Compare commits
1
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
e766cedd1a |
@@ -6,7 +6,7 @@
|
||||
"plugins": [
|
||||
{
|
||||
"name": "gitnexus",
|
||||
"version": "1.6.10-rc.94",
|
||||
"version": "1.6.9",
|
||||
"source": {
|
||||
"source": "local",
|
||||
"path": "./gitnexus-claude-plugin"
|
||||
|
||||
@@ -11,7 +11,7 @@
|
||||
"plugins": [
|
||||
{
|
||||
"name": "gitnexus",
|
||||
"version": "1.6.10-rc.94",
|
||||
"version": "1.6.9",
|
||||
"source": "./gitnexus-claude-plugin",
|
||||
"description": "Code intelligence powered by a knowledge graph. Provides execution flow tracing, blast radius analysis, and augmented search across your codebase."
|
||||
}
|
||||
|
||||
@@ -1,5 +0,0 @@
|
||||
# Custom self-hosted runner labels actionlint can't discover on its own.
|
||||
# gitnexus-evolution: the skill-evolution EC2 runner (infra/gitnexus-evolution/).
|
||||
self-hosted-runner:
|
||||
labels:
|
||||
- gitnexus-evolution
|
||||
+1
-1
@@ -11,7 +11,7 @@
|
||||
"@anthropic-ai/claude-code": "2.1.214"
|
||||
},
|
||||
"engines": {
|
||||
"node": "22.18.0"
|
||||
"node": "22.16.0"
|
||||
}
|
||||
},
|
||||
"node_modules/@anthropic-ai/claude-code": {
|
||||
|
||||
@@ -3,7 +3,7 @@
|
||||
"version": "0.0.0",
|
||||
"private": true,
|
||||
"engines": {
|
||||
"node": "22.18.0"
|
||||
"node": "22.16.0"
|
||||
},
|
||||
"dependencies": {
|
||||
"@anthropic-ai/claude-code": "2.1.214"
|
||||
|
||||
+1
-1
@@ -11,7 +11,7 @@
|
||||
"gitnexus": "1.6.9"
|
||||
},
|
||||
"engines": {
|
||||
"node": "22.18.0"
|
||||
"node": "22.16.0"
|
||||
}
|
||||
},
|
||||
"node_modules/@emnapi/runtime": {
|
||||
|
||||
@@ -3,7 +3,7 @@
|
||||
"private": true,
|
||||
"version": "1.0.0",
|
||||
"engines": {
|
||||
"node": "22.18.0"
|
||||
"node": "22.16.0"
|
||||
},
|
||||
"dependencies": {
|
||||
"gitnexus": "1.6.9"
|
||||
|
||||
@@ -352,7 +352,7 @@ jobs:
|
||||
with:
|
||||
persist-credentials: false # this job uploads artifacts (artipacked)
|
||||
|
||||
- uses: actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v7.0.0
|
||||
- uses: actions/setup-node@48b55a011bda9f5d6aeb4c2d9c7362e8dae4041e # v6.4.0
|
||||
with:
|
||||
node-version: 22
|
||||
|
||||
|
||||
@@ -39,7 +39,7 @@ jobs:
|
||||
- uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0
|
||||
with:
|
||||
persist-credentials: false
|
||||
- uses: actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v7.0.0
|
||||
- uses: actions/setup-node@48b55a011bda9f5d6aeb4c2d9c7362e8dae4041e # v6.4.0
|
||||
with:
|
||||
node-version: 22
|
||||
- name: Unit-test the host->container config transforms
|
||||
@@ -60,7 +60,7 @@ jobs:
|
||||
- uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0
|
||||
with:
|
||||
persist-credentials: false
|
||||
- uses: actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v7.0.0
|
||||
- uses: actions/setup-node@48b55a011bda9f5d6aeb4c2d9c7362e8dae4041e # v6.4.0
|
||||
with:
|
||||
node-version: 22
|
||||
# Builds the image the same way a developer's "Reopen in Container" does.
|
||||
|
||||
@@ -14,7 +14,7 @@ jobs:
|
||||
- uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0
|
||||
with:
|
||||
persist-credentials: false
|
||||
- uses: actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v7.0.0
|
||||
- uses: actions/setup-node@48b55a011bda9f5d6aeb4c2d9c7362e8dae4041e # v6.4.0
|
||||
with:
|
||||
node-version: 22
|
||||
cache: npm
|
||||
@@ -29,7 +29,7 @@ jobs:
|
||||
- uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0
|
||||
with:
|
||||
persist-credentials: false
|
||||
- uses: actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v7.0.0
|
||||
- uses: actions/setup-node@48b55a011bda9f5d6aeb4c2d9c7362e8dae4041e # v6.4.0
|
||||
with:
|
||||
node-version: 22
|
||||
cache: npm
|
||||
|
||||
@@ -46,7 +46,7 @@ jobs:
|
||||
with:
|
||||
path: ~/.lbdb/extension
|
||||
key: lbug-fts-${{ runner.os }}-${{ hashFiles('gitnexus/package-lock.json') }}
|
||||
- name: Ensure FTS + VECTOR extensions installed
|
||||
- name: Ensure FTS extension installed
|
||||
run: npx tsx scripts/ensure-fts.ts
|
||||
working-directory: gitnexus
|
||||
- name: Run sharded tests with coverage (blob)
|
||||
@@ -205,10 +205,6 @@ jobs:
|
||||
# tsx-on-source path in CI (both entry points stay covered).
|
||||
env:
|
||||
GITNEXUS_REQUIRE_FTS: '1'
|
||||
# #2623: the win32 VECTOR gate is gone, so the vector suites genuinely
|
||||
# run here — require the extension so an unavailable VECTOR is a loud
|
||||
# failure, never a silent skip (same contract as GITNEXUS_REQUIRE_FTS).
|
||||
GITNEXUS_REQUIRE_VECTOR: '1'
|
||||
GITNEXUS_E2E_CLI: dist
|
||||
# #2449: hosted Windows runners intermittently push the busiest shard past
|
||||
# the default 15-minute watchdog. 20 minutes restores real headroom while
|
||||
@@ -223,21 +219,19 @@ jobs:
|
||||
- uses: ./.github/actions/setup-gitnexus
|
||||
with:
|
||||
build: 'true'
|
||||
# Warm-cache the installed LadybugDB FTS + VECTOR extensions
|
||||
# (~/.lbdb/extension) per OS + lockfile so a warm run skips the network
|
||||
# install entirely, and the parallel shards share one download across
|
||||
# runs. Pure reliability/speed: on a cache miss the tests self-install on
|
||||
# demand (see test/helpers/fts-availability.ts), so a miss just falls
|
||||
# back to install — never a correctness dependency. Keyed by lockfile
|
||||
# hash so a LadybugDB version bump re-installs; per-OS because the
|
||||
# extensions are native binaries. (Key name kept as lbug-fts for cache
|
||||
# continuity — the path covers every extension in the shared home.)
|
||||
# Warm-cache the installed LadybugDB FTS extension (~/.lbdb/extension) per
|
||||
# OS + lockfile so a warm run skips the network install entirely, and the
|
||||
# parallel shards share one download across runs. Pure reliability/speed:
|
||||
# on a cache miss the tests self-install FTS on demand (see
|
||||
# test/helpers/fts-availability.ts), so a miss just falls back to install —
|
||||
# never a correctness dependency. Keyed by lockfile hash so a LadybugDB
|
||||
# version bump re-installs; per-OS because the extension is a native binary.
|
||||
- name: Cache LadybugDB FTS extension
|
||||
uses: actions/cache@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v5
|
||||
with:
|
||||
path: ~/.lbdb/extension
|
||||
key: lbug-fts-${{ runner.os }}-${{ hashFiles('gitnexus/package-lock.json') }}
|
||||
- name: Ensure FTS + VECTOR extensions installed
|
||||
- name: Ensure FTS extension installed
|
||||
run: npx tsx scripts/ensure-fts.ts
|
||||
working-directory: gitnexus
|
||||
- name: Run platform-sensitive tests
|
||||
@@ -384,16 +378,15 @@ jobs:
|
||||
"$PREFIX/bin/gitnexus" --version
|
||||
fi
|
||||
|
||||
# Node engines-floor gate (#2372). A module that statically names an API
|
||||
# newer than the supported floor (e.g. `module.registerHooks`, added in
|
||||
# 22.15) fails to LINK on the floor — a class vitest/tsx transforms
|
||||
# structurally mask, and the default `node-version: 22` (resolves to latest)
|
||||
# never hits. Build the dist on 22.x, then import-link every module R1 names
|
||||
# as a load surface on the pinned engines floor (22.18.0, per package.json
|
||||
# `engines: ^22.18.0 || >=24.11.0`) so a regression fails here instead of
|
||||
# shipping to users on the minimum supported Node.
|
||||
# Node engines-floor gate (#2372). The embedding resolvers statically named
|
||||
# `module.registerHooks`, which only exists on Node >= 22.15 / >= 23.5, so on
|
||||
# the supported floor (engines: >=22.0.0) those ESM modules failed to LINK —
|
||||
# a class vitest/tsx transforms structurally mask, and the default
|
||||
# `node-version: 22` (resolves to latest) never hits. Build the dist on 22.x,
|
||||
# then import-link every module R1 names as a load surface on a pinned 22.14
|
||||
# so a regression fails here instead of shipping to users on that Node range.
|
||||
node-floor-compat:
|
||||
name: node floor compat (22.18)
|
||||
name: node floor compat (22.14)
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 15
|
||||
steps:
|
||||
@@ -402,7 +395,7 @@ jobs:
|
||||
- uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0
|
||||
with:
|
||||
persist-credentials: false
|
||||
- uses: actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v7.0.0
|
||||
- uses: actions/setup-node@48b55a011bda9f5d6aeb4c2d9c7362e8dae4041e # v6.4.0
|
||||
with:
|
||||
node-version: '22'
|
||||
cache: npm
|
||||
@@ -420,16 +413,16 @@ jobs:
|
||||
# Switch to the engines-floor Node AFTER building — native deps built on
|
||||
# 22.x load across the whole 22.x ABI line, and nothing installs after this
|
||||
# (so no package-manager cache is needed).
|
||||
- uses: actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v7.0.0
|
||||
- uses: actions/setup-node@48b55a011bda9f5d6aeb4c2d9c7362e8dae4041e # v6.4.0
|
||||
with:
|
||||
node-version: '22.18.0'
|
||||
node-version: '22.14.0'
|
||||
package-manager-cache: false
|
||||
- name: Import-link the built dist on Node 22.18
|
||||
- name: Import-link the built dist on Node 22.14
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
node --version
|
||||
node --version | grep -q '^v22\.18\.' || { echo "expected Node 22.18.x" >&2; exit 1; }
|
||||
node --version | grep -q '^v22\.14\.' || { echo "expected Node 22.14.x" >&2; exit 1; }
|
||||
for m in \
|
||||
core/embeddings/runtime-install \
|
||||
core/embeddings/onnxruntime-node-resolver \
|
||||
@@ -561,9 +554,9 @@ jobs:
|
||||
- uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0
|
||||
with:
|
||||
persist-credentials: false
|
||||
- uses: actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v7.0.0
|
||||
- uses: actions/setup-node@48b55a011bda9f5d6aeb4c2d9c7362e8dae4041e # v6.4.0
|
||||
with:
|
||||
node-version: '22.18.0'
|
||||
node-version: '22.16.0'
|
||||
cache: npm
|
||||
cache-dependency-path: |
|
||||
gitnexus/package-lock.json
|
||||
|
||||
@@ -323,9 +323,9 @@ jobs:
|
||||
- name: Set up pinned Node.js
|
||||
id: setup-node
|
||||
if: steps.context.outputs.ready == 'true'
|
||||
uses: actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v7.0.0
|
||||
uses: actions/setup-node@48b55a011bda9f5d6aeb4c2d9c7362e8dae4041e # v6.4.0
|
||||
with:
|
||||
node-version: '22.18.0'
|
||||
node-version: '22.16.0'
|
||||
|
||||
- name: Install and preflight Claude subprocess isolation
|
||||
id: isolation
|
||||
@@ -377,7 +377,7 @@ jobs:
|
||||
.github/claude-canary-runtime/package-lock.json \
|
||||
"${runtime_dir}/package-lock.json"
|
||||
printf '%s\n' 'registry=https://registry.npmjs.org/' 'audit=false' 'fund=false' > "${npmrc}"
|
||||
test "$(node --version)" = 'v22.18.0'
|
||||
test "$(node --version)" = 'v22.16.0'
|
||||
test "$(uname -m)" = 'x86_64'
|
||||
|
||||
# The trusted lock and these independent receipts pin both the thin
|
||||
@@ -398,7 +398,7 @@ jobs:
|
||||
if (
|
||||
lock.lockfileVersion !== 3 ||
|
||||
lock.packages?.['']?.dependencies?.['@anthropic-ai/claude-code'] !== '2.1.214' ||
|
||||
lock.packages?.['']?.engines?.node !== '22.18.0'
|
||||
lock.packages?.['']?.engines?.node !== '22.16.0'
|
||||
) {
|
||||
throw new Error('Claude runtime lock root is not exact');
|
||||
}
|
||||
@@ -506,7 +506,7 @@ jobs:
|
||||
install -m 0600 .github/gitnexus-review-runtime/package.json "${runtime_dir}/package.json"
|
||||
install -m 0600 .github/gitnexus-review-runtime/package-lock.json "${runtime_dir}/package-lock.json"
|
||||
printf '%s\n' 'registry=https://registry.npmjs.org/' 'audit=false' 'fund=false' > "${npmrc}"
|
||||
test "$(node --version)" = 'v22.18.0'
|
||||
test "$(node --version)" = 'v22.16.0'
|
||||
npm ci \
|
||||
--prefix "${runtime_dir}" \
|
||||
--userconfig "${npmrc}" \
|
||||
@@ -1241,7 +1241,7 @@ jobs:
|
||||
CLAUDE_CONFIG_DIR: ${{ runner.temp }}/gitnexus-review-claude-config
|
||||
CLAUDE_WORKING_DIR: ${{ runner.temp }}/gitnexus-review-control
|
||||
NPM_CONFIG_IGNORE_SCRIPTS: 'true'
|
||||
NODE_VERSION: '22.18.0'
|
||||
NODE_VERSION: '22.16.0'
|
||||
with:
|
||||
claude_code_oauth_token: ${{ secrets.CLAUDE_CODE_OAUTH_TOKEN }}
|
||||
path_to_claude_code_executable: ${{ runner.temp }}/gitnexus-review-claude-runtime/node_modules/@anthropic-ai/claude-code/bin/claude.exe
|
||||
|
||||
@@ -12,34 +12,12 @@
|
||||
# App that opens the promotion PR). The Mint-App-Token step hard-fails
|
||||
# without them once a promotion is detected. Verify the App installation
|
||||
# is scoped to this repo with only Contents: RW + Pull requests: RW.
|
||||
# [x] Create the protected Environment `gitnexus-evolution` with a
|
||||
# [ ] Create the protected Environment `gitnexus-evolution` with a
|
||||
# deployment-branch rule restricting it to `main`, and ideally scope the
|
||||
# three secrets above to that Environment. workflow_dispatch runs this
|
||||
# workflow (and eval/workflow_bench/evolve.py) from the *dispatched ref*,
|
||||
# so this server-side rule — not a code-side guard the branch could edit
|
||||
# away — is what stops a non-main branch from running with the secrets.
|
||||
# [x] Register a self-hosted runner labeled `gitnexus-evolution` (a dedicated
|
||||
# EC2 box works well). GitHub-hosted runners hard-cap job execution at 6
|
||||
# hours, non-configurable — too short once a benchmark session actually
|
||||
# invokes Skill/MCP tools for real. Self-hosted runners cap at 5 days
|
||||
# instead. This job only ever runs on schedule/workflow_dispatch, never
|
||||
# on fork-PR content, so the usual public-repo self-hosted-runner risk
|
||||
# doesn't apply — still keep the box dedicated to this workflow, with
|
||||
# outbound-only network access, and prefer on-demand over Spot (a Spot
|
||||
# reclaim mid-run loses the same way a 6-hour timeout does). Instance,
|
||||
# security group, and IAM setup are documented privately, not in this
|
||||
# repo — publishing the exact topology of a real, live AWS account
|
||||
# isn't safe to do in a public repo even without literal secrets.
|
||||
# Accepted tradeoff: the box is stopped between runs (an EventBridge
|
||||
# schedule starts it ~15min before the Saturday cron and stops it 24h
|
||||
# later) but is not destroyed/recreated per run, so it isn't fully
|
||||
# ephemeral — a compromise between the review-flagged ideal (re-image
|
||||
# between runs, bounding how long the injected model API key could
|
||||
# matter if the box were ever compromised some other way) and the added
|
||||
# complexity of per-job ephemeral provisioning for a job that runs at
|
||||
# most weekly. Revisit if run frequency increases or the threat model
|
||||
# changes; stopping already bounds the exposure window to the job's own
|
||||
# runtime on 1 day out of 7.
|
||||
# [ ] Run workflow_dispatch once and confirm: containment preflight passes,
|
||||
# the benchmark completes inside the job timeout, the results artifact
|
||||
# uploads, and a promotion (if any) opens a well-formed PR.
|
||||
@@ -99,13 +77,13 @@ jobs:
|
||||
github.event_name == 'workflow_dispatch' ||
|
||||
vars.GITNEXUS_EVOLUTION_ENABLED == 'true'
|
||||
)
|
||||
runs-on: [self-hosted, linux, x64, gitnexus-evolution]
|
||||
runs-on: ubuntu-latest
|
||||
# Gate promotion runs on a protected Environment. An admin must attach a
|
||||
# deployment-branch rule (main only) and ideally scope the three secrets to
|
||||
# it — server-side enforcement a dispatched non-main ref cannot bypass by
|
||||
# editing its own workflow copy. See the activation checklist above.
|
||||
environment: gitnexus-evolution
|
||||
timeout-minutes: 1440 # self-hosted ceiling is 5 days (7200min); 24h is a generous margin over a single-generation serial run
|
||||
timeout-minutes: 355 # ceiling just under GitHub's 360-minute hard cap
|
||||
permissions:
|
||||
contents: read # The promotion PR uses a short-lived App token minted below.
|
||||
env:
|
||||
@@ -130,9 +108,9 @@ jobs:
|
||||
persist-credentials: false
|
||||
fetch-depth: 0
|
||||
|
||||
- uses: actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v7.0.0
|
||||
- uses: actions/setup-node@48b55a011bda9f5d6aeb4c2d9c7362e8dae4041e # v6.4.0
|
||||
with:
|
||||
node-version: '22.18.0'
|
||||
node-version: '22.16.0'
|
||||
cache: npm
|
||||
cache-dependency-path: |
|
||||
gitnexus/package-lock.json
|
||||
|
||||
@@ -48,7 +48,7 @@ jobs:
|
||||
with:
|
||||
persist-credentials: false
|
||||
|
||||
- uses: actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v7.0.0
|
||||
- uses: actions/setup-node@48b55a011bda9f5d6aeb4c2d9c7362e8dae4041e # v6.4.0
|
||||
with:
|
||||
node-version: 22
|
||||
|
||||
|
||||
@@ -59,7 +59,7 @@ jobs:
|
||||
repository: ${{ github.event.pull_request.head.repo.full_name }}
|
||||
persist-credentials: false
|
||||
|
||||
- uses: actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v7.0.0
|
||||
- uses: actions/setup-node@48b55a011bda9f5d6aeb4c2d9c7362e8dae4041e # v6.4.0
|
||||
with:
|
||||
node-version: 22
|
||||
cache: npm
|
||||
|
||||
@@ -369,7 +369,7 @@ jobs:
|
||||
exit 1
|
||||
fi
|
||||
|
||||
- uses: actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v7.0.0
|
||||
- uses: actions/setup-node@48b55a011bda9f5d6aeb4c2d9c7362e8dae4041e # v6.4.0
|
||||
with:
|
||||
# Node 24 ships with npm >= 11.5.x, which is the minimum that
|
||||
# supports npm Trusted Publishing OIDC. Node 22 ships with npm
|
||||
@@ -828,7 +828,7 @@ jobs:
|
||||
fi
|
||||
|
||||
- name: Create GitHub Release
|
||||
uses: softprops/action-gh-release@3d0d9888cb7fd7b750713d6e236d1fcb99157228 # v2
|
||||
uses: softprops/action-gh-release@718ea10b132b3b2eba29c1007bb80653f286566b # v2
|
||||
with:
|
||||
tag_name: ${{ steps.vtag-gate.outputs.vtag }}
|
||||
name: >-
|
||||
|
||||
@@ -50,7 +50,7 @@ jobs:
|
||||
- uses: actions/checkout@9c091bb21b7c1c1d1991bb908d89e4e9dddfe3e0 # v7.0.0
|
||||
with:
|
||||
persist-credentials: false
|
||||
- uses: actions/setup-node@820762786026740c76f36085b0efc47a31fe5020 # v7.0.0
|
||||
- uses: actions/setup-node@48b55a011bda9f5d6aeb4c2d9c7362e8dae4041e # v6.4.0
|
||||
with:
|
||||
node-version: '22'
|
||||
cache: npm
|
||||
|
||||
+2
-1
@@ -68,8 +68,9 @@ gitnexus-web/test-results/
|
||||
eval/.coverage
|
||||
eval/.hypothesis/
|
||||
|
||||
# Local docs — planning output (gitnexus-plan / gitnexus-work) stays local, not tracked
|
||||
# Local docs (docs/plans/ stays tracked — gitnexus-plan output travels with the work)
|
||||
docs/*
|
||||
!docs/plans/
|
||||
|
||||
gitnexus/test/fixtures/mini-repo/*.md
|
||||
gitnexus/test/fixtures/mini-repo/.claude
|
||||
|
||||
+1
-1
@@ -13,7 +13,7 @@ This project uses the [PolyForm Noncommercial License 1.0.0](https://polyformpro
|
||||
|
||||
## Development setup
|
||||
|
||||
**Prerequisites:** Node.js — `gitnexus/` requires `^22.18.0 || >=24.11.0` and `gitnexus-web/` requires `^20.19.0 || >=22.12.0` (enforced via the `engines` field in each package). Use `nvm install` to match the local version.
|
||||
**Prerequisites:** Node.js — `gitnexus/` requires `>=22.0.0` and `gitnexus-web/` requires `^20.19.0 || >=22.12.0` (enforced via the `engines` field in each package). Use `nvm install` to match the local version.
|
||||
|
||||
1. Clone the repository.
|
||||
2. **Shared package:** `cd gitnexus-shared && npm install && npm run build`
|
||||
|
||||
@@ -488,10 +488,9 @@ Most `analyze` knobs are also CLI flags (`--workers`, `--worker-timeout`, `--max
|
||||
| `PROF_LBUG_LOAD` | unset | When `1`, emits one `[lbug-load prof]` summary line per `loadGraphToLbug` call breaking the graph-DB persistence wall into stages (`csv-emit` / `copy-nodes` / `copy-rels` / `fallback` / `total`) plus node & edge counts. Zero-cost when unset. | Attributing large-repo analyze wall time across CSV generation vs. LadybugDB `COPY` (issue #2203) — the analyze "emit" timing is the scope-resolution bucket, not this DB-write path. |
|
||||
| `GITNEXUS_MAX_FILE_SIZE` | `512` (KB) | Walker skip threshold in KB. Hard cap is `32768` (tree-sitter buffer ceiling). Equivalent to `--max-file-size <kb>`. | Indexing repos with intentionally-large source files (generated parsers, vendored bundles) that should still be parsed. |
|
||||
| `GITNEXUS_WORKER_SUB_BATCH_TIMEOUT_MS` | `30000` | Worker idle timeout in milliseconds before retry/fallback. Equivalent to `--worker-timeout <seconds>` × 1000. | Slow-parsing files (large minified JS, deeply-nested TS types) that legitimately need more than 30s. |
|
||||
| `GITNEXUS_WORKER_READY_TIMEOUT_MS` | `5000` | Startup budget in milliseconds for a parse worker to load its grammar bindings and report `{type:'ready'}`. Slots that miss it are treated as startup crashes. | Slow or heavily loaded hosts where a full pool cold-starting concurrently needs more than 5s, and analyze aborts with "did not report ready within 5000ms". |
|
||||
| `GITNEXUS_FTS_STEMMER` | `porter` | Stemmer used when rebuilding BM25/FTS indexes. Use `none` for CJK-heavy repositories, or a language stemmer such as `german`, `french`, or `spanish` for matching repository comments. Re-run `gitnexus analyze --repair-fts` after changing it. | Keyword search quality is poor for non-English comments or identifiers under English stemming. |
|
||||
| `GITNEXUS_WAL_CHECKPOINT_THRESHOLD` | `67108864` (64 MiB) | LadybugDB WAL auto-checkpoint threshold in bytes. Equivalent to `--wal-checkpoint-threshold <bytes>`. `-1` keeps LadybugDB's stock threshold (~16 MiB). Larger thresholds reduce checkpoint frequency but increase the WAL size at rotation time — choose a smaller value on disk-constrained environments. | You need a larger or smaller WAL auto-checkpoint threshold for your analyze workload. |
|
||||
| `GITNEXUS_LBUG_BUFFER_POOL_SIZE` | min(2 GiB, 80% RAM) | LadybugDB buffer-pool ceiling in bytes for every GitNexus database (analyze, MCP server, serve, group bridges). `0` restores LadybugDB's native unbounded default of 80% of system RAM; invalid values warn and fall back to the default (#2557). During `analyze` the pool is right-sized to the graph, scaled on non-4 KiB-page hosts by the page-size granule ratio up to min(2 GiB × pageSize/4 KiB, 80% RAM) (#2631); this env var overrides all of that as an absolute value. | A long-lived `gitnexus mcp` or a big incremental `analyze` uses too much memory, or a huge repo's working set genuinely needs a pool larger than 2 GiB. |
|
||||
| `GITNEXUS_LBUG_BUFFER_POOL_SIZE` | min(2 GiB, 80% RAM) | LadybugDB buffer-pool ceiling in bytes for every GitNexus database (analyze, MCP server, serve, group bridges). `0` restores LadybugDB's native unbounded default of 80% of system RAM; invalid values warn and fall back to the default (#2557). | A long-lived `gitnexus mcp` or a big incremental `analyze` uses too much memory, or a huge repo's working set genuinely needs a pool larger than 2 GiB. |
|
||||
| `GITNEXUS_LBUG_MAX_DB_SIZE` | `17179869184` (16 GiB) | Maximum size in bytes of a single LadybugDB database file — an mmap/disk-address-space ceiling, not a memory limit (it does not constrain the buffer pool). Invalid values silently fall back to the default. | Indexing a genuinely huge monorepo whose on-disk graph index approaches 16 GiB. |
|
||||
| `GITNEXUS_WORKER_SUB_BATCH_MAX_BYTES` | `8388608` (8 MB) | Per-job byte budget the pool will send to a worker in one `postMessage`. | Very large individual files; mostly diagnostic — bumping past 8 MB risks structured-clone memory pressure. |
|
||||
| `GITNEXUS_WORKER_MAX_RESPAWNS_PER_SLOT` | `3` | Max replacement spawns per worker slot before the slot is dropped from the active rotation. Bounds respawn loops on a chronically-crashing slot. | Hosts where a flaky worker should retry more (raise) or fail-fast (lower) before the slot is dropped. |
|
||||
|
||||
@@ -4,7 +4,6 @@ from __future__ import annotations
|
||||
|
||||
import json
|
||||
import os
|
||||
import shutil
|
||||
import stat
|
||||
import subprocess
|
||||
import sys
|
||||
@@ -20,16 +19,10 @@ from workflow_bench.process_control import ManagedProcessResult, run_managed
|
||||
from workflow_bench.proposer_sandbox import (
|
||||
MAX_BUNDLE_BYTES,
|
||||
MAX_EVIDENCE_FILE_BYTES,
|
||||
SANDBOX_NODE,
|
||||
SANDBOX_NODE_PREFIX,
|
||||
VITE_TEMP_DIR,
|
||||
SANDBOX_PATH,
|
||||
SANDBOX_PYTHON3,
|
||||
SANDBOX_SHELL_PREFIX,
|
||||
SANDBOX_USER_SKILLS,
|
||||
ReadOnlyMount,
|
||||
SandboxError,
|
||||
_runtime_mount_args,
|
||||
build_claude_settings,
|
||||
build_sandbox_environment,
|
||||
prepare_sandbox,
|
||||
@@ -168,28 +161,7 @@ def test_sandbox_command_has_minimal_mounts_and_no_host_root_bind(tmp_path: Path
|
||||
check=False,
|
||||
)
|
||||
assert probe.returncode == 0, probe.stderr
|
||||
assert probe.stdout == f"/home/agent|{SANDBOX_PATH}"
|
||||
|
||||
# The evidence-provenance.mjs plan-writer's PATH-scan trusts a Python 3
|
||||
# candidate only if it (and its directory) is owned by root or by the
|
||||
# current process — real /usr/bin/python3 is root-owned on the host,
|
||||
# which surfaces as the kernel's overflow uid inside this
|
||||
# --unshare-user sandbox (root itself is never mapped in). This wrapper
|
||||
# is freshly created by the host process instead, so it's trusted, and
|
||||
# it must still exec through to a real, working Python 3.
|
||||
python3_index = argv.index(SANDBOX_PYTHON3)
|
||||
assert argv[python3_index - 2] == "--ro-bind"
|
||||
python3_wrapper = Path(argv[python3_index - 1])
|
||||
assert stat.S_IMODE(python3_wrapper.stat().st_mode) == 0o500
|
||||
version = subprocess.run(
|
||||
[str(python3_wrapper), "-I", "-S", "-c", "import sys; print(sys.version_info[0])"],
|
||||
text=True,
|
||||
capture_output=True,
|
||||
check=False,
|
||||
)
|
||||
assert version.returncode == 0, version.stderr
|
||||
assert version.stdout.strip() == "3"
|
||||
|
||||
assert probe.stdout == "/home/agent|/opt/claude:/usr/local/bin:/usr/bin:/bin"
|
||||
assert SANDBOX_USER_SKILLS in argv
|
||||
user_skills_index = argv.index(SANDBOX_USER_SKILLS)
|
||||
assert argv[user_skills_index - 2] == "--ro-bind"
|
||||
@@ -197,223 +169,6 @@ def test_sandbox_command_has_minimal_mounts_and_no_host_root_bind(tmp_path: Path
|
||||
assert not private_root.exists()
|
||||
|
||||
|
||||
def test_runtime_mounts_bind_the_resolved_node_to_a_fresh_sandbox_path(monkeypatch) -> None:
|
||||
# sanitized_graph.py and runner_sessions.py invoke the sandboxed graph CLI
|
||||
# via SANDBOX_NODE. node's real host location varies (GitHub-hosted
|
||||
# runner images happen to have one under /usr/local/bin; a self-hosted
|
||||
# runner's actions/setup-node installs into its own tool-cache directory
|
||||
# instead), so this must bind to a FRESH sandbox path like /opt/claude/...
|
||||
# rather than anywhere under /usr, /bin, /lib, or /lib64: those are
|
||||
# already read-only bound by this same function, and bwrap can't create
|
||||
# a new mount-point file inside an already-read-only tree when the real
|
||||
# path doesn't already exist there on the host (observed empirically:
|
||||
# "bwrap: Can't create file at /usr/local/bin/node: Read-only file
|
||||
# system" when this bind first targeted that path on a self-hosted
|
||||
# runner where node isn't really there).
|
||||
monkeypatch.setattr(
|
||||
"workflow_bench.proposer_sandbox.shutil.which",
|
||||
lambda name: "/opt/hostedtoolcache/node/22.18.0/x64/bin/node" if name == "node" else None,
|
||||
)
|
||||
args = _runtime_mount_args()
|
||||
node_index = args.index("/opt/hostedtoolcache/node/22.18.0/x64/bin/node")
|
||||
assert args[node_index - 1] == "--ro-bind"
|
||||
assert args[node_index + 1] == SANDBOX_NODE
|
||||
assert not any(SANDBOX_NODE.startswith(bound + "/") for bound in ("/usr", "/bin", "/lib", "/lib64"))
|
||||
|
||||
|
||||
def test_runtime_mounts_bind_the_node_prefix_so_npx_and_npm_resolve(monkeypatch, tmp_path) -> None:
|
||||
# npx and npm are not standalone binaries -- they are symlinks into
|
||||
# ../lib/node_modules/npm/bin/*-cli.js -- so binding the sibling files is
|
||||
# not enough; the install prefix carrying both bin/ and lib/node_modules
|
||||
# has to be mounted. Without this, a self-hosted runner (where
|
||||
# actions/setup-node installs into its own tool cache, outside /usr) gets
|
||||
# a sandbox with node but no npx, and every task verify command dies with
|
||||
# "/bin/sh: 1: npx: not found" -- all 18 runs of skill-evolution run
|
||||
# 29861768554 did exactly that.
|
||||
prefix = tmp_path / "hostedtoolcache" / "node" / "22.18.0" / "x64"
|
||||
(prefix / "bin").mkdir(parents=True)
|
||||
(prefix / "bin" / "node").write_text("#!/bin/sh\nexit 0\n")
|
||||
(prefix / "lib" / "node_modules" / "npm" / "bin").mkdir(parents=True)
|
||||
(prefix / "lib" / "node_modules" / "npm" / "bin" / "npx-cli.js").write_text("")
|
||||
(prefix / "bin" / "npx").symlink_to("../lib/node_modules/npm/bin/npx-cli.js")
|
||||
monkeypatch.setattr(
|
||||
"workflow_bench.proposer_sandbox.shutil.which",
|
||||
lambda name: str(prefix / "bin" / "node") if name == "node" else None,
|
||||
)
|
||||
args = _runtime_mount_args()
|
||||
prefix_index = args.index(str(prefix))
|
||||
assert args[prefix_index - 1] == "--ro-bind"
|
||||
assert args[prefix_index + 1] == SANDBOX_NODE_PREFIX
|
||||
# the single-binary bind stays: sanitized_graph.py and runner_sessions.py
|
||||
# invoke SANDBOX_NODE directly.
|
||||
node_index = args.index(str(prefix / "bin" / "node"))
|
||||
assert args[node_index + 1] == SANDBOX_NODE
|
||||
# and the prefix's bin/ must actually be on PATH for npx to resolve.
|
||||
assert f"{SANDBOX_NODE_PREFIX}/bin" in SANDBOX_PATH.split(":")
|
||||
|
||||
|
||||
def test_runtime_mounts_skip_the_prefix_bind_for_an_unrecognized_node_layout(monkeypatch, tmp_path) -> None:
|
||||
# The prefix is derived from the node binary's path, so it must only be
|
||||
# trusted when the layout really is <prefix>/bin/node carrying npm.
|
||||
# Otherwise parent.parent names an unrelated ancestor: /opt/bin/node would
|
||||
# bind ALL of /opt (every tool cache on a hosted runner) and a bare
|
||||
# <dir>/node would bind <dir>'s parent -- an over-broad mount into a
|
||||
# sandbox that runs untrusted model-authored code. The pre-existing
|
||||
# real-Bubblewrap node canary builds exactly this bare <dir>/node shape.
|
||||
bare = tmp_path / "toolcache"
|
||||
bare.mkdir()
|
||||
(bare / "node").write_text("#!/bin/sh\nexit 0\n")
|
||||
monkeypatch.setattr(
|
||||
"workflow_bench.proposer_sandbox.shutil.which",
|
||||
lambda name: str(bare / "node") if name == "node" else None,
|
||||
)
|
||||
args = _runtime_mount_args()
|
||||
assert SANDBOX_NODE_PREFIX not in args
|
||||
assert str(tmp_path) not in args
|
||||
# the node bind itself is unaffected -- SANDBOX_NODE still works.
|
||||
assert args[args.index(str(bare / "node")) + 1] == SANDBOX_NODE
|
||||
|
||||
|
||||
def test_runtime_mounts_skip_the_prefix_bind_without_npx_beside_node(monkeypatch, tmp_path) -> None:
|
||||
# Right <prefix>/bin/node shape, but no working npx beside it: binding the
|
||||
# prefix would widen the mount surface without making npx resolvable.
|
||||
prefix = tmp_path / "x64"
|
||||
(prefix / "bin").mkdir(parents=True)
|
||||
(prefix / "bin" / "node").write_text("#!/bin/sh\nexit 0\n")
|
||||
monkeypatch.setattr(
|
||||
"workflow_bench.proposer_sandbox.shutil.which",
|
||||
lambda name: str(prefix / "bin" / "node") if name == "node" else None,
|
||||
)
|
||||
args = _runtime_mount_args()
|
||||
assert SANDBOX_NODE_PREFIX not in args
|
||||
|
||||
|
||||
def test_runtime_mounts_bind_a_real_tool_cache_layout(monkeypatch, tmp_path) -> None:
|
||||
# The positive counterpart: a genuine <prefix>/bin/node install carrying
|
||||
# npm, outside the system trees, is bound so npx resolves.
|
||||
prefix = tmp_path / "node" / "22.18.0" / "x64"
|
||||
(prefix / "bin").mkdir(parents=True)
|
||||
(prefix / "bin" / "node").write_text("#!/bin/sh\nexit 0\n")
|
||||
(prefix / "lib" / "node_modules" / "npm" / "bin").mkdir(parents=True)
|
||||
(prefix / "lib" / "node_modules" / "npm" / "bin" / "npx-cli.js").write_text("")
|
||||
(prefix / "bin" / "npx").symlink_to("../lib/node_modules/npm/bin/npx-cli.js")
|
||||
monkeypatch.setattr(
|
||||
"workflow_bench.proposer_sandbox.shutil.which",
|
||||
lambda name: str(prefix / "bin" / "node") if name == "node" else None,
|
||||
)
|
||||
args = _runtime_mount_args()
|
||||
prefix_index = args.index(SANDBOX_NODE_PREFIX)
|
||||
assert args[prefix_index - 2] == "--ro-bind"
|
||||
assert args[prefix_index - 1] == str(prefix)
|
||||
|
||||
|
||||
def test_runtime_mounts_skip_the_prefix_bind_when_it_is_already_bound(monkeypatch) -> None:
|
||||
# On an image where node genuinely lives in /usr/local/bin, the prefix is
|
||||
# /usr/local -- already inside the wholesale /usr read-only bind. Binding
|
||||
# it again would be redundant and would needlessly widen the argv, so the
|
||||
# containment surface stays minimal.
|
||||
monkeypatch.setattr(
|
||||
"workflow_bench.proposer_sandbox.shutil.which",
|
||||
lambda name: "/usr/local/bin/node" if name == "node" else None,
|
||||
)
|
||||
args = _runtime_mount_args()
|
||||
assert SANDBOX_NODE_PREFIX not in args
|
||||
assert args[args.index("/usr/local/bin/node") + 1] == SANDBOX_NODE
|
||||
|
||||
|
||||
def test_runtime_mounts_skip_the_node_bind_when_node_is_unresolvable(monkeypatch) -> None:
|
||||
monkeypatch.setattr("workflow_bench.proposer_sandbox.shutil.which", lambda name: None)
|
||||
args = _runtime_mount_args()
|
||||
assert SANDBOX_NODE not in args
|
||||
|
||||
|
||||
def test_node_modules_mounts_get_a_writable_vite_temp_overlay(tmp_path: Path) -> None:
|
||||
# vite writes <node_modules>/.vite-temp/<config>.timestamp-*.mjs before
|
||||
# loading a TypeScript config, so a read-only dependency mount makes vitest
|
||||
# fail with EROFS before any test runs -- and every task verify command and
|
||||
# every hidden oracle ends in "npx vitest run <test>". Reproduced on the
|
||||
# self-hosted runner with npx bypassed entirely, proving it is independent
|
||||
# of the node-prefix mount.
|
||||
clone = tmp_path / "clone"
|
||||
clone.mkdir()
|
||||
deps = tmp_path / "deps"
|
||||
deps.mkdir()
|
||||
# task_assets.py captures this directory into the dependency snapshot; the
|
||||
# overlay is gated on the mount source actually carrying it.
|
||||
(deps / VITE_TEMP_DIR).mkdir()
|
||||
executable = tmp_path / "executable"
|
||||
executable.write_text("#!/bin/sh\nexit 0\n")
|
||||
executable.chmod(0o755)
|
||||
|
||||
with prepare_sandbox(
|
||||
clone=clone,
|
||||
claude_bin=executable,
|
||||
bwrap_bin=executable,
|
||||
preflight=False,
|
||||
read_only_mounts=(ReadOnlyMount(source=deps, target="/workspace/gitnexus/node_modules"),),
|
||||
) as sandbox:
|
||||
argv = sandbox.command_prefix
|
||||
|
||||
bind_index = argv.index("/workspace/gitnexus/node_modules")
|
||||
assert argv[bind_index - 2 : bind_index + 1] == ["--ro-bind", str(deps), "/workspace/gitnexus/node_modules"]
|
||||
overlay = f"/workspace/gitnexus/node_modules/{VITE_TEMP_DIR}"
|
||||
overlay_index = argv.index(overlay)
|
||||
assert argv[overlay_index - 1] == "--tmpfs"
|
||||
# the overlay must come AFTER the read-only bind, or the bind would mask it
|
||||
assert overlay_index > bind_index
|
||||
|
||||
|
||||
def test_node_modules_mount_without_a_captured_vite_temp_gets_no_overlay(tmp_path: Path) -> None:
|
||||
# The trusted GitNexus runtime mounts /opt/gitnexus/node_modules, whose
|
||||
# source is the built runtime and does NOT carry a .vite-temp. bwrap cannot
|
||||
# mkdir a mount point inside a read-only bind, so overlaying it would fail
|
||||
# with "Can't mkdir .../node_modules/.vite-temp: Read-only file system".
|
||||
# Regression for that CI failure: the overlay must fire only where the
|
||||
# source actually contains the directory, not for every node_modules mount.
|
||||
clone = tmp_path / "clone"
|
||||
clone.mkdir()
|
||||
runtime = tmp_path / "runtime-node-modules"
|
||||
runtime.mkdir() # deliberately no .vite-temp
|
||||
executable = tmp_path / "executable"
|
||||
executable.write_text("#!/bin/sh\nexit 0\n")
|
||||
executable.chmod(0o755)
|
||||
|
||||
with prepare_sandbox(
|
||||
clone=clone,
|
||||
claude_bin=executable,
|
||||
bwrap_bin=executable,
|
||||
preflight=False,
|
||||
read_only_mounts=(ReadOnlyMount(source=runtime, target="/opt/gitnexus/node_modules"),),
|
||||
) as sandbox:
|
||||
argv = sandbox.command_prefix
|
||||
|
||||
assert "/opt/gitnexus/node_modules" in argv
|
||||
assert not any(str(item).endswith(f"/{VITE_TEMP_DIR}") for item in argv)
|
||||
|
||||
|
||||
def test_non_node_modules_mounts_get_no_vite_temp_overlay(tmp_path: Path) -> None:
|
||||
# Scoped to dependency mounts: a hidden-oracle or skill mount stays wholly
|
||||
# read-only, with no writable island inside it.
|
||||
clone = tmp_path / "clone"
|
||||
clone.mkdir()
|
||||
other = tmp_path / "oracle"
|
||||
other.mkdir()
|
||||
executable = tmp_path / "executable"
|
||||
executable.write_text("#!/bin/sh\nexit 0\n")
|
||||
executable.chmod(0o755)
|
||||
|
||||
with prepare_sandbox(
|
||||
clone=clone,
|
||||
claude_bin=executable,
|
||||
bwrap_bin=executable,
|
||||
preflight=False,
|
||||
read_only_mounts=(ReadOnlyMount(source=other, target="/workspace/.wfbench-oracle-abc"),),
|
||||
) as sandbox:
|
||||
argv = sandbox.command_prefix
|
||||
|
||||
assert not any(str(item).endswith(f"/{VITE_TEMP_DIR}") for item in argv)
|
||||
|
||||
|
||||
def test_stricter_prefix_freezes_evaluated_skills_and_can_unshare_network(tmp_path: Path) -> None:
|
||||
clone = tmp_path / "clone"
|
||||
skill = clone / ".claude" / "skills" / "gitnexus-work"
|
||||
@@ -442,78 +197,6 @@ def test_stricter_prefix_freezes_evaluated_skills_and_can_unshare_network(tmp_pa
|
||||
assert prefix[user_index - 2] == "--ro-bind"
|
||||
|
||||
|
||||
@pytest.mark.skipif(
|
||||
os.environ.get("GITNEXUS_REQUIRE_BWRAP_CANARY") != "1",
|
||||
reason="real Bubblewrap canary is mandatory in the named Ubuntu CI job",
|
||||
)
|
||||
def test_real_bubblewrap_runs_node_from_outside_the_bound_trees(tmp_path: Path, monkeypatch) -> None:
|
||||
# Reproduces the self-hosted-runner failure directly: node resolved from
|
||||
# a path outside /usr, /bin, /lib, /lib64 (actions/setup-node's own
|
||||
# tool-cache convention) must still be reachable inside the sandbox at
|
||||
# SANDBOX_NODE. A real node copied to a fresh, non-system location stands
|
||||
# in for the tool-cache install; argv-construction tests alone can't
|
||||
# catch a bwrap-level "Can't create file ...: Read-only file system"
|
||||
# (the actual error this fix resolves), only a real bwrap invocation can.
|
||||
real_node = shutil.which("node")
|
||||
if not real_node:
|
||||
pytest.skip("no node on PATH to relocate for this canary")
|
||||
toolcache = tmp_path / "toolcache"
|
||||
toolcache.mkdir()
|
||||
relocated_node = toolcache / "node"
|
||||
shutil.copy2(real_node, relocated_node)
|
||||
relocated_node.chmod(0o755)
|
||||
# Only fake "node"'s resolution -- prepare_sandbox's own bwrap/claude
|
||||
# lookups (_resolve_executable) also go through shutil.which, and must
|
||||
# keep resolving for real or preflight fails before the sandbox is even
|
||||
# built.
|
||||
real_which = shutil.which
|
||||
monkeypatch.setattr(
|
||||
"workflow_bench.proposer_sandbox.shutil.which",
|
||||
lambda name: str(relocated_node) if name == "node" else real_which(name),
|
||||
)
|
||||
|
||||
clone = tmp_path / "clone"
|
||||
clone.mkdir()
|
||||
with prepare_sandbox(clone=clone, claude_bin=Path(sys.executable), preflight=True) as sandbox:
|
||||
result = sandbox.run([SANDBOX_NODE, "--version"], timeout=10)
|
||||
assert result.ok, result.stderr_tail
|
||||
|
||||
|
||||
@pytest.mark.skipif(
|
||||
os.environ.get("GITNEXUS_REQUIRE_BWRAP_CANARY") != "1",
|
||||
reason="real Bubblewrap canary is mandatory in the named Ubuntu CI job",
|
||||
)
|
||||
def test_real_bubblewrap_runs_npx_from_outside_the_bound_trees(tmp_path: Path, monkeypatch) -> None:
|
||||
# The npx half of the self-hosted-runner failure. Relocating a real node
|
||||
# INSTALL (bin/ + lib/node_modules, not just the binary) to a fresh path
|
||||
# outside /usr, /bin, /lib and /lib64 reproduces actions/setup-node's
|
||||
# tool-cache convention. Every task verify command is
|
||||
# "cd gitnexus && npx tsc ... && npx vitest ...", so npx must resolve
|
||||
# inside the sandbox; argv assertions cannot prove a bwrap-level mount
|
||||
# actually works, only a real invocation can.
|
||||
real_node = shutil.which("node")
|
||||
if not real_node:
|
||||
pytest.skip("no node on PATH to relocate for this canary")
|
||||
real_prefix = Path(real_node).resolve().parent.parent
|
||||
if not (real_prefix / "lib" / "node_modules" / "npm").is_dir():
|
||||
pytest.skip(f"node at {real_node} has no npm under its install prefix")
|
||||
toolcache = tmp_path / "toolcache" / "node" / "22.18.0" / "x64"
|
||||
shutil.copytree(real_prefix, toolcache, symlinks=True)
|
||||
relocated_node = toolcache / "bin" / "node"
|
||||
assert relocated_node.exists()
|
||||
real_which = shutil.which
|
||||
monkeypatch.setattr(
|
||||
"workflow_bench.proposer_sandbox.shutil.which",
|
||||
lambda name: str(relocated_node) if name == "node" else real_which(name),
|
||||
)
|
||||
|
||||
clone = tmp_path / "clone"
|
||||
clone.mkdir()
|
||||
with prepare_sandbox(clone=clone, claude_bin=Path(sys.executable), preflight=True) as sandbox:
|
||||
result = sandbox.run(["/bin/sh", "-c", "command -v npx && npx --version"], timeout=60)
|
||||
assert result.ok, result.stderr_tail
|
||||
|
||||
|
||||
@pytest.mark.skipif(
|
||||
os.environ.get("GITNEXUS_REQUIRE_BWRAP_CANARY") != "1",
|
||||
reason="real Bubblewrap canary is mandatory in the named Ubuntu CI job",
|
||||
@@ -1067,3 +750,4 @@ for line in sys.stdin:
|
||||
assert bash_result.get("is_error") is not True, bash_result
|
||||
assert (clone / "bash-called").read_text() == "canary"
|
||||
assert (clone / "mcp-called").read_text() == "ok"
|
||||
|
||||
|
||||
@@ -262,122 +262,3 @@ def test_phase_workspace_accepts_new_regular_review_output(tmp_path):
|
||||
artifact.write_text("new review")
|
||||
|
||||
runner_artifacts.enforce_phase_workspace(tmp_path, before, allowed_artifact=artifact)
|
||||
|
||||
|
||||
def test_phase_workspace_ignores_claude_sandbox_bootstrap_noise(tmp_path):
|
||||
# Reproduced empirically: Claude Code's own enableWeakerNestedSandbox
|
||||
# bootstrap creates this exact set of paths on every session regardless
|
||||
# of task or model output (a trivial "say OK" prompt was enough). None
|
||||
# of it is something the model decided to write, so it must not read as
|
||||
# an unauthorized planning-phase change.
|
||||
before = runner_artifacts.workspace_snapshot(tmp_path)
|
||||
(tmp_path / ".claude" / "agents").mkdir(parents=True)
|
||||
(tmp_path / ".claude" / "commands").mkdir(parents=True)
|
||||
(tmp_path / ".claude" / ".cc-writes").write_text("{}")
|
||||
(tmp_path / ".env").write_text("")
|
||||
(tmp_path / ".env.development.local").write_text("")
|
||||
(tmp_path / ".npmrc").write_text("")
|
||||
(tmp_path / "package.json").write_text("{}")
|
||||
(tmp_path / "node_modules").mkdir()
|
||||
(tmp_path / "node_modules" / ".bin").mkdir()
|
||||
artifact = tmp_path / "review-output.md"
|
||||
artifact.write_text("new review")
|
||||
|
||||
runner_artifacts.enforce_phase_workspace(tmp_path, before, allowed_artifact=artifact)
|
||||
|
||||
|
||||
def test_phase_workspace_still_rejects_a_genuinely_unauthorized_change(tmp_path):
|
||||
# The bootstrap-noise exclusion must stay narrow: an actual source-file
|
||||
# edit outside the allowed artifact still has to be caught.
|
||||
before = runner_artifacts.workspace_snapshot(tmp_path)
|
||||
(tmp_path / "src.py").write_text("changed")
|
||||
artifact = tmp_path / "review-output.md"
|
||||
artifact.write_text("new review")
|
||||
|
||||
with pytest.raises(ValueError, match="unauthorized workspace path"):
|
||||
runner_artifacts.enforce_phase_workspace(tmp_path, before, allowed_artifact=artifact)
|
||||
|
||||
|
||||
def test_phase_workspace_ignores_nested_claude_sandbox_bootstrap_noise(tmp_path):
|
||||
# Claude Code bootstraps into whatever directory it is running in, not just
|
||||
# the workspace root. The benchmark's task prompts cd into gitnexus/, so the
|
||||
# same noise lands one level down -- observed verbatim in skill-evolution run
|
||||
# 29861768554, where 13 of 18 sessions failed with
|
||||
# "phase changed unauthorized workspace path(s): gitnexus/.claude/.cc-writes".
|
||||
nested = tmp_path / "gitnexus" / ".claude"
|
||||
nested.mkdir(parents=True)
|
||||
(nested / "settings.local.json").write_text("{}")
|
||||
before = runner_artifacts.workspace_snapshot(tmp_path)
|
||||
(nested / ".cc-writes").write_text("{}")
|
||||
artifact = tmp_path / "review-output.md"
|
||||
artifact.write_text("new review")
|
||||
|
||||
runner_artifacts.enforce_phase_workspace(tmp_path, before, allowed_artifact=artifact)
|
||||
|
||||
|
||||
def test_phase_workspace_does_not_descend_into_nested_bootstrap_directories(tmp_path):
|
||||
# The exclusion must skip an entry before it is queued for traversal, so
|
||||
# content created *inside* the ignored directory stays invisible too.
|
||||
nested = tmp_path / "gitnexus" / ".claude" / ".cc-writes"
|
||||
nested.mkdir(parents=True)
|
||||
before = runner_artifacts.workspace_snapshot(tmp_path)
|
||||
(nested / "pending.json").write_text('{"writes": 1}')
|
||||
artifact = tmp_path / "review-output.md"
|
||||
artifact.write_text("new review")
|
||||
|
||||
runner_artifacts.enforce_phase_workspace(tmp_path, before, allowed_artifact=artifact)
|
||||
|
||||
|
||||
def test_phase_workspace_still_rejects_nested_real_claude_config(tmp_path):
|
||||
# gitnexus/.claude/settings.local.json is real tracked repository content.
|
||||
# Excluding ".claude" wholesale at depth would blind the check to it, so the
|
||||
# exclusion must name only the entries Claude Code itself creates.
|
||||
nested = tmp_path / "gitnexus" / ".claude"
|
||||
nested.mkdir(parents=True)
|
||||
settings = nested / "settings.local.json"
|
||||
settings.write_text("{}")
|
||||
before = runner_artifacts.workspace_snapshot(tmp_path)
|
||||
settings.write_text('{"permissions": "changed"}')
|
||||
artifact = tmp_path / "review-output.md"
|
||||
artifact.write_text("new review")
|
||||
|
||||
with pytest.raises(ValueError, match="unauthorized workspace path"):
|
||||
runner_artifacts.enforce_phase_workspace(tmp_path, before, allowed_artifact=artifact)
|
||||
|
||||
|
||||
def test_phase_workspace_still_rejects_nested_package_json(tmp_path):
|
||||
# package.json is in WORKSPACE_SNAPSHOT_BOOTSTRAP_NOISE, but only as a
|
||||
# workspace-root entry: gitnexus/package.json is real tracked content whose
|
||||
# edits must still be caught.
|
||||
nested = tmp_path / "gitnexus"
|
||||
nested.mkdir()
|
||||
manifest = nested / "package.json"
|
||||
manifest.write_text("{}")
|
||||
before = runner_artifacts.workspace_snapshot(tmp_path)
|
||||
manifest.write_text('{"version": "9.9.9"}')
|
||||
artifact = tmp_path / "review-output.md"
|
||||
artifact.write_text("new review")
|
||||
|
||||
with pytest.raises(ValueError, match="unauthorized workspace path"):
|
||||
runner_artifacts.enforce_phase_workspace(tmp_path, before, allowed_artifact=artifact)
|
||||
|
||||
|
||||
def test_phase_workspace_still_sees_writes_under_a_pre_existing_nested_claude_dir(tmp_path):
|
||||
# Every excluded name is a blind spot. .claude/agents and .claude/commands
|
||||
# are deliberately NOT excluded at depth: once a .claude directory exists
|
||||
# (gitnexus/.claude/settings.local.json is tracked), anything written
|
||||
# underneath an excluded entry is invisible to this check, and Claude Code
|
||||
# loads .claude/agents relative to its cwd -- which these tasks point at
|
||||
# gitnexus/. A planning phase must not be able to plant a definition there
|
||||
# for the later work phase to read.
|
||||
nested = tmp_path / "gitnexus" / ".claude"
|
||||
nested.mkdir(parents=True)
|
||||
(nested / "settings.local.json").write_text("{}")
|
||||
before = runner_artifacts.workspace_snapshot(tmp_path)
|
||||
(nested / "agents").mkdir()
|
||||
(nested / "agents" / "planted.md").write_text("planted agent definition")
|
||||
artifact = tmp_path / "review-output.md"
|
||||
artifact.write_text("new review")
|
||||
|
||||
with pytest.raises(ValueError, match="unauthorized workspace path"):
|
||||
runner_artifacts.enforce_phase_workspace(tmp_path, before, allowed_artifact=artifact)
|
||||
|
||||
@@ -9,7 +9,7 @@ from pathlib import Path
|
||||
|
||||
import pytest
|
||||
|
||||
from workflow_bench.proposer_sandbox import VITE_TEMP_DIR, SandboxError
|
||||
from workflow_bench.proposer_sandbox import SandboxError
|
||||
from workflow_bench.oracle_assets import TaskOracleSnapshot
|
||||
from workflow_bench.runner_tasks import resolve_task_bindings
|
||||
from workflow_bench.task_assets import TaskAssetCache, stage_task_assets
|
||||
@@ -113,27 +113,6 @@ def test_small_assets_use_a_bounded_buffered_fallback(monkeypatch, tmp_path: Pat
|
||||
assert (clone / "second").read_bytes() == b"def"
|
||||
|
||||
|
||||
def test_default_buffered_fallback_budget_covers_a_realistic_large_asset(
|
||||
monkeypatch,
|
||||
tmp_path: Path,
|
||||
) -> None:
|
||||
# 20 MiB exceeds the old 16 MiB default but must fit comfortably under
|
||||
# the current default, proving the real (non-monkeypatched) budget
|
||||
# constant is sized for a realistic large sandbox_copy asset such as the
|
||||
# harness's own pre-built graph index, not just tiny fixtures.
|
||||
payload = os.urandom(20 * 1024 * 1024)
|
||||
repo, task = _repo_and_task(tmp_path, {"large": payload})
|
||||
clone = tmp_path / "clone"
|
||||
clone.mkdir()
|
||||
monkeypatch.setattr(task_assets, "_try_reflink", lambda *_args: False)
|
||||
|
||||
with TaskAssetCache(tmp_path / "cache") as cache:
|
||||
snapshot = cache.prepare(task, repo=repo, resolved_sha=SHA)
|
||||
snapshot.materialize(clone)
|
||||
|
||||
assert (clone / "large").read_bytes() == payload
|
||||
|
||||
|
||||
def test_large_asset_without_reflink_fails_before_publish_and_cleans_staging(
|
||||
monkeypatch,
|
||||
tmp_path: Path,
|
||||
@@ -410,36 +389,3 @@ def test_resolved_task_binding_carries_dependency_digests_and_rejects_live_drift
|
||||
(repo / "dependency" / "package.json").write_bytes(b'{"version":2}')
|
||||
with pytest.raises(ValueError, match="definition drifted"):
|
||||
resolve_task_bindings([task], [binding], oracle_snapshots=[oracle])
|
||||
|
||||
|
||||
def test_node_modules_dependency_snapshot_captures_the_vite_temp_mount_point(tmp_path: Path) -> None:
|
||||
# bwrap cannot mkdir a mount point inside an already-read-only bind, so the
|
||||
# directory vite needs must exist in the captured dependency bytes. It is
|
||||
# recorded during capture, which puts it inside the manifest and both
|
||||
# dependency digests rather than leaving it an untracked mutation of a
|
||||
# digest-bound snapshot.
|
||||
repo, _ = _repo_and_task(tmp_path, {"dependency/package.json": b'{"version":1}'})
|
||||
task = {
|
||||
"sandbox_copy": [],
|
||||
"sandbox_dependencies": [{"source": "dependency", "target": "gitnexus/node_modules"}],
|
||||
}
|
||||
with TaskAssetCache(tmp_path / "cache") as cache:
|
||||
snapshot = cache.prepare(task, repo=repo, resolved_sha=SHA)
|
||||
captured = {entry.path.as_posix() for entry in snapshot.dependencies[0].entries}
|
||||
assert f"payload/{VITE_TEMP_DIR}" in captured
|
||||
vite_temp = next((snapshot.root / "dependencies").glob(f"*/payload/{VITE_TEMP_DIR}"))
|
||||
assert vite_temp.is_dir()
|
||||
|
||||
|
||||
def test_non_node_modules_dependency_snapshot_has_no_vite_temp(tmp_path: Path) -> None:
|
||||
# The capture is scoped to dependency mounts whose target is node_modules;
|
||||
# an unrelated vendored dependency is captured byte-for-byte as declared.
|
||||
repo, _ = _repo_and_task(tmp_path, {"dependency/package.json": b'{"version":1}'})
|
||||
task = {
|
||||
"sandbox_copy": [],
|
||||
"sandbox_dependencies": [{"source": "dependency", "target": "vendor/dependency"}],
|
||||
}
|
||||
with TaskAssetCache(tmp_path / "cache") as cache:
|
||||
snapshot = cache.prepare(task, repo=repo, resolved_sha=SHA)
|
||||
captured = {entry.path.as_posix() for entry in snapshot.dependencies[0].entries}
|
||||
assert not any(path.endswith(VITE_TEMP_DIR) for path in captured)
|
||||
|
||||
@@ -10,7 +10,6 @@ import yaml
|
||||
|
||||
from workflow_bench.runner import (
|
||||
aggregate,
|
||||
broken_incumbent_arms,
|
||||
build_parser,
|
||||
infra_error_record,
|
||||
normalized_model_identifier,
|
||||
@@ -65,7 +64,6 @@ def test_aggregate_takes_medians_and_counts_resolved():
|
||||
"valid_runs": 3,
|
||||
"excluded_runs": 0,
|
||||
"transcripts_missing": 0,
|
||||
"error_kinds": {},
|
||||
}
|
||||
|
||||
|
||||
@@ -174,7 +172,7 @@ def test_eval_ci_uses_locked_uv_and_blocking_native_containment_jobs():
|
||||
}
|
||||
assert containment["timeout-minutes"] == 20
|
||||
assert containment_node_setup["with"] == {
|
||||
"node-version": "22.18.0",
|
||||
"node-version": "22.16.0",
|
||||
"cache": "npm",
|
||||
"cache-dependency-path": "gitnexus/package-lock.json\ngitnexus-shared/package-lock.json\n",
|
||||
}
|
||||
@@ -335,61 +333,6 @@ def test_render_report_surfaces_excluded_and_unverified_runs():
|
||||
assert "no locatable session transcript" in report
|
||||
|
||||
|
||||
def test_render_report_surfaces_why_each_row_failed():
|
||||
results = {
|
||||
"t": {
|
||||
"workflow": aggregate(
|
||||
[record(resolved=False, error_kind="plan-evidence-invalid")],
|
||||
),
|
||||
}
|
||||
}
|
||||
report = render_report(results)
|
||||
assert "plan-evidence-invalid×1" in report
|
||||
|
||||
|
||||
def test_broken_incumbent_arms_flags_an_incumbent_that_resolved_nothing():
|
||||
results = {
|
||||
"t1": {"workflow": aggregate([record(resolved=False, error_kind="plan-evidence-invalid")])},
|
||||
"t2": {"workflow": aggregate([record(resolved=False, error_kind="plan-evidence-invalid")])},
|
||||
}
|
||||
assert broken_incumbent_arms(results, {"workflow"}) == ["workflow"]
|
||||
|
||||
|
||||
def test_broken_incumbent_arms_ignores_a_merely_underperforming_candidate():
|
||||
# The incumbent works fine; only the candidate arm fails. That's a normal,
|
||||
# expected "bad candidate" outcome and must not read as a broken harness.
|
||||
results = {
|
||||
"t1": {
|
||||
"workflow": aggregate([record(resolved=True)]),
|
||||
"candidate_workflow": aggregate([record(resolved=False, error_kind="verify-failed")]),
|
||||
},
|
||||
}
|
||||
assert broken_incumbent_arms(results, {"workflow"}) == []
|
||||
|
||||
|
||||
def test_broken_incumbent_arms_flags_an_incumbent_with_zero_valid_runs():
|
||||
# Every run excluded via an excluded-but-non-systemic error_kind
|
||||
# ("evidence-unverified"): valid_runs == 0 for every task, which the old
|
||||
# `valid_runs > 0` guard let sail through silently, and which the outage
|
||||
# streak breaker also doesn't catch (it resets rather than accumulates
|
||||
# on this exact error_kind -- see test_systemic_outage_streak_resets_on_non_outage).
|
||||
results = {
|
||||
"t1": {"workflow": aggregate([record(resolved=False, error_kind="evidence-unverified")])},
|
||||
"t2": {"workflow": aggregate([record(resolved=False, error_kind="evidence-unverified")])},
|
||||
}
|
||||
assert results["t1"]["workflow"]["valid_runs"] == 0
|
||||
assert broken_incumbent_arms(results, {"workflow"}) == ["workflow"]
|
||||
|
||||
|
||||
def test_broken_incumbent_arms_ignores_partial_incumbent_failure():
|
||||
# Resolved in at least one task — struggling, not broken.
|
||||
results = {
|
||||
"t1": {"workflow": aggregate([record(resolved=False, error_kind="verify-failed")])},
|
||||
"t2": {"workflow": aggregate([record(resolved=True)])},
|
||||
}
|
||||
assert broken_incumbent_arms(results, {"workflow"}) == []
|
||||
|
||||
|
||||
def test_infra_error_record_captures_the_failure_and_is_excluded():
|
||||
exc = subprocess.TimeoutExpired(cmd="claude -p", timeout=5)
|
||||
rec = infra_error_record(exc)
|
||||
|
||||
@@ -167,56 +167,6 @@ def test_run_claude_forwards_the_named_model_to_every_session(monkeypatch, tmp_p
|
||||
assert captured[captured.index("--model") + 1] == "claude-sonnet-4-20250514"
|
||||
|
||||
|
||||
def test_run_claude_restricts_tools_via_tools_flag_outside_bare(monkeypatch, tmp_path):
|
||||
# Outside --bare, the built-in toolset defaults to everything (subagents,
|
||||
# WebFetch, Task, ...) and --allowedTools only pre-approves within that —
|
||||
# it does not narrow it. --tools is what actually restricts the set, so a
|
||||
# non-bare arm session must pass it or it silently gets a far wider
|
||||
# toolset than intended.
|
||||
captured: list[str] = []
|
||||
|
||||
def fake_run(command, **kwargs):
|
||||
captured.extend(command)
|
||||
return fake_cli_result(VALID_REPORT)
|
||||
|
||||
monkeypatch.setattr(runner_sessions, "run_managed", fake_run)
|
||||
runner.run_claude(
|
||||
"task",
|
||||
tmp_path,
|
||||
claude_bin="claude",
|
||||
timeout=5,
|
||||
bare=False,
|
||||
allowed_tools=["Read", "Edit", "Bash", "Skill"],
|
||||
)
|
||||
tools_idx = captured.index("--tools")
|
||||
assert captured[tools_idx + 1 : tools_idx + 5] == ["Read", "Edit", "Bash", "Skill"]
|
||||
allowed_idx = captured.index("--allowedTools")
|
||||
assert captured[allowed_idx + 1 : allowed_idx + 5] == ["Read", "Edit", "Bash", "Skill"]
|
||||
|
||||
|
||||
def test_run_claude_omits_tools_flag_under_bare(monkeypatch, tmp_path):
|
||||
# --bare already hard-restricts to Bash/Edit/Read on its own (a Claude
|
||||
# Code design choice, not something --tools/--allowedTools can widen or
|
||||
# narrow further), so bare sessions must not also pass --tools.
|
||||
captured: list[str] = []
|
||||
|
||||
def fake_run(command, **kwargs):
|
||||
captured.extend(command)
|
||||
return fake_cli_result(VALID_REPORT)
|
||||
|
||||
monkeypatch.setattr(runner_sessions, "run_managed", fake_run)
|
||||
runner.run_claude(
|
||||
"task",
|
||||
tmp_path,
|
||||
claude_bin="claude",
|
||||
timeout=5,
|
||||
bare=True,
|
||||
allowed_tools=["Read", "Edit", "Bash", "Skill"],
|
||||
)
|
||||
assert "--tools" not in captured
|
||||
assert "--allowedTools" in captured
|
||||
|
||||
|
||||
@pytest.mark.parametrize(
|
||||
("proc", "expected_kind"),
|
||||
[
|
||||
@@ -332,15 +282,6 @@ def test_agent_tool_grants_are_exact_and_nomcp_has_no_graph_tools(monkeypatch, t
|
||||
assert captured[3]["mcp_config_json"] == '{"mcpServers":{}}'
|
||||
assert captured[3]["disallowed_tools"] == ["Skill", "mcp__gitnexus"]
|
||||
|
||||
# --bare hard-disables the Skill tool and every mcp__* tool regardless of
|
||||
# --allowedTools (a Claude Code design choice, not something the harness
|
||||
# can override) -- every arm here except baseline_nomcp needs Skill
|
||||
# and/or MCP tools, so only baseline_nomcp may still run under --bare.
|
||||
assert captured[0]["bare"] is False # workflow: planning session
|
||||
assert captured[1]["bare"] is False # review
|
||||
assert captured[2]["bare"] is False # workflow_direct
|
||||
assert captured[3]["bare"] is True # baseline_nomcp
|
||||
|
||||
|
||||
def test_mcp_config_uses_only_the_minimal_pinned_harness_runtime(monkeypatch, tmp_path):
|
||||
runtime = tmp_path / "gitnexus"
|
||||
@@ -349,12 +290,10 @@ def test_mcp_config_uses_only_the_minimal_pinned_harness_runtime(monkeypatch, tm
|
||||
runtime / "dist" / "cli",
|
||||
runtime / "node_modules",
|
||||
runtime / "vendor",
|
||||
runtime / "hooks" / "claude",
|
||||
shared / "dist",
|
||||
):
|
||||
directory.mkdir(parents=True)
|
||||
(runtime / "dist" / "cli" / "index.js").write_text("")
|
||||
(runtime / "hooks" / "claude" / "resolve-analyze-cmd.cjs").write_text("")
|
||||
(runtime / "package.json").write_text(json.dumps({"version": runner.PINNED_GITNEXUS_VERSION}))
|
||||
(runtime / "node_modules" / "gitnexus-shared").symlink_to(shared, target_is_directory=True)
|
||||
(shared / "package.json").write_text(json.dumps({"name": "gitnexus-shared"}))
|
||||
@@ -377,7 +316,6 @@ def test_mcp_config_uses_only_the_minimal_pinned_harness_runtime(monkeypatch, tm
|
||||
(runtime / "vendor", f"{runner.SANDBOX_GITNEXUS}/vendor"),
|
||||
(shared / "dist", f"{runner.SANDBOX_GITNEXUS_SHARED}/dist"),
|
||||
(shared / "package.json", f"{runner.SANDBOX_GITNEXUS_SHARED}/package.json"),
|
||||
(runtime / "hooks" / "claude", f"{runner.SANDBOX_GITNEXUS}/hooks/claude"),
|
||||
]
|
||||
package = json.loads((runtime / "package.json").read_text())
|
||||
assert package["version"] == runner.PINNED_GITNEXUS_VERSION
|
||||
@@ -392,12 +330,6 @@ def test_mcp_config_uses_only_the_minimal_pinned_harness_runtime(monkeypatch, tm
|
||||
assert shared / forbidden not in mounted_sources
|
||||
assert f"{runner.SANDBOX_GITNEXUS_SHARED}/{forbidden}" not in mounted_targets
|
||||
|
||||
# Only hooks/claude is exposed, not the whole hooks/ directory (which also
|
||||
# has an unrelated hooks/antigravity/ tree) and not the runtime root itself.
|
||||
assert runtime / "hooks" not in mounted_sources
|
||||
assert runtime / "hooks" / "antigravity" not in mounted_sources
|
||||
assert f"{runner.SANDBOX_GITNEXUS}/hooks" not in mounted_targets
|
||||
|
||||
|
||||
@pytest.mark.skipif(
|
||||
os.environ.get("GITNEXUS_REQUIRE_BWRAP_CANARY") != "1",
|
||||
@@ -415,7 +347,6 @@ def test_real_bubblewrap_runtime_mount_imports_cli_without_exposing_checkout(tmp
|
||||
f"{runner.SANDBOX_GITNEXUS}/vendor",
|
||||
f"{runner.SANDBOX_GITNEXUS_SHARED}/dist/index.js",
|
||||
f"{runner.SANDBOX_GITNEXUS_SHARED}/package.json",
|
||||
f"{runner.SANDBOX_GITNEXUS}/hooks/claude/resolve-analyze-cmd.cjs",
|
||||
]
|
||||
forbidden = [
|
||||
f"{runner.SANDBOX_GITNEXUS}/{relative}"
|
||||
@@ -438,26 +369,16 @@ def test_real_bubblewrap_runtime_mount_imports_cli_without_exposing_checkout(tmp
|
||||
preflight=True,
|
||||
) as sandbox:
|
||||
visibility = sandbox.run(
|
||||
[runner.SANDBOX_NODE, "-e", visibility_script],
|
||||
["/usr/local/bin/node", "-e", visibility_script],
|
||||
timeout=10,
|
||||
)
|
||||
imported = sandbox.run(
|
||||
[runner.SANDBOX_NODE, runner.SANDBOX_GITNEXUS_ENTRYPOINT, "--version"],
|
||||
timeout=10,
|
||||
)
|
||||
# --version never reaches the `analyze` command, which is loaded via a
|
||||
# lazy dynamic import and is the only path that pulls in
|
||||
# resolve-invocation.ts's module-load-time require of hooks/claude/
|
||||
# resolve-analyze-cmd.cjs. Require the compiled analyze module
|
||||
# directly so this canary actually exercises that chain.
|
||||
analyze_imported = sandbox.run(
|
||||
[runner.SANDBOX_NODE, "-e", f"require('{runner.SANDBOX_GITNEXUS}/dist/cli/analyze.js')"],
|
||||
["/usr/local/bin/node", runner.SANDBOX_GITNEXUS_ENTRYPOINT, "--version"],
|
||||
timeout=10,
|
||||
)
|
||||
|
||||
assert visibility.ok, visibility.stderr_tail
|
||||
assert imported.ok, imported.stderr_tail
|
||||
assert analyze_imported.ok, analyze_imported.stderr_tail
|
||||
assert imported.stdout_tail.strip() == runner.PINNED_GITNEXUS_VERSION
|
||||
|
||||
|
||||
|
||||
@@ -26,18 +26,7 @@ SANDBOX_HOME = "/home/agent"
|
||||
SANDBOX_TMP = "/tmp"
|
||||
SANDBOX_CLAUDE = "/opt/claude/claude"
|
||||
SANDBOX_SHELL_PREFIX = "/opt/claude/shell-prefix"
|
||||
SANDBOX_PYTHON3 = "/opt/claude/python3"
|
||||
SANDBOX_NODE = "/opt/claude/node"
|
||||
SANDBOX_NODE_PREFIX = "/opt/claude/nodejs"
|
||||
# Vite transpiles a TypeScript config into <node_modules>/.vite-temp before it
|
||||
# loads anything, so a read-only dependency mount makes `vitest` die with EROFS
|
||||
# before a single test runs -- and every task verify command and every hidden
|
||||
# oracle ends in `npx vitest run <test>`. bwrap cannot create a mount point
|
||||
# inside an already-read-only bind, so the directory is captured into the
|
||||
# dependency snapshot (task_assets.py) and a tmpfs is overlaid on it here.
|
||||
VITE_TEMP_DIR = ".vite-temp"
|
||||
DEPENDENCY_MOUNT_BASENAME = "node_modules"
|
||||
SANDBOX_PATH = f"/opt/claude:{SANDBOX_NODE_PREFIX}/bin:/usr/local/bin:/usr/bin:/bin"
|
||||
SANDBOX_PATH = "/opt/claude:/usr/local/bin:/usr/bin:/bin"
|
||||
SANDBOX_GITNEXUS = "/opt/gitnexus"
|
||||
SANDBOX_GITNEXUS_SHARED = "/opt/gitnexus-shared"
|
||||
SANDBOX_GITNEXUS_REGISTRY = "/opt/gitnexus-registry"
|
||||
@@ -360,58 +349,10 @@ def build_claude_settings() -> str:
|
||||
|
||||
def _runtime_mount_args() -> list[str]:
|
||||
args: list[str] = []
|
||||
system_trees = ("/usr", "/bin", "/lib", "/lib64")
|
||||
for raw in system_trees:
|
||||
for raw in ("/usr", "/bin", "/lib", "/lib64"):
|
||||
path = Path(raw)
|
||||
if path.exists():
|
||||
args += ["--ro-bind", raw, raw]
|
||||
# sanitized_graph.py and runner_sessions.py invoke the sandboxed graph
|
||||
# CLI via SANDBOX_NODE. Bind whatever `node` actually resolves to on PATH
|
||||
# there -- true node location varies by host (GitHub-hosted runner images
|
||||
# happen to have one under /usr/local/bin; a self-hosted runner's
|
||||
# actions/setup-node installs into its own tool-cache directory instead).
|
||||
# Target must be a fresh path like /opt/claude/... rather than anywhere
|
||||
# under /usr, /bin, /lib, or /lib64: those are already read-only bound
|
||||
# above, and bwrap can't create a new mount-point file inside an
|
||||
# already-read-only tree when the real path doesn't already exist there
|
||||
# (the exact case a self-hosted runner hits, and the reason this bind
|
||||
# exists at all).
|
||||
node_bin = shutil.which("node")
|
||||
if node_bin:
|
||||
args += ["--ro-bind", node_bin, SANDBOX_NODE]
|
||||
# The single-binary bind above gives SANDBOX_NODE but NOT npm or npx:
|
||||
# those are symlinks into ../lib/node_modules/npm/bin/*-cli.js, so the
|
||||
# install prefix carrying both bin/ and lib/node_modules has to be
|
||||
# mounted for them to resolve at all. When node really lives under a
|
||||
# system tree (/usr/local/bin on GitHub-hosted images) the prefix is
|
||||
# already inside the wholesale read-only binds above and npm/npx came
|
||||
# along for free -- which is exactly why this gap stayed invisible
|
||||
# until a self-hosted runner put node in actions/setup-node's tool
|
||||
# cache, outside /usr, and every task verify command
|
||||
# ("cd gitnexus && npx tsc ... && npx vitest ...") died with
|
||||
# "/bin/sh: 1: npx: not found". Skip the redundant bind in the
|
||||
# already-covered case so the mount surface stays minimal.
|
||||
#
|
||||
# The prefix is only ever derived from a real <prefix>/bin/node layout
|
||||
# that actually carries npm. Deriving it as parent.parent unconditionally
|
||||
# would mount an unrelated ancestor whenever node sits somewhere else:
|
||||
# /opt/bin/node would bind all of /opt (every tool cache on a hosted
|
||||
# runner) and a bare <dir>/node would bind <dir>'s parent. This function
|
||||
# exists to keep the sandbox surface minimal, so an unrecognized layout
|
||||
# binds nothing extra and simply leaves npx unavailable, exactly as
|
||||
# before.
|
||||
node_bin_dir = Path(node_bin).resolve().parent
|
||||
node_prefix = node_bin_dir.parent
|
||||
# Test the property actually needed -- a working npx next to node in a
|
||||
# real bin/ directory -- rather than a proxy like lib/node_modules/npm.
|
||||
# .exists() follows the symlink, so a dangling npx correctly fails: it
|
||||
# would not survive the mount either. Requiring the "bin" name keeps
|
||||
# the parent.parent derivation honest; an npx sitting directly beside
|
||||
# node in a flat directory would make that derivation name the wrong
|
||||
# prefix.
|
||||
provides_npx = node_bin_dir.name == "bin" and (node_bin_dir / "npx").exists()
|
||||
if provides_npx and not any(node_prefix.is_relative_to(tree) for tree in system_trees):
|
||||
args += ["--ro-bind", str(node_prefix), SANDBOX_NODE_PREFIX]
|
||||
for raw in (
|
||||
"/etc/ssl",
|
||||
"/etc/hosts",
|
||||
@@ -443,24 +384,6 @@ def _create_shell_prefix_wrapper(private_root: Path) -> Path:
|
||||
return wrapper
|
||||
|
||||
|
||||
def _create_python3_wrapper(private_root: Path) -> Path:
|
||||
"""A trusted, self-owned Python 3 launcher for evidence-provenance.mjs's atomic mover.
|
||||
|
||||
/usr/bin/python3 is a real system binary, but it's root-owned on the host.
|
||||
Inside this --unshare-user sandbox only the calling uid is mapped (root is
|
||||
not), so root-owned files surface as the kernel's overflow uid — which
|
||||
evidence-provenance.mjs's PATH-scan correctly refuses to trust. This
|
||||
wrapper is freshly created by the same host process that owns
|
||||
home/temp/shell-prefix, so it maps to the sandbox's own trusted uid
|
||||
instead, and simply execs the real interpreter through to do the work.
|
||||
"""
|
||||
|
||||
wrapper = private_root / "python3"
|
||||
wrapper.write_text('#!/bin/bash\nset -eu\nexec /usr/bin/python3 "$@"\n')
|
||||
wrapper.chmod(0o500)
|
||||
return wrapper
|
||||
|
||||
|
||||
def _resolve_executable(executable: Path | str | None, default: str) -> Path:
|
||||
raw = os.fspath(executable) if executable is not None else shutil.which(default)
|
||||
if not raw:
|
||||
@@ -686,20 +609,6 @@ def _sandbox_command_prefix(
|
||||
]
|
||||
for mount in mounts:
|
||||
args += ["--ro-bind", str(mount.source), mount.target]
|
||||
# Overlay an empty writable tmpfs on the one path vite must write.
|
||||
# Everything else in the mount, and the whole workspace, stays
|
||||
# read-only, and the overlay lives only inside the sandbox -- it never
|
||||
# reaches the host clone the credited patch is captured from.
|
||||
#
|
||||
# Gate on the mount SOURCE actually containing the directory, not on
|
||||
# the target name: bwrap cannot create a mount point inside an
|
||||
# already-read-only bind, so a tmpfs can only be overlaid where the
|
||||
# directory already exists in the bound bytes. task_assets.py captures
|
||||
# it into dependency-snapshot node_modules; other node_modules mounts
|
||||
# (e.g. the trusted GitNexus runtime at /opt/gitnexus/node_modules) do
|
||||
# not carry it, and overlaying them would fail with EROFS.
|
||||
if PurePosixPath(mount.target).name == DEPENDENCY_MOUNT_BASENAME and (mount.source / VITE_TEMP_DIR).is_dir():
|
||||
args += ["--tmpfs", f"{mount.target}/{VITE_TEMP_DIR}"]
|
||||
args += ["--chdir", SANDBOX_WORKSPACE, "--"]
|
||||
return args
|
||||
|
||||
@@ -732,7 +641,6 @@ def prepare_sandbox(
|
||||
directory.mkdir(mode=0o700)
|
||||
directory.chmod(0o700)
|
||||
shell_prefix = _create_shell_prefix_wrapper(private_root)
|
||||
python3_wrapper = _create_python3_wrapper(private_root)
|
||||
# Claude may discover user-level skills below HOME. Keep the rest of HOME
|
||||
# writable for normal CLI state, but overlay an immutable empty skills root
|
||||
# so a model cannot shadow the evaluated repository/plugin skill by name.
|
||||
@@ -743,7 +651,6 @@ def prepare_sandbox(
|
||||
*read_only_mounts,
|
||||
ReadOnlyMount(source=user_skills, target=SANDBOX_USER_SKILLS),
|
||||
ReadOnlyMount(source=shell_prefix, target=SANDBOX_SHELL_PREFIX),
|
||||
ReadOnlyMount(source=python3_wrapper, target=SANDBOX_PYTHON3),
|
||||
)
|
||||
primary: BaseException | None = None
|
||||
try:
|
||||
|
||||
@@ -78,7 +78,6 @@ from .proposer_sandbox import (
|
||||
SANDBOX_GITNEXUS as SANDBOX_GITNEXUS,
|
||||
SANDBOX_GITNEXUS_REGISTRY,
|
||||
SANDBOX_GITNEXUS_SHARED as SANDBOX_GITNEXUS_SHARED,
|
||||
SANDBOX_NODE as SANDBOX_NODE,
|
||||
SANDBOX_WORKSPACE,
|
||||
ReadOnlyMount,
|
||||
SandboxError,
|
||||
@@ -403,13 +402,6 @@ def run_arm(
|
||||
auth_token=args.auth_token,
|
||||
base_url=args.base_url,
|
||||
)
|
||||
# --bare hard-disables the Skill tool and every mcp__* tool — by Claude
|
||||
# Code design, not a bug (--allowedTools can't restore what --bare
|
||||
# removes). Every arm except baseline_nomcp needs Skill and/or MCP tools,
|
||||
# so only baseline_nomcp can keep --bare's tighter isolation; the rest
|
||||
# rely on ANTHROPIC_API_KEY alone (the sandboxed HOME has no OAuth/
|
||||
# keychain state to conflict with it).
|
||||
bare = arm == "baseline_nomcp"
|
||||
common = {
|
||||
"claude_bin": sandbox.claude_bin,
|
||||
"timeout": args.timeout,
|
||||
@@ -420,7 +412,7 @@ def run_arm(
|
||||
read_only_paths=_evaluated_skill_roots(worktree, arm),
|
||||
),
|
||||
"require_pid_namespace": True,
|
||||
"bare": bare,
|
||||
"bare": True,
|
||||
"settings_json": sandbox.settings_json,
|
||||
"strict_mcp_config": True,
|
||||
"mcp_config_json": sandbox_mcp_config(),
|
||||
@@ -680,21 +672,13 @@ def aggregate(records: list[dict[str, Any]]) -> dict[str, Any]:
|
||||
# unmeasured run makes the whole median unavailable so the gate won't rank
|
||||
# a candidate on a cost that was never actually captured.
|
||||
valid_costs = [r.get("cost_usd") for r in valid]
|
||||
out["cost_usd"] = (
|
||||
None if (not valid or any(cost is None for cost in valid_costs)) else statistics.median(valid_costs)
|
||||
)
|
||||
out["cost_usd"] = None if (not valid or any(cost is None for cost in valid_costs)) else statistics.median(valid_costs)
|
||||
out["resolved"] = sum(1 for r in records if r["resolved"])
|
||||
out["runs"] = len(records)
|
||||
out["valid_runs"] = len(valid)
|
||||
out["excluded_runs"] = len(records) - len(valid)
|
||||
out["transcripts_missing"] = sum(1 for r in records if r.get("transcript_missing"))
|
||||
out["class"] = records[0].get("class", "")
|
||||
error_kinds: dict[str, int] = {}
|
||||
for r in records:
|
||||
kind = r.get("error_kind")
|
||||
if kind:
|
||||
error_kinds[kind] = error_kinds.get(kind, 0) + 1
|
||||
out["error_kinds"] = error_kinds
|
||||
return out
|
||||
|
||||
|
||||
@@ -711,33 +695,6 @@ def savings(baseline: dict[str, Any], workflow: dict[str, Any]) -> dict[str, Any
|
||||
return out
|
||||
|
||||
|
||||
def broken_incumbent_arms(
|
||||
results: dict[str, dict[str, dict[str, Any]]],
|
||||
incumbent_arms: set[str],
|
||||
) -> list[str]:
|
||||
"""Incumbent arms that resolved nothing across every task they ran.
|
||||
|
||||
An incumbent arm is the currently-shipped, presumably-working skill: if it
|
||||
resolves NOTHING across every task it ran, that reads as an environment or
|
||||
harness failure (missing trusted interpreter, stale skill fingerprint,
|
||||
sandbox misconfiguration), not a skill regression. A candidate merely
|
||||
underperforming is a normal, expected outcome and must not trip this —
|
||||
only checking incumbents keeps that distinction.
|
||||
|
||||
Deliberately does NOT require valid_runs > 0 per task: an incumbent that
|
||||
fails every run with an excluded-but-non-systemic error_kind (e.g.
|
||||
"evidence-unverified", which the outage-streak breaker explicitly resets
|
||||
on rather than accumulates) would otherwise never accumulate a single
|
||||
valid run and sail through silently — the exact "quiet no-promotion"
|
||||
outcome this guard exists to catch, and arguably worse than the
|
||||
some-runs-resolved-zero case since here nothing completed at all.
|
||||
aggregate() never marks an excluded/unverifiable row resolved=True, so
|
||||
resolved == 0 alone already covers both cases.
|
||||
"""
|
||||
present = incumbent_arms & {arm for arms in results.values() for arm in arms}
|
||||
return sorted(arm for arm in present if all(arms[arm]["resolved"] == 0 for arms in results.values() if arm in arms))
|
||||
|
||||
|
||||
def _na(value: Any) -> Any:
|
||||
"""Render an unmeasured metric as ``n/a`` instead of a misleading number."""
|
||||
return "n/a" if value is None else value
|
||||
@@ -762,8 +719,8 @@ def render_report(results: dict[str, dict[str, dict[str, Any]]]) -> str:
|
||||
"efficiency, sum usage from the session transcripts instead",
|
||||
"(dedup events sharing one message.id).",
|
||||
"",
|
||||
"| task | class | arm | resolved | input | cache_create | cache_read | output | cost $ | wall s | turns | churn | errors |",
|
||||
"| --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- |",
|
||||
"| task | class | arm | resolved | input | cache_create | cache_read | output | cost $ | wall s | turns | churn |",
|
||||
"| --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- | --- |",
|
||||
]
|
||||
for task_id, arms in results.items():
|
||||
for arm, agg in arms.items():
|
||||
@@ -771,14 +728,12 @@ def render_report(results: dict[str, dict[str, dict[str, Any]]]) -> str:
|
||||
resolved_cell = f"{agg['resolved']}/{agg.get('valid_runs', agg['runs'])}"
|
||||
if excluded:
|
||||
resolved_cell += f" ({excluded} excluded)"
|
||||
error_cell = ", ".join(f"{kind}×{count}" for kind, count in sorted(agg.get("error_kinds", {}).items()))
|
||||
lines.append(
|
||||
f"| {task_id} | {agg['class']} | {arm} | {resolved_cell} "
|
||||
f"| {agg['input_tokens']:.0f} | {agg['cache_creation_input_tokens']:.0f} "
|
||||
f"| {agg['cache_read_input_tokens']:.0f} | {agg['output_tokens']:.0f} "
|
||||
f"| {_cost_cell(agg['cost_usd'])} | {agg['duration_s']:.0f} | {agg['num_turns']:.0f} "
|
||||
f"| {agg['diff_files']:.0f}/+{agg['diff_insertions']:.0f}/−{agg['diff_deletions']:.0f} "
|
||||
f"| {error_cell} |"
|
||||
f"| {agg['diff_files']:.0f}/+{agg['diff_insertions']:.0f}/−{agg['diff_deletions']:.0f} |"
|
||||
)
|
||||
for arm in arms:
|
||||
if arm != "baseline" and "baseline" in arms:
|
||||
@@ -787,7 +742,7 @@ def render_report(results: dict[str, dict[str, dict[str, Any]]]) -> str:
|
||||
f"| {task_id} | {arms[arm]['class']} | **{arm} savings %** | — "
|
||||
f"| {s['input_tokens']} | {s['cache_creation_input_tokens']} "
|
||||
f"| {s['cache_read_input_tokens']} | {s['output_tokens']} "
|
||||
f"| {_na(s['cost_usd'])} | {s['duration_s']} | — | — | — |"
|
||||
f"| {_na(s['cost_usd'])} | {s['duration_s']} | — | — |"
|
||||
)
|
||||
lines.append("")
|
||||
all_aggs = [agg for arms in results.values() for agg in arms.values()]
|
||||
@@ -1371,17 +1326,6 @@ def main() -> None:
|
||||
}
|
||||
(out_dir / "promotion.json").write_text(json.dumps(promotion, indent=2) + "\n")
|
||||
print(f"\n{report}\n\nWritten to {out_dir}/")
|
||||
broken_incumbents = broken_incumbent_arms(results, set(CANDIDATE_ARMS.values()))
|
||||
if broken_incumbents:
|
||||
# Fail loudly rather than let a broken environment read as a quiet
|
||||
# "no promotion, incumbent stands."
|
||||
print(
|
||||
f"[harness-health] incumbent arm(s) {', '.join(broken_incumbents)} resolved zero "
|
||||
"tasks across every valid run — this looks like an environment/harness failure, "
|
||||
"not a normal candidate miss. See the errors column in report.md and error_detail "
|
||||
"in results.jsonl. Exiting non-zero rather than reporting a quiet no-promotion."
|
||||
)
|
||||
raise SystemExit(1)
|
||||
if outage_tripped:
|
||||
# Non-zero exit so a driver (evolve.py) treats the partial benchmark as a
|
||||
# failed run and halts instead of proposing from outage-truncated evidence.
|
||||
|
||||
@@ -21,64 +21,6 @@ MAX_WORKSPACE_SNAPSHOT_ENTRIES = 100_000
|
||||
MAX_WORKSPACE_SNAPSHOT_PATH_BYTES = 16 * 1024 * 1024
|
||||
MAX_WORKSPACE_SNAPSHOT_FILE_BYTES = 1024 * 1024 * 1024
|
||||
|
||||
# Claude Code's own enableWeakerNestedSandbox bootstrap creates these paths on
|
||||
# EVERY session regardless of task or model output -- reproduced empirically
|
||||
# with a trivial "say OK" prompt: a synthetic package.json/lockfiles/
|
||||
# node_modules, a full set of .env variants, and .claude/agents,
|
||||
# .claude/commands, .claude/.cc-writes. None of this is something the model
|
||||
# decided to write, so it must not count as an "unauthorized" workspace
|
||||
# change during the planning-phase boundary check (the one thing this
|
||||
# snapshot is used for -- see workspace_snapshot's callers). Mirrors the
|
||||
# pre-existing .git exclusion below, which is the same kind of harness/tool
|
||||
# noise rather than substantive diff.
|
||||
WORKSPACE_SNAPSHOT_BOOTSTRAP_NOISE = frozenset(
|
||||
{
|
||||
".claude",
|
||||
".env",
|
||||
".env.development",
|
||||
".env.development.local",
|
||||
".env.local",
|
||||
".env.production",
|
||||
".env.production.local",
|
||||
".env.test",
|
||||
".env.test.local",
|
||||
".gitmodules",
|
||||
".npmrc",
|
||||
".yarnrc",
|
||||
".yarnrc.yml",
|
||||
"bunfig.toml",
|
||||
"node_modules",
|
||||
"package-lock.json",
|
||||
"package.json",
|
||||
"pnpm-lock.yaml",
|
||||
"yarn.lock",
|
||||
}
|
||||
)
|
||||
|
||||
# The set above is matched at the workspace ROOT only, because most of its
|
||||
# entries (package.json, node_modules, the .env family) are also legitimate
|
||||
# repository content further down the tree -- gitnexus/package.json and
|
||||
# gitnexus/.claude/settings.local.json are both tracked files whose edits must
|
||||
# still be caught. But Claude Code bootstraps into whatever directory it is
|
||||
# running in, so a task whose prompt cd's into a subdirectory gets the same
|
||||
# noise one level down. Observed in skill-evolution run 29861768554: 13 of 18
|
||||
# sessions failed with "phase changed unauthorized workspace path(s):
|
||||
# gitnexus/.claude/.cc-writes". That entry is matched at ANY depth -- never
|
||||
# ".claude" itself, which holds real configuration.
|
||||
#
|
||||
# Deliberately only .cc-writes. Every excluded name is a blind spot: once a
|
||||
# .claude directory already exists (gitnexus/.claude/settings.local.json is
|
||||
# tracked), anything a phase writes underneath an excluded entry becomes
|
||||
# invisible to this check, and Claude Code loads .claude/agents relative to
|
||||
# its cwd -- which these tasks point at gitnexus/. Adding "agents" and
|
||||
# "commands" here on the theory that they might also appear nested would let a
|
||||
# planning phase plant a definition that the later work phase reads, with no
|
||||
# evidence in the boundary check. Only .cc-writes was ever observed nested, so
|
||||
# only .cc-writes is excluded; extend this set from an observed failure, never
|
||||
# pre-emptively.
|
||||
CLAUDE_BOOTSTRAP_DIR = ".claude"
|
||||
CLAUDE_BOOTSTRAP_ENTRIES = frozenset({".cc-writes"})
|
||||
|
||||
IMPLEMENTATION_ARMS = frozenset(
|
||||
{
|
||||
"workflow",
|
||||
@@ -110,19 +52,8 @@ class VerificationResult:
|
||||
yield self.output
|
||||
|
||||
|
||||
def _is_bootstrap_noise(relative: PurePosixPath) -> bool:
|
||||
"""Report whether a walked entry is harness noise rather than workspace change."""
|
||||
|
||||
parts = relative.parts
|
||||
if parts[0] == ".git" or parts[0] in WORKSPACE_SNAPSHOT_BOOTSTRAP_NOISE:
|
||||
return True
|
||||
return len(parts) >= 2 and parts[-2] == CLAUDE_BOOTSTRAP_DIR and parts[-1] in CLAUDE_BOOTSTRAP_ENTRIES
|
||||
|
||||
|
||||
def workspace_snapshot(worktree: Path) -> dict[str, str]:
|
||||
"""Hash the workspace without following links, excluding Git internals
|
||||
and Claude Code's own sandbox-bootstrap noise (see
|
||||
WORKSPACE_SNAPSHOT_BOOTSTRAP_NOISE)."""
|
||||
"""Hash the workspace without following links, excluding Git internals."""
|
||||
|
||||
root = worktree.expanduser().absolute()
|
||||
mode = root.lstat().st_mode
|
||||
@@ -143,7 +74,7 @@ def workspace_snapshot(worktree: Path) -> dict[str, str]:
|
||||
raise ValueError(f"workspace snapshot directory is unreadable: {directory}: {exc}") from exc
|
||||
for entry in children:
|
||||
relative = relative_dir / entry.name
|
||||
if _is_bootstrap_noise(relative):
|
||||
if relative.parts[0] == ".git":
|
||||
continue
|
||||
entry_count += 1
|
||||
path_bytes += len(relative.as_posix().encode())
|
||||
|
||||
@@ -18,7 +18,6 @@ from .proposer_sandbox import (
|
||||
SANDBOX_GITNEXUS,
|
||||
SANDBOX_GITNEXUS_REGISTRY,
|
||||
SANDBOX_HOME,
|
||||
SANDBOX_NODE,
|
||||
SANDBOX_TMP,
|
||||
SANDBOX_WORKSPACE,
|
||||
SandboxError,
|
||||
@@ -51,8 +50,6 @@ def measured_cost(raw: Any) -> float | None:
|
||||
if not math.isfinite(raw) or raw < 0:
|
||||
return None
|
||||
return float(raw)
|
||||
|
||||
|
||||
SANDBOX_GITNEXUS_ENTRYPOINT = f"{SANDBOX_GITNEXUS}/dist/cli/index.js"
|
||||
SENSITIVE_EVENT_KEYS = frozenset(
|
||||
{
|
||||
@@ -116,7 +113,7 @@ def sandbox_mcp_config() -> str:
|
||||
"PATH=/usr/local/bin:/usr/bin:/bin",
|
||||
"LANG=C.UTF-8",
|
||||
"GIT_TERMINAL_PROMPT=0",
|
||||
SANDBOX_NODE,
|
||||
"/usr/local/bin/node",
|
||||
SANDBOX_GITNEXUS_ENTRYPOINT,
|
||||
"mcp",
|
||||
],
|
||||
@@ -380,13 +377,6 @@ def run_claude(
|
||||
if strict_mcp_config:
|
||||
cmd += ["--strict-mcp-config", "--mcp-config", mcp_config_json or '{"mcpServers":{}}']
|
||||
if allowed_tools:
|
||||
# --bare's own hard-coded Bash/Edit/Read ceiling already scopes bare
|
||||
# sessions; outside --bare the built-in toolset defaults to
|
||||
# everything (subagents, WebFetch, Task, ...), so --tools is needed
|
||||
# to actually restrict it — --allowedTools only pre-approves within
|
||||
# whatever set is available, it does not narrow that set.
|
||||
if not bare:
|
||||
cmd += ["--tools", *allowed_tools]
|
||||
cmd += ["--allowedTools", *allowed_tools]
|
||||
if disable_slash_commands:
|
||||
cmd.append("--disable-slash-commands")
|
||||
|
||||
@@ -217,12 +217,6 @@ def trusted_gitnexus_runtime_mounts() -> tuple[ReadOnlyMount, ...]:
|
||||
f"{SANDBOX_GITNEXUS_SHARED}/package.json",
|
||||
directory=False,
|
||||
),
|
||||
_validated_runtime_component(
|
||||
runtime,
|
||||
"hooks/claude",
|
||||
f"{SANDBOX_GITNEXUS}/hooks/claude",
|
||||
directory=True,
|
||||
),
|
||||
)
|
||||
|
||||
entrypoint = mounts[0].source / "cli" / "index.js"
|
||||
|
||||
@@ -16,7 +16,6 @@ from .process_control import ManagedProcessError, run_managed
|
||||
from .proposer_sandbox import (
|
||||
SANDBOX_GITNEXUS,
|
||||
SANDBOX_HOME,
|
||||
SANDBOX_NODE,
|
||||
SANDBOX_WORKSPACE,
|
||||
ReadOnlyMount,
|
||||
SandboxError,
|
||||
@@ -249,7 +248,7 @@ def _run_graph_cli(
|
||||
) -> bytes | None:
|
||||
command = [
|
||||
*prefix,
|
||||
SANDBOX_NODE,
|
||||
"/usr/local/bin/node",
|
||||
SANDBOX_GITNEXUS_ENTRYPOINT,
|
||||
*arguments,
|
||||
]
|
||||
|
||||
@@ -25,9 +25,7 @@ from pathlib import Path, PurePosixPath
|
||||
from typing import Any
|
||||
|
||||
from .proposer_sandbox import (
|
||||
DEPENDENCY_MOUNT_BASENAME,
|
||||
SANDBOX_WORKSPACE,
|
||||
VITE_TEMP_DIR,
|
||||
ReadOnlyMount,
|
||||
SandboxError,
|
||||
_prepare_clone_target,
|
||||
@@ -41,14 +39,9 @@ MAX_TASK_ASSET_ENTRIES = 100_000
|
||||
MAX_TASK_ASSET_PATH_BYTES = 4_096
|
||||
MAX_TASK_ASSET_BYTES = 2 * 1024 * 1024 * 1024
|
||||
|
||||
# The largest known real sandbox_copy asset in this harness is the shipped
|
||||
# index above (~428 MiB estimated, ~290 MiB measured); budget comfortably
|
||||
# above that so it can still materialize via buffered copy on a filesystem
|
||||
# that cannot reflink (ext4 CI runners, 9p-backed dev mounts), while staying
|
||||
# well below MAX_TASK_ASSET_BYTES so a genuinely oversized or malformed
|
||||
# declaration still fails closed instead of silently paying for a slow full
|
||||
# copy.
|
||||
MAX_BUFFERED_FALLBACK_BYTES = 512 * 1024 * 1024
|
||||
# A filesystem without reflink support may still run tiny fixtures. Large
|
||||
# assets fail closed instead of silently returning to one full copy per arm.
|
||||
MAX_BUFFERED_FALLBACK_BYTES = 16 * 1024 * 1024
|
||||
COPY_CHUNK_BYTES = 1024 * 1024
|
||||
|
||||
# linux/fs.h: #define FICLONE _IOW(0x94, 9, int)
|
||||
@@ -162,11 +155,9 @@ class TaskAssetSnapshot:
|
||||
source = snapshot_root / Path(*dependency.snapshot_path.parts)
|
||||
metadata = source.lstat()
|
||||
expected_directory = dependency.kind == "directory"
|
||||
if (
|
||||
stat.S_ISLNK(metadata.st_mode)
|
||||
or (expected_directory and not stat.S_ISDIR(metadata.st_mode))
|
||||
or (not expected_directory and not stat.S_ISREG(metadata.st_mode))
|
||||
):
|
||||
if stat.S_ISLNK(metadata.st_mode) or (
|
||||
expected_directory and not stat.S_ISDIR(metadata.st_mode)
|
||||
) or (not expected_directory and not stat.S_ISREG(metadata.st_mode)):
|
||||
raise SandboxError(f"dependency snapshot changed: {dependency.source}")
|
||||
target = PurePosixPath(dependency.target)
|
||||
_prepare_clone_target(
|
||||
@@ -217,7 +208,9 @@ class TaskAssetCache:
|
||||
repo_identity = _real_directory(repo, label="task asset repository")
|
||||
declarations, relative_paths = _sandbox_copy_declarations(task)
|
||||
dependency_declarations = _sandbox_dependency_declarations(task)
|
||||
dependency_identity = tuple((declaration.source, declaration.target) for declaration in dependency_declarations)
|
||||
dependency_identity = tuple(
|
||||
(declaration.source, declaration.target) for declaration in dependency_declarations
|
||||
)
|
||||
definition = (str(repo_identity), resolved_sha, declarations, dependency_identity)
|
||||
existing = self._by_definition.get(definition)
|
||||
if existing is not None:
|
||||
@@ -260,21 +253,6 @@ class TaskAssetCache:
|
||||
dependency_builder.copy_descriptor(descriptor, PurePosixPath("payload"))
|
||||
finally:
|
||||
os.close(descriptor)
|
||||
# vitest cannot start against a read-only node_modules: vite
|
||||
# writes <node_modules>/.vite-temp/<config>.timestamp-*.mjs
|
||||
# before loading a TypeScript config. bwrap cannot create
|
||||
# that mount point inside an already-read-only bind, so the
|
||||
# empty directory is captured here -- before the manifest and
|
||||
# both dependency digests are computed, so it is part of the
|
||||
# snapshot rather than an untracked mutation of it. The
|
||||
# sandbox overlays a tmpfs on it; see VITE_TEMP_DIR.
|
||||
payload_entry = dependency_builder.entries.get(PurePosixPath("payload"))
|
||||
if (
|
||||
payload_entry is not None
|
||||
and payload_entry.kind == "directory"
|
||||
and PurePosixPath(declaration.target).name == DEPENDENCY_MOUNT_BASENAME
|
||||
):
|
||||
dependency_builder.ensure_directory(PurePosixPath("payload") / VITE_TEMP_DIR)
|
||||
dependency_entries = dependency_builder.finished_entries()
|
||||
_validate_dependency_symlinks(
|
||||
container,
|
||||
@@ -479,14 +457,10 @@ class _SnapshotBuilder:
|
||||
destination = self.destination / Path(*relative.parts)
|
||||
os.symlink(target, destination)
|
||||
after = os.stat(name, dir_fd=parent_descriptor, follow_symlinks=False)
|
||||
if (
|
||||
_mutation_identity(before) != _mutation_identity(after)
|
||||
or os.readlink(
|
||||
name,
|
||||
dir_fd=parent_descriptor,
|
||||
)
|
||||
!= target
|
||||
):
|
||||
if _mutation_identity(before) != _mutation_identity(after) or os.readlink(
|
||||
name,
|
||||
dir_fd=parent_descriptor,
|
||||
) != target:
|
||||
raise SandboxError(f"dependency symlink changed while snapshotting: {relative}")
|
||||
self.total_bytes += len(target_bytes)
|
||||
self.budget.total_bytes += len(target_bytes)
|
||||
@@ -524,15 +498,6 @@ class _SnapshotBuilder:
|
||||
self.entries[entry.path] = entry
|
||||
self.budget.entries += 1
|
||||
|
||||
def ensure_directory(self, relative: PurePosixPath) -> None:
|
||||
"""Record and create one extra directory inside this snapshot.
|
||||
|
||||
Used for harness-owned mount points that must exist in the captured
|
||||
bytes rather than be created against a read-only bind at runtime.
|
||||
"""
|
||||
|
||||
self._record_directory(relative)
|
||||
|
||||
def finished_entries(self) -> tuple[AssetManifestEntry, ...]:
|
||||
return tuple(sorted(self.entries.values(), key=lambda entry: entry.path.as_posix()))
|
||||
|
||||
@@ -597,7 +562,9 @@ def _sandbox_dependency_declarations(
|
||||
or declaration.target_path in other.target_path.parents
|
||||
or other.target_path in declaration.target_path.parents
|
||||
):
|
||||
raise SandboxError(f"sandbox dependency targets overlap: {declaration.target} and {other.target}")
|
||||
raise SandboxError(
|
||||
f"sandbox dependency targets overlap: {declaration.target} and {other.target}"
|
||||
)
|
||||
return tuple(declarations)
|
||||
|
||||
|
||||
@@ -679,7 +646,9 @@ def _validate_dependency_symlinks(
|
||||
)
|
||||
if sandbox_resolved != sandbox_boundary and sandbox_boundary not in sandbox_resolved.parents:
|
||||
raise SandboxError(f"dependency symlink escapes the sandbox workspace: {entry.path}")
|
||||
manifest_resolved = PurePosixPath(posixpath.normpath((entry.path.parent / target).as_posix()))
|
||||
manifest_resolved = PurePosixPath(
|
||||
posixpath.normpath((entry.path.parent / target).as_posix())
|
||||
)
|
||||
if manifest_resolved != manifest_boundary and manifest_boundary not in manifest_resolved.parents:
|
||||
continue
|
||||
link = container / Path(*entry.path.parts)
|
||||
@@ -1047,7 +1016,8 @@ def _dependency_mounts(
|
||||
snapshot: TaskAssetSnapshot,
|
||||
) -> list[ReadOnlyMount]:
|
||||
declarations = tuple(
|
||||
(declaration.source, declaration.target) for declaration in _sandbox_dependency_declarations(task)
|
||||
(declaration.source, declaration.target)
|
||||
for declaration in _sandbox_dependency_declarations(task)
|
||||
)
|
||||
if snapshot.dependency_declarations != declarations:
|
||||
raise SandboxError("task asset snapshot does not match this dependency declaration")
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
{
|
||||
"name": "gitnexus",
|
||||
"description": "Code intelligence powered by a knowledge graph. Provides execution flow tracing, blast radius analysis, and augmented search across your codebase.",
|
||||
"version": "1.6.10-rc.94",
|
||||
"version": "1.6.9",
|
||||
"author": {
|
||||
"name": "GitNexus"
|
||||
},
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
{
|
||||
"name": "gitnexus",
|
||||
"description": "Code intelligence powered by a knowledge graph. Provides execution flow tracing, blast radius analysis, and augmented search across your codebase.",
|
||||
"version": "1.6.10-rc.94",
|
||||
"version": "1.6.9",
|
||||
"skills": "./skills",
|
||||
"mcpServers": "./.mcp.json",
|
||||
"hooks": "./hooks/hooks.json",
|
||||
|
||||
Generated
+53
-111
@@ -11,14 +11,14 @@
|
||||
"@langchain/anthropic": "^1.5.1",
|
||||
"@langchain/core": "^1.2.2",
|
||||
"@langchain/google-genai": "^2.2.0",
|
||||
"@langchain/langgraph": "^1.4.8",
|
||||
"@langchain/langgraph": "^1.4.7",
|
||||
"@langchain/ollama": "^1.3.0",
|
||||
"@langchain/openai": "^1.5.3",
|
||||
"@sigma/edge-curve": "^3.1.0",
|
||||
"@tailwindcss/vite": "^4.3.2",
|
||||
"axios": "^1.18.1",
|
||||
"d3": "^7.9.0",
|
||||
"dompurify": "^3.4.12",
|
||||
"dompurify": "^3.4.11",
|
||||
"gitnexus-shared": "file:../gitnexus-shared",
|
||||
"graphology": "^0.26.0",
|
||||
"graphology-indices": "^0.17.0",
|
||||
@@ -29,14 +29,14 @@
|
||||
"i18next": "^26.3.0",
|
||||
"i18next-browser-languagedetector": "^8.2.1",
|
||||
"langchain": "^1.4.6",
|
||||
"lru-cache": "^11.5.2",
|
||||
"lru-cache": "^11.5.1",
|
||||
"lucide-react": "^1.23.0",
|
||||
"mermaid": "^11.15.0",
|
||||
"mnemonist": "^0.40.4",
|
||||
"pandemonium": "^2.4.0",
|
||||
"react": "^19.2.5",
|
||||
"react-dom": "^19.2.7",
|
||||
"react-i18next": "^17.0.10",
|
||||
"react-i18next": "^17.0.8",
|
||||
"react-markdown": "^10.1.0",
|
||||
"react-syntax-highlighter": "^16.1.1",
|
||||
"react-zoom-pan-pinch": "^4.0.3",
|
||||
@@ -47,7 +47,7 @@
|
||||
"zod": "^4.4.3"
|
||||
},
|
||||
"devDependencies": {
|
||||
"@babel/types": "^8.0.0",
|
||||
"@babel/types": "^7.29.0",
|
||||
"@playwright/test": "^1.61.1",
|
||||
"@testing-library/jest-dom": "^6.9.1",
|
||||
"@testing-library/react": "^16.3.2",
|
||||
@@ -63,7 +63,7 @@
|
||||
"jsdom": "^29.1.1",
|
||||
"tree-sitter-wasms": "^0.1.13",
|
||||
"typescript": "^5.4.5",
|
||||
"vite": "^8.1.5",
|
||||
"vite": "^8.1.4",
|
||||
"vitest": "^4.1.10",
|
||||
"wait-on": "^9.0.10"
|
||||
},
|
||||
@@ -186,13 +186,13 @@
|
||||
}
|
||||
},
|
||||
"node_modules/@babel/helper-string-parser": {
|
||||
"version": "8.0.0",
|
||||
"resolved": "https://registry.npmjs.org/@babel/helper-string-parser/-/helper-string-parser-8.0.0.tgz",
|
||||
"integrity": "sha512-6mJgmFFFIIO82vvoLt9XtRC7/TkzXfts1t/SpRX4IHSzMgqoPYCWesVu1udUPUWioAE/2fcG6WuI8zrkE1gwrg==",
|
||||
"version": "7.29.7",
|
||||
"resolved": "https://registry.npmjs.org/@babel/helper-string-parser/-/helper-string-parser-7.29.7.tgz",
|
||||
"integrity": "sha512-Pb5ijPrZ89GDH8223L4UP8i6QApWxs04RbPQJTeWDV0/keR2E36MeKnyr6LYmUUvqRRI+Iv87SuF1W6ErINzYw==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"engines": {
|
||||
"node": "^22.18.0 || >=24.11.0"
|
||||
"node": ">=6.9.0"
|
||||
}
|
||||
},
|
||||
"node_modules/@babel/helper-validator-identifier": {
|
||||
@@ -221,17 +221,16 @@
|
||||
"node": ">=6.0.0"
|
||||
}
|
||||
},
|
||||
"node_modules/@babel/parser/node_modules/@babel/helper-string-parser": {
|
||||
"version": "7.29.7",
|
||||
"resolved": "https://registry.npmjs.org/@babel/helper-string-parser/-/helper-string-parser-7.29.7.tgz",
|
||||
"integrity": "sha512-Pb5ijPrZ89GDH8223L4UP8i6QApWxs04RbPQJTeWDV0/keR2E36MeKnyr6LYmUUvqRRI+Iv87SuF1W6ErINzYw==",
|
||||
"dev": true,
|
||||
"node_modules/@babel/runtime": {
|
||||
"version": "7.29.2",
|
||||
"resolved": "https://registry.npmjs.org/@babel/runtime/-/runtime-7.29.2.tgz",
|
||||
"integrity": "sha512-JiDShH45zKHWyGe4ZNVRrCjBz8Nh9TMmZG1kh4QTK8hCBTWBi8Da+i7s1fJw7/lYpM4ccepSNfqzZ/QvABBi5g==",
|
||||
"license": "MIT",
|
||||
"engines": {
|
||||
"node": ">=6.9.0"
|
||||
}
|
||||
},
|
||||
"node_modules/@babel/parser/node_modules/@babel/types": {
|
||||
"node_modules/@babel/types": {
|
||||
"version": "7.29.7",
|
||||
"resolved": "https://registry.npmjs.org/@babel/types/-/types-7.29.7.tgz",
|
||||
"integrity": "sha512-4zBIxpPzowiZpusoFkyGVwakdRJUyuH5PxQ/PrqghfdFWWasvnCdPfQXHrenDai+gyLARulZjZowCOj6fjT4pA==",
|
||||
@@ -245,39 +244,6 @@
|
||||
"node": ">=6.9.0"
|
||||
}
|
||||
},
|
||||
"node_modules/@babel/runtime": {
|
||||
"version": "7.29.2",
|
||||
"resolved": "https://registry.npmjs.org/@babel/runtime/-/runtime-7.29.2.tgz",
|
||||
"integrity": "sha512-JiDShH45zKHWyGe4ZNVRrCjBz8Nh9TMmZG1kh4QTK8hCBTWBi8Da+i7s1fJw7/lYpM4ccepSNfqzZ/QvABBi5g==",
|
||||
"license": "MIT",
|
||||
"engines": {
|
||||
"node": ">=6.9.0"
|
||||
}
|
||||
},
|
||||
"node_modules/@babel/types": {
|
||||
"version": "8.0.0",
|
||||
"resolved": "https://registry.npmjs.org/@babel/types/-/types-8.0.0.tgz",
|
||||
"integrity": "sha512-K8ponJDxBwDHigkeFqaqT5wLGl4bTlwMafR8k7b5CPxr6Ww+UG9ls8Yx6Tcpboxu97eeGVEEyKcHmEyOwN1vSw==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"@babel/helper-string-parser": "^8.0.0",
|
||||
"@babel/helper-validator-identifier": "^8.0.0"
|
||||
},
|
||||
"engines": {
|
||||
"node": "^22.18.0 || >=24.11.0"
|
||||
}
|
||||
},
|
||||
"node_modules/@babel/types/node_modules/@babel/helper-validator-identifier": {
|
||||
"version": "8.0.4",
|
||||
"resolved": "https://registry.npmjs.org/@babel/helper-validator-identifier/-/helper-validator-identifier-8.0.4.tgz",
|
||||
"integrity": "sha512-4wFaiLd0bVo4cIoTXI3zKI038NIWE/cr3jvBjejOVYVxV/m8Ltav1USiGzG1fmS5J2RhgEOgXNNK46cRPnRsrg==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"engines": {
|
||||
"node": "^22.18.0 || >=24.11.0"
|
||||
}
|
||||
},
|
||||
"node_modules/@bcoe/v8-coverage": {
|
||||
"version": "1.0.2",
|
||||
"resolved": "https://registry.npmjs.org/@bcoe/v8-coverage/-/v8-coverage-1.0.2.tgz",
|
||||
@@ -1172,13 +1138,13 @@
|
||||
}
|
||||
},
|
||||
"node_modules/@langchain/langgraph": {
|
||||
"version": "1.4.8",
|
||||
"resolved": "https://registry.npmjs.org/@langchain/langgraph/-/langgraph-1.4.8.tgz",
|
||||
"integrity": "sha512-DN1Np1XefdBEbp1qBKlt39cwoL743AAGpR5Ipja0gY2YbWvsoQnOTIrjnj/orSAhaUYsdTKS8VSWdFzsHZo6Ig==",
|
||||
"version": "1.4.7",
|
||||
"resolved": "https://registry.npmjs.org/@langchain/langgraph/-/langgraph-1.4.7.tgz",
|
||||
"integrity": "sha512-2tcyf3QGC7v89kqSxMCtRvzg/3L/4yHtOaWC49A8KieCciWJs7LGaxHoPB6QRxXyUgyR+Zg9Q1ss/XJIE+JuSQ==",
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"@langchain/langgraph-checkpoint": "^1.1.3",
|
||||
"@langchain/langgraph-sdk": "~1.9.26",
|
||||
"@langchain/langgraph-sdk": "~1.9.25",
|
||||
"@langchain/protocol": "^0.0.18",
|
||||
"@standard-schema/spec": "1.1.0"
|
||||
},
|
||||
@@ -1203,9 +1169,9 @@
|
||||
}
|
||||
},
|
||||
"node_modules/@langchain/langgraph-sdk": {
|
||||
"version": "1.9.28",
|
||||
"resolved": "https://registry.npmjs.org/@langchain/langgraph-sdk/-/langgraph-sdk-1.9.28.tgz",
|
||||
"integrity": "sha512-4j3XuM0PvtmAbL8mPfBS99ez3+ytRfgbOpAR/nOeaejTRF3Q9dNw2QnaGLGng8wLPtGLoSj+SYgUOVxy9Bv9vg==",
|
||||
"version": "1.9.25",
|
||||
"resolved": "https://registry.npmjs.org/@langchain/langgraph-sdk/-/langgraph-sdk-1.9.25.tgz",
|
||||
"integrity": "sha512-mRKW8zyQUaHox+HirRFMRrPqOvNbQI3xeXDt6kkk4PbBg77V92bsO1WzUVNrmJ81zCkvxyOrWSK8D6ioCj0a8A==",
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"@langchain/protocol": "^0.0.18",
|
||||
@@ -1242,9 +1208,9 @@
|
||||
"license": "MIT"
|
||||
},
|
||||
"node_modules/@langchain/langgraph-sdk/node_modules/p-queue": {
|
||||
"version": "9.3.3",
|
||||
"resolved": "https://registry.npmjs.org/p-queue/-/p-queue-9.3.3.tgz",
|
||||
"integrity": "sha512-NXAOdnEe5FsZJfT4oK84lE1Y5cFFdWlRuOo5tww8DyNMxyRXwn39fIkUtNLKppcPC+UYU/bXujNCUGDv01y7CA==",
|
||||
"version": "9.3.0",
|
||||
"resolved": "https://registry.npmjs.org/p-queue/-/p-queue-9.3.0.tgz",
|
||||
"integrity": "sha512-7NED7xhQ74Ngp4JP/2e0VZHp7vSWfJfqeiR92jPgxsz6m0Se4P03YoTKa9dDXyZ3r6P616gUXttrB6nnHYKang==",
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"eventemitter3": "^5.0.4",
|
||||
@@ -4109,9 +4075,9 @@
|
||||
"peer": true
|
||||
},
|
||||
"node_modules/dompurify": {
|
||||
"version": "3.4.12",
|
||||
"resolved": "https://registry.npmjs.org/dompurify/-/dompurify-3.4.12.tgz",
|
||||
"integrity": "sha512-zQvGet8Z2sWbQhCmfFz/T5QWH2oBmjnqK3qvOjaqaNLrLEF912WamU+ohnTp0TCep/MFVHpdJuCZEdFOdTnEFg==",
|
||||
"version": "3.4.11",
|
||||
"resolved": "https://registry.npmjs.org/dompurify/-/dompurify-3.4.11.tgz",
|
||||
"integrity": "sha512-zhlUV12GsaRzMsf9q5M254YhA4+VuF0fG+QFqu6aYpoGlKtz+w8//jBcGVYBgQkR5GHjUomejY84AV+/uPbWdw==",
|
||||
"license": "(MPL-2.0 OR Apache-2.0)",
|
||||
"optionalDependencies": {
|
||||
"@types/trusted-types": "^2.0.7"
|
||||
@@ -4391,9 +4357,9 @@
|
||||
"license": "Unlicense"
|
||||
},
|
||||
"node_modules/fast-uri": {
|
||||
"version": "3.1.4",
|
||||
"resolved": "https://registry.npmjs.org/fast-uri/-/fast-uri-3.1.4.tgz",
|
||||
"integrity": "sha512-8JnbkQ4juDyvYs4mgFGQqg4yCYtFDtUtmp2QIQq11ZZe5CFQ5wcqm1rqDgAh/QdMySuBnPzMUiJUNZG5N/AiQw==",
|
||||
"version": "3.1.2",
|
||||
"resolved": "https://registry.npmjs.org/fast-uri/-/fast-uri-3.1.2.tgz",
|
||||
"integrity": "sha512-rVjf7ArG3LTk+FS6Yw81V1DLuZl1bRbNrev6Tmd/9RaroeeRRJhAt7jg/6YFxbvAQXUCavSoZhPPj6oOx+5KjQ==",
|
||||
"dev": true,
|
||||
"funding": [
|
||||
{
|
||||
@@ -5706,9 +5672,9 @@
|
||||
}
|
||||
},
|
||||
"node_modules/lru-cache": {
|
||||
"version": "11.5.2",
|
||||
"resolved": "https://registry.npmjs.org/lru-cache/-/lru-cache-11.5.2.tgz",
|
||||
"integrity": "sha512-4pfM1Ff0x50o0tQwb5ucw/RzNyD0/YJME6IVcStalZuMWxdt3sR3huStTtxz4PUmvZfRguvDejasvQ2kifR11g==",
|
||||
"version": "11.5.1",
|
||||
"resolved": "https://registry.npmjs.org/lru-cache/-/lru-cache-11.5.1.tgz",
|
||||
"integrity": "sha512-RPimw/7aMdv2oqRrxKwvZXcPfwBrn/JZ2xYcY9Hus/6LaS3VOAKVWKWgNLCFSiOm1ESXinjsDlidVU7JlnCN2A==",
|
||||
"license": "BlueOak-1.0.0",
|
||||
"engines": {
|
||||
"node": "20 || >=22"
|
||||
@@ -5755,30 +5721,6 @@
|
||||
"source-map-js": "^1.2.1"
|
||||
}
|
||||
},
|
||||
"node_modules/magicast/node_modules/@babel/helper-string-parser": {
|
||||
"version": "7.29.7",
|
||||
"resolved": "https://registry.npmjs.org/@babel/helper-string-parser/-/helper-string-parser-7.29.7.tgz",
|
||||
"integrity": "sha512-Pb5ijPrZ89GDH8223L4UP8i6QApWxs04RbPQJTeWDV0/keR2E36MeKnyr6LYmUUvqRRI+Iv87SuF1W6ErINzYw==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"engines": {
|
||||
"node": ">=6.9.0"
|
||||
}
|
||||
},
|
||||
"node_modules/magicast/node_modules/@babel/types": {
|
||||
"version": "7.29.7",
|
||||
"resolved": "https://registry.npmjs.org/@babel/types/-/types-7.29.7.tgz",
|
||||
"integrity": "sha512-4zBIxpPzowiZpusoFkyGVwakdRJUyuH5PxQ/PrqghfdFWWasvnCdPfQXHrenDai+gyLARulZjZowCOj6fjT4pA==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"@babel/helper-string-parser": "^7.29.7",
|
||||
"@babel/helper-validator-identifier": "^7.29.7"
|
||||
},
|
||||
"engines": {
|
||||
"node": ">=6.9.0"
|
||||
}
|
||||
},
|
||||
"node_modules/make-dir": {
|
||||
"version": "4.0.0",
|
||||
"resolved": "https://registry.npmjs.org/make-dir/-/make-dir-4.0.0.tgz",
|
||||
@@ -6884,9 +6826,9 @@
|
||||
}
|
||||
},
|
||||
"node_modules/nanoid": {
|
||||
"version": "3.3.16",
|
||||
"resolved": "https://registry.npmjs.org/nanoid/-/nanoid-3.3.16.tgz",
|
||||
"integrity": "sha512-bzlKTyNJ7+LdGIIwy8ijFpIqEQIvafahV7eYykJ8Cvh42EdJeODoJ6gUJXpQJvej1BddH8OqTXZNE/KfbWAu8Q==",
|
||||
"version": "3.3.15",
|
||||
"resolved": "https://registry.npmjs.org/nanoid/-/nanoid-3.3.15.tgz",
|
||||
"integrity": "sha512-y7Wygv/7mEOvxTuEQDB8StXdMRBWf1kR/tlhAzBRUFkB2jfcLOAxO/SHmOO2zgz1pVgK29/kyupn059/bCHdjA==",
|
||||
"funding": [
|
||||
{
|
||||
"type": "github",
|
||||
@@ -7274,9 +7216,9 @@
|
||||
}
|
||||
},
|
||||
"node_modules/postcss": {
|
||||
"version": "8.5.22",
|
||||
"resolved": "https://registry.npmjs.org/postcss/-/postcss-8.5.22.tgz",
|
||||
"integrity": "sha512-KBDEIpLrvpv16pp3K0Fw+UCoZfopFjjgeB+0tA/aaThfEE74kKDLrgg603YvOWJyg3+WYtyq3xYsQWsIyZlPqQ==",
|
||||
"version": "8.5.16",
|
||||
"resolved": "https://registry.npmjs.org/postcss/-/postcss-8.5.16.tgz",
|
||||
"integrity": "sha512-vuwillviilfKZsg0VGj5R/YwwcHx4SLsIOI/7K6mQkWx+l5cUHTjj5g0AasTBcyXsbfTgrwsUNmVUb5xVwyPwg==",
|
||||
"funding": [
|
||||
{
|
||||
"type": "opencollective",
|
||||
@@ -7293,7 +7235,7 @@
|
||||
],
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"nanoid": "^3.3.16",
|
||||
"nanoid": "^3.3.12",
|
||||
"picocolors": "^1.1.1",
|
||||
"source-map-js": "^1.2.1"
|
||||
},
|
||||
@@ -7414,9 +7356,9 @@
|
||||
}
|
||||
},
|
||||
"node_modules/react-i18next": {
|
||||
"version": "17.0.10",
|
||||
"resolved": "https://registry.npmjs.org/react-i18next/-/react-i18next-17.0.10.tgz",
|
||||
"integrity": "sha512-XneHftyYA774MJkkccSkZ5oKrUpCnXIPmxio3wemqrVzCRLWiGXOMbIzObrer03fNDEnm8g8R5yYls4HcE+esg==",
|
||||
"version": "17.0.8",
|
||||
"resolved": "https://registry.npmjs.org/react-i18next/-/react-i18next-17.0.8.tgz",
|
||||
"integrity": "sha512-0ooKbGLU8JXhe1zwpQUWIeXSgLPOfwJmgheWRIUpcoA0CpyabpGhayjdG+/eA5esC1AQ8h2jWpXjJfzQzeDOCw==",
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"@babel/runtime": "^7.29.2",
|
||||
@@ -7426,7 +7368,7 @@
|
||||
"peerDependencies": {
|
||||
"i18next": ">= 26.2.0",
|
||||
"react": ">= 16.8.0",
|
||||
"typescript": "^5 || ^6 || ^7"
|
||||
"typescript": "^5 || ^6"
|
||||
},
|
||||
"peerDependenciesMeta": {
|
||||
"react-dom": {
|
||||
@@ -7952,9 +7894,9 @@
|
||||
}
|
||||
},
|
||||
"node_modules/tar": {
|
||||
"version": "7.5.20",
|
||||
"resolved": "https://registry.npmjs.org/tar/-/tar-7.5.20.tgz",
|
||||
"integrity": "sha512-9FcyK4PA6+WbzlTM9WhQm6vB5W7cP7dUiPsv1g7YDwEQnQ1CGpK3MGlKk/ITVWMk05kHZuBhmVhiv8LZoy/PFQ==",
|
||||
"version": "7.5.16",
|
||||
"resolved": "https://registry.npmjs.org/tar/-/tar-7.5.16.tgz",
|
||||
"integrity": "sha512-56adEpPMouktRlBLXiaYFFzZ/3+JXa8P9n7WbR+ibIjtviN55mEaOkiysCnPnWm+7kkui1Dn8J9l+g6zV8731w==",
|
||||
"dev": true,
|
||||
"license": "BlueOak-1.0.0",
|
||||
"dependencies": {
|
||||
@@ -8354,15 +8296,15 @@
|
||||
}
|
||||
},
|
||||
"node_modules/vite": {
|
||||
"version": "8.1.5",
|
||||
"resolved": "https://registry.npmjs.org/vite/-/vite-8.1.5.tgz",
|
||||
"integrity": "sha512-7ULLwsCdYx/nRyrpiEwvqb5TFHrMVZyBt+rg/OAXT7rgj/z+DtTDyKFeLAdDkubDVDKD8jOsndmy7m55XcfUsw==",
|
||||
"version": "8.1.4",
|
||||
"resolved": "https://registry.npmjs.org/vite/-/vite-8.1.4.tgz",
|
||||
"integrity": "sha512-bTT9PsdWO+MQMNG9ZXIP/qM9wGh37DFxTV/sPq9cFpHr3w4jkgef032PkAL9jAqhk3Nz8NQw3O8n6/xFkqO4QQ==",
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"lightningcss": "^1.32.0",
|
||||
"picomatch": "^4.0.5",
|
||||
"postcss": "^8.5.17",
|
||||
"rolldown": "~1.1.5",
|
||||
"postcss": "^8.5.16",
|
||||
"rolldown": "~1.1.4",
|
||||
"tinyglobby": "^0.2.17"
|
||||
},
|
||||
"bin": {
|
||||
|
||||
@@ -21,14 +21,14 @@
|
||||
"@langchain/anthropic": "^1.5.1",
|
||||
"@langchain/core": "^1.2.2",
|
||||
"@langchain/google-genai": "^2.2.0",
|
||||
"@langchain/langgraph": "^1.4.8",
|
||||
"@langchain/langgraph": "^1.4.7",
|
||||
"@langchain/ollama": "^1.3.0",
|
||||
"@langchain/openai": "^1.5.3",
|
||||
"@sigma/edge-curve": "^3.1.0",
|
||||
"@tailwindcss/vite": "^4.3.2",
|
||||
"axios": "^1.18.1",
|
||||
"d3": "^7.9.0",
|
||||
"dompurify": "^3.4.12",
|
||||
"dompurify": "^3.4.11",
|
||||
"gitnexus-shared": "file:../gitnexus-shared",
|
||||
"graphology": "^0.26.0",
|
||||
"graphology-indices": "^0.17.0",
|
||||
@@ -39,14 +39,14 @@
|
||||
"i18next": "^26.3.0",
|
||||
"i18next-browser-languagedetector": "^8.2.1",
|
||||
"langchain": "^1.4.6",
|
||||
"lru-cache": "^11.5.2",
|
||||
"lru-cache": "^11.5.1",
|
||||
"lucide-react": "^1.23.0",
|
||||
"mermaid": "^11.15.0",
|
||||
"mnemonist": "^0.40.4",
|
||||
"pandemonium": "^2.4.0",
|
||||
"react": "^19.2.5",
|
||||
"react-dom": "^19.2.7",
|
||||
"react-i18next": "^17.0.10",
|
||||
"react-i18next": "^17.0.8",
|
||||
"react-markdown": "^10.1.0",
|
||||
"react-syntax-highlighter": "^16.1.1",
|
||||
"react-zoom-pan-pinch": "^4.0.3",
|
||||
@@ -57,7 +57,7 @@
|
||||
"zod": "^4.4.3"
|
||||
},
|
||||
"devDependencies": {
|
||||
"@babel/types": "^8.0.0",
|
||||
"@babel/types": "^7.29.0",
|
||||
"@playwright/test": "^1.61.1",
|
||||
"@testing-library/jest-dom": "^6.9.1",
|
||||
"@testing-library/react": "^16.3.2",
|
||||
@@ -73,7 +73,7 @@
|
||||
"jsdom": "^29.1.1",
|
||||
"tree-sitter-wasms": "^0.1.13",
|
||||
"typescript": "^5.4.5",
|
||||
"vite": "^8.1.5",
|
||||
"vite": "^8.1.4",
|
||||
"vitest": "^4.1.10",
|
||||
"wait-on": "^9.0.10"
|
||||
},
|
||||
|
||||
+3
-24
@@ -284,7 +284,6 @@ Set these env vars to use a remote OpenAI-compatible `/v1/embeddings` endpoint i
|
||||
export GITNEXUS_EMBEDDING_URL=http://your-server:8080/v1
|
||||
export GITNEXUS_EMBEDDING_MODEL=BAAI/bge-large-en-v1.5
|
||||
export GITNEXUS_EMBEDDING_DIMS=1024 # optional, default 384
|
||||
export GITNEXUS_EMBEDDING_REQUEST_DIMS=omit # optional: omit "dimensions", or an integer to override it
|
||||
export GITNEXUS_EMBEDDING_API_KEY=your-key # optional, default: "unused"
|
||||
export GITNEXUS_EMBEDDING_MAX_ATTEMPTS=3 # optional, total attempts (1-20)
|
||||
export GITNEXUS_EMBEDDING_RETRY_CAP_MS=5000 # optional, maximum retry delay
|
||||
@@ -292,15 +291,6 @@ export GITNEXUS_EMBEDDING_MIN_INTERVAL_MS=0 # optional, minimum request spacing
|
||||
gitnexus analyze . --embeddings
|
||||
```
|
||||
|
||||
`GITNEXUS_EMBEDDING_REQUEST_DIMS` controls only the `dimensions` field sent in
|
||||
the request body, independently of `GITNEXUS_EMBEDDING_DIMS` (which still
|
||||
validates the returned vector's length):
|
||||
|
||||
- `omit` (or `none`, `off`, `false`, `0`) — do not send `dimensions` at all, for
|
||||
strict backends that return the right vector size but reject the field.
|
||||
- a positive integer — send that value instead of `GITNEXUS_EMBEDDING_DIMS`.
|
||||
- unset — send `GITNEXUS_EMBEDDING_DIMS` (the previous behavior).
|
||||
|
||||
Works with Infinity, vLLM, TEI, llama.cpp, Ollama, LM Studio, or OpenAI. Retry and pacing settings are provider-neutral; provider-specific limits should be supplied through configuration. When unset, local embeddings are used unchanged.
|
||||
|
||||
## Multi-Repo Support
|
||||
@@ -484,7 +474,7 @@ Configure the behavior with these environment variables:
|
||||
| `GITNEXUS_FTS_CJK_SEGMENTATION` | `none`, `bigram` | `none` | `bigram` inserts overlapping character-bigram boundaries into Chinese/Japanese Han-ideograph spans in `content`/`description` before FTS indexing, so LadybugDB's space-only tokenizer can see sub-phrase word boundaries. Scoped to CJK Unified Ideographs only — Japanese Hiragana/Katakana and Korean Hangul are not currently segmented. Unlike `GITNEXUS_FTS_STEMMER`, this rewrites stored text — enabling it on an already-indexed repo requires a full `gitnexus analyze --force`; neither `--repair-fts` nor a plain incremental `analyze` applies it to previously-indexed files. Set the same value wherever `analyze` and search-serving processes (CLI query, MCP server, web server) run. |
|
||||
| `GITNEXUS_COMMUNITY_ENGINE` | `graphology`, `icebug`, `auto` | `graphology` | Community-detection engine used during analyze. `graphology` uses the bundled default path. `icebug` and `auto` currently behave identically: both try the experimental Icebug CSR path and fall back to Graphology if the optional native module is unavailable or incompatible. |
|
||||
| `GITNEXUS_WAL_CHECKPOINT_THRESHOLD` | integer `>= -1` | `67108864` (64 MiB) | LadybugDB WAL auto-checkpoint threshold during analyze (bytes). Auto-checkpoint remains enabled; `-1` keeps Ladybug's stock ~16 MiB. Larger thresholds reduce checkpoint frequency but increase the WAL size at rotation time — choose a smaller value on disk-constrained environments. |
|
||||
| `GITNEXUS_LBUG_BUFFER_POOL_SIZE` | integer `>= 0` (bytes) | min(2 GiB, 80% RAM) | LadybugDB buffer-pool ceiling for every GitNexus database (analyze, MCP server, serve, group bridges). Bounded so a long-lived `gitnexus mcp` process or a large incremental `analyze` cannot grow toward LadybugDB's native 80%-of-RAM default and OOM the host (#2557). `0` restores that native unbounded default; invalid values warn and fall back to the default. During `analyze` the pool is right-sized to the graph and, on non-4 KiB-page hosts (Apple Silicon 16 KiB, Ascend/aarch64 64 KiB), scaled by the page-size granule ratio up to min(2 GiB × pageSize/4 KiB, 80% RAM) (#2631); this env var overrides all of that as an absolute value. |
|
||||
| `GITNEXUS_LBUG_BUFFER_POOL_SIZE` | integer `>= 0` (bytes) | min(2 GiB, 80% RAM) | LadybugDB buffer-pool ceiling for every GitNexus database (analyze, MCP server, serve, group bridges). Bounded so a long-lived `gitnexus mcp` process or a large incremental `analyze` cannot grow toward LadybugDB's native 80%-of-RAM default and OOM the host (#2557). `0` restores that native unbounded default; invalid values warn and fall back to the default. |
|
||||
| `GITNEXUS_LBUG_MAX_DB_SIZE` | positive integer (bytes) | `17179869184` (16 GiB) | Upper bound for a single LadybugDB database file. This is an mmap/disk-address-space ceiling, not a memory limit — it does not constrain the buffer pool (use `GITNEXUS_LBUG_BUFFER_POOL_SIZE` for that). Raise it when indexing genuinely huge monorepos; invalid values silently fall back to the default. |
|
||||
|
||||
```bash
|
||||
@@ -511,19 +501,9 @@ For very large repositories:
|
||||
# Increase Node.js heap size
|
||||
NODE_OPTIONS="--max-old-space-size=16384" npx gitnexus analyze
|
||||
|
||||
# Exclude large directories (this repo only)
|
||||
# Exclude large directories
|
||||
echo "vendor/" >> .gitnexusignore
|
||||
echo "dist/" >> .gitnexusignore
|
||||
|
||||
# Exclude a directory across every repo you index, without touching each
|
||||
# repo's own .gitnexusignore or needing push/commit access to it. GitNexus
|
||||
# reads the same sources `git` itself does: core.excludesFile (all repos)
|
||||
# and $GIT_DIR/info/exclude (this repo only, untracked). A repo's own
|
||||
# .gitignore/.gitnexusignore can still override either with a `!pattern`
|
||||
# negation. Skip both entirely with GITNEXUS_NO_GLOBAL_IGNORE=1.
|
||||
git config --global core.excludesFile ~/.gitignore_global # applies to every repo
|
||||
echo "docs/" >> ~/.gitignore_global
|
||||
echo "build/" >> .git/info/exclude # this repo only, untracked
|
||||
```
|
||||
|
||||
### Large files are being skipped
|
||||
@@ -558,7 +538,7 @@ For repositories with very large source files, `GITNEXUS_WORKER_SUB_BATCH_MAX_BY
|
||||
|
||||
### Worker pool resilience tuning
|
||||
|
||||
Four env vars expose the pool's resilience layers (respawn budget, cumulative-timeout cap, circuit breaker, startup handshake). Defaults are tuned for typical repos; bump them when an analyze legitimately needs more retries, or lower them to fail-fast on a known-bad shape.
|
||||
Three env vars expose the pool's resilience layers (respawn budget, cumulative-timeout cap, circuit breaker). Defaults are tuned for typical repos; bump them when an analyze legitimately needs more retries, or lower them to fail-fast on a known-bad shape.
|
||||
|
||||
| Variable | Default | Effect |
|
||||
| ----------------------------------------------- | ----------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ |
|
||||
@@ -566,7 +546,6 @@ Four env vars expose the pool's resilience layers (respawn budget, cumulative-ti
|
||||
| `GITNEXUS_WORKER_MAX_CUMULATIVE_TIMEOUT_MS` | `5 × subBatchTimeoutMs` | Total retry wall-time budget per job before quarantining. Bounds exponentially-growing retry waits. |
|
||||
| `GITNEXUS_WORKER_CONSECUTIVE_FAILURE_THRESHOLD` | `max(3, poolSize)` | Per-slot consecutive deaths before the pool's circuit breaker trips. After tripping, dispatches require a fresh pool. |
|
||||
| `GITNEXUS_WORKER_SHUTDOWN_DRAIN_MS` | `30000` | Max wait at pool shutdown for a retired worker still inside native code — terminated at its next JS-safe point instead of mid-native-call, which would abort the process (`Napi::Error`, #2432). |
|
||||
| `GITNEXUS_WORKER_READY_TIMEOUT_MS` | `5000` | Startup budget for a parse worker to load its grammar bindings and report `{type:'ready'}`. Slots that miss it are treated as startup crashes. Raise it on a slow or heavily loaded host where a full pool cold-starting concurrently needs more than 5s. |
|
||||
| `GITNEXUS_CPP_CAPTURE_BUDGET_MS` | `20000` | Per-file wall-clock budget for C++ capture extraction; on breach the file keeps partial captures with a warning (#2432). `0` expires immediately. |
|
||||
|
||||
### Graph cleanup tuning
|
||||
|
||||
@@ -1 +1 @@
|
||||
36e29abc0780bc857b6df6dd180a0b6036c8a28f927ccc2d4fe50eede24d0c99
|
||||
a99e69ab2dfb897ed771c6a8e29c5b32843a7f734db701e0699afc07c090e4d5
|
||||
|
||||
@@ -46,9 +46,8 @@
|
||||
"_note": "#2046: F35 qualified-constructor captures now emit @reference.qualified-name + a simple-name @reference.name on `new Ns.Foo()`/`new A.B.Foo()`; namespace_declaration/file_scoped_namespace_declaration now emit @declaration.namespace name captures (feeding the non-destructive namespacePrefix sidecar for `new B.Foo()` same-tail disambiguation). + csharp-interface-only-base and csharp-namespace-qualified-ctor fixtures. Pure capture-additive + fixture-corpus drift; scaling stays linear (~1.11)."
|
||||
},
|
||||
"rust": {
|
||||
"fingerprint": "f7742f65f14d7d6590df7f16303fc3cc9dc0c233cd80bf90c98b084933cd3846",
|
||||
"fingerprint": "df369c5a5f8de7753fc8bab8b4108ef5081750974ea5085ba9a867675ac9eb29",
|
||||
"scaling_budget": 1.5,
|
||||
"_rebaselined_dyn_trait_object_2604": "#2604: RUST_SCOPE_QUERY now captures function_signature_item (abstract trait methods, no body) as a scope + declaration, so a &dyn Trait receiver can dispatch a CALLS edge to the trait's own method. Additive capture shift across every bench fixture with a required trait method. Prior df369c5a5f8de7753fc8bab8b4108ef5081750974ea5085ba9a867675ac9eb29 -> f7742f65f14d7d6590df7f16303fc3cc9dc0c233cd80bf90c98b084933cd3846; scaling 1.033 < 1.5.",
|
||||
"_rebaselined_callable_flow_2522_review": "PR #2522 review hardening: callable operands retain expression/qualified identity and formals retain signature metadata. Prior 65e5bca66bb1ca117949409e8fb5c80ee69d6f1b5318908eaaecf08da0482e5c -> df369c5a5f8de7753fc8bab8b4108ef5081750974ea5085ba9a867675ac9eb29; scaling 1.065 < 1.5.",
|
||||
"_rebaselined_callable_flow_2522_followup": "PR #2522 follow-up: Rust fn-value callable flow facts with invocation/constructor-result suppression. Prior ac610bbe97666bf285923479dd7b43a2fe4c5354aae8df1bcbafdc04fb220f82 -> 65e5bca66bb1ca117949409e8fb5c80ee69d6f1b5318908eaaecf08da0482e5c; scaling 1.024 < 1.5.",
|
||||
"_rebaselined": "#1956 tri-review U1: rust-qualified-trait fixture (scoped + generic-of-scoped impl trait paths); bareTypeIdentifier now resolves scoped_type_identifier bases by their name: tail (additive, no existing-fixture drift); linear (~1.04). #1975: + rust-scoped-impl fixture (impl a::Inner / b::Inner inherent scoped impls) \u2014 legacy @definition.impl scoped arm + findEnclosingClassInfo inherent-impl scoped target; rust scope-extractor captures byte-identical. | #942: scope-resolution-only cleanup reworded fixture comments; capture byte-positions shift, capture LOGIC unchanged.",
|
||||
@@ -90,7 +89,7 @@
|
||||
"_rebaselined": "#1919 review CF3 fix: extended kotlin-local-property-owner (init/accessor destructuring) + new dart-accessor-owner fixture (getter/setter ownership). Fingerprint-only corpus drift; scaling ~1.0."
|
||||
},
|
||||
"java": {
|
||||
"fingerprint": "d04298a91beec76d0fa7099b3d71265723be60c1df688969aa954f135dd49686",
|
||||
"fingerprint": "975b68aaac6d06094260fb0c67f9b1bc03692ba7220669d192aca9dccd5fc0ca",
|
||||
"scaling_budget": 1.5,
|
||||
"_rebaselined_callable_flow_2522_review": "PR #2522 review hardening: callable operands retain expression/qualified identity and formals retain signature metadata; same-name lexical regions use an O(ancestor-depth) ID-set lookup. Prior d5c59d7dc9e206637515d5aea1163f7c1cdd76410c38c5fe6143d13d19677d6a -> 004a3592998dca1193bd1429a8284513725de7764f2a3eceedaaa984cfd763b4; scaling 0.992 < 1.5.",
|
||||
"_rebaselined_callable_flow_2522_followup": "PR #2522 follow-up: Java method-reference/SAM callable flow facts with invocation-result suppression. Prior 062d754764aaa8a6772fb90875c710502a63e3e7a300e633942381ed914faada -> d5c59d7dc9e206637515d5aea1163f7c1cdd76410c38c5fe6143d13d19677d6a; scaling 1.074 < 1.5.",
|
||||
@@ -98,9 +97,7 @@
|
||||
"_note": "#1928 / #2045: F35 adds qualified + qualified-generic constructor query captures (`new pkg.Foo()`, `new a.b.Foo()`, `new pkg.Box<T>()`); F38 synthesizes `@reference.call.constructor` on `super(...)`/`this(...)` explicit_constructor_invocation nodes; F41 generic-aware stripQualifier in interpret (type-binding normalization). + java-qualified-constructor and java-explicit-constructor fixtures. Pure capture-additive + fixture-corpus drift; scaling stays linear (~1.06).",
|
||||
"_rebaselined_2522_review_fixes": "PR #2522 review fixes: get/test dropped from callableProtocolMethods. Prior 004a3592998dca1193bd1429a8284513725de7764f2a3eceedaaa984cfd763b4 -> f3b4f4b6610e07c3ac90deb1c53d3572b6ad55a36e5d7134984876d30031ff67; scaling ratio re-verified within budget.",
|
||||
"_rebaselined_2550_instance_model": "PR #2549 (#2550): anonymous class bodies emit synthesized @declaration.class/@declaration.name (Worker$N), an @reference.inherits to the constructed type, and receiver @type-binding.* captures; six new java-* fixtures joined the corpus. Prior f3b4f4b6610e07c3ac90deb1c53d3572b6ad55a36e5d7134984876d30031ff67 -> d79c3b92acfc866094981499b977388ca14f90839bca0c040342ab1cec00aa90; scaling 1.058 < 1.5.",
|
||||
"_rebaselined_2555_enum_constant_bodies": "PR for #2555: enum constant bodies emit synthesized E$N classes + @reference.inherits to the host enum; anonymous naming follows JLS 13.1 immediately-enclosing-type chains INCLUDING anonymous enclosing types (NestHost$1$1, N$1$1); six new java-* fixtures joined the corpus. Prior d79c3b92acfc866094981499b977388ca14f90839bca0c040342ab1cec00aa90 -> 975b68aaac6d06094260fb0c67f9b1bc03692ba7220669d192aca9dccd5fc0ca; scaling 1.05 < 1.5.",
|
||||
"_rebaselined_2564_record_capture": "PR for #2564: JAVA_QUERIES gained a (record_declaration name: (identifier) @name) @definition.record capture, previously entirely missing (record_declaration had no structure-phase capture at all, unlike class/interface/enum) - a record's methods existed as ownerless Method nodes with no HAS_METHOD edge. Two new java-* fixtures (java-record-methods, java-new-expr-chain-call) joined the corpus. Prior 975b68aaac6d06094260fb0c67f9b1bc03692ba7220669d192aca9dccd5fc0ca -> 85fc7af9c3c1bceac76cb4f27214410b04967682a2eaa7e468e26efd1f4e2537; scaling 1.059 < 1.5.",
|
||||
"_rebaselined_2561_enum_constant_receiver": "PR for #2561: synthesizeJavaAnonymousClassDeclarations now emits a class-scope @type-binding.annotation/name/type per enum constant (constant simple name -> its E$N synthesized class when bodied, else the host enum) so E.CONST.method() resolves through the existing compound-receiver chain walk. Two drivers of the drift, both in the java-enum-constant-body fixture (this bench's corpus IS test/fixtures/lang-resolution): (1) one extra type-binding match per enum_constant from the capture change; (2) review follow-up added a body-less Plain.java enum + EnumConst.dispatchToConstant/dispatchInherited methods (bodied-override, inherited-via-MRO, and body-less dispatch call sites). The review's fail-safe hardening (bodied constant binds ONLY to E$N, never the host enum, when name synthesis fails on a malformed tree) is output-neutral on this well-formed corpus (verified: fingerprint identical with and without it). Prior 85fc7af9c3c1bceac76cb4f27214410b04967682a2eaa7e468e26efd1f4e2537 -> d04298a91beec76d0fa7099b3d71265723be60c1df688969aa954f135dd49686; scaling < 1.5."
|
||||
"_rebaselined_2555_enum_constant_bodies": "PR for #2555: enum constant bodies emit synthesized E$N classes + @reference.inherits to the host enum; anonymous naming follows JLS 13.1 immediately-enclosing-type chains INCLUDING anonymous enclosing types (NestHost$1$1, N$1$1); six new java-* fixtures joined the corpus. Prior d79c3b92acfc866094981499b977388ca14f90839bca0c040342ab1cec00aa90 -> 975b68aaac6d06094260fb0c67f9b1bc03692ba7220669d192aca9dccd5fc0ca; scaling 1.05 < 1.5."
|
||||
},
|
||||
"typescript": {
|
||||
"fingerprint": "3280b13d3f9378ab23eee31c2edc779b5a9ae1e7bb510c23a24855b44406d2f4",
|
||||
|
||||
Generated
+81
-74
@@ -1,16 +1,16 @@
|
||||
{
|
||||
"name": "gitnexus",
|
||||
"version": "1.6.10-rc.94",
|
||||
"version": "1.6.9",
|
||||
"lockfileVersion": 3,
|
||||
"requires": true,
|
||||
"packages": {
|
||||
"": {
|
||||
"name": "gitnexus",
|
||||
"version": "1.6.10-rc.94",
|
||||
"version": "1.6.9",
|
||||
"hasInstallScript": true,
|
||||
"license": "PolyForm-Noncommercial-1.0.0",
|
||||
"dependencies": {
|
||||
"@ladybugdb/core": "^0.18.3",
|
||||
"@ladybugdb/core": "^0.18.0",
|
||||
"@modelcontextprotocol/sdk": "^1.0.0",
|
||||
"@scarf/scarf": "^1.4.0",
|
||||
"busboy": "^1.6.0",
|
||||
@@ -24,7 +24,7 @@
|
||||
"graphology-indices": "^0.17.0",
|
||||
"graphology-utils": "^2.3.0",
|
||||
"ignore": "^7.0.5",
|
||||
"js-yaml": "^5.0.0",
|
||||
"js-yaml": "^4.1.1",
|
||||
"jsonc-parser": "^3.3.1",
|
||||
"mnemonist": "^0.40.3",
|
||||
"node-addon-api": "^8.0.0",
|
||||
@@ -58,7 +58,9 @@
|
||||
"@types/cli-progress": "^3.11.6",
|
||||
"@types/cors": "^2.8.17",
|
||||
"@types/express": "^5.0.6",
|
||||
"@types/node": "^26.0.0",
|
||||
"@types/js-yaml": "^4.0.9",
|
||||
"@types/node": "^25.6.0",
|
||||
"@types/uuid": "^11.0.0",
|
||||
"@vitest/coverage-v8": "^4.0.18",
|
||||
"gitnexus-shared": "file:../gitnexus-shared",
|
||||
"tsx": "^4.0.0",
|
||||
@@ -66,7 +68,7 @@
|
||||
"vitest": "^4.0.18"
|
||||
},
|
||||
"engines": {
|
||||
"node": "^22.18.0 || >=24.11.0"
|
||||
"node": ">=22.0.0"
|
||||
},
|
||||
"optionalDependencies": {
|
||||
"@huggingface/transformers": "^4.1.0",
|
||||
@@ -1252,9 +1254,9 @@
|
||||
}
|
||||
},
|
||||
"node_modules/@ladybugdb/core": {
|
||||
"version": "0.18.3",
|
||||
"resolved": "https://registry.npmjs.org/@ladybugdb/core/-/core-0.18.3.tgz",
|
||||
"integrity": "sha512-XjpPKW4MrL28D2gYGTZuIjiEcPx12L21lx58QggrdrItw8o/e9Lmg/Ejoo4Kz08lZj+rIcC1Fu9thzIYOTUlJw==",
|
||||
"version": "0.18.1",
|
||||
"resolved": "https://registry.npmjs.org/@ladybugdb/core/-/core-0.18.1.tgz",
|
||||
"integrity": "sha512-0c1kXDpdv7z/GB0oyFYnLEjLsXFwPHz1YD4wxtrk9hav8zJX5T1PHQMr+XRfdDI1NQjx4iNdbPQGGT7Bx/X2aw==",
|
||||
"hasInstallScript": true,
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
@@ -1263,17 +1265,17 @@
|
||||
"node-addon-api": "^6.0.0"
|
||||
},
|
||||
"optionalDependencies": {
|
||||
"@ladybugdb/core-darwin-arm64": "0.18.3",
|
||||
"@ladybugdb/core-darwin-x64": "0.18.3",
|
||||
"@ladybugdb/core-linux-arm64": "0.18.3",
|
||||
"@ladybugdb/core-linux-x64": "0.18.3",
|
||||
"@ladybugdb/core-win32-x64": "0.18.3"
|
||||
"@ladybugdb/core-darwin-arm64": "0.18.1",
|
||||
"@ladybugdb/core-darwin-x64": "0.18.1",
|
||||
"@ladybugdb/core-linux-arm64": "0.18.1",
|
||||
"@ladybugdb/core-linux-x64": "0.18.1",
|
||||
"@ladybugdb/core-win32-x64": "0.18.1"
|
||||
}
|
||||
},
|
||||
"node_modules/@ladybugdb/core-darwin-arm64": {
|
||||
"version": "0.18.3",
|
||||
"resolved": "https://registry.npmjs.org/@ladybugdb/core-darwin-arm64/-/core-darwin-arm64-0.18.3.tgz",
|
||||
"integrity": "sha512-DGZTOlvSS4esEb1vTekY5IDoAvZAeYzR5cXVkECtQj9BVkk05zsvCAdTPo1Rz1BuI0qvqUVF+2WlIerI67iA2g==",
|
||||
"version": "0.18.1",
|
||||
"resolved": "https://registry.npmjs.org/@ladybugdb/core-darwin-arm64/-/core-darwin-arm64-0.18.1.tgz",
|
||||
"integrity": "sha512-M5YZuAONRAv3awkr+cfaibn9Da+3pgDzRiek/JabWQuz48xgzW3Vh9yQH4s8Dq/bfQo6YTsaLIBRcUCCUzCtcg==",
|
||||
"cpu": [
|
||||
"arm64"
|
||||
],
|
||||
@@ -1284,9 +1286,9 @@
|
||||
]
|
||||
},
|
||||
"node_modules/@ladybugdb/core-darwin-x64": {
|
||||
"version": "0.18.3",
|
||||
"resolved": "https://registry.npmjs.org/@ladybugdb/core-darwin-x64/-/core-darwin-x64-0.18.3.tgz",
|
||||
"integrity": "sha512-Qp6j0CM/orBlK6KD0p/s4ofkIhNUwi1hdCgMw+fj81UHugWHkVLiYV4grRBdHhyplw+snchZpTxvfpxFbkG1Cw==",
|
||||
"version": "0.18.1",
|
||||
"resolved": "https://registry.npmjs.org/@ladybugdb/core-darwin-x64/-/core-darwin-x64-0.18.1.tgz",
|
||||
"integrity": "sha512-kq+pyTskfCx++Mrbk7QssE/f/CpSuU50T8lhRtv4PaOKhC2Jf8/wAUOA17UxI594wAru3ERpqVBFUBWGcPk2ag==",
|
||||
"cpu": [
|
||||
"x64"
|
||||
],
|
||||
@@ -1297,9 +1299,9 @@
|
||||
]
|
||||
},
|
||||
"node_modules/@ladybugdb/core-linux-arm64": {
|
||||
"version": "0.18.3",
|
||||
"resolved": "https://registry.npmjs.org/@ladybugdb/core-linux-arm64/-/core-linux-arm64-0.18.3.tgz",
|
||||
"integrity": "sha512-F9miYjBuS43I7uNG199FNMqwdHJ98WA6dU3v2SZCeLXmXCdRzmYcuHQWlbNr2Tba9CX58w2XvBZoUaXZKJ/yKQ==",
|
||||
"version": "0.18.1",
|
||||
"resolved": "https://registry.npmjs.org/@ladybugdb/core-linux-arm64/-/core-linux-arm64-0.18.1.tgz",
|
||||
"integrity": "sha512-fu7ke1haa5rPINcQn0+kxQijZ0A8ZDWP9e+X8xcDH94RagDbPWwG8yFC890cGSdc/j7mTV+xkA/y/kVHpmVI6w==",
|
||||
"cpu": [
|
||||
"arm64"
|
||||
],
|
||||
@@ -1310,9 +1312,9 @@
|
||||
]
|
||||
},
|
||||
"node_modules/@ladybugdb/core-linux-x64": {
|
||||
"version": "0.18.3",
|
||||
"resolved": "https://registry.npmjs.org/@ladybugdb/core-linux-x64/-/core-linux-x64-0.18.3.tgz",
|
||||
"integrity": "sha512-AfG5RDp/f/IDctDMpTAT5+2MYNtlWT191xiQNjSaWD4X85DhY3Dzps8Qu5VteIAPih5d6mmoaKGs8q0XIjfkFA==",
|
||||
"version": "0.18.1",
|
||||
"resolved": "https://registry.npmjs.org/@ladybugdb/core-linux-x64/-/core-linux-x64-0.18.1.tgz",
|
||||
"integrity": "sha512-qp5HilHzDGuArfOyD+VyA7lVJ7IwQDKd81NZKKTmUwIAOJtdwqniYx6JZICPnlr36zFJBx/lGYoSsEzbC+TVdw==",
|
||||
"cpu": [
|
||||
"x64"
|
||||
],
|
||||
@@ -1323,9 +1325,9 @@
|
||||
]
|
||||
},
|
||||
"node_modules/@ladybugdb/core-win32-x64": {
|
||||
"version": "0.18.3",
|
||||
"resolved": "https://registry.npmjs.org/@ladybugdb/core-win32-x64/-/core-win32-x64-0.18.3.tgz",
|
||||
"integrity": "sha512-bHuFk0m9cnq0WGd9I4D8or8g6cC/BS58iatMtilqM3JpDPIQIFk6MQl6exL7P4xyWbkLwQgsrv2ToDnyoQNKvg==",
|
||||
"version": "0.18.1",
|
||||
"resolved": "https://registry.npmjs.org/@ladybugdb/core-win32-x64/-/core-win32-x64-0.18.1.tgz",
|
||||
"integrity": "sha512-vHcXr7Df2X1dbb5ORK+SBmNstd/3tApGFImbAnaWiTuLDFlAdfY8lbiSBSp3OgFjc0BB7F3GYUUdvgDRJjK3zA==",
|
||||
"cpu": [
|
||||
"x64"
|
||||
],
|
||||
@@ -1929,6 +1931,13 @@
|
||||
"dev": true,
|
||||
"license": "MIT"
|
||||
},
|
||||
"node_modules/@types/js-yaml": {
|
||||
"version": "4.0.9",
|
||||
"resolved": "https://registry.npmjs.org/@types/js-yaml/-/js-yaml-4.0.9.tgz",
|
||||
"integrity": "sha512-k4MGaQl5TGo/iipqb2UDG2UwjXziSWkh0uysQelTlJpX1qGlpUZYm8PnO4DxG1qBomtJUdYJ6qR6xdIah10JLg==",
|
||||
"dev": true,
|
||||
"license": "MIT"
|
||||
},
|
||||
"node_modules/@types/jsesc": {
|
||||
"version": "2.5.1",
|
||||
"resolved": "https://registry.npmjs.org/@types/jsesc/-/jsesc-2.5.1.tgz",
|
||||
@@ -1937,13 +1946,13 @@
|
||||
"license": "MIT"
|
||||
},
|
||||
"node_modules/@types/node": {
|
||||
"version": "26.1.1",
|
||||
"resolved": "https://registry.npmjs.org/@types/node/-/node-26.1.1.tgz",
|
||||
"integrity": "sha512-nxAkRSVkN1Y0JC1W8ky/fTfkGsMmcrRsbx+3XoZE+rMOX71kLYTV7fLXpqud1GpbpP5TuffXFqfX7fH2GgZREw==",
|
||||
"version": "25.9.5",
|
||||
"resolved": "https://registry.npmjs.org/@types/node/-/node-25.9.5.tgz",
|
||||
"integrity": "sha512-OScDchr2fwuUmWdf4kZ9h7PcJiYDVInhJizG/biAq3cAvqwYktuy/TYGGdZNMtNTFUP7rnb0NU4TUdm82kt4Rg==",
|
||||
"devOptional": true,
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"undici-types": "~8.3.0"
|
||||
"undici-types": ">=7.24.0 <7.24.7"
|
||||
}
|
||||
},
|
||||
"node_modules/@types/qs": {
|
||||
@@ -1981,6 +1990,17 @@
|
||||
"@types/node": "*"
|
||||
}
|
||||
},
|
||||
"node_modules/@types/uuid": {
|
||||
"version": "11.0.0",
|
||||
"resolved": "https://registry.npmjs.org/@types/uuid/-/uuid-11.0.0.tgz",
|
||||
"integrity": "sha512-HVyk8nj2m+jcFRNazzqyVKiZezyhDKrGUA3jlEcg/nZ6Ms+qHwocba1Y/AaVaznJTAM9xpdFSh+ptbNrhOGvZA==",
|
||||
"deprecated": "This is a stub types definition. uuid provides its own type definitions, so you do not need this installed.",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"uuid": "*"
|
||||
}
|
||||
},
|
||||
"node_modules/@vitest/coverage-v8": {
|
||||
"version": "4.1.10",
|
||||
"resolved": "https://registry.npmjs.org/@vitest/coverage-v8/-/coverage-v8-4.1.10.tgz",
|
||||
@@ -2296,20 +2316,20 @@
|
||||
}
|
||||
},
|
||||
"node_modules/body-parser": {
|
||||
"version": "2.3.0",
|
||||
"resolved": "https://registry.npmjs.org/body-parser/-/body-parser-2.3.0.tgz",
|
||||
"integrity": "sha512-2cGmJupaNgg+QUwVLAucDuWuoMZ6EX9iHDRswZ5lsNYEmwPaRknMPCLZz07yTzVq/83p4o/wzbDZbBrTvGGTIw==",
|
||||
"version": "2.2.2",
|
||||
"resolved": "https://registry.npmjs.org/body-parser/-/body-parser-2.2.2.tgz",
|
||||
"integrity": "sha512-oP5VkATKlNwcgvxi0vM0p/D3n2C3EReYVX+DNYs5TjZFn/oQt2j+4sVJtSMr18pdRr8wjTcBl6LoV+FUwzPmNA==",
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"bytes": "^3.1.2",
|
||||
"content-type": "^2.0.0",
|
||||
"content-type": "^1.0.5",
|
||||
"debug": "^4.4.3",
|
||||
"http-errors": "^2.0.1",
|
||||
"iconv-lite": "^0.7.2",
|
||||
"http-errors": "^2.0.0",
|
||||
"iconv-lite": "^0.7.0",
|
||||
"on-finished": "^2.4.1",
|
||||
"qs": "^6.15.2",
|
||||
"raw-body": "^3.0.2",
|
||||
"type-is": "^2.1.0"
|
||||
"qs": "^6.14.1",
|
||||
"raw-body": "^3.0.1",
|
||||
"type-is": "^2.0.1"
|
||||
},
|
||||
"engines": {
|
||||
"node": ">=18"
|
||||
@@ -2319,23 +2339,10 @@
|
||||
"url": "https://opencollective.com/express"
|
||||
}
|
||||
},
|
||||
"node_modules/body-parser/node_modules/content-type": {
|
||||
"version": "2.0.0",
|
||||
"resolved": "https://registry.npmjs.org/content-type/-/content-type-2.0.0.tgz",
|
||||
"integrity": "sha512-j/O/d7GcZCyNl7/hwZAb606rzqkyvaDctLmckbxLzHvFBzTJHuGEdodATcP3yIRoDrLHkIATJuvzbFlp/ki2cQ==",
|
||||
"license": "MIT",
|
||||
"engines": {
|
||||
"node": ">=18"
|
||||
},
|
||||
"funding": {
|
||||
"type": "opencollective",
|
||||
"url": "https://opencollective.com/express"
|
||||
}
|
||||
},
|
||||
"node_modules/brace-expansion": {
|
||||
"version": "5.0.7",
|
||||
"resolved": "https://registry.npmjs.org/brace-expansion/-/brace-expansion-5.0.7.tgz",
|
||||
"integrity": "sha512-7oFy703dxfY3/NLxC1fh2SUCQ0H9rmAY+5EpDVfXjUTTs+HEwR2nYaqLv+GWcTsumwxPfiz6CzCNkwXwBUwqCA==",
|
||||
"version": "5.0.6",
|
||||
"resolved": "https://registry.npmjs.org/brace-expansion/-/brace-expansion-5.0.6.tgz",
|
||||
"integrity": "sha512-kLpxurY4Z4r9sgMsyG0Z9uzsBlgiU/EFKhj/h91/8yHu0edo7XuixOIH3VcJ8kkxs6/jPzoI6U9Vj3WqbMQ94g==",
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"balanced-match": "^4.0.2"
|
||||
@@ -3042,9 +3049,9 @@
|
||||
"license": "MIT"
|
||||
},
|
||||
"node_modules/fast-uri": {
|
||||
"version": "3.1.4",
|
||||
"resolved": "https://registry.npmjs.org/fast-uri/-/fast-uri-3.1.4.tgz",
|
||||
"integrity": "sha512-8JnbkQ4juDyvYs4mgFGQqg4yCYtFDtUtmp2QIQq11ZZe5CFQ5wcqm1rqDgAh/QdMySuBnPzMUiJUNZG5N/AiQw==",
|
||||
"version": "3.1.2",
|
||||
"resolved": "https://registry.npmjs.org/fast-uri/-/fast-uri-3.1.2.tgz",
|
||||
"integrity": "sha512-rVjf7ArG3LTk+FS6Yw81V1DLuZl1bRbNrev6Tmd/9RaroeeRRJhAt7jg/6YFxbvAQXUCavSoZhPPj6oOx+5KjQ==",
|
||||
"funding": [
|
||||
{
|
||||
"type": "github",
|
||||
@@ -3403,9 +3410,9 @@
|
||||
"license": "MIT"
|
||||
},
|
||||
"node_modules/hono": {
|
||||
"version": "4.12.31",
|
||||
"resolved": "https://registry.npmjs.org/hono/-/hono-4.12.31.tgz",
|
||||
"integrity": "sha512-zJIHFrl6bq3RDd2YusFNCDlM8qUprxKswyi/OPzPyzKDdyBXDqWx8bZlZ7R+saTdSTatUmb3O7K4SspGPaEOQg==",
|
||||
"version": "4.12.26",
|
||||
"resolved": "https://registry.npmjs.org/hono/-/hono-4.12.26.tgz",
|
||||
"integrity": "sha512-uyZtpnYxM9CmQ7QsQknM4zN8EftNqhON1qYeIKM0Se67CCEe2c44xyGURwB0axX2fBDu1dqHrHAc1hmNT8ITkw==",
|
||||
"license": "MIT",
|
||||
"engines": {
|
||||
"node": ">=16.9.0"
|
||||
@@ -3582,9 +3589,9 @@
|
||||
"license": "MIT"
|
||||
},
|
||||
"node_modules/js-yaml": {
|
||||
"version": "5.0.0",
|
||||
"resolved": "https://registry.npmjs.org/js-yaml/-/js-yaml-5.0.0.tgz",
|
||||
"integrity": "sha512-GSvaPUbk1U+FMZ7rJzF+F8e5YVtu7KnD40et/5rBXXRBv2jCO9L3qCewvIDDdudC0QycTFlf6EAA+h3kxBsuUw==",
|
||||
"version": "4.3.0",
|
||||
"resolved": "https://registry.npmjs.org/js-yaml/-/js-yaml-4.3.0.tgz",
|
||||
"integrity": "sha512-1td788aAnnZ5qs7V2QIRl1owjtYpbKt749Y3xauqQgwIIGF/xXWz1wMTEBx5O3LK3lXLVuqXPdPxj2BoFHaW9Q==",
|
||||
"funding": [
|
||||
{
|
||||
"type": "github",
|
||||
@@ -3600,7 +3607,7 @@
|
||||
"argparse": "^2.0.1"
|
||||
},
|
||||
"bin": {
|
||||
"js-yaml": "bin/js-yaml.mjs"
|
||||
"js-yaml": "bin/js-yaml.js"
|
||||
}
|
||||
},
|
||||
"node_modules/jsesc": {
|
||||
@@ -5128,9 +5135,9 @@
|
||||
}
|
||||
},
|
||||
"node_modules/tar": {
|
||||
"version": "7.5.20",
|
||||
"resolved": "https://registry.npmjs.org/tar/-/tar-7.5.20.tgz",
|
||||
"integrity": "sha512-9FcyK4PA6+WbzlTM9WhQm6vB5W7cP7dUiPsv1g7YDwEQnQ1CGpK3MGlKk/ITVWMk05kHZuBhmVhiv8LZoy/PFQ==",
|
||||
"version": "7.5.16",
|
||||
"resolved": "https://registry.npmjs.org/tar/-/tar-7.5.16.tgz",
|
||||
"integrity": "sha512-56adEpPMouktRlBLXiaYFFzZ/3+JXa8P9n7WbR+ibIjtviN55mEaOkiysCnPnWm+7kkui1Dn8J9l+g6zV8731w==",
|
||||
"license": "BlueOak-1.0.0",
|
||||
"dependencies": {
|
||||
"@isaacs/fs-minipass": "^4.0.0",
|
||||
@@ -5503,9 +5510,9 @@
|
||||
}
|
||||
},
|
||||
"node_modules/undici-types": {
|
||||
"version": "8.3.0",
|
||||
"resolved": "https://registry.npmjs.org/undici-types/-/undici-types-8.3.0.tgz",
|
||||
"integrity": "sha512-j375ScV60dom+YkPFIfTLcOiPxkN/buHz5GobjLhixFuANaNs3C9l4GmrWqejgXWJ7BbJcFYpTEUkS1Ge8bpZQ==",
|
||||
"version": "7.24.6",
|
||||
"resolved": "https://registry.npmjs.org/undici-types/-/undici-types-7.24.6.tgz",
|
||||
"integrity": "sha512-WRNW+sJgj5OBN4/0JpHFqtqzhpbnV0GuB+OozA9gCL7a993SmU+1JBZCzLNxYsbMfIeDL+lTsphD5jN5N+n0zg==",
|
||||
"devOptional": true,
|
||||
"license": "MIT"
|
||||
},
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "gitnexus",
|
||||
"version": "1.6.10-rc.94",
|
||||
"version": "1.6.9",
|
||||
"description": "Graph-powered code intelligence for AI agents. Index any codebase, query via MCP or CLI.",
|
||||
"author": "Abhigyan Patwari",
|
||||
"license": "PolyForm-Noncommercial-1.0.0",
|
||||
@@ -56,7 +56,7 @@
|
||||
"version": "node scripts/sync-plugin-manifests.mjs"
|
||||
},
|
||||
"dependencies": {
|
||||
"@ladybugdb/core": "^0.18.3",
|
||||
"@ladybugdb/core": "^0.18.0",
|
||||
"@modelcontextprotocol/sdk": "^1.0.0",
|
||||
"@scarf/scarf": "^1.4.0",
|
||||
"busboy": "^1.6.0",
|
||||
@@ -70,7 +70,7 @@
|
||||
"graphology-indices": "^0.17.0",
|
||||
"graphology-utils": "^2.3.0",
|
||||
"ignore": "^7.0.5",
|
||||
"js-yaml": "^5.0.0",
|
||||
"js-yaml": "^4.1.1",
|
||||
"jsonc-parser": "^3.3.1",
|
||||
"mnemonist": "^0.40.3",
|
||||
"node-addon-api": "^8.0.0",
|
||||
@@ -105,7 +105,9 @@
|
||||
"@types/cli-progress": "^3.11.6",
|
||||
"@types/cors": "^2.8.17",
|
||||
"@types/express": "^5.0.6",
|
||||
"@types/node": "^26.0.0",
|
||||
"@types/js-yaml": "^4.0.9",
|
||||
"@types/node": "^25.6.0",
|
||||
"@types/uuid": "^11.0.0",
|
||||
"@vitest/coverage-v8": "^4.0.18",
|
||||
"gitnexus-shared": "file:../gitnexus-shared",
|
||||
"tsx": "^4.0.0",
|
||||
@@ -118,6 +120,6 @@
|
||||
}
|
||||
},
|
||||
"engines": {
|
||||
"node": "^22.18.0 || >=24.11.0"
|
||||
"node": ">=22.0.0"
|
||||
}
|
||||
}
|
||||
|
||||
@@ -117,12 +117,6 @@ const LBUG_NATIVE = [
|
||||
// to a live native DB, rm-then-rename over an existing parked copy) before
|
||||
// any open — rename semantics are exactly what differs on Windows.
|
||||
'test/unit/incremental-dirty-recovery.test.ts',
|
||||
// #2623: the incremental writeback must load VECTOR before the CodeEmbedding
|
||||
// join-delete, and the blocked path must escalate instead of crashing. The
|
||||
// win32 VECTOR gate was removed in the same PR, so this ordering must be
|
||||
// proven on the windows-latest native addon, not just Ubuntu. Budget: ~25s
|
||||
// on Linux → expect ~2min on the slowest Windows shard.
|
||||
'test/unit/incremental-vector-extension-ordering.test.ts',
|
||||
];
|
||||
|
||||
// Process spawning and CLI tests — exercise child_process with real
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
/**
|
||||
* Install the LadybugDB FTS and VECTOR extensions into the shared home (~/.lbdb)
|
||||
* up front, so every test in a sharded CI run finds them regardless of shard.
|
||||
* Install the LadybugDB FTS extension into the shared home (~/.lbdb) up front, so
|
||||
* every test in a sharded CI run finds it regardless of which shard it lands in.
|
||||
*
|
||||
* FTS-dependent tests split two ways: the LOAD-path gate (skipUnlessFtsAvailable)
|
||||
* self-installs on miss, but the FILE-path gate (requireFtsResourceOrSkip, e.g.
|
||||
@@ -17,24 +17,13 @@
|
||||
import { mkdtempSync, rmSync } from 'node:fs';
|
||||
import { tmpdir } from 'node:os';
|
||||
import { join } from 'node:path';
|
||||
import {
|
||||
initLbug,
|
||||
loadFTSExtension,
|
||||
loadVectorExtension,
|
||||
closeLbug,
|
||||
} from '../src/core/lbug/lbug-adapter.js';
|
||||
import { initLbug, loadFTSExtension, closeLbug } from '../src/core/lbug/lbug-adapter.js';
|
||||
|
||||
const dir = mkdtempSync(join(tmpdir(), 'gn-ensure-fts-'));
|
||||
try {
|
||||
await initLbug(join(dir, 'ensure-fts.lbug'));
|
||||
const ok = await loadFTSExtension(undefined, { policy: 'auto' });
|
||||
console.log(ok ? 'FTS extension ready.' : 'FTS extension unavailable (continuing).');
|
||||
// VECTOR rides the same pre-install (#2623): the win32 gate is gone, so the
|
||||
// vector suites genuinely run on Windows/macOS — installing once here means
|
||||
// every sharded test process LOADs from ~/.lbdb instead of racing its own
|
||||
// out-of-process INSTALL (bounded 15s each when the server is unreachable).
|
||||
const vec = await loadVectorExtension(undefined, { policy: 'auto' });
|
||||
console.log(vec ? 'VECTOR extension ready.' : 'VECTOR extension unavailable (continuing).');
|
||||
} catch (err) {
|
||||
console.warn(`ensure-fts: skipped (${err instanceof Error ? err.message : String(err)})`);
|
||||
} finally {
|
||||
|
||||
@@ -18,7 +18,6 @@ import { boundedCheckpointBeforeExit } from '../core/lbug/shutdown-helpers.js';
|
||||
import {
|
||||
getOsPageSize,
|
||||
isLbugCheckpointIoError,
|
||||
isLbugCheckpointBusyError,
|
||||
isLbugPageSizeFrameError,
|
||||
isPageSizeAwareLadybug,
|
||||
isWalCorruptionError,
|
||||
@@ -1625,16 +1624,8 @@ const analyzeCommandImpl = async (
|
||||
}
|
||||
|
||||
if (isLbugCheckpointIoError(err)) {
|
||||
// #2599: when the checkpoint IO error also looks busy/locked, another
|
||||
// handle holds the store open — name that actionable cause alongside the
|
||||
// threshold hint (the original error is preserved so the hint still fires).
|
||||
const heldOpen = isLbugCheckpointBusyError(err)
|
||||
? ` Another process may hold the store open (a running \`gitnexus mcp\` server, or a\n` +
|
||||
` stale reader) — close other GitNexus processes on this repo, then retry.\n`
|
||||
: '';
|
||||
cliError(
|
||||
` LadybugDB failed while rotating/removing WAL checkpoint files.\n` +
|
||||
heldOpen +
|
||||
` This can happen when auto-checkpoint runs at the default threshold (~16MB).\n` +
|
||||
` Retry with a larger checkpoint threshold to reduce checkpoint frequency:\n` +
|
||||
` gitnexus analyze --wal-checkpoint-threshold ${RECOMMENDED_WAL_CHECKPOINT_THRESHOLD}\n` +
|
||||
|
||||
@@ -12,16 +12,8 @@ import {
|
||||
type EmbeddingRuntimeResolution,
|
||||
} from '../core/embeddings/runtime-install.js';
|
||||
import { cudaRedirectDoctorStatus } from '../core/embeddings/onnxruntime-node-resolver.js';
|
||||
import {
|
||||
checkLbugNative,
|
||||
probeFtsExtensionLoad,
|
||||
probeVectorExtensionLoad,
|
||||
} from '../core/lbug/native-check.js';
|
||||
import {
|
||||
getEffectiveBufferPoolSize,
|
||||
getOsPageSize,
|
||||
isPageSizeAwareLadybug,
|
||||
} from '../core/lbug/lbug-config.js';
|
||||
import { checkLbugNative, probeFtsExtensionLoad } from '../core/lbug/native-check.js';
|
||||
import { getOsPageSize, isPageSizeAwareLadybug } from '../core/lbug/lbug-config.js';
|
||||
import { diagnoseExtensionLoad } from '../core/lbug/extension-load-error.js';
|
||||
import { getExtensionInstallPolicy } from '../core/lbug/extension-loader.js';
|
||||
import { t } from './i18n/index.js';
|
||||
@@ -154,22 +146,6 @@ export function pageSizeDoctorLines(
|
||||
return lines;
|
||||
}
|
||||
|
||||
/**
|
||||
* The hintless buffer-pool doctor line (#2631) — the pool the next Database
|
||||
* open in THIS process would get. Same plain-params testable-helper shape as
|
||||
* pageSizeDoctorLines above. `pool` is getEffectiveBufferPoolSize(): `0` is
|
||||
* the pass-through sentinel for LadybugDB's native 80%-of-RAM default, never
|
||||
* printed as "0 MiB". `envRaw` (the raw GITNEXUS_LBUG_BUFFER_POOL_SIZE value)
|
||||
* marks operator-supplied absolute values as "(env override)" — no scaling
|
||||
* suffix: the hintless default is deliberately unscaled (#2557), and an env
|
||||
* value is absolute, so a "×N" note would misdescribe both.
|
||||
*/
|
||||
export function poolSizeDoctorLine(pool: number, envRaw: string | undefined): string {
|
||||
const value = pool === 0 ? 'native 80% of RAM' : `${Math.round(pool / (1024 * 1024))} MiB`;
|
||||
const envNote = envRaw !== undefined && envRaw.trim().length > 0 ? ' (env override)' : '';
|
||||
return ` ${padDisplayEnd('pool size', 10)}${value}${envNote}`;
|
||||
}
|
||||
|
||||
export const doctorCommand = async () => {
|
||||
const fingerprint = getRuntimeFingerprint();
|
||||
const capabilities = getRuntimeCapabilities();
|
||||
@@ -188,11 +164,6 @@ export const doctorCommand = async () => {
|
||||
for (const line of pageSizeDoctorLines(getOsPageSize(), fingerprint.ladybugdb)) {
|
||||
console.log(line);
|
||||
}
|
||||
// Hintless buffer pool for the next DB open (#2631). Literal label like
|
||||
// the page size line above (no i18n key).
|
||||
console.log(
|
||||
poolSizeDoctorLine(getEffectiveBufferPoolSize(), process.env.GITNEXUS_LBUG_BUFFER_POOL_SIZE),
|
||||
);
|
||||
const nativeCheck = checkLbugNative();
|
||||
if (nativeCheck.ok) {
|
||||
console.log(` ${padDisplayEnd('native', 10)}✓ lbugjs.node loaded`);
|
||||
@@ -224,32 +195,8 @@ export const doctorCommand = async () => {
|
||||
console.log(` ${padDisplayEnd('', 18)}${remedy}`);
|
||||
}
|
||||
}
|
||||
// Live LOAD probe for VECTOR too (#2623). The static capability is just
|
||||
// `platform !== 'win32'`, so it printed "available" on the very machines
|
||||
// where analyze was failing to load the extension — the same contradiction
|
||||
// #2374 fixed for FTS above, and exactly what #2623's reporter saw while
|
||||
// every incremental analyze died on an unloaded VECTOR extension.
|
||||
const vectorProbe = nativeCheck.ok
|
||||
? await probeVectorExtensionLoad()
|
||||
: { loaded: false, reason: 'LadybugDB native module (lbugjs.node) failed to load' };
|
||||
console.log(
|
||||
` ${label('doctor.labels.vectorIndex', 18)}${vectorProbe.loaded ? 'available' : 'unavailable'}`,
|
||||
);
|
||||
if (!vectorProbe.loaded && vectorProbe.reason) {
|
||||
console.log(` ${padDisplayEnd('', 18)}${vectorProbe.reason}`);
|
||||
const { kind, remedy } = diagnoseExtensionLoad(vectorProbe.reason, 'VECTOR');
|
||||
if (kind !== 'unknown') {
|
||||
console.log(` ${padDisplayEnd('', 18)}${remedy}`);
|
||||
}
|
||||
}
|
||||
// Semantic mode follows the probe, not the platform: without a loadable
|
||||
// VECTOR extension the index can be neither built nor queried, so search is
|
||||
// really on exact scan no matter what the platform would allow.
|
||||
console.log(
|
||||
` ${label('doctor.labels.semanticMode', 18)}${
|
||||
vectorProbe.loaded ? capabilities.semanticMode : 'exact-scan'
|
||||
}`,
|
||||
);
|
||||
console.log(` ${label('doctor.labels.vectorIndex', 18)}${capabilities.vector}`);
|
||||
console.log(` ${label('doctor.labels.semanticMode', 18)}${capabilities.semanticMode}`);
|
||||
// Surface the optional-extension install policy so offline users can see
|
||||
// whether analyze/query will reach the network (extension.ladybugdb.com).
|
||||
// Literal label (like the 'native' line) to avoid adding i18n keys.
|
||||
|
||||
@@ -3,7 +3,6 @@ import fs from 'fs/promises';
|
||||
import nodePath from 'path';
|
||||
import type { Path } from 'path-scurry';
|
||||
import { logger } from '../core/logger.js';
|
||||
import { getCoreExcludesFilePath, getGitInfoExcludePath } from '../storage/git.js';
|
||||
|
||||
const DEFAULT_IGNORE_LIST = new Set([
|
||||
// Version Control
|
||||
@@ -351,8 +350,6 @@ export const isHardcodedIgnoredDirectory = (name: string): boolean => {
|
||||
export interface IgnoreOptions {
|
||||
/** Skip .gitignore parsing, only read .gitnexusignore. Defaults to GITNEXUS_NO_GITIGNORE env var. */
|
||||
noGitignore?: boolean;
|
||||
/** Skip core.excludesFile and $GIT_COMMON_DIR/info/exclude. Defaults to GITNEXUS_NO_GLOBAL_IGNORE env var. */
|
||||
noGlobalIgnore?: boolean;
|
||||
}
|
||||
|
||||
export const loadIgnoreRules = async (
|
||||
@@ -362,32 +359,6 @@ export const loadIgnoreRules = async (
|
||||
const ig = ignore();
|
||||
let hasRules = false;
|
||||
|
||||
// Mirror git's own precedence for ignore sources (gitignore(5)): patterns
|
||||
// from core.excludesFile are consulted first (lowest precedence — git's
|
||||
// real global, all-repos file), then $GIT_COMMON_DIR/info/exclude
|
||||
// (per-repo, untracked — no write access to the repo needed), then
|
||||
// .gitignore/.gitnexusignore below. Later ig.add() calls win on
|
||||
// conflicting patterns, matching git's own last-match-wins semantics (#2606).
|
||||
const skipGlobalIgnore = options?.noGlobalIgnore ?? !!process.env.GITNEXUS_NO_GLOBAL_IGNORE;
|
||||
if (!skipGlobalIgnore) {
|
||||
const globalSources = [
|
||||
getCoreExcludesFilePath(repoPath),
|
||||
getGitInfoExcludePath(repoPath),
|
||||
].filter((candidate): candidate is string => candidate !== null);
|
||||
for (const sourcePath of globalSources) {
|
||||
try {
|
||||
const content = await fs.readFile(sourcePath, 'utf-8');
|
||||
ig.add(content);
|
||||
hasRules = true;
|
||||
} catch (err: unknown) {
|
||||
const code = (err as NodeJS.ErrnoException).code;
|
||||
if (code !== 'ENOENT') {
|
||||
logger.warn(` Warning: could not read ${sourcePath}: ${(err as Error).message}`);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Allow users to bypass .gitignore parsing (e.g. when .gitignore accidentally excludes source files)
|
||||
const skipGitignore = options?.noGitignore ?? !!process.env.GITNEXUS_NO_GITIGNORE;
|
||||
const filenames = skipGitignore ? ['.gitnexusignore'] : ['.gitignore', '.gitnexusignore'];
|
||||
|
||||
@@ -29,7 +29,6 @@ interface HttpConfig {
|
||||
maxAttempts: number;
|
||||
retryCapMs: number;
|
||||
minIntervalMs: number;
|
||||
requestDimensions?: number;
|
||||
}
|
||||
|
||||
export interface EmbeddingRequestOptions {
|
||||
@@ -107,26 +106,20 @@ const paceHttpRequest = async (minIntervalMs: number, signal?: AbortSignal): Pro
|
||||
};
|
||||
|
||||
/**
|
||||
* Stable lead of a {@link readConfig} malformed dims-env error. `readConfig`
|
||||
* throws a plain `Error` (not an {@link HttpEmbeddingError}) for a malformed
|
||||
* `GITNEXUS_EMBEDDING_DIMS` or `GITNEXUS_EMBEDDING_REQUEST_DIMS` because it's a
|
||||
* *config* mistake, not an endpoint failure — so the CLI recognizes it by this
|
||||
* lead ({@link isHttpEmbeddingDimsError}) and prints a clean config message
|
||||
* instead of a raw stack dump. Each var names itself so the message points the
|
||||
* operator at the variable they actually set, not a sibling. See #2385.
|
||||
* Stable lead of the {@link readConfig} malformed-`GITNEXUS_EMBEDDING_DIMS`
|
||||
* error. `readConfig` throws a plain `Error` (not an {@link HttpEmbeddingError})
|
||||
* because this is a *config* mistake, not an endpoint failure — so the CLI
|
||||
* recognizes it by this lead ({@link isHttpEmbeddingDimsError}) and prints a
|
||||
* clean config message instead of a raw stack dump. See #2385.
|
||||
*/
|
||||
const dimsEnvErrorLead = (name: string): string => `${name} must be a positive integer`;
|
||||
const EMBEDDING_DIMS_ENV_ERROR_LEAD = dimsEnvErrorLead('GITNEXUS_EMBEDDING_DIMS');
|
||||
const EMBEDDING_REQUEST_DIMS_ENV_ERROR_LEAD = dimsEnvErrorLead('GITNEXUS_EMBEDDING_REQUEST_DIMS');
|
||||
const EMBEDDING_DIMS_ENV_ERROR_LEAD = 'GITNEXUS_EMBEDDING_DIMS must be a positive integer';
|
||||
|
||||
/**
|
||||
* @internal Exported for the CLI analyze error handler. True when `message` is a
|
||||
* {@link readConfig} malformed dims-env config error (a plain `Error`) — for
|
||||
* either `GITNEXUS_EMBEDDING_DIMS` or `GITNEXUS_EMBEDDING_REQUEST_DIMS`.
|
||||
* @internal Exported for the CLI analyze error handler. True when `message` is
|
||||
* the {@link readConfig} malformed-DIMS config error (a plain `Error`).
|
||||
*/
|
||||
export const isHttpEmbeddingDimsError = (message: string): boolean =>
|
||||
message.includes(EMBEDDING_DIMS_ENV_ERROR_LEAD) ||
|
||||
message.includes(EMBEDDING_REQUEST_DIMS_ENV_ERROR_LEAD);
|
||||
message.includes(EMBEDDING_DIMS_ENV_ERROR_LEAD);
|
||||
|
||||
/**
|
||||
* Build config from the current process.env snapshot.
|
||||
@@ -154,23 +147,6 @@ const readConfig = (): HttpConfig | null => {
|
||||
dimensions = parsed;
|
||||
}
|
||||
|
||||
const rawRequestDims = process.env.GITNEXUS_EMBEDDING_REQUEST_DIMS?.trim();
|
||||
let requestDimensions = dimensions;
|
||||
if (rawRequestDims) {
|
||||
if (/^(omit|none|off|false|0)$/i.test(rawRequestDims)) {
|
||||
requestDimensions = undefined;
|
||||
} else {
|
||||
if (!/^\d+$/.test(rawRequestDims)) {
|
||||
throw new Error(`${EMBEDDING_REQUEST_DIMS_ENV_ERROR_LEAD}, got "${rawRequestDims}"`);
|
||||
}
|
||||
const parsed = parseInt(rawRequestDims, 10);
|
||||
if (parsed <= 0) {
|
||||
throw new Error(`${EMBEDDING_REQUEST_DIMS_ENV_ERROR_LEAD}, got "${rawRequestDims}"`);
|
||||
}
|
||||
requestDimensions = parsed;
|
||||
}
|
||||
}
|
||||
|
||||
return {
|
||||
baseUrl: baseUrl.replace(/\/+$/, ''),
|
||||
model,
|
||||
@@ -187,7 +163,6 @@ const readConfig = (): HttpConfig | null => {
|
||||
300_000,
|
||||
),
|
||||
minIntervalMs: parseNonNegativeIntegerEnv('GITNEXUS_EMBEDDING_MIN_INTERVAL_MS', 0, 300_000),
|
||||
requestDimensions,
|
||||
};
|
||||
};
|
||||
|
||||
@@ -308,9 +283,9 @@ const isEmbeddingItem = (item: unknown): item is EmbeddingItem =>
|
||||
* the `dimensions` field in the request body. Endpoints that implement
|
||||
* Matryoshka truncation (OpenAI text-embedding-3-*, Cohere embed-v3,
|
||||
* Voyage) return a truncated vector at that size; endpoints that do not
|
||||
* recognise the field may ignore it or return 400. Set
|
||||
* `GITNEXUS_EMBEDDING_REQUEST_DIMS=omit` for strict backends while keeping
|
||||
* `GITNEXUS_EMBEDDING_DIMS` set to the returned vector size.
|
||||
* recognise the field may ignore it or return 400. Leave
|
||||
* `GITNEXUS_EMBEDDING_DIMS` unset for strict backends that reject
|
||||
* unknown fields.
|
||||
*/
|
||||
const httpEmbedBatch = async (
|
||||
url: string,
|
||||
@@ -459,7 +434,7 @@ export const httpEmbed = async (
|
||||
config.model,
|
||||
config.apiKey,
|
||||
batchIndex,
|
||||
config.requestDimensions,
|
||||
config.dimensions,
|
||||
requestOptions,
|
||||
config.maxAttempts,
|
||||
config.retryCapMs,
|
||||
@@ -516,7 +491,7 @@ export const httpEmbedQuery = async (
|
||||
config.model,
|
||||
config.apiKey,
|
||||
0,
|
||||
config.requestDimensions,
|
||||
config.dimensions,
|
||||
requestOptions,
|
||||
config.maxAttempts,
|
||||
config.retryCapMs,
|
||||
|
||||
@@ -3,10 +3,8 @@
|
||||
*
|
||||
* `module.registerHooks` — the synchronous ESM/CJS resolution-hook API the
|
||||
* embedding-stack resolvers rely on — was added in Node 22.15.0 (and 23.5.0 on
|
||||
* the 23.x line). The gitnexus engines floor is `^22.18.0 || >=24.11.0`, so
|
||||
* every supported runtime exposes it — but `engines` is advisory (not
|
||||
* engine-strict), so a below-floor Node (22.0–22.14, or the unsupported
|
||||
* 23.0–23.4 line) can still run, where the export is absent.
|
||||
* the 23.x line). The gitnexus engines floor is `>=22.0.0`, which admits Node
|
||||
* 22.0–22.14 AND 23.0–23.4, where the export is absent.
|
||||
*
|
||||
* In this `"type": "module"` package, a *static named* import of a missing
|
||||
* builtin export (`import { registerHooks } from 'node:module'`) is a
|
||||
|
||||
@@ -52,8 +52,8 @@
|
||||
* per-resolution cost is a single string comparison.
|
||||
*
|
||||
* `module.registerHooks` is marked `@experimental` and requires Node >= 22.15
|
||||
* (below the gitnexus engines floor of `^22.18.0 || >=24.11.0`). On below-floor
|
||||
* runtimes it is absent and this is a graceful no-op: embeddings then resolve onnxruntime-common exactly
|
||||
* (the gitnexus engines floor is >= 22.0.0). On older runtimes it is absent and
|
||||
* this is a graceful no-op: embeddings then resolve onnxruntime-common exactly
|
||||
* as before — fine on hoisted layouts. Any failure during installation is
|
||||
* swallowed.
|
||||
*/
|
||||
@@ -100,9 +100,9 @@ export const ensureOnnxRuntimeCommonResolvable = (): void => {
|
||||
attempted = true;
|
||||
|
||||
try {
|
||||
// Node < 22.15 / < 23.5 (below the gitnexus engines floor of
|
||||
// ^22.18.0 || >=24.11.0): no synchronous hooks API. Degrade gracefully —
|
||||
// the import still works on hoisted layouts.
|
||||
// Node < 22.15 / < 23.5 (the gitnexus engines floor is >= 22.0.0): no
|
||||
// synchronous hooks API. Degrade gracefully — the import still works on
|
||||
// hoisted layouts.
|
||||
const registerHooks = getRegisterHooks();
|
||||
if (typeof registerHooks !== 'function') return;
|
||||
|
||||
|
||||
@@ -36,8 +36,8 @@
|
||||
* So CUDA-12 hosts, Windows (DirectML), macOS, and CPU-only hosts are
|
||||
* untouched. Idempotent; any failure is swallowed and leaves the default
|
||||
* resolution exactly as before. `module.registerHooks` requires Node >= 22.15
|
||||
* (below the gitnexus engines floor of `^22.18.0 || >=24.11.0`); on below-floor
|
||||
* runtimes the redirect is a no-op, but the default copy's CUDA major is still probed so an
|
||||
* (the gitnexus engines floor is >= 22.0.0); on older runtimes the redirect is
|
||||
* a no-op, but the default copy's CUDA major is still probed so an
|
||||
* already-matching host (e.g. CUDA 12 + transformers' CUDA-12 build) keeps
|
||||
* auto-selecting the GPU.
|
||||
* `npm link` / symlinked local-dev checkouts are a known caveat: `resolveOurOrtNodeDir`/
|
||||
|
||||
@@ -191,7 +191,7 @@ export const ensureEmbeddingStackResolvable = (): void => {
|
||||
hookAttempted = true;
|
||||
|
||||
try {
|
||||
// Node < 22.15 / < 23.5 (below the engines floor of ^22.18.0 || >=24.11.0): no synchronous hooks
|
||||
// Node < 22.15 / < 23.5 (engines floor is >= 22.0.0): no synchronous hooks
|
||||
// API. Degrade gracefully — normally-installed stacks still resolve; only
|
||||
// the runtime-prefix fallback is unavailable. Reachable now that the import
|
||||
// is a namespace access (see node-module-compat.ts) rather than a static
|
||||
|
||||
@@ -1031,8 +1031,6 @@ const LBUG_OPEN_RETRY_PATTERNS = [
|
||||
'lock held by another process',
|
||||
];
|
||||
|
||||
// Cross-repo bridge RO open retry. Catalogued as entry 5 of the lbug-config
|
||||
// retry-budget registry; caps back-off so total wait ~3s.
|
||||
const LBUG_OPEN_RETRY_ATTEMPTS = 10;
|
||||
const LBUG_OPEN_RETRY_BASE_MS = 100;
|
||||
/** Cap individual back-off delays so the total wait is bounded (~3s). */
|
||||
|
||||
@@ -59,7 +59,6 @@ export interface SpringBeanCandidateAdapter {
|
||||
}
|
||||
|
||||
type OwnedTypeNamesByOwner = ReadonlyMap<string, ReadonlySet<string>>;
|
||||
type RecognizedAnnotationNames = { readonly has: (value: string) => boolean };
|
||||
|
||||
function simpleNameOf(def: SymbolDefinition): string | undefined {
|
||||
const qualifiedName = def.qualifiedName;
|
||||
@@ -153,11 +152,7 @@ function hasVisibleTypeBinding(
|
||||
return false;
|
||||
}
|
||||
|
||||
function wildcardImportTarget(
|
||||
parsed: ParsedFile,
|
||||
simpleName: string,
|
||||
recognizedAnnotations: RecognizedAnnotationNames,
|
||||
): string | undefined {
|
||||
function wildcardImportTarget(parsed: ParsedFile, simpleName: string): string | undefined {
|
||||
const wildcardPackages = new Set(
|
||||
parsed.parsedImports
|
||||
.filter((entry) => entry.kind === 'wildcard')
|
||||
@@ -166,40 +161,37 @@ function wildcardImportTarget(
|
||||
if (wildcardPackages.size !== 1) return undefined;
|
||||
const [packageName] = wildcardPackages;
|
||||
const target = `${packageName}.${simpleName}`;
|
||||
return recognizedAnnotations.has(target) ? target : undefined;
|
||||
return SPRING_BEAN_STEREOTYPES.has(target) ? target : undefined;
|
||||
}
|
||||
|
||||
/** Build a scope-aware Spring annotation resolver shared by framework hooks. */
|
||||
export function createSpringAnnotationNameResolver(indexes: ScopeResolutionIndexes) {
|
||||
const ownedTypeNamesByOwner = buildOwnedTypeNamesByOwner(indexes);
|
||||
return (
|
||||
rawName: string,
|
||||
parsed: ParsedFile,
|
||||
enclosingScope: ScopeId | null,
|
||||
recognizedAnnotations: RecognizedAnnotationNames,
|
||||
isPackageVisibilityIncomplete: boolean,
|
||||
): string | undefined => {
|
||||
if (rawName.includes('.')) {
|
||||
return recognizedAnnotations.has(rawName) ? rawName : undefined;
|
||||
}
|
||||
function resolveSpringAnnotation(
|
||||
rawName: string,
|
||||
parsed: ParsedFile,
|
||||
enclosingScope: ScopeId | null,
|
||||
indexes: ScopeResolutionIndexes,
|
||||
ownedTypeNamesByOwner: OwnedTypeNamesByOwner,
|
||||
isPackageVisibilityIncomplete: boolean,
|
||||
): string | undefined {
|
||||
if (rawName.includes('.')) {
|
||||
return SPRING_BEAN_STEREOTYPES.has(rawName) ? rawName : undefined;
|
||||
}
|
||||
|
||||
if (hasLexicalTypeDeclaration(enclosingScope, rawName, indexes)) return undefined;
|
||||
if (hasInheritedTypeDeclaration(enclosingScope, rawName, indexes, ownedTypeNamesByOwner)) {
|
||||
return undefined;
|
||||
}
|
||||
if (hasLexicalTypeDeclaration(enclosingScope, rawName, indexes)) return undefined;
|
||||
if (hasInheritedTypeDeclaration(enclosingScope, rawName, indexes, ownedTypeNamesByOwner)) {
|
||||
return undefined;
|
||||
}
|
||||
|
||||
const explicitImports = explicitImportTargets(parsed, rawName);
|
||||
if (explicitImports.size > 0) {
|
||||
if (explicitImports.size !== 1) return undefined;
|
||||
const [imported] = explicitImports;
|
||||
return recognizedAnnotations.has(imported) ? imported : undefined;
|
||||
}
|
||||
const explicitImports = explicitImportTargets(parsed, rawName);
|
||||
if (explicitImports.size > 0) {
|
||||
if (explicitImports.size !== 1) return undefined;
|
||||
const [imported] = explicitImports;
|
||||
return SPRING_BEAN_STEREOTYPES.has(imported) ? imported : undefined;
|
||||
}
|
||||
|
||||
const wildcardTarget = wildcardImportTarget(parsed, rawName, recognizedAnnotations);
|
||||
if (wildcardTarget === undefined || isPackageVisibilityIncomplete) return undefined;
|
||||
const wildcardTarget = wildcardImportTarget(parsed, rawName);
|
||||
if (wildcardTarget === undefined || isPackageVisibilityIncomplete) return undefined;
|
||||
|
||||
return hasVisibleTypeBinding(enclosingScope, rawName, indexes) ? undefined : wildcardTarget;
|
||||
};
|
||||
return hasVisibleTypeBinding(enclosingScope, rawName, indexes) ? undefined : wildcardTarget;
|
||||
}
|
||||
|
||||
/** Build a language hook that enriches Class nodes after scope resolution. */
|
||||
@@ -210,7 +202,7 @@ export function createSpringBeanCandidateAttacher(adapter: SpringBeanCandidateAd
|
||||
nodeLookup: GraphNodeLookup,
|
||||
indexes: ScopeResolutionIndexes,
|
||||
): void => {
|
||||
const resolveSpringAnnotation = createSpringAnnotationNameResolver(indexes);
|
||||
const ownedTypeNamesByOwner = buildOwnedTypeNamesByOwner(indexes);
|
||||
for (const parsed of parsedFiles) {
|
||||
for (const fact of adapter.getClassAnnotationFacts(parsed.filePath)) {
|
||||
const classScope = indexes.scopeTree.getScope(fact.classScopeId);
|
||||
@@ -229,7 +221,8 @@ export function createSpringBeanCandidateAttacher(adapter: SpringBeanCandidateAd
|
||||
rawName,
|
||||
parsed,
|
||||
classScope.parent,
|
||||
SPRING_BEAN_STEREOTYPES,
|
||||
indexes,
|
||||
ownedTypeNamesByOwner,
|
||||
adapter.isPackageVisibilityIncomplete(parsed.filePath),
|
||||
);
|
||||
if (annotation !== undefined) recognized.add(annotation);
|
||||
|
||||
@@ -1,166 +0,0 @@
|
||||
import type { GraphNode } from 'gitnexus-shared';
|
||||
import type { KnowledgeGraph } from '../../../graph/types.js';
|
||||
import { generateId } from '../../../../lib/utils.js';
|
||||
|
||||
export const SPRING_CONFIG_DESCRIPTION = 'Spring configuration property';
|
||||
|
||||
export interface SpringValueConsumer {
|
||||
readonly kind: 'value';
|
||||
readonly fieldName: string;
|
||||
readonly line: number;
|
||||
readonly keys: readonly string[];
|
||||
}
|
||||
|
||||
export interface SpringConfigurationPropertiesConsumer {
|
||||
readonly kind: 'configuration-properties';
|
||||
readonly className: string;
|
||||
readonly line: number;
|
||||
readonly prefix: string;
|
||||
}
|
||||
|
||||
export type SpringConfigConsumer = SpringValueConsumer | SpringConfigurationPropertiesConsumer;
|
||||
|
||||
export interface SpringConfigConsumerBatch {
|
||||
readonly filePath: string;
|
||||
readonly consumers: readonly SpringConfigConsumer[];
|
||||
}
|
||||
|
||||
function closestNode(
|
||||
candidates: readonly GraphNode[],
|
||||
filePath: string,
|
||||
name: string,
|
||||
line: number,
|
||||
): GraphNode | undefined {
|
||||
return candidates
|
||||
.filter((node) => node.properties.filePath === filePath && node.properties.name === name)
|
||||
.sort(
|
||||
(left, right) =>
|
||||
Math.abs(Number(left.properties.startLine ?? 0) - line) -
|
||||
Math.abs(Number(right.properties.startLine ?? 0) - line),
|
||||
)[0];
|
||||
}
|
||||
|
||||
function markUnresolved(node: GraphNode, key: string): void {
|
||||
const marker = `Spring config unresolved: ${key}`;
|
||||
const existing =
|
||||
typeof node.properties.description === 'string' ? node.properties.description : '';
|
||||
if (existing.includes(marker)) return;
|
||||
node.properties.description = existing.length > 0 ? `${existing}; ${marker}` : marker;
|
||||
}
|
||||
|
||||
function relaxedName(value: string): string {
|
||||
return value.toLowerCase().replace(/[-_.]/g, '');
|
||||
}
|
||||
|
||||
function isSpringConfigNode(node: GraphNode): boolean {
|
||||
return (
|
||||
node.label === 'Property' &&
|
||||
typeof node.properties.description === 'string' &&
|
||||
node.properties.description.startsWith(SPRING_CONFIG_DESCRIPTION)
|
||||
);
|
||||
}
|
||||
|
||||
/**
|
||||
* Attach normalized, language-provider-produced Spring consumers to config
|
||||
* keys already present in the shared graph.
|
||||
*/
|
||||
export function bindSpringConfigConsumers(
|
||||
graph: KnowledgeGraph,
|
||||
batches: readonly SpringConfigConsumerBatch[],
|
||||
): void {
|
||||
if (batches.length === 0) return;
|
||||
|
||||
const configNodes: GraphNode[] = [];
|
||||
const propertyNodes: GraphNode[] = [];
|
||||
const classNodes: GraphNode[] = [];
|
||||
for (const node of graph.iterNodes()) {
|
||||
if (isSpringConfigNode(node)) configNodes.push(node);
|
||||
else if (node.label === 'Property') propertyNodes.push(node);
|
||||
else if (node.label === 'Class' || node.label === 'Record') classNodes.push(node);
|
||||
}
|
||||
|
||||
const keyNodes = new Map<string, GraphNode[]>();
|
||||
for (const node of configNodes) {
|
||||
const key = String(node.properties.name);
|
||||
const bucket = keyNodes.get(key) ?? [];
|
||||
bucket.push(node);
|
||||
keyNodes.set(key, bucket);
|
||||
}
|
||||
|
||||
const propertiesByOwner = new Map<string, GraphNode[]>();
|
||||
for (const rel of graph.iterRelationshipsByType('HAS_PROPERTY')) {
|
||||
const property = graph.getNode(rel.targetId);
|
||||
if (property?.label !== 'Property' || isSpringConfigNode(property)) continue;
|
||||
const members = propertiesByOwner.get(rel.sourceId) ?? [];
|
||||
members.push(property);
|
||||
propertiesByOwner.set(rel.sourceId, members);
|
||||
}
|
||||
|
||||
const addBinding = (
|
||||
source: GraphNode,
|
||||
target: GraphNode,
|
||||
reason: string,
|
||||
confidence: number,
|
||||
): void => {
|
||||
const edgeId = generateId('USES', `${source.id}->${target.id}:${reason}`);
|
||||
graph.addRelationship({
|
||||
id: edgeId,
|
||||
sourceId: source.id,
|
||||
targetId: target.id,
|
||||
type: 'USES',
|
||||
confidence,
|
||||
reason,
|
||||
});
|
||||
};
|
||||
|
||||
for (const { filePath, consumers } of batches) {
|
||||
for (const consumer of consumers) {
|
||||
if (consumer.kind === 'value') {
|
||||
const field = closestNode(propertyNodes, filePath, consumer.fieldName, consumer.line);
|
||||
if (field === undefined) continue;
|
||||
for (const key of consumer.keys) {
|
||||
const matches = keyNodes.get(key) ?? [];
|
||||
if (matches.length === 0) {
|
||||
markUnresolved(field, key);
|
||||
continue;
|
||||
}
|
||||
for (const match of matches) {
|
||||
addBinding(field, match, `spring-config:@Value ${key}`, 1);
|
||||
}
|
||||
}
|
||||
continue;
|
||||
}
|
||||
|
||||
const owner = closestNode(classNodes, filePath, consumer.className, consumer.line);
|
||||
if (owner === undefined) continue;
|
||||
const prefix = `${consumer.prefix}.`;
|
||||
const matches = configNodes.filter((node) => {
|
||||
const key = String(node.properties.name);
|
||||
return key === consumer.prefix || key.startsWith(prefix);
|
||||
});
|
||||
if (matches.length === 0) {
|
||||
markUnresolved(owner, consumer.prefix);
|
||||
continue;
|
||||
}
|
||||
for (const match of matches) {
|
||||
addBinding(owner, match, `spring-config:@ConfigurationProperties ${consumer.prefix}`, 0.95);
|
||||
}
|
||||
|
||||
for (const field of propertiesByOwner.get(owner.id) ?? []) {
|
||||
const fieldName = relaxedName(String(field.properties.name));
|
||||
for (const match of matches) {
|
||||
const key = String(match.properties.name);
|
||||
const suffix = key === consumer.prefix ? '' : key.slice(prefix.length);
|
||||
const firstSegment = suffix.split(/[.\[]/, 1)[0];
|
||||
if (firstSegment.length === 0 || relaxedName(firstSegment) !== fieldName) continue;
|
||||
addBinding(
|
||||
field,
|
||||
match,
|
||||
`spring-config:@ConfigurationProperties field ${consumer.prefix}`,
|
||||
0.95,
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1,19 +0,0 @@
|
||||
import type { AnalysisFeatureDescriptor } from '../../../analysis-features.js';
|
||||
|
||||
function isSpringApplicationConfig(filePath: string): boolean {
|
||||
const base = filePath.replaceAll('\\', '/').split('/').pop() ?? '';
|
||||
return /^application(?:-[^.]+)?\.(?:properties|ya?ml)$/i.test(base);
|
||||
}
|
||||
|
||||
/** Durable completeness contract for Java Spring configuration bindings. */
|
||||
export const SPRING_CONFIG_BINDINGS_FEATURE: AnalysisFeatureDescriptor = {
|
||||
id: 'spring.config-bindings',
|
||||
version: 1,
|
||||
// Java sources need consumer extraction even without config files (missing
|
||||
// placeholders still get unresolved markers). Config-only repositories also
|
||||
// need a one-time rebuild to backfill language-agnostic Property nodes.
|
||||
appliesTo: (filePaths) =>
|
||||
filePaths.some(
|
||||
(filePath) => filePath.toLowerCase().endsWith('.java') || isSpringApplicationConfig(filePath),
|
||||
),
|
||||
};
|
||||
@@ -9,7 +9,6 @@ import {
|
||||
type JvmPackageFact,
|
||||
} from '../jvm/package-facts.js';
|
||||
import { getJavaPackageFact, setJavaPackageFact } from './package-facts.js';
|
||||
import type { JavaSpringConfigConsumerFact } from './spring-config-bindings.js';
|
||||
|
||||
export type JavaClassAnnotationFact = ClassAnnotationFact;
|
||||
|
||||
@@ -17,16 +16,13 @@ export interface JavaCaptureSideChannel {
|
||||
readonly kind: 'java';
|
||||
readonly packageFact: JvmPackageFact;
|
||||
readonly classAnnotations: readonly JavaClassAnnotationFact[];
|
||||
readonly springConfigConsumers?: readonly JavaSpringConfigConsumerFact[];
|
||||
}
|
||||
|
||||
const classAnnotations = createClassAnnotationFactStore();
|
||||
const springConfigConsumers = new Map<string, readonly JavaSpringConfigConsumerFact[]>();
|
||||
|
||||
/** Clear facts retained by a prior workspace pass in a long-lived process. */
|
||||
export function clearJavaClassAnnotationFacts(): void {
|
||||
classAnnotations.clear();
|
||||
springConfigConsumers.clear();
|
||||
}
|
||||
|
||||
/** Store the annotation syntax collected by Java's existing scope-query traversal. */
|
||||
@@ -37,35 +33,17 @@ export function setJavaClassAnnotationFacts(
|
||||
classAnnotations.set(filePath, facts);
|
||||
}
|
||||
|
||||
export function setJavaSpringConfigConsumerFacts(
|
||||
filePath: string,
|
||||
facts: readonly JavaSpringConfigConsumerFact[],
|
||||
): void {
|
||||
if (facts.length === 0) springConfigConsumers.delete(filePath);
|
||||
else springConfigConsumers.set(filePath, facts);
|
||||
}
|
||||
|
||||
export function getJavaSpringConfigConsumerFacts(
|
||||
filePath: string,
|
||||
): readonly JavaSpringConfigConsumerFact[] {
|
||||
return springConfigConsumers.get(filePath) ?? [];
|
||||
}
|
||||
|
||||
/** Snapshot worker-local Java annotation facts for ParsedFile serialization. */
|
||||
export function collectJavaCaptureSideChannel(
|
||||
filePath: string,
|
||||
): JavaCaptureSideChannel | undefined {
|
||||
const facts = classAnnotations.get(filePath);
|
||||
const configConsumers = springConfigConsumers.get(filePath) ?? [];
|
||||
const packageFact = getJavaPackageFact(filePath);
|
||||
if (facts.length === 0 && configConsumers.length === 0 && packageFact === undefined) {
|
||||
return undefined;
|
||||
}
|
||||
if (facts.length === 0 && packageFact === undefined) return undefined;
|
||||
return {
|
||||
kind: 'java',
|
||||
packageFact: packageFact ?? UNKNOWN_JVM_PACKAGE_FACT,
|
||||
classAnnotations: facts,
|
||||
...(configConsumers.length > 0 ? { springConfigConsumers: configConsumers } : {}),
|
||||
};
|
||||
}
|
||||
|
||||
@@ -84,15 +62,10 @@ export function applyJavaCaptureSideChannel(parsed: ParsedFile): void {
|
||||
!Array.isArray(data.classAnnotations)
|
||||
) {
|
||||
setJavaClassAnnotationFacts(parsed.filePath, []);
|
||||
setJavaSpringConfigConsumerFacts(parsed.filePath, []);
|
||||
setJavaPackageFact(parsed.filePath, UNKNOWN_JVM_PACKAGE_FACT);
|
||||
return;
|
||||
}
|
||||
setJavaClassAnnotationFacts(parsed.filePath, data.classAnnotations);
|
||||
setJavaSpringConfigConsumerFacts(
|
||||
parsed.filePath,
|
||||
Array.isArray(data.springConfigConsumers) ? data.springConfigConsumers : [],
|
||||
);
|
||||
setJavaPackageFact(
|
||||
parsed.filePath,
|
||||
isJvmPackageFact(data.packageFact) ? data.packageFact : UNKNOWN_JVM_PACKAGE_FACT,
|
||||
|
||||
@@ -32,13 +32,9 @@ import { getJavaParser, getJavaScopeQuery } from './query.js';
|
||||
import { recordCacheHit, recordCacheMiss } from './cache-stats.js';
|
||||
import { getTreeSitterBufferSize } from '../../constants.js';
|
||||
import { parseSourceSafe } from '../../../tree-sitter/safe-parse.js';
|
||||
import {
|
||||
setJavaClassAnnotationFacts,
|
||||
setJavaSpringConfigConsumerFacts,
|
||||
} from './capture-side-channel.js';
|
||||
import { setJavaClassAnnotationFacts } from './capture-side-channel.js';
|
||||
import { captureJavaPackageFact } from './package-facts.js';
|
||||
import { synthesizeCallableFlowCaptures } from '../../utils/callable-flow-captures.js';
|
||||
import { captureJavaSpringConfigConsumerFacts } from './spring-config-bindings.js';
|
||||
|
||||
/** Declaration anchors that carry function-like arity metadata. */
|
||||
const FUNCTION_DECL_TAGS = ['@declaration.method', '@declaration.constructor'] as const;
|
||||
@@ -154,29 +150,6 @@ export function emitJavaScopeCaptures(
|
||||
continue;
|
||||
}
|
||||
|
||||
// Normalize a `new`-expression receiver to its constructed type's simple
|
||||
// name: `new Local().inner()` binds the WHOLE `object_creation_expression`
|
||||
// as `@reference.receiver`, so its raw text is `"new Local()"` — a string
|
||||
// that can never match a scope binding, so the compound-receiver resolver
|
||||
// silently falls through to name-only fallback resolution and picks the
|
||||
// wrong same-named method on a collision (#2564). Rewriting the text to
|
||||
// just `Local` lets Case 2 (class-name / static receiver) in
|
||||
// receiver-bound-calls.ts resolve it via its normal MRO walk. Mirrors the
|
||||
// established `normalizePhpReceiver` precedent (php/captures.ts) — a
|
||||
// language-local capture rewrite, no shared-pipeline change.
|
||||
if (grouped['@reference.receiver'] !== undefined) {
|
||||
const receiverNode = nodeIfType(nodeMap['@reference.receiver'], 'object_creation_expression');
|
||||
const typeNode = receiverNode?.childForFieldName('type');
|
||||
const simpleName = typeNode ? javaBaseSimpleNameOf(typeNode) : undefined;
|
||||
if (simpleName !== undefined) {
|
||||
grouped['@reference.receiver'] = syntheticCapture(
|
||||
'@reference.receiver',
|
||||
receiverNode!,
|
||||
simpleName,
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
// Filter read.member when it's a child of method_invocation or assignment.
|
||||
// `@reference.read.member` is captured directly on the `field_access` node.
|
||||
if (grouped['@reference.read.member'] !== undefined) {
|
||||
@@ -284,10 +257,6 @@ export function emitJavaScopeCaptures(
|
||||
}
|
||||
|
||||
setJavaClassAnnotationFacts(filePath, materializeClassAnnotationFacts(classAnnotations));
|
||||
setJavaSpringConfigConsumerFacts(
|
||||
filePath,
|
||||
captureJavaSpringConfigConsumerFacts(tree.rootNode, filePath),
|
||||
);
|
||||
|
||||
return [
|
||||
...resolveVarTypeBindings(out),
|
||||
@@ -372,49 +341,23 @@ function synthesizeJavaAnonymousClassDeclarations(rootNode: SyntaxNode): Capture
|
||||
// constant's class extends its HOST ENUM (javac semantics), so the
|
||||
// inherits reference names the enum — giving `mroFor(E$N) ∋ E` and
|
||||
// keeping bare calls from the body to the enum's own helpers alive
|
||||
// through the ownership gate's MRO arm.
|
||||
// through the ownership gate's MRO arm. No receiver typeBinding piece:
|
||||
// constants are not variable initializers; `E.A.hook()` dispatch rides
|
||||
// the existing enum receiver machinery.
|
||||
for (const constant of rootNode.descendantsOfType('enum_constant')) {
|
||||
const name = synthesizeJavaAnonymousClassName(constant);
|
||||
if (name === undefined) continue;
|
||||
const body = constant.childForFieldName?.('body');
|
||||
if (body === null || body === undefined || body.type !== 'class_body') continue;
|
||||
out.push({
|
||||
'@declaration.class': nodeToCapture('@declaration.class', body),
|
||||
'@declaration.name': syntheticCapture('@declaration.name', body, name),
|
||||
});
|
||||
const hostEnum = javaEnclosingEnumNameOf(constant);
|
||||
const bodyNode = constant.childForFieldName?.('body');
|
||||
const isBodied = bodyNode !== null && bodyNode !== undefined && bodyNode.type === 'class_body';
|
||||
const bodiedName = synthesizeJavaAnonymousClassName(constant);
|
||||
if (bodiedName !== undefined && isBodied) {
|
||||
if (hostEnum !== undefined) {
|
||||
out.push({
|
||||
'@declaration.class': nodeToCapture('@declaration.class', bodyNode),
|
||||
'@declaration.name': syntheticCapture('@declaration.name', bodyNode, bodiedName),
|
||||
});
|
||||
if (hostEnum !== undefined) {
|
||||
out.push({
|
||||
'@reference.inherits': nodeToCapture('@reference.inherits', bodyNode),
|
||||
'@reference.name': syntheticCapture('@reference.name', bodyNode, hostEnum),
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
// Receiver dispatch (#2561): `E.CONST.method()` resolves through the
|
||||
// generic compound-receiver chain walk, which looks up each dotted
|
||||
// segment via the owning class scope's `typeBindings` map — the same
|
||||
// mechanism a field declaration uses (`private User user;` binds
|
||||
// `user` on the class scope). Binding the constant's own simple name
|
||||
// there — to its synthesized `E$N` class when bodied (MRO includes E,
|
||||
// so members inherited from the enum still resolve), or to the host
|
||||
// enum itself when body-less — makes `E.CONST.method()` resolve with
|
||||
// no changes to the shared receiver-binding machinery.
|
||||
//
|
||||
// A bodied constant binds ONLY to its `E$N` class, never the host enum:
|
||||
// if name synthesis fails on a malformed/error-recovery tree (`bodiedName`
|
||||
// undefined despite a real body), emit nothing rather than silently
|
||||
// misattributing an OVERRIDING constant's receiver to the enum's own
|
||||
// (non-overridden) method — a wrong edge is worse than no edge. Mirrors
|
||||
// the `object_creation_expression` branch, which skips on synthesis
|
||||
// failure. `hostEnum` is used only for genuinely body-less constants.
|
||||
const constantNameNode = constant.childForFieldName?.('name');
|
||||
const constantType = isBodied ? bodiedName : hostEnum;
|
||||
if (constantNameNode !== null && constantNameNode !== undefined && constantType !== undefined) {
|
||||
out.push({
|
||||
'@type-binding.annotation': nodeToCapture('@type-binding.annotation', constant),
|
||||
'@type-binding.name': nodeToCapture('@type-binding.name', constantNameNode),
|
||||
'@type-binding.type': syntheticCapture('@type-binding.type', constant, constantType),
|
||||
'@reference.inherits': nodeToCapture('@reference.inherits', body),
|
||||
'@reference.name': syntheticCapture('@reference.name', body, hostEnum),
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
@@ -30,7 +30,6 @@ import {
|
||||
} from './index.js';
|
||||
import { populateJavaPackageSiblings } from './package-siblings.js';
|
||||
import { attachSpringBeanCandidateMetadata } from './spring-bean-metadata.js';
|
||||
import { attachJavaSpringConfigBindings } from './spring-config-bindings.js';
|
||||
import {
|
||||
applyJavaCaptureSideChannel,
|
||||
clearJavaClassAnnotationFacts,
|
||||
@@ -84,10 +83,7 @@ const javaScopeResolver: ScopeResolver = {
|
||||
|
||||
populateNamespaceSiblings: populateJavaPackageSiblings,
|
||||
populateRangeBindings: populateJavaCrossFileReturnTypes,
|
||||
emitPostResolutionEdges: (graph, parsedFiles, nodeLookup, indexes, ctx) => {
|
||||
attachSpringBeanCandidateMetadata(graph, parsedFiles, nodeLookup, indexes);
|
||||
attachJavaSpringConfigBindings(graph, parsedFiles, nodeLookup, indexes, ctx);
|
||||
},
|
||||
emitPostResolutionEdges: attachSpringBeanCandidateMetadata,
|
||||
};
|
||||
|
||||
export { javaScopeResolver };
|
||||
|
||||
@@ -1,267 +0,0 @@
|
||||
import type { KnowledgeGraph } from '../../../graph/types.js';
|
||||
import type { GraphNodeLookup } from '../../scope-resolution/graph-bridge/node-lookup.js';
|
||||
import type { ScopeResolutionIndexes } from '../../model/scope-resolution-indexes.js';
|
||||
import { makeScopeId, type ParsedFile, type ScopeId } from 'gitnexus-shared';
|
||||
import {
|
||||
bindSpringConfigConsumers,
|
||||
type SpringConfigConsumer,
|
||||
} from '../../frameworks/spring/config-bindings.js';
|
||||
import { createSpringAnnotationNameResolver } from '../../frameworks/spring/bean-candidates.js';
|
||||
import { parseSourceSafe } from '../../../tree-sitter/safe-parse.js';
|
||||
import { nodeToCapture, type SyntaxNode } from '../../utils/ast-helpers.js';
|
||||
import { getJavaParser } from './query.js';
|
||||
import { getJavaSpringConfigConsumerFacts } from './capture-side-channel.js';
|
||||
import { isJavaPackageSiblingVisibilityIncomplete } from './package-siblings.js';
|
||||
|
||||
const VALUE_ANNOTATION = 'org.springframework.beans.factory.annotation.Value';
|
||||
const CONFIGURATION_PROPERTIES_ANNOTATION =
|
||||
'org.springframework.boot.context.properties.ConfigurationProperties';
|
||||
|
||||
interface JavaAnnotation {
|
||||
readonly name: string;
|
||||
readonly node: SyntaxNode;
|
||||
}
|
||||
|
||||
interface JavaImports {
|
||||
readonly exact: ReadonlySet<string>;
|
||||
readonly wildcard: ReadonlySet<string>;
|
||||
readonly localTypes: ReadonlySet<string>;
|
||||
}
|
||||
|
||||
export interface JavaSpringConfigConsumerFact {
|
||||
readonly consumer: SpringConfigConsumer;
|
||||
readonly annotationName: string;
|
||||
readonly classScopeId: ScopeId;
|
||||
}
|
||||
|
||||
function collectJavaImports(root: SyntaxNode): JavaImports {
|
||||
const exact = new Set<string>();
|
||||
const wildcard = new Set<string>();
|
||||
const localTypes = new Set<string>();
|
||||
|
||||
for (const node of root.descendantsOfType('import_declaration')) {
|
||||
const imported = node.text
|
||||
.replace(/^\s*import\s+(?:static\s+)?/, '')
|
||||
.replace(/;\s*$/, '')
|
||||
.trim();
|
||||
if (imported.endsWith('.*')) wildcard.add(imported.slice(0, -2));
|
||||
else exact.add(imported);
|
||||
}
|
||||
|
||||
for (const type of [
|
||||
'class_declaration',
|
||||
'interface_declaration',
|
||||
'enum_declaration',
|
||||
'record_declaration',
|
||||
'annotation_type_declaration',
|
||||
]) {
|
||||
for (const node of root.descendantsOfType(type)) {
|
||||
const name = node.childForFieldName('name')?.text;
|
||||
if (name) localTypes.add(name);
|
||||
}
|
||||
}
|
||||
return { exact, wildcard, localTypes };
|
||||
}
|
||||
|
||||
function annotationsOn(node: SyntaxNode): JavaAnnotation[] {
|
||||
const modifiers = node.namedChildren.find((child) => child.type === 'modifiers');
|
||||
if (modifiers === undefined) return [];
|
||||
const annotations: JavaAnnotation[] = [];
|
||||
for (const child of modifiers.namedChildren) {
|
||||
if (child.type !== 'annotation' && child.type !== 'marker_annotation') continue;
|
||||
const name = child.childForFieldName('name')?.text ?? child.firstNamedChild?.text;
|
||||
if (name) annotations.push({ name, node: child });
|
||||
}
|
||||
return annotations;
|
||||
}
|
||||
|
||||
function resolvesToAnnotation(
|
||||
rawName: string,
|
||||
canonicalName: string,
|
||||
imports: JavaImports,
|
||||
): boolean {
|
||||
if (rawName.includes('.')) return rawName === canonicalName;
|
||||
if (imports.localTypes.has(rawName)) return false;
|
||||
if (imports.exact.has(canonicalName)) return true;
|
||||
const packageName = canonicalName.slice(0, canonicalName.lastIndexOf('.'));
|
||||
return imports.wildcard.has(packageName);
|
||||
}
|
||||
|
||||
function decodeJavaStringLiteral(literal: string): string {
|
||||
const delimiterLength = literal.startsWith('"""') && literal.endsWith('"""') ? 3 : 1;
|
||||
return literal
|
||||
.slice(delimiterLength, -delimiterLength)
|
||||
.replace(/\\u([0-9a-fA-F]{4})/g, (_match, hex: string) =>
|
||||
String.fromCharCode(Number.parseInt(hex, 16)),
|
||||
)
|
||||
.replace(/\\(["'\\btnfr])/g, (_match, escaped: string) => {
|
||||
const controls: Record<string, string> = {
|
||||
b: '\b',
|
||||
t: '\t',
|
||||
n: '\n',
|
||||
f: '\f',
|
||||
r: '\r',
|
||||
};
|
||||
return controls[escaped] ?? escaped;
|
||||
});
|
||||
}
|
||||
|
||||
function javaStringLiterals(annotation: SyntaxNode): string[] {
|
||||
return annotation
|
||||
.descendantsOfType('string_literal')
|
||||
.map((literal) => decodeJavaStringLiteral(literal.text));
|
||||
}
|
||||
|
||||
/** Extract statically readable Spring placeholder keys from a Java annotation. */
|
||||
export function parseValuePlaceholderKeys(annotation: SyntaxNode): string[] {
|
||||
const keys = new Set<string>();
|
||||
for (const literal of javaStringLiterals(annotation)) {
|
||||
for (const match of literal.matchAll(/\$\{([^{}]+)\}/g)) {
|
||||
const key = match[1].split(':', 1)[0].trim();
|
||||
if (/^[A-Za-z0-9_.-]+$/.test(key)) keys.add(key);
|
||||
}
|
||||
}
|
||||
return [...keys];
|
||||
}
|
||||
|
||||
/** Extract `prefix`/`value` (or the positional value) from the annotation. */
|
||||
export function parseConfigurationPropertiesPrefix(annotation: SyntaxNode): string | null {
|
||||
const named = annotation.descendantsOfType('element_value_pair').find((pair) => {
|
||||
const key = pair.childForFieldName('key')?.text;
|
||||
return key === 'prefix' || key === 'value';
|
||||
});
|
||||
const namedValue = named?.childForFieldName('value');
|
||||
const argumentsNode = annotation.childForFieldName('arguments');
|
||||
const literalNode =
|
||||
(namedValue?.type === 'string_literal'
|
||||
? namedValue
|
||||
: namedValue?.descendantsOfType('string_literal')[0]) ??
|
||||
(named === undefined
|
||||
? argumentsNode?.namedChildren.find((child) => child.type === 'string_literal')
|
||||
: undefined);
|
||||
if (literalNode === undefined) return null;
|
||||
const prefix = decodeJavaStringLiteral(literalNode.text)
|
||||
.trim()
|
||||
.replace(/^\.+|\.+$/g, '');
|
||||
return /^[A-Za-z0-9_.-]+$/.test(prefix) ? prefix : null;
|
||||
}
|
||||
|
||||
function classScopeId(filePath: string, declaration: SyntaxNode): ScopeId {
|
||||
return makeScopeId({
|
||||
filePath,
|
||||
range: nodeToCapture('@scope.class', declaration).range,
|
||||
kind: 'Class',
|
||||
});
|
||||
}
|
||||
|
||||
function enclosingClass(node: SyntaxNode): SyntaxNode | undefined {
|
||||
let current = node.parent;
|
||||
while (current !== null) {
|
||||
if (current.type === 'class_declaration' || current.type === 'record_declaration') {
|
||||
return current;
|
||||
}
|
||||
current = current.parent;
|
||||
}
|
||||
return undefined;
|
||||
}
|
||||
|
||||
/** Collect config facts from the Java parser's existing AST (no reparse). */
|
||||
export function captureJavaSpringConfigConsumerFacts(
|
||||
root: SyntaxNode,
|
||||
filePath: string,
|
||||
): JavaSpringConfigConsumerFact[] {
|
||||
const imports = collectJavaImports(root);
|
||||
const facts: JavaSpringConfigConsumerFact[] = [];
|
||||
|
||||
for (const field of root.descendantsOfType('field_declaration')) {
|
||||
const annotations = annotationsOn(field).filter((annotation) =>
|
||||
resolvesToAnnotation(annotation.name, VALUE_ANNOTATION, imports),
|
||||
);
|
||||
if (annotations.length === 0) continue;
|
||||
const owner = enclosingClass(field);
|
||||
if (owner === undefined) continue;
|
||||
for (const declarator of field.namedChildren.filter(
|
||||
(child) => child.type === 'variable_declarator',
|
||||
)) {
|
||||
const fieldName = declarator.childForFieldName('name')?.text;
|
||||
if (!fieldName) continue;
|
||||
for (const annotation of annotations) {
|
||||
const keys = parseValuePlaceholderKeys(annotation.node);
|
||||
if (keys.length > 0) {
|
||||
facts.push({
|
||||
consumer: { kind: 'value', fieldName, line: field.startPosition.row + 1, keys },
|
||||
annotationName: annotation.name,
|
||||
classScopeId: classScopeId(filePath, owner),
|
||||
});
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
for (const type of ['class_declaration', 'record_declaration']) {
|
||||
for (const declaration of root.descendantsOfType(type)) {
|
||||
const className = declaration.childForFieldName('name')?.text;
|
||||
if (!className) continue;
|
||||
for (const annotation of annotationsOn(declaration)) {
|
||||
if (!resolvesToAnnotation(annotation.name, CONFIGURATION_PROPERTIES_ANNOTATION, imports)) {
|
||||
continue;
|
||||
}
|
||||
const prefix = parseConfigurationPropertiesPrefix(annotation.node);
|
||||
if (prefix !== null) {
|
||||
facts.push({
|
||||
consumer: {
|
||||
kind: 'configuration-properties',
|
||||
className,
|
||||
line: declaration.startPosition.row + 1,
|
||||
prefix,
|
||||
},
|
||||
annotationName: annotation.name,
|
||||
classScopeId: classScopeId(filePath, declaration),
|
||||
});
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
return facts;
|
||||
}
|
||||
|
||||
/** Parse Java consumers for focused unit tests; production reuses the worker AST. */
|
||||
export function extractJavaSpringConfigConsumers(source: string): SpringConfigConsumer[] {
|
||||
const tree = parseSourceSafe(getJavaParser(), source);
|
||||
return captureJavaSpringConfigConsumerFacts(tree.rootNode, '<memory>').map(
|
||||
(fact) => fact.consumer,
|
||||
);
|
||||
}
|
||||
|
||||
/** Java ScopeResolver post-resolution hook for Spring configuration consumers. */
|
||||
export function attachJavaSpringConfigBindings(
|
||||
graph: KnowledgeGraph,
|
||||
parsedFiles: readonly ParsedFile[],
|
||||
_nodeLookup: GraphNodeLookup,
|
||||
indexes: ScopeResolutionIndexes,
|
||||
_ctx: { readonly fileContents: ReadonlyMap<string, string> },
|
||||
): void {
|
||||
const resolveAnnotation = createSpringAnnotationNameResolver(indexes);
|
||||
const recognizedAnnotations = new Set([VALUE_ANNOTATION, CONFIGURATION_PROPERTIES_ANNOTATION]);
|
||||
const batches: Array<{ filePath: string; consumers: SpringConfigConsumer[] }> = [];
|
||||
for (const parsed of parsedFiles) {
|
||||
const consumers: SpringConfigConsumer[] = [];
|
||||
for (const fact of getJavaSpringConfigConsumerFacts(parsed.filePath)) {
|
||||
const classScope = indexes.scopeTree.getScope(fact.classScopeId);
|
||||
if (classScope === undefined || classScope.kind !== 'Class') continue;
|
||||
const expectedAnnotation =
|
||||
fact.consumer.kind === 'value' ? VALUE_ANNOTATION : CONFIGURATION_PROPERTIES_ANNOTATION;
|
||||
const enclosingScope = fact.consumer.kind === 'value' ? classScope.id : classScope.parent;
|
||||
const resolved = resolveAnnotation(
|
||||
fact.annotationName,
|
||||
parsed,
|
||||
enclosingScope,
|
||||
recognizedAnnotations,
|
||||
isJavaPackageSiblingVisibilityIncomplete(parsed.filePath),
|
||||
);
|
||||
if (resolved === expectedAnnotation) consumers.push(fact.consumer);
|
||||
}
|
||||
if (consumers.length > 0) batches.push({ filePath: parsed.filePath, consumers });
|
||||
}
|
||||
bindSpringConfigConsumers(graph, batches);
|
||||
}
|
||||
@@ -8,9 +8,10 @@
|
||||
* 1. **Per-name import statements** — `import a, b` and
|
||||
* `from m import x, y` decompose to one match per imported name
|
||||
* (see `import-decomposer.ts`).
|
||||
* 2. **Receiver type bindings** — methods emit an implicit `self` / `cls`
|
||||
* binding, and `__init__` assignments from annotated parameters emit
|
||||
* class-scoped instance-field bindings (see `receiver-binding.ts`).
|
||||
* 2. **Receiver type bindings** — each `function_definition` inside a
|
||||
* class body emits a `@type-binding.self` (or `@type-binding.cls`
|
||||
* for `@classmethod`) capture so Pass-4 attaches the implicit
|
||||
* receiver (see `receiver-binding.ts`).
|
||||
*
|
||||
* Pure given the input source text. No I/O, no globals consulted.
|
||||
*/
|
||||
@@ -24,10 +25,7 @@ import {
|
||||
} from '../../utils/ast-helpers.js';
|
||||
import { splitImportStatement } from './import-decomposer.js';
|
||||
import { getPythonParser, getPythonScopeQuery } from './query.js';
|
||||
import {
|
||||
synthesizeConstructorFieldTypeBindings,
|
||||
synthesizeReceiverTypeBinding,
|
||||
} from './receiver-binding.js';
|
||||
import { synthesizeReceiverTypeBinding } from './receiver-binding.js';
|
||||
import { synthesizeDependsReferences } from './depends-references.js';
|
||||
import { computePythonArityMetadata } from './arity-metadata.js';
|
||||
import { recordCacheHit, recordCacheMiss } from './cache-stats.js';
|
||||
@@ -135,7 +133,6 @@ export function emitPythonScopeCaptures(
|
||||
if (fnNode !== null) {
|
||||
const synth = synthesizeReceiverTypeBinding(fnNode);
|
||||
if (synth !== null) out.push(synth);
|
||||
out.push(...synthesizeConstructorFieldTypeBindings(fnNode));
|
||||
for (const depRef of synthesizeDependsReferences(fnNode)) out.push(depRef);
|
||||
}
|
||||
continue;
|
||||
|
||||
@@ -119,10 +119,7 @@ export function interpretPythonTypeBinding(captures: CaptureMatch): ParsedTypeBi
|
||||
// `cls` is a self-like receiver; share the source label so downstream
|
||||
// `Registry.lookup` Step 2 treats them identically.
|
||||
else if (captures['@type-binding.cls'] !== undefined) source = 'self';
|
||||
else if (captures['@type-binding.instance-field'] !== undefined) {
|
||||
source =
|
||||
captures['@type-binding.parameter'] !== undefined ? 'parameter-annotation' : 'annotation';
|
||||
} else if (captures['@type-binding.constructor'] !== undefined) source = 'constructor-inferred';
|
||||
else if (captures['@type-binding.constructor'] !== undefined) source = 'constructor-inferred';
|
||||
else if (captures['@type-binding.annotation'] !== undefined) source = 'annotation';
|
||||
else if (captures['@type-binding.alias'] !== undefined) source = 'assignment-inferred';
|
||||
else if (captures['@type-binding.return'] !== undefined) source = 'return-annotation';
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
/**
|
||||
* Synthesize implicit receiver and constructor-assigned field type bindings
|
||||
* for methods.
|
||||
* Synthesize `@type-binding.self` / `@type-binding.cls` captures for
|
||||
* methods.
|
||||
*
|
||||
* Tree-sitter can't easily express "the first parameter of a function
|
||||
* defined directly inside a class body" via a single static query.
|
||||
@@ -113,114 +113,3 @@ export function synthesizeReceiverTypeBinding(fnNode: SyntaxNode): CaptureMatch
|
||||
'@type-binding.type': syntheticCapture('@type-binding.type', first, className),
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Synthesize class-scope field bindings for the common Python constructor
|
||||
* injection pattern:
|
||||
*
|
||||
* def __init__(self, service: Service):
|
||||
* self.service = service
|
||||
*
|
||||
* An explicit field annotation (`self.service: Service = ...`) is also
|
||||
* accepted and takes precedence over a parameter annotation. Deliberately do
|
||||
* not infer from arbitrary unannotated RHS expressions: the receiver resolver
|
||||
* needs a declared type, not a name-only guess.
|
||||
*/
|
||||
export function synthesizeConstructorFieldTypeBindings(fnNode: SyntaxNode): CaptureMatch[] {
|
||||
if (fnNode.childForFieldName('name')?.text !== '__init__') return [];
|
||||
if (findEnclosingClassDefinition(fnNode) === null) return [];
|
||||
if (hasDecorator(fnNode, 'staticmethod') || hasDecorator(fnNode, 'classmethod')) return [];
|
||||
|
||||
const receiver = synthesizeReceiverTypeBinding(fnNode);
|
||||
const receiverName = receiver?.['@type-binding.self']?.text;
|
||||
if (receiverName === undefined) return [];
|
||||
|
||||
const parameters = fnNode.childForFieldName('parameters');
|
||||
const body = fnNode.childForFieldName('body');
|
||||
if (parameters === null || body === null) return [];
|
||||
|
||||
const parameterTypes = new Map<string, string>();
|
||||
for (let i = 0; i < parameters.namedChildCount; i++) {
|
||||
const parameter = parameters.namedChild(i);
|
||||
if (parameter === null) continue;
|
||||
const name = firstParameterName(parameter);
|
||||
const annotation = parameter.childForFieldName('type');
|
||||
if (name !== null && annotation !== null) parameterTypes.set(name, annotation.text);
|
||||
}
|
||||
|
||||
type Candidate = { readonly match: CaptureMatch; readonly explicit: boolean };
|
||||
const candidates = new Map<string, Candidate>();
|
||||
|
||||
const stack: SyntaxNode[] = [body];
|
||||
while (stack.length > 0) {
|
||||
const node = stack.pop()!;
|
||||
if (
|
||||
node !== body &&
|
||||
(node.type === 'function_definition' ||
|
||||
node.type === 'lambda' ||
|
||||
node.type === 'class_definition' ||
|
||||
node.type === 'if_statement' ||
|
||||
node.type === 'for_statement' ||
|
||||
node.type === 'while_statement' ||
|
||||
node.type === 'try_statement' ||
|
||||
node.type === 'match_statement')
|
||||
) {
|
||||
continue;
|
||||
}
|
||||
|
||||
if (node.type === 'assignment') {
|
||||
const left = node.childForFieldName('left');
|
||||
const right = node.childForFieldName('right');
|
||||
if (left?.type === 'attribute') {
|
||||
const object = left.childForFieldName('object');
|
||||
const field = left.childForFieldName('attribute');
|
||||
if (object?.type === 'identifier' && object.text === receiverName && field !== null) {
|
||||
const explicitType = node.childForFieldName('type');
|
||||
const parameterType =
|
||||
right?.type === 'identifier' ? parameterTypes.get(right.text) : undefined;
|
||||
const typeName = explicitType?.text ?? parameterType;
|
||||
if (typeName !== undefined) {
|
||||
const explicit = explicitType !== null;
|
||||
const existing = candidates.get(field.text);
|
||||
if (existing === undefined || explicit || !existing.explicit) {
|
||||
candidates.set(field.text, {
|
||||
explicit,
|
||||
match: {
|
||||
'@type-binding.name': syntheticCapture('@type-binding.name', field, field.text),
|
||||
'@type-binding.type': syntheticCapture(
|
||||
'@type-binding.type',
|
||||
explicitType ?? right ?? field,
|
||||
typeName,
|
||||
),
|
||||
...(explicit
|
||||
? {}
|
||||
: {
|
||||
'@type-binding.parameter': syntheticCapture(
|
||||
'@type-binding.parameter',
|
||||
right ?? field,
|
||||
'1',
|
||||
),
|
||||
}),
|
||||
'@type-binding.instance-field': syntheticCapture(
|
||||
'@type-binding.instance-field',
|
||||
node,
|
||||
'1',
|
||||
),
|
||||
},
|
||||
});
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Push in reverse so the LIFO walk visits source order. That keeps Map
|
||||
// insertion order (and therefore emitted capture order) deterministic.
|
||||
for (let i = node.namedChildCount - 1; i >= 0; i--) {
|
||||
const child = node.namedChild(i);
|
||||
if (child !== null) stack.push(child);
|
||||
}
|
||||
}
|
||||
|
||||
return [...candidates.values()].map(({ match }) => match);
|
||||
}
|
||||
|
||||
@@ -36,23 +36,15 @@ export function pythonFunctionDefinitionLabel(
|
||||
// ─── bindingScopeFor ──────────────────────────────────────────────────────
|
||||
|
||||
/** Python has no block scope, so the central extractor's "innermost
|
||||
* enclosing scope" default is already correct for ordinary bindings.
|
||||
* Constructor-injected instance fields are the exception: their marker is
|
||||
* anchored inside `__init__`, but compound receiver resolution needs the
|
||||
* field type on the enclosing Class scope. */
|
||||
* enclosing scope" default is already correct: `for x in …` creates
|
||||
* `x` in the enclosing function/module scope (because we never emit a
|
||||
* `@scope.block` for the for-loop body), comprehension variables stay
|
||||
* in their expression context, etc. Returns `null` to delegate. */
|
||||
export function pythonBindingScopeFor(
|
||||
decl: CaptureMatch,
|
||||
innermost: Scope,
|
||||
tree: ScopeTree,
|
||||
_decl: CaptureMatch,
|
||||
_innermost: Scope,
|
||||
_tree: ScopeTree,
|
||||
): ScopeId | null {
|
||||
if (decl['@type-binding.instance-field'] !== undefined) {
|
||||
let current: Scope | undefined = innermost;
|
||||
while (current !== undefined) {
|
||||
if (current.kind === 'Class') return current.id;
|
||||
if (current.parent === null) break;
|
||||
current = tree.getScope(current.parent);
|
||||
}
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
|
||||
@@ -2,23 +2,8 @@ import type { CaptureMatch, ParsedImport, ParsedTypeBinding, TypeRef } from 'git
|
||||
|
||||
const REF_PREFIX_RE = /^&\s*(mut\s+)?/;
|
||||
const PTR_PREFIX_RE = /^\*\s*(const|mut)?\s*/;
|
||||
const DYN_PREFIX_RE = /^dyn\s+/;
|
||||
const ENUM_VARIANT_NAMES = new Set(['Some', 'None', 'Ok', 'Err']);
|
||||
|
||||
// `dyn Trait`, `&dyn Trait`, `Box<dyn Trait>` all name a trait object whose
|
||||
// receiver-dispatch target is the trait itself (#2604) — strip the `dyn`
|
||||
// keyword and any auto-trait/lifetime bound list (`dyn Trait + Send`) down to
|
||||
// the principal trait name. Reference/pointer sigils are stripped by the
|
||||
// caller first; wrapper unwrapping (Box<T> etc.) runs before this so the
|
||||
// unwrapped inner text still gets the same treatment.
|
||||
function stripDynBound(t: string): string {
|
||||
if (!DYN_PREFIX_RE.test(t)) return t;
|
||||
t = t.replace(DYN_PREFIX_RE, '');
|
||||
const plus = t.indexOf('+');
|
||||
if (plus !== -1) t = t.slice(0, plus);
|
||||
return t.trim();
|
||||
}
|
||||
|
||||
// ─── interpretImport ──────────────────────────────────────────────────────
|
||||
|
||||
export function interpretRustImport(captures: CaptureMatch): ParsedImport | null {
|
||||
@@ -113,7 +98,6 @@ export function normalizeRustTypeName(text: string): string {
|
||||
const inner = extractFirstGenericArg(t);
|
||||
if (inner !== null) t = inner;
|
||||
}
|
||||
t = stripDynBound(t);
|
||||
const bracket = t.indexOf('<');
|
||||
if (bracket !== -1) t = t.slice(0, bracket);
|
||||
// Take last segment of qualified paths (crate::foo::Bar → Bar)
|
||||
@@ -174,7 +158,6 @@ function normalizeRustReturnType(text: string): string {
|
||||
}
|
||||
}
|
||||
}
|
||||
t = stripDynBound(t);
|
||||
const bracket = t.indexOf('<');
|
||||
if (bracket !== -1) t = t.slice(0, bracket);
|
||||
const lastColon = t.lastIndexOf('::');
|
||||
|
||||
@@ -10,7 +10,6 @@ const RUST_SCOPE_QUERY = `
|
||||
(enum_item) @scope.class
|
||||
(union_item) @scope.class
|
||||
(function_item) @scope.function
|
||||
(function_signature_item) @scope.function
|
||||
(closure_expression) @scope.function
|
||||
(block) @scope.block
|
||||
(if_expression) @scope.block
|
||||
@@ -56,14 +55,6 @@ const RUST_SCOPE_QUERY = `
|
||||
(function_item
|
||||
name: (identifier) @declaration.name) @declaration.function
|
||||
|
||||
;; Declarations — trait method signature (required method, no body,
|
||||
;; e.g. fn foo(self) -> T; inside a trait body). Without this, an abstract
|
||||
;; trait method is invisible to scope resolution — never owned by its
|
||||
;; trait's Class scope, so a dyn Trait receiver can never dispatch to
|
||||
;; it (#2604).
|
||||
(function_signature_item
|
||||
name: (identifier) @declaration.name) @declaration.function
|
||||
|
||||
;; Declarations — struct fields
|
||||
(field_declaration
|
||||
name: (field_identifier) @declaration.name
|
||||
|
||||
@@ -20,7 +20,6 @@ export {
|
||||
scopeResolutionPhase,
|
||||
type ScopeResolutionOutput,
|
||||
} from '../scope-resolution/pipeline/phase.js';
|
||||
export { springConfigPhase, type SpringConfigOutput } from './spring-config.js';
|
||||
export { pruneLocalSymbolsPhase, type PruneLocalSymbolsOutput } from './prune-local-symbols.js';
|
||||
export { taintSummariesPhase, type TaintSummariesOutput } from './taint-summaries.js';
|
||||
export { callSummariesPhase, type CallSummariesOutput } from './call-summaries.js';
|
||||
|
||||
@@ -25,17 +25,6 @@ export interface ProcessesOutput {
|
||||
processResult: ProcessDetectionResult;
|
||||
}
|
||||
|
||||
/**
|
||||
* Compute the dynamic max-processes budget from the symbol count.
|
||||
*
|
||||
* Scales proportionally (symbolCount / 10) with a floor of 20.
|
||||
* Prior to #2198 this was capped at 300 via `Math.min(300, …)`,
|
||||
* silently truncating process detection on large repositories.
|
||||
*/
|
||||
export function computeDynamicMaxProcesses(symbolCount: number): number {
|
||||
return Math.max(20, Math.round(symbolCount / 10));
|
||||
}
|
||||
|
||||
export const processesPhase: PipelinePhase<ProcessesOutput> = {
|
||||
name: 'processes',
|
||||
// `structure` supplies `totalFiles` (progress counter) without the spurious
|
||||
@@ -64,7 +53,7 @@ export const processesPhase: PipelinePhase<ProcessesOutput> = {
|
||||
ctx.graph.forEachNode((n) => {
|
||||
if (n.label !== 'File') symbolCount++;
|
||||
});
|
||||
const dynamicMaxProcesses = computeDynamicMaxProcesses(symbolCount);
|
||||
const dynamicMaxProcesses = Math.max(20, Math.min(300, Math.round(symbolCount / 10)));
|
||||
|
||||
const processResult = await processProcesses(
|
||||
ctx.graph,
|
||||
|
||||
@@ -1,551 +0,0 @@
|
||||
/**
|
||||
* Phase: springConfig
|
||||
*
|
||||
* Adds key-only nodes for statically readable Spring
|
||||
* `application*.properties` / `application*.yml` / `application*.yaml` files.
|
||||
* Language-specific ScopeResolver hooks attach consumers later. Configuration
|
||||
* values are deliberately never copied into the graph because they may contain
|
||||
* credentials and key identity is sufficient for impact analysis.
|
||||
*
|
||||
* @deps structure
|
||||
* @reads Spring application configuration files
|
||||
* @writes Property nodes and DEFINES edges
|
||||
*/
|
||||
|
||||
import fs from 'node:fs/promises';
|
||||
import path from 'node:path';
|
||||
import { createRequire } from 'node:module';
|
||||
import type { Event as YamlEvent } from 'js-yaml';
|
||||
import { SPRING_CONFIG_DESCRIPTION } from '../frameworks/spring/config-bindings.js';
|
||||
import { generateId } from '../../../lib/utils.js';
|
||||
import type { PipelineContext, PipelinePhase, PhaseResult } from './types.js';
|
||||
import { getPhaseOutput } from './types.js';
|
||||
import type { StructureOutput } from './structure.js';
|
||||
|
||||
const require = createRequire(import.meta.url);
|
||||
const yaml = require('js-yaml') as typeof import('js-yaml');
|
||||
// js-yaml 5 dropped DEFAULT_SCHEMA; CORE plus these tags is what it used to be, so
|
||||
// explicitly tagged values keep parsing instead of throwing (an unknown tag aborts
|
||||
// the whole file). None of them can execute code.
|
||||
const SPRING_YAML_SCHEMA = yaml.CORE_SCHEMA.withTags(
|
||||
yaml.mergeTag,
|
||||
yaml.timestampTag,
|
||||
yaml.binaryTag,
|
||||
yaml.omapTag,
|
||||
yaml.pairsTag,
|
||||
yaml.setTag,
|
||||
);
|
||||
const MAX_CONFIG_FILE_BYTES = 2 * 1024 * 1024;
|
||||
const MAX_YAML_TRAVERSAL_DEPTH = 128;
|
||||
const MAX_YAML_TRAVERSAL_NODES = 100_000;
|
||||
|
||||
export interface SpringConfigKey {
|
||||
readonly key: string;
|
||||
readonly filePath: string;
|
||||
readonly line: number;
|
||||
readonly profile?: string;
|
||||
readonly format: 'properties' | 'yaml';
|
||||
}
|
||||
|
||||
interface SpringConfigFile {
|
||||
readonly filePath: string;
|
||||
readonly profile?: string;
|
||||
readonly format: SpringConfigKey['format'];
|
||||
}
|
||||
|
||||
export interface SpringConfigOutput {
|
||||
readonly configKeys: number;
|
||||
}
|
||||
|
||||
/** Match only Spring Boot's conventional application config file names. */
|
||||
export function classifySpringConfigFile(filePath: string): SpringConfigFile | null {
|
||||
const base = path.posix.basename(filePath.replaceAll('\\', '/'));
|
||||
const match = /^application(?:-([^.]+))?\.(properties|ya?ml)$/i.exec(base);
|
||||
if (match === null) return null;
|
||||
return {
|
||||
filePath,
|
||||
...(match[1] ? { profile: match[1] } : {}),
|
||||
format: match[2].toLowerCase() === 'properties' ? 'properties' : 'yaml',
|
||||
};
|
||||
}
|
||||
|
||||
function unescapePropertyKey(raw: string): string {
|
||||
return raw
|
||||
.replace(/\\u([0-9a-fA-F]{4})/g, (_match, hex: string) =>
|
||||
String.fromCharCode(Number.parseInt(hex, 16)),
|
||||
)
|
||||
.replace(/\\([:=#!\\ ])/g, '$1');
|
||||
}
|
||||
|
||||
function logicalPropertiesLines(content: string): Array<{ text: string; line: number }> {
|
||||
const physical = content.split(/\r?\n/);
|
||||
const logical: Array<{ text: string; line: number }> = [];
|
||||
let current = '';
|
||||
let startLine = 1;
|
||||
|
||||
for (let index = 0; index < physical.length; index++) {
|
||||
const line = physical[index];
|
||||
if (current.length === 0) startLine = index + 1;
|
||||
current += current.length === 0 ? line : line.trimStart();
|
||||
|
||||
let trailingBackslashes = 0;
|
||||
for (let cursor = current.length - 1; cursor >= 0 && current[cursor] === '\\'; cursor--) {
|
||||
trailingBackslashes++;
|
||||
}
|
||||
if (trailingBackslashes % 2 === 1) {
|
||||
current = current.slice(0, -1);
|
||||
continue;
|
||||
}
|
||||
logical.push({ text: current, line: startLine });
|
||||
current = '';
|
||||
}
|
||||
if (current.length > 0) logical.push({ text: current, line: startLine });
|
||||
return logical;
|
||||
}
|
||||
|
||||
/** Parse `.properties` keys without retaining their values. */
|
||||
export function parseSpringProperties(
|
||||
content: string,
|
||||
filePath: string,
|
||||
profile?: string,
|
||||
): SpringConfigKey[] {
|
||||
const keys: SpringConfigKey[] = [];
|
||||
const seen = new Set<string>();
|
||||
|
||||
for (const logical of logicalPropertiesLines(content)) {
|
||||
const trimmed = logical.text.trimStart();
|
||||
if (trimmed.length === 0 || trimmed.startsWith('#') || trimmed.startsWith('!')) continue;
|
||||
|
||||
let separator = -1;
|
||||
let escaped = false;
|
||||
for (let index = 0; index < trimmed.length; index++) {
|
||||
const char = trimmed[index];
|
||||
if (!escaped && (char === '=' || char === ':' || /\s/.test(char))) {
|
||||
separator = index;
|
||||
break;
|
||||
}
|
||||
escaped = !escaped && char === '\\';
|
||||
if (char !== '\\') escaped = false;
|
||||
}
|
||||
const rawKey = (separator === -1 ? trimmed : trimmed.slice(0, separator)).trim();
|
||||
const key = unescapePropertyKey(rawKey);
|
||||
if (key.length === 0 || seen.has(key)) continue;
|
||||
seen.add(key);
|
||||
keys.push({
|
||||
key,
|
||||
filePath,
|
||||
line: logical.line,
|
||||
...(profile ? { profile } : {}),
|
||||
format: 'properties',
|
||||
});
|
||||
}
|
||||
|
||||
return keys;
|
||||
}
|
||||
|
||||
interface YamlParseEvent {
|
||||
readonly startLine: number;
|
||||
readonly kind: 'scalar' | 'sequence' | 'mapping' | 'alias' | null;
|
||||
readonly result: unknown;
|
||||
readonly aliasOf: YamlParseEvent | undefined;
|
||||
readonly children: YamlParseEvent[];
|
||||
}
|
||||
|
||||
interface YamlMappingLocation {
|
||||
readonly valueEvent: YamlParseEvent;
|
||||
readonly line: number;
|
||||
}
|
||||
|
||||
interface YamlTraversalState {
|
||||
remainingNodes: number;
|
||||
readonly activeObjects: Set<object>;
|
||||
}
|
||||
|
||||
function consumeYamlTraversalBudget(state: YamlTraversalState, depth: number): void {
|
||||
if (depth > MAX_YAML_TRAVERSAL_DEPTH) {
|
||||
throw new Error(`Spring YAML traversal depth exceeds ${MAX_YAML_TRAVERSAL_DEPTH}`);
|
||||
}
|
||||
state.remainingNodes--;
|
||||
if (state.remainingNodes < 0) {
|
||||
throw new Error(`Spring YAML traversal exceeds ${MAX_YAML_TRAVERSAL_NODES} nodes`);
|
||||
}
|
||||
}
|
||||
|
||||
function isObjectValue(value: unknown): value is object {
|
||||
return value !== null && typeof value === 'object';
|
||||
}
|
||||
|
||||
// Aliases are resolved to their anchor event by name while the tree is built,
|
||||
// so following one here is a single pointer hop.
|
||||
function resolveYamlAliasEvent(event: YamlParseEvent | undefined): YamlParseEvent | undefined {
|
||||
return event?.aliasOf ?? event;
|
||||
}
|
||||
|
||||
function yamlMappingPairs(event: YamlParseEvent): Array<{
|
||||
key: string;
|
||||
keyEvent: YamlParseEvent;
|
||||
valueEvent: YamlParseEvent;
|
||||
}> {
|
||||
const pairs: Array<{ key: string; keyEvent: YamlParseEvent; valueEvent: YamlParseEvent }> = [];
|
||||
for (let index = 0; index + 1 < event.children.length; index += 2) {
|
||||
const keyEvent = event.children[index];
|
||||
const valueEvent = event.children[index + 1];
|
||||
if (keyEvent.kind !== 'scalar') continue;
|
||||
pairs.push({ key: String(keyEvent.result), keyEvent, valueEvent });
|
||||
}
|
||||
return pairs;
|
||||
}
|
||||
|
||||
/**
|
||||
* First match in a pre-order walk of `event`, following sequences and `<<` merge
|
||||
* chains. Iterative: children are pushed in reverse so the explicit stack pops
|
||||
* them in declaration order, which is what makes "first match" mean the same
|
||||
* thing it did when this recursed.
|
||||
*/
|
||||
function findYamlMappingLocation(
|
||||
event: YamlParseEvent | undefined,
|
||||
key: string,
|
||||
traversal: YamlTraversalState,
|
||||
): YamlMappingLocation | undefined {
|
||||
const visited = new Set<YamlParseEvent>();
|
||||
const stack: Array<{ event: YamlParseEvent | undefined; depth: number }> = [{ event, depth: 0 }];
|
||||
|
||||
while (stack.length > 0) {
|
||||
const step = stack.pop();
|
||||
if (step === undefined) break;
|
||||
consumeYamlTraversalBudget(traversal, step.depth);
|
||||
const resolved = resolveYamlAliasEvent(step.event);
|
||||
if (resolved === undefined || visited.has(resolved)) continue;
|
||||
visited.add(resolved);
|
||||
|
||||
if (resolved.kind === 'sequence') {
|
||||
for (let index = resolved.children.length - 1; index >= 0; index--) {
|
||||
stack.push({ event: resolved.children[index], depth: step.depth + 1 });
|
||||
}
|
||||
continue;
|
||||
}
|
||||
if (resolved.kind !== 'mapping') continue;
|
||||
|
||||
const pairs = yamlMappingPairs(resolved);
|
||||
const direct = pairs.find((pair) => pair.key === key);
|
||||
if (direct !== undefined) {
|
||||
return { valueEvent: direct.valueEvent, line: direct.keyEvent.startLine };
|
||||
}
|
||||
const merges = pairs.filter((pair) => pair.key === '<<');
|
||||
for (let index = merges.length - 1; index >= 0; index--) {
|
||||
stack.push({ event: merges[index].valueEvent, depth: step.depth + 1 });
|
||||
}
|
||||
}
|
||||
return undefined;
|
||||
}
|
||||
|
||||
type YamlFlattenStep =
|
||||
| {
|
||||
readonly kind: 'visit';
|
||||
readonly value: unknown;
|
||||
readonly event: YamlParseEvent | undefined;
|
||||
readonly prefix: string;
|
||||
readonly sourceLine: number;
|
||||
readonly depth: number;
|
||||
}
|
||||
// Pops after every descendant of the object that pushed it, which is where the
|
||||
// recursive form's `finally` used to release the cycle guard.
|
||||
| { readonly kind: 'leave'; readonly object: object };
|
||||
|
||||
/**
|
||||
* Flatten a document to `dotted.key -> line`, iteratively. Children are pushed in
|
||||
* reverse so the stack pops them in declaration order, keeping `out` in the same
|
||||
* insertion order — and the traversal budget consumed in the same sequence — as
|
||||
* the recursive walk this replaced.
|
||||
*/
|
||||
function flattenYamlValue(
|
||||
value: unknown,
|
||||
event: YamlParseEvent | undefined,
|
||||
prefix: string,
|
||||
out: Map<string, number>,
|
||||
traversal: YamlTraversalState,
|
||||
): void {
|
||||
const stack: YamlFlattenStep[] = [
|
||||
{ kind: 'visit', value, event, prefix, sourceLine: event?.startLine ?? 1, depth: 0 },
|
||||
];
|
||||
|
||||
while (stack.length > 0) {
|
||||
const step = stack.pop();
|
||||
if (step === undefined) break;
|
||||
if (step.kind === 'leave') {
|
||||
traversal.activeObjects.delete(step.object);
|
||||
continue;
|
||||
}
|
||||
|
||||
const { value: current, prefix: currentPrefix, sourceLine, depth } = step;
|
||||
consumeYamlTraversalBudget(traversal, depth);
|
||||
const resolvedEvent = resolveYamlAliasEvent(step.event);
|
||||
const trackedObject = isObjectValue(current) ? current : undefined;
|
||||
if (trackedObject !== undefined) {
|
||||
if (traversal.activeObjects.has(trackedObject)) continue;
|
||||
traversal.activeObjects.add(trackedObject);
|
||||
stack.push({ kind: 'leave', object: trackedObject });
|
||||
}
|
||||
|
||||
if (Array.isArray(current)) {
|
||||
if (current.length === 0 && currentPrefix.length > 0 && !out.has(currentPrefix)) {
|
||||
out.set(currentPrefix, sourceLine);
|
||||
}
|
||||
for (let index = current.length - 1; index >= 0; index--) {
|
||||
stack.push({
|
||||
kind: 'visit',
|
||||
value: current[index],
|
||||
event: resolvedEvent?.children[index],
|
||||
prefix: `${currentPrefix}[${index}]`,
|
||||
sourceLine,
|
||||
depth: depth + 1,
|
||||
});
|
||||
}
|
||||
continue;
|
||||
}
|
||||
|
||||
if (
|
||||
current !== null &&
|
||||
typeof current === 'object' &&
|
||||
(resolvedEvent?.kind === 'mapping' || resolvedEvent === undefined)
|
||||
) {
|
||||
// js-yaml 5 builds `!!set` as a native Set, whose members are not own
|
||||
// properties; v4 built a plain `{member: null}` object. Enumerate them so a
|
||||
// tagged set still contributes one key per member instead of a bare leaf.
|
||||
const entries: Array<[string, unknown]> =
|
||||
current instanceof Set
|
||||
? [...current].map((member) => [String(member), null])
|
||||
: Object.entries(current as Record<string, unknown>);
|
||||
if (entries.length === 0 && currentPrefix.length > 0 && !out.has(currentPrefix)) {
|
||||
out.set(currentPrefix, sourceLine);
|
||||
}
|
||||
for (let index = entries.length - 1; index >= 0; index--) {
|
||||
const [key, nested] = entries[index];
|
||||
const location = findYamlMappingLocation(resolvedEvent, key, traversal);
|
||||
stack.push({
|
||||
kind: 'visit',
|
||||
value: nested,
|
||||
event: location?.valueEvent,
|
||||
prefix: currentPrefix.length === 0 ? key : `${currentPrefix}.${key}`,
|
||||
sourceLine: location?.line ?? sourceLine,
|
||||
depth: depth + 1,
|
||||
});
|
||||
}
|
||||
continue;
|
||||
}
|
||||
|
||||
if (currentPrefix.length > 0 && !out.has(currentPrefix)) out.set(currentPrefix, sourceLine);
|
||||
}
|
||||
}
|
||||
|
||||
// js-yaml 5 reports node positions as source offsets; map them to 1-based lines.
|
||||
function makeLineResolver(source: string): (offset: number) => number {
|
||||
const lineStarts = [0];
|
||||
for (let index = 0; index < source.length; index++) {
|
||||
if (source[index] === '\n') lineStarts.push(index + 1);
|
||||
}
|
||||
return (offset: number): number => {
|
||||
let low = 0;
|
||||
let high = lineStarts.length - 1;
|
||||
let line = 0;
|
||||
while (low <= high) {
|
||||
const mid = (low + high) >> 1;
|
||||
if (lineStarts[mid] <= offset) {
|
||||
line = mid;
|
||||
low = mid + 1;
|
||||
} else {
|
||||
high = mid - 1;
|
||||
}
|
||||
}
|
||||
return line + 1;
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Rebuild the parse tree from js-yaml 5's event stream (v4's `listener` option
|
||||
* was removed). Returns each document's root event, with aliases already
|
||||
* resolved to their anchor event so merged/aliased keys keep the line where
|
||||
* they were declared.
|
||||
*
|
||||
* One node per event, so this pass is bounded by MAX_CONFIG_FILE_BYTES alone —
|
||||
* MAX_YAML_TRAVERSAL_NODES governs the later walk, which can revisit a shared
|
||||
* anchor many times and so needs a budget this linear pass does not.
|
||||
*/
|
||||
function buildYamlEventTree(
|
||||
events: readonly YamlEvent[],
|
||||
source: string,
|
||||
): Array<YamlParseEvent | undefined> {
|
||||
const lineOf = makeLineResolver(source);
|
||||
const anchors = new Map<string, YamlParseEvent>();
|
||||
const stack: YamlParseEvent[] = [];
|
||||
const documentRoots: Array<YamlParseEvent | undefined> = [];
|
||||
|
||||
const anchorName = (start: number, end: number): string | null =>
|
||||
start >= 0 && end > start ? source.slice(start, end) : null;
|
||||
const attach = (node: YamlParseEvent): void => {
|
||||
stack[stack.length - 1]?.children.push(node);
|
||||
};
|
||||
const register = (name: string | null, node: YamlParseEvent): void => {
|
||||
if (name !== null) anchors.set(name, node);
|
||||
};
|
||||
|
||||
for (const event of events) {
|
||||
switch (event.type) {
|
||||
case yaml.EVENT_DOCUMENT:
|
||||
// Anchors are document-scoped. constructFromEvents already rejects a
|
||||
// cross-document alias before we get here, so this only keeps the two
|
||||
// layers from disagreeing.
|
||||
anchors.clear();
|
||||
stack.push({
|
||||
startLine: 1,
|
||||
kind: null,
|
||||
result: undefined,
|
||||
aliasOf: undefined,
|
||||
children: [],
|
||||
});
|
||||
break;
|
||||
case yaml.EVENT_MAPPING:
|
||||
case yaml.EVENT_SEQUENCE: {
|
||||
const node: YamlParseEvent = {
|
||||
startLine: lineOf(event.start),
|
||||
kind: event.type === yaml.EVENT_MAPPING ? 'mapping' : 'sequence',
|
||||
result: undefined,
|
||||
aliasOf: undefined,
|
||||
children: [],
|
||||
};
|
||||
register(anchorName(event.anchorStart, event.anchorEnd), node);
|
||||
attach(node);
|
||||
stack.push(node);
|
||||
break;
|
||||
}
|
||||
case yaml.EVENT_SCALAR: {
|
||||
const node: YamlParseEvent = {
|
||||
startLine: lineOf(event.valueStart),
|
||||
kind: 'scalar',
|
||||
result: yaml.getScalarValue(source, event),
|
||||
aliasOf: undefined,
|
||||
children: [],
|
||||
};
|
||||
register(anchorName(event.anchorStart, event.anchorEnd), node);
|
||||
attach(node);
|
||||
break;
|
||||
}
|
||||
case yaml.EVENT_ALIAS: {
|
||||
const target = anchors.get(anchorName(event.anchorStart, event.anchorEnd) ?? '');
|
||||
attach({ startLine: 1, kind: 'alias', result: undefined, aliasOf: target, children: [] });
|
||||
break;
|
||||
}
|
||||
case yaml.EVENT_POP: {
|
||||
const done = stack.pop();
|
||||
// Documents are the only top-level containers, so a pop that empties the
|
||||
// stack closes a document; its single child is the document's root value.
|
||||
if (done !== undefined && stack.length === 0) documentRoots.push(done.children[0]);
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
return documentRoots;
|
||||
}
|
||||
|
||||
/** Parse and flatten YAML leaves without retaining their values. */
|
||||
export function parseSpringYaml(
|
||||
content: string,
|
||||
filePath: string,
|
||||
profile?: string,
|
||||
): SpringConfigKey[] {
|
||||
const flattened = new Map<string, number>();
|
||||
const traversal: YamlTraversalState = {
|
||||
remainingNodes: MAX_YAML_TRAVERSAL_NODES,
|
||||
activeObjects: new Set<object>(),
|
||||
};
|
||||
|
||||
const events = yaml.parseEvents(content, { maxDepth: MAX_YAML_TRAVERSAL_DEPTH });
|
||||
const documents = yaml.constructFromEvents(events, {
|
||||
source: content,
|
||||
schema: SPRING_YAML_SCHEMA,
|
||||
json: true,
|
||||
});
|
||||
const documentEvents = buildYamlEventTree(events, content);
|
||||
|
||||
documents.forEach((document, index) =>
|
||||
flattenYamlValue(document, documentEvents[index], '', flattened, traversal),
|
||||
);
|
||||
return [...flattened.entries()]
|
||||
.sort(([left], [right]) => left.localeCompare(right))
|
||||
.map(([key, line]) => ({
|
||||
key,
|
||||
filePath,
|
||||
line,
|
||||
...(profile ? { profile } : {}),
|
||||
format: 'yaml' as const,
|
||||
}));
|
||||
}
|
||||
|
||||
function configKeyNodeId(entry: SpringConfigKey): string {
|
||||
return generateId('Property', `spring-config:${entry.filePath}:${entry.key}`);
|
||||
}
|
||||
|
||||
async function readConfigKeys(
|
||||
repoPath: string,
|
||||
scannedFiles: StructureOutput['scannedFiles'],
|
||||
): Promise<SpringConfigKey[]> {
|
||||
const keys: SpringConfigKey[] = [];
|
||||
for (const scanned of scannedFiles) {
|
||||
const classified = classifySpringConfigFile(scanned.path);
|
||||
if (classified === null || scanned.size > MAX_CONFIG_FILE_BYTES) continue;
|
||||
try {
|
||||
const content = await fs.readFile(path.join(repoPath, scanned.path), 'utf8');
|
||||
keys.push(
|
||||
...(classified.format === 'properties'
|
||||
? parseSpringProperties(content, classified.filePath, classified.profile)
|
||||
: parseSpringYaml(content, classified.filePath, classified.profile)),
|
||||
);
|
||||
} catch {
|
||||
// Malformed configuration is not a reason to fail the entire code index.
|
||||
// Fail closed: no keys and therefore no misleading bindings for this file.
|
||||
}
|
||||
}
|
||||
return keys;
|
||||
}
|
||||
|
||||
export const springConfigPhase: PipelinePhase<SpringConfigOutput> = {
|
||||
name: 'springConfig',
|
||||
deps: ['structure'],
|
||||
|
||||
async execute(
|
||||
ctx: PipelineContext,
|
||||
deps: ReadonlyMap<string, PhaseResult<unknown>>,
|
||||
): Promise<SpringConfigOutput> {
|
||||
const { scannedFiles } = getPhaseOutput<StructureOutput>(deps, 'structure');
|
||||
const configKeys = await readConfigKeys(ctx.repoPath, scannedFiles);
|
||||
for (const entry of configKeys) {
|
||||
const nodeId = configKeyNodeId(entry);
|
||||
ctx.graph.addNode({
|
||||
id: nodeId,
|
||||
label: 'Property',
|
||||
properties: {
|
||||
name: entry.key,
|
||||
filePath: entry.filePath,
|
||||
startLine: entry.line,
|
||||
endLine: entry.line,
|
||||
description: entry.profile
|
||||
? `${SPRING_CONFIG_DESCRIPTION} (profile: ${entry.profile})`
|
||||
: SPRING_CONFIG_DESCRIPTION,
|
||||
},
|
||||
});
|
||||
const fileId = generateId('File', entry.filePath);
|
||||
if (ctx.graph.getNode(fileId) !== undefined) {
|
||||
ctx.graph.addRelationship({
|
||||
id: generateId('DEFINES', `${fileId}->${nodeId}`),
|
||||
sourceId: fileId,
|
||||
targetId: nodeId,
|
||||
type: 'DEFINES',
|
||||
confidence: 1,
|
||||
reason: 'spring-config:key',
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
return { configKeys: configKeys.length };
|
||||
},
|
||||
};
|
||||
@@ -31,7 +31,6 @@ import {
|
||||
ormPhase,
|
||||
crossFilePhase,
|
||||
scopeResolutionPhase,
|
||||
springConfigPhase,
|
||||
pruneLocalSymbolsPhase,
|
||||
taintSummariesPhase,
|
||||
callSummariesPhase,
|
||||
@@ -243,7 +242,7 @@ export interface PipelineOptions {
|
||||
*
|
||||
* Phase dependency graph:
|
||||
*
|
||||
* scan → structure → [springConfig, markdown, cobol] → parse → [routes, tools, orm]
|
||||
* scan → structure → [markdown, cobol] → parse → [routes, tools, orm]
|
||||
* → crossFile → scopeResolution → pruneLocalSymbols
|
||||
* → mro → di → communities → processes
|
||||
*
|
||||
@@ -262,7 +261,6 @@ export function buildPhaseList(options?: PipelineOptions): PipelinePhase[] {
|
||||
new PhaseRegistry<PipelineOptions>()
|
||||
.register(scanPhase)
|
||||
.register(structurePhase)
|
||||
.register(springConfigPhase)
|
||||
.register(markdownPhase)
|
||||
.register(cobolPhase)
|
||||
.register(parsePhase)
|
||||
|
||||
@@ -982,15 +982,13 @@ function followChainedRef(start: TypeRef, draftById: ReadonlyMap<ScopeId, ScopeD
|
||||
* name in the same scope. Higher number wins; ties keep the later match
|
||||
* (last-write-wins preserves historical order within a tier).
|
||||
*
|
||||
* Rationale: explicit variable and field annotations always beat bindings
|
||||
* derived from parameter annotations or inference because they reflect the
|
||||
* most specific user intent. `self`/`cls` are treated as strongly as other
|
||||
* declared types because they are language-required receiver types.
|
||||
* Rationale: explicit annotations always beat inferred ones because they
|
||||
* reflect user intent. `self`/`cls` are treated as strongly as annotations
|
||||
* because they are language-required receiver types.
|
||||
*/
|
||||
function typeBindingStrength(source: TypeRef['source']): number {
|
||||
switch (source) {
|
||||
case 'annotation':
|
||||
return 3;
|
||||
case 'parameter-annotation':
|
||||
case 'return-annotation':
|
||||
case 'self':
|
||||
|
||||
@@ -743,11 +743,10 @@ export const PYTHON_QUERIES = `
|
||||
|
||||
// Java queries - works with tree-sitter-java
|
||||
export const JAVA_QUERIES = `
|
||||
; Classes, Interfaces, Enums, Records, Annotations
|
||||
; Classes, Interfaces, Enums, Annotations
|
||||
(class_declaration name: (identifier) @name) @definition.class
|
||||
(interface_declaration name: (identifier) @name) @definition.interface
|
||||
(enum_declaration name: (identifier) @name) @definition.enum
|
||||
(record_declaration name: (identifier) @name) @definition.record
|
||||
(annotation_type_declaration name: (identifier) @name) @definition.annotation
|
||||
|
||||
; Anonymous class bodies: new Runnable() { ... } — no @name capture; the
|
||||
|
||||
@@ -205,19 +205,6 @@ export interface WorkerPoolOptions {
|
||||
* created. Default `Math.max(3, poolSize)`.
|
||||
*/
|
||||
consecutiveFailureThreshold?: number;
|
||||
/**
|
||||
* Startup budget in milliseconds for a replacement worker to emit the
|
||||
* `{type:'ready'}` handshake before the pool treats it as a startup
|
||||
* crash (see {@link waitForWorkerReady}). Default 5000; also overridable
|
||||
* via `GITNEXUS_WORKER_READY_TIMEOUT_MS`, mirroring
|
||||
* `GITNEXUS_WORKER_SUB_BATCH_TIMEOUT_MS`. On a slow or heavily loaded
|
||||
* host, a full pool of workers cold-starting concurrently can
|
||||
* legitimately need more than 5s to load the native grammar bindings —
|
||||
* without the override every slot times out and the pool misclassifies
|
||||
* the slow start as a deterministic startup crash-loop, aborting the
|
||||
* whole analyze.
|
||||
*/
|
||||
workerReadyTimeoutMs?: number;
|
||||
/**
|
||||
* Test-only injection point for the Worker constructor. When provided,
|
||||
* the pool uses this factory instead of `new Worker(workerUrl)`. Production
|
||||
@@ -419,7 +406,17 @@ const DEFAULT_TIMEOUT_BACKOFF_FACTOR = 2;
|
||||
const DEFAULT_MAX_RESPAWNS_PER_SLOT = 3;
|
||||
const DEFAULT_MAX_CUMULATIVE_TIMEOUT_FACTOR = 5;
|
||||
const DEFAULT_CONSECUTIVE_FAILURE_THRESHOLD_FLOOR = 3;
|
||||
const DEFAULT_WORKER_READY_TIMEOUT_MS = 5_000;
|
||||
/**
|
||||
* Bounded wait for a replacement worker to emit the `{type:'ready'}`
|
||||
* handshake from `parse-worker.ts`. Trusting Node's `online` event alone
|
||||
* lets a worker that crashes during top-of-script init slip past pool
|
||||
* startup — the pool only notices on the first dispatch's idle timeout
|
||||
* (default 30s). 5 seconds is a generous budget for parser + grammar
|
||||
* imports; if the worker hasn't reported ready by then, it's almost
|
||||
* certainly stuck or crashed and the pool should surface the failure
|
||||
* fast rather than wait out the dispatch idle timeout.
|
||||
*/
|
||||
const WORKER_READY_TIMEOUT_MS = 5_000;
|
||||
/**
|
||||
* Default upper bound on auto-resolved pool size. Past 16 workers the
|
||||
* dominant cost shifts from worker-side parsing to main-thread merge /
|
||||
@@ -550,7 +547,6 @@ interface ResolvedWorkerPoolOptions {
|
||||
maxCumulativeTimeoutMs: number;
|
||||
consecutiveFailureThreshold: number;
|
||||
shutdownDrainMs: number;
|
||||
workerReadyTimeoutMs: number;
|
||||
}
|
||||
|
||||
export function resolveWorkerPoolOptions(
|
||||
@@ -587,10 +583,6 @@ export function resolveWorkerPoolOptions(
|
||||
nonNegativeInteger(options.shutdownDrainMs) ??
|
||||
nonNegativeInteger(process.env.GITNEXUS_WORKER_SHUTDOWN_DRAIN_MS) ??
|
||||
DEFAULT_SHUTDOWN_DRAIN_MS,
|
||||
workerReadyTimeoutMs:
|
||||
positiveInteger(options.workerReadyTimeoutMs) ??
|
||||
positiveInteger(process.env.GITNEXUS_WORKER_READY_TIMEOUT_MS) ??
|
||||
DEFAULT_WORKER_READY_TIMEOUT_MS,
|
||||
};
|
||||
}
|
||||
|
||||
@@ -691,27 +683,6 @@ function captureWorkerStderr(worker: Worker): void {
|
||||
stream.on('error', () => undefined);
|
||||
}
|
||||
|
||||
/**
|
||||
* Forward a worker's piped stdout to the parent process's stdout, so worker
|
||||
* logs stay visible now that the production factory spawns with
|
||||
* `{ stdout: true }`. Workers with INHERITED stdout have been observed to
|
||||
* crash silently during top-of-script init (exit code 1, nothing on stderr,
|
||||
* roughly half of a concurrently spawned pool) on macOS 26.5 under both
|
||||
* Node 22 and 26; piping stdout eliminates the crash entirely. Piping also
|
||||
* matches the existing stderr handling, so worker output no longer races the
|
||||
* parent's raw fd. No-op when the worker has no `stdout` stream (test
|
||||
* factories).
|
||||
*/
|
||||
function forwardWorkerStdout(worker: Worker): void {
|
||||
const stream = worker.stdout;
|
||||
if (!stream) return;
|
||||
stream.on('data', (chunk: Buffer | string) => {
|
||||
process.stdout.write(chunk);
|
||||
});
|
||||
// A stdout stream error must never crash the pool.
|
||||
stream.on('error', () => undefined);
|
||||
}
|
||||
|
||||
/** Captured stderr tail for a worker, trimmed; '' when nothing was captured. */
|
||||
function workerStderrTail(worker: Worker): string {
|
||||
return workerStderrTails.get(worker)?.text.trim() ?? '';
|
||||
@@ -751,14 +722,13 @@ function workerErrorReason(workerIndex: number, message: string, stack?: string)
|
||||
* (parser/grammar import failure, missing native binding) slip past
|
||||
* pool startup. The pool then only noticed the dead replacement on the
|
||||
* first dispatch's idle timeout (default 30s) — a long stall masking
|
||||
* an actual crash. This handshake bounds the wait at `readyTimeoutMs`
|
||||
* (see {@link WorkerPoolOptions.workerReadyTimeoutMs}) and surfaces init
|
||||
* failures as `error` / `exit` / `messageerror` events directly.
|
||||
* `messageerror` is wired the same way: a V8 deserialization failure
|
||||
* during startup is treated as worker death and rejects the readiness
|
||||
* promise.
|
||||
* an actual crash. This handshake bounds the wait at
|
||||
* {@link WORKER_READY_TIMEOUT_MS} and surfaces init failures as
|
||||
* `error` / `exit` / `messageerror` events directly. `messageerror` is
|
||||
* wired the same way: a V8 deserialization failure during startup is
|
||||
* treated as worker death and rejects the readiness promise.
|
||||
*/
|
||||
function waitForWorkerReady(worker: Worker, readyTimeoutMs: number): Promise<void> {
|
||||
function waitForWorkerReady(worker: Worker): Promise<void> {
|
||||
return new Promise<void>((resolve, reject) => {
|
||||
const cleanup = () => {
|
||||
clearTimeout(timer);
|
||||
@@ -811,11 +781,11 @@ function waitForWorkerReady(worker: Worker, readyTimeoutMs: number): Promise<voi
|
||||
new Error(
|
||||
withStderr(
|
||||
worker,
|
||||
`Replacement worker did not report ready within ${readyTimeoutMs}ms — likely crashed during top-of-script init (slow host? raise GITNEXUS_WORKER_READY_TIMEOUT_MS)`,
|
||||
`Replacement worker did not report ready within ${WORKER_READY_TIMEOUT_MS}ms — likely crashed during top-of-script init`,
|
||||
),
|
||||
),
|
||||
);
|
||||
}, readyTimeoutMs);
|
||||
}, WORKER_READY_TIMEOUT_MS);
|
||||
worker.on('message', onMessage);
|
||||
worker.once('error', onError);
|
||||
worker.once('exit', onExit);
|
||||
@@ -961,10 +931,6 @@ export const createWorkerPool = (
|
||||
options?.workerFactory ??
|
||||
((url: URL) =>
|
||||
new Worker(url, {
|
||||
// Piped (not inherited) stdio: stderr for crash capture (#1741),
|
||||
// stdout because inherited stdout triggers silent startup crashes on
|
||||
// some hosts (see forwardWorkerStdout).
|
||||
stdout: true,
|
||||
stderr: true,
|
||||
workerData: workerStoreData,
|
||||
// The CFG visitors build per-function control-flow graphs by RECURSIVE
|
||||
@@ -978,11 +944,10 @@ export const createWorkerPool = (
|
||||
// try/catch) and only that function's PDG is skipped, never a crash.
|
||||
resourceLimits: { stackSizeMb: 16 },
|
||||
}));
|
||||
/** Spawn + wire stdio capture/forwarding in one step (used by all spawn sites). */
|
||||
/** Spawn + wire stderr capture in one step (used by all spawn sites). */
|
||||
const spawnAndCapture = (url: URL): Worker => {
|
||||
const worker = spawnWorker(url);
|
||||
captureWorkerStderr(worker);
|
||||
forwardWorkerStdout(worker);
|
||||
return worker;
|
||||
};
|
||||
const workers: (Worker | undefined)[] = new Array(size);
|
||||
@@ -1134,7 +1099,7 @@ export const createWorkerPool = (
|
||||
const worker = workers[i];
|
||||
if (!worker) return; // terminated mid-startup
|
||||
try {
|
||||
await waitForWorkerReady(worker, poolOptions.workerReadyTimeoutMs);
|
||||
await waitForWorkerReady(worker);
|
||||
anyWorkerReachedReady = true;
|
||||
return; // ready — slot stays in activeSlots
|
||||
} catch (err) {
|
||||
@@ -1196,7 +1161,7 @@ export const createWorkerPool = (
|
||||
chunkHash?: string,
|
||||
): Promise<TResult[]> => {
|
||||
// Await the initial-spawn readiness gate (F13). On first dispatch
|
||||
// this blocks for up to poolOptions.workerReadyTimeoutMs while every initial
|
||||
// this blocks for up to WORKER_READY_TIMEOUT_MS while every initial
|
||||
// worker's `{type:'ready'}` handshake is checked; on subsequent
|
||||
// dispatches the promise is already settled and resolves
|
||||
// synchronously. Slots whose initial worker crashed in top-of-
|
||||
@@ -1395,7 +1360,7 @@ export const createWorkerPool = (
|
||||
if (stopped) return false;
|
||||
const replacement = spawnAndCapture(workerUrl);
|
||||
try {
|
||||
await waitForWorkerReady(replacement, poolOptions.workerReadyTimeoutMs);
|
||||
await waitForWorkerReady(replacement);
|
||||
} catch (err) {
|
||||
await replacement.terminate().catch(() => undefined);
|
||||
logger.warn(
|
||||
|
||||
@@ -101,24 +101,18 @@ const POSIX_MISSING_DEPENDENCY_SIGNATURES: readonly RegExp[] = [
|
||||
* display language — the only localized part is the OS-error tail after it. So
|
||||
* it is the language-independent fallback signal once the specific tails miss: a
|
||||
* French/German/Japanese Windows 126 has a localized tail we cannot enumerate,
|
||||
* but it still carries this wrapper. See hedgedLoadFailureRemedy.
|
||||
* but it still carries this wrapper. See HEDGED_LOAD_FAILURE_REMEDY.
|
||||
*/
|
||||
const LOAD_FAILURE_WRAPPER = /failed to load library/i;
|
||||
|
||||
// Remedies are label-parameterized (#2623 follow-up): doctor now live-probes
|
||||
// VECTOR through the same classifier, and FTS-specific advice (`--repair-fts`
|
||||
// repairs FTS indexes only) must not be dispensed for other extensions.
|
||||
const repairFtsHint = (label: string, lead: string): string =>
|
||||
label === 'FTS' ? ` (${lead}\`gitnexus analyze --repair-fts\`)` : '';
|
||||
const MISSING_FILE_REMEDY =
|
||||
'The FTS extension is not installed. Re-run with network access and ' +
|
||||
'GITNEXUS_LBUG_EXTENSION_INSTALL=auto (or `gitnexus analyze --repair-fts`) to download it.';
|
||||
|
||||
const missingFileRemedy = (label: string): string =>
|
||||
`The ${label} extension is not installed. Re-run with network access and ` +
|
||||
`GITNEXUS_LBUG_EXTENSION_INSTALL=auto${repairFtsHint(label, 'or ')} to download it.`;
|
||||
|
||||
const corruptFileRemedy = (label: string): string =>
|
||||
`The ${label} extension file is present but unreadable (corrupt, truncated, or built for another ` +
|
||||
`platform). Re-download it with network access and ` +
|
||||
`GITNEXUS_LBUG_EXTENSION_INSTALL=auto${repairFtsHint(label, '')}.`;
|
||||
const CORRUPT_FILE_REMEDY =
|
||||
'The FTS extension file is present but unreadable (corrupt, truncated, or built for another ' +
|
||||
'platform). Re-download it with network access and GITNEXUS_LBUG_EXTENSION_INSTALL=auto ' +
|
||||
'(`gitnexus analyze --repair-fts`).';
|
||||
|
||||
// Single source of truth for the VC++ runtime-install pointer, shared by the
|
||||
// Windows-126 and structural missing-dependency remedies so the name/URL cannot
|
||||
@@ -128,15 +122,15 @@ const VC_REDIST_INSTALL_HINT =
|
||||
'https://aka.ms/vs/17/release/vc_redist.x64.exe';
|
||||
|
||||
// MSVC-first per DuckDB's canonical answer for this exact error; OpenSSL second.
|
||||
const windowsMissingDependencyRemedy = (label: string): string =>
|
||||
`The ${label} extension is present but a required runtime library is missing (Windows error 126). ` +
|
||||
const WINDOWS_MISSING_DEPENDENCY_REMEDY =
|
||||
'The FTS extension is present but a required runtime library is missing (Windows error 126). ' +
|
||||
'Reinstalling the extension will NOT help. Install ' +
|
||||
VC_REDIST_INSTALL_HINT +
|
||||
'; if the error persists, the extension also needs OpenSSL 3 ' +
|
||||
'(libcrypto-3-x64.dll / libssl-3-x64.dll) on the DLL search path.';
|
||||
|
||||
const posixMissingDependencyRemedy = (label: string): string =>
|
||||
`The ${label} extension is present but a shared library it depends on could not be loaded (named in ` +
|
||||
const POSIX_MISSING_DEPENDENCY_REMEDY =
|
||||
'The FTS extension is present but a shared library it depends on could not be loaded (named in ' +
|
||||
'the error above). Reinstalling the extension will NOT help — install that library or add it to ' +
|
||||
'your loader search path.';
|
||||
|
||||
@@ -146,18 +140,16 @@ const posixMissingDependencyRemedy = (label: string): string =>
|
||||
// branches — rather than confidently prescribing the wrong single fix. The clean
|
||||
// long-term fix is upstream: have LadybugDB include the numeric GetLastError/errno
|
||||
// in the message (as it already does elsewhere), so this becomes a code match.
|
||||
const hedgedLoadFailureRemedy = (label: string): string =>
|
||||
`The ${label} extension file was found but could not be loaded — see the "Error:" text above (shown ` +
|
||||
const HEDGED_LOAD_FAILURE_REMEDY =
|
||||
'The FTS extension file was found but could not be loaded — see the "Error:" text above (shown ' +
|
||||
"in your system's language). Reinstalling usually will not help. If it names a missing module or " +
|
||||
'library, install the required runtime (on Windows: the Microsoft Visual C++ 2015-2022 ' +
|
||||
'Redistributable x64 and OpenSSL 3); if it names a corrupt or invalid file, ' +
|
||||
(label === 'FTS'
|
||||
? 'run `gitnexus analyze --repair-fts` to re-download.'
|
||||
: 're-run analyze with network access and GITNEXUS_LBUG_EXTENSION_INSTALL=auto to re-download.');
|
||||
'Redistributable x64 and OpenSSL 3); if it names a corrupt or invalid file, run ' +
|
||||
'`gitnexus analyze --repair-fts` to re-download.';
|
||||
|
||||
const unknownRemedy = (label: string): string =>
|
||||
`The ${label} extension failed to load for an unrecognized reason. Run \`gitnexus doctor\` for live ` +
|
||||
`${label} status and verify the extension file and platform.`;
|
||||
const UNKNOWN_REMEDY =
|
||||
'The FTS extension failed to load for an unrecognized reason. Run `gitnexus doctor` for live ' +
|
||||
'FTS status and verify the extension file and platform.';
|
||||
|
||||
const matchesAny = (reason: string, signatures: readonly RegExp[]): boolean =>
|
||||
signatures.some((re) => re.test(reason));
|
||||
@@ -170,20 +162,19 @@ const matchesAny = (reason: string, signatures: readonly RegExp[]): boolean =>
|
||||
*/
|
||||
export function classifyExtensionLoadError(
|
||||
reason: string | undefined | null,
|
||||
label: string = 'FTS',
|
||||
): ExtensionLoadDiagnosis {
|
||||
const text = reason ?? '';
|
||||
if (matchesAny(text, MISSING_FILE_SIGNATURES)) {
|
||||
return { kind: 'missing_file', remedy: missingFileRemedy(label) };
|
||||
return { kind: 'missing_file', remedy: MISSING_FILE_REMEDY };
|
||||
}
|
||||
if (matchesAny(text, FILE_CORRUPTION_SIGNATURES)) {
|
||||
return { kind: 'corrupt_file', remedy: corruptFileRemedy(label) };
|
||||
return { kind: 'corrupt_file', remedy: CORRUPT_FILE_REMEDY };
|
||||
}
|
||||
if (matchesAny(text, WINDOWS_MISSING_DEPENDENCY_SIGNATURES)) {
|
||||
return { kind: 'missing_dependency', remedy: windowsMissingDependencyRemedy(label) };
|
||||
return { kind: 'missing_dependency', remedy: WINDOWS_MISSING_DEPENDENCY_REMEDY };
|
||||
}
|
||||
if (matchesAny(text, POSIX_MISSING_DEPENDENCY_SIGNATURES)) {
|
||||
return { kind: 'missing_dependency', remedy: posixMissingDependencyRemedy(label) };
|
||||
return { kind: 'missing_dependency', remedy: POSIX_MISSING_DEPENDENCY_REMEDY };
|
||||
}
|
||||
// Language-independent fallback: the extension demonstrably failed to load
|
||||
// (lbug's English wrapper is present) but the localized OS tail matched no
|
||||
@@ -191,9 +182,9 @@ export function classifyExtensionLoadError(
|
||||
// remedy — strictly better than the generic `unknown` for non-English hosts,
|
||||
// and it never prescribes the wrong fix.
|
||||
if (LOAD_FAILURE_WRAPPER.test(text)) {
|
||||
return { kind: 'missing_dependency', remedy: hedgedLoadFailureRemedy(label) };
|
||||
return { kind: 'missing_dependency', remedy: HEDGED_LOAD_FAILURE_REMEDY };
|
||||
}
|
||||
return { kind: 'unknown', remedy: unknownRemedy(label) };
|
||||
return { kind: 'unknown', remedy: UNKNOWN_REMEDY };
|
||||
}
|
||||
|
||||
// ── Language-independent structural layer ────────────────────────────────────
|
||||
@@ -201,8 +192,8 @@ export function classifyExtensionLoadError(
|
||||
/** Well-formedness of the extension binary for the host platform + arch. */
|
||||
export type ExtensionBinaryState = 'absent' | 'corrupt' | 'valid' | 'indeterminate';
|
||||
|
||||
const structuralMissingDependencyRemedy = (label: string): string =>
|
||||
`The ${label} extension file is valid, so the failure is a missing or incompatible runtime dependency, ` +
|
||||
const STRUCTURAL_MISSING_DEPENDENCY_REMEDY =
|
||||
'The FTS extension file is valid, so the failure is a missing or incompatible runtime dependency, ' +
|
||||
'not the extension itself — reinstalling will NOT help. On Windows, install ' +
|
||||
VC_REDIST_INSTALL_HINT +
|
||||
' and ensure OpenSSL 3 is available; on Linux/macOS install the shared library named in the error above.';
|
||||
@@ -341,16 +332,13 @@ export function inspectExtensionBinary(
|
||||
* classifier (which still carries the language-independent hedged fallback). This
|
||||
* is the entry point every surface should call.
|
||||
*/
|
||||
export function diagnoseExtensionLoad(
|
||||
reason: string | undefined | null,
|
||||
label: string = 'FTS',
|
||||
): ExtensionLoadDiagnosis {
|
||||
export function diagnoseExtensionLoad(reason: string | undefined | null): ExtensionLoadDiagnosis {
|
||||
const text = reason ?? '';
|
||||
const stringResult = classifyExtensionLoadError(text, label);
|
||||
const stringResult = classifyExtensionLoadError(text);
|
||||
const fileState = inspectExtensionBinary(extractExtensionPath(text));
|
||||
|
||||
if (fileState === 'corrupt') {
|
||||
return { kind: 'corrupt_file', remedy: corruptFileRemedy(label) };
|
||||
return { kind: 'corrupt_file', remedy: CORRUPT_FILE_REMEDY };
|
||||
}
|
||||
if (fileState === 'valid') {
|
||||
// The structural probe only inspects the first BINARY_HEADER_BYTES, so a file
|
||||
@@ -369,7 +357,7 @@ export function diagnoseExtensionLoad(
|
||||
const remedy =
|
||||
stringResult.kind === 'missing_dependency'
|
||||
? stringResult.remedy
|
||||
: structuralMissingDependencyRemedy(label);
|
||||
: STRUCTURAL_MISSING_DEPENDENCY_REMEDY;
|
||||
return { kind: 'missing_dependency', remedy };
|
||||
}
|
||||
// 'absent' or 'indeterminate' → no positive structural evidence, so defer to the
|
||||
|
||||
@@ -323,7 +323,7 @@ export class ExtensionManager {
|
||||
name,
|
||||
loaded: false,
|
||||
reason,
|
||||
diagnosis: diagnoseExtensionLoad(reason, label),
|
||||
diagnosis: diagnoseExtensionLoad(reason),
|
||||
});
|
||||
const key = `${name}:${reason}`;
|
||||
if (this.warnedKeys.has(key)) return;
|
||||
|
||||
@@ -23,11 +23,7 @@ import { streamAllCSVsToDisk, type StreamedCSVResult } from './csv-generator.js'
|
||||
import type { PdgEmitManifest } from './pdg-emit-sink.js';
|
||||
import { getNodeLabel as deriveNodeLabel, type WriteStreamFactory } from './rel-pair-routing.js';
|
||||
import { EMBEDDABLE_LABELS, type CachedEmbedding } from '../embeddings/types.js';
|
||||
import {
|
||||
extensionManager,
|
||||
resolveAnalyzeInstallPolicy,
|
||||
type ExtensionEnsureOptions,
|
||||
} from './extension-loader.js';
|
||||
import { extensionManager, type ExtensionEnsureOptions } from './extension-loader.js';
|
||||
import {
|
||||
classifyDeleteAllError,
|
||||
closeLbugConnection,
|
||||
@@ -36,7 +32,6 @@ import {
|
||||
isDbBusyError,
|
||||
isOpenRetryExhausted,
|
||||
isWalCorruptionError,
|
||||
bufferPoolExhaustionRemedy,
|
||||
openLbugConnection,
|
||||
sleep,
|
||||
toNativeSafePath,
|
||||
@@ -46,7 +41,6 @@ import {
|
||||
type LbugConnectionHandle,
|
||||
} from './lbug-config.js';
|
||||
import {
|
||||
cleanQuarantinedMissingShadowWals,
|
||||
finalizeLbugSidecarsAfterClose,
|
||||
guardWalQuarantine,
|
||||
isMissingShadowSidecarError,
|
||||
@@ -56,8 +50,8 @@ import {
|
||||
quarantineWalForMissingShadow,
|
||||
renameFailureMessage,
|
||||
shadowSidecarRecoveryMessage,
|
||||
sidecarPreflightDisabled,
|
||||
} from './sidecar-recovery.js';
|
||||
import { isVectorExtensionSupportedByPlatform } from '../platform/capabilities.js';
|
||||
|
||||
import { logger } from '../logger.js';
|
||||
// ---------------------------------------------------------------------------
|
||||
@@ -824,30 +818,6 @@ const doInitLbug = async (dbPath: string, readOnly: boolean = false) => {
|
||||
// -------------------------------------------------------------------------
|
||||
const releaseInitLock = await acquireInitLock(dbPath);
|
||||
try {
|
||||
// Reclaim missing-shadow WAL quarantines from a PRIOR crash (#2637).
|
||||
// LadybugDB renames an unrecoverable WAL aside as
|
||||
// `${dbPath}.wal.missing-shadow.<ts>-<rand>` (quarantineWalForMissingShadow)
|
||||
// instead of deleting it. Once quarantined it is permanently detached from
|
||||
// the live store and never reopened, so reclaiming it is safe regardless of
|
||||
// whether the main DB file exists this run — unlike the orphan-sidecar
|
||||
// cleanup below, this must NOT be gated on "main DB missing": a quarantine
|
||||
// event and a healthy main DB are independent facts. Never let a reclaim
|
||||
// failure (e.g. a transient EBUSY from an antivirus scan) block DB startup.
|
||||
if (!sidecarPreflightDisabled()) {
|
||||
try {
|
||||
const reclaimed = await cleanQuarantinedMissingShadowWals(dbPath);
|
||||
for (const file of reclaimed) {
|
||||
logger.warn(
|
||||
`GitNexus: reclaimed quarantined WAL ${path.basename(file)} from a prior crash`,
|
||||
);
|
||||
}
|
||||
} catch (err) {
|
||||
logger.warn(
|
||||
`GitNexus: failed to reclaim missing-shadow WAL quarantines: ${summarizeError(err)}`,
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
// Crash-recovery cleanup: if the main DB file is missing, stale sidecars
|
||||
// from an interrupted run can block fresh opens indefinitely.
|
||||
try {
|
||||
@@ -979,14 +949,7 @@ const copyNodeCSVs = async (
|
||||
const copyQuery = getCopyQuery(table, normalizeCopyPath(csvPath));
|
||||
await copyCsvWithRetry(targetConn, copyQuery, (retryErr) => {
|
||||
const retryMsg = retryErr instanceof Error ? retryErr.message : String(retryErr);
|
||||
// Pool exhaustion gets a remedy (#2631): the raw binder text gives the
|
||||
// operator nothing to act on, and on non-4K-page hosts (Ascend aarch64,
|
||||
// Apple Silicon) the pool bills up to pageSize/4KiB x faster than the
|
||||
// sizing was calibrated for — name the knob and the mechanism.
|
||||
const remedy = bufferPoolExhaustionRemedy(retryMsg);
|
||||
throw new Error(
|
||||
`COPY failed for ${table}: ${retryMsg.slice(0, 200)}${remedy ? ` ${remedy}` : ''}`,
|
||||
);
|
||||
throw new Error(`COPY failed for ${table}: ${retryMsg.slice(0, 200)}`);
|
||||
});
|
||||
}
|
||||
};
|
||||
@@ -1158,7 +1121,6 @@ export const loadGraphToLbug = async (
|
||||
|
||||
const insertedRels = totalValidRels;
|
||||
const warnings: string[] = [];
|
||||
let poolRemedyIssued = false;
|
||||
if (insertedRels > 0) {
|
||||
log(`Loading edges: ${insertedRels.toLocaleString()} across ${relsByPair.size} types`);
|
||||
|
||||
@@ -1185,17 +1147,6 @@ export const loadGraphToLbug = async (
|
||||
await copyCsvWithRetry(writeConn, copyQuery, (retryErr) => {
|
||||
const retryMsg = retryErr instanceof Error ? retryErr.message : String(retryErr);
|
||||
warnings.push(`${fromLabel}->${toLabel} (${rows} edges): ${retryMsg.slice(0, 80)}`);
|
||||
// One remedy per bulk load, not per pair (#2631): pool exhaustion
|
||||
// repeats for every remaining pair once it starts. logger.warn, not
|
||||
// just warnings.push — the returned warnings array has no consumer at
|
||||
// any call site, so a push alone would leave the remedy invisible
|
||||
// while the row-by-row fallback quietly degrades the load.
|
||||
const remedy = poolRemedyIssued ? undefined : bufferPoolExhaustionRemedy(retryMsg);
|
||||
if (remedy) {
|
||||
poolRemedyIssued = true;
|
||||
warnings.push(remedy);
|
||||
logger.warn(remedy);
|
||||
}
|
||||
failedPairEdges += rows;
|
||||
failedPairCsvPaths.add(pairCsvPath);
|
||||
});
|
||||
@@ -2745,16 +2696,14 @@ export const loadVectorExtension = async (
|
||||
): Promise<boolean> => {
|
||||
const useModuleState = targetConn === undefined;
|
||||
if (useModuleState && vectorExtensionLoaded) return true;
|
||||
// No platform gate. Windows was hard-refused here for years on the strength
|
||||
// of an early-era report that in-process INSTALL VECTOR could SIGSEGV
|
||||
// (#1365) — but the extension server ships win_amd64 VECTOR artifacts for
|
||||
// every 0.18.x extension version (probed live: v0.18.0 and v0.18.1 both
|
||||
// serve a real PE32+ DLL; the pinned 0.18.2 core resolves its extension
|
||||
// directory to 0.18.1, strace-verified), and INSTALL now runs in a spawned
|
||||
// child process (installDuckDbExtensionOutOfProcess), so even a crashing
|
||||
// installer kills only the child and degrades to `false` here. LOAD of a
|
||||
// present extension file is an ordinary in-process load whose failures
|
||||
// surface as catchable errors, exactly like FTS.
|
||||
// INSTALL VECTOR crashes with SIGSEGV on Windows: the KuzuDB native extension
|
||||
// installer has an unhandled error path on Windows that raises a fatal signal
|
||||
// that JS try/catch cannot intercept. Skip loading — vector/embedding search
|
||||
// is unavailable but all graph index queries still work. Do NOT set
|
||||
// vectorExtensionLoaded here: the flag means "successfully loaded", and a
|
||||
// subsequent call would otherwise short-circuit to `return true` at the top.
|
||||
if (process.platform === 'win32') return false;
|
||||
if (!isVectorExtensionSupportedByPlatform()) return false;
|
||||
|
||||
const c: lbug.Connection | null = targetConn ?? conn;
|
||||
if (!c) {
|
||||
@@ -2863,78 +2812,6 @@ export const createVectorIndex = async (): Promise<boolean> => {
|
||||
}
|
||||
};
|
||||
|
||||
/**
|
||||
* Make DML against {@link EMBEDDING_TABLE_NAME} legal on the writable
|
||||
* connection when it can be, and report whether it is.
|
||||
*
|
||||
* LadybugDB refuses EVERY mutation of a table carrying an HNSW index while
|
||||
* the VECTOR extension is not loaded on that connection: `DELETE` fails with
|
||||
* "Trying to delete from an index on table CodeEmbedding but its extension is
|
||||
* not loaded", `CREATE` with the matching "insert into an index" variant,
|
||||
* `DROP TABLE` is refused while the index references it, and `SET` — even on
|
||||
* a NON-indexed property — segfaults the process outright. Probed against
|
||||
* @ladybugdb/core 0.18.2 (the lockfile-pinned version) and 0.18.0 — every
|
||||
* result identical on both (#2623).
|
||||
*
|
||||
* Dropping the index is NOT an available recovery: `CALL DROP_VECTOR_INDEX`
|
||||
* is itself a VECTOR-extension function and resolves to "Catalog exception:
|
||||
* function DROP_VECTOR_INDEX is not defined" in exactly the state it would
|
||||
* need to rescue. Loading the extension is the only in-place repair, which is
|
||||
* why this returns a verdict instead of attempting a fixup.
|
||||
*
|
||||
* `true` = embedding-row DML is safe: either VECTOR is now loaded, or the
|
||||
* table carries no index to trip over. `false` = genuinely blocked (index
|
||||
* present, extension unloadable); the analyze orchestrator answers that by
|
||||
* escalating to the wipe-and-rebuild write plan instead of failing
|
||||
* mid-writeback.
|
||||
*
|
||||
* Cheap by construction: one local `SHOW_INDEXES` read settles the common
|
||||
* "this repo never built an embedding index" case without touching the
|
||||
* extension machinery at all, so a VECTOR-less machine is not charged a
|
||||
* bounded INSTALL attempt on every incremental analyze. `SHOW_INDEXES` is
|
||||
* readable WITHOUT the extension and reports `extension_loaded` per index, so
|
||||
* no error-string sniffing is needed; it runs through the unprepared
|
||||
* `conn.query()` path like every other `CALL` procedure here (#2114).
|
||||
*/
|
||||
export const ensureEmbeddingRowDmlSafe = async (): Promise<boolean> => {
|
||||
const targetConn = conn;
|
||||
if (!targetConn) {
|
||||
throw new Error('LadybugDB not initialized. Call initLbug first.');
|
||||
}
|
||||
// Catalog FIRST. The overwhelmingly common case on a repo that never enabled
|
||||
// embeddings is "no index at all", and that is provable with one local read
|
||||
// — no extension needed. Loading first would make every incremental analyze
|
||||
// on a VECTOR-less machine pay a bounded out-of-process INSTALL attempt (the
|
||||
// `auto` policy) plus an "extension unavailable" warning, for a repo that
|
||||
// can never hit this hazard.
|
||||
let indexRows: any[] | undefined;
|
||||
try {
|
||||
indexRows = await withConnLock(async () =>
|
||||
readQueryRows(await targetConn.query('CALL SHOW_INDEXES() RETURN *')),
|
||||
);
|
||||
} catch (err) {
|
||||
// Fall through to the load attempt: unable to prove the index is absent,
|
||||
// so the extension is the only thing that can make DML safe.
|
||||
logger.warn(
|
||||
{ err },
|
||||
`Could not read the index catalog to check for a ${EMBEDDING_TABLE_NAME} vector index; ` +
|
||||
'falling back to loading the VECTOR extension.',
|
||||
);
|
||||
}
|
||||
// Any non-HASH index on the embedding table gates DML. Keyed on index TYPE,
|
||||
// not name, so an index built under a different name still counts; the
|
||||
// implicit primary-key HASH index is engine-internal and never gates.
|
||||
const indexGatesDml =
|
||||
indexRows === undefined ||
|
||||
indexRows.some((row) => {
|
||||
const table = row?.table_name ?? row?.[0];
|
||||
if (table !== EMBEDDING_TABLE_NAME) return false;
|
||||
return (row?.index_type ?? row?.[2]) !== 'HASH';
|
||||
});
|
||||
if (!indexGatesDml) return true;
|
||||
return await loadVectorExtension(undefined, { policy: resolveAnalyzeInstallPolicy() });
|
||||
};
|
||||
|
||||
/**
|
||||
* Lazy-create an FTS index, caching the fact in-process.
|
||||
*
|
||||
@@ -3027,30 +2904,7 @@ export const queryFTS = async (
|
||||
};
|
||||
|
||||
/**
|
||||
* True for the two benign "nothing to drop" `DROP_FTS_INDEX` failures —
|
||||
* both catalog/binder exceptions, LadybugDB's classes for "this name isn't
|
||||
* bound to anything right now" (probe-verified end-to-end through
|
||||
* `dropFTSIndex`'s real `conn.query()` path against @ladybugdb/core
|
||||
* 0.18.x): the named index was never created (`Binder exception: Table <T>
|
||||
* doesn't have an index with name <name>.`), or the FTS extension/function
|
||||
* isn't registered at all (`Catalog exception: function DROP_FTS_INDEX is
|
||||
* not defined...`). A real engine failure — e.g. the `Runtime exception:
|
||||
* FTS index '<name>' is inconsistent: ...` class from #2589 — is a
|
||||
* DIFFERENT exception class (an execution-time failure, not a catalog/bind
|
||||
* lookup miss), so this returns false for it. Anchored to the START of the
|
||||
* message (not a bare substring search): every probed LadybugDB error leads
|
||||
* with its exception class, and anchoring means a future message that merely
|
||||
* mentions "Binder exception" or "Catalog exception" further in in the body
|
||||
* of an otherwise-genuine failure can't be misclassified as benign. Pure
|
||||
* string logic so it is unit-testable without a native LadybugDB connection.
|
||||
*/
|
||||
export const isBenignDropFtsIndexError = (message: string): boolean =>
|
||||
message.startsWith('Binder exception:') || message.startsWith('Catalog exception:');
|
||||
|
||||
/**
|
||||
* Drop an FTS index. Tolerates only {@link isBenignDropFtsIndexError} —
|
||||
* anything else rethrows instead of being silently masked, which previously
|
||||
* let a corrupted index persist across analyze runs undetected.
|
||||
* Drop an FTS index
|
||||
*/
|
||||
export const dropFTSIndex = async (tableName: string, indexName: string): Promise<void> => {
|
||||
if (!conn) {
|
||||
@@ -3059,11 +2913,8 @@ export const dropFTSIndex = async (tableName: string, indexName: string): Promis
|
||||
|
||||
try {
|
||||
await queryAndDrain(conn, `CALL DROP_FTS_INDEX('${tableName}', '${indexName}')`);
|
||||
} catch (e: unknown) {
|
||||
const msg = e instanceof Error ? e.message : String(e);
|
||||
if (!isBenignDropFtsIndexError(msg)) {
|
||||
throw e;
|
||||
}
|
||||
} catch {
|
||||
// Index may not exist
|
||||
} finally {
|
||||
ensuredFTSIndexes.delete(ftsIndexKey(tableName, indexName));
|
||||
}
|
||||
|
||||
@@ -321,16 +321,6 @@ const resolveCheckpointThreshold = (): number => {
|
||||
const DEFAULT_BUFFER_POOL_CAP = 2 * 1024 * 1024 * 1024;
|
||||
const BUFFER_POOL_FLOOR = 64 * 1024 * 1024;
|
||||
|
||||
// COPY-safety floor for the adaptive hint (below). LadybugDB's bulk COPY needs
|
||||
// working buffer-pool memory that scales with the repo: a 64 MiB pool fails
|
||||
// ("buffer pool is full and no memory could be freed") on any non-trivial repo,
|
||||
// and even the ~1800-file GitNexus checkout needs ≥256 MiB. So the adaptive
|
||||
// size never drops a repo below this — a distinct, higher floor than
|
||||
// BUFFER_POOL_FLOOR, which only guards defaultBufferPoolSize on tiny-RAM
|
||||
// machines. It is still clamped up to defaultBufferPoolSize, so a machine whose
|
||||
// default is below this floor keeps its default rather than over-committing.
|
||||
const ADAPTIVE_POOL_FLOOR = 256 * 1024 * 1024;
|
||||
|
||||
const parseBufferPoolSize = (raw: string | undefined): number | undefined => {
|
||||
if (raw === undefined) return undefined;
|
||||
const normalized = raw.trim();
|
||||
@@ -340,151 +330,19 @@ const parseBufferPoolSize = (raw: string | undefined): number | undefined => {
|
||||
return Math.floor(parsed);
|
||||
};
|
||||
|
||||
/**
|
||||
* The buffer-manager frame size compiled into every shipped `@ladybugdb/core`
|
||||
* binary (`LBUG_PAGE_SIZE_LOG2 = 12` in the engine's CMake) — frames are 4 KiB
|
||||
* on every platform, independent of the OS page size.
|
||||
*/
|
||||
const LBUG_ASSUMED_FRAME_SIZE = 4096;
|
||||
|
||||
/**
|
||||
* How much the OS page size amplifies buffer-pool consumption (#2631).
|
||||
*
|
||||
* LadybugDB's VM region charges pool budget per DISCARD GRANULE, not per
|
||||
* frame: `discardGranuleSize = max(frameSize, osPageSize)` (vm_region.cpp),
|
||||
* `claimFrame` bills the whole granule when its first 4 KiB frame becomes
|
||||
* resident, and `releaseFrame` refunds only when the granule's LAST frame
|
||||
* leaves. On a 64 KiB-page kernel (aarch64 openEuler — Ascend hosts) that is
|
||||
* 16 frames per granule: scattered access is billed up to 16× its real bytes,
|
||||
* and whole eviction passes can evict frames yet refund nothing — which is
|
||||
* exactly the engine's "buffer pool is full and no memory could be freed"
|
||||
* throw. Apple Silicon macOS (16 KiB pages) is the same mechanism at 4×.
|
||||
*
|
||||
* So the ANALYZE-path pool sizes (the per-element estimate, the COPY-safety
|
||||
* floor, and the cap the hint is clamped against) are scaled by this ratio:
|
||||
* the budget must cover worst-case granule charging or COPY dies on non-4K
|
||||
* hosts with a pool that would be ample on x86. The hintless default
|
||||
* (defaultBufferPoolSize — MCP serve, doctor, native-check) is deliberately
|
||||
* NOT scaled: the pool is a native eager allocation committed at DB open
|
||||
* (measured — see POOL_BYTES_PER_ELEMENT below), so scaling the global
|
||||
* default would revert the #2557 OOM cap on every 16 KiB/64 KiB host. If the
|
||||
* engine ever charges per-frame (or ships page-size-matched frames), this
|
||||
* collapses back to 1 and the scaling disappears.
|
||||
*
|
||||
* Fail-safe: an undetectable page size (win32 — where the granule mechanism
|
||||
* is absent anyway — or a failed `getconf`) means ratio 1, i.e. today's
|
||||
* behavior.
|
||||
*/
|
||||
export const granuleRatio = (pageSize: number | undefined = getOsPageSize()): number => {
|
||||
if (pageSize === undefined || !Number.isFinite(pageSize)) return 1;
|
||||
return Math.max(1, Math.floor(pageSize / LBUG_ASSUMED_FRAME_SIZE));
|
||||
};
|
||||
|
||||
/**
|
||||
* Hintless pool default — MCP serve, doctor, native-check, any open without a
|
||||
* per-run hint. Deliberately UNSCALED (#2557): the pool is an eager native
|
||||
* allocation at DB open, so a page-size-scaled default would hand a
|
||||
* long-lived `gitnexus mcp` on a 16 KiB/64 KiB host up to 80% of RAM — the
|
||||
* exact OOM exposure the 2 GiB cap was added to remove.
|
||||
*/
|
||||
const defaultBufferPoolSize = (): number =>
|
||||
Math.min(DEFAULT_BUFFER_POOL_CAP, Math.max(BUFFER_POOL_FLOOR, Math.floor(os.totalmem() * 0.8)));
|
||||
|
||||
/**
|
||||
* Upper bound for the ANALYZE-path (hinted) pool: the #2557 cap scaled by the
|
||||
* granule ratio, still bounded by 80% of RAM. Scaling only this bound — and
|
||||
* not defaultBufferPoolSize — is what lets the #2631 fix take effect during
|
||||
* the bulk COPY without touching hintless opens: with an unscaled cap the
|
||||
* min() below would clamp the scaled COPY floor straight back to 2 GiB.
|
||||
*/
|
||||
const scaledAnalyzePoolCap = (pageSize: number | undefined): number =>
|
||||
Math.min(
|
||||
DEFAULT_BUFFER_POOL_CAP * granuleRatio(pageSize),
|
||||
Math.max(BUFFER_POOL_FLOOR, Math.floor(os.totalmem() * 0.8)),
|
||||
);
|
||||
|
||||
/**
|
||||
* Clamp an adaptive pool request to [ADAPTIVE_POOL_FLOOR × granuleRatio,
|
||||
* scaledAnalyzePoolCap]. The lower bound keeps LadybugDB's COPY viable
|
||||
* (scaled because the granule accounting inflates consumption on non-4K
|
||||
* hosts, see granuleRatio); the upper bound means the hint can never exceed
|
||||
* the page-size-scaled #2557 cap or 80% of RAM — and on a machine whose cap
|
||||
* is below the COPY floor, the cap wins, so the pool is never over-committed.
|
||||
* On 4 KiB hosts (ratio 1) this is byte-identical to clamping against the
|
||||
* hintless default.
|
||||
*/
|
||||
const clampBufferPool = (bytes: number, pageSize: number | undefined = getOsPageSize()): number =>
|
||||
Math.min(
|
||||
scaledAnalyzePoolCap(pageSize),
|
||||
Math.max(ADAPTIVE_POOL_FLOOR * granuleRatio(pageSize), Math.floor(bytes)),
|
||||
);
|
||||
|
||||
/**
|
||||
* Buffer-pool bytes to provision per graph element (node + relationship).
|
||||
*
|
||||
* The fixed 2 GiB default is far larger than most repos' working set, and
|
||||
* LadybugDB eagerly commits the pool at DB open — measured: a full
|
||||
* `analyze --force` of the GitNexus checkout takes ~51 s with the 2 GiB pool
|
||||
* vs ~35 s with the ~414 MiB this factor yields (31% faster; the oversized
|
||||
* pool's commit dominates). The pool is a page cache over the on-disk index,
|
||||
* which scales with node/edge count, so a per-element budget sizes it to the
|
||||
* repo. Kept generous so the whole index stays resident (no COPY thrash) and
|
||||
* always clamped to at least ADAPTIVE_POOL_FLOOR; tuned by timing a real
|
||||
* large-repo `analyze --force` at this factor vs a forced 2 GiB pool (the pool
|
||||
* is a native eager allocation, measured with a real analyze, not a build-free
|
||||
* bench — see the emit-path COPY timing note in bench/emit-persistence).
|
||||
*/
|
||||
const POOL_BYTES_PER_ELEMENT = 4 * 1024;
|
||||
|
||||
/**
|
||||
* Size the buffer pool to an estimated graph size (node + relationship count),
|
||||
* clamped to [ADAPTIVE_POOL_FLOOR, scaledAnalyzePoolCap], with every term
|
||||
* scaled by granuleRatio (#2631): on non-4K hosts the engine bills pool
|
||||
* budget per OS-page-sized granule, so the same graph consumes up to
|
||||
* pageSize/4096 × the budget it needs on x86. On 4 KiB hosts the ratio is 1
|
||||
* and this is byte-identical to the pre-#2631 behavior. The estimate is never
|
||||
* above the page-size-scaled #2557 cap bounded by 80% of RAM, never below the
|
||||
* scaled COPY-safety floor; the hintless default stays unscaled.
|
||||
*
|
||||
* `pageSize` is a test seam (the pageSizeDoctorLines convention); production
|
||||
* callers omit it and get the memoized real OS page size.
|
||||
*/
|
||||
export const estimateBufferPool = (
|
||||
graphElementCount: number,
|
||||
pageSize: number | undefined = getOsPageSize(),
|
||||
): number =>
|
||||
clampBufferPool(graphElementCount * POOL_BYTES_PER_ELEMENT * granuleRatio(pageSize), pageSize);
|
||||
|
||||
/**
|
||||
* Optional per-run buffer-pool size hint (bytes). The analyze orchestrator sets
|
||||
* it once the graph size is known (after the pipeline, before the DB open) so
|
||||
* the pool is sized to the repo instead of the fixed 2 GiB default, and clears
|
||||
* it at run end. Non-analyze opens (MCP serve, `native-check` `:memory:`) never
|
||||
* set it and keep the default.
|
||||
*/
|
||||
let bufferPoolSizeHint: number | undefined;
|
||||
|
||||
/** Set (bytes) or clear (`undefined`) the per-run buffer-pool size hint. */
|
||||
export const setBufferPoolSizeHint = (bytes: number | undefined): void => {
|
||||
bufferPoolSizeHint = bytes;
|
||||
};
|
||||
|
||||
/**
|
||||
* Resolve the `bufferManagerSize` passed to every `new lbug.Database(...)`.
|
||||
* `GITNEXUS_LBUG_BUFFER_POOL_SIZE` (bytes) overrides everything; `0` is a
|
||||
* `GITNEXUS_LBUG_BUFFER_POOL_SIZE` (bytes) overrides the default; `0` is a
|
||||
* deliberate escape hatch that restores LadybugDB's native unbounded
|
||||
* 80%-of-RAM default. With no env override, a per-run `setBufferPoolSizeHint`
|
||||
* (clamped to [floor, default]) sizes the pool to the repo; otherwise the
|
||||
* default. Resolved at call time (not module load) so tests can stub the env
|
||||
* var, the hint, and `os.totalmem`.
|
||||
* 80%-of-RAM default. Resolved at call time (not module load) so tests can
|
||||
* stub the env var and `os.totalmem`.
|
||||
*/
|
||||
const resolveBufferManagerSize = (): number => {
|
||||
const raw = process.env.GITNEXUS_LBUG_BUFFER_POOL_SIZE;
|
||||
if (raw === undefined) {
|
||||
return bufferPoolSizeHint !== undefined
|
||||
? clampBufferPool(bufferPoolSizeHint)
|
||||
: defaultBufferPoolSize();
|
||||
}
|
||||
if (raw === undefined) return defaultBufferPoolSize();
|
||||
const parsed = parseBufferPoolSize(raw);
|
||||
if (parsed !== undefined) return parsed;
|
||||
// Non-empty but unparseable input: warn the operator and fall back —
|
||||
@@ -492,64 +350,12 @@ const resolveBufferManagerSize = (): number => {
|
||||
if (raw.trim().length > 0) {
|
||||
logger.warn(
|
||||
{ rawValue: raw, fallback: defaultBufferPoolSize() },
|
||||
`Ignoring invalid GITNEXUS_LBUG_BUFFER_POOL_SIZE=${raw}; expected integer >= 0 (bytes; 0 restores the native 80%-of-RAM default); falling back to the platform default pool size.`,
|
||||
`Ignoring invalid GITNEXUS_LBUG_BUFFER_POOL_SIZE=${raw}; expected integer >= 0 (bytes; 0 restores the native 80%-of-RAM default); falling back to min(2 GiB, 80% of RAM).`,
|
||||
);
|
||||
}
|
||||
return defaultBufferPoolSize();
|
||||
};
|
||||
|
||||
/**
|
||||
* Doctor-facing view of the pool size the next Database open would get
|
||||
* (#2631): env override > clamped hint > unscaled hintless default. Read-only;
|
||||
* doctor prints it next to the page-size lines so support triage sees the
|
||||
* sizing inputs at a glance. `0` is the pass-through sentinel for LadybugDB's
|
||||
* native 80%-of-RAM default — callers must label it, not print "0 MiB".
|
||||
*/
|
||||
export const getEffectiveBufferPoolSize = (): number => resolveBufferManagerSize();
|
||||
|
||||
/**
|
||||
* Matches the engine's buffer-pool exhaustion throw (buffer_manager.cpp:
|
||||
* "Unable to allocate memory! The buffer pool is full and no memory could be
|
||||
* freed!"). Distinct from isLbugPageSizeFrameError above, which matches the
|
||||
* madvise/frame-release failure class.
|
||||
*/
|
||||
const BUFFER_POOL_EXHAUSTION_RE = /buffer pool is full|unable to allocate memory/i;
|
||||
|
||||
const formatMiB = (bytes: number): string => `${Math.round(bytes / (1024 * 1024))} MiB`;
|
||||
|
||||
/**
|
||||
* Actionable remedy for a buffer-pool exhaustion error (#2631), or undefined
|
||||
* when `message` is not that class. Cause → consequence → remedy, the
|
||||
* diagnoseExtensionLoad convention: names the effective pool, the override
|
||||
* knob, and — on non-4K hosts — the granule amplification that makes the
|
||||
* budget exhaust early (the reporter's Ascend/aarch64 64 KiB kernel billed a
|
||||
* pool up to 16× faster than the same analyze on x86).
|
||||
*/
|
||||
export const bufferPoolExhaustionRemedy = (
|
||||
message: string,
|
||||
pageSize: number | undefined = getOsPageSize(),
|
||||
): string | undefined => {
|
||||
if (!BUFFER_POOL_EXHAUSTION_RE.test(message)) return undefined;
|
||||
const ratio = granuleRatio(pageSize);
|
||||
const pool = resolveBufferManagerSize();
|
||||
// 0 is the pass-through sentinel (GITNEXUS_LBUG_BUFFER_POOL_SIZE=0 →
|
||||
// LadybugDB's native 80%-of-RAM default) — "0 MiB" would be nonsense in the
|
||||
// very triage text this remedy exists to provide.
|
||||
const poolLabel = pool === 0 ? "LadybugDB's native 80%-of-RAM default" : formatMiB(pool);
|
||||
const pageNote =
|
||||
ratio > 1
|
||||
? ` This host's ${(pageSize ?? 0) / 1024} KiB OS page size makes the engine bill pool ` +
|
||||
`memory in ${(pageSize ?? 0) / 1024} KiB granules — up to ${ratio}× faster budget use ` +
|
||||
`than a 4 KiB-page host running the same analyze.`
|
||||
: '';
|
||||
return (
|
||||
`The LadybugDB buffer pool (${poolLabel}) was exhausted during the bulk COPY.` +
|
||||
pageNote +
|
||||
` Set GITNEXUS_LBUG_BUFFER_POOL_SIZE=<bytes> to raise it (e.g. ${4 * 1024 * 1024 * 1024}` +
|
||||
` for 4 GiB); 0 restores LadybugDB's native 80%-of-RAM default.`
|
||||
);
|
||||
};
|
||||
|
||||
/** Matches WAL corruption errors from the LadybugDB engine. */
|
||||
const WAL_CORRUPTION_RE = /corrupt(ed)?\s+wal|invalid\s+wal\s+record|wal.*corrupt|checksum.*wal/i;
|
||||
|
||||
@@ -636,12 +442,8 @@ const LBUG_PAGE_COMBO_RE = /unsupported page size combination/i;
|
||||
* True when `err` looks like the LadybugDB buffer manager failing to release
|
||||
* frame memory — the failure mode of a 4 KiB page-size assumption on a
|
||||
* 16 KiB/64 KiB-page kernel (#1231). Deliberately does NOT match the
|
||||
* generic "buffer pool is full" exhaustion error: that one is handled as a
|
||||
* SIZING problem — though since #2631 we know page size drives sizing too
|
||||
* (the engine bills pool budget per OS-page-sized discard granule, so non-4K
|
||||
* hosts exhaust the same budget up to pageSize/4096× earlier; see
|
||||
* granuleRatio, which scales the pool accordingly, and
|
||||
* bufferPoolExhaustionRemedy, which explains it to the operator).
|
||||
* generic "buffer pool is full" exhaustion error, which is a sizing
|
||||
* problem, not a page-size one.
|
||||
*/
|
||||
export const isLbugPageSizeFrameError = (err: unknown): boolean => {
|
||||
if (!err) return false;
|
||||
@@ -668,16 +470,6 @@ export const isPageSizeAwareLadybug = (version: string | undefined): boolean =>
|
||||
// because analyze error paths and doctor may both ask, and getconf forks.
|
||||
let cachedOsPageSize: number | null | undefined;
|
||||
|
||||
/**
|
||||
* Test seam (the `_captureLogger` convention): pin the memoized OS page size
|
||||
* so sizing tests are host-independent — without this they would silently
|
||||
* drift on 16 KiB-page Apple Silicon runners. `number` pins a value, `null`
|
||||
* pins "undetectable", `undefined` clears the memo so the next call re-probes.
|
||||
*/
|
||||
export const _setOsPageSizeForTests = (pageSize: number | null | undefined): void => {
|
||||
cachedOsPageSize = pageSize;
|
||||
};
|
||||
|
||||
/**
|
||||
* OS memory page size in bytes, or `undefined` when it cannot be determined
|
||||
* (Windows, missing getconf, sandboxed exec). Node exposes no page-size API,
|
||||
@@ -716,6 +508,11 @@ export const getOsPageSize = (): number | undefined => {
|
||||
return cachedOsPageSize ?? undefined;
|
||||
};
|
||||
|
||||
/** Exported only for unit tests — clears the getconf probe cache. */
|
||||
export const _resetOsPageSizeCacheForTest = (): void => {
|
||||
cachedOsPageSize = undefined;
|
||||
};
|
||||
|
||||
type LbugModule = typeof lbug;
|
||||
|
||||
export interface LbugDatabaseOptions {
|
||||
@@ -764,29 +561,6 @@ export const isDbBusyError = (err: unknown): boolean => {
|
||||
);
|
||||
};
|
||||
|
||||
/**
|
||||
* True when a WAL-checkpoint IO error ALSO carries a busy/lock signal — the
|
||||
* rotation failed because another handle (a `gitnexus mcp` server, or this
|
||||
* process's own reader) holds the store's WAL open, rather than a permanent
|
||||
* disk error. Reuses `isDbBusyError`'s already-tested keyword set instead of a
|
||||
* fresh regex, so an unmatched message degrades to "IO error" rather than
|
||||
* silently claiming a held-open cause. (#2599)
|
||||
*/
|
||||
export const isLbugCheckpointBusyError = (err: unknown): boolean => {
|
||||
if (!isLbugCheckpointIoError(err)) return false;
|
||||
// Anchor to real held-open wording rather than isDbBusyError's broad
|
||||
// `.includes('lock')`, which matches the DB PATH embedded in the checkpoint
|
||||
// error message (e.g. a repo under `blockchain-app`) and would misclassify a
|
||||
// pure disk fault as held-open (#2614 LOW).
|
||||
const msg = (err instanceof Error ? err.message : String(err)).toLowerCase();
|
||||
return (
|
||||
msg.includes('could not set lock') ||
|
||||
msg.includes('lock is held') ||
|
||||
msg.includes('being used by another process') ||
|
||||
msg.includes('is busy')
|
||||
);
|
||||
};
|
||||
|
||||
/** See {@link classifyDeleteAllError}. */
|
||||
export type DeleteAllErrorClass = 'benign-missing-table' | 'rethrow';
|
||||
|
||||
@@ -874,19 +648,6 @@ export const HANDLE_RELEASE_PROBE_ATTEMPTS = 5;
|
||||
export const HANDLE_RELEASE_PROBE_DELAY_MS = 50;
|
||||
const HANDLE_RELEASE_LOCK_CODES = new Set(['EBUSY', 'EPERM', 'EACCES']);
|
||||
|
||||
// Retry-budget registry, part 2 (retry-budget consolidation): the remaining
|
||||
// open-time lock retries live next to their call sites but are catalogued here
|
||||
// so all lbug retry budgets surface in one grep. They retry the same lock class
|
||||
// as 1–3 ("Could not set lock" while a writer rebuilds the index):
|
||||
// 4. LOCK_RETRY_ATTEMPTS / LOCK_RETRY_DELAY_MS (pool-adapter.ts)
|
||||
// → read pool's read-only open while `gitnexus analyze` is writing
|
||||
// (3 attempts, linear 2s·n back-off ≈ 6s total)
|
||||
// 5. LBUG_OPEN_RETRY_ATTEMPTS / _BASE_MS / _MAX_MS (group/bridge-db.ts)
|
||||
// → cross-repo bridge RO open race (10 attempts, linear 100ms·n capped
|
||||
// at 500ms ≈ 3.5s total)
|
||||
// Kept in-file (not moved here) so explicit `lbug-config` test mocks don't have
|
||||
// to enumerate them; change a budget in its call site and update this catalogue.
|
||||
|
||||
/**
|
||||
* Test-fixture directory prefixes recognized by `isTestFixturePath`.
|
||||
*
|
||||
|
||||
@@ -96,9 +96,6 @@ export interface FtsProbeResult {
|
||||
reason?: string;
|
||||
}
|
||||
|
||||
/** Same shape for every optional extension; `FtsProbeResult` is the legacy name. */
|
||||
export type ExtensionProbeResult = FtsProbeResult;
|
||||
|
||||
const DEFAULT_FTS_PROBE_TIMEOUT_MS = 10_000;
|
||||
|
||||
/** A LadybugDB query result exposes a synchronous `close()`. */
|
||||
@@ -139,39 +136,8 @@ const closeProbeResults = (result: unknown): void => {
|
||||
export async function probeFtsExtensionLoad(
|
||||
timeoutMs: number = DEFAULT_FTS_PROBE_TIMEOUT_MS,
|
||||
): Promise<FtsProbeResult> {
|
||||
return await probeExtensionLoad('fts', timeoutMs);
|
||||
}
|
||||
|
||||
/**
|
||||
* Live-probe `LOAD EXTENSION vector`, the VECTOR counterpart of the FTS probe.
|
||||
*
|
||||
* Needed for the same reason #2374 needed the FTS one, and reported the same
|
||||
* way: #2623's reporter saw `doctor` print `VECTOR index: available` while
|
||||
* every incremental `analyze` was dying because the extension had not loaded.
|
||||
* `doctor` derived that line from a static platform capability, so it read
|
||||
* "available" no matter what the extension file was doing.
|
||||
*
|
||||
* Probes for real on every platform, Windows included: the extension server
|
||||
* ships win_amd64 VECTOR artifacts for every 0.18.x extension version (the
|
||||
* old blanket Windows refusal was stale, #1365-era). LOAD never touches the
|
||||
* network and never invokes the installer, so this probe is exactly as safe
|
||||
* as the FTS one above.
|
||||
*/
|
||||
export async function probeVectorExtensionLoad(
|
||||
timeoutMs: number = DEFAULT_FTS_PROBE_TIMEOUT_MS,
|
||||
): Promise<ExtensionProbeResult> {
|
||||
return await probeExtensionLoad('vector', timeoutMs);
|
||||
}
|
||||
|
||||
/**
|
||||
* Shared LOAD probe. `extension` is a fixed internal literal, never user input.
|
||||
*/
|
||||
async function probeExtensionLoad(
|
||||
extension: 'fts' | 'vector',
|
||||
timeoutMs: number,
|
||||
): Promise<ExtensionProbeResult> {
|
||||
let timer: ReturnType<typeof setTimeout> | undefined;
|
||||
const timeout = new Promise<ExtensionProbeResult>((resolve) => {
|
||||
const timeout = new Promise<FtsProbeResult>((resolve) => {
|
||||
timer = setTimeout(
|
||||
() =>
|
||||
resolve({
|
||||
@@ -182,7 +148,7 @@ async function probeExtensionLoad(
|
||||
);
|
||||
});
|
||||
|
||||
const probe = (async (): Promise<ExtensionProbeResult> => {
|
||||
const probe = (async (): Promise<FtsProbeResult> => {
|
||||
try {
|
||||
const { default: lbug } = await import('@ladybugdb/core');
|
||||
const db = new lbug.Database(':memory:');
|
||||
@@ -190,7 +156,7 @@ async function probeExtensionLoad(
|
||||
try {
|
||||
const conn = new lbug.Connection(db);
|
||||
try {
|
||||
const result = await conn.query(`LOAD EXTENSION ${extension}`);
|
||||
const result = await conn.query('LOAD EXTENSION fts');
|
||||
closeProbeResults(result);
|
||||
return { loaded: true };
|
||||
} finally {
|
||||
|
||||
@@ -17,7 +17,7 @@
|
||||
|
||||
import fs from 'fs/promises';
|
||||
import lbug from '@ladybugdb/core';
|
||||
import { isReadOnlyDbError, loadFTSExtension, loadVectorExtension } from './lbug-adapter.js';
|
||||
import { isReadOnlyDbError, loadFTSExtension } from './lbug-adapter.js';
|
||||
import { closeQueryResults } from './query-result-utils.js';
|
||||
import {
|
||||
createLbugDatabase,
|
||||
@@ -53,44 +53,10 @@ interface PoolEntry {
|
||||
}>;
|
||||
lastUsed: number;
|
||||
dbPath: string;
|
||||
/** Filesystem identity of the on-disk DB at open time. When `analyze`
|
||||
* rebuilds or mutates the index, this diverges from the current file and
|
||||
* initLbug re-opens the pool onto the new file instead of serving the
|
||||
* stale open inode. Null for injected/external databases (initLbugWithDb),
|
||||
* which are never invalidated this way. */
|
||||
dbIdentity: DbIdentity | null;
|
||||
/** Set to true when the pool entry is closed — checkin will close orphaned connections */
|
||||
closed: boolean;
|
||||
}
|
||||
|
||||
/** Filesystem identity used to detect an index rebuilt/mutated under a live
|
||||
* read pool. `ino` catches a full-rebuild unlink+recreate or an atomic-rename
|
||||
* swap; `mtimeMs`+`size` catch an in-place incremental writeback. */
|
||||
interface DbIdentity {
|
||||
ino: number;
|
||||
mtimeMs: number;
|
||||
size: number;
|
||||
}
|
||||
|
||||
export async function statDbIdentity(dbPath: string): Promise<DbIdentity | null> {
|
||||
try {
|
||||
const s = await fs.stat(dbPath);
|
||||
return { ino: s.ino, mtimeMs: s.mtimeMs, size: s.size };
|
||||
} catch {
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
/** True only when both identities are known AND differ. A stat failure
|
||||
* (ENOENT during the brief unlink window of a full rebuild) yields false, so
|
||||
* the reader keeps serving its still-valid open inode until the NEW file
|
||||
* appears with a different identity — avoiding a churn into a failed reopen
|
||||
* mid-rebuild. */
|
||||
export function dbIdentityChanged(prev: DbIdentity | null, next: DbIdentity | null): boolean {
|
||||
if (!prev || !next) return false;
|
||||
return prev.ino !== next.ino || prev.mtimeMs !== next.mtimeMs || prev.size !== next.size;
|
||||
}
|
||||
|
||||
const pool = new Map<string, PoolEntry>();
|
||||
|
||||
/**
|
||||
@@ -126,18 +92,6 @@ interface SharedDB {
|
||||
db: lbug.Database;
|
||||
refCount: number;
|
||||
ftsLoaded: boolean;
|
||||
/** VECTOR loaded on this Database. Extension load scope is per-Database
|
||||
* (probe-verified on @ladybugdb/core 0.18.x): loading on any one
|
||||
* connection enables QUERY_VECTOR_INDEX on every connection of the same
|
||||
* Database. Without this load the pool's vector lane raised a Catalog
|
||||
* exception on every semantic query and silently fell back to the exact
|
||||
* scan (#2623 follow-up). Optional with `?? false` semantics so the
|
||||
* construction sites stay minimal. */
|
||||
vectorLoaded?: boolean;
|
||||
/** File identity at open — used to detect reuse of a shared read-only handle
|
||||
* whose on-disk index was rebuilt/swapped since it opened (only reachable
|
||||
* when a second pool consumer shares this dbPath; #2614 F2). */
|
||||
dbIdentity?: DbIdentity | null;
|
||||
/** When true, closeOne skips db.close() — the Database is owned externally. */
|
||||
external?: boolean;
|
||||
}
|
||||
@@ -366,7 +320,6 @@ function closeOne(repoId: string): void {
|
||||
// for the same dbPath reuse it instead of hitting a file lock.
|
||||
shared.refCount = 0;
|
||||
shared.ftsLoaded = false;
|
||||
shared.vectorLoaded = false;
|
||||
} else {
|
||||
shared.db.close().catch(() => {});
|
||||
dbCache.delete(entry.dbPath);
|
||||
@@ -436,16 +389,7 @@ setInterval(() => {
|
||||
function createConnection(db: lbug.Database): lbug.Connection {
|
||||
silenceStdout();
|
||||
try {
|
||||
const conn = new lbug.Connection(db);
|
||||
// Bound a single query at the engine level so a pathological query cannot
|
||||
// hang a pooled connection past the JS-side Promise.race guard (which frees
|
||||
// the waiter but not the native call). Matches QUERY_TIMEOUT_MS. Guarded so
|
||||
// test doubles that don't model the engine method don't break connection
|
||||
// creation.
|
||||
if (typeof conn.setQueryTimeout === 'function') {
|
||||
conn.setQueryTimeout(QUERY_TIMEOUT_MS);
|
||||
}
|
||||
return conn;
|
||||
return new lbug.Connection(db);
|
||||
} finally {
|
||||
restoreStdout();
|
||||
}
|
||||
@@ -456,8 +400,6 @@ const QUERY_TIMEOUT_MS = 30_000;
|
||||
/** Waiter queue timeout in milliseconds */
|
||||
const WAITER_TIMEOUT_MS = 15_000;
|
||||
|
||||
// Read-only open retry while `gitnexus analyze` writes. Catalogued as entry 4
|
||||
// of the lbug-config retry-budget registry.
|
||||
const LOCK_RETRY_ATTEMPTS = 3;
|
||||
const LOCK_RETRY_DELAY_MS = 2000;
|
||||
const SHADOW_REPLAY_PROBE_QUERY = 'MATCH (n) RETURN n LIMIT 1';
|
||||
@@ -651,45 +593,18 @@ const initPromises = new Map<string, Promise<void>>();
|
||||
* Concurrent calls for the same repoId are deduplicated — the second caller
|
||||
* awaits the first's in-progress init rather than starting a redundant one.
|
||||
*/
|
||||
/**
|
||||
* Returns `true` when this call (re)opened a fresh handle onto the current
|
||||
* on-disk file, `false` when it reused/served the existing handle (unchanged,
|
||||
* or changed-but-a-query-is-in-flight). Callers that gate their own freshness
|
||||
* bookkeeping on "did the pool actually roll over" (LocalBackend) use the
|
||||
* return value; callers that only need the pool ready can ignore it.
|
||||
*/
|
||||
export const initLbug = async (repoId: string, dbPath: string): Promise<boolean> => {
|
||||
export const initLbug = async (repoId: string, dbPath: string): Promise<void> => {
|
||||
const existing = pool.get(repoId);
|
||||
if (existing) {
|
||||
existing.lastUsed = Date.now();
|
||||
// Detect an index that `analyze` rebuilt or mutated under this live read
|
||||
// pool. Without this, the pool keeps serving the old (POSIX:
|
||||
// unlinked-but-open) inode until LRU/idle eviction — a stale-read window
|
||||
// of up to IDLE_TIMEOUT_MS after analyze finishes.
|
||||
const current = await statDbIdentity(dbPath);
|
||||
if (!dbIdentityChanged(existing.dbIdentity, current)) return false; // unchanged → reuse
|
||||
// A query is in flight on this entry; closing its connection (and the
|
||||
// shared Database at refCount 0) mid-use is a native use-after-free. Serve
|
||||
// the current handle for this dispatch — the next initLbug that finds the
|
||||
// entry idle (checkedOut === 0) reopens, since the identity stays divergent
|
||||
// until then. Under sustained overlapping queries `checkedOut` may never
|
||||
// reach 0 and `lastUsed` keeps the idle timer from evicting, so this window
|
||||
// is bounded by the load, not IDLE_TIMEOUT_MS — the data stays consistent
|
||||
// (a complete older snapshot), just not the newest. Callers that route
|
||||
// freshness THROUGH initLbug (rather than calling closeLbug directly) get
|
||||
// this guard for free; that is why LocalBackend delegates here (#2614).
|
||||
if (existing.checkedOut > 0) return false;
|
||||
closeOne(repoId); // idle & changed → evict, then fall through to reopen the new file
|
||||
return;
|
||||
}
|
||||
|
||||
// Deduplicate concurrent init calls for the same repoId —
|
||||
// prevents double-init race when multiple parallel tool calls
|
||||
// trigger initialization for the same repo simultaneously.
|
||||
const pending = initPromises.get(repoId);
|
||||
if (pending) {
|
||||
await pending;
|
||||
return true;
|
||||
}
|
||||
if (pending) return pending;
|
||||
|
||||
const promise = doInitLbug(repoId, dbPath);
|
||||
initPromises.set(repoId, promise);
|
||||
@@ -698,7 +613,6 @@ export const initLbug = async (repoId: string, dbPath: string): Promise<boolean>
|
||||
} finally {
|
||||
initPromises.delete(repoId);
|
||||
}
|
||||
return true;
|
||||
};
|
||||
|
||||
/**
|
||||
@@ -719,23 +633,6 @@ async function doInitLbug(repoId: string, dbPath: string): Promise<void> {
|
||||
// Reuse an existing native Database if another repoId already opened this path.
|
||||
// This prevents buffer manager exhaustion from multiple mmap regions on the same file.
|
||||
let shared = dbCache.get(dbPath);
|
||||
if (shared && !shared.external && shared.dbIdentity) {
|
||||
// #2614 F2: a cached read-only Database is keyed by dbPath and shared across
|
||||
// pool consumers. If the on-disk index was rebuilt/swapped (new inode) while
|
||||
// ANOTHER consumer still holds this handle (refCount kept it alive), reusing
|
||||
// it serves a superseded index. Unreachable via the MCP backend (one
|
||||
// consumer per lbugPath ⇒ refCount hits 0 ⇒ closeOne reopens fresh); a
|
||||
// complete fix needs per-inode handles rather than a dbPath-keyed cache.
|
||||
// Surface it so the corner is observable instead of silently stale.
|
||||
const current = await statDbIdentity(dbPath);
|
||||
if (dbIdentityChanged(shared.dbIdentity, current)) {
|
||||
realStderrWrite(
|
||||
`GitNexus: reusing a shared read-only handle for ${dbPath} whose on-disk ` +
|
||||
`index was rebuilt while another consumer holds it — results may be stale ` +
|
||||
`until that consumer releases it.\n`,
|
||||
);
|
||||
}
|
||||
}
|
||||
if (!shared) {
|
||||
// Open in read-only mode — MCP server never writes to the database.
|
||||
// This allows multiple MCP server instances to read concurrently, and
|
||||
@@ -744,7 +641,7 @@ async function doInitLbug(repoId: string, dbPath: string): Promise<void> {
|
||||
for (let attempt = 1; attempt <= LOCK_RETRY_ATTEMPTS; attempt++) {
|
||||
try {
|
||||
const db = await openReadOnlyDatabase(dbPath);
|
||||
shared = { db, refCount: 0, ftsLoaded: false, dbIdentity: await statDbIdentity(dbPath) };
|
||||
shared = { db, refCount: 0, ftsLoaded: false };
|
||||
dbCache.set(dbPath, shared);
|
||||
break;
|
||||
} catch (err: any) {
|
||||
@@ -753,12 +650,7 @@ async function doInitLbug(repoId: string, dbPath: string): Promise<void> {
|
||||
if (isWalCorruptionError(lastError)) {
|
||||
try {
|
||||
const db = await tryQuarantineAndReopen(dbPath, repoId);
|
||||
shared = {
|
||||
db,
|
||||
refCount: 0,
|
||||
ftsLoaded: false,
|
||||
dbIdentity: await statDbIdentity(dbPath),
|
||||
};
|
||||
shared = { db, refCount: 0, ftsLoaded: false };
|
||||
dbCache.set(dbPath, shared);
|
||||
break;
|
||||
} catch (retryErr) {
|
||||
@@ -819,20 +711,10 @@ async function doInitLbug(repoId: string, dbPath: string): Promise<void> {
|
||||
if (!shared.ftsLoaded) {
|
||||
shared.ftsLoaded = await loadFTSExtension(available[0], { policy: 'load-only' });
|
||||
}
|
||||
// VECTOR too — extension load scope is per-Database, so this one load
|
||||
// makes QUERY_VECTOR_INDEX legal on every pooled connection. Same
|
||||
// load-only contract as FTS above; on failure the semantic-query lane
|
||||
// falls back to the exact scan with its own diagnostic (#2623 follow-up).
|
||||
if (!shared.vectorLoaded) {
|
||||
shared.vectorLoaded = await loadVectorExtension(available[0], { policy: 'load-only' });
|
||||
}
|
||||
|
||||
// Register pool entry only after all connections are pre-warmed and FTS is
|
||||
// loaded. Concurrent executeQuery calls see either "not initialized"
|
||||
// (and throw cleanly) or a fully ready pool — never a half-built one.
|
||||
// Record the on-disk identity so a later initLbug can detect an analyze
|
||||
// rebuild/mutation and re-open onto the new file (pool staleness invalidation).
|
||||
const dbIdentity = await statDbIdentity(dbPath);
|
||||
pool.set(repoId, {
|
||||
db,
|
||||
available,
|
||||
@@ -840,7 +722,6 @@ async function doInitLbug(repoId: string, dbPath: string): Promise<void> {
|
||||
waiters: [],
|
||||
lastUsed: Date.now(),
|
||||
dbPath,
|
||||
dbIdentity,
|
||||
closed: false,
|
||||
});
|
||||
ensureIdleTimer();
|
||||
@@ -896,11 +777,6 @@ export async function initLbugWithDb(
|
||||
if (!shared.ftsLoaded) {
|
||||
shared.ftsLoaded = await loadFTSExtension(available[0], { policy: 'load-only' });
|
||||
}
|
||||
// VECTOR too — same per-Database scope and load-only contract as the
|
||||
// doInitLbug site above (#2623 follow-up).
|
||||
if (!shared.vectorLoaded) {
|
||||
shared.vectorLoaded = await loadVectorExtension(available[0], { policy: 'load-only' });
|
||||
}
|
||||
|
||||
pool.set(repoId, {
|
||||
db: existingDb,
|
||||
@@ -909,8 +785,6 @@ export async function initLbugWithDb(
|
||||
waiters: [],
|
||||
lastUsed: Date.now(),
|
||||
dbPath,
|
||||
// Injected/external DB (tests) — not tracked for rebuild invalidation.
|
||||
dbIdentity: null,
|
||||
closed: false,
|
||||
});
|
||||
ensureIdleTimer();
|
||||
|
||||
@@ -402,7 +402,6 @@ CREATE REL TABLE ${REL_TABLE_NAME} (
|
||||
FROM \`Static\` TO Community,
|
||||
FROM \`Variable\` TO Community,
|
||||
FROM \`Property\` TO Community,
|
||||
FROM \`Property\` TO \`Property\`,
|
||||
FROM \`Record\` TO Method,
|
||||
FROM \`Record\` TO \`Constructor\`,
|
||||
FROM \`Record\` TO \`Property\`,
|
||||
|
||||
@@ -60,7 +60,7 @@ export const isMissingFsError = (err: unknown): boolean =>
|
||||
|
||||
const missing = isMissingFsError;
|
||||
|
||||
export const sidecarPreflightDisabled = (): boolean =>
|
||||
const sidecarPreflightDisabled = (): boolean =>
|
||||
/^(1|true|yes|on)$/i.test(process.env.GITNEXUS_DISABLE_LBUG_SIDECAR_PREFLIGHT ?? '');
|
||||
|
||||
export const statIfExists = async (filePath: string): Promise<{ size: number } | null> => {
|
||||
|
||||
@@ -118,9 +118,6 @@ export const runCheckpointWithRetry = async (
|
||||
{ attempts: CHECKPOINT_RETRY_ATTEMPTS },
|
||||
'GitNexus: manual WAL checkpoint exhausted retry budget — surfacing IO error to caller',
|
||||
);
|
||||
// The held-open cause (#2599) is named at the CLI layer (analyze.ts) where the
|
||||
// --wal-checkpoint-threshold recovery hint already renders, so the original IO
|
||||
// error is preserved intact for that classifier rather than re-wrapped here.
|
||||
throw lastError;
|
||||
};
|
||||
|
||||
|
||||
@@ -86,24 +86,23 @@ export const getRuntimeFingerprint = (): RuntimeFingerprint => ({
|
||||
onnxruntime: packageVersion('onnxruntime-node'),
|
||||
});
|
||||
|
||||
export const isVectorExtensionSupportedByPlatform = (
|
||||
platform: NodeJS.Platform = process.platform,
|
||||
): boolean => platform !== 'win32';
|
||||
|
||||
export const getRuntimeCapabilities = (): RuntimeCapabilities => {
|
||||
const vector = isVectorExtensionSupportedByPlatform() ? 'available' : 'unavailable';
|
||||
const exactScanLimit = getExactScanLimit();
|
||||
// Static PLATFORM capability only. LadybugDB ships the VECTOR extension for
|
||||
// every platform gitnexus supports — the extension server hosts win_amd64
|
||||
// artifacts for every 0.18.x extension version (probed: v0.18.0 and v0.18.1
|
||||
// both return a real 14 MB PE32+ DLL; the pinned 0.18.2 core resolves its
|
||||
// extension directory to 0.18.1, strace-verified), so the old
|
||||
// `platform !== 'win32'` gate was stale (#1365-era). Whether the extension
|
||||
// actually LOADS on a given machine is a runtime question — doctor answers
|
||||
// it with probeVectorExtensionLoad, and analyze/query degrade to exact scan
|
||||
// when the load fails.
|
||||
return {
|
||||
graph: 'available',
|
||||
fts: 'available',
|
||||
vector: 'available',
|
||||
semanticMode: 'vector-index',
|
||||
vector,
|
||||
semanticMode: vector === 'available' ? 'vector-index' : 'exact-scan',
|
||||
exactScanLimit,
|
||||
reason: undefined,
|
||||
reason:
|
||||
vector === 'unavailable'
|
||||
? 'LadybugDB VECTOR is disabled on this platform; semantic search uses exact scan when embeddings exist.'
|
||||
: undefined,
|
||||
};
|
||||
};
|
||||
|
||||
|
||||
@@ -11,7 +11,6 @@
|
||||
|
||||
import path from 'path';
|
||||
import fs from 'fs/promises';
|
||||
import { retryRename } from '../storage/fs-atomic.js';
|
||||
import { runPipelineFromRepo } from './ingestion/pipeline.js';
|
||||
import type { KnowledgeGraph } from './graph/types.js';
|
||||
import { resetDegradedParseCounter } from './tree-sitter/safe-parse.js';
|
||||
@@ -25,7 +24,6 @@ import {
|
||||
closeLbugBeforeExit,
|
||||
loadCachedEmbeddings,
|
||||
deleteNodesForFiles,
|
||||
ensureEmbeddingRowDmlSafe,
|
||||
deleteAllCommunitiesAndProcesses,
|
||||
deleteAllInterprocTaintPaths,
|
||||
deleteAllCallSummaries,
|
||||
@@ -36,12 +34,10 @@ import {
|
||||
LbugWipeError,
|
||||
DELETE_FILES_CHUNK_SIZE,
|
||||
} from './lbug/lbug-adapter.js';
|
||||
import { estimateBufferPool, setBufferPoolSizeHint } from './lbug/lbug-config.js';
|
||||
import { escapeCypherString } from './lbug/cypher-escape.js';
|
||||
import {
|
||||
buildSearchIndexesOrDegrade,
|
||||
createSearchFTSIndexes,
|
||||
dropSearchFTSIndexes,
|
||||
initialiseSearchFTSStemmer,
|
||||
verifySearchFTSIndexes,
|
||||
} from './search/fts-indexes.js';
|
||||
@@ -57,10 +53,7 @@ import {
|
||||
checkpointOnce,
|
||||
type WalCheckpointDriver,
|
||||
} from './lbug/wal-checkpoint-driver.js';
|
||||
import {
|
||||
quarantineSidecarsForDirtyRecovery,
|
||||
inspectLbugSidecars,
|
||||
} from './lbug/sidecar-recovery.js';
|
||||
import { quarantineSidecarsForDirtyRecovery } from './lbug/sidecar-recovery.js';
|
||||
import type { EmbeddingIdentity } from './embeddings/embedding-identity.js';
|
||||
import {
|
||||
getStoragePaths,
|
||||
@@ -130,7 +123,6 @@ import { EMBEDDING_TABLE_NAME } from './lbug/schema.js';
|
||||
import { STALE_HASH_SENTINEL } from './lbug/schema.js';
|
||||
import { isSpringBeanCandidateSourceFile } from './ingestion/frameworks/spring/bean-catalog.js';
|
||||
import { SPRING_BEAN_INVENTORY_FEATURE } from './ingestion/frameworks/spring/analysis-features.js';
|
||||
import { SPRING_CONFIG_BINDINGS_FEATURE } from './ingestion/languages/java/analysis-features.js';
|
||||
import {
|
||||
CLASS_FRAMEWORK_ANNOTATIONS_FEATURE,
|
||||
findAnalysisFeatureMismatches,
|
||||
@@ -145,7 +137,6 @@ import {
|
||||
const ANALYSIS_FEATURES = [
|
||||
CLASS_FRAMEWORK_ANNOTATIONS_FEATURE,
|
||||
SPRING_BEAN_INVENTORY_FEATURE,
|
||||
SPRING_CONFIG_BINDINGS_FEATURE,
|
||||
] as const;
|
||||
|
||||
interface PersistedFrameworkAnnotationRow {
|
||||
@@ -654,11 +645,6 @@ export async function runFullAnalysis(
|
||||
// and are shared across branches (#2106 KTD7).
|
||||
const { storagePath } = getStoragePaths(repoPath);
|
||||
|
||||
// Start each analyze with a clean buffer-pool hint: any pre-pipeline DB open
|
||||
// (e.g. the embeddings-cache open) falls back to the default until the hint is
|
||||
// set from the built graph below, so a prior run's size can't leak in.
|
||||
setBufferPoolSizeHint(undefined);
|
||||
|
||||
// Clean up stale KuzuDB files from before the LadybugDB migration.
|
||||
const kuzuResult = await cleanupOldKuzuFiles(storagePath);
|
||||
if (kuzuResult.found && kuzuResult.needsReindex) {
|
||||
@@ -1334,54 +1320,6 @@ export async function runFullAnalysis(
|
||||
? diffFileHashes(newFileHashes, existingMeta!.fileHashes)
|
||||
: undefined;
|
||||
|
||||
// #2 atomic index publish: on a full rebuild, build the fresh DB at a temp
|
||||
// path and swap it over the live index in one rename at the very end, so a
|
||||
// concurrent MCP reader opening mid-build only ever sees the previous
|
||||
// complete index (never a wiped/half-built file) and a crash leaves the old
|
||||
// index intact. The whole build flows through the singleton connection, so
|
||||
// only initLbug/wipeLbugDbFiles below take the temp target.
|
||||
//
|
||||
// POSIX only: the common CLI/serve-worker analyze paths skip the native close
|
||||
// (closeLbugBeforeExit, #2264) and leave the build handle open at swap time.
|
||||
// POSIX renames an open file cleanly; a same-process open handle blocks the
|
||||
// rename on Windows. Windows keeps the current in-place behavior
|
||||
// (buildPath === lbugPath, no swap) until that is resolved (see §12/follow-up).
|
||||
const isFullRebuild = !(isIncremental && hashDiff);
|
||||
// Where the swap is allowed:
|
||||
// - POSIX renames an open file, so the usual skip-native-close (#2264) is
|
||||
// fine and the swap always applies.
|
||||
// - Windows can swap only when a real close is safe to release the build
|
||||
// handle before the rename — i.e. NOT a --pdg run (the #2264 destructor
|
||||
// crash). Unverified on Windows CI; falls back to in-place otherwise.
|
||||
const posixSwap = process.platform !== 'win32';
|
||||
// #2614 Windows: the forced real-close before the rename re-bets that #2264 is
|
||||
// --pdg-only, which is unproven (the CLI/worker skip the native close
|
||||
// UNCONDITIONALLY) and unverifiable without a Windows runner. Keep it opt-in
|
||||
// (GITNEXUS_ATOMIC_WINDOWS_SWAP=1) so the default Windows analyze stays on the
|
||||
// proven in-place path; enable it only to test the Windows swap.
|
||||
const windowsSwapOk =
|
||||
process.platform === 'win32' &&
|
||||
options.pdg !== true &&
|
||||
process.env.GITNEXUS_ATOMIC_WINDOWS_SWAP === '1';
|
||||
// Incremental atomicity copies the whole index into the temp before mutating
|
||||
// it, which negates incremental's speed premise — so it is opt-in
|
||||
// (GITNEXUS_ATOMIC_INCREMENTAL=1) pending a benchmark. Full rebuilds always
|
||||
// swap where the platform allows.
|
||||
const wantAtomicIncremental =
|
||||
isIncremental && !!hashDiff && process.env.GITNEXUS_ATOMIC_INCREMENTAL === '1';
|
||||
// #2614 F3: the copy-then-swap stages ONLY the main lbug file, so a live index
|
||||
// carrying an orphan .wal/.shadow (a silently-failed prior checkpoint) would
|
||||
// be copied incompletely and lose that delta. Only take the atomic path when
|
||||
// the live index is a consolidated single file; otherwise fall back to the
|
||||
// in-place writeback, which the next open replays correctly.
|
||||
const atomicIncremental =
|
||||
wantAtomicIncremental && (await inspectLbugSidecars(lbugPath)).kind === 'clean';
|
||||
if (wantAtomicIncremental && !atomicIncremental) {
|
||||
log('atomic-incremental: live index carries orphan sidecars — using in-place writeback');
|
||||
}
|
||||
const useAtomicSwap = (isFullRebuild || atomicIncremental) && (posixSwap || windowsSwapOk);
|
||||
const buildPath = useAtomicSwap ? `${lbugPath}.new` : lbugPath;
|
||||
|
||||
if (isIncremental && hashDiff) {
|
||||
log(
|
||||
`Incremental: changed=${hashDiff.changed.length}, ` +
|
||||
@@ -1404,14 +1342,6 @@ export async function runFullAnalysis(
|
||||
directWriteCount: hashDiff.toWrite.length,
|
||||
},
|
||||
});
|
||||
if (atomicIncremental) {
|
||||
// Stage the live index into the temp so the in-place delete/writeback
|
||||
// below mutates the COPY, and the end-of-run swap publishes it atomically.
|
||||
// Clear any stale temp first (a crashed run), then copy the (consolidated,
|
||||
// single-file) live index. Whole-file copy — hence opt-in.
|
||||
await wipeLbugDbFiles(buildPath);
|
||||
await fs.copyFile(lbugPath, buildPath);
|
||||
}
|
||||
} else {
|
||||
// Full rebuild path: wipe DB files first.
|
||||
// Set the dirty flag BEFORE the wipe whenever a prior meta exists,
|
||||
@@ -1441,27 +1371,10 @@ export async function runFullAnalysis(
|
||||
// valve below can never drift. Failures now throw a typed LbugWipeError
|
||||
// (ENOENT-verified removal) instead of silently letting initLbug reopen
|
||||
// a still-populated DB this run believes it wiped.
|
||||
//
|
||||
// With the atomic swap (POSIX), this wipes the TEMP build target
|
||||
// (`buildPath` = `<lbugPath>.new`, clearing any stragglers from a crashed
|
||||
// run) and leaves the live index untouched until the end-of-run swap. On
|
||||
// Windows buildPath === lbugPath, so this is the original in-place wipe.
|
||||
await wipeLbugDbFiles(buildPath);
|
||||
await wipeLbugDbFiles(lbugPath);
|
||||
}
|
||||
|
||||
// Size the buffer pool to the graph just built by the pipeline (a page cache
|
||||
// over the on-disk index, which scales with node/edge count) instead of the
|
||||
// fixed 2 GiB default, whose eager commit dominates large-repo analyze. The
|
||||
// size is clamped to [COPY-safety floor, default], so it only ever shrinks
|
||||
// the pool; env override / no-hint paths are unchanged. See
|
||||
// resolveBufferManagerSize / estimateBufferPool.
|
||||
setBufferPoolSizeHint(
|
||||
estimateBufferPool(pipelineResult.graph.nodeCount + pipelineResult.graph.relationshipCount),
|
||||
);
|
||||
|
||||
// Full rebuild (POSIX) builds into the temp `buildPath`; incremental and
|
||||
// Windows use `buildPath === lbugPath` in place.
|
||||
await initLbug(buildPath);
|
||||
await initLbug(lbugPath);
|
||||
|
||||
// Manual WAL checkpoint driver (#1741): periodically drain the WAL
|
||||
// from JS so the un-retriable native auto-checkpoint almost never
|
||||
@@ -1687,44 +1600,7 @@ export async function runFullAnalysis(
|
||||
// DB write plan changes here; fileHashes/meta bookkeeping is identical.
|
||||
// Thresholds + the AND-gate live in incremental/escalation-gate.ts.
|
||||
const writeFraction = effectiveWriteSet.size / Math.max(1, allFilePaths.length);
|
||||
// VECTOR gate (#2623) — load the extension BEFORE a single embedding row
|
||||
// is touched. `deleteNodesForFiles` below opens with the CodeEmbedding
|
||||
// join-delete, and LadybugDB refuses all DML on a table carrying its HNSW
|
||||
// index unless VECTOR is loaded on this connection; nothing else on this
|
||||
// path loads it until Phase 4, so every incremental run over a DB that
|
||||
// already built `code_embedding_idx` died here. Same seam the FTS drop
|
||||
// occupies at the head of this branch (#2589): index lifecycle first,
|
||||
// then rows. UNCONDITIONAL — not gated on `shouldGenerateEmbeddings` —
|
||||
// because a DB carrying the index from an earlier `--embeddings` run hits
|
||||
// the identical wall on a plain incremental run.
|
||||
//
|
||||
// When VECTOR genuinely cannot load, the table is immutable (the index
|
||||
// cannot be dropped without the extension either), so surgery is
|
||||
// impossible: fall through to the escalation valve's wipe-and-COPY plan,
|
||||
// which rebuilds the DB files outright and needs no embedding-row DML.
|
||||
const embeddingRowDmlSafe = await ensureEmbeddingRowDmlSafe();
|
||||
if (!embeddingRowDmlSafe && cachedEmbeddings.length === 0) {
|
||||
// The escalation below WIPES the DB files, and Phase 3.5 restores
|
||||
// embedding rows from `cachedEmbeddings` — which is only populated when
|
||||
// `deriveEmbeddingMode` saw `meta.stats.embeddings > 0`. A DB whose meta
|
||||
// under-reports its embeddings (meta restored from an older run, or a
|
||||
// count that never got stamped) would therefore have every vector
|
||||
// silently destroyed by a rebuild it did not ask for. Read them now,
|
||||
// while the DB is still intact — a plain MATCH, which needs no VECTOR
|
||||
// extension. Rows whose owning node is gone are dropped by Phase 3.5's
|
||||
// live-graph filter, exactly as on any other wiped path.
|
||||
const rescued = await loadCachedEmbeddings();
|
||||
if (rescued.embeddings.length > 0) {
|
||||
cachedEmbeddings = rescued.embeddings;
|
||||
cachedEmbeddingNodeIds = rescued.embeddingNodeIds;
|
||||
log(
|
||||
`Preserving ${rescued.embeddings.length} embedding row(s) across the forced rebuild ` +
|
||||
`(the index metadata did not account for them).`,
|
||||
);
|
||||
}
|
||||
}
|
||||
if (
|
||||
!embeddingRowDmlSafe ||
|
||||
shouldEscalateIncrementalWrite(
|
||||
filesToDelete.length,
|
||||
effectiveWriteSet.size,
|
||||
@@ -1733,20 +1609,13 @@ export async function runFullAnalysis(
|
||||
) {
|
||||
escalatedFullWrite = true;
|
||||
log(
|
||||
!embeddingRowDmlSafe
|
||||
? `Incremental: the ${EMBEDDING_TABLE_NAME} vector index exists but the VECTOR ` +
|
||||
`extension could not be loaded, so embedding rows cannot be rewritten in place — ` +
|
||||
`switching to a full DB write (wipe + bulk COPY) for this run. Semantic search ` +
|
||||
`falls back to exact scan until VECTOR is available; run \`gitnexus doctor\` for ` +
|
||||
`live extension status, or set GITNEXUS_LBUG_EXTENSION_INSTALL=auto to allow one ` +
|
||||
`bounded install attempt.`
|
||||
: `Incremental: effective write set covers ${effectiveWriteSet.size}/${allFilePaths.length} ` +
|
||||
// Display clamp only (predicate unchanged): BFS-found deleted
|
||||
// importers can push the numerator past the CURRENT file list, so
|
||||
// the raw fraction can exceed 1 — see the population-mismatch note
|
||||
// on shouldEscalateIncrementalWrite (tri-review 4669518496).
|
||||
`files (${Math.min(100, Math.round(writeFraction * 100))}%) — switching to a full DB write ` +
|
||||
`(wipe + bulk COPY) for this run; file-level incremental bookkeeping is unaffected.`,
|
||||
`Incremental: effective write set covers ${effectiveWriteSet.size}/${allFilePaths.length} ` +
|
||||
// Display clamp only (predicate unchanged): BFS-found deleted
|
||||
// importers can push the numerator past the CURRENT file list, so
|
||||
// the raw fraction can exceed 1 — see the population-mismatch note
|
||||
// on shouldEscalateIncrementalWrite (tri-review 4669518496).
|
||||
`files (${Math.min(100, Math.round(writeFraction * 100))}%) — switching to a full DB write ` +
|
||||
`(wipe + bulk COPY) for this run; file-level incremental bookkeeping is unaffected.`,
|
||||
);
|
||||
// toWriteCount: 0 is the established full-path dirty-flag sentinel;
|
||||
// the real counters ride along for crash diagnostics.
|
||||
@@ -1767,8 +1636,8 @@ export async function runFullAnalysis(
|
||||
// to replace wholesale.
|
||||
await walCheckpointDriver.stop();
|
||||
await closeLbug();
|
||||
await wipeLbugDbFiles(buildPath);
|
||||
await initLbug(buildPath);
|
||||
await wipeLbugDbFiles(lbugPath);
|
||||
await initLbug(lbugPath);
|
||||
walCheckpointDriver = startWalCheckpointDriver();
|
||||
await loadGraphToLbug(pipelineResult.graph, pipelineResult.repoPath, storagePath, (msg) => {
|
||||
lbugMsgCount++;
|
||||
@@ -1776,20 +1645,7 @@ export async function runFullAnalysis(
|
||||
progress('lbug', pct, msg);
|
||||
});
|
||||
} else {
|
||||
// 1a. Drop every FTS index before touching a single row (#2589).
|
||||
// `deleteNodesForFiles` below DETACH DELETEs rows out of tables
|
||||
// that otherwise still carry the FTS index built at the end of
|
||||
// the PREVIOUS analyze run — Phase 3 doesn't drop+rebuild it
|
||||
// until well after this delete completes. LadybugDB's FTS
|
||||
// extension is not proven to survive DML against an indexed
|
||||
// table (its own docs never demonstrate it), and that ordering
|
||||
// is exactly what produced "FTS index 'file_fts' is
|
||||
// inconsistent: term is missing during delete". Dropping first
|
||||
// removes the hazard outright; Phase 3's createSearchFTSIndexes
|
||||
// rebuilds every index from the final row set regardless, so
|
||||
// this is a no-op on its own drop step there.
|
||||
await dropSearchFTSIndexes();
|
||||
// 1b. Remove the write set's existing rows — batched (#2409): one
|
||||
// 1a. Remove the write set's existing rows — batched (#2409): one
|
||||
// DETACH DELETE per table per 200-file chunk. The former per-file
|
||||
// loop issued a count + delete per table per FILE — ~13k
|
||||
// single-row write transactions on a ~700-file write set — which
|
||||
@@ -2112,8 +1968,8 @@ export async function runFullAnalysis(
|
||||
// the case a naive gate would leave index-less again.
|
||||
// buildVectorIndex carries its own extension-policy gate and
|
||||
// warn-on-failure; the boolean feeds semanticMode so the finalize stamp
|
||||
// reflects the DB's ACTUAL state even when recreation fails (extension
|
||||
// unavailable → 'exact-scan').
|
||||
// reflects the DB's ACTUAL state even when recreation fails (win32 /
|
||||
// extension unavailable → 'exact-scan').
|
||||
const dbWasWiped = !isIncremental || escalatedFullWrite;
|
||||
if (restoredEmbeddingCount > 0 && dbWasWiped && embeddingSkipped) {
|
||||
// Re-import at the seam rather than thread a mutable capture from
|
||||
@@ -2369,11 +2225,7 @@ export async function runFullAnalysis(
|
||||
// inside the resolver, and a mismatch leaves the dirty flag intact so the
|
||||
// next run takes the established full-recovery path.
|
||||
meta.runnerIdentity = finalizeAnalyzerRunnerIdentity(import.meta.url, runnerIdentity);
|
||||
// #2614 F1: the freshness stamp (saveMeta) is written AFTER the atomic swap
|
||||
// below — never here — so a concurrent MCP reader can't observe
|
||||
// meta.indexedAt = T_new while lbugPath still resolves to the pre-swap
|
||||
// inode (which latched the reader on the stale index permanently). The meta
|
||||
// object is fully computed at this point; only its write is deferred.
|
||||
await saveMeta(metaDir, meta);
|
||||
|
||||
// Persist the incremental parse cache for the next run. Wraps in
|
||||
// try/catch so a cache-write failure never breaks an otherwise
|
||||
@@ -2511,59 +2363,7 @@ export async function runFullAnalysis(
|
||||
// LadybugDB destructor double-free after --pdg writes — closeLbugBeforeExit
|
||||
// CHECKPOINTs for durability then leaves the handles for process exit to
|
||||
// reclaim (#2264). Long-lived callers close for real.
|
||||
//
|
||||
// On Windows a swap must release the build handle before the rename (a
|
||||
// same-process open file can't be renamed), so it forces a real close —
|
||||
// safe because windowsSwapOk excludes --pdg (the #2264 case). POSIX renames
|
||||
// an open file, so it keeps the skip-native-close there.
|
||||
const forceRealCloseForSwap = useAtomicSwap && process.platform === 'win32';
|
||||
await (options.skipNativeCloseOnExit && !forceRealCloseForSwap
|
||||
? closeLbugBeforeExit()
|
||||
: closeLbug());
|
||||
|
||||
// #2 atomic publish: the fresh index was built at buildPath (a full rebuild,
|
||||
// or an opt-in atomic incremental that copied the live index in first). Swap
|
||||
// it over the live lbugPath in one rename so an MCP reader that opened
|
||||
// mid-build only ever saw the previous complete index — never a wiped/
|
||||
// half-built file. The close above checkpoint-consolidated buildPath to a
|
||||
// single file (no .wal), so the rename publishes a complete index; a reader
|
||||
// holding the old inode keeps a consistent stale snapshot until the pool
|
||||
// re-opens onto the new one (the pool staleness invalidation). Runs only on
|
||||
// success — a thrown error skips this, leaving the live index intact and the
|
||||
// temp build to be cleared by the next run's wipe.
|
||||
// Only publish if the build actually produced a DB at buildPath. A
|
||||
// degenerate run (empty repo, or a mocked pipeline that never opened the
|
||||
// store) leaves nothing to swap — skip rather than throw ENOENT.
|
||||
const builtDbExists = useAtomicSwap
|
||||
? await fs.stat(buildPath).then(
|
||||
() => true,
|
||||
() => false,
|
||||
)
|
||||
: false;
|
||||
if (useAtomicSwap && builtDbExists) {
|
||||
await retryRename(buildPath, lbugPath);
|
||||
// Clear any sidecars orphaned beside the replaced file. A cleanly-closed
|
||||
// prior index has none; a crashed one could, and it would be replay
|
||||
// poison next to the freshly published index. Best-effort.
|
||||
for (const suffix of ['.wal', '.shadow', '.wal.checkpoint'] as const) {
|
||||
await fs.rm(`${lbugPath}${suffix}`, { force: true }).catch(() => {});
|
||||
}
|
||||
// #2614 F4: if the final checkpoint silently failed, the build may still
|
||||
// carry a residual .wal/.shadow under the temp name. MOVE it beside the
|
||||
// published index (not orphan/delete it) so the next open replays the
|
||||
// delta, rather than leaving it under a name LadybugDB never reconciles.
|
||||
for (const suffix of ['.wal', '.shadow'] as const) {
|
||||
await fs.rename(`${buildPath}${suffix}`, `${lbugPath}${suffix}`).catch(() => {});
|
||||
}
|
||||
}
|
||||
|
||||
// #2614 F1: stamp the freshness metadata now that the index is published.
|
||||
// When meta.indexedAt becomes visible, lbugPath already resolves to the new
|
||||
// inode, so a reader reiniting on the stamp opens the fresh graph rather
|
||||
// than latching on the old one. Leaving the dirty flag set across the swap
|
||||
// is a crash-safety improvement: a failed swap leaves the previous index
|
||||
// live and the next run recovers via the full-rebuild path.
|
||||
await saveMeta(metaDir, meta);
|
||||
await (options.skipNativeCloseOnExit ? closeLbugBeforeExit() : closeLbug());
|
||||
|
||||
progress('done', 100, 'Done');
|
||||
|
||||
|
||||
@@ -121,20 +121,6 @@ export function getSearchFTSStemmer(): string {
|
||||
return resolvedStemmer ?? resolveFTSStemmer();
|
||||
}
|
||||
|
||||
/**
|
||||
* Drop every configured FTS index (no-op per index when absent or unloadable
|
||||
* — `dropFTSIndex` tolerates both). Callable ahead of any DML that mutates an
|
||||
* FTS-indexed table's rows: LadybugDB's FTS extension is not proven to
|
||||
* survive a DETACH DELETE against a table that still carries a live index
|
||||
* from a prior run (#2589) — dropping first removes that hazard entirely,
|
||||
* regardless of whether it also fixed a specific native inconsistency.
|
||||
*/
|
||||
export async function dropSearchFTSIndexes(): Promise<void> {
|
||||
for (const { table, indexName } of FTS_INDEXES) {
|
||||
await dropFTSIndex(table, indexName);
|
||||
}
|
||||
}
|
||||
|
||||
export async function createSearchFTSIndexes(
|
||||
options?: CreateSearchFTSIndexesOptions,
|
||||
): Promise<void> {
|
||||
|
||||
@@ -15,8 +15,6 @@ import {
|
||||
executeParameterized,
|
||||
closeLbug,
|
||||
isLbugReady,
|
||||
statDbIdentity,
|
||||
dbIdentityChanged,
|
||||
} from '../../core/lbug/pool-adapter.js';
|
||||
import { queryClassBeanMetadata } from './bean-metadata.js';
|
||||
import { isValidQueryParams } from '../../core/lbug/query-params.js';
|
||||
@@ -61,7 +59,10 @@ import {
|
||||
type ExactEmbeddingRow,
|
||||
} from '../../core/embeddings/exact-search.js';
|
||||
import { EMBEDDING_TABLE_NAME, EMBEDDING_INDEX_NAME } from '../../core/lbug/schema.js';
|
||||
import { getExactScanLimit } from '../../core/platform/capabilities.js';
|
||||
import {
|
||||
getExactScanLimit,
|
||||
isVectorExtensionSupportedByPlatform,
|
||||
} from '../../core/platform/capabilities.js';
|
||||
import { PhaseTimer } from '../../core/search/phase-timer.js';
|
||||
import { ftsDegradedWarning } from '../../core/search/fts-indexes.js';
|
||||
import {
|
||||
@@ -724,12 +725,6 @@ export class LocalBackend {
|
||||
// not persist across calls and the staleness check would reinit forever
|
||||
// (#2106).
|
||||
private lastObservedIndexedAt: Map<string, string> = new Map();
|
||||
// #2614 F1: file identity of the lbug the pool last opened. An atomic swap or
|
||||
// an in-place incremental changes the inode; reiniting on that reinit-covers
|
||||
// the window where meta.indexedAt hasn't caught up (and the incremental case),
|
||||
// so a rebuilt index is never served stale even when the stamp looks current.
|
||||
private lastObservedDbIdentity: Map<string, Awaited<ReturnType<typeof statDbIdentity>>> =
|
||||
new Map();
|
||||
private groupToolSvc: GroupService | null = null;
|
||||
/**
|
||||
* One-shot stderr warnings for sibling-clone drift, keyed by
|
||||
@@ -1065,7 +1060,6 @@ export class LocalBackend {
|
||||
this.initializedRepos.delete(key);
|
||||
this.lastStalenessCheck.delete(key);
|
||||
this.lastObservedIndexedAt.delete(key);
|
||||
this.lastObservedDbIdentity.delete(key);
|
||||
this.reinitPromises.delete(key);
|
||||
closeLbug(key).catch(() => {});
|
||||
}
|
||||
@@ -1469,40 +1463,22 @@ export class LocalBackend {
|
||||
// Reading the flat meta for a branch handle would compare the branch
|
||||
// index's indexedAt against the primary's and thrash the pool (#2106).
|
||||
const meta = await loadMeta(path.dirname(repo.lbugPath));
|
||||
if (!meta) return;
|
||||
// Compare against the last indexedAt OBSERVED for this pool (keyed by
|
||||
// lbugPath), not the handle's — branch handles are fresh spreads so a
|
||||
// handle mutation would not persist and would reinit on every check.
|
||||
const observed = this.lastObservedIndexedAt.get(poolKey) ?? repo.indexedAt;
|
||||
const stampChanged = !!meta?.indexedAt && meta.indexedAt !== observed;
|
||||
// #2614 F1: also reinit on a file-identity change. An atomic swap (or an
|
||||
// in-place incremental) changes the lbug inode; keying only on
|
||||
// meta.indexedAt let a reader that reinited inside the pre-swap window
|
||||
// latch on the old inode forever (its stamp already == meta.indexedAt).
|
||||
const currentIdentity = await statDbIdentity(repo.lbugPath);
|
||||
const identityChanged = dbIdentityChanged(
|
||||
this.lastObservedDbIdentity.get(poolKey) ?? null,
|
||||
currentIdentity,
|
||||
);
|
||||
if (stampChanged || identityChanged) {
|
||||
// Index was rebuilt/swapped — DELEGATE the close/reopen to the pool's
|
||||
// initLbug, which refuses to evict (and close the shared Database)
|
||||
// while a query is in flight (its checkedOut>0 guard). Calling
|
||||
// closeLbug directly here bypassed that guard and could close a
|
||||
// Database mid-query — a native use-after-free (#2614). Wrap in
|
||||
// reinitPromises to serialize concurrent detectors.
|
||||
if (meta.indexedAt && meta.indexedAt !== observed) {
|
||||
// Index was rebuilt — close stale connection and re-init.
|
||||
// Wrap in reinitPromises to prevent TOCTOU race where concurrent
|
||||
// callers both detect staleness and double-close the pool.
|
||||
const reinit = (async () => {
|
||||
try {
|
||||
// Advance the observed stamp regardless: a stamp change with an
|
||||
// unchanged file must not re-trigger on every check.
|
||||
if (meta?.indexedAt) this.lastObservedIndexedAt.set(poolKey, meta.indexedAt);
|
||||
const reopened = await initLbug(poolKey, repo.lbugPath);
|
||||
// Advance the observed IDENTITY only when the pool actually rolled
|
||||
// over. If a query was in flight, initLbug served the current
|
||||
// handle and returned false; leaving the identity divergent
|
||||
// re-triggers the reopen on a later idle check instead of latching.
|
||||
if (reopened) {
|
||||
this.lastObservedDbIdentity.set(poolKey, await statDbIdentity(repo.lbugPath));
|
||||
}
|
||||
await closeLbug(poolKey);
|
||||
this.initializedRepos.delete(poolKey);
|
||||
this.lastObservedIndexedAt.set(poolKey, meta.indexedAt);
|
||||
await initLbug(poolKey, repo.lbugPath);
|
||||
this.initializedRepos.add(poolKey);
|
||||
} finally {
|
||||
this.reinitPromises.delete(poolKey);
|
||||
}
|
||||
@@ -1521,7 +1497,6 @@ export class LocalBackend {
|
||||
await initLbug(poolKey, repo.lbugPath);
|
||||
this.initializedRepos.add(poolKey);
|
||||
this.lastObservedIndexedAt.set(poolKey, repo.indexedAt);
|
||||
this.lastObservedDbIdentity.set(poolKey, await statDbIdentity(repo.lbugPath));
|
||||
} catch (err: any) {
|
||||
// If lock error, mark as not initialized so next call retries
|
||||
this.initializedRepos.delete(poolKey);
|
||||
@@ -2416,16 +2391,10 @@ export class LocalBackend {
|
||||
string,
|
||||
{ distance: number; chunkIndex: number; startLine: number; endLine: number }
|
||||
>();
|
||||
// Always TRY the vector lane — no platform gate. LadybugDB ships the
|
||||
// VECTOR extension for every supported platform, Windows included
|
||||
// (#2623 follow-up; the old `platform !== 'win32'` gate was stale), so
|
||||
// whether the index is queryable is a per-machine runtime fact. The
|
||||
// catch below is the fallback: any failure (extension unloadable, index
|
||||
// absent, older DB) degrades to the exact scan with a once-per-backend
|
||||
// diagnostic instead of being silently swallowed.
|
||||
try {
|
||||
bestChunks = await collectBestChunks(limit, async (fetchLimit) => {
|
||||
const vectorQuery = `
|
||||
if (isVectorExtensionSupportedByPlatform()) {
|
||||
try {
|
||||
bestChunks = await collectBestChunks(limit, async (fetchLimit) => {
|
||||
const vectorQuery = `
|
||||
CALL QUERY_VECTOR_INDEX('${EMBEDDING_TABLE_NAME}', '${EMBEDDING_INDEX_NAME}',
|
||||
CAST(${queryVecStr} AS FLOAT[${dims}]), ${fetchLimit})
|
||||
YIELD node AS emb, distance
|
||||
@@ -2436,27 +2405,27 @@ export class LocalBackend {
|
||||
ORDER BY distance
|
||||
`;
|
||||
|
||||
const embResults = await executeQuery(repo.lbugPath, vectorQuery);
|
||||
return embResults.map((row) => ({
|
||||
nodeId: row.nodeId ?? row[0],
|
||||
chunkIndex: row.chunkIndex ?? row[1] ?? 0,
|
||||
startLine: row.startLine ?? row[2] ?? 0,
|
||||
endLine: row.endLine ?? row[3] ?? 0,
|
||||
distance: row.distance ?? row[4],
|
||||
}));
|
||||
});
|
||||
} catch (err) {
|
||||
bestChunks = new Map();
|
||||
if (!this.warnedVectorUnsupported) {
|
||||
// Rare diagnostic: surface why semantic search fell back to the
|
||||
// exact scan. Emitted once per `LocalBackend` instance lifetime to
|
||||
// avoid noisy stderr on hot semantic-search paths (DoD §2.8).
|
||||
this.warnedVectorUnsupported = true;
|
||||
logger.warn(
|
||||
{ err },
|
||||
'GitNexus [query:vector]: vector index query failed; using exact scan fallback',
|
||||
);
|
||||
const embResults = await executeQuery(repo.lbugPath, vectorQuery);
|
||||
return embResults.map((row) => ({
|
||||
nodeId: row.nodeId ?? row[0],
|
||||
chunkIndex: row.chunkIndex ?? row[1] ?? 0,
|
||||
startLine: row.startLine ?? row[2] ?? 0,
|
||||
endLine: row.endLine ?? row[3] ?? 0,
|
||||
distance: row.distance ?? row[4],
|
||||
}));
|
||||
});
|
||||
} catch {
|
||||
bestChunks = new Map();
|
||||
}
|
||||
} else if (!this.warnedVectorUnsupported) {
|
||||
// Rare diagnostic: surface why we fell back to the exact scan path so
|
||||
// operators can see at a glance that VECTOR is disabled by platform
|
||||
// policy. Emitted once per `LocalBackend` instance lifetime to avoid
|
||||
// noisy stderr on hot semantic-search paths (DoD §2.8).
|
||||
this.warnedVectorUnsupported = true;
|
||||
logger.warn(
|
||||
'GitNexus [query:vector]: VECTOR extension not supported on this platform; using exact scan fallback',
|
||||
);
|
||||
}
|
||||
|
||||
if (bestChunks.size === 0) {
|
||||
@@ -4392,31 +4361,44 @@ export class LocalBackend {
|
||||
return { error: 'New name is the same as the current name.' };
|
||||
}
|
||||
|
||||
// Steps 2+3: Determine the set of files the apply step will rewrite, then
|
||||
// enumerate every occurrence in each. The apply step (Step 4) does a
|
||||
// whole-file `\boldName\b` global replace on every file in `changes`, so the
|
||||
// reported edit list MUST enumerate every matching line in every such file —
|
||||
// otherwise the preview under-reports what lands, and the same partial list
|
||||
// comes back after apply (#2605). Building `changes` from one file set makes
|
||||
// the preview enumerate exactly the files the apply loop rewrites, using the
|
||||
// same word-boundary regex. (This is per-call consistency; the apply loop
|
||||
// still re-reads each file, so an external write landing between preview and
|
||||
// apply is a pre-existing gap this method does not lock against.)
|
||||
type RenameEdit = {
|
||||
line: number;
|
||||
old_text: string;
|
||||
new_text: string;
|
||||
confidence: 'graph' | 'text_search';
|
||||
// Step 2: Collect edits from graph (high confidence)
|
||||
const changes = new Map<string, { file_path: string; edits: any[] }>();
|
||||
|
||||
const addEdit = (
|
||||
filePath: string,
|
||||
line: number,
|
||||
oldText: string,
|
||||
newText: string,
|
||||
confidence: string,
|
||||
) => {
|
||||
if (!changes.has(filePath)) {
|
||||
changes.set(filePath, { file_path: filePath, edits: [] });
|
||||
}
|
||||
changes.get(filePath)!.edits.push({ line, old_text: oldText, new_text: newText, confidence });
|
||||
};
|
||||
const escapedOldName = oldName.replace(/[.*+?^${}()|[\]\\]/g, '\\$&');
|
||||
|
||||
// Classify each file to rewrite by how it was discovered. Definition and
|
||||
// graph-ref files carry graph confidence; files found only by text search
|
||||
// carry text_search confidence. A graph-classified file is never downgraded.
|
||||
const fileConfidence = new Map<string, 'graph' | 'text_search'>();
|
||||
|
||||
if (sym.filePath) {
|
||||
fileConfidence.set(sym.filePath, 'graph');
|
||||
// The definition itself
|
||||
if (sym.filePath && sym.startLine) {
|
||||
try {
|
||||
const content = await fs.readFile(assertSafePath(sym.filePath), 'utf-8');
|
||||
const lines = content.split('\n');
|
||||
const lineIdx = sym.startLine - 1;
|
||||
if (lineIdx >= 0 && lineIdx < lines.length && lines[lineIdx].includes(oldName)) {
|
||||
const defRegex = new RegExp(
|
||||
`\\b${oldName.replace(/[.*+?^${}()|[\]\\]/g, '\\$&')}\\b`,
|
||||
'g',
|
||||
);
|
||||
addEdit(
|
||||
sym.filePath,
|
||||
sym.startLine,
|
||||
lines[lineIdx].trim(),
|
||||
lines[lineIdx].replace(defRegex, new_name).trim(),
|
||||
'graph',
|
||||
);
|
||||
}
|
||||
} catch (e) {
|
||||
logQueryError('rename:read-definition', e);
|
||||
}
|
||||
}
|
||||
|
||||
// All incoming refs from graph (callers, importers, etc.)
|
||||
@@ -4426,13 +4408,44 @@ export class LocalBackend {
|
||||
...(lookupResult.incoming.extends || []),
|
||||
...(lookupResult.incoming.implements || []),
|
||||
];
|
||||
|
||||
let graphEdits = changes.size > 0 ? 1 : 0; // count definition edit
|
||||
|
||||
for (const ref of allIncoming) {
|
||||
if (ref.filePath) {
|
||||
fileConfidence.set(ref.filePath, 'graph');
|
||||
if (!ref.filePath) continue;
|
||||
try {
|
||||
const content = await fs.readFile(assertSafePath(ref.filePath), 'utf-8');
|
||||
const lines = content.split('\n');
|
||||
for (let i = 0; i < lines.length; i++) {
|
||||
if (lines[i].includes(oldName)) {
|
||||
addEdit(
|
||||
ref.filePath,
|
||||
i + 1,
|
||||
lines[i].trim(),
|
||||
lines[i]
|
||||
.replace(
|
||||
new RegExp(`\\b${oldName.replace(/[.*+?^${}()|[\]\\]/g, '\\$&')}\\b`, 'g'),
|
||||
new_name,
|
||||
)
|
||||
.trim(),
|
||||
'graph',
|
||||
);
|
||||
graphEdits++;
|
||||
break; // one edit per file from graph refs
|
||||
}
|
||||
}
|
||||
} catch (e) {
|
||||
logQueryError('rename:read-ref', e);
|
||||
}
|
||||
}
|
||||
|
||||
// Text search for files the graph might have missed entirely.
|
||||
// Step 3: Text search for refs the graph might have missed
|
||||
let astSearchEdits = 0;
|
||||
const graphFiles = new Set(
|
||||
[sym.filePath, ...allIncoming.map((r) => r.filePath)].filter(Boolean),
|
||||
);
|
||||
|
||||
// Simple text search across the repo for the old name (in files not already covered by graph)
|
||||
try {
|
||||
const { execFileSync } = await import('child_process');
|
||||
const rgArgs = [
|
||||
@@ -4459,98 +4472,67 @@ export class LocalBackend {
|
||||
|
||||
for (const file of files) {
|
||||
const normalizedFile = file.replace(/\\/g, '/').replace(/^\.\//, '');
|
||||
// Never downgrade a graph-classified file to text_search.
|
||||
if (!fileConfidence.has(normalizedFile)) {
|
||||
fileConfidence.set(normalizedFile, 'text_search');
|
||||
if (graphFiles.has(normalizedFile)) continue; // already covered by graph
|
||||
|
||||
try {
|
||||
const content = await fs.readFile(assertSafePath(normalizedFile), 'utf-8');
|
||||
const lines = content.split('\n');
|
||||
const regex = new RegExp(`\\b${oldName.replace(/[.*+?^${}()|[\]\\]/g, '\\$&')}\\b`, 'g');
|
||||
for (let i = 0; i < lines.length; i++) {
|
||||
regex.lastIndex = 0;
|
||||
if (regex.test(lines[i])) {
|
||||
regex.lastIndex = 0;
|
||||
addEdit(
|
||||
normalizedFile,
|
||||
i + 1,
|
||||
lines[i].trim(),
|
||||
lines[i].replace(regex, new_name).trim(),
|
||||
'text_search',
|
||||
);
|
||||
astSearchEdits++;
|
||||
}
|
||||
}
|
||||
} catch (e) {
|
||||
logQueryError('rename:text-search-read', e);
|
||||
}
|
||||
}
|
||||
} catch (e) {
|
||||
logQueryError('rename:ripgrep', e);
|
||||
}
|
||||
|
||||
// Enumerate every `\boldName\b` line in each file to rewrite, so the previewed
|
||||
// file set is exactly the set the apply loop below rewrites. A file with no
|
||||
// matching line is dropped (apply would write nothing to it). `wordTest`
|
||||
// (non-global) probes each line; `wordReplace` (global) rewrites it and is
|
||||
// reused by the apply loop — compiled once each rather than once per line,
|
||||
// and one escaping formula serves both passes.
|
||||
const wordTest = new RegExp(`\\b${escapedOldName}\\b`);
|
||||
const wordReplace = new RegExp(`\\b${escapedOldName}\\b`, 'g');
|
||||
const changes = new Map<string, { file_path: string; edits: RenameEdit[] }>();
|
||||
// Step 4: Apply or preview
|
||||
const allChanges = Array.from(changes.values());
|
||||
const totalEdits = allChanges.reduce((sum, c) => sum + c.edits.length, 0);
|
||||
|
||||
for (const [filePath, confidence] of fileConfidence) {
|
||||
try {
|
||||
const content = await fs.readFile(assertSafePath(filePath), 'utf-8');
|
||||
const lines = content.split('\n');
|
||||
const edits: RenameEdit[] = [];
|
||||
for (let i = 0; i < lines.length; i++) {
|
||||
if (!wordTest.test(lines[i])) {
|
||||
continue;
|
||||
}
|
||||
edits.push({
|
||||
line: i + 1,
|
||||
old_text: lines[i].trim(),
|
||||
new_text: lines[i].replace(wordReplace, new_name).trim(),
|
||||
confidence,
|
||||
});
|
||||
}
|
||||
if (edits.length > 0) {
|
||||
changes.set(filePath, { file_path: filePath, edits });
|
||||
}
|
||||
} catch (e) {
|
||||
logQueryError('rename:enumerate', e);
|
||||
}
|
||||
}
|
||||
|
||||
// Step 4: Apply or preview.
|
||||
const failedFiles: string[] = [];
|
||||
if (!dry_run) {
|
||||
for (const change of changes.values()) {
|
||||
// Apply edits to files
|
||||
for (const change of allChanges) {
|
||||
try {
|
||||
const fullPath = assertSafePath(change.file_path);
|
||||
const content = await fs.readFile(fullPath, 'utf-8');
|
||||
await fs.writeFile(fullPath, content.replace(wordReplace, new_name), 'utf-8');
|
||||
let content = await fs.readFile(fullPath, 'utf-8');
|
||||
const regex = new RegExp(`\\b${oldName.replace(/[.*+?^${}()|[\]\\]/g, '\\$&')}\\b`, 'g');
|
||||
content = content.replace(regex, new_name);
|
||||
await fs.writeFile(fullPath, content, 'utf-8');
|
||||
} catch (e) {
|
||||
// A swallowed write failure must not be reported as success (#2283):
|
||||
// record the file so the result degrades to 'partial'.
|
||||
// A swallowed write failure must not be reported as a full success
|
||||
// (#2283): record the file so the result can degrade to 'partial'
|
||||
// with the unwritten files listed, rather than masquerading as done.
|
||||
logQueryError('rename:apply-edit', e);
|
||||
failedFiles.push(change.file_path);
|
||||
}
|
||||
}
|
||||
// A file whose write threw did not land, so drop its edits from the
|
||||
// reported result — total_edits/changes must describe what actually
|
||||
// reached disk, not what was attempted (#2605: the report matches reality
|
||||
// even on partial failure). failed_files still names every dropped file.
|
||||
for (const f of failedFiles) {
|
||||
changes.delete(f);
|
||||
}
|
||||
}
|
||||
|
||||
// Counts derive from the reported set (dry-run: every enumerated file;
|
||||
// apply: only files that landed), so the graph/text_search split always
|
||||
// sums to total_edits and never overstates a partial apply.
|
||||
const reported = Array.from(changes.values());
|
||||
let graphEdits = 0;
|
||||
let astSearchEdits = 0;
|
||||
for (const change of reported) {
|
||||
for (const edit of change.edits) {
|
||||
if (edit.confidence === 'graph') {
|
||||
graphEdits++;
|
||||
} else {
|
||||
astSearchEdits++;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return {
|
||||
status: failedFiles.length > 0 ? 'partial' : 'success',
|
||||
old_name: oldName,
|
||||
new_name,
|
||||
files_affected: reported.length,
|
||||
total_edits: graphEdits + astSearchEdits,
|
||||
files_affected: allChanges.length,
|
||||
total_edits: totalEdits,
|
||||
graph_edits: graphEdits,
|
||||
text_search_edits: astSearchEdits,
|
||||
changes: reported,
|
||||
changes: allChanges,
|
||||
applied: !dry_run,
|
||||
...(failedFiles.length > 0 && { failed_files: failedFiles }),
|
||||
};
|
||||
|
||||
@@ -1,7 +1,6 @@
|
||||
import { execFileSync, execSync } from 'child_process';
|
||||
import { statSync } from 'fs';
|
||||
import path from 'path';
|
||||
import os from 'os';
|
||||
|
||||
// Git utilities for repository detection, commit tracking, and diff analysis
|
||||
|
||||
@@ -210,84 +209,6 @@ export const getCanonicalRepoRoot = (fromPath: string): string | null => {
|
||||
}
|
||||
};
|
||||
|
||||
// getGitInfoExcludePath/getCoreExcludesFilePath are called once per repo
|
||||
// PER language/contract extractor during group sync (#2606) — an N-repo
|
||||
// group fans out to 6+ extractors each calling these, so an uncached
|
||||
// execSync per call turns into O(extractors × repos) blocking subprocess
|
||||
// spawns. Both resolve to the same value for the same fromPath for the
|
||||
// life of the process (git config/exclude files don't change mid-run), so
|
||||
// memoize by fromPath. ponytail: process-lifetime cache, never invalidated
|
||||
// — fine for one-shot CLI runs; the long-lived MCP server would need a
|
||||
// TTL or explicit invalidation if a user edits core.excludesFile mid-session.
|
||||
const gitInfoExcludePathCache = new Map<string, string | null>();
|
||||
const coreExcludesFilePathCache = new Map<string, string>();
|
||||
|
||||
/**
|
||||
* Path to the repo's `$GIT_COMMON_DIR/info/exclude` file — git's own
|
||||
* per-repo, untracked exclude list (same tier as `.gitignore` in
|
||||
* precedence, but never committed, so it works even when the caller has
|
||||
* no write access to the repo's tracked content). Shared across every
|
||||
* linked worktree of a repo, matching git's own resolution (#2606).
|
||||
*
|
||||
* Returns `null` when `fromPath` is not inside a git repository or `git`
|
||||
* is unavailable; callers should treat that the same as "no file".
|
||||
*/
|
||||
export const getGitInfoExcludePath = (fromPath: string): string | null => {
|
||||
const cached = gitInfoExcludePathCache.get(fromPath);
|
||||
if (cached !== undefined) return cached;
|
||||
|
||||
let result: string | null;
|
||||
try {
|
||||
const commonDir = chompGitOutput(
|
||||
execSync('git rev-parse --path-format=absolute --git-common-dir', {
|
||||
cwd: fromPath,
|
||||
stdio: ['ignore', 'pipe', 'ignore'],
|
||||
windowsHide: true,
|
||||
}),
|
||||
);
|
||||
result = commonDir ? path.join(path.resolve(commonDir), 'info', 'exclude') : null;
|
||||
} catch {
|
||||
result = null;
|
||||
}
|
||||
gitInfoExcludePathCache.set(fromPath, result);
|
||||
return result;
|
||||
};
|
||||
|
||||
/**
|
||||
* Path to git's own global, all-repos ignore file: the value of
|
||||
* `core.excludesFile` (any config scope — system/global/local, resolved
|
||||
* the same way `git` itself would from `fromPath`), or git's documented
|
||||
* default of `$XDG_CONFIG_HOME/git/ignore` when unset (gitignore(5)).
|
||||
* Lowest-precedence source, mirroring git's own behavior (#2606).
|
||||
*
|
||||
* Never throws: an unset key or unavailable `git` falls through to the
|
||||
* default path, which is always computable without `git`.
|
||||
*/
|
||||
export const getCoreExcludesFilePath = (fromPath: string): string => {
|
||||
const cached = coreExcludesFilePathCache.get(fromPath);
|
||||
if (cached !== undefined) return cached;
|
||||
|
||||
let result: string | undefined;
|
||||
try {
|
||||
const configured = chompGitOutput(
|
||||
execSync('git config --get --type=path core.excludesFile', {
|
||||
cwd: fromPath,
|
||||
stdio: ['ignore', 'pipe', 'ignore'],
|
||||
windowsHide: true,
|
||||
}),
|
||||
);
|
||||
if (configured) result = configured;
|
||||
} catch {
|
||||
// Unset, or git unavailable — fall through to git's documented default.
|
||||
}
|
||||
if (!result) {
|
||||
const xdgConfigHome = process.env.XDG_CONFIG_HOME || path.join(os.homedir(), '.config');
|
||||
result = path.join(xdgConfigHome, 'git', 'ignore');
|
||||
}
|
||||
coreExcludesFilePathCache.set(fromPath, result);
|
||||
return result;
|
||||
};
|
||||
|
||||
/**
|
||||
* Resolve `fromPath` to the directory whose basename should drive the
|
||||
* registry name (#1259) — the *identity root*. Three outcomes:
|
||||
|
||||
@@ -429,23 +429,8 @@ export interface RepoMeta {
|
||||
* `E.hook` to `E$1.hook`, and nested-host anonymous names re-key
|
||||
* (`EnumWrap$1` → `EnumWrap$Mode$1`). Same contract as v8: identities move
|
||||
* on unchanged files; force a full re-analyze.
|
||||
* v10: Java `record_declaration` now emits a first-class `Record` graph node
|
||||
* (#2564): a record's container node was previously never created (JAVA_QUERIES
|
||||
* had no capture for it), so its methods existed as ownerless Method nodes
|
||||
* with no `HAS_METHOD` edge. The incremental write set only covers changed
|
||||
* files — a top-up against a pre-v10 index would keep silently omitting the
|
||||
* `Record` node and its `HAS_METHOD` edges for every unchanged record file
|
||||
* (same v7 contract: new nodes/edges the incremental path would otherwise
|
||||
* never backfill); force a full re-analyze instead.
|
||||
* v11: Rust abstract trait methods (`fn foo(&self) -> T;`, no body) now get a
|
||||
* scope + declaration capture (#2604): RUST_SCOPE_QUERY had no
|
||||
* `function_signature_item` pattern, so a `&dyn Trait` receiver could never
|
||||
* dispatch a CALLS edge to the trait's own method. Same v7/v10 contract: the
|
||||
* incremental write set only covers changed files, so a top-up against a
|
||||
* pre-v11 index would keep silently missing these CALLS edges for every
|
||||
* unchanged Rust trait file; force a full re-analyze instead.
|
||||
*/
|
||||
export const INCREMENTAL_SCHEMA_VERSION = 11;
|
||||
export const INCREMENTAL_SCHEMA_VERSION = 9;
|
||||
|
||||
export interface IndexedRepo {
|
||||
repoPath: string;
|
||||
|
||||
-8
@@ -21,12 +21,4 @@ class Unrelated {
|
||||
public void caller() {
|
||||
hook();
|
||||
}
|
||||
|
||||
public void dispatchToConstant() {
|
||||
EnumConst.A.hook();
|
||||
}
|
||||
|
||||
public void dispatchInherited() {
|
||||
EnumConst.A.log();
|
||||
}
|
||||
}
|
||||
|
||||
-13
@@ -1,13 +0,0 @@
|
||||
public enum Plain {
|
||||
A;
|
||||
|
||||
public void m() {
|
||||
System.out.println("plain m");
|
||||
}
|
||||
}
|
||||
|
||||
class PlainCaller {
|
||||
public void callPlain() {
|
||||
Plain.A.m();
|
||||
}
|
||||
}
|
||||
-12
@@ -1,12 +0,0 @@
|
||||
package probe;
|
||||
|
||||
public class LocalChain {
|
||||
void m() {
|
||||
class Local {
|
||||
void inner() {
|
||||
System.out.println("right target");
|
||||
}
|
||||
}
|
||||
new Local().inner();
|
||||
}
|
||||
}
|
||||
@@ -1,7 +0,0 @@
|
||||
package probe;
|
||||
|
||||
class Other {
|
||||
void inner() {
|
||||
System.out.println("wrong target");
|
||||
}
|
||||
}
|
||||
@@ -1,11 +0,0 @@
|
||||
package probe;
|
||||
|
||||
public record Point(int x, int y) {
|
||||
public int sum() {
|
||||
return x + y;
|
||||
}
|
||||
|
||||
public int scaled(int factor) {
|
||||
return sum() * factor;
|
||||
}
|
||||
}
|
||||
Vendored
-3
@@ -1,3 +0,0 @@
|
||||
class KnowledgeGraphService:
|
||||
def extract_and_store_graph(self, text: str) -> None:
|
||||
pass
|
||||
-23
@@ -1,23 +0,0 @@
|
||||
from knowledge_graph_service import KnowledgeGraphService
|
||||
|
||||
|
||||
class MemoryService:
|
||||
def __init__(self, knowledge_graph_service: KnowledgeGraphService):
|
||||
self.knowledge_graph_service = knowledge_graph_service
|
||||
|
||||
def store_memory(self, text: str) -> None:
|
||||
self.knowledge_graph_service.extract_and_store_graph(text)
|
||||
|
||||
def archive_memory(self, text: str) -> None:
|
||||
self.knowledge_graph_service.extract_and_store_graph(text)
|
||||
|
||||
def restore_memory(self, text: str) -> None:
|
||||
self.knowledge_graph_service.extract_and_store_graph(text)
|
||||
|
||||
|
||||
class ExplicitFieldMemoryService:
|
||||
def __init__(self, knowledge_graph_service):
|
||||
self.knowledge_graph_service: KnowledgeGraphService = knowledge_graph_service
|
||||
|
||||
def ingest_memory(self, text: str) -> None:
|
||||
self.knowledge_graph_service.extract_and_store_graph(text)
|
||||
-6
@@ -1,6 +0,0 @@
|
||||
def extract_and_store_graph(text: str) -> None:
|
||||
pass
|
||||
|
||||
|
||||
def exercise_decoy(text: str) -> None:
|
||||
extract_and_store_graph(text)
|
||||
@@ -1,15 +0,0 @@
|
||||
pub trait Behaviour {
|
||||
fn trait_target(&self) -> u32;
|
||||
}
|
||||
|
||||
pub struct Impl1;
|
||||
|
||||
impl Behaviour for Impl1 {
|
||||
fn trait_target(&self) -> u32 {
|
||||
7
|
||||
}
|
||||
}
|
||||
|
||||
pub fn calls_via_dyn(b: &dyn Behaviour) -> u32 {
|
||||
b.trait_target()
|
||||
}
|
||||
@@ -88,8 +88,8 @@
|
||||
"digest": "338c3922981604e71ddfc60ad61eba4b17f68ca654644e01add942c729b422cf"
|
||||
},
|
||||
"python-call-result-binding/models.py": {
|
||||
"captureGroups": 17,
|
||||
"digest": "441e2596001c4eaea4808ae6dd195a031f99d8ffe38c8fd450415c109d3365e2"
|
||||
"captureGroups": 16,
|
||||
"digest": "cbbb5168c28123820a70fe24016b0ed26ab02a6339d21ffe40fbccad940c1d70"
|
||||
},
|
||||
"python-call-result-binding/service.py": {
|
||||
"captureGroups": 9,
|
||||
@@ -171,25 +171,13 @@
|
||||
"captureGroups": 15,
|
||||
"digest": "201ce01b83b21d729aca89c6299570df55393a95f1899a2eaacd989873950177"
|
||||
},
|
||||
"python-constructor-field-receiver/knowledge_graph_service.py": {
|
||||
"captureGroups": 10,
|
||||
"digest": "83dcf9f81ac7d0a9e9ed5e467e41acd07e2ec581926d85eea5d4b5e1e4157744"
|
||||
},
|
||||
"python-constructor-field-receiver/memory_service.py": {
|
||||
"captureGroups": 59,
|
||||
"digest": "ad9c3be5a7c10e112bb20eac20b8603196fc2e739b7567159c58a607639ce192"
|
||||
},
|
||||
"python-constructor-field-receiver/test_fixture.py": {
|
||||
"captureGroups": 13,
|
||||
"digest": "6f642e752086a5e9337d21accb65ca90ef5af2b3e139239ea12c5cd031844dd2"
|
||||
},
|
||||
"python-constructor-type-inference/models/repo.py": {
|
||||
"captureGroups": 17,
|
||||
"digest": "3c400c7a331d7796a730e1ba53b91c4f5ec4799121044b0c160844988fca8662"
|
||||
"captureGroups": 16,
|
||||
"digest": "ad11823ee187cc3e1efab34a67a5013119b4ada87c5701b080921c0b0be09e62"
|
||||
},
|
||||
"python-constructor-type-inference/models/user.py": {
|
||||
"captureGroups": 17,
|
||||
"digest": "441e2596001c4eaea4808ae6dd195a031f99d8ffe38c8fd450415c109d3365e2"
|
||||
"captureGroups": 16,
|
||||
"digest": "cbbb5168c28123820a70fe24016b0ed26ab02a6339d21ffe40fbccad940c1d70"
|
||||
},
|
||||
"python-constructor-type-inference/services/app.py": {
|
||||
"captureGroups": 13,
|
||||
@@ -204,12 +192,12 @@
|
||||
"digest": "dd51c32d705934b1384991ad2291869f446327752481abc20600d4ad9f553ea3"
|
||||
},
|
||||
"python-dict-items-loop/repo.py": {
|
||||
"captureGroups": 16,
|
||||
"digest": "2d283b4acbc71e318a4520cb7557508b9084213ba53fbdac07457e348a8b24c6"
|
||||
"captureGroups": 15,
|
||||
"digest": "8116cf4cbf4dca377e88f97ca645f40fab648a4e9a8e790b5b3761e4a3e17d7c"
|
||||
},
|
||||
"python-dict-items-loop/user.py": {
|
||||
"captureGroups": 16,
|
||||
"digest": "6568834a7f228e78a11b08282196e138795980aed6c55b9953cc89e87c998e52"
|
||||
"captureGroups": 15,
|
||||
"digest": "15984fa30be4603f3e47c27342352dd602d104b33ac5224dc911d78a82d87926"
|
||||
},
|
||||
"python-django-app-imports/accounts/__init__.py": {
|
||||
"captureGroups": 0,
|
||||
@@ -300,8 +288,8 @@
|
||||
"digest": "d1e23831dcae38034b278bfefa2b8c4e21ca722ab2c79bfb3126744338a2a401"
|
||||
},
|
||||
"python-enumerate-loop/user.py": {
|
||||
"captureGroups": 17,
|
||||
"digest": "441e2596001c4eaea4808ae6dd195a031f99d8ffe38c8fd450415c109d3365e2"
|
||||
"captureGroups": 16,
|
||||
"digest": "cbbb5168c28123820a70fe24016b0ed26ab02a6339d21ffe40fbccad940c1d70"
|
||||
},
|
||||
"python-field-type-disambig/address.py": {
|
||||
"captureGroups": 9,
|
||||
@@ -328,8 +316,8 @@
|
||||
"digest": "5c290b3b34f3f5e9dcdd6ee3ae72ba4223337cd64592330f9a8c6c6f13b2fd2d"
|
||||
},
|
||||
"python-for-call-expr/models.py": {
|
||||
"captureGroups": 41,
|
||||
"digest": "e8807a9969732197feb04204d200d5810b895c006f1424a7a6f5f0d5762ef49a"
|
||||
"captureGroups": 39,
|
||||
"digest": "133a14c0543a41412d9d4fd5485d6270e1f4b0cd247f0ec0d71974850e4918c1"
|
||||
},
|
||||
"python-function-local-import-chain/app.py": {
|
||||
"captureGroups": 9,
|
||||
@@ -476,8 +464,8 @@
|
||||
"digest": "01fe4805f59723a5f163d26b7be3ed3e456eb3a093e3df3a8f034cecee22ebb9"
|
||||
},
|
||||
"python-method-chain-binding/models.py": {
|
||||
"captureGroups": 48,
|
||||
"digest": "f049817034428759193fce03b417ca23c133f6ead87f960e173afba9b79b88e8"
|
||||
"captureGroups": 45,
|
||||
"digest": "ea3f745514a330d86447796734faff29f0c88ea49a7ef8781087c537426d29bf"
|
||||
},
|
||||
"python-method-enrichment/app.py": {
|
||||
"captureGroups": 13,
|
||||
@@ -692,8 +680,8 @@
|
||||
"digest": "741f690b6330491303b9b58cb31027a33600973265b59428facfefbabf0cf7e1"
|
||||
},
|
||||
"python-return-type-inference/models.py": {
|
||||
"captureGroups": 17,
|
||||
"digest": "441e2596001c4eaea4808ae6dd195a031f99d8ffe38c8fd450415c109d3365e2"
|
||||
"captureGroups": 16,
|
||||
"digest": "cbbb5168c28123820a70fe24016b0ed26ab02a6339d21ffe40fbccad940c1d70"
|
||||
},
|
||||
"python-return-type-inference/service.py": {
|
||||
"captureGroups": 9,
|
||||
@@ -772,8 +760,8 @@
|
||||
"digest": "4df7ea089c43552ca4ea5a51f8e985d11d351b86949efb2f7ebefdf6a9ffd689"
|
||||
},
|
||||
"python-walrus-operator/models.py": {
|
||||
"captureGroups": 22,
|
||||
"digest": "9ab1a8a69970e8c875bd103251ecf9c8af2c1464a6bdc1213fe7560c4fa34063"
|
||||
"captureGroups": 21,
|
||||
"digest": "cf014d1bad66ea61e327c146fd1a52159a96faaf264555bb164230a5218e6948"
|
||||
},
|
||||
"python-write-access/models.py": {
|
||||
"captureGroups": 11,
|
||||
@@ -784,7 +772,7 @@
|
||||
"digest": "6e3690ec68d8de54f376bb6f5f7a29da829a8a6a24001eabae9ec66a327f3409"
|
||||
},
|
||||
"synthetic:dao-20": {
|
||||
"captureGroups": 773,
|
||||
"digest": "37e047eda37477bbc33f4dd8ba259c3f876580378566c0c14dfe580795a952af"
|
||||
"captureGroups": 733,
|
||||
"digest": "c540f2143882137e6c6f7996bf9b87a0117f41915b1d0989d3d6bb9db5a5a1ab"
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
{
|
||||
"rust-abstract-dispatch/src/lib.rs": {
|
||||
"captureGroups": 34,
|
||||
"digest": "973679363065ecd54c4e5128a9fab214ea27eca24f0f079c63a3c6f285e678b0"
|
||||
"captureGroups": 30,
|
||||
"digest": "88309004d1ab00054f81bc55c1d058fc4ca25781162d6b94b7e7ce631a5d61b2"
|
||||
},
|
||||
"rust-abstract-dispatch/src/main.rs": {
|
||||
"captureGroups": 21,
|
||||
@@ -148,8 +148,8 @@
|
||||
"digest": "e0120e3f215282e68d83b4f8f5d8918945e0b3e7ce4e0128c6afd2aa43caa1c0"
|
||||
},
|
||||
"rust-cross-module-collision/src/traits.rs": {
|
||||
"captureGroups": 5,
|
||||
"digest": "c7150a5052e0e2b5fd7fc21cc8ce361530fe5cc60f67ad0e2ef945bb97965b53"
|
||||
"captureGroups": 3,
|
||||
"digest": "88eef9d92ea6e370bd8ef7fbf64c42ec622ec933fb53c56b32bc67db87fa8e03"
|
||||
},
|
||||
"rust-deep-field-chain/models.rs": {
|
||||
"captureGroups": 24,
|
||||
@@ -171,10 +171,6 @@
|
||||
"captureGroups": 22,
|
||||
"digest": "c53db401a81fde2ffd5665393acb9cd605a62ec51c015c3aafb3f41c0897471f"
|
||||
},
|
||||
"rust-dyn-trait-object/src/lib.rs": {
|
||||
"captureGroups": 23,
|
||||
"digest": "720618dff6a43ab8e5b59aa354c0c448b9057dd6f2f7b3b22b13b82d53745943"
|
||||
},
|
||||
"rust-err-unwrap/src/error.rs": {
|
||||
"captureGroups": 9,
|
||||
"digest": "798c8e01c6e54792ba69e845248efc8abf0cba38fa3d16fb8e0d1f6dd2ad2b7e"
|
||||
@@ -320,8 +316,8 @@
|
||||
"digest": "cd836a2a9c15ab240961d2e15f192f7e33d65eb5ebf2e1a8af2f620a47fe66ae"
|
||||
},
|
||||
"rust-method-enrichment/src/lib.rs": {
|
||||
"captureGroups": 42,
|
||||
"digest": "a4d9ca570fbb1ff1859a0b4f737aa3507b236f99700d37ded2c8c36518add567"
|
||||
"captureGroups": 40,
|
||||
"digest": "71627a8218e32514b6945e4e310686eb631ce37451c90c9644e3f5336a37820b"
|
||||
},
|
||||
"rust-method-enrichment/src/main.rs": {
|
||||
"captureGroups": 18,
|
||||
@@ -364,8 +360,8 @@
|
||||
"digest": "141388068614e16d96f27cfdf18ac9001b9e202ce832fe10f38dab990637b3ab"
|
||||
},
|
||||
"rust-parent-resolution/src/serializable.rs": {
|
||||
"captureGroups": 5,
|
||||
"digest": "f33bb881dd937cdd5af2eca6b0284ea79ca2d296c79217513f882dcba8a82fd8"
|
||||
"captureGroups": 3,
|
||||
"digest": "f35d44f44d81e3a0be40f68ba9dbd4bde6f01659fa15b6db34a458ad460f904e"
|
||||
},
|
||||
"rust-parent-resolution/src/user.rs": {
|
||||
"captureGroups": 13,
|
||||
@@ -376,8 +372,8 @@
|
||||
"digest": "bc8946d31db81b85d780633608fdaa7565258cd788285fa00cd6dcb0de3dd16c"
|
||||
},
|
||||
"rust-qualified-trait/src/traits.rs": {
|
||||
"captureGroups": 9,
|
||||
"digest": "10f3bba4c2a16cdac77de0498ff506910ac09cc1a85e7daebfc5743c54e26015"
|
||||
"captureGroups": 5,
|
||||
"digest": "15be069f28f1400e4beb0b0860acb59979f78549960486f36a92f56578f05a06"
|
||||
},
|
||||
"rust-qualified-trait/src/widget.rs": {
|
||||
"captureGroups": 23,
|
||||
@@ -496,12 +492,12 @@
|
||||
"digest": "f1b9f72d74467be55a8b7679215b49bcabb4d0fced6080f752672070b32ed93d"
|
||||
},
|
||||
"rust-traits/src/traits/clickable.rs": {
|
||||
"captureGroups": 7,
|
||||
"digest": "83c36832f24446fe03a07288dd394bc7494a54e5979c9b71c1dcb8b19ab54341"
|
||||
"captureGroups": 3,
|
||||
"digest": "3ed5b27c172d48f83929715ba92d1030a282f9f1e29ec2fcdd3d7e9efbc54a84"
|
||||
},
|
||||
"rust-traits/src/traits/drawable.rs": {
|
||||
"captureGroups": 9,
|
||||
"digest": "cee5091f041038722f1f012394a75ba4e16870d05b2dafa37371200e785198f0"
|
||||
"captureGroups": 5,
|
||||
"digest": "1dca39bbc7c1b1b66f1a34730b9a5b4dba04c54ee9d2688255e0fd4e6bc48499"
|
||||
},
|
||||
"rust-union/lib.rs": {
|
||||
"captureGroups": 10,
|
||||
|
||||
-25
@@ -1,25 +0,0 @@
|
||||
package com.example;
|
||||
|
||||
import org.springframework.beans.factory.annotation.Value;
|
||||
import org.springframework.boot.context.properties.ConfigurationProperties;
|
||||
|
||||
class DirectValues {
|
||||
@Value("${payment.timeout:30}")
|
||||
private int timeout;
|
||||
|
||||
@Value("${payment.missing}")
|
||||
private String missing;
|
||||
}
|
||||
|
||||
@ConfigurationProperties(prefix = "service")
|
||||
class ServiceProperties {
|
||||
private String endpoint;
|
||||
private Retry retry;
|
||||
}
|
||||
|
||||
@ConfigurationProperties("service")
|
||||
class UnmatchedServiceProperties {
|
||||
private String unrelated;
|
||||
}
|
||||
|
||||
class Retry {}
|
||||
-6
@@ -1,6 +0,0 @@
|
||||
defaults: &defaults
|
||||
retry:
|
||||
max-attempts: 3
|
||||
service:
|
||||
<<: *defaults
|
||||
endpoint: https://service.example.test
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user