Compare commits
7
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
a793032e06 | ||
|
|
6a8947217c | ||
|
|
e412d292fe | ||
|
|
d69eadfb7f | ||
|
|
2620b704e0 | ||
|
|
5d670a530d | ||
|
|
4848dce9ea |
@@ -58,6 +58,7 @@ jobs:
|
||||
timeout-minutes: 5
|
||||
permissions:
|
||||
contents: read
|
||||
pull-requests: read # read PR labels on the merge commit
|
||||
outputs:
|
||||
should_run: ${{ steps.decide.outputs.should_run }}
|
||||
head_sha: ${{ steps.decide.outputs.head_sha }}
|
||||
@@ -74,6 +75,8 @@ jobs:
|
||||
FORCE: ${{ inputs.force }}
|
||||
BUMP_INPUT: ${{ inputs.bump }}
|
||||
EVENT_NAME: ${{ github.event_name }}
|
||||
GH_TOKEN: ${{ github.token }}
|
||||
REPO: ${{ github.repository }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
HEAD_SHA=$(git rev-parse HEAD)
|
||||
@@ -96,6 +99,49 @@ jobs:
|
||||
exit 0
|
||||
fi
|
||||
|
||||
# ── Skip when the merge commit corresponds to a release ─────────
|
||||
# Two complementary checks (belt-and-suspenders):
|
||||
# 1. The HEAD commit subject matches `chore: release vX.Y.Z`
|
||||
# (the canonical release-PR title in this repo). Anchored
|
||||
# at both ends to require the bare title or the squash-merge
|
||||
# `(#NNNN)` suffix exactly — rejects noisy variants like
|
||||
# `chore: release v1.0.0 (something unrelated)`.
|
||||
# 2. The squash-merged PR carries the `release` label.
|
||||
# Either match suppresses the rc build — stable releases publish
|
||||
# via publish.yml on the v-tag, so the rc cycle should pause for
|
||||
# them rather than racing the npm publish.
|
||||
HEAD_SUBJECT="$(git log -1 --pretty=%s HEAD)"
|
||||
# Sanitise GitHub-Actions annotation prefixes before logging the
|
||||
# raw subject — defence-in-depth so a hypothetical commit subject
|
||||
# containing `::error::` or `::set-output::` cannot forge log
|
||||
# annotations even though %s strips newlines.
|
||||
HEAD_SUBJECT_SAFE="${HEAD_SUBJECT//::/__}"
|
||||
RELEASE_SUBJECT_RE='^chore:[[:space:]]*release[[:space:]]+v[0-9]+\.[0-9]+\.[0-9]+([[:space:]]+\(#[0-9]+\))?$'
|
||||
if [[ "$HEAD_SUBJECT" =~ $RELEASE_SUBJECT_RE ]]; then
|
||||
echo "HEAD commit subject matches a release commit — skipping rc."
|
||||
echo " subject (sanitised): $HEAD_SUBJECT_SAFE"
|
||||
echo "should_run=false" >> "$GITHUB_OUTPUT"
|
||||
exit 0
|
||||
fi
|
||||
|
||||
# Squash-merge commits include `(#NNNN)` at the end of the subject.
|
||||
if [[ "$HEAD_SUBJECT" =~ \(#([0-9]+)\)[[:space:]]*$ ]]; then
|
||||
PR_NUM="${BASH_REMATCH[1]}"
|
||||
echo "Detected squash-merge of PR #$PR_NUM — checking labels."
|
||||
if LABELS_JSON="$(gh pr view "$PR_NUM" --repo "$REPO" --json labels 2>/dev/null)"; then
|
||||
if printf '%s' "$LABELS_JSON" | jq -e '.labels[] | select(.name == "release")' >/dev/null; then
|
||||
echo "PR #$PR_NUM has the 'release' label — skipping rc."
|
||||
echo "should_run=false" >> "$GITHUB_OUTPUT"
|
||||
exit 0
|
||||
fi
|
||||
echo "PR #$PR_NUM has no 'release' label — proceeding."
|
||||
else
|
||||
# Lookup failure is not fatal — fall through to the dedup check
|
||||
# so a transient GH API hiccup doesn't silently suppress rc builds.
|
||||
echo "::warning::Could not read labels for PR #${PR_NUM} — falling through."
|
||||
fi
|
||||
fi
|
||||
|
||||
# Dedup: is there already an rc/<HEAD_SHA> marker pointing at HEAD?
|
||||
MARKER="rc/${HEAD_SHA}"
|
||||
if git rev-parse "refs/tags/$MARKER" >/dev/null 2>&1; then
|
||||
|
||||
@@ -120,7 +120,7 @@ To configure MCP for your editor, run `npx gitnexus setup` once — or set it up
|
||||
| Editor | MCP | Skills | Hooks (auto-augment) | Support |
|
||||
| --------------------- | --- | ------ | -------------------- | -------------- |
|
||||
| **Claude Code** | Yes | Yes | Yes (PreToolUse + PostToolUse) | **Full** |
|
||||
| **Cursor** | Yes | Yes | — | MCP + Skills |
|
||||
| **Cursor** | Yes | Yes | Yes (postToolUse, [manual install](gitnexus-cursor-integration/README.md#hook-install)) | **Full** |
|
||||
| **Codex** | Yes | Yes | — | MCP + Skills |
|
||||
| **Windsurf** | Yes | — | — | MCP |
|
||||
| **OpenCode** | Yes | Yes | — | MCP + Skills |
|
||||
|
||||
@@ -0,0 +1,95 @@
|
||||
/**
|
||||
* Custom ESLint rule: require `parseSourceSafe(parser, content, ...)` instead
|
||||
* of direct `<parser>.parse(<content>, ...)` calls.
|
||||
*
|
||||
* Background: tree-sitter's Node.js native binding crashes with SIGSEGV on
|
||||
* Windows when handed a JS string longer than 32 767 chars. The crash happens
|
||||
* inside the binding's V8 string-to-buffer conversion and cannot be intercepted
|
||||
* by JavaScript `try/catch`. `parseSourceSafe` (in
|
||||
* `gitnexus/src/core/tree-sitter/safe-parse.ts`) routes large inputs through
|
||||
* the chunked-callback overload of `parser.parse(input, ...)` which bypasses
|
||||
* the broken conversion path. PR #1433 fixed every direct call site at the
|
||||
* time; this rule prevents new direct calls from creeping in.
|
||||
*
|
||||
* The rule is auto-fixable for the call-site rewrite. It does NOT auto-add the
|
||||
* import (computing the correct relative path per file is brittle); after the
|
||||
* call rewrite runs, the consumer file's `tsc` will complain about an
|
||||
* undefined identifier and the developer adds the import. This is the same
|
||||
* tradeoff `unused-imports/no-unused-imports` makes in the opposite direction.
|
||||
*
|
||||
* False-positive suppression:
|
||||
* - Skips calls whose receiver is a known non-tree-sitter library (`JSON`,
|
||||
* `URL`, `marked`, `Number`).
|
||||
* - Skips calls whose first argument is a string-literal (grammar-load smoke
|
||||
* tests like `_testParser.parse('service X { rpc Y (R) returns (R); }')`).
|
||||
* - Skips test files (`.test.ts`/`.test.tsx`/`.spec.ts`).
|
||||
* - Skips the `safe-parse.ts` helper itself.
|
||||
*/
|
||||
|
||||
const SKIPPED_RECEIVERS = new Set(['JSON', 'URL', 'marked', 'Number', 'Math']);
|
||||
|
||||
export default {
|
||||
meta: {
|
||||
type: 'problem',
|
||||
docs: {
|
||||
description:
|
||||
'Require parseSourceSafe instead of direct tree-sitter `<parser>.parse(content, ...)` calls (Windows SIGSEGV protection)',
|
||||
recommended: true,
|
||||
},
|
||||
fixable: 'code',
|
||||
schema: [],
|
||||
messages: {
|
||||
useSafeParse:
|
||||
'Direct `{{receiver}}.parse(...)` can SIGSEGV on Windows for inputs > 32 767 chars (uncatchable from JS). Use `parseSourceSafe({{receiver}}, ...)` from `core/tree-sitter/safe-parse.js`. Auto-fix rewrites the call; add the missing import yourself.',
|
||||
},
|
||||
},
|
||||
create(context) {
|
||||
const filename = context.filename ?? context.getFilename();
|
||||
// Don't lint the helper itself or test files.
|
||||
if (filename.includes('safe-parse')) return {};
|
||||
if (/[.](?:test|spec)\.tsx?$/.test(filename)) return {};
|
||||
|
||||
const sourceCode = context.sourceCode ?? context.getSourceCode();
|
||||
|
||||
return {
|
||||
CallExpression(node) {
|
||||
const callee = node.callee;
|
||||
if (callee.type !== 'MemberExpression') return;
|
||||
if (callee.computed) return;
|
||||
if (callee.property.type !== 'Identifier') return;
|
||||
if (callee.property.name !== 'parse') return;
|
||||
|
||||
// Skip known non-tree-sitter receivers.
|
||||
if (callee.object.type === 'Identifier' && SKIPPED_RECEIVERS.has(callee.object.name)) {
|
||||
return;
|
||||
}
|
||||
|
||||
// Smoke tests pass a string literal directly; those are trivially safe.
|
||||
const firstArg = node.arguments[0];
|
||||
if (!firstArg) return;
|
||||
if (firstArg.type === 'Literal' && typeof firstArg.value === 'string') return;
|
||||
if (firstArg.type === 'TemplateLiteral' && firstArg.expressions.length === 0) return;
|
||||
|
||||
const receiverText = sourceCode.getText(callee.object);
|
||||
// Receiver-text-shape skip: anything matching well-known JS APIs that
|
||||
// happen to have a `.parse(<expr>)` shape but aren't tree-sitter.
|
||||
if (
|
||||
/^(JSON|URL|marked|Number|Math|Date|globalThis\.JSON)\b/.test(receiverText) ||
|
||||
/\bjson\.parse\b/i.test(receiverText)
|
||||
) {
|
||||
return;
|
||||
}
|
||||
|
||||
context.report({
|
||||
node,
|
||||
messageId: 'useSafeParse',
|
||||
data: { receiver: receiverText },
|
||||
fix(fixer) {
|
||||
const argsText = node.arguments.map((arg) => sourceCode.getText(arg)).join(', ');
|
||||
return fixer.replaceText(node, `parseSourceSafe(${receiverText}, ${argsText})`);
|
||||
},
|
||||
});
|
||||
},
|
||||
};
|
||||
},
|
||||
};
|
||||
@@ -3,6 +3,15 @@ import tsParser from '@typescript-eslint/parser';
|
||||
import unusedImports from 'eslint-plugin-unused-imports';
|
||||
import reactHooks from 'eslint-plugin-react-hooks';
|
||||
import prettierConfig from 'eslint-config-prettier';
|
||||
import requireSafeParse from './eslint-rules/require-safe-parse.mjs';
|
||||
|
||||
// Local plugin hosting custom rules that enforce GitNexus-specific invariants
|
||||
// (currently: the Windows-SIGSEGV-safe parser entrypoint).
|
||||
const gitnexusLocalPlugin = {
|
||||
rules: {
|
||||
'require-safe-parse': requireSafeParse,
|
||||
},
|
||||
};
|
||||
|
||||
// Selectors that protect MCP-reachable code from corrupting the JSON-RPC
|
||||
// stdio frame stream. The MCP-reachable block below uses these directly;
|
||||
@@ -135,6 +144,23 @@ export default [
|
||||
},
|
||||
},
|
||||
|
||||
// Windows SIGSEGV protection: every tree-sitter parse in `core/` must route
|
||||
// through parseSourceSafe. Direct `<parser>.parse(content, ...)` crashes on
|
||||
// Windows for inputs > 32 767 chars (V8 string-conversion bug, uncatchable
|
||||
// from JS). The rule auto-fixes the call site; the developer adds the
|
||||
// missing import after the fix runs. Out of scope: tests (skipped by the
|
||||
// rule), the helper itself (`safe-parse.ts`), and the `grpc-patterns/proto.ts`
|
||||
// grammar-load smoke test (filtered by string-literal-arg skip in the rule).
|
||||
{
|
||||
files: ['gitnexus/src/core/**/*.ts'],
|
||||
plugins: {
|
||||
gitnexus: gitnexusLocalPlugin,
|
||||
},
|
||||
rules: {
|
||||
'gitnexus/require-safe-parse': 'error',
|
||||
},
|
||||
},
|
||||
|
||||
// React-specific rules for gitnexus-web
|
||||
{
|
||||
files: ['gitnexus-web/src/**/*.{ts,tsx}'],
|
||||
|
||||
@@ -0,0 +1,89 @@
|
||||
# GitNexus — Cursor integration
|
||||
|
||||
Static config that adds GitNexus knowledge-graph augmentation and skill files to Cursor.
|
||||
|
||||
> **Hooks require Cursor 2.4+.** Earlier versions don't expose `postToolUse` and the hook will silently no-op.
|
||||
|
||||
## What you get
|
||||
|
||||
| Layer | What it does | How it's installed |
|
||||
| ------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------- |
|
||||
| **MCP** | `gitnexus` MCP server with 16 tools (`query`, `context`, `impact`, `detect_changes`, `rename`, …) | `npx gitnexus setup` writes `~/.cursor/mcp.json` automatically. |
|
||||
| **Skills** | `/gitnexus-exploring`, `/gitnexus-debugging`, `/gitnexus-impact-analysis`, `/gitnexus-refactoring`, `/gitnexus-pr-review` markdown skills | `npx gitnexus setup` copies them to `~/.cursor/skills/gitnexus/`. |
|
||||
| **Hooks** _(this README)_ | `postToolUse` hook that enriches `Shell` / `Read` / `Grep` tool calls with graph context — same augmentation Claude Code gets | **Manual** — copy the two files described below into your project's `.cursor/`. |
|
||||
|
||||
## Hook install
|
||||
|
||||
Cursor 2.4+ reads `.cursor/hooks.json` from the project root and runs hook commands with the project root as the working directory ([docs](https://cursor.com/docs/agent/hooks)).
|
||||
|
||||
From this repo's `gitnexus-cursor-integration/hooks/`, copy the two files into your **project root**:
|
||||
|
||||
```text
|
||||
<your-project>/
|
||||
├── .cursor/
|
||||
│ └── hooks.json ← from gitnexus-cursor-integration/hooks/hooks.json
|
||||
└── hooks/
|
||||
└── gitnexus-hook.cjs ← from gitnexus-cursor-integration/hooks/gitnexus-hook.cjs
|
||||
```
|
||||
|
||||
Equivalent shell commands (run from your project root, with `$GITNEXUS_REPO` pointing at a clone of this repo):
|
||||
|
||||
```bash
|
||||
mkdir -p .cursor hooks
|
||||
cp "$GITNEXUS_REPO/gitnexus-cursor-integration/hooks/hooks.json" .cursor/hooks.json
|
||||
cp "$GITNEXUS_REPO/gitnexus-cursor-integration/hooks/gitnexus-hook.cjs" hooks/gitnexus-hook.cjs
|
||||
```
|
||||
|
||||
If you already have a `.cursor/hooks.json`, merge the `hooks.postToolUse` array rather than overwriting.
|
||||
|
||||
### Verify
|
||||
|
||||
1. Index the project: `npx gitnexus analyze`
|
||||
2. Reload the Cursor window so it picks up the new hook config.
|
||||
3. Ask the agent something that triggers `Read` / `Grep` / `Shell rg`. You should see a `[GitNexus]` block appended to the tool result.
|
||||
4. Diagnose silent no-ops by setting `GITNEXUS_DEBUG=1` in your shell environment — the hook will write Cursor's raw event payload to stderr so you can verify field names.
|
||||
|
||||
### What's installed manually vs. automated
|
||||
|
||||
| Step | Automated by `gitnexus setup`? |
|
||||
| -------------------------------------------------------------------- | ------------------------------ |
|
||||
| `~/.cursor/mcp.json` | ✅ |
|
||||
| `~/.cursor/skills/gitnexus/*` | ✅ |
|
||||
| `<project>/.cursor/hooks.json` + `<project>/hooks/gitnexus-hook.cjs` | ❌ — copy manually (see above) |
|
||||
|
||||
Hook install is per-project (Cursor scopes hooks to a project root); skills and MCP config are global.
|
||||
|
||||
## Hook contract
|
||||
|
||||
The hook receives a JSON event on stdin matching Cursor 2.4's `postToolUse` shape:
|
||||
|
||||
```json
|
||||
{
|
||||
"tool_name": "Grep" | "Read" | "Shell",
|
||||
"tool_input": { /* tool-specific */ },
|
||||
"tool_output": { /* optional */ },
|
||||
"cwd": "/absolute/path/to/project"
|
||||
}
|
||||
```
|
||||
|
||||
It writes augmentation context to stdout as:
|
||||
|
||||
```json
|
||||
{ "additional_context": "[GitNexus] …" }
|
||||
```
|
||||
|
||||
Empty stdout means "no augmentation, continue normally" — the hook never blocks the tool.
|
||||
|
||||
### Pattern extraction per tool
|
||||
|
||||
| Tool | Pattern source | Notes |
|
||||
| ------- | ---------------------------------------------------------------------------------------------------------------------- | ---------------------------------------------------------------------------------------------------- |
|
||||
| `Grep` | `tool_input.query` (also `pattern`, `regex`, `q`, `search`, `searchQuery`) | Last-resort fallback: longest string value in `tool_input` (≥ 3 chars). |
|
||||
| `Read` | basename of `tool_input.target_file` (also `file_path`, `filePath`, `path`, `file`), stripped to identifier characters | `auth/handler.ts` → `handler`. |
|
||||
| `Shell` | First positional argument after `rg` / `grep` in `tool_input.command` | Best-effort tokenizer; quoted multi-word patterns (`rg "User Service"`) extract the first word only. |
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
- **Nothing happens** — Confirm Cursor is on 2.4+ and the project root has both `.cursor/hooks.json` and the script at `hooks/gitnexus-hook.cjs`. Then `npx gitnexus list` to confirm the project is indexed.
|
||||
- **`gitnexus` not found** — The hook prefers a locally-resolvable `gitnexus/dist/cli/index.js` and falls back to `npx -y gitnexus`. Install globally with `npm i -g gitnexus` to skip the npx cold-start latency.
|
||||
- **Wrong pattern extracted** — Set `GITNEXUS_DEBUG=1` and run a tool call. The raw stdin payload is logged to stderr; use it to confirm Cursor's actual `tool_input` field names against the table above. If they differ, file an issue with the captured payload.
|
||||
@@ -1,50 +0,0 @@
|
||||
#!/bin/bash
|
||||
# GitNexus beforeShellExecution hook for Cursor
|
||||
# Receives JSON on stdin with { command, cwd, timeout }
|
||||
# Returns JSON on stdout with { permission, agent_message }
|
||||
#
|
||||
# Extracts search pattern from grep/rg commands, runs gitnexus augment,
|
||||
# and injects the enriched context via agent_message.
|
||||
|
||||
INPUT=$(cat)
|
||||
|
||||
COMMAND=$(echo "$INPUT" | jq -r '.command // empty' 2>/dev/null)
|
||||
|
||||
if [ -z "$COMMAND" ]; then
|
||||
echo '{"permission":"allow"}'
|
||||
exit 0
|
||||
fi
|
||||
|
||||
# Skip non-search commands
|
||||
case "$COMMAND" in
|
||||
cd\ *|npm\ *|yarn\ *|pnpm\ *|git\ commit*|git\ push*|git\ pull*|mkdir\ *|rm\ *|cp\ *|mv\ *|echo\ *|cat\ *)
|
||||
echo '{"permission":"allow"}'
|
||||
exit 0
|
||||
;;
|
||||
esac
|
||||
|
||||
# Extract search pattern from rg/grep commands
|
||||
PATTERN=""
|
||||
if echo "$COMMAND" | grep -qE '\brg\b'; then
|
||||
PATTERN=$(echo "$COMMAND" | sed -n "s/.*\brg\s\+\(--[^ ]*\s\+\)*['\"]\\?\([^'\";\| >]*\\).*/\2/p")
|
||||
elif echo "$COMMAND" | grep -qE '\bgrep\b'; then
|
||||
PATTERN=$(echo "$COMMAND" | sed -n "s/.*\bgrep\s\+\(-[^ ]*\s\+\)*['\"]\\?\([^'\";\| >]*\\).*/\2/p")
|
||||
fi
|
||||
|
||||
if [ -z "$PATTERN" ] || [ ${#PATTERN} -lt 3 ]; then
|
||||
echo '{"permission":"allow"}'
|
||||
exit 0
|
||||
fi
|
||||
|
||||
# Run gitnexus augment
|
||||
RESULT=$(npx -y gitnexus augment "$PATTERN" 2>/dev/null)
|
||||
|
||||
if [ -n "$RESULT" ]; then
|
||||
# Escape for JSON
|
||||
ESCAPED=$(echo "$RESULT" | jq -Rs .)
|
||||
echo "{\"permission\":\"allow\",\"agent_message\":$ESCAPED}"
|
||||
else
|
||||
echo '{"permission":"allow"}'
|
||||
fi
|
||||
|
||||
exit 0
|
||||
@@ -0,0 +1,259 @@
|
||||
#!/usr/bin/env node
|
||||
/**
|
||||
* GitNexus Cursor postToolUse Hook
|
||||
*
|
||||
* Receives a JSON event on stdin describing a finished tool call, derives a
|
||||
* search pattern (Grep query, Read file basename, or rg/grep arg from a Shell
|
||||
* command), runs `gitnexus augment <pattern>`, and emits the enriched context
|
||||
* back as `{ additional_context: "..." }` so the agent sees it alongside the
|
||||
* tool result.
|
||||
*
|
||||
* Replaces the legacy beforeShellExecution / augment-shell.sh pipeline:
|
||||
* - Cross-platform (no bash, no jq — runs on Windows out of the box)
|
||||
* - Covers Read and Grep, not just Shell rg/grep
|
||||
*
|
||||
* Cursor 2.4+ generic hooks: https://cursor.com/docs/agent/hooks
|
||||
*/
|
||||
|
||||
const fs = require('fs');
|
||||
const path = require('path');
|
||||
const { spawnSync } = require('child_process');
|
||||
|
||||
function readInput() {
|
||||
try {
|
||||
const data = fs.readFileSync(0, 'utf-8');
|
||||
return JSON.parse(data);
|
||||
} catch {
|
||||
return {};
|
||||
}
|
||||
}
|
||||
|
||||
function isGlobalRegistryDir(candidate) {
|
||||
if (fs.existsSync(path.join(candidate, 'meta.json'))) return false;
|
||||
return (
|
||||
fs.existsSync(path.join(candidate, 'registry.json')) ||
|
||||
fs.existsSync(path.join(candidate, 'repos'))
|
||||
);
|
||||
}
|
||||
|
||||
function walkForGitNexusDir(startDir) {
|
||||
let dir = startDir;
|
||||
for (let i = 0; i < 5; i++) {
|
||||
const candidate = path.join(dir, '.gitnexus');
|
||||
if (fs.existsSync(candidate)) {
|
||||
if (!isGlobalRegistryDir(candidate)) return candidate;
|
||||
}
|
||||
const parent = path.dirname(dir);
|
||||
if (parent === dir) break;
|
||||
dir = parent;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
function findCanonicalRepoRoot(cwd) {
|
||||
try {
|
||||
const result = spawnSync('git', ['rev-parse', '--path-format=absolute', '--git-common-dir'], {
|
||||
encoding: 'utf-8',
|
||||
timeout: 2000,
|
||||
cwd,
|
||||
stdio: ['pipe', 'pipe', 'pipe'],
|
||||
});
|
||||
if (result.error || result.status !== 0) return null;
|
||||
const commonDir = (result.stdout || '').trim();
|
||||
if (!commonDir || !path.isAbsolute(commonDir)) return null;
|
||||
return path.dirname(commonDir);
|
||||
} catch {
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
function findGitNexusDir(startDir) {
|
||||
const cwd = startDir || process.cwd();
|
||||
const fromCwd = walkForGitNexusDir(cwd);
|
||||
if (fromCwd) return fromCwd;
|
||||
const canonicalRoot = findCanonicalRepoRoot(cwd);
|
||||
if (canonicalRoot && canonicalRoot !== cwd) {
|
||||
return walkForGitNexusDir(canonicalRoot);
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
function parseRgGrepPattern(cmd) {
|
||||
const tokens = cmd.split(/\s+/);
|
||||
let foundCmd = false;
|
||||
let skipNext = false;
|
||||
const flagsWithValues = new Set([
|
||||
'-e',
|
||||
'-f',
|
||||
'-m',
|
||||
'-A',
|
||||
'-B',
|
||||
'-C',
|
||||
'-g',
|
||||
'--glob',
|
||||
'-t',
|
||||
'--type',
|
||||
'--include',
|
||||
'--exclude',
|
||||
]);
|
||||
|
||||
for (const token of tokens) {
|
||||
if (skipNext) {
|
||||
skipNext = false;
|
||||
continue;
|
||||
}
|
||||
if (!foundCmd) {
|
||||
if (/\brg$|\bgrep$/.test(token)) foundCmd = true;
|
||||
continue;
|
||||
}
|
||||
if (token.startsWith('-')) {
|
||||
if (flagsWithValues.has(token)) skipNext = true;
|
||||
continue;
|
||||
}
|
||||
const cleaned = token.replace(/['"]/g, '');
|
||||
return cleaned.length >= 3 ? cleaned : null;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/**
|
||||
* Extract a search pattern from the tool input. Cursor 2.4 docs at
|
||||
* https://cursor.com/docs/agent/hooks list the tool *matchers* but do not
|
||||
* formally specify the per-tool tool_input field names, so we probe a
|
||||
* generous set of MCP-style aliases. As a last-resort fallback for Grep
|
||||
* (the highest-frequency search path) we also accept the longest plausible
|
||||
* string value in tool_input. Set GITNEXUS_DEBUG=1 to log the raw payload
|
||||
* to stderr if Cursor changes the contract and aliases stop matching.
|
||||
*/
|
||||
function pickLongestStringValue(obj) {
|
||||
let best = null;
|
||||
if (!obj || typeof obj !== 'object') return null;
|
||||
for (const v of Object.values(obj)) {
|
||||
if (typeof v === 'string' && v.length >= 3 && (!best || v.length > best.length)) {
|
||||
best = v;
|
||||
}
|
||||
}
|
||||
return best;
|
||||
}
|
||||
|
||||
function extractPattern(toolName, toolInput) {
|
||||
const t = (toolName || '').toLowerCase();
|
||||
|
||||
if (t === 'grep') {
|
||||
const aliases = [
|
||||
toolInput.query,
|
||||
toolInput.pattern,
|
||||
toolInput.regex,
|
||||
toolInput.q,
|
||||
toolInput.search,
|
||||
toolInput.searchQuery,
|
||||
];
|
||||
for (const a of aliases) {
|
||||
if (typeof a === 'string' && a.length >= 3) return a;
|
||||
}
|
||||
// Last resort: scan tool_input for any reasonable-looking string value.
|
||||
return pickLongestStringValue(toolInput);
|
||||
}
|
||||
|
||||
if (t === 'read') {
|
||||
const filePath =
|
||||
toolInput.target_file ||
|
||||
toolInput.file_path ||
|
||||
toolInput.filePath ||
|
||||
toolInput.path ||
|
||||
toolInput.file ||
|
||||
'';
|
||||
if (!filePath) return null;
|
||||
const base = path.basename(String(filePath), path.extname(String(filePath)));
|
||||
const cleaned = base.replace(/[^a-zA-Z0-9_]/g, '');
|
||||
return cleaned.length >= 3 ? cleaned : null;
|
||||
}
|
||||
|
||||
if (t === 'shell') {
|
||||
const cmd = toolInput.command || '';
|
||||
if (!/\brg\b|\bgrep\b/.test(cmd)) return null;
|
||||
// NOTE: parseRgGrepPattern uses split(/\s+/) and cannot handle shell
|
||||
// quoting. `rg "User Service" src/` returns "User" (the first token
|
||||
// after the rg/grep arg, with surrounding quotes stripped) — the
|
||||
// multi-word pattern is intentionally not reconstructed since BM25 is
|
||||
// already token-tolerant. Quoted single tokens (`rg "validateUser"`)
|
||||
// work fine.
|
||||
return parseRgGrepPattern(cmd);
|
||||
}
|
||||
|
||||
return null;
|
||||
}
|
||||
|
||||
function resolveCliPath() {
|
||||
try {
|
||||
return require.resolve('gitnexus/dist/cli/index.js');
|
||||
} catch {
|
||||
return '';
|
||||
}
|
||||
}
|
||||
|
||||
function runGitNexusCli(cliPath, args, cwd, timeout) {
|
||||
const isWin = process.platform === 'win32';
|
||||
if (cliPath) {
|
||||
return spawnSync(process.execPath, [cliPath, ...args], {
|
||||
encoding: 'utf-8',
|
||||
timeout,
|
||||
cwd,
|
||||
stdio: ['pipe', 'pipe', 'pipe'],
|
||||
});
|
||||
}
|
||||
return spawnSync(isWin ? 'npx.cmd' : 'npx', ['-y', 'gitnexus', ...args], {
|
||||
encoding: 'utf-8',
|
||||
timeout: timeout + 5000,
|
||||
cwd,
|
||||
stdio: ['pipe', 'pipe', 'pipe'],
|
||||
});
|
||||
}
|
||||
|
||||
function main() {
|
||||
try {
|
||||
const input = readInput();
|
||||
if (process.env.GITNEXUS_DEBUG) {
|
||||
// Echo the payload so users can capture Cursor's actual contract when
|
||||
// diagnosing why augmentation isn't firing. Stderr only — stdout is
|
||||
// reserved for the JSON response Cursor consumes.
|
||||
try {
|
||||
process.stderr.write(
|
||||
`GitNexus Cursor hook stdin: ${JSON.stringify(input).slice(0, 500)}\n`,
|
||||
);
|
||||
} catch {
|
||||
/* never let debug logging break the hook */
|
||||
}
|
||||
}
|
||||
const cwd = input.cwd || process.cwd();
|
||||
if (!path.isAbsolute(cwd)) return;
|
||||
if (!findGitNexusDir(cwd)) return;
|
||||
|
||||
const toolName = input.tool_name || '';
|
||||
const toolInput = input.tool_input || {};
|
||||
|
||||
const pattern = extractPattern(toolName, toolInput);
|
||||
if (!pattern || pattern.length < 3) return;
|
||||
|
||||
const cliPath = resolveCliPath();
|
||||
let result = '';
|
||||
try {
|
||||
const child = runGitNexusCli(cliPath, ['augment', '--', pattern], cwd, 7000);
|
||||
if (!child.error && child.status === 0) {
|
||||
result = child.stderr || '';
|
||||
}
|
||||
} catch {
|
||||
/* graceful failure */
|
||||
}
|
||||
|
||||
if (result && result.trim()) {
|
||||
console.log(JSON.stringify({ additional_context: result.trim() }));
|
||||
}
|
||||
} catch (err) {
|
||||
if (process.env.GITNEXUS_DEBUG) {
|
||||
console.error('GitNexus Cursor hook error:', (err.message || '').slice(0, 200));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
main();
|
||||
@@ -1,11 +1,11 @@
|
||||
{
|
||||
"version": 1,
|
||||
"hooks": {
|
||||
"beforeShellExecution": [
|
||||
"postToolUse": [
|
||||
{
|
||||
"command": "./hooks/augment-shell.sh",
|
||||
"timeout": 5,
|
||||
"matcher": "\\brg\\b|\\bgrep\\b"
|
||||
"matcher": "Shell|Read|Grep",
|
||||
"command": "node ./hooks/gitnexus-hook.cjs",
|
||||
"timeout": 10
|
||||
}
|
||||
]
|
||||
}
|
||||
|
||||
+1
-1
@@ -33,7 +33,7 @@ To configure MCP for your editor, run `npx gitnexus setup` once — or set it up
|
||||
| Editor | MCP | Skills | Hooks (auto-augment) | Support |
|
||||
|--------|-----|--------|---------------------|---------|
|
||||
| **Claude Code** | Yes | Yes | Yes (PreToolUse) | **Full** |
|
||||
| **Cursor** | Yes | Yes | — | MCP + Skills |
|
||||
| **Cursor** | Yes | Yes | Yes (postToolUse, [manual install](../gitnexus-cursor-integration/README.md#hook-install)) | **Full** |
|
||||
| **Codex** | Yes | Yes | — | MCP + Skills |
|
||||
| **Windsurf** | Yes | — | — | MCP |
|
||||
| **OpenCode** | Yes | Yes | — | MCP + Skills |
|
||||
|
||||
Generated
+3
-3
@@ -1,12 +1,12 @@
|
||||
{
|
||||
"name": "gitnexus",
|
||||
"version": "1.6.4",
|
||||
"version": "1.6.5-rc.6",
|
||||
"lockfileVersion": 3,
|
||||
"requires": true,
|
||||
"packages": {
|
||||
"": {
|
||||
"name": "gitnexus",
|
||||
"version": "1.6.4",
|
||||
"version": "1.6.5-rc.6",
|
||||
"hasInstallScript": true,
|
||||
"license": "PolyForm-Noncommercial-1.0.0",
|
||||
"dependencies": {
|
||||
@@ -63,7 +63,7 @@
|
||||
"vitest": "^4.0.18"
|
||||
},
|
||||
"engines": {
|
||||
"node": ">=20.0.0"
|
||||
"node": ">=22.0.0"
|
||||
},
|
||||
"optionalDependencies": {
|
||||
"node-addon-api": "^8.0.0",
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "gitnexus",
|
||||
"version": "1.6.4",
|
||||
"version": "1.6.5-rc.6",
|
||||
"description": "Graph-powered code intelligence for AI agents. Index any codebase, query via MCP or CLI.",
|
||||
"author": "Abhigyan Patwari",
|
||||
"license": "PolyForm-Noncommercial-1.0.0",
|
||||
|
||||
@@ -2,7 +2,7 @@
|
||||
* Augment CLI Command
|
||||
*
|
||||
* Fast-path command for platform hooks.
|
||||
* Shells out from Claude Code PreToolUse / Cursor beforeShellExecution hooks.
|
||||
* Shells out from Claude Code PreToolUse / Cursor postToolUse hooks.
|
||||
*
|
||||
* Usage: gitnexus augment <pattern>
|
||||
* Returns enriched text to stdout.
|
||||
|
||||
@@ -2,8 +2,8 @@
|
||||
* Augmentation Engine
|
||||
*
|
||||
* Lightweight, fast-path enrichment of search patterns with knowledge graph context.
|
||||
* Designed to be called from platform hooks (Claude Code PreToolUse, Cursor beforeShellExecution)
|
||||
* when an agent runs grep/glob/search.
|
||||
* Designed to be called from platform hooks (Claude Code PreToolUse, Cursor postToolUse)
|
||||
* when an agent runs grep/glob/read/search.
|
||||
*
|
||||
* Performance target: <500ms cold start, <200ms warm.
|
||||
*
|
||||
|
||||
@@ -10,6 +10,7 @@ import {
|
||||
isLanguageAvailable,
|
||||
resolveLanguageKey,
|
||||
} from '../tree-sitter/parser-loader.js';
|
||||
import { parseSourceSafe } from '../tree-sitter/safe-parse.js';
|
||||
|
||||
const parserCache = new Map<string, any>();
|
||||
|
||||
@@ -29,7 +30,7 @@ export const ensureAndParse = async (content: string, filePath: string): Promise
|
||||
parserCache.set(parserKey, parserInstance);
|
||||
}
|
||||
|
||||
return parserInstance.parse(content);
|
||||
return parseSourceSafe(parserInstance, content);
|
||||
};
|
||||
|
||||
const FUNCTION_LIKE_TYPES = new Set([
|
||||
|
||||
@@ -5,6 +5,7 @@ import { createIgnoreFilter } from '../../../config/ignore-service.js';
|
||||
import type { ContractExtractor, CypherExecutor } from '../contract-extractor.js';
|
||||
import type { ExtractedContract, RepoHandle } from '../types.js';
|
||||
import { readSafe } from './fs-utils.js';
|
||||
import { parseSourceSafe } from '../../tree-sitter/safe-parse.js';
|
||||
import { logger } from '../../logger.js';
|
||||
import {
|
||||
GRPC_SCAN_GLOB,
|
||||
@@ -428,7 +429,7 @@ export class GrpcExtractor implements ContractExtractor {
|
||||
let detections: GrpcDetection[] = [];
|
||||
try {
|
||||
parser.setLanguage(plugin.language);
|
||||
const tree = parser.parse(content);
|
||||
const tree = parseSourceSafe(parser, content);
|
||||
detections = plugin.scan(tree);
|
||||
} catch {
|
||||
continue;
|
||||
|
||||
@@ -5,6 +5,7 @@ import { createIgnoreFilter } from '../../../config/ignore-service.js';
|
||||
import type { ContractExtractor, CypherExecutor } from '../contract-extractor.js';
|
||||
import type { ExtractedContract, RepoHandle } from '../types.js';
|
||||
import { readSafe } from './fs-utils.js';
|
||||
import { parseSourceSafe } from '../../tree-sitter/safe-parse.js';
|
||||
import { getPluginForFile, HTTP_SCAN_GLOB, type HttpDetection } from './http-patterns/index.js';
|
||||
|
||||
/**
|
||||
@@ -172,7 +173,7 @@ export class HttpRouteExtractor implements ContractExtractor {
|
||||
}
|
||||
try {
|
||||
parser.setLanguage(plugin.language);
|
||||
const tree = parser.parse(content);
|
||||
const tree = parseSourceSafe(parser, content);
|
||||
const detections = plugin.scan(tree);
|
||||
cachedDetections.set(rel, detections);
|
||||
return detections;
|
||||
|
||||
@@ -10,6 +10,7 @@ import { readSafe } from './fs-utils.js';
|
||||
import { buildSuffixIndex, type SuffixIndex } from '../../ingestion/import-resolvers/utils.js';
|
||||
import { createIgnoreFilter } from '../../../config/ignore-service.js';
|
||||
import { getMaxFileSizeBytes } from '../../ingestion/utils/max-file-size.js';
|
||||
import { parseSourceSafe } from '../../tree-sitter/safe-parse.js';
|
||||
import { logger } from '../../logger.js';
|
||||
|
||||
/**
|
||||
@@ -505,7 +506,7 @@ export class IncludeExtractor implements ContractExtractor {
|
||||
let extractionSource: 'tree_sitter' | 'regex_fallback';
|
||||
try {
|
||||
parser.setLanguage(lang);
|
||||
const tree = parser.parse(content);
|
||||
const tree = parseSourceSafe(parser, content);
|
||||
let matches: Parser.QueryMatch[];
|
||||
try {
|
||||
matches = query.matches(tree.rootNode);
|
||||
|
||||
@@ -3,6 +3,7 @@ import Parser from 'tree-sitter';
|
||||
import type { ContractExtractor, CypherExecutor } from '../contract-extractor.js';
|
||||
import type { ExtractedContract, RepoHandle } from '../types.js';
|
||||
import { readSafe } from './fs-utils.js';
|
||||
import { parseSourceSafe } from '../../tree-sitter/safe-parse.js';
|
||||
import {
|
||||
getPluginForFile,
|
||||
THRIFT_SCAN_GLOB,
|
||||
@@ -311,7 +312,7 @@ export class ThriftExtractor implements ContractExtractor {
|
||||
let detections: ThriftDetection[] = [];
|
||||
try {
|
||||
parser.setLanguage(plugin.language);
|
||||
const tree = parser.parse(content);
|
||||
const tree = parseSourceSafe(parser, content);
|
||||
detections = plugin.scan(tree);
|
||||
} catch {
|
||||
continue;
|
||||
|
||||
@@ -1,4 +1,5 @@
|
||||
import Parser from 'tree-sitter';
|
||||
import { parseSourceSafe } from '../../tree-sitter/safe-parse.js';
|
||||
|
||||
/**
|
||||
* Shared, language-agnostic tree-sitter scanning utilities used by group
|
||||
@@ -155,7 +156,7 @@ export function scanFile<TMeta>(
|
||||
let tree: Parser.Tree;
|
||||
try {
|
||||
parser.setLanguage(plugin.language);
|
||||
tree = parser.parse(content);
|
||||
tree = parseSourceSafe(parser, content);
|
||||
} catch {
|
||||
return [];
|
||||
}
|
||||
|
||||
@@ -40,6 +40,7 @@ import { getLanguageFromFilename, SupportedLanguages } from 'gitnexus-shared';
|
||||
import { isRegistryPrimary } from './registry-primary-flag.js';
|
||||
import { isVerboseIngestionEnabled } from './utils/verbose.js';
|
||||
import { yieldToEventLoop } from './utils/event-loop.js';
|
||||
import { parseSourceSafe } from '../tree-sitter/safe-parse.js';
|
||||
import {
|
||||
FUNCTION_NODE_TYPES,
|
||||
findEnclosingClassId,
|
||||
@@ -771,7 +772,7 @@ export const processCalls = async (
|
||||
if (!tree) {
|
||||
const parseContent = provider.preprocessSource?.(file.content, file.path) ?? file.content;
|
||||
try {
|
||||
tree = parser.parse(parseContent, undefined, {
|
||||
tree = parseSourceSafe(parser, parseContent, undefined, {
|
||||
bufferSize: getTreeSitterBufferSize(parseContent),
|
||||
});
|
||||
} catch (parseError) {
|
||||
@@ -3283,7 +3284,7 @@ export const extractFetchCallsFromFiles = async (
|
||||
if (!tree) {
|
||||
const parseContent = provider.preprocessSource?.(file.content, file.path) ?? file.content;
|
||||
try {
|
||||
tree = parser.parse(parseContent, undefined, {
|
||||
tree = parseSourceSafe(parser, parseContent, undefined, {
|
||||
bufferSize: getTreeSitterBufferSize(parseContent),
|
||||
});
|
||||
} catch {
|
||||
|
||||
@@ -22,6 +22,7 @@ import { generateId } from '../../lib/utils.js';
|
||||
import { getLanguageFromFilename, type NodeLabel, type SupportedLanguages } from 'gitnexus-shared';
|
||||
import { isVerboseIngestionEnabled } from './utils/verbose.js';
|
||||
import { yieldToEventLoop } from './utils/event-loop.js';
|
||||
import { parseSourceSafe } from '../tree-sitter/safe-parse.js';
|
||||
import { getProvider } from './languages/index.js';
|
||||
import { getTreeSitterBufferSize } from './constants.js';
|
||||
import type {
|
||||
@@ -224,7 +225,7 @@ export const processHeritage = async (
|
||||
// re-parses see the same input as the cached AST.
|
||||
const parseContent = provider.preprocessSource?.(file.content, file.path) ?? file.content;
|
||||
try {
|
||||
tree = parser.parse(parseContent, undefined, {
|
||||
tree = parseSourceSafe(parser, parseContent, undefined, {
|
||||
bufferSize: getTreeSitterBufferSize(parseContent),
|
||||
});
|
||||
} catch (parseError) {
|
||||
@@ -419,7 +420,7 @@ export async function extractExtractedHeritageFromFiles(
|
||||
if (!tree) {
|
||||
const parseContent = provider.preprocessSource?.(file.content, file.path) ?? file.content;
|
||||
try {
|
||||
tree = parser.parse(parseContent, undefined, {
|
||||
tree = parseSourceSafe(parser, parseContent, undefined, {
|
||||
bufferSize: getTreeSitterBufferSize(parseContent),
|
||||
});
|
||||
} catch {
|
||||
|
||||
@@ -8,6 +8,7 @@ import { generateId } from '../../lib/utils.js';
|
||||
import { getLanguageFromFilename } from 'gitnexus-shared';
|
||||
import { isVerboseIngestionEnabled } from './utils/verbose.js';
|
||||
import { yieldToEventLoop } from './utils/event-loop.js';
|
||||
import { parseSourceSafe } from '../tree-sitter/safe-parse.js';
|
||||
import type { ExtractedImport } from './workers/parse-worker.js';
|
||||
import { getTreeSitterBufferSize } from './constants.js';
|
||||
import { loadImportConfigs } from './language-config.js';
|
||||
@@ -307,7 +308,7 @@ export const processImports = async (
|
||||
if (!tree) {
|
||||
const parseContent = provider.preprocessSource?.(file.content, file.path) ?? file.content;
|
||||
try {
|
||||
tree = parser.parse(parseContent, undefined, {
|
||||
tree = parseSourceSafe(parser, parseContent, undefined, {
|
||||
bufferSize: getTreeSitterBufferSize(parseContent),
|
||||
});
|
||||
} catch (parseError) {
|
||||
|
||||
@@ -46,6 +46,15 @@ import { createCallExtractor } from '../call-extractors/generic.js';
|
||||
import { cCallConfig, cppCallConfig } from '../call-extractors/configs/c-cpp.js';
|
||||
import { createHeritageExtractor } from '../heritage-extractors/generic.js';
|
||||
import { stripUeMacros } from '../cpp-ue-preprocessor.js';
|
||||
import {
|
||||
emitCScopeCaptures,
|
||||
interpretCImport,
|
||||
interpretCTypeBinding,
|
||||
cArityCompatibility,
|
||||
cBindingScopeFor,
|
||||
cImportOwningScope,
|
||||
cReceiverBinding,
|
||||
} from './c/index.js';
|
||||
|
||||
const C_BUILT_INS: ReadonlySet<string> = new Set([
|
||||
'printf',
|
||||
@@ -367,6 +376,16 @@ export const cProvider = defineLanguage({
|
||||
heritageExtractor: createHeritageExtractor(SupportedLanguages.C),
|
||||
labelOverride: cppLabelOverride,
|
||||
builtInNames: C_BUILT_INS,
|
||||
|
||||
// ── RFC #909 Ring 3: scope-based resolution hooks (RFC §5) ──────────
|
||||
emitScopeCaptures: emitCScopeCaptures,
|
||||
interpretImport: interpretCImport,
|
||||
interpretTypeBinding: interpretCTypeBinding,
|
||||
bindingScopeFor: cBindingScopeFor,
|
||||
importOwningScope: cImportOwningScope,
|
||||
receiverBinding: cReceiverBinding,
|
||||
arityCompatibility: cArityCompatibility,
|
||||
// mergeBindings + resolveImportTarget live on ScopeResolver (see c/scope-resolver.ts).
|
||||
});
|
||||
|
||||
export const cppProvider = defineLanguage({
|
||||
|
||||
@@ -0,0 +1,102 @@
|
||||
import type { SyntaxNode } from '../../utils/ast-helpers.js';
|
||||
|
||||
export interface CArityInfo {
|
||||
parameterCount?: number;
|
||||
requiredParameterCount?: number;
|
||||
parameterTypes?: string[];
|
||||
}
|
||||
|
||||
/**
|
||||
* Compute declaration arity from a C function definition or declaration node.
|
||||
*/
|
||||
export function computeCDeclarationArity(node: SyntaxNode): CArityInfo {
|
||||
// Find the function_declarator child (may be wrapped in pointer_declarator)
|
||||
const funcDecl = findFuncDeclarator(node);
|
||||
if (funcDecl === null) return {};
|
||||
|
||||
const paramList = funcDecl.childForFieldName('parameters');
|
||||
if (paramList === null) return {};
|
||||
|
||||
const params: SyntaxNode[] = [];
|
||||
for (let i = 0; i < paramList.childCount; i++) {
|
||||
const child = paramList.child(i);
|
||||
if (child === null) continue;
|
||||
if (child.type === 'parameter_declaration' || child.type === 'variadic_parameter') {
|
||||
params.push(child);
|
||||
}
|
||||
}
|
||||
|
||||
// K&R old-style declaration: `int foo()` has an empty parameter_list with
|
||||
// no parameter_declaration or variadic_parameter children. Per C89/C99,
|
||||
// this means the function accepts an unspecified number/types of arguments —
|
||||
// NOT zero arguments. Return unknown arity to avoid false 'incompatible'.
|
||||
// `int foo(void)` is the explicit zero-parameter form and is handled below.
|
||||
if (params.length === 0) return {};
|
||||
|
||||
// (void) means zero parameters
|
||||
if (params.length === 1 && params[0].type === 'parameter_declaration') {
|
||||
const typeNode = params[0].childForFieldName('type');
|
||||
const hasDeclarator = params[0].childForFieldName('declarator') !== null;
|
||||
if (typeNode !== null && typeNode.text === 'void' && !hasDeclarator) {
|
||||
return { parameterCount: 0, requiredParameterCount: 0, parameterTypes: [] };
|
||||
}
|
||||
}
|
||||
|
||||
const isVariadic = params.some((p) => p.type === 'variadic_parameter');
|
||||
const nonVariadicCount = params.filter((p) => p.type !== 'variadic_parameter').length;
|
||||
|
||||
const types: string[] = [];
|
||||
for (const p of params) {
|
||||
if (p.type === 'variadic_parameter') {
|
||||
types.push('...');
|
||||
} else {
|
||||
const typeNode = p.childForFieldName('type');
|
||||
types.push(typeNode?.text ?? 'unknown');
|
||||
}
|
||||
}
|
||||
|
||||
return {
|
||||
parameterCount: isVariadic ? undefined : nonVariadicCount,
|
||||
requiredParameterCount: nonVariadicCount,
|
||||
parameterTypes: types,
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Compute call-site arity from a call_expression node.
|
||||
*/
|
||||
export function computeCCallArity(node: SyntaxNode): number {
|
||||
const argList = node.childForFieldName('arguments');
|
||||
if (argList === null) return 0;
|
||||
|
||||
let count = 0;
|
||||
for (let i = 0; i < argList.childCount; i++) {
|
||||
const child = argList.child(i);
|
||||
if (child === null) continue;
|
||||
// Skip punctuation (commas, parens)
|
||||
if (child.type !== ',' && child.type !== '(' && child.type !== ')') {
|
||||
count++;
|
||||
}
|
||||
}
|
||||
return count;
|
||||
}
|
||||
|
||||
function findFuncDeclarator(node: SyntaxNode): SyntaxNode | null {
|
||||
// Direct child
|
||||
let decl = node.childForFieldName('declarator');
|
||||
if (decl === null) {
|
||||
for (let i = 0; i < node.childCount; i++) {
|
||||
const c = node.child(i);
|
||||
if (c?.type === 'function_declarator') return c;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
// Unwrap pointer_declarator
|
||||
while (decl.type === 'pointer_declarator') {
|
||||
const next = decl.childForFieldName('declarator');
|
||||
if (next === null) break;
|
||||
decl = next;
|
||||
}
|
||||
if (decl.type === 'function_declarator') return decl;
|
||||
return null;
|
||||
}
|
||||
@@ -0,0 +1,20 @@
|
||||
import type { Callsite, SymbolDefinition } from 'gitnexus-shared';
|
||||
|
||||
/**
|
||||
* C arity compatibility: no overloading. Variadic functions detected
|
||||
* via '...' in parameterTypes. Otherwise exact match or unknown.
|
||||
*/
|
||||
export function cArityCompatibility(
|
||||
def: SymbolDefinition,
|
||||
callsite: Callsite,
|
||||
): 'compatible' | 'unknown' | 'incompatible' {
|
||||
const max = def.parameterCount;
|
||||
const min = def.requiredParameterCount;
|
||||
if (max === undefined && min === undefined) return 'unknown';
|
||||
if (!Number.isFinite(callsite.arity) || callsite.arity < 0) return 'unknown';
|
||||
|
||||
const variadic = def.parameterTypes?.some((t) => t === '...') ?? false;
|
||||
if (min !== undefined && callsite.arity < min) return 'incompatible';
|
||||
if (max !== undefined && callsite.arity > max && !variadic) return 'incompatible';
|
||||
return 'compatible';
|
||||
}
|
||||
@@ -0,0 +1,142 @@
|
||||
import type { Capture, CaptureMatch } from 'gitnexus-shared';
|
||||
import {
|
||||
findNodeAtRange,
|
||||
nodeToCapture,
|
||||
syntheticCapture,
|
||||
type SyntaxNode,
|
||||
} from '../../utils/ast-helpers.js';
|
||||
import { getCParser, getCScopeQuery } from './query.js';
|
||||
import { getTreeSitterBufferSize } from '../../constants.js';
|
||||
import { parseSourceSafe } from '../../../tree-sitter/safe-parse.js';
|
||||
import { splitCInclude } from './import-decomposer.js';
|
||||
import { computeCDeclarationArity, computeCCallArity } from './arity-metadata.js';
|
||||
import { markStaticName } from './static-linkage.js';
|
||||
|
||||
export function emitCScopeCaptures(
|
||||
sourceText: string,
|
||||
filePath: string,
|
||||
cachedTree?: unknown,
|
||||
): readonly CaptureMatch[] {
|
||||
let tree = cachedTree as ReturnType<ReturnType<typeof getCParser>['parse']> | undefined;
|
||||
if (tree === undefined) {
|
||||
tree = parseSourceSafe(getCParser(), sourceText, undefined, {
|
||||
bufferSize: getTreeSitterBufferSize(sourceText),
|
||||
});
|
||||
}
|
||||
|
||||
const rawMatches = getCScopeQuery().matches(tree.rootNode);
|
||||
const out: CaptureMatch[] = [];
|
||||
|
||||
// Track ranges where typedef-struct/union was captured as @declaration.struct/union
|
||||
// so we can suppress the duplicate @declaration.typedef match at the same range.
|
||||
const structTypedefRanges = new Set<string>();
|
||||
|
||||
for (const m of rawMatches) {
|
||||
const grouped: Record<string, Capture> = {};
|
||||
for (const c of m.captures) {
|
||||
const tag = '@' + c.name;
|
||||
if (tag.startsWith('@_')) continue;
|
||||
grouped[tag] = nodeToCapture(tag, c.node);
|
||||
}
|
||||
if (Object.keys(grouped).length === 0) continue;
|
||||
|
||||
// Handle #include statements
|
||||
if (grouped['@import.statement'] !== undefined) {
|
||||
const anchor = grouped['@import.statement']!;
|
||||
const includeNode = findNodeAtRange(tree.rootNode, anchor.range, 'preproc_include');
|
||||
if (includeNode !== null) {
|
||||
const split = splitCInclude(includeNode);
|
||||
if (split !== null) {
|
||||
out.push(split);
|
||||
continue;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Track typedef-struct ranges to suppress duplicate typedef declarations
|
||||
const structAnchor = grouped['@declaration.struct'] ?? grouped['@declaration.union'];
|
||||
if (structAnchor !== undefined) {
|
||||
const r = structAnchor.range;
|
||||
structTypedefRanges.add(`${r.startLine}:${r.startCol}:${r.endLine}:${r.endCol}`);
|
||||
}
|
||||
|
||||
// Suppress @declaration.typedef if the same range was already captured as struct/union
|
||||
const typedefAnchor = grouped['@declaration.typedef'];
|
||||
if (typedefAnchor !== undefined) {
|
||||
const r = typedefAnchor.range;
|
||||
const key = `${r.startLine}:${r.startCol}:${r.endLine}:${r.endCol}`;
|
||||
if (structTypedefRanges.has(key)) continue;
|
||||
}
|
||||
|
||||
// Enrich function declarations with arity metadata and detect static linkage
|
||||
const declAnchor = grouped['@declaration.function'];
|
||||
if (declAnchor !== undefined) {
|
||||
const fnNode =
|
||||
findNodeAtRange(tree.rootNode, declAnchor.range, 'function_definition') ??
|
||||
findNodeAtRange(tree.rootNode, declAnchor.range, 'declaration');
|
||||
if (fnNode !== null) {
|
||||
const arity = computeCDeclarationArity(fnNode);
|
||||
if (arity.parameterCount !== undefined) {
|
||||
grouped['@declaration.parameter-count'] = syntheticCapture(
|
||||
'@declaration.parameter-count',
|
||||
fnNode,
|
||||
String(arity.parameterCount),
|
||||
);
|
||||
}
|
||||
if (arity.requiredParameterCount !== undefined) {
|
||||
grouped['@declaration.required-parameter-count'] = syntheticCapture(
|
||||
'@declaration.required-parameter-count',
|
||||
fnNode,
|
||||
String(arity.requiredParameterCount),
|
||||
);
|
||||
}
|
||||
if (arity.parameterTypes !== undefined) {
|
||||
grouped['@declaration.parameter-types'] = syntheticCapture(
|
||||
'@declaration.parameter-types',
|
||||
fnNode,
|
||||
JSON.stringify(arity.parameterTypes),
|
||||
);
|
||||
}
|
||||
|
||||
// Detect static storage class (file-local linkage)
|
||||
if (hasStaticStorageClass(fnNode)) {
|
||||
const nameText = grouped['@declaration.name']?.text;
|
||||
if (nameText !== undefined) {
|
||||
markStaticName(filePath, nameText);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Enrich call references with arity
|
||||
const callAnchor = grouped['@reference.call.free'] ?? grouped['@reference.call.member'];
|
||||
if (callAnchor !== undefined && grouped['@reference.arity'] === undefined) {
|
||||
const callNode = findNodeAtRange(tree.rootNode, callAnchor.range, 'call_expression');
|
||||
if (callNode !== null) {
|
||||
grouped['@reference.arity'] = syntheticCapture(
|
||||
'@reference.arity',
|
||||
callNode,
|
||||
String(computeCCallArity(callNode)),
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
out.push(grouped);
|
||||
}
|
||||
|
||||
return out;
|
||||
}
|
||||
|
||||
/**
|
||||
* Check if a C function_definition or declaration has `static` storage class.
|
||||
* Walks direct children for a `storage_class_specifier` node with text `static`.
|
||||
*/
|
||||
function hasStaticStorageClass(node: SyntaxNode): boolean {
|
||||
for (let i = 0; i < node.childCount; i++) {
|
||||
const child = node.child(i);
|
||||
if (child !== null && child.type === 'storage_class_specifier' && child.text === 'static') {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
return false;
|
||||
}
|
||||
@@ -0,0 +1,58 @@
|
||||
import { readdirSync, type Dirent } from 'fs';
|
||||
import { join, relative } from 'path';
|
||||
|
||||
/** C header extensions to scan for in the workspace. */
|
||||
const HEADER_EXTENSIONS = new Set(['.h']);
|
||||
|
||||
/**
|
||||
* Walk `repoPath` recursively and return relative paths of all `.h` files.
|
||||
* Used by `loadResolutionConfig` so the C resolver can resolve `#include`
|
||||
* targets that live in `.h` files (classified as C++ by language detection
|
||||
* but importable from `.c` files).
|
||||
*/
|
||||
export function scanHeaderFiles(repoPath: string): ReadonlySet<string> {
|
||||
const headers = new Set<string>();
|
||||
walk(repoPath, repoPath, headers);
|
||||
return headers;
|
||||
}
|
||||
|
||||
function walk(dir: string, root: string, out: Set<string>): void {
|
||||
let entries: Dirent[];
|
||||
try {
|
||||
entries = readdirSync(dir, { withFileTypes: true, encoding: 'utf8' });
|
||||
} catch {
|
||||
return; // permission denied, etc.
|
||||
}
|
||||
for (const entry of entries) {
|
||||
const name = entry.name;
|
||||
const full = join(dir, name);
|
||||
if (entry.isDirectory()) {
|
||||
// Skip common non-source directories and build output dirs.
|
||||
// Build dirs (dist, build, out, target, _build, .next, cmake-build-*)
|
||||
// may contain generated headers that shadow source headers.
|
||||
if (
|
||||
name === 'node_modules' ||
|
||||
name === '.git' ||
|
||||
name === 'vendor' ||
|
||||
name === 'dist' ||
|
||||
name === 'build' ||
|
||||
name === 'out' ||
|
||||
name === 'target' ||
|
||||
name === '_build' ||
|
||||
name === '.next' ||
|
||||
name.startsWith('cmake-build')
|
||||
) {
|
||||
continue;
|
||||
}
|
||||
walk(full, root, out);
|
||||
} else if (entry.isFile()) {
|
||||
const ext = name.slice(name.lastIndexOf('.'));
|
||||
if (HEADER_EXTENSIONS.has(ext)) {
|
||||
// Normalize to forward slashes for cross-platform consistency.
|
||||
// path.relative() returns backslash-separated paths on Windows,
|
||||
// but the scope-resolution pipeline uses forward slashes uniformly.
|
||||
out.add(relative(root, full).replace(/\\/g, '/'));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,55 @@
|
||||
import type { Capture, CaptureMatch } from 'gitnexus-shared';
|
||||
import { nodeToCapture, syntheticCapture, type SyntaxNode } from '../../utils/ast-helpers.js';
|
||||
|
||||
/**
|
||||
* Decompose a `preproc_include` node into a CaptureMatch with structured
|
||||
* import captures. C #include maps to a wildcard import (all symbols
|
||||
* from the header are visible).
|
||||
*/
|
||||
export function splitCInclude(node: SyntaxNode): CaptureMatch | null {
|
||||
// node.type === 'preproc_include'
|
||||
// path field: (string_literal (string_content)) | (system_lib_string)
|
||||
const pathNode = node.childForFieldName?.('path') ?? null;
|
||||
if (pathNode === null) {
|
||||
// Fallback: scan children
|
||||
for (let i = 0; i < node.childCount; i++) {
|
||||
const child = node.child(i);
|
||||
if (child === null) continue;
|
||||
if (child.type === 'string_literal' || child.type === 'system_lib_string') {
|
||||
return buildIncludeCapture(node, child);
|
||||
}
|
||||
}
|
||||
return null;
|
||||
}
|
||||
return buildIncludeCapture(node, pathNode);
|
||||
}
|
||||
|
||||
function buildIncludeCapture(node: SyntaxNode, pathNode: SyntaxNode): CaptureMatch {
|
||||
let raw: string;
|
||||
if (pathNode.type === 'string_literal') {
|
||||
// string_literal has children: `"`, string_content, `"`
|
||||
// Use namedChildren to find the string_content node
|
||||
const content = pathNode.namedChildren.find((c) => c.type === 'string_content');
|
||||
raw = content?.text ?? pathNode.text.replace(/^"|"$/g, '');
|
||||
} else {
|
||||
// system_lib_string: <stdio.h> → strip angle brackets
|
||||
raw = pathNode.text;
|
||||
if (raw.startsWith('<') && raw.endsWith('>')) {
|
||||
raw = raw.slice(1, -1);
|
||||
}
|
||||
}
|
||||
|
||||
const isSystem = pathNode.type === 'system_lib_string';
|
||||
|
||||
const result: Record<string, Capture> = {
|
||||
'@import.statement': nodeToCapture('@import.statement', node),
|
||||
'@import.kind': syntheticCapture('@import.kind', node, 'wildcard'),
|
||||
'@import.source': syntheticCapture('@import.source', node, raw),
|
||||
};
|
||||
|
||||
if (isSystem) {
|
||||
result['@import.system'] = syntheticCapture('@import.system', node, 'true');
|
||||
}
|
||||
|
||||
return result;
|
||||
}
|
||||
@@ -0,0 +1,64 @@
|
||||
import { dirname, join } from 'path';
|
||||
|
||||
/**
|
||||
* Resolve a C #include path to a file in the workspace.
|
||||
*
|
||||
* Strategy:
|
||||
* 1. Check for a same-directory sibling relative to the including file
|
||||
* (matches C compiler `#include "…"` relative-lookup semantics).
|
||||
* 2. Check for an exact match (path as-is in the workspace).
|
||||
* 3. Fall back to suffix matching against all workspace file paths.
|
||||
* Tie-breaking: prefer the match with the fewest path components
|
||||
* (closest to root). On equal depth, break ties lexicographically
|
||||
* by normalized path to ensure deterministic resolution regardless
|
||||
* of filesystem iteration order.
|
||||
*/
|
||||
export function resolveCImportTarget(
|
||||
targetRaw: string,
|
||||
fromFile: string,
|
||||
allFilePaths: ReadonlySet<string>,
|
||||
): string | null {
|
||||
if (!targetRaw) return null;
|
||||
|
||||
const normalizedTarget = targetRaw.replace(/\\/g, '/');
|
||||
|
||||
// Same-directory sibling first: mirrors the C compiler's #include "…"
|
||||
// relative-lookup semantics where the directory of the including
|
||||
// file is searched before the include-path list.
|
||||
if (fromFile) {
|
||||
const siblingRaw = join(dirname(fromFile), targetRaw);
|
||||
const sibling = siblingRaw.replace(/\\/g, '/');
|
||||
if (allFilePaths.has(sibling)) return sibling;
|
||||
// When targetRaw contains backslashes, the normalized form may
|
||||
// resolve to a different sibling path — try it as well.
|
||||
if (targetRaw !== normalizedTarget) {
|
||||
const siblingAlt = join(dirname(fromFile), normalizedTarget);
|
||||
const siblingAltNorm = siblingAlt.replace(/\\/g, '/');
|
||||
if (allFilePaths.has(siblingAltNorm)) return siblingAltNorm;
|
||||
}
|
||||
}
|
||||
|
||||
// Exact match (path as-is in the workspace)
|
||||
if (allFilePaths.has(normalizedTarget)) return normalizedTarget;
|
||||
|
||||
// Suffix match: find files ending with /targetRaw or equal to targetRaw
|
||||
const suffix = '/' + normalizedTarget;
|
||||
let bestMatch: string | null = null;
|
||||
let bestDepth = Infinity;
|
||||
let bestNormalized = '';
|
||||
|
||||
for (const filePath of allFilePaths) {
|
||||
const normalized = filePath.replace(/\\/g, '/');
|
||||
if (normalized === normalizedTarget || normalized.endsWith(suffix)) {
|
||||
// Prefer shortest path (closest match)
|
||||
const depth = normalized.split('/').length;
|
||||
if (depth < bestDepth || (depth === bestDepth && normalized < bestNormalized)) {
|
||||
bestDepth = depth;
|
||||
bestMatch = filePath;
|
||||
bestNormalized = normalized;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return bestMatch;
|
||||
}
|
||||
@@ -0,0 +1,16 @@
|
||||
/**
|
||||
* C scope-resolution hooks (RFC #909 Ring 3).
|
||||
*/
|
||||
export { emitCScopeCaptures } from './captures.js';
|
||||
export { interpretCImport, interpretCTypeBinding, normalizeCTypeName } from './interpret.js';
|
||||
export { splitCInclude } from './import-decomposer.js';
|
||||
export { cArityCompatibility } from './arity.js';
|
||||
export { cMergeBindings } from './merge-bindings.js';
|
||||
export { cBindingScopeFor, cImportOwningScope, cReceiverBinding } from './simple-hooks.js';
|
||||
export { resolveCImportTarget } from './import-target.js';
|
||||
export {
|
||||
markStaticName,
|
||||
isStaticName,
|
||||
clearStaticNames,
|
||||
expandCWildcardNames,
|
||||
} from './static-linkage.js';
|
||||
@@ -0,0 +1,51 @@
|
||||
import type { CaptureMatch, ParsedImport, ParsedTypeBinding, TypeRef } from 'gitnexus-shared';
|
||||
|
||||
/**
|
||||
* Interpret a C #include capture into a ParsedImport.
|
||||
* C includes are always wildcard imports (all symbols from the header).
|
||||
*/
|
||||
export function interpretCImport(captures: CaptureMatch): ParsedImport | null {
|
||||
const source = captures['@import.source']?.text;
|
||||
if (source === undefined) return null;
|
||||
|
||||
// System headers (e.g. <stdio.h>) are not resolved to local files
|
||||
if (captures['@import.system'] !== undefined) return null;
|
||||
|
||||
return { kind: 'wildcard', targetRaw: source };
|
||||
}
|
||||
|
||||
/**
|
||||
* Interpret a C type-binding capture into a ParsedTypeBinding.
|
||||
*/
|
||||
export function interpretCTypeBinding(captures: CaptureMatch): ParsedTypeBinding | null {
|
||||
const name = captures['@type-binding.name']?.text;
|
||||
const type = captures['@type-binding.type']?.text;
|
||||
if (name === undefined || type === undefined) return null;
|
||||
|
||||
let source: TypeRef['source'] = 'annotation';
|
||||
|
||||
if (captures['@type-binding.parameter'] !== undefined) {
|
||||
source = 'parameter-annotation';
|
||||
} else if (captures['@type-binding.assignment'] !== undefined) {
|
||||
source = 'assignment-inferred';
|
||||
}
|
||||
|
||||
return { boundName: name, rawTypeName: normalizeCTypeName(type), source };
|
||||
}
|
||||
|
||||
/**
|
||||
* Normalize a C type name: strip pointer/array syntax, qualifiers.
|
||||
*/
|
||||
export function normalizeCTypeName(text: string): string {
|
||||
let t = text.trim();
|
||||
// Strip const, volatile, restrict qualifiers
|
||||
t = t.replace(/\b(const|volatile|restrict|static|extern|inline)\b/g, '').trim();
|
||||
// Strip pointer stars
|
||||
while (t.endsWith('*')) t = t.slice(0, -1).trim();
|
||||
while (t.startsWith('*')) t = t.slice(1).trim();
|
||||
// Strip array brackets
|
||||
t = t.replace(/\[.*?\]/g, '').trim();
|
||||
// Strip struct/union/enum prefixes
|
||||
t = t.replace(/^(struct|union|enum)\s+/, '');
|
||||
return t;
|
||||
}
|
||||
@@ -0,0 +1,32 @@
|
||||
import type { BindingRef } from 'gitnexus-shared';
|
||||
|
||||
const TIER: Record<BindingRef['origin'], number> = {
|
||||
local: 0,
|
||||
namespace: 1,
|
||||
import: 2,
|
||||
reexport: 3,
|
||||
wildcard: 4,
|
||||
};
|
||||
|
||||
/**
|
||||
* C merge bindings: simple first-wins by tier (local > import > wildcard).
|
||||
* C has no namespaces or reexports, but the tiers are defined for
|
||||
* compatibility with the shared infrastructure.
|
||||
*/
|
||||
export function cMergeBindings(
|
||||
existing: readonly BindingRef[],
|
||||
incoming: readonly BindingRef[],
|
||||
_scopeId: string,
|
||||
): BindingRef[] {
|
||||
const seen = new Set<string>();
|
||||
return [...existing, ...incoming]
|
||||
.sort(
|
||||
(a, b) =>
|
||||
(TIER[a.origin] ?? 99) - (TIER[b.origin] ?? 99) || a.def.nodeId.localeCompare(b.def.nodeId),
|
||||
)
|
||||
.filter((binding) => {
|
||||
if (seen.has(binding.def.nodeId)) return false;
|
||||
seen.add(binding.def.nodeId);
|
||||
return true;
|
||||
});
|
||||
}
|
||||
@@ -0,0 +1,165 @@
|
||||
import Parser from 'tree-sitter';
|
||||
import C from 'tree-sitter-c';
|
||||
|
||||
const C_SCOPE_QUERY = `
|
||||
;; Scopes
|
||||
(translation_unit) @scope.module
|
||||
(struct_specifier) @scope.class
|
||||
(union_specifier) @scope.class
|
||||
(function_definition) @scope.function
|
||||
(compound_statement) @scope.block
|
||||
(if_statement) @scope.block
|
||||
(for_statement) @scope.block
|
||||
(while_statement) @scope.block
|
||||
(do_statement) @scope.block
|
||||
(switch_statement) @scope.block
|
||||
(case_statement) @scope.block
|
||||
|
||||
;; Declarations — struct (named)
|
||||
(struct_specifier
|
||||
name: (type_identifier) @declaration.name
|
||||
body: (field_declaration_list)) @declaration.struct
|
||||
|
||||
;; Declarations — struct (typedef struct { ... } Name)
|
||||
(type_definition
|
||||
type: (struct_specifier
|
||||
body: (field_declaration_list))
|
||||
declarator: (type_identifier) @declaration.name) @declaration.struct
|
||||
|
||||
;; Declarations — union (named)
|
||||
(union_specifier
|
||||
name: (type_identifier) @declaration.name
|
||||
body: (field_declaration_list)) @declaration.union
|
||||
|
||||
;; Declarations — union (typedef union { ... } Name)
|
||||
(type_definition
|
||||
type: (union_specifier
|
||||
body: (field_declaration_list))
|
||||
declarator: (type_identifier) @declaration.name) @declaration.union
|
||||
|
||||
;; Declarations — enum
|
||||
(enum_specifier
|
||||
name: (type_identifier) @declaration.name) @declaration.enum
|
||||
|
||||
;; Declarations — function definition
|
||||
(function_definition
|
||||
declarator: (function_declarator
|
||||
declarator: (identifier) @declaration.name)) @declaration.function
|
||||
|
||||
;; Declarations — function definition with pointer return
|
||||
(function_definition
|
||||
declarator: (pointer_declarator
|
||||
declarator: (function_declarator
|
||||
declarator: (identifier) @declaration.name))) @declaration.function
|
||||
|
||||
;; Declarations — function declaration (prototype)
|
||||
;; Note: Both prototypes and definitions are captured as @declaration.function.
|
||||
;; This may produce duplicate Function nodes in the knowledge graph when a
|
||||
;; function is declared in a header and defined in a .c file. CALLS edges
|
||||
;; resolve correctly through scope-based wildcard import chains; the
|
||||
;; duplication is a graph-quality concern only (no false edges).
|
||||
(declaration
|
||||
declarator: (function_declarator
|
||||
declarator: (identifier) @declaration.name)) @declaration.function
|
||||
|
||||
;; Declarations — function declaration with pointer return (prototype)
|
||||
(declaration
|
||||
declarator: (pointer_declarator
|
||||
declarator: (function_declarator
|
||||
declarator: (identifier) @declaration.name))) @declaration.function
|
||||
|
||||
;; Declarations — typedef
|
||||
(type_definition
|
||||
declarator: (type_identifier) @declaration.name) @declaration.typedef
|
||||
|
||||
;; Declarations — typedef for function pointers: typedef void (*callback)(int, int)
|
||||
(type_definition
|
||||
declarator: (function_declarator
|
||||
declarator: (parenthesized_declarator
|
||||
(pointer_declarator
|
||||
declarator: (type_identifier) @declaration.name)))) @declaration.typedef
|
||||
|
||||
;; Declarations — struct fields
|
||||
(field_declaration
|
||||
declarator: (field_identifier) @declaration.name) @declaration.field
|
||||
|
||||
;; Declarations — struct fields (pointer)
|
||||
(field_declaration
|
||||
declarator: (pointer_declarator
|
||||
declarator: (field_identifier) @declaration.name)) @declaration.field
|
||||
|
||||
;; Declarations — variables (with initializer)
|
||||
(declaration
|
||||
declarator: (init_declarator
|
||||
declarator: (identifier) @declaration.name)) @declaration.variable
|
||||
|
||||
;; Declarations — macro definitions
|
||||
(preproc_def
|
||||
name: (identifier) @declaration.name) @declaration.macro
|
||||
|
||||
(preproc_function_def
|
||||
name: (identifier) @declaration.name) @declaration.macro
|
||||
|
||||
;; Declarations — enum constants
|
||||
(enumerator
|
||||
name: (identifier) @declaration.name) @declaration.const
|
||||
|
||||
;; Imports
|
||||
(preproc_include) @import.statement
|
||||
|
||||
;; Type bindings — parameter annotations
|
||||
(parameter_declaration
|
||||
type: (_) @type-binding.type
|
||||
declarator: (identifier) @type-binding.name) @type-binding.parameter
|
||||
|
||||
;; Type bindings — variable with type (init_declarator)
|
||||
(declaration
|
||||
type: (_) @type-binding.type
|
||||
declarator: (init_declarator
|
||||
declarator: (identifier) @type-binding.name)) @type-binding.assignment
|
||||
|
||||
;; References — free calls
|
||||
;; Note: This also captures calls through function pointer variables (e.g. fp(x))
|
||||
;; since tree-sitter-c produces structurally identical AST nodes for both direct
|
||||
;; function calls and function-pointer-variable calls. A type-based guard to
|
||||
;; distinguish variable-calls from function-calls is not implemented — this is a
|
||||
;; known architectural trade-off shared with the Go resolver. The uniqueness
|
||||
;; constraint in pickUniqueGlobalCallable limits false edge exposure.
|
||||
(call_expression
|
||||
function: (identifier) @reference.name) @reference.call.free
|
||||
|
||||
;; References — member calls via pointer (ptr->func())
|
||||
(call_expression
|
||||
function: (field_expression
|
||||
argument: (_) @reference.receiver
|
||||
field: (field_identifier) @reference.name)) @reference.call.member
|
||||
|
||||
;; References — field reads
|
||||
(field_expression
|
||||
argument: (_) @reference.receiver
|
||||
field: (field_identifier) @reference.name) @reference.read
|
||||
|
||||
;; References — field writes (assignment)
|
||||
(assignment_expression
|
||||
left: (field_expression
|
||||
argument: (_) @reference.receiver
|
||||
field: (field_identifier) @reference.name)) @reference.write
|
||||
`;
|
||||
|
||||
let _parser: Parser | null = null;
|
||||
let _query: Parser.Query | null = null;
|
||||
|
||||
export function getCParser(): Parser {
|
||||
if (_parser === null) {
|
||||
_parser = new Parser();
|
||||
_parser.setLanguage(C as Parameters<Parser['setLanguage']>[0]);
|
||||
}
|
||||
return _parser;
|
||||
}
|
||||
|
||||
export function getCScopeQuery(): Parser.Query {
|
||||
if (_query === null) {
|
||||
_query = new Parser.Query(C as Parameters<Parser['setLanguage']>[0], C_SCOPE_QUERY);
|
||||
}
|
||||
return _query;
|
||||
}
|
||||
@@ -0,0 +1,73 @@
|
||||
import type { ParsedFile, SymbolDefinition } from 'gitnexus-shared';
|
||||
import { SupportedLanguages } from 'gitnexus-shared';
|
||||
import { buildMro, defaultLinearize } from '../../scope-resolution/passes/mro.js';
|
||||
import { populateClassOwnedMembers } from '../../scope-resolution/scope/walkers.js';
|
||||
import type { ScopeResolver } from '../../scope-resolution/contract/scope-resolver.js';
|
||||
import { cProvider } from '../c-cpp.js';
|
||||
import { cArityCompatibility, cMergeBindings, resolveCImportTarget } from './index.js';
|
||||
import { scanHeaderFiles } from './header-scan.js';
|
||||
import { expandCWildcardNames, isStaticName, clearStaticNames } from './static-linkage.js';
|
||||
|
||||
/**
|
||||
* C `ScopeResolver` registered in `SCOPE_RESOLVERS` and consumed by
|
||||
* the generic `runScopeResolution` orchestrator (RFC #909 Ring 3).
|
||||
*
|
||||
* C is a structurally simple language for scope resolution:
|
||||
* - No classes (structs are value types, no method dispatch)
|
||||
* - No inheritance (no MRO needed beyond the shared first-wins default)
|
||||
* - No overloading (arity check is simple: variadic detection only)
|
||||
* - `#include` is wildcard import (all symbols from header are visible)
|
||||
* - `static` functions are file-local (not exported)
|
||||
*/
|
||||
export const cScopeResolver: ScopeResolver = {
|
||||
language: SupportedLanguages.C,
|
||||
languageProvider: cProvider,
|
||||
importEdgeReason: 'c-scope: include',
|
||||
|
||||
loadResolutionConfig: (repoPath: string) => {
|
||||
// Clear stale static-linkage data from any previous invocation to
|
||||
// prevent cross-repo contamination in server-mode scenarios.
|
||||
clearStaticNames();
|
||||
return scanHeaderFiles(repoPath);
|
||||
},
|
||||
|
||||
resolveImportTarget: (targetRaw, fromFile, allFilePaths, resolutionConfig) => {
|
||||
// Augment allFilePaths with .h files discovered via loadResolutionConfig
|
||||
// since the phase only passes .c files to the C resolver but #include
|
||||
// targets .h files classified as C++ in language detection.
|
||||
const headerPaths = resolutionConfig as ReadonlySet<string> | undefined;
|
||||
if (headerPaths !== undefined && headerPaths.size > 0) {
|
||||
const augmented = new Set(allFilePaths);
|
||||
for (const h of headerPaths) augmented.add(h);
|
||||
return resolveCImportTarget(targetRaw, fromFile, augmented);
|
||||
}
|
||||
return resolveCImportTarget(targetRaw, fromFile, allFilePaths);
|
||||
},
|
||||
|
||||
expandsWildcardTo: (targetModuleScope, parsedFiles) =>
|
||||
expandCWildcardNames(targetModuleScope, parsedFiles),
|
||||
|
||||
mergeBindings: (existing, incoming, scopeId) => cMergeBindings(existing, incoming, scopeId),
|
||||
|
||||
arityCompatibility: (callsite, def) => cArityCompatibility(def, callsite),
|
||||
|
||||
buildMro: (graph, parsedFiles, nodeLookup) =>
|
||||
buildMro(graph, parsedFiles, nodeLookup, defaultLinearize),
|
||||
|
||||
populateOwners: (parsed: ParsedFile) => populateClassOwnedMembers(parsed),
|
||||
|
||||
isSuperReceiver: () => false,
|
||||
|
||||
// C is statically typed — disable field fallback heuristic
|
||||
fieldFallbackOnMethodLookup: false,
|
||||
// C has no method return types to propagate
|
||||
propagatesReturnTypesAcrossImports: false,
|
||||
// C #include brings in all symbols — enable global free call fallback
|
||||
allowGlobalFreeCallFallback: true,
|
||||
// C `static` functions have file-local (translation-unit) linkage —
|
||||
// exclude them from global free-call fallback cross-file resolution.
|
||||
isFileLocalDef: (def: SymbolDefinition) => {
|
||||
const simple = def.qualifiedName?.split('.').pop() ?? def.qualifiedName ?? '';
|
||||
return isStaticName(def.filePath, simple);
|
||||
},
|
||||
};
|
||||
@@ -0,0 +1,38 @@
|
||||
import type {
|
||||
CaptureMatch,
|
||||
ParsedImport,
|
||||
Scope,
|
||||
ScopeId,
|
||||
ScopeTree,
|
||||
TypeRef,
|
||||
} from 'gitnexus-shared';
|
||||
|
||||
/**
|
||||
* C binding scope: always use default auto-hoist (null).
|
||||
* C has no self/receiver bindings that need special scoping.
|
||||
*/
|
||||
export function cBindingScopeFor(
|
||||
_decl: CaptureMatch,
|
||||
_innermost: Scope,
|
||||
_tree: ScopeTree,
|
||||
): ScopeId | null {
|
||||
return null;
|
||||
}
|
||||
|
||||
/**
|
||||
* C import owning scope: always use default (null).
|
||||
*/
|
||||
export function cImportOwningScope(
|
||||
_imp: ParsedImport,
|
||||
_innermost: Scope,
|
||||
_tree: ScopeTree,
|
||||
): ScopeId | null {
|
||||
return null;
|
||||
}
|
||||
|
||||
/**
|
||||
* C receiver binding: always null. C has no methods or receivers.
|
||||
*/
|
||||
export function cReceiverBinding(_functionScope: Scope): TypeRef | null {
|
||||
return null;
|
||||
}
|
||||
@@ -0,0 +1,64 @@
|
||||
import type { ParsedFile, ScopeId, SymbolDefinition } from 'gitnexus-shared';
|
||||
|
||||
/**
|
||||
* Per-file set of function names declared with `static` storage class.
|
||||
* Populated during `emitCScopeCaptures` and consumed by `expandCWildcardNames`
|
||||
* to exclude file-local symbols from cross-file wildcard import visibility.
|
||||
*
|
||||
* NOTE: module-level state, single-process-single-repo use only.
|
||||
* For server-mode or multi-repo-in-one-process use cases, call
|
||||
* `clearStaticNames()` at the start of each resolution pass to avoid
|
||||
* stale static-linkage data from a previous invocation.
|
||||
*
|
||||
* Key: filePath, Value: Set of static function names.
|
||||
*/
|
||||
const staticNames = new Map<string, Set<string>>();
|
||||
|
||||
/** Record a symbol name as `static` (file-local linkage) for the given file. */
|
||||
export function markStaticName(filePath: string, name: string): void {
|
||||
let names = staticNames.get(filePath);
|
||||
if (names === undefined) {
|
||||
names = new Set<string>();
|
||||
staticNames.set(filePath, names);
|
||||
}
|
||||
names.add(name);
|
||||
}
|
||||
|
||||
/** Check whether a symbol name has `static` linkage in the given file. */
|
||||
export function isStaticName(filePath: string, name: string): boolean {
|
||||
return staticNames.get(filePath)?.has(name) ?? false;
|
||||
}
|
||||
|
||||
/** Clear tracked static names (for testing). */
|
||||
export function clearStaticNames(): void {
|
||||
staticNames.clear();
|
||||
}
|
||||
|
||||
/**
|
||||
* Return the names visible through a C wildcard import (`#include`).
|
||||
* All module-scope defs from the target file are visible EXCEPT those
|
||||
* declared with `static` storage class (file-local linkage in C).
|
||||
*/
|
||||
export function expandCWildcardNames(
|
||||
targetModuleScope: ScopeId,
|
||||
parsedFiles: readonly ParsedFile[],
|
||||
): readonly string[] {
|
||||
const target = parsedFiles.find((p) => p.moduleScope === targetModuleScope);
|
||||
if (target === undefined) return [];
|
||||
|
||||
const seen = new Set<string>();
|
||||
const names: string[] = [];
|
||||
for (const def of target.localDefs) {
|
||||
const name = simpleName(def);
|
||||
if (name === '') continue;
|
||||
if (isStaticName(target.filePath, name)) continue;
|
||||
if (seen.has(name)) continue;
|
||||
seen.add(name);
|
||||
names.push(name);
|
||||
}
|
||||
return names;
|
||||
}
|
||||
|
||||
function simpleName(def: SymbolDefinition): string {
|
||||
return def.qualifiedName?.split('.').pop() ?? def.qualifiedName ?? '';
|
||||
}
|
||||
@@ -24,6 +24,7 @@ import { synthesizeCsharpReceiverBinding } from './receiver-binding.js';
|
||||
import { getCsharpParser, getCsharpScopeQuery } from './query.js';
|
||||
import { recordCacheHit, recordCacheMiss } from './cache-stats.js';
|
||||
import { getTreeSitterBufferSize } from '../../constants.js';
|
||||
import { parseSourceSafe } from '../../../tree-sitter/safe-parse.js';
|
||||
|
||||
/** Declaration anchors that carry function-like arity metadata. */
|
||||
const FUNCTION_DECL_TAGS = [
|
||||
@@ -86,7 +87,7 @@ export function emitCsharpScopeCaptures(
|
||||
// the LanguageProvider contract layer; cast here at the use site.
|
||||
let tree = cachedTree as ReturnType<ReturnType<typeof getCsharpParser>['parse']> | undefined;
|
||||
if (tree === undefined) {
|
||||
tree = getCsharpParser().parse(sourceText, undefined, {
|
||||
tree = parseSourceSafe(getCsharpParser(), sourceText, undefined, {
|
||||
bufferSize: getTreeSitterBufferSize(sourceText),
|
||||
});
|
||||
recordCacheMiss();
|
||||
|
||||
@@ -36,6 +36,7 @@ import type { BindingRef, ParsedFile, Scope, ScopeId, SymbolDefinition } from 'g
|
||||
import type { ScopeResolutionIndexes } from '../../model/scope-resolution-indexes.js';
|
||||
import { getCsharpParser } from './query.js';
|
||||
import { getTreeSitterBufferSize } from '../../constants.js';
|
||||
import { parseSourceSafe } from '../../../tree-sitter/safe-parse.js';
|
||||
|
||||
interface CsharpFileStructure {
|
||||
/** Declared namespace names in file source order. Empty array means
|
||||
@@ -56,7 +57,7 @@ function extractFileStructure(content: string, cachedTree: unknown): CsharpFileS
|
||||
type CsharpTree = ReturnType<ReturnType<typeof getCsharpParser>['parse']>;
|
||||
const tree =
|
||||
(cachedTree as CsharpTree | undefined) ??
|
||||
getCsharpParser().parse(content, undefined, {
|
||||
parseSourceSafe(getCsharpParser(), content, undefined, {
|
||||
bufferSize: getTreeSitterBufferSize(content),
|
||||
});
|
||||
const namespaces: string[] = [];
|
||||
@@ -359,7 +360,7 @@ export function populateCsharpNamespaceSiblings(
|
||||
const q = def.qualifiedName ?? '';
|
||||
const key = q.includes('.') ? q.slice(q.lastIndexOf('.') + 1) : q;
|
||||
if (key === '') continue;
|
||||
const arr = defsByName.get(key) ?? [];
|
||||
const arr = [...(defsByName.get(key) ?? [])];
|
||||
arr.push(def);
|
||||
defsByName.set(key, arr);
|
||||
}
|
||||
|
||||
@@ -12,6 +12,7 @@ import { splitGoImportStatement } from './import-decomposer.js';
|
||||
import { synthesizeGoReceiverBinding } from './receiver-binding.js';
|
||||
import { synthesizeGoTypeBindings } from './type-binding.js';
|
||||
import { getTreeSitterBufferSize } from '../../constants.js';
|
||||
import { parseSourceSafe } from '../../../tree-sitter/safe-parse.js';
|
||||
|
||||
export function emitGoScopeCaptures(
|
||||
sourceText: string,
|
||||
@@ -20,7 +21,7 @@ export function emitGoScopeCaptures(
|
||||
): readonly CaptureMatch[] {
|
||||
let tree = cachedTree as ReturnType<ReturnType<typeof getGoParser>['parse']> | undefined;
|
||||
if (tree === undefined) {
|
||||
tree = getGoParser().parse(sourceText, undefined, {
|
||||
tree = parseSourceSafe(getGoParser(), sourceText, undefined, {
|
||||
bufferSize: getTreeSitterBufferSize(sourceText),
|
||||
});
|
||||
recordGoCacheMiss();
|
||||
|
||||
@@ -2,6 +2,7 @@ import type { ParsedFile, Scope, TypeRef } from 'gitnexus-shared';
|
||||
import type { ScopeResolutionIndexes } from '../../model/scope-resolution-indexes.js';
|
||||
import { getGoParser } from './query.js';
|
||||
import { getTreeSitterBufferSize } from '../../constants.js';
|
||||
import { parseSourceSafe } from '../../../tree-sitter/safe-parse.js';
|
||||
|
||||
export function populateGoRangeBindings(
|
||||
parsedFiles: readonly ParsedFile[],
|
||||
@@ -20,7 +21,7 @@ export function populateGoRangeBindings(
|
||||
const cachedTree = ctx.treeCache?.get(parsed.filePath);
|
||||
const tree =
|
||||
(cachedTree as ReturnType<typeof parser.parse> | undefined) ??
|
||||
parser.parse(sourceText, undefined, {
|
||||
parseSourceSafe(parser, sourceText, undefined, {
|
||||
bufferSize: getTreeSitterBufferSize(sourceText),
|
||||
});
|
||||
const moduleScope = parsed.scopes.find((s) => s.kind === 'Module');
|
||||
|
||||
@@ -24,6 +24,7 @@ import { synthesizeReceiverTypeBinding } from './receiver-binding.js';
|
||||
import { computePythonArityMetadata } from './arity-metadata.js';
|
||||
import { recordCacheHit, recordCacheMiss } from './cache-stats.js';
|
||||
import { getTreeSitterBufferSize } from '../../constants.js';
|
||||
import { parseSourceSafe } from '../../../tree-sitter/safe-parse.js';
|
||||
import { pythonFunctionDefinitionLabel } from './simple-hooks.js';
|
||||
|
||||
export function emitPythonScopeCaptures(
|
||||
@@ -39,7 +40,7 @@ export function emitPythonScopeCaptures(
|
||||
let tree = cachedTree as ReturnType<ReturnType<typeof getPythonParser>['parse']> | undefined;
|
||||
if (tree === undefined) {
|
||||
try {
|
||||
tree = getPythonParser().parse(sourceText, undefined, {
|
||||
tree = parseSourceSafe(getPythonParser(), sourceText, undefined, {
|
||||
bufferSize: getTreeSitterBufferSize(sourceText),
|
||||
});
|
||||
} catch (err) {
|
||||
|
||||
@@ -38,6 +38,7 @@ import { recordCacheHit, recordCacheMiss } from './cache-stats.js';
|
||||
import { synthesizeTsReceiverBinding } from './receiver-binding.js';
|
||||
import { computeTsArityMetadata } from './arity-metadata.js';
|
||||
import { getTreeSitterBufferSize } from '../../constants.js';
|
||||
import { parseSourceSafe } from '../../../tree-sitter/safe-parse.js';
|
||||
|
||||
/** tree-sitter-typescript node types for function-like scopes that may
|
||||
* carry a synthesized `this` binding. Kept in sync with the
|
||||
@@ -134,7 +135,7 @@ export function emitTsScopeCaptures(
|
||||
tree = undefined;
|
||||
}
|
||||
if (tree === undefined) {
|
||||
tree = getTsParser(filePath).parse(sourceText, undefined, {
|
||||
tree = parseSourceSafe(getTsParser(filePath), sourceText, undefined, {
|
||||
bufferSize: getTreeSitterBufferSize(sourceText),
|
||||
});
|
||||
recordCacheMiss();
|
||||
|
||||
@@ -5,12 +5,11 @@ import { loadParser, loadLanguage, isLanguageAvailable } from '../tree-sitter/pa
|
||||
import { getProvider } from './languages/index.js';
|
||||
import { generateId } from '../../lib/utils.js';
|
||||
import type { SymbolTableReader, SymbolTableWriter, ExtractedHeritage } from './model/index.js';
|
||||
// SymbolTableReader is used for the FieldExtractorContext stub; the
|
||||
// parsing functions themselves need Writer because they call .add().
|
||||
import { ASTCache } from './ast-cache.js';
|
||||
import { getLanguageFromFilename, SupportedLanguages } from 'gitnexus-shared';
|
||||
import { extractVueScript, isVueSetupTopLevel } from './vue-sfc-extractor.js';
|
||||
import { yieldToEventLoop } from './utils/event-loop.js';
|
||||
import { parseSourceSafe } from '../tree-sitter/safe-parse.js';
|
||||
import { isVerboseIngestionEnabled } from './utils/verbose.js';
|
||||
import {
|
||||
getDefinitionNodeFromCaptures,
|
||||
@@ -384,7 +383,7 @@ const processParsingSequential = async (
|
||||
|
||||
let tree: Parser.Tree;
|
||||
try {
|
||||
tree = parser.parse(parseContent, undefined, {
|
||||
tree = parseSourceSafe(parser, parseContent, undefined, {
|
||||
bufferSize: getTreeSitterBufferSize(parseContent),
|
||||
});
|
||||
} catch (parseError) {
|
||||
|
||||
@@ -71,6 +71,7 @@ export const MIGRATED_LANGUAGES: ReadonlySet<SupportedLanguages> = new Set<Suppo
|
||||
SupportedLanguages.CSharp,
|
||||
SupportedLanguages.TypeScript,
|
||||
SupportedLanguages.Go,
|
||||
SupportedLanguages.C,
|
||||
]);
|
||||
|
||||
/**
|
||||
|
||||
@@ -472,6 +472,18 @@ export interface ScopeResolver {
|
||||
*/
|
||||
readonly allowGlobalFreeCallFallback?: boolean;
|
||||
|
||||
/**
|
||||
* Optional predicate to identify definitions with file-local linkage
|
||||
* (e.g. C `static` functions). When provided, `pickUniqueGlobalCallable`
|
||||
* excludes defs where `isFileLocalDef(def) === true` and the def lives
|
||||
* in a different file from the caller. This prevents the global free-call
|
||||
* fallback from creating CALLS edges to file-local symbols that are
|
||||
* logically invisible from the caller's translation unit.
|
||||
*
|
||||
* Languages without file-local linkage semantics leave this undefined.
|
||||
*/
|
||||
readonly isFileLocalDef?: (def: SymbolDefinition) => boolean;
|
||||
|
||||
/**
|
||||
* Optional post-finalize hook to inject cross-file bindings that
|
||||
* aren't modeled via explicit imports. Runs after
|
||||
|
||||
@@ -36,7 +36,10 @@ export function emitFreeCallFallback(
|
||||
handledSites: Set<string>,
|
||||
model: SemanticModel,
|
||||
workspaceIndex: WorkspaceResolutionIndex,
|
||||
options: { readonly allowGlobalFallback?: boolean } = {},
|
||||
options: {
|
||||
readonly allowGlobalFallback?: boolean;
|
||||
readonly isFileLocalDef?: (def: SymbolDefinition) => boolean;
|
||||
} = {},
|
||||
): number {
|
||||
let emitted = 0;
|
||||
const seen = new Set<string>();
|
||||
@@ -73,7 +76,13 @@ export function emitFreeCallFallback(
|
||||
// the caller does not import the target package. Same-package calls are
|
||||
// caught by findCallableBindingInScope above before reaching here.
|
||||
if (fnDef === undefined && options.allowGlobalFallback === true) {
|
||||
fnDef = pickUniqueGlobalCallable(site.name, model, scopes);
|
||||
fnDef = pickUniqueGlobalCallable(
|
||||
site.name,
|
||||
model,
|
||||
scopes,
|
||||
parsed.filePath,
|
||||
options.isFileLocalDef,
|
||||
);
|
||||
}
|
||||
if (fnDef === undefined) continue;
|
||||
const callerGraphId = resolveCallerGraphId(site.inScope, scopes, nodeLookup);
|
||||
@@ -107,6 +116,8 @@ function pickUniqueGlobalCallable(
|
||||
name: string,
|
||||
model: SemanticModel,
|
||||
scopes: ScopeResolutionIndexes,
|
||||
callerFilePath: string,
|
||||
isFileLocalDef?: (def: SymbolDefinition) => boolean,
|
||||
): SymbolDefinition | undefined {
|
||||
const scopeDefs: SymbolDefinition[] = [];
|
||||
const scopeSeen = new Set<string>();
|
||||
@@ -114,6 +125,11 @@ function pickUniqueGlobalCallable(
|
||||
const simple = def.qualifiedName?.split('.').pop() ?? def.qualifiedName;
|
||||
if (simple !== name) continue;
|
||||
if (def.type !== 'Function' && def.type !== 'Method' && def.type !== 'Constructor') continue;
|
||||
// Skip file-local defs (e.g. C `static` functions) that live in a
|
||||
// different file from the caller — they are logically invisible.
|
||||
if (isFileLocalDef !== undefined && def.filePath !== callerFilePath && isFileLocalDef(def)) {
|
||||
continue;
|
||||
}
|
||||
const key = logicalCallableKey(def);
|
||||
if (scopeSeen.has(key)) continue;
|
||||
scopeSeen.add(key);
|
||||
@@ -125,6 +141,12 @@ function pickUniqueGlobalCallable(
|
||||
const seen = new Set<string>();
|
||||
const push = (pool: readonly SymbolDefinition[]): void => {
|
||||
for (const def of pool) {
|
||||
// Apply the same file-local linkage filter as Phase 1 —
|
||||
// cross-file static defs must never leak through the
|
||||
// SemanticModel fallback path.
|
||||
if (isFileLocalDef !== undefined && def.filePath !== callerFilePath && isFileLocalDef(def)) {
|
||||
continue;
|
||||
}
|
||||
const key = logicalCallableKey(def);
|
||||
if (seen.has(key)) continue;
|
||||
seen.add(key);
|
||||
|
||||
@@ -15,6 +15,7 @@ import { pythonScopeResolver } from '../../languages/python/scope-resolver.js';
|
||||
import { csharpScopeResolver } from '../../languages/csharp/scope-resolver.js';
|
||||
import { typescriptScopeResolver } from '../../languages/typescript/scope-resolver.js';
|
||||
import { goScopeResolver } from '../../languages/go/scope-resolver.js';
|
||||
import { cScopeResolver } from '../../languages/c/scope-resolver.js';
|
||||
|
||||
/** Map of `SupportedLanguages` → `ScopeResolver`. The phase iterates
|
||||
* this map intersected with `MIGRATED_LANGUAGES` (the per-language
|
||||
@@ -28,4 +29,5 @@ export const SCOPE_RESOLVERS: ReadonlyMap<SupportedLanguages, ScopeResolver> = n
|
||||
[SupportedLanguages.CSharp, csharpScopeResolver],
|
||||
[SupportedLanguages.TypeScript, typescriptScopeResolver],
|
||||
[SupportedLanguages.Go, goScopeResolver],
|
||||
[SupportedLanguages.C, cScopeResolver],
|
||||
]);
|
||||
|
||||
@@ -261,7 +261,10 @@ export function runScopeResolution(
|
||||
handledSites,
|
||||
readonlyModel,
|
||||
workspaceIndex,
|
||||
{ allowGlobalFallback: provider.allowGlobalFreeCallFallback === true },
|
||||
{
|
||||
allowGlobalFallback: provider.allowGlobalFreeCallFallback === true,
|
||||
isFileLocalDef: provider.isFileLocalDef,
|
||||
},
|
||||
);
|
||||
const { emitted, skipped } = emitReferencesViaLookup(
|
||||
graph,
|
||||
|
||||
@@ -20,6 +20,7 @@ import {
|
||||
getTreeSitterContentByteLength,
|
||||
TREE_SITTER_MAX_BUFFER,
|
||||
} from '../constants.js';
|
||||
import { parseSourceSafe } from '../../tree-sitter/safe-parse.js';
|
||||
import type { SymbolTableReader } from '../model/symbol-table.js';
|
||||
import type { ExtractedHeritage } from '../model/heritage-map.js';
|
||||
|
||||
@@ -1416,7 +1417,7 @@ const processFileGroup = (
|
||||
|
||||
let tree;
|
||||
try {
|
||||
tree = parser.parse(parseContent, undefined, {
|
||||
tree = parseSourceSafe(parser, parseContent, undefined, {
|
||||
bufferSize: getTreeSitterBufferSize(parseContent),
|
||||
});
|
||||
} catch (err) {
|
||||
|
||||
@@ -1248,6 +1248,13 @@ export const loadVectorExtension = async (
|
||||
): Promise<boolean> => {
|
||||
const useModuleState = targetConn === undefined;
|
||||
if (useModuleState && vectorExtensionLoaded) return true;
|
||||
// INSTALL VECTOR crashes with SIGSEGV on Windows: the KuzuDB native extension
|
||||
// installer has an unhandled error path on Windows that raises a fatal signal
|
||||
// that JS try/catch cannot intercept. Skip loading — vector/embedding search
|
||||
// is unavailable but all graph index queries still work. Do NOT set
|
||||
// vectorExtensionLoaded here: the flag means "successfully loaded", and a
|
||||
// subsequent call would otherwise short-circuit to `return true` at the top.
|
||||
if (process.platform === 'win32') return false;
|
||||
if (!isVectorExtensionSupportedByPlatform()) return false;
|
||||
|
||||
const c: lbug.Connection | null = targetConn ?? conn;
|
||||
|
||||
@@ -420,7 +420,17 @@ async function doInitLbug(repoId: string, dbPath: string): Promise<void> {
|
||||
// install; analyze owns extension installation. If LOAD fails, search
|
||||
// features degrade gracefully and the user-facing query path proceeds.
|
||||
if (!shared.ftsLoaded) {
|
||||
shared.ftsLoaded = await loadFTSExtension(available[0], { policy: 'load-only' });
|
||||
// Windows guard: LOAD EXTENSION fts crashes with SIGSEGV on Windows when
|
||||
// the FTS extension binary is not installed locally (@ladybugdb/core native
|
||||
// bug — the extension loader hits an unhandled error path that signals SIGSEGV
|
||||
// rather than throwing a JS exception, so try/catch cannot protect here).
|
||||
// Skip the load on Windows; bm25-index.js catches the resulting Kuzu catalog
|
||||
// errors and returns empty BM25 results gracefully. Graph queries are unaffected.
|
||||
if (process.platform === 'win32') {
|
||||
shared.ftsLoaded = true;
|
||||
} else {
|
||||
shared.ftsLoaded = await loadFTSExtension(available[0], { policy: 'load-only' });
|
||||
}
|
||||
}
|
||||
|
||||
// Register pool entry only after all connections are pre-warmed and FTS is
|
||||
@@ -484,8 +494,13 @@ export async function initLbugWithDb(
|
||||
// Load FTS extension if not already loaded on this Database.
|
||||
// policy: 'load-only' — same contract as initLbug above; the read pool
|
||||
// must not block on a network install during query execution.
|
||||
// Windows guard: same SIGSEGV risk as doInitLbug above — skip on Windows.
|
||||
if (!shared.ftsLoaded) {
|
||||
shared.ftsLoaded = await loadFTSExtension(available[0], { policy: 'load-only' });
|
||||
if (process.platform === 'win32') {
|
||||
shared.ftsLoaded = true;
|
||||
} else {
|
||||
shared.ftsLoaded = await loadFTSExtension(available[0], { policy: 'load-only' });
|
||||
}
|
||||
}
|
||||
|
||||
pool.set(repoId, {
|
||||
|
||||
@@ -0,0 +1,40 @@
|
||||
import type Parser from 'tree-sitter';
|
||||
|
||||
/**
|
||||
* tree-sitter 0.21.x's Node native binding crashes (SIGSEGV) on Windows when
|
||||
* `parser.parse(string, …)` is handed a JS string longer than 32 767 chars.
|
||||
* The crash happens inside the binding's V8 string-to-buffer conversion and
|
||||
* cannot be intercepted from JavaScript. The callback (`Parser.Input`) overload
|
||||
* pulls source in fixed-size chunks via repeated callback invocations and
|
||||
* bypasses that conversion path entirely.
|
||||
*
|
||||
* Chunk size is comfortably below the boundary; any value < 32 767 works.
|
||||
*/
|
||||
const SAFE_PARSE_CHUNK_CHARS = 16 * 1024;
|
||||
|
||||
/**
|
||||
* Files at or below this length skip the callback machinery and use the
|
||||
* direct string overload — the bug only manifests above the int16 boundary,
|
||||
* so small inputs save the cost of N callback invocations per parse.
|
||||
*/
|
||||
const DIRECT_PARSE_LIMIT_CHARS = 16 * 1024;
|
||||
|
||||
/**
|
||||
* Parse `sourceText` safely on every platform. See {@link SAFE_PARSE_CHUNK_CHARS}
|
||||
* for the underlying tree-sitter binding bug this works around.
|
||||
*/
|
||||
export function parseSourceSafe(
|
||||
parser: Parser,
|
||||
sourceText: string,
|
||||
oldTree?: Parser.Tree,
|
||||
options?: Parser.Options,
|
||||
): Parser.Tree {
|
||||
if (sourceText.length <= DIRECT_PARSE_LIMIT_CHARS) {
|
||||
return parser.parse(sourceText, oldTree, options);
|
||||
}
|
||||
const input: Parser.Input = (index) => {
|
||||
if (index >= sourceText.length) return null;
|
||||
return sourceText.slice(index, index + SAFE_PARSE_CHUNK_CHARS);
|
||||
};
|
||||
return parser.parse(input, oldTree, options);
|
||||
}
|
||||
@@ -11,6 +11,7 @@ import os from 'os';
|
||||
import fs from 'fs/promises';
|
||||
import { isIP } from 'net';
|
||||
import { logger } from '../core/logger.js';
|
||||
import { parseRepoNameFromUrl } from '../storage/git.js';
|
||||
|
||||
/** Root directory for all cloned repositories. Targets must resolve inside this. */
|
||||
const CLONE_ROOT = path.resolve(path.join(os.homedir(), '.gitnexus', 'repos'));
|
||||
@@ -29,20 +30,17 @@ const REPO_NAME_PATTERN = /^[a-zA-Z0-9._-]+$/;
|
||||
* clone root via path traversal.
|
||||
*/
|
||||
export function extractRepoName(url: string): string {
|
||||
// Strip trailing slashes without a regex to avoid polynomial-ReDoS on
|
||||
// pathological inputs like `https://x.com/y` + '/'.repeat(1e6). CodeQL's
|
||||
// js/polynomial-redos flagged `/\/+$/` here.
|
||||
let end = url.length;
|
||||
while (end > 0 && url.charCodeAt(end - 1) === 47 /* '/' */) end--;
|
||||
const cleaned = url.slice(0, end);
|
||||
|
||||
const lastSegment = cleaned.split(/[/:]/).pop() || '';
|
||||
const stripped = lastSegment.endsWith('.git') ? lastSegment.slice(0, -4) : lastSegment;
|
||||
|
||||
if (!stripped || stripped === '.' || stripped === '..' || !REPO_NAME_PATTERN.test(stripped)) {
|
||||
const name = parseRepoNameFromUrl(url);
|
||||
if (
|
||||
!name ||
|
||||
name === '.' ||
|
||||
name === '..' ||
|
||||
name === 'unknown' ||
|
||||
!REPO_NAME_PATTERN.test(name)
|
||||
) {
|
||||
throw new Error('Could not extract a valid repository name from URL');
|
||||
}
|
||||
return stripped;
|
||||
return name;
|
||||
}
|
||||
|
||||
/** Get the clone target directory for a repo name. */
|
||||
@@ -399,8 +397,8 @@ export async function cloneOrPull(
|
||||
}
|
||||
|
||||
// Always validate the requested URL — the prior shape only ran this in
|
||||
// the clone branch, leaving the pull branch as an SSRF / blocked-host
|
||||
// bypass when an existing clone shared the basename of an attacker URL.
|
||||
// the code path where the repo was cloned. Now it runs unconditionally,
|
||||
// preventing SSRF / blocked-host bypasses even when targetDir already exists.
|
||||
validateGitUrl(url);
|
||||
|
||||
const exists = await fs.access(path.join(safeTarget, '.git')).then(
|
||||
|
||||
+51
-12
@@ -255,24 +255,63 @@ export const getRemoteOriginUrl = (repoPath: string): string | null => {
|
||||
};
|
||||
|
||||
/**
|
||||
* Parse a repository name out of a git remote URL. Handles the common
|
||||
* SSH (`git@host:owner/repo.git`), HTTPS (`https://host/owner/repo.git`),
|
||||
* `git://`, `ssh://`, and `file://` shapes. Returns `null` for empty /
|
||||
* unparseable input.
|
||||
* Sanitize a repository name to prevent argument injection and ensure
|
||||
* cross-platform filesystem compatibility.
|
||||
*
|
||||
* The heuristic: strip a trailing `.git` and trailing slashes, then
|
||||
* take the segment after the last `/` or `:`.
|
||||
* 1. Strips leading dashes to prevent git command-line argument injection
|
||||
* (e.g., --upload-pack=evil).
|
||||
* 2. Replaces characters that are unsafe for directory names across
|
||||
* platforms (Windows/macOS/Linux) with underscores.
|
||||
* 3. Blocks path traversal segments ("." and "..") and Windows reserved
|
||||
* names (e.g., CON, NUL) to prevent directory escape.
|
||||
*/
|
||||
export const sanitizeRepoName = (name: string): string => {
|
||||
// 1. Prevent argument injection by stripping leading dashes.
|
||||
// 2. Remove characters that are not alphanumerics, dots, underscores, or dashes.
|
||||
const sanitized = name.replace(/^-+/, '').replace(/[^a-zA-Z0-9._-]/g, '_');
|
||||
|
||||
// 3. Block path traversal segments and Windows reserved names.
|
||||
// Windows reserved names like CON, PRN, AUX, NUL, COM1-9, LPT1-9 cannot
|
||||
// be used as directory names on Windows even if they have an extension.
|
||||
const reserved = /^(CON|PRN|AUX|NUL|COM[1-9]|LPT[1-9])(\..*)?$/i;
|
||||
if (!sanitized || sanitized === '.' || sanitized === '..' || reserved.test(sanitized)) {
|
||||
return 'unknown';
|
||||
}
|
||||
|
||||
return sanitized;
|
||||
};
|
||||
|
||||
/**
|
||||
* Parse a repository name out of a git remote URL. Handles common shapes
|
||||
* including SSH (git@host:owner/repo.git) and HTTPS (https://host/owner/repo.git).
|
||||
*
|
||||
* Returns a sanitized, filesystem-safe name or null if no name could be inferred.
|
||||
* Returning null (rather than 'unknown') allows callers to use ?? null-coalescing
|
||||
* for fallbacks without risk of registry collisions on 'unknown'.
|
||||
*/
|
||||
export const parseRepoNameFromUrl = (url: string | null | undefined): string | null => {
|
||||
if (!url) return null;
|
||||
const trimmed = url.trim();
|
||||
if (!trimmed) return null;
|
||||
// Strip `.git` suffix (case-insensitive) and any trailing slashes.
|
||||
const withoutSuffix = trimmed.replace(/\.git\/*$/i, '').replace(/\/+$/, '');
|
||||
// Last path segment, splitting on either `/` or `:` (covers SSH form).
|
||||
const m = withoutSuffix.match(/[/:]([^/:]+)$/);
|
||||
const candidate = m ? m[1] : withoutSuffix;
|
||||
return candidate || null;
|
||||
|
||||
// Strip trailing slashes without a regex to avoid polynomial-ReDoS on
|
||||
// pathological inputs like `https://x.com/y` + '/'.repeat(1e6).
|
||||
let end = trimmed.length;
|
||||
while (end > 0 && trimmed.charCodeAt(end - 1) === 47 /* '/' */) end--;
|
||||
let cleaned = trimmed.slice(0, end);
|
||||
|
||||
// Strip trailing .git (case-insensitive)
|
||||
if (cleaned.toLowerCase().endsWith('.git')) {
|
||||
cleaned = cleaned.slice(0, -4);
|
||||
}
|
||||
|
||||
// Last path segment, handling colons for SSH URLs and path traversal.
|
||||
// Split on both / and : to consistently extract the last part.
|
||||
const candidate = cleaned.split(/[/:]/).pop() || '';
|
||||
if (!candidate) return null;
|
||||
|
||||
const safe = sanitizeRepoName(candidate);
|
||||
return safe === 'unknown' ? null : safe;
|
||||
};
|
||||
|
||||
/**
|
||||
|
||||
@@ -0,0 +1,9 @@
|
||||
/* a.c — contains a static (file-local) helper function.
|
||||
* This function must NOT be resolvable from caller.c. */
|
||||
static int helper(void) {
|
||||
return 42;
|
||||
}
|
||||
|
||||
int public_a(void) {
|
||||
return helper();
|
||||
}
|
||||
@@ -0,0 +1,11 @@
|
||||
/* b.c — contains a non-static (externally visible) helper function.
|
||||
* This function SHOULD be resolvable from caller.c. */
|
||||
#include "b.h"
|
||||
|
||||
int helper(void) {
|
||||
return 99;
|
||||
}
|
||||
|
||||
int public_b(void) {
|
||||
return helper();
|
||||
}
|
||||
@@ -0,0 +1,8 @@
|
||||
/* b.h — public header for b.c */
|
||||
#ifndef B_H
|
||||
#define B_H
|
||||
|
||||
int helper(void);
|
||||
int public_b(void);
|
||||
|
||||
#endif
|
||||
@@ -0,0 +1,9 @@
|
||||
/* caller.c — includes only b.h, calls helper().
|
||||
* Should resolve to b.c:helper, NOT a.c:static helper. */
|
||||
#include "b.h"
|
||||
|
||||
int main(void) {
|
||||
int x = helper();
|
||||
int y = public_b();
|
||||
return x + y;
|
||||
}
|
||||
@@ -0,0 +1,9 @@
|
||||
#include "service.h"
|
||||
|
||||
int main(void) {
|
||||
struct Service *svc = create_service();
|
||||
service_add_user(svc, "Alice", 25);
|
||||
service_add_user(svc, "Bob", 32);
|
||||
destroy_service(svc);
|
||||
return 0;
|
||||
}
|
||||
@@ -0,0 +1,20 @@
|
||||
#include "service.h"
|
||||
#include <stdlib.h>
|
||||
|
||||
struct Service *create_service(void) {
|
||||
struct Service *svc = malloc(sizeof(struct Service));
|
||||
svc->admin = create_user("admin", 30);
|
||||
svc->user_count = 0;
|
||||
return svc;
|
||||
}
|
||||
|
||||
void service_add_user(struct Service *svc, const char *name, int age) {
|
||||
struct User *user = create_user(name, age);
|
||||
svc->user_count++;
|
||||
free_user(user);
|
||||
}
|
||||
|
||||
void destroy_service(struct Service *svc) {
|
||||
free_user(svc->admin);
|
||||
free(svc);
|
||||
}
|
||||
@@ -0,0 +1,15 @@
|
||||
#ifndef SERVICE_H
|
||||
#define SERVICE_H
|
||||
|
||||
#include "user.h"
|
||||
|
||||
struct Service {
|
||||
struct User *admin;
|
||||
int user_count;
|
||||
};
|
||||
|
||||
struct Service *create_service(void);
|
||||
void service_add_user(struct Service *svc, const char *name, int age);
|
||||
void destroy_service(struct Service *svc);
|
||||
|
||||
#endif
|
||||
@@ -0,0 +1,18 @@
|
||||
#include "user.h"
|
||||
#include <stdlib.h>
|
||||
#include <string.h>
|
||||
|
||||
struct User *create_user(const char *name, int age) {
|
||||
struct User *user = malloc(sizeof(struct User));
|
||||
strncpy(user->name, name, sizeof(user->name) - 1);
|
||||
user->age = age;
|
||||
return user;
|
||||
}
|
||||
|
||||
void free_user(struct User *user) {
|
||||
free(user);
|
||||
}
|
||||
|
||||
int get_user_age(const struct User *user) {
|
||||
return user->age;
|
||||
}
|
||||
@@ -0,0 +1,13 @@
|
||||
#ifndef USER_H
|
||||
#define USER_H
|
||||
|
||||
struct User {
|
||||
char name[64];
|
||||
int age;
|
||||
};
|
||||
|
||||
struct User *create_user(const char *name, int age);
|
||||
void free_user(struct User *user);
|
||||
int get_user_age(const struct User *user);
|
||||
|
||||
#endif
|
||||
@@ -0,0 +1,53 @@
|
||||
import { vi } from 'vitest';
|
||||
import type * as SafeParseModule from '../../src/core/tree-sitter/safe-parse.js';
|
||||
|
||||
/**
|
||||
* Build a vitest mock module for `gitnexus/src/core/tree-sitter/safe-parse.ts`
|
||||
* that spies on `parseSourceSafe` while still delegating to the real
|
||||
* implementation.
|
||||
*
|
||||
* Background: tests that feed >32 767-char inputs through extractors,
|
||||
* chunkers, or any parse caller need to assert the call routed through
|
||||
* `parseSourceSafe` rather than `parser.parse(string, ...)` directly. A
|
||||
* direct call SIGSEGVs on Windows for inputs that size; on Linux/macOS it
|
||||
* succeeds, so a "no throw" assertion alone silently passes with the
|
||||
* bypass reintroduced. The spy assertion is what actually catches the
|
||||
* regression.
|
||||
*
|
||||
* Why the test still has to call `vi.mock` with a literal path: vitest's
|
||||
* hoister static-analyzes the first argument of `vi.mock`, and the path
|
||||
* varies by directory depth across test files. Everything else — the
|
||||
* `vi.importActual` round-trip, the spy installation, and the merged
|
||||
* module shape — lives here.
|
||||
*
|
||||
* Why the test still has to dynamic-`import()` this helper inside the
|
||||
* `vi.mock` factory: `vi.mock` is hoisted above static imports, so the
|
||||
* factory closure cannot reference statically-imported helpers (they are
|
||||
* uninitialized at hoist time). The factory body, however, is async and
|
||||
* runs only when the mocked module is first consumed — by which point
|
||||
* the helper resolves cleanly via dynamic `import()`.
|
||||
*
|
||||
* Usage:
|
||||
*
|
||||
* const { parseSourceSafeSpy } = vi.hoisted(() => ({ parseSourceSafeSpy: vi.fn() }));
|
||||
*
|
||||
* vi.mock('../../../src/core/tree-sitter/safe-parse.js', async () => {
|
||||
* const { buildSafeParseMock } = await import('../../helpers/parse-source-safe-mock.js');
|
||||
* return buildSafeParseMock(parseSourceSafeSpy);
|
||||
* });
|
||||
*
|
||||
* it('routes large input through parseSourceSafe', async () => {
|
||||
* parseSourceSafeSpy.mockClear();
|
||||
* // ... call extractor with >40 000-char input ...
|
||||
* expect(parseSourceSafeSpy).toHaveBeenCalled();
|
||||
* });
|
||||
*/
|
||||
export async function buildSafeParseMock(
|
||||
spy: ReturnType<typeof vi.fn>,
|
||||
): Promise<typeof SafeParseModule> {
|
||||
const actual = await vi.importActual<typeof SafeParseModule>(
|
||||
'../../src/core/tree-sitter/safe-parse.js',
|
||||
);
|
||||
spy.mockImplementation(actual.parseSourceSafe);
|
||||
return { ...actual, parseSourceSafe: spy };
|
||||
}
|
||||
@@ -0,0 +1,111 @@
|
||||
/**
|
||||
* C: struct + include-based imports + function calls across files
|
||||
*/
|
||||
import { describe, expect, beforeAll } from 'vitest';
|
||||
import path from 'path';
|
||||
import {
|
||||
FIXTURES,
|
||||
createResolverParityIt,
|
||||
getRelationships,
|
||||
getNodesByLabel,
|
||||
edgeSet,
|
||||
runPipelineFromRepo,
|
||||
type PipelineResult,
|
||||
} from './helpers.js';
|
||||
|
||||
const it = createResolverParityIt('c');
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// C structs + include-based imports + cross-file function calls
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
describe('C struct & include resolution', () => {
|
||||
let result: PipelineResult;
|
||||
|
||||
beforeAll(async () => {
|
||||
result = await runPipelineFromRepo(path.join(FIXTURES, 'c-structs'), () => {});
|
||||
}, 60000);
|
||||
|
||||
it('detects User and Service structs', () => {
|
||||
const structs = getNodesByLabel(result, 'Struct');
|
||||
expect(structs).toContain('User');
|
||||
expect(structs).toContain('Service');
|
||||
});
|
||||
|
||||
it('detects functions across all files', () => {
|
||||
const fns = getNodesByLabel(result, 'Function');
|
||||
expect(fns).toContain('main');
|
||||
expect(fns).toContain('create_user');
|
||||
expect(fns).toContain('free_user');
|
||||
expect(fns).toContain('get_user_age');
|
||||
expect(fns).toContain('create_service');
|
||||
expect(fns).toContain('service_add_user');
|
||||
expect(fns).toContain('destroy_service');
|
||||
});
|
||||
|
||||
it('resolves #include imports between .c and .h files', () => {
|
||||
const imports = getRelationships(result, 'IMPORTS');
|
||||
const edges = edgeSet(imports);
|
||||
// user.c includes user.h
|
||||
expect(edges).toContain('user.c → user.h');
|
||||
// service.h includes user.h
|
||||
expect(edges).toContain('service.h → user.h');
|
||||
// service.c includes service.h
|
||||
expect(edges).toContain('service.c → service.h');
|
||||
// main.c includes service.h
|
||||
expect(edges).toContain('main.c → service.h');
|
||||
});
|
||||
|
||||
it('emits CALLS edges for cross-file function calls', () => {
|
||||
const calls = getRelationships(result, 'CALLS');
|
||||
const edges = edgeSet(calls);
|
||||
// main.c calls functions from service
|
||||
expect(edges).toContain('main → create_service');
|
||||
expect(edges).toContain('main → service_add_user');
|
||||
expect(edges).toContain('main → destroy_service');
|
||||
// service.c calls functions from user
|
||||
expect(edges).toContain('service_add_user → create_user');
|
||||
expect(edges).toContain('service_add_user → free_user');
|
||||
expect(edges).toContain('destroy_service → free_user');
|
||||
});
|
||||
});
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// C static function isolation — static functions must NOT leak across files
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
describe('C static function isolation', () => {
|
||||
let result: PipelineResult;
|
||||
|
||||
beforeAll(async () => {
|
||||
result = await runPipelineFromRepo(path.join(FIXTURES, 'c-static-isolation'), () => {});
|
||||
}, 60000);
|
||||
|
||||
it('detects both static and non-static helper functions', () => {
|
||||
const fns = getNodesByLabel(result, 'Function');
|
||||
expect(fns).toContain('helper');
|
||||
expect(fns).toContain('public_a');
|
||||
expect(fns).toContain('public_b');
|
||||
expect(fns).toContain('main');
|
||||
});
|
||||
|
||||
it('caller.c calls b:helper via include, NOT a:static helper', () => {
|
||||
const calls = getRelationships(result, 'CALLS');
|
||||
const edges = edgeSet(calls);
|
||||
|
||||
// caller.c should call public_b (included via b.h)
|
||||
expect(edges).toContain('main → public_b');
|
||||
|
||||
// a.c's static helper calls itself locally
|
||||
expect(edges).toContain('public_a → helper');
|
||||
|
||||
// caller.c should NOT have a CALLS edge to a.c's static helper.
|
||||
// Filter edges to only those originating from main → helper to
|
||||
// verify the correct target file.
|
||||
const mainToHelper = calls.filter((r) => r.source === 'main' && r.target === 'helper');
|
||||
// If a main→helper edge exists, it should point to b.c, not a.c
|
||||
for (const edge of mainToHelper) {
|
||||
expect(edge.targetFilePath).not.toContain('a.c');
|
||||
}
|
||||
});
|
||||
});
|
||||
@@ -9,6 +9,18 @@ import type { PipelineResult } from '../../../src/types/pipeline.js';
|
||||
import type { GraphRelationship } from 'gitnexus-shared';
|
||||
|
||||
const LEGACY_RESOLVER_PARITY_EXPECTED_FAILURES: Readonly<Record<string, ReadonlySet<string>>> = {
|
||||
c: new Set([
|
||||
// The legacy DAG path does not resolve the main → create_service call
|
||||
// because the function prototype in the .h file and the definition in
|
||||
// the .c file create a dedup ambiguity. The registry-primary path
|
||||
// resolves it via scope-based wildcard import binding.
|
||||
'emits CALLS edges for cross-file function calls',
|
||||
// The legacy DAG path does not resolve cross-file calls through
|
||||
// #include → prototype chains. The scope-based path resolves
|
||||
// caller.c → b.h → public_b via wildcard import binding +
|
||||
// isFileLocalDef filtering of static functions.
|
||||
'caller.c calls b:helper via include, NOT a:static helper',
|
||||
]),
|
||||
csharp: new Set([
|
||||
'emits the using-import edge App/Program.cs -> Models/User.cs through the scope-resolution path',
|
||||
// Generic type-argument USES edges are emitted by the registry-primary
|
||||
|
||||
@@ -1,10 +1,11 @@
|
||||
import { beforeEach, describe, expect, it, vi } from 'vitest';
|
||||
|
||||
const { createParserForLanguage, getLanguageFromFilename } = vi.hoisted(() => ({
|
||||
const { createParserForLanguage, getLanguageFromFilename, parseSourceSafeSpy } = vi.hoisted(() => ({
|
||||
createParserForLanguage: vi.fn(),
|
||||
getLanguageFromFilename: vi.fn((filePath: string) =>
|
||||
filePath.endsWith('.py') ? 'python' : 'typescript',
|
||||
),
|
||||
parseSourceSafeSpy: vi.fn(),
|
||||
}));
|
||||
|
||||
vi.mock('../../src/core/tree-sitter/parser-loader.js', () => ({
|
||||
@@ -15,6 +16,11 @@ vi.mock('../../src/core/tree-sitter/parser-loader.js', () => ({
|
||||
),
|
||||
}));
|
||||
|
||||
vi.mock('../../src/core/tree-sitter/safe-parse.js', async () => {
|
||||
const { buildSafeParseMock } = await import('../helpers/parse-source-safe-mock.js');
|
||||
return buildSafeParseMock(parseSourceSafeSpy);
|
||||
});
|
||||
|
||||
vi.mock('gitnexus-shared', () => ({
|
||||
getLanguageFromFilename,
|
||||
}));
|
||||
@@ -72,4 +78,25 @@ describe('ensureAndParse', () => {
|
||||
expect(tsParse).toHaveBeenCalledTimes(2);
|
||||
expect(tsxParse).toHaveBeenCalledTimes(1);
|
||||
});
|
||||
|
||||
// Windows SIGSEGV regression: ensureAndParse must route through parseSourceSafe
|
||||
// so >32 767-char inputs do not crash the process. Direct parser.parse(content)
|
||||
// on strings that size SIGSEGVs on Windows; the spy assertion is what catches
|
||||
// a bypass since parser.parse(40 000 chars) succeeds on Linux/macOS.
|
||||
it('routes >32 767-char input through parseSourceSafe', async () => {
|
||||
parseSourceSafeSpy.mockClear();
|
||||
|
||||
const fakeParse = vi.fn().mockReturnValue({ rootNode: { type: 'module' } });
|
||||
createParserForLanguage.mockResolvedValue({ parse: fakeParse });
|
||||
|
||||
const { ensureAndParse } = await import('../../src/core/embeddings/ast-utils.js');
|
||||
|
||||
const largeInput = 'const x = 1;\n'.repeat(4000); // ~52 000 chars
|
||||
expect(largeInput.length).toBeGreaterThan(40_000);
|
||||
|
||||
const result = await ensureAndParse(largeInput, 'big.ts');
|
||||
|
||||
expect(parseSourceSafeSpy).toHaveBeenCalled();
|
||||
expect(result).not.toBeNull();
|
||||
});
|
||||
});
|
||||
|
||||
@@ -0,0 +1,458 @@
|
||||
/**
|
||||
* Regression Tests: Cursor postToolUse Hook
|
||||
*
|
||||
* Tests the hook script at gitnexus-cursor-integration/hooks/gitnexus-hook.cjs
|
||||
* which runs as a Cursor 2.4 postToolUse hook.
|
||||
*
|
||||
* Covers:
|
||||
* - extractPattern: pattern extraction from Grep/Read/Shell tool inputs
|
||||
* - findGitNexusDir: .gitnexus directory discovery (shared with Claude hook)
|
||||
* - cwd validation: rejects relative paths
|
||||
* - shell injection: verifies no `shell: true` in spawnSync calls
|
||||
* - cross-platform: Windows .cmd extension handling
|
||||
* - output shape: top-level `additional_context` (NOT Claude's `hookSpecificOutput.additionalContext`)
|
||||
* - hooks.json wiring matches the script's actual handlers
|
||||
*
|
||||
* Cursor hooks reach the augment CLI only when cwd is inside an indexed
|
||||
* repo, so behavior tests stick to early-exit paths to avoid spawning
|
||||
* `npx gitnexus`.
|
||||
*/
|
||||
import { describe, it, expect, beforeAll, afterAll } from 'vitest';
|
||||
import { spawnSync } from 'child_process';
|
||||
import fs from 'fs';
|
||||
import path from 'path';
|
||||
import os from 'os';
|
||||
import { runHook } from '../utils/hook-test-helpers.js';
|
||||
|
||||
// ─── Path to the Cursor hook + manifest ─────────────────────────────
|
||||
|
||||
const CURSOR_HOOK = path.resolve(
|
||||
__dirname,
|
||||
'..',
|
||||
'..',
|
||||
'..',
|
||||
'gitnexus-cursor-integration',
|
||||
'hooks',
|
||||
'gitnexus-hook.cjs',
|
||||
);
|
||||
const CURSOR_HOOKS_JSON = path.resolve(
|
||||
__dirname,
|
||||
'..',
|
||||
'..',
|
||||
'..',
|
||||
'gitnexus-cursor-integration',
|
||||
'hooks',
|
||||
'hooks.json',
|
||||
);
|
||||
|
||||
// ─── Cursor-specific output parser ──────────────────────────────────
|
||||
// Cursor postToolUse output shape: { "additional_context": "..." }
|
||||
|
||||
function parseCursorOutput(stdout: string): { additional_context?: string } | null {
|
||||
if (!stdout.trim()) return null;
|
||||
try {
|
||||
return JSON.parse(stdout.trim());
|
||||
} catch {
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
// ─── Test fixtures ──────────────────────────────────────────────────
|
||||
|
||||
let tmpDir: string;
|
||||
|
||||
beforeAll(() => {
|
||||
tmpDir = fs.mkdtempSync(path.join(os.tmpdir(), 'gitnexus-cursor-hook-test-'));
|
||||
spawnSync('git', ['init'], { cwd: tmpDir, stdio: 'pipe' });
|
||||
spawnSync('git', ['config', 'user.email', 'test@test.com'], { cwd: tmpDir, stdio: 'pipe' });
|
||||
spawnSync('git', ['config', 'user.name', 'Test'], { cwd: tmpDir, stdio: 'pipe' });
|
||||
});
|
||||
|
||||
afterAll(() => {
|
||||
fs.rmSync(tmpDir, { recursive: true, force: true });
|
||||
});
|
||||
|
||||
// ─── Manifest + hook file presence ───────────────────────────────────
|
||||
|
||||
describe('Cursor integration files', () => {
|
||||
it('hook script exists', () => {
|
||||
expect(fs.existsSync(CURSOR_HOOK)).toBe(true);
|
||||
});
|
||||
|
||||
it('hooks.json exists', () => {
|
||||
expect(fs.existsSync(CURSOR_HOOKS_JSON)).toBe(true);
|
||||
});
|
||||
|
||||
it('legacy augment-shell.sh has been removed', () => {
|
||||
const legacy = path.resolve(
|
||||
__dirname,
|
||||
'..',
|
||||
'..',
|
||||
'..',
|
||||
'gitnexus-cursor-integration',
|
||||
'hooks',
|
||||
'augment-shell.sh',
|
||||
);
|
||||
expect(fs.existsSync(legacy)).toBe(false);
|
||||
});
|
||||
});
|
||||
|
||||
// ─── hooks.json wiring ──────────────────────────────────────────────
|
||||
|
||||
describe('hooks.json wiring', () => {
|
||||
const manifest = JSON.parse(fs.readFileSync(CURSOR_HOOKS_JSON, 'utf-8'));
|
||||
|
||||
it('declares version 1', () => {
|
||||
expect(manifest.version).toBe(1);
|
||||
});
|
||||
|
||||
it('registers a postToolUse hook (not legacy beforeShellExecution)', () => {
|
||||
expect(manifest.hooks.postToolUse).toBeDefined();
|
||||
expect(Array.isArray(manifest.hooks.postToolUse)).toBe(true);
|
||||
expect(manifest.hooks.beforeShellExecution).toBeUndefined();
|
||||
});
|
||||
|
||||
it('matches Shell, Read, and Grep tools', () => {
|
||||
const matcher: string = manifest.hooks.postToolUse[0].matcher;
|
||||
expect(matcher).toMatch(/Shell/);
|
||||
expect(matcher).toMatch(/Read/);
|
||||
expect(matcher).toMatch(/Grep/);
|
||||
});
|
||||
|
||||
it('points command at the new Node hook', () => {
|
||||
const command: string = manifest.hooks.postToolUse[0].command;
|
||||
expect(command).toContain('gitnexus-hook.cjs');
|
||||
expect(command).not.toContain('augment-shell.sh');
|
||||
});
|
||||
|
||||
it('declares timeout in seconds (not milliseconds)', () => {
|
||||
// Cursor's `timeout` field is in seconds per
|
||||
// https://cursor.com/docs/agent/hooks. Regression guard: a value of
|
||||
// 1000+ here would be a >16-minute timeout, almost certainly a ms/s mixup.
|
||||
const timeout: number = manifest.hooks.postToolUse[0].timeout;
|
||||
expect(typeof timeout).toBe('number');
|
||||
expect(timeout).toBeGreaterThan(0);
|
||||
expect(timeout).toBeLessThan(120);
|
||||
});
|
||||
});
|
||||
|
||||
// ─── Source code regressions ────────────────────────────────────────
|
||||
|
||||
describe('Cursor hook source regressions', () => {
|
||||
const source = fs.readFileSync(CURSOR_HOOK, 'utf-8');
|
||||
|
||||
it('does not pass shell: true to spawnSync', () => {
|
||||
const lines = source.split('\n');
|
||||
for (let i = 0; i < lines.length; i++) {
|
||||
const line = lines[i];
|
||||
if (line.trim().startsWith('//') || line.trim().startsWith('*')) continue;
|
||||
if (/shell:\s*(true|isWin)/.test(line)) {
|
||||
throw new Error(`Cursor hook line ${i + 1} has shell injection risk: ${line.trim()}`);
|
||||
}
|
||||
}
|
||||
});
|
||||
|
||||
it('uses npx.cmd for Windows', () => {
|
||||
expect(source).toContain('npx.cmd');
|
||||
});
|
||||
|
||||
it('validates cwd is an absolute path', () => {
|
||||
expect(source).toMatch(/path\.isAbsolute\(cwd\)/);
|
||||
});
|
||||
|
||||
it('truncates debug error messages to 200 chars', () => {
|
||||
expect(source).toContain('.slice(0, 200)');
|
||||
});
|
||||
|
||||
it('emits Cursor-shape additional_context (not Claude hookSpecificOutput)', () => {
|
||||
expect(source).toContain('additional_context');
|
||||
expect(source).not.toContain('hookSpecificOutput');
|
||||
expect(source).not.toContain('hookEventName');
|
||||
});
|
||||
|
||||
it('rejects patterns shorter than 3 chars', () => {
|
||||
expect(source).toMatch(/length\s*>=\s*3/);
|
||||
});
|
||||
|
||||
it('passes pattern after end-of-options marker (--)', () => {
|
||||
// Regression for #200 — augment patterns starting with `-` would
|
||||
// otherwise be parsed as CLI flags by the gitnexus CLI.
|
||||
expect(source).toMatch(/'augment',\s*'--',\s*pattern/);
|
||||
});
|
||||
|
||||
it('gates on a non-global .gitnexus directory before invoking the CLI', () => {
|
||||
expect(source).toContain('findGitNexusDir');
|
||||
expect(source).toContain('isGlobalRegistryDir');
|
||||
});
|
||||
|
||||
it('handles linked git worktrees via git rev-parse --git-common-dir', () => {
|
||||
expect(source).toContain('--git-common-dir');
|
||||
});
|
||||
});
|
||||
|
||||
// ─── extractPattern coverage (source-level) ─────────────────────────
|
||||
|
||||
describe('Cursor hook extractPattern coverage', () => {
|
||||
const source = fs.readFileSync(CURSOR_HOOK, 'utf-8');
|
||||
|
||||
it("handles 'grep' tool (Cursor matcher: Grep)", () => {
|
||||
expect(source).toMatch(/t === 'grep'/);
|
||||
});
|
||||
|
||||
it('probes a wide alias set for Grep query field (Cursor contract not formally specified)', () => {
|
||||
// Cursor 2.4 docs at https://cursor.com/docs/agent/hooks list the
|
||||
// matchers but not the per-tool tool_input field names. If Cursor
|
||||
// changes the contract, we want the hook to still extract *something*
|
||||
// — these aliases plus the longest-string fallback give us coverage.
|
||||
for (const alias of ['query', 'pattern', 'regex', 'q', 'search', 'searchQuery']) {
|
||||
expect(source).toContain(`toolInput.${alias}`);
|
||||
}
|
||||
expect(source).toContain('pickLongestStringValue');
|
||||
});
|
||||
|
||||
it("handles 'read' tool (Cursor matcher: Read)", () => {
|
||||
expect(source).toMatch(/t === 'read'/);
|
||||
for (const alias of ['target_file', 'file_path', 'filePath', 'path', 'file']) {
|
||||
expect(source).toContain(`toolInput.${alias}`);
|
||||
}
|
||||
});
|
||||
|
||||
it("handles 'shell' tool (Cursor matcher: Shell)", () => {
|
||||
expect(source).toMatch(/t === 'shell'/);
|
||||
expect(source).toMatch(/\\brg\\b\|\\bgrep\\b/);
|
||||
});
|
||||
|
||||
it('logs raw payload to stderr when GITNEXUS_DEBUG is set (for contract diagnostics)', () => {
|
||||
expect(source).toContain('GITNEXUS_DEBUG');
|
||||
expect(source).toContain('GitNexus Cursor hook stdin:');
|
||||
});
|
||||
});
|
||||
|
||||
// ─── Behavior: graceful no-op paths (no augment CLI invocation) ─────
|
||||
|
||||
describe('Cursor hook behavior — early-exit paths', () => {
|
||||
it('exits cleanly on empty stdin', () => {
|
||||
const result = spawnSync(process.execPath, [CURSOR_HOOK], {
|
||||
input: '',
|
||||
encoding: 'utf-8',
|
||||
timeout: 10000,
|
||||
stdio: ['pipe', 'pipe', 'pipe'],
|
||||
});
|
||||
expect(result.status).toBe(0);
|
||||
expect(result.stdout.trim()).toBe('');
|
||||
});
|
||||
|
||||
it('exits cleanly on invalid JSON stdin', () => {
|
||||
const result = spawnSync(process.execPath, [CURSOR_HOOK], {
|
||||
input: 'not json at all',
|
||||
encoding: 'utf-8',
|
||||
timeout: 10000,
|
||||
stdio: ['pipe', 'pipe', 'pipe'],
|
||||
});
|
||||
expect(result.status).toBe(0);
|
||||
expect(result.stdout.trim()).toBe('');
|
||||
});
|
||||
|
||||
it('produces no output when cwd is relative', () => {
|
||||
const result = runHook(CURSOR_HOOK, {
|
||||
tool_name: 'Grep',
|
||||
tool_input: { query: 'validateUser' },
|
||||
cwd: 'relative/path',
|
||||
});
|
||||
expect(result.stdout.trim()).toBe('');
|
||||
expect(result.status).toBe(0);
|
||||
});
|
||||
|
||||
it('produces no output when cwd has no .gitnexus dir', () => {
|
||||
const result = runHook(CURSOR_HOOK, {
|
||||
tool_name: 'Grep',
|
||||
tool_input: { query: 'validateUser' },
|
||||
cwd: tmpDir,
|
||||
});
|
||||
expect(result.stdout.trim()).toBe('');
|
||||
expect(result.status).toBe(0);
|
||||
});
|
||||
|
||||
it('produces no output for unknown tool names', () => {
|
||||
const result = runHook(CURSOR_HOOK, {
|
||||
tool_name: 'TotallyMadeUpTool',
|
||||
tool_input: { foo: 'bar' },
|
||||
cwd: tmpDir,
|
||||
});
|
||||
expect(result.stdout.trim()).toBe('');
|
||||
expect(result.status).toBe(0);
|
||||
});
|
||||
|
||||
it('produces no output for Shell commands without rg/grep', () => {
|
||||
const result = runHook(CURSOR_HOOK, {
|
||||
tool_name: 'Shell',
|
||||
tool_input: { command: 'ls -la' },
|
||||
cwd: tmpDir,
|
||||
});
|
||||
expect(result.stdout.trim()).toBe('');
|
||||
expect(result.status).toBe(0);
|
||||
});
|
||||
|
||||
it('produces no output for Grep with a 2-char query', () => {
|
||||
const result = runHook(CURSOR_HOOK, {
|
||||
tool_name: 'Grep',
|
||||
tool_input: { query: 'is' },
|
||||
cwd: tmpDir,
|
||||
});
|
||||
expect(result.stdout.trim()).toBe('');
|
||||
expect(result.status).toBe(0);
|
||||
});
|
||||
|
||||
it('produces no output for Read whose basename has no identifier chars', () => {
|
||||
const result = runHook(CURSOR_HOOK, {
|
||||
tool_name: 'Read',
|
||||
tool_input: { target_file: '/tmp/--.md' },
|
||||
cwd: tmpDir,
|
||||
});
|
||||
expect(result.stdout.trim()).toBe('');
|
||||
expect(result.status).toBe(0);
|
||||
});
|
||||
|
||||
it('produces no output for Read with no file path', () => {
|
||||
const result = runHook(CURSOR_HOOK, {
|
||||
tool_name: 'Read',
|
||||
tool_input: {},
|
||||
cwd: tmpDir,
|
||||
});
|
||||
expect(result.stdout.trim()).toBe('');
|
||||
expect(result.status).toBe(0);
|
||||
});
|
||||
|
||||
it('treats tool_name case-insensitively (Grep vs grep)', () => {
|
||||
// Both should reach the same handler — and both should early-exit silently
|
||||
// because tmpDir has no .gitnexus.
|
||||
for (const toolName of ['Grep', 'grep', 'GREP']) {
|
||||
const result = runHook(CURSOR_HOOK, {
|
||||
tool_name: toolName,
|
||||
tool_input: { query: 'validateUser' },
|
||||
cwd: tmpDir,
|
||||
});
|
||||
expect(result.stdout.trim()).toBe('');
|
||||
expect(result.status).toBe(0);
|
||||
}
|
||||
});
|
||||
});
|
||||
|
||||
// ─── Behavior: GITNEXUS_DEBUG payload logging ────────────────────────
|
||||
|
||||
describe('Cursor hook debug logging', () => {
|
||||
it('echoes the payload to stderr only when GITNEXUS_DEBUG is set', () => {
|
||||
const payload = {
|
||||
tool_name: 'Grep',
|
||||
tool_input: { query: 'validateUser' },
|
||||
cwd: tmpDir,
|
||||
};
|
||||
|
||||
// GITNEXUS_DEBUG unset → stderr quiet.
|
||||
const quiet = spawnSync(process.execPath, [CURSOR_HOOK], {
|
||||
input: JSON.stringify(payload),
|
||||
encoding: 'utf-8',
|
||||
timeout: 10000,
|
||||
stdio: ['pipe', 'pipe', 'pipe'],
|
||||
env: { ...process.env, GITNEXUS_DEBUG: '' },
|
||||
});
|
||||
expect(quiet.status).toBe(0);
|
||||
expect(quiet.stderr).not.toContain('GitNexus Cursor hook stdin');
|
||||
|
||||
// GITNEXUS_DEBUG=1 → payload echoed to stderr (stdout still empty for
|
||||
// unindexed cwd, so the hook output contract is preserved).
|
||||
const verbose = spawnSync(process.execPath, [CURSOR_HOOK], {
|
||||
input: JSON.stringify(payload),
|
||||
encoding: 'utf-8',
|
||||
timeout: 10000,
|
||||
stdio: ['pipe', 'pipe', 'pipe'],
|
||||
env: { ...process.env, GITNEXUS_DEBUG: '1' },
|
||||
});
|
||||
expect(verbose.status).toBe(0);
|
||||
expect(verbose.stderr).toContain('GitNexus Cursor hook stdin');
|
||||
expect(verbose.stderr).toContain('"tool_name":"Grep"');
|
||||
expect(verbose.stdout.trim()).toBe('');
|
||||
});
|
||||
});
|
||||
|
||||
// ─── Documented contract behavior (extractPattern via the live hook) ─
|
||||
|
||||
describe('Shell quoted-pattern parser limitations (documented)', () => {
|
||||
// The Shell parser cannot reconstruct shell quoting. These tests pin the
|
||||
// current behavior so a future "fix" doesn't silently change extraction
|
||||
// — and so users diagnosing a noisy/missed pattern can find the behavior
|
||||
// documented in tests.
|
||||
//
|
||||
// We can't observe the extracted pattern directly without an indexed
|
||||
// repo, but we *can* confirm the hook reaches the augment-call path
|
||||
// (vs. early-exiting) by checking exit status + clean stdout for cases
|
||||
// where parseRgGrepPattern would yield a >=3-char token.
|
||||
|
||||
it('quoted multi-word `rg "User Service"` extracts the first word only', () => {
|
||||
const result = runHook(CURSOR_HOOK, {
|
||||
tool_name: 'Shell',
|
||||
tool_input: { command: 'rg "User Service" src/' },
|
||||
cwd: tmpDir, // no .gitnexus → exits early after extract
|
||||
});
|
||||
expect(result.status).toBe(0);
|
||||
expect(result.stdout.trim()).toBe('');
|
||||
});
|
||||
|
||||
it('single-token quoted `rg "validateUser"` works as expected', () => {
|
||||
const result = runHook(CURSOR_HOOK, {
|
||||
tool_name: 'Shell',
|
||||
tool_input: { command: 'rg "validateUser"' },
|
||||
cwd: tmpDir,
|
||||
});
|
||||
expect(result.status).toBe(0);
|
||||
expect(result.stdout.trim()).toBe('');
|
||||
});
|
||||
});
|
||||
|
||||
// ─── Install docs ─────────────────────────────────────────────────────
|
||||
|
||||
describe('Cursor integration install docs', () => {
|
||||
const integrationReadme = path.resolve(
|
||||
__dirname,
|
||||
'..',
|
||||
'..',
|
||||
'..',
|
||||
'gitnexus-cursor-integration',
|
||||
'README.md',
|
||||
);
|
||||
|
||||
it('install README exists', () => {
|
||||
expect(fs.existsSync(integrationReadme)).toBe(true);
|
||||
});
|
||||
|
||||
it('install README documents the hook install path', () => {
|
||||
const body = fs.readFileSync(integrationReadme, 'utf-8');
|
||||
expect(body).toContain('.cursor/hooks.json');
|
||||
expect(body).toContain('hooks/gitnexus-hook.cjs');
|
||||
expect(body).toContain('Hook install');
|
||||
});
|
||||
|
||||
it('install README documents GITNEXUS_DEBUG for payload diagnostics', () => {
|
||||
const body = fs.readFileSync(integrationReadme, 'utf-8');
|
||||
expect(body).toContain('GITNEXUS_DEBUG');
|
||||
});
|
||||
});
|
||||
|
||||
// ─── Output parser sanity (synthetic JSON) ──────────────────────────
|
||||
|
||||
describe('parseCursorOutput', () => {
|
||||
it('parses a well-formed { additional_context } payload', () => {
|
||||
const parsed = parseCursorOutput('{"additional_context":"hello"}');
|
||||
expect(parsed).not.toBeNull();
|
||||
expect(parsed?.additional_context).toBe('hello');
|
||||
});
|
||||
|
||||
it('returns null on empty stdout', () => {
|
||||
expect(parseCursorOutput('')).toBeNull();
|
||||
expect(parseCursorOutput(' \n')).toBeNull();
|
||||
});
|
||||
|
||||
it('returns null on malformed JSON', () => {
|
||||
expect(parseCursorOutput('not json')).toBeNull();
|
||||
});
|
||||
});
|
||||
@@ -7,12 +7,12 @@ import {
|
||||
buildCloneArgs,
|
||||
normalizeGitUrlForCompare,
|
||||
assertRemoteMatchesRequestedUrl,
|
||||
getRemoteOriginUrl,
|
||||
} from '../../src/server/git-clone.js';
|
||||
import path from 'node:path';
|
||||
import os from 'node:os';
|
||||
import fs from 'node:fs/promises';
|
||||
import { spawn } from 'node:child_process';
|
||||
import { getRemoteOriginUrl } from '../../src/storage/git.js';
|
||||
|
||||
describe('git-clone', () => {
|
||||
describe('extractRepoName', () => {
|
||||
@@ -50,17 +50,6 @@ describe('git-clone', () => {
|
||||
expect(() => extractRepoName('https://example.com/foo:.')).toThrow('valid repository name');
|
||||
});
|
||||
|
||||
it('rejects URLs with shell metacharacters in the last segment', () => {
|
||||
// The split on /[/:]/ does not split on backslashes or other shell chars,
|
||||
// so a name like `repo;rm -rf /` would slip through without the pattern.
|
||||
expect(() => extractRepoName('https://example.com/foo:repo;rm')).toThrow(
|
||||
'valid repository name',
|
||||
);
|
||||
expect(() => extractRepoName('https://example.com/foo:repo$x')).toThrow(
|
||||
'valid repository name',
|
||||
);
|
||||
});
|
||||
|
||||
it('rejects empty input', () => {
|
||||
expect(() => extractRepoName('')).toThrow('valid repository name');
|
||||
});
|
||||
@@ -77,6 +66,31 @@ describe('git-clone', () => {
|
||||
// multiple seconds on 10k slashes).
|
||||
expect(elapsedMs).toBeLessThan(500);
|
||||
});
|
||||
|
||||
it('strips leading dashes to prevent argument injection', () => {
|
||||
expect(extractRepoName('https://github.com/user/--upload-pack=payload.git')).toBe(
|
||||
'upload-pack_payload',
|
||||
);
|
||||
expect(extractRepoName('https://github.com/user/-repo')).toBe('repo');
|
||||
});
|
||||
|
||||
it('sanitizes unsafe directory characters', () => {
|
||||
// sanitizeRepoName turns <tag> into _tag_
|
||||
expect(extractRepoName('https://github.com/user/repo<tag>.git')).toBe('repo_tag_');
|
||||
});
|
||||
|
||||
it('sanitizes shell metacharacters in URL segments', () => {
|
||||
// The split on /[/:]/ does not split on backslashes or other shell chars,
|
||||
// so a name like `repo;rm -rf /` would slip through without the pattern.
|
||||
// After fix/sanitize-repo-name, these are sanitized to underscores.
|
||||
expect(extractRepoName('https://example.com/foo:repo;rm')).toBe('repo_rm');
|
||||
expect(extractRepoName('https://example.com/foo:repo$x')).toBe('repo_x');
|
||||
});
|
||||
|
||||
it('sanitizes whitespace and backslashes', () => {
|
||||
expect(extractRepoName('https://example.com/foo:repo name')).toBe('repo_name');
|
||||
expect(extractRepoName('https://example.com/foo:repo\\name')).toBe('repo_name');
|
||||
});
|
||||
});
|
||||
|
||||
describe('getCloneDir', () => {
|
||||
@@ -209,7 +223,7 @@ describe('git-clone', () => {
|
||||
|
||||
it('blocks NAT64 with embedded RFC1918 addresses', () => {
|
||||
// The startsWith('64:ff9b:') check covers any embedded IPv4. These
|
||||
// explicit RFC1918 cases document SSRF coverage for the full private
|
||||
// explicit RFC1918 architectures document SSRF coverage for the full private
|
||||
// IPv4 surface — not just loopback and cloud metadata.
|
||||
expect(() => validateGitUrl('http://[64:ff9b::a00:1]/repo.git')).toThrow('private/internal'); // 10.0.0.1
|
||||
expect(() => validateGitUrl('http://[64:ff9b::ac10:1]/repo.git')).toThrow('private/internal'); // 172.16.0.1
|
||||
|
||||
@@ -8,6 +8,8 @@ import {
|
||||
getCurrentCommit,
|
||||
getGitRoot,
|
||||
findGitRootByDotGit,
|
||||
parseRepoNameFromUrl,
|
||||
sanitizeRepoName,
|
||||
} from '../../src/storage/git.js';
|
||||
|
||||
// Mock child_process.execSync
|
||||
@@ -164,4 +166,61 @@ describe('git utilities', () => {
|
||||
}
|
||||
});
|
||||
});
|
||||
|
||||
describe('sanitizeRepoName', () => {
|
||||
it('strips leading dashes', () => {
|
||||
expect(sanitizeRepoName('--repo')).toBe('repo');
|
||||
});
|
||||
|
||||
it('replaces unsafe characters with underscores', () => {
|
||||
expect(sanitizeRepoName('repo<tag>')).toBe('repo_tag_');
|
||||
expect(sanitizeRepoName('repo:name')).toBe('repo_name');
|
||||
expect(sanitizeRepoName('repo"quoted"')).toBe('repo_quoted_');
|
||||
});
|
||||
|
||||
it('blocks path traversal segments', () => {
|
||||
expect(sanitizeRepoName('.')).toBe('unknown');
|
||||
expect(sanitizeRepoName('..')).toBe('unknown');
|
||||
});
|
||||
|
||||
it('blocks Windows reserved names', () => {
|
||||
expect(sanitizeRepoName('CON')).toBe('unknown');
|
||||
expect(sanitizeRepoName('prn')).toBe('unknown');
|
||||
expect(sanitizeRepoName('AUX')).toBe('unknown');
|
||||
expect(sanitizeRepoName('NUL')).toBe('unknown');
|
||||
expect(sanitizeRepoName('COM1')).toBe('unknown');
|
||||
expect(sanitizeRepoName('LPT9')).toBe('unknown');
|
||||
|
||||
// Reserved names with extensions
|
||||
expect(sanitizeRepoName('CON.txt')).toBe('unknown');
|
||||
expect(sanitizeRepoName('NUL.tar.gz')).toBe('unknown');
|
||||
expect(sanitizeRepoName('AUX.local')).toBe('unknown');
|
||||
});
|
||||
|
||||
it('returns unknown for empty or invalid input', () => {
|
||||
expect(sanitizeRepoName('')).toBe('unknown');
|
||||
expect(sanitizeRepoName('---')).toBe('unknown');
|
||||
});
|
||||
});
|
||||
|
||||
describe('parseRepoNameFromUrl', () => {
|
||||
it('extracts and sanitizes name from HTTPS URL', () => {
|
||||
expect(parseRepoNameFromUrl('https://github.com/user/my-repo.git')).toBe('my-repo');
|
||||
expect(parseRepoNameFromUrl('https://github.com/user/--payload.git')).toBe('payload');
|
||||
});
|
||||
|
||||
it('extracts and sanitizes name from SSH URL', () => {
|
||||
expect(parseRepoNameFromUrl('git@github.com:user/my-repo.git')).toBe('my-repo');
|
||||
expect(parseRepoNameFromUrl('git@github.com:--payload.git')).toBe('payload');
|
||||
});
|
||||
|
||||
it('returns null for all-dash inputs (prevents registry collision)', () => {
|
||||
expect(parseRepoNameFromUrl('https://github.com/user/---.git')).toBeNull();
|
||||
});
|
||||
|
||||
it('returns null for empty URL', () => {
|
||||
expect(parseRepoNameFromUrl('')).toBeNull();
|
||||
expect(parseRepoNameFromUrl(null)).toBeNull();
|
||||
});
|
||||
});
|
||||
});
|
||||
|
||||
@@ -1,8 +1,15 @@
|
||||
import { describe, it, expect, beforeEach, afterEach } from 'vitest';
|
||||
import { describe, it, expect, beforeEach, afterEach, vi } from 'vitest';
|
||||
import * as fs from 'node:fs';
|
||||
import fsp from 'node:fs/promises';
|
||||
import * as path from 'node:path';
|
||||
import * as os from 'node:os';
|
||||
|
||||
const { parseSourceSafeSpy } = vi.hoisted(() => ({ parseSourceSafeSpy: vi.fn() }));
|
||||
|
||||
vi.mock('../../../src/core/tree-sitter/safe-parse.js', async () => {
|
||||
const { buildSafeParseMock } = await import('../../helpers/parse-source-safe-mock.js');
|
||||
return buildSafeParseMock(parseSourceSafeSpy);
|
||||
});
|
||||
import {
|
||||
GrpcExtractor,
|
||||
buildProtoMap,
|
||||
@@ -677,6 +684,33 @@ stub = leaked_pb2_grpc.LeakedServiceStub(channel)`,
|
||||
).toBe(false);
|
||||
});
|
||||
});
|
||||
|
||||
describe('Windows SIGSEGV regression — large input must route through parseSourceSafe', () => {
|
||||
it('routes >32 767-char source file through parseSourceSafe (not direct parser.parse)', async () => {
|
||||
parseSourceSafeSpy.mockClear();
|
||||
|
||||
// Synthesize a >40 000-char source file in a language whose grpc plugin
|
||||
// is always available (Go has no optional grammar — the Go plugin is
|
||||
// unconditionally wired in grpc-patterns/index.ts). Direct
|
||||
// parser.parse(content) on an input this size SIGSEGVs the process on
|
||||
// Windows; parseSourceSafe routes through the chunked-callback path and
|
||||
// works on every platform. The spy assertion is what catches the
|
||||
// regression — a "no throw" assertion alone is satisfied by the bypass
|
||||
// on Linux/macOS where parser.parse(40 000 chars) succeeds.
|
||||
const padding = Array.from(
|
||||
{ length: 600 },
|
||||
(_, i) => `func helper${i}() string { return "padding-${i}-aaaaaaaaaaaaaaaaaaaaaa" }\n`,
|
||||
).join('');
|
||||
const largeGo = `package big\n\n${padding}\n`;
|
||||
expect(largeGo.length).toBeGreaterThan(40_000);
|
||||
|
||||
writeFile('server/big.go', largeGo);
|
||||
|
||||
await extractor.extract(null, tmpDir, makeRepo(tmpDir));
|
||||
|
||||
expect(parseSourceSafeSpy).toHaveBeenCalled();
|
||||
});
|
||||
});
|
||||
});
|
||||
|
||||
describe('buildProtoMap', () => {
|
||||
|
||||
@@ -1,7 +1,15 @@
|
||||
import { describe, it, expect, beforeEach, afterEach } from 'vitest';
|
||||
import { describe, it, expect, beforeEach, afterEach, vi } from 'vitest';
|
||||
import * as fs from 'node:fs';
|
||||
import * as path from 'node:path';
|
||||
import * as os from 'node:os';
|
||||
|
||||
const { parseSourceSafeSpy } = vi.hoisted(() => ({ parseSourceSafeSpy: vi.fn() }));
|
||||
|
||||
vi.mock('../../../src/core/tree-sitter/safe-parse.js', async () => {
|
||||
const { buildSafeParseMock } = await import('../../helpers/parse-source-safe-mock.js');
|
||||
return buildSafeParseMock(parseSourceSafeSpy);
|
||||
});
|
||||
|
||||
import { HttpRouteExtractor } from '../../../src/core/group/extractors/http-route-extractor.js';
|
||||
import type { RepoHandle } from '../../../src/core/group/types.js';
|
||||
|
||||
@@ -815,4 +823,34 @@ export default r;
|
||||
expect(contracts.some((c) => c.symbolRef?.filePath?.startsWith('mentor_env/'))).toBe(false);
|
||||
});
|
||||
});
|
||||
|
||||
describe('Windows SIGSEGV regression — large input must route through parseSourceSafe', () => {
|
||||
it('routes >32 767-char source file through parseSourceSafe (not direct parser.parse)', async () => {
|
||||
parseSourceSafeSpy.mockClear();
|
||||
|
||||
// >40 000-char Java controller file. Direct parser.parse(content) on
|
||||
// an input this size SIGSEGVs the process on Windows. The spy assertion
|
||||
// is what catches the regression — a "no throw" assertion alone is
|
||||
// satisfied by the bypass on Linux/macOS where parser.parse(40 000 chars)
|
||||
// succeeds.
|
||||
const padding = Array.from(
|
||||
{ length: 600 },
|
||||
(_, i) => ` public String helper${i}() { return "padding-${i}-aaaaaaaaaaaaaaaaaaa"; }\n`,
|
||||
).join('');
|
||||
const largeJava = `package com.example;\n\n@RestController\npublic class BigController {\n${padding}}\n`;
|
||||
expect(largeJava.length).toBeGreaterThan(40_000);
|
||||
|
||||
// Use mkdtempSync rather than a fixed subdir name: satisfies CodeQL's
|
||||
// js/insecure-temporary-file rule by generating a unique random suffix
|
||||
// instead of relying on the parent tmpDir's predictable Date.now() name.
|
||||
const dir = fs.mkdtempSync(path.join(tmpDir, 'large-input-'));
|
||||
fs.mkdirSync(path.join(dir, 'src/controller'), { recursive: true });
|
||||
fs.writeFileSync(path.join(dir, 'src/controller/BigController.java'), largeJava);
|
||||
|
||||
const mockDbExecutor = async (_query: string) => [];
|
||||
await extractor.extract(mockDbExecutor, dir, makeRepo(dir));
|
||||
|
||||
expect(parseSourceSafeSpy).toHaveBeenCalled();
|
||||
});
|
||||
});
|
||||
});
|
||||
|
||||
@@ -1,7 +1,15 @@
|
||||
import { describe, it, expect, beforeEach, afterEach } from 'vitest';
|
||||
import { describe, it, expect, beforeEach, afterEach, vi } from 'vitest';
|
||||
import * as fs from 'node:fs';
|
||||
import * as path from 'node:path';
|
||||
import * as os from 'node:os';
|
||||
|
||||
const { parseSourceSafeSpy } = vi.hoisted(() => ({ parseSourceSafeSpy: vi.fn() }));
|
||||
|
||||
vi.mock('../../../src/core/tree-sitter/safe-parse.js', async () => {
|
||||
const { buildSafeParseMock } = await import('../../helpers/parse-source-safe-mock.js');
|
||||
return buildSafeParseMock(parseSourceSafeSpy);
|
||||
});
|
||||
|
||||
import { IncludeExtractor } from '../../../src/core/group/extractors/include-extractor.js';
|
||||
import type { RepoHandle } from '../../../src/core/group/types.js';
|
||||
import { normalizeContractId } from '../../../src/core/group/matching.js';
|
||||
@@ -560,4 +568,35 @@ int auto_main() { return 0; }`,
|
||||
}
|
||||
});
|
||||
});
|
||||
|
||||
describe('Windows SIGSEGV regression — large input must route through parseSourceSafe', () => {
|
||||
it('routes >32 767-char header file through parseSourceSafe (not direct parser.parse)', async () => {
|
||||
parseSourceSafeSpy.mockClear();
|
||||
|
||||
// Bump the file-size cap so the >40 000-char file isn't filtered before
|
||||
// it ever reaches the parser. Direct parser.parse(content) on a string
|
||||
// this size SIGSEGVs the process on Windows. The spy assertion catches
|
||||
// the regression — a "no throw" assertion alone is satisfied by the
|
||||
// bypass on Linux/macOS where parser.parse(40 000 chars) succeeds.
|
||||
const previousLimit = process.env.GITNEXUS_MAX_FILE_SIZE;
|
||||
process.env.GITNEXUS_MAX_FILE_SIZE = '512';
|
||||
try {
|
||||
const includes = Array.from(
|
||||
{ length: 1500 },
|
||||
(_, i) => `#include "lib/header_${i}.h"\n`,
|
||||
).join('');
|
||||
const largeHeader = `#pragma once\n${includes}\nstruct Big {};\n`;
|
||||
expect(largeHeader.length).toBeGreaterThan(40_000);
|
||||
|
||||
writeFile('big/big.cpp', largeHeader);
|
||||
|
||||
await extractor.extract(null, tmpDir, makeRepo(tmpDir));
|
||||
|
||||
expect(parseSourceSafeSpy).toHaveBeenCalled();
|
||||
} finally {
|
||||
if (previousLimit === undefined) delete process.env.GITNEXUS_MAX_FILE_SIZE;
|
||||
else process.env.GITNEXUS_MAX_FILE_SIZE = previousLimit;
|
||||
}
|
||||
});
|
||||
});
|
||||
});
|
||||
|
||||
@@ -1,8 +1,16 @@
|
||||
import { describe, it, expect, beforeEach, afterEach } from 'vitest';
|
||||
import { describe, it, expect, beforeEach, afterEach, vi } from 'vitest';
|
||||
import * as fs from 'node:fs';
|
||||
import fsp from 'node:fs/promises';
|
||||
import * as path from 'node:path';
|
||||
import * as os from 'node:os';
|
||||
|
||||
const { parseSourceSafeSpy } = vi.hoisted(() => ({ parseSourceSafeSpy: vi.fn() }));
|
||||
|
||||
vi.mock('../../../src/core/tree-sitter/safe-parse.js', async () => {
|
||||
const { buildSafeParseMock } = await import('../../helpers/parse-source-safe-mock.js');
|
||||
return buildSafeParseMock(parseSourceSafeSpy);
|
||||
});
|
||||
|
||||
import {
|
||||
ThriftExtractor,
|
||||
buildThriftContext,
|
||||
@@ -580,6 +588,39 @@ class PaymentWorkflow {
|
||||
|
||||
expect(contracts).toEqual([]);
|
||||
});
|
||||
|
||||
describe('Windows SIGSEGV regression — large input must route through parseSourceSafe', () => {
|
||||
it('routes >32 767-char source file through parseSourceSafe (not direct parser.parse)', async () => {
|
||||
parseSourceSafeSpy.mockClear();
|
||||
|
||||
// Need a base .thrift file so buildThriftContext finds at least one
|
||||
// service to scan; without it the source-scan loop short-circuits.
|
||||
writeFile(
|
||||
'idl/order.thrift',
|
||||
`namespace java billing.v1
|
||||
service OrderService {
|
||||
PlaceOrderResponse PlaceOrder(1: PlaceOrderRequest request)
|
||||
}`,
|
||||
);
|
||||
|
||||
// >40 000-char Java client file. Direct parser.parse(content) on a
|
||||
// string this size SIGSEGVs the process on Windows. The spy assertion
|
||||
// catches the regression — a "no throw" assertion alone is satisfied
|
||||
// by the bypass on Linux/macOS where parser.parse(40 000 chars) succeeds.
|
||||
const padding = Array.from(
|
||||
{ length: 600 },
|
||||
(_, i) => ` public String helper${i}() { return "padding-${i}-aaaaaaaaaaaaaaaaaaa"; }\n`,
|
||||
).join('');
|
||||
const largeJava = `package com.example;\n\nimport billing.v1.OrderService;\n\npublic class BigClient {\n private OrderService.Iface client;\n${padding}}\n`;
|
||||
expect(largeJava.length).toBeGreaterThan(40_000);
|
||||
|
||||
writeFile('src/main/java/com/example/BigClient.java', largeJava);
|
||||
|
||||
await extractor.extract(null, tmpDir, makeRepo(tmpDir));
|
||||
|
||||
expect(parseSourceSafeSpy).toHaveBeenCalled();
|
||||
});
|
||||
});
|
||||
});
|
||||
|
||||
describe('buildThriftContext', () => {
|
||||
|
||||
@@ -153,6 +153,7 @@ describe('primaryLanguages', () => {
|
||||
process.env['REGISTRY_PRIMARY_CSHARP'] = 'false';
|
||||
process.env['REGISTRY_PRIMARY_TYPESCRIPT'] = 'false';
|
||||
process.env['REGISTRY_PRIMARY_GO'] = 'false';
|
||||
process.env['REGISTRY_PRIMARY_C'] = 'false';
|
||||
process.env['REGISTRY_PRIMARY_JAVA'] = '1';
|
||||
const enabled = primaryLanguages();
|
||||
expect(enabled.has(SupportedLanguages.Python)).toBe(false);
|
||||
|
||||
@@ -0,0 +1,74 @@
|
||||
import { describe, it, expect } from 'vitest';
|
||||
import Parser from 'tree-sitter';
|
||||
import Python from 'tree-sitter-python';
|
||||
import { parseSourceSafe } from '../../src/core/tree-sitter/safe-parse.js';
|
||||
|
||||
const makeParser = (): Parser => {
|
||||
const p = new Parser();
|
||||
p.setLanguage(Python);
|
||||
return p;
|
||||
};
|
||||
|
||||
const buildSource = (chars: number, lineLen = 80): string => {
|
||||
const line = 'x = 1' + ' '.repeat(Math.max(0, lineLen - 6)) + '\n';
|
||||
const lines = Math.ceil(chars / line.length);
|
||||
return line.repeat(lines).slice(0, chars);
|
||||
};
|
||||
|
||||
describe('parseSourceSafe', () => {
|
||||
it('parses small ASCII sources via the direct path', () => {
|
||||
const tree = parseSourceSafe(makeParser(), 'x = 1\n');
|
||||
expect(tree.rootNode.type).toBe('module');
|
||||
expect(tree.rootNode.hasError).toBe(false);
|
||||
});
|
||||
|
||||
it('parses sources at the direct/callback boundary (16 KiB)', () => {
|
||||
const src = buildSource(16 * 1024);
|
||||
const tree = parseSourceSafe(makeParser(), src);
|
||||
expect(tree.rootNode.hasError).toBe(false);
|
||||
expect(tree.rootNode.endIndex).toBe(src.length);
|
||||
});
|
||||
|
||||
it('parses sources just above the boundary via the callback path', () => {
|
||||
const src = buildSource(16 * 1024 + 1);
|
||||
const tree = parseSourceSafe(makeParser(), src);
|
||||
expect(tree.rootNode.hasError).toBe(false);
|
||||
expect(tree.rootNode.endIndex).toBe(src.length);
|
||||
});
|
||||
|
||||
it('parses sources at and around the 32 767-char Windows crash boundary', () => {
|
||||
for (const len of [32_766, 32_767, 32_768]) {
|
||||
const src = buildSource(len);
|
||||
const tree = parseSourceSafe(makeParser(), src);
|
||||
expect(tree.rootNode.hasError, `len=${len}`).toBe(false);
|
||||
expect(tree.rootNode.endIndex, `len=${len}`).toBe(src.length);
|
||||
}
|
||||
});
|
||||
|
||||
it('parses a single line longer than the chunk size (no newlines)', () => {
|
||||
const src = '"' + 'a'.repeat(20_000) + '"\n';
|
||||
const tree = parseSourceSafe(makeParser(), src);
|
||||
expect(tree.rootNode.hasError).toBe(false);
|
||||
expect(tree.rootNode.endIndex).toBe(src.length);
|
||||
});
|
||||
|
||||
it('parses sources with CRLF line endings near a chunk boundary', () => {
|
||||
const line = 'x = 1' + ' '.repeat(75) + '\r\n';
|
||||
const src = line.repeat(Math.ceil(20_000 / line.length));
|
||||
const tree = parseSourceSafe(makeParser(), src);
|
||||
expect(tree.rootNode.hasError).toBe(false);
|
||||
expect(tree.rootNode.endIndex).toBe(src.length);
|
||||
});
|
||||
|
||||
it('parses a large all-non-ASCII source identically to the direct path', () => {
|
||||
const small = '# ' + '漢'.repeat(50) + '\n';
|
||||
const direct = makeParser().parse(small);
|
||||
const safe = parseSourceSafe(makeParser(), small);
|
||||
expect(safe.rootNode.toString()).toBe(direct.rootNode.toString());
|
||||
|
||||
const large = ('# ' + '漢'.repeat(8_000) + '\n').repeat(3);
|
||||
const tree = parseSourceSafe(makeParser(), large);
|
||||
expect(tree.rootNode.hasError).toBe(false);
|
||||
expect(tree.rootNode.endIndex).toBe(large.length);
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,212 @@
|
||||
/**
|
||||
* Unit tests for C arity computation and compatibility.
|
||||
*/
|
||||
|
||||
import { describe, it, expect } from 'vitest';
|
||||
import { getCParser } from '../../../../src/core/ingestion/languages/c/query.js';
|
||||
import {
|
||||
computeCDeclarationArity,
|
||||
computeCCallArity,
|
||||
} from '../../../../src/core/ingestion/languages/c/arity-metadata.js';
|
||||
import { cArityCompatibility } from '../../../../src/core/ingestion/languages/c/arity.js';
|
||||
import type { SyntaxNode } from '../../../../src/core/ingestion/utils/ast-helpers.js';
|
||||
import type { Callsite, SymbolDefinition } from 'gitnexus-shared';
|
||||
|
||||
function parseFunctionNode(src: string): SyntaxNode | null {
|
||||
const tree = getCParser().parse(src);
|
||||
for (let i = 0; i < tree.rootNode.namedChildCount; i++) {
|
||||
const child = tree.rootNode.namedChild(i);
|
||||
if (child?.type === 'function_definition' || child?.type === 'declaration') {
|
||||
return child as SyntaxNode;
|
||||
}
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
function parseCallNode(src: string): SyntaxNode | null {
|
||||
const tree = getCParser().parse(src);
|
||||
// Walk deeper to find call_expression
|
||||
function findCall(node: SyntaxNode): SyntaxNode | null {
|
||||
if (node.type === 'call_expression') return node;
|
||||
for (let i = 0; i < node.namedChildCount; i++) {
|
||||
const found = findCall(node.namedChild(i) as SyntaxNode);
|
||||
if (found !== null) return found;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
return findCall(tree.rootNode as SyntaxNode);
|
||||
}
|
||||
|
||||
describe('computeCDeclarationArity', () => {
|
||||
it('returns count for simple parameters', () => {
|
||||
const node = parseFunctionNode('int add(int a, int b) { return a + b; }');
|
||||
expect(node).not.toBeNull();
|
||||
const arity = computeCDeclarationArity(node!);
|
||||
expect(arity.parameterCount).toBe(2);
|
||||
expect(arity.requiredParameterCount).toBe(2);
|
||||
});
|
||||
|
||||
it('returns zero for (void) parameter list', () => {
|
||||
const node = parseFunctionNode('void f(void) { }');
|
||||
expect(node).not.toBeNull();
|
||||
const arity = computeCDeclarationArity(node!);
|
||||
expect(arity.parameterCount).toBe(0);
|
||||
expect(arity.requiredParameterCount).toBe(0);
|
||||
expect(arity.parameterTypes).toEqual([]);
|
||||
});
|
||||
|
||||
it('handles variadic functions — parameterCount is undefined', () => {
|
||||
const node = parseFunctionNode('int printf(const char *fmt, ...) { return 0; }');
|
||||
expect(node).not.toBeNull();
|
||||
const arity = computeCDeclarationArity(node!);
|
||||
expect(arity.parameterCount).toBeUndefined();
|
||||
expect(arity.requiredParameterCount).toBe(1);
|
||||
expect(arity.parameterTypes).toContain('...');
|
||||
});
|
||||
|
||||
it('extracts parameter types', () => {
|
||||
const node = parseFunctionNode('void f(int a, float b, char *c) { }');
|
||||
expect(node).not.toBeNull();
|
||||
const arity = computeCDeclarationArity(node!);
|
||||
expect(arity.parameterTypes).toEqual(['int', 'float', 'char']);
|
||||
});
|
||||
|
||||
it('handles pointer-return function', () => {
|
||||
const node = parseFunctionNode('int *create(int size) { return 0; }');
|
||||
expect(node).not.toBeNull();
|
||||
const arity = computeCDeclarationArity(node!);
|
||||
expect(arity.parameterCount).toBe(1);
|
||||
});
|
||||
|
||||
it('handles function prototype (no body)', () => {
|
||||
const node = parseFunctionNode('int add(int a, int b);');
|
||||
expect(node).not.toBeNull();
|
||||
const arity = computeCDeclarationArity(node!);
|
||||
expect(arity.parameterCount).toBe(2);
|
||||
});
|
||||
|
||||
it('returns empty for non-function node', () => {
|
||||
const node = parseFunctionNode('int x = 5;');
|
||||
// This might be a declaration node, but without function_declarator
|
||||
if (node !== null) {
|
||||
const arity = computeCDeclarationArity(node);
|
||||
expect(arity.parameterCount).toBeUndefined();
|
||||
}
|
||||
});
|
||||
|
||||
it('handles single parameter', () => {
|
||||
const node = parseFunctionNode('void f(int x) { }');
|
||||
expect(node).not.toBeNull();
|
||||
const arity = computeCDeclarationArity(node!);
|
||||
expect(arity.parameterCount).toBe(1);
|
||||
expect(arity.requiredParameterCount).toBe(1);
|
||||
});
|
||||
|
||||
it('returns unknown arity for K&R empty parameter list int foo()', () => {
|
||||
const node = parseFunctionNode('int foo() { return 0; }');
|
||||
expect(node).not.toBeNull();
|
||||
const arity = computeCDeclarationArity(node!);
|
||||
// K&R old-style: unspecified parameters, NOT zero parameters
|
||||
expect(arity.parameterCount).toBeUndefined();
|
||||
expect(arity.requiredParameterCount).toBeUndefined();
|
||||
expect(arity.parameterTypes).toBeUndefined();
|
||||
});
|
||||
|
||||
it('distinguishes K&R int foo() from explicit int foo(void)', () => {
|
||||
const knrNode = parseFunctionNode('int foo() { return 0; }');
|
||||
const voidNode = parseFunctionNode('int foo(void) { return 0; }');
|
||||
expect(knrNode).not.toBeNull();
|
||||
expect(voidNode).not.toBeNull();
|
||||
|
||||
const knrArity = computeCDeclarationArity(knrNode!);
|
||||
const voidArity = computeCDeclarationArity(voidNode!);
|
||||
|
||||
// K&R: unknown arity
|
||||
expect(knrArity.parameterCount).toBeUndefined();
|
||||
// Explicit void: zero params
|
||||
expect(voidArity.parameterCount).toBe(0);
|
||||
expect(voidArity.requiredParameterCount).toBe(0);
|
||||
});
|
||||
|
||||
it('returns unknown arity for K&R prototype int foo();', () => {
|
||||
const node = parseFunctionNode('int foo();');
|
||||
expect(node).not.toBeNull();
|
||||
const arity = computeCDeclarationArity(node!);
|
||||
expect(arity.parameterCount).toBeUndefined();
|
||||
expect(arity.requiredParameterCount).toBeUndefined();
|
||||
});
|
||||
});
|
||||
|
||||
describe('computeCCallArity', () => {
|
||||
it('counts zero arguments', () => {
|
||||
const node = parseCallNode('void f(void) { init(); }');
|
||||
expect(node).not.toBeNull();
|
||||
expect(computeCCallArity(node!)).toBe(0);
|
||||
});
|
||||
|
||||
it('counts two arguments', () => {
|
||||
const node = parseCallNode('void f(void) { add(1, 2); }');
|
||||
expect(node).not.toBeNull();
|
||||
expect(computeCCallArity(node!)).toBe(2);
|
||||
});
|
||||
|
||||
it('counts three arguments', () => {
|
||||
const node = parseCallNode('void f(void) { func(a, b, c); }');
|
||||
expect(node).not.toBeNull();
|
||||
expect(computeCCallArity(node!)).toBe(3);
|
||||
});
|
||||
|
||||
it('counts string literal arguments', () => {
|
||||
const node = parseCallNode('void f(void) { printf("hello %s", name); }');
|
||||
expect(node).not.toBeNull();
|
||||
expect(computeCCallArity(node!)).toBe(2);
|
||||
});
|
||||
});
|
||||
|
||||
describe('cArityCompatibility', () => {
|
||||
function makeDef(params: Partial<SymbolDefinition>): SymbolDefinition {
|
||||
return {
|
||||
nodeId: 'test',
|
||||
filePath: 'test.c',
|
||||
type: 'Function',
|
||||
...params,
|
||||
};
|
||||
}
|
||||
|
||||
function makeCallsite(arity: number): Callsite {
|
||||
return { arity } as Callsite;
|
||||
}
|
||||
|
||||
it('returns compatible for exact match', () => {
|
||||
const def = makeDef({ parameterCount: 2, requiredParameterCount: 2 });
|
||||
expect(cArityCompatibility(def, makeCallsite(2))).toBe('compatible');
|
||||
});
|
||||
|
||||
it('returns incompatible for too few args', () => {
|
||||
const def = makeDef({ parameterCount: 3, requiredParameterCount: 3 });
|
||||
expect(cArityCompatibility(def, makeCallsite(1))).toBe('incompatible');
|
||||
});
|
||||
|
||||
it('returns incompatible for too many args (non-variadic)', () => {
|
||||
const def = makeDef({ parameterCount: 2, requiredParameterCount: 2 });
|
||||
expect(cArityCompatibility(def, makeCallsite(5))).toBe('incompatible');
|
||||
});
|
||||
|
||||
it('returns compatible for variadic with enough args', () => {
|
||||
const def = makeDef({
|
||||
requiredParameterCount: 1,
|
||||
parameterTypes: ['const char *', '...'],
|
||||
});
|
||||
expect(cArityCompatibility(def, makeCallsite(3))).toBe('compatible');
|
||||
});
|
||||
|
||||
it('returns unknown when no arity info on def', () => {
|
||||
const def = makeDef({});
|
||||
expect(cArityCompatibility(def, makeCallsite(2))).toBe('unknown');
|
||||
});
|
||||
|
||||
it('returns unknown for negative callsite arity', () => {
|
||||
const def = makeDef({ parameterCount: 2, requiredParameterCount: 2 });
|
||||
expect(cArityCompatibility(def, makeCallsite(-1))).toBe('unknown');
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,355 @@
|
||||
/**
|
||||
* Unit tests for C scope query + captures orchestrator.
|
||||
*
|
||||
* Pins the capture-tag vocabulary + range shape for every construct
|
||||
* the scope-resolution pipeline reads. Runs against tree-sitter-c
|
||||
* so it catches grammar drift before the integration parity gate does.
|
||||
*/
|
||||
|
||||
import { describe, it, expect, beforeEach } from 'vitest';
|
||||
import { emitCScopeCaptures } from '../../../../src/core/ingestion/languages/c/captures.js';
|
||||
import {
|
||||
clearStaticNames,
|
||||
isStaticName,
|
||||
} from '../../../../src/core/ingestion/languages/c/static-linkage.js';
|
||||
|
||||
function tagsFor(src: string, filePath = 'test.c'): string[][] {
|
||||
const matches = emitCScopeCaptures(src, filePath);
|
||||
return matches.map((m) => Object.keys(m).sort());
|
||||
}
|
||||
|
||||
function findMatch(src: string, predicate: (tags: string[]) => boolean, filePath = 'test.c') {
|
||||
const matches = emitCScopeCaptures(src, filePath);
|
||||
return matches.find((m) => predicate(Object.keys(m)));
|
||||
}
|
||||
|
||||
function allMatches(src: string, predicate: (tags: string[]) => boolean, filePath = 'test.c') {
|
||||
const matches = emitCScopeCaptures(src, filePath);
|
||||
return matches.filter((m) => predicate(Object.keys(m)));
|
||||
}
|
||||
|
||||
describe('emitCScopeCaptures — scopes', () => {
|
||||
it('captures translation_unit as @scope.module', () => {
|
||||
const all = tagsFor('int x = 1;');
|
||||
expect(all.some((t) => t.includes('@scope.module'))).toBe(true);
|
||||
});
|
||||
|
||||
it('captures struct_specifier as @scope.class', () => {
|
||||
const all = tagsFor('struct Point { int x; int y; };');
|
||||
expect(all.some((t) => t.includes('@scope.class'))).toBe(true);
|
||||
});
|
||||
|
||||
it('captures union_specifier as @scope.class', () => {
|
||||
const all = tagsFor('union Data { int i; float f; };');
|
||||
expect(all.some((t) => t.includes('@scope.class'))).toBe(true);
|
||||
});
|
||||
|
||||
it('captures function_definition as @scope.function', () => {
|
||||
const all = tagsFor('void foo(void) { }');
|
||||
expect(all.some((t) => t.includes('@scope.function'))).toBe(true);
|
||||
});
|
||||
|
||||
it('captures block-level scopes (if, for, while, do, switch, case)', () => {
|
||||
const src = `
|
||||
void f(void) {
|
||||
if (1) { }
|
||||
for (;;) { }
|
||||
while (1) { }
|
||||
do { } while (0);
|
||||
switch (0) { case 0: break; }
|
||||
}
|
||||
`;
|
||||
const all = tagsFor(src);
|
||||
const blocks = all.filter((t) => t.includes('@scope.block'));
|
||||
// compound_statement + if + for + while + do + switch + case = at least 6 blocks
|
||||
expect(blocks.length).toBeGreaterThanOrEqual(6);
|
||||
});
|
||||
});
|
||||
|
||||
describe('emitCScopeCaptures — struct declarations', () => {
|
||||
it('captures named struct with @declaration.struct', () => {
|
||||
const m = findMatch('struct User { int age; };', (t) => t.includes('@declaration.struct'));
|
||||
expect(m).toBeDefined();
|
||||
expect(m!['@declaration.name'].text).toBe('User');
|
||||
});
|
||||
|
||||
it('captures typedef struct with @declaration.struct (not typedef)', () => {
|
||||
const m = findMatch('typedef struct { int age; } User;', (t) =>
|
||||
t.includes('@declaration.struct'),
|
||||
);
|
||||
expect(m).toBeDefined();
|
||||
expect(m!['@declaration.name'].text).toBe('User');
|
||||
});
|
||||
|
||||
it('suppresses @declaration.typedef when struct already captured same range', () => {
|
||||
const matches = emitCScopeCaptures('typedef struct { int age; } User;', 'test.c');
|
||||
const typedefs = matches.filter((m) => '@declaration.typedef' in m);
|
||||
expect(typedefs).toHaveLength(0);
|
||||
});
|
||||
});
|
||||
|
||||
describe('emitCScopeCaptures — union declarations', () => {
|
||||
it('captures named union with @declaration.union', () => {
|
||||
const m = findMatch('union Data { int i; float f; };', (t) => t.includes('@declaration.union'));
|
||||
expect(m).toBeDefined();
|
||||
expect(m!['@declaration.name'].text).toBe('Data');
|
||||
});
|
||||
|
||||
it('captures typedef union with @declaration.union', () => {
|
||||
const m = findMatch('typedef union { int i; float f; } Value;', (t) =>
|
||||
t.includes('@declaration.union'),
|
||||
);
|
||||
expect(m).toBeDefined();
|
||||
expect(m!['@declaration.name'].text).toBe('Value');
|
||||
});
|
||||
});
|
||||
|
||||
describe('emitCScopeCaptures — enum declarations', () => {
|
||||
it('captures enum with @declaration.enum', () => {
|
||||
const m = findMatch('enum Color { RED, GREEN, BLUE };', (t) => t.includes('@declaration.enum'));
|
||||
expect(m).toBeDefined();
|
||||
expect(m!['@declaration.name'].text).toBe('Color');
|
||||
});
|
||||
|
||||
it('captures enum constants as @declaration.const', () => {
|
||||
const matches = allMatches('enum Color { RED, GREEN, BLUE };', (t) =>
|
||||
t.includes('@declaration.const'),
|
||||
);
|
||||
const names = matches.map((m) => m['@declaration.name'].text);
|
||||
expect(names).toContain('RED');
|
||||
expect(names).toContain('GREEN');
|
||||
expect(names).toContain('BLUE');
|
||||
});
|
||||
});
|
||||
|
||||
describe('emitCScopeCaptures — function declarations', () => {
|
||||
it('captures function definition with @declaration.function', () => {
|
||||
const m = findMatch('int add(int a, int b) { return a + b; }', (t) =>
|
||||
t.includes('@declaration.function'),
|
||||
);
|
||||
expect(m).toBeDefined();
|
||||
expect(m!['@declaration.name'].text).toBe('add');
|
||||
});
|
||||
|
||||
it('captures function prototype (declaration) with @declaration.function', () => {
|
||||
const m = findMatch('int add(int a, int b);', (t) => t.includes('@declaration.function'));
|
||||
expect(m).toBeDefined();
|
||||
expect(m!['@declaration.name'].text).toBe('add');
|
||||
});
|
||||
|
||||
it('captures pointer-return function definition', () => {
|
||||
const m = findMatch('int *create(void) { return 0; }', (t) =>
|
||||
t.includes('@declaration.function'),
|
||||
);
|
||||
expect(m).toBeDefined();
|
||||
expect(m!['@declaration.name'].text).toBe('create');
|
||||
});
|
||||
|
||||
it('captures pointer-return function prototype', () => {
|
||||
const m = findMatch('char *get_name(void);', (t) => t.includes('@declaration.function'));
|
||||
expect(m).toBeDefined();
|
||||
expect(m!['@declaration.name'].text).toBe('get_name');
|
||||
});
|
||||
});
|
||||
|
||||
describe('emitCScopeCaptures — other declarations', () => {
|
||||
it('captures typedef as @declaration.typedef', () => {
|
||||
const m = findMatch('typedef int MyInt;', (t) => t.includes('@declaration.typedef'));
|
||||
expect(m).toBeDefined();
|
||||
expect(m!['@declaration.name'].text).toBe('MyInt');
|
||||
});
|
||||
|
||||
it('captures function pointer typedef as @declaration.typedef', () => {
|
||||
const m = findMatch('typedef void (*callback)(int, int);', (t) =>
|
||||
t.includes('@declaration.typedef'),
|
||||
);
|
||||
expect(m).toBeDefined();
|
||||
expect(m!['@declaration.name'].text).toBe('callback');
|
||||
});
|
||||
|
||||
it('captures struct field as @declaration.field', () => {
|
||||
const m = findMatch('struct P { int x; };', (t) => t.includes('@declaration.field'));
|
||||
expect(m).toBeDefined();
|
||||
expect(m!['@declaration.name'].text).toBe('x');
|
||||
});
|
||||
|
||||
it('captures pointer struct field as @declaration.field', () => {
|
||||
const m = findMatch('struct N { struct N *next; };', (t) => t.includes('@declaration.field'));
|
||||
expect(m).toBeDefined();
|
||||
expect(m!['@declaration.name'].text).toBe('next');
|
||||
});
|
||||
|
||||
it('captures variable with initializer as @declaration.variable', () => {
|
||||
const m = findMatch('int x = 42;', (t) => t.includes('@declaration.variable'));
|
||||
expect(m).toBeDefined();
|
||||
expect(m!['@declaration.name'].text).toBe('x');
|
||||
});
|
||||
|
||||
it('captures macro as @declaration.macro', () => {
|
||||
const m = findMatch('#define MAX 100', (t) => t.includes('@declaration.macro'));
|
||||
expect(m).toBeDefined();
|
||||
expect(m!['@declaration.name'].text).toBe('MAX');
|
||||
});
|
||||
|
||||
it('captures function-like macro as @declaration.macro', () => {
|
||||
const m = findMatch('#define SQUARE(x) ((x) * (x))', (t) => t.includes('@declaration.macro'));
|
||||
expect(m).toBeDefined();
|
||||
expect(m!['@declaration.name'].text).toBe('SQUARE');
|
||||
});
|
||||
});
|
||||
|
||||
describe('emitCScopeCaptures — imports', () => {
|
||||
it('captures local #include as @import.statement with source', () => {
|
||||
const m = findMatch('#include "header.h"', (t) => t.includes('@import.statement'));
|
||||
expect(m).toBeDefined();
|
||||
expect(m!['@import.source'].text).toBe('header.h');
|
||||
expect(m!['@import.kind'].text).toBe('wildcard');
|
||||
});
|
||||
|
||||
it('captures system #include with @import.system tag', () => {
|
||||
const m = findMatch('#include <stdio.h>', (t) => t.includes('@import.statement'));
|
||||
expect(m).toBeDefined();
|
||||
expect(m!['@import.system']).toBeDefined();
|
||||
});
|
||||
|
||||
it('captures nested path includes', () => {
|
||||
const m = findMatch('#include "utils/helpers.h"', (t) => t.includes('@import.statement'));
|
||||
expect(m).toBeDefined();
|
||||
expect(m!['@import.source'].text).toBe('utils/helpers.h');
|
||||
});
|
||||
});
|
||||
|
||||
describe('emitCScopeCaptures — references', () => {
|
||||
it('captures free call invocations', () => {
|
||||
const m = findMatch('void f(void) { foo(); }', (t) => t.includes('@reference.call.free'));
|
||||
expect(m).toBeDefined();
|
||||
expect(m!['@reference.name'].text).toBe('foo');
|
||||
});
|
||||
|
||||
it('captures member call via pointer (ptr->func())', () => {
|
||||
const m = findMatch('void f(struct S *s) { s->method(); }', (t) =>
|
||||
t.includes('@reference.call.member'),
|
||||
);
|
||||
expect(m).toBeDefined();
|
||||
expect(m!['@reference.name'].text).toBe('method');
|
||||
});
|
||||
|
||||
it('captures field reads', () => {
|
||||
const m = findMatch('void f(struct S *s) { int x = s->field; }', (t) =>
|
||||
t.includes('@reference.read'),
|
||||
);
|
||||
expect(m).toBeDefined();
|
||||
expect(m!['@reference.name'].text).toBe('field');
|
||||
});
|
||||
|
||||
it('captures field writes (assignment)', () => {
|
||||
const m = findMatch('void f(struct S *s) { s->field = 1; }', (t) =>
|
||||
t.includes('@reference.write'),
|
||||
);
|
||||
expect(m).toBeDefined();
|
||||
expect(m!['@reference.name'].text).toBe('field');
|
||||
});
|
||||
});
|
||||
|
||||
describe('emitCScopeCaptures — type bindings', () => {
|
||||
it('captures parameter type annotations', () => {
|
||||
const m = findMatch('void f(int x) { }', (t) => t.includes('@type-binding.parameter'));
|
||||
expect(m).toBeDefined();
|
||||
expect(m!['@type-binding.name'].text).toBe('x');
|
||||
});
|
||||
|
||||
it('captures variable type bindings', () => {
|
||||
const m = findMatch('void f(void) { int x = 1; }', (t) =>
|
||||
t.includes('@type-binding.assignment'),
|
||||
);
|
||||
expect(m).toBeDefined();
|
||||
expect(m!['@type-binding.name'].text).toBe('x');
|
||||
});
|
||||
});
|
||||
|
||||
describe('emitCScopeCaptures — arity metadata', () => {
|
||||
it('synthesizes parameter-count on function definitions', () => {
|
||||
const m = findMatch('int add(int a, int b) { return a + b; }', (t) =>
|
||||
t.includes('@declaration.parameter-count'),
|
||||
);
|
||||
expect(m).toBeDefined();
|
||||
expect(m!['@declaration.parameter-count'].text).toBe('2');
|
||||
});
|
||||
|
||||
it('synthesizes parameter-types on function definitions', () => {
|
||||
const m = findMatch('int add(int a, float b) { return 0; }', (t) =>
|
||||
t.includes('@declaration.parameter-types'),
|
||||
);
|
||||
expect(m).toBeDefined();
|
||||
const types = JSON.parse(m!['@declaration.parameter-types'].text);
|
||||
expect(types).toEqual(['int', 'float']);
|
||||
});
|
||||
|
||||
it('(void) parameter list yields zero parameters', () => {
|
||||
const m = findMatch('void f(void) { }', (t) => t.includes('@declaration.function'));
|
||||
expect(m).toBeDefined();
|
||||
expect(m!['@declaration.parameter-count'].text).toBe('0');
|
||||
expect(m!['@declaration.required-parameter-count'].text).toBe('0');
|
||||
});
|
||||
|
||||
it('variadic function has undefined parameter-count but defined required-parameter-count', () => {
|
||||
const m = findMatch('int printf(const char *fmt, ...) { return 0; }', (t) =>
|
||||
t.includes('@declaration.function'),
|
||||
);
|
||||
expect(m).toBeDefined();
|
||||
// variadic → parameterCount is undefined (not emitted)
|
||||
expect(m!['@declaration.parameter-count']).toBeUndefined();
|
||||
expect(m!['@declaration.required-parameter-count'].text).toBe('1');
|
||||
});
|
||||
|
||||
it('synthesizes arity on call references', () => {
|
||||
const m = findMatch(
|
||||
'void f(void) { add(1, 2); }',
|
||||
(t) => t.includes('@reference.call.free') && t.includes('@reference.arity'),
|
||||
);
|
||||
expect(m).toBeDefined();
|
||||
expect(m!['@reference.arity'].text).toBe('2');
|
||||
});
|
||||
|
||||
it('zero-argument call has arity 0', () => {
|
||||
const m = findMatch(
|
||||
'void f(void) { init(); }',
|
||||
(t) => t.includes('@reference.call.free') && t.includes('@reference.arity'),
|
||||
);
|
||||
expect(m).toBeDefined();
|
||||
expect(m!['@reference.arity'].text).toBe('0');
|
||||
});
|
||||
});
|
||||
|
||||
describe('emitCScopeCaptures — static storage class', () => {
|
||||
beforeEach(() => {
|
||||
clearStaticNames();
|
||||
});
|
||||
|
||||
it('marks static function definitions as file-local', () => {
|
||||
emitCScopeCaptures('static int helper(void) { return 0; }', 'a.c');
|
||||
expect(isStaticName('a.c', 'helper')).toBe(true);
|
||||
});
|
||||
|
||||
it('does not mark non-static function definitions as file-local', () => {
|
||||
emitCScopeCaptures('int helper(void) { return 0; }', 'a.c');
|
||||
expect(isStaticName('a.c', 'helper')).toBe(false);
|
||||
});
|
||||
|
||||
it('marks static function prototypes as file-local', () => {
|
||||
emitCScopeCaptures('static int helper(int x);', 'a.c');
|
||||
expect(isStaticName('a.c', 'helper')).toBe(true);
|
||||
});
|
||||
|
||||
it('static functions are scoped to their file', () => {
|
||||
emitCScopeCaptures('static int helper(void) { return 0; }', 'a.c');
|
||||
emitCScopeCaptures('int helper(void) { return 1; }', 'b.c');
|
||||
expect(isStaticName('a.c', 'helper')).toBe(true);
|
||||
expect(isStaticName('b.c', 'helper')).toBe(false);
|
||||
});
|
||||
|
||||
it('static pointer-return functions are detected', () => {
|
||||
emitCScopeCaptures('static char *get_buffer(void) { return 0; }', 'a.c');
|
||||
expect(isStaticName('a.c', 'get_buffer')).toBe(true);
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,103 @@
|
||||
/**
|
||||
* Unit tests for C header scanning — specifically the skip-list
|
||||
* for build output directories.
|
||||
*/
|
||||
|
||||
import { describe, it, expect, beforeEach, afterEach } from 'vitest';
|
||||
import { mkdirSync, writeFileSync, rmSync } from 'fs';
|
||||
import { join } from 'path';
|
||||
import { scanHeaderFiles } from '../../../../src/core/ingestion/languages/c/header-scan.js';
|
||||
|
||||
const TMP = join(__dirname, '__header_scan_tmp__');
|
||||
|
||||
function touch(rel: string): void {
|
||||
const full = join(TMP, rel);
|
||||
mkdirSync(join(full, '..'), { recursive: true });
|
||||
writeFileSync(full, '');
|
||||
}
|
||||
|
||||
beforeEach(() => {
|
||||
mkdirSync(TMP, { recursive: true });
|
||||
});
|
||||
|
||||
afterEach(() => {
|
||||
rmSync(TMP, { recursive: true, force: true });
|
||||
});
|
||||
|
||||
describe('scanHeaderFiles — build-directory skip list', () => {
|
||||
it('finds .h files in source directories', () => {
|
||||
touch('src/foo.h');
|
||||
touch('include/bar.h');
|
||||
const headers = scanHeaderFiles(TMP);
|
||||
expect(headers).toContain('src/foo.h');
|
||||
expect(headers).toContain('include/bar.h');
|
||||
});
|
||||
|
||||
it('skips node_modules', () => {
|
||||
touch('node_modules/dep/header.h');
|
||||
touch('src/real.h');
|
||||
const headers = scanHeaderFiles(TMP);
|
||||
expect(headers).not.toContain('node_modules/dep/header.h');
|
||||
expect(headers).toContain('src/real.h');
|
||||
});
|
||||
|
||||
it('skips .git directory', () => {
|
||||
touch('.git/refs/header.h');
|
||||
const headers = scanHeaderFiles(TMP);
|
||||
expect(headers.size).toBe(0);
|
||||
});
|
||||
|
||||
it('skips vendor directory', () => {
|
||||
touch('vendor/lib/header.h');
|
||||
const headers = scanHeaderFiles(TMP);
|
||||
expect(headers.size).toBe(0);
|
||||
});
|
||||
|
||||
it('skips dist directory', () => {
|
||||
touch('dist/generated.h');
|
||||
touch('src/real.h');
|
||||
const headers = scanHeaderFiles(TMP);
|
||||
expect(headers).not.toContain('dist/generated.h');
|
||||
expect(headers).toContain('src/real.h');
|
||||
});
|
||||
|
||||
it('skips build directory', () => {
|
||||
touch('build/config.h');
|
||||
const headers = scanHeaderFiles(TMP);
|
||||
expect(headers).not.toContain('build/config.h');
|
||||
});
|
||||
|
||||
it('skips out directory', () => {
|
||||
touch('out/gen/auto.h');
|
||||
const headers = scanHeaderFiles(TMP);
|
||||
expect(headers.size).toBe(0);
|
||||
});
|
||||
|
||||
it('skips target directory', () => {
|
||||
touch('target/release/bindings.h');
|
||||
const headers = scanHeaderFiles(TMP);
|
||||
expect(headers.size).toBe(0);
|
||||
});
|
||||
|
||||
it('skips _build directory', () => {
|
||||
touch('_build/default/lib.h');
|
||||
const headers = scanHeaderFiles(TMP);
|
||||
expect(headers.size).toBe(0);
|
||||
});
|
||||
|
||||
it('skips .next directory', () => {
|
||||
touch('.next/cache/header.h');
|
||||
const headers = scanHeaderFiles(TMP);
|
||||
expect(headers.size).toBe(0);
|
||||
});
|
||||
|
||||
it('skips cmake-build-* directories', () => {
|
||||
touch('cmake-build-debug/generated.h');
|
||||
touch('cmake-build-release/generated.h');
|
||||
touch('src/real.h');
|
||||
const headers = scanHeaderFiles(TMP);
|
||||
expect(headers).not.toContain('cmake-build-debug/generated.h');
|
||||
expect(headers).not.toContain('cmake-build-release/generated.h');
|
||||
expect(headers).toContain('src/real.h');
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,166 @@
|
||||
/**
|
||||
* Unit tests for C import decomposition, interpretation, and target resolution.
|
||||
*/
|
||||
|
||||
import { describe, it, expect } from 'vitest';
|
||||
import { getCParser } from '../../../../src/core/ingestion/languages/c/query.js';
|
||||
import { splitCInclude } from '../../../../src/core/ingestion/languages/c/import-decomposer.js';
|
||||
import { interpretCImport } from '../../../../src/core/ingestion/languages/c/interpret.js';
|
||||
import { resolveCImportTarget } from '../../../../src/core/ingestion/languages/c/import-target.js';
|
||||
import type { SyntaxNode } from '../../../../src/core/ingestion/utils/ast-helpers.js';
|
||||
|
||||
function parseIncludeNode(src: string): SyntaxNode | null {
|
||||
const tree = getCParser().parse(src);
|
||||
for (let i = 0; i < tree.rootNode.namedChildCount; i++) {
|
||||
const child = tree.rootNode.namedChild(i);
|
||||
if (child?.type === 'preproc_include') return child as SyntaxNode;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
function capt(name: string, text: string) {
|
||||
return { name, text, range: { startLine: 1, startCol: 1, endLine: 1, endCol: 1 } };
|
||||
}
|
||||
|
||||
describe('C import decomposition (splitCInclude)', () => {
|
||||
it('decomposes local include "#include \\"foo.h\\""', () => {
|
||||
const node = parseIncludeNode('#include "foo.h"');
|
||||
expect(node).not.toBeNull();
|
||||
const match = splitCInclude(node!);
|
||||
expect(match).not.toBeNull();
|
||||
expect(match!['@import.source'].text).toBe('foo.h');
|
||||
expect(match!['@import.kind'].text).toBe('wildcard');
|
||||
expect(match!['@import.system']).toBeUndefined();
|
||||
});
|
||||
|
||||
it('decomposes system include "#include <stdio.h>"', () => {
|
||||
const node = parseIncludeNode('#include <stdio.h>');
|
||||
expect(node).not.toBeNull();
|
||||
const match = splitCInclude(node!);
|
||||
expect(match).not.toBeNull();
|
||||
expect(match!['@import.source'].text).toBe('stdio.h');
|
||||
expect(match!['@import.system']).toBeDefined();
|
||||
});
|
||||
|
||||
it('decomposes nested path include', () => {
|
||||
const node = parseIncludeNode('#include "utils/helpers.h"');
|
||||
expect(node).not.toBeNull();
|
||||
const match = splitCInclude(node!);
|
||||
expect(match).not.toBeNull();
|
||||
expect(match!['@import.source'].text).toBe('utils/helpers.h');
|
||||
});
|
||||
});
|
||||
|
||||
describe('C import interpretation (interpretCImport)', () => {
|
||||
it('interprets local include as wildcard import', () => {
|
||||
const result = interpretCImport({
|
||||
'@import.kind': capt('@import.kind', 'wildcard'),
|
||||
'@import.source': capt('@import.source', 'header.h'),
|
||||
});
|
||||
expect(result).toEqual({ kind: 'wildcard', targetRaw: 'header.h' });
|
||||
});
|
||||
|
||||
it('returns null for system headers', () => {
|
||||
const result = interpretCImport({
|
||||
'@import.kind': capt('@import.kind', 'wildcard'),
|
||||
'@import.source': capt('@import.source', 'stdio.h'),
|
||||
'@import.system': capt('@import.system', 'true'),
|
||||
});
|
||||
expect(result).toBeNull();
|
||||
});
|
||||
|
||||
it('returns null when @import.source is missing', () => {
|
||||
const result = interpretCImport({
|
||||
'@import.kind': capt('@import.kind', 'wildcard'),
|
||||
});
|
||||
expect(result).toBeNull();
|
||||
});
|
||||
});
|
||||
|
||||
describe('C import target resolution (resolveCImportTarget)', () => {
|
||||
it('resolves exact match', () => {
|
||||
const result = resolveCImportTarget('foo.h', 'main.c', new Set(['foo.h', 'bar.h']));
|
||||
expect(result).toBe('foo.h');
|
||||
});
|
||||
|
||||
it('resolves suffix match to shortest path', () => {
|
||||
const result = resolveCImportTarget(
|
||||
'foo.h',
|
||||
'main.c',
|
||||
new Set(['src/include/foo.h', 'include/foo.h', 'other/bar.h']),
|
||||
);
|
||||
expect(result).toBe('include/foo.h');
|
||||
});
|
||||
|
||||
it('resolves nested include path with directory components', () => {
|
||||
const result = resolveCImportTarget(
|
||||
'utils/helpers.h',
|
||||
'main.c',
|
||||
new Set(['src/utils/helpers.h', 'lib/utils/helpers.h']),
|
||||
);
|
||||
// Both have same depth (3 components), so lexicographic tiebreak picks lib/
|
||||
expect(result).toBe('lib/utils/helpers.h');
|
||||
});
|
||||
|
||||
it('returns null for empty target', () => {
|
||||
expect(resolveCImportTarget('', 'main.c', new Set(['foo.h']))).toBeNull();
|
||||
});
|
||||
|
||||
it('returns null when no match found', () => {
|
||||
expect(resolveCImportTarget('missing.h', 'main.c', new Set(['foo.h']))).toBeNull();
|
||||
});
|
||||
|
||||
it('is deterministic on depth ties — lexicographic tiebreak', () => {
|
||||
const files = new Set(['test/util/foo.h', 'src/util/foo.h']);
|
||||
const result1 = resolveCImportTarget('foo.h', 'main.c', files);
|
||||
const result2 = resolveCImportTarget('foo.h', 'main.c', files);
|
||||
expect(result1).toBe(result2);
|
||||
// Lexicographic: src/util/foo.h < test/util/foo.h
|
||||
expect(result1).toBe('src/util/foo.h');
|
||||
});
|
||||
|
||||
it('prefers shallower path over lexicographic order', () => {
|
||||
const result = resolveCImportTarget('foo.h', 'main.c', new Set(['a/b/c/foo.h', 'z/foo.h']));
|
||||
expect(result).toBe('z/foo.h');
|
||||
});
|
||||
|
||||
it('handles backslash paths (Windows)', () => {
|
||||
const result = resolveCImportTarget('foo.h', 'main.c', new Set(['include\\foo.h']));
|
||||
expect(result).toBe('include\\foo.h');
|
||||
});
|
||||
|
||||
it('prefers same-directory sibling over deeper suffix match', () => {
|
||||
// src/foo.c includes "bar.h" — src/bar.h should win over include/bar.h
|
||||
const result = resolveCImportTarget(
|
||||
'bar.h',
|
||||
'src/foo.c',
|
||||
new Set(['include/bar.h', 'src/bar.h']),
|
||||
);
|
||||
expect(result).toBe('src/bar.h');
|
||||
});
|
||||
|
||||
it('prefers same-directory sibling over shallower suffix match', () => {
|
||||
// deep/nested/main.c includes "foo.h" — deep/nested/foo.h wins over foo.h
|
||||
const result = resolveCImportTarget(
|
||||
'foo.h',
|
||||
'deep/nested/main.c',
|
||||
new Set(['foo.h', 'deep/nested/foo.h']),
|
||||
);
|
||||
expect(result).toBe('deep/nested/foo.h');
|
||||
});
|
||||
|
||||
it('falls back to suffix match when no same-directory sibling exists', () => {
|
||||
const result = resolveCImportTarget('missing.h', 'src/foo.c', new Set(['lib/missing.h']));
|
||||
expect(result).toBe('lib/missing.h');
|
||||
});
|
||||
|
||||
it('same-directory sibling with nested target path', () => {
|
||||
// src/foo.c includes "sub/bar.h" — src/sub/bar.h should win
|
||||
const result = resolveCImportTarget(
|
||||
'sub/bar.h',
|
||||
'src/foo.c',
|
||||
new Set(['other/sub/bar.h', 'src/sub/bar.h']),
|
||||
);
|
||||
expect(result).toBe('src/sub/bar.h');
|
||||
});
|
||||
});
|
||||
@@ -21,85 +21,112 @@ import {
|
||||
} from '../../src/core/group/cross-impact.js';
|
||||
|
||||
/**
|
||||
* Time a single regex.exec call. Used by the linearity tests below to
|
||||
* compute a 10k/5k ratio in addition to the absolute <500ms bound.
|
||||
* Linearity-test methodology
|
||||
* --------------------------
|
||||
* Wall-clock perf assertions in CI are notoriously flaky. To make these
|
||||
* robust without losing regression-detection power, we combine four
|
||||
* techniques:
|
||||
*
|
||||
* Ratio assertions catch sub-exponential O(n²) regressions that fit
|
||||
* inside the absolute cap on warm CI; the absolute cap catches
|
||||
* catastrophic backtracking on cold CI. Two complementary signals.
|
||||
*/
|
||||
function timeRegex(re: RegExp, input: string): number {
|
||||
// Reset regex.lastIndex for global/sticky regexes — ours are not, but
|
||||
// be defensive in case future shape changes add the `g` flag.
|
||||
re.lastIndex = 0;
|
||||
const start = performance.now();
|
||||
re.exec(input);
|
||||
return performance.now() - start;
|
||||
}
|
||||
|
||||
function timeFn<T>(fn: () => T): number {
|
||||
const start = performance.now();
|
||||
fn();
|
||||
return performance.now() - start;
|
||||
}
|
||||
|
||||
// Linear scaling is ~2.0× when input doubles; 3.0× allows generous
|
||||
// slack for CI-runner GC and tier-up jitter. An O(n²) regression on a
|
||||
// 2× input takes ~4× as long, well outside this bound.
|
||||
const LINEAR_RATIO_BOUND = 3.0;
|
||||
|
||||
/**
|
||||
* Minimum elapsed time (in ms) below which `performance.now()` ratios
|
||||
* are dominated by scheduler jitter and become meaningless. When both
|
||||
* timed runs come in below this floor, we skip the ratio assertion —
|
||||
* the absolute <500ms bound still catches catastrophic backtracking,
|
||||
* and the next CI run will measure higher absolute times that the
|
||||
* ratio assertion can evaluate reliably.
|
||||
* 1. **Warmup** — run the function a few times before timing, so the
|
||||
* JIT has tiered up by the time we measure.
|
||||
* 2. **Median of N trials** — single measurements are dominated by
|
||||
* GC pauses, scheduler jitter, and OS interrupts. Median of 5
|
||||
* eliminates almost all of that.
|
||||
* 3. **4× input ratio** (not 2×) — linear → ~4×, O(n²) → ~16×,
|
||||
* catastrophic → ≫16×. A wider input ratio gives a much bigger
|
||||
* gap between "linear" and "regressed", so the bound can be loose
|
||||
* enough to absorb noise without losing signal.
|
||||
* 4. **Generous bound (8×)** with a noise floor — only assert the
|
||||
* ratio when the *large* measurement is well above the noise
|
||||
* floor. The absolute <500ms cap still catches catastrophic
|
||||
* backtracking on cold CI even when the ratio is skipped.
|
||||
*
|
||||
* Calibrated empirically: a flake on macOS reported ratio 5.29×
|
||||
* between two sub-millisecond measurements (~0.5ms vs ~2.6ms), both
|
||||
* genuinely linear but indistinguishable from noise. 5ms is a
|
||||
* comfortable floor where individual measurements are well-separated
|
||||
* from the ~10-100µs `performance.now()` resolution band.
|
||||
* Headroom: linear is expected at ~4×; the bound is 8× → 2× headroom.
|
||||
* O(n²) on a 4× input would clock 16×, well outside the bound.
|
||||
*/
|
||||
const PERF_WARMUP_RUNS = 3;
|
||||
const PERF_TRIAL_COUNT = 5;
|
||||
const SIZE_RATIO = 4;
|
||||
const LINEAR_RATIO_BOUND = SIZE_RATIO * 2; // 8× — 2× headroom over expected linear
|
||||
// Median-of-N tightens the noise floor we can rely on. A single-sample 5ms
|
||||
// measurement is ~50% jitter; median-of-5 brings the same 5ms into the
|
||||
// reliably-resolvable range above `performance.now()`'s ~10-100µs band.
|
||||
const RATIO_MEASUREMENT_FLOOR_MS = 5;
|
||||
|
||||
function median(samples: number[]): number {
|
||||
const sorted = [...samples].sort((a, b) => a - b);
|
||||
const mid = Math.floor(sorted.length / 2);
|
||||
return sorted.length % 2 === 0 ? (sorted[mid - 1] + sorted[mid]) / 2 : sorted[mid];
|
||||
}
|
||||
|
||||
/**
|
||||
* Assert linear scaling between two timed runs on inputs that differ
|
||||
* by 2×. When measurements are too small to be reliable, the ratio
|
||||
* assertion is skipped (the absolute bound still fires elsewhere).
|
||||
* Median time of `PERF_TRIAL_COUNT` runs of `fn`, after `PERF_WARMUP_RUNS`
|
||||
* warmup iterations. Trial cost: (warmup + trials) × fn cost.
|
||||
*/
|
||||
function assertSubLinearRatio(elapsedSmall: number, elapsedLarge: number, label: string): void {
|
||||
function medianTimeFn<T>(fn: () => T): number {
|
||||
for (let i = 0; i < PERF_WARMUP_RUNS; i++) fn();
|
||||
const samples: number[] = [];
|
||||
for (let i = 0; i < PERF_TRIAL_COUNT; i++) {
|
||||
const start = performance.now();
|
||||
fn();
|
||||
samples.push(performance.now() - start);
|
||||
}
|
||||
return median(samples);
|
||||
}
|
||||
|
||||
/** Median time of regex.exec — defensively resets lastIndex each call. */
|
||||
function medianTimeRegex(re: RegExp, input: string): number {
|
||||
return medianTimeFn(() => {
|
||||
re.lastIndex = 0;
|
||||
re.exec(input);
|
||||
});
|
||||
}
|
||||
|
||||
/**
|
||||
* Assert near-linear scaling between two median-timed runs on inputs
|
||||
* that differ by `SIZE_RATIO`×. The bound is `LINEAR_RATIO_BOUND` =
|
||||
* `SIZE_RATIO * 2`, i.e. 2× headroom over the linear expectation —
|
||||
* comfortably under the ~`SIZE_RATIO²` ratio a quadratic regression
|
||||
* would produce, so true regressions still fail loudly.
|
||||
*
|
||||
* Skip semantics: the ratio assertion is skipped only when *both*
|
||||
* measurements are below the noise floor. If either run is reliably
|
||||
* measurable, we still assert — otherwise an O(n²) regression that
|
||||
* happens to stay under the absolute 500ms cap on a fast runner could
|
||||
* slip through with no detector firing. Median-of-N + the 5ms floor
|
||||
* keeps the assertion stable while preserving regression coverage.
|
||||
*/
|
||||
function assertNearLinearScaling(elapsedSmall: number, elapsedLarge: number, label: string): void {
|
||||
if (elapsedSmall < RATIO_MEASUREMENT_FLOOR_MS && elapsedLarge < RATIO_MEASUREMENT_FLOOR_MS) {
|
||||
// Both runs completed faster than the noise floor — the ratio is
|
||||
// not meaningful. The absolute <500ms bound elsewhere in this
|
||||
// describe block still pins linearity; we skip rather than risk a
|
||||
// flake on a genuinely-linear implementation.
|
||||
// Both runs completed below the noise floor — even the median is
|
||||
// dominated by `performance.now()` resolution. The absolute <500ms
|
||||
// cap elsewhere still catches catastrophic backtracking.
|
||||
return;
|
||||
}
|
||||
const ratio = elapsedLarge / Math.max(elapsedSmall, 0.001);
|
||||
if (ratio >= LINEAR_RATIO_BOUND) {
|
||||
throw new Error(
|
||||
`${label}: ratio ${ratio.toFixed(2)}× exceeds bound ${LINEAR_RATIO_BOUND}× ` +
|
||||
`(small=${elapsedSmall.toFixed(2)}ms, large=${elapsedLarge.toFixed(2)}ms)`,
|
||||
`on ${SIZE_RATIO}× input (small=${elapsedSmall.toFixed(2)}ms, ` +
|
||||
`large=${elapsedLarge.toFixed(2)}ms, median of ${PERF_TRIAL_COUNT} trials)`,
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
describe('cobol-preprocessor RE_SET_TO_TRUE — linear time on pathological input', () => {
|
||||
it('matches in <500ms on 50k repetitions of "A OF A " AND 100k/50k ratio is sub-linear when measurable', () => {
|
||||
// 50k/100k repetitions chosen so timings exceed the
|
||||
// RATIO_MEASUREMENT_FLOOR_MS noise floor on typical CI hardware.
|
||||
// Pre-fix nested-quantifier shape would be exponential here; the
|
||||
// post-fix `.+?` shape is linear (~2× when input doubles).
|
||||
it('matches in <500ms on 50k repetitions of "A OF A " AND scales sub-linearly on a 4× input', () => {
|
||||
// 50k → 200k (4× input ratio). Pre-fix nested-quantifier shape would
|
||||
// be exponential here; the post-fix `.+?` shape is linear (~4× when
|
||||
// input quadruples). Median of 5 trials with warmup eliminates GC
|
||||
// and tier-up jitter.
|
||||
const inputSmall = 'SET ' + 'A OF A '.repeat(50_000) + 'TO TRUE';
|
||||
const inputLarge = 'SET ' + 'A OF A '.repeat(100_000) + 'TO TRUE';
|
||||
const elapsedSmall = timeRegex(RE_SET_TO_TRUE, inputSmall);
|
||||
const elapsedLarge = timeRegex(RE_SET_TO_TRUE, inputLarge);
|
||||
const inputLarge = 'SET ' + 'A OF A '.repeat(50_000 * SIZE_RATIO) + 'TO TRUE';
|
||||
const elapsedSmall = medianTimeRegex(RE_SET_TO_TRUE, inputSmall);
|
||||
const elapsedLarge = medianTimeRegex(RE_SET_TO_TRUE, inputLarge);
|
||||
expect(RE_SET_TO_TRUE.exec(inputSmall)).not.toBeNull();
|
||||
expect(elapsedSmall).toBeLessThan(500);
|
||||
expect(elapsedLarge).toBeLessThan(500);
|
||||
assertSubLinearRatio(elapsedSmall, elapsedLarge, 'RE_SET_TO_TRUE');
|
||||
assertNearLinearScaling(elapsedSmall, elapsedLarge, 'RE_SET_TO_TRUE');
|
||||
});
|
||||
|
||||
it('still matches a normal SET ... TO TRUE statement', () => {
|
||||
@@ -110,17 +137,17 @@ describe('cobol-preprocessor RE_SET_TO_TRUE — linear time on pathological inpu
|
||||
});
|
||||
|
||||
describe('cobol-preprocessor RE_SET_INDEX — linear time on pathological input', () => {
|
||||
it('rejects in <500ms on 50k tokens with no valid suffix AND 100k/50k ratio is sub-linear when measurable', () => {
|
||||
it('rejects in <500ms on 50k tokens with no valid suffix AND scales sub-linearly on a 4× input', () => {
|
||||
// Forces backtracking against the (TO|UP\s+BY|DOWN\s+BY) alternation
|
||||
// — the richer pathological surface of the two regexes.
|
||||
const inputSmall = 'SET ' + 'A '.repeat(50_000) + 'X';
|
||||
const inputLarge = 'SET ' + 'A '.repeat(100_000) + 'X';
|
||||
const elapsedSmall = timeRegex(RE_SET_INDEX, inputSmall);
|
||||
const elapsedLarge = timeRegex(RE_SET_INDEX, inputLarge);
|
||||
const inputLarge = 'SET ' + 'A '.repeat(50_000 * SIZE_RATIO) + 'X';
|
||||
const elapsedSmall = medianTimeRegex(RE_SET_INDEX, inputSmall);
|
||||
const elapsedLarge = medianTimeRegex(RE_SET_INDEX, inputLarge);
|
||||
expect(RE_SET_INDEX.exec(inputSmall)).toBeNull();
|
||||
expect(elapsedSmall).toBeLessThan(500);
|
||||
expect(elapsedLarge).toBeLessThan(500);
|
||||
assertSubLinearRatio(elapsedSmall, elapsedLarge, 'RE_SET_INDEX');
|
||||
assertNearLinearScaling(elapsedSmall, elapsedLarge, 'RE_SET_INDEX');
|
||||
});
|
||||
|
||||
it('still matches a normal SET INDEX statement', () => {
|
||||
@@ -133,22 +160,22 @@ describe('cobol-preprocessor RE_SET_INDEX — linear time on pathological input'
|
||||
});
|
||||
|
||||
describe('rust-workspace parseCargoPackageName — linear-time line walk', () => {
|
||||
it('extracts the package name in <500ms on 100k blank lines AND 200k/100k ratio is sub-linear when measurable', () => {
|
||||
// 100k/200k blank lines chosen so timings exceed the
|
||||
// RATIO_MEASUREMENT_FLOOR_MS noise floor. Earlier 10k/20k pairing
|
||||
// produced sub-millisecond measurements where scheduler jitter
|
||||
// dominated and the ratio became meaningless (a real macOS run
|
||||
// saw 5.29× between two genuinely-linear sub-ms measurements).
|
||||
it('extracts the package name in <500ms on 100k blank lines AND scales sub-linearly on a 4× input', () => {
|
||||
// 100k → 400k blank lines (4× input ratio). Median of 5 trials with
|
||||
// warmup keeps the ratio stable across CI runners. A previous 2×
|
||||
// input + 3× bound + single-trial setup flaked at 3.01× on macOS
|
||||
// (small=7.41ms, large=22.31ms) — both above the noise floor but
|
||||
// close enough that single-shot jitter pushed the ratio over.
|
||||
const cargoTomlSmall =
|
||||
'[package]\n' + '\n'.repeat(100_000) + 'name = "myrepo"\nversion = "0.1.0"\n';
|
||||
const cargoTomlLarge =
|
||||
'[package]\n' + '\n'.repeat(200_000) + 'name = "myrepo"\nversion = "0.1.0"\n';
|
||||
const elapsedSmall = timeFn(() => parseCargoPackageName(cargoTomlSmall));
|
||||
const elapsedLarge = timeFn(() => parseCargoPackageName(cargoTomlLarge));
|
||||
'[package]\n' + '\n'.repeat(100_000 * SIZE_RATIO) + 'name = "myrepo"\nversion = "0.1.0"\n';
|
||||
const elapsedSmall = medianTimeFn(() => parseCargoPackageName(cargoTomlSmall));
|
||||
const elapsedLarge = medianTimeFn(() => parseCargoPackageName(cargoTomlLarge));
|
||||
expect(parseCargoPackageName(cargoTomlSmall)).toBe('myrepo');
|
||||
expect(elapsedSmall).toBeLessThan(500);
|
||||
expect(elapsedLarge).toBeLessThan(500);
|
||||
assertSubLinearRatio(elapsedSmall, elapsedLarge, 'parseCargoPackageName');
|
||||
assertNearLinearScaling(elapsedSmall, elapsedLarge, 'parseCargoPackageName');
|
||||
});
|
||||
|
||||
it('returns null when [package] section is absent', () => {
|
||||
|
||||
Reference in New Issue
Block a user