Compare commits
91
Commits
release/v1.5.0
...
pr-823
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
3768e3bfd0 | ||
|
|
3384575ac6 | ||
|
|
dd194d56b1 | ||
|
|
80a6fde2ba | ||
|
|
8d38cc99fa | ||
|
|
41844edf88 | ||
|
|
9ad1984b17 | ||
|
|
1a597f3cc6 | ||
|
|
759c983dce | ||
|
|
988b905abe | ||
|
|
3fbee2d3d2 | ||
|
|
26ff700e37 | ||
|
|
6388113e10 | ||
|
|
c672697012 | ||
|
|
d786e692af | ||
|
|
a6421b3b1b | ||
|
|
9f4109a33f | ||
|
|
79e1d933fa | ||
|
|
4d4756fe86 | ||
|
|
b10d25bbca | ||
|
|
a94d6ef80b | ||
|
|
1ff324ca16 | ||
|
|
9364739fb4 | ||
|
|
75635638b1 | ||
|
|
5be0537ce4 | ||
|
|
a162f66254 | ||
|
|
6d9ec1009e | ||
|
|
4911201664 | ||
|
|
08541e2857 | ||
|
|
d9960c62bf | ||
|
|
7c983d798f | ||
|
|
d87744fffc | ||
|
|
100858f8c8 | ||
|
|
e5dafce9f2 | ||
|
|
ab956f113c | ||
|
|
ad2a397137 | ||
|
|
6147579e54 | ||
|
|
4a1f912aee | ||
|
|
d09078925e | ||
|
|
338cb01ee0 | ||
|
|
4450a14b98 | ||
|
|
3f0b8c1a5b | ||
|
|
bb68cc1eb0 | ||
|
|
d6debf3324 | ||
|
|
9ab92a97d0 | ||
|
|
4fde5f241b | ||
|
|
9f9bbcd744 | ||
|
|
fd67cfd5a7 | ||
|
|
3f28f7ead5 | ||
|
|
d9ba9aa998 | ||
|
|
c19e76a4a3 | ||
|
|
b75e76d44a | ||
|
|
83b5bec293 | ||
|
|
3388ae16d7 | ||
|
|
d784f591b2 | ||
|
|
0f43190543 | ||
|
|
fe87ff8f74 | ||
|
|
be2401061e | ||
|
|
b73233d232 | ||
|
|
1c8ae5eb46 | ||
|
|
b73928f732 | ||
|
|
19faf3b326 | ||
|
|
cb772b9e29 | ||
|
|
10f8815639 | ||
|
|
6ead5e5986 | ||
|
|
14791ded4a | ||
|
|
9eeb20bb04 | ||
|
|
5a7c0fdbb1 | ||
|
|
0561d24efd | ||
|
|
153262304c | ||
|
|
16cf4c503e | ||
|
|
57951a197b | ||
|
|
63fc4c795f | ||
|
|
5c4fca21c3 | ||
|
|
dd0f5eed7d | ||
|
|
e3d73a7aed | ||
|
|
255e3e79eb | ||
|
|
be4458b650 | ||
|
|
4fed097abb | ||
|
|
4fa395f4b6 | ||
|
|
52277247fe | ||
|
|
ba5de0bde4 | ||
|
|
dc86ea96dc | ||
|
|
d1adc8331a | ||
|
|
80d363f145 | ||
|
|
12be2025f1 | ||
|
|
dcf40da3a8 | ||
|
|
c617350c58 | ||
|
|
71255a096c | ||
|
|
39322e37df | ||
|
|
0e6f8f1720 |
@@ -30,6 +30,10 @@ jobs:
|
||||
registry-url: https://registry.npmjs.org
|
||||
cache: npm
|
||||
cache-dependency-path: gitnexus/package-lock.json
|
||||
- name: Build gitnexus-shared
|
||||
run: npm install && npm run build
|
||||
working-directory: gitnexus-shared
|
||||
|
||||
- run: npm ci
|
||||
working-directory: gitnexus
|
||||
|
||||
@@ -63,7 +67,22 @@ jobs:
|
||||
env:
|
||||
NODE_AUTH_TOKEN: ${{ secrets.NPM_TOKEN }}
|
||||
|
||||
- name: Extract release notes from CHANGELOG
|
||||
id: changelog
|
||||
shell: bash
|
||||
run: |
|
||||
VERSION="${GITHUB_REF#refs/tags/v}"
|
||||
NOTES=$(awk "/^## \\[$VERSION\\]/{found=1; next} /^## \\[/{if(found) exit} found" gitnexus/CHANGELOG.md)
|
||||
if [ -z "$NOTES" ]; then
|
||||
echo "::warning::No CHANGELOG entry found for v$VERSION, falling back to auto-generated notes"
|
||||
echo "fallback=true" >> "$GITHUB_OUTPUT"
|
||||
else
|
||||
echo "$NOTES" > /tmp/release-notes.md
|
||||
echo "fallback=false" >> "$GITHUB_OUTPUT"
|
||||
fi
|
||||
|
||||
- name: Create GitHub Release
|
||||
uses: softprops/action-gh-release@a06a81a03ee405af7f2048a818ed3f03bbf83c7b # v2
|
||||
with:
|
||||
generate_release_notes: true
|
||||
body_path: ${{ steps.changelog.outputs.fallback == 'false' && '/tmp/release-notes.md' || '' }}
|
||||
generate_release_notes: ${{ steps.changelog.outputs.fallback == 'true' }}
|
||||
|
||||
@@ -1,10 +1,10 @@
|
||||
<!-- version: 1.2.0 -->
|
||||
<!-- version: 1.3.0 -->
|
||||
<!--
|
||||
Metadata: version, last reviewed, scope, model policy, reference docs, changelog.
|
||||
Last updated: 2026-03-22
|
||||
-->
|
||||
|
||||
Last reviewed: 2026-03-24
|
||||
Last reviewed: 2026-04-13
|
||||
|
||||
**Project:** GitNexus · **Environment:** dev · **Maintainer:** repository maintainers (see GitHub)
|
||||
|
||||
@@ -54,6 +54,7 @@ Generic “core standards” playbooks are often long and stack-specific. For th
|
||||
|
||||
| Date | Version | Change |
|
||||
|------|---------|--------|
|
||||
| 2026-04-13 | 1.3.0 | Updated GitNexus index stats after DAG refactor. |
|
||||
| 2026-03-24 | 1.2.0 | Fixed gitnexus:start block duplication (was inlined in Reference Docs bullet). |
|
||||
| 2026-03-23 | 1.1.0 | Updated agent instructions (sections, references, Cursor layout). |
|
||||
| 2026-03-22 | 1.0.0 | Added structured agent header and changelog. |
|
||||
@@ -63,7 +64,7 @@ Generic “core standards” playbooks are often long and stack-specific. For th
|
||||
<!-- gitnexus:start -->
|
||||
# GitNexus — Code Intelligence
|
||||
|
||||
This project is indexed by GitNexus as **GitNexus**. Use the GitNexus MCP tools to understand code, assess impact, and navigate safely. For current symbol stats, run `npx gitnexus analyze` and inspect `.gitnexus/meta.json`.
|
||||
This project is indexed by GitNexus as **GitNexus** (4325 symbols, 10556 relationships, 300 execution flows). Use the GitNexus MCP tools to understand code, assess impact, and navigate safely.
|
||||
|
||||
> If any GitNexus tool warns the index is stale, run `npx gitnexus analyze` in terminal first.
|
||||
|
||||
|
||||
+117
-2
@@ -16,7 +16,7 @@ This repository is a **monorepo** with two main products: the **CLI / MCP packag
|
||||
|
||||
1. **Ingestion** (`gitnexus analyze`)
|
||||
- Entry: `gitnexus/src/cli/analyze.ts` → `runPipelineFromRepo` in `gitnexus/src/core/ingestion/pipeline.ts`.
|
||||
- Walks the git working tree, parses supported languages via **Tree-sitter**, resolves imports/calls/inheritance, detects **communities** and **processes** (execution flows), and builds an in-memory **knowledge graph** (`gitnexus/src/core/graph/`).
|
||||
- The pipeline is structured as a **DAG (Directed Acyclic Graph)** of named phases (see [Pipeline Phase DAG](#pipeline-phase-dag) below).
|
||||
- Output is loaded into **LadybugDB** under **`.gitnexus/`** at the repo root (`lbug/`, `meta.json`, etc.). Optional **FTS** indexes and **embeddings** attach to the same store.
|
||||
- The repo is registered in **`~/.gitnexus/registry.json`** so MCP can find it from any working directory.
|
||||
|
||||
@@ -49,7 +49,7 @@ This repository is a **monorepo** with two main products: the **CLI / MCP packag
|
||||
| If you are changing… | Start in… |
|
||||
|----------------------|-----------|
|
||||
| CLI commands / flags | `gitnexus/src/cli/` (`index.ts`, per-command modules). |
|
||||
| Parsing or graph construction | `gitnexus/src/core/ingestion/` (pipeline, processors, resolvers, type-extractors). |
|
||||
| Parsing or graph construction | `gitnexus/src/core/ingestion/pipeline-phases/` (individual phase files), `pipeline.ts` (orchestrator). |
|
||||
| Graph schema / DB access | `gitnexus/src/core/lbug/` (`schema.ts`, `lbug-adapter.ts`), `gitnexus/src/mcp/core/lbug-adapter.ts` if MCP-specific. |
|
||||
| MCP protocol, tools, resources | `gitnexus/src/mcp/server.ts`, `tools.ts`, `resources.ts`. |
|
||||
| Search ranking | `gitnexus/src/core/search/` (BM25, hybrid fusion). |
|
||||
@@ -58,8 +58,123 @@ This repository is a **monorepo** with two main products: the **CLI / MCP packag
|
||||
| Web UI behavior | `gitnexus-web/src/` (components, workers, graph client). |
|
||||
| CI | `.github/workflows/*.yml`, `.github/actions/setup-gitnexus/`. |
|
||||
|
||||
## Pipeline Phase DAG
|
||||
|
||||
The ingestion pipeline is a DAG of named phases. Each phase is defined in its own file under `gitnexus/src/core/ingestion/pipeline-phases/` with explicit dependencies, typed inputs, and typed outputs.
|
||||
|
||||
```
|
||||
scan → structure → [markdown, cobol] → parse → [routes, tools, orm]
|
||||
→ crossFile → mro → communities → processes
|
||||
```
|
||||
|
||||
### Phase files
|
||||
|
||||
| Phase | File | Dependencies | What it does |
|
||||
|-------|------|-------------|--------------|
|
||||
| `scan` | `scan.ts` | (root) | Walk repo filesystem, collect paths + sizes |
|
||||
| `structure` | `structure.ts` | `scan` | Build File/Folder nodes + CONTAINS edges |
|
||||
| `markdown` | `markdown.ts` | `structure` | Extract headings and cross-links from .md/.mdx |
|
||||
| `cobol` | `cobol.ts` | `structure` | Regex-based COBOL/JCL extraction |
|
||||
| `parse` | `parse.ts` + `parse-impl.ts` | `structure`, `markdown`, `cobol` | Chunked tree-sitter parse, import/call/heritage resolution |
|
||||
| `routes` | `routes.ts` | `parse` | Route registry (Next.js, Expo, PHP, decorator-based) |
|
||||
| `tools` | `tools.ts` | `parse` | MCP/RPC tool detection |
|
||||
| `orm` | `orm.ts` | `parse` | Prisma/Supabase ORM query edges |
|
||||
| `crossFile` | `cross-file.ts` + `cross-file-impl.ts` | `parse`, `routes`, `tools`, `orm` | Cross-file type propagation in topological order |
|
||||
| `mro` | `mro.ts` | `crossFile` | Method Resolution Order, METHOD_OVERRIDES edges |
|
||||
| `communities` | `communities.ts` | `mro` | Leiden community detection |
|
||||
| `processes` | `processes.ts` | `communities`, `routes`, `tools` | Execution flow detection, Route/Tool → Process links |
|
||||
|
||||
### How to add a new phase
|
||||
|
||||
1. Create a new file in `pipeline-phases/` (e.g. `my-phase.ts`)
|
||||
2. Define a `PipelinePhase<MyOutput>` object with `name`, `deps`, and `execute(ctx, deps)`
|
||||
3. Export it from `pipeline-phases/index.ts`
|
||||
4. Add it to the `buildPhaseList()` function in `pipeline.ts`
|
||||
|
||||
```typescript
|
||||
// pipeline-phases/my-phase.ts
|
||||
import type { PipelinePhase, PipelineContext, PhaseResult } from './types.js';
|
||||
import { getPhaseOutput } from './types.js';
|
||||
import type { ParseOutput } from './parse.js';
|
||||
|
||||
export interface MyPhaseOutput { /* ... */ }
|
||||
|
||||
export const myPhase: PipelinePhase<MyPhaseOutput> = {
|
||||
name: 'myPhase',
|
||||
deps: ['parse'], // runs after parse completes
|
||||
async execute(ctx, deps) {
|
||||
const { allPaths } = getPhaseOutput<ParseOutput>(deps, 'parse');
|
||||
// ... do work, write to ctx.graph ...
|
||||
return { /* typed output */ };
|
||||
},
|
||||
};
|
||||
```
|
||||
|
||||
### DAG runner
|
||||
|
||||
The runner (`pipeline-phases/runner.ts`) validates the DAG at startup (detects cycles and missing deps via topological sort), then executes phases in dependency order. Each phase receives:
|
||||
- `ctx: PipelineContext` — shared graph, repoPath, progress callback
|
||||
- `deps: Map<string, PhaseResult>` — outputs from all upstream phases
|
||||
|
||||
## Known limitations
|
||||
|
||||
### Overloaded method resolution
|
||||
|
||||
Method and Constructor node IDs include an arity suffix (`#<paramCount>`) to
|
||||
disambiguate overloaded methods. Two overloads with different parameter counts
|
||||
produce distinct graph nodes: `Method:file:Class.method#1` vs
|
||||
`Method:file:Class.method#2`.
|
||||
|
||||
**Same-arity overload disambiguation:** When two overloads share the same
|
||||
parameter count but differ in types (e.g. `save(int)` vs `save(String)`), a
|
||||
type-hash suffix `~type1,type2` is appended to produce distinct node IDs:
|
||||
`Method:file:Class.save#1~int` vs `Method:file:Class.save#1~String`. The suffix
|
||||
is only added when a same-arity collision is detected within a class and all
|
||||
parameters have non-null type annotations. Languages without type info (Python,
|
||||
Ruby, JS) fall back to arity-only IDs. TypeScript/JavaScript overload signatures
|
||||
are intentionally excluded from type-hashing because they are declaration-only
|
||||
contracts that should collapse to the implementation body's node ID. See issue
|
||||
\#651.
|
||||
|
||||
**C++ const-qualified overload disambiguation:** Methods overloaded by const
|
||||
qualification (e.g. `begin()` vs `begin() const`) are disambiguated via an
|
||||
`isConst` property and a `$const` ID suffix appended to the const-qualified
|
||||
variant when a non-const collision exists. The `$const` suffix appears after the
|
||||
type-hash suffix: e.g. `Method:file:Container.begin#0$const`.
|
||||
|
||||
**Generic/template type preservation in type-hash:** The type-hash suffix uses
|
||||
`rawType` (full AST text including generic/template args) rather than the
|
||||
simplified `type` from `extractSimpleTypeName`. This means C++ template overloads
|
||||
like `process(vector<int>)` vs `process(vector<string>)` produce distinct IDs:
|
||||
`~vector<int>` vs `~vector<std::string>`. Java generic overloads like
|
||||
`process(List<String>)` vs `process(List<Integer>)` are a compile error due to
|
||||
type erasure, so this gap is theoretical for Java.
|
||||
|
||||
**ID stability on first overload:** Type and const tags are collision-only. When
|
||||
a class has `save(int)` as its only `save` method, the ID is `save#1` (no tag).
|
||||
Adding `save(String)` changes the original to `save#1~int`. This is correct for
|
||||
fresh analysis but means IDs are not stable across overload additions. Future
|
||||
incremental re-analysis should account for this.
|
||||
|
||||
**Variadic method matching:** When one side is variadic (`parameterCount`
|
||||
undefined) and the other has a fixed count, `METHOD_IMPLEMENTS` edges are
|
||||
emitted with confidence 0.7 instead of 1.0. Variadic methods like
|
||||
`foo(String... args)` may superficially match `foo(String s)` by type but
|
||||
are not guaranteed to be interchangeable across all languages (Java/Kotlin
|
||||
accept this via varargs sugar; TypeScript, C#, Rust do not).
|
||||
|
||||
**Confidence tiering** for `METHOD_IMPLEMENTS` edges:
|
||||
|
||||
| Match quality | Confidence | When |
|
||||
|---|---|---|
|
||||
| Exact parameter types match | 1.0 | Both sides have `parameterTypes` arrays and they match |
|
||||
| Arity (count) matches | 1.0 | Both sides have `parameterCount`, types unavailable |
|
||||
| Variadic vs fixed | 0.7 | One side is variadic, other has fixed count |
|
||||
| Lenient (insufficient info) | 0.7 | One or both sides lack type and count data |
|
||||
|
||||
## Related docs
|
||||
|
||||
- [MIGRATION.md](MIGRATION.md) — breaking changes and migration guidance.
|
||||
- [RUNBOOK.md](RUNBOOK.md) — operational commands and recovery.
|
||||
- [GUARDRAILS.md](GUARDRAILS.md) — safety boundaries for humans and agents.
|
||||
- [TESTING.md](TESTING.md) — how to run tests.
|
||||
|
||||
@@ -10,6 +10,19 @@ All notable changes to GitNexus will be documented in this file.
|
||||
- Added automatic cleanup of stale KuzuDB index files
|
||||
- LadybugDB v0.15 requires explicit VECTOR extension loading for semantic search
|
||||
|
||||
## [1.5.3] - 2026-04-01
|
||||
|
||||
### Added
|
||||
|
||||
- **TypeScript/JavaScript MethodExtractor config** — shared extraction config covering abstract methods, visibility modifiers, async/override keywords, decorators, rest/optional/destructured parameters, and return types (#588) — @compound-ai
|
||||
|
||||
### Fixed
|
||||
|
||||
- **Azure OpenAI compatibility** — use `max_completion_tokens` instead of deprecated `max_tokens` (newer models reject `max_tokens`); skip `temperature` for Azure provider (some models reject non-default values) (#618)
|
||||
- **Simplified Azure interactive setup** — 3 prompts (endpoint, deployment, key) instead of 7 (#618)
|
||||
- **Wiki HTML viewer script injection** — escape `</script>` in embedded JSON so LLM-generated markdown no longer breaks the viewer (#618)
|
||||
- Ensure import rewrites survive npm publish lifecycle
|
||||
|
||||
## [1.4.0] - 2026-03-13
|
||||
|
||||
### Added
|
||||
|
||||
@@ -1,10 +1,10 @@
|
||||
<!-- version: 1.2.0 -->
|
||||
<!-- version: 1.3.0 -->
|
||||
<!--
|
||||
Metadata: version, last reviewed, scope, model policy, reference docs, changelog.
|
||||
Last updated: 2026-03-22
|
||||
-->
|
||||
|
||||
Last reviewed: 2026-03-24
|
||||
Last reviewed: 2026-04-13
|
||||
|
||||
**Project:** GitNexus · **Environment:** dev · **Maintainer:** repository maintainers (see GitHub)
|
||||
|
||||
@@ -41,6 +41,7 @@ If always-on instructions grow, load deep conventions via conditional reads (e.g
|
||||
|
||||
| Date | Version | Change |
|
||||
|------|---------|--------|
|
||||
| 2026-04-13 | 1.3.0 | Updated GitNexus index stats after DAG refactor. |
|
||||
| 2026-03-24 | 1.2.0 | Removed duplicated gitnexus:start block and scope table; replaced with pointers to AGENTS.md. |
|
||||
| 2026-03-23 | 1.1.0 | Updated agent instructions to match AGENTS.md. |
|
||||
| 2026-03-22 | 1.0.0 | Added structured header and changelog. |
|
||||
@@ -49,4 +50,206 @@ If always-on instructions grow, load deep conventions via conditional reads (e.g
|
||||
|
||||
## GitNexus rules
|
||||
|
||||
GitNexus MCP rules are in the `<!-- gitnexus:start -->` … `<!-- gitnexus:end -->` block in **[AGENTS.md](AGENTS.md)** — load that section when working with MCP tools or the graph index.
|
||||
GitNexus MCP rules are in the `<!-- gitnexus:start -->
|
||||
# GitNexus — Code Intelligence
|
||||
|
||||
This project is indexed by GitNexus as **GitNexus** (4325 symbols, 10556 relationships, 300 execution flows). Use the GitNexus MCP tools to understand code, assess impact, and navigate safely.
|
||||
|
||||
> If any GitNexus tool warns the index is stale, run `npx gitnexus analyze` in terminal first.
|
||||
|
||||
## Always Do
|
||||
|
||||
- **MUST run impact analysis before editing any symbol.** Before modifying a function, class, or method, run `gitnexus_impact({target: "symbolName", direction: "upstream"})` and report the blast radius (direct callers, affected processes, risk level) to the user.
|
||||
- **MUST run `gitnexus_detect_changes()` before committing** to verify your changes only affect expected symbols and execution flows.
|
||||
- **MUST warn the user** if impact analysis returns HIGH or CRITICAL risk before proceeding with edits.
|
||||
- When exploring unfamiliar code, use `gitnexus_query({query: "concept"})` to find execution flows instead of grepping. It returns process-grouped results ranked by relevance.
|
||||
- When you need full context on a specific symbol — callers, callees, which execution flows it participates in — use `gitnexus_context({name: "symbolName"})`.
|
||||
|
||||
## When Debugging
|
||||
|
||||
1. `gitnexus_query({query: "<error or symptom>"})` — find execution flows related to the issue
|
||||
2. `gitnexus_context({name: "<suspect function>"})` — see all callers, callees, and process participation
|
||||
3. `READ gitnexus://repo/GitNexus/process/{processName}` — trace the full execution flow step by step
|
||||
4. For regressions: `gitnexus_detect_changes({scope: "compare", base_ref: "main"})` — see what your branch changed
|
||||
|
||||
## When Refactoring
|
||||
|
||||
- **Renaming**: MUST use `gitnexus_rename({symbol_name: "old", new_name: "new", dry_run: true})` first. Review the preview — graph edits are safe, text_search edits need manual review. Then run with `dry_run: false`.
|
||||
- **Extracting/Splitting**: MUST run `gitnexus_context({name: "target"})` to see all incoming/outgoing refs, then `gitnexus_impact({target: "target", direction: "upstream"})` to find all external callers before moving code.
|
||||
- After any refactor: run `gitnexus_detect_changes({scope: "all"})` to verify only expected files changed.
|
||||
|
||||
## Never Do
|
||||
|
||||
- NEVER edit a function, class, or method without first running `gitnexus_impact` on it.
|
||||
- NEVER ignore HIGH or CRITICAL risk warnings from impact analysis.
|
||||
- NEVER rename symbols with find-and-replace — use `gitnexus_rename` which understands the call graph.
|
||||
- NEVER commit changes without running `gitnexus_detect_changes()` to check affected scope.
|
||||
|
||||
## Tools Quick Reference
|
||||
|
||||
| Tool | When to use | Command |
|
||||
|------|-------------|---------|
|
||||
| `query` | Find code by concept | `gitnexus_query({query: "auth validation"})` |
|
||||
| `context` | 360-degree view of one symbol | `gitnexus_context({name: "validateUser"})` |
|
||||
| `impact` | Blast radius before editing | `gitnexus_impact({target: "X", direction: "upstream"})` |
|
||||
| `detect_changes` | Pre-commit scope check | `gitnexus_detect_changes({scope: "staged"})` |
|
||||
| `rename` | Safe multi-file rename | `gitnexus_rename({symbol_name: "old", new_name: "new", dry_run: true})` |
|
||||
| `cypher` | Custom graph queries | `gitnexus_cypher({query: "MATCH ..."})` |
|
||||
|
||||
## Impact Risk Levels
|
||||
|
||||
| Depth | Meaning | Action |
|
||||
|-------|---------|--------|
|
||||
| d=1 | WILL BREAK — direct callers/importers | MUST update these |
|
||||
| d=2 | LIKELY AFFECTED — indirect deps | Should test |
|
||||
| d=3 | MAY NEED TESTING — transitive | Test if critical path |
|
||||
|
||||
## Resources
|
||||
|
||||
| Resource | Use for |
|
||||
|----------|---------|
|
||||
| `gitnexus://repo/GitNexus/context` | Codebase overview, check index freshness |
|
||||
| `gitnexus://repo/GitNexus/clusters` | All functional areas |
|
||||
| `gitnexus://repo/GitNexus/processes` | All execution flows |
|
||||
| `gitnexus://repo/GitNexus/process/{name}` | Step-by-step execution trace |
|
||||
|
||||
## Self-Check Before Finishing
|
||||
|
||||
Before completing any code modification task, verify:
|
||||
1. `gitnexus_impact` was run for all modified symbols
|
||||
2. No HIGH/CRITICAL risk warnings were ignored
|
||||
3. `gitnexus_detect_changes()` confirms changes match expected scope
|
||||
4. All d=1 (WILL BREAK) dependents were updated
|
||||
|
||||
## Keeping the Index Fresh
|
||||
|
||||
After committing code changes, the GitNexus index becomes stale. Re-run analyze to update it:
|
||||
|
||||
```bash
|
||||
npx gitnexus analyze
|
||||
```
|
||||
|
||||
If the index previously included embeddings, preserve them by adding `--embeddings`:
|
||||
|
||||
```bash
|
||||
npx gitnexus analyze --embeddings
|
||||
```
|
||||
|
||||
To check whether embeddings exist, inspect `.gitnexus/meta.json` — the `stats.embeddings` field shows the count (0 means no embeddings). **Running analyze without `--embeddings` will delete any previously generated embeddings.**
|
||||
|
||||
> Claude Code users: A PostToolUse hook handles this automatically after `git commit` and `git merge`.
|
||||
|
||||
## CLI
|
||||
|
||||
| Task | Read this skill file |
|
||||
|------|---------------------|
|
||||
| Understand architecture / "How does X work?" | `.claude/skills/gitnexus/gitnexus-exploring/SKILL.md` |
|
||||
| Blast radius / "What breaks if I change X?" | `.claude/skills/gitnexus/gitnexus-impact-analysis/SKILL.md` |
|
||||
| Trace bugs / "Why is X failing?" | `.claude/skills/gitnexus/gitnexus-debugging/SKILL.md` |
|
||||
| Rename / extract / split / refactor | `.claude/skills/gitnexus/gitnexus-refactoring/SKILL.md` |
|
||||
| Tools, resources, schema reference | `.claude/skills/gitnexus/gitnexus-guide/SKILL.md` |
|
||||
| Index, status, clean, wiki CLI commands | `.claude/skills/gitnexus/gitnexus-cli/SKILL.md` |
|
||||
|
||||
<!-- gitnexus:end -->` block in **[AGENTS.md](AGENTS.md)** — load that section when working with MCP tools or the graph index.
|
||||
|
||||
<!-- gitnexus:start -->
|
||||
# GitNexus — Code Intelligence
|
||||
|
||||
This project is indexed by GitNexus as **GitNexus** (3298 symbols, 7954 relationships, 185 execution flows). Use the GitNexus MCP tools to understand code, assess impact, and navigate safely.
|
||||
|
||||
> If any GitNexus tool warns the index is stale, run `npx gitnexus analyze` in terminal first.
|
||||
|
||||
## Always Do
|
||||
|
||||
- **MUST run impact analysis before editing any symbol.** Before modifying a function, class, or method, run `gitnexus_impact({target: "symbolName", direction: "upstream"})` and report the blast radius (direct callers, affected processes, risk level) to the user.
|
||||
- **MUST run `gitnexus_detect_changes()` before committing** to verify your changes only affect expected symbols and execution flows.
|
||||
- **MUST warn the user** if impact analysis returns HIGH or CRITICAL risk before proceeding with edits.
|
||||
- When exploring unfamiliar code, use `gitnexus_query({query: "concept"})` to find execution flows instead of grepping. It returns process-grouped results ranked by relevance.
|
||||
- When you need full context on a specific symbol — callers, callees, which execution flows it participates in — use `gitnexus_context({name: "symbolName"})`.
|
||||
|
||||
## When Debugging
|
||||
|
||||
1. `gitnexus_query({query: "<error or symptom>"})` — find execution flows related to the issue
|
||||
2. `gitnexus_context({name: "<suspect function>"})` — see all callers, callees, and process participation
|
||||
3. `READ gitnexus://repo/GitNexus/process/{processName}` — trace the full execution flow step by step
|
||||
4. For regressions: `gitnexus_detect_changes({scope: "compare", base_ref: "main"})` — see what your branch changed
|
||||
|
||||
## When Refactoring
|
||||
|
||||
- **Renaming**: MUST use `gitnexus_rename({symbol_name: "old", new_name: "new", dry_run: true})` first. Review the preview — graph edits are safe, text_search edits need manual review. Then run with `dry_run: false`.
|
||||
- **Extracting/Splitting**: MUST run `gitnexus_context({name: "target"})` to see all incoming/outgoing refs, then `gitnexus_impact({target: "target", direction: "upstream"})` to find all external callers before moving code.
|
||||
- After any refactor: run `gitnexus_detect_changes({scope: "all"})` to verify only expected files changed.
|
||||
|
||||
## Never Do
|
||||
|
||||
- NEVER edit a function, class, or method without first running `gitnexus_impact` on it.
|
||||
- NEVER ignore HIGH or CRITICAL risk warnings from impact analysis.
|
||||
- NEVER rename symbols with find-and-replace — use `gitnexus_rename` which understands the call graph.
|
||||
- NEVER commit changes without running `gitnexus_detect_changes()` to check affected scope.
|
||||
|
||||
## Tools Quick Reference
|
||||
|
||||
| Tool | When to use | Command |
|
||||
|------|-------------|---------|
|
||||
| `query` | Find code by concept | `gitnexus_query({query: "auth validation"})` |
|
||||
| `context` | 360-degree view of one symbol | `gitnexus_context({name: "validateUser"})` |
|
||||
| `impact` | Blast radius before editing | `gitnexus_impact({target: "X", direction: "upstream"})` |
|
||||
| `detect_changes` | Pre-commit scope check | `gitnexus_detect_changes({scope: "staged"})` |
|
||||
| `rename` | Safe multi-file rename | `gitnexus_rename({symbol_name: "old", new_name: "new", dry_run: true})` |
|
||||
| `cypher` | Custom graph queries | `gitnexus_cypher({query: "MATCH ..."})` |
|
||||
|
||||
## Impact Risk Levels
|
||||
|
||||
| Depth | Meaning | Action |
|
||||
|-------|---------|--------|
|
||||
| d=1 | WILL BREAK — direct callers/importers | MUST update these |
|
||||
| d=2 | LIKELY AFFECTED — indirect deps | Should test |
|
||||
| d=3 | MAY NEED TESTING — transitive | Test if critical path |
|
||||
|
||||
## Resources
|
||||
|
||||
| Resource | Use for |
|
||||
|----------|---------|
|
||||
| `gitnexus://repo/GitNexus/context` | Codebase overview, check index freshness |
|
||||
| `gitnexus://repo/GitNexus/clusters` | All functional areas |
|
||||
| `gitnexus://repo/GitNexus/processes` | All execution flows |
|
||||
| `gitnexus://repo/GitNexus/process/{name}` | Step-by-step execution trace |
|
||||
|
||||
## Self-Check Before Finishing
|
||||
|
||||
Before completing any code modification task, verify:
|
||||
1. `gitnexus_impact` was run for all modified symbols
|
||||
2. No HIGH/CRITICAL risk warnings were ignored
|
||||
3. `gitnexus_detect_changes()` confirms changes match expected scope
|
||||
4. All d=1 (WILL BREAK) dependents were updated
|
||||
|
||||
## Keeping the Index Fresh
|
||||
|
||||
After committing code changes, the GitNexus index becomes stale. Re-run analyze to update it:
|
||||
|
||||
```bash
|
||||
npx gitnexus analyze
|
||||
```
|
||||
|
||||
If the index previously included embeddings, preserve them by adding `--embeddings`:
|
||||
|
||||
```bash
|
||||
npx gitnexus analyze --embeddings
|
||||
```
|
||||
|
||||
To check whether embeddings exist, inspect `.gitnexus/meta.json` — the `stats.embeddings` field shows the count (0 means no embeddings). **Running analyze without `--embeddings` will delete any previously generated embeddings.**
|
||||
|
||||
> Claude Code users: A PostToolUse hook handles this automatically after `git commit` and `git merge`.
|
||||
|
||||
## CLI
|
||||
|
||||
| Task | Read this skill file |
|
||||
|------|---------------------|
|
||||
| Understand architecture / "How does X work?" | `.claude/skills/gitnexus/gitnexus-exploring/SKILL.md` |
|
||||
| Blast radius / "What breaks if I change X?" | `.claude/skills/gitnexus/gitnexus-impact-analysis/SKILL.md` |
|
||||
| Trace bugs / "Why is X failing?" | `.claude/skills/gitnexus/gitnexus-debugging/SKILL.md` |
|
||||
| Rename / extract / split / refactor | `.claude/skills/gitnexus/gitnexus-refactoring/SKILL.md` |
|
||||
| Tools, resources, schema reference | `.claude/skills/gitnexus/gitnexus-guide/SKILL.md` |
|
||||
| Index, status, clean, wiki CLI commands | `.claude/skills/gitnexus/gitnexus-cli/SKILL.md` |
|
||||
|
||||
<!-- gitnexus:end -->
|
||||
|
||||
@@ -0,0 +1,27 @@
|
||||
# Migration Guide
|
||||
|
||||
## OVERRIDES → METHOD_OVERRIDES (PR #642)
|
||||
|
||||
The `OVERRIDES` relationship type has been renamed to `METHOD_OVERRIDES` for
|
||||
consistency with the new `METHOD_IMPLEMENTS` edge type.
|
||||
|
||||
### Do I need to migrate?
|
||||
|
||||
**No.** Backward compatibility is handled automatically at runtime:
|
||||
|
||||
- `local-backend.ts` dual-reads both `OVERRIDES` and `METHOD_OVERRIDES` in all
|
||||
impact-analysis and context queries. Existing stored graphs with `OVERRIDES`
|
||||
edges continue to return correct results without any manual intervention.
|
||||
- The `REL_TYPES` array in `schema-constants.ts` includes both names so Cypher
|
||||
queries that reference either will work.
|
||||
|
||||
### What happens on re-index?
|
||||
|
||||
Running `npx gitnexus analyze` on a repository produces `METHOD_OVERRIDES`
|
||||
edges going forward. The old `OVERRIDES` edges are replaced as part of the
|
||||
normal full re-index.
|
||||
|
||||
### When will the legacy alias be removed?
|
||||
|
||||
The `OVERRIDES` compat alias will remain until a future major version. Removal
|
||||
will be announced in this file and in the changelog before it happens.
|
||||
@@ -52,7 +52,7 @@ https://github.com/user-attachments/assets/172685ba-8e54-4ea7-9ad1-e31a3398da72
|
||||
| **What** | Index repos locally, connect AI agents via MCP | Visual graph explorer + AI chat in browser |
|
||||
| **For** | Daily development with Cursor, Claude Code, Codex, Windsurf, OpenCode | Quick exploration, demos, one-off analysis |
|
||||
| **Scale** | Full repos, any size | Limited by browser memory (~5k files), or unlimited via backend mode |
|
||||
| **Install** | `npm install -g gitnexus` | No install —[gitnexus.vercel.app](https://gitnexus.vercel.app) |
|
||||
| **Install** | `npm install -g gitnexus` | No install — [gitnexus.vercel.app](https://gitnexus.vercel.app) |
|
||||
| **Storage** | LadybugDB native (fast, persistent) | LadybugDB WASM (in-memory, per session) |
|
||||
| **Parsing** | Tree-sitter native bindings | Tree-sitter WASM |
|
||||
| **Privacy** | Everything local, no network | Everything in-browser, no server |
|
||||
@@ -119,7 +119,6 @@ To configure MCP for your editor, run `npx gitnexus setup` once — or set it up
|
||||
| **Codex** | Yes | Yes | — | MCP + Skills |
|
||||
| **Windsurf** | Yes | — | — | MCP |
|
||||
| **OpenCode** | Yes | Yes | — | MCP + Skills |
|
||||
| **Codex** | Yes | — | — | MCP |
|
||||
|
||||
> **Claude Code** gets the deepest integration: MCP tools + agent skills + PreToolUse hooks that enrich searches with graph context + PostToolUse hooks that auto-reindex after commits.
|
||||
|
||||
@@ -206,11 +205,21 @@ gitnexus clean --all --force # Delete all indexes
|
||||
gitnexus wiki [path] # Generate repository wiki from knowledge graph
|
||||
gitnexus wiki --model <model> # Wiki with custom LLM model (default: gpt-4o-mini)
|
||||
gitnexus wiki --base-url <url> # Wiki with custom LLM API base URL
|
||||
|
||||
# Repository groups (multi-repo / monorepo service tracking)
|
||||
gitnexus group create <name> # Create a repository group
|
||||
gitnexus group add <name> <repo> # Add a repo to a group
|
||||
gitnexus group remove <name> <repo> # Remove a repo from a group
|
||||
gitnexus group list [name] # List groups, or show one group's config
|
||||
gitnexus group sync <name> # Extract contracts and match across repos/services
|
||||
gitnexus group contracts <name> # Inspect extracted contracts and cross-links
|
||||
gitnexus group query <name> <q> # Search execution flows across all repos in a group
|
||||
gitnexus group status <name> # Check staleness of repos in a group
|
||||
```
|
||||
|
||||
### What Your AI Agent Gets
|
||||
|
||||
**7 tools** exposed via MCP:
|
||||
**16 tools** exposed via MCP (11 per-repo + 5 group):
|
||||
|
||||
| Tool | What It Does | `repo` Param |
|
||||
| ------------------ | ----------------------------------------------------------------- | -------------- |
|
||||
@@ -221,6 +230,11 @@ gitnexus wiki --base-url <url> # Wiki with custom LLM API base URL
|
||||
| `detect_changes` | Git-diff impact — maps changed lines to affected processes | Optional |
|
||||
| `rename` | Multi-file coordinated rename with graph + text search | Optional |
|
||||
| `cypher` | Raw Cypher graph queries | Optional |
|
||||
| `group_list` | List configured repository groups | — |
|
||||
| `group_sync` | Extract contracts and match across repos/services | — |
|
||||
| `group_contracts`| Inspect extracted contracts and cross-links | — |
|
||||
| `group_query` | Search execution flows across all repos in a group | — |
|
||||
| `group_status` | Check staleness of repos in a group | — |
|
||||
|
||||
> When only one repo is indexed, the `repo` parameter is optional. With multiple repos, specify which one: `query({query: "auth", repo: "my-app"})`.
|
||||
|
||||
|
||||
@@ -0,0 +1,725 @@
|
||||
# PR #626 HIGH-Priority Fixes Implementation Plan
|
||||
|
||||
> **For agentic workers:** REQUIRED SUB-SKILL: Use superpowers:subagent-driven-development (recommended) or superpowers:executing-plans to implement this plan task-by-task. Steps use checkbox (`- [ ]`) syntax for tracking.
|
||||
|
||||
**Goal:** Fix 4 HIGH-priority issues from PR #626 code review before merge.
|
||||
|
||||
**Architecture:** Minimal targeted fixes — each task is independent. TDD: tests first, then implementation. No refactoring beyond what's needed.
|
||||
|
||||
**Tech Stack:** TypeScript, Vitest, Node.js fs/path APIs
|
||||
|
||||
**Spec:** `docs/superpowers/specs/2026-04-02-pr626-high-fixes-design.md`
|
||||
|
||||
**Paths:** All file paths are relative to the monorepo root (`GitNexus/`). Git commands run from the root. The `gitnexus/` prefix is a package subdirectory, not a separate repo.
|
||||
|
||||
---
|
||||
|
||||
### Task 1: Path Traversal — Validate Group Name
|
||||
|
||||
**Files:**
|
||||
- Modify: `gitnexus/src/core/group/storage.ts:17-19` (getGroupDir) and `:63-68` (createGroupDir)
|
||||
- Test: `gitnexus/test/unit/group/storage.test.ts`
|
||||
|
||||
- [ ] **Step 1: Write failing tests for validateGroupName**
|
||||
|
||||
In `gitnexus/test/unit/group/storage.test.ts`, add `createGroupDir` and `validateGroupName` to the existing import from `'../../../src/core/group/storage.js'` (line 6-11). Then add these describe blocks at the end of the outer `describe('Group storage', ...)`:
|
||||
|
||||
```typescript
|
||||
describe('validateGroupName', () => {
|
||||
it('test_validateGroupName_traversal_path_throws', () => {
|
||||
expect(() => validateGroupName('../../evil')).toThrow(/Invalid group name/);
|
||||
});
|
||||
|
||||
it('test_validateGroupName_slash_in_name_throws', () => {
|
||||
expect(() => validateGroupName('foo/bar')).toThrow(/Invalid group name/);
|
||||
});
|
||||
|
||||
it('test_validateGroupName_empty_string_throws', () => {
|
||||
expect(() => validateGroupName('')).toThrow(/Invalid group name/);
|
||||
});
|
||||
|
||||
it('test_validateGroupName_starts_with_dash_throws', () => {
|
||||
expect(() => validateGroupName('-leading-dash')).toThrow(/Invalid group name/);
|
||||
});
|
||||
|
||||
it('test_validateGroupName_starts_with_underscore_throws', () => {
|
||||
expect(() => validateGroupName('_leading')).toThrow(/Invalid group name/);
|
||||
});
|
||||
|
||||
it('test_validateGroupName_dots_throws', () => {
|
||||
expect(() => validateGroupName('com.example')).toThrow(/Invalid group name/);
|
||||
});
|
||||
|
||||
it('test_validateGroupName_valid_alphanumeric_passes', () => {
|
||||
expect(() => validateGroupName('my-group_01')).not.toThrow();
|
||||
});
|
||||
|
||||
it('test_validateGroupName_single_char_passes', () => {
|
||||
expect(() => validateGroupName('A')).not.toThrow();
|
||||
});
|
||||
|
||||
it('test_validateGroupName_all_digits_passes', () => {
|
||||
expect(() => validateGroupName('123')).not.toThrow();
|
||||
});
|
||||
});
|
||||
|
||||
describe('getGroupDir rejects invalid names', () => {
|
||||
it('test_getGroupDir_traversal_throws', () => {
|
||||
expect(() => getGroupDir(tmpDir, '../../etc')).toThrow(/Invalid group name/);
|
||||
});
|
||||
|
||||
it('test_getGroupDir_valid_name_returns_path', () => {
|
||||
const dir = getGroupDir(tmpDir, 'company');
|
||||
expect(dir).toBe(path.join(tmpDir, 'groups', 'company'));
|
||||
});
|
||||
});
|
||||
|
||||
describe('createGroupDir rejects invalid names', () => {
|
||||
it('test_createGroupDir_traversal_throws', async () => {
|
||||
await expect(createGroupDir(tmpDir, '../evil')).rejects.toThrow(/Invalid group name/);
|
||||
});
|
||||
});
|
||||
```
|
||||
|
||||
- [ ] **Step 2: Run tests to verify they fail**
|
||||
|
||||
Run: `cd gitnexus && npx vitest run test/unit/group/storage.test.ts`
|
||||
Expected: FAIL — `validateGroupName` is not exported, `getGroupDir` does not throw.
|
||||
|
||||
- [ ] **Step 3: Implement validateGroupName and wire into getGroupDir and createGroupDir**
|
||||
|
||||
In `gitnexus/src/core/group/storage.ts`, add the validation function before `getGroupDir` and call it:
|
||||
|
||||
```typescript
|
||||
const GROUP_NAME_RE = /^[a-zA-Z0-9][a-zA-Z0-9_-]*$/;
|
||||
|
||||
export function validateGroupName(name: string): void {
|
||||
if (!GROUP_NAME_RE.test(name)) {
|
||||
throw new Error(
|
||||
`Invalid group name "${name}". Names must start with a letter or digit and contain only [a-zA-Z0-9_-].`,
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
export function getGroupDir(gitnexusDir: string, groupName: string): string {
|
||||
validateGroupName(groupName);
|
||||
return path.join(gitnexusDir, 'groups', groupName);
|
||||
}
|
||||
```
|
||||
|
||||
`createGroupDir` already calls `getGroupDir` at line 68, so it inherits validation automatically. No change needed in `createGroupDir`.
|
||||
|
||||
- [ ] **Step 4: Run tests to verify they pass**
|
||||
|
||||
Run: `cd gitnexus && npx vitest run test/unit/group/storage.test.ts`
|
||||
Expected: ALL PASS
|
||||
|
||||
- [ ] **Step 5: Commit**
|
||||
|
||||
```bash
|
||||
cd gitnexus && git add src/core/group/storage.ts test/unit/group/storage.test.ts
|
||||
git commit -m "fix(group): validate group name to prevent path traversal
|
||||
|
||||
Add validateGroupName() with regex [a-zA-Z0-9][a-zA-Z0-9_-]*.
|
||||
Called in getGroupDir (defense in depth) which covers all CLI entry
|
||||
points: create, add, remove, status, sync.
|
||||
|
||||
Addresses PR #626 review item 1 (HIGH).
|
||||
|
||||
Co-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com>"
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
### Task 2: Directory Exclusions in Service Boundary Detector
|
||||
|
||||
**Files:**
|
||||
- Modify: `gitnexus/src/core/group/service-boundary-detector.ts:24-51` (add constant), `:78` (walkForBoundaries), `:130` (hasSourceFilesInSubdirs)
|
||||
- Test: `gitnexus/test/unit/group/service-boundary-detector.test.ts`
|
||||
|
||||
- [ ] **Step 1: Write failing tests for excluded directories**
|
||||
|
||||
Add this describe block inside the existing `detectServiceBoundaries` describe in `gitnexus/test/unit/group/service-boundary-detector.test.ts`:
|
||||
|
||||
```typescript
|
||||
it('test_detect_skips_vendor_directory', async () => {
|
||||
writeFile('services/auth/package.json', '{}');
|
||||
writeFile('services/auth/src/index.ts', '');
|
||||
// vendor should be skipped — its contents should not create a boundary
|
||||
writeFile('vendor/some-dep/package.json', '{}');
|
||||
writeFile('vendor/some-dep/src/lib.go', '');
|
||||
|
||||
const boundaries = await detectServiceBoundaries(tmpDir);
|
||||
|
||||
const paths = boundaries.map((b) => b.servicePath);
|
||||
expect(paths).toContain('services/auth');
|
||||
expect(paths).not.toContain('vendor/some-dep');
|
||||
});
|
||||
|
||||
it('test_detect_skips_target_directory', async () => {
|
||||
writeFile('services/api/go.mod', 'module api');
|
||||
writeFile('services/api/main.go', '');
|
||||
writeFile('target/classes/Main.java', '');
|
||||
writeFile('target/pom.xml', '<project/>');
|
||||
|
||||
const boundaries = await detectServiceBoundaries(tmpDir);
|
||||
|
||||
const paths = boundaries.map((b) => b.servicePath);
|
||||
expect(paths).toContain('services/api');
|
||||
expect(paths).not.toContain('target');
|
||||
});
|
||||
|
||||
it('test_detect_skips_pycache_directory', async () => {
|
||||
writeFile('services/ml/pyproject.toml', '[project]');
|
||||
writeFile('services/ml/model.py', '');
|
||||
// __pycache__ with a marker + source files — would be detected as
|
||||
// a boundary if not excluded, since it has package.json + .py file
|
||||
writeFile('__pycache__/package.json', '{}');
|
||||
writeFile('__pycache__/cached.py', '');
|
||||
|
||||
const boundaries = await detectServiceBoundaries(tmpDir);
|
||||
|
||||
const paths = boundaries.map((b) => b.servicePath);
|
||||
expect(paths).toContain('services/ml');
|
||||
expect(paths.every((p) => !p.includes('__pycache__'))).toBe(true);
|
||||
});
|
||||
|
||||
it('test_detect_skips_dotfile_directories_regression', async () => {
|
||||
writeFile('services/api/package.json', '{}');
|
||||
writeFile('services/api/src/index.ts', '');
|
||||
writeFile('.hidden/package.json', '{}');
|
||||
writeFile('.hidden/src/index.ts', '');
|
||||
|
||||
const boundaries = await detectServiceBoundaries(tmpDir);
|
||||
|
||||
const paths = boundaries.map((b) => b.servicePath);
|
||||
expect(paths).toContain('services/api');
|
||||
expect(paths).not.toContain('.hidden');
|
||||
});
|
||||
|
||||
it('test_detect_does_not_skip_regular_source_directories', async () => {
|
||||
writeFile('services/api/package.json', '{}');
|
||||
writeFile('services/api/src/index.ts', '');
|
||||
|
||||
const boundaries = await detectServiceBoundaries(tmpDir);
|
||||
|
||||
expect(boundaries).toHaveLength(1);
|
||||
expect(boundaries[0].serviceName).toBe('api');
|
||||
});
|
||||
```
|
||||
|
||||
- [ ] **Step 2: Run tests to verify `vendor` and `target` tests fail**
|
||||
|
||||
Run: `cd gitnexus && npx vitest run test/unit/group/service-boundary-detector.test.ts`
|
||||
Expected: `test_detect_skips_vendor_directory` and `test_detect_skips_target_directory` FAIL (vendor/target not excluded). Other new tests may pass since dotfile exclusion already exists.
|
||||
|
||||
- [ ] **Step 3: Add EXCLUDED_DIRS constant and update both walking functions**
|
||||
|
||||
In `gitnexus/src/core/group/service-boundary-detector.ts`:
|
||||
|
||||
After `SOURCE_EXTENSIONS` (after line 51), add:
|
||||
|
||||
```typescript
|
||||
const EXCLUDED_DIRS = new Set([
|
||||
'node_modules',
|
||||
'vendor',
|
||||
'target',
|
||||
'build',
|
||||
'dist',
|
||||
'__pycache__',
|
||||
'.venv',
|
||||
'venv',
|
||||
'.tox',
|
||||
'.mypy_cache',
|
||||
'.gradle',
|
||||
'.mvn',
|
||||
'out',
|
||||
'bin',
|
||||
]);
|
||||
```
|
||||
|
||||
In `walkForBoundaries`, replace line 78:
|
||||
```typescript
|
||||
if (entry.name.startsWith('.') || entry.name === 'node_modules') continue;
|
||||
```
|
||||
with:
|
||||
```typescript
|
||||
if (entry.name.startsWith('.') || EXCLUDED_DIRS.has(entry.name)) continue;
|
||||
```
|
||||
|
||||
In `hasSourceFilesInSubdirs`, replace line 130:
|
||||
```typescript
|
||||
if (entry.isDirectory() && !entry.name.startsWith('.') && entry.name !== 'node_modules') {
|
||||
```
|
||||
with:
|
||||
```typescript
|
||||
if (entry.isDirectory() && !entry.name.startsWith('.') && !EXCLUDED_DIRS.has(entry.name)) {
|
||||
```
|
||||
|
||||
- [ ] **Step 4: Run tests to verify they pass**
|
||||
|
||||
Run: `cd gitnexus && npx vitest run test/unit/group/service-boundary-detector.test.ts`
|
||||
Expected: ALL PASS
|
||||
|
||||
- [ ] **Step 5: Commit**
|
||||
|
||||
```bash
|
||||
cd gitnexus && git add src/core/group/service-boundary-detector.ts test/unit/group/service-boundary-detector.test.ts
|
||||
git commit -m "fix(group): add directory exclusions to service boundary detector
|
||||
|
||||
Add EXCLUDED_DIRS set: vendor, target, build, dist, __pycache__,
|
||||
.venv, venv, .tox, .mypy_cache, .gradle, .mvn, out, bin.
|
||||
Applied in walkForBoundaries and hasSourceFilesInSubdirs.
|
||||
Replaces inline node_modules check.
|
||||
|
||||
Addresses PR #626 review item 3 (HIGH).
|
||||
|
||||
Co-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com>"
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
### Task 3: Remove Double-Close of LadybugDB Pools
|
||||
|
||||
**Files:**
|
||||
- Modify: `gitnexus/src/cli/group.ts:160` (remove import), `:187-189` (remove finally block body)
|
||||
- Test: `gitnexus/test/unit/group/sync.test.ts` (add pool cleanup test)
|
||||
- Test: `gitnexus/test/integration/group/group-cli.test.ts` (verify no blanket close in source)
|
||||
|
||||
- [ ] **Step 1: Write unit tests for per-id pool cleanup in sync.ts**
|
||||
|
||||
Add to `gitnexus/test/unit/group/sync.test.ts`, inside the existing `describe('syncGroup', ...)`:
|
||||
|
||||
```typescript
|
||||
it('test_syncGroup_closes_only_opened_pools', async () => {
|
||||
const config = makeConfig({
|
||||
'app/backend': 'backend-repo',
|
||||
'app/frontend': 'frontend-repo',
|
||||
});
|
||||
|
||||
const closedIds: string[] = [];
|
||||
|
||||
// Mock initLbug/closeLbug via per-repo override that tracks pool lifecycle
|
||||
const { vi } = await import('vitest');
|
||||
const poolAdapter = await import('../../../src/core/lbug/pool-adapter.js');
|
||||
const initSpy = vi.spyOn(poolAdapter, 'initLbug').mockResolvedValue(undefined);
|
||||
const closeSpy = vi.spyOn(poolAdapter, 'closeLbug').mockImplementation(async (id?: string) => {
|
||||
if (id) closedIds.push(id);
|
||||
});
|
||||
|
||||
try {
|
||||
await syncGroup(config, {
|
||||
resolveRepoHandle: async (_name, groupPath) => ({
|
||||
id: groupPath.replace(/\//g, '-'),
|
||||
path: groupPath,
|
||||
repoPath: '/tmp/' + groupPath,
|
||||
storagePath: '/tmp/' + groupPath + '/.gitnexus',
|
||||
}),
|
||||
skipWrite: true,
|
||||
}).catch(() => {});
|
||||
// Regardless of extraction errors, closeLbug should be called per id
|
||||
// closeLbug should only receive specific pool ids, never undefined/empty
|
||||
for (const id of closedIds) {
|
||||
expect(id).toBeTruthy();
|
||||
expect(typeof id).toBe('string');
|
||||
}
|
||||
// No blanket close (no-arg call)
|
||||
const blanketCalls = closeSpy.mock.calls.filter((args) => args.length === 0 || !args[0]);
|
||||
expect(blanketCalls).toHaveLength(0);
|
||||
} finally {
|
||||
initSpy.mockRestore();
|
||||
closeSpy.mockRestore();
|
||||
}
|
||||
});
|
||||
```
|
||||
|
||||
- [ ] **Step 2: Run sync unit test to verify it passes (sync.ts already does per-id cleanup)**
|
||||
|
||||
Run: `cd gitnexus && npx vitest run test/unit/group/sync.test.ts`
|
||||
Expected: PASS — sync.ts already cleans up correctly. This test locks the behavior.
|
||||
|
||||
- [ ] **Step 3: Write test verifying CLI source has no blanket closeLbug()**
|
||||
|
||||
Add to `gitnexus/test/integration/group/group-cli.test.ts`:
|
||||
|
||||
```typescript
|
||||
it('test_sync_command_source_does_not_call_blanket_closeLbug', () => {
|
||||
const cliGroupPath = path.join(repoRoot, 'src', 'cli', 'group.ts');
|
||||
const source = fs.readFileSync(cliGroupPath, 'utf-8');
|
||||
|
||||
// closeLbug() without arguments (blanket close) must not appear.
|
||||
// closeLbug(id) with argument is fine (that's in sync.ts, not here).
|
||||
// Match closeLbug() but not closeLbug(someArg)
|
||||
const blanketClosePattern = /closeLbug\s*\(\s*\)/;
|
||||
expect(source).not.toMatch(blanketClosePattern);
|
||||
});
|
||||
```
|
||||
|
||||
- [ ] **Step 4: Run test to verify it fails**
|
||||
|
||||
Run: `cd gitnexus && npx vitest run test/integration/group/group-cli.test.ts`
|
||||
Expected: FAIL — `closeLbug()` (no args) exists at line 188.
|
||||
|
||||
- [ ] **Step 5: Remove blanket closeLbug() from cli/group.ts**
|
||||
|
||||
In `gitnexus/src/cli/group.ts`:
|
||||
|
||||
Remove the `closeLbug` import at line 160:
|
||||
```typescript
|
||||
const { closeLbug } = await import('../core/lbug/pool-adapter.js');
|
||||
```
|
||||
|
||||
Replace the try/finally wrapper (lines 162-189):
|
||||
```typescript
|
||||
try {
|
||||
const groupDir = getGroupDir(getDefaultGitnexusDir(), name);
|
||||
const config = await loadGroupConfig(groupDir);
|
||||
|
||||
console.log(`Syncing group "${name}" (${Object.keys(config.repos).length} repos)...\n`);
|
||||
|
||||
const result = await syncGroup(config, {
|
||||
groupDir,
|
||||
allowStale: Boolean(opts.allowStale),
|
||||
verbose: Boolean(opts.verbose),
|
||||
skipEmbeddings: Boolean(opts.skipEmbeddings),
|
||||
exactOnly: Boolean(opts.exactOnly),
|
||||
});
|
||||
|
||||
if (opts.json) {
|
||||
console.log(JSON.stringify(result, null, 2));
|
||||
} else {
|
||||
console.log(`\nMatching cascade:`);
|
||||
const exactLinks = result.crossLinks.filter((l) => l.matchType === 'exact');
|
||||
console.log(` exact: ${exactLinks.length} cross-links (confidence 1.0)`);
|
||||
console.log(` unmatched: ${result.unmatched.length} contracts`);
|
||||
console.log(
|
||||
`\nWrote contracts.json (${result.contracts.length} contracts, ${result.crossLinks.length} cross-links)`,
|
||||
);
|
||||
}
|
||||
} finally {
|
||||
await closeLbug().catch(() => {});
|
||||
}
|
||||
```
|
||||
|
||||
Becomes (remove try/finally entirely, since sync.ts handles its own cleanup):
|
||||
```typescript
|
||||
const groupDir = getGroupDir(getDefaultGitnexusDir(), name);
|
||||
const config = await loadGroupConfig(groupDir);
|
||||
|
||||
console.log(`Syncing group "${name}" (${Object.keys(config.repos).length} repos)...\n`);
|
||||
|
||||
const result = await syncGroup(config, {
|
||||
groupDir,
|
||||
allowStale: Boolean(opts.allowStale),
|
||||
verbose: Boolean(opts.verbose),
|
||||
skipEmbeddings: Boolean(opts.skipEmbeddings),
|
||||
exactOnly: Boolean(opts.exactOnly),
|
||||
});
|
||||
|
||||
if (opts.json) {
|
||||
console.log(JSON.stringify(result, null, 2));
|
||||
} else {
|
||||
console.log(`\nMatching cascade:`);
|
||||
const exactLinks = result.crossLinks.filter((l) => l.matchType === 'exact');
|
||||
console.log(` exact: ${exactLinks.length} cross-links (confidence 1.0)`);
|
||||
console.log(` unmatched: ${result.unmatched.length} contracts`);
|
||||
console.log(
|
||||
`\nWrote contracts.json (${result.contracts.length} contracts, ${result.crossLinks.length} cross-links)`,
|
||||
);
|
||||
}
|
||||
```
|
||||
|
||||
- [ ] **Step 6: Run tests to verify they pass**
|
||||
|
||||
Run: `cd gitnexus && npx vitest run test/integration/group/group-cli.test.ts test/unit/group/sync.test.ts`
|
||||
Expected: ALL PASS
|
||||
|
||||
- [ ] **Step 7: Commit**
|
||||
|
||||
```bash
|
||||
cd gitnexus && git add src/cli/group.ts test/integration/group/group-cli.test.ts test/unit/group/sync.test.ts
|
||||
git commit -m "fix(group): remove blanket closeLbug() from CLI sync command
|
||||
|
||||
sync.ts already closes pools per-id in its finally block.
|
||||
The blanket closeLbug() in cli/group.ts tears down ALL active pools
|
||||
including unrelated ones in MCP server context.
|
||||
|
||||
Addresses PR #626 review item 4 (HIGH).
|
||||
|
||||
Co-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com>"
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
### Task 4: gRPC Proto Regex — Brace-Depth Counter
|
||||
|
||||
**Files:**
|
||||
- Modify: `gitnexus/src/core/group/extractors/grpc-extractor.ts:101-130` (parseProtoFile)
|
||||
- Test: `gitnexus/test/unit/group/grpc-extractor.test.ts`
|
||||
|
||||
- [ ] **Step 1: Write failing tests for nested braces in proto services**
|
||||
|
||||
Add this describe block inside the existing `proto file parsing` describe in `gitnexus/test/unit/group/grpc-extractor.test.ts`:
|
||||
|
||||
```typescript
|
||||
it('test_extract_proto_with_google_api_http_nested_braces', async () => {
|
||||
writeFile(
|
||||
'api/gateway.proto',
|
||||
`syntax = "proto3";
|
||||
package gateway.v1;
|
||||
|
||||
import "google/api/annotations.proto";
|
||||
|
||||
service GatewayService {
|
||||
rpc GetUser (GetUserRequest) returns (UserResponse) {
|
||||
option (google.api.http) = {
|
||||
get: "/v1/users/{user_id}"
|
||||
};
|
||||
}
|
||||
rpc CreateUser (CreateUserRequest) returns (UserResponse) {
|
||||
option (google.api.http) = {
|
||||
post: "/v1/users"
|
||||
body: "*"
|
||||
};
|
||||
}
|
||||
}`,
|
||||
);
|
||||
|
||||
const contracts = await extractor.extract(null, tmpDir, makeRepo(tmpDir));
|
||||
const providers = contracts.filter(
|
||||
(c) => c.role === 'provider' && c.symbolRef.filePath === 'api/gateway.proto',
|
||||
);
|
||||
|
||||
expect(providers).toHaveLength(2);
|
||||
const ids = providers.map((c) => c.contractId).sort();
|
||||
expect(ids).toEqual([
|
||||
'grpc::gateway.v1.GatewayService/CreateUser',
|
||||
'grpc::gateway.v1.GatewayService/GetUser',
|
||||
]);
|
||||
});
|
||||
|
||||
it('test_extract_proto_with_multiple_services', async () => {
|
||||
writeFile(
|
||||
'api/multi.proto',
|
||||
`syntax = "proto3";
|
||||
package multi;
|
||||
|
||||
service ServiceA {
|
||||
rpc MethodA (Req) returns (Res);
|
||||
}
|
||||
|
||||
service ServiceB {
|
||||
rpc MethodB1 (Req) returns (Res);
|
||||
rpc MethodB2 (Req) returns (Res);
|
||||
}`,
|
||||
);
|
||||
|
||||
const contracts = await extractor.extract(null, tmpDir, makeRepo(tmpDir));
|
||||
const providers = contracts.filter(
|
||||
(c) => c.role === 'provider' && c.symbolRef.filePath === 'api/multi.proto',
|
||||
);
|
||||
|
||||
expect(providers).toHaveLength(3);
|
||||
const ids = providers.map((c) => c.contractId).sort();
|
||||
expect(ids).toEqual([
|
||||
'grpc::multi.ServiceA/MethodA',
|
||||
'grpc::multi.ServiceB/MethodB1',
|
||||
'grpc::multi.ServiceB/MethodB2',
|
||||
]);
|
||||
});
|
||||
|
||||
it('test_extract_proto_with_nested_option_blocks_in_rpc', async () => {
|
||||
writeFile(
|
||||
'api/nested.proto',
|
||||
`syntax = "proto3";
|
||||
package nested;
|
||||
|
||||
service DeepService {
|
||||
rpc DeepMethod (Req) returns (Res) {
|
||||
option (google.api.http) = {
|
||||
post: "/v1/deep"
|
||||
body: "*"
|
||||
additional_bindings {
|
||||
get: "/v1/deep/{id}"
|
||||
}
|
||||
};
|
||||
}
|
||||
}`,
|
||||
);
|
||||
|
||||
const contracts = await extractor.extract(null, tmpDir, makeRepo(tmpDir));
|
||||
const providers = contracts.filter(
|
||||
(c) => c.role === 'provider' && c.symbolRef.filePath === 'api/nested.proto',
|
||||
);
|
||||
|
||||
expect(providers).toHaveLength(1);
|
||||
expect(providers[0].contractId).toBe('grpc::nested.DeepService/DeepMethod');
|
||||
});
|
||||
|
||||
it('test_extract_proto_malformed_unclosed_brace_skips_service', async () => {
|
||||
writeFile(
|
||||
'api/broken.proto',
|
||||
`syntax = "proto3";
|
||||
package broken;
|
||||
|
||||
service IncompleteService {
|
||||
rpc SomeMethod (Req) returns (Res);
|
||||
// Missing closing brace — EOF before depth returns to 0
|
||||
`,
|
||||
);
|
||||
|
||||
// Should not throw; incomplete service is silently skipped
|
||||
const contracts = await extractor.extract(null, tmpDir, makeRepo(tmpDir));
|
||||
const providers = contracts.filter(
|
||||
(c) => c.role === 'provider' && c.symbolRef.filePath === 'api/broken.proto',
|
||||
);
|
||||
|
||||
// The old regex would find partial match; the new parser should skip it
|
||||
expect(providers).toHaveLength(0);
|
||||
});
|
||||
```
|
||||
|
||||
- [ ] **Step 2: Run tests to verify the nested brace test fails**
|
||||
|
||||
Run: `cd gitnexus && npx vitest run test/unit/group/grpc-extractor.test.ts`
|
||||
Expected: `test_extract_proto_with_google_api_http_nested_braces` FAIL — regex stops at first `}` inside the `option` block.
|
||||
|
||||
- [ ] **Step 3: Replace serviceRe regex with extractServiceBlocks function**
|
||||
|
||||
In `gitnexus/src/core/group/extractors/grpc-extractor.ts`, replace the `parseProtoFile` method (lines 101-130):
|
||||
|
||||
```typescript
|
||||
private parseProtoFile(content: string, filePath: string): ExtractedContract[] {
|
||||
const out: ExtractedContract[] = [];
|
||||
|
||||
const pkgMatch = content.match(/^package\s+([\w.]+)\s*;/m);
|
||||
const pkg = pkgMatch ? pkgMatch[1] : '';
|
||||
|
||||
for (const { name: serviceName, body } of extractServiceBlocks(content)) {
|
||||
const rpcRe = /rpc\s+(\w+)\s*\(/g;
|
||||
let rpcMatch: RegExpExecArray | null;
|
||||
while ((rpcMatch = rpcRe.exec(body)) !== null) {
|
||||
const methodName = rpcMatch[1];
|
||||
const cid = contractId(pkg, serviceName, methodName);
|
||||
out.push(
|
||||
makeContract(cid, 'provider', filePath, `${serviceName}.${methodName}`, 0.85, {
|
||||
package: pkg,
|
||||
service: serviceName,
|
||||
method: methodName,
|
||||
source: 'proto',
|
||||
}),
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
return out;
|
||||
}
|
||||
```
|
||||
|
||||
Add this function before the class (e.g. after `serviceOnlyContractId`, around line 26):
|
||||
|
||||
```typescript
|
||||
function extractServiceBlocks(content: string): Array<{ name: string; body: string }> {
|
||||
const results: Array<{ name: string; body: string }> = [];
|
||||
const headerRe = /service\s+(\w+)\s*\{/g;
|
||||
let headerMatch: RegExpExecArray | null;
|
||||
|
||||
while ((headerMatch = headerRe.exec(content)) !== null) {
|
||||
const serviceName = headerMatch[1];
|
||||
const bodyStart = headerMatch.index + headerMatch[0].length;
|
||||
let depth = 1;
|
||||
let pos = bodyStart;
|
||||
|
||||
while (pos < content.length && depth > 0) {
|
||||
const ch = content[pos];
|
||||
if (ch === '{') depth++;
|
||||
else if (ch === '}') depth--;
|
||||
pos++;
|
||||
}
|
||||
|
||||
// If EOF before depth returns to 0, skip incomplete service
|
||||
if (depth !== 0) continue;
|
||||
|
||||
// body is between opening { (consumed by regex) and closing } (pos is one past it)
|
||||
const body = content.slice(bodyStart, pos - 1);
|
||||
results.push({ name: serviceName, body });
|
||||
}
|
||||
|
||||
return results;
|
||||
}
|
||||
```
|
||||
|
||||
- [ ] **Step 4: Run tests to verify they pass**
|
||||
|
||||
Run: `cd gitnexus && npx vitest run test/unit/group/grpc-extractor.test.ts`
|
||||
Expected: ALL PASS (including existing regression tests)
|
||||
|
||||
- [ ] **Step 5: Commit**
|
||||
|
||||
```bash
|
||||
cd gitnexus && git add src/core/group/extractors/grpc-extractor.ts test/unit/group/grpc-extractor.test.ts
|
||||
git commit -m "fix(group): replace gRPC proto regex with brace-depth counter
|
||||
|
||||
The serviceRe regex used [^}]* which stopped at the first '}'.
|
||||
Proto services with google.api.http annotations contain nested {}
|
||||
blocks, causing methods to be missed.
|
||||
|
||||
New extractServiceBlocks() uses a brace-depth counter (init depth=1
|
||||
after opening {, scan char-by-char). Malformed protos with unclosed
|
||||
braces are silently skipped.
|
||||
|
||||
Addresses PR #626 review item 2 (HIGH).
|
||||
|
||||
Co-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com>"
|
||||
```
|
||||
|
||||
---
|
||||
|
||||
### Task 5: Run Full Test Suite
|
||||
|
||||
- [ ] **Step 1: Run all group-related tests**
|
||||
|
||||
Run: `cd gitnexus && npx vitest run test/unit/group/ test/integration/group/`
|
||||
Expected: ALL PASS
|
||||
|
||||
- [ ] **Step 2: Run full test suite to catch regressions**
|
||||
|
||||
Run: `cd gitnexus && npx vitest run`
|
||||
Expected: ALL PASS, 0 failures
|
||||
|
||||
- [ ] **Step 3: Run typecheck**
|
||||
|
||||
Run: `cd gitnexus && npx tsc --noEmit`
|
||||
Expected: No errors
|
||||
|
||||
---
|
||||
|
||||
### Task 6: CLI Integration Smoke Test
|
||||
|
||||
- [ ] **Step 1: Add CLI smoke test for path traversal**
|
||||
|
||||
Add to `gitnexus/test/integration/group/group-cli.test.ts` inside the existing `group CLI` describe:
|
||||
|
||||
```typescript
|
||||
it('test_create_with_invalid_name_fails', () => {
|
||||
const result = runGroup(['create', '../../evil']);
|
||||
expect(result.status).not.toBe(0);
|
||||
expect(result.stderr).toContain('Invalid group name');
|
||||
});
|
||||
```
|
||||
|
||||
- [ ] **Step 2: Run test**
|
||||
|
||||
Run: `cd gitnexus && npx vitest run test/integration/group/group-cli.test.ts`
|
||||
Expected: ALL PASS
|
||||
|
||||
- [ ] **Step 3: Commit**
|
||||
|
||||
```bash
|
||||
cd gitnexus && git add test/integration/group/group-cli.test.ts
|
||||
git commit -m "test(group): add CLI smoke test for path traversal rejection
|
||||
|
||||
Verifies that 'group create ../../evil' fails with Invalid group name.
|
||||
|
||||
Co-Authored-By: Claude Opus 4.6 (1M context) <noreply@anthropic.com>"
|
||||
```
|
||||
@@ -0,0 +1,175 @@
|
||||
# PR #626 HIGH-Priority Fixes Design
|
||||
|
||||
**Date:** 2026-04-02
|
||||
**PR:** abhigyanpatwari/GitNexus#626 — Intra-repo service communication tracking
|
||||
**Scope:** 4 HIGH-priority issues identified by abhigyanpatwari and xkonjin
|
||||
**Approach:** Minimal targeted fixes (option A) — no refactoring, no scope creep
|
||||
|
||||
---
|
||||
|
||||
## Fix 1: Path Traversal via Group Name
|
||||
|
||||
**File:** `gitnexus/src/core/group/storage.ts`
|
||||
**Risk:** A group name like `../../etc` creates directories outside the intended path.
|
||||
|
||||
### Solution
|
||||
|
||||
Add `validateGroupName(name: string): void` that enforces `/^[a-zA-Z0-9][a-zA-Z0-9_-]*$/`.
|
||||
|
||||
- Call in `createGroupDir` (primary entry point)
|
||||
- Call in `getGroupDir` (defense in depth)
|
||||
- Throw descriptive error on invalid names
|
||||
|
||||
**Legacy:** Groups already on disk with names outside this pattern are not auto-renamed; only new `create` / resolved paths are validated.
|
||||
|
||||
### Why regex over path.resolve + startsWith
|
||||
|
||||
- abhigyanpatwari explicitly requested `[a-zA-Z0-9_-]`
|
||||
- Stricter: disallows spaces, dots, Unicode edge cases
|
||||
- Simpler to reason about
|
||||
|
||||
### Tests
|
||||
|
||||
- `../../evil` throws
|
||||
- `foo/bar` throws
|
||||
- Empty string throws
|
||||
- `my-group_01` passes
|
||||
- `A` (single char) passes
|
||||
- CLI smoke: one integration test that hits `getGroupDir` / `createGroupDir` (e.g. `group create` or `group add`) with an invalid name proves wiring for every subcommand that resolves a group through storage
|
||||
|
||||
### CLI/API entry points accepting groupName
|
||||
|
||||
All paths flow through `getGroupDir` (which validates), so coverage is implicit. For reference:
|
||||
|
||||
| Command | Entry | Calls |
|
||||
|---------|-------|-------|
|
||||
| `group create` | `cli/group.ts` action | `createGroupDir` -> `getGroupDir` |
|
||||
| `group add` | `cli/group.ts` action | `getGroupDir` |
|
||||
| `group remove` | `cli/group.ts` action | `getGroupDir` |
|
||||
| `group list` | `cli/group.ts` action | reads `groups/` dir directly — no traversal risk (reads, not writes) |
|
||||
| `group status` | `cli/group.ts` action | `getGroupDir` |
|
||||
| `group sync` | `cli/group.ts` action | `getGroupDir` |
|
||||
|
||||
**`listGroups`:** Reads directory names from disk without validation. Not a write path, so no traversal risk. May surface manually-created directories with non-conforming names — accepted as-is, not in scope.
|
||||
|
||||
---
|
||||
|
||||
## Fix 2: gRPC Proto Regex -> Brace-Depth Counter
|
||||
|
||||
**File:** `gitnexus/src/core/group/extractors/grpc-extractor.ts`
|
||||
**Risk:** `serviceRe = /service\s+(\w+)\s*\{([^}]*)}/gs` stops at first `}`. Proto services with `google.api.http` annotations inside RPCs contain nested `{ }` blocks.
|
||||
|
||||
### Solution
|
||||
|
||||
Replace `serviceRe` regex with `extractServiceBlocks(content: string): Array<{ name: string; body: string }>`:
|
||||
|
||||
1. Use regex only to find `service <Name> {` start positions (regex consumes the opening `{`)
|
||||
2. Initialise depth to 1 immediately after the opening `{`
|
||||
3. Scan forward char by char: `{` -> depth++, `}` -> depth--; collect into body
|
||||
4. Stop when depth reaches 0 (the matching closing `}`)
|
||||
5. Return name + body pairs
|
||||
|
||||
Inner `rpcRe` regex remains unchanged — it operates on the already-extracted body.
|
||||
|
||||
**Malformed input:** If EOF is reached before `depth` returns to 0, skip the incomplete service (do not add to results). Lock this in the test.
|
||||
|
||||
**Scope limitation (v1):** Brace-depth only — no lexer for string literals or comments containing `{`/`}`. Sufficient for `google.api.http` annotations. Known false positive: braces inside `//` comments or quoted strings within proto options. Accepted for v1; a proper proto lexer is out of scope.
|
||||
|
||||
### Tests
|
||||
|
||||
- Proto with single service, no nesting (regression)
|
||||
- Proto with `google.api.http` nested braces inside RPC options
|
||||
- Proto with multiple services
|
||||
- Proto with nested `option` blocks inside RPC (e.g. `google.api.http`)
|
||||
- Malformed proto with unclosed brace (graceful handling)
|
||||
|
||||
---
|
||||
|
||||
## Fix 3: Directory Exclusions in Service Boundary Detector
|
||||
|
||||
**File:** `gitnexus/src/core/group/service-boundary-detector.ts`
|
||||
**Risk:** Walks entire repo tree, only skipping dotfiles and `node_modules`. Extremely slow on repos with `vendor/`, `target/`, `__pycache__/`, `.venv/`.
|
||||
|
||||
### Solution
|
||||
|
||||
Create `EXCLUDED_DIRS` as a `Set<string>` (alongside existing `SERVICE_MARKERS`, `SOURCE_EXTENSIONS`), for example:
|
||||
|
||||
```text
|
||||
node_modules, vendor, target, build, dist,
|
||||
__pycache__, .venv, venv, .tox, .mypy_cache,
|
||||
.gradle, .mvn, out, bin
|
||||
```
|
||||
|
||||
(Implement as `new Set([...])` — the list above is the membership, not a string literal.)
|
||||
|
||||
Apply in both:
|
||||
- `walkForBoundaries` (line 77-78) — replace current inline `=== 'node_modules'` check with `EXCLUDED_DIRS.has(entry.name)`
|
||||
- `hasSourceFilesInSubdirs` (line 130) — replace `entry.name !== 'node_modules'` with `!EXCLUDED_DIRS.has(entry.name)`
|
||||
|
||||
Note: remove the old `=== 'node_modules'` literal from both locations — it is covered by `EXCLUDED_DIRS`.
|
||||
Dotfile exclusion (`.` prefix) remains as a separate check since it's a pattern, not a name.
|
||||
Exclusions apply only to `isDirectory()` entries — file names are never checked against `EXCLUDED_DIRS`.
|
||||
|
||||
**Tradeoff:** Rare layouts that keep source under names like `out/` or `bin/` will be skipped; accepted for performance on typical monorepos.
|
||||
|
||||
**Case sensitivity:** `Set.has` is case-sensitive (matches current `=== 'node_modules'` behavior). Windows case-insensitive FS not handled — accepted as-is, consistent with existing code.
|
||||
|
||||
### Tests
|
||||
|
||||
- Directory named `vendor/` is skipped
|
||||
- Directory named `target/` is skipped
|
||||
- Directory named `__pycache__/` is skipped
|
||||
- Regular source directories are NOT skipped
|
||||
- Dotfile directories still skipped (regression)
|
||||
|
||||
---
|
||||
|
||||
## Fix 4: Double-Close of LadybugDB Pools
|
||||
|
||||
**Files:**
|
||||
- `gitnexus/src/core/group/sync.ts` (lines 155-157) — per-id cleanup (KEEP)
|
||||
- `gitnexus/src/cli/group.ts` (line 188) — blanket `closeLbug()` (REMOVE)
|
||||
|
||||
**Risk:** In MCP server context, `closeLbug()` without arguments tears down ALL active pools, including ones from unrelated operations.
|
||||
|
||||
### Solution
|
||||
|
||||
Remove the `closeLbug()` call (no arguments) from `cli/group.ts` finally block. The per-id cleanup in `sync.ts` is sufficient:
|
||||
|
||||
```typescript
|
||||
// sync.ts — KEEP: cleans up only pools opened by this sync
|
||||
finally {
|
||||
for (const id of [...new Set(openPoolIds)]) {
|
||||
await closeLbug(id).catch(() => {});
|
||||
}
|
||||
}
|
||||
```
|
||||
|
||||
```typescript
|
||||
// cli/group.ts — REMOVE: blanket close that kills all pools
|
||||
finally {
|
||||
await closeLbug().catch(() => {}); // DELETE THIS
|
||||
}
|
||||
```
|
||||
|
||||
Remove the `closeLbug` import from `cli/group.ts` — after removing the `finally` call it has no remaining usages.
|
||||
|
||||
### Tests (unit level — mock pool adapter)
|
||||
|
||||
- `syncGroup` closes only the pools it opened (mock `closeLbug`, assert called with specific ids)
|
||||
- Two-pool scenario: sync opens pools A and B, both closed in finally; pool C (opened elsewhere) not touched
|
||||
- CLI `sync` command does not call blanket `closeLbug()` (verify no zero-arg call in source — static check or grep-based test)
|
||||
|
||||
---
|
||||
|
||||
## Out of Scope
|
||||
|
||||
- JSON -> LadybugDB migration (tracked in #606)
|
||||
- MEDIUM/LOW issues (items 5-10 from review summary)
|
||||
- Test gap coverage beyond what's needed for these 4 fixes
|
||||
- Any refactoring or architectural changes
|
||||
|
||||
## Execution Order
|
||||
|
||||
Fixes are independent — can be implemented in parallel or any order.
|
||||
Recommended order for review clarity: 1 -> 3 -> 4 -> 2 (simplest to most complex).
|
||||
@@ -97,7 +97,8 @@ export type RelationshipType =
|
||||
| 'CONTAINS'
|
||||
| 'CALLS'
|
||||
| 'INHERITS'
|
||||
| 'OVERRIDES'
|
||||
| 'METHOD_OVERRIDES'
|
||||
| 'METHOD_IMPLEMENTS'
|
||||
| 'IMPORTS'
|
||||
| 'USES'
|
||||
| 'DEFINES'
|
||||
|
||||
@@ -19,6 +19,7 @@ export type { NodeTableName, RelType } from './lbug/schema-constants.js';
|
||||
// Language support
|
||||
export { SupportedLanguages } from './languages.js';
|
||||
export { getLanguageFromFilename, getSyntaxLanguageFromFilename } from './language-detection.js';
|
||||
export type { MroStrategy } from './mro-strategy.js';
|
||||
|
||||
// Pipeline progress
|
||||
export type { PipelinePhase, PipelineProgress } from './pipeline.js';
|
||||
|
||||
@@ -41,6 +41,7 @@ const EXTENSION_MAP: Record<SupportedLanguages, readonly string[]> = {
|
||||
[SupportedLanguages.Kotlin]: ['.kt', '.kts'],
|
||||
[SupportedLanguages.Swift]: ['.swift'],
|
||||
[SupportedLanguages.Dart]: ['.dart'],
|
||||
[SupportedLanguages.Vue]: ['.vue'],
|
||||
[SupportedLanguages.Cobol]: ['.cbl', '.cob', '.cpy', '.cobol'],
|
||||
} satisfies Record<SupportedLanguages, readonly string[]>; // Ensure exhaustiveness
|
||||
|
||||
@@ -98,6 +99,7 @@ const SYNTAX_MAP: Record<SupportedLanguages, string> = {
|
||||
[SupportedLanguages.Kotlin]: 'kotlin',
|
||||
[SupportedLanguages.Swift]: 'swift',
|
||||
[SupportedLanguages.Dart]: 'dart',
|
||||
[SupportedLanguages.Vue]: 'typescript',
|
||||
[SupportedLanguages.Cobol]: 'cobol',
|
||||
} satisfies Record<SupportedLanguages, string>; // Ensure exhaustiveness
|
||||
|
||||
|
||||
@@ -19,6 +19,7 @@ export enum SupportedLanguages {
|
||||
Kotlin = 'kotlin',
|
||||
Swift = 'swift',
|
||||
Dart = 'dart',
|
||||
Vue = 'vue',
|
||||
/** Standalone regex processor — no tree-sitter, no LanguageProvider. */
|
||||
Cobol = 'cobol',
|
||||
}
|
||||
|
||||
@@ -55,7 +55,9 @@ export const REL_TYPES = [
|
||||
'HAS_METHOD',
|
||||
'HAS_PROPERTY',
|
||||
'ACCESSES',
|
||||
'OVERRIDES',
|
||||
'METHOD_OVERRIDES',
|
||||
'OVERRIDES', // Legacy compat alias — kept until all stored indexes are migrated
|
||||
'METHOD_IMPLEMENTS',
|
||||
'MEMBER_OF',
|
||||
'STEP_IN_PROCESS',
|
||||
'HANDLES_ROUTE',
|
||||
|
||||
@@ -0,0 +1,23 @@
|
||||
/**
|
||||
* MRO (Method Resolution Order) strategy — shared between CLI and any
|
||||
* future consumer that reasons about multiple-inheritance semantics.
|
||||
*
|
||||
* Lives in `gitnexus-shared` so the low-level resolution module
|
||||
* (`core/ingestion/model/resolve.ts`) does not need to import from
|
||||
* `languages/` — keeping the `model/` layer free of language-registry
|
||||
* coupling.
|
||||
*
|
||||
* Strategy semantics:
|
||||
* - `first-wins`: BFS ancestor walk, first match wins (default).
|
||||
* - `leftmost-base`: BFS ancestor walk, leftmost base wins (C++).
|
||||
* - `c3`: C3-linearized ancestor order, first match wins (Python).
|
||||
* - `implements-split`: BFS walk, first match wins (Java/C#/Kotlin) — full
|
||||
* interface-default ambiguity is handled at graph level.
|
||||
* - `qualified-syntax`: No auto-resolution (Rust — requires `<T as Trait>::m`).
|
||||
*/
|
||||
export type MroStrategy =
|
||||
| 'first-wins'
|
||||
| 'c3'
|
||||
| 'leftmost-base'
|
||||
| 'implements-split'
|
||||
| 'qualified-syntax';
|
||||
@@ -0,0 +1,109 @@
|
||||
import { test, expect } from '@playwright/test';
|
||||
|
||||
/**
|
||||
* E2E tests for heartbeat disconnect/reconnect behavior.
|
||||
*
|
||||
* Verifies the key regression: when the heartbeat fails, the UI shows a
|
||||
* "reconnecting" banner instead of resetting to the onboarding screen.
|
||||
*
|
||||
* Strategy: block /api/heartbeat via route interception BEFORE loading the
|
||||
* graph. The heartbeat EventSource can never connect, so onReconnecting
|
||||
* fires on the first retry attempt. This reliably tests the banner behavior
|
||||
* without depending on setOffline timing (which varies across CI environments).
|
||||
*/
|
||||
|
||||
const BACKEND_URL = process.env.BACKEND_URL ?? 'http://localhost:4747';
|
||||
const FRONTEND_URL = process.env.FRONTEND_URL ?? 'http://localhost:5173';
|
||||
|
||||
test.beforeAll(async () => {
|
||||
if (process.env.E2E) return;
|
||||
try {
|
||||
const [backendRes, frontendRes] = await Promise.allSettled([
|
||||
fetch(`${BACKEND_URL}/api/repos`),
|
||||
fetch(FRONTEND_URL),
|
||||
]);
|
||||
if (
|
||||
backendRes.status === 'rejected' ||
|
||||
(backendRes.status === 'fulfilled' && !backendRes.value.ok)
|
||||
) {
|
||||
test.skip(true, 'gitnexus serve not available');
|
||||
return;
|
||||
}
|
||||
if (
|
||||
frontendRes.status === 'rejected' ||
|
||||
(frontendRes.status === 'fulfilled' && !frontendRes.value.ok)
|
||||
) {
|
||||
test.skip(true, 'Vite dev server not available');
|
||||
return;
|
||||
}
|
||||
if (backendRes.status === 'fulfilled') {
|
||||
const repos = await backendRes.value.json();
|
||||
if (!repos.length) {
|
||||
test.skip(true, 'No indexed repos');
|
||||
return;
|
||||
}
|
||||
}
|
||||
} catch {
|
||||
test.skip(true, 'servers not available');
|
||||
}
|
||||
});
|
||||
|
||||
test.describe('Heartbeat Reconnect', () => {
|
||||
test('shows reconnecting banner instead of onboarding reset when heartbeat is unavailable', async ({
|
||||
page,
|
||||
}) => {
|
||||
// Block the heartbeat BEFORE navigating — the EventSource will fail
|
||||
// immediately on every connection attempt, triggering onReconnecting.
|
||||
await page.route('**/api/heartbeat', (route) => route.abort('connectionrefused'));
|
||||
|
||||
// Load the app and connect to a repo (all other endpoints work normally)
|
||||
await page.goto('/');
|
||||
|
||||
const landingCard = page.locator('[data-testid="landing-repo-card"]').first();
|
||||
try {
|
||||
await landingCard.waitFor({ state: 'visible', timeout: 15_000 });
|
||||
await landingCard.click();
|
||||
} catch {
|
||||
// auto-connect may skip the landing screen
|
||||
}
|
||||
|
||||
// Wait for graph to load (heartbeat is blocked, but graph loads fine)
|
||||
await expect(page.locator('[data-testid="status-ready"]')).toBeVisible({ timeout: 30_000 });
|
||||
|
||||
// The reconnecting banner should appear (heartbeat is failing)
|
||||
const banner = page.getByText('Server connection lost');
|
||||
await expect(banner).toBeVisible({ timeout: 15_000 });
|
||||
|
||||
// The graph canvas should STILL be visible — NOT reset to onboarding
|
||||
await expect(page.locator('canvas').first()).toBeVisible();
|
||||
});
|
||||
|
||||
test('banner clears when heartbeat becomes available', async ({ page }) => {
|
||||
// Start with heartbeat blocked
|
||||
await page.route('**/api/heartbeat', (route) => route.abort('connectionrefused'));
|
||||
|
||||
await page.goto('/');
|
||||
const landingCard = page.locator('[data-testid="landing-repo-card"]').first();
|
||||
try {
|
||||
await landingCard.waitFor({ state: 'visible', timeout: 15_000 });
|
||||
await landingCard.click();
|
||||
} catch {
|
||||
// auto-connect may skip the landing screen
|
||||
}
|
||||
|
||||
await expect(page.locator('[data-testid="status-ready"]')).toBeVisible({ timeout: 30_000 });
|
||||
|
||||
// Verify banner appears
|
||||
const banner = page.getByText('Server connection lost');
|
||||
await expect(banner).toBeVisible({ timeout: 15_000 });
|
||||
|
||||
// Unblock heartbeat — the real server is running, so reconnect will succeed
|
||||
await page.unroute('**/api/heartbeat');
|
||||
|
||||
// Banner should disappear as heartbeat reconnects
|
||||
await expect(banner).not.toBeVisible({ timeout: 30_000 });
|
||||
|
||||
// Graph should still be there
|
||||
await expect(page.locator('[data-testid="status-ready"]')).toBeVisible();
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,112 @@
|
||||
import { test, expect } from '@playwright/test';
|
||||
|
||||
/**
|
||||
* E2E tests for multi-repo scoping and URL persistence.
|
||||
*
|
||||
* Verifies that:
|
||||
* - Connecting via ?server= loads data and sets ?project= in the URL
|
||||
* - The repo name appears in the UI after connecting
|
||||
* - F5 with ?server=&project= reconnects to the correct repo
|
||||
*
|
||||
* Runs against the single indexed repo in CI — validates the plumbing
|
||||
* works end-to-end even with one repo.
|
||||
*/
|
||||
|
||||
const BACKEND_URL = process.env.BACKEND_URL ?? 'http://localhost:4747';
|
||||
const FRONTEND_URL = process.env.FRONTEND_URL ?? 'http://localhost:5173';
|
||||
|
||||
let firstRepoName: string;
|
||||
|
||||
test.beforeAll(async () => {
|
||||
if (process.env.E2E) {
|
||||
// Still need to fetch the repo name for assertions
|
||||
try {
|
||||
const res = await fetch(`${BACKEND_URL}/api/repos`);
|
||||
const repos = await res.json();
|
||||
firstRepoName = repos[0]?.name ?? '';
|
||||
} catch {
|
||||
firstRepoName = '';
|
||||
}
|
||||
return;
|
||||
}
|
||||
try {
|
||||
const [backendRes, frontendRes] = await Promise.allSettled([
|
||||
fetch(`${BACKEND_URL}/api/repos`),
|
||||
fetch(FRONTEND_URL),
|
||||
]);
|
||||
if (
|
||||
backendRes.status === 'rejected' ||
|
||||
(backendRes.status === 'fulfilled' && !backendRes.value.ok)
|
||||
) {
|
||||
test.skip(true, 'gitnexus serve not available');
|
||||
return;
|
||||
}
|
||||
if (
|
||||
frontendRes.status === 'rejected' ||
|
||||
(frontendRes.status === 'fulfilled' && !frontendRes.value.ok)
|
||||
) {
|
||||
test.skip(true, 'Vite dev server not available');
|
||||
return;
|
||||
}
|
||||
if (backendRes.status === 'fulfilled') {
|
||||
const repos = await backendRes.value.json();
|
||||
if (!repos.length) {
|
||||
test.skip(true, 'No indexed repos');
|
||||
return;
|
||||
}
|
||||
firstRepoName = repos[0].name;
|
||||
}
|
||||
} catch {
|
||||
test.skip(true, 'servers not available');
|
||||
}
|
||||
});
|
||||
|
||||
test.describe('Multi-Repo Scoping', () => {
|
||||
test('auto-connect via ?server= sets ?project= in URL', async ({ page }) => {
|
||||
// Navigate with ?server= param (the bookmarkable shortcut)
|
||||
await page.goto(`/?server=${encodeURIComponent(BACKEND_URL)}`);
|
||||
|
||||
// Wait for graph to load
|
||||
await expect(page.locator('[data-testid="status-ready"]')).toBeVisible({ timeout: 30_000 });
|
||||
|
||||
// URL should now contain ?project= with the repo name
|
||||
const url = new URL(page.url());
|
||||
const project = url.searchParams.get('project');
|
||||
expect(project).toBeTruthy();
|
||||
expect(project).toBe(firstRepoName);
|
||||
});
|
||||
|
||||
test('?server= is preserved in URL for F5 recovery', async ({ page }) => {
|
||||
await page.goto(`/?server=${encodeURIComponent(BACKEND_URL)}`);
|
||||
await expect(page.locator('[data-testid="status-ready"]')).toBeVisible({ timeout: 30_000 });
|
||||
|
||||
// URL should still have ?server=
|
||||
const url = new URL(page.url());
|
||||
expect(url.searchParams.get('server')).toBeTruthy();
|
||||
|
||||
// F5 should reconnect (not show onboarding)
|
||||
await page.reload();
|
||||
await expect(page.locator('[data-testid="status-ready"]')).toBeVisible({ timeout: 30_000 });
|
||||
});
|
||||
|
||||
test('node count in status bar matches backend data', async ({ page }) => {
|
||||
await page.goto(`/?server=${encodeURIComponent(BACKEND_URL)}`);
|
||||
await expect(page.locator('[data-testid="status-ready"]')).toBeVisible({ timeout: 30_000 });
|
||||
|
||||
// Fetch expected node count from backend
|
||||
const res = await fetch(`${BACKEND_URL}/api/repo?repo=${encodeURIComponent(firstRepoName)}`);
|
||||
const repoInfo = await res.json();
|
||||
const expectedNodes = repoInfo.stats?.nodes;
|
||||
|
||||
if (expectedNodes) {
|
||||
// Status bar shows node count — use the status-ready area to avoid
|
||||
// matching multiple elements (file tree, header may also show counts)
|
||||
const statusBar = page.locator('footer');
|
||||
const nodeText = statusBar.getByText(/\d+ nodes/).first();
|
||||
await expect(nodeText).toBeVisible({ timeout: 10_000 });
|
||||
const text = await nodeText.textContent();
|
||||
const displayedNodes = parseInt(text?.match(/(\d+)\s*nodes/)?.[1] ?? '0', 10);
|
||||
expect(displayedNodes).toBeGreaterThan(0);
|
||||
}
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,168 @@
|
||||
import { test, expect } from '@playwright/test';
|
||||
|
||||
/**
|
||||
* E2E tests for the repo-switching and false-404 fixes.
|
||||
*
|
||||
* Most tests use the live backend (same pattern as multi-repo-scoping.spec.ts).
|
||||
* The 503 hold-queue test uses route interception to simulate a slow analysis.
|
||||
*/
|
||||
|
||||
const BACKEND_URL = process.env.BACKEND_URL ?? 'http://localhost:4747';
|
||||
const FRONTEND_URL = process.env.FRONTEND_URL ?? 'http://localhost:5173';
|
||||
|
||||
let firstRepoName: string;
|
||||
|
||||
test.beforeAll(async () => {
|
||||
if (process.env.E2E) {
|
||||
try {
|
||||
const res = await fetch(`${BACKEND_URL}/api/repos`);
|
||||
const repos = await res.json();
|
||||
firstRepoName = repos[0]?.name ?? '';
|
||||
} catch {
|
||||
firstRepoName = '';
|
||||
}
|
||||
return;
|
||||
}
|
||||
try {
|
||||
const [backendRes, frontendRes] = await Promise.allSettled([
|
||||
fetch(`${BACKEND_URL}/api/repos`),
|
||||
fetch(FRONTEND_URL),
|
||||
]);
|
||||
if (
|
||||
backendRes.status === 'rejected' ||
|
||||
(backendRes.status === 'fulfilled' && !backendRes.value.ok)
|
||||
) {
|
||||
test.skip(true, 'gitnexus serve not available');
|
||||
return;
|
||||
}
|
||||
if (
|
||||
frontendRes.status === 'rejected' ||
|
||||
(frontendRes.status === 'fulfilled' && !frontendRes.value.ok)
|
||||
) {
|
||||
test.skip(true, 'Vite dev server not available');
|
||||
return;
|
||||
}
|
||||
if (backendRes.status === 'fulfilled') {
|
||||
const repos = await backendRes.value.json();
|
||||
if (!repos.length) {
|
||||
test.skip(true, 'No indexed repos');
|
||||
return;
|
||||
}
|
||||
firstRepoName = repos[0].name;
|
||||
}
|
||||
} catch {
|
||||
test.skip(true, 'servers not available');
|
||||
}
|
||||
});
|
||||
|
||||
// ── 1. Hold-queue: 503 → descriptive user message ────────────────────────────
|
||||
|
||||
test.describe('Hold-queue timeout error', () => {
|
||||
test('shows descriptive message when /api/repo returns 503', async ({ page }, testInfo) => {
|
||||
// Intercept only /api/repo (singular) — not /api/repos — to return a 503
|
||||
// regex: /api/repo followed by end, ?, or # — NOT /api/repos
|
||||
await page.route(/\/api\/repo(?!s)(\?.*)?$/, (route) =>
|
||||
route.fulfill({
|
||||
status: 503,
|
||||
contentType: 'application/json',
|
||||
body: JSON.stringify({
|
||||
error: `Repository analysis for "${firstRepoName}" is taking longer than expected. Please try again in a moment.`,
|
||||
}),
|
||||
}),
|
||||
);
|
||||
|
||||
await page.goto(`/?server=${encodeURIComponent(BACKEND_URL)}`);
|
||||
|
||||
// UI should show the 503 error message
|
||||
await expect(page.getByText(/taking longer than expected/i)).toBeVisible({
|
||||
timeout: 20_000,
|
||||
});
|
||||
|
||||
await page.screenshot({ path: testInfo.outputPath('hold-queue-503.png') });
|
||||
});
|
||||
});
|
||||
|
||||
// ── 2. ?project= URL persistence ─────────────────────────────────────────────
|
||||
|
||||
test.describe('?project= URL persistence', () => {
|
||||
test('?project= is set in URL after connecting via ?server=', async ({ page }) => {
|
||||
await page.goto(`/?server=${encodeURIComponent(BACKEND_URL)}`);
|
||||
|
||||
await expect(page.locator('[data-testid="status-ready"]')).toBeVisible({ timeout: 30_000 });
|
||||
|
||||
const url = new URL(page.url());
|
||||
const project = url.searchParams.get('project');
|
||||
expect(project).toBeTruthy();
|
||||
// first repo returned by the live backend
|
||||
if (firstRepoName) expect(project).toBe(firstRepoName);
|
||||
});
|
||||
|
||||
test('?project= is still present after F5 reload', async ({ page }) => {
|
||||
await page.goto(`/?server=${encodeURIComponent(BACKEND_URL)}`);
|
||||
await expect(page.locator('[data-testid="status-ready"]')).toBeVisible({ timeout: 30_000 });
|
||||
|
||||
// After connect, URL has ?server=&project= — F5 re-uses both params
|
||||
await page.reload();
|
||||
await expect(page.locator('[data-testid="status-ready"]')).toBeVisible({ timeout: 30_000 });
|
||||
|
||||
const url = new URL(page.url());
|
||||
expect(url.searchParams.get('project')).toBeTruthy();
|
||||
});
|
||||
});
|
||||
|
||||
// ── 3. ?project= + ?server= combined auto-connect ────────────────────────────
|
||||
|
||||
test.describe('?project= auto-connect', () => {
|
||||
test('navigating with ?server=&project= connects to the correct repo', async ({
|
||||
page,
|
||||
}, testInfo) => {
|
||||
if (!firstRepoName) test.skip(true, 'no repo name available');
|
||||
|
||||
await page.goto(
|
||||
`/?server=${encodeURIComponent(BACKEND_URL)}&project=${encodeURIComponent(firstRepoName)}`,
|
||||
);
|
||||
|
||||
await expect(page.locator('[data-testid="status-ready"]')).toBeVisible({ timeout: 30_000 });
|
||||
|
||||
// ?project= in URL should match what we passed in
|
||||
const url = new URL(page.url());
|
||||
expect(url.searchParams.get('project')).toBe(firstRepoName);
|
||||
|
||||
await page.screenshot({ path: testInfo.outputPath('project-param-connect.png') });
|
||||
});
|
||||
});
|
||||
|
||||
// ── 4. Windows path normalization ─────────────────────────────────────────────
|
||||
|
||||
test.describe('Windows path normalization', () => {
|
||||
test('project name uses basename when /api/repo returns a Windows-style repoPath', async ({
|
||||
page,
|
||||
}) => {
|
||||
const repoName = firstRepoName || 'test-repo';
|
||||
const windowsPath = `C:\\Users\\LENOVO\\.gitnexus\\repos\\${repoName}`;
|
||||
|
||||
// Mock /api/repo to return a Windows backslash path while keeping name correct
|
||||
await page.route(/\/api\/repo(?!s)(\?.*)?$/, (route) =>
|
||||
route.fulfill({
|
||||
contentType: 'application/json',
|
||||
body: JSON.stringify({
|
||||
// intentionally omit `name` to force path-based extraction
|
||||
path: windowsPath,
|
||||
repoPath: windowsPath,
|
||||
}),
|
||||
}),
|
||||
);
|
||||
|
||||
await page.goto(`/?server=${encodeURIComponent(BACKEND_URL)}`);
|
||||
|
||||
await expect(page.locator('[data-testid="status-ready"]')).toBeVisible({ timeout: 30_000 });
|
||||
|
||||
// URL ?project= must be the short basename, NOT the full Windows path
|
||||
const url = new URL(page.url());
|
||||
const project = url.searchParams.get('project');
|
||||
expect(project).toBeTruthy();
|
||||
expect(project).not.toContain('\\');
|
||||
expect(project).not.toContain('LENOVO');
|
||||
expect(project).toBe(repoName);
|
||||
});
|
||||
});
|
||||
@@ -1,4 +1,4 @@
|
||||
import { test, expect, type TestInfo } from '@playwright/test';
|
||||
import { test, expect } from '@playwright/test';
|
||||
|
||||
/**
|
||||
* E2E tests for the GitNexus web UI — exploring view features.
|
||||
@@ -58,36 +58,41 @@ test.beforeAll(async () => {
|
||||
* For these tests we require at least one indexed repo, so pick the first
|
||||
* landing card when present and then wait for the exploring view.
|
||||
*/
|
||||
async function waitForGraphLoaded(page: import('@playwright/test').Page, testInfo: TestInfo) {
|
||||
async function waitForGraphLoaded(page: import('@playwright/test').Page) {
|
||||
await page.goto('/');
|
||||
|
||||
const landingCard = page.locator('[data-testid="landing-repo-card"]').first();
|
||||
const landingCards = page.locator('[data-testid="landing-repo-card"]');
|
||||
const preferredLandingCard = landingCards
|
||||
.filter({ hasText: /GitNexus|local-integration/ })
|
||||
.first();
|
||||
try {
|
||||
await landingCard.waitFor({ state: 'visible', timeout: 15_000 });
|
||||
await landingCards.first().waitFor({ state: 'visible', timeout: 15_000 });
|
||||
const landingCard =
|
||||
(await preferredLandingCard.count()) > 0 ? preferredLandingCard : landingCards.first();
|
||||
await landingCard.click();
|
||||
} catch {
|
||||
// Landing screen may not appear (e.g. ?server auto-connect)
|
||||
}
|
||||
|
||||
await expect(page.locator('[data-testid="status-ready"]')).toBeVisible({ timeout: 30_000 });
|
||||
await expect(page.getByText(/\d+ nodes/).first()).toBeVisible();
|
||||
await page.screenshot({ path: testInfo.outputPath('graph-loaded.png') });
|
||||
const statusBar = page.getByRole('contentinfo');
|
||||
await expect(statusBar.getByText('Ready', { exact: true })).toBeVisible({ timeout: 45_000 });
|
||||
await expect(statusBar).toContainText(/nodes/, {
|
||||
timeout: 20_000,
|
||||
});
|
||||
}
|
||||
|
||||
test.describe('Server Connection & Graph Loading', () => {
|
||||
test('selects a repo from landing and loads graph', async ({ page }, testInfo) => {
|
||||
await waitForGraphLoaded(page, testInfo);
|
||||
await page.screenshot({ path: testInfo.outputPath('graph-loaded-full.png'), fullPage: true });
|
||||
test('selects a repo from landing and loads graph', async ({ page }) => {
|
||||
await waitForGraphLoaded(page);
|
||||
});
|
||||
});
|
||||
|
||||
test.describe('Nexus AI', () => {
|
||||
test('panel opens and agent initializes without error', async ({ page }, testInfo) => {
|
||||
await waitForGraphLoaded(page, testInfo);
|
||||
test('panel opens and agent initializes without error', async ({ page }) => {
|
||||
await waitForGraphLoaded(page);
|
||||
|
||||
await page.getByRole('button', { name: 'Nexus AI' }).click();
|
||||
await expect(page.getByText('Ask me anything')).toBeVisible({ timeout: 15_000 });
|
||||
await page.screenshot({ path: testInfo.outputPath('nexus-ai-panel.png'), fullPage: true });
|
||||
|
||||
const errorBanner = page.getByText('Database not ready');
|
||||
expect(await errorBanner.isVisible().catch(() => false)).toBe(false);
|
||||
@@ -95,8 +100,8 @@ test.describe('Nexus AI', () => {
|
||||
});
|
||||
|
||||
test.describe('Processes Panel', () => {
|
||||
test('shows process list and View button works', async ({ page }, testInfo) => {
|
||||
await waitForGraphLoaded(page, testInfo);
|
||||
test('shows process list and View button works', async ({ page }) => {
|
||||
await waitForGraphLoaded(page);
|
||||
|
||||
await page.getByRole('button', { name: 'Nexus AI' }).click();
|
||||
await page.getByText('Processes').click();
|
||||
@@ -104,7 +109,6 @@ test.describe('Processes Panel', () => {
|
||||
await expect(page.locator('[data-testid="process-list-loaded"]')).toBeVisible({
|
||||
timeout: 15_000,
|
||||
});
|
||||
await page.screenshot({ path: testInfo.outputPath('processes-panel.png'), fullPage: true });
|
||||
|
||||
const processRow = page.locator('[data-testid="process-row"]').first();
|
||||
await expect(processRow).toBeVisible({ timeout: 10_000 });
|
||||
@@ -114,14 +118,10 @@ test.describe('Processes Panel', () => {
|
||||
await viewBtn.waitFor({ state: 'visible', timeout: 5_000 });
|
||||
await viewBtn.click();
|
||||
await expect(page.locator('[data-testid="process-modal"]')).toBeVisible({ timeout: 5_000 });
|
||||
await page.screenshot({
|
||||
path: testInfo.outputPath('process-view-clicked.png'),
|
||||
fullPage: true,
|
||||
});
|
||||
});
|
||||
|
||||
test('lightbulb highlights nodes in graph', async ({ page }, testInfo) => {
|
||||
await waitForGraphLoaded(page, testInfo);
|
||||
test('lightbulb highlights nodes in graph', async ({ page }) => {
|
||||
await waitForGraphLoaded(page);
|
||||
|
||||
await page.getByRole('button', { name: 'Nexus AI' }).click();
|
||||
await page.getByText('Processes').click();
|
||||
@@ -137,13 +137,12 @@ test.describe('Processes Panel', () => {
|
||||
await lightbulb.waitFor({ state: 'visible', timeout: 5_000 });
|
||||
await lightbulb.click();
|
||||
await expect(processRow).toHaveClass(/bg-amber-950/, { timeout: 5_000 });
|
||||
await page.screenshot({ path: testInfo.outputPath('after-highlight.png'), fullPage: true });
|
||||
});
|
||||
});
|
||||
|
||||
test.describe('Turn Off All Highlights', () => {
|
||||
test('selecting a node dims others, button clears it', async ({ page }, testInfo) => {
|
||||
await waitForGraphLoaded(page, testInfo);
|
||||
test('selecting a node dims others, button clears it', async ({ page }) => {
|
||||
await waitForGraphLoaded(page);
|
||||
|
||||
await expect(page.locator('canvas').first()).toBeVisible({ timeout: 10_000 });
|
||||
|
||||
@@ -160,6 +159,5 @@ test.describe('Turn Off All Highlights', () => {
|
||||
await expect(highlightToggle).toHaveAttribute('title', 'Turn on AI highlights', {
|
||||
timeout: 5_000,
|
||||
});
|
||||
await page.screenshot({ path: testInfo.outputPath('highlights-cleared.png'), fullPage: true });
|
||||
});
|
||||
});
|
||||
|
||||
+84
-49
@@ -1,4 +1,4 @@
|
||||
import { useCallback, useEffect, useRef } from 'react';
|
||||
import { useCallback, useEffect, useRef, useState } from 'react';
|
||||
import { AppStateProvider, useAppState } from './hooks/useAppState';
|
||||
import { DropZone } from './components/DropZone';
|
||||
import { LoadingOverlay } from './components/LoadingOverlay';
|
||||
@@ -36,7 +36,6 @@ const AppContent = () => {
|
||||
refreshLLMSettings,
|
||||
initializeAgent,
|
||||
startEmbeddingsWithFallback,
|
||||
embeddingStatus,
|
||||
codeReferences,
|
||||
selectedNode,
|
||||
isCodePanelOpen,
|
||||
@@ -45,17 +44,25 @@ const AppContent = () => {
|
||||
availableRepos,
|
||||
setAvailableRepos,
|
||||
switchRepo,
|
||||
setCurrentRepo,
|
||||
} = useAppState();
|
||||
|
||||
const graphCanvasRef = useRef<GraphCanvasHandle>(null);
|
||||
const [serverDisconnected, setServerDisconnected] = useState(false);
|
||||
|
||||
const handleServerConnect = useCallback(
|
||||
async (result: ConnectResult): Promise<void> => {
|
||||
// Extract project name from repoPath
|
||||
// Use the canonical repo name from the server response so all subsequent
|
||||
// backend calls (queries, search, grep, readFile) scope to this repo.
|
||||
const repoName = result.repoInfo.name;
|
||||
const repoPath = result.repoInfo.repoPath ?? result.repoInfo.path;
|
||||
const parts = (repoPath || '').split('/').filter((p) => p && !p.startsWith('.'));
|
||||
const projectName = parts[parts.length - 1] || parts[0] || 'server-project';
|
||||
// Normalize both Windows (\) and Unix (/) path separators before splitting
|
||||
const projectName =
|
||||
result.repoInfo.name ||
|
||||
(repoPath || '').replace(/\\/g, '/').split('/').filter(Boolean).pop() ||
|
||||
'server-project';
|
||||
setProjectName(projectName);
|
||||
setCurrentRepo(projectName);
|
||||
|
||||
// Build KnowledgeGraph from server data for visualization
|
||||
const graph = createKnowledgeGraph();
|
||||
@@ -67,6 +74,11 @@ const AppContent = () => {
|
||||
}
|
||||
setGraph(graph);
|
||||
|
||||
// Persist the active project in the URL for bookmarkability and F5 refresh resilience
|
||||
const urlObj = new URL(window.location.href);
|
||||
urlObj.searchParams.set('project', projectName);
|
||||
window.history.replaceState(null, '', urlObj.toString());
|
||||
|
||||
// Transition directly to exploring view
|
||||
setViewMode('exploring');
|
||||
|
||||
@@ -80,20 +92,26 @@ const AppContent = () => {
|
||||
console.warn('Failed to initialize agent:', err);
|
||||
}
|
||||
},
|
||||
[setViewMode, setGraph, setProjectName, initializeAgent, startEmbeddingsWithFallback],
|
||||
[
|
||||
setViewMode,
|
||||
setGraph,
|
||||
setProjectName,
|
||||
setCurrentRepo,
|
||||
initializeAgent,
|
||||
startEmbeddingsWithFallback,
|
||||
],
|
||||
);
|
||||
|
||||
// Auto-connect when ?server query param is present (bookmarkable shortcut)
|
||||
// Auto-connect when ?server or ?project query param is present (bookmarkable shortcut)
|
||||
const autoConnectRan = useRef(false);
|
||||
useEffect(() => {
|
||||
if (autoConnectRan.current) return;
|
||||
const params = new URLSearchParams(window.location.search);
|
||||
if (!params.has('server')) return;
|
||||
autoConnectRan.current = true;
|
||||
const serverUrlParam = params.get('server');
|
||||
const projectParam = params.get('project');
|
||||
|
||||
// Clean the URL so a refresh won't re-trigger
|
||||
const cleanUrl = window.location.pathname + window.location.hash;
|
||||
window.history.replaceState(null, '', cleanUrl);
|
||||
if (!serverUrlParam && !projectParam) return;
|
||||
autoConnectRan.current = true;
|
||||
|
||||
setProgress({
|
||||
phase: 'extracting',
|
||||
@@ -103,36 +121,45 @@ const AppContent = () => {
|
||||
});
|
||||
setViewMode('loading');
|
||||
|
||||
const serverUrl = params.get('server') || window.location.origin;
|
||||
|
||||
const serverUrl = serverUrlParam || window.location.origin;
|
||||
const baseUrl = normalizeServerUrl(serverUrl);
|
||||
|
||||
connectToServer(serverUrl, (phase, downloaded, total) => {
|
||||
if (phase === 'validating') {
|
||||
setProgress({
|
||||
phase: 'extracting',
|
||||
percent: 5,
|
||||
message: 'Connecting to server...',
|
||||
detail: 'Validating server',
|
||||
});
|
||||
} else if (phase === 'downloading') {
|
||||
const pct = total ? Math.round((downloaded / total) * 90) + 5 : 50;
|
||||
const mb = (downloaded / (1024 * 1024)).toFixed(1);
|
||||
setProgress({
|
||||
phase: 'extracting',
|
||||
percent: pct,
|
||||
message: 'Downloading graph...',
|
||||
detail: `${mb} MB downloaded`,
|
||||
});
|
||||
} else if (phase === 'extracting') {
|
||||
setProgress({
|
||||
phase: 'extracting',
|
||||
percent: 97,
|
||||
message: 'Processing...',
|
||||
detail: 'Extracting file contents',
|
||||
});
|
||||
}
|
||||
})
|
||||
const tryConnect = async () => {
|
||||
return await connectToServer(
|
||||
serverUrl,
|
||||
(phase, downloaded, total) => {
|
||||
if (phase === 'validating') {
|
||||
setProgress({
|
||||
phase: 'extracting',
|
||||
percent: 5,
|
||||
message: 'Connecting to server...',
|
||||
detail: 'Validating server',
|
||||
});
|
||||
} else if (phase === 'downloading') {
|
||||
const pct = total ? Math.round((downloaded / total) * 90) + 5 : 50;
|
||||
const mb = (downloaded / (1024 * 1024)).toFixed(1);
|
||||
setProgress({
|
||||
phase: 'extracting',
|
||||
percent: pct,
|
||||
message: 'Downloading graph...',
|
||||
detail: `${mb} MB downloaded`,
|
||||
});
|
||||
} else if (phase === 'extracting') {
|
||||
setProgress({
|
||||
phase: 'extracting',
|
||||
percent: 97,
|
||||
message: 'Processing...',
|
||||
detail: 'Extracting file contents',
|
||||
});
|
||||
}
|
||||
},
|
||||
undefined,
|
||||
projectParam || undefined,
|
||||
{ awaitAnalysis: true }, // enable backend hold-queue for repos still being analyzed
|
||||
);
|
||||
};
|
||||
|
||||
tryConnect()
|
||||
.then(async (result) => {
|
||||
await handleServerConnect(result);
|
||||
setProgress(null);
|
||||
@@ -169,21 +196,18 @@ const AppContent = () => {
|
||||
|
||||
// ── Server heartbeat: detect when server goes down while exploring ────────
|
||||
// Uses SSE (EventSource) for instant detection — no polling delay.
|
||||
// On disconnect: show a reconnecting banner instead of resetting to onboarding.
|
||||
// The heartbeat retries indefinitely with capped backoff and recovers automatically.
|
||||
useEffect(() => {
|
||||
if (viewMode !== 'exploring') return;
|
||||
|
||||
const cleanup = connectHeartbeat(
|
||||
() => {}, // onConnect — already connected, no action needed
|
||||
() => {
|
||||
// Server went down — return to onboarding
|
||||
setViewMode('onboarding');
|
||||
setGraph(null);
|
||||
setProgress(null);
|
||||
},
|
||||
() => setServerDisconnected(false),
|
||||
() => setServerDisconnected(true),
|
||||
);
|
||||
|
||||
return cleanup;
|
||||
}, [viewMode, setViewMode, setGraph, setProgress]);
|
||||
}, [viewMode]);
|
||||
|
||||
// Render based on view mode
|
||||
if (viewMode === 'onboarding') {
|
||||
@@ -196,7 +220,12 @@ const AppContent = () => {
|
||||
await handleServerConnect(result);
|
||||
setProgress(null);
|
||||
if (serverUrl) {
|
||||
setServerBaseUrl(normalizeServerUrl(serverUrl));
|
||||
const base = normalizeServerUrl(serverUrl);
|
||||
setServerBaseUrl(base);
|
||||
// Add ?server= so F5 reconnects to this server
|
||||
const url = new URL(window.location.href);
|
||||
url.searchParams.set('server', base);
|
||||
window.history.replaceState(null, '', url.toString());
|
||||
}
|
||||
}}
|
||||
/>
|
||||
@@ -268,6 +297,12 @@ const AppContent = () => {
|
||||
|
||||
<StatusBar />
|
||||
|
||||
{serverDisconnected && (
|
||||
<div className="fixed bottom-12 left-1/2 z-50 -translate-x-1/2 rounded-lg border border-yellow-500/30 bg-yellow-900/80 px-4 py-2 text-sm text-yellow-200 shadow-lg backdrop-blur">
|
||||
Server connection lost — reconnecting…
|
||||
</div>
|
||||
)}
|
||||
|
||||
{/* Settings Panel (modal) */}
|
||||
<SettingsPanel
|
||||
isOpen={isSettingsPanelOpen}
|
||||
|
||||
@@ -54,6 +54,7 @@ export const CodeReferencesPanel = ({ onFocusNode }: CodeReferencesPanelProps) =
|
||||
clearCodeReferences,
|
||||
setSelectedNode,
|
||||
codeReferenceFocus,
|
||||
projectName,
|
||||
} = useAppState();
|
||||
|
||||
const nodeById = useMemo(() => {
|
||||
@@ -174,7 +175,7 @@ export const CodeReferencesPanel = ({ onFocusNode }: CodeReferencesPanelProps) =
|
||||
return () => {
|
||||
rafIds.forEach((id) => cancelAnimationFrame(id));
|
||||
};
|
||||
}, [codeReferenceFocus?.ts, aiReferences]);
|
||||
}, [codeReferenceFocus, aiReferences]);
|
||||
|
||||
const refsWithSnippets = useMemo(() => {
|
||||
return aiReferences.map((ref) => {
|
||||
@@ -223,13 +224,14 @@ export const CodeReferencesPanel = ({ onFocusNode }: CodeReferencesPanelProps) =
|
||||
const isWholeFile = selectedIsFile || startLine === undefined;
|
||||
|
||||
const options = isWholeFile
|
||||
? undefined
|
||||
? { repo: projectName }
|
||||
: {
|
||||
startLine: Math.max(0, startLine - CONTEXT_LINES),
|
||||
endLine: (endLine ?? startLine) + CONTEXT_LINES,
|
||||
repo: projectName,
|
||||
};
|
||||
|
||||
readFile(selectedFilePath, options)
|
||||
readFile(selectedFilePath, { ...options, repo: projectName || undefined })
|
||||
.then((result) => {
|
||||
if (!cancelled) {
|
||||
setFileResult(result);
|
||||
@@ -251,6 +253,7 @@ export const CodeReferencesPanel = ({ onFocusNode }: CodeReferencesPanelProps) =
|
||||
selectedNode?.properties?.startLine,
|
||||
selectedNode?.properties?.endLine,
|
||||
selectedIsFile,
|
||||
projectName,
|
||||
]);
|
||||
|
||||
// Scroll to the selected node's startLine after content loads
|
||||
|
||||
@@ -201,7 +201,7 @@ interface OnboardingGuideProps {
|
||||
}
|
||||
|
||||
export const OnboardingGuide = ({ isPolling }: OnboardingGuideProps) => {
|
||||
const primary = isDev ? 'cd gitnexus && npm run serve' : 'npx gitnexus@latest serve';
|
||||
const primary = isDev ? 'npm run --prefix gitnexus serve' : 'npx gitnexus@latest serve';
|
||||
const termLabel = isDev ? 'Start backend' : 'Terminal';
|
||||
|
||||
// Step states: step 1 = copy command, step 2 = run/wait, step 3 = auto-connect
|
||||
@@ -277,7 +277,9 @@ export const OnboardingGuide = ({ isPolling }: OnboardingGuideProps) => {
|
||||
state={step2State}
|
||||
number={2}
|
||||
title={isPolling ? 'Waiting for server to start' : 'Paste and run in your terminal'}
|
||||
description={isPolling ? undefined : 'Open a new terminal window, paste, and hit Enter.'}
|
||||
description={
|
||||
isPolling ? undefined : 'Open a terminal at the project root, paste, and hit Enter.'
|
||||
}
|
||||
>
|
||||
{isPolling && <PollingBar />}
|
||||
</StepRow>
|
||||
|
||||
@@ -8,8 +8,10 @@ import {
|
||||
Loader2,
|
||||
AlertTriangle,
|
||||
GitBranch,
|
||||
ArrowDown,
|
||||
} from '@/lib/lucide-icons';
|
||||
import { useAppState } from '../hooks/useAppState';
|
||||
import { useAutoScroll } from '../hooks/useAutoScroll';
|
||||
import { ToolCallCard } from './ToolCallCard';
|
||||
import { isProviderConfigured } from '../core/llm/settings-service';
|
||||
import { MarkdownRenderer } from './MarkdownRenderer';
|
||||
@@ -35,14 +37,11 @@ export const RightPanel = () => {
|
||||
const [chatInput, setChatInput] = useState('');
|
||||
const [activeTab, setActiveTab] = useState<'chat' | 'processes'>('chat');
|
||||
const textareaRef = useRef<HTMLTextAreaElement>(null);
|
||||
const messagesEndRef = useRef<HTMLDivElement>(null);
|
||||
|
||||
// Auto-scroll to bottom when messages update or while streaming
|
||||
useEffect(() => {
|
||||
if (messagesEndRef.current) {
|
||||
messagesEndRef.current.scrollIntoView({ behavior: 'smooth' });
|
||||
}
|
||||
}, [chatMessages, isChatLoading]);
|
||||
// Keep streamed replies pinned unless the user intentionally scrolls away from the bottom.
|
||||
const { scrollContainerRef, messagesContainerRef, isAtBottom, scrollToBottom } = useAutoScroll(
|
||||
chatMessages,
|
||||
isChatLoading,
|
||||
);
|
||||
|
||||
const resolveFilePathForUI = useCallback((_requestedPath: string): string | null => {
|
||||
return null;
|
||||
@@ -265,7 +264,7 @@ export const RightPanel = () => {
|
||||
|
||||
{/* Chat Content - only show when chat tab is active */}
|
||||
{activeTab === 'chat' && (
|
||||
<div className="flex flex-1 flex-col overflow-hidden">
|
||||
<div className="relative flex flex-1 flex-col overflow-hidden">
|
||||
{/* Status bar */}
|
||||
<div className="flex items-center gap-2.5 border-b border-border-subtle bg-elevated/50 px-4 py-3">
|
||||
<div className="ml-auto flex items-center gap-2">
|
||||
@@ -291,7 +290,7 @@ export const RightPanel = () => {
|
||||
)}
|
||||
|
||||
{/* Messages */}
|
||||
<div className="scrollbar-thin flex-1 overflow-y-auto p-4">
|
||||
<div ref={scrollContainerRef} className="scrollbar-thin flex-1 overflow-y-auto p-4">
|
||||
{chatMessages.length === 0 ? (
|
||||
<div className="flex h-full flex-col items-center justify-center px-4 text-center">
|
||||
<div className="mb-4 flex h-14 w-14 items-center justify-center rounded-xl bg-gradient-to-br from-accent to-node-interface text-2xl shadow-glow">
|
||||
@@ -315,7 +314,7 @@ export const RightPanel = () => {
|
||||
</div>
|
||||
</div>
|
||||
) : (
|
||||
<div className="flex flex-col gap-6">
|
||||
<div ref={messagesContainerRef} className="flex flex-col gap-6">
|
||||
{chatMessages.map((message) => (
|
||||
<div key={message.id} className="animate-fade-in">
|
||||
{/* User message - compact label style */}
|
||||
@@ -391,10 +390,22 @@ export const RightPanel = () => {
|
||||
))}
|
||||
</div>
|
||||
)}
|
||||
{/* Scroll anchor for auto-scroll */}
|
||||
<div ref={messagesEndRef} />
|
||||
</div>
|
||||
|
||||
{/* Scroll to bottom */}
|
||||
<button
|
||||
aria-label="Scroll to bottom"
|
||||
onClick={() => scrollToBottom()}
|
||||
className={`absolute bottom-20 left-1/2 z-10 -translate-x-1/2 rounded-full border border-border-subtle bg-elevated px-3 py-1.5 text-xs text-text-secondary shadow-lg transition-all duration-200 hover:border-accent hover:text-accent ${
|
||||
!isAtBottom && chatMessages.length > 0
|
||||
? 'translate-y-0 opacity-100'
|
||||
: 'pointer-events-none translate-y-2 opacity-0'
|
||||
}`}
|
||||
>
|
||||
<ArrowDown className="mr-1 inline h-3.5 w-3.5" />
|
||||
Scroll to bottom
|
||||
</button>
|
||||
|
||||
{/* Input */}
|
||||
<div className="border-t border-border-subtle bg-surface p-3">
|
||||
<div className="flex items-end gap-2 rounded-xl border border-border-subtle bg-elevated px-3 py-2 transition-all focus-within:border-accent focus-within:ring-2 focus-within:ring-accent/20">
|
||||
|
||||
@@ -64,7 +64,7 @@ export const StatusBar = () => {
|
||||
</a>
|
||||
|
||||
{/* Right - Stats */}
|
||||
<div className="flex items-center gap-3">
|
||||
<div className="flex items-center gap-3" data-testid="graph-stats">
|
||||
{graph && (
|
||||
<>
|
||||
<span>{nodeCount} nodes</span>
|
||||
|
||||
@@ -145,6 +145,7 @@ interface AppState {
|
||||
availableRepos: BackendRepo[];
|
||||
setAvailableRepos: (repos: BackendRepo[]) => void;
|
||||
switchRepo: (repoName: string) => Promise<void>;
|
||||
setCurrentRepo: (repoName: string) => void;
|
||||
|
||||
// Worker API (shared across app)
|
||||
runQuery: (cypher: string) => Promise<any[]>;
|
||||
@@ -456,6 +457,10 @@ const AppStateProviderInner = ({ children }: { children: ReactNode }) => {
|
||||
// Backend client — direct HTTP calls (no Worker/Comlink)
|
||||
const repoRef = useRef<string | undefined>(undefined);
|
||||
|
||||
const setCurrentRepo = useCallback((repoName: string) => {
|
||||
repoRef.current = repoName;
|
||||
}, []);
|
||||
|
||||
const runQuery = useCallback(async (cypher: string): Promise<any[]> => {
|
||||
return backendRunQuery(cypher, repoRef.current);
|
||||
}, []);
|
||||
@@ -574,6 +579,13 @@ const AppStateProviderInner = ({ children }: { children: ReactNode }) => {
|
||||
|
||||
try {
|
||||
const effectiveProjectName = overrideProjectName || projectName || 'project';
|
||||
|
||||
// Sync repoRef so all agent backend calls target the correct repo.
|
||||
// initializeAgent can be called from App.tsx (handleServerConnect) which
|
||||
// never sets repoRef.current directly — without this, queries default to repo[0].
|
||||
if (overrideProjectName) {
|
||||
repoRef.current = overrideProjectName;
|
||||
}
|
||||
const repo = repoRef.current;
|
||||
|
||||
// Build backend interface for Graph RAG tools
|
||||
@@ -605,7 +617,8 @@ const AppStateProviderInner = ({ children }: { children: ReactNode }) => {
|
||||
setIsAgentInitializing(false);
|
||||
}
|
||||
},
|
||||
[projectName],
|
||||
// eslint-disable-next-line react-hooks/exhaustive-deps
|
||||
[], // repoRef is a stable ref — we sync it explicitly on entry; no state deps needed
|
||||
);
|
||||
|
||||
const sendChatMessage = useCallback(
|
||||
@@ -1037,6 +1050,9 @@ const AppStateProviderInner = ({ children }: { children: ReactNode }) => {
|
||||
setCodePanelOpen(false);
|
||||
setCodeReferenceFocus(null);
|
||||
|
||||
let connectedRepo: BackendRepo | undefined;
|
||||
let pNameStr = repoName || 'server-project';
|
||||
|
||||
try {
|
||||
const result: ConnectResult = await connectToServer(
|
||||
serverBaseUrl,
|
||||
@@ -1068,39 +1084,28 @@ const AppStateProviderInner = ({ children }: { children: ReactNode }) => {
|
||||
},
|
||||
undefined,
|
||||
repoName,
|
||||
{ awaitAnalysis: true }, // enable backend hold-queue for repos still being analyzed
|
||||
);
|
||||
|
||||
// Build graph for visualization
|
||||
const repoPath = result.repoInfo.repoPath ?? result.repoInfo.path;
|
||||
// Prefer the registry name, then normalize Windows \ and Unix / paths
|
||||
const pName =
|
||||
repoName || result.repoInfo.name || repoPath?.split('/').pop() || 'server-project';
|
||||
repoName ||
|
||||
result.repoInfo.name ||
|
||||
(repoPath || '').replace(/\\/g, '/').split('/').filter(Boolean).pop() ||
|
||||
'server-project';
|
||||
setProjectName(pName);
|
||||
repoRef.current = pName;
|
||||
|
||||
connectedRepo = result.repoInfo;
|
||||
pNameStr = pName;
|
||||
|
||||
const newGraph = createKnowledgeGraph();
|
||||
for (const node of result.nodes) newGraph.addNode(node);
|
||||
for (const rel of result.relationships) newGraph.addRelationship(rel);
|
||||
setGraph(newGraph);
|
||||
|
||||
// No fileContents needed — grep/read tools use backend HTTP
|
||||
|
||||
// Initialize agent with backend queries, then start embeddings
|
||||
try {
|
||||
if (getActiveProviderConfig()) {
|
||||
await initializeAgent(pName);
|
||||
}
|
||||
setViewMode('exploring');
|
||||
startEmbeddingsWithFallback();
|
||||
setProgress(null);
|
||||
} catch (err) {
|
||||
console.warn('Failed to initialize agent:', err);
|
||||
setIsAgentReady(false);
|
||||
agentRef.current = null;
|
||||
setAgentError('Failed to initialize agent');
|
||||
setViewMode('exploring');
|
||||
setProgress(null);
|
||||
}
|
||||
} catch (err) {
|
||||
} catch (err: unknown) {
|
||||
console.error('Repo switch failed:', err);
|
||||
setProgress({
|
||||
phase: 'error',
|
||||
@@ -1114,6 +1119,36 @@ const AppStateProviderInner = ({ children }: { children: ReactNode }) => {
|
||||
setViewMode('exploring');
|
||||
setProgress(null);
|
||||
}, ERROR_RESET_DELAY_MS);
|
||||
return; // Abort the whole switchRepo process
|
||||
}
|
||||
|
||||
if (pNameStr) {
|
||||
// Persist the selected project in the URL so a refresh re-opens it
|
||||
const urlObj = new URL(window.location.href);
|
||||
urlObj.searchParams.set('project', pNameStr);
|
||||
window.history.replaceState(null, '', urlObj.toString());
|
||||
}
|
||||
|
||||
// Reset the agent and clear chat history so the AI starts fresh for the new repo
|
||||
agentRef.current = null;
|
||||
setIsAgentReady(false);
|
||||
setChatMessages([]);
|
||||
|
||||
// Re-initialize agent with the new repo's graph context
|
||||
try {
|
||||
if (getActiveProviderConfig()) {
|
||||
await initializeAgent(pNameStr);
|
||||
}
|
||||
setViewMode('exploring');
|
||||
startEmbeddingsWithFallback();
|
||||
setProgress(null);
|
||||
} catch (err) {
|
||||
console.warn('Failed to initialize agent:', err);
|
||||
setIsAgentReady(false);
|
||||
agentRef.current = null;
|
||||
setAgentError('Failed to initialize agent');
|
||||
setViewMode('exploring');
|
||||
setProgress(null);
|
||||
}
|
||||
},
|
||||
[
|
||||
@@ -1133,6 +1168,7 @@ const AppStateProviderInner = ({ children }: { children: ReactNode }) => {
|
||||
setCodeReferences,
|
||||
setCodePanelOpen,
|
||||
setCodeReferenceFocus,
|
||||
setChatMessages,
|
||||
],
|
||||
);
|
||||
|
||||
@@ -1219,6 +1255,7 @@ const AppStateProviderInner = ({ children }: { children: ReactNode }) => {
|
||||
availableRepos,
|
||||
setAvailableRepos,
|
||||
switchRepo,
|
||||
setCurrentRepo,
|
||||
runQuery,
|
||||
isDatabaseReady,
|
||||
// Embedding state and methods
|
||||
|
||||
@@ -0,0 +1,145 @@
|
||||
import { useCallback, useEffect, useLayoutEffect, useRef, useState } from 'react';
|
||||
|
||||
const DEFAULT_BOTTOM_THRESHOLD = 100;
|
||||
const USER_SCROLL_EPSILON = 5;
|
||||
|
||||
export interface UseAutoScrollResult {
|
||||
scrollContainerRef: React.RefObject<HTMLDivElement>;
|
||||
messagesContainerRef: React.RefObject<HTMLDivElement>;
|
||||
isAtBottom: boolean;
|
||||
scrollToBottom: (behavior?: ScrollBehavior) => void;
|
||||
}
|
||||
|
||||
function isNearBottom(element: HTMLElement, threshold: number): boolean {
|
||||
return element.scrollHeight - element.scrollTop - element.clientHeight <= threshold;
|
||||
}
|
||||
|
||||
export function useAutoScroll<T>(
|
||||
chatMessages: T[],
|
||||
isChatLoading: boolean,
|
||||
bottomThreshold = DEFAULT_BOTTOM_THRESHOLD,
|
||||
): UseAutoScrollResult {
|
||||
const scrollContainerRef = useRef<HTMLDivElement>(null);
|
||||
const messagesContainerRef = useRef<HTMLDivElement>(null);
|
||||
|
||||
const [isAtBottom, setIsAtBottom] = useState(true);
|
||||
|
||||
const shouldStickToBottomRef = useRef(true);
|
||||
const lastScrollTopRef = useRef(0);
|
||||
const scrollFrameIdRef = useRef<number | null>(null);
|
||||
|
||||
const syncScrollState = useCallback(() => {
|
||||
const element = scrollContainerRef.current;
|
||||
if (!element) return;
|
||||
|
||||
const currentScrollTop = element.scrollTop;
|
||||
const nearBottom = isNearBottom(element, bottomThreshold);
|
||||
|
||||
if (nearBottom) {
|
||||
shouldStickToBottomRef.current = true;
|
||||
} else if (currentScrollTop < lastScrollTopRef.current - USER_SCROLL_EPSILON) {
|
||||
shouldStickToBottomRef.current = false;
|
||||
}
|
||||
|
||||
lastScrollTopRef.current = currentScrollTop;
|
||||
setIsAtBottom(nearBottom);
|
||||
}, [bottomThreshold]);
|
||||
|
||||
const scrollToBottom = useCallback(
|
||||
(behavior: ScrollBehavior = 'smooth') => {
|
||||
const element = scrollContainerRef.current;
|
||||
if (!element) return;
|
||||
|
||||
shouldStickToBottomRef.current = true;
|
||||
|
||||
if (behavior === 'auto') {
|
||||
element.scrollTop = element.scrollHeight;
|
||||
lastScrollTopRef.current = element.scrollTop;
|
||||
setIsAtBottom(isNearBottom(element, bottomThreshold));
|
||||
return;
|
||||
}
|
||||
|
||||
element.scrollTo({
|
||||
top: element.scrollHeight,
|
||||
behavior,
|
||||
});
|
||||
},
|
||||
[bottomThreshold],
|
||||
);
|
||||
|
||||
useEffect(() => {
|
||||
const element = scrollContainerRef.current;
|
||||
if (!element) return;
|
||||
|
||||
lastScrollTopRef.current = element.scrollTop;
|
||||
|
||||
const handleScroll = () => {
|
||||
if (scrollFrameIdRef.current !== null) {
|
||||
cancelAnimationFrame(scrollFrameIdRef.current);
|
||||
}
|
||||
|
||||
scrollFrameIdRef.current = requestAnimationFrame(() => {
|
||||
scrollFrameIdRef.current = null;
|
||||
syncScrollState();
|
||||
});
|
||||
};
|
||||
|
||||
element.addEventListener('scroll', handleScroll, { passive: true });
|
||||
syncScrollState();
|
||||
|
||||
return () => {
|
||||
element.removeEventListener('scroll', handleScroll);
|
||||
|
||||
if (scrollFrameIdRef.current !== null) {
|
||||
cancelAnimationFrame(scrollFrameIdRef.current);
|
||||
scrollFrameIdRef.current = null;
|
||||
}
|
||||
};
|
||||
}, [syncScrollState]);
|
||||
|
||||
useEffect(() => {
|
||||
const content = messagesContainerRef.current;
|
||||
const scrollEl = scrollContainerRef.current;
|
||||
if (!content || !scrollEl || typeof ResizeObserver === 'undefined') return;
|
||||
|
||||
let resizeFrameId: number | null = null;
|
||||
|
||||
const observer = new ResizeObserver(() => {
|
||||
if (shouldStickToBottomRef.current) {
|
||||
if (resizeFrameId !== null) {
|
||||
cancelAnimationFrame(resizeFrameId);
|
||||
}
|
||||
|
||||
resizeFrameId = requestAnimationFrame(() => {
|
||||
resizeFrameId = null;
|
||||
scrollToBottom('auto');
|
||||
});
|
||||
} else {
|
||||
syncScrollState();
|
||||
}
|
||||
});
|
||||
|
||||
observer.observe(content);
|
||||
|
||||
return () => {
|
||||
observer.disconnect();
|
||||
|
||||
if (resizeFrameId !== null) {
|
||||
cancelAnimationFrame(resizeFrameId);
|
||||
resizeFrameId = null;
|
||||
}
|
||||
};
|
||||
}, [chatMessages.length, scrollToBottom, syncScrollState]);
|
||||
|
||||
useLayoutEffect(() => {
|
||||
if (!shouldStickToBottomRef.current) return;
|
||||
scrollToBottom('auto');
|
||||
}, [chatMessages.length, isChatLoading, scrollToBottom]);
|
||||
|
||||
return {
|
||||
scrollContainerRef,
|
||||
messagesContainerRef,
|
||||
isAtBottom,
|
||||
scrollToBottom,
|
||||
};
|
||||
}
|
||||
@@ -9,6 +9,7 @@
|
||||
export {
|
||||
AlertCircle,
|
||||
AlertTriangle,
|
||||
ArrowDown,
|
||||
ArrowRight,
|
||||
AtSign,
|
||||
Brain,
|
||||
|
||||
@@ -222,7 +222,7 @@ export function normalizeServerUrl(input: string): string {
|
||||
|
||||
// ── Internal Helpers ───────────────────────────────────────────────────────
|
||||
|
||||
const DEFAULT_TIMEOUT_MS = 10_000;
|
||||
const DEFAULT_TIMEOUT_MS = 30_000;
|
||||
const PROBE_TIMEOUT_MS = 2_000;
|
||||
|
||||
const fetchWithTimeout = async (
|
||||
@@ -264,11 +264,13 @@ const fetchWithTimeout = async (
|
||||
const assertOk = async (response: Response): Promise<void> => {
|
||||
if (response.ok) return;
|
||||
|
||||
let message = `Backend returned ${response.status} ${response.statusText}`;
|
||||
let message = response.statusText;
|
||||
try {
|
||||
const body = await response.json();
|
||||
if (body && typeof body.error === 'string') {
|
||||
message = body.error;
|
||||
} else if (body && typeof body.message === 'string') {
|
||||
message = body.message;
|
||||
}
|
||||
} catch {
|
||||
// Response body was not JSON
|
||||
@@ -302,16 +304,26 @@ export const fetchServerInfo = async (): Promise<ServerInfo> => {
|
||||
};
|
||||
|
||||
/**
|
||||
* Connect an SSE heartbeat to the backend. Fires `onDisconnect` when the
|
||||
* server goes down (after one retry to avoid false positives from transient
|
||||
* network hiccups). Returns a cleanup function.
|
||||
* Connect an SSE heartbeat to the backend. Retries indefinitely with capped
|
||||
* exponential backoff so transient hiccups don't reset the UI.
|
||||
*
|
||||
* - `onConnect` fires on every successful (re)connection.
|
||||
* - `onReconnecting` fires on the first retry after a drop — use it to show
|
||||
* a "reconnecting" banner while keeping the current view intact.
|
||||
*
|
||||
* Returns a cleanup function that tears down the EventSource and timers.
|
||||
*/
|
||||
export const connectHeartbeat = (onConnect: () => void, onDisconnect: () => void): (() => void) => {
|
||||
export const connectHeartbeat = (
|
||||
onConnect: () => void,
|
||||
onReconnecting: () => void,
|
||||
): (() => void) => {
|
||||
let closed = false;
|
||||
let retryTimer: ReturnType<typeof setTimeout> | null = null;
|
||||
let es: EventSource | null = null;
|
||||
let attempt = 0;
|
||||
const MAX_RETRIES = 3;
|
||||
/** Whether we've already fired onReconnecting for the current drop. */
|
||||
let notifiedReconnecting = false;
|
||||
const MAX_BACKOFF_MS = 15_000;
|
||||
|
||||
const connect = () => {
|
||||
if (closed) return;
|
||||
@@ -319,6 +331,7 @@ export const connectHeartbeat = (onConnect: () => void, onDisconnect: () => void
|
||||
es.onopen = () => {
|
||||
if (!closed) {
|
||||
attempt = 0;
|
||||
notifiedReconnecting = false;
|
||||
onConnect();
|
||||
}
|
||||
};
|
||||
@@ -326,13 +339,15 @@ export const connectHeartbeat = (onConnect: () => void, onDisconnect: () => void
|
||||
es?.close();
|
||||
es = null;
|
||||
if (closed) return;
|
||||
if (attempt < MAX_RETRIES) {
|
||||
const delay = 1_000 * Math.pow(2, attempt);
|
||||
attempt++;
|
||||
retryTimer = setTimeout(connect, delay);
|
||||
} else {
|
||||
onDisconnect();
|
||||
|
||||
if (!notifiedReconnecting) {
|
||||
notifiedReconnecting = true;
|
||||
onReconnecting();
|
||||
}
|
||||
|
||||
const delay = Math.min(1_000 * Math.pow(2, attempt), MAX_BACKOFF_MS);
|
||||
attempt++;
|
||||
retryTimer = setTimeout(connect, delay);
|
||||
};
|
||||
};
|
||||
|
||||
@@ -373,10 +388,22 @@ export const fetchRepos = async (): Promise<BackendRepo[]> => {
|
||||
return response.json() as Promise<BackendRepo[]>;
|
||||
};
|
||||
|
||||
/** Fetch repo metadata. */
|
||||
export const fetchRepoInfo = async (repo?: string): Promise<BackendRepo> => {
|
||||
/** Fetch repo metadata.
|
||||
* Pass `awaitAnalysis: true` when connecting to a repo that may still be cloning/analyzing —
|
||||
* this enables the backend's hold-queue and uses a 5-minute timeout to match.
|
||||
* Normal calls (e.g. repo switching between already-indexed repos) use the default 10s timeout.
|
||||
*
|
||||
* Must stay in sync with HOLD_QUEUE_TIMEOUT_SECS in gitnexus/src/server/api.ts.
|
||||
*/
|
||||
const HOLD_QUEUE_TIMEOUT_MS = 300_000; // 5 minutes — matches backend HOLD_QUEUE_TIMEOUT_SECS
|
||||
|
||||
export const fetchRepoInfo = async (
|
||||
repo?: string,
|
||||
opts?: { awaitAnalysis?: boolean },
|
||||
): Promise<BackendRepo> => {
|
||||
const url = `${_backendUrl}/api/repo${repo ? `?${repoParam(repo)}` : ''}`;
|
||||
const response = await fetchWithTimeout(url);
|
||||
const timeout = opts?.awaitAnalysis ? HOLD_QUEUE_TIMEOUT_MS : undefined;
|
||||
const response = await fetchWithTimeout(url, {}, timeout);
|
||||
await assertOk(response);
|
||||
const data = await response.json();
|
||||
return { ...data, repoPath: data.repoPath ?? data.path };
|
||||
@@ -391,13 +418,19 @@ export const fetchGraph = async (
|
||||
onProgress?: (downloaded: number, total: number | null) => void;
|
||||
},
|
||||
): Promise<{ nodes: GraphNode[]; relationships: GraphRelationship[] }> => {
|
||||
const params = [repoParam(repo), opts?.includeContent ? 'includeContent=true' : '']
|
||||
const params = [repoParam(repo), opts?.includeContent ? 'includeContent=true' : '', 'stream=true']
|
||||
.filter(Boolean)
|
||||
.join('&');
|
||||
const url = `${_backendUrl}/api/graph${params ? `?${params}` : ''}`;
|
||||
const response = await fetchWithTimeout(url, { signal: opts?.signal }, 60_000);
|
||||
// Large repos can take a while to serialize the graph — use an elevated timeout
|
||||
const response = await fetchWithTimeout(url, { signal: opts?.signal }, 120_000);
|
||||
await assertOk(response);
|
||||
|
||||
const contentType = response.headers.get('Content-Type') || '';
|
||||
if (contentType.includes('application/x-ndjson')) {
|
||||
return parseNdjsonGraphResponse(response, opts?.onProgress);
|
||||
}
|
||||
|
||||
if (!opts?.onProgress || !response.body) {
|
||||
return response.json() as Promise<{ nodes: GraphNode[]; relationships: GraphRelationship[] }>;
|
||||
}
|
||||
@@ -426,6 +459,66 @@ export const fetchGraph = async (
|
||||
return JSON.parse(new TextDecoder().decode(combined));
|
||||
};
|
||||
|
||||
const parseNdjsonGraphResponse = async (
|
||||
response: Response,
|
||||
onProgress?: (downloaded: number, total: number | null) => void,
|
||||
): Promise<{ nodes: GraphNode[]; relationships: GraphRelationship[] }> => {
|
||||
if (!response.body) {
|
||||
throw new BackendError('No response body', response.status, 'server');
|
||||
}
|
||||
|
||||
const contentLength = response.headers.get('Content-Length');
|
||||
const total = contentLength ? parseInt(contentLength, 10) : null;
|
||||
const reader = response.body.getReader();
|
||||
const decoder = new TextDecoder();
|
||||
const nodes: GraphNode[] = [];
|
||||
const relationships: GraphRelationship[] = [];
|
||||
let buffer = '';
|
||||
let downloaded = 0;
|
||||
|
||||
const parseLine = (line: string) => {
|
||||
const trimmed = line.trim();
|
||||
if (!trimmed) return;
|
||||
|
||||
const record = JSON.parse(trimmed) as
|
||||
| { type: 'node'; data: GraphNode }
|
||||
| { type: 'relationship'; data: GraphRelationship }
|
||||
| { type: 'error'; error: string };
|
||||
|
||||
if (record.type === 'node') {
|
||||
nodes.push(record.data);
|
||||
return;
|
||||
}
|
||||
if (record.type === 'relationship') {
|
||||
relationships.push(record.data);
|
||||
return;
|
||||
}
|
||||
if (record.type === 'error') {
|
||||
throw new BackendError(record.error, response.status || 500, 'server');
|
||||
}
|
||||
};
|
||||
|
||||
while (true) {
|
||||
const { done, value } = await reader.read();
|
||||
if (done) break;
|
||||
|
||||
downloaded += value.length;
|
||||
onProgress?.(downloaded, total);
|
||||
buffer += decoder.decode(value, { stream: true });
|
||||
|
||||
const lines = buffer.split('\n');
|
||||
buffer = lines.pop() || '';
|
||||
for (const line of lines) {
|
||||
parseLine(line);
|
||||
}
|
||||
}
|
||||
|
||||
buffer += decoder.decode();
|
||||
parseLine(buffer);
|
||||
|
||||
return { nodes, relationships };
|
||||
};
|
||||
|
||||
/** Execute a Cypher query. Returns rows. */
|
||||
export const runQuery = async (
|
||||
cypher: string,
|
||||
@@ -657,18 +750,21 @@ export interface ConnectResult {
|
||||
/**
|
||||
* Connect to a server: validate, fetch repo info, download graph.
|
||||
* Content is NOT included (use readFile/grep for file access).
|
||||
* Pass `awaitAnalysis: true` when the repo may still be cloning/analyzing —
|
||||
* this enables the backend hold-queue and a 5-minute fetch timeout.
|
||||
*/
|
||||
export async function connectToServer(
|
||||
url: string,
|
||||
onProgress?: (phase: string, downloaded: number, total: number | null) => void,
|
||||
signal?: AbortSignal,
|
||||
repoName?: string,
|
||||
opts?: { awaitAnalysis?: boolean },
|
||||
): Promise<ConnectResult> {
|
||||
const baseUrl = normalizeServerUrl(url);
|
||||
setBackendUrl(baseUrl);
|
||||
|
||||
onProgress?.('validating', 0, null);
|
||||
const repoInfo = await fetchRepoInfo(repoName);
|
||||
const repoInfo = await fetchRepoInfo(repoName, { awaitAnalysis: opts?.awaitAnalysis });
|
||||
|
||||
onProgress?.('downloading', 0, null);
|
||||
const { nodes, relationships } = await fetchGraph(repoName, {
|
||||
|
||||
@@ -0,0 +1,147 @@
|
||||
import { describe, it, expect, vi, beforeEach, afterEach } from 'vitest';
|
||||
import { connectHeartbeat } from '../../src/services/backend-client';
|
||||
|
||||
// Mock EventSource to simulate SSE behavior
|
||||
class MockEventSource {
|
||||
onopen: (() => void) | null = null;
|
||||
onerror: (() => void) | null = null;
|
||||
closed = false;
|
||||
|
||||
close() {
|
||||
this.closed = true;
|
||||
}
|
||||
}
|
||||
|
||||
let lastEventSource: MockEventSource | null = null;
|
||||
|
||||
beforeEach(() => {
|
||||
lastEventSource = null;
|
||||
vi.stubGlobal(
|
||||
'EventSource',
|
||||
vi.fn().mockImplementation(() => {
|
||||
lastEventSource = new MockEventSource();
|
||||
return lastEventSource;
|
||||
}),
|
||||
);
|
||||
vi.useFakeTimers();
|
||||
});
|
||||
|
||||
afterEach(() => {
|
||||
vi.useRealTimers();
|
||||
vi.unstubAllGlobals();
|
||||
});
|
||||
|
||||
describe('connectHeartbeat', () => {
|
||||
it('calls onConnect when EventSource opens', () => {
|
||||
const onConnect = vi.fn();
|
||||
const onReconnecting = vi.fn();
|
||||
connectHeartbeat(onConnect, onReconnecting);
|
||||
|
||||
lastEventSource!.onopen!();
|
||||
expect(onConnect).toHaveBeenCalledOnce();
|
||||
expect(onReconnecting).not.toHaveBeenCalled();
|
||||
});
|
||||
|
||||
it('calls onReconnecting on first error, then retries', () => {
|
||||
const onConnect = vi.fn();
|
||||
const onReconnecting = vi.fn();
|
||||
connectHeartbeat(onConnect, onReconnecting);
|
||||
|
||||
// Simulate connection drop
|
||||
lastEventSource!.onerror!();
|
||||
|
||||
expect(onReconnecting).toHaveBeenCalledOnce();
|
||||
expect(lastEventSource!.closed).toBe(true);
|
||||
|
||||
// Advance past first retry delay (1s)
|
||||
vi.advanceTimersByTime(1_000);
|
||||
|
||||
// A new EventSource should have been created
|
||||
expect(EventSource).toHaveBeenCalledTimes(2);
|
||||
});
|
||||
|
||||
it('fires onReconnecting only once per disconnect', () => {
|
||||
const onConnect = vi.fn();
|
||||
const onReconnecting = vi.fn();
|
||||
connectHeartbeat(onConnect, onReconnecting);
|
||||
|
||||
// First error
|
||||
lastEventSource!.onerror!();
|
||||
expect(onReconnecting).toHaveBeenCalledOnce();
|
||||
|
||||
// Second retry fires error again
|
||||
vi.advanceTimersByTime(1_000);
|
||||
lastEventSource!.onerror!();
|
||||
expect(onReconnecting).toHaveBeenCalledOnce(); // still 1
|
||||
|
||||
// Third retry fires error
|
||||
vi.advanceTimersByTime(2_000);
|
||||
lastEventSource!.onerror!();
|
||||
expect(onReconnecting).toHaveBeenCalledOnce(); // still 1
|
||||
});
|
||||
|
||||
it('retries indefinitely instead of giving up after 3 attempts', () => {
|
||||
const onConnect = vi.fn();
|
||||
const onReconnecting = vi.fn();
|
||||
connectHeartbeat(onConnect, onReconnecting);
|
||||
|
||||
// Simulate 10 consecutive failures — should never stop retrying
|
||||
for (let i = 0; i < 10; i++) {
|
||||
lastEventSource!.onerror!();
|
||||
// Advance past the max backoff (15s) to ensure the next retry fires
|
||||
vi.advanceTimersByTime(16_000);
|
||||
}
|
||||
|
||||
// Should have created 11 EventSources (1 initial + 10 retries)
|
||||
expect(EventSource).toHaveBeenCalledTimes(11);
|
||||
});
|
||||
|
||||
it('resets reconnecting state when connection recovers', () => {
|
||||
const onConnect = vi.fn();
|
||||
const onReconnecting = vi.fn();
|
||||
connectHeartbeat(onConnect, onReconnecting);
|
||||
|
||||
// Drop
|
||||
lastEventSource!.onerror!();
|
||||
expect(onReconnecting).toHaveBeenCalledOnce();
|
||||
|
||||
// Retry succeeds
|
||||
vi.advanceTimersByTime(1_000);
|
||||
lastEventSource!.onopen!();
|
||||
expect(onConnect).toHaveBeenCalledOnce();
|
||||
|
||||
// Drop again — should fire onReconnecting again (reset after recovery)
|
||||
lastEventSource!.onerror!();
|
||||
expect(onReconnecting).toHaveBeenCalledTimes(2);
|
||||
});
|
||||
|
||||
it('caps backoff at 15 seconds', () => {
|
||||
const onConnect = vi.fn();
|
||||
const onReconnecting = vi.fn();
|
||||
connectHeartbeat(onConnect, onReconnecting);
|
||||
|
||||
// Fail many times to push backoff past the cap
|
||||
for (let i = 0; i < 6; i++) {
|
||||
lastEventSource!.onerror!();
|
||||
// The delay for attempt i is min(1000 * 2^i, 15000)
|
||||
// i=0: 1s, i=1: 2s, i=2: 4s, i=3: 8s, i=4: 15s (capped), i=5: 15s (capped)
|
||||
vi.advanceTimersByTime(16_000);
|
||||
}
|
||||
|
||||
// All retries should have fired — 7 EventSources total
|
||||
expect(EventSource).toHaveBeenCalledTimes(7);
|
||||
});
|
||||
|
||||
it('stops retrying when cleanup is called', () => {
|
||||
const onConnect = vi.fn();
|
||||
const onReconnecting = vi.fn();
|
||||
const cleanup = connectHeartbeat(onConnect, onReconnecting);
|
||||
|
||||
lastEventSource!.onerror!();
|
||||
cleanup();
|
||||
|
||||
// Advance time — no new EventSource should be created
|
||||
vi.advanceTimersByTime(30_000);
|
||||
expect(EventSource).toHaveBeenCalledTimes(1);
|
||||
});
|
||||
});
|
||||
@@ -1,5 +1,5 @@
|
||||
import { describe, expect, it } from 'vitest';
|
||||
import { normalizeServerUrl } from '../../src/services/backend-client';
|
||||
import { afterEach, describe, expect, it, vi } from 'vitest';
|
||||
import { fetchGraph, normalizeServerUrl, setBackendUrl } from '../../src/services/backend-client';
|
||||
|
||||
describe('normalizeServerUrl', () => {
|
||||
it('adds http:// to localhost', () => {
|
||||
@@ -31,3 +31,137 @@ describe('normalizeServerUrl', () => {
|
||||
expect(normalizeServerUrl('https://gitnexus.example.com')).toBe('https://gitnexus.example.com');
|
||||
});
|
||||
});
|
||||
|
||||
afterEach(() => {
|
||||
vi.restoreAllMocks();
|
||||
});
|
||||
|
||||
describe('fetchGraph', () => {
|
||||
it('requests streamed graph responses from the backend', async () => {
|
||||
setBackendUrl('http://localhost:4747');
|
||||
|
||||
const fetchMock = vi.fn().mockResolvedValue(
|
||||
new Response('{"nodes":[],"relationships":[]}', {
|
||||
status: 200,
|
||||
headers: {
|
||||
'Content-Type': 'application/json',
|
||||
},
|
||||
}),
|
||||
);
|
||||
vi.stubGlobal('fetch', fetchMock);
|
||||
|
||||
await fetchGraph('big-repo');
|
||||
|
||||
expect(fetchMock).toHaveBeenCalledWith(
|
||||
expect.stringContaining('/api/graph?repo=big-repo&stream=true'),
|
||||
expect.any(Object),
|
||||
);
|
||||
});
|
||||
|
||||
it('parses NDJSON graph streams incrementally', async () => {
|
||||
setBackendUrl('http://localhost:4747');
|
||||
|
||||
const encoder = new TextEncoder();
|
||||
const stream = new ReadableStream<Uint8Array>({
|
||||
start(controller) {
|
||||
controller.enqueue(
|
||||
encoder.encode(
|
||||
[
|
||||
'{"type":"node","data":{"id":"File:src/app.ts","label":"File","properties":{"name":"app.ts","filePath":"src/app.ts"}}}\n',
|
||||
'{"type":"relationship","data":{"id":"File:src/app.ts_CONTAINS_Function:src/app.ts:main","type":"CONTAINS","sourceId":"File:src/app.ts","targetId":"Function:src/app.ts:main"}}\n',
|
||||
].join(''),
|
||||
),
|
||||
);
|
||||
controller.close();
|
||||
},
|
||||
});
|
||||
|
||||
vi.stubGlobal(
|
||||
'fetch',
|
||||
vi.fn().mockResolvedValue(
|
||||
new Response(stream, {
|
||||
status: 200,
|
||||
headers: {
|
||||
'Content-Type': 'application/x-ndjson',
|
||||
},
|
||||
}),
|
||||
),
|
||||
);
|
||||
|
||||
const progress = vi.fn();
|
||||
const result = await fetchGraph('big-repo', { onProgress: progress });
|
||||
|
||||
expect(result.nodes).toHaveLength(1);
|
||||
expect(result.relationships).toHaveLength(1);
|
||||
expect(result.nodes[0].id).toBe('File:src/app.ts');
|
||||
expect(result.relationships[0].type).toBe('CONTAINS');
|
||||
expect(progress).toHaveBeenCalled();
|
||||
});
|
||||
|
||||
it('parses NDJSON graph lines split across chunks', async () => {
|
||||
setBackendUrl('http://localhost:4747');
|
||||
|
||||
const encoder = new TextEncoder();
|
||||
const stream = new ReadableStream<Uint8Array>({
|
||||
start(controller) {
|
||||
controller.enqueue(
|
||||
encoder.encode(
|
||||
'{"type":"node","data":{"id":"File:src/app.ts","label":"File","properties":{"name":"app.ts"',
|
||||
),
|
||||
);
|
||||
controller.enqueue(
|
||||
encoder.encode(
|
||||
',"filePath":"src/app.ts"}}}\n{"type":"relationship","data":{"id":"File:src/app.ts_CONTAINS_Function:src/app.ts:main","type":"CONTAINS","sourceId":"File:src/app.ts","targetId":"Function:src/app.ts:main"}}\n',
|
||||
),
|
||||
);
|
||||
controller.close();
|
||||
},
|
||||
});
|
||||
|
||||
vi.stubGlobal(
|
||||
'fetch',
|
||||
vi.fn().mockResolvedValue(
|
||||
new Response(stream, {
|
||||
status: 200,
|
||||
headers: {
|
||||
'Content-Type': 'application/x-ndjson',
|
||||
},
|
||||
}),
|
||||
),
|
||||
);
|
||||
|
||||
const result = await fetchGraph('big-repo');
|
||||
|
||||
expect(result.nodes).toHaveLength(1);
|
||||
expect(result.relationships).toHaveLength(1);
|
||||
expect(result.nodes[0].properties.filePath).toBe('src/app.ts');
|
||||
});
|
||||
|
||||
it('throws backend errors emitted in the NDJSON stream', async () => {
|
||||
setBackendUrl('http://localhost:4747');
|
||||
|
||||
const encoder = new TextEncoder();
|
||||
const stream = new ReadableStream<Uint8Array>({
|
||||
start(controller) {
|
||||
controller.enqueue(encoder.encode('{"type":"error","error":"stream failed"}\n'));
|
||||
controller.close();
|
||||
},
|
||||
});
|
||||
|
||||
vi.stubGlobal(
|
||||
'fetch',
|
||||
vi.fn().mockResolvedValue(
|
||||
new Response(stream, {
|
||||
status: 200,
|
||||
headers: {
|
||||
'Content-Type': 'application/x-ndjson',
|
||||
},
|
||||
}),
|
||||
),
|
||||
);
|
||||
|
||||
await expect(fetchGraph('big-repo')).rejects.toMatchObject({
|
||||
message: 'stream failed',
|
||||
});
|
||||
});
|
||||
});
|
||||
|
||||
@@ -0,0 +1,287 @@
|
||||
import { act, fireEvent, render, screen } from '@testing-library/react';
|
||||
import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest';
|
||||
import { useAutoScroll } from '../../src/hooks/useAutoScroll';
|
||||
|
||||
interface HarnessProps {
|
||||
messages: unknown[];
|
||||
isChatLoading: boolean;
|
||||
}
|
||||
|
||||
function AutoScrollHarness({ messages, isChatLoading }: HarnessProps) {
|
||||
const { scrollContainerRef, messagesContainerRef, isAtBottom, scrollToBottom } = useAutoScroll(
|
||||
messages,
|
||||
isChatLoading,
|
||||
);
|
||||
|
||||
return (
|
||||
<>
|
||||
<div data-testid="is-at-bottom">{String(isAtBottom)}</div>
|
||||
<div data-testid="container" ref={scrollContainerRef}>
|
||||
{messages.length > 0 ? (
|
||||
<div data-testid="messages-container" ref={messagesContainerRef}>
|
||||
{messages.map((message, index) => (
|
||||
<div key={index}>{String(message)}</div>
|
||||
))}
|
||||
</div>
|
||||
) : null}
|
||||
</div>
|
||||
<button type="button" onClick={() => scrollToBottom()}>
|
||||
Scroll to bottom
|
||||
</button>
|
||||
</>
|
||||
);
|
||||
}
|
||||
|
||||
function setScrollMetrics(
|
||||
element: HTMLDivElement,
|
||||
metrics: { scrollTop?: number; scrollHeight?: number; clientHeight?: number },
|
||||
) {
|
||||
if (metrics.scrollTop !== undefined) {
|
||||
Object.defineProperty(element, 'scrollTop', {
|
||||
configurable: true,
|
||||
writable: true,
|
||||
value: metrics.scrollTop,
|
||||
});
|
||||
}
|
||||
|
||||
if (metrics.scrollHeight !== undefined) {
|
||||
Object.defineProperty(element, 'scrollHeight', {
|
||||
configurable: true,
|
||||
value: metrics.scrollHeight,
|
||||
});
|
||||
}
|
||||
|
||||
if (metrics.clientHeight !== undefined) {
|
||||
Object.defineProperty(element, 'clientHeight', {
|
||||
configurable: true,
|
||||
value: metrics.clientHeight,
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
async function flushAnimationFrame() {
|
||||
await act(async () => {
|
||||
vi.runAllTimers();
|
||||
});
|
||||
}
|
||||
|
||||
async function scrollContainer(element: HTMLDivElement, scrollTop: number) {
|
||||
setScrollMetrics(element, { scrollTop });
|
||||
fireEvent.scroll(element);
|
||||
await flushAnimationFrame();
|
||||
}
|
||||
|
||||
const resizeObserverInstances: ResizeObserverMock[] = [];
|
||||
|
||||
class ResizeObserverMock {
|
||||
callback: ResizeObserverCallback;
|
||||
observedElements: Element[] = [];
|
||||
observe = vi.fn((element: Element) => {
|
||||
this.observedElements.push(element);
|
||||
});
|
||||
unobserve = vi.fn();
|
||||
disconnect = vi.fn();
|
||||
|
||||
constructor(callback: ResizeObserverCallback) {
|
||||
this.callback = callback;
|
||||
resizeObserverInstances.push(this);
|
||||
}
|
||||
}
|
||||
|
||||
async function triggerResize(instance: ResizeObserverMock) {
|
||||
await act(async () => {
|
||||
instance.callback([], instance as unknown as ResizeObserver);
|
||||
});
|
||||
await flushAnimationFrame();
|
||||
}
|
||||
|
||||
describe('useAutoScroll', () => {
|
||||
beforeEach(() => {
|
||||
vi.useFakeTimers();
|
||||
resizeObserverInstances.length = 0;
|
||||
vi.stubGlobal(
|
||||
'requestAnimationFrame',
|
||||
vi.fn((callback: FrameRequestCallback) => {
|
||||
return window.setTimeout(() => callback(performance.now()), 0);
|
||||
}),
|
||||
);
|
||||
vi.stubGlobal(
|
||||
'cancelAnimationFrame',
|
||||
vi.fn((frameId: number) => {
|
||||
clearTimeout(frameId);
|
||||
}),
|
||||
);
|
||||
Object.defineProperty(HTMLElement.prototype, 'scrollTo', {
|
||||
configurable: true,
|
||||
value: function (options: ScrollToOptions) {
|
||||
if (options.top !== undefined) {
|
||||
Object.defineProperty(this, 'scrollTop', {
|
||||
configurable: true,
|
||||
writable: true,
|
||||
value: options.top,
|
||||
});
|
||||
}
|
||||
},
|
||||
});
|
||||
vi.stubGlobal('ResizeObserver', ResizeObserverMock);
|
||||
});
|
||||
|
||||
afterEach(() => {
|
||||
vi.useRealTimers();
|
||||
vi.unstubAllGlobals();
|
||||
});
|
||||
|
||||
it('starts with isAtBottom true and auto-scrolls the very first message', () => {
|
||||
const { rerender } = render(<AutoScrollHarness messages={[]} isChatLoading={false} />);
|
||||
|
||||
expect(screen.getByTestId('is-at-bottom')).toHaveTextContent('true');
|
||||
|
||||
const container = screen.getByTestId('container') as HTMLDivElement;
|
||||
setScrollMetrics(container, { scrollTop: 0, scrollHeight: 500, clientHeight: 200 });
|
||||
|
||||
rerender(<AutoScrollHarness messages={[{ id: 1 }]} isChatLoading={false} />);
|
||||
|
||||
expect(container.scrollTop).toBe(500);
|
||||
expect(screen.getByTestId('is-at-bottom')).toHaveTextContent('true');
|
||||
});
|
||||
|
||||
it('follows streaming updates while the view stays pinned to the bottom', () => {
|
||||
const { rerender } = render(<AutoScrollHarness messages={[{ id: 1 }]} isChatLoading={false} />);
|
||||
const container = screen.getByTestId('container') as HTMLDivElement;
|
||||
|
||||
setScrollMetrics(container, { scrollTop: 700, scrollHeight: 1000, clientHeight: 200 });
|
||||
|
||||
rerender(<AutoScrollHarness messages={[{ id: 1 }]} isChatLoading={true} />);
|
||||
|
||||
expect(container.scrollTop).toBe(1000);
|
||||
expect(screen.getByTestId('is-at-bottom')).toHaveTextContent('true');
|
||||
});
|
||||
|
||||
it('stops auto-scroll after the user scrolls up', async () => {
|
||||
const { rerender } = render(<AutoScrollHarness messages={[{ id: 1 }]} isChatLoading={false} />);
|
||||
const container = screen.getByTestId('container') as HTMLDivElement;
|
||||
|
||||
setScrollMetrics(container, { scrollTop: 700, scrollHeight: 1000, clientHeight: 200 });
|
||||
await scrollContainer(container, 700);
|
||||
await scrollContainer(container, 250);
|
||||
|
||||
expect(screen.getByTestId('is-at-bottom')).toHaveTextContent('false');
|
||||
|
||||
setScrollMetrics(container, { scrollTop: 250, scrollHeight: 1400, clientHeight: 200 });
|
||||
rerender(<AutoScrollHarness messages={[{ id: 1 }, { id: 2 }]} isChatLoading={true} />);
|
||||
|
||||
expect(container.scrollTop).toBe(250);
|
||||
});
|
||||
|
||||
it('re-enables auto-scroll once the user returns near the bottom', async () => {
|
||||
const { rerender } = render(<AutoScrollHarness messages={[{ id: 1 }]} isChatLoading={false} />);
|
||||
const container = screen.getByTestId('container') as HTMLDivElement;
|
||||
|
||||
setScrollMetrics(container, { scrollTop: 700, scrollHeight: 1000, clientHeight: 200 });
|
||||
await scrollContainer(container, 700);
|
||||
await scrollContainer(container, 250);
|
||||
|
||||
setScrollMetrics(container, { scrollTop: 1120, scrollHeight: 1400, clientHeight: 200 });
|
||||
await scrollContainer(container, 1120);
|
||||
|
||||
expect(screen.getByTestId('is-at-bottom')).toHaveTextContent('true');
|
||||
|
||||
setScrollMetrics(container, { scrollTop: 1120, scrollHeight: 1800, clientHeight: 200 });
|
||||
rerender(<AutoScrollHarness messages={[{ id: 1 }, { id: 2 }]} isChatLoading={true} />);
|
||||
|
||||
expect(container.scrollTop).toBe(1800);
|
||||
});
|
||||
|
||||
it('scrollToBottom re-engages auto-scroll and scrolls to the container bottom', async () => {
|
||||
const { rerender } = render(<AutoScrollHarness messages={[{ id: 1 }]} isChatLoading={false} />);
|
||||
const container = screen.getByTestId('container') as HTMLDivElement;
|
||||
const scrollTo = vi.spyOn(container, 'scrollTo');
|
||||
|
||||
setScrollMetrics(container, { scrollTop: 700, scrollHeight: 1000, clientHeight: 200 });
|
||||
await scrollContainer(container, 700);
|
||||
await scrollContainer(container, 250);
|
||||
|
||||
fireEvent.click(screen.getByRole('button', { name: 'Scroll to bottom' }));
|
||||
|
||||
expect(scrollTo).toHaveBeenCalledWith({ top: 1000, behavior: 'smooth' });
|
||||
expect(screen.getByTestId('is-at-bottom')).toHaveTextContent('false');
|
||||
|
||||
setScrollMetrics(container, { scrollTop: 250, scrollHeight: 1600, clientHeight: 200 });
|
||||
rerender(<AutoScrollHarness messages={[{ id: 1 }, { id: 2 }]} isChatLoading={true} />);
|
||||
|
||||
expect(container.scrollTop).toBe(1600);
|
||||
expect(screen.getByTestId('is-at-bottom')).toHaveTextContent('true');
|
||||
});
|
||||
|
||||
it('re-pins to the latest bottom when inner content grows asynchronously', async () => {
|
||||
render(<AutoScrollHarness messages={[{ id: 1 }]} isChatLoading={false} />);
|
||||
const container = screen.getByTestId('container') as HTMLDivElement;
|
||||
|
||||
setScrollMetrics(container, { scrollTop: 700, scrollHeight: 1000, clientHeight: 200 });
|
||||
await scrollContainer(container, 700);
|
||||
|
||||
setScrollMetrics(container, { scrollTop: 1000, scrollHeight: 1450, clientHeight: 200 });
|
||||
await triggerResize(resizeObserverInstances[0]);
|
||||
|
||||
expect(container.scrollTop).toBe(1450);
|
||||
expect(screen.getByTestId('is-at-bottom')).toHaveTextContent('true');
|
||||
});
|
||||
|
||||
it('does not auto-scroll on async growth after user intentionally scrolls away', async () => {
|
||||
render(<AutoScrollHarness messages={[{ id: 1 }]} isChatLoading={false} />);
|
||||
const container = screen.getByTestId('container') as HTMLDivElement;
|
||||
|
||||
setScrollMetrics(container, { scrollTop: 700, scrollHeight: 1000, clientHeight: 200 });
|
||||
await scrollContainer(container, 700);
|
||||
await scrollContainer(container, 250);
|
||||
|
||||
setScrollMetrics(container, { scrollTop: 250, scrollHeight: 1400, clientHeight: 200 });
|
||||
await triggerResize(resizeObserverInstances[0]);
|
||||
|
||||
expect(container.scrollTop).toBe(250);
|
||||
expect(screen.getByTestId('is-at-bottom')).toHaveTextContent('false');
|
||||
});
|
||||
|
||||
it('cancels the pending ResizeObserver rAF when the component unmounts', () => {
|
||||
const cancelRAF = vi.mocked(cancelAnimationFrame);
|
||||
|
||||
const { unmount } = render(<AutoScrollHarness messages={[{ id: 1 }]} isChatLoading={false} />);
|
||||
const container = screen.getByTestId('container') as HTMLDivElement;
|
||||
|
||||
setScrollMetrics(container, { scrollTop: 950, scrollHeight: 1000, clientHeight: 200 });
|
||||
|
||||
const callsBefore = cancelRAF.mock.calls.length;
|
||||
|
||||
act(() => {
|
||||
resizeObserverInstances[0].callback(
|
||||
[],
|
||||
resizeObserverInstances[0] as unknown as ResizeObserver,
|
||||
);
|
||||
});
|
||||
|
||||
unmount();
|
||||
|
||||
expect(cancelRAF.mock.calls.length).toBeGreaterThan(callsBefore);
|
||||
|
||||
expect(() => vi.runAllTimers()).not.toThrow();
|
||||
});
|
||||
|
||||
it('attaches the observer when the messages wrapper first appears and disconnects on unmount', () => {
|
||||
const { rerender, unmount } = render(<AutoScrollHarness messages={[]} isChatLoading={false} />);
|
||||
|
||||
expect(screen.queryByTestId('messages-container')).toBeNull();
|
||||
expect(resizeObserverInstances).toHaveLength(0);
|
||||
|
||||
rerender(<AutoScrollHarness messages={[{ id: 1 }]} isChatLoading={false} />);
|
||||
|
||||
const messagesContainer = screen.getByTestId('messages-container');
|
||||
const resizeObserver = resizeObserverInstances[0];
|
||||
|
||||
expect(resizeObserverInstances).toHaveLength(1);
|
||||
expect(resizeObserver.observe).toHaveBeenCalledWith(messagesContainer);
|
||||
|
||||
unmount();
|
||||
|
||||
expect(resizeObserver.disconnect).toHaveBeenCalledTimes(1);
|
||||
});
|
||||
});
|
||||
@@ -2,6 +2,120 @@
|
||||
|
||||
All notable changes to GitNexus will be documented in this file.
|
||||
|
||||
## [1.6.1] - 2026-04-13
|
||||
|
||||
### Added
|
||||
- **Service group extractor expansion** — manifest extractor and broader extractor coverage (2/4 of #606 split) (#796)
|
||||
- **Dart call patterns** for `await`, cascade, lambda, and widget-tree contexts (#801)
|
||||
|
||||
### Fixed
|
||||
- **Stack overflow and memory exhaustion** on large repository analysis (#814)
|
||||
- **`tree-sitter-dart` install crash** — switched from git URL to npm tarball (#811)
|
||||
- **Generic TypeScript awaited function calls** missing from the call graph (#804)
|
||||
- **Runtime dependency on `file:../gitnexus-shared`** removed from the published package (#803)
|
||||
- **Ruby `singleton_class` context** preserved during sequential parsing (#774)
|
||||
|
||||
### Changed
|
||||
- **DAG-based ingestion pipeline architecture** — pipeline phases now declare typed dependencies and run via a topologically sorted DAG; container-node logic extracted to `LanguageProvider`. Includes hardened lifecycle (try/finally cleanup, error wrapping, cycle reporting), tightened `ParseOutput.exportedTypeMap` immutability, and corrected phase dependencies (#809)
|
||||
|
||||
## [1.6.0] - 2026-04-12
|
||||
|
||||
### Added
|
||||
- **SemanticModel architecture refactor (SM-8 through SM-19)** — extracted registries into `model/` module with ISP-compliant interfaces: TypeRegistry, MethodRegistry, FieldRegistry, RegistrationTable, ResolutionContext (#786)
|
||||
- HeritageMap built from accumulated `ExtractedHeritage[]` for MRO-aware resolution (#739)
|
||||
- `lookupMethodByOwnerWithMRO` using HeritageMap for cross-class method dispatch (#740)
|
||||
- MRO fast path before D2 fuzzy widening in call resolution (#741)
|
||||
- BindingAccumulator for cross-file return type propagation (#743, #763)
|
||||
- Restructured `resolveUncached` replacing `lookupFuzzy` data source for all tiers (#764)
|
||||
- Deleted `lookupFuzzy`, `lookupFuzzyCallable`, `globalIndex`, `callableIndex` — replaced with structured lookups (#769)
|
||||
- Deleted `resolveCallTarget` god-method — replaced with thin dispatcher delegating to `resolveMemberCall` (#744), `resolveStaticCall` (#754), `resolveFreeCall` (#756) (#770)
|
||||
- **Service group infrastructure** — service boundary detection, contract extractors, sync pipeline, CLI/MCP tools, monorepo fixture; bridge.lbug storage and contract matching expansion (#795)
|
||||
- **C# interface-to-interface heritage** capture (#789)
|
||||
- **Vue SFC support** with destructured call result tracking (#604)
|
||||
- **Java method reference** resolution — `obj::method` as call sites (#622)
|
||||
- **C/C++ MethodExtractor** config with pure virtual detection (#617)
|
||||
- **MethodExtractor configs** for Python, PHP, Swift, Dart, Rust, Ruby (#624)
|
||||
- **METHOD_IMPLEMENTS edges** with overload disambiguation and MethodExtractor unification (#642)
|
||||
- **Same-arity overload disambiguation** via type-hash suffix (#658)
|
||||
- **`GITNEXUS_HOME` env var** to customize global directory (#746)
|
||||
- **Verbose analyze output** prints skipped large file paths (#745)
|
||||
- **Class name lookup index** for O(1) qualified lookups (#707, #716)
|
||||
- **`lookupMethodByOwner` index** for O(1) cross-class chain resolution (#665)
|
||||
- **Fuzzy lookup counters** for performance visibility (#708)
|
||||
|
||||
### Fixed
|
||||
- **Stack overflow on large PHP files** — iterative AST traversal (#783)
|
||||
- **Large repository graph loading** failure (#732)
|
||||
- **Windows multi-repo switching** — false 404 errors and stale repo context (#633)
|
||||
- **`detect_changes` diff mapping** — map diff hunks to symbol line ranges (#779)
|
||||
- **HTTP client vs Express route detection** and Spring interface attribution (#780)
|
||||
- **VECTOR extension** not loaded during DB init for semantic search (#782)
|
||||
- **tree-sitter-swift** postinstall patch for macOS ARM64 (#788)
|
||||
- **tree-sitter-c** peer dependency conflict pinned (#723)
|
||||
- **Constructor indexing** in methodByOwner (#694, #753)
|
||||
- **Named binding processor** — `lookupExact` replaced with `lookupExactAll` (#755)
|
||||
- **`.gitnexusignore` negation patterns** now respected (#654)
|
||||
- **MCP setup** prefers global gitnexus binary over npx (#653)
|
||||
- **CORS rejection** returns clean error instead of 500 (#646)
|
||||
- **Array.push stack overflow** — replaced spread with loop (#650)
|
||||
- **MCP stdout silencing** prevents embedder/pool-adapter conflicts (#645)
|
||||
- **Web heartbeat** — graceful reconnection replaces aggressive disconnect (#643)
|
||||
- **Web repo scoping** — backend calls scoped to active repo (#644)
|
||||
- **OpenCode config path** and FTS extension load order (#781)
|
||||
- **OnboardingGuide** dev-mode serve command corrected (#725)
|
||||
- **Security issues** and critical bugs from code review (#709)
|
||||
|
||||
### Changed
|
||||
- Replaced class-type fuzzy lookups with structured indices in type-env (#733, #734, #736)
|
||||
- Extracted `CLASS_LIKE_TYPES` constant (#693)
|
||||
|
||||
## [1.5.3] - 2026-04-01
|
||||
|
||||
### Added
|
||||
- **TypeScript/JavaScript MethodExtractor** config (#588)
|
||||
|
||||
### Fixed
|
||||
- **Wiki Azure OpenAI** compat and HTML viewer script injection (#618)
|
||||
|
||||
## [1.5.2] - 2026-04-01
|
||||
|
||||
### Fixed
|
||||
- **`gitnexus-shared` module not found** — `gitnexus-shared` was a `file:` workspace dependency never published to npm, causing `ERR_MODULE_NOT_FOUND` when installing `gitnexus` globally. The build now bundles shared code into `dist/_shared/` and rewrites imports to relative paths (#613)
|
||||
- **v1.5.1 publish regression** — npm's `prepare` lifecycle ran `tsc` after `prepack`, overwriting the rewritten imports before packing; both scripts now run the full build so the final tarball is always correct
|
||||
|
||||
## [1.5.1] - 2026-04-01 [YANKED]
|
||||
|
||||
### Fixed
|
||||
- Incomplete fix for `gitnexus-shared` bundling — `prepare` script overwrote rewritten imports during publish
|
||||
|
||||
## [1.5.0] - 2026-04-01
|
||||
|
||||
### Added
|
||||
- **Repo landing screen** — when the backend detects indexed repositories, the web UI now shows a landing page with selectable repo cards (name, stats, indexed date) instead of auto-loading the first repo; users can also analyze new repos directly from the landing screen (#607)
|
||||
- **Unified web & CLI ingestion pipeline** — complete architectural migration of the web app from a self-contained WASM browser app to a thin client backed by the CLI server; new `gitnexus-shared` package for cross-package type unification (#536)
|
||||
- New server endpoints: `/api/heartbeat` (SSE liveness), `/api/info`, `/api/repos`, `/api/file`, `/api/grep`, `/api/analyze` (SSE progress), `/api/embed`, `/api/mcp` (MCP-over-StreamableHTTP)
|
||||
- Onboarding flow: auto-detect server → connect → repo landing or analyze
|
||||
- Header repo dropdown: switch, re-analyze, or delete repos
|
||||
- **Azure OpenAI support for wiki command** — fixed broken Azure auth (`api-key` header), `api-version` URL parameter, reasoning model handling (`max_completion_tokens`, no `temperature`), content filter error messages; added interactive setup wizard, `--api-version` and `--reasoning-model` CLI flags (#562)
|
||||
- **Java method references & interface dispatch** — `obj::method` treated as call sites, overload selection via typed variable args (not just literals), interface dispatch emits additional CALLS edges to implementing classes (#540)
|
||||
- **MethodExtractor abstraction** — structured method metadata extraction (isAbstract, isFinal, annotations, visibility, parameter types) with config-driven factory pattern (#576)
|
||||
- Java and Kotlin configs with overload-safe `methodInfoCache` keyed by `name:line`
|
||||
- C# config with `sealed`, `params`/`out`/`ref`/optional parameters, `[Attribute]` syntax, `internal` visibility (#582)
|
||||
- **`--skip-agents-md` CLI flag** — opt out of overwriting GitNexus-managed sections in AGENTS.md and CLAUDE.md during `gitnexus analyze` (#517)
|
||||
- **Prettier** — monorepo-wide code formatter with lint-staged + Husky pre-commit hook, `.prettierrc` config, Tailwind CSS v4 plugin, `endOfLine: "lf"` + `.gitattributes` for Windows consistency (#563)
|
||||
- **ESLint v9** — flat config with `unused-imports` auto-removal, `@typescript-eslint` rules, React hooks rules, CI `lint` job (#564)
|
||||
|
||||
### Fixed
|
||||
- **OpenCode MCP configuration** — corrected README MCP setup for OpenCode which requires `command` as an array containing both executable and arguments (#363)
|
||||
- **litellm security** — excluded vulnerable versions 1.82.7 and 1.82.8 in eval harness `pyproject.toml` (#580)
|
||||
|
||||
### Changed
|
||||
- **Reduced explicit `any` types** — 128 `no-explicit-any` warnings eliminated (689 → 561, 19% reduction) across `NodeProperties` index signature, ~80 `SyntaxNode` substitutions, typed worker protocol, and graphology community detection (#566)
|
||||
|
||||
### Docs
|
||||
- Added `gitnexus-shared` build step to web UI quick start instructions (#585)
|
||||
- Added enterprise offering section to README (#579)
|
||||
|
||||
## [1.4.10] - 2026-03-27
|
||||
|
||||
### Fixed
|
||||
|
||||
@@ -164,6 +164,16 @@ gitnexus clean # Delete index for current repo
|
||||
gitnexus clean --all --force # Delete all indexes
|
||||
gitnexus wiki [path] # Generate LLM-powered docs from knowledge graph
|
||||
gitnexus wiki --model <model> # Wiki with custom LLM model (default: gpt-4o-mini)
|
||||
|
||||
# Repository groups (multi-repo / monorepo service tracking)
|
||||
gitnexus group create <name> # Create a repository group
|
||||
gitnexus group add <name> <repo> # Add a repo to a group
|
||||
gitnexus group remove <name> <repo> # Remove a repo from a group
|
||||
gitnexus group list [name] # List groups, or show one group's config
|
||||
gitnexus group sync <name> # Extract contracts and match across repos/services
|
||||
gitnexus group contracts <name> # Inspect extracted contracts and cross-links
|
||||
gitnexus group query <name> <q> # Search execution flows across all repos in a group
|
||||
gitnexus group status <name> # Check staleness of repos in a group
|
||||
```
|
||||
|
||||
## Remote Embeddings
|
||||
@@ -224,6 +234,56 @@ Installed automatically by both `gitnexus analyze` (per-repo) and `gitnexus setu
|
||||
- Node.js >= 18
|
||||
- Git repository (uses git for commit tracking)
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
### `Cannot destructure property 'package' of 'node.target' as it is null`
|
||||
|
||||
This crash was caused by a dependency URL format that is incompatible with
|
||||
certain npm/arborist versions ([npm/cli#8126](https://github.com/npm/cli/issues/8126)).
|
||||
It is fixed in **gitnexus v1.6.2+**. Upgrade to the latest version:
|
||||
|
||||
```bash
|
||||
npx gitnexus@latest analyze # always uses the newest release
|
||||
# — or —
|
||||
npm install -g gitnexus@latest # upgrade a global install
|
||||
```
|
||||
|
||||
If you still hit npm install issues after upgrading, these generic workarounds
|
||||
may help:
|
||||
|
||||
```bash
|
||||
npm install -g npm@latest # update npm itself
|
||||
npm cache clean --force # clear a possibly corrupt cache
|
||||
```
|
||||
|
||||
### Installation fails with native module errors
|
||||
|
||||
Some optional language grammars (Dart, Kotlin, Swift) require native compilation. If they fail, GitNexus still works — those languages will be skipped.
|
||||
|
||||
If `npm install -g gitnexus` fails on native modules:
|
||||
|
||||
```bash
|
||||
# Ensure build tools are available (Linux/macOS)
|
||||
# Ubuntu/Debian: sudo apt install python3 make g++
|
||||
# macOS: xcode-select --install
|
||||
|
||||
# Retry installation
|
||||
npm install -g gitnexus
|
||||
```
|
||||
|
||||
### Analysis runs out of memory
|
||||
|
||||
For very large repositories:
|
||||
|
||||
```bash
|
||||
# Increase Node.js heap size
|
||||
NODE_OPTIONS="--max-old-space-size=16384" npx gitnexus analyze
|
||||
|
||||
# Exclude large directories
|
||||
echo "vendor/" >> .gitnexusignore
|
||||
echo "dist/" >> .gitnexusignore
|
||||
```
|
||||
|
||||
## Privacy
|
||||
|
||||
- All processing happens locally on your machine
|
||||
|
||||
Generated
+70
-5
@@ -1,27 +1,29 @@
|
||||
{
|
||||
"name": "gitnexus",
|
||||
"version": "1.4.10",
|
||||
"version": "1.6.1",
|
||||
"lockfileVersion": 3,
|
||||
"requires": true,
|
||||
"packages": {
|
||||
"": {
|
||||
"name": "gitnexus",
|
||||
"version": "1.4.10",
|
||||
"version": "1.6.1",
|
||||
"hasInstallScript": true,
|
||||
"license": "PolyForm-Noncommercial-1.0.0",
|
||||
"dependencies": {
|
||||
"@huggingface/transformers": "^3.0.0",
|
||||
"@ladybugdb/core": "^0.15.2",
|
||||
"@modelcontextprotocol/sdk": "^1.0.0",
|
||||
"@scarf/scarf": "^1.4.0",
|
||||
"cli-progress": "^3.12.0",
|
||||
"commander": "^12.0.0",
|
||||
"cors": "^2.8.5",
|
||||
"express": "^4.19.2",
|
||||
"gitnexus-shared": "file:../gitnexus-shared",
|
||||
"glob": "^11.0.0",
|
||||
"graphology": "^0.25.4",
|
||||
"graphology-indices": "^0.17.0",
|
||||
"graphology-utils": "^2.3.0",
|
||||
"ignore": "^7.0.5",
|
||||
"js-yaml": "^4.1.1",
|
||||
"lru-cache": "^11.0.0",
|
||||
"mnemonist": "^0.39.0",
|
||||
"onnxruntime-node": "^1.24.0",
|
||||
@@ -47,9 +49,11 @@
|
||||
"@types/cli-progress": "^3.11.6",
|
||||
"@types/cors": "^2.8.17",
|
||||
"@types/express": "^4.17.21",
|
||||
"@types/js-yaml": "^4.0.9",
|
||||
"@types/node": "^20.0.0",
|
||||
"@types/uuid": "^10.0.0",
|
||||
"@vitest/coverage-v8": "^4.0.18",
|
||||
"gitnexus-shared": "file:../gitnexus-shared",
|
||||
"tsx": "^4.0.0",
|
||||
"typescript": "^5.4.5",
|
||||
"vitest": "^4.0.18"
|
||||
@@ -58,13 +62,15 @@
|
||||
"node": ">=20.0.0"
|
||||
},
|
||||
"optionalDependencies": {
|
||||
"tree-sitter-dart": "github:UserNobody14/tree-sitter-dart#80e23c07b64494f7e21090bb3450223ef0b192f4",
|
||||
"tree-sitter-dart": "git+https://github.com/UserNobody14/tree-sitter-dart.git#80e23c07b64494f7e21090bb3450223ef0b192f4",
|
||||
"tree-sitter-kotlin": "^0.3.8",
|
||||
"tree-sitter-proto": "file:./vendor/tree-sitter-proto",
|
||||
"tree-sitter-swift": "^0.6.0"
|
||||
}
|
||||
},
|
||||
"../gitnexus-shared": {
|
||||
"version": "1.0.0",
|
||||
"dev": true,
|
||||
"devDependencies": {
|
||||
"typescript": "^6.0.2"
|
||||
}
|
||||
@@ -1875,6 +1881,13 @@
|
||||
"dev": true,
|
||||
"license": "MIT"
|
||||
},
|
||||
"node_modules/@scarf/scarf": {
|
||||
"version": "1.4.0",
|
||||
"resolved": "https://registry.npmjs.org/@scarf/scarf/-/scarf-1.4.0.tgz",
|
||||
"integrity": "sha512-xxeapPiUXdZAE3che6f3xogoJPeZgig6omHEy1rIY5WVsB3H2BHNnZH+gHG6x91SCWyQCzWGsuL2Hh3ClO5/qQ==",
|
||||
"hasInstallScript": true,
|
||||
"license": "Apache-2.0"
|
||||
},
|
||||
"node_modules/@standard-schema/spec": {
|
||||
"version": "1.1.0",
|
||||
"resolved": "https://registry.npmjs.org/@standard-schema/spec/-/spec-1.1.0.tgz",
|
||||
@@ -1992,6 +2005,13 @@
|
||||
"dev": true,
|
||||
"license": "MIT"
|
||||
},
|
||||
"node_modules/@types/js-yaml": {
|
||||
"version": "4.0.9",
|
||||
"resolved": "https://registry.npmjs.org/@types/js-yaml/-/js-yaml-4.0.9.tgz",
|
||||
"integrity": "sha512-k4MGaQl5TGo/iipqb2UDG2UwjXziSWkh0uysQelTlJpX1qGlpUZYm8PnO4DxG1qBomtJUdYJ6qR6xdIah10JLg==",
|
||||
"dev": true,
|
||||
"license": "MIT"
|
||||
},
|
||||
"node_modules/@types/mime": {
|
||||
"version": "1.3.5",
|
||||
"resolved": "https://registry.npmjs.org/@types/mime/-/mime-1.3.5.tgz",
|
||||
@@ -2285,6 +2305,12 @@
|
||||
"url": "https://github.com/chalk/ansi-styles?sponsor=1"
|
||||
}
|
||||
},
|
||||
"node_modules/argparse": {
|
||||
"version": "2.0.1",
|
||||
"resolved": "https://registry.npmjs.org/argparse/-/argparse-2.0.1.tgz",
|
||||
"integrity": "sha512-8+9WqebbFzpX9OR+Wa6O29asIogeRMzcGtAINdpMHHyAg10f05aSFVBbcEqGf/PXw1EjAZ+q2/bEBg3DvurK3Q==",
|
||||
"license": "Python-2.0"
|
||||
},
|
||||
"node_modules/array-flatten": {
|
||||
"version": "1.1.1",
|
||||
"resolved": "https://registry.npmjs.org/array-flatten/-/array-flatten-1.1.1.tgz",
|
||||
@@ -3566,6 +3592,18 @@
|
||||
"dev": true,
|
||||
"license": "MIT"
|
||||
},
|
||||
"node_modules/js-yaml": {
|
||||
"version": "4.1.1",
|
||||
"resolved": "https://registry.npmjs.org/js-yaml/-/js-yaml-4.1.1.tgz",
|
||||
"integrity": "sha512-qQKT4zQxXl8lLwBtHMWwaTcGfFOZviOJet3Oy/xmGk2gZH677CJM9EvtfdSkgWcATZhj/55JZ0rmy3myCT5lsA==",
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"argparse": "^2.0.1"
|
||||
},
|
||||
"bin": {
|
||||
"js-yaml": "bin/js-yaml.js"
|
||||
}
|
||||
},
|
||||
"node_modules/json-schema-traverse": {
|
||||
"version": "1.0.0",
|
||||
"resolved": "https://registry.npmjs.org/json-schema-traverse/-/json-schema-traverse-1.0.0.tgz",
|
||||
@@ -5094,7 +5132,7 @@
|
||||
},
|
||||
"node_modules/tree-sitter-dart": {
|
||||
"version": "1.0.0",
|
||||
"resolved": "git+ssh://git@github.com/UserNobody14/tree-sitter-dart.git#80e23c07b64494f7e21090bb3450223ef0b192f4",
|
||||
"resolved": "git+https://github.com/UserNobody14/tree-sitter-dart.git#80e23c07b64494f7e21090bb3450223ef0b192f4",
|
||||
"integrity": "sha512-Bs/1wAOIJ2akPEXlE/XVpuES19Oo3NqoSJRJ/0N2r38qAd9nTXdqmaGHQ44/JXnA6QHcbgD2YzCCc4wUc98cyQ==",
|
||||
"hasInstallScript": true,
|
||||
"license": "ISC",
|
||||
@@ -5258,6 +5296,10 @@
|
||||
"node": "^18 || ^20 || >= 21"
|
||||
}
|
||||
},
|
||||
"node_modules/tree-sitter-proto": {
|
||||
"resolved": "vendor/tree-sitter-proto",
|
||||
"link": true
|
||||
},
|
||||
"node_modules/tree-sitter-python": {
|
||||
"version": "0.23.4",
|
||||
"resolved": "https://registry.npmjs.org/tree-sitter-python/-/tree-sitter-python-0.23.4.tgz",
|
||||
@@ -5839,6 +5881,29 @@
|
||||
"peerDependencies": {
|
||||
"zod": "^3.25.28 || ^4"
|
||||
}
|
||||
},
|
||||
"vendor/tree-sitter-proto": {
|
||||
"version": "0.4.1",
|
||||
"hasInstallScript": true,
|
||||
"license": "MIT",
|
||||
"optional": true,
|
||||
"dependencies": {
|
||||
"node-addon-api": "^8.0.0",
|
||||
"node-gyp-build": "^4.8.0"
|
||||
},
|
||||
"peerDependencies": {
|
||||
"tree-sitter": ">=0.21.0"
|
||||
}
|
||||
},
|
||||
"vendor/tree-sitter-proto/node_modules/node-addon-api": {
|
||||
"version": "8.7.0",
|
||||
"resolved": "https://registry.npmjs.org/node-addon-api/-/node-addon-api-8.7.0.tgz",
|
||||
"integrity": "sha512-9MdFxmkKaOYVTV+XVRG8ArDwwQ77XIgIPyKASB1k3JPq3M8fGQQQE3YpMOrKm6g//Ktx8ivZr8xo1Qmtqub+GA==",
|
||||
"license": "MIT",
|
||||
"optional": true,
|
||||
"engines": {
|
||||
"node": "^18 || ^20 || >= 21"
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
+13
-7
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "gitnexus",
|
||||
"version": "1.5.0",
|
||||
"version": "1.6.1",
|
||||
"description": "Graph-powered code intelligence for AI agents. Index any codebase, query via MCP or CLI.",
|
||||
"author": "Abhigyan Patwari",
|
||||
"license": "PolyForm-Noncommercial-1.0.0",
|
||||
@@ -38,7 +38,7 @@
|
||||
"vendor"
|
||||
],
|
||||
"scripts": {
|
||||
"build": "tsc",
|
||||
"build": "node scripts/build.js",
|
||||
"serve": "tsx src/cli/index.ts serve",
|
||||
"dev": "tsx watch src/cli/index.ts",
|
||||
"test": "vitest run",
|
||||
@@ -46,14 +46,15 @@
|
||||
"test:integration": "vitest run test/integration",
|
||||
"test:watch": "vitest",
|
||||
"test:coverage": "vitest run --coverage",
|
||||
"prepare": "npm run build",
|
||||
"prepack": "npm run build && chmod +x dist/cli/index.js"
|
||||
"postinstall": "node scripts/patch-tree-sitter-swift.cjs",
|
||||
"prepare": "node scripts/build.js",
|
||||
"prepack": "node scripts/build.js"
|
||||
},
|
||||
"dependencies": {
|
||||
"gitnexus-shared": "file:../gitnexus-shared",
|
||||
"@huggingface/transformers": "^3.0.0",
|
||||
"@ladybugdb/core": "^0.15.2",
|
||||
"@modelcontextprotocol/sdk": "^1.0.0",
|
||||
"@scarf/scarf": "^1.4.0",
|
||||
"cli-progress": "^3.12.0",
|
||||
"commander": "^12.0.0",
|
||||
"cors": "^2.8.5",
|
||||
@@ -63,6 +64,7 @@
|
||||
"graphology-indices": "^0.17.0",
|
||||
"graphology-utils": "^2.3.0",
|
||||
"ignore": "^7.0.5",
|
||||
"js-yaml": "^4.1.1",
|
||||
"lru-cache": "^11.0.0",
|
||||
"mnemonist": "^0.39.0",
|
||||
"onnxruntime-node": "^1.24.0",
|
||||
@@ -82,14 +84,17 @@
|
||||
"uuid": "^13.0.0"
|
||||
},
|
||||
"optionalDependencies": {
|
||||
"tree-sitter-dart": "github:UserNobody14/tree-sitter-dart#80e23c07b64494f7e21090bb3450223ef0b192f4",
|
||||
"tree-sitter-dart": "git+https://github.com/UserNobody14/tree-sitter-dart.git#80e23c07b64494f7e21090bb3450223ef0b192f4",
|
||||
"tree-sitter-kotlin": "^0.3.8",
|
||||
"tree-sitter-proto": "file:./vendor/tree-sitter-proto",
|
||||
"tree-sitter-swift": "^0.6.0"
|
||||
},
|
||||
"devDependencies": {
|
||||
"gitnexus-shared": "file:../gitnexus-shared",
|
||||
"@types/cli-progress": "^3.11.6",
|
||||
"@types/cors": "^2.8.17",
|
||||
"@types/express": "^4.17.21",
|
||||
"@types/js-yaml": "^4.0.9",
|
||||
"@types/node": "^20.0.0",
|
||||
"@types/uuid": "^10.0.0",
|
||||
"@vitest/coverage-v8": "^4.0.18",
|
||||
@@ -100,7 +105,8 @@
|
||||
"overrides": {
|
||||
"@huggingface/transformers": {
|
||||
"onnxruntime-node": "$onnxruntime-node"
|
||||
}
|
||||
},
|
||||
"tree-sitter-c": "0.23.2"
|
||||
},
|
||||
"engines": {
|
||||
"node": ">=20.0.0"
|
||||
|
||||
@@ -0,0 +1,73 @@
|
||||
#!/usr/bin/env node
|
||||
/**
|
||||
* Build script that compiles gitnexus and inlines gitnexus-shared into the dist.
|
||||
*
|
||||
* Steps:
|
||||
* 1. Build gitnexus-shared (tsc)
|
||||
* 2. Build gitnexus (tsc)
|
||||
* 3. Copy gitnexus-shared/dist → dist/_shared
|
||||
* 4. Rewrite bare 'gitnexus-shared' specifiers → relative paths
|
||||
*/
|
||||
import { execSync } from 'node:child_process';
|
||||
import fs from 'node:fs';
|
||||
import path from 'node:path';
|
||||
import { fileURLToPath } from 'node:url';
|
||||
|
||||
const __dirname = path.dirname(fileURLToPath(import.meta.url));
|
||||
const ROOT = path.resolve(__dirname, '..');
|
||||
const SHARED_ROOT = path.resolve(ROOT, '..', 'gitnexus-shared');
|
||||
const DIST = path.join(ROOT, 'dist');
|
||||
const SHARED_DEST = path.join(DIST, '_shared');
|
||||
|
||||
// ── 1. Build gitnexus-shared ───────────────────────────────────────
|
||||
console.log('[build] compiling gitnexus-shared…');
|
||||
execSync('npx tsc', { cwd: SHARED_ROOT, stdio: 'inherit' });
|
||||
|
||||
// ── 2. Build gitnexus ──────────────────────────────────────────────
|
||||
console.log('[build] compiling gitnexus…');
|
||||
execSync('npx tsc', { cwd: ROOT, stdio: 'inherit' });
|
||||
|
||||
// ── 3. Copy shared dist ────────────────────────────────────────────
|
||||
console.log('[build] copying shared module into dist/_shared…');
|
||||
fs.cpSync(path.join(SHARED_ROOT, 'dist'), SHARED_DEST, { recursive: true });
|
||||
|
||||
// ── 4. Rewrite imports ─────────────────────────────────────────────
|
||||
console.log('[build] rewriting gitnexus-shared imports…');
|
||||
let rewritten = 0;
|
||||
|
||||
function rewriteFile(filePath) {
|
||||
const content = fs.readFileSync(filePath, 'utf-8');
|
||||
if (!content.includes('gitnexus-shared')) return;
|
||||
|
||||
const relDir = path.relative(path.dirname(filePath), SHARED_DEST);
|
||||
// Always use posix separators and point to the package index
|
||||
const relImport = relDir.split(path.sep).join('/') + '/index.js';
|
||||
|
||||
const updated = content
|
||||
.replace(/from\s+['"]gitnexus-shared['"]/g, `from '${relImport}'`)
|
||||
.replace(/import\(\s*['"]gitnexus-shared['"]\s*\)/g, `import('${relImport}')`);
|
||||
|
||||
if (updated !== content) {
|
||||
fs.writeFileSync(filePath, updated);
|
||||
rewritten++;
|
||||
}
|
||||
}
|
||||
|
||||
function walk(dir, extensions, cb) {
|
||||
for (const entry of fs.readdirSync(dir, { withFileTypes: true })) {
|
||||
const full = path.join(dir, entry.name);
|
||||
if (entry.isDirectory()) {
|
||||
walk(full, extensions, cb);
|
||||
} else if (extensions.some((ext) => entry.name.endsWith(ext))) {
|
||||
cb(full);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
walk(DIST, ['.js', '.d.ts'], rewriteFile);
|
||||
|
||||
// ── 5. Make CLI entry executable ────────────────────────────────────
|
||||
const cliEntry = path.join(DIST, 'cli', 'index.js');
|
||||
if (fs.existsSync(cliEntry)) fs.chmodSync(cliEntry, 0o755);
|
||||
|
||||
console.log(`[build] done — rewrote ${rewritten} files.`);
|
||||
@@ -0,0 +1,78 @@
|
||||
#!/usr/bin/env node
|
||||
/**
|
||||
* WORKAROUND: tree-sitter-swift@0.6.0 binding.gyp build failure
|
||||
*
|
||||
* Background:
|
||||
* tree-sitter-swift@0.6.0's binding.gyp contains an "actions" array that
|
||||
* invokes `tree-sitter generate` to regenerate parser.c from grammar.js.
|
||||
* This is intended for grammar developers, but the published npm package
|
||||
* already ships pre-generated parser files (parser.c, scanner.c), so the
|
||||
* actions are unnecessary for consumers. Since consumers don't have
|
||||
* tree-sitter-cli installed, the actions always fail during `npm install`.
|
||||
*
|
||||
* Why we can't just upgrade:
|
||||
* tree-sitter-swift@0.7.1 fixes this (removes postinstall, ships prebuilds),
|
||||
* but it requires tree-sitter@^0.22.1. The upstream project pins tree-sitter
|
||||
* to ^0.21.0 and all other grammar packages depend on that version.
|
||||
* Upgrading tree-sitter would be a separate breaking change.
|
||||
*
|
||||
* How this workaround works:
|
||||
* 1. tree-sitter-swift's own postinstall fails (npm warns but continues)
|
||||
* 2. This script runs as gitnexus's postinstall
|
||||
* 3. It removes the "actions" array from binding.gyp
|
||||
* 4. It rebuilds the native binding with the cleaned binding.gyp
|
||||
*
|
||||
* TODO: Remove this script when tree-sitter is upgraded to ^0.22.x,
|
||||
* which allows using tree-sitter-swift@0.7.1+ directly.
|
||||
*/
|
||||
const fs = require('fs');
|
||||
const path = require('path');
|
||||
const { execSync } = require('child_process');
|
||||
|
||||
const swiftDir = path.join(__dirname, '..', 'node_modules', 'tree-sitter-swift');
|
||||
const bindingPath = path.join(swiftDir, 'binding.gyp');
|
||||
|
||||
try {
|
||||
if (!fs.existsSync(bindingPath)) {
|
||||
process.exit(0);
|
||||
}
|
||||
|
||||
const content = fs.readFileSync(bindingPath, 'utf8');
|
||||
let needsRebuild = false;
|
||||
|
||||
if (content.includes('"actions"')) {
|
||||
// Strip Python-style comments (#) and trailing commas before JSON parsing
|
||||
const cleaned = content
|
||||
.replace(/#[^\n]*/g, '') // Remove # comments
|
||||
.replace(/,(\s*[\]}])/g, '$1'); // Remove trailing commas before ] or }
|
||||
const gyp = JSON.parse(cleaned);
|
||||
|
||||
if (gyp.targets && gyp.targets[0] && gyp.targets[0].actions) {
|
||||
delete gyp.targets[0].actions;
|
||||
fs.writeFileSync(bindingPath, JSON.stringify(gyp, null, 2) + '\n');
|
||||
console.log('[tree-sitter-swift] Patched binding.gyp (removed actions array)');
|
||||
needsRebuild = true;
|
||||
}
|
||||
}
|
||||
|
||||
// Check if native binding exists
|
||||
const bindingNode = path.join(swiftDir, 'build', 'Release', 'tree_sitter_swift_binding.node');
|
||||
if (!fs.existsSync(bindingNode)) {
|
||||
needsRebuild = true;
|
||||
}
|
||||
|
||||
if (needsRebuild) {
|
||||
console.log('[tree-sitter-swift] Rebuilding native binding...');
|
||||
execSync('npx node-gyp rebuild', {
|
||||
cwd: swiftDir,
|
||||
stdio: 'pipe',
|
||||
timeout: 120000,
|
||||
});
|
||||
console.log('[tree-sitter-swift] Native binding built successfully');
|
||||
}
|
||||
} catch (err) {
|
||||
console.warn('[tree-sitter-swift] Could not build native binding:', err.message);
|
||||
console.warn(
|
||||
'[tree-sitter-swift] You may need to manually run: cd node_modules/tree-sitter-swift && npx node-gyp rebuild',
|
||||
);
|
||||
}
|
||||
@@ -26,6 +26,7 @@ interface RepoStats {
|
||||
|
||||
export interface AIContextOptions {
|
||||
skipAgentsMd?: boolean;
|
||||
noStats?: boolean;
|
||||
}
|
||||
|
||||
const GITNEXUS_START_MARKER = '<!-- gitnexus:start -->';
|
||||
@@ -42,10 +43,29 @@ const GITNEXUS_END_MARKER = '<!-- gitnexus:end -->';
|
||||
* - Exact tool commands with parameters — vague directives get ignored
|
||||
* - Self-review checklist — forces model to verify its own work
|
||||
*/
|
||||
async function findGroupsContainingRegistryName(registryName: string): Promise<string[]> {
|
||||
const { listGroups, getDefaultGitnexusDir, getGroupDir } =
|
||||
await import('../core/group/storage.js');
|
||||
const { loadGroupConfig } = await import('../core/group/config-parser.js');
|
||||
const names = await listGroups();
|
||||
const hits: string[] = [];
|
||||
for (const g of names) {
|
||||
try {
|
||||
const config = await loadGroupConfig(getGroupDir(getDefaultGitnexusDir(), g));
|
||||
if (Object.values(config.repos).some((r) => r === registryName)) hits.push(config.name);
|
||||
} catch {
|
||||
// skip invalid or unreadable groups
|
||||
}
|
||||
}
|
||||
return hits;
|
||||
}
|
||||
|
||||
function generateGitNexusContent(
|
||||
projectName: string,
|
||||
stats: RepoStats,
|
||||
generatedSkills?: GeneratedSkillInfo[],
|
||||
groupNames?: string[],
|
||||
noStats?: boolean,
|
||||
): string {
|
||||
const generatedRows =
|
||||
generatedSkills && generatedSkills.length > 0
|
||||
@@ -69,7 +89,7 @@ function generateGitNexusContent(
|
||||
return `${GITNEXUS_START_MARKER}
|
||||
# GitNexus — Code Intelligence
|
||||
|
||||
This project is indexed by GitNexus as **${projectName}** (${stats.nodes || 0} symbols, ${stats.edges || 0} relationships, ${stats.processes || 0} execution flows). Use the GitNexus MCP tools to understand code, assess impact, and navigate safely.
|
||||
This project is indexed by GitNexus as **${projectName}**${noStats ? '' : ` (${stats.nodes || 0} symbols, ${stats.edges || 0} relationships, ${stats.processes || 0} execution flows)`}. Use the GitNexus MCP tools to understand code, assess impact, and navigate safely.
|
||||
|
||||
> If any GitNexus tool warns the index is stale, run \`npx gitnexus analyze\` in terminal first.
|
||||
|
||||
@@ -155,7 +175,15 @@ To check whether embeddings exist, inspect \`.gitnexus/meta.json\` — the \`sta
|
||||
|
||||
> Claude Code users: A PostToolUse hook handles this automatically after \`git commit\` and \`git merge\`.
|
||||
|
||||
## CLI
|
||||
${
|
||||
groupNames && groupNames.length > 0
|
||||
? `## Cross-Repo Groups
|
||||
|
||||
This repository is listed under GitNexus **group(s): ${groupNames.join(', ')}** (see \`~/.gitnexus/groups/\`). For blast radius across repository boundaries, use MCP tools \`group_impact\`, \`group_sync\`, \`group_query\`, \`group_contracts\`, \`group_status\`, and \`group_list\`. From the terminal: \`npx gitnexus group list\`, \`npx gitnexus group sync <name>\`, \`npx gitnexus group impact <name> --target <symbol> --repo <group-path>\`.
|
||||
|
||||
`
|
||||
: ''
|
||||
}## CLI
|
||||
|
||||
${skillsTable}
|
||||
|
||||
@@ -305,7 +333,14 @@ export async function generateAIContextFiles(
|
||||
generatedSkills?: GeneratedSkillInfo[],
|
||||
options?: AIContextOptions,
|
||||
): Promise<{ files: string[] }> {
|
||||
const content = generateGitNexusContent(projectName, stats, generatedSkills);
|
||||
const groupNames = await findGroupsContainingRegistryName(projectName);
|
||||
const content = generateGitNexusContent(
|
||||
projectName,
|
||||
stats,
|
||||
generatedSkills,
|
||||
groupNames,
|
||||
options?.noStats,
|
||||
);
|
||||
const createdFiles: string[] = [];
|
||||
|
||||
if (!options?.skipAgentsMd) {
|
||||
|
||||
@@ -20,8 +20,11 @@ import fs from 'fs/promises';
|
||||
|
||||
const HEAP_MB = 8192;
|
||||
const HEAP_FLAG = `--max-old-space-size=${HEAP_MB}`;
|
||||
/** Increase default stack size (KB) to prevent stack overflow on deep class hierarchies. */
|
||||
const STACK_KB = 4096;
|
||||
const STACK_FLAG = `--stack-size=${STACK_KB}`;
|
||||
|
||||
/** Re-exec the process with an 8GB heap if we're currently below that. */
|
||||
/** Re-exec the process with an 8GB heap and larger stack if we're currently below that. */
|
||||
function ensureHeap(): boolean {
|
||||
const nodeOpts = process.env.NODE_OPTIONS || '';
|
||||
if (nodeOpts.includes('--max-old-space-size')) return false;
|
||||
@@ -29,8 +32,13 @@ function ensureHeap(): boolean {
|
||||
const v8Heap = v8.getHeapStatistics().heap_size_limit;
|
||||
if (v8Heap >= HEAP_MB * 1024 * 1024 * 0.9) return false;
|
||||
|
||||
// --stack-size is a V8 flag not allowed in NODE_OPTIONS on Node 24+,
|
||||
// so pass it only as a direct CLI argument, not via the environment.
|
||||
const cliFlags = [HEAP_FLAG];
|
||||
if (!nodeOpts.includes('--stack-size')) cliFlags.push(STACK_FLAG);
|
||||
|
||||
try {
|
||||
execFileSync(process.execPath, [HEAP_FLAG, ...process.argv.slice(1)], {
|
||||
execFileSync(process.execPath, [...cliFlags, ...process.argv.slice(1)], {
|
||||
stdio: 'inherit',
|
||||
env: { ...process.env, NODE_OPTIONS: `${nodeOpts} ${HEAP_FLAG}`.trim() },
|
||||
});
|
||||
@@ -47,6 +55,8 @@ export interface AnalyzeOptions {
|
||||
verbose?: boolean;
|
||||
/** Skip AGENTS.md and CLAUDE.md gitnexus block updates. */
|
||||
skipAgentsMd?: boolean;
|
||||
/** Omit volatile symbol/relationship counts from AGENTS.md and CLAUDE.md. */
|
||||
noStats?: boolean;
|
||||
/** Index the folder even when no .git directory is present. */
|
||||
skipGit?: boolean;
|
||||
}
|
||||
@@ -177,6 +187,7 @@ export const analyzeCommand = async (inputPath?: string, options?: AnalyzeOption
|
||||
embeddings: options?.embeddings,
|
||||
skipGit: options?.skipGit,
|
||||
skipAgentsMd: options?.skipAgentsMd,
|
||||
noStats: options?.noStats,
|
||||
},
|
||||
{
|
||||
onProgress: (_phase, percent, message) => {
|
||||
@@ -240,7 +251,7 @@ export const analyzeCommand = async (inputPath?: string, options?: AnalyzeOption
|
||||
processes: s.processes,
|
||||
},
|
||||
skillResult.skills,
|
||||
{ skipAgentsMd: options?.skipAgentsMd },
|
||||
{ skipAgentsMd: options?.skipAgentsMd, noStats: options?.noStats },
|
||||
);
|
||||
}
|
||||
} catch {
|
||||
@@ -282,7 +293,51 @@ export const analyzeCommand = async (inputPath?: string, options?: AnalyzeOption
|
||||
console.warn = origWarn;
|
||||
console.error = origError;
|
||||
bar.stop();
|
||||
console.error(`\n Analysis failed: ${err.message}\n`);
|
||||
|
||||
const msg = err.message || String(err);
|
||||
console.error(`\n Analysis failed: ${msg}\n`);
|
||||
|
||||
// Provide helpful guidance for known failure modes
|
||||
if (
|
||||
msg.includes('Maximum call stack size exceeded') ||
|
||||
msg.includes('call stack') ||
|
||||
msg.includes('Map maximum size') ||
|
||||
msg.includes('Invalid array length') ||
|
||||
msg.includes('Invalid string length') ||
|
||||
msg.includes('allocation failed') ||
|
||||
msg.includes('heap out of memory') ||
|
||||
msg.includes('JavaScript heap')
|
||||
) {
|
||||
console.error(' This error typically occurs on very large repositories.');
|
||||
console.error(' Suggestions:');
|
||||
console.error(' 1. Add large vendored/generated directories to .gitnexusignore');
|
||||
console.error(' 2. Increase Node.js heap: NODE_OPTIONS="--max-old-space-size=16384"');
|
||||
console.error(' 3. Increase stack size: NODE_OPTIONS="--stack-size=4096"');
|
||||
console.error('');
|
||||
} else if (msg.includes('ERESOLVE') || msg.includes('Could not resolve dependency')) {
|
||||
// Note: the original arborist "Cannot destructure property 'package' of
|
||||
// 'node.target'" crash happens inside npm *before* gitnexus code runs,
|
||||
// so it can't be caught here. This branch handles dependency-resolution
|
||||
// errors that surface at runtime (e.g. dynamic require failures).
|
||||
console.error(' This looks like an npm dependency resolution issue.');
|
||||
console.error(' Suggestions:');
|
||||
console.error(' 1. Clear the npm cache: npm cache clean --force');
|
||||
console.error(' 2. Update npm: npm install -g npm@latest');
|
||||
console.error(' 3. Reinstall gitnexus: npm install -g gitnexus@latest');
|
||||
console.error(' 4. Or try npx directly: npx gitnexus@latest analyze');
|
||||
console.error('');
|
||||
} else if (
|
||||
msg.includes('MODULE_NOT_FOUND') ||
|
||||
msg.includes('Cannot find module') ||
|
||||
msg.includes('ERR_MODULE_NOT_FOUND')
|
||||
) {
|
||||
console.error(' A required module could not be loaded. The installation may be corrupt.');
|
||||
console.error(' Suggestions:');
|
||||
console.error(' 1. Reinstall: npm install -g gitnexus@latest');
|
||||
console.error(' 2. Clear cache: npm cache clean --force && npx gitnexus@latest analyze');
|
||||
console.error('');
|
||||
}
|
||||
|
||||
process.exitCode = 1;
|
||||
return;
|
||||
}
|
||||
|
||||
@@ -0,0 +1,298 @@
|
||||
// gitnexus/src/cli/group.ts
|
||||
import { createRequire } from 'node:module';
|
||||
import type { Command } from 'commander';
|
||||
|
||||
const _require = createRequire(import.meta.url);
|
||||
const yaml = _require('js-yaml') as typeof import('js-yaml');
|
||||
|
||||
export function registerGroupCommands(program: Command): void {
|
||||
const group = program
|
||||
.command('group')
|
||||
.description('Manage repository groups for cross-index impact analysis');
|
||||
|
||||
group
|
||||
.command('create <name>')
|
||||
.description('Create a new group with template group.yaml')
|
||||
.option('--force', 'Overwrite existing group')
|
||||
.action(async (name: string, opts: { force?: boolean }) => {
|
||||
const { createGroupDir, getDefaultGitnexusDir } = await import('../core/group/storage.js');
|
||||
const dir = await createGroupDir(getDefaultGitnexusDir(), name, opts.force);
|
||||
console.log(`Created group "${name}" at ${dir}`);
|
||||
console.log('Edit group.yaml to add repos, then run: gitnexus group sync ' + name);
|
||||
});
|
||||
|
||||
group
|
||||
.command('add <group> <groupPath> <registryName>')
|
||||
.description(
|
||||
'Add a repo to a group. <groupPath> = hierarchy path (e.g. hr/hiring/backend), <registryName> = name from registry',
|
||||
)
|
||||
.action(async (groupName: string, groupPath: string, registryName: string) => {
|
||||
const { getGroupDir, getDefaultGitnexusDir } = await import('../core/group/storage.js');
|
||||
const { loadGroupConfig } = await import('../core/group/config-parser.js');
|
||||
const path = await import('node:path');
|
||||
const fs = await import('node:fs/promises');
|
||||
const groupDir = getGroupDir(getDefaultGitnexusDir(), groupName);
|
||||
const config = await loadGroupConfig(groupDir);
|
||||
config.repos[groupPath] = registryName;
|
||||
|
||||
await fs.writeFile(path.join(groupDir, 'group.yaml'), yaml.dump(config), 'utf-8');
|
||||
console.log(`Added ${registryName} as "${groupPath}" to group "${groupName}"`);
|
||||
console.log(`Run: gitnexus group sync ${groupName}`);
|
||||
});
|
||||
|
||||
group
|
||||
.command('remove <group> <path>')
|
||||
.description('Remove a repo from a group')
|
||||
.action(async (groupName: string, repoPath: string) => {
|
||||
const { getGroupDir, getDefaultGitnexusDir } = await import('../core/group/storage.js');
|
||||
const { loadGroupConfig } = await import('../core/group/config-parser.js');
|
||||
const path = await import('node:path');
|
||||
const fs = await import('node:fs/promises');
|
||||
const groupDir = getGroupDir(getDefaultGitnexusDir(), groupName);
|
||||
const config = await loadGroupConfig(groupDir);
|
||||
if (!(repoPath in config.repos)) {
|
||||
console.error(`Repo path "${repoPath}" not found in group "${groupName}"`);
|
||||
process.exitCode = 1;
|
||||
return;
|
||||
}
|
||||
delete config.repos[repoPath];
|
||||
await fs.writeFile(path.join(groupDir, 'group.yaml'), yaml.dump(config), 'utf-8');
|
||||
console.log(`Removed "${repoPath}" from group "${groupName}"`);
|
||||
});
|
||||
|
||||
group
|
||||
.command('list [name]')
|
||||
.description('List all groups or details of one')
|
||||
.action(async (name?: string) => {
|
||||
const { listGroups, getDefaultGitnexusDir, getGroupDir } =
|
||||
await import('../core/group/storage.js');
|
||||
if (!name) {
|
||||
const groups = await listGroups();
|
||||
if (groups.length === 0) {
|
||||
console.log('No groups configured. Create one with: gitnexus group create <name>');
|
||||
return;
|
||||
}
|
||||
console.log('Groups:');
|
||||
groups.forEach((g) => console.log(` ${g}`));
|
||||
return;
|
||||
}
|
||||
const { loadGroupConfig } = await import('../core/group/config-parser.js');
|
||||
const groupDir = getGroupDir(getDefaultGitnexusDir(), name);
|
||||
const config = await loadGroupConfig(groupDir);
|
||||
console.log(`Group: ${config.name}`);
|
||||
if (config.description) console.log(`Description: ${config.description}`);
|
||||
console.log(`\nRepos (${Object.keys(config.repos).length}):`);
|
||||
for (const [p, id] of Object.entries(config.repos)) {
|
||||
console.log(` ${p} -> ${id}`);
|
||||
}
|
||||
if (config.links.length > 0) {
|
||||
console.log(`\nManifest links (${config.links.length}):`);
|
||||
for (const link of config.links) {
|
||||
console.log(` ${link.from} -> ${link.to} [${link.type}: ${link.contract}]`);
|
||||
}
|
||||
}
|
||||
});
|
||||
|
||||
group
|
||||
.command('status <name>')
|
||||
.description('Check staleness of group and repos')
|
||||
.action(async (name: string) => {
|
||||
const { readContractRegistry, getGroupDir, getDefaultGitnexusDir } =
|
||||
await import('../core/group/storage.js');
|
||||
const { LocalBackend } = await import('../mcp/local/local-backend.js');
|
||||
|
||||
const groupDir = getGroupDir(getDefaultGitnexusDir(), name);
|
||||
const registry = await readContractRegistry(groupDir);
|
||||
|
||||
console.log(
|
||||
`Group: ${name}${registry ? ` (last sync: ${registry.generatedAt})` : ' (never synced)'}\n`,
|
||||
);
|
||||
|
||||
const backend = new LocalBackend();
|
||||
try {
|
||||
await backend.init();
|
||||
const raw = await backend.getGroupService().groupStatus({ name });
|
||||
const st = raw as {
|
||||
repos?: Record<
|
||||
string,
|
||||
{
|
||||
indexStale: boolean;
|
||||
contractsStale: boolean;
|
||||
missing: boolean;
|
||||
commitsBehind?: number;
|
||||
}
|
||||
>;
|
||||
missingRepos?: string[];
|
||||
};
|
||||
|
||||
console.log(' Repo index / contracts staleness:');
|
||||
for (const [repoPath, row] of Object.entries(st.repos || {})) {
|
||||
if (row.missing) {
|
||||
console.log(` ${repoPath.padEnd(25)} MISSING (not in registry or unreadable)`);
|
||||
continue;
|
||||
}
|
||||
const idx = row.indexStale
|
||||
? `STALE (${row.commitsBehind ?? '?'} commits behind)`
|
||||
: 'OK ';
|
||||
const ctr = row.contractsStale ? ' CONTRACTS_STALE' : '';
|
||||
console.log(` ${repoPath.padEnd(25)} ${idx}${ctr}`);
|
||||
}
|
||||
if ((st.missingRepos || []).length > 0) {
|
||||
console.log(`\n Last sync missing repos: ${st.missingRepos!.join(', ')}`);
|
||||
}
|
||||
} finally {
|
||||
await backend.dispose().catch(() => {});
|
||||
}
|
||||
});
|
||||
|
||||
group
|
||||
.command('sync <name>')
|
||||
.description('Sync Contract Registry — extract contracts and build cross-links')
|
||||
.option('--skip-embeddings', 'Exact + BM25 only (no embedding fallback)')
|
||||
.option('--exact-only', 'Exact match only')
|
||||
.option('--allow-stale', 'Skip stale index warnings')
|
||||
.option('--verbose', 'Show each cross-link detail')
|
||||
.option('--json', 'JSON output')
|
||||
.action(async (name: string, opts: Record<string, boolean | undefined>) => {
|
||||
const { getGroupDir, getDefaultGitnexusDir } = await import('../core/group/storage.js');
|
||||
const { loadGroupConfig } = await import('../core/group/config-parser.js');
|
||||
const { syncGroup } = await import('../core/group/sync.js');
|
||||
|
||||
const groupDir = getGroupDir(getDefaultGitnexusDir(), name);
|
||||
const config = await loadGroupConfig(groupDir);
|
||||
|
||||
console.log(`Syncing group "${name}" (${Object.keys(config.repos).length} repos)...\n`);
|
||||
|
||||
const result = await syncGroup(config, {
|
||||
groupDir,
|
||||
allowStale: Boolean(opts.allowStale),
|
||||
verbose: Boolean(opts.verbose),
|
||||
skipEmbeddings: Boolean(opts.skipEmbeddings),
|
||||
exactOnly: Boolean(opts.exactOnly),
|
||||
});
|
||||
|
||||
if (opts.json) {
|
||||
console.log(JSON.stringify(result, null, 2));
|
||||
} else {
|
||||
console.log(`\nMatching cascade:`);
|
||||
const exactLinks = result.crossLinks.filter((l) => l.matchType === 'exact');
|
||||
console.log(` exact: ${exactLinks.length} cross-links (confidence 1.0)`);
|
||||
console.log(` unmatched: ${result.unmatched.length} contracts`);
|
||||
console.log(
|
||||
`\nWrote contracts.json (${result.contracts.length} contracts, ${result.crossLinks.length} cross-links)`,
|
||||
);
|
||||
}
|
||||
});
|
||||
|
||||
group
|
||||
.command('query <name> <query>')
|
||||
.description('Search execution flows across all repos in a group')
|
||||
.option('--subgroup <path>', 'Limit search scope')
|
||||
.option('--limit <n>', 'Max merged results', '5')
|
||||
.option('--json', 'JSON output')
|
||||
.action(
|
||||
async (
|
||||
name: string,
|
||||
queryText: string,
|
||||
opts: Record<string, string | boolean | undefined>,
|
||||
) => {
|
||||
const { LocalBackend } = await import('../mcp/local/local-backend.js');
|
||||
|
||||
const limit = parseInt(String(opts.limit ?? '5'), 10) || 5;
|
||||
const subgroup = opts.subgroup as string | undefined;
|
||||
const backend = new LocalBackend();
|
||||
try {
|
||||
await backend.init();
|
||||
|
||||
console.log(`Searching "${queryText}" across group "${name}"...\n`);
|
||||
|
||||
const raw = await backend.getGroupService().groupQuery({
|
||||
name,
|
||||
query: queryText,
|
||||
limit,
|
||||
subgroup,
|
||||
});
|
||||
const merged = raw as {
|
||||
results: Array<Record<string, unknown>>;
|
||||
per_repo: Array<{ repo: string; count: number }>;
|
||||
};
|
||||
|
||||
if (opts.json) {
|
||||
console.log(JSON.stringify(raw, null, 2));
|
||||
} else {
|
||||
console.log(`Results (top ${merged.results.length}):\n`);
|
||||
for (const p of merged.results) {
|
||||
const label = (p.summary || p.heuristicLabel || p.name || 'unnamed') as string;
|
||||
console.log(` [${p._repo}] ${label} (rrf: ${(p._rrf_score as number).toFixed(4)})`);
|
||||
}
|
||||
if (merged.results.length === 0) {
|
||||
console.log(' No matching execution flows found.');
|
||||
}
|
||||
}
|
||||
} finally {
|
||||
await backend.dispose().catch(() => {});
|
||||
}
|
||||
},
|
||||
);
|
||||
|
||||
group
|
||||
.command('contracts <name>')
|
||||
.description('Inspect Contract Registry')
|
||||
.option('--type <type>', 'Filter by contract type')
|
||||
.option('--repo <repo>', 'Filter by repo')
|
||||
.option('--unmatched', 'Show only unmatched contracts')
|
||||
.option('--json', 'JSON output')
|
||||
.action(async (name: string, opts: Record<string, string | boolean | undefined>) => {
|
||||
const { LocalBackend } = await import('../mcp/local/local-backend.js');
|
||||
|
||||
const backend = new LocalBackend();
|
||||
try {
|
||||
await backend.init();
|
||||
const raw = await backend.getGroupService().groupContracts({
|
||||
name,
|
||||
type: opts.type as string | undefined,
|
||||
repo: opts.repo as string | undefined,
|
||||
unmatchedOnly: Boolean(opts.unmatched),
|
||||
});
|
||||
|
||||
if (raw && typeof raw === 'object' && 'error' in raw) {
|
||||
console.error(String((raw as { error: string }).error));
|
||||
process.exitCode = 1;
|
||||
return;
|
||||
}
|
||||
|
||||
const { contracts, crossLinks } = raw as {
|
||||
contracts: Array<{
|
||||
role: string;
|
||||
contractId: string;
|
||||
repo: string;
|
||||
symbolRef: { name: string };
|
||||
}>;
|
||||
crossLinks: Array<{
|
||||
from: { repo: string };
|
||||
to: { repo: string };
|
||||
matchType: string;
|
||||
confidence: number;
|
||||
contractId: string;
|
||||
}>;
|
||||
};
|
||||
|
||||
if (opts.json) {
|
||||
console.log(JSON.stringify({ contracts, crossLinks }, null, 2));
|
||||
} else {
|
||||
console.log(`Contracts (${contracts.length}):`);
|
||||
for (const c of contracts) {
|
||||
console.log(` [${c.role}] ${c.contractId} (${c.repo}) ${c.symbolRef.name}`);
|
||||
}
|
||||
console.log(`\nCross-links (${crossLinks.length}):`);
|
||||
for (const l of crossLinks) {
|
||||
console.log(
|
||||
` ${l.from.repo} -> ${l.to.repo} [${l.matchType}, conf=${l.confidence}] ${l.contractId}`,
|
||||
);
|
||||
}
|
||||
}
|
||||
} finally {
|
||||
await backend.dispose().catch(() => {});
|
||||
}
|
||||
});
|
||||
}
|
||||
@@ -6,6 +6,7 @@
|
||||
import { Command } from 'commander';
|
||||
import { createRequire } from 'node:module';
|
||||
import { createLazyAction } from './lazy-action.js';
|
||||
import { registerGroupCommands } from './group.js';
|
||||
|
||||
const _require = createRequire(import.meta.url);
|
||||
const pkg = _require('../../package.json');
|
||||
@@ -25,6 +26,7 @@ program
|
||||
.option('--embeddings', 'Enable embedding generation for semantic search (off by default)')
|
||||
.option('--skills', 'Generate repo-specific skill files from detected communities')
|
||||
.option('--skip-agents-md', 'Skip updating the gitnexus section in AGENTS.md and CLAUDE.md')
|
||||
.option('--no-stats', 'Omit volatile file/symbol counts from AGENTS.md and CLAUDE.md')
|
||||
.option('--skip-git', 'Index a folder without requiring a .git directory')
|
||||
.option('-v, --verbose', 'Enable verbose ingestion warnings (default: false)')
|
||||
.addHelpText(
|
||||
@@ -148,4 +150,6 @@ program
|
||||
.option('--idle-timeout <seconds>', 'Auto-shutdown after N seconds idle (0 = disabled)', '0')
|
||||
.action(createLazyAction(() => import('./eval-server.js'), 'evalServerCommand'));
|
||||
|
||||
registerGroupCommands(program);
|
||||
|
||||
program.parse(process.argv);
|
||||
|
||||
@@ -14,7 +14,10 @@ process.on('unhandledRejection', (reason: any) => {
|
||||
|
||||
export const serveCommand = async (options?: { port?: string; host?: string }) => {
|
||||
const port = Number(options?.port ?? 4747);
|
||||
const host = options?.host ?? '127.0.0.1';
|
||||
// Default to 'localhost' so the OS decides whether to bind to 127.0.0.1 or
|
||||
// ::1 based on system configuration, avoiding spurious CORS errors when the
|
||||
// hosted frontend at gitnexus.vercel.app connects to localhost.
|
||||
const host = options?.host ?? 'localhost';
|
||||
|
||||
try {
|
||||
await createServer(port, host);
|
||||
|
||||
@@ -9,7 +9,7 @@
|
||||
import fs from 'fs/promises';
|
||||
import path from 'path';
|
||||
import os from 'os';
|
||||
import { execFile } from 'child_process';
|
||||
import { execFile, execFileSync } from 'child_process';
|
||||
import { promisify } from 'util';
|
||||
import { fileURLToPath } from 'url';
|
||||
import { glob } from 'glob';
|
||||
@@ -25,11 +25,44 @@ interface SetupResult {
|
||||
errors: string[];
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolve the absolute path to the `gitnexus` binary if it's installed
|
||||
* globally (or via npm -g / yarn global). Returns null when not found.
|
||||
*/
|
||||
function resolveGitnexusBin(): string | null {
|
||||
try {
|
||||
const cmd = process.platform === 'win32' ? 'where' : 'which';
|
||||
const resolved = execFileSync(cmd, ['gitnexus'], {
|
||||
encoding: 'utf-8',
|
||||
timeout: 5000,
|
||||
stdio: ['ignore', 'pipe', 'ignore'],
|
||||
})
|
||||
.split('\n')[0]
|
||||
.trim();
|
||||
return resolved || null;
|
||||
} catch {
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* The MCP server entry for all editors.
|
||||
* On Windows, npx must be invoked via cmd /c since it's a .cmd script.
|
||||
*
|
||||
* Prefers the globally-installed `gitnexus` binary (starts in ~1 s) over
|
||||
* `npx -y gitnexus@latest` (cold-cache install of native deps can take
|
||||
* >60 s, exceeding Claude Code's 30 s MCP connection timeout).
|
||||
*
|
||||
* Falls back to npx when the binary isn't on PATH — e.g. first-time
|
||||
* users who ran `npx gitnexus analyze` but haven't done `npm i -g`.
|
||||
*/
|
||||
function getMcpEntry() {
|
||||
const bin = resolveGitnexusBin();
|
||||
|
||||
if (bin) {
|
||||
return { command: bin, args: ['mcp'] };
|
||||
}
|
||||
|
||||
// Fallback: npx (works without a global install, but slow cold-start)
|
||||
if (process.platform === 'win32') {
|
||||
return {
|
||||
command: 'cmd',
|
||||
@@ -232,7 +265,7 @@ async function setupOpenCode(result: SetupResult): Promise<void> {
|
||||
return;
|
||||
}
|
||||
|
||||
const configPath = path.join(opencodeDir, 'config.json');
|
||||
const configPath = path.join(opencodeDir, 'opencode.json');
|
||||
try {
|
||||
const existing = await readJsonFile(configPath);
|
||||
const config = existing || {};
|
||||
|
||||
+18
-58
@@ -232,65 +232,28 @@ export const wikiCommand = async (inputPath?: string, options?: WikiCommandOptio
|
||||
|
||||
llmConfig = { ...llmConfig, provider: 'cursor', model, apiKey: '', baseUrl: '' };
|
||||
} else if (choice === '3') {
|
||||
// Azure OpenAI guided setup
|
||||
console.log('\n Azure OpenAI setup.');
|
||||
console.log(
|
||||
' You need: your resource name, deployment name, and API key from the Azure portal.\n',
|
||||
);
|
||||
// Azure OpenAI guided setup — minimal prompts
|
||||
console.log('\n Azure OpenAI setup.\n');
|
||||
|
||||
const resourceName = (
|
||||
await prompt(' Azure resource name (e.g. my-openai-resource): ')
|
||||
).trim();
|
||||
if (!resourceName) {
|
||||
console.log('\n No resource name provided. Aborting.\n');
|
||||
const endpoint = (
|
||||
await prompt(' Endpoint URL (e.g. https://my-resource.openai.azure.com): ')
|
||||
)
|
||||
.trim()
|
||||
.replace(/\/+$/, '');
|
||||
if (!endpoint) {
|
||||
console.log('\n No endpoint provided. Aborting.\n');
|
||||
process.exitCode = 1;
|
||||
return;
|
||||
}
|
||||
|
||||
const deploymentName = (
|
||||
await prompt(' Deployment name (the name you gave your model deployment): ')
|
||||
).trim();
|
||||
const deploymentName = (await prompt(' Deployment name: ')).trim();
|
||||
if (!deploymentName) {
|
||||
console.log('\n No deployment name provided. Aborting.\n');
|
||||
process.exitCode = 1;
|
||||
return;
|
||||
}
|
||||
|
||||
// Offer v1 or legacy URL
|
||||
console.log('\n API format:');
|
||||
console.log(' [1] v1 API — recommended (no api-version needed)');
|
||||
console.log(' [2] Legacy — uses api-version query param\n');
|
||||
const apiFormat = await prompt(' Select format (1/2, default: 1): ');
|
||||
|
||||
let azureApiVersion: string | undefined;
|
||||
let azureBaseUrl: string;
|
||||
if (apiFormat === '2') {
|
||||
const versionInput = await prompt(' api-version (default: 2024-10-21): ');
|
||||
azureApiVersion = versionInput || '2024-10-21';
|
||||
azureBaseUrl = `https://${resourceName}.openai.azure.com/openai/deployments/${deploymentName}`;
|
||||
} else {
|
||||
azureBaseUrl = `https://${resourceName}.openai.azure.com/openai/v1`;
|
||||
azureApiVersion = undefined;
|
||||
}
|
||||
|
||||
defaultModel = deploymentName;
|
||||
|
||||
// Ask if this is a reasoning model deployment
|
||||
const reasoningAnswer = await prompt(
|
||||
' Is this a reasoning model (o1, o3, o4-mini)? (y/N): ',
|
||||
);
|
||||
const isReasoningModelDeployment = ['y', 'yes'].includes(reasoningAnswer.toLowerCase());
|
||||
|
||||
if (isReasoningModelDeployment) {
|
||||
console.log(
|
||||
' Note: temperature and max_tokens will be omitted for this deployment (Azure reasoning model requirement).\n',
|
||||
);
|
||||
}
|
||||
|
||||
const modelInput = await prompt(` Model / deployment name (default: ${defaultModel}): `);
|
||||
const model = modelInput || defaultModel;
|
||||
|
||||
// API key
|
||||
// API key — use env var if available
|
||||
const envKey = process.env.GITNEXUS_API_KEY || process.env.OPENAI_API_KEY || '';
|
||||
let azureKey: string;
|
||||
if (envKey) {
|
||||
@@ -311,26 +274,23 @@ export const wikiCommand = async (inputPath?: string, options?: WikiCommandOptio
|
||||
return;
|
||||
}
|
||||
|
||||
// Save Azure config including optional apiVersion and isReasoningModel
|
||||
const azureConfig: Parameters<typeof saveCLIConfig>[0] = {
|
||||
// Always use v1 API format — no need for api-version
|
||||
const azureBaseUrl = `${endpoint}/openai/v1`;
|
||||
|
||||
await saveCLIConfig({
|
||||
apiKey: azureKey,
|
||||
baseUrl: azureBaseUrl,
|
||||
model,
|
||||
model: deploymentName,
|
||||
provider: 'azure',
|
||||
isReasoningModel: isReasoningModelDeployment,
|
||||
};
|
||||
if (azureApiVersion) azureConfig.apiVersion = azureApiVersion;
|
||||
await saveCLIConfig(azureConfig);
|
||||
});
|
||||
console.log(' Config saved to ~/.gitnexus/config.json\n');
|
||||
|
||||
llmConfig = {
|
||||
...llmConfig,
|
||||
apiKey: azureKey,
|
||||
baseUrl: azureBaseUrl,
|
||||
model,
|
||||
model: deploymentName,
|
||||
provider: 'azure',
|
||||
apiVersion: azureApiVersion,
|
||||
isReasoningModel: isReasoningModelDeployment,
|
||||
};
|
||||
} else {
|
||||
// OpenAI-compatible provider (OpenAI, OpenRouter, Custom)
|
||||
|
||||
@@ -398,11 +398,16 @@ export const createIgnoreFilter = async (repoPath: string, options?: IgnoreOptio
|
||||
// defense-in-depth — do not remove `dot: false` assuming this covers it.
|
||||
if (DEFAULT_IGNORE_LIST.has(p.name)) return true;
|
||||
// Check against .gitignore / .gitnexusignore patterns.
|
||||
// Test both bare path and path with trailing slash to handle
|
||||
// bare-name patterns (e.g. `local`) and dir-only patterns (e.g. `local/`).
|
||||
// Since childrenIgnored is only called for directories, always test with
|
||||
// a trailing slash. This ensures directory-only negation patterns (e.g.
|
||||
// `!iOS/`) are applied correctly — without the slash, `ig.ignores('iOS')`
|
||||
// treats the path as a file and misses the negation.
|
||||
// Bare-name patterns (e.g. `local`) still match `local/` per gitignore spec:
|
||||
// the `ignore` package normalizes `dir` and `dir/` to match directories.
|
||||
// See: https://github.com/kaelzhang/node-ignore#2-filenames-and-dirnames
|
||||
if (ig) {
|
||||
const rel = p.relative();
|
||||
if (rel && (ig.ignores(rel) || ig.ignores(rel + '/'))) return true;
|
||||
if (rel && ig.ignores(rel + '/')) return true;
|
||||
}
|
||||
return false;
|
||||
},
|
||||
|
||||
@@ -93,7 +93,7 @@ export async function augment(pattern: string, cwd?: string): Promise<string> {
|
||||
if (!repo) return '';
|
||||
|
||||
// Lazy-load lbug adapter (skip unnecessary init)
|
||||
const { initLbug, executeQuery, isLbugReady } = await import('../../mcp/core/lbug-adapter.js');
|
||||
const { initLbug, executeQuery, isLbugReady } = await import('../lbug/pool-adapter.js');
|
||||
const { searchFTSFromLbug } = await import('../search/bm25-index.js');
|
||||
|
||||
const repoId = repo.name.toLowerCase();
|
||||
|
||||
@@ -100,8 +100,8 @@ const batchInsertEmbeddings = async (
|
||||
) => Promise<void>,
|
||||
updates: Array<{ id: string; embedding: number[] }>,
|
||||
): Promise<void> => {
|
||||
// INSERT into separate embedding table - much more memory efficient!
|
||||
const cypher = `CREATE (e:CodeEmbedding {nodeId: $nodeId, embedding: $embedding})`;
|
||||
// MERGE instead of CREATE — idempotent, handles concurrent analyzes and partial prior runs
|
||||
const cypher = `MERGE (e:CodeEmbedding {nodeId: $nodeId}) SET e.embedding = $embedding`;
|
||||
const paramsList = updates.map((u) => ({ nodeId: u.id, embedding: u.embedding }));
|
||||
await executeWithReusedStatement(cypher, paramsList);
|
||||
};
|
||||
|
||||
@@ -0,0 +1,39 @@
|
||||
/**
|
||||
* Git working tree vs index commit staleness (used by MCP resources, group status, etc.).
|
||||
* Lives in core/ so application code does not depend on the MCP package layer.
|
||||
*/
|
||||
|
||||
import { execFileSync } from 'node:child_process';
|
||||
|
||||
export interface StalenessInfo {
|
||||
isStale: boolean;
|
||||
commitsBehind: number;
|
||||
hint?: string;
|
||||
}
|
||||
|
||||
/**
|
||||
* Check how many commits the index is behind HEAD (synchronous; uses git CLI).
|
||||
*/
|
||||
export function checkStaleness(repoPath: string, lastCommit: string): StalenessInfo {
|
||||
try {
|
||||
const result = execFileSync('git', ['rev-list', '--count', `${lastCommit}..HEAD`], {
|
||||
cwd: repoPath,
|
||||
encoding: 'utf-8',
|
||||
stdio: ['pipe', 'pipe', 'pipe'],
|
||||
}).trim();
|
||||
|
||||
const commitsBehind = parseInt(result, 10) || 0;
|
||||
|
||||
if (commitsBehind > 0) {
|
||||
return {
|
||||
isStale: true,
|
||||
commitsBehind,
|
||||
hint: `⚠️ Index is ${commitsBehind} commit${commitsBehind > 1 ? 's' : ''} behind HEAD. Run analyze tool to update.`,
|
||||
};
|
||||
}
|
||||
|
||||
return { isStale: false, commitsBehind: 0 };
|
||||
} catch {
|
||||
return { isStale: false, commitsBehind: 0 };
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,139 @@
|
||||
# Group Analysis Pipeline
|
||||
|
||||
Flow chart of the cross-repo contract extraction + matching pipeline.
|
||||
This covers what runs **inside this PR** (extractors + manifest) and
|
||||
the downstream handoff to the bridge storage (PR #795) and
|
||||
cross-impact query (PR #606).
|
||||
|
||||
## High-level overview
|
||||
|
||||
```mermaid
|
||||
flowchart TD
|
||||
A[group.yaml] --> B[GroupConfig parser]
|
||||
B --> C{For each repo<br/>in group}
|
||||
C --> D[Per-repo LadybugDB<br/>indexed by main pipeline]
|
||||
|
||||
D --> E1[TopicExtractor]
|
||||
D --> E2[HttpRouteExtractor]
|
||||
D --> E3[GrpcExtractor]
|
||||
|
||||
E1 --> F[ExtractedContract array<br/>per repo]
|
||||
E2 --> F
|
||||
E3 --> F
|
||||
|
||||
B --> M[ManifestExtractor]
|
||||
M --> G[Manifest contracts<br/>+ cross-links]
|
||||
|
||||
F --> H[Contract matching<br/>exact + wildcard]
|
||||
G --> H
|
||||
|
||||
H --> I[(bridge.lbug<br/>#795)]
|
||||
|
||||
I --> J[runGroupImpact<br/>#606]
|
||||
J --> K[CrossRepoImpact]
|
||||
```
|
||||
|
||||
## Per-repo extractor pipeline
|
||||
|
||||
Each extractor under `src/core/group/extractors/` follows the same
|
||||
two-strategy shape:
|
||||
|
||||
```mermaid
|
||||
flowchart TD
|
||||
R[RepoHandle + CypherExecutor<br/>for this repo] --> S{Graph-assisted<br/>Strategy A<br/>available?}
|
||||
|
||||
S -->|yes| A1[Cypher query against<br/>per-repo LadybugDB]
|
||||
A1 --> A2{non-empty<br/>result?}
|
||||
A2 -->|yes| OUT[ExtractedContract array]
|
||||
A2 -->|no| B1
|
||||
|
||||
S -->|no| B1[Source-scan Strategy B]
|
||||
B1 --> B2[glob repo source files]
|
||||
B2 --> B3{ext in registry?}
|
||||
B3 -->|yes| B4[Per-language plugin<br/>scan parsed tree]
|
||||
B3 -->|no| SKIP[skip file]
|
||||
B4 --> OUT
|
||||
|
||||
SKIP --> B2
|
||||
```
|
||||
|
||||
**Strategy A** (graph-assisted) uses Cypher over edges already produced
|
||||
by the main ingestion pipeline:
|
||||
- HTTP: `HANDLES_ROUTE` / `FETCHES` edges from `(File)-[]->(Route)`
|
||||
- topic: none (pipeline doesn't yet produce topic nodes — Strategy B only)
|
||||
- gRPC: none (Strategy B + proto map only)
|
||||
|
||||
**Strategy B** (source-scan) is 100% tree-sitter based after this PR.
|
||||
Each `*-patterns/<lang>.ts` plugin owns its grammar + S-expression
|
||||
queries; the top-level orchestrator imports neither.
|
||||
|
||||
## Plugin architecture
|
||||
|
||||
```mermaid
|
||||
flowchart LR
|
||||
O[Orchestrator<br/>topic|http|grpc-extractor.ts] --> REG[REGISTRY<br/>*-patterns/index.ts]
|
||||
REG --> P1[java.ts<br/>tree-sitter-java]
|
||||
REG --> P2[go.ts<br/>tree-sitter-go]
|
||||
REG --> P3[python.ts<br/>tree-sitter-python]
|
||||
REG --> P4[node.ts<br/>JS + TS + TSX]
|
||||
REG --> P5[php.ts<br/>tree-sitter-php<br/>HTTP only]
|
||||
REG --> P6[proto.ts<br/>tree-sitter-proto<br/>gRPC only, optional]
|
||||
|
||||
P1 --> SCAN[tree-sitter-scanner.ts<br/>compilePatterns + runCompiledPatterns]
|
||||
P2 --> SCAN
|
||||
P3 --> SCAN
|
||||
P4 --> SCAN
|
||||
P5 --> SCAN
|
||||
P6 --> SCAN
|
||||
|
||||
SCAN --> DET[Detection objects<br/>TopicMeta / HttpDetection / GrpcDetection]
|
||||
DET --> O
|
||||
O --> CT[ExtractedContract array]
|
||||
```
|
||||
|
||||
The orchestrator never imports a grammar. Adding a new language /
|
||||
framework = drop one file in `*-patterns/`, register it in
|
||||
`index.ts`. No orchestrator edits required.
|
||||
|
||||
## Manifest extraction
|
||||
|
||||
```mermaid
|
||||
flowchart TD
|
||||
Y[group.yaml links] --> ME[ManifestExtractor]
|
||||
ME --> LOOP{for each link}
|
||||
LOOP --> RES[resolveSymbol<br/>label-scoped Cypher]
|
||||
RES --> OK{found?}
|
||||
OK -->|yes| REF[real symbol uid + ref]
|
||||
OK -->|no| SYN[synthetic uid<br/>manifest::repo::cid]
|
||||
|
||||
REF --> EMIT[emit provider + consumer<br/>Contract objects<br/>+ CrossLink]
|
||||
SYN --> EMIT
|
||||
|
||||
EMIT --> BRIDGE[(bridge.lbug<br/>#795)]
|
||||
```
|
||||
|
||||
Label-scoped queries in `resolveSymbol` keep accidental cross-matches
|
||||
out:
|
||||
- `topic` → `(n:Function|Method|Class|Interface)`
|
||||
- `grpc` method → `(n:Function|Method)`, service → `(n:Class|Interface)`
|
||||
- `lib` → `(n:Package|Module)`
|
||||
|
||||
## Cross-impact query (PR #606)
|
||||
|
||||
```mermaid
|
||||
flowchart TD
|
||||
U[User changes symbol S<br/>in repo R] --> LI[Local impact engine<br/>per-repo uid expansion]
|
||||
LI --> IDS[Affected uid set]
|
||||
|
||||
IDS --> BR[Bridge query<br/>MATCH Contract WHERE uid IN ids]
|
||||
BR --> CL[CrossLink traversal]
|
||||
CL --> OTHER[Matching contract in<br/>other repo]
|
||||
|
||||
OTHER --> FE[Fan-out impact<br/>to consuming repo]
|
||||
FE --> OUT[CrossRepoImpact<br/>per affected repo]
|
||||
```
|
||||
|
||||
The bridge stores every extracted contract keyed by `symbolUid`.
|
||||
Manifest-sourced contracts use the synthetic uid form so both sides
|
||||
of the `(local impact) ↔ (bridge query)` join derive the same uid
|
||||
without coordinating through any shared state.
|
||||
@@ -0,0 +1,588 @@
|
||||
import fsp from 'node:fs/promises';
|
||||
import path from 'node:path';
|
||||
import { createHash } from 'node:crypto';
|
||||
import lbug from '@ladybugdb/core';
|
||||
import type { LbugValue } from '@ladybugdb/core';
|
||||
import type { BridgeHandle, BridgeMeta, StoredContract, CrossLink, RepoSnapshot } from './types.js';
|
||||
import { BRIDGE_SCHEMA_QUERIES, BRIDGE_SCHEMA_VERSION } from './bridge-schema.js';
|
||||
import { dedupeContracts, dedupeCrossLinks } from './normalization.js';
|
||||
|
||||
export function contractNodeId(
|
||||
repo: string,
|
||||
contractId: string,
|
||||
role: string,
|
||||
filePath: string,
|
||||
): string {
|
||||
return createHash('sha256').update(`${repo}\0${contractId}\0${role}\0${filePath}`).digest('hex');
|
||||
}
|
||||
|
||||
/* ------------------------------------------------------------------ */
|
||||
/* ContractLookupIndex — in-memory lookup for findContractNode */
|
||||
/* ------------------------------------------------------------------ */
|
||||
|
||||
/**
|
||||
* In-memory index of contract node IDs keyed three ways, mirroring the
|
||||
* three-tier fallback lookup in {@link findContractNode}. Built once per
|
||||
* `writeBridge` call after all contracts are successfully inserted, then
|
||||
* consulted for every cross-link — which eliminates the former N+1 query
|
||||
* pattern (up to `6 × cross-links` DB round-trips) and turns cross-link
|
||||
* resolution into constant-time per link.
|
||||
*
|
||||
* Keys are deliberately flat strings (not tuples) so `Map<string, ...>`
|
||||
* works; the separator `\0` can't occur in any legal repo path / file
|
||||
* path / symbol identifier, which makes the encoding injection-safe.
|
||||
*/
|
||||
export interface ContractLookupIndex {
|
||||
/** tier 1: `repo + role + symbolUid` → contract node id */
|
||||
byUid: Map<string, string>;
|
||||
/** tier 2: `repo + role + filePath + symbolName` → contract node id */
|
||||
byRef: Map<string, string>;
|
||||
/** tier 3: `repo + role + filePath` → list of contract node ids in that file */
|
||||
byFile: Map<string, string[]>;
|
||||
}
|
||||
|
||||
export function createContractLookupIndex(): ContractLookupIndex {
|
||||
return {
|
||||
byUid: new Map(),
|
||||
byRef: new Map(),
|
||||
byFile: new Map(),
|
||||
};
|
||||
}
|
||||
|
||||
function uidKey(repo: string, role: string, symbolUid: string): string {
|
||||
return `${repo}\0${role}\0${symbolUid}`;
|
||||
}
|
||||
|
||||
function refKey(repo: string, role: string, filePath: string, symbolName: string): string {
|
||||
return `${repo}\0${role}\0${filePath}\0${symbolName}`;
|
||||
}
|
||||
|
||||
function fileKey(repo: string, role: string, filePath: string): string {
|
||||
return `${repo}\0${role}\0${filePath}`;
|
||||
}
|
||||
|
||||
/**
|
||||
* Add a successfully-inserted contract to the lookup index. Must be called
|
||||
* AFTER the DB insert succeeds (not before) so failed inserts don't poison
|
||||
* the index and cause cross-links to point at non-existent rows.
|
||||
*/
|
||||
export function indexContract(
|
||||
index: ContractLookupIndex,
|
||||
contract: StoredContract,
|
||||
nodeId: string,
|
||||
): void {
|
||||
if (contract.symbolUid) {
|
||||
index.byUid.set(uidKey(contract.repo, contract.role, contract.symbolUid), nodeId);
|
||||
}
|
||||
index.byRef.set(
|
||||
refKey(contract.repo, contract.role, contract.symbolRef.filePath, contract.symbolRef.name),
|
||||
nodeId,
|
||||
);
|
||||
const fk = fileKey(contract.repo, contract.role, contract.symbolRef.filePath);
|
||||
const existing = index.byFile.get(fk);
|
||||
if (existing) {
|
||||
existing.push(nodeId);
|
||||
} else {
|
||||
index.byFile.set(fk, [nodeId]);
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolve a cross-link endpoint (consumer or provider reference) to an
|
||||
* already-inserted contract node id. Returns `null` if no match — the
|
||||
* caller is expected to count that as a dropped link in `WriteBridgeReport`.
|
||||
*
|
||||
* The resolution order matches the pre-cache DB-query behavior:
|
||||
* 1. exact `symbolUid` match in the same `(repo, role)` scope
|
||||
* 2. exact `(filePath, symbolName)` match
|
||||
* 3. if exactly one contract lives in the file → that one (fallback for
|
||||
* legacy graph-assisted extractors that couldn't resolve a symbol name)
|
||||
*
|
||||
* This is a pure function — no I/O, no DB — so it's trivial to unit-test
|
||||
* in isolation (which was the reviewer's main clean-code concern on the
|
||||
* original 35-line inner closure in `writeBridge`).
|
||||
*/
|
||||
export function findContractNode(
|
||||
index: ContractLookupIndex,
|
||||
repo: string,
|
||||
role: 'consumer' | 'provider',
|
||||
symbolUid: string,
|
||||
filePath: string,
|
||||
symbolName: string,
|
||||
): string | null {
|
||||
if (symbolUid) {
|
||||
const uidHit = index.byUid.get(uidKey(repo, role, symbolUid));
|
||||
if (uidHit !== undefined) return uidHit;
|
||||
}
|
||||
|
||||
const refHit = index.byRef.get(refKey(repo, role, filePath, symbolName));
|
||||
if (refHit !== undefined) return refHit;
|
||||
|
||||
const fileCandidates = index.byFile.get(fileKey(repo, role, filePath));
|
||||
if (fileCandidates && fileCandidates.length === 1) return fileCandidates[0];
|
||||
|
||||
return null;
|
||||
}
|
||||
|
||||
export async function openBridgeDb(dbPath: string): Promise<BridgeHandle> {
|
||||
const parentDir = path.dirname(dbPath);
|
||||
await fsp.mkdir(parentDir, { recursive: true });
|
||||
const db = new lbug.Database(dbPath, 0, false, false); // writable
|
||||
const conn = new lbug.Connection(db);
|
||||
return { _db: db, _conn: conn, groupDir: parentDir } as BridgeHandle;
|
||||
}
|
||||
|
||||
/**
|
||||
* LadybugDB returns an error whose message contains this substring when a
|
||||
* CREATE NODE TABLE or CREATE REL TABLE statement hits an already-existing
|
||||
* table. LadybugDB DDL doesn't support IF NOT EXISTS, and its JS driver
|
||||
* doesn't expose typed error codes, so we match on the message substring —
|
||||
* the same pattern used by `core/lbug/lbug-adapter.ts`. If a future
|
||||
* LadybugDB release changes the wording, update this constant.
|
||||
*/
|
||||
const LBUG_ALREADY_EXISTS_MSG = 'already exists';
|
||||
|
||||
export async function ensureBridgeSchema(handle: BridgeHandle): Promise<void> {
|
||||
const conn = handle._conn as lbug.Connection;
|
||||
for (const q of BRIDGE_SCHEMA_QUERIES) {
|
||||
try {
|
||||
await conn.query(q);
|
||||
} catch (err: unknown) {
|
||||
const msg = err instanceof Error ? err.message : String(err);
|
||||
if (!msg.includes(LBUG_ALREADY_EXISTS_MSG)) throw err;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
export async function queryBridge<T>(
|
||||
handle: BridgeHandle,
|
||||
cypher: string,
|
||||
params?: Record<string, LbugValue>,
|
||||
): Promise<T[]> {
|
||||
const conn = handle._conn as lbug.Connection;
|
||||
if (params && Object.keys(params).length > 0) {
|
||||
const stmt = await conn.prepare(cypher);
|
||||
if (!stmt.isSuccess()) {
|
||||
const errMsg = await stmt.getErrorMessage();
|
||||
throw new Error(`Bridge query prepare failed: ${errMsg}`);
|
||||
}
|
||||
const queryResult = await conn.execute(stmt, params);
|
||||
const result = unwrapQueryResult(queryResult);
|
||||
return (await result.getAll()) as T[];
|
||||
}
|
||||
const queryResult = await conn.query(cypher);
|
||||
const result = unwrapQueryResult(queryResult);
|
||||
return (await result.getAll()) as T[];
|
||||
}
|
||||
|
||||
/**
|
||||
* LadybugDB's `conn.query` / `conn.execute` can return either a single
|
||||
* `QueryResult` (for a single statement) or an array of them (when a
|
||||
* multi-statement script is dispatched). We always pass a single statement,
|
||||
* so the array form is a wrapper we unwrap here — but an empty top-level
|
||||
* array would cause `.getAll()` on `undefined` and crash with a confusing
|
||||
* stack. Throwing an explicit error makes a driver-contract regression
|
||||
* visible immediately instead of masking it.
|
||||
*/
|
||||
function unwrapQueryResult(queryResult: lbug.QueryResult | lbug.QueryResult[]): lbug.QueryResult {
|
||||
if (Array.isArray(queryResult)) {
|
||||
if (queryResult.length === 0) {
|
||||
throw new Error('Bridge query returned an empty QueryResult array');
|
||||
}
|
||||
return queryResult[0];
|
||||
}
|
||||
return queryResult;
|
||||
}
|
||||
|
||||
export async function closeBridgeDb(handle: BridgeHandle): Promise<void> {
|
||||
try {
|
||||
await (handle._conn as lbug.Connection).close();
|
||||
} catch {
|
||||
/* ignore */
|
||||
}
|
||||
try {
|
||||
await (handle._db as lbug.Database).close();
|
||||
} catch {
|
||||
/* ignore */
|
||||
}
|
||||
}
|
||||
|
||||
/* ------------------------------------------------------------------ */
|
||||
/* retryRename — handles transient EBUSY/EPERM/EACCES on Windows */
|
||||
/* ------------------------------------------------------------------ */
|
||||
|
||||
const RETRY_CODES = new Set(['EBUSY', 'EPERM', 'EACCES']);
|
||||
|
||||
export async function retryRename(src: string, dst: string, attempts = 3): Promise<void> {
|
||||
for (let i = 1; i <= attempts; i++) {
|
||||
try {
|
||||
await fsp.rename(src, dst);
|
||||
return;
|
||||
} catch (err: unknown) {
|
||||
const code = (err as NodeJS.ErrnoException).code;
|
||||
if (!code || !RETRY_CODES.has(code) || i === attempts) throw err;
|
||||
await new Promise((r) => setTimeout(r, 100 * Math.pow(2, i - 1)));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/* ------------------------------------------------------------------ */
|
||||
/* writeBridgeMeta / readBridgeMeta */
|
||||
/* ------------------------------------------------------------------ */
|
||||
|
||||
export async function writeBridgeMeta(groupDir: string, meta: BridgeMeta): Promise<void> {
|
||||
const target = path.join(groupDir, 'meta.json');
|
||||
const tmp = `${target}.tmp.${Date.now()}`;
|
||||
await fsp.writeFile(tmp, JSON.stringify(meta, null, 2), 'utf-8');
|
||||
// Use retryRename for consistency with writeBridge's atomic swap — on
|
||||
// Windows a concurrent reader can cause EBUSY/EPERM even on a tiny
|
||||
// meta.json, and we don't want meta write to be less robust than the
|
||||
// bridge.lbug swap it accompanies.
|
||||
await retryRename(tmp, target);
|
||||
}
|
||||
|
||||
export async function readBridgeMeta(groupDir: string): Promise<BridgeMeta> {
|
||||
try {
|
||||
const content = await fsp.readFile(path.join(groupDir, 'meta.json'), 'utf-8');
|
||||
return JSON.parse(content) as BridgeMeta;
|
||||
} catch {
|
||||
return { version: 0, generatedAt: '', missingRepos: [] };
|
||||
}
|
||||
}
|
||||
|
||||
/* ------------------------------------------------------------------ */
|
||||
/* writeBridge — atomic write-to-temp-then-rename */
|
||||
/* ------------------------------------------------------------------ */
|
||||
|
||||
export interface WriteBridgeInput {
|
||||
contracts: StoredContract[];
|
||||
crossLinks: CrossLink[];
|
||||
repoSnapshots: Record<string, RepoSnapshot>;
|
||||
missingRepos: string[];
|
||||
}
|
||||
|
||||
/**
|
||||
* Non-fatal issues encountered during writeBridge. Callers can log these to
|
||||
* surface partial-success state without aborting the whole sync.
|
||||
* `sampleErrors` is capped at MAX_SAMPLE_ERRORS per category to bound memory.
|
||||
*/
|
||||
export interface WriteBridgeReport {
|
||||
contractsInserted: number;
|
||||
contractsFailed: number;
|
||||
snapshotsInserted: number;
|
||||
snapshotsFailed: number;
|
||||
linksInserted: number;
|
||||
linksFailed: number;
|
||||
/** Cross-links skipped because their from/to contract nodes weren't found. */
|
||||
linksDroppedMissingNode: number;
|
||||
sampleErrors: Array<{
|
||||
kind: 'contract' | 'snapshot' | 'link';
|
||||
id: string;
|
||||
message: string;
|
||||
}>;
|
||||
}
|
||||
|
||||
const MAX_SAMPLE_ERRORS = 10;
|
||||
|
||||
function errMessage(err: unknown): string {
|
||||
if (err instanceof Error) return err.message;
|
||||
try {
|
||||
return String(err);
|
||||
} catch {
|
||||
return 'unknown error';
|
||||
}
|
||||
}
|
||||
|
||||
export async function writeBridge(
|
||||
groupDir: string,
|
||||
input: WriteBridgeInput,
|
||||
): Promise<WriteBridgeReport> {
|
||||
await fsp.mkdir(groupDir, { recursive: true });
|
||||
const contracts = dedupeContracts(input.contracts);
|
||||
const crossLinks = dedupeCrossLinks(input.crossLinks);
|
||||
|
||||
const finalPath = path.join(groupDir, 'bridge.lbug');
|
||||
const tmpPath = path.join(groupDir, 'bridge.lbug.tmp');
|
||||
const bakPath = path.join(groupDir, 'bridge.lbug.bak');
|
||||
|
||||
const report: WriteBridgeReport = {
|
||||
contractsInserted: 0,
|
||||
contractsFailed: 0,
|
||||
snapshotsInserted: 0,
|
||||
snapshotsFailed: 0,
|
||||
linksInserted: 0,
|
||||
linksFailed: 0,
|
||||
linksDroppedMissingNode: 0,
|
||||
sampleErrors: [],
|
||||
};
|
||||
|
||||
const recordError = (kind: 'contract' | 'snapshot' | 'link', id: string, err: unknown) => {
|
||||
if (report.sampleErrors.length < MAX_SAMPLE_ERRORS) {
|
||||
report.sampleErrors.push({ kind, id, message: errMessage(err) });
|
||||
}
|
||||
};
|
||||
|
||||
// Clean up any leftover tmp
|
||||
try {
|
||||
await fsp.rm(tmpPath, { recursive: true, force: true });
|
||||
} catch {
|
||||
/* ignore */
|
||||
}
|
||||
|
||||
// 1. Create temp DB, insert all data.
|
||||
//
|
||||
// Everything after `openBridgeDb` must run inside a try/finally so that
|
||||
// if ANY step before the explicit `closeBridgeDb` throws — schema
|
||||
// creation, a contract insert loop that rethrows, a snapshot write, the
|
||||
// cross-link loop, or anything else — the handle is still released. A
|
||||
// leaked handle holds the native LadybugDB file lock on tmpPath, which
|
||||
// (a) leaks a FD and (b) prevents the next writeBridge call from
|
||||
// reusing the same tmp slot.
|
||||
const handle = await openBridgeDb(tmpPath);
|
||||
let handleClosed = false;
|
||||
try {
|
||||
await ensureBridgeSchema(handle);
|
||||
|
||||
// Build the lookup index incrementally as contracts are inserted, so
|
||||
// failed inserts are never in the index (and therefore never resolved
|
||||
// by the cross-link loop below). This replaces a previous N+1 query
|
||||
// pattern where each link made up to 6 DB round-trips to find its
|
||||
// endpoints — see ContractLookupIndex.
|
||||
const lookupIndex = createContractLookupIndex();
|
||||
|
||||
// Insert contracts — tolerate individual failures (e.g., a corrupt meta
|
||||
// that can't be serialized). The whole sync must not fail because one
|
||||
// contract is broken.
|
||||
for (const c of contracts) {
|
||||
const id = contractNodeId(c.repo, c.contractId, c.role, c.symbolRef.filePath);
|
||||
try {
|
||||
await queryBridge(
|
||||
handle,
|
||||
`CREATE (n:Contract {
|
||||
id: $id,
|
||||
contractId: $contractId,
|
||||
type: $type,
|
||||
role: $role,
|
||||
repo: $repo,
|
||||
service: $service,
|
||||
symbolUid: $symbolUid,
|
||||
filePath: $filePath,
|
||||
symbolName: $symbolName,
|
||||
confidence: $confidence,
|
||||
meta: $meta
|
||||
})`,
|
||||
{
|
||||
id,
|
||||
contractId: c.contractId,
|
||||
type: c.type,
|
||||
role: c.role,
|
||||
repo: c.repo,
|
||||
service: c.service ?? '',
|
||||
symbolUid: c.symbolUid,
|
||||
filePath: c.symbolRef.filePath,
|
||||
symbolName: c.symbolName,
|
||||
confidence: c.confidence,
|
||||
meta: JSON.stringify(c.meta),
|
||||
},
|
||||
);
|
||||
report.contractsInserted++;
|
||||
// Only index on successful insert — the cross-link loop must never
|
||||
// resolve to a row that isn't actually in the DB.
|
||||
indexContract(lookupIndex, c, id);
|
||||
} catch (err) {
|
||||
report.contractsFailed++;
|
||||
recordError('contract', id, err);
|
||||
}
|
||||
}
|
||||
|
||||
// Insert repo snapshots
|
||||
for (const [repoId, snap] of Object.entries(input.repoSnapshots)) {
|
||||
try {
|
||||
await queryBridge(
|
||||
handle,
|
||||
`CREATE (s:RepoSnapshot {
|
||||
id: $id,
|
||||
indexedAt: $indexedAt,
|
||||
lastCommit: $lastCommit
|
||||
})`,
|
||||
{
|
||||
id: repoId,
|
||||
indexedAt: snap.indexedAt,
|
||||
lastCommit: snap.lastCommit,
|
||||
},
|
||||
);
|
||||
report.snapshotsInserted++;
|
||||
} catch (err) {
|
||||
report.snapshotsFailed++;
|
||||
recordError('snapshot', repoId, err);
|
||||
}
|
||||
}
|
||||
|
||||
// Insert cross-links (tolerating missing nodes).
|
||||
//
|
||||
// `findContractNode` consults the in-memory lookup index built above,
|
||||
// not the DB — that's an O(1) pure-function lookup per endpoint instead
|
||||
// of the previous 2-3 DB queries. For M cross-links, the previous code
|
||||
// issued up to 6M round-trips; this version issues zero.
|
||||
//
|
||||
// `link.contractId` may differ between the consumer and provider sides
|
||||
// (e.g. wildcard consumer `grpc::Service/*` → method-level provider
|
||||
// `grpc::Service/Method`) — that's why we resolve each endpoint
|
||||
// independently via its own `(repo, role, symbolUid, filePath, symbolName)`
|
||||
// tuple rather than matching on contractId.
|
||||
for (const link of crossLinks) {
|
||||
const linkId = `${link.from.repo}::${link.contractId}->${link.to.repo}::${link.contractId}`;
|
||||
try {
|
||||
const fromId = findContractNode(
|
||||
lookupIndex,
|
||||
link.from.repo,
|
||||
'consumer',
|
||||
link.from.symbolUid,
|
||||
link.from.symbolRef.filePath,
|
||||
link.from.symbolRef.name,
|
||||
);
|
||||
const toId = findContractNode(
|
||||
lookupIndex,
|
||||
link.to.repo,
|
||||
'provider',
|
||||
link.to.symbolUid,
|
||||
link.to.symbolRef.filePath,
|
||||
link.to.symbolRef.name,
|
||||
);
|
||||
if (!fromId || !toId) {
|
||||
report.linksDroppedMissingNode++;
|
||||
continue;
|
||||
}
|
||||
await queryBridge(
|
||||
handle,
|
||||
`
|
||||
MATCH (a:Contract), (b:Contract)
|
||||
WHERE a.id = $fromId AND b.id = $toId
|
||||
CREATE (a)-[:ContractLink {
|
||||
matchType: $matchType,
|
||||
confidence: $confidence,
|
||||
contractId: $contractId,
|
||||
fromRepo: $fromRepo,
|
||||
toRepo: $toRepo
|
||||
}]->(b)
|
||||
`,
|
||||
{
|
||||
fromId,
|
||||
toId,
|
||||
matchType: link.matchType,
|
||||
confidence: link.confidence,
|
||||
contractId: link.contractId,
|
||||
fromRepo: link.from.repo,
|
||||
toRepo: link.to.repo,
|
||||
},
|
||||
);
|
||||
report.linksInserted++;
|
||||
} catch (err) {
|
||||
report.linksFailed++;
|
||||
recordError('link', linkId, err);
|
||||
}
|
||||
}
|
||||
|
||||
// 2. Close temp DB (happy path). The finally block also calls
|
||||
// closeBridgeDb if we threw above; `handleClosed` prevents a
|
||||
// double-close on the native handle.
|
||||
await closeBridgeDb(handle);
|
||||
handleClosed = true;
|
||||
} finally {
|
||||
if (!handleClosed) {
|
||||
await closeBridgeDb(handle).catch(() => {
|
||||
/* ignore: cleanup path, best effort */
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
// 3. Atomic swap: old→.bak, tmp→final, rm .bak
|
||||
try {
|
||||
await fsp.access(finalPath);
|
||||
await retryRename(finalPath, bakPath);
|
||||
} catch {
|
||||
/* no existing db */
|
||||
}
|
||||
await retryRename(tmpPath, finalPath);
|
||||
try {
|
||||
await fsp.rm(bakPath, { recursive: true, force: true });
|
||||
} catch {
|
||||
/* ignore */
|
||||
}
|
||||
|
||||
// 4. Write meta.json
|
||||
await writeBridgeMeta(groupDir, {
|
||||
version: BRIDGE_SCHEMA_VERSION,
|
||||
generatedAt: new Date().toISOString(),
|
||||
missingRepos: input.missingRepos,
|
||||
});
|
||||
|
||||
return report;
|
||||
}
|
||||
|
||||
/* ------------------------------------------------------------------ */
|
||||
/* openBridgeDbReadOnly */
|
||||
/* ------------------------------------------------------------------ */
|
||||
|
||||
export async function openBridgeDbReadOnly(groupDir: string): Promise<BridgeHandle | null> {
|
||||
const dbPath = path.join(groupDir, 'bridge.lbug');
|
||||
try {
|
||||
await fsp.access(dbPath);
|
||||
} catch {
|
||||
// Check for .bak recovery. Use `retryRename` (not `fsp.rename`) for the
|
||||
// exact same reason the rest of this file does: the scenario that
|
||||
// triggers bak recovery is an interrupted writer, which on Windows may
|
||||
// still be holding an open handle on `.bak` for a few milliseconds when
|
||||
// a reader races in. EBUSY/EPERM retries recover that case silently.
|
||||
const bakPath = path.join(groupDir, 'bridge.lbug.bak');
|
||||
try {
|
||||
await fsp.access(bakPath);
|
||||
await retryRename(bakPath, dbPath);
|
||||
} catch {
|
||||
return null;
|
||||
}
|
||||
}
|
||||
// Version gate: check meta.json version compatibility
|
||||
const meta = await readBridgeMeta(groupDir);
|
||||
if (meta.version > 0 && meta.version !== BRIDGE_SCHEMA_VERSION) {
|
||||
return null; // incompatible schema version — fallback to JSON or re-sync
|
||||
}
|
||||
|
||||
// Open the native handle. If Connection construction throws AFTER
|
||||
// Database was successfully allocated, we'd leak the native Database
|
||||
// object. Wrap each step separately and tear down the partial handle.
|
||||
let db: lbug.Database | undefined;
|
||||
let conn: lbug.Connection | undefined;
|
||||
try {
|
||||
db = new lbug.Database(dbPath, 0, false, true); // readOnly
|
||||
conn = new lbug.Connection(db);
|
||||
return { _db: db, _conn: conn, groupDir } as BridgeHandle;
|
||||
} catch {
|
||||
if (conn) {
|
||||
try {
|
||||
await conn.close();
|
||||
} catch {
|
||||
/* ignore */
|
||||
}
|
||||
}
|
||||
if (db) {
|
||||
try {
|
||||
await db.close();
|
||||
} catch {
|
||||
/* ignore */
|
||||
}
|
||||
}
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
/* ------------------------------------------------------------------ */
|
||||
/* bridgeExists */
|
||||
/* ------------------------------------------------------------------ */
|
||||
|
||||
export async function bridgeExists(groupDir: string): Promise<boolean> {
|
||||
const handle = await openBridgeDbReadOnly(groupDir);
|
||||
if (!handle) return false;
|
||||
await closeBridgeDb(handle);
|
||||
return true;
|
||||
}
|
||||
@@ -0,0 +1,60 @@
|
||||
/**
|
||||
* Bridge LadybugDB schema for cross-repo Contract Registry.
|
||||
* Separate from per-repo schema in lbug/schema.ts.
|
||||
*/
|
||||
|
||||
/**
|
||||
* Version of the bridge.lbug schema below. `openBridgeDbReadOnly` compares
|
||||
* this against `meta.json`'s version field and returns `null` on mismatch,
|
||||
* which trips the caller into either the JSON fallback path or a fresh
|
||||
* `group sync` that rebuilds `bridge.lbug` from scratch.
|
||||
*
|
||||
* Migration contract for contributors bumping this constant:
|
||||
* 1. Bump the number (e.g. `1` → `2`).
|
||||
* 2. Update the DDL below to match the new schema.
|
||||
* 3. DO NOT attempt an online migration in this file — the version gate
|
||||
* is intentionally a "discard and re-sync" strategy for V1. An old
|
||||
* bridge.lbug whose version doesn't match is treated as opaque and
|
||||
* rebuilt by the next `group sync`.
|
||||
* 4. If online migration becomes necessary (e.g. when groups accumulate
|
||||
* large amounts of embedding data), add a migration path as a
|
||||
* separate `bridge-migrations.ts` module rather than bloating this
|
||||
* file — keep schema and migration concerns separate.
|
||||
*/
|
||||
export const BRIDGE_SCHEMA_VERSION = 1;
|
||||
|
||||
export const CONTRACT_SCHEMA = `
|
||||
CREATE NODE TABLE Contract (
|
||||
id STRING,
|
||||
contractId STRING,
|
||||
type STRING,
|
||||
role STRING,
|
||||
repo STRING,
|
||||
service STRING DEFAULT '',
|
||||
symbolUid STRING DEFAULT '',
|
||||
filePath STRING DEFAULT '',
|
||||
symbolName STRING DEFAULT '',
|
||||
confidence DOUBLE DEFAULT 0.0,
|
||||
meta STRING DEFAULT '{}',
|
||||
PRIMARY KEY (id)
|
||||
)`;
|
||||
|
||||
export const REPO_SNAPSHOT_SCHEMA = `
|
||||
CREATE NODE TABLE RepoSnapshot (
|
||||
id STRING,
|
||||
indexedAt STRING DEFAULT '',
|
||||
lastCommit STRING DEFAULT '',
|
||||
PRIMARY KEY (id)
|
||||
)`;
|
||||
|
||||
export const CONTRACT_LINK_SCHEMA = `
|
||||
CREATE REL TABLE ContractLink (
|
||||
FROM Contract TO Contract,
|
||||
matchType STRING,
|
||||
confidence DOUBLE,
|
||||
contractId STRING,
|
||||
fromRepo STRING,
|
||||
toRepo STRING
|
||||
)`;
|
||||
|
||||
export const BRIDGE_SCHEMA_QUERIES = [CONTRACT_SCHEMA, REPO_SNAPSHOT_SCHEMA, CONTRACT_LINK_SCHEMA];
|
||||
@@ -0,0 +1,98 @@
|
||||
import { createRequire } from 'node:module';
|
||||
import type { GroupConfig, GroupManifestLink, ContractType, ContractRole } from './types.js';
|
||||
|
||||
const _require = createRequire(import.meta.url);
|
||||
const yaml = _require('js-yaml') as typeof import('js-yaml');
|
||||
|
||||
const VALID_CONTRACT_TYPES: ContractType[] = ['http', 'grpc', 'topic', 'lib', 'custom'];
|
||||
const VALID_ROLES: ContractRole[] = ['provider', 'consumer'];
|
||||
|
||||
const DEFAULT_DETECT = {
|
||||
http: true,
|
||||
grpc: true,
|
||||
topics: true,
|
||||
shared_libs: true,
|
||||
embedding_fallback: true,
|
||||
};
|
||||
|
||||
const DEFAULT_MATCHING = {
|
||||
bm25_threshold: 0.7,
|
||||
embedding_threshold: 0.65,
|
||||
max_candidates_per_step: 3,
|
||||
};
|
||||
|
||||
export function parseGroupConfig(yamlContent: string): GroupConfig {
|
||||
const raw = yaml.load(yamlContent, { schema: yaml.JSON_SCHEMA }) as Record<string, unknown>;
|
||||
|
||||
if (!raw || typeof raw !== 'object' || Array.isArray(raw)) {
|
||||
throw new Error('Invalid YAML: expected an object');
|
||||
}
|
||||
|
||||
if (raw.version === undefined) throw new Error('version is required in group.yaml');
|
||||
if (raw.version !== 1) {
|
||||
throw new Error(`Unsupported group.yaml version: ${raw.version}. Expected 1.`);
|
||||
}
|
||||
if (!raw.name || typeof raw.name !== 'string') throw new Error('name is required in group.yaml');
|
||||
if (!raw.repos || typeof raw.repos !== 'object' || Array.isArray(raw.repos)) {
|
||||
throw new Error('repos is required in group.yaml (must be a mapping)');
|
||||
}
|
||||
|
||||
const repos = raw.repos as Record<string, string>;
|
||||
const repoPaths = new Set(Object.keys(repos));
|
||||
|
||||
const rawLinks = (raw.links as unknown[]) || [];
|
||||
const links: GroupManifestLink[] = rawLinks.map((l: unknown, i: number) => {
|
||||
const link = l as Record<string, unknown>;
|
||||
if (!link.from || !repoPaths.has(link.from as string)) {
|
||||
throw new Error(`links[${i}].from "${link.from}" does not match any repo path in group`);
|
||||
}
|
||||
if (!link.to || !repoPaths.has(link.to as string)) {
|
||||
throw new Error(`links[${i}].to "${link.to}" does not match any repo path in group`);
|
||||
}
|
||||
if (!VALID_CONTRACT_TYPES.includes(link.type as ContractType)) {
|
||||
throw new Error(
|
||||
`links[${i}].type "${link.type}" is invalid. Expected: ${VALID_CONTRACT_TYPES.join(', ')}`,
|
||||
);
|
||||
}
|
||||
if (!VALID_ROLES.includes(link.role as ContractRole)) {
|
||||
throw new Error(`links[${i}].role "${link.role}" is invalid. Expected: provider | consumer`);
|
||||
}
|
||||
if (
|
||||
link.contract === undefined ||
|
||||
link.contract === null ||
|
||||
String(link.contract).trim() === ''
|
||||
) {
|
||||
throw new Error(`links[${i}].contract is required`);
|
||||
}
|
||||
return {
|
||||
from: link.from as string,
|
||||
to: link.to as string,
|
||||
type: link.type as ContractType,
|
||||
contract: String(link.contract),
|
||||
role: link.role as ContractRole,
|
||||
};
|
||||
});
|
||||
|
||||
const detect = { ...DEFAULT_DETECT, ...((raw.detect as object) || {}) };
|
||||
const matching = { ...DEFAULT_MATCHING, ...((raw.matching as object) || {}) };
|
||||
const packages = (raw.packages as Record<string, Record<string, string>>) || {};
|
||||
|
||||
return {
|
||||
version: 1,
|
||||
name: raw.name as string,
|
||||
description: (raw.description as string) || '',
|
||||
repos,
|
||||
links,
|
||||
packages,
|
||||
detect,
|
||||
matching,
|
||||
};
|
||||
}
|
||||
|
||||
export async function loadGroupConfig(groupDir: string): Promise<GroupConfig> {
|
||||
const fsp = await import('node:fs/promises');
|
||||
const path = await import('node:path');
|
||||
const yamlPath = path.join(groupDir, 'group.yaml');
|
||||
const content = await fsp.readFile(yamlPath, 'utf-8');
|
||||
return parseGroupConfig(content);
|
||||
}
|
||||
@@ -0,0 +1,16 @@
|
||||
import type { ContractType, ExtractedContract, RepoHandle } from './types.js';
|
||||
|
||||
export interface ContractExtractor {
|
||||
type: ContractType;
|
||||
canExtract(repo: RepoHandle): Promise<boolean>;
|
||||
extract(
|
||||
dbExecutor: CypherExecutor | null,
|
||||
repoPath: string,
|
||||
repo: RepoHandle,
|
||||
): Promise<ExtractedContract[]>;
|
||||
}
|
||||
|
||||
export type CypherExecutor = (
|
||||
query: string,
|
||||
params?: Record<string, unknown>,
|
||||
) => Promise<Record<string, unknown>[]>;
|
||||
@@ -0,0 +1,23 @@
|
||||
import * as fs from 'node:fs';
|
||||
import * as path from 'node:path';
|
||||
|
||||
/**
|
||||
* Safely read a file inside a repo, rejecting any path that escapes
|
||||
* `repoPath` via `..` traversal or absolute segments. Returns `null` if
|
||||
* the path is outside the repo or the file can't be read.
|
||||
*
|
||||
* Used by every source-scan extractor under this directory. Kept as a
|
||||
* single shared implementation so the path-traversal guard (security-
|
||||
* sensitive) lives in exactly one place.
|
||||
*/
|
||||
export function readSafe(repoPath: string, rel: string): string | null {
|
||||
const abs = path.resolve(repoPath, rel);
|
||||
const base = path.resolve(repoPath);
|
||||
const relToBase = path.relative(base, abs);
|
||||
if (relToBase.startsWith('..') || path.isAbsolute(relToBase)) return null;
|
||||
try {
|
||||
return fs.readFileSync(abs, 'utf-8');
|
||||
} catch {
|
||||
return null;
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,482 @@
|
||||
import * as path from 'node:path';
|
||||
import { glob } from 'glob';
|
||||
import Parser from 'tree-sitter';
|
||||
import type { ContractExtractor, CypherExecutor } from '../contract-extractor.js';
|
||||
import type { ExtractedContract, RepoHandle } from '../types.js';
|
||||
import { readSafe } from './fs-utils.js';
|
||||
import {
|
||||
GRPC_SCAN_GLOB,
|
||||
getPluginForFile,
|
||||
hasProtoPlugin,
|
||||
type GrpcDetection,
|
||||
} from './grpc-patterns/index.js';
|
||||
|
||||
/**
|
||||
* Language-agnostic orchestrator for gRPC (provider + consumer) contract
|
||||
* extraction.
|
||||
*
|
||||
* Two parts:
|
||||
*
|
||||
* 1. **`.proto` parsing** — tree-sitter when `tree-sitter-proto` is
|
||||
* installed (optionalDependency vendored in `vendor/tree-sitter-proto/`),
|
||||
* via the `.proto` entry in `grpc-patterns/` and `hasProtoPlugin`.
|
||||
* When the grammar isn't available (platform incompatibility, native
|
||||
* build failure) the orchestrator falls back to the in-process
|
||||
* string-sanitizing parser defined below (`stripProtoCommentsAndStrings`
|
||||
* + `extractServiceBlocks`). The fallback preserves offsets so any
|
||||
* downstream regex scans run against a sanitized copy without
|
||||
* affecting line numbers of the original.
|
||||
*
|
||||
* 2. **Source-scan providers / consumers** — delegated to per-language
|
||||
* plugins in `./grpc-patterns/`. The orchestrator imports NO
|
||||
* tree-sitter grammars or query strings — each plugin owns its own.
|
||||
*/
|
||||
|
||||
// ─── .proto fallback parser (used only when tree-sitter-proto is absent) ───
|
||||
|
||||
function contractId(pkg: string, service: string, method: string): string {
|
||||
const prefix = pkg ? `${pkg}.${service}` : service;
|
||||
return `grpc::${prefix}/${method}`;
|
||||
}
|
||||
|
||||
function serviceOnlyContractId(serviceName: string): string {
|
||||
return `grpc::${serviceName}/*`;
|
||||
}
|
||||
|
||||
/**
|
||||
* Replace all .proto comments and string literals with spaces, preserving the
|
||||
* original length and character offsets of the input. This lets downstream
|
||||
* regex / brace-depth parsers run on a "sanitized" copy without having to
|
||||
* understand proto syntax, while any RegExp.exec/index-based lookups that
|
||||
* were already positional against `content` continue to work against the
|
||||
* original string.
|
||||
*
|
||||
* Supported comment forms: `// line comment`, `/* block comment * /`.
|
||||
* Supported strings: double-quoted ("…") and single-quoted ('…') with `\`
|
||||
* escape handling. Raw/unterminated strings are not supported — we stop
|
||||
* on a line break for line-style comments and on EOF for unterminated
|
||||
* strings/blocks, which matches how most real proto files parse.
|
||||
*/
|
||||
function stripProtoCommentsAndStrings(content: string): string {
|
||||
const out = new Array<string>(content.length);
|
||||
let i = 0;
|
||||
while (i < content.length) {
|
||||
const ch = content[i];
|
||||
const next = content[i + 1];
|
||||
|
||||
// Line comment: // ... \n
|
||||
if (ch === '/' && next === '/') {
|
||||
out[i] = ' ';
|
||||
out[i + 1] = ' ';
|
||||
i += 2;
|
||||
while (i < content.length && content[i] !== '\n') {
|
||||
out[i] = content[i] === '\r' ? '\r' : ' ';
|
||||
i++;
|
||||
}
|
||||
continue;
|
||||
}
|
||||
|
||||
// Block comment: /* ... */
|
||||
if (ch === '/' && next === '*') {
|
||||
out[i] = ' ';
|
||||
out[i + 1] = ' ';
|
||||
i += 2;
|
||||
while (i < content.length) {
|
||||
if (content[i] === '*' && content[i + 1] === '/') {
|
||||
out[i] = ' ';
|
||||
out[i + 1] = ' ';
|
||||
i += 2;
|
||||
break;
|
||||
}
|
||||
// Preserve newlines so line numbers stay stable for downstream code.
|
||||
out[i] = content[i] === '\n' || content[i] === '\r' ? content[i] : ' ';
|
||||
i++;
|
||||
}
|
||||
continue;
|
||||
}
|
||||
|
||||
// String literal: "..." or '...'
|
||||
if (ch === '"' || ch === "'") {
|
||||
const quote = ch;
|
||||
out[i] = ' '; // replace opening quote
|
||||
i++;
|
||||
while (i < content.length) {
|
||||
const c = content[i];
|
||||
if (c === '\\' && i + 1 < content.length) {
|
||||
// Skip escaped pair (e.g. \" \n \\)
|
||||
out[i] = ' ';
|
||||
out[i + 1] = ' ';
|
||||
i += 2;
|
||||
continue;
|
||||
}
|
||||
if (c === quote) {
|
||||
out[i] = ' ';
|
||||
i++;
|
||||
break;
|
||||
}
|
||||
// Preserve newlines; proto technically disallows unescaped newlines
|
||||
// inside strings, but real files occasionally have them.
|
||||
out[i] = c === '\n' || c === '\r' ? c : ' ';
|
||||
i++;
|
||||
}
|
||||
continue;
|
||||
}
|
||||
|
||||
out[i] = ch;
|
||||
i++;
|
||||
}
|
||||
return out.join('');
|
||||
}
|
||||
|
||||
function extractServiceBlocks(content: string): Array<{ name: string; body: string }> {
|
||||
const results: Array<{ name: string; body: string }> = [];
|
||||
// Sanitize comments and string literals so braces inside them don't
|
||||
// throw off the depth counter. The sanitized copy has the same length
|
||||
// and offsets as the original, so we use it ONLY to scan for service
|
||||
// headers and braces; the service body we return is sliced from the
|
||||
// ORIGINAL content to preserve exact source text for downstream use.
|
||||
const sanitized = stripProtoCommentsAndStrings(content);
|
||||
const headerRe = /service\s+(\w+)\s*\{/g;
|
||||
let headerMatch: RegExpExecArray | null;
|
||||
|
||||
while ((headerMatch = headerRe.exec(sanitized)) !== null) {
|
||||
const serviceName = headerMatch[1];
|
||||
const bodyStart = headerMatch.index + headerMatch[0].length;
|
||||
let depth = 1;
|
||||
let pos = bodyStart;
|
||||
|
||||
while (pos < sanitized.length && depth > 0) {
|
||||
const ch = sanitized[pos];
|
||||
if (ch === '{') depth++;
|
||||
else if (ch === '}') depth--;
|
||||
pos++;
|
||||
}
|
||||
|
||||
// If EOF before depth returns to 0, skip incomplete service
|
||||
if (depth !== 0) continue;
|
||||
|
||||
// body is between opening { (consumed by regex) and closing } (pos is one past it)
|
||||
const body = content.slice(bodyStart, pos - 1);
|
||||
results.push({ name: serviceName, body });
|
||||
}
|
||||
|
||||
return results;
|
||||
}
|
||||
|
||||
function makeContract(
|
||||
cid: string,
|
||||
role: 'provider' | 'consumer',
|
||||
filePath: string,
|
||||
symbolName: string,
|
||||
confidence: number,
|
||||
meta: Record<string, unknown>,
|
||||
): ExtractedContract {
|
||||
return {
|
||||
contractId: cid,
|
||||
type: 'grpc',
|
||||
role,
|
||||
symbolUid: '',
|
||||
symbolRef: { filePath: filePath.replace(/\\/g, '/'), name: symbolName },
|
||||
symbolName,
|
||||
confidence,
|
||||
meta: { ...meta, extractionStrategy: 'source_scan' },
|
||||
};
|
||||
}
|
||||
|
||||
export interface ProtoServiceInfo {
|
||||
package: string;
|
||||
serviceName: string;
|
||||
methods: string[];
|
||||
protoPath: string;
|
||||
}
|
||||
|
||||
function normalizeProtoPath(rel: string): string {
|
||||
return rel.replace(/\\/g, '/');
|
||||
}
|
||||
|
||||
function extractProtoImports(content: string): string[] {
|
||||
const imports: string[] = [];
|
||||
const re = /^\s*import\s+"([^"]+)"\s*;/gm;
|
||||
let match: RegExpExecArray | null;
|
||||
while ((match = re.exec(content)) !== null) {
|
||||
imports.push(match[1]);
|
||||
}
|
||||
return imports;
|
||||
}
|
||||
|
||||
function longestSharedSegmentRun(aPath: string, bPath: string): number {
|
||||
const a = aPath.split('/').filter(Boolean);
|
||||
const b = bPath.split('/').filter(Boolean);
|
||||
let best = 0;
|
||||
|
||||
for (let i = 0; i < a.length; i++) {
|
||||
for (let j = 0; j < b.length; j++) {
|
||||
let run = 0;
|
||||
while (a[i + run] && b[j + run] && a[i + run] === b[j + run]) {
|
||||
run++;
|
||||
}
|
||||
if (run > best) best = run;
|
||||
}
|
||||
}
|
||||
|
||||
return best;
|
||||
}
|
||||
|
||||
async function buildProtoContext(repoPath: string): Promise<{
|
||||
packagesByProto: Map<string, string>;
|
||||
servicesByName: Map<string, ProtoServiceInfo[]>;
|
||||
}> {
|
||||
const servicesByName = new Map<string, ProtoServiceInfo[]>();
|
||||
const protoFiles = await glob('**/*.proto', {
|
||||
cwd: repoPath,
|
||||
absolute: false,
|
||||
nodir: true,
|
||||
ignore: ['**/node_modules/**', '**/.git/**', '**/vendor/**'],
|
||||
});
|
||||
const contents = new Map<string, string>();
|
||||
|
||||
for (const rel of protoFiles) {
|
||||
const content = readSafe(repoPath, rel);
|
||||
if (!content) continue;
|
||||
contents.set(normalizeProtoPath(rel), content);
|
||||
}
|
||||
|
||||
const packagesByProto = new Map<string, string>();
|
||||
|
||||
const resolvePackage = (protoPath: string, seen = new Set<string>()): string => {
|
||||
if (packagesByProto.has(protoPath)) return packagesByProto.get(protoPath) ?? '';
|
||||
if (seen.has(protoPath)) return '';
|
||||
|
||||
const content = contents.get(protoPath);
|
||||
if (!content) return '';
|
||||
|
||||
seen.add(protoPath);
|
||||
const pkgMatch = content.match(/^\s*package\s+([\w.]+)\s*;/m);
|
||||
if (pkgMatch?.[1]) {
|
||||
packagesByProto.set(protoPath, pkgMatch[1]);
|
||||
return pkgMatch[1];
|
||||
}
|
||||
|
||||
for (const importPath of extractProtoImports(content)) {
|
||||
const normalizedImport = normalizeProtoPath(importPath);
|
||||
const candidates = [
|
||||
normalizeProtoPath(
|
||||
path.posix.normalize(path.posix.join(path.posix.dirname(protoPath), normalizedImport)),
|
||||
),
|
||||
normalizedImport,
|
||||
];
|
||||
for (const candidate of candidates) {
|
||||
if (!contents.has(candidate)) continue;
|
||||
const inheritedPackage = resolvePackage(candidate, seen);
|
||||
if (inheritedPackage) {
|
||||
packagesByProto.set(protoPath, inheritedPackage);
|
||||
return inheritedPackage;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
packagesByProto.set(protoPath, '');
|
||||
return '';
|
||||
};
|
||||
|
||||
for (const rel of protoFiles) {
|
||||
const normalizedRel = normalizeProtoPath(rel);
|
||||
const content = contents.get(normalizedRel);
|
||||
if (!content) continue;
|
||||
const pkg = resolvePackage(normalizedRel);
|
||||
|
||||
const serviceBlocks = extractServiceBlocks(content);
|
||||
for (const block of serviceBlocks) {
|
||||
const rpcRe = /rpc\s+(\w+)\s*\(/g;
|
||||
const methods: string[] = [];
|
||||
let m: RegExpExecArray | null;
|
||||
while ((m = rpcRe.exec(block.body)) !== null) {
|
||||
methods.push(m[1]);
|
||||
}
|
||||
const info: ProtoServiceInfo = {
|
||||
package: pkg,
|
||||
serviceName: block.name,
|
||||
methods,
|
||||
protoPath: normalizedRel,
|
||||
};
|
||||
const existing = servicesByName.get(block.name) ?? [];
|
||||
existing.push(info);
|
||||
servicesByName.set(block.name, existing);
|
||||
}
|
||||
}
|
||||
|
||||
return { packagesByProto, servicesByName };
|
||||
}
|
||||
|
||||
export async function buildProtoMap(repoPath: string): Promise<Map<string, ProtoServiceInfo[]>> {
|
||||
const { servicesByName } = await buildProtoContext(repoPath);
|
||||
return servicesByName;
|
||||
}
|
||||
|
||||
export function resolveProtoConflict(
|
||||
serviceName: string,
|
||||
sourceFilePath: string,
|
||||
candidates: ProtoServiceInfo[],
|
||||
): ProtoServiceInfo | null {
|
||||
if (candidates.length === 0) return null;
|
||||
if (candidates.length === 1) return candidates[0];
|
||||
|
||||
const sourceDir = normalizeProtoPath(path.dirname(sourceFilePath));
|
||||
const scored = candidates.map((c) => {
|
||||
const protoDir = normalizeProtoPath(path.dirname(c.protoPath));
|
||||
return { candidate: c, score: longestSharedSegmentRun(sourceDir, protoDir) };
|
||||
});
|
||||
|
||||
let maxScore = -1;
|
||||
for (const s of scored) {
|
||||
if (s.score > maxScore) maxScore = s.score;
|
||||
}
|
||||
const winners = scored.filter((s) => s.score === maxScore);
|
||||
|
||||
// Path heuristic cannot uniquely identify a winner — refuse to guess.
|
||||
// Ties (including all-zero ties) would otherwise silently merge unrelated
|
||||
// services under a fabricated package-qualified contract id.
|
||||
if (winners.length !== 1) {
|
||||
const paths = candidates.map((c) => c.protoPath).join(', ');
|
||||
console.warn(
|
||||
`[grpc-extractor] Ambiguous proto resolution for service "${serviceName}" from ${sourceFilePath}: ${winners.length} candidates tied at score ${maxScore} among [${paths}] — skipping canonical contract`,
|
||||
);
|
||||
return null;
|
||||
}
|
||||
|
||||
return winners[0].candidate;
|
||||
}
|
||||
|
||||
export function serviceContractId(pkg: string, serviceName: string): string {
|
||||
const prefix = pkg ? `${pkg}.${serviceName}` : serviceName;
|
||||
return `grpc::${prefix}/*`;
|
||||
}
|
||||
|
||||
// ─── Orchestrator ────────────────────────────────────────────────────
|
||||
|
||||
export class GrpcExtractor implements ContractExtractor {
|
||||
type = 'grpc' as const;
|
||||
|
||||
async canExtract(_repo: RepoHandle): Promise<boolean> {
|
||||
return true;
|
||||
}
|
||||
|
||||
async extract(
|
||||
_dbExecutor: CypherExecutor | null,
|
||||
repoPath: string,
|
||||
_repo: RepoHandle,
|
||||
): Promise<ExtractedContract[]> {
|
||||
const out: ExtractedContract[] = [];
|
||||
const protoContext = await buildProtoContext(repoPath);
|
||||
const protoMap = protoContext.servicesByName;
|
||||
|
||||
// ─── Proto files — definitive provider source ─────────────────
|
||||
// When tree-sitter-proto is available, .proto files are handled by
|
||||
// the plugin loop below (they're in GRPC_SCAN_GLOB). Otherwise
|
||||
// emit provider contracts directly from the proto map that
|
||||
// `buildProtoContext` already built — no second glob / parse pass.
|
||||
if (!hasProtoPlugin) {
|
||||
for (const infos of protoMap.values()) {
|
||||
for (const info of infos) {
|
||||
for (const methodName of info.methods) {
|
||||
const cid = contractId(info.package, info.serviceName, methodName);
|
||||
out.push(
|
||||
makeContract(
|
||||
cid,
|
||||
'provider',
|
||||
info.protoPath,
|
||||
`${info.serviceName}.${methodName}`,
|
||||
0.85,
|
||||
{
|
||||
package: info.package,
|
||||
service: info.serviceName,
|
||||
method: methodName,
|
||||
source: 'proto',
|
||||
},
|
||||
),
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// ─── Source files (+ .proto when plugin available) ────────────
|
||||
const sourceFiles = await glob(GRPC_SCAN_GLOB, {
|
||||
cwd: repoPath,
|
||||
ignore: ['**/node_modules/**', '**/.git/**', '**/vendor/**', '**/dist/**', '**/build/**'],
|
||||
nodir: true,
|
||||
});
|
||||
|
||||
const parser = new Parser();
|
||||
for (const rel of sourceFiles) {
|
||||
const plugin = getPluginForFile(rel);
|
||||
if (!plugin) continue;
|
||||
const content = readSafe(repoPath, rel);
|
||||
if (!content) continue;
|
||||
let detections: GrpcDetection[] = [];
|
||||
try {
|
||||
parser.setLanguage(plugin.language);
|
||||
const tree = parser.parse(content);
|
||||
detections = plugin.scan(tree);
|
||||
} catch {
|
||||
continue;
|
||||
}
|
||||
for (const d of detections) {
|
||||
const contract = this.detectionToContract(d, rel, protoMap);
|
||||
if (contract) out.push(contract);
|
||||
}
|
||||
}
|
||||
|
||||
return this.dedupe(out);
|
||||
}
|
||||
|
||||
/**
|
||||
* Convert a plugin `GrpcDetection` into a concrete `ExtractedContract`
|
||||
* by resolving the short service name against the proto map, building
|
||||
* either a service-level (`grpc::pkg.Svc/*`) or method-level
|
||||
* (`grpc::pkg.Svc/Method`) contract id, and selecting confidence
|
||||
* based on whether the proto map had an entry.
|
||||
*/
|
||||
private detectionToContract(
|
||||
d: GrpcDetection,
|
||||
filePath: string,
|
||||
protoMap: Map<string, ProtoServiceInfo[]>,
|
||||
): ExtractedContract | null {
|
||||
const candidates = protoMap.get(d.serviceName) ?? [];
|
||||
const proto = resolveProtoConflict(d.serviceName, filePath, candidates);
|
||||
// If there were proto candidates but resolution was ambiguous, skip
|
||||
// contract emission rather than fabricating a package-qualified id from
|
||||
// an arbitrary candidate. resolveProtoConflict already warned.
|
||||
if (candidates.length > 0 && proto === null) return null;
|
||||
const pkg = proto?.package ?? '';
|
||||
const cid = d.methodName
|
||||
? contractId(pkg, d.serviceName, d.methodName)
|
||||
: proto
|
||||
? serviceContractId(pkg, d.serviceName)
|
||||
: serviceOnlyContractId(d.serviceName);
|
||||
const confidence = proto ? d.confidenceWithProto : d.confidenceWithoutProto;
|
||||
const meta: Record<string, unknown> = {
|
||||
service: d.serviceName,
|
||||
source: d.source,
|
||||
};
|
||||
if (d.methodName) meta.method = d.methodName;
|
||||
return makeContract(cid, d.role, filePath, d.symbolName, confidence, meta);
|
||||
}
|
||||
|
||||
private dedupe(items: ExtractedContract[]): ExtractedContract[] {
|
||||
const byKey = new Map<string, ExtractedContract>();
|
||||
for (const c of items) {
|
||||
const k = `${c.contractId}|${c.role}|${c.symbolRef.filePath}`;
|
||||
const existing = byKey.get(k);
|
||||
if (
|
||||
!existing ||
|
||||
c.confidence > existing.confidence ||
|
||||
(c.confidence === existing.confidence &&
|
||||
String(c.meta.source) < String(existing.meta.source))
|
||||
) {
|
||||
byKey.set(k, c);
|
||||
}
|
||||
}
|
||||
return Array.from(byKey.values());
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,109 @@
|
||||
import Go from 'tree-sitter-go';
|
||||
import {
|
||||
compilePatterns,
|
||||
runCompiledPatterns,
|
||||
type LanguagePatterns,
|
||||
} from '../tree-sitter-scanner.js';
|
||||
import type { GrpcDetection, GrpcLanguagePlugin } from './types.js';
|
||||
|
||||
/**
|
||||
* Go gRPC plugin. Detects:
|
||||
* - Provider: `pb.RegisterXxxServer(...)` calls
|
||||
* - Provider: `pb.UnimplementedXxxServer` embedded in a struct
|
||||
* - Consumer: `pb.NewXxxClient(conn)` calls
|
||||
*/
|
||||
|
||||
const REGISTER_RE = /^Register(\w+)Server$/;
|
||||
const UNIMPLEMENTED_RE = /^Unimplemented(\w+)Server$/;
|
||||
const NEW_CLIENT_RE = /^New(\w+)Client$/;
|
||||
|
||||
// Any `xxx.<fn>(...)` call — plugin filters the field identifier text.
|
||||
const SELECTOR_CALL_PATTERNS = compilePatterns({
|
||||
name: 'go-grpc-selector-call',
|
||||
language: Go,
|
||||
patterns: [
|
||||
{
|
||||
meta: {},
|
||||
query: `
|
||||
(call_expression
|
||||
function: (selector_expression
|
||||
field: (field_identifier) @fn))
|
||||
`,
|
||||
},
|
||||
],
|
||||
} satisfies LanguagePatterns<Record<string, never>>);
|
||||
|
||||
// Any `qualified_type` used as a struct field — for `pb.UnimplementedXxxServer`.
|
||||
const STRUCT_EMBEDDING_PATTERNS = compilePatterns({
|
||||
name: 'go-grpc-struct-embedding',
|
||||
language: Go,
|
||||
patterns: [
|
||||
{
|
||||
meta: {},
|
||||
query: `
|
||||
(struct_type
|
||||
(field_declaration_list
|
||||
(field_declaration
|
||||
type: (qualified_type
|
||||
name: (type_identifier) @field_type))))
|
||||
`,
|
||||
},
|
||||
],
|
||||
} satisfies LanguagePatterns<Record<string, never>>);
|
||||
|
||||
export const GO_GRPC_PLUGIN: GrpcLanguagePlugin = {
|
||||
name: 'go-grpc',
|
||||
language: Go,
|
||||
scan(tree) {
|
||||
const out: GrpcDetection[] = [];
|
||||
|
||||
for (const match of runCompiledPatterns(SELECTOR_CALL_PATTERNS, tree)) {
|
||||
const fnNode = match.captures.fn;
|
||||
if (!fnNode) continue;
|
||||
const fnText = fnNode.text;
|
||||
|
||||
const registerMatch = REGISTER_RE.exec(fnText);
|
||||
if (registerMatch) {
|
||||
out.push({
|
||||
role: 'provider',
|
||||
serviceName: registerMatch[1],
|
||||
symbolName: fnText,
|
||||
source: 'go_register',
|
||||
confidenceWithProto: 0.8,
|
||||
confidenceWithoutProto: 0.65,
|
||||
});
|
||||
continue;
|
||||
}
|
||||
|
||||
const newClientMatch = NEW_CLIENT_RE.exec(fnText);
|
||||
if (newClientMatch) {
|
||||
out.push({
|
||||
role: 'consumer',
|
||||
serviceName: newClientMatch[1],
|
||||
symbolName: fnText,
|
||||
source: 'go_client',
|
||||
confidenceWithProto: 0.75,
|
||||
confidenceWithoutProto: 0.55,
|
||||
});
|
||||
continue;
|
||||
}
|
||||
}
|
||||
|
||||
for (const match of runCompiledPatterns(STRUCT_EMBEDDING_PATTERNS, tree)) {
|
||||
const fieldNode = match.captures.field_type;
|
||||
if (!fieldNode) continue;
|
||||
const unimpl = UNIMPLEMENTED_RE.exec(fieldNode.text);
|
||||
if (!unimpl) continue;
|
||||
out.push({
|
||||
role: 'provider',
|
||||
serviceName: unimpl[1],
|
||||
symbolName: fieldNode.text,
|
||||
source: 'go_unimplemented',
|
||||
confidenceWithProto: 0.8,
|
||||
confidenceWithoutProto: 0.65,
|
||||
});
|
||||
}
|
||||
|
||||
return out;
|
||||
},
|
||||
};
|
||||
@@ -0,0 +1,53 @@
|
||||
import * as path from 'node:path';
|
||||
import type { GrpcLanguagePlugin } from './types.js';
|
||||
import { GO_GRPC_PLUGIN } from './go.js';
|
||||
import { JAVA_GRPC_PLUGIN } from './java.js';
|
||||
import { PYTHON_GRPC_PLUGIN } from './python.js';
|
||||
import { JAVASCRIPT_GRPC_PLUGIN, TYPESCRIPT_GRPC_PLUGIN, TSX_GRPC_PLUGIN } from './node.js';
|
||||
import { PROTO_GRPC_PLUGIN } from './proto.js';
|
||||
|
||||
export type { GrpcDetection, GrpcLanguagePlugin, GrpcRole } from './types.js';
|
||||
export { PROTO_GRPC_PLUGIN, extractPackageFromTree } from './proto.js';
|
||||
|
||||
/**
|
||||
* File-extension → gRPC language plugin registry. Mirrors the shape
|
||||
* of `http-patterns/index.ts` and `topic-patterns/index.ts`.
|
||||
*
|
||||
* `.proto` files are registered only when `tree-sitter-proto` is
|
||||
* available (it's an optionalDependency). When absent, the orchestrator
|
||||
* falls back to the built-in manual proto parser.
|
||||
*/
|
||||
const REGISTRY: Record<string, GrpcLanguagePlugin> = {
|
||||
'.go': GO_GRPC_PLUGIN,
|
||||
'.java': JAVA_GRPC_PLUGIN,
|
||||
'.py': PYTHON_GRPC_PLUGIN,
|
||||
'.js': JAVASCRIPT_GRPC_PLUGIN,
|
||||
'.jsx': JAVASCRIPT_GRPC_PLUGIN,
|
||||
'.ts': TYPESCRIPT_GRPC_PLUGIN,
|
||||
'.tsx': TSX_GRPC_PLUGIN,
|
||||
...(PROTO_GRPC_PLUGIN ? { '.proto': PROTO_GRPC_PLUGIN } : {}),
|
||||
};
|
||||
|
||||
/**
|
||||
* Glob for source files worth scanning for gRPC server/client patterns.
|
||||
* Includes `.proto` when the grammar is available.
|
||||
*/
|
||||
export const GRPC_SCAN_GLOB = PROTO_GRPC_PLUGIN
|
||||
? '**/*.{go,java,py,ts,tsx,js,jsx,proto}'
|
||||
: '**/*.{go,java,py,ts,tsx,js,jsx}';
|
||||
|
||||
/**
|
||||
* Whether the tree-sitter proto plugin is available. The orchestrator
|
||||
* uses this to decide between the tree-sitter path and the fallback
|
||||
* manual parser for `.proto` files.
|
||||
*/
|
||||
export const hasProtoPlugin = PROTO_GRPC_PLUGIN !== null;
|
||||
|
||||
/**
|
||||
* Return the gRPC plugin registered for the given file's extension,
|
||||
* or `undefined` if the extension is not registered.
|
||||
*/
|
||||
export function getPluginForFile(rel: string): GrpcLanguagePlugin | undefined {
|
||||
const ext = path.extname(rel).toLowerCase();
|
||||
return REGISTRY[ext];
|
||||
}
|
||||
@@ -0,0 +1,179 @@
|
||||
import Parser from 'tree-sitter';
|
||||
import Java from 'tree-sitter-java';
|
||||
import {
|
||||
compilePatterns,
|
||||
runCompiledPatterns,
|
||||
type LanguagePatterns,
|
||||
} from '../tree-sitter-scanner.js';
|
||||
import type { GrpcDetection, GrpcLanguagePlugin } from './types.js';
|
||||
|
||||
/**
|
||||
* Java gRPC plugin. Detects:
|
||||
* - Provider: classes extending `XxxServiceGrpc.XxxServiceImplBase`
|
||||
* (with or without a `@GrpcService` annotation; the annotation
|
||||
* only affects confidence labelling in the original regex version
|
||||
* — here we emit a single detection per class and pick the source
|
||||
* label based on whether the annotation is present).
|
||||
* - Consumer: `XxxServiceGrpc.newBlockingStub(ch)` /
|
||||
* `XxxServiceGrpc.newStub(ch)` calls.
|
||||
*/
|
||||
|
||||
const IMPL_BASE_RE = /^(\w+)ImplBase$/;
|
||||
const GRPC_SUFFIX_RE = /^(\w+)Grpc$/;
|
||||
|
||||
// Classes extending `ScopedType.ScopedType` where the inner name ends
|
||||
// in ImplBase. Covers `XxxServiceGrpc.XxxServiceImplBase`.
|
||||
// Note: tree-sitter-java's `scoped_type_identifier` exposes its two
|
||||
// segments as positional `type_identifier` children, NOT as named
|
||||
// `scope:`/`name:` fields. We match positionally here and rely on the
|
||||
// grammar's left-to-right ordering: first child = outer, second = inner.
|
||||
const SCOPED_IMPL_BASE_PATTERNS = compilePatterns({
|
||||
name: 'java-grpc-scoped-impl-base',
|
||||
language: Java,
|
||||
patterns: [
|
||||
{
|
||||
meta: {},
|
||||
query: `
|
||||
(class_declaration
|
||||
name: (identifier) @class_name
|
||||
superclass: (superclass
|
||||
(scoped_type_identifier
|
||||
(type_identifier) @outer
|
||||
(type_identifier) @inner (#match? @inner "ImplBase$")))) @class
|
||||
`,
|
||||
},
|
||||
],
|
||||
} satisfies LanguagePatterns<Record<string, never>>);
|
||||
|
||||
// Classes extending a simple `XxxImplBase` identifier (no scope).
|
||||
const PLAIN_IMPL_BASE_PATTERNS = compilePatterns({
|
||||
name: 'java-grpc-plain-impl-base',
|
||||
language: Java,
|
||||
patterns: [
|
||||
{
|
||||
meta: {},
|
||||
query: `
|
||||
(class_declaration
|
||||
name: (identifier) @class_name
|
||||
superclass: (superclass
|
||||
(type_identifier) @plain_type (#match? @plain_type "ImplBase$"))) @class
|
||||
`,
|
||||
},
|
||||
],
|
||||
} satisfies LanguagePatterns<Record<string, never>>);
|
||||
|
||||
// gRPC stub factories: `XxxGrpc.newStub(ch)` / `XxxGrpc.newBlockingStub(ch)`.
|
||||
const STUB_PATTERNS = compilePatterns({
|
||||
name: 'java-grpc-stub',
|
||||
language: Java,
|
||||
patterns: [
|
||||
{
|
||||
meta: {},
|
||||
query: `
|
||||
(method_invocation
|
||||
object: (identifier) @grpc_cls
|
||||
name: (identifier) @method (#match? @method "^new(Blocking)?Stub$"))
|
||||
`,
|
||||
},
|
||||
],
|
||||
} satisfies LanguagePatterns<Record<string, never>>);
|
||||
|
||||
/**
|
||||
* Check whether a `class_declaration` node has a `@GrpcService`
|
||||
* annotation in its modifiers list. In tree-sitter-java, class-level
|
||||
* annotations live under `(class_declaration (modifiers (marker_annotation|annotation)))`.
|
||||
*/
|
||||
function hasGrpcServiceAnnotation(classNode: Parser.SyntaxNode): boolean {
|
||||
for (let i = 0; i < classNode.namedChildCount; i++) {
|
||||
const child = classNode.namedChild(i);
|
||||
if (!child || child.type !== 'modifiers') continue;
|
||||
for (let j = 0; j < child.namedChildCount; j++) {
|
||||
const mod = child.namedChild(j);
|
||||
if (!mod) continue;
|
||||
if (mod.type !== 'marker_annotation' && mod.type !== 'annotation') continue;
|
||||
const nameNode = mod.childForFieldName('name');
|
||||
if (nameNode?.text === 'GrpcService') return true;
|
||||
}
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
/**
|
||||
* Given the inner type_identifier text like `AuthServiceImplBase`,
|
||||
* return the service name (`AuthService`), or null if the text
|
||||
* doesn't end in `ImplBase`.
|
||||
*/
|
||||
function extractServiceFromImplBase(text: string): string | null {
|
||||
const m = IMPL_BASE_RE.exec(text);
|
||||
if (!m) return null;
|
||||
// Strip a trailing `Grpc` on the service name too — the original
|
||||
// regex replaces `Grpc$` on the extracted prefix.
|
||||
return m[1].replace(/Grpc$/, '');
|
||||
}
|
||||
|
||||
export const JAVA_GRPC_PLUGIN: GrpcLanguagePlugin = {
|
||||
name: 'java-grpc',
|
||||
language: Java,
|
||||
scan(tree) {
|
||||
const out: GrpcDetection[] = [];
|
||||
const emittedClassIds = new Set<number>();
|
||||
|
||||
// ─── Providers: scoped form (`...Grpc.XxxImplBase`) ─────────────
|
||||
for (const match of runCompiledPatterns(SCOPED_IMPL_BASE_PATTERNS, tree)) {
|
||||
const classNode = match.captures.class;
|
||||
const innerNode = match.captures.inner;
|
||||
if (!classNode || !innerNode) continue;
|
||||
const serviceName = extractServiceFromImplBase(innerNode.text);
|
||||
if (!serviceName) continue;
|
||||
emittedClassIds.add(classNode.id);
|
||||
const annotated = hasGrpcServiceAnnotation(classNode);
|
||||
out.push({
|
||||
role: 'provider',
|
||||
serviceName,
|
||||
symbolName: serviceName,
|
||||
source: annotated ? 'java_grpc_service' : 'java_impl_base',
|
||||
confidenceWithProto: 0.8,
|
||||
confidenceWithoutProto: 0.65,
|
||||
});
|
||||
}
|
||||
|
||||
// ─── Providers: plain form (`XxxImplBase`) ──────────────────────
|
||||
for (const match of runCompiledPatterns(PLAIN_IMPL_BASE_PATTERNS, tree)) {
|
||||
const classNode = match.captures.class;
|
||||
const plainNode = match.captures.plain_type;
|
||||
if (!classNode || !plainNode) continue;
|
||||
if (emittedClassIds.has(classNode.id)) continue;
|
||||
const serviceName = extractServiceFromImplBase(plainNode.text);
|
||||
if (!serviceName) continue;
|
||||
emittedClassIds.add(classNode.id);
|
||||
const annotated = hasGrpcServiceAnnotation(classNode);
|
||||
out.push({
|
||||
role: 'provider',
|
||||
serviceName,
|
||||
symbolName: serviceName,
|
||||
source: annotated ? 'java_grpc_service' : 'java_impl_base',
|
||||
confidenceWithProto: 0.8,
|
||||
confidenceWithoutProto: 0.65,
|
||||
});
|
||||
}
|
||||
|
||||
// ─── Consumers: `XxxGrpc.newBlockingStub(...)` / `newStub(...)` ─
|
||||
for (const match of runCompiledPatterns(STUB_PATTERNS, tree)) {
|
||||
const grpcClsNode = match.captures.grpc_cls;
|
||||
if (!grpcClsNode) continue;
|
||||
const grpcMatch = GRPC_SUFFIX_RE.exec(grpcClsNode.text);
|
||||
if (!grpcMatch) continue;
|
||||
const serviceName = grpcMatch[1];
|
||||
out.push({
|
||||
role: 'consumer',
|
||||
serviceName,
|
||||
symbolName: `${serviceName}Stub`,
|
||||
source: 'java_stub',
|
||||
confidenceWithProto: 0.75,
|
||||
confidenceWithoutProto: 0.55,
|
||||
});
|
||||
}
|
||||
|
||||
return out;
|
||||
},
|
||||
};
|
||||
@@ -0,0 +1,314 @@
|
||||
import Parser from 'tree-sitter';
|
||||
import JavaScript from 'tree-sitter-javascript';
|
||||
import TypeScript from 'tree-sitter-typescript';
|
||||
import {
|
||||
compilePatterns,
|
||||
runCompiledPatterns,
|
||||
unquoteLiteral,
|
||||
type CompiledPatterns,
|
||||
type LanguagePatterns,
|
||||
type PatternSpec,
|
||||
} from '../tree-sitter-scanner.js';
|
||||
import type { GrpcDetection, GrpcLanguagePlugin } from './types.js';
|
||||
|
||||
/**
|
||||
* Node.js / TypeScript gRPC plugin family. Detects:
|
||||
* - Provider: NestJS `@GrpcMethod('Service', 'Method')` decorators
|
||||
* - Consumer: NestJS `@GrpcClient(...) readonly x!: XxxServiceClient`
|
||||
* - Consumer: `client.getService<X>('AuthService')`
|
||||
* - Consumer: `new XxxServiceClient(...)` (generated client constructor)
|
||||
* - Consumer: `new foo.bar.Xxx(...)` when the file uses
|
||||
* `loadPackageDefinition` (gRPC dynamic proto loader)
|
||||
*
|
||||
* As with the HTTP `node.ts`, pattern sources are defined once and
|
||||
* compiled against three grammar variants (JS / TS / TSX) because
|
||||
* `Parser.Query` is not portable across grammar objects.
|
||||
*/
|
||||
|
||||
const SERVICE_CLIENT_RE = /^(\w+Service)Client$/;
|
||||
const CAPITALIZED_SERVICE_RE = /^[A-Z]\w+$/;
|
||||
|
||||
// @GrpcMethod('Service', 'Method')
|
||||
const GRPC_METHOD_SPEC: PatternSpec<Record<string, never>> = {
|
||||
meta: {},
|
||||
query: `
|
||||
(decorator
|
||||
(call_expression
|
||||
function: (identifier) @dec (#eq? @dec "GrpcMethod")
|
||||
arguments: (arguments
|
||||
. [(string) (template_string)] @service
|
||||
. [(string) (template_string)] @method)))
|
||||
`,
|
||||
};
|
||||
|
||||
// @GrpcClient(...) standalone decorator — the plugin walks to the next
|
||||
// sibling (a field definition) to read its type annotation.
|
||||
const GRPC_CLIENT_SPEC: PatternSpec<Record<string, never>> = {
|
||||
meta: {},
|
||||
query: `
|
||||
(decorator
|
||||
(call_expression
|
||||
function: (identifier) @dec (#eq? @dec "GrpcClient"))) @grpc_client_decorator
|
||||
`,
|
||||
};
|
||||
|
||||
// `.getService<X>('AuthService')` / `.getService('AuthService')`
|
||||
const GET_SERVICE_SPEC: PatternSpec<Record<string, never>> = {
|
||||
meta: {},
|
||||
query: `
|
||||
(call_expression
|
||||
function: (member_expression
|
||||
property: (property_identifier) @method (#eq? @method "getService"))
|
||||
arguments: (arguments . [(string) (template_string)] @service))
|
||||
`,
|
||||
};
|
||||
|
||||
// `new XxxServiceClient(...)` — bare identifier constructor.
|
||||
const NEW_SIMPLE_CTOR_SPEC: PatternSpec<Record<string, never>> = {
|
||||
meta: {},
|
||||
query: `
|
||||
(new_expression
|
||||
constructor: (identifier) @ctor)
|
||||
`,
|
||||
};
|
||||
|
||||
// `new foo.bar.XxxService(...)` — qualified constructor.
|
||||
const NEW_QUALIFIED_CTOR_SPEC: PatternSpec<Record<string, never>> = {
|
||||
meta: {},
|
||||
query: `
|
||||
(new_expression
|
||||
constructor: (member_expression
|
||||
property: (property_identifier) @ctor))
|
||||
`,
|
||||
};
|
||||
|
||||
// Detect whether the file uses `loadPackageDefinition` (gRPC dynamic
|
||||
// proto loader). Matches either a bare call or an `obj.loadPackageDefinition(...)`
|
||||
// call. Plugin gates the qualified-constructor consumer on this —
|
||||
// structural check avoids materializing `tree.rootNode.text` for every file.
|
||||
const LOAD_PACKAGE_DEFINITION_SPEC: PatternSpec<Record<string, never>> = {
|
||||
meta: {},
|
||||
query: `
|
||||
(call_expression
|
||||
function: [
|
||||
(identifier) @fn (#eq? @fn "loadPackageDefinition")
|
||||
(member_expression property: (property_identifier) @fn (#eq? @fn "loadPackageDefinition"))
|
||||
])
|
||||
`,
|
||||
};
|
||||
|
||||
interface NodeGrpcPatternBundle {
|
||||
grpcMethod: CompiledPatterns<Record<string, never>>;
|
||||
grpcClient: CompiledPatterns<Record<string, never>>;
|
||||
getService: CompiledPatterns<Record<string, never>>;
|
||||
newSimpleCtor: CompiledPatterns<Record<string, never>>;
|
||||
newQualifiedCtor: CompiledPatterns<Record<string, never>>;
|
||||
loadPackageDefinition: CompiledPatterns<Record<string, never>>;
|
||||
}
|
||||
|
||||
function compileBundle(language: unknown, name: string): NodeGrpcPatternBundle {
|
||||
const mk = (spec: PatternSpec<Record<string, never>>, suffix: string) =>
|
||||
compilePatterns({
|
||||
name: `${name}-${suffix}`,
|
||||
language,
|
||||
patterns: [spec],
|
||||
} satisfies LanguagePatterns<Record<string, never>>);
|
||||
return {
|
||||
grpcMethod: mk(GRPC_METHOD_SPEC, 'grpc-method'),
|
||||
grpcClient: mk(GRPC_CLIENT_SPEC, 'grpc-client'),
|
||||
getService: mk(GET_SERVICE_SPEC, 'get-service'),
|
||||
newSimpleCtor: mk(NEW_SIMPLE_CTOR_SPEC, 'new-simple-ctor'),
|
||||
newQualifiedCtor: mk(NEW_QUALIFIED_CTOR_SPEC, 'new-qualified-ctor'),
|
||||
loadPackageDefinition: mk(LOAD_PACKAGE_DEFINITION_SPEC, 'load-package-definition'),
|
||||
};
|
||||
}
|
||||
|
||||
const JAVASCRIPT_BUNDLE = compileBundle(JavaScript, 'javascript-grpc');
|
||||
const TYPESCRIPT_BUNDLE = compileBundle(TypeScript.typescript, 'typescript-grpc');
|
||||
const TSX_BUNDLE = compileBundle(TypeScript.tsx, 'tsx-grpc');
|
||||
|
||||
/**
|
||||
* Given a `@GrpcClient(...)` decorator node, find the type annotation
|
||||
* text of the field it decorates (e.g. `AuthServiceClient`).
|
||||
*
|
||||
* In tree-sitter-typescript, decorators on class fields can appear in
|
||||
* two configurations:
|
||||
* - As a CHILD of `public_field_definition` alongside the field's
|
||||
* type annotation (the common case for NestJS `@GrpcClient`).
|
||||
* - As a SIBLING of the field in `class_body` (for method
|
||||
* decorators, but kept for resilience against grammar variants).
|
||||
* We walk the parent container and search for a type annotation.
|
||||
*/
|
||||
function resolveGrpcClientFieldType(decoratorNode: Parser.SyntaxNode): string | null {
|
||||
const parent = decoratorNode.parent;
|
||||
if (!parent) return null;
|
||||
|
||||
// Case 1: decorator is a child of the field definition — search
|
||||
// the parent itself (which is the field definition) for a
|
||||
// type_annotation child.
|
||||
if (parent.type === 'public_field_definition' || parent.type.endsWith('field_definition')) {
|
||||
return findFirstTypeAnnotationText(parent);
|
||||
}
|
||||
|
||||
// Case 2: decorator is a sibling of the field in a class_body — walk
|
||||
// forward through subsequent siblings until we find a node containing
|
||||
// a type annotation.
|
||||
for (let i = 0; i < parent.namedChildCount; i++) {
|
||||
const child = parent.namedChild(i);
|
||||
if (child && child.id === decoratorNode.id) {
|
||||
for (let j = i + 1; j < parent.namedChildCount; j++) {
|
||||
const next = parent.namedChild(j);
|
||||
if (!next) continue;
|
||||
if (next.type === 'decorator') continue;
|
||||
const typeText = findFirstTypeAnnotationText(next);
|
||||
if (typeText) return typeText;
|
||||
return null;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/**
|
||||
* Recursively search `node` for the first `type_annotation` child and
|
||||
* return the text of its inner `type_identifier`, or null. Handles
|
||||
* both `public_field_definition` and its variants.
|
||||
*/
|
||||
function findFirstTypeAnnotationText(node: Parser.SyntaxNode): string | null {
|
||||
if (node.type === 'type_annotation') {
|
||||
for (let i = 0; i < node.namedChildCount; i++) {
|
||||
const child = node.namedChild(i);
|
||||
if (!child) continue;
|
||||
if (child.type === 'type_identifier') return child.text;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
for (let i = 0; i < node.namedChildCount; i++) {
|
||||
const child = node.namedChild(i);
|
||||
if (!child) continue;
|
||||
const found = findFirstTypeAnnotationText(child);
|
||||
if (found) return found;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
function scanBundle(bundle: NodeGrpcPatternBundle, tree: Parser.Tree): GrpcDetection[] {
|
||||
const out: GrpcDetection[] = [];
|
||||
|
||||
// ─── Provider: @GrpcMethod('Service', 'Method') ──────────────────
|
||||
for (const match of runCompiledPatterns(bundle.grpcMethod, tree)) {
|
||||
const svcNode = match.captures.service;
|
||||
const methodNode = match.captures.method;
|
||||
if (!svcNode || !methodNode) continue;
|
||||
const svc = unquoteLiteral(svcNode.text);
|
||||
const mth = unquoteLiteral(methodNode.text);
|
||||
if (!svc || !mth) continue;
|
||||
out.push({
|
||||
role: 'provider',
|
||||
serviceName: svc,
|
||||
symbolName: `${svc}.${mth}`,
|
||||
source: 'ts_grpc_method',
|
||||
methodName: mth,
|
||||
// @GrpcMethod hard-coded confidence 0.8 in the original code
|
||||
// regardless of whether the proto map has a match.
|
||||
confidenceWithProto: 0.8,
|
||||
confidenceWithoutProto: 0.8,
|
||||
});
|
||||
}
|
||||
|
||||
// ─── Consumer: @GrpcClient() field with XxxServiceClient type ────
|
||||
for (const match of runCompiledPatterns(bundle.grpcClient, tree)) {
|
||||
const decoratorNode = match.captures.grpc_client_decorator;
|
||||
if (!decoratorNode) continue;
|
||||
const typeText = resolveGrpcClientFieldType(decoratorNode);
|
||||
if (!typeText) continue;
|
||||
const svcMatch = SERVICE_CLIENT_RE.exec(typeText);
|
||||
if (!svcMatch) continue;
|
||||
const serviceName = svcMatch[1];
|
||||
out.push({
|
||||
role: 'consumer',
|
||||
serviceName,
|
||||
symbolName: `${serviceName}Client`,
|
||||
source: 'ts_grpc_client_decorator',
|
||||
confidenceWithProto: 0.75,
|
||||
confidenceWithoutProto: 0.55,
|
||||
});
|
||||
}
|
||||
|
||||
// ─── Consumer: client.getService<X>('Service') ───────────────────
|
||||
for (const match of runCompiledPatterns(bundle.getService, tree)) {
|
||||
const svcNode = match.captures.service;
|
||||
if (!svcNode) continue;
|
||||
const svc = unquoteLiteral(svcNode.text);
|
||||
if (!svc) continue;
|
||||
out.push({
|
||||
role: 'consumer',
|
||||
serviceName: svc,
|
||||
symbolName: `${svc}Client`,
|
||||
source: 'ts_client_grpc_get_service',
|
||||
confidenceWithProto: 0.75,
|
||||
confidenceWithoutProto: 0.55,
|
||||
});
|
||||
}
|
||||
|
||||
// ─── Consumer: new XxxServiceClient(...) ─────────────────────────
|
||||
for (const match of runCompiledPatterns(bundle.newSimpleCtor, tree)) {
|
||||
const ctorNode = match.captures.ctor;
|
||||
if (!ctorNode) continue;
|
||||
const svcMatch = SERVICE_CLIENT_RE.exec(ctorNode.text);
|
||||
if (!svcMatch) continue;
|
||||
const serviceName = svcMatch[1];
|
||||
out.push({
|
||||
role: 'consumer',
|
||||
serviceName,
|
||||
symbolName: `${serviceName}Client`,
|
||||
source: 'ts_generated_client',
|
||||
confidenceWithProto: 0.75,
|
||||
confidenceWithoutProto: 0.55,
|
||||
});
|
||||
}
|
||||
|
||||
// ─── Consumer: loadPackageDefinition dynamic proto loader ────────
|
||||
// Only emit when the file uses loadPackageDefinition, otherwise a
|
||||
// generic `new foo.bar.Something()` in unrelated code would falsely
|
||||
// register as a gRPC consumer. Check structurally via a dedicated
|
||||
// query — avoids materializing `tree.rootNode.text` for the whole
|
||||
// file (expensive on large files).
|
||||
const usesLoadPackage = runCompiledPatterns(bundle.loadPackageDefinition, tree).length > 0;
|
||||
if (usesLoadPackage) {
|
||||
for (const match of runCompiledPatterns(bundle.newQualifiedCtor, tree)) {
|
||||
const ctorNode = match.captures.ctor;
|
||||
if (!ctorNode) continue;
|
||||
if (!CAPITALIZED_SERVICE_RE.test(ctorNode.text)) continue;
|
||||
out.push({
|
||||
role: 'consumer',
|
||||
serviceName: ctorNode.text,
|
||||
symbolName: `${ctorNode.text}Client`,
|
||||
source: 'ts_load_package_definition',
|
||||
confidenceWithProto: 0.75,
|
||||
confidenceWithoutProto: 0.55,
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
return out;
|
||||
}
|
||||
|
||||
export const JAVASCRIPT_GRPC_PLUGIN: GrpcLanguagePlugin = {
|
||||
name: 'javascript-grpc',
|
||||
language: JavaScript,
|
||||
scan: (tree) => scanBundle(JAVASCRIPT_BUNDLE, tree),
|
||||
};
|
||||
|
||||
export const TYPESCRIPT_GRPC_PLUGIN: GrpcLanguagePlugin = {
|
||||
name: 'typescript-grpc',
|
||||
language: TypeScript.typescript,
|
||||
scan: (tree) => scanBundle(TYPESCRIPT_BUNDLE, tree),
|
||||
};
|
||||
|
||||
export const TSX_GRPC_PLUGIN: GrpcLanguagePlugin = {
|
||||
name: 'tsx-grpc',
|
||||
language: TypeScript.tsx,
|
||||
scan: (tree) => scanBundle(TSX_BUNDLE, tree),
|
||||
};
|
||||
@@ -0,0 +1,147 @@
|
||||
import { createRequire } from 'node:module';
|
||||
import {
|
||||
compilePatterns,
|
||||
runCompiledPatterns,
|
||||
type CompiledPatterns,
|
||||
type LanguagePatterns,
|
||||
} from '../tree-sitter-scanner.js';
|
||||
import type { GrpcDetection, GrpcLanguagePlugin } from './types.js';
|
||||
|
||||
/**
|
||||
* Protobuf (.proto) tree-sitter plugin for gRPC contract extraction.
|
||||
*
|
||||
* Uses `tree-sitter-proto` (coder3101/tree-sitter-proto) as an
|
||||
* optionalDependency — if the grammar is not installed (e.g. native
|
||||
* compilation failed on an unusual platform), the plugin exports
|
||||
* `null` and the orchestrator falls back to the existing manual
|
||||
* string-sanitizing parser.
|
||||
*
|
||||
* The grammar is vendored in `vendor/tree-sitter-proto/` with
|
||||
* parser.c regenerated against tree-sitter-cli 0.24 (ABI version 14)
|
||||
* so it is compatible with the project's tree-sitter 0.25 runtime.
|
||||
*/
|
||||
|
||||
const _require = createRequire(import.meta.url);
|
||||
let ProtoGrammar: unknown = null;
|
||||
try {
|
||||
ProtoGrammar = _require('tree-sitter-proto');
|
||||
} catch {
|
||||
// Grammar not installed — PROTO_GRPC_PLUGIN will be null.
|
||||
}
|
||||
|
||||
let PACKAGE_PATTERNS: CompiledPatterns<Record<string, never>> | null = null;
|
||||
let SERVICE_PATTERNS: CompiledPatterns<Record<string, never>> | null = null;
|
||||
|
||||
if (ProtoGrammar) {
|
||||
try {
|
||||
// Validate that the grammar actually loads end-to-end: compile queries
|
||||
// AND parse + walk a trivial proto file. tree-sitter's internal
|
||||
// `initializeLanguageNodeClasses` can fail with a TDZ error in some
|
||||
// test runners (vitest forks) when SyntaxNode isn't fully initialized
|
||||
// yet. Catching that here ensures `PROTO_GRPC_PLUGIN` stays null and
|
||||
// the orchestrator falls back to the manual parser.
|
||||
// eslint-disable-next-line @typescript-eslint/no-explicit-any
|
||||
const _Parser = _require('tree-sitter') as any;
|
||||
// Smoke-test: parse + setLanguage to verify the grammar is
|
||||
// end-to-end compatible with this tree-sitter runtime.
|
||||
const _testParser = new _Parser();
|
||||
_testParser.setLanguage(ProtoGrammar);
|
||||
_testParser.parse('service X { rpc Y (R) returns (R); }');
|
||||
|
||||
PACKAGE_PATTERNS = compilePatterns({
|
||||
name: 'proto-package',
|
||||
language: ProtoGrammar,
|
||||
patterns: [
|
||||
{
|
||||
meta: {},
|
||||
query: `(package (full_ident) @pkg)`,
|
||||
},
|
||||
],
|
||||
} satisfies LanguagePatterns<Record<string, never>>);
|
||||
|
||||
SERVICE_PATTERNS = compilePatterns({
|
||||
name: 'proto-service',
|
||||
language: ProtoGrammar,
|
||||
patterns: [
|
||||
{
|
||||
meta: {},
|
||||
query: `
|
||||
(service
|
||||
(service_name) @service_name
|
||||
(rpc
|
||||
(rpc_name) @rpc_name))
|
||||
`,
|
||||
},
|
||||
],
|
||||
} satisfies LanguagePatterns<Record<string, never>>);
|
||||
} catch {
|
||||
// Compilation failed (grammar ABI mismatch?) — fall back to null.
|
||||
PACKAGE_PATTERNS = null;
|
||||
SERVICE_PATTERNS = null;
|
||||
ProtoGrammar = null;
|
||||
}
|
||||
}
|
||||
|
||||
function buildPlugin(): GrpcLanguagePlugin | null {
|
||||
if (!ProtoGrammar || !PACKAGE_PATTERNS || !SERVICE_PATTERNS) return null;
|
||||
const pkgPatterns = PACKAGE_PATTERNS;
|
||||
const svcPatterns = SERVICE_PATTERNS;
|
||||
|
||||
return {
|
||||
name: 'proto-grpc',
|
||||
language: ProtoGrammar,
|
||||
scan(tree) {
|
||||
const out: GrpcDetection[] = [];
|
||||
|
||||
// Extract `package` declaration (first match wins).
|
||||
let pkg = '';
|
||||
for (const match of runCompiledPatterns(pkgPatterns, tree)) {
|
||||
const pkgNode = match.captures.pkg;
|
||||
if (pkgNode) {
|
||||
pkg = pkgNode.text;
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
// Extract `service → rpc` pairs. The query returns one match per
|
||||
// (service, rpc) combination thanks to the nested structure.
|
||||
for (const match of runCompiledPatterns(svcPatterns, tree)) {
|
||||
const serviceNode = match.captures.service_name;
|
||||
const rpcNode = match.captures.rpc_name;
|
||||
if (!serviceNode || !rpcNode) continue;
|
||||
const serviceName = serviceNode.text;
|
||||
const methodName = rpcNode.text;
|
||||
out.push({
|
||||
role: 'provider',
|
||||
serviceName,
|
||||
symbolName: `${serviceName}.${methodName}`,
|
||||
source: 'proto',
|
||||
methodName,
|
||||
// Proto definitions are the canonical source of truth — always
|
||||
// high confidence regardless of cross-referencing.
|
||||
confidenceWithProto: 0.85,
|
||||
confidenceWithoutProto: 0.85,
|
||||
});
|
||||
}
|
||||
|
||||
return out;
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* The proto plugin, or `null` if tree-sitter-proto is not available.
|
||||
* The orchestrator checks this at import time and decides whether to
|
||||
* use the tree-sitter path or the fallback manual parser.
|
||||
*/
|
||||
export const PROTO_GRPC_PLUGIN: GrpcLanguagePlugin | null = buildPlugin();
|
||||
|
||||
/** The package declaration text from a proto file's tree. */
|
||||
export function extractPackageFromTree(tree: import('tree-sitter').Tree): string {
|
||||
if (!PACKAGE_PATTERNS) return '';
|
||||
for (const match of runCompiledPatterns(PACKAGE_PATTERNS, tree)) {
|
||||
const pkgNode = match.captures.pkg;
|
||||
if (pkgNode) return pkgNode.text;
|
||||
}
|
||||
return '';
|
||||
}
|
||||
@@ -0,0 +1,77 @@
|
||||
import Python from 'tree-sitter-python';
|
||||
import {
|
||||
compilePatterns,
|
||||
runCompiledPatterns,
|
||||
type LanguagePatterns,
|
||||
} from '../tree-sitter-scanner.js';
|
||||
import type { GrpcDetection, GrpcLanguagePlugin } from './types.js';
|
||||
|
||||
/**
|
||||
* Python gRPC plugin. Detects:
|
||||
* - Provider: `add_XxxServicer_to_server(...)` calls (bare identifier
|
||||
* or qualified attribute form `auth_pb2_grpc.add_XxxServicer_to_server`)
|
||||
* - Consumer: `XxxStub(channel)` calls (bare or `auth_pb2_grpc.XxxStub`)
|
||||
*/
|
||||
|
||||
const ADD_SERVICER_RE = /^add_(\w+)Servicer_to_server$/;
|
||||
const STUB_RE = /^(\w+)Stub$/;
|
||||
/** Reserved names that would produce garbage service names. */
|
||||
const STUB_IGNORE = new Set(['Mock', 'Test', 'Fake', 'Stub']);
|
||||
|
||||
// Any call whose target is either a bare identifier or an attribute
|
||||
// access (`obj.method`). The plugin filters the function name in JS.
|
||||
const CALL_PATTERNS = compilePatterns({
|
||||
name: 'python-grpc-call',
|
||||
language: Python,
|
||||
patterns: [
|
||||
{
|
||||
meta: {},
|
||||
query: `
|
||||
(call
|
||||
function: [
|
||||
(identifier) @fn
|
||||
(attribute attribute: (identifier) @fn)
|
||||
])
|
||||
`,
|
||||
},
|
||||
],
|
||||
} satisfies LanguagePatterns<Record<string, never>>);
|
||||
|
||||
export const PYTHON_GRPC_PLUGIN: GrpcLanguagePlugin = {
|
||||
name: 'python-grpc',
|
||||
language: Python,
|
||||
scan(tree) {
|
||||
const out: GrpcDetection[] = [];
|
||||
for (const match of runCompiledPatterns(CALL_PATTERNS, tree)) {
|
||||
const fnNode = match.captures.fn;
|
||||
if (!fnNode) continue;
|
||||
const fnText = fnNode.text;
|
||||
|
||||
const addServicer = ADD_SERVICER_RE.exec(fnText);
|
||||
if (addServicer) {
|
||||
out.push({
|
||||
role: 'provider',
|
||||
serviceName: addServicer[1],
|
||||
symbolName: fnText,
|
||||
source: 'python_servicer',
|
||||
confidenceWithProto: 0.8,
|
||||
confidenceWithoutProto: 0.65,
|
||||
});
|
||||
continue;
|
||||
}
|
||||
|
||||
const stubMatch = STUB_RE.exec(fnText);
|
||||
if (stubMatch && !STUB_IGNORE.has(stubMatch[1])) {
|
||||
out.push({
|
||||
role: 'consumer',
|
||||
serviceName: stubMatch[1],
|
||||
symbolName: fnText,
|
||||
source: 'python_stub',
|
||||
confidenceWithProto: 0.75,
|
||||
confidenceWithoutProto: 0.55,
|
||||
});
|
||||
}
|
||||
}
|
||||
return out;
|
||||
},
|
||||
};
|
||||
@@ -0,0 +1,54 @@
|
||||
import type Parser from 'tree-sitter';
|
||||
|
||||
/**
|
||||
* Shared types for the grpc-extractor language plugins.
|
||||
*
|
||||
* Each plugin lives in its own file (java.ts, go.ts, ...) and owns the
|
||||
* tree-sitter grammar import + query sources. The top-level
|
||||
* `grpc-extractor.ts` orchestrator only knows about this type module
|
||||
* and the plugin registry (`./index.ts`). It MUST NOT import any
|
||||
* grammar or query text directly.
|
||||
*/
|
||||
|
||||
export type GrpcRole = 'provider' | 'consumer';
|
||||
|
||||
/**
|
||||
* One raw gRPC detection produced by a plugin's `scan()` function. The
|
||||
* orchestrator uses the proto map to resolve the full package-qualified
|
||||
* contract id and choose a confidence based on whether the proto was
|
||||
* found.
|
||||
*
|
||||
* Most patterns produce service-level detections; `TS @GrpcMethod` is
|
||||
* the only pattern that captures an explicit `methodName`, producing
|
||||
* a method-level contract (`grpc::pkg.Service/Method`).
|
||||
*/
|
||||
export interface GrpcDetection {
|
||||
role: GrpcRole;
|
||||
/** Short service name, e.g. `"AuthService"`. */
|
||||
serviceName: string;
|
||||
/** Symbol name emitted into the contract's symbolRef. */
|
||||
symbolName: string;
|
||||
/** Metadata source label (goes into `meta.source`). */
|
||||
source: string;
|
||||
/** Explicit method name; set only by TS `@GrpcMethod`. */
|
||||
methodName?: string;
|
||||
/** Confidence when the proto map resolves the service. */
|
||||
confidenceWithProto: number;
|
||||
/** Confidence when the proto map has no entry. */
|
||||
confidenceWithoutProto: number;
|
||||
}
|
||||
|
||||
/**
|
||||
* One language-scoped gRPC plugin. Plugins own the tree-sitter grammar
|
||||
* and a `scan(tree)` function that returns zero or more
|
||||
* `GrpcDetection`s. The plugin is free to run multiple compiled query
|
||||
* bundles and walk the AST to cross-reference captures.
|
||||
*
|
||||
* `language` is typed `unknown` for the same reason as in
|
||||
* `tree-sitter-scanner.ts`.
|
||||
*/
|
||||
export interface GrpcLanguagePlugin {
|
||||
name: string;
|
||||
language: unknown;
|
||||
scan(tree: Parser.Tree): GrpcDetection[];
|
||||
}
|
||||
@@ -0,0 +1,224 @@
|
||||
import Go from 'tree-sitter-go';
|
||||
import {
|
||||
compilePatterns,
|
||||
runCompiledPatterns,
|
||||
unquoteLiteral,
|
||||
type LanguagePatterns,
|
||||
} from '../tree-sitter-scanner.js';
|
||||
import type { HttpDetection, HttpLanguagePlugin } from './types.js';
|
||||
|
||||
/**
|
||||
* Go HTTP plugin. Handles:
|
||||
* - gin / echo / chi framework routing — `r.GET("/path", handler)`
|
||||
* - net/http stdlib — `http.HandleFunc("/path", handler)`
|
||||
* - net/http consumer — `http.Get(...)`, `http.NewRequest("METHOD", ...)`
|
||||
* - resty consumer — `client.R().Delete("/path")`
|
||||
*/
|
||||
|
||||
// ─── Provider: framework routing ──────────────────────────────────────
|
||||
// Matches `\w+\.GET(...)` etc. (gin, echo, chi all share this shape).
|
||||
// Captures the HTTP method (field name), path literal, and handler
|
||||
// identifier passed as the second argument.
|
||||
const FRAMEWORK_ROUTE_PATTERNS = compilePatterns({
|
||||
name: 'go-framework-route',
|
||||
language: Go,
|
||||
patterns: [
|
||||
{
|
||||
meta: {},
|
||||
query: `
|
||||
(call_expression
|
||||
function: (selector_expression
|
||||
field: (field_identifier) @http_method (#match? @http_method "^(GET|POST|PUT|DELETE|PATCH)$"))
|
||||
arguments: (argument_list
|
||||
(interpreted_string_literal) @path
|
||||
(identifier) @handler))
|
||||
`,
|
||||
},
|
||||
],
|
||||
} satisfies LanguagePatterns<Record<string, never>>);
|
||||
|
||||
// ─── Provider: net/http `http.HandleFunc("/p", handler)` ─────────────
|
||||
const HANDLE_FUNC_PATTERNS = compilePatterns({
|
||||
name: 'go-handle-func',
|
||||
language: Go,
|
||||
patterns: [
|
||||
{
|
||||
meta: {},
|
||||
query: `
|
||||
(call_expression
|
||||
function: (selector_expression
|
||||
operand: (identifier) @pkg (#eq? @pkg "http")
|
||||
field: (field_identifier) @fn (#eq? @fn "HandleFunc"))
|
||||
arguments: (argument_list
|
||||
(interpreted_string_literal) @path
|
||||
(identifier) @handler))
|
||||
`,
|
||||
},
|
||||
],
|
||||
} satisfies LanguagePatterns<Record<string, never>>);
|
||||
|
||||
// ─── Consumer: net/http stdlib Get / Post / Head ─────────────────────
|
||||
const HTTP_CLIENT_METHOD_TO_HTTP: Record<string, string> = {
|
||||
Get: 'GET',
|
||||
Post: 'POST',
|
||||
Head: 'GET', // HEAD has no body semantics we care about — treat as GET for contract matching
|
||||
};
|
||||
|
||||
const HTTP_CLIENT_PATTERNS = compilePatterns({
|
||||
name: 'go-http-client',
|
||||
language: Go,
|
||||
patterns: [
|
||||
{
|
||||
meta: {},
|
||||
query: `
|
||||
(call_expression
|
||||
function: (selector_expression
|
||||
operand: (identifier) @pkg (#eq? @pkg "http")
|
||||
field: (field_identifier) @fn (#match? @fn "^(Get|Post|Head)$"))
|
||||
arguments: (argument_list . (interpreted_string_literal) @path))
|
||||
`,
|
||||
},
|
||||
],
|
||||
} satisfies LanguagePatterns<Record<string, never>>);
|
||||
|
||||
// ─── Consumer: net/http `http.NewRequest("METHOD", "/path", ...)` ────
|
||||
const NEW_REQUEST_PATTERNS = compilePatterns({
|
||||
name: 'go-new-request',
|
||||
language: Go,
|
||||
patterns: [
|
||||
{
|
||||
meta: {},
|
||||
query: `
|
||||
(call_expression
|
||||
function: (selector_expression
|
||||
operand: (identifier) @pkg (#eq? @pkg "http")
|
||||
field: (field_identifier) @fn (#eq? @fn "NewRequest"))
|
||||
arguments: (argument_list
|
||||
.
|
||||
(interpreted_string_literal) @http_method
|
||||
(interpreted_string_literal) @path))
|
||||
`,
|
||||
},
|
||||
],
|
||||
} satisfies LanguagePatterns<Record<string, never>>);
|
||||
|
||||
// ─── Consumer: resty `client.R().Delete("/path")` ─────────────────────
|
||||
// Matches any chained call whose receiver is `something.R()` and whose
|
||||
// method name is an HTTP verb. This is how go-resty's fluent API looks.
|
||||
const RESTY_PATTERNS = compilePatterns({
|
||||
name: 'go-resty',
|
||||
language: Go,
|
||||
patterns: [
|
||||
{
|
||||
meta: {},
|
||||
query: `
|
||||
(call_expression
|
||||
function: (selector_expression
|
||||
operand: (call_expression
|
||||
function: (selector_expression
|
||||
field: (field_identifier) @r (#eq? @r "R")))
|
||||
field: (field_identifier) @http_method (#match? @http_method "^(Get|Post|Put|Delete|Patch)$"))
|
||||
arguments: (argument_list . (interpreted_string_literal) @path))
|
||||
`,
|
||||
},
|
||||
],
|
||||
} satisfies LanguagePatterns<Record<string, never>>);
|
||||
|
||||
export const GO_HTTP_PLUGIN: HttpLanguagePlugin = {
|
||||
name: 'go-http',
|
||||
language: Go,
|
||||
scan(tree) {
|
||||
const out: HttpDetection[] = [];
|
||||
|
||||
// Framework providers: r.GET/POST/... with handler identifier
|
||||
for (const match of runCompiledPatterns(FRAMEWORK_ROUTE_PATTERNS, tree)) {
|
||||
const methodNode = match.captures.http_method;
|
||||
const pathNode = match.captures.path;
|
||||
const handlerNode = match.captures.handler;
|
||||
if (!methodNode || !pathNode) continue;
|
||||
const path = unquoteLiteral(pathNode.text);
|
||||
if (path === null) continue;
|
||||
out.push({
|
||||
role: 'provider',
|
||||
framework: 'go-framework',
|
||||
method: methodNode.text.toUpperCase(),
|
||||
path,
|
||||
name: handlerNode?.text ?? null,
|
||||
confidence: 0.8,
|
||||
});
|
||||
}
|
||||
|
||||
// net/http HandleFunc: default method GET
|
||||
for (const match of runCompiledPatterns(HANDLE_FUNC_PATTERNS, tree)) {
|
||||
const pathNode = match.captures.path;
|
||||
const handlerNode = match.captures.handler;
|
||||
if (!pathNode) continue;
|
||||
const path = unquoteLiteral(pathNode.text);
|
||||
if (path === null) continue;
|
||||
out.push({
|
||||
role: 'provider',
|
||||
framework: 'go-stdlib',
|
||||
method: 'GET',
|
||||
path,
|
||||
name: handlerNode?.text ?? null,
|
||||
confidence: 0.8,
|
||||
});
|
||||
}
|
||||
|
||||
// net/http client: http.Get/Post/Head
|
||||
for (const match of runCompiledPatterns(HTTP_CLIENT_PATTERNS, tree)) {
|
||||
const fnNode = match.captures.fn;
|
||||
const pathNode = match.captures.path;
|
||||
if (!fnNode || !pathNode) continue;
|
||||
const httpMethod = HTTP_CLIENT_METHOD_TO_HTTP[fnNode.text];
|
||||
if (!httpMethod) continue;
|
||||
const path = unquoteLiteral(pathNode.text);
|
||||
if (path === null) continue;
|
||||
out.push({
|
||||
role: 'consumer',
|
||||
framework: 'go-stdlib',
|
||||
method: httpMethod,
|
||||
path,
|
||||
name: null,
|
||||
confidence: 0.7,
|
||||
});
|
||||
}
|
||||
|
||||
// net/http NewRequest
|
||||
for (const match of runCompiledPatterns(NEW_REQUEST_PATTERNS, tree)) {
|
||||
const methodNode = match.captures.http_method;
|
||||
const pathNode = match.captures.path;
|
||||
if (!methodNode || !pathNode) continue;
|
||||
const method = unquoteLiteral(methodNode.text);
|
||||
const path = unquoteLiteral(pathNode.text);
|
||||
if (method === null || path === null) continue;
|
||||
out.push({
|
||||
role: 'consumer',
|
||||
framework: 'go-stdlib',
|
||||
method: method.toUpperCase(),
|
||||
path,
|
||||
name: null,
|
||||
confidence: 0.7,
|
||||
});
|
||||
}
|
||||
|
||||
// resty
|
||||
for (const match of runCompiledPatterns(RESTY_PATTERNS, tree)) {
|
||||
const methodNode = match.captures.http_method;
|
||||
const pathNode = match.captures.path;
|
||||
if (!methodNode || !pathNode) continue;
|
||||
const path = unquoteLiteral(pathNode.text);
|
||||
if (path === null) continue;
|
||||
out.push({
|
||||
role: 'consumer',
|
||||
framework: 'go-resty',
|
||||
method: methodNode.text.toUpperCase(),
|
||||
path,
|
||||
name: null,
|
||||
confidence: 0.7,
|
||||
});
|
||||
}
|
||||
|
||||
return out;
|
||||
},
|
||||
};
|
||||
@@ -0,0 +1,50 @@
|
||||
import * as path from 'node:path';
|
||||
import type { HttpLanguagePlugin } from './types.js';
|
||||
import { JAVA_HTTP_PLUGIN } from './java.js';
|
||||
import { GO_HTTP_PLUGIN } from './go.js';
|
||||
import { PYTHON_HTTP_PLUGIN } from './python.js';
|
||||
import { PHP_HTTP_PLUGIN } from './php.js';
|
||||
import { JAVASCRIPT_HTTP_PLUGIN, TYPESCRIPT_HTTP_PLUGIN, TSX_HTTP_PLUGIN } from './node.js';
|
||||
|
||||
export type { HttpDetection, HttpLanguagePlugin, HttpRole } from './types.js';
|
||||
|
||||
/**
|
||||
* File-extension → HTTP language plugin registry. The top-level
|
||||
* orchestrator (`http-route-extractor.ts`) looks up the plugin for each
|
||||
* file it visits and delegates the tree-sitter scanning to the plugin.
|
||||
*
|
||||
* Keys are lowercase extensions including the leading dot. To add a
|
||||
* new language, drop a `http-patterns/<lang>.ts` that exports a
|
||||
* `HttpLanguagePlugin`, import it here and register the extension(s).
|
||||
* No edits to `http-route-extractor.ts` are required.
|
||||
*/
|
||||
const REGISTRY: Record<string, HttpLanguagePlugin> = {
|
||||
'.java': JAVA_HTTP_PLUGIN,
|
||||
'.go': GO_HTTP_PLUGIN,
|
||||
'.py': PYTHON_HTTP_PLUGIN,
|
||||
'.php': PHP_HTTP_PLUGIN,
|
||||
'.js': JAVASCRIPT_HTTP_PLUGIN,
|
||||
'.jsx': JAVASCRIPT_HTTP_PLUGIN,
|
||||
'.ts': TYPESCRIPT_HTTP_PLUGIN,
|
||||
'.tsx': TSX_HTTP_PLUGIN,
|
||||
};
|
||||
|
||||
/**
|
||||
* Glob for files worth scanning for HTTP routes. Kept alongside the
|
||||
* registry so adding a new language widens the glob in one edit.
|
||||
*
|
||||
* `.vue` / `.svelte` files are intentionally omitted for the source-scan
|
||||
* path — they need their own grammar-aware extraction and the existing
|
||||
* regex fallback for them was never very accurate. The graph-assisted
|
||||
* Strategy A still handles them via the ingestion pipeline.
|
||||
*/
|
||||
export const HTTP_SCAN_GLOB = '**/*.{ts,tsx,js,jsx,java,go,py,php}';
|
||||
|
||||
/**
|
||||
* Return the HTTP plugin registered for the given file's extension,
|
||||
* or `undefined` if the extension is not registered.
|
||||
*/
|
||||
export function getPluginForFile(rel: string): HttpLanguagePlugin | undefined {
|
||||
const ext = path.extname(rel).toLowerCase();
|
||||
return REGISTRY[ext];
|
||||
}
|
||||
@@ -0,0 +1,267 @@
|
||||
import Parser from 'tree-sitter';
|
||||
import Java from 'tree-sitter-java';
|
||||
import {
|
||||
compilePatterns,
|
||||
runCompiledPatterns,
|
||||
unquoteLiteral,
|
||||
type LanguagePatterns,
|
||||
} from '../tree-sitter-scanner.js';
|
||||
import type { HttpDetection, HttpLanguagePlugin } from './types.js';
|
||||
|
||||
/**
|
||||
* Java HTTP plugin. Handles:
|
||||
* - Spring `@RequestMapping` class prefixes + `@(Get|Post|...)Mapping` method annotations
|
||||
* - Spring `RestTemplate.getForObject/...`, `WebClient.method(HttpMethod.X, ...)`
|
||||
* - OkHttp `new Request.Builder().url("...")`
|
||||
*
|
||||
* The plugin runs two pattern bundles: one to collect class-level
|
||||
* `@RequestMapping` prefixes keyed by the enclosing class node, and a
|
||||
* second to match method-level annotations. The `scan` function walks
|
||||
* up from each matched annotation to find its enclosing class and
|
||||
* combines the prefix with the method path.
|
||||
*/
|
||||
|
||||
const METHOD_ANNOTATION_TO_HTTP: Record<string, string> = {
|
||||
GetMapping: 'GET',
|
||||
PostMapping: 'POST',
|
||||
PutMapping: 'PUT',
|
||||
DeleteMapping: 'DELETE',
|
||||
PatchMapping: 'PATCH',
|
||||
};
|
||||
|
||||
// ─── Provider: Spring class-level @RequestMapping prefix ──────────────
|
||||
const SPRING_CLASS_PREFIX_PATTERNS = compilePatterns({
|
||||
name: 'java-spring-class-prefix',
|
||||
language: Java,
|
||||
patterns: [
|
||||
{
|
||||
meta: {},
|
||||
query: `
|
||||
(class_declaration
|
||||
(modifiers
|
||||
(annotation
|
||||
name: (identifier) @ann (#eq? @ann "RequestMapping")
|
||||
arguments: (annotation_argument_list (string_literal) @prefix)))) @class
|
||||
`,
|
||||
},
|
||||
],
|
||||
} satisfies LanguagePatterns<Record<string, never>>);
|
||||
|
||||
// ─── Provider: Spring @(Get|Post|...)Mapping method annotations ───────
|
||||
const SPRING_METHOD_ROUTE_PATTERNS = compilePatterns({
|
||||
name: 'java-spring-method-route',
|
||||
language: Java,
|
||||
patterns: [
|
||||
{
|
||||
meta: {},
|
||||
query: `
|
||||
(method_declaration
|
||||
(modifiers
|
||||
(annotation
|
||||
name: (identifier) @ann (#match? @ann "^(Get|Post|Put|Delete|Patch)Mapping$")
|
||||
arguments: (annotation_argument_list (string_literal) @path)))
|
||||
name: (identifier) @method_name) @method
|
||||
`,
|
||||
},
|
||||
],
|
||||
} satisfies LanguagePatterns<Record<string, never>>);
|
||||
|
||||
// ─── Consumer: Spring RestTemplate (object-named + method-named) ──────
|
||||
// RestTemplate.getForObject / getForEntity → GET
|
||||
// RestTemplate.postForObject / postForEntity → POST
|
||||
// RestTemplate.put → PUT
|
||||
// RestTemplate.delete → DELETE
|
||||
// RestTemplate.patchForObject → PATCH
|
||||
const REST_TEMPLATE_TO_HTTP: Record<string, string> = {
|
||||
getForObject: 'GET',
|
||||
getForEntity: 'GET',
|
||||
postForObject: 'POST',
|
||||
postForEntity: 'POST',
|
||||
put: 'PUT',
|
||||
delete: 'DELETE',
|
||||
patchForObject: 'PATCH',
|
||||
};
|
||||
|
||||
interface RestTemplateMeta {
|
||||
framework: 'spring-rest-template';
|
||||
}
|
||||
|
||||
const REST_TEMPLATE_PATTERNS = compilePatterns({
|
||||
name: 'java-rest-template',
|
||||
language: Java,
|
||||
patterns: [
|
||||
{
|
||||
meta: { framework: 'spring-rest-template' },
|
||||
query: `
|
||||
(method_invocation
|
||||
object: (identifier) @obj (#eq? @obj "restTemplate")
|
||||
name: (identifier) @method
|
||||
arguments: (argument_list . (string_literal) @path))
|
||||
`,
|
||||
},
|
||||
],
|
||||
} satisfies LanguagePatterns<RestTemplateMeta>);
|
||||
|
||||
// ─── Consumer: Spring WebClient — webClient.method(HttpMethod.X, "path") ─
|
||||
const WEB_CLIENT_PATTERNS = compilePatterns({
|
||||
name: 'java-web-client',
|
||||
language: Java,
|
||||
patterns: [
|
||||
{
|
||||
meta: {},
|
||||
query: `
|
||||
(method_invocation
|
||||
object: (identifier) @obj (#eq? @obj "webClient")
|
||||
name: (identifier) @method (#eq? @method "method")
|
||||
arguments: (argument_list
|
||||
(field_access
|
||||
object: (identifier) @httpMethodCls (#eq? @httpMethodCls "HttpMethod")
|
||||
field: (identifier) @http_method)
|
||||
(string_literal) @path))
|
||||
`,
|
||||
},
|
||||
],
|
||||
} satisfies LanguagePatterns<Record<string, never>>);
|
||||
|
||||
// ─── Consumer: OkHttp `new Request.Builder().url("path")` ─────────────
|
||||
// Note: `Request.Builder` is a `scoped_type_identifier` whose text includes
|
||||
// the dot, so `#eq?` against the literal string matches cleanly (no need
|
||||
// to escape a regex dot).
|
||||
const OK_HTTP_PATTERNS = compilePatterns({
|
||||
name: 'java-okhttp',
|
||||
language: Java,
|
||||
patterns: [
|
||||
{
|
||||
meta: {},
|
||||
query: `
|
||||
(method_invocation
|
||||
object: (object_creation_expression
|
||||
type: (scoped_type_identifier) @type (#eq? @type "Request.Builder"))
|
||||
name: (identifier) @method (#eq? @method "url")
|
||||
arguments: (argument_list . (string_literal) @path))
|
||||
`,
|
||||
},
|
||||
],
|
||||
} satisfies LanguagePatterns<Record<string, never>>);
|
||||
|
||||
/**
|
||||
* Find the nearest enclosing class_declaration ancestor for a node, or
|
||||
* null if the node is top-level. Tree-sitter's SyntaxNode.parent walks
|
||||
* one level at a time.
|
||||
*/
|
||||
function findEnclosingClass(node: Parser.SyntaxNode): Parser.SyntaxNode | null {
|
||||
let cur: Parser.SyntaxNode | null = node.parent;
|
||||
while (cur) {
|
||||
if (cur.type === 'class_declaration') return cur;
|
||||
cur = cur.parent;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/**
|
||||
* Join a class-level prefix and a method-level path into a single URL
|
||||
* path. Mirrors the semantics of the original regex implementation:
|
||||
* strip trailing slashes on the prefix, then ensure a single slash
|
||||
* between prefix and method path.
|
||||
*/
|
||||
function joinPath(prefix: string, methodPath: string): string {
|
||||
const cleanPrefix = prefix.replace(/^\/+/, '').replace(/\/+$/, '');
|
||||
const cleanSub = methodPath.replace(/^\/+/, '');
|
||||
if (!cleanPrefix) return `/${cleanSub}`;
|
||||
return `/${cleanPrefix}/${cleanSub}`;
|
||||
}
|
||||
|
||||
export const JAVA_HTTP_PLUGIN: HttpLanguagePlugin = {
|
||||
name: 'java-http',
|
||||
language: Java,
|
||||
scan(tree) {
|
||||
const out: HttpDetection[] = [];
|
||||
|
||||
// ─── Providers: Spring class prefix + method annotations ────────
|
||||
const prefixByClassId = new Map<number, string>();
|
||||
for (const match of runCompiledPatterns(SPRING_CLASS_PREFIX_PATTERNS, tree)) {
|
||||
const prefixNode = match.captures.prefix;
|
||||
const classNode = match.captures.class;
|
||||
if (!prefixNode || !classNode) continue;
|
||||
const prefix = unquoteLiteral(prefixNode.text);
|
||||
if (prefix !== null) prefixByClassId.set(classNode.id, prefix);
|
||||
}
|
||||
|
||||
for (const match of runCompiledPatterns(SPRING_METHOD_ROUTE_PATTERNS, tree)) {
|
||||
const annNode = match.captures.ann;
|
||||
const pathNode = match.captures.path;
|
||||
const nameNode = match.captures.method_name;
|
||||
const methodNode = match.captures.method;
|
||||
if (!annNode || !pathNode || !methodNode) continue;
|
||||
const httpMethod = METHOD_ANNOTATION_TO_HTTP[annNode.text];
|
||||
if (!httpMethod) continue;
|
||||
const rawPath = unquoteLiteral(pathNode.text);
|
||||
if (rawPath === null) continue;
|
||||
const enclosingClass = findEnclosingClass(methodNode);
|
||||
const prefix = enclosingClass ? (prefixByClassId.get(enclosingClass.id) ?? '') : '';
|
||||
const fullPath = joinPath(prefix, rawPath);
|
||||
out.push({
|
||||
role: 'provider',
|
||||
framework: 'spring',
|
||||
method: httpMethod,
|
||||
path: fullPath,
|
||||
name: nameNode?.text ?? null,
|
||||
confidence: 0.8,
|
||||
});
|
||||
}
|
||||
|
||||
// ─── Consumers: RestTemplate ────────────────────────────────────
|
||||
for (const match of runCompiledPatterns(REST_TEMPLATE_PATTERNS, tree)) {
|
||||
const methodNode = match.captures.method;
|
||||
const pathNode = match.captures.path;
|
||||
if (!methodNode || !pathNode) continue;
|
||||
const httpMethod = REST_TEMPLATE_TO_HTTP[methodNode.text];
|
||||
if (!httpMethod) continue;
|
||||
const path = unquoteLiteral(pathNode.text);
|
||||
if (path === null) continue;
|
||||
out.push({
|
||||
role: 'consumer',
|
||||
framework: 'spring-rest-template',
|
||||
method: httpMethod,
|
||||
path,
|
||||
name: null,
|
||||
confidence: 0.7,
|
||||
});
|
||||
}
|
||||
|
||||
// ─── Consumers: WebClient.method(HttpMethod.X, "path") ──────────
|
||||
for (const match of runCompiledPatterns(WEB_CLIENT_PATTERNS, tree)) {
|
||||
const httpMethodNode = match.captures.http_method;
|
||||
const pathNode = match.captures.path;
|
||||
if (!httpMethodNode || !pathNode) continue;
|
||||
const path = unquoteLiteral(pathNode.text);
|
||||
if (path === null) continue;
|
||||
out.push({
|
||||
role: 'consumer',
|
||||
framework: 'spring-web-client',
|
||||
method: httpMethodNode.text.toUpperCase(),
|
||||
path,
|
||||
name: null,
|
||||
confidence: 0.7,
|
||||
});
|
||||
}
|
||||
|
||||
// ─── Consumers: OkHttp Request.Builder().url("path") ────────────
|
||||
for (const match of runCompiledPatterns(OK_HTTP_PATTERNS, tree)) {
|
||||
const pathNode = match.captures.path;
|
||||
if (!pathNode) continue;
|
||||
const path = unquoteLiteral(pathNode.text);
|
||||
if (path === null) continue;
|
||||
out.push({
|
||||
role: 'consumer',
|
||||
framework: 'okhttp',
|
||||
method: 'GET',
|
||||
path,
|
||||
name: null,
|
||||
confidence: 0.7,
|
||||
});
|
||||
}
|
||||
|
||||
return out;
|
||||
},
|
||||
};
|
||||
@@ -0,0 +1,373 @@
|
||||
import Parser from 'tree-sitter';
|
||||
import JavaScript from 'tree-sitter-javascript';
|
||||
import TypeScript from 'tree-sitter-typescript';
|
||||
import {
|
||||
compilePatterns,
|
||||
runCompiledPatterns,
|
||||
unquoteLiteral,
|
||||
type CompiledPatterns,
|
||||
type LanguagePatterns,
|
||||
type PatternSpec,
|
||||
} from '../tree-sitter-scanner.js';
|
||||
import type { HttpDetection, HttpLanguagePlugin } from './types.js';
|
||||
|
||||
/**
|
||||
* Node.js / TypeScript HTTP plugin family. Handles:
|
||||
* - NestJS `@Controller('prefix')` classes with `@Get(':id')` methods
|
||||
* - Express `router.get(...)` / `app.post(...)` providers
|
||||
* - `fetch(url)` / `fetch(url, { method: 'POST' })` consumers
|
||||
* - `axios.get(url)` / `axios.delete(url)` consumers
|
||||
*
|
||||
* Because the JavaScript and TypeScript tree-sitter grammars share
|
||||
* node type names for every construct we query, pattern sources are
|
||||
* defined once and compiled against each grammar variant. The plugin
|
||||
* exports three `HttpLanguagePlugin`s (JS, TS, TSX) that share the
|
||||
* same `scan` function but bind to different grammars.
|
||||
*/
|
||||
|
||||
// ─── Provider: NestJS — class-level @Controller('prefix') ────────────
|
||||
// In tree-sitter-typescript decorators are NOT children of
|
||||
// class_declaration / method_definition — they're siblings in the
|
||||
// surrounding class_body / program node. We therefore match the
|
||||
// decorator standalone and walk to its related class/method in JS.
|
||||
const NEST_CONTROLLER_SPEC: PatternSpec<Record<string, never>> = {
|
||||
meta: {},
|
||||
query: `
|
||||
(decorator
|
||||
(call_expression
|
||||
function: (identifier) @dec (#eq? @dec "Controller")
|
||||
arguments: (arguments . [(string) (template_string)] @prefix))) @ctrl_decorator
|
||||
`,
|
||||
};
|
||||
|
||||
// ─── Provider: NestJS — method-level @Get/@Post/... decorators ───────
|
||||
// Matches either `@Get('path')` or `@Get()`. The `@path` capture is
|
||||
// optional — when the first argument isn't a string, the plugin falls
|
||||
// back to '/' for the method-level path.
|
||||
const NEST_METHOD_SPEC: PatternSpec<Record<string, never>> = {
|
||||
meta: {},
|
||||
query: `
|
||||
(decorator
|
||||
(call_expression
|
||||
function: (identifier) @dec (#match? @dec "^(Get|Post|Put|Delete|Patch)$")
|
||||
arguments: (arguments) @args)) @method_decorator
|
||||
`,
|
||||
};
|
||||
|
||||
// ─── Provider: Express — router.get/app.post/... ─────────────────────
|
||||
const EXPRESS_SPEC: PatternSpec<Record<string, never>> = {
|
||||
meta: {},
|
||||
query: `
|
||||
(call_expression
|
||||
function: (member_expression
|
||||
object: (identifier) @obj (#match? @obj "^(router|app)$")
|
||||
property: (property_identifier) @http_method (#match? @http_method "^(get|post|put|delete|patch)$"))
|
||||
arguments: (arguments . [(string) (template_string)] @path))
|
||||
`,
|
||||
};
|
||||
|
||||
// ─── Consumer: fetch(url) with NO options ─────────────────────────────
|
||||
const FETCH_NO_OPTIONS_SPEC: PatternSpec<Record<string, never>> = {
|
||||
meta: {},
|
||||
query: `
|
||||
(call_expression
|
||||
function: (identifier) @fn (#eq? @fn "fetch")
|
||||
arguments: (arguments . [(string) (template_string)] @path .))
|
||||
`,
|
||||
};
|
||||
|
||||
// ─── Consumer: fetch(url, { method: 'X', ... }) ──────────────────────
|
||||
const FETCH_WITH_OPTIONS_SPEC: PatternSpec<Record<string, never>> = {
|
||||
meta: {},
|
||||
query: `
|
||||
(call_expression
|
||||
function: (identifier) @fn (#eq? @fn "fetch")
|
||||
arguments: (arguments
|
||||
. [(string) (template_string)] @path
|
||||
(object
|
||||
(pair
|
||||
key: (property_identifier) @key (#eq? @key "method")
|
||||
value: (string) @http_method))))
|
||||
`,
|
||||
};
|
||||
|
||||
// ─── Consumer: axios.get/post/... ────────────────────────────────────
|
||||
const AXIOS_SPEC: PatternSpec<Record<string, never>> = {
|
||||
meta: {},
|
||||
query: `
|
||||
(call_expression
|
||||
function: (member_expression
|
||||
object: (identifier) @obj (#eq? @obj "axios")
|
||||
property: (property_identifier) @http_method (#match? @http_method "^(get|post|put|delete|patch)$"))
|
||||
arguments: (arguments . [(string) (template_string)] @path))
|
||||
`,
|
||||
};
|
||||
|
||||
interface NodePatternBundle {
|
||||
controller: CompiledPatterns<Record<string, never>>;
|
||||
methodDecorator: CompiledPatterns<Record<string, never>>;
|
||||
express: CompiledPatterns<Record<string, never>>;
|
||||
fetchNoOptions: CompiledPatterns<Record<string, never>>;
|
||||
fetchWithOptions: CompiledPatterns<Record<string, never>>;
|
||||
axios: CompiledPatterns<Record<string, never>>;
|
||||
}
|
||||
|
||||
function compileBundle(language: unknown, name: string): NodePatternBundle {
|
||||
const mk = (spec: PatternSpec<Record<string, never>>, suffix: string) =>
|
||||
compilePatterns({
|
||||
name: `${name}-${suffix}`,
|
||||
language,
|
||||
patterns: [spec],
|
||||
} satisfies LanguagePatterns<Record<string, never>>);
|
||||
return {
|
||||
controller: mk(NEST_CONTROLLER_SPEC, 'nest-controller'),
|
||||
methodDecorator: mk(NEST_METHOD_SPEC, 'nest-method-decorator'),
|
||||
express: mk(EXPRESS_SPEC, 'express'),
|
||||
fetchNoOptions: mk(FETCH_NO_OPTIONS_SPEC, 'fetch-no-options'),
|
||||
fetchWithOptions: mk(FETCH_WITH_OPTIONS_SPEC, 'fetch-with-options'),
|
||||
axios: mk(AXIOS_SPEC, 'axios'),
|
||||
};
|
||||
}
|
||||
|
||||
const JAVASCRIPT_BUNDLE = compileBundle(JavaScript, 'javascript-http');
|
||||
const TYPESCRIPT_BUNDLE = compileBundle(TypeScript.typescript, 'typescript-http');
|
||||
const TSX_BUNDLE = compileBundle(TypeScript.tsx, 'tsx-http');
|
||||
|
||||
const NEST_DECORATOR_TO_HTTP: Record<string, string> = {
|
||||
Get: 'GET',
|
||||
Post: 'POST',
|
||||
Put: 'PUT',
|
||||
Delete: 'DELETE',
|
||||
Patch: 'PATCH',
|
||||
};
|
||||
|
||||
/**
|
||||
* Find the nearest enclosing class_declaration for a node, or null.
|
||||
*/
|
||||
function findEnclosingClass(node: Parser.SyntaxNode): Parser.SyntaxNode | null {
|
||||
let cur: Parser.SyntaxNode | null = node.parent;
|
||||
while (cur) {
|
||||
if (cur.type === 'class_declaration') return cur;
|
||||
cur = cur.parent;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
function joinPath(prefix: string, sub: string): string {
|
||||
const cleanPrefix = prefix.replace(/^\/+/, '').replace(/\/+$/, '');
|
||||
const cleanSub = sub.replace(/^\/+/, '');
|
||||
if (!cleanPrefix) return `/${cleanSub}`;
|
||||
return `/${cleanPrefix}/${cleanSub}`;
|
||||
}
|
||||
|
||||
/**
|
||||
* For a standalone `decorator` node (child of class_body / program),
|
||||
* find the related `class_declaration` node that it decorates. In
|
||||
* tree-sitter-typescript the decorator is placed before the class
|
||||
* declaration as a sibling (when decorating a class) or inside the
|
||||
* class_body before a method_definition (when decorating a method);
|
||||
* we walk the parent chain until we find the enclosing class.
|
||||
*/
|
||||
function findDecoratedClass(decoratorNode: Parser.SyntaxNode): Parser.SyntaxNode | null {
|
||||
const parent = decoratorNode.parent;
|
||||
if (!parent) return null;
|
||||
// Case 1: decorator is a sibling of the class_declaration at program /
|
||||
// export_statement level. Walk forward through siblings until we find
|
||||
// the class_declaration this decorator belongs to.
|
||||
for (let i = 0; i < parent.namedChildCount; i++) {
|
||||
const child = parent.namedChild(i);
|
||||
if (child && child.id === decoratorNode.id) {
|
||||
for (let j = i + 1; j < parent.namedChildCount; j++) {
|
||||
const next = parent.namedChild(j);
|
||||
if (!next) continue;
|
||||
if (next.type === 'decorator') continue; // adjacent decorators stack
|
||||
if (next.type === 'class_declaration') return next;
|
||||
if (next.type === 'export_statement') {
|
||||
// `export class Foo { ... }` wraps the declaration.
|
||||
for (let k = 0; k < next.namedChildCount; k++) {
|
||||
const inner = next.namedChild(k);
|
||||
if (inner?.type === 'class_declaration') return inner;
|
||||
}
|
||||
}
|
||||
break;
|
||||
}
|
||||
break;
|
||||
}
|
||||
}
|
||||
// Case 2: decorator is inside a class_body (decorating a method) —
|
||||
// walk up to the enclosing class_declaration.
|
||||
return findEnclosingClass(decoratorNode);
|
||||
}
|
||||
|
||||
/**
|
||||
* For a method-level decorator node (child of class_body before a
|
||||
* method_definition), find the method_definition it decorates.
|
||||
*/
|
||||
function findDecoratedMethod(decoratorNode: Parser.SyntaxNode): Parser.SyntaxNode | null {
|
||||
const parent = decoratorNode.parent;
|
||||
if (!parent || parent.type !== 'class_body') return null;
|
||||
for (let i = 0; i < parent.namedChildCount; i++) {
|
||||
const child = parent.namedChild(i);
|
||||
if (child && child.id === decoratorNode.id) {
|
||||
for (let j = i + 1; j < parent.namedChildCount; j++) {
|
||||
const next = parent.namedChild(j);
|
||||
if (!next) continue;
|
||||
if (next.type === 'decorator') continue;
|
||||
if (next.type === 'method_definition') return next;
|
||||
return null;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
function scanBundle(bundle: NodePatternBundle, tree: Parser.Tree): HttpDetection[] {
|
||||
const out: HttpDetection[] = [];
|
||||
|
||||
// NestJS: collect `@Controller('prefix')` class decorators, keyed by
|
||||
// the `class_declaration` they decorate.
|
||||
const prefixByClassId = new Map<number, string>();
|
||||
for (const match of runCompiledPatterns(bundle.controller, tree)) {
|
||||
const prefixNode = match.captures.prefix;
|
||||
const decoratorNode = match.captures.ctrl_decorator;
|
||||
if (!prefixNode || !decoratorNode) continue;
|
||||
const prefix = unquoteLiteral(prefixNode.text);
|
||||
if (prefix === null) continue;
|
||||
const classNode = findDecoratedClass(decoratorNode);
|
||||
if (!classNode) continue;
|
||||
prefixByClassId.set(classNode.id, prefix);
|
||||
}
|
||||
|
||||
// NestJS: method-level @Get/@Post/... decorators. The decorator's
|
||||
// arguments list may be empty (`@Get()`), a string (`@Get('path')`),
|
||||
// or something else (which we skip).
|
||||
for (const match of runCompiledPatterns(bundle.methodDecorator, tree)) {
|
||||
const decNode = match.captures.dec;
|
||||
const argsNode = match.captures.args;
|
||||
const decoratorNode = match.captures.method_decorator;
|
||||
if (!decNode || !argsNode || !decoratorNode) continue;
|
||||
const httpMethod = NEST_DECORATOR_TO_HTTP[decNode.text];
|
||||
if (!httpMethod) continue;
|
||||
const methodNode = findDecoratedMethod(decoratorNode);
|
||||
if (!methodNode) continue;
|
||||
const enclosingClass = findEnclosingClass(methodNode);
|
||||
// Only emit NestJS detections when the class actually has a
|
||||
// @Controller decorator — without it, the match is almost certainly
|
||||
// something else (e.g. an unrelated library using similar names).
|
||||
if (!enclosingClass || !prefixByClassId.has(enclosingClass.id)) continue;
|
||||
const prefix = prefixByClassId.get(enclosingClass.id) ?? '';
|
||||
|
||||
let rawPath = '/';
|
||||
const firstArg = argsNode.namedChild(0);
|
||||
if (firstArg && (firstArg.type === 'string' || firstArg.type === 'template_string')) {
|
||||
const unquoted = unquoteLiteral(firstArg.text);
|
||||
if (unquoted !== null) rawPath = unquoted;
|
||||
}
|
||||
|
||||
// Get the method name from the decorated method_definition.
|
||||
const methodNameNode = methodNode.childForFieldName('name');
|
||||
const name = methodNameNode?.text ?? null;
|
||||
|
||||
out.push({
|
||||
role: 'provider',
|
||||
framework: 'nest',
|
||||
method: httpMethod,
|
||||
path: joinPath(prefix, rawPath),
|
||||
name,
|
||||
confidence: 0.8,
|
||||
});
|
||||
}
|
||||
|
||||
// Express: router/app.<verb>(...)
|
||||
for (const match of runCompiledPatterns(bundle.express, tree)) {
|
||||
const methodNode = match.captures.http_method;
|
||||
const pathNode = match.captures.path;
|
||||
if (!methodNode || !pathNode) continue;
|
||||
const path = unquoteLiteral(pathNode.text);
|
||||
if (path === null) continue;
|
||||
out.push({
|
||||
role: 'provider',
|
||||
framework: 'express',
|
||||
method: methodNode.text.toUpperCase(),
|
||||
path,
|
||||
name: 'handler',
|
||||
confidence: 0.8,
|
||||
});
|
||||
}
|
||||
|
||||
// Consumer: fetch with options { method: 'X' }
|
||||
const fetchSeen = new Set<number>();
|
||||
for (const match of runCompiledPatterns(bundle.fetchWithOptions, tree)) {
|
||||
const pathNode = match.captures.path;
|
||||
const methodNode = match.captures.http_method;
|
||||
if (!pathNode || !methodNode) continue;
|
||||
const path = unquoteLiteral(pathNode.text);
|
||||
const method = unquoteLiteral(methodNode.text);
|
||||
if (path === null || method === null) continue;
|
||||
fetchSeen.add(pathNode.id);
|
||||
out.push({
|
||||
role: 'consumer',
|
||||
framework: 'fetch',
|
||||
method: method.toUpperCase(),
|
||||
path,
|
||||
name: null,
|
||||
confidence: 0.7,
|
||||
});
|
||||
}
|
||||
|
||||
// Consumer: plain fetch(path) — default GET. Skip path nodes we already
|
||||
// matched with the options variant so we don't double-emit.
|
||||
for (const match of runCompiledPatterns(bundle.fetchNoOptions, tree)) {
|
||||
const pathNode = match.captures.path;
|
||||
if (!pathNode) continue;
|
||||
if (fetchSeen.has(pathNode.id)) continue;
|
||||
const path = unquoteLiteral(pathNode.text);
|
||||
if (path === null) continue;
|
||||
out.push({
|
||||
role: 'consumer',
|
||||
framework: 'fetch',
|
||||
method: 'GET',
|
||||
path,
|
||||
name: null,
|
||||
confidence: 0.7,
|
||||
});
|
||||
}
|
||||
|
||||
// Consumer: axios.<verb>(url)
|
||||
for (const match of runCompiledPatterns(bundle.axios, tree)) {
|
||||
const methodNode = match.captures.http_method;
|
||||
const pathNode = match.captures.path;
|
||||
if (!methodNode || !pathNode) continue;
|
||||
const path = unquoteLiteral(pathNode.text);
|
||||
if (path === null) continue;
|
||||
out.push({
|
||||
role: 'consumer',
|
||||
framework: 'axios',
|
||||
method: methodNode.text.toUpperCase(),
|
||||
path,
|
||||
name: null,
|
||||
confidence: 0.7,
|
||||
});
|
||||
}
|
||||
|
||||
return out;
|
||||
}
|
||||
|
||||
export const JAVASCRIPT_HTTP_PLUGIN: HttpLanguagePlugin = {
|
||||
name: 'javascript-http',
|
||||
language: JavaScript,
|
||||
scan: (tree) => scanBundle(JAVASCRIPT_BUNDLE, tree),
|
||||
};
|
||||
|
||||
export const TYPESCRIPT_HTTP_PLUGIN: HttpLanguagePlugin = {
|
||||
name: 'typescript-http',
|
||||
language: TypeScript.typescript,
|
||||
scan: (tree) => scanBundle(TYPESCRIPT_BUNDLE, tree),
|
||||
};
|
||||
|
||||
export const TSX_HTTP_PLUGIN: HttpLanguagePlugin = {
|
||||
name: 'tsx-http',
|
||||
language: TypeScript.tsx,
|
||||
scan: (tree) => scanBundle(TSX_BUNDLE, tree),
|
||||
};
|
||||
@@ -0,0 +1,79 @@
|
||||
import PHP from 'tree-sitter-php';
|
||||
import {
|
||||
compilePatterns,
|
||||
runCompiledPatterns,
|
||||
unquoteLiteral,
|
||||
type LanguagePatterns,
|
||||
} from '../tree-sitter-scanner.js';
|
||||
import type { HttpDetection, HttpLanguagePlugin } from './types.js';
|
||||
|
||||
/**
|
||||
* PHP HTTP plugin — Laravel `Route::get/post/...` declarations.
|
||||
*
|
||||
* The pipeline already uses `PHP.php_only` for ingesting plain `.php`
|
||||
* files (see `core/tree-sitter/parser-loader.ts`), and we do the same
|
||||
* here so Laravel route files are parsed with the right grammar dialect.
|
||||
*/
|
||||
|
||||
const LARAVEL_PATTERNS = compilePatterns({
|
||||
name: 'php-laravel',
|
||||
language: PHP.php_only,
|
||||
patterns: [
|
||||
{
|
||||
meta: {},
|
||||
query: `
|
||||
(scoped_call_expression
|
||||
scope: (name) @scope (#eq? @scope "Route")
|
||||
name: (name) @method (#match? @method "^(get|post|put|delete|patch)$")
|
||||
arguments: (arguments . (argument (string) @path)))
|
||||
`,
|
||||
},
|
||||
],
|
||||
} satisfies LanguagePatterns<Record<string, never>>);
|
||||
|
||||
/**
|
||||
* Extract the inner text of a PHP `string` node. The tree-sitter-php
|
||||
* grammar wraps single / double-quoted literals differently depending
|
||||
* on content; we try both the raw `text` (with quotes) through
|
||||
* `unquoteLiteral`, and a fallback via the `string_value` / `string_content`
|
||||
* child nodes.
|
||||
*/
|
||||
function phpStringText(node: import('tree-sitter').SyntaxNode): string | null {
|
||||
// Most single-quoted strings expose their inner content through the
|
||||
// full node text (including quotes), which unquoteLiteral strips.
|
||||
const direct = unquoteLiteral(node.text);
|
||||
if (direct !== null && direct !== node.text) return direct;
|
||||
// Fall back to child string_content / string_value node if present.
|
||||
for (const child of node.children) {
|
||||
if (child.type === 'string_content' || child.type === 'string_value') {
|
||||
return child.text;
|
||||
}
|
||||
}
|
||||
return direct;
|
||||
}
|
||||
|
||||
export const PHP_HTTP_PLUGIN: HttpLanguagePlugin = {
|
||||
name: 'php-http',
|
||||
language: PHP.php_only,
|
||||
scan(tree) {
|
||||
const out: HttpDetection[] = [];
|
||||
|
||||
for (const match of runCompiledPatterns(LARAVEL_PATTERNS, tree)) {
|
||||
const methodNode = match.captures.method;
|
||||
const pathNode = match.captures.path;
|
||||
if (!methodNode || !pathNode) continue;
|
||||
const path = phpStringText(pathNode);
|
||||
if (path === null) continue;
|
||||
out.push({
|
||||
role: 'provider',
|
||||
framework: 'laravel',
|
||||
method: methodNode.text.toUpperCase(),
|
||||
path,
|
||||
name: 'route',
|
||||
confidence: 0.8,
|
||||
});
|
||||
}
|
||||
|
||||
return out;
|
||||
},
|
||||
};
|
||||
@@ -0,0 +1,142 @@
|
||||
import Python from 'tree-sitter-python';
|
||||
import {
|
||||
compilePatterns,
|
||||
runCompiledPatterns,
|
||||
unquoteLiteral,
|
||||
type LanguagePatterns,
|
||||
} from '../tree-sitter-scanner.js';
|
||||
import type { HttpDetection, HttpLanguagePlugin } from './types.js';
|
||||
|
||||
/**
|
||||
* Python HTTP plugin. Handles:
|
||||
* - FastAPI `@app.get("/path")` provider decorators
|
||||
* - `requests.get/post/...("url")` consumer calls
|
||||
* - Generic `requests.request("METHOD", "url")` consumer calls
|
||||
*/
|
||||
|
||||
const FASTAPI_VERBS: Record<string, string> = {
|
||||
get: 'GET',
|
||||
post: 'POST',
|
||||
put: 'PUT',
|
||||
delete: 'DELETE',
|
||||
patch: 'PATCH',
|
||||
};
|
||||
|
||||
// ─── Provider: FastAPI @app.get/... ──────────────────────────────────
|
||||
const FASTAPI_PATTERNS = compilePatterns({
|
||||
name: 'python-fastapi',
|
||||
language: Python,
|
||||
patterns: [
|
||||
{
|
||||
meta: {},
|
||||
query: `
|
||||
(decorator
|
||||
(call
|
||||
function: (attribute
|
||||
object: (identifier) @obj (#eq? @obj "app")
|
||||
attribute: (identifier) @method (#match? @method "^(get|post|put|delete|patch)$"))
|
||||
arguments: (argument_list . (string) @path)))
|
||||
`,
|
||||
},
|
||||
],
|
||||
} satisfies LanguagePatterns<Record<string, never>>);
|
||||
|
||||
// ─── Consumer: requests.get/post/... ──────────────────────────────────
|
||||
const REQUESTS_VERB_PATTERNS = compilePatterns({
|
||||
name: 'python-requests-verb',
|
||||
language: Python,
|
||||
patterns: [
|
||||
{
|
||||
meta: {},
|
||||
query: `
|
||||
(call
|
||||
function: (attribute
|
||||
object: (identifier) @obj (#eq? @obj "requests")
|
||||
attribute: (identifier) @method (#match? @method "^(get|post|put|delete|patch)$"))
|
||||
arguments: (argument_list . (string) @path))
|
||||
`,
|
||||
},
|
||||
],
|
||||
} satisfies LanguagePatterns<Record<string, never>>);
|
||||
|
||||
// ─── Consumer: requests.request("METHOD", "url") ─────────────────────
|
||||
const REQUESTS_GENERIC_PATTERNS = compilePatterns({
|
||||
name: 'python-requests-generic',
|
||||
language: Python,
|
||||
patterns: [
|
||||
{
|
||||
meta: {},
|
||||
query: `
|
||||
(call
|
||||
function: (attribute
|
||||
object: (identifier) @obj (#eq? @obj "requests")
|
||||
attribute: (identifier) @method (#eq? @method "request"))
|
||||
arguments: (argument_list . (string) @http_method (string) @path))
|
||||
`,
|
||||
},
|
||||
],
|
||||
} satisfies LanguagePatterns<Record<string, never>>);
|
||||
|
||||
export const PYTHON_HTTP_PLUGIN: HttpLanguagePlugin = {
|
||||
name: 'python-http',
|
||||
language: Python,
|
||||
scan(tree) {
|
||||
const out: HttpDetection[] = [];
|
||||
|
||||
// Providers: FastAPI
|
||||
for (const match of runCompiledPatterns(FASTAPI_PATTERNS, tree)) {
|
||||
const methodNode = match.captures.method;
|
||||
const pathNode = match.captures.path;
|
||||
if (!methodNode || !pathNode) continue;
|
||||
const httpMethod = FASTAPI_VERBS[methodNode.text];
|
||||
if (!httpMethod) continue;
|
||||
const path = unquoteLiteral(pathNode.text);
|
||||
if (path === null) continue;
|
||||
out.push({
|
||||
role: 'provider',
|
||||
framework: 'fastapi',
|
||||
method: httpMethod,
|
||||
path,
|
||||
name: null,
|
||||
confidence: 0.8,
|
||||
});
|
||||
}
|
||||
|
||||
// Consumers: requests.<verb>
|
||||
for (const match of runCompiledPatterns(REQUESTS_VERB_PATTERNS, tree)) {
|
||||
const methodNode = match.captures.method;
|
||||
const pathNode = match.captures.path;
|
||||
if (!methodNode || !pathNode) continue;
|
||||
const path = unquoteLiteral(pathNode.text);
|
||||
if (path === null) continue;
|
||||
out.push({
|
||||
role: 'consumer',
|
||||
framework: 'python-requests',
|
||||
method: methodNode.text.toUpperCase(),
|
||||
path,
|
||||
name: null,
|
||||
confidence: 0.7,
|
||||
});
|
||||
}
|
||||
|
||||
// Consumers: requests.request("METHOD", "url")
|
||||
for (const match of runCompiledPatterns(REQUESTS_GENERIC_PATTERNS, tree)) {
|
||||
const methodNode = match.captures.http_method;
|
||||
const pathNode = match.captures.path;
|
||||
if (!methodNode || !pathNode) continue;
|
||||
const methodRaw = unquoteLiteral(methodNode.text);
|
||||
const path = unquoteLiteral(pathNode.text);
|
||||
if (methodRaw === null || path === null) continue;
|
||||
out.push({
|
||||
role: 'consumer',
|
||||
framework: 'python-requests',
|
||||
method: methodRaw.toUpperCase(),
|
||||
path,
|
||||
name: null,
|
||||
confidence: 0.7,
|
||||
});
|
||||
}
|
||||
|
||||
return out;
|
||||
},
|
||||
};
|
||||
@@ -0,0 +1,65 @@
|
||||
import type Parser from 'tree-sitter';
|
||||
|
||||
/**
|
||||
* Shared types for the http-route-extractor language plugins.
|
||||
*
|
||||
* Each plugin lives in its own file (java.ts, node.ts, ...) and owns
|
||||
* the tree-sitter grammar import + queries. The top-level
|
||||
* `http-route-extractor.ts` orchestrator only knows about this type
|
||||
* module and the plugin registry (`./index.ts`). It MUST NOT import
|
||||
* any grammar or query text directly — language-specific knowledge
|
||||
* belongs in the plugins.
|
||||
*/
|
||||
|
||||
export type HttpRole = 'provider' | 'consumer';
|
||||
|
||||
/**
|
||||
* One raw HTTP detection produced by a plugin's `scan()` function. The
|
||||
* orchestrator converts this into a full `ExtractedContract` by running
|
||||
* path normalization and building the contract id.
|
||||
*
|
||||
* `path` is the raw literal string as it appeared in source (with
|
||||
* `${...}` template placeholders still in place); the orchestrator
|
||||
* runs the appropriate normalizer for provider vs. consumer paths.
|
||||
*/
|
||||
export interface HttpDetection {
|
||||
role: HttpRole;
|
||||
/** Short framework label, e.g. `'spring'`, `'nest'`, `'express'`. */
|
||||
framework: string;
|
||||
/** HTTP method in upper case (`'GET'`, `'POST'`, ...). */
|
||||
method: string;
|
||||
/** Raw path literal as seen in source (template placeholders intact). */
|
||||
path: string;
|
||||
/**
|
||||
* Symbol name of the handler (for providers) or calling function
|
||||
* (for consumers) when the plugin can determine it structurally.
|
||||
* Null when no good candidate is available.
|
||||
*/
|
||||
name: string | null;
|
||||
/** Confidence in (0, 1]. Source-scan plugins typically use 0.7–0.8. */
|
||||
confidence: number;
|
||||
}
|
||||
|
||||
/**
|
||||
* One language-scoped HTTP plugin. The plugin owns the tree-sitter
|
||||
* grammar and the `scan` function that translates a parsed tree into
|
||||
* zero or more `HttpDetection`s. Plugins are free to run multiple
|
||||
* compiled pattern bundles internally (see the shared scanner's
|
||||
* `runCompiledPatterns` helper).
|
||||
*
|
||||
* `language` is typed as `unknown` for the same reason as
|
||||
* `LanguagePatterns.language` in `tree-sitter-scanner.ts` — the
|
||||
* grammar modules export different shapes.
|
||||
*/
|
||||
export interface HttpLanguagePlugin {
|
||||
/** Human-readable plugin name for diagnostics. */
|
||||
name: string;
|
||||
/** tree-sitter grammar object (passed to the shared parser). */
|
||||
language: unknown;
|
||||
/**
|
||||
* Scan a parsed tree and return zero or more HTTP detections. Plugins
|
||||
* must not throw — they should swallow per-match errors so a single
|
||||
* malformed construct does not abort the whole file.
|
||||
*/
|
||||
scan(tree: Parser.Tree): HttpDetection[];
|
||||
}
|
||||
@@ -0,0 +1,467 @@
|
||||
import * as path from 'node:path';
|
||||
import { glob } from 'glob';
|
||||
import Parser from 'tree-sitter';
|
||||
import type { ContractExtractor, CypherExecutor } from '../contract-extractor.js';
|
||||
import type { ExtractedContract, RepoHandle } from '../types.js';
|
||||
import { readSafe } from './fs-utils.js';
|
||||
import { getPluginForFile, HTTP_SCAN_GLOB, type HttpDetection } from './http-patterns/index.js';
|
||||
|
||||
/**
|
||||
* Language-agnostic orchestrator for HTTP route (provider + consumer)
|
||||
* contract extraction. Two strategies, in order of preference per role:
|
||||
*
|
||||
* 1. **Graph-assisted (Strategy A)** — if a per-repo LadybugDB executor
|
||||
* is available, read `HANDLES_ROUTE` / `FETCHES` Cypher edges that
|
||||
* the ingestion pipeline already produced via tree-sitter. This is
|
||||
* the preferred path because the graph has richer symbol metadata
|
||||
* (real uids, class/method structure, etc.).
|
||||
*
|
||||
* 2. **Source-scan fallback (Strategy B)** — parse files directly with
|
||||
* the per-language plugin registry in `./http-patterns/`. Used when
|
||||
* the graph has no routes/fetches for this repo (e.g. a repo that
|
||||
* hasn't been indexed yet, or whose indexer doesn't know the
|
||||
* framework). Each plugin owns its tree-sitter grammar and query
|
||||
* sources — this orchestrator imports NO grammars or query strings.
|
||||
*
|
||||
* Adding a new language for Strategy B is a one-file edit in
|
||||
* `http-patterns/index.ts`: register a new `HttpLanguagePlugin` and
|
||||
* widen `HTTP_SCAN_GLOB` if needed.
|
||||
*/
|
||||
|
||||
// ─── Graph-assisted queries ──────────────────────────────────────────
|
||||
|
||||
const HANDLES_ROUTE_QUERY = `
|
||||
MATCH (handlerFile:File)-[r:CodeRelation {type: 'HANDLES_ROUTE'}]->(route:Route)
|
||||
RETURN handlerFile.id AS fileId, handlerFile.filePath AS filePath,
|
||||
route.name AS routePath, route.id AS routeId,
|
||||
route.responseKeys AS responseKeys,
|
||||
r.reason AS routeSource`;
|
||||
|
||||
const FETCHES_QUERY = `
|
||||
MATCH (callerFile:File)-[r:CodeRelation {type: 'FETCHES'}]->(route:Route)
|
||||
RETURN callerFile.id AS fileId, callerFile.filePath AS filePath,
|
||||
route.name AS routePath, route.id AS routeId,
|
||||
r.reason AS fetchReason`;
|
||||
|
||||
const CONTAINS_QUERY = `
|
||||
MATCH (file:File {id: $fileId})<-[:CodeRelation {type: 'CONTAINS'}]-(sym)
|
||||
WHERE sym.startLine IS NOT NULL
|
||||
RETURN sym.id AS uid, sym.name AS name, sym.filePath AS filePath, labels(sym) AS labels
|
||||
ORDER BY sym.startLine`;
|
||||
|
||||
// ─── Path normalization (shared between provider / consumer paths) ──
|
||||
|
||||
/**
|
||||
* Canonicalize a provider-side HTTP path for contract-id generation:
|
||||
* - strip query string
|
||||
* - lower-case
|
||||
* - drop trailing slash
|
||||
* - collapse `:id`, `{id}`, `[id]` path params into a single `{param}`
|
||||
*/
|
||||
export function normalizeHttpPath(p: string): string {
|
||||
let s = p.trim().split('?')[0].toLowerCase().replace(/\/+$/, '');
|
||||
s = s.replace(/:\w+/g, '{param}');
|
||||
s = s.replace(/\{[^}]+\}/g, '{param}');
|
||||
s = s.replace(/\[[^\]]+\]/g, '{param}');
|
||||
// Preserve root: after stripping trailing slashes, the root "/"
|
||||
// collapses to "" which would produce malformed contract ids like
|
||||
// `http::GET::`. Restore a single slash for the root case.
|
||||
return s === '' ? '/' : s;
|
||||
}
|
||||
|
||||
/**
|
||||
* Consumer-side normalization is more aggressive:
|
||||
* - template literals (`${x}`) → `{param}`
|
||||
* - strip protocol + host if the URL is absolute
|
||||
* - numeric segments → `{param}` (so `/api/orders/42` → `/api/orders/{param}`)
|
||||
*/
|
||||
function normalizeConsumerPath(url: string): string {
|
||||
const templated = url.replace(/\$\{[^}]+\}/g, '{param}').trim();
|
||||
let pathOnly = templated;
|
||||
if (/^https?:\/\//i.test(templated)) {
|
||||
try {
|
||||
pathOnly = new URL(templated).pathname;
|
||||
} catch {
|
||||
pathOnly = templated.replace(/^https?:\/\/[^/]+/i, '');
|
||||
}
|
||||
}
|
||||
const normalized = normalizeHttpPath(pathOnly || '/');
|
||||
const segments = normalized
|
||||
.split('/')
|
||||
.filter(Boolean)
|
||||
.map((segment) => (/^\d+$/.test(segment) ? '{param}' : segment));
|
||||
return `/${segments.join('/')}`.replace(/\/+$/, '') || '/';
|
||||
}
|
||||
|
||||
function contractIdFor(method: string, pathNorm: string): string {
|
||||
return `http::${method.toUpperCase()}::${pathNorm}`;
|
||||
}
|
||||
|
||||
// ─── Graph row helpers ───────────────────────────────────────────────
|
||||
|
||||
function methodFromRouteReason(reason: string): string | null {
|
||||
const r = reason || '';
|
||||
if (/GetMapping|decorator-Get/i.test(r)) return 'GET';
|
||||
if (/PostMapping|decorator-Post/i.test(r)) return 'POST';
|
||||
if (/PutMapping|decorator-Put/i.test(r)) return 'PUT';
|
||||
if (/DeleteMapping|decorator-Delete/i.test(r)) return 'DELETE';
|
||||
if (/PatchMapping|decorator-Patch/i.test(r)) return 'PATCH';
|
||||
return null;
|
||||
}
|
||||
|
||||
function pickSymbolUid(
|
||||
rows: Record<string, unknown>[],
|
||||
preferredName: string | null,
|
||||
): { uid: string; name: string; filePath: string } {
|
||||
const norm = (x: unknown) => String(x ?? '');
|
||||
const labeled = rows.filter((r) => {
|
||||
const labels = r.labels ?? r[3];
|
||||
const s = JSON.stringify(labels);
|
||||
return s.includes('Method') || s.includes('Function');
|
||||
});
|
||||
const pool = labeled.length > 0 ? labeled : rows;
|
||||
if (preferredName) {
|
||||
const hit = pool.find((r) => norm(r.name ?? r[1]) === preferredName);
|
||||
if (hit) {
|
||||
return {
|
||||
uid: norm(hit.uid ?? hit[0]),
|
||||
name: norm(hit.name ?? hit[1]),
|
||||
filePath: norm(hit.filePath ?? hit[2]),
|
||||
};
|
||||
}
|
||||
}
|
||||
const first = pool[0] || rows[0];
|
||||
return {
|
||||
uid: norm(first?.uid ?? first?.[0]),
|
||||
name: norm(first?.name ?? first?.[1]),
|
||||
filePath: norm(first?.filePath ?? first?.[2]),
|
||||
};
|
||||
}
|
||||
|
||||
// ─── Orchestrator ────────────────────────────────────────────────────
|
||||
|
||||
export class HttpRouteExtractor implements ContractExtractor {
|
||||
type = 'http' as const;
|
||||
|
||||
async canExtract(_repo: RepoHandle): Promise<boolean> {
|
||||
return true;
|
||||
}
|
||||
|
||||
async extract(
|
||||
dbExecutor: CypherExecutor | null,
|
||||
repoPath: string,
|
||||
_repo: RepoHandle,
|
||||
): Promise<ExtractedContract[]> {
|
||||
// Parse each file at most once and reuse the plugin results across
|
||||
// both graph-assisted enrichment and source-scan emission.
|
||||
const parser = new Parser();
|
||||
const cachedDetections = new Map<string, HttpDetection[]>();
|
||||
const getDetections = (rel: string): HttpDetection[] => {
|
||||
const cached = cachedDetections.get(rel);
|
||||
if (cached) return cached;
|
||||
const plugin = getPluginForFile(rel);
|
||||
if (!plugin) {
|
||||
cachedDetections.set(rel, []);
|
||||
return [];
|
||||
}
|
||||
const content = readSafe(repoPath, rel);
|
||||
if (!content) {
|
||||
cachedDetections.set(rel, []);
|
||||
return [];
|
||||
}
|
||||
try {
|
||||
parser.setLanguage(plugin.language);
|
||||
const tree = parser.parse(content);
|
||||
const detections = plugin.scan(tree);
|
||||
cachedDetections.set(rel, detections);
|
||||
return detections;
|
||||
} catch {
|
||||
cachedDetections.set(rel, []);
|
||||
return [];
|
||||
}
|
||||
};
|
||||
|
||||
// Glob the source-scan file list at most once per extract() —
|
||||
// both provider and consumer fallback paths share the same list.
|
||||
let scannedFiles: string[] | null = null;
|
||||
const getScannedFiles = async (): Promise<string[]> => {
|
||||
if (scannedFiles) return scannedFiles;
|
||||
scannedFiles = await this.scanFiles(repoPath);
|
||||
return scannedFiles;
|
||||
};
|
||||
|
||||
const graphProviders =
|
||||
dbExecutor != null ? await this.extractProvidersGraph(dbExecutor, getDetections) : [];
|
||||
const providers =
|
||||
graphProviders.length > 0
|
||||
? graphProviders
|
||||
: this.extractProvidersSourceScan(await getScannedFiles(), getDetections);
|
||||
|
||||
const graphConsumers =
|
||||
dbExecutor != null ? await this.extractConsumersGraph(dbExecutor, getDetections) : [];
|
||||
const consumers =
|
||||
graphConsumers.length > 0
|
||||
? graphConsumers
|
||||
: this.extractConsumersSourceScan(await getScannedFiles(), getDetections);
|
||||
|
||||
return [...providers, ...consumers];
|
||||
}
|
||||
|
||||
private async scanFiles(repoPath: string): Promise<string[]> {
|
||||
return glob(HTTP_SCAN_GLOB, {
|
||||
cwd: repoPath,
|
||||
ignore: ['**/node_modules/**', '**/.git/**', '**/dist/**', '**/build/**', '**/vendor/**'],
|
||||
nodir: true,
|
||||
});
|
||||
}
|
||||
|
||||
// ─── Graph-assisted providers ──────────────────────────────────────
|
||||
|
||||
private async extractProvidersGraph(
|
||||
db: CypherExecutor,
|
||||
getDetections: (rel: string) => HttpDetection[],
|
||||
): Promise<ExtractedContract[]> {
|
||||
const out: ExtractedContract[] = [];
|
||||
let rows: Record<string, unknown>[];
|
||||
try {
|
||||
rows = await db(HANDLES_ROUTE_QUERY);
|
||||
} catch {
|
||||
return [];
|
||||
}
|
||||
|
||||
for (const row of rows) {
|
||||
const filePath = String(row.filePath ?? '');
|
||||
const routePath = String(row.routePath ?? '');
|
||||
const routeSource = String(row.routeSource ?? row.routeReason ?? '');
|
||||
let method = methodFromRouteReason(routeSource);
|
||||
|
||||
// Look up handler name (and backfill method if missing) from the
|
||||
// plugin's scan of the handler file. This replaces the old
|
||||
// regex-based `inferMethodFromFileScan` and `pickJavaHandlerName`
|
||||
// helpers — tree-sitter gives both pieces of information
|
||||
// structurally. Always run the lookup: even when method is set by
|
||||
// `methodFromRouteReason`, we still need the handler name.
|
||||
const detections = filePath ? getDetections(filePath) : [];
|
||||
const providerDetections = detections.filter((d) => d.role === 'provider');
|
||||
let handlerName: string | null = null;
|
||||
const normalizedRoute = normalizeHttpPath(routePath);
|
||||
// Candidates share the same normalized path. When multiple
|
||||
// detections at the same path exist (e.g. GET + POST /api/orders
|
||||
// in one router), a blind `.find()` silently returned the first
|
||||
// verb — attaching the wrong handler and, when method was not
|
||||
// already pinned by the route reason, the wrong method too.
|
||||
// Disambiguate by method when we know it; refuse to guess when
|
||||
// we don't.
|
||||
const candidates = providerDetections.filter(
|
||||
(d) => normalizeHttpPath(d.path) === normalizedRoute,
|
||||
);
|
||||
let match: (typeof candidates)[number] | undefined;
|
||||
const ambiguousCandidates = !method && candidates.length > 1;
|
||||
if (method) {
|
||||
match = candidates.find((d) => d.method === method);
|
||||
} else if (candidates.length === 1) {
|
||||
match = candidates[0];
|
||||
}
|
||||
// else: multiple candidates + unknown method → leave match
|
||||
// undefined so handlerName stays null and skip symbol
|
||||
// enrichment below, keeping the file-basename fallback instead
|
||||
// of letting pickSymbolUid silently pick the first Function /
|
||||
// Method in the file (which reintroduces the mis-attribution
|
||||
// we were trying to avoid). Method stays at the conservative
|
||||
// 'GET' default set below.
|
||||
if (match) {
|
||||
if (!method) method = match.method;
|
||||
handlerName = match.name;
|
||||
}
|
||||
if (!method) method = 'GET';
|
||||
|
||||
const pathNorm = normalizeHttpPath(routePath);
|
||||
const cid = contractIdFor(method, pathNorm);
|
||||
|
||||
let symbolUid = '';
|
||||
let symbolName = path.basename(filePath) || 'handler';
|
||||
let symPath = filePath;
|
||||
const fileId = row.fileId ?? row[0];
|
||||
if (fileId && !ambiguousCandidates) {
|
||||
try {
|
||||
const syms = await db(CONTAINS_QUERY, { fileId });
|
||||
if (syms.length > 0) {
|
||||
const picked = pickSymbolUid(syms, handlerName);
|
||||
symbolUid = picked.uid;
|
||||
symbolName = picked.name;
|
||||
symPath = picked.filePath || filePath;
|
||||
}
|
||||
} catch {
|
||||
/* ignore */
|
||||
}
|
||||
}
|
||||
|
||||
out.push({
|
||||
contractId: cid,
|
||||
type: 'http',
|
||||
role: 'provider',
|
||||
symbolUid,
|
||||
symbolRef: { filePath: symPath, name: symbolName },
|
||||
symbolName,
|
||||
confidence: 0.9,
|
||||
meta: {
|
||||
method,
|
||||
path: pathNorm,
|
||||
pathSegments: pathNorm.split('/').filter(Boolean),
|
||||
extractionStrategy: 'graph_assisted',
|
||||
routeSource,
|
||||
},
|
||||
});
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
// ─── Source-scan providers ─────────────────────────────────────────
|
||||
|
||||
private extractProvidersSourceScan(
|
||||
files: string[],
|
||||
getDetections: (rel: string) => HttpDetection[],
|
||||
): ExtractedContract[] {
|
||||
const out: ExtractedContract[] = [];
|
||||
for (const rel of files) {
|
||||
const detections = getDetections(rel);
|
||||
for (const d of detections) {
|
||||
if (d.role !== 'provider') continue;
|
||||
const pathNorm = normalizeHttpPath(d.path);
|
||||
out.push({
|
||||
contractId: contractIdFor(d.method, pathNorm),
|
||||
type: 'http',
|
||||
role: 'provider',
|
||||
symbolUid: '',
|
||||
symbolRef: { filePath: rel, name: d.name ?? 'handler' },
|
||||
symbolName: d.name ?? 'handler',
|
||||
confidence: d.confidence,
|
||||
meta: {
|
||||
method: d.method,
|
||||
path: pathNorm,
|
||||
pathSegments: pathNorm.split('/').filter(Boolean),
|
||||
extractionStrategy: 'source_scan',
|
||||
framework: d.framework,
|
||||
},
|
||||
});
|
||||
}
|
||||
}
|
||||
return this.dedupeContracts(out);
|
||||
}
|
||||
|
||||
// ─── Graph-assisted consumers ──────────────────────────────────────
|
||||
|
||||
private async extractConsumersGraph(
|
||||
db: CypherExecutor,
|
||||
getDetections: (rel: string) => HttpDetection[],
|
||||
): Promise<ExtractedContract[]> {
|
||||
const out: ExtractedContract[] = [];
|
||||
let rows: Record<string, unknown>[];
|
||||
try {
|
||||
rows = await db(FETCHES_QUERY);
|
||||
} catch {
|
||||
return [];
|
||||
}
|
||||
for (const row of rows) {
|
||||
const filePath = String(row.filePath ?? '');
|
||||
const routePath = String(row.routePath ?? '');
|
||||
const pathNorm = normalizeHttpPath(routePath);
|
||||
let method = 'GET';
|
||||
// Prefer the plugin's detected method if we can find a matching
|
||||
// fetch/axios call in the same file.
|
||||
const detections = filePath ? getDetections(filePath) : [];
|
||||
// Symmetric to the provider path: if multiple consumer calls in
|
||||
// the same file share the same normalized path (e.g. a GET
|
||||
// fetch AND a POST fetch to `/api/orders`), `.find()` silently
|
||||
// picked the first verb and keyed the contract id on the wrong
|
||||
// method. With no upstream method signal here, refuse to guess
|
||||
// when candidates are ambiguous — leave `method` at its
|
||||
// conservative 'GET' default.
|
||||
const consumerCandidates = detections.filter(
|
||||
(d) => d.role === 'consumer' && normalizeConsumerPath(d.path) === pathNorm,
|
||||
);
|
||||
if (consumerCandidates.length === 1) {
|
||||
method = consumerCandidates[0].method;
|
||||
}
|
||||
|
||||
const cid = contractIdFor(method, pathNorm);
|
||||
let symbolUid = '';
|
||||
let symbolName = 'fetch';
|
||||
let symPath = filePath;
|
||||
const fileId = row.fileId ?? row[0];
|
||||
if (fileId) {
|
||||
try {
|
||||
const syms = await db(CONTAINS_QUERY, { fileId });
|
||||
if (syms.length > 0) {
|
||||
const picked = pickSymbolUid(syms, null);
|
||||
symbolUid = picked.uid;
|
||||
symbolName = picked.name;
|
||||
symPath = picked.filePath || filePath;
|
||||
}
|
||||
} catch {
|
||||
/* ignore */
|
||||
}
|
||||
}
|
||||
out.push({
|
||||
contractId: cid,
|
||||
type: 'http',
|
||||
role: 'consumer',
|
||||
symbolUid,
|
||||
symbolRef: { filePath: symPath, name: symbolName },
|
||||
symbolName,
|
||||
confidence: 0.9,
|
||||
meta: {
|
||||
method,
|
||||
path: pathNorm,
|
||||
extractionStrategy: 'graph_assisted',
|
||||
fetchReason: String(row.fetchReason ?? ''),
|
||||
},
|
||||
});
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
// ─── Source-scan consumers ─────────────────────────────────────────
|
||||
|
||||
private extractConsumersSourceScan(
|
||||
files: string[],
|
||||
getDetections: (rel: string) => HttpDetection[],
|
||||
): ExtractedContract[] {
|
||||
const out: ExtractedContract[] = [];
|
||||
for (const rel of files) {
|
||||
const detections = getDetections(rel);
|
||||
for (const d of detections) {
|
||||
if (d.role !== 'consumer') continue;
|
||||
const pathNorm = normalizeConsumerPath(d.path);
|
||||
out.push({
|
||||
contractId: contractIdFor(d.method, pathNorm),
|
||||
type: 'http',
|
||||
role: 'consumer',
|
||||
symbolUid: '',
|
||||
symbolRef: { filePath: rel, name: 'fetch' },
|
||||
symbolName: 'fetch',
|
||||
confidence: d.confidence,
|
||||
meta: {
|
||||
method: d.method,
|
||||
path: pathNorm,
|
||||
extractionStrategy: 'source_scan',
|
||||
framework: d.framework,
|
||||
},
|
||||
});
|
||||
}
|
||||
}
|
||||
return this.dedupeContracts(out);
|
||||
}
|
||||
|
||||
private dedupeContracts(items: ExtractedContract[]): ExtractedContract[] {
|
||||
const seen = new Set<string>();
|
||||
const out: ExtractedContract[] = [];
|
||||
for (const c of items) {
|
||||
const k = `${c.contractId}|${c.symbolRef.filePath}|${c.symbolRef.name}`;
|
||||
if (seen.has(k)) continue;
|
||||
seen.add(k);
|
||||
out.push(c);
|
||||
}
|
||||
return out;
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,311 @@
|
||||
import type { ContractType, CrossLink, GroupManifestLink, StoredContract } from '../types.js';
|
||||
import type { CypherExecutor } from '../contract-extractor.js';
|
||||
|
||||
export interface ManifestExtractResult {
|
||||
contracts: StoredContract[];
|
||||
crossLinks: CrossLink[];
|
||||
}
|
||||
|
||||
/**
|
||||
* Canonicalize an HTTP path for matching against Route.name in the graph.
|
||||
* Mirrors core/ingestion/pipeline.ts ensureSlash semantics:
|
||||
* - Ensures a leading slash.
|
||||
* - Strips trailing slashes (except the root "/").
|
||||
* - Normalizes consecutive slashes.
|
||||
* - Does NOT lowercase (route matching is case-sensitive).
|
||||
*/
|
||||
function normalizeRoutePath(raw: string): string {
|
||||
const trimmed = raw.trim();
|
||||
if (!trimmed) return '/';
|
||||
const withLeading = trimmed.startsWith('/') ? trimmed : `/${trimmed}`;
|
||||
const collapsed = withLeading.replace(/\/+/g, '/');
|
||||
if (collapsed === '/') return '/';
|
||||
return collapsed.replace(/\/+$/, '');
|
||||
}
|
||||
|
||||
/**
|
||||
* Split a manifest HTTP contract into its optional `METHOD::` prefix and
|
||||
* its path portion.
|
||||
*
|
||||
* `buildContractId` recommends the explicit-method form `GET::/api/orders`
|
||||
* in group.yaml; if we hand that raw string to `normalizeRoutePath` we get
|
||||
* `/GET::/api/orders`, which can never match `Route.name = "/api/orders"`
|
||||
* in the graph. This helper extracts the path portion so the Cypher
|
||||
* lookup uses the canonical route name.
|
||||
*
|
||||
* The method prefix regex mirrors `buildContractId` (line ~251) for
|
||||
* symmetry: case-insensitive `[A-Za-z]+` followed by `::`. The captured
|
||||
* method is upper-cased for downstream use; method-constrained matching
|
||||
* against `HANDLES_ROUTE` is a future enhancement (not yet wired).
|
||||
*
|
||||
* Edge cases:
|
||||
* - `"::/api/orders"` — empty method portion, no alpha prefix match, so
|
||||
* the whole string is treated as a bare path (matches buildContractId
|
||||
* which also requires `[A-Za-z]+`).
|
||||
* - `"GET::"` — method with empty path, returns `{ method: 'GET', path: '' }`;
|
||||
* `normalizeRoutePath('')` resolves to `/` for caller.
|
||||
*/
|
||||
function parseHttpContract(raw: string): { method: string | null; path: string } {
|
||||
const match = raw.match(/^([A-Za-z]+)::/);
|
||||
if (!match) return { method: null, path: raw };
|
||||
return { method: match[1].toUpperCase(), path: raw.slice(match[0].length) };
|
||||
}
|
||||
|
||||
/**
|
||||
* Stable synthetic symbolUid for a manifest-declared contract whose target
|
||||
* symbol could not be resolved against the per-repo graph (resolveSymbol
|
||||
* returned null). Two reasons we don't leave the uid empty:
|
||||
*
|
||||
* 1. The bridge stores Contract nodes keyed in part by symbolUid; an empty
|
||||
* uid means downstream Cypher queries that anchor on `provider.symbolUid`
|
||||
* can't tell two different unresolved manifest contracts apart.
|
||||
* 2. The cross-impact bridge query in cross-impact.ts joins local impact
|
||||
* results to bridge contracts via `WHERE provider.symbolUid IN $localUids`.
|
||||
* If the local impact engine produces a deterministic identifier for the
|
||||
* unresolved target, it must agree with the value the bridge stored. A
|
||||
* synthetic uid keyed off (repo, contractId) is the only thing both sides
|
||||
* can derive without knowing about each other.
|
||||
*
|
||||
* Format: `manifest::<repo>::<contractId>`. Stable across syncs, scoped to a
|
||||
* single repo within a group, and never collides with real indexer uids
|
||||
* (which never start with `manifest::`).
|
||||
*/
|
||||
export function manifestSymbolUid(repo: string, contractId: string): string {
|
||||
return `manifest::${repo}::${contractId}`;
|
||||
}
|
||||
|
||||
export class ManifestExtractor {
|
||||
async extractFromManifest(
|
||||
links: GroupManifestLink[],
|
||||
dbExecutors?: Map<string, CypherExecutor>,
|
||||
): Promise<ManifestExtractResult> {
|
||||
const contracts: StoredContract[] = [];
|
||||
const crossLinks: CrossLink[] = [];
|
||||
|
||||
for (const link of links) {
|
||||
const contractId = this.buildContractId(link.type, link.contract);
|
||||
|
||||
const providerRepo = link.role === 'provider' ? link.from : link.to;
|
||||
const consumerRepo = link.role === 'provider' ? link.to : link.from;
|
||||
|
||||
const providerSymbol = await this.resolveSymbol(providerRepo, link, dbExecutors);
|
||||
const consumerSymbol = await this.resolveSymbol(consumerRepo, link, dbExecutors);
|
||||
const providerRef = providerSymbol || { filePath: '', name: link.contract };
|
||||
const consumerRef = consumerSymbol || { filePath: '', name: link.contract };
|
||||
// When the resolver finds a real graph symbol we keep its uid, otherwise
|
||||
// fall back to the deterministic synthetic uid (see manifestSymbolUid).
|
||||
const providerUid = providerSymbol?.uid || manifestSymbolUid(providerRepo, contractId);
|
||||
const consumerUid = consumerSymbol?.uid || manifestSymbolUid(consumerRepo, contractId);
|
||||
|
||||
contracts.push({
|
||||
contractId,
|
||||
type: link.type,
|
||||
role: 'provider',
|
||||
symbolUid: providerUid,
|
||||
symbolRef: providerRef,
|
||||
symbolName: link.contract,
|
||||
confidence: 1.0,
|
||||
meta: { source: 'manifest' },
|
||||
repo: providerRepo,
|
||||
});
|
||||
|
||||
contracts.push({
|
||||
contractId,
|
||||
type: link.type,
|
||||
role: 'consumer',
|
||||
symbolUid: consumerUid,
|
||||
symbolRef: consumerRef,
|
||||
symbolName: link.contract,
|
||||
confidence: 1.0,
|
||||
meta: { source: 'manifest' },
|
||||
repo: consumerRepo,
|
||||
});
|
||||
|
||||
crossLinks.push({
|
||||
from: { repo: consumerRepo, symbolUid: consumerUid, symbolRef: consumerRef },
|
||||
to: { repo: providerRepo, symbolUid: providerUid, symbolRef: providerRef },
|
||||
type: link.type,
|
||||
contractId,
|
||||
matchType: 'manifest',
|
||||
confidence: 1.0,
|
||||
});
|
||||
}
|
||||
|
||||
return { contracts, crossLinks };
|
||||
}
|
||||
|
||||
private async resolveSymbol(
|
||||
repoPathKey: string,
|
||||
link: GroupManifestLink,
|
||||
dbExecutors?: Map<string, CypherExecutor>,
|
||||
): Promise<{ filePath: string; name: string; uid: string } | null> {
|
||||
const executor = dbExecutors?.get(repoPathKey);
|
||||
if (!executor) return null;
|
||||
|
||||
// NOTE: All lookups use EXACT equality on the relevant name field and
|
||||
// deterministic ORDER BY before LIMIT 1. Previous versions used CONTAINS
|
||||
// for fuzzy matching (plus an unconditional ".proto" fallback for gRPC)
|
||||
// which produced silent false positives: e.g. manifest "/orders" would
|
||||
// match "/suborders", and a gRPC manifest entry in a repo with any
|
||||
// .proto file would attach to a random proto symbol.
|
||||
//
|
||||
// If resolveSymbol returns null, the extractor falls back to a
|
||||
// deterministic synthetic uid via `manifestSymbolUid(repo, contractId)`
|
||||
// (see the function's docstring for why synthetic rather than empty).
|
||||
// Cross-impact still works: the bridge query joins on the synthetic
|
||||
// uid, and the local impact engine derives the same uid for the
|
||||
// unresolved symbol — name-based hints are the additional safety net.
|
||||
try {
|
||||
let rows: Record<string, unknown>[];
|
||||
if (link.type === 'http') {
|
||||
// Route.name is the canonicalized URL path (see
|
||||
// core/ingestion/pipeline.ts ensureSlash + generateId('Route', ...)).
|
||||
// Normalize the manifest contract the same way so a user-written
|
||||
// "/api/orders" matches "api/orders" in the graph.
|
||||
//
|
||||
// The contract may also use the explicit-method form "GET::/api/orders"
|
||||
// recommended by buildContractId. Strip the METHOD:: prefix before
|
||||
// normalizing — otherwise `normalizeRoutePath('GET::/api/orders')`
|
||||
// returns `/GET::/api/orders` and never matches Route.name. The
|
||||
// captured method is not yet used to constrain the Cypher query
|
||||
// (method-aware HANDLES_ROUTE matching is a future enhancement).
|
||||
const parsed = parseHttpContract(link.contract);
|
||||
const normalized = normalizeRoutePath(parsed.path);
|
||||
rows = await executor(
|
||||
`MATCH (handler)-[r:CodeRelation {type: 'HANDLES_ROUTE'}]->(route:Route)
|
||||
WHERE route.name = $normalized
|
||||
RETURN handler.id AS uid, handler.name AS name, handler.filePath AS filePath
|
||||
ORDER BY handler.filePath ASC
|
||||
LIMIT 1`,
|
||||
{ normalized },
|
||||
);
|
||||
} else if (link.type === 'topic') {
|
||||
// Topic names aren't a first-class NodeLabel in the graph —
|
||||
// topics are referenced by function/method symbols (Kafka
|
||||
// listeners, publishers). Restrict to symbol-like labels to
|
||||
// avoid cross-matching Files/Variables/Imports that happen to
|
||||
// share the topic name.
|
||||
rows = await executor(
|
||||
`MATCH (n:Function|Method|Class|Interface) WHERE n.name = $contract
|
||||
RETURN n.id AS uid, n.name AS name, n.filePath AS filePath
|
||||
ORDER BY n.filePath ASC
|
||||
LIMIT 1`,
|
||||
{ contract: link.contract },
|
||||
);
|
||||
} else if (link.type === 'grpc') {
|
||||
// Contract is "Service/Method" or just "Service" (or package.Service
|
||||
// variants). Prefer matching by method name when present, otherwise
|
||||
// by service name. NO .proto path fallback — that's guaranteed to
|
||||
// return a wrong symbol in any repo with more than one proto file.
|
||||
// Label filters scope lookups: methods → Function|Method, services
|
||||
// → Class|Interface (no label match = no silent wrong hits on
|
||||
// File/Variable nodes that happen to share the name).
|
||||
const parts = link.contract.split('/');
|
||||
const serviceName = parts[0]?.trim() ?? '';
|
||||
const methodName = parts[1]?.trim() ?? '';
|
||||
if (methodName) {
|
||||
rows = await executor(
|
||||
`MATCH (n:Function|Method) WHERE n.name = $methodName
|
||||
RETURN n.id AS uid, n.name AS name, n.filePath AS filePath
|
||||
ORDER BY n.filePath ASC
|
||||
LIMIT 1`,
|
||||
{ methodName },
|
||||
);
|
||||
} else if (serviceName) {
|
||||
rows = await executor(
|
||||
`MATCH (n:Class|Interface) WHERE n.name = $serviceName
|
||||
RETURN n.id AS uid, n.name AS name, n.filePath AS filePath
|
||||
ORDER BY n.filePath ASC
|
||||
LIMIT 1`,
|
||||
{ serviceName },
|
||||
);
|
||||
} else {
|
||||
rows = [];
|
||||
}
|
||||
} else if (link.type === 'lib') {
|
||||
// Only exact match on the symbol's name. Previous fallback to
|
||||
// CONTAINS on n.filePath would promote "react" to "react-native"
|
||||
// or "@types/react" — silent wrong attribution. Restrict to
|
||||
// package-level labels so we don't return arbitrary symbols
|
||||
// named after a library.
|
||||
rows = await executor(
|
||||
`MATCH (n:Package|Module) WHERE n.name = $contract
|
||||
RETURN n.id AS uid, n.name AS name, n.filePath AS filePath
|
||||
ORDER BY n.filePath ASC
|
||||
LIMIT 1`,
|
||||
{ contract: link.contract },
|
||||
);
|
||||
} else {
|
||||
return null;
|
||||
}
|
||||
if (rows.length > 0) {
|
||||
return {
|
||||
filePath: rows[0].filePath as string,
|
||||
name: rows[0].name as string,
|
||||
uid: String(rows[0].uid ?? ''),
|
||||
};
|
||||
}
|
||||
} catch (err) {
|
||||
// Log but don't throw: a broken graph query in one repo shouldn't
|
||||
// fail the whole manifest extraction. Unresolved contracts still
|
||||
// get a synthetic symbolUid below, so cross-impact can proceed.
|
||||
const message = err instanceof Error ? err.message : String(err);
|
||||
console.warn(
|
||||
`[manifest-extractor] resolveSymbol failed for ${link.type}:${link.contract} ` +
|
||||
`in ${repoPathKey}: ${message}`,
|
||||
);
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
/**
|
||||
* Build a canonical contract id for a manifest link.
|
||||
*
|
||||
* HTTP is the only type with two valid forms:
|
||||
* - Explicit method: `"GET::/api/orders"` → `"http::GET::/api/orders"`
|
||||
* (matches exactly against `HttpRouteExtractor` provider/consumer
|
||||
* contracts, which are also keyed by `http::<METHOD>::<path>`).
|
||||
* - Method-agnostic: `"/api/orders"` → `"http::*::/api/orders"`
|
||||
* — the `*` is a wildcard and is intended to match any concrete
|
||||
* HTTP method on that path. Wildcard-aware matching is the
|
||||
* responsibility of the sync / cross-impact layer (see #793);
|
||||
* downstream code should treat `http::*::<path>` as matching
|
||||
* every `http::<METHOD>::<path>` for the same path.
|
||||
*
|
||||
* Recommend the explicit-method form in group.yaml whenever the
|
||||
* manifest author knows the method — it round-trips through exact
|
||||
* equality matching without requiring wildcard logic downstream.
|
||||
*
|
||||
* NOTE on exhaustiveness: the switch covers every current
|
||||
* `ContractType` variant and falls through to a `never` assertion so
|
||||
* TypeScript fails the build if a new variant is added without a
|
||||
* corresponding case.
|
||||
*/
|
||||
private buildContractId(type: ContractType, contract: string): string {
|
||||
switch (type) {
|
||||
case 'http': {
|
||||
// Canonicalize method casing and path separators so logically
|
||||
// equivalent inputs (`get::/api/orders` vs `GET::/api/orders`,
|
||||
// or trailing-slash variants) produce the same contractId and
|
||||
// matching `manifestSymbolUid` fallback. Without this, raw
|
||||
// user casing leaks into cross-impact join keys and fragments
|
||||
// matches across repos.
|
||||
const { method, path: rawPath } = parseHttpContract(contract);
|
||||
const normalizedPath = normalizeRoutePath(rawPath);
|
||||
return method ? `http::${method}::${normalizedPath}` : `http::*::${normalizedPath}`;
|
||||
}
|
||||
case 'grpc':
|
||||
return `grpc::${contract}`;
|
||||
case 'topic':
|
||||
return `topic::${contract}`;
|
||||
case 'lib':
|
||||
return `lib::${contract}`;
|
||||
case 'custom':
|
||||
return `custom::${contract}`;
|
||||
default: {
|
||||
const _exhaustive: never = type;
|
||||
throw new Error(`Unhandled ContractType: ${String(_exhaustive)}`);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,114 @@
|
||||
import { glob } from 'glob';
|
||||
import Parser from 'tree-sitter';
|
||||
import type { ContractExtractor, CypherExecutor } from '../contract-extractor.js';
|
||||
import type { ExtractedContract, RepoHandle } from '../types.js';
|
||||
import { readSafe } from './fs-utils.js';
|
||||
import { scanFile, unquoteLiteral } from './tree-sitter-scanner.js';
|
||||
import {
|
||||
TOPIC_SCAN_GLOB,
|
||||
getProviderForFile,
|
||||
type Broker,
|
||||
type TopicMeta,
|
||||
} from './topic-patterns/index.js';
|
||||
|
||||
/**
|
||||
* Language-agnostic orchestrator for topic (message broker) contract
|
||||
* extraction. All grammar-specific knowledge lives in `topic-patterns/*`
|
||||
* — this file must not import any tree-sitter grammar directly.
|
||||
*
|
||||
* Flow per file:
|
||||
* 1. `getProviderForFile(rel)` → compiled plugin (or `undefined` if the
|
||||
* file's extension isn't registered, in which case we skip it).
|
||||
* 2. `scanFile(parser, provider, content)` → list of `{meta, valueText}`
|
||||
* pairs, one per matched literal.
|
||||
* 3. `unquoteLiteral(valueText)` → the raw topic string.
|
||||
* 4. `makeContract(topic, meta, relPath)` → `ExtractedContract`.
|
||||
*
|
||||
* Adding a new language is a one-file edit in `topic-patterns/index.ts`.
|
||||
*/
|
||||
|
||||
function makeContract(topicName: string, meta: TopicMeta, filePath: string): ExtractedContract {
|
||||
return {
|
||||
contractId: `topic::${topicName}`,
|
||||
type: 'topic',
|
||||
role: meta.role,
|
||||
symbolUid: '',
|
||||
symbolRef: { filePath: filePath.replace(/\\/g, '/'), name: meta.symbolName },
|
||||
symbolName: meta.symbolName,
|
||||
confidence: meta.confidence,
|
||||
meta: {
|
||||
broker: meta.broker satisfies Broker,
|
||||
topicName,
|
||||
extractionStrategy: 'tree_sitter',
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
export class TopicExtractor implements ContractExtractor {
|
||||
type = 'topic' as const;
|
||||
|
||||
async canExtract(_repo: RepoHandle): Promise<boolean> {
|
||||
return true;
|
||||
}
|
||||
|
||||
async extract(
|
||||
_dbExecutor: CypherExecutor | null,
|
||||
repoPath: string,
|
||||
_repo: RepoHandle,
|
||||
): Promise<ExtractedContract[]> {
|
||||
const files = await glob(TOPIC_SCAN_GLOB, {
|
||||
cwd: repoPath,
|
||||
ignore: [
|
||||
'**/node_modules/**',
|
||||
'**/.git/**',
|
||||
'**/vendor/**',
|
||||
'**/dist/**',
|
||||
'**/build/**',
|
||||
// Language-level test file conventions. Go test files
|
||||
// `*_test.go` live next to source; other languages either use
|
||||
// separate test directories (Python's `tests/`, Java's
|
||||
// `src/test/`) or are already covered by the dist/build ignores.
|
||||
// Pushed to the glob level so the orchestrator stays
|
||||
// language-agnostic.
|
||||
'**/*_test.go',
|
||||
],
|
||||
nodir: true,
|
||||
});
|
||||
|
||||
// One parser reused across files; the scanner calls `setLanguage` per
|
||||
// file based on which plugin the registry returns.
|
||||
const parser = new Parser();
|
||||
const out: ExtractedContract[] = [];
|
||||
|
||||
for (const rel of files) {
|
||||
const provider = getProviderForFile(rel);
|
||||
if (!provider) continue;
|
||||
|
||||
const content = readSafe(repoPath, rel);
|
||||
if (!content) continue;
|
||||
|
||||
const matches = scanFile(parser, provider, content);
|
||||
for (const match of matches) {
|
||||
const valueNode = match.captures.value;
|
||||
if (!valueNode) continue;
|
||||
const topicName = unquoteLiteral(valueNode.text);
|
||||
if (!topicName) continue;
|
||||
out.push(makeContract(topicName, match.meta, rel));
|
||||
}
|
||||
}
|
||||
|
||||
return this.dedupe(out);
|
||||
}
|
||||
|
||||
private dedupe(items: ExtractedContract[]): ExtractedContract[] {
|
||||
const seen = new Set<string>();
|
||||
const out: ExtractedContract[] = [];
|
||||
for (const c of items) {
|
||||
const k = `${c.contractId}|${c.role}|${c.symbolRef.filePath}`;
|
||||
if (seen.has(k)) continue;
|
||||
seen.add(k);
|
||||
out.push(c);
|
||||
}
|
||||
return out;
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,123 @@
|
||||
import Go from 'tree-sitter-go';
|
||||
import { compilePatterns, type LanguagePatterns } from '../tree-sitter-scanner.js';
|
||||
import type { TopicMeta } from './types.js';
|
||||
|
||||
/**
|
||||
* Go topic extraction patterns.
|
||||
*
|
||||
* Detects Sarama, segmentio/kafka-go and nats.go producer/consumer APIs:
|
||||
* - `X.ConsumePartition("topic", ...)`
|
||||
* - `sarama.ProducerMessage{Topic: "xxx"}`
|
||||
* - `kafka.Writer{Topic: "xxx"}` / `kafka.WriterConfig{Topic: ...}`
|
||||
* - `kafka.Reader{Topic: "xxx"}` / `kafka.ReaderConfig{Topic: ...}`
|
||||
* - `nc.Subscribe("topic", ...)` / `js.Subscribe("topic", ...)`
|
||||
* - `nc.Publish("topic", ...)` / `js.Publish("topic", ...)`
|
||||
*
|
||||
* Every query MUST bind `@value` to the topic literal node.
|
||||
*/
|
||||
const GO_TOPIC_SPEC: LanguagePatterns<TopicMeta> = {
|
||||
name: 'go-topic',
|
||||
language: Go,
|
||||
patterns: [
|
||||
{
|
||||
meta: {
|
||||
role: 'consumer',
|
||||
broker: 'kafka',
|
||||
confidence: 0.7,
|
||||
symbolName: 'ConsumePartition',
|
||||
},
|
||||
query: `
|
||||
(call_expression
|
||||
function: (selector_expression
|
||||
field: (field_identifier) @method (#eq? @method "ConsumePartition"))
|
||||
arguments: (argument_list . (interpreted_string_literal) @value))
|
||||
`,
|
||||
},
|
||||
{
|
||||
meta: {
|
||||
role: 'provider',
|
||||
broker: 'kafka',
|
||||
confidence: 0.75,
|
||||
symbolName: 'sarama.ProducerMessage',
|
||||
},
|
||||
query: `
|
||||
(composite_literal
|
||||
type: (qualified_type
|
||||
package: (package_identifier) @pkg (#eq? @pkg "sarama")
|
||||
name: (type_identifier) @ty (#eq? @ty "ProducerMessage"))
|
||||
body: (literal_value
|
||||
(keyed_element
|
||||
(literal_element (identifier) @field (#eq? @field "Topic"))
|
||||
(literal_element (interpreted_string_literal) @value))))
|
||||
`,
|
||||
},
|
||||
{
|
||||
meta: {
|
||||
role: 'provider',
|
||||
broker: 'kafka',
|
||||
confidence: 0.75,
|
||||
symbolName: 'kafka.Writer',
|
||||
},
|
||||
query: `
|
||||
(composite_literal
|
||||
type: (qualified_type
|
||||
package: (package_identifier) @pkg (#eq? @pkg "kafka")
|
||||
name: (type_identifier) @ty (#match? @ty "^(Writer|WriterConfig)$"))
|
||||
body: (literal_value
|
||||
(keyed_element
|
||||
(literal_element (identifier) @field (#eq? @field "Topic"))
|
||||
(literal_element (interpreted_string_literal) @value))))
|
||||
`,
|
||||
},
|
||||
{
|
||||
meta: {
|
||||
role: 'consumer',
|
||||
broker: 'kafka',
|
||||
confidence: 0.75,
|
||||
symbolName: 'kafka.Reader',
|
||||
},
|
||||
query: `
|
||||
(composite_literal
|
||||
type: (qualified_type
|
||||
package: (package_identifier) @pkg (#eq? @pkg "kafka")
|
||||
name: (type_identifier) @ty (#match? @ty "^(Reader|ReaderConfig)$"))
|
||||
body: (literal_value
|
||||
(keyed_element
|
||||
(literal_element (identifier) @field (#eq? @field "Topic"))
|
||||
(literal_element (interpreted_string_literal) @value))))
|
||||
`,
|
||||
},
|
||||
{
|
||||
meta: {
|
||||
role: 'consumer',
|
||||
broker: 'nats',
|
||||
confidence: 0.8,
|
||||
symbolName: 'nc.Subscribe',
|
||||
},
|
||||
query: `
|
||||
(call_expression
|
||||
function: (selector_expression
|
||||
operand: (identifier) @obj (#match? @obj "^(nc|js)$")
|
||||
field: (field_identifier) @method (#match? @method "^[Ss]ubscribe$"))
|
||||
arguments: (argument_list . (interpreted_string_literal) @value))
|
||||
`,
|
||||
},
|
||||
{
|
||||
meta: {
|
||||
role: 'provider',
|
||||
broker: 'nats',
|
||||
confidence: 0.8,
|
||||
symbolName: 'nc.Publish',
|
||||
},
|
||||
query: `
|
||||
(call_expression
|
||||
function: (selector_expression
|
||||
operand: (identifier) @obj (#match? @obj "^(nc|js)$")
|
||||
field: (field_identifier) @method (#match? @method "^[Pp]ublish$"))
|
||||
arguments: (argument_list . (interpreted_string_literal) @value))
|
||||
`,
|
||||
},
|
||||
],
|
||||
};
|
||||
|
||||
export const GO_TOPIC_PROVIDER = compilePatterns(GO_TOPIC_SPEC);
|
||||
@@ -0,0 +1,49 @@
|
||||
import * as path from 'node:path';
|
||||
import type { CompiledPatterns } from '../tree-sitter-scanner.js';
|
||||
import type { TopicMeta } from './types.js';
|
||||
import { JAVA_TOPIC_PROVIDER } from './java.js';
|
||||
import { GO_TOPIC_PROVIDER } from './go.js';
|
||||
import { PYTHON_TOPIC_PROVIDER } from './python.js';
|
||||
import {
|
||||
JAVASCRIPT_TOPIC_PROVIDER,
|
||||
TYPESCRIPT_TOPIC_PROVIDER,
|
||||
TSX_TOPIC_PROVIDER,
|
||||
} from './node.js';
|
||||
|
||||
export type { TopicMeta, Broker } from './types.js';
|
||||
|
||||
/**
|
||||
* File-extension → compiled-plugin registry for topic extraction. The
|
||||
* top-level orchestrator (`topic-extractor.ts`) looks up the plugin for
|
||||
* each file it visits and delegates the scanning to `tree-sitter-scanner`.
|
||||
*
|
||||
* Keys are lowercase extensions including the leading dot. To add a new
|
||||
* language, drop a `topic-patterns/<lang>.ts` that exports a compiled
|
||||
* provider, import it here and register the extension(s). No edits to
|
||||
* `topic-extractor.ts` are required.
|
||||
*/
|
||||
const REGISTRY: Record<string, CompiledPatterns<TopicMeta>> = {
|
||||
'.java': JAVA_TOPIC_PROVIDER,
|
||||
'.go': GO_TOPIC_PROVIDER,
|
||||
'.py': PYTHON_TOPIC_PROVIDER,
|
||||
'.js': JAVASCRIPT_TOPIC_PROVIDER,
|
||||
'.jsx': JAVASCRIPT_TOPIC_PROVIDER,
|
||||
'.ts': TYPESCRIPT_TOPIC_PROVIDER,
|
||||
'.tsx': TSX_TOPIC_PROVIDER,
|
||||
};
|
||||
|
||||
/**
|
||||
* Glob pattern for files worth scanning. Kept here so adding a new
|
||||
* language to the registry also widens the glob automatically via a
|
||||
* single edit.
|
||||
*/
|
||||
export const TOPIC_SCAN_GLOB = '**/*.{ts,tsx,js,jsx,java,go,py}';
|
||||
|
||||
/**
|
||||
* Return the compiled provider registered for the given file's
|
||||
* extension, or `undefined` if the extension is not registered.
|
||||
*/
|
||||
export function getProviderForFile(rel: string): CompiledPatterns<TopicMeta> | undefined {
|
||||
const ext = path.extname(rel).toLowerCase();
|
||||
return REGISTRY[ext];
|
||||
}
|
||||
@@ -0,0 +1,83 @@
|
||||
import Java from 'tree-sitter-java';
|
||||
import { compilePatterns, type LanguagePatterns } from '../tree-sitter-scanner.js';
|
||||
import type { TopicMeta } from './types.js';
|
||||
|
||||
/**
|
||||
* Java topic extraction patterns.
|
||||
*
|
||||
* Detects Kafka and RabbitMQ (Spring conventions) producer/consumer APIs:
|
||||
* - `@KafkaListener(topics = "xxx")`
|
||||
* - `@RabbitListener(queues = "xxx")`
|
||||
* - `kafkaTemplate.send("xxx", ...)`
|
||||
* - `rabbitTemplate.convertAndSend("xxx", ...)`
|
||||
*
|
||||
* Every query MUST bind `@value` to the topic literal node.
|
||||
*/
|
||||
const JAVA_TOPIC_SPEC: LanguagePatterns<TopicMeta> = {
|
||||
name: 'java-topic',
|
||||
language: Java,
|
||||
patterns: [
|
||||
{
|
||||
meta: {
|
||||
role: 'consumer',
|
||||
broker: 'kafka',
|
||||
confidence: 0.8,
|
||||
symbolName: 'kafkaListener',
|
||||
},
|
||||
query: `
|
||||
(annotation
|
||||
name: (identifier) @name (#eq? @name "KafkaListener")
|
||||
arguments: (annotation_argument_list
|
||||
(element_value_pair
|
||||
key: (identifier) @key (#eq? @key "topics")
|
||||
value: (string_literal) @value)))
|
||||
`,
|
||||
},
|
||||
{
|
||||
meta: {
|
||||
role: 'consumer',
|
||||
broker: 'rabbitmq',
|
||||
confidence: 0.8,
|
||||
symbolName: 'rabbitListener',
|
||||
},
|
||||
query: `
|
||||
(annotation
|
||||
name: (identifier) @name (#eq? @name "RabbitListener")
|
||||
arguments: (annotation_argument_list
|
||||
(element_value_pair
|
||||
key: (identifier) @key (#eq? @key "queues")
|
||||
value: (string_literal) @value)))
|
||||
`,
|
||||
},
|
||||
{
|
||||
meta: {
|
||||
role: 'provider',
|
||||
broker: 'kafka',
|
||||
confidence: 0.8,
|
||||
symbolName: 'kafkaTemplate.send',
|
||||
},
|
||||
query: `
|
||||
(method_invocation
|
||||
object: (identifier) @obj (#eq? @obj "kafkaTemplate")
|
||||
name: (identifier) @method (#eq? @method "send")
|
||||
arguments: (argument_list . (string_literal) @value))
|
||||
`,
|
||||
},
|
||||
{
|
||||
meta: {
|
||||
role: 'provider',
|
||||
broker: 'rabbitmq',
|
||||
confidence: 0.8,
|
||||
symbolName: 'rabbitTemplate.convertAndSend',
|
||||
},
|
||||
query: `
|
||||
(method_invocation
|
||||
object: (identifier) @obj (#eq? @obj "rabbitTemplate")
|
||||
name: (identifier) @method (#eq? @method "convertAndSend")
|
||||
arguments: (argument_list . (string_literal) @value))
|
||||
`,
|
||||
},
|
||||
],
|
||||
};
|
||||
|
||||
export const JAVA_TOPIC_PROVIDER = compilePatterns(JAVA_TOPIC_SPEC);
|
||||
@@ -0,0 +1,165 @@
|
||||
import JavaScript from 'tree-sitter-javascript';
|
||||
import TypeScript from 'tree-sitter-typescript';
|
||||
import {
|
||||
compilePatterns,
|
||||
type LanguagePatterns,
|
||||
type PatternSpec,
|
||||
} from '../tree-sitter-scanner.js';
|
||||
import type { TopicMeta } from './types.js';
|
||||
|
||||
/**
|
||||
* Node.js / TypeScript topic extraction patterns.
|
||||
*
|
||||
* Detects kafkajs, amqplib (RabbitMQ), and nats.js producer/consumer APIs:
|
||||
* - `producer.send({ topic: 'xxx', ... })` (kafkajs)
|
||||
* - `consumer.subscribe({ topic: 'xxx', ... })` (kafkajs)
|
||||
* - `channel.consume("queue", ...)` / `channel.publish(...)` / `channel.sendToQueue(...)`
|
||||
* - `nc.subscribe("topic")` / `js.subscribe("topic")`
|
||||
* - `nc.publish("topic", ...)` / `js.publish("topic", ...)`
|
||||
*
|
||||
* The JavaScript and TypeScript tree-sitter grammars share node type
|
||||
* names for every construct we query here, so the pattern sources are
|
||||
* defined once and compiled against each grammar variant. We export three
|
||||
* providers because Parser.Query objects are NOT portable across grammar
|
||||
* instances — `.js` files use the JavaScript grammar, `.ts` uses
|
||||
* TypeScript.typescript, and `.tsx` uses TypeScript.tsx.
|
||||
*
|
||||
* Every query MUST bind `@value` to the topic literal node.
|
||||
*/
|
||||
const NODE_TOPIC_PATTERNS: PatternSpec<TopicMeta>[] = [
|
||||
{
|
||||
meta: {
|
||||
role: 'provider',
|
||||
broker: 'kafka',
|
||||
confidence: 0.8,
|
||||
symbolName: 'producer.send',
|
||||
},
|
||||
query: `
|
||||
(call_expression
|
||||
function: (member_expression
|
||||
object: (identifier) @obj (#eq? @obj "producer")
|
||||
property: (property_identifier) @prop (#eq? @prop "send"))
|
||||
arguments: (arguments
|
||||
(object
|
||||
(pair
|
||||
key: (property_identifier) @key (#eq? @key "topic")
|
||||
value: [(string) (template_string)] @value))))
|
||||
`,
|
||||
},
|
||||
{
|
||||
meta: {
|
||||
role: 'consumer',
|
||||
broker: 'kafka',
|
||||
confidence: 0.8,
|
||||
symbolName: 'consumer.subscribe',
|
||||
},
|
||||
query: `
|
||||
(call_expression
|
||||
function: (member_expression
|
||||
object: (identifier) @obj (#eq? @obj "consumer")
|
||||
property: (property_identifier) @prop (#eq? @prop "subscribe"))
|
||||
arguments: (arguments
|
||||
(object
|
||||
(pair
|
||||
key: (property_identifier) @key (#eq? @key "topic")
|
||||
value: [(string) (template_string)] @value))))
|
||||
`,
|
||||
},
|
||||
{
|
||||
meta: {
|
||||
role: 'consumer',
|
||||
broker: 'rabbitmq',
|
||||
confidence: 0.8,
|
||||
symbolName: 'channel.consume',
|
||||
},
|
||||
query: `
|
||||
(call_expression
|
||||
function: (member_expression
|
||||
object: (identifier) @obj (#eq? @obj "channel")
|
||||
property: (property_identifier) @prop (#eq? @prop "consume"))
|
||||
arguments: (arguments . [(string) (template_string)] @value))
|
||||
`,
|
||||
},
|
||||
{
|
||||
meta: {
|
||||
role: 'provider',
|
||||
broker: 'rabbitmq',
|
||||
confidence: 0.8,
|
||||
symbolName: 'channel.publish',
|
||||
},
|
||||
query: `
|
||||
(call_expression
|
||||
function: (member_expression
|
||||
object: (identifier) @obj (#eq? @obj "channel")
|
||||
property: (property_identifier) @prop (#eq? @prop "publish"))
|
||||
arguments: (arguments . [(string) (template_string)] @value))
|
||||
`,
|
||||
},
|
||||
{
|
||||
meta: {
|
||||
role: 'provider',
|
||||
broker: 'rabbitmq',
|
||||
confidence: 0.8,
|
||||
symbolName: 'channel.sendToQueue',
|
||||
},
|
||||
query: `
|
||||
(call_expression
|
||||
function: (member_expression
|
||||
object: (identifier) @obj (#eq? @obj "channel")
|
||||
property: (property_identifier) @prop (#eq? @prop "sendToQueue"))
|
||||
arguments: (arguments . [(string) (template_string)] @value))
|
||||
`,
|
||||
},
|
||||
{
|
||||
meta: {
|
||||
role: 'consumer',
|
||||
broker: 'nats',
|
||||
confidence: 0.8,
|
||||
symbolName: 'nc.subscribe',
|
||||
},
|
||||
query: `
|
||||
(call_expression
|
||||
function: (member_expression
|
||||
object: (identifier) @obj (#match? @obj "^(nc|js)$")
|
||||
property: (property_identifier) @prop (#match? @prop "^[Ss]ubscribe$"))
|
||||
arguments: (arguments . [(string) (template_string)] @value))
|
||||
`,
|
||||
},
|
||||
{
|
||||
meta: {
|
||||
role: 'provider',
|
||||
broker: 'nats',
|
||||
confidence: 0.8,
|
||||
symbolName: 'nc.publish',
|
||||
},
|
||||
query: `
|
||||
(call_expression
|
||||
function: (member_expression
|
||||
object: (identifier) @obj (#match? @obj "^(nc|js)$")
|
||||
property: (property_identifier) @prop (#match? @prop "^[Pp]ublish$"))
|
||||
arguments: (arguments . [(string) (template_string)] @value))
|
||||
`,
|
||||
},
|
||||
];
|
||||
|
||||
const JAVASCRIPT_TOPIC_SPEC: LanguagePatterns<TopicMeta> = {
|
||||
name: 'javascript-topic',
|
||||
language: JavaScript,
|
||||
patterns: NODE_TOPIC_PATTERNS,
|
||||
};
|
||||
|
||||
const TYPESCRIPT_TOPIC_SPEC: LanguagePatterns<TopicMeta> = {
|
||||
name: 'typescript-topic',
|
||||
language: TypeScript.typescript,
|
||||
patterns: NODE_TOPIC_PATTERNS,
|
||||
};
|
||||
|
||||
const TSX_TOPIC_SPEC: LanguagePatterns<TopicMeta> = {
|
||||
name: 'tsx-topic',
|
||||
language: TypeScript.tsx,
|
||||
patterns: NODE_TOPIC_PATTERNS,
|
||||
};
|
||||
|
||||
export const JAVASCRIPT_TOPIC_PROVIDER = compilePatterns(JAVASCRIPT_TOPIC_SPEC);
|
||||
export const TYPESCRIPT_TOPIC_PROVIDER = compilePatterns(TYPESCRIPT_TOPIC_SPEC);
|
||||
export const TSX_TOPIC_PROVIDER = compilePatterns(TSX_TOPIC_SPEC);
|
||||
@@ -0,0 +1,119 @@
|
||||
import Python from 'tree-sitter-python';
|
||||
import { compilePatterns, type LanguagePatterns } from '../tree-sitter-scanner.js';
|
||||
import type { TopicMeta } from './types.js';
|
||||
|
||||
/**
|
||||
* Python topic extraction patterns.
|
||||
*
|
||||
* Detects kafka-python, pika (RabbitMQ), and nats-py producer/consumer APIs:
|
||||
* - `KafkaConsumer('topic', ...)`
|
||||
* - `producer.send('topic', ...)` / `producer.produce('topic', ...)`
|
||||
* - `channel.basic_consume(queue='xxx', ...)`
|
||||
* - `channel.basic_publish(exchange='xxx', ...)`
|
||||
* - `await nc.subscribe('topic')`
|
||||
* - `await nc.publish('topic', ...)`
|
||||
*
|
||||
* Every query MUST bind `@value` to the topic literal node.
|
||||
*/
|
||||
const PYTHON_TOPIC_SPEC: LanguagePatterns<TopicMeta> = {
|
||||
name: 'python-topic',
|
||||
language: Python,
|
||||
patterns: [
|
||||
{
|
||||
meta: {
|
||||
role: 'consumer',
|
||||
broker: 'kafka',
|
||||
confidence: 0.7,
|
||||
symbolName: 'KafkaConsumer',
|
||||
},
|
||||
query: `
|
||||
(call
|
||||
function: (identifier) @func (#eq? @func "KafkaConsumer")
|
||||
arguments: (argument_list . (string) @value))
|
||||
`,
|
||||
},
|
||||
{
|
||||
meta: {
|
||||
role: 'provider',
|
||||
broker: 'kafka',
|
||||
confidence: 0.7,
|
||||
symbolName: 'producer.send',
|
||||
},
|
||||
query: `
|
||||
(call
|
||||
function: (attribute
|
||||
object: (identifier) @obj (#eq? @obj "producer")
|
||||
attribute: (identifier) @method (#match? @method "^(send|produce)$"))
|
||||
arguments: (argument_list . (string) @value))
|
||||
`,
|
||||
},
|
||||
{
|
||||
meta: {
|
||||
role: 'consumer',
|
||||
broker: 'rabbitmq',
|
||||
confidence: 0.7,
|
||||
symbolName: 'basic_consume',
|
||||
},
|
||||
query: `
|
||||
(call
|
||||
function: (attribute
|
||||
object: (identifier) @obj (#eq? @obj "channel")
|
||||
attribute: (identifier) @method (#eq? @method "basic_consume"))
|
||||
arguments: (argument_list
|
||||
(keyword_argument
|
||||
name: (identifier) @kw (#eq? @kw "queue")
|
||||
value: (string) @value)))
|
||||
`,
|
||||
},
|
||||
{
|
||||
meta: {
|
||||
role: 'provider',
|
||||
broker: 'rabbitmq',
|
||||
confidence: 0.7,
|
||||
symbolName: 'basic_publish',
|
||||
},
|
||||
query: `
|
||||
(call
|
||||
function: (attribute
|
||||
object: (identifier) @obj (#eq? @obj "channel")
|
||||
attribute: (identifier) @method (#eq? @method "basic_publish"))
|
||||
arguments: (argument_list
|
||||
(keyword_argument
|
||||
name: (identifier) @kw (#eq? @kw "exchange")
|
||||
value: (string) @value)))
|
||||
`,
|
||||
},
|
||||
{
|
||||
meta: {
|
||||
role: 'consumer',
|
||||
broker: 'nats',
|
||||
confidence: 0.75,
|
||||
symbolName: 'nc.subscribe',
|
||||
},
|
||||
query: `
|
||||
(call
|
||||
function: (attribute
|
||||
object: (identifier) @obj (#eq? @obj "nc")
|
||||
attribute: (identifier) @method (#eq? @method "subscribe"))
|
||||
arguments: (argument_list . (string) @value))
|
||||
`,
|
||||
},
|
||||
{
|
||||
meta: {
|
||||
role: 'provider',
|
||||
broker: 'nats',
|
||||
confidence: 0.75,
|
||||
symbolName: 'nc.publish',
|
||||
},
|
||||
query: `
|
||||
(call
|
||||
function: (attribute
|
||||
object: (identifier) @obj (#eq? @obj "nc")
|
||||
attribute: (identifier) @method (#eq? @method "publish"))
|
||||
arguments: (argument_list . (string) @value))
|
||||
`,
|
||||
},
|
||||
],
|
||||
};
|
||||
|
||||
export const PYTHON_TOPIC_PROVIDER = compilePatterns(PYTHON_TOPIC_SPEC);
|
||||
@@ -0,0 +1,27 @@
|
||||
/**
|
||||
* Shared types for the topic-extractor language plugins.
|
||||
*
|
||||
* Each plugin lives in its own file (java.ts, go.ts, ...) and owns the
|
||||
* tree-sitter grammar import + query sources. The top-level
|
||||
* `topic-extractor.ts` orchestrator only knows about this type module and
|
||||
* the plugin registry (`./index.ts`). It MUST NOT import any grammar or
|
||||
* query text directly — that's the whole point of the split.
|
||||
*/
|
||||
|
||||
export type Broker = 'kafka' | 'rabbitmq' | 'nats';
|
||||
|
||||
/**
|
||||
* Per-pattern payload every topic plugin attaches to its query. Whatever
|
||||
* the pattern matches, the orchestrator receives this object verbatim
|
||||
* and uses it to build an `ExtractedContract`.
|
||||
*
|
||||
* Plugins produce one `TopicMeta` per pattern (not per match) because a
|
||||
* single query uniquely identifies its broker/role/confidence triple.
|
||||
*/
|
||||
export interface TopicMeta {
|
||||
role: 'provider' | 'consumer';
|
||||
broker: Broker;
|
||||
confidence: number;
|
||||
/** Short human-readable label of the API being detected. */
|
||||
symbolName: string;
|
||||
}
|
||||
@@ -0,0 +1,193 @@
|
||||
import Parser from 'tree-sitter';
|
||||
|
||||
/**
|
||||
* Shared, language-agnostic tree-sitter scanning utilities used by group
|
||||
* extractors (topic, http, grpc, ...).
|
||||
*
|
||||
* Design goals:
|
||||
* - The top-level extractors must not import any tree-sitter grammar.
|
||||
* - Per-language plugins own their grammar import, their query sources,
|
||||
* and the mapping from capture → meta.
|
||||
* - This module provides the plumbing: compile queries once per plugin,
|
||||
* parse a file with a given grammar, run all patterns, and return the
|
||||
* captured `string_literal`-style nodes together with the plugin's meta.
|
||||
*/
|
||||
|
||||
/**
|
||||
* One pattern owned by a language plugin. Each pattern owns a tree-sitter
|
||||
* S-expression query. Plugins can freely choose which capture names to
|
||||
* use — the scanner exposes every capture in the returned `captures`
|
||||
* map and does not privilege any particular name.
|
||||
*
|
||||
* `TMeta` is the plugin-specific payload the orchestrator receives back
|
||||
* when this pattern matches — e.g. for topic extraction it carries the
|
||||
* broker name, role, confidence, symbol name.
|
||||
*/
|
||||
export interface PatternSpec<TMeta> {
|
||||
/** Tree-sitter S-expression. */
|
||||
query: string;
|
||||
/** Plugin-specific payload returned on every match. */
|
||||
meta: TMeta;
|
||||
}
|
||||
|
||||
/**
|
||||
* A set of patterns owned by one language plugin, bound to a specific
|
||||
* tree-sitter grammar.
|
||||
*
|
||||
* `language` is typed as `unknown` because tree-sitter's TypeScript
|
||||
* declarations use `any` for the grammar object, and the grammar modules
|
||||
* export different shapes (plain grammar vs. namespace with `typescript`
|
||||
* / `tsx` members). Callers pass the concrete grammar object; this
|
||||
* module forwards it to `parser.setLanguage` / `new Parser.Query`.
|
||||
*/
|
||||
export interface LanguagePatterns<TMeta> {
|
||||
/** Human-readable plugin name for diagnostics. */
|
||||
name: string;
|
||||
/** tree-sitter grammar object. */
|
||||
language: unknown;
|
||||
/** Patterns authored against `language`. */
|
||||
patterns: PatternSpec<TMeta>[];
|
||||
}
|
||||
|
||||
/**
|
||||
* Compiled form of a `LanguagePatterns` bundle. Queries are compiled
|
||||
* eagerly at module load time so a broken grammar/query pair fails
|
||||
* loudly the first time the plugin is imported, instead of silently
|
||||
* at scan time when no contract is produced.
|
||||
*/
|
||||
export interface CompiledPatterns<TMeta> {
|
||||
name: string;
|
||||
language: unknown;
|
||||
patterns: CompiledPattern<TMeta>[];
|
||||
}
|
||||
|
||||
export interface CompiledPattern<TMeta> {
|
||||
query: Parser.Query;
|
||||
meta: TMeta;
|
||||
}
|
||||
|
||||
/**
|
||||
* Map from capture name → syntax node. Every named capture the query
|
||||
* binds is exposed as an entry. If a query captures the same name more
|
||||
* than once (unusual), the first occurrence wins — plugins that need
|
||||
* all occurrences should use distinct capture names or fall back to
|
||||
* `match.captures` array directly by iterating `query.matches()`
|
||||
* themselves.
|
||||
*/
|
||||
export type CaptureMap = Record<string, Parser.SyntaxNode>;
|
||||
|
||||
/**
|
||||
* One match returned by `scanFile` / `runCompiledPatterns`. The caller
|
||||
* receives the full capture map plus the plugin meta, and is
|
||||
* responsible for turning it into a domain object.
|
||||
*/
|
||||
export interface ScanMatch<TMeta> {
|
||||
meta: TMeta;
|
||||
captures: CaptureMap;
|
||||
}
|
||||
|
||||
/**
|
||||
* Compile a LanguagePatterns bundle. Call this once per plugin, at
|
||||
* module load time, and export the result. Throws if any pattern
|
||||
* fails to compile against the grammar — that's a bug in the plugin
|
||||
* author's query, not a runtime condition.
|
||||
*/
|
||||
export function compilePatterns<TMeta>(bundle: LanguagePatterns<TMeta>): CompiledPatterns<TMeta> {
|
||||
const compiled: CompiledPattern<TMeta>[] = [];
|
||||
for (const spec of bundle.patterns) {
|
||||
try {
|
||||
const query = new Parser.Query(bundle.language, spec.query);
|
||||
compiled.push({ query, meta: spec.meta });
|
||||
} catch (err) {
|
||||
const message = err instanceof Error ? err.message : String(err);
|
||||
throw new Error(
|
||||
`[tree-sitter-scanner] Failed to compile pattern in ${bundle.name}: ${message}\n` +
|
||||
`Query source:\n${spec.query}`,
|
||||
);
|
||||
}
|
||||
}
|
||||
return { name: bundle.name, language: bundle.language, patterns: compiled };
|
||||
}
|
||||
|
||||
/**
|
||||
* Run every compiled pattern in `plugin` against an already-parsed
|
||||
* tree. Use this when a plugin needs multiple query bundles against
|
||||
* the same file (e.g. one query for class-level prefixes and another
|
||||
* for method-level annotations) and wants to avoid re-parsing.
|
||||
*/
|
||||
export function runCompiledPatterns<TMeta>(
|
||||
plugin: CompiledPatterns<TMeta>,
|
||||
tree: Parser.Tree,
|
||||
): ScanMatch<TMeta>[] {
|
||||
const out: ScanMatch<TMeta>[] = [];
|
||||
for (const compiled of plugin.patterns) {
|
||||
let matches: Parser.QueryMatch[];
|
||||
try {
|
||||
matches = compiled.query.matches(tree.rootNode);
|
||||
} catch {
|
||||
continue;
|
||||
}
|
||||
for (const match of matches) {
|
||||
const captures: CaptureMap = {};
|
||||
for (const cap of match.captures) {
|
||||
if (!(cap.name in captures)) captures[cap.name] = cap.node;
|
||||
}
|
||||
out.push({ meta: compiled.meta, captures });
|
||||
}
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
/**
|
||||
* Parse `content` with the plugin's grammar and run every compiled
|
||||
* pattern against the AST. Returns one `ScanMatch` per matched query
|
||||
* occurrence, carrying the plugin's meta payload.
|
||||
*
|
||||
* Errors are swallowed at the file level (malformed file must not abort
|
||||
* the whole extract). Individual pattern failures are swallowed too so
|
||||
* a single unusable query doesn't block the rest of the plugin.
|
||||
*/
|
||||
export function scanFile<TMeta>(
|
||||
parser: Parser,
|
||||
plugin: CompiledPatterns<TMeta>,
|
||||
content: string,
|
||||
): ScanMatch<TMeta>[] {
|
||||
let tree: Parser.Tree;
|
||||
try {
|
||||
parser.setLanguage(plugin.language);
|
||||
tree = parser.parse(content);
|
||||
} catch {
|
||||
return [];
|
||||
}
|
||||
return runCompiledPatterns(plugin, tree);
|
||||
}
|
||||
|
||||
/**
|
||||
* Strip enclosing quotes from a tree-sitter string literal node's text.
|
||||
* Handles single / double / template quotes, Python triple-quoted strings,
|
||||
* and Go raw string literals (backticks).
|
||||
*
|
||||
* Returns null for empty/nullish input so callers can uniformly skip
|
||||
* captures whose value is missing.
|
||||
*/
|
||||
export function unquoteLiteral(raw: string): string | null {
|
||||
if (!raw) return null;
|
||||
|
||||
// Python triple-quoted
|
||||
if (
|
||||
(raw.startsWith('"""') && raw.endsWith('"""')) ||
|
||||
(raw.startsWith("'''") && raw.endsWith("'''"))
|
||||
) {
|
||||
return raw.slice(3, -3);
|
||||
}
|
||||
|
||||
const first = raw[0];
|
||||
const last = raw[raw.length - 1];
|
||||
if ((first === '"' || first === "'" || first === '`') && last === first && raw.length >= 2) {
|
||||
return raw.slice(1, -1);
|
||||
}
|
||||
|
||||
// Some grammars expose the string content without quotes already (e.g.
|
||||
// Python `string_content` child). Return as-is.
|
||||
return raw;
|
||||
}
|
||||
@@ -0,0 +1,237 @@
|
||||
import type { StoredContract, CrossLink } from './types.js';
|
||||
|
||||
export interface MatchResult {
|
||||
matched: CrossLink[];
|
||||
unmatched: StoredContract[];
|
||||
}
|
||||
|
||||
export interface WildcardMatchResult {
|
||||
matched: CrossLink[];
|
||||
remaining: StoredContract[];
|
||||
}
|
||||
|
||||
function isGrpcWildcard(cid: string): boolean {
|
||||
return cid.startsWith('grpc::') && cid.endsWith('/*');
|
||||
}
|
||||
|
||||
export function normalizeContractId(id: string): string {
|
||||
const colonIdx = id.indexOf('::');
|
||||
if (colonIdx === -1) return id;
|
||||
|
||||
const type = id.substring(0, colonIdx);
|
||||
const rest = id.substring(colonIdx + 2);
|
||||
|
||||
switch (type) {
|
||||
case 'http': {
|
||||
const parts = rest.split('::');
|
||||
if (parts.length >= 2) {
|
||||
const method = parts[0].toUpperCase();
|
||||
let pathPart = parts.slice(1).join('::');
|
||||
pathPart = pathPart.replace(/\/+$/, '');
|
||||
return `http::${method}::${pathPart}`;
|
||||
}
|
||||
return id;
|
||||
}
|
||||
case 'grpc': {
|
||||
// Canonical form: `grpc::<lowercased-package-or-service>[/<method>]`.
|
||||
//
|
||||
// The package/service segment is lowercased because gRPC package
|
||||
// names are effectively case-insensitive across language bindings
|
||||
// (`auth.AuthService`, `auth.authservice`, `AUTH.AUTHSERVICE` all
|
||||
// describe the same wire protocol service). The RPC method segment
|
||||
// is preserved as-is because the HTTP/2 path used on the wire is
|
||||
// case-sensitive per the gRPC spec (`/Service/MethodName`), and
|
||||
// method names in generated clients match the proto source exactly.
|
||||
//
|
||||
// A package-only id (no slash) and a package/method id are treated
|
||||
// as DISTINCT canonical forms: `grpc::userservice` does not match
|
||||
// `grpc::userservice/Login`. That's by design — callers that want
|
||||
// service-level manifest matching against method-level providers
|
||||
// should use the gRPC wildcard form `grpc::UserService/*` which is
|
||||
// handled by runWildcardMatch below.
|
||||
const slashIdx = rest.indexOf('/');
|
||||
if (slashIdx > 0) {
|
||||
const pkg = rest.substring(0, slashIdx).toLowerCase();
|
||||
const method = rest.substring(slashIdx);
|
||||
return `grpc::${pkg}${method}`;
|
||||
}
|
||||
if (slashIdx === 0) {
|
||||
// Malformed "/method" with leading slash — keep as-is so two
|
||||
// equally malformed ids can still match each other.
|
||||
return `grpc::${rest}`;
|
||||
}
|
||||
// No slash: package/service only. Lowercase to match the package
|
||||
// segment produced by the pkg/method branch above.
|
||||
return `grpc::${rest.toLowerCase()}`;
|
||||
}
|
||||
case 'topic':
|
||||
return `topic::${rest.trim().toLowerCase()}`;
|
||||
case 'lib':
|
||||
return `lib::${rest.toLowerCase()}`;
|
||||
default:
|
||||
return id;
|
||||
}
|
||||
}
|
||||
|
||||
function findMatchingKeys(contractId: string, index: Map<string, StoredContract[]>): string[] {
|
||||
const normalized = normalizeContractId(contractId);
|
||||
if (index.has(normalized)) return [normalized];
|
||||
|
||||
if (normalized.startsWith('http::*::')) {
|
||||
const pathPart = normalized.substring('http::*::'.length);
|
||||
const matches: string[] = [];
|
||||
for (const key of index.keys()) {
|
||||
if (key.startsWith('http::') && key.endsWith(`::${pathPart}`)) {
|
||||
matches.push(key);
|
||||
}
|
||||
}
|
||||
return matches;
|
||||
}
|
||||
|
||||
return [];
|
||||
}
|
||||
|
||||
export function buildProviderIndex(contracts: StoredContract[]): Map<string, StoredContract[]> {
|
||||
const providers = contracts.filter((c) => c.role === 'provider');
|
||||
const index = new Map<string, StoredContract[]>();
|
||||
for (const p of providers) {
|
||||
const key = normalizeContractId(p.contractId);
|
||||
const list = index.get(key) || [];
|
||||
list.push(p);
|
||||
index.set(key, list);
|
||||
}
|
||||
return index;
|
||||
}
|
||||
|
||||
export function runExactMatch(
|
||||
contracts: StoredContract[],
|
||||
providerIndex?: Map<string, StoredContract[]>,
|
||||
): MatchResult {
|
||||
const index = providerIndex ?? buildProviderIndex(contracts);
|
||||
|
||||
// Skip gRPC wildcard consumers — they go to wildcard pass only
|
||||
const consumers = contracts.filter((c) => c.role === 'consumer' && !isGrpcWildcard(c.contractId));
|
||||
|
||||
const matched: CrossLink[] = [];
|
||||
const matchedConsumerIds = new Set<string>();
|
||||
const matchedProviderIds = new Set<string>();
|
||||
|
||||
for (const consumer of consumers) {
|
||||
const matchingKeys = findMatchingKeys(consumer.contractId, index);
|
||||
if (matchingKeys.length === 0) continue;
|
||||
|
||||
const allMatchingProviders = matchingKeys.flatMap((k) => index.get(k) || []);
|
||||
for (const provider of allMatchingProviders) {
|
||||
if (provider.repo === consumer.repo) {
|
||||
if (!provider.service || !consumer.service || provider.service === consumer.service) {
|
||||
continue;
|
||||
}
|
||||
}
|
||||
|
||||
matched.push({
|
||||
from: {
|
||||
repo: consumer.repo,
|
||||
service: consumer.service,
|
||||
symbolUid: consumer.symbolUid,
|
||||
symbolRef: consumer.symbolRef,
|
||||
},
|
||||
to: {
|
||||
repo: provider.repo,
|
||||
service: provider.service,
|
||||
symbolUid: provider.symbolUid,
|
||||
symbolRef: provider.symbolRef,
|
||||
},
|
||||
type: consumer.type,
|
||||
contractId: consumer.contractId,
|
||||
matchType: 'exact',
|
||||
confidence: 1.0,
|
||||
});
|
||||
|
||||
matchedConsumerIds.add(`${consumer.repo}::${consumer.contractId}`);
|
||||
matchedProviderIds.add(`${provider.repo}::${provider.contractId}`);
|
||||
}
|
||||
}
|
||||
|
||||
// normalUnmatched: contracts that weren't matched in exact pass
|
||||
const normalUnmatched = contracts.filter((c) => {
|
||||
if (isGrpcWildcard(c.contractId)) return false; // excluded from exact, handled separately
|
||||
const id = `${c.repo}::${c.contractId}`;
|
||||
return c.role === 'provider' ? !matchedProviderIds.has(id) : !matchedConsumerIds.has(id);
|
||||
});
|
||||
|
||||
// Re-add gRPC wildcard contracts — they were never in exact matching
|
||||
const grpcWildcards = contracts.filter((c) => isGrpcWildcard(c.contractId));
|
||||
const unmatched = [...normalUnmatched, ...grpcWildcards];
|
||||
|
||||
return { matched, unmatched };
|
||||
}
|
||||
|
||||
export function runWildcardMatch(
|
||||
unmatched: StoredContract[],
|
||||
providerIndex: Map<string, StoredContract[]>,
|
||||
): WildcardMatchResult {
|
||||
const wildcardConsumers = unmatched.filter(
|
||||
(c) => c.role === 'consumer' && isGrpcWildcard(c.contractId),
|
||||
);
|
||||
const matched: CrossLink[] = [];
|
||||
const matchedConsumerIds = new Set<string>();
|
||||
|
||||
for (const consumer of wildcardConsumers) {
|
||||
const normalized = normalizeContractId(consumer.contractId);
|
||||
// "grpc::com.example.userservice/*" → "com.example.userservice"
|
||||
// "grpc::userservice/*" → "userservice"
|
||||
const fqService = normalized.slice(normalized.indexOf('::') + 2, -2); // strip "grpc::" and "/*"
|
||||
|
||||
for (const [key, providers] of providerIndex) {
|
||||
// Only match against non-wildcard gRPC providers (method-level IDs)
|
||||
if (!key.startsWith('grpc::') || key.endsWith('/*')) continue;
|
||||
const afterPrefix = key.slice(6); // strip "grpc::"
|
||||
const slashIdx = afterPrefix.indexOf('/');
|
||||
if (slashIdx < 0) continue;
|
||||
const providerFqService = afterPrefix.slice(0, slashIdx);
|
||||
|
||||
// Match: exact FQ service, or bare-name match when consumer has no package
|
||||
const isMatch =
|
||||
providerFqService === fqService ||
|
||||
(!fqService.includes('.') && providerFqService.endsWith('.' + fqService));
|
||||
|
||||
if (!isMatch) continue;
|
||||
|
||||
for (const provider of providers) {
|
||||
// Skip same-repo same-service (same logic as runExactMatch)
|
||||
if (provider.repo === consumer.repo) {
|
||||
if (!provider.service || !consumer.service || provider.service === consumer.service) {
|
||||
continue;
|
||||
}
|
||||
}
|
||||
|
||||
matched.push({
|
||||
from: {
|
||||
repo: consumer.repo,
|
||||
service: consumer.service,
|
||||
symbolUid: consumer.symbolUid,
|
||||
symbolRef: consumer.symbolRef,
|
||||
},
|
||||
to: {
|
||||
repo: provider.repo,
|
||||
service: provider.service,
|
||||
symbolUid: provider.symbolUid,
|
||||
symbolRef: provider.symbolRef,
|
||||
},
|
||||
type: consumer.type,
|
||||
contractId: consumer.contractId, // consumer's wildcard ID
|
||||
matchType: 'wildcard',
|
||||
confidence: Math.min(provider.confidence, consumer.confidence),
|
||||
});
|
||||
matchedConsumerIds.add(`${consumer.repo}::${consumer.contractId}`);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
const remaining = unmatched.filter((c) => {
|
||||
if (c.role !== 'consumer' || !isGrpcWildcard(c.contractId)) return true;
|
||||
return !matchedConsumerIds.has(`${c.repo}::${c.contractId}`);
|
||||
});
|
||||
|
||||
return { matched, remaining };
|
||||
}
|
||||
@@ -0,0 +1,124 @@
|
||||
import type { CrossLink, CrossLinkEndpoint, StoredContract } from './types.js';
|
||||
|
||||
function contractKey(contract: StoredContract): string {
|
||||
return [contract.repo, contract.contractId, contract.role, contract.symbolRef.filePath].join(
|
||||
'\0',
|
||||
);
|
||||
}
|
||||
|
||||
function endpointKey(endpoint: CrossLinkEndpoint): string {
|
||||
return [
|
||||
endpoint.repo,
|
||||
endpoint.service ?? '',
|
||||
endpoint.symbolRef.filePath,
|
||||
endpoint.symbolRef.name,
|
||||
].join('\0');
|
||||
}
|
||||
|
||||
/**
|
||||
* Score a contract by how much information it carries, so `dedupeContracts`
|
||||
* can prefer the "richer" record when two contracts collide on the same
|
||||
* `(repo, contractId, role, filePath)` key.
|
||||
*
|
||||
* Weights express a priority ordering, not calibrated probabilities:
|
||||
* +3 — `symbolUid` resolved (tier 1 of the downstream lookup — highest
|
||||
* signal because it's the strongest anchor for cross-impact traversal
|
||||
* and the only one that's robust to renames)
|
||||
* +2 — any of `filePath`, `symbolRef.name`, or `symbolName` that's more
|
||||
* specific than the contractId itself (tier 2 signal — resolves
|
||||
* uniquely in most cases and survives across syncs)
|
||||
* +1 — `service` tag (monorepo attribution — useful but not sufficient
|
||||
* on its own) or non-manifest origin (auto-extracted contracts are
|
||||
* preferred over manifest-declared synthetic ones because the former
|
||||
* are grounded in real source code)
|
||||
*
|
||||
* The absolute numbers don't matter, only their relative ordering.
|
||||
*/
|
||||
function contractRichness(contract: StoredContract): number {
|
||||
let score = 0;
|
||||
if (contract.symbolUid) score += 3;
|
||||
if (contract.symbolRef.filePath) score += 2;
|
||||
if (contract.symbolRef.name && contract.symbolRef.name !== contract.contractId) score += 2;
|
||||
if (contract.symbolName && contract.symbolName !== contract.contractId) score += 2;
|
||||
if (contract.service) score += 1;
|
||||
if (contract.meta.source !== 'manifest') score += 1;
|
||||
return score;
|
||||
}
|
||||
|
||||
function mergeContracts(existing: StoredContract, incoming: StoredContract): StoredContract {
|
||||
const [primary, secondary] =
|
||||
contractRichness(incoming) > contractRichness(existing)
|
||||
? [incoming, existing]
|
||||
: [existing, incoming];
|
||||
const symbolRefName = primary.symbolRef.name || secondary.symbolRef.name;
|
||||
return {
|
||||
...secondary,
|
||||
...primary,
|
||||
symbolUid: primary.symbolUid || secondary.symbolUid,
|
||||
symbolRef: {
|
||||
filePath: primary.symbolRef.filePath || secondary.symbolRef.filePath,
|
||||
name: symbolRefName,
|
||||
},
|
||||
symbolName: primary.symbolName || secondary.symbolName || symbolRefName,
|
||||
confidence: Math.max(existing.confidence, incoming.confidence),
|
||||
service: primary.service ?? secondary.service,
|
||||
meta: { ...secondary.meta, ...primary.meta },
|
||||
};
|
||||
}
|
||||
|
||||
function mergeEndpoints(
|
||||
existing: CrossLinkEndpoint,
|
||||
incoming: CrossLinkEndpoint,
|
||||
): CrossLinkEndpoint {
|
||||
return {
|
||||
repo: existing.repo,
|
||||
service: existing.service ?? incoming.service,
|
||||
symbolUid: existing.symbolUid || incoming.symbolUid,
|
||||
symbolRef: {
|
||||
filePath: existing.symbolRef.filePath || incoming.symbolRef.filePath,
|
||||
name: existing.symbolRef.name || incoming.symbolRef.name,
|
||||
},
|
||||
};
|
||||
}
|
||||
|
||||
function crossLinkKey(link: CrossLink): string {
|
||||
return [
|
||||
link.type,
|
||||
link.contractId,
|
||||
link.matchType,
|
||||
endpointKey(link.from),
|
||||
endpointKey(link.to),
|
||||
].join('\0');
|
||||
}
|
||||
|
||||
export function dedupeContracts(items: StoredContract[]): StoredContract[] {
|
||||
const deduped = new Map<string, StoredContract>();
|
||||
for (const contract of items) {
|
||||
const key = contractKey(contract);
|
||||
const existing = deduped.get(key);
|
||||
deduped.set(key, existing ? mergeContracts(existing, contract) : contract);
|
||||
}
|
||||
return [...deduped.values()];
|
||||
}
|
||||
|
||||
export function dedupeCrossLinks(items: CrossLink[]): CrossLink[] {
|
||||
const deduped = new Map<string, CrossLink>();
|
||||
for (const link of items) {
|
||||
const key = crossLinkKey(link);
|
||||
const existing = deduped.get(key);
|
||||
if (!existing) {
|
||||
deduped.set(key, link);
|
||||
continue;
|
||||
}
|
||||
const keepIncoming = link.confidence > existing.confidence;
|
||||
const primary = keepIncoming ? link : existing;
|
||||
const secondary = keepIncoming ? existing : link;
|
||||
deduped.set(key, {
|
||||
...primary,
|
||||
confidence: Math.max(existing.confidence, link.confidence),
|
||||
from: mergeEndpoints(primary.from, secondary.from),
|
||||
to: mergeEndpoints(primary.to, secondary.to),
|
||||
});
|
||||
}
|
||||
return [...deduped.values()];
|
||||
}
|
||||
@@ -0,0 +1,177 @@
|
||||
import fs from 'node:fs/promises';
|
||||
import path from 'node:path';
|
||||
|
||||
export interface ServiceBoundary {
|
||||
servicePath: string;
|
||||
serviceName: string;
|
||||
markers: string[];
|
||||
confidence: number;
|
||||
}
|
||||
|
||||
const SERVICE_MARKERS = [
|
||||
'package.json',
|
||||
'go.mod',
|
||||
'Dockerfile',
|
||||
'pom.xml',
|
||||
'build.gradle',
|
||||
'build.gradle.kts',
|
||||
'Cargo.toml',
|
||||
'pyproject.toml',
|
||||
'requirements.txt',
|
||||
'mix.exs',
|
||||
] as const;
|
||||
|
||||
const SOURCE_EXTENSIONS = new Set([
|
||||
'.ts',
|
||||
'.tsx',
|
||||
'.js',
|
||||
'.jsx',
|
||||
'.mjs',
|
||||
'.cjs',
|
||||
'.go',
|
||||
'.java',
|
||||
'.kt',
|
||||
'.kts',
|
||||
'.py',
|
||||
'.pyi',
|
||||
'.rs',
|
||||
'.c',
|
||||
'.cpp',
|
||||
'.h',
|
||||
'.hpp',
|
||||
'.cs',
|
||||
'.rb',
|
||||
'.php',
|
||||
'.swift',
|
||||
'.dart',
|
||||
'.ex',
|
||||
'.exs',
|
||||
'.erl',
|
||||
'.proto',
|
||||
]);
|
||||
|
||||
const EXCLUDED_DIRS = new Set([
|
||||
'node_modules',
|
||||
'vendor',
|
||||
'target',
|
||||
'build',
|
||||
'dist',
|
||||
'__pycache__',
|
||||
'.venv',
|
||||
'venv',
|
||||
'.tox',
|
||||
'.mypy_cache',
|
||||
'.gradle',
|
||||
'.mvn',
|
||||
'out',
|
||||
'bin',
|
||||
]);
|
||||
|
||||
export async function detectServiceBoundaries(repoPath: string): Promise<ServiceBoundary[]> {
|
||||
const boundaries: ServiceBoundary[] = [];
|
||||
await walkForBoundaries(repoPath, repoPath, boundaries);
|
||||
return boundaries;
|
||||
}
|
||||
|
||||
async function walkForBoundaries(
|
||||
dir: string,
|
||||
repoRoot: string,
|
||||
results: ServiceBoundary[],
|
||||
): Promise<void> {
|
||||
let entries: import('node:fs').Dirent[];
|
||||
try {
|
||||
entries = await fs.readdir(dir, { withFileTypes: true });
|
||||
} catch {
|
||||
return;
|
||||
}
|
||||
|
||||
const isRoot = path.resolve(dir) === path.resolve(repoRoot);
|
||||
|
||||
const foundMarkers: string[] = [];
|
||||
let hasSourceFiles = false;
|
||||
const subdirs: string[] = [];
|
||||
|
||||
for (const entry of entries) {
|
||||
if (entry.name.startsWith('.') || EXCLUDED_DIRS.has(entry.name)) continue;
|
||||
|
||||
if (entry.isDirectory()) {
|
||||
subdirs.push(path.join(dir, entry.name));
|
||||
} else if (entry.isFile()) {
|
||||
if (SERVICE_MARKERS.includes(entry.name as (typeof SERVICE_MARKERS)[number])) {
|
||||
foundMarkers.push(entry.name);
|
||||
}
|
||||
const ext = path.extname(entry.name).toLowerCase();
|
||||
if (SOURCE_EXTENSIONS.has(ext)) {
|
||||
hasSourceFiles = true;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Check subdirectories for source files if not found at this level
|
||||
if (!hasSourceFiles && foundMarkers.length > 0) {
|
||||
hasSourceFiles = await hasSourceFilesInSubdirs(subdirs);
|
||||
}
|
||||
|
||||
if (!isRoot && foundMarkers.length >= 1 && hasSourceFiles) {
|
||||
const relativePath = path.relative(repoRoot, dir).replace(/\\/g, '/');
|
||||
const serviceName = path.basename(dir);
|
||||
const confidence = computeConfidence(foundMarkers.length);
|
||||
|
||||
results.push({
|
||||
servicePath: relativePath,
|
||||
serviceName,
|
||||
markers: foundMarkers,
|
||||
confidence,
|
||||
});
|
||||
}
|
||||
|
||||
// Recurse into subdirectories
|
||||
for (const subdir of subdirs) {
|
||||
await walkForBoundaries(subdir, repoRoot, results);
|
||||
}
|
||||
}
|
||||
|
||||
async function hasSourceFilesInSubdirs(subdirs: string[]): Promise<boolean> {
|
||||
for (const subdir of subdirs) {
|
||||
let entries: import('node:fs').Dirent[];
|
||||
try {
|
||||
entries = await fs.readdir(subdir, { withFileTypes: true });
|
||||
} catch {
|
||||
continue;
|
||||
}
|
||||
for (const entry of entries) {
|
||||
if (entry.isFile()) {
|
||||
const ext = path.extname(entry.name).toLowerCase();
|
||||
if (SOURCE_EXTENSIONS.has(ext)) return true;
|
||||
}
|
||||
if (entry.isDirectory() && !entry.name.startsWith('.') && !EXCLUDED_DIRS.has(entry.name)) {
|
||||
const deeper = await hasSourceFilesInSubdirs([path.join(subdir, entry.name)]);
|
||||
if (deeper) return true;
|
||||
}
|
||||
}
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
function computeConfidence(markerCount: number): number {
|
||||
if (markerCount >= 3) return 1.0;
|
||||
if (markerCount === 2) return 0.9;
|
||||
return 0.75;
|
||||
}
|
||||
|
||||
export function assignService(filePath: string, boundaries: ServiceBoundary[]): string | undefined {
|
||||
const normalized = filePath.replace(/\\/g, '/');
|
||||
|
||||
let bestMatch: ServiceBoundary | undefined;
|
||||
let bestLength = 0;
|
||||
|
||||
for (const boundary of boundaries) {
|
||||
const prefix = boundary.servicePath + '/';
|
||||
if (normalized.startsWith(prefix) && boundary.servicePath.length > bestLength) {
|
||||
bestMatch = boundary;
|
||||
bestLength = boundary.servicePath.length;
|
||||
}
|
||||
}
|
||||
|
||||
return bestMatch?.servicePath;
|
||||
}
|
||||
@@ -0,0 +1,223 @@
|
||||
/**
|
||||
* Group orchestration shared by MCP (LocalBackend) and CLI.
|
||||
* DB access is injected via GroupToolPort so this module stays free of LocalBackend private API.
|
||||
*/
|
||||
|
||||
import { checkStaleness } from '../git-staleness.js';
|
||||
import { loadGroupConfig } from './config-parser.js';
|
||||
import { getDefaultGitnexusDir, getGroupDir, listGroups, readContractRegistry } from './storage.js';
|
||||
import { syncGroup } from './sync.js';
|
||||
|
||||
export interface GroupRepoHandle {
|
||||
id: string;
|
||||
name: string;
|
||||
repoPath: string;
|
||||
storagePath: string;
|
||||
indexedAt?: string;
|
||||
lastCommit?: string;
|
||||
}
|
||||
|
||||
export interface GroupToolPort {
|
||||
resolveRepo(repoParam?: string): Promise<GroupRepoHandle>;
|
||||
impact(
|
||||
repo: GroupRepoHandle,
|
||||
params: {
|
||||
target: string;
|
||||
direction: 'upstream' | 'downstream';
|
||||
maxDepth?: number;
|
||||
relationTypes?: string[];
|
||||
includeTests?: boolean;
|
||||
minConfidence?: number;
|
||||
},
|
||||
): Promise<unknown>;
|
||||
query(
|
||||
repo: GroupRepoHandle,
|
||||
params: {
|
||||
query: string;
|
||||
task_context?: string;
|
||||
goal?: string;
|
||||
limit?: number;
|
||||
max_symbols?: number;
|
||||
include_content?: boolean;
|
||||
},
|
||||
): Promise<unknown>;
|
||||
impactByUid(
|
||||
repoId: string,
|
||||
uid: string,
|
||||
direction: string,
|
||||
opts: {
|
||||
maxDepth: number;
|
||||
relationTypes: string[];
|
||||
minConfidence: number;
|
||||
includeTests: boolean;
|
||||
},
|
||||
): Promise<unknown | null>;
|
||||
}
|
||||
|
||||
function repoInSubgroup(repoPath: string, subgroup?: string): boolean {
|
||||
if (!subgroup?.trim()) return true;
|
||||
const s = subgroup.replace(/\/+$/, '');
|
||||
return repoPath === s || repoPath.startsWith(`${s}/`);
|
||||
}
|
||||
|
||||
export class GroupService {
|
||||
constructor(private readonly port: GroupToolPort) {}
|
||||
|
||||
async groupList(params: Record<string, unknown>): Promise<unknown> {
|
||||
const name = typeof params.name === 'string' ? params.name.trim() : '';
|
||||
if (!name) {
|
||||
const groups = await listGroups();
|
||||
return { groups };
|
||||
}
|
||||
const groupDir = getGroupDir(getDefaultGitnexusDir(), name);
|
||||
const config = await loadGroupConfig(groupDir);
|
||||
return {
|
||||
name: config.name,
|
||||
description: config.description,
|
||||
repos: config.repos,
|
||||
links: config.links,
|
||||
};
|
||||
}
|
||||
|
||||
async groupSync(params: Record<string, unknown>): Promise<unknown> {
|
||||
const name = String(params.name ?? '').trim();
|
||||
if (!name) return { error: 'name is required' };
|
||||
const groupDir = getGroupDir(getDefaultGitnexusDir(), name);
|
||||
const config = await loadGroupConfig(groupDir);
|
||||
const result = await syncGroup(config, {
|
||||
groupDir,
|
||||
exactOnly: Boolean(params.exactOnly),
|
||||
skipEmbeddings: Boolean(params.skipEmbeddings),
|
||||
allowStale: Boolean(params.allowStale),
|
||||
verbose: Boolean(params.verbose),
|
||||
});
|
||||
return {
|
||||
contracts: result.contracts.length,
|
||||
crossLinks: result.crossLinks.length,
|
||||
unmatched: result.unmatched.length,
|
||||
missingRepos: result.missingRepos,
|
||||
};
|
||||
}
|
||||
|
||||
async groupContracts(params: Record<string, unknown>): Promise<unknown> {
|
||||
const name = String(params.name ?? '').trim();
|
||||
if (!name) return { error: 'name is required' };
|
||||
const groupDir = getGroupDir(getDefaultGitnexusDir(), name);
|
||||
const registry = await readContractRegistry(groupDir);
|
||||
if (!registry) {
|
||||
return { error: `No contracts.json for group "${name}". Run group_sync first.` };
|
||||
}
|
||||
let contracts = registry.contracts;
|
||||
if (params.type) contracts = contracts.filter((c) => c.type === params.type);
|
||||
if (params.repo) contracts = contracts.filter((c) => c.repo === params.repo);
|
||||
if (params.unmatchedOnly) {
|
||||
const matchedIds = new Set(
|
||||
registry.crossLinks.flatMap((l) => [
|
||||
`${l.from.repo}::${l.contractId}`,
|
||||
`${l.to.repo}::${l.contractId}`,
|
||||
]),
|
||||
);
|
||||
contracts = contracts.filter((c) => !matchedIds.has(`${c.repo}::${c.contractId}`));
|
||||
}
|
||||
return { contracts, crossLinks: registry.crossLinks };
|
||||
}
|
||||
|
||||
async groupQuery(params: Record<string, unknown>): Promise<unknown> {
|
||||
const name = String(params.name ?? '').trim();
|
||||
const queryText = String(params.query ?? '').trim();
|
||||
if (!name || !queryText) return { error: 'name and query are required' };
|
||||
|
||||
const limit = typeof params.limit === 'number' && params.limit > 0 ? params.limit : 5;
|
||||
const subgroup = typeof params.subgroup === 'string' ? params.subgroup : undefined;
|
||||
const groupDir = getGroupDir(getDefaultGitnexusDir(), name);
|
||||
const config = await loadGroupConfig(groupDir);
|
||||
|
||||
const perRepo: Array<{ repo: string; score: number; processes: unknown[] }> = [];
|
||||
for (const [repoPath, registryName] of Object.entries(config.repos)) {
|
||||
if (!repoInSubgroup(repoPath, subgroup)) continue;
|
||||
try {
|
||||
const repoObj = await this.port.resolveRepo(registryName);
|
||||
const queryResult = (await this.port.query(repoObj, {
|
||||
query: queryText,
|
||||
limit,
|
||||
max_symbols: 10,
|
||||
include_content: false,
|
||||
})) as { processes?: Array<Record<string, unknown>> };
|
||||
const processes = queryResult.processes || [];
|
||||
const scored = processes.map((p, idx) => ({
|
||||
...p,
|
||||
_rrf_score: 1 / (idx + 1 + 60),
|
||||
_repo: repoPath,
|
||||
}));
|
||||
perRepo.push({ repo: repoPath, score: 0, processes: scored });
|
||||
} catch {
|
||||
perRepo.push({ repo: repoPath, score: 0, processes: [] });
|
||||
}
|
||||
}
|
||||
|
||||
const allProcesses = perRepo.flatMap((r) => r.processes as Array<Record<string, unknown>>);
|
||||
allProcesses.sort((a, b) => (b._rrf_score as number) - (a._rrf_score as number));
|
||||
const topN = allProcesses.slice(0, limit);
|
||||
|
||||
return {
|
||||
group: name,
|
||||
query: queryText,
|
||||
results: topN,
|
||||
per_repo: perRepo.map((r) => ({ repo: r.repo, count: r.processes.length })),
|
||||
};
|
||||
}
|
||||
|
||||
async groupStatus(params: Record<string, unknown>): Promise<unknown> {
|
||||
const name = String(params.name ?? '').trim();
|
||||
if (!name) return { error: 'name is required' };
|
||||
const groupDir = getGroupDir(getDefaultGitnexusDir(), name);
|
||||
const config = await loadGroupConfig(groupDir);
|
||||
const registry = await readContractRegistry(groupDir);
|
||||
|
||||
const repoStatuses: Record<
|
||||
string,
|
||||
{
|
||||
indexStale: boolean;
|
||||
contractsStale: boolean;
|
||||
missing: boolean;
|
||||
commitsBehind?: number;
|
||||
}
|
||||
> = {};
|
||||
|
||||
const fsp = await import('node:fs/promises');
|
||||
const pathMod = await import('node:path');
|
||||
|
||||
for (const [repoPath, registryName] of Object.entries(config.repos)) {
|
||||
try {
|
||||
const repoObj = await this.port.resolveRepo(registryName);
|
||||
const metaPath = pathMod.join(repoObj.storagePath, 'meta.json');
|
||||
const metaRaw = await fsp.readFile(metaPath, 'utf-8').catch(() => '{}');
|
||||
const meta = JSON.parse(metaRaw) as { lastCommit?: string; indexedAt?: string };
|
||||
|
||||
const staleness = meta.lastCommit
|
||||
? checkStaleness(repoObj.repoPath, meta.lastCommit)
|
||||
: { isStale: true, commitsBehind: -1 };
|
||||
|
||||
const snapshot = registry?.repoSnapshots[repoPath];
|
||||
const contractsStale =
|
||||
snapshot && meta.indexedAt ? snapshot.indexedAt !== meta.indexedAt : !snapshot;
|
||||
|
||||
repoStatuses[repoPath] = {
|
||||
indexStale: staleness.isStale,
|
||||
contractsStale: Boolean(contractsStale),
|
||||
missing: false,
|
||||
commitsBehind: staleness.commitsBehind,
|
||||
};
|
||||
} catch {
|
||||
repoStatuses[repoPath] = { indexStale: false, contractsStale: false, missing: true };
|
||||
}
|
||||
}
|
||||
|
||||
return {
|
||||
group: name,
|
||||
lastSync: registry?.generatedAt || null,
|
||||
missingRepos: registry?.missingRepos || [],
|
||||
repos: repoStatuses,
|
||||
};
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,109 @@
|
||||
import * as fs from 'node:fs';
|
||||
import * as fsp from 'node:fs/promises';
|
||||
import * as path from 'node:path';
|
||||
import * as os from 'node:os';
|
||||
import type { ContractRegistry } from './types.js';
|
||||
|
||||
const CONTRACTS_FILE = 'contracts.json';
|
||||
|
||||
export function getDefaultGitnexusDir(): string {
|
||||
return process.env.GITNEXUS_HOME || path.join(os.homedir(), '.gitnexus');
|
||||
}
|
||||
|
||||
export function getGroupsBaseDir(gitnexusDir?: string): string {
|
||||
return path.join(gitnexusDir || getDefaultGitnexusDir(), 'groups');
|
||||
}
|
||||
|
||||
const GROUP_NAME_RE = /^[a-zA-Z0-9][a-zA-Z0-9_-]*$/;
|
||||
|
||||
export function validateGroupName(name: string): void {
|
||||
if (!GROUP_NAME_RE.test(name)) {
|
||||
throw new Error(
|
||||
`Invalid group name "${name}". Names must start with a letter or digit and contain only [a-zA-Z0-9_-].`,
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
export function getGroupDir(gitnexusDir: string, groupName: string): string {
|
||||
validateGroupName(groupName);
|
||||
return path.join(gitnexusDir, 'groups', groupName);
|
||||
}
|
||||
|
||||
export async function writeContractRegistry(
|
||||
groupDir: string,
|
||||
registry: ContractRegistry,
|
||||
): Promise<void> {
|
||||
const targetPath = path.join(groupDir, CONTRACTS_FILE);
|
||||
const tmpPath = `${targetPath}.tmp.${Date.now()}`;
|
||||
|
||||
await fsp.writeFile(tmpPath, JSON.stringify(registry, null, 2), 'utf-8');
|
||||
await fsp.rename(tmpPath, targetPath);
|
||||
}
|
||||
|
||||
export async function readContractRegistry(groupDir: string): Promise<ContractRegistry | null> {
|
||||
const filePath = path.join(groupDir, CONTRACTS_FILE);
|
||||
try {
|
||||
const content = await fsp.readFile(filePath, 'utf-8');
|
||||
return JSON.parse(content) as ContractRegistry;
|
||||
} catch (err: unknown) {
|
||||
if ((err as NodeJS.ErrnoException).code === 'ENOENT') return null;
|
||||
throw err;
|
||||
}
|
||||
}
|
||||
|
||||
export async function listGroups(gitnexusDir?: string): Promise<string[]> {
|
||||
const groupsDir = getGroupsBaseDir(gitnexusDir);
|
||||
try {
|
||||
const entries = await fsp.readdir(groupsDir, { withFileTypes: true });
|
||||
const names: string[] = [];
|
||||
for (const entry of entries) {
|
||||
if (entry.isDirectory()) {
|
||||
const yamlPath = path.join(groupsDir, entry.name, 'group.yaml');
|
||||
if (fs.existsSync(yamlPath)) {
|
||||
names.push(entry.name);
|
||||
}
|
||||
}
|
||||
}
|
||||
return names;
|
||||
} catch (err: unknown) {
|
||||
if ((err as NodeJS.ErrnoException).code === 'ENOENT') return [];
|
||||
throw err;
|
||||
}
|
||||
}
|
||||
|
||||
export async function createGroupDir(
|
||||
gitnexusDir: string,
|
||||
groupName: string,
|
||||
force: boolean = false,
|
||||
): Promise<string> {
|
||||
const groupDir = getGroupDir(gitnexusDir, groupName);
|
||||
if (fs.existsSync(path.join(groupDir, 'group.yaml')) && !force) {
|
||||
throw new Error(`Group "${groupName}" already exists. Use --force to overwrite.`);
|
||||
}
|
||||
await fsp.mkdir(groupDir, { recursive: true });
|
||||
|
||||
const template = `version: 1
|
||||
name: ${groupName}
|
||||
description: ""
|
||||
|
||||
repos: {}
|
||||
|
||||
links: []
|
||||
|
||||
packages: {}
|
||||
|
||||
detect:
|
||||
http: true
|
||||
grpc: true
|
||||
topics: true
|
||||
shared_libs: true
|
||||
embedding_fallback: true
|
||||
|
||||
matching:
|
||||
bm25_threshold: 0.7
|
||||
embedding_threshold: 0.65
|
||||
max_candidates_per_step: 3
|
||||
`;
|
||||
await fsp.writeFile(path.join(groupDir, 'group.yaml'), template, 'utf-8');
|
||||
return groupDir;
|
||||
}
|
||||
@@ -0,0 +1,185 @@
|
||||
import fs from 'node:fs/promises';
|
||||
import path from 'node:path';
|
||||
import { Buffer } from 'node:buffer';
|
||||
import { initLbug, closeLbug, executeParameterized } from '../lbug/pool-adapter.js';
|
||||
import { readRegistry, type RegistryEntry } from '../../storage/repo-manager.js';
|
||||
import type { GroupConfig, RepoHandle, RepoSnapshot, StoredContract, CrossLink } from './types.js';
|
||||
import { HttpRouteExtractor } from './extractors/http-route-extractor.js';
|
||||
import { GrpcExtractor } from './extractors/grpc-extractor.js';
|
||||
import { TopicExtractor } from './extractors/topic-extractor.js';
|
||||
import { runExactMatch } from './matching.js';
|
||||
import { detectServiceBoundaries, assignService } from './service-boundary-detector.js';
|
||||
import type { CypherExecutor } from './contract-extractor.js';
|
||||
import { writeContractRegistry } from './storage.js';
|
||||
import type { ContractRegistry } from './types.js';
|
||||
|
||||
export interface SyncOptions {
|
||||
extractorOverride?:
|
||||
| ((repo: RepoHandle) => Promise<StoredContract[]>)
|
||||
| (() => Promise<StoredContract[]>);
|
||||
resolveRepoHandle?: (registryName: string, groupPath: string) => Promise<RepoHandle | null>;
|
||||
skipWrite?: boolean;
|
||||
groupDir?: string;
|
||||
allowStale?: boolean;
|
||||
verbose?: boolean;
|
||||
exactOnly?: boolean;
|
||||
skipEmbeddings?: boolean;
|
||||
}
|
||||
|
||||
export interface SyncResult {
|
||||
contracts: StoredContract[];
|
||||
crossLinks: CrossLink[];
|
||||
unmatched: StoredContract[];
|
||||
missingRepos: string[];
|
||||
repoSnapshots: Record<string, RepoSnapshot>;
|
||||
}
|
||||
|
||||
export function stableRepoPoolId(entry: RegistryEntry, allEntries: RegistryEntry[]): string {
|
||||
const base = entry.name.toLowerCase();
|
||||
const resolved = path.resolve(entry.path);
|
||||
for (const other of allEntries) {
|
||||
if (other.name.toLowerCase() === base && path.resolve(other.path) !== resolved) {
|
||||
const hash = Buffer.from(entry.path).toString('base64url').slice(0, 6);
|
||||
return `${base}-${hash}`;
|
||||
}
|
||||
}
|
||||
return base;
|
||||
}
|
||||
|
||||
function defaultResolveHandle(allEntries: RegistryEntry[]) {
|
||||
return async (registryName: string, groupPath: string): Promise<RepoHandle | null> => {
|
||||
const e = allEntries.find((en) => en.name === registryName);
|
||||
if (!e) return null;
|
||||
const poolId = stableRepoPoolId(e, allEntries);
|
||||
return {
|
||||
id: poolId,
|
||||
path: groupPath,
|
||||
repoPath: e.path,
|
||||
storagePath: e.storagePath,
|
||||
};
|
||||
};
|
||||
}
|
||||
|
||||
export async function syncGroup(config: GroupConfig, opts?: SyncOptions): Promise<SyncResult> {
|
||||
const missingRepos: string[] = [];
|
||||
const repoSnapshots: Record<string, RepoSnapshot> = {};
|
||||
let autoContracts: StoredContract[] = [];
|
||||
let dbExecutors: Map<string, CypherExecutor> | undefined;
|
||||
|
||||
const eo = opts?.extractorOverride;
|
||||
if (eo && eo.length === 0) {
|
||||
autoContracts = await (eo as () => Promise<StoredContract[]>)();
|
||||
} else {
|
||||
const entries = await readRegistry();
|
||||
const resolve = opts?.resolveRepoHandle ?? defaultResolveHandle(entries);
|
||||
const httpEx = new HttpRouteExtractor();
|
||||
const grpcEx = new GrpcExtractor();
|
||||
const topicEx = new TopicExtractor();
|
||||
dbExecutors = new Map<string, CypherExecutor>();
|
||||
const openPoolIds: string[] = [];
|
||||
|
||||
try {
|
||||
for (const [groupPath, regName] of Object.entries(config.repos)) {
|
||||
const handle = await resolve(regName, groupPath);
|
||||
if (!handle) {
|
||||
missingRepos.push(groupPath);
|
||||
continue;
|
||||
}
|
||||
|
||||
const poolId = handle.id;
|
||||
const lbugPath = path.join(handle.storagePath, 'lbug');
|
||||
try {
|
||||
await initLbug(poolId, lbugPath);
|
||||
openPoolIds.push(poolId);
|
||||
|
||||
const executor: CypherExecutor = (query, params) =>
|
||||
executeParameterized(poolId, query, params ?? {});
|
||||
|
||||
dbExecutors.set(groupPath, executor);
|
||||
|
||||
const boundaries = await detectServiceBoundaries(handle.repoPath);
|
||||
|
||||
if (config.detect.http) {
|
||||
const extracted = await httpEx.extract(executor, handle.repoPath, handle);
|
||||
for (const c of extracted) {
|
||||
autoContracts.push({
|
||||
...c,
|
||||
repo: groupPath,
|
||||
service: assignService(c.symbolRef.filePath, boundaries),
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
if (config.detect.grpc) {
|
||||
const extracted = await grpcEx.extract(executor, handle.repoPath, handle);
|
||||
for (const c of extracted) {
|
||||
autoContracts.push({
|
||||
...c,
|
||||
repo: groupPath,
|
||||
service: assignService(c.symbolRef.filePath, boundaries),
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
if (config.detect.topics) {
|
||||
const extracted = await topicEx.extract(executor, handle.repoPath, handle);
|
||||
for (const c of extracted) {
|
||||
autoContracts.push({
|
||||
...c,
|
||||
repo: groupPath,
|
||||
service: assignService(c.symbolRef.filePath, boundaries),
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
const metaPath = path.join(handle.storagePath, 'meta.json');
|
||||
try {
|
||||
const raw = await fs.readFile(metaPath, 'utf-8');
|
||||
const m = JSON.parse(raw) as { indexedAt?: string; lastCommit?: string };
|
||||
repoSnapshots[groupPath] = {
|
||||
indexedAt: m.indexedAt || '',
|
||||
lastCommit: m.lastCommit || '',
|
||||
};
|
||||
} catch {
|
||||
const e = entries.find((en) => en.name === regName);
|
||||
repoSnapshots[groupPath] = {
|
||||
indexedAt: e?.indexedAt || '',
|
||||
lastCommit: e?.lastCommit || '',
|
||||
};
|
||||
}
|
||||
} catch {
|
||||
missingRepos.push(groupPath);
|
||||
}
|
||||
}
|
||||
} finally {
|
||||
for (const id of [...new Set(openPoolIds)]) {
|
||||
await closeLbug(id).catch(() => {});
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
const { matched, unmatched } = runExactMatch(autoContracts);
|
||||
const crossLinks: CrossLink[] = matched;
|
||||
const allContracts: StoredContract[] = autoContracts;
|
||||
|
||||
const registry: ContractRegistry = {
|
||||
version: 1,
|
||||
generatedAt: new Date().toISOString(),
|
||||
repoSnapshots,
|
||||
missingRepos,
|
||||
contracts: allContracts,
|
||||
crossLinks,
|
||||
};
|
||||
|
||||
if (opts?.groupDir && !opts.skipWrite) {
|
||||
await writeContractRegistry(opts.groupDir, registry);
|
||||
}
|
||||
|
||||
return {
|
||||
contracts: allContracts,
|
||||
crossLinks,
|
||||
unmatched,
|
||||
missingRepos,
|
||||
repoSnapshots,
|
||||
};
|
||||
}
|
||||
@@ -0,0 +1,147 @@
|
||||
export type ContractType = 'http' | 'grpc' | 'topic' | 'lib' | 'custom';
|
||||
export type MatchType = 'exact' | 'manifest' | 'wildcard' | 'bm25' | 'embedding';
|
||||
export type ContractRole = 'provider' | 'consumer';
|
||||
|
||||
export interface GroupConfig {
|
||||
version: number;
|
||||
name: string;
|
||||
description: string;
|
||||
repos: Record<string, string>;
|
||||
links: GroupManifestLink[];
|
||||
packages: Record<string, Record<string, string>>;
|
||||
detect: DetectConfig;
|
||||
matching: MatchingConfig;
|
||||
}
|
||||
|
||||
export interface GroupManifestLink {
|
||||
from: string;
|
||||
to: string;
|
||||
type: ContractType;
|
||||
contract: string;
|
||||
role: ContractRole;
|
||||
}
|
||||
|
||||
export interface DetectConfig {
|
||||
http: boolean;
|
||||
grpc: boolean;
|
||||
topics: boolean;
|
||||
shared_libs: boolean;
|
||||
embedding_fallback: boolean;
|
||||
}
|
||||
|
||||
export interface MatchingConfig {
|
||||
bm25_threshold: number;
|
||||
embedding_threshold: number;
|
||||
max_candidates_per_step: number;
|
||||
}
|
||||
|
||||
export interface SymbolRef {
|
||||
filePath: string;
|
||||
name: string;
|
||||
}
|
||||
|
||||
export interface ExtractedContract {
|
||||
contractId: string;
|
||||
type: ContractType;
|
||||
role: ContractRole;
|
||||
symbolUid: string;
|
||||
symbolRef: SymbolRef;
|
||||
symbolName: string;
|
||||
confidence: number;
|
||||
meta: Record<string, unknown>;
|
||||
/** Service boundary within a monorepo (relative path from repo root, e.g. "services/auth"). */
|
||||
service?: string;
|
||||
}
|
||||
|
||||
export interface CrossLinkEndpoint {
|
||||
repo: string;
|
||||
/** Service boundary within a monorepo (relative path from repo root). */
|
||||
service?: string;
|
||||
symbolUid: string;
|
||||
symbolRef: SymbolRef;
|
||||
}
|
||||
|
||||
export interface CrossLink {
|
||||
from: CrossLinkEndpoint;
|
||||
to: CrossLinkEndpoint;
|
||||
type: ContractType;
|
||||
contractId: string;
|
||||
matchType: MatchType;
|
||||
confidence: number;
|
||||
}
|
||||
|
||||
export interface RepoSnapshot {
|
||||
indexedAt: string;
|
||||
lastCommit: string;
|
||||
}
|
||||
|
||||
export interface ContractRegistry {
|
||||
version: number;
|
||||
generatedAt: string;
|
||||
repoSnapshots: Record<string, RepoSnapshot>;
|
||||
missingRepos: string[];
|
||||
contracts: StoredContract[];
|
||||
crossLinks: CrossLink[];
|
||||
}
|
||||
|
||||
export interface StoredContract extends ExtractedContract {
|
||||
repo: string;
|
||||
}
|
||||
|
||||
/** Repo within a group (group path + paths; name collision with MCP RepoHandle — import from group/types only). */
|
||||
export interface RepoHandle {
|
||||
id: string;
|
||||
path: string;
|
||||
repoPath: string;
|
||||
storagePath: string;
|
||||
}
|
||||
|
||||
export interface GroupImpactResult {
|
||||
local: unknown;
|
||||
group: string;
|
||||
cross: CrossRepoImpact[];
|
||||
outOfScope: OutOfScopeLink[];
|
||||
truncated: boolean;
|
||||
truncatedRepos: string[];
|
||||
summary: {
|
||||
direct: number;
|
||||
processes_affected: number;
|
||||
modules_affected: number;
|
||||
cross_repo_hits: number;
|
||||
};
|
||||
risk: string;
|
||||
}
|
||||
|
||||
export interface CrossRepoImpact {
|
||||
repo: string;
|
||||
repo_path: string;
|
||||
contract: {
|
||||
id: string;
|
||||
type: ContractType;
|
||||
match_type: MatchType;
|
||||
confidence: number;
|
||||
};
|
||||
by_depth: Record<string, unknown[]>;
|
||||
affected_processes: string[];
|
||||
}
|
||||
|
||||
export interface OutOfScopeLink {
|
||||
from: string;
|
||||
to: string;
|
||||
contractId: string;
|
||||
confidence: number;
|
||||
}
|
||||
|
||||
/** Opaque handle to an open bridge LadybugDB. */
|
||||
export interface BridgeHandle {
|
||||
/** Internal — do not access directly. */
|
||||
readonly _db: unknown;
|
||||
readonly _conn: unknown;
|
||||
readonly groupDir: string;
|
||||
}
|
||||
|
||||
export interface BridgeMeta {
|
||||
version: number;
|
||||
generatedAt: string;
|
||||
missingRepos: string[];
|
||||
}
|
||||
@@ -0,0 +1,391 @@
|
||||
/**
|
||||
* BindingAccumulator — read-append-only accumulator that collects TypeEnv
|
||||
* bindings across files in the GitNexus analyzer pipeline.
|
||||
*
|
||||
* **Current behavior (both execution paths):** The accumulator carries only
|
||||
* file-scope (`scope = ''`) entries. Function-scope bindings are stripped
|
||||
* at both write sites:
|
||||
*
|
||||
* - **Worker path**: `parse-worker.ts` serializes only
|
||||
* `typeEnv.fileScope()` entries across the IPC boundary.
|
||||
* - **Sequential path**: `type-env.ts::flush()` iterates only the FILE_SCOPE
|
||||
* entry of the env map and writes `BindingEntry` records with
|
||||
* `scope: ''` hardcoded.
|
||||
*
|
||||
* The narrowing exists because function-scope bindings have zero downstream
|
||||
* consumers today and were previously costing ~4.9 MB of heap + IPC on
|
||||
* every pipeline run. See `type-env.ts::flush()` and the `FileScopeBindings`
|
||||
* JSDoc in `parse-worker.ts` for the paired Phase 9 reversion checklist.
|
||||
*
|
||||
* **Historical quality asymmetry (Phase 9 consideration):** Even though
|
||||
* both paths now carry only file-scope data, the two paths were built
|
||||
* under different resolution capabilities, and a future Phase 9 reverter
|
||||
* that widens them back to all scopes will inherit that asymmetry:
|
||||
*
|
||||
* - **Sequential path** had (and would regain) access to the full
|
||||
* `SymbolTable` and `importedBindings`, so its bindings benefit from
|
||||
* Tier 2 cross-file propagation.
|
||||
* - **Worker path** runs without `SymbolTable` / `importedBindings` and
|
||||
* can only produce Tier 0 (annotation-declared) and local Tier 1
|
||||
* (same-file constructor inference) bindings.
|
||||
*
|
||||
* Phase 9 consumers that trust every entry equally will silently produce
|
||||
* worse results for large repos (worker-dominant) than small ones
|
||||
* (sequential-dominant). If Phase 9 needs homogeneous quality, either
|
||||
* (a) tag entries with their tier at insert time so consumers can filter,
|
||||
* or (b) post-process worker-path entries through a follow-up resolution
|
||||
* pass after the main-thread `SymbolTable` is complete.
|
||||
*
|
||||
* **Lifecycle contract**: single-use — `append* → finalize → consume → dispose`.
|
||||
* After `dispose()` the accumulator is permanently dead: any mutating call
|
||||
* (`appendFile`) throws, and read methods return empty/undefined as if the
|
||||
* accumulator had never been appended to. The instance is not recyclable;
|
||||
* construct a new one for a new pipeline run. Finalization and disposal are
|
||||
* orthogonal state dimensions and may be invoked in either order.
|
||||
*/
|
||||
|
||||
export interface BindingEntry {
|
||||
readonly scope: string; // '' for file-level, 'funcName@startIndex' for function-local
|
||||
readonly varName: string;
|
||||
readonly typeName: string;
|
||||
}
|
||||
|
||||
/**
|
||||
* Minimal graph-node shape required by `enrichExportedTypeMap()`. Intentionally
|
||||
* narrower than the full `GraphNode` type in `graph/types.ts` so tests can
|
||||
* construct a minimal mock without depending on the full graph module, and
|
||||
* so the enrichment logic is a pure function over this contract.
|
||||
*
|
||||
* Matches the shape of the real `KnowledgeGraph` node's `properties.isExported`
|
||||
* access path — tests that use a different shape silently pass while
|
||||
* production fails.
|
||||
*/
|
||||
export interface EnrichmentGraphNode {
|
||||
readonly id: string;
|
||||
readonly properties?: { readonly isExported?: boolean } | undefined;
|
||||
}
|
||||
|
||||
/**
|
||||
* Minimal graph lookup interface used by `enrichExportedTypeMap()`.
|
||||
* Consumes only the method the enrichment loop actually calls.
|
||||
*/
|
||||
export interface EnrichmentGraphLookup {
|
||||
getNode(id: string): EnrichmentGraphNode | undefined;
|
||||
}
|
||||
|
||||
/**
|
||||
* Merge file-scope bindings from a (finalized) `BindingAccumulator` into an
|
||||
* `exportedTypeMap` for symbols whose graph nodes are marked as exported.
|
||||
*
|
||||
* This is the single source of truth for the worker-path ExportedTypeMap
|
||||
* enrichment loop. Previously the logic lived inline in `pipeline.ts` and
|
||||
* the test suite reimplemented it as a `runEnrichmentLoop` helper — a
|
||||
* drift-prone pattern that meant tests could pass while production regressed.
|
||||
* Extracting it here makes the production code call the same function the
|
||||
* tests call.
|
||||
*
|
||||
* **Node ID candidate order**: `Function:{filePath}:{name}` →
|
||||
* `Variable:{filePath}:{name}` → `Const:{filePath}:{name}`. First match wins.
|
||||
*
|
||||
* **Tier 0 priority**: if `exportedTypeMap` already has an entry for a
|
||||
* `(filePath, name)` pair, the accumulator entry does NOT overwrite it —
|
||||
* the SymbolTable tier-0 pass is authoritative. Without this guard, a
|
||||
* worker-path binding could clobber a higher-quality type from SymbolTable.
|
||||
*
|
||||
* **Finalize precondition**: the accumulator should be finalized before
|
||||
* calling this function. The lifecycle contract is
|
||||
* `append → finalize → enrich → dispose`. Finalization is not asserted
|
||||
* here (the test suite and pipeline both honor it separately), but any
|
||||
* append happening concurrently with this enrichment would be a lifecycle
|
||||
* bug at the caller level.
|
||||
*
|
||||
* @returns The number of new entries written into `exportedTypeMap`
|
||||
* (0 on empty accumulator or when every candidate was filtered
|
||||
* out by the export check or the Tier 0 guard).
|
||||
*/
|
||||
export function enrichExportedTypeMap(
|
||||
bindingAccumulator: BindingAccumulator,
|
||||
graph: EnrichmentGraphLookup,
|
||||
exportedTypeMap: Map<string, Map<string, string>>,
|
||||
): number {
|
||||
if (bindingAccumulator.fileCount === 0) return 0;
|
||||
let enriched = 0;
|
||||
for (const filePath of bindingAccumulator.files()) {
|
||||
for (const [name, type] of bindingAccumulator.fileScopeEntries(filePath)) {
|
||||
// Three-candidate-ID lookup mirrors the sequential-path export check
|
||||
// in `collectExportedBindings()` (call-processor.ts).
|
||||
const functionNodeId = `Function:${filePath}:${name}`;
|
||||
const variableNodeId = `Variable:${filePath}:${name}`;
|
||||
const constNodeId = `Const:${filePath}:${name}`;
|
||||
const node =
|
||||
graph.getNode(functionNodeId) ??
|
||||
graph.getNode(variableNodeId) ??
|
||||
graph.getNode(constNodeId);
|
||||
if (!node?.properties?.isExported) continue;
|
||||
|
||||
let fileExports = exportedTypeMap.get(filePath);
|
||||
if (!fileExports) {
|
||||
fileExports = new Map();
|
||||
exportedTypeMap.set(filePath, fileExports);
|
||||
}
|
||||
// Tier 0 priority: SymbolTable-populated entries are authoritative.
|
||||
if (!fileExports.has(name)) {
|
||||
fileExports.set(name, type);
|
||||
enriched++;
|
||||
}
|
||||
}
|
||||
}
|
||||
return enriched;
|
||||
}
|
||||
|
||||
const ENTRY_OVERHEAD = 64; // bytes per entry (object overhead + property refs)
|
||||
const MAP_ENTRY_OVERHEAD = 80; // bytes per file entry in the map
|
||||
|
||||
export class BindingAccumulator {
|
||||
// Storage is split into two parallel maps so file-scope reads are fast.
|
||||
// - _allByFile holds every BindingEntry (used by getFile, memory estimate).
|
||||
// - _fileScopeByFile is a nested Map<filePath, Map<varName, typeName>> for
|
||||
// O(1) point-lookup via fileScopeGet(). For iteration-based consumers
|
||||
// (enrichExportedTypeMap), fileScopeEntries() iterates the inner Map.
|
||||
// Both maps carry the same key set modulo the `scope === ''` precondition:
|
||||
// _allByFile has a key as soon as any entry is appended; _fileScopeByFile
|
||||
// only has a key once a file-scope entry arrives. Code that iterates via
|
||||
// files() uses _allByFile so files with only function-scope entries
|
||||
// remain visible.
|
||||
//
|
||||
// Note: Map.set semantics mean a duplicate varName for the same file
|
||||
// overwrites the previous value (last-write-wins). This is the correct
|
||||
// behavior — duplicate top-level bindings in the same file shouldn't
|
||||
// happen in well-formed source, and if they do the last declaration
|
||||
// is typically the one the compiler sees.
|
||||
private readonly _allByFile = new Map<string, BindingEntry[]>();
|
||||
private readonly _fileScopeByFile = new Map<string, Map<string, string>>();
|
||||
private _totalBindings = 0;
|
||||
private _finalized = false;
|
||||
private _disposed = false;
|
||||
|
||||
/**
|
||||
* Append bindings for a file. Safe to call multiple times for the same file.
|
||||
* Throws if the accumulator has been finalized. Skips if entries is empty.
|
||||
*
|
||||
* The `entries` parameter is `readonly` — this method never mutates the
|
||||
* caller's array. Internally, the first `appendFile` call per filePath
|
||||
* makes a defensive copy (`slice()`), and subsequent calls push into the
|
||||
* accumulator's own storage.
|
||||
*/
|
||||
appendFile(filePath: string, entries: readonly BindingEntry[]): void {
|
||||
if (this._finalized) {
|
||||
throw new Error(
|
||||
'[BindingAccumulator] appendFile after finalize — no further appends allowed',
|
||||
);
|
||||
}
|
||||
// Single-use lifecycle: once disposed, the accumulator is dead. A
|
||||
// post-dispose append almost always indicates a missed wiring step
|
||||
// (the consumer is reading state that was supposed to be released),
|
||||
// so convert the silent use-after-dispose into a loud failure.
|
||||
if (this._disposed) {
|
||||
throw new Error('BindingAccumulator: use after dispose');
|
||||
}
|
||||
if (entries.length === 0) {
|
||||
return;
|
||||
}
|
||||
// Note on the file-scope-only invariant:
|
||||
// The accumulator does NOT reject function-scope entries at this
|
||||
// boundary. The narrowing contract is enforced by the two production
|
||||
// write sites — `parse-worker.ts` (which uses `typeEnv.fileScope()`
|
||||
// and hardcodes `scope: ''` in the pipeline adapter) and
|
||||
// `type-env.ts::flush()` (which iterates only `env.get(FILE_SCOPE)`).
|
||||
// The class JSDoc documents the invariant and the Phase 9 reversion
|
||||
// path. Making `appendFile` runtime-reject non-file-scope entries
|
||||
// would break the accumulator's own storage-split tests which
|
||||
// legitimately exercise mixed-scope entries. If a future write path
|
||||
// violates the invariant, tests should fail via missing exports in
|
||||
// the enrichment loop, not via an assertion here.
|
||||
// All-scope store.
|
||||
const existingAll = this._allByFile.get(filePath);
|
||||
if (existingAll !== undefined) {
|
||||
for (const e of entries) {
|
||||
existingAll.push(e);
|
||||
}
|
||||
} else {
|
||||
this._allByFile.set(filePath, entries.slice());
|
||||
}
|
||||
// File-scope fast-path store (nested Map for O(1) point-lookup via fileScopeGet).
|
||||
// Populated lazily on first file-scope entry per file.
|
||||
let fileScopeMap = this._fileScopeByFile.get(filePath);
|
||||
for (const e of entries) {
|
||||
if (e.scope === '') {
|
||||
if (fileScopeMap === undefined) {
|
||||
fileScopeMap = new Map();
|
||||
this._fileScopeByFile.set(filePath, fileScopeMap);
|
||||
}
|
||||
fileScopeMap.set(e.varName, e.typeName);
|
||||
}
|
||||
}
|
||||
this._totalBindings += entries.length;
|
||||
}
|
||||
|
||||
/** Lock the accumulator — no further appends. Idempotent. */
|
||||
finalize(): void {
|
||||
// Dev-mode invariant: verify the parallel storage split is consistent.
|
||||
// `_fileScopeByFile` must be a proper projection of `_allByFile`
|
||||
// where the outer key is a subset and the inner entries are exactly
|
||||
// the `scope === ''` subset of `_allByFile[key]`. A drift would
|
||||
// indicate a bug in `appendFile()` where one map was updated but
|
||||
// not the other.
|
||||
if (process.env.NODE_ENV !== 'production' && !this._finalized) {
|
||||
for (const [filePath, fileScopeMap] of this._fileScopeByFile) {
|
||||
const allEntries = this._allByFile.get(filePath);
|
||||
if (allEntries === undefined) {
|
||||
throw new Error(
|
||||
`[BindingAccumulator] storage split drift: file ${filePath} has file-scope entries ` +
|
||||
`but no _allByFile entry`,
|
||||
);
|
||||
}
|
||||
// Count unique file-scope varNames in _allByFile (to match Map dedup
|
||||
// semantics in _fileScopeByFile where Map.set deduplicates same-name).
|
||||
const projectedNames = new Set(
|
||||
allEntries.filter((e) => e.scope === '').map((e) => e.varName),
|
||||
);
|
||||
if (projectedNames.size !== fileScopeMap.size) {
|
||||
throw new Error(
|
||||
`[BindingAccumulator] storage split drift: file ${filePath} has ` +
|
||||
`${fileScopeMap.size} file-scope names in Map but ${projectedNames.size} unique ` +
|
||||
`file-scope varNames in _allByFile`,
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
this._finalized = true;
|
||||
}
|
||||
|
||||
/**
|
||||
* Release the accumulator's heap footprint. Clears both internal storage
|
||||
* maps and resets `_totalBindings` to zero. Idempotent — calling twice
|
||||
* is a no-op. Orthogonal to `finalize()` — calling `dispose()` does not
|
||||
* change the finalized state.
|
||||
*
|
||||
* **Single-use lifecycle.** This is a one-way terminal transition: the
|
||||
* accumulator is not recyclable. Any subsequent `appendFile` call throws
|
||||
* (`'BindingAccumulator: use after dispose'`), regardless of whether
|
||||
* `finalize()` was called first. Post-dispose reads do not throw —
|
||||
* they return empty/undefined state matching a never-appended-to
|
||||
* accumulator:
|
||||
* - `fileCount === 0`
|
||||
* - `totalBindings === 0`
|
||||
* - `files()` yields an empty iterator
|
||||
* - `getFile(x)` returns `undefined` for all `x`
|
||||
* - `fileScopeEntries(x)` returns `[]` for all `x`
|
||||
* - `fileScopeGet(x, y)` returns `undefined` for all `x, y`
|
||||
* - `estimateMemoryBytes()` returns `0`
|
||||
*
|
||||
* Lifecycle note: the pipeline disposes the accumulator inside the
|
||||
* `finally` of the `crossFile` phase, which is scheduled after every
|
||||
* other accumulator consumer (Phase 9 call/assignment processing and
|
||||
* the ExportedTypeMap enrichment loop). The dispose call therefore
|
||||
* runs once, on both the happy path and the throw path of the
|
||||
* crossFile phase.
|
||||
*/
|
||||
dispose(): void {
|
||||
this._allByFile.clear();
|
||||
this._fileScopeByFile.clear();
|
||||
this._totalBindings = 0;
|
||||
this._disposed = true;
|
||||
}
|
||||
|
||||
/** Get all bindings for a file, or undefined if the file is unknown. */
|
||||
getFile(filePath: string): readonly BindingEntry[] | undefined {
|
||||
return this._allByFile.get(filePath);
|
||||
}
|
||||
|
||||
/**
|
||||
* Get only scope='' (file-level) entries as [varName, typeName] tuples.
|
||||
* For iteration-based consumers (e.g., `enrichExportedTypeMap`).
|
||||
* Returns an empty array for an unknown file.
|
||||
*
|
||||
* O(1) map lookup + O(n_file_scope) tuple reconstruction from the inner
|
||||
* Map. Does NOT walk function-scope entries.
|
||||
*
|
||||
* For point-lookup consumers (e.g., Phase 9 fallback), prefer
|
||||
* `fileScopeGet(filePath, name)` — O(1) with no allocation.
|
||||
*/
|
||||
fileScopeEntries(filePath: string): readonly (readonly [string, string])[] {
|
||||
const map = this._fileScopeByFile.get(filePath);
|
||||
return map ? [...map.entries()] : [];
|
||||
}
|
||||
|
||||
/**
|
||||
* O(1) point-lookup for a single file-scope binding by (filePath, name).
|
||||
* Returns the typeName if found, `undefined` otherwise.
|
||||
*
|
||||
* This is the preferred lookup path for Phase 9 consumers that resolve
|
||||
* a single callee's return type — avoids the O(n_file_scope) iteration
|
||||
* and defensive-copy allocation of `fileScopeEntries()`.
|
||||
*/
|
||||
fileScopeGet(filePath: string, name: string): string | undefined {
|
||||
return this._fileScopeByFile.get(filePath)?.get(name);
|
||||
}
|
||||
|
||||
/** Iterate over all file paths in insertion order. */
|
||||
files(): IterableIterator<string> {
|
||||
return this._allByFile.keys();
|
||||
}
|
||||
|
||||
/** Number of distinct files with at least one binding. */
|
||||
get fileCount(): number {
|
||||
return this._allByFile.size;
|
||||
}
|
||||
|
||||
/** Total number of binding entries across all files. */
|
||||
get totalBindings(): number {
|
||||
return this._totalBindings;
|
||||
}
|
||||
|
||||
/** Whether the accumulator has been finalized. */
|
||||
get finalized(): boolean {
|
||||
return this._finalized;
|
||||
}
|
||||
|
||||
/**
|
||||
* Whether the accumulator has been disposed. Exposed for symmetry with
|
||||
* `finalized` so debug tooling and future Phase 9 consumers can detect a
|
||||
* disposed accumulator without inspecting empty state heuristically.
|
||||
*
|
||||
* Disposal and finalization are orthogonal: a disposed accumulator may or
|
||||
* may not be finalized, and vice versa. See `dispose()` for the full
|
||||
* lifecycle contract.
|
||||
*/
|
||||
get disposed(): boolean {
|
||||
return this._disposed;
|
||||
}
|
||||
|
||||
/**
|
||||
* Rough memory estimate in bytes (intentionally pessimistic).
|
||||
* Formula: sum of (ENTRY_OVERHEAD + char bytes of scope+varName+typeName) per entry
|
||||
* + MAP_ENTRY_OVERHEAD + char bytes of filePath per file.
|
||||
*
|
||||
* Note: V8 stores all-ASCII strings as Latin-1 (1 byte/char) and only upgrades
|
||||
* to UCS-2 (2 bytes/char) for non-Latin-1 code points. Source paths and type names
|
||||
* are typically all-ASCII, so actual heap cost is roughly half what this returns.
|
||||
* The pessimistic factor is intentional — better to over-budget than under-budget.
|
||||
*
|
||||
* **⚠ Cost profile**: O(totalBindings) — iterates every entry in
|
||||
* `_allByFile` and reads three string `.length` properties per entry.
|
||||
* At a typical repo scale (10k files × ~20 file-scope bindings) this is
|
||||
* ~200k property reads per call. Call at most once per pipeline run,
|
||||
* NOT per file, per chunk, or per progress tick. The current single
|
||||
* call site is the dev-mode telemetry log at the pipeline finalize
|
||||
* seam. Adding a per-file-progress caller would silently make it
|
||||
* quadratic in repo size.
|
||||
*/
|
||||
estimateMemoryBytes(): number {
|
||||
let total = 0;
|
||||
for (const [filePath, entries] of this._allByFile) {
|
||||
total += MAP_ENTRY_OVERHEAD + filePath.length * 2;
|
||||
for (const e of entries) {
|
||||
total += ENTRY_OVERHEAD + (e.scope.length + e.varName.length + e.typeName.length) * 2;
|
||||
}
|
||||
}
|
||||
return total;
|
||||
}
|
||||
}
|
||||
File diff suppressed because it is too large
Load Diff
@@ -0,0 +1,177 @@
|
||||
import type { SyntaxNode } from '../utils/ast-helpers.js';
|
||||
import type { NodeLabel } from 'gitnexus-shared';
|
||||
import type {
|
||||
ClassExtractionConfig,
|
||||
ClassExtractor,
|
||||
ClassLikeNodeLabel,
|
||||
ExtractedClassSymbol,
|
||||
} from '../class-types.js';
|
||||
|
||||
const DEFAULT_SCOPE_NAME_NODE_TYPES = new Set([
|
||||
'nested_namespace_specifier',
|
||||
'scoped_identifier',
|
||||
'scoped_type_identifier',
|
||||
'qualified_name',
|
||||
'namespace_name',
|
||||
'namespace_identifier',
|
||||
'package_identifier',
|
||||
'type_identifier',
|
||||
'identifier',
|
||||
'name',
|
||||
'constant',
|
||||
]);
|
||||
|
||||
const DEFAULT_TYPE_NAME_NODE_TYPES = new Set([
|
||||
'type_identifier',
|
||||
'identifier',
|
||||
'simple_identifier',
|
||||
'namespace_identifier',
|
||||
'constant',
|
||||
'name',
|
||||
]);
|
||||
|
||||
const DEFAULT_LABEL_BY_NODE_TYPE: Record<string, ClassLikeNodeLabel> = {
|
||||
class_declaration: 'Class',
|
||||
abstract_class_declaration: 'Class',
|
||||
interface_declaration: 'Interface',
|
||||
struct_declaration: 'Struct',
|
||||
record_declaration: 'Record',
|
||||
enum_declaration: 'Enum',
|
||||
class_definition: 'Class',
|
||||
struct_specifier: 'Struct',
|
||||
class_specifier: 'Class',
|
||||
enum_specifier: 'Enum',
|
||||
struct_item: 'Struct',
|
||||
enum_item: 'Enum',
|
||||
class: 'Class',
|
||||
object_declaration: 'Class',
|
||||
companion_object: 'Class',
|
||||
protocol_declaration: 'Interface',
|
||||
extension_declaration: 'Class',
|
||||
};
|
||||
|
||||
const CLASS_LIKE_LABELS = new Set<ClassLikeNodeLabel>([
|
||||
'Class',
|
||||
'Struct',
|
||||
'Interface',
|
||||
'Enum',
|
||||
'Record',
|
||||
]);
|
||||
|
||||
const normalizeQualifiedName = (value: string): string =>
|
||||
value
|
||||
.replace(/\s+/g, '')
|
||||
.replace(/^::/, '')
|
||||
.replace(/::/g, '.')
|
||||
.replace(/\\/g, '.')
|
||||
.replace(/\.+/g, '.')
|
||||
.replace(/^\.+|\.+$/g, '');
|
||||
|
||||
const splitQualifiedName = (value: string): string[] => {
|
||||
const normalized = normalizeQualifiedName(value);
|
||||
return normalized ? normalized.split('.').filter(Boolean) : [];
|
||||
};
|
||||
|
||||
const extractScopeSegmentsFromNode = (
|
||||
scopeNode: SyntaxNode,
|
||||
scopeNameNodeTypes: ReadonlySet<string>,
|
||||
): string[] => {
|
||||
const nameNode =
|
||||
scopeNode.childForFieldName?.('name') ??
|
||||
scopeNode.namedChildren?.find((child) => scopeNameNodeTypes.has(child.type));
|
||||
return nameNode ? splitQualifiedName(nameNode.text) : [];
|
||||
};
|
||||
|
||||
const extractTypeNameFromNode = (node: SyntaxNode): string | undefined => {
|
||||
const nameField = node.childForFieldName?.('name');
|
||||
if (nameField) return nameField.text;
|
||||
const nameChild = node.namedChildren?.find((child) =>
|
||||
DEFAULT_TYPE_NAME_NODE_TYPES.has(child.type),
|
||||
);
|
||||
return nameChild?.text;
|
||||
};
|
||||
|
||||
const isClassLikeLabel = (label: NodeLabel | null | undefined): label is ClassLikeNodeLabel =>
|
||||
label !== undefined && label !== null && CLASS_LIKE_LABELS.has(label as ClassLikeNodeLabel);
|
||||
|
||||
export function createClassExtractor(config: ClassExtractionConfig): ClassExtractor {
|
||||
const typeDeclarationSet = new Set(config.typeDeclarationNodes);
|
||||
const fileScopeSet = new Set(config.fileScopeNodeTypes ?? []);
|
||||
const ancestorScopeSet = new Set(config.ancestorScopeNodeTypes ?? []);
|
||||
const scopeNameNodeTypes = new Set([
|
||||
...DEFAULT_SCOPE_NAME_NODE_TYPES,
|
||||
...(config.scopeNameNodeTypes ?? []),
|
||||
]);
|
||||
|
||||
const buildQualifiedName = (node: SyntaxNode, simpleName: string): string => {
|
||||
let root = node;
|
||||
while (root.parent) root = root.parent;
|
||||
|
||||
const readScopeSegments = (scopeNode: SyntaxNode): string[] =>
|
||||
config.extractScopeSegments?.(scopeNode) ??
|
||||
extractScopeSegmentsFromNode(scopeNode, scopeNameNodeTypes);
|
||||
|
||||
const fileScopeSegments: string[] = [];
|
||||
for (const child of root.namedChildren ?? []) {
|
||||
if (fileScopeSet.has(child.type)) {
|
||||
fileScopeSegments.push(...readScopeSegments(child));
|
||||
}
|
||||
}
|
||||
|
||||
const ancestorScopes: string[][] = [];
|
||||
let current = node.parent;
|
||||
while (current) {
|
||||
if (ancestorScopeSet.has(current.type)) {
|
||||
const segments = readScopeSegments(current);
|
||||
if (segments.length > 0) ancestorScopes.push(segments);
|
||||
}
|
||||
current = current.parent;
|
||||
}
|
||||
|
||||
return [
|
||||
...fileScopeSegments,
|
||||
...ancestorScopes.reverse().flat(),
|
||||
...splitQualifiedName(simpleName),
|
||||
]
|
||||
.filter(Boolean)
|
||||
.join('.');
|
||||
};
|
||||
|
||||
const extract = (
|
||||
node: SyntaxNode,
|
||||
fallback?: {
|
||||
name?: string;
|
||||
type?: NodeLabel | null;
|
||||
},
|
||||
): ExtractedClassSymbol | null => {
|
||||
if (!typeDeclarationSet.has(node.type)) return null;
|
||||
|
||||
const name = config.extractName?.(node) ?? extractTypeNameFromNode(node) ?? fallback?.name;
|
||||
const type =
|
||||
config.extractType?.(node) ??
|
||||
DEFAULT_LABEL_BY_NODE_TYPE[node.type] ??
|
||||
(isClassLikeLabel(fallback?.type) ? fallback.type : undefined);
|
||||
|
||||
if (!name || !type) return null;
|
||||
|
||||
return {
|
||||
name,
|
||||
type,
|
||||
qualifiedName: buildQualifiedName(node, name) || name,
|
||||
};
|
||||
};
|
||||
|
||||
return {
|
||||
language: config.language,
|
||||
|
||||
isTypeDeclaration(node: SyntaxNode): boolean {
|
||||
return typeDeclarationSet.has(node.type);
|
||||
},
|
||||
|
||||
extract,
|
||||
|
||||
extractQualifiedName(node: SyntaxNode, simpleName: string): string | null {
|
||||
return extract(node, { name: simpleName })?.qualifiedName ?? null;
|
||||
},
|
||||
};
|
||||
}
|
||||
@@ -0,0 +1,44 @@
|
||||
import type { NodeLabel, SupportedLanguages } from 'gitnexus-shared';
|
||||
import type { SyntaxNode } from './utils/ast-helpers.js';
|
||||
|
||||
export type ClassLikeNodeLabel = Extract<
|
||||
NodeLabel,
|
||||
'Class' | 'Struct' | 'Interface' | 'Enum' | 'Record'
|
||||
>;
|
||||
|
||||
export interface ExtractedClassSymbol {
|
||||
name: string;
|
||||
type: ClassLikeNodeLabel;
|
||||
qualifiedName: string;
|
||||
}
|
||||
|
||||
/**
|
||||
* Cross-language qualified type names are normalized to dot-separated scope
|
||||
* segments:
|
||||
* - file/package scope contributes leading segments when the language has one
|
||||
* - lexical namespace/module/type scope contributes enclosing segments
|
||||
* - the simple type name is always the trailing segment
|
||||
*/
|
||||
export interface ClassExtractor {
|
||||
language: SupportedLanguages;
|
||||
isTypeDeclaration(node: SyntaxNode): boolean;
|
||||
extract(
|
||||
node: SyntaxNode,
|
||||
fallback?: {
|
||||
name?: string;
|
||||
type?: NodeLabel | null;
|
||||
},
|
||||
): ExtractedClassSymbol | null;
|
||||
extractQualifiedName(node: SyntaxNode, simpleName: string): string | null;
|
||||
}
|
||||
|
||||
export interface ClassExtractionConfig {
|
||||
language: SupportedLanguages;
|
||||
typeDeclarationNodes: string[];
|
||||
fileScopeNodeTypes?: string[];
|
||||
ancestorScopeNodeTypes?: string[];
|
||||
scopeNameNodeTypes?: string[];
|
||||
extractName?: (node: SyntaxNode) => string | undefined;
|
||||
extractType?: (node: SyntaxNode) => ClassLikeNodeLabel | undefined;
|
||||
extractScopeSegments?: (node: SyntaxNode) => string[] | null | undefined;
|
||||
}
|
||||
@@ -92,7 +92,7 @@ function isCopybook(filePath: string): boolean {
|
||||
export const processCobol = (
|
||||
graph: KnowledgeGraph,
|
||||
files: CobolFile[],
|
||||
allPathSet: Set<string>,
|
||||
allPathSet: ReadonlySet<string>,
|
||||
): CobolProcessResult => {
|
||||
const result: CobolProcessResult = {
|
||||
programs: 0,
|
||||
|
||||
@@ -226,6 +226,7 @@ export const ENTRY_POINT_PATTERNS = {
|
||||
/^onEvent$/, // BLoC event handler
|
||||
/^mapEventToState$/, // Legacy BLoC pattern
|
||||
],
|
||||
[SupportedLanguages.Vue]: [], // Vue uses TypeScript queries — entry points handled via TS patterns
|
||||
[SupportedLanguages.Cobol]: [], // Standalone regex processor — no tree-sitter entry points
|
||||
} satisfies Record<SupportedLanguages, RegExp[]>;
|
||||
|
||||
|
||||
@@ -15,12 +15,17 @@ import type { FieldVisibility } from '../../field-types.js';
|
||||
|
||||
/**
|
||||
* Check whether any child of `node` (named or unnamed) has .text matching
|
||||
* one of the given `keywords`.
|
||||
* the given `keyword`.
|
||||
*
|
||||
* Skips the `name` field child to avoid false positives when a method is
|
||||
* named after a contextual keyword (e.g. `abstract()` in TypeScript).
|
||||
*/
|
||||
export function hasKeyword(node: SyntaxNode, keyword: string): boolean {
|
||||
const nameNode = node.childForFieldName('name');
|
||||
for (let i = 0; i < node.childCount; i++) {
|
||||
const child = node.child(i);
|
||||
if (child && child.text.trim() === keyword) return true;
|
||||
if (!child || child === nameNode) continue;
|
||||
if (child.text.trim() === keyword) return true;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
@@ -46,6 +51,7 @@ export function hasModifier(node: SyntaxNode, modifierType: string, keyword: str
|
||||
/**
|
||||
* Return the first matching visibility keyword found either as a direct keyword
|
||||
* child or inside a modifier wrapper node.
|
||||
* Skips the `name` field child (same rationale as hasKeyword).
|
||||
*/
|
||||
export function findVisibility(
|
||||
node: SyntaxNode,
|
||||
@@ -53,10 +59,12 @@ export function findVisibility(
|
||||
defaultVis: FieldVisibility,
|
||||
modifierNodeType?: string,
|
||||
): FieldVisibility {
|
||||
const nameNode = node.childForFieldName('name');
|
||||
// Direct keyword children
|
||||
for (let i = 0; i < node.childCount; i++) {
|
||||
const child = node.child(i);
|
||||
const text = child?.text.trim() as FieldVisibility | undefined;
|
||||
if (!child || child === nameNode) continue;
|
||||
const text = child.text.trim() as FieldVisibility | undefined;
|
||||
if (text && (keywords as ReadonlySet<string>).has(text)) return text;
|
||||
}
|
||||
// Modifier wrapper
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
// gitnexus/src/core/ingestion/field-types.ts
|
||||
|
||||
import type { TypeEnvironment } from './type-env.js';
|
||||
import type { SymbolTable } from './symbol-table.js';
|
||||
import type { SymbolTableReader } from './model/symbol-table.js';
|
||||
import { SupportedLanguages } from 'gitnexus-shared';
|
||||
|
||||
/**
|
||||
@@ -57,7 +57,7 @@ export interface FieldExtractorContext {
|
||||
/** Type environment for resolution */
|
||||
typeEnv: TypeEnvironment;
|
||||
/** Symbol table for FQN lookups */
|
||||
symbolTable: SymbolTable;
|
||||
symbolTable: SymbolTableReader;
|
||||
/** Current file path */
|
||||
filePath: string;
|
||||
/** Language ID */
|
||||
|
||||
@@ -1,3 +1,4 @@
|
||||
import { isVerboseIngestionEnabled } from './utils/verbose.js';
|
||||
import fs from 'fs/promises';
|
||||
import path from 'path';
|
||||
import { glob } from 'glob';
|
||||
@@ -43,6 +44,7 @@ export const walkRepositoryPaths = async (
|
||||
const entries: ScannedFile[] = [];
|
||||
let processed = 0;
|
||||
let skippedLarge = 0;
|
||||
const skippedLargePaths: string[] = [];
|
||||
|
||||
for (let start = 0; start < filtered.length; start += READ_CONCURRENCY) {
|
||||
const batch = filtered.slice(start, start + READ_CONCURRENCY);
|
||||
@@ -52,6 +54,7 @@ export const walkRepositoryPaths = async (
|
||||
const stat = await fs.stat(fullPath);
|
||||
if (stat.size > MAX_FILE_SIZE) {
|
||||
skippedLarge++;
|
||||
skippedLargePaths.push(relativePath.replace(/\\/g, '/'));
|
||||
return null;
|
||||
}
|
||||
return { path: relativePath.replace(/\\/g, '/'), size: stat.size };
|
||||
@@ -73,6 +76,11 @@ export const walkRepositoryPaths = async (
|
||||
console.warn(
|
||||
` Skipped ${skippedLarge} large files (>${MAX_FILE_SIZE / 1024}KB, likely generated/vendored)`,
|
||||
);
|
||||
if (isVerboseIngestionEnabled()) {
|
||||
for (const p of skippedLargePaths) {
|
||||
console.warn(` - ${p}`);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return entries;
|
||||
|
||||
@@ -891,6 +891,7 @@ export const AST_FRAMEWORK_PATTERNS_BY_LANGUAGE = {
|
||||
patterns: FRAMEWORK_AST_PATTERNS.riverpod,
|
||||
},
|
||||
],
|
||||
[SupportedLanguages.Vue]: [], // Vue uses TypeScript AST framework detection
|
||||
[SupportedLanguages.Cobol]: [], // Standalone regex processor — no AST framework patterns
|
||||
} satisfies Record<SupportedLanguages, AstFrameworkPatternConfig[]>;
|
||||
|
||||
|
||||
@@ -19,47 +19,34 @@ import { ASTCache } from './ast-cache.js';
|
||||
import Parser from 'tree-sitter';
|
||||
import { isLanguageAvailable, loadParser, loadLanguage } from '../tree-sitter/parser-loader.js';
|
||||
import { generateId } from '../../lib/utils.js';
|
||||
import { getLanguageFromFilename } from 'gitnexus-shared';
|
||||
import { getLanguageFromFilename, type SupportedLanguages } from 'gitnexus-shared';
|
||||
import { isVerboseIngestionEnabled } from './utils/verbose.js';
|
||||
import { yieldToEventLoop } from './utils/event-loop.js';
|
||||
import { SupportedLanguages } from 'gitnexus-shared';
|
||||
import { getProvider } from './languages/index.js';
|
||||
import { getTreeSitterBufferSize } from './constants.js';
|
||||
import type { ExtractedHeritage } from './workers/parse-worker.js';
|
||||
import type { ResolutionContext } from './resolution-context.js';
|
||||
import { TIER_CONFIDENCE } from './resolution-context.js';
|
||||
import type {
|
||||
ExtractedHeritage,
|
||||
HeritageResolutionStrategy,
|
||||
HeritageStrategyLookup,
|
||||
} from './model/heritage-map.js';
|
||||
import { resolveExtendsType } from './model/heritage-map.js';
|
||||
import type { ResolutionContext } from './model/resolution-context.js';
|
||||
import { TIER_CONFIDENCE } from './model/resolution-context.js';
|
||||
|
||||
/**
|
||||
* Determine whether a heritage.extends capture is actually an IMPLEMENTS relationship.
|
||||
* Uses the symbol table first (authoritative — Tier 1); falls back to provider-defined
|
||||
* heuristics for external symbols not present in the graph:
|
||||
* - interfaceNamePattern: matched against parent name (e.g., /^I[A-Z]/ for C#/Java)
|
||||
* - heritageDefaultEdge: 'IMPLEMENTS' causes all unresolved parents to map to IMPLEMENTS
|
||||
* - All others: default EXTENDS
|
||||
* Derive the heritage-resolution strategy for a language from its
|
||||
* `LanguageProvider`. This is the production wiring that `buildHeritageMap`
|
||||
* and the standalone `resolveExtendsType` call site use — the model layer
|
||||
* itself stays unaware of the provider registry.
|
||||
*/
|
||||
/** Exported for implementor-map construction (C#/Java: `extends` rows in base_list may be interfaces). */
|
||||
export const resolveExtendsType = (
|
||||
parentName: string,
|
||||
currentFilePath: string,
|
||||
ctx: ResolutionContext,
|
||||
language: SupportedLanguages,
|
||||
): { type: 'EXTENDS' | 'IMPLEMENTS'; idPrefix: string } => {
|
||||
const resolved = ctx.resolve(parentName, currentFilePath);
|
||||
if (resolved && resolved.candidates.length > 0) {
|
||||
const isInterface = resolved.candidates[0].type === 'Interface';
|
||||
return isInterface
|
||||
? { type: 'IMPLEMENTS', idPrefix: 'Interface' }
|
||||
: { type: 'EXTENDS', idPrefix: 'Class' };
|
||||
}
|
||||
// Unresolved symbol — fall back to provider-defined heuristics
|
||||
const provider = getProvider(language);
|
||||
if (provider.interfaceNamePattern?.test(parentName)) {
|
||||
return { type: 'IMPLEMENTS', idPrefix: 'Interface' };
|
||||
}
|
||||
if (provider.heritageDefaultEdge === 'IMPLEMENTS') {
|
||||
return { type: 'IMPLEMENTS', idPrefix: 'Interface' };
|
||||
}
|
||||
return { type: 'EXTENDS', idPrefix: 'Class' };
|
||||
export const getHeritageStrategyForLanguage: HeritageStrategyLookup = (
|
||||
lang: SupportedLanguages,
|
||||
): HeritageResolutionStrategy => {
|
||||
const provider = getProvider(lang);
|
||||
return {
|
||||
interfaceNamePattern: provider.interfaceNamePattern,
|
||||
defaultEdge: provider.heritageDefaultEdge ?? 'EXTENDS',
|
||||
};
|
||||
};
|
||||
|
||||
/**
|
||||
@@ -180,7 +167,7 @@ export const processHeritage = async (
|
||||
parentClassName,
|
||||
file.path,
|
||||
ctx,
|
||||
language,
|
||||
getHeritageStrategyForLanguage(language),
|
||||
);
|
||||
|
||||
const child = resolveHeritageId(
|
||||
@@ -296,7 +283,7 @@ export const processHeritageFromExtracted = async (
|
||||
h.parentName,
|
||||
h.filePath,
|
||||
ctx,
|
||||
fileLanguage,
|
||||
getHeritageStrategyForLanguage(fileLanguage),
|
||||
);
|
||||
|
||||
const child = resolveHeritageId(
|
||||
@@ -372,7 +359,7 @@ export const processHeritageFromExtracted = async (
|
||||
/**
|
||||
* Walk source files with the same heritage captures as parse-worker, producing
|
||||
* {@link ExtractedHeritage} rows without mutating the graph. Used on the
|
||||
* sequential pipeline path so `buildImplementorMap(..., ctx)` can run before
|
||||
* sequential pipeline path so `buildHeritageMap(..., ctx)` can run before
|
||||
* `processCalls` (worker path defers calls until heritage from all chunks exists).
|
||||
*/
|
||||
export async function extractExtractedHeritageFromFiles(
|
||||
|
||||
@@ -12,7 +12,11 @@ import type { ExtractedImport } from './workers/parse-worker.js';
|
||||
import { getTreeSitterBufferSize } from './constants.js';
|
||||
import { loadImportConfigs } from './language-config.js';
|
||||
import { buildSuffixIndex } from './import-resolvers/utils.js';
|
||||
import type { ResolutionContext, ModuleAliasMap } from './resolution-context.js';
|
||||
import type {
|
||||
ResolutionContext,
|
||||
ModuleAliasMap,
|
||||
NamedImportMap,
|
||||
} from './model/resolution-context.js';
|
||||
import type {
|
||||
ImportResult,
|
||||
ResolveCtx,
|
||||
@@ -20,8 +24,7 @@ import type {
|
||||
} from './import-resolvers/types.js';
|
||||
import type { NamedBinding } from './named-bindings/types.js';
|
||||
import type { SyntaxNode } from './utils/ast-helpers.js';
|
||||
|
||||
const isDev = process.env.NODE_ENV === 'development';
|
||||
import { isDev } from './utils/env.js';
|
||||
|
||||
// Type: Map<FilePath, Set<ResolvedFilePath>>
|
||||
// Stores all files that a given file imports from
|
||||
@@ -61,30 +64,6 @@ function wireImplicitImports(
|
||||
// Avoids expanding every Go package import into N individual ImportMap edges.
|
||||
export type PackageMap = Map<string, Set<string>>;
|
||||
|
||||
// Type: Map<ImportingFilePath, Map<LocalName, {sourcePath, exportedName}>>
|
||||
// Tracks which specific names a file imports from which sources (TS/Python only).
|
||||
// Used to tighten Tier 2a resolution: `import { User } from './models'`
|
||||
// means only `User` (not `Repo`) is visible from models.ts via this import.
|
||||
// Stores both the resolved source path and the original exported name so that
|
||||
// aliased imports (`import { User as U }`) can resolve U → User in the source file.
|
||||
export interface NamedImportBinding {
|
||||
sourcePath: string;
|
||||
exportedName: string;
|
||||
}
|
||||
export type NamedImportMap = Map<string, Map<string, NamedImportBinding>>;
|
||||
|
||||
/**
|
||||
* Check if a file path is directly inside a package directory identified by its suffix.
|
||||
* Used by the symbol resolver for Go and C# directory-level import matching.
|
||||
*/
|
||||
export function isFileInPackageDir(filePath: string, dirSuffix: string): boolean {
|
||||
// Prepend '/' so paths like "internal/auth/service.go" match suffix "/internal/auth/"
|
||||
const normalized = '/' + filePath.replace(/\\/g, '/');
|
||||
if (!normalized.includes(dirSuffix)) return false;
|
||||
const afterDir = normalized.substring(normalized.indexOf(dirSuffix) + dirSuffix.length);
|
||||
return !afterDir.includes('/');
|
||||
}
|
||||
|
||||
// ImportResolutionContext is defined in ./import-resolvers/types.ts — re-exported here for consumers.
|
||||
|
||||
export function buildImportResolutionContext(allPaths: string[]): ImportResolutionContext {
|
||||
|
||||
@@ -11,6 +11,7 @@ export const EXTENSIONS = [
|
||||
'.ts',
|
||||
'.jsx',
|
||||
'.js',
|
||||
'.vue',
|
||||
'/index.tsx',
|
||||
'/index.ts',
|
||||
'/index.jsx',
|
||||
|
||||
@@ -0,0 +1,13 @@
|
||||
/**
|
||||
* Vue import resolver — delegates to TypeScript's standard resolver.
|
||||
*
|
||||
* Vue <script> blocks use the same import syntax as TypeScript (including
|
||||
* tsconfig path aliases like `@/`), so no custom resolution logic is needed.
|
||||
*/
|
||||
|
||||
import { SupportedLanguages } from 'gitnexus-shared';
|
||||
import { resolveStandard } from './standard.js';
|
||||
import type { ImportResolverFn } from './types.js';
|
||||
|
||||
export const resolveVueImport: ImportResolverFn = (raw, fp, ctx) =>
|
||||
resolveStandard(raw, fp, ctx, SupportedLanguages.TypeScript);
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user