Compare commits

..
Author SHA1 Message Date
Gergo Magyar 3768e3bfd0 fix(server): log skip-embedding count and table-not-found swallow path
Addresses review feedback on PR #823:
- Log count of already-embedded nodes when skipNodeIds is populated
  (aids debugging if Kuzu driver row shape changes).
- Log when the 'table does not exist' swallow path fires so ops can
  catch it if Kuzu ever changes error wording.
- Document the {} config positional argument with an inline comment
  referencing the runEmbeddingPipeline signature.
2026-04-15 07:40:58 +01:00
Gergo Magyar 3384575ac6 style: prettier format gitnexus/src/server/api.ts 2026-04-15 07:38:16 +01:00
jonasvanderhaegen-xve dd194d56b1 fix(server): narrow catch to table-not-exist errors only in POST /api/embed
Bare catch{} would silently swallow connection errors and proceed to
re-embed all nodes, hiding infrastructure issues. Now only swallows
errors where the CodeEmbedding table does not yet exist.
2026-04-14 14:27:03 +02:00
jonasvanderhaegen-xve 80a6fde2ba fix(server): skip already-embedded nodes in POST /api/embed to avoid vector-index SET error
Kuzu/LadybugDB forbids SET on a property that is part of a vector index.
The /api/embed endpoint was calling runEmbeddingPipeline without skipNodeIds,
causing it to attempt MERGE+SET on every node including those already embedded.

Fix: query existing CodeEmbedding nodeIds before running the pipeline and pass
them as skipNodeIds so only new (unembedded) nodes are processed.
2026-04-14 13:33:48 +02:00
jonasvanderhaegen-xve 8d38cc99fa fix(embeddings): use MERGE instead of CREATE for CodeEmbedding inserts
CREATE fails with duplicate PK when a CodeEmbedding node already exists,
which happens when:
- A PostToolUse hook triggers a concurrent gitnexus analyze during an
  active analyze run (git commits fire the hook)
- A partial prior run left some embeddings in the DB before a crash

Switching to MERGE makes the insert idempotent: existing embeddings are
updated in place, new ones are created, no PK violations.

Fixes: #822
2026-04-14 12:59:14 +02:00
jonasvanderhaegen-xve 41844edf88 fix(csv-generator): deduplicate all node types, not just File nodes
The pipeline can produce duplicate node IDs across all symbol types
(Class, Method, Function, etc.). Only File nodes were guarded by a
seenFileIds Set, leaving every other type unprotected. When the CSV
was COPY'd into LadybugDB, duplicate PKs caused mass "Batch execution
error: Found duplicated primary key value" warnings on gitnexus serve.

Replace the per-type seenFileIds with a single seenNodeIds Set checked
at the top of the iteration loop, before the switch, so every label is
covered by the same O(1) deduplication guard.

Fixes: #822
2026-04-14 12:32:08 +02:00
554 changed files with 6172 additions and 59750 deletions
@@ -17,13 +17,12 @@ npx gitnexus analyze
Run from the project root. This parses all source files, builds the knowledge graph, writes it to `.gitnexus/`, and generates CLAUDE.md / AGENTS.md context files.
| Flag | Effect |
| ------------------- | ------------------------------------------------------------------------------------------------------- |
| `--force` | Force full re-index even if up to date |
| `--embeddings` | Enable embedding generation for semantic search (off by default) |
| `--drop-embeddings` | Drop existing embeddings on rebuild. By default, an `analyze` without `--embeddings` preserves them. |
| Flag | Effect |
| -------------- | ---------------------------------------------------------------- |
| `--force` | Force full re-index even if up to date |
| `--embeddings` | Enable embedding generation for semantic search (off by default) |
**When to run:** First time in a project, after major code changes, or when `gitnexus://repo/{name}/context` reports the index is stale. In Claude Code, a PostToolUse hook detects staleness after `git commit` and `git merge` and notifies the agent to run `analyze` — the hook does not run analyze itself, to avoid blocking the agent for up to 120s and risking KuzuDB corruption on timeout.
**When to run:** First time in a project, after major code changes, or when `gitnexus://repo/{name}/context` reports the index is stale. In Claude Code, a PostToolUse hook runs `analyze` automatically after `git commit` and `git merge`, preserving embeddings if previously generated.
### status — Check index freshness
-1
View File
@@ -1 +0,0 @@
plans/
-21
View File
@@ -1,21 +0,0 @@
.git
.gitignore
.DS_Store
node_modules
**/node_modules
dist
**/dist
coverage
**/coverage
.env
.env.local
.env.*.local
**/*.tsbuildinfo
.gitnexus
gitnexus-web/playwright-report
gitnexus-web/test-results
-19
View File
@@ -1,19 +0,0 @@
# Images (signed Cosign keyless on every push from main / vX.Y.Z tags).
# Available from both GHCR (default below) and Docker Hub — pick one:
# GHCR: ghcr.io/abhigyanpatwari/gitnexus{,-web}:latest
# Docker Hub: akonlabs/gitnexus{,-web}:latest
# Both registries receive the same digest from a single signed build.
SERVER_IMAGE=ghcr.io/abhigyanpatwari/gitnexus:latest
WEB_IMAGE=ghcr.io/abhigyanpatwari/gitnexus-web:latest
# Container names
SERVER_CONTAINER_NAME=gitnexus-server
WEB_CONTAINER_NAME=gitnexus-web
# Host ports — the web UI expects the server on http://localhost:4747 by default.
SERVER_HOST_PORT=4747
WEB_HOST_PORT=4173
# Optional read-only mount, exposed to the server as /workspace.
# Override with the directory that contains the repos you want to index.
WORKSPACE_DIR=./
@@ -1,105 +0,0 @@
# Wraps docker/build-push-action with one automatic retry. Upstream explicitly
# keeps retry out of the action (docker/build-push-action#1422); a local
# composite keeps docker.yml readable and pins the same action SHA in one place.
name: Docker build-push (with retry)
description: >-
Runs docker/build-push-action twice on failure with a configurable backoff,
then exposes the digest from whichever attempt succeeded.
inputs:
context:
description: Build context path
required: false
default: '.'
file:
description: Dockerfile path (relative to repo root)
required: true
platforms:
description: Comma-separated platforms list for buildx
required: true
push:
description: Whether to push (string 'true' or 'false')
required: true
tags:
description: Newline-separated image tags (from docker/metadata-action)
required: true
labels:
description: Labels string (from docker/metadata-action)
required: true
cache-from:
description: buildx cache-from value
required: true
cache-to:
description: buildx cache-to value (include ignore-error=true for GHA cache flakes)
required: true
retry-wait-seconds:
description: Seconds to sleep before the second attempt
required: false
default: '45'
outputs:
digest:
description: Manifest digest from the successful build attempt
value: ${{ steps.resolve.outputs.digest }}
runs:
using: composite
steps:
- name: Build and push (attempt 1)
id: try1
continue-on-error: true
uses: docker/build-push-action@bcafcacb16a39f128d818304e6c9c0c18556b85f # v7.1.0
with:
context: ${{ inputs.context }}
file: ${{ inputs.file }}
platforms: ${{ inputs.platforms }}
push: ${{ inputs.push == 'true' }}
tags: ${{ inputs.tags }}
labels: ${{ inputs.labels }}
cache-from: ${{ inputs.cache-from }}
cache-to: ${{ inputs.cache-to }}
provenance: mode=max
sbom: true
- name: Backoff before Docker build retry
if: steps.try1.outcome == 'failure'
shell: bash
env:
RETRY_WAIT_SECONDS: ${{ inputs.retry-wait-seconds }}
run: |
echo "::warning::Docker build-push attempt 1 failed; retrying in ${RETRY_WAIT_SECONDS}s…"
sleep "${RETRY_WAIT_SECONDS}"
- name: Build and push (attempt 2)
id: try2
if: steps.try1.outcome == 'failure'
uses: docker/build-push-action@bcafcacb16a39f128d818304e6c9c0c18556b85f # v7.1.0
with:
context: ${{ inputs.context }}
file: ${{ inputs.file }}
platforms: ${{ inputs.platforms }}
push: ${{ inputs.push == 'true' }}
tags: ${{ inputs.tags }}
labels: ${{ inputs.labels }}
cache-from: ${{ inputs.cache-from }}
cache-to: ${{ inputs.cache-to }}
provenance: mode=max
sbom: true
- name: Resolve image digest
id: resolve
if: always()
shell: bash
run: |
set -euo pipefail
if [ "${{ steps.try1.outcome }}" = "success" ]; then
echo "digest=${{ steps.try1.outputs.digest }}" >> "$GITHUB_OUTPUT"
exit 0
fi
if [ "${{ steps.try2.outcome }}" = "success" ]; then
echo "::notice::docker-build-push retry succeeded (attempt 2); investigate if this recurs across runs."
echo "digest=${{ steps.try2.outputs.digest }}" >> "$GITHUB_OUTPUT"
exit 0
fi
echo "::error::Docker build and push failed after two attempts (registry/cache flake or real build error)."
exit 1
@@ -1,15 +1,12 @@
name: Setup GitNexus Web
description: Setup Node.js 20.19+ (vite 7 floor), build gitnexus-shared, install web dependencies
description: Setup Node.js 20, build gitnexus-shared, install web dependencies
runs:
using: composite
steps:
- uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4
with:
# Vite 7 requires Node ^20.19.0 || >=22.12.0 (require(esm) support).
# Pin explicitly so we don't depend on the floating "20" alias resolving
# to a high enough patch version on every runner image.
node-version: '20.19.0'
node-version: 20
cache: npm
cache-dependency-path: gitnexus-web/package-lock.json
-75
View File
@@ -1,75 +0,0 @@
version: 2
updates:
# Keep third-party Actions SHA pins current. See CONTRIBUTING.md — when
# reviewing these bumps, verify the SHA corresponds to the claimed tag by
# running `gh api repos/<owner>/<action>/git/refs/tags/<tag>` before merge.
- package-ecosystem: github-actions
directory: /
schedule:
interval: weekly
open-pull-requests-limit: 5
commit-message:
prefix: chore
include: scope
labels:
- dependencies
- ci
# Gitnexus npm deps — tree-sitter grammars checked daily so we catch
# new releases that unblock the tree-sitter 0.25 upgrade ASAP. Grammars
# are grouped so lockstep bumps produce a single PR. The tree-sitter
# RUNTIME is pinned — upgrade deliberately via the drift check workflow.
# See .github/scripts/check-tree-sitter-upgrade-readiness.py for
# the upgrade readiness tracker.
- package-ecosystem: npm
directory: /gitnexus
schedule:
interval: daily
open-pull-requests-limit: 10
commit-message:
prefix: chore(deps)
include: scope
labels:
- dependencies
groups:
tree-sitter-grammars:
patterns:
- tree-sitter-*
exclude-patterns:
- tree-sitter
- tree-sitter-cli
ignore:
# Pin the tree-sitter runtime at 0.21.x until the drift check
# reports all grammars are peer-dep compatible with 0.25.
- dependency-name: tree-sitter
update-types:
- version-update:semver-major
- version-update:semver-minor
# tree-sitter-cli follows the runtime's version cadence. Bump when
# regenerating vendor/tree-sitter-proto/src/parser.c, not on a schedule.
- dependency-name: tree-sitter-cli
# gitnexus-web (thin frontend client).
- package-ecosystem: npm
directory: /gitnexus-web
schedule:
interval: weekly
open-pull-requests-limit: 5
commit-message:
prefix: chore(deps)
include: scope
labels:
- dependencies
- frontend
# Shared types package.
- package-ecosystem: npm
directory: /gitnexus-shared
schedule:
interval: weekly
open-pull-requests-limit: 5
commit-message:
prefix: chore(deps)
include: scope
labels:
- dependencies
-53
View File
@@ -1,53 +0,0 @@
# release-drafter config — used only for PR autolabeling by
# `.github/workflows/pr-labeler.yml` (the workflow passes `disable-releaser: true`,
# so the draft-release side of release-drafter never runs).
#
# The labels applied here are the same ones `.github/release.yml` maps to
# categorized release-notes sections.
#
# `sync-labels: true` removes managed autolabels that no longer match the PR —
# critical for the breaking-change case: if a PR title drops the `!` or the body
# drops `BREAKING CHANGE:`, the `breaking` label is pulled off automatically.
# Required by release-drafter; not used because releaser is disabled.
name-template: 'unused'
tag-template: 'unused'
template: |
$CHANGES
sync-labels: true
autolabeler:
- label: enhancement
title:
- '/^feat(\([^)]+\))?!?:/i'
- label: bug
title:
- '/^fix(\([^)]+\))?!?:/i'
- label: performance
title:
- '/^perf(\([^)]+\))?!?:/i'
- label: refactor
title:
- '/^refactor(\([^)]+\))?!?:/i'
- label: documentation
title:
- '/^docs(\([^)]+\))?!?:/i'
- label: test
title:
- '/^test(\([^)]+\))?!?:/i'
- label: ci
title:
- '/^ci(\([^)]+\))?!?:/i'
- label: dependencies
title:
- '/^(build|deps)(\([^)]+\))?!?:/i'
- label: chore
title:
- '/^(chore|revert)(\([^)]+\))?!?:/i'
# Breaking-change marker: either `!` in the type prefix or `BREAKING CHANGE:` in body.
- label: breaking
title:
- '/^[a-z]+(\([^)]+\))?!:/i'
body:
- '/BREAKING[ -]CHANGE:/i'
@@ -1,358 +0,0 @@
#!/usr/bin/env python3
"""Monitor tree-sitter 0.25 upgrade readiness.
Tracks two things Dependabot cannot see:
1. Peer-dep compatibility. Each tree-sitter-* grammar declares a peer
dependency on the tree-sitter runtime. We want to know when every
grammar's *latest npm release* satisfies tree-sitter@0.25.0 so we
can upgrade without --legacy-peer-deps.
2. Vendored upstream drift. vendor/tree-sitter-proto/ is a snapshot of
coder3101/tree-sitter-proto's parser.c. When upstream moves, we want
to know whether we can pick it up.
Invoked from .github/workflows/tree-sitter-upgrade-readiness.yml daily.
Runs locally too:
python3 .github/scripts/check-tree-sitter-upgrade-readiness.py
Outputs Markdown to stdout. Exit 0 when every grammar is upgrade-ready
and the vendored proto is in sync. Exit 1 when blockers remain (the
workflow uses this to open or update a tracking issue).
No external deps -- stdlib only, so it runs on any vanilla runner.
"""
from __future__ import annotations
import json
import os
import pathlib
import re
import sys
import urllib.error
import urllib.request
REPO_ROOT = pathlib.Path(__file__).resolve().parents[2]
GITNEXUS_DIR = REPO_ROOT / "gitnexus"
VENDOR_PROTO_DIR = GITNEXUS_DIR / "vendor" / "tree-sitter-proto"
# ── Upgrade target ──────────────────────────────────────────────────────
# The runtime version we want to upgrade TO. Update this when the goal
# changes (e.g. once 0.25 lands and we target 0.26).
TARGET_RUNTIME = "0.25.0"
TARGET_RUNTIME_MAJOR_MINOR = ".".join(TARGET_RUNTIME.split(".")[:2])
# Tree-sitter runtime -> (min_abi, max_abi) it can load. Only the current
# and target entries matter; extend when changing TARGET_RUNTIME.
RUNTIME_ABI_RANGES: dict[str, tuple[int, int]] = {
"0.21": (13, 14),
"0.25": (13, 15),
}
assert TARGET_RUNTIME_MAJOR_MINOR in RUNTIME_ABI_RANGES, (
f"RUNTIME_ABI_RANGES has no entry for {TARGET_RUNTIME_MAJOR_MINOR!r}. "
f"Add the ABI range after auditing the upstream release notes."
)
# Grammars we use. Values are the upstream GitHub repos to check for
# unreleased ABI bumps (owner/repo, branch, parser.c path).
GRAMMARS: dict[str, tuple[str, str, str]] = {
"tree-sitter-c": ("tree-sitter/tree-sitter-c", "master", "src/parser.c"),
"tree-sitter-c-sharp": ("tree-sitter/tree-sitter-c-sharp", "master", "src/parser.c"),
"tree-sitter-cpp": ("tree-sitter/tree-sitter-cpp", "master", "src/parser.c"),
"tree-sitter-dart": ("UserNobody14/tree-sitter-dart", "master", "src/parser.c"),
"tree-sitter-go": ("tree-sitter/tree-sitter-go", "master", "src/parser.c"),
"tree-sitter-java": ("tree-sitter/tree-sitter-java", "master", "src/parser.c"),
"tree-sitter-javascript": ("tree-sitter/tree-sitter-javascript", "master", "src/parser.c"),
"tree-sitter-kotlin": ("fwcd/tree-sitter-kotlin", "main", "src/parser.c"),
"tree-sitter-php": ("tree-sitter/tree-sitter-php", "master", "php/src/parser.c"),
"tree-sitter-python": ("tree-sitter/tree-sitter-python", "master", "src/parser.c"),
"tree-sitter-ruby": ("tree-sitter/tree-sitter-ruby", "master", "src/parser.c"),
"tree-sitter-rust": ("tree-sitter/tree-sitter-rust", "master", "src/parser.c"),
"tree-sitter-swift": ("alex-pinkus/tree-sitter-swift", "main", "src/parser.c"),
"tree-sitter-typescript": ("tree-sitter/tree-sitter-typescript", "master", "typescript/src/parser.c"),
}
UPSTREAM_PROTO_OWNER = "coder3101"
UPSTREAM_PROTO_REPO = "tree-sitter-proto"
UPSTREAM_PROTO_BRANCH = "main"
# ── Helpers ─────────────────────────────────────────────────────────────
def read_current_runtime() -> str:
"""Return the tree-sitter runtime version pinned in package.json (e.g. '0.21')."""
pkg = json.loads((GITNEXUS_DIR / "package.json").read_text())
raw = pkg["dependencies"]["tree-sitter"]
match = re.search(r"(\d+)\.(\d+)", raw)
if not match:
raise SystemExit(f"could not parse tree-sitter version: {raw!r}")
return f"{match.group(1)}.{match.group(2)}"
def npm_view_json(pkg: str) -> dict | None:
"""Fetch package metadata from the npm registry via HTTPS.
Uses the registry API directly so we don't depend on the npm CLI
being available (it's a batch file on Windows which complicates
subprocess calls).
"""
url = f"https://registry.npmjs.org/{pkg}/latest"
try:
req = urllib.request.Request(url, headers={"Accept": "application/json"})
with urllib.request.urlopen(req, timeout=8) as resp:
return json.loads(resp.read().decode("utf-8"))
except (urllib.error.URLError, urllib.error.HTTPError, json.JSONDecodeError):
return None
def satisfies_target(peer_range: str | None, target: str) -> bool:
"""Check if a semver range like '^0.22.4' or '^0.25.0' satisfies the target.
Simple heuristic: extract the minimum version from the range and check
if target >= min. For caret ranges (^X.Y.Z), the upper bound is the
next major (for X>0) or next minor (for X==0). We check both bounds.
"""
if peer_range is None:
# No peer dep declared = no constraint = compatible.
return True
match = re.search(r"(\d+)\.(\d+)\.(\d+)", peer_range)
if not match:
return False
min_major, min_minor, min_patch = int(match.group(1)), int(match.group(2)), int(match.group(3))
t_match = re.search(r"(\d+)\.(\d+)\.(\d+)", target)
if not t_match:
return False
t_major, t_minor, t_patch = int(t_match.group(1)), int(t_match.group(2)), int(t_match.group(3))
# Target must be >= minimum.
target_tuple = (t_major, t_minor, t_patch)
min_tuple = (min_major, min_minor, min_patch)
if target_tuple < min_tuple:
return False
# For caret ranges with major 0: ^0.X.Y allows [0.X.Y, 0.(X+1).0).
if peer_range.startswith("^") and min_major == 0:
if t_major != 0 or t_minor >= min_minor + 1:
return False
# For caret ranges with major >0: ^X.Y.Z allows [X.Y.Z, (X+1).0.0).
elif peer_range.startswith("^") and min_major > 0:
if t_major >= min_major + 1:
return False
return True
_GITHUB_TOKEN = os.environ.get("GITHUB_TOKEN")
def fetch_text(url: str, timeout: int = 8) -> str | None:
"""Fetch a URL and return its text, or None on failure.
Adds an Authorization header for github.com URLs when GITHUB_TOKEN is
set (raises the rate limit from 60 to 5 000 requests/hour).
"""
headers: dict[str, str] = {}
if _GITHUB_TOKEN and ("github.com" in url or "githubusercontent.com" in url):
headers["Authorization"] = f"Bearer {_GITHUB_TOKEN}"
try:
req = urllib.request.Request(url, headers=headers)
with urllib.request.urlopen(req, timeout=timeout) as resp:
return resp.read().decode("utf-8", errors="ignore")
except (urllib.error.URLError, urllib.error.HTTPError):
return None
def extract_abi_from_text(text: str) -> int | None:
"""Extract LANGUAGE_VERSION from parser.c text."""
match = re.search(r"#define\s+LANGUAGE_VERSION\s+(\d+)", text[:4096])
return int(match.group(1)) if match else None
def extract_language_version(parser_c: pathlib.Path) -> int | None:
"""Return the LANGUAGE_VERSION defined in a parser.c, or None if absent."""
if not parser_c.is_file():
return None
with parser_c.open("r", encoding="utf-8", errors="ignore") as fh:
head = fh.read(4096)
return extract_abi_from_text(head)
def md_h(text: str, level: int = 2) -> str:
return f"{'#' * level} {text}\n"
# ── Main ────────────────────────────────────────────────────────────────
def main() -> int:
blockers: dict[str, str] = {}
lines: list[str] = []
lines.append(md_h("Tree-sitter 0.25 upgrade readiness", 1))
lines.append("")
current_runtime = read_current_runtime()
current_abi_range = RUNTIME_ABI_RANGES.get(current_runtime, (0, 0))
target_abi_range = RUNTIME_ABI_RANGES.get(TARGET_RUNTIME_MAJOR_MINOR, (0, 0))
lines.append(f"- Current runtime: `tree-sitter@{current_runtime}.x` (ABI {current_abi_range[0]}..{current_abi_range[1]})")
lines.append(f"- Target runtime: `tree-sitter@{TARGET_RUNTIME}` (ABI {target_abi_range[0]}..{target_abi_range[1]})")
lines.append("")
# ── Grammar peer-dep compatibility ───────────────────────────────
lines.append(md_h("Grammar compatibility", 2))
lines.append("| Grammar | npm latest | Peer dep | Satisfies 0.25? | ABI | Upstream ABI | Status |")
lines.append("|---|---|---|---|---|---|---|")
ready_count = 0
total_count = len(GRAMMARS)
for name, (upstream_repo, upstream_branch, parser_path) in sorted(GRAMMARS.items()):
# Fetch latest npm metadata.
info = npm_view_json(name)
fetch_failed = info is None
npm_version = "?"
peer_range = None
peer_optional = True
if info:
npm_version = info.get("version", "?")
peers = info.get("peerDependencies") or {}
peer_range = peers.get("tree-sitter")
meta = info.get("peerDependenciesMeta") or {}
ts_meta = meta.get("tree-sitter") or {}
peer_optional = ts_meta.get("optional", False) if peer_range else True
if fetch_failed:
peer_display = "? (fetch failed)"
compatible = False
else:
peer_display = peer_range or "none"
if peer_range and not peer_optional:
peer_display += " (required)"
compatible = satisfies_target(peer_range, TARGET_RUNTIME)
# Check installed ABI using the same parser_path from GRAMMARS.
installed_parser = GITNEXUS_DIR / "node_modules" / name / parser_path
if not installed_parser.is_file():
# Fallback to default location.
installed_parser = GITNEXUS_DIR / "node_modules" / name / "src" / "parser.c"
installed_abi = extract_language_version(installed_parser)
abi_display = str(installed_abi) if installed_abi else "?"
# Check upstream (main/master branch) ABI for unreleased work.
upstream_url = (
f"https://raw.githubusercontent.com/{upstream_repo}/"
f"{upstream_branch}/{parser_path}"
)
upstream_text = fetch_text(upstream_url)
upstream_abi = extract_abi_from_text(upstream_text) if upstream_text else None
upstream_abi_display = str(upstream_abi) if upstream_abi else "?"
# Determine status.
if fetch_failed:
status = "Unknown (fetch failed)"
blockers[name] = f"`{name}`: npm registry fetch failed — could not verify peer dep"
elif compatible:
status = "Ready"
ready_count += 1
elif upstream_abi and upstream_abi >= 15:
status = "Unreleased (ABI 15 on main)"
blockers[name] = f"`{name}`: ABI 15 on `{upstream_repo}` main but not published to npm"
else:
status = "Blocking"
blockers[name] = f"`{name}@{npm_version}`: peer `{peer_display}` incompatible with 0.25"
# Also check upstream package.json for relaxed peer dep.
if not compatible and not fetch_failed:
upstream_pkg_url = (
f"https://raw.githubusercontent.com/{upstream_repo}/"
f"{upstream_branch}/package.json"
)
upstream_pkg_text = fetch_text(upstream_pkg_url)
if upstream_pkg_text:
try:
upstream_pkg = json.loads(upstream_pkg_text)
upstream_peer = (upstream_pkg.get("peerDependencies") or {}).get("tree-sitter")
if upstream_peer and satisfies_target(upstream_peer, TARGET_RUNTIME):
status = "Unreleased (peer relaxed on main)"
blockers[name] = f"`{name}`: peer dep relaxed on `{upstream_repo}` main but not published to npm"
except json.JSONDecodeError:
pass
compat_icon = "Yes" if compatible else "**No**"
lines.append(
f"| `{name}` | {npm_version} | {peer_display} | {compat_icon} | {abi_display} | {upstream_abi_display} | {status} |"
)
lines.append("")
lines.append(f"**{ready_count}/{total_count}** grammars ready for `tree-sitter@{TARGET_RUNTIME}`.")
lines.append("")
# ── Vendored proto drift ─────────────────────────────────────────
lines.append(md_h("Vendored tree-sitter-proto", 2))
vendored_abi = extract_language_version(VENDOR_PROTO_DIR / "src" / "parser.c")
upstream_proto_url = (
f"https://raw.githubusercontent.com/{UPSTREAM_PROTO_OWNER}/"
f"{UPSTREAM_PROTO_REPO}/{UPSTREAM_PROTO_BRANCH}/src/parser.c"
)
upstream_proto_text = fetch_text(upstream_proto_url)
upstream_proto_abi = extract_abi_from_text(upstream_proto_text) if upstream_proto_text else None
sha_url = (
f"https://api.github.com/repos/{UPSTREAM_PROTO_OWNER}/"
f"{UPSTREAM_PROTO_REPO}/commits/{UPSTREAM_PROTO_BRANCH}"
)
sha_text = fetch_text(sha_url)
upstream_sha = "?"
if sha_text:
try:
upstream_sha = json.loads(sha_text).get("sha", "?")[:12]
except json.JSONDecodeError:
pass
local_proto_path = VENDOR_PROTO_DIR / "src" / "parser.c"
local_proto_text = local_proto_path.read_text(encoding="utf-8", errors="ignore") if local_proto_path.is_file() else ""
in_sync = bool(
upstream_proto_text
and local_proto_text.replace("\r\n", "\n")
== upstream_proto_text.replace("\r\n", "\n")
)
lines.append(f"- Upstream: `{UPSTREAM_PROTO_OWNER}/{UPSTREAM_PROTO_REPO}@{UPSTREAM_PROTO_BRANCH}` (HEAD `{upstream_sha}`)")
lines.append(f"- Upstream ABI: **{upstream_proto_abi}**")
lines.append(f"- Vendored ABI: **{vendored_abi}**")
lines.append(f"- In sync: {'yes' if in_sync else 'no — upstream has diverged'}")
if upstream_proto_abi and vendored_abi and upstream_proto_abi > vendored_abi:
can_upgrade = upstream_proto_abi <= target_abi_range[1]
lines.append(f"- Upstream ABI {upstream_proto_abi} {'is' if can_upgrade else 'is NOT'} within target runtime range ({target_abi_range[0]}..{target_abi_range[1]})")
if can_upgrade:
lines.append(f"- **Action:** after upgrading to tree-sitter@{TARGET_RUNTIME}, regenerate vendored parser.c from upstream `{upstream_sha}`")
else:
lines.append(f"- **Action:** wait for runtime upgrade beyond {TARGET_RUNTIME} that supports ABI {upstream_proto_abi}")
blockers["vendored-proto-abi"] = f"vendored tree-sitter-proto: upstream ABI {upstream_proto_abi} outside target range"
elif not in_sync:
lines.append("- **Action:** review upstream changes; vendored copy may need updating")
blockers["vendored-proto-sync"] = "vendored tree-sitter-proto: out of sync with upstream"
# ── Summary ──────────────────────────────────────────────────────
lines.append("")
lines.append(md_h("Summary", 2))
if blockers:
lines.append(f"**{len(blockers)} blocker(s) remaining:**\n")
for b in blockers.values():
lines.append(f"- {b}")
lines.append("")
lines.append("Upgrade to `tree-sitter@0.25` is **blocked**.")
else:
lines.append("All grammars are compatible. Upgrade to `tree-sitter@0.25` is **ready**.")
print("\n".join(lines))
return 1 if blockers else 0
if __name__ == "__main__":
sys.exit(main())
@@ -1,179 +0,0 @@
#!/usr/bin/env python3
"""Enforce the GitHub Actions concurrency convention.
See CONTRIBUTING.md -> "GitHub Actions — Concurrency Convention" for the rules.
Invoked from .github/workflows/ci-quality.yml. Runs locally too:
python3 .github/scripts/check-workflow-concurrency.py .github/workflows
Rules:
1. Every entry-point (non-reusable) workflow declares a top-level
`concurrency:` block.
2. Reusable workflows (on: workflow_call ONLY) do NOT declare one.
3. The `concurrency.group` expression MUST reference either
`${{ github.workflow }}` or one of the approved hardcoded literal prefixes
for workflows that are simultaneously entry-points AND reusable (on: push/
workflow_call). Two such exceptions are currently approved:
- `CI-` for ci.yml (the original canonical form)
- `docker-build-push-` for docker.yml
This is checked by substring containment rather than prefix match because
the group value is a conditional expression that resolves to a `CI-…` or
`docker-build-push-…` literal at runtime.
We deliberately do not use a YAML library — keeps the script dependency-free
on any vanilla runner. `on:` block parsing is line-based and handles both the
flat (`on: workflow_call`) and mapping (`on:\n workflow_call:`) forms.
"""
from __future__ import annotations
import pathlib
import re
import sys
REQUIRED_TOKENS = ("${{ github.workflow }}", "CI-", "docker-build-push-")
def is_reusable(lines: list[str]) -> bool:
"""Return True iff the workflow's `on:` block names only `workflow_call`."""
in_on = False
on_indent: int | None = None
keys: list[str] = []
for raw in lines:
# Skip blank lines and comments
stripped = raw.strip()
if not stripped or stripped.startswith("#"):
continue
indent = len(raw) - len(raw.lstrip(" "))
if not in_on:
if raw.startswith("on:"):
remainder = raw[len("on:"):].strip()
if not remainder:
# `on:` followed by indented mapping on next lines
in_on = True
on_indent = indent
continue
if remainder.startswith("[") and remainder.endswith("]"):
# Flow-style list: on: [workflow_call]
items = [
item.strip() for item in remainder.strip("[]").split(",")
]
return items == ["workflow_call"]
# Scalar form: on: workflow_call (or a single other event)
return remainder == "workflow_call"
continue
# Inside the `on:` block; stop when indentation returns to <= on_indent
if on_indent is not None and indent <= on_indent:
break
# Only consider keys at on_indent + indentation step (anything deeper
# is nested config like `types:`)
if ":" not in stripped:
continue
# Heuristic: first-level event keys are those with indent == on_indent + 2
# (the canonical step for a 2-space YAML doc). We collect all first-level
# keys by tracking the smallest indent seen inside the block.
keys.append((indent, stripped.split(":", 1)[0].strip()))
if not keys:
return False
# Take only the outermost-indented keys as the event list
min_indent = min(i for i, _ in keys)
events = [name for i, name in keys if i == min_indent]
return events == ["workflow_call"]
CONCURRENCY_RE = re.compile(r"^concurrency:\s*$")
GROUP_RE = re.compile(r"^\s+group:\s*(.+?)\s*$")
def extract_group_key(lines: list[str]) -> str | None:
"""Return the `group:` value of the top-level `concurrency:` block, or None."""
for idx, raw in enumerate(lines):
if CONCURRENCY_RE.match(raw):
# Scan forward until we leave the concurrency block (next top-level key
# is at column 0 and ends with `:`).
for follow in lines[idx + 1:]:
if follow and not follow.startswith(" ") and follow.rstrip().endswith(":"):
break
m = GROUP_RE.match(follow)
if m:
return m.group(1).strip().strip("'").strip('"')
break
return None
def has_top_level_concurrency(lines: list[str]) -> bool:
return any(CONCURRENCY_RE.match(raw) for raw in lines)
def check(workflows_dir: pathlib.Path) -> int:
fail = 0
files = sorted(
list(workflows_dir.glob("*.yml")) + list(workflows_dir.glob("*.yaml"))
)
for path in files:
lines = path.read_text(encoding="utf-8").splitlines()
reusable = is_reusable(lines)
has_conc = has_top_level_concurrency(lines)
if reusable:
if has_conc:
print(
f"::error file={path}::Reusable workflow (on: workflow_call) "
"must NOT declare its own concurrency block — it inherits "
"from the caller. See CONTRIBUTING.md -> GitHub Actions — "
"Concurrency Convention."
)
fail = 1
continue
if not has_conc:
print(
f"::error file={path}::Missing top-level concurrency block. "
"See CONTRIBUTING.md -> GitHub Actions — Concurrency Convention."
)
fail = 1
continue
group = extract_group_key(lines)
if group is None:
print(
f"::error file={path}::concurrency block is missing a "
"`group:` key."
)
fail = 1
continue
if not any(token in group for token in REQUIRED_TOKENS):
print(
f"::error file={path}::concurrency.group `{group}` must "
f"reference one of {REQUIRED_TOKENS} (use ${{{{ github.workflow }}}} "
"for normal entry-point workflows; use an approved literal prefix "
"only for workflows that are both entry-points AND reusable — "
"see CONTRIBUTING.md -> GitHub Actions — Concurrency Convention)."
)
fail = 1
return fail
def main(argv: list[str]) -> int:
if len(argv) != 2:
print(f"usage: {argv[0]} <workflows-dir>", file=sys.stderr)
return 2
workflows_dir = pathlib.Path(argv[1])
if not workflows_dir.is_dir():
print(f"not a directory: {workflows_dir}", file=sys.stderr)
return 2
return check(workflows_dir)
if __name__ == "__main__":
sys.exit(main(sys.argv))
+4 -4
View File
@@ -11,8 +11,8 @@ jobs:
outputs:
web_changed: ${{ steps.filter.outputs.web }}
steps:
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
- uses: dorny/paths-filter@fbd0ab8f3e69293af611ebaee6363fc25e6d187d # v3
- uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4
- uses: dorny/paths-filter@de90cc6fb38fc0963ad72b210f1f284cd68cea36 # v3
id: filter
with:
filters: |
@@ -26,7 +26,7 @@ jobs:
runs-on: ubuntu-latest
timeout-minutes: 20
steps:
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
- uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4
- uses: ./.github/actions/setup-gitnexus-web
@@ -74,7 +74,7 @@ jobs:
- name: Upload test results
if: always()
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1
uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4
with:
name: e2e-results
path: |
+6 -29
View File
@@ -8,8 +8,8 @@ jobs:
runs-on: ubuntu-latest
timeout-minutes: 5
steps:
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
- uses: actions/setup-node@48b55a011bda9f5d6aeb4c2d9c7362e8dae4041e # v6.4.0
- uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4
- uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4
with:
node-version: 20
cache: npm
@@ -21,8 +21,8 @@ jobs:
runs-on: ubuntu-latest
timeout-minutes: 10
steps:
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
- uses: actions/setup-node@48b55a011bda9f5d6aeb4c2d9c7362e8dae4041e # v6.4.0
- uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4
- uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4
with:
node-version: 20
cache: npm
@@ -34,7 +34,7 @@ jobs:
runs-on: ubuntu-latest
timeout-minutes: 10
steps:
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
- uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4
- uses: ./.github/actions/setup-gitnexus
- run: npx tsc --noEmit
working-directory: gitnexus
@@ -43,30 +43,7 @@ jobs:
runs-on: ubuntu-latest
timeout-minutes: 10
steps:
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
- uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4
- uses: ./.github/actions/setup-gitnexus-web
- run: npx tsc -b --noEmit
working-directory: gitnexus-web
# Enforces the convention documented in CONTRIBUTING.md → "GitHub Actions —
# Concurrency Convention":
# 1. Every entry-point (non-reusable) workflow declares a top-level
# `concurrency:` block.
# 2. Reusable workflows (`on: workflow_call` only) do NOT declare one —
# they inherit concurrency from the caller.
# 3. The concurrency group key starts with `${{ github.workflow }}` or
# the literal `CI-` prefix (the documented ci.yml exception for
# reusable-workflow-safe grouping).
# Reusability is detected by parsing each workflow's `on:` block, not an
# allowlist, so new reusable workflows never produce false positives.
workflow-convention:
name: Workflow concurrency convention
runs-on: ubuntu-latest
timeout-minutes: 5
steps:
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
- name: Validate workflow concurrency convention
shell: bash
run: |
set -euo pipefail
python3 .github/scripts/check-workflow-concurrency.py .github/workflows
+4 -14
View File
@@ -14,16 +14,6 @@ permissions:
contents: read # needed for sparse checkout of vitest.config.ts
pull-requests: write # needed to post sticky PR comment
# Concurrency convention: see CONTRIBUTING.md → "GitHub Actions — Concurrency Convention".
# Serialize sticky-comment writes per PR so two rapid CI completions don't race.
# Internal PRs surface in `pull_requests[0].number`. Fork PRs leave that array empty,
# so we fall back to `<head-repo-full-name>/<head-branch>`, which is stable across
# reruns and subsequent pushes for the same fork PR (unlike `workflow_run.id` which
# is unique per run and therefore does not serialize anything).
concurrency:
group: ${{ github.workflow }}-${{ github.event.workflow_run.pull_requests[0].number || format('{0}/{1}', github.event.workflow_run.head_repository.full_name, github.event.workflow_run.head_branch) }}
cancel-in-progress: false
jobs:
pr-report:
name: PR Report
@@ -36,7 +26,7 @@ jobs:
steps:
# ── Download artifacts from the CI run ────────────────────────
- name: Download artifacts
uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v7
uses: actions/github-script@60a0d83039c74a4aee543508d2ffcb1c3799cdea # v7
with:
script: |
const fs = require('fs');
@@ -123,7 +113,7 @@ jobs:
- name: Checkout (for vitest config)
if: steps.meta.outputs.skip != 'true'
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4
with:
sparse-checkout: gitnexus/vitest.config.ts
sparse-checkout-cone-mode: false
@@ -132,7 +122,7 @@ jobs:
- name: Fetch base branch coverage
if: steps.meta.outputs.skip != 'true'
id: base-coverage
uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v7
uses: actions/github-script@60a0d83039c74a4aee543508d2ffcb1c3799cdea # v7
with:
script: |
const fs = require('fs');
@@ -416,7 +406,7 @@ jobs:
- name: Comment on PR
if: steps.meta.outputs.skip != 'true'
uses: marocchino/sticky-pull-request-comment@0ea0beb66eb9baf113663a64ec522f60e49231c0 # v2
uses: marocchino/sticky-pull-request-comment@773744901bac0e8cbb5a0dc842800d45e9b2b405 # v2
with:
header: ci-report
number: ${{ steps.meta.outputs.pr_number }}
-110
View File
@@ -1,110 +0,0 @@
name: Scope Resolution Parity
# Reusable workflow — called from ci.yml. Does NOT declare concurrency;
# it inherits the caller's concurrency group per the convention documented
# in CONTRIBUTING.md → "GitHub Actions — Concurrency Convention".
#
# ── Purpose (RFC #909 Ring 3, §6.4 "Observability gates") ──────────────
# For every language in `MIGRATED_LANGUAGES` (exported from
# `gitnexus/src/core/ingestion/registry-primary-flag.ts`), run the
# resolver integration test at `test/integration/resolvers/<slug>.test.ts`
# TWICE on every PR:
#
# 1. `REGISTRY_PRIMARY_<LANG>=0` — legacy DAG path (guarantees we haven't
# broken the old path while migrating). Known legacy gaps may be skipped
# through the resolver test helper's expected-failure list.
# 2. `REGISTRY_PRIMARY_<LANG>=1` — registry-primary path (guarantees the
# new path carries the same behavior — the parity gate).
#
# BOTH must pass. The source of truth is the TypeScript constant — adding
# a language to that `Set` is the ONLY contributor action; CI auto-
# discovers it, runs parity, and the language's default production path
# flips to registry-primary in the same change.
#
# When the set is empty (e.g. mid-Ring-3 for every language), the parity
# matrix is skipped and the workflow reports success — no-op until a
# language is explicitly claimed migrated.
on:
workflow_call:
jobs:
discover:
name: Discover migrated languages
runs-on: ubuntu-latest
timeout-minutes: 5
outputs:
languages: ${{ steps.read.outputs.languages }}
has-any: ${{ steps.read.outputs.has-any }}
steps:
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
- uses: ./.github/actions/setup-gitnexus
- name: Extract MIGRATED_LANGUAGES from registry-primary-flag.ts
id: read
shell: bash
working-directory: gitnexus
run: |
set -euo pipefail
# `tsx` evaluates the TS source directly (no build step), imports
# the exported `Set`, and emits a GH-Actions-friendly JSON matrix.
LANGS=$(npx tsx scripts/ci-list-migrated-languages.ts)
COUNT=$(printf '%s' "$LANGS" | jq 'length')
HAS_ANY="false"
if [[ "$COUNT" -gt 0 ]]; then HAS_ANY="true"; fi
echo "languages=$LANGS" >> "$GITHUB_OUTPUT"
echo "has-any=$HAS_ANY" >> "$GITHUB_OUTPUT"
echo "Discovered $COUNT migrated language(s): $LANGS"
echo "Parity matrix will run: $HAS_ANY"
parity:
name: ${{ matrix.lang.slug }} parity
needs: discover
if: needs.discover.outputs.has-any == 'true'
runs-on: ubuntu-latest
timeout-minutes: 20
strategy:
# One language failing must not abort the others — we want the full
# parity matrix result on a single CI run so a reviewer sees every
# regression at once rather than one-at-a-time.
fail-fast: false
matrix:
lang: ${{ fromJSON(needs.discover.outputs.languages) }}
steps:
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
- uses: ./.github/actions/setup-gitnexus
with:
build: 'true'
- name: Verify resolver test file exists
shell: bash
working-directory: gitnexus
run: |
set -euo pipefail
TEST_FILE="test/integration/resolvers/${{ matrix.lang.slug }}.test.ts"
if [[ ! -f "$TEST_FILE" ]]; then
echo "::error title=Missing resolver test::\
Expected $TEST_FILE for '${{ matrix.lang.slug }}' (listed in \
MIGRATED_LANGUAGES). Either fix the slug or add the test file \
before listing this language as migrated."
exit 1
fi
- name: Resolver tests — legacy DAG (REGISTRY_PRIMARY_${{ matrix.lang.envvar }}=0)
shell: bash
working-directory: gitnexus
env:
FLAG_NAME: REGISTRY_PRIMARY_${{ matrix.lang.envvar }}
# Explicitly force the flag to `0` even though it also defaults to
# `MIGRATED_LANGUAGES.has(lang)` — once a language is in the set,
# the default flips to registry-primary, so an unset env var would
# silently re-run the same path as step #2. `env FOO=0 cmd` spawns
# `cmd` with the override scoped to just this invocation.
run: env "$FLAG_NAME=0" npx vitest run "test/integration/resolvers/${{ matrix.lang.slug }}.test.ts"
- name: Resolver tests — registry-primary (REGISTRY_PRIMARY_${{ matrix.lang.envvar }}=1)
shell: bash
working-directory: gitnexus
env:
FLAG_NAME: REGISTRY_PRIMARY_${{ matrix.lang.envvar }}
run: env "$FLAG_NAME=1" npx vitest run "test/integration/resolvers/${{ matrix.lang.slug }}.test.ts"
+3 -6
View File
@@ -9,7 +9,7 @@ jobs:
runs-on: ubuntu-latest
timeout-minutes: 25
steps:
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
- uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4
- uses: ./.github/actions/setup-gitnexus
with:
build: 'true'
@@ -41,12 +41,9 @@ jobs:
--outputFile=web-test-results.json
working-directory: gitnexus-web
- name: Run docker-server integration tests
run: node --test docker-server.test.mjs
- name: Upload test reports
if: always()
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1
uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4
with:
name: test-reports
path: |
@@ -66,7 +63,7 @@ jobs:
runs-on: ${{ matrix.os }}
timeout-minutes: 25
steps:
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
- uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4
- uses: ./.github/actions/setup-gitnexus
with:
build: 'true'
+12 -45
View File
@@ -9,28 +9,15 @@ on:
paths-ignore: ['**.md', 'docs/**', 'LICENSE']
workflow_call:
# Concurrency convention: see CONTRIBUTING.md → "GitHub Actions — Concurrency Convention".
# Hardcoded `CI-` prefix (not `${{ github.workflow }}`) because this workflow is
# invoked as a reusable workflow from publish.yml and release-candidate.yml. In
# called-workflow context `github.workflow` evaluation is ambiguous across GitHub
# Actions versions, and a prefix that could resolve to the caller's name would
# share a concurrency group with the caller → deadlock. A literal prefix is
# immune. Direct `push`/`pull_request` invocations use `CI-<ref>`; invocations
# from a reusable-workflow caller fall into a per-run-unique group that never
# serializes with the caller.
# cancel-in-progress is event-aware: cancel superseded PR runs, queue every other
# event (push to main, workflow_call from publish.yml, etc.).
concurrency:
group: ${{ (github.event_name == 'pull_request' || github.event_name == 'push') && format('CI-{0}', github.ref) || format('CI-nested-{0}', github.run_id) }}
cancel-in-progress: ${{ github.event_name == 'pull_request' }}
group: ci-${{ github.ref }}
cancel-in-progress: true
# ── Reusable workflow orchestration ─────────────────────────────────
# Each concern lives in its own workflow file for maintainability:
# ci-quality.yml — typecheck (tsc --noEmit)
# ci-tests.yml — unit + integration tests with coverage + cross-platform
# ci-e2e.yml — E2E tests (only when gitnexus-web/ changes)
# ci-scope-parity.yml — RFC #909 Ring 3 parity gate: legacy DAG + registry-primary
# both pass, per migrated language in the JSON registry
#
# Shared setup is DRY via .github/actions/setup-gitnexus composite action.
@@ -50,11 +37,6 @@ jobs:
permissions:
contents: read
scope-parity:
uses: ./.github/workflows/ci-scope-parity.yml
permissions:
contents: read
# ── Save PR metadata for the reporting workflow ─────────────────
# The ci-report.yml workflow (triggered by workflow_run) needs the
# PR number and job results to post a comment. We save them as an
@@ -63,7 +45,7 @@ jobs:
save-pr-meta:
name: Save PR Metadata
if: always() && github.event_name == 'pull_request'
needs: [quality, tests, e2e, scope-parity]
needs: [quality, tests, e2e]
runs-on: ubuntu-latest
timeout-minutes: 5
steps:
@@ -74,14 +56,12 @@ jobs:
QUALITY: ${{ needs.quality.result }}
TESTS: ${{ needs.tests.result }}
E2E: ${{ needs.e2e.result }}
SCOPE_PARITY: ${{ needs.scope-parity.result }}
run: |
mkdir -p pr-meta
echo "$PR_NUMBER" > pr-meta/pr_number
echo "$QUALITY" > pr-meta/quality_result
echo "$TESTS" > pr-meta/tests_result
echo "$E2E" > pr-meta/e2e_result
echo "$SCOPE_PARITY" > pr-meta/scope_parity_result
echo "$PR_NUMBER" > pr-meta/pr_number
echo "$QUALITY" > pr-meta/quality_result
echo "$TESTS" > pr-meta/tests_result
echo "$E2E" > pr-meta/e2e_result
# TODO(post-merge): remove backward-compat copies once ci-report.yml
# on main reads underscore names.
# Backward-compat: ci-report.yml on main still reads hyphenated
@@ -94,7 +74,7 @@ jobs:
cp pr-meta/e2e_result pr-meta/e2e-result
- name: Upload PR metadata
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1
uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4
with:
name: pr-meta
path: pr-meta/
@@ -104,7 +84,7 @@ jobs:
# Single required check for branch protection.
ci-status:
name: CI Gate
needs: [quality, tests, e2e, scope-parity]
needs: [quality, tests, e2e]
if: always()
runs-on: ubuntu-latest
timeout-minutes: 5
@@ -115,12 +95,10 @@ jobs:
QUALITY: ${{ needs.quality.result }}
TESTS: ${{ needs.tests.result }}
E2E: ${{ needs.e2e.result }}
SCOPE_PARITY: ${{ needs.scope-parity.result }}
run: |
echo "Quality: $QUALITY"
echo "Tests: $TESTS"
echo "E2E: $E2E"
echo "Scope parity: $SCOPE_PARITY"
echo "Quality: $QUALITY"
echo "Tests: $TESTS"
echo "E2E: $E2E"
if [[ "$QUALITY" != "success" ]] ||
[[ "$TESTS" != "success" ]]; then
echo "::error::Quality or test jobs failed"
@@ -130,14 +108,3 @@ jobs:
echo "::error::E2E job failed"
exit 1
fi
# scope-parity is a reusable workflow. With an empty migrated-
# languages list, its parity matrix is skipped and the outer
# workflow still reports `success`. If any entry's legacy-DAG or
# registry-primary run fails, the workflow reports `failure`.
# Accept only `success`; `skipped` would mean the entire
# discover job was skipped too (upstream failure), which should
# still block.
if [[ "$SCOPE_PARITY" != "success" ]]; then
echo "::error::Scope-resolution parity gate failed (RFC #909 Ring 3)"
exit 1
fi
+3 -4
View File
@@ -16,10 +16,9 @@ on:
issue_comment:
types: [created]
# Concurrency convention: see CONTRIBUTING.md → "GitHub Actions — Concurrency Convention".
# Serialize per-PR to avoid racing review comments.
concurrency:
group: ${{ github.workflow }}-${{ github.event.issue.number || github.event.pull_request.number }}
group: claude-review-${{ github.event.issue.number || github.event.pull_request.number }}
cancel-in-progress: false
jobs:
@@ -57,7 +56,7 @@ jobs:
# For issue_comment triggers, resolve the PR number, head SHA, and fork repo
- name: Resolve PR context
id: pr
uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v7
uses: actions/github-script@60a0d83039c74a4aee543508d2ffcb1c3799cdea # v7
with:
script: |
let pr;
@@ -77,7 +76,7 @@ jobs:
core.setOutput('branch', pr.head.ref);
- name: Checkout PR head
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4
with:
repository: ${{ steps.pr.outputs.repo }}
ref: ${{ steps.pr.outputs.sha }}
+3 -4
View File
@@ -10,10 +10,9 @@ on:
pull_request_review:
types: [submitted]
# Concurrency convention: see CONTRIBUTING.md → "GitHub Actions — Concurrency Convention".
# Serialize per-PR/issue to avoid racing comments.
concurrency:
group: ${{ github.workflow }}-${{ github.event.issue.number || github.event.pull_request.number || github.event.issue.id }}
group: claude-code-${{ github.event.issue.number || github.event.pull_request.number || github.event.issue.id }}
cancel-in-progress: false
jobs:
@@ -59,7 +58,7 @@ jobs:
# For PR-related triggers, resolve the fork repo so we can checkout correctly.
- name: Resolve PR context
id: pr
uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v7
uses: actions/github-script@60a0d83039c74a4aee543508d2ffcb1c3799cdea # v7
with:
script: |
// Determine if this event is PR-related
@@ -91,7 +90,7 @@ jobs:
core.setOutput('branch', pr.head.ref);
- name: Checkout repository
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4
with:
repository: ${{ steps.pr.outputs.is_pr == 'true' && steps.pr.outputs.repo || github.repository }}
ref: ${{ steps.pr.outputs.is_pr == 'true' && steps.pr.outputs.sha || '' }}
-259
View File
@@ -1,259 +0,0 @@
name: Docker Build & Push
on:
push:
tags:
- 'v*'
pull_request:
# workflow_dispatch is allowed for dry-run testing only. Publishing is still
# exclusively tag-driven so that every signed image corresponds 1:1 to a
# published `gitnexus@X.Y.Z` on npm. dry_run:true (the default) skips all
# push, sign, and attestation steps — the build runs but nothing is published.
workflow_dispatch:
inputs:
dry_run:
description: 'Build only — skip push, signing, and attestations'
required: false
default: true
type: boolean
workflow_call:
inputs:
tag:
description: >-
The full v-prefixed tag to build (e.g. v1.2.3-rc.1).
The tag must already exist in the repo and its tree must contain
a gitnexus/package.json whose version matches the tag.
required: true
type: string
# Concurrency convention: see CONTRIBUTING.md → "GitHub Actions — Concurrency Convention".
# Tag refs are unique per release, so distinct tags run in parallel.
# Re-pushes of the same tag serialize. cancel-in-progress: false — never cancel a publish mid-flight.
# Hardcoded `docker-build-push-` prefix (not `${{ github.workflow }}`) when invoked as a reusable
# workflow: in called-workflow context `github.workflow` is ambiguous and could resolve to the
# caller's name, sharing a concurrency group with the caller → deadlock.
# Direct tag-push invocations use `docker-build-push-<ref>`; workflow_call invocations get a
# per-run-unique group (they are already serialized by the caller's own concurrency group).
concurrency:
group: ${{ (github.event_name == 'push') && format('docker-build-push-{0}', github.ref) || format('docker-build-push-nested-{0}', github.run_id) }}
cancel-in-progress: false
jobs:
build-push:
name: Build & Push ${{ matrix.image.name }}
runs-on: ubuntu-latest
timeout-minutes: 60
permissions:
contents: read
packages: write
# Required for Cosign keyless signing via the OIDC token exchange,
# and for build provenance / SBOM attestations.
id-token: write
attestations: write
strategy:
fail-fast: false
matrix:
image:
# Static UI bundle. Small, fast image. Drop-in replacement for the
# legacy single-image setup at the same `gitnexus` repository slug
# is intentionally avoided — the UI now lives at `gitnexus-web` and
# the CLI/server takes the canonical `gitnexus` slug below.
- name: gitnexus-web
dockerfile: Dockerfile.web
slug: gitnexus-web
# CLI / `gitnexus serve` backend. Heavy native deps (tree-sitter,
# onnxruntime-node) live only in this image.
- name: gitnexus
dockerfile: Dockerfile.cli
slug: gitnexus
steps:
# Only the workflow_call path requires a non-empty `inputs.tag` — callers
# (e.g. release-candidate.yml) must pass the RC tag explicitly. On direct
# tag pushes the tag comes from `github.ref`, so `inputs.tag` is always
# empty and validating it here would break every real release (#1064).
# The downstream "Verify tag matches gitnexus/package.json version" step
# handles both event types by falling back to GITHUB_REF.
- name: Validate tag input
if: github.event_name == 'workflow_call'
shell: bash
env:
TAG_INPUT: ${{ inputs.tag }}
run: |
if [ -z "${TAG_INPUT}" ]; then
echo "::error::No tag provided to docker.yml — refusing to build/push."
exit 1
fi
# When triggered by workflow_call the caller passes the RC tag as an input;
# we check out that tag so the Dockerfile and package.json match the built image.
# For tag-push events github.ref is already the tag ref — no override needed.
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
with:
ref: ${{ inputs.tag || github.ref }}
# ── Lock the docker image version to the npm package version ──────────
# Mirrors the check in publish.yml: refuse to build unless the git tag
# exactly matches `gitnexus/package.json`'s version. This guarantees
# `ghcr.io/<owner>/gitnexus:X.Y.Z` always corresponds to the same
# `gitnexus@X.Y.Z` published to npm — no drift, no surprises.
- name: Verify tag matches gitnexus/package.json version
id: version
if: github.event_name != 'workflow_dispatch' && github.event_name != 'pull_request'
shell: bash
env:
# For workflow_call the tag comes from the caller input; for push events
# it is derived from GITHUB_REF (set to empty so the else-branch fires).
INPUT_TAG: ${{ inputs.tag }}
run: |
if [ -n "$INPUT_TAG" ]; then
TAG_VERSION="${INPUT_TAG#v}"
else
TAG_VERSION="${GITHUB_REF#refs/tags/v}"
fi
if ! [[ "$TAG_VERSION" =~ ^[0-9]+\.[0-9]+\.[0-9]+(-[a-zA-Z0-9.]+)?$ ]]; then
echo "::error::Tag does not follow semver: v$TAG_VERSION"
exit 1
fi
PKG_VERSION=$(node -p "require('./gitnexus/package.json').version")
if [ "$TAG_VERSION" != "$PKG_VERSION" ]; then
echo "::error::Tag version (v$TAG_VERSION) does not match gitnexus/package.json version ($PKG_VERSION)"
exit 1
fi
echo "version=$PKG_VERSION" >> "$GITHUB_OUTPUT"
echo "Version verified: $PKG_VERSION"
# Required for multi-platform (linux/arm64) emulation.
- name: Set up QEMU
uses: docker/setup-qemu-action@ce360397dd3f832beb865e1373c09c0e9f86d70a # v4.0.0
- name: Set up Docker Buildx
uses: docker/setup-buildx-action@4d04d5d9486b7bd6fa91e7baf45bbb4f8b9deedd # v4.0.0
- name: Install Cosign
uses: sigstore/cosign-installer@cad07c2e89fa2edd6e2d7bab4c1aa38e53f76003 # v4.1.1
- name: Log in to GitHub Container Registry
if: ${{ github.event_name != 'pull_request' && !inputs.dry_run }}
uses: docker/login-action@4907a6ddec9925e35a0a9e82d7399ccc52663121 # v4.1.0
with:
registry: ghcr.io
username: ${{ github.actor }}
password: ${{ secrets.GITHUB_TOKEN }}
# Docker Hub is a mirror of GHCR: same tags, same digests, same Cosign
# signatures. GHCR remains authoritative (it is the registry the
# ClusterImagePolicy globs against by default), but Docker Hub is the
# registry most users reach for first, so we publish there too.
# Requires repo secrets DOCKERHUB_USERNAME and DOCKERHUB_TOKEN (a scoped
# access token, NOT the account password) with write access to the
# `akonlabs/gitnexus` and `akonlabs/gitnexus-web` repos.
- name: Log in to Docker Hub
if: ${{ github.event_name != 'pull_request' && !inputs.dry_run }}
uses: docker/login-action@4907a6ddec9925e35a0a9e82d7399ccc52663121 # v4.1.0
with:
username: ${{ secrets.DOCKERHUB_USERNAME }}
password: ${{ secrets.DOCKERHUB_TOKEN }}
# Computes image tags and labels from the verified semver tag:
# v1.2.3 → :1.2.3, :1.2, :1, :latest (auto, only for non-prerelease)
# v1.2.3-rc.1 → :1.2.3-rc.1 only (prereleases never become :latest)
# `:latest` is only emitted for tag pushes thanks to `flavor: latest=auto`,
# ensuring it always points at a real npm-published version.
#
# For workflow_call invocations github.ref is the caller's branch ref, so
# the type=semver patterns would not match. In that case we add an explicit
# type=raw tag using the version already verified above, so the same
# image-naming rules apply regardless of how the workflow was triggered.
# NOTE: We check `inputs.tag` rather than `github.event_name` because in a
# reusable workflow the github context is inherited from the caller —
# `github.event_name` would still be "push", not "workflow_call".
- name: Extract Docker metadata
id: meta
uses: docker/metadata-action@030e881283bb7a6894de51c315a6bfe6a94e05cf # v6.0.0
with:
# Dual-registry publish. metadata-action expands the same tag set
# against every image ref listed here, and build-push-action pushes
# one build to all of them, so the GHCR and Docker Hub images share
# a digest and are byte-identical. The Docker Hub namespace
# (`akonlabs`) is hardcoded because it differs from the GitHub org
# (`abhigyanpatwari`) — `github.repository_owner` would produce the
# wrong ref.
images: |
ghcr.io/${{ github.repository_owner }}/${{ matrix.image.slug }}
docker.io/akonlabs/${{ matrix.image.slug }}
flavor: latest=auto
tags: |
type=semver,pattern={{version}}
type=semver,pattern={{major}}.{{minor}}
type=semver,pattern={{major}}
type=raw,value=${{ steps.version.outputs.version }},enable=${{ inputs.tag != '' }}
# Transient 502s from GHCR / Docker Hub / GHA cache during multi-platform
# exports are retried inside `.github/actions/docker-build-push-retry`
# (see docker/build-push-action#1422 — retry policy stays out of the
# upstream action). `ignore-error=true` on cache-to avoids cache export
# flakes failing an otherwise successful push.
- name: Build and push
id: build
uses: ./.github/actions/docker-build-push-retry
with:
context: .
file: ${{ matrix.image.dockerfile }}
platforms: linux/amd64,linux/arm64
push: ${{ github.event_name != 'pull_request' && !inputs.dry_run }}
tags: ${{ steps.meta.outputs.tags }}
labels: ${{ steps.meta.outputs.labels }}
cache-from: type=gha,scope=${{ matrix.image.slug }}
cache-to: type=gha,mode=max,scope=${{ matrix.image.slug }},ignore-error=true
# Cosign keyless signing. Each pushed tag is signed by the workflow's
# OIDC identity, so consumers can verify the image with the strict,
# fully-anchored identity regex (kept in sync with README.md and
# deploy/kubernetes/cluster-image-policy.yaml — update all three together).
# NOTE: `${...}` expression syntax is NOT evaluated inside YAML comments, so
# the example below uses literal `<owner>/<repo>` placeholders that consumers
# substitute themselves; the canonical, fully-rendered command lives in README.md.
# cosign verify ghcr.io/<owner>/<slug>:<tag> \
# --certificate-identity-regexp '^https://github\.com/<owner>/<repo>/\.github/workflows/docker\.yml@refs/tags/v[0-9]+\.[0-9]+\.[0-9]+(-[a-zA-Z0-9.]+)?$' \
# --certificate-oidc-issuer https://token.actions.githubusercontent.com
# Do NOT relax to `@.*` — that accepts signatures from any ref, including
# unprotected branches and PRs, and defeats the supply-chain guarantee.
- name: Sign image with Cosign (keyless)
if: ${{ github.event_name != 'pull_request' && !inputs.dry_run }}
env:
# Cosign v2 (installed by sigstore/cosign-installer above) makes
# keyless the default. COSIGN_EXPERIMENTAL is a v1-only opt-in flag
# that is now deprecated/no-op, so it is intentionally omitted.
DIGEST: ${{ steps.build.outputs.digest }}
TAGS: ${{ steps.meta.outputs.tags }}
run: |
# Sign every tag at the same digest so consumers can verify by tag or by digest.
# Use `while read` instead of `for $TAGS` to be robust against tags that
# could ever contain whitespace (the metadata-action output is newline-
# separated, not space-separated).
while IFS= read -r tag; do
[[ -n "$tag" ]] && cosign sign --yes "${tag}@${DIGEST}"
done <<< "$TAGS"
# Attach the SBOM produced by buildx as a verifiable attestation on the
# digest. Attestations are pushed as OCI referrers to the registry named
# in `subject-name`, so we call the action once per registry. The digest
# is identical across registries (same build, same push), so consumers
# pulling from either GHCR or Docker Hub see the same provenance.
- name: Generate build provenance attestation (GHCR)
if: ${{ github.event_name != 'pull_request' && !inputs.dry_run }}
uses: actions/attest-build-provenance@a2bbfa25375fe432b6a289bc6b6cd05ecd0c4c32 # v4.1.0
with:
subject-name: ghcr.io/${{ github.repository_owner }}/${{ matrix.image.slug }}
subject-digest: ${{ steps.build.outputs.digest }}
push-to-registry: true
- name: Generate build provenance attestation (Docker Hub)
if: ${{ github.event_name != 'pull_request' && !inputs.dry_run }}
uses: actions/attest-build-provenance@a2bbfa25375fe432b6a289bc6b6cd05ecd0c4c32 # v4.1.0
with:
subject-name: docker.io/akonlabs/${{ matrix.image.slug }}
subject-digest: ${{ steps.build.outputs.digest }}
push-to-registry: true
+2 -3
View File
@@ -8,9 +8,8 @@ on:
permissions:
pull-requests: write
# Concurrency convention: see CONTRIBUTING.md → "GitHub Actions — Concurrency Convention".
concurrency:
group: ${{ github.workflow }}-${{ github.event.pull_request.number }}
group: pr-desc-${{ github.event.pull_request.number }}
cancel-in-progress: true
jobs:
@@ -19,7 +18,7 @@ jobs:
timeout-minutes: 5
steps:
- name: Check PR description quality
uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v9.0.0
uses: actions/github-script@ed597411d8f924073f98dfc5c65a23a2325f34cd # v8
with:
script: |
const MIN_BODY_LENGTH = 50;
-113
View File
@@ -1,113 +0,0 @@
name: PR Conventional Labeler
# Two workflows in one file with different triggers, matched to the minimum
# privilege each needs:
#
# validate-title (on: pull_request)
# Fork-safe. Runs with the PR-head's read-only GITHUB_TOKEN. Uses
# `amannn/action-semantic-pull-request` to fail the check when the PR
# title doesn't follow the conventional-commit format. Because the
# action only reads the event payload, no fork-controlled code runs.
#
# autolabel (on: pull_request_target)
# Needs `pull-requests: write` to apply labels, so must be
# pull_request_target. Uses `release-drafter/release-drafter` with
# `dry-run: true` to only run the autolabeler against the
# `.github/release-drafter.yml` config from the BASE ref (release-
# drafter reads the config from the repository's default branch, NOT
# the PR head — verify with `gh api repos/release-drafter/release-drafter/contents/...`
# or a fork-test PR before merging if the repo is high-value).
# `sync-labels: true` in the config removes managed autolabels that no
# longer match (e.g. when `!` or `BREAKING CHANGE:` is dropped).
#
# Title format: <type>[(scope)][!]: <subject>
# Allowed types: feat, fix, perf, refactor, docs, test, ci, build, chore, revert, deps
# Trailing `!` on the type marks a breaking change.
# See CONTRIBUTING.md → "Pull request titles".
on:
pull_request:
# Title-only changes fire `edited`. `opened` and `reopened` cover creation.
# `synchronize` (push to the PR branch) is intentionally excluded — titles
# don't change on push, so it only wastes CI minutes and broadens the
# privileged-token exposure window on the autolabel job.
types: [opened, edited, reopened]
pull_request_target:
types: [opened, edited, reopened]
# Concurrency convention: see CONTRIBUTING.md → "GitHub Actions — Concurrency Convention".
# Include `github.event_name` so `pull_request` (validate-title) and
# `pull_request_target` (autolabel) runs for the same PR do NOT share a slot
# and therefore cannot cancel each other — a cancelled required-check would
# permanently block merge until the next title edit.
# Within each trigger the latest title edit still supersedes the prior run.
concurrency:
group: ${{ github.workflow }}-${{ github.event_name }}-${{ github.event.pull_request.number }}
cancel-in-progress: true
jobs:
validate-title:
# Fork-safe job — only runs on `pull_request` (not `pull_request_target`).
# Token is read-only; writes a commit status that branch protection can
# require before merge.
name: Validate PR title
if: github.event_name == 'pull_request'
runs-on: ubuntu-latest
timeout-minutes: 5
permissions:
pull-requests: read
steps:
# Pinned to v6.1.1. Verify SHA via:
# gh api repos/amannn/action-semantic-pull-request/git/refs/tags/v6.1.1
- uses: amannn/action-semantic-pull-request@48f256284bd46cdaab1048c3721360e808335d50 # v6.1.1
env:
GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
with:
types: |
feat
fix
perf
refactor
docs
test
ci
build
chore
revert
deps
requireScope: false
# Subject must be non-empty. We DO allow capitalized proper nouns
# (MCP, GitHub, API, etc.) — the old `^(?![A-Z]).+$` pattern
# rejected legitimate titles like `fix: MCP tool schema`.
subjectPattern: ^\S.{2,}$
subjectPatternError: |
The subject "{subject}" in PR title "{title}" is invalid.
Subjects must be at least 3 characters and must not start with whitespace.
wip: false
autolabel:
# Privileged job — runs only on `pull_request_target` so it can write labels.
# Never checks out fork code, never executes fork-controlled input; only
# reads the PR metadata (title, body, labels) and calls the GitHub API.
name: Apply conventional label
if: github.event_name == 'pull_request_target'
runs-on: ubuntu-latest
timeout-minutes: 5
permissions:
# `contents: read` is required — release-drafter's context.config() reads
# `.github/release-drafter.yml` from the repo's default branch via the
# repo-contents API. Without it the job silently 403s and no labels are
# applied. Job-level permissions nullify all unlisted scopes, so an
# explicit grant is necessary here.
contents: read
pull-requests: write
steps:
# Pinned to v7.2.0. Verify SHA via:
# gh api repos/release-drafter/release-drafter/git/refs/tags/v7.2.0
# v7 removed `disable-releaser`; use `dry-run: true` to only autolabel.
- uses: release-drafter/release-drafter@5de93583980a40bd78603b6dfdcda5b4df377b32 # v7.2.0
with:
config-name: release-drafter.yml
dry-run: true
env:
GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
+4 -13
View File
@@ -7,22 +7,13 @@ on:
# No workflow-level permissions — scoped per job below.
# Concurrency convention: see CONTRIBUTING.md → "GitHub Actions — Concurrency Convention".
# Tag refs are unique per release, so distinct tags run in parallel. Re-pushes of the
# same tag serialize. cancel-in-progress: false — never cancel a publish mid-flight.
concurrency:
group: ${{ github.workflow }}-${{ github.ref }}
cancel-in-progress: false
jobs:
ci:
uses: ./.github/workflows/ci.yml
permissions:
contents: read
actions: read
# No pull-requests:write — `ci.yml`'s save-pr-meta job is gated on
# `github.event_name == 'pull_request'`, so it never runs during a
# tag-triggered publish. Least-privilege for release-critical paths.
pull-requests: write
publish:
needs: ci
@@ -32,8 +23,8 @@ jobs:
contents: write
id-token: write
steps:
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
- uses: actions/setup-node@48b55a011bda9f5d6aeb4c2d9c7362e8dae4041e # v6.4.0
- uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4
- uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4
with:
node-version: 20
registry-url: https://registry.npmjs.org
@@ -91,7 +82,7 @@ jobs:
fi
- name: Create GitHub Release
uses: softprops/action-gh-release@b4309332981a82ec1c5618f44dd2e27cc8bfbfda # v2
uses: softprops/action-gh-release@a06a81a03ee405af7f2048a818ed3f03bbf83c7b # v2
with:
body_path: ${{ steps.changelog.outputs.fallback == 'false' && '/tmp/release-notes.md' || '' }}
generate_release_notes: ${{ steps.changelog.outputs.fallback == 'true' }}
-392
View File
@@ -1,392 +0,0 @@
name: Release Candidate
on:
# Publish a release-candidate build whenever a merge/commit lands on main.
# Docs/README-only changes are filtered out so prose updates don't
# cut a release.
push:
branches: [main]
paths-ignore:
- '**.md'
- 'docs/**'
- 'LICENSE'
workflow_dispatch:
inputs:
bump:
description: >-
Cycle policy. 'auto' (default) continues the active rc cycle on
this branch if there is one, otherwise bumps patch from latest.
Choose 'patch' / 'minor' / 'major' to explicitly start or reset
an rc cycle.
required: false
default: 'auto'
type: choice
options:
- auto
- patch
- minor
- major
force:
description: 'Publish even when HEAD already has an rc marker'
required: false
default: 'false'
type: choice
options:
- 'false'
- 'true'
# No workflow-level permissions — scoped per job below.
permissions: {}
# Concurrency convention: see CONTRIBUTING.md → "GitHub Actions — Concurrency Convention".
# Serialize all runs on the same ref (push + workflow_dispatch) to prevent two publishes
# racing on the rc counter. cancel-in-progress: false — the earlier merge publishes first.
concurrency:
group: ${{ github.workflow }}-${{ github.ref }}
cancel-in-progress: false
jobs:
# ── Skip when HEAD already has an rc marker (retry / duplicate dispatch) ──
# The marker is a lightweight tag `rc/<HEAD_SHA>` pushed *before* `npm
# publish`, so a failed publish leaves the marker in place and the guard
# refuses to re-publish. Recovery path after a partial failure:
# git push --delete origin rc/<HEAD_SHA> v<RC_VERSION>
# then redispatch with force=true.
guard:
name: Check if release candidate should run
runs-on: ubuntu-latest
timeout-minutes: 5
permissions:
contents: read
outputs:
should_run: ${{ steps.decide.outputs.should_run }}
head_sha: ${{ steps.decide.outputs.head_sha }}
steps:
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
with:
fetch-depth: 0
fetch-tags: true
- name: Decide
id: decide
shell: bash
env:
FORCE: ${{ inputs.force }}
BUMP_INPUT: ${{ inputs.bump }}
EVENT_NAME: ${{ github.event_name }}
run: |
set -euo pipefail
HEAD_SHA=$(git rev-parse HEAD)
echo "head_sha=$HEAD_SHA" >> "$GITHUB_OUTPUT"
if [ "$FORCE" = "true" ]; then
echo "Force flag set — running regardless of marker tag."
echo "should_run=true" >> "$GITHUB_OUTPUT"
exit 0
fi
# An explicit cycle reset on dispatch (bump != auto) also bypasses
# the dedup guard — the maintainer is deliberately asking for a
# new rc from the same commit.
if [ "$EVENT_NAME" = "workflow_dispatch" ] \
&& [ -n "${BUMP_INPUT:-}" ] \
&& [ "${BUMP_INPUT:-auto}" != "auto" ]; then
echo "Explicit bump=$BUMP_INPUT — bypassing marker dedup."
echo "should_run=true" >> "$GITHUB_OUTPUT"
exit 0
fi
# Dedup: is there already an rc/<HEAD_SHA> marker pointing at HEAD?
MARKER="rc/${HEAD_SHA}"
if git rev-parse "refs/tags/$MARKER" >/dev/null 2>&1; then
echo "HEAD already has marker $MARKER — skipping."
echo "should_run=false" >> "$GITHUB_OUTPUT"
else
echo "No marker on HEAD — proceeding."
echo "should_run=true" >> "$GITHUB_OUTPUT"
fi
# ── Reuse the stable CI workflow ─────────────────────────────────────
ci:
needs: guard
if: needs.guard.outputs.should_run == 'true'
uses: ./.github/workflows/ci.yml
permissions:
contents: read
secrets: inherit
# ── Publish the rc build to npm + create GitHub prerelease ───────────
publish:
name: Publish release candidate to npm
needs: [guard, ci]
if: needs.guard.outputs.should_run == 'true'
runs-on: ubuntu-latest
timeout-minutes: 20
permissions:
contents: write # push rc tag + marker
id-token: write # npm provenance
outputs:
vtag: ${{ steps.reltag.outputs.vtag }}
steps:
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
with:
fetch-depth: 0
fetch-tags: true
- uses: actions/setup-node@48b55a011bda9f5d6aeb4c2d9c7362e8dae4041e # v6.4.0
with:
node-version: 20
registry-url: https://registry.npmjs.org
cache: npm
cache-dependency-path: gitnexus/package-lock.json
- name: Build gitnexus-shared
run: npm install && npm run build
working-directory: gitnexus-shared
- name: Install gitnexus dependencies
run: npm ci
working-directory: gitnexus
- name: Resolve rc version
id: version
shell: bash
working-directory: gitnexus
env:
BUMP_INPUT: ${{ inputs.bump }}
EVENT_NAME: ${{ github.event_name }}
PKG_NAME: gitnexus
run: |
set -euo pipefail
# 1. Current published `latest` — the floor for any new rc base.
# Only E404 ("never published") falls back to package.json; any
# other error (network, auth, malformed response) fails fast.
NPM_STDERR_LATEST="$(mktemp)"
if CURRENT_LATEST="$(npm view "$PKG_NAME" version 2>"$NPM_STDERR_LATEST")"; then
:
else
if grep -q 'E404' "$NPM_STDERR_LATEST"; then
CURRENT_LATEST="$(node -p "require('./package.json').version")"
echo "Package not on registry (E404) — seeding from package.json: $CURRENT_LATEST"
else
echo "::error::npm registry unreachable for 'view version':" >&2
cat "$NPM_STDERR_LATEST" >&2
rm -f "$NPM_STDERR_LATEST"
exit 1
fi
fi
rm -f "$NPM_STDERR_LATEST"
CURRENT_LATEST_CLEAN="${CURRENT_LATEST%%-*}"
# 2. Full version list — needed for the counter and for active-cycle
# inference. Same E404-only fallback.
NPM_STDERR_VERSIONS="$(mktemp)"
if VERSIONS_JSON="$(npm view "$PKG_NAME" versions --json 2>"$NPM_STDERR_VERSIONS")"; then
:
else
if grep -q 'E404' "$NPM_STDERR_VERSIONS"; then
VERSIONS_JSON='[]'
echo "No published versions for $PKG_NAME yet (E404)."
else
echo "::error::npm registry unreachable for 'view versions':" >&2
cat "$NPM_STDERR_VERSIONS" >&2
rm -f "$NPM_STDERR_VERSIONS"
exit 1
fi
fi
rm -f "$NPM_STDERR_VERSIONS"
# 3. Base selection.
# - workflow_dispatch + bump ∈ {patch,minor,major} → explicit cycle
# reset from latest.
# - Everything else (push, or dispatch with bump=auto) → continue
# the highest active rc base > latest if one exists; else
# default to patch from latest.
if [ "$EVENT_NAME" = "workflow_dispatch" ] \
&& [ -n "${BUMP_INPUT:-}" ] \
&& [ "${BUMP_INPUT:-auto}" != "auto" ]; then
BASE="$(npx --yes -p semver@7 semver -i "$BUMP_INPUT" "$CURRENT_LATEST_CLEAN")"
echo "Explicit bump=$BUMP_INPUT → BASE=$BASE"
else
cat > /tmp/active_base.mjs <<'NODESCRIPT'
const latest = process.env.LATEST;
let v;
try { v = JSON.parse(process.env.VERSIONS_JSON); } catch { v = []; }
if (!Array.isArray(v)) v = [v];
const parse = s => s.split(".").map(n => parseInt(n, 10));
const gt = (a, b) => {
const [A, B] = [parse(a), parse(b)];
for (let i = 0; i < 3; i++) if (A[i] !== B[i]) return A[i] > B[i];
return false;
};
const bases = new Set();
for (const s of v) {
const m = /^(\d+\.\d+\.\d+)-rc\.\d+$/.exec(s);
if (m && gt(m[1], latest)) bases.add(m[1]);
}
if (!bases.size) { process.stdout.write(""); process.exit(0); }
const sorted = [...bases].sort((a, b) => gt(a, b) ? 1 : -1);
process.stdout.write(sorted[sorted.length - 1]);
NODESCRIPT
ACTIVE_BASE="$(LATEST="$CURRENT_LATEST_CLEAN" VERSIONS_JSON="$VERSIONS_JSON" node /tmp/active_base.mjs)"
if [ -n "$ACTIVE_BASE" ]; then
BASE="$ACTIVE_BASE"
echo "Continuing active rc cycle → BASE=$BASE"
else
BASE="$(npx --yes -p semver@7 semver -i patch "$CURRENT_LATEST_CLEAN")"
echo "No active rc cycle → patch bump from latest → BASE=$BASE"
fi
fi
# 4. Counter: 1 + max existing N for `${BASE}-rc.*`, else 1.
cat > /tmp/next_rc.mjs <<'NODESCRIPT'
const base = process.env.BASE;
const prefix = base + "-rc.";
let v;
try { v = JSON.parse(process.env.VERSIONS_JSON); } catch { v = []; }
if (!Array.isArray(v)) v = [v];
const ns = v
.filter(s => typeof s === "string" && s.startsWith(prefix))
.map(s => parseInt(s.slice(prefix.length), 10))
.filter(n => Number.isInteger(n) && n >= 0);
process.stdout.write(String(ns.length ? Math.max(...ns) + 1 : 1));
NODESCRIPT
NEXT_N="$(BASE="$BASE" VERSIONS_JSON="$VERSIONS_JSON" node /tmp/next_rc.mjs)"
RC_VERSION="${BASE}-rc.${NEXT_N}"
echo "Computed rc: $RC_VERSION"
# 5. Defensive: if the exact version already exists on the registry
# (e.g., race with another run), abort before re-publishing.
# Same E404-only pattern used above — a transient network
# failure must fail loudly, not pretend the version is missing.
NPM_STDERR_EXISTS="$(mktemp)"
if npm view "$PKG_NAME@$RC_VERSION" version 2>"$NPM_STDERR_EXISTS" >/dev/null; then
rm -f "$NPM_STDERR_EXISTS"
echo "::error::Version $RC_VERSION already exists on npm — aborting."
exit 1
else
if grep -qiE 'E404|not found' "$NPM_STDERR_EXISTS"; then
rm -f "$NPM_STDERR_EXISTS"
# Version doesn't exist — safe to proceed.
else
echo "::error::npm registry unreachable for existence check:" >&2
cat "$NPM_STDERR_EXISTS" >&2
rm -f "$NPM_STDERR_EXISTS"
exit 1
fi
fi
echo "base=$BASE" >> "$GITHUB_OUTPUT"
echo "rc_n=$NEXT_N" >> "$GITHUB_OUTPUT"
echo "rc_version=$RC_VERSION" >> "$GITHUB_OUTPUT"
- name: Apply rc version in-CI
shell: bash
working-directory: gitnexus
run: |
set -euo pipefail
npm version "${{ steps.version.outputs.rc_version }}" \
--no-git-tag-version --allow-same-version
- name: Build gitnexus
run: npm run build
working-directory: gitnexus
- name: Dry-run publish
run: npm publish --dry-run --tag rc
working-directory: gitnexus
# ── Acquire the "rc lock" BEFORE publishing (fixes idempotency) ─────
# We create two tags and push them atomically:
# v<RC_VERSION> → annotated tag on a detached release commit
# whose tree contains the rewritten package.json
# (so the tag's source matches the npm tarball)
# rc/<HEAD_SHA> → lightweight tag on HEAD; the guard's dedup key
# If this push fails, nothing is published — safe.
# If this push succeeds but npm publish fails, the marker stays on
# the remote and blocks retries until an operator manually cleans up.
- name: Create and push rc tags
id: reltag
shell: bash
working-directory: gitnexus
env:
RC_VERSION: ${{ steps.version.outputs.rc_version }}
HEAD_SHA: ${{ needs.guard.outputs.head_sha }}
run: |
set -euo pipefail
VTAG="v${RC_VERSION}"
MARKER="rc/${HEAD_SHA}"
git config user.name 'github-actions[bot]'
git config user.email '41898282+github-actions[bot]@users.noreply.github.com'
# Detached release commit with the version bump — keeps `main`
# pristine but gives the v-tag a tree that matches the published
# package contents exactly (fixes release-integrity gap).
git add package.json package-lock.json 2>/dev/null || git add package.json
git commit -m "release: ${VTAG}" --allow-empty
RELEASE_SHA="$(git rev-parse HEAD)"
echo "Detached release commit: $RELEASE_SHA"
# Annotated release tag on the release commit.
git tag -a "$VTAG" "$RELEASE_SHA" -m "$VTAG"
# Lightweight marker on the user-visible HEAD for the guard.
git tag "$MARKER" "$HEAD_SHA"
# Atomic push of both refs. If either would clobber an existing
# remote ref, the push fails and we stop before npm publish.
git push --atomic origin "refs/tags/$VTAG" "refs/tags/$MARKER"
echo "vtag=$VTAG" >> "$GITHUB_OUTPUT"
echo "marker=$MARKER" >> "$GITHUB_OUTPUT"
echo "release_sha=$RELEASE_SHA" >> "$GITHUB_OUTPUT"
- name: Publish to npm (rc dist-tag)
run: npm publish --provenance --access public --tag rc
working-directory: gitnexus
env:
NODE_AUTH_TOKEN: ${{ secrets.NPM_TOKEN }}
- name: Create GitHub prerelease
uses: softprops/action-gh-release@b4309332981a82ec1c5618f44dd2e27cc8bfbfda # v2
with:
tag_name: ${{ steps.reltag.outputs.vtag }}
name: Release Candidate ${{ steps.reltag.outputs.vtag }}
prerelease: true
make_latest: 'false'
generate_release_notes: true
body: |
Automated release candidate build from `main`.
**npm:** `npm install gitnexus@rc`
**Version:** `${{ steps.version.outputs.rc_version }}`
**Target base:** `${{ steps.version.outputs.base }}` (rc #${{ steps.version.outputs.rc_n }})
**Source commit (main):** ${{ needs.guard.outputs.head_sha }}
**Release commit (versioned tree):** ${{ steps.reltag.outputs.release_sha }}
Release candidates are pre-stable builds intended for early testing.
Stable releases remain on the `latest` dist-tag.
# ── Build & push RC Docker images ────────────────────────────────────
# Calls docker.yml as a reusable workflow so that the build, signing, and
# attestation logic stays in one place. The publish job exposes `vtag`
# (e.g. `v1.2.3-rc.1`) as an output so we can pass it as the tag input.
# RC images are signed with Cosign keyless signing; the OIDC identity
# will be `docker.yml@refs/heads/main` (the caller's ref) rather than a
# tag ref — see README.md § Docker for the correct verify command for RCs.
docker:
name: Build & Push RC Docker images
needs: [guard, publish]
if: needs.guard.outputs.should_run == 'true' && needs.publish.outputs.vtag != ''
uses: ./.github/workflows/docker.yml
# Reusable workflows do not receive caller secrets unless inherited; without
# this, DOCKERHUB_* / GITHUB_TOKEN are empty in docker.yml → "Username and
# password required" on Docker Hub login (see same pattern on `ci:` above).
secrets: inherit
permissions:
contents: read
packages: write
id-token: write
attestations: write
with:
tag: ${{ needs.publish.outputs.vtag }}
@@ -1,185 +0,0 @@
name: Tree-sitter Upgrade Readiness
# Monitors readiness for upgrading tree-sitter to 0.25.x. Tracks:
# 1. Peer-dep compatibility — can each grammar install cleanly with
# tree-sitter@0.25.0 without --legacy-peer-deps?
# 2. Vendored proto drift — has coder3101/tree-sitter-proto moved
# ahead of our vendored snapshot?
# See .github/scripts/check-tree-sitter-upgrade-readiness.py for the logic.
#
# Concurrency convention: see CONTRIBUTING.md → "GitHub Actions — Concurrency Convention".
on:
schedule:
# Daily at 09:00 UTC. Matches Dependabot's daily cadence so drift
# and dep PRs surface together.
- cron: '0 9 * * *'
workflow_dispatch:
pull_request:
paths:
- '.github/scripts/check-tree-sitter-upgrade-readiness.py'
- '.github/workflows/tree-sitter-upgrade-readiness.yml'
concurrency:
group: ${{ github.workflow }}-${{ github.ref }}
cancel-in-progress: ${{ github.event_name == 'pull_request' }}
permissions:
contents: read
jobs:
readiness:
name: Check upgrade readiness
runs-on: ubuntu-latest
timeout-minutes: 10
permissions:
contents: read
# Needed to open/update the tracking issue on scheduled runs.
issues: write
steps:
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
- uses: ./.github/actions/setup-gitnexus
with:
build: 'false'
- name: Run upgrade readiness check
id: readiness
shell: bash
env:
GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
run: |
set +e
python3 .github/scripts/check-tree-sitter-upgrade-readiness.py > drift-report.md
code=$?
set -e
echo "exit_code=$code" >> "$GITHUB_OUTPUT"
{
echo 'report<<DRIFT_EOF'
cat drift-report.md
echo 'DRIFT_EOF'
} >> "$GITHUB_OUTPUT"
echo "=== Report ==="
cat drift-report.md
# On PR runs, the script validates that it runs correctly. Blockers
# are informational — the scheduled run opens a tracking issue.
- name: Annotate PR with readiness status
if: github.event_name == 'pull_request' && steps.readiness.outputs.exit_code != '0'
run: |
echo "::warning::Tree-sitter 0.25 upgrade has blockers. See job output for the full readiness report."
- name: Upsert tracking issue on scheduled runs
if: >
github.event_name == 'schedule' &&
steps.readiness.outputs.exit_code != '0'
uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v9.0.0
env:
REPORT: ${{ steps.readiness.outputs.report }}
with:
script: |
const title = 'Tree-sitter 0.25 upgrade readiness';
const report = process.env.REPORT;
const body = report + '\n\n' +
'<sub>Generated daily by `.github/workflows/tree-sitter-upgrade-readiness.yml`. ' +
'Closes automatically when all blockers are resolved.</sub>';
const { data: open } = await github.rest.issues.listForRepo({
owner: context.repo.owner,
repo: context.repo.repo,
state: 'open',
labels: 'tree-sitter-drift',
per_page: 10,
});
const existing = open.find(i => i.title === title);
if (existing) {
// Extract ready/total count for the changelog comment.
const readyMatch = report.match(/\*\*(\d+)\/(\d+)\*\* grammars ready/);
const blockerMatch = report.match(/\*\*(\d+) blocker/);
const ready = readyMatch ? readyMatch[1] : '?';
const total = readyMatch ? readyMatch[2] : '?';
const blockers = blockerMatch ? blockerMatch[1] : '?';
// Find grammars whose status changed by diffing the old and
// new table rows. Each row looks like:
// | `tree-sitter-foo` | ... | Ready |
// | `tree-sitter-foo` | ... | Blocking |
const parseRows = (md) => {
const map = {};
for (const m of md.matchAll(/\| `(tree-sitter-[^`]+)` \|.*?\| (\S+(?:\s\S+)*?) \|$/gm)) {
map[m[1]] = m[2].trim();
}
return map;
};
const oldRows = parseRows(existing.body || '');
const newRows = parseRows(report);
const changes = [];
for (const [name, newStatus] of Object.entries(newRows)) {
const oldStatus = oldRows[name];
if (oldStatus && oldStatus !== newStatus) {
changes.push(`\`${name}\`: ${oldStatus} → ${newStatus}`);
}
}
const today = new Date().toISOString().slice(0, 10);
let comment = `**${today}:** ${ready}/${total} ready. ${blockers} blocker(s) remaining.`;
if (changes.length > 0) {
comment += '\n\nChanges:\n' + changes.map(c => `- ${c}`).join('\n');
} else {
comment += ' No changes from previous run.';
}
await github.rest.issues.createComment({
owner: context.repo.owner,
repo: context.repo.repo,
issue_number: existing.number,
body: comment,
});
await github.rest.issues.update({
owner: context.repo.owner,
repo: context.repo.repo,
issue_number: existing.number,
body,
});
core.info(`Updated existing issue #${existing.number}`);
} else {
const { data: created } = await github.rest.issues.create({
owner: context.repo.owner,
repo: context.repo.repo,
title,
body,
labels: ['tree-sitter-drift', 'dependencies'],
});
core.info(`Opened issue #${created.number}`);
}
- name: Close tracking issue on clean scheduled runs
if: >
github.event_name == 'schedule' &&
steps.readiness.outputs.exit_code == '0'
uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v9.0.0
with:
script: |
const title = 'Tree-sitter 0.25 upgrade readiness';
const { data: open } = await github.rest.issues.listForRepo({
owner: context.repo.owner,
repo: context.repo.repo,
state: 'open',
labels: 'tree-sitter-drift',
per_page: 10,
});
const existing = open.find(i => i.title === title);
if (existing) {
await github.rest.issues.createComment({
owner: context.repo.owner,
repo: context.repo.repo,
issue_number: existing.number,
body: 'All grammars are now compatible with tree-sitter@0.25. Upgrade is ready! Closing automatically.',
});
await github.rest.issues.update({
owner: context.repo.owner,
repo: context.repo.repo,
issue_number: existing.number,
state: 'closed',
});
core.info(`Closed issue #${existing.number}`);
}
+2 -4
View File
@@ -47,10 +47,8 @@ permissions:
issues: write
pull-requests: write
# Concurrency convention: see CONTRIBUTING.md → "GitHub Actions — Concurrency Convention".
# Single global slot — newest manual dispatch supersedes any in-flight run.
concurrency:
group: ${{ github.workflow }}
group: triage-sweep
cancel-in-progress: true
jobs:
@@ -76,7 +74,7 @@ jobs:
run: pip install -r .github/scripts/triage/requirements.txt
- name: Cache FastEmbed model weights
uses: actions/cache@27d5ce7f107fe9357f9df03efb73ab90386fccae # v5
uses: actions/cache@668228422ae6a00e4ad889ee87cd7109ec5666a7 # v5
with:
path: ${{ github.workspace }}/.fastembed_cache
key: fastembed-bge-small-en-v1.5
+1 -12
View File
@@ -23,7 +23,6 @@ Thumbs.db
.env
.env.local
.env.*.local
docker/.env
# Logs
*.log
@@ -82,11 +81,6 @@ GitNexus.sln
# Git worktrees
.worktrees/
# Vendored tree-sitter grammar build artifacts (created at install time,
# never committed). See docs/plans/2026-04-15-002-fix-tree-sitter-proto-vendor-deps-plan.md
gitnexus/vendor/**/build/
gitnexus/vendor/**/node_modules/
/github/scripts/triage/__pycache__/
.claude-flow/
@@ -101,9 +95,4 @@ gitnexus/vendor/**/node_modules/
.swarm/
local_docs/
# Local agent scratch / review prompts (never commit)
.tmp/
.agents/
.context/
local_docs/
+104 -113
View File
@@ -1,130 +1,117 @@
<!-- version: 1.7.0 -->
<!-- Last updated: 2026-04-23 -->
<!-- version: 1.3.0 -->
<!--
Metadata: version, last reviewed, scope, model policy, reference docs, changelog.
Last updated: 2026-03-22
-->
Last reviewed: 2026-04-23
Last reviewed: 2026-04-13
**Project:** GitNexus · **Environment:** dev · **Maintainer:** repository maintainers (see GitHub)
This file uses a standard agent header (version, scope, model policy, reference docs, changelog), adapted for this **TypeScript/JavaScript monorepo**.
## Scope
| Boundary | Rule |
|----------|------|
| **Reads** | `gitnexus/`, `gitnexus-web/`, `eval/`, plugin packages, `.github/`, `.gitnexus/`, docs. |
| **Writes** | Only paths required for the change; keep diffs minimal. Update lockfiles when deps change. |
| **Executes** | `npm`, `npx`, `node` under `gitnexus/` and `gitnexus-web/`; `uv run` for Python under `eval/`; documented CI/dev workflows. |
| **Off-limits** | Real `.env` / secrets, production credentials, unrelated repos, destructive git ops without confirmation. |
| | |
|--|--|
| **Reads** | Repository tree as needed for the task: `gitnexus/`, `gitnexus-web/`, `eval/`, plugin packages, `.github/`, `.gitnexus/` when present, and docs. |
| **Writes** | Only paths required for the requested change; keep diffs minimal. Update lockfiles when dependencies change. |
| **Executes** | `npm`, `npx`, `node` under `gitnexus/` and `gitnexus-web/`; `uv run` for Python under `eval/` when applicable; shell utilities for documented CI/dev workflows. |
| **Off-limits** | User secrets (e.g. real `.env`), production deployment credentials, unrelated repositories, destructive git history operations without explicit human confirmation. |
## Model Configuration
- **Primary:** Use a named model (e.g. Claude Sonnet 4.x). Avoid `Auto` or unversioned `latest` when reproducibility matters.
- **Notes:** The GitNexus CLI indexer does not call an LLM.
- **Primary:** Pin in **Cursor** (Settings → model). Use a **named** model (e.g. GPT-5.2, Claude Sonnet 4.x). Avoid relying on **Auto** when reproducibility or audit trail matters.
- **Fallback:** As configured in Cursor or your organization (do not encode `latest` or wildcards in automation configs).
- **Notes:** The open-source GitNexus CLI indexer does not call an LLM. Optional Nexus AI in the web UI uses end-user provider keys and models.
## Execution Sequence (complex tasks)
For multi-step work, state up front:
1. Which rules in this file and **[GUARDRAILS.md](GUARDRAILS.md)** apply (and any relevant Signs).
2. Current **Scope** boundaries.
3. Which **validation commands** you will run (`cd gitnexus && npm test`, `npx tsc --noEmit`).
Long sessions dilute instructions. For **multi-step** work, state up front:
On long threads, *"Remember: apply all AGENTS.md rules"* re-weights these instructions against context dilution.
1. Which rules in this file and **[GUARDRAILS.md](GUARDRAILS.md)** apply (and any relevant Signs).
2. Current **Scope** boundaries (Reads / Writes / Off-limits).
3. Which **validation commands** you will run (e.g. `cd gitnexus && npm test`, `npx tsc --noEmit`).
On very long threads, the human may add *“Remember: apply all AGENTS.md rules”* to re-weight rule tokens against context dilution.
## Claude Code hooks
**PreToolUse** hooks can block tools (e.g. `git_commit`) until checks pass. Adapt to this repo: `cd gitnexus && npm test` before commit.
Hooks enforce gates that prompts cannot. In **Claude Code**, **PreToolUse** hooks can block tools such as `git_commit` until checks pass. Adapt to this repo: e.g. `cd gitnexus && npm test` before commit.
## Context budget
## Context budget (Cursor / standards)
Commands and gotchas live under **Repo reference** below and in **[CONTRIBUTING.md](CONTRIBUTING.md)**. If always-on rules grow, split into **`.cursor/rules/*.mdc`** (globs). **Cursor:** project-wide rules in `.cursor/index.mdc`. **Claude Code:** load `STANDARDS.md` only when needed.
Generic “core standards” playbooks are often long and stack-specific. For this monorepo, commands and gotchas live under **Cursor Cloud specific instructions** below and in **[CONTRIBUTING.md](CONTRIBUTING.md)**. If always-on rules grow, split domain rules into **`.cursor/rules/*.mdc`** (globs). **Cursor:** project-wide rules live in **`.cursor/index.mdc`** (YAML frontmatter with `alwaysApply: true`). **Claude Code:** optionally load a **`STANDARDS.md`** only when needed (e.g. *“When writing new code, read STANDARDS.md”*) to save context.
## Reference docs
## Reference Documentation
- **[ARCHITECTURE.md](ARCHITECTURE.md)**, **[CONTRIBUTING.md](CONTRIBUTING.md)**, **[GUARDRAILS.md](GUARDRAILS.md)**
- **Call-resolution DAG (legacy path):** See ARCHITECTURE.md § Call-Resolution DAG. Typed 6-stage DAG inside the `parse` phase; language-specific behavior behind `inferImplicitReceiver` / `selectDispatch` hooks on `LanguageProvider`. Shared code in `gitnexus/src/core/ingestion/` must not name languages. Types: `gitnexus/src/core/ingestion/call-types.ts`.
- **Scope-resolution pipeline (RFC #909 Ring 3):** See ARCHITECTURE.md § Scope-Resolution Pipeline. Replaces the legacy DAG for languages in `MIGRATED_LANGUAGES` (see `registry-primary-flag.ts`). A language plugs in by implementing `ScopeResolver` (`scope-resolution/contract/scope-resolver.ts`) and registering it in `SCOPE_RESOLVERS`. CI parity gate runs BOTH paths per migrated language on every PR.
- **Cursor:** `.cursor/index.mdc` (always-on); `.cursor/rules/*.mdc` (glob-scoped). Legacy `.cursorrules` deprecated.
- **GitNexus:** skills in `.claude/skills/gitnexus/`; MCP rules in `gitnexus:start` block below.
- **This repository:** **[ARCHITECTURE.md](ARCHITECTURE.md)**, **[CONTRIBUTING.md](CONTRIBUTING.md)**, **[GUARDRAILS.md](GUARDRAILS.md)**.
- **Cursor:** `.cursor/index.mdc` (always-on rules); optional `.cursor/rules/*.mdc` (glob-scoped). Legacy `.cursorrules` is deprecated — see `.cursor/index.mdc`.
- **Optional local files:** `NOTES.md` (short vendor-neutral project snapshot). For handoffs, keep notes local (e.g., a scratch file outside the repo) rather than committing `HANDOFF.md`.
- **GitNexus:** skills under `.claude/skills/gitnexus/`; machine-oriented rules in the `gitnexus:start` … `gitnexus:end` block below.
## Changelog
| Date | Version | Change |
|------|---------|--------|
| 2026-04-23 | 1.7.0 | TypeScript added to `MIGRATED_LANGUAGES` (registry-primary call resolution by default). |
| 2026-04-20 | 1.6.0 | Added scope-resolution pipeline pointer (RFC #909 Ring 3); Python migrated to registry-primary. |
| 2026-04-19 | 1.5.0 | Cross-repo impact (#794): `impact`/`query`/`context` accept `repo: "@<group>"` + `service`. Removed `group_query`/`group_contracts`/`group_status` MCP tools; added `gitnexus://group/{name}/contracts` and `gitnexus://group/{name}/status` resources. |
| 2026-04-16 | 1.4.0 | Fixed: web UI description, pre-commit behavior, MCP tools (7->16), added gitnexus-shared, removed stale vite-plugin-wasm gotcha. |
| 2026-04-13 | 1.3.0 | Updated GitNexus index stats after DAG refactor. |
| 2026-03-24 | 1.2.0 | Fixed gitnexus:start block duplication. |
| 2026-03-23 | 1.1.0 | Updated agent instructions, references, Cursor layout. |
| 2026-03-22 | 1.0.0 | Initial structured header and changelog. |
| 2026-03-24 | 1.2.0 | Fixed gitnexus:start block duplication (was inlined in Reference Docs bullet). |
| 2026-03-23 | 1.1.0 | Updated agent instructions (sections, references, Cursor layout). |
| 2026-03-22 | 1.0.0 | Added structured agent header and changelog. |
---
<!-- gitnexus:start -->
# GitNexus — Code Intelligence
Indexed as **GitNexus** (4325 symbols, 10556 relationships, 300 execution flows). Use MCP tools to understand code, assess impact, and navigate safely.
This project is indexed by GitNexus as **GitNexus** (4325 symbols, 10556 relationships, 300 execution flows). Use the GitNexus MCP tools to understand code, assess impact, and navigate safely.
> If any tool warns the index is stale, run `npx gitnexus analyze` first.
> If any GitNexus tool warns the index is stale, run `npx gitnexus analyze` in terminal first.
## Always Do
- **MUST run impact analysis before editing any symbol.** `gitnexus_impact({target: "symbolName", direction: "upstream"})` — report blast radius to the user.
- **MUST run `gitnexus_detect_changes()` before committing** — verify only expected symbols and flows are affected.
- **MUST warn the user** if impact returns HIGH or CRITICAL risk.
- Explore unfamiliar code with `gitnexus_query({query: "concept"})` (process-grouped, ranked) instead of grepping.
- Full context on a symbol: `gitnexus_context({name: "symbolName"})`.
- **MUST run impact analysis before editing any symbol.** Before modifying a function, class, or method, run `gitnexus_impact({target: "symbolName", direction: "upstream"})` and report the blast radius (direct callers, affected processes, risk level) to the user.
- **MUST run `gitnexus_detect_changes()` before committing** to verify your changes only affect expected symbols and execution flows.
- **MUST warn the user** if impact analysis returns HIGH or CRITICAL risk before proceeding with edits.
- When exploring unfamiliar code, use `gitnexus_query({query: "concept"})` to find execution flows instead of grepping. It returns process-grouped results ranked by relevance.
- When you need full context on a specific symbol — callers, callees, which execution flows it participates in — use `gitnexus_context({name: "symbolName"})`.
## When Debugging
1. `gitnexus_query({query: "<error or symptom>"})` — find related execution flows
2. `gitnexus_context({name: "<suspect function>"})` — callers, callees, process participation
3. `READ gitnexus://repo/GitNexus/process/{processName}` — trace flow step by step
4. Regressions: `gitnexus_detect_changes({scope: "compare", base_ref: "main"})`
1. `gitnexus_query({query: "<error or symptom>"})` — find execution flows related to the issue
2. `gitnexus_context({name: "<suspect function>"})` — see all callers, callees, and process participation
3. `READ gitnexus://repo/GitNexus/process/{processName}` — trace the full execution flow step by step
4. For regressions: `gitnexus_detect_changes({scope: "compare", base_ref: "main"})` — see what your branch changed
## When Refactoring
- **Rename:** `gitnexus_rename({symbol_name: "old", new_name: "new", dry_run: true})` first. Graph edits are safe; text_search edits need manual review.
- **Extract/Split:** `gitnexus_context` (incoming/outgoing refs) then `gitnexus_impact` (upstream callers) before moving code.
- **After any refactor:** `gitnexus_detect_changes({scope: "all"})` to verify scope.
- **Renaming**: MUST use `gitnexus_rename({symbol_name: "old", new_name: "new", dry_run: true})` first. Review the preview — graph edits are safe, text_search edits need manual review. Then run with `dry_run: false`.
- **Extracting/Splitting**: MUST run `gitnexus_context({name: "target"})` to see all incoming/outgoing refs, then `gitnexus_impact({target: "target", direction: "upstream"})` to find all external callers before moving code.
- After any refactor: run `gitnexus_detect_changes({scope: "all"})` to verify only expected files changed.
## Never Do
- Edit a symbol without running `gitnexus_impact` first.
- Ignore HIGH/CRITICAL risk warnings.
- Rename with find-and-replace — use `gitnexus_rename`.
- Commit without `gitnexus_detect_changes()`.
- Add language-specific behavior to shared ingestion code (`gitnexus/src/core/ingestion/`) — use a `LanguageProvider` hook. Seeing `provider.mroStrategy === 'xxx'` or an import from `languages/xxx.ts` in shared code means stop and add a hook.
- NEVER edit a function, class, or method without first running `gitnexus_impact` on it.
- NEVER ignore HIGH or CRITICAL risk warnings from impact analysis.
- NEVER rename symbols with find-and-replace — use `gitnexus_rename` which understands the call graph.
- NEVER commit changes without running `gitnexus_detect_changes()` to check affected scope.
## Tools Quick Reference
| Tool | When to use | Example |
| Tool | When to use | Command |
|------|-------------|---------|
| `list_repos` | Discover indexed repos | `gitnexus_list_repos({})` |
| `query` | Find code by concept | `gitnexus_query({query: "auth validation"})` |
| `context` | 360-degree view of one symbol | `gitnexus_context({name: "validateUser"})` |
| `impact` | Blast radius before editing | `gitnexus_impact({target: "X", direction: "upstream"})` |
| `detect_changes` | Pre-commit scope check | `gitnexus_detect_changes({scope: "staged"})` |
| `rename` | Safe multi-file rename | `gitnexus_rename({symbol_name: "old", new_name: "new", dry_run: true})` |
| `cypher` | Custom graph queries | `gitnexus_cypher({query: "MATCH ..."})` |
| `api_impact` | Pre-change API route impact | `gitnexus_api_impact({route: "/api/users", method: "GET"})` |
| `route_map` | Route → handler → consumer map | `gitnexus_route_map({})` |
| `tool_map` | MCP/RPC tool definitions | `gitnexus_tool_map({})` |
| `shape_check` | Response shape vs consumer access | `gitnexus_shape_check({route: "/api/users"})` |
| `group_list` | List repo groups | `gitnexus_group_list({})` |
| `group_sync` | Rebuild group Contract Registry | `gitnexus_group_sync({name: "myGroup"})` |
| `query` (group mode) | Cross-repo search in a group (RRF-merged) | `gitnexus_query({repo: "@myGroup", query: "auth"})` |
| `context` (group mode) | 360° view across all member repos | `gitnexus_context({repo: "@myGroup", name: "validateUser"})` |
| `impact` (group mode) | Cross-repo blast radius via Contract Bridge | `gitnexus_impact({repo: "@myGroup", target: "X", direction: "upstream"})` |
> Group mode: pass `repo: "@<groupName>"` to fan out across all member repos, or `repo: "@<groupName>/<memberPath>"` to target a single member (path keys from `group.yaml`). Optional `service: "<monorepo/path>"` filters by service root. Group-level state (contracts, staleness) lives in the resources table below — there are **no** `group_query` / `group_context` / `group_impact` / `group_contracts` / `group_status` MCP tools.
>
> For a full walkthrough of setting up a group across multiple repos that communicate over gRPC, see [docs/guides/microservices-grpc.md](docs/guides/microservices-grpc.md).
## Impact Risk Levels
| Depth | Meaning | Action |
|-------|---------|--------|
| d=1 | WILL BREAK — direct callers/importers | MUST update |
| d=1 | WILL BREAK — direct callers/importers | MUST update these |
| d=2 | LIKELY AFFECTED — indirect deps | Should test |
| d=3 | MAY NEED TESTING — transitive | Test if critical path |
@@ -132,83 +119,87 @@ Indexed as **GitNexus** (4325 symbols, 10556 relationships, 300 execution flows)
| Resource | Use for |
|----------|---------|
| `gitnexus://repo/GitNexus/context` | Codebase overview, index freshness |
| `gitnexus://repo/GitNexus/context` | Codebase overview, check index freshness |
| `gitnexus://repo/GitNexus/clusters` | All functional areas |
| `gitnexus://repo/GitNexus/processes` | All execution flows |
| `gitnexus://repo/GitNexus/process/{name}` | Step-by-step execution trace |
| `gitnexus://group/{name}/contracts` | Group Contract Registry (provider/consumer rows + cross-links) |
| `gitnexus://group/{name}/status` | Per-member index + Contract Registry staleness report |
## Self-Check Before Finishing
Before completing any code modification task, verify:
1. `gitnexus_impact` was run for all modified symbols
2. No HIGH/CRITICAL warnings were ignored
3. `gitnexus_detect_changes()` confirms expected scope
4. All d=1 dependents were updated
2. No HIGH/CRITICAL risk warnings were ignored
3. `gitnexus_detect_changes()` confirms changes match expected scope
4. All d=1 (WILL BREAK) dependents were updated
## Keeping the Index Fresh
After committing code changes, the GitNexus index becomes stale. Re-run analyze to update it:
```bash
npx gitnexus analyze # basic refresh; preserves any existing embeddings
npx gitnexus analyze --embeddings # also generate embeddings for new/changed nodes
npx gitnexus analyze --drop-embeddings # explicit opt-in to wipe existing embeddings
npx gitnexus analyze
```
Check `.gitnexus/meta.json` `stats.embeddings` (0 = none). A plain `analyze` no longer drops existing vectors — pass `--drop-embeddings` to wipe.
If the index previously included embeddings, preserve them by adding `--embeddings`:
> Claude Code: PostToolUse hook detects a stale index after `git commit` and `git merge` and prompts the agent to run `analyze`. The hook does not invoke `analyze` itself.
```bash
npx gitnexus analyze --embeddings
```
## CLI Skills
To check whether embeddings exist, inspect `.gitnexus/meta.json` — the `stats.embeddings` field shows the count (0 means no embeddings). **Running analyze without `--embeddings` will delete any previously generated embeddings.**
| Task | Skill file |
|------|-----------|
| Architecture / "How does X work?" | `.claude/skills/gitnexus/gitnexus-exploring/SKILL.md` |
| Blast radius / "What breaks?" | `.claude/skills/gitnexus/gitnexus-impact-analysis/SKILL.md` |
| Debugging / "Why is X failing?" | `.claude/skills/gitnexus/gitnexus-debugging/SKILL.md` |
| Refactoring | `.claude/skills/gitnexus/gitnexus-refactoring/SKILL.md` |
| Tools/resources/schema reference | `.claude/skills/gitnexus/gitnexus-guide/SKILL.md` |
| CLI commands (index, status, clean, wiki) | `.claude/skills/gitnexus/gitnexus-cli/SKILL.md` |
> Claude Code users: A PostToolUse hook handles this automatically after `git commit` and `git merge`.
## CLI
| Task | Read this skill file |
|------|---------------------|
| Understand architecture / "How does X work?" | `.claude/skills/gitnexus/gitnexus-exploring/SKILL.md` |
| Blast radius / "What breaks if I change X?" | `.claude/skills/gitnexus/gitnexus-impact-analysis/SKILL.md` |
| Trace bugs / "Why is X failing?" | `.claude/skills/gitnexus/gitnexus-debugging/SKILL.md` |
| Rename / extract / split / refactor | `.claude/skills/gitnexus/gitnexus-refactoring/SKILL.md` |
| Tools, resources, schema reference | `.claude/skills/gitnexus/gitnexus-guide/SKILL.md` |
| Index, status, clean, wiki CLI commands | `.claude/skills/gitnexus/gitnexus-cli/SKILL.md` |
<!-- gitnexus:end -->
## Repo reference
## Cursor Cloud specific instructions
### Packages
### Repository structure
| Package | Path | Purpose |
|---------|------|---------|
| **CLI/Core** | `gitnexus/` | TypeScript CLI, indexing pipeline, MCP server. Published to npm. |
| **Web UI** | `gitnexus-web/` | React/Vite thin client. All queries via `gitnexus serve` HTTP API. |
| **Shared** | `gitnexus-shared/` | Shared TypeScript types and constants. |
| Claude Plugin | `gitnexus-claude-plugin/` | Static config for Claude marketplace. |
| Cursor Integration | `gitnexus-cursor-integration/` | Static config for Cursor editor. |
| Eval | `eval/` | Python evaluation harness (Docker + LLM API keys). |
This is a monorepo with two main products and supporting config packages:
| Component | Path | Purpose |
|-----------|------|---------|
| **GitNexus CLI/Core** | `gitnexus/` | Main product — TypeScript CLI, indexing pipeline, MCP server. Published to npm. |
| **GitNexus Web UI** | `gitnexus-web/` | React/Vite browser app — graph explorer + AI chat. Runs entirely in WASM. |
| Claude Plugin | `gitnexus-claude-plugin/` | Static config for Claude marketplace (no build). |
| Cursor Integration | `gitnexus-cursor-integration/` | Static config for Cursor editor (no build). |
| SWE-bench Eval | `eval/` | Python evaluation harness (optional; needs Docker + LLM API keys). |
### Running services
```bash
cd gitnexus && npm run dev # CLI: tsx watch mode
cd gitnexus-web && npm run dev # Web UI: Vite on port 5173
npx gitnexus serve # HTTP API on port 4747 (from any indexed repo)
```
- **CLI/Core**: `cd gitnexus && npm run dev` (tsx watch mode) or `npm run build && node dist/cli/index.js <command>`
- **Web UI**: `cd gitnexus-web && npm run dev` (Vite on port 5173)
- **Backend mode**: `cd <indexed-repo> && node /workspace/gitnexus/dist/cli/index.js serve` (HTTP API on port 3741 by default)
### Testing
**CLI / Core (`gitnexus/`)**
- `npm test` — full vitest suite (~2000 tests)
- `npm run test:unit` — unit tests only
- `npm run test:integration` — integration (~1850 tests). LadybugDB file-locking tests may fail in containers (known env issue).
- `npx tsc --noEmit` — typecheck
- **Unit tests**: `cd gitnexus && npm test` (vitest, ~2000 tests)
- **Integration tests**: `cd gitnexus && npm run test:integration` (vitest, ~1850 tests). Two LadybugDB file-locking tests (`lbug-core-adapter`, `search-core`) may fail in containerized environments due to `/tmp` locking limitations — this is a known environment issue, not a code bug.
- **TypeScript check**: `cd gitnexus && npx tsc --noEmit`
**Web UI (`gitnexus-web/`)**
- `npm test` — vitest (~200 tests)
- `npm run test:e2e` — Playwright (7 spec files; requires `gitnexus serve` + `npm run dev`)
- `npx tsc -b --noEmit` — typecheck
- **Unit tests**: `cd gitnexus-web && npm test` (vitest, ~200 tests)
- **E2E tests**: `cd gitnexus-web && E2E=1 npx playwright test` (Playwright, 5 tests — requires `gitnexus serve` + `npm run dev` running)
- **TypeScript check**: `cd gitnexus-web && npx tsc -b --noEmit`
**Pre-commit hook** (`.husky/pre-commit`): formatting (prettier via lint-staged) + typecheck for staged packages. Tests do **not** run in pre-commit — CI only.
No separate lint command is configured; TypeScript strict checking serves as the primary static analysis.
### Gotchas
- `npm install` in `gitnexus/` triggers `prepare` (builds via `tsc`) and `postinstall` (patches tree-sitter-swift, builds tree-sitter-proto). Native bindings need `python3`, `make`, `g++`.
- `tree-sitter-kotlin` and `tree-sitter-swift` are optional — install warnings expected.
- ESLint configured via `eslint.config.mjs` (TS, React Hooks, unused-imports). No `npm run lint` script; use `npx eslint .`. Prettier runs via lint-staged. CI checks both in `ci-quality.yml`.
- `npm install` in `gitnexus/` triggers `prepare` (builds via `tsc`) and `postinstall` (patches tree-sitter-swift). Native tree-sitter bindings require `python3`, `make`, and `g++` to be present.
- `tree-sitter-kotlin` and `tree-sitter-swift` are optional dependencies — install warnings for these are expected and non-blocking.
- The Web UI uses `vite-plugin-wasm` and requires `Cross-Origin-Opener-Policy`/`Cross-Origin-Embedder-Policy` headers for `SharedArrayBuffer` (handled automatically by Vite dev server).
- There is no ESLint/Prettier configuration in this repo.
+116 -437
View File
@@ -1,134 +1,99 @@
# Architecture — GitNexus
Monorepo: **CLI/MCP** (`gitnexus/`) + **browser UI** (`gitnexus-web/`).
This repository is a **monorepo** with two main products: the **CLI / MCP package** (`gitnexus/`) and the **browser UI** (`gitnexus-web/`). Supporting folders ship editor integrations and plugins without changing the core graph engine.
## Repository layout
| Path | Role |
|------|------|
| `gitnexus/` | npm package `gitnexus`: CLI, MCP server (stdio), HTTP API, ingestion pipeline, LadybugDB graph, embeddings. |
| `gitnexus-web/` | Vite + React thin client: graph explorer + AI chat. All queries via `gitnexus serve` HTTP API. |
| `gitnexus-shared/` | Shared TypeScript types and constants (consumed by CLI and Web). |
| `.claude/`, `gitnexus-claude-plugin/`, `gitnexus-cursor-integration/` | Agent skills and plugin metadata. |
| `eval/` | Evaluation harnesses for benchmarking tool usage. |
| `.github/` | CI workflows + composite actions (`setup-gitnexus/`, `setup-gitnexus-web/`). |
| `gitnexus/` | Published npm package `gitnexus`: CLI, MCP server (stdio), local HTTP API for bridge mode, ingestion pipeline, LadybugDB graph, embeddings (optional). |
| `gitnexus-web/` | Vite + React UI: in-browser indexing (WASM), graph visualization, optional connection to `gitnexus serve`. |
| `.claude/`, `gitnexus-claude-plugin/`, `gitnexus-cursor-integration/` | Packaged **skills** and plugin metadata so agents discover the same workflows as documented in `AGENTS.md`. |
| `eval/` | Evaluation harnesses and docs for benchmarking tool usage. |
| `.github/` | CI workflows (quality, unit, integration, E2E) and composite actions. |
## End-to-end flow: index → graph → tools
1. **Ingestion** — `analyze.ts` → `runFullAnalysis` (`run-analyze.ts`) → `runPipelineFromRepo` (`pipeline.ts`). DAG of 12 phases builds a `KnowledgeGraph` in memory, then loads into LadybugDB under `.gitnexus/`. Repo registered in `~/.gitnexus/registry.json` for MCP discovery.
1. **Ingestion** (`gitnexus analyze`)
- Entry: `gitnexus/src/cli/analyze.ts` → `runPipelineFromRepo` in `gitnexus/src/core/ingestion/pipeline.ts`.
- The pipeline is structured as a **DAG (Directed Acyclic Graph)** of named phases (see [Pipeline Phase DAG](#pipeline-phase-dag) below).
- Output is loaded into **LadybugDB** under **`.gitnexus/`** at the repo root (`lbug/`, `meta.json`, etc.). Optional **FTS** indexes and **embeddings** attach to the same store.
- The repo is registered in **`~/.gitnexus/registry.json`** so MCP can find it from any working directory.
2. **Persistence** — `repo-manager.ts` (paths, registry, KuzuDB cleanup). `lbug-adapter.ts` (graph load, queries, embedding batches).
2. **Persistence & metadata**
- `gitnexus/src/storage/repo-manager.ts` — paths, registry, cleanup of legacy Kuzu artifacts.
- `gitnexus/src/core/lbug/lbug-adapter.ts` — graph load, queries, embedding restore batches.
3. **Query layer** — three interfaces to the same backend:
- **MCP (stdio):** `mcp.ts` → `LocalBackend` → tools (`tools.ts`) + resources (`resources.ts`)
- **HTTP bridge:** `serve.ts` → Express (`api.ts`, `mcp-http.ts`) for web UI
- **CLI direct:** `gitnexus query|context|impact|cypher` in `tool.ts`
3. **Query & agents**
- **MCP (stdio):** `gitnexus/src/cli/mcp.ts` → `startMCPServer` → `LocalBackend` (`gitnexus/src/mcp/local/local-backend.ts`) opens registered repos and serves **tools** from `gitnexus/src/mcp/tools.ts` and **resources** from `gitnexus/src/mcp/resources.ts`.
- **Bridge HTTP:** `gitnexus/src/cli/serve.ts` → Express app in `gitnexus/src/server/api.ts` (CORS-limited) exposes REST + MCP-over-HTTP for the web UI.
- **CLI tools (no MCP):** `gitnexus query`, `context`, `impact`, `cypher` in `gitnexus/src/cli/tool.ts` call the same backend for scripts and CI.
4. **Staleness** — `staleness.ts` compares indexed `lastCommit` to `HEAD`, surfaces hints.
4. **Staleness**
- `gitnexus/src/mcp/staleness.ts` compares indexed `lastCommit` to `HEAD` and surfaces hints when the graph is behind git.
## MCP tools
## MCP tools (summary)
| Tool | Purpose |
|------|---------|
| `list_repos` | Discover indexed repos |
| `query` | Hybrid BM25 + vector search over the graph |
| `cypher` | Ad hoc Cypher against the schema |
| `context` | Callers, callees, processes for one symbol |
| `impact` | Blast radius (upstream/downstream) with risk summary |
| `detect_changes` | Map git diffs to affected symbols and processes |
| `rename` | Graph-assisted multi-file rename with `dry_run` preview |
| `api_impact` | Pre-change impact report for an API route handler |
| `route_map` | API route → handler → consumer mappings |
| `tool_map` | MCP/RPC tool definitions and handlers |
| `shape_check` | Response shape vs consumer property access mismatches |
| `group_list` | List repo groups or details for one group |
| `group_sync` | Rebuild group Contract Registry (`contracts.json`) and bridge graph |
`query`, `context`, and `impact` are group-aware: pass `repo: "@<groupName>"` (or `"@<groupName>/<memberPath>"` to scope to one member) plus optional `service: "<monorepo/path>"`. Group-mode `query` merges per-repo results via Reciprocal Rank Fusion; group-mode `impact` runs the local walk in the chosen member and fans out across boundaries via the Contract Bridge (`gitnexus/src/core/group/cross-impact.ts`). The previously-planned `group_query`, `group_context`, `group_impact`, `group_contracts`, `group_status` MCP tools are intentionally not introduced — group-level state is exposed via resources instead:
| Resource URI | Purpose |
|--------------|---------|
| `gitnexus://group/{name}/contracts` | Contract Registry (provider/consumer rows + cross-links) |
| `gitnexus://group/{name}/status` | Per-member index + Contract Registry staleness |
| `list_repos` | Discover indexed repositories when more than one is registered. |
| `query` | Natural-language / keyword search over the graph (hybrid BM25 + optional vectors). |
| `cypher` | Ad hoc **Cypher** against the schema (see resource `gitnexus://repo/{name}/schema`). |
| `context` | Callers, callees, processes for one symbol (with disambiguation). |
| `impact` | Blast radius (upstream/downstream) with depth and risk summary. |
| `detect_changes` | Map git diffs to affected symbols and processes. |
| `rename` | Graph-assisted rename with `dry_run` preview (`graph` vs `text_search` confidence). |
## Where to change what
| Concern | Start in |
|---------|----------|
| CLI commands/flags | `src/cli/` (`index.ts`, per-command modules) |
| Parsing/graph construction | `src/core/ingestion/pipeline-phases/` + `pipeline.ts` |
| Graph schema/DB | `src/core/lbug/` (`schema.ts`, `lbug-adapter.ts`) |
| MCP tools/resources | `src/mcp/server.ts`, `tools.ts`, `resources.ts` |
| Cross-repo groups (sync, contracts, `@<group>` routing) | `src/core/group/` (`service.ts`, `cross-impact.ts`, `sync.ts`, `bridge-db.ts`) |
| Search ranking | `src/core/search/` (BM25, hybrid fusion) |
| Embeddings | `src/core/embeddings/` + `src/core/run-analyze.ts` |
| Wiki generation | `src/core/wiki/` |
| Language support | `src/core/ingestion/languages/` + `tree-sitter-queries.ts` + `gitnexus-shared/src/languages.ts` |
| Import resolution | `src/core/ingestion/import-processor.ts` + `import-resolvers/configs/` + `model/resolution-context.ts` |
| Call resolution/MRO | `src/core/ingestion/call-processor.ts` + `model/resolve.ts` |
| Type extraction | `src/core/ingestion/type-extractors/` |
| Worker pool | `src/core/ingestion/workers/` |
| Web UI | `gitnexus-web/src/` |
| CI | `.github/workflows/*.yml`, `.github/actions/` |
> Paths above are relative to `gitnexus/` unless they start with `gitnexus-web/` or `.github/`.
---
| If you are changing… | Start in… |
|----------------------|-----------|
| CLI commands / flags | `gitnexus/src/cli/` (`index.ts`, per-command modules). |
| Parsing or graph construction | `gitnexus/src/core/ingestion/pipeline-phases/` (individual phase files), `pipeline.ts` (orchestrator). |
| Graph schema / DB access | `gitnexus/src/core/lbug/` (`schema.ts`, `lbug-adapter.ts`), `gitnexus/src/mcp/core/lbug-adapter.ts` if MCP-specific. |
| MCP protocol, tools, resources | `gitnexus/src/mcp/server.ts`, `tools.ts`, `resources.ts`. |
| Search ranking | `gitnexus/src/core/search/` (BM25, hybrid fusion). |
| Embeddings | `gitnexus/src/core/embeddings/`, phases in `analyze.ts`. |
| Wiki generation | `gitnexus/src/core/wiki/`. |
| Web UI behavior | `gitnexus-web/src/` (components, workers, graph client). |
| CI | `.github/workflows/*.yml`, `.github/actions/setup-gitnexus/`. |
## Pipeline Phase DAG
12 phases defined in `gitnexus/src/core/ingestion/pipeline-phases/`, each with explicit `deps` and typed output.
The ingestion pipeline is a DAG of named phases. Each phase is defined in its own file under `gitnexus/src/core/ingestion/pipeline-phases/` with explicit dependencies, typed inputs, and typed outputs.
```
scan → structure → [markdown, cobol] → parse → [routes, tools, orm]
→ crossFile → mro → communities → processes
```
| Phase | File | Deps | Output |
|-------|------|------|--------|
| `scan` | `scan.ts` | (root) | File paths + sizes |
| `structure` | `structure.ts` | `scan` | File/Folder nodes, CONTAINS edges, `allPathSet` |
| `markdown` | `markdown.ts` | `structure` | Section nodes, cross-link edges from .md/.mdx |
| `cobol` | `cobol.ts` | `structure` | COBOL program/paragraph/section nodes (regex, no tree-sitter) |
| `parse` | `parse.ts` + `parse-impl.ts` | `structure`, `markdown`, `cobol` | Symbol nodes, IMPORTS/CALLS/EXTENDS edges, extracted routes/tools/ORM queries |
| `routes` | `routes.ts` | `parse` | Route nodes + HANDLES_ROUTE edges (Next.js, Expo, PHP, decorators) |
| `tools` | `tools.ts` | `parse` | Tool nodes + HANDLES_TOOL edges |
| `orm` | `orm.ts` | `parse` | QUERIES edges (Prisma, Supabase) |
| `crossFile` | `cross-file.ts` + `cross-file-impl.ts` | `parse`, `routes`, `tools`, `orm` | Cross-file type propagation in topological import order |
| `mro` | `mro.ts` | `crossFile`, `structure` | METHOD_OVERRIDES + METHOD_IMPLEMENTS edges |
| `communities` | `communities.ts` | `mro`, `structure` | Community nodes + MEMBER_OF edges (Leiden algorithm) |
| `processes` | `processes.ts` | `communities`, `routes`, `tools`, `structure` | Process nodes + STEP_IN_PROCESS edges |
### Phase files
**Non-phase files in the same directory:** `parse-impl.ts`, `cross-file-impl.ts` (implementation), `wildcard-synthesis.ts` (whole-module import expansion), `orm-extraction.ts` (sequential ORM fallback), `types.ts`, `runner.ts`, `index.ts`.
### DAG runner
`runner.ts` — static phase graph, no plugins, compile-time type safety.
1. **Validation** — Kahn's topological sort. Rejects on: duplicate names, missing deps, cycles (DFS traces the concrete cycle path, e.g., `A -> B -> C -> A`, plus count of transitively blocked dependents).
2. **Execution** — sequential in topological order. Each phase receives:
- `ctx: PipelineContext` — shared mutable `KnowledgeGraph`, `repoPath`, progress callback, options
- `deps: ReadonlyMap<string, PhaseResult>` — **declared deps only** (runner filters the results map to prevent hidden coupling)
3. **Error handling** — wraps phase errors with the phase name, emits terminal `error` progress event, swallows progress handler errors to preserve the original cause.
4. **Timing** — per-phase `durationMs` in `PhaseResult`, dev-mode console logging.
**Design patterns:**
- **Single graph accumulator** — all phases mutate the same `KnowledgeGraph` in `ctx`; the graph is the primary output.
- **Typed phase access** — `getPhaseOutput<T>(deps, 'name')` for type-safe upstream results.
- **Binding accumulator lifecycle** — created in `parse`, disposed by `crossFile` (in `finally`). No other phase should take ownership.
- **Skippable phases** — `skipGraphPhases` omits MRO/communities/processes (faster tests). `skipWorkers` forces sequential parsing.
| Phase | File | Dependencies | What it does |
|-------|------|-------------|--------------|
| `scan` | `scan.ts` | (root) | Walk repo filesystem, collect paths + sizes |
| `structure` | `structure.ts` | `scan` | Build File/Folder nodes + CONTAINS edges |
| `markdown` | `markdown.ts` | `structure` | Extract headings and cross-links from .md/.mdx |
| `cobol` | `cobol.ts` | `structure` | Regex-based COBOL/JCL extraction |
| `parse` | `parse.ts` + `parse-impl.ts` | `structure`, `markdown`, `cobol` | Chunked tree-sitter parse, import/call/heritage resolution |
| `routes` | `routes.ts` | `parse` | Route registry (Next.js, Expo, PHP, decorator-based) |
| `tools` | `tools.ts` | `parse` | MCP/RPC tool detection |
| `orm` | `orm.ts` | `parse` | Prisma/Supabase ORM query edges |
| `crossFile` | `cross-file.ts` + `cross-file-impl.ts` | `parse`, `routes`, `tools`, `orm` | Cross-file type propagation in topological order |
| `mro` | `mro.ts` | `crossFile` | Method Resolution Order, METHOD_OVERRIDES edges |
| `communities` | `communities.ts` | `mro` | Leiden community detection |
| `processes` | `processes.ts` | `communities`, `routes`, `tools` | Execution flow detection, Route/Tool → Process links |
### How to add a new phase
1. Create `pipeline-phases/my-phase.ts` with a `PipelinePhase<MyOutput>` (name, deps, execute)
2. Export from `pipeline-phases/index.ts`
3. Add to `buildPhaseList()` in `pipeline.ts`
1. Create a new file in `pipeline-phases/` (e.g. `my-phase.ts`)
2. Define a `PipelinePhase<MyOutput>` object with `name`, `deps`, and `execute(ctx, deps)`
3. Export it from `pipeline-phases/index.ts`
4. Add it to the `buildPhaseList()` function in `pipeline.ts`
```typescript
import type { PipelinePhase, PhaseResult } from './types.js';
// pipeline-phases/my-phase.ts
import type { PipelinePhase, PipelineContext, PhaseResult } from './types.js';
import { getPhaseOutput } from './types.js';
import type { ParseOutput } from './parse.js';
@@ -136,367 +101,81 @@ export interface MyPhaseOutput { /* ... */ }
export const myPhase: PipelinePhase<MyPhaseOutput> = {
name: 'myPhase',
deps: ['parse'],
deps: ['parse'], // runs after parse completes
async execute(ctx, deps) {
const { allPaths } = getPhaseOutput<ParseOutput>(deps, 'parse');
// ... write to ctx.graph ...
// ... do work, write to ctx.graph ...
return { /* typed output */ };
},
};
```
---
### DAG runner
## Call-Resolution DAG
Typed 6-stage pipeline in `call-processor.ts` (inside the `parse` phase) that resolves method/function calls and emits CALLS edges. Language behavior plugs in at two `LanguageProvider` hook points (stages 3–4); shared code names no languages. Scope: call resolution only — import resolution, type extraction, heritage, and symbol-table population live in other phases.
### Stages
```
extract-call ──▶ classify-form ──▶ infer-receiver ──▶ select-dispatch ──▶ resolve-target ──▶ emit-edge
(1) (2) (3) [hook] (4) [hook] (5) (6)
```
| Stage | Produces | Location |
|-------|----------|----------|
| **extract-call** | `ExtractedCallSite` (name, form, receiver, argCount) | `call-extractors/` (per-language); runs in worker |
| **classify-form** | callForm (`free`/`member`/`constructor`) + arity | `call-analysis.ts` → `inferCallForm`; shared, runs in worker |
| **infer-receiver** | `ReceiverEnriched` (receiver type finalized) | `call-processor.ts`; shared default chain, then `inferImplicitReceiver` hook |
| **select-dispatch** | `DispatchDecision` (primary, fallback, ancestryView) | `selectDispatch` hook, falls back to shared default |
| **resolve-target** | `TieredCandidates` | `model/resolve.ts` → `lookupMethodByOwnerWithMRO` (MRO walk) |
| **emit-edge** | CALLS edge in graph | `call-processor.ts`; writes edge with confidence tier |
### Provider hooks
Both hooks are optional on `LanguageProvider`. Ruby is the only current implementer.
**`inferImplicitReceiver`** — called after shared infer-receiver defaults. Returns `ImplicitReceiverOverride | null`.
| | |
|---|---|
| Inputs | `calledName`, `callForm`, `receiverName`, `receiverTypeName`, `callNode` (AST), `filePath` |
| Non-null fields | `callForm`, `receiverName`, `receiverTypeName` (required); `receiverSource: 'implicit-self'` (fixed); `hint?` (opaque, passed to `selectDispatch`) |
| Null | Keep existing `ReceiverEnriched` state |
**`selectDispatch`** — called after infer-receiver (including hook). Returns `DispatchDecision | null`; null uses shared default (constructor → `primary:'constructor'`; typed receiver → `primary:'owner-scoped'`; else → `primary:'free'`).
| | |
|---|---|
| Inputs | `calledName`, `callForm`, `receiverName`, `receiverTypeName`, `receiverSource`, `hint` |
| Non-null fields | `primary: 'owner-scoped' \| 'free' \| 'constructor'`; `fallback?: 'free-arity-narrowed'`; `ancestryView?: 'instance' \| 'singleton'`; `hint?` |
**`DispatchDecision` field semantics:**
- `primary: 'owner-scoped'` — MRO walk from receiver's type; used when receiver type is known.
- `fallback: 'free-arity-narrowed'` — after owner-scoped miss, search free-call candidates by arity only (Ruby uses this for implicit-self calls that miss their owner's MRO).
- `ancestryView: 'singleton'` — walk singleton/class ancestry instead of instance ancestry (Ruby `def self.foo` bodies, so `extend`-ed methods are found).
### Adding language behavior
1. **Implicit receivers** — implement `inferImplicitReceiver`: return null if call already has a receiver; otherwise use `findEnclosingClassInfo` (`ast-helpers.ts`) to find the enclosing context, return `ImplicitReceiverOverride` with `receiverSource: 'implicit-self'`, and optionally set `hint` for `selectDispatch`.
2. **Custom dispatch** — implement `selectDispatch`: inspect `receiverSource` and `hint`, return `DispatchDecision` with `primary`, optional `fallback`, optional `ancestryView`; return null to keep shared defaults.
3. **MRO strategy** — confirm `mroStrategy` is `'first-wins'`, `'c3'`, `'ruby-mixin'`, or `'none'`; consumed by `lookupMethodByOwnerWithMRO`.
**Ruby example** (`languages/ruby.ts` + `utils/ruby-self-call.ts`): `inferImplicitReceiver` rewrites bare-identifier calls to `self.method` and sets `hint` to `'instance'`/`'singleton'`; `selectDispatch` uses hint for `ancestryView` and adds `fallback: 'free-arity-narrowed'` for implicit-self calls.
### Code references
| Module | Purpose |
|--------|---------|
| `core/ingestion/call-types.ts` | DAG types: `ReceiverEnriched`, `DispatchDecision`, `ImplicitReceiverOverride` |
| `core/ingestion/language-provider.ts` | Hook signatures: `inferImplicitReceiver`, `selectDispatch` |
| `core/ingestion/call-processor.ts` | `processCalls`: stages 3–6 |
| `core/ingestion/model/resolve.ts` | `lookupMethodByOwnerWithMRO`: stage 5 MRO walk |
| `core/ingestion/languages/ruby.ts` | Both hooks + `mroStrategy: 'ruby-mixin'` |
| `core/ingestion/utils/ruby-self-call.ts` | Bare-call rewrite for `inferImplicitReceiver` |
### Coexistence with the scope-resolution pipeline
The Call-Resolution DAG is the **legacy path**. RFC #909 Ring 3 introduces a parallel **scope-resolution pipeline** (next section) that replaces stages 1–6 with a scope-indexed registry lookup. Both paths ship side-by-side and are gated per-language via `MIGRATED_LANGUAGES` + the `REGISTRY_PRIMARY_<LANG>` env var.
- **Unmigrated language** → Call-Resolution DAG runs; scope-resolution phase is a no-op.
- **Migrated language** (currently: Python, C#) → scope-resolution owns CALLS/ACCESSES/USES emission; the legacy DAG gates off for that language via `isRegistryPrimary(lang)` checks in `call-processor.ts` and `import-processor.ts`.
- `import-processor` still populates `importMap` for migrated languages — heritage's `ctx.resolve` reads it to disambiguate parent classes. Only edge emission is gated.
- CI runs BOTH paths for every migrated language on every PR (`.github/workflows/ci-scope-parity.yml`); both must pass.
#### Same-graph guarantee
Edges emitted by the scope-resolution pipeline and edges emitted by the legacy DAG are indistinguishable to downstream consumers (MCP tools, HTTP API, embeddings, group bridge):
- **Node identity** — both paths use `generateId(...)` from `lib/utils.ts`, the same qualified-name keyspace, and the same node labels (`File`, `Folder`, `Class`, `Method`, `Function`, …). Overload disambiguation suffixes `parameterTypes` into the id consistently — see `scope-resolution/graph-bridge/ids.ts` and the legacy emitter in `call-processor.ts`.
- **Edge vocabulary** — both paths emit the same reasons: `'import-resolved' | 'global' | 'local-call' | 'same-file' | 'interface-dispatch' | 'read' | 'write'`. Migrating a language must not change which reasons consumers see for previously-resolved edges.
- **Confidence tier** — both paths attach a numeric `confidence` to each edge using the same scale.
The CI parity workflow (`.github/workflows/ci-scope-parity.yml`) runs both paths against every migrated language's fixture corpus and fails on any divergence.
#### Semantic-model source of truth
Two independent invariants.
**ParsedFile = the AST-level truth.** `ParsedFile` (`gitnexus-shared/src/scope-resolution/parsed-file.ts`) is the single per-file artifact both resolution paths consume. Scope-resolution passes MUST NOT build a parallel parse representation. If a per-language hook needs AST-level facts that `ParsedFile` doesn't expose, it should reuse the orchestrator's `treeCache` (`RunScopeResolutionInput.treeCache`) rather than re-invoking `parser.parse(...)` on its own — the C# `populateNamespaceSiblings` hook is the reference implementation of this pattern.
**SemanticModel = the symbol-level truth.** `SemanticModel` (`gitnexus/src/core/ingestion/model/semantic-model.ts`) is the authoritative store for every symbol-indexed lookup (by `nodeId`, `simpleName`, `qualifiedName`, or `filePath`). Both paths read from here:
- Legacy Call-Resolution DAG → `call-processor` Tier 1/2/3 via `model.symbols.lookupExactAll`, `model.methods.lookupMethodByName`, `model.types.lookupClassByName`, `lookupMethodByOwnerWithMRO`.
- Scope-resolution pipeline → `findOwnedMember`, `pickOverload`, `findExportedDefByName` all consult `model.methods` / `model.fields` / `model.symbols`.
The scope-resolution pipeline additionally carries `WorkspaceResolutionIndex` for `Scope`-valued lookups (`classScopeByDefId`, `moduleScopeByFile`) that `SemanticModel` structurally cannot hold. No symbol-indexed duplicates exist outside `SemanticModel`.
**Write / read phase contract.** The model is mutable during three ordered phases and read-only afterward:
```
Phase 1: legacy parse ──► symbolTable.add fans into types/methods/fields
Phase 2: scope-resolution ──► reconcileOwnership() registers corrected ownerIds
Phase 3: finalize ──► model.attachScopeIndexes(bundle) — one-shot freeze
─────────────────────────── phase boundary ───────────────────────────
Read phase: all resolution passes + MCP + HTTP + embeddings see
SemanticModel (read-only handle); writes are type-errors.
```
`runScopeResolution` narrows `MutableSemanticModel` → `SemanticModel` at the phase boundary so downstream passes physically cannot mutate the model even accidentally.
**Transitional: reconciliation pass.** `reconcileOwnership` (`scope-resolution/pipeline/reconcile-ownership.ts`) is a shim for languages whose legacy extractor doesn't resolve `enclosingClassId` at parse time (Python class-body methods are the canonical case). It walks `parsed.localDefs[i].ownerId` after `populateOwners` and registers any missed methods/fields into the model. Idempotent — safe to re-run, safe alongside languages whose legacy extractor already carries `ownerId` (C#).
The architectural end state is for every language's parse-time extractor to emit the correct `ownerId` directly, making reconciliation a no-op (tracked as a follow-up refactor). The dev-mode validator `validateOwnershipParity` surfaces any drift via `onWarn` under `NODE_ENV !== 'production' && VALIDATE_SEMANTIC_MODEL !== '0'`.
References: `semantic-model.ts` file-head (full write/read contract); `contract/scope-resolver.ts` Contract Invariant I9 (scope-resolution-side rule).
---
## Scope-Resolution Pipeline (RFC #909 Ring 3)
Language-agnostic registry-primary resolver. Replaces the Call-Resolution DAG for migrated languages. Adding a language is one interface implementation (`ScopeResolver`) plus two registrations — no changes to shared code, no new pipeline phase.
### Pipeline stages
```
ParsedFile[] (extractParsedFile per file)
│ finalizeScopeModel (+ provider hooks)
▼
ScopeResolutionIndexes
│ resolveReferenceSites (via MethodRegistry.lookup)
▼
ReferenceIndex
│ emitReceiverBoundCalls ── FIRST
│ emitFreeCallFallback ── THEN
│ emitReferencesViaLookup ── LAST (uses handledSites)
│ emitImportEdges
▼
KnowledgeGraph (IMPORTS / CALLS / ACCESSES / INHERITS / USES)
```
Orchestrator: `runScopeResolution(input, provider)` in `scope-resolution/pipeline/run.ts`.
Pipeline phase: `scopeResolutionPhase` in `scope-resolution/pipeline/phase.ts` — iterates `SCOPE_RESOLVERS ∩ MIGRATED_LANGUAGES`, reads per-file Trees from the parse phase's `scopeTreeCache`, disposes the cache at the end.
### `ScopeResolver` contract
Single interface a language implements to plug into the pipeline. Contract fully documented in `scope-resolution/contract/scope-resolver.ts`.
| Hook | Purpose |
|------|---------|
| `languageProvider` | Base `LanguageProvider` (tree-sitter query, `emitScopeCaptures`, import/binding interpreters, hooks) |
| `populateOwners(parsed)` | Fill deferred `ownerId` fields on method defs (captures can't always know the owning class at parse time) |
| `buildMro(graph, parsed, nodeLookup)` | Produce `mroByClassDefId: Map<DefId, DefId[]>` — C3, Ruby-mixin, or first-wins per language |
| `resolveImportTarget(target, fromFile, allFiles)` | `(rawImportPath, sourceFile) → targetFilePath` (PEP-328 for Python, etc.) |
| `mergeBindings(existing, incoming, scopeId)` | Shadowing / LEGB precedence |
| `arityCompatibility` | Provider consumed by registry during `MethodRegistry.lookup` Step 2 |
| `importEdgeReason` | Confidence-tier string for IMPORTS edge reason field |
| `propagatesReturnTypesAcrossImports?` | Opt out of cross-file return-type propagation (default on) |
| `fieldFallbackOnMethodLookup?` | Statically-typed languages turn this OFF — the heuristic over-connects (default on) |
| `unwrapCollectionAccessor?` | Property-style collection views (`data.Values` on Dictionary-like receivers) — default off |
| `collapseMemberCallsByCallerTarget?` | One CALLS edge per (caller, target) instead of per-site — default off |
| `populateNamespaceSiblings?` | Cross-file implicit visibility (compiler-implicit namespace sharing) — default off; ctx carries `treeCache` |
| `hoistTypeBindingsToModule?` | Walk up to Module scope when looking up a method's return-type typeBinding — default off; enable only when bindings are stored at module level |
### Per-language registration
1. Implement `ScopeResolver` in `languages/<lang>/scope-resolver.ts`.
2. Add entry to `SCOPE_RESOLVERS` in `scope-resolution/pipeline/registry.ts`.
3. Add the language to `MIGRATED_LANGUAGES` in `registry-primary-flag.ts` when the shadow-harness corpus parity ≥ 99% fixtures / ≥ 98% corpus.
CI auto-discovers the set via `tsx`. No workflow edit required.
### Code references
| Module | Purpose |
|--------|---------|
| `scope-resolution/contract/scope-resolver.ts` | `ScopeResolver` interface + shared types |
| `scope-resolution/pipeline/run.ts` | Generic orchestrator |
| `scope-resolution/pipeline/phase.ts` | Pipeline-phase wrapper (deps: `parse`, `structure`) |
| `scope-resolution/pipeline/registry.ts` | `SCOPE_RESOLVERS` map |
| `scope-resolution/passes/*.ts` | Reference-resolution passes (receiver-bound, free-call fallback, compound-receiver, MRO, cross-file return-type propagation) |
| `scope-resolution/graph-bridge/*.ts` | CLI-local translation from resolved references → `KnowledgeGraph` edges |
| `scope-resolution/scope/*.ts` | Generic scope-chain walkers + namespace targets |
| `scope-resolution/workspace-index.ts` | Build-once O(1) lookup index |
| `registry-primary-flag.ts` | `MIGRATED_LANGUAGES` set + `isRegistryPrimary(lang)` |
| `languages/python/index.ts` | Python `ScopeResolver` hooks + known-limitation docs |
| `languages/python/captures.ts` | `emitPythonScopeCaptures` (honors cross-phase Tree cache) |
| `languages/csharp/index.ts` | C# `ScopeResolver` hooks + known-limitation docs |
| `languages/csharp/captures.ts` | `emitCsharpScopeCaptures` (honors cross-phase Tree cache) |
| `languages/csharp/namespace-siblings.ts` | Cross-file implicit-namespace visibility hook (reads `treeCache`) |
### Performance notes
- **Cross-phase Tree cache**: parse phase writes Trees into `scopeTreeCache` (separate from the chunk-local `astCache`) ONLY for languages with `emitScopeCaptures`. Scope-resolution reads from it to skip the second parse. Cleared at end of the phase. Workers leave the cache empty — Trees can't cross MessageChannels; cache miss = fresh parse. `PROF_SCOPE_RESOLUTION=1` emits hit/miss counters and a worker-engaged warning.
- **Typed relationship iteration**: heritage + MRO walk only the EXTENDS / IMPLEMENTS / HAS_METHOD edges via `iterRelationshipsByType`, not the full relationship map.
- **Workspace-resolution-index**: O(1) `findOwnedMember` / `findExportedDef` / `classScopeByDefId` built once per run.
- **SCC-ordered cross-file return-type propagation** (PR #1050): `propagateImportedReturnTypes` walks `indexes.sccs` in reverse-topological order (leaves first), so multi-hop alias chains like `models.User → service.user → app.user` collapse to the terminal class in a single linear pass. Within each importer, the source module's `typeBindings` is chain-followed BEFORE mirroring (so we mirror terminal types, not intermediate refs), and the importer's own `typeBindings` is chain-followed AFTER mirroring (so local `const x = importedFn()` resolves before downstream importers run). Cyclic SCCs reach a partial fixpoint within a single pass without iterating to convergence — see the `ts-circular` cross-file-binding fixture which only asserts pipeline-no-throw. PROF output (`PROF_SCOPE_RESOLUTION=1`) splits `finalize` from `propagate` so quadratic regressions in the chain-follow surface independently.
---
## Language-agnostic graph feeding
16 languages → single unified graph. Four abstraction layers:
```
Unified Graph Schema (44 node types, 21 relationship types)
↑
Unified Resolution (3-tier name lookup + MRO walk)
↑
Language Providers (import semantics, type config, export checker, MRO strategy)
↑
Tree-Sitter Queries (per-language S-expressions, unified capture tags)
```
### Language providers
Each language implements `LanguageProvider` (`language-provider.ts`). Key fields:
| Field | Purpose |
|-------|---------|
| `id`, `extensions` | Language identity and file matching |
| `treeSitterQueries` | S-expression queries for AST extraction |
| `importSemantics` | `named` / `wildcard-leaf` / `wildcard-transitive` / `namespace` |
| `importResolver` | Language-specific path → file resolution |
| `exportChecker` | Public/exported symbol detection |
| `typeConfig` | Type annotation extraction rules |
| `mroStrategy` | `first-wins` / `c3` / `none` |
16 providers in `languages/index.ts` via `satisfies Record<SupportedLanguages, LanguageProvider>` — missing a language is a compile error.
### Unified capture tags
Per-language tree-sitter queries use different AST node names but produce the **same semantic capture tags**: `@definition.class`, `@definition.function`, `@call.name`, `@import.source`, `@heritage.extends`. Downstream extraction needs no language branching. Defined in `tree-sitter-queries.ts`.
### Import resolution
Per-language import resolution uses the **configs + factory** pattern (like call/method/class extractors). Each language declares an `ImportResolutionConfig` in `import-resolvers/configs/`, listing an ordered chain of `ImportResolverStrategy` functions. `createImportResolver()` (in `resolver-factory.ts`) composes them: first non-null result wins. Low-level helpers shared across strategies live alongside the configs in `import-resolvers/` (e.g. `go.ts`, `rust.ts`, `python.ts`).
Unified 3-tier algorithm (`model/resolution-context.ts`), per-language `importSemantics` controls which tier activates:
| Tier | Confidence | Mechanism |
|------|-----------|-----------|
| 1 — same-file | 0.95 | Symbol table for caller's file |
| 2 — import-scoped | 0.9 | `NamedImportMap` chains (named) or all files in `importMap` (wildcard) |
| 3 — global | 0.5 | O(1) index lookups: class, impl, callable. Fallback only |
| Import strategy | Languages | Behavior |
|----------------|-----------|----------|
| `named` | TS, JS, Java, C#, Rust, PHP, Kotlin | Only explicitly imported names visible |
| `wildcard-leaf` | Go, Ruby, Swift, Dart | Whole-package import, no transitive re-exports |
| `wildcard-transitive` | C, C++ | `#include` closure chains through re-exports |
| `namespace` | Python | Module aliases resolved at call site |
### Chunked parse-and-resolve
`parse` processes files in ~20 MB byte-budget chunks to bound memory. Per chunk:
1. Worker pool dispatches files (or sequential fallback via `skipWorkers`)
2. Each worker: detect language → load grammar → run queries → return unified `ParseWorkerResult`
3. Synthesize wildcard bindings (`wildcard-synthesis.ts`)
4. Resolve imports and heritage
5. Collect `BindingAccumulator` entries for cross-file propagation
Workers: `workers/worker-pool.ts`, `workers/parse-worker.ts`.
### Heritage and MRO
All languages emit unified `ExtractedHeritage` (child, parent, `EXTENDS`/`IMPLEMENTS`). MRO phase walks the heritage graph using per-language strategy:
- **`first-wins`** — Java, C#, C++, TS, Ruby, Go
- **`c3`** — Python (C3 linearization)
- **`none`** — single-inheritance languages
Unified walk: `lookupMethodByOwnerWithMRO()` in `model/resolve.ts`.
---
## Full analysis flow
`runFullAnalysis` in `run-analyze.ts` orchestrates everything around the pipeline:
```
CLI (analyze.ts) → runFullAnalysis(repoPath, options, callbacks)
1. Early exit if lastCommit == HEAD (unless --force) [0%]
2. Cache existing embeddings from prior index [0%]
3. runPipelineFromRepo() → KnowledgeGraph [0-60%]
4. Clean up legacy KuzuDB files [60%]
5. initLbug() → loadGraphToLbug() via CSV streaming [60-85%]
6. Create FTS indexes (File, Function, Class, Method...) [85-90%]
7. Restore cached embeddings (batch insert) [88%]
8. Generate new embeddings if --embeddings [90-98%]
9. Save metadata + register repo + update .gitignore [98-100%]
10. Generate AI context files (AGENTS.md, CLAUDE.md) [100%]
```
**Options:** `--force` (rebuild regardless), `--embeddings` (opt-in, skipped if >50k nodes), `--skipGit`, `--noStats`.
## Storage
```
<repo>/.gitnexus/
├── lbug # LadybugDB database
├── lbug.wal # Write-ahead log
├── lbug.lock # Single-writer lock
└── meta.json # lastCommit, indexedAt, stats
~/.gitnexus/
└── registry.json # Global repo registry (MCP discovery)
```
Managed by `repo-manager.ts`.
## LadybugDB schema
Defined in `lbug/schema.ts`. Separate node tables per type, single `CodeRelation` table.
**Node tables:** File, Folder, Function, Class, Interface, Method, Constructor, CodeElement, Struct, Enum, Macro, Typedef, Union, Namespace, Trait, Impl, TypeAlias, Const, Static, Property, Record, Delegate, Annotation, Template, Module, Community, Process, Route, Tool, Section, Embedding.
**Relation types** (`CodeRelation.type`): CONTAINS, DEFINES, CALLS, IMPORTS, EXTENDS, IMPLEMENTS, HAS_METHOD, HAS_PROPERTY, ACCESSES, METHOD_OVERRIDES, METHOD_IMPLEMENTS, MEMBER_OF, STEP_IN_PROCESS, HANDLES_ROUTE, FETCHES, HANDLES_TOOL, ENTRY_POINT_OF.
## Embeddings and search
**Embeddings** (`src/core/embeddings/`): Snowflake arctic-embed-xs (384D). Embeddable: File, Function, Class, Method, Interface. Incremental via SHA1 content hash. Separate `Embedding` table.
**Search** (`src/core/search/`): Hybrid BM25 + semantic vector, merged via Reciprocal Rank Fusion (K=60).
The runner (`pipeline-phases/runner.ts`) validates the DAG at startup (detects cycles and missing deps via topological sort), then executes phases in dependency order. Each phase receives:
- `ctx: PipelineContext` — shared graph, repoPath, progress callback
- `deps: Map<string, PhaseResult>` — outputs from all upstream phases
## Known limitations
### Overloaded method resolution
Node IDs use arity suffix (`#<paramCount>`): `Method:file:Class.method#1` vs `#2`.
Method and Constructor node IDs include an arity suffix (`#<paramCount>`) to
disambiguate overloaded methods. Two overloads with different parameter counts
produce distinct graph nodes: `Method:file:Class.method#1` vs
`Method:file:Class.method#2`.
**Same-arity disambiguation:** type-hash suffix `~type1,type2` when collision detected and type annotations present. Languages without types (Python, Ruby, JS) use arity-only. TS/JS overload signatures excluded (collapse to implementation body). See #651.
**Same-arity overload disambiguation:** When two overloads share the same
parameter count but differ in types (e.g. `save(int)` vs `save(String)`), a
type-hash suffix `~type1,type2` is appended to produce distinct node IDs:
`Method:file:Class.save#1~int` vs `Method:file:Class.save#1~String`. The suffix
is only added when a same-arity collision is detected within a class and all
parameters have non-null type annotations. Languages without type info (Python,
Ruby, JS) fall back to arity-only IDs. TypeScript/JavaScript overload signatures
are intentionally excluded from type-hashing because they are declaration-only
contracts that should collapse to the implementation body's node ID. See issue
\#651.
**C++ const-qualified:** `$const` suffix after type-hash when non-const collision exists: `Method:file:Container.begin#0$const`.
**C++ const-qualified overload disambiguation:** Methods overloaded by const
qualification (e.g. `begin()` vs `begin() const`) are disambiguated via an
`isConst` property and a `$const` ID suffix appended to the const-qualified
variant when a non-const collision exists. The `$const` suffix appears after the
type-hash suffix: e.g. `Method:file:Container.begin#0$const`.
**Generic/template types:** type-hash uses `rawType` (full AST text including generics): `~vector<int>` vs `~vector<std::string>`.
**Generic/template type preservation in type-hash:** The type-hash suffix uses
`rawType` (full AST text including generic/template args) rather than the
simplified `type` from `extractSimpleTypeName`. This means C++ template overloads
like `process(vector<int>)` vs `process(vector<string>)` produce distinct IDs:
`~vector<int>` vs `~vector<std::string>`. Java generic overloads like
`process(List<String>)` vs `process(List<Integer>)` are a compile error due to
type erasure, so this gap is theoretical for Java.
**ID stability:** collision-only tags mean IDs change when overloads are added. `save#1` becomes `save#1~int` when `save(String)` is added.
**ID stability on first overload:** Type and const tags are collision-only. When
a class has `save(int)` as its only `save` method, the ID is `save#1` (no tag).
Adding `save(String)` changes the original to `save#1~int`. This is correct for
fresh analysis but means IDs are not stable across overload additions. Future
incremental re-analysis should account for this.
**Variadic matching:** confidence 0.7 when one side is variadic and the other has fixed count.
**Variadic method matching:** When one side is variadic (`parameterCount`
undefined) and the other has a fixed count, `METHOD_IMPLEMENTS` edges are
emitted with confidence 0.7 instead of 1.0. Variadic methods like
`foo(String... args)` may superficially match `foo(String s)` by type but
are not guaranteed to be interchangeable across all languages (Java/Kotlin
accept this via varargs sugar; TypeScript, C#, Rust do not).
**METHOD_IMPLEMENTS confidence tiering:**
**Confidence tiering** for `METHOD_IMPLEMENTS` edges:
| Match quality | Confidence |
|---|---|
| Exact parameter types match | 1.0 |
| Arity match, types unavailable | 1.0 |
| Variadic vs fixed | 0.7 |
| Insufficient info | 0.7 |
| Match quality | Confidence | When |
|---|---|---|
| Exact parameter types match | 1.0 | Both sides have `parameterTypes` arrays and they match |
| Arity (count) matches | 1.0 | Both sides have `parameterCount`, types unavailable |
| Variadic vs fixed | 0.7 | One side is variadic, other has fixed count |
| Lenient (insufficient info) | 0.7 | One or both sides lack type and count data |
## Related docs
- [MIGRATION.md](MIGRATION.md) — breaking changes and migration guidance
- [RUNBOOK.md](RUNBOOK.md) — operational commands and recovery
- [GUARDRAILS.md](GUARDRAILS.md) — safety boundaries for humans and agents
- [TESTING.md](TESTING.md) — how to run tests
- `AGENTS.md` / `CLAUDE.md` — agent workflows and tool usage
- [MIGRATION.md](MIGRATION.md) — breaking changes and migration guidance.
- [RUNBOOK.md](RUNBOOK.md) — operational commands and recovery.
- [GUARDRAILS.md](GUARDRAILS.md) — safety boundaries for humans and agents.
- [TESTING.md](TESTING.md) — how to run tests.
- `AGENTS.md` / `CLAUDE.md` — agent workflows and tool usage expectations for **this** repo when indexed by GitNexus.
+203 -2
View File
@@ -35,7 +35,6 @@ If always-on instructions grow, load deep conventions via conditional reads (e.g
## Reference Documentation
- **This repository:** [AGENTS.md](AGENTS.md) (Cursor + monorepo notes), [ARCHITECTURE.md](ARCHITECTURE.md), [CONTRIBUTING.md](CONTRIBUTING.md), [GUARDRAILS.md](GUARDRAILS.md).
- **Call-resolution DAG:** See ARCHITECTURE.md § Call-Resolution DAG. Shared pipeline code in `gitnexus/src/core/ingestion/` must not name languages — use `LanguageProvider` hooks instead (see AGENTS.md).
- **GitNexus:** `.claude/skills/gitnexus/`; MCP and indexed-repo rules live only in [AGENTS.md](AGENTS.md) (`gitnexus:start` … `gitnexus:end`). See **GitNexus rules** below.
## Changelog
@@ -51,4 +50,206 @@ If always-on instructions grow, load deep conventions via conditional reads (e.g
## GitNexus rules
See the `<!-- gitnexus:start --> … <!-- gitnexus:end -->` block in **[AGENTS.md](AGENTS.md)** for the canonical MCP tools, impact analysis rules, and index instructions.
GitNexus MCP rules are in the `<!-- gitnexus:start -->
# GitNexus — Code Intelligence
This project is indexed by GitNexus as **GitNexus** (4325 symbols, 10556 relationships, 300 execution flows). Use the GitNexus MCP tools to understand code, assess impact, and navigate safely.
> If any GitNexus tool warns the index is stale, run `npx gitnexus analyze` in terminal first.
## Always Do
- **MUST run impact analysis before editing any symbol.** Before modifying a function, class, or method, run `gitnexus_impact({target: "symbolName", direction: "upstream"})` and report the blast radius (direct callers, affected processes, risk level) to the user.
- **MUST run `gitnexus_detect_changes()` before committing** to verify your changes only affect expected symbols and execution flows.
- **MUST warn the user** if impact analysis returns HIGH or CRITICAL risk before proceeding with edits.
- When exploring unfamiliar code, use `gitnexus_query({query: "concept"})` to find execution flows instead of grepping. It returns process-grouped results ranked by relevance.
- When you need full context on a specific symbol — callers, callees, which execution flows it participates in — use `gitnexus_context({name: "symbolName"})`.
## When Debugging
1. `gitnexus_query({query: "<error or symptom>"})` — find execution flows related to the issue
2. `gitnexus_context({name: "<suspect function>"})` — see all callers, callees, and process participation
3. `READ gitnexus://repo/GitNexus/process/{processName}` — trace the full execution flow step by step
4. For regressions: `gitnexus_detect_changes({scope: "compare", base_ref: "main"})` — see what your branch changed
## When Refactoring
- **Renaming**: MUST use `gitnexus_rename({symbol_name: "old", new_name: "new", dry_run: true})` first. Review the preview — graph edits are safe, text_search edits need manual review. Then run with `dry_run: false`.
- **Extracting/Splitting**: MUST run `gitnexus_context({name: "target"})` to see all incoming/outgoing refs, then `gitnexus_impact({target: "target", direction: "upstream"})` to find all external callers before moving code.
- After any refactor: run `gitnexus_detect_changes({scope: "all"})` to verify only expected files changed.
## Never Do
- NEVER edit a function, class, or method without first running `gitnexus_impact` on it.
- NEVER ignore HIGH or CRITICAL risk warnings from impact analysis.
- NEVER rename symbols with find-and-replace — use `gitnexus_rename` which understands the call graph.
- NEVER commit changes without running `gitnexus_detect_changes()` to check affected scope.
## Tools Quick Reference
| Tool | When to use | Command |
|------|-------------|---------|
| `query` | Find code by concept | `gitnexus_query({query: "auth validation"})` |
| `context` | 360-degree view of one symbol | `gitnexus_context({name: "validateUser"})` |
| `impact` | Blast radius before editing | `gitnexus_impact({target: "X", direction: "upstream"})` |
| `detect_changes` | Pre-commit scope check | `gitnexus_detect_changes({scope: "staged"})` |
| `rename` | Safe multi-file rename | `gitnexus_rename({symbol_name: "old", new_name: "new", dry_run: true})` |
| `cypher` | Custom graph queries | `gitnexus_cypher({query: "MATCH ..."})` |
## Impact Risk Levels
| Depth | Meaning | Action |
|-------|---------|--------|
| d=1 | WILL BREAK — direct callers/importers | MUST update these |
| d=2 | LIKELY AFFECTED — indirect deps | Should test |
| d=3 | MAY NEED TESTING — transitive | Test if critical path |
## Resources
| Resource | Use for |
|----------|---------|
| `gitnexus://repo/GitNexus/context` | Codebase overview, check index freshness |
| `gitnexus://repo/GitNexus/clusters` | All functional areas |
| `gitnexus://repo/GitNexus/processes` | All execution flows |
| `gitnexus://repo/GitNexus/process/{name}` | Step-by-step execution trace |
## Self-Check Before Finishing
Before completing any code modification task, verify:
1. `gitnexus_impact` was run for all modified symbols
2. No HIGH/CRITICAL risk warnings were ignored
3. `gitnexus_detect_changes()` confirms changes match expected scope
4. All d=1 (WILL BREAK) dependents were updated
## Keeping the Index Fresh
After committing code changes, the GitNexus index becomes stale. Re-run analyze to update it:
```bash
npx gitnexus analyze
```
If the index previously included embeddings, preserve them by adding `--embeddings`:
```bash
npx gitnexus analyze --embeddings
```
To check whether embeddings exist, inspect `.gitnexus/meta.json` — the `stats.embeddings` field shows the count (0 means no embeddings). **Running analyze without `--embeddings` will delete any previously generated embeddings.**
> Claude Code users: A PostToolUse hook handles this automatically after `git commit` and `git merge`.
## CLI
| Task | Read this skill file |
|------|---------------------|
| Understand architecture / "How does X work?" | `.claude/skills/gitnexus/gitnexus-exploring/SKILL.md` |
| Blast radius / "What breaks if I change X?" | `.claude/skills/gitnexus/gitnexus-impact-analysis/SKILL.md` |
| Trace bugs / "Why is X failing?" | `.claude/skills/gitnexus/gitnexus-debugging/SKILL.md` |
| Rename / extract / split / refactor | `.claude/skills/gitnexus/gitnexus-refactoring/SKILL.md` |
| Tools, resources, schema reference | `.claude/skills/gitnexus/gitnexus-guide/SKILL.md` |
| Index, status, clean, wiki CLI commands | `.claude/skills/gitnexus/gitnexus-cli/SKILL.md` |
<!-- gitnexus:end -->` block in **[AGENTS.md](AGENTS.md)** — load that section when working with MCP tools or the graph index.
<!-- gitnexus:start -->
# GitNexus — Code Intelligence
This project is indexed by GitNexus as **GitNexus** (3298 symbols, 7954 relationships, 185 execution flows). Use the GitNexus MCP tools to understand code, assess impact, and navigate safely.
> If any GitNexus tool warns the index is stale, run `npx gitnexus analyze` in terminal first.
## Always Do
- **MUST run impact analysis before editing any symbol.** Before modifying a function, class, or method, run `gitnexus_impact({target: "symbolName", direction: "upstream"})` and report the blast radius (direct callers, affected processes, risk level) to the user.
- **MUST run `gitnexus_detect_changes()` before committing** to verify your changes only affect expected symbols and execution flows.
- **MUST warn the user** if impact analysis returns HIGH or CRITICAL risk before proceeding with edits.
- When exploring unfamiliar code, use `gitnexus_query({query: "concept"})` to find execution flows instead of grepping. It returns process-grouped results ranked by relevance.
- When you need full context on a specific symbol — callers, callees, which execution flows it participates in — use `gitnexus_context({name: "symbolName"})`.
## When Debugging
1. `gitnexus_query({query: "<error or symptom>"})` — find execution flows related to the issue
2. `gitnexus_context({name: "<suspect function>"})` — see all callers, callees, and process participation
3. `READ gitnexus://repo/GitNexus/process/{processName}` — trace the full execution flow step by step
4. For regressions: `gitnexus_detect_changes({scope: "compare", base_ref: "main"})` — see what your branch changed
## When Refactoring
- **Renaming**: MUST use `gitnexus_rename({symbol_name: "old", new_name: "new", dry_run: true})` first. Review the preview — graph edits are safe, text_search edits need manual review. Then run with `dry_run: false`.
- **Extracting/Splitting**: MUST run `gitnexus_context({name: "target"})` to see all incoming/outgoing refs, then `gitnexus_impact({target: "target", direction: "upstream"})` to find all external callers before moving code.
- After any refactor: run `gitnexus_detect_changes({scope: "all"})` to verify only expected files changed.
## Never Do
- NEVER edit a function, class, or method without first running `gitnexus_impact` on it.
- NEVER ignore HIGH or CRITICAL risk warnings from impact analysis.
- NEVER rename symbols with find-and-replace — use `gitnexus_rename` which understands the call graph.
- NEVER commit changes without running `gitnexus_detect_changes()` to check affected scope.
## Tools Quick Reference
| Tool | When to use | Command |
|------|-------------|---------|
| `query` | Find code by concept | `gitnexus_query({query: "auth validation"})` |
| `context` | 360-degree view of one symbol | `gitnexus_context({name: "validateUser"})` |
| `impact` | Blast radius before editing | `gitnexus_impact({target: "X", direction: "upstream"})` |
| `detect_changes` | Pre-commit scope check | `gitnexus_detect_changes({scope: "staged"})` |
| `rename` | Safe multi-file rename | `gitnexus_rename({symbol_name: "old", new_name: "new", dry_run: true})` |
| `cypher` | Custom graph queries | `gitnexus_cypher({query: "MATCH ..."})` |
## Impact Risk Levels
| Depth | Meaning | Action |
|-------|---------|--------|
| d=1 | WILL BREAK — direct callers/importers | MUST update these |
| d=2 | LIKELY AFFECTED — indirect deps | Should test |
| d=3 | MAY NEED TESTING — transitive | Test if critical path |
## Resources
| Resource | Use for |
|----------|---------|
| `gitnexus://repo/GitNexus/context` | Codebase overview, check index freshness |
| `gitnexus://repo/GitNexus/clusters` | All functional areas |
| `gitnexus://repo/GitNexus/processes` | All execution flows |
| `gitnexus://repo/GitNexus/process/{name}` | Step-by-step execution trace |
## Self-Check Before Finishing
Before completing any code modification task, verify:
1. `gitnexus_impact` was run for all modified symbols
2. No HIGH/CRITICAL risk warnings were ignored
3. `gitnexus_detect_changes()` confirms changes match expected scope
4. All d=1 (WILL BREAK) dependents were updated
## Keeping the Index Fresh
After committing code changes, the GitNexus index becomes stale. Re-run analyze to update it:
```bash
npx gitnexus analyze
```
If the index previously included embeddings, preserve them by adding `--embeddings`:
```bash
npx gitnexus analyze --embeddings
```
To check whether embeddings exist, inspect `.gitnexus/meta.json` — the `stats.embeddings` field shows the count (0 means no embeddings). **Running analyze without `--embeddings` will delete any previously generated embeddings.**
> Claude Code users: A PostToolUse hook handles this automatically after `git commit` and `git merge`.
## CLI
| Task | Read this skill file |
|------|---------------------|
| Understand architecture / "How does X work?" | `.claude/skills/gitnexus/gitnexus-exploring/SKILL.md` |
| Blast radius / "What breaks if I change X?" | `.claude/skills/gitnexus/gitnexus-impact-analysis/SKILL.md` |
| Trace bugs / "Why is X failing?" | `.claude/skills/gitnexus/gitnexus-debugging/SKILL.md` |
| Rename / extract / split / refactor | `.claude/skills/gitnexus/gitnexus-refactoring/SKILL.md` |
| Tools, resources, schema reference | `.claude/skills/gitnexus/gitnexus-guide/SKILL.md` |
| Index, status, clean, wiki CLI commands | `.claude/skills/gitnexus/gitnexus-cli/SKILL.md` |
<!-- gitnexus:end -->
+11 -135
View File
@@ -21,154 +21,30 @@ This project uses the [PolyForm Noncommercial License 1.0.0](https://polyformpro
## Branch and pull requests
- Use short-lived branches off the default branch of the repo you are targeting.
- **PR titles MUST follow the conventional-commit format** — `pr-labeler.yml` enforces this on every PR and auto-applies the matching label so release notes group the change correctly.
- Prefer **conventional commits** (short prefix + description), for example:
```text
feat: add graph export option
fix: correct MCP tool schema for query
test: cover cluster merge edge case
docs: clarify analyze flags
```
- **PR title:** `[area] Short description` (e.g. `[cli] Fix index refresh race`).
- **PR description:** what changed, why, how to verify (commands), and any risk or rollback notes.
### Pull request titles
Format: `<type>[(scope)][!]: <subject>`
Allowed types and the release-notes section each one lands in (defined in `.github/release.yml`):
| Type | Label applied | Release-notes section |
|------|---------------|-----------------------|
| `feat` | `enhancement` | 🚀 Features |
| `fix` | `bug` | 🐛 Bug Fixes |
| `perf` | `performance` | 🏎️ Performance |
| `refactor` | `refactor` | 🔄 Refactoring |
| `test` | `test` | 🧪 Tests |
| `ci` | `ci` | 👷 CI/CD |
| `build` / `deps` | `dependencies` | 📦 Dependencies |
| `docs` | `documentation` | (grouped under Other Changes unless a Docs section is added) |
| `chore` / `revert` | `chore` | (excluded from release notes) |
Append `!` to the type (e.g. `feat(api)!: drop /v1 endpoint`) or include `BREAKING CHANGE:` in the PR body to flag a breaking change — the labeler then adds the `breaking` label and the 💥 Breaking Changes section is rendered first.
Examples:
```text
feat(web): add smart chat scroll
fix(extractors): resolve silent contract mis-resolution
perf: avoid O(n²) traversal in heritage walker
chore(deps): bump vitest to 3.0.0
ci: standardize workflow concurrency
```
Commits within a PR may use any style — only the **merged PR title** shows up in release notes, so that's the one the convention applies to.
## Before you open a PR
- [ ] Tests pass for the packages you touched (`gitnexus` and/or `gitnexus-web`).
- [ ] Typecheck passes: `npx tsc --noEmit` in `gitnexus/` and `npx tsc -b --noEmit` in `gitnexus-web/`.
- [ ] No secrets, tokens, or machine-specific paths committed.
- [ ] Documentation updated if behavior or public CLI/MCP contract changes.
- [ ] Pre-commit hook runs clean (`.husky/pre-commit` — formatting via lint-staged + typecheck for staged packages; tests run in CI only).
- [ ] Pre-commit hook runs clean (`.husky/pre-commit` — typecheck + unit tests for staged packages).
## Code review
Maintainers may request changes for correctness, tests, performance, or consistency with existing patterns. Keeping diffs focused makes review faster.
## GitHub Actions — Concurrency Convention
Every workflow under `.github/workflows/` MUST declare a top-level `concurrency:` block using this convention:
- **Group key** starts with `${{ github.workflow }}` so no two workflows can collide on the same group name. The discriminator that follows is chosen per event shape:
- Branch/tag scope: `${{ github.workflow }}-${{ github.ref }}`
- Per-PR scope (for `issue_comment`, `pull_request_review*`, `pull_request` meta events): `${{ github.workflow }}-${{ github.event.pull_request.number || github.event.issue.number }}`
- `workflow_run` scope (e.g. `ci-report.yml`): `${{ github.workflow }}-${{ github.event.workflow_run.pull_requests[0].number || format('{0}/{1}', github.event.workflow_run.head_repository.full_name, github.event.workflow_run.head_branch) }}` — the fork fallback must be stable across reruns (never `workflow_run.id`, which is per-run-unique and defeats serialization).
- Global single-slot (manual dispatch utilities): `${{ github.workflow }}`
- **Reusable workflows invoked via `workflow_call`:** do NOT use `${{ github.workflow }}` in the group key — in called-workflow context its evaluation is ambiguous and can resolve to the caller's name, which would deadlock against the caller's own group. Use a hardcoded literal prefix and a `github.event_name`-aware expression that falls through to `github.run_id` for reusable invocations (see `ci.yml` for the canonical form). Approved literal prefixes: `CI-` (`ci.yml`) and `docker-build-push-` (`docker.yml`). The `check-workflow-concurrency.py` validation script must be updated whenever a new approved literal prefix is added.
- **Merge queue (`merge_group`)**: when this event is added, use `${{ github.workflow }}-${{ github.event.merge_group.head_ref }}` with `cancel-in-progress: false` (every queue entry is a distinct ref; never cancel).
- **`cancel-in-progress` policy:**
| Event | `cancel-in-progress` | Why |
|-------|----------------------|-----|
| `pull_request` CI run | `true` | New push supersedes old run |
| `push` to `main` | `false` | Every main commit gets validated |
| Tag push (`v*` publish) | `false` | Never cancel mid-publish |
| `push` to `main` for release-candidate | `false` | Never cancel mid-RC publish |
| `workflow_dispatch` (release/publish) | `false` | Manual runs are intentional |
| `workflow_run` (sticky-comment reports) | `false` | Serialize, don't race |
| Per-PR bot workflows (`@claude`, review) | `false` | Serialize comments per PR |
| PR-meta re-checks (pr-description-check) | `true` | Cheap, latest wins |
| Single-slot utilities (triage sweep) | `true` | Latest dispatch supersedes |
- For workflows that serve multiple events at once (e.g. `ci.yml` handles `pull_request`, `push`, and `workflow_call`), make `cancel-in-progress` event-aware:
```yaml
concurrency:
group: ${{ github.workflow }}-${{ github.ref }}
cancel-in-progress: ${{ github.event_name == 'pull_request' }}
```
- When adding a new workflow, copy the concurrency block from an existing workflow of the same event shape.
## AI-assisted contributions
If you use coding agents, follow project context files (e.g. `AGENTS.md`, `CLAUDE.md`) and avoid drive-by refactors unrelated to the issue. Prefer incremental, test-backed changes.
## Releases
Two publish workflows ship `gitnexus` to npm:
- **Stable** (`.github/workflows/publish.yml`) — triggered by pushing any `v*`
tag. Publishes to the `latest` dist-tag with a changelog-backed GitHub
release. Maintainers are expected to tag from `main` as a convention; the
workflow itself does not enforce branch reachability.
- **Release Candidate** (`.github/workflows/release-candidate.yml`) — runs on
every push to `main` (typically a merged PR) plus manual dispatch. Docs-only
changes are skipped via `paths-ignore`. Publishes to the `rc` dist-tag with
version `X.Y.Z-rc.N` and a GitHub prerelease, where:
- `X.Y.Z` is selected automatically. On push (and on dispatch with
`bump: auto`, the default) the workflow **continues the active rc cycle**:
if the registry already has `X.Y.Z-rc.*` versions with `X.Y.Z` > current
`latest`, it reuses the highest such base; otherwise it patch-bumps
from `latest`. Dispatching with `bump: patch|minor|major` **resets**
the cycle from `latest`.
- `N` is auto-incremented against existing `X.Y.Z-rc.*` entries on the
registry. First rc for a given base is `rc.1`.
- After the npm publish succeeds, the workflow calls `docker.yml` as a
reusable workflow to build and push the corresponding RC Docker images
(e.g. `ghcr.io/abhigyanpatwari/gitnexus:1.7.0-rc.1`, mirrored to
`docker.io/akonlabs/gitnexus:1.7.0-rc.1`). The images are signed
with Cosign; the OIDC identity is `docker.yml@refs/heads/main` (the
caller's ref — see README.md § Docker for the verify command).
Idempotency: the workflow pushes an `rc/<HEAD_SHA>` marker tag and a
`v<RC>` release tag **atomically, before** calling `npm publish`. The guard
refuses to re-run once the marker exists, so a post-publish failure will
not mint a duplicate rc for the same commit. The `v<RC>` tag points at a
detached release commit whose `package.json` matches the npm tarball
exactly (traceable releases). Recovery after a partial failure:
```bash
git push --delete origin rc/<HEAD_SHA> v<RC>
# then redispatch the workflow with force: true
```
**Docker-only partial failure:** if `publish` succeeds (npm tarball + tags
are live) but the `docker` job subsequently fails (e.g. GHCR flakiness),
the npm RC is already published and the `rc/<HEAD_SHA>` marker is in place.
Re-running `release-candidate.yml` with `force: true` will abort at the
"Version already exists on npm" guard. To recover without cutting a new RC:
```bash
# 1. Manually trigger only the docker workflow, passing the existing RC tag:
gh workflow run docker.yml --ref main -f tag=v<RC_VERSION>
# (requires a workflow_dispatch trigger on docker.yml — see note below)
```
Because `docker.yml` intentionally has no `workflow_dispatch` (images are
tag-driven by design), the practical recovery options are:
- Wait for the next commit on `main`, which will cut a new RC that includes
the Docker build.
- Manually run `docker build` + `docker push` locally and sign with Cosign
against the same digest.
- Delete `rc/<HEAD_SHA>` and `v<RC>` tags, then redispatch with `force:
true` to re-run the full RC pipeline (cuts a new RC number).
The rc workflow never moves `latest`. To verify after a change, inspect dist-tags:
```bash
npm view gitnexus dist-tags
```
-209
View File
@@ -1,209 +0,0 @@
# Definition of Done — GitNexus
Last reviewed: 2026-04-23 · Version: 2.0.0
This document defines the repo-wide completion bar for production-ready changes in GitNexus. It is the stable baseline. Implementation prompts, agent behavior, and review workflows may add task-specific checks, but they must never weaken this bar.
Use it together with:
- `AGENTS.md` — agent-facing rules of engagement
- `GUARDRAILS.md` — hard safety constraints
- `CONTRIBUTING.md` — contributor workflow
- `TESTING.md` — test strategy and coverage expectations
- `ARCHITECTURE.md` — pipeline boundaries, Call-Resolution DAG, LanguageProvider contract
## 1. Scope and Intent
A change is **Done** when it is correct, safely integrated, appropriately tested, operationally sound, and a net improvement to the codebase — not merely "the code compiles and a test passes."
This DoD applies to:
- CLI, MCP, and HTTP-bridge behavior in `gitnexus/`
- Browser UI in `gitnexus-web/`
- Shared contracts in `gitnexus-shared/`
- CI workflows, release pipelines, and repo-level docs
Out of scope: full agent personas, step-by-step implementation prompts, verbose review formatting rules, repo walkthroughs already covered elsewhere, temporary task-specific acceptance criteria. Those belong in prompts, PR templates, or other repo docs.
## 2. Core Definition of Done
Every change must satisfy **every relevant item** below. If an item does not apply, say so explicitly in the PR description.
### 2.1 Correctness and Completeness
- [ ] The requested behavior is implemented end-to-end in the **real runtime path** for the affected surface — no dead code, partial wiring, test-only shims, or "works in isolation but not in production" seams.
- [ ] Edge cases relevant to the changed surface are handled or explicitly documented as out of scope.
- [ ] Error handling is proportionate: inputs at system boundaries (user input, external APIs, filesystem, process spawn) are validated; internal, framework-guaranteed paths are trusted.
- [ ] The change produces the same result on re-run (idempotent where expected) and does not rely on accidental ordering.
### 2.2 Architecture and Placement
- [ ] The change is placed in the correct package and layer:
- `gitnexus/` for CLI, MCP, HTTP bridge, ingestion, graph, and runtime logic
- `gitnexus-web/` for browser UI (thin client — no WASM workers, all queries via HTTP API)
- `gitnexus-shared/` for shared contracts, types, and constants
- [ ] Pipeline and architecture boundaries remain explicit. Shared ingestion code in `gitnexus/src/core/ingestion/` must not name languages — use `LanguageProvider` hooks (see `AGENTS.md` and `ARCHITECTURE.md` § Call-Resolution DAG).
- [ ] No hidden cross-phase coupling; no leaking of language-specific logic into shared infrastructure without a documented architectural reason.
- [ ] Runtime and graph behavior are consistent — the real source of truth is fixed at the source, not symptom-patched in a downstream layer.
- [ ] Direct imports from `gitnexus-shared` are used. No barrel re-exports introduced to paper over drift between packages.
### 2.3 Design and Readability
- [ ] The implementation is the **smallest correct solution** for the requirement. No speculative abstraction, unnecessary indirection, clever but hard-to-follow control flow, or unrelated cleanup.
- [ ] Naming, control flow, ownership, and extension points are clear enough that the next contributor can extend the code without archaeology.
- [ ] Comments are minimal and useful — they explain intent, invariants, contracts, or non-obvious constraints. No stale comments, placeholder comments, narrated code, commented-out code, or "what" comments where a good name would do.
- [ ] No copy-paste duplication created for convenience; no premature deduplication of three similar lines.
### 2.4 Contracts and Compatibility
- [ ] Existing contracts (types in `gitnexus-shared/`, CLI flags, MCP tools/resources, HTTP routes, graph node/edge shapes, persisted IDs) are preserved unless the task explicitly requires a contract change.
- [ ] Any contract change is intentional, explicit, and reflected in **every direct consumer** in the same change, with types aligned end-to-end.
- [ ] Persisted data changes (graph schema, IDs, embeddings) are backward-compatible or accompanied by a documented migration / reindex path.
- [ ] If user-visible behavior, public usage, CLI help, or README examples change, the relevant docs, examples, help text, or migration notes are updated in the same change.
### 2.5 Security
- [ ] No new injection surfaces (command, path, SQL/Cypher-style, prompt) introduced on paths that consume untrusted input.
- [ ] No secrets, tokens, or credentials committed to the repo, to logs, or to error messages.
- [ ] Filesystem access honors the repo-scope and indexed-repo boundaries documented in `AGENTS.md` and `GUARDRAILS.md`.
- [ ] Third-party dependencies added or bumped are justified, from reputable sources, and do not regress the supply-chain posture.
### 2.6 Performance and Resource Use
- [ ] No repeated avoidable work, unnecessary scans, unnecessary round-trips, unbounded caches, or obvious hot-path regressions.
- [ ] Tree-sitter buffer sizing follows the adaptive 512KB–32MB convention (`getTreeSitterBufferSize`) — do not hard-code new buffer sizes.
- [ ] Memory and handle lifecycles are explicit: database handles (LadybugDB) close cleanly, no dangling process watchers, no leaked tree-sitter parsers.
- [ ] Long-running or large-graph paths remain bounded or are measurably streamed; degradation on large real repos is considered, not assumed benign.
### 2.7 Tests
- [ ] Tests cover the **real changed path** — they would fail if behavior, wiring, or contracts were broken, not only if a mock were misconfigured.
- [ ] Integration tests hit a real database where the production path does; do not introduce mocks that hide migration or schema drift.
- [ ] Assertions are meaningful. Use `toBe` / `toEqual` for exact expectations; avoid `toBeGreaterThanOrEqual` and other bounds-only assertions that mask regressions.
- [ ] Fixtures are realistic enough for the risk of the change — a one-file fixture is not sufficient for a pipeline-wide behavior change.
- [ ] New tests are deterministic and do not depend on network, clock, or host-specific paths without explicit isolation.
### 2.8 Observability and Operability
- [ ] Errors surfaced to users or callers are actionable: they name what failed, what input was involved (without leaking secrets), and how to recover where possible.
- [ ] Logging is proportionate — no noisy debug logs left in hot paths, no silent catches that swallow diagnostics.
- [ ] CLI exit codes and MCP tool responses are correct for each outcome (success, user error, internal error).
- [ ] Progress reporting (`PipelineProgress` and similar shared contracts) remains accurate after the change.
### 2.9 Reversibility and Risk
- [ ] The change has a clear rollback story: revert is safe, or migration is accompanied by a documented rollback / reindex procedure.
- [ ] Residual risks, compatibility impacts, and operational concerns are either resolved or **clearly stated** in the PR description.
- [ ] Destructive or hard-to-reverse operations (graph rebuild, schema change, `git` state manipulation) are opt-in or guarded.
## 3. Agent-Assisted Workflow Guardrails
When the change is produced with or reviewed by an AI agent, the following additional gates apply:
- [ ] **Scope match.** The final diff matches the intended symbols, files, and processes — no speculative refactors, unrelated formatting churn, or collateral edits outside the task scope.
- [ ] **Evidence-based edits.** Claims about repo state are verified against the current code, not trusted from memory or stale documentation.
- [ ] **Impact analysis.** Where GitNexus graph tooling is available and relevant, impact of non-trivial symbol, contract, or runtime-path changes is checked **before** editing.
- [ ] **Embeddings preserved.** If an indexed repo already has embeddings and re-analysis is required, embeddings are preserved — not accidentally dropped by a destructive reindex.
- [ ] **No false-done.** "Done" is claimed only after the Validation Baseline below has been run or any gap is explicitly named. Green tests on an unrelated path do not constitute validation.
- [ ] **Five-axis self-review** before handing off: correctness, readability, architecture, security, performance.
## 4. Validation Baseline
Run the commands relevant to the touched area. If something cannot be run in the current environment, state it explicitly in the handoff.
### 4.1 Build ordering
- [ ] `gitnexus-shared/` dist is built before consuming packages are typechecked or tested (CI uses the `setup-gitnexus` action for this — local runs must match).
### 4.2 If `gitnexus/` changed
- [ ] `cd gitnexus && npx tsc --noEmit`
- [ ] `cd gitnexus && npm test`
- [ ] `cd gitnexus && npx prettier --check .` for files in the diff (pre-commit runs the affected-tests subset; do not expand scope)
### 4.3 If `gitnexus-web/` changed
- [ ] `cd gitnexus-web && npx tsc -b --noEmit`
- [ ] `cd gitnexus-web && npm test`
- [ ] `cd gitnexus-web && npm run test:e2e` when browser flows or user-facing UI behavior changed
### 4.4 If `gitnexus-shared/` changed
- [ ] Shared package builds cleanly (`npm run build` in `gitnexus-shared/`)
- [ ] Dependent packages still typecheck and test after the shared change — verify both CLI and web consumers together
### 4.5 If CI workflows or release pipelines changed
- [ ] The workflow passes a dry-run or triggered run before merge; concurrency (`cancel-in-progress`) and the `setup-gitnexus` action remain wired correctly.
- [ ] `CHANGELOG.md` is **not** edited here — it is owned by the release process.
## 5. Review Gates
A reviewer (human or agent) should be able to answer **yes** to each of the following before approving:
1. **Correctness** — Does the change do what it claims on the real runtime path?
2. **Readability** — Will the next contributor understand this in six months without asking?
3. **Architecture** — Is it in the right package, layer, and phase? Are boundaries respected?
4. **Security** — No new injection, leak, or trust-boundary violation?
5. **Performance** — No obvious regression on realistic inputs?
6. **Tests** — Would a regression in the changed behavior fail loudly?
7. **Scope** — Does the diff match the intended change, with no unrelated churn?
## 6. "Not Done" Signals
A change is **not** Done if any of the following is true, even if CI is green:
- The runtime path is not actually exercised by the tests.
- A contract drifted between `gitnexus/`, `gitnexus-web/`, and `gitnexus-shared/` and only one side was updated.
- A language-specific concern leaked into shared ingestion code.
- The diff contains unrelated reformatting, refactors, or cleanup beyond the stated task.
- Logs, comments, or TODOs were added as placeholders for work not done.
- The change depends on a manual step that is not documented.
- `CHANGELOG.md` was edited during PR work.
- Pre-commit, prettier, or typecheck was bypassed without explicit justification.
## 7. Task-Specific DoD Template
Use this in implementation and review prompts. Keep it short and tailor it to the actual change:
```md
# Definition of Done for this implementation
- [ ] Runtime wiring is complete for the affected path.
- [ ] Requested behavior is correct and relevant contracts are preserved or explicitly updated.
- [ ] The design stays scoped, readable, and proportionate to the task.
- [ ] Tests prove the changed behavior and catch broken wiring.
- [ ] Required validation for touched packages has been run, or any gap is explicitly noted.
- [ ] Repo boundaries, security, performance, and operational safety are respected.
- [ ] The diff contains only the intended change — no unrelated churn.
```
## 8. How to Use This File in Claude Review
Reference this file as the repo-wide completion bar. Add a task-specific review instruction such as:
```md
Review this change against `DoD.md` and the repo docs (`AGENTS.md`, `GUARDRAILS.md`,
`CONTRIBUTING.md`, `TESTING.md`, `ARCHITECTURE.md`). Treat `DoD.md` as the minimum
bar for production readiness. Flag anything that is partially wired, contract-unsafe,
under-tested, architecturally misplaced, scope-creeping, or harder to maintain than
necessary. Apply the five-axis review gate: correctness, readability, architecture,
security, performance.
```
## 9. Evolution
This DoD is living. Revisit it when:
- A class of incident slips past it (add a gate).
- A gate becomes consistently ceremonial without catching issues (remove or merge it).
- The architecture evolves in a way that changes what "done" means (update placement, validation, or contracts sections).
Track material updates in the changelog below. Keep the file tight — if it grows past a single read-in-one-sitting, something has drifted into the wrong place.
## Changelog
| Date | Version | Change |
| ---------- | ------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------- |
| 2026-04-23 | 2.0.0 | Restructured into numbered sections; added Security, Observability, Reversibility, Agent-Assisted Guardrails, Review Gates, Not-Done Signals; expanded validation baseline (shared-first build, prettier, CI workflow checks). |
| 2026-04-13 | 1.0.0 | Initial repo-wide Definition of Done. |
-58
View File
@@ -1,58 +0,0 @@
ARG BUILDPLATFORM
ARG TARGETPLATFORM
# ── Builder ────────────────────────────────────────────────────────────
# Native modules (tree-sitter-*, onnxruntime-node, node-gyp builds for
# tree-sitter-proto / tree-sitter-swift) require python3 + a C/C++ toolchain.
FROM node:22-trixie-slim AS builder
WORKDIR /app
# Toolchain for node-gyp / native builds.
RUN apt-get update && apt-get install -y --no-install-recommends python3 make g++ git && rm -rf /var/lib/apt/lists/*
# Build gitnexus-shared first — gitnexus depends on it as a workspace.
COPY gitnexus-shared/package.json gitnexus-shared/package-lock.json ./gitnexus-shared/
RUN npm ci --prefix gitnexus-shared
COPY gitnexus-shared ./gitnexus-shared
RUN rm -f gitnexus-shared/tsconfig.tsbuildinfo
RUN npm run build --prefix gitnexus-shared
# Copy the full gitnexus package before installing — `npm ci` triggers
# `postinstall` (patches tree-sitter-swift, builds the vendored
# tree-sitter-proto) and `prepare` (compiles TypeScript via scripts/build.js),
# both of which need the source tree.
COPY gitnexus ./gitnexus
RUN npm ci --prefix gitnexus
# Drop dev dependencies for a smaller runtime layer.
RUN npm prune --omit=dev --prefix gitnexus
# ── Runtime ────────────────────────────────────────────────────────────
FROM node:22-trixie-slim AS runtime
# curl for the healthcheck; git so `gitnexus` can clone repos at runtime.
RUN apt-get update && apt-get install -y --no-install-recommends curl git && rm -rf /var/lib/apt/lists/*
WORKDIR /app
# Pre-create the data directory and hand it to the unprivileged `node` user
# so the bind-mounted volume is writable without root.
RUN mkdir -p /data/gitnexus && chown -R node:node /data
COPY --from=builder --chown=node:node /app/gitnexus/dist ./gitnexus/dist
COPY --from=builder --chown=node:node /app/gitnexus/node_modules ./gitnexus/node_modules
COPY --from=builder --chown=node:node /app/gitnexus/package.json ./gitnexus/package.json
COPY --from=builder --chown=node:node /app/gitnexus/vendor ./gitnexus/vendor
USER node
# The web UI defaults to http://localhost:4747 — keep that contract.
ENV GITNEXUS_HOME=/data/gitnexus \
NODE_ENV=production \
PORT=4747
EXPOSE 4747
# Bind to 0.0.0.0 so the server is reachable from the host's mapped port.
CMD ["node", "gitnexus/dist/cli/index.js", "serve", "--host", "0.0.0.0", "--port", "4747"]
-37
View File
@@ -1,37 +0,0 @@
ARG BUILDPLATFORM
ARG TARGETPLATFORM
FROM --platform=$BUILDPLATFORM node:22-alpine AS builder
WORKDIR /app
COPY gitnexus-shared/package.json gitnexus-shared/package-lock.json ./gitnexus-shared/
RUN npm ci --prefix gitnexus-shared
COPY gitnexus-shared ./gitnexus-shared
RUN npm run build --prefix gitnexus-shared
COPY gitnexus/package.json ./gitnexus/
COPY gitnexus-web/package.json gitnexus-web/package-lock.json ./gitnexus-web/
RUN npm ci --prefix gitnexus-web
COPY gitnexus-web ./gitnexus-web
RUN npm run build --prefix gitnexus-web
FROM node:22-alpine AS runtime
RUN apk add --no-cache curl
WORKDIR /app
COPY --from=builder /app/gitnexus-web/dist ./dist
COPY docker-server.mjs ./docker-server.mjs
RUN chown -R node:node /app
USER node
EXPOSE 4173
CMD ["node", "docker-server.mjs"]
+46 -43
View File
@@ -1,69 +1,72 @@
# Guardrails — GitNexus
# Guardrails — GitNexus (repo + agents)
Rules for **human contributors** and **AI agents**. Complements `AGENTS.md` (workflows) and `CONTRIBUTING.md` (PR process).
Rules for **human contributors** and **AI agents** working on this codebase or publishing artifacts. These complement `AGENTS.md` / `CLAUDE.md` (which focus on GitNexus-in-GitNexus workflows).
## Scope (least privilege)
## Scope (typical agent session)
- **Read:** Source, tests, docs, public config as needed.
- **Write:** Only files required for the fix or feature; no unrelated formatting or refactors.
- **Execute:** Tests, typecheck, documented CLI commands. No destructive commands on user data without approval.
- **Off-limits:** Other people's machines, production deployments you don't own, credentials you lack permission to use.
When automating changes in this repository, treat scope as **least privilege**:
Maintainer may widen scope per task.
- **Read:** Source, tests, docs, public config as needed for the task.
- **Write:** Only files required for the requested fix or feature; avoid unrelated formatting or refactors.
- **Execute:** Tests, typecheck, and documented CLI commands; do not run destructive commands on user data outside the repo without explicit approval.
- **Off-limits:** Other people’s machines, production deployments you don’t own, and credentials you didn’t receive permission to use.
Adjust explicitly if the maintainer defines a different scope for a task.
---
## Non-negotiables
1. **Never commit secrets** — API keys, tokens, real `.env` values, private URLs, session cookies. Use `.env.example` with placeholders.
2. **Never rename with find-and-replace** in GitNexus-indexed projects — use `rename` MCP tool with `dry_run: true` first, review `graph` vs `text_search` edits. No separate `gitnexus rename` CLI exists.
3. **Run impact analysis before editing shared symbols** — `impact` (upstream) for functions/classes/methods others call. Do not ignore HIGH/CRITICAL without maintainer sign-off.
4. **Run `detect_changes` before commit** — confirm diffs map to expected symbols/processes when the graph is available.
5. **Preserve embeddings** — plain `npx gitnexus analyze` now preserves any embeddings recorded in `.gitnexus/meta.json` (the previous behavior wiped them). Use `--embeddings` to also generate vectors for new/changed nodes; use `--drop-embeddings` only when an explicit wipe is intended (e.g., model swap).
1. **Never commit secrets** — API keys, tokens, `.env` with real values, private URLs, or session cookies. Use `.env.example` with placeholders only.
2. **Never rename symbols with blind find-and-replace** when working in a GitNexus-indexed project — use the **`rename` MCP tool** with **`dry_run: true` first**, then review `graph` vs `text_search` edits. (There is no separate `gitnexus rename` CLI; renaming goes through MCP or editor integration.)
3. **Run impact analysis before editing shared symbols** — use **`impact`** (upstream) for functions/classes/methods others call; do not ignore **HIGH** / **CRITICAL** risk without maintainer sign-off.
4. **Prefer `detect_changes` before commit** — confirm diffs map to expected symbols/processes when the graph is available.
5. **Preserve embeddings** — if `.gitnexus/meta.json` shows embeddings, run `npx gitnexus analyze --embeddings` when refreshing the index; plain `analyze` can drop them.
---
## Signs (recurring failure patterns)
Format: **Trigger → Instruction → Reason**. Append new Signs when the same mistake repeats.
Use this format: **Trigger → Instruction → Reason**.
Append new Signs here when the same mistake repeats (e.g. CI broken twice the same way).
### Stale graph after edits
### Sign: Stale graph after edits
- **Trigger:** MCP warns index is behind `HEAD`, or search doesn't match latest commit.
- **Do:** `npx gitnexus analyze` (plus `--embeddings` if used).
- **Why:** Tools query LadybugDB from last analyze; git changes are invisible until re-indexed.
- **Trigger:** MCP or resources warn the index is behind `HEAD`, or code search doesn’t match latest commit.
- **Instruction:** Run `npx gitnexus analyze` from the repo root (plus `--embeddings` if the project used them).
- **Reason:** Tools query LadybugDB built at last analyze; git changes are invisible until re-indexed.
### Embeddings vanished after analyze
### Sign: Embeddings vanished after analyze
- **Trigger:** Semantic search quality drops; `stats.embeddings` in `meta.json` is 0 after refresh.
- **Do:** Re-run `npx gitnexus analyze --embeddings` to regenerate. Check the analyze log for a `Warning: could not load cached embeddings` line — if present, the cache restore failed (corrupt DB / schema mismatch) and the rebuild had nothing to preserve. If you intentionally passed `--drop-embeddings`, this is expected.
- **Why:** Plain `analyze` preserves prior vectors by re-inserting them after the rebuild; the only ways to end up at zero are an explicit `--drop-embeddings`, a cache-load failure (now logged), or a model/dimension change that invalidates the cache.
- **Trigger:** Semantic search quality drops; `stats.embeddings` in `.gitnexus/meta.json` is 0 after a refresh.
- **Instruction:** Re-run `npx gitnexus analyze --embeddings` and confirm `meta.json` reflects stored embeddings.
- **Reason:** Embedding generation is opt-in; analyze without the flag does not preserve prior vectors.
### MCP lists no repos
### Sign: MCP lists no repos
- **Trigger:** MCP stderr says no indexed repos.
- **Do:** `npx gitnexus analyze` in the target repo; verify `npx gitnexus list` shows it.
- **Why:** MCP discovers repos via `~/.gitnexus/registry.json`, populated by analyze.
- **Trigger:** MCP stderr says no indexed repos.
- **Instruction:** Run `npx gitnexus analyze` in the target repository; verify `npx gitnexus list` shows it.
- **Reason:** The MCP server discovers repos via `~/.gitnexus/registry.json`, populated by analyze.
### Wrong repo in multi-repo setups
### Sign: Wrong repo in multi-repo setups
- **Trigger:** Query/impact results belong to another project.
- **Do:** Call `list_repos`, then pass `repo` on subsequent tools.
- **Why:** Default target is ambiguous when multiple repos are registered.
- **Trigger:** Query/impact results clearly belong to another project.
- **Instruction:** Call `list_repos`, then pass **`repo`** on subsequent tools (or use per-workspace MCP config).
- **Reason:** Default target may be ambiguous when multiple repos are registered.
### LadybugDB lock / "database busy"
### Sign: LadybugDB lock / “database busy”
- **Trigger:** Errors opening `.gitnexus/lbug` while MCP and analyze both run.
- **Do:** Stop overlapping processes (one writer at a time). Retry analyze or restart MCP.
- **Why:** Embedded DB expects single-process ownership.
- **Trigger:** Errors opening `.gitnexus/lbug` while MCP and analyze both run.
- **Instruction:** Stop overlapping processes; one writer at a time. Retry analyze or restart MCP.
- **Reason:** Embedded DB expects single-process ownership of the store.
---
## Publishing & supply chain
- **npm:** Do not publish from unreviewed automation. Bump version intentionally; tag releases to match `package.json`.
- **Dependencies:** Minimal, auditable `package.json` changes; run tests and CI after lockfile updates.
- **License:** PolyForm Noncommercial 1.0.0 — do not relicense without maintainer approval.
- **npm:** Do not publish from unreviewed automation; follow maintainer release process. Bump version intentionally; tag releases to match `package.json`.
- **Dependencies:** Prefer minimal, auditable changes to `package.json`; run tests and CI after lockfile updates.
- **License:** This project ships under **PolyForm Noncommercial 1.0.0** — do not relicense or imply a different license in docs or metadata without maintainer approval.
---
@@ -71,15 +74,15 @@ Format: **Trigger → Instruction → Reason**. Append new Signs when the same m
Stop and ask a **human maintainer** when:
- Impact analysis shows HIGH/CRITICAL risk and the task still requires the change.
- You need to alter CI, release, or security-sensitive config.
- Requirements conflict (e.g. "speed up analyze" vs "must keep all embeddings on huge repo").
- Impact analysis shows **HIGH** / **CRITICAL** risk and the task still requires the change.
- You need to alter **CI**, **release**, or **security-sensitive** config.
- Requirements conflict (e.g. “speed up analyze” vs “must keep all embeddings on huge repo”).
- You are unsure whether data loss is acceptable (`clean`, forced migrations, schema changes).
---
## Related docs
- [ARCHITECTURE.md](ARCHITECTURE.md) — components and data flow
- [RUNBOOK.md](RUNBOOK.md) — commands for recovery
- [CONTRIBUTING.md](CONTRIBUTING.md) — PR and commit expectations
- [ARCHITECTURE.md](ARCHITECTURE.md) — components and data flow.
- [RUNBOOK.md](RUNBOOK.md) — commands for recovery.
- [CONTRIBUTING.md](CONTRIBUTING.md) — PR and commit expectations.
-44
View File
@@ -1,49 +1,5 @@
# Migration Guide
## `impact` tool may now return `{ status: 'ambiguous' }` (PR #888, issue #470)
Before this change the `impact` MCP tool silently picked the first match
when the `target` name hit multiple symbols (Class → Interface → Function
→ Method → Constructor priority UNION). This often produced analysis for
the wrong symbol with no signal back to the caller.
After this change, when the resolver finds more than one viable match
and the caller supplied none of `target_uid` / `file_path` / `kind`,
`impact` returns a disambiguation response shaped like:
```json
{
"status": "ambiguous",
"message": "Found N symbols matching '<target>'. Use target_uid, file_path, or kind to disambiguate.",
"target": { "name": "<target>" },
"direction": "upstream",
"impactedCount": 0,
"risk": "UNKNOWN",
"candidates": [
{ "uid": "...", "name": "...", "kind": "Function", "filePath": "...", "line": 42, "score": 0.76 }
]
}
```
### Do I need to migrate?
**Probably not, but check for assumptions.** Callers that unconditionally
read `result.byDepth` / `result.summary` / `result.affected_processes`
without first checking `result.status` will now see `undefined` in the
ambiguous case. The fix is to branch on `result.status === 'ambiguous'`
first and follow up with `target_uid` (preferred) or `file_path` / `kind`.
The `context` tool's ambiguous response is a strict superset of the
existing shape — every candidate gains a `score` field, no existing field
has changed. No migration required for `context` callers.
### What happens on re-index?
Nothing — this is an MCP-surface change only. The graph schema, indexer,
and stored data are untouched.
---
## OVERRIDES → METHOD_OVERRIDES (PR #642)
The `OVERRIDES` relationship type has been renamed to `METHOD_OVERRIDES` for
+8 -168
View File
@@ -1,5 +1,5 @@
# GitNexus
**⚠️ Important Notice:** GitNexus has NO official cryptocurrency, token, or coin. Any token/coin using the GitNexus name on Pump.fun or any other platform is **not affiliated with, endorsed by, or created by** this project or its maintainers. Do not purchase any cryptocurrency claiming association with GitNexus.
⚠️ Important Notice:** GitNexus has NO official cryptocurrency, token, or coin. Any token/coin using the GitNexus name on Pump.fun or any other platform is **not affiliated with, endorsed by, or created by** this project or its maintainers. Do not purchase any cryptocurrency claiming association with GitNexus.
<div align="center">
@@ -9,7 +9,7 @@
<h2>Join the official Discord to discuss ideas, issues etc!</h2>
<a href="https://discord.gg/MgJrmsqr62">
<a href="https://discord.gg/AAsRVT6fGb">
<img src="https://img.shields.io/discord/1477255801545429032?color=5865F2&logo=discord&logoColor=white" alt="Discord"/>
</a>
<a href="https://www.npmjs.com/package/gitnexus">
@@ -120,7 +120,7 @@ To configure MCP for your editor, run `npx gitnexus setup` once — or set it up
| **Windsurf** | Yes | — | — | MCP |
| **OpenCode** | Yes | Yes | — | MCP + Skills |
> **Claude Code** gets the deepest integration: MCP tools + agent skills + PreToolUse hooks that enrich searches with graph context + PostToolUse hooks that detect a stale index after commits and prompt the agent to reindex.
> **Claude Code** gets the deepest integration: MCP tools + agent skills + PreToolUse hooks that enrich searches with graph context + PostToolUse hooks that auto-reindex after commits.
## Community Integrations
@@ -194,7 +194,6 @@ gitnexus analyze --force # Force full re-index
gitnexus analyze --skills # Generate repo-specific skill files from detected communities
gitnexus analyze --skip-embeddings # Skip embedding generation (faster)
gitnexus analyze --skip-agents-md # Preserve custom AGENTS.md/CLAUDE.md gitnexus section edits
gitnexus analyze --skip-git # Index folders that are not Git repositories
gitnexus analyze --embeddings # Enable embedding generation (slower, better search)
gitnexus analyze --verbose # Log skipped files when parsers are unavailable
gitnexus mcp # Start MCP server (stdio) — serves all indexed repos
@@ -208,11 +207,11 @@ gitnexus wiki --model <model> # Wiki with custom LLM model (default: gpt-4o-m
gitnexus wiki --base-url <url> # Wiki with custom LLM API base URL
# Repository groups (multi-repo / monorepo service tracking)
gitnexus group create <name> # Create a repository group
gitnexus group add <group> <groupPath> <registryName> # Add a repo to a group. <groupPath> is a hierarchy path (e.g. hr/hiring/backend); <registryName> is the repo's name from the registry (see `gitnexus list`)
gitnexus group remove <group> <groupPath> # Remove a repo from a group by its hierarchy path
gitnexus group list [name] # List groups, or show one group's config
gitnexus group sync <name> # Extract contracts and match across repos/services
gitnexus group create <name> # Create a repository group
gitnexus group add <name> <repo> # Add a repo to a group
gitnexus group remove <name> <repo> # Remove a repo from a group
gitnexus group list [name] # List groups, or show one group's config
gitnexus group sync <name> # Extract contracts and match across repos/services
gitnexus group contracts <name> # Inspect extracted contracts and cross-links
gitnexus group query <name> <q> # Search execution flows across all repos in a group
gitnexus group status <name> # Check staleness of repos in a group
@@ -336,165 +335,6 @@ cd ../gitnexus-web && npm install
npm run dev
```
## Docker
The official Docker setup ships **two signed images** orchestrated by `docker-compose.yaml`. Each image is published to both **GitHub Container Registry** (GHCR) and **Docker Hub** — same build, same digest, same Cosign signature — so pick whichever registry you prefer:
| Purpose | GHCR (default in `docker-compose.yaml`) | Docker Hub mirror |
| ---------------------------------------------------------------------- | --------------------------------------------- | ------------------------------------------- |
| CLI / `gitnexus serve` backend (HTTP API on port `4747`, MCP, indexer) | `ghcr.io/abhigyanpatwari/gitnexus:latest` | `akonlabs/gitnexus:latest` |
| Static web UI (port `4173`) | `ghcr.io/abhigyanpatwari/gitnexus-web:latest` | `akonlabs/gitnexus-web:latest` |
> **Heads-up — image rename.** Earlier releases published the web UI under
> `ghcr.io/abhigyanpatwari/gitnexus`. Starting with the introduction of the
> bundled backend, that slug now hosts the CLI/server image and the UI moved
> to `ghcr.io/abhigyanpatwari/gitnexus-web`. The previous tags remain
> available for pulling, but new versions are only published under the new
> slugs. Update your `docker run` / compose files accordingly (or just adopt
> the bundled compose).
### One-command setup
```bash
docker compose up -d
```
This starts the server on `http://localhost:4747` and the web UI on
`http://localhost:4173`. The UI auto-detects the server because the browser
runs on the host and reaches the container via the mapped port.
A named volume (`gitnexus-data`) persists the global registry, indexes, and
cloned repos at `/data/gitnexus` inside the server container. To make repos on
your host machine indexable, set `WORKSPACE_DIR` before bringing the stack up:
```bash
WORKSPACE_DIR=$HOME/code docker compose up -d
# Inside the server container the directory is mounted read-only at /workspace.
docker compose exec gitnexus-server gitnexus index /workspace/my-repo
```
### Direct `docker run`
```bash
# Server
docker run --rm -d \
--name gitnexus-server \
-p 4747:4747 \
-v gitnexus-data:/data/gitnexus \
ghcr.io/abhigyanpatwari/gitnexus:latest
# Web UI
docker run --rm -d \
--name gitnexus-web \
-p 4173:4173 \
ghcr.io/abhigyanpatwari/gitnexus-web:latest
```
Optional env file (override image tags, container names, ports, workspace dir):
```bash
cp .env.example .env
docker compose --env-file .env up -d
```
### Versioning & supply-chain protection
The Docker images are version-locked to the npm package:
- Stable images are **only published from `vX.Y.Z` git tags** (via `docker.yml`
triggered directly by the tag push), and the workflow refuses to build unless
the tag exactly matches `gitnexus/package.json`'s version. So
`ghcr.io/abhigyanpatwari/gitnexus:1.6.2` (and its Docker Hub mirror
`akonlabs/gitnexus:1.6.2`) is byte-for-byte the same release as
`npm install gitnexus@1.6.2` — no drift, no floating builds from `main`.
Both registries receive the same digest from a single build step, so you can
pull from either and the signature verifies identically.
- Release-candidate images (e.g. `:1.7.0-rc.1`) are published alongside each
RC npm release. They are built by `release-candidate.yml` calling `docker.yml`
as a reusable workflow after the RC tag is created and pushed.
- `:latest` is auto-promoted only from non-prerelease tags by the Docker
metadata action, so it always points at a real, npm-published version.
Both images are signed with [Cosign keyless signing][cosign-keyless] using the
workflow's GitHub OIDC identity, and shipped with build provenance and SBOM
attestations. **This is your protection against supply-chain attacks**: even if
an attacker republishes a same-named image elsewhere (or somehow pushes to a
typo-squatted registry), they cannot forge a Cosign signature tied to
`abhigyanpatwari/GitNexus`'s `docker.yml`. Always verify before pulling into
sensitive environments:
**Stable releases** — signed from the `v*` tag ref:
```bash
cosign verify ghcr.io/abhigyanpatwari/gitnexus:1.6.2 \
--certificate-identity-regexp '^https://github\.com/abhigyanpatwari/GitNexus/\.github/workflows/docker\.yml@refs/tags/v[0-9]+\.[0-9]+\.[0-9]+(-[a-zA-Z0-9.]+)?$' \
--certificate-oidc-issuer https://token.actions.githubusercontent.com
# Same signature verifies the Docker Hub mirror (identical digest):
cosign verify docker.io/akonlabs/gitnexus:1.6.2 \
--certificate-identity-regexp '^https://github\.com/abhigyanpatwari/GitNexus/\.github/workflows/docker\.yml@refs/tags/v[0-9]+\.[0-9]+\.[0-9]+(-[a-zA-Z0-9.]+)?$' \
--certificate-oidc-issuer https://token.actions.githubusercontent.com
```
The regex pins the certificate identity to this repo's `docker.yml` workflow
**run from a `v*` tag** — rejecting unsigned images, images signed by other
workflows, and images signed from unprotected refs. It is identical for both
registries because both sets of tags were signed at the same digest in one
workflow run.
**Release candidates** — signed from `refs/heads/main` (the caller's ref when
`release-candidate.yml` invokes `docker.yml` as a reusable workflow):
```bash
cosign verify ghcr.io/abhigyanpatwari/gitnexus:1.7.0-rc.1 \
--certificate-identity 'https://github.com/abhigyanpatwari/GitNexus/.github/workflows/docker.yml@refs/heads/main' \
--certificate-oidc-issuer https://token.actions.githubusercontent.com
```
You can also inspect the build provenance and SBOM:
```bash
cosign download attestation ghcr.io/abhigyanpatwari/gitnexus:1.6.2 \
--predicate-type https://slsa.dev/provenance/v1
```
#### Kubernetes: enforce signatures at admission
For Kubernetes deployments, ship the bundled
[`ClusterImagePolicy`](deploy/kubernetes/cluster-image-policy.yaml) so the
[Sigstore policy-controller][policy-controller] rejects any GitNexus pod whose
image is not signed by this repo's `docker.yml` running from a `vX.Y.Z` tag —
the same identity the `cosign verify` snippet above pins.
```bash
# 1. Install the controller (one-time, cluster-wide)
helm repo add sigstore https://sigstore.github.io/helm-charts && helm repo update
helm install policy-controller -n cosign-system --create-namespace \
sigstore/policy-controller
# 2. Opt your namespace in
kubectl label namespace <your-ns> policy.sigstore.dev/include=true
# 3. Apply the policy
kubectl apply -f deploy/kubernetes/cluster-image-policy.yaml
```
After this, attempting to deploy an unsigned image — or one signed by anything
other than `abhigyanpatwari/GitNexus`'s `docker.yml` at a `v*` tag — fails the
admission webhook before a pod is ever created. This turns the verifiable
signature into an enforced policy, which is the supply-chain control most
clusters actually need.
[cosign-keyless]: https://docs.sigstore.dev/cosign/signing/overview/
[policy-controller]: https://docs.sigstore.dev/policy-controller/overview/
### Files
- [Dockerfile.web](Dockerfile.web) — builds `gitnexus-shared` and `gitnexus-web`, then serves the production frontend.
- [Dockerfile.cli](Dockerfile.cli) — builds the CLI/server (with its native deps) and runs `gitnexus serve --host 0.0.0.0`.
- [docker-compose.yaml](docker-compose.yaml) — starts both signed images side by side.
- [.env.example](.env.example) — overrides for image names, container names, ports, and the workspace mount.
The web UI uses the same indexing pipeline as the CLI but runs entirely in WebAssembly (Tree-sitter WASM, LadybugDB WASM, in-browser embeddings). It's great for quick exploration but limited by browser memory for larger repos.
**Local Backend Mode:** Run `gitnexus serve` and open the web UI locally — it auto-detects the server and shows all your indexed repos, with full AI chat support. No need to re-upload or re-index. The agent's tools (Cypher queries, search, code navigation) route through the backend HTTP API automatically.
+5 -8
View File
@@ -20,9 +20,9 @@ From repository root, unless noted:
cd gitnexus
npm install
npm run build
npm test # full suite: vitest run
npm run test:unit # unit only: vitest run test/unit
npm test # unit: vitest run test/unit
npm run test:integration # integration suite
npm run test:all
npm run test:coverage
npx tsc --noEmit # typecheck (matches CI)
```
@@ -42,11 +42,8 @@ npm run test:e2e # Playwright (requires gitnexus serve + npm run dev)
A husky pre-commit hook (`.husky/pre-commit`) runs automatically on every `git commit`:
1. **Formatting** — `lint-staged` runs prettier on staged files
2. **`gitnexus-web/` files staged** → `tsc -b --noEmit`
3. **`gitnexus/` files staged** → `tsc --noEmit`
Tests do **not** run in the pre-commit hook — they run in CI (`ci-tests.yml`) only.
- **`gitnexus-web/` files staged** → `tsc -b --noEmit` + `vitest run`
- **`gitnexus/` files staged** → `tsc --noEmit` + `vitest run --project default`
Skip with `git commit --no-verify` (use sparingly).
@@ -80,7 +77,7 @@ Re-run the full relevant suite when:
GitHub Actions (`.github/workflows/ci.yml`) orchestrate:
- **`ci-quality.yml`** — prettier format check, eslint lint, `tsc --noEmit` for `gitnexus/`, `tsc -b --noEmit` for `gitnexus-web/`
- **`ci-quality.yml`** — `tsc --noEmit` for `gitnexus/` + `tsc -b --noEmit` for `gitnexus-web/`
- **`ci-tests.yml`** — `vitest run` with coverage (ubuntu) + cross-platform (macOS, Windows)
- **`ci-e2e.yml`** — Playwright E2E tests, gated on `gitnexus-web/**` changes
@@ -1,76 +0,0 @@
# Sigstore policy-controller ClusterImagePolicy for GitNexus container images.
#
# This enforces — at admission time — that every Pod pulling a
# `ghcr.io/abhigyanpatwari/gitnexus` or `gitnexus-web` image is using a build
# that was Cosign-keyless-signed by this repository's `docker.yml` workflow
# running from a `vX.Y.Z` git tag. Unsigned images, images signed by other
# workflows, and images signed from unprotected refs (e.g. `main`, PR branches)
# are rejected.
#
# Prerequisites
# -------------
# 1. Install the Sigstore policy-controller in your cluster (Helm):
#
# helm repo add sigstore https://sigstore.github.io/helm-charts
# helm repo update
# helm install policy-controller -n cosign-system --create-namespace \
# sigstore/policy-controller
#
# 2. Opt namespaces in to verification:
#
# kubectl label namespace <your-ns> policy.sigstore.dev/include=true
#
# 3. Apply this policy:
#
# kubectl apply -f deploy/kubernetes/cluster-image-policy.yaml
#
# After this, `kubectl run --image=ghcr.io/abhigyanpatwari/gitnexus:<tag>` in
# any opted-in namespace will only succeed if the image carries a valid
# Sigstore signature with the pinned identity.
#
# References
# - https://docs.sigstore.dev/policy-controller/overview/
# - https://github.com/sigstore/policy-controller
apiVersion: policy.sigstore.dev/v1beta1
kind: ClusterImagePolicy
metadata:
name: gitnexus-signed-images
spec:
# Apply to both published GitNexus images on both registries. Image
# references always carry a tag or digest at admission time, so these globs
# cover every `gitnexus:<tag>`, `gitnexus@sha256:...`, `gitnexus-web:<tag>`,
# and `gitnexus-web@sha256:...` reference on either GHCR or Docker Hub.
# The Docker Hub images are byte-for-byte mirrors of the GHCR images (same
# build, same digest, same Cosign signature), so the same keyless identity
# authority verifies both.
images:
- glob: 'ghcr.io/abhigyanpatwari/gitnexus*'
# Docker Hub references can appear in three forms at admission time
# (`docker.io/...`, `index.docker.io/...`, and bare `akonlabs/...` with
# the default registry implied). List all three so the policy cannot be
# sidestepped by the choice of registry prefix. The Docker Hub namespace
# is `akonlabs` rather than `abhigyanpatwari` because the Docker Hub org
# differs from the GitHub org.
- glob: 'docker.io/akonlabs/gitnexus*'
- glob: 'index.docker.io/akonlabs/gitnexus*'
- glob: 'akonlabs/gitnexus*'
authorities:
- name: gitnexus-cosign-keyless
keyless:
# Public-good Sigstore Fulcio root.
url: https://fulcio.sigstore.dev
identities:
# Pin both the OIDC issuer (GitHub Actions) AND the exact workflow
# path running from a `vX.Y.Z` (or `vX.Y.Z-prerelease`) tag. Same
# regex the README's `cosign verify` example uses; it rejects:
# * unsigned images
# * signatures from any other repo / workflow
# * signatures from non-tag refs (main, PRs, release branches)
# * signatures from arbitrary non-semver tags
- issuer: https://token.actions.githubusercontent.com
subjectRegExp: ^https://github\.com/abhigyanpatwari/GitNexus/\.github/workflows/docker\.yml@refs/tags/v[0-9]+\.[0-9]+\.[0-9]+(-[a-zA-Z0-9.]+)?$
# Cross-check the signature against the public Rekor transparency log,
# so an attacker who briefly compromised Fulcio cannot retroactively
# mint a signature without leaving a public, append-only audit record.
ctlog:
url: https://rekor.sigstore.dev
-45
View File
@@ -1,45 +0,0 @@
services:
gitnexus-server:
image: ${SERVER_IMAGE:-ghcr.io/abhigyanpatwari/gitnexus:latest}
container_name: ${SERVER_CONTAINER_NAME:-gitnexus-server}
# Map the server to the same host port the web UI expects by default
# (http://localhost:4747). The browser runs on the host, so the UI's
# built-in default works without any reconfiguration.
ports:
- '${SERVER_HOST_PORT:-4747}:4747'
volumes:
# Persist the global registry, indexes, and cloned repos across runs.
- gitnexus-data:/data/gitnexus
# Optional: mount a host workspace so `gitnexus index <path>` can see
# repos you already have on disk. The default points at an empty
# `./workspace/` sibling that compose will create on first start —
# it intentionally does NOT bind-mount the repo root, which would
# expose `.git`, `.env`, and CI secrets to the container.
# Override with `WORKSPACE_DIR=/abs/path/to/your/repos`.
- ${WORKSPACE_DIR:-./workspace}:/workspace:ro
restart: unless-stopped
healthcheck:
test: ['CMD', 'curl', '-fsS', 'http://localhost:4747/api/heartbeat']
interval: 30s
timeout: 5s
retries: 3
start_period: 15s
gitnexus-web:
image: ${WEB_IMAGE:-ghcr.io/abhigyanpatwari/gitnexus-web:latest}
container_name: ${WEB_CONTAINER_NAME:-gitnexus-web}
ports:
- '${WEB_HOST_PORT:-4173}:4173'
depends_on:
gitnexus-server:
condition: service_healthy
restart: unless-stopped
healthcheck:
test: ['CMD', 'curl', '-f', 'http://localhost:4173/']
interval: 30s
timeout: 5s
retries: 3
start_period: 10s
volumes:
gitnexus-data:
-81
View File
@@ -1,81 +0,0 @@
import { createReadStream } from 'node:fs';
import { stat } from 'node:fs/promises';
import { createServer } from 'node:http';
import { extname, join, normalize, sep } from 'node:path';
const host = '0.0.0.0';
const port = Number(process.env.PORT || '4173');
const root = join(process.cwd(), 'dist');
const contentTypes = {
'.css': 'text/css; charset=utf-8',
'.html': 'text/html; charset=utf-8',
'.js': 'text/javascript; charset=utf-8',
'.json': 'application/json; charset=utf-8',
'.map': 'application/json; charset=utf-8',
'.png': 'image/png',
'.svg': 'image/svg+xml',
'.txt': 'text/plain; charset=utf-8',
'.woff': 'font/woff',
'.woff2': 'font/woff2',
};
function resolvePath(urlPath) {
let decoded;
try {
decoded = decodeURIComponent(urlPath);
} catch {
return null;
}
if (decoded.includes('\0')) return null;
const cleanPath = normalize(decoded.replace(/^\/+/, ''));
const candidate = join(root, cleanPath);
if (candidate !== root && !candidate.startsWith(root + sep)) return null;
return candidate;
}
const server = createServer(async (req, res) => {
const requestPath = req.url?.split('?')[0] || '/';
let filePath = resolvePath(requestPath);
if (!filePath) {
res.writeHead(400);
res.end('Bad request');
return;
}
try {
const fileStat = await stat(filePath).catch(() => null);
if (fileStat?.isDirectory()) {
filePath = join(filePath, 'index.html');
} else if (!fileStat?.isFile()) {
filePath = join(root, 'index.html');
}
const finalStat = await stat(filePath).catch(() => null);
if (!finalStat?.isFile()) {
res.writeHead(404);
res.end('Not found');
return;
}
res.writeHead(200, {
'Cache-Control': filePath.includes('/assets/')
? 'public, max-age=31536000, immutable'
: 'no-cache',
'Content-Type': contentTypes[extname(filePath)] || 'application/octet-stream',
'Cross-Origin-Opener-Policy': 'same-origin',
'Cross-Origin-Embedder-Policy': 'require-corp',
});
const stream = createReadStream(filePath);
stream.on('error', () => res.destroy());
stream.pipe(res);
} catch (error) {
res.writeHead(500);
res.end(error instanceof Error ? error.message : 'Internal server error');
}
});
server.listen(port, host, () => {
console.log(`gitnexus-web listening on http://${host}:${port}`);
});
-107
View File
@@ -1,107 +0,0 @@
import { mkdir, mkdtemp, rm, unlink, writeFile } from 'node:fs/promises';
import http, { createServer } from 'node:http';
import { tmpdir } from 'node:os';
import { dirname, join } from 'node:path';
import { spawn } from 'node:child_process';
import { fileURLToPath } from 'node:url';
import { after, before, it } from 'node:test';
import assert from 'node:assert/strict';
const __dirname = dirname(fileURLToPath(import.meta.url));
const serverScript = join(__dirname, 'docker-server.mjs');
function getFreePort() {
return new Promise((resolve) => {
const s = createServer();
s.listen(0, '127.0.0.1', () => {
const { port } = s.address();
s.close(() => resolve(port));
});
});
}
function rawGet(port, path) {
return new Promise((resolve, reject) => {
const req = http.request({ host: '127.0.0.1', port, path }, (res) => {
let body = '';
res.setEncoding('utf8');
res.on('data', (chunk) => {
body += chunk;
});
res.on('end', () => resolve({ status: res.statusCode, headers: res.headers, body }));
});
req.on('error', reject);
req.end();
});
}
async function waitForServer(port, retries = 30) {
for (let i = 0; i < retries; i++) {
try {
await rawGet(port, '/');
return;
} catch {
await new Promise((r) => setTimeout(r, 100));
}
}
throw new Error('Server did not start in time');
}
let tmpDir, serverPort, child;
before(async () => {
tmpDir = await mkdtemp(join(tmpdir(), 'gitnexus-docker-test-'));
const distDir = join(tmpDir, 'dist');
const assetsDir = join(distDir, 'assets');
await mkdir(assetsDir, { recursive: true });
await writeFile(join(distDir, 'index.html'), '<html><body>spa</body></html>');
await writeFile(join(assetsDir, 'app.abc123.js'), 'console.log("app")');
serverPort = await getFreePort();
child = spawn(process.execPath, [serverScript], {
cwd: tmpDir,
env: { ...process.env, PORT: String(serverPort) },
stdio: 'pipe',
});
child.on('error', (err) => {
throw err;
});
await waitForServer(serverPort);
});
after(async () => {
child?.kill();
if (tmpDir) await rm(tmpDir, { recursive: true, force: true });
});
it('serves a valid asset with immutable cache header', async () => {
const res = await rawGet(serverPort, '/assets/app.abc123.js');
assert.equal(res.status, 200);
assert.match(res.headers['cache-control'], /immutable/);
assert.equal(res.headers['cross-origin-opener-policy'], 'same-origin');
assert.equal(res.headers['cross-origin-embedder-policy'], 'require-corp');
});
it('serves SPA fallback for unknown routes', async () => {
const res = await rawGet(serverPort, '/some/unknown/route');
assert.equal(res.status, 200);
assert.match(res.body, /spa/);
assert.match(res.headers['cache-control'], /no-cache/);
});
it('rejects path traversal with 400', async () => {
const res = await rawGet(serverPort, '/../../../etc/passwd');
assert.equal(res.status, 400);
});
it('rejects percent-encoded null bytes with 400', async () => {
const res = await rawGet(serverPort, '/foo%00bar');
assert.equal(res.status, 400);
});
it('returns 404 when dist/index.html is missing', async () => {
await unlink(join(tmpDir, 'dist', 'index.html'));
const res = await rawGet(serverPort, '/nonexistent-page');
assert.equal(res.status, 404);
});
-295
View File
@@ -1,295 +0,0 @@
# Using GitNexus across gRPC microservices
## When to use this guide
This guide is for teams whose product lives in **several separate Git repositories** — one per service — and whose services talk to each other over **gRPC** (possibly alongside HTTP and message topics). GitNexus indexes each repo independently, then a _group_ stitches the per-repo indexes into a single cross-repo view that the `impact`, `query`, and `context` tools can traverse. If your services live in one monorepo, much of this still applies — set each service as a member of a group and use the `service` prefix to scope queries — but the walkthrough assumes the harder multi-repo case.
## Mental model
- Each repository has its own `.gitnexus/` index (a LadybugDB graph of symbols, relationships, processes). `gitnexus analyze` in each repo produces that index completely independently.
- A **group** is a higher-level construct stored at `~/.gitnexus/groups/<group>/` that references the per-repo indexes by their registry name.
- Sync-time extractors walk each member repo and emit **contracts** — provider or consumer records keyed by a canonical `contractId` (`grpc::auth.AuthService/Login`, `http::GET::/orders`, etc.).
- The sync step matches providers and consumers that share a `contractId` and writes **cross-links** to `<groupDir>/contracts.json`. Those cross-links are what lets `impact({repo: "@<group>", target: "X"})` hop from one repo into another.
- Contracts come from three places: automatic contract extractors (`grpc-extractor`, `http-route-extractor`, `topic-extractor`), a manifest escape hatch (`config.links` in `group.yaml`), and — for same-name symbol matches where no contract is declared — the exact-match matching cascade in [`matching.ts`](../../gitnexus/src/core/group/matching.ts).
- Each repo stays editable and re-indexable on its own. Re-run `gitnexus analyze` in a repo when it changes, then `gitnexus group sync <group>` to refresh `contracts.json`. `gitnexus group status` reports which members are stale.
## Prerequisites
- GitNexus installed and runnable as `gitnexus` or `npx gitnexus` (see the root [README.md](../../README.md)).
- Each service repository checked out locally. No requirement that they share a parent directory — the group references them by registry name.
- Write access to `~/.gitnexus/` (the default gitnexus home; see `getDefaultGitnexusDir` in [`storage.ts`](../../gitnexus/src/core/group/storage.ts)).
## Step-by-step walkthrough
The example uses three services — a TypeScript API gateway, a Go orders service, and a Python inventory service — with gRPC between them. The gateway is an `orders` consumer; the orders service is both an `orders` provider and an `inventory` consumer; the inventory service is an `inventory` provider.
### 1. Index each repository
Run `analyze` from inside each service repo (or pass the path). The CLI surface lives in [`gitnexus/src/cli/analyze.ts`](../../gitnexus/src/cli/analyze.ts) and is wired in [`gitnexus/src/cli/index.ts`](../../gitnexus/src/cli/index.ts).
```bash
cd ~/code/gateway && npx gitnexus analyze
cd ~/code/orders && npx gitnexus analyze
cd ~/code/inventory && npx gitnexus analyze
```
Useful flags:
- `--force` — reindex even if up to date.
- `--embeddings` — generate embedding vectors (needed only if you want semantic search; the exact-match cross-repo cascade does **not** need them).
- `--name <alias>` — register the repo under a specific alias when two repos share a basename (e.g. two `api/` folders).
- `--skip-git` — index a checkout that isn't a git repo.
Each run writes a `.gitnexus/` folder in the repo and registers the repo in `~/.gitnexus/registry.json`. Confirm with `npx gitnexus list`.
### 2. Author `group.yaml`
Create the group directory and edit the config. Either use the CLI scaffolder or write the file directly — both produce the same shape consumed by [`config-parser.ts`](../../gitnexus/src/core/group/config-parser.ts).
```bash
npx gitnexus group create payments-platform
# or manually:
mkdir -p ~/.gitnexus/groups/payments-platform
$EDITOR ~/.gitnexus/groups/payments-platform/group.yaml
```
Minimal working `group.yaml`:
```yaml
version: 1
name: payments-platform
description: Gateway + orders + inventory (gRPC)
repos:
gateway: gateway
orders: orders
inventory: inventory
# Only add explicit links when the automatic extractors miss something —
# see "When automatic extraction isn't enough" below.
links: []
packages: {}
detect:
http: true
grpc: true
topics: true
shared_libs: true
embedding_fallback: false
matching:
bm25_threshold: 0.7
embedding_threshold: 0.65
max_candidates_per_step: 3
```
Field notes (schema in [`types.ts`](../../gitnexus/src/core/group/types.ts)):
- `version` — must be `1`. The parser rejects anything else.
- `name` — required; used for the group directory name and all CLI / MCP calls.
- `repos` — a mapping from **group path** (a logical name you choose; can be a hierarchy like `backend/orders`) to **registry name** (the name shown by `npx gitnexus list`). Both sides appear throughout the tooling: contract rows use the group path; `@<group>/<groupPath>` routes tools to a single member.
- `links` — optional manifest escape hatch, one entry per explicit cross-repo contract. Validated by the parser: `from` and `to` must be known repo paths, `type` must be one of `http | grpc | topic | lib | custom`, and `role` must be `provider | consumer`.
- `detect` — toggles per extractor family. Defaults (set in `config-parser.ts`) turn `http`, `grpc`, `topics`, and `shared_libs` on; disable the ones you don't use to speed up sync.
- `matching` — thresholds for the matching cascade. The exact match is always run; other strategies depend on indexer state.
### 3. Sync the group
```bash
npx gitnexus group sync payments-platform --verbose
```
What this does (see [`sync.ts`](../../gitnexus/src/core/group/sync.ts)):
1. Opens each member's per-repo LadybugDB.
2. Runs the HTTP, gRPC, and topic extractors against the source files.
3. Applies manifest `links` through [`manifest-extractor.ts`](../../gitnexus/src/core/group/extractors/manifest-extractor.ts).
4. Runs the exact-match cascade, joining providers and consumers that share a normalized `contractId`.
5. Writes `contracts.json` in the group directory.
Flags:
- `--exact-only` — stop after the exact cascade; skip BM25 and embedding fallback.
- `--skip-embeddings` — run exact plus BM25 but not embedding-based matching.
- `--allow-stale` — don't warn if a member's index is stale.
- `--json` — machine-readable output.
The same operation is available over MCP as `group_sync({ name: "payments-platform" })` — see [`tools.ts`](../../gitnexus/src/mcp/tools.ts).
### 4. Inspect the registry
Use `gitnexus group contracts` for the CLI view or read the `gitnexus://group/<name>/contracts` MCP resource for the same data.
```bash
npx gitnexus group contracts payments-platform --type grpc --json
```
A shortened response:
```json
{
"contracts": [
{
"contractId": "grpc::orders.OrderService/PlaceOrder",
"type": "grpc",
"role": "provider",
"repo": "orders",
"symbolRef": { "filePath": "internal/grpc/order_server.go", "name": "RegisterOrderServiceServer" },
"confidence": 0.8,
"meta": { "service": "OrderService", "method": "PlaceOrder", "source": "go_register" }
},
{
"contractId": "grpc::orders.OrderService/PlaceOrder",
"type": "grpc",
"role": "consumer",
"repo": "gateway",
"symbolRef": { "filePath": "src/clients/orders.ts", "name": "OrderServiceClient" },
"confidence": 0.75,
"meta": { "service": "OrderService", "source": "ts_generated_client" }
}
],
"crossLinks": [
{
"from": { "repo": "gateway", "symbolUid": "…", "symbolRef": { "filePath": "src/clients/orders.ts", "name": "OrderServiceClient" } },
"to": { "repo": "orders", "symbolUid": "…", "symbolRef": { "filePath": "internal/grpc/order_server.go", "name": "RegisterOrderServiceServer" } },
"type": "grpc",
"contractId": "grpc::orders.OrderService/PlaceOrder",
"matchType": "exact",
"confidence": 1.0
}
]
}
```
Staleness of the underlying indexes shows up in `npx gitnexus group status payments-platform` or the `gitnexus://group/<name>/status` resource.
### 5. Run cross-repo impact with `@<group>` routing
From any shell (you do **not** have to `cd` into a member repo), the normal `impact` / `query` / `context` tools accept `repo: "@<group>"` to fan out across all members, or `repo: "@<group>/<memberPath>"` to target one member. Routing is implemented in [`resolve-at-member.ts`](../../gitnexus/src/core/group/resolve-at-member.ts) and described in [`tools.ts`](../../gitnexus/src/mcp/tools.ts).
Example MCP calls:
```json
{"tool": "impact", "arguments": {
"repo": "@payments-platform/orders",
"target": "PlaceOrder",
"direction": "upstream",
"crossDepth": 2
}}
```
```json
{"tool": "query", "arguments": {
"repo": "@payments-platform",
"query": "retry logic around PlaceOrder"
}}
```
The CLI equivalents still exist for scripting:
```bash
npx gitnexus group impact payments-platform \
--repo orders --target PlaceOrder --direction upstream --cross-depth 2
```
Phase 1 walks within the anchor member; Phase 2 hops across the Contract Bridge wherever a cross-link endpoint matches an impacted symbol. See [`cross-impact.ts`](../../gitnexus/src/core/group/cross-impact.ts) for the bridge query.
## How gRPC extraction works
`GrpcExtractor` ([`grpc-extractor.ts`](../../gitnexus/src/core/group/extractors/grpc-extractor.ts)) runs two passes per member repo:
1. **Proto map.** Every `**/*.proto` file is parsed to enumerate `service Foo { rpc Bar(...) }` blocks and (transitively) resolve the package name. Each RPC method becomes a provider contract with `contractId = grpc::<package>.<Service>/<Method>` and `confidence = 0.85`. Parsing uses the vendored `tree-sitter-proto` grammar when available and falls back to a length-preserving manual parser (`extractServiceBlocks`) otherwise, so `.proto` extraction works on platforms where the grammar fails to build.
2. **Source scan.** Every source file whose extension matches [`GRPC_SCAN_GLOB`](../../gitnexus/src/core/group/extractors/grpc-patterns/index.ts) is parsed by its language plugin:
| Language | Provider signal | Consumer signal |
|----------|-----------------|-----------------|
| Go ([`go.ts`](../../gitnexus/src/core/group/extractors/grpc-patterns/go.ts)) | `pb.RegisterXxxServer(...)`, `pb.UnimplementedXxxServer` embedded in struct | `pb.NewXxxClient(conn)` |
| Java ([`java.ts`](../../gitnexus/src/core/group/extractors/grpc-patterns/java.ts)) | `extends XxxServiceGrpc.XxxServiceImplBase` (with or without `@GrpcService`) | `XxxServiceGrpc.newBlockingStub(...)`, `newStub(...)` |
| Python ([`python.ts`](../../gitnexus/src/core/group/extractors/grpc-patterns/python.ts)) | `add_XxxServicer_to_server(...)` (bare or `_pb2_grpc.` attribute form) | `XxxStub(channel)` (ignores `Mock`/`Test`/`Fake`/`Stub`) |
| Node / TS ([`node.ts`](../../gitnexus/src/core/group/extractors/grpc-patterns/node.ts)) | NestJS `@GrpcMethod('Service','Method')` | `@GrpcClient` field typed `XxxServiceClient`, `client.getService<X>('Service')`, `new XxxServiceClient(...)`, `new foo.bar.XxxService(...)` in files that call `loadPackageDefinition` |
For each source-scan detection the extractor looks up the short service name in the proto map and picks:
- `grpc::<package>.<Service>/<Method>` when a method is named and the service resolves against the proto map,
- `grpc::<package>.<Service>/*` (wildcard) when only the service is known, or
- `grpc::<ServiceName>/*` when no `.proto` is available at all.
Provider detections land at confidence 0.8 (with proto) or 0.65 (without); consumers at 0.75 or 0.55. NestJS `@GrpcMethod` is fixed at 0.8 because the decorator is self-describing.
### Matching
`matching.ts` lowercases the package/service segment before comparing contract ids, so bindings that capitalize names differently (`auth.AuthService` vs `auth.authservice`) still match. Method names are compared case-sensitively because gRPC's wire path is case-sensitive. Service-only wildcards (`grpc::pkg.Svc/*`) match any method on the same service during cross-linking.
### Known limitations
- **Ambiguous proto resolution.** If a short service name exists in more than one `.proto` file and the source-scan hit can't be narrowed down by shared directory segments (`resolveProtoConflict` refuses to guess), the extractor skips contract emission and logs a warning.
- **Proto packages must be resolvable locally.** Transitive imports that point outside the repo produce an empty package segment, which means the contract id collapses to `grpc::<Service>/<Method>`. Cross-repo matches still work as long as both sides agree on the empty package.
- **Rewrite rules are not implemented.** If the provider repo writes `grpc::orders.OrderService/PlaceOrder` and the consumer repo writes `grpc::orderspb.OrderService/PlaceOrder`, they won't cross-link automatically. Use `config.links` to declare the correspondence (see below).
- **One sync = one snapshot.** Contracts are extracted against the indexed snapshot of each repo. Re-index first, then re-sync; the `status` command and resource surface staleness.
## When automatic extraction isn't enough
The escape hatch is the `links` list in `group.yaml`, handled by [`ManifestExtractor`](../../gitnexus/src/core/group/extractors/manifest-extractor.ts). Each entry is a **one-directional** provider/consumer declaration:
```yaml
version: 1
name: payments-platform
repos:
gateway: gateway
orders: orders
inventory: inventory
links:
# Explicit gRPC method: use when naming mismatches stop the
# automatic matcher from cross-linking.
- from: gateway
to: orders
type: grpc
contract: OrderService/PlaceOrder
role: consumer
# Service-level link when you don't want to enumerate methods.
- from: orders
to: inventory
type: grpc
contract: InventoryService
role: consumer
# Works for HTTP too — use `METHOD::/path` form for the exact
# handler, or just `/path` for a method-agnostic wildcard.
- from: gateway
to: orders
type: http
contract: POST::/orders
role: consumer
```
What the manifest extractor does (see [`manifest-extractor.ts`](../../gitnexus/src/core/group/extractors/manifest-extractor.ts)):
1. Builds a canonical `contractId` with `buildContractId` — the same canonicalization used by the automatic extractors, so manifest links cross-match automatic contracts on the other side.
2. Tries to resolve each side to a real graph symbol (the `Route` node for HTTP, a `Function|Method` / `Class|Interface` for gRPC, a `Package|Module` for `lib`).
3. If resolution fails, falls back to a deterministic synthetic uid (`manifest::<repo>::<contractId>`) so both sides still line up in cross-impact — name-only links still work when the symbol isn't in the graph.
4. Emits both a provider and a consumer `StoredContract` (confidence `1.0`, `source: "manifest"`) and a `CrossLink` with `matchType: "manifest"`.
Use `links` for exactly the cases the extractor can't infer: different package names across repos (see #701), hand-rolled transports, cases where the provider repo isn't checked out locally but you still want a record, or any contract whose provider and consumer simply don't share a surface the extractors know how to pattern-match.
History: the manifest extractor used to be silently skipped by the sync pipeline; that was fixed in [#827](https://github.com/abhigyanpatwari/GitNexus/pull/827) (tracking issue #826). If you ever see `config.links` with zero cross-links in `contracts.json`, make sure you're on a build that includes that fix, then re-run `group sync`.
## Troubleshooting
1. **`contracts.json` is empty after a sync.** Either no member repo contained a recognizable gRPC pattern, or the extractors are disabled in `detect`. Confirm `detect.grpc: true` and re-run with `--verbose`.
2. **A known provider/consumer pair doesn't cross-link.** Most common cause: the package segment differs. Check the raw contract ids with `gitnexus group contracts <name> --unmatched` — if you see two same-method contracts with different package prefixes, add a manifest `links:` entry to bridge them (no automatic rewrite rules yet).
3. **`matchType: "manifest"` is missing entirely.** The extractor needs `config.links` to be non-empty and the sync pipeline to actually call it — verify you're on a post-#827 build. Empty contract rows for manifest links usually mean `resolveSymbol` couldn't find a graph match; the synthetic uid still lets cross-impact work, it just won't carry a file path.
4. **Ambiguous proto warnings.** Look for `[grpc-extractor] Ambiguous proto resolution` in the sync logs; that means a service name exists in multiple `.proto` files under the same repo and the path-distance heuristic couldn't pick a winner. Resolve by renaming the service or declaring the intended pairing in `config.links`.
5. **Cross-impact says "stale".** Both sides need a fresh per-repo index _and_ a fresh group sync. Order matters: `gitnexus analyze` in each changed repo, then `gitnexus group sync <name>`. Use `gitnexus group status <name>` to see which side is behind.
## Related docs and references
- [AGENTS.md](../../AGENTS.md) — authoritative list of MCP tools and resources, including group-mode routing and the `gitnexus://group/…` resources.
- [ARCHITECTURE.md](../../ARCHITECTURE.md) — overall data flow and the call-resolution DAG that the per-repo indexer uses.
- [`gitnexus/src/core/group/`](../../gitnexus/src/core/group/) — `service.ts`, `sync.ts`, `config-parser.ts`, `matching.ts`.
- [`gitnexus/src/core/group/extractors/grpc-extractor.ts`](../../gitnexus/src/core/group/extractors/grpc-extractor.ts) and [`grpc-patterns/`](../../gitnexus/src/core/group/extractors/grpc-patterns/) — gRPC detection.
- [`gitnexus/src/core/group/extractors/manifest-extractor.ts`](../../gitnexus/src/core/group/extractors/manifest-extractor.ts) — the `config.links` escape hatch.
- [`gitnexus/src/mcp/tools.ts`](../../gitnexus/src/mcp/tools.ts) — MCP tool schemas (`group_list`, `group_sync`, plus `@<group>` routing on `impact` / `query` / `context`).
- [`gitnexus/src/cli/group.ts`](../../gitnexus/src/cli/group.ts) — CLI command definitions and flags.
- Upstream issues: [#701](https://github.com/abhigyanpatwari/GitNexus/issues/701), [#826](https://github.com/abhigyanpatwari/GitNexus/issues/826), [#906](https://github.com/abhigyanpatwari/GitNexus/issues/906).
@@ -21,7 +21,6 @@ Run from the project root. This parses all source files, builds the knowledge gr
|------|--------|
| `--force` | Force full re-index even if up to date |
| `--embeddings` | Enable embedding generation for semantic search (off by default) |
| `--drop-embeddings` | Drop existing embeddings on rebuild. By default, an `analyze` without `--embeddings` preserves them. |
**When to run:** First time in a project, after major code changes, or when `gitnexus://repo/{name}/context` reports the index is stale.
+4 -4
View File
@@ -8,13 +8,13 @@
"name": "gitnexus-shared",
"version": "1.0.0",
"devDependencies": {
"typescript": "^6.0.3"
"typescript": "^6.0.2"
}
},
"node_modules/typescript": {
"version": "6.0.3",
"resolved": "https://registry.npmjs.org/typescript/-/typescript-6.0.3.tgz",
"integrity": "sha512-y2TvuxSZPDyQakkFRPZHKFm+KKVqIisdg9/CZwm9ftvKXLP8NRWj38/ODjNbr43SsoXqNuAisEf1GdCxqWcdBw==",
"version": "6.0.2",
"resolved": "https://registry.npmjs.org/typescript/-/typescript-6.0.2.tgz",
"integrity": "sha512-bGdAIrZ0wiGDo5l8c++HWtbaNCWTS4UTv7RaTH/ThVIgjkveJt83m74bBHMJkuCbslY8ixgLBVZJIOiQlQTjfQ==",
"dev": true,
"license": "Apache-2.0",
"bin": {
+1 -1
View File
@@ -20,6 +20,6 @@
"src"
],
"devDependencies": {
"typescript": "^6.0.3"
"typescript": "^6.0.2"
}
}
-16
View File
@@ -131,20 +131,4 @@ export interface GraphRelationship {
confidence: number;
reason: string;
step?: number;
/**
* Per-signal evidence trace for edges emitted by the scope-based
* resolution pipeline (RFC #909 Ring 2 PKG #925). Populated by
* `emit-references.ts` when draining `ReferenceIndex` into the graph
* so downstream query / audit tools can inspect *why* a given edge
* was emitted with its confidence value.
*
* Optional and additive — every existing edge emitter ignores this
* field, and every existing query continues to work whether or not
* an edge carries it.
*/
evidence?: readonly {
readonly kind: string;
readonly weight: number;
readonly note?: string;
}[];
}
-129
View File
@@ -23,132 +23,3 @@ export type { MroStrategy } from './mro-strategy.js';
// Pipeline progress
export type { PipelinePhase, PipelineProgress } from './pipeline.js';
// ─── Scope-based resolution — RFC #909 (Ring 1 #910) ────────────────────────
// Data model (RFC §2)
export type { SymbolDefinition } from './scope-resolution/symbol-definition.js';
export type {
ScopeId,
DefId,
ScopeKind,
Range,
Capture,
CaptureMatch,
BindingRef,
ImportEdge,
TypeRef,
Scope,
ResolutionEvidence,
Resolution,
Reference,
ReferenceIndex,
LookupParams,
RegistryContributor,
ParsedImport,
ParsedTypeBinding,
WorkspaceIndex,
Callsite,
ScopeLookup,
} from './scope-resolution/types.js';
// Evidence + tie-break constants (RFC Appendix A, Appendix B)
export { EvidenceWeights, typeBindingWeightAtDepth } from './scope-resolution/evidence-weights.js';
export { ORIGIN_PRIORITY } from './scope-resolution/origin-priority.js';
export type { OriginForTieBreak } from './scope-resolution/origin-priority.js';
// Language classification (RFC §6.1 Ring 3/4 governance)
export {
LanguageClassifications,
isProductionLanguage,
} from './scope-resolution/language-classification.js';
export type { LanguageClassification } from './scope-resolution/language-classification.js';
// Core indexes over per-file artifacts (RFC §3.1; Ring 2 SHARED #913)
export { buildDefIndex } from './scope-resolution/def-index.js';
export type { DefIndex } from './scope-resolution/def-index.js';
export { buildModuleScopeIndex } from './scope-resolution/module-scope-index.js';
export type { ModuleScopeIndex, ModuleScopeEntry } from './scope-resolution/module-scope-index.js';
export { buildQualifiedNameIndex } from './scope-resolution/qualified-name-index.js';
export type { QualifiedNameIndex } from './scope-resolution/qualified-name-index.js';
// Strict type-reference resolver (RFC §4.6; Ring 2 SHARED #916)
// `ScopeLookup` is defined in `./scope-resolution/types.js` and exported
// from the type-export block above — not from this module.
export { resolveTypeRef } from './scope-resolution/resolve-type-ref.js';
export type { ResolveTypeRefContext } from './scope-resolution/resolve-type-ref.js';
// ScopeExtractor output contracts (RFC §3.2 Phase 1; Ring 2 PKG #919)
export type { ParsedFile } from './scope-resolution/parsed-file.js';
export type { ReferenceSite, ReferenceKind, CallForm } from './scope-resolution/reference-site.js';
// Method-dispatch materialized view over HeritageMap (RFC §3.1; Ring 2 SHARED #914)
export { buildMethodDispatchIndex } from './scope-resolution/method-dispatch-index.js';
export type {
MethodDispatchIndex,
MethodDispatchInput,
} from './scope-resolution/method-dispatch-index.js';
// SCC-aware cross-file finalize (RFC §3.2 Phase 2; Ring 2 SHARED #915)
export { finalize } from './scope-resolution/finalize-algorithm.js';
export type {
FinalizeInput,
FinalizeFile,
FinalizeHooks,
FinalizeOutput,
FinalizedScc,
FinalizeStats,
} from './scope-resolution/finalize-algorithm.js';
// Scope-aware registries + 7-step lookup (RFC §4; Ring 2 SHARED #917)
export { buildClassRegistry } from './scope-resolution/registries/class-registry.js';
export type { ClassRegistry } from './scope-resolution/registries/class-registry.js';
export { buildMethodRegistry } from './scope-resolution/registries/method-registry.js';
export type {
MethodRegistry,
MethodLookupOptions,
} from './scope-resolution/registries/method-registry.js';
export { buildFieldRegistry } from './scope-resolution/registries/field-registry.js';
export type {
FieldRegistry,
FieldLookupOptions,
} from './scope-resolution/registries/field-registry.js';
export { lookupCore } from './scope-resolution/registries/lookup-core.js';
export type { CoreLookupParams } from './scope-resolution/registries/lookup-core.js';
export { lookupQualified } from './scope-resolution/registries/lookup-qualified.js';
export type { LookupQualifiedParams } from './scope-resolution/registries/lookup-qualified.js';
export { composeEvidence, confidenceFromEvidence } from './scope-resolution/registries/evidence.js';
export type { RawSignals } from './scope-resolution/registries/evidence.js';
export {
compareByConfidenceWithTiebreaks,
CONFIDENCE_EPSILON,
} from './scope-resolution/registries/tie-breaks.js';
export type { TieBreakKey } from './scope-resolution/registries/tie-breaks.js';
export { CLASS_KINDS, METHOD_KINDS, FIELD_KINDS } from './scope-resolution/registries/context.js';
export type {
RegistryContext,
RegistryProviders,
OwnerScopedContributor,
ArityVerdict,
} from './scope-resolution/registries/context.js';
// Scope tree spine + position lookup (RFC §2.2 + §3.1; Ring 2 SHARED #912)
export { makeScopeId, clearScopeIdInternPool } from './scope-resolution/scope-id.js';
export type { ScopeIdInput } from './scope-resolution/scope-id.js';
export {
buildScopeTree,
canParentScope,
ScopeTreeInvariantError,
} from './scope-resolution/scope-tree.js';
export type { ScopeTree } from './scope-resolution/scope-tree.js';
export { buildPositionIndex } from './scope-resolution/position-index.js';
export type { PositionIndex } from './scope-resolution/position-index.js';
// Shadow-mode diff + aggregation (RFC §6.3; Ring 2 SHARED #918)
export { diffResolutions } from './scope-resolution/shadow/diff.js';
export type {
ShadowAgreement,
ShadowCallsite,
ShadowDiff,
} from './scope-resolution/shadow/diff.js';
export { aggregateDiffs } from './scope-resolution/shadow/aggregate.js';
export type { LanguageParityRow, ShadowParityReport } from './scope-resolution/shadow/aggregate.js';
@@ -30,7 +30,6 @@ export const NODE_TABLES = [
'TypeAlias',
'Const',
'Static',
'Variable',
'Property',
'Record',
'Delegate',
+14 -37
View File
@@ -1,46 +1,23 @@
/**
* MRO (Method Resolution Order) strategy — shared canonical definition.
* MRO (Method Resolution Order) strategy — shared between CLI and any
* future consumer that reasons about multiple-inheritance semantics.
*
* Lives in `gitnexus-shared` so `model/resolve.ts` and `mro-processor.ts` share
* the type without importing the language registry (avoids circular coupling).
* Lives in `gitnexus-shared` so the low-level resolution module
* (`core/ingestion/model/resolve.ts`) does not need to import from
* `languages/` — keeping the `model/` layer free of language-registry
* coupling.
*
* `first-wins` (default, Java/C#/Kotlin/Go/Swift/Dart):
* BFS ancestor walk in declaration order; first match wins.
*
* `leftmost-base` (C++):
* BFS walk; HeritageMap preserves source insertion order, so BFS naturally
* picks the leftmost base in diamond inheritance.
*
* `c3` (Python):
* C3-linearization; falls back to BFS on cyclic/inconsistent hierarchy.
* See model/resolve.ts § c3Linearize.
*
* `implements-split` (Java/C#/Kotlin):
* Low-level lookup is BFS; graph-level mro-processor detects and warns on
* interface-default method ambiguity.
*
* `qualified-syntax` (Rust):
* No auto-resolution — `lookupMethodByOwnerWithMRO` returns undefined immediately.
* Rust requires explicit `<Type as Trait>::method` syntax.
*
* `ruby-mixin` (Ruby):
* Kind-aware walk that does NOT short-circuit on direct owner first (`prepend`
* must beat the class's own method). Walk order:
* 1. Prepend providers (reverse declaration — last-prepended wins)
* 2. Direct owner's own methods
* 3. Include providers (reverse declaration)
* 4. Transitive ancestors (BFS fallback)
* Singleton dispatch: caller passes `ancestryOverride` (extend providers only);
* becomes a simple left-to-right scan. Miss NEVER falls through to file-scoped
* lookup — null-routes or honors `fallback`.
*
* @see model/resolve.ts § lookupMethodByOwnerWithMRO
* @see languages/ruby.ts § selectDispatch
* Strategy semantics:
* - `first-wins`: BFS ancestor walk, first match wins (default).
* - `leftmost-base`: BFS ancestor walk, leftmost base wins (C++).
* - `c3`: C3-linearized ancestor order, first match wins (Python).
* - `implements-split`: BFS walk, first match wins (Java/C#/Kotlin) — full
* interface-default ambiguity is handled at graph level.
* - `qualified-syntax`: No auto-resolution (Rust — requires `<T as Trait>::m`).
*/
export type MroStrategy =
| 'first-wins'
| 'c3'
| 'leftmost-base'
| 'implements-split'
| 'qualified-syntax'
| 'ruby-mixin';
| 'qualified-syntax';
@@ -1,62 +0,0 @@
/**
* `DefIndex` — O(1) `DefId → SymbolDefinition` materialization.
*
* The global "what is this id?" lookup. Every per-kind registry (ClassRegistry,
* MethodRegistry, FieldRegistry) returns `DefId[]` and resolves them back to
* full `SymbolDefinition` records through this index — one central hash map,
* one allocation per def.
*
* Part of RFC #909 Ring 2 SHARED — #913.
*
* Consumed by: #917 (`Registry.lookup` implementations), #915 (SCC finalize).
*/
import type { SymbolDefinition } from './symbol-definition.js';
import type { DefId } from './types.js';
export interface DefIndex {
readonly byId: ReadonlyMap<DefId, SymbolDefinition>;
readonly size: number;
get(id: DefId): SymbolDefinition | undefined;
has(id: DefId): boolean;
}
/**
* Build a `DefIndex` from a flat list of `SymbolDefinition` records.
*
* **Collision policy: first-write-wins.** `DefId` is meant to be unique
* (`nodeId` is the stable graph identifier), so a collision indicates an
* upstream bug — most likely the same symbol parsed twice or a duplicate
* commit into the pipeline. Rather than silently overwriting with a later
* definition that may be partial or wrong, the first record wins and
* subsequent records for the same id are dropped. Pipeline bugs surface
* later as `has(id) === true` but the def looking older than expected,
* which is easier to debug than a silent overwrite.
*
* Pure function — safe to call repeatedly; no side effects.
*/
export function buildDefIndex(defs: readonly SymbolDefinition[]): DefIndex {
const byId = new Map<DefId, SymbolDefinition>();
for (const def of defs) {
if (byId.has(def.nodeId)) continue; // first-write-wins
byId.set(def.nodeId, def);
}
return wrapIndex(byId);
}
// ─── Internal ───────────────────────────────────────────────────────────────
function wrapIndex(byId: Map<DefId, SymbolDefinition>): DefIndex {
return {
byId,
get size() {
return byId.size;
},
get(id: DefId): SymbolDefinition | undefined {
return byId.get(id);
},
has(id: DefId): boolean {
return byId.has(id);
},
};
}
@@ -1,90 +0,0 @@
/**
* `EvidenceWeights` — RFC Appendix A (authoritative values).
*
* Starting calibration for scope-based resolution. Shadow-first rollout
* tunes these against legacy DAG parity. Every `ResolutionEvidence.weight`
* value in the codebase MUST reference this map; inline magic numbers are a
* lint violation. Extends issue #429 (centralize hardcoded confidence values).
*
* Evidence composes additively inside `composeEvidence`; the sum is capped
* at 1.0 in `Resolution.confidence`.
*/
/**
* Authoritative weight map. Keys are a mix of `ResolutionEvidence.kind`
* values and special modifiers (scope-chain depth, MRO depth decay,
* unlinked-import multiplicative cap).
*/
export const EvidenceWeights = {
// ─── Where-found signals (visibility) ─────────────────────────────────────
/** `BindingRef.origin === 'local'` */
local: 0.55,
/** `BindingRef.origin === 'import'` */
import: 0.45,
/** `BindingRef.origin === 'reexport'` */
reexport: 0.4,
/** `BindingRef.origin === 'namespace'` */
namespace: 0.4,
/** `BindingRef.origin === 'wildcard'` */
wildcard: 0.3,
// ─── Scope-chain deduction (per-hop) ──────────────────────────────────────
/** Deducted per parent-hop taken (depth-0 = 0, depth-1 = −0.02, …). */
scopeChainPerDepth: -0.02,
// ─── Receiver-type-binding signal (decays by MRO depth) ───────────────────
/**
* Weight applied when the receiver's type binding resolves to a class that
* declares the candidate as a method/field. Decays by MRO depth: direct
* class = index 0; 1 parent hop = index 1; etc. Falls back to the last
* value for depths beyond the table.
*/
typeBindingByMroDepth: [0.5, 0.42, 0.36, 0.32, 0.3] as const,
// ─── Corroborating signals ────────────────────────────────────────────────
/** `def.ownerId === resolvedReceiver.def.id` (exact owner match). */
ownerMatch: 0.2,
/** Explanatory only — retained for debuggability. Never discriminates
* because surviving candidates already passed `acceptedKinds`. */
kindMatch: 0.0,
// ─── Arity compatibility (from `provider.arityCompatibility`) ─────────────
/** `provider.arityCompatibility(...) === 'compatible'` */
arityMatchCompatible: 0.1,
/** `provider.arityCompatibility(...) === 'unknown'` */
arityMatchUnknown: 0.0,
/** `provider.arityCompatibility(...) === 'incompatible'` — penalizes;
* candidates filtered only when a compatible candidate exists. */
arityMatchIncompatible: -0.15,
// ─── Global fallback (only when nothing lexically visible) ────────────────
/** Hit via `QualifiedNameIndex.byQualifiedName`. */
globalQualified: 0.35,
/** Fallback hit in a `byName` index (and nothing was lexically visible). */
globalName: 0.1,
// ─── Degraded signals ─────────────────────────────────────────────────────
/** Call/reference flowing through a `dynamic-unresolved` edge. */
dynamicImportUnresolved: 0.02,
// ─── Unresolved-import cap (multiplicative, applied per-signal) ───────────
/**
* Multiplicative cap on the edge-derived evidence signal
* (`import`/`wildcard`/`reexport`/`namespace`) when
* `ImportEdge.linkStatus === 'unresolved'`. Independent corroborating
* signals on the same candidate (`owner-match`, `arity-match`,
* `type-binding`) are NOT penalized.
*/
unlinkedImportMultiplier: 0.5,
} as const;
/**
* Look up the `type-binding` signal weight for a given MRO depth, falling
* back to the last tabulated value for depths beyond the table.
*/
export function typeBindingWeightAtDepth(mroDepth: number): number {
const table = EvidenceWeights.typeBindingByMroDepth;
if (mroDepth < 0) return table[0];
if (mroDepth >= table.length) return table[table.length - 1];
return table[mroDepth];
}
@@ -1,969 +0,0 @@
/**
* `finalize` — cross-file finalize algorithm for the SemanticModel
* (RFC §3.2 Phase 2; Ring 2 SHARED #915).
*
* Pure logic that takes per-file parse output (`ParsedImport[]` +
* `SymbolDefinition[]`) and returns:
*
* - Linked `ImportEdge[]` per module scope, with `targetModuleScope` and
* `targetDefId` filled where resolvable; edges that could not be
* resolved within the hard fixpoint cap are marked
* `linkStatus: 'unresolved'`.
* - Materialized `bindings` per module scope — local defs merged with
* imported / wildcard-expanded / re-exported names via the provider's
* `mergeBindings` precedence.
* - The SCC condensation of the import graph, exposed so disjoint SCCs
* can be processed in parallel by callers that want that.
*
* The algorithm is **SCC-aware**: it runs Tarjan SCC over the file-level
* import graph, processes SCCs in reverse-topological order (leaves
* first), and within each SCC runs a bounded fixpoint link pass capped at
* `N = |edges in SCC|`. Cyclic imports finalize without hanging; malformed
* inputs are bounded by the cap.
*
* **No language-specific logic.** Target resolution, wildcard expansion,
* and binding precedence all go through caller-supplied hooks
* (`resolveImportTarget`, `expandsWildcardTo`, `mergeBindings`) that
* match the LanguageProvider surface from #911.
*
* **Non-binding imports rule.** `dynamic-unresolved` passes through with
* `targetFile: null`; `dynamic-resolved` and `side-effect` resolve to
* file-level `ImportEdge`s. None of these materialize `BindingRef`s.
*/
import type { SymbolDefinition } from './symbol-definition.js';
import type { BindingRef, ImportEdge, ParsedImport, ScopeId, WorkspaceIndex } from './types.js';
// ─── Public contracts ───────────────────────────────────────────────────────
/** Per-file input for the finalize pass. */
export interface FinalizeFile {
readonly filePath: string;
/** The module scope id for this file; owns the finalized imports + bindings. */
readonly moduleScope: ScopeId;
readonly parsedImports: readonly ParsedImport[];
/**
* Defs exported from this file — the "what other files can import by name"
* surface. Typically those with `isExported: true` (the module's own
* declarations); parsers MAY also surface re-exported names here as a
* shortcut, but it is no longer required for correctness.
*
* **Multi-hop re-export contract.** `finalize` resolves an edge
* `A → B (importedName: 'X')` by first looking up `X` in `B.localDefs`.
* If `B` only has `export { X } from './C'` and does NOT surface `X` in
* its own `localDefs`, `finalize` falls back to the precomputed
* per-file re-export closure (`buildReexportClosures`), which encodes
* every name reachable through `B`'s named and wildcard re-exports —
* including transitively through cyclic SCCs. The lookup is O(1) and
* inherits the upstream `targetDefId`, populating `transitiveVia` with
* the file paths traversed to reach the leaf def.
*
* Surfacing re-exported names in `localDefs` is still a valid (and
* slightly cheaper) optimization: the direct lookup short-circuits the
* closure consult. Parsers SHOULD prefer surfacing names they can resolve
* statically (e.g., `export { X } from './c'` when `c.ts` is parsed in
* the same workspace), and rely on the closure for the long tail of
* barrel patterns.
*
* The fixpoint does NOT mutate `localDefs` across iterations — it is
* static input.
*/
readonly localDefs: readonly SymbolDefinition[];
}
/** Input to `finalize`. */
export interface FinalizeInput {
readonly files: readonly FinalizeFile[];
/** Opaque workspace context forwarded to provider hooks. */
readonly workspaceIndex: WorkspaceIndex;
}
/**
* Provider-supplied hooks. Mirror the optional LanguageProvider scope-
* resolution hooks declared in #911; `finalize` calls them pure-ly and
* expects pure answers.
*/
export interface FinalizeHooks {
/**
* Resolve a raw import target to the concrete file path that owns it.
* Return `null` when no target file is resolvable (e.g., `np.foo` when
* `numpy` is external to the workspace).
*/
resolveImportTarget(
targetRaw: string,
fromFile: string,
workspaceIndex: WorkspaceIndex,
): string | null;
/**
* For a wildcard `import * from M`, return the names visible in the
* exporting module scope `M`. The finalize pass looks each name up in
* `M`'s local defs to produce a concrete `BindingRef`; names with no
* matching export are dropped.
*/
expandsWildcardTo(targetModuleScope: ScopeId, workspaceIndex: WorkspaceIndex): readonly string[];
/**
* Merge `incoming` bindings into `existing` for a given name. Called
* once per name at each scope. Typical rules:
* - Python: local > imported > wildcard (last-write-wins within tier).
* - Rust: explicit `use` > glob; `pub use` overrides.
* Return value replaces the bucket entirely — no implicit append.
*/
mergeBindings(
existing: readonly BindingRef[],
incoming: readonly BindingRef[],
scope: ScopeId,
): readonly BindingRef[];
}
/** One SCC in the file-level import graph. */
export interface FinalizedScc {
readonly files: readonly string[];
/** True iff this SCC has ≥ 2 files OR a single file that self-imports. */
readonly isCycle: boolean;
}
/**
* Counters reported by `finalize`.
*
* **Counting granularity** — all edge counters are **per-`ParsedImport`**,
* not per-materialized-`ImportEdge`. A single `wildcard` ParsedImport that
* expands to N exports counts as one linked edge in these stats; the
* materialized output (`FinalizeOutput.imports`) will have N edges for
* that input. `dynamic-unresolved` ParsedImports count as linked (they
* pass through with no `linkStatus`), so `linkedEdges` ≠ "has a
* BindingRef" — use the `bindings` map for that.
*
* In other words: `totalEdges === input.parsedImports.length` summed
* across files, and `linkedEdges + unresolvedEdges === totalEdges`.
*/
export interface FinalizeStats {
readonly totalFiles: number;
/** Total `ParsedImport` records seen across all files. */
readonly totalEdges: number;
/**
* `ParsedImport`s whose finalized edge does NOT carry
* `linkStatus: 'unresolved'`. Includes `dynamic-unresolved` pass-throughs.
*/
readonly linkedEdges: number;
/** `ParsedImport`s whose finalized edge carries `linkStatus: 'unresolved'`. */
readonly unresolvedEdges: number;
readonly sccCount: number;
readonly largestSccSize: number;
}
export interface FinalizeOutput {
/** Linked `ImportEdge[]` per module scope, in original input order. */
readonly imports: ReadonlyMap<ScopeId, readonly ImportEdge[]>;
/** Materialized bindings per module scope. */
readonly bindings: ReadonlyMap<ScopeId, ReadonlyMap<string, readonly BindingRef[]>>;
/** SCCs in reverse-topological order (leaves first). */
readonly sccs: readonly FinalizedScc[];
readonly stats: FinalizeStats;
}
// ─── Entry point ───────────────────────────────────────────────────────────
export function finalize(input: FinalizeInput, hooks: FinalizeHooks): FinalizeOutput {
const byFilePath = new Map<string, FinalizeFile>();
for (const f of input.files) byFilePath.set(f.filePath, f);
// ── Phase 0: pre-resolve raw import targets (one syscall-equivalent per
// (file, parsedImport)). Edges with no resolvable target become
// `linkStatus: 'unresolved'` or, for dynamic-unresolved, pass through
// with `targetFile: null`.
const edgeIndex = new Map<string, ImportEdgeDraft[]>(); // filePath → drafts
let totalEdges = 0;
for (const file of input.files) {
const drafts: ImportEdgeDraft[] = [];
for (const parsed of file.parsedImports) {
const draft = makeEdgeDraft(parsed, file, hooks, input.workspaceIndex);
drafts.push(draft);
totalEdges++;
}
edgeIndex.set(file.filePath, drafts);
}
// ── Phase 1: build file-level import graph (only resolvable edges form
// graph edges; unresolvable ones are terminal and contribute no
// fixpoint obligation).
const graph = new Map<string, Set<string>>();
for (const file of input.files) {
graph.set(file.filePath, new Set());
}
for (const [fromFile, drafts] of edgeIndex) {
const edges = graph.get(fromFile);
if (edges === undefined) continue;
for (const d of drafts) {
if (d.targetFile !== null && byFilePath.has(d.targetFile)) {
edges.add(d.targetFile);
}
}
}
// ── Phase 2: Tarjan SCC → reverse-topological list of SCCs.
const sccs = tarjanSccs(graph);
// ── Phase 2.5: precompute the per-file re-export closure (iterative,
// SCC-condensed). Eliminates the recursive crawl that the per-edge
// `tryFinalize` call site used to do; lookups are O(1) afterwards.
// See `buildReexportClosures` for the algorithm.
const reexportClosures = buildReexportClosures(input.files, byFilePath, edgeIndex);
// ── Phase 3: process SCCs in reverse-topological order (leaves first).
// Within each SCC, run a bounded fixpoint that resolves intra-SCC edges.
// Edges leaving the SCC are already resolved (their target SCC is
// already finalized); edges inside the SCC may need multiple passes.
const linkedByScope = new Map<ScopeId, readonly ImportEdge[]>();
let linkedEdges = 0;
for (const scc of sccs) {
const sccFiles = new Set(scc.files);
const capacity = countEdgesWithin(edgeIndex, sccFiles);
// Run the fixpoint up to `capacity` iterations. Each iteration tries to
// resolve every still-unlinked edge in the SCC; stops early if a pass
// makes no progress.
let progressed = true;
let iterations = 0;
while (progressed && iterations < capacity) {
progressed = false;
iterations++;
for (const filePath of scc.files) {
const drafts = edgeIndex.get(filePath);
if (drafts === undefined) continue;
for (const draft of drafts) {
if (draft.finalized !== null) continue;
const finalized = tryFinalize(draft, byFilePath, reexportClosures);
if (finalized !== null) {
draft.finalized = finalized;
progressed = true;
}
}
}
}
// Any drafts still not finalized within this SCC hit the cap → unresolved.
for (const filePath of scc.files) {
const drafts = edgeIndex.get(filePath);
if (drafts === undefined) continue;
for (const draft of drafts) {
if (draft.finalized !== null) continue;
draft.finalized = {
...draft.base,
linkStatus: 'unresolved' as const,
};
}
}
}
// ── Phase 4: collect finalized `ImportEdge[]` per module scope, preserving
// input order within each file, and wildcard-expand where applicable.
for (const file of input.files) {
const drafts = edgeIndex.get(file.filePath);
if (drafts === undefined) continue;
const finalized: ImportEdge[] = [];
for (const d of drafts) {
const edge = d.finalized;
if (edge === null) {
throw new Error(`Invariant violated: import edge was not finalized for ${file.filePath}`);
}
if (d.source.kind === 'wildcard' && edge.linkStatus !== 'unresolved') {
// Produce one `wildcard-expanded` ImportEdge per exported name.
const expanded = expandWildcard(edge, byFilePath, hooks, input.workspaceIndex);
for (const e of expanded) finalized.push(e);
} else {
finalized.push(edge);
}
if (edge.linkStatus !== 'unresolved') linkedEdges++;
}
linkedByScope.set(file.moduleScope, Object.freeze(finalized));
}
// ── Phase 5: materialize module-scope bindings (local + imports + wildcards),
// delegating precedence to `provider.mergeBindings`.
const bindingsByScope = materializeBindings(input.files, linkedByScope, hooks);
// ── Stats.
const sccCount = sccs.length;
let largestSccSize = 0;
for (const scc of sccs) {
if (scc.files.length > largestSccSize) largestSccSize = scc.files.length;
}
const stats: FinalizeStats = {
totalFiles: input.files.length,
totalEdges,
linkedEdges,
unresolvedEdges: totalEdges - linkedEdges,
sccCount,
largestSccSize,
};
return Object.freeze({
imports: linkedByScope,
bindings: bindingsByScope,
sccs,
stats,
});
}
// ─── Internal: edge drafting (phase 0) ──────────────────────────────────────
interface ImportEdgeDraft {
readonly source: ParsedImport;
readonly fromFile: string;
readonly fromScope: ScopeId;
readonly targetFile: string | null;
readonly base: ImportEdge;
finalized: ImportEdge | null;
}
function makeEdgeDraft(
parsed: ParsedImport,
file: FinalizeFile,
hooks: FinalizeHooks,
workspace: WorkspaceIndex,
): ImportEdgeDraft {
// Dynamic-unresolved passes through — no `BindingRef`, no target file.
if (parsed.kind === 'dynamic-unresolved') {
const base: ImportEdge = {
localName: parsed.localName,
targetFile: null,
targetExportedName: '',
kind: 'dynamic-unresolved',
};
return {
source: parsed,
fromFile: file.filePath,
fromScope: file.moduleScope,
targetFile: null,
base,
finalized: base, // already fully finalized
};
}
const targetFile = hooks.resolveImportTarget(parsed.targetRaw ?? '', file.filePath, workspace);
// Edge is unresolvable at the file level — mark unresolved now.
if (targetFile === null) {
const base: ImportEdge = {
localName: extractLocalName(parsed),
targetFile: null,
targetExportedName: extractExportedName(parsed),
kind: edgeKindFor(parsed),
linkStatus: 'unresolved',
};
return {
source: parsed,
fromFile: file.filePath,
fromScope: file.moduleScope,
targetFile: null,
base,
finalized: base,
};
}
// Resolvable at the file level; intra-SCC fixpoint may still fail to fill
// in `targetDefId` (e.g., symbol not exported from target). Side-effect
// and resolved-dynamic imports are terminal at the file level — no
// `targetDefId` needed since they materialize no `BindingRef`. Pre-
// finalize them here so the fixpoint loop skips them entirely.
const base: ImportEdge = {
localName: extractLocalName(parsed),
targetFile,
targetExportedName: extractExportedName(parsed),
kind: edgeKindFor(parsed),
};
const isFileLevelTerminal = parsed.kind === 'side-effect' || parsed.kind === 'dynamic-resolved';
return {
source: parsed,
fromFile: file.filePath,
fromScope: file.moduleScope,
targetFile,
base,
finalized: isFileLevelTerminal ? base : null,
};
}
function edgeKindFor(parsed: ParsedImport): ImportEdge['kind'] {
if (parsed.kind === 'wildcard') return 'wildcard-expanded';
return parsed.kind;
}
function extractLocalName(parsed: ParsedImport): string {
switch (parsed.kind) {
case 'wildcard':
case 'side-effect':
case 'dynamic-resolved':
return '';
default:
return parsed.localName;
}
}
function extractExportedName(parsed: ParsedImport): string {
switch (parsed.kind) {
case 'named':
case 'alias':
case 'namespace':
case 'reexport':
return parsed.importedName;
case 'wildcard':
case 'dynamic-unresolved':
case 'dynamic-resolved':
case 'side-effect':
return '';
}
}
// ─── Internal: per-edge finalization (phase 3) ─────────────────────────────
function tryFinalize(
draft: ImportEdgeDraft,
byFilePath: Map<string, FinalizeFile>,
reexportClosures: ReadonlyMap<string, FileReexportClosure>,
): ImportEdge | null {
const targetFile = draft.targetFile;
if (targetFile === null) return draft.base; // already terminal
const targetModule = byFilePath.get(targetFile);
if (targetModule === undefined) return draft.base; // external target — leave as-is
// Wildcards finalize at the file level; their per-name expansion happens
// in phase 4. At this stage we just record the target module scope.
if (draft.source.kind === 'wildcard') {
return {
...draft.base,
targetModuleScope: targetModule.moduleScope,
};
}
// Namespace imports alias the target *module*; they don't name a
// specific export. Link the module scope unconditionally. If the target
// also exposes a def whose simple name matches `importedName` (some
// languages emit a synthetic module-def), pick it up as the `targetDefId`
// so consumers can reach the module as a symbol — but its absence is not
// a failure.
if (draft.source.kind === 'namespace') {
const moduleDef = findExportByName(targetModule.localDefs, extractExportedName(draft.source));
return {
...draft.base,
targetModuleScope: targetModule.moduleScope,
...(moduleDef !== undefined ? { targetDefId: moduleDef.nodeId } : {}),
};
}
// named / alias / reexport: look up the imported name in the target's
// local defs. Multi-hop re-export chains settle iteratively — each hop
// resolves once its prior hop is finalized.
const importedName = extractExportedName(draft.source);
const exported = findExportByName(targetModule.localDefs, importedName);
if (exported !== undefined) {
const transitiveVia =
draft.source.kind === 'reexport' ? Object.freeze([targetFile]) : undefined;
return {
...draft.base,
targetModuleScope: targetModule.moduleScope,
targetDefId: exported.nodeId,
...(transitiveVia !== undefined ? { transitiveVia } : {}),
};
}
// Multi-hop re-export follow. Barrel modules like
// // models.ts
// export { User } from './base';
// emit no local def for `User`; the name surfaces only via their own
// `reexport` edge. The per-file re-export closure built in phase 2.5
// already encodes every name reachable through that file's named and
// wildcard re-exports — including transitively through cyclic SCCs —
// so the lookup is O(1) and never recurses.
const followed = lookupReexportedName(reexportClosures, targetFile, importedName);
if (followed === null) {
// Target resolvable but the name isn't exported — keep trying in case a
// re-export inside the target's SCC surfaces it in a later iteration.
return null;
}
const viaFiles = [targetFile, ...followed.via];
const transitiveVia =
draft.source.kind === 'reexport' || viaFiles.length > 1 ? Object.freeze(viaFiles) : undefined;
return {
...draft.base,
targetModuleScope: targetModule.moduleScope,
targetDefId: followed.def.nodeId,
...(transitiveVia !== undefined ? { transitiveVia } : {}),
};
}
// ─── Internal: re-export closure (phase 2.5) ───────────────────────────────
/**
* Per-file map of `name → terminal def + via path` — i.e. every name
* importable from this file via its named/wildcard re-export chain
* (excluding the file's own `localDefs`, which the caller checks first
* via `findExportByName`). `via` is the ordered list of intermediate
* files traversed to reach the def.
*
* Built once per finalize pass. Lookups are O(1).
*/
type ReexportClosureEntry = { readonly def: SymbolDefinition; readonly via: readonly string[] };
type FileReexportClosure = ReadonlyMap<string, ReexportClosureEntry>;
/**
* Build per-file re-export closures.
*
* **Algorithm.** Iterative SCC-condensed reverse-topological propagation,
* structurally identical to how `finalize` itself processes the file-
* level import graph. Replaces the legacy recursive
* `followReexportChain` crawl with a bounded, stack-safe pass:
*
* 1. **Sub-graph.** Build a directed graph whose edges are
* `reexport` and `wildcard` drafts only (regular imports do not
* contribute to the export surface, and `namespace`/
* `reexport-namespace` are terminal — their target def lives in
* `localDefs`).
* 2. **SCC condensation.** Run the same iterative `tarjanSccs` over
* the sub-graph. Output is in reverse-topological order (leaves
* first), so when we process an SCC every out-of-SCC neighbor
* already has its closure populated.
* 3. **Per-SCC propagation.**
* * Acyclic singleton: one pass — read neighbors' (already
* fully populated) closures.
* * Cyclic SCC (cycle ≥ 2 files, or self-loop): bounded
* fixpoint inside the SCC, capped at `|SCC| + 1` iterations
* (each iteration propagates names one hop further around
* the cycle; first-wins precedence keeps the map monotone
* so the fixpoint converges in at most |SCC| hops).
*
* **Precedence semantics — preserved from the recursive crawl.**
* * Named re-exports take precedence over wildcards.
* * Within each kind, declaration order wins (first match for a
* given exported name is kept; later drafts skip).
*
* **Complexity.**
* * Pre-pass: O(V + E_re) for SCC, plus O(|SCC| × Σ drafts) per cyclic
* SCC. For tree-shaped barrel graphs (the common case) it
* collapses to O(E_re) total.
* * Per-edge lookup at finalize time: O(1).
* * `transitiveVia` preserves the exact file path chain for diagnostics
* and graph provenance. Building those arrays copies the inherited path,
* which is O(depth²) in a pathological single-name barrel chain; practical
* TypeScript barrel chains are shallow enough that we keep exact paths
* instead of capping or summarizing them.
* * Pathological deep chains that previously needed
* `MAX_REEXPORT_DEPTH=100` to bound stack growth now resolve
* in full and are bounded only by available memory — the
* iterative formulation has no call-stack ceiling.
*/
function buildReexportClosures(
files: readonly FinalizeFile[],
byFilePath: ReadonlyMap<string, FinalizeFile>,
edgeIndex: ReadonlyMap<string, ImportEdgeDraft[]>,
): ReadonlyMap<string, FileReexportClosure> {
const closures = new Map<string, Map<string, ReexportClosureEntry>>();
for (const file of files) closures.set(file.filePath, new Map());
// ── Step 1: build the re-export sub-graph (only resolvable
// reexport/wildcard targets contribute edges).
const subGraph = new Map<string, Set<string>>();
for (const file of files) {
const targets = new Set<string>();
const drafts = edgeIndex.get(file.filePath);
if (drafts !== undefined) {
for (const d of drafts) {
if (d.source.kind !== 'reexport' && d.source.kind !== 'wildcard') continue;
if (d.targetFile === null) continue;
if (!byFilePath.has(d.targetFile)) continue;
targets.add(d.targetFile);
}
}
subGraph.set(file.filePath, targets);
}
// ── Step 2: SCC over the sub-graph. Reuses the same iterative Tarjan
// implementation that drives the file-level finalize loop, so any
// call-stack-safety guarantees there transfer here unchanged.
const subSccs = tarjanSccs(subGraph);
// ── Step 3: process SCCs in reverse-topological order. Acyclic
// singletons settle in one pass; cyclic SCCs run a bounded fixpoint.
for (const scc of subSccs) {
if (!scc.isCycle) {
const filePath = scc.files[0];
if (filePath !== undefined) {
populateFileClosure(filePath, byFilePath, edgeIndex, closures);
}
continue;
}
// Cap = |SCC| + 1. With first-wins precedence each name needs at
// most |SCC| iterations to propagate fully around the cycle; the
// extra iteration confirms no progress and breaks the loop.
const cap = scc.files.length + 1;
let progressed = true;
let iter = 0;
while (progressed && iter < cap) {
progressed = false;
iter++;
for (const filePath of scc.files) {
if (populateFileClosure(filePath, byFilePath, edgeIndex, closures)) {
progressed = true;
}
}
}
}
return closures;
}
/**
* Populate one file's re-export closure for one pass. Returns `true`
* iff the closure grew (signalling fixpoint progress to the caller).
*
* Walks the file's drafts in declaration order, named re-exports first
* (precedence), then wildcards. For each draft, attempts:
* 1. **Direct hit** — name exists in the target file's `localDefs`.
* 2. **Inherited** — name exists in the target file's already-populated
* closure (which encodes the target's own re-export chain).
*
* `closures.get(targetFile)` may itself still be empty for in-SCC
* targets on the first iteration; the outer fixpoint loop handles
* that by re-invoking this function.
*/
function populateFileClosure(
filePath: string,
byFilePath: ReadonlyMap<string, FinalizeFile>,
edgeIndex: ReadonlyMap<string, ImportEdgeDraft[]>,
closures: Map<string, Map<string, ReexportClosureEntry>>,
): boolean {
const myClosure = closures.get(filePath);
if (myClosure === undefined) return false;
const before = myClosure.size;
const drafts = edgeIndex.get(filePath);
if (drafts === undefined) return false;
// Named re-exports — precedence over wildcards, declaration order
// first-wins for duplicates of the same exported name.
for (const draft of drafts) {
if (draft.source.kind !== 'reexport') continue;
const targetFile = draft.targetFile;
if (targetFile === null) continue;
const targetModule = byFilePath.get(targetFile);
if (targetModule === undefined) continue;
const localName = draft.source.localName;
if (myClosure.has(localName)) continue;
const importedName = draft.source.importedName;
const direct = findExportByName(targetModule.localDefs, importedName);
if (direct !== undefined) {
myClosure.set(localName, { def: direct, via: Object.freeze([targetFile]) });
continue;
}
const inherited = closures.get(targetFile)?.get(importedName);
if (inherited !== undefined) {
myClosure.set(localName, {
def: inherited.def,
via: Object.freeze([targetFile, ...inherited.via]),
});
}
// Else: target's closure is still empty (in-SCC, awaiting next
// iteration). Outer loop will revisit.
}
// Wildcard re-exports — fan out the target's own surface (localDefs
// + transitive closure). `myClosure.has(name)` checks below preserve
// the named-precedence and first-wins semantics from above.
for (const draft of drafts) {
if (draft.source.kind !== 'wildcard') continue;
const targetFile = draft.targetFile;
if (targetFile === null) continue;
const targetModule = byFilePath.get(targetFile);
if (targetModule === undefined) continue;
for (const def of targetModule.localDefs) {
const name = deriveSimpleName(def);
if (name === null || myClosure.has(name)) continue;
myClosure.set(name, { def, via: Object.freeze([targetFile]) });
}
const targetClosure = closures.get(targetFile);
if (targetClosure !== undefined) {
for (const [name, entry] of targetClosure) {
if (myClosure.has(name)) continue;
myClosure.set(name, {
def: entry.def,
via: Object.freeze([targetFile, ...entry.via]),
});
}
}
}
return myClosure.size > before;
}
/**
* O(1) lookup into a precomputed re-export closure. Replaces the legacy
* recursive `followReexportChain` traversal with a single map indexing.
*/
function lookupReexportedName(
closures: ReadonlyMap<string, FileReexportClosure>,
filePath: string,
name: string,
): { def: SymbolDefinition; via: readonly string[] } | null {
const closure = closures.get(filePath);
if (closure === undefined) return null;
const entry = closure.get(name);
if (entry === undefined) return null;
return { def: entry.def, via: entry.via };
}
/**
* The "simple" (unqualified) name of a def, for import-name matching.
*
* Canonical source: `def.qualifiedName` — the tail after the last `.` (or
* the whole string if no dot). Defs without a qualifiedName can't be
* resolved by name here and return `null`; callers treat that as "name
* not exported" and either retry in a later fixpoint iteration or mark
* the edge unresolved.
*/
function deriveSimpleName(def: SymbolDefinition): string | null {
const q = def.qualifiedName;
if (q === undefined || q.length === 0) return null;
const dot = q.lastIndexOf('.');
return dot === -1 ? q : q.slice(dot + 1);
}
function findExportByName(
defs: readonly SymbolDefinition[],
name: string,
): SymbolDefinition | undefined {
for (const d of defs) {
if (deriveSimpleName(d) === name) return d;
}
return undefined;
}
function countEdgesWithin(edgeIndex: Map<string, ImportEdgeDraft[]>, files: Set<string>): number {
let n = 0;
for (const filePath of files) {
const drafts = edgeIndex.get(filePath);
if (drafts === undefined) continue;
for (const d of drafts) {
if (d.targetFile !== null && files.has(d.targetFile)) n++;
}
}
// Guarantee at least one pass even for a trivial SCC (ensures deterministic
// fixpoint termination even when a single-file SCC has zero intra-SCC edges
// but still needs one settle pass).
return Math.max(n, 1);
}
// ─── Internal: wildcard expansion (phase 4) ────────────────────────────────
function expandWildcard(
edge: ImportEdge,
byFilePath: Map<string, FinalizeFile>,
hooks: FinalizeHooks,
workspace: WorkspaceIndex,
): readonly ImportEdge[] {
if (edge.targetModuleScope === undefined || edge.targetFile === null) {
return [edge]; // unresolvable wildcard survives as a single unlinked edge
}
const target = byFilePath.get(edge.targetFile);
if (target === undefined) return [edge];
const names = hooks.expandsWildcardTo(edge.targetModuleScope, workspace);
if (names.length === 0) return [];
const expanded: ImportEdge[] = [];
for (const name of names) {
const def = findExportByName(target.localDefs, name);
if (def === undefined) continue;
expanded.push({
localName: name,
targetFile: edge.targetFile,
targetExportedName: name,
kind: 'wildcard-expanded',
targetModuleScope: edge.targetModuleScope,
targetDefId: def.nodeId,
});
}
return expanded;
}
// ─── Internal: bindings materialization (phase 5) ───────────────────────────
function materializeBindings(
files: readonly FinalizeFile[],
linkedByScope: ReadonlyMap<ScopeId, readonly ImportEdge[]>,
hooks: FinalizeHooks,
): ReadonlyMap<ScopeId, ReadonlyMap<string, readonly BindingRef[]>> {
const out = new Map<ScopeId, ReadonlyMap<string, readonly BindingRef[]>>();
// Build a `nodeId → SymbolDefinition` index once across all files
// (O(N_files × D_defs)) so the per-edge lookup below is O(1) instead
// of a full linear scan. At realistic TypeScript monorepo scale
// (~5k files × ~50 defs × ~100k linked import edges) this is the
// difference between ~25 s and a few ms inside finalize. The map
// is local to this pass — no cross-pass state leaks.
const defById = new Map<string, SymbolDefinition>();
for (const f of files) {
for (const d of f.localDefs) defById.set(d.nodeId, d);
}
for (const file of files) {
const scopeBindings = new Map<string, readonly BindingRef[]>();
// Start with local defs as `origin: 'local'` bindings.
for (const def of file.localDefs) {
const name = deriveSimpleName(def);
if (name === null) continue;
const incoming: BindingRef[] = [{ def, origin: 'local' }];
const existing = scopeBindings.get(name) ?? [];
scopeBindings.set(name, hooks.mergeBindings(existing, incoming, file.moduleScope));
}
// Layer in finalized imports.
const imports = linkedByScope.get(file.moduleScope) ?? [];
for (const edge of imports) {
if (edge.targetDefId === undefined || edge.linkStatus === 'unresolved') continue;
const def = defById.get(edge.targetDefId);
if (def === undefined) continue;
const origin: BindingRef['origin'] =
edge.kind === 'namespace'
? 'namespace'
: edge.kind === 'wildcard-expanded'
? 'wildcard'
: edge.kind === 'reexport'
? 'reexport'
: 'import';
const fallback = deriveSimpleName(def);
const name = edge.localName.length > 0 ? edge.localName : fallback;
if (name === null) continue;
const incoming: BindingRef[] = [{ def, origin, via: edge }];
const existing = scopeBindings.get(name) ?? [];
scopeBindings.set(name, hooks.mergeBindings(existing, incoming, file.moduleScope));
}
// Freeze nested buckets for immutability.
const frozen = new Map<string, readonly BindingRef[]>();
for (const [name, refs] of scopeBindings) {
frozen.set(name, Object.freeze(refs.slice()));
}
out.set(file.moduleScope, frozen);
}
return out;
}
// ─── Internal: Tarjan SCC ──────────────────────────────────────────────────
/**
* Iterative Tarjan SCC. Returns SCCs in **reverse-topological** order
* (leaves first — a property Tarjan gives for free, and the order
* `finalize` wants so leaves are fully resolved before their dependents).
*/
function tarjanSccs(graph: ReadonlyMap<string, ReadonlySet<string>>): FinalizedScc[] {
const index = new Map<string, number>();
const lowlink = new Map<string, number>();
const onStack = new Set<string>();
const stack: string[] = [];
const sccs: FinalizedScc[] = [];
let idx = 0;
// Iterative DFS to avoid stack overflow on deep import chains.
const allNodes = Array.from(graph.keys()).sort(); // deterministic order
const iterStack: Array<{ node: string; children: Iterator<string>; entered: boolean }> = [];
for (const root of allNodes) {
if (index.has(root)) continue;
iterStack.push({
node: root,
children: (graph.get(root) ?? new Set<string>()).values(),
entered: false,
});
while (iterStack.length > 0) {
const frame = iterStack[iterStack.length - 1];
if (frame === undefined) break;
if (!frame.entered) {
frame.entered = true;
index.set(frame.node, idx);
lowlink.set(frame.node, idx);
idx++;
stack.push(frame.node);
onStack.add(frame.node);
}
const nextChild = frame.children.next();
if (nextChild.done) {
// Post-visit: compute SCC membership if frame.node is a root.
if (lowlink.get(frame.node) === index.get(frame.node)) {
const scc: string[] = [];
let selfInCycle = false;
while (true) {
const w = stack.pop();
if (w === undefined) {
throw new Error(`Invariant violated: Tarjan stack exhausted at ${frame.node}`);
}
onStack.delete(w);
scc.push(w);
// A single-file self-loop counts as a cycle.
if (w === frame.node) {
selfInCycle = (graph.get(w) ?? new Set()).has(w);
break;
}
}
const isCycle = scc.length > 1 || selfInCycle;
sccs.push({ files: Object.freeze(scc), isCycle });
}
iterStack.pop();
// Propagate lowlink to parent.
if (iterStack.length > 0) {
const parent = iterStack[iterStack.length - 1];
if (parent !== undefined) {
lowlink.set(
parent.node,
Math.min(
requiredNumber(lowlink, parent.node, 'lowlink'),
requiredNumber(lowlink, frame.node, 'lowlink'),
),
);
}
}
continue;
}
const child = nextChild.value;
if (!index.has(child)) {
iterStack.push({
node: child,
children: (graph.get(child) ?? new Set<string>()).values(),
entered: false,
});
} else if (onStack.has(child)) {
lowlink.set(
frame.node,
Math.min(
requiredNumber(lowlink, frame.node, 'lowlink'),
requiredNumber(index, child, 'index'),
),
);
}
}
}
return sccs;
}
function requiredNumber(map: ReadonlyMap<string, number>, key: string, label: string): number {
const value = map.get(key);
if (value === undefined) {
throw new Error(`Invariant violated: missing Tarjan ${label} for ${key}`);
}
return value;
}
@@ -1,49 +0,0 @@
/**
* `LanguageClassification` — RFC §6.1 Ring 3 / Ring 4 governance.
*
* Classifies each `SupportedLanguages` member for the rollout. Ring 4 (DAG
* retirement) is gated on *all production languages* being registry-primary
* and stable for one release cycle; `experimental` and `quarantined`
* languages do not block.
*
* Initial classification (locked in Ring 1 #910):
* - production: javascript, typescript, python, java, c, cpp, csharp, go,
* ruby, rust, php, kotlin, swift, dart
* - experimental: vue (embedded-language / SFC complexity),
* cobol (regex-provider path)
* - quarantined: (none)
*/
import { SupportedLanguages } from '../languages.js';
export type LanguageClassification = 'production' | 'experimental' | 'quarantined';
/**
* The canonical classification for each supported language. Governance
* changes (promote `experimental` → `production`, quarantine a language, …)
* update this map in a dedicated PR.
*/
export const LanguageClassifications: Readonly<Record<SupportedLanguages, LanguageClassification>> =
{
[SupportedLanguages.JavaScript]: 'production',
[SupportedLanguages.TypeScript]: 'production',
[SupportedLanguages.Python]: 'production',
[SupportedLanguages.Java]: 'production',
[SupportedLanguages.C]: 'production',
[SupportedLanguages.CPlusPlus]: 'production',
[SupportedLanguages.CSharp]: 'production',
[SupportedLanguages.Go]: 'production',
[SupportedLanguages.Ruby]: 'production',
[SupportedLanguages.Rust]: 'production',
[SupportedLanguages.PHP]: 'production',
[SupportedLanguages.Kotlin]: 'production',
[SupportedLanguages.Swift]: 'production',
[SupportedLanguages.Dart]: 'production',
[SupportedLanguages.Vue]: 'experimental',
[SupportedLanguages.Cobol]: 'experimental',
};
/** Convenience predicate: is this language gating Ring 4 retirement? */
export function isProductionLanguage(lang: SupportedLanguages): boolean {
return LanguageClassifications[lang] === 'production';
}
@@ -1,145 +0,0 @@
/**
* `MethodDispatchIndex` — materialized view of class hierarchies keyed by
* `DefId` (RFC §3.1; Ring 2 SHARED #914).
*
* Two O(1)-access maps used by `Registry.lookupMethod` and interface-
* dispatch callers:
*
* - `mroByOwnerDefId` : owner class → full MRO ancestor chain
* (excludes the owner itself, in per-language
* strategy order).
* - `implsByInterfaceDefId` : interface/trait → classes that implement it.
*
* **Not an MRO implementation.** The build function is a pure aggregator: it
* asks the caller (via `computeMro` and `implementsOf` callbacks) for the
* per-language answers and materializes the two-way index. MRO strategies
* live where they already do today (`model/resolve.ts § c3Linearize`,
* `languages/ruby.ts § selectDispatch`, etc.) — this index does not
* reimplement them.
*
* Why callbacks and not a shared strategy registry: the five strategies
* (Python C3, Ruby kind-aware, Java/Kotlin linear, Rust qualified-syntax,
* COBOL none) already exist in the CLI package and depend on the CLI's
* `HeritageMap` + `SemanticModel`. Pulling them into `gitnexus-shared` would
* require migrating both — out of scope for #914. Callbacks let the shared
* build stay pure while honoring existing strategies verbatim.
*
* Consumed by: #917 (`Registry.lookupMethod` MRO fast path, interface
* dispatch resolver).
*/
import type { DefId } from './types.js';
// ─── Public contracts ───────────────────────────────────────────────────────
export interface MethodDispatchIndex {
/**
* Full MRO ancestor chain per owner class (excludes the owner itself).
* Order reflects the per-language strategy used by `computeMro`.
*/
readonly mroByOwnerDefId: ReadonlyMap<DefId, readonly DefId[]>;
/** Interfaces / traits → classes that implement them. */
readonly implsByInterfaceDefId: ReadonlyMap<DefId, readonly DefId[]>;
/** `mroByOwnerDefId.get`, with an empty frozen array on miss. */
mroFor(ownerDefId: DefId): readonly DefId[];
/** `implsByInterfaceDefId.get`, with an empty frozen array on miss. */
implementorsOf(interfaceDefId: DefId): readonly DefId[];
}
export interface MethodDispatchInput {
/**
* Owner defs to index (classes, structs, traits, interfaces — any kind
* that can appear on the owner side of a method-dispatch graph).
*/
readonly owners: readonly DefId[];
/**
* Return the full MRO ancestor chain for `ownerDefId`, **excluding the
* owner itself**, in the order dictated by the owner's language-specific
* MRO strategy.
*
* Contract:
* - Pure (no side effects).
* - Deterministic per input.
* - `undefined` not allowed — return `[]` when the owner has no parents.
*/
readonly computeMro: (ownerDefId: DefId) => readonly DefId[];
/**
* Return the set of interface/trait defs that `ownerDefId` implements.
* Transitive inclusion (e.g., `implements` on a parent class) is the
* caller's choice — the build function simply inverts whatever is
* returned.
*
* Repeated IDs in the output are deduplicated automatically.
*
* **Call-count contract.** `implementsOf` is invoked **once per
* occurrence** of an owner in `input.owners`, not once per unique
* owner. Duplicate owners therefore re-invoke it; dedup happens at
* the bucket layer (after the callback returns). Callers with
* expensive `implementsOf` implementations should pass a deduplicated
* `owners` list. `computeMro`, by contrast, is memoized by the first-
* write-wins policy and fires at most once per unique owner.
*/
readonly implementsOf: (ownerDefId: DefId) => readonly DefId[];
}
// ─── Builder ────────────────────────────────────────────────────────────────
export function buildMethodDispatchIndex(input: MethodDispatchInput): MethodDispatchIndex {
const mroByOwnerDefId = new Map<DefId, readonly DefId[]>();
const implsBuilding = new Map<DefId, DefId[]>();
const implsSeen = new Map<DefId, Set<DefId>>();
for (const ownerId of input.owners) {
// First-write-wins on duplicate owner ids: a stable policy consistent
// with sibling indexes (#913 DefIndex / ModuleScopeIndex).
if (!mroByOwnerDefId.has(ownerId)) {
const chain = input.computeMro(ownerId);
mroByOwnerDefId.set(ownerId, Object.freeze(chain.slice()));
}
for (const ifaceId of input.implementsOf(ownerId)) {
let seen = implsSeen.get(ifaceId);
if (seen === undefined) {
seen = new Set<DefId>();
implsSeen.set(ifaceId, seen);
}
if (seen.has(ownerId)) continue;
seen.add(ownerId);
let bucket = implsBuilding.get(ifaceId);
if (bucket === undefined) {
bucket = [];
implsBuilding.set(ifaceId, bucket);
}
bucket.push(ownerId);
}
}
const implsByInterfaceDefId = new Map<DefId, readonly DefId[]>();
for (const [ifaceId, owners] of implsBuilding) {
implsByInterfaceDefId.set(ifaceId, Object.freeze(owners.slice()));
}
return wrapIndex(mroByOwnerDefId, implsByInterfaceDefId);
}
// ─── Internal ───────────────────────────────────────────────────────────────
const EMPTY: readonly DefId[] = Object.freeze([]);
function wrapIndex(
mroByOwnerDefId: Map<DefId, readonly DefId[]>,
implsByInterfaceDefId: Map<DefId, readonly DefId[]>,
): MethodDispatchIndex {
return {
mroByOwnerDefId,
implsByInterfaceDefId,
mroFor(ownerDefId: DefId): readonly DefId[] {
return mroByOwnerDefId.get(ownerDefId) ?? EMPTY;
},
implementorsOf(interfaceDefId: DefId): readonly DefId[] {
return implsByInterfaceDefId.get(interfaceDefId) ?? EMPTY;
},
};
}
@@ -1,73 +0,0 @@
/**
* `ModuleScopeIndex` — O(1) `filePath → moduleScopeId` lookup.
*
* Every file parsed produces exactly one `Module` scope at its root. The
* finalize algorithm needs to resolve `ImportEdge.targetFile` to a concrete
* module scope id in constant time during the link pass; this index is that
* mapping.
*
* Part of RFC #909 Ring 2 SHARED — #913.
*
* Consumed by: #915 (SCC finalize link pass), #923 (shadow harness when
* resolving callsite file → enclosing module).
*/
import type { ScopeId } from './types.js';
export interface ModuleScopeIndex {
readonly byFilePath: ReadonlyMap<string, ScopeId>;
readonly size: number;
get(filePath: string): ScopeId | undefined;
has(filePath: string): boolean;
}
export interface ModuleScopeEntry {
readonly filePath: string;
readonly moduleScopeId: ScopeId;
}
/**
* Build a `ModuleScopeIndex` from a flat list of `{ filePath, moduleScopeId }`
* pairs.
*
* **Collision policy: first-write-wins.** A file should appear exactly once
* in a single ingestion run; collisions indicate the same file was parsed
* twice or a `filePath` normalization bug upstream. Dropping the later
* entry preserves the first-stable id the rest of the pipeline may already
* have registered against.
*
* **Caller contract: filePath keys must be pre-normalized.** This index
* keys on the raw `filePath` string and does NOT canonicalize separators,
* case, or trailing slashes. Callers upstream of this function must agree
* on a canonical form (typically repo-root-relative, POSIX separators,
* no trailing slash) before constructing entries — otherwise `C:\foo\bar.ts`,
* `C:/foo/bar.ts`, and `foo/bar.ts` will all hash to distinct buckets and
* `get()` will miss.
*
* Pure function — safe to call repeatedly; no side effects.
*/
export function buildModuleScopeIndex(entries: readonly ModuleScopeEntry[]): ModuleScopeIndex {
const byFilePath = new Map<string, ScopeId>();
for (const { filePath, moduleScopeId } of entries) {
if (byFilePath.has(filePath)) continue; // first-write-wins
byFilePath.set(filePath, moduleScopeId);
}
return wrapIndex(byFilePath);
}
// ─── Internal ───────────────────────────────────────────────────────────────
function wrapIndex(byFilePath: Map<string, ScopeId>): ModuleScopeIndex {
return {
byFilePath,
get size() {
return byFilePath.size;
},
get(filePath: string): ScopeId | undefined {
return byFilePath.get(filePath);
},
has(filePath: string): boolean {
return byFilePath.has(filePath);
},
};
}
@@ -1,30 +0,0 @@
/**
* `ORIGIN_PRIORITY` — RFC Appendix B (authoritative values).
*
* Tie-break ordering applied inside `Registry.lookup` Step 7 when
* `|Δconfidence| < 0.001` between two `Resolution` candidates. Lower number
* = stronger (wins the tie).
*
* Full tie-break order (§4.2 Step 7):
* confidence DESC → scope depth ASC → MRO depth ASC → ORIGIN_PRIORITY ASC
* → DefId.localeCompare
*/
export type OriginForTieBreak =
| 'local'
| 'import'
| 'reexport'
| 'namespace'
| 'wildcard'
| 'global-qualified'
| 'global-name';
export const ORIGIN_PRIORITY: Readonly<Record<OriginForTieBreak, number>> = {
local: 0,
import: 1,
reexport: 2,
namespace: 3,
wildcard: 4,
'global-qualified': 5,
'global-name': 6,
};
@@ -1,77 +0,0 @@
/**
* `ParsedFile` — the per-file artifact produced by `ScopeExtractor`
* (RFC §3.2 Phase 1; Ring 2 PKG #919).
*
* The boundary between Phase 1 (extraction, per-file, parallelizable) and
* Phase 2 (finalize, cross-file). One `ParsedFile` is emitted per source
* file; the finalize orchestrator (#921) collects them into a workspace-
* wide set and feeds them to the shared `finalize` algorithm (#915).
*
* ## Shape
*
* - `scopes` — every `Scope` created for this file, in tree-
* topological order (module first, then children).
* `Scope.bindings` carry **local-only** bindings at
* this stage; finalize merges imports/wildcards on top.
* - `parsedImports` — raw `ParsedImport[]` for this file; finalize
* resolves each to a concrete `ImportEdge`.
* - `localDefs` — defs structurally declared in this file. A
* superset of every `Scope.ownedDefs` union.
* Listed separately so `finalize` can dedup-index
* without re-walking scopes.
* - `referenceSites` — pre-resolution usage facts; populated by the
* resolution phase into `ReferenceIndex`.
*
* ## What `ParsedFile` deliberately does NOT carry
*
* - Linked `ImportEdge`s. Those are finalize output.
* - A `ScopeTree` instance. Callers build one from `scopes` (cheap —
* `buildScopeTree(parsedFile.scopes)`). Keeping the ParsedFile flat
* makes IPC serialization from worker threads straightforward.
* - Merged module-scope bindings. Finalize owns that materialization.
*
* ## Compatibility with `FinalizeFile`
*
* `FinalizeFile` (defined in `./finalize-algorithm.ts`) is a structural
* subset of `ParsedFile` — `filePath`, `moduleScope`, `parsedImports`,
* `localDefs`. A `ParsedFile` is trivially convertible to a `FinalizeFile`
* by picking those four fields, so the finalize orchestrator threads
* ParsedFile through to the shared algorithm without shape-shifting.
*
* ## Source-of-truth invariant
*
* `ParsedFile` is the single semantic model consumed by both the legacy
* DAG (`gitnexus/src/core/ingestion/` outside `scope-resolution/`) and
* the scope-resolution pipeline (`gitnexus/src/core/ingestion/scope-resolution/`).
* Downstream passes MUST NOT build a parallel parse representation; if
* a pass needs AST-level facts that `ParsedFile` doesn't expose, it
* should reuse the orchestrator's `treeCache` rather than re-invoke
* `parser.parse(...)` on its own. See the
* `ScopeResolver` contract (`gitnexus/src/core/ingestion/scope-resolution/contract/scope-resolver.ts`)
* for the full list of invariants downstream consumers rely on.
*/
import type { Scope, ScopeId } from './types.js';
import type { ParsedImport } from './types.js';
import type { SymbolDefinition } from './symbol-definition.js';
import type { ReferenceSite } from './reference-site.js';
export interface ParsedFile {
readonly filePath: string;
/** `Scope.id` of the file's root `Module` scope. */
readonly moduleScope: ScopeId;
/**
* All scopes in this file, typically emitted in tree-topological order.
* Caller reconstructs a `ScopeTree` via `buildScopeTree(scopes)` when
* navigation or invariant re-validation is needed.
*/
readonly scopes: readonly Scope[];
readonly parsedImports: readonly ParsedImport[];
/**
* All defs structurally declared in this file (classes, methods, fields,
* variables). Mirrors the union of `Scope.ownedDefs` across `scopes`,
* pre-flattened for O(N) consumption by finalize.
*/
readonly localDefs: readonly SymbolDefinition[];
readonly referenceSites: readonly ReferenceSite[];
}
@@ -1,166 +0,0 @@
/**
* `PositionIndex` — O(log N_file) scope-at-position lookup
* (RFC §3.1; Ring 2 SHARED #912).
*
* Per-file sorted array of `(range, scopeId)` entries, sorted by start
* position ASC (`startLine`, then `startCol`). `atPosition(filePath, line,
* col)` binary-searches for the last entry whose start ≤ (line, col), then
* scans backward through the sorted prefix and returns the first entry
* whose range contains the query position.
*
* **Why this works.** `ScopeTree`'s invariants (parent strictly contains
* child; siblings don't overlap) guarantee that the scopes containing a
* given point form an **ancestor chain**. When scanning backward through
* entries sorted by start position ASC, the first scope we find that
* contains the query is the innermost one — any deeper-starting scope
* that also contained the query would appear *later* in the sorted array,
* but we're only scanning entries with start ≤ query, so anything later
* necessarily starts after the query and can't contain it.
*
* Expected complexity: `O(log N_file + D)` where `D` is the lexical depth
* at the query position (typically ≤ 10). Worst-case degrades to `O(N_file)`
* only under pathological inputs (many scopes starting at the same line).
*
* **Line/column conventions.** Matches `Range` in `types.ts`: lines are
* 1-based, columns are 0-based. Ranges are **inclusive on both ends** —
* a scope whose `endLine:endCol` equals the query position still contains
* it. That matches how tree-sitter captures bodies (closing brace
* included) and how closed PR #902's `enclosingFunctions` behaved.
*/
import type { Range, Scope, ScopeId } from './types.js';
export interface PositionIndex {
/** Total scope entries indexed across all files. */
readonly size: number;
/**
* Innermost scope containing `(line, col)` in `filePath`, or `undefined`
* when nothing contains it (position before file start, after file end,
* or filePath not indexed).
*
* **Touching-boundary semantics.** Ranges are inclusive on both ends.
* When two sibling scopes share a boundary point — e.g.
* `[5:0, 10:0]` and `[10:0, 15:0]`, which is legal under `ScopeTree`'s
* non-overlap invariant — a query at the shared point `(10, 0)` is
* contained by **both**. The innermost-wins tie-break rule applies as
* usual: since neither is nested inside the other, the one that
* **starts latest** wins, i.e. the **right** sibling. The mechanism
* is the backward scan through the start-position-sorted array (see
* `findLastStartLteIndex` below) — both siblings land before the
* upper-bound cursor, and the right sibling is scanned first. Queries at non-boundary positions between them naturally
* fall to the unique containing scope.
*/
atPosition(filePath: string, line: number, col: number): ScopeId | undefined;
}
/**
* Build a `PositionIndex` from a flat list of `Scope` records.
*
* Duplicate `id`s are tolerated and deduplicated — the caller's
* `ScopeTree.buildScopeTree` is the authoritative validator of scope
* identity, and the position index does not need to re-check that
* invariant.
*/
export function buildPositionIndex(scopes: readonly Scope[]): PositionIndex {
const entriesByFile = new Map<string, Entry[]>();
const seen = new Set<ScopeId>();
for (const scope of scopes) {
if (seen.has(scope.id)) continue;
seen.add(scope.id);
let bucket = entriesByFile.get(scope.filePath);
if (bucket === undefined) {
bucket = [];
entriesByFile.set(scope.filePath, bucket);
}
bucket.push({ id: scope.id, range: scope.range });
}
for (const bucket of entriesByFile.values()) {
bucket.sort(compareEntry);
}
return wrapIndex(entriesByFile, seen.size);
}
// ─── Internals ──────────────────────────────────────────────────────────────
interface Entry {
readonly id: ScopeId;
readonly range: Range;
}
/**
* Sort by start position ASC, breaking ties by end position DESC so that
* larger (outer) scopes appear before their smaller (inner) co-starting
* siblings in the array. Makes the backward-scan contract crisp: the
* first containing hit from the end of the scanned prefix is the
* innermost scope.
*/
function compareEntry(a: Entry, b: Entry): number {
if (a.range.startLine !== b.range.startLine) return a.range.startLine - b.range.startLine;
if (a.range.startCol !== b.range.startCol) return a.range.startCol - b.range.startCol;
if (a.range.endLine !== b.range.endLine) return b.range.endLine - a.range.endLine;
return b.range.endCol - a.range.endCol;
}
/** Whether `(line, col)` is at or after `range`'s start. */
function startIsAtOrBefore(range: Range, line: number, col: number): boolean {
if (range.startLine < line) return true;
if (range.startLine > line) return false;
return range.startCol <= col;
}
/** Whether `(line, col)` is at or before `range`'s end (inclusive). */
function endIsAtOrAfter(range: Range, line: number, col: number): boolean {
if (range.endLine > line) return true;
if (range.endLine < line) return false;
return range.endCol >= col;
}
/**
* Return the largest index `i` in `arr` where `arr[i].range` starts at or
* before `(line, col)`. Returns `-1` if no entry starts ≤ the query.
*
* Classic "upper bound - 1" binary search: find the first entry that
* starts *after* the query, then step back one.
*/
function findLastStartLteIndex(arr: readonly Entry[], line: number, col: number): number {
let lo = 0;
let hi = arr.length;
while (lo < hi) {
const mid = (lo + hi) >>> 1;
if (startIsAtOrBefore(arr[mid]!.range, line, col)) {
lo = mid + 1;
} else {
hi = mid;
}
}
return lo - 1;
}
function wrapIndex(entriesByFile: Map<string, Entry[]>, size: number): PositionIndex {
return {
get size() {
return size;
},
atPosition(filePath: string, line: number, col: number): ScopeId | undefined {
const bucket = entriesByFile.get(filePath);
if (bucket === undefined || bucket.length === 0) return undefined;
const endIdx = findLastStartLteIndex(bucket, line, col);
if (endIdx < 0) return undefined;
// Scan backward; first containing hit is innermost (see file header).
for (let i = endIdx; i >= 0; i--) {
const entry = bucket[i]!;
if (endIsAtOrAfter(entry.range, line, col)) {
// `startIsAtOrBefore` is guaranteed true by the binary search.
return entry.id;
}
}
return undefined;
},
};
}
@@ -1,92 +0,0 @@
/**
* `QualifiedNameIndex` — O(1) `qualifiedName → DefId[]` lookup across all kinds.
*
* Cross-kind fast path for qualified-name resolution
* (`lookupQualified(qname, scope, params)` in RFC §4.5). Class, method,
* field, and namespace defs all contribute to a single index here; consumers
* filter the returned `DefId[]` by `p.acceptedKinds` at the call site.
*
* Returns `DefId[]` (not a single `DefId`) because multiple defs can legally
* share a qualified name — partial classes in C#, method overloads, or
* accidental cross-kind collisions. The lookup caller filters to the expected
* kind(s) and ranks the survivors.
*
* Part of RFC #909 Ring 2 SHARED — #913.
*
* Consumed by: #917 (`Registry.lookup` qualified fast path, `resolveTypeRef`
* dotted fallback via #916).
*/
import type { SymbolDefinition } from './symbol-definition.js';
import type { DefId } from './types.js';
export interface QualifiedNameIndex {
readonly byQualifiedName: ReadonlyMap<string, readonly DefId[]>;
readonly size: number;
/** Returns all `DefId`s registered under this qualified name; empty frozen
* array on miss so callers can iterate without null checks. */
get(qualifiedName: string): readonly DefId[];
has(qualifiedName: string): boolean;
}
/**
* Build a `QualifiedNameIndex` from a flat list of `SymbolDefinition` records.
*
* Only defs with a non-empty `qualifiedName` contribute; defs without one are
* silently skipped (not every kind carries a qualified name — anonymous or
* top-level symbols, dynamic-unresolved imports, etc.).
*
* **Duplicate policy: appended in input order.** Each unique `(qname, DefId)`
* pair contributes at most once — repeated entries for the same pair are
* deduplicated. Distinct `DefId`s sharing a `qname` accumulate in insertion
* order (stable output for deterministic lookup ranking at the call site).
*
* Pure function — safe to call repeatedly; no side effects.
*/
export function buildQualifiedNameIndex(defs: readonly SymbolDefinition[]): QualifiedNameIndex {
const byQualifiedName = new Map<string, DefId[]>();
const seenPairs = new Set<string>();
for (const def of defs) {
const qname = def.qualifiedName;
if (qname === undefined || qname.length === 0) continue;
const pairKey = `${qname}\0${def.nodeId}`;
if (seenPairs.has(pairKey)) continue;
seenPairs.add(pairKey);
const bucket = byQualifiedName.get(qname);
if (bucket === undefined) {
byQualifiedName.set(qname, [def.nodeId]);
} else {
bucket.push(def.nodeId);
}
}
// Freeze bucket arrays so consumers can't mutate the index.
const frozen = new Map<string, readonly DefId[]>();
for (const [k, v] of byQualifiedName) {
frozen.set(k, Object.freeze(v.slice()));
}
return wrapIndex(frozen);
}
// ─── Internal ───────────────────────────────────────────────────────────────
const EMPTY: readonly DefId[] = Object.freeze([]);
function wrapIndex(byQualifiedName: Map<string, readonly DefId[]>): QualifiedNameIndex {
return {
byQualifiedName,
get size() {
return byQualifiedName.size;
},
get(qualifiedName: string): readonly DefId[] {
return byQualifiedName.get(qualifiedName) ?? EMPTY;
},
has(qualifiedName: string): boolean {
return byQualifiedName.has(qualifiedName);
},
};
}
@@ -1,82 +0,0 @@
/**
* `ReferenceSite` — a pre-resolution usage fact collected by `ScopeExtractor`
* (RFC §3.2 Phase 1; Ring 2 PKG #919).
*
* One record per `@reference.*` capture. The extractor records:
* - the name being referenced (method/field/class name),
* - the source range,
* - the innermost lexical scope containing the reference,
* - the reference kind (call, read, write, inherits, etc.),
* - optional call-form classification from `provider.classifyCallForm`,
* - optional explicit-receiver hint for dotted calls (`user.save()`),
* - optional arity for call sites.
*
* Reference sites are consumed by the resolution phase (RFC §3.2 Phase 4)
* which routes each through `Registry.lookup` / `resolveTypeRef` and
* emits the final `Reference` record into `ReferenceIndex`.
*
* **Pre-resolution only.** `ReferenceSite` intentionally carries no
* `toDef`, `confidence`, or `evidence`. Those are populated by the
* resolution step that reads this record and produces a `Reference`
* (defined in `./types.ts`).
*/
import type { Range, ScopeId } from './types.js';
/**
* What kind of usage this reference represents — the graph-edge kind
* emitted after resolution (`CALLS`, `READS`, `WRITES`, etc.).
*
* Matches the `kind` field on `Reference` in `./types.ts` so the
* resolution phase can pass it through without re-classification.
*/
export type ReferenceKind =
| 'call'
| 'read'
| 'write'
| 'type-reference'
| 'inherits'
| 'import-use';
/**
* How a call site binds its target. Informs `Registry.lookup` Step 2
* (type-binding path):
* - `'free'` — bare call (no receiver); resolution via lexical chain.
* - `'member'` — dotted call (`x.foo()`); resolution via receiver type.
* - `'constructor'` — `new Foo()`; receiver is the class itself.
* - `'index'` — index expression (`arr[0]`); rare as a dispatch site.
*
* Only meaningful for `kind === 'call'`; ignored for reads/writes.
*/
export type CallForm = 'free' | 'member' | 'constructor' | 'index';
export interface ReferenceSite {
/** The name being referenced (e.g., `'save'`, `'User'`, `'count'`). */
readonly name: string;
/** Source-text range of this reference. */
readonly atRange: Range;
/**
* Innermost lexical scope that contains `atRange`. Resolved by the
* extractor via position lookup and frozen here so the resolution
* phase doesn't re-compute it per call.
*/
readonly inScope: ScopeId;
readonly kind: ReferenceKind;
/** Set when `kind === 'call'`. */
readonly callForm?: CallForm;
/**
* Explicit receiver for dotted calls (`user.save()` → `{ name: 'user' }`).
* Passed through to `Registry.lookup.explicitReceiver`.
*/
readonly explicitReceiver?: { readonly name: string };
/** Argument count at the call site; used by `provider.arityCompatibility`. */
readonly arity?: number;
/**
* Inferred argument types at the call site, one per argument. An
* empty-string entry means "unknown" — consumers narrowing overload
* candidates treat unknown as any-match. Populated by languages
* that can derive types from literals / constructor expressions
* (C#: `42` → `'int'`, `"alice"` → `'string'`).
*/
readonly argumentTypes?: readonly string[];
}
@@ -1,41 +0,0 @@
/**
* `ClassRegistry` — scope-aware lookup for class-like symbols
* (RFC §4.4; Ring 2 SHARED #917).
*
* Thin wrapper over `lookupCore`, specialized for class kinds:
*
* - `acceptedKinds` = Class / Interface / Enum / Struct / Union /
* Trait / TypeAlias / Typedef / Record / Delegate / Annotation /
* Template / Namespace.
* - `useReceiverTypeBinding` is **false** — classes are resolved by
* name through the lexical chain + global qualified fallback, not
* via a receiver type.
* - Arity filter is not applicable (classes are not called with
* argument counts at lookup time).
*/
import type { Resolution, ScopeId } from '../types.js';
import { lookupCore, type CoreLookupParams } from './lookup-core.js';
import { CLASS_KINDS, type RegistryContext } from './context.js';
export interface ClassRegistry {
/**
* Look up a class-like symbol by simple or dotted name anchored at
* `scope`. Returns a confidence-ranked `Resolution[]`; consume `[0]`
* for the best answer.
*/
lookup(name: string, scope: ScopeId): readonly Resolution[];
}
export function buildClassRegistry(ctx: RegistryContext): ClassRegistry {
const params: CoreLookupParams = {
acceptedKinds: CLASS_KINDS,
useReceiverTypeBinding: false,
ownerScopedContributor: null,
};
return {
lookup(name: string, scope: ScopeId) {
return lookupCore(name, scope, params, ctx);
},
};
}
@@ -1,110 +0,0 @@
/**
* `RegistryContext` — the injected state required by the scope-aware
* registry lookups (RFC §4; Ring 2 SHARED #917).
*
* Bundles every Ring 2 index + every provider hook the 7-step algorithm
* might consult. Threaded through `lookupCore` and the three public
* registries unchanged; construction is the caller's responsibility
* (typically once per workspace-indexing pass in Ring 2 PKG).
*
* The design intent is **pure-logic in `gitnexus-shared`, data + hooks
* supplied by the caller**. Nothing here loads files, parses AST, or
* reaches into the CLI package.
*/
import type { NodeLabel } from '../../graph/types.js';
import type { SymbolDefinition } from '../symbol-definition.js';
import type { Callsite, DefId } from '../types.js';
import type { DefIndex } from '../def-index.js';
import type { QualifiedNameIndex } from '../qualified-name-index.js';
import type { ModuleScopeIndex } from '../module-scope-index.js';
import type { ScopeTree } from '../scope-tree.js';
import type { MethodDispatchIndex } from '../method-dispatch-index.js';
// ─── Provider hooks consumed by the registries ─────────────────────────────
export interface RegistryProviders {
/**
* Language-specific arity compatibility between a callsite and a candidate
* `def`. Mirrors `LanguageProvider.arityCompatibility` from #911. Optional:
* when absent, every candidate receives `'unknown'` (neutral signal).
*/
arityCompatibility?(callsite: Callsite, def: SymbolDefinition): ArityVerdict;
}
export type ArityVerdict = 'compatible' | 'unknown' | 'incompatible';
// ─── Owner-scoped contributor (concrete shape for `RegistryContributor`) ────
/**
* Per-owner membership view plugged into `LookupParams.ownerScopedContributor`.
*
* When the caller knows a receiver is of type `Owner` (e.g., after
* resolving an explicit receiver or via `self`), it can supply the
* `Owner`'s own member bucket here. `lookupCore` treats hits from this
* contributor as `origin: 'local'` inside the owner's body scope —
* strongest-visibility evidence, unaffected by the scope-chain hop
* deduction that punishes outer-scope hits.
*
* Ring 1's `RegistryContributor = unknown` opaque placeholder is narrowed
* to this concrete shape here in Ring 2 SHARED (#917).
*/
export interface OwnerScopedContributor {
/** The owner (class/struct/trait/interface) that bounds this view. */
readonly ownerDefId: DefId;
/**
* Methods / fields directly declared on the owner, keyed by simple name.
* Return empty array on miss; implementations should NOT walk the MRO —
* that's `MethodDispatchIndex`'s job, handled in the type-binding step.
*/
byName(name: string): readonly SymbolDefinition[];
}
// ─── Top-level context threaded through every lookup ───────────────────────
export interface RegistryContext {
readonly scopes: ScopeTree;
readonly defs: DefIndex;
readonly qualifiedNames: QualifiedNameIndex;
readonly moduleScopes: ModuleScopeIndex;
/**
* Method-dispatch index; required for method/field registries that
* honor `useReceiverTypeBinding`. Omit for class-only lookups.
*/
readonly methodDispatch?: MethodDispatchIndex;
readonly providers: RegistryProviders;
}
// ─── Per-kind default `acceptedKinds` sets ─────────────────────────────────
//
// Exported so the three public registries stay declarative (each one just
// points at the right constant + passes it to `lookupCore`).
export const CLASS_KINDS: readonly NodeLabel[] = Object.freeze([
'Class',
'Interface',
'Enum',
'Struct',
'Union',
'Trait',
'TypeAlias',
'Typedef',
'Record',
'Delegate',
'Annotation',
'Template',
'Namespace',
]);
export const METHOD_KINDS: readonly NodeLabel[] = Object.freeze([
'Method',
'Function',
'Constructor',
]);
export const FIELD_KINDS: readonly NodeLabel[] = Object.freeze([
'Variable',
'Property',
'Const',
'Static',
]);
@@ -1,196 +0,0 @@
/**
* `composeEvidence` — translate accumulated raw signals per candidate
* into a `ResolutionEvidence[]` using the authoritative `EvidenceWeights`
* map (RFC §4.3 + Appendix A; Ring 2 SHARED #917).
*
* Each `RawSignals` record describes what was observed about a candidate
* during the 7-step walk: where it was found, at what depth, whether
* anything corroborates it. This module turns those raw facts into the
* typed evidence list attached to the outgoing `Resolution`.
*
* **Every weight comes from `EvidenceWeights`.** No inline magic numbers.
* Extends issue #429 (centralize hardcoded confidence values).
*
* **Confidence compose rule.** Signals add; the sum is capped at 1.0 at
* the call site (inside `lookupCore`). This module only emits the list;
* it does NOT compute the capped sum so callers can inspect per-signal
* contributions for debugging.
*/
import type { BindingRef, ResolutionEvidence } from '../types.js';
import { EvidenceWeights, typeBindingWeightAtDepth } from '../evidence-weights.js';
/**
* Raw signals observed for a single candidate during the 7-step walk.
* Optional fields encode "this signal did not fire"; presence encodes
* "emit an evidence record".
*/
export interface RawSignals {
// ── Where-found ────────────────────────────────────────────────────────
/** Visibility origin of the binding that produced this candidate. */
readonly origin?: BindingRef['origin'] | 'global-qualified' | 'global-name';
/** Depth at which the binding was found (hops up from start scope). */
readonly scopeChainDepth?: number;
/** `ImportEdge` that brought the name in; present when origin is a non-local. */
readonly viaUnlinkedImport?: boolean;
// ── Type-binding path ──────────────────────────────────────────────────
/** Set when the candidate came via the receiver's type-binding MRO walk. */
readonly typeBindingMroDepth?: number;
// ── Corroborators ──────────────────────────────────────────────────────
/** `def.ownerId === resolvedReceiver.def.nodeId`. */
readonly ownerMatch?: boolean;
/** Always fires for candidates that pass `acceptedKinds`; weight 0. */
readonly kindMatch: true;
// ── Arity ──────────────────────────────────────────────────────────────
readonly arityVerdict?: 'compatible' | 'unknown' | 'incompatible';
// ── Dynamic-unresolved passthrough ─────────────────────────────────────
/** Candidate flows through a `kind: 'dynamic-unresolved'` ImportEdge. */
readonly dynamicUnresolved?: boolean;
}
/**
* Compose the raw signals into a stable `ResolutionEvidence[]` list.
*
* Emission order mirrors the `EvidenceWeights` layout: where-found →
* type-binding → corroborators → arity → degraded. Stable order makes
* the per-signal contributions easy to reason about in tests and in the
* shadow-mode parity dashboard.
*/
export function composeEvidence(signals: RawSignals): readonly ResolutionEvidence[] {
const out: ResolutionEvidence[] = [];
// ── Where-found visibility ─────────────────────────────────────────────
if (signals.origin !== undefined) {
const baseWeight = getOriginWeight(signals.origin);
const capped = signals.viaUnlinkedImport
? baseWeight * EvidenceWeights.unlinkedImportMultiplier
: baseWeight;
const evidenceKind = whereFoundEvidenceKind(signals.origin);
out.push({
kind: evidenceKind,
weight: capped,
...(signals.viaUnlinkedImport
? { note: `via unresolved import (${EvidenceWeights.unlinkedImportMultiplier}× cap)` }
: {}),
});
}
// ── Scope-chain depth deduction (per-hop, only meaningful for lexical
// hits where scopeChainDepth ≥ 1). Depth 0 = no deduction; depth N ≥ 1
// emits a single `scope-chain` evidence with the accumulated penalty.
if (signals.scopeChainDepth !== undefined && signals.scopeChainDepth > 0) {
out.push({
kind: 'scope-chain',
weight: EvidenceWeights.scopeChainPerDepth * signals.scopeChainDepth,
note: `depth=${signals.scopeChainDepth}`,
});
}
// ── Type-binding / MRO path ────────────────────────────────────────────
if (signals.typeBindingMroDepth !== undefined) {
out.push({
kind: 'type-binding',
weight: typeBindingWeightAtDepth(signals.typeBindingMroDepth),
note: `mroDepth=${signals.typeBindingMroDepth}`,
});
}
// ── Owner match (explanatory for debug) ────────────────────────────────
if (signals.ownerMatch === true) {
out.push({
kind: 'owner-match',
weight: EvidenceWeights.ownerMatch,
});
}
// ── Kind match (always present; weight 0; retained for debuggability) ──
out.push({
kind: 'kind-match',
weight: EvidenceWeights.kindMatch,
});
// ── Arity ──────────────────────────────────────────────────────────────
if (signals.arityVerdict !== undefined) {
const weight =
signals.arityVerdict === 'compatible'
? EvidenceWeights.arityMatchCompatible
: signals.arityVerdict === 'incompatible'
? EvidenceWeights.arityMatchIncompatible
: EvidenceWeights.arityMatchUnknown;
out.push({
kind: 'arity-match',
weight,
note: signals.arityVerdict,
});
}
// ── Dynamic-unresolved (degraded signal) ───────────────────────────────
if (signals.dynamicUnresolved === true) {
out.push({
kind: 'dynamic-import-unresolved',
weight: EvidenceWeights.dynamicImportUnresolved,
});
}
return out;
}
/**
* Sum evidence weights and clamp to `[0, 1]`. Separate from `composeEvidence`
* so tests and the parity dashboard can inspect the raw evidence list.
*/
export function confidenceFromEvidence(evidence: readonly ResolutionEvidence[]): number {
let sum = 0;
for (const e of evidence) sum += e.weight;
if (sum < 0) return 0;
if (sum > 1) return 1;
return sum;
}
// ─── Internal ───────────────────────────────────────────────────────────────
function getOriginWeight(origin: NonNullable<RawSignals['origin']>): number {
switch (origin) {
case 'local':
return EvidenceWeights.local;
case 'import':
return EvidenceWeights.import;
case 'reexport':
return EvidenceWeights.reexport;
case 'namespace':
return EvidenceWeights.namespace;
case 'wildcard':
return EvidenceWeights.wildcard;
case 'global-qualified':
return EvidenceWeights.globalQualified;
case 'global-name':
// Reserved for Ring 3 byName global index. `lookupCore` today only
// emits `'global-qualified'` (via `lookupQualified`, dotted-name
// fallback); no code path constructs `origin: 'global-name'` yet.
// Kept here so the Appendix A weight stays live and `composeEvidence`
// remains exhaustive over the origin union.
return EvidenceWeights.globalName;
}
}
function whereFoundEvidenceKind(
origin: NonNullable<RawSignals['origin']>,
): ResolutionEvidence['kind'] {
switch (origin) {
case 'local':
return 'local';
case 'import':
case 'reexport':
case 'namespace':
case 'wildcard':
return 'import';
case 'global-qualified':
return 'global-qualified';
case 'global-name':
return 'global-name';
}
}
@@ -1,43 +0,0 @@
/**
* `FieldRegistry` — scope-aware lookup for field / property / variable
* access (RFC §4.4; Ring 2 SHARED #917).
*
* Thin wrapper over `lookupCore`, specialized for data-member kinds:
*
* - `acceptedKinds` = Variable / Property / Const / Static.
* - `useReceiverTypeBinding` is **true** — fields are resolved against
* the receiver type's MRO first, then via the lexical chain for
* free variables.
* - `callsite` is not meaningful for field access (no arity), but the
* `explicitReceiver` and `ownerScopedContributor` knobs are.
*/
import type { Resolution, ScopeId } from '../types.js';
import { lookupCore, type CoreLookupParams } from './lookup-core.js';
import type { OwnerScopedContributor, RegistryContext } from './context.js';
import { FIELD_KINDS } from './context.js';
export interface FieldLookupOptions {
readonly explicitReceiver?: { readonly name: string };
readonly ownerScopedContributor?: OwnerScopedContributor;
}
export interface FieldRegistry {
lookup(name: string, scope: ScopeId, options?: FieldLookupOptions): readonly Resolution[];
}
export function buildFieldRegistry(ctx: RegistryContext): FieldRegistry {
return {
lookup(name: string, scope: ScopeId, options: FieldLookupOptions = {}) {
const params: CoreLookupParams = {
acceptedKinds: FIELD_KINDS,
useReceiverTypeBinding: true,
ownerScopedContributor: options.ownerScopedContributor ?? null,
...(options.explicitReceiver !== undefined
? { explicitReceiver: options.explicitReceiver }
: {}),
};
return lookupCore(name, scope, params, ctx);
},
};
}
@@ -1,461 +0,0 @@
/**
* `lookupCore` — the shared 7-step canonical resolution algorithm
* (RFC §4.2; Ring 2 SHARED #917).
*
* Pure function. Given a name, a starting scope, and per-kind parameters,
* walks lexical scopes + optional type-binding MRO + optional owner
* contributor + global qualified-name fallback, and returns a ranked
* `Resolution[]` with per-candidate evidence.
*
* All three public registries (`ClassRegistry` / `MethodRegistry` /
* `FieldRegistry`) dispatch into this function, differing only in the
* parameters they pass. The CHOICE of which steps fire is expressed
* through `LookupParams`, not through different algorithms per kind.
*
* ## Algorithm (RFC §4.2, verbatim names)
*
* **Step 1 — Lexical scope-chain walk.** From `startScope`, walk
* parent-ward. At each scope, consult `scope.bindings.get(name)`:
* - Filter candidates whose `def.type ∈ acceptedKinds`.
* - For each surviving candidate, record a raw signal with the
* binding's origin + the current scope-chain depth.
* - **Hard shadow.** If `bindings.get(name)` is non-empty (including
* non-kind-matching candidates), stop walking. The name is
* lexically bound here; outer scopes are not consulted.
*
* **Step 2 — Type-binding resolution.** When `useReceiverTypeBinding`
* is true, resolve the receiver's type at `startScope` (from
* `scope.typeBindings`), then walk the MRO via
* `MethodDispatchIndex.mroFor(ownerDefId)`. Membership per owner comes
* through `RegistryContext.methodDispatch` + owner lookups into
* `scope.ownedDefs`; each hit records a raw signal with the owner's
* MRO depth.
*
* **Step 3 — Owner-scoped contributor.** When
* `params.ownerScopedContributor` is present, merge its `byName(name)`
* hits with `origin: 'local'` (they are declared directly on the
* receiver). Distinct from Step 2 — Step 2 walks the MRO; Step 3 only
* looks at the directly-declared owner members.
*
* **Step 4 — Kind filter (emit `kind-match` evidence).** Already
* applied during Steps 1-3; this step just adds a `kind-match` signal
* at weight 0 to every candidate for debuggability (so the evidence
* array is self-describing).
*
* **Step 5 — Arity filter.** Call `providers.arityCompatibility(callsite,
* def)` per surviving candidate. Verdicts: `compatible` / `unknown` /
* `incompatible`. If at least one candidate is `compatible`, drop
* `incompatible` ones. Otherwise keep all (the penalty weight alone
* will rank them lower but they remain in the result).
*
* **Step 6 — Global fallback.** When Steps 1-3 produced **no**
* candidates and the name contains a `.`, consult the
* `QualifiedNameIndex` via `lookupQualified` — see §4.5. The `scope`
* argument is NOT passed here because global lookup is scope-agnostic.
*
* **Step 7 — Rank + tie-break.** Compose evidence, compute confidence
* (sum capped at 1.0), sort by the RFC Appendix B cascade.
*
* ## What this module does NOT do
*
* - No AST reads (pure data in, pure data out).
* - No `gitnexus/` imports.
* - No language switches. Language-specific behavior flows exclusively
* through `providers.*` and the `params` object.
* - No caching. Callers that want memoization can wrap this function.
*/
import type { NodeLabel } from '../../graph/types.js';
import type { SymbolDefinition } from '../symbol-definition.js';
import type {
BindingRef,
Callsite,
DefId,
LookupParams,
Resolution,
Scope,
ScopeId,
} from '../types.js';
import type { OriginForTieBreak } from '../origin-priority.js';
import { composeEvidence, confidenceFromEvidence, type RawSignals } from './evidence.js';
import { compareByConfidenceWithTiebreaks, type TieBreakKey } from './tie-breaks.js';
import { lookupQualified } from './lookup-qualified.js';
import type { ArityVerdict, OwnerScopedContributor, RegistryContext } from './context.js';
// ─── Public entry point ─────────────────────────────────────────────────────
/** Extended `LookupParams` narrowing `ownerScopedContributor` to the concrete shape. */
export interface CoreLookupParams extends Omit<LookupParams, 'ownerScopedContributor'> {
readonly ownerScopedContributor: OwnerScopedContributor | null;
/** Call-site description forwarded to `arityCompatibility`. Optional — for non-call lookups. */
readonly callsite?: Callsite;
}
/**
* Run the 7-step lookup. Returns a non-empty `Resolution[]` when any
* candidate was found; an empty array otherwise. Callers consume `[0]`
* for the best answer and optionally inspect the rest for alternates.
*/
export function lookupCore(
name: string,
startScope: ScopeId,
params: CoreLookupParams,
ctx: RegistryContext,
): readonly Resolution[] {
const acceptedKinds = new Set<NodeLabel>(params.acceptedKinds);
const perCandidate = new Map<DefId, CandidateState>();
// ── Step 1: lexical scope-chain walk ──────────────────────────────────
const lexicalShadowed = walkLexicalChain(name, startScope, acceptedKinds, ctx, perCandidate);
// ── Step 2: type-binding / MRO walk (methods/fields) ──────────────────
if (params.useReceiverTypeBinding && ctx.methodDispatch !== undefined) {
walkReceiverTypeBinding(name, startScope, acceptedKinds, params, ctx, perCandidate);
}
// ── Step 3: owner-scoped contributor ──────────────────────────────────
if (params.ownerScopedContributor !== null) {
seedFromOwnerScopedContributor(
name,
params.ownerScopedContributor,
acceptedKinds,
perCandidate,
);
}
// ── Step 4: kind-match evidence (emitted by composeEvidence directly) ──
// Handled inside `composeEvidence`.
// ── Step 5: arity filter ──────────────────────────────────────────────
if (params.callsite !== undefined) {
applyArityFilter(params.callsite, perCandidate, ctx);
}
// ── Step 6: global fallback (only when Steps 1-3 produced nothing) ──
if (perCandidate.size === 0 && !lexicalShadowed && name.includes('.')) {
const globals = lookupQualified(name, { acceptedKinds: params.acceptedKinds }, ctx);
if (globals.length > 0) return globals;
}
if (perCandidate.size === 0) return EMPTY;
// ── Step 7: compose evidence + rank ──────────────────────────────────
return rankCandidates(perCandidate);
}
// ─── Internal state ────────────────────────────────────────────────────────
interface CandidateState {
readonly def: SymbolDefinition;
readonly signals: MutableRawSignals;
readonly tieBreakKey: MutableTieBreakKey;
}
interface MutableRawSignals {
origin?: BindingRef['origin'] | 'global-qualified' | 'global-name';
scopeChainDepth?: number;
viaUnlinkedImport?: boolean;
typeBindingMroDepth?: number;
ownerMatch?: boolean;
kindMatch: true;
arityVerdict?: ArityVerdict;
dynamicUnresolved?: boolean;
}
interface MutableTieBreakKey {
scopeDepth: number;
mroDepth: number;
origin: OriginForTieBreak;
}
function ensureCandidate(
perCandidate: Map<DefId, CandidateState>,
def: SymbolDefinition,
): CandidateState {
const existing = perCandidate.get(def.nodeId);
if (existing !== undefined) return existing;
const fresh: CandidateState = {
def,
signals: { kindMatch: true },
tieBreakKey: { scopeDepth: 0, mroDepth: 0, origin: 'local' },
};
perCandidate.set(def.nodeId, fresh);
return fresh;
}
// ─── Step 1 implementation ─────────────────────────────────────────────────
/**
* Walk the lexical scope chain from `startScope` upward. Returns `true`
* iff a scope with any `bindings.get(name)` entries was found — the
* caller uses this to decide whether to run the global fallback.
*/
function walkLexicalChain(
name: string,
startScope: ScopeId,
acceptedKinds: ReadonlySet<NodeLabel>,
ctx: RegistryContext,
perCandidate: Map<DefId, CandidateState>,
): boolean {
let currentId: ScopeId | null = startScope;
let depth = 0;
const visited = new Set<ScopeId>();
while (currentId !== null) {
if (visited.has(currentId)) return false;
visited.add(currentId);
const scope: Scope | undefined = ctx.scopes.getScope(currentId);
if (scope === undefined) return false;
const bindings = scope.bindings.get(name);
if (bindings !== undefined && bindings.length > 0) {
for (const binding of bindings) {
if (!acceptedKinds.has(binding.def.type)) continue;
recordLexicalHit(perCandidate, binding, depth);
}
return true; // hard shadow regardless of kind-filter survivorship
}
currentId = scope.parent;
depth++;
}
return false;
}
function recordLexicalHit(
perCandidate: Map<DefId, CandidateState>,
binding: BindingRef,
scopeChainDepth: number,
): void {
const state = ensureCandidate(perCandidate, binding.def);
state.signals.origin = binding.origin;
state.signals.scopeChainDepth = scopeChainDepth;
if (binding.via?.linkStatus === 'unresolved') {
state.signals.viaUnlinkedImport = true;
}
if (binding.via?.kind === 'dynamic-unresolved') {
state.signals.dynamicUnresolved = true;
}
state.tieBreakKey.scopeDepth = scopeChainDepth;
state.tieBreakKey.origin = binding.origin as OriginForTieBreak;
}
// ─── Step 2 implementation ─────────────────────────────────────────────────
function walkReceiverTypeBinding(
name: string,
startScope: ScopeId,
acceptedKinds: ReadonlySet<NodeLabel>,
params: CoreLookupParams,
ctx: RegistryContext,
perCandidate: Map<DefId, CandidateState>,
): void {
const ownerDefId = resolveReceiverOwner(startScope, params, ctx);
if (ownerDefId === undefined) return;
if (ctx.methodDispatch === undefined) return;
const ownerDef = ctx.defs.get(ownerDefId);
if (ownerDef === undefined) return;
// Walk the owner itself at depth 0, then its MRO chain.
const walk: DefId[] = [ownerDefId, ...ctx.methodDispatch.mroFor(ownerDefId)];
for (let mroDepth = 0; mroDepth < walk.length; mroDepth++) {
const currentOwnerId = walk[mroDepth]!;
const members = collectOwnedMembers(currentOwnerId, name, ctx);
for (const def of members) {
if (!acceptedKinds.has(def.type)) continue;
recordTypeBindingHit(perCandidate, def, mroDepth, ownerDefId);
}
}
}
function resolveReceiverOwner(
startScope: ScopeId,
params: CoreLookupParams,
ctx: RegistryContext,
): DefId | undefined {
// Explicit receiver: consult the callsite scope's typeBindings for the
// named receiver; the attached TypeRef identifies the owner. Without a
// ready resolveTypeRef call (that module is separate), we do a direct
// lookup and trust the caller to have populated the binding.
if (params.explicitReceiver !== undefined) {
return lookupReceiverType(startScope, params.explicitReceiver.name, ctx);
}
// Implicit `self` / `this` — the scope's typeBindings should carry it.
for (const implicitName of IMPLICIT_RECEIVERS) {
const owner = lookupReceiverType(startScope, implicitName, ctx);
if (owner !== undefined) return owner;
}
return undefined;
}
const IMPLICIT_RECEIVERS: readonly string[] = Object.freeze(['self', 'this']);
function lookupReceiverType(
startScope: ScopeId,
receiverName: string,
ctx: RegistryContext,
): DefId | undefined {
let currentId: ScopeId | null = startScope;
const visited = new Set<ScopeId>();
while (currentId !== null) {
if (visited.has(currentId)) return undefined;
visited.add(currentId);
const scope = ctx.scopes.getScope(currentId);
if (scope === undefined) return undefined;
const typeRef = scope.typeBindings.get(receiverName);
if (typeRef !== undefined) {
// rawName must resolve to a def via qualifiedNames; if it doesn't, we
// can't claim the receiver type. No fallback — that's what
// `resolveTypeRef` would do, but we keep this path lean and let
// callers pre-resolve if they want the richer semantics.
const candidateIds = ctx.qualifiedNames.get(typeRef.rawName);
if (candidateIds.length === 1) return candidateIds[0];
// Ambiguous (≥ 2) or missing (0) — caller must pre-resolve via
// `resolveTypeRef` (#916) if they want the richer semantics. We
// intentionally do NOT re-implement a simple-name fallback here.
return undefined;
}
currentId = scope.parent;
}
return undefined;
}
function collectOwnedMembers(
ownerDefId: DefId,
memberName: string,
ctx: RegistryContext,
): readonly SymbolDefinition[] {
// An owner's members are defs whose `ownerId === ownerDefId` and whose
// simple name matches `memberName`. We iterate `defs.byId` — O(D) per
// call today. A future by-owner index would make this O(K); tracked as
// a follow-up optimization before Ring 3 flips go production.
const out: SymbolDefinition[] = [];
for (const def of ctx.defs.byId.values()) {
if (def.ownerId !== ownerDefId) continue;
if (simpleNameOf(def) !== memberName) continue;
out.push(def);
}
return out;
}
function simpleNameOf(def: SymbolDefinition): string | undefined {
if (def.qualifiedName === undefined || def.qualifiedName.length === 0) return undefined;
const dot = def.qualifiedName.lastIndexOf('.');
return dot === -1 ? def.qualifiedName : def.qualifiedName.slice(dot + 1);
}
function recordTypeBindingHit(
perCandidate: Map<DefId, CandidateState>,
def: SymbolDefinition,
mroDepth: number,
receiverOwner: DefId,
): void {
const state = ensureCandidate(perCandidate, def);
const existingMroDepth = state.signals.typeBindingMroDepth;
const firstHit = existingMroDepth === undefined;
// Only replace if this hit is shallower (smaller MRO depth). The local
// const lets TS narrow to `number` in the `else` branch so no `!`
// assertion is needed.
if (firstHit || mroDepth < existingMroDepth) {
state.signals.typeBindingMroDepth = mroDepth;
state.tieBreakKey.mroDepth = mroDepth;
}
if (def.ownerId === receiverOwner) {
state.signals.ownerMatch = true;
}
// Pure type-binding candidates (no lexical hit) would otherwise keep the
// `ensureCandidate` default `tieBreakKey.origin === 'local'`, making the
// Appendix B cascade lump them with local-origin candidates. Demote them
// to `'import'` — the strongest non-local origin — only when no earlier
// phase set an origin for this candidate. Lexical hits from Step 1 set
// `signals.origin` before Step 2 runs, so the guard skips them; Step 3
// (`seedFromOwnerScopedContributor`) runs AFTER Step 2 and unconditionally
// overrides `tieBreakKey.origin` back to `'local'` for direct-owner
// members, so any same-def overlap still ends up ranked correctly.
if (firstHit && state.signals.origin === undefined) {
state.tieBreakKey.origin = 'import';
}
}
// ─── Step 3 implementation ─────────────────────────────────────────────────
function seedFromOwnerScopedContributor(
name: string,
contributor: OwnerScopedContributor,
acceptedKinds: ReadonlySet<NodeLabel>,
perCandidate: Map<DefId, CandidateState>,
): void {
for (const def of contributor.byName(name)) {
if (!acceptedKinds.has(def.type)) continue;
const state = ensureCandidate(perCandidate, def);
// Treat the contributor's direct membership as `origin: 'local'` —
// strongest visibility, no scope-chain penalty.
state.signals.origin = 'local';
state.signals.scopeChainDepth = 0;
state.signals.ownerMatch = def.ownerId === contributor.ownerDefId;
state.tieBreakKey.origin = 'local';
}
}
// ─── Step 5 implementation ─────────────────────────────────────────────────
function applyArityFilter(
callsite: Callsite,
perCandidate: Map<DefId, CandidateState>,
ctx: RegistryContext,
): void {
const arityFn = ctx.providers.arityCompatibility;
if (arityFn === undefined) {
// No provider → record 'unknown' for every candidate; keeps signal
// shape uniform for composeEvidence.
for (const state of perCandidate.values()) {
state.signals.arityVerdict = 'unknown';
}
return;
}
let anyCompatible = false;
for (const state of perCandidate.values()) {
const verdict = arityFn(callsite, state.def);
state.signals.arityVerdict = verdict;
if (verdict === 'compatible') anyCompatible = true;
}
if (!anyCompatible) return;
// Filter: when at least one compatible candidate exists, drop incompatibles.
for (const [defId, state] of perCandidate) {
if (state.signals.arityVerdict === 'incompatible') {
perCandidate.delete(defId);
}
}
}
// ─── Step 7 implementation ─────────────────────────────────────────────────
function rankCandidates(perCandidate: Map<DefId, CandidateState>): readonly Resolution[] {
const resolutions: Resolution[] = [];
const tieKeys = new Map<string, TieBreakKey>();
for (const state of perCandidate.values()) {
const evidence = composeEvidence(state.signals as RawSignals);
const confidence = confidenceFromEvidence(evidence);
resolutions.push({ def: state.def, confidence, evidence });
tieKeys.set(state.def.nodeId, { ...state.tieBreakKey });
}
resolutions.sort((a, b) => compareByConfidenceWithTiebreaks(a, b, tieKeys));
return Object.freeze(resolutions);
}
// ─── Constants ──────────────────────────────────────────────────────────────
const EMPTY: readonly Resolution[] = Object.freeze([]);
@@ -1,71 +0,0 @@
/**
* `lookupQualified` — qualified-name fast path (RFC §4.5; Ring 2 SHARED #917).
*
* Consults `QualifiedNameIndex` directly, filters by `acceptedKinds`, and
* returns `Resolution[]` with `origin: 'global-qualified'` evidence. Used by:
*
* - `resolveTypeRef` dotted fallback (#916)
* - `Registry.lookup` Step 6 when no lexical candidate survived
* - Explicit dotted identifiers in Cypher / MCP tools where the caller
* knows the target's canonical qualified name
*
* **Strict + deterministic.** No receiver-type resolution, no scope walk.
* Every surviving candidate gets the same base confidence (from
* `EvidenceWeights.globalQualified`), then the tie-break cascade
* disambiguates.
*/
import type { NodeLabel } from '../../graph/types.js';
import type { Resolution } from '../types.js';
import { composeEvidence, confidenceFromEvidence } from './evidence.js';
import { compareByConfidenceWithTiebreaks, type TieBreakKey } from './tie-breaks.js';
import type { RegistryContext } from './context.js';
export interface LookupQualifiedParams {
readonly acceptedKinds: readonly NodeLabel[];
}
/**
* Look up a canonical qualified name (e.g., `app.models.User`) across all
* defs, filtered by `acceptedKinds`. Returns an empty array when the name
* is not indexed or no candidate matches the kind filter.
*
* Callers consume `[0]` for the strict single-return answer; the remainder
* carries alternate candidates (partial classes, overloads, accidental
* cross-kind hits) ordered by the tie-break cascade.
*/
export function lookupQualified(
qualifiedName: string,
params: LookupQualifiedParams,
ctx: RegistryContext,
): readonly Resolution[] {
const defIds = ctx.qualifiedNames.get(qualifiedName);
if (defIds.length === 0) return EMPTY;
const acceptedKinds = new Set<NodeLabel>(params.acceptedKinds);
const resolutions: Resolution[] = [];
const tieKeys = new Map<string, TieBreakKey>();
for (const defId of defIds) {
const def = ctx.defs.get(defId);
if (def === undefined) continue;
if (!acceptedKinds.has(def.type)) continue;
const evidence = composeEvidence({ origin: 'global-qualified', kindMatch: true });
const confidence = confidenceFromEvidence(evidence);
resolutions.push({ def, confidence, evidence });
tieKeys.set(def.nodeId, {
scopeDepth: 0,
mroDepth: 0,
origin: 'global-qualified',
});
}
if (resolutions.length === 0) return EMPTY;
resolutions.sort((a, b) => compareByConfidenceWithTiebreaks(a, b, tieKeys));
return Object.freeze(resolutions);
}
const EMPTY: readonly Resolution[] = Object.freeze([]);
@@ -1,54 +0,0 @@
/**
* `MethodRegistry` — scope-aware lookup for method / function / constructor
* dispatch (RFC §4.4; Ring 2 SHARED #917).
*
* Thin wrapper over `lookupCore`, specialized for callable kinds:
*
* - `acceptedKinds` = Method / Function / Constructor.
* - `useReceiverTypeBinding` is **true** — the type-binding + MRO walk
* (Step 2) is the primary evidence path for receiver-dispatched calls.
* - `callsite.arity` flows through to `provider.arityCompatibility`
* when provided. When the provider is absent, arity evidence is
* `unknown` (neutral signal).
*/
import type { Callsite, Resolution, ScopeId } from '../types.js';
import { lookupCore, type CoreLookupParams } from './lookup-core.js';
import type { OwnerScopedContributor, RegistryContext } from './context.js';
import { METHOD_KINDS } from './context.js';
/**
* Extra per-call parameters that vary across call sites but NOT across
* registries. Kept as a separate shape so `MethodRegistry.lookup` stays
* concise while still exposing the explicit-receiver + owner-contributor +
* arity knobs the RFC algorithm needs.
*/
export interface MethodLookupOptions {
/** Call-site arity for `provider.arityCompatibility`. */
readonly callsite?: Callsite;
/** Explicit receiver (e.g., `user` in `user.save()`). See §4.1. */
readonly explicitReceiver?: { readonly name: string };
/** Optional per-owner contributor (Step 3). */
readonly ownerScopedContributor?: OwnerScopedContributor;
}
export interface MethodRegistry {
lookup(name: string, scope: ScopeId, options?: MethodLookupOptions): readonly Resolution[];
}
export function buildMethodRegistry(ctx: RegistryContext): MethodRegistry {
return {
lookup(name: string, scope: ScopeId, options: MethodLookupOptions = {}) {
const params: CoreLookupParams = {
acceptedKinds: METHOD_KINDS,
useReceiverTypeBinding: true,
ownerScopedContributor: options.ownerScopedContributor ?? null,
...(options.callsite !== undefined ? { callsite: options.callsite } : {}),
...(options.explicitReceiver !== undefined
? { explicitReceiver: options.explicitReceiver }
: {}),
};
return lookupCore(name, scope, params, ctx);
},
};
}
@@ -1,76 +0,0 @@
/**
* `compareByConfidenceWithTiebreaks` — the RFC §4.2 Step 7 total order
* over `Resolution` candidates (Ring 2 SHARED #917).
*
* Primary key is confidence (DESC). Remaining ties within `CONFIDENCE_EPSILON`
* fall through a deterministic cascade so the same inputs always produce
* the same winner, independent of insertion order.
*
* Tie-break cascade (per RFC Appendix B):
*
* 1. confidence DESC (primary)
* 2. scope depth ASC (nearer lexical scope wins)
* 3. MRO depth ASC (nearer class in hierarchy wins)
* 4. `ORIGIN_PRIORITY` ASC (local > import > … > global-name)
* 5. DefId.localeCompare (final deterministic tiebreaker)
*
* The per-candidate inputs needed beyond `Resolution.confidence` —
* `scopeDepth`, `mroDepth`, `origin` — are supplied via a sidecar
* `TieBreakKey` so the comparator stays pure and `Resolution` itself
* doesn't need to carry book-keeping fields.
*/
import { ORIGIN_PRIORITY, type OriginForTieBreak } from '../origin-priority.js';
import type { Resolution } from '../types.js';
export const CONFIDENCE_EPSILON = 0.001;
/** Side-information per candidate used for secondary tie-breaks. */
export interface TieBreakKey {
readonly scopeDepth: number;
readonly mroDepth: number;
readonly origin: OriginForTieBreak;
}
/**
* Pure comparator suitable for `Array.prototype.sort`. Return value follows
* the JavaScript convention: negative → `a` wins, positive → `b` wins.
*
* **Important:** `keys` is keyed by `Resolution.def.nodeId`, not by array
* index — stable across reorderings. Missing keys fall back to neutral
* values (`scopeDepth: 0`, `mroDepth: 0`, `origin: 'local'`), which means
* the tie-break degrades gracefully to defId-lexicographic ordering when
* side-info is unavailable. That keeps the total order deterministic
* even on malformed inputs.
*/
export function compareByConfidenceWithTiebreaks(
a: Resolution,
b: Resolution,
keys: ReadonlyMap<string, TieBreakKey>,
): number {
// Primary: confidence DESC, treating values within epsilon as equal.
const delta = b.confidence - a.confidence;
if (Math.abs(delta) >= CONFIDENCE_EPSILON) return delta < 0 ? -1 : 1;
const ka = keys.get(a.def.nodeId) ?? DEFAULT_KEY;
const kb = keys.get(b.def.nodeId) ?? DEFAULT_KEY;
// Secondary: scope depth ASC.
if (ka.scopeDepth !== kb.scopeDepth) return ka.scopeDepth - kb.scopeDepth;
// Tertiary: MRO depth ASC.
if (ka.mroDepth !== kb.mroDepth) return ka.mroDepth - kb.mroDepth;
// Quaternary: ORIGIN_PRIORITY ASC.
const po = ORIGIN_PRIORITY[ka.origin] - ORIGIN_PRIORITY[kb.origin];
if (po !== 0) return po;
// Final: DefId lexicographic, locale-aware for deterministic cross-platform output.
return a.def.nodeId.localeCompare(b.def.nodeId);
}
const DEFAULT_KEY: TieBreakKey = Object.freeze({
scopeDepth: 0,
mroDepth: 0,
origin: 'local',
});
@@ -1,148 +0,0 @@
/**
* `resolveTypeRef` — strict single-return resolver for `TypeRef`s
* (RFC §4.6; Ring 2 SHARED #916).
*
* Narrower contract than `Registry.lookup`: no name-only global fallback, no
* confidence ranking, no arity check. Used by `Registry.lookup` Step 2 (type-
* binding propagation) and by any caller that wants the single best type-
* target for an annotation without paying for the full evidence pipeline.
*
* **Algorithm (strict).** Walk the scope chain from `ref.declaredAtScope`:
*
* 1. At each scope, inspect `bindings.get(ref.rawName)`:
* - If one of the bindings is a **type-kind** def with a **strict origin**
* (`'local' | 'import' | 'namespace' | 'reexport'`), return it.
* - If any binding for this name exists at this scope but none qualifies
* (e.g., a local variable named `User` shadows an outer import of class
* `User`), return `null`. The nearer binding shadows; we do NOT fall
* through to the global qualified-name index.
* - Otherwise continue to the parent scope.
* 2. If the raw name is a dotted path (e.g., `'models.User'`) and the scope
* walk produced no match, consult `QualifiedNameIndex.byQualifiedName`.
* Only accept **exactly one** type-kind hit — anything ambiguous returns
* `null` rather than a guess.
* 3. Return `null`.
*
* **What `'strict' origins' means.** `'wildcard'` is intentionally excluded.
* A wildcard-expanded name (`from x import *`) is too loose to use as an
* anchor for type resolution — it gives no signal about whether the name was
* actually imported. `Registry.lookup` may accept wildcard bindings at its
* own discretion (with lower evidence weight); `resolveTypeRef` does not.
*
* **What 'type-kind' means.** The subset of `NodeLabel` that a type annotation
* may legitimately reference: class-like, interface-like, enum-like, and
* alias-like kinds. See `TYPE_KINDS` below.
*
* Pure function — safe to call repeatedly; no side effects.
*/
import type { NodeLabel } from '../graph/types.js';
import type { SymbolDefinition } from './symbol-definition.js';
import type { BindingRef, ScopeId, ScopeLookup, TypeRef } from './types.js';
import type { DefIndex } from './def-index.js';
import type { QualifiedNameIndex } from './qualified-name-index.js';
// ─── Public contracts ───────────────────────────────────────────────────────
/**
* All inputs `resolveTypeRef` needs from the semantic model. Bundled into a
* context object so the call site stays short and the interface is stable as
* additional indexes get threaded through in later rings.
*/
export interface ResolveTypeRefContext {
readonly scopes: ScopeLookup;
readonly defIndex: DefIndex;
readonly qualifiedNameIndex: QualifiedNameIndex;
}
// ─── Strict policy constants ────────────────────────────────────────────────
/** `'wildcard'` is deliberately absent. See file header. */
const STRICT_ORIGINS: ReadonlySet<BindingRef['origin']> = new Set<BindingRef['origin']>([
'local',
'import',
'namespace',
'reexport',
]);
/**
* `NodeLabel` values that may appear on the RHS of a type annotation.
*
* Includes the usual class-like and interface-like kinds plus the alias-like
* ones (`TypeAlias`, `Typedef`). `Namespace` is excluded — it is a scope
* container, not a value type. `Function` / `Method` / `Variable` are
* excluded by design: a `rawName` bound to them at a strict origin is a
* *shadowing* binding, which the algorithm short-circuits to `null`.
*
* `'Type'` (the generic `NodeLabel` value) is also excluded — verified
* against `gitnexus/src/core/ingestion/` at the time of writing, no
* production extractor emits `type: 'Type'` for annotation-relevant
* symbols. Should a future extractor start emitting it, add `'Type'`
* here and add a test asserting the new path.
*/
const TYPE_KINDS: ReadonlySet<NodeLabel> = new Set<NodeLabel>([
'Class',
'Interface',
'Enum',
'Struct',
'Union',
'Trait',
'TypeAlias',
'Typedef',
'Record',
'Delegate',
'Annotation',
'Template',
]);
// ─── Main entry point ──────────────────────────────────────────────────────
export function resolveTypeRef(ref: TypeRef, ctx: ResolveTypeRefContext): SymbolDefinition | null {
// Phase 1: scope-chain walk anchored at the declaration site.
let currentId: ScopeId | null = ref.declaredAtScope;
const visited = new Set<ScopeId>();
while (currentId !== null) {
// Cycle guard — a well-formed scope tree never loops, but a bug in the
// construction path should fail fast here rather than hanging.
if (visited.has(currentId)) return null;
visited.add(currentId);
const scope = ctx.scopes.getScope(currentId);
if (scope === undefined) return null; // broken chain = unresolvable
const bindings = scope.bindings.get(ref.rawName);
if (bindings !== undefined && bindings.length > 0) {
// At least one binding exists at this scope → it is the shadowing site.
// Either one of them qualifies, or the name is shadowed by a non-type.
for (const binding of bindings) {
if (!STRICT_ORIGINS.has(binding.origin)) continue;
if (TYPE_KINDS.has(binding.def.type)) {
return binding.def;
}
}
// Shadowed by a non-type / non-strict-origin binding. Fail fast — no
// global fallback, no walk to the parent.
return null;
}
currentId = scope.parent;
}
// Phase 2: dotted fallback via `QualifiedNameIndex`. Only accept a unique
// type-kind hit; anything ambiguous returns null (strict: no guesses).
if (ref.rawName.includes('.')) {
const candidates = ctx.qualifiedNameIndex.get(ref.rawName);
let onlyTypeDef: SymbolDefinition | null = null;
for (const defId of candidates) {
const def = ctx.defIndex.get(defId);
if (def === undefined) continue;
if (!TYPE_KINDS.has(def.type)) continue;
if (onlyTypeDef !== null) return null; // ambiguous
onlyTypeDef = def;
}
if (onlyTypeDef !== null) return onlyTypeDef;
}
return null;
}
@@ -1,57 +0,0 @@
/**
* `ScopeId` canonical constructor + string intern pool
* (RFC §2.2; Ring 2 SHARED #912).
*
* `ScopeId` is a deterministic string derived from the scope's file path,
* byte range, and kind:
*
* scope:{filePath}#{startLine}:{startCol}-{endLine}:{endCol}:{kind}
*
* Two scopes produced by reparsing the same file at the same positions are
* `===`-equal as strings. Beyond the canonical shape, `makeScopeId` also
* **interns** the string through a process-local pool, so repeated calls
* with structurally identical inputs return the same string reference —
* making `Map<ScopeId, ...>` lookups and cache keys identity-fast.
*
* The intern pool is unbounded. The number of distinct `ScopeId`s across a
* single indexing run is O(total scopes in workspace), which is bounded by
* source-text size and already in memory; interning adds no asymptotic
* pressure. `clearScopeIdInternPool` is exported for test isolation.
*/
import type { Range } from './types.js';
import type { ScopeId, ScopeKind } from './types.js';
/** Inputs required to construct a canonical `ScopeId`. */
export interface ScopeIdInput {
readonly filePath: string;
readonly range: Range;
readonly kind: ScopeKind;
}
/**
* Build a canonical `ScopeId` from its structural parts and intern it.
*
* Pure + referentially transparent: given the same input shape, always
* returns the same string reference for the lifetime of the pool.
*/
export function makeScopeId(input: ScopeIdInput): ScopeId {
const raw = `scope:${input.filePath}#${input.range.startLine}:${input.range.startCol}-${input.range.endLine}:${input.range.endCol}:${input.kind}`;
const existing = INTERN_POOL.get(raw);
if (existing !== undefined) return existing;
INTERN_POOL.set(raw, raw);
return raw;
}
/**
* Drop the intern pool. Intended for test setup/teardown — production code
* should not need this, since the pool's memory usage is bounded by the
* number of live scopes and cleaning it mid-run would break identity
* equality for existing scope ids.
*/
export function clearScopeIdInternPool(): void {
INTERN_POOL.clear();
}
/** Internal: shared intern pool (process-local). */
const INTERN_POOL = new Map<string, string>();
@@ -1,295 +0,0 @@
/**
* `ScopeTree` — the lexical-scope spine of the `SemanticModel`
* (RFC §2.2 + §3.1; Ring 2 SHARED #912).
*
* Generalizes the `enclosingFunctions` pattern from closed PR #902 to
* arbitrary `ScopeKind`s. Owns the (parent ↔ children) relationship
* derived from each `Scope.parent` pointer, and validates the structural
* invariants a well-formed scope tree must satisfy.
*
* Invariants enforced at build time (throw on violation):
*
* - Every non-`Module` scope has a non-null parent.
* - Every parent pointer references a scope that was also supplied to
* `buildScopeTree`.
* - Parent range **strictly contains** child range.
* - Sibling ranges under the same parent do not overlap.
* - Parent and child live in the same `filePath`. (Cross-file parent
* pointers would be a category error — a `File` scope is not the
* parent of another file's scopes; imports do that job.)
*
* Satisfies the `ScopeLookup` contract (defined in `./types.js`), so
* `resolveTypeRef` (#916) and the scope-aware registries (#917) can take a
* `ScopeTree` directly without adapters.
*
* Immutable surface: `byId` is a `ReadonlyMap`; children arrays are
* `Object.freeze`d; miss lookups return a shared frozen empty array.
*/
import type { Scope, ScopeId, ScopeLookup, Range } from './types.js';
// ─── Public contract ────────────────────────────────────────────────────────
export interface ScopeTree extends ScopeLookup {
readonly size: number;
readonly byId: ReadonlyMap<ScopeId, Scope>;
getScope(id: ScopeId): Scope | undefined;
getParent(id: ScopeId): Scope | undefined;
/** Child `ScopeId`s of `id`, in input order. Frozen empty array on miss. */
getChildren(id: ScopeId): readonly ScopeId[];
/**
* Ancestor chain from the immediate parent up to (and including) the
* root module scope. Excludes the starting scope itself. Frozen empty
* array on miss / for a root scope.
*/
getAncestors(id: ScopeId): readonly ScopeId[];
has(id: ScopeId): boolean;
}
// ─── Build errors ───────────────────────────────────────────────────────────
/**
* Thrown by `buildScopeTree` when the input violates a structural
* invariant. Carries the offending ids + the invariant name so failed
* extraction pipelines can report actionable diagnostics.
*/
export class ScopeTreeInvariantError extends Error {
constructor(
readonly invariant:
| 'non-module-requires-parent'
| 'parent-not-found'
| 'parent-must-contain-child'
| 'sibling-ranges-overlap'
| 'parent-must-share-filepath'
| 'duplicate-scope-id',
message: string,
) {
super(message);
this.name = 'ScopeTreeInvariantError';
}
}
// ─── Builder ───────────────────────────────────────────────────────────────
/**
* Build an immutable `ScopeTree` from a flat list of `Scope` records.
*
* Throws `ScopeTreeInvariantError` on the first invariant violation; a
* malformed tree is a bug in the extraction pipeline, not a data case for
* consumers to handle, so fail-fast is the correct posture.
*/
export function buildScopeTree(scopes: readonly Scope[]): ScopeTree {
const byId = new Map<ScopeId, Scope>();
const childrenById = new Map<ScopeId, ScopeId[]>();
// ── Pass 1: collect by id + duplicate check ───────────────────────────
for (const scope of scopes) {
if (byId.has(scope.id)) {
throw new ScopeTreeInvariantError(
'duplicate-scope-id',
`Two scopes share id '${scope.id}'. Scope ids must be unique per tree.`,
);
}
byId.set(scope.id, scope);
}
// ── Pass 2: validate parent pointers + build children buckets ─────────
for (const scope of scopes) {
if (scope.parent === null) {
if (scope.kind !== 'Module') {
throw new ScopeTreeInvariantError(
'non-module-requires-parent',
`Scope '${scope.id}' has kind '${scope.kind}' but no parent. Only 'Module' scopes may be root-level.`,
);
}
continue;
}
const parent = byId.get(scope.parent);
if (parent === undefined) {
throw new ScopeTreeInvariantError(
'parent-not-found',
`Scope '${scope.id}' references parent '${scope.parent}' which is not part of this tree.`,
);
}
if (parent.filePath !== scope.filePath) {
throw new ScopeTreeInvariantError(
'parent-must-share-filepath',
`Scope '${scope.id}' (${scope.filePath}) has parent '${parent.id}' in a different file (${parent.filePath}). Parent/child scopes must share filePath.`,
);
}
if (!canParentScope(parent.range, scope.range, parent.kind, scope.kind)) {
throw new ScopeTreeInvariantError(
'parent-must-contain-child',
`Parent scope '${parent.id}' at ${formatRange(parent.range)} does not contain child '${scope.id}' at ${formatRange(scope.range)} (allowed: strict containment, or equal-range Module-as-parent).`,
);
}
let bucket = childrenById.get(parent.id);
if (bucket === undefined) {
bucket = [];
childrenById.set(parent.id, bucket);
}
bucket.push(scope.id);
}
// ── Pass 3: sibling-overlap check ─────────────────────────────────────
for (const [parentId, childIds] of childrenById) {
if (childIds.length < 2) continue;
// Sort siblings by (startLine, startCol) for an O(n log n) pairwise
// scan instead of O(n²) all-pairs.
const children = childIds.map((id) => byId.get(id)!).slice();
children.sort((a, b) => comparePosition(a.range, b.range));
for (let i = 1; i < children.length; i++) {
const prev = children[i - 1]!;
const curr = children[i]!;
if (rangesOverlap(prev.range, curr.range)) {
throw new ScopeTreeInvariantError(
'sibling-ranges-overlap',
`Sibling scopes under parent '${parentId}' overlap: '${prev.id}' ${formatRange(prev.range)} and '${curr.id}' ${formatRange(curr.range)}.`,
);
}
}
}
// Freeze children arrays so the surface is truly read-only.
const frozenChildren = new Map<ScopeId, readonly ScopeId[]>();
for (const [parentId, childIds] of childrenById) {
frozenChildren.set(parentId, Object.freeze(childIds.slice()));
}
return freezeTree(byId, frozenChildren);
}
// ─── Internals ──────────────────────────────────────────────────────────────
const EMPTY_CHILDREN: readonly ScopeId[] = Object.freeze([]);
function freezeTree(
byId: Map<ScopeId, Scope>,
childrenById: Map<ScopeId, readonly ScopeId[]>,
): ScopeTree {
return {
byId,
get size() {
return byId.size;
},
getScope(id: ScopeId): Scope | undefined {
return byId.get(id);
},
getParent(id: ScopeId): Scope | undefined {
const scope = byId.get(id);
if (scope === undefined || scope.parent === null) return undefined;
return byId.get(scope.parent);
},
getChildren(id: ScopeId): readonly ScopeId[] {
return childrenById.get(id) ?? EMPTY_CHILDREN;
},
getAncestors(id: ScopeId): readonly ScopeId[] {
const start = byId.get(id);
if (start === undefined || start.parent === null) return EMPTY_CHILDREN;
const out: ScopeId[] = [];
const visited = new Set<ScopeId>([id]);
let cursor: ScopeId | null = start.parent;
while (cursor !== null && !visited.has(cursor)) {
visited.add(cursor);
out.push(cursor);
const next = byId.get(cursor);
cursor = next === undefined ? null : next.parent;
}
return Object.freeze(out);
},
has(id: ScopeId): boolean {
return byId.has(id);
},
};
}
/**
* `outer` strictly contains `inner` when `outer`'s start is at or before
* `inner`'s start, `outer`'s end is at or after `inner`'s end, and they are
* not the exact same range. Equal ranges are rejected — a child cannot
* occupy the exact same span as its parent.
*/
function rangeStrictlyContains(outer: Range, inner: Range): boolean {
if (
outer.startLine === inner.startLine &&
outer.startCol === inner.startCol &&
outer.endLine === inner.endLine &&
outer.endCol === inner.endCol
) {
return false;
}
const outerStartsAtOrBefore =
outer.startLine < inner.startLine ||
(outer.startLine === inner.startLine && outer.startCol <= inner.startCol);
const outerEndsAtOrAfter =
outer.endLine > inner.endLine ||
(outer.endLine === inner.endLine && outer.endCol >= inner.endCol);
return outerStartsAtOrBefore && outerEndsAtOrAfter;
}
function rangesEqual(a: Range, b: Range): boolean {
return (
a.startLine === b.startLine &&
a.startCol === b.startCol &&
a.endLine === b.endLine &&
a.endCol === b.endCol
);
}
/**
* Whether `outer` (kind `outerKind`) is a valid parent for `inner` (kind
* `innerKind`).
*
* Strict containment is the general rule. The single carve-out is the
* `Module`/non-`Module` pair whose ranges are exactly equal — this happens
* naturally when tree-sitter reports identical byte spans for the
* `compilation_unit` (or equivalent file-root construct) and the file's
* single top-level scope. Common shape: a C# file consisting of nothing
* but `namespace X { ... }` with no leading or trailing trivia outside the
* namespace's `{}` body — `compilation_unit` and `namespace_declaration`
* both span exactly the same byte range. The `Module` is the universal
* outer of any file-level scope by language semantics, so coincident
* ranges should not break the parent chain.
*
* The carve-out is direction-asymmetric: only `Module`-as-outer parents a
* same-range non-`Module`, never the reverse. This preserves the
* acyclicity buildScopeTree relies on, and matches the corresponding
* helper in `scope-extractor.ts` so `pass1BuildScopes` and the validator
* agree on what a well-formed parent edge looks like.
*/
export function canParentScope(
outer: Range,
inner: Range,
outerKind: Scope['kind'],
innerKind: Scope['kind'],
): boolean {
if (rangeStrictlyContains(outer, inner)) return true;
if (outerKind === 'Module' && innerKind !== 'Module' && rangesEqual(outer, inner)) return true;
return false;
}
/**
* Two ranges overlap when neither finishes before the other begins. Ranges
* that merely touch at a single boundary point (`a.end === b.start`) do
* NOT overlap — this matches tree-sitter's half-open-like range semantics
* and the typical "sibling blocks meet but don't overlap" pattern.
*/
function rangesOverlap(a: Range, b: Range): boolean {
const aEndsBeforeB =
a.endLine < b.startLine || (a.endLine === b.startLine && a.endCol <= b.startCol);
const bEndsBeforeA =
b.endLine < a.startLine || (b.endLine === a.startLine && b.endCol <= a.startCol);
return !(aEndsBeforeB || bEndsBeforeA);
}
function comparePosition(a: Range, b: Range): number {
if (a.startLine !== b.startLine) return a.startLine - b.startLine;
return a.startCol - b.startCol;
}
function formatRange(r: Range): string {
return `${r.startLine}:${r.startCol}-${r.endLine}:${r.endCol}`;
}
@@ -1,188 +0,0 @@
/**
* Shadow-mode aggregation — per-language parity %, per-evidence-kind
* breakdown of divergences. Consumed by the parity dashboard (RING2-PKG-5).
*
* Pure functions; no I/O. The harness persists per-run JSON; the dashboard
* reads `.gitnexus/shadow-parity/latest.json` and renders.
*
* Related types — `ShadowAgreement`, `ShadowCallsite`, `ShadowDiff` — are
* defined alongside `diffResolutions` in `./diff.ts` and re-exported
* through the top-level `gitnexus-shared` barrel. Consumers import all
* three from `gitnexus-shared`, not from this module.
*
* Part of RFC #909 Ring 2 SHARED — #918.
*/
import type { SupportedLanguages } from '../../languages.js';
import type { ResolutionEvidence } from '../types.js';
import type { ShadowAgreement, ShadowDiff } from './diff.js';
// ─── Aggregated report shape ────────────────────────────────────────────────
export interface LanguageParityRow {
readonly language: SupportedLanguages;
readonly totalCalls: number;
readonly bothAgree: number;
readonly onlyLegacy: number;
readonly onlyNew: number;
readonly bothDisagree: number;
readonly bothEmpty: number;
/**
* Fraction in [0, 1]. Numerator = `bothAgree`; denominator = "calls where
* at least one side resolved" = `totalCalls - bothEmpty`.
*
* When the denominator is 0 (all calls for this language were
* `both-empty`), returns 0. Callers rendering the dashboard should treat
* a 0 parity alongside `totalCalls === bothEmpty` as "no signal" rather
* than "total disagreement".
*/
readonly parity: number;
/**
* Divergence signals broken down by `ResolutionEvidence.kind`. Sourced
* from `ShadowDiff.evidenceDelta` on non-agreeing rows only — `both-agree`
* and `both-empty` do not contribute.
*/
readonly evidenceBreakdown: ReadonlyMap<ResolutionEvidence['kind'], number>;
}
export interface ShadowParityReport {
readonly generatedAt: string; // ISO 8601
readonly perLanguage: readonly LanguageParityRow[];
readonly overall: Omit<LanguageParityRow, 'language' | 'evidenceBreakdown'>;
}
// ─── Public API ─────────────────────────────────────────────────────────────
/**
* Aggregate a stream of `ShadowDiff` records into a `ShadowParityReport`,
* bucketed by language. Pure function.
*
* - `perLanguage` rows are sorted alphabetically by `SupportedLanguages`
* value for stable JSON output (the dashboard reads
* `.gitnexus/shadow-parity/latest.json` and diffing snapshots is useful).
* - `overall` is the column-wise sum across languages.
* - `generatedAt` is injected via the `now` parameter so tests stay
* deterministic; production callers let it default to `new Date()`.
*/
export function aggregateDiffs(
diffs: readonly { readonly language: SupportedLanguages; readonly diff: ShadowDiff }[],
now: Date = new Date(),
): ShadowParityReport {
const perLanguageMap = new Map<SupportedLanguages, MutableCounts>();
for (const { language, diff } of diffs) {
let counts = perLanguageMap.get(language);
if (!counts) {
counts = makeEmptyCounts();
perLanguageMap.set(language, counts);
}
tallyDiff(counts, diff);
}
const perLanguage: LanguageParityRow[] = Array.from(perLanguageMap.entries())
.map(([language, counts]) => buildRow(language, counts))
.sort((a, b) => a.language.localeCompare(b.language));
const overall = buildOverallRow(perLanguage);
return {
generatedAt: now.toISOString(),
perLanguage,
overall,
};
}
// ─── Internal helpers ───────────────────────────────────────────────────────
interface MutableCounts {
totalCalls: number;
bothAgree: number;
onlyLegacy: number;
onlyNew: number;
bothDisagree: number;
bothEmpty: number;
evidenceBreakdown: Map<ResolutionEvidence['kind'], number>;
}
function makeEmptyCounts(): MutableCounts {
return {
totalCalls: 0,
bothAgree: 0,
onlyLegacy: 0,
onlyNew: 0,
bothDisagree: 0,
bothEmpty: 0,
evidenceBreakdown: new Map(),
};
}
function tallyDiff(counts: MutableCounts, diff: ShadowDiff): void {
counts.totalCalls += 1;
incrementAgreement(counts, diff.agreement);
if (diff.agreement === 'both-agree' || diff.agreement === 'both-empty') return;
for (const ev of diff.evidenceDelta) {
counts.evidenceBreakdown.set(ev.kind, (counts.evidenceBreakdown.get(ev.kind) ?? 0) + 1);
}
}
function incrementAgreement(counts: MutableCounts, agreement: ShadowAgreement): void {
switch (agreement) {
case 'both-agree':
counts.bothAgree += 1;
return;
case 'only-legacy':
counts.onlyLegacy += 1;
return;
case 'only-new':
counts.onlyNew += 1;
return;
case 'both-disagree':
counts.bothDisagree += 1;
return;
case 'both-empty':
counts.bothEmpty += 1;
return;
}
}
function buildRow(language: SupportedLanguages, counts: MutableCounts): LanguageParityRow {
const resolved = counts.totalCalls - counts.bothEmpty;
const parity = resolved > 0 ? counts.bothAgree / resolved : 0;
return {
language,
totalCalls: counts.totalCalls,
bothAgree: counts.bothAgree,
onlyLegacy: counts.onlyLegacy,
onlyNew: counts.onlyNew,
bothDisagree: counts.bothDisagree,
bothEmpty: counts.bothEmpty,
parity,
// Freeze via `new Map` on a sorted-kind copy so downstream consumers
// can't mutate the aggregator's internal state.
evidenceBreakdown: new Map(
Array.from(counts.evidenceBreakdown.entries()).sort(([a], [b]) => a.localeCompare(b)),
),
};
}
function buildOverallRow(
perLanguage: readonly LanguageParityRow[],
): Omit<LanguageParityRow, 'language' | 'evidenceBreakdown'> {
let totalCalls = 0;
let bothAgree = 0;
let onlyLegacy = 0;
let onlyNew = 0;
let bothDisagree = 0;
let bothEmpty = 0;
for (const row of perLanguage) {
totalCalls += row.totalCalls;
bothAgree += row.bothAgree;
onlyLegacy += row.onlyLegacy;
onlyNew += row.onlyNew;
bothDisagree += row.bothDisagree;
bothEmpty += row.bothEmpty;
}
const resolved = totalCalls - bothEmpty;
const parity = resolved > 0 ? bothAgree / resolved : 0;
return { totalCalls, bothAgree, onlyLegacy, onlyNew, bothDisagree, bothEmpty, parity };
}
@@ -1,126 +0,0 @@
/**
* Shadow-mode diff logic — RFC §6.3.
*
* Pure comparison logic for shadow mode. Takes two `Resolution[]` (legacy
* DAG result + new scope-based registry result) and produces a structured
* diff record for the parity dashboard.
*
* Consumed by the Ring 2 PKG shadow harness (#923), which dual-runs each
* call through legacy + new paths, diffs results, and persists per-run JSON
* for the parity dashboard.
*
* Part of RFC #909 Ring 2 SHARED — #918.
*/
import type { Resolution, ResolutionEvidence } from '../types.js';
// ─── Diff record shape ──────────────────────────────────────────────────────
export type ShadowAgreement =
| 'both-agree' // top match identical (same DefId)
| 'only-legacy' // legacy resolved; new did not
| 'only-new' // new resolved; legacy did not
| 'both-disagree' // both resolved, but to different targets
| 'both-empty'; // both returned empty
export interface ShadowDiff {
readonly callsite: ShadowCallsite;
readonly legacy: Resolution | null;
readonly newResult: Resolution | null;
readonly agreement: ShadowAgreement;
/**
* Symmetric difference of the two top resolutions' `evidence` arrays,
* keyed on `ResolutionEvidence.kind`.
*
* - For `'both-agree'` and `'both-empty'` agreements, always empty.
* - For `'both-disagree'`, contains evidence kinds present on exactly one
* side (not in both).
* - For `'only-legacy'`, contains all of legacy's top evidence.
* - For `'only-new'`, contains all of new's top evidence.
*/
readonly evidenceDelta: readonly ResolutionEvidence[];
}
export interface ShadowCallsite {
readonly filePath: string;
readonly line: number;
readonly col: number;
readonly calledName: string;
}
// ─── Public API ─────────────────────────────────────────────────────────────
/**
* Compare two `Resolution[]` arrays (top matches at `[0]`) and produce a
* `ShadowDiff`. Pure function.
*
* Agreement rules:
* - both arrays empty → `'both-empty'`, `evidenceDelta: []`
* - legacy empty, new non-empty → `'only-new'`, `evidenceDelta` = new's top evidence
* - legacy non-empty, new empty → `'only-legacy'`, `evidenceDelta` = legacy's top evidence
* - both non-empty, same top `def.nodeId` → `'both-agree'`, `evidenceDelta: []`
* - both non-empty, different top `def.nodeId` → `'both-disagree'`,
* `evidenceDelta` = symmetric difference by `ResolutionEvidence.kind`
* (first occurrence of a kind-only-on-legacy then kind-only-on-new; order
* preserved from input arrays)
*
* Evidence-delta rationale: callers aggregating divergences want to know
* which signal kinds explain a disagreement. Keying on `kind` (not full
* equality over `weight`/`note`) avoids spurious deltas when the same
* signal fires with slightly different calibration weights on each side.
*/
export function diffResolutions(
callsite: ShadowCallsite,
legacy: readonly Resolution[],
newResult: readonly Resolution[],
): ShadowDiff {
const legacyTop: Resolution | null = legacy.length > 0 ? legacy[0] : null;
const newTop: Resolution | null = newResult.length > 0 ? newResult[0] : null;
const agreement: ShadowAgreement = (() => {
if (legacyTop === null && newTop === null) return 'both-empty';
if (legacyTop === null) return 'only-new';
if (newTop === null) return 'only-legacy';
return legacyTop.def.nodeId === newTop.def.nodeId ? 'both-agree' : 'both-disagree';
})();
const evidenceDelta = computeEvidenceDelta(legacyTop, newTop, agreement);
return {
callsite,
legacy: legacyTop,
newResult: newTop,
agreement,
evidenceDelta,
};
}
// ─── Internal helpers ───────────────────────────────────────────────────────
/**
* Symmetric difference of two evidence arrays, keyed on
* `ResolutionEvidence.kind`. Preserves input order: legacy-only signals
* first (in legacy's original order), then new-only signals (in new's order).
*
* For `'both-agree'` / `'both-empty'` the delta is empty by contract. For
* `'only-legacy'` / `'only-new'` one side's evidence is the delta (nothing to
* subtract against).
*/
function computeEvidenceDelta(
legacy: Resolution | null,
newResult: Resolution | null,
agreement: ShadowAgreement,
): readonly ResolutionEvidence[] {
if (agreement === 'both-agree' || agreement === 'both-empty') return [];
if (agreement === 'only-legacy') return legacy!.evidence;
if (agreement === 'only-new') return newResult!.evidence;
// both-disagree: symmetric difference keyed on `kind`
const legacyKinds = new Set(legacy!.evidence.map((e) => e.kind));
const newKinds = new Set(newResult!.evidence.map((e) => e.kind));
const onlyInLegacy = legacy!.evidence.filter((e) => !newKinds.has(e.kind));
const onlyInNew = newResult!.evidence.filter((e) => !legacyKinds.has(e.kind));
return [...onlyInLegacy, ...onlyInNew];
}
@@ -1,35 +0,0 @@
/**
* `SymbolDefinition` — the canonical shape of an indexed symbol record.
*
* Historically defined in `gitnexus/src/core/ingestion/model/symbol-table.ts`;
* moved into `gitnexus-shared` as part of RFC #909 Ring 1 (#910) so the
* scope-resolution types that reference it can live in the shared package
* alongside their consumers (`gitnexus/` and `gitnexus-web/`).
*
* Shape is unchanged from the prior local definition.
*/
import type { NodeLabel } from '../graph/types.js';
export interface SymbolDefinition {
nodeId: string;
filePath: string;
type: NodeLabel;
/** Canonical dot-separated qualified type name for class-like symbols
* (e.g. `App.Models.User`). Falls back to the simple symbol name when no
* package/namespace/module scope exists or no explicit qualified metadata is provided. */
qualifiedName?: string;
parameterCount?: number;
/** Number of required (non-optional, non-default) parameters.
* Enables range-based arity filtering: argCount >= requiredParameterCount && argCount <= parameterCount. */
requiredParameterCount?: number;
/** Per-parameter type names for overload disambiguation (e.g. ['int', 'String']).
* Populated when parameter types are resolvable from AST (any typed language). */
parameterTypes?: string[];
/** Raw return type text extracted from AST (e.g. 'User', 'Promise<User>') */
returnType?: string;
/** Declared type for non-callable symbols — fields/properties (e.g. 'Address', 'List<User>') */
declaredType?: string;
/** Links Method/Constructor/Property to owning Class/Struct/Trait nodeId */
ownerId?: string;
}
@@ -1,478 +0,0 @@
/**
* Scope-resolution type definitions — RFC §2 data model (authoritative source).
*
* See: https://www.notion.so/346dc50b6ed281cfaacbe480bf231d50
*
* Anti-drift rule: every type, interface, and enum defined here is the single
* source of truth. Later code that references these names must import them
* from `gitnexus-shared`; it must not re-define them locally.
*
* Lifecycle contract (RFC §2.8): scopes are **constructed during extraction,
* linked during finalize, immutable after finalize**. All fields are
* `readonly` at the type level; `Object.freeze` is applied at runtime in dev
* builds.
*
* Two structures are populated after freeze:
* 1. `ReferenceIndex` — by resolution, before emission.
* 2. `ScopeResolutionIndexes.bindingAugmentations` — the dedicated
* append-only post-finalize binding channel (e.g. C# same-namespace
* cross-file fanout). The companion `indexes.bindings` is the
* finalize-output channel and is deep-frozen by `materializeBindings`;
* walkers consult both via `lookupBindingsAt`. See `ScopeResolver`
* Invariant I8 for the full lifecycle contract.
*/
import type { NodeLabel } from '../graph/types.js';
import type { SymbolDefinition } from './symbol-definition.js';
// ─── §2.1 Type aliases ──────────────────────────────────────────────────────
/** Stable per-(file, range, kind) scope identifier; interned for identity-fast equality. */
export type ScopeId = string;
/** Stable symbol-definition identifier (graph nodeId). */
export type DefId = string;
/** Kinds of lexical scope a `Scope` node can represent. */
export type ScopeKind =
| 'Module' // file root
| 'Namespace' // C++ namespace, C# namespace, Kotlin package-object, Rust mod
| 'Class' // class/struct/trait/interface body
| 'Function' // function/method/closure/lambda body
| 'Block' // { ... }, if-body, for-body, with-body, match arms
| 'Expression'; // comprehensions, for-init, pattern bindings, lambda param lists
// ─── Range + Capture (parser-agnostic) ──────────────────────────────────────
/** Source-text range. 1-based `startLine`/`endLine`; 0-based `startCol`/`endCol`. */
export interface Range {
readonly startLine: number;
readonly startCol: number;
readonly endLine: number;
readonly endCol: number;
}
/**
* Tagged capture emitted by a LanguageProvider's `emitScopeCaptures` hook.
*
* Parser-agnostic: tree-sitter queries and COBOL's regex tagger both produce
* `Capture[]`. The central `ScopeExtractor` consumes captures without
* knowing which parser produced them.
*/
export interface Capture {
/** Capture name, including leading `@` (e.g., `'@scope.module'`, `'@declaration.class'`). */
readonly name: string;
readonly range: Range;
/** The captured source text. */
readonly text: string;
}
/**
* A grouping of `Capture`s that came from a single query match (e.g., one
* `@import.statement` match carries `@import.source`, `@import.name`,
* `@import.alias?` as child captures). Keyed by capture name for O(1)
* child access.
*/
export type CaptureMatch = Readonly<Record<string, Capture>>;
// ─── Hook input/output types (RFC §5.2) ─────────────────────────────────────
/**
* Provider-interpreted raw import, consumed by finalize (Phase 2) to produce
* linked `ImportEdge[]`. The provider's `interpretImport` hook turns a
* `CaptureMatch` for an `@import.statement` into one of these; the central
* finalize algorithm resolves `targetRaw` to a concrete file via
* `resolveImportTarget` and materializes the final `ImportEdge`.
*
* Discriminated union — each variant carries only the fields that make sense
* for its kind. Invalid shapes (e.g., a `namespace` import with an alias-like
* `importedName` mismatch) are compile errors, not latent bugs. `'wildcard-
* expanded'` is deliberately NOT a variant: that kind is finalize output only,
* produced when `expandsWildcardTo` materializes a wildcard against target
* exports — a provider must never emit it at parse time.
*/
export type ParsedImport =
/**
* Per-name import without rename.
*
* Examples:
* - Python `from foo import X` → `{ kind: 'named', localName: 'X', importedName: 'X', targetRaw: 'foo' }`
* - TS `import { X } from './foo'` → `{ kind: 'named', localName: 'X', importedName: 'X', targetRaw: './foo' }`
* - Java `import foo.bar.X` → `{ kind: 'named', localName: 'X', importedName: 'X', targetRaw: 'foo.bar' }`
*/
| {
readonly kind: 'named';
readonly localName: string;
readonly importedName: string;
readonly targetRaw: string;
}
/**
* Per-name import with rename.
*
* Examples:
* - Python `from foo import X as Y` → `{ kind: 'alias', localName: 'Y', importedName: 'X', alias: 'Y', targetRaw: 'foo' }`
* - TS `import { X as Y } from './foo'` → `{ kind: 'alias', localName: 'Y', importedName: 'X', alias: 'Y', targetRaw: './foo' }`
*/
| {
readonly kind: 'alias';
readonly localName: string;
readonly importedName: string;
readonly alias: string;
readonly targetRaw: string;
}
/**
* Qualified module handle, with or without rename. `importedName` is the
* module being aliased; `localName` is the scope-visible handle (often the
* same unless renamed).
*
* Examples:
* - Python `import numpy` → `{ kind: 'namespace', localName: 'numpy', importedName: 'numpy', targetRaw: 'numpy' }`
* - Python `import numpy as np` → `{ kind: 'namespace', localName: 'np', importedName: 'numpy', targetRaw: 'numpy' }`
* - TS `import * as np from 'numpy'` → `{ kind: 'namespace', localName: 'np', importedName: 'numpy', targetRaw: 'numpy' }`
* - Go `import foo "pkg/bar"` → `{ kind: 'namespace', localName: 'foo', importedName: 'bar', targetRaw: 'pkg/bar' }`
*/
| {
readonly kind: 'namespace';
/** Scope-visible handle (e.g. `np` in `import numpy as np`; `numpy` when unaliased). */
readonly localName: string;
/** Module being aliased (e.g. `numpy` in `import numpy as np`). */
readonly importedName: string;
readonly targetRaw: string;
}
/**
* Syntactically-detectable parse-time re-export. Finalize may still produce
* `ImportEdge { kind: 'reexport', transitiveVia }` when flattening chains;
* this variant preserves the *parse-time* signal so finalize doesn't have
* to re-derive it from scratch.
*
* Examples:
* - TS `export { X } from './y'` → `{ kind: 'reexport', localName: 'X', importedName: 'X', targetRaw: './y' }`
* - TS `export { X as Y } from './y'` → `{ kind: 'reexport', localName: 'Y', importedName: 'X', alias: 'Y', targetRaw: './y' }`
* - Rust `pub use foo::bar` → `{ kind: 'reexport', localName: 'bar', importedName: 'bar', targetRaw: 'foo' }`
*/
| {
readonly kind: 'reexport';
/** Name as re-exported in the current module. */
readonly localName: string;
/** Name in the source module. */
readonly importedName: string;
readonly targetRaw: string;
/** Set when the re-export renames the symbol (e.g. `export { X as Y } from './y'`). */
readonly alias?: string;
}
/**
* Wildcard import — brings every exported name from the target module into
* the importing scope. The finalize algorithm expands this into one
* `BindingRef` per exported name via the provider's `expandsWildcardTo`
* hook, producing the finalize-only `ImportEdge` kind `'wildcard-expanded'`.
*
* Examples:
* - Python `from foo import *` → `{ kind: 'wildcard', targetRaw: 'foo' }`
* - JS `export * from './foo'` → `{ kind: 'wildcard', targetRaw: './foo' }`
* - Rust `pub use foo::*` → `{ kind: 'wildcard', targetRaw: 'foo' }`
*/
| {
readonly kind: 'wildcard';
readonly targetRaw: string;
}
/**
* Runtime-computed target — the import path is not a static literal at
* parse time. Providers SHOULD emit the unresolvable expression's source
* text as `targetRaw` to aid diagnostics; `null` only when no string form
* exists.
*
* Examples:
* - JS `await import(expr)` → `{ kind: 'dynamic-unresolved', localName: '', targetRaw: 'expr' }`
* - Python `importlib.import_module(f'pkg.{name}')` → `{ kind: 'dynamic-unresolved', localName: '', targetRaw: "f'pkg.{name}'" }`
*/
| {
readonly kind: 'dynamic-unresolved';
readonly localName: string;
/** Source text of the unresolved expression when available; `null` otherwise. */
readonly targetRaw: string | null;
}
/**
* Lazy / dynamic import whose target IS a static string literal at parse
* time, so it can be linked to a concrete `targetFile`. No local name
* binding is materialized — `import('./m')` returns `Promise<Module>` and
* any consumer-visible names appear via subsequent `.then(({ X }) => …)`
* destructuring, which is outside the static-import surface. The edge
* exists for module-reachability and impact analysis (so editing `./m`
* still flags the dynamic importer as affected).
*
* Providers MUST only emit this kind when `targetRaw` is a literal
* string they can hand to `resolveImportTarget`; expression arguments
* stay `dynamic-unresolved`.
*
* Examples:
* - JS `import('./feature')` → `{ kind: 'dynamic-resolved', targetRaw: './feature' }`
* - JS `await import('@scope/pkg/sub')` → `{ kind: 'dynamic-resolved', targetRaw: '@scope/pkg/sub' }`
*/
| {
readonly kind: 'dynamic-resolved';
readonly targetRaw: string;
}
/**
* Bare-source / side-effect import that introduces no local name binding
* but still establishes a file-level dependency. Resolves to a concrete
* `targetFile` via `resolveImportTarget` and produces a file→file
* `ImportEdge` for module-reachability and impact analysis, with no
* `BindingRef` materialized.
*
* Examples:
* - JS / TS `import './polyfill'` → `{ kind: 'side-effect', targetRaw: './polyfill' }`
* - Rust `use foo::bar as _` → side-effect (binding hidden under `_`)
*/
| {
readonly kind: 'side-effect';
readonly targetRaw: string;
};
/**
* Provider-interpreted type binding. The provider's `interpretTypeBinding`
* hook turns a `CaptureMatch` (e.g., `@type-binding.parameter`) into one of
* these; the central extractor attaches the resulting `TypeRef` to the
* appropriate scope's `typeBindings` map.
*/
export interface ParsedTypeBinding {
/** The name being bound (parameter name, `self`, assignment LHS, …). */
readonly boundName: string;
/** The raw type name as written in source (`'User'`, `'models.User'`, …). */
readonly rawTypeName: string;
readonly source: TypeRef['source'];
}
/**
* Cross-file workspace index consumed by finalize-phase hooks
* (`resolveImportTarget`, `expandsWildcardTo`). Opaque placeholder in Ring 1;
* concretely typed in Ring 2 SHARED (#915).
*/
export type WorkspaceIndex = unknown;
// `ScopeTree` is exported from `./scope-tree.js` as of Ring 2 SHARED (#912).
// The former opaque placeholder lived here during Ring 1; removed now that
// the concrete type exists. Consumers import from `gitnexus-shared` directly.
/**
* Minimal scope-lookup contract: map a `ScopeId` back to its `Scope` record.
*
* Lives in the data-model layer so both `ScopeTree` (§3.1) and
* `resolveTypeRef` / `Registry.lookup` (§4) can depend on it without
* inverting each other. `ScopeTree` is the canonical implementation;
* tests and future alternative containers may supply their own.
*/
export interface ScopeLookup {
getScope(id: ScopeId): Scope | undefined;
}
/** Call-site description passed to `arityCompatibility`. */
export interface Callsite {
/** Number of arguments at the call site. */
readonly arity: number;
}
// ─── §2.4 ImportEdge ────────────────────────────────────────────────────────
/**
* A cross-file import edge attached to a module/namespace scope.
*
* Raw (unlinked) edges are emitted during parse (Phase 1); `targetModuleScope`
* and `targetDefId` are filled in during finalize (Phase 2) via SCC-aware
* bounded-fixpoint linking (RFC §3.2).
*/
export interface ImportEdge {
/** How this scope sees the imported name (after alias). */
readonly localName: string;
/** Exporting file; `null` only when `kind === 'dynamic-unresolved'`. */
readonly targetFile: string | null;
/** The name under which the target exports this symbol. */
readonly targetExportedName: string;
/** Pre-resolved at finalize: the module scope of the exporting file. */
readonly targetModuleScope?: ScopeId;
/** Pre-resolved at finalize: the exported symbol's `DefId`. */
readonly targetDefId?: DefId;
readonly kind:
| 'named'
| 'alias'
| 'namespace'
| 'wildcard-expanded'
| 'reexport'
| 'dynamic-unresolved'
| 'dynamic-resolved'
| 'side-effect';
/** Re-export chain, for provenance (e.g., `['./y']` when re-exported via `./y`). */
readonly transitiveVia?: readonly string[];
/** Set to `'unresolved'` when the SCC fixpoint could not link this edge. */
readonly linkStatus?: 'unresolved';
}
// ─── §2.3 BindingRef ────────────────────────────────────────────────────────
/**
* A name binding visible at a scope, with provenance.
*
* Provenance stays at the visibility layer — a name being visible because it
* is local vs imported vs wildcard-expanded vs re-exported is a property of
* the binding itself. This keeps evidence emission and `import-use` reference
* stamping first-class instead of reconstructing provenance from a side table.
*/
export interface BindingRef {
readonly def: SymbolDefinition;
readonly origin: 'local' | 'import' | 'namespace' | 'wildcard' | 'reexport';
/** Non-null for non-local origins; carries the `ImportEdge` that brought the name into this scope. */
readonly via?: ImportEdge;
}
// ─── §2.5 TypeRef ───────────────────────────────────────────────────────────
/**
* A reference to a named type, anchored at its declaration site.
*
* Design choice: raw name + declaration-site scope, resolved at lookup time.
* Pre-resolution would invert the extraction/resolution wall. Deferred thunks
* add no capability. Structured type systems are months of work per language.
* This shape keeps V1 tractable while preserving correctness for aliases,
* re-exports, and nested modules. Generics deferred to V2 via `typeArgs`.
*/
export interface TypeRef {
/** The name as written in source (e.g., `'User'`, `'models.User'`, `'List'`). */
readonly rawName: string;
/** Anchor for resolving `rawName` — the scope where the annotation/inference was written. */
readonly declaredAtScope: ScopeId;
readonly source:
| 'annotation'
| 'parameter-annotation'
| 'return-annotation'
| 'self'
| 'assignment-inferred'
| 'constructor-inferred'
| 'receiver-propagated';
/** Reserved for V2+: generic type arguments (`List<User>` → `[TypeRef('User')]`). V1 ignores. */
readonly typeArgs?: readonly TypeRef[];
}
// ─── §2.2 Scope ─────────────────────────────────────────────────────────────
/**
* The canonical lexical-scope node. Forms the spine of the SemanticModel.
*
* ScopeId shape (RFC §2.2): `scope:{filePath}#{startLine}:{startCol}-{endLine}:{endCol}:{kind}`
* — deterministic, stable across reparses of the same source, interned.
*/
export interface Scope {
readonly id: ScopeId;
readonly parent: ScopeId | null;
readonly kind: ScopeKind;
readonly range: Range;
readonly filePath: string;
/** Names visible from this scope. Provenance preserved via `BindingRef.origin`. */
readonly bindings: ReadonlyMap<string, readonly BindingRef[]>;
/** Defs structurally owned by this scope (e.g., methods owned by a class body scope). */
readonly ownedDefs: readonly SymbolDefinition[];
/** Import edges attached to this scope. Mostly module/namespace scopes, but some
* languages allow local imports (Python `def f(): from x import Y`, Rust
* fn-local `use`, TS dynamic `import()`). */
readonly imports: readonly ImportEdge[];
/** Local type facts visible from this scope (parameter annotations, `self` binding, etc.). */
readonly typeBindings: ReadonlyMap<string, TypeRef>;
}
// ─── §2.6 Resolution + ResolutionEvidence ───────────────────────────────────
/**
* One piece of evidence for a `Resolution`. Multiple signals corroborate a
* single match; their weights compose additively to produce `confidence`.
*
* Weights come from `EvidenceWeights` (see `./evidence-weights.ts`).
*/
export interface ResolutionEvidence {
readonly kind:
| 'local'
| 'scope-chain'
| 'import'
| 'type-binding'
| 'owner-match'
| 'kind-match'
| 'arity-match'
| 'global-name'
| 'global-qualified'
| 'dynamic-import-unresolved';
/** Signal weight, sourced from `EvidenceWeights`. Additive; sum capped at 1.0. */
readonly weight: number;
/** Optional debug annotation (e.g., `'matched via self: User'`). */
readonly note?: string;
}
/**
* A ranked resolution candidate returned by `ClassRegistry.lookup` /
* `MethodRegistry.lookup` / `FieldRegistry.lookup`. Evidence composes
* additively; callers read `[0]` for the one-shot answer or inspect the
* evidence trace for debugging.
*/
export interface Resolution {
readonly def: SymbolDefinition;
/** Σ of `evidence[].weight`, capped at 1.0. */
readonly confidence: number;
readonly evidence: readonly ResolutionEvidence[];
/** Optional debug trace: scopes walked to reach `def`. */
readonly path?: readonly ScopeId[];
}
// ─── §2.7 Reference + ReferenceIndex ────────────────────────────────────────
/**
* A post-resolution usage fact: some code at `atRange` inside `fromScope`
* references `toDef` with the given confidence/evidence. Materialized by the
* resolution phase; emitted as graph edges (`CALLS`/`READS`/`WRITES`/etc.)
* during the emit phase.
*/
export interface Reference {
/** Innermost lexical scope containing `atRange`. */
readonly fromScope: ScopeId;
readonly toDef: DefId;
/** Location of the reference in source. */
readonly atRange: Range;
readonly kind: 'call' | 'read' | 'write' | 'type-reference' | 'inherits' | 'import-use';
readonly confidence: number;
readonly evidence: readonly ResolutionEvidence[];
}
/**
* Two-way index over `Reference` records, populated during the resolution
* phase. Scopes stay immutable after finalize; references accumulate here.
*/
export interface ReferenceIndex {
readonly bySourceScope: ReadonlyMap<ScopeId, readonly Reference[]>;
readonly byTargetDef: ReadonlyMap<DefId, readonly Reference[]>;
}
// ─── §4.1 LookupParams ──────────────────────────────────────────────────────
/**
* Opaque placeholder for the per-kind registry passed as the owner-scoped
* contributor. Typed concretely in Ring 2 SHARED (#917); kept as `unknown`
* here so Ring 1 can ship without pulling in the registry implementation.
*/
export type RegistryContributor = unknown;
/**
* Parameters accepted by `Registry.lookup`. Three registries (Class/Method/
* Field) run the same 7-step algorithm with different parameter tuples; see
* RFC §4.4 for per-registry specializations.
*/
export interface LookupParams {
readonly acceptedKinds: readonly NodeLabel[];
/** Class lookups: false. Method/Field lookups: true. */
readonly useReceiverTypeBinding: boolean;
readonly ownerScopedContributor: RegistryContributor | null;
/** Optional arity hint fed to `provider.arityCompatibility`. */
readonly arityHint?: number;
/** Explicit receiver name (e.g., `'user'` in `user.save()`). When present,
* the receiver's type binding at the callsite scope is used; otherwise
* the enclosing method's implicit `self`/`this` is consulted. See §4.1. */
readonly explicitReceiver?: { readonly name: string };
}
+4 -23
View File
@@ -61,22 +61,13 @@ test.beforeAll(async () => {
}
});
// Auto-connect downloads the full graph from the backend; under parallel
// workers in CI the same backend serves multiple downloads concurrently, so
// reaching the "Ready" state can take noticeably longer than a single-worker
// run. Match the 45s budget used by waitForGraphLoaded() in
// server-connect.spec.ts which has been stable on the same backend.
const READY_TIMEOUT_MS = 45_000;
test.describe('Multi-Repo Scoping', () => {
test('auto-connect via ?server= sets ?project= in URL', async ({ page }) => {
// Navigate with ?server= param (the bookmarkable shortcut)
await page.goto(`/?server=${encodeURIComponent(BACKEND_URL)}`);
// Wait for graph to load
await expect(page.locator('[data-testid="status-ready"]')).toBeVisible({
timeout: READY_TIMEOUT_MS,
});
await expect(page.locator('[data-testid="status-ready"]')).toBeVisible({ timeout: 30_000 });
// URL should now contain ?project= with the repo name
const url = new URL(page.url());
@@ -86,14 +77,8 @@ test.describe('Multi-Repo Scoping', () => {
});
test('?server= is preserved in URL for F5 recovery', async ({ page }) => {
// Two sequential auto-connects (initial + reload), each up to READY_TIMEOUT_MS,
// can exceed the default 60s test timeout under parallel workers.
test.slow();
await page.goto(`/?server=${encodeURIComponent(BACKEND_URL)}`);
await expect(page.locator('[data-testid="status-ready"]')).toBeVisible({
timeout: READY_TIMEOUT_MS,
});
await expect(page.locator('[data-testid="status-ready"]')).toBeVisible({ timeout: 30_000 });
// URL should still have ?server=
const url = new URL(page.url());
@@ -101,16 +86,12 @@ test.describe('Multi-Repo Scoping', () => {
// F5 should reconnect (not show onboarding)
await page.reload();
await expect(page.locator('[data-testid="status-ready"]')).toBeVisible({
timeout: READY_TIMEOUT_MS,
});
await expect(page.locator('[data-testid="status-ready"]')).toBeVisible({ timeout: 30_000 });
});
test('node count in status bar matches backend data', async ({ page }) => {
await page.goto(`/?server=${encodeURIComponent(BACKEND_URL)}`);
await expect(page.locator('[data-testid="status-ready"]')).toBeVisible({
timeout: READY_TIMEOUT_MS,
});
await expect(page.locator('[data-testid="status-ready"]')).toBeVisible({ timeout: 30_000 });
// Fetch expected node count from backend
const res = await fetch(`${BACKEND_URL}/api/repo?repo=${encodeURIComponent(firstRepoName)}`);
+1 -8
View File
@@ -26,10 +26,7 @@ async function enterExploringView(page: import('@playwright/test').Page) {
// Landing screen may not appear (e.g. ?server auto-connect)
}
// Match the 45s budget used by waitForGraphLoaded() in
// server-connect.spec.ts; under parallel CI workers, downloading the full
// graph can occasionally exceed 30s.
await expect(page.locator('[data-testid="status-ready"]')).toBeVisible({ timeout: 45_000 });
await expect(page.locator('[data-testid="status-ready"]')).toBeVisible({ timeout: 30_000 });
}
// ── Flow 1: Onboarding (no server running) ─────────────────────────────────
@@ -247,10 +244,6 @@ test.describe('Flow 3: Analyze form', () => {
test.describe('Flow 4: Repo dropdown in exploring view', () => {
const SKIP_MSG = 'Requires running gitnexus server with indexed repos';
// enterExploringView() can take up to ~45s under parallel CI workers; combined
// with the dropdown interactions this can exceed the default 60s test budget.
test.slow();
test.beforeAll(async () => {
if (process.env.E2E) return;
try {
+5 -26
View File
@@ -84,20 +84,11 @@ test.describe('Hold-queue timeout error', () => {
// ── 2. ?project= URL persistence ─────────────────────────────────────────────
// Auto-connect downloads the full graph from the backend; under parallel
// workers in CI the same backend serves multiple downloads concurrently, so
// reaching the "Ready" state can take noticeably longer than a single-worker
// run. Match the 45s budget used by waitForGraphLoaded() in
// server-connect.spec.ts which has been stable on the same backend.
const READY_TIMEOUT_MS = 45_000;
test.describe('?project= URL persistence', () => {
test('?project= is set in URL after connecting via ?server=', async ({ page }) => {
await page.goto(`/?server=${encodeURIComponent(BACKEND_URL)}`);
await expect(page.locator('[data-testid="status-ready"]')).toBeVisible({
timeout: READY_TIMEOUT_MS,
});
await expect(page.locator('[data-testid="status-ready"]')).toBeVisible({ timeout: 30_000 });
const url = new URL(page.url());
const project = url.searchParams.get('project');
@@ -107,20 +98,12 @@ test.describe('?project= URL persistence', () => {
});
test('?project= is still present after F5 reload', async ({ page }) => {
// Two sequential auto-connects (initial + reload), each up to READY_TIMEOUT_MS,
// can exceed the default 60s test timeout under parallel workers.
test.slow();
await page.goto(`/?server=${encodeURIComponent(BACKEND_URL)}`);
await expect(page.locator('[data-testid="status-ready"]')).toBeVisible({
timeout: READY_TIMEOUT_MS,
});
await expect(page.locator('[data-testid="status-ready"]')).toBeVisible({ timeout: 30_000 });
// After connect, URL has ?server=&project= — F5 re-uses both params
await page.reload();
await expect(page.locator('[data-testid="status-ready"]')).toBeVisible({
timeout: READY_TIMEOUT_MS,
});
await expect(page.locator('[data-testid="status-ready"]')).toBeVisible({ timeout: 30_000 });
const url = new URL(page.url());
expect(url.searchParams.get('project')).toBeTruthy();
@@ -139,9 +122,7 @@ test.describe('?project= auto-connect', () => {
`/?server=${encodeURIComponent(BACKEND_URL)}&project=${encodeURIComponent(firstRepoName)}`,
);
await expect(page.locator('[data-testid="status-ready"]')).toBeVisible({
timeout: READY_TIMEOUT_MS,
});
await expect(page.locator('[data-testid="status-ready"]')).toBeVisible({ timeout: 30_000 });
// ?project= in URL should match what we passed in
const url = new URL(page.url());
@@ -174,9 +155,7 @@ test.describe('Windows path normalization', () => {
await page.goto(`/?server=${encodeURIComponent(BACKEND_URL)}`);
await expect(page.locator('[data-testid="status-ready"]')).toBeVisible({
timeout: READY_TIMEOUT_MS,
});
await expect(page.locator('[data-testid="status-ready"]')).toBeVisible({ timeout: 30_000 });
// URL ?project= must be the short basename, NOT the full Windows path
const url = new URL(page.url());
+2337 -908
View File
File diff suppressed because it is too large Load Diff
+19 -19
View File
@@ -3,7 +3,7 @@
"private": true,
"version": "0.0.0",
"engines": {
"node": "^20.19.0 || >=22.12.0"
"node": ">=20.0.0"
},
"type": "module",
"scripts": {
@@ -19,14 +19,14 @@
},
"dependencies": {
"gitnexus-shared": "file:../gitnexus-shared",
"@langchain/anthropic": "^1.3.27",
"@langchain/core": "^1.1.41",
"@langchain/google-genai": "^2.1.28",
"@langchain/langgraph": "^1.2.9",
"@langchain/ollama": "^1.2.6",
"@langchain/openai": "^1.4.4",
"@langchain/anthropic": "^1.3.10",
"@langchain/core": "^1.1.15",
"@langchain/google-genai": "^2.1.10",
"@langchain/langgraph": "^1.1.0",
"@langchain/ollama": "^1.2.0",
"@langchain/openai": "^1.2.2",
"@sigma/edge-curve": "^3.1.0",
"@tailwindcss/vite": "^4.2.4",
"@tailwindcss/vite": "^4.1.18",
"axios": "^1.13.2",
"d3": "^7.9.0",
"dompurify": "^3.3.3",
@@ -36,10 +36,10 @@
"graphology-layout-forceatlas2": "^0.10.1",
"graphology-layout-noverlap": "^0.4.2",
"graphology-utils": "^2.3.0",
"langchain": "^1.3.4",
"langchain": "^1.2.10",
"lru-cache": "^11.2.4",
"lucide-react": "^1.11.0",
"mermaid": "^11.14.0",
"lucide-react": "^0.562.0",
"mermaid": "^11.12.2",
"mnemonist": "^0.39.0",
"pandemonium": "^2.4.0",
"react": "^18.3.1",
@@ -49,12 +49,12 @@
"react-zoom-pan-pinch": "^3.7.0",
"remark-gfm": "^4.0.1",
"sigma": "^3.0.2",
"tailwindcss": "^4.2.4",
"tailwindcss": "^4.1.18",
"uuid": "^13.0.0",
"zod": "^3.25.76"
},
"devDependencies": {
"@babel/types": "^7.29.0",
"@babel/types": "^7.28.5",
"@playwright/test": "^1.58.2",
"@testing-library/jest-dom": "^6.9.1",
"@testing-library/react": "^16.3.2",
@@ -65,13 +65,13 @@
"@types/react-dom": "^18.3.0",
"@types/react-syntax-highlighter": "^15.5.13",
"@vercel/node": "^5.5.16",
"@vitejs/plugin-react": "^5.1.4",
"@vitest/coverage-v8": "^4.1.5",
"jsdom": "^29.0.2",
"@vitejs/plugin-react": "^5.1.0",
"@vitest/coverage-v8": "^3.2.4",
"jsdom": "^29.0.0",
"tree-sitter-wasms": "^0.1.13",
"typescript": "^5.4.5",
"vite": "^8.0.10",
"vitest": "^4.1.5",
"wait-on": "^9.0.5"
"vite": "^5.2.0",
"vitest": "^3.2.4",
"wait-on": "^8.0.5"
}
}
+1 -95
View File
@@ -5,42 +5,7 @@
* than directly from lucide-react. This provides a single place to manage
* which icons are used and allows future optimization (e.g., tree-shaking
* configuration, icon subset bundling) without touching every component.
*
* --- Why a local `Github` icon? ---
*
* Lucide removed all brand icons in v1
* (https://lucide.dev/guide/react/migration,
* https://github.com/lucide-icons/lucide/blob/main/BRAND_LOGOS_STATEMENT.md),
* so `import { Github } from 'lucide-react'` no longer compiles.
*
* We replace it with the official mark from Primer Octicons — the icon set
* GitHub itself uses on github.com — copied verbatim. This was preferred over
* the alternatives because it:
*
* 1. Is the canonical GitHub-maintained source for the mark, kept in sync
* with what users see on github.com.
* 2. Is MIT-licensed (Copyright (c) GitHub Inc.,
* https://github.com/primer/octicons/blob/main/LICENSE), so embedding the
* path data is permitted.
* 3. Adds zero new npm dependencies (vs. `@primer/octicons-react`,
* `react-icons`, or `simple-icons`), keeping the web bundle lean.
* 4. Ships per-size hand-tuned glyphs (16 + 24) — the same approach Primer
* uses — so the mark stays crisp at the small `h-4 w-4` sites in the
* header and onboarding screens as well as at full size.
*
* Trademark note: the GitHub mark is a registered trademark of GitHub, Inc.
* (https://brand.github.com/foundations/logo). The MIT license covers our
* right to copy the SVG; trademark rules still govern *use*. We use the mark
* here only to link to GitHub and to indicate GitHub source-repo integration,
* which are explicitly permitted by GitHub's brand toolkit.
*
* SVG sources (copied verbatim, MIT, Copyright (c) GitHub Inc.):
* - https://github.com/primer/octicons/blob/main/icons/mark-github-16.svg
* - https://github.com/primer/octicons/blob/main/icons/mark-github-24.svg
*/
import { forwardRef } from 'react';
import type { LucideProps } from 'lucide-react';
export {
AlertCircle,
AlertTriangle,
@@ -66,6 +31,7 @@ export {
Folder,
FolderOpen,
GitBranch,
Github,
Globe,
Hash,
Heart,
@@ -109,63 +75,3 @@ export {
ZoomIn,
ZoomOut,
} from 'lucide-react';
/**
* GitHub mark — local copy of Primer Octicons `mark-github-{16,24}`.
*
* Why this exists, why Primer Octicons, and the trademark caveat are documented
* at the top of this file. Please read that header before changing the SVG
* paths or swapping the source.
*
* API-compatible with `lucide-react` icons (`LucideProps`). The Octicons mark
* is a *filled* glyph, so the lucide-only stroke props (`strokeWidth`,
* `absoluteStrokeWidth`) are accepted for type parity but ignored. Color
* defaults to `currentColor`, so Tailwind `text-*` utilities work the same as
* with any other icon in this module.
*/
export const Github = forwardRef<SVGSVGElement, LucideProps>(function Github(
{
size = 24,
color = 'currentColor',
className,
strokeWidth: _strokeWidth,
absoluteStrokeWidth: _absoluteStrokeWidth,
...rest
},
ref,
) {
const numericSize = typeof size === 'string' ? Number.parseFloat(size) : size;
const useSmallVariant = Number.isFinite(numericSize) && (numericSize as number) <= 16;
if (useSmallVariant) {
return (
<svg
ref={ref}
xmlns="http://www.w3.org/2000/svg"
width={size}
height={size}
viewBox="0 0 16 16"
fill={color}
className={className}
{...rest}
>
<path d="M6.766 11.328c-2.063-.25-3.516-1.734-3.516-3.656 0-.781.281-1.625.75-2.188-.203-.515-.172-1.609.063-2.062.625-.078 1.468.25 1.968.703.594-.187 1.219-.281 1.985-.281.765 0 1.39.094 1.953.265.484-.437 1.344-.765 1.969-.687.218.422.25 1.515.046 2.047.5.593.766 1.39.766 2.203 0 1.922-1.453 3.375-3.547 3.64.531.344.89 1.094.89 1.954v1.625c0 .468.391.734.86.547C13.781 14.359 16 11.53 16 8.03 16 3.61 12.406 0 7.984 0 3.563 0 0 3.61 0 8.031a7.88 7.88 0 0 0 5.172 7.422c.422.156.828-.125.828-.547v-1.25c-.219.094-.5.156-.75.156-1.031 0-1.64-.562-2.078-1.609-.172-.422-.36-.672-.719-.719-.187-.015-.25-.093-.25-.187 0-.188.313-.328.625-.328.453 0 .844.281 1.25.86.313.452.64.655 1.031.655s.641-.14 1-.5c.266-.265.47-.5.657-.656" />
</svg>
);
}
return (
<svg
ref={ref}
xmlns="http://www.w3.org/2000/svg"
width={size}
height={size}
viewBox="0 0 24 24"
fill={color}
className={className}
{...rest}
>
<path d="M10.226 17.284c-2.965-.36-5.054-2.493-5.054-5.256 0-1.123.404-2.336 1.078-3.144-.292-.741-.247-2.314.09-2.965.898-.112 2.111.36 2.83 1.01.853-.269 1.752-.404 2.853-.404 1.1 0 1.999.135 2.807.382.696-.629 1.932-1.1 2.83-.988.315.606.36 2.179.067 2.942.72.854 1.101 2 1.101 3.167 0 2.763-2.089 4.852-5.098 5.234.763.494 1.28 1.572 1.28 2.807v2.336c0 .674.561 1.056 1.235.786 4.066-1.55 7.255-5.615 7.255-10.646C23.5 6.188 18.334 1 11.978 1 5.62 1 .5 6.188.5 12.545c0 4.986 3.167 9.12 7.435 10.669.606.225 1.19-.18 1.19-.786V20.63a2.9 2.9 0 0 1-1.078.224c-1.483 0-2.359-.808-2.987-2.313-.247-.607-.517-.966-1.034-1.033-.27-.023-.359-.135-.359-.27 0-.27.45-.471.898-.471.652 0 1.213.404 1.797 1.235.45.651.921.943 1.483.943.561 0 .92-.202 1.437-.719.382-.381.674-.718.944-.943" />
</svg>
);
});
+1 -4
View File
@@ -16,12 +16,9 @@ let lastEventSource: MockEventSource | null = null;
beforeEach(() => {
lastEventSource = null;
// vitest 4 enforces that mock implementations used with `new` must have a
// [[Construct]] slot. Arrow functions don't, so we use a regular function
// declaration here. The production code calls `new EventSource(...)`.
vi.stubGlobal(
'EventSource',
vi.fn().mockImplementation(function () {
vi.fn().mockImplementation(() => {
lastEventSource = new MockEventSource();
return lastEventSource;
}),
-1
View File
@@ -16,7 +16,6 @@ export default defineConfig({
alias: {
'@': path.resolve(__dirname, './src'),
'@shared': path.resolve(__dirname, '../shared'),
'gitnexus-shared': path.resolve(__dirname, '../gitnexus-shared/src/index.ts'),
// Fix for Rollup failing to resolve this deep import from @langchain/anthropic
'@anthropic-ai/sdk/lib/transform-json-schema': path.resolve(
__dirname,
+4 -8
View File
@@ -38,15 +38,11 @@ export default defineConfig({
'src/main.tsx', // Entry point
'src/vite-env.d.ts', // Type declarations
],
// Thresholds set to the post-vitest-4 baseline (AST-aware remapping
// measures coverage more accurately than the old istanbul-style mapping,
// so the same 220 tests now report slightly lower percentages). These
// are soft floors for regression detection, not coverage targets.
thresholds: {
statements: 9,
branches: 4,
functions: 7,
lines: 9,
statements: 10,
branches: 10,
functions: 10,
lines: 10,
},
},
},
-4
View File
@@ -9,10 +9,6 @@ tsconfig.json
.gitignore
node_modules/
# Vendor build artifacts (created during install, not shipped)
vendor/**/node_modules
vendor/**/build
# Package lock (consumers use their own)
package-lock.json
-110
View File
@@ -2,116 +2,6 @@
All notable changes to GitNexus will be documented in this file.
## [Unreleased]
## [1.6.3] - 2026-04-24
### Added
- **Cross-repo impact analysis** — `@repo` MCP routing plus group resources let impact queries span multiple indexed repositories in a group (#794, #984)
- **Python scope-based call resolution** — registry-primary flip, performance, and generalization work from RFC #909 Ring 3 (#980)
- **C# scope-resolution migration** — C# now runs on the registry-primary path alongside Python (#934, #1019)
- **RFC #909 Ring 1 & Ring 2 scope-resolution infrastructure** — the shared foundation for language-agnostic scope resolution:
- Scope-resolution types and constants, `LanguageProvider` hook extension (#910, #911, #949, #950)
- `ScopeTree` + `PositionIndex` + `makeScopeId` (#912, #961)
- `DefIndex` / `ModuleScopeIndex` / `QualifiedNameIndex` (#913, #958)
- `MethodDispatchIndex` materialized view over `HeritageMap` (#914, #960)
- `resolveTypeRef` strict single-return type resolver (#916, #959)
- SCC-aware finalize with bounded fixpoint (#915, #962)
- `ClassRegistry` / `MethodRegistry` / `FieldRegistry` + 7-step lookup (#917, #963)
- Shadow-mode diff + aggregate, parity harness + static dashboard (#918, #923, #951, #972)
- `ScopeExtractor` driver with 5-pass CaptureMatch → ParsedFile (#919, #965)
- `ScopeExtractor` wired into parse-worker + processor (#920, #969)
- `finalize-orchestrator` materializes `ScopeResolutionIndexes` (#921, #970)
- Per-language `resolveImportTarget` adapter (#922, #971)
- `REGISTRY_PRIMARY_<LANG>` per-language flag reader (#924, #968)
- `emit-references` drains `ReferenceIndex` to graph edges (#925, #973)
- **`gitnexus analyze --name <alias>`** with duplicate-name guard in the repo registry (#955)
- **`gitnexus remove <target>`** unindexes a registered repo by name or path (#664, #1003)
- **Auto-infer registry name** from `git remote.origin.url` when `--name` is omitted (#981)
- **Sibling-clone drift detection** — indexed repos are fingerprinted by remote URL so duplicate registrations are caught before graph divergence (#982)
- **Configurable large-file skip threshold** — the walker's 512 KB default is now overridable via `GITNEXUS_MAX_FILE_SIZE` (KB) or `gitnexus analyze --max-file-size <kb>`. Values are clamped to the 32 MB tree-sitter ceiling, invalid inputs fall back to the default with a one-time warning, and the CLI banner reports the effective post-clamp threshold when an override is active (#991, #1044, #1045)
- **`GITNEXUS_INDEX_TEST_DIRS` opt-in** for `__tests__` / `__mocks__` traversal (#771, #1046)
- **`analyze` embedding preservation** — existing embeddings are preserved by default, `--force` regenerates them, `--drop-embeddings` opts out entirely (CLI + HTTP API) (#1055)
- **Structural embedding chunking** with data-driven `CHUNKING_RULES` dispatch, replacing the flat line-based split (#987)
- **PHP HTTP consumer detection** for the extractor catalogue (#993)
- **Per-phase search timing** instrumentation across the query pipeline (#953)
- **MCP disambiguation ranking** — `context` / `impact` candidates are ranked and expose `kind` / `file_path` hints (#888)
- **Docker images for UI + CLI/server** shipped via `docker-compose` with cosign signing (#967), RC image builds (#978), and GHCR → Docker Hub mirroring (#1029)
### Fixed
- **Go CALLS edges for receiver methods** — worker source IDs now align with the main pipeline, restoring receiver-method call edges (#1043)
- **Node 22 DEP0151 warning** from `tree-sitter-c-sharp` import silenced (#1013, #1049)
- **FTS index bootstrap** tries a local `LOAD` before `INSTALL` so offline/air-gapped runs no longer fail on network errors (#726)
- **FTS ensure failures** are no longer cached and are invalidated on pool teardown (#1006)
- **`groupImpact` local-impact errors** now bubble to the caller instead of being swallowed (#1004, #1007)
- **Friendly error** when a group name is not found, with regression test for #903 (#989)
- **`bm25` results** return FTS-matched symbols instead of an arbitrary `LIMIT 3` slice (#806)
- **Embedding AST traversal** switched from recursion to iterative DFS, fixing stack overflow on deeply nested files (#990)
- **React component path detection** runs before lowercasing, so mixed-case `.jsx`/`.tsx` files are recognised (#260)
- **`detect-changes` ENOBUFS** by setting `maxBuffer` on `git` / `rg` `execFileSync` invocations (#957)
- **`detect-changes` in direct CLI** — command was wired to MCP only; now exposed on the CLI as well (#892)
- **CLI gitnexus markers** — `<!-- gitnexus:* -->` is only matched at section position, no longer inside code/prose (#1041, #1042)
- **`opencode.json` setup** preserves existing comments and config during install (#998)
- **Sequential parser logging** — skipped languages are now logged instead of silently dropped (#1021)
- **`cli-e2e` fixture isolation** from the shared mini-repo, plus stabilised `rel-csv-split` stream teardown on Windows via `expect.poll` (#954, #1052)
- **Docker** — RC build guarded against empty `vtag`, `inputs.tag` used to detect `workflow_call` context, web builder stage now copies `gitnexus/package.json`, base image switched from alpine to debian (#983, #996, #997, #1014)
- **CI** — reusable `docker.yml` now inherits secrets from `release-candidate.yml` (#1054)
### Changed
- **`setup` config I/O unified** on `mergeJsoncFile` across all writers (#1031)
- **Docker CI** gains a retry wrapper for `build-push` with visibility and hardened shell
### Chore / Dependencies
- Dependency bumps: `graphology` 0.25.4 → 0.26.0 (#1001), `uuid` 13 → 14 (#1000), `@huggingface/transformers` (#1035), `@types/node` (#1002), `@types/uuid` (#1016), `vitest` 4.1.4 → 4.1.5 (#1017), `@vitest/coverage-v8` (#1018)
- gitnexus-web dependency bumps: `vite` 5.4.21 → 6.4.2 → 7.3.2 → 8.0.10 + `vitest` 4 (#1061, #1062, #1063), `lucide-react` 0.562.0 → 1.11.0 with local GitHub SVG fallback (#1038), `@langchain/anthropic` 1.3.10 → 1.3.27 (#1039), `@babel/types` (#1037)
- gitnexus-shared dependency bumps: `typescript` (#1034)
- GitHub Actions bumps: `actions/setup-node` 6.3.0 → 6.4.0 (#1033)
- Documentation: repo-wide `DoD.md` Definition of Done (#1032), gRPC microservices group guide (#906, #994), `group add` / `group remove` README fixes (#1020), CLI docs include `--skip-git` (#750), README Discord link updated
## [1.6.2] - 2026-04-18
### Added
- **Docker support** — containerized ingestion and MCP serving for reproducible runs on CI and container platforms (#848)
- **Language-agnostic heritage extractor** — config+factory pattern for class-heritage extraction (EXTENDS / IMPLEMENTS), completing the extractor refactor alongside method/field/call/variable (#890)
- **Language-agnostic call extractor** — config+factory pattern that collapses ~225 lines of inline parse-worker logic into declarative per-language configs (#877)
- **Language-agnostic variable extractor** — structured metadata for `Const` / `Static` / `Variable` nodes via config+factory pattern (#878)
- **AST-aware embedding chunking** — offset-based splitting preserves symbol boundaries, improving semantic search precision on large files (#889)
- **HTTP consumer detection for jQuery and axios object-form** — `$.ajax` / `$.get` / `$.post` and `axios({ url, method })` now recognized as HTTP call sites (#887)
### Fixed
- **Python external dotted imports** — avoid spurious same-file matches when an import path like `foo.bar.baz` refers to a third-party module (#899)
- **Worker warnings no longer terminate ingestion** — non-fatal parser warnings keep the pipeline running instead of aborting the run (#900, #261)
- **Global-install upgrade `ENOTEMPTY`** — devendored `tree-sitter-proto` install lifecycle + preinstall cleanup so `npm i -g gitnexus@latest` succeeds on top of an older install (#843, #846)
- **`env.cacheDir`** now defaults to a user-writable location, unblocking ingestion on systems where the install directory is read-only (#845)
- **Content-hash staleness detection for embeddings** — zero-node rebuilds no longer skip vector-index creation, fixing semantic search after selective re-analysis (#831)
- **`tree-sitter-c-sharp` version pin** — locked to 0.23.1 to avoid a breaking change in a transitive prerelease (#834)
- **`release-drafter` v7 CI** — replaced the removed `disable-releaser` flag with `dry-run` so release-note drafts still work
- **`npm arborist` crash from `tree-sitter-dart`** — switched the dependency URL format so `npm install` no longer crashes on clean installs
- **Service-group `ManifestExtractor`** — `config.links` now wires the manifest extractor properly, restoring cross-link discovery that had silently dropped to zero
### Changed
- **SemanticModel wired as a first-class resolution input (SM-20)** — `call-processor`, `resolution-context`, `type-env`, and `heritage-map` now consult `table.model.*` directly; 37 internal call sites migrated off the SymbolTable wrapper (#885)
- **Per-strategy `ImportSemantics` hooks** — `named` / `wildcard-transitive` / `wildcard-leaf` / `namespace` strategies split into composable hooks, replacing the monolithic conditional (Strategies 1–4 of #886)
- **Class extraction configs moved to `configs/` subdirectory** — per-language class configs now co-locate with the other extractor configs, completing the extractor layer's directory convention (#879)
- **CLI AI-context trimmed** — duplicated CLAUDE.md block removed from the shipped context, reducing token usage in LLM-consuming workflows (#904)
- **LLM context files optimized** — AI-consumed documentation tuned for accuracy and token efficiency (#857)
- **Workflow concurrency standardized** — all CI workflows adopt the consistent concurrency key pattern documented in CONTRIBUTING.md; release-note labeling automated (#837)
- **E2E status-ready timeout raised** — 45s accommodates parallel-worker startup variance on CI (#908)
### Chore / Dependencies
- **tree-sitter 0.25 upgrade readiness** — daily Dependabot monitor for the upcoming major-version bump (#847)
- Dependency bumps: `glob` 11.1.0 → 13.0.6 (#867), `commander` 12.1.0 → 14.0.3 (#868), `@huggingface/transformers` (#869), `@modelcontextprotocol/sdk` (#866), `lru-cache` 11.2.7 → 11.3.5 (#870), `mnemonist` 0.39.8 → 0.40.3 (#871), `@ladybugdb/core` (#873)
- gitnexus-web dependency bumps: `mermaid` 11.12.2 → 11.14.0 (#860), `tailwindcss` (#861), `jsdom` 29.0.0 → 29.0.2 (#863), `wait-on` 8.0.5 → 9.0.5 (#859), `@vitest/coverage-v8` (#864)
- GitHub Actions bumps: `actions/checkout` 4.3.1 → 6.0.2 (#842), `actions/upload-artifact` 4.6.2 → 7.0.1 (#838), `actions/setup-node` 4.4.0 → 6.3.0 (#841), `actions/cache` 5.0.4 → 5.0.5 (#840), `actions/github-script` 7.0.1 → 9.0.0 (#850), `dorny/paths-filter` 3.0.2 → 4.0.1 (#839), `amannn/action-semantic-pull-request` 6.1.1 (#853), `release-drafter/release-drafter` 6.0.0 → 7.2.0 (#852), `marocchino/sticky-pull-request-comment` 3.0.4 (#851), `softprops/action-gh-release` 2.5.0 → 3.0.0 (#849)
## [1.6.1] - 2026-04-13
### Added
+1 -1
View File
@@ -1,6 +1,6 @@
FROM node:20-bookworm
WORKDIR /app
RUN apt-get -o Acquire::Check-Valid-Until=false -o Acquire::Check-Date=false update && apt-get install -y python3 make g++ && rm -rf /var/lib/apt/lists/*
RUN apt-get update && apt-get install -y python3 make g++ && rm -rf /var/lib/apt/lists/*
COPY . .
RUN npm ci --ignore-scripts \
&& node scripts/patch-tree-sitter-swift.cjs \
+5 -44
View File
@@ -155,7 +155,6 @@ gitnexus analyze --force # Force full re-index
gitnexus analyze --embeddings # Enable embedding generation (slower, better search)
gitnexus analyze --skip-agents-md # Preserve custom AGENTS.md/CLAUDE.md gitnexus section edits
gitnexus analyze --verbose # Log skipped files when parsers are unavailable
gitnexus analyze --max-file-size 1024 # Skip files larger than N KB (default: 512, cap: 32768)
gitnexus mcp # Start MCP server (stdio) — serves all indexed repos
gitnexus serve # Start local HTTP server (multi-repo) for web UI
gitnexus index # Register an existing .gitnexus/ folder into the global registry
@@ -167,11 +166,11 @@ gitnexus wiki [path] # Generate LLM-powered docs from knowledge grap
gitnexus wiki --model <model> # Wiki with custom LLM model (default: gpt-4o-mini)
# Repository groups (multi-repo / monorepo service tracking)
gitnexus group create <name> # Create a repository group
gitnexus group add <group> <groupPath> <registryName> # Add a repo to a group. <groupPath> is a hierarchy path (e.g. hr/hiring/backend); <registryName> is the repo's name from the registry (see `gitnexus list`)
gitnexus group remove <group> <groupPath> # Remove a repo from a group by its hierarchy path
gitnexus group list [name] # List groups, or show one group's config
gitnexus group sync <name> # Extract contracts and match across repos/services
gitnexus group create <name> # Create a repository group
gitnexus group add <name> <repo> # Add a repo to a group
gitnexus group remove <name> <repo> # Remove a repo from a group
gitnexus group list [name] # List groups, or show one group's config
gitnexus group sync <name> # Extract contracts and match across repos/services
gitnexus group contracts <name> # Inspect extracted contracts and cross-links
gitnexus group query <name> <q> # Search execution flows across all repos in a group
gitnexus group status <name> # Check staleness of repos in a group
@@ -235,29 +234,6 @@ Installed automatically by both `gitnexus analyze` (per-repo) and `gitnexus setu
- Node.js >= 18
- Git repository (uses git for commit tracking)
## Release candidates
Stable releases publish to the default `latest` dist-tag. When a pull request
with non-documentation changes merges into `main`, an automated workflow also
publishes a prerelease build under the `rc` dist-tag, so early adopters can
try in-flight fixes without waiting for the next stable cut. (Docs-only
merges are skipped.)
```bash
# Try the latest release candidate (pre-stable — may change at any time)
npm install -g gitnexus@rc
# — or —
npx gitnexus@rc analyze
```
Release-candidate versions follow the standard semver prerelease format
`X.Y.Z-rc.N`, where `X.Y.Z` is the next stable target (bumped from the
current `latest` by patch by default; `minor` or `major` when kicking off a
bigger cycle) and `N` increments per published rc. Example sequence:
`1.6.2-rc.1`, `1.6.2-rc.2`, …, then once `1.6.2` ships stable,
`1.6.3-rc.1`. See the [Releases page](https://github.com/abhigyanpatwari/GitNexus/releases)
for the full list; stable `latest` is unaffected.
## Troubleshooting
### `Cannot destructure property 'package' of 'node.target' as it is null`
@@ -308,21 +284,6 @@ echo "vendor/" >> .gitnexusignore
echo "dist/" >> .gitnexusignore
```
### Large files are being skipped
By default the walker skips files larger than **512 KB** (see log line `Skipped N large files (>512KB)`). Raise the threshold via either the CLI flag or the environment variable — both accept a value in **KB**:
```bash
# CLI flag (takes precedence over the env var)
npx gitnexus analyze --max-file-size 2048 # skip only files > 2 MB
# Environment variable (persists across commands)
export GITNEXUS_MAX_FILE_SIZE=2048
npx gitnexus analyze
```
Values above **32768 KB (32 MB)** are clamped to the tree-sitter parser ceiling; invalid values fall back to the 512 KB default with a one-time warning. When an override is active, `analyze` prints the effective threshold in its startup banner (e.g. `GITNEXUS_MAX_FILE_SIZE: effective threshold 2048KB (default 512KB)`).
## Privacy
- All processing happens locally on your machine
+437 -308
View File
File diff suppressed because it is too large Load Diff
+13 -17
View File
@@ -1,6 +1,6 @@
{
"name": "gitnexus",
"version": "1.6.4-rc.13",
"version": "1.6.1",
"description": "Graph-powered code intelligence for AI agents. Index any codebase, query via MCP or CLI.",
"author": "Abhigyan Patwari",
"license": "PolyForm-Noncommercial-1.0.0",
@@ -35,8 +35,7 @@
"hooks",
"scripts",
"skills",
"vendor",
"web"
"vendor"
],
"scripts": {
"build": "node scripts/build.js",
@@ -47,33 +46,32 @@
"test:integration": "vitest run test/integration",
"test:watch": "vitest",
"test:coverage": "vitest run --coverage",
"postinstall": "node scripts/patch-tree-sitter-swift.cjs && node scripts/build-tree-sitter-proto.cjs",
"postinstall": "node scripts/patch-tree-sitter-swift.cjs",
"prepare": "node scripts/build.js",
"prepack": "node scripts/build.js"
},
"dependencies": {
"@huggingface/transformers": "^4.1.0",
"@huggingface/transformers": "^3.0.0",
"@ladybugdb/core": "^0.15.2",
"@modelcontextprotocol/sdk": "^1.0.0",
"@scarf/scarf": "^1.4.0",
"cli-progress": "^3.12.0",
"commander": "^14.0.3",
"commander": "^12.0.0",
"cors": "^2.8.5",
"express": "^4.19.2",
"glob": "^13.0.6",
"graphology": "^0.26.0",
"glob": "^11.0.0",
"graphology": "^0.25.4",
"graphology-indices": "^0.17.0",
"graphology-utils": "^2.3.0",
"ignore": "^7.0.5",
"js-yaml": "^4.1.1",
"jsonc-parser": "^3.3.1",
"lru-cache": "^11.0.0",
"mnemonist": "^0.40.3",
"mnemonist": "^0.39.0",
"onnxruntime-node": "^1.24.0",
"pandemonium": "^2.4.0",
"tree-sitter": "^0.21.1",
"tree-sitter-c": "0.23.2",
"tree-sitter-c-sharp": "0.23.1",
"tree-sitter-c-sharp": "^0.23.1",
"tree-sitter-cpp": "^0.23.4",
"tree-sitter-go": "^0.23.0",
"tree-sitter-java": "^0.23.5",
@@ -83,25 +81,23 @@
"tree-sitter-ruby": "^0.23.1",
"tree-sitter-rust": "0.23.1",
"tree-sitter-typescript": "^0.23.2",
"uuid": "^14.0.0"
"uuid": "^13.0.0"
},
"optionalDependencies": {
"node-addon-api": "^8.0.0",
"node-gyp-build": "^4.8.0",
"tree-sitter-dart": "git+https://github.com/UserNobody14/tree-sitter-dart.git#80e23c07b64494f7e21090bb3450223ef0b192f4",
"tree-sitter-kotlin": "^0.3.8",
"tree-sitter-proto": "file:./vendor/tree-sitter-proto",
"tree-sitter-swift": "^0.6.0"
},
"devDependencies": {
"gitnexus-shared": "file:../gitnexus-shared",
"@types/cli-progress": "^3.11.6",
"@types/cors": "^2.8.17",
"@types/express": "^4.17.21",
"@types/js-yaml": "^4.0.9",
"@types/node": "^25.6.0",
"@types/uuid": "^11.0.0",
"@types/node": "^20.0.0",
"@types/uuid": "^10.0.0",
"@vitest/coverage-v8": "^4.0.18",
"gitnexus-shared": "file:../gitnexus-shared",
"tsx": "^4.0.0",
"typescript": "^5.4.5",
"vitest": "^4.0.18"
-134
View File
@@ -1,134 +0,0 @@
/**
* Synthetic benchmark for scope-resolution. Builds a large in-memory
* Python workspace and times runScopeResolution against it directly,
* isolating the resolution cost from parse / heritage / pipeline
* overhead.
*
* Usage: REGISTRY_PRIMARY_PYTHON=1 npx tsx scripts/bench-scope-resolution.ts
*/
process.env.REGISTRY_PRIMARY_PYTHON = '1';
import { generateId } from '../src/lib/utils.js';
import { createKnowledgeGraph } from '../src/core/graph/graph.js';
import { runScopeResolution } from '../src/core/ingestion/scope-resolution/index.js';
import { pythonScopeResolver } from '../src/core/ingestion/languages/python/scope-resolver.js';
const N_CLASSES = Number(process.env.BENCH_CLASSES ?? '60');
const N_USERS = Number(process.env.BENCH_USERS ?? '40');
const ITERS = Number(process.env.BENCH_ITERS ?? '5');
function buildWorkspace(): { path: string; content: string }[] {
const files: { path: string; content: string }[] = [];
// Build N_CLASSES "model" files, each defining a class with a few methods.
for (let i = 0; i < N_CLASSES; i++) {
const lines: string[] = [];
for (let j = 0; j < 5; j++) {
lines.push(`class Model${i}_${j}:`);
lines.push(` name: str`);
lines.push(` def save(self) -> bool:`);
lines.push(` return True`);
lines.push(` def update(self, name: str) -> "Model${i}_${j}":`);
lines.push(` self.name = name`);
lines.push(` return self`);
lines.push(` def get_other(self) -> "Model${i}_${(j + 1) % 5}":`);
lines.push(` return Model${i}_${(j + 1) % 5}()`);
lines.push('');
}
files.push({ path: `models/m${i}.py`, content: lines.join('\n') });
}
// Build N_USERS "user" files that import from a few model files
// and exercise the receiver-bound dispatcher heavily.
for (let u = 0; u < N_USERS; u++) {
const targets = [u % N_CLASSES, (u + 1) % N_CLASSES, (u + 2) % N_CLASSES];
const imports = targets
.map((t) => `from models.m${t} import Model${t}_0, Model${t}_1, Model${t}_2`)
.join('\n');
const calls: string[] = [];
for (let k = 0; k < 30; k++) {
const t = targets[k % 3]!;
const j = k % 3;
calls.push(` m${k} = Model${t}_${j}()`);
calls.push(` m${k}.save()`);
calls.push(` m${k}.update("x").save()`);
calls.push(` m${k}.get_other().save()`);
}
const content = `${imports}\n\ndef use_${u}() -> None:\n${calls.join('\n')}\n`;
files.push({ path: `app/u${u}.py`, content });
}
return files;
}
function buildGraph(files: { path: string; content: string }[]) {
const graph = createKnowledgeGraph();
// Pre-populate File / Class / Function nodes the resolver expects.
for (const f of files) {
const fileId = generateId('File', f.path);
graph.addNode({
id: fileId,
label: 'File',
properties: { name: f.path, filePath: f.path },
});
// Lightweight regex-extract class & def names so the lookup index
// has something to find. Real pipeline builds these via parse phase;
// for the bench this stand-in is enough to exercise the resolver.
const classRe = /^class (\w+)/gm;
const defRe = /^\s*def (\w+)/gm;
let m: RegExpExecArray | null;
while ((m = classRe.exec(f.content)) !== null) {
const name = m[1]!;
const id = generateId('Class', `${f.path}:${name}`);
graph.addNode({
id,
label: 'Class',
properties: { name, filePath: f.path, qualifiedName: name },
});
}
while ((m = defRe.exec(f.content)) !== null) {
const name = m[1]!;
const id = generateId('Function', `${f.path}:${name}`);
graph.addNode({
id,
label: 'Function',
properties: { name, filePath: f.path, qualifiedName: name },
});
}
}
return graph;
}
async function main() {
const files = buildWorkspace();
console.log(`bench: ${files.length} files (${N_CLASSES} models × 5 classes + ${N_USERS} users)`);
console.log(` × ${ITERS} iterations\n`);
// Warmup
for (let i = 0; i < 2; i++) {
const graph = buildGraph(files);
runScopeResolution({ graph, files, onWarn: () => {} }, pythonScopeResolver);
}
const samples: number[] = [];
for (let i = 0; i < ITERS; i++) {
const graph = buildGraph(files);
const start = process.hrtime.bigint();
runScopeResolution({ graph, files, onWarn: () => {} }, pythonScopeResolver);
const end = process.hrtime.bigint();
const ms = Number(end - start) / 1_000_000;
samples.push(ms);
console.log(` iter ${i + 1}: ${ms.toFixed(0)} ms`);
}
samples.sort((a, b) => a - b);
const median = samples[Math.floor(samples.length / 2)]!;
const min = samples[0]!;
console.log(`\nmin: ${min.toFixed(0)} ms · median: ${median.toFixed(0)} ms`);
}
main().catch((err) => {
console.error(err);
process.exit(1);
});
@@ -1,82 +0,0 @@
#!/usr/bin/env node
/**
* Build tree-sitter-proto native binding.
*
* Why this script exists:
* tree-sitter-proto is vendored under gitnexus/vendor/tree-sitter-proto/
* and declared as a `file:` optionalDependency. Previously, the vendored
* package had its own `dependencies` and `install` script, which caused
* npm to create `vendor/tree-sitter-proto/node_modules/` and
* `vendor/tree-sitter-proto/build/` during install. Those directories
* blocked `rmdir` on global-install upgrade, producing:
*
* ENOTEMPTY: directory not empty, rmdir
* '.../gitnexus/vendor/tree-sitter-proto/node_modules/node-addon-api'
*
* (See https://github.com/abhigyanpatwari/GitNexus/issues/836.)
*
* We stripped `dependencies` and the `install` script from the vendored
* package.json, hoisted `node-addon-api` and `node-gyp-build` into
* gitnexus's own optionalDependencies, and moved native compilation here.
*
* What this does:
* Runs `npx node-gyp rebuild` inside `node_modules/tree-sitter-proto/`
* (which npm creates as a copy of vendor/tree-sitter-proto/ when
* resolving the file: dep). Build output lands in
* `node_modules/tree-sitter-proto/build/Release/tree_sitter_proto_binding.node`
* — under npm-managed territory, safe on upgrade.
*
* Mirrors scripts/patch-tree-sitter-swift.cjs. Best-effort: if any
* precondition fails (optional dep absent, no toolchain, --ignore-scripts),
* warn and exit 0 so gitnexus install still succeeds.
*/
const fs = require('fs');
const path = require('path');
const { execSync } = require('child_process');
const protoDir = path.join(__dirname, '..', 'node_modules', 'tree-sitter-proto');
const bindingGyp = path.join(protoDir, 'binding.gyp');
const bindingNode = path.join(protoDir, 'build', 'Release', 'tree_sitter_proto_binding.node');
try {
if (!fs.existsSync(bindingGyp)) {
// tree-sitter-proto is an optionalDependency; absent when install
// skipped optional deps or the file: dep was not resolved.
process.exit(0);
}
// Skip if the native binding already exists (idempotent re-run).
if (fs.existsSync(bindingNode)) {
process.exit(0);
}
// Pre-flight: the hoisted build deps must be resolvable.
try {
require.resolve('node-addon-api');
require.resolve('node-gyp-build');
} catch (resolveErr) {
console.warn(
'[tree-sitter-proto] Skipping build: hoisted build deps not resolvable (%s).',
resolveErr.message,
);
console.warn(
'[tree-sitter-proto] Proto parsing will be unavailable. Install without --no-optional and with scripts enabled to build.',
);
process.exit(0);
}
console.log('[tree-sitter-proto] Building native binding...');
execSync('npx node-gyp rebuild', {
cwd: protoDir,
stdio: 'pipe',
timeout: 180000,
});
console.log('[tree-sitter-proto] Native binding built successfully');
} catch (err) {
console.warn('[tree-sitter-proto] Could not build native binding:', err.message);
console.warn(
'[tree-sitter-proto] Proto (.proto) parsing will be unavailable. Non-proto gitnexus functionality is unaffected.',
);
// Exit 0: optionalDependency failures must not fail the gitnexus install.
process.exit(0);
}
+2 -22
View File
@@ -21,11 +21,11 @@ const SHARED_DEST = path.join(DIST, '_shared');
// ── 1. Build gitnexus-shared ───────────────────────────────────────
console.log('[build] compiling gitnexus-shared…');
execSync('npx tsc', { cwd: SHARED_ROOT, stdio: 'inherit', timeout: 120_000 });
execSync('npx tsc', { cwd: SHARED_ROOT, stdio: 'inherit' });
// ── 2. Build gitnexus ──────────────────────────────────────────────
console.log('[build] compiling gitnexus…');
execSync('npx tsc', { cwd: ROOT, stdio: 'inherit', timeout: 120_000 });
execSync('npx tsc', { cwd: ROOT, stdio: 'inherit' });
// ── 3. Copy shared dist ────────────────────────────────────────────
console.log('[build] copying shared module into dist/_shared…');
@@ -70,24 +70,4 @@ walk(DIST, ['.js', '.d.ts'], rewriteFile);
const cliEntry = path.join(DIST, 'cli', 'index.js');
if (fs.existsSync(cliEntry)) fs.chmodSync(cliEntry, 0o755);
// ── 6. Build & copy web UI ──────────────────────────────────────────
const WEB_ROOT = path.resolve(ROOT, '..', 'gitnexus-web');
const WEB_DEST = path.join(DIST, '..', 'web');
if (fs.existsSync(path.join(WEB_ROOT, 'package.json'))) {
console.log('[build] building gitnexus-web…');
if (!fs.existsSync(path.join(WEB_ROOT, 'node_modules'))) {
console.log('[build] installing gitnexus-web dependencies…');
execSync('npm ci', { cwd: WEB_ROOT, stdio: 'inherit', timeout: 120_000 });
}
execSync('npm run build', { cwd: WEB_ROOT, stdio: 'inherit', timeout: 120_000 });
// Copy dist → gitnexus/web/ (shipped in the npm package)
fs.rmSync(WEB_DEST, { recursive: true, force: true });
fs.cpSync(path.join(WEB_ROOT, 'dist'), WEB_DEST, { recursive: true });
console.log('[build] copied web UI → gitnexus/web/');
} else {
console.log('[build] skipping web UI (gitnexus-web not found)');
}
console.log(`[build] done — rewrote ${rewritten} files.`);
@@ -1,24 +0,0 @@
/**
* CI helper — emits the `MIGRATED_LANGUAGES` set as a JSON matrix array for
* GitHub Actions (`.github/workflows/ci-scope-parity.yml`).
*
* Consumed by the `discover` job in that workflow. Each entry has:
* - `slug`: lowercase language id, matching `test/integration/resolvers/<slug>.test.ts`.
* - `envvar`: uppercase suffix used to build the `REGISTRY_PRIMARY_<envvar>` toggle.
*
* Run with `npx tsx scripts/ci-list-migrated-languages.ts`. The script
* writes a single JSON array to stdout (no wrapper object) so the
* workflow can pipe it straight into `$GITHUB_OUTPUT`.
*/
import { MIGRATED_LANGUAGES } from '../src/core/ingestion/registry-primary-flag.js';
const entries = [...MIGRATED_LANGUAGES].map((slug) => {
const s = String(slug);
return {
slug: s,
envvar: s.toUpperCase().replace(/-/g, '_'),
};
});
process.stdout.write(JSON.stringify(entries));
-291
View File
@@ -1,291 +0,0 @@
<!doctype html>
<html lang="en">
<head>
<meta charset="utf-8" />
<meta name="viewport" content="width=device-width,initial-scale=1" />
<title>GitNexus — Shadow Parity Dashboard</title>
<!--
Static dashboard for the RFC #909 shadow-mode parity report.
Reads `latest.json` from this directory and renders a per-language
parity table. Zero build step, zero runtime dependencies — a
single file that any browser or file:// context can open.
Usage:
# from repo root, after a shadow-mode run
cp .gitnexus/shadow-parity/latest.json gitnexus/shadow-parity-dashboard/
open gitnexus/shadow-parity-dashboard/index.html
CI artifact wiring (follow-up): the CI job publishes a snapshot
of this directory + latest.json as a downloadable bundle per run.
-->
<style>
:root {
color-scheme: light dark;
--fg: #1f2937;
--fg-muted: #6b7280;
--bg: #ffffff;
--bg-muted: #f9fafb;
--border: #e5e7eb;
--good: #16a34a;
--warn: #d97706;
--bad: #dc2626;
--primary-tag-legacy: #7c3aed;
--primary-tag-registry: #0ea5e9;
}
@media (prefers-color-scheme: dark) {
:root {
--fg: #e5e7eb;
--fg-muted: #9ca3af;
--bg: #111827;
--bg-muted: #1f2937;
--border: #374151;
}
}
html,
body {
margin: 0;
padding: 0;
background: var(--bg);
color: var(--fg);
font:
14px/1.45 system-ui,
-apple-system,
sans-serif;
}
main {
max-width: 1200px;
margin: 0 auto;
padding: 24px 16px;
}
h1 {
font-size: 20px;
margin: 0 0 4px;
}
.meta {
color: var(--fg-muted);
font-size: 12px;
margin-bottom: 20px;
}
.cards {
display: grid;
grid-template-columns: repeat(auto-fit, minmax(180px, 1fr));
gap: 10px;
margin-bottom: 20px;
}
.card {
border: 1px solid var(--border);
border-radius: 6px;
padding: 10px 12px;
background: var(--bg-muted);
}
.card .k {
color: var(--fg-muted);
font-size: 11px;
text-transform: uppercase;
letter-spacing: 0.04em;
}
.card .v {
font-size: 20px;
font-weight: 600;
}
table {
width: 100%;
border-collapse: collapse;
font-variant-numeric: tabular-nums;
}
th,
td {
padding: 6px 10px;
text-align: right;
border-bottom: 1px solid var(--border);
}
th:first-child,
td:first-child {
text-align: left;
}
thead th {
font-weight: 600;
color: var(--fg-muted);
font-size: 12px;
background: var(--bg-muted);
}
tbody tr:hover {
background: var(--bg-muted);
}
.parity {
font-weight: 600;
}
.parity.good {
color: var(--good);
}
.parity.warn {
color: var(--warn);
}
.parity.bad {
color: var(--bad);
}
.tag {
display: inline-block;
padding: 1px 6px;
border-radius: 10px;
font-size: 10px;
margin-left: 6px;
color: white;
}
.tag.legacy {
background: var(--primary-tag-legacy);
}
.tag.registry {
background: var(--primary-tag-registry);
}
.empty {
padding: 40px;
text-align: center;
color: var(--fg-muted);
}
code {
font-family: ui-monospace, SFMono-Regular, Menlo, monospace;
background: var(--bg-muted);
padding: 1px 4px;
border-radius: 3px;
}
</style>
</head>
<body>
<main>
<h1>Shadow Parity — RFC #909</h1>
<div class="meta" id="meta">loading <code>latest.json</code>…</div>
<div class="cards" id="cards"></div>
<table id="per-language">
<thead>
<tr>
<th>Language</th>
<th>Total</th>
<th>Agree</th>
<th>Only legacy</th>
<th>Only new</th>
<th>Disagree</th>
<th>Both empty</th>
<th>Parity</th>
</tr>
</thead>
<tbody></tbody>
</table>
<div id="empty" class="empty" style="display: none">
No records yet. Enable <code>GITNEXUS_SHADOW_MODE=1</code> and run ingestion to populate.
</div>
</main>
<script>
/* global fetch, document */
(async function () {
const tbody = document.querySelector('#per-language tbody');
const cards = document.getElementById('cards');
const meta = document.getElementById('meta');
const empty = document.getElementById('empty');
const table = document.getElementById('per-language');
let payload;
try {
const r = await fetch('./latest.json', { cache: 'no-store' });
if (!r.ok) throw new Error('HTTP ' + r.status);
payload = await r.json();
} catch (err) {
meta.textContent = 'Failed to load latest.json: ' + err.message;
table.style.display = 'none';
empty.style.display = 'block';
return;
}
const primary = payload.primaryByLanguage || {};
const report = payload.report || {};
const perLang = report.perLanguage || [];
const overall = report.overall || {};
meta.textContent =
'Run ' +
payload.runId +
' — generated ' +
payload.generatedAt +
' (schema v' +
payload.schemaVersion +
')';
// Overall summary cards.
cards.innerHTML = '';
const overallParity = overall.parity !== undefined ? overall.parity : 0;
cards.appendChild(makeCard('Total calls', overall.totalCalls ?? 0));
cards.appendChild(makeCard('Both agree', overall.bothAgree ?? 0));
cards.appendChild(makeCard('Disagree', overall.bothDisagree ?? 0));
cards.appendChild(makeCard('Overall parity', formatPct(overallParity)));
if (!perLang.length) {
table.style.display = 'none';
empty.style.display = 'block';
return;
}
for (const row of perLang) {
const tr = document.createElement('tr');
const primaryTag = primary[row.language];
const tag = primaryTag
? '<span class="tag ' + primaryTag + '">primary: ' + primaryTag + '</span>'
: '';
const parityClass = parityClassFor(row.parity);
tr.innerHTML =
'<td>' +
escape(row.language) +
tag +
'</td>' +
'<td>' +
row.totalCalls +
'</td>' +
'<td>' +
row.bothAgree +
'</td>' +
'<td>' +
row.onlyLegacy +
'</td>' +
'<td>' +
row.onlyNew +
'</td>' +
'<td>' +
row.bothDisagree +
'</td>' +
'<td>' +
row.bothEmpty +
'</td>' +
'<td class="parity ' +
parityClass +
'">' +
formatPct(row.parity) +
'</td>';
tbody.appendChild(tr);
}
function makeCard(k, v) {
const div = document.createElement('div');
div.className = 'card';
div.innerHTML =
'<div class="k">' + escape(k) + '</div><div class="v">' + escape(String(v)) + '</div>';
return div;
}
function formatPct(x) {
if (typeof x !== 'number' || !isFinite(x)) return '—';
return (x * 100).toFixed(1) + '%';
}
function parityClassFor(x) {
if (typeof x !== 'number') return '';
if (x >= 0.95) return 'good';
if (x >= 0.8) return 'warn';
return 'bad';
}
function escape(s) {
return String(s).replace(/[&<>"']/g, function (c) {
return { '&': '&amp;', '<': '&lt;', '>': '&gt;', '"': '&quot;', "'": '&#39;' }[c];
});
}
})();
</script>
</body>
</html>
+1 -2
View File
@@ -21,9 +21,8 @@ Run from the project root. This parses all source files, builds the knowledge gr
| -------------- | ---------------------------------------------------------------- |
| `--force` | Force full re-index even if up to date |
| `--embeddings` | Enable embedding generation for semantic search (off by default) |
| `--drop-embeddings` | Drop existing embeddings on rebuild. By default, an `analyze` without `--embeddings` preserves them. |
**When to run:** First time in a project, after major code changes, or when `gitnexus://repo/{name}/context` reports the index is stale. In Claude Code, a PostToolUse hook detects staleness after `git commit` and `git merge` and notifies the agent to run `analyze` — the hook does not run analyze itself, to avoid blocking the agent for up to 120s and risking KuzuDB corruption on timeout.
**When to run:** First time in a project, after major code changes, or when `gitnexus://repo/{name}/context` reports the index is stale. In Claude Code, a PostToolUse hook runs `analyze` automatically after `git commit` and `git merge`, preserving embeddings if previously generated.
### status — Check index freshness
+62 -40
View File
@@ -32,33 +32,6 @@ export interface AIContextOptions {
const GITNEXUS_START_MARKER = '<!-- gitnexus:start -->';
const GITNEXUS_END_MARKER = '<!-- gitnexus:end -->';
/**
* Find the index of a section marker that occupies its own line.
* Unlike `indexOf`, this rejects inline prose references like
* `` See the `<!-- gitnexus:start -->` block `` that appear
* mid-sentence (#1041). A marker counts as section-position only when:
* - preceded by newline or start-of-file, AND
* - followed by newline, `\r` (CRLF files), or end-of-file.
* The generator always emits each marker alone on its line, so this
* matches every legitimate section and none of the inline mentions.
*
* `startFrom` lets the end-marker lookup start after the already-found
* start marker, avoiding a scan from 0 and guaranteeing we never pick
* up an end marker that appears earlier in the file than the start.
*/
function findSectionMarkerIndex(content: string, marker: string, startFrom = 0): number {
let idx = content.indexOf(marker, startFrom);
while (idx !== -1) {
const atLineStart = idx === 0 || content[idx - 1] === '\n';
const endPos = idx + marker.length;
const atLineEnd =
endPos === content.length || content[endPos] === '\n' || content[endPos] === '\r';
if (atLineStart && atLineEnd) return idx;
idx = content.indexOf(marker, idx + 1);
}
return -1;
}
/**
* Generate the full GitNexus context content.
*
@@ -128,6 +101,19 @@ This project is indexed by GitNexus as **${projectName}**${noStats ? '' : ` (${s
- When exploring unfamiliar code, use \`gitnexus_query({query: "concept"})\` to find execution flows instead of grepping. It returns process-grouped results ranked by relevance.
- When you need full context on a specific symbol — callers, callees, which execution flows it participates in — use \`gitnexus_context({name: "symbolName"})\`.
## When Debugging
1. \`gitnexus_query({query: "<error or symptom>"})\` — find execution flows related to the issue
2. \`gitnexus_context({name: "<suspect function>"})\` — see all callers, callees, and process participation
3. \`READ gitnexus://repo/${projectName}/process/{processName}\` — trace the full execution flow step by step
4. For regressions: \`gitnexus_detect_changes({scope: "compare", base_ref: "main"})\` — see what your branch changed
## When Refactoring
- **Renaming**: MUST use \`gitnexus_rename({symbol_name: "old", new_name: "new", dry_run: true})\` first. Review the preview — graph edits are safe, text_search edits need manual review. Then run with \`dry_run: false\`.
- **Extracting/Splitting**: MUST run \`gitnexus_context({name: "target"})\` to see all incoming/outgoing refs, then \`gitnexus_impact({target: "target", direction: "upstream"})\` to find all external callers before moving code.
- After any refactor: run \`gitnexus_detect_changes({scope: "all"})\` to verify only expected files changed.
## Never Do
- NEVER edit a function, class, or method without first running \`gitnexus_impact\` on it.
@@ -135,6 +121,25 @@ This project is indexed by GitNexus as **${projectName}**${noStats ? '' : ` (${s
- NEVER rename symbols with find-and-replace — use \`gitnexus_rename\` which understands the call graph.
- NEVER commit changes without running \`gitnexus_detect_changes()\` to check affected scope.
## Tools Quick Reference
| Tool | When to use | Command |
|------|-------------|---------|
| \`query\` | Find code by concept | \`gitnexus_query({query: "auth validation"})\` |
| \`context\` | 360-degree view of one symbol | \`gitnexus_context({name: "validateUser"})\` |
| \`impact\` | Blast radius before editing | \`gitnexus_impact({target: "X", direction: "upstream"})\` |
| \`detect_changes\` | Pre-commit scope check | \`gitnexus_detect_changes({scope: "staged"})\` |
| \`rename\` | Safe multi-file rename | \`gitnexus_rename({symbol_name: "old", new_name: "new", dry_run: true})\` |
| \`cypher\` | Custom graph queries | \`gitnexus_cypher({query: "MATCH ..."})\` |
## Impact Risk Levels
| Depth | Meaning | Action |
|-------|---------|--------|
| d=1 | WILL BREAK — direct callers/importers | MUST update these |
| d=2 | LIKELY AFFECTED — indirect deps | Should test |
| d=3 | MAY NEED TESTING — transitive | Test if critical path |
## Resources
| Resource | Use for |
@@ -144,11 +149,37 @@ This project is indexed by GitNexus as **${projectName}**${noStats ? '' : ` (${s
| \`gitnexus://repo/${projectName}/processes\` | All execution flows |
| \`gitnexus://repo/${projectName}/process/{name}\` | Step-by-step execution trace |
## Self-Check Before Finishing
Before completing any code modification task, verify:
1. \`gitnexus_impact\` was run for all modified symbols
2. No HIGH/CRITICAL risk warnings were ignored
3. \`gitnexus_detect_changes()\` confirms changes match expected scope
4. All d=1 (WILL BREAK) dependents were updated
## Keeping the Index Fresh
After committing code changes, the GitNexus index becomes stale. Re-run analyze to update it:
\`\`\`bash
npx gitnexus analyze
\`\`\`
If the index previously included embeddings, preserve them by adding \`--embeddings\`:
\`\`\`bash
npx gitnexus analyze --embeddings
\`\`\`
To check whether embeddings exist, inspect \`.gitnexus/meta.json\` — the \`stats.embeddings\` field shows the count (0 means no embeddings). **Running analyze without \`--embeddings\` will delete any previously generated embeddings.**
> Claude Code users: A PostToolUse hook handles this automatically after \`git commit\` and \`git merge\`.
${
groupNames && groupNames.length > 0
? `## Cross-Repo Groups
This repository is listed under GitNexus **group(s): ${groupNames.join(', ')}** (see \`~/.gitnexus/groups/\`). For cross-repo analysis, use MCP tools \`impact\`, \`query\`, and \`context\` with \`repo\` set to \`@<groupName>\` or \`@<groupName>/<memberPath>\` (paths match keys in that group’s \`group.yaml\`). Use \`group_list\` / \`group_sync\` for membership and sync. From the terminal: \`npx gitnexus group list\`, \`npx gitnexus group sync <name>\`, \`npx gitnexus group impact <name> --target <symbol> --repo <group-path>\`.
This repository is listed under GitNexus **group(s): ${groupNames.join(', ')}** (see \`~/.gitnexus/groups/\`). For blast radius across repository boundaries, use MCP tools \`group_impact\`, \`group_sync\`, \`group_query\`, \`group_contracts\`, \`group_status\`, and \`group_list\`. From the terminal: \`npx gitnexus group list\`, \`npx gitnexus group sync <name>\`, \`npx gitnexus group impact <name> --target <symbol> --repo <group-path>\`.
`
: ''
@@ -190,18 +221,9 @@ async function upsertGitNexusSection(
const existingContent = await fs.readFile(filePath, 'utf-8');
// Check if GitNexus section already exists. Matching is restricted
// to markers that occupy their own line so that inline prose
// references (e.g. `` See the `<!-- gitnexus:start -->` block `` in
// the shipped CLAUDE.md) are NOT treated as section delimiters
// (#1041). The end-marker scan starts after the start-marker so it
// can never pick up an earlier end in the file.
const startIdx = findSectionMarkerIndex(existingContent, GITNEXUS_START_MARKER);
const endIdx = findSectionMarkerIndex(
existingContent,
GITNEXUS_END_MARKER,
startIdx === -1 ? 0 : startIdx,
);
// Check if GitNexus section already exists
const startIdx = existingContent.indexOf(GITNEXUS_START_MARKER);
const endIdx = existingContent.indexOf(GITNEXUS_END_MARKER);
if (startIdx !== -1 && endIdx !== -1 && endIdx > startIdx) {
// Replace existing section
+1 -71
View File
@@ -13,14 +13,9 @@ import { execFileSync } from 'child_process';
import v8 from 'v8';
import cliProgress from 'cli-progress';
import { closeLbug } from '../core/lbug/lbug-adapter.js';
import {
getStoragePaths,
getGlobalRegistryPath,
RegistryNameCollisionError,
} from '../storage/repo-manager.js';
import { getStoragePaths, getGlobalRegistryPath } from '../storage/repo-manager.js';
import { getGitRoot, hasGitDir } from '../storage/git.js';
import { runFullAnalysis } from '../core/run-analyze.js';
import { getMaxFileSizeBannerMessage } from '../core/ingestion/utils/max-file-size.js';
import fs from 'fs/promises';
const HEAP_MB = 8192;
@@ -56,12 +51,6 @@ function ensureHeap(): boolean {
export interface AnalyzeOptions {
force?: boolean;
embeddings?: boolean;
/**
* Explicitly drop existing embeddings on rebuild instead of preserving
* them. Without this flag, a routine `analyze` keeps any embeddings
* already present in the index even when `--embeddings` is omitted.
*/
dropEmbeddings?: boolean;
skills?: boolean;
verbose?: boolean;
/** Skip AGENTS.md and CLAUDE.md gitnexus block updates. */
@@ -70,27 +59,6 @@ export interface AnalyzeOptions {
noStats?: boolean;
/** Index the folder even when no .git directory is present. */
skipGit?: boolean;
/**
* Override the default basename-derived registry `name` with a
* user-supplied alias (#829). Disambiguates repos whose paths share a
* basename. Persisted — subsequent re-analyses of the same path without
* `--name` preserve the alias.
*/
name?: string;
/**
* Allow registration even when another path already uses the same
* `--name` alias (#829). Intentionally a distinct flag from `--force`
* because the user may want to coexist under the same name WITHOUT
* paying the cost of a pipeline re-index. Maps to registerRepo's
* `allowDuplicateName` option end-to-end.
*/
allowDuplicateName?: boolean;
/**
* Override the walker's large-file skip threshold (#991). Value in KB;
* clamped downstream to the tree-sitter 32 MB ceiling. Sets
* `GITNEXUS_MAX_FILE_SIZE` for the rest of the pipeline.
*/
maxFileSize?: string;
}
export const analyzeCommand = async (inputPath?: string, options?: AnalyzeOptions) => {
@@ -100,10 +68,6 @@ export const analyzeCommand = async (inputPath?: string, options?: AnalyzeOption
process.env.GITNEXUS_VERBOSE = '1';
}
if (options?.maxFileSize) {
process.env.GITNEXUS_MAX_FILE_SIZE = options.maxFileSize;
}
console.log('\n GitNexus Analyzer\n');
let repoPath: string;
@@ -149,11 +113,6 @@ export const analyzeCommand = async (inputPath?: string, options?: AnalyzeOption
);
}
const maxFileSizeBanner = getMaxFileSizeBannerMessage();
if (maxFileSizeBanner) {
console.log(`${maxFileSizeBanner}\n`);
}
// ── CLI progress bar setup ─────────────────────────────────────────
const bar = new cliProgress.SingleBar(
{
@@ -188,11 +147,9 @@ export const analyzeCommand = async (inputPath?: string, options?: AnalyzeOption
const origLog = console.log.bind(console);
const origWarn = console.warn.bind(console);
const origError = console.error.bind(console);
let barCurrentValue = 0;
const barLog = (...args: any[]) => {
process.stdout.write('\x1b[2K\r');
origLog(args.map((a) => (typeof a === 'string' ? a : String(a))).join(' '));
bar.update(barCurrentValue);
};
console.log = barLog;
console.warn = barLog;
@@ -203,7 +160,6 @@ export const analyzeCommand = async (inputPath?: string, options?: AnalyzeOption
let phaseStart = Date.now();
const updateBar = (value: number, phaseLabel: string) => {
barCurrentValue = value;
if (phaseLabel !== lastPhaseLabel) {
lastPhaseLabel = phaseLabel;
phaseStart = Date.now();
@@ -227,21 +183,11 @@ export const analyzeCommand = async (inputPath?: string, options?: AnalyzeOption
const result = await runFullAnalysis(
repoPath,
{
// Pipeline re-index — OR'd with --skills because skill generation
// needs a fresh pipelineResult. Has no bearing on the registry
// collision guard (see allowDuplicateName below).
force: options?.force || options?.skills,
embeddings: options?.embeddings,
dropEmbeddings: options?.dropEmbeddings,
skipGit: options?.skipGit,
skipAgentsMd: options?.skipAgentsMd,
noStats: options?.noStats,
registryName: options?.name,
// Registry-collision bypass — its own CLI flag, intentionally NOT
// overloading --force. A user who hits the collision guard should
// be able to accept the duplicate name without also paying the
// cost of a full pipeline re-index. See #829 review round 2.
allowDuplicateName: options?.allowDuplicateName,
},
{
onProgress: (_phase, percent, message) => {
@@ -349,22 +295,6 @@ export const analyzeCommand = async (inputPath?: string, options?: AnalyzeOption
bar.stop();
const msg = err.message || String(err);
// Registry name-collision from --name (#829) — surface as an
// actionable error rather than a generic stack-trace.
if (err instanceof RegistryNameCollisionError) {
console.error(`\n Registry name collision:\n`);
console.error(` "${err.registryName}" is already used by "${err.existingPath}".\n`);
console.error(` Options:`);
console.error(` • Pick a different alias: gitnexus analyze --name <alias>`);
console.error(
` • Allow the duplicate: gitnexus analyze --allow-duplicate-name (leaves "-r ${err.registryName}" ambiguous)`,
);
console.error('');
process.exitCode = 1;
return;
}
console.error(`\n Analysis failed: ${msg}\n`);
// Provide helpful guidance for known failure modes
+1 -25
View File
@@ -6,13 +6,7 @@
*/
import fs from 'fs/promises';
import {
findRepo,
unregisterRepo,
listRegisteredRepos,
assertSafeStoragePath,
UnsafeStoragePathError,
} from '../storage/repo-manager.js';
import { findRepo, unregisterRepo, listRegisteredRepos } from '../storage/repo-manager.js';
export const cleanCommand = async (options?: { force?: boolean; all?: boolean }) => {
// --all flag: clean all indexed repos
@@ -33,24 +27,6 @@ export const cleanCommand = async (options?: { force?: boolean; all?: boolean })
const entries = await listRegisteredRepos();
for (const entry of entries) {
// Safety guard (#1003 review — @magyargergo): same rationale as
// remove.ts. `~/.gitnexus/registry.json` is user-writable, so a
// corrupted or hand-edited entry could point storagePath at the
// repo root, an empty string, or anywhere else — and
// fs.rm(recursive: true) on any of those would be catastrophic.
// Skip poisoned entries without touching disk, but keep going
// through the rest of the registry (preserves the existing
// per-repo error-tolerance semantics of `clean --all`).
try {
assertSafeStoragePath(entry);
} catch (err) {
if (err instanceof UnsafeStoragePathError) {
console.error(`Refusing to clean ${entry.name}: ${err.message}`);
continue;
}
throw err;
}
try {
await fs.rm(entry.storagePath, { recursive: true, force: true });
await unregisterRepo(entry.path);
-77
View File
@@ -184,83 +184,6 @@ export function registerGroupCommands(program: Command): void {
}
});
group
.command('impact <name>')
.description('Cross-repo impact for a symbol in one member repo of a group')
.requiredOption('--target <symbol>', 'Symbol or file name to analyze')
.requiredOption(
'--repo <groupPath>',
'Member path from group.yaml (e.g. app/backend), not the indexed repo name',
)
.option('--direction <dir>', 'upstream or downstream', 'upstream')
.option('--service <path>', 'Optional monorepo service directory prefix (path filter)')
.option(
'--subgroup <path>',
'Optional prefix limiting which group repos participate in cross fan-out',
)
.option('--max-depth <n>', 'Max graph traversal depth')
.option('--cross-depth <n>', 'Cross-repository hop depth')
.option('--min-confidence <n>', 'Minimum relation confidence (0–1)')
.option('--include-tests', 'Include test files in traversal', false)
.option('--timeout-ms <n>', 'Phase-1 local impact wall time in milliseconds')
.option('--json', 'JSON output')
.action(async (name: string, opts: Record<string, string | boolean | undefined>) => {
const { LocalBackend } = await import('../mcp/local/local-backend.js');
const backend = new LocalBackend();
try {
await backend.init();
const payload: Record<string, unknown> = {
name,
repo: opts.repo,
target: opts.target,
direction: (opts.direction as string) || 'upstream',
};
if (opts.service) payload.service = opts.service;
if (opts.subgroup) payload.subgroup = opts.subgroup;
if (opts.maxDepth !== undefined && opts.maxDepth !== '') {
const n = parseInt(String(opts.maxDepth), 10);
if (!Number.isNaN(n)) payload.maxDepth = n;
}
if (opts.crossDepth !== undefined && opts.crossDepth !== '') {
const n = parseInt(String(opts.crossDepth), 10);
if (!Number.isNaN(n)) payload.crossDepth = n;
}
if (opts.minConfidence !== undefined && opts.minConfidence !== '') {
const n = parseFloat(String(opts.minConfidence));
if (!Number.isNaN(n)) payload.minConfidence = n;
}
if (opts.timeoutMs !== undefined && opts.timeoutMs !== '') {
const n = parseInt(String(opts.timeoutMs), 10);
if (!Number.isNaN(n)) payload.timeoutMs = n;
}
if (opts.includeTests) payload.includeTests = true;
const raw = await backend.getGroupService().groupImpact(payload);
if (raw && typeof raw === 'object' && 'error' in raw) {
console.error(String((raw as { error: string }).error));
process.exitCode = 1;
return;
}
if (opts.json) {
console.log(JSON.stringify(raw, null, 2));
} else {
const summary = (raw as { summary?: Record<string, number> })?.summary;
const risk = (raw as { risk?: string })?.risk;
console.log(`Group impact for "${name}" (${String(opts.repo)}): risk=${risk ?? '?'}`);
if (summary) {
console.log(
` direct=${summary.direct ?? 0} processes=${summary.processes_affected ?? 0} cross=${summary.cross_repo_hits ?? 0}`,
);
}
}
} finally {
await backend.dispose().catch(() => {});
}
});
group
.command('query <name> <query>')
.description('Search execution flows across all repos in a group')

Some files were not shown because too many files have changed in this diff Show More