Compare commits

..
Author SHA1 Message Date
Gergo Magyar 3768e3bfd0 fix(server): log skip-embedding count and table-not-found swallow path
Addresses review feedback on PR #823:
- Log count of already-embedded nodes when skipNodeIds is populated
  (aids debugging if Kuzu driver row shape changes).
- Log when the 'table does not exist' swallow path fires so ops can
  catch it if Kuzu ever changes error wording.
- Document the {} config positional argument with an inline comment
  referencing the runEmbeddingPipeline signature.
2026-04-15 07:40:58 +01:00
Gergo Magyar 3384575ac6 style: prettier format gitnexus/src/server/api.ts 2026-04-15 07:38:16 +01:00
jonasvanderhaegen-xve dd194d56b1 fix(server): narrow catch to table-not-exist errors only in POST /api/embed
Bare catch{} would silently swallow connection errors and proceed to
re-embed all nodes, hiding infrastructure issues. Now only swallows
errors where the CodeEmbedding table does not yet exist.
2026-04-14 14:27:03 +02:00
jonasvanderhaegen-xve 80a6fde2ba fix(server): skip already-embedded nodes in POST /api/embed to avoid vector-index SET error
Kuzu/LadybugDB forbids SET on a property that is part of a vector index.
The /api/embed endpoint was calling runEmbeddingPipeline without skipNodeIds,
causing it to attempt MERGE+SET on every node including those already embedded.

Fix: query existing CodeEmbedding nodeIds before running the pipeline and pass
them as skipNodeIds so only new (unembedded) nodes are processed.
2026-04-14 13:33:48 +02:00
jonasvanderhaegen-xve 8d38cc99fa fix(embeddings): use MERGE instead of CREATE for CodeEmbedding inserts
CREATE fails with duplicate PK when a CodeEmbedding node already exists,
which happens when:
- A PostToolUse hook triggers a concurrent gitnexus analyze during an
  active analyze run (git commits fire the hook)
- A partial prior run left some embeddings in the DB before a crash

Switching to MERGE makes the insert idempotent: existing embeddings are
updated in place, new ones are created, no PK violations.

Fixes: #822
2026-04-14 12:59:14 +02:00
jonasvanderhaegen-xve 41844edf88 fix(csv-generator): deduplicate all node types, not just File nodes
The pipeline can produce duplicate node IDs across all symbol types
(Class, Method, Function, etc.). Only File nodes were guarded by a
seenFileIds Set, leaving every other type unprotected. When the CSV
was COPY'd into LadybugDB, duplicate PKs caused mass "Batch execution
error: Found duplicated primary key value" warnings on gitnexus serve.

Replace the per-type seenFileIds with a single seenNodeIds Set checked
at the top of the iteration loop, before the switch, so every label is
covered by the same O(1) deduplication guard.

Fixes: #822
2026-04-14 12:32:08 +02:00
260 changed files with 3493 additions and 18105 deletions
-1
View File
@@ -1 +0,0 @@
plans/
-19
View File
@@ -1,19 +0,0 @@
.git
.gitignore
.DS_Store
node_modules
**/node_modules
dist
**/dist
coverage
**/coverage
.env
.env.local
.env.*.local
.gitnexus
gitnexus-web/playwright-report
gitnexus-web/test-results
-3
View File
@@ -1,3 +0,0 @@
IMAGE_NAME=ghcr.io/abhigyanpatwari/gitnexus:latest
CONTAINER_NAME=gitnexus
HOST_PORT=4173
-75
View File
@@ -1,75 +0,0 @@
version: 2
updates:
# Keep third-party Actions SHA pins current. See CONTRIBUTING.md — when
# reviewing these bumps, verify the SHA corresponds to the claimed tag by
# running `gh api repos/<owner>/<action>/git/refs/tags/<tag>` before merge.
- package-ecosystem: github-actions
directory: /
schedule:
interval: weekly
open-pull-requests-limit: 5
commit-message:
prefix: chore
include: scope
labels:
- dependencies
- ci
# Gitnexus npm deps — tree-sitter grammars checked daily so we catch
# new releases that unblock the tree-sitter 0.25 upgrade ASAP. Grammars
# are grouped so lockstep bumps produce a single PR. The tree-sitter
# RUNTIME is pinned — upgrade deliberately via the drift check workflow.
# See .github/scripts/check-tree-sitter-upgrade-readiness.py for
# the upgrade readiness tracker.
- package-ecosystem: npm
directory: /gitnexus
schedule:
interval: daily
open-pull-requests-limit: 10
commit-message:
prefix: chore(deps)
include: scope
labels:
- dependencies
groups:
tree-sitter-grammars:
patterns:
- tree-sitter-*
exclude-patterns:
- tree-sitter
- tree-sitter-cli
ignore:
# Pin the tree-sitter runtime at 0.21.x until the drift check
# reports all grammars are peer-dep compatible with 0.25.
- dependency-name: tree-sitter
update-types:
- version-update:semver-major
- version-update:semver-minor
# tree-sitter-cli follows the runtime's version cadence. Bump when
# regenerating vendor/tree-sitter-proto/src/parser.c, not on a schedule.
- dependency-name: tree-sitter-cli
# gitnexus-web (thin frontend client).
- package-ecosystem: npm
directory: /gitnexus-web
schedule:
interval: weekly
open-pull-requests-limit: 5
commit-message:
prefix: chore(deps)
include: scope
labels:
- dependencies
- frontend
# Shared types package.
- package-ecosystem: npm
directory: /gitnexus-shared
schedule:
interval: weekly
open-pull-requests-limit: 5
commit-message:
prefix: chore(deps)
include: scope
labels:
- dependencies
-53
View File
@@ -1,53 +0,0 @@
# release-drafter config — used only for PR autolabeling by
# `.github/workflows/pr-labeler.yml` (the workflow passes `disable-releaser: true`,
# so the draft-release side of release-drafter never runs).
#
# The labels applied here are the same ones `.github/release.yml` maps to
# categorized release-notes sections.
#
# `sync-labels: true` removes managed autolabels that no longer match the PR —
# critical for the breaking-change case: if a PR title drops the `!` or the body
# drops `BREAKING CHANGE:`, the `breaking` label is pulled off automatically.
# Required by release-drafter; not used because releaser is disabled.
name-template: 'unused'
tag-template: 'unused'
template: |
$CHANGES
sync-labels: true
autolabeler:
- label: enhancement
title:
- '/^feat(\([^)]+\))?!?:/i'
- label: bug
title:
- '/^fix(\([^)]+\))?!?:/i'
- label: performance
title:
- '/^perf(\([^)]+\))?!?:/i'
- label: refactor
title:
- '/^refactor(\([^)]+\))?!?:/i'
- label: documentation
title:
- '/^docs(\([^)]+\))?!?:/i'
- label: test
title:
- '/^test(\([^)]+\))?!?:/i'
- label: ci
title:
- '/^ci(\([^)]+\))?!?:/i'
- label: dependencies
title:
- '/^(build|deps)(\([^)]+\))?!?:/i'
- label: chore
title:
- '/^(chore|revert)(\([^)]+\))?!?:/i'
# Breaking-change marker: either `!` in the type prefix or `BREAKING CHANGE:` in body.
- label: breaking
title:
- '/^[a-z]+(\([^)]+\))?!:/i'
body:
- '/BREAKING[ -]CHANGE:/i'
@@ -1,358 +0,0 @@
#!/usr/bin/env python3
"""Monitor tree-sitter 0.25 upgrade readiness.
Tracks two things Dependabot cannot see:
1. Peer-dep compatibility. Each tree-sitter-* grammar declares a peer
dependency on the tree-sitter runtime. We want to know when every
grammar's *latest npm release* satisfies tree-sitter@0.25.0 so we
can upgrade without --legacy-peer-deps.
2. Vendored upstream drift. vendor/tree-sitter-proto/ is a snapshot of
coder3101/tree-sitter-proto's parser.c. When upstream moves, we want
to know whether we can pick it up.
Invoked from .github/workflows/tree-sitter-upgrade-readiness.yml daily.
Runs locally too:
python3 .github/scripts/check-tree-sitter-upgrade-readiness.py
Outputs Markdown to stdout. Exit 0 when every grammar is upgrade-ready
and the vendored proto is in sync. Exit 1 when blockers remain (the
workflow uses this to open or update a tracking issue).
No external deps -- stdlib only, so it runs on any vanilla runner.
"""
from __future__ import annotations
import json
import os
import pathlib
import re
import sys
import urllib.error
import urllib.request
REPO_ROOT = pathlib.Path(__file__).resolve().parents[2]
GITNEXUS_DIR = REPO_ROOT / "gitnexus"
VENDOR_PROTO_DIR = GITNEXUS_DIR / "vendor" / "tree-sitter-proto"
# ── Upgrade target ──────────────────────────────────────────────────────
# The runtime version we want to upgrade TO. Update this when the goal
# changes (e.g. once 0.25 lands and we target 0.26).
TARGET_RUNTIME = "0.25.0"
TARGET_RUNTIME_MAJOR_MINOR = ".".join(TARGET_RUNTIME.split(".")[:2])
# Tree-sitter runtime -> (min_abi, max_abi) it can load. Only the current
# and target entries matter; extend when changing TARGET_RUNTIME.
RUNTIME_ABI_RANGES: dict[str, tuple[int, int]] = {
"0.21": (13, 14),
"0.25": (13, 15),
}
assert TARGET_RUNTIME_MAJOR_MINOR in RUNTIME_ABI_RANGES, (
f"RUNTIME_ABI_RANGES has no entry for {TARGET_RUNTIME_MAJOR_MINOR!r}. "
f"Add the ABI range after auditing the upstream release notes."
)
# Grammars we use. Values are the upstream GitHub repos to check for
# unreleased ABI bumps (owner/repo, branch, parser.c path).
GRAMMARS: dict[str, tuple[str, str, str]] = {
"tree-sitter-c": ("tree-sitter/tree-sitter-c", "master", "src/parser.c"),
"tree-sitter-c-sharp": ("tree-sitter/tree-sitter-c-sharp", "master", "src/parser.c"),
"tree-sitter-cpp": ("tree-sitter/tree-sitter-cpp", "master", "src/parser.c"),
"tree-sitter-dart": ("UserNobody14/tree-sitter-dart", "master", "src/parser.c"),
"tree-sitter-go": ("tree-sitter/tree-sitter-go", "master", "src/parser.c"),
"tree-sitter-java": ("tree-sitter/tree-sitter-java", "master", "src/parser.c"),
"tree-sitter-javascript": ("tree-sitter/tree-sitter-javascript", "master", "src/parser.c"),
"tree-sitter-kotlin": ("fwcd/tree-sitter-kotlin", "main", "src/parser.c"),
"tree-sitter-php": ("tree-sitter/tree-sitter-php", "master", "php/src/parser.c"),
"tree-sitter-python": ("tree-sitter/tree-sitter-python", "master", "src/parser.c"),
"tree-sitter-ruby": ("tree-sitter/tree-sitter-ruby", "master", "src/parser.c"),
"tree-sitter-rust": ("tree-sitter/tree-sitter-rust", "master", "src/parser.c"),
"tree-sitter-swift": ("alex-pinkus/tree-sitter-swift", "main", "src/parser.c"),
"tree-sitter-typescript": ("tree-sitter/tree-sitter-typescript", "master", "typescript/src/parser.c"),
}
UPSTREAM_PROTO_OWNER = "coder3101"
UPSTREAM_PROTO_REPO = "tree-sitter-proto"
UPSTREAM_PROTO_BRANCH = "main"
# ── Helpers ─────────────────────────────────────────────────────────────
def read_current_runtime() -> str:
"""Return the tree-sitter runtime version pinned in package.json (e.g. '0.21')."""
pkg = json.loads((GITNEXUS_DIR / "package.json").read_text())
raw = pkg["dependencies"]["tree-sitter"]
match = re.search(r"(\d+)\.(\d+)", raw)
if not match:
raise SystemExit(f"could not parse tree-sitter version: {raw!r}")
return f"{match.group(1)}.{match.group(2)}"
def npm_view_json(pkg: str) -> dict | None:
"""Fetch package metadata from the npm registry via HTTPS.
Uses the registry API directly so we don't depend on the npm CLI
being available (it's a batch file on Windows which complicates
subprocess calls).
"""
url = f"https://registry.npmjs.org/{pkg}/latest"
try:
req = urllib.request.Request(url, headers={"Accept": "application/json"})
with urllib.request.urlopen(req, timeout=8) as resp:
return json.loads(resp.read().decode("utf-8"))
except (urllib.error.URLError, urllib.error.HTTPError, json.JSONDecodeError):
return None
def satisfies_target(peer_range: str | None, target: str) -> bool:
"""Check if a semver range like '^0.22.4' or '^0.25.0' satisfies the target.
Simple heuristic: extract the minimum version from the range and check
if target >= min. For caret ranges (^X.Y.Z), the upper bound is the
next major (for X>0) or next minor (for X==0). We check both bounds.
"""
if peer_range is None:
# No peer dep declared = no constraint = compatible.
return True
match = re.search(r"(\d+)\.(\d+)\.(\d+)", peer_range)
if not match:
return False
min_major, min_minor, min_patch = int(match.group(1)), int(match.group(2)), int(match.group(3))
t_match = re.search(r"(\d+)\.(\d+)\.(\d+)", target)
if not t_match:
return False
t_major, t_minor, t_patch = int(t_match.group(1)), int(t_match.group(2)), int(t_match.group(3))
# Target must be >= minimum.
target_tuple = (t_major, t_minor, t_patch)
min_tuple = (min_major, min_minor, min_patch)
if target_tuple < min_tuple:
return False
# For caret ranges with major 0: ^0.X.Y allows [0.X.Y, 0.(X+1).0).
if peer_range.startswith("^") and min_major == 0:
if t_major != 0 or t_minor >= min_minor + 1:
return False
# For caret ranges with major >0: ^X.Y.Z allows [X.Y.Z, (X+1).0.0).
elif peer_range.startswith("^") and min_major > 0:
if t_major >= min_major + 1:
return False
return True
_GITHUB_TOKEN = os.environ.get("GITHUB_TOKEN")
def fetch_text(url: str, timeout: int = 8) -> str | None:
"""Fetch a URL and return its text, or None on failure.
Adds an Authorization header for github.com URLs when GITHUB_TOKEN is
set (raises the rate limit from 60 to 5 000 requests/hour).
"""
headers: dict[str, str] = {}
if _GITHUB_TOKEN and ("github.com" in url or "githubusercontent.com" in url):
headers["Authorization"] = f"Bearer {_GITHUB_TOKEN}"
try:
req = urllib.request.Request(url, headers=headers)
with urllib.request.urlopen(req, timeout=timeout) as resp:
return resp.read().decode("utf-8", errors="ignore")
except (urllib.error.URLError, urllib.error.HTTPError):
return None
def extract_abi_from_text(text: str) -> int | None:
"""Extract LANGUAGE_VERSION from parser.c text."""
match = re.search(r"#define\s+LANGUAGE_VERSION\s+(\d+)", text[:4096])
return int(match.group(1)) if match else None
def extract_language_version(parser_c: pathlib.Path) -> int | None:
"""Return the LANGUAGE_VERSION defined in a parser.c, or None if absent."""
if not parser_c.is_file():
return None
with parser_c.open("r", encoding="utf-8", errors="ignore") as fh:
head = fh.read(4096)
return extract_abi_from_text(head)
def md_h(text: str, level: int = 2) -> str:
return f"{'#' * level} {text}\n"
# ── Main ────────────────────────────────────────────────────────────────
def main() -> int:
blockers: dict[str, str] = {}
lines: list[str] = []
lines.append(md_h("Tree-sitter 0.25 upgrade readiness", 1))
lines.append("")
current_runtime = read_current_runtime()
current_abi_range = RUNTIME_ABI_RANGES.get(current_runtime, (0, 0))
target_abi_range = RUNTIME_ABI_RANGES.get(TARGET_RUNTIME_MAJOR_MINOR, (0, 0))
lines.append(f"- Current runtime: `tree-sitter@{current_runtime}.x` (ABI {current_abi_range[0]}..{current_abi_range[1]})")
lines.append(f"- Target runtime: `tree-sitter@{TARGET_RUNTIME}` (ABI {target_abi_range[0]}..{target_abi_range[1]})")
lines.append("")
# ── Grammar peer-dep compatibility ───────────────────────────────
lines.append(md_h("Grammar compatibility", 2))
lines.append("| Grammar | npm latest | Peer dep | Satisfies 0.25? | ABI | Upstream ABI | Status |")
lines.append("|---|---|---|---|---|---|---|")
ready_count = 0
total_count = len(GRAMMARS)
for name, (upstream_repo, upstream_branch, parser_path) in sorted(GRAMMARS.items()):
# Fetch latest npm metadata.
info = npm_view_json(name)
fetch_failed = info is None
npm_version = "?"
peer_range = None
peer_optional = True
if info:
npm_version = info.get("version", "?")
peers = info.get("peerDependencies") or {}
peer_range = peers.get("tree-sitter")
meta = info.get("peerDependenciesMeta") or {}
ts_meta = meta.get("tree-sitter") or {}
peer_optional = ts_meta.get("optional", False) if peer_range else True
if fetch_failed:
peer_display = "? (fetch failed)"
compatible = False
else:
peer_display = peer_range or "none"
if peer_range and not peer_optional:
peer_display += " (required)"
compatible = satisfies_target(peer_range, TARGET_RUNTIME)
# Check installed ABI using the same parser_path from GRAMMARS.
installed_parser = GITNEXUS_DIR / "node_modules" / name / parser_path
if not installed_parser.is_file():
# Fallback to default location.
installed_parser = GITNEXUS_DIR / "node_modules" / name / "src" / "parser.c"
installed_abi = extract_language_version(installed_parser)
abi_display = str(installed_abi) if installed_abi else "?"
# Check upstream (main/master branch) ABI for unreleased work.
upstream_url = (
f"https://raw.githubusercontent.com/{upstream_repo}/"
f"{upstream_branch}/{parser_path}"
)
upstream_text = fetch_text(upstream_url)
upstream_abi = extract_abi_from_text(upstream_text) if upstream_text else None
upstream_abi_display = str(upstream_abi) if upstream_abi else "?"
# Determine status.
if fetch_failed:
status = "Unknown (fetch failed)"
blockers[name] = f"`{name}`: npm registry fetch failed — could not verify peer dep"
elif compatible:
status = "Ready"
ready_count += 1
elif upstream_abi and upstream_abi >= 15:
status = "Unreleased (ABI 15 on main)"
blockers[name] = f"`{name}`: ABI 15 on `{upstream_repo}` main but not published to npm"
else:
status = "Blocking"
blockers[name] = f"`{name}@{npm_version}`: peer `{peer_display}` incompatible with 0.25"
# Also check upstream package.json for relaxed peer dep.
if not compatible and not fetch_failed:
upstream_pkg_url = (
f"https://raw.githubusercontent.com/{upstream_repo}/"
f"{upstream_branch}/package.json"
)
upstream_pkg_text = fetch_text(upstream_pkg_url)
if upstream_pkg_text:
try:
upstream_pkg = json.loads(upstream_pkg_text)
upstream_peer = (upstream_pkg.get("peerDependencies") or {}).get("tree-sitter")
if upstream_peer and satisfies_target(upstream_peer, TARGET_RUNTIME):
status = "Unreleased (peer relaxed on main)"
blockers[name] = f"`{name}`: peer dep relaxed on `{upstream_repo}` main but not published to npm"
except json.JSONDecodeError:
pass
compat_icon = "Yes" if compatible else "**No**"
lines.append(
f"| `{name}` | {npm_version} | {peer_display} | {compat_icon} | {abi_display} | {upstream_abi_display} | {status} |"
)
lines.append("")
lines.append(f"**{ready_count}/{total_count}** grammars ready for `tree-sitter@{TARGET_RUNTIME}`.")
lines.append("")
# ── Vendored proto drift ─────────────────────────────────────────
lines.append(md_h("Vendored tree-sitter-proto", 2))
vendored_abi = extract_language_version(VENDOR_PROTO_DIR / "src" / "parser.c")
upstream_proto_url = (
f"https://raw.githubusercontent.com/{UPSTREAM_PROTO_OWNER}/"
f"{UPSTREAM_PROTO_REPO}/{UPSTREAM_PROTO_BRANCH}/src/parser.c"
)
upstream_proto_text = fetch_text(upstream_proto_url)
upstream_proto_abi = extract_abi_from_text(upstream_proto_text) if upstream_proto_text else None
sha_url = (
f"https://api.github.com/repos/{UPSTREAM_PROTO_OWNER}/"
f"{UPSTREAM_PROTO_REPO}/commits/{UPSTREAM_PROTO_BRANCH}"
)
sha_text = fetch_text(sha_url)
upstream_sha = "?"
if sha_text:
try:
upstream_sha = json.loads(sha_text).get("sha", "?")[:12]
except json.JSONDecodeError:
pass
local_proto_path = VENDOR_PROTO_DIR / "src" / "parser.c"
local_proto_text = local_proto_path.read_text(encoding="utf-8", errors="ignore") if local_proto_path.is_file() else ""
in_sync = bool(
upstream_proto_text
and local_proto_text.replace("\r\n", "\n")
== upstream_proto_text.replace("\r\n", "\n")
)
lines.append(f"- Upstream: `{UPSTREAM_PROTO_OWNER}/{UPSTREAM_PROTO_REPO}@{UPSTREAM_PROTO_BRANCH}` (HEAD `{upstream_sha}`)")
lines.append(f"- Upstream ABI: **{upstream_proto_abi}**")
lines.append(f"- Vendored ABI: **{vendored_abi}**")
lines.append(f"- In sync: {'yes' if in_sync else 'no — upstream has diverged'}")
if upstream_proto_abi and vendored_abi and upstream_proto_abi > vendored_abi:
can_upgrade = upstream_proto_abi <= target_abi_range[1]
lines.append(f"- Upstream ABI {upstream_proto_abi} {'is' if can_upgrade else 'is NOT'} within target runtime range ({target_abi_range[0]}..{target_abi_range[1]})")
if can_upgrade:
lines.append(f"- **Action:** after upgrading to tree-sitter@{TARGET_RUNTIME}, regenerate vendored parser.c from upstream `{upstream_sha}`")
else:
lines.append(f"- **Action:** wait for runtime upgrade beyond {TARGET_RUNTIME} that supports ABI {upstream_proto_abi}")
blockers["vendored-proto-abi"] = f"vendored tree-sitter-proto: upstream ABI {upstream_proto_abi} outside target range"
elif not in_sync:
lines.append("- **Action:** review upstream changes; vendored copy may need updating")
blockers["vendored-proto-sync"] = "vendored tree-sitter-proto: out of sync with upstream"
# ── Summary ──────────────────────────────────────────────────────
lines.append("")
lines.append(md_h("Summary", 2))
if blockers:
lines.append(f"**{len(blockers)} blocker(s) remaining:**\n")
for b in blockers.values():
lines.append(f"- {b}")
lines.append("")
lines.append("Upgrade to `tree-sitter@0.25` is **blocked**.")
else:
lines.append("All grammars are compatible. Upgrade to `tree-sitter@0.25` is **ready**.")
print("\n".join(lines))
return 1 if blockers else 0
if __name__ == "__main__":
sys.exit(main())
@@ -1,173 +0,0 @@
#!/usr/bin/env python3
"""Enforce the GitHub Actions concurrency convention.
See CONTRIBUTING.md -> "GitHub Actions — Concurrency Convention" for the rules.
Invoked from .github/workflows/ci-quality.yml. Runs locally too:
python3 .github/scripts/check-workflow-concurrency.py .github/workflows
Rules:
1. Every entry-point (non-reusable) workflow declares a top-level
`concurrency:` block.
2. Reusable workflows (on: workflow_call ONLY) do NOT declare one.
3. The `concurrency.group` expression MUST reference either
`${{ github.workflow }}` or a literal `CI-` prefix (the documented
ci.yml reusable-workflow-safe exception). This is checked by substring
containment rather than prefix match because ci.yml's group is a
conditional expression that resolves to a `CI-…` literal at runtime.
We deliberately do not use a YAML library — keeps the script dependency-free
on any vanilla runner. `on:` block parsing is line-based and handles both the
flat (`on: workflow_call`) and mapping (`on:\n workflow_call:`) forms.
"""
from __future__ import annotations
import pathlib
import re
import sys
REQUIRED_TOKENS = ("${{ github.workflow }}", "CI-")
def is_reusable(lines: list[str]) -> bool:
"""Return True iff the workflow's `on:` block names only `workflow_call`."""
in_on = False
on_indent: int | None = None
keys: list[str] = []
for raw in lines:
# Skip blank lines and comments
stripped = raw.strip()
if not stripped or stripped.startswith("#"):
continue
indent = len(raw) - len(raw.lstrip(" "))
if not in_on:
if raw.startswith("on:"):
remainder = raw[len("on:"):].strip()
if not remainder:
# `on:` followed by indented mapping on next lines
in_on = True
on_indent = indent
continue
if remainder.startswith("[") and remainder.endswith("]"):
# Flow-style list: on: [workflow_call]
items = [
item.strip() for item in remainder.strip("[]").split(",")
]
return items == ["workflow_call"]
# Scalar form: on: workflow_call (or a single other event)
return remainder == "workflow_call"
continue
# Inside the `on:` block; stop when indentation returns to <= on_indent
if on_indent is not None and indent <= on_indent:
break
# Only consider keys at on_indent + indentation step (anything deeper
# is nested config like `types:`)
if ":" not in stripped:
continue
# Heuristic: first-level event keys are those with indent == on_indent + 2
# (the canonical step for a 2-space YAML doc). We collect all first-level
# keys by tracking the smallest indent seen inside the block.
keys.append((indent, stripped.split(":", 1)[0].strip()))
if not keys:
return False
# Take only the outermost-indented keys as the event list
min_indent = min(i for i, _ in keys)
events = [name for i, name in keys if i == min_indent]
return events == ["workflow_call"]
CONCURRENCY_RE = re.compile(r"^concurrency:\s*$")
GROUP_RE = re.compile(r"^\s+group:\s*(.+?)\s*$")
def extract_group_key(lines: list[str]) -> str | None:
"""Return the `group:` value of the top-level `concurrency:` block, or None."""
for idx, raw in enumerate(lines):
if CONCURRENCY_RE.match(raw):
# Scan forward until we leave the concurrency block (next top-level key
# is at column 0 and ends with `:`).
for follow in lines[idx + 1:]:
if follow and not follow.startswith(" ") and follow.rstrip().endswith(":"):
break
m = GROUP_RE.match(follow)
if m:
return m.group(1).strip().strip("'").strip('"')
break
return None
def has_top_level_concurrency(lines: list[str]) -> bool:
return any(CONCURRENCY_RE.match(raw) for raw in lines)
def check(workflows_dir: pathlib.Path) -> int:
fail = 0
files = sorted(
list(workflows_dir.glob("*.yml")) + list(workflows_dir.glob("*.yaml"))
)
for path in files:
lines = path.read_text(encoding="utf-8").splitlines()
reusable = is_reusable(lines)
has_conc = has_top_level_concurrency(lines)
if reusable:
if has_conc:
print(
f"::error file={path}::Reusable workflow (on: workflow_call) "
"must NOT declare its own concurrency block — it inherits "
"from the caller. See CONTRIBUTING.md -> GitHub Actions — "
"Concurrency Convention."
)
fail = 1
continue
if not has_conc:
print(
f"::error file={path}::Missing top-level concurrency block. "
"See CONTRIBUTING.md -> GitHub Actions — Concurrency Convention."
)
fail = 1
continue
group = extract_group_key(lines)
if group is None:
print(
f"::error file={path}::concurrency block is missing a "
"`group:` key."
)
fail = 1
continue
if not any(token in group for token in REQUIRED_TOKENS):
print(
f"::error file={path}::concurrency.group `{group}` must "
f"reference one of {REQUIRED_TOKENS}. See CONTRIBUTING.md -> "
"GitHub Actions — Concurrency Convention."
)
fail = 1
return fail
def main(argv: list[str]) -> int:
if len(argv) != 2:
print(f"usage: {argv[0]} <workflows-dir>", file=sys.stderr)
return 2
workflows_dir = pathlib.Path(argv[1])
if not workflows_dir.is_dir():
print(f"not a directory: {workflows_dir}", file=sys.stderr)
return 2
return check(workflows_dir)
if __name__ == "__main__":
sys.exit(main(sys.argv))
+4 -4
View File
@@ -11,8 +11,8 @@ jobs:
outputs:
web_changed: ${{ steps.filter.outputs.web }}
steps:
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
- uses: dorny/paths-filter@fbd0ab8f3e69293af611ebaee6363fc25e6d187d # v3
- uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4
- uses: dorny/paths-filter@de90cc6fb38fc0963ad72b210f1f284cd68cea36 # v3
id: filter
with:
filters: |
@@ -26,7 +26,7 @@ jobs:
runs-on: ubuntu-latest
timeout-minutes: 20
steps:
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
- uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4
- uses: ./.github/actions/setup-gitnexus-web
@@ -74,7 +74,7 @@ jobs:
- name: Upload test results
if: always()
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1
uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4
with:
name: e2e-results
path: |
+6 -29
View File
@@ -8,8 +8,8 @@ jobs:
runs-on: ubuntu-latest
timeout-minutes: 5
steps:
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
- uses: actions/setup-node@53b83947a5a98c8d113130e565377fae1a50d02f # v6.3.0
- uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4
- uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4
with:
node-version: 20
cache: npm
@@ -21,8 +21,8 @@ jobs:
runs-on: ubuntu-latest
timeout-minutes: 10
steps:
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
- uses: actions/setup-node@53b83947a5a98c8d113130e565377fae1a50d02f # v6.3.0
- uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4
- uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4
with:
node-version: 20
cache: npm
@@ -34,7 +34,7 @@ jobs:
runs-on: ubuntu-latest
timeout-minutes: 10
steps:
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
- uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4
- uses: ./.github/actions/setup-gitnexus
- run: npx tsc --noEmit
working-directory: gitnexus
@@ -43,30 +43,7 @@ jobs:
runs-on: ubuntu-latest
timeout-minutes: 10
steps:
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
- uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4
- uses: ./.github/actions/setup-gitnexus-web
- run: npx tsc -b --noEmit
working-directory: gitnexus-web
# Enforces the convention documented in CONTRIBUTING.md → "GitHub Actions —
# Concurrency Convention":
# 1. Every entry-point (non-reusable) workflow declares a top-level
# `concurrency:` block.
# 2. Reusable workflows (`on: workflow_call` only) do NOT declare one —
# they inherit concurrency from the caller.
# 3. The concurrency group key starts with `${{ github.workflow }}` or
# the literal `CI-` prefix (the documented ci.yml exception for
# reusable-workflow-safe grouping).
# Reusability is detected by parsing each workflow's `on:` block, not an
# allowlist, so new reusable workflows never produce false positives.
workflow-convention:
name: Workflow concurrency convention
runs-on: ubuntu-latest
timeout-minutes: 5
steps:
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
- name: Validate workflow concurrency convention
shell: bash
run: |
set -euo pipefail
python3 .github/scripts/check-workflow-concurrency.py .github/workflows
+4 -14
View File
@@ -14,16 +14,6 @@ permissions:
contents: read # needed for sparse checkout of vitest.config.ts
pull-requests: write # needed to post sticky PR comment
# Concurrency convention: see CONTRIBUTING.md → "GitHub Actions — Concurrency Convention".
# Serialize sticky-comment writes per PR so two rapid CI completions don't race.
# Internal PRs surface in `pull_requests[0].number`. Fork PRs leave that array empty,
# so we fall back to `<head-repo-full-name>/<head-branch>`, which is stable across
# reruns and subsequent pushes for the same fork PR (unlike `workflow_run.id` which
# is unique per run and therefore does not serialize anything).
concurrency:
group: ${{ github.workflow }}-${{ github.event.workflow_run.pull_requests[0].number || format('{0}/{1}', github.event.workflow_run.head_repository.full_name, github.event.workflow_run.head_branch) }}
cancel-in-progress: false
jobs:
pr-report:
name: PR Report
@@ -36,7 +26,7 @@ jobs:
steps:
# ── Download artifacts from the CI run ────────────────────────
- name: Download artifacts
uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v7
uses: actions/github-script@60a0d83039c74a4aee543508d2ffcb1c3799cdea # v7
with:
script: |
const fs = require('fs');
@@ -123,7 +113,7 @@ jobs:
- name: Checkout (for vitest config)
if: steps.meta.outputs.skip != 'true'
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4
with:
sparse-checkout: gitnexus/vitest.config.ts
sparse-checkout-cone-mode: false
@@ -132,7 +122,7 @@ jobs:
- name: Fetch base branch coverage
if: steps.meta.outputs.skip != 'true'
id: base-coverage
uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v7
uses: actions/github-script@60a0d83039c74a4aee543508d2ffcb1c3799cdea # v7
with:
script: |
const fs = require('fs');
@@ -416,7 +406,7 @@ jobs:
- name: Comment on PR
if: steps.meta.outputs.skip != 'true'
uses: marocchino/sticky-pull-request-comment@0ea0beb66eb9baf113663a64ec522f60e49231c0 # v2
uses: marocchino/sticky-pull-request-comment@773744901bac0e8cbb5a0dc842800d45e9b2b405 # v2
with:
header: ci-report
number: ${{ steps.meta.outputs.pr_number }}
+3 -6
View File
@@ -9,7 +9,7 @@ jobs:
runs-on: ubuntu-latest
timeout-minutes: 25
steps:
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
- uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4
- uses: ./.github/actions/setup-gitnexus
with:
build: 'true'
@@ -41,12 +41,9 @@ jobs:
--outputFile=web-test-results.json
working-directory: gitnexus-web
- name: Run docker-server integration tests
run: node --test docker-server.test.mjs
- name: Upload test reports
if: always()
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1
uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4
with:
name: test-reports
path: |
@@ -66,7 +63,7 @@ jobs:
runs-on: ${{ matrix.os }}
timeout-minutes: 25
steps:
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
- uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4
- uses: ./.github/actions/setup-gitnexus
with:
build: 'true'
+6 -17
View File
@@ -9,20 +9,9 @@ on:
paths-ignore: ['**.md', 'docs/**', 'LICENSE']
workflow_call:
# Concurrency convention: see CONTRIBUTING.md → "GitHub Actions — Concurrency Convention".
# Hardcoded `CI-` prefix (not `${{ github.workflow }}`) because this workflow is
# invoked as a reusable workflow from publish.yml and release-candidate.yml. In
# called-workflow context `github.workflow` evaluation is ambiguous across GitHub
# Actions versions, and a prefix that could resolve to the caller's name would
# share a concurrency group with the caller → deadlock. A literal prefix is
# immune. Direct `push`/`pull_request` invocations use `CI-<ref>`; invocations
# from a reusable-workflow caller fall into a per-run-unique group that never
# serializes with the caller.
# cancel-in-progress is event-aware: cancel superseded PR runs, queue every other
# event (push to main, workflow_call from publish.yml, etc.).
concurrency:
group: ${{ (github.event_name == 'pull_request' || github.event_name == 'push') && format('CI-{0}', github.ref) || format('CI-nested-{0}', github.run_id) }}
cancel-in-progress: ${{ github.event_name == 'pull_request' }}
group: ci-${{ github.ref }}
cancel-in-progress: true
# ── Reusable workflow orchestration ─────────────────────────────────
# Each concern lives in its own workflow file for maintainability:
@@ -85,7 +74,7 @@ jobs:
cp pr-meta/e2e_result pr-meta/e2e-result
- name: Upload PR metadata
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1
uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4
with:
name: pr-meta
path: pr-meta/
@@ -107,9 +96,9 @@ jobs:
TESTS: ${{ needs.tests.result }}
E2E: ${{ needs.e2e.result }}
run: |
echo "Quality: $QUALITY"
echo "Tests: $TESTS"
echo "E2E: $E2E"
echo "Quality: $QUALITY"
echo "Tests: $TESTS"
echo "E2E: $E2E"
if [[ "$QUALITY" != "success" ]] ||
[[ "$TESTS" != "success" ]]; then
echo "::error::Quality or test jobs failed"
+3 -4
View File
@@ -16,10 +16,9 @@ on:
issue_comment:
types: [created]
# Concurrency convention: see CONTRIBUTING.md → "GitHub Actions — Concurrency Convention".
# Serialize per-PR to avoid racing review comments.
concurrency:
group: ${{ github.workflow }}-${{ github.event.issue.number || github.event.pull_request.number }}
group: claude-review-${{ github.event.issue.number || github.event.pull_request.number }}
cancel-in-progress: false
jobs:
@@ -57,7 +56,7 @@ jobs:
# For issue_comment triggers, resolve the PR number, head SHA, and fork repo
- name: Resolve PR context
id: pr
uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v7
uses: actions/github-script@60a0d83039c74a4aee543508d2ffcb1c3799cdea # v7
with:
script: |
let pr;
@@ -77,7 +76,7 @@ jobs:
core.setOutput('branch', pr.head.ref);
- name: Checkout PR head
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4
with:
repository: ${{ steps.pr.outputs.repo }}
ref: ${{ steps.pr.outputs.sha }}
+3 -4
View File
@@ -10,10 +10,9 @@ on:
pull_request_review:
types: [submitted]
# Concurrency convention: see CONTRIBUTING.md → "GitHub Actions — Concurrency Convention".
# Serialize per-PR/issue to avoid racing comments.
concurrency:
group: ${{ github.workflow }}-${{ github.event.issue.number || github.event.pull_request.number || github.event.issue.id }}
group: claude-code-${{ github.event.issue.number || github.event.pull_request.number || github.event.issue.id }}
cancel-in-progress: false
jobs:
@@ -59,7 +58,7 @@ jobs:
# For PR-related triggers, resolve the fork repo so we can checkout correctly.
- name: Resolve PR context
id: pr
uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v7
uses: actions/github-script@60a0d83039c74a4aee543508d2ffcb1c3799cdea # v7
with:
script: |
// Determine if this event is PR-related
@@ -91,7 +90,7 @@ jobs:
core.setOutput('branch', pr.head.ref);
- name: Checkout repository
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4
with:
repository: ${{ steps.pr.outputs.is_pr == 'true' && steps.pr.outputs.repo || github.repository }}
ref: ${{ steps.pr.outputs.is_pr == 'true' && steps.pr.outputs.sha || '' }}
-71
View File
@@ -1,71 +0,0 @@
name: Docker Build & Push
on:
push:
tags:
- 'v*'
branches:
- main
paths-ignore: ['**.md', 'docs/**', 'LICENSE']
workflow_dispatch:
# Concurrency convention: see CONTRIBUTING.md → "GitHub Actions — Concurrency Convention".
# Tag refs are unique per release — distinct tags run in parallel.
# Pushes to main serialize; cancel superseded runs.
concurrency:
group: ${{ github.workflow }}-${{ github.ref }}
cancel-in-progress: ${{ github.ref == 'refs/heads/main' }}
jobs:
build-push:
name: Build & Push image
runs-on: ubuntu-latest
timeout-minutes: 30
permissions:
contents: read
packages: write
steps:
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
# Required for multi-platform (linux/arm64) emulation.
- name: Set up QEMU
uses: docker/setup-qemu-action@ce360397dd3f832beb865e1373c09c0e9f86d70a # v4.0.0
- name: Set up Docker Buildx
uses: docker/setup-buildx-action@4d04d5d9486b7bd6fa91e7baf45bbb4f8b9deedd # v4.0.0
- name: Log in to GitHub Container Registry
uses: docker/login-action@4907a6ddec9925e35a0a9e82d7399ccc52663121 # v4.1.0
with:
registry: ghcr.io
username: ${{ github.actor }}
password: ${{ secrets.GITHUB_TOKEN }}
# Computes image tags and labels from Git metadata:
# v* tag → ghcr.io/<owner>/<repo>:<semver> (e.g. 1.2.3, 1.2, 1)
# main push → ghcr.io/<owner>/<repo>:latest
- name: Extract Docker metadata
id: meta
uses: docker/metadata-action@030e881283bb7a6894de51c315a6bfe6a94e05cf # v6.0.0
with:
images: ghcr.io/${{ github.repository }}
tags: |
type=semver,pattern={{version}}
type=semver,pattern={{major}}.{{minor}}
type=semver,pattern={{major}}
type=raw,value=latest,enable={{is_default_branch}}
type=sha,prefix=sha-,format=short
- name: Build and push
uses: docker/build-push-action@bcafcacb16a39f128d818304e6c9c0c18556b85f # v7.1.0
with:
context: .
platforms: linux/amd64,linux/arm64
push: true
tags: ${{ steps.meta.outputs.tags }}
labels: ${{ steps.meta.outputs.labels }}
cache-from: type=gha
cache-to: type=gha,mode=max
build-args: |
BUILDPLATFORM=${{ runner.os == 'Linux' && 'linux/amd64' || 'linux/amd64' }}
+2 -3
View File
@@ -8,9 +8,8 @@ on:
permissions:
pull-requests: write
# Concurrency convention: see CONTRIBUTING.md → "GitHub Actions — Concurrency Convention".
concurrency:
group: ${{ github.workflow }}-${{ github.event.pull_request.number }}
group: pr-desc-${{ github.event.pull_request.number }}
cancel-in-progress: true
jobs:
@@ -19,7 +18,7 @@ jobs:
timeout-minutes: 5
steps:
- name: Check PR description quality
uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v9.0.0
uses: actions/github-script@ed597411d8f924073f98dfc5c65a23a2325f34cd # v8
with:
script: |
const MIN_BODY_LENGTH = 50;
-113
View File
@@ -1,113 +0,0 @@
name: PR Conventional Labeler
# Two workflows in one file with different triggers, matched to the minimum
# privilege each needs:
#
# validate-title (on: pull_request)
# Fork-safe. Runs with the PR-head's read-only GITHUB_TOKEN. Uses
# `amannn/action-semantic-pull-request` to fail the check when the PR
# title doesn't follow the conventional-commit format. Because the
# action only reads the event payload, no fork-controlled code runs.
#
# autolabel (on: pull_request_target)
# Needs `pull-requests: write` to apply labels, so must be
# pull_request_target. Uses `release-drafter/release-drafter` with
# `dry-run: true` to only run the autolabeler against the
# `.github/release-drafter.yml` config from the BASE ref (release-
# drafter reads the config from the repository's default branch, NOT
# the PR head — verify with `gh api repos/release-drafter/release-drafter/contents/...`
# or a fork-test PR before merging if the repo is high-value).
# `sync-labels: true` in the config removes managed autolabels that no
# longer match (e.g. when `!` or `BREAKING CHANGE:` is dropped).
#
# Title format: <type>[(scope)][!]: <subject>
# Allowed types: feat, fix, perf, refactor, docs, test, ci, build, chore, revert, deps
# Trailing `!` on the type marks a breaking change.
# See CONTRIBUTING.md → "Pull request titles".
on:
pull_request:
# Title-only changes fire `edited`. `opened` and `reopened` cover creation.
# `synchronize` (push to the PR branch) is intentionally excluded — titles
# don't change on push, so it only wastes CI minutes and broadens the
# privileged-token exposure window on the autolabel job.
types: [opened, edited, reopened]
pull_request_target:
types: [opened, edited, reopened]
# Concurrency convention: see CONTRIBUTING.md → "GitHub Actions — Concurrency Convention".
# Include `github.event_name` so `pull_request` (validate-title) and
# `pull_request_target` (autolabel) runs for the same PR do NOT share a slot
# and therefore cannot cancel each other — a cancelled required-check would
# permanently block merge until the next title edit.
# Within each trigger the latest title edit still supersedes the prior run.
concurrency:
group: ${{ github.workflow }}-${{ github.event_name }}-${{ github.event.pull_request.number }}
cancel-in-progress: true
jobs:
validate-title:
# Fork-safe job — only runs on `pull_request` (not `pull_request_target`).
# Token is read-only; writes a commit status that branch protection can
# require before merge.
name: Validate PR title
if: github.event_name == 'pull_request'
runs-on: ubuntu-latest
timeout-minutes: 5
permissions:
pull-requests: read
steps:
# Pinned to v6.1.1. Verify SHA via:
# gh api repos/amannn/action-semantic-pull-request/git/refs/tags/v6.1.1
- uses: amannn/action-semantic-pull-request@48f256284bd46cdaab1048c3721360e808335d50 # v6.1.1
env:
GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
with:
types: |
feat
fix
perf
refactor
docs
test
ci
build
chore
revert
deps
requireScope: false
# Subject must be non-empty. We DO allow capitalized proper nouns
# (MCP, GitHub, API, etc.) — the old `^(?![A-Z]).+$` pattern
# rejected legitimate titles like `fix: MCP tool schema`.
subjectPattern: ^\S.{2,}$
subjectPatternError: |
The subject "{subject}" in PR title "{title}" is invalid.
Subjects must be at least 3 characters and must not start with whitespace.
wip: false
autolabel:
# Privileged job — runs only on `pull_request_target` so it can write labels.
# Never checks out fork code, never executes fork-controlled input; only
# reads the PR metadata (title, body, labels) and calls the GitHub API.
name: Apply conventional label
if: github.event_name == 'pull_request_target'
runs-on: ubuntu-latest
timeout-minutes: 5
permissions:
# `contents: read` is required — release-drafter's context.config() reads
# `.github/release-drafter.yml` from the repo's default branch via the
# repo-contents API. Without it the job silently 403s and no labels are
# applied. Job-level permissions nullify all unlisted scopes, so an
# explicit grant is necessary here.
contents: read
pull-requests: write
steps:
# Pinned to v7.2.0. Verify SHA via:
# gh api repos/release-drafter/release-drafter/git/refs/tags/v7.2.0
# v7 removed `disable-releaser`; use `dry-run: true` to only autolabel.
- uses: release-drafter/release-drafter@5de93583980a40bd78603b6dfdcda5b4df377b32 # v7.2.0
with:
config-name: release-drafter.yml
dry-run: true
env:
GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
+4 -13
View File
@@ -7,22 +7,13 @@ on:
# No workflow-level permissions — scoped per job below.
# Concurrency convention: see CONTRIBUTING.md → "GitHub Actions — Concurrency Convention".
# Tag refs are unique per release, so distinct tags run in parallel. Re-pushes of the
# same tag serialize. cancel-in-progress: false — never cancel a publish mid-flight.
concurrency:
group: ${{ github.workflow }}-${{ github.ref }}
cancel-in-progress: false
jobs:
ci:
uses: ./.github/workflows/ci.yml
permissions:
contents: read
actions: read
# No pull-requests:write — `ci.yml`'s save-pr-meta job is gated on
# `github.event_name == 'pull_request'`, so it never runs during a
# tag-triggered publish. Least-privilege for release-critical paths.
pull-requests: write
publish:
needs: ci
@@ -32,8 +23,8 @@ jobs:
contents: write
id-token: write
steps:
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
- uses: actions/setup-node@53b83947a5a98c8d113130e565377fae1a50d02f # v6.3.0
- uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4
- uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4
with:
node-version: 20
registry-url: https://registry.npmjs.org
@@ -91,7 +82,7 @@ jobs:
fi
- name: Create GitHub Release
uses: softprops/action-gh-release@b4309332981a82ec1c5618f44dd2e27cc8bfbfda # v2
uses: softprops/action-gh-release@a06a81a03ee405af7f2048a818ed3f03bbf83c7b # v2
with:
body_path: ${{ steps.changelog.outputs.fallback == 'false' && '/tmp/release-notes.md' || '' }}
generate_release_notes: ${{ steps.changelog.outputs.fallback == 'true' }}
-366
View File
@@ -1,366 +0,0 @@
name: Release Candidate
on:
# Publish a release-candidate build whenever a merge/commit lands on main.
# Docs/README-only changes are filtered out so prose updates don't
# cut a release.
push:
branches: [main]
paths-ignore:
- '**.md'
- 'docs/**'
- 'LICENSE'
workflow_dispatch:
inputs:
bump:
description: >-
Cycle policy. 'auto' (default) continues the active rc cycle on
this branch if there is one, otherwise bumps patch from latest.
Choose 'patch' / 'minor' / 'major' to explicitly start or reset
an rc cycle.
required: false
default: 'auto'
type: choice
options:
- auto
- patch
- minor
- major
force:
description: 'Publish even when HEAD already has an rc marker'
required: false
default: 'false'
type: choice
options:
- 'false'
- 'true'
# No workflow-level permissions — scoped per job below.
permissions: {}
# Concurrency convention: see CONTRIBUTING.md → "GitHub Actions — Concurrency Convention".
# Serialize all runs on the same ref (push + workflow_dispatch) to prevent two publishes
# racing on the rc counter. cancel-in-progress: false — the earlier merge publishes first.
concurrency:
group: ${{ github.workflow }}-${{ github.ref }}
cancel-in-progress: false
jobs:
# ── Skip when HEAD already has an rc marker (retry / duplicate dispatch) ──
# The marker is a lightweight tag `rc/<HEAD_SHA>` pushed *before* `npm
# publish`, so a failed publish leaves the marker in place and the guard
# refuses to re-publish. Recovery path after a partial failure:
# git push --delete origin rc/<HEAD_SHA> v<RC_VERSION>
# then redispatch with force=true.
guard:
name: Check if release candidate should run
runs-on: ubuntu-latest
timeout-minutes: 5
permissions:
contents: read
outputs:
should_run: ${{ steps.decide.outputs.should_run }}
head_sha: ${{ steps.decide.outputs.head_sha }}
steps:
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
with:
fetch-depth: 0
fetch-tags: true
- name: Decide
id: decide
shell: bash
env:
FORCE: ${{ inputs.force }}
BUMP_INPUT: ${{ inputs.bump }}
EVENT_NAME: ${{ github.event_name }}
run: |
set -euo pipefail
HEAD_SHA=$(git rev-parse HEAD)
echo "head_sha=$HEAD_SHA" >> "$GITHUB_OUTPUT"
if [ "$FORCE" = "true" ]; then
echo "Force flag set — running regardless of marker tag."
echo "should_run=true" >> "$GITHUB_OUTPUT"
exit 0
fi
# An explicit cycle reset on dispatch (bump != auto) also bypasses
# the dedup guard — the maintainer is deliberately asking for a
# new rc from the same commit.
if [ "$EVENT_NAME" = "workflow_dispatch" ] \
&& [ -n "${BUMP_INPUT:-}" ] \
&& [ "${BUMP_INPUT:-auto}" != "auto" ]; then
echo "Explicit bump=$BUMP_INPUT — bypassing marker dedup."
echo "should_run=true" >> "$GITHUB_OUTPUT"
exit 0
fi
# Dedup: is there already an rc/<HEAD_SHA> marker pointing at HEAD?
MARKER="rc/${HEAD_SHA}"
if git rev-parse "refs/tags/$MARKER" >/dev/null 2>&1; then
echo "HEAD already has marker $MARKER — skipping."
echo "should_run=false" >> "$GITHUB_OUTPUT"
else
echo "No marker on HEAD — proceeding."
echo "should_run=true" >> "$GITHUB_OUTPUT"
fi
# ── Reuse the stable CI workflow ─────────────────────────────────────
ci:
needs: guard
if: needs.guard.outputs.should_run == 'true'
uses: ./.github/workflows/ci.yml
permissions:
contents: read
secrets: inherit
# ── Publish the rc build to npm + create GitHub prerelease ───────────
publish:
name: Publish release candidate to npm
needs: [guard, ci]
if: needs.guard.outputs.should_run == 'true'
runs-on: ubuntu-latest
timeout-minutes: 20
permissions:
contents: write # push rc tag + marker
id-token: write # npm provenance
steps:
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
with:
fetch-depth: 0
fetch-tags: true
- uses: actions/setup-node@53b83947a5a98c8d113130e565377fae1a50d02f # v6.3.0
with:
node-version: 20
registry-url: https://registry.npmjs.org
cache: npm
cache-dependency-path: gitnexus/package-lock.json
- name: Build gitnexus-shared
run: npm install && npm run build
working-directory: gitnexus-shared
- name: Install gitnexus dependencies
run: npm ci
working-directory: gitnexus
- name: Resolve rc version
id: version
shell: bash
working-directory: gitnexus
env:
BUMP_INPUT: ${{ inputs.bump }}
EVENT_NAME: ${{ github.event_name }}
PKG_NAME: gitnexus
run: |
set -euo pipefail
# 1. Current published `latest` — the floor for any new rc base.
# Only E404 ("never published") falls back to package.json; any
# other error (network, auth, malformed response) fails fast.
NPM_STDERR_LATEST="$(mktemp)"
if CURRENT_LATEST="$(npm view "$PKG_NAME" version 2>"$NPM_STDERR_LATEST")"; then
:
else
if grep -q 'E404' "$NPM_STDERR_LATEST"; then
CURRENT_LATEST="$(node -p "require('./package.json').version")"
echo "Package not on registry (E404) — seeding from package.json: $CURRENT_LATEST"
else
echo "::error::npm registry unreachable for 'view version':" >&2
cat "$NPM_STDERR_LATEST" >&2
rm -f "$NPM_STDERR_LATEST"
exit 1
fi
fi
rm -f "$NPM_STDERR_LATEST"
CURRENT_LATEST_CLEAN="${CURRENT_LATEST%%-*}"
# 2. Full version list — needed for the counter and for active-cycle
# inference. Same E404-only fallback.
NPM_STDERR_VERSIONS="$(mktemp)"
if VERSIONS_JSON="$(npm view "$PKG_NAME" versions --json 2>"$NPM_STDERR_VERSIONS")"; then
:
else
if grep -q 'E404' "$NPM_STDERR_VERSIONS"; then
VERSIONS_JSON='[]'
echo "No published versions for $PKG_NAME yet (E404)."
else
echo "::error::npm registry unreachable for 'view versions':" >&2
cat "$NPM_STDERR_VERSIONS" >&2
rm -f "$NPM_STDERR_VERSIONS"
exit 1
fi
fi
rm -f "$NPM_STDERR_VERSIONS"
# 3. Base selection.
# - workflow_dispatch + bump ∈ {patch,minor,major} → explicit cycle
# reset from latest.
# - Everything else (push, or dispatch with bump=auto) → continue
# the highest active rc base > latest if one exists; else
# default to patch from latest.
if [ "$EVENT_NAME" = "workflow_dispatch" ] \
&& [ -n "${BUMP_INPUT:-}" ] \
&& [ "${BUMP_INPUT:-auto}" != "auto" ]; then
BASE="$(npx --yes -p semver@7 semver -i "$BUMP_INPUT" "$CURRENT_LATEST_CLEAN")"
echo "Explicit bump=$BUMP_INPUT → BASE=$BASE"
else
cat > /tmp/active_base.mjs <<'NODESCRIPT'
const latest = process.env.LATEST;
let v;
try { v = JSON.parse(process.env.VERSIONS_JSON); } catch { v = []; }
if (!Array.isArray(v)) v = [v];
const parse = s => s.split(".").map(n => parseInt(n, 10));
const gt = (a, b) => {
const [A, B] = [parse(a), parse(b)];
for (let i = 0; i < 3; i++) if (A[i] !== B[i]) return A[i] > B[i];
return false;
};
const bases = new Set();
for (const s of v) {
const m = /^(\d+\.\d+\.\d+)-rc\.\d+$/.exec(s);
if (m && gt(m[1], latest)) bases.add(m[1]);
}
if (!bases.size) { process.stdout.write(""); process.exit(0); }
const sorted = [...bases].sort((a, b) => gt(a, b) ? 1 : -1);
process.stdout.write(sorted[sorted.length - 1]);
NODESCRIPT
ACTIVE_BASE="$(LATEST="$CURRENT_LATEST_CLEAN" VERSIONS_JSON="$VERSIONS_JSON" node /tmp/active_base.mjs)"
if [ -n "$ACTIVE_BASE" ]; then
BASE="$ACTIVE_BASE"
echo "Continuing active rc cycle → BASE=$BASE"
else
BASE="$(npx --yes -p semver@7 semver -i patch "$CURRENT_LATEST_CLEAN")"
echo "No active rc cycle → patch bump from latest → BASE=$BASE"
fi
fi
# 4. Counter: 1 + max existing N for `${BASE}-rc.*`, else 1.
cat > /tmp/next_rc.mjs <<'NODESCRIPT'
const base = process.env.BASE;
const prefix = base + "-rc.";
let v;
try { v = JSON.parse(process.env.VERSIONS_JSON); } catch { v = []; }
if (!Array.isArray(v)) v = [v];
const ns = v
.filter(s => typeof s === "string" && s.startsWith(prefix))
.map(s => parseInt(s.slice(prefix.length), 10))
.filter(n => Number.isInteger(n) && n >= 0);
process.stdout.write(String(ns.length ? Math.max(...ns) + 1 : 1));
NODESCRIPT
NEXT_N="$(BASE="$BASE" VERSIONS_JSON="$VERSIONS_JSON" node /tmp/next_rc.mjs)"
RC_VERSION="${BASE}-rc.${NEXT_N}"
echo "Computed rc: $RC_VERSION"
# 5. Defensive: if the exact version already exists on the registry
# (e.g., race with another run), abort before re-publishing.
# Same E404-only pattern used above — a transient network
# failure must fail loudly, not pretend the version is missing.
NPM_STDERR_EXISTS="$(mktemp)"
if npm view "$PKG_NAME@$RC_VERSION" version 2>"$NPM_STDERR_EXISTS" >/dev/null; then
rm -f "$NPM_STDERR_EXISTS"
echo "::error::Version $RC_VERSION already exists on npm — aborting."
exit 1
else
if grep -qiE 'E404|not found' "$NPM_STDERR_EXISTS"; then
rm -f "$NPM_STDERR_EXISTS"
# Version doesn't exist — safe to proceed.
else
echo "::error::npm registry unreachable for existence check:" >&2
cat "$NPM_STDERR_EXISTS" >&2
rm -f "$NPM_STDERR_EXISTS"
exit 1
fi
fi
echo "base=$BASE" >> "$GITHUB_OUTPUT"
echo "rc_n=$NEXT_N" >> "$GITHUB_OUTPUT"
echo "rc_version=$RC_VERSION" >> "$GITHUB_OUTPUT"
- name: Apply rc version in-CI
shell: bash
working-directory: gitnexus
run: |
set -euo pipefail
npm version "${{ steps.version.outputs.rc_version }}" \
--no-git-tag-version --allow-same-version
- name: Build gitnexus
run: npm run build
working-directory: gitnexus
- name: Dry-run publish
run: npm publish --dry-run --tag rc
working-directory: gitnexus
# ── Acquire the "rc lock" BEFORE publishing (fixes idempotency) ─────
# We create two tags and push them atomically:
# v<RC_VERSION> → annotated tag on a detached release commit
# whose tree contains the rewritten package.json
# (so the tag's source matches the npm tarball)
# rc/<HEAD_SHA> → lightweight tag on HEAD; the guard's dedup key
# If this push fails, nothing is published — safe.
# If this push succeeds but npm publish fails, the marker stays on
# the remote and blocks retries until an operator manually cleans up.
- name: Create and push rc tags
id: reltag
shell: bash
working-directory: gitnexus
env:
RC_VERSION: ${{ steps.version.outputs.rc_version }}
HEAD_SHA: ${{ needs.guard.outputs.head_sha }}
run: |
set -euo pipefail
VTAG="v${RC_VERSION}"
MARKER="rc/${HEAD_SHA}"
git config user.name 'github-actions[bot]'
git config user.email '41898282+github-actions[bot]@users.noreply.github.com'
# Detached release commit with the version bump — keeps `main`
# pristine but gives the v-tag a tree that matches the published
# package contents exactly (fixes release-integrity gap).
git add package.json package-lock.json 2>/dev/null || git add package.json
git commit -m "release: ${VTAG}" --allow-empty
RELEASE_SHA="$(git rev-parse HEAD)"
echo "Detached release commit: $RELEASE_SHA"
# Annotated release tag on the release commit.
git tag -a "$VTAG" "$RELEASE_SHA" -m "$VTAG"
# Lightweight marker on the user-visible HEAD for the guard.
git tag "$MARKER" "$HEAD_SHA"
# Atomic push of both refs. If either would clobber an existing
# remote ref, the push fails and we stop before npm publish.
git push --atomic origin "refs/tags/$VTAG" "refs/tags/$MARKER"
echo "vtag=$VTAG" >> "$GITHUB_OUTPUT"
echo "marker=$MARKER" >> "$GITHUB_OUTPUT"
echo "release_sha=$RELEASE_SHA" >> "$GITHUB_OUTPUT"
- name: Publish to npm (rc dist-tag)
run: npm publish --provenance --access public --tag rc
working-directory: gitnexus
env:
NODE_AUTH_TOKEN: ${{ secrets.NPM_TOKEN }}
- name: Create GitHub prerelease
uses: softprops/action-gh-release@b4309332981a82ec1c5618f44dd2e27cc8bfbfda # v2
with:
tag_name: ${{ steps.reltag.outputs.vtag }}
name: Release Candidate ${{ steps.reltag.outputs.vtag }}
prerelease: true
make_latest: 'false'
generate_release_notes: true
body: |
Automated release candidate build from `main`.
**npm:** `npm install gitnexus@rc`
**Version:** `${{ steps.version.outputs.rc_version }}`
**Target base:** `${{ steps.version.outputs.base }}` (rc #${{ steps.version.outputs.rc_n }})
**Source commit (main):** ${{ needs.guard.outputs.head_sha }}
**Release commit (versioned tree):** ${{ steps.reltag.outputs.release_sha }}
Release candidates are pre-stable builds intended for early testing.
Stable releases remain on the `latest` dist-tag.
@@ -1,185 +0,0 @@
name: Tree-sitter Upgrade Readiness
# Monitors readiness for upgrading tree-sitter to 0.25.x. Tracks:
# 1. Peer-dep compatibility — can each grammar install cleanly with
# tree-sitter@0.25.0 without --legacy-peer-deps?
# 2. Vendored proto drift — has coder3101/tree-sitter-proto moved
# ahead of our vendored snapshot?
# See .github/scripts/check-tree-sitter-upgrade-readiness.py for the logic.
#
# Concurrency convention: see CONTRIBUTING.md → "GitHub Actions — Concurrency Convention".
on:
schedule:
# Daily at 09:00 UTC. Matches Dependabot's daily cadence so drift
# and dep PRs surface together.
- cron: '0 9 * * *'
workflow_dispatch:
pull_request:
paths:
- '.github/scripts/check-tree-sitter-upgrade-readiness.py'
- '.github/workflows/tree-sitter-upgrade-readiness.yml'
concurrency:
group: ${{ github.workflow }}-${{ github.ref }}
cancel-in-progress: ${{ github.event_name == 'pull_request' }}
permissions:
contents: read
jobs:
readiness:
name: Check upgrade readiness
runs-on: ubuntu-latest
timeout-minutes: 10
permissions:
contents: read
# Needed to open/update the tracking issue on scheduled runs.
issues: write
steps:
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
- uses: ./.github/actions/setup-gitnexus
with:
build: 'false'
- name: Run upgrade readiness check
id: readiness
shell: bash
env:
GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
run: |
set +e
python3 .github/scripts/check-tree-sitter-upgrade-readiness.py > drift-report.md
code=$?
set -e
echo "exit_code=$code" >> "$GITHUB_OUTPUT"
{
echo 'report<<DRIFT_EOF'
cat drift-report.md
echo 'DRIFT_EOF'
} >> "$GITHUB_OUTPUT"
echo "=== Report ==="
cat drift-report.md
# On PR runs, the script validates that it runs correctly. Blockers
# are informational — the scheduled run opens a tracking issue.
- name: Annotate PR with readiness status
if: github.event_name == 'pull_request' && steps.readiness.outputs.exit_code != '0'
run: |
echo "::warning::Tree-sitter 0.25 upgrade has blockers. See job output for the full readiness report."
- name: Upsert tracking issue on scheduled runs
if: >
github.event_name == 'schedule' &&
steps.readiness.outputs.exit_code != '0'
uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v9.0.0
env:
REPORT: ${{ steps.readiness.outputs.report }}
with:
script: |
const title = 'Tree-sitter 0.25 upgrade readiness';
const report = process.env.REPORT;
const body = report + '\n\n' +
'<sub>Generated daily by `.github/workflows/tree-sitter-upgrade-readiness.yml`. ' +
'Closes automatically when all blockers are resolved.</sub>';
const { data: open } = await github.rest.issues.listForRepo({
owner: context.repo.owner,
repo: context.repo.repo,
state: 'open',
labels: 'tree-sitter-drift',
per_page: 10,
});
const existing = open.find(i => i.title === title);
if (existing) {
// Extract ready/total count for the changelog comment.
const readyMatch = report.match(/\*\*(\d+)\/(\d+)\*\* grammars ready/);
const blockerMatch = report.match(/\*\*(\d+) blocker/);
const ready = readyMatch ? readyMatch[1] : '?';
const total = readyMatch ? readyMatch[2] : '?';
const blockers = blockerMatch ? blockerMatch[1] : '?';
// Find grammars whose status changed by diffing the old and
// new table rows. Each row looks like:
// | `tree-sitter-foo` | ... | Ready |
// | `tree-sitter-foo` | ... | Blocking |
const parseRows = (md) => {
const map = {};
for (const m of md.matchAll(/\| `(tree-sitter-[^`]+)` \|.*?\| (\S+(?:\s\S+)*?) \|$/gm)) {
map[m[1]] = m[2].trim();
}
return map;
};
const oldRows = parseRows(existing.body || '');
const newRows = parseRows(report);
const changes = [];
for (const [name, newStatus] of Object.entries(newRows)) {
const oldStatus = oldRows[name];
if (oldStatus && oldStatus !== newStatus) {
changes.push(`\`${name}\`: ${oldStatus} → ${newStatus}`);
}
}
const today = new Date().toISOString().slice(0, 10);
let comment = `**${today}:** ${ready}/${total} ready. ${blockers} blocker(s) remaining.`;
if (changes.length > 0) {
comment += '\n\nChanges:\n' + changes.map(c => `- ${c}`).join('\n');
} else {
comment += ' No changes from previous run.';
}
await github.rest.issues.createComment({
owner: context.repo.owner,
repo: context.repo.repo,
issue_number: existing.number,
body: comment,
});
await github.rest.issues.update({
owner: context.repo.owner,
repo: context.repo.repo,
issue_number: existing.number,
body,
});
core.info(`Updated existing issue #${existing.number}`);
} else {
const { data: created } = await github.rest.issues.create({
owner: context.repo.owner,
repo: context.repo.repo,
title,
body,
labels: ['tree-sitter-drift', 'dependencies'],
});
core.info(`Opened issue #${created.number}`);
}
- name: Close tracking issue on clean scheduled runs
if: >
github.event_name == 'schedule' &&
steps.readiness.outputs.exit_code == '0'
uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v9.0.0
with:
script: |
const title = 'Tree-sitter 0.25 upgrade readiness';
const { data: open } = await github.rest.issues.listForRepo({
owner: context.repo.owner,
repo: context.repo.repo,
state: 'open',
labels: 'tree-sitter-drift',
per_page: 10,
});
const existing = open.find(i => i.title === title);
if (existing) {
await github.rest.issues.createComment({
owner: context.repo.owner,
repo: context.repo.repo,
issue_number: existing.number,
body: 'All grammars are now compatible with tree-sitter@0.25. Upgrade is ready! Closing automatically.',
});
await github.rest.issues.update({
owner: context.repo.owner,
repo: context.repo.repo,
issue_number: existing.number,
state: 'closed',
});
core.info(`Closed issue #${existing.number}`);
}
+2 -4
View File
@@ -47,10 +47,8 @@ permissions:
issues: write
pull-requests: write
# Concurrency convention: see CONTRIBUTING.md → "GitHub Actions — Concurrency Convention".
# Single global slot — newest manual dispatch supersedes any in-flight run.
concurrency:
group: ${{ github.workflow }}
group: triage-sweep
cancel-in-progress: true
jobs:
@@ -76,7 +74,7 @@ jobs:
run: pip install -r .github/scripts/triage/requirements.txt
- name: Cache FastEmbed model weights
uses: actions/cache@27d5ce7f107fe9357f9df03efb73ab90386fccae # v5
uses: actions/cache@668228422ae6a00e4ad889ee87cd7109ec5666a7 # v5
with:
path: ${{ github.workspace }}/.fastembed_cache
key: fastembed-bge-small-en-v1.5
+1 -8
View File
@@ -23,7 +23,6 @@ Thumbs.db
.env
.env.local
.env.*.local
docker/.env
# Logs
*.log
@@ -82,11 +81,6 @@ GitNexus.sln
# Git worktrees
.worktrees/
# Vendored tree-sitter grammar build artifacts (created at install time,
# never committed). See docs/plans/2026-04-15-002-fix-tree-sitter-proto-vendor-deps-plan.md
gitnexus/vendor/**/build/
gitnexus/vendor/**/node_modules/
/github/scripts/triage/__pycache__/
.claude-flow/
@@ -101,5 +95,4 @@ gitnexus/vendor/**/node_modules/
.swarm/
local_docs/
local_docs/
+104 -102
View File
@@ -1,122 +1,117 @@
<!-- version: 1.4.0 -->
<!-- Last updated: 2026-04-16 -->
<!-- version: 1.3.0 -->
<!--
Metadata: version, last reviewed, scope, model policy, reference docs, changelog.
Last updated: 2026-03-22
-->
Last reviewed: 2026-04-16
Last reviewed: 2026-04-13
**Project:** GitNexus · **Environment:** dev · **Maintainer:** repository maintainers (see GitHub)
This file uses a standard agent header (version, scope, model policy, reference docs, changelog), adapted for this **TypeScript/JavaScript monorepo**.
## Scope
| Boundary | Rule |
|----------|------|
| **Reads** | `gitnexus/`, `gitnexus-web/`, `eval/`, plugin packages, `.github/`, `.gitnexus/`, docs. |
| **Writes** | Only paths required for the change; keep diffs minimal. Update lockfiles when deps change. |
| **Executes** | `npm`, `npx`, `node` under `gitnexus/` and `gitnexus-web/`; `uv run` for Python under `eval/`; documented CI/dev workflows. |
| **Off-limits** | Real `.env` / secrets, production credentials, unrelated repos, destructive git ops without confirmation. |
| | |
|--|--|
| **Reads** | Repository tree as needed for the task: `gitnexus/`, `gitnexus-web/`, `eval/`, plugin packages, `.github/`, `.gitnexus/` when present, and docs. |
| **Writes** | Only paths required for the requested change; keep diffs minimal. Update lockfiles when dependencies change. |
| **Executes** | `npm`, `npx`, `node` under `gitnexus/` and `gitnexus-web/`; `uv run` for Python under `eval/` when applicable; shell utilities for documented CI/dev workflows. |
| **Off-limits** | User secrets (e.g. real `.env`), production deployment credentials, unrelated repositories, destructive git history operations without explicit human confirmation. |
## Model Configuration
- **Primary:** Use a named model (e.g. Claude Sonnet 4.x). Avoid `Auto` or unversioned `latest` when reproducibility matters.
- **Notes:** The GitNexus CLI indexer does not call an LLM.
- **Primary:** Pin in **Cursor** (Settings → model). Use a **named** model (e.g. GPT-5.2, Claude Sonnet 4.x). Avoid relying on **Auto** when reproducibility or audit trail matters.
- **Fallback:** As configured in Cursor or your organization (do not encode `latest` or wildcards in automation configs).
- **Notes:** The open-source GitNexus CLI indexer does not call an LLM. Optional Nexus AI in the web UI uses end-user provider keys and models.
## Execution Sequence (complex tasks)
For multi-step work, state up front:
1. Which rules in this file and **[GUARDRAILS.md](GUARDRAILS.md)** apply (and any relevant Signs).
2. Current **Scope** boundaries.
3. Which **validation commands** you will run (`cd gitnexus && npm test`, `npx tsc --noEmit`).
Long sessions dilute instructions. For **multi-step** work, state up front:
On long threads, *"Remember: apply all AGENTS.md rules"* re-weights these instructions against context dilution.
1. Which rules in this file and **[GUARDRAILS.md](GUARDRAILS.md)** apply (and any relevant Signs).
2. Current **Scope** boundaries (Reads / Writes / Off-limits).
3. Which **validation commands** you will run (e.g. `cd gitnexus && npm test`, `npx tsc --noEmit`).
On very long threads, the human may add *“Remember: apply all AGENTS.md rules”* to re-weight rule tokens against context dilution.
## Claude Code hooks
**PreToolUse** hooks can block tools (e.g. `git_commit`) until checks pass. Adapt to this repo: `cd gitnexus && npm test` before commit.
Hooks enforce gates that prompts cannot. In **Claude Code**, **PreToolUse** hooks can block tools such as `git_commit` until checks pass. Adapt to this repo: e.g. `cd gitnexus && npm test` before commit.
## Context budget
## Context budget (Cursor / standards)
Commands and gotchas live under **Repo reference** below and in **[CONTRIBUTING.md](CONTRIBUTING.md)**. If always-on rules grow, split into **`.cursor/rules/*.mdc`** (globs). **Cursor:** project-wide rules in `.cursor/index.mdc`. **Claude Code:** load `STANDARDS.md` only when needed.
Generic “core standards” playbooks are often long and stack-specific. For this monorepo, commands and gotchas live under **Cursor Cloud specific instructions** below and in **[CONTRIBUTING.md](CONTRIBUTING.md)**. If always-on rules grow, split domain rules into **`.cursor/rules/*.mdc`** (globs). **Cursor:** project-wide rules live in **`.cursor/index.mdc`** (YAML frontmatter with `alwaysApply: true`). **Claude Code:** optionally load a **`STANDARDS.md`** only when needed (e.g. *“When writing new code, read STANDARDS.md”*) to save context.
## Reference docs
## Reference Documentation
- **[ARCHITECTURE.md](ARCHITECTURE.md)**, **[CONTRIBUTING.md](CONTRIBUTING.md)**, **[GUARDRAILS.md](GUARDRAILS.md)**
- **Call-resolution DAG:** See ARCHITECTURE.md § Call-Resolution DAG. Typed 6-stage DAG inside the `parse` phase; language-specific behavior behind `inferImplicitReceiver` / `selectDispatch` hooks on `LanguageProvider`. Shared code in `gitnexus/src/core/ingestion/` must not name languages. Types: `gitnexus/src/core/ingestion/call-types.ts`.
- **Cursor:** `.cursor/index.mdc` (always-on); `.cursor/rules/*.mdc` (glob-scoped). Legacy `.cursorrules` deprecated.
- **GitNexus:** skills in `.claude/skills/gitnexus/`; MCP rules in `gitnexus:start` block below.
- **This repository:** **[ARCHITECTURE.md](ARCHITECTURE.md)**, **[CONTRIBUTING.md](CONTRIBUTING.md)**, **[GUARDRAILS.md](GUARDRAILS.md)**.
- **Cursor:** `.cursor/index.mdc` (always-on rules); optional `.cursor/rules/*.mdc` (glob-scoped). Legacy `.cursorrules` is deprecated — see `.cursor/index.mdc`.
- **Optional local files:** `NOTES.md` (short vendor-neutral project snapshot). For handoffs, keep notes local (e.g., a scratch file outside the repo) rather than committing `HANDOFF.md`.
- **GitNexus:** skills under `.claude/skills/gitnexus/`; machine-oriented rules in the `gitnexus:start` … `gitnexus:end` block below.
## Changelog
| Date | Version | Change |
|------|---------|--------|
| 2026-04-16 | 1.4.0 | Fixed: web UI description, pre-commit behavior, MCP tools (7->16), added gitnexus-shared, removed stale vite-plugin-wasm gotcha. |
| 2026-04-13 | 1.3.0 | Updated GitNexus index stats after DAG refactor. |
| 2026-03-24 | 1.2.0 | Fixed gitnexus:start block duplication. |
| 2026-03-23 | 1.1.0 | Updated agent instructions, references, Cursor layout. |
| 2026-03-22 | 1.0.0 | Initial structured header and changelog. |
| 2026-03-24 | 1.2.0 | Fixed gitnexus:start block duplication (was inlined in Reference Docs bullet). |
| 2026-03-23 | 1.1.0 | Updated agent instructions (sections, references, Cursor layout). |
| 2026-03-22 | 1.0.0 | Added structured agent header and changelog. |
---
<!-- gitnexus:start -->
# GitNexus — Code Intelligence
Indexed as **GitNexus** (4325 symbols, 10556 relationships, 300 execution flows). Use MCP tools to understand code, assess impact, and navigate safely.
This project is indexed by GitNexus as **GitNexus** (4325 symbols, 10556 relationships, 300 execution flows). Use the GitNexus MCP tools to understand code, assess impact, and navigate safely.
> If any tool warns the index is stale, run `npx gitnexus analyze` first.
> If any GitNexus tool warns the index is stale, run `npx gitnexus analyze` in terminal first.
## Always Do
- **MUST run impact analysis before editing any symbol.** `gitnexus_impact({target: "symbolName", direction: "upstream"})` — report blast radius to the user.
- **MUST run `gitnexus_detect_changes()` before committing** — verify only expected symbols and flows are affected.
- **MUST warn the user** if impact returns HIGH or CRITICAL risk.
- Explore unfamiliar code with `gitnexus_query({query: "concept"})` (process-grouped, ranked) instead of grepping.
- Full context on a symbol: `gitnexus_context({name: "symbolName"})`.
- **MUST run impact analysis before editing any symbol.** Before modifying a function, class, or method, run `gitnexus_impact({target: "symbolName", direction: "upstream"})` and report the blast radius (direct callers, affected processes, risk level) to the user.
- **MUST run `gitnexus_detect_changes()` before committing** to verify your changes only affect expected symbols and execution flows.
- **MUST warn the user** if impact analysis returns HIGH or CRITICAL risk before proceeding with edits.
- When exploring unfamiliar code, use `gitnexus_query({query: "concept"})` to find execution flows instead of grepping. It returns process-grouped results ranked by relevance.
- When you need full context on a specific symbol — callers, callees, which execution flows it participates in — use `gitnexus_context({name: "symbolName"})`.
## When Debugging
1. `gitnexus_query({query: "<error or symptom>"})` — find related execution flows
2. `gitnexus_context({name: "<suspect function>"})` — callers, callees, process participation
3. `READ gitnexus://repo/GitNexus/process/{processName}` — trace flow step by step
4. Regressions: `gitnexus_detect_changes({scope: "compare", base_ref: "main"})`
1. `gitnexus_query({query: "<error or symptom>"})` — find execution flows related to the issue
2. `gitnexus_context({name: "<suspect function>"})` — see all callers, callees, and process participation
3. `READ gitnexus://repo/GitNexus/process/{processName}` — trace the full execution flow step by step
4. For regressions: `gitnexus_detect_changes({scope: "compare", base_ref: "main"})` — see what your branch changed
## When Refactoring
- **Rename:** `gitnexus_rename({symbol_name: "old", new_name: "new", dry_run: true})` first. Graph edits are safe; text_search edits need manual review.
- **Extract/Split:** `gitnexus_context` (incoming/outgoing refs) then `gitnexus_impact` (upstream callers) before moving code.
- **After any refactor:** `gitnexus_detect_changes({scope: "all"})` to verify scope.
- **Renaming**: MUST use `gitnexus_rename({symbol_name: "old", new_name: "new", dry_run: true})` first. Review the preview — graph edits are safe, text_search edits need manual review. Then run with `dry_run: false`.
- **Extracting/Splitting**: MUST run `gitnexus_context({name: "target"})` to see all incoming/outgoing refs, then `gitnexus_impact({target: "target", direction: "upstream"})` to find all external callers before moving code.
- After any refactor: run `gitnexus_detect_changes({scope: "all"})` to verify only expected files changed.
## Never Do
- Edit a symbol without running `gitnexus_impact` first.
- Ignore HIGH/CRITICAL risk warnings.
- Rename with find-and-replace — use `gitnexus_rename`.
- Commit without `gitnexus_detect_changes()`.
- Add language-specific behavior to shared ingestion code (`gitnexus/src/core/ingestion/`) — use a `LanguageProvider` hook. Seeing `provider.mroStrategy === 'xxx'` or an import from `languages/xxx.ts` in shared code means stop and add a hook.
- NEVER edit a function, class, or method without first running `gitnexus_impact` on it.
- NEVER ignore HIGH or CRITICAL risk warnings from impact analysis.
- NEVER rename symbols with find-and-replace — use `gitnexus_rename` which understands the call graph.
- NEVER commit changes without running `gitnexus_detect_changes()` to check affected scope.
## Tools Quick Reference
| Tool | When to use | Example |
| Tool | When to use | Command |
|------|-------------|---------|
| `list_repos` | Discover indexed repos | `gitnexus_list_repos({})` |
| `query` | Find code by concept | `gitnexus_query({query: "auth validation"})` |
| `context` | 360-degree view of one symbol | `gitnexus_context({name: "validateUser"})` |
| `impact` | Blast radius before editing | `gitnexus_impact({target: "X", direction: "upstream"})` |
| `detect_changes` | Pre-commit scope check | `gitnexus_detect_changes({scope: "staged"})` |
| `rename` | Safe multi-file rename | `gitnexus_rename({symbol_name: "old", new_name: "new", dry_run: true})` |
| `cypher` | Custom graph queries | `gitnexus_cypher({query: "MATCH ..."})` |
| `api_impact` | Pre-change API route impact | `gitnexus_api_impact({route: "/api/users", method: "GET"})` |
| `route_map` | Route → handler → consumer map | `gitnexus_route_map({})` |
| `tool_map` | MCP/RPC tool definitions | `gitnexus_tool_map({})` |
| `shape_check` | Response shape vs consumer access | `gitnexus_shape_check({route: "/api/users"})` |
| `group_list` | List repo groups | `gitnexus_group_list({})` |
| `group_query` | Cross-repo search in a group | `gitnexus_group_query({name: "myGroup", query: "auth"})` |
| `group_sync` | Rebuild group Contract Registry | `gitnexus_group_sync({name: "myGroup"})` |
| `group_contracts` | Inspect group contracts | `gitnexus_group_contracts({name: "myGroup"})` |
| `group_status` | Group staleness report | `gitnexus_group_status({name: "myGroup"})` |
## Impact Risk Levels
| Depth | Meaning | Action |
|-------|---------|--------|
| d=1 | WILL BREAK — direct callers/importers | MUST update |
| d=1 | WILL BREAK — direct callers/importers | MUST update these |
| d=2 | LIKELY AFFECTED — indirect deps | Should test |
| d=3 | MAY NEED TESTING — transitive | Test if critical path |
@@ -124,80 +119,87 @@ Indexed as **GitNexus** (4325 symbols, 10556 relationships, 300 execution flows)
| Resource | Use for |
|----------|---------|
| `gitnexus://repo/GitNexus/context` | Codebase overview, index freshness |
| `gitnexus://repo/GitNexus/context` | Codebase overview, check index freshness |
| `gitnexus://repo/GitNexus/clusters` | All functional areas |
| `gitnexus://repo/GitNexus/processes` | All execution flows |
| `gitnexus://repo/GitNexus/process/{name}` | Step-by-step execution trace |
## Self-Check Before Finishing
Before completing any code modification task, verify:
1. `gitnexus_impact` was run for all modified symbols
2. No HIGH/CRITICAL warnings were ignored
3. `gitnexus_detect_changes()` confirms expected scope
4. All d=1 dependents were updated
2. No HIGH/CRITICAL risk warnings were ignored
3. `gitnexus_detect_changes()` confirms changes match expected scope
4. All d=1 (WILL BREAK) dependents were updated
## Keeping the Index Fresh
After committing code changes, the GitNexus index becomes stale. Re-run analyze to update it:
```bash
npx gitnexus analyze # basic refresh
npx gitnexus analyze --embeddings # preserve embeddings
npx gitnexus analyze
```
Check `.gitnexus/meta.json` `stats.embeddings` (0 = none). Running without `--embeddings` deletes existing vectors.
If the index previously included embeddings, preserve them by adding `--embeddings`:
> Claude Code: PostToolUse hook handles this after `git commit` and `git merge`.
```bash
npx gitnexus analyze --embeddings
```
## CLI Skills
To check whether embeddings exist, inspect `.gitnexus/meta.json` — the `stats.embeddings` field shows the count (0 means no embeddings). **Running analyze without `--embeddings` will delete any previously generated embeddings.**
| Task | Skill file |
|------|-----------|
| Architecture / "How does X work?" | `.claude/skills/gitnexus/gitnexus-exploring/SKILL.md` |
| Blast radius / "What breaks?" | `.claude/skills/gitnexus/gitnexus-impact-analysis/SKILL.md` |
| Debugging / "Why is X failing?" | `.claude/skills/gitnexus/gitnexus-debugging/SKILL.md` |
| Refactoring | `.claude/skills/gitnexus/gitnexus-refactoring/SKILL.md` |
| Tools/resources/schema reference | `.claude/skills/gitnexus/gitnexus-guide/SKILL.md` |
| CLI commands (index, status, clean, wiki) | `.claude/skills/gitnexus/gitnexus-cli/SKILL.md` |
> Claude Code users: A PostToolUse hook handles this automatically after `git commit` and `git merge`.
## CLI
| Task | Read this skill file |
|------|---------------------|
| Understand architecture / "How does X work?" | `.claude/skills/gitnexus/gitnexus-exploring/SKILL.md` |
| Blast radius / "What breaks if I change X?" | `.claude/skills/gitnexus/gitnexus-impact-analysis/SKILL.md` |
| Trace bugs / "Why is X failing?" | `.claude/skills/gitnexus/gitnexus-debugging/SKILL.md` |
| Rename / extract / split / refactor | `.claude/skills/gitnexus/gitnexus-refactoring/SKILL.md` |
| Tools, resources, schema reference | `.claude/skills/gitnexus/gitnexus-guide/SKILL.md` |
| Index, status, clean, wiki CLI commands | `.claude/skills/gitnexus/gitnexus-cli/SKILL.md` |
<!-- gitnexus:end -->
## Repo reference
## Cursor Cloud specific instructions
### Packages
### Repository structure
| Package | Path | Purpose |
|---------|------|---------|
| **CLI/Core** | `gitnexus/` | TypeScript CLI, indexing pipeline, MCP server. Published to npm. |
| **Web UI** | `gitnexus-web/` | React/Vite thin client. All queries via `gitnexus serve` HTTP API. |
| **Shared** | `gitnexus-shared/` | Shared TypeScript types and constants. |
| Claude Plugin | `gitnexus-claude-plugin/` | Static config for Claude marketplace. |
| Cursor Integration | `gitnexus-cursor-integration/` | Static config for Cursor editor. |
| Eval | `eval/` | Python evaluation harness (Docker + LLM API keys). |
This is a monorepo with two main products and supporting config packages:
| Component | Path | Purpose |
|-----------|------|---------|
| **GitNexus CLI/Core** | `gitnexus/` | Main product — TypeScript CLI, indexing pipeline, MCP server. Published to npm. |
| **GitNexus Web UI** | `gitnexus-web/` | React/Vite browser app — graph explorer + AI chat. Runs entirely in WASM. |
| Claude Plugin | `gitnexus-claude-plugin/` | Static config for Claude marketplace (no build). |
| Cursor Integration | `gitnexus-cursor-integration/` | Static config for Cursor editor (no build). |
| SWE-bench Eval | `eval/` | Python evaluation harness (optional; needs Docker + LLM API keys). |
### Running services
```bash
cd gitnexus && npm run dev # CLI: tsx watch mode
cd gitnexus-web && npm run dev # Web UI: Vite on port 5173
npx gitnexus serve # HTTP API on port 4747 (from any indexed repo)
```
- **CLI/Core**: `cd gitnexus && npm run dev` (tsx watch mode) or `npm run build && node dist/cli/index.js <command>`
- **Web UI**: `cd gitnexus-web && npm run dev` (Vite on port 5173)
- **Backend mode**: `cd <indexed-repo> && node /workspace/gitnexus/dist/cli/index.js serve` (HTTP API on port 3741 by default)
### Testing
**CLI / Core (`gitnexus/`)**
- `npm test` — full vitest suite (~2000 tests)
- `npm run test:unit` — unit tests only
- `npm run test:integration` — integration (~1850 tests). LadybugDB file-locking tests may fail in containers (known env issue).
- `npx tsc --noEmit` — typecheck
- **Unit tests**: `cd gitnexus && npm test` (vitest, ~2000 tests)
- **Integration tests**: `cd gitnexus && npm run test:integration` (vitest, ~1850 tests). Two LadybugDB file-locking tests (`lbug-core-adapter`, `search-core`) may fail in containerized environments due to `/tmp` locking limitations — this is a known environment issue, not a code bug.
- **TypeScript check**: `cd gitnexus && npx tsc --noEmit`
**Web UI (`gitnexus-web/`)**
- `npm test` — vitest (~200 tests)
- `npm run test:e2e` — Playwright (7 spec files; requires `gitnexus serve` + `npm run dev`)
- `npx tsc -b --noEmit` — typecheck
- **Unit tests**: `cd gitnexus-web && npm test` (vitest, ~200 tests)
- **E2E tests**: `cd gitnexus-web && E2E=1 npx playwright test` (Playwright, 5 tests — requires `gitnexus serve` + `npm run dev` running)
- **TypeScript check**: `cd gitnexus-web && npx tsc -b --noEmit`
**Pre-commit hook** (`.husky/pre-commit`): formatting (prettier via lint-staged) + typecheck for staged packages. Tests do **not** run in pre-commit — CI only.
No separate lint command is configured; TypeScript strict checking serves as the primary static analysis.
### Gotchas
- `npm install` in `gitnexus/` triggers `prepare` (builds via `tsc`) and `postinstall` (patches tree-sitter-swift, builds tree-sitter-proto). Native bindings need `python3`, `make`, `g++`.
- `tree-sitter-kotlin` and `tree-sitter-swift` are optional — install warnings expected.
- ESLint configured via `eslint.config.mjs` (TS, React Hooks, unused-imports). No `npm run lint` script; use `npx eslint .`. Prettier runs via lint-staged. CI checks both in `ci-quality.yml`.
- `npm install` in `gitnexus/` triggers `prepare` (builds via `tsc`) and `postinstall` (patches tree-sitter-swift). Native tree-sitter bindings require `python3`, `make`, and `g++` to be present.
- `tree-sitter-kotlin` and `tree-sitter-swift` are optional dependencies — install warnings for these are expected and non-blocking.
- The Web UI uses `vite-plugin-wasm` and requires `Cross-Origin-Opener-Policy`/`Cross-Origin-Embedder-Policy` headers for `SharedArrayBuffer` (handled automatically by Vite dev server).
- There is no ESLint/Prettier configuration in this repo.
+116 -300
View File
@@ -1,129 +1,99 @@
# Architecture — GitNexus
Monorepo: **CLI/MCP** (`gitnexus/`) + **browser UI** (`gitnexus-web/`).
This repository is a **monorepo** with two main products: the **CLI / MCP package** (`gitnexus/`) and the **browser UI** (`gitnexus-web/`). Supporting folders ship editor integrations and plugins without changing the core graph engine.
## Repository layout
| Path | Role |
|------|------|
| `gitnexus/` | npm package `gitnexus`: CLI, MCP server (stdio), HTTP API, ingestion pipeline, LadybugDB graph, embeddings. |
| `gitnexus-web/` | Vite + React thin client: graph explorer + AI chat. All queries via `gitnexus serve` HTTP API. |
| `gitnexus-shared/` | Shared TypeScript types and constants (consumed by CLI and Web). |
| `.claude/`, `gitnexus-claude-plugin/`, `gitnexus-cursor-integration/` | Agent skills and plugin metadata. |
| `eval/` | Evaluation harnesses for benchmarking tool usage. |
| `.github/` | CI workflows + composite actions (`setup-gitnexus/`, `setup-gitnexus-web/`). |
| `gitnexus/` | Published npm package `gitnexus`: CLI, MCP server (stdio), local HTTP API for bridge mode, ingestion pipeline, LadybugDB graph, embeddings (optional). |
| `gitnexus-web/` | Vite + React UI: in-browser indexing (WASM), graph visualization, optional connection to `gitnexus serve`. |
| `.claude/`, `gitnexus-claude-plugin/`, `gitnexus-cursor-integration/` | Packaged **skills** and plugin metadata so agents discover the same workflows as documented in `AGENTS.md`. |
| `eval/` | Evaluation harnesses and docs for benchmarking tool usage. |
| `.github/` | CI workflows (quality, unit, integration, E2E) and composite actions. |
## End-to-end flow: index → graph → tools
1. **Ingestion** — `analyze.ts` → `runFullAnalysis` (`run-analyze.ts`) → `runPipelineFromRepo` (`pipeline.ts`). DAG of 12 phases builds a `KnowledgeGraph` in memory, then loads into LadybugDB under `.gitnexus/`. Repo registered in `~/.gitnexus/registry.json` for MCP discovery.
1. **Ingestion** (`gitnexus analyze`)
- Entry: `gitnexus/src/cli/analyze.ts` → `runPipelineFromRepo` in `gitnexus/src/core/ingestion/pipeline.ts`.
- The pipeline is structured as a **DAG (Directed Acyclic Graph)** of named phases (see [Pipeline Phase DAG](#pipeline-phase-dag) below).
- Output is loaded into **LadybugDB** under **`.gitnexus/`** at the repo root (`lbug/`, `meta.json`, etc.). Optional **FTS** indexes and **embeddings** attach to the same store.
- The repo is registered in **`~/.gitnexus/registry.json`** so MCP can find it from any working directory.
2. **Persistence** — `repo-manager.ts` (paths, registry, KuzuDB cleanup). `lbug-adapter.ts` (graph load, queries, embedding batches).
2. **Persistence & metadata**
- `gitnexus/src/storage/repo-manager.ts` — paths, registry, cleanup of legacy Kuzu artifacts.
- `gitnexus/src/core/lbug/lbug-adapter.ts` — graph load, queries, embedding restore batches.
3. **Query layer** — three interfaces to the same backend:
- **MCP (stdio):** `mcp.ts` → `LocalBackend` → tools (`tools.ts`) + resources (`resources.ts`)
- **HTTP bridge:** `serve.ts` → Express (`api.ts`, `mcp-http.ts`) for web UI
- **CLI direct:** `gitnexus query|context|impact|cypher` in `tool.ts`
3. **Query & agents**
- **MCP (stdio):** `gitnexus/src/cli/mcp.ts` → `startMCPServer` → `LocalBackend` (`gitnexus/src/mcp/local/local-backend.ts`) opens registered repos and serves **tools** from `gitnexus/src/mcp/tools.ts` and **resources** from `gitnexus/src/mcp/resources.ts`.
- **Bridge HTTP:** `gitnexus/src/cli/serve.ts` → Express app in `gitnexus/src/server/api.ts` (CORS-limited) exposes REST + MCP-over-HTTP for the web UI.
- **CLI tools (no MCP):** `gitnexus query`, `context`, `impact`, `cypher` in `gitnexus/src/cli/tool.ts` call the same backend for scripts and CI.
4. **Staleness** — `staleness.ts` compares indexed `lastCommit` to `HEAD`, surfaces hints.
4. **Staleness**
- `gitnexus/src/mcp/staleness.ts` compares indexed `lastCommit` to `HEAD` and surfaces hints when the graph is behind git.
## MCP tools
## MCP tools (summary)
| Tool | Purpose |
|------|---------|
| `list_repos` | Discover indexed repos |
| `query` | Hybrid BM25 + vector search over the graph |
| `cypher` | Ad hoc Cypher against the schema |
| `context` | Callers, callees, processes for one symbol |
| `impact` | Blast radius (upstream/downstream) with risk summary |
| `detect_changes` | Map git diffs to affected symbols and processes |
| `rename` | Graph-assisted multi-file rename with `dry_run` preview |
| `api_impact` | Pre-change impact report for an API route handler |
| `route_map` | API route → handler → consumer mappings |
| `tool_map` | MCP/RPC tool definitions and handlers |
| `shape_check` | Response shape vs consumer property access mismatches |
| `group_list` | List repo groups or details for one group |
| `group_query` | Cross-repo search in a group (reciprocal rank fusion) |
| `group_sync` | Rebuild group Contract Registry (`contracts.json`) |
| `group_contracts` | Inspect group contracts and cross-links |
| `group_status` | Index and Contract Registry staleness per repo in a group |
| `list_repos` | Discover indexed repositories when more than one is registered. |
| `query` | Natural-language / keyword search over the graph (hybrid BM25 + optional vectors). |
| `cypher` | Ad hoc **Cypher** against the schema (see resource `gitnexus://repo/{name}/schema`). |
| `context` | Callers, callees, processes for one symbol (with disambiguation). |
| `impact` | Blast radius (upstream/downstream) with depth and risk summary. |
| `detect_changes` | Map git diffs to affected symbols and processes. |
| `rename` | Graph-assisted rename with `dry_run` preview (`graph` vs `text_search` confidence). |
## Where to change what
| Concern | Start in |
|---------|----------|
| CLI commands/flags | `src/cli/` (`index.ts`, per-command modules) |
| Parsing/graph construction | `src/core/ingestion/pipeline-phases/` + `pipeline.ts` |
| Graph schema/DB | `src/core/lbug/` (`schema.ts`, `lbug-adapter.ts`) |
| MCP tools/resources | `src/mcp/server.ts`, `tools.ts`, `resources.ts` |
| Search ranking | `src/core/search/` (BM25, hybrid fusion) |
| Embeddings | `src/core/embeddings/` + `src/core/run-analyze.ts` |
| Wiki generation | `src/core/wiki/` |
| Language support | `src/core/ingestion/languages/` + `tree-sitter-queries.ts` + `gitnexus-shared/src/languages.ts` |
| Import resolution | `src/core/ingestion/import-processor.ts` + `import-resolvers/configs/` + `model/resolution-context.ts` |
| Call resolution/MRO | `src/core/ingestion/call-processor.ts` + `model/resolve.ts` |
| Type extraction | `src/core/ingestion/type-extractors/` |
| Worker pool | `src/core/ingestion/workers/` |
| Web UI | `gitnexus-web/src/` |
| CI | `.github/workflows/*.yml`, `.github/actions/` |
> Paths above are relative to `gitnexus/` unless they start with `gitnexus-web/` or `.github/`.
---
| If you are changing… | Start in… |
|----------------------|-----------|
| CLI commands / flags | `gitnexus/src/cli/` (`index.ts`, per-command modules). |
| Parsing or graph construction | `gitnexus/src/core/ingestion/pipeline-phases/` (individual phase files), `pipeline.ts` (orchestrator). |
| Graph schema / DB access | `gitnexus/src/core/lbug/` (`schema.ts`, `lbug-adapter.ts`), `gitnexus/src/mcp/core/lbug-adapter.ts` if MCP-specific. |
| MCP protocol, tools, resources | `gitnexus/src/mcp/server.ts`, `tools.ts`, `resources.ts`. |
| Search ranking | `gitnexus/src/core/search/` (BM25, hybrid fusion). |
| Embeddings | `gitnexus/src/core/embeddings/`, phases in `analyze.ts`. |
| Wiki generation | `gitnexus/src/core/wiki/`. |
| Web UI behavior | `gitnexus-web/src/` (components, workers, graph client). |
| CI | `.github/workflows/*.yml`, `.github/actions/setup-gitnexus/`. |
## Pipeline Phase DAG
12 phases defined in `gitnexus/src/core/ingestion/pipeline-phases/`, each with explicit `deps` and typed output.
The ingestion pipeline is a DAG of named phases. Each phase is defined in its own file under `gitnexus/src/core/ingestion/pipeline-phases/` with explicit dependencies, typed inputs, and typed outputs.
```
scan → structure → [markdown, cobol] → parse → [routes, tools, orm]
→ crossFile → mro → communities → processes
```
| Phase | File | Deps | Output |
|-------|------|------|--------|
| `scan` | `scan.ts` | (root) | File paths + sizes |
| `structure` | `structure.ts` | `scan` | File/Folder nodes, CONTAINS edges, `allPathSet` |
| `markdown` | `markdown.ts` | `structure` | Section nodes, cross-link edges from .md/.mdx |
| `cobol` | `cobol.ts` | `structure` | COBOL program/paragraph/section nodes (regex, no tree-sitter) |
| `parse` | `parse.ts` + `parse-impl.ts` | `structure`, `markdown`, `cobol` | Symbol nodes, IMPORTS/CALLS/EXTENDS edges, extracted routes/tools/ORM queries |
| `routes` | `routes.ts` | `parse` | Route nodes + HANDLES_ROUTE edges (Next.js, Expo, PHP, decorators) |
| `tools` | `tools.ts` | `parse` | Tool nodes + HANDLES_TOOL edges |
| `orm` | `orm.ts` | `parse` | QUERIES edges (Prisma, Supabase) |
| `crossFile` | `cross-file.ts` + `cross-file-impl.ts` | `parse`, `routes`, `tools`, `orm` | Cross-file type propagation in topological import order |
| `mro` | `mro.ts` | `crossFile`, `structure` | METHOD_OVERRIDES + METHOD_IMPLEMENTS edges |
| `communities` | `communities.ts` | `mro`, `structure` | Community nodes + MEMBER_OF edges (Leiden algorithm) |
| `processes` | `processes.ts` | `communities`, `routes`, `tools`, `structure` | Process nodes + STEP_IN_PROCESS edges |
### Phase files
**Non-phase files in the same directory:** `parse-impl.ts`, `cross-file-impl.ts` (implementation), `wildcard-synthesis.ts` (whole-module import expansion), `orm-extraction.ts` (sequential ORM fallback), `types.ts`, `runner.ts`, `index.ts`.
### DAG runner
`runner.ts` — static phase graph, no plugins, compile-time type safety.
1. **Validation** — Kahn's topological sort. Rejects on: duplicate names, missing deps, cycles (DFS traces the concrete cycle path, e.g., `A -> B -> C -> A`, plus count of transitively blocked dependents).
2. **Execution** — sequential in topological order. Each phase receives:
- `ctx: PipelineContext` — shared mutable `KnowledgeGraph`, `repoPath`, progress callback, options
- `deps: ReadonlyMap<string, PhaseResult>` — **declared deps only** (runner filters the results map to prevent hidden coupling)
3. **Error handling** — wraps phase errors with the phase name, emits terminal `error` progress event, swallows progress handler errors to preserve the original cause.
4. **Timing** — per-phase `durationMs` in `PhaseResult`, dev-mode console logging.
**Design patterns:**
- **Single graph accumulator** — all phases mutate the same `KnowledgeGraph` in `ctx`; the graph is the primary output.
- **Typed phase access** — `getPhaseOutput<T>(deps, 'name')` for type-safe upstream results.
- **Binding accumulator lifecycle** — created in `parse`, disposed by `crossFile` (in `finally`). No other phase should take ownership.
- **Skippable phases** — `skipGraphPhases` omits MRO/communities/processes (faster tests). `skipWorkers` forces sequential parsing.
| Phase | File | Dependencies | What it does |
|-------|------|-------------|--------------|
| `scan` | `scan.ts` | (root) | Walk repo filesystem, collect paths + sizes |
| `structure` | `structure.ts` | `scan` | Build File/Folder nodes + CONTAINS edges |
| `markdown` | `markdown.ts` | `structure` | Extract headings and cross-links from .md/.mdx |
| `cobol` | `cobol.ts` | `structure` | Regex-based COBOL/JCL extraction |
| `parse` | `parse.ts` + `parse-impl.ts` | `structure`, `markdown`, `cobol` | Chunked tree-sitter parse, import/call/heritage resolution |
| `routes` | `routes.ts` | `parse` | Route registry (Next.js, Expo, PHP, decorator-based) |
| `tools` | `tools.ts` | `parse` | MCP/RPC tool detection |
| `orm` | `orm.ts` | `parse` | Prisma/Supabase ORM query edges |
| `crossFile` | `cross-file.ts` + `cross-file-impl.ts` | `parse`, `routes`, `tools`, `orm` | Cross-file type propagation in topological order |
| `mro` | `mro.ts` | `crossFile` | Method Resolution Order, METHOD_OVERRIDES edges |
| `communities` | `communities.ts` | `mro` | Leiden community detection |
| `processes` | `processes.ts` | `communities`, `routes`, `tools` | Execution flow detection, Route/Tool → Process links |
### How to add a new phase
1. Create `pipeline-phases/my-phase.ts` with a `PipelinePhase<MyOutput>` (name, deps, execute)
2. Export from `pipeline-phases/index.ts`
3. Add to `buildPhaseList()` in `pipeline.ts`
1. Create a new file in `pipeline-phases/` (e.g. `my-phase.ts`)
2. Define a `PipelinePhase<MyOutput>` object with `name`, `deps`, and `execute(ctx, deps)`
3. Export it from `pipeline-phases/index.ts`
4. Add it to the `buildPhaseList()` function in `pipeline.ts`
```typescript
import type { PipelinePhase, PhaseResult } from './types.js';
// pipeline-phases/my-phase.ts
import type { PipelinePhase, PipelineContext, PhaseResult } from './types.js';
import { getPhaseOutput } from './types.js';
import type { ParseOutput } from './parse.js';
@@ -131,235 +101,81 @@ export interface MyPhaseOutput { /* ... */ }
export const myPhase: PipelinePhase<MyPhaseOutput> = {
name: 'myPhase',
deps: ['parse'],
deps: ['parse'], // runs after parse completes
async execute(ctx, deps) {
const { allPaths } = getPhaseOutput<ParseOutput>(deps, 'parse');
// ... write to ctx.graph ...
// ... do work, write to ctx.graph ...
return { /* typed output */ };
},
};
```
---
### DAG runner
## Call-Resolution DAG
Typed 6-stage pipeline in `call-processor.ts` (inside the `parse` phase) that resolves method/function calls and emits CALLS edges. Language behavior plugs in at two `LanguageProvider` hook points (stages 3–4); shared code names no languages. Scope: call resolution only — import resolution, type extraction, heritage, and symbol-table population live in other phases.
### Stages
```
extract-call ──▶ classify-form ──▶ infer-receiver ──▶ select-dispatch ──▶ resolve-target ──▶ emit-edge
(1) (2) (3) [hook] (4) [hook] (5) (6)
```
| Stage | Produces | Location |
|-------|----------|----------|
| **extract-call** | `ExtractedCallSite` (name, form, receiver, argCount) | `call-extractors/` (per-language); runs in worker |
| **classify-form** | callForm (`free`/`member`/`constructor`) + arity | `call-analysis.ts` → `inferCallForm`; shared, runs in worker |
| **infer-receiver** | `ReceiverEnriched` (receiver type finalized) | `call-processor.ts`; shared default chain, then `inferImplicitReceiver` hook |
| **select-dispatch** | `DispatchDecision` (primary, fallback, ancestryView) | `selectDispatch` hook, falls back to shared default |
| **resolve-target** | `TieredCandidates` | `model/resolve.ts` → `lookupMethodByOwnerWithMRO` (MRO walk) |
| **emit-edge** | CALLS edge in graph | `call-processor.ts`; writes edge with confidence tier |
### Provider hooks
Both hooks are optional on `LanguageProvider`. Ruby is the only current implementer.
**`inferImplicitReceiver`** — called after shared infer-receiver defaults. Returns `ImplicitReceiverOverride | null`.
| | |
|---|---|
| Inputs | `calledName`, `callForm`, `receiverName`, `receiverTypeName`, `callNode` (AST), `filePath` |
| Non-null fields | `callForm`, `receiverName`, `receiverTypeName` (required); `receiverSource: 'implicit-self'` (fixed); `hint?` (opaque, passed to `selectDispatch`) |
| Null | Keep existing `ReceiverEnriched` state |
**`selectDispatch`** — called after infer-receiver (including hook). Returns `DispatchDecision | null`; null uses shared default (constructor → `primary:'constructor'`; typed receiver → `primary:'owner-scoped'`; else → `primary:'free'`).
| | |
|---|---|
| Inputs | `calledName`, `callForm`, `receiverName`, `receiverTypeName`, `receiverSource`, `hint` |
| Non-null fields | `primary: 'owner-scoped' \| 'free' \| 'constructor'`; `fallback?: 'free-arity-narrowed'`; `ancestryView?: 'instance' \| 'singleton'`; `hint?` |
**`DispatchDecision` field semantics:**
- `primary: 'owner-scoped'` — MRO walk from receiver's type; used when receiver type is known.
- `fallback: 'free-arity-narrowed'` — after owner-scoped miss, search free-call candidates by arity only (Ruby uses this for implicit-self calls that miss their owner's MRO).
- `ancestryView: 'singleton'` — walk singleton/class ancestry instead of instance ancestry (Ruby `def self.foo` bodies, so `extend`-ed methods are found).
### Adding language behavior
1. **Implicit receivers** — implement `inferImplicitReceiver`: return null if call already has a receiver; otherwise use `findEnclosingClassInfo` (`ast-helpers.ts`) to find the enclosing context, return `ImplicitReceiverOverride` with `receiverSource: 'implicit-self'`, and optionally set `hint` for `selectDispatch`.
2. **Custom dispatch** — implement `selectDispatch`: inspect `receiverSource` and `hint`, return `DispatchDecision` with `primary`, optional `fallback`, optional `ancestryView`; return null to keep shared defaults.
3. **MRO strategy** — confirm `mroStrategy` is `'first-wins'`, `'c3'`, `'ruby-mixin'`, or `'none'`; consumed by `lookupMethodByOwnerWithMRO`.
**Ruby example** (`languages/ruby.ts` + `utils/ruby-self-call.ts`): `inferImplicitReceiver` rewrites bare-identifier calls to `self.method` and sets `hint` to `'instance'`/`'singleton'`; `selectDispatch` uses hint for `ancestryView` and adds `fallback: 'free-arity-narrowed'` for implicit-self calls.
### Code references
| Module | Purpose |
|--------|---------|
| `core/ingestion/call-types.ts` | DAG types: `ReceiverEnriched`, `DispatchDecision`, `ImplicitReceiverOverride` |
| `core/ingestion/language-provider.ts` | Hook signatures: `inferImplicitReceiver`, `selectDispatch` |
| `core/ingestion/call-processor.ts` | `processCalls`: stages 3–6 |
| `core/ingestion/model/resolve.ts` | `lookupMethodByOwnerWithMRO`: stage 5 MRO walk |
| `core/ingestion/languages/ruby.ts` | Both hooks + `mroStrategy: 'ruby-mixin'` |
| `core/ingestion/utils/ruby-self-call.ts` | Bare-call rewrite for `inferImplicitReceiver` |
---
## Language-agnostic graph feeding
16 languages → single unified graph. Four abstraction layers:
```
Unified Graph Schema (44 node types, 21 relationship types)
↑
Unified Resolution (3-tier name lookup + MRO walk)
↑
Language Providers (import semantics, type config, export checker, MRO strategy)
↑
Tree-Sitter Queries (per-language S-expressions, unified capture tags)
```
### Language providers
Each language implements `LanguageProvider` (`language-provider.ts`). Key fields:
| Field | Purpose |
|-------|---------|
| `id`, `extensions` | Language identity and file matching |
| `treeSitterQueries` | S-expression queries for AST extraction |
| `importSemantics` | `named` / `wildcard-leaf` / `wildcard-transitive` / `namespace` |
| `importResolver` | Language-specific path → file resolution |
| `exportChecker` | Public/exported symbol detection |
| `typeConfig` | Type annotation extraction rules |
| `mroStrategy` | `first-wins` / `c3` / `none` |
16 providers in `languages/index.ts` via `satisfies Record<SupportedLanguages, LanguageProvider>` — missing a language is a compile error.
### Unified capture tags
Per-language tree-sitter queries use different AST node names but produce the **same semantic capture tags**: `@definition.class`, `@definition.function`, `@call.name`, `@import.source`, `@heritage.extends`. Downstream extraction needs no language branching. Defined in `tree-sitter-queries.ts`.
### Import resolution
Per-language import resolution uses the **configs + factory** pattern (like call/method/class extractors). Each language declares an `ImportResolutionConfig` in `import-resolvers/configs/`, listing an ordered chain of `ImportResolverStrategy` functions. `createImportResolver()` (in `resolver-factory.ts`) composes them: first non-null result wins. Low-level helpers shared across strategies live alongside the configs in `import-resolvers/` (e.g. `go.ts`, `rust.ts`, `python.ts`).
Unified 3-tier algorithm (`model/resolution-context.ts`), per-language `importSemantics` controls which tier activates:
| Tier | Confidence | Mechanism |
|------|-----------|-----------|
| 1 — same-file | 0.95 | Symbol table for caller's file |
| 2 — import-scoped | 0.9 | `NamedImportMap` chains (named) or all files in `importMap` (wildcard) |
| 3 — global | 0.5 | O(1) index lookups: class, impl, callable. Fallback only |
| Import strategy | Languages | Behavior |
|----------------|-----------|----------|
| `named` | TS, JS, Java, C#, Rust, PHP, Kotlin | Only explicitly imported names visible |
| `wildcard-leaf` | Go, Ruby, Swift, Dart | Whole-package import, no transitive re-exports |
| `wildcard-transitive` | C, C++ | `#include` closure chains through re-exports |
| `namespace` | Python | Module aliases resolved at call site |
### Chunked parse-and-resolve
`parse` processes files in ~20 MB byte-budget chunks to bound memory. Per chunk:
1. Worker pool dispatches files (or sequential fallback via `skipWorkers`)
2. Each worker: detect language → load grammar → run queries → return unified `ParseWorkerResult`
3. Synthesize wildcard bindings (`wildcard-synthesis.ts`)
4. Resolve imports and heritage
5. Collect `BindingAccumulator` entries for cross-file propagation
Workers: `workers/worker-pool.ts`, `workers/parse-worker.ts`.
### Heritage and MRO
All languages emit unified `ExtractedHeritage` (child, parent, `EXTENDS`/`IMPLEMENTS`). MRO phase walks the heritage graph using per-language strategy:
- **`first-wins`** — Java, C#, C++, TS, Ruby, Go
- **`c3`** — Python (C3 linearization)
- **`none`** — single-inheritance languages
Unified walk: `lookupMethodByOwnerWithMRO()` in `model/resolve.ts`.
---
## Full analysis flow
`runFullAnalysis` in `run-analyze.ts` orchestrates everything around the pipeline:
```
CLI (analyze.ts) → runFullAnalysis(repoPath, options, callbacks)
1. Early exit if lastCommit == HEAD (unless --force) [0%]
2. Cache existing embeddings from prior index [0%]
3. runPipelineFromRepo() → KnowledgeGraph [0-60%]
4. Clean up legacy KuzuDB files [60%]
5. initLbug() → loadGraphToLbug() via CSV streaming [60-85%]
6. Create FTS indexes (File, Function, Class, Method...) [85-90%]
7. Restore cached embeddings (batch insert) [88%]
8. Generate new embeddings if --embeddings [90-98%]
9. Save metadata + register repo + update .gitignore [98-100%]
10. Generate AI context files (AGENTS.md, CLAUDE.md) [100%]
```
**Options:** `--force` (rebuild regardless), `--embeddings` (opt-in, skipped if >50k nodes), `--skipGit`, `--noStats`.
## Storage
```
<repo>/.gitnexus/
├── lbug # LadybugDB database
├── lbug.wal # Write-ahead log
├── lbug.lock # Single-writer lock
└── meta.json # lastCommit, indexedAt, stats
~/.gitnexus/
└── registry.json # Global repo registry (MCP discovery)
```
Managed by `repo-manager.ts`.
## LadybugDB schema
Defined in `lbug/schema.ts`. Separate node tables per type, single `CodeRelation` table.
**Node tables:** File, Folder, Function, Class, Interface, Method, Constructor, CodeElement, Struct, Enum, Macro, Typedef, Union, Namespace, Trait, Impl, TypeAlias, Const, Static, Property, Record, Delegate, Annotation, Template, Module, Community, Process, Route, Tool, Section, Embedding.
**Relation types** (`CodeRelation.type`): CONTAINS, DEFINES, CALLS, IMPORTS, EXTENDS, IMPLEMENTS, HAS_METHOD, HAS_PROPERTY, ACCESSES, METHOD_OVERRIDES, METHOD_IMPLEMENTS, MEMBER_OF, STEP_IN_PROCESS, HANDLES_ROUTE, FETCHES, HANDLES_TOOL, ENTRY_POINT_OF.
## Embeddings and search
**Embeddings** (`src/core/embeddings/`): Snowflake arctic-embed-xs (384D). Embeddable: File, Function, Class, Method, Interface. Incremental via SHA1 content hash. Separate `Embedding` table.
**Search** (`src/core/search/`): Hybrid BM25 + semantic vector, merged via Reciprocal Rank Fusion (K=60).
The runner (`pipeline-phases/runner.ts`) validates the DAG at startup (detects cycles and missing deps via topological sort), then executes phases in dependency order. Each phase receives:
- `ctx: PipelineContext` — shared graph, repoPath, progress callback
- `deps: Map<string, PhaseResult>` — outputs from all upstream phases
## Known limitations
### Overloaded method resolution
Node IDs use arity suffix (`#<paramCount>`): `Method:file:Class.method#1` vs `#2`.
Method and Constructor node IDs include an arity suffix (`#<paramCount>`) to
disambiguate overloaded methods. Two overloads with different parameter counts
produce distinct graph nodes: `Method:file:Class.method#1` vs
`Method:file:Class.method#2`.
**Same-arity disambiguation:** type-hash suffix `~type1,type2` when collision detected and type annotations present. Languages without types (Python, Ruby, JS) use arity-only. TS/JS overload signatures excluded (collapse to implementation body). See #651.
**Same-arity overload disambiguation:** When two overloads share the same
parameter count but differ in types (e.g. `save(int)` vs `save(String)`), a
type-hash suffix `~type1,type2` is appended to produce distinct node IDs:
`Method:file:Class.save#1~int` vs `Method:file:Class.save#1~String`. The suffix
is only added when a same-arity collision is detected within a class and all
parameters have non-null type annotations. Languages without type info (Python,
Ruby, JS) fall back to arity-only IDs. TypeScript/JavaScript overload signatures
are intentionally excluded from type-hashing because they are declaration-only
contracts that should collapse to the implementation body's node ID. See issue
\#651.
**C++ const-qualified:** `$const` suffix after type-hash when non-const collision exists: `Method:file:Container.begin#0$const`.
**C++ const-qualified overload disambiguation:** Methods overloaded by const
qualification (e.g. `begin()` vs `begin() const`) are disambiguated via an
`isConst` property and a `$const` ID suffix appended to the const-qualified
variant when a non-const collision exists. The `$const` suffix appears after the
type-hash suffix: e.g. `Method:file:Container.begin#0$const`.
**Generic/template types:** type-hash uses `rawType` (full AST text including generics): `~vector<int>` vs `~vector<std::string>`.
**Generic/template type preservation in type-hash:** The type-hash suffix uses
`rawType` (full AST text including generic/template args) rather than the
simplified `type` from `extractSimpleTypeName`. This means C++ template overloads
like `process(vector<int>)` vs `process(vector<string>)` produce distinct IDs:
`~vector<int>` vs `~vector<std::string>`. Java generic overloads like
`process(List<String>)` vs `process(List<Integer>)` are a compile error due to
type erasure, so this gap is theoretical for Java.
**ID stability:** collision-only tags mean IDs change when overloads are added. `save#1` becomes `save#1~int` when `save(String)` is added.
**ID stability on first overload:** Type and const tags are collision-only. When
a class has `save(int)` as its only `save` method, the ID is `save#1` (no tag).
Adding `save(String)` changes the original to `save#1~int`. This is correct for
fresh analysis but means IDs are not stable across overload additions. Future
incremental re-analysis should account for this.
**Variadic matching:** confidence 0.7 when one side is variadic and the other has fixed count.
**Variadic method matching:** When one side is variadic (`parameterCount`
undefined) and the other has a fixed count, `METHOD_IMPLEMENTS` edges are
emitted with confidence 0.7 instead of 1.0. Variadic methods like
`foo(String... args)` may superficially match `foo(String s)` by type but
are not guaranteed to be interchangeable across all languages (Java/Kotlin
accept this via varargs sugar; TypeScript, C#, Rust do not).
**METHOD_IMPLEMENTS confidence tiering:**
**Confidence tiering** for `METHOD_IMPLEMENTS` edges:
| Match quality | Confidence |
|---|---|
| Exact parameter types match | 1.0 |
| Arity match, types unavailable | 1.0 |
| Variadic vs fixed | 0.7 |
| Insufficient info | 0.7 |
| Match quality | Confidence | When |
|---|---|---|
| Exact parameter types match | 1.0 | Both sides have `parameterTypes` arrays and they match |
| Arity (count) matches | 1.0 | Both sides have `parameterCount`, types unavailable |
| Variadic vs fixed | 0.7 | One side is variadic, other has fixed count |
| Lenient (insufficient info) | 0.7 | One or both sides lack type and count data |
## Related docs
- [MIGRATION.md](MIGRATION.md) — breaking changes and migration guidance
- [RUNBOOK.md](RUNBOOK.md) — operational commands and recovery
- [GUARDRAILS.md](GUARDRAILS.md) — safety boundaries for humans and agents
- [TESTING.md](TESTING.md) — how to run tests
- `AGENTS.md` / `CLAUDE.md` — agent workflows and tool usage
- [MIGRATION.md](MIGRATION.md) — breaking changes and migration guidance.
- [RUNBOOK.md](RUNBOOK.md) — operational commands and recovery.
- [GUARDRAILS.md](GUARDRAILS.md) — safety boundaries for humans and agents.
- [TESTING.md](TESTING.md) — how to run tests.
- `AGENTS.md` / `CLAUDE.md` — agent workflows and tool usage expectations for **this** repo when indexed by GitNexus.
+203 -2
View File
@@ -35,7 +35,6 @@ If always-on instructions grow, load deep conventions via conditional reads (e.g
## Reference Documentation
- **This repository:** [AGENTS.md](AGENTS.md) (Cursor + monorepo notes), [ARCHITECTURE.md](ARCHITECTURE.md), [CONTRIBUTING.md](CONTRIBUTING.md), [GUARDRAILS.md](GUARDRAILS.md).
- **Call-resolution DAG:** See ARCHITECTURE.md § Call-Resolution DAG. Shared pipeline code in `gitnexus/src/core/ingestion/` must not name languages — use `LanguageProvider` hooks instead (see AGENTS.md).
- **GitNexus:** `.claude/skills/gitnexus/`; MCP and indexed-repo rules live only in [AGENTS.md](AGENTS.md) (`gitnexus:start` … `gitnexus:end`). See **GitNexus rules** below.
## Changelog
@@ -51,4 +50,206 @@ If always-on instructions grow, load deep conventions via conditional reads (e.g
## GitNexus rules
See the `<!-- gitnexus:start --> … <!-- gitnexus:end -->` block in **[AGENTS.md](AGENTS.md)** for the canonical MCP tools, impact analysis rules, and index instructions.
GitNexus MCP rules are in the `<!-- gitnexus:start -->
# GitNexus — Code Intelligence
This project is indexed by GitNexus as **GitNexus** (4325 symbols, 10556 relationships, 300 execution flows). Use the GitNexus MCP tools to understand code, assess impact, and navigate safely.
> If any GitNexus tool warns the index is stale, run `npx gitnexus analyze` in terminal first.
## Always Do
- **MUST run impact analysis before editing any symbol.** Before modifying a function, class, or method, run `gitnexus_impact({target: "symbolName", direction: "upstream"})` and report the blast radius (direct callers, affected processes, risk level) to the user.
- **MUST run `gitnexus_detect_changes()` before committing** to verify your changes only affect expected symbols and execution flows.
- **MUST warn the user** if impact analysis returns HIGH or CRITICAL risk before proceeding with edits.
- When exploring unfamiliar code, use `gitnexus_query({query: "concept"})` to find execution flows instead of grepping. It returns process-grouped results ranked by relevance.
- When you need full context on a specific symbol — callers, callees, which execution flows it participates in — use `gitnexus_context({name: "symbolName"})`.
## When Debugging
1. `gitnexus_query({query: "<error or symptom>"})` — find execution flows related to the issue
2. `gitnexus_context({name: "<suspect function>"})` — see all callers, callees, and process participation
3. `READ gitnexus://repo/GitNexus/process/{processName}` — trace the full execution flow step by step
4. For regressions: `gitnexus_detect_changes({scope: "compare", base_ref: "main"})` — see what your branch changed
## When Refactoring
- **Renaming**: MUST use `gitnexus_rename({symbol_name: "old", new_name: "new", dry_run: true})` first. Review the preview — graph edits are safe, text_search edits need manual review. Then run with `dry_run: false`.
- **Extracting/Splitting**: MUST run `gitnexus_context({name: "target"})` to see all incoming/outgoing refs, then `gitnexus_impact({target: "target", direction: "upstream"})` to find all external callers before moving code.
- After any refactor: run `gitnexus_detect_changes({scope: "all"})` to verify only expected files changed.
## Never Do
- NEVER edit a function, class, or method without first running `gitnexus_impact` on it.
- NEVER ignore HIGH or CRITICAL risk warnings from impact analysis.
- NEVER rename symbols with find-and-replace — use `gitnexus_rename` which understands the call graph.
- NEVER commit changes without running `gitnexus_detect_changes()` to check affected scope.
## Tools Quick Reference
| Tool | When to use | Command |
|------|-------------|---------|
| `query` | Find code by concept | `gitnexus_query({query: "auth validation"})` |
| `context` | 360-degree view of one symbol | `gitnexus_context({name: "validateUser"})` |
| `impact` | Blast radius before editing | `gitnexus_impact({target: "X", direction: "upstream"})` |
| `detect_changes` | Pre-commit scope check | `gitnexus_detect_changes({scope: "staged"})` |
| `rename` | Safe multi-file rename | `gitnexus_rename({symbol_name: "old", new_name: "new", dry_run: true})` |
| `cypher` | Custom graph queries | `gitnexus_cypher({query: "MATCH ..."})` |
## Impact Risk Levels
| Depth | Meaning | Action |
|-------|---------|--------|
| d=1 | WILL BREAK — direct callers/importers | MUST update these |
| d=2 | LIKELY AFFECTED — indirect deps | Should test |
| d=3 | MAY NEED TESTING — transitive | Test if critical path |
## Resources
| Resource | Use for |
|----------|---------|
| `gitnexus://repo/GitNexus/context` | Codebase overview, check index freshness |
| `gitnexus://repo/GitNexus/clusters` | All functional areas |
| `gitnexus://repo/GitNexus/processes` | All execution flows |
| `gitnexus://repo/GitNexus/process/{name}` | Step-by-step execution trace |
## Self-Check Before Finishing
Before completing any code modification task, verify:
1. `gitnexus_impact` was run for all modified symbols
2. No HIGH/CRITICAL risk warnings were ignored
3. `gitnexus_detect_changes()` confirms changes match expected scope
4. All d=1 (WILL BREAK) dependents were updated
## Keeping the Index Fresh
After committing code changes, the GitNexus index becomes stale. Re-run analyze to update it:
```bash
npx gitnexus analyze
```
If the index previously included embeddings, preserve them by adding `--embeddings`:
```bash
npx gitnexus analyze --embeddings
```
To check whether embeddings exist, inspect `.gitnexus/meta.json` — the `stats.embeddings` field shows the count (0 means no embeddings). **Running analyze without `--embeddings` will delete any previously generated embeddings.**
> Claude Code users: A PostToolUse hook handles this automatically after `git commit` and `git merge`.
## CLI
| Task | Read this skill file |
|------|---------------------|
| Understand architecture / "How does X work?" | `.claude/skills/gitnexus/gitnexus-exploring/SKILL.md` |
| Blast radius / "What breaks if I change X?" | `.claude/skills/gitnexus/gitnexus-impact-analysis/SKILL.md` |
| Trace bugs / "Why is X failing?" | `.claude/skills/gitnexus/gitnexus-debugging/SKILL.md` |
| Rename / extract / split / refactor | `.claude/skills/gitnexus/gitnexus-refactoring/SKILL.md` |
| Tools, resources, schema reference | `.claude/skills/gitnexus/gitnexus-guide/SKILL.md` |
| Index, status, clean, wiki CLI commands | `.claude/skills/gitnexus/gitnexus-cli/SKILL.md` |
<!-- gitnexus:end -->` block in **[AGENTS.md](AGENTS.md)** — load that section when working with MCP tools or the graph index.
<!-- gitnexus:start -->
# GitNexus — Code Intelligence
This project is indexed by GitNexus as **GitNexus** (3298 symbols, 7954 relationships, 185 execution flows). Use the GitNexus MCP tools to understand code, assess impact, and navigate safely.
> If any GitNexus tool warns the index is stale, run `npx gitnexus analyze` in terminal first.
## Always Do
- **MUST run impact analysis before editing any symbol.** Before modifying a function, class, or method, run `gitnexus_impact({target: "symbolName", direction: "upstream"})` and report the blast radius (direct callers, affected processes, risk level) to the user.
- **MUST run `gitnexus_detect_changes()` before committing** to verify your changes only affect expected symbols and execution flows.
- **MUST warn the user** if impact analysis returns HIGH or CRITICAL risk before proceeding with edits.
- When exploring unfamiliar code, use `gitnexus_query({query: "concept"})` to find execution flows instead of grepping. It returns process-grouped results ranked by relevance.
- When you need full context on a specific symbol — callers, callees, which execution flows it participates in — use `gitnexus_context({name: "symbolName"})`.
## When Debugging
1. `gitnexus_query({query: "<error or symptom>"})` — find execution flows related to the issue
2. `gitnexus_context({name: "<suspect function>"})` — see all callers, callees, and process participation
3. `READ gitnexus://repo/GitNexus/process/{processName}` — trace the full execution flow step by step
4. For regressions: `gitnexus_detect_changes({scope: "compare", base_ref: "main"})` — see what your branch changed
## When Refactoring
- **Renaming**: MUST use `gitnexus_rename({symbol_name: "old", new_name: "new", dry_run: true})` first. Review the preview — graph edits are safe, text_search edits need manual review. Then run with `dry_run: false`.
- **Extracting/Splitting**: MUST run `gitnexus_context({name: "target"})` to see all incoming/outgoing refs, then `gitnexus_impact({target: "target", direction: "upstream"})` to find all external callers before moving code.
- After any refactor: run `gitnexus_detect_changes({scope: "all"})` to verify only expected files changed.
## Never Do
- NEVER edit a function, class, or method without first running `gitnexus_impact` on it.
- NEVER ignore HIGH or CRITICAL risk warnings from impact analysis.
- NEVER rename symbols with find-and-replace — use `gitnexus_rename` which understands the call graph.
- NEVER commit changes without running `gitnexus_detect_changes()` to check affected scope.
## Tools Quick Reference
| Tool | When to use | Command |
|------|-------------|---------|
| `query` | Find code by concept | `gitnexus_query({query: "auth validation"})` |
| `context` | 360-degree view of one symbol | `gitnexus_context({name: "validateUser"})` |
| `impact` | Blast radius before editing | `gitnexus_impact({target: "X", direction: "upstream"})` |
| `detect_changes` | Pre-commit scope check | `gitnexus_detect_changes({scope: "staged"})` |
| `rename` | Safe multi-file rename | `gitnexus_rename({symbol_name: "old", new_name: "new", dry_run: true})` |
| `cypher` | Custom graph queries | `gitnexus_cypher({query: "MATCH ..."})` |
## Impact Risk Levels
| Depth | Meaning | Action |
|-------|---------|--------|
| d=1 | WILL BREAK — direct callers/importers | MUST update these |
| d=2 | LIKELY AFFECTED — indirect deps | Should test |
| d=3 | MAY NEED TESTING — transitive | Test if critical path |
## Resources
| Resource | Use for |
|----------|---------|
| `gitnexus://repo/GitNexus/context` | Codebase overview, check index freshness |
| `gitnexus://repo/GitNexus/clusters` | All functional areas |
| `gitnexus://repo/GitNexus/processes` | All execution flows |
| `gitnexus://repo/GitNexus/process/{name}` | Step-by-step execution trace |
## Self-Check Before Finishing
Before completing any code modification task, verify:
1. `gitnexus_impact` was run for all modified symbols
2. No HIGH/CRITICAL risk warnings were ignored
3. `gitnexus_detect_changes()` confirms changes match expected scope
4. All d=1 (WILL BREAK) dependents were updated
## Keeping the Index Fresh
After committing code changes, the GitNexus index becomes stale. Re-run analyze to update it:
```bash
npx gitnexus analyze
```
If the index previously included embeddings, preserve them by adding `--embeddings`:
```bash
npx gitnexus analyze --embeddings
```
To check whether embeddings exist, inspect `.gitnexus/meta.json` — the `stats.embeddings` field shows the count (0 means no embeddings). **Running analyze without `--embeddings` will delete any previously generated embeddings.**
> Claude Code users: A PostToolUse hook handles this automatically after `git commit` and `git merge`.
## CLI
| Task | Read this skill file |
|------|---------------------|
| Understand architecture / "How does X work?" | `.claude/skills/gitnexus/gitnexus-exploring/SKILL.md` |
| Blast radius / "What breaks if I change X?" | `.claude/skills/gitnexus/gitnexus-impact-analysis/SKILL.md` |
| Trace bugs / "Why is X failing?" | `.claude/skills/gitnexus/gitnexus-debugging/SKILL.md` |
| Rename / extract / split / refactor | `.claude/skills/gitnexus/gitnexus-refactoring/SKILL.md` |
| Tools, resources, schema reference | `.claude/skills/gitnexus/gitnexus-guide/SKILL.md` |
| Index, status, clean, wiki CLI commands | `.claude/skills/gitnexus/gitnexus-cli/SKILL.md` |
<!-- gitnexus:end -->
+11 -108
View File
@@ -21,127 +21,30 @@ This project uses the [PolyForm Noncommercial License 1.0.0](https://polyformpro
## Branch and pull requests
- Use short-lived branches off the default branch of the repo you are targeting.
- **PR titles MUST follow the conventional-commit format** — `pr-labeler.yml` enforces this on every PR and auto-applies the matching label so release notes group the change correctly.
- Prefer **conventional commits** (short prefix + description), for example:
```text
feat: add graph export option
fix: correct MCP tool schema for query
test: cover cluster merge edge case
docs: clarify analyze flags
```
- **PR title:** `[area] Short description` (e.g. `[cli] Fix index refresh race`).
- **PR description:** what changed, why, how to verify (commands), and any risk or rollback notes.
### Pull request titles
Format: `<type>[(scope)][!]: <subject>`
Allowed types and the release-notes section each one lands in (defined in `.github/release.yml`):
| Type | Label applied | Release-notes section |
|------|---------------|-----------------------|
| `feat` | `enhancement` | 🚀 Features |
| `fix` | `bug` | 🐛 Bug Fixes |
| `perf` | `performance` | 🏎️ Performance |
| `refactor` | `refactor` | 🔄 Refactoring |
| `test` | `test` | 🧪 Tests |
| `ci` | `ci` | 👷 CI/CD |
| `build` / `deps` | `dependencies` | 📦 Dependencies |
| `docs` | `documentation` | (grouped under Other Changes unless a Docs section is added) |
| `chore` / `revert` | `chore` | (excluded from release notes) |
Append `!` to the type (e.g. `feat(api)!: drop /v1 endpoint`) or include `BREAKING CHANGE:` in the PR body to flag a breaking change — the labeler then adds the `breaking` label and the 💥 Breaking Changes section is rendered first.
Examples:
```text
feat(web): add smart chat scroll
fix(extractors): resolve silent contract mis-resolution
perf: avoid O(n²) traversal in heritage walker
chore(deps): bump vitest to 3.0.0
ci: standardize workflow concurrency
```
Commits within a PR may use any style — only the **merged PR title** shows up in release notes, so that's the one the convention applies to.
## Before you open a PR
- [ ] Tests pass for the packages you touched (`gitnexus` and/or `gitnexus-web`).
- [ ] Typecheck passes: `npx tsc --noEmit` in `gitnexus/` and `npx tsc -b --noEmit` in `gitnexus-web/`.
- [ ] No secrets, tokens, or machine-specific paths committed.
- [ ] Documentation updated if behavior or public CLI/MCP contract changes.
- [ ] Pre-commit hook runs clean (`.husky/pre-commit` — formatting via lint-staged + typecheck for staged packages; tests run in CI only).
- [ ] Pre-commit hook runs clean (`.husky/pre-commit` — typecheck + unit tests for staged packages).
## Code review
Maintainers may request changes for correctness, tests, performance, or consistency with existing patterns. Keeping diffs focused makes review faster.
## GitHub Actions — Concurrency Convention
Every workflow under `.github/workflows/` MUST declare a top-level `concurrency:` block using this convention:
- **Group key** starts with `${{ github.workflow }}` so no two workflows can collide on the same group name. The discriminator that follows is chosen per event shape:
- Branch/tag scope: `${{ github.workflow }}-${{ github.ref }}`
- Per-PR scope (for `issue_comment`, `pull_request_review*`, `pull_request` meta events): `${{ github.workflow }}-${{ github.event.pull_request.number || github.event.issue.number }}`
- `workflow_run` scope (e.g. `ci-report.yml`): `${{ github.workflow }}-${{ github.event.workflow_run.pull_requests[0].number || format('{0}/{1}', github.event.workflow_run.head_repository.full_name, github.event.workflow_run.head_branch) }}` — the fork fallback must be stable across reruns (never `workflow_run.id`, which is per-run-unique and defeats serialization).
- Global single-slot (manual dispatch utilities): `${{ github.workflow }}`
- **Reusable workflows invoked via `workflow_call`:** do NOT use `${{ github.workflow }}` in the group key — in called-workflow context its evaluation is ambiguous and can resolve to the caller's name, which would deadlock against the caller's own group. Use a hardcoded literal prefix and a `github.event_name`-aware expression that falls through to `github.run_id` for reusable invocations (see `ci.yml` for the canonical form).
- **Merge queue (`merge_group`)**: when this event is added, use `${{ github.workflow }}-${{ github.event.merge_group.head_ref }}` with `cancel-in-progress: false` (every queue entry is a distinct ref; never cancel).
- **`cancel-in-progress` policy:**
| Event | `cancel-in-progress` | Why |
|-------|----------------------|-----|
| `pull_request` CI run | `true` | New push supersedes old run |
| `push` to `main` | `false` | Every main commit gets validated |
| Tag push (`v*` publish) | `false` | Never cancel mid-publish |
| `push` to `main` for release-candidate | `false` | Never cancel mid-RC publish |
| `workflow_dispatch` (release/publish) | `false` | Manual runs are intentional |
| `workflow_run` (sticky-comment reports) | `false` | Serialize, don't race |
| Per-PR bot workflows (`@claude`, review) | `false` | Serialize comments per PR |
| PR-meta re-checks (pr-description-check) | `true` | Cheap, latest wins |
| Single-slot utilities (triage sweep) | `true` | Latest dispatch supersedes |
- For workflows that serve multiple events at once (e.g. `ci.yml` handles `pull_request`, `push`, and `workflow_call`), make `cancel-in-progress` event-aware:
```yaml
concurrency:
group: ${{ github.workflow }}-${{ github.ref }}
cancel-in-progress: ${{ github.event_name == 'pull_request' }}
```
- When adding a new workflow, copy the concurrency block from an existing workflow of the same event shape.
## AI-assisted contributions
If you use coding agents, follow project context files (e.g. `AGENTS.md`, `CLAUDE.md`) and avoid drive-by refactors unrelated to the issue. Prefer incremental, test-backed changes.
## Releases
Two publish workflows ship `gitnexus` to npm:
- **Stable** (`.github/workflows/publish.yml`) — triggered by pushing any `v*`
tag. Publishes to the `latest` dist-tag with a changelog-backed GitHub
release. Maintainers are expected to tag from `main` as a convention; the
workflow itself does not enforce branch reachability.
- **Release Candidate** (`.github/workflows/release-candidate.yml`) — runs on
every push to `main` (typically a merged PR) plus manual dispatch. Docs-only
changes are skipped via `paths-ignore`. Publishes to the `rc` dist-tag with
version `X.Y.Z-rc.N` and a GitHub prerelease, where:
- `X.Y.Z` is selected automatically. On push (and on dispatch with
`bump: auto`, the default) the workflow **continues the active rc cycle**:
if the registry already has `X.Y.Z-rc.*` versions with `X.Y.Z` > current
`latest`, it reuses the highest such base; otherwise it patch-bumps
from `latest`. Dispatching with `bump: patch|minor|major` **resets**
the cycle from `latest`.
- `N` is auto-incremented against existing `X.Y.Z-rc.*` entries on the
registry. First rc for a given base is `rc.1`.
Idempotency: the workflow pushes an `rc/<HEAD_SHA>` marker tag and a
`v<RC>` release tag **atomically, before** calling `npm publish`. The guard
refuses to re-run once the marker exists, so a post-publish failure will
not mint a duplicate rc for the same commit. The `v<RC>` tag points at a
detached release commit whose `package.json` matches the npm tarball
exactly (traceable releases). Recovery after a partial failure:
```bash
git push --delete origin rc/<HEAD_SHA> v<RC>
# then redispatch the workflow with force: true
```
The rc workflow never moves `latest`. To verify after a change, inspect dist-tags:
```bash
npm view gitnexus dist-tags
```
-36
View File
@@ -1,36 +0,0 @@
ARG BUILDPLATFORM
ARG TARGETPLATFORM
FROM --platform=$BUILDPLATFORM node:22-alpine AS builder
WORKDIR /app
COPY gitnexus-shared/package.json gitnexus-shared/package-lock.json ./gitnexus-shared/
RUN npm ci --prefix gitnexus-shared
COPY gitnexus-shared ./gitnexus-shared
RUN npm run build --prefix gitnexus-shared
COPY gitnexus/package.json ./gitnexus/package.json
COPY gitnexus-web/package.json gitnexus-web/package-lock.json ./gitnexus-web/
RUN npm ci --prefix gitnexus-web
COPY gitnexus-web ./gitnexus-web
RUN npm run build --prefix gitnexus-web
FROM node:22-alpine AS runtime
RUN apk add --no-cache curl
WORKDIR /app
COPY --from=builder /app/gitnexus-web/dist ./dist
COPY docker-server.mjs ./docker-server.mjs
RUN chown -R node:node /app
USER node
EXPOSE 4173
CMD ["node", "docker-server.mjs"]
+46 -43
View File
@@ -1,69 +1,72 @@
# Guardrails — GitNexus
# Guardrails — GitNexus (repo + agents)
Rules for **human contributors** and **AI agents**. Complements `AGENTS.md` (workflows) and `CONTRIBUTING.md` (PR process).
Rules for **human contributors** and **AI agents** working on this codebase or publishing artifacts. These complement `AGENTS.md` / `CLAUDE.md` (which focus on GitNexus-in-GitNexus workflows).
## Scope (least privilege)
## Scope (typical agent session)
- **Read:** Source, tests, docs, public config as needed.
- **Write:** Only files required for the fix or feature; no unrelated formatting or refactors.
- **Execute:** Tests, typecheck, documented CLI commands. No destructive commands on user data without approval.
- **Off-limits:** Other people's machines, production deployments you don't own, credentials you lack permission to use.
When automating changes in this repository, treat scope as **least privilege**:
Maintainer may widen scope per task.
- **Read:** Source, tests, docs, public config as needed for the task.
- **Write:** Only files required for the requested fix or feature; avoid unrelated formatting or refactors.
- **Execute:** Tests, typecheck, and documented CLI commands; do not run destructive commands on user data outside the repo without explicit approval.
- **Off-limits:** Other people’s machines, production deployments you don’t own, and credentials you didn’t receive permission to use.
Adjust explicitly if the maintainer defines a different scope for a task.
---
## Non-negotiables
1. **Never commit secrets** — API keys, tokens, real `.env` values, private URLs, session cookies. Use `.env.example` with placeholders.
2. **Never rename with find-and-replace** in GitNexus-indexed projects — use `rename` MCP tool with `dry_run: true` first, review `graph` vs `text_search` edits. No separate `gitnexus rename` CLI exists.
3. **Run impact analysis before editing shared symbols** — `impact` (upstream) for functions/classes/methods others call. Do not ignore HIGH/CRITICAL without maintainer sign-off.
4. **Run `detect_changes` before commit** — confirm diffs map to expected symbols/processes when the graph is available.
5. **Preserve embeddings** — if `.gitnexus/meta.json` shows embeddings, use `npx gitnexus analyze --embeddings`; plain `analyze` drops them.
1. **Never commit secrets** — API keys, tokens, `.env` with real values, private URLs, or session cookies. Use `.env.example` with placeholders only.
2. **Never rename symbols with blind find-and-replace** when working in a GitNexus-indexed project — use the **`rename` MCP tool** with **`dry_run: true` first**, then review `graph` vs `text_search` edits. (There is no separate `gitnexus rename` CLI; renaming goes through MCP or editor integration.)
3. **Run impact analysis before editing shared symbols** — use **`impact`** (upstream) for functions/classes/methods others call; do not ignore **HIGH** / **CRITICAL** risk without maintainer sign-off.
4. **Prefer `detect_changes` before commit** — confirm diffs map to expected symbols/processes when the graph is available.
5. **Preserve embeddings** — if `.gitnexus/meta.json` shows embeddings, run `npx gitnexus analyze --embeddings` when refreshing the index; plain `analyze` can drop them.
---
## Signs (recurring failure patterns)
Format: **Trigger → Instruction → Reason**. Append new Signs when the same mistake repeats.
Use this format: **Trigger → Instruction → Reason**.
Append new Signs here when the same mistake repeats (e.g. CI broken twice the same way).
### Stale graph after edits
### Sign: Stale graph after edits
- **Trigger:** MCP warns index is behind `HEAD`, or search doesn't match latest commit.
- **Do:** `npx gitnexus analyze` (plus `--embeddings` if used).
- **Why:** Tools query LadybugDB from last analyze; git changes are invisible until re-indexed.
- **Trigger:** MCP or resources warn the index is behind `HEAD`, or code search doesn’t match latest commit.
- **Instruction:** Run `npx gitnexus analyze` from the repo root (plus `--embeddings` if the project used them).
- **Reason:** Tools query LadybugDB built at last analyze; git changes are invisible until re-indexed.
### Embeddings vanished after analyze
### Sign: Embeddings vanished after analyze
- **Trigger:** Semantic search quality drops; `stats.embeddings` in `meta.json` is 0 after refresh.
- **Do:** `npx gitnexus analyze --embeddings`, confirm `meta.json` reflects stored embeddings.
- **Why:** Embedding generation is opt-in; analyze without the flag does not preserve prior vectors.
- **Trigger:** Semantic search quality drops; `stats.embeddings` in `.gitnexus/meta.json` is 0 after a refresh.
- **Instruction:** Re-run `npx gitnexus analyze --embeddings` and confirm `meta.json` reflects stored embeddings.
- **Reason:** Embedding generation is opt-in; analyze without the flag does not preserve prior vectors.
### MCP lists no repos
### Sign: MCP lists no repos
- **Trigger:** MCP stderr says no indexed repos.
- **Do:** `npx gitnexus analyze` in the target repo; verify `npx gitnexus list` shows it.
- **Why:** MCP discovers repos via `~/.gitnexus/registry.json`, populated by analyze.
- **Trigger:** MCP stderr says no indexed repos.
- **Instruction:** Run `npx gitnexus analyze` in the target repository; verify `npx gitnexus list` shows it.
- **Reason:** The MCP server discovers repos via `~/.gitnexus/registry.json`, populated by analyze.
### Wrong repo in multi-repo setups
### Sign: Wrong repo in multi-repo setups
- **Trigger:** Query/impact results belong to another project.
- **Do:** Call `list_repos`, then pass `repo` on subsequent tools.
- **Why:** Default target is ambiguous when multiple repos are registered.
- **Trigger:** Query/impact results clearly belong to another project.
- **Instruction:** Call `list_repos`, then pass **`repo`** on subsequent tools (or use per-workspace MCP config).
- **Reason:** Default target may be ambiguous when multiple repos are registered.
### LadybugDB lock / "database busy"
### Sign: LadybugDB lock / “database busy”
- **Trigger:** Errors opening `.gitnexus/lbug` while MCP and analyze both run.
- **Do:** Stop overlapping processes (one writer at a time). Retry analyze or restart MCP.
- **Why:** Embedded DB expects single-process ownership.
- **Trigger:** Errors opening `.gitnexus/lbug` while MCP and analyze both run.
- **Instruction:** Stop overlapping processes; one writer at a time. Retry analyze or restart MCP.
- **Reason:** Embedded DB expects single-process ownership of the store.
---
## Publishing & supply chain
- **npm:** Do not publish from unreviewed automation. Bump version intentionally; tag releases to match `package.json`.
- **Dependencies:** Minimal, auditable `package.json` changes; run tests and CI after lockfile updates.
- **License:** PolyForm Noncommercial 1.0.0 — do not relicense without maintainer approval.
- **npm:** Do not publish from unreviewed automation; follow maintainer release process. Bump version intentionally; tag releases to match `package.json`.
- **Dependencies:** Prefer minimal, auditable changes to `package.json`; run tests and CI after lockfile updates.
- **License:** This project ships under **PolyForm Noncommercial 1.0.0** — do not relicense or imply a different license in docs or metadata without maintainer approval.
---
@@ -71,15 +74,15 @@ Format: **Trigger → Instruction → Reason**. Append new Signs when the same m
Stop and ask a **human maintainer** when:
- Impact analysis shows HIGH/CRITICAL risk and the task still requires the change.
- You need to alter CI, release, or security-sensitive config.
- Requirements conflict (e.g. "speed up analyze" vs "must keep all embeddings on huge repo").
- Impact analysis shows **HIGH** / **CRITICAL** risk and the task still requires the change.
- You need to alter **CI**, **release**, or **security-sensitive** config.
- Requirements conflict (e.g. “speed up analyze” vs “must keep all embeddings on huge repo”).
- You are unsure whether data loss is acceptable (`clean`, forced migrations, schema changes).
---
## Related docs
- [ARCHITECTURE.md](ARCHITECTURE.md) — components and data flow
- [RUNBOOK.md](RUNBOOK.md) — commands for recovery
- [CONTRIBUTING.md](CONTRIBUTING.md) — PR and commit expectations
- [ARCHITECTURE.md](ARCHITECTURE.md) — components and data flow.
- [RUNBOOK.md](RUNBOOK.md) — commands for recovery.
- [CONTRIBUTING.md](CONTRIBUTING.md) — PR and commit expectations.
-44
View File
@@ -1,49 +1,5 @@
# Migration Guide
## `impact` tool may now return `{ status: 'ambiguous' }` (PR #888, issue #470)
Before this change the `impact` MCP tool silently picked the first match
when the `target` name hit multiple symbols (Class → Interface → Function
→ Method → Constructor priority UNION). This often produced analysis for
the wrong symbol with no signal back to the caller.
After this change, when the resolver finds more than one viable match
and the caller supplied none of `target_uid` / `file_path` / `kind`,
`impact` returns a disambiguation response shaped like:
```json
{
"status": "ambiguous",
"message": "Found N symbols matching '<target>'. Use target_uid, file_path, or kind to disambiguate.",
"target": { "name": "<target>" },
"direction": "upstream",
"impactedCount": 0,
"risk": "UNKNOWN",
"candidates": [
{ "uid": "...", "name": "...", "kind": "Function", "filePath": "...", "line": 42, "score": 0.76 }
]
}
```
### Do I need to migrate?
**Probably not, but check for assumptions.** Callers that unconditionally
read `result.byDepth` / `result.summary` / `result.affected_processes`
without first checking `result.status` will now see `undefined` in the
ambiguous case. The fix is to branch on `result.status === 'ambiguous'`
first and follow up with `target_uid` (preferred) or `file_path` / `kind`.
The `context` tool's ambiguous response is a strict superset of the
existing shape — every candidate gains a `score` field, no existing field
has changed. No migration required for `context` callers.
### What happens on re-index?
Nothing — this is an MCP-surface change only. The graph schema, indexer,
and stored data are untouched.
---
## OVERRIDES → METHOD_OVERRIDES (PR #642)
The `OVERRIDES` relationship type has been renamed to `METHOD_OVERRIDES` for
-36
View File
@@ -335,42 +335,6 @@ cd ../gitnexus-web && npm install
npm run dev
```
## Docker
```bash
docker run --rm \
--name gitnexus \
-p 4173:4173 \
ghcr.io/abhigyanpatwari/gitnexus:latest
```
Or with Docker Compose:
```bash
docker compose up -d
```
Optional env file:
```bash
cp .env.example .env
set -a
source .env
set +a
```
Docker files:
- [Dockerfile](Dockerfile) is the source for the published `gitnexus` image. It builds `gitnexus-shared` and `gitnexus-web`, then serves the production frontend.
- [docker-compose.yaml](docker-compose.yaml) starts the published image with Docker Compose.
- [.env.example](.env.example) sets the image name, container name, and exposed port for the example commands.
Notes:
- The published image serves the production frontend only. It does not start `gitnexus serve`.
- In backend mode, the app still defaults to `http://localhost:4747` unless you change the server URL in the UI.
- If you do not want an env file, the defaults are `ghcr.io/abhigyanpatwari/gitnexus:latest`, container name `gitnexus`, and port `4173`.
The web UI uses the same indexing pipeline as the CLI but runs entirely in WebAssembly (Tree-sitter WASM, LadybugDB WASM, in-browser embeddings). It's great for quick exploration but limited by browser memory for larger repos.
**Local Backend Mode:** Run `gitnexus serve` and open the web UI locally — it auto-detects the server and shows all your indexed repos, with full AI chat support. No need to re-upload or re-index. The agent's tools (Cypher queries, search, code navigation) route through the backend HTTP API automatically.
+5 -8
View File
@@ -20,9 +20,9 @@ From repository root, unless noted:
cd gitnexus
npm install
npm run build
npm test # full suite: vitest run
npm run test:unit # unit only: vitest run test/unit
npm test # unit: vitest run test/unit
npm run test:integration # integration suite
npm run test:all
npm run test:coverage
npx tsc --noEmit # typecheck (matches CI)
```
@@ -42,11 +42,8 @@ npm run test:e2e # Playwright (requires gitnexus serve + npm run dev)
A husky pre-commit hook (`.husky/pre-commit`) runs automatically on every `git commit`:
1. **Formatting** — `lint-staged` runs prettier on staged files
2. **`gitnexus-web/` files staged** → `tsc -b --noEmit`
3. **`gitnexus/` files staged** → `tsc --noEmit`
Tests do **not** run in the pre-commit hook — they run in CI (`ci-tests.yml`) only.
- **`gitnexus-web/` files staged** → `tsc -b --noEmit` + `vitest run`
- **`gitnexus/` files staged** → `tsc --noEmit` + `vitest run --project default`
Skip with `git commit --no-verify` (use sparingly).
@@ -80,7 +77,7 @@ Re-run the full relevant suite when:
GitHub Actions (`.github/workflows/ci.yml`) orchestrate:
- **`ci-quality.yml`** — prettier format check, eslint lint, `tsc --noEmit` for `gitnexus/`, `tsc -b --noEmit` for `gitnexus-web/`
- **`ci-quality.yml`** — `tsc --noEmit` for `gitnexus/` + `tsc -b --noEmit` for `gitnexus-web/`
- **`ci-tests.yml`** — `vitest run` with coverage (ubuntu) + cross-platform (macOS, Windows)
- **`ci-e2e.yml`** — Playwright E2E tests, gated on `gitnexus-web/**` changes
-13
View File
@@ -1,13 +0,0 @@
services:
gitnexus:
image: ${IMAGE_NAME:-ghcr.io/brainifii/gitnexus:latest}
container_name: ${CONTAINER_NAME:-gitnexus}
ports:
- '${HOST_PORT:-4173}:4173'
restart: unless-stopped
healthcheck:
test: ['CMD', 'curl', '-f', 'http://localhost:4173/']
interval: 30s
timeout: 5s
retries: 3
start_period: 10s
-81
View File
@@ -1,81 +0,0 @@
import { createReadStream } from 'node:fs';
import { stat } from 'node:fs/promises';
import { createServer } from 'node:http';
import { extname, join, normalize, sep } from 'node:path';
const host = '0.0.0.0';
const port = Number(process.env.PORT || '4173');
const root = join(process.cwd(), 'dist');
const contentTypes = {
'.css': 'text/css; charset=utf-8',
'.html': 'text/html; charset=utf-8',
'.js': 'text/javascript; charset=utf-8',
'.json': 'application/json; charset=utf-8',
'.map': 'application/json; charset=utf-8',
'.png': 'image/png',
'.svg': 'image/svg+xml',
'.txt': 'text/plain; charset=utf-8',
'.woff': 'font/woff',
'.woff2': 'font/woff2',
};
function resolvePath(urlPath) {
let decoded;
try {
decoded = decodeURIComponent(urlPath);
} catch {
return null;
}
if (decoded.includes('\0')) return null;
const cleanPath = normalize(decoded.replace(/^\/+/, ''));
const candidate = join(root, cleanPath);
if (candidate !== root && !candidate.startsWith(root + sep)) return null;
return candidate;
}
const server = createServer(async (req, res) => {
const requestPath = req.url?.split('?')[0] || '/';
let filePath = resolvePath(requestPath);
if (!filePath) {
res.writeHead(400);
res.end('Bad request');
return;
}
try {
const fileStat = await stat(filePath).catch(() => null);
if (fileStat?.isDirectory()) {
filePath = join(filePath, 'index.html');
} else if (!fileStat?.isFile()) {
filePath = join(root, 'index.html');
}
const finalStat = await stat(filePath).catch(() => null);
if (!finalStat?.isFile()) {
res.writeHead(404);
res.end('Not found');
return;
}
res.writeHead(200, {
'Cache-Control': filePath.includes('/assets/')
? 'public, max-age=31536000, immutable'
: 'no-cache',
'Content-Type': contentTypes[extname(filePath)] || 'application/octet-stream',
'Cross-Origin-Opener-Policy': 'same-origin',
'Cross-Origin-Embedder-Policy': 'require-corp',
});
const stream = createReadStream(filePath);
stream.on('error', () => res.destroy());
stream.pipe(res);
} catch (error) {
res.writeHead(500);
res.end(error instanceof Error ? error.message : 'Internal server error');
}
});
server.listen(port, host, () => {
console.log(`gitnexus-web listening on http://${host}:${port}`);
});
-107
View File
@@ -1,107 +0,0 @@
import { mkdir, mkdtemp, rm, unlink, writeFile } from 'node:fs/promises';
import http, { createServer } from 'node:http';
import { tmpdir } from 'node:os';
import { dirname, join } from 'node:path';
import { spawn } from 'node:child_process';
import { fileURLToPath } from 'node:url';
import { after, before, it } from 'node:test';
import assert from 'node:assert/strict';
const __dirname = dirname(fileURLToPath(import.meta.url));
const serverScript = join(__dirname, 'docker-server.mjs');
function getFreePort() {
return new Promise((resolve) => {
const s = createServer();
s.listen(0, '127.0.0.1', () => {
const { port } = s.address();
s.close(() => resolve(port));
});
});
}
function rawGet(port, path) {
return new Promise((resolve, reject) => {
const req = http.request({ host: '127.0.0.1', port, path }, (res) => {
let body = '';
res.setEncoding('utf8');
res.on('data', (chunk) => {
body += chunk;
});
res.on('end', () => resolve({ status: res.statusCode, headers: res.headers, body }));
});
req.on('error', reject);
req.end();
});
}
async function waitForServer(port, retries = 30) {
for (let i = 0; i < retries; i++) {
try {
await rawGet(port, '/');
return;
} catch {
await new Promise((r) => setTimeout(r, 100));
}
}
throw new Error('Server did not start in time');
}
let tmpDir, serverPort, child;
before(async () => {
tmpDir = await mkdtemp(join(tmpdir(), 'gitnexus-docker-test-'));
const distDir = join(tmpDir, 'dist');
const assetsDir = join(distDir, 'assets');
await mkdir(assetsDir, { recursive: true });
await writeFile(join(distDir, 'index.html'), '<html><body>spa</body></html>');
await writeFile(join(assetsDir, 'app.abc123.js'), 'console.log("app")');
serverPort = await getFreePort();
child = spawn(process.execPath, [serverScript], {
cwd: tmpDir,
env: { ...process.env, PORT: String(serverPort) },
stdio: 'pipe',
});
child.on('error', (err) => {
throw err;
});
await waitForServer(serverPort);
});
after(async () => {
child?.kill();
if (tmpDir) await rm(tmpDir, { recursive: true, force: true });
});
it('serves a valid asset with immutable cache header', async () => {
const res = await rawGet(serverPort, '/assets/app.abc123.js');
assert.equal(res.status, 200);
assert.match(res.headers['cache-control'], /immutable/);
assert.equal(res.headers['cross-origin-opener-policy'], 'same-origin');
assert.equal(res.headers['cross-origin-embedder-policy'], 'require-corp');
});
it('serves SPA fallback for unknown routes', async () => {
const res = await rawGet(serverPort, '/some/unknown/route');
assert.equal(res.status, 200);
assert.match(res.body, /spa/);
assert.match(res.headers['cache-control'], /no-cache/);
});
it('rejects path traversal with 400', async () => {
const res = await rawGet(serverPort, '/../../../etc/passwd');
assert.equal(res.status, 400);
});
it('rejects percent-encoded null bytes with 400', async () => {
const res = await rawGet(serverPort, '/foo%00bar');
assert.equal(res.status, 400);
});
it('returns 404 when dist/index.html is missing', async () => {
await unlink(join(tmpDir, 'dist', 'index.html'));
const res = await rawGet(serverPort, '/nonexistent-page');
assert.equal(res.status, 404);
});
-57
View File
@@ -23,60 +23,3 @@ export type { MroStrategy } from './mro-strategy.js';
// Pipeline progress
export type { PipelinePhase, PipelineProgress } from './pipeline.js';
// ─── Scope-based resolution — RFC #909 (Ring 1 #910) ────────────────────────
// Data model (RFC §2)
export type { SymbolDefinition } from './scope-resolution/symbol-definition.js';
export type {
ScopeId,
DefId,
ScopeKind,
Range,
Capture,
CaptureMatch,
BindingRef,
ImportEdge,
TypeRef,
Scope,
ResolutionEvidence,
Resolution,
Reference,
ReferenceIndex,
LookupParams,
RegistryContributor,
ParsedImport,
ParsedTypeBinding,
WorkspaceIndex,
ScopeTree,
Callsite,
} from './scope-resolution/types.js';
// Evidence + tie-break constants (RFC Appendix A, Appendix B)
export { EvidenceWeights, typeBindingWeightAtDepth } from './scope-resolution/evidence-weights.js';
export { ORIGIN_PRIORITY } from './scope-resolution/origin-priority.js';
export type { OriginForTieBreak } from './scope-resolution/origin-priority.js';
// Language classification (RFC §6.1 Ring 3/4 governance)
export {
LanguageClassifications,
isProductionLanguage,
} from './scope-resolution/language-classification.js';
export type { LanguageClassification } from './scope-resolution/language-classification.js';
// Core indexes over per-file artifacts (RFC §3.1; Ring 2 SHARED #913)
export { buildDefIndex } from './scope-resolution/def-index.js';
export type { DefIndex } from './scope-resolution/def-index.js';
export { buildModuleScopeIndex } from './scope-resolution/module-scope-index.js';
export type { ModuleScopeIndex, ModuleScopeEntry } from './scope-resolution/module-scope-index.js';
export { buildQualifiedNameIndex } from './scope-resolution/qualified-name-index.js';
export type { QualifiedNameIndex } from './scope-resolution/qualified-name-index.js';
// Shadow-mode diff + aggregation (RFC §6.3; Ring 2 SHARED #918)
export { diffResolutions } from './scope-resolution/shadow/diff.js';
export type {
ShadowAgreement,
ShadowCallsite,
ShadowDiff,
} from './scope-resolution/shadow/diff.js';
export { aggregateDiffs } from './scope-resolution/shadow/aggregate.js';
export type { LanguageParityRow, ShadowParityReport } from './scope-resolution/shadow/aggregate.js';
@@ -30,7 +30,6 @@ export const NODE_TABLES = [
'TypeAlias',
'Const',
'Static',
'Variable',
'Property',
'Record',
'Delegate',
+14 -37
View File
@@ -1,46 +1,23 @@
/**
* MRO (Method Resolution Order) strategy — shared canonical definition.
* MRO (Method Resolution Order) strategy — shared between CLI and any
* future consumer that reasons about multiple-inheritance semantics.
*
* Lives in `gitnexus-shared` so `model/resolve.ts` and `mro-processor.ts` share
* the type without importing the language registry (avoids circular coupling).
* Lives in `gitnexus-shared` so the low-level resolution module
* (`core/ingestion/model/resolve.ts`) does not need to import from
* `languages/` — keeping the `model/` layer free of language-registry
* coupling.
*
* `first-wins` (default, Java/C#/Kotlin/Go/Swift/Dart):
* BFS ancestor walk in declaration order; first match wins.
*
* `leftmost-base` (C++):
* BFS walk; HeritageMap preserves source insertion order, so BFS naturally
* picks the leftmost base in diamond inheritance.
*
* `c3` (Python):
* C3-linearization; falls back to BFS on cyclic/inconsistent hierarchy.
* See model/resolve.ts § c3Linearize.
*
* `implements-split` (Java/C#/Kotlin):
* Low-level lookup is BFS; graph-level mro-processor detects and warns on
* interface-default method ambiguity.
*
* `qualified-syntax` (Rust):
* No auto-resolution — `lookupMethodByOwnerWithMRO` returns undefined immediately.
* Rust requires explicit `<Type as Trait>::method` syntax.
*
* `ruby-mixin` (Ruby):
* Kind-aware walk that does NOT short-circuit on direct owner first (`prepend`
* must beat the class's own method). Walk order:
* 1. Prepend providers (reverse declaration — last-prepended wins)
* 2. Direct owner's own methods
* 3. Include providers (reverse declaration)
* 4. Transitive ancestors (BFS fallback)
* Singleton dispatch: caller passes `ancestryOverride` (extend providers only);
* becomes a simple left-to-right scan. Miss NEVER falls through to file-scoped
* lookup — null-routes or honors `fallback`.
*
* @see model/resolve.ts § lookupMethodByOwnerWithMRO
* @see languages/ruby.ts § selectDispatch
* Strategy semantics:
* - `first-wins`: BFS ancestor walk, first match wins (default).
* - `leftmost-base`: BFS ancestor walk, leftmost base wins (C++).
* - `c3`: C3-linearized ancestor order, first match wins (Python).
* - `implements-split`: BFS walk, first match wins (Java/C#/Kotlin) — full
* interface-default ambiguity is handled at graph level.
* - `qualified-syntax`: No auto-resolution (Rust — requires `<T as Trait>::m`).
*/
export type MroStrategy =
| 'first-wins'
| 'c3'
| 'leftmost-base'
| 'implements-split'
| 'qualified-syntax'
| 'ruby-mixin';
| 'qualified-syntax';
@@ -1,62 +0,0 @@
/**
* `DefIndex` — O(1) `DefId → SymbolDefinition` materialization.
*
* The global "what is this id?" lookup. Every per-kind registry (ClassRegistry,
* MethodRegistry, FieldRegistry) returns `DefId[]` and resolves them back to
* full `SymbolDefinition` records through this index — one central hash map,
* one allocation per def.
*
* Part of RFC #909 Ring 2 SHARED — #913.
*
* Consumed by: #917 (`Registry.lookup` implementations), #915 (SCC finalize).
*/
import type { SymbolDefinition } from './symbol-definition.js';
import type { DefId } from './types.js';
export interface DefIndex {
readonly byId: ReadonlyMap<DefId, SymbolDefinition>;
readonly size: number;
get(id: DefId): SymbolDefinition | undefined;
has(id: DefId): boolean;
}
/**
* Build a `DefIndex` from a flat list of `SymbolDefinition` records.
*
* **Collision policy: first-write-wins.** `DefId` is meant to be unique
* (`nodeId` is the stable graph identifier), so a collision indicates an
* upstream bug — most likely the same symbol parsed twice or a duplicate
* commit into the pipeline. Rather than silently overwriting with a later
* definition that may be partial or wrong, the first record wins and
* subsequent records for the same id are dropped. Pipeline bugs surface
* later as `has(id) === true` but the def looking older than expected,
* which is easier to debug than a silent overwrite.
*
* Pure function — safe to call repeatedly; no side effects.
*/
export function buildDefIndex(defs: readonly SymbolDefinition[]): DefIndex {
const byId = new Map<DefId, SymbolDefinition>();
for (const def of defs) {
if (byId.has(def.nodeId)) continue; // first-write-wins
byId.set(def.nodeId, def);
}
return freezeIndex(byId);
}
// ─── Internal ───────────────────────────────────────────────────────────────
function freezeIndex(byId: Map<DefId, SymbolDefinition>): DefIndex {
return {
byId,
get size() {
return byId.size;
},
get(id: DefId): SymbolDefinition | undefined {
return byId.get(id);
},
has(id: DefId): boolean {
return byId.has(id);
},
};
}
@@ -1,90 +0,0 @@
/**
* `EvidenceWeights` — RFC Appendix A (authoritative values).
*
* Starting calibration for scope-based resolution. Shadow-first rollout
* tunes these against legacy DAG parity. Every `ResolutionEvidence.weight`
* value in the codebase MUST reference this map; inline magic numbers are a
* lint violation. Extends issue #429 (centralize hardcoded confidence values).
*
* Evidence composes additively inside `composeEvidence`; the sum is capped
* at 1.0 in `Resolution.confidence`.
*/
/**
* Authoritative weight map. Keys are a mix of `ResolutionEvidence.kind`
* values and special modifiers (scope-chain depth, MRO depth decay,
* unlinked-import multiplicative cap).
*/
export const EvidenceWeights = {
// ─── Where-found signals (visibility) ─────────────────────────────────────
/** `BindingRef.origin === 'local'` */
local: 0.55,
/** `BindingRef.origin === 'import'` */
import: 0.45,
/** `BindingRef.origin === 'reexport'` */
reexport: 0.4,
/** `BindingRef.origin === 'namespace'` */
namespace: 0.4,
/** `BindingRef.origin === 'wildcard'` */
wildcard: 0.3,
// ─── Scope-chain deduction (per-hop) ──────────────────────────────────────
/** Deducted per parent-hop taken (depth-0 = 0, depth-1 = −0.02, …). */
scopeChainPerDepth: -0.02,
// ─── Receiver-type-binding signal (decays by MRO depth) ───────────────────
/**
* Weight applied when the receiver's type binding resolves to a class that
* declares the candidate as a method/field. Decays by MRO depth: direct
* class = index 0; 1 parent hop = index 1; etc. Falls back to the last
* value for depths beyond the table.
*/
typeBindingByMroDepth: [0.5, 0.42, 0.36, 0.32, 0.3] as const,
// ─── Corroborating signals ────────────────────────────────────────────────
/** `def.ownerId === resolvedReceiver.def.id` (exact owner match). */
ownerMatch: 0.2,
/** Explanatory only — retained for debuggability. Never discriminates
* because surviving candidates already passed `acceptedKinds`. */
kindMatch: 0.0,
// ─── Arity compatibility (from `provider.arityCompatibility`) ─────────────
/** `provider.arityCompatibility(...) === 'compatible'` */
arityMatchCompatible: 0.1,
/** `provider.arityCompatibility(...) === 'unknown'` */
arityMatchUnknown: 0.0,
/** `provider.arityCompatibility(...) === 'incompatible'` — penalizes;
* candidates filtered only when a compatible candidate exists. */
arityMatchIncompatible: -0.15,
// ─── Global fallback (only when nothing lexically visible) ────────────────
/** Hit via `QualifiedNameIndex.byQualifiedName`. */
globalQualified: 0.35,
/** Fallback hit in a `byName` index (and nothing was lexically visible). */
globalName: 0.1,
// ─── Degraded signals ─────────────────────────────────────────────────────
/** Call/reference flowing through a `dynamic-unresolved` edge. */
dynamicImportUnresolved: 0.02,
// ─── Unresolved-import cap (multiplicative, applied per-signal) ───────────
/**
* Multiplicative cap on the edge-derived evidence signal
* (`import`/`wildcard`/`reexport`/`namespace`) when
* `ImportEdge.linkStatus === 'unresolved'`. Independent corroborating
* signals on the same candidate (`owner-match`, `arity-match`,
* `type-binding`) are NOT penalized.
*/
unlinkedImportMultiplier: 0.5,
} as const;
/**
* Look up the `type-binding` signal weight for a given MRO depth, falling
* back to the last tabulated value for depths beyond the table.
*/
export function typeBindingWeightAtDepth(mroDepth: number): number {
const table = EvidenceWeights.typeBindingByMroDepth;
if (mroDepth < 0) return table[0];
if (mroDepth >= table.length) return table[table.length - 1];
return table[mroDepth];
}
@@ -1,49 +0,0 @@
/**
* `LanguageClassification` — RFC §6.1 Ring 3 / Ring 4 governance.
*
* Classifies each `SupportedLanguages` member for the rollout. Ring 4 (DAG
* retirement) is gated on *all production languages* being registry-primary
* and stable for one release cycle; `experimental` and `quarantined`
* languages do not block.
*
* Initial classification (locked in Ring 1 #910):
* - production: javascript, typescript, python, java, c, cpp, csharp, go,
* ruby, rust, php, kotlin, swift, dart
* - experimental: vue (embedded-language / SFC complexity),
* cobol (regex-provider path)
* - quarantined: (none)
*/
import { SupportedLanguages } from '../languages.js';
export type LanguageClassification = 'production' | 'experimental' | 'quarantined';
/**
* The canonical classification for each supported language. Governance
* changes (promote `experimental` → `production`, quarantine a language, …)
* update this map in a dedicated PR.
*/
export const LanguageClassifications: Readonly<Record<SupportedLanguages, LanguageClassification>> =
{
[SupportedLanguages.JavaScript]: 'production',
[SupportedLanguages.TypeScript]: 'production',
[SupportedLanguages.Python]: 'production',
[SupportedLanguages.Java]: 'production',
[SupportedLanguages.C]: 'production',
[SupportedLanguages.CPlusPlus]: 'production',
[SupportedLanguages.CSharp]: 'production',
[SupportedLanguages.Go]: 'production',
[SupportedLanguages.Ruby]: 'production',
[SupportedLanguages.Rust]: 'production',
[SupportedLanguages.PHP]: 'production',
[SupportedLanguages.Kotlin]: 'production',
[SupportedLanguages.Swift]: 'production',
[SupportedLanguages.Dart]: 'production',
[SupportedLanguages.Vue]: 'experimental',
[SupportedLanguages.Cobol]: 'experimental',
};
/** Convenience predicate: is this language gating Ring 4 retirement? */
export function isProductionLanguage(lang: SupportedLanguages): boolean {
return LanguageClassifications[lang] === 'production';
}
@@ -1,65 +0,0 @@
/**
* `ModuleScopeIndex` — O(1) `filePath → moduleScopeId` lookup.
*
* Every file parsed produces exactly one `Module` scope at its root. The
* finalize algorithm needs to resolve `ImportEdge.targetFile` to a concrete
* module scope id in constant time during the link pass; this index is that
* mapping.
*
* Part of RFC #909 Ring 2 SHARED — #913.
*
* Consumed by: #915 (SCC finalize link pass), #923 (shadow harness when
* resolving callsite file → enclosing module).
*/
import type { ScopeId } from './types.js';
export interface ModuleScopeIndex {
readonly byFilePath: ReadonlyMap<string, ScopeId>;
readonly size: number;
get(filePath: string): ScopeId | undefined;
has(filePath: string): boolean;
}
export interface ModuleScopeEntry {
readonly filePath: string;
readonly moduleScopeId: ScopeId;
}
/**
* Build a `ModuleScopeIndex` from a flat list of `{ filePath, moduleScopeId }`
* pairs.
*
* **Collision policy: first-write-wins.** A file should appear exactly once
* in a single ingestion run; collisions indicate the same file was parsed
* twice or a `filePath` normalization bug upstream. Dropping the later
* entry preserves the first-stable id the rest of the pipeline may already
* have registered against.
*
* Pure function — safe to call repeatedly; no side effects.
*/
export function buildModuleScopeIndex(entries: readonly ModuleScopeEntry[]): ModuleScopeIndex {
const byFilePath = new Map<string, ScopeId>();
for (const { filePath, moduleScopeId } of entries) {
if (byFilePath.has(filePath)) continue; // first-write-wins
byFilePath.set(filePath, moduleScopeId);
}
return freezeIndex(byFilePath);
}
// ─── Internal ───────────────────────────────────────────────────────────────
function freezeIndex(byFilePath: Map<string, ScopeId>): ModuleScopeIndex {
return {
byFilePath,
get size() {
return byFilePath.size;
},
get(filePath: string): ScopeId | undefined {
return byFilePath.get(filePath);
},
has(filePath: string): boolean {
return byFilePath.has(filePath);
},
};
}
@@ -1,30 +0,0 @@
/**
* `ORIGIN_PRIORITY` — RFC Appendix B (authoritative values).
*
* Tie-break ordering applied inside `Registry.lookup` Step 7 when
* `|Δconfidence| < 0.001` between two `Resolution` candidates. Lower number
* = stronger (wins the tie).
*
* Full tie-break order (§4.2 Step 7):
* confidence DESC → scope depth ASC → MRO depth ASC → ORIGIN_PRIORITY ASC
* → DefId.localeCompare
*/
export type OriginForTieBreak =
| 'local'
| 'import'
| 'reexport'
| 'namespace'
| 'wildcard'
| 'global-qualified'
| 'global-name';
export const ORIGIN_PRIORITY: Readonly<Record<OriginForTieBreak, number>> = {
local: 0,
import: 1,
reexport: 2,
namespace: 3,
wildcard: 4,
'global-qualified': 5,
'global-name': 6,
};
@@ -1,92 +0,0 @@
/**
* `QualifiedNameIndex` — O(1) `qualifiedName → DefId[]` lookup across all kinds.
*
* Cross-kind fast path for qualified-name resolution
* (`lookupQualified(qname, scope, params)` in RFC §4.5). Class, method,
* field, and namespace defs all contribute to a single index here; consumers
* filter the returned `DefId[]` by `p.acceptedKinds` at the call site.
*
* Returns `DefId[]` (not a single `DefId`) because multiple defs can legally
* share a qualified name — partial classes in C#, method overloads, or
* accidental cross-kind collisions. The lookup caller filters to the expected
* kind(s) and ranks the survivors.
*
* Part of RFC #909 Ring 2 SHARED — #913.
*
* Consumed by: #917 (`Registry.lookup` qualified fast path, `resolveTypeRef`
* dotted fallback via #916).
*/
import type { SymbolDefinition } from './symbol-definition.js';
import type { DefId } from './types.js';
export interface QualifiedNameIndex {
readonly byQualifiedName: ReadonlyMap<string, readonly DefId[]>;
readonly size: number;
/** Returns all `DefId`s registered under this qualified name; empty frozen
* array on miss so callers can iterate without null checks. */
get(qualifiedName: string): readonly DefId[];
has(qualifiedName: string): boolean;
}
/**
* Build a `QualifiedNameIndex` from a flat list of `SymbolDefinition` records.
*
* Only defs with a non-empty `qualifiedName` contribute; defs without one are
* silently skipped (not every kind carries a qualified name — anonymous or
* top-level symbols, dynamic-unresolved imports, etc.).
*
* **Duplicate policy: appended in input order.** Each unique `(qname, DefId)`
* pair contributes at most once — repeated entries for the same pair are
* deduplicated. Distinct `DefId`s sharing a `qname` accumulate in insertion
* order (stable output for deterministic lookup ranking at the call site).
*
* Pure function — safe to call repeatedly; no side effects.
*/
export function buildQualifiedNameIndex(defs: readonly SymbolDefinition[]): QualifiedNameIndex {
const byQualifiedName = new Map<string, DefId[]>();
const seenPairs = new Set<string>();
for (const def of defs) {
const qname = def.qualifiedName;
if (qname === undefined || qname.length === 0) continue;
const pairKey = `${qname}\0${def.nodeId}`;
if (seenPairs.has(pairKey)) continue;
seenPairs.add(pairKey);
const bucket = byQualifiedName.get(qname);
if (bucket === undefined) {
byQualifiedName.set(qname, [def.nodeId]);
} else {
bucket.push(def.nodeId);
}
}
// Freeze bucket arrays so consumers can't mutate the index.
const frozen = new Map<string, readonly DefId[]>();
for (const [k, v] of byQualifiedName) {
frozen.set(k, Object.freeze(v.slice()));
}
return freezeIndex(frozen);
}
// ─── Internal ───────────────────────────────────────────────────────────────
const EMPTY: readonly DefId[] = Object.freeze([]);
function freezeIndex(byQualifiedName: Map<string, readonly DefId[]>): QualifiedNameIndex {
return {
byQualifiedName,
get size() {
return byQualifiedName.size;
},
get(qualifiedName: string): readonly DefId[] {
return byQualifiedName.get(qualifiedName) ?? EMPTY;
},
has(qualifiedName: string): boolean {
return byQualifiedName.has(qualifiedName);
},
};
}
@@ -1,185 +0,0 @@
/**
* Shadow-mode aggregation — per-language parity %, per-evidence-kind
* breakdown of divergences. Consumed by the parity dashboard (RING2-PKG-5).
*
* Pure functions; no I/O. The harness persists per-run JSON; the dashboard
* reads `.gitnexus/shadow-parity/latest.json` and renders.
*
* Part of RFC #909 Ring 2 SHARED — #918.
*/
import type { SupportedLanguages } from '../../languages.js';
import type { ResolutionEvidence } from '../types.js';
import type { ShadowAgreement, ShadowDiff } from './diff.js';
// ─── Aggregated report shape ────────────────────────────────────────────────
export interface LanguageParityRow {
readonly language: SupportedLanguages;
readonly totalCalls: number;
readonly bothAgree: number;
readonly onlyLegacy: number;
readonly onlyNew: number;
readonly bothDisagree: number;
readonly bothEmpty: number;
/**
* Fraction in [0, 1]. Numerator = `bothAgree`; denominator = "calls where
* at least one side resolved" = `totalCalls - bothEmpty`.
*
* When the denominator is 0 (all calls for this language were
* `both-empty`), returns 0. Callers rendering the dashboard should treat
* a 0 parity alongside `totalCalls === bothEmpty` as "no signal" rather
* than "total disagreement".
*/
readonly parity: number;
/**
* Divergence signals broken down by `ResolutionEvidence.kind`. Sourced
* from `ShadowDiff.evidenceDelta` on non-agreeing rows only — `both-agree`
* and `both-empty` do not contribute.
*/
readonly evidenceBreakdown: ReadonlyMap<ResolutionEvidence['kind'], number>;
}
export interface ShadowParityReport {
readonly generatedAt: string; // ISO 8601
readonly perLanguage: readonly LanguageParityRow[];
readonly overall: Omit<LanguageParityRow, 'language' | 'evidenceBreakdown'>;
}
// ─── Public API ─────────────────────────────────────────────────────────────
/**
* Aggregate a stream of `ShadowDiff` records into a `ShadowParityReport`,
* bucketed by language. Pure function.
*
* - `perLanguage` rows are sorted alphabetically by `SupportedLanguages`
* value for stable JSON output (the dashboard reads
* `.gitnexus/shadow-parity/latest.json` and diffing snapshots is useful).
* - `overall` is the column-wise sum across languages.
* - `generatedAt` is injected via the `now` parameter so tests stay
* deterministic; production callers let it default to `new Date()`.
*/
export function aggregateDiffs(
diffs: readonly { readonly language: SupportedLanguages; readonly diff: ShadowDiff }[],
now: Date = new Date(),
): ShadowParityReport {
const perLanguageMap = new Map<SupportedLanguages, MutableCounts>();
for (const { language, diff } of diffs) {
let counts = perLanguageMap.get(language);
if (!counts) {
counts = makeEmptyCounts();
perLanguageMap.set(language, counts);
}
tallyDiff(counts, diff);
}
const perLanguage: LanguageParityRow[] = Array.from(perLanguageMap.entries())
.map(([language, counts]) => buildRow(language, counts))
.sort((a, b) => a.language.localeCompare(b.language));
const overall = buildOverallRow(perLanguage);
return {
generatedAt: now.toISOString(),
perLanguage,
overall,
};
}
// ─── Internal helpers ───────────────────────────────────────────────────────
interface MutableCounts {
totalCalls: number;
bothAgree: number;
onlyLegacy: number;
onlyNew: number;
bothDisagree: number;
bothEmpty: number;
evidenceBreakdown: Map<ResolutionEvidence['kind'], number>;
}
function makeEmptyCounts(): MutableCounts {
return {
totalCalls: 0,
bothAgree: 0,
onlyLegacy: 0,
onlyNew: 0,
bothDisagree: 0,
bothEmpty: 0,
evidenceBreakdown: new Map(),
};
}
function tallyDiff(counts: MutableCounts, diff: ShadowDiff): void {
counts.totalCalls += 1;
incrementAgreement(counts, diff.agreement);
if (diff.agreement === 'both-agree' || diff.agreement === 'both-empty') return;
for (const ev of diff.evidenceDelta) {
counts.evidenceBreakdown.set(ev.kind, (counts.evidenceBreakdown.get(ev.kind) ?? 0) + 1);
}
}
function incrementAgreement(counts: MutableCounts, agreement: ShadowAgreement): void {
switch (agreement) {
case 'both-agree':
counts.bothAgree += 1;
return;
case 'only-legacy':
counts.onlyLegacy += 1;
return;
case 'only-new':
counts.onlyNew += 1;
return;
case 'both-disagree':
counts.bothDisagree += 1;
return;
case 'both-empty':
counts.bothEmpty += 1;
return;
}
}
function buildRow(language: SupportedLanguages, counts: MutableCounts): LanguageParityRow {
const resolved = counts.totalCalls - counts.bothEmpty;
const parity = resolved > 0 ? counts.bothAgree / resolved : 0;
return {
language,
totalCalls: counts.totalCalls,
bothAgree: counts.bothAgree,
onlyLegacy: counts.onlyLegacy,
onlyNew: counts.onlyNew,
bothDisagree: counts.bothDisagree,
bothEmpty: counts.bothEmpty,
parity,
// Freeze via `new Map` on a sorted-kind copy so downstream consumers
// can't mutate the aggregator's internal state.
evidenceBreakdown: new Map(
Array.from(counts.evidenceBreakdown.entries()).sort(([a], [b]) => a.localeCompare(b)),
),
};
}
function buildOverallRow(
perLanguage: readonly LanguageParityRow[],
): Omit<LanguageParityRow, 'language' | 'evidenceBreakdown'> {
let totalCalls = 0;
let bothAgree = 0;
let onlyLegacy = 0;
let onlyNew = 0;
let bothDisagree = 0;
let bothEmpty = 0;
for (const row of perLanguage) {
totalCalls += row.totalCalls;
bothAgree += row.bothAgree;
onlyLegacy += row.onlyLegacy;
onlyNew += row.onlyNew;
bothDisagree += row.bothDisagree;
bothEmpty += row.bothEmpty;
}
const resolved = totalCalls - bothEmpty;
const parity = resolved > 0 ? bothAgree / resolved : 0;
return { totalCalls, bothAgree, onlyLegacy, onlyNew, bothDisagree, bothEmpty, parity };
}
export type { ShadowAgreement, ShadowDiff };
@@ -1,126 +0,0 @@
/**
* Shadow-mode diff logic — RFC §6.3.
*
* Pure comparison logic for shadow mode. Takes two `Resolution[]` (legacy
* DAG result + new scope-based registry result) and produces a structured
* diff record for the parity dashboard.
*
* Consumed by the Ring 2 PKG shadow harness (#923), which dual-runs each
* call through legacy + new paths, diffs results, and persists per-run JSON
* for the parity dashboard.
*
* Part of RFC #909 Ring 2 SHARED — #918.
*/
import type { Resolution, ResolutionEvidence } from '../types.js';
// ─── Diff record shape ──────────────────────────────────────────────────────
export type ShadowAgreement =
| 'both-agree' // top match identical (same DefId)
| 'only-legacy' // legacy resolved; new did not
| 'only-new' // new resolved; legacy did not
| 'both-disagree' // both resolved, but to different targets
| 'both-empty'; // both returned empty
export interface ShadowDiff {
readonly callsite: ShadowCallsite;
readonly legacy: Resolution | null;
readonly newResult: Resolution | null;
readonly agreement: ShadowAgreement;
/**
* Symmetric difference of the two top resolutions' `evidence` arrays,
* keyed on `ResolutionEvidence.kind`.
*
* - For `'both-agree'` and `'both-empty'` agreements, always empty.
* - For `'both-disagree'`, contains evidence kinds present on exactly one
* side (not in both).
* - For `'only-legacy'`, contains all of legacy's top evidence.
* - For `'only-new'`, contains all of new's top evidence.
*/
readonly evidenceDelta: readonly ResolutionEvidence[];
}
export interface ShadowCallsite {
readonly filePath: string;
readonly line: number;
readonly col: number;
readonly calledName: string;
}
// ─── Public API ─────────────────────────────────────────────────────────────
/**
* Compare two `Resolution[]` arrays (top matches at `[0]`) and produce a
* `ShadowDiff`. Pure function.
*
* Agreement rules:
* - both arrays empty → `'both-empty'`, `evidenceDelta: []`
* - legacy empty, new non-empty → `'only-new'`, `evidenceDelta` = new's top evidence
* - legacy non-empty, new empty → `'only-legacy'`, `evidenceDelta` = legacy's top evidence
* - both non-empty, same top `def.nodeId` → `'both-agree'`, `evidenceDelta: []`
* - both non-empty, different top `def.nodeId` → `'both-disagree'`,
* `evidenceDelta` = symmetric difference by `ResolutionEvidence.kind`
* (first occurrence of a kind-only-on-legacy then kind-only-on-new; order
* preserved from input arrays)
*
* Evidence-delta rationale: callers aggregating divergences want to know
* which signal kinds explain a disagreement. Keying on `kind` (not full
* equality over `weight`/`note`) avoids spurious deltas when the same
* signal fires with slightly different calibration weights on each side.
*/
export function diffResolutions(
callsite: ShadowCallsite,
legacy: readonly Resolution[],
newResult: readonly Resolution[],
): ShadowDiff {
const legacyTop: Resolution | null = legacy.length > 0 ? legacy[0] : null;
const newTop: Resolution | null = newResult.length > 0 ? newResult[0] : null;
const agreement: ShadowAgreement = (() => {
if (legacyTop === null && newTop === null) return 'both-empty';
if (legacyTop === null) return 'only-new';
if (newTop === null) return 'only-legacy';
return legacyTop.def.nodeId === newTop.def.nodeId ? 'both-agree' : 'both-disagree';
})();
const evidenceDelta = computeEvidenceDelta(legacyTop, newTop, agreement);
return {
callsite,
legacy: legacyTop,
newResult: newTop,
agreement,
evidenceDelta,
};
}
// ─── Internal helpers ───────────────────────────────────────────────────────
/**
* Symmetric difference of two evidence arrays, keyed on
* `ResolutionEvidence.kind`. Preserves input order: legacy-only signals
* first (in legacy's original order), then new-only signals (in new's order).
*
* For `'both-agree'` / `'both-empty'` the delta is empty by contract. For
* `'only-legacy'` / `'only-new'` one side's evidence is the delta (nothing to
* subtract against).
*/
function computeEvidenceDelta(
legacy: Resolution | null,
newResult: Resolution | null,
agreement: ShadowAgreement,
): readonly ResolutionEvidence[] {
if (agreement === 'both-agree' || agreement === 'both-empty') return [];
if (agreement === 'only-legacy') return legacy!.evidence;
if (agreement === 'only-new') return newResult!.evidence;
// both-disagree: symmetric difference keyed on `kind`
const legacyKinds = new Set(legacy!.evidence.map((e) => e.kind));
const newKinds = new Set(newResult!.evidence.map((e) => e.kind));
const onlyInLegacy = legacy!.evidence.filter((e) => !newKinds.has(e.kind));
const onlyInNew = newResult!.evidence.filter((e) => !legacyKinds.has(e.kind));
return [...onlyInLegacy, ...onlyInNew];
}
@@ -1,35 +0,0 @@
/**
* `SymbolDefinition` — the canonical shape of an indexed symbol record.
*
* Historically defined in `gitnexus/src/core/ingestion/model/symbol-table.ts`;
* moved into `gitnexus-shared` as part of RFC #909 Ring 1 (#910) so the
* scope-resolution types that reference it can live in the shared package
* alongside their consumers (`gitnexus/` and `gitnexus-web/`).
*
* Shape is unchanged from the prior local definition.
*/
import type { NodeLabel } from '../graph/types.js';
export interface SymbolDefinition {
nodeId: string;
filePath: string;
type: NodeLabel;
/** Canonical dot-separated qualified type name for class-like symbols
* (e.g. `App.Models.User`). Falls back to the simple symbol name when no
* package/namespace/module scope exists or no explicit qualified metadata is provided. */
qualifiedName?: string;
parameterCount?: number;
/** Number of required (non-optional, non-default) parameters.
* Enables range-based arity filtering: argCount >= requiredParameterCount && argCount <= parameterCount. */
requiredParameterCount?: number;
/** Per-parameter type names for overload disambiguation (e.g. ['int', 'String']).
* Populated when parameter types are resolvable from AST (any typed language). */
parameterTypes?: string[];
/** Raw return type text extracted from AST (e.g. 'User', 'Promise<User>') */
returnType?: string;
/** Declared type for non-callable symbols — fields/properties (e.g. 'Address', 'List<User>') */
declaredType?: string;
/** Links Method/Constructor/Property to owning Class/Struct/Trait nodeId */
ownerId?: string;
}
@@ -1,408 +0,0 @@
/**
* Scope-resolution type definitions — RFC §2 data model (authoritative source).
*
* See: https://www.notion.so/346dc50b6ed281cfaacbe480bf231d50
*
* Anti-drift rule: every type, interface, and enum defined here is the single
* source of truth. Later code that references these names must import them
* from `gitnexus-shared`; it must not re-define them locally.
*
* Lifecycle contract (RFC §2.8): scopes are **constructed during extraction,
* linked during finalize, immutable after finalize**. All fields are
* `readonly` at the type level; `Object.freeze` is applied at runtime in dev
* builds. `ReferenceIndex` is the sole structure populated after freeze — by
* resolution, before emission.
*/
import type { NodeLabel } from '../graph/types.js';
import type { SymbolDefinition } from './symbol-definition.js';
// ─── §2.1 Type aliases ──────────────────────────────────────────────────────
/** Stable per-(file, range, kind) scope identifier; interned for identity-fast equality. */
export type ScopeId = string;
/** Stable symbol-definition identifier (graph nodeId). */
export type DefId = string;
/** Kinds of lexical scope a `Scope` node can represent. */
export type ScopeKind =
| 'Module' // file root
| 'Namespace' // C++ namespace, C# namespace, Kotlin package-object, Rust mod
| 'Class' // class/struct/trait/interface body
| 'Function' // function/method/closure/lambda body
| 'Block' // { ... }, if-body, for-body, with-body, match arms
| 'Expression'; // comprehensions, for-init, pattern bindings, lambda param lists
// ─── Range + Capture (parser-agnostic) ──────────────────────────────────────
/** Source-text range. 1-based `startLine`/`endLine`; 0-based `startCol`/`endCol`. */
export interface Range {
readonly startLine: number;
readonly startCol: number;
readonly endLine: number;
readonly endCol: number;
}
/**
* Tagged capture emitted by a LanguageProvider's `emitScopeCaptures` hook.
*
* Parser-agnostic: tree-sitter queries and COBOL's regex tagger both produce
* `Capture[]`. The central `ScopeExtractor` consumes captures without
* knowing which parser produced them.
*/
export interface Capture {
/** Capture name, including leading `@` (e.g., `'@scope.module'`, `'@declaration.class'`). */
readonly name: string;
readonly range: Range;
/** The captured source text. */
readonly text: string;
}
/**
* A grouping of `Capture`s that came from a single query match (e.g., one
* `@import.statement` match carries `@import.source`, `@import.name`,
* `@import.alias?` as child captures). Keyed by capture name for O(1)
* child access.
*/
export type CaptureMatch = Readonly<Record<string, Capture>>;
// ─── Hook input/output types (RFC §5.2) ─────────────────────────────────────
/**
* Provider-interpreted raw import, consumed by finalize (Phase 2) to produce
* linked `ImportEdge[]`. The provider's `interpretImport` hook turns a
* `CaptureMatch` for an `@import.statement` into one of these; the central
* finalize algorithm resolves `targetRaw` to a concrete file via
* `resolveImportTarget` and materializes the final `ImportEdge`.
*
* Discriminated union — each variant carries only the fields that make sense
* for its kind. Invalid shapes (e.g., a `namespace` import with an alias-like
* `importedName` mismatch) are compile errors, not latent bugs. `'wildcard-
* expanded'` is deliberately NOT a variant: that kind is finalize output only,
* produced when `expandsWildcardTo` materializes a wildcard against target
* exports — a provider must never emit it at parse time.
*/
export type ParsedImport =
/**
* Per-name import without rename.
*
* Examples:
* - Python `from foo import X` → `{ kind: 'named', localName: 'X', importedName: 'X', targetRaw: 'foo' }`
* - TS `import { X } from './foo'` → `{ kind: 'named', localName: 'X', importedName: 'X', targetRaw: './foo' }`
* - Java `import foo.bar.X` → `{ kind: 'named', localName: 'X', importedName: 'X', targetRaw: 'foo.bar' }`
*/
| {
readonly kind: 'named';
readonly localName: string;
readonly importedName: string;
readonly targetRaw: string;
}
/**
* Per-name import with rename.
*
* Examples:
* - Python `from foo import X as Y` → `{ kind: 'alias', localName: 'Y', importedName: 'X', alias: 'Y', targetRaw: 'foo' }`
* - TS `import { X as Y } from './foo'` → `{ kind: 'alias', localName: 'Y', importedName: 'X', alias: 'Y', targetRaw: './foo' }`
*/
| {
readonly kind: 'alias';
readonly localName: string;
readonly importedName: string;
readonly alias: string;
readonly targetRaw: string;
}
/**
* Qualified module handle, with or without rename. `importedName` is the
* module being aliased; `localName` is the scope-visible handle (often the
* same unless renamed).
*
* Examples:
* - Python `import numpy` → `{ kind: 'namespace', localName: 'numpy', importedName: 'numpy', targetRaw: 'numpy' }`
* - Python `import numpy as np` → `{ kind: 'namespace', localName: 'np', importedName: 'numpy', targetRaw: 'numpy' }`
* - TS `import * as np from 'numpy'` → `{ kind: 'namespace', localName: 'np', importedName: 'numpy', targetRaw: 'numpy' }`
* - Go `import foo "pkg/bar"` → `{ kind: 'namespace', localName: 'foo', importedName: 'bar', targetRaw: 'pkg/bar' }`
*/
| {
readonly kind: 'namespace';
/** Scope-visible handle (e.g. `np` in `import numpy as np`; `numpy` when unaliased). */
readonly localName: string;
/** Module being aliased (e.g. `numpy` in `import numpy as np`). */
readonly importedName: string;
readonly targetRaw: string;
}
/**
* Syntactically-detectable parse-time re-export. Finalize may still produce
* `ImportEdge { kind: 'reexport', transitiveVia }` when flattening chains;
* this variant preserves the *parse-time* signal so finalize doesn't have
* to re-derive it from scratch.
*
* Examples:
* - TS `export { X } from './y'` → `{ kind: 'reexport', localName: 'X', importedName: 'X', targetRaw: './y' }`
* - TS `export { X as Y } from './y'` → `{ kind: 'reexport', localName: 'Y', importedName: 'X', alias: 'Y', targetRaw: './y' }`
* - Rust `pub use foo::bar` → `{ kind: 'reexport', localName: 'bar', importedName: 'bar', targetRaw: 'foo' }`
*/
| {
readonly kind: 'reexport';
/** Name as re-exported in the current module. */
readonly localName: string;
/** Name in the source module. */
readonly importedName: string;
readonly targetRaw: string;
/** Set when the re-export renames the symbol (e.g. `export { X as Y } from './y'`). */
readonly alias?: string;
}
/**
* Runtime-computed target — the import path is not a static literal at
* parse time. Providers SHOULD emit the unresolvable expression's source
* text as `targetRaw` to aid diagnostics; `null` only when no string form
* exists.
*
* Examples:
* - JS `await import(expr)` → `{ kind: 'dynamic-unresolved', localName: '', targetRaw: 'expr' }`
* - Python `importlib.import_module(f'pkg.{name}')` → `{ kind: 'dynamic-unresolved', localName: '', targetRaw: "f'pkg.{name}'" }`
*/
| {
readonly kind: 'dynamic-unresolved';
readonly localName: string;
/** Source text of the unresolved expression when available; `null` otherwise. */
readonly targetRaw: string | null;
};
/**
* Provider-interpreted type binding. The provider's `interpretTypeBinding`
* hook turns a `CaptureMatch` (e.g., `@type-binding.parameter`) into one of
* these; the central extractor attaches the resulting `TypeRef` to the
* appropriate scope's `typeBindings` map.
*/
export interface ParsedTypeBinding {
/** The name being bound (parameter name, `self`, assignment LHS, …). */
readonly boundName: string;
/** The raw type name as written in source (`'User'`, `'models.User'`, …). */
readonly rawTypeName: string;
readonly source: TypeRef['source'];
}
/**
* Cross-file workspace index consumed by finalize-phase hooks
* (`resolveImportTarget`, `expandsWildcardTo`). Opaque placeholder in Ring 1;
* concretely typed in Ring 2 SHARED (#915).
*/
export type WorkspaceIndex = unknown;
/**
* Scope tree handle consumed by parse-phase hooks (`bindingScopeFor`,
* `importOwningScope`) to navigate the in-progress scope tree. Opaque
* placeholder in Ring 1; concretely typed in Ring 2 SHARED (#912).
*/
export type ScopeTree = unknown;
/** Call-site description passed to `arityCompatibility`. */
export interface Callsite {
/** Number of arguments at the call site. */
readonly arity: number;
}
// ─── §2.4 ImportEdge ────────────────────────────────────────────────────────
/**
* A cross-file import edge attached to a module/namespace scope.
*
* Raw (unlinked) edges are emitted during parse (Phase 1); `targetModuleScope`
* and `targetDefId` are filled in during finalize (Phase 2) via SCC-aware
* bounded-fixpoint linking (RFC §3.2).
*/
export interface ImportEdge {
/** How this scope sees the imported name (after alias). */
readonly localName: string;
/** Exporting file; `null` only when `kind === 'dynamic-unresolved'`. */
readonly targetFile: string | null;
/** The name under which the target exports this symbol. */
readonly targetExportedName: string;
/** Pre-resolved at finalize: the module scope of the exporting file. */
readonly targetModuleScope?: ScopeId;
/** Pre-resolved at finalize: the exported symbol's `DefId`. */
readonly targetDefId?: DefId;
readonly kind:
| 'named'
| 'alias'
| 'namespace'
| 'wildcard-expanded'
| 'reexport'
| 'dynamic-unresolved';
/** Re-export chain, for provenance (e.g., `['./y']` when re-exported via `./y`). */
readonly transitiveVia?: readonly string[];
/** Set to `'unresolved'` when the SCC fixpoint could not link this edge. */
readonly linkStatus?: 'unresolved';
}
// ─── §2.3 BindingRef ────────────────────────────────────────────────────────
/**
* A name binding visible at a scope, with provenance.
*
* Provenance stays at the visibility layer — a name being visible because it
* is local vs imported vs wildcard-expanded vs re-exported is a property of
* the binding itself. This keeps evidence emission and `import-use` reference
* stamping first-class instead of reconstructing provenance from a side table.
*/
export interface BindingRef {
readonly def: SymbolDefinition;
readonly origin: 'local' | 'import' | 'namespace' | 'wildcard' | 'reexport';
/** Non-null for non-local origins; carries the `ImportEdge` that brought the name into this scope. */
readonly via?: ImportEdge;
}
// ─── §2.5 TypeRef ───────────────────────────────────────────────────────────
/**
* A reference to a named type, anchored at its declaration site.
*
* Design choice: raw name + declaration-site scope, resolved at lookup time.
* Pre-resolution would invert the extraction/resolution wall. Deferred thunks
* add no capability. Structured type systems are months of work per language.
* This shape keeps V1 tractable while preserving correctness for aliases,
* re-exports, and nested modules. Generics deferred to V2 via `typeArgs`.
*/
export interface TypeRef {
/** The name as written in source (e.g., `'User'`, `'models.User'`, `'List'`). */
readonly rawName: string;
/** Anchor for resolving `rawName` — the scope where the annotation/inference was written. */
readonly declaredAtScope: ScopeId;
readonly source:
| 'annotation'
| 'parameter-annotation'
| 'return-annotation'
| 'self'
| 'assignment-inferred'
| 'constructor-inferred'
| 'receiver-propagated';
/** Reserved for V2+: generic type arguments (`List<User>` → `[TypeRef('User')]`). V1 ignores. */
readonly typeArgs?: readonly TypeRef[];
}
// ─── §2.2 Scope ─────────────────────────────────────────────────────────────
/**
* The canonical lexical-scope node. Forms the spine of the SemanticModel.
*
* ScopeId shape (RFC §2.2): `scope:{filePath}#{startLine}:{startCol}-{endLine}:{endCol}:{kind}`
* — deterministic, stable across reparses of the same source, interned.
*/
export interface Scope {
readonly id: ScopeId;
readonly parent: ScopeId | null;
readonly kind: ScopeKind;
readonly range: Range;
readonly filePath: string;
/** Names visible from this scope. Provenance preserved via `BindingRef.origin`. */
readonly bindings: ReadonlyMap<string, readonly BindingRef[]>;
/** Defs structurally owned by this scope (e.g., methods owned by a class body scope). */
readonly ownedDefs: readonly SymbolDefinition[];
/** Import edges attached to this scope. Mostly module/namespace scopes, but some
* languages allow local imports (Python `def f(): from x import Y`, Rust
* fn-local `use`, TS dynamic `import()`). */
readonly imports: readonly ImportEdge[];
/** Local type facts visible from this scope (parameter annotations, `self` binding, etc.). */
readonly typeBindings: ReadonlyMap<string, TypeRef>;
}
// ─── §2.6 Resolution + ResolutionEvidence ───────────────────────────────────
/**
* One piece of evidence for a `Resolution`. Multiple signals corroborate a
* single match; their weights compose additively to produce `confidence`.
*
* Weights come from `EvidenceWeights` (see `./evidence-weights.ts`).
*/
export interface ResolutionEvidence {
readonly kind:
| 'local'
| 'scope-chain'
| 'import'
| 'type-binding'
| 'owner-match'
| 'kind-match'
| 'arity-match'
| 'global-name'
| 'global-qualified'
| 'dynamic-import-unresolved';
/** Signal weight, sourced from `EvidenceWeights`. Additive; sum capped at 1.0. */
readonly weight: number;
/** Optional debug annotation (e.g., `'matched via self: User'`). */
readonly note?: string;
}
/**
* A ranked resolution candidate returned by `ClassRegistry.lookup` /
* `MethodRegistry.lookup` / `FieldRegistry.lookup`. Evidence composes
* additively; callers read `[0]` for the one-shot answer or inspect the
* evidence trace for debugging.
*/
export interface Resolution {
readonly def: SymbolDefinition;
/** Σ of `evidence[].weight`, capped at 1.0. */
readonly confidence: number;
readonly evidence: readonly ResolutionEvidence[];
/** Optional debug trace: scopes walked to reach `def`. */
readonly path?: readonly ScopeId[];
}
// ─── §2.7 Reference + ReferenceIndex ────────────────────────────────────────
/**
* A post-resolution usage fact: some code at `atRange` inside `fromScope`
* references `toDef` with the given confidence/evidence. Materialized by the
* resolution phase; emitted as graph edges (`CALLS`/`READS`/`WRITES`/etc.)
* during the emit phase.
*/
export interface Reference {
/** Innermost lexical scope containing `atRange`. */
readonly fromScope: ScopeId;
readonly toDef: DefId;
/** Location of the reference in source. */
readonly atRange: Range;
readonly kind: 'call' | 'read' | 'write' | 'type-reference' | 'inherits' | 'import-use';
readonly confidence: number;
readonly evidence: readonly ResolutionEvidence[];
}
/**
* Two-way index over `Reference` records, populated during the resolution
* phase. Scopes stay immutable after finalize; references accumulate here.
*/
export interface ReferenceIndex {
readonly bySourceScope: ReadonlyMap<ScopeId, readonly Reference[]>;
readonly byTargetDef: ReadonlyMap<DefId, readonly Reference[]>;
}
// ─── §4.1 LookupParams ──────────────────────────────────────────────────────
/**
* Opaque placeholder for the per-kind registry passed as the owner-scoped
* contributor. Typed concretely in Ring 2 SHARED (#917); kept as `unknown`
* here so Ring 1 can ship without pulling in the registry implementation.
*/
export type RegistryContributor = unknown;
/**
* Parameters accepted by `Registry.lookup`. Three registries (Class/Method/
* Field) run the same 7-step algorithm with different parameter tuples; see
* RFC §4.4 for per-registry specializations.
*/
export interface LookupParams {
readonly acceptedKinds: readonly NodeLabel[];
/** Class lookups: false. Method/Field lookups: true. */
readonly useReceiverTypeBinding: boolean;
readonly ownerScopedContributor: RegistryContributor | null;
/** Optional arity hint fed to `provider.arityCompatibility`. */
readonly arityHint?: number;
/** Explicit receiver name (e.g., `'user'` in `user.save()`). When present,
* the receiver's type binding at the callsite scope is used; otherwise
* the enclosing method's implicit `self`/`this` is consulted. See §4.1. */
readonly explicitReceiver?: { readonly name: string };
}
+4 -23
View File
@@ -61,22 +61,13 @@ test.beforeAll(async () => {
}
});
// Auto-connect downloads the full graph from the backend; under parallel
// workers in CI the same backend serves multiple downloads concurrently, so
// reaching the "Ready" state can take noticeably longer than a single-worker
// run. Match the 45s budget used by waitForGraphLoaded() in
// server-connect.spec.ts which has been stable on the same backend.
const READY_TIMEOUT_MS = 45_000;
test.describe('Multi-Repo Scoping', () => {
test('auto-connect via ?server= sets ?project= in URL', async ({ page }) => {
// Navigate with ?server= param (the bookmarkable shortcut)
await page.goto(`/?server=${encodeURIComponent(BACKEND_URL)}`);
// Wait for graph to load
await expect(page.locator('[data-testid="status-ready"]')).toBeVisible({
timeout: READY_TIMEOUT_MS,
});
await expect(page.locator('[data-testid="status-ready"]')).toBeVisible({ timeout: 30_000 });
// URL should now contain ?project= with the repo name
const url = new URL(page.url());
@@ -86,14 +77,8 @@ test.describe('Multi-Repo Scoping', () => {
});
test('?server= is preserved in URL for F5 recovery', async ({ page }) => {
// Two sequential auto-connects (initial + reload), each up to READY_TIMEOUT_MS,
// can exceed the default 60s test timeout under parallel workers.
test.slow();
await page.goto(`/?server=${encodeURIComponent(BACKEND_URL)}`);
await expect(page.locator('[data-testid="status-ready"]')).toBeVisible({
timeout: READY_TIMEOUT_MS,
});
await expect(page.locator('[data-testid="status-ready"]')).toBeVisible({ timeout: 30_000 });
// URL should still have ?server=
const url = new URL(page.url());
@@ -101,16 +86,12 @@ test.describe('Multi-Repo Scoping', () => {
// F5 should reconnect (not show onboarding)
await page.reload();
await expect(page.locator('[data-testid="status-ready"]')).toBeVisible({
timeout: READY_TIMEOUT_MS,
});
await expect(page.locator('[data-testid="status-ready"]')).toBeVisible({ timeout: 30_000 });
});
test('node count in status bar matches backend data', async ({ page }) => {
await page.goto(`/?server=${encodeURIComponent(BACKEND_URL)}`);
await expect(page.locator('[data-testid="status-ready"]')).toBeVisible({
timeout: READY_TIMEOUT_MS,
});
await expect(page.locator('[data-testid="status-ready"]')).toBeVisible({ timeout: 30_000 });
// Fetch expected node count from backend
const res = await fetch(`${BACKEND_URL}/api/repo?repo=${encodeURIComponent(firstRepoName)}`);
+1 -8
View File
@@ -26,10 +26,7 @@ async function enterExploringView(page: import('@playwright/test').Page) {
// Landing screen may not appear (e.g. ?server auto-connect)
}
// Match the 45s budget used by waitForGraphLoaded() in
// server-connect.spec.ts; under parallel CI workers, downloading the full
// graph can occasionally exceed 30s.
await expect(page.locator('[data-testid="status-ready"]')).toBeVisible({ timeout: 45_000 });
await expect(page.locator('[data-testid="status-ready"]')).toBeVisible({ timeout: 30_000 });
}
// ── Flow 1: Onboarding (no server running) ─────────────────────────────────
@@ -247,10 +244,6 @@ test.describe('Flow 3: Analyze form', () => {
test.describe('Flow 4: Repo dropdown in exploring view', () => {
const SKIP_MSG = 'Requires running gitnexus server with indexed repos';
// enterExploringView() can take up to ~45s under parallel CI workers; combined
// with the dropdown interactions this can exceed the default 60s test budget.
test.slow();
test.beforeAll(async () => {
if (process.env.E2E) return;
try {
+5 -26
View File
@@ -84,20 +84,11 @@ test.describe('Hold-queue timeout error', () => {
// ── 2. ?project= URL persistence ─────────────────────────────────────────────
// Auto-connect downloads the full graph from the backend; under parallel
// workers in CI the same backend serves multiple downloads concurrently, so
// reaching the "Ready" state can take noticeably longer than a single-worker
// run. Match the 45s budget used by waitForGraphLoaded() in
// server-connect.spec.ts which has been stable on the same backend.
const READY_TIMEOUT_MS = 45_000;
test.describe('?project= URL persistence', () => {
test('?project= is set in URL after connecting via ?server=', async ({ page }) => {
await page.goto(`/?server=${encodeURIComponent(BACKEND_URL)}`);
await expect(page.locator('[data-testid="status-ready"]')).toBeVisible({
timeout: READY_TIMEOUT_MS,
});
await expect(page.locator('[data-testid="status-ready"]')).toBeVisible({ timeout: 30_000 });
const url = new URL(page.url());
const project = url.searchParams.get('project');
@@ -107,20 +98,12 @@ test.describe('?project= URL persistence', () => {
});
test('?project= is still present after F5 reload', async ({ page }) => {
// Two sequential auto-connects (initial + reload), each up to READY_TIMEOUT_MS,
// can exceed the default 60s test timeout under parallel workers.
test.slow();
await page.goto(`/?server=${encodeURIComponent(BACKEND_URL)}`);
await expect(page.locator('[data-testid="status-ready"]')).toBeVisible({
timeout: READY_TIMEOUT_MS,
});
await expect(page.locator('[data-testid="status-ready"]')).toBeVisible({ timeout: 30_000 });
// After connect, URL has ?server=&project= — F5 re-uses both params
await page.reload();
await expect(page.locator('[data-testid="status-ready"]')).toBeVisible({
timeout: READY_TIMEOUT_MS,
});
await expect(page.locator('[data-testid="status-ready"]')).toBeVisible({ timeout: 30_000 });
const url = new URL(page.url());
expect(url.searchParams.get('project')).toBeTruthy();
@@ -139,9 +122,7 @@ test.describe('?project= auto-connect', () => {
`/?server=${encodeURIComponent(BACKEND_URL)}&project=${encodeURIComponent(firstRepoName)}`,
);
await expect(page.locator('[data-testid="status-ready"]')).toBeVisible({
timeout: READY_TIMEOUT_MS,
});
await expect(page.locator('[data-testid="status-ready"]')).toBeVisible({ timeout: 30_000 });
// ?project= in URL should match what we passed in
const url = new URL(page.url());
@@ -174,9 +155,7 @@ test.describe('Windows path normalization', () => {
await page.goto(`/?server=${encodeURIComponent(BACKEND_URL)}`);
await expect(page.locator('[data-testid="status-ready"]')).toBeVisible({
timeout: READY_TIMEOUT_MS,
});
await expect(page.locator('[data-testid="status-ready"]')).toBeVisible({ timeout: 30_000 });
// URL ?project= must be the short basename, NOT the full Windows path
const url = new URL(page.url());
+393 -162
View File
@@ -29,7 +29,7 @@
"langchain": "^1.2.10",
"lru-cache": "^11.2.4",
"lucide-react": "^0.562.0",
"mermaid": "^11.14.0",
"mermaid": "^11.12.2",
"mnemonist": "^0.39.0",
"pandemonium": "^2.4.0",
"react": "^18.3.1",
@@ -39,7 +39,7 @@
"react-zoom-pan-pinch": "^3.7.0",
"remark-gfm": "^4.0.1",
"sigma": "^3.0.2",
"tailwindcss": "^4.2.2",
"tailwindcss": "^4.1.18",
"uuid": "^13.0.0",
"zod": "^3.25.76"
},
@@ -57,12 +57,12 @@
"@vercel/node": "^5.5.16",
"@vitejs/plugin-react": "^5.1.0",
"@vitest/coverage-v8": "^3.2.4",
"jsdom": "^29.0.2",
"jsdom": "^29.0.0",
"tree-sitter-wasms": "^0.1.13",
"typescript": "^5.4.5",
"vite": "^5.2.0",
"vitest": "^3.2.4",
"wait-on": "^9.0.5"
"wait-on": "^8.0.5"
},
"engines": {
"node": ">=20.0.0"
@@ -129,49 +129,39 @@
}
},
"node_modules/@asamuzakjp/css-color": {
"version": "5.1.11",
"resolved": "https://registry.npmjs.org/@asamuzakjp/css-color/-/css-color-5.1.11.tgz",
"integrity": "sha512-KVw6qIiCTUQhByfTd78h2yD1/00waTmm9uy/R7Ck/ctUyAPj+AEDLkQIdJW0T8+qGgj3j5bpNKK7Q3G+LedJWg==",
"version": "5.0.1",
"resolved": "https://registry.npmjs.org/@asamuzakjp/css-color/-/css-color-5.0.1.tgz",
"integrity": "sha512-2SZFvqMyvboVV1d15lMf7XiI3m7SDqXUuKaTymJYLN6dSGadqp+fVojqJlVoMlbZnlTmu3S0TLwLTJpvBMO1Aw==",
"dev": true,
"license": "MIT",
"dependencies": {
"@asamuzakjp/generational-cache": "^1.0.1",
"@csstools/css-calc": "^3.2.0",
"@csstools/css-color-parser": "^4.1.0",
"@csstools/css-calc": "^3.1.1",
"@csstools/css-color-parser": "^4.0.2",
"@csstools/css-parser-algorithms": "^4.0.0",
"@csstools/css-tokenizer": "^4.0.0"
"@csstools/css-tokenizer": "^4.0.0",
"lru-cache": "^11.2.6"
},
"engines": {
"node": "^20.19.0 || ^22.12.0 || >=24.0.0"
}
},
"node_modules/@asamuzakjp/dom-selector": {
"version": "7.0.10",
"resolved": "https://registry.npmjs.org/@asamuzakjp/dom-selector/-/dom-selector-7.0.10.tgz",
"integrity": "sha512-KyOb19eytNSELkmdqzZZUXWCU25byIlOld5qVFg0RYdS0T3tt7jeDByxk9hIAC73frclD8GKrHttr0SUjKCCdQ==",
"version": "7.0.3",
"resolved": "https://registry.npmjs.org/@asamuzakjp/dom-selector/-/dom-selector-7.0.3.tgz",
"integrity": "sha512-Q6mU0Z6bfj6YvnX2k9n0JxiIwrCFN59x/nWmYQnAqP000ruX/yV+5bp/GRcF5T8ncvfwJQ7fgfP74DlpKExILA==",
"dev": true,
"license": "MIT",
"dependencies": {
"@asamuzakjp/generational-cache": "^1.0.1",
"@asamuzakjp/nwsapi": "^2.3.9",
"bidi-js": "^1.0.3",
"css-tree": "^3.2.1",
"is-potential-custom-element-name": "^1.0.1"
"is-potential-custom-element-name": "^1.0.1",
"lru-cache": "^11.2.7"
},
"engines": {
"node": "^20.19.0 || ^22.12.0 || >=24.0.0"
}
},
"node_modules/@asamuzakjp/generational-cache": {
"version": "1.0.1",
"resolved": "https://registry.npmjs.org/@asamuzakjp/generational-cache/-/generational-cache-1.0.1.tgz",
"integrity": "sha512-wajfB8KqzMCN2KGNFdLkReeHncd0AslUSrvHVvvYWuU8ghncRJoA50kT3zP9MVL0+9g4/67H+cdvBskj9THPzg==",
"dev": true,
"license": "MIT",
"engines": {
"node": "^20.19.0 || ^22.12.0 || >=24.0.0"
}
},
"node_modules/@asamuzakjp/nwsapi": {
"version": "2.3.9",
"resolved": "https://registry.npmjs.org/@asamuzakjp/nwsapi/-/nwsapi-2.3.9.tgz",
@@ -543,40 +533,54 @@
"license": "MIT"
},
"node_modules/@chevrotain/cst-dts-gen": {
"version": "12.0.0",
"resolved": "https://registry.npmjs.org/@chevrotain/cst-dts-gen/-/cst-dts-gen-12.0.0.tgz",
"integrity": "sha512-fSL4KXjTl7cDgf0B5Rip9Q05BOrYvkJV/RrBTE/bKDN096E4hN/ySpcBK5B24T76dlQ2i32Zc3PAE27jFnFrKg==",
"version": "11.0.3",
"resolved": "https://registry.npmjs.org/@chevrotain/cst-dts-gen/-/cst-dts-gen-11.0.3.tgz",
"integrity": "sha512-BvIKpRLeS/8UbfxXxgC33xOumsacaeCKAjAeLyOn7Pcp95HiRbrpl14S+9vaZLolnbssPIUuiUd8IvgkRyt6NQ==",
"license": "Apache-2.0",
"dependencies": {
"@chevrotain/gast": "12.0.0",
"@chevrotain/types": "12.0.0"
"@chevrotain/gast": "11.0.3",
"@chevrotain/types": "11.0.3",
"lodash-es": "4.17.21"
}
},
"node_modules/@chevrotain/cst-dts-gen/node_modules/lodash-es": {
"version": "4.17.21",
"resolved": "https://registry.npmjs.org/lodash-es/-/lodash-es-4.17.21.tgz",
"integrity": "sha512-mKnC+QJ9pWVzv+C4/U3rRsHapFfHvQFoFB92e52xeyGMcX6/OlIl78je1u8vePzYZSkkogMPJ2yjxxsb89cxyw==",
"license": "MIT"
},
"node_modules/@chevrotain/gast": {
"version": "12.0.0",
"resolved": "https://registry.npmjs.org/@chevrotain/gast/-/gast-12.0.0.tgz",
"integrity": "sha512-1ne/m3XsIT8aEdrvT33so0GUC+wkctpUPK6zU9IlOyJLUbR0rg4G7ZiApiJbggpgPir9ERy3FRjT6T7lpgetnQ==",
"version": "11.0.3",
"resolved": "https://registry.npmjs.org/@chevrotain/gast/-/gast-11.0.3.tgz",
"integrity": "sha512-+qNfcoNk70PyS/uxmj3li5NiECO+2YKZZQMbmjTqRI3Qchu8Hig/Q9vgkHpI3alNjr7M+a2St5pw5w5F6NL5/Q==",
"license": "Apache-2.0",
"dependencies": {
"@chevrotain/types": "12.0.0"
"@chevrotain/types": "11.0.3",
"lodash-es": "4.17.21"
}
},
"node_modules/@chevrotain/gast/node_modules/lodash-es": {
"version": "4.17.21",
"resolved": "https://registry.npmjs.org/lodash-es/-/lodash-es-4.17.21.tgz",
"integrity": "sha512-mKnC+QJ9pWVzv+C4/U3rRsHapFfHvQFoFB92e52xeyGMcX6/OlIl78je1u8vePzYZSkkogMPJ2yjxxsb89cxyw==",
"license": "MIT"
},
"node_modules/@chevrotain/regexp-to-ast": {
"version": "12.0.0",
"resolved": "https://registry.npmjs.org/@chevrotain/regexp-to-ast/-/regexp-to-ast-12.0.0.tgz",
"integrity": "sha512-p+EW9MaJwgaHguhoqwOtx/FwuGr+DnNn857sXWOi/mClXIkPGl3rn7hGNWvo31HA3vyeQxjqe+H36yZJwYU8cA==",
"version": "11.0.3",
"resolved": "https://registry.npmjs.org/@chevrotain/regexp-to-ast/-/regexp-to-ast-11.0.3.tgz",
"integrity": "sha512-1fMHaBZxLFvWI067AVbGJav1eRY7N8DDvYCTwGBiE/ytKBgP8azTdgyrKyWZ9Mfh09eHWb5PgTSO8wi7U824RA==",
"license": "Apache-2.0"
},
"node_modules/@chevrotain/types": {
"version": "12.0.0",
"resolved": "https://registry.npmjs.org/@chevrotain/types/-/types-12.0.0.tgz",
"integrity": "sha512-S+04vjFQKeuYw0/eW3U52LkAHQsB1ASxsPGsLPUyQgrZ2iNNibQrsidruDzjEX2JYfespXMG0eZmXlhA6z7nWA==",
"version": "11.0.3",
"resolved": "https://registry.npmjs.org/@chevrotain/types/-/types-11.0.3.tgz",
"integrity": "sha512-gsiM3G8b58kZC2HaWR50gu6Y1440cHiJ+i3JUvcp/35JchYejb2+5MVeJK0iKThYpAa/P2PYFV4hoi44HD+aHQ==",
"license": "Apache-2.0"
},
"node_modules/@chevrotain/utils": {
"version": "12.0.0",
"resolved": "https://registry.npmjs.org/@chevrotain/utils/-/utils-12.0.0.tgz",
"integrity": "sha512-lB59uJoaGIfOOL9knQqQRfhl9g7x8/wqFkp13zTdkRu1huG9kg6IJs1O8hqj9rs6h7orGxHJUKb+mX3rPbWGhA==",
"version": "11.0.3",
"resolved": "https://registry.npmjs.org/@chevrotain/utils/-/utils-11.0.3.tgz",
"integrity": "sha512-YslZMgtJUyuMbZ+aKvfF3x1f5liK4mWNxghFRv7jqRR9C3R3fAOGTTKvxXDa2Y1s9zSbcpuO0cAxDYsc9SrXoQ==",
"license": "Apache-2.0"
},
"node_modules/@cspotcode/source-map-support": {
@@ -624,9 +628,9 @@
}
},
"node_modules/@csstools/css-calc": {
"version": "3.2.0",
"resolved": "https://registry.npmjs.org/@csstools/css-calc/-/css-calc-3.2.0.tgz",
"integrity": "sha512-bR9e6o2BDB12jzN/gIbjHa5wLJ4UjD1CB9pM7ehlc0ddk6EBz+yYS1EV2MF55/HUxrHcB/hehAyt5vhsA3hx7w==",
"version": "3.1.1",
"resolved": "https://registry.npmjs.org/@csstools/css-calc/-/css-calc-3.1.1.tgz",
"integrity": "sha512-HJ26Z/vmsZQqs/o3a6bgKslXGFAungXGbinULZO3eMsOyNJHeBBZfup5FiZInOghgoM4Hwnmw+OgbJCNg1wwUQ==",
"dev": true,
"funding": [
{
@@ -648,9 +652,9 @@
}
},
"node_modules/@csstools/css-color-parser": {
"version": "4.1.0",
"resolved": "https://registry.npmjs.org/@csstools/css-color-parser/-/css-color-parser-4.1.0.tgz",
"integrity": "sha512-U0KhLYmy2GVj6q4T3WaAe6NPuFYCPQoE3b0dRGxejWDgcPp8TP7S5rVdM5ZrFaqu4N67X8YaPBw14dQSYx3IyQ==",
"version": "4.0.2",
"resolved": "https://registry.npmjs.org/@csstools/css-color-parser/-/css-color-parser-4.0.2.tgz",
"integrity": "sha512-0GEfbBLmTFf0dJlpsNU7zwxRIH0/BGEMuXLTCvFYxuL1tNhqzTbtnFICyJLTNK4a+RechKP75e7w42ClXSnJQw==",
"dev": true,
"funding": [
{
@@ -665,7 +669,7 @@
"license": "MIT",
"dependencies": {
"@csstools/color-helpers": "^6.0.2",
"@csstools/css-calc": "^3.2.0"
"@csstools/css-calc": "^3.1.1"
},
"engines": {
"node": ">=20.19.0"
@@ -1760,12 +1764,12 @@
}
},
"node_modules/@mermaid-js/parser": {
"version": "1.1.0",
"resolved": "https://registry.npmjs.org/@mermaid-js/parser/-/parser-1.1.0.tgz",
"integrity": "sha512-gxK9ZX2+Fex5zu8LhRQoMeMPEHbc73UKZ0FQ54YrQtUxE1VVhMwzeNtKRPAu5aXks4FasbMe4xB4bWrmq6Jlxw==",
"version": "0.6.3",
"resolved": "https://registry.npmjs.org/@mermaid-js/parser/-/parser-0.6.3.tgz",
"integrity": "sha512-lnjOhe7zyHjc+If7yT4zoedx2vo4sHaTmtkl1+or8BRTnCtDmcTpAjpzDSfCZrshM5bCoz0GyidzadJAH1xobA==",
"license": "MIT",
"dependencies": {
"langium": "^4.0.0"
"langium": "3.3.1"
}
},
"node_modules/@napi-rs/wasm-runtime": {
@@ -2220,6 +2224,257 @@
"dev": true,
"license": "MIT"
},
"node_modules/@swc/core": {
"version": "1.15.8",
"resolved": "https://registry.npmjs.org/@swc/core/-/core-1.15.8.tgz",
"integrity": "sha512-T8keoJjXaSUoVBCIjgL6wAnhADIb09GOELzKg10CjNg+vLX48P93SME6jTfte9MZIm5m+Il57H3rTSk/0kzDUw==",
"dev": true,
"hasInstallScript": true,
"license": "Apache-2.0",
"optional": true,
"peer": true,
"dependencies": {
"@swc/counter": "^0.1.3",
"@swc/types": "^0.1.25"
},
"engines": {
"node": ">=10"
},
"funding": {
"type": "opencollective",
"url": "https://opencollective.com/swc"
},
"optionalDependencies": {
"@swc/core-darwin-arm64": "1.15.8",
"@swc/core-darwin-x64": "1.15.8",
"@swc/core-linux-arm-gnueabihf": "1.15.8",
"@swc/core-linux-arm64-gnu": "1.15.8",
"@swc/core-linux-arm64-musl": "1.15.8",
"@swc/core-linux-x64-gnu": "1.15.8",
"@swc/core-linux-x64-musl": "1.15.8",
"@swc/core-win32-arm64-msvc": "1.15.8",
"@swc/core-win32-ia32-msvc": "1.15.8",
"@swc/core-win32-x64-msvc": "1.15.8"
},
"peerDependencies": {
"@swc/helpers": ">=0.5.17"
},
"peerDependenciesMeta": {
"@swc/helpers": {
"optional": true
}
}
},
"node_modules/@swc/core-darwin-arm64": {
"version": "1.15.8",
"resolved": "https://registry.npmjs.org/@swc/core-darwin-arm64/-/core-darwin-arm64-1.15.8.tgz",
"integrity": "sha512-M9cK5GwyWWRkRGwwCbREuj6r8jKdES/haCZ3Xckgkl8MUQJZA3XB7IXXK1IXRNeLjg6m7cnoMICpXv1v1hlJOg==",
"cpu": [
"arm64"
],
"dev": true,
"license": "Apache-2.0 AND MIT",
"optional": true,
"os": [
"darwin"
],
"peer": true,
"engines": {
"node": ">=10"
}
},
"node_modules/@swc/core-darwin-x64": {
"version": "1.15.8",
"resolved": "https://registry.npmjs.org/@swc/core-darwin-x64/-/core-darwin-x64-1.15.8.tgz",
"integrity": "sha512-j47DasuOvXl80sKJHSi2X25l44CMc3VDhlJwA7oewC1nV1VsSzwX+KOwE5tLnfORvVJJyeiXgJORNYg4jeIjYQ==",
"cpu": [
"x64"
],
"dev": true,
"license": "Apache-2.0 AND MIT",
"optional": true,
"os": [
"darwin"
],
"peer": true,
"engines": {
"node": ">=10"
}
},
"node_modules/@swc/core-linux-arm-gnueabihf": {
"version": "1.15.8",
"resolved": "https://registry.npmjs.org/@swc/core-linux-arm-gnueabihf/-/core-linux-arm-gnueabihf-1.15.8.tgz",
"integrity": "sha512-siAzDENu2rUbwr9+fayWa26r5A9fol1iORG53HWxQL1J8ym4k7xt9eME0dMPXlYZDytK5r9sW8zEA10F2U3Xwg==",
"cpu": [
"arm"
],
"dev": true,
"license": "Apache-2.0",
"optional": true,
"os": [
"linux"
],
"peer": true,
"engines": {
"node": ">=10"
}
},
"node_modules/@swc/core-linux-arm64-gnu": {
"version": "1.15.8",
"resolved": "https://registry.npmjs.org/@swc/core-linux-arm64-gnu/-/core-linux-arm64-gnu-1.15.8.tgz",
"integrity": "sha512-o+1y5u6k2FfPYbTRUPvurwzNt5qd0NTumCTFscCNuBksycloXY16J8L+SMW5QRX59n4Hp9EmFa3vpvNHRVv1+Q==",
"cpu": [
"arm64"
],
"dev": true,
"license": "Apache-2.0 AND MIT",
"optional": true,
"os": [
"linux"
],
"peer": true,
"engines": {
"node": ">=10"
}
},
"node_modules/@swc/core-linux-arm64-musl": {
"version": "1.15.8",
"resolved": "https://registry.npmjs.org/@swc/core-linux-arm64-musl/-/core-linux-arm64-musl-1.15.8.tgz",
"integrity": "sha512-koiCqL09EwOP1S2RShCI7NbsQuG6r2brTqUYE7pV7kZm9O17wZ0LSz22m6gVibpwEnw8jI3IE1yYsQTVpluALw==",
"cpu": [
"arm64"
],
"dev": true,
"license": "Apache-2.0 AND MIT",
"optional": true,
"os": [
"linux"
],
"peer": true,
"engines": {
"node": ">=10"
}
},
"node_modules/@swc/core-linux-x64-gnu": {
"version": "1.15.8",
"resolved": "https://registry.npmjs.org/@swc/core-linux-x64-gnu/-/core-linux-x64-gnu-1.15.8.tgz",
"integrity": "sha512-4p6lOMU3bC+Vd5ARtKJ/FxpIC5G8v3XLoPEZ5s7mLR8h7411HWC/LmTXDHcrSXRC55zvAVia1eldy6zDLz8iFQ==",
"cpu": [
"x64"
],
"dev": true,
"license": "Apache-2.0 AND MIT",
"optional": true,
"os": [
"linux"
],
"peer": true,
"engines": {
"node": ">=10"
}
},
"node_modules/@swc/core-linux-x64-musl": {
"version": "1.15.8",
"resolved": "https://registry.npmjs.org/@swc/core-linux-x64-musl/-/core-linux-x64-musl-1.15.8.tgz",
"integrity": "sha512-z3XBnbrZAL+6xDGAhJoN4lOueIxC/8rGrJ9tg+fEaeqLEuAtHSW2QHDHxDwkxZMjuF/pZ6MUTjHjbp8wLbuRLA==",
"cpu": [
"x64"
],
"dev": true,
"license": "Apache-2.0 AND MIT",
"optional": true,
"os": [
"linux"
],
"peer": true,
"engines": {
"node": ">=10"
}
},
"node_modules/@swc/core-win32-arm64-msvc": {
"version": "1.15.8",
"resolved": "https://registry.npmjs.org/@swc/core-win32-arm64-msvc/-/core-win32-arm64-msvc-1.15.8.tgz",
"integrity": "sha512-djQPJ9Rh9vP8GTS/Df3hcc6XP6xnG5c8qsngWId/BLA9oX6C7UzCPAn74BG/wGb9a6j4w3RINuoaieJB3t+7iQ==",
"cpu": [
"arm64"
],
"dev": true,
"license": "Apache-2.0 AND MIT",
"optional": true,
"os": [
"win32"
],
"peer": true,
"engines": {
"node": ">=10"
}
},
"node_modules/@swc/core-win32-ia32-msvc": {
"version": "1.15.8",
"resolved": "https://registry.npmjs.org/@swc/core-win32-ia32-msvc/-/core-win32-ia32-msvc-1.15.8.tgz",
"integrity": "sha512-/wfAgxORg2VBaUoFdytcVBVCgf1isWZIEXB9MZEUty4wwK93M/PxAkjifOho9RN3WrM3inPLabICRCEgdHpKKQ==",
"cpu": [
"ia32"
],
"dev": true,
"license": "Apache-2.0 AND MIT",
"optional": true,
"os": [
"win32"
],
"peer": true,
"engines": {
"node": ">=10"
}
},
"node_modules/@swc/core-win32-x64-msvc": {
"version": "1.15.8",
"resolved": "https://registry.npmjs.org/@swc/core-win32-x64-msvc/-/core-win32-x64-msvc-1.15.8.tgz",
"integrity": "sha512-GpMePrh9Sl4d61o4KAHOOv5is5+zt6BEXCOCgs/H0FLGeii7j9bWDE8ExvKFy2GRRZVNR1ugsnzaGWHKM6kuzA==",
"cpu": [
"x64"
],
"dev": true,
"license": "Apache-2.0 AND MIT",
"optional": true,
"os": [
"win32"
],
"peer": true,
"engines": {
"node": ">=10"
}
},
"node_modules/@swc/counter": {
"version": "0.1.3",
"resolved": "https://registry.npmjs.org/@swc/counter/-/counter-0.1.3.tgz",
"integrity": "sha512-e2BR4lsJkkRlKZ/qCHPw9ZaSxc0MVUd7gtbtaB7aMvHeJVYe8sOB8DBZkP2DtISHGSku9sCK6T6cnY0CtXrOCQ==",
"dev": true,
"license": "Apache-2.0",
"optional": true,
"peer": true
},
"node_modules/@swc/types": {
"version": "0.1.25",
"resolved": "https://registry.npmjs.org/@swc/types/-/types-0.1.25.tgz",
"integrity": "sha512-iAoY/qRhNH8a/hBvm3zKj9qQ4oc2+3w1unPJa2XvTK3XjeLXtzcCingVPw/9e5mn1+0yPqxcBGp9Jf0pkfMb1g==",
"dev": true,
"license": "Apache-2.0",
"optional": true,
"peer": true,
"dependencies": {
"@swc/counter": "^0.1.3"
}
},
"node_modules/@swc/wasm": {
"version": "1.15.8",
"resolved": "https://registry.npmjs.org/@swc/wasm/-/wasm-1.15.8.tgz",
"integrity": "sha512-RG2BxGbbsjtddFCo1ghKH6A/BMXbY1eMBfpysV0lJMCpI4DZOjW1BNBnxvBt7YsYmlJtmy5UXIg9/4ekBTFFaQ==",
"dev": true,
"license": "Apache-2.0",
"optional": true,
"peer": true
},
"node_modules/@tailwindcss/node": {
"version": "4.1.18",
"resolved": "https://registry.npmjs.org/@tailwindcss/node/-/node-4.1.18.tgz",
@@ -2235,12 +2490,6 @@
"tailwindcss": "4.1.18"
}
},
"node_modules/@tailwindcss/node/node_modules/tailwindcss": {
"version": "4.1.18",
"resolved": "https://registry.npmjs.org/tailwindcss/-/tailwindcss-4.1.18.tgz",
"integrity": "sha512-4+Z+0yiYyEtUVCScyfHCxOYP06L5Ne+JiHhY2IjR2KWMIWhJOYZKLSGZaP5HkZ8+bY0cxfzwDE5uOmzFXyIwxw==",
"license": "MIT"
},
"node_modules/@tailwindcss/oxide": {
"version": "4.1.18",
"resolved": "https://registry.npmjs.org/@tailwindcss/oxide/-/oxide-4.1.18.tgz",
@@ -2483,12 +2732,6 @@
"vite": "^5.2.0 || ^6 || ^7"
}
},
"node_modules/@tailwindcss/vite/node_modules/tailwindcss": {
"version": "4.1.18",
"resolved": "https://registry.npmjs.org/tailwindcss/-/tailwindcss-4.1.18.tgz",
"integrity": "sha512-4+Z+0yiYyEtUVCScyfHCxOYP06L5Ne+JiHhY2IjR2KWMIWhJOYZKLSGZaP5HkZ8+bY0cxfzwDE5uOmzFXyIwxw==",
"license": "MIT"
},
"node_modules/@testing-library/dom": {
"version": "10.4.1",
"resolved": "https://registry.npmjs.org/@testing-library/dom/-/dom-10.4.1.tgz",
@@ -3115,16 +3358,6 @@
"integrity": "sha512-WmoN8qaIAo7WTYWbAZuG8PYEhn5fkz7dZrqTBZ7dtt//lL2Gwms1IcnQ5yHqjDfX8Ft5j4YzDM23f87zBfDe9g==",
"license": "ISC"
},
"node_modules/@upsetjs/venn.js": {
"version": "2.0.0",
"resolved": "https://registry.npmjs.org/@upsetjs/venn.js/-/venn.js-2.0.0.tgz",
"integrity": "sha512-WbBhLrooyePuQ1VZxrJjtLvTc4NVfpOyKx0sKqioq9bX1C1m7Jgykkn8gLrtwumBioXIqam8DLxp88Adbue6Hw==",
"license": "MIT",
"optionalDependencies": {
"d3-selection": "^3.0.0",
"d3-transition": "^3.0.1"
}
},
"node_modules/@vercel/build-utils": {
"version": "13.2.11",
"resolved": "https://registry.npmjs.org/@vercel/build-utils/-/build-utils-13.2.11.tgz",
@@ -3596,14 +3829,14 @@
"license": "MIT"
},
"node_modules/axios": {
"version": "1.15.0",
"resolved": "https://registry.npmjs.org/axios/-/axios-1.15.0.tgz",
"integrity": "sha512-wWyJDlAatxk30ZJer+GeCWS209sA42X+N5jU2jy6oHTp7ufw8uzUTVFBX9+wTfAlhiJXGS0Bq7X6efruWjuK9Q==",
"version": "1.13.2",
"resolved": "https://registry.npmjs.org/axios/-/axios-1.13.2.tgz",
"integrity": "sha512-VPk9ebNqPcy5lRGuSlKx752IlDatOjT9paPlm8A7yOuW2Fbvp4X3JznJtT4f0GzGLLiWE9W8onz51SqLYwzGaA==",
"license": "MIT",
"dependencies": {
"follow-redirects": "^1.15.11",
"form-data": "^4.0.5",
"proxy-from-env": "^2.1.0"
"follow-redirects": "^1.15.6",
"form-data": "^4.0.4",
"proxy-from-env": "^1.1.0"
}
},
"node_modules/bail": {
@@ -3896,33 +4129,37 @@
}
},
"node_modules/chevrotain": {
"version": "12.0.0",
"resolved": "https://registry.npmjs.org/chevrotain/-/chevrotain-12.0.0.tgz",
"integrity": "sha512-csJvb+6kEiQaqo1woTdSAuOWdN0WTLIydkKrBnS+V5gZz0oqBrp4kQ35519QgK6TpBThiG3V1vNSHlIkv4AglQ==",
"version": "11.0.3",
"resolved": "https://registry.npmjs.org/chevrotain/-/chevrotain-11.0.3.tgz",
"integrity": "sha512-ci2iJH6LeIkvP9eJW6gpueU8cnZhv85ELY8w8WiFtNjMHA5ad6pQLaJo9mEly/9qUyCpvqX8/POVUTf18/HFdw==",
"license": "Apache-2.0",
"dependencies": {
"@chevrotain/cst-dts-gen": "12.0.0",
"@chevrotain/gast": "12.0.0",
"@chevrotain/regexp-to-ast": "12.0.0",
"@chevrotain/types": "12.0.0",
"@chevrotain/utils": "12.0.0"
},
"engines": {
"node": ">=22.0.0"
"@chevrotain/cst-dts-gen": "11.0.3",
"@chevrotain/gast": "11.0.3",
"@chevrotain/regexp-to-ast": "11.0.3",
"@chevrotain/types": "11.0.3",
"@chevrotain/utils": "11.0.3",
"lodash-es": "4.17.21"
}
},
"node_modules/chevrotain-allstar": {
"version": "0.4.1",
"resolved": "https://registry.npmjs.org/chevrotain-allstar/-/chevrotain-allstar-0.4.1.tgz",
"integrity": "sha512-PvVJm3oGqrveUVW2Vt/eZGeiAIsJszYweUcYwcskg9e+IubNYKKD+rHHem7A6XVO22eDAL+inxNIGAzZ/VIWlA==",
"version": "0.3.1",
"resolved": "https://registry.npmjs.org/chevrotain-allstar/-/chevrotain-allstar-0.3.1.tgz",
"integrity": "sha512-b7g+y9A0v4mxCW1qUhf3BSVPg+/NvGErk/dOkrDaHA0nQIQGAtrOjlX//9OQtRlSCy+x9rfB5N8yC71lH1nvMw==",
"license": "MIT",
"dependencies": {
"lodash-es": "^4.17.21"
},
"peerDependencies": {
"chevrotain": "^12.0.0"
"chevrotain": "^11.0.0"
}
},
"node_modules/chevrotain/node_modules/lodash-es": {
"version": "4.17.21",
"resolved": "https://registry.npmjs.org/lodash-es/-/lodash-es-4.17.21.tgz",
"integrity": "sha512-mKnC+QJ9pWVzv+C4/U3rRsHapFfHvQFoFB92e52xeyGMcX6/OlIl78je1u8vePzYZSkkogMPJ2yjxxsb89cxyw==",
"license": "MIT"
},
"node_modules/chownr": {
"version": "3.0.0",
"resolved": "https://registry.npmjs.org/chownr/-/chownr-3.0.0.tgz",
@@ -4593,9 +4830,9 @@
}
},
"node_modules/dagre-d3-es": {
"version": "7.0.14",
"resolved": "https://registry.npmjs.org/dagre-d3-es/-/dagre-d3-es-7.0.14.tgz",
"integrity": "sha512-P4rFMVq9ESWqmOgK+dlXvOtLwYg0i7u0HBGJER0LZDJT2VHIPAMZ/riPxqJceWMStH5+E61QxFra9kIS3AqdMg==",
"version": "7.0.13",
"resolved": "https://registry.npmjs.org/dagre-d3-es/-/dagre-d3-es-7.0.13.tgz",
"integrity": "sha512-efEhnxpSuwpYOKRm/L5KbqoZmNNukHa/Flty4Wp62JRvgH2ojwVgPgdYyr4twpieZnyRDdIH7PY2mopX26+j2Q==",
"license": "MIT",
"dependencies": {
"d3": "^7.9.0",
@@ -5827,9 +6064,9 @@
}
},
"node_modules/joi": {
"version": "18.1.2",
"resolved": "https://registry.npmjs.org/joi/-/joi-18.1.2.tgz",
"integrity": "sha512-rF5MAmps5esSlhCA+N1b6IYHDw9j/btzGaqfgie522jS02Ju/HXBxamlXVlKEHAxoMKQL77HWI8jlqWsFuekZA==",
"version": "18.0.2",
"resolved": "https://registry.npmjs.org/joi/-/joi-18.0.2.tgz",
"integrity": "sha512-RuCOQMIt78LWnktPoeBL0GErkNaJPTBGcYuyaBvUOQSpcpcLfWrHPPihYdOGbV5pam9VTWbeoF7TsGiHugcjGA==",
"dev": true,
"license": "BSD-3-Clause",
"dependencies": {
@@ -5839,7 +6076,7 @@
"@hapi/pinpoint": "^2.0.1",
"@hapi/tlds": "^1.1.1",
"@hapi/topo": "^6.0.2",
"@standard-schema/spec": "^1.1.0"
"@standard-schema/spec": "^1.0.0"
},
"engines": {
"node": ">= 20"
@@ -5861,14 +6098,14 @@
"license": "MIT"
},
"node_modules/jsdom": {
"version": "29.0.2",
"resolved": "https://registry.npmjs.org/jsdom/-/jsdom-29.0.2.tgz",
"integrity": "sha512-9VnGEBosc/ZpwyOsJBCQ/3I5p7Q5ngOY14a9bf5btenAORmZfDse1ZEheMiWcJ3h81+Fv7HmJFdS0szo/waF2w==",
"version": "29.0.0",
"resolved": "https://registry.npmjs.org/jsdom/-/jsdom-29.0.0.tgz",
"integrity": "sha512-9FshNB6OepopZ08unmmGpsF7/qCjxGPbo3NbgfJAnPeHXnsODE9WWffXZtRFRFe0ntzaAOcSKNJFz8wiyvF1jQ==",
"dev": true,
"license": "MIT",
"dependencies": {
"@asamuzakjp/css-color": "^5.1.5",
"@asamuzakjp/dom-selector": "^7.0.6",
"@asamuzakjp/css-color": "^5.0.1",
"@asamuzakjp/dom-selector": "^7.0.2",
"@bramus/specificity": "^2.4.2",
"@csstools/css-syntax-patches-for-csstree": "^1.1.1",
"@exodus/bytes": "^1.15.0",
@@ -5882,7 +6119,7 @@
"saxes": "^6.0.0",
"symbol-tree": "^3.2.4",
"tough-cookie": "^6.0.1",
"undici": "^7.24.5",
"undici": "^7.24.3",
"w3c-xmlserializer": "^5.0.0",
"webidl-conversions": "^8.0.1",
"whatwg-mimetype": "^5.0.0",
@@ -5915,9 +6152,9 @@
}
},
"node_modules/jsdom/node_modules/undici": {
"version": "7.25.0",
"resolved": "https://registry.npmjs.org/undici/-/undici-7.25.0.tgz",
"integrity": "sha512-xXnp4kTyor2Zq+J1FfPI6Eq3ew5h6Vl0F/8d9XU5zZQf1tX9s2Su1/3PiMmUANFULpmksxkClamIZcaUqryHsQ==",
"version": "7.24.3",
"resolved": "https://registry.npmjs.org/undici/-/undici-7.24.3.tgz",
"integrity": "sha512-eJdUmK/Wrx2d+mnWWmwwLRyA7OQCkLap60sk3dOK4ViZR7DKwwptwuIvFBg2HaiP9ESaEdhtpSymQPvytpmkCA==",
"dev": true,
"license": "MIT",
"engines": {
@@ -6058,21 +6295,19 @@
}
},
"node_modules/langium": {
"version": "4.2.2",
"resolved": "https://registry.npmjs.org/langium/-/langium-4.2.2.tgz",
"integrity": "sha512-JUshTRAfHI4/MF9dH2WupvjSXyn8JBuUEWazB8ZVJUtXutT0doDlAv1XKbZ1Pb5sMexa8FF4CFBc0iiul7gbUQ==",
"version": "3.3.1",
"resolved": "https://registry.npmjs.org/langium/-/langium-3.3.1.tgz",
"integrity": "sha512-QJv/h939gDpvT+9SiLVlY7tZC3xB2qK57v0J04Sh9wpMb6MP1q8gB21L3WIo8T5P1MSMg3Ep14L7KkDCFG3y4w==",
"license": "MIT",
"dependencies": {
"@chevrotain/regexp-to-ast": "~12.0.0",
"chevrotain": "~12.0.0",
"chevrotain-allstar": "~0.4.1",
"chevrotain": "~11.0.3",
"chevrotain-allstar": "~0.3.0",
"vscode-languageserver": "~9.0.1",
"vscode-languageserver-textdocument": "~1.0.11",
"vscode-uri": "~3.1.0"
"vscode-uri": "~3.0.8"
},
"engines": {
"node": ">=20.10.0",
"npm": ">=10.2.3"
"node": ">=16.0.0"
}
},
"node_modules/langsmith": {
@@ -6378,16 +6613,16 @@
}
},
"node_modules/lodash": {
"version": "4.18.1",
"resolved": "https://registry.npmjs.org/lodash/-/lodash-4.18.1.tgz",
"integrity": "sha512-dMInicTPVE8d1e5otfwmmjlxkZoUpiVLwyeTdUsi/Caj/gfzzblBcCE5sRHV/AsjuCmxWrte2TNGSYuCeCq+0Q==",
"version": "4.17.23",
"resolved": "https://registry.npmjs.org/lodash/-/lodash-4.17.23.tgz",
"integrity": "sha512-LgVTMpQtIopCi79SJeDiP0TfWi5CNEc/L/aRdTh3yIvmZXTnheWpKjSZhnvMl8iXbC1tFg9gdHHDMLoV7CnG+w==",
"dev": true,
"license": "MIT"
},
"node_modules/lodash-es": {
"version": "4.18.1",
"resolved": "https://registry.npmjs.org/lodash-es/-/lodash-es-4.18.1.tgz",
"integrity": "sha512-J8xewKD/Gk22OZbhpOVSwcs60zhd95ESDwezOFuA3/099925PdHJ7OFHNTGtajL3AlZkykD32HykiMo+BIBI8A==",
"version": "4.17.22",
"resolved": "https://registry.npmjs.org/lodash-es/-/lodash-es-4.17.22.tgz",
"integrity": "sha512-XEawp1t0gxSi9x01glktRZ5HDy0HXqrM0x5pXQM98EaI0NxO6jVM7omDOxsuEo5UIASAnm2bRp1Jt/e0a2XU8Q==",
"license": "MIT"
},
"node_modules/longest-streak": {
@@ -6837,28 +7072,27 @@
}
},
"node_modules/mermaid": {
"version": "11.14.0",
"resolved": "https://registry.npmjs.org/mermaid/-/mermaid-11.14.0.tgz",
"integrity": "sha512-GSGloRsBs+JINmmhl0JDwjpuezCsHB4WGI4NASHxL3fHo3o/BRXTxhDLKnln8/Q0lRFRyDdEjmk1/d5Sn1Xz8g==",
"version": "11.12.2",
"resolved": "https://registry.npmjs.org/mermaid/-/mermaid-11.12.2.tgz",
"integrity": "sha512-n34QPDPEKmaeCG4WDMGy0OT6PSyxKCfy2pJgShP+Qow2KLrvWjclwbc3yXfSIf4BanqWEhQEpngWwNp/XhZt6w==",
"license": "MIT",
"dependencies": {
"@braintree/sanitize-url": "^7.1.1",
"@iconify/utils": "^3.0.2",
"@mermaid-js/parser": "^1.1.0",
"@iconify/utils": "^3.0.1",
"@mermaid-js/parser": "^0.6.3",
"@types/d3": "^7.4.3",
"@upsetjs/venn.js": "^2.0.0",
"cytoscape": "^3.33.1",
"cytoscape": "^3.29.3",
"cytoscape-cose-bilkent": "^4.1.0",
"cytoscape-fcose": "^2.2.0",
"d3": "^7.9.0",
"d3-sankey": "^0.12.3",
"dagre-d3-es": "7.0.14",
"dayjs": "^1.11.19",
"dompurify": "^3.3.1",
"katex": "^0.16.25",
"dagre-d3-es": "7.0.13",
"dayjs": "^1.11.18",
"dompurify": "^3.2.5",
"katex": "^0.16.22",
"khroma": "^2.1.0",
"lodash-es": "^4.17.23",
"marked": "^16.3.0",
"lodash-es": "^4.17.21",
"marked": "^16.2.1",
"roughjs": "^4.6.6",
"stylis": "^4.3.6",
"ts-dedent": "^2.2.0",
@@ -8083,13 +8317,10 @@
}
},
"node_modules/proxy-from-env": {
"version": "2.1.0",
"resolved": "https://registry.npmjs.org/proxy-from-env/-/proxy-from-env-2.1.0.tgz",
"integrity": "sha512-cJ+oHTW1VAEa8cJslgmUZrc+sjRKgAKl3Zyse6+PV38hZe/V6Z14TbCuXcan9F9ghlz4QrFr2c92TNF82UkYHA==",
"license": "MIT",
"engines": {
"node": ">=10"
}
"version": "1.1.0",
"resolved": "https://registry.npmjs.org/proxy-from-env/-/proxy-from-env-1.1.0.tgz",
"integrity": "sha512-D+zkORCbA9f1tdWRK0RaCR3GPv50cMxcrz4X8k5LTSUD1Dkw47mKJEZQNunItRTkWwgtaUSo1RVFRIG9ZXiFYg==",
"license": "MIT"
},
"node_modules/punycode": {
"version": "2.3.1",
@@ -8775,9 +9006,9 @@
"license": "MIT"
},
"node_modules/tailwindcss": {
"version": "4.2.2",
"resolved": "https://registry.npmjs.org/tailwindcss/-/tailwindcss-4.2.2.tgz",
"integrity": "sha512-KWBIxs1Xb6NoLdMVqhbhgwZf2PGBpPEiwOqgI4pFIYbNTfBXiKYyWoTsXgBQ9WFg/OlhnvHaY+AEpW7wSmFo2Q==",
"version": "4.1.18",
"resolved": "https://registry.npmjs.org/tailwindcss/-/tailwindcss-4.1.18.tgz",
"integrity": "sha512-4+Z+0yiYyEtUVCScyfHCxOYP06L5Ne+JiHhY2IjR2KWMIWhJOYZKLSGZaP5HkZ8+bY0cxfzwDE5uOmzFXyIwxw==",
"license": "MIT"
},
"node_modules/tapable": {
@@ -10039,9 +10270,9 @@
"license": "MIT"
},
"node_modules/vscode-uri": {
"version": "3.1.0",
"resolved": "https://registry.npmjs.org/vscode-uri/-/vscode-uri-3.1.0.tgz",
"integrity": "sha512-/BpdSx+yCQGnCvecbyXdxHDkuk55/G3xwnC0GqY4gmQ3j+A+g8kzzgB4Nk/SINjqn6+waqw3EgbVF2QKExkRxQ==",
"version": "3.0.8",
"resolved": "https://registry.npmjs.org/vscode-uri/-/vscode-uri-3.0.8.tgz",
"integrity": "sha512-AyFQ0EVmsOZOlAnxoFOGOq1SQDWAB7C6aqMGS23svWAllfOaxbuFvcT8D1i8z3Gyn8fraVeZNNmN6e9bxxXkKw==",
"license": "MIT"
},
"node_modules/w3c-xmlserializer": {
@@ -10058,15 +10289,15 @@
}
},
"node_modules/wait-on": {
"version": "9.0.5",
"resolved": "https://registry.npmjs.org/wait-on/-/wait-on-9.0.5.tgz",
"integrity": "sha512-qgnbHDfDTRIp73ANEJNRW/7kn8CrDUcvZz18xotJQku/P4saTGkbIzvnMZebPmVvVNUiRq1qWAPyqCH+W4H8KA==",
"version": "8.0.5",
"resolved": "https://registry.npmjs.org/wait-on/-/wait-on-8.0.5.tgz",
"integrity": "sha512-J3WlS0txVHkhLRb2FsmRg3dkMTCV1+M6Xra3Ho7HzZDHpE7DCOnoSoCJsZotrmW3uRMhvIJGSKUKrh/MeF4iag==",
"dev": true,
"license": "MIT",
"dependencies": {
"axios": "^1.15.0",
"joi": "^18.1.2",
"lodash": "^4.18.1",
"axios": "^1.12.1",
"joi": "^18.0.1",
"lodash": "^4.17.21",
"minimist": "^1.2.8",
"rxjs": "^7.8.2"
},
@@ -10074,7 +10305,7 @@
"wait-on": "bin/wait-on"
},
"engines": {
"node": ">=20.0.0"
"node": ">=12.0.0"
}
},
"node_modules/webidl-conversions": {
+4 -4
View File
@@ -39,7 +39,7 @@
"langchain": "^1.2.10",
"lru-cache": "^11.2.4",
"lucide-react": "^0.562.0",
"mermaid": "^11.14.0",
"mermaid": "^11.12.2",
"mnemonist": "^0.39.0",
"pandemonium": "^2.4.0",
"react": "^18.3.1",
@@ -49,7 +49,7 @@
"react-zoom-pan-pinch": "^3.7.0",
"remark-gfm": "^4.0.1",
"sigma": "^3.0.2",
"tailwindcss": "^4.2.2",
"tailwindcss": "^4.1.18",
"uuid": "^13.0.0",
"zod": "^3.25.76"
},
@@ -67,11 +67,11 @@
"@vercel/node": "^5.5.16",
"@vitejs/plugin-react": "^5.1.0",
"@vitest/coverage-v8": "^3.2.4",
"jsdom": "^29.0.2",
"jsdom": "^29.0.0",
"tree-sitter-wasms": "^0.1.13",
"typescript": "^5.4.5",
"vite": "^5.2.0",
"vitest": "^3.2.4",
"wait-on": "^9.0.5"
"wait-on": "^8.0.5"
}
}
-1
View File
@@ -16,7 +16,6 @@ export default defineConfig({
alias: {
'@': path.resolve(__dirname, './src'),
'@shared': path.resolve(__dirname, '../shared'),
'gitnexus-shared': path.resolve(__dirname, '../gitnexus-shared/src/index.ts'),
// Fix for Rollup failing to resolve this deep import from @langchain/anthropic
'@anthropic-ai/sdk/lib/transform-json-schema': path.resolve(
__dirname,
-4
View File
@@ -9,10 +9,6 @@ tsconfig.json
.gitignore
node_modules/
# Vendor build artifacts (created during install, not shipped)
vendor/**/node_modules
vendor/**/build
# Package lock (consumers use their own)
package-lock.json
-40
View File
@@ -2,46 +2,6 @@
All notable changes to GitNexus will be documented in this file.
## [1.6.2] - 2026-04-18
### Added
- **Docker support** — containerized ingestion and MCP serving for reproducible runs on CI and container platforms (#848)
- **Language-agnostic heritage extractor** — config+factory pattern for class-heritage extraction (EXTENDS / IMPLEMENTS), completing the extractor refactor alongside method/field/call/variable (#890)
- **Language-agnostic call extractor** — config+factory pattern that collapses ~225 lines of inline parse-worker logic into declarative per-language configs (#877)
- **Language-agnostic variable extractor** — structured metadata for `Const` / `Static` / `Variable` nodes via config+factory pattern (#878)
- **AST-aware embedding chunking** — offset-based splitting preserves symbol boundaries, improving semantic search precision on large files (#889)
- **HTTP consumer detection for jQuery and axios object-form** — `$.ajax` / `$.get` / `$.post` and `axios({ url, method })` now recognized as HTTP call sites (#887)
### Fixed
- **Python external dotted imports** — avoid spurious same-file matches when an import path like `foo.bar.baz` refers to a third-party module (#899)
- **Worker warnings no longer terminate ingestion** — non-fatal parser warnings keep the pipeline running instead of aborting the run (#900, #261)
- **Global-install upgrade `ENOTEMPTY`** — devendored `tree-sitter-proto` install lifecycle + preinstall cleanup so `npm i -g gitnexus@latest` succeeds on top of an older install (#843, #846)
- **`env.cacheDir`** now defaults to a user-writable location, unblocking ingestion on systems where the install directory is read-only (#845)
- **Content-hash staleness detection for embeddings** — zero-node rebuilds no longer skip vector-index creation, fixing semantic search after selective re-analysis (#831)
- **`tree-sitter-c-sharp` version pin** — locked to 0.23.1 to avoid a breaking change in a transitive prerelease (#834)
- **`release-drafter` v7 CI** — replaced the removed `disable-releaser` flag with `dry-run` so release-note drafts still work
- **`npm arborist` crash from `tree-sitter-dart`** — switched the dependency URL format so `npm install` no longer crashes on clean installs
- **Service-group `ManifestExtractor`** — `config.links` now wires the manifest extractor properly, restoring cross-link discovery that had silently dropped to zero
### Changed
- **SemanticModel wired as a first-class resolution input (SM-20)** — `call-processor`, `resolution-context`, `type-env`, and `heritage-map` now consult `table.model.*` directly; 37 internal call sites migrated off the SymbolTable wrapper (#885)
- **Per-strategy `ImportSemantics` hooks** — `named` / `wildcard-transitive` / `wildcard-leaf` / `namespace` strategies split into composable hooks, replacing the monolithic conditional (Strategies 1–4 of #886)
- **Class extraction configs moved to `configs/` subdirectory** — per-language class configs now co-locate with the other extractor configs, completing the extractor layer's directory convention (#879)
- **CLI AI-context trimmed** — duplicated CLAUDE.md block removed from the shipped context, reducing token usage in LLM-consuming workflows (#904)
- **LLM context files optimized** — AI-consumed documentation tuned for accuracy and token efficiency (#857)
- **Workflow concurrency standardized** — all CI workflows adopt the consistent concurrency key pattern documented in CONTRIBUTING.md; release-note labeling automated (#837)
- **E2E status-ready timeout raised** — 45s accommodates parallel-worker startup variance on CI (#908)
### Chore / Dependencies
- **tree-sitter 0.25 upgrade readiness** — daily Dependabot monitor for the upcoming major-version bump (#847)
- Dependency bumps: `glob` 11.1.0 → 13.0.6 (#867), `commander` 12.1.0 → 14.0.3 (#868), `@huggingface/transformers` (#869), `@modelcontextprotocol/sdk` (#866), `lru-cache` 11.2.7 → 11.3.5 (#870), `mnemonist` 0.39.8 → 0.40.3 (#871), `@ladybugdb/core` (#873)
- gitnexus-web dependency bumps: `mermaid` 11.12.2 → 11.14.0 (#860), `tailwindcss` (#861), `jsdom` 29.0.0 → 29.0.2 (#863), `wait-on` 8.0.5 → 9.0.5 (#859), `@vitest/coverage-v8` (#864)
- GitHub Actions bumps: `actions/checkout` 4.3.1 → 6.0.2 (#842), `actions/upload-artifact` 4.6.2 → 7.0.1 (#838), `actions/setup-node` 4.4.0 → 6.3.0 (#841), `actions/cache` 5.0.4 → 5.0.5 (#840), `actions/github-script` 7.0.1 → 9.0.0 (#850), `dorny/paths-filter` 3.0.2 → 4.0.1 (#839), `amannn/action-semantic-pull-request` 6.1.1 (#853), `release-drafter/release-drafter` 6.0.0 → 7.2.0 (#852), `marocchino/sticky-pull-request-comment` 3.0.4 (#851), `softprops/action-gh-release` 2.5.0 → 3.0.0 (#849)
## [1.6.1] - 2026-04-13
### Added
+1 -1
View File
@@ -1,6 +1,6 @@
FROM node:20-bookworm
WORKDIR /app
RUN apt-get -o Acquire::Check-Valid-Until=false -o Acquire::Check-Date=false update && apt-get install -y python3 make g++ && rm -rf /var/lib/apt/lists/*
RUN apt-get update && apt-get install -y python3 make g++ && rm -rf /var/lib/apt/lists/*
COPY . .
RUN npm ci --ignore-scripts \
&& node scripts/patch-tree-sitter-swift.cjs \
-23
View File
@@ -234,29 +234,6 @@ Installed automatically by both `gitnexus analyze` (per-repo) and `gitnexus setu
- Node.js >= 18
- Git repository (uses git for commit tracking)
## Release candidates
Stable releases publish to the default `latest` dist-tag. When a pull request
with non-documentation changes merges into `main`, an automated workflow also
publishes a prerelease build under the `rc` dist-tag, so early adopters can
try in-flight fixes without waiting for the next stable cut. (Docs-only
merges are skipped.)
```bash
# Try the latest release candidate (pre-stable — may change at any time)
npm install -g gitnexus@rc
# — or —
npx gitnexus@rc analyze
```
Release-candidate versions follow the standard semver prerelease format
`X.Y.Z-rc.N`, where `X.Y.Z` is the next stable target (bumped from the
current `latest` by patch by default; `minor` or `major` when kicking off a
bigger cycle) and `N` increments per published rc. Example sequence:
`1.6.2-rc.1`, `1.6.2-rc.2`, …, then once `1.6.2` ships stable,
`1.6.3-rc.1`. See the [Releases page](https://github.com/abhigyanpatwari/GitNexus/releases)
for the full list; stable `latest` is unaffected.
## Troubleshooting
### `Cannot destructure property 'package' of 'node.target' as it is null`
+411 -272
View File
File diff suppressed because it is too large Load Diff
+7 -9
View File
@@ -1,6 +1,6 @@
{
"name": "gitnexus",
"version": "1.6.3-rc.5",
"version": "1.6.1",
"description": "Graph-powered code intelligence for AI agents. Index any codebase, query via MCP or CLI.",
"author": "Abhigyan Patwari",
"license": "PolyForm-Noncommercial-1.0.0",
@@ -46,32 +46,32 @@
"test:integration": "vitest run test/integration",
"test:watch": "vitest",
"test:coverage": "vitest run --coverage",
"postinstall": "node scripts/patch-tree-sitter-swift.cjs && node scripts/build-tree-sitter-proto.cjs",
"postinstall": "node scripts/patch-tree-sitter-swift.cjs",
"prepare": "node scripts/build.js",
"prepack": "node scripts/build.js"
},
"dependencies": {
"@huggingface/transformers": "^4.1.0",
"@huggingface/transformers": "^3.0.0",
"@ladybugdb/core": "^0.15.2",
"@modelcontextprotocol/sdk": "^1.0.0",
"@scarf/scarf": "^1.4.0",
"cli-progress": "^3.12.0",
"commander": "^14.0.3",
"commander": "^12.0.0",
"cors": "^2.8.5",
"express": "^4.19.2",
"glob": "^13.0.6",
"glob": "^11.0.0",
"graphology": "^0.25.4",
"graphology-indices": "^0.17.0",
"graphology-utils": "^2.3.0",
"ignore": "^7.0.5",
"js-yaml": "^4.1.1",
"lru-cache": "^11.0.0",
"mnemonist": "^0.40.3",
"mnemonist": "^0.39.0",
"onnxruntime-node": "^1.24.0",
"pandemonium": "^2.4.0",
"tree-sitter": "^0.21.1",
"tree-sitter-c": "0.23.2",
"tree-sitter-c-sharp": "0.23.1",
"tree-sitter-c-sharp": "^0.23.1",
"tree-sitter-cpp": "^0.23.4",
"tree-sitter-go": "^0.23.0",
"tree-sitter-java": "^0.23.5",
@@ -84,8 +84,6 @@
"uuid": "^13.0.0"
},
"optionalDependencies": {
"node-addon-api": "^8.0.0",
"node-gyp-build": "^4.8.0",
"tree-sitter-dart": "git+https://github.com/UserNobody14/tree-sitter-dart.git#80e23c07b64494f7e21090bb3450223ef0b192f4",
"tree-sitter-kotlin": "^0.3.8",
"tree-sitter-proto": "file:./vendor/tree-sitter-proto",
@@ -1,82 +0,0 @@
#!/usr/bin/env node
/**
* Build tree-sitter-proto native binding.
*
* Why this script exists:
* tree-sitter-proto is vendored under gitnexus/vendor/tree-sitter-proto/
* and declared as a `file:` optionalDependency. Previously, the vendored
* package had its own `dependencies` and `install` script, which caused
* npm to create `vendor/tree-sitter-proto/node_modules/` and
* `vendor/tree-sitter-proto/build/` during install. Those directories
* blocked `rmdir` on global-install upgrade, producing:
*
* ENOTEMPTY: directory not empty, rmdir
* '.../gitnexus/vendor/tree-sitter-proto/node_modules/node-addon-api'
*
* (See https://github.com/abhigyanpatwari/GitNexus/issues/836.)
*
* We stripped `dependencies` and the `install` script from the vendored
* package.json, hoisted `node-addon-api` and `node-gyp-build` into
* gitnexus's own optionalDependencies, and moved native compilation here.
*
* What this does:
* Runs `npx node-gyp rebuild` inside `node_modules/tree-sitter-proto/`
* (which npm creates as a copy of vendor/tree-sitter-proto/ when
* resolving the file: dep). Build output lands in
* `node_modules/tree-sitter-proto/build/Release/tree_sitter_proto_binding.node`
* — under npm-managed territory, safe on upgrade.
*
* Mirrors scripts/patch-tree-sitter-swift.cjs. Best-effort: if any
* precondition fails (optional dep absent, no toolchain, --ignore-scripts),
* warn and exit 0 so gitnexus install still succeeds.
*/
const fs = require('fs');
const path = require('path');
const { execSync } = require('child_process');
const protoDir = path.join(__dirname, '..', 'node_modules', 'tree-sitter-proto');
const bindingGyp = path.join(protoDir, 'binding.gyp');
const bindingNode = path.join(protoDir, 'build', 'Release', 'tree_sitter_proto_binding.node');
try {
if (!fs.existsSync(bindingGyp)) {
// tree-sitter-proto is an optionalDependency; absent when install
// skipped optional deps or the file: dep was not resolved.
process.exit(0);
}
// Skip if the native binding already exists (idempotent re-run).
if (fs.existsSync(bindingNode)) {
process.exit(0);
}
// Pre-flight: the hoisted build deps must be resolvable.
try {
require.resolve('node-addon-api');
require.resolve('node-gyp-build');
} catch (resolveErr) {
console.warn(
'[tree-sitter-proto] Skipping build: hoisted build deps not resolvable (%s).',
resolveErr.message,
);
console.warn(
'[tree-sitter-proto] Proto parsing will be unavailable. Install without --no-optional and with scripts enabled to build.',
);
process.exit(0);
}
console.log('[tree-sitter-proto] Building native binding...');
execSync('npx node-gyp rebuild', {
cwd: protoDir,
stdio: 'pipe',
timeout: 180000,
});
console.log('[tree-sitter-proto] Native binding built successfully');
} catch (err) {
console.warn('[tree-sitter-proto] Could not build native binding:', err.message);
console.warn(
'[tree-sitter-proto] Proto (.proto) parsing will be unavailable. Non-proto gitnexus functionality is unaffected.',
);
// Exit 0: optionalDependency failures must not fail the gitnexus install.
process.exit(0);
}
+58
View File
@@ -101,6 +101,19 @@ This project is indexed by GitNexus as **${projectName}**${noStats ? '' : ` (${s
- When exploring unfamiliar code, use \`gitnexus_query({query: "concept"})\` to find execution flows instead of grepping. It returns process-grouped results ranked by relevance.
- When you need full context on a specific symbol — callers, callees, which execution flows it participates in — use \`gitnexus_context({name: "symbolName"})\`.
## When Debugging
1. \`gitnexus_query({query: "<error or symptom>"})\` — find execution flows related to the issue
2. \`gitnexus_context({name: "<suspect function>"})\` — see all callers, callees, and process participation
3. \`READ gitnexus://repo/${projectName}/process/{processName}\` — trace the full execution flow step by step
4. For regressions: \`gitnexus_detect_changes({scope: "compare", base_ref: "main"})\` — see what your branch changed
## When Refactoring
- **Renaming**: MUST use \`gitnexus_rename({symbol_name: "old", new_name: "new", dry_run: true})\` first. Review the preview — graph edits are safe, text_search edits need manual review. Then run with \`dry_run: false\`.
- **Extracting/Splitting**: MUST run \`gitnexus_context({name: "target"})\` to see all incoming/outgoing refs, then \`gitnexus_impact({target: "target", direction: "upstream"})\` to find all external callers before moving code.
- After any refactor: run \`gitnexus_detect_changes({scope: "all"})\` to verify only expected files changed.
## Never Do
- NEVER edit a function, class, or method without first running \`gitnexus_impact\` on it.
@@ -108,6 +121,25 @@ This project is indexed by GitNexus as **${projectName}**${noStats ? '' : ` (${s
- NEVER rename symbols with find-and-replace — use \`gitnexus_rename\` which understands the call graph.
- NEVER commit changes without running \`gitnexus_detect_changes()\` to check affected scope.
## Tools Quick Reference
| Tool | When to use | Command |
|------|-------------|---------|
| \`query\` | Find code by concept | \`gitnexus_query({query: "auth validation"})\` |
| \`context\` | 360-degree view of one symbol | \`gitnexus_context({name: "validateUser"})\` |
| \`impact\` | Blast radius before editing | \`gitnexus_impact({target: "X", direction: "upstream"})\` |
| \`detect_changes\` | Pre-commit scope check | \`gitnexus_detect_changes({scope: "staged"})\` |
| \`rename\` | Safe multi-file rename | \`gitnexus_rename({symbol_name: "old", new_name: "new", dry_run: true})\` |
| \`cypher\` | Custom graph queries | \`gitnexus_cypher({query: "MATCH ..."})\` |
## Impact Risk Levels
| Depth | Meaning | Action |
|-------|---------|--------|
| d=1 | WILL BREAK — direct callers/importers | MUST update these |
| d=2 | LIKELY AFFECTED — indirect deps | Should test |
| d=3 | MAY NEED TESTING — transitive | Test if critical path |
## Resources
| Resource | Use for |
@@ -117,6 +149,32 @@ This project is indexed by GitNexus as **${projectName}**${noStats ? '' : ` (${s
| \`gitnexus://repo/${projectName}/processes\` | All execution flows |
| \`gitnexus://repo/${projectName}/process/{name}\` | Step-by-step execution trace |
## Self-Check Before Finishing
Before completing any code modification task, verify:
1. \`gitnexus_impact\` was run for all modified symbols
2. No HIGH/CRITICAL risk warnings were ignored
3. \`gitnexus_detect_changes()\` confirms changes match expected scope
4. All d=1 (WILL BREAK) dependents were updated
## Keeping the Index Fresh
After committing code changes, the GitNexus index becomes stale. Re-run analyze to update it:
\`\`\`bash
npx gitnexus analyze
\`\`\`
If the index previously included embeddings, preserve them by adding \`--embeddings\`:
\`\`\`bash
npx gitnexus analyze --embeddings
\`\`\`
To check whether embeddings exist, inspect \`.gitnexus/meta.json\` — the \`stats.embeddings\` field shows the count (0 means no embeddings). **Running analyze without \`--embeddings\` will delete any previously generated embeddings.**
> Claude Code users: A PostToolUse hook handles this automatically after \`git commit\` and \`git merge\`.
${
groupNames && groupNames.length > 0
? `## Cross-Repo Groups
-3
View File
@@ -147,11 +147,9 @@ export const analyzeCommand = async (inputPath?: string, options?: AnalyzeOption
const origLog = console.log.bind(console);
const origWarn = console.warn.bind(console);
const origError = console.error.bind(console);
let barCurrentValue = 0;
const barLog = (...args: any[]) => {
process.stdout.write('\x1b[2K\r');
origLog(args.map((a) => (typeof a === 'string' ? a : String(a))).join(' '));
bar.update(barCurrentValue);
};
console.log = barLog;
console.warn = barLog;
@@ -162,7 +160,6 @@ export const analyzeCommand = async (inputPath?: string, options?: AnalyzeOption
let phaseStart = Date.now();
const updateBar = (value: number, phaseLabel: string) => {
barCurrentValue = value;
if (phaseLabel !== lastPhaseLabel) {
lastPhaseLabel = phaseLabel;
phaseStart = Date.now();
-112
View File
@@ -1,112 +0,0 @@
/**
* Shared AST utilities for the embedding pipeline.
* Centralizes parser caching and tree-sitter node lookups
* used by both chunker.ts and structural-extractor.ts.
*/
import { getLanguageFromFilename } from 'gitnexus-shared';
import {
createParserForLanguage,
isLanguageAvailable,
resolveLanguageKey,
} from '../tree-sitter/parser-loader.js';
const parserCache = new Map<string, any>();
/**
* Ensure parser is initialized and language is loaded, then parse content.
* Returns null if language is unavailable or parsing fails.
*/
export const ensureAndParse = async (content: string, filePath: string): Promise<any | null> => {
const language = getLanguageFromFilename(filePath);
if (!language) return null;
if (!isLanguageAvailable(language)) return null;
const parserKey = resolveLanguageKey(language, filePath);
let parserInstance = parserCache.get(parserKey);
if (!parserInstance) {
parserInstance = await createParserForLanguage(language, filePath);
parserCache.set(parserKey, parserInstance);
}
return parserInstance.parse(content);
};
const FUNCTION_LIKE_TYPES = new Set([
'function_declaration',
'function_definition',
'method_declaration',
'method_definition',
'function_item',
'function_signature_item',
'arrow_function',
'function_expression',
'generator_function_declaration',
'generator_function',
'async_function_declaration',
'async_arrow_function',
'constructor_declaration',
'constructor_definition',
'compact_constructor_declaration',
'short_function_declaration',
'proc_declaration',
'func_literal',
'local_function_statement',
'anonymous_function',
'lambda_literal',
'init_declaration',
'deinit_declaration',
]);
/**
* Find the first function/method-like declaration in a snippet AST.
* Used by the chunker when parsing node.content where absolute line
* numbers don't apply.
*/
export const findFunctionNode = (root: any): any | null => {
if (FUNCTION_LIKE_TYPES.has(root.type)) return root;
for (let i = 0; i < root.namedChildCount; i++) {
const child = root.namedChild(i);
if (!child) continue;
if (FUNCTION_LIKE_TYPES.has(child.type)) return child;
const found = findFunctionNode(child);
if (found) return found;
}
return null;
};
/**
* Find the first class/struct/interface/enum-like declaration in an AST.
* Used when parsing node.content (a snippet, not a full file) where
* absolute line numbers don't apply.
*/
export const findDeclarationNode = (root: any): any | null => {
const CLASS_LIKE_TYPES = new Set([
'class_declaration',
'class_definition',
'struct_declaration',
'struct_item',
'interface_declaration',
'interface_definition',
'enum_declaration',
'enum_item',
'type_declaration', // Go: type X struct
'declaration', // Go: type X struct
'object_declaration', // Kotlin: object
'impl_item', // Rust: impl
]);
if (CLASS_LIKE_TYPES.has(root.type)) return root;
for (let i = 0; i < root.namedChildCount; i++) {
const child = root.namedChild(i);
if (!child) continue;
if (CLASS_LIKE_TYPES.has(child.type)) return child;
const found = findDeclarationNode(child);
if (found) return found;
}
return null;
};
@@ -1,63 +0,0 @@
/**
* Character-based sliding window chunking (pure, no tree-sitter dependency)
*/
import { buildLineIndex, resolveChunkLines } from './line-index.js';
export interface Chunk {
text: string;
chunkIndex: number;
startOffset: number;
endOffset: number;
startLine: number;
endLine: number;
}
export const characterChunk = (
content: string,
startLine: number,
endLine: number,
chunkSize: number = 1200,
overlap: number = 120,
): Chunk[] => {
if (content.length <= chunkSize) {
return [
{
text: content,
chunkIndex: 0,
startOffset: 0,
endOffset: content.length,
startLine,
endLine,
},
];
}
const chunks: Chunk[] = [];
let offset = 0;
const lineOffsets = buildLineIndex(content);
while (offset < content.length) {
const end = Math.min(offset + chunkSize, content.length);
const chunkText = content.slice(offset, end);
const lineRange = resolveChunkLines(lineOffsets, offset, end, startLine);
chunks.push({
text: chunkText,
chunkIndex: chunks.length,
startOffset: offset,
endOffset: end,
startLine: lineRange.startLine,
endLine: lineRange.endLine,
});
offset = end - overlap;
if (offset >= content.length) break;
if (end >= content.length) break;
if (offset <= (chunks.length > 1 ? end - chunkSize : 0)) {
offset = end;
}
}
return chunks;
};
-363
View File
@@ -1,363 +0,0 @@
/**
* Chunker Module
*
* Splits code nodes into chunks for embedding.
* - Function/Method: AST-aware chunking by statement boundaries
* - Other types: character-based sliding window fallback
* - Short content (≤ chunkSize): no chunking
*/
export { type Chunk, characterChunk } from './character-chunk.js';
import { characterChunk } from './character-chunk.js';
import type { Chunk } from './character-chunk.js';
import { ensureAndParse, findDeclarationNode, findFunctionNode } from './ast-utils.js';
import { buildLineIndex, resolveChunkLines } from './line-index.js';
/**
* Main chunkNode function: dispatches by label
*/
export const chunkNode = async (
label: string,
content: string,
filePath: string,
startLine: number,
endLine: number,
chunkSize: number = 1200,
overlap: number = 120,
): Promise<Chunk[]> => {
// Content fits in one chunk — no splitting needed
if (content.length <= chunkSize) {
return [
{
text: content,
chunkIndex: 0,
startOffset: 0,
endOffset: content.length,
startLine,
endLine,
},
];
}
// Only function-like labels get AST chunking
if (label === 'Function' || label === 'Method' || label === 'Constructor') {
try {
const astChunks = await astChunk(content, filePath, startLine, endLine, chunkSize, overlap);
if (astChunks.length > 0) return astChunks;
} catch {
// AST parsing failed — fall through to character fallback
}
}
if (label === 'Class' || label === 'Interface') {
try {
const declarationChunks = await declarationChunk(
label,
content,
filePath,
startLine,
endLine,
chunkSize,
overlap,
);
if (declarationChunks.length > 0) return declarationChunks;
} catch {
// AST parsing failed — fall through to character fallback
}
}
// Character-based fallback for everything else
return characterChunk(content, startLine, endLine, chunkSize, overlap);
};
/**
* AST-based chunking for Function/Method
* Parse snippet content, locate the function declaration node,
* split body by statement boundaries.
*/
const astChunk = async (
content: string,
filePath: string,
startLine: number,
endLine: number,
chunkSize: number,
overlap: number,
): Promise<Chunk[]> => {
const tree = await ensureAndParse(content, filePath);
if (!tree) return [];
const root = tree.rootNode;
const lineOffsets = buildLineIndex(content);
// Find the function/method declaration in the snippet AST.
// tree-sitter parses node.content (a snippet), so rows are relative (0-based).
const targetNode = findFunctionNode(root);
if (!targetNode) return [];
// Get the body (statements) via childForFieldName('body')
const bodyNode = targetNode.childForFieldName('body');
if (!bodyNode) return [];
// Extract individual statements
const statements: Array<{ startIndex: number; endIndex: number }> = [];
for (let i = 0; i < bodyNode.namedChildCount; i++) {
const child = bodyNode.namedChild(i);
if (!child) continue;
statements.push({
startIndex: child.startIndex,
endIndex: child.endIndex,
});
}
if (statements.length === 0) return [];
return chunkByUnits(
content,
lineOffsets,
startLine,
chunkSize,
overlap,
statements,
targetNode.startIndex,
targetNode.endIndex,
true,
true,
);
};
const DECLARATION_BODY_NODE_TYPES = new Set([
'class_body',
'object_type',
'declaration_list',
'interface_body',
]);
const FIELD_LIKE_MEMBER_TYPES = new Set([
'field_definition',
'public_field_definition',
'property_definition',
'property_signature',
'variable_declarator',
'lexical_declaration',
'pair',
'enum_assignment',
]);
const declarationChunk = async (
label: 'Class' | 'Interface',
content: string,
filePath: string,
startLine: number,
endLine: number,
chunkSize: number,
overlap: number,
): Promise<Chunk[]> => {
const tree = await ensureAndParse(content, filePath);
if (!tree) return [];
const targetNode = findDeclarationNode(tree.rootNode);
if (!targetNode) return [];
const bodyNode = getDeclarationBodyNode(targetNode);
if (!bodyNode) return [];
const members = collectDeclarationUnits(bodyNode, label);
if (members.length === 0) return [];
return chunkByUnits(
content,
buildLineIndex(content),
startLine,
chunkSize,
overlap,
members,
targetNode.startIndex,
targetNode.endIndex,
false,
false,
);
};
const buildChunk = (
content: string,
lineOffsets: Int32Array,
chunkIndex: number,
startOffset: number,
endOffset: number,
baseStartLine: number,
): Chunk => {
const lineRange = resolveChunkLines(lineOffsets, startOffset, endOffset, baseStartLine);
return {
text: content.slice(startOffset, endOffset),
chunkIndex,
startOffset,
endOffset,
startLine: lineRange.startLine,
endLine: lineRange.endLine,
};
};
const chunkByUnits = (
content: string,
lineOffsets: Int32Array,
baseStartLine: number,
chunkSize: number,
overlap: number,
units: Array<{ startIndex: number; endIndex: number }>,
containerStartOffset: number,
containerEndOffset: number,
includeContainerPrefixOnFirstChunk: boolean,
includeContainerSuffixOnLastChunk: boolean,
): Chunk[] => {
const chunks: Chunk[] = [];
let chunkStartUnitIdx = 0;
while (chunkStartUnitIdx < units.length) {
const chunkStartOffset =
chunkStartUnitIdx === 0 && includeContainerPrefixOnFirstChunk
? containerStartOffset
: units[chunkStartUnitIdx].startIndex;
let chunkEndUnitIdx = chunkStartUnitIdx;
let candidateEndOffset =
chunkEndUnitIdx === units.length - 1 && includeContainerSuffixOnLastChunk
? containerEndOffset
: units[chunkEndUnitIdx].endIndex;
while (chunkEndUnitIdx + 1 < units.length) {
const nextEndOffset =
chunkEndUnitIdx + 1 === units.length - 1 && includeContainerSuffixOnLastChunk
? containerEndOffset
: units[chunkEndUnitIdx + 1].endIndex;
if (nextEndOffset - chunkStartOffset > chunkSize) break;
chunkEndUnitIdx += 1;
candidateEndOffset = nextEndOffset;
}
if (candidateEndOffset - chunkStartOffset > chunkSize) {
const oversizedUnit = units[chunkStartUnitIdx];
const oversizedLineRange = resolveChunkLines(
lineOffsets,
oversizedUnit.startIndex,
oversizedUnit.endIndex,
baseStartLine,
);
const oversizedChunks = characterChunk(
content.slice(oversizedUnit.startIndex, oversizedUnit.endIndex),
oversizedLineRange.startLine,
oversizedLineRange.endLine,
chunkSize,
overlap,
).map((chunk, offsetIdx) => ({
...chunk,
chunkIndex: chunks.length + offsetIdx,
startOffset: chunk.startOffset + oversizedUnit.startIndex,
endOffset: chunk.endOffset + oversizedUnit.startIndex,
}));
chunks.push(...oversizedChunks);
chunkStartUnitIdx += 1;
continue;
}
chunks.push(
buildChunk(
content,
lineOffsets,
chunks.length,
chunkStartOffset,
candidateEndOffset,
baseStartLine,
),
);
if (chunkEndUnitIdx === units.length - 1) {
break;
}
const nextChunkStartUnitIdx = findOverlapStartIndex(
units,
chunkStartUnitIdx,
chunkEndUnitIdx,
overlap,
);
if (nextChunkStartUnitIdx <= chunkStartUnitIdx) {
chunkStartUnitIdx = chunkEndUnitIdx + 1;
} else {
chunkStartUnitIdx = nextChunkStartUnitIdx;
}
}
return chunks;
};
const findOverlapStartIndex = (
statements: Array<{ startIndex: number; endIndex: number }>,
chunkStartStmtIdx: number,
chunkEndStmtIdx: number,
overlapSize: number,
): number => {
if (overlapSize <= 0) return chunkEndStmtIdx + 1;
let overlapStartIdx = chunkEndStmtIdx;
while (overlapStartIdx > chunkStartStmtIdx) {
const overlapLength =
statements[chunkEndStmtIdx].endIndex - statements[overlapStartIdx - 1].startIndex;
if (overlapLength > overlapSize) break;
overlapStartIdx -= 1;
}
return overlapStartIdx;
};
const getDeclarationBodyNode = (node: any): any | null => {
const bodyNode = node.childForFieldName?.('body');
if (bodyNode) return bodyNode;
for (let i = 0; i < node.namedChildCount; i++) {
const child = node.namedChild(i);
if (!child) continue;
if (DECLARATION_BODY_NODE_TYPES.has(child.type)) return child;
}
return null;
};
const collectDeclarationUnits = (
bodyNode: any,
label: 'Class' | 'Interface',
): Array<{ startIndex: number; endIndex: number }> => {
const members: Array<{ startIndex: number; endIndex: number; groupable: boolean }> = [];
for (let i = 0; i < bodyNode.namedChildCount; i++) {
const child = bodyNode.namedChild(i);
if (!child) continue;
members.push({
startIndex: child.startIndex,
endIndex: child.endIndex,
groupable: label === 'Class' && FIELD_LIKE_MEMBER_TYPES.has(child.type),
});
}
if (members.length === 0) return [];
const grouped: Array<{ startIndex: number; endIndex: number }> = [];
let current = members[0];
for (let i = 1; i < members.length; i++) {
const next = members[i];
if (current.groupable && next.groupable) {
current = {
startIndex: current.startIndex,
endIndex: next.endIndex,
groupable: true,
};
continue;
}
grouped.push({ startIndex: current.startIndex, endIndex: current.endIndex });
current = next;
}
grouped.push({ startIndex: current.startIndex, endIndex: current.endIndex });
return grouped;
};
-5
View File
@@ -157,11 +157,6 @@ export const initEmbedder = async (
try {
// Configure transformers.js environment
env.allowLocalModels = false;
// Default cache to user-writable location. transformers.js defaults to
// ./node_modules/.cache inside its own install dir, which is unwritable
// when gitnexus is installed globally (e.g. /usr/lib/node_modules/).
// Respect HF_HOME if set, otherwise fall back to ~/.cache/huggingface.
env.cacheDir = process.env.HF_HOME ?? `${process.env.HOME}/.cache/huggingface`;
const isDev = process.env.NODE_ENV === 'development';
if (isDev) {
+125 -305
View File
@@ -3,13 +3,12 @@
*
* Orchestrates the background embedding process:
* 1. Query embeddable nodes from LadybugDB
* 2. Generate text representations with enriched metadata
* 3. Chunk long nodes, batch embed
* 4. Update LadybugDB with chunk-aware embeddings
* 2. Generate text representations
* 3. Batch embed using transformers.js
* 4. Update LadybugDB with embeddings
* 5. Create vector index for semantic search
*/
import { createHash } from 'crypto';
import {
initEmbedder,
embedBatch,
@@ -17,54 +16,19 @@ import {
embeddingToArray,
isEmbedderReady,
} from './embedder.js';
import { generateEmbeddingText } from './text-generator.js';
import { chunkNode, characterChunk } from './chunker.js';
import { extractStructuralNames } from './structural-extractor.js';
import { generateBatchEmbeddingTexts } from './text-generator.js';
import {
type EmbeddingProgress,
type EmbeddingConfig,
type EmbeddableNode,
type SemanticSearchResult,
type ModelProgress,
type EmbeddingContext,
DEFAULT_EMBEDDING_CONFIG,
EMBEDDABLE_LABELS,
isShortLabel,
LABELS_WITH_EXPORTED,
STRUCTURAL_LABELS,
collectBestChunks,
} from './types.js';
import {
EMBEDDING_TABLE_NAME,
EMBEDDING_INDEX_NAME,
CREATE_VECTOR_INDEX_QUERY,
STALE_HASH_SENTINEL,
} from '../lbug/schema.js';
import { loadVectorExtension } from '../lbug/lbug-adapter.js';
const isDev = process.env.NODE_ENV === 'development';
/**
* Compute a stable content fingerprint for an embeddable node.
* Used to detect when the underlying text has changed so stale vectors
* can be replaced (DELETE-then-INSERT, the Kuzu-sanctioned pattern for
* vector-indexed rows).
*/
export const contentHashForNode = (
node: EmbeddableNode,
config: Partial<EmbeddingConfig> = {},
): string => {
// Hash must be deterministic across runs, so exclude methodNames/fieldNames
// which are populated during the batch loop via AST extraction.
// Using only node.content ensures the hash stays stable.
const text = generateEmbeddingText(
{ ...node, methodNames: undefined, fieldNames: undefined },
node.content,
config,
);
return createHash('sha1').update(text).digest('hex');
};
/**
* Progress callback type
*/
@@ -72,50 +36,37 @@ export type EmbeddingProgressCallback = (progress: EmbeddingProgress) => void;
/**
* Query all embeddable nodes from LadybugDB
* Uses table-specific queries for different label types
* Uses table-specific queries (File has different schema than code elements)
*/
const queryEmbeddableNodes = async (
executeQuery: (cypher: string) => Promise<any[]>,
): Promise<EmbeddableNode[]> => {
const allNodes: EmbeddableNode[] = [];
// Query each embeddable table with table-specific columns
for (const label of EMBEDDABLE_LABELS) {
try {
let query: string;
if (label === 'Method') {
// Method has parameterCount and returnType
if (label === 'File') {
// File nodes don't have startLine/endLine
query = `
MATCH (n:Method)
RETURN n.id AS id, n.name AS name, 'Method' AS label,
n.filePath AS filePath, n.content AS content,
n.startLine AS startLine, n.endLine AS endLine,
n.isExported AS isExported, n.description AS description,
n.parameterCount AS parameterCount, n.returnType AS returnType
`;
} else if (LABELS_WITH_EXPORTED.has(label)) {
// Function, Class, Interface have isExported and description
query = `
MATCH (n:\`${label}\`)
RETURN n.id AS id, n.name AS name, '${label}' AS label,
n.filePath AS filePath, n.content AS content,
n.startLine AS startLine, n.endLine AS endLine,
n.isExported AS isExported, n.description AS description
MATCH (n:File)
RETURN n.id AS id, n.name AS name, 'File' AS label,
n.filePath AS filePath, n.content AS content
`;
} else {
// Multi-language tables (Struct, Enum, etc.) — have description but no isExported
// Code elements have startLine/endLine
query = `
MATCH (n:\`${label}\`)
RETURN n.id AS id, n.name AS name, '${label}' AS label,
MATCH (n:${label})
RETURN n.id AS id, n.name AS name, '${label}' AS label,
n.filePath AS filePath, n.content AS content,
n.startLine AS startLine, n.endLine AS endLine,
n.description AS description
n.startLine AS startLine, n.endLine AS endLine
`;
}
const rows = await executeQuery(query);
for (const row of rows) {
const hasExportedColumn = label === 'Method' || LABELS_WITH_EXPORTED.has(label);
allNodes.push({
id: row.id ?? row[0],
name: row.name ?? row[1],
@@ -124,17 +75,10 @@ const queryEmbeddableNodes = async (
content: row.content ?? row[4] ?? '',
startLine: row.startLine ?? row[5],
endLine: row.endLine ?? row[6],
isExported: hasExportedColumn ? (row.isExported ?? row[7]) : undefined,
description: row.description ?? (hasExportedColumn ? row[8] : row[7]),
...(label === 'Method'
? {
parameterCount: row.parameterCount ?? row[9],
returnType: row.returnType ?? row[10],
}
: {}),
});
}
} catch (error) {
// Table might not exist or be empty, continue
if (isDev) {
console.warn(`Query for ${label} nodes failed:`, error);
}
@@ -145,52 +89,52 @@ const queryEmbeddableNodes = async (
};
/**
* Batch INSERT chunk-aware embeddings into CodeEmbedding table
* Batch INSERT embeddings into separate CodeEmbedding table
* Using a separate lightweight table avoids copy-on-write overhead
* that occurs when UPDATEing nodes with large content fields
*/
export const batchInsertEmbeddings = async (
const batchInsertEmbeddings = async (
executeWithReusedStatement: (
cypher: string,
paramsList: Array<Record<string, any>>,
) => Promise<void>,
updates: Array<{
nodeId: string;
chunkIndex: number;
startLine: number;
endLine: number;
embedding: number[];
contentHash?: string;
}>,
updates: Array<{ id: string; embedding: number[] }>,
): Promise<void> => {
const cypher = `CREATE (e:${EMBEDDING_TABLE_NAME} {id: $id, nodeId: $nodeId, chunkIndex: $chunkIndex, startLine: $startLine, endLine: $endLine, embedding: $embedding, contentHash: $contentHash})`;
const paramsList = updates.map((u) => ({
id: `${u.nodeId}:${u.chunkIndex}`,
nodeId: u.nodeId,
chunkIndex: u.chunkIndex,
startLine: u.startLine,
endLine: u.endLine,
embedding: u.embedding,
contentHash: u.contentHash ?? STALE_HASH_SENTINEL,
}));
// MERGE instead of CREATE — idempotent, handles concurrent analyzes and partial prior runs
const cypher = `MERGE (e:CodeEmbedding {nodeId: $nodeId}) SET e.embedding = $embedding`;
const paramsList = updates.map((u) => ({ nodeId: u.id, embedding: u.embedding }));
await executeWithReusedStatement(cypher, paramsList);
};
/**
* Create the vector index for semantic search
* Now indexes the separate CodeEmbedding table.
* Delegates extension loading to lbug-adapter's loadVectorExtension(),
* which owns the VECTOR extension lifecycle and state tracking.
* Now indexes the separate CodeEmbedding table
*/
let vectorExtensionLoaded = false;
const createVectorIndex = async (
executeQuery: (cypher: string) => Promise<any[]>,
): Promise<void> => {
// Delegate to the adapter which tracks loaded state and handles DB reconnect resets
await loadVectorExtension();
// LadybugDB v0.15+ requires explicit VECTOR extension loading (once per session)
if (!vectorExtensionLoaded) {
try {
await executeQuery('INSTALL VECTOR');
await executeQuery('LOAD EXTENSION VECTOR');
vectorExtensionLoaded = true;
} catch {
// Extension may already be loaded — CREATE_VECTOR_INDEX will fail clearly if not
vectorExtensionLoaded = true;
}
}
const cypher = `
CALL CREATE_VECTOR_INDEX('CodeEmbedding', 'code_embedding_idx', 'embedding', metric := 'cosine')
`;
try {
await executeQuery(CREATE_VECTOR_INDEX_QUERY);
await executeQuery(cypher);
} catch (error) {
// Index might already exist
if (isDev) {
console.warn('Vector index creation warning:', error);
}
@@ -205,11 +149,6 @@ const createVectorIndex = async (
* @param onProgress - Callback for progress updates
* @param config - Optional configuration override
* @param skipNodeIds - Optional set of node IDs that already have embeddings (incremental mode)
* @param context - Optional repo/server context for metadata enrichment
* @param existingEmbeddings - Optional map of nodeId → contentHash for incremental mode.
* Nodes whose hash matches are skipped; nodes with a changed hash are DELETE'd
* and re-embedded; nodes not in the map are embedded fresh.
*/
export const runEmbeddingPipeline = async (
executeQuery: (cypher: string) => Promise<any[]>,
@@ -220,8 +159,6 @@ export const runEmbeddingPipeline = async (
onProgress: EmbeddingProgressCallback,
config: Partial<EmbeddingConfig> = {},
skipNodeIds?: Set<string>,
context?: EmbeddingContext,
existingEmbeddings?: Map<string, string>,
): Promise<void> => {
const finalConfig = { ...DEFAULT_EMBEDDING_CONFIG, ...config };
@@ -257,65 +194,13 @@ export const runEmbeddingPipeline = async (
// Phase 2: Query embeddable nodes
let nodes = await queryEmbeddableNodes(executeQuery);
// Apply context metadata
if (context?.repoName) {
for (const node of nodes) {
node.repoName = context.repoName;
node.serverName = context.serverName;
}
}
// Incremental mode: compare content hashes, delete stale rows, skip fresh ones.
// Computed hashes for stale nodes are cached so batchInsertEmbeddings can reuse them
// (avoids double computation).
const computedStaleHashes = new Map<string, string>();
if (existingEmbeddings && existingEmbeddings.size > 0) {
// Incremental mode: filter out nodes that already have embeddings
if (skipNodeIds && skipNodeIds.size > 0) {
const beforeCount = nodes.length;
const staleNodeIds: string[] = [];
nodes = nodes.filter((n) => {
const existingHash = existingEmbeddings.get(n.id);
if (existingHash === undefined) {
// New node — needs embedding
return true;
}
const currentHash = contentHashForNode(n, finalConfig);
if (currentHash !== existingHash) {
// Content changed — cache hash for reuse during insert, mark for DELETE + re-embed
computedStaleHashes.set(n.id, currentHash);
staleNodeIds.push(n.id);
return true;
}
// Hash matches — skip (fresh); no need to cache hash for skipped nodes
return false;
});
// DELETE stale embedding rows so they can be re-inserted
// (Kuzu forbids SET on vector-indexed properties; DELETE-then-INSERT is the sanctioned pattern)
if (staleNodeIds.length > 0) {
if (isDev) {
console.log(`🔄 Deleting ${staleNodeIds.length} stale embedding rows for re-embed`);
}
try {
await executeWithReusedStatement(
`MATCH (e:${EMBEDDING_TABLE_NAME} {nodeId: $nodeId}) DELETE e`,
staleNodeIds.map((nodeId) => ({ nodeId })),
);
} catch (err) {
// "does not exist" = rows already gone — safe to proceed.
// All other errors risk vector-index corruption (Kuzu requires DELETE-before-INSERT
// for vector-indexed properties) — propagate so the pipeline aborts cleanly.
const msg = err instanceof Error ? err.message : String(err);
if (!msg.includes('does not exist')) {
throw new Error(
`[embed] Failed to delete stale embedding rows — aborting to prevent vector-index corruption: ${msg}`,
);
}
}
}
nodes = nodes.filter((n) => !skipNodeIds.has(n.id));
if (isDev) {
console.log(
`📦 Incremental embeddings: ${beforeCount} total, ${existingEmbeddings.size} cached, ${staleNodeIds.length} stale, ${nodes.length} to embed`,
`📦 Incremental embeddings: ${beforeCount} total, ${skipNodeIds.size} cached, ${nodes.length} to embed`,
);
}
}
@@ -327,11 +212,6 @@ export const runEmbeddingPipeline = async (
}
if (totalNodes === 0) {
// Ensure the vector index exists even when no new nodes need embedding.
// A prior crash or first-time incremental run may have left CodeEmbedding
// rows without ever reaching index creation.
await createVectorIndex(executeQuery);
onProgress({
phase: 'ready',
percent: 100,
@@ -341,12 +221,10 @@ export const runEmbeddingPipeline = async (
return;
}
// Phase 3: Chunk + embed nodes
// Phase 3: Batch embed nodes
const batchSize = finalConfig.batchSize;
const chunkSize = finalConfig.chunkSize;
const overlap = finalConfig.overlap;
const totalBatches = Math.ceil(totalNodes / batchSize);
let processedNodes = 0;
let totalChunks = 0;
onProgress({
phase: 'embedding',
@@ -354,116 +232,39 @@ export const runEmbeddingPipeline = async (
nodesProcessed: 0,
totalNodes,
currentBatch: 0,
totalBatches: Math.ceil(totalNodes / batchSize),
totalBatches,
});
// Process in batches of nodes
for (let batchIndex = 0; batchIndex < totalNodes; batchIndex += batchSize) {
const batch = nodes.slice(batchIndex, batchIndex + batchSize);
for (let batchIndex = 0; batchIndex < totalBatches; batchIndex++) {
const start = batchIndex * batchSize;
const end = Math.min(start + batchSize, totalNodes);
const batch = nodes.slice(start, end);
// Chunk each node and generate text
const allTexts: string[] = [];
const allUpdates: Array<{
nodeId: string;
chunkIndex: number;
startLine: number;
endLine: number;
contentHash: string;
}> = [];
// Generate texts for this batch
const texts = generateBatchEmbeddingTexts(batch, finalConfig);
for (const node of batch) {
const isShort = isShortLabel(node.label);
const startLine = node.startLine ?? 0;
const endLine = node.endLine ?? 0;
// Embed the batch
const embeddings = await embedBatch(texts);
// Extract structural names for class-like nodes via AST extractors
if (!isShort && STRUCTURAL_LABELS.has(node.label)) {
try {
const names = await extractStructuralNames(node.content, node.filePath);
node.methodNames = names.methodNames;
node.fieldNames = names.fieldNames;
} catch {
// AST extraction failed — names stay undefined, text-generator handles gracefully
}
}
// Update LadybugDB with embeddings
const updates = batch.map((node, i) => ({
id: node.id,
embedding: embeddingToArray(embeddings[i]),
}));
// Compute content hash once per node (re-use cached value for stale nodes)
const hash = computedStaleHashes.get(node.id) ?? contentHashForNode(node, finalConfig);
let chunks: Array<{ text: string; chunkIndex: number; startLine: number; endLine: number }>;
if (isShort) {
chunks = [{ text: node.content, chunkIndex: 0, startLine, endLine }];
} else {
try {
chunks = await chunkNode(
node.label,
node.content,
node.filePath,
startLine,
endLine,
chunkSize,
overlap,
);
} catch (chunkErr) {
if (isDev) {
console.warn(
`⚠️ AST chunking failed for ${node.label} "${node.name}" (${node.filePath}), falling back to character-based chunking:`,
chunkErr,
);
}
chunks = characterChunk(node.content, startLine, endLine, chunkSize, overlap);
}
}
for (const chunk of chunks) {
const text = generateEmbeddingText(node, chunk.text, finalConfig);
allTexts.push(text);
allUpdates.push({
nodeId: node.id,
chunkIndex: chunk.chunkIndex,
startLine: chunk.startLine,
endLine: chunk.endLine,
contentHash: hash,
});
}
}
// Embed chunk texts in sub-batches to control memory
const EMBED_SUB_BATCH = 8;
for (let si = 0; si < allTexts.length; si += EMBED_SUB_BATCH) {
const subTexts = allTexts.slice(si, si + EMBED_SUB_BATCH);
const subUpdates = allUpdates.slice(si, si + EMBED_SUB_BATCH);
let embeddings: Float32Array[];
try {
embeddings = await embedBatch(subTexts);
} catch (embedErr) {
console.error(
`❌ embedBatch failed for ${subTexts.length} texts (first: "${subTexts[0]?.substring(0, 80)}..."):`,
embedErr,
);
throw embedErr;
}
const dbUpdates = subUpdates.map((u, i) => ({
...u,
embedding: embeddingToArray(embeddings[i]),
}));
await batchInsertEmbeddings(executeWithReusedStatement, dbUpdates);
}
await batchInsertEmbeddings(executeWithReusedStatement, updates);
processedNodes += batch.length;
totalChunks += allUpdates.length;
// Report progress (20-90% for embedding phase)
const embeddingProgress = 20 + (processedNodes / totalNodes) * 70;
onProgress({
phase: 'embedding',
percent: Math.round(embeddingProgress),
nodesProcessed: processedNodes,
totalNodes,
currentBatch: Math.floor(batchIndex / batchSize) + 1,
totalBatches: Math.ceil(totalNodes / batchSize),
currentBatch: batchIndex + 1,
totalBatches,
});
}
@@ -481,6 +282,7 @@ export const runEmbeddingPipeline = async (
await createVectorIndex(executeQuery);
// Complete
onProgress({
phase: 'ready',
percent: 100,
@@ -489,9 +291,7 @@ export const runEmbeddingPipeline = async (
});
if (isDev) {
console.log(
`✅ Embedding pipeline complete! (${totalChunks} chunks from ${totalNodes} nodes)`,
);
console.log('✅ Embedding pipeline complete!');
}
} catch (error) {
const errorMessage = error instanceof Error ? error.message : 'Unknown error';
@@ -511,7 +311,15 @@ export const runEmbeddingPipeline = async (
};
/**
* Perform semantic search using the vector index with chunk deduplication
* Perform semantic search using the vector index
*
* Uses CodeEmbedding table and queries each node table to get metadata
*
* @param executeQuery - Function to execute Cypher queries
* @param query - Search query text
* @param k - Number of results to return (default: 10)
* @param maxDistance - Maximum distance threshold (default: 0.5)
* @returns Array of search results ordered by relevance
*/
export const semanticSearch = async (
executeQuery: (cypher: string) => Promise<any[]>,
@@ -523,46 +331,37 @@ export const semanticSearch = async (
throw new Error('Embedding model not initialized. Run embedding pipeline first.');
}
// Embed the query
const queryEmbedding = await embedText(query);
const queryVec = embeddingToArray(queryEmbedding);
const queryVecStr = `[${queryVec.join(',')}]`;
const bestChunks = await collectBestChunks(k, async (fetchLimit) => {
const vectorQuery = `
CALL QUERY_VECTOR_INDEX('${EMBEDDING_TABLE_NAME}', '${EMBEDDING_INDEX_NAME}',
CAST(${queryVecStr} AS FLOAT[${queryVec.length}]), ${fetchLimit})
YIELD node AS emb, distance
WITH emb, distance
WHERE distance < ${maxDistance}
RETURN emb.nodeId AS nodeId, emb.chunkIndex AS chunkIndex,
emb.startLine AS startLine, emb.endLine AS endLine, distance
ORDER BY distance
`;
// Query the vector index on CodeEmbedding to get nodeIds and distances
const vectorQuery = `
CALL QUERY_VECTOR_INDEX('CodeEmbedding', 'code_embedding_idx',
CAST(${queryVecStr} AS FLOAT[${queryVec.length}]), ${k})
YIELD node AS emb, distance
WITH emb, distance
WHERE distance < ${maxDistance}
RETURN emb.nodeId AS nodeId, distance
ORDER BY distance
`;
const embResults = await executeQuery(vectorQuery);
return embResults.map((row) => ({
nodeId: row.nodeId ?? row[0],
chunkIndex: row.chunkIndex ?? row[1] ?? 0,
startLine: row.startLine ?? row[2] ?? 0,
endLine: row.endLine ?? row[3] ?? 0,
distance: row.distance ?? row[4],
}));
});
const embResults = await executeQuery(vectorQuery);
if (bestChunks.size === 0) {
if (embResults.length === 0) {
return [];
}
// Group results by label for batched metadata queries
const byLabel = new Map<
string,
Array<{ nodeId: string; distance: number } & Record<string, any>>
>();
for (const [nodeId, chunk] of Array.from(bestChunks.entries()).slice(0, k)) {
const byLabel = new Map<string, Array<{ nodeId: string; distance: number }>>();
for (const embRow of embResults) {
const nodeId = embRow.nodeId ?? embRow[0];
const distance = embRow.distance ?? embRow[1];
const labelEndIdx = nodeId.indexOf(':');
const label = labelEndIdx > 0 ? nodeId.substring(0, labelEndIdx) : 'Unknown';
if (!byLabel.has(label)) byLabel.set(label, []);
byLabel.get(label)!.push({ nodeId, ...chunk });
byLabel.get(label)!.push({ nodeId, distance });
}
// Batch-fetch metadata per label
@@ -571,11 +370,19 @@ export const semanticSearch = async (
for (const [label, items] of byLabel) {
const idList = items.map((i) => `'${i.nodeId.replace(/'/g, "''")}'`).join(', ');
try {
const nodeQuery = `
MATCH (n:\`${label}\`) WHERE n.id IN [${idList}]
RETURN n.id AS id, n.name AS name, n.filePath AS filePath,
n.startLine AS startLine, n.endLine AS endLine
`;
let nodeQuery: string;
if (label === 'File') {
nodeQuery = `
MATCH (n:File) WHERE n.id IN [${idList}]
RETURN n.id AS id, n.name AS name, n.filePath AS filePath
`;
} else {
nodeQuery = `
MATCH (n:${label}) WHERE n.id IN [${idList}]
RETURN n.id AS id, n.name AS name, n.filePath AS filePath,
n.startLine AS startLine, n.endLine AS endLine
`;
}
const nodeRows = await executeQuery(nodeQuery);
const rowMap = new Map<string, any>();
for (const row of nodeRows) {
@@ -591,8 +398,8 @@ export const semanticSearch = async (
label,
filePath: nodeRow.filePath ?? nodeRow[2] ?? '',
distance: item.distance,
startLine: item.startLine,
endLine: item.endLine,
startLine: label !== 'File' ? (nodeRow.startLine ?? nodeRow[3]) : undefined,
endLine: label !== 'File' ? (nodeRow.endLine ?? nodeRow[4]) : undefined,
});
}
}
@@ -601,6 +408,7 @@ export const semanticSearch = async (
}
}
// Re-sort by distance since batch queries may have mixed order
results.sort((a, b) => a.distance - b.distance);
return results;
@@ -608,6 +416,16 @@ export const semanticSearch = async (
/**
* Semantic search with graph expansion (flattened results)
*
* Note: With multi-table schema, graph traversal is simplified.
* Returns semantic matches with their metadata.
* For full graph traversal, use execute_vector_cypher tool directly.
*
* @param executeQuery - Function to execute Cypher queries
* @param query - Search query text
* @param k - Number of initial semantic matches (default: 5)
* @param _hops - Unused (kept for API compatibility).
* @returns Semantic matches with metadata
*/
export const semanticSearchWithContext = async (
executeQuery: (cypher: string) => Promise<any[]>,
@@ -615,6 +433,8 @@ export const semanticSearchWithContext = async (
k: number = 5,
_hops: number = 1,
): Promise<any[]> => {
// For multi-table schema, just return semantic search results
// Graph traversal is complex with separate tables - use execute_vector_cypher instead
const results = await semanticSearch(executeQuery, query, k, 0.5);
return results.map((r) => ({
@@ -1,50 +0,0 @@
export interface ResolvedLineRange {
startLine: number;
endLine: number;
}
export const buildLineIndex = (content: string): Int32Array => {
const offsets: number[] = [0];
for (let i = 0; i < content.length; i++) {
if (content.charCodeAt(i) === 10) offsets.push(i + 1);
}
return new Int32Array(offsets);
};
const clampOffset = (lineOffsets: Int32Array, charOffset: number): number => {
if (lineOffsets.length === 0) return 0;
const maxOffset = lineOffsets[lineOffsets.length - 1];
if (charOffset < 0) return 0;
if (charOffset > maxOffset) return maxOffset;
return charOffset;
};
export const lineFromOffset = (lineOffsets: Int32Array, charOffset: number): number => {
if (lineOffsets.length === 0) return 0;
const clamped = clampOffset(lineOffsets, charOffset);
let lo = 0;
let hi = lineOffsets.length - 1;
while (lo < hi) {
const mid = (lo + hi + 1) >> 1;
if (lineOffsets[mid] <= clamped) lo = mid;
else hi = mid - 1;
}
return lo;
};
export const resolveChunkLines = (
lineOffsets: Int32Array,
startOffset: number,
endOffset: number,
baseStartLine: number,
): ResolvedLineRange => {
const relativeStartLine = lineFromOffset(lineOffsets, startOffset);
const effectiveEndOffset = endOffset > startOffset ? endOffset - 1 : startOffset;
const relativeEndLine = lineFromOffset(lineOffsets, effectiveEndOffset);
return {
startLine: baseStartLine + relativeStartLine,
endLine: baseStartLine + relativeEndLine,
};
};
@@ -1,37 +0,0 @@
/**
* Server Mapping Configuration
*
* Reads ~/.gitnexus/server-mapping.json to map repo names to service names.
* Used in embedding text to enrich metadata with microservice context.
*/
import fs from 'fs/promises';
import path from 'path';
import os from 'os';
const MAPPING_FILE = path.join(os.homedir(), '.gitnexus', 'server-mapping.json');
let cachedMapping: Record<string, string> | null = null;
/**
* Read the server mapping file and return the serverName for a given repoName.
* Returns undefined if no mapping exists.
*/
export const readServerMapping = async (repoName: string): Promise<string | undefined> => {
try {
if (!cachedMapping) {
const raw = await fs.readFile(MAPPING_FILE, 'utf-8');
cachedMapping = JSON.parse(raw);
}
return cachedMapping[repoName];
} catch {
return undefined;
}
};
/**
* Clear the cached mapping (useful for testing or after file changes)
*/
export const clearServerMappingCache = (): void => {
cachedMapping = null;
};
@@ -1,89 +0,0 @@
/**
* Structural Extractor Module
*
* Reuses ingestion pipeline's AST-based MethodExtractor / FieldExtractor
* to extract method and field names for embedding text generation.
*/
import { getProviderForFile } from '../ingestion/languages/index.js';
import type { MethodExtractorContext, ExtractedMethods } from '../ingestion/method-types.js';
import type { FieldExtractorContext, ExtractedFields } from '../ingestion/field-types.js';
import type { LanguageProvider } from '../ingestion/language-provider.js';
import { buildTypeEnv } from '../ingestion/type-env.js';
import { SupportedLanguages } from 'gitnexus-shared';
import { ensureAndParse, findDeclarationNode } from './ast-utils.js';
export interface StructuralNames {
methodNames: string[];
fieldNames: string[];
}
const NOOP_SYMBOL_TABLE = {
lookupExactAll: () => [],
lookupExact: () => undefined,
lookupExactFull: () => undefined,
} as any;
/**
* Extract method and field names from a class/struct/interface node
* using the ingestion pipeline's AST extractors.
*/
export const extractStructuralNames = async (
content: string,
filePath: string,
): Promise<StructuralNames> => {
const provider = getProviderForFile(filePath);
if (!provider) return { methodNames: [], fieldNames: [] };
const tree = await ensureAndParse(content, filePath);
if (!tree) return { methodNames: [], fieldNames: [] };
// Parse node.content (a snippet) — find declaration directly, not by range
const classNode = findDeclarationNode(tree.rootNode);
if (!classNode) return { methodNames: [], fieldNames: [] };
const language = provider.id;
const methodNames = extractMethodNames(classNode, provider, filePath, language);
const fieldNames = extractFieldNames(classNode, provider, tree, filePath, language);
return { methodNames, fieldNames };
};
function extractMethodNames(
classNode: any,
provider: LanguageProvider,
filePath: string,
language: SupportedLanguages,
): string[] {
if (!provider.methodExtractor) return [];
const context: MethodExtractorContext = { filePath, language };
const result: ExtractedMethods | null = provider.methodExtractor.extract(classNode, context);
if (!result?.methods?.length) return [];
return result.methods.map((m) => m.name);
}
function extractFieldNames(
classNode: any,
provider: LanguageProvider,
tree: any,
filePath: string,
language: SupportedLanguages,
): string[] {
if (!provider.fieldExtractor) return [];
const typeEnv = buildTypeEnv(tree, language);
const context: FieldExtractorContext = {
typeEnv,
symbolTable: NOOP_SYMBOL_TABLE,
filePath,
language,
};
const result: ExtractedFields | null = provider.fieldExtractor.extract(classNode, context);
if (!result?.fields?.length) return [];
return result.fields.map((f) => f.name);
}
+141 -188
View File
@@ -1,253 +1,206 @@
/**
* Text Generator Module
*
* Generates enriched embedding text from code nodes with metadata.
* Supports chunkable labels (Function/Method with AST chunking),
* Class-specific structural text, and short-node direct embed.
*
* Method/field names for Class nodes are extracted by the ingestion
* pipeline's AST extractors and passed via node.methodNames/node.fieldNames.
* Pure functions to generate embedding text from code nodes.
* Combines node metadata with code snippets for semantic matching.
*/
import type { EmbeddableNode, EmbeddingConfig } from './types.js';
import { DEFAULT_EMBEDDING_CONFIG, isShortLabel } from './types.js';
import { DEFAULT_EMBEDDING_CONFIG } from './types.js';
/**
* Truncate description to max length at sentence/word boundary
* Extract the filename from a file path
*/
const truncateDescription = (text: string, maxLength: number): string => {
if (text.length <= maxLength) return text;
const getFileName = (filePath: string): string => {
const parts = filePath.split('/');
return parts[parts.length - 1] || filePath;
};
const truncated = text.slice(0, maxLength);
/**
* Extract the directory path from a file path
*/
const getDirectory = (filePath: string): string => {
const parts = filePath.split('/');
parts.pop();
return parts.join('/') || '';
};
// Try sentence boundary (. ! ?)
const sentenceEnd = Math.max(
truncated.lastIndexOf('. '),
truncated.lastIndexOf('! '),
truncated.lastIndexOf('? '),
);
if (sentenceEnd > maxLength * 0.5) {
return truncated.slice(0, sentenceEnd + 1);
/**
* Truncate content to max length, preserving word boundaries
*/
const truncateContent = (content: string, maxLength: number): string => {
if (content.length <= maxLength) {
return content;
}
// Try word boundary
// Find last space before maxLength to avoid cutting words
const truncated = content.slice(0, maxLength);
const lastSpace = truncated.lastIndexOf(' ');
if (lastSpace > maxLength * 0.5) {
return truncated.slice(0, lastSpace);
if (lastSpace > maxLength * 0.8) {
return truncated.slice(0, lastSpace) + '...';
}
return truncated;
return truncated + '...';
};
/**
* Clean code content for embedding
* Removes excessive whitespace while preserving structure
*/
const cleanContent = (content: string): string => {
return content
.replace(/\r\n/g, '\n')
.replace(/\n{3,}/g, '\n\n')
.split('\n')
.map((line) => line.trimEnd())
.join('\n')
.trim();
return (
content
// Normalize line endings
.replace(/\r\n/g, '\n')
// Remove excessive blank lines (more than 2)
.replace(/\n{3,}/g, '\n\n')
// Trim each line
.split('\n')
.map((line) => line.trimEnd())
.join('\n')
.trim()
);
};
/**
* Build metadata header for a node
* Generate embedding text for a Function node
*/
const buildMetadataHeader = (node: EmbeddableNode, config: Partial<EmbeddingConfig>): string => {
const parts: string[] = [];
const generateFunctionText = (node: EmbeddableNode, maxSnippetLength: number): string => {
const parts: string[] = [`Function: ${node.name}`, `File: ${getFileName(node.filePath)}`];
// Label + name
parts.push(`${node.label}: ${node.name}`);
// Repo name
if (node.repoName) {
parts.push(`Repo: ${node.repoName}`);
const dir = getDirectory(node.filePath);
if (dir) {
parts.push(`Directory: ${dir}`);
}
// Server name (optional)
if (node.serverName) {
parts.push(`Server: ${node.serverName}`);
}
// Full file path
parts.push(`Path: ${node.filePath}`);
// Export status
if (node.isExported !== undefined) {
parts.push(`Export: ${node.isExported}`);
}
// Description (truncated)
if (node.description) {
const maxLen = config.maxDescriptionLength ?? DEFAULT_EMBEDDING_CONFIG.maxDescriptionLength;
const truncated = truncateDescription(node.description, maxLen);
if (truncated) {
parts.push(truncated);
}
if (node.content) {
const cleanedContent = cleanContent(node.content);
const snippet = truncateContent(cleanedContent, maxSnippetLength);
parts.push('', snippet);
}
return parts.join('\n');
};
const generateCodeBodyText = (
node: EmbeddableNode,
codeBody: string,
config: Partial<EmbeddingConfig>,
): string => {
const header = buildMetadataHeader(node, config);
const cleaned = cleanContent(codeBody);
return `${header}\n\n${cleaned}`;
};
/**
* Generate embedding text for Class nodes
* Signature + properties + method name list only (no method bodies)
* Method/field names come from AST extractors via node.methodNames/node.fieldNames.
* Generate embedding text for a Class node
*/
const generateClassText = (
node: EmbeddableNode,
codeBody: string,
config: Partial<EmbeddingConfig>,
): string => {
return generateStructuralTypeText(node, codeBody, config);
};
const generateClassText = (node: EmbeddableNode, maxSnippetLength: number): string => {
const parts: string[] = [`Class: ${node.name}`, `File: ${getFileName(node.filePath)}`];
const generateStructuralTypeText = (
node: EmbeddableNode,
codeBody: string,
config: Partial<EmbeddingConfig>,
): string => {
const header = buildMetadataHeader(node, config);
const parts: string[] = [header];
if (node.methodNames?.length) {
parts.push(`Methods: ${node.methodNames.join(', ')}`);
}
if (node.fieldNames?.length) {
parts.push(`Properties: ${node.fieldNames.join(', ')}`);
const dir = getDirectory(node.filePath);
if (dir) {
parts.push(`Directory: ${dir}`);
}
const declarationOnly = extractDeclarationOnly(cleanContent(node.content));
if (declarationOnly) {
parts.push('', declarationOnly);
}
const cleanedChunk = cleanContent(codeBody);
if (cleanedChunk && cleanedChunk !== cleanContent(node.content)) {
parts.push('', cleanedChunk);
if (node.content) {
const cleanedContent = cleanContent(node.content);
const snippet = truncateContent(cleanedContent, maxSnippetLength);
parts.push('', snippet);
}
return parts.join('\n');
};
const DECL_START_RE =
/^(?:(?:export|pub|data|abstract)\s+)*(?:type\s+\w+\s+struct|(?:class|struct|enum|interface)\s)/;
/**
* Extract class/interface/struct declaration lines, skipping method bodies.
* - Brace-based languages: detects method signatures (lines with `(` and `{`)
* and skips until depth returns to class body level.
* - Non-brace languages (Python/Ruby): returns empty string (patterns handle extraction).
* Generate embedding text for a Method node
*/
export const extractDeclarationOnly = (content: string): string => {
const lines = content.split('\n');
const declLines: string[] = [];
let depth = 0;
let started = false;
let classDepth = 0;
let skipDepth = 0;
const generateMethodText = (node: EmbeddableNode, maxSnippetLength: number): string => {
const parts: string[] = [`Method: ${node.name}`, `File: ${getFileName(node.filePath)}`];
for (const [idx, line] of lines.entries()) {
const trimmed = line.trim();
if (!started) {
if (DECL_START_RE.test(trimmed)) {
// Non-brace language check: current line or next 3 lines must have `{`
const nextLines = lines.slice(idx + 1, idx + 4);
if (!trimmed.includes('{') && !nextLines.some((l) => l.includes('{'))) {
return '';
}
started = true;
declLines.push(trimmed);
for (const ch of trimmed) {
if (ch === '{') depth++;
else if (ch === '}') depth--;
}
if (depth > 0) classDepth = depth;
}
continue;
}
// Always update depth (even when skipping)
const opens = (trimmed.match(/{/g) || []).length;
const closes = (trimmed.match(/}/g) || []).length;
const prevDepth = depth;
depth += opens - closes;
if (skipDepth > 0) {
if (depth <= classDepth) {
skipDepth = 0;
// Closing brace of class
if (depth <= 0) {
declLines.push(trimmed);
break;
}
}
continue;
}
// Detect method signature: line has both `(` and `{` and goes deeper than class body
const hasParens = trimmed.includes('(');
const hasOpenBrace = opens > 0;
if (hasParens && hasOpenBrace && prevDepth + opens > classDepth) {
if (opens === closes && trimmed.endsWith(';')) {
// Property with function/object initializer like `config = { timeout: 5000 };` — keep
declLines.push(trimmed);
}
// else: single-line or multi-line method — skip entirely
if (opens !== closes) {
skipDepth = classDepth;
}
continue;
}
declLines.push(trimmed);
if (depth <= 0 && declLines.length > 1) break;
const dir = getDirectory(node.filePath);
if (dir) {
parts.push(`Directory: ${dir}`);
}
return declLines.join('\n').trim();
if (node.content) {
const cleanedContent = cleanContent(node.content);
const snippet = truncateContent(cleanedContent, maxSnippetLength);
parts.push('', snippet);
}
return parts.join('\n');
};
/**
* Generate embedding text for an Interface node
*/
const generateInterfaceText = (node: EmbeddableNode, maxSnippetLength: number): string => {
const parts: string[] = [`Interface: ${node.name}`, `File: ${getFileName(node.filePath)}`];
const dir = getDirectory(node.filePath);
if (dir) {
parts.push(`Directory: ${dir}`);
}
if (node.content) {
const cleanedContent = cleanContent(node.content);
const snippet = truncateContent(cleanedContent, maxSnippetLength);
parts.push('', snippet);
}
return parts.join('\n');
};
/**
* Generate embedding text for a File node
* Uses file name and first N characters of content
*/
const generateFileText = (node: EmbeddableNode, maxSnippetLength: number): string => {
const parts: string[] = [`File: ${node.name}`, `Path: ${node.filePath}`];
if (node.content) {
const cleanedContent = cleanContent(node.content);
// For files, use a shorter snippet since they can be very long
const snippet = truncateContent(cleanedContent, Math.min(maxSnippetLength, 300));
parts.push('', snippet);
}
return parts.join('\n');
};
/**
* Generate embedding text for any embeddable node
* Dispatches to the appropriate generator based on node label
*
* @param node - The node to generate text for
* @param config - Optional configuration for max snippet length
* @returns Text suitable for embedding
*/
export const generateEmbeddingText = (
node: EmbeddableNode,
codeBody: string,
config: Partial<EmbeddingConfig> = {},
): string => {
if (isShortLabel(node.label)) {
const header = buildMetadataHeader(node, config);
const cleaned = cleanContent(node.content);
return `${header}\n\n${cleaned}`;
}
const maxSnippetLength = config.maxSnippetLength ?? DEFAULT_EMBEDDING_CONFIG.maxSnippetLength;
if (node.label === 'Class') {
return generateClassText(node, codeBody, config);
switch (node.label) {
case 'Function':
return generateFunctionText(node, maxSnippetLength);
case 'Class':
return generateClassText(node, maxSnippetLength);
case 'Method':
return generateMethodText(node, maxSnippetLength);
case 'Interface':
return generateInterfaceText(node, maxSnippetLength);
case 'File':
return generateFileText(node, maxSnippetLength);
default:
// Fallback for any other embeddable type
return `${node.label}: ${node.name}\nPath: ${node.filePath}`;
}
if (node.label === 'Interface') {
return generateStructuralTypeText(node, codeBody, config);
}
return generateCodeBodyText(node, codeBody, config);
};
/**
* Export truncation helper for testing
* Generate embedding texts for a batch of nodes
*
* @param nodes - Array of nodes to generate text for
* @param config - Optional configuration
* @returns Array of texts in the same order as input nodes
*/
export { truncateDescription };
export const generateBatchEmbeddingTexts = (
nodes: EmbeddableNode[],
config: Partial<EmbeddingConfig> = {},
): string[] => {
return nodes.map((node) => generateEmbeddingText(node, config));
};
+3 -175
View File
@@ -5,40 +5,10 @@
*/
/**
* Node labels that need chunking (have code body, potentially long)
* Node labels that should be embedded for semantic search
* These are code elements that benefit from semantic matching
*/
export const CHUNKABLE_LABELS = [
'Function',
'Method',
'Constructor',
'Class',
'Interface',
'Struct',
'Enum',
'Trait',
'Impl',
'Macro',
'Namespace',
] as const;
/**
* Node labels that are short (no chunking needed, embed directly)
*/
export const SHORT_LABELS = [
'TypeAlias',
'Typedef',
'Const',
'Property',
'Record',
'Union',
'Static',
'Variable',
] as const;
/**
* All embeddable labels (union of CHUNKABLE + SHORT)
*/
export const EMBEDDABLE_LABELS = [...CHUNKABLE_LABELS, ...SHORT_LABELS] as const;
export const EMBEDDABLE_LABELS = ['Function', 'Class', 'Method', 'Interface', 'File'] as const;
export type EmbeddableLabel = (typeof EMBEDDABLE_LABELS)[number];
@@ -48,39 +18,6 @@ export type EmbeddableLabel = (typeof EMBEDDABLE_LABELS)[number];
export const isEmbeddableLabel = (label: string): label is EmbeddableLabel =>
EMBEDDABLE_LABELS.includes(label as EmbeddableLabel);
/**
* Check if a label needs chunking
*/
export const isChunkableLabel = (label: string): boolean =>
(CHUNKABLE_LABELS as readonly string[]).includes(label);
/**
* Check if a label is a short type (no chunking)
*/
export const isShortLabel = (label: string): boolean =>
(SHORT_LABELS as readonly string[]).includes(label);
/**
* Node labels that have structural names (methods/fields) extractable via AST
*/
export const STRUCTURAL_LABELS: ReadonlySet<string> = new Set([
'Class',
'Struct',
'Interface',
'Enum',
]);
/**
* Node labels that have isExported column in their schema
*/
export const LABELS_WITH_EXPORTED = new Set([
'Function',
'Class',
'Interface',
'Method',
'CodeElement',
]) as ReadonlySet<string>;
/**
* Embedding pipeline phases
*/
@@ -120,12 +57,6 @@ export interface EmbeddingConfig {
device: 'auto' | 'dml' | 'cuda' | 'cpu' | 'wasm';
/** Maximum characters of code snippet to include */
maxSnippetLength: number;
/** Maximum code chunk size in characters (for chunking long code) */
chunkSize: number;
/** Overlap between chunks in characters */
overlap: number;
/** Maximum description length in characters */
maxDescriptionLength: number;
}
/**
@@ -139,9 +70,6 @@ export const DEFAULT_EMBEDDING_CONFIG: EmbeddingConfig = {
dimensions: 384,
device: 'auto',
maxSnippetLength: 500,
chunkSize: 1200,
overlap: 120,
maxDescriptionLength: 150,
};
/**
@@ -168,34 +96,6 @@ export interface EmbeddableNode {
content: string;
startLine?: number;
endLine?: number;
isExported?: boolean;
description?: string;
parameterCount?: number;
returnType?: string;
repoName?: string;
serverName?: string;
methodNames?: string[];
fieldNames?: string[];
}
/**
* Cached embedding entry restored from LadybugDB before a graph rebuild
*/
export interface CachedEmbedding {
nodeId: string;
chunkIndex: number;
startLine: number;
endLine: number;
embedding: number[];
contentHash?: string;
}
/**
* Context info for embedding pipeline (repo/server metadata enrichment)
*/
export interface EmbeddingContext {
repoName?: string;
serverName?: string;
}
/**
@@ -208,75 +108,3 @@ export interface ModelProgress {
loaded?: number;
total?: number;
}
export interface ChunkSearchRow {
nodeId: string;
chunkIndex: number;
startLine: number;
endLine: number;
distance: number;
}
export interface BestChunkMatch {
chunkIndex: number;
startLine: number;
endLine: number;
distance: number;
}
/**
* Deduplicate vector search chunk results by nodeId,
* keeping the chunk with smallest distance for each node.
*/
export const dedupBestChunks = (
rows: ChunkSearchRow[],
limit?: number,
): Map<string, BestChunkMatch> => {
const best = new Map<string, BestChunkMatch>();
for (const row of rows) {
const existing = best.get(row.nodeId);
if (!existing || row.distance < existing.distance) {
best.set(row.nodeId, {
chunkIndex: row.chunkIndex,
startLine: row.startLine,
endLine: row.endLine,
distance: row.distance,
});
}
if (limit !== undefined && best.size >= limit) break;
}
return best;
};
const DEFAULT_FETCH_MULTIPLIER = 4;
const DEFAULT_FETCH_BUFFER = 8;
const DEFAULT_MAX_FETCH = 200;
/**
* Fetch vector-search chunks until we have enough unique nodeIds
* or can tell the result set is exhausted.
*/
export const collectBestChunks = async (
limit: number,
fetchRows: (fetchLimit: number) => Promise<ChunkSearchRow[]>,
maxFetch: number = DEFAULT_MAX_FETCH,
): Promise<Map<string, BestChunkMatch>> => {
if (limit <= 0) return new Map();
let fetchLimit = Math.max(limit * DEFAULT_FETCH_MULTIPLIER, limit + DEFAULT_FETCH_BUFFER);
let previousFetchLimit = 0;
while (fetchLimit > previousFetchLimit) {
const rows = await fetchRows(fetchLimit);
const bestChunks = dedupBestChunks(rows, limit);
if (bestChunks.size >= limit || rows.length < fetchLimit) {
return bestChunks;
}
previousFetchLimit = fetchLimit;
fetchLimit = fetchLimit >= maxFetch ? fetchLimit * 2 : Math.min(maxFetch, fetchLimit * 2);
}
return new Map();
};
@@ -17,9 +17,6 @@ import type { HttpDetection, HttpLanguagePlugin } from './types.js';
* - Express `router.get(...)` / `app.post(...)` providers
* - `fetch(url)` / `fetch(url, { method: 'POST' })` consumers
* - `axios.get(url)` / `axios.delete(url)` consumers
* - `axios({ method, url })` object-form consumers
* - jQuery `$.get(url)` / `$.post(url, ...)` shorthand consumers
* - jQuery `$.ajax({ url, method | type })` consumers
*
* Because the JavaScript and TypeScript tree-sitter grammars share
* node type names for every construct we query, pattern sources are
@@ -106,48 +103,6 @@ const AXIOS_SPEC: PatternSpec<Record<string, never>> = {
`,
};
// ─── Consumer: jQuery shorthand $.get(url) / $.post(url, ...) ────────
// `$` is a valid JS identifier, so tree-sitter parses `$.get(...)` as a
// call_expression whose function is a member_expression on identifier `$`.
const JQUERY_SHORTHAND_SPEC: PatternSpec<Record<string, never>> = {
meta: {},
query: `
(call_expression
function: (member_expression
object: (identifier) @obj (#eq? @obj "$")
property: (property_identifier) @http_method (#match? @http_method "^(get|post)$"))
arguments: (arguments . [(string) (template_string)] @path))
`,
};
// ─── Consumer: jQuery $.ajax({ url, method|type }) ───────────────────
// The query captures the options object only; key/value pairs are read
// programmatically via `readStringProp` below, which tolerates any key
// order and accepts either `method:` or `type:` (jQuery supports both).
const JQUERY_AJAX_SPEC: PatternSpec<Record<string, never>> = {
meta: {},
query: `
(call_expression
function: (member_expression
object: (identifier) @obj (#eq? @obj "$")
property: (property_identifier) @fn (#eq? @fn "ajax"))
arguments: (arguments (object) @options))
`,
};
// ─── Consumer: axios({ method, url }) object form ────────────────────
// Distinct from AXIOS_SPEC above because the call target is an identifier
// (`axios`) rather than a member expression (`axios.get`). As with the
// jQuery ajax form, option keys are resolved programmatically.
const AXIOS_OBJECT_SPEC: PatternSpec<Record<string, never>> = {
meta: {},
query: `
(call_expression
function: (identifier) @fn (#eq? @fn "axios")
arguments: (arguments (object) @options))
`,
};
interface NodePatternBundle {
controller: CompiledPatterns<Record<string, never>>;
methodDecorator: CompiledPatterns<Record<string, never>>;
@@ -155,9 +110,6 @@ interface NodePatternBundle {
fetchNoOptions: CompiledPatterns<Record<string, never>>;
fetchWithOptions: CompiledPatterns<Record<string, never>>;
axios: CompiledPatterns<Record<string, never>>;
jqueryShorthand: CompiledPatterns<Record<string, never>>;
jqueryAjax: CompiledPatterns<Record<string, never>>;
axiosObject: CompiledPatterns<Record<string, never>>;
}
function compileBundle(language: unknown, name: string): NodePatternBundle {
@@ -174,9 +126,6 @@ function compileBundle(language: unknown, name: string): NodePatternBundle {
fetchNoOptions: mk(FETCH_NO_OPTIONS_SPEC, 'fetch-no-options'),
fetchWithOptions: mk(FETCH_WITH_OPTIONS_SPEC, 'fetch-with-options'),
axios: mk(AXIOS_SPEC, 'axios'),
jqueryShorthand: mk(JQUERY_SHORTHAND_SPEC, 'jquery-shorthand'),
jqueryAjax: mk(JQUERY_AJAX_SPEC, 'jquery-ajax'),
axiosObject: mk(AXIOS_OBJECT_SPEC, 'axios-object'),
};
}
@@ -211,28 +160,6 @@ function joinPath(prefix: string, sub: string): string {
return `/${cleanPrefix}/${cleanSub}`;
}
/**
* Walk `pair` children of an `object` literal and return the unquoted
* string/template_string value for the first pair whose key matches one
* of `keyNames`. Returns null when no matching pair is present or the
* value is not a string literal. Used by the jQuery ajax / axios object
* consumers to resolve `url` / `method` / `type` keys in any order.
*/
function readStringProp(objectNode: Parser.SyntaxNode, keyNames: readonly string[]): string | null {
for (let i = 0; i < objectNode.namedChildCount; i++) {
const pair = objectNode.namedChild(i);
if (!pair || pair.type !== 'pair') continue;
const keyNode = pair.childForFieldName('key');
const valueNode = pair.childForFieldName('value');
if (!keyNode || !valueNode) continue;
if (!keyNames.includes(keyNode.text)) continue;
if (valueNode.type !== 'string' && valueNode.type !== 'template_string') continue;
const lit = unquoteLiteral(valueNode.text);
if (lit !== null) return lit;
}
return null;
}
/**
* For a standalone `decorator` node (child of class_body / program),
* find the related `class_declaration` node that it decorates. In
@@ -424,62 +351,6 @@ function scanBundle(bundle: NodePatternBundle, tree: Parser.Tree): HttpDetection
});
}
// Consumer: jQuery shorthand $.get(url) / $.post(url, ...)
for (const match of runCompiledPatterns(bundle.jqueryShorthand, tree)) {
const methodNode = match.captures.http_method;
const pathNode = match.captures.path;
if (!methodNode || !pathNode) continue;
const path = unquoteLiteral(pathNode.text);
if (path === null) continue;
out.push({
role: 'consumer',
framework: 'jquery',
method: methodNode.text.toUpperCase(),
path,
name: null,
confidence: 0.7,
});
}
// Consumer: jQuery $.ajax({ url, method|type }). jQuery accepts either
// `method:` or `type:`; both default to GET when absent.
for (const match of runCompiledPatterns(bundle.jqueryAjax, tree)) {
const optionsNode = match.captures.options;
if (!optionsNode) continue;
const path = readStringProp(optionsNode, ['url']);
if (path === null) continue;
const rawMethod = readStringProp(optionsNode, ['method', 'type']);
const method = (rawMethod ?? 'GET').toUpperCase();
out.push({
role: 'consumer',
framework: 'jquery',
method,
path,
name: null,
confidence: 0.7,
});
}
// Consumer: axios({ method, url }) object form. Structurally distinct
// from axios.<verb>(url) (identifier vs member_expression call), so no
// dedup against the member-form loop above is required.
for (const match of runCompiledPatterns(bundle.axiosObject, tree)) {
const optionsNode = match.captures.options;
if (!optionsNode) continue;
const path = readStringProp(optionsNode, ['url']);
if (path === null) continue;
const rawMethod = readStringProp(optionsNode, ['method']);
const method = (rawMethod ?? 'GET').toUpperCase();
out.push({
role: 'consumer',
framework: 'axios',
method,
path,
name: null,
confidence: 0.7,
});
}
return out;
}
@@ -79,50 +79,17 @@ export class ManifestExtractor {
links: GroupManifestLink[],
dbExecutors?: Map<string, CypherExecutor>,
): Promise<ManifestExtractResult> {
// Resolve all (repo, link) pairs in parallel. The previous sequential
// await-per-link produced 2N round-trips; parallel resolution uses the
// per-repo executor pool directly and scales linearly with manifest size.
//
// Memoization: a manifest can list the same contract multiple times
// (e.g. a consumer and provider declaration, or cross-referenced groups).
// Key on (repo, type, contract) — the canonical input to the Cypher
// query — so duplicate links resolve to one DB hit.
type ResolvedSymbol = { filePath: string; name: string; uid: string } | null;
const resolveCache = new Map<string, Promise<ResolvedSymbol>>();
const resolveOnce = (repo: string, link: GroupManifestLink): Promise<ResolvedSymbol> => {
const key = `${repo}\u0000${link.type}\u0000${link.contract}`;
let pending = resolveCache.get(key);
if (!pending) {
pending = this.resolveSymbol(repo, link, dbExecutors);
resolveCache.set(key, pending);
}
return pending;
};
const perLink = await Promise.all(
links.map(async (link) => {
const contractId = this.buildContractId(link.type, link.contract);
const providerRepo = link.role === 'provider' ? link.from : link.to;
const consumerRepo = link.role === 'provider' ? link.to : link.from;
const [providerSymbol, consumerSymbol] = await Promise.all([
resolveOnce(providerRepo, link),
resolveOnce(consumerRepo, link),
]);
return { link, contractId, providerRepo, consumerRepo, providerSymbol, consumerSymbol };
}),
);
const contracts: StoredContract[] = [];
const crossLinks: CrossLink[] = [];
for (const {
link,
contractId,
providerRepo,
consumerRepo,
providerSymbol,
consumerSymbol,
} of perLink) {
for (const link of links) {
const contractId = this.buildContractId(link.type, link.contract);
const providerRepo = link.role === 'provider' ? link.from : link.to;
const consumerRepo = link.role === 'provider' ? link.to : link.from;
const providerSymbol = await this.resolveSymbol(providerRepo, link, dbExecutors);
const consumerSymbol = await this.resolveSymbol(consumerRepo, link, dbExecutors);
const providerRef = providerSymbol || { filePath: '', name: link.contract };
const consumerRef = consumerSymbol || { filePath: '', name: link.contract };
// When the resolver finds a real graph symbol we keep its uid, otherwise
+1 -56
View File
@@ -7,7 +7,6 @@ import type { GroupConfig, RepoHandle, RepoSnapshot, StoredContract, CrossLink }
import { HttpRouteExtractor } from './extractors/http-route-extractor.js';
import { GrpcExtractor } from './extractors/grpc-extractor.js';
import { TopicExtractor } from './extractors/topic-extractor.js';
import { ManifestExtractor } from './extractors/manifest-extractor.js';
import { runExactMatch } from './matching.js';
import { detectServiceBoundaries, assignService } from './service-boundary-detector.js';
import type { CypherExecutor } from './contract-extractor.js';
@@ -61,28 +60,10 @@ function defaultResolveHandle(allEntries: RegistryEntry[]) {
};
}
/**
* Dedupe cross-links that point from the same consumer endpoint to the same
* provider endpoint for the same contract. Preserves first-seen order so the
* caller controls precedence (e.g., pass manifest links first).
*/
function dedupeCrossLinks(links: CrossLink[]): CrossLink[] {
const seen = new Set<string>();
const out: CrossLink[] = [];
for (const link of links) {
const key = `${link.from.repo}::${link.from.symbolUid}|${link.to.repo}::${link.to.symbolUid}|${link.type}|${link.contractId}`;
if (seen.has(key)) continue;
seen.add(key);
out.push(link);
}
return out;
}
export async function syncGroup(config: GroupConfig, opts?: SyncOptions): Promise<SyncResult> {
const missingRepos: string[] = [];
const repoSnapshots: Record<string, RepoSnapshot> = {};
let autoContracts: StoredContract[] = [];
let manifestCrossLinks: CrossLink[] = [];
let dbExecutors: Map<string, CypherExecutor> | undefined;
const eo = opts?.extractorOverride;
@@ -177,44 +158,8 @@ export async function syncGroup(config: GroupConfig, opts?: SyncOptions): Promis
}
}
// Process manifest links declared in group.yaml.
// ManifestExtractor is fully implemented but was never wired into this
// pipeline — config.links were parsed and validated but silently dropped.
// Placed after the DB try/finally: resolveSymbol falls back to synthetic
// UIDs when dbExecutors is undefined or a pool is closed, so cross-links
// are always generated regardless of whether real DB executors are available.
if (config.links.length > 0) {
// Warn about dangling links that reference repos not declared in config.repos.
// They still generate cross-links via synthetic UIDs (determinism is preserved),
// but the operator probably meant something that now silently does nothing useful.
const knownRepos = new Set(Object.keys(config.repos));
for (const link of config.links) {
const dangling = [link.from, link.to].filter((r) => !knownRepos.has(r));
if (dangling.length > 0) {
console.warn(
`[group/sync] manifest link ${link.type}:${link.contract} references repos not in config.repos: ${dangling.join(', ')} — cross-links will use synthetic UIDs`,
);
}
}
const manifestEx = new ManifestExtractor();
const manifestResult = await manifestEx.extractFromManifest(config.links, dbExecutors);
autoContracts.push(...manifestResult.contracts);
manifestCrossLinks = manifestResult.crossLinks;
if (opts?.verbose) {
console.log(
` manifest: ${manifestCrossLinks.length} cross-links from ${config.links.length} declared links`,
);
}
}
const { matched, unmatched } = runExactMatch(autoContracts);
// Dedupe cross-links. Manifest contracts participate in runExactMatch, so a
// manifest-declared link can also emit a matchType:'exact' CrossLink with the
// same endpoints. Prefer the manifest version — it reflects operator intent
// and carries matchType:'manifest' which downstream consumers may rely on.
const crossLinks = dedupeCrossLinks([...manifestCrossLinks, ...matched]);
const crossLinks: CrossLink[] = matched;
const allContracts: StoredContract[] = autoContracts;
const registry: ContractRegistry = {
@@ -1,12 +0,0 @@
// gitnexus/src/core/ingestion/call-extractors/configs/c-cpp.ts
import { SupportedLanguages } from 'gitnexus-shared';
import type { CallExtractionConfig } from '../../call-types.js';
export const cCallConfig: CallExtractionConfig = {
language: SupportedLanguages.C,
};
export const cppCallConfig: CallExtractionConfig = {
language: SupportedLanguages.CPlusPlus,
};
@@ -1,9 +0,0 @@
// gitnexus/src/core/ingestion/call-extractors/configs/csharp.ts
import { SupportedLanguages } from 'gitnexus-shared';
import type { CallExtractionConfig } from '../../call-types.js';
export const csharpCallConfig: CallExtractionConfig = {
language: SupportedLanguages.CSharp,
typeAsReceiverHeuristic: true,
};
@@ -1,8 +0,0 @@
// gitnexus/src/core/ingestion/call-extractors/configs/dart.ts
import { SupportedLanguages } from 'gitnexus-shared';
import type { CallExtractionConfig } from '../../call-types.js';
export const dartCallConfig: CallExtractionConfig = {
language: SupportedLanguages.Dart,
};
@@ -1,8 +0,0 @@
// gitnexus/src/core/ingestion/call-extractors/configs/go.ts
import { SupportedLanguages } from 'gitnexus-shared';
import type { CallExtractionConfig } from '../../call-types.js';
export const goCallConfig: CallExtractionConfig = {
language: SupportedLanguages.Go,
};
@@ -1,59 +0,0 @@
// gitnexus/src/core/ingestion/call-extractors/configs/jvm.ts
import { SupportedLanguages } from 'gitnexus-shared';
import type { CallExtractionConfig, ExtractedCallSite } from '../../call-types.js';
import type { SyntaxNode } from '../../utils/ast-helpers.js';
// ---------------------------------------------------------------------------
// Java method_reference (::) parsing — absorbs call-sites/java.ts
// ---------------------------------------------------------------------------
/**
* Parse Java `method_reference` nodes (`expr::method`, `Type::new`,
* `this::m`, `super::m`).
*/
function parseJavaMethodReference(callNode: SyntaxNode): ExtractedCallSite | null {
if (callNode.type !== 'method_reference') return null;
const recv = callNode.namedChild(0);
if (!recv) return null;
// Type::new → constructor call
for (const c of callNode.children) {
if (c.type === 'new') {
if (recv.type !== 'identifier') return null;
return { calledName: recv.text, callForm: 'constructor' };
}
}
// expr::method → member call with receiver
const rhs = callNode.child(callNode.childCount - 1);
if (!rhs || rhs.type !== 'identifier') return null;
const methodName = rhs.text;
if (recv.type === 'identifier') {
return { calledName: methodName, callForm: 'member', receiverName: recv.text };
}
if (recv.type === 'this') {
return { calledName: methodName, callForm: 'member', receiverName: 'this' };
}
if (recv.type === 'super') {
return { calledName: methodName, callForm: 'member', receiverName: 'super' };
}
return null;
}
// ---------------------------------------------------------------------------
// Configs
// ---------------------------------------------------------------------------
export const javaCallConfig: CallExtractionConfig = {
language: SupportedLanguages.Java,
extractLanguageCallSite: parseJavaMethodReference,
typeAsReceiverHeuristic: true,
};
export const kotlinCallConfig: CallExtractionConfig = {
language: SupportedLanguages.Kotlin,
typeAsReceiverHeuristic: true,
};
@@ -1,8 +0,0 @@
// gitnexus/src/core/ingestion/call-extractors/configs/php.ts
import { SupportedLanguages } from 'gitnexus-shared';
import type { CallExtractionConfig } from '../../call-types.js';
export const phpCallConfig: CallExtractionConfig = {
language: SupportedLanguages.PHP,
};
@@ -1,8 +0,0 @@
// gitnexus/src/core/ingestion/call-extractors/configs/python.ts
import { SupportedLanguages } from 'gitnexus-shared';
import type { CallExtractionConfig } from '../../call-types.js';
export const pythonCallConfig: CallExtractionConfig = {
language: SupportedLanguages.Python,
};
@@ -1,8 +0,0 @@
// gitnexus/src/core/ingestion/call-extractors/configs/ruby.ts
import { SupportedLanguages } from 'gitnexus-shared';
import type { CallExtractionConfig } from '../../call-types.js';
export const rubyCallConfig: CallExtractionConfig = {
language: SupportedLanguages.Ruby,
};
@@ -1,8 +0,0 @@
// gitnexus/src/core/ingestion/call-extractors/configs/rust.ts
import { SupportedLanguages } from 'gitnexus-shared';
import type { CallExtractionConfig } from '../../call-types.js';
export const rustCallConfig: CallExtractionConfig = {
language: SupportedLanguages.Rust,
};
@@ -1,8 +0,0 @@
// gitnexus/src/core/ingestion/call-extractors/configs/swift.ts
import { SupportedLanguages } from 'gitnexus-shared';
import type { CallExtractionConfig } from '../../call-types.js';
export const swiftCallConfig: CallExtractionConfig = {
language: SupportedLanguages.Swift,
};
@@ -1,12 +0,0 @@
// gitnexus/src/core/ingestion/call-extractors/configs/typescript-javascript.ts
import { SupportedLanguages } from 'gitnexus-shared';
import type { CallExtractionConfig } from '../../call-types.js';
export const typescriptCallConfig: CallExtractionConfig = {
language: SupportedLanguages.TypeScript,
};
export const javascriptCallConfig: CallExtractionConfig = {
language: SupportedLanguages.JavaScript,
};
@@ -1,86 +0,0 @@
// gitnexus/src/core/ingestion/call-extractors/generic.ts
/**
* Generic table-driven call extractor factory.
*
* Mirrors method-extractors/generic.ts and field-extractors/generic.ts —
* define a config per language and generate extractors from configs.
*
* The factory converts a declarative {@link CallExtractionConfig} into a
* runtime {@link CallExtractor} whose `extract()` method:
* 1. Tries `config.extractLanguageCallSite(callNode)` for non-standard shapes.
* 2. Falls through to the generic path using shared utilities from
* `utils/call-analysis.ts` (`inferCallForm`, `extractReceiverName`, etc.).
*/
import type { SyntaxNode } from '../utils/ast-helpers.js';
import {
inferCallForm,
extractReceiverName,
extractReceiverNode,
extractMixedChain,
countCallArguments,
} from '../utils/call-analysis.js';
import type { CallExtractor, CallExtractionConfig, ExtractedCallSite } from '../call-types.js';
/**
* Create a CallExtractor from a declarative config.
*/
export function createCallExtractor(config: CallExtractionConfig): CallExtractor {
return {
language: config.language,
extract(callNode: SyntaxNode, callNameNode: SyntaxNode | undefined): ExtractedCallSite | null {
// ── Path 1: Language-specific call site ──────────────────────────
// Non-standard call shapes (e.g. Java `::` method references) are
// handled entirely by the config hook. When it returns a result,
// the generic path is skipped — no argCount, no mixed chain.
//
// Note: `extractLanguageCallSite` is called on every `extract()`
// invocation — both `extract(callNode, undefined)` (parse-worker
// Path 1) and `extract(callNode, callNameNode)` (Path 2).
// Language hooks must therefore be idempotent and cheap (e.g. a
// single node-type check).
if (config.extractLanguageCallSite) {
const seed = config.extractLanguageCallSite(callNode);
if (seed) {
return {
...seed,
...(config.typeAsReceiverHeuristic ? { typeAsReceiverHeuristic: true } : {}),
};
}
}
// ── Path 2: Generic extraction via @call.name ────────────────────
if (!callNameNode) return null;
const calledName = callNameNode.text;
const callForm = inferCallForm(callNode, callNameNode);
let receiverName = callForm === 'member' ? extractReceiverName(callNameNode) : undefined;
let receiverMixedChain: ExtractedCallSite['receiverMixedChain'];
// When the receiver is a complex expression (call chain, field chain,
// or mixed), extractReceiverName returns undefined. Walk the receiver
// node to build a unified mixed chain for deferred resolution.
if (callForm === 'member' && receiverName === undefined) {
const receiverNode = extractReceiverNode(callNameNode);
if (receiverNode) {
const extracted = extractMixedChain(receiverNode);
if (extracted && extracted.chain.length > 0) {
receiverMixedChain = extracted.chain;
receiverName = extracted.baseReceiverName;
}
}
}
return {
calledName,
...(callForm !== undefined ? { callForm } : {}),
...(receiverName !== undefined ? { receiverName } : {}),
argCount: countCallArguments(callNode),
...(receiverMixedChain !== undefined ? { receiverMixedChain } : {}),
...(config.typeAsReceiverHeuristic ? { typeAsReceiverHeuristic: true } : {}),
};
},
};
}
+108 -265
View File
@@ -1,34 +1,7 @@
import { KnowledgeGraph } from '../graph/types.js';
import { ASTCache } from './ast-cache.js';
import type { SymbolDefinition } from 'gitnexus-shared';
import type { SymbolTableReader, HeritageMap, ExtractedHeritage } from './model/index.js';
import { CLASS_TYPES, CALL_TARGET_TYPES, lookupMethodByOwnerWithMRO } from './model/index.js';
import type { DispatchDecision, ReceiverEnriched } from './call-types.js';
/** Shorthand for the receiver-source discriminant shared across the DAG. */
type ReceiverSource = ReceiverEnriched['receiverSource'];
/**
* DAG stage 4 fallback: used when `selectDispatch` is absent or returns null.
* Preserves pre-DAG dispatch semantics:
* - 'constructor' → constructor branch
* - 'free' → free branch (admits Swift/Kotlin class-target fast path)
* - 'member' or undefined → owner-scoped branch
*
* `undefined` callForm MUST route through owner-scoped (not free) so bare
* identifiers without a classified shape do NOT trigger `resolveFreeCall`'s
* class-target fast path. Without a `receiverTypeName`, the owner-scoped
* branch falls through to `resolveModuleAliasedCall` + `singleCandidate`,
* matching legacy behavior where non-callable symbols (Class, Interface)
* null-route instead of producing spurious Constructor edges.
*/
const defaultDispatchDecision = (
callForm: 'free' | 'member' | 'constructor' | undefined,
): DispatchDecision => {
if (callForm === 'constructor') return { primary: 'constructor' };
if (callForm === 'free') return { primary: 'free' };
return { primary: 'owner-scoped' };
};
import type { SymbolDefinition, SymbolTableReader } from './model/symbol-table.js';
import { CLASS_TYPES, CALL_TARGET_TYPES } from './model/symbol-table.js';
import Parser from 'tree-sitter';
import type { ResolutionContext } from './model/resolution-context.js';
import { TIER_CONFIDENCE, type ResolutionTier } from './model/resolution-context.js';
@@ -59,6 +32,7 @@ import {
} from './utils/call-analysis.js';
import { buildTypeEnv, isSubclassOf } from './type-env.js';
import type { ConstructorBinding, TypeEnvironment } from './type-env.js';
import type { HeritageMap } from './model/heritage-map.js';
import type { BindingAccumulator } from './binding-accumulator.js';
import { getTreeSitterBufferSize } from './constants.js';
import type {
@@ -68,11 +42,14 @@ import type {
ExtractedFetchCall,
FileConstructorBindings,
} from './workers/parse-worker.js';
import type { ExtractedHeritage } from './model/heritage-map.js';
import { normalizeFetchURL, routeMatches } from './route-extractors/nextjs.js';
import { extractTemplateComponents } from './vue-sfc-extractor.js';
import { extractReturnTypeName, stripNullable } from './type-extractors/shared.js';
import type { LiteralTypeInferrer } from './type-extractors/types.js';
import type { SyntaxNode } from './utils/ast-helpers.js';
import { extractParsedCallSite } from './call-sites/extract-language-call-site.js';
import { lookupMethodByOwnerWithMRO } from './model/resolve.js';
/** Per-file resolved type bindings for exported symbols.
* Populated during call processing, consumed by Phase 14 re-resolution pass. */
@@ -788,26 +765,22 @@ export const processCalls = async (
// Extract heritage from query matches to build parentMap for buildTypeEnv.
// Heritage-processor runs in PARALLEL, so graph edges don't exist when buildTypeEnv runs.
const fileParentMap = new Map<string, string[]>();
if (provider.heritageExtractor) {
for (const match of matches) {
const captureMap: Record<string, any> = {};
match.captures.forEach((c) => (captureMap[c.name] = c.node));
if (captureMap['heritage.class']) {
const heritageItems = provider.heritageExtractor.extract(captureMap, {
filePath: file.path,
language,
});
for (const item of heritageItems) {
if (item.kind === 'extends') {
let parents = fileParentMap.get(item.className);
if (!parents) {
parents = [];
fileParentMap.set(item.className, parents);
}
if (!parents.includes(item.parentName)) parents.push(item.parentName);
}
}
for (const match of matches) {
const captureMap: Record<string, any> = {};
match.captures.forEach((c) => (captureMap[c.name] = c.node));
if (captureMap['heritage.class'] && captureMap['heritage.extends']) {
const className: string = captureMap['heritage.class'].text;
const parentName: string = captureMap['heritage.extends'].text;
const extendsNode = captureMap['heritage.extends'];
const fieldDecl = extendsNode.parent;
if (fieldDecl?.type === 'field_declaration' && fieldDecl.childForFieldName('name'))
continue;
let parents = fileParentMap.get(className);
if (!parents) {
parents = [];
fileParentMap.set(className, parents);
}
if (!parents.includes(parentName)) parents.push(parentName);
}
}
const parentMap: ReadonlyMap<string, readonly string[]> = fileParentMap;
@@ -937,79 +910,74 @@ export const processCalls = async (
if (!captureMap['call']) return;
const callNode = captureMap['call'];
const callExtractor = provider.callExtractor;
const languageSeed = extractParsedCallSite(language, callNode);
if (languageSeed) {
if (provider.isBuiltInName(languageSeed.calledName)) return;
// ── Language-specific call site (e.g. Java :: method references) ──
if (callExtractor) {
const langCallSite = callExtractor.extract(callNode, undefined);
if (langCallSite) {
if (provider.isBuiltInName(langCallSite.calledName)) return;
const sourceId =
findEnclosingFunction(callNode, file.path, ctx, provider) ||
generateId('File', file.path);
const receiverName =
languageSeed.callForm === 'member' ? languageSeed.receiverName : undefined;
let receiverTypeName =
receiverName && typeEnv ? typeEnv.lookup(receiverName, callNode) : undefined;
const sourceId =
findEnclosingFunction(callNode, file.path, ctx, provider) ||
generateId('File', file.path);
const receiverName =
langCallSite.callForm === 'member' ? langCallSite.receiverName : undefined;
let receiverTypeName =
receiverName && typeEnv ? typeEnv.lookup(receiverName, callNode) : undefined;
if (
receiverName !== undefined &&
receiverTypeName === undefined &&
languageSeed.callForm === 'member' &&
(language === 'java' || language === 'csharp' || language === 'kotlin')
) {
const c0 = receiverName.charCodeAt(0);
if (c0 >= 65 && c0 <= 90) receiverTypeName = receiverName;
}
if (
langCallSite.typeAsReceiverHeuristic &&
receiverName !== undefined &&
receiverTypeName === undefined &&
langCallSite.callForm === 'member'
) {
const c0 = receiverName.charCodeAt(0);
if (c0 >= 65 && c0 <= 90) receiverTypeName = receiverName;
}
const resolved = resolveCallTarget(
{
calledName: languageSeed.calledName,
callForm: languageSeed.callForm,
...(receiverTypeName !== undefined ? { receiverTypeName } : {}),
...(receiverName !== undefined ? { receiverName } : {}),
},
file.path,
ctx,
undefined,
widenCache,
undefined,
heritageMap,
);
const resolved = resolveCallTarget(
{
calledName: langCallSite.calledName,
callForm: langCallSite.callForm,
...(receiverTypeName !== undefined ? { receiverTypeName } : {}),
...(receiverName !== undefined ? { receiverName } : {}),
},
if (!resolved) return;
graph.addRelationship({
id: generateId('CALLS', `${sourceId}:${languageSeed.calledName}->${resolved.nodeId}`),
sourceId,
targetId: resolved.nodeId,
type: 'CALLS',
confidence: resolved.confidence,
reason: resolved.reason,
});
if (heritageMap && languageSeed.callForm === 'member' && receiverTypeName) {
const implTargets = findInterfaceDispatchTargets(
languageSeed.calledName,
receiverTypeName,
file.path,
ctx,
undefined,
widenCache,
undefined,
heritageMap,
resolved.nodeId,
);
if (!resolved) return;
graph.addRelationship({
id: generateId('CALLS', `${sourceId}:${langCallSite.calledName}->${resolved.nodeId}`),
sourceId,
targetId: resolved.nodeId,
type: 'CALLS',
confidence: resolved.confidence,
reason: resolved.reason,
});
if (heritageMap && langCallSite.callForm === 'member' && receiverTypeName) {
const implTargets = findInterfaceDispatchTargets(
langCallSite.calledName,
receiverTypeName,
file.path,
ctx,
heritageMap,
resolved.nodeId,
);
for (const impl of implTargets) {
graph.addRelationship({
id: generateId('CALLS', `${sourceId}:${langCallSite.calledName}->${impl.nodeId}`),
sourceId,
targetId: impl.nodeId,
type: 'CALLS',
confidence: impl.confidence,
reason: impl.reason,
});
}
for (const impl of implTargets) {
graph.addRelationship({
id: generateId('CALLS', `${sourceId}:${languageSeed.calledName}->${impl.nodeId}`),
sourceId,
targetId: impl.nodeId,
type: 'CALLS',
confidence: impl.confidence,
reason: impl.reason,
});
}
return;
}
return;
}
const nameNode = captureMap['call.name'];
@@ -1017,28 +985,6 @@ export const processCalls = async (
const calledName = nameNode.text;
// Check heritage extractor for call-based heritage (e.g., Ruby include/extend/prepend)
if (provider.heritageExtractor?.extractFromCall) {
const heritageItems = provider.heritageExtractor.extractFromCall(
calledName,
captureMap['call'],
{ filePath: file.path, language },
);
if (heritageItems !== null) {
for (const item of heritageItems) {
collectedHeritage.push({
filePath: file.path,
className: item.className,
parentName: item.parentName,
kind: item.kind,
});
}
return;
}
}
// Dispatch: route language-specific calls (properties, imports)
// Heritage routing is handled by heritageExtractor.extractFromCall above.
const routed = callRouter?.(calledName, captureMap['call']);
if (routed) {
switch (routed.kind) {
@@ -1046,6 +992,17 @@ export const processCalls = async (
case 'import':
return;
case 'heritage':
for (const item of routed.items) {
collectedHeritage.push({
filePath: file.path,
className: item.enclosingClass,
parentName: item.mixinName,
kind: item.heritageKind,
});
}
return;
case 'properties': {
const fileId = generateId('File', file.path);
const propEnclosingClassId = findEnclosingClassId(captureMap['call'], file.path);
@@ -1098,17 +1055,10 @@ export const processCalls = async (
if (provider.isBuiltInName(calledName)) return;
// --- DAG stage 2-3: classify-form + infer-receiver (shared defaults) ---
// These stages run the shared inference chain. Language providers can
// customize infer-receiver (stage 3) via the inferImplicitReceiver hook
// which runs AFTER this default chain (typed-binding → constructor-map →
// module-alias → class-as-receiver → mixed-chain), and selectDispatch
// (stage 4) which picks the resolver branch.
let callForm = inferCallForm(callNode, nameNode);
let receiverName = callForm === 'member' ? extractReceiverName(nameNode) : undefined;
const callForm = inferCallForm(callNode, nameNode);
const receiverName = callForm === 'member' ? extractReceiverName(nameNode) : undefined;
let receiverTypeName =
receiverName && typeEnv ? typeEnv.lookup(receiverName, callNode) : undefined;
let receiverSource: ReceiverSource = receiverTypeName ? 'typed-binding' : 'none';
// Phase P: virtual dispatch override — when the declared type is a base class but
// the constructor created a known subclass, prefer the more specific type.
// Checks per-file parentMap first, then falls back to globalParentMap for
@@ -1147,7 +1097,6 @@ export const processCalls = async (
ctx.model.types.lookupClassByName(receiverTypeName).length > 0)
) {
receiverTypeName = ctorType;
receiverSource = 'constructor-map';
}
}
}
@@ -1156,14 +1105,10 @@ export const processCalls = async (
const enclosingFunc = findEnclosingFunction(callNode, file.path, ctx, provider);
const funcName = enclosingFunc ? extractFuncNameFromSourceId(enclosingFunc) : '';
receiverTypeName = lookupReceiverType(receiverIndex, funcName, receiverName);
if (receiverTypeName) receiverSource = 'constructor-map';
}
// Fall back to class-as-receiver for static method calls (e.g. UserService.find_user(),
// Greetable.format()). When the receiver name is not a variable in TypeEnv but
// resolves to a class-like symbol (Class / Interface / Struct / Enum / Trait) via
// tiered resolution, use it directly as the receiver type. `Trait` is included so
// Ruby module class-method calls flow through the class-as-receiver path and reach
// the `selectDispatch` hook's singleton branch.
// Fall back to class-as-receiver for static method calls (e.g. UserService.find_user()).
// When the receiver name is not a variable in TypeEnv but resolves to a Class/Struct/Interface
// through the standard tiered resolution, use it directly as the receiver type.
if (!receiverTypeName && receiverName && callForm === 'member') {
const typeResolved = ctx.resolve(receiverName, file.path);
if (
@@ -1173,12 +1118,10 @@ export const processCalls = async (
d.type === 'Class' ||
d.type === 'Interface' ||
d.type === 'Struct' ||
d.type === 'Enum' ||
d.type === 'Trait',
d.type === 'Enum',
)
) {
receiverTypeName = receiverName;
receiverSource = 'class-as-receiver';
}
}
// Hoist sourceId so it's available for ACCESSES edge emission during chain walk.
@@ -1224,51 +1167,11 @@ export const processCalls = async (
makeAccessEmitter(graph, sourceId),
heritageMap,
);
if (receiverTypeName) receiverSource = 'mixed-chain';
}
}
}
}
// --- DAG stage 3: infer-receiver (provider hook) ---
// Synthesize implicit receivers for languages that omit them (e.g., Ruby bare-call).
// This hook runs AFTER the shared inference chain so explicit receivers /
// typed bindings always take precedence. Output (if non-null) overlays onto
// the ReceiverEnriched for the next stage.
let dispatchHint: string | undefined;
if (provider.inferImplicitReceiver) {
const override = provider.inferImplicitReceiver({
calledName,
callForm,
receiverName,
receiverTypeName,
callNode,
filePath: file.path,
});
if (override) {
callForm = override.callForm;
receiverName = override.receiverName;
receiverTypeName = override.receiverTypeName;
receiverSource = override.receiverSource;
dispatchHint = override.hint;
}
}
// --- DAG stage 4: select-dispatch (provider hook + default fallback) ---
// Decide which resolver path to try first (primary) and fallback strategy.
// Language providers can customize dispatch via selectDispatch hook; all
// others use the shared defaultDispatchDecision. Always non-null after this
// block so downstream resolvers are table-driven.
const dispatchDecision: DispatchDecision =
provider.selectDispatch?.({
calledName,
callForm,
receiverName,
receiverTypeName,
receiverSource,
hint: dispatchHint,
}) ?? defaultDispatchDecision(callForm);
// Build overload hints for languages with inferLiteralType (Java/Kotlin/C#/C++).
// Only used when multiple candidates survive arity filtering — ~1-3% of calls.
const langConfig = provider.typeConfig;
@@ -1290,7 +1193,6 @@ export const processCalls = async (
widenCache,
undefined,
heritageMap,
dispatchDecision,
);
if (!resolved) return;
@@ -1829,20 +1731,11 @@ const resolveCallTarget = (
widenCache?: WidenCache,
preComputedArgTypes?: (string | undefined)[],
heritageMap?: HeritageMap,
dispatchDecision?: DispatchDecision,
): ResolveResult | null => {
const tiered = ctx.resolve(call.calledName, currentFile);
if (!tiered) return null;
// DAG dispatch: use decision.primary to pick the resolver branch.
// Callers that own the DAG (processCalls + crossFile deferred paths)
// pass a decision; other callers use the shared default ladder.
// Language-specific primary / fallback / ancestryView overrides come from
// the provider's `selectDispatch` hook.
const decision = dispatchDecision ?? defaultDispatchDecision(call.callForm);
const primary = decision.primary;
if (primary === 'free') {
if (call.callForm === 'free') {
return resolveFreeCall(
call.calledName,
currentFile,
@@ -1853,7 +1746,7 @@ const resolveCallTarget = (
preComputedArgTypes,
);
}
if (primary === 'constructor') {
if (call.callForm === 'constructor') {
return (
resolveStaticCall(
call.calledName,
@@ -1866,7 +1759,6 @@ const resolveCallTarget = (
) ?? singleCandidate(tiered, call.argCount, 'constructor')
);
}
// primary === 'owner-scoped'
if (call.receiverTypeName) {
// Skip the owner-scoped MRO path when the tiered pool has genuine
// overload ambiguity that needs D1-D4+E handling, not D0.
@@ -1874,15 +1766,6 @@ const resolveCallTarget = (
(!!overloadHints || !!preComputedArgTypes) &&
countCallableCandidates(tiered.candidates, call.argCount, call.callForm) > 1;
// Try owner-scoped (resolveMemberCall) then file-scoped (resolveMemberCallByFile).
// DAG: dispatchDecision.ancestryView selects instance vs singleton ancestry
// for kind-aware MRO strategies. Ruby `Account.log` flows via 'singleton'.
//
// Singleton-ancestry miss MUST NOT degrade to the file-scoped fallback:
// resolveMemberCallByFile matches by ownerId and would happily pick an
// instance method defined on the same class, leaking instance dispatch
// onto what was declared a class-method call. For singleton dispatch,
// a miss either null-routes or falls through to `decision.fallback`.
const singletonDispatch = decision.ancestryView === 'singleton';
const memberResult =
(!skipMember
? resolveMemberCall(
@@ -1892,21 +1775,18 @@ const resolveCallTarget = (
ctx,
heritageMap,
call.argCount,
decision.ancestryView,
)
: null) ??
(singletonDispatch
? null
: resolveMemberCallByFile(
call.calledName,
call.receiverTypeName,
currentFile,
ctx,
call.argCount,
call.callForm,
overloadHints,
preComputedArgTypes,
));
resolveMemberCallByFile(
call.calledName,
call.receiverTypeName,
currentFile,
ctx,
call.argCount,
call.callForm,
overloadHints,
preComputedArgTypes,
);
if (memberResult) return memberResult;
// Module-alias narrowing runs as a FALLBACK, after owner/file-scoped
@@ -1942,26 +1822,7 @@ const resolveCallTarget = (
// hierarchy. When the type is NOT in the index (PHP `mixed`, dynamic
// types, unresolvable aliases), the scoped resolvers had nothing to
// work with and singleCandidate is the correct last resort.
//
// DAG fallback override: when `select-dispatch` returned
// `fallback: 'free-arity-narrowed'` (today: Ruby implicit-self bare
// calls whose enclosing class doesn't define the method), fall through
// to free-call resolution instead of null-routing. This preserves
// existing free-call arity-narrowing heuristics for bare calls that
// happen to target methods on unrelated classes.
if (typeResolves && typeResolves.candidates.length > 0) {
if (decision.fallback === 'free-arity-narrowed') {
const free = resolveFreeCall(
call.calledName,
currentFile,
ctx,
call.argCount,
tiered,
overloadHints,
preComputedArgTypes,
);
if (free) return free;
}
return null; // null-route: type resolved, no candidate matched
}
return singleCandidate(tiered, call.argCount, call.callForm);
@@ -2157,13 +2018,6 @@ const resolveMethodByOwner = (
ctx: ResolutionContext,
heritageMap?: HeritageMap,
argCount?: number,
/**
* DAG-sourced ancestry selector. `'singleton'` routes through
* `heritageMap.getSingletonAncestry(owner)` for class-method dispatch
* (Ruby `Account.log` via `extend LoggerMixin`). Default / undefined
* uses the walker's instance-dispatch behavior.
*/
ancestryView?: 'instance' | 'singleton',
): { def: SymbolDefinition; tier: ResolutionTier } | undefined => {
const typeResolved = ctx.resolve(receiverTypeName, filePath);
if (!typeResolved) return undefined;
@@ -2192,14 +2046,6 @@ const resolveMethodByOwner = (
let ambiguous = false;
for (const candidate of typeResolved.candidates) {
if (!CLASS_LIKE_TYPES.has(candidate.type)) continue;
// Singleton dispatch: when the DAG decision requested the singleton
// ancestry view, pass `heritageMap.getSingletonAncestry` as the walker's
// ancestry override. Kind-aware strategies (e.g. MroStrategy 'ruby-mixin')
// honor the override by scanning it linearly in place of their default walk.
const singletonOverride =
ancestryView === 'singleton' && canWalkMRO && heritageMap
? heritageMap.getSingletonAncestry(candidate.nodeId).map((e) => e.parentId)
: undefined;
const def = canWalkMRO
? lookupMethodByOwnerWithMRO(
candidate.nodeId,
@@ -2208,7 +2054,6 @@ const resolveMethodByOwner = (
ctx.model,
mroStrategy,
argCount,
singletonOverride,
)
: ctx.model.methods.lookupMethodByOwner(candidate.nodeId, methodName, argCount);
if (!def) continue;
@@ -2263,7 +2108,6 @@ export const resolveMemberCall = (
ctx: ResolutionContext,
heritageMap?: HeritageMap,
argCount?: number,
ancestryView?: 'instance' | 'singleton',
): ResolveResult | null => {
const resolved = resolveMethodByOwner(
ownerType,
@@ -2272,7 +2116,6 @@ export const resolveMemberCall = (
ctx,
heritageMap,
argCount,
ancestryView,
);
if (!resolved) return null;
return toResolveResult(resolved.def, resolved.tier);
+42 -13
View File
@@ -1,14 +1,10 @@
/**
* Shared Ruby call routing logic.
*
* Ruby expresses imports and property definitions as method calls rather
* than syntax-level constructs. This module provides a routing function
* used by the CLI call-processor, CLI parse-worker, and the web
* call-processor so that the classification logic lives in one place.
*
* Heritage (mixins: include/extend/prepend) was previously routed here
* but is now handled by heritageExtractor.extractFromCall before the
* call router runs. The router still returns 'skip' for these calls.
* Ruby expresses imports, heritage (mixins), and property definitions as
* method calls rather than syntax-level constructs. This module provides a
* routing function used by the CLI call-processor, CLI parse-worker, and
* the web call-processor so that the classification logic lives in one place.
*
* NOTE: This file is intentionally duplicated in gitnexus-web/ because the
* two packages have separate build targets (Node native vs WASM/browser).
@@ -34,10 +30,17 @@ export type CallRouter = (calledName: string, callNode: SyntaxNode) => CallRouti
export type RubyCallRouting =
| { kind: 'import'; importPath: string; isRelative: boolean }
| { kind: 'heritage'; items: RubyHeritageItem[] }
| { kind: 'properties'; items: RubyPropertyItem[] }
| { kind: 'call' }
| { kind: 'skip' };
export interface RubyHeritageItem {
enclosingClass: string;
mixinName: string;
heritageKind: 'include' | 'extend' | 'prepend';
}
export type RubyAccessorType = 'attr_accessor' | 'attr_reader' | 'attr_writer';
export interface RubyPropertyItem {
@@ -53,6 +56,9 @@ export interface RubyPropertyItem {
const CALL_RESULT: RubyCallRouting = { kind: 'call' };
const SKIP_RESULT: RubyCallRouting = { kind: 'skip' };
/** Max depth for parent-walking loops to prevent pathological AST traversals */
const MAX_PARENT_DEPTH = 50;
// ── Routing function ────────────────────────────────────────────────────────
/**
@@ -82,12 +88,35 @@ export function routeRubyCall(calledName: string, callNode: SyntaxNode): RubyCal
return { kind: 'import', importPath, isRelative };
}
// ── include / extend / prepend — heritage (now handled by heritageExtractor) ─
// Call-based heritage is intercepted by heritageExtractor.extractFromCall
// before the call router runs. Return SKIP_RESULT so these calls don't
// fall through to normal call processing.
// ── include / extend / prepend → heritage (mixin) ──────────────────────
if (calledName === 'include' || calledName === 'extend' || calledName === 'prepend') {
return SKIP_RESULT;
let enclosingClass: string | null = null;
let current = callNode.parent;
let depth = 0;
while (current && ++depth <= MAX_PARENT_DEPTH) {
if (current.type === 'class' || current.type === 'module') {
const nameNode = current.childForFieldName?.('name');
if (nameNode) {
enclosingClass = nameNode.text;
break;
}
}
current = current.parent;
}
if (!enclosingClass) return SKIP_RESULT;
const items: RubyHeritageItem[] = [];
const argList = callNode.childForFieldName?.('arguments');
for (const arg of argList?.children ?? []) {
if (arg.type === 'constant' || arg.type === 'scope_resolution') {
items.push({
enclosingClass,
mixinName: arg.text,
heritageKind: calledName as 'include' | 'extend' | 'prepend',
});
}
}
return items.length > 0 ? { kind: 'heritage', items } : SKIP_RESULT;
}
// ── attr_accessor / attr_reader / attr_writer → property definitions ───
@@ -0,0 +1,33 @@
/** Non-generic @call shapes → { calledName, callForm, receiverName? } (used from call-processor / parse-worker). */
import { SupportedLanguages } from '../../../config/supported-languages.js';
import type { SyntaxNode } from '../utils/ast-helpers.js';
import { parseJavaMethodReference } from './java.js';
export type ParsedCallSite = {
calledName: string;
callForm: 'free' | 'member' | 'constructor';
receiverName?: string;
};
/** Non-null → seed replaces @call.name; null → use @call.name + inferCallForm / extractReceiverName. */
export function extractParsedCallSite(
language: SupportedLanguages,
callNode: SyntaxNode,
): ParsedCallSite | null {
switch (language) {
case SupportedLanguages.Java:
if (callNode.type === 'method_reference') {
const parsed = parseJavaMethodReference(callNode);
if (!parsed) return null;
return {
calledName: parsed.calledName,
callForm: parsed.callForm,
...(parsed.receiverName !== undefined ? { receiverName: parsed.receiverName } : {}),
};
}
return null;
default:
return null;
}
}
@@ -0,0 +1,41 @@
/** Java `method_reference` (`::`) nodes (tree-sitter-java). `super::` still lacks TypeEnv receiver typing. */
import type { SyntaxNode } from '../utils/ast-helpers.js';
export type ParsedJavaMethodReference = {
calledName: string;
callForm: 'member' | 'constructor';
receiverName?: string;
};
/** Parse `expr::method`, `Type::new`, `this::m`, `super::m`. */
export const parseJavaMethodReference = (
callNode: SyntaxNode,
): ParsedJavaMethodReference | null => {
if (callNode.type !== 'method_reference') return null;
const recv = callNode.namedChild(0);
if (!recv) return null;
for (const c of callNode.children) {
if (c.type === 'new') {
if (recv.type !== 'identifier') return null;
return { calledName: recv.text, callForm: 'constructor' };
}
}
const rhs = callNode.child(callNode.childCount - 1);
if (!rhs || rhs.type !== 'identifier') return null;
const methodName = rhs.text;
if (recv.type === 'identifier') {
return { calledName: methodName, callForm: 'member', receiverName: recv.text };
}
if (recv.type === 'this') {
return { calledName: methodName, callForm: 'member', receiverName: 'this' };
}
if (recv.type === 'super') {
return { calledName: methodName, callForm: 'member', receiverName: 'super' };
}
return null;
};
-177
View File
@@ -1,177 +0,0 @@
// gitnexus/src/core/ingestion/call-types.ts
/**
* Types for the language-agnostic call extraction pipeline.
*
* Mirrors method-types.ts / field-types.ts: defines the domain interfaces
* consumed by createCallExtractor() and the per-language configs.
*/
import type { SupportedLanguages } from 'gitnexus-shared';
import type { SyntaxNode } from './utils/ast-helpers.js';
import type { MixedChainStep } from './utils/call-analysis.js';
// ---------------------------------------------------------------------------
// Extracted result
// ---------------------------------------------------------------------------
/**
* Per-node call extraction result. The parse worker enriches this with
* file-level context (filePath, sourceId, TypeEnv lookups, arg types) to
* produce the final `ExtractedCall` that enters the resolution pipeline.
*/
export interface ExtractedCallSite {
calledName: string;
callForm?: 'free' | 'member' | 'constructor';
receiverName?: string;
argCount?: number;
/** Unified mixed chain for complex receivers (field + call chains). */
receiverMixedChain?: MixedChainStep[];
/** When true, the type-as-receiver heuristic applies: if receiverName
* starts with an uppercase letter and has no TypeEnv binding, treat it
* as a type name (e.g. Java `User::getName`). */
typeAsReceiverHeuristic?: boolean;
}
// ---------------------------------------------------------------------------
// Extractor interface (produced by createCallExtractor)
// ---------------------------------------------------------------------------
export interface CallExtractor {
readonly language: SupportedLanguages;
/**
* Extract a call site from captured AST nodes.
*
* @param callNode The @call capture (call_expression, method_invocation, …)
* @param callNameNode The @call.name capture (identifier inside the call).
* May be undefined when the call shape has no name capture
* (e.g. Java method_reference via `::`).
* @returns Extracted call site, or null when no call can be derived.
*/
extract(callNode: SyntaxNode, callNameNode: SyntaxNode | undefined): ExtractedCallSite | null;
}
// ---------------------------------------------------------------------------
// Config interface (one per language / language group)
// ---------------------------------------------------------------------------
export interface CallExtractionConfig {
language: SupportedLanguages;
/**
* Language-specific call site extraction. Called **before** the generic
* path. If it returns non-null, the generic `inferCallForm` /
* `extractReceiverName` path is skipped entirely.
*
* Use this for call shapes that don't follow the standard `@call` /
* `@call.name` pattern (e.g. Java `method_reference` via `::`).
*/
extractLanguageCallSite?: (callNode: SyntaxNode) => ExtractedCallSite | null;
/**
* Whether the type-as-receiver heuristic applies for this language.
* When true and the receiver name starts with an uppercase letter,
* the receiver is treated as a type name when no TypeEnv binding exists.
*
* Applies to JVM and C# languages where `Type.method()` and `Type::method`
* are common patterns.
*/
typeAsReceiverHeuristic?: boolean;
}
// ---------------------------------------------------------------------------
// Call-resolution DAG types
// ---------------------------------------------------------------------------
//
// The call-resolution pipeline is a typed DAG:
//
// extract-call ──▶ classify-form ──▶ infer-receiver ──▶ select-dispatch ──▶ resolve-target ──▶ emit-edge
//
// Provider hooks plug in at infer-receiver and select-dispatch; shared stages
// stay language-agnostic. Stages 1-2 run in the parse worker; stages 3-6 run
// on the main thread. DAG-internal types below are main-thread-only and never
// serialize to the graph.
/**
* DAG stage 3 output: call record with receiver type and source discriminant.
*
* `receiverTypeName` is resolved via TypeEnv → constructor-map → class-as-receiver →
* mixed-chain, or synthesized by `inferImplicitReceiver`. `receiverSource` tags
* which path won and drives MRO strategy selection in stage 4.
*
* Invariants:
* - `receiverSource` MUST match how `receiverTypeName` was resolved; every
* discriminant must have a live reader and writer.
* - `hint` is opaque to shared stages; only the same provider's `selectDispatch` reads it.
*
* @see language-provider.ts § inferImplicitReceiver, selectDispatch
*/
export interface ReceiverEnriched {
readonly calledName: string;
readonly callForm: 'free' | 'member' | 'constructor' | undefined;
readonly receiverName: string | undefined;
readonly receiverTypeName: string | undefined;
readonly receiverSource:
| 'none'
| 'typed-binding'
| 'constructor-map'
| 'class-as-receiver'
| 'mixed-chain'
| 'implicit-self';
/** Free-form hint from the provider hook; opaque to shared stages. */
readonly hint?: string;
}
/**
* Provider hook output for `LanguageProvider.inferImplicitReceiver` (DAG stage 3).
*
* Overlay applied to `ReceiverEnriched` when an implicit receiver is synthesized.
* Ruby example: bare `serialize` inside `Account#call_serialize` →
* `{ callForm: 'member', receiverName: 'self', receiverTypeName: 'Account',
* receiverSource: 'implicit-self', hint: 'instance' }`
*
* Invariants:
* - `receiverSource` is always `'implicit-self'` — the only variant this type produces.
* - `callForm` is always `'member'` — the rewrite converts bare-call to method invocation.
* - `hint` is opaque to shared stages; consumed by the same language's `selectDispatch`.
*/
export interface ImplicitReceiverOverride {
readonly callForm: 'free' | 'member' | 'constructor';
readonly receiverName: string;
readonly receiverTypeName: string;
readonly receiverSource: Extract<ReceiverEnriched['receiverSource'], 'implicit-self'>;
/** Free-form language tag (e.g. Ruby sets 'singleton' for `def self.foo`
* method bodies). Consumed by the same language's `selectDispatch` hook. */
readonly hint?: string;
}
/**
* DAG stage 4 output: dispatch strategy for resolving the target method.
*
* Encodes which resolver branch to try first and an optional fallback.
* Stage 5 delegates to `resolveMemberCall`, `resolveFreeCall`, or
* `resolveStaticCall` based on `primary`.
*
* - `primary`: `'owner-scoped'` = MRO walk, `'free'` = arity-tiered global lookup,
* `'constructor'` = type instantiation.
* - `fallback`: Only `'free-arity-narrowed'` exists; used by Ruby implicit-self
* to degrade to arity-tiered free lookup when the MRO walk misses.
* - `ancestryView`: Ruby `'ruby-mixin'` only. `'singleton'` walks extend providers
* only; a miss NEVER falls through to file-scoped lookup (enforced in
* resolveCallTarget). `'instance'` is the default.
*
* Common patterns:
* - `{primary: 'constructor'}` — constructor call
* - `{primary: 'owner-scoped'}` — member call with known type
* - `{primary: 'owner-scoped', fallback: 'free-arity-narrowed', ancestryView: 'instance'}` — Ruby implicit-self
* - `{primary: 'owner-scoped', ancestryView: 'singleton'}` — Ruby class-method call
*
* @see language-provider.ts § selectDispatch
* @see call-processor.ts § defaultDispatchDecision, resolveCallTarget
*/
export interface DispatchDecision {
readonly primary: 'owner-scoped' | 'free' | 'constructor';
readonly fallback?: 'free-arity-narrowed';
readonly ancestryView?: 'instance' | 'singleton';
readonly hint?: string;
}
@@ -1,15 +0,0 @@
// gitnexus/src/core/ingestion/class-extractors/configs/c-cpp.ts
import { SupportedLanguages } from 'gitnexus-shared';
import type { ClassExtractionConfig } from '../../class-types.js';
export const cClassConfig: ClassExtractionConfig = {
language: SupportedLanguages.C,
typeDeclarationNodes: ['struct_specifier', 'enum_specifier'],
};
export const cppClassConfig: ClassExtractionConfig = {
language: SupportedLanguages.CPlusPlus,
typeDeclarationNodes: ['class_specifier', 'struct_specifier', 'enum_specifier'],
ancestorScopeNodeTypes: ['namespace_definition', 'class_specifier', 'struct_specifier'],
};
@@ -1,24 +0,0 @@
// gitnexus/src/core/ingestion/class-extractors/configs/csharp.ts
import { SupportedLanguages } from 'gitnexus-shared';
import type { ClassExtractionConfig } from '../../class-types.js';
export const csharpClassConfig: ClassExtractionConfig = {
language: SupportedLanguages.CSharp,
typeDeclarationNodes: [
'class_declaration',
'interface_declaration',
'struct_declaration',
'enum_declaration',
'record_declaration',
],
fileScopeNodeTypes: ['file_scoped_namespace_declaration'],
ancestorScopeNodeTypes: [
'namespace_declaration',
'class_declaration',
'interface_declaration',
'struct_declaration',
'enum_declaration',
'record_declaration',
],
};
@@ -1,10 +0,0 @@
// gitnexus/src/core/ingestion/class-extractors/configs/dart.ts
import { SupportedLanguages } from 'gitnexus-shared';
import type { ClassExtractionConfig } from '../../class-types.js';
export const dartClassConfig: ClassExtractionConfig = {
language: SupportedLanguages.Dart,
typeDeclarationNodes: ['class_definition', 'extension_declaration', 'enum_declaration'],
ancestorScopeNodeTypes: ['class_definition', 'extension_declaration', 'enum_declaration'],
};
@@ -1,21 +0,0 @@
// gitnexus/src/core/ingestion/class-extractors/configs/go.ts
import { SupportedLanguages } from 'gitnexus-shared';
import type { ClassExtractionConfig } from '../../class-types.js';
export const goClassConfig: ClassExtractionConfig = {
language: SupportedLanguages.Go,
typeDeclarationNodes: ['type_declaration'],
fileScopeNodeTypes: ['package_clause'],
extractName(node) {
const typeSpec = node.namedChildren.find((child) => child.type === 'type_spec');
return typeSpec?.childForFieldName('name')?.text;
},
extractType(node) {
const typeSpec = node.namedChildren.find((child) => child.type === 'type_spec');
const typeNode = typeSpec?.childForFieldName('type');
if (typeNode?.type === 'struct_type') return 'Struct';
if (typeNode?.type === 'interface_type') return 'Interface';
return undefined;
},
};
@@ -1,40 +0,0 @@
// gitnexus/src/core/ingestion/class-extractors/configs/jvm.ts
import { SupportedLanguages } from 'gitnexus-shared';
import type { ClassExtractionConfig } from '../../class-types.js';
// ---------------------------------------------------------------------------
// Java
// ---------------------------------------------------------------------------
export const javaClassConfig: ClassExtractionConfig = {
language: SupportedLanguages.Java,
typeDeclarationNodes: [
'class_declaration',
'interface_declaration',
'enum_declaration',
'record_declaration',
],
fileScopeNodeTypes: ['package_declaration'],
ancestorScopeNodeTypes: [
'class_declaration',
'interface_declaration',
'enum_declaration',
'record_declaration',
],
};
// ---------------------------------------------------------------------------
// Kotlin
// ---------------------------------------------------------------------------
export const kotlinClassConfig: ClassExtractionConfig = {
language: SupportedLanguages.Kotlin,
typeDeclarationNodes: ['class_declaration', 'object_declaration', 'companion_object'],
fileScopeNodeTypes: ['package_header'],
ancestorScopeNodeTypes: ['class_declaration', 'object_declaration', 'companion_object'],
extractType(node) {
if (node.type !== 'class_declaration') return undefined;
return node.children.some((child) => child?.text === 'interface') ? 'Interface' : 'Class';
},
};
@@ -1,10 +0,0 @@
// gitnexus/src/core/ingestion/class-extractors/configs/php.ts
import { SupportedLanguages } from 'gitnexus-shared';
import type { ClassExtractionConfig } from '../../class-types.js';
export const phpClassConfig: ClassExtractionConfig = {
language: SupportedLanguages.PHP,
typeDeclarationNodes: ['class_declaration', 'interface_declaration', 'enum_declaration'],
ancestorScopeNodeTypes: ['namespace_definition'],
};
@@ -1,10 +0,0 @@
// gitnexus/src/core/ingestion/class-extractors/configs/python.ts
import { SupportedLanguages } from 'gitnexus-shared';
import type { ClassExtractionConfig } from '../../class-types.js';
export const pythonClassConfig: ClassExtractionConfig = {
language: SupportedLanguages.Python,
typeDeclarationNodes: ['class_definition'],
ancestorScopeNodeTypes: ['class_definition'],
};
@@ -1,10 +0,0 @@
// gitnexus/src/core/ingestion/class-extractors/configs/ruby.ts
import { SupportedLanguages } from 'gitnexus-shared';
import type { ClassExtractionConfig } from '../../class-types.js';
export const rubyClassConfig: ClassExtractionConfig = {
language: SupportedLanguages.Ruby,
typeDeclarationNodes: ['class'],
ancestorScopeNodeTypes: ['module', 'class'],
};

Some files were not shown because too many files have changed in this diff Show More