Compare commits
30
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
c9c5fd169b | ||
|
|
d024119b33 | ||
|
|
0a4b31b3c5 | ||
|
|
54d02fcc22 | ||
|
|
d50a523837 | ||
|
|
54c7e45a2f | ||
|
|
e947a82a18 | ||
|
|
978187b34a | ||
|
|
185cec70a1 | ||
|
|
1d0fb782a3 | ||
|
|
7001e8e4b4 | ||
|
|
ed07c18b8d | ||
|
|
df429ea60c | ||
|
|
b43cb5ee55 | ||
|
|
fec06b823c | ||
|
|
eb0d9c51a0 | ||
|
|
7a5ab57bd3 | ||
|
|
3fd4346bcb | ||
|
|
c2734cd25e | ||
|
|
8cb2f278cc | ||
|
|
f3df8ab7ba | ||
|
|
93cdfb27fa | ||
|
|
109a3c6946 | ||
|
|
1df79c2eab | ||
|
|
32c9ddaf32 | ||
|
|
385ee037bd | ||
|
|
28ddbe5d54 | ||
|
|
c100577e5e | ||
|
|
baf3f9e37d | ||
|
|
b340c5d87a |
@@ -0,0 +1,75 @@
|
||||
version: 2
|
||||
updates:
|
||||
# Keep third-party Actions SHA pins current. See CONTRIBUTING.md — when
|
||||
# reviewing these bumps, verify the SHA corresponds to the claimed tag by
|
||||
# running `gh api repos/<owner>/<action>/git/refs/tags/<tag>` before merge.
|
||||
- package-ecosystem: github-actions
|
||||
directory: /
|
||||
schedule:
|
||||
interval: weekly
|
||||
open-pull-requests-limit: 5
|
||||
commit-message:
|
||||
prefix: chore
|
||||
include: scope
|
||||
labels:
|
||||
- dependencies
|
||||
- ci
|
||||
|
||||
# Gitnexus npm deps — tree-sitter grammars checked daily so we catch
|
||||
# new releases that unblock the tree-sitter 0.25 upgrade ASAP. Grammars
|
||||
# are grouped so lockstep bumps produce a single PR. The tree-sitter
|
||||
# RUNTIME is pinned — upgrade deliberately via the drift check workflow.
|
||||
# See .github/scripts/check-tree-sitter-upgrade-readiness.py for
|
||||
# the upgrade readiness tracker.
|
||||
- package-ecosystem: npm
|
||||
directory: /gitnexus
|
||||
schedule:
|
||||
interval: daily
|
||||
open-pull-requests-limit: 10
|
||||
commit-message:
|
||||
prefix: chore(deps)
|
||||
include: scope
|
||||
labels:
|
||||
- dependencies
|
||||
groups:
|
||||
tree-sitter-grammars:
|
||||
patterns:
|
||||
- tree-sitter-*
|
||||
exclude-patterns:
|
||||
- tree-sitter
|
||||
- tree-sitter-cli
|
||||
ignore:
|
||||
# Pin the tree-sitter runtime at 0.21.x until the drift check
|
||||
# reports all grammars are peer-dep compatible with 0.25.
|
||||
- dependency-name: tree-sitter
|
||||
update-types:
|
||||
- version-update:semver-major
|
||||
- version-update:semver-minor
|
||||
# tree-sitter-cli follows the runtime's version cadence. Bump when
|
||||
# regenerating vendor/tree-sitter-proto/src/parser.c, not on a schedule.
|
||||
- dependency-name: tree-sitter-cli
|
||||
|
||||
# gitnexus-web (thin frontend client).
|
||||
- package-ecosystem: npm
|
||||
directory: /gitnexus-web
|
||||
schedule:
|
||||
interval: weekly
|
||||
open-pull-requests-limit: 5
|
||||
commit-message:
|
||||
prefix: chore(deps)
|
||||
include: scope
|
||||
labels:
|
||||
- dependencies
|
||||
- frontend
|
||||
|
||||
# Shared types package.
|
||||
- package-ecosystem: npm
|
||||
directory: /gitnexus-shared
|
||||
schedule:
|
||||
interval: weekly
|
||||
open-pull-requests-limit: 5
|
||||
commit-message:
|
||||
prefix: chore(deps)
|
||||
include: scope
|
||||
labels:
|
||||
- dependencies
|
||||
@@ -0,0 +1,53 @@
|
||||
# release-drafter config — used only for PR autolabeling by
|
||||
# `.github/workflows/pr-labeler.yml` (the workflow passes `disable-releaser: true`,
|
||||
# so the draft-release side of release-drafter never runs).
|
||||
#
|
||||
# The labels applied here are the same ones `.github/release.yml` maps to
|
||||
# categorized release-notes sections.
|
||||
#
|
||||
# `sync-labels: true` removes managed autolabels that no longer match the PR —
|
||||
# critical for the breaking-change case: if a PR title drops the `!` or the body
|
||||
# drops `BREAKING CHANGE:`, the `breaking` label is pulled off automatically.
|
||||
|
||||
# Required by release-drafter; not used because releaser is disabled.
|
||||
name-template: 'unused'
|
||||
tag-template: 'unused'
|
||||
template: |
|
||||
$CHANGES
|
||||
|
||||
sync-labels: true
|
||||
|
||||
autolabeler:
|
||||
- label: enhancement
|
||||
title:
|
||||
- '/^feat(\([^)]+\))?!?:/i'
|
||||
- label: bug
|
||||
title:
|
||||
- '/^fix(\([^)]+\))?!?:/i'
|
||||
- label: performance
|
||||
title:
|
||||
- '/^perf(\([^)]+\))?!?:/i'
|
||||
- label: refactor
|
||||
title:
|
||||
- '/^refactor(\([^)]+\))?!?:/i'
|
||||
- label: documentation
|
||||
title:
|
||||
- '/^docs(\([^)]+\))?!?:/i'
|
||||
- label: test
|
||||
title:
|
||||
- '/^test(\([^)]+\))?!?:/i'
|
||||
- label: ci
|
||||
title:
|
||||
- '/^ci(\([^)]+\))?!?:/i'
|
||||
- label: dependencies
|
||||
title:
|
||||
- '/^(build|deps)(\([^)]+\))?!?:/i'
|
||||
- label: chore
|
||||
title:
|
||||
- '/^(chore|revert)(\([^)]+\))?!?:/i'
|
||||
# Breaking-change marker: either `!` in the type prefix or `BREAKING CHANGE:` in body.
|
||||
- label: breaking
|
||||
title:
|
||||
- '/^[a-z]+(\([^)]+\))?!:/i'
|
||||
body:
|
||||
- '/BREAKING[ -]CHANGE:/i'
|
||||
@@ -0,0 +1,358 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Monitor tree-sitter 0.25 upgrade readiness.
|
||||
|
||||
Tracks two things Dependabot cannot see:
|
||||
|
||||
1. Peer-dep compatibility. Each tree-sitter-* grammar declares a peer
|
||||
dependency on the tree-sitter runtime. We want to know when every
|
||||
grammar's *latest npm release* satisfies tree-sitter@0.25.0 so we
|
||||
can upgrade without --legacy-peer-deps.
|
||||
|
||||
2. Vendored upstream drift. vendor/tree-sitter-proto/ is a snapshot of
|
||||
coder3101/tree-sitter-proto's parser.c. When upstream moves, we want
|
||||
to know whether we can pick it up.
|
||||
|
||||
Invoked from .github/workflows/tree-sitter-upgrade-readiness.yml daily.
|
||||
Runs locally too:
|
||||
|
||||
python3 .github/scripts/check-tree-sitter-upgrade-readiness.py
|
||||
|
||||
Outputs Markdown to stdout. Exit 0 when every grammar is upgrade-ready
|
||||
and the vendored proto is in sync. Exit 1 when blockers remain (the
|
||||
workflow uses this to open or update a tracking issue).
|
||||
|
||||
No external deps -- stdlib only, so it runs on any vanilla runner.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import os
|
||||
import pathlib
|
||||
import re
|
||||
import sys
|
||||
import urllib.error
|
||||
import urllib.request
|
||||
|
||||
REPO_ROOT = pathlib.Path(__file__).resolve().parents[2]
|
||||
GITNEXUS_DIR = REPO_ROOT / "gitnexus"
|
||||
VENDOR_PROTO_DIR = GITNEXUS_DIR / "vendor" / "tree-sitter-proto"
|
||||
|
||||
# ── Upgrade target ──────────────────────────────────────────────────────
|
||||
# The runtime version we want to upgrade TO. Update this when the goal
|
||||
# changes (e.g. once 0.25 lands and we target 0.26).
|
||||
TARGET_RUNTIME = "0.25.0"
|
||||
TARGET_RUNTIME_MAJOR_MINOR = ".".join(TARGET_RUNTIME.split(".")[:2])
|
||||
|
||||
# Tree-sitter runtime -> (min_abi, max_abi) it can load. Only the current
|
||||
# and target entries matter; extend when changing TARGET_RUNTIME.
|
||||
RUNTIME_ABI_RANGES: dict[str, tuple[int, int]] = {
|
||||
"0.21": (13, 14),
|
||||
"0.25": (13, 15),
|
||||
}
|
||||
|
||||
assert TARGET_RUNTIME_MAJOR_MINOR in RUNTIME_ABI_RANGES, (
|
||||
f"RUNTIME_ABI_RANGES has no entry for {TARGET_RUNTIME_MAJOR_MINOR!r}. "
|
||||
f"Add the ABI range after auditing the upstream release notes."
|
||||
)
|
||||
|
||||
# Grammars we use. Values are the upstream GitHub repos to check for
|
||||
# unreleased ABI bumps (owner/repo, branch, parser.c path).
|
||||
GRAMMARS: dict[str, tuple[str, str, str]] = {
|
||||
"tree-sitter-c": ("tree-sitter/tree-sitter-c", "master", "src/parser.c"),
|
||||
"tree-sitter-c-sharp": ("tree-sitter/tree-sitter-c-sharp", "master", "src/parser.c"),
|
||||
"tree-sitter-cpp": ("tree-sitter/tree-sitter-cpp", "master", "src/parser.c"),
|
||||
"tree-sitter-dart": ("UserNobody14/tree-sitter-dart", "master", "src/parser.c"),
|
||||
"tree-sitter-go": ("tree-sitter/tree-sitter-go", "master", "src/parser.c"),
|
||||
"tree-sitter-java": ("tree-sitter/tree-sitter-java", "master", "src/parser.c"),
|
||||
"tree-sitter-javascript": ("tree-sitter/tree-sitter-javascript", "master", "src/parser.c"),
|
||||
"tree-sitter-kotlin": ("fwcd/tree-sitter-kotlin", "main", "src/parser.c"),
|
||||
"tree-sitter-php": ("tree-sitter/tree-sitter-php", "master", "php/src/parser.c"),
|
||||
"tree-sitter-python": ("tree-sitter/tree-sitter-python", "master", "src/parser.c"),
|
||||
"tree-sitter-ruby": ("tree-sitter/tree-sitter-ruby", "master", "src/parser.c"),
|
||||
"tree-sitter-rust": ("tree-sitter/tree-sitter-rust", "master", "src/parser.c"),
|
||||
"tree-sitter-swift": ("alex-pinkus/tree-sitter-swift", "main", "src/parser.c"),
|
||||
"tree-sitter-typescript": ("tree-sitter/tree-sitter-typescript", "master", "typescript/src/parser.c"),
|
||||
}
|
||||
|
||||
UPSTREAM_PROTO_OWNER = "coder3101"
|
||||
UPSTREAM_PROTO_REPO = "tree-sitter-proto"
|
||||
UPSTREAM_PROTO_BRANCH = "main"
|
||||
|
||||
|
||||
# ── Helpers ─────────────────────────────────────────────────────────────
|
||||
|
||||
def read_current_runtime() -> str:
|
||||
"""Return the tree-sitter runtime version pinned in package.json (e.g. '0.21')."""
|
||||
pkg = json.loads((GITNEXUS_DIR / "package.json").read_text())
|
||||
raw = pkg["dependencies"]["tree-sitter"]
|
||||
match = re.search(r"(\d+)\.(\d+)", raw)
|
||||
if not match:
|
||||
raise SystemExit(f"could not parse tree-sitter version: {raw!r}")
|
||||
return f"{match.group(1)}.{match.group(2)}"
|
||||
|
||||
|
||||
def npm_view_json(pkg: str) -> dict | None:
|
||||
"""Fetch package metadata from the npm registry via HTTPS.
|
||||
|
||||
Uses the registry API directly so we don't depend on the npm CLI
|
||||
being available (it's a batch file on Windows which complicates
|
||||
subprocess calls).
|
||||
"""
|
||||
url = f"https://registry.npmjs.org/{pkg}/latest"
|
||||
try:
|
||||
req = urllib.request.Request(url, headers={"Accept": "application/json"})
|
||||
with urllib.request.urlopen(req, timeout=8) as resp:
|
||||
return json.loads(resp.read().decode("utf-8"))
|
||||
except (urllib.error.URLError, urllib.error.HTTPError, json.JSONDecodeError):
|
||||
return None
|
||||
|
||||
|
||||
def satisfies_target(peer_range: str | None, target: str) -> bool:
|
||||
"""Check if a semver range like '^0.22.4' or '^0.25.0' satisfies the target.
|
||||
|
||||
Simple heuristic: extract the minimum version from the range and check
|
||||
if target >= min. For caret ranges (^X.Y.Z), the upper bound is the
|
||||
next major (for X>0) or next minor (for X==0). We check both bounds.
|
||||
"""
|
||||
if peer_range is None:
|
||||
# No peer dep declared = no constraint = compatible.
|
||||
return True
|
||||
match = re.search(r"(\d+)\.(\d+)\.(\d+)", peer_range)
|
||||
if not match:
|
||||
return False
|
||||
min_major, min_minor, min_patch = int(match.group(1)), int(match.group(2)), int(match.group(3))
|
||||
|
||||
t_match = re.search(r"(\d+)\.(\d+)\.(\d+)", target)
|
||||
if not t_match:
|
||||
return False
|
||||
t_major, t_minor, t_patch = int(t_match.group(1)), int(t_match.group(2)), int(t_match.group(3))
|
||||
|
||||
# Target must be >= minimum.
|
||||
target_tuple = (t_major, t_minor, t_patch)
|
||||
min_tuple = (min_major, min_minor, min_patch)
|
||||
if target_tuple < min_tuple:
|
||||
return False
|
||||
|
||||
# For caret ranges with major 0: ^0.X.Y allows [0.X.Y, 0.(X+1).0).
|
||||
if peer_range.startswith("^") and min_major == 0:
|
||||
if t_major != 0 or t_minor >= min_minor + 1:
|
||||
return False
|
||||
# For caret ranges with major >0: ^X.Y.Z allows [X.Y.Z, (X+1).0.0).
|
||||
elif peer_range.startswith("^") and min_major > 0:
|
||||
if t_major >= min_major + 1:
|
||||
return False
|
||||
|
||||
return True
|
||||
|
||||
|
||||
_GITHUB_TOKEN = os.environ.get("GITHUB_TOKEN")
|
||||
|
||||
|
||||
def fetch_text(url: str, timeout: int = 8) -> str | None:
|
||||
"""Fetch a URL and return its text, or None on failure.
|
||||
|
||||
Adds an Authorization header for github.com URLs when GITHUB_TOKEN is
|
||||
set (raises the rate limit from 60 to 5 000 requests/hour).
|
||||
"""
|
||||
headers: dict[str, str] = {}
|
||||
if _GITHUB_TOKEN and ("github.com" in url or "githubusercontent.com" in url):
|
||||
headers["Authorization"] = f"Bearer {_GITHUB_TOKEN}"
|
||||
try:
|
||||
req = urllib.request.Request(url, headers=headers)
|
||||
with urllib.request.urlopen(req, timeout=timeout) as resp:
|
||||
return resp.read().decode("utf-8", errors="ignore")
|
||||
except (urllib.error.URLError, urllib.error.HTTPError):
|
||||
return None
|
||||
|
||||
|
||||
def extract_abi_from_text(text: str) -> int | None:
|
||||
"""Extract LANGUAGE_VERSION from parser.c text."""
|
||||
match = re.search(r"#define\s+LANGUAGE_VERSION\s+(\d+)", text[:4096])
|
||||
return int(match.group(1)) if match else None
|
||||
|
||||
|
||||
def extract_language_version(parser_c: pathlib.Path) -> int | None:
|
||||
"""Return the LANGUAGE_VERSION defined in a parser.c, or None if absent."""
|
||||
if not parser_c.is_file():
|
||||
return None
|
||||
with parser_c.open("r", encoding="utf-8", errors="ignore") as fh:
|
||||
head = fh.read(4096)
|
||||
return extract_abi_from_text(head)
|
||||
|
||||
|
||||
def md_h(text: str, level: int = 2) -> str:
|
||||
return f"{'#' * level} {text}\n"
|
||||
|
||||
|
||||
# ── Main ────────────────────────────────────────────────────────────────
|
||||
|
||||
def main() -> int:
|
||||
blockers: dict[str, str] = {}
|
||||
lines: list[str] = []
|
||||
lines.append(md_h("Tree-sitter 0.25 upgrade readiness", 1))
|
||||
lines.append("")
|
||||
|
||||
current_runtime = read_current_runtime()
|
||||
current_abi_range = RUNTIME_ABI_RANGES.get(current_runtime, (0, 0))
|
||||
target_abi_range = RUNTIME_ABI_RANGES.get(TARGET_RUNTIME_MAJOR_MINOR, (0, 0))
|
||||
|
||||
lines.append(f"- Current runtime: `tree-sitter@{current_runtime}.x` (ABI {current_abi_range[0]}..{current_abi_range[1]})")
|
||||
lines.append(f"- Target runtime: `tree-sitter@{TARGET_RUNTIME}` (ABI {target_abi_range[0]}..{target_abi_range[1]})")
|
||||
lines.append("")
|
||||
|
||||
# ── Grammar peer-dep compatibility ───────────────────────────────
|
||||
lines.append(md_h("Grammar compatibility", 2))
|
||||
lines.append("| Grammar | npm latest | Peer dep | Satisfies 0.25? | ABI | Upstream ABI | Status |")
|
||||
lines.append("|---|---|---|---|---|---|---|")
|
||||
|
||||
ready_count = 0
|
||||
total_count = len(GRAMMARS)
|
||||
|
||||
for name, (upstream_repo, upstream_branch, parser_path) in sorted(GRAMMARS.items()):
|
||||
# Fetch latest npm metadata.
|
||||
info = npm_view_json(name)
|
||||
fetch_failed = info is None
|
||||
npm_version = "?"
|
||||
peer_range = None
|
||||
peer_optional = True
|
||||
if info:
|
||||
npm_version = info.get("version", "?")
|
||||
peers = info.get("peerDependencies") or {}
|
||||
peer_range = peers.get("tree-sitter")
|
||||
meta = info.get("peerDependenciesMeta") or {}
|
||||
ts_meta = meta.get("tree-sitter") or {}
|
||||
peer_optional = ts_meta.get("optional", False) if peer_range else True
|
||||
|
||||
if fetch_failed:
|
||||
peer_display = "? (fetch failed)"
|
||||
compatible = False
|
||||
else:
|
||||
peer_display = peer_range or "none"
|
||||
if peer_range and not peer_optional:
|
||||
peer_display += " (required)"
|
||||
compatible = satisfies_target(peer_range, TARGET_RUNTIME)
|
||||
|
||||
# Check installed ABI using the same parser_path from GRAMMARS.
|
||||
installed_parser = GITNEXUS_DIR / "node_modules" / name / parser_path
|
||||
if not installed_parser.is_file():
|
||||
# Fallback to default location.
|
||||
installed_parser = GITNEXUS_DIR / "node_modules" / name / "src" / "parser.c"
|
||||
installed_abi = extract_language_version(installed_parser)
|
||||
abi_display = str(installed_abi) if installed_abi else "?"
|
||||
|
||||
# Check upstream (main/master branch) ABI for unreleased work.
|
||||
upstream_url = (
|
||||
f"https://raw.githubusercontent.com/{upstream_repo}/"
|
||||
f"{upstream_branch}/{parser_path}"
|
||||
)
|
||||
upstream_text = fetch_text(upstream_url)
|
||||
upstream_abi = extract_abi_from_text(upstream_text) if upstream_text else None
|
||||
upstream_abi_display = str(upstream_abi) if upstream_abi else "?"
|
||||
|
||||
# Determine status.
|
||||
if fetch_failed:
|
||||
status = "Unknown (fetch failed)"
|
||||
blockers[name] = f"`{name}`: npm registry fetch failed — could not verify peer dep"
|
||||
elif compatible:
|
||||
status = "Ready"
|
||||
ready_count += 1
|
||||
elif upstream_abi and upstream_abi >= 15:
|
||||
status = "Unreleased (ABI 15 on main)"
|
||||
blockers[name] = f"`{name}`: ABI 15 on `{upstream_repo}` main but not published to npm"
|
||||
else:
|
||||
status = "Blocking"
|
||||
blockers[name] = f"`{name}@{npm_version}`: peer `{peer_display}` incompatible with 0.25"
|
||||
|
||||
# Also check upstream package.json for relaxed peer dep.
|
||||
if not compatible and not fetch_failed:
|
||||
upstream_pkg_url = (
|
||||
f"https://raw.githubusercontent.com/{upstream_repo}/"
|
||||
f"{upstream_branch}/package.json"
|
||||
)
|
||||
upstream_pkg_text = fetch_text(upstream_pkg_url)
|
||||
if upstream_pkg_text:
|
||||
try:
|
||||
upstream_pkg = json.loads(upstream_pkg_text)
|
||||
upstream_peer = (upstream_pkg.get("peerDependencies") or {}).get("tree-sitter")
|
||||
if upstream_peer and satisfies_target(upstream_peer, TARGET_RUNTIME):
|
||||
status = "Unreleased (peer relaxed on main)"
|
||||
blockers[name] = f"`{name}`: peer dep relaxed on `{upstream_repo}` main but not published to npm"
|
||||
except json.JSONDecodeError:
|
||||
pass
|
||||
|
||||
compat_icon = "Yes" if compatible else "**No**"
|
||||
lines.append(
|
||||
f"| `{name}` | {npm_version} | {peer_display} | {compat_icon} | {abi_display} | {upstream_abi_display} | {status} |"
|
||||
)
|
||||
|
||||
lines.append("")
|
||||
lines.append(f"**{ready_count}/{total_count}** grammars ready for `tree-sitter@{TARGET_RUNTIME}`.")
|
||||
lines.append("")
|
||||
|
||||
# ── Vendored proto drift ─────────────────────────────────────────
|
||||
lines.append(md_h("Vendored tree-sitter-proto", 2))
|
||||
vendored_abi = extract_language_version(VENDOR_PROTO_DIR / "src" / "parser.c")
|
||||
|
||||
upstream_proto_url = (
|
||||
f"https://raw.githubusercontent.com/{UPSTREAM_PROTO_OWNER}/"
|
||||
f"{UPSTREAM_PROTO_REPO}/{UPSTREAM_PROTO_BRANCH}/src/parser.c"
|
||||
)
|
||||
upstream_proto_text = fetch_text(upstream_proto_url)
|
||||
upstream_proto_abi = extract_abi_from_text(upstream_proto_text) if upstream_proto_text else None
|
||||
|
||||
sha_url = (
|
||||
f"https://api.github.com/repos/{UPSTREAM_PROTO_OWNER}/"
|
||||
f"{UPSTREAM_PROTO_REPO}/commits/{UPSTREAM_PROTO_BRANCH}"
|
||||
)
|
||||
sha_text = fetch_text(sha_url)
|
||||
upstream_sha = "?"
|
||||
if sha_text:
|
||||
try:
|
||||
upstream_sha = json.loads(sha_text).get("sha", "?")[:12]
|
||||
except json.JSONDecodeError:
|
||||
pass
|
||||
|
||||
local_proto_path = VENDOR_PROTO_DIR / "src" / "parser.c"
|
||||
local_proto_text = local_proto_path.read_text(encoding="utf-8", errors="ignore") if local_proto_path.is_file() else ""
|
||||
in_sync = bool(
|
||||
upstream_proto_text
|
||||
and local_proto_text.replace("\r\n", "\n")
|
||||
== upstream_proto_text.replace("\r\n", "\n")
|
||||
)
|
||||
|
||||
lines.append(f"- Upstream: `{UPSTREAM_PROTO_OWNER}/{UPSTREAM_PROTO_REPO}@{UPSTREAM_PROTO_BRANCH}` (HEAD `{upstream_sha}`)")
|
||||
lines.append(f"- Upstream ABI: **{upstream_proto_abi}**")
|
||||
lines.append(f"- Vendored ABI: **{vendored_abi}**")
|
||||
lines.append(f"- In sync: {'yes' if in_sync else 'no — upstream has diverged'}")
|
||||
|
||||
if upstream_proto_abi and vendored_abi and upstream_proto_abi > vendored_abi:
|
||||
can_upgrade = upstream_proto_abi <= target_abi_range[1]
|
||||
lines.append(f"- Upstream ABI {upstream_proto_abi} {'is' if can_upgrade else 'is NOT'} within target runtime range ({target_abi_range[0]}..{target_abi_range[1]})")
|
||||
if can_upgrade:
|
||||
lines.append(f"- **Action:** after upgrading to tree-sitter@{TARGET_RUNTIME}, regenerate vendored parser.c from upstream `{upstream_sha}`")
|
||||
else:
|
||||
lines.append(f"- **Action:** wait for runtime upgrade beyond {TARGET_RUNTIME} that supports ABI {upstream_proto_abi}")
|
||||
blockers["vendored-proto-abi"] = f"vendored tree-sitter-proto: upstream ABI {upstream_proto_abi} outside target range"
|
||||
elif not in_sync:
|
||||
lines.append("- **Action:** review upstream changes; vendored copy may need updating")
|
||||
blockers["vendored-proto-sync"] = "vendored tree-sitter-proto: out of sync with upstream"
|
||||
|
||||
# ── Summary ──────────────────────────────────────────────────────
|
||||
lines.append("")
|
||||
lines.append(md_h("Summary", 2))
|
||||
if blockers:
|
||||
lines.append(f"**{len(blockers)} blocker(s) remaining:**\n")
|
||||
for b in blockers.values():
|
||||
lines.append(f"- {b}")
|
||||
lines.append("")
|
||||
lines.append("Upgrade to `tree-sitter@0.25` is **blocked**.")
|
||||
else:
|
||||
lines.append("All grammars are compatible. Upgrade to `tree-sitter@0.25` is **ready**.")
|
||||
|
||||
print("\n".join(lines))
|
||||
return 1 if blockers else 0
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main())
|
||||
@@ -0,0 +1,173 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Enforce the GitHub Actions concurrency convention.
|
||||
|
||||
See CONTRIBUTING.md -> "GitHub Actions — Concurrency Convention" for the rules.
|
||||
|
||||
Invoked from .github/workflows/ci-quality.yml. Runs locally too:
|
||||
python3 .github/scripts/check-workflow-concurrency.py .github/workflows
|
||||
|
||||
Rules:
|
||||
1. Every entry-point (non-reusable) workflow declares a top-level
|
||||
`concurrency:` block.
|
||||
2. Reusable workflows (on: workflow_call ONLY) do NOT declare one.
|
||||
3. The `concurrency.group` expression MUST reference either
|
||||
`${{ github.workflow }}` or a literal `CI-` prefix (the documented
|
||||
ci.yml reusable-workflow-safe exception). This is checked by substring
|
||||
containment rather than prefix match because ci.yml's group is a
|
||||
conditional expression that resolves to a `CI-…` literal at runtime.
|
||||
|
||||
We deliberately do not use a YAML library — keeps the script dependency-free
|
||||
on any vanilla runner. `on:` block parsing is line-based and handles both the
|
||||
flat (`on: workflow_call`) and mapping (`on:\n workflow_call:`) forms.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import pathlib
|
||||
import re
|
||||
import sys
|
||||
|
||||
|
||||
REQUIRED_TOKENS = ("${{ github.workflow }}", "CI-")
|
||||
|
||||
|
||||
def is_reusable(lines: list[str]) -> bool:
|
||||
"""Return True iff the workflow's `on:` block names only `workflow_call`."""
|
||||
in_on = False
|
||||
on_indent: int | None = None
|
||||
keys: list[str] = []
|
||||
|
||||
for raw in lines:
|
||||
# Skip blank lines and comments
|
||||
stripped = raw.strip()
|
||||
if not stripped or stripped.startswith("#"):
|
||||
continue
|
||||
|
||||
indent = len(raw) - len(raw.lstrip(" "))
|
||||
|
||||
if not in_on:
|
||||
if raw.startswith("on:"):
|
||||
remainder = raw[len("on:"):].strip()
|
||||
if not remainder:
|
||||
# `on:` followed by indented mapping on next lines
|
||||
in_on = True
|
||||
on_indent = indent
|
||||
continue
|
||||
if remainder.startswith("[") and remainder.endswith("]"):
|
||||
# Flow-style list: on: [workflow_call]
|
||||
items = [
|
||||
item.strip() for item in remainder.strip("[]").split(",")
|
||||
]
|
||||
return items == ["workflow_call"]
|
||||
# Scalar form: on: workflow_call (or a single other event)
|
||||
return remainder == "workflow_call"
|
||||
continue
|
||||
|
||||
# Inside the `on:` block; stop when indentation returns to <= on_indent
|
||||
if on_indent is not None and indent <= on_indent:
|
||||
break
|
||||
|
||||
# Only consider keys at on_indent + indentation step (anything deeper
|
||||
# is nested config like `types:`)
|
||||
if ":" not in stripped:
|
||||
continue
|
||||
# Heuristic: first-level event keys are those with indent == on_indent + 2
|
||||
# (the canonical step for a 2-space YAML doc). We collect all first-level
|
||||
# keys by tracking the smallest indent seen inside the block.
|
||||
keys.append((indent, stripped.split(":", 1)[0].strip()))
|
||||
|
||||
if not keys:
|
||||
return False
|
||||
|
||||
# Take only the outermost-indented keys as the event list
|
||||
min_indent = min(i for i, _ in keys)
|
||||
events = [name for i, name in keys if i == min_indent]
|
||||
return events == ["workflow_call"]
|
||||
|
||||
|
||||
CONCURRENCY_RE = re.compile(r"^concurrency:\s*$")
|
||||
GROUP_RE = re.compile(r"^\s+group:\s*(.+?)\s*$")
|
||||
|
||||
|
||||
def extract_group_key(lines: list[str]) -> str | None:
|
||||
"""Return the `group:` value of the top-level `concurrency:` block, or None."""
|
||||
for idx, raw in enumerate(lines):
|
||||
if CONCURRENCY_RE.match(raw):
|
||||
# Scan forward until we leave the concurrency block (next top-level key
|
||||
# is at column 0 and ends with `:`).
|
||||
for follow in lines[idx + 1:]:
|
||||
if follow and not follow.startswith(" ") and follow.rstrip().endswith(":"):
|
||||
break
|
||||
m = GROUP_RE.match(follow)
|
||||
if m:
|
||||
return m.group(1).strip().strip("'").strip('"')
|
||||
break
|
||||
return None
|
||||
|
||||
|
||||
def has_top_level_concurrency(lines: list[str]) -> bool:
|
||||
return any(CONCURRENCY_RE.match(raw) for raw in lines)
|
||||
|
||||
|
||||
def check(workflows_dir: pathlib.Path) -> int:
|
||||
fail = 0
|
||||
files = sorted(
|
||||
list(workflows_dir.glob("*.yml")) + list(workflows_dir.glob("*.yaml"))
|
||||
)
|
||||
for path in files:
|
||||
lines = path.read_text(encoding="utf-8").splitlines()
|
||||
reusable = is_reusable(lines)
|
||||
has_conc = has_top_level_concurrency(lines)
|
||||
|
||||
if reusable:
|
||||
if has_conc:
|
||||
print(
|
||||
f"::error file={path}::Reusable workflow (on: workflow_call) "
|
||||
"must NOT declare its own concurrency block — it inherits "
|
||||
"from the caller. See CONTRIBUTING.md -> GitHub Actions — "
|
||||
"Concurrency Convention."
|
||||
)
|
||||
fail = 1
|
||||
continue
|
||||
|
||||
if not has_conc:
|
||||
print(
|
||||
f"::error file={path}::Missing top-level concurrency block. "
|
||||
"See CONTRIBUTING.md -> GitHub Actions — Concurrency Convention."
|
||||
)
|
||||
fail = 1
|
||||
continue
|
||||
|
||||
group = extract_group_key(lines)
|
||||
if group is None:
|
||||
print(
|
||||
f"::error file={path}::concurrency block is missing a "
|
||||
"`group:` key."
|
||||
)
|
||||
fail = 1
|
||||
continue
|
||||
|
||||
if not any(token in group for token in REQUIRED_TOKENS):
|
||||
print(
|
||||
f"::error file={path}::concurrency.group `{group}` must "
|
||||
f"reference one of {REQUIRED_TOKENS}. See CONTRIBUTING.md -> "
|
||||
"GitHub Actions — Concurrency Convention."
|
||||
)
|
||||
fail = 1
|
||||
|
||||
return fail
|
||||
|
||||
|
||||
def main(argv: list[str]) -> int:
|
||||
if len(argv) != 2:
|
||||
print(f"usage: {argv[0]} <workflows-dir>", file=sys.stderr)
|
||||
return 2
|
||||
workflows_dir = pathlib.Path(argv[1])
|
||||
if not workflows_dir.is_dir():
|
||||
print(f"not a directory: {workflows_dir}", file=sys.stderr)
|
||||
return 2
|
||||
return check(workflows_dir)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main(sys.argv))
|
||||
@@ -11,8 +11,8 @@ jobs:
|
||||
outputs:
|
||||
web_changed: ${{ steps.filter.outputs.web }}
|
||||
steps:
|
||||
- uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4
|
||||
- uses: dorny/paths-filter@de90cc6fb38fc0963ad72b210f1f284cd68cea36 # v3
|
||||
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
- uses: dorny/paths-filter@fbd0ab8f3e69293af611ebaee6363fc25e6d187d # v3
|
||||
id: filter
|
||||
with:
|
||||
filters: |
|
||||
@@ -26,7 +26,7 @@ jobs:
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 20
|
||||
steps:
|
||||
- uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4
|
||||
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
|
||||
- uses: ./.github/actions/setup-gitnexus-web
|
||||
|
||||
@@ -74,7 +74,7 @@ jobs:
|
||||
|
||||
- name: Upload test results
|
||||
if: always()
|
||||
uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4
|
||||
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1
|
||||
with:
|
||||
name: e2e-results
|
||||
path: |
|
||||
|
||||
@@ -8,8 +8,8 @@ jobs:
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 5
|
||||
steps:
|
||||
- uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4
|
||||
- uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4
|
||||
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
- uses: actions/setup-node@53b83947a5a98c8d113130e565377fae1a50d02f # v6.3.0
|
||||
with:
|
||||
node-version: 20
|
||||
cache: npm
|
||||
@@ -21,8 +21,8 @@ jobs:
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 10
|
||||
steps:
|
||||
- uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4
|
||||
- uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4
|
||||
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
- uses: actions/setup-node@53b83947a5a98c8d113130e565377fae1a50d02f # v6.3.0
|
||||
with:
|
||||
node-version: 20
|
||||
cache: npm
|
||||
@@ -34,7 +34,7 @@ jobs:
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 10
|
||||
steps:
|
||||
- uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4
|
||||
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
- uses: ./.github/actions/setup-gitnexus
|
||||
- run: npx tsc --noEmit
|
||||
working-directory: gitnexus
|
||||
@@ -43,7 +43,30 @@ jobs:
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 10
|
||||
steps:
|
||||
- uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4
|
||||
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
- uses: ./.github/actions/setup-gitnexus-web
|
||||
- run: npx tsc -b --noEmit
|
||||
working-directory: gitnexus-web
|
||||
|
||||
# Enforces the convention documented in CONTRIBUTING.md → "GitHub Actions —
|
||||
# Concurrency Convention":
|
||||
# 1. Every entry-point (non-reusable) workflow declares a top-level
|
||||
# `concurrency:` block.
|
||||
# 2. Reusable workflows (`on: workflow_call` only) do NOT declare one —
|
||||
# they inherit concurrency from the caller.
|
||||
# 3. The concurrency group key starts with `${{ github.workflow }}` or
|
||||
# the literal `CI-` prefix (the documented ci.yml exception for
|
||||
# reusable-workflow-safe grouping).
|
||||
# Reusability is detected by parsing each workflow's `on:` block, not an
|
||||
# allowlist, so new reusable workflows never produce false positives.
|
||||
workflow-convention:
|
||||
name: Workflow concurrency convention
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 5
|
||||
steps:
|
||||
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
- name: Validate workflow concurrency convention
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
python3 .github/scripts/check-workflow-concurrency.py .github/workflows
|
||||
|
||||
@@ -14,6 +14,16 @@ permissions:
|
||||
contents: read # needed for sparse checkout of vitest.config.ts
|
||||
pull-requests: write # needed to post sticky PR comment
|
||||
|
||||
# Concurrency convention: see CONTRIBUTING.md → "GitHub Actions — Concurrency Convention".
|
||||
# Serialize sticky-comment writes per PR so two rapid CI completions don't race.
|
||||
# Internal PRs surface in `pull_requests[0].number`. Fork PRs leave that array empty,
|
||||
# so we fall back to `<head-repo-full-name>/<head-branch>`, which is stable across
|
||||
# reruns and subsequent pushes for the same fork PR (unlike `workflow_run.id` which
|
||||
# is unique per run and therefore does not serialize anything).
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.event.workflow_run.pull_requests[0].number || format('{0}/{1}', github.event.workflow_run.head_repository.full_name, github.event.workflow_run.head_branch) }}
|
||||
cancel-in-progress: false
|
||||
|
||||
jobs:
|
||||
pr-report:
|
||||
name: PR Report
|
||||
@@ -26,7 +36,7 @@ jobs:
|
||||
steps:
|
||||
# ── Download artifacts from the CI run ────────────────────────
|
||||
- name: Download artifacts
|
||||
uses: actions/github-script@60a0d83039c74a4aee543508d2ffcb1c3799cdea # v7
|
||||
uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v7
|
||||
with:
|
||||
script: |
|
||||
const fs = require('fs');
|
||||
@@ -113,7 +123,7 @@ jobs:
|
||||
|
||||
- name: Checkout (for vitest config)
|
||||
if: steps.meta.outputs.skip != 'true'
|
||||
uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4
|
||||
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
with:
|
||||
sparse-checkout: gitnexus/vitest.config.ts
|
||||
sparse-checkout-cone-mode: false
|
||||
@@ -122,7 +132,7 @@ jobs:
|
||||
- name: Fetch base branch coverage
|
||||
if: steps.meta.outputs.skip != 'true'
|
||||
id: base-coverage
|
||||
uses: actions/github-script@60a0d83039c74a4aee543508d2ffcb1c3799cdea # v7
|
||||
uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v7
|
||||
with:
|
||||
script: |
|
||||
const fs = require('fs');
|
||||
@@ -406,7 +416,7 @@ jobs:
|
||||
|
||||
- name: Comment on PR
|
||||
if: steps.meta.outputs.skip != 'true'
|
||||
uses: marocchino/sticky-pull-request-comment@773744901bac0e8cbb5a0dc842800d45e9b2b405 # v2
|
||||
uses: marocchino/sticky-pull-request-comment@0ea0beb66eb9baf113663a64ec522f60e49231c0 # v2
|
||||
with:
|
||||
header: ci-report
|
||||
number: ${{ steps.meta.outputs.pr_number }}
|
||||
|
||||
@@ -9,7 +9,7 @@ jobs:
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 25
|
||||
steps:
|
||||
- uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4
|
||||
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
- uses: ./.github/actions/setup-gitnexus
|
||||
with:
|
||||
build: 'true'
|
||||
@@ -43,7 +43,7 @@ jobs:
|
||||
|
||||
- name: Upload test reports
|
||||
if: always()
|
||||
uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4
|
||||
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1
|
||||
with:
|
||||
name: test-reports
|
||||
path: |
|
||||
@@ -63,7 +63,7 @@ jobs:
|
||||
runs-on: ${{ matrix.os }}
|
||||
timeout-minutes: 25
|
||||
steps:
|
||||
- uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4
|
||||
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
- uses: ./.github/actions/setup-gitnexus
|
||||
with:
|
||||
build: 'true'
|
||||
|
||||
@@ -9,9 +9,20 @@ on:
|
||||
paths-ignore: ['**.md', 'docs/**', 'LICENSE']
|
||||
workflow_call:
|
||||
|
||||
# Concurrency convention: see CONTRIBUTING.md → "GitHub Actions — Concurrency Convention".
|
||||
# Hardcoded `CI-` prefix (not `${{ github.workflow }}`) because this workflow is
|
||||
# invoked as a reusable workflow from publish.yml and release-candidate.yml. In
|
||||
# called-workflow context `github.workflow` evaluation is ambiguous across GitHub
|
||||
# Actions versions, and a prefix that could resolve to the caller's name would
|
||||
# share a concurrency group with the caller → deadlock. A literal prefix is
|
||||
# immune. Direct `push`/`pull_request` invocations use `CI-<ref>`; invocations
|
||||
# from a reusable-workflow caller fall into a per-run-unique group that never
|
||||
# serializes with the caller.
|
||||
# cancel-in-progress is event-aware: cancel superseded PR runs, queue every other
|
||||
# event (push to main, workflow_call from publish.yml, etc.).
|
||||
concurrency:
|
||||
group: ci-${{ github.ref }}
|
||||
cancel-in-progress: true
|
||||
group: ${{ (github.event_name == 'pull_request' || github.event_name == 'push') && format('CI-{0}', github.ref) || format('CI-nested-{0}', github.run_id) }}
|
||||
cancel-in-progress: ${{ github.event_name == 'pull_request' }}
|
||||
|
||||
# ── Reusable workflow orchestration ─────────────────────────────────
|
||||
# Each concern lives in its own workflow file for maintainability:
|
||||
@@ -74,7 +85,7 @@ jobs:
|
||||
cp pr-meta/e2e_result pr-meta/e2e-result
|
||||
|
||||
- name: Upload PR metadata
|
||||
uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4
|
||||
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1
|
||||
with:
|
||||
name: pr-meta
|
||||
path: pr-meta/
|
||||
@@ -96,9 +107,9 @@ jobs:
|
||||
TESTS: ${{ needs.tests.result }}
|
||||
E2E: ${{ needs.e2e.result }}
|
||||
run: |
|
||||
echo "Quality: $QUALITY"
|
||||
echo "Tests: $TESTS"
|
||||
echo "E2E: $E2E"
|
||||
echo "Quality: $QUALITY"
|
||||
echo "Tests: $TESTS"
|
||||
echo "E2E: $E2E"
|
||||
if [[ "$QUALITY" != "success" ]] ||
|
||||
[[ "$TESTS" != "success" ]]; then
|
||||
echo "::error::Quality or test jobs failed"
|
||||
|
||||
@@ -16,9 +16,10 @@ on:
|
||||
issue_comment:
|
||||
types: [created]
|
||||
|
||||
# Concurrency convention: see CONTRIBUTING.md → "GitHub Actions — Concurrency Convention".
|
||||
# Serialize per-PR to avoid racing review comments.
|
||||
concurrency:
|
||||
group: claude-review-${{ github.event.issue.number || github.event.pull_request.number }}
|
||||
group: ${{ github.workflow }}-${{ github.event.issue.number || github.event.pull_request.number }}
|
||||
cancel-in-progress: false
|
||||
|
||||
jobs:
|
||||
@@ -56,7 +57,7 @@ jobs:
|
||||
# For issue_comment triggers, resolve the PR number, head SHA, and fork repo
|
||||
- name: Resolve PR context
|
||||
id: pr
|
||||
uses: actions/github-script@60a0d83039c74a4aee543508d2ffcb1c3799cdea # v7
|
||||
uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v7
|
||||
with:
|
||||
script: |
|
||||
let pr;
|
||||
@@ -76,7 +77,7 @@ jobs:
|
||||
core.setOutput('branch', pr.head.ref);
|
||||
|
||||
- name: Checkout PR head
|
||||
uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4
|
||||
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
with:
|
||||
repository: ${{ steps.pr.outputs.repo }}
|
||||
ref: ${{ steps.pr.outputs.sha }}
|
||||
|
||||
@@ -10,9 +10,10 @@ on:
|
||||
pull_request_review:
|
||||
types: [submitted]
|
||||
|
||||
# Concurrency convention: see CONTRIBUTING.md → "GitHub Actions — Concurrency Convention".
|
||||
# Serialize per-PR/issue to avoid racing comments.
|
||||
concurrency:
|
||||
group: claude-code-${{ github.event.issue.number || github.event.pull_request.number || github.event.issue.id }}
|
||||
group: ${{ github.workflow }}-${{ github.event.issue.number || github.event.pull_request.number || github.event.issue.id }}
|
||||
cancel-in-progress: false
|
||||
|
||||
jobs:
|
||||
@@ -58,7 +59,7 @@ jobs:
|
||||
# For PR-related triggers, resolve the fork repo so we can checkout correctly.
|
||||
- name: Resolve PR context
|
||||
id: pr
|
||||
uses: actions/github-script@60a0d83039c74a4aee543508d2ffcb1c3799cdea # v7
|
||||
uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v7
|
||||
with:
|
||||
script: |
|
||||
// Determine if this event is PR-related
|
||||
@@ -90,7 +91,7 @@ jobs:
|
||||
core.setOutput('branch', pr.head.ref);
|
||||
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4
|
||||
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
with:
|
||||
repository: ${{ steps.pr.outputs.is_pr == 'true' && steps.pr.outputs.repo || github.repository }}
|
||||
ref: ${{ steps.pr.outputs.is_pr == 'true' && steps.pr.outputs.sha || '' }}
|
||||
|
||||
@@ -8,8 +8,9 @@ on:
|
||||
permissions:
|
||||
pull-requests: write
|
||||
|
||||
# Concurrency convention: see CONTRIBUTING.md → "GitHub Actions — Concurrency Convention".
|
||||
concurrency:
|
||||
group: pr-desc-${{ github.event.pull_request.number }}
|
||||
group: ${{ github.workflow }}-${{ github.event.pull_request.number }}
|
||||
cancel-in-progress: true
|
||||
|
||||
jobs:
|
||||
@@ -18,7 +19,7 @@ jobs:
|
||||
timeout-minutes: 5
|
||||
steps:
|
||||
- name: Check PR description quality
|
||||
uses: actions/github-script@ed597411d8f924073f98dfc5c65a23a2325f34cd # v8
|
||||
uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v9.0.0
|
||||
with:
|
||||
script: |
|
||||
const MIN_BODY_LENGTH = 50;
|
||||
|
||||
@@ -0,0 +1,113 @@
|
||||
name: PR Conventional Labeler
|
||||
|
||||
# Two workflows in one file with different triggers, matched to the minimum
|
||||
# privilege each needs:
|
||||
#
|
||||
# validate-title (on: pull_request)
|
||||
# Fork-safe. Runs with the PR-head's read-only GITHUB_TOKEN. Uses
|
||||
# `amannn/action-semantic-pull-request` to fail the check when the PR
|
||||
# title doesn't follow the conventional-commit format. Because the
|
||||
# action only reads the event payload, no fork-controlled code runs.
|
||||
#
|
||||
# autolabel (on: pull_request_target)
|
||||
# Needs `pull-requests: write` to apply labels, so must be
|
||||
# pull_request_target. Uses `release-drafter/release-drafter` with
|
||||
# `dry-run: true` to only run the autolabeler against the
|
||||
# `.github/release-drafter.yml` config from the BASE ref (release-
|
||||
# drafter reads the config from the repository's default branch, NOT
|
||||
# the PR head — verify with `gh api repos/release-drafter/release-drafter/contents/...`
|
||||
# or a fork-test PR before merging if the repo is high-value).
|
||||
# `sync-labels: true` in the config removes managed autolabels that no
|
||||
# longer match (e.g. when `!` or `BREAKING CHANGE:` is dropped).
|
||||
#
|
||||
# Title format: <type>[(scope)][!]: <subject>
|
||||
# Allowed types: feat, fix, perf, refactor, docs, test, ci, build, chore, revert, deps
|
||||
# Trailing `!` on the type marks a breaking change.
|
||||
# See CONTRIBUTING.md → "Pull request titles".
|
||||
|
||||
on:
|
||||
pull_request:
|
||||
# Title-only changes fire `edited`. `opened` and `reopened` cover creation.
|
||||
# `synchronize` (push to the PR branch) is intentionally excluded — titles
|
||||
# don't change on push, so it only wastes CI minutes and broadens the
|
||||
# privileged-token exposure window on the autolabel job.
|
||||
types: [opened, edited, reopened]
|
||||
pull_request_target:
|
||||
types: [opened, edited, reopened]
|
||||
|
||||
# Concurrency convention: see CONTRIBUTING.md → "GitHub Actions — Concurrency Convention".
|
||||
# Include `github.event_name` so `pull_request` (validate-title) and
|
||||
# `pull_request_target` (autolabel) runs for the same PR do NOT share a slot
|
||||
# and therefore cannot cancel each other — a cancelled required-check would
|
||||
# permanently block merge until the next title edit.
|
||||
# Within each trigger the latest title edit still supersedes the prior run.
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.event_name }}-${{ github.event.pull_request.number }}
|
||||
cancel-in-progress: true
|
||||
|
||||
jobs:
|
||||
validate-title:
|
||||
# Fork-safe job — only runs on `pull_request` (not `pull_request_target`).
|
||||
# Token is read-only; writes a commit status that branch protection can
|
||||
# require before merge.
|
||||
name: Validate PR title
|
||||
if: github.event_name == 'pull_request'
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 5
|
||||
permissions:
|
||||
pull-requests: read
|
||||
steps:
|
||||
# Pinned to v6.1.1. Verify SHA via:
|
||||
# gh api repos/amannn/action-semantic-pull-request/git/refs/tags/v6.1.1
|
||||
- uses: amannn/action-semantic-pull-request@48f256284bd46cdaab1048c3721360e808335d50 # v6.1.1
|
||||
env:
|
||||
GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||
with:
|
||||
types: |
|
||||
feat
|
||||
fix
|
||||
perf
|
||||
refactor
|
||||
docs
|
||||
test
|
||||
ci
|
||||
build
|
||||
chore
|
||||
revert
|
||||
deps
|
||||
requireScope: false
|
||||
# Subject must be non-empty. We DO allow capitalized proper nouns
|
||||
# (MCP, GitHub, API, etc.) — the old `^(?![A-Z]).+$` pattern
|
||||
# rejected legitimate titles like `fix: MCP tool schema`.
|
||||
subjectPattern: ^\S.{2,}$
|
||||
subjectPatternError: |
|
||||
The subject "{subject}" in PR title "{title}" is invalid.
|
||||
Subjects must be at least 3 characters and must not start with whitespace.
|
||||
wip: false
|
||||
|
||||
autolabel:
|
||||
# Privileged job — runs only on `pull_request_target` so it can write labels.
|
||||
# Never checks out fork code, never executes fork-controlled input; only
|
||||
# reads the PR metadata (title, body, labels) and calls the GitHub API.
|
||||
name: Apply conventional label
|
||||
if: github.event_name == 'pull_request_target'
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 5
|
||||
permissions:
|
||||
# `contents: read` is required — release-drafter's context.config() reads
|
||||
# `.github/release-drafter.yml` from the repo's default branch via the
|
||||
# repo-contents API. Without it the job silently 403s and no labels are
|
||||
# applied. Job-level permissions nullify all unlisted scopes, so an
|
||||
# explicit grant is necessary here.
|
||||
contents: read
|
||||
pull-requests: write
|
||||
steps:
|
||||
# Pinned to v7.2.0. Verify SHA via:
|
||||
# gh api repos/release-drafter/release-drafter/git/refs/tags/v7.2.0
|
||||
# v7 removed `disable-releaser`; use `dry-run: true` to only autolabel.
|
||||
- uses: release-drafter/release-drafter@5de93583980a40bd78603b6dfdcda5b4df377b32 # v7.2.0
|
||||
with:
|
||||
config-name: release-drafter.yml
|
||||
dry-run: true
|
||||
env:
|
||||
GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||
@@ -7,13 +7,22 @@ on:
|
||||
|
||||
# No workflow-level permissions — scoped per job below.
|
||||
|
||||
# Concurrency convention: see CONTRIBUTING.md → "GitHub Actions — Concurrency Convention".
|
||||
# Tag refs are unique per release, so distinct tags run in parallel. Re-pushes of the
|
||||
# same tag serialize. cancel-in-progress: false — never cancel a publish mid-flight.
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.ref }}
|
||||
cancel-in-progress: false
|
||||
|
||||
jobs:
|
||||
ci:
|
||||
uses: ./.github/workflows/ci.yml
|
||||
permissions:
|
||||
contents: read
|
||||
actions: read
|
||||
pull-requests: write
|
||||
# No pull-requests:write — `ci.yml`'s save-pr-meta job is gated on
|
||||
# `github.event_name == 'pull_request'`, so it never runs during a
|
||||
# tag-triggered publish. Least-privilege for release-critical paths.
|
||||
|
||||
publish:
|
||||
needs: ci
|
||||
@@ -23,8 +32,8 @@ jobs:
|
||||
contents: write
|
||||
id-token: write
|
||||
steps:
|
||||
- uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4
|
||||
- uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4
|
||||
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
- uses: actions/setup-node@53b83947a5a98c8d113130e565377fae1a50d02f # v6.3.0
|
||||
with:
|
||||
node-version: 20
|
||||
registry-url: https://registry.npmjs.org
|
||||
@@ -82,7 +91,7 @@ jobs:
|
||||
fi
|
||||
|
||||
- name: Create GitHub Release
|
||||
uses: softprops/action-gh-release@a06a81a03ee405af7f2048a818ed3f03bbf83c7b # v2
|
||||
uses: softprops/action-gh-release@b4309332981a82ec1c5618f44dd2e27cc8bfbfda # v2
|
||||
with:
|
||||
body_path: ${{ steps.changelog.outputs.fallback == 'false' && '/tmp/release-notes.md' || '' }}
|
||||
generate_release_notes: ${{ steps.changelog.outputs.fallback == 'true' }}
|
||||
|
||||
@@ -0,0 +1,366 @@
|
||||
name: Release Candidate
|
||||
|
||||
on:
|
||||
# Publish a release-candidate build whenever a merge/commit lands on main.
|
||||
# Docs/README-only changes are filtered out so prose updates don't
|
||||
# cut a release.
|
||||
push:
|
||||
branches: [main]
|
||||
paths-ignore:
|
||||
- '**.md'
|
||||
- 'docs/**'
|
||||
- 'LICENSE'
|
||||
workflow_dispatch:
|
||||
inputs:
|
||||
bump:
|
||||
description: >-
|
||||
Cycle policy. 'auto' (default) continues the active rc cycle on
|
||||
this branch if there is one, otherwise bumps patch from latest.
|
||||
Choose 'patch' / 'minor' / 'major' to explicitly start or reset
|
||||
an rc cycle.
|
||||
required: false
|
||||
default: 'auto'
|
||||
type: choice
|
||||
options:
|
||||
- auto
|
||||
- patch
|
||||
- minor
|
||||
- major
|
||||
force:
|
||||
description: 'Publish even when HEAD already has an rc marker'
|
||||
required: false
|
||||
default: 'false'
|
||||
type: choice
|
||||
options:
|
||||
- 'false'
|
||||
- 'true'
|
||||
|
||||
# No workflow-level permissions — scoped per job below.
|
||||
permissions: {}
|
||||
|
||||
# Concurrency convention: see CONTRIBUTING.md → "GitHub Actions — Concurrency Convention".
|
||||
# Serialize all runs on the same ref (push + workflow_dispatch) to prevent two publishes
|
||||
# racing on the rc counter. cancel-in-progress: false — the earlier merge publishes first.
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.ref }}
|
||||
cancel-in-progress: false
|
||||
|
||||
jobs:
|
||||
# ── Skip when HEAD already has an rc marker (retry / duplicate dispatch) ──
|
||||
# The marker is a lightweight tag `rc/<HEAD_SHA>` pushed *before* `npm
|
||||
# publish`, so a failed publish leaves the marker in place and the guard
|
||||
# refuses to re-publish. Recovery path after a partial failure:
|
||||
# git push --delete origin rc/<HEAD_SHA> v<RC_VERSION>
|
||||
# then redispatch with force=true.
|
||||
guard:
|
||||
name: Check if release candidate should run
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 5
|
||||
permissions:
|
||||
contents: read
|
||||
outputs:
|
||||
should_run: ${{ steps.decide.outputs.should_run }}
|
||||
head_sha: ${{ steps.decide.outputs.head_sha }}
|
||||
steps:
|
||||
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
with:
|
||||
fetch-depth: 0
|
||||
fetch-tags: true
|
||||
|
||||
- name: Decide
|
||||
id: decide
|
||||
shell: bash
|
||||
env:
|
||||
FORCE: ${{ inputs.force }}
|
||||
BUMP_INPUT: ${{ inputs.bump }}
|
||||
EVENT_NAME: ${{ github.event_name }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
HEAD_SHA=$(git rev-parse HEAD)
|
||||
echo "head_sha=$HEAD_SHA" >> "$GITHUB_OUTPUT"
|
||||
|
||||
if [ "$FORCE" = "true" ]; then
|
||||
echo "Force flag set — running regardless of marker tag."
|
||||
echo "should_run=true" >> "$GITHUB_OUTPUT"
|
||||
exit 0
|
||||
fi
|
||||
|
||||
# An explicit cycle reset on dispatch (bump != auto) also bypasses
|
||||
# the dedup guard — the maintainer is deliberately asking for a
|
||||
# new rc from the same commit.
|
||||
if [ "$EVENT_NAME" = "workflow_dispatch" ] \
|
||||
&& [ -n "${BUMP_INPUT:-}" ] \
|
||||
&& [ "${BUMP_INPUT:-auto}" != "auto" ]; then
|
||||
echo "Explicit bump=$BUMP_INPUT — bypassing marker dedup."
|
||||
echo "should_run=true" >> "$GITHUB_OUTPUT"
|
||||
exit 0
|
||||
fi
|
||||
|
||||
# Dedup: is there already an rc/<HEAD_SHA> marker pointing at HEAD?
|
||||
MARKER="rc/${HEAD_SHA}"
|
||||
if git rev-parse "refs/tags/$MARKER" >/dev/null 2>&1; then
|
||||
echo "HEAD already has marker $MARKER — skipping."
|
||||
echo "should_run=false" >> "$GITHUB_OUTPUT"
|
||||
else
|
||||
echo "No marker on HEAD — proceeding."
|
||||
echo "should_run=true" >> "$GITHUB_OUTPUT"
|
||||
fi
|
||||
|
||||
# ── Reuse the stable CI workflow ─────────────────────────────────────
|
||||
ci:
|
||||
needs: guard
|
||||
if: needs.guard.outputs.should_run == 'true'
|
||||
uses: ./.github/workflows/ci.yml
|
||||
permissions:
|
||||
contents: read
|
||||
secrets: inherit
|
||||
|
||||
# ── Publish the rc build to npm + create GitHub prerelease ───────────
|
||||
publish:
|
||||
name: Publish release candidate to npm
|
||||
needs: [guard, ci]
|
||||
if: needs.guard.outputs.should_run == 'true'
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 20
|
||||
permissions:
|
||||
contents: write # push rc tag + marker
|
||||
id-token: write # npm provenance
|
||||
steps:
|
||||
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
with:
|
||||
fetch-depth: 0
|
||||
fetch-tags: true
|
||||
|
||||
- uses: actions/setup-node@53b83947a5a98c8d113130e565377fae1a50d02f # v6.3.0
|
||||
with:
|
||||
node-version: 20
|
||||
registry-url: https://registry.npmjs.org
|
||||
cache: npm
|
||||
cache-dependency-path: gitnexus/package-lock.json
|
||||
|
||||
- name: Build gitnexus-shared
|
||||
run: npm install && npm run build
|
||||
working-directory: gitnexus-shared
|
||||
|
||||
- name: Install gitnexus dependencies
|
||||
run: npm ci
|
||||
working-directory: gitnexus
|
||||
|
||||
- name: Resolve rc version
|
||||
id: version
|
||||
shell: bash
|
||||
working-directory: gitnexus
|
||||
env:
|
||||
BUMP_INPUT: ${{ inputs.bump }}
|
||||
EVENT_NAME: ${{ github.event_name }}
|
||||
PKG_NAME: gitnexus
|
||||
run: |
|
||||
set -euo pipefail
|
||||
|
||||
# 1. Current published `latest` — the floor for any new rc base.
|
||||
# Only E404 ("never published") falls back to package.json; any
|
||||
# other error (network, auth, malformed response) fails fast.
|
||||
NPM_STDERR_LATEST="$(mktemp)"
|
||||
if CURRENT_LATEST="$(npm view "$PKG_NAME" version 2>"$NPM_STDERR_LATEST")"; then
|
||||
:
|
||||
else
|
||||
if grep -q 'E404' "$NPM_STDERR_LATEST"; then
|
||||
CURRENT_LATEST="$(node -p "require('./package.json').version")"
|
||||
echo "Package not on registry (E404) — seeding from package.json: $CURRENT_LATEST"
|
||||
else
|
||||
echo "::error::npm registry unreachable for 'view version':" >&2
|
||||
cat "$NPM_STDERR_LATEST" >&2
|
||||
rm -f "$NPM_STDERR_LATEST"
|
||||
exit 1
|
||||
fi
|
||||
fi
|
||||
rm -f "$NPM_STDERR_LATEST"
|
||||
CURRENT_LATEST_CLEAN="${CURRENT_LATEST%%-*}"
|
||||
|
||||
# 2. Full version list — needed for the counter and for active-cycle
|
||||
# inference. Same E404-only fallback.
|
||||
NPM_STDERR_VERSIONS="$(mktemp)"
|
||||
if VERSIONS_JSON="$(npm view "$PKG_NAME" versions --json 2>"$NPM_STDERR_VERSIONS")"; then
|
||||
:
|
||||
else
|
||||
if grep -q 'E404' "$NPM_STDERR_VERSIONS"; then
|
||||
VERSIONS_JSON='[]'
|
||||
echo "No published versions for $PKG_NAME yet (E404)."
|
||||
else
|
||||
echo "::error::npm registry unreachable for 'view versions':" >&2
|
||||
cat "$NPM_STDERR_VERSIONS" >&2
|
||||
rm -f "$NPM_STDERR_VERSIONS"
|
||||
exit 1
|
||||
fi
|
||||
fi
|
||||
rm -f "$NPM_STDERR_VERSIONS"
|
||||
|
||||
# 3. Base selection.
|
||||
# - workflow_dispatch + bump ∈ {patch,minor,major} → explicit cycle
|
||||
# reset from latest.
|
||||
# - Everything else (push, or dispatch with bump=auto) → continue
|
||||
# the highest active rc base > latest if one exists; else
|
||||
# default to patch from latest.
|
||||
if [ "$EVENT_NAME" = "workflow_dispatch" ] \
|
||||
&& [ -n "${BUMP_INPUT:-}" ] \
|
||||
&& [ "${BUMP_INPUT:-auto}" != "auto" ]; then
|
||||
BASE="$(npx --yes -p semver@7 semver -i "$BUMP_INPUT" "$CURRENT_LATEST_CLEAN")"
|
||||
echo "Explicit bump=$BUMP_INPUT → BASE=$BASE"
|
||||
else
|
||||
cat > /tmp/active_base.mjs <<'NODESCRIPT'
|
||||
const latest = process.env.LATEST;
|
||||
let v;
|
||||
try { v = JSON.parse(process.env.VERSIONS_JSON); } catch { v = []; }
|
||||
if (!Array.isArray(v)) v = [v];
|
||||
const parse = s => s.split(".").map(n => parseInt(n, 10));
|
||||
const gt = (a, b) => {
|
||||
const [A, B] = [parse(a), parse(b)];
|
||||
for (let i = 0; i < 3; i++) if (A[i] !== B[i]) return A[i] > B[i];
|
||||
return false;
|
||||
};
|
||||
const bases = new Set();
|
||||
for (const s of v) {
|
||||
const m = /^(\d+\.\d+\.\d+)-rc\.\d+$/.exec(s);
|
||||
if (m && gt(m[1], latest)) bases.add(m[1]);
|
||||
}
|
||||
if (!bases.size) { process.stdout.write(""); process.exit(0); }
|
||||
const sorted = [...bases].sort((a, b) => gt(a, b) ? 1 : -1);
|
||||
process.stdout.write(sorted[sorted.length - 1]);
|
||||
NODESCRIPT
|
||||
ACTIVE_BASE="$(LATEST="$CURRENT_LATEST_CLEAN" VERSIONS_JSON="$VERSIONS_JSON" node /tmp/active_base.mjs)"
|
||||
if [ -n "$ACTIVE_BASE" ]; then
|
||||
BASE="$ACTIVE_BASE"
|
||||
echo "Continuing active rc cycle → BASE=$BASE"
|
||||
else
|
||||
BASE="$(npx --yes -p semver@7 semver -i patch "$CURRENT_LATEST_CLEAN")"
|
||||
echo "No active rc cycle → patch bump from latest → BASE=$BASE"
|
||||
fi
|
||||
fi
|
||||
|
||||
# 4. Counter: 1 + max existing N for `${BASE}-rc.*`, else 1.
|
||||
cat > /tmp/next_rc.mjs <<'NODESCRIPT'
|
||||
const base = process.env.BASE;
|
||||
const prefix = base + "-rc.";
|
||||
let v;
|
||||
try { v = JSON.parse(process.env.VERSIONS_JSON); } catch { v = []; }
|
||||
if (!Array.isArray(v)) v = [v];
|
||||
const ns = v
|
||||
.filter(s => typeof s === "string" && s.startsWith(prefix))
|
||||
.map(s => parseInt(s.slice(prefix.length), 10))
|
||||
.filter(n => Number.isInteger(n) && n >= 0);
|
||||
process.stdout.write(String(ns.length ? Math.max(...ns) + 1 : 1));
|
||||
NODESCRIPT
|
||||
NEXT_N="$(BASE="$BASE" VERSIONS_JSON="$VERSIONS_JSON" node /tmp/next_rc.mjs)"
|
||||
RC_VERSION="${BASE}-rc.${NEXT_N}"
|
||||
echo "Computed rc: $RC_VERSION"
|
||||
|
||||
# 5. Defensive: if the exact version already exists on the registry
|
||||
# (e.g., race with another run), abort before re-publishing.
|
||||
# Same E404-only pattern used above — a transient network
|
||||
# failure must fail loudly, not pretend the version is missing.
|
||||
NPM_STDERR_EXISTS="$(mktemp)"
|
||||
if npm view "$PKG_NAME@$RC_VERSION" version 2>"$NPM_STDERR_EXISTS" >/dev/null; then
|
||||
rm -f "$NPM_STDERR_EXISTS"
|
||||
echo "::error::Version $RC_VERSION already exists on npm — aborting."
|
||||
exit 1
|
||||
else
|
||||
if grep -qiE 'E404|not found' "$NPM_STDERR_EXISTS"; then
|
||||
rm -f "$NPM_STDERR_EXISTS"
|
||||
# Version doesn't exist — safe to proceed.
|
||||
else
|
||||
echo "::error::npm registry unreachable for existence check:" >&2
|
||||
cat "$NPM_STDERR_EXISTS" >&2
|
||||
rm -f "$NPM_STDERR_EXISTS"
|
||||
exit 1
|
||||
fi
|
||||
fi
|
||||
|
||||
echo "base=$BASE" >> "$GITHUB_OUTPUT"
|
||||
echo "rc_n=$NEXT_N" >> "$GITHUB_OUTPUT"
|
||||
echo "rc_version=$RC_VERSION" >> "$GITHUB_OUTPUT"
|
||||
|
||||
- name: Apply rc version in-CI
|
||||
shell: bash
|
||||
working-directory: gitnexus
|
||||
run: |
|
||||
set -euo pipefail
|
||||
npm version "${{ steps.version.outputs.rc_version }}" \
|
||||
--no-git-tag-version --allow-same-version
|
||||
|
||||
- name: Build gitnexus
|
||||
run: npm run build
|
||||
working-directory: gitnexus
|
||||
|
||||
- name: Dry-run publish
|
||||
run: npm publish --dry-run --tag rc
|
||||
working-directory: gitnexus
|
||||
|
||||
# ── Acquire the "rc lock" BEFORE publishing (fixes idempotency) ─────
|
||||
# We create two tags and push them atomically:
|
||||
# v<RC_VERSION> → annotated tag on a detached release commit
|
||||
# whose tree contains the rewritten package.json
|
||||
# (so the tag's source matches the npm tarball)
|
||||
# rc/<HEAD_SHA> → lightweight tag on HEAD; the guard's dedup key
|
||||
# If this push fails, nothing is published — safe.
|
||||
# If this push succeeds but npm publish fails, the marker stays on
|
||||
# the remote and blocks retries until an operator manually cleans up.
|
||||
- name: Create and push rc tags
|
||||
id: reltag
|
||||
shell: bash
|
||||
working-directory: gitnexus
|
||||
env:
|
||||
RC_VERSION: ${{ steps.version.outputs.rc_version }}
|
||||
HEAD_SHA: ${{ needs.guard.outputs.head_sha }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
VTAG="v${RC_VERSION}"
|
||||
MARKER="rc/${HEAD_SHA}"
|
||||
git config user.name 'github-actions[bot]'
|
||||
git config user.email '41898282+github-actions[bot]@users.noreply.github.com'
|
||||
|
||||
# Detached release commit with the version bump — keeps `main`
|
||||
# pristine but gives the v-tag a tree that matches the published
|
||||
# package contents exactly (fixes release-integrity gap).
|
||||
git add package.json package-lock.json 2>/dev/null || git add package.json
|
||||
git commit -m "release: ${VTAG}" --allow-empty
|
||||
RELEASE_SHA="$(git rev-parse HEAD)"
|
||||
echo "Detached release commit: $RELEASE_SHA"
|
||||
|
||||
# Annotated release tag on the release commit.
|
||||
git tag -a "$VTAG" "$RELEASE_SHA" -m "$VTAG"
|
||||
# Lightweight marker on the user-visible HEAD for the guard.
|
||||
git tag "$MARKER" "$HEAD_SHA"
|
||||
|
||||
# Atomic push of both refs. If either would clobber an existing
|
||||
# remote ref, the push fails and we stop before npm publish.
|
||||
git push --atomic origin "refs/tags/$VTAG" "refs/tags/$MARKER"
|
||||
|
||||
echo "vtag=$VTAG" >> "$GITHUB_OUTPUT"
|
||||
echo "marker=$MARKER" >> "$GITHUB_OUTPUT"
|
||||
echo "release_sha=$RELEASE_SHA" >> "$GITHUB_OUTPUT"
|
||||
|
||||
- name: Publish to npm (rc dist-tag)
|
||||
run: npm publish --provenance --access public --tag rc
|
||||
working-directory: gitnexus
|
||||
env:
|
||||
NODE_AUTH_TOKEN: ${{ secrets.NPM_TOKEN }}
|
||||
|
||||
- name: Create GitHub prerelease
|
||||
uses: softprops/action-gh-release@b4309332981a82ec1c5618f44dd2e27cc8bfbfda # v2
|
||||
with:
|
||||
tag_name: ${{ steps.reltag.outputs.vtag }}
|
||||
name: Release Candidate ${{ steps.reltag.outputs.vtag }}
|
||||
prerelease: true
|
||||
make_latest: 'false'
|
||||
generate_release_notes: true
|
||||
body: |
|
||||
Automated release candidate build from `main`.
|
||||
|
||||
**npm:** `npm install gitnexus@rc`
|
||||
**Version:** `${{ steps.version.outputs.rc_version }}`
|
||||
**Target base:** `${{ steps.version.outputs.base }}` (rc #${{ steps.version.outputs.rc_n }})
|
||||
**Source commit (main):** ${{ needs.guard.outputs.head_sha }}
|
||||
**Release commit (versioned tree):** ${{ steps.reltag.outputs.release_sha }}
|
||||
|
||||
Release candidates are pre-stable builds intended for early testing.
|
||||
Stable releases remain on the `latest` dist-tag.
|
||||
@@ -0,0 +1,185 @@
|
||||
name: Tree-sitter Upgrade Readiness
|
||||
|
||||
# Monitors readiness for upgrading tree-sitter to 0.25.x. Tracks:
|
||||
# 1. Peer-dep compatibility — can each grammar install cleanly with
|
||||
# tree-sitter@0.25.0 without --legacy-peer-deps?
|
||||
# 2. Vendored proto drift — has coder3101/tree-sitter-proto moved
|
||||
# ahead of our vendored snapshot?
|
||||
# See .github/scripts/check-tree-sitter-upgrade-readiness.py for the logic.
|
||||
#
|
||||
# Concurrency convention: see CONTRIBUTING.md → "GitHub Actions — Concurrency Convention".
|
||||
|
||||
on:
|
||||
schedule:
|
||||
# Daily at 09:00 UTC. Matches Dependabot's daily cadence so drift
|
||||
# and dep PRs surface together.
|
||||
- cron: '0 9 * * *'
|
||||
workflow_dispatch:
|
||||
pull_request:
|
||||
paths:
|
||||
- '.github/scripts/check-tree-sitter-upgrade-readiness.py'
|
||||
- '.github/workflows/tree-sitter-upgrade-readiness.yml'
|
||||
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.ref }}
|
||||
cancel-in-progress: ${{ github.event_name == 'pull_request' }}
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
jobs:
|
||||
readiness:
|
||||
name: Check upgrade readiness
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 10
|
||||
permissions:
|
||||
contents: read
|
||||
# Needed to open/update the tracking issue on scheduled runs.
|
||||
issues: write
|
||||
steps:
|
||||
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
|
||||
- uses: ./.github/actions/setup-gitnexus
|
||||
with:
|
||||
build: 'false'
|
||||
|
||||
- name: Run upgrade readiness check
|
||||
id: readiness
|
||||
shell: bash
|
||||
env:
|
||||
GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||
run: |
|
||||
set +e
|
||||
python3 .github/scripts/check-tree-sitter-upgrade-readiness.py > drift-report.md
|
||||
code=$?
|
||||
set -e
|
||||
echo "exit_code=$code" >> "$GITHUB_OUTPUT"
|
||||
{
|
||||
echo 'report<<DRIFT_EOF'
|
||||
cat drift-report.md
|
||||
echo 'DRIFT_EOF'
|
||||
} >> "$GITHUB_OUTPUT"
|
||||
echo "=== Report ==="
|
||||
cat drift-report.md
|
||||
|
||||
# On PR runs, the script validates that it runs correctly. Blockers
|
||||
# are informational — the scheduled run opens a tracking issue.
|
||||
- name: Annotate PR with readiness status
|
||||
if: github.event_name == 'pull_request' && steps.readiness.outputs.exit_code != '0'
|
||||
run: |
|
||||
echo "::warning::Tree-sitter 0.25 upgrade has blockers. See job output for the full readiness report."
|
||||
|
||||
- name: Upsert tracking issue on scheduled runs
|
||||
if: >
|
||||
github.event_name == 'schedule' &&
|
||||
steps.readiness.outputs.exit_code != '0'
|
||||
uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v9.0.0
|
||||
env:
|
||||
REPORT: ${{ steps.readiness.outputs.report }}
|
||||
with:
|
||||
script: |
|
||||
const title = 'Tree-sitter 0.25 upgrade readiness';
|
||||
const report = process.env.REPORT;
|
||||
const body = report + '\n\n' +
|
||||
'<sub>Generated daily by `.github/workflows/tree-sitter-upgrade-readiness.yml`. ' +
|
||||
'Closes automatically when all blockers are resolved.</sub>';
|
||||
const { data: open } = await github.rest.issues.listForRepo({
|
||||
owner: context.repo.owner,
|
||||
repo: context.repo.repo,
|
||||
state: 'open',
|
||||
labels: 'tree-sitter-drift',
|
||||
per_page: 10,
|
||||
});
|
||||
const existing = open.find(i => i.title === title);
|
||||
if (existing) {
|
||||
// Extract ready/total count for the changelog comment.
|
||||
const readyMatch = report.match(/\*\*(\d+)\/(\d+)\*\* grammars ready/);
|
||||
const blockerMatch = report.match(/\*\*(\d+) blocker/);
|
||||
const ready = readyMatch ? readyMatch[1] : '?';
|
||||
const total = readyMatch ? readyMatch[2] : '?';
|
||||
const blockers = blockerMatch ? blockerMatch[1] : '?';
|
||||
|
||||
// Find grammars whose status changed by diffing the old and
|
||||
// new table rows. Each row looks like:
|
||||
// | `tree-sitter-foo` | ... | Ready |
|
||||
// | `tree-sitter-foo` | ... | Blocking |
|
||||
const parseRows = (md) => {
|
||||
const map = {};
|
||||
for (const m of md.matchAll(/\| `(tree-sitter-[^`]+)` \|.*?\| (\S+(?:\s\S+)*?) \|$/gm)) {
|
||||
map[m[1]] = m[2].trim();
|
||||
}
|
||||
return map;
|
||||
};
|
||||
const oldRows = parseRows(existing.body || '');
|
||||
const newRows = parseRows(report);
|
||||
const changes = [];
|
||||
for (const [name, newStatus] of Object.entries(newRows)) {
|
||||
const oldStatus = oldRows[name];
|
||||
if (oldStatus && oldStatus !== newStatus) {
|
||||
changes.push(`\`${name}\`: ${oldStatus} → ${newStatus}`);
|
||||
}
|
||||
}
|
||||
|
||||
const today = new Date().toISOString().slice(0, 10);
|
||||
let comment = `**${today}:** ${ready}/${total} ready. ${blockers} blocker(s) remaining.`;
|
||||
if (changes.length > 0) {
|
||||
comment += '\n\nChanges:\n' + changes.map(c => `- ${c}`).join('\n');
|
||||
} else {
|
||||
comment += ' No changes from previous run.';
|
||||
}
|
||||
await github.rest.issues.createComment({
|
||||
owner: context.repo.owner,
|
||||
repo: context.repo.repo,
|
||||
issue_number: existing.number,
|
||||
body: comment,
|
||||
});
|
||||
|
||||
await github.rest.issues.update({
|
||||
owner: context.repo.owner,
|
||||
repo: context.repo.repo,
|
||||
issue_number: existing.number,
|
||||
body,
|
||||
});
|
||||
core.info(`Updated existing issue #${existing.number}`);
|
||||
} else {
|
||||
const { data: created } = await github.rest.issues.create({
|
||||
owner: context.repo.owner,
|
||||
repo: context.repo.repo,
|
||||
title,
|
||||
body,
|
||||
labels: ['tree-sitter-drift', 'dependencies'],
|
||||
});
|
||||
core.info(`Opened issue #${created.number}`);
|
||||
}
|
||||
|
||||
- name: Close tracking issue on clean scheduled runs
|
||||
if: >
|
||||
github.event_name == 'schedule' &&
|
||||
steps.readiness.outputs.exit_code == '0'
|
||||
uses: actions/github-script@3a2844b7e9c422d3c10d287c895573f7108da1b3 # v9.0.0
|
||||
with:
|
||||
script: |
|
||||
const title = 'Tree-sitter 0.25 upgrade readiness';
|
||||
const { data: open } = await github.rest.issues.listForRepo({
|
||||
owner: context.repo.owner,
|
||||
repo: context.repo.repo,
|
||||
state: 'open',
|
||||
labels: 'tree-sitter-drift',
|
||||
per_page: 10,
|
||||
});
|
||||
const existing = open.find(i => i.title === title);
|
||||
if (existing) {
|
||||
await github.rest.issues.createComment({
|
||||
owner: context.repo.owner,
|
||||
repo: context.repo.repo,
|
||||
issue_number: existing.number,
|
||||
body: 'All grammars are now compatible with tree-sitter@0.25. Upgrade is ready! Closing automatically.',
|
||||
});
|
||||
await github.rest.issues.update({
|
||||
owner: context.repo.owner,
|
||||
repo: context.repo.repo,
|
||||
issue_number: existing.number,
|
||||
state: 'closed',
|
||||
});
|
||||
core.info(`Closed issue #${existing.number}`);
|
||||
}
|
||||
@@ -47,8 +47,10 @@ permissions:
|
||||
issues: write
|
||||
pull-requests: write
|
||||
|
||||
# Concurrency convention: see CONTRIBUTING.md → "GitHub Actions — Concurrency Convention".
|
||||
# Single global slot — newest manual dispatch supersedes any in-flight run.
|
||||
concurrency:
|
||||
group: triage-sweep
|
||||
group: ${{ github.workflow }}
|
||||
cancel-in-progress: true
|
||||
|
||||
jobs:
|
||||
@@ -74,7 +76,7 @@ jobs:
|
||||
run: pip install -r .github/scripts/triage/requirements.txt
|
||||
|
||||
- name: Cache FastEmbed model weights
|
||||
uses: actions/cache@668228422ae6a00e4ad889ee87cd7109ec5666a7 # v5
|
||||
uses: actions/cache@27d5ce7f107fe9357f9df03efb73ab90386fccae # v5
|
||||
with:
|
||||
path: ${{ github.workspace }}/.fastembed_cache
|
||||
key: fastembed-bge-small-en-v1.5
|
||||
|
||||
@@ -81,6 +81,11 @@ GitNexus.sln
|
||||
# Git worktrees
|
||||
.worktrees/
|
||||
|
||||
# Vendored tree-sitter grammar build artifacts (created at install time,
|
||||
# never committed). See docs/plans/2026-04-15-002-fix-tree-sitter-proto-vendor-deps-plan.md
|
||||
gitnexus/vendor/**/build/
|
||||
gitnexus/vendor/**/node_modules/
|
||||
|
||||
/github/scripts/triage/__pycache__/
|
||||
|
||||
.claude-flow/
|
||||
|
||||
@@ -1,117 +1,120 @@
|
||||
<!-- version: 1.3.0 -->
|
||||
<!--
|
||||
Metadata: version, last reviewed, scope, model policy, reference docs, changelog.
|
||||
Last updated: 2026-03-22
|
||||
-->
|
||||
<!-- version: 1.4.0 -->
|
||||
<!-- Last updated: 2026-04-16 -->
|
||||
|
||||
Last reviewed: 2026-04-13
|
||||
Last reviewed: 2026-04-16
|
||||
|
||||
**Project:** GitNexus · **Environment:** dev · **Maintainer:** repository maintainers (see GitHub)
|
||||
|
||||
This file uses a standard agent header (version, scope, model policy, reference docs, changelog), adapted for this **TypeScript/JavaScript monorepo**.
|
||||
|
||||
## Scope
|
||||
|
||||
| | |
|
||||
|--|--|
|
||||
| **Reads** | Repository tree as needed for the task: `gitnexus/`, `gitnexus-web/`, `eval/`, plugin packages, `.github/`, `.gitnexus/` when present, and docs. |
|
||||
| **Writes** | Only paths required for the requested change; keep diffs minimal. Update lockfiles when dependencies change. |
|
||||
| **Executes** | `npm`, `npx`, `node` under `gitnexus/` and `gitnexus-web/`; `uv run` for Python under `eval/` when applicable; shell utilities for documented CI/dev workflows. |
|
||||
| **Off-limits** | User secrets (e.g. real `.env`), production deployment credentials, unrelated repositories, destructive git history operations without explicit human confirmation. |
|
||||
| Boundary | Rule |
|
||||
|----------|------|
|
||||
| **Reads** | `gitnexus/`, `gitnexus-web/`, `eval/`, plugin packages, `.github/`, `.gitnexus/`, docs. |
|
||||
| **Writes** | Only paths required for the change; keep diffs minimal. Update lockfiles when deps change. |
|
||||
| **Executes** | `npm`, `npx`, `node` under `gitnexus/` and `gitnexus-web/`; `uv run` for Python under `eval/`; documented CI/dev workflows. |
|
||||
| **Off-limits** | Real `.env` / secrets, production credentials, unrelated repos, destructive git ops without confirmation. |
|
||||
|
||||
## Model Configuration
|
||||
|
||||
- **Primary:** Pin in **Cursor** (Settings → model). Use a **named** model (e.g. GPT-5.2, Claude Sonnet 4.x). Avoid relying on **Auto** when reproducibility or audit trail matters.
|
||||
- **Fallback:** As configured in Cursor or your organization (do not encode `latest` or wildcards in automation configs).
|
||||
- **Notes:** The open-source GitNexus CLI indexer does not call an LLM. Optional Nexus AI in the web UI uses end-user provider keys and models.
|
||||
- **Primary:** Use a named model (e.g. Claude Sonnet 4.x). Avoid `Auto` or unversioned `latest` when reproducibility matters.
|
||||
- **Notes:** The GitNexus CLI indexer does not call an LLM.
|
||||
|
||||
## Execution Sequence (complex tasks)
|
||||
|
||||
Long sessions dilute instructions. For **multi-step** work, state up front:
|
||||
|
||||
For multi-step work, state up front:
|
||||
1. Which rules in this file and **[GUARDRAILS.md](GUARDRAILS.md)** apply (and any relevant Signs).
|
||||
2. Current **Scope** boundaries (Reads / Writes / Off-limits).
|
||||
3. Which **validation commands** you will run (e.g. `cd gitnexus && npm test`, `npx tsc --noEmit`).
|
||||
2. Current **Scope** boundaries.
|
||||
3. Which **validation commands** you will run (`cd gitnexus && npm test`, `npx tsc --noEmit`).
|
||||
|
||||
On very long threads, the human may add *“Remember: apply all AGENTS.md rules”* to re-weight rule tokens against context dilution.
|
||||
On long threads, *"Remember: apply all AGENTS.md rules"* re-weights these instructions against context dilution.
|
||||
|
||||
## Claude Code hooks
|
||||
|
||||
Hooks enforce gates that prompts cannot. In **Claude Code**, **PreToolUse** hooks can block tools such as `git_commit` until checks pass. Adapt to this repo: e.g. `cd gitnexus && npm test` before commit.
|
||||
**PreToolUse** hooks can block tools (e.g. `git_commit`) until checks pass. Adapt to this repo: `cd gitnexus && npm test` before commit.
|
||||
|
||||
## Context budget (Cursor / standards)
|
||||
## Context budget
|
||||
|
||||
Generic “core standards” playbooks are often long and stack-specific. For this monorepo, commands and gotchas live under **Cursor Cloud specific instructions** below and in **[CONTRIBUTING.md](CONTRIBUTING.md)**. If always-on rules grow, split domain rules into **`.cursor/rules/*.mdc`** (globs). **Cursor:** project-wide rules live in **`.cursor/index.mdc`** (YAML frontmatter with `alwaysApply: true`). **Claude Code:** optionally load a **`STANDARDS.md`** only when needed (e.g. *“When writing new code, read STANDARDS.md”*) to save context.
|
||||
Commands and gotchas live under **Repo reference** below and in **[CONTRIBUTING.md](CONTRIBUTING.md)**. If always-on rules grow, split into **`.cursor/rules/*.mdc`** (globs). **Cursor:** project-wide rules in `.cursor/index.mdc`. **Claude Code:** load `STANDARDS.md` only when needed.
|
||||
|
||||
## Reference Documentation
|
||||
## Reference docs
|
||||
|
||||
- **This repository:** **[ARCHITECTURE.md](ARCHITECTURE.md)**, **[CONTRIBUTING.md](CONTRIBUTING.md)**, **[GUARDRAILS.md](GUARDRAILS.md)**.
|
||||
- **Cursor:** `.cursor/index.mdc` (always-on rules); optional `.cursor/rules/*.mdc` (glob-scoped). Legacy `.cursorrules` is deprecated — see `.cursor/index.mdc`.
|
||||
- **Optional local files:** `NOTES.md` (short vendor-neutral project snapshot). For handoffs, keep notes local (e.g., a scratch file outside the repo) rather than committing `HANDOFF.md`.
|
||||
- **GitNexus:** skills under `.claude/skills/gitnexus/`; machine-oriented rules in the `gitnexus:start` … `gitnexus:end` block below.
|
||||
- **[ARCHITECTURE.md](ARCHITECTURE.md)**, **[CONTRIBUTING.md](CONTRIBUTING.md)**, **[GUARDRAILS.md](GUARDRAILS.md)**
|
||||
- **Cursor:** `.cursor/index.mdc` (always-on); `.cursor/rules/*.mdc` (glob-scoped). Legacy `.cursorrules` deprecated.
|
||||
- **GitNexus:** skills in `.claude/skills/gitnexus/`; MCP rules in `gitnexus:start` block below.
|
||||
|
||||
## Changelog
|
||||
|
||||
| Date | Version | Change |
|
||||
|------|---------|--------|
|
||||
| 2026-04-16 | 1.4.0 | Fixed: web UI description, pre-commit behavior, MCP tools (7->16), added gitnexus-shared, removed stale vite-plugin-wasm gotcha. |
|
||||
| 2026-04-13 | 1.3.0 | Updated GitNexus index stats after DAG refactor. |
|
||||
| 2026-03-24 | 1.2.0 | Fixed gitnexus:start block duplication (was inlined in Reference Docs bullet). |
|
||||
| 2026-03-23 | 1.1.0 | Updated agent instructions (sections, references, Cursor layout). |
|
||||
| 2026-03-22 | 1.0.0 | Added structured agent header and changelog. |
|
||||
| 2026-03-24 | 1.2.0 | Fixed gitnexus:start block duplication. |
|
||||
| 2026-03-23 | 1.1.0 | Updated agent instructions, references, Cursor layout. |
|
||||
| 2026-03-22 | 1.0.0 | Initial structured header and changelog. |
|
||||
|
||||
---
|
||||
|
||||
<!-- gitnexus:start -->
|
||||
# GitNexus — Code Intelligence
|
||||
|
||||
This project is indexed by GitNexus as **GitNexus** (4325 symbols, 10556 relationships, 300 execution flows). Use the GitNexus MCP tools to understand code, assess impact, and navigate safely.
|
||||
Indexed as **GitNexus** (4325 symbols, 10556 relationships, 300 execution flows). Use MCP tools to understand code, assess impact, and navigate safely.
|
||||
|
||||
> If any GitNexus tool warns the index is stale, run `npx gitnexus analyze` in terminal first.
|
||||
> If any tool warns the index is stale, run `npx gitnexus analyze` first.
|
||||
|
||||
## Always Do
|
||||
|
||||
- **MUST run impact analysis before editing any symbol.** Before modifying a function, class, or method, run `gitnexus_impact({target: "symbolName", direction: "upstream"})` and report the blast radius (direct callers, affected processes, risk level) to the user.
|
||||
- **MUST run `gitnexus_detect_changes()` before committing** to verify your changes only affect expected symbols and execution flows.
|
||||
- **MUST warn the user** if impact analysis returns HIGH or CRITICAL risk before proceeding with edits.
|
||||
- When exploring unfamiliar code, use `gitnexus_query({query: "concept"})` to find execution flows instead of grepping. It returns process-grouped results ranked by relevance.
|
||||
- When you need full context on a specific symbol — callers, callees, which execution flows it participates in — use `gitnexus_context({name: "symbolName"})`.
|
||||
- **MUST run impact analysis before editing any symbol.** `gitnexus_impact({target: "symbolName", direction: "upstream"})` — report blast radius to the user.
|
||||
- **MUST run `gitnexus_detect_changes()` before committing** — verify only expected symbols and flows are affected.
|
||||
- **MUST warn the user** if impact returns HIGH or CRITICAL risk.
|
||||
- Explore unfamiliar code with `gitnexus_query({query: "concept"})` (process-grouped, ranked) instead of grepping.
|
||||
- Full context on a symbol: `gitnexus_context({name: "symbolName"})`.
|
||||
|
||||
## When Debugging
|
||||
|
||||
1. `gitnexus_query({query: "<error or symptom>"})` — find execution flows related to the issue
|
||||
2. `gitnexus_context({name: "<suspect function>"})` — see all callers, callees, and process participation
|
||||
3. `READ gitnexus://repo/GitNexus/process/{processName}` — trace the full execution flow step by step
|
||||
4. For regressions: `gitnexus_detect_changes({scope: "compare", base_ref: "main"})` — see what your branch changed
|
||||
1. `gitnexus_query({query: "<error or symptom>"})` — find related execution flows
|
||||
2. `gitnexus_context({name: "<suspect function>"})` — callers, callees, process participation
|
||||
3. `READ gitnexus://repo/GitNexus/process/{processName}` — trace flow step by step
|
||||
4. Regressions: `gitnexus_detect_changes({scope: "compare", base_ref: "main"})`
|
||||
|
||||
## When Refactoring
|
||||
|
||||
- **Renaming**: MUST use `gitnexus_rename({symbol_name: "old", new_name: "new", dry_run: true})` first. Review the preview — graph edits are safe, text_search edits need manual review. Then run with `dry_run: false`.
|
||||
- **Extracting/Splitting**: MUST run `gitnexus_context({name: "target"})` to see all incoming/outgoing refs, then `gitnexus_impact({target: "target", direction: "upstream"})` to find all external callers before moving code.
|
||||
- After any refactor: run `gitnexus_detect_changes({scope: "all"})` to verify only expected files changed.
|
||||
- **Rename:** `gitnexus_rename({symbol_name: "old", new_name: "new", dry_run: true})` first. Graph edits are safe; text_search edits need manual review.
|
||||
- **Extract/Split:** `gitnexus_context` (incoming/outgoing refs) then `gitnexus_impact` (upstream callers) before moving code.
|
||||
- **After any refactor:** `gitnexus_detect_changes({scope: "all"})` to verify scope.
|
||||
|
||||
## Never Do
|
||||
|
||||
- NEVER edit a function, class, or method without first running `gitnexus_impact` on it.
|
||||
- NEVER ignore HIGH or CRITICAL risk warnings from impact analysis.
|
||||
- NEVER rename symbols with find-and-replace — use `gitnexus_rename` which understands the call graph.
|
||||
- NEVER commit changes without running `gitnexus_detect_changes()` to check affected scope.
|
||||
- Edit a symbol without running `gitnexus_impact` first.
|
||||
- Ignore HIGH/CRITICAL risk warnings.
|
||||
- Rename with find-and-replace — use `gitnexus_rename`.
|
||||
- Commit without `gitnexus_detect_changes()`.
|
||||
|
||||
## Tools Quick Reference
|
||||
|
||||
| Tool | When to use | Command |
|
||||
| Tool | When to use | Example |
|
||||
|------|-------------|---------|
|
||||
| `list_repos` | Discover indexed repos | `gitnexus_list_repos({})` |
|
||||
| `query` | Find code by concept | `gitnexus_query({query: "auth validation"})` |
|
||||
| `context` | 360-degree view of one symbol | `gitnexus_context({name: "validateUser"})` |
|
||||
| `impact` | Blast radius before editing | `gitnexus_impact({target: "X", direction: "upstream"})` |
|
||||
| `detect_changes` | Pre-commit scope check | `gitnexus_detect_changes({scope: "staged"})` |
|
||||
| `rename` | Safe multi-file rename | `gitnexus_rename({symbol_name: "old", new_name: "new", dry_run: true})` |
|
||||
| `cypher` | Custom graph queries | `gitnexus_cypher({query: "MATCH ..."})` |
|
||||
| `api_impact` | Pre-change API route impact | `gitnexus_api_impact({route: "/api/users", method: "GET"})` |
|
||||
| `route_map` | Route → handler → consumer map | `gitnexus_route_map({})` |
|
||||
| `tool_map` | MCP/RPC tool definitions | `gitnexus_tool_map({})` |
|
||||
| `shape_check` | Response shape vs consumer access | `gitnexus_shape_check({route: "/api/users"})` |
|
||||
| `group_list` | List repo groups | `gitnexus_group_list({})` |
|
||||
| `group_query` | Cross-repo search in a group | `gitnexus_group_query({name: "myGroup", query: "auth"})` |
|
||||
| `group_sync` | Rebuild group Contract Registry | `gitnexus_group_sync({name: "myGroup"})` |
|
||||
| `group_contracts` | Inspect group contracts | `gitnexus_group_contracts({name: "myGroup"})` |
|
||||
| `group_status` | Group staleness report | `gitnexus_group_status({name: "myGroup"})` |
|
||||
|
||||
## Impact Risk Levels
|
||||
|
||||
| Depth | Meaning | Action |
|
||||
|-------|---------|--------|
|
||||
| d=1 | WILL BREAK — direct callers/importers | MUST update these |
|
||||
| d=1 | WILL BREAK — direct callers/importers | MUST update |
|
||||
| d=2 | LIKELY AFFECTED — indirect deps | Should test |
|
||||
| d=3 | MAY NEED TESTING — transitive | Test if critical path |
|
||||
|
||||
@@ -119,87 +122,80 @@ This project is indexed by GitNexus as **GitNexus** (4325 symbols, 10556 relatio
|
||||
|
||||
| Resource | Use for |
|
||||
|----------|---------|
|
||||
| `gitnexus://repo/GitNexus/context` | Codebase overview, check index freshness |
|
||||
| `gitnexus://repo/GitNexus/context` | Codebase overview, index freshness |
|
||||
| `gitnexus://repo/GitNexus/clusters` | All functional areas |
|
||||
| `gitnexus://repo/GitNexus/processes` | All execution flows |
|
||||
| `gitnexus://repo/GitNexus/process/{name}` | Step-by-step execution trace |
|
||||
|
||||
## Self-Check Before Finishing
|
||||
|
||||
Before completing any code modification task, verify:
|
||||
1. `gitnexus_impact` was run for all modified symbols
|
||||
2. No HIGH/CRITICAL risk warnings were ignored
|
||||
3. `gitnexus_detect_changes()` confirms changes match expected scope
|
||||
4. All d=1 (WILL BREAK) dependents were updated
|
||||
2. No HIGH/CRITICAL warnings were ignored
|
||||
3. `gitnexus_detect_changes()` confirms expected scope
|
||||
4. All d=1 dependents were updated
|
||||
|
||||
## Keeping the Index Fresh
|
||||
|
||||
After committing code changes, the GitNexus index becomes stale. Re-run analyze to update it:
|
||||
|
||||
```bash
|
||||
npx gitnexus analyze
|
||||
npx gitnexus analyze # basic refresh
|
||||
npx gitnexus analyze --embeddings # preserve embeddings
|
||||
```
|
||||
|
||||
If the index previously included embeddings, preserve them by adding `--embeddings`:
|
||||
Check `.gitnexus/meta.json` `stats.embeddings` (0 = none). Running without `--embeddings` deletes existing vectors.
|
||||
|
||||
```bash
|
||||
npx gitnexus analyze --embeddings
|
||||
```
|
||||
> Claude Code: PostToolUse hook handles this after `git commit` and `git merge`.
|
||||
|
||||
To check whether embeddings exist, inspect `.gitnexus/meta.json` — the `stats.embeddings` field shows the count (0 means no embeddings). **Running analyze without `--embeddings` will delete any previously generated embeddings.**
|
||||
## CLI Skills
|
||||
|
||||
> Claude Code users: A PostToolUse hook handles this automatically after `git commit` and `git merge`.
|
||||
|
||||
## CLI
|
||||
|
||||
| Task | Read this skill file |
|
||||
|------|---------------------|
|
||||
| Understand architecture / "How does X work?" | `.claude/skills/gitnexus/gitnexus-exploring/SKILL.md` |
|
||||
| Blast radius / "What breaks if I change X?" | `.claude/skills/gitnexus/gitnexus-impact-analysis/SKILL.md` |
|
||||
| Trace bugs / "Why is X failing?" | `.claude/skills/gitnexus/gitnexus-debugging/SKILL.md` |
|
||||
| Rename / extract / split / refactor | `.claude/skills/gitnexus/gitnexus-refactoring/SKILL.md` |
|
||||
| Tools, resources, schema reference | `.claude/skills/gitnexus/gitnexus-guide/SKILL.md` |
|
||||
| Index, status, clean, wiki CLI commands | `.claude/skills/gitnexus/gitnexus-cli/SKILL.md` |
|
||||
| Task | Skill file |
|
||||
|------|-----------|
|
||||
| Architecture / "How does X work?" | `.claude/skills/gitnexus/gitnexus-exploring/SKILL.md` |
|
||||
| Blast radius / "What breaks?" | `.claude/skills/gitnexus/gitnexus-impact-analysis/SKILL.md` |
|
||||
| Debugging / "Why is X failing?" | `.claude/skills/gitnexus/gitnexus-debugging/SKILL.md` |
|
||||
| Refactoring | `.claude/skills/gitnexus/gitnexus-refactoring/SKILL.md` |
|
||||
| Tools/resources/schema reference | `.claude/skills/gitnexus/gitnexus-guide/SKILL.md` |
|
||||
| CLI commands (index, status, clean, wiki) | `.claude/skills/gitnexus/gitnexus-cli/SKILL.md` |
|
||||
|
||||
<!-- gitnexus:end -->
|
||||
|
||||
## Cursor Cloud specific instructions
|
||||
## Repo reference
|
||||
|
||||
### Repository structure
|
||||
### Packages
|
||||
|
||||
This is a monorepo with two main products and supporting config packages:
|
||||
|
||||
| Component | Path | Purpose |
|
||||
|-----------|------|---------|
|
||||
| **GitNexus CLI/Core** | `gitnexus/` | Main product — TypeScript CLI, indexing pipeline, MCP server. Published to npm. |
|
||||
| **GitNexus Web UI** | `gitnexus-web/` | React/Vite browser app — graph explorer + AI chat. Runs entirely in WASM. |
|
||||
| Claude Plugin | `gitnexus-claude-plugin/` | Static config for Claude marketplace (no build). |
|
||||
| Cursor Integration | `gitnexus-cursor-integration/` | Static config for Cursor editor (no build). |
|
||||
| SWE-bench Eval | `eval/` | Python evaluation harness (optional; needs Docker + LLM API keys). |
|
||||
| Package | Path | Purpose |
|
||||
|---------|------|---------|
|
||||
| **CLI/Core** | `gitnexus/` | TypeScript CLI, indexing pipeline, MCP server. Published to npm. |
|
||||
| **Web UI** | `gitnexus-web/` | React/Vite thin client. All queries via `gitnexus serve` HTTP API. |
|
||||
| **Shared** | `gitnexus-shared/` | Shared TypeScript types and constants. |
|
||||
| Claude Plugin | `gitnexus-claude-plugin/` | Static config for Claude marketplace. |
|
||||
| Cursor Integration | `gitnexus-cursor-integration/` | Static config for Cursor editor. |
|
||||
| Eval | `eval/` | Python evaluation harness (Docker + LLM API keys). |
|
||||
|
||||
### Running services
|
||||
|
||||
- **CLI/Core**: `cd gitnexus && npm run dev` (tsx watch mode) or `npm run build && node dist/cli/index.js <command>`
|
||||
- **Web UI**: `cd gitnexus-web && npm run dev` (Vite on port 5173)
|
||||
- **Backend mode**: `cd <indexed-repo> && node /workspace/gitnexus/dist/cli/index.js serve` (HTTP API on port 3741 by default)
|
||||
```bash
|
||||
cd gitnexus && npm run dev # CLI: tsx watch mode
|
||||
cd gitnexus-web && npm run dev # Web UI: Vite on port 5173
|
||||
npx gitnexus serve # HTTP API on port 4747 (from any indexed repo)
|
||||
```
|
||||
|
||||
### Testing
|
||||
|
||||
**CLI / Core (`gitnexus/`)**
|
||||
- **Unit tests**: `cd gitnexus && npm test` (vitest, ~2000 tests)
|
||||
- **Integration tests**: `cd gitnexus && npm run test:integration` (vitest, ~1850 tests). Two LadybugDB file-locking tests (`lbug-core-adapter`, `search-core`) may fail in containerized environments due to `/tmp` locking limitations — this is a known environment issue, not a code bug.
|
||||
- **TypeScript check**: `cd gitnexus && npx tsc --noEmit`
|
||||
- `npm test` — full vitest suite (~2000 tests)
|
||||
- `npm run test:unit` — unit tests only
|
||||
- `npm run test:integration` — integration (~1850 tests). LadybugDB file-locking tests may fail in containers (known env issue).
|
||||
- `npx tsc --noEmit` — typecheck
|
||||
|
||||
**Web UI (`gitnexus-web/`)**
|
||||
- **Unit tests**: `cd gitnexus-web && npm test` (vitest, ~200 tests)
|
||||
- **E2E tests**: `cd gitnexus-web && E2E=1 npx playwright test` (Playwright, 5 tests — requires `gitnexus serve` + `npm run dev` running)
|
||||
- **TypeScript check**: `cd gitnexus-web && npx tsc -b --noEmit`
|
||||
- `npm test` — vitest (~200 tests)
|
||||
- `npm run test:e2e` — Playwright (7 spec files; requires `gitnexus serve` + `npm run dev`)
|
||||
- `npx tsc -b --noEmit` — typecheck
|
||||
|
||||
No separate lint command is configured; TypeScript strict checking serves as the primary static analysis.
|
||||
**Pre-commit hook** (`.husky/pre-commit`): formatting (prettier via lint-staged) + typecheck for staged packages. Tests do **not** run in pre-commit — CI only.
|
||||
|
||||
### Gotchas
|
||||
|
||||
- `npm install` in `gitnexus/` triggers `prepare` (builds via `tsc`) and `postinstall` (patches tree-sitter-swift). Native tree-sitter bindings require `python3`, `make`, and `g++` to be present.
|
||||
- `tree-sitter-kotlin` and `tree-sitter-swift` are optional dependencies — install warnings for these are expected and non-blocking.
|
||||
- The Web UI uses `vite-plugin-wasm` and requires `Cross-Origin-Opener-Policy`/`Cross-Origin-Embedder-Policy` headers for `SharedArrayBuffer` (handled automatically by Vite dev server).
|
||||
- There is no ESLint/Prettier configuration in this repo.
|
||||
- `npm install` in `gitnexus/` triggers `prepare` (builds via `tsc`) and `postinstall` (patches tree-sitter-swift, builds tree-sitter-proto). Native bindings need `python3`, `make`, `g++`.
|
||||
- `tree-sitter-kotlin` and `tree-sitter-swift` are optional — install warnings expected.
|
||||
- ESLint configured via `eslint.config.mjs` (TS, React Hooks, unused-imports). No `npm run lint` script; use `npx eslint .`. Prettier runs via lint-staged. CI checks both in `ci-quality.yml`.
|
||||
|
||||
+233
-116
@@ -1,99 +1,129 @@
|
||||
# Architecture — GitNexus
|
||||
|
||||
This repository is a **monorepo** with two main products: the **CLI / MCP package** (`gitnexus/`) and the **browser UI** (`gitnexus-web/`). Supporting folders ship editor integrations and plugins without changing the core graph engine.
|
||||
Monorepo: **CLI/MCP** (`gitnexus/`) + **browser UI** (`gitnexus-web/`).
|
||||
|
||||
## Repository layout
|
||||
|
||||
| Path | Role |
|
||||
|------|------|
|
||||
| `gitnexus/` | Published npm package `gitnexus`: CLI, MCP server (stdio), local HTTP API for bridge mode, ingestion pipeline, LadybugDB graph, embeddings (optional). |
|
||||
| `gitnexus-web/` | Vite + React UI: in-browser indexing (WASM), graph visualization, optional connection to `gitnexus serve`. |
|
||||
| `.claude/`, `gitnexus-claude-plugin/`, `gitnexus-cursor-integration/` | Packaged **skills** and plugin metadata so agents discover the same workflows as documented in `AGENTS.md`. |
|
||||
| `eval/` | Evaluation harnesses and docs for benchmarking tool usage. |
|
||||
| `.github/` | CI workflows (quality, unit, integration, E2E) and composite actions. |
|
||||
| `gitnexus/` | npm package `gitnexus`: CLI, MCP server (stdio), HTTP API, ingestion pipeline, LadybugDB graph, embeddings. |
|
||||
| `gitnexus-web/` | Vite + React thin client: graph explorer + AI chat. All queries via `gitnexus serve` HTTP API. |
|
||||
| `gitnexus-shared/` | Shared TypeScript types and constants (consumed by CLI and Web). |
|
||||
| `.claude/`, `gitnexus-claude-plugin/`, `gitnexus-cursor-integration/` | Agent skills and plugin metadata. |
|
||||
| `eval/` | Evaluation harnesses for benchmarking tool usage. |
|
||||
| `.github/` | CI workflows + composite actions (`setup-gitnexus/`, `setup-gitnexus-web/`). |
|
||||
|
||||
## End-to-end flow: index → graph → tools
|
||||
|
||||
1. **Ingestion** (`gitnexus analyze`)
|
||||
- Entry: `gitnexus/src/cli/analyze.ts` → `runPipelineFromRepo` in `gitnexus/src/core/ingestion/pipeline.ts`.
|
||||
- The pipeline is structured as a **DAG (Directed Acyclic Graph)** of named phases (see [Pipeline Phase DAG](#pipeline-phase-dag) below).
|
||||
- Output is loaded into **LadybugDB** under **`.gitnexus/`** at the repo root (`lbug/`, `meta.json`, etc.). Optional **FTS** indexes and **embeddings** attach to the same store.
|
||||
- The repo is registered in **`~/.gitnexus/registry.json`** so MCP can find it from any working directory.
|
||||
1. **Ingestion** — `analyze.ts` → `runFullAnalysis` (`run-analyze.ts`) → `runPipelineFromRepo` (`pipeline.ts`). DAG of 12 phases builds a `KnowledgeGraph` in memory, then loads into LadybugDB under `.gitnexus/`. Repo registered in `~/.gitnexus/registry.json` for MCP discovery.
|
||||
|
||||
2. **Persistence & metadata**
|
||||
- `gitnexus/src/storage/repo-manager.ts` — paths, registry, cleanup of legacy Kuzu artifacts.
|
||||
- `gitnexus/src/core/lbug/lbug-adapter.ts` — graph load, queries, embedding restore batches.
|
||||
2. **Persistence** — `repo-manager.ts` (paths, registry, KuzuDB cleanup). `lbug-adapter.ts` (graph load, queries, embedding batches).
|
||||
|
||||
3. **Query & agents**
|
||||
- **MCP (stdio):** `gitnexus/src/cli/mcp.ts` → `startMCPServer` → `LocalBackend` (`gitnexus/src/mcp/local/local-backend.ts`) opens registered repos and serves **tools** from `gitnexus/src/mcp/tools.ts` and **resources** from `gitnexus/src/mcp/resources.ts`.
|
||||
- **Bridge HTTP:** `gitnexus/src/cli/serve.ts` → Express app in `gitnexus/src/server/api.ts` (CORS-limited) exposes REST + MCP-over-HTTP for the web UI.
|
||||
- **CLI tools (no MCP):** `gitnexus query`, `context`, `impact`, `cypher` in `gitnexus/src/cli/tool.ts` call the same backend for scripts and CI.
|
||||
3. **Query layer** — three interfaces to the same backend:
|
||||
- **MCP (stdio):** `mcp.ts` → `LocalBackend` → tools (`tools.ts`) + resources (`resources.ts`)
|
||||
- **HTTP bridge:** `serve.ts` → Express (`api.ts`, `mcp-http.ts`) for web UI
|
||||
- **CLI direct:** `gitnexus query|context|impact|cypher` in `tool.ts`
|
||||
|
||||
4. **Staleness**
|
||||
- `gitnexus/src/mcp/staleness.ts` compares indexed `lastCommit` to `HEAD` and surfaces hints when the graph is behind git.
|
||||
4. **Staleness** — `staleness.ts` compares indexed `lastCommit` to `HEAD`, surfaces hints.
|
||||
|
||||
## MCP tools (summary)
|
||||
## MCP tools
|
||||
|
||||
| Tool | Purpose |
|
||||
|------|---------|
|
||||
| `list_repos` | Discover indexed repositories when more than one is registered. |
|
||||
| `query` | Natural-language / keyword search over the graph (hybrid BM25 + optional vectors). |
|
||||
| `cypher` | Ad hoc **Cypher** against the schema (see resource `gitnexus://repo/{name}/schema`). |
|
||||
| `context` | Callers, callees, processes for one symbol (with disambiguation). |
|
||||
| `impact` | Blast radius (upstream/downstream) with depth and risk summary. |
|
||||
| `detect_changes` | Map git diffs to affected symbols and processes. |
|
||||
| `rename` | Graph-assisted rename with `dry_run` preview (`graph` vs `text_search` confidence). |
|
||||
| `list_repos` | Discover indexed repos |
|
||||
| `query` | Hybrid BM25 + vector search over the graph |
|
||||
| `cypher` | Ad hoc Cypher against the schema |
|
||||
| `context` | Callers, callees, processes for one symbol |
|
||||
| `impact` | Blast radius (upstream/downstream) with risk summary |
|
||||
| `detect_changes` | Map git diffs to affected symbols and processes |
|
||||
| `rename` | Graph-assisted multi-file rename with `dry_run` preview |
|
||||
| `api_impact` | Pre-change impact report for an API route handler |
|
||||
| `route_map` | API route → handler → consumer mappings |
|
||||
| `tool_map` | MCP/RPC tool definitions and handlers |
|
||||
| `shape_check` | Response shape vs consumer property access mismatches |
|
||||
| `group_list` | List repo groups or details for one group |
|
||||
| `group_query` | Cross-repo search in a group (reciprocal rank fusion) |
|
||||
| `group_sync` | Rebuild group Contract Registry (`contracts.json`) |
|
||||
| `group_contracts` | Inspect group contracts and cross-links |
|
||||
| `group_status` | Index and Contract Registry staleness per repo in a group |
|
||||
|
||||
## Where to change what
|
||||
|
||||
| If you are changing… | Start in… |
|
||||
|----------------------|-----------|
|
||||
| CLI commands / flags | `gitnexus/src/cli/` (`index.ts`, per-command modules). |
|
||||
| Parsing or graph construction | `gitnexus/src/core/ingestion/pipeline-phases/` (individual phase files), `pipeline.ts` (orchestrator). |
|
||||
| Graph schema / DB access | `gitnexus/src/core/lbug/` (`schema.ts`, `lbug-adapter.ts`), `gitnexus/src/mcp/core/lbug-adapter.ts` if MCP-specific. |
|
||||
| MCP protocol, tools, resources | `gitnexus/src/mcp/server.ts`, `tools.ts`, `resources.ts`. |
|
||||
| Search ranking | `gitnexus/src/core/search/` (BM25, hybrid fusion). |
|
||||
| Embeddings | `gitnexus/src/core/embeddings/`, phases in `analyze.ts`. |
|
||||
| Wiki generation | `gitnexus/src/core/wiki/`. |
|
||||
| Web UI behavior | `gitnexus-web/src/` (components, workers, graph client). |
|
||||
| CI | `.github/workflows/*.yml`, `.github/actions/setup-gitnexus/`. |
|
||||
| Concern | Start in |
|
||||
|---------|----------|
|
||||
| CLI commands/flags | `src/cli/` (`index.ts`, per-command modules) |
|
||||
| Parsing/graph construction | `src/core/ingestion/pipeline-phases/` + `pipeline.ts` |
|
||||
| Graph schema/DB | `src/core/lbug/` (`schema.ts`, `lbug-adapter.ts`) |
|
||||
| MCP tools/resources | `src/mcp/server.ts`, `tools.ts`, `resources.ts` |
|
||||
| Search ranking | `src/core/search/` (BM25, hybrid fusion) |
|
||||
| Embeddings | `src/core/embeddings/` + `src/core/run-analyze.ts` |
|
||||
| Wiki generation | `src/core/wiki/` |
|
||||
| Language support | `src/core/ingestion/languages/` + `tree-sitter-queries.ts` + `gitnexus-shared/src/languages.ts` |
|
||||
| Import resolution | `src/core/ingestion/import-processor.ts` + `model/resolution-context.ts` |
|
||||
| Call resolution/MRO | `src/core/ingestion/call-processor.ts` + `model/resolve.ts` |
|
||||
| Type extraction | `src/core/ingestion/type-extractors/` |
|
||||
| Worker pool | `src/core/ingestion/workers/` |
|
||||
| Web UI | `gitnexus-web/src/` |
|
||||
| CI | `.github/workflows/*.yml`, `.github/actions/` |
|
||||
|
||||
> Paths above are relative to `gitnexus/` unless they start with `gitnexus-web/` or `.github/`.
|
||||
|
||||
---
|
||||
|
||||
## Pipeline Phase DAG
|
||||
|
||||
The ingestion pipeline is a DAG of named phases. Each phase is defined in its own file under `gitnexus/src/core/ingestion/pipeline-phases/` with explicit dependencies, typed inputs, and typed outputs.
|
||||
12 phases defined in `gitnexus/src/core/ingestion/pipeline-phases/`, each with explicit `deps` and typed output.
|
||||
|
||||
```
|
||||
scan → structure → [markdown, cobol] → parse → [routes, tools, orm]
|
||||
→ crossFile → mro → communities → processes
|
||||
```
|
||||
|
||||
### Phase files
|
||||
| Phase | File | Deps | Output |
|
||||
|-------|------|------|--------|
|
||||
| `scan` | `scan.ts` | (root) | File paths + sizes |
|
||||
| `structure` | `structure.ts` | `scan` | File/Folder nodes, CONTAINS edges, `allPathSet` |
|
||||
| `markdown` | `markdown.ts` | `structure` | Section nodes, cross-link edges from .md/.mdx |
|
||||
| `cobol` | `cobol.ts` | `structure` | COBOL program/paragraph/section nodes (regex, no tree-sitter) |
|
||||
| `parse` | `parse.ts` + `parse-impl.ts` | `structure`, `markdown`, `cobol` | Symbol nodes, IMPORTS/CALLS/EXTENDS edges, extracted routes/tools/ORM queries |
|
||||
| `routes` | `routes.ts` | `parse` | Route nodes + HANDLES_ROUTE edges (Next.js, Expo, PHP, decorators) |
|
||||
| `tools` | `tools.ts` | `parse` | Tool nodes + HANDLES_TOOL edges |
|
||||
| `orm` | `orm.ts` | `parse` | QUERIES edges (Prisma, Supabase) |
|
||||
| `crossFile` | `cross-file.ts` + `cross-file-impl.ts` | `parse`, `routes`, `tools`, `orm` | Cross-file type propagation in topological import order |
|
||||
| `mro` | `mro.ts` | `crossFile`, `structure` | METHOD_OVERRIDES + METHOD_IMPLEMENTS edges |
|
||||
| `communities` | `communities.ts` | `mro`, `structure` | Community nodes + MEMBER_OF edges (Leiden algorithm) |
|
||||
| `processes` | `processes.ts` | `communities`, `routes`, `tools`, `structure` | Process nodes + STEP_IN_PROCESS edges |
|
||||
|
||||
| Phase | File | Dependencies | What it does |
|
||||
|-------|------|-------------|--------------|
|
||||
| `scan` | `scan.ts` | (root) | Walk repo filesystem, collect paths + sizes |
|
||||
| `structure` | `structure.ts` | `scan` | Build File/Folder nodes + CONTAINS edges |
|
||||
| `markdown` | `markdown.ts` | `structure` | Extract headings and cross-links from .md/.mdx |
|
||||
| `cobol` | `cobol.ts` | `structure` | Regex-based COBOL/JCL extraction |
|
||||
| `parse` | `parse.ts` + `parse-impl.ts` | `structure`, `markdown`, `cobol` | Chunked tree-sitter parse, import/call/heritage resolution |
|
||||
| `routes` | `routes.ts` | `parse` | Route registry (Next.js, Expo, PHP, decorator-based) |
|
||||
| `tools` | `tools.ts` | `parse` | MCP/RPC tool detection |
|
||||
| `orm` | `orm.ts` | `parse` | Prisma/Supabase ORM query edges |
|
||||
| `crossFile` | `cross-file.ts` + `cross-file-impl.ts` | `parse`, `routes`, `tools`, `orm` | Cross-file type propagation in topological order |
|
||||
| `mro` | `mro.ts` | `crossFile` | Method Resolution Order, METHOD_OVERRIDES edges |
|
||||
| `communities` | `communities.ts` | `mro` | Leiden community detection |
|
||||
| `processes` | `processes.ts` | `communities`, `routes`, `tools` | Execution flow detection, Route/Tool → Process links |
|
||||
**Non-phase files in the same directory:** `parse-impl.ts`, `cross-file-impl.ts` (implementation), `wildcard-synthesis.ts` (whole-module import expansion), `orm-extraction.ts` (sequential ORM fallback), `types.ts`, `runner.ts`, `index.ts`.
|
||||
|
||||
### DAG runner
|
||||
|
||||
`runner.ts` — static phase graph, no plugins, compile-time type safety.
|
||||
|
||||
1. **Validation** — Kahn's topological sort. Rejects on: duplicate names, missing deps, cycles (DFS traces the concrete cycle path, e.g., `A -> B -> C -> A`, plus count of transitively blocked dependents).
|
||||
|
||||
2. **Execution** — sequential in topological order. Each phase receives:
|
||||
- `ctx: PipelineContext` — shared mutable `KnowledgeGraph`, `repoPath`, progress callback, options
|
||||
- `deps: ReadonlyMap<string, PhaseResult>` — **declared deps only** (runner filters the results map to prevent hidden coupling)
|
||||
|
||||
3. **Error handling** — wraps phase errors with the phase name, emits terminal `error` progress event, swallows progress handler errors to preserve the original cause.
|
||||
|
||||
4. **Timing** — per-phase `durationMs` in `PhaseResult`, dev-mode console logging.
|
||||
|
||||
**Design patterns:**
|
||||
- **Single graph accumulator** — all phases mutate the same `KnowledgeGraph` in `ctx`; the graph is the primary output.
|
||||
- **Typed phase access** — `getPhaseOutput<T>(deps, 'name')` for type-safe upstream results.
|
||||
- **Binding accumulator lifecycle** — created in `parse`, disposed by `crossFile` (in `finally`). No other phase should take ownership.
|
||||
- **Skippable phases** — `skipGraphPhases` omits MRO/communities/processes (faster tests). `skipWorkers` forces sequential parsing.
|
||||
|
||||
### How to add a new phase
|
||||
|
||||
1. Create a new file in `pipeline-phases/` (e.g. `my-phase.ts`)
|
||||
2. Define a `PipelinePhase<MyOutput>` object with `name`, `deps`, and `execute(ctx, deps)`
|
||||
3. Export it from `pipeline-phases/index.ts`
|
||||
4. Add it to the `buildPhaseList()` function in `pipeline.ts`
|
||||
1. Create `pipeline-phases/my-phase.ts` with a `PipelinePhase<MyOutput>` (name, deps, execute)
|
||||
2. Export from `pipeline-phases/index.ts`
|
||||
3. Add to `buildPhaseList()` in `pipeline.ts`
|
||||
|
||||
```typescript
|
||||
// pipeline-phases/my-phase.ts
|
||||
import type { PipelinePhase, PipelineContext, PhaseResult } from './types.js';
|
||||
import type { PipelinePhase, PhaseResult } from './types.js';
|
||||
import { getPhaseOutput } from './types.js';
|
||||
import type { ParseOutput } from './parse.js';
|
||||
|
||||
@@ -101,81 +131,168 @@ export interface MyPhaseOutput { /* ... */ }
|
||||
|
||||
export const myPhase: PipelinePhase<MyPhaseOutput> = {
|
||||
name: 'myPhase',
|
||||
deps: ['parse'], // runs after parse completes
|
||||
deps: ['parse'],
|
||||
async execute(ctx, deps) {
|
||||
const { allPaths } = getPhaseOutput<ParseOutput>(deps, 'parse');
|
||||
// ... do work, write to ctx.graph ...
|
||||
// ... write to ctx.graph ...
|
||||
return { /* typed output */ };
|
||||
},
|
||||
};
|
||||
```
|
||||
|
||||
### DAG runner
|
||||
---
|
||||
|
||||
The runner (`pipeline-phases/runner.ts`) validates the DAG at startup (detects cycles and missing deps via topological sort), then executes phases in dependency order. Each phase receives:
|
||||
- `ctx: PipelineContext` — shared graph, repoPath, progress callback
|
||||
- `deps: Map<string, PhaseResult>` — outputs from all upstream phases
|
||||
## Language-agnostic graph feeding
|
||||
|
||||
16 languages → single unified graph. Four abstraction layers:
|
||||
|
||||
```
|
||||
Unified Graph Schema (44 node types, 21 relationship types)
|
||||
↑
|
||||
Unified Resolution (3-tier name lookup + MRO walk)
|
||||
↑
|
||||
Language Providers (import semantics, type config, export checker, MRO strategy)
|
||||
↑
|
||||
Tree-Sitter Queries (per-language S-expressions, unified capture tags)
|
||||
```
|
||||
|
||||
### Language providers
|
||||
|
||||
Each language implements `LanguageProvider` (`language-provider.ts`). Key fields:
|
||||
|
||||
| Field | Purpose |
|
||||
|-------|---------|
|
||||
| `id`, `extensions` | Language identity and file matching |
|
||||
| `treeSitterQueries` | S-expression queries for AST extraction |
|
||||
| `importSemantics` | `named` / `wildcard-leaf` / `wildcard-transitive` / `namespace` |
|
||||
| `importResolver` | Language-specific path → file resolution |
|
||||
| `exportChecker` | Public/exported symbol detection |
|
||||
| `typeConfig` | Type annotation extraction rules |
|
||||
| `mroStrategy` | `first-wins` / `c3` / `none` |
|
||||
|
||||
16 providers in `languages/index.ts` via `satisfies Record<SupportedLanguages, LanguageProvider>` — missing a language is a compile error.
|
||||
|
||||
### Unified capture tags
|
||||
|
||||
Per-language tree-sitter queries use different AST node names but produce the **same semantic capture tags**: `@definition.class`, `@definition.function`, `@call.name`, `@import.source`, `@heritage.extends`. Downstream extraction needs no language branching. Defined in `tree-sitter-queries.ts`.
|
||||
|
||||
### Import resolution
|
||||
|
||||
Unified 3-tier algorithm (`model/resolution-context.ts`), per-language `importSemantics` controls which tier activates:
|
||||
|
||||
| Tier | Confidence | Mechanism |
|
||||
|------|-----------|-----------|
|
||||
| 1 — same-file | 0.95 | Symbol table for caller's file |
|
||||
| 2 — import-scoped | 0.9 | `NamedImportMap` chains (named) or all files in `importMap` (wildcard) |
|
||||
| 3 — global | 0.5 | O(1) index lookups: class, impl, callable. Fallback only |
|
||||
|
||||
| Import strategy | Languages | Behavior |
|
||||
|----------------|-----------|----------|
|
||||
| `named` | TS, JS, Java, C#, Rust, PHP, Kotlin | Only explicitly imported names visible |
|
||||
| `wildcard-leaf` | Go, Ruby, Swift, Dart | Whole-package import, no transitive re-exports |
|
||||
| `wildcard-transitive` | C, C++ | `#include` closure chains through re-exports |
|
||||
| `namespace` | Python | Module aliases resolved at call site |
|
||||
|
||||
### Chunked parse-and-resolve
|
||||
|
||||
`parse` processes files in ~20 MB byte-budget chunks to bound memory. Per chunk:
|
||||
1. Worker pool dispatches files (or sequential fallback via `skipWorkers`)
|
||||
2. Each worker: detect language → load grammar → run queries → return unified `ParseWorkerResult`
|
||||
3. Synthesize wildcard bindings (`wildcard-synthesis.ts`)
|
||||
4. Resolve imports and heritage
|
||||
5. Collect `BindingAccumulator` entries for cross-file propagation
|
||||
|
||||
Workers: `workers/worker-pool.ts`, `workers/parse-worker.ts`.
|
||||
|
||||
### Heritage and MRO
|
||||
|
||||
All languages emit unified `ExtractedHeritage` (child, parent, `EXTENDS`/`IMPLEMENTS`). MRO phase walks the heritage graph using per-language strategy:
|
||||
- **`first-wins`** — Java, C#, C++, TS, Ruby, Go
|
||||
- **`c3`** — Python (C3 linearization)
|
||||
- **`none`** — single-inheritance languages
|
||||
|
||||
Unified walk: `lookupMethodByOwnerWithMRO()` in `model/resolve.ts`.
|
||||
|
||||
---
|
||||
|
||||
## Full analysis flow
|
||||
|
||||
`runFullAnalysis` in `run-analyze.ts` orchestrates everything around the pipeline:
|
||||
|
||||
```
|
||||
CLI (analyze.ts) → runFullAnalysis(repoPath, options, callbacks)
|
||||
1. Early exit if lastCommit == HEAD (unless --force) [0%]
|
||||
2. Cache existing embeddings from prior index [0%]
|
||||
3. runPipelineFromRepo() → KnowledgeGraph [0-60%]
|
||||
4. Clean up legacy KuzuDB files [60%]
|
||||
5. initLbug() → loadGraphToLbug() via CSV streaming [60-85%]
|
||||
6. Create FTS indexes (File, Function, Class, Method...) [85-90%]
|
||||
7. Restore cached embeddings (batch insert) [88%]
|
||||
8. Generate new embeddings if --embeddings [90-98%]
|
||||
9. Save metadata + register repo + update .gitignore [98-100%]
|
||||
10. Generate AI context files (AGENTS.md, CLAUDE.md) [100%]
|
||||
```
|
||||
|
||||
**Options:** `--force` (rebuild regardless), `--embeddings` (opt-in, skipped if >50k nodes), `--skipGit`, `--noStats`.
|
||||
|
||||
## Storage
|
||||
|
||||
```
|
||||
<repo>/.gitnexus/
|
||||
├── lbug # LadybugDB database
|
||||
├── lbug.wal # Write-ahead log
|
||||
├── lbug.lock # Single-writer lock
|
||||
└── meta.json # lastCommit, indexedAt, stats
|
||||
|
||||
~/.gitnexus/
|
||||
└── registry.json # Global repo registry (MCP discovery)
|
||||
```
|
||||
|
||||
Managed by `repo-manager.ts`.
|
||||
|
||||
## LadybugDB schema
|
||||
|
||||
Defined in `lbug/schema.ts`. Separate node tables per type, single `CodeRelation` table.
|
||||
|
||||
**Node tables:** File, Folder, Function, Class, Interface, Method, Constructor, CodeElement, Struct, Enum, Macro, Typedef, Union, Namespace, Trait, Impl, TypeAlias, Const, Static, Property, Record, Delegate, Annotation, Template, Module, Community, Process, Route, Tool, Section, Embedding.
|
||||
|
||||
**Relation types** (`CodeRelation.type`): CONTAINS, DEFINES, CALLS, IMPORTS, EXTENDS, IMPLEMENTS, HAS_METHOD, HAS_PROPERTY, ACCESSES, METHOD_OVERRIDES, METHOD_IMPLEMENTS, MEMBER_OF, STEP_IN_PROCESS, HANDLES_ROUTE, FETCHES, HANDLES_TOOL, ENTRY_POINT_OF.
|
||||
|
||||
## Embeddings and search
|
||||
|
||||
**Embeddings** (`src/core/embeddings/`): Snowflake arctic-embed-xs (384D). Embeddable: File, Function, Class, Method, Interface. Incremental via SHA1 content hash. Separate `Embedding` table.
|
||||
|
||||
**Search** (`src/core/search/`): Hybrid BM25 + semantic vector, merged via Reciprocal Rank Fusion (K=60).
|
||||
|
||||
## Known limitations
|
||||
|
||||
### Overloaded method resolution
|
||||
|
||||
Method and Constructor node IDs include an arity suffix (`#<paramCount>`) to
|
||||
disambiguate overloaded methods. Two overloads with different parameter counts
|
||||
produce distinct graph nodes: `Method:file:Class.method#1` vs
|
||||
`Method:file:Class.method#2`.
|
||||
Node IDs use arity suffix (`#<paramCount>`): `Method:file:Class.method#1` vs `#2`.
|
||||
|
||||
**Same-arity overload disambiguation:** When two overloads share the same
|
||||
parameter count but differ in types (e.g. `save(int)` vs `save(String)`), a
|
||||
type-hash suffix `~type1,type2` is appended to produce distinct node IDs:
|
||||
`Method:file:Class.save#1~int` vs `Method:file:Class.save#1~String`. The suffix
|
||||
is only added when a same-arity collision is detected within a class and all
|
||||
parameters have non-null type annotations. Languages without type info (Python,
|
||||
Ruby, JS) fall back to arity-only IDs. TypeScript/JavaScript overload signatures
|
||||
are intentionally excluded from type-hashing because they are declaration-only
|
||||
contracts that should collapse to the implementation body's node ID. See issue
|
||||
\#651.
|
||||
**Same-arity disambiguation:** type-hash suffix `~type1,type2` when collision detected and type annotations present. Languages without types (Python, Ruby, JS) use arity-only. TS/JS overload signatures excluded (collapse to implementation body). See #651.
|
||||
|
||||
**C++ const-qualified overload disambiguation:** Methods overloaded by const
|
||||
qualification (e.g. `begin()` vs `begin() const`) are disambiguated via an
|
||||
`isConst` property and a `$const` ID suffix appended to the const-qualified
|
||||
variant when a non-const collision exists. The `$const` suffix appears after the
|
||||
type-hash suffix: e.g. `Method:file:Container.begin#0$const`.
|
||||
**C++ const-qualified:** `$const` suffix after type-hash when non-const collision exists: `Method:file:Container.begin#0$const`.
|
||||
|
||||
**Generic/template type preservation in type-hash:** The type-hash suffix uses
|
||||
`rawType` (full AST text including generic/template args) rather than the
|
||||
simplified `type` from `extractSimpleTypeName`. This means C++ template overloads
|
||||
like `process(vector<int>)` vs `process(vector<string>)` produce distinct IDs:
|
||||
`~vector<int>` vs `~vector<std::string>`. Java generic overloads like
|
||||
`process(List<String>)` vs `process(List<Integer>)` are a compile error due to
|
||||
type erasure, so this gap is theoretical for Java.
|
||||
**Generic/template types:** type-hash uses `rawType` (full AST text including generics): `~vector<int>` vs `~vector<std::string>`.
|
||||
|
||||
**ID stability on first overload:** Type and const tags are collision-only. When
|
||||
a class has `save(int)` as its only `save` method, the ID is `save#1` (no tag).
|
||||
Adding `save(String)` changes the original to `save#1~int`. This is correct for
|
||||
fresh analysis but means IDs are not stable across overload additions. Future
|
||||
incremental re-analysis should account for this.
|
||||
**ID stability:** collision-only tags mean IDs change when overloads are added. `save#1` becomes `save#1~int` when `save(String)` is added.
|
||||
|
||||
**Variadic method matching:** When one side is variadic (`parameterCount`
|
||||
undefined) and the other has a fixed count, `METHOD_IMPLEMENTS` edges are
|
||||
emitted with confidence 0.7 instead of 1.0. Variadic methods like
|
||||
`foo(String... args)` may superficially match `foo(String s)` by type but
|
||||
are not guaranteed to be interchangeable across all languages (Java/Kotlin
|
||||
accept this via varargs sugar; TypeScript, C#, Rust do not).
|
||||
**Variadic matching:** confidence 0.7 when one side is variadic and the other has fixed count.
|
||||
|
||||
**Confidence tiering** for `METHOD_IMPLEMENTS` edges:
|
||||
**METHOD_IMPLEMENTS confidence tiering:**
|
||||
|
||||
| Match quality | Confidence | When |
|
||||
|---|---|---|
|
||||
| Exact parameter types match | 1.0 | Both sides have `parameterTypes` arrays and they match |
|
||||
| Arity (count) matches | 1.0 | Both sides have `parameterCount`, types unavailable |
|
||||
| Variadic vs fixed | 0.7 | One side is variadic, other has fixed count |
|
||||
| Lenient (insufficient info) | 0.7 | One or both sides lack type and count data |
|
||||
| Match quality | Confidence |
|
||||
|---|---|
|
||||
| Exact parameter types match | 1.0 |
|
||||
| Arity match, types unavailable | 1.0 |
|
||||
| Variadic vs fixed | 0.7 |
|
||||
| Insufficient info | 0.7 |
|
||||
|
||||
## Related docs
|
||||
|
||||
- [MIGRATION.md](MIGRATION.md) — breaking changes and migration guidance.
|
||||
- [RUNBOOK.md](RUNBOOK.md) — operational commands and recovery.
|
||||
- [GUARDRAILS.md](GUARDRAILS.md) — safety boundaries for humans and agents.
|
||||
- [TESTING.md](TESTING.md) — how to run tests.
|
||||
- `AGENTS.md` / `CLAUDE.md` — agent workflows and tool usage expectations for **this** repo when indexed by GitNexus.
|
||||
- [MIGRATION.md](MIGRATION.md) — breaking changes and migration guidance
|
||||
- [RUNBOOK.md](RUNBOOK.md) — operational commands and recovery
|
||||
- [GUARDRAILS.md](GUARDRAILS.md) — safety boundaries for humans and agents
|
||||
- [TESTING.md](TESTING.md) — how to run tests
|
||||
- `AGENTS.md` / `CLAUDE.md` — agent workflows and tool usage
|
||||
|
||||
+108
-11
@@ -21,30 +21,127 @@ This project uses the [PolyForm Noncommercial License 1.0.0](https://polyformpro
|
||||
## Branch and pull requests
|
||||
|
||||
- Use short-lived branches off the default branch of the repo you are targeting.
|
||||
- Prefer **conventional commits** (short prefix + description), for example:
|
||||
|
||||
```text
|
||||
feat: add graph export option
|
||||
fix: correct MCP tool schema for query
|
||||
test: cover cluster merge edge case
|
||||
docs: clarify analyze flags
|
||||
```
|
||||
|
||||
- **PR title:** `[area] Short description` (e.g. `[cli] Fix index refresh race`).
|
||||
- **PR titles MUST follow the conventional-commit format** — `pr-labeler.yml` enforces this on every PR and auto-applies the matching label so release notes group the change correctly.
|
||||
- **PR description:** what changed, why, how to verify (commands), and any risk or rollback notes.
|
||||
|
||||
### Pull request titles
|
||||
|
||||
Format: `<type>[(scope)][!]: <subject>`
|
||||
|
||||
Allowed types and the release-notes section each one lands in (defined in `.github/release.yml`):
|
||||
|
||||
| Type | Label applied | Release-notes section |
|
||||
|------|---------------|-----------------------|
|
||||
| `feat` | `enhancement` | 🚀 Features |
|
||||
| `fix` | `bug` | 🐛 Bug Fixes |
|
||||
| `perf` | `performance` | 🏎️ Performance |
|
||||
| `refactor` | `refactor` | 🔄 Refactoring |
|
||||
| `test` | `test` | 🧪 Tests |
|
||||
| `ci` | `ci` | 👷 CI/CD |
|
||||
| `build` / `deps` | `dependencies` | 📦 Dependencies |
|
||||
| `docs` | `documentation` | (grouped under Other Changes unless a Docs section is added) |
|
||||
| `chore` / `revert` | `chore` | (excluded from release notes) |
|
||||
|
||||
Append `!` to the type (e.g. `feat(api)!: drop /v1 endpoint`) or include `BREAKING CHANGE:` in the PR body to flag a breaking change — the labeler then adds the `breaking` label and the 💥 Breaking Changes section is rendered first.
|
||||
|
||||
Examples:
|
||||
|
||||
```text
|
||||
feat(web): add smart chat scroll
|
||||
fix(extractors): resolve silent contract mis-resolution
|
||||
perf: avoid O(n²) traversal in heritage walker
|
||||
chore(deps): bump vitest to 3.0.0
|
||||
ci: standardize workflow concurrency
|
||||
```
|
||||
|
||||
Commits within a PR may use any style — only the **merged PR title** shows up in release notes, so that's the one the convention applies to.
|
||||
|
||||
## Before you open a PR
|
||||
|
||||
- [ ] Tests pass for the packages you touched (`gitnexus` and/or `gitnexus-web`).
|
||||
- [ ] Typecheck passes: `npx tsc --noEmit` in `gitnexus/` and `npx tsc -b --noEmit` in `gitnexus-web/`.
|
||||
- [ ] No secrets, tokens, or machine-specific paths committed.
|
||||
- [ ] Documentation updated if behavior or public CLI/MCP contract changes.
|
||||
- [ ] Pre-commit hook runs clean (`.husky/pre-commit` — typecheck + unit tests for staged packages).
|
||||
- [ ] Pre-commit hook runs clean (`.husky/pre-commit` — formatting via lint-staged + typecheck for staged packages; tests run in CI only).
|
||||
|
||||
## Code review
|
||||
|
||||
Maintainers may request changes for correctness, tests, performance, or consistency with existing patterns. Keeping diffs focused makes review faster.
|
||||
|
||||
## GitHub Actions — Concurrency Convention
|
||||
|
||||
Every workflow under `.github/workflows/` MUST declare a top-level `concurrency:` block using this convention:
|
||||
|
||||
- **Group key** starts with `${{ github.workflow }}` so no two workflows can collide on the same group name. The discriminator that follows is chosen per event shape:
|
||||
- Branch/tag scope: `${{ github.workflow }}-${{ github.ref }}`
|
||||
- Per-PR scope (for `issue_comment`, `pull_request_review*`, `pull_request` meta events): `${{ github.workflow }}-${{ github.event.pull_request.number || github.event.issue.number }}`
|
||||
- `workflow_run` scope (e.g. `ci-report.yml`): `${{ github.workflow }}-${{ github.event.workflow_run.pull_requests[0].number || format('{0}/{1}', github.event.workflow_run.head_repository.full_name, github.event.workflow_run.head_branch) }}` — the fork fallback must be stable across reruns (never `workflow_run.id`, which is per-run-unique and defeats serialization).
|
||||
- Global single-slot (manual dispatch utilities): `${{ github.workflow }}`
|
||||
- **Reusable workflows invoked via `workflow_call`:** do NOT use `${{ github.workflow }}` in the group key — in called-workflow context its evaluation is ambiguous and can resolve to the caller's name, which would deadlock against the caller's own group. Use a hardcoded literal prefix and a `github.event_name`-aware expression that falls through to `github.run_id` for reusable invocations (see `ci.yml` for the canonical form).
|
||||
- **Merge queue (`merge_group`)**: when this event is added, use `${{ github.workflow }}-${{ github.event.merge_group.head_ref }}` with `cancel-in-progress: false` (every queue entry is a distinct ref; never cancel).
|
||||
- **`cancel-in-progress` policy:**
|
||||
|
||||
| Event | `cancel-in-progress` | Why |
|
||||
|-------|----------------------|-----|
|
||||
| `pull_request` CI run | `true` | New push supersedes old run |
|
||||
| `push` to `main` | `false` | Every main commit gets validated |
|
||||
| Tag push (`v*` publish) | `false` | Never cancel mid-publish |
|
||||
| `push` to `main` for release-candidate | `false` | Never cancel mid-RC publish |
|
||||
| `workflow_dispatch` (release/publish) | `false` | Manual runs are intentional |
|
||||
| `workflow_run` (sticky-comment reports) | `false` | Serialize, don't race |
|
||||
| Per-PR bot workflows (`@claude`, review) | `false` | Serialize comments per PR |
|
||||
| PR-meta re-checks (pr-description-check) | `true` | Cheap, latest wins |
|
||||
| Single-slot utilities (triage sweep) | `true` | Latest dispatch supersedes |
|
||||
|
||||
- For workflows that serve multiple events at once (e.g. `ci.yml` handles `pull_request`, `push`, and `workflow_call`), make `cancel-in-progress` event-aware:
|
||||
|
||||
```yaml
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.ref }}
|
||||
cancel-in-progress: ${{ github.event_name == 'pull_request' }}
|
||||
```
|
||||
|
||||
- When adding a new workflow, copy the concurrency block from an existing workflow of the same event shape.
|
||||
|
||||
## AI-assisted contributions
|
||||
|
||||
If you use coding agents, follow project context files (e.g. `AGENTS.md`, `CLAUDE.md`) and avoid drive-by refactors unrelated to the issue. Prefer incremental, test-backed changes.
|
||||
|
||||
## Releases
|
||||
|
||||
Two publish workflows ship `gitnexus` to npm:
|
||||
|
||||
- **Stable** (`.github/workflows/publish.yml`) — triggered by pushing any `v*`
|
||||
tag. Publishes to the `latest` dist-tag with a changelog-backed GitHub
|
||||
release. Maintainers are expected to tag from `main` as a convention; the
|
||||
workflow itself does not enforce branch reachability.
|
||||
- **Release Candidate** (`.github/workflows/release-candidate.yml`) — runs on
|
||||
every push to `main` (typically a merged PR) plus manual dispatch. Docs-only
|
||||
changes are skipped via `paths-ignore`. Publishes to the `rc` dist-tag with
|
||||
version `X.Y.Z-rc.N` and a GitHub prerelease, where:
|
||||
- `X.Y.Z` is selected automatically. On push (and on dispatch with
|
||||
`bump: auto`, the default) the workflow **continues the active rc cycle**:
|
||||
if the registry already has `X.Y.Z-rc.*` versions with `X.Y.Z` > current
|
||||
`latest`, it reuses the highest such base; otherwise it patch-bumps
|
||||
from `latest`. Dispatching with `bump: patch|minor|major` **resets**
|
||||
the cycle from `latest`.
|
||||
- `N` is auto-incremented against existing `X.Y.Z-rc.*` entries on the
|
||||
registry. First rc for a given base is `rc.1`.
|
||||
|
||||
Idempotency: the workflow pushes an `rc/<HEAD_SHA>` marker tag and a
|
||||
`v<RC>` release tag **atomically, before** calling `npm publish`. The guard
|
||||
refuses to re-run once the marker exists, so a post-publish failure will
|
||||
not mint a duplicate rc for the same commit. The `v<RC>` tag points at a
|
||||
detached release commit whose `package.json` matches the npm tarball
|
||||
exactly (traceable releases). Recovery after a partial failure:
|
||||
|
||||
```bash
|
||||
git push --delete origin rc/<HEAD_SHA> v<RC>
|
||||
# then redispatch the workflow with force: true
|
||||
```
|
||||
|
||||
The rc workflow never moves `latest`. To verify after a change, inspect dist-tags:
|
||||
|
||||
```bash
|
||||
npm view gitnexus dist-tags
|
||||
```
|
||||
|
||||
+43
-46
@@ -1,72 +1,69 @@
|
||||
# Guardrails — GitNexus (repo + agents)
|
||||
# Guardrails — GitNexus
|
||||
|
||||
Rules for **human contributors** and **AI agents** working on this codebase or publishing artifacts. These complement `AGENTS.md` / `CLAUDE.md` (which focus on GitNexus-in-GitNexus workflows).
|
||||
Rules for **human contributors** and **AI agents**. Complements `AGENTS.md` (workflows) and `CONTRIBUTING.md` (PR process).
|
||||
|
||||
## Scope (typical agent session)
|
||||
## Scope (least privilege)
|
||||
|
||||
When automating changes in this repository, treat scope as **least privilege**:
|
||||
- **Read:** Source, tests, docs, public config as needed.
|
||||
- **Write:** Only files required for the fix or feature; no unrelated formatting or refactors.
|
||||
- **Execute:** Tests, typecheck, documented CLI commands. No destructive commands on user data without approval.
|
||||
- **Off-limits:** Other people's machines, production deployments you don't own, credentials you lack permission to use.
|
||||
|
||||
- **Read:** Source, tests, docs, public config as needed for the task.
|
||||
- **Write:** Only files required for the requested fix or feature; avoid unrelated formatting or refactors.
|
||||
- **Execute:** Tests, typecheck, and documented CLI commands; do not run destructive commands on user data outside the repo without explicit approval.
|
||||
- **Off-limits:** Other people’s machines, production deployments you don’t own, and credentials you didn’t receive permission to use.
|
||||
|
||||
Adjust explicitly if the maintainer defines a different scope for a task.
|
||||
Maintainer may widen scope per task.
|
||||
|
||||
---
|
||||
|
||||
## Non-negotiables
|
||||
|
||||
1. **Never commit secrets** — API keys, tokens, `.env` with real values, private URLs, or session cookies. Use `.env.example` with placeholders only.
|
||||
2. **Never rename symbols with blind find-and-replace** when working in a GitNexus-indexed project — use the **`rename` MCP tool** with **`dry_run: true` first**, then review `graph` vs `text_search` edits. (There is no separate `gitnexus rename` CLI; renaming goes through MCP or editor integration.)
|
||||
3. **Run impact analysis before editing shared symbols** — use **`impact`** (upstream) for functions/classes/methods others call; do not ignore **HIGH** / **CRITICAL** risk without maintainer sign-off.
|
||||
4. **Prefer `detect_changes` before commit** — confirm diffs map to expected symbols/processes when the graph is available.
|
||||
5. **Preserve embeddings** — if `.gitnexus/meta.json` shows embeddings, run `npx gitnexus analyze --embeddings` when refreshing the index; plain `analyze` can drop them.
|
||||
1. **Never commit secrets** — API keys, tokens, real `.env` values, private URLs, session cookies. Use `.env.example` with placeholders.
|
||||
2. **Never rename with find-and-replace** in GitNexus-indexed projects — use `rename` MCP tool with `dry_run: true` first, review `graph` vs `text_search` edits. No separate `gitnexus rename` CLI exists.
|
||||
3. **Run impact analysis before editing shared symbols** — `impact` (upstream) for functions/classes/methods others call. Do not ignore HIGH/CRITICAL without maintainer sign-off.
|
||||
4. **Run `detect_changes` before commit** — confirm diffs map to expected symbols/processes when the graph is available.
|
||||
5. **Preserve embeddings** — if `.gitnexus/meta.json` shows embeddings, use `npx gitnexus analyze --embeddings`; plain `analyze` drops them.
|
||||
|
||||
---
|
||||
|
||||
## Signs (recurring failure patterns)
|
||||
|
||||
Use this format: **Trigger → Instruction → Reason**.
|
||||
Append new Signs here when the same mistake repeats (e.g. CI broken twice the same way).
|
||||
Format: **Trigger → Instruction → Reason**. Append new Signs when the same mistake repeats.
|
||||
|
||||
### Sign: Stale graph after edits
|
||||
### Stale graph after edits
|
||||
|
||||
- **Trigger:** MCP or resources warn the index is behind `HEAD`, or code search doesn’t match latest commit.
|
||||
- **Instruction:** Run `npx gitnexus analyze` from the repo root (plus `--embeddings` if the project used them).
|
||||
- **Reason:** Tools query LadybugDB built at last analyze; git changes are invisible until re-indexed.
|
||||
- **Trigger:** MCP warns index is behind `HEAD`, or search doesn't match latest commit.
|
||||
- **Do:** `npx gitnexus analyze` (plus `--embeddings` if used).
|
||||
- **Why:** Tools query LadybugDB from last analyze; git changes are invisible until re-indexed.
|
||||
|
||||
### Sign: Embeddings vanished after analyze
|
||||
### Embeddings vanished after analyze
|
||||
|
||||
- **Trigger:** Semantic search quality drops; `stats.embeddings` in `.gitnexus/meta.json` is 0 after a refresh.
|
||||
- **Instruction:** Re-run `npx gitnexus analyze --embeddings` and confirm `meta.json` reflects stored embeddings.
|
||||
- **Reason:** Embedding generation is opt-in; analyze without the flag does not preserve prior vectors.
|
||||
- **Trigger:** Semantic search quality drops; `stats.embeddings` in `meta.json` is 0 after refresh.
|
||||
- **Do:** `npx gitnexus analyze --embeddings`, confirm `meta.json` reflects stored embeddings.
|
||||
- **Why:** Embedding generation is opt-in; analyze without the flag does not preserve prior vectors.
|
||||
|
||||
### Sign: MCP lists no repos
|
||||
### MCP lists no repos
|
||||
|
||||
- **Trigger:** MCP stderr says no indexed repos.
|
||||
- **Instruction:** Run `npx gitnexus analyze` in the target repository; verify `npx gitnexus list` shows it.
|
||||
- **Reason:** The MCP server discovers repos via `~/.gitnexus/registry.json`, populated by analyze.
|
||||
- **Trigger:** MCP stderr says no indexed repos.
|
||||
- **Do:** `npx gitnexus analyze` in the target repo; verify `npx gitnexus list` shows it.
|
||||
- **Why:** MCP discovers repos via `~/.gitnexus/registry.json`, populated by analyze.
|
||||
|
||||
### Sign: Wrong repo in multi-repo setups
|
||||
### Wrong repo in multi-repo setups
|
||||
|
||||
- **Trigger:** Query/impact results clearly belong to another project.
|
||||
- **Instruction:** Call `list_repos`, then pass **`repo`** on subsequent tools (or use per-workspace MCP config).
|
||||
- **Reason:** Default target may be ambiguous when multiple repos are registered.
|
||||
- **Trigger:** Query/impact results belong to another project.
|
||||
- **Do:** Call `list_repos`, then pass `repo` on subsequent tools.
|
||||
- **Why:** Default target is ambiguous when multiple repos are registered.
|
||||
|
||||
### Sign: LadybugDB lock / “database busy”
|
||||
### LadybugDB lock / "database busy"
|
||||
|
||||
- **Trigger:** Errors opening `.gitnexus/lbug` while MCP and analyze both run.
|
||||
- **Instruction:** Stop overlapping processes; one writer at a time. Retry analyze or restart MCP.
|
||||
- **Reason:** Embedded DB expects single-process ownership of the store.
|
||||
- **Trigger:** Errors opening `.gitnexus/lbug` while MCP and analyze both run.
|
||||
- **Do:** Stop overlapping processes (one writer at a time). Retry analyze or restart MCP.
|
||||
- **Why:** Embedded DB expects single-process ownership.
|
||||
|
||||
---
|
||||
|
||||
## Publishing & supply chain
|
||||
|
||||
- **npm:** Do not publish from unreviewed automation; follow maintainer release process. Bump version intentionally; tag releases to match `package.json`.
|
||||
- **Dependencies:** Prefer minimal, auditable changes to `package.json`; run tests and CI after lockfile updates.
|
||||
- **License:** This project ships under **PolyForm Noncommercial 1.0.0** — do not relicense or imply a different license in docs or metadata without maintainer approval.
|
||||
- **npm:** Do not publish from unreviewed automation. Bump version intentionally; tag releases to match `package.json`.
|
||||
- **Dependencies:** Minimal, auditable `package.json` changes; run tests and CI after lockfile updates.
|
||||
- **License:** PolyForm Noncommercial 1.0.0 — do not relicense without maintainer approval.
|
||||
|
||||
---
|
||||
|
||||
@@ -74,15 +71,15 @@ Append new Signs here when the same mistake repeats (e.g. CI broken twice the sa
|
||||
|
||||
Stop and ask a **human maintainer** when:
|
||||
|
||||
- Impact analysis shows **HIGH** / **CRITICAL** risk and the task still requires the change.
|
||||
- You need to alter **CI**, **release**, or **security-sensitive** config.
|
||||
- Requirements conflict (e.g. “speed up analyze” vs “must keep all embeddings on huge repo”).
|
||||
- Impact analysis shows HIGH/CRITICAL risk and the task still requires the change.
|
||||
- You need to alter CI, release, or security-sensitive config.
|
||||
- Requirements conflict (e.g. "speed up analyze" vs "must keep all embeddings on huge repo").
|
||||
- You are unsure whether data loss is acceptable (`clean`, forced migrations, schema changes).
|
||||
|
||||
---
|
||||
|
||||
## Related docs
|
||||
|
||||
- [ARCHITECTURE.md](ARCHITECTURE.md) — components and data flow.
|
||||
- [RUNBOOK.md](RUNBOOK.md) — commands for recovery.
|
||||
- [CONTRIBUTING.md](CONTRIBUTING.md) — PR and commit expectations.
|
||||
- [ARCHITECTURE.md](ARCHITECTURE.md) — components and data flow
|
||||
- [RUNBOOK.md](RUNBOOK.md) — commands for recovery
|
||||
- [CONTRIBUTING.md](CONTRIBUTING.md) — PR and commit expectations
|
||||
|
||||
+8
-5
@@ -20,9 +20,9 @@ From repository root, unless noted:
|
||||
cd gitnexus
|
||||
npm install
|
||||
npm run build
|
||||
npm test # unit: vitest run test/unit
|
||||
npm test # full suite: vitest run
|
||||
npm run test:unit # unit only: vitest run test/unit
|
||||
npm run test:integration # integration suite
|
||||
npm run test:all
|
||||
npm run test:coverage
|
||||
npx tsc --noEmit # typecheck (matches CI)
|
||||
```
|
||||
@@ -42,8 +42,11 @@ npm run test:e2e # Playwright (requires gitnexus serve + npm run dev)
|
||||
|
||||
A husky pre-commit hook (`.husky/pre-commit`) runs automatically on every `git commit`:
|
||||
|
||||
- **`gitnexus-web/` files staged** → `tsc -b --noEmit` + `vitest run`
|
||||
- **`gitnexus/` files staged** → `tsc --noEmit` + `vitest run --project default`
|
||||
1. **Formatting** — `lint-staged` runs prettier on staged files
|
||||
2. **`gitnexus-web/` files staged** → `tsc -b --noEmit`
|
||||
3. **`gitnexus/` files staged** → `tsc --noEmit`
|
||||
|
||||
Tests do **not** run in the pre-commit hook — they run in CI (`ci-tests.yml`) only.
|
||||
|
||||
Skip with `git commit --no-verify` (use sparingly).
|
||||
|
||||
@@ -77,7 +80,7 @@ Re-run the full relevant suite when:
|
||||
|
||||
GitHub Actions (`.github/workflows/ci.yml`) orchestrate:
|
||||
|
||||
- **`ci-quality.yml`** — `tsc --noEmit` for `gitnexus/` + `tsc -b --noEmit` for `gitnexus-web/`
|
||||
- **`ci-quality.yml`** — prettier format check, eslint lint, `tsc --noEmit` for `gitnexus/`, `tsc -b --noEmit` for `gitnexus-web/`
|
||||
- **`ci-tests.yml`** — `vitest run` with coverage (ubuntu) + cross-platform (macOS, Windows)
|
||||
- **`ci-e2e.yml`** — Playwright E2E tests, gated on `gitnexus-web/**` changes
|
||||
|
||||
|
||||
@@ -9,6 +9,10 @@ tsconfig.json
|
||||
.gitignore
|
||||
node_modules/
|
||||
|
||||
# Vendor build artifacts (created during install, not shipped)
|
||||
vendor/**/node_modules
|
||||
vendor/**/build
|
||||
|
||||
# Package lock (consumers use their own)
|
||||
package-lock.json
|
||||
|
||||
|
||||
@@ -234,6 +234,29 @@ Installed automatically by both `gitnexus analyze` (per-repo) and `gitnexus setu
|
||||
- Node.js >= 18
|
||||
- Git repository (uses git for commit tracking)
|
||||
|
||||
## Release candidates
|
||||
|
||||
Stable releases publish to the default `latest` dist-tag. When a pull request
|
||||
with non-documentation changes merges into `main`, an automated workflow also
|
||||
publishes a prerelease build under the `rc` dist-tag, so early adopters can
|
||||
try in-flight fixes without waiting for the next stable cut. (Docs-only
|
||||
merges are skipped.)
|
||||
|
||||
```bash
|
||||
# Try the latest release candidate (pre-stable — may change at any time)
|
||||
npm install -g gitnexus@rc
|
||||
# — or —
|
||||
npx gitnexus@rc analyze
|
||||
```
|
||||
|
||||
Release-candidate versions follow the standard semver prerelease format
|
||||
`X.Y.Z-rc.N`, where `X.Y.Z` is the next stable target (bumped from the
|
||||
current `latest` by patch by default; `minor` or `major` when kicking off a
|
||||
bigger cycle) and `N` increments per published rc. Example sequence:
|
||||
`1.6.2-rc.1`, `1.6.2-rc.2`, …, then once `1.6.2` ships stable,
|
||||
`1.6.3-rc.1`. See the [Releases page](https://github.com/abhigyanpatwari/GitNexus/releases)
|
||||
for the full list; stable `latest` is unaffected.
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
### `Cannot destructure property 'package' of 'node.target' as it is null`
|
||||
|
||||
Generated
+19
-141
@@ -1,12 +1,12 @@
|
||||
{
|
||||
"name": "gitnexus",
|
||||
"version": "1.6.1",
|
||||
"version": "1.6.2-rc.13",
|
||||
"lockfileVersion": 3,
|
||||
"requires": true,
|
||||
"packages": {
|
||||
"": {
|
||||
"name": "gitnexus",
|
||||
"version": "1.6.1",
|
||||
"version": "1.6.2-rc.13",
|
||||
"hasInstallScript": true,
|
||||
"license": "PolyForm-Noncommercial-1.0.0",
|
||||
"dependencies": {
|
||||
@@ -30,7 +30,7 @@
|
||||
"pandemonium": "^2.4.0",
|
||||
"tree-sitter": "^0.21.1",
|
||||
"tree-sitter-c": "0.23.2",
|
||||
"tree-sitter-c-sharp": "^0.23.1",
|
||||
"tree-sitter-c-sharp": "0.23.1",
|
||||
"tree-sitter-cpp": "^0.23.4",
|
||||
"tree-sitter-go": "^0.23.0",
|
||||
"tree-sitter-java": "^0.23.5",
|
||||
@@ -62,6 +62,8 @@
|
||||
"node": ">=20.0.0"
|
||||
},
|
||||
"optionalDependencies": {
|
||||
"node-addon-api": "^8.0.0",
|
||||
"node-gyp-build": "^4.8.0",
|
||||
"tree-sitter-dart": "git+https://github.com/UserNobody14/tree-sitter-dart.git#80e23c07b64494f7e21090bb3450223ef0b192f4",
|
||||
"tree-sitter-kotlin": "^0.3.8",
|
||||
"tree-sitter-proto": "file:./vendor/tree-sitter-proto",
|
||||
@@ -1226,6 +1228,12 @@
|
||||
"win32"
|
||||
]
|
||||
},
|
||||
"node_modules/@ladybugdb/core/node_modules/node-addon-api": {
|
||||
"version": "6.1.0",
|
||||
"resolved": "https://registry.npmjs.org/node-addon-api/-/node-addon-api-6.1.0.tgz",
|
||||
"integrity": "sha512-+eawOlIgy680F0kBzPUNFhMZGtJ1YmqM6l4+Crf4IkImjYrO/mqPwRMh352g23uIaQKFItcQ64I7KMaJxHgAVA==",
|
||||
"license": "MIT"
|
||||
},
|
||||
"node_modules/@modelcontextprotocol/sdk": {
|
||||
"version": "1.28.0",
|
||||
"resolved": "https://registry.npmjs.org/@modelcontextprotocol/sdk/-/sdk-1.28.0.tgz",
|
||||
@@ -4118,10 +4126,13 @@
|
||||
}
|
||||
},
|
||||
"node_modules/node-addon-api": {
|
||||
"version": "6.1.0",
|
||||
"resolved": "https://registry.npmjs.org/node-addon-api/-/node-addon-api-6.1.0.tgz",
|
||||
"integrity": "sha512-+eawOlIgy680F0kBzPUNFhMZGtJ1YmqM6l4+Crf4IkImjYrO/mqPwRMh352g23uIaQKFItcQ64I7KMaJxHgAVA==",
|
||||
"license": "MIT"
|
||||
"version": "8.7.0",
|
||||
"resolved": "https://registry.npmjs.org/node-addon-api/-/node-addon-api-8.7.0.tgz",
|
||||
"integrity": "sha512-9MdFxmkKaOYVTV+XVRG8ArDwwQ77XIgIPyKASB1k3JPq3M8fGQQQE3YpMOrKm6g//Ktx8ivZr8xo1Qmtqub+GA==",
|
||||
"license": "MIT",
|
||||
"engines": {
|
||||
"node": "^18 || ^20 || >= 21"
|
||||
}
|
||||
},
|
||||
"node_modules/node-api-headers": {
|
||||
"version": "1.8.0",
|
||||
@@ -5069,24 +5080,6 @@
|
||||
}
|
||||
}
|
||||
},
|
||||
"node_modules/tree-sitter-c-sharp/node_modules/node-addon-api": {
|
||||
"version": "8.7.0",
|
||||
"resolved": "https://registry.npmjs.org/node-addon-api/-/node-addon-api-8.7.0.tgz",
|
||||
"integrity": "sha512-9MdFxmkKaOYVTV+XVRG8ArDwwQ77XIgIPyKASB1k3JPq3M8fGQQQE3YpMOrKm6g//Ktx8ivZr8xo1Qmtqub+GA==",
|
||||
"license": "MIT",
|
||||
"engines": {
|
||||
"node": "^18 || ^20 || >= 21"
|
||||
}
|
||||
},
|
||||
"node_modules/tree-sitter-c/node_modules/node-addon-api": {
|
||||
"version": "8.7.0",
|
||||
"resolved": "https://registry.npmjs.org/node-addon-api/-/node-addon-api-8.7.0.tgz",
|
||||
"integrity": "sha512-9MdFxmkKaOYVTV+XVRG8ArDwwQ77XIgIPyKASB1k3JPq3M8fGQQQE3YpMOrKm6g//Ktx8ivZr8xo1Qmtqub+GA==",
|
||||
"license": "MIT",
|
||||
"engines": {
|
||||
"node": "^18 || ^20 || >= 21"
|
||||
}
|
||||
},
|
||||
"node_modules/tree-sitter-cli": {
|
||||
"version": "0.23.2",
|
||||
"resolved": "https://registry.npmjs.org/tree-sitter-cli/-/tree-sitter-cli-0.23.2.tgz",
|
||||
@@ -5121,18 +5114,9 @@
|
||||
}
|
||||
}
|
||||
},
|
||||
"node_modules/tree-sitter-cpp/node_modules/node-addon-api": {
|
||||
"version": "8.7.0",
|
||||
"resolved": "https://registry.npmjs.org/node-addon-api/-/node-addon-api-8.7.0.tgz",
|
||||
"integrity": "sha512-9MdFxmkKaOYVTV+XVRG8ArDwwQ77XIgIPyKASB1k3JPq3M8fGQQQE3YpMOrKm6g//Ktx8ivZr8xo1Qmtqub+GA==",
|
||||
"license": "MIT",
|
||||
"engines": {
|
||||
"node": "^18 || ^20 || >= 21"
|
||||
}
|
||||
},
|
||||
"node_modules/tree-sitter-dart": {
|
||||
"version": "1.0.0",
|
||||
"resolved": "git+https://github.com/UserNobody14/tree-sitter-dart.git#80e23c07b64494f7e21090bb3450223ef0b192f4",
|
||||
"resolved": "git+ssh://git@github.com/UserNobody14/tree-sitter-dart.git#80e23c07b64494f7e21090bb3450223ef0b192f4",
|
||||
"integrity": "sha512-Bs/1wAOIJ2akPEXlE/XVpuES19Oo3NqoSJRJ/0N2r38qAd9nTXdqmaGHQ44/JXnA6QHcbgD2YzCCc4wUc98cyQ==",
|
||||
"hasInstallScript": true,
|
||||
"license": "ISC",
|
||||
@@ -5176,15 +5160,6 @@
|
||||
}
|
||||
}
|
||||
},
|
||||
"node_modules/tree-sitter-go/node_modules/node-addon-api": {
|
||||
"version": "8.7.0",
|
||||
"resolved": "https://registry.npmjs.org/node-addon-api/-/node-addon-api-8.7.0.tgz",
|
||||
"integrity": "sha512-9MdFxmkKaOYVTV+XVRG8ArDwwQ77XIgIPyKASB1k3JPq3M8fGQQQE3YpMOrKm6g//Ktx8ivZr8xo1Qmtqub+GA==",
|
||||
"license": "MIT",
|
||||
"engines": {
|
||||
"node": "^18 || ^20 || >= 21"
|
||||
}
|
||||
},
|
||||
"node_modules/tree-sitter-java": {
|
||||
"version": "0.23.5",
|
||||
"resolved": "https://registry.npmjs.org/tree-sitter-java/-/tree-sitter-java-0.23.5.tgz",
|
||||
@@ -5204,15 +5179,6 @@
|
||||
}
|
||||
}
|
||||
},
|
||||
"node_modules/tree-sitter-java/node_modules/node-addon-api": {
|
||||
"version": "8.7.0",
|
||||
"resolved": "https://registry.npmjs.org/node-addon-api/-/node-addon-api-8.7.0.tgz",
|
||||
"integrity": "sha512-9MdFxmkKaOYVTV+XVRG8ArDwwQ77XIgIPyKASB1k3JPq3M8fGQQQE3YpMOrKm6g//Ktx8ivZr8xo1Qmtqub+GA==",
|
||||
"license": "MIT",
|
||||
"engines": {
|
||||
"node": "^18 || ^20 || >= 21"
|
||||
}
|
||||
},
|
||||
"node_modules/tree-sitter-javascript": {
|
||||
"version": "0.23.1",
|
||||
"resolved": "https://registry.npmjs.org/tree-sitter-javascript/-/tree-sitter-javascript-0.23.1.tgz",
|
||||
@@ -5232,15 +5198,6 @@
|
||||
}
|
||||
}
|
||||
},
|
||||
"node_modules/tree-sitter-javascript/node_modules/node-addon-api": {
|
||||
"version": "8.7.0",
|
||||
"resolved": "https://registry.npmjs.org/node-addon-api/-/node-addon-api-8.7.0.tgz",
|
||||
"integrity": "sha512-9MdFxmkKaOYVTV+XVRG8ArDwwQ77XIgIPyKASB1k3JPq3M8fGQQQE3YpMOrKm6g//Ktx8ivZr8xo1Qmtqub+GA==",
|
||||
"license": "MIT",
|
||||
"engines": {
|
||||
"node": "^18 || ^20 || >= 21"
|
||||
}
|
||||
},
|
||||
"node_modules/tree-sitter-kotlin": {
|
||||
"version": "0.3.8",
|
||||
"resolved": "https://registry.npmjs.org/tree-sitter-kotlin/-/tree-sitter-kotlin-0.3.8.tgz",
|
||||
@@ -5287,15 +5244,6 @@
|
||||
}
|
||||
}
|
||||
},
|
||||
"node_modules/tree-sitter-php/node_modules/node-addon-api": {
|
||||
"version": "8.7.0",
|
||||
"resolved": "https://registry.npmjs.org/node-addon-api/-/node-addon-api-8.7.0.tgz",
|
||||
"integrity": "sha512-9MdFxmkKaOYVTV+XVRG8ArDwwQ77XIgIPyKASB1k3JPq3M8fGQQQE3YpMOrKm6g//Ktx8ivZr8xo1Qmtqub+GA==",
|
||||
"license": "MIT",
|
||||
"engines": {
|
||||
"node": "^18 || ^20 || >= 21"
|
||||
}
|
||||
},
|
||||
"node_modules/tree-sitter-proto": {
|
||||
"resolved": "vendor/tree-sitter-proto",
|
||||
"link": true
|
||||
@@ -5319,15 +5267,6 @@
|
||||
}
|
||||
}
|
||||
},
|
||||
"node_modules/tree-sitter-python/node_modules/node-addon-api": {
|
||||
"version": "8.7.0",
|
||||
"resolved": "https://registry.npmjs.org/node-addon-api/-/node-addon-api-8.7.0.tgz",
|
||||
"integrity": "sha512-9MdFxmkKaOYVTV+XVRG8ArDwwQ77XIgIPyKASB1k3JPq3M8fGQQQE3YpMOrKm6g//Ktx8ivZr8xo1Qmtqub+GA==",
|
||||
"license": "MIT",
|
||||
"engines": {
|
||||
"node": "^18 || ^20 || >= 21"
|
||||
}
|
||||
},
|
||||
"node_modules/tree-sitter-ruby": {
|
||||
"version": "0.23.1",
|
||||
"resolved": "https://registry.npmjs.org/tree-sitter-ruby/-/tree-sitter-ruby-0.23.1.tgz",
|
||||
@@ -5347,15 +5286,6 @@
|
||||
}
|
||||
}
|
||||
},
|
||||
"node_modules/tree-sitter-ruby/node_modules/node-addon-api": {
|
||||
"version": "8.7.0",
|
||||
"resolved": "https://registry.npmjs.org/node-addon-api/-/node-addon-api-8.7.0.tgz",
|
||||
"integrity": "sha512-9MdFxmkKaOYVTV+XVRG8ArDwwQ77XIgIPyKASB1k3JPq3M8fGQQQE3YpMOrKm6g//Ktx8ivZr8xo1Qmtqub+GA==",
|
||||
"license": "MIT",
|
||||
"engines": {
|
||||
"node": "^18 || ^20 || >= 21"
|
||||
}
|
||||
},
|
||||
"node_modules/tree-sitter-rust": {
|
||||
"version": "0.23.1",
|
||||
"resolved": "https://registry.npmjs.org/tree-sitter-rust/-/tree-sitter-rust-0.23.1.tgz",
|
||||
@@ -5375,15 +5305,6 @@
|
||||
}
|
||||
}
|
||||
},
|
||||
"node_modules/tree-sitter-rust/node_modules/node-addon-api": {
|
||||
"version": "8.7.0",
|
||||
"resolved": "https://registry.npmjs.org/node-addon-api/-/node-addon-api-8.7.0.tgz",
|
||||
"integrity": "sha512-9MdFxmkKaOYVTV+XVRG8ArDwwQ77XIgIPyKASB1k3JPq3M8fGQQQE3YpMOrKm6g//Ktx8ivZr8xo1Qmtqub+GA==",
|
||||
"license": "MIT",
|
||||
"engines": {
|
||||
"node": "^18 || ^20 || >= 21"
|
||||
}
|
||||
},
|
||||
"node_modules/tree-sitter-swift": {
|
||||
"version": "0.6.0",
|
||||
"resolved": "https://registry.npmjs.org/tree-sitter-swift/-/tree-sitter-swift-0.6.0.tgz",
|
||||
@@ -5413,16 +5334,6 @@
|
||||
"license": "ISC",
|
||||
"optional": true
|
||||
},
|
||||
"node_modules/tree-sitter-swift/node_modules/node-addon-api": {
|
||||
"version": "8.7.0",
|
||||
"resolved": "https://registry.npmjs.org/node-addon-api/-/node-addon-api-8.7.0.tgz",
|
||||
"integrity": "sha512-9MdFxmkKaOYVTV+XVRG8ArDwwQ77XIgIPyKASB1k3JPq3M8fGQQQE3YpMOrKm6g//Ktx8ivZr8xo1Qmtqub+GA==",
|
||||
"license": "MIT",
|
||||
"optional": true,
|
||||
"engines": {
|
||||
"node": "^18 || ^20 || >= 21"
|
||||
}
|
||||
},
|
||||
"node_modules/tree-sitter-swift/node_modules/which": {
|
||||
"version": "2.0.2",
|
||||
"resolved": "https://registry.npmjs.org/which/-/which-2.0.2.tgz",
|
||||
@@ -5459,24 +5370,6 @@
|
||||
}
|
||||
}
|
||||
},
|
||||
"node_modules/tree-sitter-typescript/node_modules/node-addon-api": {
|
||||
"version": "8.7.0",
|
||||
"resolved": "https://registry.npmjs.org/node-addon-api/-/node-addon-api-8.7.0.tgz",
|
||||
"integrity": "sha512-9MdFxmkKaOYVTV+XVRG8ArDwwQ77XIgIPyKASB1k3JPq3M8fGQQQE3YpMOrKm6g//Ktx8ivZr8xo1Qmtqub+GA==",
|
||||
"license": "MIT",
|
||||
"engines": {
|
||||
"node": "^18 || ^20 || >= 21"
|
||||
}
|
||||
},
|
||||
"node_modules/tree-sitter/node_modules/node-addon-api": {
|
||||
"version": "8.7.0",
|
||||
"resolved": "https://registry.npmjs.org/node-addon-api/-/node-addon-api-8.7.0.tgz",
|
||||
"integrity": "sha512-9MdFxmkKaOYVTV+XVRG8ArDwwQ77XIgIPyKASB1k3JPq3M8fGQQQE3YpMOrKm6g//Ktx8ivZr8xo1Qmtqub+GA==",
|
||||
"license": "MIT",
|
||||
"engines": {
|
||||
"node": "^18 || ^20 || >= 21"
|
||||
}
|
||||
},
|
||||
"node_modules/tslib": {
|
||||
"version": "2.8.1",
|
||||
"resolved": "https://registry.npmjs.org/tslib/-/tslib-2.8.1.tgz",
|
||||
@@ -5884,26 +5777,11 @@
|
||||
},
|
||||
"vendor/tree-sitter-proto": {
|
||||
"version": "0.4.1",
|
||||
"hasInstallScript": true,
|
||||
"license": "MIT",
|
||||
"optional": true,
|
||||
"dependencies": {
|
||||
"node-addon-api": "^8.0.0",
|
||||
"node-gyp-build": "^4.8.0"
|
||||
},
|
||||
"peerDependencies": {
|
||||
"tree-sitter": ">=0.21.0"
|
||||
}
|
||||
},
|
||||
"vendor/tree-sitter-proto/node_modules/node-addon-api": {
|
||||
"version": "8.7.0",
|
||||
"resolved": "https://registry.npmjs.org/node-addon-api/-/node-addon-api-8.7.0.tgz",
|
||||
"integrity": "sha512-9MdFxmkKaOYVTV+XVRG8ArDwwQ77XIgIPyKASB1k3JPq3M8fGQQQE3YpMOrKm6g//Ktx8ivZr8xo1Qmtqub+GA==",
|
||||
"license": "MIT",
|
||||
"optional": true,
|
||||
"engines": {
|
||||
"node": "^18 || ^20 || >= 21"
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "gitnexus",
|
||||
"version": "1.6.1",
|
||||
"version": "1.6.2-rc.13",
|
||||
"description": "Graph-powered code intelligence for AI agents. Index any codebase, query via MCP or CLI.",
|
||||
"author": "Abhigyan Patwari",
|
||||
"license": "PolyForm-Noncommercial-1.0.0",
|
||||
@@ -46,7 +46,7 @@
|
||||
"test:integration": "vitest run test/integration",
|
||||
"test:watch": "vitest",
|
||||
"test:coverage": "vitest run --coverage",
|
||||
"postinstall": "node scripts/patch-tree-sitter-swift.cjs",
|
||||
"postinstall": "node scripts/patch-tree-sitter-swift.cjs && node scripts/build-tree-sitter-proto.cjs",
|
||||
"prepare": "node scripts/build.js",
|
||||
"prepack": "node scripts/build.js"
|
||||
},
|
||||
@@ -71,7 +71,7 @@
|
||||
"pandemonium": "^2.4.0",
|
||||
"tree-sitter": "^0.21.1",
|
||||
"tree-sitter-c": "0.23.2",
|
||||
"tree-sitter-c-sharp": "^0.23.1",
|
||||
"tree-sitter-c-sharp": "0.23.1",
|
||||
"tree-sitter-cpp": "^0.23.4",
|
||||
"tree-sitter-go": "^0.23.0",
|
||||
"tree-sitter-java": "^0.23.5",
|
||||
@@ -84,6 +84,8 @@
|
||||
"uuid": "^13.0.0"
|
||||
},
|
||||
"optionalDependencies": {
|
||||
"node-addon-api": "^8.0.0",
|
||||
"node-gyp-build": "^4.8.0",
|
||||
"tree-sitter-dart": "git+https://github.com/UserNobody14/tree-sitter-dart.git#80e23c07b64494f7e21090bb3450223ef0b192f4",
|
||||
"tree-sitter-kotlin": "^0.3.8",
|
||||
"tree-sitter-proto": "file:./vendor/tree-sitter-proto",
|
||||
|
||||
@@ -0,0 +1,82 @@
|
||||
#!/usr/bin/env node
|
||||
/**
|
||||
* Build tree-sitter-proto native binding.
|
||||
*
|
||||
* Why this script exists:
|
||||
* tree-sitter-proto is vendored under gitnexus/vendor/tree-sitter-proto/
|
||||
* and declared as a `file:` optionalDependency. Previously, the vendored
|
||||
* package had its own `dependencies` and `install` script, which caused
|
||||
* npm to create `vendor/tree-sitter-proto/node_modules/` and
|
||||
* `vendor/tree-sitter-proto/build/` during install. Those directories
|
||||
* blocked `rmdir` on global-install upgrade, producing:
|
||||
*
|
||||
* ENOTEMPTY: directory not empty, rmdir
|
||||
* '.../gitnexus/vendor/tree-sitter-proto/node_modules/node-addon-api'
|
||||
*
|
||||
* (See https://github.com/abhigyanpatwari/GitNexus/issues/836.)
|
||||
*
|
||||
* We stripped `dependencies` and the `install` script from the vendored
|
||||
* package.json, hoisted `node-addon-api` and `node-gyp-build` into
|
||||
* gitnexus's own optionalDependencies, and moved native compilation here.
|
||||
*
|
||||
* What this does:
|
||||
* Runs `npx node-gyp rebuild` inside `node_modules/tree-sitter-proto/`
|
||||
* (which npm creates as a copy of vendor/tree-sitter-proto/ when
|
||||
* resolving the file: dep). Build output lands in
|
||||
* `node_modules/tree-sitter-proto/build/Release/tree_sitter_proto_binding.node`
|
||||
* — under npm-managed territory, safe on upgrade.
|
||||
*
|
||||
* Mirrors scripts/patch-tree-sitter-swift.cjs. Best-effort: if any
|
||||
* precondition fails (optional dep absent, no toolchain, --ignore-scripts),
|
||||
* warn and exit 0 so gitnexus install still succeeds.
|
||||
*/
|
||||
const fs = require('fs');
|
||||
const path = require('path');
|
||||
const { execSync } = require('child_process');
|
||||
|
||||
const protoDir = path.join(__dirname, '..', 'node_modules', 'tree-sitter-proto');
|
||||
const bindingGyp = path.join(protoDir, 'binding.gyp');
|
||||
const bindingNode = path.join(protoDir, 'build', 'Release', 'tree_sitter_proto_binding.node');
|
||||
|
||||
try {
|
||||
if (!fs.existsSync(bindingGyp)) {
|
||||
// tree-sitter-proto is an optionalDependency; absent when install
|
||||
// skipped optional deps or the file: dep was not resolved.
|
||||
process.exit(0);
|
||||
}
|
||||
|
||||
// Skip if the native binding already exists (idempotent re-run).
|
||||
if (fs.existsSync(bindingNode)) {
|
||||
process.exit(0);
|
||||
}
|
||||
|
||||
// Pre-flight: the hoisted build deps must be resolvable.
|
||||
try {
|
||||
require.resolve('node-addon-api');
|
||||
require.resolve('node-gyp-build');
|
||||
} catch (resolveErr) {
|
||||
console.warn(
|
||||
'[tree-sitter-proto] Skipping build: hoisted build deps not resolvable (%s).',
|
||||
resolveErr.message,
|
||||
);
|
||||
console.warn(
|
||||
'[tree-sitter-proto] Proto parsing will be unavailable. Install without --no-optional and with scripts enabled to build.',
|
||||
);
|
||||
process.exit(0);
|
||||
}
|
||||
|
||||
console.log('[tree-sitter-proto] Building native binding...');
|
||||
execSync('npx node-gyp rebuild', {
|
||||
cwd: protoDir,
|
||||
stdio: 'pipe',
|
||||
timeout: 180000,
|
||||
});
|
||||
console.log('[tree-sitter-proto] Native binding built successfully');
|
||||
} catch (err) {
|
||||
console.warn('[tree-sitter-proto] Could not build native binding:', err.message);
|
||||
console.warn(
|
||||
'[tree-sitter-proto] Proto (.proto) parsing will be unavailable. Non-proto gitnexus functionality is unaffected.',
|
||||
);
|
||||
// Exit 0: optionalDependency failures must not fail the gitnexus install.
|
||||
process.exit(0);
|
||||
}
|
||||
@@ -157,6 +157,11 @@ export const initEmbedder = async (
|
||||
try {
|
||||
// Configure transformers.js environment
|
||||
env.allowLocalModels = false;
|
||||
// Default cache to user-writable location. transformers.js defaults to
|
||||
// ./node_modules/.cache inside its own install dir, which is unwritable
|
||||
// when gitnexus is installed globally (e.g. /usr/lib/node_modules/).
|
||||
// Respect HF_HOME if set, otherwise fall back to ~/.cache/huggingface.
|
||||
env.cacheDir = process.env.HF_HOME ?? `${process.env.HOME}/.cache/huggingface`;
|
||||
|
||||
const isDev = process.env.NODE_ENV === 'development';
|
||||
if (isDev) {
|
||||
|
||||
@@ -9,6 +9,7 @@
|
||||
* 5. Create vector index for semantic search
|
||||
*/
|
||||
|
||||
import { createHash } from 'crypto';
|
||||
import {
|
||||
initEmbedder,
|
||||
embedBatch,
|
||||
@@ -16,7 +17,7 @@ import {
|
||||
embeddingToArray,
|
||||
isEmbedderReady,
|
||||
} from './embedder.js';
|
||||
import { generateBatchEmbeddingTexts } from './text-generator.js';
|
||||
import { generateEmbeddingText, generateBatchEmbeddingTexts } from './text-generator.js';
|
||||
import {
|
||||
type EmbeddingProgress,
|
||||
type EmbeddingConfig,
|
||||
@@ -26,9 +27,29 @@ import {
|
||||
DEFAULT_EMBEDDING_CONFIG,
|
||||
EMBEDDABLE_LABELS,
|
||||
} from './types.js';
|
||||
import {
|
||||
EMBEDDING_TABLE_NAME,
|
||||
EMBEDDING_INDEX_NAME,
|
||||
CREATE_VECTOR_INDEX_QUERY,
|
||||
} from '../lbug/schema.js';
|
||||
import { loadVectorExtension } from '../lbug/lbug-adapter.js';
|
||||
|
||||
const isDev = process.env.NODE_ENV === 'development';
|
||||
|
||||
/**
|
||||
* Compute a stable content fingerprint for an embeddable node.
|
||||
* Used to detect when the underlying text has changed so stale vectors
|
||||
* can be replaced (DELETE-then-INSERT, the Kuzu-sanctioned pattern for
|
||||
* vector-indexed rows).
|
||||
*/
|
||||
export const contentHashForNode = (
|
||||
node: EmbeddableNode,
|
||||
config: Partial<EmbeddingConfig> = {},
|
||||
): string => {
|
||||
const text = generateEmbeddingText(node, config);
|
||||
return createHash('sha1').update(text).digest('hex');
|
||||
};
|
||||
|
||||
/**
|
||||
* Progress callback type
|
||||
*/
|
||||
@@ -98,41 +119,32 @@ const batchInsertEmbeddings = async (
|
||||
cypher: string,
|
||||
paramsList: Array<Record<string, any>>,
|
||||
) => Promise<void>,
|
||||
updates: Array<{ id: string; embedding: number[] }>,
|
||||
updates: Array<{ id: string; embedding: number[]; contentHash: string }>,
|
||||
): Promise<void> => {
|
||||
// INSERT into separate embedding table - much more memory efficient!
|
||||
const cypher = `CREATE (e:CodeEmbedding {nodeId: $nodeId, embedding: $embedding})`;
|
||||
const paramsList = updates.map((u) => ({ nodeId: u.id, embedding: u.embedding }));
|
||||
// MERGE instead of CREATE — idempotent, handles concurrent analyzes and partial prior runs
|
||||
const cypher = `MERGE (e:${EMBEDDING_TABLE_NAME} {nodeId: $nodeId}) SET e.embedding = $embedding, e.contentHash = $contentHash`;
|
||||
const paramsList = updates.map((u) => ({
|
||||
nodeId: u.id,
|
||||
embedding: u.embedding,
|
||||
contentHash: u.contentHash,
|
||||
}));
|
||||
await executeWithReusedStatement(cypher, paramsList);
|
||||
};
|
||||
|
||||
/**
|
||||
* Create the vector index for semantic search
|
||||
* Now indexes the separate CodeEmbedding table
|
||||
* Now indexes the separate CodeEmbedding table.
|
||||
* Delegates extension loading to lbug-adapter's loadVectorExtension(),
|
||||
* which owns the VECTOR extension lifecycle and state tracking.
|
||||
*/
|
||||
let vectorExtensionLoaded = false;
|
||||
|
||||
const createVectorIndex = async (
|
||||
executeQuery: (cypher: string) => Promise<any[]>,
|
||||
): Promise<void> => {
|
||||
// LadybugDB v0.15+ requires explicit VECTOR extension loading (once per session)
|
||||
if (!vectorExtensionLoaded) {
|
||||
try {
|
||||
await executeQuery('INSTALL VECTOR');
|
||||
await executeQuery('LOAD EXTENSION VECTOR');
|
||||
vectorExtensionLoaded = true;
|
||||
} catch {
|
||||
// Extension may already be loaded — CREATE_VECTOR_INDEX will fail clearly if not
|
||||
vectorExtensionLoaded = true;
|
||||
}
|
||||
}
|
||||
|
||||
const cypher = `
|
||||
CALL CREATE_VECTOR_INDEX('CodeEmbedding', 'code_embedding_idx', 'embedding', metric := 'cosine')
|
||||
`;
|
||||
// Delegate to the adapter which tracks loaded state and handles DB reconnect resets
|
||||
await loadVectorExtension();
|
||||
|
||||
try {
|
||||
await executeQuery(cypher);
|
||||
await executeQuery(CREATE_VECTOR_INDEX_QUERY);
|
||||
} catch (error) {
|
||||
// Index might already exist
|
||||
if (isDev) {
|
||||
@@ -148,7 +160,9 @@ const createVectorIndex = async (
|
||||
* @param executeWithReusedStatement - Function to execute with reused prepared statement
|
||||
* @param onProgress - Callback for progress updates
|
||||
* @param config - Optional configuration override
|
||||
* @param skipNodeIds - Optional set of node IDs that already have embeddings (incremental mode)
|
||||
* @param existingEmbeddings - Optional map of nodeId → contentHash for incremental mode.
|
||||
* Nodes whose hash matches are skipped; nodes with a changed hash are DELETE'd
|
||||
* and re-embedded; nodes not in the map are embedded fresh.
|
||||
*/
|
||||
export const runEmbeddingPipeline = async (
|
||||
executeQuery: (cypher: string) => Promise<any[]>,
|
||||
@@ -158,7 +172,7 @@ export const runEmbeddingPipeline = async (
|
||||
) => Promise<void>,
|
||||
onProgress: EmbeddingProgressCallback,
|
||||
config: Partial<EmbeddingConfig> = {},
|
||||
skipNodeIds?: Set<string>,
|
||||
existingEmbeddings?: Map<string, string>,
|
||||
): Promise<void> => {
|
||||
const finalConfig = { ...DEFAULT_EMBEDDING_CONFIG, ...config };
|
||||
|
||||
@@ -194,13 +208,57 @@ export const runEmbeddingPipeline = async (
|
||||
// Phase 2: Query embeddable nodes
|
||||
let nodes = await queryEmbeddableNodes(executeQuery);
|
||||
|
||||
// Incremental mode: filter out nodes that already have embeddings
|
||||
if (skipNodeIds && skipNodeIds.size > 0) {
|
||||
// Incremental mode: compare content hashes, delete stale rows, skip fresh ones.
|
||||
// Computed hashes for stale nodes are cached so batchInsertEmbeddings can reuse them
|
||||
// (avoids double computation).
|
||||
const computedStaleHashes = new Map<string, string>();
|
||||
if (existingEmbeddings && existingEmbeddings.size > 0) {
|
||||
const beforeCount = nodes.length;
|
||||
nodes = nodes.filter((n) => !skipNodeIds.has(n.id));
|
||||
const staleNodeIds: string[] = [];
|
||||
nodes = nodes.filter((n) => {
|
||||
const existingHash = existingEmbeddings.get(n.id);
|
||||
if (existingHash === undefined) {
|
||||
// New node — needs embedding
|
||||
return true;
|
||||
}
|
||||
const currentHash = contentHashForNode(n, finalConfig);
|
||||
if (currentHash !== existingHash) {
|
||||
// Content changed — cache hash for reuse during insert, mark for DELETE + re-embed
|
||||
computedStaleHashes.set(n.id, currentHash);
|
||||
staleNodeIds.push(n.id);
|
||||
return true;
|
||||
}
|
||||
// Hash matches — skip (fresh); no need to cache hash for skipped nodes
|
||||
return false;
|
||||
});
|
||||
|
||||
// DELETE stale embedding rows so they can be re-inserted
|
||||
// (Kuzu forbids SET on vector-indexed properties; DELETE-then-INSERT is the sanctioned pattern)
|
||||
if (staleNodeIds.length > 0) {
|
||||
if (isDev) {
|
||||
console.log(`🔄 Deleting ${staleNodeIds.length} stale embedding rows for re-embed`);
|
||||
}
|
||||
try {
|
||||
await executeWithReusedStatement(
|
||||
`MATCH (e:${EMBEDDING_TABLE_NAME} {nodeId: $nodeId}) DELETE e`,
|
||||
staleNodeIds.map((nodeId) => ({ nodeId })),
|
||||
);
|
||||
} catch (err) {
|
||||
// "does not exist" = rows already gone — safe to proceed.
|
||||
// All other errors risk vector-index corruption (Kuzu requires DELETE-before-INSERT
|
||||
// for vector-indexed properties) — propagate so the pipeline aborts cleanly.
|
||||
const msg = err instanceof Error ? err.message : String(err);
|
||||
if (!msg.includes('does not exist')) {
|
||||
throw new Error(
|
||||
`[embed] Failed to delete stale embedding rows — aborting to prevent vector-index corruption: ${msg}`,
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (isDev) {
|
||||
console.log(
|
||||
`📦 Incremental embeddings: ${beforeCount} total, ${skipNodeIds.size} cached, ${nodes.length} to embed`,
|
||||
`📦 Incremental embeddings: ${beforeCount} total, ${existingEmbeddings.size} cached, ${staleNodeIds.length} stale, ${nodes.length} to embed`,
|
||||
);
|
||||
}
|
||||
}
|
||||
@@ -212,6 +270,11 @@ export const runEmbeddingPipeline = async (
|
||||
}
|
||||
|
||||
if (totalNodes === 0) {
|
||||
// Ensure the vector index exists even when no new nodes need embedding.
|
||||
// A prior crash or first-time incremental run may have left CodeEmbedding
|
||||
// rows without ever reaching index creation.
|
||||
await createVectorIndex(executeQuery);
|
||||
|
||||
onProgress({
|
||||
phase: 'ready',
|
||||
percent: 100,
|
||||
@@ -250,6 +313,7 @@ export const runEmbeddingPipeline = async (
|
||||
const updates = batch.map((node, i) => ({
|
||||
id: node.id,
|
||||
embedding: embeddingToArray(embeddings[i]),
|
||||
contentHash: computedStaleHashes.get(node.id) ?? contentHashForNode(node, finalConfig),
|
||||
}));
|
||||
|
||||
await batchInsertEmbeddings(executeWithReusedStatement, updates);
|
||||
@@ -338,7 +402,7 @@ export const semanticSearch = async (
|
||||
|
||||
// Query the vector index on CodeEmbedding to get nodeIds and distances
|
||||
const vectorQuery = `
|
||||
CALL QUERY_VECTOR_INDEX('CodeEmbedding', 'code_embedding_idx',
|
||||
CALL QUERY_VECTOR_INDEX('${EMBEDDING_TABLE_NAME}', '${EMBEDDING_INDEX_NAME}',
|
||||
CAST(${queryVecStr} AS FLOAT[${queryVec.length}]), ${k})
|
||||
YIELD node AS emb, distance
|
||||
WITH emb, distance
|
||||
|
||||
@@ -79,17 +79,50 @@ export class ManifestExtractor {
|
||||
links: GroupManifestLink[],
|
||||
dbExecutors?: Map<string, CypherExecutor>,
|
||||
): Promise<ManifestExtractResult> {
|
||||
// Resolve all (repo, link) pairs in parallel. The previous sequential
|
||||
// await-per-link produced 2N round-trips; parallel resolution uses the
|
||||
// per-repo executor pool directly and scales linearly with manifest size.
|
||||
//
|
||||
// Memoization: a manifest can list the same contract multiple times
|
||||
// (e.g. a consumer and provider declaration, or cross-referenced groups).
|
||||
// Key on (repo, type, contract) — the canonical input to the Cypher
|
||||
// query — so duplicate links resolve to one DB hit.
|
||||
type ResolvedSymbol = { filePath: string; name: string; uid: string } | null;
|
||||
const resolveCache = new Map<string, Promise<ResolvedSymbol>>();
|
||||
const resolveOnce = (repo: string, link: GroupManifestLink): Promise<ResolvedSymbol> => {
|
||||
const key = `${repo}\u0000${link.type}\u0000${link.contract}`;
|
||||
let pending = resolveCache.get(key);
|
||||
if (!pending) {
|
||||
pending = this.resolveSymbol(repo, link, dbExecutors);
|
||||
resolveCache.set(key, pending);
|
||||
}
|
||||
return pending;
|
||||
};
|
||||
|
||||
const perLink = await Promise.all(
|
||||
links.map(async (link) => {
|
||||
const contractId = this.buildContractId(link.type, link.contract);
|
||||
const providerRepo = link.role === 'provider' ? link.from : link.to;
|
||||
const consumerRepo = link.role === 'provider' ? link.to : link.from;
|
||||
const [providerSymbol, consumerSymbol] = await Promise.all([
|
||||
resolveOnce(providerRepo, link),
|
||||
resolveOnce(consumerRepo, link),
|
||||
]);
|
||||
return { link, contractId, providerRepo, consumerRepo, providerSymbol, consumerSymbol };
|
||||
}),
|
||||
);
|
||||
|
||||
const contracts: StoredContract[] = [];
|
||||
const crossLinks: CrossLink[] = [];
|
||||
|
||||
for (const link of links) {
|
||||
const contractId = this.buildContractId(link.type, link.contract);
|
||||
|
||||
const providerRepo = link.role === 'provider' ? link.from : link.to;
|
||||
const consumerRepo = link.role === 'provider' ? link.to : link.from;
|
||||
|
||||
const providerSymbol = await this.resolveSymbol(providerRepo, link, dbExecutors);
|
||||
const consumerSymbol = await this.resolveSymbol(consumerRepo, link, dbExecutors);
|
||||
for (const {
|
||||
link,
|
||||
contractId,
|
||||
providerRepo,
|
||||
consumerRepo,
|
||||
providerSymbol,
|
||||
consumerSymbol,
|
||||
} of perLink) {
|
||||
const providerRef = providerSymbol || { filePath: '', name: link.contract };
|
||||
const consumerRef = consumerSymbol || { filePath: '', name: link.contract };
|
||||
// When the resolver finds a real graph symbol we keep its uid, otherwise
|
||||
|
||||
@@ -7,6 +7,7 @@ import type { GroupConfig, RepoHandle, RepoSnapshot, StoredContract, CrossLink }
|
||||
import { HttpRouteExtractor } from './extractors/http-route-extractor.js';
|
||||
import { GrpcExtractor } from './extractors/grpc-extractor.js';
|
||||
import { TopicExtractor } from './extractors/topic-extractor.js';
|
||||
import { ManifestExtractor } from './extractors/manifest-extractor.js';
|
||||
import { runExactMatch } from './matching.js';
|
||||
import { detectServiceBoundaries, assignService } from './service-boundary-detector.js';
|
||||
import type { CypherExecutor } from './contract-extractor.js';
|
||||
@@ -60,10 +61,28 @@ function defaultResolveHandle(allEntries: RegistryEntry[]) {
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Dedupe cross-links that point from the same consumer endpoint to the same
|
||||
* provider endpoint for the same contract. Preserves first-seen order so the
|
||||
* caller controls precedence (e.g., pass manifest links first).
|
||||
*/
|
||||
function dedupeCrossLinks(links: CrossLink[]): CrossLink[] {
|
||||
const seen = new Set<string>();
|
||||
const out: CrossLink[] = [];
|
||||
for (const link of links) {
|
||||
const key = `${link.from.repo}::${link.from.symbolUid}|${link.to.repo}::${link.to.symbolUid}|${link.type}|${link.contractId}`;
|
||||
if (seen.has(key)) continue;
|
||||
seen.add(key);
|
||||
out.push(link);
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
export async function syncGroup(config: GroupConfig, opts?: SyncOptions): Promise<SyncResult> {
|
||||
const missingRepos: string[] = [];
|
||||
const repoSnapshots: Record<string, RepoSnapshot> = {};
|
||||
let autoContracts: StoredContract[] = [];
|
||||
let manifestCrossLinks: CrossLink[] = [];
|
||||
let dbExecutors: Map<string, CypherExecutor> | undefined;
|
||||
|
||||
const eo = opts?.extractorOverride;
|
||||
@@ -158,8 +177,44 @@ export async function syncGroup(config: GroupConfig, opts?: SyncOptions): Promis
|
||||
}
|
||||
}
|
||||
|
||||
// Process manifest links declared in group.yaml.
|
||||
// ManifestExtractor is fully implemented but was never wired into this
|
||||
// pipeline — config.links were parsed and validated but silently dropped.
|
||||
// Placed after the DB try/finally: resolveSymbol falls back to synthetic
|
||||
// UIDs when dbExecutors is undefined or a pool is closed, so cross-links
|
||||
// are always generated regardless of whether real DB executors are available.
|
||||
if (config.links.length > 0) {
|
||||
// Warn about dangling links that reference repos not declared in config.repos.
|
||||
// They still generate cross-links via synthetic UIDs (determinism is preserved),
|
||||
// but the operator probably meant something that now silently does nothing useful.
|
||||
const knownRepos = new Set(Object.keys(config.repos));
|
||||
for (const link of config.links) {
|
||||
const dangling = [link.from, link.to].filter((r) => !knownRepos.has(r));
|
||||
if (dangling.length > 0) {
|
||||
console.warn(
|
||||
`[group/sync] manifest link ${link.type}:${link.contract} references repos not in config.repos: ${dangling.join(', ')} — cross-links will use synthetic UIDs`,
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
const manifestEx = new ManifestExtractor();
|
||||
const manifestResult = await manifestEx.extractFromManifest(config.links, dbExecutors);
|
||||
autoContracts.push(...manifestResult.contracts);
|
||||
manifestCrossLinks = manifestResult.crossLinks;
|
||||
if (opts?.verbose) {
|
||||
console.log(
|
||||
` manifest: ${manifestCrossLinks.length} cross-links from ${config.links.length} declared links`,
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
const { matched, unmatched } = runExactMatch(autoContracts);
|
||||
const crossLinks: CrossLink[] = matched;
|
||||
|
||||
// Dedupe cross-links. Manifest contracts participate in runExactMatch, so a
|
||||
// manifest-declared link can also emit a matchType:'exact' CrossLink with the
|
||||
// same endpoints. Prefer the manifest version — it reflects operator intent
|
||||
// and carries matchType:'manifest' which downstream consumers may rely on.
|
||||
const crossLinks = dedupeCrossLinks([...manifestCrossLinks, ...matched]);
|
||||
const allContracts: StoredContract[] = autoContracts;
|
||||
|
||||
const registry: ContractRegistry = {
|
||||
|
||||
@@ -315,14 +315,18 @@ export const streamAllCSVsToDisk = async (
|
||||
CodeElement: codeElemWriter,
|
||||
};
|
||||
|
||||
const seenFileIds = new Set<string>();
|
||||
// Deduplicate all node types — the pipeline can produce duplicate IDs across
|
||||
// all symbol types (Class, Method, Function, etc.), not just File nodes.
|
||||
// A single Set covering every label prevents PK violations on COPY.
|
||||
const seenNodeIds = new Set<string>();
|
||||
|
||||
// --- SINGLE PASS over all nodes ---
|
||||
for (const node of graph.iterNodes()) {
|
||||
if (seenNodeIds.has(node.id)) continue;
|
||||
seenNodeIds.add(node.id);
|
||||
|
||||
switch (node.label) {
|
||||
case 'File': {
|
||||
if (seenFileIds.has(node.id)) break;
|
||||
seenFileIds.add(node.id);
|
||||
const content = await extractContent(node, contentCache);
|
||||
await fileWriter.addRow(
|
||||
[
|
||||
|
||||
@@ -1,6 +1,8 @@
|
||||
import fs from 'fs/promises';
|
||||
import { createReadStream, createWriteStream } from 'fs';
|
||||
import { createInterface } from 'readline';
|
||||
import { once } from 'events';
|
||||
import { finished } from 'stream/promises';
|
||||
import path from 'path';
|
||||
import lbug from '@ladybugdb/core';
|
||||
import { KnowledgeGraph } from '../graph/types.js';
|
||||
@@ -9,16 +11,148 @@ import {
|
||||
REL_TABLE_NAME,
|
||||
SCHEMA_QUERIES,
|
||||
EMBEDDING_TABLE_NAME,
|
||||
STALE_HASH_SENTINEL,
|
||||
NodeTableName,
|
||||
} from './schema.js';
|
||||
import { streamAllCSVsToDisk } from './csv-generator.js';
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Relationship CSV splitting — extracted for testability (PR #818)
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/** Factory for creating WriteStreams — injectable for testing. */
|
||||
export type WriteStreamFactory = (filePath: string) => import('fs').WriteStream;
|
||||
|
||||
/** Result of splitting the relationship CSV into per-label-pair files. */
|
||||
export interface RelCsvSplitResult {
|
||||
relHeader: string;
|
||||
relsByPairMeta: Map<string, { csvPath: string; rows: number }>;
|
||||
pairWriteStreams: Map<string, import('fs').WriteStream>;
|
||||
skippedRels: number;
|
||||
totalValidRels: number;
|
||||
}
|
||||
|
||||
/**
|
||||
* Split a relationship CSV into per-label-pair files on disk.
|
||||
*
|
||||
* Streams the CSV line-by-line, routing each relationship to a file named
|
||||
* `rel_{fromLabel}_{toLabel}.csv`. Handles backpressure correctly: only one
|
||||
* drain listener per stream at a time, and readline resumes only when ALL
|
||||
* backpressured streams have drained.
|
||||
*
|
||||
* @param csvPath Path to the combined relationship CSV
|
||||
* @param csvDir Directory to write per-pair CSV files
|
||||
* @param validTables Set of valid node table names
|
||||
* @param getNodeLabel Function to extract the label from a node ID
|
||||
* @param wsFactory Optional WriteStream factory (defaults to fs.createWriteStream)
|
||||
*/
|
||||
export const splitRelCsvByLabelPair = async (
|
||||
csvPath: string,
|
||||
csvDir: string,
|
||||
validTables: Set<string>,
|
||||
getNodeLabel: (id: string) => string,
|
||||
wsFactory: WriteStreamFactory = (p) => createWriteStream(p, 'utf-8'),
|
||||
): Promise<RelCsvSplitResult> => {
|
||||
let relHeader = '';
|
||||
const relsByPairMeta = new Map<string, { csvPath: string; rows: number }>();
|
||||
const pairWriteStreams = new Map<string, import('fs').WriteStream>();
|
||||
let skippedRels = 0;
|
||||
let totalValidRels = 0;
|
||||
|
||||
const inputStream = createReadStream(csvPath, 'utf-8');
|
||||
const rl = createInterface({ input: inputStream, crlfDelay: Infinity });
|
||||
|
||||
// If any pair WriteStream errors (disk full, EMFILE, etc.) or the input
|
||||
// stream fails, we need to abort the pending `once(ws, 'drain')` await.
|
||||
// An AbortController gives us one signal to cancel all pending waits
|
||||
// without a custom state machine.
|
||||
const abortOnError = new AbortController();
|
||||
let streamError: Error | null = null;
|
||||
const markStreamError = (err: Error): void => {
|
||||
streamError ??= err;
|
||||
abortOnError.abort(err);
|
||||
};
|
||||
|
||||
try {
|
||||
// `for await (const line of rl)` replaces the old manual
|
||||
// on('line')/pause()/resume()/waitingForDrain state machine: readline's
|
||||
// async iterator naturally serializes line delivery with our awaits, so
|
||||
// at most one ws can be in backpressure at a time and we just await its
|
||||
// 'drain' event.
|
||||
let isFirst = true;
|
||||
for await (const line of rl) {
|
||||
if (streamError) throw streamError;
|
||||
if (isFirst) {
|
||||
relHeader = line;
|
||||
isFirst = false;
|
||||
continue;
|
||||
}
|
||||
if (!line.trim()) continue;
|
||||
const match = line.match(/"([^"]*)","([^"]*)"/);
|
||||
if (!match) {
|
||||
skippedRels++;
|
||||
continue;
|
||||
}
|
||||
const fromLabel = getNodeLabel(match[1]);
|
||||
const toLabel = getNodeLabel(match[2]);
|
||||
if (!validTables.has(fromLabel) || !validTables.has(toLabel)) {
|
||||
skippedRels++;
|
||||
continue;
|
||||
}
|
||||
|
||||
const pairKey = `${fromLabel}|${toLabel}`;
|
||||
let ws = pairWriteStreams.get(pairKey);
|
||||
if (!ws) {
|
||||
const pairCsvPath = path.join(csvDir, `rel_${fromLabel}_${toLabel}.csv`);
|
||||
ws = wsFactory(pairCsvPath);
|
||||
ws.on('error', markStreamError);
|
||||
pairWriteStreams.set(pairKey, ws);
|
||||
relsByPairMeta.set(pairKey, { csvPath: pairCsvPath, rows: 0 });
|
||||
if (!ws.write(relHeader + '\n')) {
|
||||
await once(ws, 'drain', { signal: abortOnError.signal });
|
||||
}
|
||||
}
|
||||
|
||||
if (!ws.write(line + '\n')) {
|
||||
await once(ws, 'drain', { signal: abortOnError.signal });
|
||||
}
|
||||
relsByPairMeta.get(pairKey)!.rows++;
|
||||
totalValidRels++;
|
||||
}
|
||||
if (streamError) throw streamError;
|
||||
} catch (err) {
|
||||
// Tear down everything so no fd is left dangling. If the abort was caused
|
||||
// by a stream error, rethrow that error (more actionable than AbortError).
|
||||
for (const ws of pairWriteStreams.values()) ws.destroy();
|
||||
inputStream.destroy();
|
||||
throw streamError ?? err;
|
||||
} finally {
|
||||
// Readline 'close' fires before the underlying fs.ReadStream releases its
|
||||
// fd — on Windows that race caused ENOTEMPTY on the parent dir.
|
||||
// stream/promises.finished is the stdlib "wait until this stream is fully
|
||||
// closed" primitive and handles both success and error paths.
|
||||
await finished(inputStream).catch(() => {});
|
||||
}
|
||||
|
||||
return { relHeader, relsByPairMeta, pairWriteStreams, skippedRels, totalValidRels };
|
||||
};
|
||||
|
||||
let db: lbug.Database | null = null;
|
||||
let conn: lbug.Connection | null = null;
|
||||
let currentDbPath: string | null = null;
|
||||
let ftsLoaded = false;
|
||||
let vectorExtensionLoaded = false;
|
||||
|
||||
/**
|
||||
* Check if an error indicates a missing column or table (schema-level problem)
|
||||
* rather than a transient/connection error. Used for legacy DB fallback logic.
|
||||
*/
|
||||
const isMissingColumnOrTableError = (msg: string): boolean =>
|
||||
msg.includes('does not exist') ||
|
||||
// Kuzu-specific: "(table|column|property) ... not found" — narrow enough to avoid
|
||||
// matching transient errors like "connection not found" or "key not found".
|
||||
/(table|column|property).*not found/i.test(msg);
|
||||
|
||||
/** Expose the current Database for pool adapter reuse in tests. */
|
||||
export const getDatabase = (): lbug.Database | null => db;
|
||||
|
||||
@@ -247,75 +381,17 @@ export const loadGraphToLbug = async (
|
||||
}
|
||||
|
||||
// Bulk COPY relationships — split by FROM→TO label pair (LadybugDB requires it)
|
||||
// Stream-read the relation CSV line by line and write directly to per-pair
|
||||
// temp files on disk. This avoids accumulating potentially millions of CSV
|
||||
// lines in memory which could exceed V8 Map or array limits on large repos.
|
||||
let relHeader = '';
|
||||
const relsByPairMeta = new Map<string, { csvPath: string; rows: number }>();
|
||||
const pairWriteStreams = new Map<string, import('fs').WriteStream>();
|
||||
let skippedRels = 0;
|
||||
let totalValidRels = 0;
|
||||
const { relHeader, relsByPairMeta, pairWriteStreams, skippedRels, totalValidRels } =
|
||||
await splitRelCsvByLabelPair(csvResult.relCsvPath, csvDir, validTables, getNodeLabel);
|
||||
|
||||
await new Promise<void>((resolve, reject) => {
|
||||
const rl = createInterface({
|
||||
input: createReadStream(csvResult.relCsvPath, 'utf-8'),
|
||||
crlfDelay: Infinity,
|
||||
});
|
||||
let isFirst = true;
|
||||
rl.on('line', (line) => {
|
||||
if (isFirst) {
|
||||
relHeader = line;
|
||||
isFirst = false;
|
||||
return;
|
||||
}
|
||||
if (!line.trim()) return;
|
||||
const match = line.match(/"([^"]*)","([^"]*)"/);
|
||||
if (!match) {
|
||||
skippedRels++;
|
||||
return;
|
||||
}
|
||||
const fromLabel = getNodeLabel(match[1]);
|
||||
const toLabel = getNodeLabel(match[2]);
|
||||
if (!validTables.has(fromLabel) || !validTables.has(toLabel)) {
|
||||
skippedRels++;
|
||||
return;
|
||||
}
|
||||
const pairKey = `${fromLabel}|${toLabel}`;
|
||||
let ws = pairWriteStreams.get(pairKey);
|
||||
if (!ws) {
|
||||
const pairCsvPath = path.join(csvDir, `rel_${fromLabel}_${toLabel}.csv`);
|
||||
ws = createWriteStream(pairCsvPath, 'utf-8');
|
||||
ws.write(relHeader + '\n');
|
||||
pairWriteStreams.set(pairKey, ws);
|
||||
relsByPairMeta.set(pairKey, { csvPath: pairCsvPath, rows: 0 });
|
||||
}
|
||||
const ok = ws.write(line + '\n');
|
||||
relsByPairMeta.get(pairKey)!.rows++;
|
||||
totalValidRels++;
|
||||
// Handle backpressure: pause reading when the write buffer is full,
|
||||
// resume when the stream drains. Prevents unbounded memory growth
|
||||
// on repos with millions of relationships.
|
||||
if (!ok) {
|
||||
rl.pause();
|
||||
ws.once('drain', () => rl.resume());
|
||||
}
|
||||
});
|
||||
rl.on('close', resolve);
|
||||
rl.on('error', (err) => {
|
||||
// Destroy all open write streams to avoid resource leaks
|
||||
for (const ws of pairWriteStreams.values()) ws.destroy();
|
||||
reject(err);
|
||||
});
|
||||
});
|
||||
|
||||
// Close all per-pair write streams before COPY
|
||||
// Close all per-pair write streams before COPY. `stream/promises.finished`
|
||||
// resolves on the stream's 'finish' event and rejects on 'error' — replaces
|
||||
// a hand-rolled promisification with the stdlib primitive.
|
||||
await Promise.all(
|
||||
Array.from(pairWriteStreams.values()).map(
|
||||
(ws) =>
|
||||
new Promise<void>((resolve, reject) =>
|
||||
ws.end((err: Error | undefined) => (err ? reject(err) : resolve())),
|
||||
),
|
||||
),
|
||||
Array.from(pairWriteStreams.values()).map(async (ws) => {
|
||||
ws.end();
|
||||
await finished(ws);
|
||||
}),
|
||||
);
|
||||
|
||||
const insertedRels = totalValidRels;
|
||||
@@ -808,18 +884,35 @@ export const getLbugStats = async (): Promise<{ nodes: number; edges: number }>
|
||||
*/
|
||||
export const loadCachedEmbeddings = async (): Promise<{
|
||||
embeddingNodeIds: Set<string>;
|
||||
embeddings: Array<{ nodeId: string; embedding: number[] }>;
|
||||
embeddings: Array<{ nodeId: string; embedding: number[]; contentHash?: string }>;
|
||||
}> => {
|
||||
if (!conn) {
|
||||
return { embeddingNodeIds: new Set(), embeddings: [] };
|
||||
}
|
||||
|
||||
const embeddingNodeIds = new Set<string>();
|
||||
const embeddings: Array<{ nodeId: string; embedding: number[] }> = [];
|
||||
const embeddings: Array<{ nodeId: string; embedding: number[]; contentHash?: string }> = [];
|
||||
try {
|
||||
const rows = await conn.query(
|
||||
`MATCH (e:${EMBEDDING_TABLE_NAME}) RETURN e.nodeId AS nodeId, e.embedding AS embedding`,
|
||||
);
|
||||
// Try to read contentHash alongside the embedding
|
||||
let rows: any;
|
||||
let hasContentHash = true;
|
||||
try {
|
||||
rows = await conn.query(
|
||||
`MATCH (e:${EMBEDDING_TABLE_NAME}) RETURN e.nodeId AS nodeId, e.embedding AS embedding, e.contentHash AS contentHash`,
|
||||
);
|
||||
} catch (err: any) {
|
||||
// Only fall back for missing-column errors (legacy DBs without contentHash).
|
||||
// Rethrow transient / connection errors so callers see them.
|
||||
const msg = err?.message ?? '';
|
||||
if (isMissingColumnOrTableError(msg)) {
|
||||
hasContentHash = false;
|
||||
rows = await conn.query(
|
||||
`MATCH (e:${EMBEDDING_TABLE_NAME}) RETURN e.nodeId AS nodeId, e.embedding AS embedding`,
|
||||
);
|
||||
} else {
|
||||
throw err;
|
||||
}
|
||||
}
|
||||
const result = Array.isArray(rows) ? rows[0] : rows;
|
||||
for (const row of await result.getAll()) {
|
||||
const nodeId = String(row.nodeId ?? row[0] ?? '');
|
||||
@@ -832,6 +925,7 @@ export const loadCachedEmbeddings = async (): Promise<{
|
||||
embedding: Array.isArray(embedding)
|
||||
? embedding.map(Number)
|
||||
: Array.from(embedding as any).map(Number),
|
||||
contentHash: hasContentHash ? (row.contentHash ?? row[2] ?? undefined) : undefined,
|
||||
});
|
||||
}
|
||||
}
|
||||
@@ -842,6 +936,63 @@ export const loadCachedEmbeddings = async (): Promise<{
|
||||
return { embeddingNodeIds, embeddings };
|
||||
};
|
||||
|
||||
/**
|
||||
* Fetch existing embedding hashes from CodeEmbedding table for incremental embedding.
|
||||
* Returns a Map<nodeId, contentHash> suitable for passing to `runEmbeddingPipeline`.
|
||||
* Handles legacy DBs without the `contentHash` column (all rows treated as stale with empty hash).
|
||||
* Returns undefined if the CodeEmbedding table does not exist.
|
||||
*
|
||||
* @param execQuery - Cypher query executor (typically pool-adapter's `executeQuery`)
|
||||
*/
|
||||
export const fetchExistingEmbeddingHashes = async (
|
||||
execQuery: (cypher: string) => Promise<any[]>,
|
||||
): Promise<Map<string, string> | undefined> => {
|
||||
try {
|
||||
const rows = await execQuery(
|
||||
`MATCH (e:${EMBEDDING_TABLE_NAME}) RETURN e.nodeId AS nodeId, e.contentHash AS contentHash`,
|
||||
);
|
||||
if (!rows || rows.length === 0) return undefined;
|
||||
const map = new Map<string, string>();
|
||||
for (const r of rows) {
|
||||
const nodeId = r.nodeId ?? r[0];
|
||||
const hash = r.contentHash ?? r[1] ?? STALE_HASH_SENTINEL;
|
||||
if (nodeId) {
|
||||
// Empty/null contentHash means legacy row — treat as stale so it gets re-embedded
|
||||
map.set(nodeId, hash || STALE_HASH_SENTINEL);
|
||||
}
|
||||
}
|
||||
return map;
|
||||
} catch (err: any) {
|
||||
const msg = err?.message ?? '';
|
||||
if (isMissingColumnOrTableError(msg)) {
|
||||
// Column or table missing — try fallback without contentHash
|
||||
try {
|
||||
const rows = await execQuery(`MATCH (e:${EMBEDDING_TABLE_NAME}) RETURN e.nodeId AS nodeId`);
|
||||
if (!rows || rows.length === 0) return undefined;
|
||||
const map = new Map<string, string>();
|
||||
for (const r of rows) {
|
||||
const nodeId = r.nodeId ?? r[0];
|
||||
if (nodeId) map.set(nodeId, STALE_HASH_SENTINEL); // no contentHash — treat as stale
|
||||
}
|
||||
console.log(
|
||||
`[embed] ${map.size} nodes in legacy DB (no contentHash) — all treated as stale`,
|
||||
);
|
||||
return map;
|
||||
} catch (fallbackErr: any) {
|
||||
const fallbackMsg = fallbackErr?.message ?? '';
|
||||
if (isMissingColumnOrTableError(fallbackMsg)) {
|
||||
console.log(
|
||||
`[embed] CodeEmbedding table not yet present — full embedding run (${fallbackMsg})`,
|
||||
);
|
||||
return undefined;
|
||||
}
|
||||
throw fallbackErr;
|
||||
}
|
||||
}
|
||||
throw err;
|
||||
}
|
||||
};
|
||||
|
||||
export const closeLbug = async (): Promise<void> => {
|
||||
if (conn) {
|
||||
try {
|
||||
|
||||
@@ -436,10 +436,20 @@ if (Number.isNaN(_rawDims) || _rawDims <= 0) {
|
||||
}
|
||||
export const EMBEDDING_DIMS = _rawDims;
|
||||
|
||||
/** HNSW vector index name for the CodeEmbedding table. */
|
||||
export const EMBEDDING_INDEX_NAME = 'code_embedding_idx';
|
||||
|
||||
/**
|
||||
* Sentinel value for "no content hash available" — used in legacy DBs and null rows.
|
||||
* Nodes with this hash are always treated as stale and re-embedded.
|
||||
*/
|
||||
export const STALE_HASH_SENTINEL = '';
|
||||
|
||||
export const EMBEDDING_SCHEMA = `
|
||||
CREATE NODE TABLE ${EMBEDDING_TABLE_NAME} (
|
||||
nodeId STRING,
|
||||
embedding FLOAT[${EMBEDDING_DIMS}],
|
||||
contentHash STRING,
|
||||
PRIMARY KEY (nodeId)
|
||||
)`;
|
||||
|
||||
@@ -448,7 +458,7 @@ CREATE NODE TABLE ${EMBEDDING_TABLE_NAME} (
|
||||
* Uses HNSW (Hierarchical Navigable Small World) algorithm with cosine similarity
|
||||
*/
|
||||
export const CREATE_VECTOR_INDEX_QUERY = `
|
||||
CALL CREATE_VECTOR_INDEX('${EMBEDDING_TABLE_NAME}', 'code_embedding_idx', 'embedding', metric := 'cosine')
|
||||
CALL CREATE_VECTOR_INDEX('${EMBEDDING_TABLE_NAME}', '${EMBEDDING_INDEX_NAME}', 'embedding', metric := 'cosine')
|
||||
`;
|
||||
|
||||
// ============================================================================
|
||||
|
||||
@@ -32,6 +32,8 @@ import {
|
||||
} from '../storage/repo-manager.js';
|
||||
import { getCurrentCommit, hasGitDir } from '../storage/git.js';
|
||||
import { generateAIContextFiles } from '../cli/ai-context.js';
|
||||
import { EMBEDDING_TABLE_NAME } from './lbug/schema.js';
|
||||
import { STALE_HASH_SENTINEL } from './lbug/schema.js';
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Public types
|
||||
@@ -138,7 +140,7 @@ export async function runFullAnalysis(
|
||||
|
||||
// ── Cache embeddings from existing index before rebuild ────────────
|
||||
let cachedEmbeddingNodeIds = new Set<string>();
|
||||
let cachedEmbeddings: Array<{ nodeId: string; embedding: number[] }> = [];
|
||||
let cachedEmbeddings: Array<{ nodeId: string; embedding: number[]; contentHash?: string }> = [];
|
||||
|
||||
if (options.embeddings && existingMeta && !options.force) {
|
||||
try {
|
||||
@@ -219,10 +221,14 @@ export async function runFullAnalysis(
|
||||
const EMBED_BATCH = 200;
|
||||
for (let i = 0; i < cachedEmbeddings.length; i += EMBED_BATCH) {
|
||||
const batch = cachedEmbeddings.slice(i, i + EMBED_BATCH);
|
||||
const paramsList = batch.map((e) => ({ nodeId: e.nodeId, embedding: e.embedding }));
|
||||
const paramsList = batch.map((e) => ({
|
||||
nodeId: e.nodeId,
|
||||
embedding: e.embedding,
|
||||
contentHash: e.contentHash ?? STALE_HASH_SENTINEL,
|
||||
}));
|
||||
try {
|
||||
await executeWithReusedStatement(
|
||||
`CREATE (e:CodeEmbedding {nodeId: $nodeId, embedding: $embedding})`,
|
||||
`MERGE (e:${EMBEDDING_TABLE_NAME} {nodeId: $nodeId}) SET e.embedding = $embedding, e.contentHash = $contentHash`,
|
||||
paramsList,
|
||||
);
|
||||
} catch {
|
||||
@@ -251,6 +257,14 @@ export async function runFullAnalysis(
|
||||
httpMode ? 'Connecting to embedding endpoint...' : 'Loading embedding model...',
|
||||
);
|
||||
const { runEmbeddingPipeline } = await import('./embeddings/embedding-pipeline.js');
|
||||
// Build a Map<nodeId, contentHash> from cached embeddings for incremental mode
|
||||
let existingEmbeddings: Map<string, string> | undefined;
|
||||
if (cachedEmbeddingNodeIds.size > 0) {
|
||||
existingEmbeddings = new Map<string, string>();
|
||||
for (const e of cachedEmbeddings) {
|
||||
existingEmbeddings.set(e.nodeId, e.contentHash ?? STALE_HASH_SENTINEL);
|
||||
}
|
||||
}
|
||||
await runEmbeddingPipeline(
|
||||
executeQuery,
|
||||
executeWithReusedStatement,
|
||||
@@ -265,7 +279,7 @@ export async function runFullAnalysis(
|
||||
progress('embeddings', scaled, label);
|
||||
},
|
||||
{},
|
||||
cachedEmbeddingNodeIds.size > 0 ? cachedEmbeddingNodeIds : undefined,
|
||||
existingEmbeddings,
|
||||
);
|
||||
}
|
||||
|
||||
@@ -275,7 +289,9 @@ export async function runFullAnalysis(
|
||||
// Count embeddings in the index (cached + newly generated)
|
||||
let embeddingCount = 0;
|
||||
try {
|
||||
const embResult = await executeQuery(`MATCH (e:CodeEmbedding) RETURN count(e) AS cnt`);
|
||||
const embResult = await executeQuery(
|
||||
`MATCH (e:${EMBEDDING_TABLE_NAME}) RETURN count(e) AS cnt`,
|
||||
);
|
||||
embeddingCount = embResult?.[0]?.cnt ?? 0;
|
||||
} catch {
|
||||
/* table may not exist if embeddings never ran */
|
||||
|
||||
@@ -42,6 +42,11 @@ export const initEmbedder = async (): Promise<FeatureExtractionPipeline> => {
|
||||
initPromise = (async () => {
|
||||
try {
|
||||
env.allowLocalModels = false;
|
||||
// Default cache to user-writable location. transformers.js defaults to
|
||||
// ./node_modules/.cache inside its own install dir, which is unwritable
|
||||
// when gitnexus is installed globally (e.g. /usr/lib/node_modules/).
|
||||
// Respect HF_HOME if set, otherwise fall back to ~/.cache/huggingface.
|
||||
env.cacheDir = process.env.HF_HOME ?? `${process.env.HOME}/.cache/huggingface`;
|
||||
|
||||
console.error('GitNexus: Loading embedding model (first search may take a moment)...');
|
||||
|
||||
|
||||
+34
-19
@@ -1449,25 +1449,40 @@ export const createServer = async (port: number, host: string = '127.0.0.1') =>
|
||||
await withLbugDb(lbugPath, async () => {
|
||||
const { runEmbeddingPipeline } =
|
||||
await import('../core/embeddings/embedding-pipeline.js');
|
||||
await runEmbeddingPipeline(executeQuery, executeWithReusedStatement, (p) => {
|
||||
embedJobManager.updateJob(job.id, {
|
||||
progress: {
|
||||
phase:
|
||||
p.phase === 'ready' ? 'complete' : p.phase === 'error' ? 'failed' : p.phase,
|
||||
percent: p.percent,
|
||||
message:
|
||||
p.phase === 'loading-model'
|
||||
? 'Loading embedding model...'
|
||||
: p.phase === 'embedding'
|
||||
? `Embedding nodes (${p.percent}%)...`
|
||||
: p.phase === 'indexing'
|
||||
? 'Creating vector index...'
|
||||
: p.phase === 'ready'
|
||||
? 'Embeddings complete'
|
||||
: `${p.phase} (${p.percent}%)`,
|
||||
},
|
||||
});
|
||||
});
|
||||
// Fetch existing content hashes for incremental embedding.
|
||||
// Delegated to lbug-adapter which owns the DB query logic and legacy-fallback handling.
|
||||
const { fetchExistingEmbeddingHashes } = await import('../core/lbug/lbug-adapter.js');
|
||||
const existingEmbeddings = await fetchExistingEmbeddingHashes(executeQuery);
|
||||
if (existingEmbeddings && existingEmbeddings.size > 0) {
|
||||
console.log(
|
||||
`[embed] ${existingEmbeddings.size} nodes already embedded — incremental run with content-hash comparison`,
|
||||
);
|
||||
}
|
||||
await runEmbeddingPipeline(
|
||||
executeQuery,
|
||||
executeWithReusedStatement,
|
||||
(p) => {
|
||||
embedJobManager.updateJob(job.id, {
|
||||
progress: {
|
||||
phase:
|
||||
p.phase === 'ready' ? 'complete' : p.phase === 'error' ? 'failed' : p.phase,
|
||||
percent: p.percent,
|
||||
message:
|
||||
p.phase === 'loading-model'
|
||||
? 'Loading embedding model...'
|
||||
: p.phase === 'embedding'
|
||||
? `Embedding nodes (${p.percent}%)...`
|
||||
: p.phase === 'indexing'
|
||||
? 'Creating vector index...'
|
||||
: p.phase === 'ready'
|
||||
? 'Embeddings complete'
|
||||
: `${p.phase} (${p.percent}%)`,
|
||||
},
|
||||
});
|
||||
},
|
||||
{}, // config: use defaults
|
||||
existingEmbeddings,
|
||||
);
|
||||
});
|
||||
|
||||
clearTimeout(embedTimeout);
|
||||
|
||||
@@ -0,0 +1,377 @@
|
||||
import { describe, it, expect, vi, beforeEach } from 'vitest';
|
||||
import { createHash } from 'crypto';
|
||||
import { contentHashForNode } from '../../src/core/embeddings/embedding-pipeline.js';
|
||||
import { generateEmbeddingText } from '../../src/core/embeddings/text-generator.js';
|
||||
import type { EmbeddableNode, EmbeddingProgress } from '../../src/core/embeddings/types.js';
|
||||
import { DEFAULT_EMBEDDING_CONFIG } from '../../src/core/embeddings/types.js';
|
||||
import { STALE_HASH_SENTINEL } from '../../src/core/lbug/schema.js';
|
||||
|
||||
// ────────────────────────────────────────────────────────────────────────────
|
||||
// contentHashForNode
|
||||
// ────────────────────────────────────────────────────────────────────────────
|
||||
describe('contentHashForNode', () => {
|
||||
const makeNode = (overrides: Partial<EmbeddableNode> = {}): EmbeddableNode => ({
|
||||
id: 'Function:foo:src/main.ts',
|
||||
name: 'foo',
|
||||
label: 'Function',
|
||||
filePath: 'src/main.ts',
|
||||
content: 'function foo() { return 1; }',
|
||||
...overrides,
|
||||
});
|
||||
|
||||
it('returns a 40-char hex SHA-1 digest', () => {
|
||||
const hash = contentHashForNode(makeNode());
|
||||
expect(hash).toMatch(/^[0-9a-f]{40}$/);
|
||||
});
|
||||
|
||||
it('is deterministic — same node always produces the same hash', () => {
|
||||
const node = makeNode();
|
||||
expect(contentHashForNode(node)).toBe(contentHashForNode(node));
|
||||
});
|
||||
|
||||
it('matches sha1(generateEmbeddingText(node))', () => {
|
||||
const node = makeNode();
|
||||
const expected = createHash('sha1').update(generateEmbeddingText(node)).digest('hex');
|
||||
expect(contentHashForNode(node)).toBe(expected);
|
||||
});
|
||||
|
||||
it('changes when node content is edited', () => {
|
||||
const original = makeNode({ content: 'function foo() { return 1; }' });
|
||||
const edited = makeNode({ content: 'function foo() { return 42; }' });
|
||||
expect(contentHashForNode(original)).not.toBe(contentHashForNode(edited));
|
||||
});
|
||||
|
||||
it('changes when filePath differs', () => {
|
||||
const a = makeNode({ filePath: 'src/a.ts' });
|
||||
const b = makeNode({ filePath: 'src/b.ts' });
|
||||
// Different filePaths lead to different embedding text ⇒ different hashes
|
||||
expect(contentHashForNode(a)).not.toBe(contentHashForNode(b));
|
||||
});
|
||||
|
||||
it('produces identical hash regardless of config vs finalConfig when config is empty', () => {
|
||||
const node = makeNode();
|
||||
const hashWithEmptyConfig = contentHashForNode(node, {});
|
||||
const hashWithFullDefaults = contentHashForNode(node, DEFAULT_EMBEDDING_CONFIG);
|
||||
expect(hashWithEmptyConfig).toBe(hashWithFullDefaults);
|
||||
});
|
||||
});
|
||||
|
||||
// ────────────────────────────────────────────────────────────────────────────
|
||||
// STALE_HASH_SENTINEL
|
||||
// ────────────────────────────────────────────────────────────────────────────
|
||||
describe('STALE_HASH_SENTINEL', () => {
|
||||
it('is the empty string', () => {
|
||||
expect(STALE_HASH_SENTINEL).toBe('');
|
||||
});
|
||||
|
||||
it('is falsy — enables consistent `hash || STALE_HASH_SENTINEL` patterns', () => {
|
||||
expect(!STALE_HASH_SENTINEL).toBe(true);
|
||||
});
|
||||
});
|
||||
|
||||
// ────────────────────────────────────────────────────────────────────────────
|
||||
// runEmbeddingPipeline — exports
|
||||
// ────────────────────────────────────────────────────────────────────────────
|
||||
describe('runEmbeddingPipeline incremental mode', () => {
|
||||
it('exports contentHashForNode as a named export', async () => {
|
||||
const mod = await import('../../src/core/embeddings/embedding-pipeline.js');
|
||||
expect(typeof mod.contentHashForNode).toBe('function');
|
||||
});
|
||||
|
||||
it('exports runEmbeddingPipeline as a named export', async () => {
|
||||
const mod = await import('../../src/core/embeddings/embedding-pipeline.js');
|
||||
expect(typeof mod.runEmbeddingPipeline).toBe('function');
|
||||
});
|
||||
});
|
||||
|
||||
// ────────────────────────────────────────────────────────────────────────────
|
||||
// EMBEDDING_SCHEMA includes contentHash column
|
||||
// ────────────────────────────────────────────────────────────────────────────
|
||||
describe('EMBEDDING_SCHEMA', () => {
|
||||
it('includes contentHash STRING column', async () => {
|
||||
const { EMBEDDING_SCHEMA } = await import('../../src/core/lbug/schema.js');
|
||||
expect(EMBEDDING_SCHEMA).toContain('contentHash STRING');
|
||||
});
|
||||
});
|
||||
|
||||
// ────────────────────────────────────────────────────────────────────────────
|
||||
// EMBEDDING_INDEX_NAME export
|
||||
// ────────────────────────────────────────────────────────────────────────────
|
||||
describe('EMBEDDING_INDEX_NAME', () => {
|
||||
it('is exported from schema.ts', async () => {
|
||||
const { EMBEDDING_INDEX_NAME } = await import('../../src/core/lbug/schema.js');
|
||||
expect(EMBEDDING_INDEX_NAME).toBe('code_embedding_idx');
|
||||
});
|
||||
});
|
||||
|
||||
// ────────────────────────────────────────────────────────────────────────────
|
||||
// runEmbeddingPipeline — incremental filter logic with mocked embedder
|
||||
//
|
||||
// Tests the three incremental-mode code paths:
|
||||
// 1. New node (not in existingEmbeddings) → embedded
|
||||
// 2. Unchanged node (hash matches) → skipped
|
||||
// 3. Stale node (hash mismatch) → DELETE old → re-embed
|
||||
// 4. Zero nodes after filter → createVectorIndex still called
|
||||
// ────────────────────────────────────────────────────────────────────────────
|
||||
describe('runEmbeddingPipeline incremental filter', () => {
|
||||
// Track mocked calls
|
||||
let queryCalls: string[];
|
||||
let stmtCalls: Array<{ cypher: string; params: Array<Record<string, any>> }>;
|
||||
let progressUpdates: EmbeddingProgress[];
|
||||
|
||||
// Helper node
|
||||
const makeNode = (overrides: Partial<EmbeddableNode> = {}): EmbeddableNode => ({
|
||||
id: 'Function:foo:src/main.ts',
|
||||
name: 'foo',
|
||||
label: 'Function',
|
||||
filePath: 'src/main.ts',
|
||||
content: 'function foo() { return 1; }',
|
||||
...overrides,
|
||||
});
|
||||
|
||||
beforeEach(() => {
|
||||
queryCalls = [];
|
||||
stmtCalls = [];
|
||||
progressUpdates = [];
|
||||
vi.restoreAllMocks();
|
||||
vi.resetModules();
|
||||
});
|
||||
|
||||
// Mock the embedder module so we never need a real model
|
||||
const mockEmbedderSetup = () => {
|
||||
vi.doMock('../../src/core/embeddings/embedder.js', () => ({
|
||||
initEmbedder: vi.fn().mockResolvedValue(undefined),
|
||||
embedBatch: vi
|
||||
.fn()
|
||||
.mockImplementation((texts: string[]) =>
|
||||
Promise.resolve(texts.map(() => new Float32Array(384))),
|
||||
),
|
||||
embedText: vi.fn().mockResolvedValue(new Float32Array(384)),
|
||||
embeddingToArray: vi.fn().mockImplementation((emb: Float32Array) => Array.from(emb)),
|
||||
isEmbedderReady: vi.fn().mockReturnValue(true),
|
||||
}));
|
||||
|
||||
// Mock loadVectorExtension (avoids needing the native lbug module)
|
||||
vi.doMock('../../src/core/lbug/lbug-adapter.js', () => ({
|
||||
loadVectorExtension: vi.fn().mockResolvedValue(undefined),
|
||||
}));
|
||||
};
|
||||
|
||||
const mockExecuteQuery = (nodes: EmbeddableNode[]) => {
|
||||
return vi.fn().mockImplementation(async (cypher: string) => {
|
||||
queryCalls.push(cypher);
|
||||
// Respond to node queries based on label
|
||||
for (const label of ['Function', 'Class', 'Method', 'Interface', 'File']) {
|
||||
if (cypher.includes(`MATCH (n:${label})`)) {
|
||||
return nodes
|
||||
.filter((n) => n.label === label)
|
||||
.map((n) => ({
|
||||
id: n.id,
|
||||
name: n.name,
|
||||
label: n.label,
|
||||
filePath: n.filePath,
|
||||
content: n.content,
|
||||
startLine: n.startLine,
|
||||
endLine: n.endLine,
|
||||
}));
|
||||
}
|
||||
}
|
||||
return [];
|
||||
});
|
||||
};
|
||||
|
||||
const mockExecuteWithReusedStatement = () => {
|
||||
return vi
|
||||
.fn()
|
||||
.mockImplementation(async (cypher: string, params: Array<Record<string, any>>) => {
|
||||
stmtCalls.push({ cypher, params });
|
||||
});
|
||||
};
|
||||
|
||||
const onProgress = (p: EmbeddingProgress) => {
|
||||
progressUpdates.push({ ...p });
|
||||
};
|
||||
|
||||
it('skips unchanged nodes when hash matches', async () => {
|
||||
mockEmbedderSetup();
|
||||
|
||||
const node = makeNode();
|
||||
const hash = contentHashForNode(node, DEFAULT_EMBEDDING_CONFIG);
|
||||
const existingEmbeddings = new Map<string, string>([[node.id, hash]]);
|
||||
|
||||
const executeQuery = mockExecuteQuery([node]);
|
||||
const executeWithReusedStatement = mockExecuteWithReusedStatement();
|
||||
|
||||
const { runEmbeddingPipeline } =
|
||||
await import('../../src/core/embeddings/embedding-pipeline.js');
|
||||
|
||||
await runEmbeddingPipeline(
|
||||
executeQuery,
|
||||
executeWithReusedStatement,
|
||||
onProgress,
|
||||
{},
|
||||
existingEmbeddings,
|
||||
);
|
||||
|
||||
// No MERGE calls — node was skipped because hash matched
|
||||
const mergeCalls = stmtCalls.filter((c) => c.cypher.includes('MERGE'));
|
||||
expect(mergeCalls).toHaveLength(0);
|
||||
|
||||
// Pipeline should reach 'ready' state
|
||||
const readyProgress = progressUpdates.find((p) => p.phase === 'ready');
|
||||
expect(readyProgress).toBeDefined();
|
||||
expect(readyProgress!.percent).toBe(100);
|
||||
});
|
||||
|
||||
it('embeds new nodes not in existingEmbeddings', async () => {
|
||||
mockEmbedderSetup();
|
||||
|
||||
const node = makeNode({
|
||||
id: 'Function:newFn:src/new.ts',
|
||||
name: 'newFn',
|
||||
filePath: 'src/new.ts',
|
||||
});
|
||||
const existingEmbeddings = new Map<string, string>(); // empty — no prior embeddings
|
||||
|
||||
const executeQuery = mockExecuteQuery([node]);
|
||||
const executeWithReusedStatement = mockExecuteWithReusedStatement();
|
||||
|
||||
const { runEmbeddingPipeline } =
|
||||
await import('../../src/core/embeddings/embedding-pipeline.js');
|
||||
|
||||
await runEmbeddingPipeline(
|
||||
executeQuery,
|
||||
executeWithReusedStatement,
|
||||
onProgress,
|
||||
{},
|
||||
existingEmbeddings,
|
||||
);
|
||||
|
||||
// Should have a MERGE call to insert the embedding
|
||||
const mergeCalls = stmtCalls.filter((c) => c.cypher.includes('MERGE'));
|
||||
expect(mergeCalls.length).toBeGreaterThanOrEqual(1);
|
||||
|
||||
// The inserted row should contain the node id and a contentHash
|
||||
const insertParams = mergeCalls[0].params;
|
||||
expect(insertParams.some((p: any) => p.nodeId === node.id)).toBe(true);
|
||||
expect(insertParams[0].contentHash).toMatch(/^[0-9a-f]{40}$/);
|
||||
});
|
||||
|
||||
it('deletes and re-embeds stale nodes (hash mismatch)', async () => {
|
||||
mockEmbedderSetup();
|
||||
|
||||
const node = makeNode({ content: 'function foo() { return 42; }' });
|
||||
const staleHash = 'aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa'; // wrong hash
|
||||
const existingEmbeddings = new Map<string, string>([[node.id, staleHash]]);
|
||||
|
||||
const executeQuery = mockExecuteQuery([node]);
|
||||
const executeWithReusedStatement = mockExecuteWithReusedStatement();
|
||||
|
||||
const { runEmbeddingPipeline } =
|
||||
await import('../../src/core/embeddings/embedding-pipeline.js');
|
||||
|
||||
await runEmbeddingPipeline(
|
||||
executeQuery,
|
||||
executeWithReusedStatement,
|
||||
onProgress,
|
||||
{},
|
||||
existingEmbeddings,
|
||||
);
|
||||
|
||||
// Should have a DELETE call for the stale node
|
||||
const deleteCalls = stmtCalls.filter((c) => c.cypher.includes('DELETE'));
|
||||
expect(deleteCalls.length).toBeGreaterThanOrEqual(1);
|
||||
expect(deleteCalls[0].params.some((p: any) => p.nodeId === node.id)).toBe(true);
|
||||
|
||||
// Should also have a MERGE call to re-insert with new hash
|
||||
const mergeCalls = stmtCalls.filter((c) => c.cypher.includes('MERGE'));
|
||||
expect(mergeCalls.length).toBeGreaterThanOrEqual(1);
|
||||
});
|
||||
|
||||
it('treats STALE_HASH_SENTINEL as stale — triggers re-embed', async () => {
|
||||
mockEmbedderSetup();
|
||||
|
||||
const node = makeNode();
|
||||
// Legacy row: nodeId present but contentHash is STALE_HASH_SENTINEL
|
||||
const existingEmbeddings = new Map<string, string>([[node.id, STALE_HASH_SENTINEL]]);
|
||||
|
||||
const executeQuery = mockExecuteQuery([node]);
|
||||
const executeWithReusedStatement = mockExecuteWithReusedStatement();
|
||||
|
||||
const { runEmbeddingPipeline } =
|
||||
await import('../../src/core/embeddings/embedding-pipeline.js');
|
||||
|
||||
await runEmbeddingPipeline(
|
||||
executeQuery,
|
||||
executeWithReusedStatement,
|
||||
onProgress,
|
||||
{},
|
||||
existingEmbeddings,
|
||||
);
|
||||
|
||||
// Should have a DELETE call (stale)
|
||||
const deleteCalls = stmtCalls.filter((c) => c.cypher.includes('DELETE'));
|
||||
expect(deleteCalls.length).toBeGreaterThanOrEqual(1);
|
||||
|
||||
// Should also have a MERGE (re-embed)
|
||||
const mergeCalls = stmtCalls.filter((c) => c.cypher.includes('MERGE'));
|
||||
expect(mergeCalls.length).toBeGreaterThanOrEqual(1);
|
||||
});
|
||||
|
||||
it('calls createVectorIndex even when zero nodes need embedding after filter', async () => {
|
||||
mockEmbedderSetup();
|
||||
|
||||
const node = makeNode();
|
||||
const hash = contentHashForNode(node, DEFAULT_EMBEDDING_CONFIG);
|
||||
// All existing hashes match — zero nodes to embed
|
||||
const existingEmbeddings = new Map<string, string>([[node.id, hash]]);
|
||||
|
||||
const executeQuery = mockExecuteQuery([node]);
|
||||
const executeWithReusedStatement = mockExecuteWithReusedStatement();
|
||||
|
||||
const { runEmbeddingPipeline } =
|
||||
await import('../../src/core/embeddings/embedding-pipeline.js');
|
||||
|
||||
await runEmbeddingPipeline(
|
||||
executeQuery,
|
||||
executeWithReusedStatement,
|
||||
onProgress,
|
||||
{},
|
||||
existingEmbeddings,
|
||||
);
|
||||
|
||||
// The CREATE_VECTOR_INDEX query should have been called via executeQuery
|
||||
const vectorIndexCalls = queryCalls.filter((c) => c.includes('CREATE_VECTOR_INDEX'));
|
||||
expect(vectorIndexCalls.length).toBeGreaterThanOrEqual(1);
|
||||
});
|
||||
|
||||
it('throws when DELETE for stale nodes fails with non-trivial error', async () => {
|
||||
mockEmbedderSetup();
|
||||
|
||||
const node = makeNode({ content: 'function foo() { return 42; }' });
|
||||
const staleHash = 'aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa';
|
||||
const existingEmbeddings = new Map<string, string>([[node.id, staleHash]]);
|
||||
|
||||
const executeQuery = mockExecuteQuery([node]);
|
||||
const executeWithReusedStatement = vi.fn().mockRejectedValue(new Error('Connection lost'));
|
||||
|
||||
const { runEmbeddingPipeline } =
|
||||
await import('../../src/core/embeddings/embedding-pipeline.js');
|
||||
|
||||
await expect(
|
||||
runEmbeddingPipeline(
|
||||
executeQuery,
|
||||
executeWithReusedStatement,
|
||||
onProgress,
|
||||
{},
|
||||
existingEmbeddings,
|
||||
),
|
||||
).rejects.toThrow('vector-index corruption');
|
||||
});
|
||||
});
|
||||
|
||||
// ────────────────────────────────────────────────────────────────────────────
|
||||
// fetchExistingEmbeddingHashes — tested in integration tests (requires native module)
|
||||
// The function is tested via lbug-core-adapter integration tests which have the
|
||||
// native @ladybugdb/core module available.
|
||||
// ────────────────────────────────────────────────────────────────────────────
|
||||
@@ -583,4 +583,34 @@ describe('ManifestExtractor', () => {
|
||||
expect(result.contracts).toHaveLength(0);
|
||||
expect(result.crossLinks).toHaveLength(0);
|
||||
});
|
||||
|
||||
it('memoizes repeated (repo, type, contract) resolutions so each tuple hits the DB once', async () => {
|
||||
const calls: Array<{ repo: string; cypher: string }> = [];
|
||||
const execFor = (repo: string) => async (cypher: string) => {
|
||||
calls.push({ repo, cypher });
|
||||
return [{ uid: `uid::${repo}`, name: 'handler', filePath: 'src/h.ts' }];
|
||||
};
|
||||
|
||||
const dbExecutors = new Map<string, (c: string) => Promise<Record<string, unknown>[]>>([
|
||||
['svc/a', execFor('svc/a')],
|
||||
['svc/b', execFor('svc/b')],
|
||||
]);
|
||||
|
||||
// Two links declare the same (repo, type, contract) triple on each side,
|
||||
// so naive sequential resolution would run 4 queries; memoization collapses
|
||||
// to 2 (one per distinct repo tuple).
|
||||
const link: GroupManifestLink = {
|
||||
from: 'svc/b',
|
||||
to: 'svc/a',
|
||||
type: 'http',
|
||||
contract: 'GET::/api/orders',
|
||||
role: 'consumer',
|
||||
};
|
||||
|
||||
await extractor.extractFromManifest([link, { ...link }], dbExecutors);
|
||||
|
||||
// One resolution per distinct (repo, type, contract) — not per (link × side).
|
||||
expect(calls).toHaveLength(2);
|
||||
expect(new Set(calls.map((c) => c.repo))).toEqual(new Set(['svc/a', 'svc/b']));
|
||||
});
|
||||
});
|
||||
|
||||
@@ -3,7 +3,12 @@ import * as fs from 'node:fs';
|
||||
import * as path from 'node:path';
|
||||
import * as os from 'node:os';
|
||||
import { syncGroup, stableRepoPoolId } from '../../../src/core/group/sync.js';
|
||||
import type { GroupConfig, StoredContract, RepoHandle } from '../../../src/core/group/types.js';
|
||||
import type {
|
||||
GroupConfig,
|
||||
StoredContract,
|
||||
RepoHandle,
|
||||
GroupManifestLink,
|
||||
} from '../../../src/core/group/types.js';
|
||||
import type { RegistryEntry } from '../../../src/storage/repo-manager.js';
|
||||
|
||||
describe('syncGroup', () => {
|
||||
@@ -202,6 +207,110 @@ describe('syncGroup', () => {
|
||||
}
|
||||
});
|
||||
|
||||
it('manifest links in config.links produce cross-links with matchType manifest', async () => {
|
||||
const links: GroupManifestLink[] = [
|
||||
{
|
||||
from: 'app/consumer',
|
||||
to: 'app/provider',
|
||||
type: 'http',
|
||||
contract: 'GET::/api/orders',
|
||||
role: 'consumer',
|
||||
},
|
||||
];
|
||||
|
||||
const config: GroupConfig = {
|
||||
version: 1,
|
||||
name: 'test',
|
||||
description: '',
|
||||
repos: { 'app/consumer': 'consumer-repo', 'app/provider': 'provider-repo' },
|
||||
links,
|
||||
packages: {},
|
||||
detect: {
|
||||
http: true,
|
||||
grpc: false,
|
||||
topics: false,
|
||||
shared_libs: false,
|
||||
embedding_fallback: false,
|
||||
},
|
||||
matching: { bm25_threshold: 0.7, embedding_threshold: 0.65, max_candidates_per_step: 3 },
|
||||
};
|
||||
|
||||
const result = await syncGroup(config, {
|
||||
extractorOverride: async () => [],
|
||||
skipWrite: true,
|
||||
});
|
||||
|
||||
// ManifestExtractor should inject 2 contracts (provider + consumer) and 1 cross-link
|
||||
expect(result.contracts).toHaveLength(2);
|
||||
const manifestLinks = result.crossLinks.filter((cl) => cl.matchType === 'manifest');
|
||||
expect(manifestLinks).toHaveLength(1);
|
||||
expect(manifestLinks[0].contractId).toBe('http::GET::/api/orders');
|
||||
expect(manifestLinks[0].from.repo).toBe('app/consumer');
|
||||
expect(manifestLinks[0].to.repo).toBe('app/provider');
|
||||
expect(manifestLinks[0].confidence).toBe(1.0);
|
||||
|
||||
// With no DB executors available, UIDs fall back to the deterministic
|
||||
// synthetic form `manifest::<repo>::<contractId>`.
|
||||
expect(manifestLinks[0].from.symbolUid).toBe('manifest::app/consumer::http::GET::/api/orders');
|
||||
expect(manifestLinks[0].to.symbolUid).toBe('manifest::app/provider::http::GET::/api/orders');
|
||||
|
||||
// Manifest contracts also participate in runExactMatch; we must not emit a
|
||||
// duplicate matchType:'exact' cross-link for the same endpoint pair.
|
||||
const exactForSameContract = result.crossLinks.filter(
|
||||
(cl) => cl.matchType === 'exact' && cl.contractId === 'http::GET::/api/orders',
|
||||
);
|
||||
expect(exactForSameContract).toHaveLength(0);
|
||||
expect(result.crossLinks).toHaveLength(1);
|
||||
});
|
||||
|
||||
it('manifest links referencing unknown repos still produce cross-links via synthetic UIDs', async () => {
|
||||
const links: GroupManifestLink[] = [
|
||||
{
|
||||
from: 'app/known',
|
||||
to: 'app/dangling', // not present in config.repos
|
||||
type: 'http',
|
||||
contract: 'POST::/api/missing',
|
||||
role: 'consumer',
|
||||
},
|
||||
];
|
||||
|
||||
const config: GroupConfig = {
|
||||
version: 1,
|
||||
name: 'test',
|
||||
description: '',
|
||||
repos: { 'app/known': 'known-repo' },
|
||||
links,
|
||||
packages: {},
|
||||
detect: {
|
||||
http: true,
|
||||
grpc: false,
|
||||
topics: false,
|
||||
shared_libs: false,
|
||||
embedding_fallback: false,
|
||||
},
|
||||
matching: { bm25_threshold: 0.7, embedding_threshold: 0.65, max_candidates_per_step: 3 },
|
||||
};
|
||||
|
||||
const warnings: string[] = [];
|
||||
const origWarn = console.warn;
|
||||
console.warn = (msg: string) => warnings.push(String(msg));
|
||||
try {
|
||||
const result = await syncGroup(config, {
|
||||
extractorOverride: async () => [],
|
||||
skipWrite: true,
|
||||
});
|
||||
|
||||
expect(result.crossLinks).toHaveLength(1);
|
||||
expect(result.crossLinks[0].matchType).toBe('manifest');
|
||||
expect(result.crossLinks[0].to.symbolUid).toBe(
|
||||
'manifest::app/dangling::http::POST::/api/missing',
|
||||
);
|
||||
expect(warnings.some((w) => w.includes('app/dangling'))).toBe(true);
|
||||
} finally {
|
||||
console.warn = origWarn;
|
||||
}
|
||||
});
|
||||
|
||||
it('writes registry to groupDir when skipWrite is false', async () => {
|
||||
const tmpDir = path.join(os.tmpdir(), `gitnexus-sync-write-${Date.now()}`);
|
||||
fs.mkdirSync(tmpDir, { recursive: true });
|
||||
|
||||
@@ -0,0 +1,283 @@
|
||||
import { describe, it, expect, beforeEach, afterEach } from 'vitest';
|
||||
import { EventEmitter } from 'events';
|
||||
import fs from 'fs';
|
||||
import path from 'path';
|
||||
import os from 'os';
|
||||
import { splitRelCsvByLabelPair } from '../../src/core/lbug/lbug-adapter.js';
|
||||
|
||||
/**
|
||||
* Regression tests for splitRelCsvByLabelPair (PR #818).
|
||||
*
|
||||
* These tests call the real exported function from lbug-adapter.ts with a
|
||||
* mock WriteStream factory, exercising the actual backpressure, error
|
||||
* handling, and drain-listener guard without touching LadybugDB.
|
||||
*/
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Mock WriteStream — controllable backpressure + error injection
|
||||
// ---------------------------------------------------------------------------
|
||||
class MockWriteStream extends EventEmitter {
|
||||
public chunks: string[] = [];
|
||||
public destroyed = false;
|
||||
public ended = false;
|
||||
public blocked = false;
|
||||
public maxDrainListenersSeen = 0;
|
||||
|
||||
write(chunk: string): boolean {
|
||||
this.chunks.push(chunk);
|
||||
this._trackDrainListeners();
|
||||
return !this.blocked;
|
||||
}
|
||||
|
||||
end(cb?: (err?: Error) => void): this {
|
||||
this.ended = true;
|
||||
if (cb) cb();
|
||||
return this;
|
||||
}
|
||||
|
||||
destroy(): this {
|
||||
this.destroyed = true;
|
||||
return this;
|
||||
}
|
||||
|
||||
unblock(): void {
|
||||
this.blocked = false;
|
||||
this.emit('drain');
|
||||
}
|
||||
|
||||
triggerError(err: Error): void {
|
||||
this.emit('error', err);
|
||||
}
|
||||
|
||||
private _trackDrainListeners(): void {
|
||||
const count = this.listenerCount('drain');
|
||||
if (count > this.maxDrainListenersSeen) {
|
||||
this.maxDrainListenersSeen = count;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Test helpers
|
||||
// ---------------------------------------------------------------------------
|
||||
const HEADER = '"from","to","type","confidence","reason","step"';
|
||||
|
||||
function csvLine(from: string, to: string, type = 'CALLS'): string {
|
||||
return `"${from}","${to}","${type}",1.0,"auto",0`;
|
||||
}
|
||||
|
||||
function getNodeLabel(id: string): string {
|
||||
return id.split(':')[0];
|
||||
}
|
||||
|
||||
/** Cast MockWriteStream factory to the real WriteStreamFactory type. */
|
||||
function mockFactory(streams: MockWriteStream[], opts?: { blocked?: boolean }) {
|
||||
return (() => {
|
||||
const ws = new MockWriteStream();
|
||||
if (opts?.blocked) ws.blocked = true;
|
||||
streams.push(ws);
|
||||
return ws;
|
||||
}) as unknown as (filePath: string) => import('fs').WriteStream;
|
||||
}
|
||||
|
||||
let tmpDir: string;
|
||||
|
||||
beforeEach(() => {
|
||||
tmpDir = fs.mkdtempSync(path.join(os.tmpdir(), 'rel-csv-test-'));
|
||||
});
|
||||
|
||||
afterEach(() => {
|
||||
// fs.rmSync's built-in retry loop handles Windows EBUSY/ENOTEMPTY/EPERM
|
||||
// when a just-closed fd hasn't been released yet (Node added this exactly
|
||||
// for cross-platform tmpdir cleanup — see Node.js fs docs). The production
|
||||
// function also waits for the input stream's 'close' event, so this is
|
||||
// defense-in-depth.
|
||||
fs.rmSync(tmpDir, { recursive: true, force: true, maxRetries: 5, retryDelay: 50 });
|
||||
});
|
||||
|
||||
function writeCsv(lines: string[]): string {
|
||||
const csvPath = path.join(tmpDir, 'relations.csv');
|
||||
fs.writeFileSync(csvPath, lines.join('\n') + '\n');
|
||||
return csvPath;
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Tests
|
||||
// ---------------------------------------------------------------------------
|
||||
describe('splitRelCsvByLabelPair', () => {
|
||||
const validTables = new Set(['Function', 'Class', 'File', 'Method']);
|
||||
|
||||
it('splits lines into per-pair files with correct row counts', async () => {
|
||||
const csvPath = writeCsv([
|
||||
HEADER,
|
||||
csvLine('Function:a', 'Class:b'),
|
||||
csvLine('Function:c', 'Class:d'),
|
||||
csvLine('File:e', 'Method:f'),
|
||||
]);
|
||||
|
||||
const streams: MockWriteStream[] = [];
|
||||
const result = await splitRelCsvByLabelPair(
|
||||
csvPath,
|
||||
tmpDir,
|
||||
validTables,
|
||||
getNodeLabel,
|
||||
mockFactory(streams),
|
||||
);
|
||||
|
||||
expect(result.totalValidRels).toBe(3);
|
||||
expect(result.relsByPairMeta.get('Function|Class')?.rows).toBe(2);
|
||||
expect(result.relsByPairMeta.get('File|Method')?.rows).toBe(1);
|
||||
});
|
||||
|
||||
it('captures the CSV header in relHeader', async () => {
|
||||
const csvPath = writeCsv([HEADER, csvLine('Function:a', 'Class:b')]);
|
||||
|
||||
const streams: MockWriteStream[] = [];
|
||||
const result = await splitRelCsvByLabelPair(
|
||||
csvPath,
|
||||
tmpDir,
|
||||
validTables,
|
||||
getNodeLabel,
|
||||
mockFactory(streams),
|
||||
);
|
||||
|
||||
expect(result.relHeader).toBe(HEADER);
|
||||
});
|
||||
|
||||
it('skips lines with unknown labels and counts them', async () => {
|
||||
const csvPath = writeCsv([
|
||||
HEADER,
|
||||
csvLine('Function:a', 'Class:b'),
|
||||
csvLine('Unknown:x', 'Class:y'),
|
||||
csvLine('Function:c', 'Bogus:d'),
|
||||
]);
|
||||
|
||||
const streams: MockWriteStream[] = [];
|
||||
const result = await splitRelCsvByLabelPair(
|
||||
csvPath,
|
||||
tmpDir,
|
||||
validTables,
|
||||
getNodeLabel,
|
||||
mockFactory(streams),
|
||||
);
|
||||
|
||||
expect(result.totalValidRels).toBe(1);
|
||||
expect(result.skippedRels).toBe(2);
|
||||
});
|
||||
|
||||
it('ignores blank lines without counting them as skipped', async () => {
|
||||
const csvPath = writeCsv([HEADER, '', csvLine('Function:a', 'Class:b'), '', '']);
|
||||
|
||||
const streams: MockWriteStream[] = [];
|
||||
const result = await splitRelCsvByLabelPair(
|
||||
csvPath,
|
||||
tmpDir,
|
||||
validTables,
|
||||
getNodeLabel,
|
||||
mockFactory(streams),
|
||||
);
|
||||
|
||||
expect(result.totalValidRels).toBe(1);
|
||||
expect(result.skippedRels).toBe(0);
|
||||
});
|
||||
|
||||
it('registers at most 1 drain listener per stream under heavy backpressure', async () => {
|
||||
const lines = [HEADER];
|
||||
for (let i = 0; i < 50; i++) {
|
||||
lines.push(csvLine(`Function:f${i}`, `Class:c${i}`));
|
||||
}
|
||||
const csvPath = writeCsv(lines);
|
||||
|
||||
const streams: MockWriteStream[] = [];
|
||||
const promise = splitRelCsvByLabelPair(
|
||||
csvPath,
|
||||
tmpDir,
|
||||
validTables,
|
||||
getNodeLabel,
|
||||
mockFactory(streams, { blocked: true }),
|
||||
);
|
||||
|
||||
// Give readline time to buffer and fire lines
|
||||
await new Promise((r) => setTimeout(r, 50));
|
||||
|
||||
// Unblock all streams so the Promise can resolve
|
||||
for (const ws of streams) ws.unblock();
|
||||
await promise;
|
||||
|
||||
// The guard should have kept drain listeners at 1
|
||||
for (const ws of streams) {
|
||||
expect(ws.maxDrainListenersSeen).toBeLessThanOrEqual(1);
|
||||
}
|
||||
});
|
||||
|
||||
it('rejects the Promise when a WriteStream emits an error', async () => {
|
||||
const csvPath = writeCsv([HEADER, csvLine('Function:a', 'Class:b')]);
|
||||
|
||||
const streams: MockWriteStream[] = [];
|
||||
const promise = splitRelCsvByLabelPair(
|
||||
csvPath,
|
||||
tmpDir,
|
||||
validTables,
|
||||
getNodeLabel,
|
||||
mockFactory(streams, { blocked: true }),
|
||||
);
|
||||
|
||||
// Wait for readline to process, then error while paused on drain
|
||||
await new Promise((r) => setTimeout(r, 50));
|
||||
expect(streams.length).toBeGreaterThan(0);
|
||||
streams[0].triggerError(new Error('disk full'));
|
||||
|
||||
await expect(promise).rejects.toThrow('disk full');
|
||||
});
|
||||
|
||||
it('destroys all streams when one errors (no lingering FDs)', async () => {
|
||||
const lines = [HEADER];
|
||||
for (let i = 0; i < 10; i++) {
|
||||
lines.push(csvLine(`Function:f${i}`, `Class:c${i}`));
|
||||
lines.push(csvLine(`File:e${i}`, `Method:m${i}`));
|
||||
}
|
||||
const csvPath = writeCsv(lines);
|
||||
|
||||
const streams: MockWriteStream[] = [];
|
||||
const promise = splitRelCsvByLabelPair(
|
||||
csvPath,
|
||||
tmpDir,
|
||||
validTables,
|
||||
getNodeLabel,
|
||||
mockFactory(streams, { blocked: true }),
|
||||
);
|
||||
|
||||
// The first pair stream is created immediately and blocks on its header
|
||||
// write. Unblock it once so the loop advances and creates the second
|
||||
// pair stream (also blocked). Now both streams exist — trigger the error.
|
||||
await new Promise((r) => setTimeout(r, 20));
|
||||
expect(streams.length).toBe(1);
|
||||
streams[0].unblock();
|
||||
await new Promise((r) => setTimeout(r, 20));
|
||||
expect(streams.length).toBeGreaterThanOrEqual(2);
|
||||
streams[0].triggerError(new Error('EMFILE'));
|
||||
|
||||
await expect(promise).rejects.toThrow('EMFILE');
|
||||
|
||||
for (const ws of streams) {
|
||||
expect(ws.destroyed).toBe(true);
|
||||
}
|
||||
});
|
||||
|
||||
it('handles empty CSV (header only) without errors', async () => {
|
||||
const csvPath = writeCsv([HEADER]);
|
||||
|
||||
const streams: MockWriteStream[] = [];
|
||||
const result = await splitRelCsvByLabelPair(
|
||||
csvPath,
|
||||
tmpDir,
|
||||
validTables,
|
||||
getNodeLabel,
|
||||
mockFactory(streams),
|
||||
);
|
||||
|
||||
expect(result.totalValidRels).toBe(0);
|
||||
expect(result.skippedRels).toBe(0);
|
||||
expect(result.relHeader).toBe(HEADER);
|
||||
});
|
||||
});
|
||||
+1
-7
@@ -5,14 +5,8 @@
|
||||
"repository": "https://github.com/coder3101/tree-sitter-proto",
|
||||
"license": "MIT",
|
||||
"main": "bindings/node",
|
||||
"scripts": {
|
||||
"install": "node-gyp-build"
|
||||
},
|
||||
"_vendoredBy": "gitnexus — build deps (node-addon-api, node-gyp-build) are hoisted into gitnexus/package.json optionalDependencies, and native compilation is performed by gitnexus/scripts/build-tree-sitter-proto.cjs at gitnexus postinstall. Do NOT re-add a dependencies block or an install script here — doing so reintroduces https://github.com/abhigyanpatwari/GitNexus/issues/836 (ENOTEMPTY on global upgrade).",
|
||||
"peerDependencies": {
|
||||
"tree-sitter": ">=0.21.0"
|
||||
},
|
||||
"dependencies": {
|
||||
"node-addon-api": "^8.0.0",
|
||||
"node-gyp-build": "^4.8.0"
|
||||
}
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user