Compare commits
21
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
68decd2e68 | ||
|
|
3fd4346bcb | ||
|
|
c2734cd25e | ||
|
|
8cb2f278cc | ||
|
|
f3df8ab7ba | ||
|
|
93cdfb27fa | ||
|
|
109a3c6946 | ||
|
|
1df79c2eab | ||
|
|
32c9ddaf32 | ||
|
|
385ee037bd | ||
|
|
28ddbe5d54 | ||
|
|
c100577e5e | ||
|
|
baf3f9e37d | ||
|
|
b340c5d87a | ||
|
|
9ad1984b17 | ||
|
|
1a597f3cc6 | ||
|
|
759c983dce | ||
|
|
988b905abe | ||
|
|
3fbee2d3d2 | ||
|
|
a162f66254 | ||
|
|
ad2a397137 |
@@ -0,0 +1,16 @@
|
||||
version: 2
|
||||
updates:
|
||||
# Keep third-party Actions SHA pins current. See CONTRIBUTING.md — when
|
||||
# reviewing these bumps, verify the SHA corresponds to the claimed tag by
|
||||
# running `gh api repos/<owner>/<action>/git/refs/tags/<tag>` before merge.
|
||||
- package-ecosystem: github-actions
|
||||
directory: /
|
||||
schedule:
|
||||
interval: weekly
|
||||
open-pull-requests-limit: 5
|
||||
commit-message:
|
||||
prefix: chore
|
||||
include: scope
|
||||
labels:
|
||||
- dependencies
|
||||
- ci
|
||||
@@ -0,0 +1,53 @@
|
||||
# release-drafter config — used only for PR autolabeling by
|
||||
# `.github/workflows/pr-labeler.yml` (the workflow passes `disable-releaser: true`,
|
||||
# so the draft-release side of release-drafter never runs).
|
||||
#
|
||||
# The labels applied here are the same ones `.github/release.yml` maps to
|
||||
# categorized release-notes sections.
|
||||
#
|
||||
# `sync-labels: true` removes managed autolabels that no longer match the PR —
|
||||
# critical for the breaking-change case: if a PR title drops the `!` or the body
|
||||
# drops `BREAKING CHANGE:`, the `breaking` label is pulled off automatically.
|
||||
|
||||
# Required by release-drafter; not used because releaser is disabled.
|
||||
name-template: 'unused'
|
||||
tag-template: 'unused'
|
||||
template: |
|
||||
$CHANGES
|
||||
|
||||
sync-labels: true
|
||||
|
||||
autolabeler:
|
||||
- label: enhancement
|
||||
title:
|
||||
- '/^feat(\([^)]+\))?!?:/i'
|
||||
- label: bug
|
||||
title:
|
||||
- '/^fix(\([^)]+\))?!?:/i'
|
||||
- label: performance
|
||||
title:
|
||||
- '/^perf(\([^)]+\))?!?:/i'
|
||||
- label: refactor
|
||||
title:
|
||||
- '/^refactor(\([^)]+\))?!?:/i'
|
||||
- label: documentation
|
||||
title:
|
||||
- '/^docs(\([^)]+\))?!?:/i'
|
||||
- label: test
|
||||
title:
|
||||
- '/^test(\([^)]+\))?!?:/i'
|
||||
- label: ci
|
||||
title:
|
||||
- '/^ci(\([^)]+\))?!?:/i'
|
||||
- label: dependencies
|
||||
title:
|
||||
- '/^(build|deps)(\([^)]+\))?!?:/i'
|
||||
- label: chore
|
||||
title:
|
||||
- '/^(chore|revert)(\([^)]+\))?!?:/i'
|
||||
# Breaking-change marker: either `!` in the type prefix or `BREAKING CHANGE:` in body.
|
||||
- label: breaking
|
||||
title:
|
||||
- '/^[a-z]+(\([^)]+\))?!:/i'
|
||||
body:
|
||||
- '/BREAKING[ -]CHANGE:/i'
|
||||
@@ -0,0 +1,173 @@
|
||||
#!/usr/bin/env python3
|
||||
"""Enforce the GitHub Actions concurrency convention.
|
||||
|
||||
See CONTRIBUTING.md -> "GitHub Actions — Concurrency Convention" for the rules.
|
||||
|
||||
Invoked from .github/workflows/ci-quality.yml. Runs locally too:
|
||||
python3 .github/scripts/check-workflow-concurrency.py .github/workflows
|
||||
|
||||
Rules:
|
||||
1. Every entry-point (non-reusable) workflow declares a top-level
|
||||
`concurrency:` block.
|
||||
2. Reusable workflows (on: workflow_call ONLY) do NOT declare one.
|
||||
3. The `concurrency.group` expression MUST reference either
|
||||
`${{ github.workflow }}` or a literal `CI-` prefix (the documented
|
||||
ci.yml reusable-workflow-safe exception). This is checked by substring
|
||||
containment rather than prefix match because ci.yml's group is a
|
||||
conditional expression that resolves to a `CI-…` literal at runtime.
|
||||
|
||||
We deliberately do not use a YAML library — keeps the script dependency-free
|
||||
on any vanilla runner. `on:` block parsing is line-based and handles both the
|
||||
flat (`on: workflow_call`) and mapping (`on:\n workflow_call:`) forms.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import pathlib
|
||||
import re
|
||||
import sys
|
||||
|
||||
|
||||
REQUIRED_TOKENS = ("${{ github.workflow }}", "CI-")
|
||||
|
||||
|
||||
def is_reusable(lines: list[str]) -> bool:
|
||||
"""Return True iff the workflow's `on:` block names only `workflow_call`."""
|
||||
in_on = False
|
||||
on_indent: int | None = None
|
||||
keys: list[str] = []
|
||||
|
||||
for raw in lines:
|
||||
# Skip blank lines and comments
|
||||
stripped = raw.strip()
|
||||
if not stripped or stripped.startswith("#"):
|
||||
continue
|
||||
|
||||
indent = len(raw) - len(raw.lstrip(" "))
|
||||
|
||||
if not in_on:
|
||||
if raw.startswith("on:"):
|
||||
remainder = raw[len("on:"):].strip()
|
||||
if not remainder:
|
||||
# `on:` followed by indented mapping on next lines
|
||||
in_on = True
|
||||
on_indent = indent
|
||||
continue
|
||||
if remainder.startswith("[") and remainder.endswith("]"):
|
||||
# Flow-style list: on: [workflow_call]
|
||||
items = [
|
||||
item.strip() for item in remainder.strip("[]").split(",")
|
||||
]
|
||||
return items == ["workflow_call"]
|
||||
# Scalar form: on: workflow_call (or a single other event)
|
||||
return remainder == "workflow_call"
|
||||
continue
|
||||
|
||||
# Inside the `on:` block; stop when indentation returns to <= on_indent
|
||||
if on_indent is not None and indent <= on_indent:
|
||||
break
|
||||
|
||||
# Only consider keys at on_indent + indentation step (anything deeper
|
||||
# is nested config like `types:`)
|
||||
if ":" not in stripped:
|
||||
continue
|
||||
# Heuristic: first-level event keys are those with indent == on_indent + 2
|
||||
# (the canonical step for a 2-space YAML doc). We collect all first-level
|
||||
# keys by tracking the smallest indent seen inside the block.
|
||||
keys.append((indent, stripped.split(":", 1)[0].strip()))
|
||||
|
||||
if not keys:
|
||||
return False
|
||||
|
||||
# Take only the outermost-indented keys as the event list
|
||||
min_indent = min(i for i, _ in keys)
|
||||
events = [name for i, name in keys if i == min_indent]
|
||||
return events == ["workflow_call"]
|
||||
|
||||
|
||||
CONCURRENCY_RE = re.compile(r"^concurrency:\s*$")
|
||||
GROUP_RE = re.compile(r"^\s+group:\s*(.+?)\s*$")
|
||||
|
||||
|
||||
def extract_group_key(lines: list[str]) -> str | None:
|
||||
"""Return the `group:` value of the top-level `concurrency:` block, or None."""
|
||||
for idx, raw in enumerate(lines):
|
||||
if CONCURRENCY_RE.match(raw):
|
||||
# Scan forward until we leave the concurrency block (next top-level key
|
||||
# is at column 0 and ends with `:`).
|
||||
for follow in lines[idx + 1:]:
|
||||
if follow and not follow.startswith(" ") and follow.rstrip().endswith(":"):
|
||||
break
|
||||
m = GROUP_RE.match(follow)
|
||||
if m:
|
||||
return m.group(1).strip().strip("'").strip('"')
|
||||
break
|
||||
return None
|
||||
|
||||
|
||||
def has_top_level_concurrency(lines: list[str]) -> bool:
|
||||
return any(CONCURRENCY_RE.match(raw) for raw in lines)
|
||||
|
||||
|
||||
def check(workflows_dir: pathlib.Path) -> int:
|
||||
fail = 0
|
||||
files = sorted(
|
||||
list(workflows_dir.glob("*.yml")) + list(workflows_dir.glob("*.yaml"))
|
||||
)
|
||||
for path in files:
|
||||
lines = path.read_text(encoding="utf-8").splitlines()
|
||||
reusable = is_reusable(lines)
|
||||
has_conc = has_top_level_concurrency(lines)
|
||||
|
||||
if reusable:
|
||||
if has_conc:
|
||||
print(
|
||||
f"::error file={path}::Reusable workflow (on: workflow_call) "
|
||||
"must NOT declare its own concurrency block — it inherits "
|
||||
"from the caller. See CONTRIBUTING.md -> GitHub Actions — "
|
||||
"Concurrency Convention."
|
||||
)
|
||||
fail = 1
|
||||
continue
|
||||
|
||||
if not has_conc:
|
||||
print(
|
||||
f"::error file={path}::Missing top-level concurrency block. "
|
||||
"See CONTRIBUTING.md -> GitHub Actions — Concurrency Convention."
|
||||
)
|
||||
fail = 1
|
||||
continue
|
||||
|
||||
group = extract_group_key(lines)
|
||||
if group is None:
|
||||
print(
|
||||
f"::error file={path}::concurrency block is missing a "
|
||||
"`group:` key."
|
||||
)
|
||||
fail = 1
|
||||
continue
|
||||
|
||||
if not any(token in group for token in REQUIRED_TOKENS):
|
||||
print(
|
||||
f"::error file={path}::concurrency.group `{group}` must "
|
||||
f"reference one of {REQUIRED_TOKENS}. See CONTRIBUTING.md -> "
|
||||
"GitHub Actions — Concurrency Convention."
|
||||
)
|
||||
fail = 1
|
||||
|
||||
return fail
|
||||
|
||||
|
||||
def main(argv: list[str]) -> int:
|
||||
if len(argv) != 2:
|
||||
print(f"usage: {argv[0]} <workflows-dir>", file=sys.stderr)
|
||||
return 2
|
||||
workflows_dir = pathlib.Path(argv[1])
|
||||
if not workflows_dir.is_dir():
|
||||
print(f"not a directory: {workflows_dir}", file=sys.stderr)
|
||||
return 2
|
||||
return check(workflows_dir)
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
sys.exit(main(sys.argv))
|
||||
@@ -11,8 +11,8 @@ jobs:
|
||||
outputs:
|
||||
web_changed: ${{ steps.filter.outputs.web }}
|
||||
steps:
|
||||
- uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4
|
||||
- uses: dorny/paths-filter@de90cc6fb38fc0963ad72b210f1f284cd68cea36 # v3
|
||||
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
- uses: dorny/paths-filter@fbd0ab8f3e69293af611ebaee6363fc25e6d187d # v3
|
||||
id: filter
|
||||
with:
|
||||
filters: |
|
||||
@@ -26,7 +26,7 @@ jobs:
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 20
|
||||
steps:
|
||||
- uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4
|
||||
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
|
||||
- uses: ./.github/actions/setup-gitnexus-web
|
||||
|
||||
@@ -74,7 +74,7 @@ jobs:
|
||||
|
||||
- name: Upload test results
|
||||
if: always()
|
||||
uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4
|
||||
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1
|
||||
with:
|
||||
name: e2e-results
|
||||
path: |
|
||||
|
||||
@@ -8,8 +8,8 @@ jobs:
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 5
|
||||
steps:
|
||||
- uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4
|
||||
- uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4
|
||||
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
- uses: actions/setup-node@53b83947a5a98c8d113130e565377fae1a50d02f # v6.3.0
|
||||
with:
|
||||
node-version: 20
|
||||
cache: npm
|
||||
@@ -21,8 +21,8 @@ jobs:
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 10
|
||||
steps:
|
||||
- uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4
|
||||
- uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4
|
||||
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
- uses: actions/setup-node@53b83947a5a98c8d113130e565377fae1a50d02f # v6.3.0
|
||||
with:
|
||||
node-version: 20
|
||||
cache: npm
|
||||
@@ -34,7 +34,7 @@ jobs:
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 10
|
||||
steps:
|
||||
- uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4
|
||||
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
- uses: ./.github/actions/setup-gitnexus
|
||||
- run: npx tsc --noEmit
|
||||
working-directory: gitnexus
|
||||
@@ -43,7 +43,30 @@ jobs:
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 10
|
||||
steps:
|
||||
- uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4
|
||||
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
- uses: ./.github/actions/setup-gitnexus-web
|
||||
- run: npx tsc -b --noEmit
|
||||
working-directory: gitnexus-web
|
||||
|
||||
# Enforces the convention documented in CONTRIBUTING.md → "GitHub Actions —
|
||||
# Concurrency Convention":
|
||||
# 1. Every entry-point (non-reusable) workflow declares a top-level
|
||||
# `concurrency:` block.
|
||||
# 2. Reusable workflows (`on: workflow_call` only) do NOT declare one —
|
||||
# they inherit concurrency from the caller.
|
||||
# 3. The concurrency group key starts with `${{ github.workflow }}` or
|
||||
# the literal `CI-` prefix (the documented ci.yml exception for
|
||||
# reusable-workflow-safe grouping).
|
||||
# Reusability is detected by parsing each workflow's `on:` block, not an
|
||||
# allowlist, so new reusable workflows never produce false positives.
|
||||
workflow-convention:
|
||||
name: Workflow concurrency convention
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 5
|
||||
steps:
|
||||
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
- name: Validate workflow concurrency convention
|
||||
shell: bash
|
||||
run: |
|
||||
set -euo pipefail
|
||||
python3 .github/scripts/check-workflow-concurrency.py .github/workflows
|
||||
|
||||
@@ -14,6 +14,16 @@ permissions:
|
||||
contents: read # needed for sparse checkout of vitest.config.ts
|
||||
pull-requests: write # needed to post sticky PR comment
|
||||
|
||||
# Concurrency convention: see CONTRIBUTING.md → "GitHub Actions — Concurrency Convention".
|
||||
# Serialize sticky-comment writes per PR so two rapid CI completions don't race.
|
||||
# Internal PRs surface in `pull_requests[0].number`. Fork PRs leave that array empty,
|
||||
# so we fall back to `<head-repo-full-name>/<head-branch>`, which is stable across
|
||||
# reruns and subsequent pushes for the same fork PR (unlike `workflow_run.id` which
|
||||
# is unique per run and therefore does not serialize anything).
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.event.workflow_run.pull_requests[0].number || format('{0}/{1}', github.event.workflow_run.head_repository.full_name, github.event.workflow_run.head_branch) }}
|
||||
cancel-in-progress: false
|
||||
|
||||
jobs:
|
||||
pr-report:
|
||||
name: PR Report
|
||||
@@ -113,7 +123,7 @@ jobs:
|
||||
|
||||
- name: Checkout (for vitest config)
|
||||
if: steps.meta.outputs.skip != 'true'
|
||||
uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4
|
||||
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
with:
|
||||
sparse-checkout: gitnexus/vitest.config.ts
|
||||
sparse-checkout-cone-mode: false
|
||||
|
||||
@@ -9,7 +9,7 @@ jobs:
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 25
|
||||
steps:
|
||||
- uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4
|
||||
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
- uses: ./.github/actions/setup-gitnexus
|
||||
with:
|
||||
build: 'true'
|
||||
@@ -43,7 +43,7 @@ jobs:
|
||||
|
||||
- name: Upload test reports
|
||||
if: always()
|
||||
uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4
|
||||
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1
|
||||
with:
|
||||
name: test-reports
|
||||
path: |
|
||||
@@ -63,7 +63,7 @@ jobs:
|
||||
runs-on: ${{ matrix.os }}
|
||||
timeout-minutes: 25
|
||||
steps:
|
||||
- uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4
|
||||
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
- uses: ./.github/actions/setup-gitnexus
|
||||
with:
|
||||
build: 'true'
|
||||
|
||||
@@ -9,9 +9,20 @@ on:
|
||||
paths-ignore: ['**.md', 'docs/**', 'LICENSE']
|
||||
workflow_call:
|
||||
|
||||
# Concurrency convention: see CONTRIBUTING.md → "GitHub Actions — Concurrency Convention".
|
||||
# Hardcoded `CI-` prefix (not `${{ github.workflow }}`) because this workflow is
|
||||
# invoked as a reusable workflow from publish.yml and release-candidate.yml. In
|
||||
# called-workflow context `github.workflow` evaluation is ambiguous across GitHub
|
||||
# Actions versions, and a prefix that could resolve to the caller's name would
|
||||
# share a concurrency group with the caller → deadlock. A literal prefix is
|
||||
# immune. Direct `push`/`pull_request` invocations use `CI-<ref>`; invocations
|
||||
# from a reusable-workflow caller fall into a per-run-unique group that never
|
||||
# serializes with the caller.
|
||||
# cancel-in-progress is event-aware: cancel superseded PR runs, queue every other
|
||||
# event (push to main, workflow_call from publish.yml, etc.).
|
||||
concurrency:
|
||||
group: ci-${{ github.ref }}
|
||||
cancel-in-progress: true
|
||||
group: ${{ (github.event_name == 'pull_request' || github.event_name == 'push') && format('CI-{0}', github.ref) || format('CI-nested-{0}', github.run_id) }}
|
||||
cancel-in-progress: ${{ github.event_name == 'pull_request' }}
|
||||
|
||||
# ── Reusable workflow orchestration ─────────────────────────────────
|
||||
# Each concern lives in its own workflow file for maintainability:
|
||||
@@ -74,7 +85,7 @@ jobs:
|
||||
cp pr-meta/e2e_result pr-meta/e2e-result
|
||||
|
||||
- name: Upload PR metadata
|
||||
uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4
|
||||
uses: actions/upload-artifact@043fb46d1a93c77aae656e7c1c64a875d1fc6a0a # v7.0.1
|
||||
with:
|
||||
name: pr-meta
|
||||
path: pr-meta/
|
||||
|
||||
@@ -16,9 +16,10 @@ on:
|
||||
issue_comment:
|
||||
types: [created]
|
||||
|
||||
# Concurrency convention: see CONTRIBUTING.md → "GitHub Actions — Concurrency Convention".
|
||||
# Serialize per-PR to avoid racing review comments.
|
||||
concurrency:
|
||||
group: claude-review-${{ github.event.issue.number || github.event.pull_request.number }}
|
||||
group: ${{ github.workflow }}-${{ github.event.issue.number || github.event.pull_request.number }}
|
||||
cancel-in-progress: false
|
||||
|
||||
jobs:
|
||||
@@ -76,7 +77,7 @@ jobs:
|
||||
core.setOutput('branch', pr.head.ref);
|
||||
|
||||
- name: Checkout PR head
|
||||
uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4
|
||||
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
with:
|
||||
repository: ${{ steps.pr.outputs.repo }}
|
||||
ref: ${{ steps.pr.outputs.sha }}
|
||||
|
||||
@@ -10,9 +10,10 @@ on:
|
||||
pull_request_review:
|
||||
types: [submitted]
|
||||
|
||||
# Concurrency convention: see CONTRIBUTING.md → "GitHub Actions — Concurrency Convention".
|
||||
# Serialize per-PR/issue to avoid racing comments.
|
||||
concurrency:
|
||||
group: claude-code-${{ github.event.issue.number || github.event.pull_request.number || github.event.issue.id }}
|
||||
group: ${{ github.workflow }}-${{ github.event.issue.number || github.event.pull_request.number || github.event.issue.id }}
|
||||
cancel-in-progress: false
|
||||
|
||||
jobs:
|
||||
@@ -90,7 +91,7 @@ jobs:
|
||||
core.setOutput('branch', pr.head.ref);
|
||||
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4
|
||||
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
with:
|
||||
repository: ${{ steps.pr.outputs.is_pr == 'true' && steps.pr.outputs.repo || github.repository }}
|
||||
ref: ${{ steps.pr.outputs.is_pr == 'true' && steps.pr.outputs.sha || '' }}
|
||||
|
||||
@@ -8,8 +8,9 @@ on:
|
||||
permissions:
|
||||
pull-requests: write
|
||||
|
||||
# Concurrency convention: see CONTRIBUTING.md → "GitHub Actions — Concurrency Convention".
|
||||
concurrency:
|
||||
group: pr-desc-${{ github.event.pull_request.number }}
|
||||
group: ${{ github.workflow }}-${{ github.event.pull_request.number }}
|
||||
cancel-in-progress: true
|
||||
|
||||
jobs:
|
||||
|
||||
@@ -0,0 +1,113 @@
|
||||
name: PR Conventional Labeler
|
||||
|
||||
# Two workflows in one file with different triggers, matched to the minimum
|
||||
# privilege each needs:
|
||||
#
|
||||
# validate-title (on: pull_request)
|
||||
# Fork-safe. Runs with the PR-head's read-only GITHUB_TOKEN. Uses
|
||||
# `amannn/action-semantic-pull-request` to fail the check when the PR
|
||||
# title doesn't follow the conventional-commit format. Because the
|
||||
# action only reads the event payload, no fork-controlled code runs.
|
||||
#
|
||||
# autolabel (on: pull_request_target)
|
||||
# Needs `pull-requests: write` to apply labels, so must be
|
||||
# pull_request_target. Uses `release-drafter/release-drafter` with
|
||||
# `disable-releaser: true` to only run the autolabeler against the
|
||||
# `.github/release-drafter.yml` config from the BASE ref (release-
|
||||
# drafter reads the config from the repository's default branch, NOT
|
||||
# the PR head — verify with `gh api repos/release-drafter/release-drafter/contents/...`
|
||||
# or a fork-test PR before merging if the repo is high-value).
|
||||
# `sync-labels: true` in the config removes managed autolabels that no
|
||||
# longer match (e.g. when `!` or `BREAKING CHANGE:` is dropped).
|
||||
#
|
||||
# Title format: <type>[(scope)][!]: <subject>
|
||||
# Allowed types: feat, fix, perf, refactor, docs, test, ci, build, chore, revert, deps
|
||||
# Trailing `!` on the type marks a breaking change.
|
||||
# See CONTRIBUTING.md → "Pull request titles".
|
||||
|
||||
on:
|
||||
pull_request:
|
||||
# Title-only changes fire `edited`. `opened` and `reopened` cover creation.
|
||||
# `synchronize` (push to the PR branch) is intentionally excluded — titles
|
||||
# don't change on push, so it only wastes CI minutes and broadens the
|
||||
# privileged-token exposure window on the autolabel job.
|
||||
types: [opened, edited, reopened]
|
||||
pull_request_target:
|
||||
types: [opened, edited, reopened]
|
||||
|
||||
# Concurrency convention: see CONTRIBUTING.md → "GitHub Actions — Concurrency Convention".
|
||||
# Include `github.event_name` so `pull_request` (validate-title) and
|
||||
# `pull_request_target` (autolabel) runs for the same PR do NOT share a slot
|
||||
# and therefore cannot cancel each other — a cancelled required-check would
|
||||
# permanently block merge until the next title edit.
|
||||
# Within each trigger the latest title edit still supersedes the prior run.
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.event_name }}-${{ github.event.pull_request.number }}
|
||||
cancel-in-progress: true
|
||||
|
||||
jobs:
|
||||
validate-title:
|
||||
# Fork-safe job — only runs on `pull_request` (not `pull_request_target`).
|
||||
# Token is read-only; writes a commit status that branch protection can
|
||||
# require before merge.
|
||||
name: Validate PR title
|
||||
if: github.event_name == 'pull_request'
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 5
|
||||
permissions:
|
||||
pull-requests: read
|
||||
steps:
|
||||
# Pinned to v5.5.3. Verify SHA via:
|
||||
# gh api repos/amannn/action-semantic-pull-request/git/refs/tags/v5.5.3
|
||||
- uses: amannn/action-semantic-pull-request@0723387faaf9b38adef4775cd42cfd5155ed6017 # v5.5.3
|
||||
env:
|
||||
GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||
with:
|
||||
types: |
|
||||
feat
|
||||
fix
|
||||
perf
|
||||
refactor
|
||||
docs
|
||||
test
|
||||
ci
|
||||
build
|
||||
chore
|
||||
revert
|
||||
deps
|
||||
requireScope: false
|
||||
# Subject must be non-empty. We DO allow capitalized proper nouns
|
||||
# (MCP, GitHub, API, etc.) — the old `^(?![A-Z]).+$` pattern
|
||||
# rejected legitimate titles like `fix: MCP tool schema`.
|
||||
subjectPattern: ^\S.{2,}$
|
||||
subjectPatternError: |
|
||||
The subject "{subject}" in PR title "{title}" is invalid.
|
||||
Subjects must be at least 3 characters and must not start with whitespace.
|
||||
wip: false
|
||||
|
||||
autolabel:
|
||||
# Privileged job — runs only on `pull_request_target` so it can write labels.
|
||||
# Never checks out fork code, never executes fork-controlled input; only
|
||||
# reads the PR metadata (title, body, labels) and calls the GitHub API.
|
||||
name: Apply conventional label
|
||||
if: github.event_name == 'pull_request_target'
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 5
|
||||
permissions:
|
||||
# `contents: read` is required — release-drafter's context.config() reads
|
||||
# `.github/release-drafter.yml` from the repo's default branch via the
|
||||
# repo-contents API. Without it the job silently 403s and no labels are
|
||||
# applied. Job-level permissions nullify all unlisted scopes, so an
|
||||
# explicit grant is necessary here.
|
||||
contents: read
|
||||
pull-requests: write
|
||||
steps:
|
||||
# Pinned to v6.0.0. Verify SHA via:
|
||||
# gh api repos/release-drafter/release-drafter/git/refs/tags/v6.0.0
|
||||
# Note: dependabot will likely propose a bump to v6.x on first run.
|
||||
- uses: release-drafter/release-drafter@3f0f87098bd6b5c5b9a36d49c41d998ea58f9348 # v6.0.0
|
||||
with:
|
||||
config-name: release-drafter.yml
|
||||
disable-releaser: true
|
||||
env:
|
||||
GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||
@@ -7,13 +7,22 @@ on:
|
||||
|
||||
# No workflow-level permissions — scoped per job below.
|
||||
|
||||
# Concurrency convention: see CONTRIBUTING.md → "GitHub Actions — Concurrency Convention".
|
||||
# Tag refs are unique per release, so distinct tags run in parallel. Re-pushes of the
|
||||
# same tag serialize. cancel-in-progress: false — never cancel a publish mid-flight.
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.ref }}
|
||||
cancel-in-progress: false
|
||||
|
||||
jobs:
|
||||
ci:
|
||||
uses: ./.github/workflows/ci.yml
|
||||
permissions:
|
||||
contents: read
|
||||
actions: read
|
||||
pull-requests: write
|
||||
# No pull-requests:write — `ci.yml`'s save-pr-meta job is gated on
|
||||
# `github.event_name == 'pull_request'`, so it never runs during a
|
||||
# tag-triggered publish. Least-privilege for release-critical paths.
|
||||
|
||||
publish:
|
||||
needs: ci
|
||||
@@ -23,8 +32,8 @@ jobs:
|
||||
contents: write
|
||||
id-token: write
|
||||
steps:
|
||||
- uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4
|
||||
- uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4
|
||||
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
- uses: actions/setup-node@53b83947a5a98c8d113130e565377fae1a50d02f # v6.3.0
|
||||
with:
|
||||
node-version: 20
|
||||
registry-url: https://registry.npmjs.org
|
||||
|
||||
@@ -0,0 +1,366 @@
|
||||
name: Release Candidate
|
||||
|
||||
on:
|
||||
# Publish a release-candidate build whenever a merge/commit lands on main.
|
||||
# Docs/README-only changes are filtered out so prose updates don't
|
||||
# cut a release.
|
||||
push:
|
||||
branches: [main]
|
||||
paths-ignore:
|
||||
- '**.md'
|
||||
- 'docs/**'
|
||||
- 'LICENSE'
|
||||
workflow_dispatch:
|
||||
inputs:
|
||||
bump:
|
||||
description: >-
|
||||
Cycle policy. 'auto' (default) continues the active rc cycle on
|
||||
this branch if there is one, otherwise bumps patch from latest.
|
||||
Choose 'patch' / 'minor' / 'major' to explicitly start or reset
|
||||
an rc cycle.
|
||||
required: false
|
||||
default: 'auto'
|
||||
type: choice
|
||||
options:
|
||||
- auto
|
||||
- patch
|
||||
- minor
|
||||
- major
|
||||
force:
|
||||
description: 'Publish even when HEAD already has an rc marker'
|
||||
required: false
|
||||
default: 'false'
|
||||
type: choice
|
||||
options:
|
||||
- 'false'
|
||||
- 'true'
|
||||
|
||||
# No workflow-level permissions — scoped per job below.
|
||||
permissions: {}
|
||||
|
||||
# Concurrency convention: see CONTRIBUTING.md → "GitHub Actions — Concurrency Convention".
|
||||
# Serialize all runs on the same ref (push + workflow_dispatch) to prevent two publishes
|
||||
# racing on the rc counter. cancel-in-progress: false — the earlier merge publishes first.
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.ref }}
|
||||
cancel-in-progress: false
|
||||
|
||||
jobs:
|
||||
# ── Skip when HEAD already has an rc marker (retry / duplicate dispatch) ──
|
||||
# The marker is a lightweight tag `rc/<HEAD_SHA>` pushed *before* `npm
|
||||
# publish`, so a failed publish leaves the marker in place and the guard
|
||||
# refuses to re-publish. Recovery path after a partial failure:
|
||||
# git push --delete origin rc/<HEAD_SHA> v<RC_VERSION>
|
||||
# then redispatch with force=true.
|
||||
guard:
|
||||
name: Check if release candidate should run
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 5
|
||||
permissions:
|
||||
contents: read
|
||||
outputs:
|
||||
should_run: ${{ steps.decide.outputs.should_run }}
|
||||
head_sha: ${{ steps.decide.outputs.head_sha }}
|
||||
steps:
|
||||
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
with:
|
||||
fetch-depth: 0
|
||||
fetch-tags: true
|
||||
|
||||
- name: Decide
|
||||
id: decide
|
||||
shell: bash
|
||||
env:
|
||||
FORCE: ${{ inputs.force }}
|
||||
BUMP_INPUT: ${{ inputs.bump }}
|
||||
EVENT_NAME: ${{ github.event_name }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
HEAD_SHA=$(git rev-parse HEAD)
|
||||
echo "head_sha=$HEAD_SHA" >> "$GITHUB_OUTPUT"
|
||||
|
||||
if [ "$FORCE" = "true" ]; then
|
||||
echo "Force flag set — running regardless of marker tag."
|
||||
echo "should_run=true" >> "$GITHUB_OUTPUT"
|
||||
exit 0
|
||||
fi
|
||||
|
||||
# An explicit cycle reset on dispatch (bump != auto) also bypasses
|
||||
# the dedup guard — the maintainer is deliberately asking for a
|
||||
# new rc from the same commit.
|
||||
if [ "$EVENT_NAME" = "workflow_dispatch" ] \
|
||||
&& [ -n "${BUMP_INPUT:-}" ] \
|
||||
&& [ "${BUMP_INPUT:-auto}" != "auto" ]; then
|
||||
echo "Explicit bump=$BUMP_INPUT — bypassing marker dedup."
|
||||
echo "should_run=true" >> "$GITHUB_OUTPUT"
|
||||
exit 0
|
||||
fi
|
||||
|
||||
# Dedup: is there already an rc/<HEAD_SHA> marker pointing at HEAD?
|
||||
MARKER="rc/${HEAD_SHA}"
|
||||
if git rev-parse "refs/tags/$MARKER" >/dev/null 2>&1; then
|
||||
echo "HEAD already has marker $MARKER — skipping."
|
||||
echo "should_run=false" >> "$GITHUB_OUTPUT"
|
||||
else
|
||||
echo "No marker on HEAD — proceeding."
|
||||
echo "should_run=true" >> "$GITHUB_OUTPUT"
|
||||
fi
|
||||
|
||||
# ── Reuse the stable CI workflow ─────────────────────────────────────
|
||||
ci:
|
||||
needs: guard
|
||||
if: needs.guard.outputs.should_run == 'true'
|
||||
uses: ./.github/workflows/ci.yml
|
||||
permissions:
|
||||
contents: read
|
||||
secrets: inherit
|
||||
|
||||
# ── Publish the rc build to npm + create GitHub prerelease ───────────
|
||||
publish:
|
||||
name: Publish release candidate to npm
|
||||
needs: [guard, ci]
|
||||
if: needs.guard.outputs.should_run == 'true'
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 20
|
||||
permissions:
|
||||
contents: write # push rc tag + marker
|
||||
id-token: write # npm provenance
|
||||
steps:
|
||||
- uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6.0.2
|
||||
with:
|
||||
fetch-depth: 0
|
||||
fetch-tags: true
|
||||
|
||||
- uses: actions/setup-node@53b83947a5a98c8d113130e565377fae1a50d02f # v6.3.0
|
||||
with:
|
||||
node-version: 20
|
||||
registry-url: https://registry.npmjs.org
|
||||
cache: npm
|
||||
cache-dependency-path: gitnexus/package-lock.json
|
||||
|
||||
- name: Build gitnexus-shared
|
||||
run: npm install && npm run build
|
||||
working-directory: gitnexus-shared
|
||||
|
||||
- name: Install gitnexus dependencies
|
||||
run: npm ci
|
||||
working-directory: gitnexus
|
||||
|
||||
- name: Resolve rc version
|
||||
id: version
|
||||
shell: bash
|
||||
working-directory: gitnexus
|
||||
env:
|
||||
BUMP_INPUT: ${{ inputs.bump }}
|
||||
EVENT_NAME: ${{ github.event_name }}
|
||||
PKG_NAME: gitnexus
|
||||
run: |
|
||||
set -euo pipefail
|
||||
|
||||
# 1. Current published `latest` — the floor for any new rc base.
|
||||
# Only E404 ("never published") falls back to package.json; any
|
||||
# other error (network, auth, malformed response) fails fast.
|
||||
NPM_STDERR_LATEST="$(mktemp)"
|
||||
if CURRENT_LATEST="$(npm view "$PKG_NAME" version 2>"$NPM_STDERR_LATEST")"; then
|
||||
:
|
||||
else
|
||||
if grep -q 'E404' "$NPM_STDERR_LATEST"; then
|
||||
CURRENT_LATEST="$(node -p "require('./package.json').version")"
|
||||
echo "Package not on registry (E404) — seeding from package.json: $CURRENT_LATEST"
|
||||
else
|
||||
echo "::error::npm registry unreachable for 'view version':" >&2
|
||||
cat "$NPM_STDERR_LATEST" >&2
|
||||
rm -f "$NPM_STDERR_LATEST"
|
||||
exit 1
|
||||
fi
|
||||
fi
|
||||
rm -f "$NPM_STDERR_LATEST"
|
||||
CURRENT_LATEST_CLEAN="${CURRENT_LATEST%%-*}"
|
||||
|
||||
# 2. Full version list — needed for the counter and for active-cycle
|
||||
# inference. Same E404-only fallback.
|
||||
NPM_STDERR_VERSIONS="$(mktemp)"
|
||||
if VERSIONS_JSON="$(npm view "$PKG_NAME" versions --json 2>"$NPM_STDERR_VERSIONS")"; then
|
||||
:
|
||||
else
|
||||
if grep -q 'E404' "$NPM_STDERR_VERSIONS"; then
|
||||
VERSIONS_JSON='[]'
|
||||
echo "No published versions for $PKG_NAME yet (E404)."
|
||||
else
|
||||
echo "::error::npm registry unreachable for 'view versions':" >&2
|
||||
cat "$NPM_STDERR_VERSIONS" >&2
|
||||
rm -f "$NPM_STDERR_VERSIONS"
|
||||
exit 1
|
||||
fi
|
||||
fi
|
||||
rm -f "$NPM_STDERR_VERSIONS"
|
||||
|
||||
# 3. Base selection.
|
||||
# - workflow_dispatch + bump ∈ {patch,minor,major} → explicit cycle
|
||||
# reset from latest.
|
||||
# - Everything else (push, or dispatch with bump=auto) → continue
|
||||
# the highest active rc base > latest if one exists; else
|
||||
# default to patch from latest.
|
||||
if [ "$EVENT_NAME" = "workflow_dispatch" ] \
|
||||
&& [ -n "${BUMP_INPUT:-}" ] \
|
||||
&& [ "${BUMP_INPUT:-auto}" != "auto" ]; then
|
||||
BASE="$(npx --yes -p semver@7 semver -i "$BUMP_INPUT" "$CURRENT_LATEST_CLEAN")"
|
||||
echo "Explicit bump=$BUMP_INPUT → BASE=$BASE"
|
||||
else
|
||||
cat > /tmp/active_base.mjs <<'NODESCRIPT'
|
||||
const latest = process.env.LATEST;
|
||||
let v;
|
||||
try { v = JSON.parse(process.env.VERSIONS_JSON); } catch { v = []; }
|
||||
if (!Array.isArray(v)) v = [v];
|
||||
const parse = s => s.split(".").map(n => parseInt(n, 10));
|
||||
const gt = (a, b) => {
|
||||
const [A, B] = [parse(a), parse(b)];
|
||||
for (let i = 0; i < 3; i++) if (A[i] !== B[i]) return A[i] > B[i];
|
||||
return false;
|
||||
};
|
||||
const bases = new Set();
|
||||
for (const s of v) {
|
||||
const m = /^(\d+\.\d+\.\d+)-rc\.\d+$/.exec(s);
|
||||
if (m && gt(m[1], latest)) bases.add(m[1]);
|
||||
}
|
||||
if (!bases.size) { process.stdout.write(""); process.exit(0); }
|
||||
const sorted = [...bases].sort((a, b) => gt(a, b) ? 1 : -1);
|
||||
process.stdout.write(sorted[sorted.length - 1]);
|
||||
NODESCRIPT
|
||||
ACTIVE_BASE="$(LATEST="$CURRENT_LATEST_CLEAN" VERSIONS_JSON="$VERSIONS_JSON" node /tmp/active_base.mjs)"
|
||||
if [ -n "$ACTIVE_BASE" ]; then
|
||||
BASE="$ACTIVE_BASE"
|
||||
echo "Continuing active rc cycle → BASE=$BASE"
|
||||
else
|
||||
BASE="$(npx --yes -p semver@7 semver -i patch "$CURRENT_LATEST_CLEAN")"
|
||||
echo "No active rc cycle → patch bump from latest → BASE=$BASE"
|
||||
fi
|
||||
fi
|
||||
|
||||
# 4. Counter: 1 + max existing N for `${BASE}-rc.*`, else 1.
|
||||
cat > /tmp/next_rc.mjs <<'NODESCRIPT'
|
||||
const base = process.env.BASE;
|
||||
const prefix = base + "-rc.";
|
||||
let v;
|
||||
try { v = JSON.parse(process.env.VERSIONS_JSON); } catch { v = []; }
|
||||
if (!Array.isArray(v)) v = [v];
|
||||
const ns = v
|
||||
.filter(s => typeof s === "string" && s.startsWith(prefix))
|
||||
.map(s => parseInt(s.slice(prefix.length), 10))
|
||||
.filter(n => Number.isInteger(n) && n >= 0);
|
||||
process.stdout.write(String(ns.length ? Math.max(...ns) + 1 : 1));
|
||||
NODESCRIPT
|
||||
NEXT_N="$(BASE="$BASE" VERSIONS_JSON="$VERSIONS_JSON" node /tmp/next_rc.mjs)"
|
||||
RC_VERSION="${BASE}-rc.${NEXT_N}"
|
||||
echo "Computed rc: $RC_VERSION"
|
||||
|
||||
# 5. Defensive: if the exact version already exists on the registry
|
||||
# (e.g., race with another run), abort before re-publishing.
|
||||
# Same E404-only pattern used above — a transient network
|
||||
# failure must fail loudly, not pretend the version is missing.
|
||||
NPM_STDERR_EXISTS="$(mktemp)"
|
||||
if npm view "$PKG_NAME@$RC_VERSION" version 2>"$NPM_STDERR_EXISTS" >/dev/null; then
|
||||
rm -f "$NPM_STDERR_EXISTS"
|
||||
echo "::error::Version $RC_VERSION already exists on npm — aborting."
|
||||
exit 1
|
||||
else
|
||||
if grep -qiE 'E404|not found' "$NPM_STDERR_EXISTS"; then
|
||||
rm -f "$NPM_STDERR_EXISTS"
|
||||
# Version doesn't exist — safe to proceed.
|
||||
else
|
||||
echo "::error::npm registry unreachable for existence check:" >&2
|
||||
cat "$NPM_STDERR_EXISTS" >&2
|
||||
rm -f "$NPM_STDERR_EXISTS"
|
||||
exit 1
|
||||
fi
|
||||
fi
|
||||
|
||||
echo "base=$BASE" >> "$GITHUB_OUTPUT"
|
||||
echo "rc_n=$NEXT_N" >> "$GITHUB_OUTPUT"
|
||||
echo "rc_version=$RC_VERSION" >> "$GITHUB_OUTPUT"
|
||||
|
||||
- name: Apply rc version in-CI
|
||||
shell: bash
|
||||
working-directory: gitnexus
|
||||
run: |
|
||||
set -euo pipefail
|
||||
npm version "${{ steps.version.outputs.rc_version }}" \
|
||||
--no-git-tag-version --allow-same-version
|
||||
|
||||
- name: Build gitnexus
|
||||
run: npm run build
|
||||
working-directory: gitnexus
|
||||
|
||||
- name: Dry-run publish
|
||||
run: npm publish --dry-run --tag rc
|
||||
working-directory: gitnexus
|
||||
|
||||
# ── Acquire the "rc lock" BEFORE publishing (fixes idempotency) ─────
|
||||
# We create two tags and push them atomically:
|
||||
# v<RC_VERSION> → annotated tag on a detached release commit
|
||||
# whose tree contains the rewritten package.json
|
||||
# (so the tag's source matches the npm tarball)
|
||||
# rc/<HEAD_SHA> → lightweight tag on HEAD; the guard's dedup key
|
||||
# If this push fails, nothing is published — safe.
|
||||
# If this push succeeds but npm publish fails, the marker stays on
|
||||
# the remote and blocks retries until an operator manually cleans up.
|
||||
- name: Create and push rc tags
|
||||
id: reltag
|
||||
shell: bash
|
||||
working-directory: gitnexus
|
||||
env:
|
||||
RC_VERSION: ${{ steps.version.outputs.rc_version }}
|
||||
HEAD_SHA: ${{ needs.guard.outputs.head_sha }}
|
||||
run: |
|
||||
set -euo pipefail
|
||||
VTAG="v${RC_VERSION}"
|
||||
MARKER="rc/${HEAD_SHA}"
|
||||
git config user.name 'github-actions[bot]'
|
||||
git config user.email '41898282+github-actions[bot]@users.noreply.github.com'
|
||||
|
||||
# Detached release commit with the version bump — keeps `main`
|
||||
# pristine but gives the v-tag a tree that matches the published
|
||||
# package contents exactly (fixes release-integrity gap).
|
||||
git add package.json package-lock.json 2>/dev/null || git add package.json
|
||||
git commit -m "release: ${VTAG}" --allow-empty
|
||||
RELEASE_SHA="$(git rev-parse HEAD)"
|
||||
echo "Detached release commit: $RELEASE_SHA"
|
||||
|
||||
# Annotated release tag on the release commit.
|
||||
git tag -a "$VTAG" "$RELEASE_SHA" -m "$VTAG"
|
||||
# Lightweight marker on the user-visible HEAD for the guard.
|
||||
git tag "$MARKER" "$HEAD_SHA"
|
||||
|
||||
# Atomic push of both refs. If either would clobber an existing
|
||||
# remote ref, the push fails and we stop before npm publish.
|
||||
git push --atomic origin "refs/tags/$VTAG" "refs/tags/$MARKER"
|
||||
|
||||
echo "vtag=$VTAG" >> "$GITHUB_OUTPUT"
|
||||
echo "marker=$MARKER" >> "$GITHUB_OUTPUT"
|
||||
echo "release_sha=$RELEASE_SHA" >> "$GITHUB_OUTPUT"
|
||||
|
||||
- name: Publish to npm (rc dist-tag)
|
||||
run: npm publish --provenance --access public --tag rc
|
||||
working-directory: gitnexus
|
||||
env:
|
||||
NODE_AUTH_TOKEN: ${{ secrets.NPM_TOKEN }}
|
||||
|
||||
- name: Create GitHub prerelease
|
||||
uses: softprops/action-gh-release@a06a81a03ee405af7f2048a818ed3f03bbf83c7b # v2
|
||||
with:
|
||||
tag_name: ${{ steps.reltag.outputs.vtag }}
|
||||
name: Release Candidate ${{ steps.reltag.outputs.vtag }}
|
||||
prerelease: true
|
||||
make_latest: 'false'
|
||||
generate_release_notes: true
|
||||
body: |
|
||||
Automated release candidate build from `main`.
|
||||
|
||||
**npm:** `npm install gitnexus@rc`
|
||||
**Version:** `${{ steps.version.outputs.rc_version }}`
|
||||
**Target base:** `${{ steps.version.outputs.base }}` (rc #${{ steps.version.outputs.rc_n }})
|
||||
**Source commit (main):** ${{ needs.guard.outputs.head_sha }}
|
||||
**Release commit (versioned tree):** ${{ steps.reltag.outputs.release_sha }}
|
||||
|
||||
Release candidates are pre-stable builds intended for early testing.
|
||||
Stable releases remain on the `latest` dist-tag.
|
||||
@@ -47,8 +47,10 @@ permissions:
|
||||
issues: write
|
||||
pull-requests: write
|
||||
|
||||
# Concurrency convention: see CONTRIBUTING.md → "GitHub Actions — Concurrency Convention".
|
||||
# Single global slot — newest manual dispatch supersedes any in-flight run.
|
||||
concurrency:
|
||||
group: triage-sweep
|
||||
group: ${{ github.workflow }}
|
||||
cancel-in-progress: true
|
||||
|
||||
jobs:
|
||||
@@ -74,7 +76,7 @@ jobs:
|
||||
run: pip install -r .github/scripts/triage/requirements.txt
|
||||
|
||||
- name: Cache FastEmbed model weights
|
||||
uses: actions/cache@668228422ae6a00e4ad889ee87cd7109ec5666a7 # v5
|
||||
uses: actions/cache@27d5ce7f107fe9357f9df03efb73ab90386fccae # v5
|
||||
with:
|
||||
path: ${{ github.workspace }}/.fastembed_cache
|
||||
key: fastembed-bge-small-en-v1.5
|
||||
|
||||
+107
-10
@@ -21,18 +21,41 @@ This project uses the [PolyForm Noncommercial License 1.0.0](https://polyformpro
|
||||
## Branch and pull requests
|
||||
|
||||
- Use short-lived branches off the default branch of the repo you are targeting.
|
||||
- Prefer **conventional commits** (short prefix + description), for example:
|
||||
|
||||
```text
|
||||
feat: add graph export option
|
||||
fix: correct MCP tool schema for query
|
||||
test: cover cluster merge edge case
|
||||
docs: clarify analyze flags
|
||||
```
|
||||
|
||||
- **PR title:** `[area] Short description` (e.g. `[cli] Fix index refresh race`).
|
||||
- **PR titles MUST follow the conventional-commit format** — `pr-labeler.yml` enforces this on every PR and auto-applies the matching label so release notes group the change correctly.
|
||||
- **PR description:** what changed, why, how to verify (commands), and any risk or rollback notes.
|
||||
|
||||
### Pull request titles
|
||||
|
||||
Format: `<type>[(scope)][!]: <subject>`
|
||||
|
||||
Allowed types and the release-notes section each one lands in (defined in `.github/release.yml`):
|
||||
|
||||
| Type | Label applied | Release-notes section |
|
||||
|------|---------------|-----------------------|
|
||||
| `feat` | `enhancement` | 🚀 Features |
|
||||
| `fix` | `bug` | 🐛 Bug Fixes |
|
||||
| `perf` | `performance` | 🏎️ Performance |
|
||||
| `refactor` | `refactor` | 🔄 Refactoring |
|
||||
| `test` | `test` | 🧪 Tests |
|
||||
| `ci` | `ci` | 👷 CI/CD |
|
||||
| `build` / `deps` | `dependencies` | 📦 Dependencies |
|
||||
| `docs` | `documentation` | (grouped under Other Changes unless a Docs section is added) |
|
||||
| `chore` / `revert` | `chore` | (excluded from release notes) |
|
||||
|
||||
Append `!` to the type (e.g. `feat(api)!: drop /v1 endpoint`) or include `BREAKING CHANGE:` in the PR body to flag a breaking change — the labeler then adds the `breaking` label and the 💥 Breaking Changes section is rendered first.
|
||||
|
||||
Examples:
|
||||
|
||||
```text
|
||||
feat(web): add smart chat scroll
|
||||
fix(extractors): resolve silent contract mis-resolution
|
||||
perf: avoid O(n²) traversal in heritage walker
|
||||
chore(deps): bump vitest to 3.0.0
|
||||
ci: standardize workflow concurrency
|
||||
```
|
||||
|
||||
Commits within a PR may use any style — only the **merged PR title** shows up in release notes, so that's the one the convention applies to.
|
||||
|
||||
## Before you open a PR
|
||||
|
||||
- [ ] Tests pass for the packages you touched (`gitnexus` and/or `gitnexus-web`).
|
||||
@@ -45,6 +68,80 @@ This project uses the [PolyForm Noncommercial License 1.0.0](https://polyformpro
|
||||
|
||||
Maintainers may request changes for correctness, tests, performance, or consistency with existing patterns. Keeping diffs focused makes review faster.
|
||||
|
||||
## GitHub Actions — Concurrency Convention
|
||||
|
||||
Every workflow under `.github/workflows/` MUST declare a top-level `concurrency:` block using this convention:
|
||||
|
||||
- **Group key** starts with `${{ github.workflow }}` so no two workflows can collide on the same group name. The discriminator that follows is chosen per event shape:
|
||||
- Branch/tag scope: `${{ github.workflow }}-${{ github.ref }}`
|
||||
- Per-PR scope (for `issue_comment`, `pull_request_review*`, `pull_request` meta events): `${{ github.workflow }}-${{ github.event.pull_request.number || github.event.issue.number }}`
|
||||
- `workflow_run` scope (e.g. `ci-report.yml`): `${{ github.workflow }}-${{ github.event.workflow_run.pull_requests[0].number || format('{0}/{1}', github.event.workflow_run.head_repository.full_name, github.event.workflow_run.head_branch) }}` — the fork fallback must be stable across reruns (never `workflow_run.id`, which is per-run-unique and defeats serialization).
|
||||
- Global single-slot (manual dispatch utilities): `${{ github.workflow }}`
|
||||
- **Reusable workflows invoked via `workflow_call`:** do NOT use `${{ github.workflow }}` in the group key — in called-workflow context its evaluation is ambiguous and can resolve to the caller's name, which would deadlock against the caller's own group. Use a hardcoded literal prefix and a `github.event_name`-aware expression that falls through to `github.run_id` for reusable invocations (see `ci.yml` for the canonical form).
|
||||
- **Merge queue (`merge_group`)**: when this event is added, use `${{ github.workflow }}-${{ github.event.merge_group.head_ref }}` with `cancel-in-progress: false` (every queue entry is a distinct ref; never cancel).
|
||||
- **`cancel-in-progress` policy:**
|
||||
|
||||
| Event | `cancel-in-progress` | Why |
|
||||
|-------|----------------------|-----|
|
||||
| `pull_request` CI run | `true` | New push supersedes old run |
|
||||
| `push` to `main` | `false` | Every main commit gets validated |
|
||||
| Tag push (`v*` publish) | `false` | Never cancel mid-publish |
|
||||
| `push` to `main` for release-candidate | `false` | Never cancel mid-RC publish |
|
||||
| `workflow_dispatch` (release/publish) | `false` | Manual runs are intentional |
|
||||
| `workflow_run` (sticky-comment reports) | `false` | Serialize, don't race |
|
||||
| Per-PR bot workflows (`@claude`, review) | `false` | Serialize comments per PR |
|
||||
| PR-meta re-checks (pr-description-check) | `true` | Cheap, latest wins |
|
||||
| Single-slot utilities (triage sweep) | `true` | Latest dispatch supersedes |
|
||||
|
||||
- For workflows that serve multiple events at once (e.g. `ci.yml` handles `pull_request`, `push`, and `workflow_call`), make `cancel-in-progress` event-aware:
|
||||
|
||||
```yaml
|
||||
concurrency:
|
||||
group: ${{ github.workflow }}-${{ github.ref }}
|
||||
cancel-in-progress: ${{ github.event_name == 'pull_request' }}
|
||||
```
|
||||
|
||||
- When adding a new workflow, copy the concurrency block from an existing workflow of the same event shape.
|
||||
|
||||
## AI-assisted contributions
|
||||
|
||||
If you use coding agents, follow project context files (e.g. `AGENTS.md`, `CLAUDE.md`) and avoid drive-by refactors unrelated to the issue. Prefer incremental, test-backed changes.
|
||||
|
||||
## Releases
|
||||
|
||||
Two publish workflows ship `gitnexus` to npm:
|
||||
|
||||
- **Stable** (`.github/workflows/publish.yml`) — triggered by pushing any `v*`
|
||||
tag. Publishes to the `latest` dist-tag with a changelog-backed GitHub
|
||||
release. Maintainers are expected to tag from `main` as a convention; the
|
||||
workflow itself does not enforce branch reachability.
|
||||
- **Release Candidate** (`.github/workflows/release-candidate.yml`) — runs on
|
||||
every push to `main` (typically a merged PR) plus manual dispatch. Docs-only
|
||||
changes are skipped via `paths-ignore`. Publishes to the `rc` dist-tag with
|
||||
version `X.Y.Z-rc.N` and a GitHub prerelease, where:
|
||||
- `X.Y.Z` is selected automatically. On push (and on dispatch with
|
||||
`bump: auto`, the default) the workflow **continues the active rc cycle**:
|
||||
if the registry already has `X.Y.Z-rc.*` versions with `X.Y.Z` > current
|
||||
`latest`, it reuses the highest such base; otherwise it patch-bumps
|
||||
from `latest`. Dispatching with `bump: patch|minor|major` **resets**
|
||||
the cycle from `latest`.
|
||||
- `N` is auto-incremented against existing `X.Y.Z-rc.*` entries on the
|
||||
registry. First rc for a given base is `rc.1`.
|
||||
|
||||
Idempotency: the workflow pushes an `rc/<HEAD_SHA>` marker tag and a
|
||||
`v<RC>` release tag **atomically, before** calling `npm publish`. The guard
|
||||
refuses to re-run once the marker exists, so a post-publish failure will
|
||||
not mint a duplicate rc for the same commit. The `v<RC>` tag points at a
|
||||
detached release commit whose `package.json` matches the npm tarball
|
||||
exactly (traceable releases). Recovery after a partial failure:
|
||||
|
||||
```bash
|
||||
git push --delete origin rc/<HEAD_SHA> v<RC>
|
||||
# then redispatch the workflow with force: true
|
||||
```
|
||||
|
||||
The rc workflow never moves `latest`. To verify after a change, inspect dist-tags:
|
||||
|
||||
```bash
|
||||
npm view gitnexus dist-tags
|
||||
```
|
||||
|
||||
@@ -8,8 +8,10 @@ import {
|
||||
Loader2,
|
||||
AlertTriangle,
|
||||
GitBranch,
|
||||
ArrowDown,
|
||||
} from '@/lib/lucide-icons';
|
||||
import { useAppState } from '../hooks/useAppState';
|
||||
import { useAutoScroll } from '../hooks/useAutoScroll';
|
||||
import { ToolCallCard } from './ToolCallCard';
|
||||
import { isProviderConfigured } from '../core/llm/settings-service';
|
||||
import { MarkdownRenderer } from './MarkdownRenderer';
|
||||
@@ -35,14 +37,11 @@ export const RightPanel = () => {
|
||||
const [chatInput, setChatInput] = useState('');
|
||||
const [activeTab, setActiveTab] = useState<'chat' | 'processes'>('chat');
|
||||
const textareaRef = useRef<HTMLTextAreaElement>(null);
|
||||
const messagesEndRef = useRef<HTMLDivElement>(null);
|
||||
|
||||
// Auto-scroll to bottom when messages update or while streaming
|
||||
useEffect(() => {
|
||||
if (messagesEndRef.current) {
|
||||
messagesEndRef.current.scrollIntoView({ behavior: 'smooth' });
|
||||
}
|
||||
}, [chatMessages, isChatLoading]);
|
||||
// Keep streamed replies pinned unless the user intentionally scrolls away from the bottom.
|
||||
const { scrollContainerRef, messagesContainerRef, isAtBottom, scrollToBottom } = useAutoScroll(
|
||||
chatMessages,
|
||||
isChatLoading,
|
||||
);
|
||||
|
||||
const resolveFilePathForUI = useCallback((_requestedPath: string): string | null => {
|
||||
return null;
|
||||
@@ -265,7 +264,7 @@ export const RightPanel = () => {
|
||||
|
||||
{/* Chat Content - only show when chat tab is active */}
|
||||
{activeTab === 'chat' && (
|
||||
<div className="flex flex-1 flex-col overflow-hidden">
|
||||
<div className="relative flex flex-1 flex-col overflow-hidden">
|
||||
{/* Status bar */}
|
||||
<div className="flex items-center gap-2.5 border-b border-border-subtle bg-elevated/50 px-4 py-3">
|
||||
<div className="ml-auto flex items-center gap-2">
|
||||
@@ -291,7 +290,7 @@ export const RightPanel = () => {
|
||||
)}
|
||||
|
||||
{/* Messages */}
|
||||
<div className="scrollbar-thin flex-1 overflow-y-auto p-4">
|
||||
<div ref={scrollContainerRef} className="scrollbar-thin flex-1 overflow-y-auto p-4">
|
||||
{chatMessages.length === 0 ? (
|
||||
<div className="flex h-full flex-col items-center justify-center px-4 text-center">
|
||||
<div className="mb-4 flex h-14 w-14 items-center justify-center rounded-xl bg-gradient-to-br from-accent to-node-interface text-2xl shadow-glow">
|
||||
@@ -315,7 +314,7 @@ export const RightPanel = () => {
|
||||
</div>
|
||||
</div>
|
||||
) : (
|
||||
<div className="flex flex-col gap-6">
|
||||
<div ref={messagesContainerRef} className="flex flex-col gap-6">
|
||||
{chatMessages.map((message) => (
|
||||
<div key={message.id} className="animate-fade-in">
|
||||
{/* User message - compact label style */}
|
||||
@@ -391,10 +390,22 @@ export const RightPanel = () => {
|
||||
))}
|
||||
</div>
|
||||
)}
|
||||
{/* Scroll anchor for auto-scroll */}
|
||||
<div ref={messagesEndRef} />
|
||||
</div>
|
||||
|
||||
{/* Scroll to bottom */}
|
||||
<button
|
||||
aria-label="Scroll to bottom"
|
||||
onClick={() => scrollToBottom()}
|
||||
className={`absolute bottom-20 left-1/2 z-10 -translate-x-1/2 rounded-full border border-border-subtle bg-elevated px-3 py-1.5 text-xs text-text-secondary shadow-lg transition-all duration-200 hover:border-accent hover:text-accent ${
|
||||
!isAtBottom && chatMessages.length > 0
|
||||
? 'translate-y-0 opacity-100'
|
||||
: 'pointer-events-none translate-y-2 opacity-0'
|
||||
}`}
|
||||
>
|
||||
<ArrowDown className="mr-1 inline h-3.5 w-3.5" />
|
||||
Scroll to bottom
|
||||
</button>
|
||||
|
||||
{/* Input */}
|
||||
<div className="border-t border-border-subtle bg-surface p-3">
|
||||
<div className="flex items-end gap-2 rounded-xl border border-border-subtle bg-elevated px-3 py-2 transition-all focus-within:border-accent focus-within:ring-2 focus-within:ring-accent/20">
|
||||
|
||||
@@ -0,0 +1,145 @@
|
||||
import { useCallback, useEffect, useLayoutEffect, useRef, useState } from 'react';
|
||||
|
||||
const DEFAULT_BOTTOM_THRESHOLD = 100;
|
||||
const USER_SCROLL_EPSILON = 5;
|
||||
|
||||
export interface UseAutoScrollResult {
|
||||
scrollContainerRef: React.RefObject<HTMLDivElement>;
|
||||
messagesContainerRef: React.RefObject<HTMLDivElement>;
|
||||
isAtBottom: boolean;
|
||||
scrollToBottom: (behavior?: ScrollBehavior) => void;
|
||||
}
|
||||
|
||||
function isNearBottom(element: HTMLElement, threshold: number): boolean {
|
||||
return element.scrollHeight - element.scrollTop - element.clientHeight <= threshold;
|
||||
}
|
||||
|
||||
export function useAutoScroll<T>(
|
||||
chatMessages: T[],
|
||||
isChatLoading: boolean,
|
||||
bottomThreshold = DEFAULT_BOTTOM_THRESHOLD,
|
||||
): UseAutoScrollResult {
|
||||
const scrollContainerRef = useRef<HTMLDivElement>(null);
|
||||
const messagesContainerRef = useRef<HTMLDivElement>(null);
|
||||
|
||||
const [isAtBottom, setIsAtBottom] = useState(true);
|
||||
|
||||
const shouldStickToBottomRef = useRef(true);
|
||||
const lastScrollTopRef = useRef(0);
|
||||
const scrollFrameIdRef = useRef<number | null>(null);
|
||||
|
||||
const syncScrollState = useCallback(() => {
|
||||
const element = scrollContainerRef.current;
|
||||
if (!element) return;
|
||||
|
||||
const currentScrollTop = element.scrollTop;
|
||||
const nearBottom = isNearBottom(element, bottomThreshold);
|
||||
|
||||
if (nearBottom) {
|
||||
shouldStickToBottomRef.current = true;
|
||||
} else if (currentScrollTop < lastScrollTopRef.current - USER_SCROLL_EPSILON) {
|
||||
shouldStickToBottomRef.current = false;
|
||||
}
|
||||
|
||||
lastScrollTopRef.current = currentScrollTop;
|
||||
setIsAtBottom(nearBottom);
|
||||
}, [bottomThreshold]);
|
||||
|
||||
const scrollToBottom = useCallback(
|
||||
(behavior: ScrollBehavior = 'smooth') => {
|
||||
const element = scrollContainerRef.current;
|
||||
if (!element) return;
|
||||
|
||||
shouldStickToBottomRef.current = true;
|
||||
|
||||
if (behavior === 'auto') {
|
||||
element.scrollTop = element.scrollHeight;
|
||||
lastScrollTopRef.current = element.scrollTop;
|
||||
setIsAtBottom(isNearBottom(element, bottomThreshold));
|
||||
return;
|
||||
}
|
||||
|
||||
element.scrollTo({
|
||||
top: element.scrollHeight,
|
||||
behavior,
|
||||
});
|
||||
},
|
||||
[bottomThreshold],
|
||||
);
|
||||
|
||||
useEffect(() => {
|
||||
const element = scrollContainerRef.current;
|
||||
if (!element) return;
|
||||
|
||||
lastScrollTopRef.current = element.scrollTop;
|
||||
|
||||
const handleScroll = () => {
|
||||
if (scrollFrameIdRef.current !== null) {
|
||||
cancelAnimationFrame(scrollFrameIdRef.current);
|
||||
}
|
||||
|
||||
scrollFrameIdRef.current = requestAnimationFrame(() => {
|
||||
scrollFrameIdRef.current = null;
|
||||
syncScrollState();
|
||||
});
|
||||
};
|
||||
|
||||
element.addEventListener('scroll', handleScroll, { passive: true });
|
||||
syncScrollState();
|
||||
|
||||
return () => {
|
||||
element.removeEventListener('scroll', handleScroll);
|
||||
|
||||
if (scrollFrameIdRef.current !== null) {
|
||||
cancelAnimationFrame(scrollFrameIdRef.current);
|
||||
scrollFrameIdRef.current = null;
|
||||
}
|
||||
};
|
||||
}, [syncScrollState]);
|
||||
|
||||
useEffect(() => {
|
||||
const content = messagesContainerRef.current;
|
||||
const scrollEl = scrollContainerRef.current;
|
||||
if (!content || !scrollEl || typeof ResizeObserver === 'undefined') return;
|
||||
|
||||
let resizeFrameId: number | null = null;
|
||||
|
||||
const observer = new ResizeObserver(() => {
|
||||
if (shouldStickToBottomRef.current) {
|
||||
if (resizeFrameId !== null) {
|
||||
cancelAnimationFrame(resizeFrameId);
|
||||
}
|
||||
|
||||
resizeFrameId = requestAnimationFrame(() => {
|
||||
resizeFrameId = null;
|
||||
scrollToBottom('auto');
|
||||
});
|
||||
} else {
|
||||
syncScrollState();
|
||||
}
|
||||
});
|
||||
|
||||
observer.observe(content);
|
||||
|
||||
return () => {
|
||||
observer.disconnect();
|
||||
|
||||
if (resizeFrameId !== null) {
|
||||
cancelAnimationFrame(resizeFrameId);
|
||||
resizeFrameId = null;
|
||||
}
|
||||
};
|
||||
}, [chatMessages.length, scrollToBottom, syncScrollState]);
|
||||
|
||||
useLayoutEffect(() => {
|
||||
if (!shouldStickToBottomRef.current) return;
|
||||
scrollToBottom('auto');
|
||||
}, [chatMessages.length, isChatLoading, scrollToBottom]);
|
||||
|
||||
return {
|
||||
scrollContainerRef,
|
||||
messagesContainerRef,
|
||||
isAtBottom,
|
||||
scrollToBottom,
|
||||
};
|
||||
}
|
||||
@@ -9,6 +9,7 @@
|
||||
export {
|
||||
AlertCircle,
|
||||
AlertTriangle,
|
||||
ArrowDown,
|
||||
ArrowRight,
|
||||
AtSign,
|
||||
Brain,
|
||||
|
||||
@@ -0,0 +1,287 @@
|
||||
import { act, fireEvent, render, screen } from '@testing-library/react';
|
||||
import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest';
|
||||
import { useAutoScroll } from '../../src/hooks/useAutoScroll';
|
||||
|
||||
interface HarnessProps {
|
||||
messages: unknown[];
|
||||
isChatLoading: boolean;
|
||||
}
|
||||
|
||||
function AutoScrollHarness({ messages, isChatLoading }: HarnessProps) {
|
||||
const { scrollContainerRef, messagesContainerRef, isAtBottom, scrollToBottom } = useAutoScroll(
|
||||
messages,
|
||||
isChatLoading,
|
||||
);
|
||||
|
||||
return (
|
||||
<>
|
||||
<div data-testid="is-at-bottom">{String(isAtBottom)}</div>
|
||||
<div data-testid="container" ref={scrollContainerRef}>
|
||||
{messages.length > 0 ? (
|
||||
<div data-testid="messages-container" ref={messagesContainerRef}>
|
||||
{messages.map((message, index) => (
|
||||
<div key={index}>{String(message)}</div>
|
||||
))}
|
||||
</div>
|
||||
) : null}
|
||||
</div>
|
||||
<button type="button" onClick={() => scrollToBottom()}>
|
||||
Scroll to bottom
|
||||
</button>
|
||||
</>
|
||||
);
|
||||
}
|
||||
|
||||
function setScrollMetrics(
|
||||
element: HTMLDivElement,
|
||||
metrics: { scrollTop?: number; scrollHeight?: number; clientHeight?: number },
|
||||
) {
|
||||
if (metrics.scrollTop !== undefined) {
|
||||
Object.defineProperty(element, 'scrollTop', {
|
||||
configurable: true,
|
||||
writable: true,
|
||||
value: metrics.scrollTop,
|
||||
});
|
||||
}
|
||||
|
||||
if (metrics.scrollHeight !== undefined) {
|
||||
Object.defineProperty(element, 'scrollHeight', {
|
||||
configurable: true,
|
||||
value: metrics.scrollHeight,
|
||||
});
|
||||
}
|
||||
|
||||
if (metrics.clientHeight !== undefined) {
|
||||
Object.defineProperty(element, 'clientHeight', {
|
||||
configurable: true,
|
||||
value: metrics.clientHeight,
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
async function flushAnimationFrame() {
|
||||
await act(async () => {
|
||||
vi.runAllTimers();
|
||||
});
|
||||
}
|
||||
|
||||
async function scrollContainer(element: HTMLDivElement, scrollTop: number) {
|
||||
setScrollMetrics(element, { scrollTop });
|
||||
fireEvent.scroll(element);
|
||||
await flushAnimationFrame();
|
||||
}
|
||||
|
||||
const resizeObserverInstances: ResizeObserverMock[] = [];
|
||||
|
||||
class ResizeObserverMock {
|
||||
callback: ResizeObserverCallback;
|
||||
observedElements: Element[] = [];
|
||||
observe = vi.fn((element: Element) => {
|
||||
this.observedElements.push(element);
|
||||
});
|
||||
unobserve = vi.fn();
|
||||
disconnect = vi.fn();
|
||||
|
||||
constructor(callback: ResizeObserverCallback) {
|
||||
this.callback = callback;
|
||||
resizeObserverInstances.push(this);
|
||||
}
|
||||
}
|
||||
|
||||
async function triggerResize(instance: ResizeObserverMock) {
|
||||
await act(async () => {
|
||||
instance.callback([], instance as unknown as ResizeObserver);
|
||||
});
|
||||
await flushAnimationFrame();
|
||||
}
|
||||
|
||||
describe('useAutoScroll', () => {
|
||||
beforeEach(() => {
|
||||
vi.useFakeTimers();
|
||||
resizeObserverInstances.length = 0;
|
||||
vi.stubGlobal(
|
||||
'requestAnimationFrame',
|
||||
vi.fn((callback: FrameRequestCallback) => {
|
||||
return window.setTimeout(() => callback(performance.now()), 0);
|
||||
}),
|
||||
);
|
||||
vi.stubGlobal(
|
||||
'cancelAnimationFrame',
|
||||
vi.fn((frameId: number) => {
|
||||
clearTimeout(frameId);
|
||||
}),
|
||||
);
|
||||
Object.defineProperty(HTMLElement.prototype, 'scrollTo', {
|
||||
configurable: true,
|
||||
value: function (options: ScrollToOptions) {
|
||||
if (options.top !== undefined) {
|
||||
Object.defineProperty(this, 'scrollTop', {
|
||||
configurable: true,
|
||||
writable: true,
|
||||
value: options.top,
|
||||
});
|
||||
}
|
||||
},
|
||||
});
|
||||
vi.stubGlobal('ResizeObserver', ResizeObserverMock);
|
||||
});
|
||||
|
||||
afterEach(() => {
|
||||
vi.useRealTimers();
|
||||
vi.unstubAllGlobals();
|
||||
});
|
||||
|
||||
it('starts with isAtBottom true and auto-scrolls the very first message', () => {
|
||||
const { rerender } = render(<AutoScrollHarness messages={[]} isChatLoading={false} />);
|
||||
|
||||
expect(screen.getByTestId('is-at-bottom')).toHaveTextContent('true');
|
||||
|
||||
const container = screen.getByTestId('container') as HTMLDivElement;
|
||||
setScrollMetrics(container, { scrollTop: 0, scrollHeight: 500, clientHeight: 200 });
|
||||
|
||||
rerender(<AutoScrollHarness messages={[{ id: 1 }]} isChatLoading={false} />);
|
||||
|
||||
expect(container.scrollTop).toBe(500);
|
||||
expect(screen.getByTestId('is-at-bottom')).toHaveTextContent('true');
|
||||
});
|
||||
|
||||
it('follows streaming updates while the view stays pinned to the bottom', () => {
|
||||
const { rerender } = render(<AutoScrollHarness messages={[{ id: 1 }]} isChatLoading={false} />);
|
||||
const container = screen.getByTestId('container') as HTMLDivElement;
|
||||
|
||||
setScrollMetrics(container, { scrollTop: 700, scrollHeight: 1000, clientHeight: 200 });
|
||||
|
||||
rerender(<AutoScrollHarness messages={[{ id: 1 }]} isChatLoading={true} />);
|
||||
|
||||
expect(container.scrollTop).toBe(1000);
|
||||
expect(screen.getByTestId('is-at-bottom')).toHaveTextContent('true');
|
||||
});
|
||||
|
||||
it('stops auto-scroll after the user scrolls up', async () => {
|
||||
const { rerender } = render(<AutoScrollHarness messages={[{ id: 1 }]} isChatLoading={false} />);
|
||||
const container = screen.getByTestId('container') as HTMLDivElement;
|
||||
|
||||
setScrollMetrics(container, { scrollTop: 700, scrollHeight: 1000, clientHeight: 200 });
|
||||
await scrollContainer(container, 700);
|
||||
await scrollContainer(container, 250);
|
||||
|
||||
expect(screen.getByTestId('is-at-bottom')).toHaveTextContent('false');
|
||||
|
||||
setScrollMetrics(container, { scrollTop: 250, scrollHeight: 1400, clientHeight: 200 });
|
||||
rerender(<AutoScrollHarness messages={[{ id: 1 }, { id: 2 }]} isChatLoading={true} />);
|
||||
|
||||
expect(container.scrollTop).toBe(250);
|
||||
});
|
||||
|
||||
it('re-enables auto-scroll once the user returns near the bottom', async () => {
|
||||
const { rerender } = render(<AutoScrollHarness messages={[{ id: 1 }]} isChatLoading={false} />);
|
||||
const container = screen.getByTestId('container') as HTMLDivElement;
|
||||
|
||||
setScrollMetrics(container, { scrollTop: 700, scrollHeight: 1000, clientHeight: 200 });
|
||||
await scrollContainer(container, 700);
|
||||
await scrollContainer(container, 250);
|
||||
|
||||
setScrollMetrics(container, { scrollTop: 1120, scrollHeight: 1400, clientHeight: 200 });
|
||||
await scrollContainer(container, 1120);
|
||||
|
||||
expect(screen.getByTestId('is-at-bottom')).toHaveTextContent('true');
|
||||
|
||||
setScrollMetrics(container, { scrollTop: 1120, scrollHeight: 1800, clientHeight: 200 });
|
||||
rerender(<AutoScrollHarness messages={[{ id: 1 }, { id: 2 }]} isChatLoading={true} />);
|
||||
|
||||
expect(container.scrollTop).toBe(1800);
|
||||
});
|
||||
|
||||
it('scrollToBottom re-engages auto-scroll and scrolls to the container bottom', async () => {
|
||||
const { rerender } = render(<AutoScrollHarness messages={[{ id: 1 }]} isChatLoading={false} />);
|
||||
const container = screen.getByTestId('container') as HTMLDivElement;
|
||||
const scrollTo = vi.spyOn(container, 'scrollTo');
|
||||
|
||||
setScrollMetrics(container, { scrollTop: 700, scrollHeight: 1000, clientHeight: 200 });
|
||||
await scrollContainer(container, 700);
|
||||
await scrollContainer(container, 250);
|
||||
|
||||
fireEvent.click(screen.getByRole('button', { name: 'Scroll to bottom' }));
|
||||
|
||||
expect(scrollTo).toHaveBeenCalledWith({ top: 1000, behavior: 'smooth' });
|
||||
expect(screen.getByTestId('is-at-bottom')).toHaveTextContent('false');
|
||||
|
||||
setScrollMetrics(container, { scrollTop: 250, scrollHeight: 1600, clientHeight: 200 });
|
||||
rerender(<AutoScrollHarness messages={[{ id: 1 }, { id: 2 }]} isChatLoading={true} />);
|
||||
|
||||
expect(container.scrollTop).toBe(1600);
|
||||
expect(screen.getByTestId('is-at-bottom')).toHaveTextContent('true');
|
||||
});
|
||||
|
||||
it('re-pins to the latest bottom when inner content grows asynchronously', async () => {
|
||||
render(<AutoScrollHarness messages={[{ id: 1 }]} isChatLoading={false} />);
|
||||
const container = screen.getByTestId('container') as HTMLDivElement;
|
||||
|
||||
setScrollMetrics(container, { scrollTop: 700, scrollHeight: 1000, clientHeight: 200 });
|
||||
await scrollContainer(container, 700);
|
||||
|
||||
setScrollMetrics(container, { scrollTop: 1000, scrollHeight: 1450, clientHeight: 200 });
|
||||
await triggerResize(resizeObserverInstances[0]);
|
||||
|
||||
expect(container.scrollTop).toBe(1450);
|
||||
expect(screen.getByTestId('is-at-bottom')).toHaveTextContent('true');
|
||||
});
|
||||
|
||||
it('does not auto-scroll on async growth after user intentionally scrolls away', async () => {
|
||||
render(<AutoScrollHarness messages={[{ id: 1 }]} isChatLoading={false} />);
|
||||
const container = screen.getByTestId('container') as HTMLDivElement;
|
||||
|
||||
setScrollMetrics(container, { scrollTop: 700, scrollHeight: 1000, clientHeight: 200 });
|
||||
await scrollContainer(container, 700);
|
||||
await scrollContainer(container, 250);
|
||||
|
||||
setScrollMetrics(container, { scrollTop: 250, scrollHeight: 1400, clientHeight: 200 });
|
||||
await triggerResize(resizeObserverInstances[0]);
|
||||
|
||||
expect(container.scrollTop).toBe(250);
|
||||
expect(screen.getByTestId('is-at-bottom')).toHaveTextContent('false');
|
||||
});
|
||||
|
||||
it('cancels the pending ResizeObserver rAF when the component unmounts', () => {
|
||||
const cancelRAF = vi.mocked(cancelAnimationFrame);
|
||||
|
||||
const { unmount } = render(<AutoScrollHarness messages={[{ id: 1 }]} isChatLoading={false} />);
|
||||
const container = screen.getByTestId('container') as HTMLDivElement;
|
||||
|
||||
setScrollMetrics(container, { scrollTop: 950, scrollHeight: 1000, clientHeight: 200 });
|
||||
|
||||
const callsBefore = cancelRAF.mock.calls.length;
|
||||
|
||||
act(() => {
|
||||
resizeObserverInstances[0].callback(
|
||||
[],
|
||||
resizeObserverInstances[0] as unknown as ResizeObserver,
|
||||
);
|
||||
});
|
||||
|
||||
unmount();
|
||||
|
||||
expect(cancelRAF.mock.calls.length).toBeGreaterThan(callsBefore);
|
||||
|
||||
expect(() => vi.runAllTimers()).not.toThrow();
|
||||
});
|
||||
|
||||
it('attaches the observer when the messages wrapper first appears and disconnects on unmount', () => {
|
||||
const { rerender, unmount } = render(<AutoScrollHarness messages={[]} isChatLoading={false} />);
|
||||
|
||||
expect(screen.queryByTestId('messages-container')).toBeNull();
|
||||
expect(resizeObserverInstances).toHaveLength(0);
|
||||
|
||||
rerender(<AutoScrollHarness messages={[{ id: 1 }]} isChatLoading={false} />);
|
||||
|
||||
const messagesContainer = screen.getByTestId('messages-container');
|
||||
const resizeObserver = resizeObserverInstances[0];
|
||||
|
||||
expect(resizeObserverInstances).toHaveLength(1);
|
||||
expect(resizeObserver.observe).toHaveBeenCalledWith(messagesContainer);
|
||||
|
||||
unmount();
|
||||
|
||||
expect(resizeObserver.disconnect).toHaveBeenCalledTimes(1);
|
||||
});
|
||||
});
|
||||
@@ -2,6 +2,22 @@
|
||||
|
||||
All notable changes to GitNexus will be documented in this file.
|
||||
|
||||
## [1.6.1] - 2026-04-13
|
||||
|
||||
### Added
|
||||
- **Service group extractor expansion** — manifest extractor and broader extractor coverage (2/4 of #606 split) (#796)
|
||||
- **Dart call patterns** for `await`, cascade, lambda, and widget-tree contexts (#801)
|
||||
|
||||
### Fixed
|
||||
- **Stack overflow and memory exhaustion** on large repository analysis (#814)
|
||||
- **`tree-sitter-dart` install crash** — switched from git URL to npm tarball (#811)
|
||||
- **Generic TypeScript awaited function calls** missing from the call graph (#804)
|
||||
- **Runtime dependency on `file:../gitnexus-shared`** removed from the published package (#803)
|
||||
- **Ruby `singleton_class` context** preserved during sequential parsing (#774)
|
||||
|
||||
### Changed
|
||||
- **DAG-based ingestion pipeline architecture** — pipeline phases now declare typed dependencies and run via a topologically sorted DAG; container-node logic extracted to `LanguageProvider`. Includes hardened lifecycle (try/finally cleanup, error wrapping, cycle reporting), tightened `ParseOutput.exportedTypeMap` immutability, and corrected phase dependencies (#809)
|
||||
|
||||
## [1.6.0] - 2026-04-12
|
||||
|
||||
### Added
|
||||
|
||||
@@ -234,6 +234,79 @@ Installed automatically by both `gitnexus analyze` (per-repo) and `gitnexus setu
|
||||
- Node.js >= 18
|
||||
- Git repository (uses git for commit tracking)
|
||||
|
||||
## Release candidates
|
||||
|
||||
Stable releases publish to the default `latest` dist-tag. When a pull request
|
||||
with non-documentation changes merges into `main`, an automated workflow also
|
||||
publishes a prerelease build under the `rc` dist-tag, so early adopters can
|
||||
try in-flight fixes without waiting for the next stable cut. (Docs-only
|
||||
merges are skipped.)
|
||||
|
||||
```bash
|
||||
# Try the latest release candidate (pre-stable — may change at any time)
|
||||
npm install -g gitnexus@rc
|
||||
# — or —
|
||||
npx gitnexus@rc analyze
|
||||
```
|
||||
|
||||
Release-candidate versions follow the standard semver prerelease format
|
||||
`X.Y.Z-rc.N`, where `X.Y.Z` is the next stable target (bumped from the
|
||||
current `latest` by patch by default; `minor` or `major` when kicking off a
|
||||
bigger cycle) and `N` increments per published rc. Example sequence:
|
||||
`1.6.2-rc.1`, `1.6.2-rc.2`, …, then once `1.6.2` ships stable,
|
||||
`1.6.3-rc.1`. See the [Releases page](https://github.com/abhigyanpatwari/GitNexus/releases)
|
||||
for the full list; stable `latest` is unaffected.
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
### `Cannot destructure property 'package' of 'node.target' as it is null`
|
||||
|
||||
This crash was caused by a dependency URL format that is incompatible with
|
||||
certain npm/arborist versions ([npm/cli#8126](https://github.com/npm/cli/issues/8126)).
|
||||
It is fixed in **gitnexus v1.6.2+**. Upgrade to the latest version:
|
||||
|
||||
```bash
|
||||
npx gitnexus@latest analyze # always uses the newest release
|
||||
# — or —
|
||||
npm install -g gitnexus@latest # upgrade a global install
|
||||
```
|
||||
|
||||
If you still hit npm install issues after upgrading, these generic workarounds
|
||||
may help:
|
||||
|
||||
```bash
|
||||
npm install -g npm@latest # update npm itself
|
||||
npm cache clean --force # clear a possibly corrupt cache
|
||||
```
|
||||
|
||||
### Installation fails with native module errors
|
||||
|
||||
Some optional language grammars (Dart, Kotlin, Swift) require native compilation. If they fail, GitNexus still works — those languages will be skipped.
|
||||
|
||||
If `npm install -g gitnexus` fails on native modules:
|
||||
|
||||
```bash
|
||||
# Ensure build tools are available (Linux/macOS)
|
||||
# Ubuntu/Debian: sudo apt install python3 make g++
|
||||
# macOS: xcode-select --install
|
||||
|
||||
# Retry installation
|
||||
npm install -g gitnexus
|
||||
```
|
||||
|
||||
### Analysis runs out of memory
|
||||
|
||||
For very large repositories:
|
||||
|
||||
```bash
|
||||
# Increase Node.js heap size
|
||||
NODE_OPTIONS="--max-old-space-size=16384" npx gitnexus analyze
|
||||
|
||||
# Exclude large directories
|
||||
echo "vendor/" >> .gitnexusignore
|
||||
echo "dist/" >> .gitnexusignore
|
||||
```
|
||||
|
||||
## Privacy
|
||||
|
||||
- All processing happens locally on your machine
|
||||
|
||||
Generated
+6
-6
@@ -1,12 +1,12 @@
|
||||
{
|
||||
"name": "gitnexus",
|
||||
"version": "1.6.0",
|
||||
"version": "1.6.2-rc.7",
|
||||
"lockfileVersion": 3,
|
||||
"requires": true,
|
||||
"packages": {
|
||||
"": {
|
||||
"name": "gitnexus",
|
||||
"version": "1.6.0",
|
||||
"version": "1.6.2-rc.7",
|
||||
"hasInstallScript": true,
|
||||
"license": "PolyForm-Noncommercial-1.0.0",
|
||||
"dependencies": {
|
||||
@@ -30,7 +30,7 @@
|
||||
"pandemonium": "^2.4.0",
|
||||
"tree-sitter": "^0.21.1",
|
||||
"tree-sitter-c": "0.23.2",
|
||||
"tree-sitter-c-sharp": "^0.23.1",
|
||||
"tree-sitter-c-sharp": "0.23.1",
|
||||
"tree-sitter-cpp": "^0.23.4",
|
||||
"tree-sitter-go": "^0.23.0",
|
||||
"tree-sitter-java": "^0.23.5",
|
||||
@@ -62,7 +62,7 @@
|
||||
"node": ">=20.0.0"
|
||||
},
|
||||
"optionalDependencies": {
|
||||
"tree-sitter-dart": "https://github.com/UserNobody14/tree-sitter-dart/archive/80e23c07b64494f7e21090bb3450223ef0b192f4.tar.gz",
|
||||
"tree-sitter-dart": "git+https://github.com/UserNobody14/tree-sitter-dart.git#80e23c07b64494f7e21090bb3450223ef0b192f4",
|
||||
"tree-sitter-kotlin": "^0.3.8",
|
||||
"tree-sitter-proto": "file:./vendor/tree-sitter-proto",
|
||||
"tree-sitter-swift": "^0.6.0"
|
||||
@@ -5132,8 +5132,8 @@
|
||||
},
|
||||
"node_modules/tree-sitter-dart": {
|
||||
"version": "1.0.0",
|
||||
"resolved": "https://github.com/UserNobody14/tree-sitter-dart/archive/80e23c07b64494f7e21090bb3450223ef0b192f4.tar.gz",
|
||||
"integrity": "sha512-aqLZTEji2vAZPdbaCSjR0SXJGzFRKD//7VtrSV3st9bgrCM2tsXxXAHZlMlQLOCt7K2yKxM5K3gNXYph8TCjCQ==",
|
||||
"resolved": "git+https://github.com/UserNobody14/tree-sitter-dart.git#80e23c07b64494f7e21090bb3450223ef0b192f4",
|
||||
"integrity": "sha512-Bs/1wAOIJ2akPEXlE/XVpuES19Oo3NqoSJRJ/0N2r38qAd9nTXdqmaGHQ44/JXnA6QHcbgD2YzCCc4wUc98cyQ==",
|
||||
"hasInstallScript": true,
|
||||
"license": "ISC",
|
||||
"optional": true,
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "gitnexus",
|
||||
"version": "1.6.0",
|
||||
"version": "1.6.2-rc.7",
|
||||
"description": "Graph-powered code intelligence for AI agents. Index any codebase, query via MCP or CLI.",
|
||||
"author": "Abhigyan Patwari",
|
||||
"license": "PolyForm-Noncommercial-1.0.0",
|
||||
@@ -71,7 +71,7 @@
|
||||
"pandemonium": "^2.4.0",
|
||||
"tree-sitter": "^0.21.1",
|
||||
"tree-sitter-c": "0.23.2",
|
||||
"tree-sitter-c-sharp": "^0.23.1",
|
||||
"tree-sitter-c-sharp": "0.23.1",
|
||||
"tree-sitter-cpp": "^0.23.4",
|
||||
"tree-sitter-go": "^0.23.0",
|
||||
"tree-sitter-java": "^0.23.5",
|
||||
@@ -84,7 +84,7 @@
|
||||
"uuid": "^13.0.0"
|
||||
},
|
||||
"optionalDependencies": {
|
||||
"tree-sitter-dart": "https://github.com/UserNobody14/tree-sitter-dart/archive/80e23c07b64494f7e21090bb3450223ef0b192f4.tar.gz",
|
||||
"tree-sitter-dart": "git+https://github.com/UserNobody14/tree-sitter-dart.git#80e23c07b64494f7e21090bb3450223ef0b192f4",
|
||||
"tree-sitter-kotlin": "^0.3.8",
|
||||
"tree-sitter-proto": "file:./vendor/tree-sitter-proto",
|
||||
"tree-sitter-swift": "^0.6.0"
|
||||
|
||||
@@ -297,7 +297,7 @@ export const analyzeCommand = async (inputPath?: string, options?: AnalyzeOption
|
||||
const msg = err.message || String(err);
|
||||
console.error(`\n Analysis failed: ${msg}\n`);
|
||||
|
||||
// Provide helpful guidance for known large-repo failure modes
|
||||
// Provide helpful guidance for known failure modes
|
||||
if (
|
||||
msg.includes('Maximum call stack size exceeded') ||
|
||||
msg.includes('call stack') ||
|
||||
@@ -314,6 +314,28 @@ export const analyzeCommand = async (inputPath?: string, options?: AnalyzeOption
|
||||
console.error(' 2. Increase Node.js heap: NODE_OPTIONS="--max-old-space-size=16384"');
|
||||
console.error(' 3. Increase stack size: NODE_OPTIONS="--stack-size=4096"');
|
||||
console.error('');
|
||||
} else if (msg.includes('ERESOLVE') || msg.includes('Could not resolve dependency')) {
|
||||
// Note: the original arborist "Cannot destructure property 'package' of
|
||||
// 'node.target'" crash happens inside npm *before* gitnexus code runs,
|
||||
// so it can't be caught here. This branch handles dependency-resolution
|
||||
// errors that surface at runtime (e.g. dynamic require failures).
|
||||
console.error(' This looks like an npm dependency resolution issue.');
|
||||
console.error(' Suggestions:');
|
||||
console.error(' 1. Clear the npm cache: npm cache clean --force');
|
||||
console.error(' 2. Update npm: npm install -g npm@latest');
|
||||
console.error(' 3. Reinstall gitnexus: npm install -g gitnexus@latest');
|
||||
console.error(' 4. Or try npx directly: npx gitnexus@latest analyze');
|
||||
console.error('');
|
||||
} else if (
|
||||
msg.includes('MODULE_NOT_FOUND') ||
|
||||
msg.includes('Cannot find module') ||
|
||||
msg.includes('ERR_MODULE_NOT_FOUND')
|
||||
) {
|
||||
console.error(' A required module could not be loaded. The installation may be corrupt.');
|
||||
console.error(' Suggestions:');
|
||||
console.error(' 1. Reinstall: npm install -g gitnexus@latest');
|
||||
console.error(' 2. Clear cache: npm cache clean --force && npx gitnexus@latest analyze');
|
||||
console.error('');
|
||||
}
|
||||
|
||||
process.exitCode = 1;
|
||||
|
||||
@@ -9,6 +9,7 @@
|
||||
* 5. Create vector index for semantic search
|
||||
*/
|
||||
|
||||
import { createHash } from 'crypto';
|
||||
import {
|
||||
initEmbedder,
|
||||
embedBatch,
|
||||
@@ -16,7 +17,7 @@ import {
|
||||
embeddingToArray,
|
||||
isEmbedderReady,
|
||||
} from './embedder.js';
|
||||
import { generateBatchEmbeddingTexts } from './text-generator.js';
|
||||
import { generateEmbeddingText, generateBatchEmbeddingTexts } from './text-generator.js';
|
||||
import {
|
||||
type EmbeddingProgress,
|
||||
type EmbeddingConfig,
|
||||
@@ -26,9 +27,29 @@ import {
|
||||
DEFAULT_EMBEDDING_CONFIG,
|
||||
EMBEDDABLE_LABELS,
|
||||
} from './types.js';
|
||||
import {
|
||||
EMBEDDING_TABLE_NAME,
|
||||
EMBEDDING_INDEX_NAME,
|
||||
CREATE_VECTOR_INDEX_QUERY,
|
||||
} from '../lbug/schema.js';
|
||||
import { loadVectorExtension } from '../lbug/lbug-adapter.js';
|
||||
|
||||
const isDev = process.env.NODE_ENV === 'development';
|
||||
|
||||
/**
|
||||
* Compute a stable content fingerprint for an embeddable node.
|
||||
* Used to detect when the underlying text has changed so stale vectors
|
||||
* can be replaced (DELETE-then-INSERT, the Kuzu-sanctioned pattern for
|
||||
* vector-indexed rows).
|
||||
*/
|
||||
export const contentHashForNode = (
|
||||
node: EmbeddableNode,
|
||||
config: Partial<EmbeddingConfig> = {},
|
||||
): string => {
|
||||
const text = generateEmbeddingText(node, config);
|
||||
return createHash('sha1').update(text).digest('hex');
|
||||
};
|
||||
|
||||
/**
|
||||
* Progress callback type
|
||||
*/
|
||||
@@ -98,41 +119,32 @@ const batchInsertEmbeddings = async (
|
||||
cypher: string,
|
||||
paramsList: Array<Record<string, any>>,
|
||||
) => Promise<void>,
|
||||
updates: Array<{ id: string; embedding: number[] }>,
|
||||
updates: Array<{ id: string; embedding: number[]; contentHash: string }>,
|
||||
): Promise<void> => {
|
||||
// INSERT into separate embedding table - much more memory efficient!
|
||||
const cypher = `CREATE (e:CodeEmbedding {nodeId: $nodeId, embedding: $embedding})`;
|
||||
const paramsList = updates.map((u) => ({ nodeId: u.id, embedding: u.embedding }));
|
||||
// MERGE instead of CREATE — idempotent, handles concurrent analyzes and partial prior runs
|
||||
const cypher = `MERGE (e:${EMBEDDING_TABLE_NAME} {nodeId: $nodeId}) SET e.embedding = $embedding, e.contentHash = $contentHash`;
|
||||
const paramsList = updates.map((u) => ({
|
||||
nodeId: u.id,
|
||||
embedding: u.embedding,
|
||||
contentHash: u.contentHash,
|
||||
}));
|
||||
await executeWithReusedStatement(cypher, paramsList);
|
||||
};
|
||||
|
||||
/**
|
||||
* Create the vector index for semantic search
|
||||
* Now indexes the separate CodeEmbedding table
|
||||
* Now indexes the separate CodeEmbedding table.
|
||||
* Delegates extension loading to lbug-adapter's loadVectorExtension(),
|
||||
* which owns the VECTOR extension lifecycle and state tracking.
|
||||
*/
|
||||
let vectorExtensionLoaded = false;
|
||||
|
||||
const createVectorIndex = async (
|
||||
executeQuery: (cypher: string) => Promise<any[]>,
|
||||
): Promise<void> => {
|
||||
// LadybugDB v0.15+ requires explicit VECTOR extension loading (once per session)
|
||||
if (!vectorExtensionLoaded) {
|
||||
try {
|
||||
await executeQuery('INSTALL VECTOR');
|
||||
await executeQuery('LOAD EXTENSION VECTOR');
|
||||
vectorExtensionLoaded = true;
|
||||
} catch {
|
||||
// Extension may already be loaded — CREATE_VECTOR_INDEX will fail clearly if not
|
||||
vectorExtensionLoaded = true;
|
||||
}
|
||||
}
|
||||
|
||||
const cypher = `
|
||||
CALL CREATE_VECTOR_INDEX('CodeEmbedding', 'code_embedding_idx', 'embedding', metric := 'cosine')
|
||||
`;
|
||||
// Delegate to the adapter which tracks loaded state and handles DB reconnect resets
|
||||
await loadVectorExtension();
|
||||
|
||||
try {
|
||||
await executeQuery(cypher);
|
||||
await executeQuery(CREATE_VECTOR_INDEX_QUERY);
|
||||
} catch (error) {
|
||||
// Index might already exist
|
||||
if (isDev) {
|
||||
@@ -148,7 +160,9 @@ const createVectorIndex = async (
|
||||
* @param executeWithReusedStatement - Function to execute with reused prepared statement
|
||||
* @param onProgress - Callback for progress updates
|
||||
* @param config - Optional configuration override
|
||||
* @param skipNodeIds - Optional set of node IDs that already have embeddings (incremental mode)
|
||||
* @param existingEmbeddings - Optional map of nodeId → contentHash for incremental mode.
|
||||
* Nodes whose hash matches are skipped; nodes with a changed hash are DELETE'd
|
||||
* and re-embedded; nodes not in the map are embedded fresh.
|
||||
*/
|
||||
export const runEmbeddingPipeline = async (
|
||||
executeQuery: (cypher: string) => Promise<any[]>,
|
||||
@@ -158,7 +172,7 @@ export const runEmbeddingPipeline = async (
|
||||
) => Promise<void>,
|
||||
onProgress: EmbeddingProgressCallback,
|
||||
config: Partial<EmbeddingConfig> = {},
|
||||
skipNodeIds?: Set<string>,
|
||||
existingEmbeddings?: Map<string, string>,
|
||||
): Promise<void> => {
|
||||
const finalConfig = { ...DEFAULT_EMBEDDING_CONFIG, ...config };
|
||||
|
||||
@@ -194,13 +208,57 @@ export const runEmbeddingPipeline = async (
|
||||
// Phase 2: Query embeddable nodes
|
||||
let nodes = await queryEmbeddableNodes(executeQuery);
|
||||
|
||||
// Incremental mode: filter out nodes that already have embeddings
|
||||
if (skipNodeIds && skipNodeIds.size > 0) {
|
||||
// Incremental mode: compare content hashes, delete stale rows, skip fresh ones.
|
||||
// Computed hashes for stale nodes are cached so batchInsertEmbeddings can reuse them
|
||||
// (avoids double computation).
|
||||
const computedStaleHashes = new Map<string, string>();
|
||||
if (existingEmbeddings && existingEmbeddings.size > 0) {
|
||||
const beforeCount = nodes.length;
|
||||
nodes = nodes.filter((n) => !skipNodeIds.has(n.id));
|
||||
const staleNodeIds: string[] = [];
|
||||
nodes = nodes.filter((n) => {
|
||||
const existingHash = existingEmbeddings.get(n.id);
|
||||
if (existingHash === undefined) {
|
||||
// New node — needs embedding
|
||||
return true;
|
||||
}
|
||||
const currentHash = contentHashForNode(n, finalConfig);
|
||||
if (currentHash !== existingHash) {
|
||||
// Content changed — cache hash for reuse during insert, mark for DELETE + re-embed
|
||||
computedStaleHashes.set(n.id, currentHash);
|
||||
staleNodeIds.push(n.id);
|
||||
return true;
|
||||
}
|
||||
// Hash matches — skip (fresh); no need to cache hash for skipped nodes
|
||||
return false;
|
||||
});
|
||||
|
||||
// DELETE stale embedding rows so they can be re-inserted
|
||||
// (Kuzu forbids SET on vector-indexed properties; DELETE-then-INSERT is the sanctioned pattern)
|
||||
if (staleNodeIds.length > 0) {
|
||||
if (isDev) {
|
||||
console.log(`🔄 Deleting ${staleNodeIds.length} stale embedding rows for re-embed`);
|
||||
}
|
||||
try {
|
||||
await executeWithReusedStatement(
|
||||
`MATCH (e:${EMBEDDING_TABLE_NAME} {nodeId: $nodeId}) DELETE e`,
|
||||
staleNodeIds.map((nodeId) => ({ nodeId })),
|
||||
);
|
||||
} catch (err) {
|
||||
// "does not exist" = rows already gone — safe to proceed.
|
||||
// All other errors risk vector-index corruption (Kuzu requires DELETE-before-INSERT
|
||||
// for vector-indexed properties) — propagate so the pipeline aborts cleanly.
|
||||
const msg = err instanceof Error ? err.message : String(err);
|
||||
if (!msg.includes('does not exist')) {
|
||||
throw new Error(
|
||||
`[embed] Failed to delete stale embedding rows — aborting to prevent vector-index corruption: ${msg}`,
|
||||
);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (isDev) {
|
||||
console.log(
|
||||
`📦 Incremental embeddings: ${beforeCount} total, ${skipNodeIds.size} cached, ${nodes.length} to embed`,
|
||||
`📦 Incremental embeddings: ${beforeCount} total, ${existingEmbeddings.size} cached, ${staleNodeIds.length} stale, ${nodes.length} to embed`,
|
||||
);
|
||||
}
|
||||
}
|
||||
@@ -212,6 +270,11 @@ export const runEmbeddingPipeline = async (
|
||||
}
|
||||
|
||||
if (totalNodes === 0) {
|
||||
// Ensure the vector index exists even when no new nodes need embedding.
|
||||
// A prior crash or first-time incremental run may have left CodeEmbedding
|
||||
// rows without ever reaching index creation.
|
||||
await createVectorIndex(executeQuery);
|
||||
|
||||
onProgress({
|
||||
phase: 'ready',
|
||||
percent: 100,
|
||||
@@ -250,6 +313,7 @@ export const runEmbeddingPipeline = async (
|
||||
const updates = batch.map((node, i) => ({
|
||||
id: node.id,
|
||||
embedding: embeddingToArray(embeddings[i]),
|
||||
contentHash: computedStaleHashes.get(node.id) ?? contentHashForNode(node, finalConfig),
|
||||
}));
|
||||
|
||||
await batchInsertEmbeddings(executeWithReusedStatement, updates);
|
||||
@@ -338,7 +402,7 @@ export const semanticSearch = async (
|
||||
|
||||
// Query the vector index on CodeEmbedding to get nodeIds and distances
|
||||
const vectorQuery = `
|
||||
CALL QUERY_VECTOR_INDEX('CodeEmbedding', 'code_embedding_idx',
|
||||
CALL QUERY_VECTOR_INDEX('${EMBEDDING_TABLE_NAME}', '${EMBEDDING_INDEX_NAME}',
|
||||
CAST(${queryVecStr} AS FLOAT[${queryVec.length}]), ${k})
|
||||
YIELD node AS emb, distance
|
||||
WITH emb, distance
|
||||
|
||||
@@ -314,7 +314,7 @@ export async function buildProtoMap(repoPath: string): Promise<Map<string, Proto
|
||||
}
|
||||
|
||||
export function resolveProtoConflict(
|
||||
_serviceName: string,
|
||||
serviceName: string,
|
||||
sourceFilePath: string,
|
||||
candidates: ProtoServiceInfo[],
|
||||
): ProtoServiceInfo | null {
|
||||
@@ -322,17 +322,29 @@ export function resolveProtoConflict(
|
||||
if (candidates.length === 1) return candidates[0];
|
||||
|
||||
const sourceDir = normalizeProtoPath(path.dirname(sourceFilePath));
|
||||
let best = candidates[0];
|
||||
let bestScore = -1;
|
||||
for (const c of candidates) {
|
||||
const scored = candidates.map((c) => {
|
||||
const protoDir = normalizeProtoPath(path.dirname(c.protoPath));
|
||||
const sharedRun = longestSharedSegmentRun(sourceDir, protoDir);
|
||||
if (sharedRun > bestScore) {
|
||||
bestScore = sharedRun;
|
||||
best = c;
|
||||
}
|
||||
return { candidate: c, score: longestSharedSegmentRun(sourceDir, protoDir) };
|
||||
});
|
||||
|
||||
let maxScore = -1;
|
||||
for (const s of scored) {
|
||||
if (s.score > maxScore) maxScore = s.score;
|
||||
}
|
||||
return best;
|
||||
const winners = scored.filter((s) => s.score === maxScore);
|
||||
|
||||
// Path heuristic cannot uniquely identify a winner — refuse to guess.
|
||||
// Ties (including all-zero ties) would otherwise silently merge unrelated
|
||||
// services under a fabricated package-qualified contract id.
|
||||
if (winners.length !== 1) {
|
||||
const paths = candidates.map((c) => c.protoPath).join(', ');
|
||||
console.warn(
|
||||
`[grpc-extractor] Ambiguous proto resolution for service "${serviceName}" from ${sourceFilePath}: ${winners.length} candidates tied at score ${maxScore} among [${paths}] — skipping canonical contract`,
|
||||
);
|
||||
return null;
|
||||
}
|
||||
|
||||
return winners[0].candidate;
|
||||
}
|
||||
|
||||
export function serviceContractId(pkg: string, serviceName: string): string {
|
||||
@@ -410,7 +422,8 @@ export class GrpcExtractor implements ContractExtractor {
|
||||
continue;
|
||||
}
|
||||
for (const d of detections) {
|
||||
out.push(this.detectionToContract(d, rel, protoMap));
|
||||
const contract = this.detectionToContract(d, rel, protoMap);
|
||||
if (contract) out.push(contract);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -428,9 +441,13 @@ export class GrpcExtractor implements ContractExtractor {
|
||||
d: GrpcDetection,
|
||||
filePath: string,
|
||||
protoMap: Map<string, ProtoServiceInfo[]>,
|
||||
): ExtractedContract {
|
||||
const candidates = protoMap.get(d.serviceName);
|
||||
const proto = resolveProtoConflict(d.serviceName, filePath, candidates ?? []);
|
||||
): ExtractedContract | null {
|
||||
const candidates = protoMap.get(d.serviceName) ?? [];
|
||||
const proto = resolveProtoConflict(d.serviceName, filePath, candidates);
|
||||
// If there were proto candidates but resolution was ambiguous, skip
|
||||
// contract emission rather than fabricating a package-qualified id from
|
||||
// an arbitrary candidate. resolveProtoConflict already warned.
|
||||
if (candidates.length > 0 && proto === null) return null;
|
||||
const pkg = proto?.package ?? '';
|
||||
const cid = d.methodName
|
||||
? contractId(pkg, d.serviceName, d.methodName)
|
||||
|
||||
@@ -245,7 +245,30 @@ export class HttpRouteExtractor implements ContractExtractor {
|
||||
const providerDetections = detections.filter((d) => d.role === 'provider');
|
||||
let handlerName: string | null = null;
|
||||
const normalizedRoute = normalizeHttpPath(routePath);
|
||||
const match = providerDetections.find((d) => normalizeHttpPath(d.path) === normalizedRoute);
|
||||
// Candidates share the same normalized path. When multiple
|
||||
// detections at the same path exist (e.g. GET + POST /api/orders
|
||||
// in one router), a blind `.find()` silently returned the first
|
||||
// verb — attaching the wrong handler and, when method was not
|
||||
// already pinned by the route reason, the wrong method too.
|
||||
// Disambiguate by method when we know it; refuse to guess when
|
||||
// we don't.
|
||||
const candidates = providerDetections.filter(
|
||||
(d) => normalizeHttpPath(d.path) === normalizedRoute,
|
||||
);
|
||||
let match: (typeof candidates)[number] | undefined;
|
||||
const ambiguousCandidates = !method && candidates.length > 1;
|
||||
if (method) {
|
||||
match = candidates.find((d) => d.method === method);
|
||||
} else if (candidates.length === 1) {
|
||||
match = candidates[0];
|
||||
}
|
||||
// else: multiple candidates + unknown method → leave match
|
||||
// undefined so handlerName stays null and skip symbol
|
||||
// enrichment below, keeping the file-basename fallback instead
|
||||
// of letting pickSymbolUid silently pick the first Function /
|
||||
// Method in the file (which reintroduces the mis-attribution
|
||||
// we were trying to avoid). Method stays at the conservative
|
||||
// 'GET' default set below.
|
||||
if (match) {
|
||||
if (!method) method = match.method;
|
||||
handlerName = match.name;
|
||||
@@ -259,7 +282,7 @@ export class HttpRouteExtractor implements ContractExtractor {
|
||||
let symbolName = path.basename(filePath) || 'handler';
|
||||
let symPath = filePath;
|
||||
const fileId = row.fileId ?? row[0];
|
||||
if (fileId) {
|
||||
if (fileId && !ambiguousCandidates) {
|
||||
try {
|
||||
const syms = await db(CONTAINS_QUERY, { fileId });
|
||||
if (syms.length > 0) {
|
||||
@@ -347,10 +370,19 @@ export class HttpRouteExtractor implements ContractExtractor {
|
||||
// Prefer the plugin's detected method if we can find a matching
|
||||
// fetch/axios call in the same file.
|
||||
const detections = filePath ? getDetections(filePath) : [];
|
||||
const inferred = detections.find(
|
||||
// Symmetric to the provider path: if multiple consumer calls in
|
||||
// the same file share the same normalized path (e.g. a GET
|
||||
// fetch AND a POST fetch to `/api/orders`), `.find()` silently
|
||||
// picked the first verb and keyed the contract id on the wrong
|
||||
// method. With no upstream method signal here, refuse to guess
|
||||
// when candidates are ambiguous — leave `method` at its
|
||||
// conservative 'GET' default.
|
||||
const consumerCandidates = detections.filter(
|
||||
(d) => d.role === 'consumer' && normalizeConsumerPath(d.path) === pathNorm,
|
||||
);
|
||||
if (inferred) method = inferred.method;
|
||||
if (consumerCandidates.length === 1) {
|
||||
method = consumerCandidates[0].method;
|
||||
}
|
||||
|
||||
const cid = contractIdFor(method, pathNorm);
|
||||
let symbolUid = '';
|
||||
|
||||
@@ -23,6 +23,34 @@ function normalizeRoutePath(raw: string): string {
|
||||
return collapsed.replace(/\/+$/, '');
|
||||
}
|
||||
|
||||
/**
|
||||
* Split a manifest HTTP contract into its optional `METHOD::` prefix and
|
||||
* its path portion.
|
||||
*
|
||||
* `buildContractId` recommends the explicit-method form `GET::/api/orders`
|
||||
* in group.yaml; if we hand that raw string to `normalizeRoutePath` we get
|
||||
* `/GET::/api/orders`, which can never match `Route.name = "/api/orders"`
|
||||
* in the graph. This helper extracts the path portion so the Cypher
|
||||
* lookup uses the canonical route name.
|
||||
*
|
||||
* The method prefix regex mirrors `buildContractId` (line ~251) for
|
||||
* symmetry: case-insensitive `[A-Za-z]+` followed by `::`. The captured
|
||||
* method is upper-cased for downstream use; method-constrained matching
|
||||
* against `HANDLES_ROUTE` is a future enhancement (not yet wired).
|
||||
*
|
||||
* Edge cases:
|
||||
* - `"::/api/orders"` — empty method portion, no alpha prefix match, so
|
||||
* the whole string is treated as a bare path (matches buildContractId
|
||||
* which also requires `[A-Za-z]+`).
|
||||
* - `"GET::"` — method with empty path, returns `{ method: 'GET', path: '' }`;
|
||||
* `normalizeRoutePath('')` resolves to `/` for caller.
|
||||
*/
|
||||
function parseHttpContract(raw: string): { method: string | null; path: string } {
|
||||
const match = raw.match(/^([A-Za-z]+)::/);
|
||||
if (!match) return { method: null, path: raw };
|
||||
return { method: match[1].toUpperCase(), path: raw.slice(match[0].length) };
|
||||
}
|
||||
|
||||
/**
|
||||
* Stable synthetic symbolUid for a manifest-declared contract whose target
|
||||
* symbol could not be resolved against the per-repo graph (resolveSymbol
|
||||
@@ -51,17 +79,50 @@ export class ManifestExtractor {
|
||||
links: GroupManifestLink[],
|
||||
dbExecutors?: Map<string, CypherExecutor>,
|
||||
): Promise<ManifestExtractResult> {
|
||||
// Resolve all (repo, link) pairs in parallel. The previous sequential
|
||||
// await-per-link produced 2N round-trips; parallel resolution uses the
|
||||
// per-repo executor pool directly and scales linearly with manifest size.
|
||||
//
|
||||
// Memoization: a manifest can list the same contract multiple times
|
||||
// (e.g. a consumer and provider declaration, or cross-referenced groups).
|
||||
// Key on (repo, type, contract) — the canonical input to the Cypher
|
||||
// query — so duplicate links resolve to one DB hit.
|
||||
type ResolvedSymbol = { filePath: string; name: string; uid: string } | null;
|
||||
const resolveCache = new Map<string, Promise<ResolvedSymbol>>();
|
||||
const resolveOnce = (repo: string, link: GroupManifestLink): Promise<ResolvedSymbol> => {
|
||||
const key = `${repo}\u0000${link.type}\u0000${link.contract}`;
|
||||
let pending = resolveCache.get(key);
|
||||
if (!pending) {
|
||||
pending = this.resolveSymbol(repo, link, dbExecutors);
|
||||
resolveCache.set(key, pending);
|
||||
}
|
||||
return pending;
|
||||
};
|
||||
|
||||
const perLink = await Promise.all(
|
||||
links.map(async (link) => {
|
||||
const contractId = this.buildContractId(link.type, link.contract);
|
||||
const providerRepo = link.role === 'provider' ? link.from : link.to;
|
||||
const consumerRepo = link.role === 'provider' ? link.to : link.from;
|
||||
const [providerSymbol, consumerSymbol] = await Promise.all([
|
||||
resolveOnce(providerRepo, link),
|
||||
resolveOnce(consumerRepo, link),
|
||||
]);
|
||||
return { link, contractId, providerRepo, consumerRepo, providerSymbol, consumerSymbol };
|
||||
}),
|
||||
);
|
||||
|
||||
const contracts: StoredContract[] = [];
|
||||
const crossLinks: CrossLink[] = [];
|
||||
|
||||
for (const link of links) {
|
||||
const contractId = this.buildContractId(link.type, link.contract);
|
||||
|
||||
const providerRepo = link.role === 'provider' ? link.from : link.to;
|
||||
const consumerRepo = link.role === 'provider' ? link.to : link.from;
|
||||
|
||||
const providerSymbol = await this.resolveSymbol(providerRepo, link, dbExecutors);
|
||||
const consumerSymbol = await this.resolveSymbol(consumerRepo, link, dbExecutors);
|
||||
for (const {
|
||||
link,
|
||||
contractId,
|
||||
providerRepo,
|
||||
consumerRepo,
|
||||
providerSymbol,
|
||||
consumerSymbol,
|
||||
} of perLink) {
|
||||
const providerRef = providerSymbol || { filePath: '', name: link.contract };
|
||||
const consumerRef = consumerSymbol || { filePath: '', name: link.contract };
|
||||
// When the resolver finds a real graph symbol we keep its uid, otherwise
|
||||
@@ -134,7 +195,15 @@ export class ManifestExtractor {
|
||||
// core/ingestion/pipeline.ts ensureSlash + generateId('Route', ...)).
|
||||
// Normalize the manifest contract the same way so a user-written
|
||||
// "/api/orders" matches "api/orders" in the graph.
|
||||
const normalized = normalizeRoutePath(link.contract);
|
||||
//
|
||||
// The contract may also use the explicit-method form "GET::/api/orders"
|
||||
// recommended by buildContractId. Strip the METHOD:: prefix before
|
||||
// normalizing — otherwise `normalizeRoutePath('GET::/api/orders')`
|
||||
// returns `/GET::/api/orders` and never matches Route.name. The
|
||||
// captured method is not yet used to constrain the Cypher query
|
||||
// (method-aware HANDLES_ROUTE matching is a future enhancement).
|
||||
const parsed = parseHttpContract(link.contract);
|
||||
const normalized = normalizeRoutePath(parsed.path);
|
||||
rows = await executor(
|
||||
`MATCH (handler)-[r:CodeRelation {type: 'HANDLES_ROUTE'}]->(route:Route)
|
||||
WHERE route.name = $normalized
|
||||
@@ -248,8 +317,15 @@ export class ManifestExtractor {
|
||||
private buildContractId(type: ContractType, contract: string): string {
|
||||
switch (type) {
|
||||
case 'http': {
|
||||
if (/^[A-Za-z]+::/.test(contract)) return `http::${contract}`;
|
||||
return `http::*::${contract}`;
|
||||
// Canonicalize method casing and path separators so logically
|
||||
// equivalent inputs (`get::/api/orders` vs `GET::/api/orders`,
|
||||
// or trailing-slash variants) produce the same contractId and
|
||||
// matching `manifestSymbolUid` fallback. Without this, raw
|
||||
// user casing leaks into cross-impact join keys and fragments
|
||||
// matches across repos.
|
||||
const { method, path: rawPath } = parseHttpContract(contract);
|
||||
const normalizedPath = normalizeRoutePath(rawPath);
|
||||
return method ? `http::${method}::${normalizedPath}` : `http::*::${normalizedPath}`;
|
||||
}
|
||||
case 'grpc':
|
||||
return `grpc::${contract}`;
|
||||
|
||||
@@ -7,6 +7,7 @@ import type { GroupConfig, RepoHandle, RepoSnapshot, StoredContract, CrossLink }
|
||||
import { HttpRouteExtractor } from './extractors/http-route-extractor.js';
|
||||
import { GrpcExtractor } from './extractors/grpc-extractor.js';
|
||||
import { TopicExtractor } from './extractors/topic-extractor.js';
|
||||
import { ManifestExtractor } from './extractors/manifest-extractor.js';
|
||||
import { runExactMatch } from './matching.js';
|
||||
import { detectServiceBoundaries, assignService } from './service-boundary-detector.js';
|
||||
import type { CypherExecutor } from './contract-extractor.js';
|
||||
@@ -60,10 +61,28 @@ function defaultResolveHandle(allEntries: RegistryEntry[]) {
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Dedupe cross-links that point from the same consumer endpoint to the same
|
||||
* provider endpoint for the same contract. Preserves first-seen order so the
|
||||
* caller controls precedence (e.g., pass manifest links first).
|
||||
*/
|
||||
function dedupeCrossLinks(links: CrossLink[]): CrossLink[] {
|
||||
const seen = new Set<string>();
|
||||
const out: CrossLink[] = [];
|
||||
for (const link of links) {
|
||||
const key = `${link.from.repo}::${link.from.symbolUid}|${link.to.repo}::${link.to.symbolUid}|${link.type}|${link.contractId}`;
|
||||
if (seen.has(key)) continue;
|
||||
seen.add(key);
|
||||
out.push(link);
|
||||
}
|
||||
return out;
|
||||
}
|
||||
|
||||
export async function syncGroup(config: GroupConfig, opts?: SyncOptions): Promise<SyncResult> {
|
||||
const missingRepos: string[] = [];
|
||||
const repoSnapshots: Record<string, RepoSnapshot> = {};
|
||||
let autoContracts: StoredContract[] = [];
|
||||
let manifestCrossLinks: CrossLink[] = [];
|
||||
let dbExecutors: Map<string, CypherExecutor> | undefined;
|
||||
|
||||
const eo = opts?.extractorOverride;
|
||||
@@ -158,8 +177,44 @@ export async function syncGroup(config: GroupConfig, opts?: SyncOptions): Promis
|
||||
}
|
||||
}
|
||||
|
||||
// Process manifest links declared in group.yaml.
|
||||
// ManifestExtractor is fully implemented but was never wired into this
|
||||
// pipeline — config.links were parsed and validated but silently dropped.
|
||||
// Placed after the DB try/finally: resolveSymbol falls back to synthetic
|
||||
// UIDs when dbExecutors is undefined or a pool is closed, so cross-links
|
||||
// are always generated regardless of whether real DB executors are available.
|
||||
if (config.links.length > 0) {
|
||||
// Warn about dangling links that reference repos not declared in config.repos.
|
||||
// They still generate cross-links via synthetic UIDs (determinism is preserved),
|
||||
// but the operator probably meant something that now silently does nothing useful.
|
||||
const knownRepos = new Set(Object.keys(config.repos));
|
||||
for (const link of config.links) {
|
||||
const dangling = [link.from, link.to].filter((r) => !knownRepos.has(r));
|
||||
if (dangling.length > 0) {
|
||||
console.warn(
|
||||
`[group/sync] manifest link ${link.type}:${link.contract} references repos not in config.repos: ${dangling.join(', ')} — cross-links will use synthetic UIDs`,
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
const manifestEx = new ManifestExtractor();
|
||||
const manifestResult = await manifestEx.extractFromManifest(config.links, dbExecutors);
|
||||
autoContracts.push(...manifestResult.contracts);
|
||||
manifestCrossLinks = manifestResult.crossLinks;
|
||||
if (opts?.verbose) {
|
||||
console.log(
|
||||
` manifest: ${manifestCrossLinks.length} cross-links from ${config.links.length} declared links`,
|
||||
);
|
||||
}
|
||||
}
|
||||
|
||||
const { matched, unmatched } = runExactMatch(autoContracts);
|
||||
const crossLinks: CrossLink[] = matched;
|
||||
|
||||
// Dedupe cross-links. Manifest contracts participate in runExactMatch, so a
|
||||
// manifest-declared link can also emit a matchType:'exact' CrossLink with the
|
||||
// same endpoints. Prefer the manifest version — it reflects operator intent
|
||||
// and carries matchType:'manifest' which downstream consumers may rely on.
|
||||
const crossLinks = dedupeCrossLinks([...manifestCrossLinks, ...matched]);
|
||||
const allContracts: StoredContract[] = autoContracts;
|
||||
|
||||
const registry: ContractRegistry = {
|
||||
|
||||
@@ -30,8 +30,30 @@ export type CaptureMap = Record<string, SyntaxNode | undefined>;
|
||||
// so `core/ingestion/model/resolve.ts` can consume it without importing from
|
||||
// this file (which would pull in the full language-registry dependency graph).
|
||||
|
||||
/** How a language handles imports — determines wildcard synthesis behavior. */
|
||||
export type ImportSemantics = 'named' | 'wildcard' | 'namespace';
|
||||
/**
|
||||
* How a language handles imports — determines wildcard synthesis behavior.
|
||||
*
|
||||
* Import resolution is a graph-traversal policy with multiple distinct strategies,
|
||||
* analogous to MRO for method resolution. Each tag picks a strategy:
|
||||
*
|
||||
* | Tag | Mechanism | Traversal | Languages |
|
||||
* |-----------------------|------------------------------------------------|---------------------|--------------------------------------------|
|
||||
* | `named` | Per-symbol imports | None (use-site) | JS/TS, Java, C#, Rust, PHP, Kotlin, Vue |
|
||||
* | `wildcard-transitive` | Textual paste, symbols chain through files | BFS closure | C, C++ (future: Obj-C, Fortran, Nim) |
|
||||
* | `wildcard-leaf` | Whole public API, single hop | None (direct only) | Go, Ruby, Swift, Dart |
|
||||
* | `namespace` | Qualified handle; symbols resolved at call site| None at import | Python |
|
||||
* | `explicit-reexport` | Opt-in per-symbol re-export (SCAFFOLD) | Topological DAG | (future: TS `export *`, Rust `pub use`) |
|
||||
*
|
||||
* The `explicit-reexport` tag is a compile-time scaffold; no provider claims it yet.
|
||||
* It falls through to `wildcard-leaf` behavior in synthesis so today's TS/Rust
|
||||
* handling is unchanged. A future PR will implement the DAG walk for `export *`.
|
||||
*/
|
||||
export type ImportSemantics =
|
||||
| 'named'
|
||||
| 'wildcard-transitive'
|
||||
| 'wildcard-leaf'
|
||||
| 'namespace'
|
||||
| 'explicit-reexport';
|
||||
|
||||
/**
|
||||
* Everything a language needs to provide.
|
||||
@@ -68,10 +90,12 @@ interface LanguageProviderConfig {
|
||||
/** Named binding extraction from import statements.
|
||||
* Default: undefined (language uses wildcard/whole-module imports). */
|
||||
readonly namedBindingExtractor?: NamedBindingExtractorFn;
|
||||
/** How this language handles imports.
|
||||
/** How this language handles imports. See `ImportSemantics` for the full taxonomy.
|
||||
* - 'named': per-symbol imports (JS/TS, Java, C#, Rust, PHP, Kotlin)
|
||||
* - 'wildcard': whole-module imports, needs synthesis (Go, Ruby, C/C++, Swift)
|
||||
* - 'namespace': namespace imports, needs moduleAliasMap (Python)
|
||||
* - 'wildcard-transitive': textual-include closure; imports chain through files (C, C++)
|
||||
* - 'wildcard-leaf': whole-module single-hop imports; no transitive chaining (Go, Ruby, Swift, Dart)
|
||||
* - 'namespace': qualified namespace imports, needs moduleAliasMap (Python)
|
||||
* - 'explicit-reexport': opt-in per-symbol re-export (scaffold; no provider uses yet)
|
||||
* Default: 'named'. */
|
||||
readonly importSemantics?: ImportSemantics;
|
||||
/** Language-specific transformation of raw import path text before resolution.
|
||||
|
||||
@@ -321,7 +321,7 @@ export const cProvider = defineLanguage({
|
||||
typeConfig: cCppConfig,
|
||||
exportChecker: cCppExportChecker,
|
||||
importResolver: resolveCImport,
|
||||
importSemantics: 'wildcard',
|
||||
importSemantics: 'wildcard-transitive',
|
||||
fieldExtractor: createFieldExtractor(cFieldConfig),
|
||||
methodExtractor: createMethodExtractor({
|
||||
...cMethodConfig,
|
||||
@@ -339,7 +339,7 @@ export const cppProvider = defineLanguage({
|
||||
typeConfig: cCppConfig,
|
||||
exportChecker: cCppExportChecker,
|
||||
importResolver: resolveCppImport,
|
||||
importSemantics: 'wildcard',
|
||||
importSemantics: 'wildcard-transitive',
|
||||
mroStrategy: 'leftmost-base',
|
||||
fieldExtractor: createFieldExtractor(cppFieldConfig),
|
||||
methodExtractor: createMethodExtractor({
|
||||
|
||||
@@ -2,7 +2,7 @@
|
||||
* Dart Language Provider
|
||||
*
|
||||
* Dart traits:
|
||||
* - importSemantics: 'wildcard' (Dart imports bring everything public into scope)
|
||||
* - importSemantics: 'wildcard-leaf' (Dart imports bring everything public into scope)
|
||||
* - exportChecker: public if no leading underscore
|
||||
* - Dart SDK imports (dart:*) and external packages are skipped
|
||||
* - enclosingFunctionFinder: Dart's tree-sitter grammar places function_body
|
||||
@@ -90,7 +90,7 @@ export const dartProvider = defineLanguage({
|
||||
typeConfig: dartConfig,
|
||||
exportChecker: dartExportChecker,
|
||||
importResolver: resolveDartImport,
|
||||
importSemantics: 'wildcard',
|
||||
importSemantics: 'wildcard-leaf',
|
||||
fieldExtractor: createFieldExtractor(dartFieldConfig),
|
||||
methodExtractor: createMethodExtractor(dartMethodConfig),
|
||||
classExtractor: createClassExtractor({
|
||||
|
||||
@@ -5,7 +5,7 @@
|
||||
* LanguageProvider, following the Strategy pattern used by the pipeline.
|
||||
*
|
||||
* Key Go traits:
|
||||
* - importSemantics: 'wildcard' (Go imports entire packages)
|
||||
* - importSemantics: 'wildcard-leaf' (Go imports entire packages)
|
||||
* - callRouter: present (Go method calls may need routing)
|
||||
*/
|
||||
|
||||
@@ -28,7 +28,7 @@ export const goProvider = defineLanguage({
|
||||
typeConfig: goConfig,
|
||||
exportChecker: goExportChecker,
|
||||
importResolver: resolveGoImport,
|
||||
importSemantics: 'wildcard',
|
||||
importSemantics: 'wildcard-leaf',
|
||||
fieldExtractor: createFieldExtractor(goFieldConfig),
|
||||
methodExtractor: createMethodExtractor(goMethodConfig),
|
||||
classExtractor: createClassExtractor({
|
||||
|
||||
@@ -107,7 +107,7 @@ export const rubyProvider = defineLanguage({
|
||||
exportChecker: rubyExportChecker,
|
||||
importResolver: resolveRubyImport,
|
||||
callRouter: routeRubyCall,
|
||||
importSemantics: 'wildcard',
|
||||
importSemantics: 'wildcard-leaf',
|
||||
resolveEnclosingOwner(node) {
|
||||
// Ruby singleton_class (class << self) should resolve to the enclosing
|
||||
// class or module for owner/container resolution (HAS_METHOD edges, class IDs).
|
||||
|
||||
@@ -5,7 +5,7 @@
|
||||
* LanguageProvider, following the Strategy pattern used by the pipeline.
|
||||
*
|
||||
* Key Swift traits:
|
||||
* - importSemantics: 'wildcard' (Swift imports entire modules)
|
||||
* - importSemantics: 'wildcard-leaf' (Swift imports entire modules)
|
||||
* - heritageDefaultEdge: 'IMPLEMENTS' (protocols are more common than class inheritance)
|
||||
* - implicitImportWirer: all files in the same SPM target see each other
|
||||
*/
|
||||
@@ -238,7 +238,7 @@ export const swiftProvider = defineLanguage({
|
||||
typeConfig: swiftConfig,
|
||||
exportChecker: swiftExportChecker,
|
||||
importResolver: resolveSwiftImport,
|
||||
importSemantics: 'wildcard',
|
||||
importSemantics: 'wildcard-leaf',
|
||||
heritageDefaultEdge: 'IMPLEMENTS',
|
||||
fieldExtractor: createFieldExtractor(swiftFieldConfig),
|
||||
methodExtractor: createMethodExtractor({
|
||||
|
||||
@@ -15,8 +15,10 @@
|
||||
|
||||
import type { KnowledgeGraph } from '../../graph/types.js';
|
||||
import type { createResolutionContext } from '../model/resolution-context.js';
|
||||
import { getLanguageFromFilename, SupportedLanguages } from 'gitnexus-shared';
|
||||
import { getLanguageFromFilename } from 'gitnexus-shared';
|
||||
import type { SupportedLanguages } from 'gitnexus-shared';
|
||||
import { providers, getProviderForFile } from '../languages/index.js';
|
||||
import type { LanguageProvider, ImportSemantics } from '../language-provider.js';
|
||||
|
||||
// ── Constants ──────────────────────────────────────────────────────────────
|
||||
|
||||
@@ -41,10 +43,29 @@ const IMPORTABLE_SYMBOL_LABELS = new Set([
|
||||
* for C/C++ files that include many large headers. */
|
||||
const MAX_SYNTHETIC_BINDINGS_PER_FILE = 1000;
|
||||
|
||||
/** Max files allowed in a single transitive include closure. Guards against
|
||||
* OOM on pathological C/C++ codebases (boost, Linux kernel-style monoheaders)
|
||||
* where a single translation unit can transitively reach many thousands of
|
||||
* headers. When the cap is hit, BFS expansion stops early — the file still
|
||||
* synthesizes bindings from the partial closure rather than failing. */
|
||||
const MAX_TRANSITIVE_CLOSURE_SIZE = 5000;
|
||||
|
||||
/** Import semantics tags whose languages need synthesis of whole-module imports.
|
||||
* `wildcard-transitive` (C/C++) and `wildcard-leaf` (Go, Ruby, Swift, Dart) are
|
||||
* the file-based wildcard strategies. `explicit-reexport` is a scaffold tag —
|
||||
* no provider uses it yet, but it goes through the same leaf-style synthesis
|
||||
* path today because a re-exporter is still an importer; only the extra DAG
|
||||
* walk to surface re-exported symbols is missing (future work). */
|
||||
const WILDCARD_SEMANTICS: ReadonlySet<ImportSemantics> = new Set<ImportSemantics>([
|
||||
'wildcard-transitive',
|
||||
'wildcard-leaf',
|
||||
'explicit-reexport',
|
||||
]);
|
||||
|
||||
/** Languages with whole-module import semantics (derived from providers at module load). */
|
||||
const WILDCARD_LANGUAGES = new Set(
|
||||
Object.values(providers)
|
||||
.filter((p) => p.importSemantics === 'wildcard')
|
||||
.filter((p) => WILDCARD_SEMANTICS.has(p.importSemantics))
|
||||
.map((p) => p.id),
|
||||
);
|
||||
|
||||
@@ -66,6 +87,84 @@ export function needsSynthesis(lang: SupportedLanguages): boolean {
|
||||
return SYNTHESIS_LANGUAGES.has(lang);
|
||||
}
|
||||
|
||||
// ── Strategy implementations ───────────────────────────────────────────────
|
||||
|
||||
/**
|
||||
* Strategy implementation for `importSemantics: 'wildcard-transitive'` (C, C++).
|
||||
*
|
||||
* Textual-include languages chain symbols through files: if `dict.c` includes
|
||||
* `server.h` and `server.h` includes `dict.h`, then `dict.c` sees symbols from
|
||||
* all three files. This helper walks the include graph (combining both the
|
||||
* ingestion-context `importMap` and the graph-level IMPORTS edges) until the
|
||||
* closure is stable.
|
||||
*
|
||||
* **Order matters.** The returned `Set` preserves iteration order (insertion
|
||||
* order). `synthesizeWildcardImportBindings` dedupes bindings by symbol name
|
||||
* on a first-seen-wins basis, so this closure's ordering determines which
|
||||
* declaration wins when multiple headers export the same name (e.g. overloaded
|
||||
* free functions like `write_audit()` vs `write_audit(const char*)` in
|
||||
* different headers). We therefore:
|
||||
* 1. Seed the closure with direct imports in declaration order (matches the
|
||||
* order of `#include` directives in the source file).
|
||||
* 2. Use FIFO / true BFS (`queue.shift()`) for transitive expansion, so
|
||||
* closer headers are seen before deeper ones.
|
||||
*
|
||||
* Cycle-safe: the `closure.has(file)` guard prevents infinite loops on circular
|
||||
* header includes, which are valid C/C++ when paired with `#pragma once` or
|
||||
* include guards.
|
||||
*
|
||||
* Size-bounded: the closure is capped at `MAX_TRANSITIVE_CLOSURE_SIZE` files to
|
||||
* prevent OOM on pathological codebases (e.g. boost, monoheader kernel code)
|
||||
* where one translation unit can transitively reach tens of thousands of
|
||||
* headers. Partial closures still yield useful bindings for the cluster of
|
||||
* headers closest to the importer, which is what overload resolution and
|
||||
* cross-file call resolution care about.
|
||||
*
|
||||
* Queue implementation: uses a head-index over a growing array (O(1) dequeue)
|
||||
* instead of `Array.prototype.shift()` (O(n)) so deep chains stay linear.
|
||||
*/
|
||||
export function expandTransitiveIncludeClosure(
|
||||
directImports: Iterable<string>,
|
||||
importMap: ReadonlyMap<string, ReadonlySet<string>>,
|
||||
graphImports: ReadonlyMap<string, ReadonlySet<string>>,
|
||||
): Set<string> {
|
||||
const closure = new Set<string>();
|
||||
const queue: string[] = [];
|
||||
let head = 0; // O(1) dequeue: advance the head index instead of shift()-ing.
|
||||
|
||||
const tryEnqueue = (file: string): boolean => {
|
||||
if (closure.has(file)) return true;
|
||||
if (closure.size >= MAX_TRANSITIVE_CLOSURE_SIZE) return false;
|
||||
closure.add(file);
|
||||
queue.push(file);
|
||||
return true;
|
||||
};
|
||||
|
||||
// Seed direct imports in declaration order (see JSDoc on order-sensitivity).
|
||||
for (const f of directImports) {
|
||||
if (!tryEnqueue(f)) break;
|
||||
}
|
||||
// True BFS for transitive reach: head-index FIFO preserves the "closer
|
||||
// headers first" ordering that overload resolution depends on.
|
||||
while (head < queue.length) {
|
||||
if (closure.size >= MAX_TRANSITIVE_CLOSURE_SIZE) break;
|
||||
const file = queue[head++]!;
|
||||
const nested = importMap.get(file);
|
||||
if (nested) {
|
||||
for (const n of nested) {
|
||||
if (!tryEnqueue(n)) break;
|
||||
}
|
||||
}
|
||||
const nestedGraph = graphImports.get(file);
|
||||
if (nestedGraph) {
|
||||
for (const n of nestedGraph) {
|
||||
if (!tryEnqueue(n)) break;
|
||||
}
|
||||
}
|
||||
}
|
||||
return closure;
|
||||
}
|
||||
|
||||
// ── Main synthesis function ────────────────────────────────────────────────
|
||||
|
||||
/**
|
||||
@@ -149,16 +248,67 @@ export function synthesizeWildcardImportBindings(
|
||||
}
|
||||
};
|
||||
|
||||
// Synthesize from ctx.importMap (Ruby, C/C++, Swift file-based imports)
|
||||
/**
|
||||
* Dispatch wildcard synthesis by the file's language provider strategy.
|
||||
*
|
||||
* Strategy tags (see `ImportSemantics`):
|
||||
* - `wildcard-transitive`: expand the include closure first (C/C++ #include
|
||||
* chains — e.g. `dict.c` → `server.h` → `dict.h` so `dictFind` resolves
|
||||
* across header chains)
|
||||
* - `wildcard-leaf`: synthesize from direct imports only (Go, Ruby, Swift, Dart)
|
||||
* - `explicit-reexport`: scaffold tag; falls through to leaf behavior.
|
||||
* TODO(#821): implement re-export DAG walk for TS `export *` / Rust
|
||||
* `pub use`. The leaf fallthrough preserves today's TS/Rust behavior
|
||||
* (their direct imports still synthesize correctly); only the extra
|
||||
* re-export DAG walk for barrel-file correctness is missing.
|
||||
* - `namespace` / `named`: no-op here (namespace handled in Loop 3 below,
|
||||
* named needs no synthesis).
|
||||
*
|
||||
* Used by both Loop 1 (ctx.importMap) and Loop 2 (graphImports) so a future
|
||||
* transitive-import language whose edges arrive via graphImports gets closure
|
||||
* expansion consistently regardless of edge source.
|
||||
*/
|
||||
const dispatchSynthesis = (
|
||||
filePath: string,
|
||||
importedFiles: ReadonlySet<string>,
|
||||
provider: LanguageProvider,
|
||||
) => {
|
||||
switch (provider.importSemantics) {
|
||||
case 'wildcard-transitive':
|
||||
synthesizeForFile(
|
||||
filePath,
|
||||
expandTransitiveIncludeClosure(importedFiles, ctx.importMap, graphImports),
|
||||
);
|
||||
return;
|
||||
case 'wildcard-leaf':
|
||||
case 'explicit-reexport':
|
||||
synthesizeForFile(filePath, importedFiles);
|
||||
return;
|
||||
case 'namespace':
|
||||
case 'named':
|
||||
return;
|
||||
default: {
|
||||
const _exhaustive: never = provider.importSemantics;
|
||||
void _exhaustive;
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
// Loop 1: synthesize from ctx.importMap (Ruby, C/C++, Swift, Dart file-based imports).
|
||||
for (const [filePath, importedFiles] of ctx.importMap) {
|
||||
const lang = getLanguageFromFilename(filePath);
|
||||
if (!lang || !isWildcardImportLanguage(lang)) continue;
|
||||
synthesizeForFile(filePath, importedFiles);
|
||||
const provider = getProviderForFile(filePath);
|
||||
if (!provider) continue;
|
||||
dispatchSynthesis(filePath, importedFiles, provider);
|
||||
}
|
||||
|
||||
// Synthesize from graph IMPORTS edges (Go and other wildcard-import languages)
|
||||
// Loop 2: synthesize from graph IMPORTS edges (Go and other wildcard-import
|
||||
// languages whose edges live in the graph rather than ctx.importMap).
|
||||
for (const [filePath, importedFiles] of graphImports) {
|
||||
synthesizeForFile(filePath, importedFiles);
|
||||
const provider = getProviderForFile(filePath);
|
||||
if (!provider) continue;
|
||||
dispatchSynthesis(filePath, importedFiles, provider);
|
||||
}
|
||||
|
||||
// Build Python module-alias maps for namespace-import languages.
|
||||
|
||||
@@ -315,14 +315,18 @@ export const streamAllCSVsToDisk = async (
|
||||
CodeElement: codeElemWriter,
|
||||
};
|
||||
|
||||
const seenFileIds = new Set<string>();
|
||||
// Deduplicate all node types — the pipeline can produce duplicate IDs across
|
||||
// all symbol types (Class, Method, Function, etc.), not just File nodes.
|
||||
// A single Set covering every label prevents PK violations on COPY.
|
||||
const seenNodeIds = new Set<string>();
|
||||
|
||||
// --- SINGLE PASS over all nodes ---
|
||||
for (const node of graph.iterNodes()) {
|
||||
if (seenNodeIds.has(node.id)) continue;
|
||||
seenNodeIds.add(node.id);
|
||||
|
||||
switch (node.label) {
|
||||
case 'File': {
|
||||
if (seenFileIds.has(node.id)) break;
|
||||
seenFileIds.add(node.id);
|
||||
const content = await extractContent(node, contentCache);
|
||||
await fileWriter.addRow(
|
||||
[
|
||||
|
||||
@@ -1,6 +1,8 @@
|
||||
import fs from 'fs/promises';
|
||||
import { createReadStream, createWriteStream } from 'fs';
|
||||
import { createInterface } from 'readline';
|
||||
import { once } from 'events';
|
||||
import { finished } from 'stream/promises';
|
||||
import path from 'path';
|
||||
import lbug from '@ladybugdb/core';
|
||||
import { KnowledgeGraph } from '../graph/types.js';
|
||||
@@ -9,16 +11,148 @@ import {
|
||||
REL_TABLE_NAME,
|
||||
SCHEMA_QUERIES,
|
||||
EMBEDDING_TABLE_NAME,
|
||||
STALE_HASH_SENTINEL,
|
||||
NodeTableName,
|
||||
} from './schema.js';
|
||||
import { streamAllCSVsToDisk } from './csv-generator.js';
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Relationship CSV splitting — extracted for testability (PR #818)
|
||||
// ---------------------------------------------------------------------------
|
||||
|
||||
/** Factory for creating WriteStreams — injectable for testing. */
|
||||
export type WriteStreamFactory = (filePath: string) => import('fs').WriteStream;
|
||||
|
||||
/** Result of splitting the relationship CSV into per-label-pair files. */
|
||||
export interface RelCsvSplitResult {
|
||||
relHeader: string;
|
||||
relsByPairMeta: Map<string, { csvPath: string; rows: number }>;
|
||||
pairWriteStreams: Map<string, import('fs').WriteStream>;
|
||||
skippedRels: number;
|
||||
totalValidRels: number;
|
||||
}
|
||||
|
||||
/**
|
||||
* Split a relationship CSV into per-label-pair files on disk.
|
||||
*
|
||||
* Streams the CSV line-by-line, routing each relationship to a file named
|
||||
* `rel_{fromLabel}_{toLabel}.csv`. Handles backpressure correctly: only one
|
||||
* drain listener per stream at a time, and readline resumes only when ALL
|
||||
* backpressured streams have drained.
|
||||
*
|
||||
* @param csvPath Path to the combined relationship CSV
|
||||
* @param csvDir Directory to write per-pair CSV files
|
||||
* @param validTables Set of valid node table names
|
||||
* @param getNodeLabel Function to extract the label from a node ID
|
||||
* @param wsFactory Optional WriteStream factory (defaults to fs.createWriteStream)
|
||||
*/
|
||||
export const splitRelCsvByLabelPair = async (
|
||||
csvPath: string,
|
||||
csvDir: string,
|
||||
validTables: Set<string>,
|
||||
getNodeLabel: (id: string) => string,
|
||||
wsFactory: WriteStreamFactory = (p) => createWriteStream(p, 'utf-8'),
|
||||
): Promise<RelCsvSplitResult> => {
|
||||
let relHeader = '';
|
||||
const relsByPairMeta = new Map<string, { csvPath: string; rows: number }>();
|
||||
const pairWriteStreams = new Map<string, import('fs').WriteStream>();
|
||||
let skippedRels = 0;
|
||||
let totalValidRels = 0;
|
||||
|
||||
const inputStream = createReadStream(csvPath, 'utf-8');
|
||||
const rl = createInterface({ input: inputStream, crlfDelay: Infinity });
|
||||
|
||||
// If any pair WriteStream errors (disk full, EMFILE, etc.) or the input
|
||||
// stream fails, we need to abort the pending `once(ws, 'drain')` await.
|
||||
// An AbortController gives us one signal to cancel all pending waits
|
||||
// without a custom state machine.
|
||||
const abortOnError = new AbortController();
|
||||
let streamError: Error | null = null;
|
||||
const markStreamError = (err: Error): void => {
|
||||
streamError ??= err;
|
||||
abortOnError.abort(err);
|
||||
};
|
||||
|
||||
try {
|
||||
// `for await (const line of rl)` replaces the old manual
|
||||
// on('line')/pause()/resume()/waitingForDrain state machine: readline's
|
||||
// async iterator naturally serializes line delivery with our awaits, so
|
||||
// at most one ws can be in backpressure at a time and we just await its
|
||||
// 'drain' event.
|
||||
let isFirst = true;
|
||||
for await (const line of rl) {
|
||||
if (streamError) throw streamError;
|
||||
if (isFirst) {
|
||||
relHeader = line;
|
||||
isFirst = false;
|
||||
continue;
|
||||
}
|
||||
if (!line.trim()) continue;
|
||||
const match = line.match(/"([^"]*)","([^"]*)"/);
|
||||
if (!match) {
|
||||
skippedRels++;
|
||||
continue;
|
||||
}
|
||||
const fromLabel = getNodeLabel(match[1]);
|
||||
const toLabel = getNodeLabel(match[2]);
|
||||
if (!validTables.has(fromLabel) || !validTables.has(toLabel)) {
|
||||
skippedRels++;
|
||||
continue;
|
||||
}
|
||||
|
||||
const pairKey = `${fromLabel}|${toLabel}`;
|
||||
let ws = pairWriteStreams.get(pairKey);
|
||||
if (!ws) {
|
||||
const pairCsvPath = path.join(csvDir, `rel_${fromLabel}_${toLabel}.csv`);
|
||||
ws = wsFactory(pairCsvPath);
|
||||
ws.on('error', markStreamError);
|
||||
pairWriteStreams.set(pairKey, ws);
|
||||
relsByPairMeta.set(pairKey, { csvPath: pairCsvPath, rows: 0 });
|
||||
if (!ws.write(relHeader + '\n')) {
|
||||
await once(ws, 'drain', { signal: abortOnError.signal });
|
||||
}
|
||||
}
|
||||
|
||||
if (!ws.write(line + '\n')) {
|
||||
await once(ws, 'drain', { signal: abortOnError.signal });
|
||||
}
|
||||
relsByPairMeta.get(pairKey)!.rows++;
|
||||
totalValidRels++;
|
||||
}
|
||||
if (streamError) throw streamError;
|
||||
} catch (err) {
|
||||
// Tear down everything so no fd is left dangling. If the abort was caused
|
||||
// by a stream error, rethrow that error (more actionable than AbortError).
|
||||
for (const ws of pairWriteStreams.values()) ws.destroy();
|
||||
inputStream.destroy();
|
||||
throw streamError ?? err;
|
||||
} finally {
|
||||
// Readline 'close' fires before the underlying fs.ReadStream releases its
|
||||
// fd — on Windows that race caused ENOTEMPTY on the parent dir.
|
||||
// stream/promises.finished is the stdlib "wait until this stream is fully
|
||||
// closed" primitive and handles both success and error paths.
|
||||
await finished(inputStream).catch(() => {});
|
||||
}
|
||||
|
||||
return { relHeader, relsByPairMeta, pairWriteStreams, skippedRels, totalValidRels };
|
||||
};
|
||||
|
||||
let db: lbug.Database | null = null;
|
||||
let conn: lbug.Connection | null = null;
|
||||
let currentDbPath: string | null = null;
|
||||
let ftsLoaded = false;
|
||||
let vectorExtensionLoaded = false;
|
||||
|
||||
/**
|
||||
* Check if an error indicates a missing column or table (schema-level problem)
|
||||
* rather than a transient/connection error. Used for legacy DB fallback logic.
|
||||
*/
|
||||
const isMissingColumnOrTableError = (msg: string): boolean =>
|
||||
msg.includes('does not exist') ||
|
||||
// Kuzu-specific: "(table|column|property) ... not found" — narrow enough to avoid
|
||||
// matching transient errors like "connection not found" or "key not found".
|
||||
/(table|column|property).*not found/i.test(msg);
|
||||
|
||||
/** Expose the current Database for pool adapter reuse in tests. */
|
||||
export const getDatabase = (): lbug.Database | null => db;
|
||||
|
||||
@@ -247,75 +381,17 @@ export const loadGraphToLbug = async (
|
||||
}
|
||||
|
||||
// Bulk COPY relationships — split by FROM→TO label pair (LadybugDB requires it)
|
||||
// Stream-read the relation CSV line by line and write directly to per-pair
|
||||
// temp files on disk. This avoids accumulating potentially millions of CSV
|
||||
// lines in memory which could exceed V8 Map or array limits on large repos.
|
||||
let relHeader = '';
|
||||
const relsByPairMeta = new Map<string, { csvPath: string; rows: number }>();
|
||||
const pairWriteStreams = new Map<string, import('fs').WriteStream>();
|
||||
let skippedRels = 0;
|
||||
let totalValidRels = 0;
|
||||
const { relHeader, relsByPairMeta, pairWriteStreams, skippedRels, totalValidRels } =
|
||||
await splitRelCsvByLabelPair(csvResult.relCsvPath, csvDir, validTables, getNodeLabel);
|
||||
|
||||
await new Promise<void>((resolve, reject) => {
|
||||
const rl = createInterface({
|
||||
input: createReadStream(csvResult.relCsvPath, 'utf-8'),
|
||||
crlfDelay: Infinity,
|
||||
});
|
||||
let isFirst = true;
|
||||
rl.on('line', (line) => {
|
||||
if (isFirst) {
|
||||
relHeader = line;
|
||||
isFirst = false;
|
||||
return;
|
||||
}
|
||||
if (!line.trim()) return;
|
||||
const match = line.match(/"([^"]*)","([^"]*)"/);
|
||||
if (!match) {
|
||||
skippedRels++;
|
||||
return;
|
||||
}
|
||||
const fromLabel = getNodeLabel(match[1]);
|
||||
const toLabel = getNodeLabel(match[2]);
|
||||
if (!validTables.has(fromLabel) || !validTables.has(toLabel)) {
|
||||
skippedRels++;
|
||||
return;
|
||||
}
|
||||
const pairKey = `${fromLabel}|${toLabel}`;
|
||||
let ws = pairWriteStreams.get(pairKey);
|
||||
if (!ws) {
|
||||
const pairCsvPath = path.join(csvDir, `rel_${fromLabel}_${toLabel}.csv`);
|
||||
ws = createWriteStream(pairCsvPath, 'utf-8');
|
||||
ws.write(relHeader + '\n');
|
||||
pairWriteStreams.set(pairKey, ws);
|
||||
relsByPairMeta.set(pairKey, { csvPath: pairCsvPath, rows: 0 });
|
||||
}
|
||||
const ok = ws.write(line + '\n');
|
||||
relsByPairMeta.get(pairKey)!.rows++;
|
||||
totalValidRels++;
|
||||
// Handle backpressure: pause reading when the write buffer is full,
|
||||
// resume when the stream drains. Prevents unbounded memory growth
|
||||
// on repos with millions of relationships.
|
||||
if (!ok) {
|
||||
rl.pause();
|
||||
ws.once('drain', () => rl.resume());
|
||||
}
|
||||
});
|
||||
rl.on('close', resolve);
|
||||
rl.on('error', (err) => {
|
||||
// Destroy all open write streams to avoid resource leaks
|
||||
for (const ws of pairWriteStreams.values()) ws.destroy();
|
||||
reject(err);
|
||||
});
|
||||
});
|
||||
|
||||
// Close all per-pair write streams before COPY
|
||||
// Close all per-pair write streams before COPY. `stream/promises.finished`
|
||||
// resolves on the stream's 'finish' event and rejects on 'error' — replaces
|
||||
// a hand-rolled promisification with the stdlib primitive.
|
||||
await Promise.all(
|
||||
Array.from(pairWriteStreams.values()).map(
|
||||
(ws) =>
|
||||
new Promise<void>((resolve, reject) =>
|
||||
ws.end((err: Error | undefined) => (err ? reject(err) : resolve())),
|
||||
),
|
||||
),
|
||||
Array.from(pairWriteStreams.values()).map(async (ws) => {
|
||||
ws.end();
|
||||
await finished(ws);
|
||||
}),
|
||||
);
|
||||
|
||||
const insertedRels = totalValidRels;
|
||||
@@ -808,18 +884,35 @@ export const getLbugStats = async (): Promise<{ nodes: number; edges: number }>
|
||||
*/
|
||||
export const loadCachedEmbeddings = async (): Promise<{
|
||||
embeddingNodeIds: Set<string>;
|
||||
embeddings: Array<{ nodeId: string; embedding: number[] }>;
|
||||
embeddings: Array<{ nodeId: string; embedding: number[]; contentHash?: string }>;
|
||||
}> => {
|
||||
if (!conn) {
|
||||
return { embeddingNodeIds: new Set(), embeddings: [] };
|
||||
}
|
||||
|
||||
const embeddingNodeIds = new Set<string>();
|
||||
const embeddings: Array<{ nodeId: string; embedding: number[] }> = [];
|
||||
const embeddings: Array<{ nodeId: string; embedding: number[]; contentHash?: string }> = [];
|
||||
try {
|
||||
const rows = await conn.query(
|
||||
`MATCH (e:${EMBEDDING_TABLE_NAME}) RETURN e.nodeId AS nodeId, e.embedding AS embedding`,
|
||||
);
|
||||
// Try to read contentHash alongside the embedding
|
||||
let rows: any;
|
||||
let hasContentHash = true;
|
||||
try {
|
||||
rows = await conn.query(
|
||||
`MATCH (e:${EMBEDDING_TABLE_NAME}) RETURN e.nodeId AS nodeId, e.embedding AS embedding, e.contentHash AS contentHash`,
|
||||
);
|
||||
} catch (err: any) {
|
||||
// Only fall back for missing-column errors (legacy DBs without contentHash).
|
||||
// Rethrow transient / connection errors so callers see them.
|
||||
const msg = err?.message ?? '';
|
||||
if (isMissingColumnOrTableError(msg)) {
|
||||
hasContentHash = false;
|
||||
rows = await conn.query(
|
||||
`MATCH (e:${EMBEDDING_TABLE_NAME}) RETURN e.nodeId AS nodeId, e.embedding AS embedding`,
|
||||
);
|
||||
} else {
|
||||
throw err;
|
||||
}
|
||||
}
|
||||
const result = Array.isArray(rows) ? rows[0] : rows;
|
||||
for (const row of await result.getAll()) {
|
||||
const nodeId = String(row.nodeId ?? row[0] ?? '');
|
||||
@@ -832,6 +925,7 @@ export const loadCachedEmbeddings = async (): Promise<{
|
||||
embedding: Array.isArray(embedding)
|
||||
? embedding.map(Number)
|
||||
: Array.from(embedding as any).map(Number),
|
||||
contentHash: hasContentHash ? (row.contentHash ?? row[2] ?? undefined) : undefined,
|
||||
});
|
||||
}
|
||||
}
|
||||
@@ -842,6 +936,63 @@ export const loadCachedEmbeddings = async (): Promise<{
|
||||
return { embeddingNodeIds, embeddings };
|
||||
};
|
||||
|
||||
/**
|
||||
* Fetch existing embedding hashes from CodeEmbedding table for incremental embedding.
|
||||
* Returns a Map<nodeId, contentHash> suitable for passing to `runEmbeddingPipeline`.
|
||||
* Handles legacy DBs without the `contentHash` column (all rows treated as stale with empty hash).
|
||||
* Returns undefined if the CodeEmbedding table does not exist.
|
||||
*
|
||||
* @param execQuery - Cypher query executor (typically pool-adapter's `executeQuery`)
|
||||
*/
|
||||
export const fetchExistingEmbeddingHashes = async (
|
||||
execQuery: (cypher: string) => Promise<any[]>,
|
||||
): Promise<Map<string, string> | undefined> => {
|
||||
try {
|
||||
const rows = await execQuery(
|
||||
`MATCH (e:${EMBEDDING_TABLE_NAME}) RETURN e.nodeId AS nodeId, e.contentHash AS contentHash`,
|
||||
);
|
||||
if (!rows || rows.length === 0) return undefined;
|
||||
const map = new Map<string, string>();
|
||||
for (const r of rows) {
|
||||
const nodeId = r.nodeId ?? r[0];
|
||||
const hash = r.contentHash ?? r[1] ?? STALE_HASH_SENTINEL;
|
||||
if (nodeId) {
|
||||
// Empty/null contentHash means legacy row — treat as stale so it gets re-embedded
|
||||
map.set(nodeId, hash || STALE_HASH_SENTINEL);
|
||||
}
|
||||
}
|
||||
return map;
|
||||
} catch (err: any) {
|
||||
const msg = err?.message ?? '';
|
||||
if (isMissingColumnOrTableError(msg)) {
|
||||
// Column or table missing — try fallback without contentHash
|
||||
try {
|
||||
const rows = await execQuery(`MATCH (e:${EMBEDDING_TABLE_NAME}) RETURN e.nodeId AS nodeId`);
|
||||
if (!rows || rows.length === 0) return undefined;
|
||||
const map = new Map<string, string>();
|
||||
for (const r of rows) {
|
||||
const nodeId = r.nodeId ?? r[0];
|
||||
if (nodeId) map.set(nodeId, STALE_HASH_SENTINEL); // no contentHash — treat as stale
|
||||
}
|
||||
console.log(
|
||||
`[embed] ${map.size} nodes in legacy DB (no contentHash) — all treated as stale`,
|
||||
);
|
||||
return map;
|
||||
} catch (fallbackErr: any) {
|
||||
const fallbackMsg = fallbackErr?.message ?? '';
|
||||
if (isMissingColumnOrTableError(fallbackMsg)) {
|
||||
console.log(
|
||||
`[embed] CodeEmbedding table not yet present — full embedding run (${fallbackMsg})`,
|
||||
);
|
||||
return undefined;
|
||||
}
|
||||
throw fallbackErr;
|
||||
}
|
||||
}
|
||||
throw err;
|
||||
}
|
||||
};
|
||||
|
||||
export const closeLbug = async (): Promise<void> => {
|
||||
if (conn) {
|
||||
try {
|
||||
|
||||
@@ -436,10 +436,20 @@ if (Number.isNaN(_rawDims) || _rawDims <= 0) {
|
||||
}
|
||||
export const EMBEDDING_DIMS = _rawDims;
|
||||
|
||||
/** HNSW vector index name for the CodeEmbedding table. */
|
||||
export const EMBEDDING_INDEX_NAME = 'code_embedding_idx';
|
||||
|
||||
/**
|
||||
* Sentinel value for "no content hash available" — used in legacy DBs and null rows.
|
||||
* Nodes with this hash are always treated as stale and re-embedded.
|
||||
*/
|
||||
export const STALE_HASH_SENTINEL = '';
|
||||
|
||||
export const EMBEDDING_SCHEMA = `
|
||||
CREATE NODE TABLE ${EMBEDDING_TABLE_NAME} (
|
||||
nodeId STRING,
|
||||
embedding FLOAT[${EMBEDDING_DIMS}],
|
||||
contentHash STRING,
|
||||
PRIMARY KEY (nodeId)
|
||||
)`;
|
||||
|
||||
@@ -448,7 +458,7 @@ CREATE NODE TABLE ${EMBEDDING_TABLE_NAME} (
|
||||
* Uses HNSW (Hierarchical Navigable Small World) algorithm with cosine similarity
|
||||
*/
|
||||
export const CREATE_VECTOR_INDEX_QUERY = `
|
||||
CALL CREATE_VECTOR_INDEX('${EMBEDDING_TABLE_NAME}', 'code_embedding_idx', 'embedding', metric := 'cosine')
|
||||
CALL CREATE_VECTOR_INDEX('${EMBEDDING_TABLE_NAME}', '${EMBEDDING_INDEX_NAME}', 'embedding', metric := 'cosine')
|
||||
`;
|
||||
|
||||
// ============================================================================
|
||||
|
||||
@@ -32,6 +32,8 @@ import {
|
||||
} from '../storage/repo-manager.js';
|
||||
import { getCurrentCommit, hasGitDir } from '../storage/git.js';
|
||||
import { generateAIContextFiles } from '../cli/ai-context.js';
|
||||
import { EMBEDDING_TABLE_NAME } from './lbug/schema.js';
|
||||
import { STALE_HASH_SENTINEL } from './lbug/schema.js';
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Public types
|
||||
@@ -138,7 +140,7 @@ export async function runFullAnalysis(
|
||||
|
||||
// ── Cache embeddings from existing index before rebuild ────────────
|
||||
let cachedEmbeddingNodeIds = new Set<string>();
|
||||
let cachedEmbeddings: Array<{ nodeId: string; embedding: number[] }> = [];
|
||||
let cachedEmbeddings: Array<{ nodeId: string; embedding: number[]; contentHash?: string }> = [];
|
||||
|
||||
if (options.embeddings && existingMeta && !options.force) {
|
||||
try {
|
||||
@@ -219,10 +221,14 @@ export async function runFullAnalysis(
|
||||
const EMBED_BATCH = 200;
|
||||
for (let i = 0; i < cachedEmbeddings.length; i += EMBED_BATCH) {
|
||||
const batch = cachedEmbeddings.slice(i, i + EMBED_BATCH);
|
||||
const paramsList = batch.map((e) => ({ nodeId: e.nodeId, embedding: e.embedding }));
|
||||
const paramsList = batch.map((e) => ({
|
||||
nodeId: e.nodeId,
|
||||
embedding: e.embedding,
|
||||
contentHash: e.contentHash ?? STALE_HASH_SENTINEL,
|
||||
}));
|
||||
try {
|
||||
await executeWithReusedStatement(
|
||||
`CREATE (e:CodeEmbedding {nodeId: $nodeId, embedding: $embedding})`,
|
||||
`MERGE (e:${EMBEDDING_TABLE_NAME} {nodeId: $nodeId}) SET e.embedding = $embedding, e.contentHash = $contentHash`,
|
||||
paramsList,
|
||||
);
|
||||
} catch {
|
||||
@@ -251,6 +257,14 @@ export async function runFullAnalysis(
|
||||
httpMode ? 'Connecting to embedding endpoint...' : 'Loading embedding model...',
|
||||
);
|
||||
const { runEmbeddingPipeline } = await import('./embeddings/embedding-pipeline.js');
|
||||
// Build a Map<nodeId, contentHash> from cached embeddings for incremental mode
|
||||
let existingEmbeddings: Map<string, string> | undefined;
|
||||
if (cachedEmbeddingNodeIds.size > 0) {
|
||||
existingEmbeddings = new Map<string, string>();
|
||||
for (const e of cachedEmbeddings) {
|
||||
existingEmbeddings.set(e.nodeId, e.contentHash ?? STALE_HASH_SENTINEL);
|
||||
}
|
||||
}
|
||||
await runEmbeddingPipeline(
|
||||
executeQuery,
|
||||
executeWithReusedStatement,
|
||||
@@ -265,7 +279,7 @@ export async function runFullAnalysis(
|
||||
progress('embeddings', scaled, label);
|
||||
},
|
||||
{},
|
||||
cachedEmbeddingNodeIds.size > 0 ? cachedEmbeddingNodeIds : undefined,
|
||||
existingEmbeddings,
|
||||
);
|
||||
}
|
||||
|
||||
@@ -275,7 +289,9 @@ export async function runFullAnalysis(
|
||||
// Count embeddings in the index (cached + newly generated)
|
||||
let embeddingCount = 0;
|
||||
try {
|
||||
const embResult = await executeQuery(`MATCH (e:CodeEmbedding) RETURN count(e) AS cnt`);
|
||||
const embResult = await executeQuery(
|
||||
`MATCH (e:${EMBEDDING_TABLE_NAME}) RETURN count(e) AS cnt`,
|
||||
);
|
||||
embeddingCount = embResult?.[0]?.cnt ?? 0;
|
||||
} catch {
|
||||
/* table may not exist if embeddings never ran */
|
||||
|
||||
+34
-19
@@ -1449,25 +1449,40 @@ export const createServer = async (port: number, host: string = '127.0.0.1') =>
|
||||
await withLbugDb(lbugPath, async () => {
|
||||
const { runEmbeddingPipeline } =
|
||||
await import('../core/embeddings/embedding-pipeline.js');
|
||||
await runEmbeddingPipeline(executeQuery, executeWithReusedStatement, (p) => {
|
||||
embedJobManager.updateJob(job.id, {
|
||||
progress: {
|
||||
phase:
|
||||
p.phase === 'ready' ? 'complete' : p.phase === 'error' ? 'failed' : p.phase,
|
||||
percent: p.percent,
|
||||
message:
|
||||
p.phase === 'loading-model'
|
||||
? 'Loading embedding model...'
|
||||
: p.phase === 'embedding'
|
||||
? `Embedding nodes (${p.percent}%)...`
|
||||
: p.phase === 'indexing'
|
||||
? 'Creating vector index...'
|
||||
: p.phase === 'ready'
|
||||
? 'Embeddings complete'
|
||||
: `${p.phase} (${p.percent}%)`,
|
||||
},
|
||||
});
|
||||
});
|
||||
// Fetch existing content hashes for incremental embedding.
|
||||
// Delegated to lbug-adapter which owns the DB query logic and legacy-fallback handling.
|
||||
const { fetchExistingEmbeddingHashes } = await import('../core/lbug/lbug-adapter.js');
|
||||
const existingEmbeddings = await fetchExistingEmbeddingHashes(executeQuery);
|
||||
if (existingEmbeddings && existingEmbeddings.size > 0) {
|
||||
console.log(
|
||||
`[embed] ${existingEmbeddings.size} nodes already embedded — incremental run with content-hash comparison`,
|
||||
);
|
||||
}
|
||||
await runEmbeddingPipeline(
|
||||
executeQuery,
|
||||
executeWithReusedStatement,
|
||||
(p) => {
|
||||
embedJobManager.updateJob(job.id, {
|
||||
progress: {
|
||||
phase:
|
||||
p.phase === 'ready' ? 'complete' : p.phase === 'error' ? 'failed' : p.phase,
|
||||
percent: p.percent,
|
||||
message:
|
||||
p.phase === 'loading-model'
|
||||
? 'Loading embedding model...'
|
||||
: p.phase === 'embedding'
|
||||
? `Embedding nodes (${p.percent}%)...`
|
||||
: p.phase === 'indexing'
|
||||
? 'Creating vector index...'
|
||||
: p.phase === 'ready'
|
||||
? 'Embeddings complete'
|
||||
: `${p.phase} (${p.percent}%)`,
|
||||
},
|
||||
});
|
||||
},
|
||||
{}, // config: use defaults
|
||||
existingEmbeddings,
|
||||
);
|
||||
});
|
||||
|
||||
clearTimeout(embedTimeout);
|
||||
|
||||
@@ -0,0 +1,12 @@
|
||||
#include "server.h"
|
||||
|
||||
void lookupKey(const char *key) {
|
||||
dictEntry *entry = dictFind(key);
|
||||
if (entry) {
|
||||
void *val = entry->val;
|
||||
}
|
||||
}
|
||||
|
||||
void dbGet(const char *key) {
|
||||
void *val = dictFetchValue(key);
|
||||
}
|
||||
@@ -0,0 +1,12 @@
|
||||
#include "dict.h"
|
||||
#include <stdlib.h>
|
||||
|
||||
dictEntry *dictFind(const char *key) {
|
||||
return NULL;
|
||||
}
|
||||
|
||||
void *dictFetchValue(const char *key) {
|
||||
dictEntry *entry = dictFind(key);
|
||||
if (entry) return entry->val;
|
||||
return NULL;
|
||||
}
|
||||
@@ -0,0 +1,12 @@
|
||||
#ifndef DICT_H
|
||||
#define DICT_H
|
||||
|
||||
typedef struct dictEntry {
|
||||
void *key;
|
||||
void *val;
|
||||
} dictEntry;
|
||||
|
||||
dictEntry *dictFind(const char *key);
|
||||
void *dictFetchValue(const char *key);
|
||||
|
||||
#endif
|
||||
@@ -0,0 +1,8 @@
|
||||
#ifndef SERVER_H
|
||||
#define SERVER_H
|
||||
|
||||
#include "dict.h"
|
||||
|
||||
void processCommand(const char *cmd);
|
||||
|
||||
#endif
|
||||
@@ -436,6 +436,42 @@ describe('Phase 9 — Cross-File Call-Result Binding: C++', () => {
|
||||
});
|
||||
});
|
||||
|
||||
describe('Cross-File Call Resolution: pure C transitive #include', () => {
|
||||
let result: PipelineResult;
|
||||
|
||||
beforeAll(async () => {
|
||||
result = await runPipelineFromRepo(path.join(CROSS_FILE_FIXTURES, 'c-cross-file'), () => {});
|
||||
}, 60000);
|
||||
|
||||
it('detects dictFind and dictFetchValue functions', () => {
|
||||
expect(getNodesByLabel(result, 'Function')).toContain('dictFind');
|
||||
expect(getNodesByLabel(result, 'Function')).toContain('dictFetchValue');
|
||||
});
|
||||
|
||||
it('detects lookupKey and dbGet in db.c', () => {
|
||||
expect(getNodesByLabel(result, 'Function')).toContain('lookupKey');
|
||||
expect(getNodesByLabel(result, 'Function')).toContain('dbGet');
|
||||
});
|
||||
|
||||
it('resolves dictFind() call in db.c to dict via transitive header chain', () => {
|
||||
const calls = getRelationships(result, 'CALLS');
|
||||
const crossFileCall = calls.find(
|
||||
(c) =>
|
||||
c.target === 'dictFind' && c.source === 'lookupKey' && c.targetFilePath.includes('dict'),
|
||||
);
|
||||
expect(crossFileCall).toBeDefined();
|
||||
});
|
||||
|
||||
it('resolves dictFetchValue() call in db.c to dict via transitive header chain', () => {
|
||||
const calls = getRelationships(result, 'CALLS');
|
||||
const crossFileCall = calls.find(
|
||||
(c) =>
|
||||
c.target === 'dictFetchValue' && c.source === 'dbGet' && c.targetFilePath.includes('dict'),
|
||||
);
|
||||
expect(crossFileCall).toBeDefined();
|
||||
});
|
||||
});
|
||||
|
||||
describe('Phase 9 — Cross-File Call-Result Binding: C#', () => {
|
||||
let result: PipelineResult;
|
||||
|
||||
|
||||
@@ -0,0 +1,377 @@
|
||||
import { describe, it, expect, vi, beforeEach } from 'vitest';
|
||||
import { createHash } from 'crypto';
|
||||
import { contentHashForNode } from '../../src/core/embeddings/embedding-pipeline.js';
|
||||
import { generateEmbeddingText } from '../../src/core/embeddings/text-generator.js';
|
||||
import type { EmbeddableNode, EmbeddingProgress } from '../../src/core/embeddings/types.js';
|
||||
import { DEFAULT_EMBEDDING_CONFIG } from '../../src/core/embeddings/types.js';
|
||||
import { STALE_HASH_SENTINEL } from '../../src/core/lbug/schema.js';
|
||||
|
||||
// ────────────────────────────────────────────────────────────────────────────
|
||||
// contentHashForNode
|
||||
// ────────────────────────────────────────────────────────────────────────────
|
||||
describe('contentHashForNode', () => {
|
||||
const makeNode = (overrides: Partial<EmbeddableNode> = {}): EmbeddableNode => ({
|
||||
id: 'Function:foo:src/main.ts',
|
||||
name: 'foo',
|
||||
label: 'Function',
|
||||
filePath: 'src/main.ts',
|
||||
content: 'function foo() { return 1; }',
|
||||
...overrides,
|
||||
});
|
||||
|
||||
it('returns a 40-char hex SHA-1 digest', () => {
|
||||
const hash = contentHashForNode(makeNode());
|
||||
expect(hash).toMatch(/^[0-9a-f]{40}$/);
|
||||
});
|
||||
|
||||
it('is deterministic — same node always produces the same hash', () => {
|
||||
const node = makeNode();
|
||||
expect(contentHashForNode(node)).toBe(contentHashForNode(node));
|
||||
});
|
||||
|
||||
it('matches sha1(generateEmbeddingText(node))', () => {
|
||||
const node = makeNode();
|
||||
const expected = createHash('sha1').update(generateEmbeddingText(node)).digest('hex');
|
||||
expect(contentHashForNode(node)).toBe(expected);
|
||||
});
|
||||
|
||||
it('changes when node content is edited', () => {
|
||||
const original = makeNode({ content: 'function foo() { return 1; }' });
|
||||
const edited = makeNode({ content: 'function foo() { return 42; }' });
|
||||
expect(contentHashForNode(original)).not.toBe(contentHashForNode(edited));
|
||||
});
|
||||
|
||||
it('changes when filePath differs', () => {
|
||||
const a = makeNode({ filePath: 'src/a.ts' });
|
||||
const b = makeNode({ filePath: 'src/b.ts' });
|
||||
// Different filePaths lead to different embedding text ⇒ different hashes
|
||||
expect(contentHashForNode(a)).not.toBe(contentHashForNode(b));
|
||||
});
|
||||
|
||||
it('produces identical hash regardless of config vs finalConfig when config is empty', () => {
|
||||
const node = makeNode();
|
||||
const hashWithEmptyConfig = contentHashForNode(node, {});
|
||||
const hashWithFullDefaults = contentHashForNode(node, DEFAULT_EMBEDDING_CONFIG);
|
||||
expect(hashWithEmptyConfig).toBe(hashWithFullDefaults);
|
||||
});
|
||||
});
|
||||
|
||||
// ────────────────────────────────────────────────────────────────────────────
|
||||
// STALE_HASH_SENTINEL
|
||||
// ────────────────────────────────────────────────────────────────────────────
|
||||
describe('STALE_HASH_SENTINEL', () => {
|
||||
it('is the empty string', () => {
|
||||
expect(STALE_HASH_SENTINEL).toBe('');
|
||||
});
|
||||
|
||||
it('is falsy — enables consistent `hash || STALE_HASH_SENTINEL` patterns', () => {
|
||||
expect(!STALE_HASH_SENTINEL).toBe(true);
|
||||
});
|
||||
});
|
||||
|
||||
// ────────────────────────────────────────────────────────────────────────────
|
||||
// runEmbeddingPipeline — exports
|
||||
// ────────────────────────────────────────────────────────────────────────────
|
||||
describe('runEmbeddingPipeline incremental mode', () => {
|
||||
it('exports contentHashForNode as a named export', async () => {
|
||||
const mod = await import('../../src/core/embeddings/embedding-pipeline.js');
|
||||
expect(typeof mod.contentHashForNode).toBe('function');
|
||||
});
|
||||
|
||||
it('exports runEmbeddingPipeline as a named export', async () => {
|
||||
const mod = await import('../../src/core/embeddings/embedding-pipeline.js');
|
||||
expect(typeof mod.runEmbeddingPipeline).toBe('function');
|
||||
});
|
||||
});
|
||||
|
||||
// ────────────────────────────────────────────────────────────────────────────
|
||||
// EMBEDDING_SCHEMA includes contentHash column
|
||||
// ────────────────────────────────────────────────────────────────────────────
|
||||
describe('EMBEDDING_SCHEMA', () => {
|
||||
it('includes contentHash STRING column', async () => {
|
||||
const { EMBEDDING_SCHEMA } = await import('../../src/core/lbug/schema.js');
|
||||
expect(EMBEDDING_SCHEMA).toContain('contentHash STRING');
|
||||
});
|
||||
});
|
||||
|
||||
// ────────────────────────────────────────────────────────────────────────────
|
||||
// EMBEDDING_INDEX_NAME export
|
||||
// ────────────────────────────────────────────────────────────────────────────
|
||||
describe('EMBEDDING_INDEX_NAME', () => {
|
||||
it('is exported from schema.ts', async () => {
|
||||
const { EMBEDDING_INDEX_NAME } = await import('../../src/core/lbug/schema.js');
|
||||
expect(EMBEDDING_INDEX_NAME).toBe('code_embedding_idx');
|
||||
});
|
||||
});
|
||||
|
||||
// ────────────────────────────────────────────────────────────────────────────
|
||||
// runEmbeddingPipeline — incremental filter logic with mocked embedder
|
||||
//
|
||||
// Tests the three incremental-mode code paths:
|
||||
// 1. New node (not in existingEmbeddings) → embedded
|
||||
// 2. Unchanged node (hash matches) → skipped
|
||||
// 3. Stale node (hash mismatch) → DELETE old → re-embed
|
||||
// 4. Zero nodes after filter → createVectorIndex still called
|
||||
// ────────────────────────────────────────────────────────────────────────────
|
||||
describe('runEmbeddingPipeline incremental filter', () => {
|
||||
// Track mocked calls
|
||||
let queryCalls: string[];
|
||||
let stmtCalls: Array<{ cypher: string; params: Array<Record<string, any>> }>;
|
||||
let progressUpdates: EmbeddingProgress[];
|
||||
|
||||
// Helper node
|
||||
const makeNode = (overrides: Partial<EmbeddableNode> = {}): EmbeddableNode => ({
|
||||
id: 'Function:foo:src/main.ts',
|
||||
name: 'foo',
|
||||
label: 'Function',
|
||||
filePath: 'src/main.ts',
|
||||
content: 'function foo() { return 1; }',
|
||||
...overrides,
|
||||
});
|
||||
|
||||
beforeEach(() => {
|
||||
queryCalls = [];
|
||||
stmtCalls = [];
|
||||
progressUpdates = [];
|
||||
vi.restoreAllMocks();
|
||||
vi.resetModules();
|
||||
});
|
||||
|
||||
// Mock the embedder module so we never need a real model
|
||||
const mockEmbedderSetup = () => {
|
||||
vi.doMock('../../src/core/embeddings/embedder.js', () => ({
|
||||
initEmbedder: vi.fn().mockResolvedValue(undefined),
|
||||
embedBatch: vi
|
||||
.fn()
|
||||
.mockImplementation((texts: string[]) =>
|
||||
Promise.resolve(texts.map(() => new Float32Array(384))),
|
||||
),
|
||||
embedText: vi.fn().mockResolvedValue(new Float32Array(384)),
|
||||
embeddingToArray: vi.fn().mockImplementation((emb: Float32Array) => Array.from(emb)),
|
||||
isEmbedderReady: vi.fn().mockReturnValue(true),
|
||||
}));
|
||||
|
||||
// Mock loadVectorExtension (avoids needing the native lbug module)
|
||||
vi.doMock('../../src/core/lbug/lbug-adapter.js', () => ({
|
||||
loadVectorExtension: vi.fn().mockResolvedValue(undefined),
|
||||
}));
|
||||
};
|
||||
|
||||
const mockExecuteQuery = (nodes: EmbeddableNode[]) => {
|
||||
return vi.fn().mockImplementation(async (cypher: string) => {
|
||||
queryCalls.push(cypher);
|
||||
// Respond to node queries based on label
|
||||
for (const label of ['Function', 'Class', 'Method', 'Interface', 'File']) {
|
||||
if (cypher.includes(`MATCH (n:${label})`)) {
|
||||
return nodes
|
||||
.filter((n) => n.label === label)
|
||||
.map((n) => ({
|
||||
id: n.id,
|
||||
name: n.name,
|
||||
label: n.label,
|
||||
filePath: n.filePath,
|
||||
content: n.content,
|
||||
startLine: n.startLine,
|
||||
endLine: n.endLine,
|
||||
}));
|
||||
}
|
||||
}
|
||||
return [];
|
||||
});
|
||||
};
|
||||
|
||||
const mockExecuteWithReusedStatement = () => {
|
||||
return vi
|
||||
.fn()
|
||||
.mockImplementation(async (cypher: string, params: Array<Record<string, any>>) => {
|
||||
stmtCalls.push({ cypher, params });
|
||||
});
|
||||
};
|
||||
|
||||
const onProgress = (p: EmbeddingProgress) => {
|
||||
progressUpdates.push({ ...p });
|
||||
};
|
||||
|
||||
it('skips unchanged nodes when hash matches', async () => {
|
||||
mockEmbedderSetup();
|
||||
|
||||
const node = makeNode();
|
||||
const hash = contentHashForNode(node, DEFAULT_EMBEDDING_CONFIG);
|
||||
const existingEmbeddings = new Map<string, string>([[node.id, hash]]);
|
||||
|
||||
const executeQuery = mockExecuteQuery([node]);
|
||||
const executeWithReusedStatement = mockExecuteWithReusedStatement();
|
||||
|
||||
const { runEmbeddingPipeline } =
|
||||
await import('../../src/core/embeddings/embedding-pipeline.js');
|
||||
|
||||
await runEmbeddingPipeline(
|
||||
executeQuery,
|
||||
executeWithReusedStatement,
|
||||
onProgress,
|
||||
{},
|
||||
existingEmbeddings,
|
||||
);
|
||||
|
||||
// No MERGE calls — node was skipped because hash matched
|
||||
const mergeCalls = stmtCalls.filter((c) => c.cypher.includes('MERGE'));
|
||||
expect(mergeCalls).toHaveLength(0);
|
||||
|
||||
// Pipeline should reach 'ready' state
|
||||
const readyProgress = progressUpdates.find((p) => p.phase === 'ready');
|
||||
expect(readyProgress).toBeDefined();
|
||||
expect(readyProgress!.percent).toBe(100);
|
||||
});
|
||||
|
||||
it('embeds new nodes not in existingEmbeddings', async () => {
|
||||
mockEmbedderSetup();
|
||||
|
||||
const node = makeNode({
|
||||
id: 'Function:newFn:src/new.ts',
|
||||
name: 'newFn',
|
||||
filePath: 'src/new.ts',
|
||||
});
|
||||
const existingEmbeddings = new Map<string, string>(); // empty — no prior embeddings
|
||||
|
||||
const executeQuery = mockExecuteQuery([node]);
|
||||
const executeWithReusedStatement = mockExecuteWithReusedStatement();
|
||||
|
||||
const { runEmbeddingPipeline } =
|
||||
await import('../../src/core/embeddings/embedding-pipeline.js');
|
||||
|
||||
await runEmbeddingPipeline(
|
||||
executeQuery,
|
||||
executeWithReusedStatement,
|
||||
onProgress,
|
||||
{},
|
||||
existingEmbeddings,
|
||||
);
|
||||
|
||||
// Should have a MERGE call to insert the embedding
|
||||
const mergeCalls = stmtCalls.filter((c) => c.cypher.includes('MERGE'));
|
||||
expect(mergeCalls.length).toBeGreaterThanOrEqual(1);
|
||||
|
||||
// The inserted row should contain the node id and a contentHash
|
||||
const insertParams = mergeCalls[0].params;
|
||||
expect(insertParams.some((p: any) => p.nodeId === node.id)).toBe(true);
|
||||
expect(insertParams[0].contentHash).toMatch(/^[0-9a-f]{40}$/);
|
||||
});
|
||||
|
||||
it('deletes and re-embeds stale nodes (hash mismatch)', async () => {
|
||||
mockEmbedderSetup();
|
||||
|
||||
const node = makeNode({ content: 'function foo() { return 42; }' });
|
||||
const staleHash = 'aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa'; // wrong hash
|
||||
const existingEmbeddings = new Map<string, string>([[node.id, staleHash]]);
|
||||
|
||||
const executeQuery = mockExecuteQuery([node]);
|
||||
const executeWithReusedStatement = mockExecuteWithReusedStatement();
|
||||
|
||||
const { runEmbeddingPipeline } =
|
||||
await import('../../src/core/embeddings/embedding-pipeline.js');
|
||||
|
||||
await runEmbeddingPipeline(
|
||||
executeQuery,
|
||||
executeWithReusedStatement,
|
||||
onProgress,
|
||||
{},
|
||||
existingEmbeddings,
|
||||
);
|
||||
|
||||
// Should have a DELETE call for the stale node
|
||||
const deleteCalls = stmtCalls.filter((c) => c.cypher.includes('DELETE'));
|
||||
expect(deleteCalls.length).toBeGreaterThanOrEqual(1);
|
||||
expect(deleteCalls[0].params.some((p: any) => p.nodeId === node.id)).toBe(true);
|
||||
|
||||
// Should also have a MERGE call to re-insert with new hash
|
||||
const mergeCalls = stmtCalls.filter((c) => c.cypher.includes('MERGE'));
|
||||
expect(mergeCalls.length).toBeGreaterThanOrEqual(1);
|
||||
});
|
||||
|
||||
it('treats STALE_HASH_SENTINEL as stale — triggers re-embed', async () => {
|
||||
mockEmbedderSetup();
|
||||
|
||||
const node = makeNode();
|
||||
// Legacy row: nodeId present but contentHash is STALE_HASH_SENTINEL
|
||||
const existingEmbeddings = new Map<string, string>([[node.id, STALE_HASH_SENTINEL]]);
|
||||
|
||||
const executeQuery = mockExecuteQuery([node]);
|
||||
const executeWithReusedStatement = mockExecuteWithReusedStatement();
|
||||
|
||||
const { runEmbeddingPipeline } =
|
||||
await import('../../src/core/embeddings/embedding-pipeline.js');
|
||||
|
||||
await runEmbeddingPipeline(
|
||||
executeQuery,
|
||||
executeWithReusedStatement,
|
||||
onProgress,
|
||||
{},
|
||||
existingEmbeddings,
|
||||
);
|
||||
|
||||
// Should have a DELETE call (stale)
|
||||
const deleteCalls = stmtCalls.filter((c) => c.cypher.includes('DELETE'));
|
||||
expect(deleteCalls.length).toBeGreaterThanOrEqual(1);
|
||||
|
||||
// Should also have a MERGE (re-embed)
|
||||
const mergeCalls = stmtCalls.filter((c) => c.cypher.includes('MERGE'));
|
||||
expect(mergeCalls.length).toBeGreaterThanOrEqual(1);
|
||||
});
|
||||
|
||||
it('calls createVectorIndex even when zero nodes need embedding after filter', async () => {
|
||||
mockEmbedderSetup();
|
||||
|
||||
const node = makeNode();
|
||||
const hash = contentHashForNode(node, DEFAULT_EMBEDDING_CONFIG);
|
||||
// All existing hashes match — zero nodes to embed
|
||||
const existingEmbeddings = new Map<string, string>([[node.id, hash]]);
|
||||
|
||||
const executeQuery = mockExecuteQuery([node]);
|
||||
const executeWithReusedStatement = mockExecuteWithReusedStatement();
|
||||
|
||||
const { runEmbeddingPipeline } =
|
||||
await import('../../src/core/embeddings/embedding-pipeline.js');
|
||||
|
||||
await runEmbeddingPipeline(
|
||||
executeQuery,
|
||||
executeWithReusedStatement,
|
||||
onProgress,
|
||||
{},
|
||||
existingEmbeddings,
|
||||
);
|
||||
|
||||
// The CREATE_VECTOR_INDEX query should have been called via executeQuery
|
||||
const vectorIndexCalls = queryCalls.filter((c) => c.includes('CREATE_VECTOR_INDEX'));
|
||||
expect(vectorIndexCalls.length).toBeGreaterThanOrEqual(1);
|
||||
});
|
||||
|
||||
it('throws when DELETE for stale nodes fails with non-trivial error', async () => {
|
||||
mockEmbedderSetup();
|
||||
|
||||
const node = makeNode({ content: 'function foo() { return 42; }' });
|
||||
const staleHash = 'aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa';
|
||||
const existingEmbeddings = new Map<string, string>([[node.id, staleHash]]);
|
||||
|
||||
const executeQuery = mockExecuteQuery([node]);
|
||||
const executeWithReusedStatement = vi.fn().mockRejectedValue(new Error('Connection lost'));
|
||||
|
||||
const { runEmbeddingPipeline } =
|
||||
await import('../../src/core/embeddings/embedding-pipeline.js');
|
||||
|
||||
await expect(
|
||||
runEmbeddingPipeline(
|
||||
executeQuery,
|
||||
executeWithReusedStatement,
|
||||
onProgress,
|
||||
{},
|
||||
existingEmbeddings,
|
||||
),
|
||||
).rejects.toThrow('vector-index corruption');
|
||||
});
|
||||
});
|
||||
|
||||
// ────────────────────────────────────────────────────────────────────────────
|
||||
// fetchExistingEmbeddingHashes — tested in integration tests (requires native module)
|
||||
// The function is tested via lbug-core-adapter integration tests which have the
|
||||
// native @ladybugdb/core module available.
|
||||
// ────────────────────────────────────────────────────────────────────────────
|
||||
@@ -1,4 +1,4 @@
|
||||
import { describe, it, expect, beforeEach, afterEach } from 'vitest';
|
||||
import { describe, it, expect, beforeEach, afterEach, vi } from 'vitest';
|
||||
import * as fs from 'node:fs';
|
||||
import fsp from 'node:fs/promises';
|
||||
import * as path from 'node:path';
|
||||
@@ -732,6 +732,120 @@ describe('resolveProtoConflict', () => {
|
||||
it('test_no_candidates_returns_null', () => {
|
||||
expect(resolveProtoConflict('Svc', 'src/main.go', [])).toBeNull();
|
||||
});
|
||||
|
||||
it('test_all_zero_tie_returns_null', () => {
|
||||
const warnSpy = vi.spyOn(console, 'warn').mockImplementation(() => {});
|
||||
const candidates = [
|
||||
makeInfo('pkgA', 'totally/unrelated/a/svc.proto'),
|
||||
makeInfo('pkgB', 'completely/different/b/svc.proto'),
|
||||
];
|
||||
const result = resolveProtoConflict('Svc', 'src/main.go', candidates);
|
||||
expect(result).toBeNull();
|
||||
warnSpy.mockRestore();
|
||||
});
|
||||
|
||||
it('test_positive_score_tie_returns_null', () => {
|
||||
const warnSpy = vi.spyOn(console, 'warn').mockImplementation(() => {});
|
||||
// Both candidates share `src/proto` with the source dir — equal shared runs.
|
||||
const candidates = [
|
||||
makeInfo('pkgA', 'src/proto/a/svc.proto'),
|
||||
makeInfo('pkgB', 'src/proto/b/svc.proto'),
|
||||
];
|
||||
const result = resolveProtoConflict('Svc', 'src/proto/main.go', candidates);
|
||||
expect(result).toBeNull();
|
||||
warnSpy.mockRestore();
|
||||
});
|
||||
|
||||
it('test_three_way_zero_tie_returns_null', () => {
|
||||
const warnSpy = vi.spyOn(console, 'warn').mockImplementation(() => {});
|
||||
const candidates = [
|
||||
makeInfo('pkgA', 'aaa/svc.proto'),
|
||||
makeInfo('pkgB', 'bbb/svc.proto'),
|
||||
makeInfo('pkgC', 'ccc/svc.proto'),
|
||||
];
|
||||
const result = resolveProtoConflict('Svc', 'src/main.go', candidates);
|
||||
expect(result).toBeNull();
|
||||
warnSpy.mockRestore();
|
||||
});
|
||||
|
||||
it('test_unique_winner_among_ties', () => {
|
||||
// Winner with shared run 2 (services/auth), two losers with score 0.
|
||||
const candidates = [
|
||||
makeInfo('winner', 'services/auth/proto/svc.proto'),
|
||||
makeInfo('loserA', 'totally/unrelated/a/svc.proto'),
|
||||
makeInfo('loserB', 'elsewhere/b/svc.proto'),
|
||||
];
|
||||
const result = resolveProtoConflict('Svc', 'services/auth/src/server.ts', candidates);
|
||||
expect(result?.package).toBe('winner');
|
||||
});
|
||||
|
||||
it('test_ambiguous_emits_single_warn_with_service_and_paths', () => {
|
||||
const warnSpy = vi.spyOn(console, 'warn').mockImplementation(() => {});
|
||||
const candidates = [
|
||||
makeInfo('pkgA', 'totally/unrelated/a/svc.proto'),
|
||||
makeInfo('pkgB', 'completely/different/b/svc.proto'),
|
||||
];
|
||||
resolveProtoConflict('MyService', 'src/main.go', candidates);
|
||||
expect(warnSpy).toHaveBeenCalledTimes(1);
|
||||
const msg = String(warnSpy.mock.calls[0][0]);
|
||||
expect(msg).toContain('MyService');
|
||||
expect(msg).toContain('src/main.go');
|
||||
expect(msg).toContain('totally/unrelated/a/svc.proto');
|
||||
expect(msg).toContain('completely/different/b/svc.proto');
|
||||
warnSpy.mockRestore();
|
||||
});
|
||||
});
|
||||
|
||||
describe('GrpcExtractor.extract ambiguous proto resolution', () => {
|
||||
let tmpDir: string;
|
||||
let extractor: GrpcExtractor;
|
||||
|
||||
beforeEach(async () => {
|
||||
tmpDir = await fsp.mkdtemp(path.join(os.tmpdir(), 'gitnexus-grpc-ambig-'));
|
||||
extractor = new GrpcExtractor();
|
||||
});
|
||||
afterEach(async () => {
|
||||
await fsp.rm(tmpDir, { recursive: true, force: true });
|
||||
});
|
||||
|
||||
const makeRepo = (repoPath: string): RepoHandle => ({
|
||||
id: 'test-repo',
|
||||
path: '',
|
||||
repoPath,
|
||||
storagePath: '',
|
||||
});
|
||||
|
||||
it('test_ambiguous_short_name_across_unrelated_protos_yields_no_source_contract', async () => {
|
||||
const warnSpy = vi.spyOn(console, 'warn').mockImplementation(() => {});
|
||||
// Two unrelated proto files defining the same short name `UserService` in
|
||||
// unrelated directories, neither sharing path segments with the Go source.
|
||||
await fsp.mkdir(path.join(tmpDir, 'billing-team', 'proto'), { recursive: true });
|
||||
await fsp.writeFile(
|
||||
path.join(tmpDir, 'billing-team', 'proto', 'user.proto'),
|
||||
'package billing.v1;\nservice UserService { rpc GetUser (R) returns (R); }',
|
||||
);
|
||||
await fsp.mkdir(path.join(tmpDir, 'auth-team', 'proto'), { recursive: true });
|
||||
await fsp.writeFile(
|
||||
path.join(tmpDir, 'auth-team', 'proto', 'user.proto'),
|
||||
'package auth.v1;\nservice UserService { rpc GetUser (R) returns (R); }',
|
||||
);
|
||||
// Consumer in an unrelated directory.
|
||||
await fsp.mkdir(path.join(tmpDir, 'apps', 'gateway'), { recursive: true });
|
||||
await fsp.writeFile(
|
||||
path.join(tmpDir, 'apps', 'gateway', 'client.go'),
|
||||
'package main\nfunc init() { client := pb.NewUserServiceClient(conn) }',
|
||||
);
|
||||
|
||||
const contracts = await extractor.extract(null, tmpDir, makeRepo(tmpDir));
|
||||
|
||||
// No source-attributed contract for UserService should be emitted.
|
||||
const sourceContracts = contracts.filter(
|
||||
(c) => c.meta.source === 'go_client' && c.meta.service === 'UserService',
|
||||
);
|
||||
expect(sourceContracts).toHaveLength(0);
|
||||
expect(warnSpy).toHaveBeenCalled();
|
||||
warnSpy.mockRestore();
|
||||
});
|
||||
});
|
||||
|
||||
describe('serviceContractId', () => {
|
||||
|
||||
@@ -0,0 +1,393 @@
|
||||
/**
|
||||
* Coverage tests for `HttpRouteExtractor` graph-assisted paths —
|
||||
* specifically the multi-verb same-path regression (Codex finding F2).
|
||||
*
|
||||
* The bug: `extractProvidersGraph` / `extractConsumersGraph` used
|
||||
* `detections.find(d => normalizeHttpPath(d.path) === routePath)` to
|
||||
* backfill handler name and (for providers) method. On a file with
|
||||
* multiple verbs at the same normalized path (e.g. `GET /api/orders`
|
||||
* and `POST /api/orders` in one router), `.find()` returned the first
|
||||
* match, silently attaching the wrong handler and/or method.
|
||||
*
|
||||
* Strategy: mock `./http-patterns/index.js` + `./fs-utils.js` so we
|
||||
* can inject a synthetic `HttpDetection[]` per file without needing
|
||||
* real tree-sitter grammars. The `db` executor is a vi.fn() that
|
||||
* returns stubbed rows for the HANDLES_ROUTE / FETCHES / CONTAINS
|
||||
* queries.
|
||||
*/
|
||||
import { describe, it, expect, vi, beforeEach } from 'vitest';
|
||||
import type Parser from 'tree-sitter';
|
||||
import type { HttpDetection } from '../../../src/core/group/extractors/http-patterns/types.js';
|
||||
|
||||
// Per-file detections injected into the mocked plugin.
|
||||
const FILE_DETECTIONS = new Map<string, HttpDetection[]>();
|
||||
|
||||
vi.mock('../../../src/core/group/extractors/fs-utils.js', () => ({
|
||||
readSafe: (_repo: string, _rel: string) => 'stub content',
|
||||
}));
|
||||
|
||||
vi.mock('../../../src/core/group/extractors/http-patterns/index.js', () => {
|
||||
return {
|
||||
HTTP_SCAN_GLOB: '**/*.fake',
|
||||
getPluginForFile: (rel: string) => ({
|
||||
name: 'fake',
|
||||
language: {},
|
||||
scan: (_tree: Parser.Tree) => FILE_DETECTIONS.get(rel) ?? [],
|
||||
}),
|
||||
};
|
||||
});
|
||||
|
||||
// Patch tree-sitter Parser so `.setLanguage()` + `.parse()` don't
|
||||
// require a real grammar — the mocked plugin's scan() ignores the
|
||||
// tree anyway.
|
||||
vi.mock('tree-sitter', () => {
|
||||
class FakeParser {
|
||||
setLanguage(_lang: unknown) {}
|
||||
parse(_src: string) {
|
||||
return {} as Parser.Tree;
|
||||
}
|
||||
}
|
||||
return { default: FakeParser };
|
||||
});
|
||||
|
||||
import { HttpRouteExtractor } from '../../../src/core/group/extractors/http-route-extractor.js';
|
||||
|
||||
function detection(
|
||||
role: 'provider' | 'consumer',
|
||||
method: string,
|
||||
p: string,
|
||||
name: string | null,
|
||||
): HttpDetection {
|
||||
return { role, framework: 'test', method, path: p, name, confidence: 0.8 };
|
||||
}
|
||||
|
||||
describe('HttpRouteExtractor — graph-assisted multi-verb disambiguation', () => {
|
||||
beforeEach(() => {
|
||||
FILE_DETECTIONS.clear();
|
||||
});
|
||||
|
||||
// Helper to build a CONTAINS response covering all handler names in a file.
|
||||
const containsFor = (names: string[]) =>
|
||||
names.map((name, i) => ({
|
||||
uid: `uid-${name}`,
|
||||
name,
|
||||
filePath: 'routes.ts',
|
||||
labels: ['Function'],
|
||||
0: `uid-${name}`,
|
||||
1: name,
|
||||
2: 'routes.ts',
|
||||
3: ['Function'],
|
||||
}));
|
||||
|
||||
// ── Provider: happy path (single match) ────────────────────────────
|
||||
it('provider: single detection backfills handler name as today', async () => {
|
||||
FILE_DETECTIONS.set('routes.ts', [detection('provider', 'GET', '/api/orders', 'listOrders')]);
|
||||
|
||||
const db = vi.fn(async (query: string) => {
|
||||
if (query.includes('HANDLES_ROUTE')) {
|
||||
return [
|
||||
{
|
||||
fileId: 'f1',
|
||||
filePath: 'routes.ts',
|
||||
routePath: '/api/orders',
|
||||
routeId: 'r1',
|
||||
responseKeys: [],
|
||||
routeSource: 'decorator-Get',
|
||||
},
|
||||
];
|
||||
}
|
||||
if (query.includes('CONTAINS')) return containsFor(['listOrders']);
|
||||
return [];
|
||||
});
|
||||
|
||||
const ex = new HttpRouteExtractor();
|
||||
const out = await ex.extract(db, '/repo', { name: 'r', url: 'r' } as never);
|
||||
expect(out).toHaveLength(1);
|
||||
expect(out[0].symbolName).toBe('listOrders');
|
||||
expect(out[0].meta.method).toBe('GET');
|
||||
});
|
||||
|
||||
// ── Provider: multi-verb, method KNOWN (POST) ──────────────────────
|
||||
it('provider: multi-verb with method known picks the matching verb (POST)', async () => {
|
||||
FILE_DETECTIONS.set('routes.ts', [
|
||||
detection('provider', 'GET', '/api/orders', 'listOrders'),
|
||||
detection('provider', 'POST', '/api/orders', 'createOrder'),
|
||||
]);
|
||||
|
||||
const db = vi.fn(async (query: string) => {
|
||||
if (query.includes('HANDLES_ROUTE')) {
|
||||
return [
|
||||
{
|
||||
fileId: 'f1',
|
||||
filePath: 'routes.ts',
|
||||
routePath: '/api/orders',
|
||||
routeSource: 'decorator-Post',
|
||||
},
|
||||
];
|
||||
}
|
||||
if (query.includes('CONTAINS')) return containsFor(['listOrders', 'createOrder']);
|
||||
return [];
|
||||
});
|
||||
|
||||
const out = await new HttpRouteExtractor().extract(db, '/repo', {
|
||||
name: 'r',
|
||||
url: 'r',
|
||||
} as never);
|
||||
expect(out).toHaveLength(1);
|
||||
expect(out[0].symbolName).toBe('createOrder');
|
||||
expect(out[0].meta.method).toBe('POST');
|
||||
});
|
||||
|
||||
it('provider: multi-verb with method known picks the matching verb (GET)', async () => {
|
||||
FILE_DETECTIONS.set('routes.ts', [
|
||||
detection('provider', 'GET', '/api/orders', 'listOrders'),
|
||||
detection('provider', 'POST', '/api/orders', 'createOrder'),
|
||||
]);
|
||||
|
||||
const db = vi.fn(async (query: string) => {
|
||||
if (query.includes('HANDLES_ROUTE')) {
|
||||
return [
|
||||
{
|
||||
fileId: 'f1',
|
||||
filePath: 'routes.ts',
|
||||
routePath: '/api/orders',
|
||||
routeSource: 'decorator-Get',
|
||||
},
|
||||
];
|
||||
}
|
||||
if (query.includes('CONTAINS')) return containsFor(['listOrders', 'createOrder']);
|
||||
return [];
|
||||
});
|
||||
|
||||
const out = await new HttpRouteExtractor().extract(db, '/repo', {
|
||||
name: 'r',
|
||||
url: 'r',
|
||||
} as never);
|
||||
expect(out).toHaveLength(1);
|
||||
expect(out[0].symbolName).toBe('listOrders');
|
||||
expect(out[0].meta.method).toBe('GET');
|
||||
});
|
||||
|
||||
// ── Provider: multi-verb, method UNKNOWN → refuse to guess ─────────
|
||||
it('provider: multi-verb with method unknown skips backfill (no silent inheritance)', async () => {
|
||||
FILE_DETECTIONS.set('routes.ts', [
|
||||
detection('provider', 'GET', '/api/orders', 'listOrders'),
|
||||
detection('provider', 'POST', '/api/orders', 'createOrder'),
|
||||
]);
|
||||
|
||||
const db = vi.fn(async (query: string) => {
|
||||
if (query.includes('HANDLES_ROUTE')) {
|
||||
return [
|
||||
{
|
||||
fileId: 'f1',
|
||||
filePath: 'routes.ts',
|
||||
routePath: '/api/orders',
|
||||
routeSource: 'unknown-reason', // methodFromRouteReason → null
|
||||
},
|
||||
];
|
||||
}
|
||||
return [];
|
||||
});
|
||||
|
||||
const out = await new HttpRouteExtractor().extract(db, '/repo', {
|
||||
name: 'r',
|
||||
url: 'r',
|
||||
} as never);
|
||||
expect(out).toHaveLength(1);
|
||||
// CRITICAL: must NOT silently inherit POST from createOrder via .find()
|
||||
expect(out[0].meta.method).toBe('GET'); // conservative default
|
||||
// CRITICAL: must NOT silently attach createOrder as handler
|
||||
expect(out[0].symbolName).not.toBe('createOrder');
|
||||
// With no CONTAINS rows, handlerName stays null and file-basename fallback wins.
|
||||
expect(out[0].symbolName).toBe('routes.ts');
|
||||
});
|
||||
|
||||
// ── Provider: multi-verb + CONTAINS rows → must still refuse to guess ──
|
||||
it('provider: ambiguous multi-verb skips CONTAINS enrichment (no silent pool[0] pick)', async () => {
|
||||
// Regression test for Copilot's review on PR #817. Before the fix,
|
||||
// the ambiguous-case code path left `handlerName` null but still ran
|
||||
// the CONTAINS DB query, and `pickSymbolUid(syms, null)` silently
|
||||
// picked pool[0] — reintroducing handler mis-attribution via a
|
||||
// different route than `.find()`.
|
||||
FILE_DETECTIONS.set('routes.ts', [
|
||||
detection('provider', 'GET', '/api/orders', 'listOrders'),
|
||||
detection('provider', 'POST', '/api/orders', 'createOrder'),
|
||||
]);
|
||||
|
||||
const db = vi.fn(async (query: string) => {
|
||||
if (query.includes('HANDLES_ROUTE')) {
|
||||
return [
|
||||
{
|
||||
fileId: 'f1',
|
||||
filePath: 'routes.ts',
|
||||
routePath: '/api/orders',
|
||||
routeSource: 'unknown-reason',
|
||||
},
|
||||
];
|
||||
}
|
||||
if (query.includes('CONTAINS')) return containsFor(['listOrders', 'createOrder']);
|
||||
return [];
|
||||
});
|
||||
|
||||
const out = await new HttpRouteExtractor().extract(db, '/repo', {
|
||||
name: 'r',
|
||||
url: 'r',
|
||||
} as never);
|
||||
expect(out).toHaveLength(1);
|
||||
// Ambiguous → do not attribute to any real handler in the file.
|
||||
expect(out[0].symbolName).not.toBe('listOrders');
|
||||
expect(out[0].symbolName).not.toBe('createOrder');
|
||||
expect(out[0].symbolUid).toBe('');
|
||||
expect(out[0].symbolName).toBe('routes.ts');
|
||||
expect(out[0].meta.method).toBe('GET');
|
||||
// CONTAINS query must have been skipped entirely under ambiguity.
|
||||
const calls = db.mock.calls.map(([q]) => q as string);
|
||||
expect(calls.some((q) => q.includes('CONTAINS'))).toBe(false);
|
||||
});
|
||||
|
||||
// ── Provider: three-verb method known ──────────────────────────────
|
||||
it('provider: three verbs at same path with method known still matches correctly', async () => {
|
||||
FILE_DETECTIONS.set('routes.ts', [
|
||||
detection('provider', 'GET', '/api/orders', 'listOrders'),
|
||||
detection('provider', 'POST', '/api/orders', 'createOrder'),
|
||||
detection('provider', 'PUT', '/api/orders', 'replaceOrder'),
|
||||
]);
|
||||
|
||||
const db = vi.fn(async (query: string) => {
|
||||
if (query.includes('HANDLES_ROUTE')) {
|
||||
return [
|
||||
{
|
||||
fileId: 'f1',
|
||||
filePath: 'routes.ts',
|
||||
routePath: '/api/orders',
|
||||
routeSource: 'decorator-Put',
|
||||
},
|
||||
];
|
||||
}
|
||||
if (query.includes('CONTAINS'))
|
||||
return containsFor(['listOrders', 'createOrder', 'replaceOrder']);
|
||||
return [];
|
||||
});
|
||||
|
||||
const out = await new HttpRouteExtractor().extract(db, '/repo', {
|
||||
name: 'r',
|
||||
url: 'r',
|
||||
} as never);
|
||||
expect(out[0].symbolName).toBe('replaceOrder');
|
||||
expect(out[0].meta.method).toBe('PUT');
|
||||
});
|
||||
|
||||
// ── Provider: unrelated path detections don't false-positive ───────
|
||||
it('provider: detection for unrelated path does not backfill', async () => {
|
||||
FILE_DETECTIONS.set('routes.ts', [detection('provider', 'POST', '/api/users', 'createUser')]);
|
||||
|
||||
const db = vi.fn(async (query: string) => {
|
||||
if (query.includes('HANDLES_ROUTE')) {
|
||||
return [
|
||||
{
|
||||
fileId: 'f1',
|
||||
filePath: 'routes.ts',
|
||||
routePath: '/api/orders',
|
||||
routeSource: 'unknown',
|
||||
},
|
||||
];
|
||||
}
|
||||
return [];
|
||||
});
|
||||
|
||||
const out = await new HttpRouteExtractor().extract(db, '/repo', {
|
||||
name: 'r',
|
||||
url: 'r',
|
||||
} as never);
|
||||
expect(out[0].meta.method).toBe('GET');
|
||||
expect(out[0].symbolName).not.toBe('createUser');
|
||||
});
|
||||
|
||||
// ── Integration: one row, two detections, one out.push ─────────────
|
||||
it('integration: one db row with two same-path detections yields exactly one contract', async () => {
|
||||
FILE_DETECTIONS.set('routes.ts', [
|
||||
detection('provider', 'GET', '/api/orders', 'listOrders'),
|
||||
detection('provider', 'POST', '/api/orders', 'createOrder'),
|
||||
]);
|
||||
|
||||
const db = vi.fn(async (query: string) => {
|
||||
if (query.includes('HANDLES_ROUTE')) {
|
||||
return [
|
||||
{
|
||||
fileId: 'f1',
|
||||
filePath: 'routes.ts',
|
||||
routePath: '/api/orders',
|
||||
routeSource: 'decorator-Post',
|
||||
},
|
||||
];
|
||||
}
|
||||
if (query.includes('CONTAINS')) return containsFor(['listOrders', 'createOrder']);
|
||||
return [];
|
||||
});
|
||||
|
||||
const out = await new HttpRouteExtractor().extract(db, '/repo', {
|
||||
name: 'r',
|
||||
url: 'r',
|
||||
} as never);
|
||||
expect(out).toHaveLength(1);
|
||||
expect(out[0].meta.method).toBe('POST');
|
||||
expect(out[0].symbolName).toBe('createOrder');
|
||||
});
|
||||
|
||||
// ── Consumer: single match ─────────────────────────────────────────
|
||||
it('consumer: single detection backfills method as today', async () => {
|
||||
FILE_DETECTIONS.set('client.ts', [detection('consumer', 'POST', '/api/orders', null)]);
|
||||
|
||||
const db = vi.fn(async (query: string) => {
|
||||
if (query.includes('FETCHES')) {
|
||||
return [
|
||||
{
|
||||
fileId: 'f1',
|
||||
filePath: 'client.ts',
|
||||
routePath: '/api/orders',
|
||||
fetchReason: 'fetch',
|
||||
},
|
||||
];
|
||||
}
|
||||
return [];
|
||||
});
|
||||
|
||||
const out = await new HttpRouteExtractor().extract(db, '/repo', {
|
||||
name: 'r',
|
||||
url: 'r',
|
||||
} as never);
|
||||
expect(out).toHaveLength(1);
|
||||
expect(out[0].meta.method).toBe('POST');
|
||||
});
|
||||
|
||||
// ── Consumer: multi-verb skips backfill ────────────────────────────
|
||||
it('consumer: multi-verb at same path skips backfill (conservative GET)', async () => {
|
||||
FILE_DETECTIONS.set('client.ts', [
|
||||
detection('consumer', 'GET', '/api/orders', null),
|
||||
detection('consumer', 'POST', '/api/orders', null),
|
||||
]);
|
||||
|
||||
const db = vi.fn(async (query: string) => {
|
||||
if (query.includes('FETCHES')) {
|
||||
return [
|
||||
{
|
||||
fileId: 'f1',
|
||||
filePath: 'client.ts',
|
||||
routePath: '/api/orders',
|
||||
fetchReason: 'fetch',
|
||||
},
|
||||
];
|
||||
}
|
||||
return [];
|
||||
});
|
||||
|
||||
const out = await new HttpRouteExtractor().extract(db, '/repo', {
|
||||
name: 'r',
|
||||
url: 'r',
|
||||
} as never);
|
||||
expect(out).toHaveLength(1);
|
||||
// CRITICAL: must NOT silently pick POST (first/last via .find)
|
||||
expect(out[0].meta.method).toBe('GET'); // conservative default
|
||||
expect(out[0].contractId).toBe('http::GET::/api/orders');
|
||||
});
|
||||
});
|
||||
@@ -300,9 +300,317 @@ describe('ManifestExtractor', () => {
|
||||
}
|
||||
});
|
||||
|
||||
it('resolves http contract with explicit METHOD prefix (GET::/api/orders)', async () => {
|
||||
// Regression test for Codex finding F1: resolveSymbol was passing the
|
||||
// raw `link.contract` through normalizeRoutePath, which turned
|
||||
// "GET::/api/orders" into "/GET::/api/orders" and never matched
|
||||
// Route.name = "/api/orders". The extractor must strip the METHOD::
|
||||
// prefix and pass only the path portion to the Cypher executor.
|
||||
const links: GroupManifestLink[] = [
|
||||
{
|
||||
from: 'gateway',
|
||||
to: 'orders-svc',
|
||||
type: 'http',
|
||||
contract: 'GET::/api/orders',
|
||||
role: 'consumer',
|
||||
},
|
||||
];
|
||||
|
||||
let seenParam: string | undefined;
|
||||
const dbExecutors = new Map<
|
||||
string,
|
||||
(cypher: string, params?: Record<string, unknown>) => Promise<Record<string, unknown>[]>
|
||||
>([
|
||||
[
|
||||
'orders-svc',
|
||||
async (_cypher, params) => {
|
||||
seenParam = params?.normalized as string;
|
||||
if (seenParam === '/api/orders') {
|
||||
return [
|
||||
{
|
||||
uid: 'uid-orders-list',
|
||||
name: 'listOrders',
|
||||
filePath: 'src/orders.ts',
|
||||
},
|
||||
];
|
||||
}
|
||||
return [];
|
||||
},
|
||||
],
|
||||
['gateway', async () => []],
|
||||
]);
|
||||
|
||||
const result = await extractor.extractFromManifest(links, dbExecutors);
|
||||
|
||||
// The key assertion: $normalized must be the path only, NOT "/GET::/api/orders".
|
||||
expect(seenParam).toBe('/api/orders');
|
||||
|
||||
const provider = result.contracts.find((c) => c.role === 'provider');
|
||||
expect(provider?.symbolUid).toBe('uid-orders-list');
|
||||
expect(provider?.symbolRef.filePath).toBe('src/orders.ts');
|
||||
});
|
||||
|
||||
it('resolves http contract with parameterised path (POST::/users/:id)', async () => {
|
||||
const links: GroupManifestLink[] = [
|
||||
{
|
||||
from: 'gateway',
|
||||
to: 'users-svc',
|
||||
type: 'http',
|
||||
contract: 'POST::/users/:id',
|
||||
role: 'consumer',
|
||||
},
|
||||
];
|
||||
|
||||
let seenParam: string | undefined;
|
||||
const dbExecutors = new Map<
|
||||
string,
|
||||
(cypher: string, params?: Record<string, unknown>) => Promise<Record<string, unknown>[]>
|
||||
>([
|
||||
[
|
||||
'users-svc',
|
||||
async (_cypher, params) => {
|
||||
seenParam = params?.normalized as string;
|
||||
if (seenParam === '/users/:id') {
|
||||
return [
|
||||
{
|
||||
uid: 'uid-update-user',
|
||||
name: 'updateUser',
|
||||
filePath: 'src/users.ts',
|
||||
},
|
||||
];
|
||||
}
|
||||
return [];
|
||||
},
|
||||
],
|
||||
['gateway', async () => []],
|
||||
]);
|
||||
|
||||
const result = await extractor.extractFromManifest(links, dbExecutors);
|
||||
expect(seenParam).toBe('/users/:id');
|
||||
const provider = result.contracts.find((c) => c.role === 'provider');
|
||||
expect(provider?.symbolUid).toBe('uid-update-user');
|
||||
});
|
||||
|
||||
it('handles http contract with empty path after METHOD:: (GET::)', async () => {
|
||||
// Edge case: "GET::" (empty path after prefix). Normalizer produces "/"
|
||||
// — either resolves to a root route or returns null cleanly.
|
||||
const links: GroupManifestLink[] = [
|
||||
{
|
||||
from: 'gateway',
|
||||
to: 'orders-svc',
|
||||
type: 'http',
|
||||
contract: 'GET::',
|
||||
role: 'consumer',
|
||||
},
|
||||
];
|
||||
|
||||
let seenParam: string | undefined;
|
||||
const dbExecutors = new Map<
|
||||
string,
|
||||
(cypher: string, params?: Record<string, unknown>) => Promise<Record<string, unknown>[]>
|
||||
>([
|
||||
[
|
||||
'orders-svc',
|
||||
async (_cypher, params) => {
|
||||
seenParam = params?.normalized as string;
|
||||
return [];
|
||||
},
|
||||
],
|
||||
['gateway', async () => []],
|
||||
]);
|
||||
|
||||
const result = await extractor.extractFromManifest(links, dbExecutors);
|
||||
expect(seenParam).toBe('/');
|
||||
// No match → synthetic uid, no crash.
|
||||
const provider = result.contracts.find((c) => c.role === 'provider');
|
||||
// buildContractId canonicalizes the empty path to `/` so contract ids
|
||||
// match regardless of trailing-slash variants in the manifest input.
|
||||
expect(provider?.symbolUid).toBe('manifest::orders-svc::http::GET::/');
|
||||
});
|
||||
|
||||
it('treats empty method portion (::/api/orders) as a bare path', async () => {
|
||||
const links: GroupManifestLink[] = [
|
||||
{
|
||||
from: 'gateway',
|
||||
to: 'orders-svc',
|
||||
type: 'http',
|
||||
contract: '::/api/orders',
|
||||
role: 'consumer',
|
||||
},
|
||||
];
|
||||
|
||||
let seenParam: string | undefined;
|
||||
const dbExecutors = new Map<
|
||||
string,
|
||||
(cypher: string, params?: Record<string, unknown>) => Promise<Record<string, unknown>[]>
|
||||
>([
|
||||
[
|
||||
'orders-svc',
|
||||
async (_cypher, params) => {
|
||||
seenParam = params?.normalized as string;
|
||||
return [];
|
||||
},
|
||||
],
|
||||
['gateway', async () => []],
|
||||
]);
|
||||
|
||||
await extractor.extractFromManifest(links, dbExecutors);
|
||||
// "::/api/orders" has no method prefix per buildContractId's regex
|
||||
// (`[A-Za-z]+::`), so the whole string is treated as a bare path.
|
||||
// Normalizer collapses leading slashes, so "::/api/orders" stays
|
||||
// essentially as-is (no alpha prefix match).
|
||||
expect(seenParam).toBe('/::/api/orders');
|
||||
});
|
||||
|
||||
it('resolves http contract with lowercase verb (get::/api/orders)', async () => {
|
||||
const links: GroupManifestLink[] = [
|
||||
{
|
||||
from: 'gateway',
|
||||
to: 'orders-svc',
|
||||
type: 'http',
|
||||
contract: 'get::/api/orders',
|
||||
role: 'consumer',
|
||||
},
|
||||
];
|
||||
|
||||
let seenParam: string | undefined;
|
||||
const dbExecutors = new Map<
|
||||
string,
|
||||
(cypher: string, params?: Record<string, unknown>) => Promise<Record<string, unknown>[]>
|
||||
>([
|
||||
[
|
||||
'orders-svc',
|
||||
async (_cypher, params) => {
|
||||
seenParam = params?.normalized as string;
|
||||
if (seenParam === '/api/orders') {
|
||||
return [
|
||||
{
|
||||
uid: 'uid-orders-list',
|
||||
name: 'listOrders',
|
||||
filePath: 'src/orders.ts',
|
||||
},
|
||||
];
|
||||
}
|
||||
return [];
|
||||
},
|
||||
],
|
||||
['gateway', async () => []],
|
||||
]);
|
||||
|
||||
const result = await extractor.extractFromManifest(links, dbExecutors);
|
||||
expect(seenParam).toBe('/api/orders');
|
||||
const provider = result.contracts.find((c) => c.role === 'provider');
|
||||
expect(provider?.symbolUid).toBe('uid-orders-list');
|
||||
});
|
||||
|
||||
it('returns null cleanly when no Route matches explicit-method http contract', async () => {
|
||||
const links: GroupManifestLink[] = [
|
||||
{
|
||||
from: 'gateway',
|
||||
to: 'orders-svc',
|
||||
type: 'http',
|
||||
contract: 'GET::/api/orders',
|
||||
role: 'consumer',
|
||||
},
|
||||
];
|
||||
|
||||
const dbExecutors = new Map<
|
||||
string,
|
||||
(cypher: string, params?: Record<string, unknown>) => Promise<Record<string, unknown>[]>
|
||||
>([
|
||||
['orders-svc', async () => []],
|
||||
['gateway', async () => []],
|
||||
]);
|
||||
|
||||
const result = await extractor.extractFromManifest(links, dbExecutors);
|
||||
const provider = result.contracts.find((c) => c.role === 'provider');
|
||||
// No match → synthetic uid, caller falls back as today.
|
||||
expect(provider?.symbolUid).toBe('manifest::orders-svc::http::GET::/api/orders');
|
||||
});
|
||||
|
||||
it('buildContractId round-trip regression for GET::/api/orders', async () => {
|
||||
// Verifies buildContractId still produces http::GET::/api/orders for
|
||||
// explicit-method form — i.e. the fix to resolveSymbol did not touch
|
||||
// buildContractId.
|
||||
const links: GroupManifestLink[] = [
|
||||
{
|
||||
from: 'gateway',
|
||||
to: 'orders-svc',
|
||||
type: 'http',
|
||||
contract: 'GET::/api/orders',
|
||||
role: 'consumer',
|
||||
},
|
||||
];
|
||||
|
||||
const result = await extractor.extractFromManifest(links);
|
||||
const provider = result.contracts.find((c) => c.role === 'provider');
|
||||
expect(provider?.contractId).toBe('http::GET::/api/orders');
|
||||
});
|
||||
|
||||
it('canonicalizes method casing so get::/api/orders and GET::/api/orders share a contractId', async () => {
|
||||
// Regression for Copilot's review on PR #817: without canonicalization,
|
||||
// `buildContractId` passed raw casing through (`http::get::/api/orders`)
|
||||
// while `parseHttpContract` upper-cased during lookup, fragmenting
|
||||
// cross-impact joins between providers and consumers that happened to
|
||||
// use different casing conventions in their group.yaml.
|
||||
const lower = await extractor.extractFromManifest([
|
||||
{
|
||||
from: 'gateway',
|
||||
to: 'orders-svc',
|
||||
type: 'http',
|
||||
contract: 'get::/api/orders',
|
||||
role: 'consumer',
|
||||
},
|
||||
]);
|
||||
const upper = await extractor.extractFromManifest([
|
||||
{
|
||||
from: 'gateway',
|
||||
to: 'orders-svc',
|
||||
type: 'http',
|
||||
contract: 'GET::/api/orders',
|
||||
role: 'consumer',
|
||||
},
|
||||
]);
|
||||
const lowerContractId = lower.contracts.find((c) => c.role === 'provider')?.contractId;
|
||||
const upperContractId = upper.contracts.find((c) => c.role === 'provider')?.contractId;
|
||||
expect(lowerContractId).toBe('http::GET::/api/orders');
|
||||
expect(upperContractId).toBe('http::GET::/api/orders');
|
||||
expect(lowerContractId).toBe(upperContractId);
|
||||
});
|
||||
|
||||
it('returns empty for no links', async () => {
|
||||
const result = await extractor.extractFromManifest([]);
|
||||
expect(result.contracts).toHaveLength(0);
|
||||
expect(result.crossLinks).toHaveLength(0);
|
||||
});
|
||||
|
||||
it('memoizes repeated (repo, type, contract) resolutions so each tuple hits the DB once', async () => {
|
||||
const calls: Array<{ repo: string; cypher: string }> = [];
|
||||
const execFor = (repo: string) => async (cypher: string) => {
|
||||
calls.push({ repo, cypher });
|
||||
return [{ uid: `uid::${repo}`, name: 'handler', filePath: 'src/h.ts' }];
|
||||
};
|
||||
|
||||
const dbExecutors = new Map<string, (c: string) => Promise<Record<string, unknown>[]>>([
|
||||
['svc/a', execFor('svc/a')],
|
||||
['svc/b', execFor('svc/b')],
|
||||
]);
|
||||
|
||||
// Two links declare the same (repo, type, contract) triple on each side,
|
||||
// so naive sequential resolution would run 4 queries; memoization collapses
|
||||
// to 2 (one per distinct repo tuple).
|
||||
const link: GroupManifestLink = {
|
||||
from: 'svc/b',
|
||||
to: 'svc/a',
|
||||
type: 'http',
|
||||
contract: 'GET::/api/orders',
|
||||
role: 'consumer',
|
||||
};
|
||||
|
||||
await extractor.extractFromManifest([link, { ...link }], dbExecutors);
|
||||
|
||||
// One resolution per distinct (repo, type, contract) — not per (link × side).
|
||||
expect(calls).toHaveLength(2);
|
||||
expect(new Set(calls.map((c) => c.repo))).toEqual(new Set(['svc/a', 'svc/b']));
|
||||
});
|
||||
});
|
||||
|
||||
@@ -3,7 +3,12 @@ import * as fs from 'node:fs';
|
||||
import * as path from 'node:path';
|
||||
import * as os from 'node:os';
|
||||
import { syncGroup, stableRepoPoolId } from '../../../src/core/group/sync.js';
|
||||
import type { GroupConfig, StoredContract, RepoHandle } from '../../../src/core/group/types.js';
|
||||
import type {
|
||||
GroupConfig,
|
||||
StoredContract,
|
||||
RepoHandle,
|
||||
GroupManifestLink,
|
||||
} from '../../../src/core/group/types.js';
|
||||
import type { RegistryEntry } from '../../../src/storage/repo-manager.js';
|
||||
|
||||
describe('syncGroup', () => {
|
||||
@@ -202,6 +207,110 @@ describe('syncGroup', () => {
|
||||
}
|
||||
});
|
||||
|
||||
it('manifest links in config.links produce cross-links with matchType manifest', async () => {
|
||||
const links: GroupManifestLink[] = [
|
||||
{
|
||||
from: 'app/consumer',
|
||||
to: 'app/provider',
|
||||
type: 'http',
|
||||
contract: 'GET::/api/orders',
|
||||
role: 'consumer',
|
||||
},
|
||||
];
|
||||
|
||||
const config: GroupConfig = {
|
||||
version: 1,
|
||||
name: 'test',
|
||||
description: '',
|
||||
repos: { 'app/consumer': 'consumer-repo', 'app/provider': 'provider-repo' },
|
||||
links,
|
||||
packages: {},
|
||||
detect: {
|
||||
http: true,
|
||||
grpc: false,
|
||||
topics: false,
|
||||
shared_libs: false,
|
||||
embedding_fallback: false,
|
||||
},
|
||||
matching: { bm25_threshold: 0.7, embedding_threshold: 0.65, max_candidates_per_step: 3 },
|
||||
};
|
||||
|
||||
const result = await syncGroup(config, {
|
||||
extractorOverride: async () => [],
|
||||
skipWrite: true,
|
||||
});
|
||||
|
||||
// ManifestExtractor should inject 2 contracts (provider + consumer) and 1 cross-link
|
||||
expect(result.contracts).toHaveLength(2);
|
||||
const manifestLinks = result.crossLinks.filter((cl) => cl.matchType === 'manifest');
|
||||
expect(manifestLinks).toHaveLength(1);
|
||||
expect(manifestLinks[0].contractId).toBe('http::GET::/api/orders');
|
||||
expect(manifestLinks[0].from.repo).toBe('app/consumer');
|
||||
expect(manifestLinks[0].to.repo).toBe('app/provider');
|
||||
expect(manifestLinks[0].confidence).toBe(1.0);
|
||||
|
||||
// With no DB executors available, UIDs fall back to the deterministic
|
||||
// synthetic form `manifest::<repo>::<contractId>`.
|
||||
expect(manifestLinks[0].from.symbolUid).toBe('manifest::app/consumer::http::GET::/api/orders');
|
||||
expect(manifestLinks[0].to.symbolUid).toBe('manifest::app/provider::http::GET::/api/orders');
|
||||
|
||||
// Manifest contracts also participate in runExactMatch; we must not emit a
|
||||
// duplicate matchType:'exact' cross-link for the same endpoint pair.
|
||||
const exactForSameContract = result.crossLinks.filter(
|
||||
(cl) => cl.matchType === 'exact' && cl.contractId === 'http::GET::/api/orders',
|
||||
);
|
||||
expect(exactForSameContract).toHaveLength(0);
|
||||
expect(result.crossLinks).toHaveLength(1);
|
||||
});
|
||||
|
||||
it('manifest links referencing unknown repos still produce cross-links via synthetic UIDs', async () => {
|
||||
const links: GroupManifestLink[] = [
|
||||
{
|
||||
from: 'app/known',
|
||||
to: 'app/dangling', // not present in config.repos
|
||||
type: 'http',
|
||||
contract: 'POST::/api/missing',
|
||||
role: 'consumer',
|
||||
},
|
||||
];
|
||||
|
||||
const config: GroupConfig = {
|
||||
version: 1,
|
||||
name: 'test',
|
||||
description: '',
|
||||
repos: { 'app/known': 'known-repo' },
|
||||
links,
|
||||
packages: {},
|
||||
detect: {
|
||||
http: true,
|
||||
grpc: false,
|
||||
topics: false,
|
||||
shared_libs: false,
|
||||
embedding_fallback: false,
|
||||
},
|
||||
matching: { bm25_threshold: 0.7, embedding_threshold: 0.65, max_candidates_per_step: 3 },
|
||||
};
|
||||
|
||||
const warnings: string[] = [];
|
||||
const origWarn = console.warn;
|
||||
console.warn = (msg: string) => warnings.push(String(msg));
|
||||
try {
|
||||
const result = await syncGroup(config, {
|
||||
extractorOverride: async () => [],
|
||||
skipWrite: true,
|
||||
});
|
||||
|
||||
expect(result.crossLinks).toHaveLength(1);
|
||||
expect(result.crossLinks[0].matchType).toBe('manifest');
|
||||
expect(result.crossLinks[0].to.symbolUid).toBe(
|
||||
'manifest::app/dangling::http::POST::/api/missing',
|
||||
);
|
||||
expect(warnings.some((w) => w.includes('app/dangling'))).toBe(true);
|
||||
} finally {
|
||||
console.warn = origWarn;
|
||||
}
|
||||
});
|
||||
|
||||
it('writes registry to groupDir when skipWrite is false', async () => {
|
||||
const tmpDir = path.join(os.tmpdir(), `gitnexus-sync-write-${Date.now()}`);
|
||||
fs.mkdirSync(tmpDir, { recursive: true });
|
||||
|
||||
@@ -0,0 +1,283 @@
|
||||
import { describe, it, expect, beforeEach, afterEach } from 'vitest';
|
||||
import { EventEmitter } from 'events';
|
||||
import fs from 'fs';
|
||||
import path from 'path';
|
||||
import os from 'os';
|
||||
import { splitRelCsvByLabelPair } from '../../src/core/lbug/lbug-adapter.js';
|
||||
|
||||
/**
|
||||
* Regression tests for splitRelCsvByLabelPair (PR #818).
|
||||
*
|
||||
* These tests call the real exported function from lbug-adapter.ts with a
|
||||
* mock WriteStream factory, exercising the actual backpressure, error
|
||||
* handling, and drain-listener guard without touching LadybugDB.
|
||||
*/
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Mock WriteStream — controllable backpressure + error injection
|
||||
// ---------------------------------------------------------------------------
|
||||
class MockWriteStream extends EventEmitter {
|
||||
public chunks: string[] = [];
|
||||
public destroyed = false;
|
||||
public ended = false;
|
||||
public blocked = false;
|
||||
public maxDrainListenersSeen = 0;
|
||||
|
||||
write(chunk: string): boolean {
|
||||
this.chunks.push(chunk);
|
||||
this._trackDrainListeners();
|
||||
return !this.blocked;
|
||||
}
|
||||
|
||||
end(cb?: (err?: Error) => void): this {
|
||||
this.ended = true;
|
||||
if (cb) cb();
|
||||
return this;
|
||||
}
|
||||
|
||||
destroy(): this {
|
||||
this.destroyed = true;
|
||||
return this;
|
||||
}
|
||||
|
||||
unblock(): void {
|
||||
this.blocked = false;
|
||||
this.emit('drain');
|
||||
}
|
||||
|
||||
triggerError(err: Error): void {
|
||||
this.emit('error', err);
|
||||
}
|
||||
|
||||
private _trackDrainListeners(): void {
|
||||
const count = this.listenerCount('drain');
|
||||
if (count > this.maxDrainListenersSeen) {
|
||||
this.maxDrainListenersSeen = count;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Test helpers
|
||||
// ---------------------------------------------------------------------------
|
||||
const HEADER = '"from","to","type","confidence","reason","step"';
|
||||
|
||||
function csvLine(from: string, to: string, type = 'CALLS'): string {
|
||||
return `"${from}","${to}","${type}",1.0,"auto",0`;
|
||||
}
|
||||
|
||||
function getNodeLabel(id: string): string {
|
||||
return id.split(':')[0];
|
||||
}
|
||||
|
||||
/** Cast MockWriteStream factory to the real WriteStreamFactory type. */
|
||||
function mockFactory(streams: MockWriteStream[], opts?: { blocked?: boolean }) {
|
||||
return (() => {
|
||||
const ws = new MockWriteStream();
|
||||
if (opts?.blocked) ws.blocked = true;
|
||||
streams.push(ws);
|
||||
return ws;
|
||||
}) as unknown as (filePath: string) => import('fs').WriteStream;
|
||||
}
|
||||
|
||||
let tmpDir: string;
|
||||
|
||||
beforeEach(() => {
|
||||
tmpDir = fs.mkdtempSync(path.join(os.tmpdir(), 'rel-csv-test-'));
|
||||
});
|
||||
|
||||
afterEach(() => {
|
||||
// fs.rmSync's built-in retry loop handles Windows EBUSY/ENOTEMPTY/EPERM
|
||||
// when a just-closed fd hasn't been released yet (Node added this exactly
|
||||
// for cross-platform tmpdir cleanup — see Node.js fs docs). The production
|
||||
// function also waits for the input stream's 'close' event, so this is
|
||||
// defense-in-depth.
|
||||
fs.rmSync(tmpDir, { recursive: true, force: true, maxRetries: 5, retryDelay: 50 });
|
||||
});
|
||||
|
||||
function writeCsv(lines: string[]): string {
|
||||
const csvPath = path.join(tmpDir, 'relations.csv');
|
||||
fs.writeFileSync(csvPath, lines.join('\n') + '\n');
|
||||
return csvPath;
|
||||
}
|
||||
|
||||
// ---------------------------------------------------------------------------
|
||||
// Tests
|
||||
// ---------------------------------------------------------------------------
|
||||
describe('splitRelCsvByLabelPair', () => {
|
||||
const validTables = new Set(['Function', 'Class', 'File', 'Method']);
|
||||
|
||||
it('splits lines into per-pair files with correct row counts', async () => {
|
||||
const csvPath = writeCsv([
|
||||
HEADER,
|
||||
csvLine('Function:a', 'Class:b'),
|
||||
csvLine('Function:c', 'Class:d'),
|
||||
csvLine('File:e', 'Method:f'),
|
||||
]);
|
||||
|
||||
const streams: MockWriteStream[] = [];
|
||||
const result = await splitRelCsvByLabelPair(
|
||||
csvPath,
|
||||
tmpDir,
|
||||
validTables,
|
||||
getNodeLabel,
|
||||
mockFactory(streams),
|
||||
);
|
||||
|
||||
expect(result.totalValidRels).toBe(3);
|
||||
expect(result.relsByPairMeta.get('Function|Class')?.rows).toBe(2);
|
||||
expect(result.relsByPairMeta.get('File|Method')?.rows).toBe(1);
|
||||
});
|
||||
|
||||
it('captures the CSV header in relHeader', async () => {
|
||||
const csvPath = writeCsv([HEADER, csvLine('Function:a', 'Class:b')]);
|
||||
|
||||
const streams: MockWriteStream[] = [];
|
||||
const result = await splitRelCsvByLabelPair(
|
||||
csvPath,
|
||||
tmpDir,
|
||||
validTables,
|
||||
getNodeLabel,
|
||||
mockFactory(streams),
|
||||
);
|
||||
|
||||
expect(result.relHeader).toBe(HEADER);
|
||||
});
|
||||
|
||||
it('skips lines with unknown labels and counts them', async () => {
|
||||
const csvPath = writeCsv([
|
||||
HEADER,
|
||||
csvLine('Function:a', 'Class:b'),
|
||||
csvLine('Unknown:x', 'Class:y'),
|
||||
csvLine('Function:c', 'Bogus:d'),
|
||||
]);
|
||||
|
||||
const streams: MockWriteStream[] = [];
|
||||
const result = await splitRelCsvByLabelPair(
|
||||
csvPath,
|
||||
tmpDir,
|
||||
validTables,
|
||||
getNodeLabel,
|
||||
mockFactory(streams),
|
||||
);
|
||||
|
||||
expect(result.totalValidRels).toBe(1);
|
||||
expect(result.skippedRels).toBe(2);
|
||||
});
|
||||
|
||||
it('ignores blank lines without counting them as skipped', async () => {
|
||||
const csvPath = writeCsv([HEADER, '', csvLine('Function:a', 'Class:b'), '', '']);
|
||||
|
||||
const streams: MockWriteStream[] = [];
|
||||
const result = await splitRelCsvByLabelPair(
|
||||
csvPath,
|
||||
tmpDir,
|
||||
validTables,
|
||||
getNodeLabel,
|
||||
mockFactory(streams),
|
||||
);
|
||||
|
||||
expect(result.totalValidRels).toBe(1);
|
||||
expect(result.skippedRels).toBe(0);
|
||||
});
|
||||
|
||||
it('registers at most 1 drain listener per stream under heavy backpressure', async () => {
|
||||
const lines = [HEADER];
|
||||
for (let i = 0; i < 50; i++) {
|
||||
lines.push(csvLine(`Function:f${i}`, `Class:c${i}`));
|
||||
}
|
||||
const csvPath = writeCsv(lines);
|
||||
|
||||
const streams: MockWriteStream[] = [];
|
||||
const promise = splitRelCsvByLabelPair(
|
||||
csvPath,
|
||||
tmpDir,
|
||||
validTables,
|
||||
getNodeLabel,
|
||||
mockFactory(streams, { blocked: true }),
|
||||
);
|
||||
|
||||
// Give readline time to buffer and fire lines
|
||||
await new Promise((r) => setTimeout(r, 50));
|
||||
|
||||
// Unblock all streams so the Promise can resolve
|
||||
for (const ws of streams) ws.unblock();
|
||||
await promise;
|
||||
|
||||
// The guard should have kept drain listeners at 1
|
||||
for (const ws of streams) {
|
||||
expect(ws.maxDrainListenersSeen).toBeLessThanOrEqual(1);
|
||||
}
|
||||
});
|
||||
|
||||
it('rejects the Promise when a WriteStream emits an error', async () => {
|
||||
const csvPath = writeCsv([HEADER, csvLine('Function:a', 'Class:b')]);
|
||||
|
||||
const streams: MockWriteStream[] = [];
|
||||
const promise = splitRelCsvByLabelPair(
|
||||
csvPath,
|
||||
tmpDir,
|
||||
validTables,
|
||||
getNodeLabel,
|
||||
mockFactory(streams, { blocked: true }),
|
||||
);
|
||||
|
||||
// Wait for readline to process, then error while paused on drain
|
||||
await new Promise((r) => setTimeout(r, 50));
|
||||
expect(streams.length).toBeGreaterThan(0);
|
||||
streams[0].triggerError(new Error('disk full'));
|
||||
|
||||
await expect(promise).rejects.toThrow('disk full');
|
||||
});
|
||||
|
||||
it('destroys all streams when one errors (no lingering FDs)', async () => {
|
||||
const lines = [HEADER];
|
||||
for (let i = 0; i < 10; i++) {
|
||||
lines.push(csvLine(`Function:f${i}`, `Class:c${i}`));
|
||||
lines.push(csvLine(`File:e${i}`, `Method:m${i}`));
|
||||
}
|
||||
const csvPath = writeCsv(lines);
|
||||
|
||||
const streams: MockWriteStream[] = [];
|
||||
const promise = splitRelCsvByLabelPair(
|
||||
csvPath,
|
||||
tmpDir,
|
||||
validTables,
|
||||
getNodeLabel,
|
||||
mockFactory(streams, { blocked: true }),
|
||||
);
|
||||
|
||||
// The first pair stream is created immediately and blocks on its header
|
||||
// write. Unblock it once so the loop advances and creates the second
|
||||
// pair stream (also blocked). Now both streams exist — trigger the error.
|
||||
await new Promise((r) => setTimeout(r, 20));
|
||||
expect(streams.length).toBe(1);
|
||||
streams[0].unblock();
|
||||
await new Promise((r) => setTimeout(r, 20));
|
||||
expect(streams.length).toBeGreaterThanOrEqual(2);
|
||||
streams[0].triggerError(new Error('EMFILE'));
|
||||
|
||||
await expect(promise).rejects.toThrow('EMFILE');
|
||||
|
||||
for (const ws of streams) {
|
||||
expect(ws.destroyed).toBe(true);
|
||||
}
|
||||
});
|
||||
|
||||
it('handles empty CSV (header only) without errors', async () => {
|
||||
const csvPath = writeCsv([HEADER]);
|
||||
|
||||
const streams: MockWriteStream[] = [];
|
||||
const result = await splitRelCsvByLabelPair(
|
||||
csvPath,
|
||||
tmpDir,
|
||||
validTables,
|
||||
getNodeLabel,
|
||||
mockFactory(streams),
|
||||
);
|
||||
|
||||
expect(result.totalValidRels).toBe(0);
|
||||
expect(result.skippedRels).toBe(0);
|
||||
expect(result.relHeader).toBe(HEADER);
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,101 @@
|
||||
/**
|
||||
* Unit tests for `expandTransitiveIncludeClosure` — the C/C++ Strategy 1
|
||||
* (`wildcard-transitive`) implementation extracted from `wildcard-synthesis.ts`.
|
||||
*
|
||||
* These tests exercise the BFS/DFS closure algorithm in isolation, without
|
||||
* running the full pipeline. They cover edge cases flagged in PR #816 review:
|
||||
* circular header includes, deep chains, and graphImports-only transitive paths.
|
||||
*/
|
||||
|
||||
import { describe, it, expect } from 'vitest';
|
||||
import { expandTransitiveIncludeClosure } from '../../src/core/ingestion/pipeline-phases/wildcard-synthesis.js';
|
||||
|
||||
const EMPTY = new Map<string, ReadonlySet<string>>();
|
||||
|
||||
describe('expandTransitiveIncludeClosure', () => {
|
||||
it('returns the direct imports when none are chained', () => {
|
||||
const direct = new Set(['a.h', 'b.h']);
|
||||
const closure = expandTransitiveIncludeClosure(direct, EMPTY, EMPTY);
|
||||
expect([...closure].sort()).toEqual(['a.h', 'b.h']);
|
||||
});
|
||||
|
||||
it('expands a two-hop chain via importMap (a.c → b.h → c.h)', () => {
|
||||
const importMap = new Map<string, ReadonlySet<string>>([['b.h', new Set(['c.h'])]]);
|
||||
const closure = expandTransitiveIncludeClosure(new Set(['b.h']), importMap, EMPTY);
|
||||
expect([...closure].sort()).toEqual(['b.h', 'c.h']);
|
||||
});
|
||||
|
||||
it('expands a deep 5-level chain (A → B → C → D → E)', () => {
|
||||
const importMap = new Map<string, ReadonlySet<string>>([
|
||||
['B.h', new Set(['C.h'])],
|
||||
['C.h', new Set(['D.h'])],
|
||||
['D.h', new Set(['E.h'])],
|
||||
]);
|
||||
const closure = expandTransitiveIncludeClosure(new Set(['B.h']), importMap, EMPTY);
|
||||
expect([...closure].sort()).toEqual(['B.h', 'C.h', 'D.h', 'E.h']);
|
||||
});
|
||||
|
||||
it('terminates on circular header includes (A.h ↔ B.h)', () => {
|
||||
const importMap = new Map<string, ReadonlySet<string>>([
|
||||
['A.h', new Set(['B.h'])],
|
||||
['B.h', new Set(['A.h'])],
|
||||
]);
|
||||
const closure = expandTransitiveIncludeClosure(new Set(['A.h']), importMap, EMPTY);
|
||||
expect([...closure].sort()).toEqual(['A.h', 'B.h']);
|
||||
});
|
||||
|
||||
it('terminates on self-referential include (A.h includes A.h)', () => {
|
||||
const importMap = new Map<string, ReadonlySet<string>>([['A.h', new Set(['A.h'])]]);
|
||||
const closure = expandTransitiveIncludeClosure(new Set(['A.h']), importMap, EMPTY);
|
||||
expect([...closure]).toEqual(['A.h']);
|
||||
});
|
||||
|
||||
it('expands through graphImports edges when importMap is empty', () => {
|
||||
const graphImports = new Map<string, ReadonlySet<string>>([['b.h', new Set(['c.h'])]]);
|
||||
const closure = expandTransitiveIncludeClosure(new Set(['b.h']), EMPTY, graphImports);
|
||||
expect([...closure].sort()).toEqual(['b.h', 'c.h']);
|
||||
});
|
||||
|
||||
it('combines importMap and graphImports in one traversal', () => {
|
||||
const importMap = new Map<string, ReadonlySet<string>>([['b.h', new Set(['c.h'])]]);
|
||||
const graphImports = new Map<string, ReadonlySet<string>>([['c.h', new Set(['d.h'])]]);
|
||||
const closure = expandTransitiveIncludeClosure(new Set(['b.h']), importMap, graphImports);
|
||||
expect([...closure].sort()).toEqual(['b.h', 'c.h', 'd.h']);
|
||||
});
|
||||
|
||||
it('returns an empty set when given no direct imports', () => {
|
||||
const closure = expandTransitiveIncludeClosure(new Set<string>(), EMPTY, EMPTY);
|
||||
expect(closure.size).toBe(0);
|
||||
});
|
||||
|
||||
it('caps closure size to prevent OOM on pathological codebases', () => {
|
||||
// Build a synthetic include graph of 10,000 files, each including the next.
|
||||
// The cap (5000) should halt BFS early with a partial but bounded closure.
|
||||
const importMap = new Map<string, ReadonlySet<string>>();
|
||||
for (let i = 0; i < 10_000; i++) {
|
||||
importMap.set(`h${i}.h`, new Set([`h${i + 1}.h`]));
|
||||
}
|
||||
const closure = expandTransitiveIncludeClosure(new Set(['h0.h']), importMap, EMPTY);
|
||||
expect(closure.size).toBe(5000);
|
||||
// Partial closure still starts from the importer's side (BFS ordering).
|
||||
expect(closure.has('h0.h')).toBe(true);
|
||||
expect(closure.has('h1.h')).toBe(true);
|
||||
expect(closure.has('h9999.h')).toBe(false);
|
||||
});
|
||||
|
||||
it('deduplicates when a file is reachable through multiple paths (diamond)', () => {
|
||||
// A
|
||||
// / \
|
||||
// B C
|
||||
// \ /
|
||||
// D
|
||||
const importMap = new Map<string, ReadonlySet<string>>([
|
||||
['A.h', new Set(['B.h', 'C.h'])],
|
||||
['B.h', new Set(['D.h'])],
|
||||
['C.h', new Set(['D.h'])],
|
||||
]);
|
||||
const closure = expandTransitiveIncludeClosure(new Set(['A.h']), importMap, EMPTY);
|
||||
expect([...closure].sort()).toEqual(['A.h', 'B.h', 'C.h', 'D.h']);
|
||||
expect(closure.size).toBe(4); // D.h appears once
|
||||
});
|
||||
});
|
||||
Reference in New Issue
Block a user