Compare commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
43d866c802 | ||
|
|
6b0c566392 | ||
|
|
cb1293b718 | ||
|
|
6c9b6eb0f6 | ||
|
|
3558cb8a7f | ||
|
|
9f2d1780d5 | ||
|
|
a57550815f | ||
|
|
ee78ebe64a | ||
|
|
141f864181 | ||
|
|
01fa5bf98e | ||
|
|
bbd95457df | ||
|
|
d43cc691f0 | ||
|
|
f90aabf9a8 | ||
|
|
beb5574d38 | ||
|
|
d48903cce6 | ||
|
|
a3fac2f672 | ||
|
|
9954f6fdfd | ||
|
|
90ca43f851 | ||
|
|
0a200a51cb | ||
|
|
c87cb81892 | ||
|
|
b8ab23a838 | ||
|
|
e70a6d80c0 | ||
|
|
cd1c0ff7dc | ||
|
|
907440cf0b | ||
|
|
843c561e9b | ||
|
|
4c8be50cb1 | ||
|
|
d07a69c3b1 | ||
|
|
d206bf6772 | ||
|
|
03156935cb | ||
|
|
22b5fce19e | ||
|
|
09e3609376 | ||
|
|
f0f384aab7 | ||
|
|
ef8b252bcc | ||
|
|
92a1086697 | ||
|
|
b272c6864c | ||
|
|
4031562f6b | ||
|
|
4dffd81b12 | ||
|
|
b75e380bd0 | ||
|
|
24421c2db1 | ||
|
|
e6eaf08382 | ||
|
|
f409fb7525 | ||
|
|
cc2b13e332 | ||
|
|
b909bbff70 | ||
|
|
56356b71db | ||
|
|
2c9e887a80 | ||
|
|
8fbfd09081 | ||
|
|
6f4281c946 | ||
|
|
60265c1d0d | ||
|
|
fa765beec1 | ||
|
|
61f2f6d954 | ||
|
|
2428e72bd2 | ||
|
|
76ed0fa53b | ||
|
|
2c17a4642c | ||
|
|
4f2e4ac24d | ||
|
|
1af9c028c5 | ||
|
|
fb20a3c752 | ||
|
|
f1fbe643df | ||
|
|
89a24d866e | ||
|
|
9baef90ae2 | ||
|
|
11575cf6c8 | ||
|
|
c2c694ac42 | ||
|
|
01ff2a7540 | ||
|
|
13b22e9879 | ||
|
|
c8480f899d | ||
|
|
1f7764c49b | ||
|
|
00758b102a | ||
|
|
abeb52e5e5 | ||
|
|
6c972079e0 | ||
|
|
cbfdae0303 | ||
|
|
66c1ffa370 | ||
|
|
88e0034771 | ||
|
|
0736bb23bc | ||
|
|
2ff4d93314 | ||
|
|
fff716dd92 | ||
|
|
56bc226a1c | ||
|
|
a6a1004e82 | ||
|
|
5b012c3351 | ||
|
|
862fdf9185 | ||
|
|
228c993bb7 | ||
|
|
c3a2815186 | ||
|
|
893f77ae89 | ||
|
|
d49c76ddc5 | ||
|
|
7dcafb647b | ||
|
|
5dcc567870 | ||
|
|
999fbf5b11 | ||
|
|
c79c717790 | ||
|
|
9dc00e47a7 | ||
|
|
c14a78a341 | ||
|
|
7a0a83e45c | ||
|
|
f2edbb4f82 | ||
|
|
1d27ad09a2 | ||
|
|
68d4c48aba | ||
|
|
bc771574d8 | ||
|
|
5d6e15eea3 | ||
|
|
eb74eb8590 | ||
|
|
700c9d16e4 | ||
|
|
19ff84fa31 | ||
|
|
2fe03d2a21 | ||
|
|
3e29f4e4b9 | ||
|
|
71353512a1 | ||
|
|
00c5126b24 | ||
|
|
06994e474a | ||
|
|
c507b4b197 | ||
|
|
b6947b0c02 | ||
|
|
e9ccec1a52 | ||
|
|
0122d9e694 | ||
|
|
00e2476eca | ||
|
|
217efcf015 | ||
|
|
7c72cefd8d | ||
|
|
58f67d07f7 | ||
|
|
a4863605e1 | ||
|
|
fc58c415f7 | ||
|
|
790d1d5b0f | ||
|
|
8273324f3c | ||
|
|
7b71b64427 | ||
|
|
e6b8edc1ac | ||
|
|
a7b8c302d4 | ||
|
|
5769872b70 | ||
|
|
60c93d7d4a | ||
|
|
b0f25e216d | ||
|
|
84f07e83ee | ||
|
|
1e19986ef3 | ||
|
|
973c7bfbf0 | ||
|
|
c0b4098c4e | ||
|
|
11a3d0515c | ||
|
|
e0a6c40b45 | ||
|
|
aa1bab597b | ||
|
|
60ede20a11 | ||
|
|
fb5270c260 | ||
|
|
604b575e4b | ||
|
|
02dfab578c | ||
|
|
1326490a5b | ||
|
|
b48cfe9894 | ||
|
|
c1703fc0a9 | ||
|
|
480fae933b | ||
|
|
3879490817 | ||
|
|
50dbd03779 | ||
|
|
1003d8b6a5 | ||
|
|
74b9701509 | ||
|
|
f0132c1077 | ||
|
|
f6b92d4f13 | ||
|
|
64b7ff0061 | ||
|
|
f2d3df48f6 | ||
|
|
5fa73bafdf | ||
|
|
fbff6d08c0 | ||
|
|
f83ef56ccb | ||
|
|
530b4be9ee | ||
|
|
6c18ae08f7 | ||
|
|
5a5850832c | ||
|
|
62242d5f44 | ||
|
|
6e38db879e | ||
|
|
0999595444 | ||
|
|
649ad80dbb | ||
|
|
2915e60630 | ||
|
|
984316d260 |
@@ -0,0 +1,257 @@
|
||||
"""Pure math utilities for triage sweep embedding analysis.
|
||||
|
||||
All functions are stateless and perform no I/O (except model loading by FastEmbed).
|
||||
Each function operates on numpy arrays and returns numpy arrays or plain Python types.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import numpy as np
|
||||
from numpy.typing import NDArray
|
||||
from fastembed import TextEmbedding
|
||||
from sklearn.decomposition import PCA
|
||||
from sklearn.covariance import EllipticEnvelope
|
||||
from sklearn.metrics.pairwise import cosine_similarity
|
||||
|
||||
# FastEmbed model — BAAI/bge-small-en-v1.5 produces 384-dimensional embeddings.
|
||||
# ~46MB quantized ONNX, runs on CPU in ~0.5s per batch of 32.
|
||||
EMBEDDING_MODEL: str = "BAAI/bge-small-en-v1.5"
|
||||
|
||||
# Embedding dimensionality (determined by model choice).
|
||||
EMBEDDING_DIM: int = 384
|
||||
|
||||
# Batch size for FastEmbed. 32 balances memory and throughput on
|
||||
# a 2-vCPU GitHub Actions runner with ~7GB RAM.
|
||||
EMBEDDING_BATCH_SIZE: int = 32
|
||||
|
||||
|
||||
def embed_texts(texts: list[str]) -> NDArray[np.float32]:
|
||||
"""Embed a list of texts into dense vectors using FastEmbed.
|
||||
|
||||
Returns an array of shape (len(texts), 384) with dtype float32.
|
||||
Empty input returns a (0, 384) array.
|
||||
"""
|
||||
if not texts:
|
||||
return np.empty((0, EMBEDDING_DIM), dtype=np.float32)
|
||||
|
||||
model = TextEmbedding(model_name=EMBEDDING_MODEL)
|
||||
vectors = list(model.embed(texts, batch_size=EMBEDDING_BATCH_SIZE))
|
||||
return np.vstack(vectors).astype(np.float32)
|
||||
|
||||
|
||||
def normalize_rows(matrix: NDArray[np.float32]) -> NDArray[np.float32]:
|
||||
"""L2-normalize each row to unit length.
|
||||
|
||||
Zero-norm rows (e.g. from empty text) remain zero vectors.
|
||||
Uses eps=1e-10 in the denominator to avoid division by zero.
|
||||
"""
|
||||
if matrix.shape[0] == 0:
|
||||
return matrix
|
||||
|
||||
norms = np.linalg.norm(matrix, axis=1, keepdims=True)
|
||||
return matrix / (norms + 1e-10)
|
||||
|
||||
|
||||
def reduce_dimensions(
|
||||
matrix: NDArray[np.float32],
|
||||
max_components: int,
|
||||
) -> NDArray[np.float32]:
|
||||
"""Reduce dimensionality via PCA.
|
||||
|
||||
Computes n_components = min(max_components, n-1, d). If n_components < 1,
|
||||
returns the matrix unchanged. Logs explained variance for observability.
|
||||
"""
|
||||
n, d = matrix.shape
|
||||
if n <= 1:
|
||||
return matrix
|
||||
|
||||
n_components = min(max_components, n - 1, d)
|
||||
if n_components < 1:
|
||||
return matrix
|
||||
|
||||
pca = PCA(n_components=n_components)
|
||||
reduced = pca.fit_transform(matrix)
|
||||
explained = pca.explained_variance_ratio_.sum()
|
||||
print(f"PCA: {d}d -> {n_components}d, explained variance: {explained:.3f}")
|
||||
return reduced.astype(np.float32)
|
||||
|
||||
|
||||
def detect_outliers(
|
||||
matrix: NDArray[np.float32],
|
||||
contamination: float = 0.1,
|
||||
iqr_multiplier: float = 3.0,
|
||||
max_outlier_pct: float = 0.05,
|
||||
) -> list[tuple[int, float]]:
|
||||
"""Flag items whose Mahalanobis distance exceeds an IQR-based cutoff.
|
||||
|
||||
Uses EllipticEnvelope (robust covariance via MCD) to estimate the
|
||||
multivariate Gaussian, then computes sqrt(squared Mahalanobis distance)
|
||||
for each sample. The cutoff is Q75 + iqr_multiplier * IQR, which
|
||||
adapts to the actual distribution of distances.
|
||||
|
||||
A hard cap ensures no more than max_outlier_pct * n items are flagged;
|
||||
when the cap is hit, only the most extreme items (sorted by distance
|
||||
descending) are kept.
|
||||
|
||||
Returns (index, distance) tuples sorted by index ascending, along with
|
||||
the cutoff value stored as an attribute on the returned list.
|
||||
"""
|
||||
n = matrix.shape[0]
|
||||
if n < 2:
|
||||
return []
|
||||
|
||||
envelope = EllipticEnvelope(contamination=contamination, random_state=42)
|
||||
envelope.fit(matrix)
|
||||
|
||||
# .mahalanobis() returns squared Mahalanobis distances
|
||||
distances = np.sqrt(envelope.mahalanobis(matrix))
|
||||
|
||||
# IQR-based cutoff
|
||||
q25, q75 = np.percentile(distances, [25, 75])
|
||||
iqr = q75 - q25
|
||||
cutoff = q75 + iqr_multiplier * iqr
|
||||
|
||||
outlier_mask = distances > cutoff
|
||||
indices = np.where(outlier_mask)[0]
|
||||
|
||||
# Hard cap: keep at most max_outlier_pct * n items
|
||||
max_count = max(1, int(max_outlier_pct * n))
|
||||
if len(indices) > max_count:
|
||||
# Sort by distance descending, take the most extreme
|
||||
sorted_by_dist = sorted(indices, key=lambda i: distances[i], reverse=True)
|
||||
indices = np.array(sorted_by_dist[:max_count])
|
||||
|
||||
# Sort by index ascending for stable output
|
||||
indices = np.sort(indices)
|
||||
result = [(int(idx), float(distances[idx])) for idx in indices]
|
||||
|
||||
# Attach cutoff as metadata so the report can use it
|
||||
result = _OutlierResult(result) # type: ignore[assignment]
|
||||
result.cutoff = float(cutoff) # type: ignore[attr-defined]
|
||||
return result # type: ignore[return-value]
|
||||
|
||||
|
||||
class _OutlierResult(list):
|
||||
"""A list subclass that carries metadata (cutoff) from outlier detection."""
|
||||
cutoff: float = 0.0
|
||||
|
||||
|
||||
def find_duplicate_pairs(
|
||||
matrix: NDArray[np.float32],
|
||||
threshold: float,
|
||||
) -> list[tuple[int, int, float]]:
|
||||
"""Find pairs of items with cosine similarity above threshold.
|
||||
|
||||
Returns (i, j, similarity) tuples where i < j. The input should be
|
||||
L2-normalized embeddings (full dimensionality, not PCA-reduced) so
|
||||
cosine similarity equals the dot product.
|
||||
"""
|
||||
n = matrix.shape[0]
|
||||
if n <= 1:
|
||||
return []
|
||||
|
||||
sim_matrix = cosine_similarity(matrix)
|
||||
# Upper triangle indices (i < j), excluding diagonal
|
||||
rows, cols = np.triu_indices(n, k=1)
|
||||
similarities = sim_matrix[rows, cols]
|
||||
|
||||
mask = similarities > threshold
|
||||
pairs: list[tuple[int, int, float]] = []
|
||||
for idx in np.where(mask)[0]:
|
||||
pairs.append((int(rows[idx]), int(cols[idx]), float(similarities[idx])))
|
||||
|
||||
return pairs
|
||||
|
||||
|
||||
# ── Label suggestion via z-score normalized embedding similarity ──────
|
||||
|
||||
# Z-score threshold: a label must be this many standard deviations above
|
||||
# the column mean to be considered a match.
|
||||
LABEL_Z_THRESHOLD: float = 1.5
|
||||
|
||||
# Margin gate: the top-1 label must beat the second-best by this many
|
||||
# z-score units to be accepted (subsequent labels don't need a margin).
|
||||
LABEL_Z_MARGIN: float = 0.5
|
||||
|
||||
# Floor for per-column standard deviation to avoid division by near-zero.
|
||||
LABEL_Z_STD_FLOOR: float = 0.01
|
||||
|
||||
# Minimum raw cosine similarity required even if z-score is high.
|
||||
# Prevents suggesting labels that are "relatively best" but still poor.
|
||||
MIN_RAW_SIMILARITY: float = 0.3
|
||||
|
||||
# Maximum number of labels to suggest per item.
|
||||
MAX_LABELS_PER_ITEM: int = 3
|
||||
|
||||
|
||||
def suggest_labels(
|
||||
item_embeddings: NDArray[np.float32],
|
||||
label_embeddings: NDArray[np.float32],
|
||||
label_names: list[str],
|
||||
z_threshold: float = LABEL_Z_THRESHOLD,
|
||||
z_margin: float = LABEL_Z_MARGIN,
|
||||
std_floor: float = LABEL_Z_STD_FLOOR,
|
||||
min_raw_sim: float = MIN_RAW_SIMILARITY,
|
||||
max_per_item: int = MAX_LABELS_PER_ITEM,
|
||||
) -> list[list[tuple[str, float]]]:
|
||||
"""Suggest labels for each item using z-score normalized similarity.
|
||||
|
||||
1. Compute raw cosine similarity matrix (n items x m labels).
|
||||
2. Column-wise z-score: for each label j, normalize across all items.
|
||||
3. For each item, rank labels by z-score descending.
|
||||
4. Accept a label only if z >= z_threshold AND raw_sim >= min_raw_sim.
|
||||
5. Margin gate: the top-1 label must beat #2 by z_margin; subsequent
|
||||
labels don't need a margin.
|
||||
6. Cap at max_per_item.
|
||||
|
||||
Returns a list of length n, where each element is a list of
|
||||
(label_name, raw_similarity) tuples. Empty list if nothing qualifies.
|
||||
"""
|
||||
n = item_embeddings.shape[0]
|
||||
m = label_embeddings.shape[0]
|
||||
if n == 0 or m == 0:
|
||||
return [[] for _ in range(n)]
|
||||
|
||||
# (n, m) raw similarity matrix
|
||||
sim_matrix = cosine_similarity(item_embeddings, label_embeddings)
|
||||
|
||||
# Column-wise z-score normalization
|
||||
col_means = sim_matrix.mean(axis=0) # shape (m,)
|
||||
col_stds = sim_matrix.std(axis=0) # shape (m,)
|
||||
col_stds = np.maximum(col_stds, std_floor)
|
||||
z_matrix = (sim_matrix - col_means) / col_stds
|
||||
|
||||
suggestions: list[list[tuple[str, float]]] = []
|
||||
for i in range(n):
|
||||
z_row = z_matrix[i]
|
||||
raw_row = sim_matrix[i]
|
||||
|
||||
# Rank labels by z-score descending
|
||||
ranked = np.argsort(z_row)[::-1]
|
||||
|
||||
item_labels: list[tuple[str, float]] = []
|
||||
|
||||
# Margin gate: top-1 z-score must beat #2 by z_margin.
|
||||
# If not, the assignment is ambiguous — skip this item entirely.
|
||||
if len(ranked) > 1:
|
||||
top1_z = float(z_row[ranked[0]])
|
||||
top2_z = float(z_row[ranked[1]])
|
||||
if top1_z - top2_z < z_margin:
|
||||
suggestions.append(item_labels)
|
||||
continue
|
||||
|
||||
for rank_pos, idx in enumerate(ranked):
|
||||
if len(item_labels) >= max_per_item:
|
||||
break
|
||||
|
||||
z_val = float(z_row[idx])
|
||||
raw_val = float(raw_row[idx])
|
||||
|
||||
# Must pass both z-threshold and raw similarity floor
|
||||
if z_val < z_threshold or raw_val < min_raw_sim:
|
||||
continue
|
||||
|
||||
item_labels.append((label_names[idx], raw_val))
|
||||
|
||||
suggestions.append(item_labels)
|
||||
|
||||
return suggestions
|
||||
@@ -0,0 +1,4 @@
|
||||
fastembed>=0.5.0
|
||||
numpy>=1.26.0
|
||||
scikit-learn>=1.4.0
|
||||
scipy>=1.10.0
|
||||
@@ -0,0 +1,600 @@
|
||||
"""Triage sweep: fetch open issues/PRs, detect outliers and duplicates, generate a report.
|
||||
|
||||
Entrypoint script for the triage-sweep workflow. Fetches all open items via
|
||||
the GitHub REST API, delegates embedding and analysis to embedding_utils,
|
||||
generates a markdown report, and optionally creates a report issue.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import os
|
||||
import sys
|
||||
import urllib.request
|
||||
import urllib.parse
|
||||
from typing import TypedDict
|
||||
from datetime import datetime, timezone
|
||||
|
||||
from embedding_utils import (
|
||||
embed_texts,
|
||||
normalize_rows,
|
||||
reduce_dimensions,
|
||||
detect_outliers,
|
||||
find_duplicate_pairs,
|
||||
suggest_labels,
|
||||
LABEL_Z_THRESHOLD,
|
||||
LABEL_Z_MARGIN,
|
||||
LABEL_Z_STD_FLOOR,
|
||||
MIN_RAW_SIMILARITY,
|
||||
MAX_LABELS_PER_ITEM,
|
||||
)
|
||||
|
||||
# ── Thresholds (overridable via workflow_dispatch inputs) ──────────────
|
||||
|
||||
# IQR multiplier for outlier cutoff: cutoff = Q75 + IQR_MULTIPLIER * IQR.
|
||||
IQR_MULTIPLIER: float = float(os.environ.get("INPUT_IQR_MULTIPLIER", "3.0"))
|
||||
|
||||
# Hard cap: at most this fraction of items can be flagged as outliers.
|
||||
MAX_OUTLIER_PCT: float = float(os.environ.get("INPUT_MAX_OUTLIER_PCT", "0.05"))
|
||||
|
||||
# EllipticEnvelope contamination: expected fraction of outliers in the data.
|
||||
# Governs how aggressively the robust covariance downweights extreme points.
|
||||
CONTAMINATION: float = float(os.environ.get("INPUT_CONTAMINATION", "0.1"))
|
||||
|
||||
# Cosine similarity above which two items are flagged as duplicates.
|
||||
# 0.92 catches near-identical issues while tolerating paraphrasing.
|
||||
COSINE_THRESHOLD: float = float(os.environ.get("INPUT_COSINE_THRESHOLD", "0.92"))
|
||||
|
||||
# Hard cap on items to process. Prevents runaway costs on very large repos.
|
||||
MAX_ITEMS: int = int(os.environ.get("INPUT_MAX_ITEMS", "500"))
|
||||
|
||||
# When true, print report to stdout/file but do not create a GitHub issue.
|
||||
DRY_RUN: bool = os.environ.get("INPUT_DRY_RUN", "false").lower() == "true"
|
||||
|
||||
# ── Fixed constants (not user-configurable) ───────────────────────────
|
||||
|
||||
# Minimum number of samples required for EllipticEnvelope to fit
|
||||
# a Gaussian reliably. Must be >= 3 * PCA_MAX_COMPONENTS so the
|
||||
# covariance matrix is estimated from enough data points.
|
||||
PCA_MAX_COMPONENTS: int = 20
|
||||
MIN_SAMPLES_FOR_OUTLIER_DETECTION: int = 100
|
||||
|
||||
# Max character length for embedding input text. bge-small-en-v1.5 has a
|
||||
# 512-token context window (~4 chars/token). We keep title + body under
|
||||
# this limit so the model sees the full text instead of silently truncating.
|
||||
MAX_EMBED_CHARS: int = 2000
|
||||
|
||||
# GitHub REST API page size (max allowed is 100).
|
||||
API_PAGE_SIZE: int = 100
|
||||
|
||||
# Report issue label.
|
||||
REPORT_LABEL: str = "triage-report"
|
||||
|
||||
# Report file path (written for the summary step to pick up).
|
||||
REPORT_FILE: str = "/tmp/triage-report.md"
|
||||
|
||||
|
||||
class TriageItem(TypedDict):
|
||||
"""One open issue or PR, with only the fields we need."""
|
||||
number: int
|
||||
title: str
|
||||
html_url: str
|
||||
is_pr: bool
|
||||
labels: list[str]
|
||||
created_at: str
|
||||
# title + body concatenated, used as embedding input
|
||||
text: str
|
||||
|
||||
|
||||
def github_api_get(path: str) -> list[dict]:
|
||||
"""Make a single authenticated GET request to the GitHub REST API.
|
||||
|
||||
Reads GITHUB_TOKEN and GITHUB_REPOSITORY from env. Raises SystemExit
|
||||
with the HTTP status and response body on any non-2xx response.
|
||||
"""
|
||||
token = os.environ["GITHUB_TOKEN"]
|
||||
repo = os.environ["GITHUB_REPOSITORY"]
|
||||
url = f"https://api.github.com/repos/{repo}{path}"
|
||||
|
||||
req = urllib.request.Request(url)
|
||||
req.add_header("Accept", "application/vnd.github+json")
|
||||
req.add_header("Authorization", f"Bearer {token}")
|
||||
req.add_header("X-GitHub-Api-Version", "2022-11-28")
|
||||
|
||||
try:
|
||||
with urllib.request.urlopen(req, timeout=30) as resp:
|
||||
return json.loads(resp.read().decode("utf-8"))
|
||||
except urllib.error.HTTPError as e:
|
||||
body = e.read().decode("utf-8", errors="replace")
|
||||
print(f"::error::GitHub API {e.code}: {body}")
|
||||
sys.exit(1)
|
||||
|
||||
|
||||
def fetch_all_open_items() -> list[TriageItem]:
|
||||
"""Paginate through all open issues and PRs.
|
||||
|
||||
Returns up to MAX_ITEMS TriageItem dicts. Items with a pull_request
|
||||
key are marked is_pr=True. The text field is title + body concatenated.
|
||||
"""
|
||||
items: list[TriageItem] = []
|
||||
page = 1
|
||||
|
||||
while len(items) < MAX_ITEMS:
|
||||
path = (
|
||||
f"/issues?state=open&per_page={API_PAGE_SIZE}"
|
||||
f"&sort=created&direction=desc&page={page}"
|
||||
)
|
||||
data = github_api_get(path)
|
||||
|
||||
if not data:
|
||||
break
|
||||
|
||||
for raw in data:
|
||||
if len(items) >= MAX_ITEMS:
|
||||
break
|
||||
|
||||
body = raw.get("body", "") or ""
|
||||
full_text = f"{raw['title']}\n\n{body}"
|
||||
# Truncate to fit the embedding model's token window.
|
||||
# Title is always preserved; body gets clipped if needed.
|
||||
if len(full_text) > MAX_EMBED_CHARS:
|
||||
full_text = full_text[:MAX_EMBED_CHARS]
|
||||
items.append(TriageItem(
|
||||
number=raw["number"],
|
||||
title=raw["title"],
|
||||
html_url=raw["html_url"],
|
||||
is_pr="pull_request" in raw,
|
||||
labels=[lbl["name"] for lbl in raw.get("labels", [])],
|
||||
created_at=raw["created_at"],
|
||||
text=full_text,
|
||||
))
|
||||
|
||||
if len(data) < API_PAGE_SIZE:
|
||||
break
|
||||
|
||||
page += 1
|
||||
|
||||
return items
|
||||
|
||||
|
||||
class RepoLabel(TypedDict):
|
||||
"""A label from the repo with its embedding text."""
|
||||
name: str
|
||||
description: str
|
||||
# "name: description" concatenated for embedding
|
||||
text: str
|
||||
|
||||
|
||||
def fetch_repo_labels() -> list[RepoLabel]:
|
||||
"""Fetch all labels from the repository, paginating if needed.
|
||||
|
||||
Returns labels with name, description, and a text field suitable
|
||||
for embedding ("name: description"). Labels with no description
|
||||
use just the name.
|
||||
"""
|
||||
labels: list[RepoLabel] = []
|
||||
page = 1
|
||||
|
||||
while True:
|
||||
data = github_api_get(f"/labels?per_page={API_PAGE_SIZE}&page={page}")
|
||||
for raw in data:
|
||||
name = raw["name"]
|
||||
desc = raw.get("description", "") or ""
|
||||
text = f"{name}: {desc}" if desc else name
|
||||
labels.append(RepoLabel(name=name, description=desc, text=text))
|
||||
|
||||
if len(data) < API_PAGE_SIZE:
|
||||
break
|
||||
page += 1
|
||||
|
||||
return labels
|
||||
|
||||
|
||||
def apply_labels_to_item(item_number: int, labels: list[str]) -> None:
|
||||
"""Add labels to a single issue/PR via the GitHub API.
|
||||
|
||||
Skips silently if labels list is empty. Uses POST which adds labels
|
||||
without removing existing ones.
|
||||
"""
|
||||
if not labels:
|
||||
return
|
||||
|
||||
token = os.environ["GITHUB_TOKEN"]
|
||||
repo = os.environ["GITHUB_REPOSITORY"]
|
||||
url = f"https://api.github.com/repos/{repo}/issues/{item_number}/labels"
|
||||
|
||||
payload = json.dumps({"labels": labels}).encode("utf-8")
|
||||
req = urllib.request.Request(url, data=payload, method="POST")
|
||||
req.add_header("Accept", "application/vnd.github+json")
|
||||
req.add_header("Authorization", f"Bearer {token}")
|
||||
req.add_header("X-GitHub-Api-Version", "2022-11-28")
|
||||
req.add_header("Content-Type", "application/json")
|
||||
|
||||
try:
|
||||
with urllib.request.urlopen(req, timeout=30) as resp:
|
||||
resp.read()
|
||||
except urllib.error.HTTPError as e:
|
||||
body = e.read().decode("utf-8", errors="replace")
|
||||
# Non-fatal: log warning but don't abort the sweep
|
||||
print(f"::warning::Failed to label #{item_number}: {e.code} {body}")
|
||||
|
||||
|
||||
def _item_age(created_at: str) -> str:
|
||||
"""Compute a human-readable age string from an ISO 8601 created_at timestamp."""
|
||||
try:
|
||||
created = datetime.fromisoformat(created_at.replace("Z", "+00:00"))
|
||||
delta = datetime.now(timezone.utc) - created
|
||||
days = delta.days
|
||||
if days < 1:
|
||||
return "<1d"
|
||||
if days < 30:
|
||||
return f"{days}d"
|
||||
if days < 365:
|
||||
return f"{days // 30}mo"
|
||||
return f"{days // 365}y"
|
||||
except (ValueError, TypeError):
|
||||
return "?"
|
||||
|
||||
|
||||
def _suggested_action(a: TriageItem, b: TriageItem) -> str:
|
||||
"""Determine a suggested action for a duplicate pair based on types and age."""
|
||||
if a["is_pr"] and b["is_pr"]:
|
||||
return "Review for overlap"
|
||||
if not a["is_pr"] and not b["is_pr"]:
|
||||
# Both issues — close the newer one
|
||||
try:
|
||||
a_dt = datetime.fromisoformat(a["created_at"].replace("Z", "+00:00"))
|
||||
b_dt = datetime.fromisoformat(b["created_at"].replace("Z", "+00:00"))
|
||||
newer = b if b_dt > a_dt else a
|
||||
except (ValueError, TypeError):
|
||||
newer = b
|
||||
return f"Close #{newer['number']} as duplicate"
|
||||
# One issue, one PR
|
||||
return "Link PR to issue"
|
||||
|
||||
|
||||
def generate_report(
|
||||
items: list[TriageItem],
|
||||
outlier_results: list[tuple[int, float]],
|
||||
duplicate_pairs: list[tuple[int, int, float]],
|
||||
label_suggestions: list[list[tuple[str, float]]] | None = None,
|
||||
) -> str:
|
||||
"""Generate a structured markdown triage report."""
|
||||
now = datetime.now(timezone.utc).strftime("%Y-%m-%d %H:%M:%S")
|
||||
repo = os.environ.get("GITHUB_REPOSITORY", "unknown/repo")
|
||||
|
||||
# Compute label suggestion counts early for the health table
|
||||
outlier_set = {idx for idx, _ in outlier_results}
|
||||
suggested_count = 0
|
||||
if label_suggestions is not None:
|
||||
suggested_count = sum(
|
||||
1 for i, s in enumerate(label_suggestions)
|
||||
if s and not items[i]["labels"] and i not in outlier_set
|
||||
)
|
||||
|
||||
# ── Health summary table at the top ──────────────────────────────
|
||||
lines: list[str] = [
|
||||
"## Triage Sweep Report",
|
||||
"",
|
||||
f"**Run:** {now} UTC",
|
||||
f"**Items analyzed:** {len(items)}",
|
||||
f"**Thresholds:** IQR multiplier {IQR_MULTIPLIER}, Cosine > {COSINE_THRESHOLD}",
|
||||
"",
|
||||
"### Health Summary",
|
||||
"",
|
||||
"| Metric | Value |",
|
||||
"|--------|-------|",
|
||||
f"| Items analyzed | {len(items)} |",
|
||||
f"| Outliers flagged | {len(outlier_results)} |",
|
||||
f"| Duplicate pairs | {len(duplicate_pairs)} |",
|
||||
f"| Label suggestions | {suggested_count} |",
|
||||
"",
|
||||
]
|
||||
|
||||
# ── Outlier section ──────────────────────────────────────────────
|
||||
# Determine cutoff for high-confidence split
|
||||
cutoff = getattr(outlier_results, "cutoff", 0.0)
|
||||
high_conf_cutoff = 2 * cutoff if cutoff > 0 else float("inf")
|
||||
|
||||
high_conf = [(idx, d) for idx, d in outlier_results if d > high_conf_cutoff]
|
||||
borderline = [(idx, d) for idx, d in outlier_results if d <= high_conf_cutoff]
|
||||
|
||||
lines.extend([
|
||||
f"### Potential Outliers / Spam ({len(outlier_results)})",
|
||||
"",
|
||||
"Items with unusually high Mahalanobis distance from the distribution center.",
|
||||
"These may be spam, off-topic, or poorly described.",
|
||||
"",
|
||||
])
|
||||
|
||||
if high_conf:
|
||||
lines.append(f"**High Confidence** ({len(high_conf)} items, distance > 2x cutoff)")
|
||||
lines.append("")
|
||||
lines.append("| # | Type | Title | Distance | Age |")
|
||||
lines.append("|---|------|-------|----------|-----|")
|
||||
for idx, distance in high_conf:
|
||||
item = items[idx]
|
||||
kind = "PR" if item["is_pr"] else "Issue"
|
||||
age = _item_age(item["created_at"])
|
||||
title = item["title"][:80] + ("..." if len(item["title"]) > 80 else "")
|
||||
lines.append(
|
||||
f"| [#{item['number']}]({item['html_url']}) "
|
||||
f"| {kind} | {title} | {distance:.2f} | {age} |"
|
||||
)
|
||||
lines.append("")
|
||||
|
||||
if borderline:
|
||||
lines.append("<details>")
|
||||
lines.append(f"<summary>Borderline ({len(borderline)} items)</summary>")
|
||||
lines.append("")
|
||||
lines.append("| # | Type | Title | Distance | Age |")
|
||||
lines.append("|---|------|-------|----------|-----|")
|
||||
for idx, distance in borderline:
|
||||
item = items[idx]
|
||||
kind = "PR" if item["is_pr"] else "Issue"
|
||||
age = _item_age(item["created_at"])
|
||||
title = item["title"][:80] + ("..." if len(item["title"]) > 80 else "")
|
||||
lines.append(
|
||||
f"| [#{item['number']}]({item['html_url']}) "
|
||||
f"| {kind} | {title} | {distance:.2f} | {age} |"
|
||||
)
|
||||
lines.append("")
|
||||
lines.append("</details>")
|
||||
lines.append("")
|
||||
|
||||
if not outlier_results:
|
||||
lines.append("None found.")
|
||||
|
||||
# ── Duplicate pairs section ──────────────────────────────────────
|
||||
lines.extend([
|
||||
"",
|
||||
f"### Potential Duplicates ({len(duplicate_pairs)} pairs)",
|
||||
"",
|
||||
"Pairs of items with cosine similarity above the threshold.",
|
||||
"",
|
||||
])
|
||||
|
||||
if duplicate_pairs:
|
||||
lines.append("| Item A | Item B | Similarity | Suggested Action |")
|
||||
lines.append("|--------|--------|------------|------------------|")
|
||||
for i, j, sim in duplicate_pairs:
|
||||
a = items[i]
|
||||
b = items[j]
|
||||
kind_a = "PR" if a["is_pr"] else "Issue"
|
||||
kind_b = "PR" if b["is_pr"] else "Issue"
|
||||
action = _suggested_action(a, b)
|
||||
lines.append(
|
||||
f"| [#{a['number']}]({a['html_url']}) {kind_a}: {a['title']} "
|
||||
f"| [#{b['number']}]({b['html_url']}) {kind_b}: {b['title']} "
|
||||
f"| {sim:.3f} | {action} |"
|
||||
)
|
||||
else:
|
||||
lines.append("None found.")
|
||||
|
||||
# ── Label suggestions section ────────────────────────────────────
|
||||
if label_suggestions is not None:
|
||||
# High confidence: top-1 label with raw_sim >= 0.5
|
||||
# Low confidence: top-1 label with raw_sim < 0.5
|
||||
high_conf_labels: list[tuple[int, list[tuple[str, float]]]] = []
|
||||
low_conf_labels: list[tuple[int, list[tuple[str, float]]]] = []
|
||||
for i, sugs in enumerate(label_suggestions):
|
||||
if sugs and not items[i]["labels"] and i not in outlier_set:
|
||||
top1 = sugs[:1]
|
||||
if top1[0][1] >= 0.5:
|
||||
high_conf_labels.append((i, top1))
|
||||
else:
|
||||
low_conf_labels.append((i, top1))
|
||||
|
||||
total_suggestions = len(high_conf_labels) + len(low_conf_labels)
|
||||
lines.extend([
|
||||
"",
|
||||
f"### Suggested Labels ({total_suggestions} unlabeled items)",
|
||||
"",
|
||||
"Labels suggested by z-score normalized embedding similarity against repo label descriptions.",
|
||||
"Only shown for unlabeled items that were not flagged as outliers.",
|
||||
"",
|
||||
])
|
||||
|
||||
# Label concentration warning
|
||||
if total_suggestions > 0:
|
||||
label_counts: dict[str, int] = {}
|
||||
for _, sugs in high_conf_labels + low_conf_labels:
|
||||
for name, _ in sugs:
|
||||
label_counts[name] = label_counts.get(name, 0) + 1
|
||||
for name, count in label_counts.items():
|
||||
if count > total_suggestions * 0.5:
|
||||
lines.append(
|
||||
f"> **Warning:** Label `{name}` accounts for "
|
||||
f"{count}/{total_suggestions} suggestions "
|
||||
f"({count * 100 // total_suggestions}%). "
|
||||
f"Consider reviewing label descriptions for specificity."
|
||||
)
|
||||
lines.append("")
|
||||
|
||||
if high_conf_labels:
|
||||
lines.append("| # | Type | Title | Suggested Label |")
|
||||
lines.append("|---|------|-------|--------------------|")
|
||||
for idx, sugs in high_conf_labels:
|
||||
item = items[idx]
|
||||
kind = "PR" if item["is_pr"] else "Issue"
|
||||
label_strs = [f"`{name}` ({score:.2f})" for name, score in sugs]
|
||||
lines.append(
|
||||
f"| [#{item['number']}]({item['html_url']}) "
|
||||
f"| {kind} | {item['title']} | {', '.join(label_strs)} |"
|
||||
)
|
||||
|
||||
if low_conf_labels:
|
||||
lines.append("")
|
||||
lines.append("<details>")
|
||||
lines.append(f"<summary>Low-confidence suggestions ({len(low_conf_labels)} items)</summary>")
|
||||
lines.append("")
|
||||
lines.append("| # | Type | Title | Suggested Label |")
|
||||
lines.append("|---|------|-------|--------------------|")
|
||||
for idx, sugs in low_conf_labels:
|
||||
item = items[idx]
|
||||
kind = "PR" if item["is_pr"] else "Issue"
|
||||
label_strs = [f"`{name}` ({score:.2f})" for name, score in sugs]
|
||||
lines.append(
|
||||
f"| [#{item['number']}]({item['html_url']}) "
|
||||
f"| {kind} | {item['title']} | {', '.join(label_strs)} |"
|
||||
)
|
||||
lines.append("")
|
||||
lines.append("</details>")
|
||||
|
||||
if not high_conf_labels and not low_conf_labels:
|
||||
lines.append("No unlabeled items need suggestions.")
|
||||
|
||||
lines.extend([
|
||||
"",
|
||||
"### Summary",
|
||||
"",
|
||||
f"- {len(outlier_results)} outliers flagged for review",
|
||||
f"- {len(duplicate_pairs)} duplicate pairs found",
|
||||
f"- {len(items)} items analyzed in total",
|
||||
])
|
||||
|
||||
if label_suggestions is not None:
|
||||
lines.append(f"- {suggested_count} items suggested for labeling")
|
||||
|
||||
lines.extend([
|
||||
"",
|
||||
"---",
|
||||
f"*Generated by [triage-sweep](https://github.com/{repo}/actions) — no LLM was used.*",
|
||||
])
|
||||
|
||||
return "\n".join(lines)
|
||||
|
||||
|
||||
def create_report_issue(report_body: str) -> None:
|
||||
"""Create a GitHub issue with the triage report.
|
||||
|
||||
Posts to the issues API with the triage-report label.
|
||||
Raises SystemExit on non-201 response.
|
||||
"""
|
||||
token = os.environ["GITHUB_TOKEN"]
|
||||
repo = os.environ["GITHUB_REPOSITORY"]
|
||||
url = f"https://api.github.com/repos/{repo}/issues"
|
||||
|
||||
today = datetime.now(timezone.utc).strftime("%Y-%m-%d")
|
||||
payload = json.dumps({
|
||||
"title": f"Triage Sweep Report — {today}",
|
||||
"body": report_body,
|
||||
"labels": [REPORT_LABEL],
|
||||
}).encode("utf-8")
|
||||
|
||||
req = urllib.request.Request(url, data=payload, method="POST")
|
||||
req.add_header("Accept", "application/vnd.github+json")
|
||||
req.add_header("Authorization", f"Bearer {token}")
|
||||
req.add_header("X-GitHub-Api-Version", "2022-11-28")
|
||||
req.add_header("Content-Type", "application/json")
|
||||
|
||||
try:
|
||||
with urllib.request.urlopen(req, timeout=30) as resp:
|
||||
resp_body = resp.read().decode("utf-8")
|
||||
if resp.status != 201:
|
||||
print(f"::error::Failed to create issue: {resp.status} {resp_body}")
|
||||
sys.exit(1)
|
||||
result = json.loads(resp_body)
|
||||
print(f"Created issue: {result.get('html_url', 'unknown')}")
|
||||
except urllib.error.HTTPError as e:
|
||||
body = e.read().decode("utf-8", errors="replace")
|
||||
print(f"::error::Failed to create issue: {e.code} {body}")
|
||||
sys.exit(1)
|
||||
|
||||
|
||||
def write_report(report: str) -> None:
|
||||
"""Write the report to the file system for the summary step."""
|
||||
with open(REPORT_FILE, "w", encoding="utf-8") as f:
|
||||
f.write(report)
|
||||
|
||||
|
||||
def main() -> None:
|
||||
"""Orchestrate the full triage sweep."""
|
||||
# 1. Validate environment
|
||||
for var in ("GITHUB_TOKEN", "GITHUB_REPOSITORY"):
|
||||
if not os.environ.get(var):
|
||||
print(f"::error::Missing required environment variable: {var}")
|
||||
sys.exit(1)
|
||||
|
||||
# 2. Fetch all open issues + PRs
|
||||
items = fetch_all_open_items()
|
||||
print(f"Fetched {len(items)} open items")
|
||||
|
||||
if len(items) == 0:
|
||||
report = "## Triage Sweep Report\n\nNo open issues or PRs found."
|
||||
write_report(report)
|
||||
print("No items to analyze.")
|
||||
return
|
||||
|
||||
# 3. Extract texts for embedding
|
||||
texts: list[str] = [item["text"] for item in items]
|
||||
|
||||
# 4. Embed all texts (returns numpy float32 array of shape [n, 384])
|
||||
embeddings = embed_texts(texts)
|
||||
|
||||
# 5. L2-normalize
|
||||
embeddings = normalize_rows(embeddings)
|
||||
|
||||
# 6. Outlier detection (Mahalanobis via EllipticEnvelope)
|
||||
outlier_results: list[tuple[int, float]] = []
|
||||
if len(items) >= MIN_SAMPLES_FOR_OUTLIER_DETECTION:
|
||||
reduced = reduce_dimensions(embeddings, PCA_MAX_COMPONENTS)
|
||||
outlier_results = detect_outliers(
|
||||
reduced,
|
||||
contamination=CONTAMINATION,
|
||||
iqr_multiplier=IQR_MULTIPLIER,
|
||||
max_outlier_pct=MAX_OUTLIER_PCT,
|
||||
)
|
||||
else:
|
||||
print(
|
||||
f"Skipping outlier detection: {len(items)} items < "
|
||||
f"{MIN_SAMPLES_FOR_OUTLIER_DETECTION} minimum"
|
||||
)
|
||||
|
||||
# 7. Duplicate detection (pairwise cosine similarity)
|
||||
duplicate_pairs = find_duplicate_pairs(embeddings, COSINE_THRESHOLD)
|
||||
|
||||
# 8. Label suggestion via embedding similarity
|
||||
label_suggestions: list[list[tuple[str, float]]] | None = None
|
||||
repo_labels = fetch_repo_labels()
|
||||
if repo_labels:
|
||||
label_texts = [lbl["text"] for lbl in repo_labels]
|
||||
label_names = [lbl["name"] for lbl in repo_labels]
|
||||
label_embeddings = embed_texts(label_texts)
|
||||
label_embeddings = normalize_rows(label_embeddings)
|
||||
label_suggestions = suggest_labels(embeddings, label_embeddings, label_names)
|
||||
print(f"Computed label suggestions against {len(repo_labels)} repo labels")
|
||||
|
||||
# NOTE: Auto-labeling is disabled. The report shows suggestions for
|
||||
# human review. To re-enable, uncomment the block below.
|
||||
#
|
||||
# # Apply top label to unlabeled items (unless dry run)
|
||||
# # Skip outliers — flagged items shouldn't get categorized
|
||||
# outlier_set = {idx for idx, _ in outlier_results}
|
||||
# if not DRY_RUN:
|
||||
# applied_count = 0
|
||||
# for i, sugs in enumerate(label_suggestions):
|
||||
# if sugs and not items[i]["labels"] and i not in outlier_set:
|
||||
# # Apply only the top-1 label (highest confidence)
|
||||
# apply_labels_to_item(items[i]["number"], [sugs[0][0]])
|
||||
# applied_count += 1
|
||||
# print(f"Applied labels to {applied_count} unlabeled items")
|
||||
else:
|
||||
print("No repo labels found — skipping label suggestions")
|
||||
|
||||
# 9. Generate report
|
||||
report = generate_report(items, outlier_results, duplicate_pairs, label_suggestions)
|
||||
|
||||
# 10. Write report to file (for summary step)
|
||||
write_report(report)
|
||||
|
||||
# 11. Create report issue (unless dry run)
|
||||
if DRY_RUN:
|
||||
print("Dry run — skipping issue creation and label application.")
|
||||
print(report)
|
||||
else:
|
||||
create_report_issue(report)
|
||||
print("Report issue created.")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,468 @@
|
||||
"""Tests for embedding_utils.py — all embedding model calls are mocked."""
|
||||
from __future__ import annotations
|
||||
|
||||
import sys
|
||||
from unittest.mock import patch, MagicMock
|
||||
import numpy as np
|
||||
import pytest
|
||||
|
||||
# Mock fastembed before importing the module under test (persistent)
|
||||
if "fastembed" not in sys.modules:
|
||||
sys.modules["fastembed"] = MagicMock()
|
||||
|
||||
from embedding_utils import (
|
||||
embed_texts,
|
||||
normalize_rows,
|
||||
reduce_dimensions,
|
||||
detect_outliers,
|
||||
find_duplicate_pairs,
|
||||
suggest_labels,
|
||||
EMBEDDING_DIM,
|
||||
EMBEDDING_MODEL,
|
||||
EMBEDDING_BATCH_SIZE,
|
||||
LABEL_Z_THRESHOLD,
|
||||
LABEL_Z_MARGIN,
|
||||
LABEL_Z_STD_FLOOR,
|
||||
MIN_RAW_SIMILARITY,
|
||||
MAX_LABELS_PER_ITEM,
|
||||
)
|
||||
|
||||
|
||||
class TestEmbedTexts:
|
||||
"""Tests for the embed_texts function."""
|
||||
|
||||
def test_empty_list_returns_empty_array(self):
|
||||
result = embed_texts([])
|
||||
assert result.shape == (0, EMBEDDING_DIM)
|
||||
assert result.dtype == np.float32
|
||||
|
||||
@patch("embedding_utils.TextEmbedding")
|
||||
def test_single_text(self, mock_cls):
|
||||
mock_model = MagicMock()
|
||||
mock_cls.return_value = mock_model
|
||||
vec = np.random.randn(EMBEDDING_DIM).astype(np.float32)
|
||||
mock_model.embed.return_value = iter([vec])
|
||||
|
||||
result = embed_texts(["hello world"])
|
||||
|
||||
mock_cls.assert_called_once_with(model_name=EMBEDDING_MODEL)
|
||||
mock_model.embed.assert_called_once_with(
|
||||
["hello world"], batch_size=EMBEDDING_BATCH_SIZE
|
||||
)
|
||||
assert result.shape == (1, EMBEDDING_DIM)
|
||||
assert result.dtype == np.float32
|
||||
np.testing.assert_array_almost_equal(result[0], vec)
|
||||
|
||||
@patch("embedding_utils.TextEmbedding")
|
||||
def test_multiple_texts(self, mock_cls):
|
||||
mock_model = MagicMock()
|
||||
mock_cls.return_value = mock_model
|
||||
vecs = [
|
||||
np.random.randn(EMBEDDING_DIM).astype(np.float32)
|
||||
for _ in range(5)
|
||||
]
|
||||
mock_model.embed.return_value = iter(vecs)
|
||||
|
||||
result = embed_texts(["a", "b", "c", "d", "e"])
|
||||
assert result.shape == (5, EMBEDDING_DIM)
|
||||
assert result.dtype == np.float32
|
||||
|
||||
|
||||
class TestNormalizeRows:
|
||||
"""Tests for L2 row normalization."""
|
||||
|
||||
def test_empty_matrix(self):
|
||||
m = np.empty((0, 10), dtype=np.float32)
|
||||
result = normalize_rows(m)
|
||||
assert result.shape == (0, 10)
|
||||
|
||||
def test_single_row(self):
|
||||
m = np.array([[3.0, 4.0]], dtype=np.float32)
|
||||
result = normalize_rows(m)
|
||||
# Norm should be ~1.0
|
||||
norm = np.linalg.norm(result[0])
|
||||
assert abs(norm - 1.0) < 1e-5
|
||||
|
||||
def test_multiple_rows(self):
|
||||
rng = np.random.default_rng(42)
|
||||
m = rng.standard_normal((10, 50)).astype(np.float32)
|
||||
result = normalize_rows(m)
|
||||
norms = np.linalg.norm(result, axis=1)
|
||||
np.testing.assert_allclose(norms, 1.0, atol=1e-5)
|
||||
|
||||
def test_zero_row_stays_near_zero(self):
|
||||
m = np.array([[0.0, 0.0, 0.0], [1.0, 0.0, 0.0]], dtype=np.float32)
|
||||
result = normalize_rows(m)
|
||||
# Zero row divided by eps -> very small values
|
||||
assert np.linalg.norm(result[0]) < 1e-3
|
||||
# Non-zero row should be unit norm
|
||||
assert abs(np.linalg.norm(result[1]) - 1.0) < 1e-5
|
||||
|
||||
def test_preserves_direction(self):
|
||||
m = np.array([[2.0, 0.0], [0.0, 3.0]], dtype=np.float32)
|
||||
result = normalize_rows(m)
|
||||
np.testing.assert_allclose(result[0], [1.0, 0.0], atol=1e-5)
|
||||
np.testing.assert_allclose(result[1], [0.0, 1.0], atol=1e-5)
|
||||
|
||||
|
||||
class TestReduceDimensions:
|
||||
"""Tests for PCA dimensionality reduction."""
|
||||
|
||||
def test_single_sample_returns_unchanged(self):
|
||||
m = np.random.randn(1, 50).astype(np.float32)
|
||||
result = reduce_dimensions(m, 10)
|
||||
np.testing.assert_array_equal(result, m)
|
||||
|
||||
def test_reduces_dimensions(self):
|
||||
rng = np.random.default_rng(42)
|
||||
m = rng.standard_normal((100, 50)).astype(np.float32)
|
||||
result = reduce_dimensions(m, 10)
|
||||
assert result.shape == (100, 10)
|
||||
assert result.dtype == np.float32
|
||||
|
||||
def test_caps_at_n_minus_1(self):
|
||||
rng = np.random.default_rng(42)
|
||||
# 5 samples, 20 features -> max components = 4 (n-1)
|
||||
m = rng.standard_normal((5, 20)).astype(np.float32)
|
||||
result = reduce_dimensions(m, 50)
|
||||
assert result.shape == (5, 4)
|
||||
|
||||
def test_caps_at_d(self):
|
||||
rng = np.random.default_rng(42)
|
||||
# 100 samples, 3 features -> max components = 3
|
||||
m = rng.standard_normal((100, 3)).astype(np.float32)
|
||||
result = reduce_dimensions(m, 50)
|
||||
assert result.shape == (100, 3)
|
||||
|
||||
def test_max_components_respected(self):
|
||||
rng = np.random.default_rng(42)
|
||||
m = rng.standard_normal((50, 30)).astype(np.float32)
|
||||
result = reduce_dimensions(m, 5)
|
||||
assert result.shape[1] == 5
|
||||
|
||||
|
||||
class TestDetectOutliers:
|
||||
"""Tests for IQR-based outlier detection."""
|
||||
|
||||
def test_single_sample_returns_empty(self):
|
||||
m = np.random.randn(1, 5).astype(np.float32)
|
||||
result = detect_outliers(m)
|
||||
assert result == []
|
||||
|
||||
def test_empty_returns_empty(self):
|
||||
# n < 2 case
|
||||
m = np.empty((0, 5), dtype=np.float32)
|
||||
result = detect_outliers(m)
|
||||
assert result == []
|
||||
|
||||
def test_finds_outliers_in_synthetic_data(self):
|
||||
rng = np.random.default_rng(42)
|
||||
# Create a tight cluster with one obvious outlier
|
||||
cluster = rng.standard_normal((50, 3)).astype(np.float32) * 0.1
|
||||
outlier = np.array([[100.0, 100.0, 100.0]], dtype=np.float32)
|
||||
m = np.vstack([cluster, outlier])
|
||||
result = detect_outliers(m)
|
||||
# The outlier (index 50) should be detected
|
||||
outlier_indices = [idx for idx, _ in result]
|
||||
assert 50 in outlier_indices
|
||||
|
||||
def test_returns_list_of_index_distance_tuples(self):
|
||||
rng = np.random.default_rng(42)
|
||||
# Tight cluster + outlier to guarantee at least one result
|
||||
cluster = rng.standard_normal((20, 3)).astype(np.float32) * 0.1
|
||||
far_point = np.array([[50.0, 50.0, 50.0]], dtype=np.float32)
|
||||
m = np.vstack([cluster, far_point])
|
||||
result = detect_outliers(m)
|
||||
assert isinstance(result, list)
|
||||
for item in result:
|
||||
assert isinstance(item, tuple)
|
||||
assert len(item) == 2
|
||||
idx, dist = item
|
||||
assert isinstance(idx, int)
|
||||
assert isinstance(dist, float)
|
||||
assert dist > 0
|
||||
|
||||
def test_iqr_cutoff_behavior(self):
|
||||
"""Lower IQR multiplier should flag more items than higher multiplier."""
|
||||
rng = np.random.default_rng(42)
|
||||
m = rng.standard_normal((100, 3)).astype(np.float32)
|
||||
low = detect_outliers(m, iqr_multiplier=1.0, max_outlier_pct=0.5)
|
||||
high = detect_outliers(m, iqr_multiplier=5.0, max_outlier_pct=0.5)
|
||||
assert len(low) >= len(high)
|
||||
|
||||
def test_dimension_aware_no_mass_flagging(self):
|
||||
"""High-dimensional clean Gaussian data should not flag everything."""
|
||||
rng = np.random.default_rng(42)
|
||||
# 500 samples, 10 dims — well-conditioned for robust covariance
|
||||
m = rng.standard_normal((500, 10)).astype(np.float32)
|
||||
result = detect_outliers(m)
|
||||
# With IQR-based cutoff on clean Gaussian data,
|
||||
# only a small fraction should be flagged (well under 50%)
|
||||
assert len(result) < 250
|
||||
|
||||
def test_contamination_parameter(self):
|
||||
rng = np.random.default_rng(42)
|
||||
m = rng.standard_normal((50, 3)).astype(np.float32)
|
||||
# Should not raise with different contamination values
|
||||
result = detect_outliers(m, contamination=0.05)
|
||||
assert isinstance(result, list)
|
||||
|
||||
def test_max_outlier_pct_hard_cap(self):
|
||||
"""The hard cap should limit outlier count to max_outlier_pct * n."""
|
||||
rng = np.random.default_rng(42)
|
||||
# Create data with many potential outliers (bimodal)
|
||||
cluster = rng.standard_normal((80, 3)).astype(np.float32) * 0.1
|
||||
outliers = rng.standard_normal((20, 3)).astype(np.float32) * 50.0
|
||||
m = np.vstack([cluster, outliers])
|
||||
# Very low IQR multiplier to flag a lot, but cap at 5%
|
||||
result = detect_outliers(m, iqr_multiplier=0.5, max_outlier_pct=0.05)
|
||||
max_allowed = max(1, int(0.05 * 100)) # 5
|
||||
assert len(result) <= max_allowed
|
||||
|
||||
def test_hard_cap_keeps_most_extreme(self):
|
||||
"""When capped, the most extreme items (highest distance) should be kept."""
|
||||
rng = np.random.default_rng(42)
|
||||
cluster = rng.standard_normal((90, 3)).astype(np.float32) * 0.1
|
||||
# Create outliers with increasing extremity
|
||||
outliers = np.array([
|
||||
[10.0, 10.0, 10.0],
|
||||
[20.0, 20.0, 20.0],
|
||||
[50.0, 50.0, 50.0],
|
||||
], dtype=np.float32)
|
||||
m = np.vstack([cluster, outliers])
|
||||
# Cap at ~1 item (0.01 * 93 = 0, but min is 1)
|
||||
result = detect_outliers(m, iqr_multiplier=0.5, max_outlier_pct=0.02)
|
||||
if len(result) > 0:
|
||||
# The most extreme (index 92, distance for [50,50,50]) should be kept
|
||||
indices = [idx for idx, _ in result]
|
||||
assert 92 in indices
|
||||
|
||||
def test_cutoff_attribute(self):
|
||||
"""Returned result should carry a cutoff attribute."""
|
||||
rng = np.random.default_rng(42)
|
||||
m = rng.standard_normal((50, 3)).astype(np.float32)
|
||||
result = detect_outliers(m)
|
||||
assert hasattr(result, "cutoff")
|
||||
assert isinstance(result.cutoff, float)
|
||||
assert result.cutoff > 0
|
||||
|
||||
|
||||
class TestFindDuplicatePairs:
|
||||
"""Tests for cosine similarity duplicate detection."""
|
||||
|
||||
def test_single_item_returns_empty(self):
|
||||
m = np.random.randn(1, 10).astype(np.float32)
|
||||
result = find_duplicate_pairs(m, 0.9)
|
||||
assert result == []
|
||||
|
||||
def test_empty_returns_empty(self):
|
||||
m = np.empty((0, 10), dtype=np.float32)
|
||||
result = find_duplicate_pairs(m, 0.9)
|
||||
assert result == []
|
||||
|
||||
def test_identical_vectors_detected(self):
|
||||
vec = np.random.randn(10).astype(np.float32)
|
||||
vec = vec / np.linalg.norm(vec)
|
||||
m = np.vstack([vec, vec, np.random.randn(10).astype(np.float32)])
|
||||
result = find_duplicate_pairs(m, 0.99)
|
||||
# Items 0 and 1 are identical, should be found
|
||||
assert any(i == 0 and j == 1 for i, j, _ in result)
|
||||
|
||||
def test_orthogonal_vectors_not_detected(self):
|
||||
m = np.eye(5, dtype=np.float32)
|
||||
result = find_duplicate_pairs(m, 0.5)
|
||||
assert result == []
|
||||
|
||||
def test_returns_correct_format(self):
|
||||
vec = np.random.randn(10).astype(np.float32)
|
||||
vec = vec / np.linalg.norm(vec)
|
||||
m = np.vstack([vec, vec])
|
||||
result = find_duplicate_pairs(m, 0.5)
|
||||
assert len(result) >= 1
|
||||
for item in result:
|
||||
assert len(item) == 3
|
||||
i, j, sim = item
|
||||
assert isinstance(i, int)
|
||||
assert isinstance(j, int)
|
||||
assert isinstance(sim, float)
|
||||
assert i < j
|
||||
|
||||
def test_i_less_than_j(self):
|
||||
rng = np.random.default_rng(42)
|
||||
# Create some similar vectors
|
||||
base = rng.standard_normal(10).astype(np.float32)
|
||||
m = np.vstack([base + rng.standard_normal(10) * 0.01 for _ in range(5)])
|
||||
result = find_duplicate_pairs(m, 0.5)
|
||||
for i, j, _ in result:
|
||||
assert i < j
|
||||
|
||||
def test_high_threshold_fewer_pairs(self):
|
||||
rng = np.random.default_rng(42)
|
||||
m = rng.standard_normal((10, 20)).astype(np.float32)
|
||||
# Normalize for meaningful cosine similarities
|
||||
norms = np.linalg.norm(m, axis=1, keepdims=True)
|
||||
m = m / norms
|
||||
low = find_duplicate_pairs(m, 0.3)
|
||||
high = find_duplicate_pairs(m, 0.9)
|
||||
assert len(low) >= len(high)
|
||||
|
||||
|
||||
class TestSuggestLabels:
|
||||
"""Tests for z-score normalized label suggestion."""
|
||||
|
||||
def test_empty_items_returns_empty_lists(self):
|
||||
items = np.empty((0, 10), dtype=np.float32)
|
||||
labels = np.random.randn(3, 10).astype(np.float32)
|
||||
result = suggest_labels(items, labels, ["a", "b", "c"])
|
||||
assert result == []
|
||||
|
||||
def test_empty_labels_returns_empty_per_item(self):
|
||||
items = np.random.randn(5, 10).astype(np.float32)
|
||||
labels = np.empty((0, 10), dtype=np.float32)
|
||||
result = suggest_labels(items, labels, [])
|
||||
assert len(result) == 5
|
||||
assert all(s == [] for s in result)
|
||||
|
||||
def test_identical_embedding_gets_that_label(self):
|
||||
"""If an item embedding strongly matches one label, z-score should highlight it."""
|
||||
# Create multiple items so z-score normalization is meaningful
|
||||
rng = np.random.default_rng(42)
|
||||
# 10 random items + 1 item that matches label "bug" exactly
|
||||
random_items = rng.standard_normal((10, 3)).astype(np.float32)
|
||||
bug_vec = np.array([[1.0, 0.0, 0.0]], dtype=np.float32)
|
||||
items = np.vstack([random_items, bug_vec])
|
||||
labels = np.array([[1.0, 0.0, 0.0], [0.0, 1.0, 0.0], [0.0, 0.0, 1.0]], dtype=np.float32)
|
||||
result = suggest_labels(
|
||||
items, labels, ["bug", "feature", "docs"],
|
||||
z_threshold=0.5, z_margin=0.0, min_raw_sim=0.1,
|
||||
)
|
||||
# The last item (matching bug_vec) should get "bug" as top suggestion
|
||||
last_item_sugs = result[-1]
|
||||
if last_item_sugs:
|
||||
assert last_item_sugs[0][0] == "bug"
|
||||
|
||||
def test_z_score_suppresses_dominant_label(self):
|
||||
"""When all items are similar to one label, z-scores should be low
|
||||
(none stands out) and that label should not be blindly suggested."""
|
||||
# All items identical — z-score for every item on every label is 0
|
||||
items = np.ones((10, 3), dtype=np.float32)
|
||||
labels = np.array([[1.0, 1.0, 1.0], [0.0, 1.0, 0.0]], dtype=np.float32)
|
||||
result = suggest_labels(
|
||||
items, labels, ["catch-all", "specific"],
|
||||
z_threshold=1.5, min_raw_sim=0.3,
|
||||
)
|
||||
# With identical items, std=0 -> z-scores are all 0 -> nothing passes z_threshold
|
||||
for sugs in result:
|
||||
assert sugs == []
|
||||
|
||||
def test_margin_gate_blocks_top1(self):
|
||||
"""Top-1 label must beat #2 by z_margin to be accepted as position 0."""
|
||||
rng = np.random.default_rng(99)
|
||||
# 20 items, each slightly different, 2 labels
|
||||
items = rng.standard_normal((20, 5)).astype(np.float32)
|
||||
# Two labels that are nearly identical -> margin gate should block top-1
|
||||
labels = np.array([[1.0, 0.5, 0.0, 0.0, 0.0],
|
||||
[1.0, 0.5, 0.01, 0.0, 0.0]], dtype=np.float32)
|
||||
result = suggest_labels(
|
||||
items, labels, ["label-a", "label-b"],
|
||||
z_threshold=0.0, z_margin=10.0, min_raw_sim=0.0, max_per_item=1,
|
||||
)
|
||||
# With a huge margin requirement and max_per_item=1, nothing should pass
|
||||
# because the only candidate (top-1) is blocked by margin gate,
|
||||
# and max_per_item=1 prevents falling through to position 2
|
||||
for sugs in result:
|
||||
assert sugs == []
|
||||
|
||||
def test_margin_gate_passes_when_clear_winner(self):
|
||||
"""When top-1 clearly beats #2, it should pass the margin gate."""
|
||||
# Create items where one strongly matches label 0 vs label 1
|
||||
items = np.array([
|
||||
[1.0, 0.0, 0.0, 0.0, 0.0], # strongly matches label-a
|
||||
[0.0, 0.0, 0.0, 0.0, 1.0], # matches neither well
|
||||
] * 5, dtype=np.float32) # 10 items for stable z-scores
|
||||
labels = np.array([
|
||||
[1.0, 0.0, 0.0, 0.0, 0.0], # label-a
|
||||
[0.0, 1.0, 0.0, 0.0, 0.0], # label-b (orthogonal)
|
||||
], dtype=np.float32)
|
||||
result = suggest_labels(
|
||||
items, labels, ["label-a", "label-b"],
|
||||
z_threshold=0.5, z_margin=0.3, min_raw_sim=0.1,
|
||||
)
|
||||
# Items matching label-a should get it suggested (clear z-score advantage)
|
||||
got_label_a = sum(1 for sugs in result if sugs and sugs[0][0] == "label-a")
|
||||
assert got_label_a > 0
|
||||
|
||||
def test_min_raw_similarity_filter(self):
|
||||
"""Even with high z-score, low raw similarity should be filtered out."""
|
||||
# Items are orthogonal to all labels -> raw similarity near 0
|
||||
items = np.array([[1.0, 0.0, 0.0]], dtype=np.float32)
|
||||
labels = np.array([[0.0, 0.0, 1.0]], dtype=np.float32)
|
||||
result = suggest_labels(
|
||||
items, labels, ["irrelevant"],
|
||||
z_threshold=0.0, z_margin=0.0, min_raw_sim=0.9,
|
||||
)
|
||||
# Raw similarity is ~0, which is below min_raw_sim=0.9
|
||||
assert result[0] == []
|
||||
|
||||
def test_max_per_item_respected(self):
|
||||
"""Even if many labels qualify, max_per_item caps the results."""
|
||||
rng = np.random.default_rng(42)
|
||||
# Create items with some variance so z-scores differentiate
|
||||
items = rng.standard_normal((20, 10)).astype(np.float32)
|
||||
base = items[0]
|
||||
# All labels very similar to item 0
|
||||
labels = np.array([base + rng.standard_normal(10) * 0.01 for _ in range(10)])
|
||||
names = [f"label-{i}" for i in range(10)]
|
||||
result = suggest_labels(
|
||||
items, labels, names,
|
||||
z_threshold=0.0, z_margin=0.0, min_raw_sim=0.0, max_per_item=2,
|
||||
)
|
||||
for sugs in result:
|
||||
assert len(sugs) <= 2
|
||||
|
||||
def test_returns_raw_similarity_not_z_score(self):
|
||||
"""Returned scores should be raw cosine similarity, not z-scores."""
|
||||
rng = np.random.default_rng(42)
|
||||
items = rng.standard_normal((15, 5)).astype(np.float32)
|
||||
labels = rng.standard_normal((3, 5)).astype(np.float32)
|
||||
names = ["bug", "feature", "docs"]
|
||||
result = suggest_labels(
|
||||
items, labels, names,
|
||||
z_threshold=0.0, z_margin=0.0, min_raw_sim=-1.0,
|
||||
)
|
||||
# Raw cosine similarity should be in [-1, 1] range
|
||||
for sugs in result:
|
||||
for name, score in sugs:
|
||||
assert -1.0 <= score <= 1.0 + 1e-5
|
||||
assert isinstance(name, str)
|
||||
assert isinstance(score, float)
|
||||
|
||||
def test_returns_correct_format(self):
|
||||
rng = np.random.default_rng(42)
|
||||
items = rng.standard_normal((3, 10)).astype(np.float32)
|
||||
labels = rng.standard_normal((5, 10)).astype(np.float32)
|
||||
names = ["bug", "feature", "docs", "ci", "test"]
|
||||
result = suggest_labels(
|
||||
items, labels, names,
|
||||
z_threshold=0.0, z_margin=0.0, min_raw_sim=-1.0,
|
||||
)
|
||||
assert len(result) == 3
|
||||
for sugs in result:
|
||||
for name, score in sugs:
|
||||
assert isinstance(name, str)
|
||||
assert isinstance(score, float)
|
||||
assert name in names
|
||||
|
||||
def test_text_truncation_in_labels(self):
|
||||
"""Label names should be returned as-is even when very long."""
|
||||
rng = np.random.default_rng(42)
|
||||
items = rng.standard_normal((10, 5)).astype(np.float32)
|
||||
long_name = "a" * 200
|
||||
labels = rng.standard_normal((1, 5)).astype(np.float32)
|
||||
result = suggest_labels(
|
||||
items, labels, [long_name],
|
||||
z_threshold=0.0, z_margin=0.0, min_raw_sim=-1.0,
|
||||
)
|
||||
for sugs in result:
|
||||
if sugs:
|
||||
assert sugs[0][0] == long_name
|
||||
@@ -0,0 +1,873 @@
|
||||
"""Tests for sweep.py — all external calls (API, embedding) are mocked."""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import os
|
||||
import sys
|
||||
from io import BytesIO
|
||||
from unittest.mock import patch, MagicMock, mock_open
|
||||
from urllib.error import HTTPError
|
||||
|
||||
import numpy as np
|
||||
import pytest
|
||||
|
||||
# Mock fastembed before importing sweep (which imports embedding_utils)
|
||||
sys.modules["fastembed"] = MagicMock()
|
||||
|
||||
# Set required env vars before importing sweep (module-level constants read env)
|
||||
os.environ.setdefault("GITHUB_TOKEN", "test-token")
|
||||
os.environ.setdefault("GITHUB_REPOSITORY", "owner/repo")
|
||||
|
||||
from sweep import (
|
||||
github_api_get,
|
||||
fetch_all_open_items,
|
||||
fetch_repo_labels,
|
||||
apply_labels_to_item,
|
||||
generate_report,
|
||||
create_report_issue,
|
||||
write_report,
|
||||
main,
|
||||
TriageItem,
|
||||
RepoLabel,
|
||||
REPORT_FILE,
|
||||
REPORT_LABEL,
|
||||
API_PAGE_SIZE,
|
||||
MIN_SAMPLES_FOR_OUTLIER_DETECTION,
|
||||
PCA_MAX_COMPONENTS,
|
||||
MAX_EMBED_CHARS,
|
||||
IQR_MULTIPLIER,
|
||||
MAX_OUTLIER_PCT,
|
||||
_item_age,
|
||||
_suggested_action,
|
||||
)
|
||||
|
||||
|
||||
def _make_api_issue(number: int, title: str = "Test issue", is_pr: bool = False,
|
||||
body: str = "Issue body", labels: list[str] | None = None,
|
||||
created_at: str = "2026-03-21T00:00:00Z") -> dict:
|
||||
"""Helper to build a mock GitHub API issue response object."""
|
||||
result: dict = {
|
||||
"number": number,
|
||||
"title": title,
|
||||
"html_url": f"https://github.com/owner/repo/issues/{number}",
|
||||
"body": body,
|
||||
"created_at": created_at,
|
||||
"labels": [{"name": lbl} for lbl in (labels or [])],
|
||||
}
|
||||
if is_pr:
|
||||
result["pull_request"] = {"url": "..."}
|
||||
return result
|
||||
|
||||
|
||||
class TestGithubApiGet:
|
||||
"""Tests for the github_api_get function."""
|
||||
|
||||
@patch("sweep.urllib.request.urlopen")
|
||||
def test_successful_request(self, mock_urlopen):
|
||||
mock_resp = MagicMock()
|
||||
mock_resp.read.return_value = json.dumps([{"id": 1}]).encode()
|
||||
mock_resp.__enter__ = lambda s: s
|
||||
mock_resp.__exit__ = MagicMock(return_value=False)
|
||||
mock_urlopen.return_value = mock_resp
|
||||
|
||||
result = github_api_get("/issues?state=open")
|
||||
assert result == [{"id": 1}]
|
||||
|
||||
@patch("sweep.urllib.request.urlopen")
|
||||
def test_http_error_exits(self, mock_urlopen):
|
||||
error = HTTPError(
|
||||
url="https://api.github.com/repos/owner/repo/issues",
|
||||
code=403,
|
||||
msg="Forbidden",
|
||||
hdrs=None, # type: ignore[arg-type]
|
||||
fp=BytesIO(b'{"message": "rate limited"}'),
|
||||
)
|
||||
mock_urlopen.side_effect = error
|
||||
|
||||
with pytest.raises(SystemExit) as exc_info:
|
||||
github_api_get("/issues")
|
||||
assert exc_info.value.code == 1
|
||||
|
||||
|
||||
class TestConstants:
|
||||
"""Tests for module-level constants."""
|
||||
|
||||
def test_min_samples_is_at_least_3x_pca_max(self):
|
||||
"""MIN_SAMPLES must be >= 3 * PCA_MAX_COMPONENTS for reliable covariance."""
|
||||
assert MIN_SAMPLES_FOR_OUTLIER_DETECTION >= 3 * PCA_MAX_COMPONENTS
|
||||
|
||||
def test_min_samples_is_100(self):
|
||||
assert MIN_SAMPLES_FOR_OUTLIER_DETECTION == 100
|
||||
|
||||
def test_pca_max_components_is_20(self):
|
||||
assert PCA_MAX_COMPONENTS == 20
|
||||
|
||||
def test_iqr_multiplier_default(self):
|
||||
assert IQR_MULTIPLIER == 3.0
|
||||
|
||||
def test_max_outlier_pct_default(self):
|
||||
assert MAX_OUTLIER_PCT == 0.05
|
||||
|
||||
|
||||
class TestFetchAllOpenItems:
|
||||
"""Tests for fetch_all_open_items."""
|
||||
|
||||
@patch("sweep.github_api_get")
|
||||
def test_empty_repo(self, mock_get):
|
||||
mock_get.return_value = []
|
||||
items = fetch_all_open_items()
|
||||
assert items == []
|
||||
|
||||
@patch("sweep.github_api_get")
|
||||
def test_single_page(self, mock_get):
|
||||
mock_get.return_value = [
|
||||
_make_api_issue(1, "Bug report"),
|
||||
_make_api_issue(2, "Feature request", is_pr=True),
|
||||
]
|
||||
items = fetch_all_open_items()
|
||||
assert len(items) == 2
|
||||
assert items[0]["number"] == 1
|
||||
assert items[0]["is_pr"] is False
|
||||
assert items[1]["is_pr"] is True
|
||||
|
||||
@patch("sweep.github_api_get")
|
||||
def test_text_field_constructed(self, mock_get):
|
||||
mock_get.return_value = [
|
||||
_make_api_issue(1, "My Title", body="My Body"),
|
||||
]
|
||||
items = fetch_all_open_items()
|
||||
assert items[0]["text"] == "My Title\n\nMy Body"
|
||||
|
||||
@patch("sweep.github_api_get")
|
||||
def test_long_body_truncated(self, mock_get):
|
||||
"""Bodies exceeding MAX_EMBED_CHARS are truncated to fit the token window."""
|
||||
long_body = "x" * (MAX_EMBED_CHARS + 500)
|
||||
mock_get.return_value = [
|
||||
_make_api_issue(1, "Title", body=long_body),
|
||||
]
|
||||
items = fetch_all_open_items()
|
||||
assert len(items[0]["text"]) == MAX_EMBED_CHARS
|
||||
|
||||
@patch("sweep.github_api_get")
|
||||
def test_short_body_not_truncated(self, mock_get):
|
||||
"""Bodies under the limit are left intact."""
|
||||
mock_get.return_value = [
|
||||
_make_api_issue(1, "Title", body="Short body"),
|
||||
]
|
||||
items = fetch_all_open_items()
|
||||
assert items[0]["text"] == "Title\n\nShort body"
|
||||
|
||||
@patch("sweep.github_api_get")
|
||||
def test_null_body_handled(self, mock_get):
|
||||
issue = _make_api_issue(1, "No body")
|
||||
issue["body"] = None
|
||||
mock_get.return_value = [issue]
|
||||
items = fetch_all_open_items()
|
||||
assert items[0]["text"] == "No body\n\n"
|
||||
|
||||
@patch("sweep.github_api_get")
|
||||
def test_labels_extracted(self, mock_get):
|
||||
mock_get.return_value = [
|
||||
_make_api_issue(1, "Labeled", labels=["bug", "high-priority"]),
|
||||
]
|
||||
items = fetch_all_open_items()
|
||||
assert items[0]["labels"] == ["bug", "high-priority"]
|
||||
|
||||
@patch("sweep.MAX_ITEMS", 3)
|
||||
@patch("sweep.github_api_get")
|
||||
def test_max_items_cap(self, mock_get):
|
||||
mock_get.return_value = [_make_api_issue(i) for i in range(100)]
|
||||
items = fetch_all_open_items()
|
||||
assert len(items) == 3
|
||||
|
||||
@patch("sweep.API_PAGE_SIZE", 2)
|
||||
@patch("sweep.github_api_get")
|
||||
def test_pagination(self, mock_get):
|
||||
# First page: 2 items (full page), second page: 1 item (partial -> stop)
|
||||
mock_get.side_effect = [
|
||||
[_make_api_issue(1), _make_api_issue(2)],
|
||||
[_make_api_issue(3)],
|
||||
]
|
||||
items = fetch_all_open_items()
|
||||
assert len(items) == 3
|
||||
assert mock_get.call_count == 2
|
||||
|
||||
|
||||
class TestItemAge:
|
||||
"""Tests for _item_age helper."""
|
||||
|
||||
def test_recent_item(self):
|
||||
from datetime import datetime, timezone, timedelta
|
||||
recent = (datetime.now(timezone.utc) - timedelta(hours=12)).isoformat()
|
||||
assert _item_age(recent) == "<1d"
|
||||
|
||||
def test_days_old(self):
|
||||
from datetime import datetime, timezone, timedelta
|
||||
old = (datetime.now(timezone.utc) - timedelta(days=15)).isoformat()
|
||||
assert _item_age(old) == "15d"
|
||||
|
||||
def test_months_old(self):
|
||||
from datetime import datetime, timezone, timedelta
|
||||
old = (datetime.now(timezone.utc) - timedelta(days=90)).isoformat()
|
||||
assert _item_age(old) == "3mo"
|
||||
|
||||
def test_years_old(self):
|
||||
from datetime import datetime, timezone, timedelta
|
||||
old = (datetime.now(timezone.utc) - timedelta(days=400)).isoformat()
|
||||
assert _item_age(old) == "1y"
|
||||
|
||||
def test_invalid_date(self):
|
||||
assert _item_age("not-a-date") == "?"
|
||||
|
||||
|
||||
class TestSuggestedAction:
|
||||
"""Tests for _suggested_action helper."""
|
||||
|
||||
def test_both_issues_close_newer(self):
|
||||
a = TriageItem(
|
||||
number=1, title="A", html_url="u", is_pr=False, labels=[],
|
||||
created_at="2026-01-01T00:00:00Z", text="t",
|
||||
)
|
||||
b = TriageItem(
|
||||
number=2, title="B", html_url="u", is_pr=False, labels=[],
|
||||
created_at="2026-02-01T00:00:00Z", text="t",
|
||||
)
|
||||
result = _suggested_action(a, b)
|
||||
assert "Close #2 as duplicate" in result
|
||||
|
||||
def test_both_prs_review(self):
|
||||
a = TriageItem(
|
||||
number=1, title="A", html_url="u", is_pr=True, labels=[],
|
||||
created_at="2026-01-01T00:00:00Z", text="t",
|
||||
)
|
||||
b = TriageItem(
|
||||
number=2, title="B", html_url="u", is_pr=True, labels=[],
|
||||
created_at="2026-01-01T00:00:00Z", text="t",
|
||||
)
|
||||
assert _suggested_action(a, b) == "Review for overlap"
|
||||
|
||||
def test_issue_pr_link(self):
|
||||
a = TriageItem(
|
||||
number=1, title="A", html_url="u", is_pr=False, labels=[],
|
||||
created_at="2026-01-01T00:00:00Z", text="t",
|
||||
)
|
||||
b = TriageItem(
|
||||
number=2, title="B", html_url="u", is_pr=True, labels=[],
|
||||
created_at="2026-01-01T00:00:00Z", text="t",
|
||||
)
|
||||
assert _suggested_action(a, b) == "Link PR to issue"
|
||||
|
||||
|
||||
class TestGenerateReport:
|
||||
"""Tests for the markdown report generator."""
|
||||
|
||||
def test_no_findings(self):
|
||||
items = [
|
||||
TriageItem(
|
||||
number=1, title="Test", html_url="https://example.com/1",
|
||||
is_pr=False, labels=[], created_at="2026-01-01T00:00:00Z", text="Test",
|
||||
),
|
||||
]
|
||||
report = generate_report(items, [], [])
|
||||
assert "## Triage Sweep Report" in report
|
||||
assert "Items analyzed:** 1" in report
|
||||
assert "None found." in report
|
||||
assert "0 outliers flagged" in report
|
||||
assert "0 duplicate pairs found" in report
|
||||
|
||||
def test_health_summary_table(self):
|
||||
items = [
|
||||
TriageItem(
|
||||
number=1, title="Test", html_url="https://example.com/1",
|
||||
is_pr=False, labels=[], created_at="2026-01-01T00:00:00Z", text="Test",
|
||||
),
|
||||
]
|
||||
report = generate_report(items, [], [])
|
||||
assert "### Health Summary" in report
|
||||
assert "| Metric | Value |" in report
|
||||
assert "| Items analyzed | 1 |" in report
|
||||
|
||||
def test_iqr_multiplier_in_thresholds(self):
|
||||
"""Report should show IQR multiplier, not percentile."""
|
||||
items = [
|
||||
TriageItem(
|
||||
number=1, title="Test", html_url="https://example.com/1",
|
||||
is_pr=False, labels=[], created_at="2026-01-01T00:00:00Z", text="Test",
|
||||
),
|
||||
]
|
||||
report = generate_report(items, [], [])
|
||||
assert "IQR multiplier" in report
|
||||
assert "percentile" not in report.lower().split("thresholds")[0] # not in thresholds line
|
||||
|
||||
def test_with_outliers_shows_distance_and_age(self):
|
||||
items = [
|
||||
TriageItem(
|
||||
number=10, title="Spam Issue", html_url="https://example.com/10",
|
||||
is_pr=False, labels=[], created_at="2026-01-01T00:00:00Z", text="spam",
|
||||
),
|
||||
TriageItem(
|
||||
number=20, title="Good Issue", html_url="https://example.com/20",
|
||||
is_pr=False, labels=[], created_at="2026-01-01T00:00:00Z", text="good",
|
||||
),
|
||||
]
|
||||
report = generate_report(items, [(0, 12.34)], [])
|
||||
assert "#10" in report
|
||||
assert "Spam Issue" in report
|
||||
assert "12.34" in report
|
||||
assert "1 outliers flagged" in report
|
||||
# Age column should be present
|
||||
assert "| Age |" in report
|
||||
|
||||
def test_outlier_borderline_in_details(self):
|
||||
"""Borderline outliers should be in a <details> section."""
|
||||
from embedding_utils import _OutlierResult
|
||||
items = [
|
||||
TriageItem(
|
||||
number=10, title="Borderline", html_url="https://example.com/10",
|
||||
is_pr=False, labels=[], created_at="2026-01-01T00:00:00Z", text="spam",
|
||||
),
|
||||
]
|
||||
# Create outlier results with cutoff=10.0, distance=12.0 (< 2*cutoff=20)
|
||||
outlier_results = _OutlierResult([(0, 12.0)])
|
||||
outlier_results.cutoff = 10.0
|
||||
report = generate_report(items, outlier_results, [])
|
||||
assert "<details>" in report
|
||||
assert "Borderline" in report
|
||||
|
||||
def test_outlier_high_confidence(self):
|
||||
"""Items with distance > 2x cutoff should be in high confidence section."""
|
||||
from embedding_utils import _OutlierResult
|
||||
items = [
|
||||
TriageItem(
|
||||
number=10, title="Definite Spam", html_url="https://example.com/10",
|
||||
is_pr=False, labels=[], created_at="2026-01-01T00:00:00Z", text="spam",
|
||||
),
|
||||
]
|
||||
outlier_results = _OutlierResult([(0, 25.0)])
|
||||
outlier_results.cutoff = 10.0
|
||||
report = generate_report(items, outlier_results, [])
|
||||
assert "High Confidence" in report
|
||||
|
||||
def test_with_duplicates_suggested_action(self):
|
||||
items = [
|
||||
TriageItem(
|
||||
number=1, title="First", html_url="https://example.com/1",
|
||||
is_pr=False, labels=[], created_at="2026-01-01T00:00:00Z", text="a",
|
||||
),
|
||||
TriageItem(
|
||||
number=2, title="Second", html_url="https://example.com/2",
|
||||
is_pr=True, labels=[], created_at="2026-02-01T00:00:00Z", text="b",
|
||||
),
|
||||
]
|
||||
report = generate_report(items, [], [(0, 1, 0.954)])
|
||||
assert "#1" in report
|
||||
assert "#2" in report
|
||||
assert "0.954" in report
|
||||
assert "1 duplicate pairs found" in report
|
||||
assert "Suggested Action" in report
|
||||
assert "Link PR to issue" in report
|
||||
|
||||
def test_duplicate_both_issues_close_newer(self):
|
||||
items = [
|
||||
TriageItem(
|
||||
number=1, title="First", html_url="https://example.com/1",
|
||||
is_pr=False, labels=[], created_at="2026-01-01T00:00:00Z", text="a",
|
||||
),
|
||||
TriageItem(
|
||||
number=2, title="Second", html_url="https://example.com/2",
|
||||
is_pr=False, labels=[], created_at="2026-02-01T00:00:00Z", text="b",
|
||||
),
|
||||
]
|
||||
report = generate_report(items, [], [(0, 1, 0.95)])
|
||||
assert "Close #2 as duplicate" in report
|
||||
|
||||
def test_duplicate_both_prs_review(self):
|
||||
items = [
|
||||
TriageItem(
|
||||
number=1, title="PR A", html_url="https://example.com/1",
|
||||
is_pr=True, labels=[], created_at="2026-01-01T00:00:00Z", text="a",
|
||||
),
|
||||
TriageItem(
|
||||
number=2, title="PR B", html_url="https://example.com/2",
|
||||
is_pr=True, labels=[], created_at="2026-01-01T00:00:00Z", text="b",
|
||||
),
|
||||
]
|
||||
report = generate_report(items, [], [(0, 1, 0.95)])
|
||||
assert "Review for overlap" in report
|
||||
|
||||
def test_pr_type_label(self):
|
||||
items = [
|
||||
TriageItem(
|
||||
number=5, title="PR Title", html_url="https://example.com/5",
|
||||
is_pr=True, labels=[], created_at="2026-01-01T00:00:00Z", text="pr",
|
||||
),
|
||||
]
|
||||
report = generate_report(items, [(0, 8.5)], [])
|
||||
assert "| PR |" in report
|
||||
|
||||
def test_footer_present(self):
|
||||
items = [
|
||||
TriageItem(
|
||||
number=1, title="T", html_url="u",
|
||||
is_pr=False, labels=[], created_at="2026-01-01T00:00:00Z", text="t",
|
||||
),
|
||||
]
|
||||
report = generate_report(items, [], [])
|
||||
assert "no LLM was used" in report
|
||||
|
||||
|
||||
class TestCreateReportIssue:
|
||||
"""Tests for creating the report GitHub issue."""
|
||||
|
||||
@patch("sweep.urllib.request.urlopen")
|
||||
def test_successful_creation(self, mock_urlopen):
|
||||
mock_resp = MagicMock()
|
||||
mock_resp.status = 201
|
||||
mock_resp.read.return_value = json.dumps({
|
||||
"html_url": "https://github.com/owner/repo/issues/99",
|
||||
}).encode()
|
||||
mock_resp.__enter__ = lambda s: s
|
||||
mock_resp.__exit__ = MagicMock(return_value=False)
|
||||
mock_urlopen.return_value = mock_resp
|
||||
|
||||
# Should not raise
|
||||
create_report_issue("# Test Report")
|
||||
|
||||
@patch("sweep.urllib.request.urlopen")
|
||||
def test_http_error_exits(self, mock_urlopen):
|
||||
error = HTTPError(
|
||||
url="https://api.github.com/repos/owner/repo/issues",
|
||||
code=422,
|
||||
msg="Unprocessable",
|
||||
hdrs=None, # type: ignore[arg-type]
|
||||
fp=BytesIO(b'{"message": "validation failed"}'),
|
||||
)
|
||||
mock_urlopen.side_effect = error
|
||||
|
||||
with pytest.raises(SystemExit) as exc_info:
|
||||
create_report_issue("# Test Report")
|
||||
assert exc_info.value.code == 1
|
||||
|
||||
|
||||
class TestWriteReport:
|
||||
"""Tests for the write_report helper."""
|
||||
|
||||
@patch("builtins.open", mock_open())
|
||||
def test_writes_to_file(self):
|
||||
write_report("# Report Content")
|
||||
from builtins import open as builtin_open # noqa
|
||||
# Verify open was called with the right path
|
||||
from unittest.mock import call
|
||||
open_mock = open # The patched version
|
||||
open_mock.assert_called_once_with(REPORT_FILE, "w", encoding="utf-8") # type: ignore[attr-defined]
|
||||
open_mock().write.assert_called_once_with("# Report Content") # type: ignore[attr-defined]
|
||||
|
||||
|
||||
class TestFetchRepoLabels:
|
||||
"""Tests for fetch_repo_labels."""
|
||||
|
||||
@patch("sweep.github_api_get")
|
||||
def test_fetches_and_constructs_labels(self, mock_get):
|
||||
mock_get.return_value = [
|
||||
{"name": "bug", "description": "Something isn't working"},
|
||||
{"name": "enhancement", "description": "New feature or request"},
|
||||
{"name": "docs", "description": ""},
|
||||
]
|
||||
labels = fetch_repo_labels()
|
||||
assert len(labels) == 3
|
||||
assert labels[0]["name"] == "bug"
|
||||
assert labels[0]["text"] == "bug: Something isn't working"
|
||||
assert labels[2]["text"] == "docs" # no description, just name
|
||||
|
||||
@patch("sweep.github_api_get")
|
||||
def test_empty_repo_labels(self, mock_get):
|
||||
mock_get.return_value = []
|
||||
labels = fetch_repo_labels()
|
||||
assert labels == []
|
||||
|
||||
@patch("sweep.github_api_get")
|
||||
def test_null_description_handled(self, mock_get):
|
||||
mock_get.return_value = [
|
||||
{"name": "wontfix", "description": None},
|
||||
]
|
||||
labels = fetch_repo_labels()
|
||||
assert labels[0]["text"] == "wontfix"
|
||||
|
||||
@patch("sweep.API_PAGE_SIZE", 2)
|
||||
@patch("sweep.github_api_get")
|
||||
def test_label_pagination(self, mock_get):
|
||||
"""Repos with more labels than one page should fetch all pages."""
|
||||
mock_get.side_effect = [
|
||||
# First page: full (2 items = API_PAGE_SIZE)
|
||||
[
|
||||
{"name": "bug", "description": "Broken"},
|
||||
{"name": "feature", "description": "New"},
|
||||
],
|
||||
# Second page: partial (1 item < API_PAGE_SIZE) -> stop
|
||||
[
|
||||
{"name": "docs", "description": "Documentation"},
|
||||
],
|
||||
]
|
||||
labels = fetch_repo_labels()
|
||||
assert len(labels) == 3
|
||||
assert mock_get.call_count == 2
|
||||
assert labels[0]["name"] == "bug"
|
||||
assert labels[2]["name"] == "docs"
|
||||
|
||||
|
||||
class TestApplyLabelsToItem:
|
||||
"""Tests for apply_labels_to_item."""
|
||||
|
||||
def test_empty_labels_skips(self):
|
||||
# Should not make any API call
|
||||
apply_labels_to_item(1, [])
|
||||
|
||||
@patch("sweep.urllib.request.urlopen")
|
||||
def test_successful_label_application(self, mock_urlopen):
|
||||
mock_resp = MagicMock()
|
||||
mock_resp.read.return_value = b'[{"name": "bug"}]'
|
||||
mock_resp.__enter__ = lambda s: s
|
||||
mock_resp.__exit__ = MagicMock(return_value=False)
|
||||
mock_urlopen.return_value = mock_resp
|
||||
|
||||
# Should not raise
|
||||
apply_labels_to_item(42, ["bug", "enhancement"])
|
||||
|
||||
@patch("sweep.urllib.request.urlopen")
|
||||
def test_http_error_is_non_fatal(self, mock_urlopen):
|
||||
error = HTTPError(
|
||||
url="https://api.github.com/repos/owner/repo/issues/1/labels",
|
||||
code=404,
|
||||
msg="Not Found",
|
||||
hdrs=None, # type: ignore[arg-type]
|
||||
fp=BytesIO(b'{"message": "not found"}'),
|
||||
)
|
||||
mock_urlopen.side_effect = error
|
||||
|
||||
# Should NOT raise — labeling failures are warnings, not fatal
|
||||
apply_labels_to_item(1, ["bug"])
|
||||
|
||||
|
||||
class TestGenerateReportWithLabels:
|
||||
"""Tests for label suggestions in the report."""
|
||||
|
||||
def test_report_includes_label_section_high_confidence(self):
|
||||
"""High-confidence label (raw_sim >= 0.5) should appear in main table."""
|
||||
items = [
|
||||
TriageItem(
|
||||
number=1, title="Fix crash", html_url="https://example.com/1",
|
||||
is_pr=False, labels=[], created_at="2026-01-01T00:00:00Z", text="crash",
|
||||
),
|
||||
]
|
||||
suggestions = [[("bug", 0.85)]]
|
||||
report = generate_report(items, [], [], label_suggestions=suggestions)
|
||||
assert "Suggested Labels" in report
|
||||
assert "`bug` (0.85)" in report
|
||||
assert "1 items suggested for labeling" in report
|
||||
|
||||
def test_report_low_confidence_in_details(self):
|
||||
"""Low-confidence label (raw_sim < 0.5) should be in <details> section."""
|
||||
items = [
|
||||
TriageItem(
|
||||
number=1, title="Something", html_url="https://example.com/1",
|
||||
is_pr=False, labels=[], created_at="2026-01-01T00:00:00Z", text="something",
|
||||
),
|
||||
]
|
||||
suggestions = [[("maybe-bug", 0.35)]]
|
||||
report = generate_report(items, [], [], label_suggestions=suggestions)
|
||||
assert "Low-confidence suggestions" in report
|
||||
assert "<details>" in report
|
||||
assert "`maybe-bug` (0.35)" in report
|
||||
|
||||
def test_report_skips_already_labeled_items(self):
|
||||
items = [
|
||||
TriageItem(
|
||||
number=1, title="Already labeled", html_url="https://example.com/1",
|
||||
is_pr=False, labels=["bug"], created_at="2026-01-01T00:00:00Z", text="bug",
|
||||
),
|
||||
]
|
||||
suggestions = [[("bug", 0.95)]]
|
||||
report = generate_report(items, [], [], label_suggestions=suggestions)
|
||||
assert "0 items suggested for labeling" in report
|
||||
assert "No unlabeled items" in report
|
||||
|
||||
def test_report_excludes_outliers_from_suggestions(self):
|
||||
items = [
|
||||
TriageItem(
|
||||
number=1, title="Spam garbage", html_url="https://example.com/1",
|
||||
is_pr=False, labels=[], created_at="2026-01-01T00:00:00Z", text="spam",
|
||||
),
|
||||
TriageItem(
|
||||
number=2, title="Real bug", html_url="https://example.com/2",
|
||||
is_pr=False, labels=[], created_at="2026-01-01T00:00:00Z", text="bug",
|
||||
),
|
||||
]
|
||||
suggestions = [[("bug", 0.85)], [("bug", 0.90)]]
|
||||
# Item 0 is an outlier (with distance) — should be excluded from label suggestions
|
||||
report = generate_report(items, [(0, 15.2)], [], label_suggestions=suggestions)
|
||||
assert "1 unlabeled items" in report # only item 2
|
||||
assert "#2" in report
|
||||
# Item 0 (outlier) should NOT be in the suggestions table
|
||||
assert "Spam garbage" not in report.split("Suggested Labels")[1]
|
||||
|
||||
def test_report_without_label_suggestions(self):
|
||||
items = [
|
||||
TriageItem(
|
||||
number=1, title="T", html_url="u",
|
||||
is_pr=False, labels=[], created_at="2026-01-01T00:00:00Z", text="t",
|
||||
),
|
||||
]
|
||||
report = generate_report(items, [], [], label_suggestions=None)
|
||||
assert "Suggested Labels" not in report
|
||||
|
||||
def test_label_concentration_warning(self):
|
||||
"""When >50% of suggestions point to the same label, a warning should appear."""
|
||||
items = [
|
||||
TriageItem(
|
||||
number=i, title=f"Item {i}", html_url=f"https://example.com/{i}",
|
||||
is_pr=False, labels=[], created_at="2026-01-01T00:00:00Z", text=f"text {i}",
|
||||
)
|
||||
for i in range(4)
|
||||
]
|
||||
# 3 out of 4 items get "bug" label -> 75% concentration
|
||||
suggestions = [
|
||||
[("bug", 0.85)],
|
||||
[("bug", 0.80)],
|
||||
[("bug", 0.75)],
|
||||
[("enhancement", 0.90)],
|
||||
]
|
||||
report = generate_report(items, [], [], label_suggestions=suggestions)
|
||||
assert "Warning" in report
|
||||
assert "`bug`" in report
|
||||
assert "3/4" in report
|
||||
|
||||
|
||||
class TestMain:
|
||||
"""Tests for the main orchestration function."""
|
||||
|
||||
@patch.dict(os.environ, {"GITHUB_TOKEN": "", "GITHUB_REPOSITORY": "owner/repo"})
|
||||
def test_missing_token_exits(self):
|
||||
with pytest.raises(SystemExit) as exc_info:
|
||||
main()
|
||||
assert exc_info.value.code == 1
|
||||
|
||||
@patch.dict(os.environ, {"GITHUB_TOKEN": "tok", "GITHUB_REPOSITORY": ""})
|
||||
def test_missing_repo_exits(self):
|
||||
with pytest.raises(SystemExit) as exc_info:
|
||||
main()
|
||||
assert exc_info.value.code == 1
|
||||
|
||||
@patch("sweep.write_report")
|
||||
@patch("sweep.fetch_all_open_items", return_value=[])
|
||||
def test_no_items(self, mock_fetch, mock_write):
|
||||
main()
|
||||
mock_write.assert_called_once()
|
||||
report = mock_write.call_args[0][0]
|
||||
assert "No open issues or PRs found" in report
|
||||
|
||||
@patch("sweep.create_report_issue")
|
||||
@patch("sweep.write_report")
|
||||
@patch("sweep.suggest_labels", return_value=[])
|
||||
@patch("sweep.find_duplicate_pairs", return_value=[])
|
||||
@patch("sweep.detect_outliers", return_value=[])
|
||||
@patch("sweep.reduce_dimensions")
|
||||
@patch("sweep.normalize_rows")
|
||||
@patch("sweep.embed_texts")
|
||||
@patch("sweep.fetch_repo_labels")
|
||||
@patch("sweep.fetch_all_open_items")
|
||||
def test_full_flow_with_enough_items(
|
||||
self, mock_fetch, mock_labels, mock_embed, mock_norm, mock_reduce,
|
||||
mock_outliers, mock_dupes, mock_suggest, mock_write, mock_create,
|
||||
):
|
||||
"""Test the full flow with >= MIN_SAMPLES items (outlier detection runs)."""
|
||||
n = MIN_SAMPLES_FOR_OUTLIER_DETECTION
|
||||
items = [
|
||||
TriageItem(
|
||||
number=i, title=f"Item {i}", html_url=f"https://example.com/{i}",
|
||||
is_pr=False, labels=[], created_at="2026-01-01T00:00:00Z", text=f"text {i}",
|
||||
)
|
||||
for i in range(n)
|
||||
]
|
||||
mock_fetch.return_value = items
|
||||
mock_labels.return_value = [
|
||||
RepoLabel(name="bug", description="Something broken", text="bug: Something broken"),
|
||||
]
|
||||
|
||||
embeddings = np.random.randn(n, 384).astype(np.float32)
|
||||
mock_embed.return_value = embeddings
|
||||
mock_norm.return_value = embeddings
|
||||
mock_reduce.return_value = np.random.randn(n, 10).astype(np.float32)
|
||||
|
||||
main()
|
||||
|
||||
mock_fetch.assert_called_once()
|
||||
mock_labels.assert_called_once()
|
||||
# embed_texts called twice: once for items, once for labels
|
||||
assert mock_embed.call_count == 2
|
||||
mock_norm.assert_called()
|
||||
mock_reduce.assert_called_once()
|
||||
mock_outliers.assert_called_once()
|
||||
mock_dupes.assert_called_once()
|
||||
mock_suggest.assert_called_once()
|
||||
mock_write.assert_called_once()
|
||||
mock_create.assert_called_once()
|
||||
|
||||
@patch("sweep.create_report_issue")
|
||||
@patch("sweep.write_report")
|
||||
@patch("sweep.suggest_labels", return_value=[])
|
||||
@patch("sweep.find_duplicate_pairs", return_value=[])
|
||||
@patch("sweep.detect_outliers")
|
||||
@patch("sweep.reduce_dimensions")
|
||||
@patch("sweep.normalize_rows")
|
||||
@patch("sweep.embed_texts")
|
||||
@patch("sweep.fetch_repo_labels", return_value=[])
|
||||
@patch("sweep.fetch_all_open_items")
|
||||
def test_skips_outlier_detection_for_few_items(
|
||||
self, mock_fetch, mock_labels, mock_embed, mock_norm, mock_reduce,
|
||||
mock_outliers, mock_dupes, mock_suggest, mock_write, mock_create,
|
||||
):
|
||||
"""With < MIN_SAMPLES items, outlier detection should be skipped."""
|
||||
n = MIN_SAMPLES_FOR_OUTLIER_DETECTION - 1
|
||||
items = [
|
||||
TriageItem(
|
||||
number=i, title=f"Item {i}", html_url=f"https://example.com/{i}",
|
||||
is_pr=False, labels=[], created_at="2026-01-01T00:00:00Z", text=f"text {i}",
|
||||
)
|
||||
for i in range(n)
|
||||
]
|
||||
mock_fetch.return_value = items
|
||||
|
||||
embeddings = np.random.randn(n, 384).astype(np.float32)
|
||||
mock_embed.return_value = embeddings
|
||||
mock_norm.return_value = embeddings
|
||||
|
||||
main()
|
||||
|
||||
# Outlier detection should not have been called
|
||||
mock_reduce.assert_not_called()
|
||||
mock_outliers.assert_not_called()
|
||||
# But duplicates should still be checked
|
||||
mock_dupes.assert_called_once()
|
||||
|
||||
@patch.dict(os.environ, {"INPUT_DRY_RUN": "true"})
|
||||
@patch("sweep.DRY_RUN", True)
|
||||
@patch("sweep.write_report")
|
||||
@patch("sweep.create_report_issue")
|
||||
@patch("sweep.apply_labels_to_item")
|
||||
@patch("sweep.suggest_labels", return_value=[[("bug", 0.85)]])
|
||||
@patch("sweep.find_duplicate_pairs", return_value=[])
|
||||
@patch("sweep.normalize_rows")
|
||||
@patch("sweep.embed_texts")
|
||||
@patch("sweep.fetch_repo_labels")
|
||||
@patch("sweep.fetch_all_open_items")
|
||||
def test_dry_run_skips_issue_creation_and_labeling(
|
||||
self, mock_fetch, mock_labels, mock_embed, mock_norm,
|
||||
mock_dupes, mock_suggest, mock_apply, mock_create, mock_write,
|
||||
):
|
||||
items = [
|
||||
TriageItem(
|
||||
number=1, title="Item", html_url="https://example.com/1",
|
||||
is_pr=False, labels=[], created_at="2026-01-01T00:00:00Z", text="text",
|
||||
)
|
||||
]
|
||||
mock_fetch.return_value = items
|
||||
mock_labels.return_value = [
|
||||
RepoLabel(name="bug", description="Broken", text="bug: Broken"),
|
||||
]
|
||||
embeddings = np.random.randn(1, 384).astype(np.float32)
|
||||
mock_embed.return_value = embeddings
|
||||
mock_norm.return_value = embeddings
|
||||
|
||||
main()
|
||||
|
||||
mock_create.assert_not_called()
|
||||
mock_apply.assert_not_called()
|
||||
mock_write.assert_called_once()
|
||||
|
||||
@patch("sweep.create_report_issue")
|
||||
@patch("sweep.write_report")
|
||||
@patch("sweep.apply_labels_to_item")
|
||||
@patch("sweep.suggest_labels")
|
||||
@patch("sweep.find_duplicate_pairs", return_value=[])
|
||||
@patch("sweep.normalize_rows")
|
||||
@patch("sweep.embed_texts")
|
||||
@patch("sweep.fetch_repo_labels")
|
||||
@patch("sweep.fetch_all_open_items")
|
||||
def test_labels_not_auto_applied(
|
||||
self, mock_fetch, mock_labels, mock_embed, mock_norm,
|
||||
mock_dupes, mock_suggest, mock_apply, mock_write, mock_create,
|
||||
):
|
||||
"""Auto-labeling is disabled; labels should appear in report only."""
|
||||
items = [
|
||||
TriageItem(
|
||||
number=1, title="Crash bug", html_url="https://example.com/1",
|
||||
is_pr=False, labels=[], created_at="2026-01-01T00:00:00Z", text="crash",
|
||||
),
|
||||
TriageItem(
|
||||
number=2, title="Already labeled", html_url="https://example.com/2",
|
||||
is_pr=False, labels=["enhancement"], created_at="2026-01-01T00:00:00Z", text="feat",
|
||||
),
|
||||
]
|
||||
mock_fetch.return_value = items
|
||||
mock_labels.return_value = [
|
||||
RepoLabel(name="bug", description="Broken", text="bug: Broken"),
|
||||
]
|
||||
mock_suggest.return_value = [
|
||||
[("bug", 0.90)],
|
||||
[("bug", 0.45)],
|
||||
]
|
||||
|
||||
embeddings = np.random.randn(2, 384).astype(np.float32)
|
||||
mock_embed.return_value = embeddings
|
||||
mock_norm.return_value = embeddings
|
||||
|
||||
main()
|
||||
|
||||
# Auto-labeling is disabled — apply_labels_to_item should never be called
|
||||
mock_apply.assert_not_called()
|
||||
|
||||
@patch("sweep.create_report_issue")
|
||||
@patch("sweep.write_report")
|
||||
@patch("sweep.apply_labels_to_item")
|
||||
@patch("sweep.suggest_labels")
|
||||
@patch("sweep.find_duplicate_pairs", return_value=[])
|
||||
@patch("sweep.detect_outliers")
|
||||
@patch("sweep.reduce_dimensions")
|
||||
@patch("sweep.normalize_rows")
|
||||
@patch("sweep.embed_texts")
|
||||
@patch("sweep.fetch_repo_labels")
|
||||
@patch("sweep.fetch_all_open_items")
|
||||
def test_outliers_excluded_from_report_suggestions(
|
||||
self, mock_fetch, mock_labels, mock_embed, mock_norm, mock_reduce,
|
||||
mock_outliers, mock_dupes, mock_suggest, mock_apply, mock_write, mock_create,
|
||||
):
|
||||
"""Items flagged as outliers should not appear in report label suggestions."""
|
||||
n = MIN_SAMPLES_FOR_OUTLIER_DETECTION
|
||||
items = [
|
||||
TriageItem(
|
||||
number=i, title=f"Item {i}", html_url=f"https://example.com/{i}",
|
||||
is_pr=False, labels=[], created_at="2026-01-01T00:00:00Z", text=f"text {i}",
|
||||
)
|
||||
for i in range(n)
|
||||
]
|
||||
mock_fetch.return_value = items
|
||||
mock_labels.return_value = [
|
||||
RepoLabel(name="bug", description="Broken", text="bug: Broken"),
|
||||
]
|
||||
mock_outliers.return_value = [(0, 12.5), (5, 15.3)]
|
||||
mock_suggest.return_value = [[("bug", 0.85)] for _ in range(n)]
|
||||
|
||||
embeddings = np.random.randn(n, 384).astype(np.float32)
|
||||
mock_embed.return_value = embeddings
|
||||
mock_norm.return_value = embeddings
|
||||
mock_reduce.return_value = np.random.randn(n, 10).astype(np.float32)
|
||||
|
||||
main()
|
||||
|
||||
# Auto-labeling is disabled
|
||||
mock_apply.assert_not_called()
|
||||
# Report should still be generated (outliers excluded from suggestions in report)
|
||||
mock_write.assert_called_once()
|
||||
report = mock_write.call_args[0][0]
|
||||
# Outlier items 0 and 5 should not appear in the label suggestions section
|
||||
assert "Item 0" not in report.split("Suggested Labels")[1] if "Suggested Labels" in report else True
|
||||
@@ -1,192 +0,0 @@
|
||||
name: Integration Tests
|
||||
|
||||
on:
|
||||
workflow_call:
|
||||
inputs:
|
||||
collect-coverage:
|
||||
description: 'Whether to run the coverage collection job (only needed for PR reports)'
|
||||
required: false
|
||||
default: true
|
||||
type: boolean
|
||||
|
||||
jobs:
|
||||
# ── Integration test matrix ─────────────────────────────────────────
|
||||
# Each test-group runs on a SEPARATE runner per OS, giving full process
|
||||
# isolation for the KuzuDB native C++ addon.
|
||||
# 3 OS x 4 groups = 12 parallel jobs.
|
||||
#
|
||||
# Groups:
|
||||
# kuzu-db — 7 files using withTestKuzuDB / kuzu-adapter (native addon)
|
||||
# Each file runs as its own `vitest run` invocation for full
|
||||
# process isolation. KuzuDB's native N-API addon registers
|
||||
# persistent handles that prevent fork workers from exiting
|
||||
# on Linux, and its C++ destructors segfault during
|
||||
# process.exit(). Running each file in its own process lets
|
||||
# the OS reclaim all resources cleanly.
|
||||
# pipeline — 12 files: ingestion pipeline + csv + 9 resolver tests
|
||||
# e2e — 2 files: child-process only (spawnSync), no in-process kuzu
|
||||
# standalone — 4 files: pure logic, no kuzu, no child processes
|
||||
test-matrix:
|
||||
name: integration (${{ matrix.os }} / ${{ matrix.test-group }})
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
os: [ubuntu-latest, windows-latest, macos-latest]
|
||||
test-group: [kuzu-db, pipeline, e2e, standalone]
|
||||
include:
|
||||
- test-group: kuzu-db
|
||||
# Marker — actual files are listed in the run step below
|
||||
test-glob: ''
|
||||
- test-group: pipeline
|
||||
test-glob: >-
|
||||
test/integration/pipeline.test.ts
|
||||
test/integration/csv-pipeline.test.ts
|
||||
test/integration/parsing.test.ts
|
||||
test/integration/resolvers/typescript.test.ts
|
||||
test/integration/resolvers/csharp.test.ts
|
||||
test/integration/resolvers/cpp.test.ts
|
||||
test/integration/resolvers/java.test.ts
|
||||
test/integration/resolvers/python.test.ts
|
||||
test/integration/resolvers/rust.test.ts
|
||||
test/integration/resolvers/go.test.ts
|
||||
test/integration/resolvers/kotlin.test.ts
|
||||
test/integration/resolvers/php.test.ts
|
||||
- test-group: e2e
|
||||
test-glob: >-
|
||||
test/integration/cli-e2e.test.ts
|
||||
test/integration/hooks-e2e.test.ts
|
||||
test/integration/skills-e2e.test.ts
|
||||
- test-group: standalone
|
||||
test-glob: >-
|
||||
test/integration/filesystem-walker.test.ts
|
||||
test/integration/enrichment.test.ts
|
||||
test/integration/tree-sitter-languages.test.ts
|
||||
test/integration/worker-pool.test.ts
|
||||
runs-on: ${{ matrix.os }}
|
||||
timeout-minutes: 25
|
||||
steps:
|
||||
- uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4
|
||||
- uses: ./.github/actions/setup-gitnexus
|
||||
with:
|
||||
build: 'true'
|
||||
|
||||
# kuzu-db: run each file in its own vitest process for full isolation.
|
||||
# KuzuDB's native addon hangs fork workers on Linux — process isolation
|
||||
# is the only reliable fix boundary.
|
||||
- name: Run integration tests — kuzu-db (process-isolated)
|
||||
if: matrix.test-group == 'kuzu-db'
|
||||
working-directory: gitnexus
|
||||
shell: bash
|
||||
run: |
|
||||
set -e
|
||||
files=(
|
||||
test/integration/kuzu-core-adapter.test.ts
|
||||
test/integration/kuzu-pool.test.ts
|
||||
test/integration/local-backend.test.ts
|
||||
test/integration/local-backend-calltool.test.ts
|
||||
test/integration/search-core.test.ts
|
||||
test/integration/search-pool.test.ts
|
||||
test/integration/augmentation.test.ts
|
||||
)
|
||||
exit_code=0
|
||||
for f in "${files[@]}"; do
|
||||
echo "::group::$f"
|
||||
if ! npx vitest run --reporter=verbose --pool=forks "$f"; then
|
||||
exit_code=1
|
||||
echo "::error::Test file failed: $f"
|
||||
fi
|
||||
echo "::endgroup::"
|
||||
done
|
||||
exit $exit_code
|
||||
|
||||
# Non-kuzu groups: run all files in a single vitest invocation
|
||||
- name: Run integration tests — ${{ matrix.test-group }}
|
||||
if: matrix.test-group != 'kuzu-db'
|
||||
shell: bash
|
||||
env:
|
||||
TEST_GLOB: ${{ matrix.test-glob }}
|
||||
run: npx vitest run --reporter=verbose $TEST_GLOB
|
||||
working-directory: gitnexus
|
||||
|
||||
# ── Coverage collection (ubuntu only) ─────────────────────────────────
|
||||
# Runs non-kuzu integration tests with coverage enabled so the PR report
|
||||
# can merge integration + unit coverage for a combined view.
|
||||
# kuzu-db tests are excluded because each file must run in its own vitest
|
||||
# process (native addon isolation) which prevents single-run coverage merge.
|
||||
coverage:
|
||||
name: integration (ubuntu / coverage)
|
||||
if: inputs.collect-coverage
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 15
|
||||
steps:
|
||||
- uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4
|
||||
- uses: ./.github/actions/setup-gitnexus
|
||||
with:
|
||||
build: 'true'
|
||||
|
||||
- name: Run integration tests with coverage
|
||||
working-directory: gitnexus
|
||||
run: >-
|
||||
npx vitest run
|
||||
--reporter=default
|
||||
--reporter=json
|
||||
--outputFile=integration-results.json
|
||||
--coverage
|
||||
--coverage.reporter=json-summary
|
||||
--coverage.reporter=json
|
||||
--coverage.reporter=text
|
||||
--coverage.thresholdAutoUpdate=false
|
||||
--coverage.reportOnFailure=true
|
||||
--coverage.thresholds.statements=0
|
||||
--coverage.thresholds.branches=0
|
||||
--coverage.thresholds.functions=0
|
||||
--coverage.thresholds.lines=0
|
||||
test/integration/pipeline.test.ts
|
||||
test/integration/csv-pipeline.test.ts
|
||||
test/integration/parsing.test.ts
|
||||
test/integration/cli-e2e.test.ts
|
||||
test/integration/hooks-e2e.test.ts
|
||||
test/integration/filesystem-walker.test.ts
|
||||
test/integration/enrichment.test.ts
|
||||
test/integration/tree-sitter-languages.test.ts
|
||||
test/integration/worker-pool.test.ts
|
||||
test/integration/resolvers/typescript.test.ts
|
||||
test/integration/resolvers/csharp.test.ts
|
||||
test/integration/resolvers/cpp.test.ts
|
||||
test/integration/resolvers/java.test.ts
|
||||
test/integration/resolvers/python.test.ts
|
||||
test/integration/resolvers/rust.test.ts
|
||||
test/integration/resolvers/go.test.ts
|
||||
test/integration/resolvers/kotlin.test.ts
|
||||
test/integration/resolvers/php.test.ts
|
||||
|
||||
- name: Upload integration coverage
|
||||
if: always()
|
||||
uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4
|
||||
with:
|
||||
name: integration-reports
|
||||
path: |
|
||||
gitnexus/coverage/coverage-summary.json
|
||||
gitnexus/coverage/coverage-final.json
|
||||
gitnexus/integration-results.json
|
||||
retention-days: 5
|
||||
|
||||
# ── Unified status gate ──────────────────────────────────────────────
|
||||
# Branch protection should require THIS job, not the matrix jobs directly.
|
||||
# ci.yml's needs.integration.result aggregates through this gate.
|
||||
status:
|
||||
name: integration (all groups)
|
||||
needs: test-matrix
|
||||
if: always()
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 5
|
||||
steps:
|
||||
- name: Check all matrix jobs passed
|
||||
shell: bash
|
||||
env:
|
||||
RESULT: ${{ needs.test-matrix.result }}
|
||||
run: |
|
||||
if [[ "$RESULT" != "success" ]]; then
|
||||
echo "::error::Integration matrix failed or cancelled: $RESULT"
|
||||
exit 1
|
||||
fi
|
||||
+302
-386
@@ -1,432 +1,348 @@
|
||||
name: CI Report
|
||||
|
||||
# Triggered after the CI workflow completes. Because workflow_run
|
||||
# always runs code from the *default branch*, it receives a read/write
|
||||
# GITHUB_TOKEN — even when the triggering PR comes from a fork.
|
||||
|
||||
on:
|
||||
workflow_run:
|
||||
workflows: ["CI"]
|
||||
workflows: ['CI']
|
||||
types: [completed]
|
||||
|
||||
permissions:
|
||||
actions: read # needed to list/download workflow run artifacts
|
||||
contents: read # needed for sparse checkout of vitest.config.ts
|
||||
pull-requests: write # needed to post sticky PR comment
|
||||
actions: read
|
||||
contents: read
|
||||
pull-requests: write
|
||||
|
||||
jobs:
|
||||
pr-report:
|
||||
name: PR Report
|
||||
# Only run for pull-request CI runs
|
||||
if: >-
|
||||
github.event.workflow_run.event == 'pull_request' &&
|
||||
github.event.workflow_run.conclusion != 'cancelled'
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 5
|
||||
steps:
|
||||
# ── Download artifacts from the CI run ────────────────────────
|
||||
- name: Download artifacts
|
||||
- name: Download PR metadata
|
||||
uses: actions/github-script@60a0d83039c74a4aee543508d2ffcb1c3799cdea # v7
|
||||
with:
|
||||
script: |
|
||||
const fs = require('fs');
|
||||
const path = require('path');
|
||||
const runId = context.payload.workflow_run.id;
|
||||
|
||||
const allArtifacts = await github.rest.actions.listWorkflowRunArtifacts({
|
||||
const artifacts = await github.rest.actions.listWorkflowRunArtifacts({
|
||||
owner: context.repo.owner,
|
||||
repo: context.repo.repo,
|
||||
run_id: runId,
|
||||
run_id: ${{ github.event.workflow_run.id }},
|
||||
});
|
||||
|
||||
async function downloadArtifact(name, dest) {
|
||||
const match = allArtifacts.data.artifacts.find(a => a.name === name);
|
||||
if (!match) {
|
||||
core.warning(`Artifact "${name}" not found`);
|
||||
return false;
|
||||
}
|
||||
const zip = await github.rest.actions.downloadArtifact({
|
||||
owner: context.repo.owner,
|
||||
repo: context.repo.repo,
|
||||
artifact_id: match.id,
|
||||
archive_format: 'zip',
|
||||
});
|
||||
fs.mkdirSync(dest, { recursive: true });
|
||||
fs.writeFileSync(path.join(dest, `${name}.zip`), Buffer.from(zip.data));
|
||||
return true;
|
||||
const meta = artifacts.data.artifacts.find(a => a.name === 'pr-meta');
|
||||
if (!meta) {
|
||||
core.setFailed('pr-meta artifact not found — skipping report');
|
||||
return;
|
||||
}
|
||||
|
||||
const temp = process.env.RUNNER_TEMP;
|
||||
await downloadArtifact('pr-meta', path.join(temp, 'dl'));
|
||||
await downloadArtifact('test-reports', path.join(temp, 'dl'));
|
||||
await downloadArtifact('integration-reports', path.join(temp, 'dl'));
|
||||
const zip = await github.rest.actions.downloadArtifact({
|
||||
owner: context.repo.owner,
|
||||
repo: context.repo.repo,
|
||||
artifact_id: meta.id,
|
||||
archive_format: 'zip',
|
||||
});
|
||||
|
||||
- name: Extract artifacts
|
||||
shell: bash
|
||||
run: |
|
||||
cd "$RUNNER_TEMP/dl"
|
||||
# Extract each artifact into its own directory to avoid filename collisions
|
||||
for z in *.zip; do
|
||||
[ -f "$z" ] || continue
|
||||
name="${z%.zip}"
|
||||
mkdir -p "$RUNNER_TEMP/artifacts/$name"
|
||||
unzip -o "$z" -d "$RUNNER_TEMP/artifacts/$name"
|
||||
done
|
||||
const dest = path.join(process.env.RUNNER_TEMP, 'pr-meta');
|
||||
fs.mkdirSync(dest, { recursive: true });
|
||||
fs.writeFileSync(path.join(dest, 'pr-meta.zip'), Buffer.from(zip.data));
|
||||
|
||||
- name: Read PR metadata
|
||||
- name: Extract PR metadata
|
||||
id: meta
|
||||
shell: bash
|
||||
run: |
|
||||
DIR="$RUNNER_TEMP/artifacts/pr-meta"
|
||||
if [ ! -f "$DIR/pr_number" ]; then
|
||||
echo "skip=true" >> "$GITHUB_OUTPUT"
|
||||
echo "::warning::pr_number artifact missing — skipping report"
|
||||
exit 0
|
||||
cd "$RUNNER_TEMP/pr-meta"
|
||||
unzip -o pr-meta.zip
|
||||
|
||||
PR_NUMBER=$(cat pr-number | tr -d '[:space:]')
|
||||
if ! [[ "$PR_NUMBER" =~ ^[0-9]+$ ]]; then
|
||||
echo "::error::Invalid PR number: '$PR_NUMBER'"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# Validate PR number is a positive integer (artifact comes from
|
||||
# untrusted fork code, so treat contents defensively).
|
||||
PR_NUM=$(cat "$DIR/pr_number" | tr -d '[:space:]')
|
||||
if ! [[ "$PR_NUM" =~ ^[0-9]+$ ]]; then
|
||||
echo "skip=true" >> "$GITHUB_OUTPUT"
|
||||
echo "::error::Invalid PR number in artifact: '$PR_NUM'"
|
||||
exit 0
|
||||
fi
|
||||
echo "pr-number=$PR_NUMBER" >> "$GITHUB_OUTPUT"
|
||||
echo "quality=$(cat quality-result | tr -d '[:space:]')" >> "$GITHUB_OUTPUT"
|
||||
echo "tests=$(cat tests-result | tr -d '[:space:]')" >> "$GITHUB_OUTPUT"
|
||||
|
||||
echo "skip=false" >> "$GITHUB_OUTPUT"
|
||||
echo "pr_number=$PR_NUM" >> "$GITHUB_OUTPUT"
|
||||
# Validate job-result strings against known GitHub Actions values.
|
||||
# Artifact contents come from the PR workflow (potentially untrusted
|
||||
# fork code), so we whitelist to prevent newline injection into
|
||||
# GITHUB_OUTPUT.
|
||||
validate_result() {
|
||||
local val
|
||||
val=$(cat "$1" | tr -d '[:space:]')
|
||||
case "$val" in
|
||||
success|failure|cancelled|skipped) echo "$val" ;;
|
||||
*) echo "unknown" ;;
|
||||
esac
|
||||
}
|
||||
|
||||
echo "quality=$(validate_result "$DIR/quality_result")" >> "$GITHUB_OUTPUT"
|
||||
echo "unit=$(validate_result "$DIR/unit_result")" >> "$GITHUB_OUTPUT"
|
||||
echo "integration=$(validate_result "$DIR/integration_result")" >> "$GITHUB_OUTPUT"
|
||||
|
||||
- name: Checkout (for vitest config)
|
||||
if: steps.meta.outputs.skip != 'true'
|
||||
uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4
|
||||
- name: Download test reports
|
||||
id: download-test-reports
|
||||
uses: actions/github-script@60a0d83039c74a4aee543508d2ffcb1c3799cdea # v7
|
||||
with:
|
||||
sparse-checkout: gitnexus/vitest.config.ts
|
||||
sparse-checkout-cone-mode: false
|
||||
script: |
|
||||
const fs = require('fs');
|
||||
const path = require('path');
|
||||
|
||||
# ── Merge coverage from unit + integration ─────────────────────
|
||||
- name: Setup Node.js
|
||||
if: steps.meta.outputs.skip != 'true'
|
||||
uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4
|
||||
with:
|
||||
node-version: 20
|
||||
const artifacts = await github.rest.actions.listWorkflowRunArtifacts({
|
||||
owner: context.repo.owner,
|
||||
repo: context.repo.repo,
|
||||
run_id: ${{ github.event.workflow_run.id }},
|
||||
});
|
||||
|
||||
- name: Install coverage merge tools
|
||||
if: steps.meta.outputs.skip != 'true'
|
||||
run: npm install --no-save istanbul-lib-coverage istanbul-lib-report istanbul-reports
|
||||
|
||||
- name: Merge coverage reports
|
||||
if: steps.meta.outputs.skip != 'true'
|
||||
id: coverage
|
||||
shell: bash
|
||||
run: |
|
||||
DIR="$RUNNER_TEMP/artifacts"
|
||||
UNIT_COV=$(find "$DIR/test-reports" -name "coverage-final.json" -type f 2>/dev/null | head -1)
|
||||
INTEG_COV=$(find "$DIR/integration-reports" -name "coverage-final.json" -type f 2>/dev/null | head -1)
|
||||
MERGED_DIR="$RUNNER_TEMP/merged-coverage"
|
||||
mkdir -p "$MERGED_DIR"
|
||||
|
||||
if [ -n "$UNIT_COV" ] && [ -n "$INTEG_COV" ]; then
|
||||
echo "has_merged=true" >> "$GITHUB_OUTPUT"
|
||||
# Merge using Node.js + istanbul-lib-coverage.
|
||||
# Paths are passed via env vars to avoid shell interpolation
|
||||
# inside the script string.
|
||||
UNIT_COV_PATH="$UNIT_COV" \
|
||||
INTEG_COV_PATH="$INTEG_COV" \
|
||||
MERGED_OUT_DIR="$MERGED_DIR" \
|
||||
node -e "
|
||||
const libCoverage = require('istanbul-lib-coverage');
|
||||
const libReport = require('istanbul-lib-report');
|
||||
const reports = require('istanbul-reports');
|
||||
const fs = require('fs');
|
||||
|
||||
const map = libCoverage.createCoverageMap({});
|
||||
map.merge(JSON.parse(fs.readFileSync(process.env.UNIT_COV_PATH, 'utf8')));
|
||||
map.merge(JSON.parse(fs.readFileSync(process.env.INTEG_COV_PATH, 'utf8')));
|
||||
|
||||
const context = libReport.createContext({
|
||||
coverageMap: map,
|
||||
dir: process.env.MERGED_OUT_DIR,
|
||||
});
|
||||
reports.create('json-summary').execute(context);
|
||||
console.log('Merged coverage written to ' + process.env.MERGED_OUT_DIR + '/coverage-summary.json');
|
||||
"
|
||||
elif [ -n "$UNIT_COV" ]; then
|
||||
echo "has_merged=false" >> "$GITHUB_OUTPUT"
|
||||
echo "::warning::Integration coverage not found — using unit coverage only"
|
||||
else
|
||||
echo "has_merged=false" >> "$GITHUB_OUTPUT"
|
||||
echo "::warning::No coverage data found"
|
||||
fi
|
||||
|
||||
- name: Build report
|
||||
if: steps.meta.outputs.skip != 'true'
|
||||
id: report
|
||||
shell: bash
|
||||
env:
|
||||
QUALITY: ${{ steps.meta.outputs.quality }}
|
||||
UNIT: ${{ steps.meta.outputs.unit }}
|
||||
INTEG: ${{ steps.meta.outputs.integration }}
|
||||
HAS_MERGED: ${{ steps.coverage.outputs.has_merged }}
|
||||
RUN_URL: ${{ github.event.workflow_run.html_url }}
|
||||
run: |
|
||||
DIR="$RUNNER_TEMP/artifacts"
|
||||
MERGED_DIR="$RUNNER_TEMP/merged-coverage"
|
||||
|
||||
# ── Helper: read coverage summary into prefixed vars ──
|
||||
# Uses printf -v for safe variable assignment (no eval).
|
||||
read_cov() {
|
||||
local prefix=$1 file=$2
|
||||
if [ -n "$file" ] && [ -f "$file" ]; then
|
||||
local val
|
||||
val=$(jq -r '.total.statements.pct // "N/A"' "$file" 2>/dev/null) || val="N/A"
|
||||
printf -v "${prefix}_STMTS" '%s' "$val"
|
||||
val=$(jq -r '.total.branches.pct // "N/A"' "$file" 2>/dev/null) || val="N/A"
|
||||
printf -v "${prefix}_BRANCH" '%s' "$val"
|
||||
val=$(jq -r '.total.functions.pct // "N/A"' "$file" 2>/dev/null) || val="N/A"
|
||||
printf -v "${prefix}_FUNCS" '%s' "$val"
|
||||
val=$(jq -r '.total.lines.pct // "N/A"' "$file" 2>/dev/null) || val="N/A"
|
||||
printf -v "${prefix}_LINES" '%s' "$val"
|
||||
val=$(jq -r '"\(.total.statements.covered)/\(.total.statements.total)"' "$file" 2>/dev/null) || val=""
|
||||
printf -v "${prefix}_STMTS_COV" '%s' "$val"
|
||||
val=$(jq -r '"\(.total.branches.covered)/\(.total.branches.total)"' "$file" 2>/dev/null) || val=""
|
||||
printf -v "${prefix}_BRANCH_COV" '%s' "$val"
|
||||
val=$(jq -r '"\(.total.functions.covered)/\(.total.functions.total)"' "$file" 2>/dev/null) || val=""
|
||||
printf -v "${prefix}_FUNCS_COV" '%s' "$val"
|
||||
val=$(jq -r '"\(.total.lines.covered)/\(.total.lines.total)"' "$file" 2>/dev/null) || val=""
|
||||
printf -v "${prefix}_LINES_COV" '%s' "$val"
|
||||
return 0
|
||||
else
|
||||
printf -v "${prefix}_STMTS" '%s' "N/A"
|
||||
printf -v "${prefix}_BRANCH" '%s' "N/A"
|
||||
printf -v "${prefix}_FUNCS" '%s' "N/A"
|
||||
printf -v "${prefix}_LINES" '%s' "N/A"
|
||||
printf -v "${prefix}_STMTS_COV" '%s' ""
|
||||
printf -v "${prefix}_BRANCH_COV" '%s' ""
|
||||
printf -v "${prefix}_FUNCS_COV" '%s' ""
|
||||
printf -v "${prefix}_LINES_COV" '%s' ""
|
||||
return 1
|
||||
fi
|
||||
}
|
||||
|
||||
# ── Read all three coverage reports ──
|
||||
UNIT_SUMMARY=$(find "$DIR/test-reports" -name "coverage-summary.json" -type f 2>/dev/null | head -1)
|
||||
INTEG_SUMMARY=$(find "$DIR/integration-reports" -name "coverage-summary.json" -type f 2>/dev/null | head -1)
|
||||
MERGED_SUMMARY="$MERGED_DIR/coverage-summary.json"
|
||||
|
||||
read_cov "U" "$UNIT_SUMMARY"
|
||||
HAS_UNIT=$?
|
||||
read_cov "I" "$INTEG_SUMMARY"
|
||||
HAS_INTEG=$?
|
||||
read_cov "M" "$MERGED_SUMMARY"
|
||||
|
||||
# ── Locate test results (unit) ──
|
||||
RESULTS_FILE=$(find "$DIR/test-reports" -name "test-results.json" -type f 2>/dev/null | head -1)
|
||||
INTEG_RESULTS=$(find "$DIR/integration-reports" -name "integration-results.json" -type f 2>/dev/null | head -1)
|
||||
|
||||
if [ -n "$RESULTS_FILE" ]; then
|
||||
U_TOTAL=$(jq -r '.numTotalTests' "$RESULTS_FILE" 2>/dev/null || echo 0)
|
||||
U_PASSED=$(jq -r '.numPassedTests' "$RESULTS_FILE" 2>/dev/null || echo 0)
|
||||
U_FAILED=$(jq -r '.numFailedTests' "$RESULTS_FILE" 2>/dev/null || echo 0)
|
||||
U_SKIPPED=$(jq -r '.numPendingTests' "$RESULTS_FILE" 2>/dev/null || echo 0)
|
||||
U_SUITES=$(jq -r '.numTotalTestSuites' "$RESULTS_FILE" 2>/dev/null || echo 0)
|
||||
U_DURATION=$(jq -r '((.testResults | map(.endTime) | max) - (.startTime)) / 1000 | floor' "$RESULTS_FILE" 2>/dev/null || echo 0)
|
||||
else
|
||||
U_TOTAL=0; U_PASSED=0; U_FAILED=0; U_SKIPPED=0; U_SUITES=0; U_DURATION=0
|
||||
fi
|
||||
|
||||
if [ -n "$INTEG_RESULTS" ]; then
|
||||
I_TOTAL=$(jq -r '.numTotalTests' "$INTEG_RESULTS" 2>/dev/null || echo 0)
|
||||
I_PASSED=$(jq -r '.numPassedTests' "$INTEG_RESULTS" 2>/dev/null || echo 0)
|
||||
I_FAILED=$(jq -r '.numFailedTests' "$INTEG_RESULTS" 2>/dev/null || echo 0)
|
||||
I_SKIPPED=$(jq -r '.numPendingTests' "$INTEG_RESULTS" 2>/dev/null || echo 0)
|
||||
I_SUITES=$(jq -r '.numTotalTestSuites' "$INTEG_RESULTS" 2>/dev/null || echo 0)
|
||||
I_DURATION=$(jq -r '((.testResults | map(.endTime) | max) - (.startTime)) / 1000 | floor' "$INTEG_RESULTS" 2>/dev/null || echo 0)
|
||||
else
|
||||
I_TOTAL=0; I_PASSED=0; I_FAILED=0; I_SKIPPED=0; I_SUITES=0; I_DURATION=0
|
||||
fi
|
||||
|
||||
# ── Sum test results ──
|
||||
TOTAL=$((U_TOTAL + I_TOTAL))
|
||||
PASSED=$((U_PASSED + I_PASSED))
|
||||
FAILED=$((U_FAILED + I_FAILED))
|
||||
SKIPPED=$((U_SKIPPED + I_SKIPPED))
|
||||
SUITES=$((U_SUITES + I_SUITES))
|
||||
DURATION=$((U_DURATION + I_DURATION))
|
||||
|
||||
# ── Coverage thresholds (read from vitest.config.ts) ──
|
||||
if [ -f gitnexus/vitest.config.ts ]; then
|
||||
THRESH_STMTS=$(grep -oP 'statements:\s*\K[0-9]+' gitnexus/vitest.config.ts || echo 0)
|
||||
THRESH_BRANCH=$(grep -oP 'branches:\s*\K[0-9]+' gitnexus/vitest.config.ts || echo 0)
|
||||
THRESH_FUNCS=$(grep -oP 'functions:\s*\K[0-9]+' gitnexus/vitest.config.ts || echo 0)
|
||||
THRESH_LINES=$(grep -oP 'lines:\s*\K[0-9]+' gitnexus/vitest.config.ts || echo 0)
|
||||
else
|
||||
THRESH_STMTS=0; THRESH_BRANCH=0; THRESH_FUNCS=0; THRESH_LINES=0
|
||||
fi
|
||||
|
||||
# ── Status helpers ──
|
||||
status_icon() {
|
||||
case "$1" in
|
||||
success) echo "✅" ;;
|
||||
failure) echo "❌" ;;
|
||||
cancelled) echo "⏭️" ;;
|
||||
*) echo "❓" ;;
|
||||
esac
|
||||
}
|
||||
|
||||
cov_bar() {
|
||||
local pct=$1 thresh=$2
|
||||
if [ "$pct" = "N/A" ]; then echo "—"; return; fi
|
||||
local filled
|
||||
filled=$(awk "BEGIN { printf \"%d\", $pct / 5 }")
|
||||
(( filled < 0 )) && filled=0
|
||||
(( filled > 20 )) && filled=20
|
||||
local empty=$((20 - filled))
|
||||
local bar=""
|
||||
for ((i=0; i<filled; i++)); do bar+="█"; done
|
||||
for ((i=0; i<empty; i++)); do bar+="░"; done
|
||||
if [ "$(awk "BEGIN { print ($pct >= $thresh) ? 1 : 0 }")" = "1" ]; then
|
||||
echo "🟢 ${bar}"
|
||||
else
|
||||
echo "🔴 ${bar}"
|
||||
fi
|
||||
}
|
||||
|
||||
# ── Overall status ──
|
||||
if [[ "$QUALITY" == "success" && "$UNIT" == "success" && "$INTEG" == "success" ]]; then
|
||||
OVERALL="✅ **All checks passed**"
|
||||
else
|
||||
OVERALL="❌ **Some checks failed**"
|
||||
fi
|
||||
|
||||
# ── Build markdown ──
|
||||
{
|
||||
echo "body<<GITNEXUS_CI_REPORT_EOF_7f3a"
|
||||
echo "## CI Report"
|
||||
echo ""
|
||||
echo "${OVERALL}"
|
||||
echo ""
|
||||
echo "### Pipeline Status"
|
||||
echo ""
|
||||
echo "| Stage | Status | Details |"
|
||||
echo "|-------|--------|---------|"
|
||||
echo "| $(status_icon "$QUALITY") Typecheck | \`${QUALITY}\` | tsc --noEmit |"
|
||||
echo "| $(status_icon "$UNIT") Unit Tests | \`${UNIT}\` | 3 platforms |"
|
||||
echo "| $(status_icon "$INTEG") Integration | \`${INTEG}\` | 3 OS x 4 groups = 12 jobs |"
|
||||
echo ""
|
||||
|
||||
if [ "$TOTAL" -gt 0 ] 2>/dev/null; then
|
||||
echo "### Test Results"
|
||||
echo ""
|
||||
if [ "$FAILED" = "0" ]; then
|
||||
echo "✅ **${PASSED}** passed"
|
||||
else
|
||||
echo "❌ **${FAILED}** failed / **${PASSED}** passed"
|
||||
fi
|
||||
if [ "$SKIPPED" != "0" ]; then
|
||||
echo " · ${SKIPPED} skipped"
|
||||
fi
|
||||
echo " · ${SUITES} suites · ${TOTAL} total"
|
||||
echo " · ⏱️ ${DURATION}s"
|
||||
if [ "$I_TOTAL" -gt 0 ] 2>/dev/null; then
|
||||
echo " · 📊 ${U_TOTAL} unit + ${I_TOTAL} integration"
|
||||
fi
|
||||
echo ""
|
||||
fi
|
||||
|
||||
# ── Coverage table helper ──
|
||||
cov_table() {
|
||||
local label=$1 s=$2 b=$3 f=$4 l=$5 sc=$6 bc=$7 fc=$8 lc=$9
|
||||
shift 9
|
||||
local ts=$1 tb=$2 tf=$3 tl=$4
|
||||
echo "#### ${label}"
|
||||
echo ""
|
||||
echo "| Metric | Coverage | Covered | Threshold | Status |"
|
||||
echo "|--------|----------|---------|-----------|--------|"
|
||||
echo "| Statements | **${s}%** | ${sc} | ${ts}% | $(cov_bar "$s" "$ts") |"
|
||||
echo "| Branches | **${b}%** | ${bc} | ${tb}% | $(cov_bar "$b" "$tb") |"
|
||||
echo "| Functions | **${f}%** | ${fc} | ${tf}% | $(cov_bar "$f" "$tf") |"
|
||||
echo "| Lines | **${l}%** | ${lc} | ${tl}% | $(cov_bar "$l" "$tl") |"
|
||||
echo ""
|
||||
const reports = artifacts.data.artifacts.find(a => a.name === 'test-reports');
|
||||
if (!reports) {
|
||||
core.warning('test-reports artifact not found');
|
||||
return;
|
||||
}
|
||||
|
||||
if [ "$M_STMTS" != "N/A" ]; then
|
||||
echo "### Code Coverage"
|
||||
echo ""
|
||||
cov_table "Combined (Unit + Integration)" \
|
||||
"$M_STMTS" "$M_BRANCH" "$M_FUNCS" "$M_LINES" \
|
||||
"$M_STMTS_COV" "$M_BRANCH_COV" "$M_FUNCS_COV" "$M_LINES_COV" \
|
||||
"$THRESH_STMTS" "$THRESH_BRANCH" "$THRESH_FUNCS" "$THRESH_LINES"
|
||||
const zip = await github.rest.actions.downloadArtifact({
|
||||
owner: context.repo.owner,
|
||||
repo: context.repo.repo,
|
||||
artifact_id: reports.id,
|
||||
archive_format: 'zip',
|
||||
});
|
||||
|
||||
echo "<details>"
|
||||
echo "<summary>Coverage breakdown by test suite</summary>"
|
||||
echo ""
|
||||
if [ "$U_STMTS" != "N/A" ]; then
|
||||
cov_table "Unit Tests" \
|
||||
"$U_STMTS" "$U_BRANCH" "$U_FUNCS" "$U_LINES" \
|
||||
"$U_STMTS_COV" "$U_BRANCH_COV" "$U_FUNCS_COV" "$U_LINES_COV" \
|
||||
"$THRESH_STMTS" "$THRESH_BRANCH" "$THRESH_FUNCS" "$THRESH_LINES"
|
||||
fi
|
||||
if [ "$I_STMTS" != "N/A" ]; then
|
||||
cov_table "Integration Tests" \
|
||||
"$I_STMTS" "$I_BRANCH" "$I_FUNCS" "$I_LINES" \
|
||||
"$I_STMTS_COV" "$I_BRANCH_COV" "$I_FUNCS_COV" "$I_LINES_COV" \
|
||||
"$THRESH_STMTS" "$THRESH_BRANCH" "$THRESH_FUNCS" "$THRESH_LINES"
|
||||
fi
|
||||
echo "</details>"
|
||||
echo ""
|
||||
echo "<details>"
|
||||
echo "<summary>Coverage thresholds are auto-ratcheted — they only go up</summary>"
|
||||
echo ""
|
||||
echo "Vitest \`thresholds.autoUpdate\` bumps the floor whenever local coverage exceeds it."
|
||||
echo "CI enforces the current thresholds; developers commit the ratcheted values."
|
||||
echo "</details>"
|
||||
echo ""
|
||||
elif [ "$U_STMTS" != "N/A" ]; then
|
||||
echo "### Code Coverage (Unit only)"
|
||||
echo ""
|
||||
cov_table "Unit Tests" \
|
||||
"$U_STMTS" "$U_BRANCH" "$U_FUNCS" "$U_LINES" \
|
||||
"$U_STMTS_COV" "$U_BRANCH_COV" "$U_FUNCS_COV" "$U_LINES_COV" \
|
||||
"$THRESH_STMTS" "$THRESH_BRANCH" "$THRESH_FUNCS" "$THRESH_LINES"
|
||||
echo "<details>"
|
||||
echo "<summary>Coverage thresholds are auto-ratcheted — they only go up</summary>"
|
||||
echo ""
|
||||
echo "Vitest \`thresholds.autoUpdate\` bumps the floor whenever local coverage exceeds it."
|
||||
echo "CI enforces the current thresholds; developers commit the ratcheted values."
|
||||
echo "</details>"
|
||||
echo ""
|
||||
else
|
||||
echo "### Code Coverage"
|
||||
echo ""
|
||||
echo "⚠️ Coverage data unavailable - check the [unit test job](${RUN_URL}) for details."
|
||||
echo ""
|
||||
fi
|
||||
const dest = path.join(process.env.RUNNER_TEMP, 'test-reports');
|
||||
fs.mkdirSync(dest, { recursive: true });
|
||||
fs.writeFileSync(path.join(dest, 'test-reports.zip'), Buffer.from(zip.data));
|
||||
|
||||
echo "---"
|
||||
echo "<sub>📋 [View full run](${RUN_URL}) · Generated by CI</sub>"
|
||||
echo "GITNEXUS_CI_REPORT_EOF_7f3a"
|
||||
} >> "$GITHUB_OUTPUT"
|
||||
- name: Extract test reports
|
||||
if: steps.download-test-reports.outcome == 'success'
|
||||
shell: bash
|
||||
run: |
|
||||
cd "$RUNNER_TEMP/test-reports"
|
||||
unzip -o test-reports.zip || true
|
||||
|
||||
- name: Comment on PR
|
||||
if: steps.meta.outputs.skip != 'true'
|
||||
uses: marocchino/sticky-pull-request-comment@773744901bac0e8cbb5a0dc842800d45e9b2b405 # v2
|
||||
- name: Fetch cross-platform job results
|
||||
id: jobs
|
||||
uses: actions/github-script@60a0d83039c74a4aee543508d2ffcb1c3799cdea # v7
|
||||
with:
|
||||
header: ci-report
|
||||
number: ${{ steps.meta.outputs.pr_number }}
|
||||
message: ${{ steps.report.outputs.body }}
|
||||
script: |
|
||||
const jobs = await github.rest.actions.listJobsForWorkflowRun({
|
||||
owner: context.repo.owner,
|
||||
repo: context.repo.repo,
|
||||
run_id: ${{ github.event.workflow_run.id }},
|
||||
per_page: 50,
|
||||
});
|
||||
|
||||
const results = {};
|
||||
for (const job of jobs.data.jobs) {
|
||||
if (job.name.includes('ubuntu')) results.ubuntu = job.conclusion || 'pending';
|
||||
else if (job.name.includes('windows')) results.windows = job.conclusion || 'pending';
|
||||
else if (job.name.includes('macos')) results.macos = job.conclusion || 'pending';
|
||||
}
|
||||
core.setOutput('ubuntu', results.ubuntu || 'unknown');
|
||||
core.setOutput('windows', results.windows || 'unknown');
|
||||
core.setOutput('macos', results.macos || 'unknown');
|
||||
|
||||
- name: Fetch base branch coverage
|
||||
id: base-coverage
|
||||
uses: actions/github-script@60a0d83039c74a4aee543508d2ffcb1c3799cdea # v7
|
||||
with:
|
||||
script: |
|
||||
const fs = require('fs');
|
||||
const path = require('path');
|
||||
|
||||
const runs = await github.rest.actions.listWorkflowRuns({
|
||||
owner: context.repo.owner,
|
||||
repo: context.repo.repo,
|
||||
workflow_id: 'ci.yml',
|
||||
branch: 'main',
|
||||
status: 'success',
|
||||
per_page: 1,
|
||||
});
|
||||
|
||||
if (runs.data.workflow_runs.length === 0) {
|
||||
core.setOutput('found', 'false');
|
||||
return;
|
||||
}
|
||||
|
||||
const mainRunId = runs.data.workflow_runs[0].id;
|
||||
const artifacts = await github.rest.actions.listWorkflowRunArtifacts({
|
||||
owner: context.repo.owner,
|
||||
repo: context.repo.repo,
|
||||
run_id: mainRunId,
|
||||
});
|
||||
|
||||
const testReports = artifacts.data.artifacts.find(a => a.name === 'test-reports');
|
||||
if (!testReports) {
|
||||
core.setOutput('found', 'false');
|
||||
return;
|
||||
}
|
||||
|
||||
const zip = await github.rest.actions.downloadArtifact({
|
||||
owner: context.repo.owner,
|
||||
repo: context.repo.repo,
|
||||
artifact_id: testReports.id,
|
||||
archive_format: 'zip',
|
||||
});
|
||||
|
||||
const dest = path.join(process.env.RUNNER_TEMP, 'base-coverage');
|
||||
fs.mkdirSync(dest, { recursive: true });
|
||||
fs.writeFileSync(path.join(dest, 'base.zip'), Buffer.from(zip.data));
|
||||
core.setOutput('found', 'true');
|
||||
core.setOutput('dir', dest);
|
||||
|
||||
- name: Extract base coverage
|
||||
if: steps.base-coverage.outputs.found == 'true'
|
||||
shell: bash
|
||||
run: |
|
||||
cd "${{ steps.base-coverage.outputs.dir }}"
|
||||
unzip -o base.zip -d base
|
||||
|
||||
- name: Build and post report
|
||||
uses: actions/github-script@60a0d83039c74a4aee543508d2ffcb1c3799cdea # v7
|
||||
env:
|
||||
PR_NUMBER: ${{ steps.meta.outputs.pr-number }}
|
||||
QUALITY: ${{ steps.meta.outputs.quality }}
|
||||
TESTS: ${{ steps.meta.outputs.tests }}
|
||||
UBUNTU: ${{ steps.jobs.outputs.ubuntu }}
|
||||
WINDOWS: ${{ steps.jobs.outputs.windows }}
|
||||
MACOS: ${{ steps.jobs.outputs.macos }}
|
||||
BASE_FOUND: ${{ steps.base-coverage.outputs.found }}
|
||||
BASE_DIR: ${{ steps.base-coverage.outputs.dir }}
|
||||
RUN_ID: ${{ github.event.workflow_run.id }}
|
||||
HEAD_SHA: ${{ github.event.workflow_run.head_sha }}
|
||||
with:
|
||||
script: |
|
||||
const fs = require('fs');
|
||||
const path = require('path');
|
||||
|
||||
const icon = (s) => ({ success: '✅', failure: '❌', cancelled: '⏭️' }[s] || '❓');
|
||||
const temp = process.env.RUNNER_TEMP;
|
||||
|
||||
// ── Read coverage ──
|
||||
function readCov(dir) {
|
||||
const out = { stmts: 'N/A', branch: 'N/A', funcs: 'N/A', lines: 'N/A',
|
||||
stmtsCov: '', branchCov: '', funcsCov: '', linesCov: '' };
|
||||
try {
|
||||
const files = require('child_process')
|
||||
.execSync(`find "${dir}" -name coverage-summary.json -type f`, { encoding: 'utf8' })
|
||||
.trim().split('\n').filter(Boolean);
|
||||
if (!files.length) return out;
|
||||
const d = JSON.parse(fs.readFileSync(files[0], 'utf8')).total;
|
||||
out.stmts = d.statements.pct; out.branch = d.branches.pct;
|
||||
out.funcs = d.functions.pct; out.lines = d.lines.pct;
|
||||
out.stmtsCov = `${d.statements.covered}/${d.statements.total}`;
|
||||
out.branchCov = `${d.branches.covered}/${d.branches.total}`;
|
||||
out.funcsCov = `${d.functions.covered}/${d.functions.total}`;
|
||||
out.linesCov = `${d.lines.covered}/${d.lines.total}`;
|
||||
} catch {}
|
||||
return out;
|
||||
}
|
||||
|
||||
const cov = readCov(path.join(temp, 'test-reports'));
|
||||
const base = process.env.BASE_FOUND === 'true'
|
||||
? readCov(path.join(process.env.BASE_DIR, 'base'))
|
||||
: { stmts: 'N/A', branch: 'N/A', funcs: 'N/A', lines: 'N/A' };
|
||||
|
||||
// ── Read test results ──
|
||||
let total = 0, passed = 0, failed = 0, skipped = 0, suites = 0, duration = '0s';
|
||||
let skippedTests = [];
|
||||
try {
|
||||
const files = require('child_process')
|
||||
.execSync(`find "${path.join(temp, 'test-reports')}" -name test-results.json -type f`, { encoding: 'utf8' })
|
||||
.trim().split('\n').filter(Boolean);
|
||||
if (files.length) {
|
||||
const r = JSON.parse(fs.readFileSync(files[0], 'utf8'));
|
||||
total = r.numTotalTests || 0;
|
||||
passed = r.numPassedTests || 0;
|
||||
failed = r.numFailedTests || 0;
|
||||
skipped = r.numPendingTests || 0;
|
||||
suites = r.numTotalTestSuites || 0;
|
||||
const durS = Math.floor((Math.max(...r.testResults.map(t => t.endTime)) - r.startTime) / 1000);
|
||||
duration = durS >= 60 ? `${Math.floor(durS / 60)}m ${durS % 60}s` : `${durS}s`;
|
||||
// Collect skipped test names
|
||||
for (const suite of r.testResults) {
|
||||
for (const t of (suite.assertionResults || [])) {
|
||||
if (t.status === 'pending' || t.status === 'skipped') {
|
||||
skippedTests.push(`- ${t.ancestorTitles.join(' > ')} > ${t.title}`);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
} catch {}
|
||||
|
||||
// ── Coverage delta ──
|
||||
function delta(pct, basePct) {
|
||||
if (pct === 'N/A' || basePct === 'N/A') return '—';
|
||||
const d = (pct - basePct).toFixed(1);
|
||||
const dNum = parseFloat(d);
|
||||
if (dNum > 0) return `📈 +${d}%`;
|
||||
if (dNum < 0) return `📉 ${d}%`;
|
||||
return '=';
|
||||
}
|
||||
|
||||
// ── Build markdown ──
|
||||
const { PR_NUMBER, QUALITY, TESTS, UBUNTU, WINDOWS, MACOS, RUN_ID, HEAD_SHA } = process.env;
|
||||
const prNumber = parseInt(PR_NUMBER, 10);
|
||||
const overall = (QUALITY === 'success' && TESTS === 'success')
|
||||
? '✅ **All checks passed**' : '❌ **Some checks failed**';
|
||||
const sha = HEAD_SHA.slice(0, 7);
|
||||
|
||||
let body = `## CI Report\n\n${overall}   \`${sha}\`\n\n`;
|
||||
|
||||
body += `### Pipeline\n\n`;
|
||||
body += `| Stage | Status | Ubuntu | Windows | macOS |\n`;
|
||||
body += `|-------|--------|--------|---------|-------|\n`;
|
||||
body += `| Typecheck | ${icon(QUALITY)} \`${QUALITY}\` | — | — | — |\n`;
|
||||
body += `| Tests | ${icon(TESTS)} \`${TESTS}\` | ${icon(UBUNTU)} | ${icon(WINDOWS)} | ${icon(MACOS)} |\n\n`;
|
||||
|
||||
if (total > 0) {
|
||||
body += `### Tests\n\n`;
|
||||
body += `| Metric | Value |\n|--------|-------|\n`;
|
||||
body += `| Total | **${total}** |\n`;
|
||||
body += `| Passed | **${passed}** |\n`;
|
||||
if (failed > 0) body += `| Failed | **${failed}** |\n`;
|
||||
if (skipped > 0) body += `| Skipped | ${skipped} |\n`;
|
||||
body += `| Files | ${suites} |\n`;
|
||||
body += `| Duration | ${duration} |\n\n`;
|
||||
|
||||
if (failed === 0) {
|
||||
body += `✅ All **${passed}** tests passed across **${suites}** files\n`;
|
||||
} else {
|
||||
body += `❌ **${failed}** failed / **${passed}** passed\n`;
|
||||
}
|
||||
|
||||
if (skippedTests.length > 0) {
|
||||
body += `\n<details>\n<summary>${skipped} test(s) skipped</summary>\n\n`;
|
||||
body += skippedTests.join('\n') + '\n\n</details>\n';
|
||||
}
|
||||
body += '\n';
|
||||
}
|
||||
|
||||
if (cov.stmts !== 'N/A') {
|
||||
body += `### Coverage\n\n`;
|
||||
body += `| Metric | Coverage | Covered | Base (main) | Delta |\n`;
|
||||
body += `|--------|----------|---------|-------------|-------|\n`;
|
||||
body += `| Statements | **${cov.stmts}%** | ${cov.stmtsCov} | ${base.stmts}% | ${delta(cov.stmts, base.stmts)} |\n`;
|
||||
body += `| Branches | **${cov.branch}%** | ${cov.branchCov} | ${base.branch}% | ${delta(cov.branch, base.branch)} |\n`;
|
||||
body += `| Functions | **${cov.funcs}%** | ${cov.funcsCov} | ${base.funcs}% | ${delta(cov.funcs, base.funcs)} |\n`;
|
||||
body += `| Lines | **${cov.lines}%** | ${cov.linesCov} | ${base.lines}% | ${delta(cov.lines, base.lines)} |\n\n`;
|
||||
} else {
|
||||
const runUrl = `${context.serverUrl}/${context.repo.owner}/${context.repo.repo}/actions/runs/${RUN_ID}`;
|
||||
body += `### Coverage\n\n⚠️ Coverage data unavailable — check the [test job](${runUrl}) for details.\n\n`;
|
||||
}
|
||||
|
||||
const runUrl = `${context.serverUrl}/${context.repo.owner}/${context.repo.repo}/actions/runs/${RUN_ID}`;
|
||||
body += `---\n<sub>📋 [Full run](${runUrl}) · Coverage from Ubuntu · Generated by CI</sub>`;
|
||||
|
||||
// ── Post sticky comment ──
|
||||
const { data: comments } = await github.rest.issues.listComments({
|
||||
owner: context.repo.owner,
|
||||
repo: context.repo.repo,
|
||||
issue_number: prNumber,
|
||||
per_page: 100,
|
||||
direction: 'desc',
|
||||
});
|
||||
|
||||
const marker = '<!-- ci-report -->';
|
||||
const existing = comments.find(c => c.body?.includes(marker));
|
||||
const fullBody = marker + '\n' + body;
|
||||
|
||||
if (existing) {
|
||||
await github.rest.issues.updateComment({
|
||||
owner: context.repo.owner,
|
||||
repo: context.repo.repo,
|
||||
comment_id: existing.id,
|
||||
body: fullBody,
|
||||
});
|
||||
} else {
|
||||
await github.rest.issues.createComment({
|
||||
owner: context.repo.owner,
|
||||
repo: context.repo.repo,
|
||||
issue_number: prNumber,
|
||||
body: fullBody,
|
||||
});
|
||||
}
|
||||
|
||||
@@ -1,20 +1,22 @@
|
||||
name: Unit Tests
|
||||
name: Tests
|
||||
|
||||
on:
|
||||
workflow_call:
|
||||
|
||||
jobs:
|
||||
unit-tests:
|
||||
name: unit (ubuntu / coverage)
|
||||
tests:
|
||||
name: ubuntu / coverage
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 15
|
||||
timeout-minutes: 25
|
||||
steps:
|
||||
- uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4
|
||||
- uses: ./.github/actions/setup-gitnexus
|
||||
with:
|
||||
build: 'true'
|
||||
|
||||
- name: Run unit tests with coverage
|
||||
- name: Run all tests with coverage
|
||||
run: >-
|
||||
npx vitest run test/unit
|
||||
npx vitest run
|
||||
--reporter=default
|
||||
--reporter=json
|
||||
--outputFile=test-results.json
|
||||
@@ -38,16 +40,18 @@ jobs:
|
||||
retention-days: 5
|
||||
|
||||
cross-platform:
|
||||
name: unit (${{ matrix.os }})
|
||||
name: ${{ matrix.os }}
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
# Ubuntu already covered by the coverage job above
|
||||
os: [windows-latest, macos-latest]
|
||||
runs-on: ${{ matrix.os }}
|
||||
timeout-minutes: 15
|
||||
timeout-minutes: 25
|
||||
steps:
|
||||
- uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4
|
||||
- uses: ./.github/actions/setup-gitnexus
|
||||
- run: npx vitest run test/unit
|
||||
with:
|
||||
build: 'true'
|
||||
- run: npx vitest run
|
||||
working-directory: gitnexus
|
||||
+38
-54
@@ -16,10 +16,8 @@ concurrency:
|
||||
# ── Reusable workflow orchestration ─────────────────────────────────
|
||||
# Each concern lives in its own workflow file for maintainability:
|
||||
# ci-quality.yml — typecheck (tsc --noEmit)
|
||||
# ci-unit-tests.yml — unit tests with coverage + cross-platform
|
||||
# ci-integration.yml — integration test matrix (3 OS x 4 groups)
|
||||
#
|
||||
# Shared setup is DRY via .github/actions/setup-gitnexus composite action.
|
||||
# ci-tests.yml — all tests with coverage (ubuntu) + cross-platform
|
||||
# ci-report.yml — PR comment (workflow_run trigger for fork write access)
|
||||
|
||||
jobs:
|
||||
quality:
|
||||
@@ -27,56 +25,16 @@ jobs:
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
unit-tests:
|
||||
uses: ./.github/workflows/ci-unit-tests.yml
|
||||
tests:
|
||||
uses: ./.github/workflows/ci-tests.yml
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
integration:
|
||||
uses: ./.github/workflows/ci-integration.yml
|
||||
with:
|
||||
collect-coverage: ${{ github.event_name == 'pull_request' }}
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
# ── Save PR metadata for the reporting workflow ─────────────────
|
||||
# The ci-report.yml workflow (triggered by workflow_run) needs the
|
||||
# PR number and job results to post a comment. We save them as an
|
||||
# artifact because workflow_run context doesn't reliably carry PR
|
||||
# info for fork PRs.
|
||||
save-pr-meta:
|
||||
name: Save PR Metadata
|
||||
if: always() && github.event_name == 'pull_request'
|
||||
needs: [quality, unit-tests, integration]
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 5
|
||||
steps:
|
||||
- name: Write metadata
|
||||
shell: bash
|
||||
env:
|
||||
PR_NUMBER: ${{ github.event.number }}
|
||||
QUALITY: ${{ needs.quality.result }}
|
||||
UNIT: ${{ needs.unit-tests.result }}
|
||||
INTEG: ${{ needs.integration.result }}
|
||||
run: |
|
||||
mkdir -p pr-meta
|
||||
echo "$PR_NUMBER" > pr-meta/pr_number
|
||||
echo "$QUALITY" > pr-meta/quality_result
|
||||
echo "$UNIT" > pr-meta/unit_result
|
||||
echo "$INTEG" > pr-meta/integration_result
|
||||
|
||||
- name: Upload PR metadata
|
||||
uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4
|
||||
with:
|
||||
name: pr-meta
|
||||
path: pr-meta/
|
||||
retention-days: 1
|
||||
|
||||
# ── Unified CI gate ──────────────────────────────────────────────
|
||||
# Single required check for branch protection.
|
||||
ci-status:
|
||||
name: CI Gate
|
||||
needs: [quality, unit-tests, integration]
|
||||
needs: [quality, tests]
|
||||
if: always()
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 5
|
||||
@@ -85,15 +43,41 @@ jobs:
|
||||
shell: bash
|
||||
env:
|
||||
QUALITY: ${{ needs.quality.result }}
|
||||
UNIT: ${{ needs.unit-tests.result }}
|
||||
INTEG: ${{ needs.integration.result }}
|
||||
TESTS: ${{ needs.tests.result }}
|
||||
run: |
|
||||
echo "Quality: $QUALITY"
|
||||
echo "Unit Tests: $UNIT"
|
||||
echo "Integration: $INTEG"
|
||||
echo "Quality: $QUALITY"
|
||||
echo "Tests: $TESTS"
|
||||
if [[ "$QUALITY" != "success" ]] ||
|
||||
[[ "$UNIT" != "success" ]] ||
|
||||
[[ "$INTEG" != "success" ]]; then
|
||||
[[ "$TESTS" != "success" ]]; then
|
||||
echo "::error::One or more CI jobs failed"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
# ── PR metadata for ci-report.yml ────────────────────────────────
|
||||
# Saves PR number and job results so the workflow_run-triggered
|
||||
# report can post comments with a write token (works for forks).
|
||||
save-pr-meta:
|
||||
name: Save PR Metadata
|
||||
if: always() && github.event_name == 'pull_request'
|
||||
needs: [quality, tests]
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 5
|
||||
steps:
|
||||
- name: Write PR metadata
|
||||
shell: bash
|
||||
env:
|
||||
PR_NUMBER: ${{ github.event.pull_request.number }}
|
||||
QUALITY: ${{ needs.quality.result }}
|
||||
TESTS: ${{ needs.tests.result }}
|
||||
run: |
|
||||
mkdir -p pr-meta
|
||||
echo "$PR_NUMBER" > pr-meta/pr-number
|
||||
echo "$QUALITY" > pr-meta/quality-result
|
||||
echo "$TESTS" > pr-meta/tests-result
|
||||
|
||||
- name: Upload PR metadata
|
||||
uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4
|
||||
with:
|
||||
name: pr-meta
|
||||
path: pr-meta/
|
||||
retention-days: 1
|
||||
|
||||
@@ -2,8 +2,9 @@ name: Claude Code Review
|
||||
|
||||
# Uses pull_request_target so the workflow runs as defined on the default branch,
|
||||
# which allows access to secrets for posting review comments on fork PRs.
|
||||
# SECURITY: The checkout below uses the PR head SHA to review the correct code.
|
||||
# The claude-code-action sandboxes execution — it does NOT run arbitrary code
|
||||
# SECURITY: The checkout pins the fork's HEAD SHA (not the branch name) to
|
||||
# prevent TOCTOU races (force-push between trigger and checkout). The
|
||||
# claude-code-action sandboxes execution — it does NOT run arbitrary code
|
||||
# from the checked-out source.
|
||||
|
||||
on:
|
||||
@@ -15,6 +16,11 @@ on:
|
||||
issue_comment:
|
||||
types: [created]
|
||||
|
||||
# Serialize per-PR to avoid racing review comments.
|
||||
concurrency:
|
||||
group: claude-review-${{ github.event.issue.number || github.event.pull_request.number }}
|
||||
cancel-in-progress: false
|
||||
|
||||
jobs:
|
||||
claude-review:
|
||||
# Run only when:
|
||||
@@ -41,13 +47,13 @@ jobs:
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 30
|
||||
permissions:
|
||||
contents: write # needed to push fork branch to origin
|
||||
contents: read
|
||||
pull-requests: write
|
||||
issues: read
|
||||
id-token: write
|
||||
|
||||
steps:
|
||||
# For issue_comment triggers, resolve the PR number, head SHA, and branch name
|
||||
# For issue_comment triggers, resolve the PR number, head SHA, and fork repo
|
||||
- name: Resolve PR context
|
||||
id: pr
|
||||
uses: actions/github-script@60a0d83039c74a4aee543508d2ffcb1c3799cdea # v7
|
||||
@@ -66,32 +72,24 @@ jobs:
|
||||
}
|
||||
core.setOutput('number', pr.number);
|
||||
core.setOutput('sha', pr.head.sha);
|
||||
core.setOutput('repo', pr.head.repo.full_name);
|
||||
core.setOutput('branch', pr.head.ref);
|
||||
core.setOutput('is_fork', String(pr.head.repo.full_name !== pr.base.repo.full_name));
|
||||
|
||||
- name: Checkout PR head
|
||||
uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4
|
||||
with:
|
||||
repository: ${{ steps.pr.outputs.repo }}
|
||||
ref: ${{ steps.pr.outputs.sha }}
|
||||
fetch-depth: 1
|
||||
|
||||
# claude-code-action fetches branches by name from origin, which fails
|
||||
# for fork PRs. Work around by pushing the fork branch to origin so
|
||||
# the action can find it. Cleaned up in the post step below.
|
||||
- name: Push fork branch to origin
|
||||
if: steps.pr.outputs.is_fork == 'true'
|
||||
run: git push origin HEAD:refs/heads/${{ steps.pr.outputs.branch }}
|
||||
|
||||
- name: Run Claude Code Review
|
||||
id: claude-review
|
||||
uses: anthropics/claude-code-action@9469d113c6afd29550c402740f22d1a97dd1209b # v1
|
||||
with:
|
||||
claude_code_oauth_token: ${{ secrets.CLAUDE_CODE_OAUTH_TOKEN }}
|
||||
github_token: ${{ secrets.GITHUB_TOKEN }}
|
||||
allowed_non_write_users: '*'
|
||||
show_full_output: true
|
||||
plugin_marketplaces: 'https://github.com/anthropics/claude-code.git'
|
||||
plugins: 'code-review@claude-code-plugins'
|
||||
prompt: '/code-review:code-review ${{ github.repository }}/pull/${{ steps.pr.outputs.number }}'
|
||||
|
||||
# Clean up the temporary branch we pushed for fork PRs
|
||||
- name: Delete fork branch from origin
|
||||
if: always() && steps.pr.outputs.is_fork == 'true'
|
||||
run: git push origin --delete refs/heads/${{ steps.pr.outputs.branch }} || true
|
||||
|
||||
@@ -10,13 +10,42 @@ on:
|
||||
pull_request_review:
|
||||
types: [submitted]
|
||||
|
||||
# Serialize per-PR/issue to avoid racing comments.
|
||||
concurrency:
|
||||
group: claude-code-${{ github.event.issue.number || github.event.pull_request.number || github.event.issue.id }}
|
||||
cancel-in-progress: false
|
||||
|
||||
jobs:
|
||||
claude:
|
||||
if: |
|
||||
(github.event_name == 'issue_comment' && contains(github.event.comment.body, '@claude')) ||
|
||||
(github.event_name == 'pull_request_review_comment' && contains(github.event.comment.body, '@claude')) ||
|
||||
(github.event_name == 'pull_request_review' && contains(github.event.review.body, '@claude')) ||
|
||||
(github.event_name == 'issues' && (contains(github.event.issue.body, '@claude') || contains(github.event.issue.title, '@claude')))
|
||||
(
|
||||
github.event_name == 'issue_comment' &&
|
||||
contains(github.event.comment.body, '@claude') &&
|
||||
(github.event.comment.author_association == 'OWNER' ||
|
||||
github.event.comment.author_association == 'MEMBER' ||
|
||||
github.event.comment.author_association == 'COLLABORATOR')
|
||||
) ||
|
||||
(
|
||||
github.event_name == 'pull_request_review_comment' &&
|
||||
contains(github.event.comment.body, '@claude') &&
|
||||
(github.event.comment.author_association == 'OWNER' ||
|
||||
github.event.comment.author_association == 'MEMBER' ||
|
||||
github.event.comment.author_association == 'COLLABORATOR')
|
||||
) ||
|
||||
(
|
||||
github.event_name == 'pull_request_review' &&
|
||||
contains(github.event.review.body, '@claude') &&
|
||||
(github.event.review.author_association == 'OWNER' ||
|
||||
github.event.review.author_association == 'MEMBER' ||
|
||||
github.event.review.author_association == 'COLLABORATOR')
|
||||
) ||
|
||||
(
|
||||
github.event_name == 'issues' &&
|
||||
(contains(github.event.issue.body, '@claude') || contains(github.event.issue.title, '@claude')) &&
|
||||
(github.event.issue.author_association == 'OWNER' ||
|
||||
github.event.issue.author_association == 'MEMBER' ||
|
||||
github.event.issue.author_association == 'COLLABORATOR')
|
||||
)
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 30
|
||||
permissions:
|
||||
@@ -24,11 +53,47 @@ jobs:
|
||||
pull-requests: write
|
||||
issues: write
|
||||
id-token: write
|
||||
actions: read # Required for Claude to read CI results on PRs
|
||||
actions: read # required for Claude to read CI results on PRs
|
||||
steps:
|
||||
# For PR-related triggers, resolve the fork repo so we can checkout correctly.
|
||||
- name: Resolve PR context
|
||||
id: pr
|
||||
uses: actions/github-script@60a0d83039c74a4aee543508d2ffcb1c3799cdea # v7
|
||||
with:
|
||||
script: |
|
||||
// Determine if this event is PR-related
|
||||
let prNumber = null;
|
||||
if (context.eventName === 'issue_comment' && context.payload.issue.pull_request) {
|
||||
prNumber = context.payload.issue.number;
|
||||
} else if (context.eventName === 'pull_request_review_comment') {
|
||||
prNumber = context.payload.pull_request.number;
|
||||
} else if (context.eventName === 'pull_request_review') {
|
||||
prNumber = context.payload.pull_request.number;
|
||||
}
|
||||
|
||||
if (!prNumber) {
|
||||
core.setOutput('is_pr', 'false');
|
||||
return;
|
||||
}
|
||||
|
||||
const resp = await github.rest.pulls.get({
|
||||
owner: context.repo.owner,
|
||||
repo: context.repo.repo,
|
||||
pull_number: prNumber,
|
||||
});
|
||||
const pr = resp.data;
|
||||
|
||||
core.setOutput('is_pr', 'true');
|
||||
core.setOutput('number', String(prNumber));
|
||||
core.setOutput('sha', pr.head.sha);
|
||||
core.setOutput('repo', pr.head.repo.full_name);
|
||||
core.setOutput('branch', pr.head.ref);
|
||||
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4
|
||||
with:
|
||||
repository: ${{ steps.pr.outputs.is_pr == 'true' && steps.pr.outputs.repo || github.repository }}
|
||||
ref: ${{ steps.pr.outputs.is_pr == 'true' && steps.pr.outputs.sha || '' }}
|
||||
fetch-depth: 1
|
||||
|
||||
- name: Run Claude Code
|
||||
@@ -36,6 +101,9 @@ jobs:
|
||||
uses: anthropics/claude-code-action@9469d113c6afd29550c402740f22d1a97dd1209b # v1
|
||||
with:
|
||||
claude_code_oauth_token: ${{ secrets.CLAUDE_CODE_OAUTH_TOKEN }}
|
||||
github_token: ${{ secrets.GITHUB_TOKEN }}
|
||||
allowed_non_write_users: '*'
|
||||
show_full_output: true
|
||||
|
||||
# This is an optional setting that allows Claude to read CI results on PRs
|
||||
additional_permissions: |
|
||||
|
||||
@@ -0,0 +1,94 @@
|
||||
name: PR Description Check
|
||||
|
||||
on:
|
||||
pull_request:
|
||||
types: [opened, edited, reopened]
|
||||
branches: [main]
|
||||
|
||||
permissions:
|
||||
pull-requests: write
|
||||
|
||||
concurrency:
|
||||
group: pr-desc-${{ github.event.pull_request.number }}
|
||||
cancel-in-progress: true
|
||||
|
||||
jobs:
|
||||
check-description:
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 5
|
||||
steps:
|
||||
- name: Check PR description quality
|
||||
uses: actions/github-script@ed597411d8f924073f98dfc5c65a23a2325f34cd # v8
|
||||
with:
|
||||
script: |
|
||||
const MIN_BODY_LENGTH = 50;
|
||||
const LABEL = 'needs-description';
|
||||
|
||||
const pr = context.payload.pull_request;
|
||||
const body = (pr.body || '').trim();
|
||||
const owner = context.repo.owner;
|
||||
const repo = context.repo.repo;
|
||||
const number = pr.number;
|
||||
|
||||
const hasLabel = pr.labels.some(l => l.name === LABEL);
|
||||
|
||||
if (body.length < MIN_BODY_LENGTH) {
|
||||
// Add label if not already present
|
||||
if (!hasLabel) {
|
||||
await github.rest.issues.addLabels({
|
||||
owner, repo, issue_number: number,
|
||||
labels: [LABEL],
|
||||
});
|
||||
}
|
||||
|
||||
// Post or update a comment
|
||||
const marker = '<!-- pr-desc-check -->';
|
||||
const message = [
|
||||
marker,
|
||||
`### PR description is too short`,
|
||||
'',
|
||||
`This PR's description is **${body.length}** characters, ` +
|
||||
`but the minimum is **${MIN_BODY_LENGTH}**.`,
|
||||
'',
|
||||
'Please update the PR description to explain:',
|
||||
'- **What** this PR changes',
|
||||
'- **Why** the change is needed',
|
||||
'',
|
||||
'Use the PR template as a guide. This check will re-run when you edit the description.',
|
||||
].join('\n');
|
||||
|
||||
// Find existing bot comment to update (avoid spam)
|
||||
const comments = await github.rest.issues.listComments({
|
||||
owner, repo, issue_number: number,
|
||||
});
|
||||
const existing = comments.data.find(c =>
|
||||
c.body && c.body.includes(marker)
|
||||
);
|
||||
|
||||
if (existing) {
|
||||
await github.rest.issues.updateComment({
|
||||
owner, repo, comment_id: existing.id,
|
||||
body: message,
|
||||
});
|
||||
} else {
|
||||
await github.rest.issues.createComment({
|
||||
owner, repo, issue_number: number,
|
||||
body: message,
|
||||
});
|
||||
}
|
||||
|
||||
core.setFailed(
|
||||
`PR description is ${body.length} chars (minimum: ${MIN_BODY_LENGTH})`
|
||||
);
|
||||
} else {
|
||||
// Description is acceptable — remove the label if present
|
||||
if (hasLabel) {
|
||||
await github.rest.issues.removeLabel({
|
||||
owner, repo, issue_number: number,
|
||||
name: LABEL,
|
||||
}).catch(() => {});
|
||||
// .catch: label may have been removed manually
|
||||
}
|
||||
|
||||
core.info(`PR description OK (${body.length} chars)`);
|
||||
}
|
||||
@@ -12,6 +12,7 @@ jobs:
|
||||
uses: ./.github/workflows/ci.yml
|
||||
permissions:
|
||||
contents: read
|
||||
actions: read
|
||||
pull-requests: write
|
||||
|
||||
publish:
|
||||
|
||||
@@ -0,0 +1,103 @@
|
||||
name: Triage Sweep
|
||||
|
||||
on:
|
||||
workflow_dispatch:
|
||||
inputs:
|
||||
iqr_multiplier:
|
||||
description: >-
|
||||
IQR multiplier for outlier cutoff.
|
||||
cutoff = Q75 + multiplier * IQR.
|
||||
Higher = fewer outliers flagged.
|
||||
type: number
|
||||
default: 3.0
|
||||
max_outlier_pct:
|
||||
description: >-
|
||||
Maximum fraction of items that can be flagged as outliers (0-1).
|
||||
Hard cap to prevent over-flagging.
|
||||
type: number
|
||||
default: 0.05
|
||||
contamination:
|
||||
description: >-
|
||||
Expected fraction of outliers in the data (0-0.5).
|
||||
Controls how aggressively EllipticEnvelope downweights extremes.
|
||||
type: number
|
||||
default: 0.1
|
||||
cosine_threshold:
|
||||
description: >-
|
||||
Cosine similarity threshold for duplicate detection.
|
||||
Pairs with similarity above this are flagged as potential duplicates.
|
||||
Higher = only very similar pairs flagged.
|
||||
type: number
|
||||
default: 0.92
|
||||
max_items:
|
||||
description: >-
|
||||
Maximum number of open issues + PRs to process.
|
||||
Hard cap to prevent runaway costs on very large repos.
|
||||
type: number
|
||||
default: 500
|
||||
dry_run:
|
||||
description: >-
|
||||
Check this to only log results to the workflow summary.
|
||||
Uncheck to create a GitHub issue with the report and apply labels.
|
||||
type: boolean
|
||||
default: false
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
issues: write
|
||||
pull-requests: write
|
||||
|
||||
concurrency:
|
||||
group: triage-sweep
|
||||
cancel-in-progress: true
|
||||
|
||||
jobs:
|
||||
sweep:
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 30
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6
|
||||
with:
|
||||
sparse-checkout: .github/scripts/triage
|
||||
sparse-checkout-cone-mode: false
|
||||
fetch-depth: 1
|
||||
|
||||
- name: Set up Python
|
||||
uses: actions/setup-python@a309ff8b426b58ec0e2a45f0f869d46889d02405 # v6
|
||||
with:
|
||||
python-version: '3.12'
|
||||
cache: pip
|
||||
cache-dependency-path: .github/scripts/triage/requirements.txt
|
||||
|
||||
- name: Install dependencies
|
||||
run: pip install -r .github/scripts/triage/requirements.txt
|
||||
|
||||
- name: Cache FastEmbed model weights
|
||||
uses: actions/cache@668228422ae6a00e4ad889ee87cd7109ec5666a7 # v5
|
||||
with:
|
||||
path: ${{ github.workspace }}/.fastembed_cache
|
||||
key: fastembed-bge-small-en-v1.5
|
||||
|
||||
- name: Run triage sweep
|
||||
id: sweep
|
||||
env:
|
||||
GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||
GITHUB_REPOSITORY: ${{ github.repository }}
|
||||
FASTEMBED_CACHE_PATH: ${{ github.workspace }}/.fastembed_cache
|
||||
INPUT_IQR_MULTIPLIER: ${{ inputs.iqr_multiplier }}
|
||||
INPUT_MAX_OUTLIER_PCT: ${{ inputs.max_outlier_pct }}
|
||||
INPUT_CONTAMINATION: ${{ inputs.contamination }}
|
||||
INPUT_COSINE_THRESHOLD: ${{ inputs.cosine_threshold }}
|
||||
INPUT_MAX_ITEMS: ${{ inputs.max_items }}
|
||||
INPUT_DRY_RUN: ${{ inputs.dry_run }}
|
||||
run: python .github/scripts/triage/sweep.py
|
||||
|
||||
- name: Post summary
|
||||
if: always()
|
||||
run: |
|
||||
if [ -f /tmp/triage-report.md ]; then
|
||||
cat /tmp/triage-report.md >> "$GITHUB_STEP_SUMMARY"
|
||||
else
|
||||
echo "No report generated." >> "$GITHUB_STEP_SUMMARY"
|
||||
fi
|
||||
+10
-1
@@ -62,4 +62,13 @@ docs/plans/
|
||||
|
||||
gitnexus/test/fixtures/mini-repo/*.md
|
||||
gitnexus/test/fixtures/mini-repo/.claude
|
||||
gitnexus/test/fixtures/mini-repo/.gitignore
|
||||
gitnexus/test/fixtures/mini-repo/.gitignore
|
||||
|
||||
# Ignore csharp generated obj and bin folders
|
||||
gitnexus/test/fixtures/lang-resolution/**/obj
|
||||
gitnexus/test/fixtures/lang-resolution/**/bin
|
||||
GitNexus.sln
|
||||
# Git worktrees
|
||||
.worktrees/
|
||||
|
||||
/github/scripts/triage/__pycache__/
|
||||
@@ -0,0 +1,33 @@
|
||||
import { defineConfig } from 'vitest/config';
|
||||
|
||||
export default defineConfig({
|
||||
test: {
|
||||
globalSetup: ['test/global-setup.ts'],
|
||||
include: ['test/**/*.test.ts'],
|
||||
testTimeout: 30000,
|
||||
hookTimeout: 120000,
|
||||
pool: 'forks',
|
||||
globals: true,
|
||||
setupFiles: ['test/setup.ts'],
|
||||
teardownTimeout: 3000,
|
||||
dangerouslyIgnoreUnhandledErrors: true, // LadybugDB N-API destructor segfaults on fork exit — not a test failure
|
||||
coverage: {
|
||||
provider: 'v8',
|
||||
include: ['src/**/*.ts'],
|
||||
exclude: [
|
||||
'src/cli/index.ts', // CLI entry point (commander wiring)
|
||||
'src/server/**', // HTTP server (requires network)
|
||||
'src/core/wiki/**', // Wiki generation (requires LLM)
|
||||
],
|
||||
// Auto-ratchet: vitest bumps thresholds when coverage exceeds them.
|
||||
// CI will fail if a PR drops below these floors.
|
||||
thresholds: {
|
||||
statements: 26,
|
||||
branches: 23,
|
||||
functions: 28,
|
||||
lines: 27,
|
||||
autoUpdate: true,
|
||||
},
|
||||
},
|
||||
},
|
||||
});
|
||||
Executable
+33
@@ -0,0 +1,33 @@
|
||||
#!/usr/bin/env bash
|
||||
# Pre-commit hook (husky): typecheck + unit tests for both packages.
|
||||
# Mirrors CI checks from ci-quality.yml and ci-tests.yml.
|
||||
# Skip with: git commit --no-verify
|
||||
#
|
||||
# CI coverage:
|
||||
# quality / typecheck → tsc --noEmit in gitnexus/
|
||||
# quality / typecheck-web → tsc -b --noEmit in gitnexus-web/
|
||||
# tests / ubuntu+coverage → vitest run in gitnexus/ (all projects)
|
||||
# e2e / chromium → playwright (requires servers — skipped)
|
||||
|
||||
ROOT="$(git rev-parse --show-toplevel)"
|
||||
|
||||
WEB_CHANGED=$(git diff --cached --name-only -- 'gitnexus-web/' | head -1)
|
||||
CLI_CHANGED=$(git diff --cached --name-only -- 'gitnexus/' | head -1)
|
||||
|
||||
if [ -n "$WEB_CHANGED" ]; then
|
||||
echo "pre-commit: typechecking gitnexus-web (tsc -b)..."
|
||||
cd "$ROOT/gitnexus-web" && npx tsc -b --noEmit
|
||||
|
||||
echo "pre-commit: running gitnexus-web unit tests..."
|
||||
npx vitest run --reporter=dot
|
||||
fi
|
||||
|
||||
if [ -n "$CLI_CHANGED" ]; then
|
||||
echo "pre-commit: typechecking gitnexus..."
|
||||
cd "$ROOT/gitnexus" && npx tsc --noEmit
|
||||
|
||||
echo "pre-commit: running gitnexus unit tests (default project)..."
|
||||
npx vitest run --project default --reporter=dot
|
||||
fi
|
||||
|
||||
echo "pre-commit: all checks passed"
|
||||
@@ -1,7 +1,7 @@
|
||||
<!-- gitnexus:start -->
|
||||
# GitNexus — Code Intelligence
|
||||
|
||||
This project is indexed by GitNexus as **GitNexus** (1747 symbols, 4569 relationships, 130 execution flows). Use the GitNexus MCP tools to understand code, assess impact, and navigate safely.
|
||||
This project is indexed by GitNexus as **GitNexus** (2273 symbols, 5419 relationships, 174 execution flows). Use the GitNexus MCP tools to understand code, assess impact, and navigate safely.
|
||||
|
||||
> If any GitNexus tool warns the index is stale, run `npx gitnexus analyze` in terminal first.
|
||||
|
||||
@@ -69,6 +69,24 @@ Before completing any code modification task, verify:
|
||||
3. `gitnexus_detect_changes()` confirms changes match expected scope
|
||||
4. All d=1 (WILL BREAK) dependents were updated
|
||||
|
||||
## Keeping the Index Fresh
|
||||
|
||||
After committing code changes, the GitNexus index becomes stale. Re-run analyze to update it:
|
||||
|
||||
```bash
|
||||
npx gitnexus analyze
|
||||
```
|
||||
|
||||
If the index previously included embeddings, preserve them by adding `--embeddings`:
|
||||
|
||||
```bash
|
||||
npx gitnexus analyze --embeddings
|
||||
```
|
||||
|
||||
To check whether embeddings exist, inspect `.gitnexus/meta.json` — the `stats.embeddings` field shows the count (0 means no embeddings). **Running analyze without `--embeddings` will delete any previously generated embeddings.**
|
||||
|
||||
> Claude Code users: A PostToolUse hook handles this automatically after `git commit` and `git merge`.
|
||||
|
||||
## CLI
|
||||
|
||||
| Task | Read this skill file |
|
||||
@@ -79,25 +97,5 @@ Before completing any code modification task, verify:
|
||||
| Rename / extract / split / refactor | `.claude/skills/gitnexus/gitnexus-refactoring/SKILL.md` |
|
||||
| Tools, resources, schema reference | `.claude/skills/gitnexus/gitnexus-guide/SKILL.md` |
|
||||
| Index, status, clean, wiki CLI commands | `.claude/skills/gitnexus/gitnexus-cli/SKILL.md` |
|
||||
| Work in the Ingestion area (135 symbols) | `.claude/skills/generated/ingestion/SKILL.md` |
|
||||
| Work in the Workers area (70 symbols) | `.claude/skills/generated/workers/SKILL.md` |
|
||||
| Work in the Cli area (63 symbols) | `.claude/skills/generated/cli/SKILL.md` |
|
||||
| Work in the Kuzu area (52 symbols) | `.claude/skills/generated/kuzu/SKILL.md` |
|
||||
| Work in the Wiki area (52 symbols) | `.claude/skills/generated/wiki/SKILL.md` |
|
||||
| Work in the Embeddings area (48 symbols) | `.claude/skills/generated/embeddings/SKILL.md` |
|
||||
| Work in the Components area (42 symbols) | `.claude/skills/generated/components/SKILL.md` |
|
||||
| Work in the Local area (36 symbols) | `.claude/skills/generated/local/SKILL.md` |
|
||||
| Work in the Storage area (36 symbols) | `.claude/skills/generated/storage/SKILL.md` |
|
||||
| Work in the Services area (35 symbols) | `.claude/skills/generated/services/SKILL.md` |
|
||||
| Work in the Mcp area (32 symbols) | `.claude/skills/generated/mcp/SKILL.md` |
|
||||
| Work in the Llm area (30 symbols) | `.claude/skills/generated/llm/SKILL.md` |
|
||||
| Work in the Eval area (18 symbols) | `.claude/skills/generated/eval/SKILL.md` |
|
||||
| Work in the Bridge area (15 symbols) | `.claude/skills/generated/bridge/SKILL.md` |
|
||||
| Work in the Hooks area (14 symbols) | `.claude/skills/generated/hooks/SKILL.md` |
|
||||
| Work in the Search area (11 symbols) | `.claude/skills/generated/search/SKILL.md` |
|
||||
| Work in the Environments area (11 symbols) | `.claude/skills/generated/environments/SKILL.md` |
|
||||
| Work in the Analysis area (10 symbols) | `.claude/skills/generated/analysis/SKILL.md` |
|
||||
| Work in the Agents area (9 symbols) | `.claude/skills/generated/agents/SKILL.md` |
|
||||
| Work in the Graph area (6 symbols) | `.claude/skills/generated/graph/SKILL.md` |
|
||||
|
||||
<!-- gitnexus:end -->
|
||||
|
||||
@@ -2,6 +2,14 @@
|
||||
|
||||
All notable changes to GitNexus will be documented in this file.
|
||||
|
||||
## [Unreleased]
|
||||
|
||||
### Changed
|
||||
- Migrated from KuzuDB to LadybugDB v0.15 (`@ladybugdb/core`, `@ladybugdb/wasm-core`)
|
||||
- Renamed all internal paths from `kuzu` to `lbug` (storage: `.gitnexus/kuzu` → `.gitnexus/lbug`)
|
||||
- Added automatic cleanup of stale KuzuDB index files
|
||||
- LadybugDB v0.15 requires explicit VECTOR extension loading for semantic search
|
||||
|
||||
## [1.4.0] - 2026-03-13
|
||||
|
||||
### Added
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
<!-- gitnexus:start -->
|
||||
# GitNexus — Code Intelligence
|
||||
|
||||
This project is indexed by GitNexus as **GitNexus** (1747 symbols, 4569 relationships, 130 execution flows). Use the GitNexus MCP tools to understand code, assess impact, and navigate safely.
|
||||
This project is indexed by GitNexus as **GitNexus** (2273 symbols, 5419 relationships, 174 execution flows). Use the GitNexus MCP tools to understand code, assess impact, and navigate safely.
|
||||
|
||||
> If any GitNexus tool warns the index is stale, run `npx gitnexus analyze` in terminal first.
|
||||
|
||||
@@ -69,6 +69,24 @@ Before completing any code modification task, verify:
|
||||
3. `gitnexus_detect_changes()` confirms changes match expected scope
|
||||
4. All d=1 (WILL BREAK) dependents were updated
|
||||
|
||||
## Keeping the Index Fresh
|
||||
|
||||
After committing code changes, the GitNexus index becomes stale. Re-run analyze to update it:
|
||||
|
||||
```bash
|
||||
npx gitnexus analyze
|
||||
```
|
||||
|
||||
If the index previously included embeddings, preserve them by adding `--embeddings`:
|
||||
|
||||
```bash
|
||||
npx gitnexus analyze --embeddings
|
||||
```
|
||||
|
||||
To check whether embeddings exist, inspect `.gitnexus/meta.json` — the `stats.embeddings` field shows the count (0 means no embeddings). **Running analyze without `--embeddings` will delete any previously generated embeddings.**
|
||||
|
||||
> Claude Code users: A PostToolUse hook handles this automatically after `git commit` and `git merge`.
|
||||
|
||||
## CLI
|
||||
|
||||
| Task | Read this skill file |
|
||||
@@ -79,25 +97,5 @@ Before completing any code modification task, verify:
|
||||
| Rename / extract / split / refactor | `.claude/skills/gitnexus/gitnexus-refactoring/SKILL.md` |
|
||||
| Tools, resources, schema reference | `.claude/skills/gitnexus/gitnexus-guide/SKILL.md` |
|
||||
| Index, status, clean, wiki CLI commands | `.claude/skills/gitnexus/gitnexus-cli/SKILL.md` |
|
||||
| Work in the Ingestion area (135 symbols) | `.claude/skills/generated/ingestion/SKILL.md` |
|
||||
| Work in the Workers area (70 symbols) | `.claude/skills/generated/workers/SKILL.md` |
|
||||
| Work in the Cli area (63 symbols) | `.claude/skills/generated/cli/SKILL.md` |
|
||||
| Work in the Kuzu area (52 symbols) | `.claude/skills/generated/kuzu/SKILL.md` |
|
||||
| Work in the Wiki area (52 symbols) | `.claude/skills/generated/wiki/SKILL.md` |
|
||||
| Work in the Embeddings area (48 symbols) | `.claude/skills/generated/embeddings/SKILL.md` |
|
||||
| Work in the Components area (42 symbols) | `.claude/skills/generated/components/SKILL.md` |
|
||||
| Work in the Local area (36 symbols) | `.claude/skills/generated/local/SKILL.md` |
|
||||
| Work in the Storage area (36 symbols) | `.claude/skills/generated/storage/SKILL.md` |
|
||||
| Work in the Services area (35 symbols) | `.claude/skills/generated/services/SKILL.md` |
|
||||
| Work in the Mcp area (32 symbols) | `.claude/skills/generated/mcp/SKILL.md` |
|
||||
| Work in the Llm area (30 symbols) | `.claude/skills/generated/llm/SKILL.md` |
|
||||
| Work in the Eval area (18 symbols) | `.claude/skills/generated/eval/SKILL.md` |
|
||||
| Work in the Bridge area (15 symbols) | `.claude/skills/generated/bridge/SKILL.md` |
|
||||
| Work in the Hooks area (14 symbols) | `.claude/skills/generated/hooks/SKILL.md` |
|
||||
| Work in the Search area (11 symbols) | `.claude/skills/generated/search/SKILL.md` |
|
||||
| Work in the Environments area (11 symbols) | `.claude/skills/generated/environments/SKILL.md` |
|
||||
| Work in the Analysis area (10 symbols) | `.claude/skills/generated/analysis/SKILL.md` |
|
||||
| Work in the Agents area (9 symbols) | `.claude/skills/generated/agents/SKILL.md` |
|
||||
| Work in the Graph area (6 symbols) | `.claude/skills/generated/graph/SKILL.md` |
|
||||
|
||||
<!-- gitnexus:end -->
|
||||
|
||||
@@ -34,7 +34,7 @@ https://github.com/user-attachments/assets/172685ba-8e54-4ea7-9ad1-e31a3398da72
|
||||
|
||||
> *Like DeepWiki, but deeper.* DeepWiki helps you *understand* code. GitNexus lets you *analyze* it — because a knowledge graph tracks every relationship, not just descriptions.
|
||||
|
||||
**TL;DR:** The **Web UI** is a quick way to chat with any repo. The **CLI + MCP** is how you make your AI agent actually reliable — it gives Cursor, Claude Code, and friends a deep architectural view of your codebase so they stop missing dependencies, breaking call chains, and shipping blind edits. Even smaller models get full architectural clarity, making it compete with goliath models.
|
||||
**TL;DR:** The **Web UI** is a quick way to chat with any repo. The **CLI + MCP** is how you make your AI agent actually reliable — it gives Cursor, Claude Code, Codex, and friends a deep architectural view of your codebase so they stop missing dependencies, breaking call chains, and shipping blind edits. Even smaller models get full architectural clarity, making it compete with goliath models.
|
||||
|
||||
---
|
||||
|
||||
@@ -48,10 +48,10 @@ https://github.com/user-attachments/assets/172685ba-8e54-4ea7-9ad1-e31a3398da72
|
||||
| | **CLI + MCP** | **Web UI** |
|
||||
| ----------------- | -------------------------------------------------------------- | ------------------------------------------------------------ |
|
||||
| **What** | Index repos locally, connect AI agents via MCP | Visual graph explorer + AI chat in browser |
|
||||
| **For** | Daily development with Cursor, Claude Code, Windsurf, OpenCode | Quick exploration, demos, one-off analysis |
|
||||
| **For** | Daily development with Cursor, Claude Code, Codex, Windsurf, OpenCode | Quick exploration, demos, one-off analysis |
|
||||
| **Scale** | Full repos, any size | Limited by browser memory (~5k files), or unlimited via backend mode |
|
||||
| **Install** | `npm install -g gitnexus` | No install —[gitnexus.vercel.app](https://gitnexus.vercel.app) |
|
||||
| **Storage** | KuzuDB native (fast, persistent) | KuzuDB WASM (in-memory, per session) |
|
||||
| **Storage** | LadybugDB native (fast, persistent) | LadybugDB WASM (in-memory, per session) |
|
||||
| **Parsing** | Tree-sitter native bindings | Tree-sitter WASM |
|
||||
| **Privacy** | Everything local, no network | Everything in-browser, no server |
|
||||
|
||||
@@ -84,16 +84,23 @@ To configure MCP for your editor, run `npx gitnexus setup` once — or set it up
|
||||
| --------------------- | --- | ------ | -------------------- | -------------- |
|
||||
| **Claude Code** | Yes | Yes | Yes (PreToolUse + PostToolUse) | **Full** |
|
||||
| **Cursor** | Yes | Yes | — | MCP + Skills |
|
||||
| **Codex** | Yes | Yes | — | MCP + Skills |
|
||||
| **Windsurf** | Yes | — | — | MCP |
|
||||
| **OpenCode** | Yes | Yes | — | MCP + Skills |
|
||||
| **Codex** | Yes | — | — | MCP |
|
||||
|
||||
> **Claude Code** gets the deepest integration: MCP tools + agent skills + PreToolUse hooks that enrich searches with graph context + PostToolUse hooks that auto-reindex after commits.
|
||||
|
||||
### Community Integrations
|
||||
## Community Integrations
|
||||
|
||||
| Agent | Install | Source |
|
||||
|-------|---------|--------|
|
||||
| [pi](https://pi.dev) | `pi install npm:pi-gitnexus` | [pi-gitnexus](https://github.com/tintinweb/pi-gitnexus) |
|
||||
Built by the community — not officially maintained, but worth checking out.
|
||||
|
||||
| Project | Author | Description |
|
||||
|---------|--------|-------------|
|
||||
| [pi-gitnexus](https://github.com/tintinweb/pi-gitnexus) | [@tintinweb](https://github.com/tintinweb) | GitNexus plugin for [pi](https://pi.dev) — `pi install npm:pi-gitnexus` |
|
||||
| [gitnexus-stable-ops](https://github.com/ShunsukeHayashi/gitnexus-stable-ops) | [@ShunsukeHayashi](https://github.com/ShunsukeHayashi) | Stable ops & deployment workflows (Miyabi ecosystem) |
|
||||
|
||||
> Have a project built on GitNexus? Open a PR to add it here!
|
||||
|
||||
If you prefer manual configuration:
|
||||
|
||||
@@ -103,6 +110,12 @@ If you prefer manual configuration:
|
||||
claude mcp add gitnexus -- npx -y gitnexus@latest mcp
|
||||
```
|
||||
|
||||
**Codex** (full support — MCP + skills):
|
||||
|
||||
```bash
|
||||
codex mcp add gitnexus -- npx -y gitnexus@latest mcp
|
||||
```
|
||||
|
||||
**Cursor** (`~/.cursor/mcp.json` — global, works for all projects):
|
||||
|
||||
```json
|
||||
@@ -129,6 +142,14 @@ claude mcp add gitnexus -- npx -y gitnexus@latest mcp
|
||||
}
|
||||
```
|
||||
|
||||
**Codex** (`~/.codex/config.toml` for system scope, or `.codex/config.toml` for project scope):
|
||||
|
||||
```toml
|
||||
[mcp_servers.gitnexus]
|
||||
command = "npx"
|
||||
args = ["-y", "gitnexus@latest", "mcp"]
|
||||
```
|
||||
|
||||
### CLI Commands
|
||||
|
||||
```bash
|
||||
@@ -224,8 +245,8 @@ flowchart TD
|
||||
Server["server.ts"]
|
||||
Backend["LocalBackend"]
|
||||
Pool["Connection Pool"]
|
||||
ConnA["KuzuDB conn A"]
|
||||
ConnB["KuzuDB conn B"]
|
||||
ConnA["LadybugDB conn A"]
|
||||
ConnB["LadybugDB conn B"]
|
||||
end
|
||||
|
||||
Setup -->|"writes global MCP config"| CursorConfig["~/.cursor/mcp.json"]
|
||||
@@ -242,7 +263,7 @@ flowchart TD
|
||||
ConnB -->|"queries"| RepoB
|
||||
```
|
||||
|
||||
**How it works:** Each `gitnexus analyze` stores the index in `.gitnexus/` inside the repo (portable, gitignored) and registers a pointer in `~/.gitnexus/registry.json`. When an AI agent starts, the MCP server reads the registry and can serve any indexed repo. KuzuDB connections are opened lazily on first query and evicted after 5 minutes of inactivity (max 5 concurrent). If only one repo is indexed, the `repo` parameter is optional on all tools — agents don't need to change anything.
|
||||
**How it works:** Each `gitnexus analyze` stores the index in `.gitnexus/` inside the repo (portable, gitignored) and registers a pointer in `~/.gitnexus/registry.json`. When an AI agent starts, the MCP server reads the registry and can serve any indexed repo. LadybugDB connections are opened lazily on first query and evicted after 5 minutes of inactivity (max 5 concurrent). If only one repo is indexed, the `repo` parameter is optional on all tools — agents don't need to change anything.
|
||||
|
||||
---
|
||||
|
||||
@@ -263,7 +284,7 @@ npm install
|
||||
npm run dev
|
||||
```
|
||||
|
||||
The web UI uses the same indexing pipeline as the CLI but runs entirely in WebAssembly (Tree-sitter WASM, KuzuDB WASM, in-browser embeddings). It's great for quick exploration but limited by browser memory for larger repos.
|
||||
The web UI uses the same indexing pipeline as the CLI but runs entirely in WebAssembly (Tree-sitter WASM, LadybugDB WASM, in-browser embeddings). It's great for quick exploration but limited by browser memory for larger repos.
|
||||
|
||||
**Local Backend Mode:** Run `gitnexus serve` and open the web UI locally — it auto-detects the server and shows all your indexed repos, with full AI chat support. No need to re-upload or re-index. The agent's tools (Cypher queries, search, code navigation) route through the backend HTTP API automatically.
|
||||
|
||||
@@ -271,7 +292,7 @@ The web UI uses the same indexing pipeline as the CLI but runs entirely in WebAs
|
||||
|
||||
## The Problem GitNexus Solves
|
||||
|
||||
Tools like **Cursor**, **Claude Code**, **Cline**, **Roo Code**, and **Windsurf** are powerful — but they don't truly know your codebase structure.
|
||||
Tools like **Cursor**, **Claude Code**, **Codex**, **Cline**, **Roo Code**, and **Windsurf** are powerful — but they don't truly know your codebase structure.
|
||||
|
||||
**What happens:**
|
||||
|
||||
@@ -320,14 +341,30 @@ GitNexus builds a complete knowledge graph of your codebase through a multi-phas
|
||||
|
||||
1. **Structure** — Walks the file tree and maps folder/file relationships
|
||||
2. **Parsing** — Extracts functions, classes, methods, and interfaces using Tree-sitter ASTs
|
||||
3. **Resolution** — Resolves imports and function calls across files with language-aware logic
|
||||
3. **Resolution** — Resolves imports, function calls, heritage, constructor inference, and `self`/`this` receiver types across files with language-aware logic
|
||||
4. **Clustering** — Groups related symbols into functional communities
|
||||
5. **Processes** — Traces execution flows from entry points through call chains
|
||||
6. **Search** — Builds hybrid search indexes for fast retrieval
|
||||
|
||||
### Supported Languages
|
||||
|
||||
TypeScript, JavaScript, Python, Java, Kotlin, C, C++, C#, Go, Rust, PHP, Swift
|
||||
| Language | Imports | Named Bindings | Exports | Heritage | Type Annotations | Constructor Inference | Config | Frameworks | Entry Points |
|
||||
|----------|---------|----------------|---------|----------|-----------------|---------------------|--------|------------|-------------|
|
||||
| TypeScript | ✓ | ✓ | ✓ | ✓ | ✓ | ✓ | ✓ | ✓ | ✓ |
|
||||
| JavaScript | ✓ | ✓ | ✓ | ✓ | — | ✓ | ✓ | ✓ | ✓ |
|
||||
| Python | ✓ | ✓ | ✓ | ✓ | ✓ | ✓ | ✓ | ✓ | ✓ |
|
||||
| Java | ✓ | ✓ | ✓ | ✓ | ✓ | ✓ | — | ✓ | ✓ |
|
||||
| Kotlin | ✓ | ✓ | ✓ | ✓ | ✓ | ✓ | — | ✓ | ✓ |
|
||||
| C# | ✓ | ✓ | ✓ | ✓ | ✓ | ✓ | ✓ | ✓ | ✓ |
|
||||
| Go | ✓ | — | ✓ | ✓ | ✓ | ✓ | ✓ | ✓ | ✓ |
|
||||
| Rust | ✓ | ✓ | ✓ | ✓ | ✓ | ✓ | — | ✓ | ✓ |
|
||||
| PHP | ✓ | ✓ | ✓ | — | ✓ | ✓ | ✓ | ✓ | ✓ |
|
||||
| Ruby | ✓ | — | ✓ | ✓ | — | ✓ | — | ✓ | ✓ |
|
||||
| Swift | — | — | ✓ | ✓ | ✓ | ✓ | ✓ | ✓ | ✓ |
|
||||
| C | — | — | ✓ | — | ✓ | ✓ | — | ✓ | ✓ |
|
||||
| C++ | — | — | ✓ | ✓ | ✓ | ✓ | — | ✓ | ✓ |
|
||||
|
||||
**Imports** — cross-file import resolution · **Named Bindings** — `import { X as Y }` / re-export tracking · **Exports** — public/exported symbol detection · **Heritage** — class inheritance, interfaces, mixins · **Type Annotations** — explicit type extraction for receiver resolution · **Constructor Inference** — infer receiver type from constructor calls (`self`/`this` resolution included for all languages) · **Config** — language toolchain config parsing (tsconfig, go.mod, etc.) · **Frameworks** — AST-based framework pattern detection · **Entry Points** — entry point scoring heuristics
|
||||
|
||||
---
|
||||
|
||||
@@ -466,7 +503,7 @@ The wiki generator reads the indexed graph structure, groups files into modules
|
||||
| ------------------------- | ------------------------------------- | --------------------------------------- |
|
||||
| **Runtime** | Node.js (native) | Browser (WASM) |
|
||||
| **Parsing** | Tree-sitter native bindings | Tree-sitter WASM |
|
||||
| **Database** | KuzuDB native | KuzuDB WASM |
|
||||
| **Database** | LadybugDB native | LadybugDB WASM |
|
||||
| **Embeddings** | HuggingFace transformers.js (GPU/CPU) | transformers.js (WebGPU/WASM) |
|
||||
| **Search** | BM25 + semantic + RRF | BM25 + semantic + RRF |
|
||||
| **Agent Interface** | MCP (stdio) | LangChain ReAct agent |
|
||||
@@ -487,9 +524,10 @@ The wiki generator reads the indexed graph structure, groups files into modules
|
||||
|
||||
### Recently Completed
|
||||
|
||||
- [X] Constructor-Inferred Type Resolution, `self`/`this` Receiver Mapping
|
||||
- [X] Wiki Generation, Multi-File Rename, Git-Diff Impact Analysis
|
||||
- [X] Process-Grouped Search, 360-Degree Context, Claude Code Hooks
|
||||
- [X] Multi-Repo MCP, Zero-Config Setup, 11 Language Support
|
||||
- [X] Multi-Repo MCP, Zero-Config Setup, 13 Language Support
|
||||
- [X] Community Detection, Process Detection, Confidence Scoring
|
||||
- [X] Hybrid Search, Vector Index
|
||||
|
||||
@@ -506,7 +544,7 @@ The wiki generator reads the indexed graph structure, groups files into modules
|
||||
## Acknowledgments
|
||||
|
||||
- [Tree-sitter](https://tree-sitter.github.io/) — AST parsing
|
||||
- [KuzuDB](https://kuzudb.com/) — Embedded graph database with vector support
|
||||
- [LadybugDB](https://ladybugdb.com/) — Embedded graph database with vector support (formerly KuzuDB)
|
||||
- [Sigma.js](https://www.sigmajs.org/) — WebGL graph rendering
|
||||
- [transformers.js](https://huggingface.co/docs/transformers.js) — Browser ML
|
||||
- [Graphology](https://graphology.github.io/) — Graph data structures
|
||||
|
||||
@@ -0,0 +1,57 @@
|
||||
---
|
||||
review_agents: [kieran-typescript-reviewer, pattern-recognition-specialist, architecture-strategist, data-integrity-guardian, security-sentinel, performance-oracle, code-simplicity-reviewer]
|
||||
plan_review_agents: [kieran-typescript-reviewer, architecture-strategist, code-simplicity-reviewer]
|
||||
voltagent_agents: [voltagent-lang:typescript-pro, voltagent-qa-sec:security-auditor, voltagent-data-ai:database-optimizer]
|
||||
---
|
||||
|
||||
# Review Context
|
||||
|
||||
## Project Overview
|
||||
GitNexus is a code intelligence tool that builds a knowledge graph from source code using tree-sitter AST parsing across 12 languages and KuzuDB for graph storage. Two packages: `gitnexus/` (CLI/MCP, TypeScript) and `gitnexus-web/` (browser).
|
||||
|
||||
## Cross-Language Pattern Consistency (pattern-recognition-specialist)
|
||||
- 12 language-specific type extractors in `gitnexus/src/core/ingestion/type-extractors/` must follow identical patterns for: async unwrapping, constructor binding, namespace handling, nullable type stripping, for-loop element typing.
|
||||
- Past bugs: C#/Rust missing `await_expression` unwrapping that TypeScript handled correctly; PHP backslash namespace splitting inconsistent with other languages' `::` / `.` splitting.
|
||||
- When reviewing type extractor changes, verify the same pattern exists in ALL applicable language files — asymmetry is the #1 source of bugs.
|
||||
|
||||
## Data Integrity (data-integrity-guardian)
|
||||
- KuzuDB graph operations: schema in `gitnexus/src/core/kuzu/schema.ts`, adapter in `kuzu-adapter.ts`.
|
||||
- The ingestion pipeline writes symbols and relationships to the graph — changes to node/relation schemas or the ingestion pipeline can corrupt the index.
|
||||
- Known issue: KuzuDB `close()` hangs on Linux due to C++ destructor — use `detachKuzu()` pattern.
|
||||
- `lbug-adapter.ts` fallback path needs quote/newline escaping for Cypher injection prevention.
|
||||
|
||||
## Security (security-sentinel)
|
||||
- Cypher query construction in `lbug-adapter.ts` and `kuzu-adapter.ts` — watch for injection via unescaped user-provided symbol names.
|
||||
- CLI accepts `--repo` parameter and file paths — validate against path traversal.
|
||||
- MCP server exposes tools to external AI agents — all tool inputs are untrusted.
|
||||
|
||||
## Performance (performance-oracle)
|
||||
- Tree-sitter buffer size is adaptive (512KB–32MB) via `getTreeSitterBufferSize()` in `constants.ts`.
|
||||
- The ingestion pipeline processes entire repositories — O(n) per file with potential O(n²) in cross-file resolution.
|
||||
- KuzuDB batch inserts vs individual inserts matter for large repos.
|
||||
|
||||
## Architecture (architecture-strategist)
|
||||
- Ingestion pipeline phases: structure → parsing → imports → calls → heritage → processes → type resolution.
|
||||
- Shared modules: `export-detection.ts`, `constants.ts`, `utils.ts` — changes here have wide blast radius.
|
||||
- `gitnexus-web` package drifts behind CLI — flag if a change should be mirrored.
|
||||
|
||||
## Voltagent Supplementary Agents
|
||||
|
||||
Invoke these via the Agent tool alongside `/ce:review` for deeper specialist analysis. These cover gaps that compound-engineering agents don't:
|
||||
|
||||
### voltagent-lang:typescript-pro
|
||||
**When:** Changes touch type-resolution logic, generics, conditional types, or complex type-level programming in `type-env.ts`, `type-extractors/*.ts`, or `types.ts`.
|
||||
**Why:** The type resolution system uses advanced TypeScript patterns (discriminated unions, mapped types, recursive generics) that benefit from deep TS type-system review beyond what kieran-typescript-reviewer covers.
|
||||
|
||||
### voltagent-qa-sec:security-auditor
|
||||
**When:** Changes touch MCP tool handlers, Cypher query construction, CLI argument parsing, or any code that processes external input.
|
||||
**Why:** GitNexus is an MCP server — all tool inputs come from untrusted AI agents. Systematic OWASP-level audit catches injection vectors that spot-checking misses. Past finding: `lbug-adapter.ts` fallback path had unescaped newlines in Cypher queries.
|
||||
|
||||
### voltagent-data-ai:database-optimizer
|
||||
**When:** Changes touch `kuzu-adapter.ts`, `schema.ts`, `lbug-adapter.ts`, or any Cypher query construction/execution.
|
||||
**Why:** No CE agent specializes in graph database optimization. KuzuDB batch insert patterns, index usage, and query planning directly affect analysis speed on large repos.
|
||||
|
||||
## Review Tooling
|
||||
- Use `gitnexus_impact()` before approving changes to any symbol — check d=1 (WILL BREAK) callers.
|
||||
- Use `gitnexus_detect_changes({scope: "compare", base_ref: "main"})` to map PR diffs to affected execution flows.
|
||||
- Use claude-mem to surface past architectural decisions relevant to the code under review.
|
||||
+2
-2
@@ -148,7 +148,7 @@ Each mode has a `system_{mode}.jinja` + `instance_{mode}.jinja` pair. The agent
|
||||
|
||||
1. Docker container starts with SWE-bench instance (repo at specific commit)
|
||||
2. **GitNexus setup**: Node.js + gitnexus installed, `gitnexus analyze` runs (or restores from cache)
|
||||
3. **Eval-server starts**: `gitnexus eval-server` daemon (persistent HTTP server, keeps KuzuDB warm)
|
||||
3. **Eval-server starts**: `gitnexus eval-server` daemon (persistent HTTP server, keeps LadybugDB warm)
|
||||
4. **Standalone tool scripts installed** in `/usr/local/bin/` — works with `subprocess.run` (no `.bashrc` needed)
|
||||
5. Agent runs with the configured model + system prompt + GitNexus tools
|
||||
6. Agent's patch is extracted as a git diff
|
||||
@@ -167,7 +167,7 @@ Each tool script in `/usr/local/bin/` is standalone — no sourcing, no env inhe
|
||||
### Eval-server
|
||||
|
||||
The eval-server is a lightweight HTTP daemon that:
|
||||
- Keeps KuzuDB warm in memory (no cold start per tool call)
|
||||
- Keeps LadybugDB warm in memory (no cold start per tool call)
|
||||
- Returns LLM-friendly text (not raw JSON — saves tokens)
|
||||
- Includes next-step hints to guide tool chaining (query → context → impact → fix)
|
||||
- Auto-shuts down after idle timeout
|
||||
|
||||
@@ -2,7 +2,7 @@
|
||||
MCP Bridge for GitNexus
|
||||
|
||||
Starts the GitNexus MCP server as a subprocess and provides a Python interface
|
||||
to call MCP tools. Used by the bash wrapper scripts and the augmentation layer.
|
||||
to call MCP tools. Used by the bash wrapper scripts and the augmentation layer..
|
||||
|
||||
The bridge communicates with the MCP server via stdio using the JSON-RPC protocol.
|
||||
"""
|
||||
|
||||
@@ -160,7 +160,7 @@ function handlePreToolUse(input) {
|
||||
* PostToolUse handler — detect index staleness after git mutations.
|
||||
*
|
||||
* Instead of spawning a full `gitnexus analyze` synchronously (which blocks
|
||||
* the agent for up to 120s and risks KuzuDB corruption on timeout), we do a
|
||||
* the agent for up to 120s and risks LadybugDB corruption on timeout), we do a
|
||||
* lightweight staleness check: compare `git rev-parse HEAD` against the
|
||||
* lastCommit stored in `.gitnexus/meta.json`. If they differ, notify the
|
||||
* agent so it can decide when to reindex.
|
||||
|
||||
Generated
+1100
-30
File diff suppressed because it is too large
Load Diff
@@ -6,7 +6,8 @@
|
||||
"scripts": {
|
||||
"dev": "vite",
|
||||
"build": "tsc -b && vite build",
|
||||
"preview": "vite preview"
|
||||
"preview": "vite preview",
|
||||
"test": "vitest run"
|
||||
},
|
||||
"dependencies": {
|
||||
"@huggingface/transformers": "^3.0.0",
|
||||
@@ -33,7 +34,7 @@
|
||||
"graphology-layout-noverlap": "^0.4.2",
|
||||
"isomorphic-git": "^1.36.1",
|
||||
"jszip": "^3.10.1",
|
||||
"kuzu-wasm": "^0.11.1",
|
||||
"@ladybugdb/wasm-core": "^0.15.2",
|
||||
"langchain": "^1.2.10",
|
||||
"lru-cache": "^11.2.4",
|
||||
"lucide-react": "^0.562.0",
|
||||
@@ -65,6 +66,7 @@
|
||||
"tree-sitter-wasms": "^0.1.13",
|
||||
"typescript": "^5.4.5",
|
||||
"vite": "^5.2.0",
|
||||
"vite-plugin-static-copy": "^3.1.4"
|
||||
"vite-plugin-static-copy": "^3.1.4",
|
||||
"vitest": "^4.0.18"
|
||||
}
|
||||
}
|
||||
|
||||
Binary file not shown.
Binary file not shown.
+47
-31
@@ -13,6 +13,7 @@ import { FileEntry } from './services/zip';
|
||||
import { getActiveProviderConfig } from './core/llm/settings-service';
|
||||
import { createKnowledgeGraph } from './core/graph/graph';
|
||||
import { connectToServer, fetchRepos, normalizeServerUrl, type ConnectToServerResult } from './services/server-connection';
|
||||
import { HelpPanel } from './components/HelpPanel';
|
||||
|
||||
const AppContent = () => {
|
||||
const {
|
||||
@@ -28,6 +29,8 @@ const AppContent = () => {
|
||||
runPipelineFromFiles,
|
||||
isSettingsPanelOpen,
|
||||
setSettingsPanelOpen,
|
||||
isHelpDialogBoxOpen,
|
||||
setHelpDialogBoxOpen,
|
||||
refreshLLMSettings,
|
||||
initializeAgent,
|
||||
startEmbeddings,
|
||||
@@ -40,6 +43,8 @@ const AppContent = () => {
|
||||
availableRepos,
|
||||
setAvailableRepos,
|
||||
switchRepo,
|
||||
loadServerGraph,
|
||||
graph
|
||||
} = useAppState();
|
||||
|
||||
const graphCanvasRef = useRef<GraphCanvasHandle>(null);
|
||||
@@ -132,13 +137,13 @@ const AppContent = () => {
|
||||
}
|
||||
}, [setViewMode, setGraph, setFileContents, setProgress, setProjectName, runPipelineFromFiles, startEmbeddings, initializeAgent]);
|
||||
|
||||
const handleServerConnect = useCallback((result: ConnectToServerResult) => {
|
||||
const handleServerConnect = useCallback((result: ConnectToServerResult): Promise<void> => {
|
||||
// Extract project name from repoPath
|
||||
const repoPath = result.repoInfo.repoPath;
|
||||
const projectName = repoPath.split('/').pop() || 'server-project';
|
||||
setProjectName(projectName);
|
||||
|
||||
// Build KnowledgeGraph from server data (bypasses WASM pipeline entirely)
|
||||
// Build KnowledgeGraph from server data for visualization
|
||||
const graph = createKnowledgeGraph();
|
||||
for (const node of result.nodes) {
|
||||
graph.addNode(node);
|
||||
@@ -158,20 +163,30 @@ const AppContent = () => {
|
||||
// Transition directly to exploring view
|
||||
setViewMode('exploring');
|
||||
|
||||
// Initialize agent if LLM is configured
|
||||
if (getActiveProviderConfig()) {
|
||||
initializeAgent(projectName);
|
||||
}
|
||||
// Load graph into LadybugDB (in-browser WASM database) for Nexus AI queries,
|
||||
// then initialize agent once the database is ready
|
||||
const loadGraphPromise = loadServerGraph(result.nodes, result.relationships, result.fileContents)
|
||||
.then(() => {
|
||||
if (getActiveProviderConfig()) {
|
||||
return initializeAgent(projectName);
|
||||
}
|
||||
})
|
||||
.then(() => {
|
||||
startEmbeddings().catch((err) => {
|
||||
if (err?.name === 'WebGPUNotAvailableError' || err?.message?.includes('WebGPU')) {
|
||||
startEmbeddings('wasm').catch(console.warn);
|
||||
} else {
|
||||
console.warn('Embeddings auto-start failed:', err);
|
||||
}
|
||||
});
|
||||
})
|
||||
.catch((err) => {
|
||||
console.warn('Failed to load graph into LadybugDB:', err);
|
||||
// Agent won't work but graph visualization still does
|
||||
});
|
||||
|
||||
// Auto-start embeddings
|
||||
startEmbeddings().catch((err) => {
|
||||
if (err?.name === 'WebGPUNotAvailableError' || err?.message?.includes('WebGPU')) {
|
||||
startEmbeddings('wasm').catch(console.warn);
|
||||
} else {
|
||||
console.warn('Embeddings auto-start failed:', err);
|
||||
}
|
||||
});
|
||||
}, [setViewMode, setGraph, setFileContents, setProjectName, initializeAgent, startEmbeddings]);
|
||||
return loadGraphPromise;
|
||||
}, [setViewMode, setGraph, setFileContents, setProjectName, loadServerGraph, initializeAgent, startEmbeddings]);
|
||||
|
||||
// Auto-connect when ?server query param is present (bookmarkable shortcut)
|
||||
const autoConnectRan = useRef(false);
|
||||
@@ -203,16 +218,12 @@ const AppContent = () => {
|
||||
setProgress({ phase: 'extracting', percent: 97, message: 'Processing...', detail: 'Extracting file contents' });
|
||||
}
|
||||
}).then(async (result) => {
|
||||
handleServerConnect(result);
|
||||
|
||||
// Store server URL and fetch available repos for the repo switcher
|
||||
await handleServerConnect(result);
|
||||
setProgress(null);
|
||||
setServerBaseUrl(baseUrl);
|
||||
try {
|
||||
const repos = await fetchRepos(baseUrl);
|
||||
setAvailableRepos(repos);
|
||||
} catch (e) {
|
||||
console.warn('Failed to fetch repo list:', e);
|
||||
}
|
||||
fetchRepos(baseUrl)
|
||||
.then((repos) => setAvailableRepos(repos))
|
||||
.catch((e) => console.warn('Failed to fetch repo list:', e));
|
||||
}).catch((err) => {
|
||||
console.error('Auto-connect failed:', err);
|
||||
setProgress({
|
||||
@@ -246,16 +257,14 @@ const AppContent = () => {
|
||||
onFileSelect={handleFileSelect}
|
||||
onGitClone={handleGitClone}
|
||||
onServerConnect={async (result, serverUrl) => {
|
||||
handleServerConnect(result);
|
||||
await handleServerConnect(result);
|
||||
setProgress(null);
|
||||
if (serverUrl) {
|
||||
const baseUrl = normalizeServerUrl(serverUrl);
|
||||
setServerBaseUrl(baseUrl);
|
||||
try {
|
||||
const repos = await fetchRepos(baseUrl);
|
||||
setAvailableRepos(repos);
|
||||
} catch (e) {
|
||||
console.warn('Failed to fetch repo list:', e);
|
||||
}
|
||||
fetchRepos(baseUrl)
|
||||
.then((repos) => setAvailableRepos(repos))
|
||||
.catch((e) => console.warn('Failed to fetch repo list:', e));
|
||||
}
|
||||
}}
|
||||
/>
|
||||
@@ -300,6 +309,13 @@ const AppContent = () => {
|
||||
onSettingsSaved={handleSettingsSaved}
|
||||
/>
|
||||
|
||||
<HelpPanel
|
||||
isOpen={isHelpDialogBoxOpen}
|
||||
onClose={() => setHelpDialogBoxOpen(false)}
|
||||
nodeCount={graph!.nodes.length}
|
||||
edgeCount={graph!.relationships.length}
|
||||
/>
|
||||
|
||||
</div>
|
||||
);
|
||||
};
|
||||
|
||||
@@ -5,6 +5,42 @@ import { vscDarkPlus } from 'react-syntax-highlighter/dist/esm/styles/prism';
|
||||
import { useAppState } from '../hooks/useAppState';
|
||||
import { NODE_COLORS } from '../lib/constants';
|
||||
|
||||
/** Map file extension to Prism syntax highlighter language identifier */
|
||||
const getSyntaxLanguage = (filePath: string | undefined): string => {
|
||||
if (!filePath) return 'text';
|
||||
const ext = filePath.split('.').pop()?.toLowerCase();
|
||||
switch (ext) {
|
||||
case 'js': case 'jsx': case 'mjs': case 'cjs': return 'javascript';
|
||||
case 'ts': case 'tsx': case 'mts': case 'cts': return 'typescript';
|
||||
case 'py': case 'pyw': return 'python';
|
||||
case 'rb': case 'rake': case 'gemspec': return 'ruby';
|
||||
case 'java': return 'java';
|
||||
case 'go': return 'go';
|
||||
case 'rs': return 'rust';
|
||||
case 'c': case 'h': return 'c';
|
||||
case 'cpp': case 'cc': case 'cxx': case 'hpp': case 'hxx': case 'hh': return 'cpp';
|
||||
case 'cs': return 'csharp';
|
||||
case 'php': return 'php';
|
||||
case 'kt': case 'kts': return 'kotlin';
|
||||
case 'swift': return 'swift';
|
||||
case 'json': return 'json';
|
||||
case 'yaml': case 'yml': return 'yaml';
|
||||
case 'md': case 'mdx': return 'markdown';
|
||||
case 'html': case 'htm': case 'erb': return 'markup';
|
||||
case 'css': case 'scss': case 'sass': return 'css';
|
||||
case 'sh': case 'bash': case 'zsh': return 'bash';
|
||||
case 'sql': return 'sql';
|
||||
case 'xml': return 'xml';
|
||||
default: break;
|
||||
}
|
||||
// Handle extensionless Ruby files
|
||||
const basename = filePath.split('/').pop() || '';
|
||||
if (['Rakefile', 'Gemfile', 'Guardfile', 'Vagrantfile', 'Brewfile'].includes(basename)) return 'ruby';
|
||||
if (['Makefile'].includes(basename)) return 'makefile';
|
||||
if (['Dockerfile'].includes(basename)) return 'docker';
|
||||
return 'text';
|
||||
};
|
||||
|
||||
// Match the code theme used elsewhere in the app
|
||||
const customTheme = {
|
||||
...vscDarkPlus,
|
||||
@@ -267,12 +303,7 @@ export const CodeReferencesPanel = ({ onFocusNode }: CodeReferencesPanelProps) =
|
||||
<div className="flex-1 min-h-0 overflow-auto scrollbar-thin">
|
||||
{selectedFileContent ? (
|
||||
<SyntaxHighlighter
|
||||
language={
|
||||
selectedFilePath?.endsWith('.py') ? 'python' :
|
||||
selectedFilePath?.endsWith('.js') || selectedFilePath?.endsWith('.jsx') ? 'javascript' :
|
||||
selectedFilePath?.endsWith('.ts') || selectedFilePath?.endsWith('.tsx') ? 'typescript' :
|
||||
'text'
|
||||
}
|
||||
language={getSyntaxLanguage(selectedFilePath)}
|
||||
style={customTheme as any}
|
||||
showLineNumbers
|
||||
startingLineNumber={1}
|
||||
@@ -339,11 +370,7 @@ export const CodeReferencesPanel = ({ onFocusNode }: CodeReferencesPanelProps) =
|
||||
const hasRange = typeof ref.startLine === 'number';
|
||||
const startDisplay = hasRange ? (ref.startLine ?? 0) + 1 : undefined;
|
||||
const endDisplay = hasRange ? (ref.endLine ?? ref.startLine ?? 0) + 1 : undefined;
|
||||
const language =
|
||||
ref.filePath.endsWith('.py') ? 'python' :
|
||||
ref.filePath.endsWith('.js') || ref.filePath.endsWith('.jsx') ? 'javascript' :
|
||||
ref.filePath.endsWith('.ts') || ref.filePath.endsWith('.tsx') ? 'typescript' :
|
||||
'text';
|
||||
const language = getSyntaxLanguage(ref.filePath);
|
||||
|
||||
const isGlowing = glowRefId === ref.id;
|
||||
|
||||
|
||||
@@ -83,7 +83,7 @@ export const EmbeddingStatus = () => {
|
||||
<button
|
||||
onClick={handleTestArrayParams}
|
||||
className="flex items-center gap-1 px-2 py-1.5 bg-surface border border-border-subtle rounded-lg text-xs text-text-muted hover:bg-hover hover:text-text-secondary transition-all"
|
||||
title="Test if KuzuDB supports array params"
|
||||
title="Test if LadybugDB supports array params"
|
||||
>
|
||||
<FlaskConical className="w-3 h-3" />
|
||||
{testResult || 'Test'}
|
||||
|
||||
@@ -26,6 +26,9 @@ export const GraphCanvas = forwardRef<GraphCanvasHandle>((_, ref) => {
|
||||
blastRadiusNodeIds,
|
||||
isAIHighlightsEnabled,
|
||||
toggleAIHighlights,
|
||||
clearAIToolHighlights,
|
||||
clearAICitationHighlights,
|
||||
clearBlastRadius,
|
||||
animatedNodes,
|
||||
} = useAppState();
|
||||
const [hoveredNodeName, setHoveredNodeName] = useState<string | null>(null);
|
||||
@@ -305,9 +308,13 @@ export const GraphCanvas = forwardRef<GraphCanvasHandle>((_, ref) => {
|
||||
<div className="absolute top-4 right-4 z-20">
|
||||
<button
|
||||
onClick={() => {
|
||||
// If turning off, also clear process highlights
|
||||
if (isAIHighlightsEnabled) {
|
||||
setHighlightedNodeIds(new Set());
|
||||
// Turning off — clear AI highlights and selection (preserve user query highlights)
|
||||
clearAIToolHighlights();
|
||||
clearAICitationHighlights();
|
||||
clearBlastRadius();
|
||||
setSelectedNode(null);
|
||||
setSigmaSelectedNode(null);
|
||||
}
|
||||
toggleAIHighlights();
|
||||
}}
|
||||
|
||||
@@ -32,6 +32,7 @@ export const Header = ({ onFocusNode, availableRepos = [], onSwitchRepo }: Heade
|
||||
isRightPanelOpen,
|
||||
rightPanelTab,
|
||||
setSettingsPanelOpen,
|
||||
setHelpDialogBoxOpen
|
||||
} = useAppState();
|
||||
const [isRepoDropdownOpen, setIsRepoDropdownOpen] = useState(false);
|
||||
const repoDropdownRef = useRef<HTMLDivElement>(null);
|
||||
@@ -266,10 +267,13 @@ export const Header = ({ onFocusNode, availableRepos = [], onSwitchRepo }: Heade
|
||||
className="w-9 h-9 flex items-center justify-center rounded-md text-text-secondary hover:bg-hover hover:text-text-primary transition-colors"
|
||||
title="AI Settings"
|
||||
>
|
||||
<Settings className="w-[18px] h-[18px]" />
|
||||
<Settings className="w-4.5 h-4.5" />
|
||||
</button>
|
||||
<button className="w-9 h-9 flex items-center justify-center rounded-md text-text-secondary hover:bg-hover hover:text-text-primary transition-colors">
|
||||
<HelpCircle className="w-[18px] h-[18px]" />
|
||||
<button
|
||||
title="Help"
|
||||
onClick={() => setHelpDialogBoxOpen(true)}
|
||||
className="w-9 h-9 flex items-center justify-center rounded-md text-text-secondary hover:bg-hover hover:text-text-primary transition-colors">
|
||||
<HelpCircle className="w-4.5 h-4.5" />
|
||||
</button>
|
||||
|
||||
{/* AI Button */}
|
||||
|
||||
@@ -0,0 +1,390 @@
|
||||
import React, { useState } from 'react';
|
||||
import { X, GitBranch, Search, Filter, Zap, Keyboard, BarChart2, HelpCircle } from 'lucide-react';
|
||||
|
||||
interface HelpPanelProps {
|
||||
isOpen: boolean;
|
||||
onClose: () => void;
|
||||
nodeCount: number;
|
||||
edgeCount: number;
|
||||
}
|
||||
|
||||
type TabId = 'overview' | 'graph' | 'search' | 'ai' | 'shortcuts' | 'status';
|
||||
|
||||
interface Tab {
|
||||
id: TabId;
|
||||
label: string;
|
||||
icon: React.ReactNode;
|
||||
}
|
||||
|
||||
const tabs: Tab[] = [
|
||||
{ id: 'overview', label: 'Overview', icon: <HelpCircle className="w-4 h-4" /> },
|
||||
{ id: 'graph', label: 'Graph & nodes', icon: <GitBranch className="w-4 h-4" /> },
|
||||
{ id: 'search', label: 'Search & filter', icon: <Search className="w-4 h-4" /> },
|
||||
{ id: 'ai', label: 'Nexus AI', icon: <Zap className="w-4 h-4" /> },
|
||||
{ id: 'shortcuts', label: 'Shortcuts', icon: <Keyboard className="w-4 h-4" /> },
|
||||
{ id: 'status', label: 'Status bar', icon: <BarChart2 className="w-4 h-4" /> },
|
||||
];
|
||||
|
||||
const shortcuts = [
|
||||
{ label: 'Search nodes', mac: '⌘ K', win: 'Ctrl K' },
|
||||
{ label: 'Deselect / close', mac: 'Esc', win: 'Esc' },
|
||||
];
|
||||
|
||||
const nodeColors = [
|
||||
{ color: '#10b981', label: 'Function', desc: 'Function declarations' },
|
||||
{ color: '#3b82f6', label: 'File', desc: 'Source files' },
|
||||
{ color: '#f59e0b', label: 'Class', desc: 'Class declarations' },
|
||||
{ color: '#14b8a6', label: 'Method', desc: 'Class methods' },
|
||||
{ color: '#ec4899', label: 'Interface', desc: 'TypeScript interfaces' },
|
||||
{ color: '#6366f1', label: 'Folder', desc: 'Directory nodes' },
|
||||
];
|
||||
|
||||
const getStatusItems = (nodeCount: number, edgeCount: number) => [
|
||||
{ badge: <span style={{ width: 8, height: 8, borderRadius: '50%', background: '#34d399', display: 'inline-block', flexShrink: 0 }} />, title: 'Ready', desc: 'Graph is fully loaded and interactive' },
|
||||
{ badge: <span style={{ fontSize: 12, fontWeight: 500, color: '#a78bfa', flexShrink: 0 }}>{nodeCount}</span>, title: 'Nodes count', desc: 'Total files and symbols in the graph' },
|
||||
{ badge: <span style={{ fontSize: 12, fontWeight: 500, color: '#60a5fa', flexShrink: 0 }}>{edgeCount}</span>, title: 'Edges count', desc: 'Import / dependency connections' },
|
||||
{ badge: <span style={{ fontSize: 11, fontWeight: 500, color: '#34d399', flexShrink: 0, whiteSpace: 'nowrap' }}>Semantic Ready</span>, title: 'AI index status', desc: 'Repo is fully indexed for AI queries' },
|
||||
// { badge: <span style={{ fontSize: 11, fontWeight: 500, color: '#9ca3af', flexShrink: 0 }}>typescript</span>, title: 'Language', desc: 'Primary language detected in the repo' },
|
||||
];
|
||||
|
||||
const kbdStyle: React.CSSProperties = {
|
||||
fontSize: 11,
|
||||
background: 'rgba(255,255,255,0.08)',
|
||||
borderRadius: 4,
|
||||
padding: '2px 8px',
|
||||
color: '#e2e2e8',
|
||||
fontFamily: 'monospace',
|
||||
border: '0.5px solid rgba(255,255,255,0.12)',
|
||||
whiteSpace: 'nowrap',
|
||||
};
|
||||
|
||||
const kbdWinStyle: React.CSSProperties = {
|
||||
...kbdStyle,
|
||||
color: '#93c5fd',
|
||||
};
|
||||
|
||||
function TabContent({ active, nodeCount, edgeCount }: {
|
||||
active: TabId;
|
||||
nodeCount: number;
|
||||
edgeCount: number;
|
||||
}) {
|
||||
if (active === 'overview') return (
|
||||
<div style={{ display: 'flex', flexDirection: 'column', gap: 10 }}>
|
||||
<p style={{ fontSize: 11, color: '#6b7280', margin: '0 0 4px', textTransform: 'uppercase', letterSpacing: '0.08em' }}>Getting started</p>
|
||||
|
||||
<div style={{ background: 'rgba(255,255,255,0.04)', borderRadius: 10, padding: '12px 14px', borderLeft: '2px solid #a78bfa' }}>
|
||||
<p style={{ fontSize: 13, fontWeight: 500, color: '#e2e2e8', margin: '0 0 4px' }}>What is GitNexus?</p>
|
||||
<p style={{ fontSize: 12, color: '#9ca3af', margin: 0, lineHeight: 1.6 }}>An interactive graph explorer for your codebase. Every file, function, and import becomes a node you can explore, query, and navigate visually.</p>
|
||||
</div>
|
||||
|
||||
<div style={{ background: 'rgba(255,255,255,0.04)', borderRadius: 10, padding: '12px 14px', borderLeft: '2px solid #34d399' }}>
|
||||
<p style={{ fontSize: 13, fontWeight: 500, color: '#e2e2e8', margin: '0 0 4px' }}>Your current repo</p>
|
||||
<p style={{ fontSize: 12, color: '#9ca3af', margin: 0, lineHeight: 1.6 }}>
|
||||
Loaded: <span style={{ color: '#a78bfa', fontFamily: 'monospace' }}></span> {nodeCount} nodes · {edgeCount} edges
|
||||
</p>
|
||||
</div>
|
||||
|
||||
<div style={{ background: 'rgba(255,255,255,0.04)', borderRadius: 10, padding: '12px 14px', borderLeft: '2px solid #60a5fa' }}>
|
||||
<p style={{ fontSize: 13, fontWeight: 500, color: '#e2e2e8', margin: '0 0 4px' }}>Three ways to explore</p>
|
||||
<p style={{ fontSize: 12, color: '#9ca3af', margin: 0, lineHeight: 1.6 }}>
|
||||
<strong style={{ color: '#e2e2e8', fontWeight: 500 }}>1.</strong> Click nodes to inspect
|
||||
<br/>
|
||||
<strong style={{ color: '#e2e2e8', fontWeight: 500 }}>2.</strong> Search by name or type
|
||||
<br/>
|
||||
<strong style={{ color: '#e2e2e8', fontWeight: 500 }}>3.</strong> Ask Nexus AI a natural language question
|
||||
</p>
|
||||
</div>
|
||||
|
||||
<div style={{ background: 'rgba(255,255,255,0.04)', borderRadius: 10, padding: '12px 14px', borderLeft: '2px solid #fbbf24' }}>
|
||||
<p style={{ fontSize: 13, fontWeight: 500, color: '#e2e2e8', margin: '0 0 4px' }}>Navigation</p>
|
||||
<p style={{ fontSize: 12, color: '#9ca3af', margin: 0, lineHeight: 1.6 }}>
|
||||
· Scroll to zoom <br/>
|
||||
· Click and drag to pan <br/>
|
||||
· Double-click a node to focus its subgraph
|
||||
</p>
|
||||
</div>
|
||||
</div>
|
||||
);
|
||||
|
||||
if (active === 'graph') return (
|
||||
<div style={{ display: 'flex', flexDirection: 'column', gap: 12 }}>
|
||||
<p style={{ fontSize: 11, color: '#6b7280', margin: '0 0 4px', textTransform: 'uppercase', letterSpacing: '0.08em' }}>Node color legend</p>
|
||||
|
||||
{nodeColors.map(({ color, label, desc }) => (
|
||||
<div key={label} style={{ display: 'flex', gap: 10, alignItems: 'flex-start' }}>
|
||||
<span style={{ width: 12, height: 12, borderRadius: '50%', background: color, flexShrink: 0, marginTop: 2 }} />
|
||||
<div>
|
||||
<p style={{ fontSize: 12, fontWeight: 500, color: '#e2e2e8', margin: '0 0 2px' }}>{label} nodes</p>
|
||||
<p style={{ fontSize: 12, color: '#9ca3af', margin: 0 }}>{desc}</p>
|
||||
</div>
|
||||
</div>
|
||||
))}
|
||||
|
||||
<div style={{ borderTop: '0.5px solid rgba(255,255,255,0.08)', margin: '4px 0' }} />
|
||||
|
||||
<p style={{ fontSize: 12, color: '#9ca3af', margin: 0, lineHeight: 1.6 }}>
|
||||
Node <strong style={{ color: '#e2e2e8', fontWeight: 500 }}>size</strong> reflects connection count — larger nodes are depended on by more files. Edges point from importer → imported.
|
||||
</p>
|
||||
|
||||
<div style={{ background: 'rgba(255,255,255,0.04)', borderRadius: 10, padding: '10px 14px' }}>
|
||||
<p style={{ fontSize: 12, color: '#9ca3af', margin: 0, lineHeight: 1.6 }}>
|
||||
Click any node to open its detail panel — showing imports, exports, and reverse dependencies.
|
||||
</p>
|
||||
</div>
|
||||
</div>
|
||||
);
|
||||
|
||||
if (active === 'search') return (
|
||||
<div style={{ display: 'flex', flexDirection: 'column', gap: 10 }}>
|
||||
<p style={{ fontSize: 11, color: '#6b7280', margin: '0 0 4px', textTransform: 'uppercase', letterSpacing: '0.08em' }}>Search & filter</p>
|
||||
|
||||
<div style={{ background: 'rgba(255,255,255,0.04)', borderRadius: 10, padding: '12px 14px' }}>
|
||||
<div style={{ display: 'flex', alignItems: 'center', gap: 8, marginBottom: 6 }}>
|
||||
<kbd style={kbdStyle}>⌘K</kbd>/
|
||||
<kbd style={kbdStyle}>Ctrl K</kbd>
|
||||
<p style={{ fontSize: 12, fontWeight: 500, color: '#e2e2e8', margin: 0 }}>Search nodes</p>
|
||||
</div>
|
||||
<p style={{ fontSize: 12, color: '#9ca3af', margin: 0, lineHeight: 1.6 }}>
|
||||
Search by filename, function name, or import path. Matching nodes are highlighted live in the graph.
|
||||
</p>
|
||||
</div>
|
||||
|
||||
<div style={{ background: 'rgba(255,255,255,0.04)', borderRadius: 10, padding: '12px 14px' }}>
|
||||
<div style={{ display: 'flex', alignItems: 'center', gap: 8, marginBottom: 6 }}>
|
||||
<Filter style={{ width: 14, height: 14, color: '#a78bfa', flexShrink: 0 }} />
|
||||
<p style={{ fontSize: 12, fontWeight: 500, color: '#e2e2e8', margin: 0 }}>Filter panel</p>
|
||||
</div>
|
||||
<p style={{ fontSize: 12, color: '#9ca3af', margin: 0, lineHeight: 1.6 }}>
|
||||
Use the filter icon in the left sidebar to isolate specific node types, hide leaf nodes, or focus on a depth range from a selected root.
|
||||
</p>
|
||||
</div>
|
||||
|
||||
<div style={{ background: 'rgba(255,255,255,0.04)', borderRadius: 10, padding: '12px 14px' }}>
|
||||
<p style={{ fontSize: 12, fontWeight: 500, color: '#e2e2e8', margin: '0 0 6px' }}>Search syntax</p>
|
||||
{[
|
||||
{ query: 'auth', hint: 'match by name fragment' },
|
||||
{ query: './utils/', hint: 'match by path prefix' },
|
||||
{ query: 'type:config', hint: 'filter by node type' },
|
||||
].map(({ query, hint }) => (
|
||||
<div key={query} style={{ display: 'flex', alignItems: 'baseline', gap: 8, marginBottom: 4 }}>
|
||||
<code style={{ fontSize: 11, color: '#a78bfa', background: 'rgba(167,139,250,0.1)', borderRadius: 4, padding: '1px 6px', fontFamily: 'monospace', flexShrink: 0 }}>{query}</code>
|
||||
<span style={{ fontSize: 12, color: '#6b7280' }}>{hint}</span>
|
||||
</div>
|
||||
))}
|
||||
</div>
|
||||
</div>
|
||||
);
|
||||
|
||||
if (active === 'ai') return (
|
||||
<div style={{ display: 'flex', flexDirection: 'column', gap: 10 }}>
|
||||
<p style={{ fontSize: 11, color: '#6b7280', margin: '0 0 4px', textTransform: 'uppercase', letterSpacing: '0.08em' }}>Nexus AI</p>
|
||||
|
||||
<div style={{ background: 'rgba(167,139,250,0.08)', border: '0.5px solid rgba(167,139,250,0.25)', borderRadius: 10, padding: '12px 14px' }}>
|
||||
<p style={{ fontSize: 12, fontWeight: 500, color: '#a78bfa', margin: '0 0 4px' }}>✓ Semantic Ready</p>
|
||||
<p style={{ fontSize: 12, color: '#9ca3af', margin: 0, lineHeight: 1.6 }}>
|
||||
Your repo is indexed and ready for semantic queries. Nexus AI understands code structure and relationships, not just file names.
|
||||
</p>
|
||||
</div>
|
||||
|
||||
<p style={{ fontSize: 12, color: '#9ca3af', margin: '4px 0 2px' }}>Try asking:</p>
|
||||
{[
|
||||
'"Which files depend on the auth module?"',
|
||||
'"Find circular dependencies in this repo"',
|
||||
'"What are the most connected components?"',
|
||||
'"Show me all files that import useEffect"',
|
||||
].map(q => (
|
||||
<div key={q} style={{ background: 'rgba(255,255,255,0.04)', borderRadius: 8, padding: '8px 12px', fontSize: 12, color: '#e2e2e8', fontStyle: 'italic' }}>{q}</div>
|
||||
))}
|
||||
|
||||
<div style={{ borderTop: '0.5px solid rgba(255,255,255,0.08)', margin: '4px 0' }} />
|
||||
|
||||
<p style={{ fontSize: 12, color: '#6b7280', margin: 0, lineHeight: 1.6 }}>
|
||||
Open the prompt via the{' '}
|
||||
<span style={{ color: '#e2e2e8' }}>Nexus AI</span> button (top-right).
|
||||
</p>
|
||||
</div>
|
||||
);
|
||||
|
||||
if (active === 'shortcuts') return (
|
||||
<div style={{ display: 'flex', flexDirection: 'column', gap: 0 }}>
|
||||
{/* Column headers */}
|
||||
<div style={{
|
||||
display: 'grid',
|
||||
gridTemplateColumns: '1fr 80px 88px',
|
||||
gap: 8,
|
||||
padding: '0 0 8px',
|
||||
borderBottom: '0.5px solid rgba(255,255,255,0.08)',
|
||||
marginBottom: 4,
|
||||
}}>
|
||||
<span style={{ fontSize: 11, color: '#6b7280', textTransform: 'uppercase', letterSpacing: '0.08em' }}>Action</span>
|
||||
<span style={{ fontSize: 11, color: '#6b7280', textTransform: 'uppercase', letterSpacing: '0.08em', textAlign: 'center' }}>Mac</span>
|
||||
<span style={{ fontSize: 11, color: '#93c5fd', textTransform: 'uppercase', letterSpacing: '0.08em', textAlign: 'center' }}>Windows</span>
|
||||
</div>
|
||||
|
||||
{shortcuts.map(({ label, mac, win }, i) => (
|
||||
<div
|
||||
key={label}
|
||||
style={{
|
||||
display: 'grid',
|
||||
gridTemplateColumns: '1fr 80px 88px',
|
||||
gap: 8,
|
||||
alignItems: 'center',
|
||||
padding: '8px 0',
|
||||
borderBottom: i < shortcuts.length - 1 ? '0.5px solid rgba(255,255,255,0.05)' : 'none',
|
||||
}}
|
||||
>
|
||||
<span style={{ fontSize: 12, color: '#9ca3af' }}>{label}</span>
|
||||
<span style={{ display: 'flex', justifyContent: 'center' }}>
|
||||
<kbd style={kbdStyle}>{mac}</kbd>
|
||||
</span>
|
||||
<span style={{ display: 'flex', justifyContent: 'center' }}>
|
||||
<kbd style={kbdWinStyle}>{win}</kbd>
|
||||
</span>
|
||||
</div>
|
||||
))}
|
||||
</div>
|
||||
);
|
||||
|
||||
if (active === 'status') return (
|
||||
<div style={{ display: 'flex', flexDirection: 'column', gap: 8 }}>
|
||||
<p style={{ fontSize: 11, color: '#6b7280', margin: '0 0 4px', textTransform: 'uppercase', letterSpacing: '0.08em' }}>Status bar explained</p>
|
||||
{getStatusItems(nodeCount, edgeCount).map(({ badge, title, desc }) => (
|
||||
<div key={title} style={{ background: 'rgba(255,255,255,0.04)', borderRadius: 10, padding: '10px 14px', display: 'flex', gap: 12, alignItems: 'center' }}>
|
||||
{badge}
|
||||
<div>
|
||||
<p style={{ fontSize: 12, fontWeight: 500, color: '#e2e2e8', margin: '0 0 2px' }}>{title}</p>
|
||||
<p style={{ fontSize: 12, color: '#9ca3af', margin: 0 }}>{desc}</p>
|
||||
</div>
|
||||
</div>
|
||||
))}
|
||||
</div>
|
||||
);
|
||||
|
||||
return null;
|
||||
}
|
||||
|
||||
export const HelpPanel = ({ isOpen, onClose, nodeCount, edgeCount }: HelpPanelProps) => {
|
||||
const [active, setActive] = useState<TabId>('overview');
|
||||
|
||||
if (!isOpen) return null;
|
||||
|
||||
return (
|
||||
<div style={{ position: 'fixed', inset: 0, zIndex: 50, display: 'flex', alignItems: 'center', justifyContent: 'center' }}>
|
||||
{/* Backdrop */}
|
||||
<div
|
||||
style={{ position: 'absolute', inset: 0, background: 'rgba(0,0,0,0.6)', backdropFilter: 'blur(4px)' }}
|
||||
onClick={onClose}
|
||||
/>
|
||||
|
||||
{/* Panel */}
|
||||
<div style={{
|
||||
position: 'relative',
|
||||
background: '#12121a',
|
||||
border: '0.5px solid rgba(255,255,255,0.12)',
|
||||
borderRadius: 16,
|
||||
boxShadow: '0 25px 60px rgba(0,0,0,0.7)',
|
||||
width: '100%',
|
||||
maxWidth: 680,
|
||||
margin: '0 16px',
|
||||
height: '60vh',
|
||||
display: 'flex',
|
||||
flexDirection: 'column',
|
||||
overflow: 'hidden',
|
||||
fontFamily: 'var(--font-mono, monospace)',
|
||||
}}>
|
||||
|
||||
{/* Header */}
|
||||
<div style={{
|
||||
display: 'flex',
|
||||
alignItems: 'center',
|
||||
justifyContent: 'space-between',
|
||||
padding: '16px 20px',
|
||||
borderBottom: '0.5px solid rgba(255,255,255,0.08)',
|
||||
background: 'rgba(255,255,255,0.02)',
|
||||
}}>
|
||||
<div style={{ display: 'flex', alignItems: 'center', gap: 12 }}>
|
||||
<div style={{ width: 40, height: 40, display: 'flex', alignItems: 'center', justifyContent: 'center', background: 'rgba(167,139,250,0.15)', borderRadius: 12 }}>
|
||||
<HelpCircle style={{ width: 20, height: 20, color: '#a78bfa' }} />
|
||||
</div>
|
||||
<div>
|
||||
<h2 style={{ fontSize: 16, fontWeight: 600, color: '#e2e2e8', margin: 0 }}>Help & Reference</h2>
|
||||
<p style={{ fontSize: 12, color: '#6b7280', margin: 0 }}>GitNexus — graph explorer</p>
|
||||
</div>
|
||||
</div>
|
||||
<button
|
||||
onClick={onClose}
|
||||
style={{ padding: 8, color: '#6b7280', background: 'transparent', border: 'none', borderRadius: 8, cursor: 'pointer', display: 'flex', alignItems: 'center', justifyContent: 'center', transition: 'color 0.15s' }}
|
||||
onMouseEnter={e => (e.currentTarget.style.color = '#e2e2e8')}
|
||||
onMouseLeave={e => (e.currentTarget.style.color = '#6b7280')}
|
||||
>
|
||||
<X style={{ width: 20, height: 20 }} />
|
||||
</button>
|
||||
</div>
|
||||
|
||||
{/* Body: sidebar + content */}
|
||||
<div style={{ display: 'grid', gridTemplateColumns: '168px 1fr', flex: 1, overflow: 'hidden' }}>
|
||||
|
||||
{/* Sidebar nav */}
|
||||
<div style={{ borderRight: '0.5px solid rgba(255,255,255,0.08)', padding: '12px 8px', display: 'flex', flexDirection: 'column', gap: 2 }}>
|
||||
{tabs.map(({ id, label, icon }) => {
|
||||
const isActive = active === id;
|
||||
return (
|
||||
<button
|
||||
key={id}
|
||||
onClick={() => setActive(id)}
|
||||
style={{
|
||||
display: 'flex',
|
||||
alignItems: 'center',
|
||||
gap: 8,
|
||||
textAlign: 'left',
|
||||
background: isActive ? 'rgba(167,139,250,0.12)' : 'transparent',
|
||||
border: 'none',
|
||||
borderRadius: 8,
|
||||
padding: '8px 10px',
|
||||
fontSize: 12,
|
||||
fontFamily: 'inherit',
|
||||
color: isActive ? '#a78bfa' : '#9ca3af',
|
||||
cursor: 'pointer',
|
||||
transition: 'all 0.15s',
|
||||
width: '100%',
|
||||
}}
|
||||
onMouseEnter={e => { if (!isActive) { e.currentTarget.style.color = '#e2e2e8'; e.currentTarget.style.background = 'rgba(255,255,255,0.04)'; } }}
|
||||
onMouseLeave={e => { if (!isActive) { e.currentTarget.style.color = '#9ca3af'; e.currentTarget.style.background = 'transparent'; } }}
|
||||
>
|
||||
<span style={{ color: isActive ? '#a78bfa' : '#6b7280', display: 'flex', flexShrink: 0 }}>{icon}</span>
|
||||
{label}
|
||||
</button>
|
||||
);
|
||||
})}
|
||||
</div>
|
||||
|
||||
{/* Content pane */}
|
||||
<div style={{ padding: '20px', overflowY: 'auto' }}>
|
||||
<TabContent active={active} nodeCount={nodeCount} edgeCount={edgeCount} />
|
||||
</div>
|
||||
</div>
|
||||
|
||||
{/* Footer */}
|
||||
<div style={{
|
||||
display: 'flex',
|
||||
alignItems: 'center',
|
||||
justifyContent: 'space-between',
|
||||
padding: '10px 20px',
|
||||
borderTop: '0.5px solid rgba(255,255,255,0.08)',
|
||||
background: 'rgba(255,255,255,0.01)',
|
||||
}}>
|
||||
<span style={{ fontSize: 11, color: '#4b5563' }}>GitNexus — open source codebase graph explorer</span>
|
||||
<a
|
||||
href="https://github.com/abhigyanpatwari/GitNexus"
|
||||
target="_blank"
|
||||
rel="noopener noreferrer"
|
||||
style={{ fontSize: 11, color: '#a78bfa', textDecoration: 'none' }}
|
||||
>
|
||||
Docs & GitHub ↗
|
||||
</a>
|
||||
</div>
|
||||
|
||||
</div>
|
||||
</div>
|
||||
);
|
||||
};
|
||||
@@ -281,7 +281,7 @@ export const SettingsPanel = ({ isOpen, onClose, onSettingsSaved, backendUrl, is
|
||||
|
||||
if (!isOpen) return null;
|
||||
|
||||
const providers: LLMProvider[] = ['openai', 'gemini', 'anthropic', 'azure-openai', 'ollama', 'openrouter'];
|
||||
const providers: LLMProvider[] = ['openai', 'gemini', 'anthropic', 'azure-openai', 'ollama', 'openrouter', 'minimax'];
|
||||
|
||||
|
||||
return (
|
||||
@@ -366,7 +366,7 @@ export const SettingsPanel = ({ isOpen, onClose, onSettingsSaved, backendUrl, is
|
||||
w-8 h-8 rounded-lg flex items-center justify-center text-lg
|
||||
${settings.activeProvider === provider ? 'bg-accent/20' : 'bg-surface'}
|
||||
`}>
|
||||
{provider === 'openai' ? '🤖' : provider === 'gemini' ? '💎' : provider === 'anthropic' ? '🧠' : provider === 'ollama' ? '🦙' : provider === 'openrouter' ? '🌐' : '☁️'}
|
||||
{provider === 'openai' ? '🤖' : provider === 'gemini' ? '💎' : provider === 'anthropic' ? '🧠' : provider === 'ollama' ? '🦙' : provider === 'openrouter' ? '🌐' : provider === 'minimax' ? '⚡' : '☁️'}
|
||||
</div>
|
||||
<span className="font-medium">{getProviderDisplayName(provider)}</span>
|
||||
</button>
|
||||
@@ -814,7 +814,64 @@ export const SettingsPanel = ({ isOpen, onClose, onSettingsSaved, backendUrl, is
|
||||
</div>
|
||||
)}
|
||||
|
||||
{/* MiniMax Settings */}
|
||||
{settings.activeProvider === 'minimax' && (
|
||||
<div className="space-y-4 animate-fade-in">
|
||||
<div className="space-y-2">
|
||||
<label className="flex items-center gap-2 text-sm font-medium text-text-secondary">
|
||||
<Key className="w-4 h-4" />
|
||||
API Key
|
||||
</label>
|
||||
<div className="relative">
|
||||
<input
|
||||
type={showApiKey['minimax'] ? 'text' : 'password'}
|
||||
value={settings.minimax?.apiKey ?? ''}
|
||||
onChange={e => setSettings(prev => ({
|
||||
...prev,
|
||||
minimax: { ...prev.minimax!, apiKey: e.target.value }
|
||||
}))}
|
||||
placeholder="Enter your MiniMax API key"
|
||||
className="w-full px-4 py-3 pr-12 bg-elevated border border-border-subtle rounded-xl text-text-primary placeholder:text-text-muted focus:border-accent focus:ring-2 focus:ring-accent/20 outline-none transition-all"
|
||||
/>
|
||||
<button
|
||||
type="button"
|
||||
onClick={() => toggleApiKeyVisibility('minimax')}
|
||||
className="absolute right-3 top-1/2 -translate-y-1/2 p-1 text-text-muted hover:text-text-primary transition-colors"
|
||||
>
|
||||
{showApiKey['minimax'] ? <EyeOff className="w-4 h-4" /> : <Eye className="w-4 h-4" />}
|
||||
</button>
|
||||
</div>
|
||||
<p className="text-xs text-text-muted">
|
||||
Get your API key from{' '}
|
||||
<a
|
||||
href="https://platform.minimax.io"
|
||||
target="_blank"
|
||||
rel="noopener noreferrer"
|
||||
className="text-accent hover:underline"
|
||||
>
|
||||
MiniMax Platform
|
||||
</a>
|
||||
</p>
|
||||
</div>
|
||||
|
||||
<div className="space-y-2">
|
||||
<label className="text-sm font-medium text-text-secondary">Model</label>
|
||||
<input
|
||||
type="text"
|
||||
value={settings.minimax?.model ?? 'MiniMax-M2.5'}
|
||||
onChange={e => setSettings(prev => ({
|
||||
...prev,
|
||||
minimax: { ...prev.minimax!, model: e.target.value }
|
||||
}))}
|
||||
placeholder="e.g., MiniMax-M2.5, MiniMax-M2.5-highspeed"
|
||||
className="w-full px-4 py-3 bg-elevated border border-border-subtle rounded-xl text-text-primary placeholder:text-text-muted focus:border-accent focus:ring-2 focus:ring-accent/20 outline-none transition-all font-mono text-sm"
|
||||
/>
|
||||
<p className="text-xs text-text-muted">
|
||||
Available models: MiniMax-M2.5 (default), MiniMax-M2.5-highspeed (faster)
|
||||
</p>
|
||||
</div>
|
||||
</div>
|
||||
)}
|
||||
|
||||
{/* Privacy Note */}
|
||||
<div className="p-4 bg-elevated/50 border border-border-subtle rounded-xl">
|
||||
|
||||
@@ -9,6 +9,7 @@ export enum SupportedLanguages {
|
||||
Go = 'go',
|
||||
Rust = 'rust',
|
||||
PHP = 'php',
|
||||
// Ruby = 'ruby',
|
||||
Ruby = 'ruby',
|
||||
Kotlin = 'kotlin',
|
||||
Swift = 'swift',
|
||||
}
|
||||
@@ -275,7 +275,7 @@ export const embedBatch = async (texts: string[]): Promise<Float32Array[]> => {
|
||||
};
|
||||
|
||||
/**
|
||||
* Convert Float32Array to regular number array (for KuzuDB storage)
|
||||
* Convert Float32Array to regular number array (for LadybugDB storage)
|
||||
*/
|
||||
export const embeddingToArray = (embedding: Float32Array): number[] => {
|
||||
return Array.from(embedding);
|
||||
|
||||
@@ -2,10 +2,10 @@
|
||||
* Embedding Pipeline Module
|
||||
*
|
||||
* Orchestrates the background embedding process:
|
||||
* 1. Query embeddable nodes from KuzuDB
|
||||
* 1. Query embeddable nodes from LadybugDB
|
||||
* 2. Generate text representations
|
||||
* 3. Batch embed using transformers.js
|
||||
* 4. Update KuzuDB with embeddings
|
||||
* 4. Update LadybugDB with embeddings
|
||||
* 5. Create vector index for semantic search
|
||||
*/
|
||||
|
||||
@@ -27,7 +27,7 @@ import {
|
||||
export type EmbeddingProgressCallback = (progress: EmbeddingProgress) => void;
|
||||
|
||||
/**
|
||||
* Query all embeddable nodes from KuzuDB
|
||||
* Query all embeddable nodes from LadybugDB
|
||||
* Uses table-specific queries (File has different schema than code elements)
|
||||
*/
|
||||
const queryEmbeddableNodes = async (
|
||||
@@ -102,9 +102,23 @@ const batchInsertEmbeddings = async (
|
||||
* Create the vector index for semantic search
|
||||
* Now indexes the separate CodeEmbedding table
|
||||
*/
|
||||
let vectorExtensionLoaded = false;
|
||||
|
||||
const createVectorIndex = async (
|
||||
executeQuery: (cypher: string) => Promise<any[]>
|
||||
): Promise<void> => {
|
||||
// LadybugDB v0.15+ requires explicit VECTOR extension loading (once per session)
|
||||
if (!vectorExtensionLoaded) {
|
||||
try {
|
||||
await executeQuery('INSTALL VECTOR');
|
||||
await executeQuery('LOAD EXTENSION VECTOR');
|
||||
vectorExtensionLoaded = true;
|
||||
} catch {
|
||||
// Extension may already be loaded — CREATE_VECTOR_INDEX will fail clearly if not
|
||||
vectorExtensionLoaded = true;
|
||||
}
|
||||
}
|
||||
|
||||
const cypher = `
|
||||
CALL CREATE_VECTOR_INDEX('CodeEmbedding', 'code_embedding_idx', 'embedding', metric := 'cosine')
|
||||
`;
|
||||
@@ -122,7 +136,7 @@ const createVectorIndex = async (
|
||||
/**
|
||||
* Run the embedding pipeline
|
||||
*
|
||||
* @param executeQuery - Function to execute Cypher queries against KuzuDB
|
||||
* @param executeQuery - Function to execute Cypher queries against LadybugDB
|
||||
* @param executeWithReusedStatement - Function to execute with reused prepared statement
|
||||
* @param onProgress - Callback for progress updates
|
||||
* @param config - Optional configuration override
|
||||
@@ -206,7 +220,7 @@ export const runEmbeddingPipeline = async (
|
||||
// Embed the batch
|
||||
const embeddings = await embedBatch(texts);
|
||||
|
||||
// Update KuzuDB with embeddings
|
||||
// Update LadybugDB with embeddings
|
||||
const updates = batch.map((node, i) => ({
|
||||
id: node.id,
|
||||
embedding: embeddingToArray(embeddings[i]),
|
||||
@@ -313,51 +327,64 @@ export const semanticSearch = async (
|
||||
return [];
|
||||
}
|
||||
|
||||
// Get metadata for each result by querying each node table
|
||||
const results: SemanticSearchResult[] = [];
|
||||
|
||||
// Group results by label for batched metadata queries
|
||||
const byLabel = new Map<string, Array<{ nodeId: string; distance: number }>>();
|
||||
for (const embRow of embResults) {
|
||||
const nodeId = embRow.nodeId ?? embRow[0];
|
||||
const distance = embRow.distance ?? embRow[1];
|
||||
|
||||
// Extract label from node ID (format: Label:path:name)
|
||||
const labelEndIdx = nodeId.indexOf(':');
|
||||
const label = labelEndIdx > 0 ? nodeId.substring(0, labelEndIdx) : 'Unknown';
|
||||
|
||||
// Query the specific table for this node
|
||||
// File nodes don't have startLine/endLine
|
||||
if (!byLabel.has(label)) byLabel.set(label, []);
|
||||
byLabel.get(label)!.push({ nodeId, distance });
|
||||
}
|
||||
|
||||
// Batch-fetch metadata per label
|
||||
const results: SemanticSearchResult[] = [];
|
||||
|
||||
for (const [label, items] of byLabel) {
|
||||
const idList = items.map(i => `'${i.nodeId.replace(/'/g, "''")}'`).join(', ');
|
||||
try {
|
||||
let nodeQuery: string;
|
||||
if (label === 'File') {
|
||||
nodeQuery = `
|
||||
MATCH (n:File {id: '${nodeId.replace(/'/g, "''")}'})
|
||||
RETURN n.name AS name, n.filePath AS filePath
|
||||
MATCH (n:File) WHERE n.id IN [${idList}]
|
||||
RETURN n.id AS id, n.name AS name, n.filePath AS filePath
|
||||
`;
|
||||
} else {
|
||||
nodeQuery = `
|
||||
MATCH (n:${label} {id: '${nodeId.replace(/'/g, "''")}'})
|
||||
RETURN n.name AS name, n.filePath AS filePath,
|
||||
MATCH (n:${label}) WHERE n.id IN [${idList}]
|
||||
RETURN n.id AS id, n.name AS name, n.filePath AS filePath,
|
||||
n.startLine AS startLine, n.endLine AS endLine
|
||||
`;
|
||||
}
|
||||
const nodeRows = await executeQuery(nodeQuery);
|
||||
if (nodeRows.length > 0) {
|
||||
const nodeRow = nodeRows[0];
|
||||
results.push({
|
||||
nodeId,
|
||||
name: nodeRow.name ?? nodeRow[0] ?? '',
|
||||
label,
|
||||
filePath: nodeRow.filePath ?? nodeRow[1] ?? '',
|
||||
distance,
|
||||
startLine: label !== 'File' ? (nodeRow.startLine ?? nodeRow[2]) : undefined,
|
||||
endLine: label !== 'File' ? (nodeRow.endLine ?? nodeRow[3]) : undefined,
|
||||
});
|
||||
const rowMap = new Map<string, any>();
|
||||
for (const row of nodeRows) {
|
||||
const id = row.id ?? row[0];
|
||||
rowMap.set(id, row);
|
||||
}
|
||||
for (const item of items) {
|
||||
const nodeRow = rowMap.get(item.nodeId);
|
||||
if (nodeRow) {
|
||||
results.push({
|
||||
nodeId: item.nodeId,
|
||||
name: nodeRow.name ?? nodeRow[1] ?? '',
|
||||
label,
|
||||
filePath: nodeRow.filePath ?? nodeRow[2] ?? '',
|
||||
distance: item.distance,
|
||||
startLine: label !== 'File' ? (nodeRow.startLine ?? nodeRow[3]) : undefined,
|
||||
endLine: label !== 'File' ? (nodeRow.endLine ?? nodeRow[4]) : undefined,
|
||||
});
|
||||
}
|
||||
}
|
||||
} catch {
|
||||
// Table might not exist, skip
|
||||
}
|
||||
}
|
||||
|
||||
// Re-sort by distance since batch queries may have mixed order
|
||||
results.sort((a, b) => a.distance - b.distance);
|
||||
|
||||
return results;
|
||||
};
|
||||
|
||||
|
||||
@@ -92,7 +92,7 @@ export interface SemanticSearchResult {
|
||||
}
|
||||
|
||||
/**
|
||||
* Node data for embedding (minimal structure from KuzuDB query)
|
||||
* Node data for embedding (minimal structure from LadybugDB query)
|
||||
*/
|
||||
export interface EmbeddableNode {
|
||||
id: string;
|
||||
|
||||
@@ -54,6 +54,7 @@ export type RelationshipType =
|
||||
| 'DECORATES'
|
||||
| 'IMPLEMENTS'
|
||||
| 'EXTENDS'
|
||||
| 'HAS_METHOD'
|
||||
| 'MEMBER_OF'
|
||||
| 'STEP_IN_PROCESS'
|
||||
|
||||
|
||||
@@ -6,6 +6,7 @@ import { loadParser, loadLanguage } from '../tree-sitter/parser-loader';
|
||||
import { LANGUAGE_QUERIES } from './tree-sitter-queries';
|
||||
import { generateId } from '../../lib/utils';
|
||||
import { getLanguageFromFilename } from './utils';
|
||||
import { callRouters } from './call-routing';
|
||||
|
||||
/**
|
||||
* Node types that represent function/method definitions across languages.
|
||||
@@ -35,6 +36,9 @@ const FUNCTION_NODE_TYPES = new Set([
|
||||
// Rust
|
||||
'function_item',
|
||||
'impl_item', // Methods inside impl blocks
|
||||
// Ruby
|
||||
'method', // def foo
|
||||
'singleton_method', // def self.foo
|
||||
]);
|
||||
|
||||
/**
|
||||
@@ -92,6 +96,18 @@ const findEnclosingFunction = (
|
||||
current.children?.find((c: any) => c.type === 'identifier');
|
||||
funcName = nameNode?.text;
|
||||
label = 'Method'; // Treat constructors as methods for process detection
|
||||
} else if (current.type === 'method') {
|
||||
// Ruby instance method: def foo
|
||||
const nameNode = current.childForFieldName?.('name') ||
|
||||
current.children?.find((c: any) => c.type === 'identifier');
|
||||
funcName = nameNode?.text;
|
||||
label = 'Method';
|
||||
} else if (current.type === 'singleton_method') {
|
||||
// Ruby class method: def self.foo
|
||||
const nameNode = current.childForFieldName?.('name') ||
|
||||
current.children?.find((c: any) => c.type === 'identifier');
|
||||
funcName = nameNode?.text;
|
||||
label = 'Method';
|
||||
} else if (current.type === 'arrow_function' || current.type === 'function_expression') {
|
||||
// Arrow/expression: const foo = () => {} - check parent variable declarator
|
||||
const parent = current.parent;
|
||||
@@ -126,6 +142,47 @@ const findEnclosingFunction = (
|
||||
return null; // Top-level call (not inside any function)
|
||||
};
|
||||
|
||||
/** AST node types that represent a class-like container */
|
||||
const CLASS_CONTAINER_TYPES = new Set([
|
||||
'class_declaration', 'abstract_class_declaration',
|
||||
'interface_declaration', 'struct_declaration', 'record_declaration',
|
||||
'class_specifier', 'struct_specifier',
|
||||
'impl_item', 'trait_item',
|
||||
'class_definition',
|
||||
'trait_declaration',
|
||||
'protocol_declaration',
|
||||
'class', 'module', // Ruby
|
||||
]);
|
||||
|
||||
const CONTAINER_TYPE_TO_LABEL: Record<string, string> = {
|
||||
class_declaration: 'Class', abstract_class_declaration: 'Class',
|
||||
interface_declaration: 'Interface',
|
||||
struct_declaration: 'Struct', struct_specifier: 'Struct',
|
||||
class_specifier: 'Class', class_definition: 'Class',
|
||||
impl_item: 'Impl', trait_item: 'Trait', trait_declaration: 'Trait',
|
||||
record_declaration: 'Record', protocol_declaration: 'Interface',
|
||||
class: 'Class', module: 'Module',
|
||||
};
|
||||
|
||||
/** Walk up AST to find enclosing class/struct/interface, return its generateId or null. */
|
||||
const findEnclosingClassId = (node: any, filePath: string): string | null => {
|
||||
let current = node.parent;
|
||||
while (current) {
|
||||
if (CLASS_CONTAINER_TYPES.has(current.type)) {
|
||||
const nameNode = current.childForFieldName?.('name')
|
||||
?? current.children?.find((c: any) =>
|
||||
c.type === 'type_identifier' || c.type === 'identifier' || c.type === 'name' || c.type === 'constant'
|
||||
);
|
||||
if (nameNode) {
|
||||
const label = CONTAINER_TYPE_TO_LABEL[current.type] || 'Class';
|
||||
return generateId(label, `${filePath}:${nameNode.text}`);
|
||||
}
|
||||
}
|
||||
current = current.parent;
|
||||
}
|
||||
return null;
|
||||
};
|
||||
|
||||
export const processCalls = async (
|
||||
graph: KnowledgeGraph,
|
||||
files: { path: string; content: string }[],
|
||||
@@ -171,6 +228,8 @@ export const processCalls = async (
|
||||
continue;
|
||||
}
|
||||
|
||||
const callRouter = callRouters[language];
|
||||
|
||||
// 3. Process each call match
|
||||
matches.forEach(match => {
|
||||
const captureMap: Record<string, any> = {};
|
||||
@@ -184,6 +243,68 @@ export const processCalls = async (
|
||||
|
||||
const calledName = nameNode.text;
|
||||
|
||||
// Dispatch: route language-specific calls (heritage, properties, imports)
|
||||
const routed = callRouter(calledName, captureMap['call']);
|
||||
if (routed) {
|
||||
switch (routed.kind) {
|
||||
case 'skip':
|
||||
case 'import': // handled by import-processor
|
||||
return;
|
||||
|
||||
case 'heritage':
|
||||
for (const item of routed.items) {
|
||||
const childId = symbolTable.lookupExact(file.path, item.enclosingClass) ||
|
||||
symbolTable.lookupFuzzy(item.enclosingClass)[0]?.nodeId ||
|
||||
generateId('Class', `${file.path}:${item.enclosingClass}`);
|
||||
const parentId = symbolTable.lookupFuzzy(item.mixinName)[0]?.nodeId ||
|
||||
generateId('Module', `${item.mixinName}`);
|
||||
if (childId && parentId) {
|
||||
const relId = generateId('IMPLEMENTS', `${childId}->${parentId}:${item.heritageKind}`);
|
||||
graph.addRelationship({
|
||||
id: relId, sourceId: childId, targetId: parentId,
|
||||
type: 'IMPLEMENTS', confidence: 1.0, reason: item.heritageKind,
|
||||
});
|
||||
}
|
||||
}
|
||||
return;
|
||||
|
||||
case 'properties': {
|
||||
const fileId = generateId('File', file.path);
|
||||
const propEnclosingClassId = findEnclosingClassId(captureMap['call'], file.path);
|
||||
for (const item of routed.items) {
|
||||
const nodeId = generateId('Property', `${file.path}:${item.propName}`);
|
||||
graph.addNode({
|
||||
id: nodeId,
|
||||
label: 'Property' as any, // TODO: add 'Property' to graph node label union
|
||||
properties: {
|
||||
name: item.propName, filePath: file.path,
|
||||
startLine: item.startLine, endLine: item.endLine,
|
||||
language, isExported: true,
|
||||
description: item.accessorType,
|
||||
},
|
||||
});
|
||||
symbolTable.add(file.path, item.propName, nodeId, 'Property');
|
||||
const relId = generateId('DEFINES', `${fileId}->${nodeId}`);
|
||||
graph.addRelationship({
|
||||
id: relId, sourceId: fileId, targetId: nodeId,
|
||||
type: 'DEFINES', confidence: 1.0, reason: '',
|
||||
});
|
||||
if (propEnclosingClassId) {
|
||||
graph.addRelationship({
|
||||
id: generateId('HAS_METHOD', `${propEnclosingClassId}->${nodeId}`),
|
||||
sourceId: propEnclosingClassId, targetId: nodeId,
|
||||
type: 'HAS_METHOD', confidence: 1.0, reason: '',
|
||||
});
|
||||
}
|
||||
}
|
||||
return;
|
||||
}
|
||||
|
||||
case 'call':
|
||||
break; // fall through to normal call processing below
|
||||
}
|
||||
}
|
||||
|
||||
// Skip common built-ins and noise
|
||||
if (isBuiltInOrNoise(calledName)) return;
|
||||
|
||||
@@ -200,10 +321,10 @@ export const processCalls = async (
|
||||
// 5. Find the enclosing function (caller)
|
||||
const callNode = captureMap['call'];
|
||||
const enclosingFuncId = findEnclosingFunction(callNode, file.path, symbolTable);
|
||||
|
||||
|
||||
// Use enclosing function as source, fallback to file for top-level calls
|
||||
const sourceId = enclosingFuncId || generateId('File', file.path);
|
||||
|
||||
|
||||
const relId = generateId('CALLS', `${sourceId}:${calledName}->${resolved.nodeId}`);
|
||||
|
||||
graph.addRelationship({
|
||||
@@ -711,37 +832,72 @@ const resolveCallTarget = (
|
||||
* Filter out common built-in functions and noise
|
||||
* that shouldn't be tracked as calls
|
||||
*/
|
||||
const isBuiltInOrNoise = (name: string): boolean => {
|
||||
const builtIns = new Set([
|
||||
// JavaScript/TypeScript built-ins
|
||||
'console', 'log', 'warn', 'error', 'info', 'debug',
|
||||
'setTimeout', 'setInterval', 'clearTimeout', 'clearInterval',
|
||||
'parseInt', 'parseFloat', 'isNaN', 'isFinite',
|
||||
'encodeURI', 'decodeURI', 'encodeURIComponent', 'decodeURIComponent',
|
||||
'JSON', 'parse', 'stringify',
|
||||
'Object', 'Array', 'String', 'Number', 'Boolean', 'Symbol', 'BigInt',
|
||||
'Map', 'Set', 'WeakMap', 'WeakSet',
|
||||
'Promise', 'resolve', 'reject', 'then', 'catch', 'finally',
|
||||
'Math', 'Date', 'RegExp', 'Error',
|
||||
'require', 'import', 'export',
|
||||
'fetch', 'Response', 'Request',
|
||||
// React hooks and common functions
|
||||
'useState', 'useEffect', 'useCallback', 'useMemo', 'useRef', 'useContext',
|
||||
'useReducer', 'useLayoutEffect', 'useImperativeHandle', 'useDebugValue',
|
||||
'createElement', 'createContext', 'createRef', 'forwardRef', 'memo', 'lazy',
|
||||
// Common array/object methods
|
||||
'map', 'filter', 'reduce', 'forEach', 'find', 'findIndex', 'some', 'every',
|
||||
'includes', 'indexOf', 'slice', 'splice', 'concat', 'join', 'split',
|
||||
'push', 'pop', 'shift', 'unshift', 'sort', 'reverse',
|
||||
'keys', 'values', 'entries', 'assign', 'freeze', 'seal',
|
||||
'hasOwnProperty', 'toString', 'valueOf',
|
||||
// Python built-ins
|
||||
'print', 'len', 'range', 'str', 'int', 'float', 'list', 'dict', 'set', 'tuple',
|
||||
'open', 'read', 'write', 'close', 'append', 'extend', 'update',
|
||||
'super', 'type', 'isinstance', 'issubclass', 'getattr', 'setattr', 'hasattr',
|
||||
'enumerate', 'zip', 'sorted', 'reversed', 'min', 'max', 'sum', 'abs',
|
||||
]);
|
||||
/** Pre-built set (module-level singleton) to avoid re-creating per call */
|
||||
const BUILT_IN_NAMES = new Set([
|
||||
// JavaScript/TypeScript built-ins
|
||||
'console', 'log', 'warn', 'error', 'info', 'debug',
|
||||
'setTimeout', 'setInterval', 'clearTimeout', 'clearInterval',
|
||||
'parseInt', 'parseFloat', 'isNaN', 'isFinite',
|
||||
'encodeURI', 'decodeURI', 'encodeURIComponent', 'decodeURIComponent',
|
||||
'JSON', 'parse', 'stringify',
|
||||
'Object', 'Array', 'String', 'Number', 'Boolean', 'Symbol', 'BigInt',
|
||||
'Map', 'Set', 'WeakMap', 'WeakSet',
|
||||
'Promise', 'resolve', 'reject', 'then', 'catch', 'finally',
|
||||
'Math', 'Date', 'RegExp', 'Error',
|
||||
'require', 'import', 'export',
|
||||
'fetch', 'Response', 'Request',
|
||||
// React hooks and common functions
|
||||
'useState', 'useEffect', 'useCallback', 'useMemo', 'useRef', 'useContext',
|
||||
'useReducer', 'useLayoutEffect', 'useImperativeHandle', 'useDebugValue',
|
||||
'createElement', 'createContext', 'createRef', 'forwardRef', 'memo', 'lazy',
|
||||
// Common array/object methods
|
||||
'map', 'filter', 'reduce', 'forEach', 'find', 'findIndex', 'some', 'every',
|
||||
'includes', 'indexOf', 'slice', 'splice', 'concat', 'join', 'split',
|
||||
'push', 'pop', 'shift', 'unshift', 'sort', 'reverse',
|
||||
'keys', 'values', 'entries', 'assign', 'freeze', 'seal',
|
||||
'hasOwnProperty', 'toString', 'valueOf',
|
||||
// Python built-ins
|
||||
'print', 'len', 'range', 'str', 'int', 'float', 'list', 'dict', 'set', 'tuple',
|
||||
'open', 'read', 'write', 'close', 'append', 'extend', 'update',
|
||||
'super', 'type', 'isinstance', 'issubclass', 'getattr', 'setattr', 'hasattr',
|
||||
'enumerate', 'zip', 'sorted', 'reversed', 'min', 'max', 'sum', 'abs',
|
||||
// C/C++ standard library and common kernel helpers
|
||||
'printf', 'fprintf', 'sprintf', 'snprintf', 'vprintf', 'vfprintf', 'vsprintf', 'vsnprintf',
|
||||
'scanf', 'fscanf', 'sscanf',
|
||||
'malloc', 'calloc', 'realloc', 'free', 'memcpy', 'memmove', 'memset', 'memcmp',
|
||||
'strlen', 'strcpy', 'strncpy', 'strcat', 'strncat', 'strcmp', 'strncmp', 'strstr', 'strchr', 'strrchr',
|
||||
'atoi', 'atol', 'atof', 'strtol', 'strtoul', 'strtoll', 'strtoull', 'strtod',
|
||||
'sizeof', 'offsetof', 'typeof',
|
||||
'assert', 'abort', 'exit', '_exit',
|
||||
'fopen', 'fclose', 'fread', 'fwrite', 'fseek', 'ftell', 'rewind', 'fflush', 'fgets', 'fputs',
|
||||
// Linux kernel common macros/helpers (not real call targets)
|
||||
'likely', 'unlikely', 'BUG', 'BUG_ON', 'WARN', 'WARN_ON', 'WARN_ONCE',
|
||||
'IS_ERR', 'PTR_ERR', 'ERR_PTR', 'IS_ERR_OR_NULL',
|
||||
'ARRAY_SIZE', 'container_of', 'list_for_each_entry', 'list_for_each_entry_safe',
|
||||
'min', 'max', 'clamp', 'abs', 'swap',
|
||||
'pr_info', 'pr_warn', 'pr_err', 'pr_debug', 'pr_notice', 'pr_crit', 'pr_emerg',
|
||||
'printk', 'dev_info', 'dev_warn', 'dev_err', 'dev_dbg',
|
||||
'GFP_KERNEL', 'GFP_ATOMIC',
|
||||
'spin_lock', 'spin_unlock', 'spin_lock_irqsave', 'spin_unlock_irqrestore',
|
||||
'mutex_lock', 'mutex_unlock', 'mutex_init',
|
||||
'kfree', 'kmalloc', 'kzalloc', 'kcalloc', 'krealloc', 'kvmalloc', 'kvfree',
|
||||
'get', 'put',
|
||||
// Ruby built-ins and Kernel methods
|
||||
'puts', 'print', 'p', 'pp', 'warn', 'raise', 'fail',
|
||||
'require', 'require_relative', 'load', 'autoload',
|
||||
'include', 'extend', 'prepend',
|
||||
'attr_accessor', 'attr_reader', 'attr_writer',
|
||||
'public', 'private', 'protected', 'module_function',
|
||||
'lambda', 'proc', 'block_given?',
|
||||
'nil?', 'is_a?', 'kind_of?', 'instance_of?', 'respond_to?',
|
||||
'freeze', 'frozen?', 'dup', 'clone', 'tap', 'then', 'yield_self',
|
||||
// Ruby enumerables
|
||||
'each', 'map', 'select', 'reject', 'find', 'detect', 'collect',
|
||||
'inject', 'reduce', 'flat_map', 'each_with_object', 'each_with_index',
|
||||
'any?', 'all?', 'none?', 'count', 'first', 'last',
|
||||
'sort', 'sort_by', 'min', 'max', 'min_by', 'max_by',
|
||||
'group_by', 'partition', 'zip', 'compact', 'flatten', 'uniq',
|
||||
]);
|
||||
|
||||
return builtIns.has(name);
|
||||
};
|
||||
const isBuiltInOrNoise = (name: string): boolean => BUILT_IN_NAMES.has(name);
|
||||
|
||||
|
||||
@@ -0,0 +1,149 @@
|
||||
/**
|
||||
* Shared Ruby call routing logic.
|
||||
*
|
||||
* Ruby expresses imports, heritage (mixins), and property definitions as
|
||||
* method calls rather than syntax-level constructs. This module provides a
|
||||
* routing function used by the CLI call-processor, CLI parse-worker, and
|
||||
* the web call-processor so that the classification logic lives in one place.
|
||||
*
|
||||
* NOTE: This file is intentionally duplicated in gitnexus-web/ because the
|
||||
* two packages have separate build targets (Node native vs WASM/browser).
|
||||
* Keep both copies in sync until a shared package is introduced.
|
||||
*/
|
||||
|
||||
import { SupportedLanguages } from '../../config/supported-languages';
|
||||
|
||||
// ── Call routing dispatch table ─────────────────────────────────────────────
|
||||
|
||||
/** null = this call was not routed; fall through to default call handling */
|
||||
export type CallRoutingResult = RubyCallRouting | null;
|
||||
|
||||
export type CallRouter = (
|
||||
calledName: string,
|
||||
callNode: any,
|
||||
) => CallRoutingResult;
|
||||
|
||||
/** No-op router: returns null for every call (passthrough to normal processing) */
|
||||
const noRouting: CallRouter = () => null;
|
||||
|
||||
/** Per-language call routing. noRouting = no special routing (normal call processing) */
|
||||
export const callRouters = {
|
||||
[SupportedLanguages.JavaScript]: noRouting,
|
||||
[SupportedLanguages.TypeScript]: noRouting,
|
||||
[SupportedLanguages.Python]: noRouting,
|
||||
[SupportedLanguages.Java]: noRouting,
|
||||
[SupportedLanguages.Go]: noRouting,
|
||||
[SupportedLanguages.Rust]: noRouting,
|
||||
[SupportedLanguages.CSharp]: noRouting,
|
||||
[SupportedLanguages.PHP]: noRouting,
|
||||
[SupportedLanguages.Swift]: noRouting,
|
||||
[SupportedLanguages.CPlusPlus]: noRouting,
|
||||
[SupportedLanguages.C]: noRouting,
|
||||
[SupportedLanguages.Ruby]: routeRubyCall,
|
||||
[SupportedLanguages.Kotlin]: noRouting,
|
||||
} satisfies Record<SupportedLanguages, CallRouter>;
|
||||
|
||||
// ── Result types ────────────────────────────────────────────────────────────
|
||||
|
||||
export type RubyCallRouting =
|
||||
| { kind: 'import'; importPath: string; isRelative: boolean }
|
||||
| { kind: 'heritage'; items: RubyHeritageItem[] }
|
||||
| { kind: 'properties'; items: RubyPropertyItem[] }
|
||||
| { kind: 'call' }
|
||||
| { kind: 'skip' };
|
||||
|
||||
export interface RubyHeritageItem {
|
||||
enclosingClass: string;
|
||||
mixinName: string;
|
||||
heritageKind: 'include' | 'extend' | 'prepend';
|
||||
}
|
||||
|
||||
export type RubyAccessorType = 'attr_accessor' | 'attr_reader' | 'attr_writer';
|
||||
|
||||
export interface RubyPropertyItem {
|
||||
propName: string;
|
||||
accessorType: RubyAccessorType;
|
||||
startLine: number;
|
||||
endLine: number;
|
||||
}
|
||||
|
||||
// ── Pre-allocated singletons for common return values ────────────────────────
|
||||
const CALL_RESULT: RubyCallRouting = { kind: 'call' };
|
||||
const SKIP_RESULT: RubyCallRouting = { kind: 'skip' };
|
||||
|
||||
/** Max depth for parent-walking loops to prevent pathological AST traversals */
|
||||
const MAX_PARENT_DEPTH = 50;
|
||||
|
||||
// ── Routing function ────────────────────────────────────────────────────────
|
||||
|
||||
/**
|
||||
* Classify a Ruby call node and extract its semantic payload.
|
||||
*
|
||||
* @param calledName - The method name (e.g. 'require', 'include', 'attr_accessor')
|
||||
* @param callNode - The tree-sitter `call` AST node
|
||||
* @returns A discriminated union describing the call's semantic role
|
||||
*/
|
||||
export function routeRubyCall(calledName: string, callNode: any): RubyCallRouting {
|
||||
// ── require / require_relative → import ─────────────────────────────────
|
||||
if (calledName === 'require' || calledName === 'require_relative') {
|
||||
const argList = callNode.childForFieldName?.('arguments');
|
||||
const stringNode = argList?.children?.find((c: any) => c.type === 'string');
|
||||
const contentNode = stringNode?.children?.find((c: any) => c.type === 'string_content');
|
||||
if (!contentNode) return SKIP_RESULT;
|
||||
|
||||
let importPath: string = contentNode.text;
|
||||
// Validate: reject null bytes, control chars, excessively long paths
|
||||
if (!importPath || importPath.length > 1024 || /[\x00-\x1f]/.test(importPath)) {
|
||||
return SKIP_RESULT;
|
||||
}
|
||||
const isRelative = calledName === 'require_relative';
|
||||
if (isRelative && !importPath.startsWith('.')) {
|
||||
importPath = './' + importPath;
|
||||
}
|
||||
return { kind: 'import', importPath, isRelative };
|
||||
}
|
||||
|
||||
// ── include / extend / prepend → heritage (mixin) ──────────────────────
|
||||
if (calledName === 'include' || calledName === 'extend' || calledName === 'prepend') {
|
||||
let enclosingClass: string | null = null;
|
||||
let current = callNode.parent;
|
||||
let depth = 0;
|
||||
while (current && ++depth <= MAX_PARENT_DEPTH) {
|
||||
if (current.type === 'class' || current.type === 'module') {
|
||||
const nameNode = current.childForFieldName?.('name');
|
||||
if (nameNode) { enclosingClass = nameNode.text; break; }
|
||||
}
|
||||
current = current.parent;
|
||||
}
|
||||
if (!enclosingClass) return SKIP_RESULT;
|
||||
|
||||
const items: RubyHeritageItem[] = [];
|
||||
const argList = callNode.childForFieldName?.('arguments');
|
||||
for (const arg of (argList?.children ?? [])) {
|
||||
if (arg.type === 'constant' || arg.type === 'scope_resolution') {
|
||||
items.push({ enclosingClass, mixinName: arg.text, heritageKind: calledName as 'include' | 'extend' | 'prepend' });
|
||||
}
|
||||
}
|
||||
return items.length > 0 ? { kind: 'heritage', items } : SKIP_RESULT;
|
||||
}
|
||||
|
||||
// ── attr_accessor / attr_reader / attr_writer → property definitions ───
|
||||
if (calledName === 'attr_accessor' || calledName === 'attr_reader' || calledName === 'attr_writer') {
|
||||
const items: RubyPropertyItem[] = [];
|
||||
const argList = callNode.childForFieldName?.('arguments');
|
||||
for (const arg of (argList?.children ?? [])) {
|
||||
if (arg.type === 'simple_symbol') {
|
||||
items.push({
|
||||
propName: arg.text.startsWith(':') ? arg.text.slice(1) : arg.text,
|
||||
accessorType: calledName as RubyAccessorType,
|
||||
startLine: arg.startPosition.row,
|
||||
endLine: arg.endPosition.row,
|
||||
});
|
||||
}
|
||||
}
|
||||
return items.length > 0 ? { kind: 'properties', items } : SKIP_RESULT;
|
||||
}
|
||||
|
||||
// ── Everything else → regular call ─────────────────────────────────────
|
||||
return CALL_RESULT;
|
||||
}
|
||||
@@ -13,7 +13,7 @@
|
||||
import { detectFrameworkFromPath } from './framework-detection';
|
||||
|
||||
// ============================================================================
|
||||
// NAME PATTERNS - All 9 supported languages
|
||||
// NAME PATTERNS - All 11 supported languages
|
||||
// ============================================================================
|
||||
|
||||
/**
|
||||
@@ -143,6 +143,13 @@ const ENTRY_POINT_PATTERNS: Record<string, RegExp[]> = {
|
||||
/^save$/, // Repository::save()
|
||||
/^delete$/, // Repository::delete()
|
||||
],
|
||||
|
||||
// Ruby
|
||||
'ruby': [
|
||||
/^call$/, // Service objects (MyService.call)
|
||||
/^perform$/, // Background jobs (Sidekiq, ActiveJob)
|
||||
/^execute$/, // Command pattern
|
||||
],
|
||||
};
|
||||
|
||||
// ============================================================================
|
||||
@@ -302,7 +309,12 @@ export function isTestFile(filePath: string): boolean {
|
||||
p.endsWith('test.php') ||
|
||||
p.endsWith('spec.php') ||
|
||||
p.includes('/tests/feature/') ||
|
||||
p.includes('/tests/unit/')
|
||||
p.includes('/tests/unit/') ||
|
||||
// Ruby test patterns
|
||||
p.endsWith('_spec.rb') ||
|
||||
p.endsWith('_test.rb') ||
|
||||
p.includes('/spec/') ||
|
||||
p.includes('/test/fixtures/')
|
||||
);
|
||||
}
|
||||
|
||||
|
||||
@@ -257,6 +257,17 @@ export function detectFrameworkFromPath(filePath: string): FrameworkHint | null
|
||||
return { framework: 'laravel', entryPointMultiplier: 1.5, reason: 'laravel-repository' };
|
||||
}
|
||||
|
||||
// ========== RUBY ==========
|
||||
|
||||
// Ruby: bin/ or exe/ (CLI entry points)
|
||||
if ((p.includes('/bin/') || p.includes('/exe/')) && p.endsWith('.rb')) {
|
||||
return { framework: 'ruby', entryPointMultiplier: 2.5, reason: 'ruby-executable' };
|
||||
}
|
||||
|
||||
// Ruby: Rakefile or *.rake (task definitions)
|
||||
if (p.endsWith('/rakefile') || p.endsWith('.rake')) {
|
||||
return { framework: 'ruby', entryPointMultiplier: 1.5, reason: 'ruby-rake' };
|
||||
}
|
||||
// ========== SWIFT / iOS ==========
|
||||
|
||||
// iOS App entry points (highest priority)
|
||||
|
||||
@@ -4,6 +4,7 @@ import { loadParser, loadLanguage } from '../tree-sitter/parser-loader';
|
||||
import { LANGUAGE_QUERIES } from './tree-sitter-queries';
|
||||
import { generateId } from '../../lib/utils';
|
||||
import { getLanguageFromFilename } from './utils';
|
||||
import { callRouters } from './call-routing';
|
||||
|
||||
// Type: Map<FilePath, Set<ResolvedFilePath>>
|
||||
// Stores all files that a given file imports from
|
||||
@@ -53,7 +54,9 @@ const resolveImportPath = (
|
||||
// Go
|
||||
'.go',
|
||||
// Rust
|
||||
'.rs', '/mod.rs'
|
||||
'.rs', '/mod.rs',
|
||||
// Ruby
|
||||
'.rb', '.rake',
|
||||
];
|
||||
|
||||
if (importPath.startsWith('.')) {
|
||||
@@ -220,6 +223,35 @@ export const processImports = async (
|
||||
importMap.get(file.path)!.add(resolvedPath);
|
||||
}
|
||||
}
|
||||
|
||||
// ---- Language-specific call-as-import routing (Ruby require, etc.) ----
|
||||
if (captureMap['call']) {
|
||||
const callNameNode = captureMap['call.name'];
|
||||
if (callNameNode) {
|
||||
const callRouter = callRouters[language];
|
||||
const routed = callRouter(callNameNode.text, captureMap['call']);
|
||||
if (routed && routed.kind === 'import') {
|
||||
totalImportsFound++;
|
||||
const resolvedPath = resolveImportPath(
|
||||
file.path, routed.importPath, allFilePaths, allFileList, resolveCache
|
||||
);
|
||||
if (resolvedPath) {
|
||||
const sourceId = generateId('File', file.path);
|
||||
const targetId = generateId('File', resolvedPath);
|
||||
const relId = generateId('IMPORTS', `${file.path}->${resolvedPath}`);
|
||||
totalImportsResolved++;
|
||||
graph.addRelationship({
|
||||
id: relId, sourceId, targetId,
|
||||
type: 'IMPORTS', confidence: 1.0, reason: '',
|
||||
});
|
||||
if (!importMap.has(file.path)) {
|
||||
importMap.set(file.path, new Set());
|
||||
}
|
||||
importMap.get(file.path)!.add(resolvedPath);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
});
|
||||
|
||||
// If re-parsed just for this, delete the tree to save memory
|
||||
|
||||
@@ -14,7 +14,7 @@ export type FileProgressCallback = (current: number, total: number, filePath: st
|
||||
|
||||
/**
|
||||
* Check if a symbol (function, class, etc.) is exported/public
|
||||
* Handles all 9 supported languages with explicit logic
|
||||
* Handles all 11 supported languages with explicit logic
|
||||
*
|
||||
* @param node - The AST node for the symbol name
|
||||
* @param name - The symbol name
|
||||
@@ -104,7 +104,11 @@ const isNodeExported = (node: any, name: string, language: string): boolean => {
|
||||
case 'c':
|
||||
case 'cpp':
|
||||
return false;
|
||||
|
||||
|
||||
// Ruby: All top-level definitions are public by default
|
||||
case 'ruby':
|
||||
return true;
|
||||
|
||||
default:
|
||||
return false;
|
||||
}
|
||||
|
||||
@@ -396,6 +396,40 @@ export const PHP_QUERIES = `
|
||||
[(name) (qualified_name)] @heritage.trait))) @heritage
|
||||
`;
|
||||
|
||||
// Ruby queries - works with tree-sitter-ruby
|
||||
// NOTE: Ruby uses `call` for require, include, extend, prepend, attr_* etc.
|
||||
// These are all captured as @call and routed in JS post-processing:
|
||||
// - require/require_relative → import extraction
|
||||
// - include/extend/prepend → heritage (mixin) extraction
|
||||
// - attr_accessor/attr_reader/attr_writer → property definition extraction
|
||||
// - everything else → regular call extraction
|
||||
export const RUBY_QUERIES = `
|
||||
; ── Modules ──────────────────────────────────────────────────────────────────
|
||||
(module
|
||||
name: (constant) @name) @definition.module
|
||||
|
||||
; ── Classes ──────────────────────────────────────────────────────────────────
|
||||
(class
|
||||
name: (constant) @name) @definition.class
|
||||
|
||||
; ── Instance methods ─────────────────────────────────────────────────────────
|
||||
(method
|
||||
name: (identifier) @name) @definition.method
|
||||
|
||||
; ── Singleton (class-level) methods ──────────────────────────────────────────
|
||||
(singleton_method
|
||||
name: (identifier) @name) @definition.function
|
||||
|
||||
; ── All calls (require, include, attr_*, and regular calls routed in JS) ─────
|
||||
(call
|
||||
method: (identifier) @call.name) @call
|
||||
|
||||
; ── Heritage: class < SuperClass ─────────────────────────────────────────────
|
||||
(class
|
||||
name: (constant) @heritage.class
|
||||
superclass: (superclass
|
||||
(constant) @heritage.extends)) @heritage`;
|
||||
|
||||
// Swift queries - works with tree-sitter-swift
|
||||
export const SWIFT_QUERIES = `
|
||||
; Classes
|
||||
@@ -460,6 +494,8 @@ export const LANGUAGE_QUERIES: Record<SupportedLanguages, string> = {
|
||||
[SupportedLanguages.CSharp]: CSHARP_QUERIES,
|
||||
[SupportedLanguages.Rust]: RUST_QUERIES,
|
||||
[SupportedLanguages.PHP]: PHP_QUERIES,
|
||||
[SupportedLanguages.Ruby]: RUBY_QUERIES,
|
||||
[SupportedLanguages.Kotlin]: '', // Kotlin WASM parser not yet available for web
|
||||
[SupportedLanguages.Swift]: SWIFT_QUERIES,
|
||||
};
|
||||
|
||||
@@ -1,5 +1,8 @@
|
||||
import { SupportedLanguages } from '../../config/supported-languages';
|
||||
|
||||
/** Ruby extensionless filenames recognised as Ruby source */
|
||||
const RUBY_EXTENSIONLESS_FILES = new Set(['Rakefile', 'Gemfile', 'Guardfile', 'Vagrantfile', 'Brewfile']);
|
||||
|
||||
/**
|
||||
* Map file extension to SupportedLanguage enum
|
||||
*/
|
||||
@@ -31,6 +34,15 @@ export const getLanguageFromFilename = (filename: string): SupportedLanguages |
|
||||
filename.endsWith('.php5') || filename.endsWith('.php8')) {
|
||||
return SupportedLanguages.PHP;
|
||||
}
|
||||
// Ruby (extensions)
|
||||
if (filename.endsWith('.rb') || filename.endsWith('.rake') || filename.endsWith('.gemspec')) {
|
||||
return SupportedLanguages.Ruby;
|
||||
}
|
||||
// Ruby (extensionless files)
|
||||
const basename = filename.split('/').pop() || filename;
|
||||
if (RUBY_EXTENSIONLESS_FILES.has(basename)) {
|
||||
return SupportedLanguages.Ruby;
|
||||
}
|
||||
// Swift
|
||||
if (filename.endsWith('.swift')) return SupportedLanguages.Swift;
|
||||
return null;
|
||||
|
||||
+5
-5
@@ -1,5 +1,5 @@
|
||||
/**
|
||||
* CSV Generator for KuzuDB Hybrid Schema
|
||||
* CSV Generator for LadybugDB Hybrid Schema
|
||||
*
|
||||
* Generates separate CSV files for each node table and one relation CSV.
|
||||
* This enables efficient bulk loading via COPY FROM for hybrid schema.
|
||||
@@ -18,10 +18,10 @@ import { NODE_TABLES, NodeTableName } from './schema';
|
||||
// ============================================================================
|
||||
|
||||
/**
|
||||
* Sanitize string to ensure valid UTF-8 and safe CSV content for KuzuDB
|
||||
* Sanitize string to ensure valid UTF-8 and safe CSV content for LadybugDB
|
||||
* Removes or replaces invalid characters that would break CSV parsing.
|
||||
*
|
||||
* Critical: KuzuDB's CSV parser can misinterpret \r\n inside quoted fields.
|
||||
* Critical: LadybugDB's CSV parser can misinterpret \r\n inside quoted fields.
|
||||
* We normalize all line endings to \n only.
|
||||
*/
|
||||
const sanitizeUTF8 = (str: string): string => {
|
||||
@@ -213,7 +213,7 @@ const generateCommunityCSV = (nodes: GraphNode[]): string => {
|
||||
for (const node of nodes) {
|
||||
if (node.label !== 'Community') continue;
|
||||
|
||||
// Handle keywords array - convert to KuzuDB array format
|
||||
// Handle keywords array - convert to LadybugDB array format
|
||||
const keywords = (node.properties as any).keywords || [];
|
||||
const keywordsStr = `[${keywords.map((k: string) => `'${k.replace(/'/g, "''")}'`).join(',')}]`;
|
||||
|
||||
@@ -221,7 +221,7 @@ const generateCommunityCSV = (nodes: GraphNode[]): string => {
|
||||
escapeCSVField(node.id),
|
||||
escapeCSVField(node.properties.name || ''), // label is stored in name
|
||||
escapeCSVField(node.properties.heuristicLabel || ''),
|
||||
keywordsStr, // Array format for KuzuDB
|
||||
keywordsStr, // Array format for LadybugDB
|
||||
escapeCSVField((node.properties as any).description || ''),
|
||||
escapeCSVField((node.properties as any).enrichedBy || 'heuristic'),
|
||||
escapeCSVNumber(node.properties.cohesion, 0),
|
||||
+116
-108
@@ -1,51 +1,52 @@
|
||||
/**
|
||||
* KuzuDB Adapter
|
||||
*
|
||||
* Manages the KuzuDB WASM instance for client-side graph database operations.
|
||||
* LadybugDB Adapter
|
||||
*
|
||||
* Manages the LadybugDB WASM instance for client-side graph database operations.
|
||||
* Uses the "Snapshot / Bulk Load" pattern with COPY FROM for performance.
|
||||
*
|
||||
*
|
||||
* Multi-table schema: separate tables for File, Function, Class, etc.
|
||||
*/
|
||||
|
||||
import { KnowledgeGraph } from '../graph/types';
|
||||
import {
|
||||
NODE_TABLES,
|
||||
import {
|
||||
NODE_TABLES,
|
||||
REL_TABLE_NAME,
|
||||
SCHEMA_QUERIES,
|
||||
SCHEMA_QUERIES,
|
||||
EMBEDDING_TABLE_NAME,
|
||||
NodeTableName,
|
||||
} from './schema';
|
||||
import { generateAllCSVs } from './csv-generator';
|
||||
import { getQueryRows } from './query-result';
|
||||
|
||||
// Holds the reference to the dynamically loaded module
|
||||
let kuzu: any = null;
|
||||
let lbug: any = null;
|
||||
let db: any = null;
|
||||
let conn: any = null;
|
||||
|
||||
/**
|
||||
* Initialize KuzuDB WASM module and create in-memory database
|
||||
* Initialize LadybugDB WASM module and create in-memory database
|
||||
*/
|
||||
export const initKuzu = async () => {
|
||||
if (conn) return { db, conn, kuzu };
|
||||
export const initLbug = async () => {
|
||||
if (conn) return { db, conn, lbug };
|
||||
|
||||
try {
|
||||
if (import.meta.env.DEV) console.log('🚀 Initializing KuzuDB...');
|
||||
if (import.meta.env.DEV) console.log('🚀 Initializing LadybugDB...');
|
||||
|
||||
// 1. Dynamic Import (Fixes the "not a function" bundler issue)
|
||||
const kuzuModule = await import('kuzu-wasm');
|
||||
|
||||
const lbugModule = await import('@ladybugdb/wasm-core');
|
||||
|
||||
// 2. Handle Vite/Webpack "default" wrapping
|
||||
kuzu = kuzuModule.default || kuzuModule;
|
||||
lbug = lbugModule.default || lbugModule;
|
||||
|
||||
// 3. Initialize WASM
|
||||
await kuzu.init();
|
||||
|
||||
// 4. Create Database with 512MB buffer pool
|
||||
await lbug.init();
|
||||
|
||||
// 4. Create Database with 512MB buffer manager
|
||||
const BUFFER_POOL_SIZE = 512 * 1024 * 1024; // 512MB
|
||||
db = new kuzu.Database(':memory:', BUFFER_POOL_SIZE);
|
||||
conn = new kuzu.Connection(db);
|
||||
|
||||
if (import.meta.env.DEV) console.log('✅ KuzuDB WASM Initialized');
|
||||
db = new lbug.Database(':memory:', BUFFER_POOL_SIZE);
|
||||
conn = new lbug.Connection(db);
|
||||
|
||||
if (import.meta.env.DEV) console.log('✅ LadybugDB WASM Initialized');
|
||||
|
||||
// 5. Initialize Schema (all node tables, then rel tables, then embedding table)
|
||||
for (const schemaQuery of SCHEMA_QUERIES) {
|
||||
@@ -58,60 +59,60 @@ export const initKuzu = async () => {
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (import.meta.env.DEV) console.log('✅ KuzuDB Multi-Table Schema Created');
|
||||
|
||||
return { db, conn, kuzu };
|
||||
if (import.meta.env.DEV) console.log('✅ LadybugDB Multi-Table Schema Created');
|
||||
|
||||
return { db, conn, lbug };
|
||||
} catch (error) {
|
||||
if (import.meta.env.DEV) console.error('❌ KuzuDB Initialization Failed:', error);
|
||||
if (import.meta.env.DEV) console.error('❌ LadybugDB Initialization Failed:', error);
|
||||
throw error;
|
||||
}
|
||||
};
|
||||
|
||||
/**
|
||||
* Load a KnowledgeGraph into KuzuDB using COPY FROM (bulk load)
|
||||
* Load a KnowledgeGraph into LadybugDB using COPY FROM (bulk load)
|
||||
* Uses batched CSV writes and COPY statements for optimal performance
|
||||
*/
|
||||
export const loadGraphToKuzu = async (
|
||||
graph: KnowledgeGraph,
|
||||
export const loadGraphToLbug = async (
|
||||
graph: KnowledgeGraph,
|
||||
fileContents: Map<string, string>
|
||||
) => {
|
||||
const { conn, kuzu } = await initKuzu();
|
||||
|
||||
const { conn, lbug } = await initLbug();
|
||||
|
||||
try {
|
||||
if (import.meta.env.DEV) console.log(`KuzuDB: Generating CSVs for ${graph.nodeCount} nodes...`);
|
||||
|
||||
if (import.meta.env.DEV) console.log(`LadybugDB: Generating CSVs for ${graph.nodeCount} nodes...`);
|
||||
|
||||
// 1. Generate all CSVs (per-table)
|
||||
const csvData = generateAllCSVs(graph, fileContents);
|
||||
|
||||
const fs = kuzu.FS;
|
||||
|
||||
|
||||
const fs = lbug.FS;
|
||||
|
||||
// 2. Write all node CSVs to virtual filesystem
|
||||
const nodeFiles: Array<{ table: NodeTableName; path: string }> = [];
|
||||
for (const [tableName, csv] of csvData.nodes.entries()) {
|
||||
// Skip empty CSVs (only header row)
|
||||
if (csv.split('\n').length <= 1) continue;
|
||||
|
||||
|
||||
const path = `/${tableName.toLowerCase()}.csv`;
|
||||
try { await fs.unlink(path); } catch {}
|
||||
await fs.writeFile(path, csv);
|
||||
nodeFiles.push({ table: tableName, path });
|
||||
}
|
||||
|
||||
|
||||
// 3. Parse relation CSV and prepare for INSERT (COPY FROM doesn't work with multi-pair tables)
|
||||
const relLines = csvData.relCSV.split('\n').slice(1).filter(line => line.trim());
|
||||
const relCount = relLines.length;
|
||||
|
||||
|
||||
if (import.meta.env.DEV) {
|
||||
console.log(`KuzuDB: Wrote ${nodeFiles.length} node CSVs, ${relCount} relations to insert`);
|
||||
console.log(`LadybugDB: Wrote ${nodeFiles.length} node CSVs, ${relCount} relations to insert`);
|
||||
}
|
||||
|
||||
|
||||
// 4. COPY all node tables (must complete before rels due to FK constraints)
|
||||
for (const { table, path } of nodeFiles) {
|
||||
const copyQuery = getCopyQuery(table, path);
|
||||
await conn.query(copyQuery);
|
||||
}
|
||||
|
||||
|
||||
// 5. INSERT relations one by one (COPY doesn't work with multi-pair REL tables)
|
||||
// Build a set of valid table names for fast lookup
|
||||
const validTables = new Set<string>(NODE_TABLES as readonly string[]);
|
||||
@@ -135,13 +136,13 @@ export const loadGraphToKuzu = async (
|
||||
// Format: "from","to","type",confidence,"reason",step
|
||||
const match = line.match(/"([^"]*)","([^"]*)","([^"]*)",([0-9.]+),"([^"]*)",([0-9-]+)/);
|
||||
if (!match) continue;
|
||||
|
||||
|
||||
const [, fromId, toId, relType, confidenceStr, reason, stepStr] = match;
|
||||
|
||||
const fromLabel = getNodeLabel(fromId);
|
||||
const toLabel = getNodeLabel(toId);
|
||||
|
||||
// Skip relationships where either node's label doesn't have a table in KuzuDB
|
||||
// Skip relationships where either node's label doesn't have a table in LadybugDB
|
||||
// Querying a non-existent table causes a fatal native crash
|
||||
if (!validTables.has(fromLabel) || !validTables.has(toLabel)) {
|
||||
skippedRels++;
|
||||
@@ -150,7 +151,7 @@ export const loadGraphToKuzu = async (
|
||||
|
||||
const confidence = parseFloat(confidenceStr) || 1.0;
|
||||
const step = parseInt(stepStr) || 0;
|
||||
|
||||
|
||||
const insertQuery = `
|
||||
MATCH (a:${escapeLabel(fromLabel)} {id: '${fromId.replace(/'/g, "''")}'}),
|
||||
(b:${escapeLabel(toLabel)} {id: '${toId.replace(/'/g, "''")}'})
|
||||
@@ -167,38 +168,39 @@ export const loadGraphToKuzu = async (
|
||||
const toLabel = getNodeLabel(toId);
|
||||
const key = `${relType}:${fromLabel}->` + toLabel;
|
||||
skippedRelStats.set(key, (skippedRelStats.get(key) || 0) + 1);
|
||||
|
||||
|
||||
if (import.meta.env.DEV) {
|
||||
console.warn(`⚠️ Skipped: ${key} | "${fromId}" → "${toId}" | ${err instanceof Error ? err.message : String(err)}`);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
if (import.meta.env.DEV) {
|
||||
console.log(`KuzuDB: Inserted ${insertedRels}/${relCount} relations`);
|
||||
console.log(`LadybugDB: Inserted ${insertedRels}/${relCount} relations`);
|
||||
if (skippedRels > 0) {
|
||||
const topSkipped = Array.from(skippedRelStats.entries())
|
||||
.sort((a, b) => b[1] - a[1])
|
||||
.slice(0, 10);
|
||||
console.warn(`KuzuDB: Skipped ${skippedRels}/${relCount} relations (top by kind/pair):`, topSkipped);
|
||||
console.warn(`LadybugDB: Skipped ${skippedRels}/${relCount} relations (top by kind/pair):`, topSkipped);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
// 6. Verify results
|
||||
let totalNodes = 0;
|
||||
for (const tableName of NODE_TABLES) {
|
||||
try {
|
||||
const countRes = await conn.query(`MATCH (n:${tableName}) RETURN count(n) AS cnt`);
|
||||
const countRow = await countRes.getNext();
|
||||
const countRows = await getQueryRows(countRes);
|
||||
const countRow = countRows[0];
|
||||
const count = countRow ? (countRow.cnt ?? countRow[0] ?? 0) : 0;
|
||||
totalNodes += Number(count);
|
||||
} catch {
|
||||
// Table might be empty, skip
|
||||
}
|
||||
}
|
||||
|
||||
if (import.meta.env.DEV) console.log(`✅ KuzuDB Bulk Load Complete. Total nodes: ${totalNodes}, edges: ${insertedRels}`);
|
||||
|
||||
if (import.meta.env.DEV) console.log(`✅ LadybugDB Bulk Load Complete. Total nodes: ${totalNodes}, edges: ${insertedRels}`);
|
||||
|
||||
// 7. Cleanup CSV files
|
||||
for (const { path } of nodeFiles) {
|
||||
@@ -208,12 +210,12 @@ export const loadGraphToKuzu = async (
|
||||
return { success: true, count: totalNodes };
|
||||
|
||||
} catch (error) {
|
||||
if (import.meta.env.DEV) console.error('❌ KuzuDB Bulk Load Failed:', error);
|
||||
if (import.meta.env.DEV) console.error('❌ LadybugDB Bulk Load Failed:', error);
|
||||
return { success: false, count: 0 };
|
||||
}
|
||||
};
|
||||
|
||||
// KuzuDB default ESCAPE is '\' (backslash), but our CSV uses RFC 4180 escaping ("" for literal quotes).
|
||||
// LadybugDB default ESCAPE is '\' (backslash), but our CSV uses RFC 4180 escaping ("" for literal quotes).
|
||||
// Source code content is full of backslashes which confuse the auto-detection.
|
||||
// We MUST explicitly set ESCAPE='"' and disable auto_detect.
|
||||
const COPY_CSV_OPTS = `(HEADER=true, ESCAPE='"', DELIM=',', QUOTE='"', PARALLEL=false, auto_detect=false)`;
|
||||
@@ -229,6 +231,9 @@ const escapeTableName = (table: string): string => {
|
||||
return BACKTICK_TABLES.has(table) ? `\`${table}\`` : table;
|
||||
};
|
||||
|
||||
/** Tables with isExported column (TypeScript/JS-native types) */
|
||||
const TABLES_WITH_EXPORTED = new Set<string>(['Function', 'Class', 'Interface', 'Method', 'CodeElement']);
|
||||
|
||||
/**
|
||||
* Get the COPY query for a node table with correct column mapping
|
||||
*/
|
||||
@@ -246,8 +251,12 @@ const getCopyQuery = (table: NodeTableName, path: string): string => {
|
||||
if (table === 'Process') {
|
||||
return `COPY ${t}(id, label, heuristicLabel, processType, stepCount, communities, entryPointId, terminalId) FROM "${path}" ${COPY_CSV_OPTS}`;
|
||||
}
|
||||
// Code element tables (Function, Class, Interface, Method, CodeElement, and multi-language)
|
||||
return `COPY ${t}(id, name, filePath, startLine, endLine, isExported, content) FROM "${path}" ${COPY_CSV_OPTS}`;
|
||||
// TypeScript/JS code element tables have isExported; multi-language tables do not
|
||||
if (TABLES_WITH_EXPORTED.has(table)) {
|
||||
return `COPY ${t}(id, name, filePath, startLine, endLine, isExported, content) FROM "${path}" ${COPY_CSV_OPTS}`;
|
||||
}
|
||||
// Multi-language tables (Struct, Impl, Trait, Macro, etc.)
|
||||
return `COPY ${t}(id, name, filePath, startLine, endLine, content) FROM "${path}" ${COPY_CSV_OPTS}`;
|
||||
};
|
||||
|
||||
/**
|
||||
@@ -256,12 +265,12 @@ const getCopyQuery = (table: NodeTableName, path: string): string => {
|
||||
*/
|
||||
export const executeQuery = async (cypher: string): Promise<any[]> => {
|
||||
if (!conn) {
|
||||
await initKuzu();
|
||||
await initLbug();
|
||||
}
|
||||
|
||||
|
||||
try {
|
||||
const result = await conn.query(cypher);
|
||||
|
||||
|
||||
// Extract column names from RETURN clause
|
||||
const returnMatch = cypher.match(/RETURN\s+(.+?)(?:\s+ORDER|\s+LIMIT|\s+SKIP|\s*$)/is);
|
||||
let columnNames: string[] = [];
|
||||
@@ -284,12 +293,11 @@ export const executeQuery = async (cypher: string): Promise<any[]> => {
|
||||
return col.replace(/[^a-zA-Z0-9_]/g, '_');
|
||||
});
|
||||
}
|
||||
|
||||
|
||||
// Collect all rows
|
||||
const allRows = await getQueryRows(result);
|
||||
const rows: any[] = [];
|
||||
while (await result.hasNext()) {
|
||||
const row = await result.getNext();
|
||||
|
||||
for (const row of allRows) {
|
||||
// Convert tuple to named object if we have column names and row is array
|
||||
if (Array.isArray(row) && columnNames.length === row.length) {
|
||||
const namedRow: Record<string, any> = {};
|
||||
@@ -302,7 +310,7 @@ export const executeQuery = async (cypher: string): Promise<any[]> => {
|
||||
rows.push(row);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
return rows;
|
||||
} catch (error) {
|
||||
if (import.meta.env.DEV) console.error('Query execution failed:', error);
|
||||
@@ -313,7 +321,7 @@ export const executeQuery = async (cypher: string): Promise<any[]> => {
|
||||
/**
|
||||
* Get database statistics
|
||||
*/
|
||||
export const getKuzuStats = async (): Promise<{ nodes: number; edges: number }> => {
|
||||
export const getLbugStats = async (): Promise<{ nodes: number; edges: number }> => {
|
||||
if (!conn) {
|
||||
return { nodes: 0, edges: 0 };
|
||||
}
|
||||
@@ -324,43 +332,45 @@ export const getKuzuStats = async (): Promise<{ nodes: number; edges: number }>
|
||||
for (const tableName of NODE_TABLES) {
|
||||
try {
|
||||
const nodeResult = await conn.query(`MATCH (n:${tableName}) RETURN count(n) AS cnt`);
|
||||
const nodeRow = await nodeResult.getNext();
|
||||
const nodeRows = await getQueryRows(nodeResult);
|
||||
const nodeRow = nodeRows[0];
|
||||
totalNodes += Number(nodeRow?.cnt ?? nodeRow?.[0] ?? 0);
|
||||
} catch {
|
||||
// Table might not exist or be empty
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
// Count edges from single relation table
|
||||
let totalEdges = 0;
|
||||
try {
|
||||
const edgeResult = await conn.query(`MATCH ()-[r:${REL_TABLE_NAME}]->() RETURN count(r) AS cnt`);
|
||||
const edgeRow = await edgeResult.getNext();
|
||||
const edgeRows = await getQueryRows(edgeResult);
|
||||
const edgeRow = edgeRows[0];
|
||||
totalEdges = Number(edgeRow?.cnt ?? edgeRow?.[0] ?? 0);
|
||||
} catch {
|
||||
// Table might not exist or be empty
|
||||
}
|
||||
|
||||
|
||||
return { nodes: totalNodes, edges: totalEdges };
|
||||
} catch (error) {
|
||||
if (import.meta.env.DEV) {
|
||||
console.warn('Failed to get Kuzu stats:', error);
|
||||
console.warn('Failed to get LadybugDB stats:', error);
|
||||
}
|
||||
return { nodes: 0, edges: 0 };
|
||||
}
|
||||
};
|
||||
|
||||
/**
|
||||
* Check if KuzuDB is initialized and has data
|
||||
* Check if LadybugDB is initialized and has data
|
||||
*/
|
||||
export const isKuzuReady = (): boolean => {
|
||||
export const isLbugReady = (): boolean => {
|
||||
return conn !== null && db !== null;
|
||||
};
|
||||
|
||||
/**
|
||||
* Close the database connection (cleanup)
|
||||
*/
|
||||
export const closeKuzu = async (): Promise<void> => {
|
||||
export const closeLbug = async (): Promise<void> => {
|
||||
if (conn) {
|
||||
try {
|
||||
await conn.close();
|
||||
@@ -373,7 +383,7 @@ export const closeKuzu = async (): Promise<void> => {
|
||||
} catch {}
|
||||
db = null;
|
||||
}
|
||||
kuzu = null;
|
||||
lbug = null;
|
||||
};
|
||||
|
||||
/**
|
||||
@@ -387,24 +397,20 @@ export const executePrepared = async (
|
||||
params: Record<string, any>
|
||||
): Promise<any[]> => {
|
||||
if (!conn) {
|
||||
await initKuzu();
|
||||
await initLbug();
|
||||
}
|
||||
|
||||
|
||||
try {
|
||||
const stmt = await conn.prepare(cypher);
|
||||
if (!stmt.isSuccess()) {
|
||||
const errMsg = await stmt.getErrorMessage();
|
||||
throw new Error(`Prepare failed: ${errMsg}`);
|
||||
}
|
||||
|
||||
|
||||
const result = await conn.execute(stmt, params);
|
||||
|
||||
const rows: any[] = [];
|
||||
while (await result.hasNext()) {
|
||||
const row = await result.getNext();
|
||||
rows.push(row);
|
||||
}
|
||||
|
||||
|
||||
const rows = await getQueryRows(result);
|
||||
|
||||
await stmt.close();
|
||||
return rows;
|
||||
} catch (error) {
|
||||
@@ -421,22 +427,22 @@ export const executeWithReusedStatement = async (
|
||||
paramsList: Array<Record<string, any>>
|
||||
): Promise<void> => {
|
||||
if (!conn) {
|
||||
await initKuzu();
|
||||
await initLbug();
|
||||
}
|
||||
|
||||
|
||||
if (paramsList.length === 0) return;
|
||||
|
||||
|
||||
const SUB_BATCH_SIZE = 4;
|
||||
|
||||
|
||||
for (let i = 0; i < paramsList.length; i += SUB_BATCH_SIZE) {
|
||||
const subBatch = paramsList.slice(i, i + SUB_BATCH_SIZE);
|
||||
|
||||
|
||||
const stmt = await conn.prepare(cypher);
|
||||
if (!stmt.isSuccess()) {
|
||||
const errMsg = await stmt.getErrorMessage();
|
||||
throw new Error(`Prepare failed: ${errMsg}`);
|
||||
}
|
||||
|
||||
|
||||
try {
|
||||
for (const params of subBatch) {
|
||||
await conn.execute(stmt, params);
|
||||
@@ -444,7 +450,7 @@ export const executeWithReusedStatement = async (
|
||||
} finally {
|
||||
await stmt.close();
|
||||
}
|
||||
|
||||
|
||||
if (i + SUB_BATCH_SIZE < paramsList.length) {
|
||||
await new Promise(r => setTimeout(r, 0));
|
||||
}
|
||||
@@ -456,65 +462,67 @@ export const executeWithReusedStatement = async (
|
||||
*/
|
||||
export const testArrayParams = async (): Promise<{ success: boolean; error?: string }> => {
|
||||
if (!conn) {
|
||||
await initKuzu();
|
||||
await initLbug();
|
||||
}
|
||||
|
||||
|
||||
try {
|
||||
const testEmbedding = new Array(384).fill(0).map((_, i) => i / 384);
|
||||
|
||||
|
||||
// Get any node ID to test with (try File first, then others)
|
||||
let testNodeId: string | null = null;
|
||||
for (const tableName of NODE_TABLES) {
|
||||
try {
|
||||
const nodeResult = await conn.query(`MATCH (n:${tableName}) RETURN n.id AS id LIMIT 1`);
|
||||
const nodeRow = await nodeResult.getNext();
|
||||
const nodeRows = await getQueryRows(nodeResult);
|
||||
const nodeRow = nodeRows[0];
|
||||
if (nodeRow) {
|
||||
testNodeId = nodeRow.id ?? nodeRow[0];
|
||||
break;
|
||||
}
|
||||
} catch {}
|
||||
}
|
||||
|
||||
|
||||
if (!testNodeId) {
|
||||
return { success: false, error: 'No nodes found to test with' };
|
||||
}
|
||||
|
||||
|
||||
if (import.meta.env.DEV) {
|
||||
console.log('🧪 Testing array params with node:', testNodeId);
|
||||
}
|
||||
|
||||
|
||||
// First create an embedding entry
|
||||
const createQuery = `CREATE (e:${EMBEDDING_TABLE_NAME} {nodeId: $nodeId, embedding: $embedding})`;
|
||||
const stmt = await conn.prepare(createQuery);
|
||||
|
||||
|
||||
if (!stmt.isSuccess()) {
|
||||
const errMsg = await stmt.getErrorMessage();
|
||||
return { success: false, error: `Prepare failed: ${errMsg}` };
|
||||
}
|
||||
|
||||
|
||||
await conn.execute(stmt, {
|
||||
nodeId: testNodeId,
|
||||
embedding: testEmbedding,
|
||||
});
|
||||
|
||||
|
||||
await stmt.close();
|
||||
|
||||
|
||||
// Verify it was stored
|
||||
const verifyResult = await conn.query(
|
||||
`MATCH (e:${EMBEDDING_TABLE_NAME} {nodeId: '${testNodeId}'}) RETURN e.embedding AS emb`
|
||||
);
|
||||
const verifyRow = await verifyResult.getNext();
|
||||
const verifyRows = await getQueryRows(verifyResult);
|
||||
const verifyRow = verifyRows[0];
|
||||
const storedEmb = verifyRow?.emb ?? verifyRow?.[0];
|
||||
|
||||
|
||||
if (storedEmb && Array.isArray(storedEmb) && storedEmb.length === 384) {
|
||||
if (import.meta.env.DEV) {
|
||||
console.log('✅ Array params WORK! Stored embedding length:', storedEmb.length);
|
||||
}
|
||||
return { success: true };
|
||||
} else {
|
||||
return {
|
||||
success: false,
|
||||
error: `Embedding not stored correctly. Got: ${typeof storedEmb}, length: ${storedEmb?.length}`
|
||||
return {
|
||||
success: false,
|
||||
error: `Embedding not stored correctly. Got: ${typeof storedEmb}, length: ${storedEmb?.length}`
|
||||
};
|
||||
}
|
||||
} catch (error) {
|
||||
@@ -0,0 +1,41 @@
|
||||
import { describe, expect, it, vi } from 'vitest';
|
||||
|
||||
import { getQueryRows } from './query-result';
|
||||
|
||||
describe('getQueryRows', () => {
|
||||
it('prefers getAllObjects when available', async () => {
|
||||
const rows = [{ name: 'foo' }];
|
||||
const getAllObjects = vi.fn().mockResolvedValue(rows);
|
||||
const getAllRows = vi.fn().mockResolvedValue([{ name: 'bar' }]);
|
||||
const getAll = vi.fn().mockResolvedValue([['baz']]);
|
||||
|
||||
await expect(getQueryRows({ getAllObjects, getAllRows, getAll })).resolves.toEqual(rows);
|
||||
|
||||
expect(getAllObjects).toHaveBeenCalledTimes(1);
|
||||
expect(getAllRows).not.toHaveBeenCalled();
|
||||
expect(getAll).not.toHaveBeenCalled();
|
||||
});
|
||||
|
||||
it('falls back to getAllRows when getAllObjects is absent', async () => {
|
||||
const rows = [{ name: 'bar' }];
|
||||
const getAllRows = vi.fn().mockResolvedValue(rows);
|
||||
const getAll = vi.fn().mockResolvedValue([['baz']]);
|
||||
|
||||
await expect(getQueryRows({ getAllRows, getAll })).resolves.toEqual(rows);
|
||||
|
||||
expect(getAllRows).toHaveBeenCalledTimes(1);
|
||||
expect(getAll).not.toHaveBeenCalled();
|
||||
});
|
||||
|
||||
it('falls back to getAll as a final fallback', async () => {
|
||||
const rows = [['baz']];
|
||||
const getAll = vi.fn().mockResolvedValue(rows);
|
||||
|
||||
await expect(getQueryRows({ getAll })).resolves.toEqual(rows);
|
||||
expect(getAll).toHaveBeenCalledTimes(1);
|
||||
});
|
||||
|
||||
it('throws when no supported query API is exposed', async () => {
|
||||
await expect(getQueryRows({})).rejects.toThrow('Unsupported LadybugDB QueryResult shape');
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,21 @@
|
||||
export const getQueryRows = async (result: unknown): Promise<any[]> => {
|
||||
if (!result || typeof result !== 'object') return [];
|
||||
|
||||
const queryResult = result as {
|
||||
getAllObjects?: () => Promise<any[]>;
|
||||
getAllRows?: () => Promise<any[]>;
|
||||
getAll?: () => Promise<any[]>;
|
||||
};
|
||||
|
||||
if (typeof queryResult.getAllObjects === 'function') {
|
||||
return await queryResult.getAllObjects();
|
||||
}
|
||||
if (typeof queryResult.getAllRows === 'function') {
|
||||
return await queryResult.getAllRows();
|
||||
}
|
||||
if (typeof queryResult.getAll === 'function') {
|
||||
return await queryResult.getAll();
|
||||
}
|
||||
|
||||
throw new Error('Unsupported LadybugDB QueryResult shape');
|
||||
};
|
||||
@@ -1,5 +1,5 @@
|
||||
/**
|
||||
* KuzuDB Schema Definitions
|
||||
* LadybugDB Schema Definitions
|
||||
*
|
||||
* Hybrid Schema:
|
||||
* - Separate node tables for each code element type (File, Function, Class, etc.)
|
||||
@@ -13,14 +13,15 @@ import { ChatAnthropic } from '@langchain/anthropic';
|
||||
import { ChatOllama } from '@langchain/ollama';
|
||||
import type { BaseChatModel } from '@langchain/core/language_models/chat_models';
|
||||
import { createGraphRAGTools } from './tools';
|
||||
import type {
|
||||
ProviderConfig,
|
||||
import type {
|
||||
ProviderConfig,
|
||||
OpenAIConfig,
|
||||
AzureOpenAIConfig,
|
||||
AzureOpenAIConfig,
|
||||
GeminiConfig,
|
||||
AnthropicConfig,
|
||||
OllamaConfig,
|
||||
OpenRouterConfig,
|
||||
MiniMaxConfig,
|
||||
AgentStreamChunk,
|
||||
} from './types';
|
||||
import {
|
||||
@@ -197,7 +198,7 @@ export const createChatModel = (config: ProviderConfig): BaseChatModel => {
|
||||
|
||||
case 'openrouter': {
|
||||
const openRouterConfig = config as OpenRouterConfig;
|
||||
|
||||
|
||||
// Debug logging
|
||||
if (import.meta.env.DEV) {
|
||||
console.log('🌐 OpenRouter config:', {
|
||||
@@ -207,11 +208,11 @@ export const createChatModel = (config: ProviderConfig): BaseChatModel => {
|
||||
baseUrl: openRouterConfig.baseUrl,
|
||||
});
|
||||
}
|
||||
|
||||
|
||||
if (!openRouterConfig.apiKey || openRouterConfig.apiKey.trim() === '') {
|
||||
throw new Error('OpenRouter API key is required but was not provided');
|
||||
}
|
||||
|
||||
|
||||
return new ChatOpenAI({
|
||||
openAIApiKey: openRouterConfig.apiKey,
|
||||
apiKey: openRouterConfig.apiKey, // Fallback for some versions
|
||||
@@ -225,7 +226,26 @@ export const createChatModel = (config: ProviderConfig): BaseChatModel => {
|
||||
streaming: true,
|
||||
});
|
||||
}
|
||||
|
||||
|
||||
case 'minimax': {
|
||||
const minimaxConfig = config as MiniMaxConfig;
|
||||
|
||||
if (!minimaxConfig.apiKey || minimaxConfig.apiKey.trim() === '') {
|
||||
throw new Error('MiniMax API key is required but was not provided');
|
||||
}
|
||||
|
||||
return new ChatAnthropic({
|
||||
anthropicApiKey: minimaxConfig.apiKey,
|
||||
model: minimaxConfig.model,
|
||||
temperature: minimaxConfig.temperature ?? 0.1,
|
||||
maxTokens: minimaxConfig.maxTokens ?? 8192,
|
||||
streaming: true,
|
||||
clientOptions: {
|
||||
baseURL: 'https://api.minimax.io/anthropic',
|
||||
},
|
||||
});
|
||||
}
|
||||
|
||||
default:
|
||||
throw new Error(`Unsupported provider: ${(config as any).provider}`);
|
||||
}
|
||||
|
||||
@@ -5,9 +5,9 @@
|
||||
* All API keys are stored locally - never sent to any server except the LLM provider.
|
||||
*/
|
||||
|
||||
import {
|
||||
LLMSettings,
|
||||
DEFAULT_LLM_SETTINGS,
|
||||
import {
|
||||
LLMSettings,
|
||||
DEFAULT_LLM_SETTINGS,
|
||||
LLMProvider,
|
||||
OpenAIConfig,
|
||||
AzureOpenAIConfig,
|
||||
@@ -15,6 +15,7 @@ import {
|
||||
AnthropicConfig,
|
||||
OllamaConfig,
|
||||
OpenRouterConfig,
|
||||
MiniMaxConfig,
|
||||
ProviderConfig,
|
||||
} from './types';
|
||||
|
||||
@@ -60,6 +61,10 @@ export const loadSettings = (): LLMSettings => {
|
||||
...DEFAULT_LLM_SETTINGS.openrouter,
|
||||
...parsed.openrouter,
|
||||
},
|
||||
minimax: {
|
||||
...DEFAULT_LLM_SETTINGS.minimax,
|
||||
...parsed.minimax,
|
||||
},
|
||||
};
|
||||
} catch (error) {
|
||||
console.warn('Failed to load LLM settings:', error);
|
||||
@@ -89,6 +94,7 @@ export const updateProviderSettings = <T extends LLMProvider>(
|
||||
T extends 'gemini' ? Partial<Omit<GeminiConfig, 'provider'>> :
|
||||
T extends 'anthropic' ? Partial<Omit<AnthropicConfig, 'provider'>> :
|
||||
T extends 'ollama' ? Partial<Omit<OllamaConfig, 'provider'>> :
|
||||
T extends 'minimax' ? Partial<Omit<MiniMaxConfig, 'provider'>> :
|
||||
never
|
||||
>
|
||||
): LLMSettings => {
|
||||
@@ -162,6 +168,17 @@ export const updateProviderSettings = <T extends LLMProvider>(
|
||||
saveSettings(updated);
|
||||
return updated;
|
||||
}
|
||||
case 'minimax': {
|
||||
const updated: LLMSettings = {
|
||||
...current,
|
||||
minimax: {
|
||||
...(current.minimax ?? {}),
|
||||
...(updates as Partial<Omit<MiniMaxConfig, 'provider'>>),
|
||||
},
|
||||
};
|
||||
saveSettings(updated);
|
||||
return updated;
|
||||
}
|
||||
default: {
|
||||
// Should be unreachable due to T extends LLMProvider, but keep a safe fallback
|
||||
const updated: LLMSettings = { ...current };
|
||||
@@ -245,7 +262,16 @@ export const getActiveProviderConfig = (): ProviderConfig | null => {
|
||||
temperature: settings.openrouter.temperature,
|
||||
maxTokens: settings.openrouter.maxTokens,
|
||||
} as OpenRouterConfig;
|
||||
|
||||
|
||||
case 'minimax':
|
||||
if (!settings.minimax?.apiKey) {
|
||||
return null;
|
||||
}
|
||||
return {
|
||||
provider: 'minimax',
|
||||
...settings.minimax,
|
||||
} as MiniMaxConfig;
|
||||
|
||||
default:
|
||||
return null;
|
||||
}
|
||||
@@ -282,6 +308,8 @@ export const getProviderDisplayName = (provider: LLMProvider): string => {
|
||||
return 'Ollama (Local)';
|
||||
case 'openrouter':
|
||||
return 'OpenRouter';
|
||||
case 'minimax':
|
||||
return 'MiniMax';
|
||||
default:
|
||||
return provider;
|
||||
}
|
||||
@@ -303,6 +331,8 @@ export const getAvailableModels = (provider: LLMProvider): string[] => {
|
||||
return ['claude-sonnet-4-20250514', 'claude-3-5-sonnet-20241022', 'claude-3-5-haiku-20241022', 'claude-3-opus-20240229'];
|
||||
case 'ollama':
|
||||
return ['llama3.2', 'llama3.1', 'mistral', 'codellama', 'deepseek-coder'];
|
||||
case 'minimax':
|
||||
return ['MiniMax-M2.5', 'MiniMax-M2.5-highspeed'];
|
||||
default:
|
||||
return [];
|
||||
}
|
||||
|
||||
@@ -17,7 +17,7 @@ import { z } from 'zod';
|
||||
import { WebGPUNotAvailableError, embedText, embeddingToArray, initEmbedder, isEmbedderReady } from '../embeddings/embedder';
|
||||
|
||||
/**
|
||||
* Tool factory - creates tools bound to the KuzuDB query functions
|
||||
* Tool factory - creates tools bound to the LadybugDB query functions
|
||||
*/
|
||||
export const createGraphRAGTools = (
|
||||
executeQuery: (cypher: string) => Promise<any[]>,
|
||||
@@ -975,7 +975,7 @@ MATCH (n:Function {id: emb.nodeId}) RETURN n`,
|
||||
// For code elements (Function, Class, etc.), use the direct id
|
||||
const isFileTarget = targetType === 'File';
|
||||
|
||||
// Query each depth level separately (KuzuDB doesn't support list comprehensions on paths)
|
||||
// Query each depth level separately (LadybugDB doesn't support list comprehensions on paths)
|
||||
// For depth 1: direct connections only
|
||||
// For depth 2+: chain multiple single-hop queries
|
||||
const depthQueries: Promise<any[]>[] = [];
|
||||
|
||||
@@ -8,7 +8,7 @@
|
||||
/**
|
||||
* Supported LLM providers
|
||||
*/
|
||||
export type LLMProvider = 'openai' | 'azure-openai' | 'gemini' | 'anthropic' | 'ollama' | 'openrouter';
|
||||
export type LLMProvider = 'openai' | 'azure-openai' | 'gemini' | 'anthropic' | 'ollama' | 'openrouter' | 'minimax';
|
||||
|
||||
/**
|
||||
* Base configuration shared by all providers
|
||||
@@ -78,10 +78,19 @@ export interface OpenRouterConfig extends BaseProviderConfig {
|
||||
baseUrl?: string; // defaults to https://openrouter.ai/api/v1
|
||||
}
|
||||
|
||||
/**
|
||||
* MiniMax configuration (Anthropic-compatible API)
|
||||
*/
|
||||
export interface MiniMaxConfig extends BaseProviderConfig {
|
||||
provider: 'minimax';
|
||||
apiKey: string;
|
||||
model: string; // e.g., 'MiniMax-M2.5', 'MiniMax-M2.5-highspeed'
|
||||
}
|
||||
|
||||
/**
|
||||
* Union type for all provider configurations
|
||||
*/
|
||||
export type ProviderConfig = OpenAIConfig | AzureOpenAIConfig | GeminiConfig | AnthropicConfig | OllamaConfig | OpenRouterConfig;
|
||||
export type ProviderConfig = OpenAIConfig | AzureOpenAIConfig | GeminiConfig | AnthropicConfig | OllamaConfig | OpenRouterConfig | MiniMaxConfig;
|
||||
|
||||
/**
|
||||
* Stored settings (what goes to localStorage)
|
||||
@@ -98,6 +107,7 @@ export interface LLMSettings {
|
||||
anthropic?: Partial<Omit<AnthropicConfig, 'provider'>>;
|
||||
ollama?: Partial<Omit<OllamaConfig, 'provider'>>;
|
||||
openrouter?: Partial<Omit<OpenRouterConfig, 'provider'>>;
|
||||
minimax?: Partial<Omit<MiniMaxConfig, 'provider'>>;
|
||||
|
||||
// Intelligent Clustering Settings
|
||||
intelligentClustering: boolean;
|
||||
@@ -148,6 +158,11 @@ export const DEFAULT_LLM_SETTINGS: LLMSettings = {
|
||||
baseUrl: 'https://openrouter.ai/api/v1',
|
||||
temperature: 0.1,
|
||||
},
|
||||
minimax: {
|
||||
apiKey: '',
|
||||
model: 'MiniMax-M2.5',
|
||||
temperature: 0.1,
|
||||
},
|
||||
};
|
||||
|
||||
/**
|
||||
@@ -224,7 +239,7 @@ export interface AgentStep {
|
||||
* Graph schema information for LLM context
|
||||
*/
|
||||
export const GRAPH_SCHEMA_DESCRIPTION = `
|
||||
KUZU GRAPH DATABASE SCHEMA (Multi-Table):
|
||||
LADYBUG GRAPH DATABASE SCHEMA (Multi-Table):
|
||||
|
||||
NODE TABLES:
|
||||
1. File - Source files
|
||||
|
||||
@@ -40,6 +40,8 @@ const getWasmPath = (language: SupportedLanguages, filePath?: string): string =>
|
||||
[SupportedLanguages.Go]: '/wasm/go/tree-sitter-go.wasm',
|
||||
[SupportedLanguages.Rust]: '/wasm/rust/tree-sitter-rust.wasm',
|
||||
[SupportedLanguages.PHP]: '/wasm/php/tree-sitter-php.wasm',
|
||||
[SupportedLanguages.Ruby]: '/wasm/ruby/tree-sitter-ruby.wasm',
|
||||
[SupportedLanguages.Kotlin]: '', // Kotlin WASM parser not yet available for web
|
||||
[SupportedLanguages.Swift]: '/wasm/swift/tree-sitter-swift.wasm',
|
||||
};
|
||||
|
||||
|
||||
@@ -1,6 +1,6 @@
|
||||
import { createContext, useContext, useState, useCallback, useRef, useEffect, ReactNode } from 'react';
|
||||
import * as Comlink from 'comlink';
|
||||
import { KnowledgeGraph, GraphNode, NodeLabel } from '../core/graph/types';
|
||||
import { KnowledgeGraph, GraphNode, GraphRelationship, NodeLabel } from '../core/graph/types';
|
||||
import { PipelineProgress, PipelineResult, deserializePipelineResult } from '../types/pipeline';
|
||||
import { createKnowledgeGraph } from '../core/graph/graph';
|
||||
import { DEFAULT_VISIBLE_LABELS } from '../lib/constants';
|
||||
@@ -95,6 +95,7 @@ interface AppState {
|
||||
isAIHighlightsEnabled: boolean;
|
||||
toggleAIHighlights: () => void;
|
||||
clearAIToolHighlights: () => void;
|
||||
clearAICitationHighlights: () => void;
|
||||
clearBlastRadius: () => void;
|
||||
queryResult: QueryResult | null;
|
||||
setQueryResult: (result: QueryResult | null) => void;
|
||||
@@ -125,6 +126,7 @@ interface AppState {
|
||||
runPipelineFromFiles: (files: FileEntry[], onProgress: (p: PipelineProgress) => void, clusteringConfig?: ProviderConfig) => Promise<PipelineResult>;
|
||||
runQuery: (cypher: string) => Promise<any[]>;
|
||||
isDatabaseReady: () => Promise<boolean>;
|
||||
loadServerGraph: (nodes: GraphNode[], relationships: GraphRelationship[], fileContents: Record<string, string>) => Promise<void>;
|
||||
|
||||
// Embedding state
|
||||
embeddingStatus: EmbeddingStatus;
|
||||
@@ -143,7 +145,9 @@ interface AppState {
|
||||
llmSettings: LLMSettings;
|
||||
updateLLMSettings: (updates: Partial<LLMSettings>) => void;
|
||||
isSettingsPanelOpen: boolean;
|
||||
isHelpDialogBoxOpen: boolean;
|
||||
setSettingsPanelOpen: (open: boolean) => void;
|
||||
setHelpDialogBoxOpen: (open: boolean) => void;
|
||||
isAgentReady: boolean;
|
||||
isAgentInitializing: boolean;
|
||||
agentError: string | null;
|
||||
@@ -225,6 +229,10 @@ export const AppStateProvider = ({ children }: { children: ReactNode }) => {
|
||||
setAIToolHighlightedNodeIds(new Set());
|
||||
}, []);
|
||||
|
||||
const clearAICitationHighlights = useCallback(() => {
|
||||
setAICitationHighlightedNodeIds(new Set());
|
||||
}, []);
|
||||
|
||||
const clearBlastRadius = useCallback(() => {
|
||||
setBlastRadiusNodeIds(new Set());
|
||||
}, []);
|
||||
@@ -290,6 +298,7 @@ export const AppStateProvider = ({ children }: { children: ReactNode }) => {
|
||||
// LLM/Agent state
|
||||
const [llmSettings, setLLMSettings] = useState<LLMSettings>(loadSettings);
|
||||
const [isSettingsPanelOpen, setSettingsPanelOpen] = useState(false);
|
||||
const [isHelpDialogBoxOpen, setHelpDialogBoxOpen] = useState(false);
|
||||
const [isAgentReady, setIsAgentReady] = useState(false);
|
||||
const [isAgentInitializing, setIsAgentInitializing] = useState(false);
|
||||
const [agentError, setAgentError] = useState<string | null>(null);
|
||||
@@ -482,6 +491,16 @@ export const AppStateProvider = ({ children }: { children: ReactNode }) => {
|
||||
}
|
||||
}, []);
|
||||
|
||||
const loadServerGraph = useCallback(async (
|
||||
nodes: GraphNode[],
|
||||
relationships: GraphRelationship[],
|
||||
fileContents: Record<string, string>
|
||||
): Promise<void> => {
|
||||
const api = apiRef.current;
|
||||
if (!api) throw new Error('Worker not initialized');
|
||||
await api.loadServerGraph(nodes, relationships, fileContents);
|
||||
}, []);
|
||||
|
||||
// Embedding methods
|
||||
const startEmbeddings = useCallback(async (forceDevice?: 'webgpu' | 'wasm'): Promise<void> => {
|
||||
const api = apiRef.current;
|
||||
@@ -979,10 +998,13 @@ export const AppStateProvider = ({ children }: { children: ReactNode }) => {
|
||||
setProgress({ phase: 'extracting', percent: 0, message: 'Switching repository...', detail: `Loading ${repoName}` });
|
||||
setViewMode('loading');
|
||||
|
||||
setIsAgentReady(false);
|
||||
|
||||
// Clear stale graph state from previous repo (highlights, selections, blast radius)
|
||||
// Without this, sigma reducers dim ALL nodes/edges because old node IDs don't match
|
||||
setHighlightedNodeIds(new Set());
|
||||
clearAIToolHighlights();
|
||||
clearAICitationHighlights();
|
||||
clearBlastRadius();
|
||||
setSelectedNode(null);
|
||||
setQueryResult(null);
|
||||
@@ -1017,17 +1039,29 @@ export const AppStateProvider = ({ children }: { children: ReactNode }) => {
|
||||
for (const [p, c] of Object.entries(result.fileContents)) fileMap.set(p, c);
|
||||
setFileContents(fileMap);
|
||||
|
||||
setViewMode('exploring');
|
||||
|
||||
if (getActiveProviderConfig()) initializeAgent(pName);
|
||||
|
||||
startEmbeddings().catch((err) => {
|
||||
if (err?.name === 'WebGPUNotAvailableError' || err?.message?.includes('WebGPU')) {
|
||||
startEmbeddings('wasm').catch(console.warn);
|
||||
} else {
|
||||
console.warn('Embeddings auto-start failed:', err);
|
||||
// Load graph into LadybugDB for Nexus AI queries, then init agent
|
||||
try {
|
||||
await loadServerGraph(result.nodes, result.relationships, result.fileContents);
|
||||
if (getActiveProviderConfig()) {
|
||||
await initializeAgent(pName);
|
||||
}
|
||||
});
|
||||
setViewMode('exploring');
|
||||
startEmbeddings().catch((err) => {
|
||||
if (err?.name === 'WebGPUNotAvailableError' || err?.message?.includes('WebGPU')) {
|
||||
startEmbeddings('wasm').catch(console.warn);
|
||||
} else {
|
||||
console.warn('Embeddings auto-start failed:', err);
|
||||
}
|
||||
});
|
||||
setProgress(null);
|
||||
} catch (err) {
|
||||
console.warn('Failed to load graph into LadybugDB:', err);
|
||||
setIsAgentReady(false);
|
||||
await apiRef.current?.disposeAgent();
|
||||
setAgentError('Failed to load graph into LadybugDB');
|
||||
setViewMode('exploring');
|
||||
setProgress(null);
|
||||
}
|
||||
} catch (err) {
|
||||
console.error('Repo switch failed:', err);
|
||||
setProgress({
|
||||
@@ -1035,9 +1069,11 @@ export const AppStateProvider = ({ children }: { children: ReactNode }) => {
|
||||
message: 'Failed to switch repository',
|
||||
detail: err instanceof Error ? err.message : 'Unknown error',
|
||||
});
|
||||
setIsAgentReady(false);
|
||||
await apiRef.current?.disposeAgent();
|
||||
setTimeout(() => { setViewMode('exploring'); setProgress(null); }, 3000);
|
||||
}
|
||||
}, [serverBaseUrl, setProgress, setViewMode, setProjectName, setGraph, setFileContents, initializeAgent, startEmbeddings, setHighlightedNodeIds, clearAIToolHighlights, clearBlastRadius, setSelectedNode, setQueryResult, setCodeReferences, setCodePanelOpen, setCodeReferenceFocus]);
|
||||
}, [serverBaseUrl, setProgress, setViewMode, setProjectName, setGraph, setFileContents, loadServerGraph, initializeAgent, startEmbeddings, setHighlightedNodeIds, clearAIToolHighlights, clearAICitationHighlights, clearBlastRadius, setSelectedNode, setQueryResult, setCodeReferences, setCodePanelOpen, setCodeReferenceFocus]);
|
||||
|
||||
const removeCodeReference = useCallback((id: string) => {
|
||||
setCodeReferences(prev => {
|
||||
@@ -1120,6 +1156,7 @@ export const AppStateProvider = ({ children }: { children: ReactNode }) => {
|
||||
isAIHighlightsEnabled,
|
||||
toggleAIHighlights,
|
||||
clearAIToolHighlights,
|
||||
clearAICitationHighlights,
|
||||
clearBlastRadius,
|
||||
queryResult,
|
||||
setQueryResult,
|
||||
@@ -1142,6 +1179,7 @@ export const AppStateProvider = ({ children }: { children: ReactNode }) => {
|
||||
runPipelineFromFiles,
|
||||
runQuery,
|
||||
isDatabaseReady,
|
||||
loadServerGraph,
|
||||
// Embedding state and methods
|
||||
embeddingStatus,
|
||||
embeddingProgress,
|
||||
@@ -1156,6 +1194,8 @@ export const AppStateProvider = ({ children }: { children: ReactNode }) => {
|
||||
updateLLMSettings,
|
||||
isSettingsPanelOpen,
|
||||
setSettingsPanelOpen,
|
||||
isHelpDialogBoxOpen,
|
||||
setHelpDialogBoxOpen,
|
||||
isAgentReady,
|
||||
isAgentInitializing,
|
||||
agentError,
|
||||
|
||||
+14
-5
@@ -1,28 +1,37 @@
|
||||
declare module 'kuzu-wasm' {
|
||||
declare module '@ladybugdb/wasm-core' {
|
||||
export function init(): Promise<void>;
|
||||
export class Database {
|
||||
constructor(path: string);
|
||||
constructor(path: string, bufferPoolSize?: number);
|
||||
close(): Promise<void>;
|
||||
}
|
||||
export class Connection {
|
||||
constructor(db: Database);
|
||||
query(cypher: string): Promise<QueryResult>;
|
||||
prepare(cypher: string): Promise<PreparedStatement>;
|
||||
execute(stmt: PreparedStatement, params?: Record<string, any>): Promise<QueryResult>;
|
||||
close(): Promise<void>;
|
||||
}
|
||||
export interface QueryResult {
|
||||
getAll?(): Promise<any[]>;
|
||||
getAllRows?(): Promise<any[]>;
|
||||
getAllObjects?(): Promise<any[]>;
|
||||
hasNext(): Promise<boolean>;
|
||||
getNext(): Promise<any>;
|
||||
}
|
||||
export interface PreparedStatement {
|
||||
isSuccess(): boolean;
|
||||
getErrorMessage(): Promise<string>;
|
||||
close(): Promise<void>;
|
||||
}
|
||||
export const FS: {
|
||||
writeFile(path: string, data: string): Promise<void>;
|
||||
unlink(path: string): Promise<void>;
|
||||
};
|
||||
const kuzu: {
|
||||
const lbug: {
|
||||
init: typeof init;
|
||||
Database: typeof Database;
|
||||
Connection: typeof Connection;
|
||||
FS: typeof FS;
|
||||
};
|
||||
export default kuzu;
|
||||
export default lbug;
|
||||
}
|
||||
|
||||
@@ -1,5 +1,7 @@
|
||||
import * as Comlink from 'comlink';
|
||||
import { runIngestionPipeline, runPipelineFromFiles } from '../core/ingestion/pipeline';
|
||||
import { createKnowledgeGraph } from '../core/graph/graph';
|
||||
import type { GraphNode, GraphRelationship } from '../core/graph/types';
|
||||
import { PipelineProgress, SerializablePipelineResult, serializePipelineResult } from '../types/pipeline';
|
||||
import { FileEntry } from '../services/zip';
|
||||
import {
|
||||
@@ -17,28 +19,83 @@ import { enrichClustersBatch, ClusterMemberInfo, ClusterEnrichment } from '../co
|
||||
import { CommunityNode } from '../core/ingestion/community-processor';
|
||||
import { PipelineResult } from '../types/pipeline';
|
||||
import { buildCodebaseContext, type CodebaseContext } from '../core/llm/context-builder';
|
||||
import {
|
||||
buildBM25Index,
|
||||
searchBM25,
|
||||
isBM25Ready,
|
||||
import {
|
||||
buildBM25Index,
|
||||
searchBM25,
|
||||
isBM25Ready,
|
||||
getBM25Stats,
|
||||
mergeWithRRF,
|
||||
type HybridSearchResult,
|
||||
} from '../core/search';
|
||||
|
||||
// Lazy import for Kuzu to avoid breaking worker if SharedArrayBuffer unavailable
|
||||
let kuzuAdapter: typeof import('../core/kuzu/kuzu-adapter') | null = null;
|
||||
const getKuzuAdapter = async () => {
|
||||
if (!kuzuAdapter) {
|
||||
kuzuAdapter = await import('../core/kuzu/kuzu-adapter');
|
||||
// Lazy import for LadybugDB to avoid breaking worker if SharedArrayBuffer unavailable
|
||||
let lbugAdapter: typeof import('../core/lbug/lbug-adapter') | null = null;
|
||||
const getLbugAdapter = async () => {
|
||||
if (!lbugAdapter) {
|
||||
lbugAdapter = await import('../core/lbug/lbug-adapter');
|
||||
}
|
||||
return kuzuAdapter;
|
||||
return lbugAdapter;
|
||||
};
|
||||
|
||||
// Embedding state
|
||||
let embeddingProgress: EmbeddingProgress | null = null;
|
||||
let isEmbeddingComplete = false;
|
||||
|
||||
/**
|
||||
* Shared post-pipeline logic: store results, build BM25 index, load LadybugDB,
|
||||
* and queue enrichment config. Used by both runPipeline and runPipelineFromFiles.
|
||||
*/
|
||||
const finalizePipeline = async (
|
||||
result: PipelineResult,
|
||||
onProgress: (progress: PipelineProgress) => void,
|
||||
clusteringConfig?: ProviderConfig
|
||||
): Promise<SerializablePipelineResult> => {
|
||||
currentGraphResult = result;
|
||||
|
||||
// Store file contents for grep/read tools (full content, not truncated)
|
||||
storedFileContents = result.fileContents;
|
||||
|
||||
// Build BM25 index for keyword search (instant, ~100ms)
|
||||
const bm25DocCount = buildBM25Index(storedFileContents);
|
||||
if (import.meta.env.DEV) {
|
||||
console.log(`🔍 BM25 index built: ${bm25DocCount} documents`);
|
||||
}
|
||||
|
||||
// Load graph into LadybugDB for querying (optional - gracefully degrades)
|
||||
try {
|
||||
onProgress({
|
||||
phase: 'complete',
|
||||
percent: 98,
|
||||
message: 'Loading into LadybugDB...',
|
||||
stats: {
|
||||
filesProcessed: result.graph.nodeCount,
|
||||
totalFiles: result.graph.nodeCount,
|
||||
nodesCreated: result.graph.nodeCount,
|
||||
},
|
||||
});
|
||||
|
||||
const lbug = await getLbugAdapter();
|
||||
await lbug.loadGraphToLbug(result.graph, result.fileContents);
|
||||
|
||||
if (import.meta.env.DEV) {
|
||||
const stats = await lbug.getLbugStats();
|
||||
console.log('LadybugDB loaded:', stats);
|
||||
console.log('📁 Stored', storedFileContents.size, 'files for grep/read tools');
|
||||
}
|
||||
} catch {
|
||||
// LadybugDB is optional - silently continue without it
|
||||
}
|
||||
|
||||
// Store clustering config for background enrichment (runs after graph loads)
|
||||
if (clusteringConfig) {
|
||||
pendingEnrichmentConfig = clusteringConfig;
|
||||
console.log('📋 Clustering config saved for background enrichment');
|
||||
}
|
||||
|
||||
// Convert to serializable format for transfer back to main thread
|
||||
return serializePipelineResult(result);
|
||||
};
|
||||
|
||||
// File contents state - stores full file contents for grep/read tools
|
||||
let storedFileContents: Map<string, string> = new Map();
|
||||
|
||||
@@ -157,67 +214,70 @@ const workerApi = {
|
||||
onProgress: (progress: PipelineProgress) => void,
|
||||
clusteringConfig?: ProviderConfig
|
||||
): Promise<SerializablePipelineResult> {
|
||||
// Debug logging
|
||||
console.log('🔧 runPipeline called with clusteringConfig:', !!clusteringConfig);
|
||||
// Run the actual pipeline
|
||||
const result = await runIngestionPipeline(file, onProgress);
|
||||
currentGraphResult = result;
|
||||
|
||||
// Store file contents for grep/read tools (full content, not truncated)
|
||||
storedFileContents = result.fileContents;
|
||||
|
||||
// Build BM25 index for keyword search (instant, ~100ms)
|
||||
const bm25DocCount = buildBM25Index(storedFileContents);
|
||||
if (import.meta.env.DEV) {
|
||||
console.log(`🔍 BM25 index built: ${bm25DocCount} documents`);
|
||||
}
|
||||
|
||||
// Load graph into KuzuDB for querying (optional - gracefully degrades)
|
||||
try {
|
||||
onProgress({
|
||||
phase: 'complete',
|
||||
percent: 98,
|
||||
message: 'Loading into KuzuDB...',
|
||||
stats: {
|
||||
filesProcessed: result.graph.nodeCount,
|
||||
totalFiles: result.graph.nodeCount,
|
||||
nodesCreated: result.graph.nodeCount,
|
||||
},
|
||||
});
|
||||
|
||||
const kuzu = await getKuzuAdapter();
|
||||
await kuzu.loadGraphToKuzu(result.graph, result.fileContents);
|
||||
|
||||
if (import.meta.env.DEV) {
|
||||
const stats = await kuzu.getKuzuStats();
|
||||
console.log('KuzuDB loaded:', stats);
|
||||
console.log('📁 Stored', storedFileContents.size, 'files for grep/read tools');
|
||||
}
|
||||
} catch {
|
||||
// KuzuDB is optional - silently continue without it
|
||||
}
|
||||
|
||||
// Store clustering config for background enrichment (runs after graph loads)
|
||||
if (clusteringConfig) {
|
||||
pendingEnrichmentConfig = clusteringConfig;
|
||||
console.log('📋 Clustering config saved for background enrichment');
|
||||
}
|
||||
|
||||
// Convert to serializable format for transfer back to main thread
|
||||
return serializePipelineResult(result);
|
||||
return finalizePipeline(result, onProgress, clusteringConfig);
|
||||
},
|
||||
|
||||
/**
|
||||
* Execute a Cypher query against the KuzuDB database
|
||||
* Load a pre-built graph from the server into LadybugDB.
|
||||
* Called when connecting via server (bypasses the WASM ingestion pipeline).
|
||||
*/
|
||||
async loadServerGraph(
|
||||
nodes: GraphNode[],
|
||||
relationships: GraphRelationship[],
|
||||
fileContents: Record<string, string>
|
||||
): Promise<void> {
|
||||
const graph = createKnowledgeGraph();
|
||||
for (const node of nodes) graph.addNode(node);
|
||||
for (const rel of relationships) graph.addRelationship(rel);
|
||||
|
||||
const fileMap = new Map<string, string>();
|
||||
for (const [path, content] of Object.entries(fileContents)) {
|
||||
fileMap.set(path, content);
|
||||
}
|
||||
|
||||
// Replace (not accumulate) stored file contents for grep/read tools
|
||||
storedFileContents = fileMap;
|
||||
|
||||
// Track graph result for downstream APIs (enrichCommunities, etc.)
|
||||
currentGraphResult = { graph, fileContents: fileMap };
|
||||
isEmbeddingComplete = false;
|
||||
embeddingProgress = null;
|
||||
|
||||
// Load graph into LadybugDB and build BM25 index (optional - gracefully degrades)
|
||||
try {
|
||||
const lbug = await getLbugAdapter();
|
||||
await lbug.loadGraphToLbug(graph, fileMap);
|
||||
|
||||
// Build BM25 index for text search
|
||||
buildBM25Index(fileMap);
|
||||
|
||||
if (import.meta.env.DEV) {
|
||||
const stats = await lbug.getLbugStats();
|
||||
console.log('LadybugDB loaded from server:', stats);
|
||||
console.log('📁 Stored', storedFileContents.size, 'files for grep/read tools');
|
||||
}
|
||||
} catch (err) {
|
||||
if (import.meta.env.DEV) {
|
||||
console.warn('LadybugDB load failed (non-fatal, continuing without it):', err);
|
||||
}
|
||||
// Still build BM25 index even if LadybugDB fails
|
||||
buildBM25Index(fileMap);
|
||||
}
|
||||
},
|
||||
|
||||
/**
|
||||
* Execute a Cypher query against the LadybugDB database
|
||||
* @param cypher - The Cypher query string
|
||||
* @returns Query results as an array of objects
|
||||
*/
|
||||
async runQuery(cypher: string): Promise<any[]> {
|
||||
const kuzu = await getKuzuAdapter();
|
||||
if (!kuzu.isKuzuReady()) {
|
||||
const lbug = await getLbugAdapter();
|
||||
if (!lbug.isLbugReady()) {
|
||||
throw new Error('Database not ready. Please load a repository first.');
|
||||
}
|
||||
return kuzu.executeQuery(cypher);
|
||||
return lbug.executeQuery(cypher);
|
||||
},
|
||||
|
||||
/**
|
||||
@@ -225,8 +285,8 @@ const workerApi = {
|
||||
*/
|
||||
async isReady(): Promise<boolean> {
|
||||
try {
|
||||
const kuzu = await getKuzuAdapter();
|
||||
return kuzu.isKuzuReady();
|
||||
const lbug = await getLbugAdapter();
|
||||
return lbug.isLbugReady();
|
||||
} catch {
|
||||
return false;
|
||||
}
|
||||
@@ -237,8 +297,8 @@ const workerApi = {
|
||||
*/
|
||||
async getStats(): Promise<{ nodes: number; edges: number }> {
|
||||
try {
|
||||
const kuzu = await getKuzuAdapter();
|
||||
return kuzu.getKuzuStats();
|
||||
const lbug = await getLbugAdapter();
|
||||
return lbug.getLbugStats();
|
||||
} catch {
|
||||
return { nodes: 0, edges: 0 };
|
||||
}
|
||||
@@ -263,52 +323,8 @@ const workerApi = {
|
||||
stats: { filesProcessed: 0, totalFiles: files.length, nodesCreated: 0 },
|
||||
});
|
||||
|
||||
// Run the pipeline
|
||||
const result = await runPipelineFromFiles(files, onProgress);
|
||||
currentGraphResult = result;
|
||||
|
||||
// Store file contents for grep/read tools (full content, not truncated)
|
||||
storedFileContents = result.fileContents;
|
||||
|
||||
// Build BM25 index for keyword search (instant, ~100ms)
|
||||
const bm25DocCount = buildBM25Index(storedFileContents);
|
||||
if (import.meta.env.DEV) {
|
||||
console.log(`🔍 BM25 index built: ${bm25DocCount} documents`);
|
||||
}
|
||||
|
||||
// Load graph into KuzuDB for querying (optional - gracefully degrades)
|
||||
try {
|
||||
onProgress({
|
||||
phase: 'complete',
|
||||
percent: 98,
|
||||
message: 'Loading into KuzuDB...',
|
||||
stats: {
|
||||
filesProcessed: result.graph.nodeCount,
|
||||
totalFiles: result.graph.nodeCount,
|
||||
nodesCreated: result.graph.nodeCount,
|
||||
},
|
||||
});
|
||||
|
||||
const kuzu = await getKuzuAdapter();
|
||||
await kuzu.loadGraphToKuzu(result.graph, result.fileContents);
|
||||
|
||||
if (import.meta.env.DEV) {
|
||||
const stats = await kuzu.getKuzuStats();
|
||||
console.log('KuzuDB loaded:', stats);
|
||||
console.log('📁 Stored', storedFileContents.size, 'files for grep/read tools');
|
||||
}
|
||||
} catch {
|
||||
// KuzuDB is optional - silently continue without it
|
||||
}
|
||||
|
||||
// Store clustering config for background enrichment (runs after graph loads)
|
||||
if (clusteringConfig) {
|
||||
pendingEnrichmentConfig = clusteringConfig;
|
||||
console.log('📋 Clustering config saved for background enrichment');
|
||||
}
|
||||
|
||||
// Convert to serializable format for transfer back to main thread
|
||||
return serializePipelineResult(result);
|
||||
return finalizePipeline(result, onProgress, clusteringConfig);
|
||||
},
|
||||
|
||||
// ============================================================
|
||||
@@ -325,8 +341,8 @@ const workerApi = {
|
||||
onProgress: (progress: EmbeddingProgress) => void,
|
||||
forceDevice?: 'webgpu' | 'wasm'
|
||||
): Promise<void> {
|
||||
const kuzu = await getKuzuAdapter();
|
||||
if (!kuzu.isKuzuReady()) {
|
||||
const lbug = await getLbugAdapter();
|
||||
if (!lbug.isLbugReady()) {
|
||||
throw new Error('Database not ready. Please load a repository first.');
|
||||
}
|
||||
|
||||
@@ -343,8 +359,8 @@ const workerApi = {
|
||||
};
|
||||
|
||||
await runEmbeddingPipeline(
|
||||
kuzu.executeQuery,
|
||||
kuzu.executeWithReusedStatement,
|
||||
lbug.executeQuery,
|
||||
lbug.executeWithReusedStatement,
|
||||
progressCallback,
|
||||
forceDevice ? { device: forceDevice } : {}
|
||||
);
|
||||
@@ -400,15 +416,15 @@ const workerApi = {
|
||||
k: number = 10,
|
||||
maxDistance: number = 0.5
|
||||
): Promise<SemanticSearchResult[]> {
|
||||
const kuzu = await getKuzuAdapter();
|
||||
if (!kuzu.isKuzuReady()) {
|
||||
const lbug = await getLbugAdapter();
|
||||
if (!lbug.isLbugReady()) {
|
||||
throw new Error('Database not ready. Please load a repository first.');
|
||||
}
|
||||
if (!isEmbeddingComplete) {
|
||||
throw new Error('Embeddings not ready. Please wait for embedding pipeline to complete.');
|
||||
}
|
||||
|
||||
return doSemanticSearch(kuzu.executeQuery, query, k, maxDistance);
|
||||
return doSemanticSearch(lbug.executeQuery, query, k, maxDistance);
|
||||
},
|
||||
|
||||
/**
|
||||
@@ -424,15 +440,15 @@ const workerApi = {
|
||||
k: number = 5,
|
||||
hops: number = 2
|
||||
): Promise<any[]> {
|
||||
const kuzu = await getKuzuAdapter();
|
||||
if (!kuzu.isKuzuReady()) {
|
||||
const lbug = await getLbugAdapter();
|
||||
if (!lbug.isLbugReady()) {
|
||||
throw new Error('Database not ready. Please load a repository first.');
|
||||
}
|
||||
if (!isEmbeddingComplete) {
|
||||
throw new Error('Embeddings not ready. Please wait for embedding pipeline to complete.');
|
||||
}
|
||||
|
||||
return doSemanticSearchWithContext(kuzu.executeQuery, query, k, hops);
|
||||
return doSemanticSearchWithContext(lbug.executeQuery, query, k, hops);
|
||||
},
|
||||
|
||||
/**
|
||||
@@ -458,9 +474,9 @@ const workerApi = {
|
||||
let semanticResults: SemanticSearchResult[] = [];
|
||||
if (isEmbeddingComplete) {
|
||||
try {
|
||||
const kuzu = await getKuzuAdapter();
|
||||
if (kuzu.isKuzuReady()) {
|
||||
semanticResults = await doSemanticSearch(kuzu.executeQuery, query, k * 3, 0.5);
|
||||
const lbug = await getLbugAdapter();
|
||||
if (lbug.isLbugReady()) {
|
||||
semanticResults = await doSemanticSearch(lbug.executeQuery, query, k * 3, 0.5);
|
||||
}
|
||||
} catch {
|
||||
// Semantic search failed, continue with BM25 only
|
||||
@@ -516,15 +532,15 @@ const workerApi = {
|
||||
},
|
||||
|
||||
/**
|
||||
* Test if KuzuDB supports array parameters in prepared statements
|
||||
* Test if LadybugDB supports array parameters in prepared statements
|
||||
* This is a diagnostic function
|
||||
*/
|
||||
async testArrayParams(): Promise<{ success: boolean; error?: string }> {
|
||||
const kuzu = await getKuzuAdapter();
|
||||
if (!kuzu.isKuzuReady()) {
|
||||
const lbug = await getLbugAdapter();
|
||||
if (!lbug.isLbugReady()) {
|
||||
return { success: false, error: 'Database not ready' };
|
||||
}
|
||||
return kuzu.testArrayParams();
|
||||
return lbug.testArrayParams();
|
||||
},
|
||||
|
||||
// ============================================================
|
||||
@@ -539,8 +555,8 @@ const workerApi = {
|
||||
*/
|
||||
async initializeAgent(config: ProviderConfig, projectName?: string): Promise<{ success: boolean; error?: string }> {
|
||||
try {
|
||||
const kuzu = await getKuzuAdapter();
|
||||
if (!kuzu.isKuzuReady()) {
|
||||
const lbug = await getLbugAdapter();
|
||||
if (!lbug.isLbugReady()) {
|
||||
return { success: false, error: 'Database not ready. Please load a repository first.' };
|
||||
}
|
||||
|
||||
@@ -549,31 +565,31 @@ const workerApi = {
|
||||
if (!isEmbeddingComplete) {
|
||||
throw new Error('Embeddings not ready');
|
||||
}
|
||||
return doSemanticSearch(kuzu.executeQuery, query, k, maxDistance);
|
||||
return doSemanticSearch(lbug.executeQuery, query, k, maxDistance);
|
||||
};
|
||||
|
||||
const semanticSearchWithContextWrapper = async (query: string, k?: number, hops?: number) => {
|
||||
if (!isEmbeddingComplete) {
|
||||
throw new Error('Embeddings not ready');
|
||||
}
|
||||
return doSemanticSearchWithContext(kuzu.executeQuery, query, k, hops);
|
||||
return doSemanticSearchWithContext(lbug.executeQuery, query, k, hops);
|
||||
};
|
||||
|
||||
// Hybrid search wrapper - combines BM25 + semantic
|
||||
const hybridSearchWrapper = async (query: string, k?: number) => {
|
||||
// Get BM25 results (always available after ingestion)
|
||||
const bm25Results = searchBM25(query, (k ?? 10) * 3);
|
||||
|
||||
|
||||
// Get semantic results if embeddings are ready
|
||||
let semanticResults: any[] = [];
|
||||
if (isEmbeddingComplete) {
|
||||
try {
|
||||
semanticResults = await doSemanticSearch(kuzu.executeQuery, query, (k ?? 10) * 3, 0.5);
|
||||
semanticResults = await doSemanticSearch(lbug.executeQuery, query, (k ?? 10) * 3, 0.5);
|
||||
} catch {
|
||||
// Semantic search failed, continue with BM25 only
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
// Merge with RRF
|
||||
return mergeWithRRF(bm25Results, semanticResults, k ?? 10);
|
||||
};
|
||||
@@ -586,7 +602,7 @@ const workerApi = {
|
||||
|
||||
let codebaseContext;
|
||||
try {
|
||||
codebaseContext = await buildCodebaseContext(kuzu.executeQuery, resolvedProjectName);
|
||||
codebaseContext = await buildCodebaseContext(lbug.executeQuery, resolvedProjectName);
|
||||
if (import.meta.env.DEV) {
|
||||
console.log('📊 Codebase context built:', {
|
||||
files: codebaseContext.stats.fileCount,
|
||||
@@ -600,7 +616,7 @@ const workerApi = {
|
||||
|
||||
currentAgent = createGraphRAGAgent(
|
||||
config,
|
||||
kuzu.executeQuery,
|
||||
lbug.executeQuery,
|
||||
semanticSearchWrapper,
|
||||
semanticSearchWithContextWrapper,
|
||||
hybridSearchWrapper,
|
||||
@@ -627,7 +643,7 @@ const workerApi = {
|
||||
|
||||
/**
|
||||
* Initialize the Graph RAG agent in backend mode (HTTP-backed tools).
|
||||
* Uses HTTP wrappers instead of local KuzuDB for all tool queries.
|
||||
* Uses HTTP wrappers instead of local LadybugDB for all tool queries.
|
||||
* @param config - Provider configuration for the LLM
|
||||
* @param backendUrl - Base URL of the gitnexus serve backend
|
||||
* @param repoName - Repository name on the backend
|
||||
@@ -848,9 +864,9 @@ const workerApi = {
|
||||
}
|
||||
});
|
||||
|
||||
// Update KuzuDB with new data
|
||||
// Update LadybugDB with new data
|
||||
try {
|
||||
const kuzu = await getKuzuAdapter();
|
||||
const lbug = await getLbugAdapter();
|
||||
|
||||
onProgress(enrichments.size, enrichments.size); // Done
|
||||
|
||||
@@ -872,11 +888,11 @@ const workerApi = {
|
||||
c.enrichedBy = "llm"
|
||||
`;
|
||||
|
||||
await kuzu.executeQuery(query);
|
||||
await lbug.executeQuery(query);
|
||||
}
|
||||
|
||||
|
||||
} catch (err) {
|
||||
console.error('Failed to update KuzuDB with enrichment:', err);
|
||||
console.error('Failed to update LadybugDB with enrichment:', err);
|
||||
}
|
||||
|
||||
// Convert Map to Record for serialization
|
||||
|
||||
@@ -0,0 +1,58 @@
|
||||
import { describe, expect, it, vi } from 'vitest';
|
||||
|
||||
import type { SerializablePipelineResult } from '../types/pipeline';
|
||||
import { buildPipelineResultFromSerialized, hydrateSerializedServerGraph } from './server-graph-hydration';
|
||||
|
||||
describe('server graph hydration helpers', () => {
|
||||
const serialized: SerializablePipelineResult = {
|
||||
nodes: [
|
||||
{
|
||||
id: 'file:src/foo.ts',
|
||||
label: 'File' as const,
|
||||
properties: {
|
||||
name: 'foo.ts',
|
||||
filePath: 'src/foo.ts',
|
||||
content: 'export function foo() {}',
|
||||
},
|
||||
},
|
||||
{
|
||||
id: 'func:src/foo.ts:foo',
|
||||
label: 'Function' as const,
|
||||
properties: {
|
||||
name: 'foo',
|
||||
filePath: 'src/foo.ts',
|
||||
},
|
||||
},
|
||||
],
|
||||
relationships: [
|
||||
{
|
||||
source: 'file:src/foo.ts',
|
||||
target: 'func:src/foo.ts:foo',
|
||||
type: 'CONTAINS' as const,
|
||||
properties: { type: 'CONTAINS' },
|
||||
},
|
||||
],
|
||||
fileContents: {
|
||||
'src/foo.ts': 'export function foo() {}',
|
||||
},
|
||||
};
|
||||
|
||||
it('rebuilds a graph and file map from serialized server payloads', () => {
|
||||
const result = buildPipelineResultFromSerialized(serialized);
|
||||
|
||||
expect(result.graph.nodeCount).toBe(2);
|
||||
expect(result.graph.relationshipCount).toBe(1);
|
||||
expect(result.fileContents.get('src/foo.ts')).toBe('export function foo() {}');
|
||||
});
|
||||
|
||||
it('delegates the rebuilt result to the worker-side loader', async () => {
|
||||
const loadResult = vi.fn().mockResolvedValue(undefined);
|
||||
|
||||
const result = await hydrateSerializedServerGraph(serialized, loadResult);
|
||||
|
||||
expect(loadResult).toHaveBeenCalledTimes(1);
|
||||
expect(loadResult).toHaveBeenCalledWith(result);
|
||||
expect(result.graph.nodeCount).toBe(2);
|
||||
expect(result.fileContents.size).toBe(1);
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,24 @@
|
||||
import { createKnowledgeGraph } from '../core/graph/graph';
|
||||
import type { PipelineResult, SerializablePipelineResult } from '../types/pipeline';
|
||||
|
||||
export const buildPipelineResultFromSerialized = (
|
||||
serialized: SerializablePipelineResult,
|
||||
): PipelineResult => {
|
||||
const graph = createKnowledgeGraph();
|
||||
serialized.nodes.forEach((node) => graph.addNode(node));
|
||||
serialized.relationships.forEach((relationship) => graph.addRelationship(relationship));
|
||||
|
||||
return {
|
||||
graph,
|
||||
fileContents: new Map(Object.entries(serialized.fileContents)),
|
||||
};
|
||||
};
|
||||
|
||||
export const hydrateSerializedServerGraph = async (
|
||||
serialized: SerializablePipelineResult,
|
||||
loadResult: (result: PipelineResult) => Promise<void>,
|
||||
): Promise<PipelineResult> => {
|
||||
const result = buildPipelineResultFromSerialized(serialized);
|
||||
await loadResult(result);
|
||||
return result;
|
||||
};
|
||||
@@ -0,0 +1,6 @@
|
||||
import { beforeEach } from 'vitest';
|
||||
|
||||
beforeEach(() => {
|
||||
sessionStorage.clear();
|
||||
localStorage.clear();
|
||||
});
|
||||
@@ -0,0 +1,259 @@
|
||||
import { describe, expect, it } from 'vitest';
|
||||
import { createKnowledgeGraph } from '../../src/core/graph/graph';
|
||||
|
||||
// ==========================================================================
|
||||
// PR1 Bug Fix Tests — positive and negative cases
|
||||
// Tests the data structures and logic underlying the 4 bug fixes without
|
||||
// requiring WASM (LadybugDB is skipped in test env via isTestEnv()).
|
||||
// ==========================================================================
|
||||
|
||||
describe('createKnowledgeGraph — data integrity for loadServerGraph', () => {
|
||||
// Positive: nodes and relationships are stored correctly
|
||||
it('stores nodes added via addNode', () => {
|
||||
const graph = createKnowledgeGraph();
|
||||
graph.addNode({
|
||||
id: 'Function:src/index.ts:main',
|
||||
label: 'Function',
|
||||
properties: { name: 'main', filePath: 'src/index.ts', startLine: 1, endLine: 10 },
|
||||
});
|
||||
|
||||
expect(graph.nodes).toHaveLength(1);
|
||||
expect(graph.nodes[0].id).toBe('Function:src/index.ts:main');
|
||||
expect(graph.nodes[0].label).toBe('Function');
|
||||
expect(graph.nodes[0].properties.name).toBe('main');
|
||||
});
|
||||
|
||||
it('stores relationships added via addRelationship', () => {
|
||||
const graph = createKnowledgeGraph();
|
||||
graph.addNode({
|
||||
id: 'Function:a.ts:foo',
|
||||
label: 'Function',
|
||||
properties: { name: 'foo', filePath: 'a.ts', startLine: 1, endLine: 5 },
|
||||
});
|
||||
graph.addNode({
|
||||
id: 'Function:a.ts:bar',
|
||||
label: 'Function',
|
||||
properties: { name: 'bar', filePath: 'a.ts', startLine: 10, endLine: 15 },
|
||||
});
|
||||
graph.addRelationship({
|
||||
sourceId: 'Function:a.ts:foo',
|
||||
targetId: 'Function:a.ts:bar',
|
||||
type: 'CALLS',
|
||||
properties: {},
|
||||
});
|
||||
|
||||
expect(graph.relationships).toHaveLength(1);
|
||||
expect(graph.relationships[0].type).toBe('CALLS');
|
||||
expect(graph.relationships[0].sourceId).toBe('Function:a.ts:foo');
|
||||
expect(graph.relationships[0].targetId).toBe('Function:a.ts:bar');
|
||||
});
|
||||
|
||||
// Positive: deduplication works
|
||||
it('deduplicates nodes with the same id', () => {
|
||||
const graph = createKnowledgeGraph();
|
||||
graph.addNode({
|
||||
id: 'File:src/index.ts',
|
||||
label: 'File',
|
||||
properties: { name: 'index.ts', filePath: 'src/index.ts' },
|
||||
});
|
||||
graph.addNode({
|
||||
id: 'File:src/index.ts',
|
||||
label: 'File',
|
||||
properties: { name: 'index.ts', filePath: 'src/index.ts' },
|
||||
});
|
||||
|
||||
expect(graph.nodes).toHaveLength(1);
|
||||
});
|
||||
|
||||
// Positive: nodeCount reflects actual count
|
||||
it('nodeCount matches number of unique nodes', () => {
|
||||
const graph = createKnowledgeGraph();
|
||||
graph.addNode({
|
||||
id: 'File:a.ts',
|
||||
label: 'File',
|
||||
properties: { name: 'a.ts', filePath: 'a.ts' },
|
||||
});
|
||||
graph.addNode({
|
||||
id: 'File:b.ts',
|
||||
label: 'File',
|
||||
properties: { name: 'b.ts', filePath: 'b.ts' },
|
||||
});
|
||||
|
||||
expect(graph.nodeCount).toBe(2);
|
||||
});
|
||||
|
||||
// Negative: empty graph has zero counts
|
||||
it('empty graph has zero nodes and relationships', () => {
|
||||
const graph = createKnowledgeGraph();
|
||||
expect(graph.nodes).toHaveLength(0);
|
||||
expect(graph.relationships).toHaveLength(0);
|
||||
expect(graph.nodeCount).toBe(0);
|
||||
});
|
||||
|
||||
// Negative: relationships with missing source/target still stored
|
||||
// (validation is upstream, graph is a dumb container)
|
||||
it('stores relationships even with non-existent node IDs', () => {
|
||||
const graph = createKnowledgeGraph();
|
||||
graph.addRelationship({
|
||||
sourceId: 'NonExistent:a',
|
||||
targetId: 'NonExistent:b',
|
||||
type: 'CALLS',
|
||||
properties: {},
|
||||
});
|
||||
|
||||
expect(graph.relationships).toHaveLength(1);
|
||||
});
|
||||
});
|
||||
|
||||
describe('loadServerGraph — data flow validation', () => {
|
||||
// Positive: server data can be reconstructed into a KnowledgeGraph
|
||||
it('reconstructs graph from server node/relationship arrays', () => {
|
||||
const graph = createKnowledgeGraph();
|
||||
const serverNodes = [
|
||||
{ id: 'File:src/app.ts', label: 'File' as const, properties: { name: 'app.ts', filePath: 'src/app.ts' } },
|
||||
{ id: 'Function:src/app.ts:main', label: 'Function' as const, properties: { name: 'main', filePath: 'src/app.ts', startLine: 1, endLine: 20 } },
|
||||
];
|
||||
const serverRels = [
|
||||
{ sourceId: 'File:src/app.ts', targetId: 'Function:src/app.ts:main', type: 'CONTAINS' as const, properties: {} },
|
||||
];
|
||||
|
||||
for (const node of serverNodes) graph.addNode(node);
|
||||
for (const rel of serverRels) graph.addRelationship(rel);
|
||||
|
||||
expect(graph.nodeCount).toBe(2);
|
||||
expect(graph.relationships).toHaveLength(1);
|
||||
expect(graph.relationships[0].type).toBe('CONTAINS');
|
||||
});
|
||||
|
||||
// Positive: file contents map is built correctly from server data
|
||||
it('builds fileContents Map from server object entries', () => {
|
||||
const serverFileContents: Record<string, string> = {
|
||||
'src/index.ts': 'export function main() {}',
|
||||
'src/utils.ts': 'export const helper = () => {}',
|
||||
};
|
||||
|
||||
const fileMap = new Map<string, string>();
|
||||
for (const [path, content] of Object.entries(serverFileContents)) {
|
||||
fileMap.set(path, content);
|
||||
}
|
||||
|
||||
expect(fileMap.size).toBe(2);
|
||||
expect(fileMap.get('src/index.ts')).toBe('export function main() {}');
|
||||
expect(fileMap.get('src/utils.ts')).toContain('helper');
|
||||
});
|
||||
|
||||
// Negative: empty server response produces empty graph
|
||||
it('handles empty server data gracefully', () => {
|
||||
const graph = createKnowledgeGraph();
|
||||
const serverNodes: any[] = [];
|
||||
const serverRels: any[] = [];
|
||||
|
||||
for (const node of serverNodes) graph.addNode(node);
|
||||
for (const rel of serverRels) graph.addRelationship(rel);
|
||||
|
||||
expect(graph.nodeCount).toBe(0);
|
||||
expect(graph.relationships).toHaveLength(0);
|
||||
});
|
||||
|
||||
// Negative: fileContents replaces (not accumulates) on reload
|
||||
it('fileContents map replacement prevents stale data', () => {
|
||||
// Simulates the storedFileContents = fileMap assignment in loadServerGraph
|
||||
let storedFileContents = new Map<string, string>();
|
||||
storedFileContents.set('old-file.ts', 'old content');
|
||||
|
||||
// Second load replaces the map entirely
|
||||
const newFileMap = new Map<string, string>();
|
||||
newFileMap.set('new-file.ts', 'new content');
|
||||
storedFileContents = newFileMap; // assignment, not merge
|
||||
|
||||
expect(storedFileContents.size).toBe(1);
|
||||
expect(storedFileContents.has('old-file.ts')).toBe(false);
|
||||
expect(storedFileContents.get('new-file.ts')).toBe('new content');
|
||||
});
|
||||
});
|
||||
|
||||
describe('BM25 index — argument type validation', () => {
|
||||
// Positive: Map<string, string> has the expected interface for BM25
|
||||
it('Map has entries() and size for BM25 indexing', () => {
|
||||
const fileMap = new Map<string, string>([
|
||||
['src/a.ts', 'function foo() {}'],
|
||||
['src/b.ts', 'function bar() {}'],
|
||||
]);
|
||||
|
||||
expect(fileMap.size).toBe(2);
|
||||
expect(typeof fileMap.entries).toBe('function');
|
||||
|
||||
// Verify iteration works (BM25 iterates entries)
|
||||
const entries = Array.from(fileMap.entries());
|
||||
expect(entries).toHaveLength(2);
|
||||
expect(entries[0][0]).toBe('src/a.ts');
|
||||
});
|
||||
|
||||
// Negative: a KnowledgeGraph object does NOT have entries()
|
||||
// This was the original bug — passing graph instead of fileMap
|
||||
it('KnowledgeGraph does not have entries() (the original bug)', () => {
|
||||
const graph = createKnowledgeGraph();
|
||||
expect((graph as any).entries).toBeUndefined();
|
||||
});
|
||||
});
|
||||
|
||||
describe('highlight clearing — state management', () => {
|
||||
// Positive: Set operations for highlight clearing
|
||||
it('clearing a Set produces an empty set', () => {
|
||||
const highlights = new Set(['node1', 'node2', 'node3']);
|
||||
const cleared = new Set<string>();
|
||||
|
||||
expect(cleared.size).toBe(0);
|
||||
expect(highlights.size).toBe(3);
|
||||
});
|
||||
|
||||
// Positive: multiple highlight sources are independent
|
||||
it('independent highlight sets can be cleared separately', () => {
|
||||
const processHighlights = new Set(['proc_1', 'proc_2']);
|
||||
const aiToolHighlights = new Set(['Function:a.ts:foo']);
|
||||
const aiCitationHighlights = new Set(['File:b.ts']);
|
||||
const blastRadius = new Set(['Function:c.ts:bar']);
|
||||
|
||||
// Simulate "Turn off all highlights" — clear all sets
|
||||
const clearedProcess = new Set<string>();
|
||||
const clearedAITool = new Set<string>();
|
||||
const clearedAICitation = new Set<string>();
|
||||
const clearedBlast = new Set<string>();
|
||||
|
||||
expect(clearedProcess.size).toBe(0);
|
||||
expect(clearedAITool.size).toBe(0);
|
||||
expect(clearedAICitation.size).toBe(0);
|
||||
expect(clearedBlast.size).toBe(0);
|
||||
|
||||
// Original sets unchanged (React state immutability)
|
||||
expect(processHighlights.size).toBe(2);
|
||||
expect(aiToolHighlights.size).toBe(1);
|
||||
});
|
||||
|
||||
// Negative: clearing highlights doesn't affect node selection
|
||||
// (selection is a separate state — verified by checking they're independent)
|
||||
it('highlight state is independent from node selection state', () => {
|
||||
const highlights = new Set(['node1']);
|
||||
let selectedNode: { id: string } | null = { id: 'node1' };
|
||||
|
||||
// Clear highlights but keep selection
|
||||
const clearedHighlights = new Set<string>();
|
||||
expect(clearedHighlights.size).toBe(0);
|
||||
expect(selectedNode).not.toBeNull();
|
||||
|
||||
// Clear selection independently
|
||||
selectedNode = null;
|
||||
expect(selectedNode).toBeNull();
|
||||
});
|
||||
|
||||
// Negative: toggling AI highlights ON should NOT clear user query highlights
|
||||
it('AI highlight toggle does not clear process highlights', () => {
|
||||
const processHighlights = new Set(['proc_1', 'proc_2']);
|
||||
let isAIEnabled = false;
|
||||
|
||||
// Turn AI on — process highlights should survive
|
||||
isAIEnabled = true;
|
||||
expect(processHighlights.size).toBe(2);
|
||||
expect(isAIEnabled).toBe(true);
|
||||
});
|
||||
});
|
||||
@@ -20,5 +20,6 @@
|
||||
},
|
||||
"types": ["vite/client"]
|
||||
},
|
||||
"include": ["src"]
|
||||
"include": ["src"],
|
||||
"exclude": ["src/**/*.test.ts", "src/**/*.test.tsx"]
|
||||
}
|
||||
|
||||
@@ -12,11 +12,11 @@ export default defineConfig({
|
||||
tailwindcss(),
|
||||
wasm(),
|
||||
topLevelAwait(),
|
||||
// Copy kuzu-wasm worker file to assets folder for production
|
||||
// Copy lbug-wasm worker file to assets folder for production
|
||||
viteStaticCopy({
|
||||
targets: [
|
||||
{
|
||||
src: 'node_modules/kuzu-wasm/kuzu_wasm_worker.js',
|
||||
src: 'node_modules/@ladybugdb/wasm-core/lbug_wasm_worker.js',
|
||||
dest: 'assets'
|
||||
}
|
||||
]
|
||||
@@ -35,12 +35,12 @@ export default defineConfig({
|
||||
define: {
|
||||
global: 'globalThis',
|
||||
},
|
||||
// Optimize deps - exclude kuzu-wasm from pre-bundling (it has WASM files)
|
||||
// Optimize deps - exclude lbug-wasm from pre-bundling (it has WASM files)
|
||||
optimizeDeps: {
|
||||
exclude: ['kuzu-wasm'],
|
||||
exclude: ['@ladybugdb/wasm-core'],
|
||||
include: ['buffer'],
|
||||
},
|
||||
// Required for KuzuDB WASM (SharedArrayBuffer needs Cross-Origin Isolation)
|
||||
// Required for LadybugDB WASM (SharedArrayBuffer needs Cross-Origin Isolation)
|
||||
server: {
|
||||
headers: {
|
||||
'Cross-Origin-Opener-Policy': 'same-origin',
|
||||
|
||||
@@ -0,0 +1,17 @@
|
||||
import { defineConfig } from 'vitest/config';
|
||||
import path from 'path';
|
||||
|
||||
export default defineConfig({
|
||||
test: {
|
||||
environment: 'jsdom',
|
||||
globals: true,
|
||||
setupFiles: ['test/setup.ts'],
|
||||
include: ['test/**/*.test.ts'],
|
||||
exclude: ['**/node_modules/**', '**/dist/**'],
|
||||
},
|
||||
resolve: {
|
||||
alias: {
|
||||
'@': path.resolve(__dirname, './src'),
|
||||
},
|
||||
},
|
||||
});
|
||||
@@ -0,0 +1,7 @@
|
||||
{
|
||||
"permissions": {
|
||||
"allow": [
|
||||
"mcp__plugin_claude-mem_mcp-search__get_observations"
|
||||
]
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,12 @@
|
||||
# GitNexus HTTP Embedding Configuration
|
||||
# Copy to .env and uncomment to use a remote OpenAI-compatible endpoint
|
||||
# instead of the local snowflake-arctic-embed-xs model.
|
||||
# When unset, local embeddings are used unchanged.
|
||||
|
||||
# GITNEXUS_EMBEDDING_URL=http://your-server:8080/v1
|
||||
# GITNEXUS_EMBEDDING_MODEL=BAAI/bge-large-en-v1.5
|
||||
# GITNEXUS_EMBEDDING_DIMS=1024
|
||||
# GITNEXUS_EMBEDDING_API_KEY=your-key
|
||||
|
||||
# Works with Infinity, vLLM, TEI, llama.cpp, Ollama, LM Studio, or OpenAI.
|
||||
# See README for details.
|
||||
@@ -0,0 +1,159 @@
|
||||
# Changelog
|
||||
|
||||
All notable changes to GitNexus will be documented in this file.
|
||||
|
||||
## [1.4.8] - 2026-03-23
|
||||
|
||||
### Added
|
||||
- **Type resolution Milestone D — Phases 10–13** consolidated into a single milestone with full integration test coverage across 11 languages (#387)
|
||||
- Phase A/B/C: overload disambiguation via argument literal types, constructor-visible virtual dispatch via `constructorTypeMap`, `parameterTypes` extraction in `extractMethodSignature`
|
||||
- Phase 14 enhancements: single-pass seeding, Tarjan's SCC for cyclic resolution, cross-file return types
|
||||
- Optional parameter arity resolution
|
||||
- Per-language cross-file binding tests and resolver fixes
|
||||
- Store all overloads in `fileIndex` instead of last-write-wins
|
||||
- **Cross-file binding propagation** for multiple languages
|
||||
- **HTTP embedding backend** for self-hosted/remote endpoints with dynamic dimensions, batch guards, and dimension mismatch handling (#395)
|
||||
- **Markdown file indexing** — headings and cross-links as graph nodes (#399)
|
||||
- **MiniMax provider support** (#224)
|
||||
- **Codex MCP and skills support** with CLI setup flow and e2e tests
|
||||
- **HelpPanel UI** — built-in help for the web interface (#465)
|
||||
- **Section node type** registered in `NODE_TABLES` and `NODE_SCHEMA_QUERIES` (#401)
|
||||
- **Community and Process node properties** documented in cypher tool description (#411)
|
||||
- **Server-mode hydration regression tests**
|
||||
- **Pre-commit hooks** via husky for typecheck + unit tests
|
||||
|
||||
### Fixed
|
||||
- **Python import alias resolution** — `import X as Y` now routes module aliases directly to `moduleAliasMap` in import processor (#417, #461)
|
||||
- **Python module-qualified calls** resolved via `moduleAliasMap` (#337)
|
||||
- **Python module-qualified constructor calls** (Issue #337)
|
||||
- **Heritage/MRO edges** now calculate confidence per resolution tier (#412)
|
||||
- **LadybugDB lock** — retry on DB lock with session-safe cleanup (#325)
|
||||
- **CORS** — allow private/LAN network origins (#390)
|
||||
- **Analyze without git** — allow indexing folders without a `.git` directory (#384)
|
||||
- **Web: LadybugDB** — `getAllRows`, `loadServerGraph`, BM25, highlight clearing (#474)
|
||||
- **Server-mode hydration** — await server connect hydration flow (#398, #404)
|
||||
- **Embedding dimensions** — validate on every vector, not just the first; hard-throw on mismatch
|
||||
- **Timeout detection** — always-on dim validation, test hardening
|
||||
- **ONNX CUDA** — prevent uncatchable native crash when CUDA libs present but ORT lacks CUDA provider; clarify linux/x64-only
|
||||
- **CLI** — run codex mcp add via shell on Windows; write tool output to stdout via fd 1
|
||||
- **Stale progress, cross-platform prepare, DEV log** fixes
|
||||
- **Import resolution API** simplified per PR #409 review findings (P0–P3)
|
||||
- **Auto-labeling** — switched from clustering to z-score method; multi-dim aware Mahalanobis threshold
|
||||
- **PR/issue filtering** — fixed prop cutoff issue
|
||||
- **Sequential enrichment queries** + stale data detection
|
||||
- **package-lock.json** synced with `onnxruntime-node ^1.24.0`
|
||||
|
||||
### Changed
|
||||
- **Unified language dispatch** with compile-time exhaustive tables
|
||||
- **Prepare script simplified** — removed `scripts/prepare.cjs`
|
||||
- **Switched from .githooks to husky** for pre-commit hooks
|
||||
- **`@claude` workflow** restricted to maintainers and above via `author_association` check
|
||||
|
||||
### Performance
|
||||
- **O(1) per-chunk synthesis guard** using `boolean[]` instead of Set
|
||||
- **`sizeBefore` optimization** in type resolution
|
||||
- **Token truncation** improvements
|
||||
|
||||
### Chore
|
||||
- Strengthened Python module-import tests, un-skipped match/case, added perf guard
|
||||
- Added positive and negative tests for all 4 bug fixes
|
||||
- E2e tests for stale detection, sequential enrichment, stability (#396)
|
||||
- Integration tests for Milestone D across all 11 languages
|
||||
- `gitnexus-stable-ops` added to community integrations
|
||||
- `.env.example` added for embedding backend configuration
|
||||
|
||||
## [1.4.7] - 2026-03-19
|
||||
|
||||
### Added
|
||||
- **Phase 8 field/property type resolution** — ACCESSES edges with `declaredType` for field reads/writes (#354)
|
||||
- **Phase 9 return-type variable binding** — call-result variable binding across 11 languages (#379)
|
||||
- `extractPendingAssignment` in per-language type extractors captures `let x = getUser()` patterns
|
||||
- Unified fixpoint loop resolves variable types from function return types after initial walk
|
||||
- Field access on call-result variables: `user.name` resolves `name` via return type's class definition
|
||||
- Method-call-result chaining: `user.getProfile().bio` resolves through intermediate return types
|
||||
- 22 new test fixtures covering call-result and method-chain binding across all supported languages
|
||||
- Integration tests added for all 10 language resolver suites
|
||||
- **ACCESSES edge type** with read/write field access tracking (#372)
|
||||
- **Python `enumerate()` for-loop support** with nested tuple patterns (#356)
|
||||
- **MCP tool/resource descriptions** updated to reflect Phase 9 ACCESSES edge semantics and `declaredType` property
|
||||
|
||||
### Fixed
|
||||
- **mcp**: server crashes under parallel tool calls (#326, #349)
|
||||
- **parsing**: undefined error on languages missing from call routers (#364)
|
||||
- **web**: add missing Kotlin entries to `Record<SupportedLanguages>` maps
|
||||
- **rust**: `await` expression unwrapping in `extractPendingAssignment` for async call-result binding
|
||||
- **tests**: update property edge and write access expectations across multiple language tests
|
||||
- **docs**: corrected stale "single-pass" claims in type-resolution-system.md to reflect walk+fixpoint architecture
|
||||
|
||||
### Changed
|
||||
- **Upgrade `@ladybugdb/core` to 0.15.2** and remove segfault workarounds (#374)
|
||||
- **type-resolution-roadmap.md** overhauled — completed phases condensed to summaries, Phases 10–14 added with full engineering specs
|
||||
|
||||
## [1.4.6] - 2026-03-18
|
||||
|
||||
### Added
|
||||
- **Phase 7 type resolution** — return-aware loop inference for call-expression iterables (#341)
|
||||
- `ReturnTypeLookup` interface with `lookupReturnType` / `lookupRawReturnType` split
|
||||
- `ForLoopExtractorContext` context object replacing positional `(node, env)` signature
|
||||
- Call-expression iterable resolution across 8 languages (TS/JS, Java, Kotlin, C#, Go, Rust, Python, PHP)
|
||||
- PHP `$this->property` foreach via `@var` class property scan (Strategy C)
|
||||
- PHP `function_call_expression` and `member_call_expression` foreach paths
|
||||
- `extractElementTypeFromString` as canonical raw-string container unwrapper in `shared.ts`
|
||||
- `extractReturnTypeName` deduplicated from `call-processor.ts` into `shared.ts` (137 lines removed)
|
||||
- `SKIP_SUBTREE_TYPES` performance optimization with documented `template_string` exclusion
|
||||
- `pendingCallResults` infrastructure (dormant — Phase 9 work)
|
||||
|
||||
### Fixed
|
||||
- **impact**: return structured error + partial results instead of crashing (#345)
|
||||
- **impact**: add `HAS_METHOD` and `OVERRIDES` to `VALID_RELATION_TYPES` (#350)
|
||||
- **cli**: write tool output to stdout via fd 1 instead of stderr (#346)
|
||||
- **postinstall**: add permission fix for CLI and hook scripts (#348)
|
||||
- **workflow**: use prefixed temporary branch name for fork PRs to prevent overwriting real branches
|
||||
- **test**: add `--repo` to CLI e2e tool tests for multi-repo environment
|
||||
- **php**: add `declaration_list` type guard on `findClassPropertyElementType` fallback
|
||||
- **docs**: correct `pendingCallResults` description in roadmap and system docs
|
||||
|
||||
### Chore
|
||||
- Add `.worktrees/` to `.gitignore`
|
||||
|
||||
## [1.4.5] - 2026-03-17
|
||||
|
||||
### Added
|
||||
- **Ruby language support** for CLI and web (#111)
|
||||
- **TypeEnvironment API** with constructor inference, self/this/super resolution (#274)
|
||||
- **Return type inference** with doc-comment parsing (JSDoc, PHPDoc, YARD) and per-language type extractors (#284)
|
||||
- **Phase 4 type resolution** — nullable unwrapping, for-loop typing, assignment chain propagation (#310)
|
||||
- **Phase 5 type resolution** — chained calls, pattern matching, class-as-receiver (#315)
|
||||
- **Phase 6 type resolution** — for-loop Tier 1c, pattern matching, container descriptors, 10-language coverage (#318)
|
||||
- Container descriptor table for generic type argument resolution (Map keys vs values)
|
||||
- Method-aware for-loop extractors with integration tests for all languages
|
||||
- Recursive pattern binding (C# `is` patterns, Kotlin `when/is` smart casts)
|
||||
- Class field declaration unwrapping for C#/Java
|
||||
- PHP `$this->property` foreach member access
|
||||
- C++ pointer dereference range-for
|
||||
- Java `this.data.values()` field access patterns
|
||||
- Position-indexed when/is bindings for branch-local narrowing
|
||||
- **Type resolution system documentation** with architecture guide and roadmap
|
||||
- `.gitignore` and `.gitnexusignore` support during file discovery (#231)
|
||||
- Codex MCP configuration documentation in README (#236)
|
||||
- `skipGraphPhases` pipeline option to skip MRO/community/process phases for faster test runs
|
||||
- `hookTimeout: 120000` in vitest config for CI beforeAll hooks
|
||||
|
||||
### Changed
|
||||
- **Migrated from KuzuDB to LadybugDB v0.15** (#275)
|
||||
- Dynamically discover and install agent skills in CLI (#270)
|
||||
|
||||
### Performance
|
||||
- Worker pool threshold — skip worker creation for small repos (<15 files or <512KB total)
|
||||
- AST walk pruning via `SKIP_SUBTREE_TYPES` for leaf-only nodes (string, comment, number literals)
|
||||
- Pre-computed `interestingNodeTypes` set — single Set.has() replaces 3 checks per AST node
|
||||
- `fastStripNullable` — skip full nullable parsing for simple identifiers (90%+ case)
|
||||
- Replace `.children?.find()` with manual for loops in `extractFunctionName` to eliminate array allocations
|
||||
|
||||
### Fixed
|
||||
- Same-directory Python import resolution (#328)
|
||||
- Ruby method-level call resolution, HAS_METHOD edges, and dispatch table (#278)
|
||||
- C++ fixture file casing for case-sensitive CI
|
||||
- Template string incorrectly included in AST pruning set (contains interpolated expressions)
|
||||
|
||||
## [1.4.0] - Previous release
|
||||
@@ -0,0 +1,9 @@
|
||||
FROM node:22-bookworm
|
||||
WORKDIR /app
|
||||
RUN apt-get update && apt-get install -y python3 make g++ && rm -rf /var/lib/apt/lists/*
|
||||
COPY . .
|
||||
RUN npm ci --ignore-scripts \
|
||||
&& node scripts/patch-tree-sitter-swift.cjs \
|
||||
&& (npm rebuild 2>&1 || true) \
|
||||
&& cd node_modules/tree-sitter-kotlin && npx --yes node-gyp rebuild 2>&1
|
||||
CMD ["npx", "vitest", "run", "test/integration", "--reporter=verbose"]
|
||||
+40
-18
@@ -2,7 +2,7 @@
|
||||
|
||||
**Graph-powered code intelligence for AI agents.** Index any codebase into a knowledge graph, then query it via MCP or CLI.
|
||||
|
||||
Works with **Cursor**, **Claude Code**, **Windsurf**, **Cline**, **OpenCode**, and any MCP-compatible tool.
|
||||
Works with **Cursor**, **Claude Code**, **Codex**, **Windsurf**, **Cline**, **OpenCode**, and any MCP-compatible tool.
|
||||
|
||||
[](https://www.npmjs.com/package/gitnexus)
|
||||
[](https://polyformproject.org/licenses/noncommercial/1.0.0/)
|
||||
@@ -34,6 +34,7 @@ To configure MCP for your editor, run `npx gitnexus setup` once — or set it up
|
||||
|--------|-----|--------|---------------------|---------|
|
||||
| **Claude Code** | Yes | Yes | Yes (PreToolUse) | **Full** |
|
||||
| **Cursor** | Yes | Yes | — | MCP + Skills |
|
||||
| **Codex** | Yes | Yes | — | MCP + Skills |
|
||||
| **Windsurf** | Yes | — | — | MCP |
|
||||
| **OpenCode** | Yes | Yes | — | MCP + Skills |
|
||||
|
||||
@@ -55,6 +56,12 @@ If you prefer to configure manually instead of using `gitnexus setup`:
|
||||
claude mcp add gitnexus -- npx -y gitnexus@latest mcp
|
||||
```
|
||||
|
||||
### Codex (full support — MCP + skills)
|
||||
|
||||
```bash
|
||||
codex mcp add gitnexus -- npx -y gitnexus@latest mcp
|
||||
```
|
||||
|
||||
### Cursor / Windsurf
|
||||
|
||||
Add to `~/.cursor/mcp.json` (global — works for all projects):
|
||||
@@ -96,7 +103,7 @@ GitNexus builds a complete knowledge graph of your codebase through a multi-phas
|
||||
5. **Processes** — Traces execution flows from entry points through call chains
|
||||
6. **Search** — Builds hybrid search indexes for fast retrieval
|
||||
|
||||
The result is a **KuzuDB graph database** stored locally in `.gitnexus/` with full-text search and semantic embeddings.
|
||||
The result is a **LadybugDB graph database** stored locally in `.gitnexus/` with full-text search and semantic embeddings.
|
||||
|
||||
## MCP Tools
|
||||
|
||||
@@ -151,32 +158,47 @@ gitnexus wiki [path] # Generate LLM-powered docs from knowledge grap
|
||||
gitnexus wiki --model <model> # Wiki with custom LLM model (default: gpt-4o-mini)
|
||||
```
|
||||
|
||||
## Remote Embeddings
|
||||
|
||||
Set these env vars to use a remote OpenAI-compatible `/v1/embeddings` endpoint instead of the local model:
|
||||
|
||||
```bash
|
||||
export GITNEXUS_EMBEDDING_URL=http://your-server:8080/v1
|
||||
export GITNEXUS_EMBEDDING_MODEL=BAAI/bge-large-en-v1.5
|
||||
export GITNEXUS_EMBEDDING_DIMS=1024 # optional, default 384
|
||||
export GITNEXUS_EMBEDDING_API_KEY=your-key # optional, default: "unused"
|
||||
gitnexus analyze . --embeddings
|
||||
```
|
||||
|
||||
Works with Infinity, vLLM, TEI, llama.cpp, Ollama, LM Studio, or OpenAI. When unset, local embeddings are used unchanged.
|
||||
|
||||
## Multi-Repo Support
|
||||
|
||||
GitNexus supports indexing multiple repositories. Each `gitnexus analyze` registers the repo in a global registry (`~/.gitnexus/registry.json`). The MCP server serves all indexed repos automatically.
|
||||
|
||||
## Supported Languages
|
||||
|
||||
TypeScript, JavaScript, Python, Java, C, C++, C#, Go, Rust, PHP, Kotlin, Swift
|
||||
TypeScript, JavaScript, Python, Java, C, C++, C#, Go, Rust, PHP, Kotlin, Swift, Ruby
|
||||
|
||||
### Language Feature Matrix
|
||||
|
||||
| Language | Imports | Types | Exports | Named Bindings | Config | Frameworks | Entry Points | Heritage |
|
||||
|----------|---------|-------|---------|----------------|--------|------------|-------------|----------|
|
||||
| TypeScript | ✓ | ✓ | ✓ | ✓ | ✓ | ✓ | ✓ | ✓ |
|
||||
| JavaScript | ✓ | ✓ | ✓ | ✓ | ✓ | ✓ | ✓ | ✓ |
|
||||
| Python | ✓ | ✓ | ✓ | ✓ | ✓ | ✓ | ✓ | ✓ |
|
||||
| C# | ✓ | ✓ | ✓ | ✓ | ✓ | ✓ | ✓ | ✓ |
|
||||
| Java | ✓ | ✓ | ✓ | ✓ | — | ✓ | ✓ | ✓ |
|
||||
| Kotlin | ✓ | ✓ | ✓ | ✓ | — | ✓ | ✓ | ✓ |
|
||||
| Go | ✓ | ✓ | ✓ | — | ✓ | ✓ | ✓ | ✓ |
|
||||
| Rust | ✓ | ✓ | ✓ | ✓ | — | ✓ | ✓ | ✓ |
|
||||
| PHP | ✓ | ✓ | ✓ | ✓ | ✓ | ✓ | ✓ | — |
|
||||
| Swift | — | ✓ | ✓ | — | ✓ | ✓ | ✓ | ✓ |
|
||||
| C | — | ✓ | ✓ | — | — | ✓ | ✓ | ✓ |
|
||||
| C++ | — | ✓ | ✓ | — | — | ✓ | ✓ | ✓ |
|
||||
| Language | Imports | Named Bindings | Exports | Heritage | Type Annotations | Constructor Inference | Config | Frameworks | Entry Points |
|
||||
|----------|---------|----------------|---------|----------|-----------------|---------------------|--------|------------|-------------|
|
||||
| TypeScript | ✓ | ✓ | ✓ | ✓ | ✓ | ✓ | ✓ | ✓ | ✓ |
|
||||
| JavaScript | ✓ | ✓ | ✓ | ✓ | — | ✓ | ✓ | ✓ | ✓ |
|
||||
| Python | ✓ | ✓ | ✓ | ✓ | ✓ | ✓ | ✓ | ✓ | ✓ |
|
||||
| Java | ✓ | ✓ | ✓ | ✓ | ✓ | ✓ | — | ✓ | ✓ |
|
||||
| Kotlin | ✓ | ✓ | ✓ | ✓ | ✓ | ✓ | — | ✓ | ✓ |
|
||||
| C# | ✓ | ✓ | ✓ | ✓ | ✓ | ✓ | ✓ | ✓ | ✓ |
|
||||
| Go | ✓ | — | ✓ | ✓ | ✓ | ✓ | ✓ | ✓ | ✓ |
|
||||
| Rust | ✓ | ✓ | ✓ | ✓ | ✓ | ✓ | — | ✓ | ✓ |
|
||||
| PHP | ✓ | ✓ | ✓ | — | ✓ | ✓ | ✓ | ✓ | ✓ |
|
||||
| Ruby | ✓ | — | ✓ | ✓ | — | ✓ | — | ✓ | ✓ |
|
||||
| Swift | — | — | ✓ | ✓ | ✓ | ✓ | ✓ | ✓ | ✓ |
|
||||
| C | — | — | ✓ | — | ✓ | ✓ | — | ✓ | ✓ |
|
||||
| C++ | — | — | ✓ | ✓ | ✓ | ✓ | — | ✓ | ✓ |
|
||||
|
||||
**Imports** — cross-file import resolution · **Types** — type annotation extraction · **Exports** — public/exported symbol detection · **Named Bindings** — `import { X }` tracking · **Config** — language toolchain config parsing (tsconfig, go.mod, etc.) · **Frameworks** — AST-based framework pattern detection · **Entry Points** — entry point scoring heuristics · **Heritage** — class inheritance / interface implementation
|
||||
**Imports** — cross-file import resolution · **Named Bindings** — `import { X as Y }` / re-export tracking · **Exports** — public/exported symbol detection · **Heritage** — class inheritance, interfaces, mixins · **Type Annotations** — explicit type extraction for receiver resolution · **Constructor Inference** — infer receiver type from constructor calls (`self`/`this` resolution included for all languages) · **Config** — language toolchain config parsing (tsconfig, go.mod, etc.) · **Frameworks** — AST-based framework pattern detection · **Entry Points** — entry point scoring heuristics
|
||||
|
||||
## Agent Skills
|
||||
|
||||
|
||||
Regular → Executable
Regular → Executable
Regular → Executable
Generated
+266
-557
File diff suppressed because it is too large
Load Diff
+17
-6
@@ -1,6 +1,6 @@
|
||||
{
|
||||
"name": "gitnexus",
|
||||
"version": "1.4.0",
|
||||
"version": "1.4.8",
|
||||
"description": "Graph-powered code intelligence for AI agents. Index any codebase, query via MCP or CLI.",
|
||||
"author": "Abhigyan Patwari",
|
||||
"license": "PolyForm-Noncommercial-1.0.0",
|
||||
@@ -20,6 +20,7 @@
|
||||
"knowledge-graph",
|
||||
"cursor",
|
||||
"claude",
|
||||
"codex",
|
||||
"ai-agent",
|
||||
"gitnexus",
|
||||
"static-analysis",
|
||||
@@ -39,16 +40,18 @@
|
||||
"scripts": {
|
||||
"build": "tsc",
|
||||
"dev": "tsx watch src/cli/index.ts",
|
||||
"test": "vitest run test/unit",
|
||||
"test": "vitest run",
|
||||
"test:unit": "vitest run test/unit",
|
||||
"test:integration": "vitest run test/integration",
|
||||
"test:all": "vitest run",
|
||||
"test:watch": "vitest",
|
||||
"test:coverage": "vitest run --coverage",
|
||||
"prepare": "npm run build",
|
||||
"postinstall": "node scripts/patch-tree-sitter-swift.cjs"
|
||||
"postinstall": "node scripts/patch-tree-sitter-swift.cjs",
|
||||
"prepack": "npm run build && chmod +x dist/cli/index.js"
|
||||
},
|
||||
"dependencies": {
|
||||
"@huggingface/transformers": "^3.0.0",
|
||||
"@ladybugdb/core": "^0.15.2",
|
||||
"@modelcontextprotocol/sdk": "^1.0.0",
|
||||
"cli-progress": "^3.12.0",
|
||||
"commander": "^12.0.0",
|
||||
@@ -58,9 +61,10 @@
|
||||
"graphology": "^0.25.4",
|
||||
"graphology-indices": "^0.17.0",
|
||||
"graphology-utils": "^2.3.0",
|
||||
"kuzu": "^0.11.3",
|
||||
"ignore": "^7.0.5",
|
||||
"lru-cache": "^11.0.0",
|
||||
"mnemonist": "^0.39.0",
|
||||
"onnxruntime-node": "^1.24.0",
|
||||
"pandemonium": "^2.4.0",
|
||||
"tree-sitter": "^0.21.0",
|
||||
"tree-sitter-c": "^0.21.0",
|
||||
@@ -69,14 +73,15 @@
|
||||
"tree-sitter-go": "^0.21.0",
|
||||
"tree-sitter-java": "^0.21.0",
|
||||
"tree-sitter-javascript": "^0.21.0",
|
||||
"tree-sitter-kotlin": "^0.3.8",
|
||||
"tree-sitter-php": "^0.23.12",
|
||||
"tree-sitter-python": "^0.21.0",
|
||||
"tree-sitter-ruby": "^0.23.1",
|
||||
"tree-sitter-rust": "^0.21.0",
|
||||
"tree-sitter-typescript": "^0.21.0",
|
||||
"uuid": "^13.0.0"
|
||||
},
|
||||
"optionalDependencies": {
|
||||
"tree-sitter-kotlin": "^0.3.8",
|
||||
"tree-sitter-swift": "^0.6.0"
|
||||
},
|
||||
"devDependencies": {
|
||||
@@ -86,10 +91,16 @@
|
||||
"@types/node": "^20.0.0",
|
||||
"@types/uuid": "^10.0.0",
|
||||
"@vitest/coverage-v8": "^4.0.18",
|
||||
"husky": "^9.1.7",
|
||||
"tsx": "^4.0.0",
|
||||
"typescript": "^5.4.5",
|
||||
"vitest": "^4.0.18"
|
||||
},
|
||||
"overrides": {
|
||||
"@huggingface/transformers": {
|
||||
"onnxruntime-node": "$onnxruntime-node"
|
||||
}
|
||||
},
|
||||
"engines": {
|
||||
"node": ">=18.0.0"
|
||||
}
|
||||
|
||||
Regular → Executable
@@ -2,7 +2,7 @@
|
||||
* AI Context Generator
|
||||
*
|
||||
* Creates AGENTS.md and CLAUDE.md with full inline GitNexus context.
|
||||
* AGENTS.md is the standard read by Cursor, Windsurf, OpenCode, Cline, etc.
|
||||
* AGENTS.md is the standard read by Cursor, Windsurf, OpenCode, Codex, Cline, etc.
|
||||
* CLAUDE.md is for Claude Code which only reads that file.
|
||||
*/
|
||||
|
||||
@@ -308,4 +308,3 @@ export async function generateAIContextFiles(
|
||||
|
||||
return { files: createdFiles };
|
||||
}
|
||||
|
||||
|
||||
+97
-52
@@ -9,13 +9,13 @@ import { execFileSync } from 'child_process';
|
||||
import v8 from 'v8';
|
||||
import cliProgress from 'cli-progress';
|
||||
import { runPipelineFromRepo } from '../core/ingestion/pipeline.js';
|
||||
import { initKuzu, loadGraphToKuzu, getKuzuStats, executeQuery, executeWithReusedStatement, closeKuzu, createFTSIndex, loadCachedEmbeddings } from '../core/kuzu/kuzu-adapter.js';
|
||||
import { initLbug, loadGraphToLbug, getLbugStats, executeQuery, executeWithReusedStatement, closeLbug, createFTSIndex, loadCachedEmbeddings } from '../core/lbug/lbug-adapter.js';
|
||||
// Embedding imports are lazy (dynamic import) so onnxruntime-node is never
|
||||
// loaded when embeddings are not requested. This avoids crashes on Node
|
||||
// versions whose ABI is not yet supported by the native binary (#89).
|
||||
// disposeEmbedder intentionally not called — ONNX Runtime segfaults on cleanup (see #38)
|
||||
import { getStoragePaths, saveMeta, loadMeta, addToGitignore, registerRepo, getGlobalRegistryPath } from '../storage/repo-manager.js';
|
||||
import { getCurrentCommit, isGitRepo, getGitRoot } from '../storage/git.js';
|
||||
import { getStoragePaths, saveMeta, loadMeta, addToGitignore, registerRepo, getGlobalRegistryPath, cleanupOldKuzuFiles } from '../storage/repo-manager.js';
|
||||
import { getCurrentCommit, getGitRoot, hasGitDir } from '../storage/git.js';
|
||||
import { generateAIContextFiles } from './ai-context.js';
|
||||
import { generateSkillFiles, type GeneratedSkillInfo } from './skill-gen.js';
|
||||
import fs from 'fs/promises';
|
||||
@@ -48,6 +48,8 @@ export interface AnalyzeOptions {
|
||||
embeddings?: boolean;
|
||||
skills?: boolean;
|
||||
verbose?: boolean;
|
||||
/** Index the folder even when no .git directory is present. */
|
||||
skipGit?: boolean;
|
||||
}
|
||||
|
||||
/** Threshold: auto-skip embeddings for repos with more nodes than this */
|
||||
@@ -63,7 +65,7 @@ const PHASE_LABELS: Record<string, string> = {
|
||||
communities: 'Detecting communities',
|
||||
processes: 'Detecting processes',
|
||||
complete: 'Pipeline complete',
|
||||
kuzu: 'Loading into KuzuDB',
|
||||
lbug: 'Loading into LadybugDB',
|
||||
fts: 'Creating search indexes',
|
||||
embeddings: 'Generating embeddings',
|
||||
done: 'Done',
|
||||
@@ -87,26 +89,50 @@ export const analyzeCommand = async (
|
||||
} else {
|
||||
const gitRoot = getGitRoot(process.cwd());
|
||||
if (!gitRoot) {
|
||||
console.log(' Not inside a git repository\n');
|
||||
process.exitCode = 1;
|
||||
return;
|
||||
if (!options?.skipGit) {
|
||||
console.log(' Not inside a git repository.\n Tip: pass --skip-git to index any folder without a .git directory.\n');
|
||||
process.exitCode = 1;
|
||||
return;
|
||||
}
|
||||
// --skip-git: fall back to cwd as the root
|
||||
repoPath = path.resolve(process.cwd());
|
||||
} else {
|
||||
repoPath = gitRoot;
|
||||
}
|
||||
repoPath = gitRoot;
|
||||
}
|
||||
|
||||
if (!isGitRepo(repoPath)) {
|
||||
console.log(' Not a git repository\n');
|
||||
const repoHasGit = hasGitDir(repoPath);
|
||||
if (!repoHasGit && !options?.skipGit) {
|
||||
console.log(' Not a git repository.\n Tip: pass --skip-git to index any folder without a .git directory.\n');
|
||||
process.exitCode = 1;
|
||||
return;
|
||||
}
|
||||
if (!repoHasGit) {
|
||||
console.log(' Warning: no .git directory found \u2014 commit-tracking and incremental updates disabled.\n');
|
||||
}
|
||||
|
||||
const { storagePath, kuzuPath } = getStoragePaths(repoPath);
|
||||
const currentCommit = getCurrentCommit(repoPath);
|
||||
const { storagePath, lbugPath } = getStoragePaths(repoPath);
|
||||
|
||||
// Clean up stale KuzuDB files from before the LadybugDB migration.
|
||||
// If kuzu existed but lbug doesn't, we're doing a migration re-index — say so.
|
||||
const kuzuResult = await cleanupOldKuzuFiles(storagePath);
|
||||
if (kuzuResult.found && kuzuResult.needsReindex) {
|
||||
console.log(' Migrating from KuzuDB to LadybugDB — rebuilding index...\n');
|
||||
}
|
||||
|
||||
const currentCommit = repoHasGit ? getCurrentCommit(repoPath) : '';
|
||||
const existingMeta = await loadMeta(storagePath);
|
||||
|
||||
if (existingMeta && !options?.force && !options?.skills && existingMeta.lastCommit === currentCommit) {
|
||||
console.log(' Already up to date\n');
|
||||
return;
|
||||
// Non-git folders have currentCommit = '' — always rebuild since we can't detect changes
|
||||
if (currentCommit !== '') {
|
||||
console.log(' Already up to date\n');
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
if (process.env.GITNEXUS_NO_GITIGNORE) {
|
||||
console.log(' GITNEXUS_NO_GITIGNORE is set — skipping .gitignore (still reading .gitnexusignore)\n');
|
||||
}
|
||||
|
||||
// Single progress bar for entire pipeline
|
||||
@@ -130,7 +156,7 @@ export const analyzeCommand = async (
|
||||
aborted = true;
|
||||
bar.stop();
|
||||
console.log('\n Interrupted — cleaning up...');
|
||||
closeKuzu().catch(() => {}).finally(() => process.exit(130));
|
||||
closeLbug().catch(() => {}).finally(() => process.exit(130));
|
||||
};
|
||||
process.on('SIGINT', sigintHandler);
|
||||
|
||||
@@ -180,13 +206,13 @@ export const analyzeCommand = async (
|
||||
if (options?.embeddings && existingMeta && !options?.force) {
|
||||
try {
|
||||
updateBar(0, 'Caching embeddings...');
|
||||
await initKuzu(kuzuPath);
|
||||
await initLbug(lbugPath);
|
||||
const cached = await loadCachedEmbeddings();
|
||||
cachedEmbeddingNodeIds = cached.embeddingNodeIds;
|
||||
cachedEmbeddings = cached.embeddings;
|
||||
await closeKuzu();
|
||||
await closeLbug();
|
||||
} catch {
|
||||
try { await closeKuzu(); } catch {}
|
||||
try { await closeLbug(); } catch {}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -197,25 +223,25 @@ export const analyzeCommand = async (
|
||||
updateBar(scaled, phaseLabel);
|
||||
});
|
||||
|
||||
// ── Phase 2: KuzuDB (60–85%) ──────────────────────────────────────
|
||||
updateBar(60, 'Loading into KuzuDB...');
|
||||
// ── Phase 2: LadybugDB (60–85%) ──────────────────────────────────────
|
||||
updateBar(60, 'Loading into LadybugDB...');
|
||||
|
||||
await closeKuzu();
|
||||
const kuzuFiles = [kuzuPath, `${kuzuPath}.wal`, `${kuzuPath}.lock`];
|
||||
for (const f of kuzuFiles) {
|
||||
await closeLbug();
|
||||
const lbugFiles = [lbugPath, `${lbugPath}.wal`, `${lbugPath}.lock`];
|
||||
for (const f of lbugFiles) {
|
||||
try { await fs.rm(f, { recursive: true, force: true }); } catch {}
|
||||
}
|
||||
|
||||
const t0Kuzu = Date.now();
|
||||
await initKuzu(kuzuPath);
|
||||
let kuzuMsgCount = 0;
|
||||
const kuzuResult = await loadGraphToKuzu(pipelineResult.graph, pipelineResult.repoPath, storagePath, (msg) => {
|
||||
kuzuMsgCount++;
|
||||
const progress = Math.min(84, 60 + Math.round((kuzuMsgCount / (kuzuMsgCount + 10)) * 24));
|
||||
const t0Lbug = Date.now();
|
||||
await initLbug(lbugPath);
|
||||
let lbugMsgCount = 0;
|
||||
const lbugResult = await loadGraphToLbug(pipelineResult.graph, pipelineResult.repoPath, storagePath, (msg) => {
|
||||
lbugMsgCount++;
|
||||
const progress = Math.min(84, 60 + Math.round((lbugMsgCount / (lbugMsgCount + 10)) * 24));
|
||||
updateBar(progress, msg);
|
||||
});
|
||||
const kuzuTime = ((Date.now() - t0Kuzu) / 1000).toFixed(1);
|
||||
const kuzuWarnings = kuzuResult.warnings;
|
||||
const lbugTime = ((Date.now() - t0Lbug) / 1000).toFixed(1);
|
||||
const lbugWarnings = lbugResult.warnings;
|
||||
|
||||
// ── Phase 3: FTS (85–90%) ─────────────────────────────────────────
|
||||
updateBar(85, 'Creating search indexes...');
|
||||
@@ -234,22 +260,32 @@ export const analyzeCommand = async (
|
||||
|
||||
// ── Phase 3.5: Re-insert cached embeddings ────────────────────────
|
||||
if (cachedEmbeddings.length > 0) {
|
||||
updateBar(88, `Restoring ${cachedEmbeddings.length} cached embeddings...`);
|
||||
const EMBED_BATCH = 200;
|
||||
for (let i = 0; i < cachedEmbeddings.length; i += EMBED_BATCH) {
|
||||
const batch = cachedEmbeddings.slice(i, i + EMBED_BATCH);
|
||||
const paramsList = batch.map(e => ({ nodeId: e.nodeId, embedding: e.embedding }));
|
||||
try {
|
||||
await executeWithReusedStatement(
|
||||
`CREATE (e:CodeEmbedding {nodeId: $nodeId, embedding: $embedding})`,
|
||||
paramsList,
|
||||
);
|
||||
} catch { /* some may fail if node was removed, that's fine */ }
|
||||
// Check if cached embedding dimensions match current schema
|
||||
const cachedDims = cachedEmbeddings[0].embedding.length;
|
||||
const { EMBEDDING_DIMS } = await import('../core/lbug/schema.js');
|
||||
if (cachedDims !== EMBEDDING_DIMS) {
|
||||
// Dimensions changed (e.g. switched embedding model) — discard cache and re-embed all
|
||||
console.error(`⚠️ Embedding dimensions changed (${cachedDims}d → ${EMBEDDING_DIMS}d), discarding cache`);
|
||||
cachedEmbeddings = [];
|
||||
cachedEmbeddingNodeIds = new Set();
|
||||
} else {
|
||||
updateBar(88, `Restoring ${cachedEmbeddings.length} cached embeddings...`);
|
||||
const EMBED_BATCH = 200;
|
||||
for (let i = 0; i < cachedEmbeddings.length; i += EMBED_BATCH) {
|
||||
const batch = cachedEmbeddings.slice(i, i + EMBED_BATCH);
|
||||
const paramsList = batch.map(e => ({ nodeId: e.nodeId, embedding: e.embedding }));
|
||||
try {
|
||||
await executeWithReusedStatement(
|
||||
`CREATE (e:CodeEmbedding {nodeId: $nodeId, embedding: $embedding})`,
|
||||
paramsList,
|
||||
);
|
||||
} catch { /* some may fail if node was removed, that's fine */ }
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// ── Phase 4: Embeddings (90–98%) ──────────────────────────────────
|
||||
const stats = await getKuzuStats();
|
||||
const stats = await getLbugStats();
|
||||
let embeddingTime = '0.0';
|
||||
let embeddingSkipped = true;
|
||||
let embeddingSkipReason = 'off (use --embeddings to enable)';
|
||||
@@ -263,7 +299,9 @@ export const analyzeCommand = async (
|
||||
}
|
||||
|
||||
if (!embeddingSkipped) {
|
||||
updateBar(90, 'Loading embedding model...');
|
||||
const { isHttpMode } = await import('../core/embeddings/http-client.js');
|
||||
const httpMode = isHttpMode();
|
||||
updateBar(90, httpMode ? 'Connecting to embedding endpoint...' : 'Loading embedding model...');
|
||||
const t0Emb = Date.now();
|
||||
const { runEmbeddingPipeline } = await import('../core/embeddings/embedding-pipeline.js');
|
||||
await runEmbeddingPipeline(
|
||||
@@ -271,7 +309,9 @@ export const analyzeCommand = async (
|
||||
executeWithReusedStatement,
|
||||
(progress) => {
|
||||
const scaled = 90 + Math.round((progress.percent / 100) * 8);
|
||||
const label = progress.phase === 'loading-model' ? 'Loading embedding model...' : `Embedding ${progress.nodesProcessed || 0}/${progress.totalNodes || '?'}`;
|
||||
const label = progress.phase === 'loading-model'
|
||||
? (httpMode ? 'Connecting to embedding endpoint...' : 'Loading embedding model...')
|
||||
: `Embedding ${progress.nodesProcessed || 0}/${progress.totalNodes || '?'}`;
|
||||
updateBar(scaled, label);
|
||||
},
|
||||
{},
|
||||
@@ -305,7 +345,12 @@ export const analyzeCommand = async (
|
||||
};
|
||||
await saveMeta(storagePath, meta);
|
||||
await registerRepo(repoPath, meta);
|
||||
await addToGitignore(repoPath);
|
||||
// Only attempt to update .gitignore when a .git directory is present.
|
||||
// Use hasGitDir (filesystem check) rather than git CLI subprocess
|
||||
// so we skip correctly for --skip-git folders even if git CLI is available.
|
||||
if (hasGitDir(repoPath)) {
|
||||
await addToGitignore(repoPath);
|
||||
}
|
||||
|
||||
const projectName = path.basename(repoPath);
|
||||
let aggregatedClusterCount = 0;
|
||||
@@ -334,7 +379,7 @@ export const analyzeCommand = async (
|
||||
processes: pipelineResult.processResult?.stats.totalProcesses,
|
||||
}, generatedSkills);
|
||||
|
||||
await closeKuzu();
|
||||
await closeLbug();
|
||||
// Note: we intentionally do NOT call disposeEmbedder() here.
|
||||
// ONNX Runtime's native cleanup segfaults on macOS and some Linux configs.
|
||||
// Since the process exits immediately after, Node.js reclaims everything.
|
||||
@@ -355,7 +400,7 @@ export const analyzeCommand = async (
|
||||
const embeddingsCached = cachedEmbeddings.length > 0;
|
||||
console.log(`\n Repository indexed successfully (${totalTime}s)${embeddingsCached ? ` [${cachedEmbeddings.length} embeddings cached]` : ''}\n`);
|
||||
console.log(` ${stats.nodes.toLocaleString()} nodes | ${stats.edges.toLocaleString()} edges | ${pipelineResult.communityResult?.stats.totalCommunities || 0} clusters | ${pipelineResult.processResult?.stats.totalProcesses || 0} flows`);
|
||||
console.log(` KuzuDB ${kuzuTime}s | FTS ${ftsTime}s | Embeddings ${embeddingSkipped ? embeddingSkipReason : embeddingTime + 's'}`);
|
||||
console.log(` LadybugDB ${lbugTime}s | FTS ${ftsTime}s | Embeddings ${embeddingSkipped ? embeddingSkipReason : embeddingTime + 's'}`);
|
||||
console.log(` ${repoPath}`);
|
||||
|
||||
if (aiContext.files.length > 0) {
|
||||
@@ -363,12 +408,12 @@ export const analyzeCommand = async (
|
||||
}
|
||||
|
||||
// Show a quiet summary if some edge types needed fallback insertion
|
||||
if (kuzuWarnings.length > 0) {
|
||||
const totalFallback = kuzuWarnings.reduce((sum, w) => {
|
||||
if (lbugWarnings.length > 0) {
|
||||
const totalFallback = lbugWarnings.reduce((sum, w) => {
|
||||
const m = w.match(/\((\d+) edges\)/);
|
||||
return sum + (m ? parseInt(m[1]) : 0);
|
||||
}, 0);
|
||||
console.log(` Note: ${totalFallback} edges across ${kuzuWarnings.length} types inserted via fallback (schema will be updated in next release)`);
|
||||
console.log(` Note: ${totalFallback} edges across ${lbugWarnings.length} types inserted via fallback (schema will be updated in next release)`);
|
||||
}
|
||||
|
||||
try {
|
||||
@@ -379,7 +424,7 @@ export const analyzeCommand = async (
|
||||
|
||||
console.log('');
|
||||
|
||||
// KuzuDB's native module holds open handles that prevent Node from exiting.
|
||||
// LadybugDB's native module holds open handles that prevent Node from exiting.
|
||||
// ONNX Runtime also registers native atexit hooks that segfault on some
|
||||
// platforms (#38, #40). Force-exit to ensure clean termination.
|
||||
process.exit(0);
|
||||
|
||||
@@ -23,7 +23,7 @@ export async function augmentCommand(pattern: string): Promise<void> {
|
||||
|
||||
if (result) {
|
||||
// IMPORTANT: Write to stderr, NOT stdout.
|
||||
// KuzuDB's native module captures stdout fd at OS level during init,
|
||||
// LadybugDB's native module captures stdout fd at OS level during init,
|
||||
// which makes stdout permanently broken in subprocess contexts.
|
||||
// stderr is never captured, so it works reliably everywhere.
|
||||
// The hook reads from the subprocess's stderr.
|
||||
|
||||
@@ -1,7 +1,7 @@
|
||||
/**
|
||||
* Eval Server — Lightweight HTTP server for SWE-bench evaluation
|
||||
*
|
||||
* Keeps KuzuDB warm in memory so tool calls from the agent are near-instant.
|
||||
* Keeps LadybugDB warm in memory so tool calls from the agent are near-instant.
|
||||
* Designed to run inside Docker containers during SWE-bench evaluation.
|
||||
*
|
||||
* KEY DESIGN: Returns LLM-friendly text, not raw JSON.
|
||||
@@ -25,6 +25,7 @@
|
||||
*/
|
||||
|
||||
import http from 'http';
|
||||
import { writeSync } from 'node:fs';
|
||||
import { LocalBackend } from '../mcp/local/local-backend.js';
|
||||
|
||||
export interface EvalServerOptions {
|
||||
@@ -142,7 +143,10 @@ export function formatContextResult(result: any): string {
|
||||
}
|
||||
|
||||
export function formatImpactResult(result: any): string {
|
||||
if (result.error) return `Error: ${result.error}`;
|
||||
if (result.error) {
|
||||
const suggestion = result.suggestion ? `\nSuggestion: ${result.suggestion}` : '';
|
||||
return `Error: ${result.error}${suggestion}`;
|
||||
}
|
||||
|
||||
const target = result.target;
|
||||
const direction = result.direction;
|
||||
@@ -155,7 +159,11 @@ export function formatImpactResult(result: any): string {
|
||||
|
||||
const lines: string[] = [];
|
||||
const dirLabel = direction === 'upstream' ? 'depends on this (will break if changed)' : 'this depends on';
|
||||
lines.push(`Blast radius for ${target?.kind || ''} ${target?.name} (${direction}): ${total} symbol(s) ${dirLabel}\n`);
|
||||
lines.push(`Blast radius for ${target?.kind || ''} ${target?.name} (${direction}): ${total} symbol(s) ${dirLabel}`);
|
||||
if (result.partial) {
|
||||
lines.push('⚠️ Partial results — graph traversal was interrupted. Deeper impacts may exist.');
|
||||
}
|
||||
lines.push('');
|
||||
|
||||
const depthLabels: Record<number, string> = {
|
||||
1: 'WILL BREAK (direct)',
|
||||
@@ -401,9 +409,10 @@ export async function evalServerCommand(options?: EvalServerOptions): Promise<vo
|
||||
console.error(` Auto-shutdown after ${idleTimeoutSec}s idle`);
|
||||
}
|
||||
try {
|
||||
process.stdout.write(`GITNEXUS_EVAL_SERVER_READY:${port}\n`);
|
||||
// Use fd 1 directly — LadybugDB captures process.stdout (#324)
|
||||
writeSync(1, `GITNEXUS_EVAL_SERVER_READY:${port}\n`);
|
||||
} catch {
|
||||
// stdout may not be available
|
||||
// stdout may not be available (e.g., broken pipe)
|
||||
}
|
||||
});
|
||||
|
||||
|
||||
@@ -18,16 +18,19 @@ program
|
||||
|
||||
program
|
||||
.command('setup')
|
||||
.description('One-time setup: configure MCP for Cursor, Claude Code, OpenCode')
|
||||
.description('One-time setup: configure MCP for Cursor, Claude Code, OpenCode, Codex')
|
||||
.action(createLazyAction(() => import('./setup.js'), 'setupCommand'));
|
||||
|
||||
|
||||
program
|
||||
.command('analyze [path]')
|
||||
.description('Index a repository (full analysis)')
|
||||
.option('-f, --force', 'Force full re-index even if up to date')
|
||||
.option('--embeddings', 'Enable embedding generation for semantic search (off by default)')
|
||||
.option('--skills', 'Generate repo-specific skill files from detected communities')
|
||||
.option('--skip-git', 'Index a folder without requiring a .git directory')
|
||||
.option('-v, --verbose', 'Enable verbose ingestion warnings (default: false)')
|
||||
.addHelpText('after', '\nEnvironment variables:\n GITNEXUS_NO_GITIGNORE=1 Skip .gitignore parsing (still reads .gitnexusignore)')
|
||||
.action(createLazyAction(() => import('./analyze.js'), 'analyzeCommand'));
|
||||
|
||||
program
|
||||
|
||||
@@ -11,7 +11,7 @@ import { LocalBackend } from '../mcp/local/local-backend.js';
|
||||
|
||||
export const mcpCommand = async () => {
|
||||
// Prevent unhandled errors from crashing the MCP server process.
|
||||
// KuzuDB lock conflicts and transient errors should degrade gracefully.
|
||||
// LadybugDB lock conflicts and transient errors should degrade gracefully.
|
||||
process.on('uncaughtException', (err) => {
|
||||
console.error(`GitNexus MCP: uncaught exception — ${err.message}`);
|
||||
// Process is in an undefined state after uncaughtException — exit after flushing
|
||||
|
||||
+114
-16
@@ -9,11 +9,15 @@
|
||||
import fs from 'fs/promises';
|
||||
import path from 'path';
|
||||
import os from 'os';
|
||||
import { execFile } from 'child_process';
|
||||
import { promisify } from 'util';
|
||||
import { fileURLToPath } from 'url';
|
||||
import { glob } from 'glob';
|
||||
import { getGlobalDir } from '../storage/repo-manager.js';
|
||||
|
||||
const __filename = fileURLToPath(import.meta.url);
|
||||
const __dirname = path.dirname(__filename);
|
||||
const execFileAsync = promisify(execFile);
|
||||
|
||||
interface SetupResult {
|
||||
configured: string[];
|
||||
@@ -238,14 +242,75 @@ async function setupOpenCode(result: SetupResult): Promise<void> {
|
||||
}
|
||||
}
|
||||
|
||||
// ─── Skill Installation ───────────────────────────────────────────
|
||||
/**
|
||||
* Build a TOML section for Codex MCP config (~/.codex/config.toml).
|
||||
*/
|
||||
function getCodexMcpTomlSection(): string {
|
||||
const entry = getMcpEntry();
|
||||
const command = JSON.stringify(entry.command);
|
||||
const args = `[${entry.args.map(arg => JSON.stringify(arg)).join(', ')}]`;
|
||||
return `[mcp_servers.gitnexus]\ncommand = ${command}\nargs = ${args}\n`;
|
||||
}
|
||||
|
||||
const SKILL_NAMES = ['gitnexus-exploring', 'gitnexus-debugging', 'gitnexus-impact-analysis', 'gitnexus-refactoring', 'gitnexus-guide', 'gitnexus-cli'];
|
||||
/**
|
||||
* Append GitNexus MCP server config to Codex's config.toml if missing.
|
||||
*/
|
||||
async function upsertCodexConfigToml(configPath: string): Promise<void> {
|
||||
let existing = '';
|
||||
try {
|
||||
existing = await fs.readFile(configPath, 'utf-8');
|
||||
} catch {
|
||||
existing = '';
|
||||
}
|
||||
|
||||
if (existing.includes('[mcp_servers.gitnexus]')) {
|
||||
return;
|
||||
}
|
||||
|
||||
const section = getCodexMcpTomlSection();
|
||||
const nextContent = existing.trim().length > 0
|
||||
? `${existing.trimEnd()}\n\n${section}`
|
||||
: section;
|
||||
|
||||
await fs.mkdir(path.dirname(configPath), { recursive: true });
|
||||
await fs.writeFile(configPath, `${nextContent.trimEnd()}\n`, 'utf-8');
|
||||
}
|
||||
|
||||
async function setupCodex(result: SetupResult): Promise<void> {
|
||||
const codexDir = path.join(os.homedir(), '.codex');
|
||||
if (!(await dirExists(codexDir))) {
|
||||
result.skipped.push('Codex (not installed)');
|
||||
return;
|
||||
}
|
||||
|
||||
try {
|
||||
const entry = getMcpEntry();
|
||||
await execFileAsync(
|
||||
'codex',
|
||||
['mcp', 'add', 'gitnexus', '--', entry.command, ...entry.args],
|
||||
{ shell: process.platform === 'win32' }
|
||||
);
|
||||
result.configured.push('Codex');
|
||||
return;
|
||||
} catch {
|
||||
// Fallback for environments where `codex` binary isn't on PATH.
|
||||
}
|
||||
|
||||
try {
|
||||
const configPath = path.join(codexDir, 'config.toml');
|
||||
await upsertCodexConfigToml(configPath);
|
||||
result.configured.push('Codex (MCP added to ~/.codex/config.toml)');
|
||||
} catch (err: any) {
|
||||
result.errors.push(`Codex: ${err.message}`);
|
||||
}
|
||||
}
|
||||
|
||||
// ─── Skill Installation ───────────────────────────────────────────
|
||||
|
||||
/**
|
||||
* Install GitNexus skills to a target directory.
|
||||
* Each skill is installed as {targetDir}/gitnexus-{skillName}/SKILL.md
|
||||
* following the Agent Skills standard (both Cursor and Claude Code).
|
||||
* following the Agent Skills standard (Cursor, Claude Code, and Codex).
|
||||
*
|
||||
* Supports two source layouts:
|
||||
* - Flat file: skills/{name}.md → copied as SKILL.md
|
||||
@@ -255,25 +320,38 @@ async function installSkillsTo(targetDir: string): Promise<string[]> {
|
||||
const installed: string[] = [];
|
||||
const skillsRoot = path.join(__dirname, '..', '..', 'skills');
|
||||
|
||||
for (const skillName of SKILL_NAMES) {
|
||||
let flatFiles: string[] = [];
|
||||
let dirSkillFiles: string[] = [];
|
||||
try {
|
||||
[flatFiles, dirSkillFiles] = await Promise.all([
|
||||
glob('*.md', { cwd: skillsRoot }),
|
||||
glob('*/SKILL.md', { cwd: skillsRoot }),
|
||||
]);
|
||||
} catch {
|
||||
return [];
|
||||
}
|
||||
|
||||
const skillSources = new Map<string, { isDirectory: boolean }>();
|
||||
|
||||
for (const relPath of dirSkillFiles) {
|
||||
skillSources.set(path.dirname(relPath), { isDirectory: true });
|
||||
}
|
||||
for (const relPath of flatFiles) {
|
||||
const skillName = path.basename(relPath, '.md');
|
||||
if (!skillSources.has(skillName)) {
|
||||
skillSources.set(skillName, { isDirectory: false });
|
||||
}
|
||||
}
|
||||
|
||||
for (const [skillName, source] of skillSources) {
|
||||
const skillDir = path.join(targetDir, skillName);
|
||||
|
||||
try {
|
||||
// Try directory-based skill first (skills/{name}/SKILL.md)
|
||||
const dirSource = path.join(skillsRoot, skillName);
|
||||
const dirSkillFile = path.join(dirSource, 'SKILL.md');
|
||||
|
||||
let isDirectory = false;
|
||||
try {
|
||||
const stat = await fs.stat(dirSource);
|
||||
isDirectory = stat.isDirectory();
|
||||
} catch { /* not a directory */ }
|
||||
|
||||
if (isDirectory) {
|
||||
if (source.isDirectory) {
|
||||
const dirSource = path.join(skillsRoot, skillName);
|
||||
await copyDirRecursive(dirSource, skillDir);
|
||||
installed.push(skillName);
|
||||
} else {
|
||||
// Fall back to flat file (skills/{name}.md)
|
||||
const flatSource = path.join(skillsRoot, `${skillName}.md`);
|
||||
const content = await fs.readFile(flatSource, 'utf-8');
|
||||
await fs.mkdir(skillDir, { recursive: true });
|
||||
@@ -341,6 +419,24 @@ async function installOpenCodeSkills(result: SetupResult): Promise<void> {
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Install global Codex skills to ~/.agents/skills/gitnexus/
|
||||
*/
|
||||
async function installCodexSkills(result: SetupResult): Promise<void> {
|
||||
const codexDir = path.join(os.homedir(), '.codex');
|
||||
if (!(await dirExists(codexDir))) return;
|
||||
|
||||
const skillsDir = path.join(os.homedir(), '.agents', 'skills');
|
||||
try {
|
||||
const installed = await installSkillsTo(skillsDir);
|
||||
if (installed.length > 0) {
|
||||
result.configured.push(`Codex skills (${installed.length} skills → ~/.agents/skills/)`);
|
||||
}
|
||||
} catch (err: any) {
|
||||
result.errors.push(`Codex skills: ${err.message}`);
|
||||
}
|
||||
}
|
||||
|
||||
// ─── Main command ──────────────────────────────────────────────────
|
||||
|
||||
export const setupCommand = async () => {
|
||||
@@ -363,12 +459,14 @@ export const setupCommand = async () => {
|
||||
await setupCursor(result);
|
||||
await setupClaudeCode(result);
|
||||
await setupOpenCode(result);
|
||||
await setupCodex(result);
|
||||
|
||||
// Install global skills for platforms that support them
|
||||
await installClaudeCodeSkills(result);
|
||||
await installClaudeCodeHooks(result);
|
||||
await installCursorSkills(result);
|
||||
await installOpenCodeSkills(result);
|
||||
await installCodexSkills(result);
|
||||
|
||||
// Print results
|
||||
if (result.configured.length > 0) {
|
||||
|
||||
@@ -4,12 +4,12 @@
|
||||
* Shows the indexing status of the current repository.
|
||||
*/
|
||||
|
||||
import { findRepo } from '../storage/repo-manager.js';
|
||||
import { getCurrentCommit, isGitRepo } from '../storage/git.js';
|
||||
import { findRepo, getStoragePaths, hasKuzuIndex } from '../storage/repo-manager.js';
|
||||
import { getCurrentCommit, isGitRepo, getGitRoot } from '../storage/git.js';
|
||||
|
||||
export const statusCommand = async () => {
|
||||
const cwd = process.cwd();
|
||||
|
||||
|
||||
if (!isGitRepo(cwd)) {
|
||||
console.log('Not a git repository.');
|
||||
return;
|
||||
@@ -17,8 +17,16 @@ export const statusCommand = async () => {
|
||||
|
||||
const repo = await findRepo(cwd);
|
||||
if (!repo) {
|
||||
console.log('Repository not indexed.');
|
||||
console.log('Run: gitnexus analyze');
|
||||
// Check if there's a stale KuzuDB index that needs migration
|
||||
const repoRoot = getGitRoot(cwd) ?? cwd;
|
||||
const { storagePath } = getStoragePaths(repoRoot);
|
||||
if (await hasKuzuIndex(storagePath)) {
|
||||
console.log('Repository has a stale KuzuDB index from a previous version.');
|
||||
console.log('Run: gitnexus analyze (rebuilds the index with LadybugDB)');
|
||||
} else {
|
||||
console.log('Repository not indexed.');
|
||||
console.log('Run: gitnexus analyze');
|
||||
}
|
||||
return;
|
||||
}
|
||||
|
||||
|
||||
+46
-13
@@ -10,10 +10,12 @@
|
||||
* gitnexus impact --target "AuthService" --direction upstream
|
||||
* gitnexus cypher "MATCH (n:Function) RETURN n.name LIMIT 10"
|
||||
*
|
||||
* Note: Output goes to stderr because KuzuDB's native module captures stdout
|
||||
* at the OS level during init. This is consistent with augment.ts.
|
||||
* Note: Output goes to stdout via fs.writeSync(fd 1), bypassing LadybugDB's
|
||||
* native module which captures the Node.js process.stdout stream during init.
|
||||
* See the output() function for details (#324).
|
||||
*/
|
||||
|
||||
import { writeSync } from 'node:fs';
|
||||
import { LocalBackend } from '../mcp/local/local-backend.js';
|
||||
|
||||
let _backend: LocalBackend | null = null;
|
||||
@@ -29,10 +31,29 @@ async function getBackend(): Promise<LocalBackend> {
|
||||
return _backend;
|
||||
}
|
||||
|
||||
/**
|
||||
* Write tool output to stdout using low-level fd write.
|
||||
*
|
||||
* LadybugDB's native module captures Node.js process.stdout during init,
|
||||
* but the underlying OS file descriptor 1 (stdout) remains intact.
|
||||
* By using fs.writeSync(1, ...) we bypass the Node.js stream layer
|
||||
* and write directly to the real stdout fd (#324).
|
||||
*
|
||||
* Falls back to stderr if the fd write fails (e.g., broken pipe).
|
||||
*/
|
||||
function output(data: any): void {
|
||||
const text = typeof data === 'string' ? data : JSON.stringify(data, null, 2);
|
||||
// stderr because KuzuDB captures stdout at OS level
|
||||
process.stderr.write(text + '\n');
|
||||
try {
|
||||
writeSync(1, text + '\n');
|
||||
} catch (err: any) {
|
||||
if (err?.code === 'EPIPE') {
|
||||
// Consumer closed the pipe (e.g., `gitnexus cypher ... | head -1`)
|
||||
// Exit cleanly per Unix convention
|
||||
process.exit(0);
|
||||
}
|
||||
// Fallback: stderr (previous behavior, works on all platforms)
|
||||
process.stderr.write(text + '\n');
|
||||
}
|
||||
}
|
||||
|
||||
export async function queryCommand(queryText: string, options?: {
|
||||
@@ -92,15 +113,27 @@ export async function impactCommand(target: string, options?: {
|
||||
process.exit(1);
|
||||
}
|
||||
|
||||
const backend = await getBackend();
|
||||
const result = await backend.callTool('impact', {
|
||||
target,
|
||||
direction: options?.direction || 'upstream',
|
||||
maxDepth: options?.depth ? parseInt(options.depth) : undefined,
|
||||
includeTests: options?.includeTests ?? false,
|
||||
repo: options?.repo,
|
||||
});
|
||||
output(result);
|
||||
try {
|
||||
const backend = await getBackend();
|
||||
const result = await backend.callTool('impact', {
|
||||
target,
|
||||
direction: options?.direction || 'upstream',
|
||||
maxDepth: options?.depth ? parseInt(options.depth, 10) : undefined,
|
||||
includeTests: options?.includeTests ?? false,
|
||||
repo: options?.repo,
|
||||
});
|
||||
output(result);
|
||||
} catch (err: unknown) {
|
||||
// Belt-and-suspenders: catch infrastructure failures (getBackend, callTool transport)
|
||||
// The backend's impact() already returns structured errors for graph query failures
|
||||
output({
|
||||
error: (err instanceof Error ? err.message : String(err)) || 'Impact analysis failed unexpectedly',
|
||||
target: { name: target },
|
||||
direction: options?.direction || 'upstream',
|
||||
suggestion: 'Try reducing --depth or using gitnexus context <symbol> as a fallback',
|
||||
});
|
||||
process.exit(1);
|
||||
}
|
||||
}
|
||||
|
||||
export async function cypherCommand(query: string, options?: {
|
||||
|
||||
@@ -101,7 +101,7 @@ export const wikiCommand = async (
|
||||
}
|
||||
|
||||
// ── Check for existing index ────────────────────────────────────────
|
||||
const { storagePath, kuzuPath } = getStoragePaths(repoPath);
|
||||
const { storagePath, lbugPath } = getStoragePaths(repoPath);
|
||||
const meta = await loadMeta(storagePath);
|
||||
|
||||
if (!meta) {
|
||||
@@ -247,7 +247,7 @@ export const wikiCommand = async (
|
||||
const generator = new WikiGenerator(
|
||||
repoPath,
|
||||
storagePath,
|
||||
kuzuPath,
|
||||
lbugPath,
|
||||
llmConfig,
|
||||
wikiOptions,
|
||||
(phase, percent, detail) => {
|
||||
|
||||
@@ -1,3 +1,8 @@
|
||||
import ignore, { type Ignore } from 'ignore';
|
||||
import fs from 'fs/promises';
|
||||
import nodePath from 'path';
|
||||
import type { Path } from 'path-scurry';
|
||||
|
||||
const DEFAULT_IGNORE_LIST = new Set([
|
||||
// Version Control
|
||||
'.git',
|
||||
@@ -186,6 +191,10 @@ const IGNORED_FILES = new Set([
|
||||
|
||||
|
||||
|
||||
// NOTE: Negation patterns in .gitnexusignore (e.g. `!vendor/`) cannot override
|
||||
// entries in DEFAULT_IGNORE_LIST — this is intentional. The hardcoded list protects
|
||||
// against indexing directories that are almost never source code (node_modules, .git, etc.).
|
||||
// Users who need to include such directories should remove them from the hardcoded list.
|
||||
export const shouldIgnorePath = (filePath: string): boolean => {
|
||||
const normalizedPath = filePath.replace(/\\/g, '/');
|
||||
const parts = normalizedPath.split('/');
|
||||
@@ -237,3 +246,86 @@ export const shouldIgnorePath = (filePath: string): boolean => {
|
||||
return false;
|
||||
}
|
||||
|
||||
/** Check if a directory name is in the hardcoded ignore list */
|
||||
export const isHardcodedIgnoredDirectory = (name: string): boolean => {
|
||||
return DEFAULT_IGNORE_LIST.has(name);
|
||||
};
|
||||
|
||||
/**
|
||||
* Load .gitignore and .gitnexusignore rules from the repo root.
|
||||
* Returns an `ignore` instance with all patterns, or null if no files found.
|
||||
*/
|
||||
export interface IgnoreOptions {
|
||||
/** Skip .gitignore parsing, only read .gitnexusignore. Defaults to GITNEXUS_NO_GITIGNORE env var. */
|
||||
noGitignore?: boolean;
|
||||
}
|
||||
|
||||
export const loadIgnoreRules = async (
|
||||
repoPath: string,
|
||||
options?: IgnoreOptions
|
||||
): Promise<Ignore | null> => {
|
||||
const ig = ignore();
|
||||
let hasRules = false;
|
||||
|
||||
// Allow users to bypass .gitignore parsing (e.g. when .gitignore accidentally excludes source files)
|
||||
const skipGitignore = options?.noGitignore ?? !!process.env.GITNEXUS_NO_GITIGNORE;
|
||||
const filenames = skipGitignore
|
||||
? ['.gitnexusignore']
|
||||
: ['.gitignore', '.gitnexusignore'];
|
||||
|
||||
for (const filename of filenames) {
|
||||
try {
|
||||
const content = await fs.readFile(nodePath.join(repoPath, filename), 'utf-8');
|
||||
ig.add(content);
|
||||
hasRules = true;
|
||||
} catch (err: unknown) {
|
||||
const code = (err as NodeJS.ErrnoException).code;
|
||||
if (code !== 'ENOENT') {
|
||||
console.warn(` Warning: could not read ${filename}: ${(err as Error).message}`);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
return hasRules ? ig : null;
|
||||
};
|
||||
|
||||
/**
|
||||
* Create a glob-compatible ignore filter combining:
|
||||
* - .gitignore / .gitnexusignore patterns (via `ignore` package)
|
||||
* - Hardcoded DEFAULT_IGNORE_LIST, IGNORED_EXTENSIONS, IGNORED_FILES
|
||||
*
|
||||
* Returns an IgnoreLike object for glob's `ignore` option,
|
||||
* enabling directory-level pruning during traversal.
|
||||
*/
|
||||
export const createIgnoreFilter = async (repoPath: string, options?: IgnoreOptions) => {
|
||||
const ig = await loadIgnoreRules(repoPath, options);
|
||||
|
||||
return {
|
||||
ignored(p: Path): boolean {
|
||||
// path-scurry's Path.relative() returns POSIX paths on all platforms,
|
||||
// which is what the `ignore` package expects. No explicit normalization needed.
|
||||
const rel = p.relative();
|
||||
if (!rel) return false;
|
||||
// Check .gitignore / .gitnexusignore patterns
|
||||
if (ig && ig.ignores(rel)) return true;
|
||||
// Fall back to hardcoded rules
|
||||
return shouldIgnorePath(rel);
|
||||
},
|
||||
childrenIgnored(p: Path): boolean {
|
||||
// Fast path: check directory name against hardcoded list.
|
||||
// Note: dot-directories (.git, .vscode, etc.) are primarily excluded by
|
||||
// glob's `dot: false` option in filesystem-walker.ts. This check is
|
||||
// defense-in-depth — do not remove `dot: false` assuming this covers it.
|
||||
if (DEFAULT_IGNORE_LIST.has(p.name)) return true;
|
||||
// Check against .gitignore / .gitnexusignore patterns.
|
||||
// Test both bare path and path with trailing slash to handle
|
||||
// bare-name patterns (e.g. `local`) and dir-only patterns (e.g. `local/`).
|
||||
if (ig) {
|
||||
const rel = p.relative();
|
||||
if (rel && (ig.ignores(rel) || ig.ignores(rel + '/'))) return true;
|
||||
}
|
||||
return false;
|
||||
},
|
||||
};
|
||||
};
|
||||
|
||||
|
||||
@@ -1,3 +1,33 @@
|
||||
/**
|
||||
* HOW TO ADD A NEW LANGUAGE:
|
||||
*
|
||||
* 1. Add the enum member below (e.g., Scala = 'scala')
|
||||
* 2. Run `tsc --noEmit` — compiler errors guide you to every dispatch table
|
||||
* 3. Use this checklist for each file:
|
||||
*
|
||||
* FILE | WHAT TO ADD | DEFAULT (simple languages)
|
||||
* ----------------------------------|------------------------------------------|---------------------------
|
||||
* tree-sitter-queries.ts | Query string + LANGUAGE_QUERIES entry | (required)
|
||||
* export-detection.ts | ExportChecker function + table entry | (required)
|
||||
* import-resolution.ts | Resolver in importResolvers | resolveStandard(...)
|
||||
* import-resolution.ts | namedBindingExtractors entry | undefined
|
||||
* call-routing.ts | callRouters entry | noRouting
|
||||
* entry-point-scoring.ts | ENTRY_POINT_PATTERNS entry | []
|
||||
* framework-detection.ts | AST_FRAMEWORK_PATTERNS entry | []
|
||||
* type-extractors/<lang>.ts | New file + index.ts import | (required)
|
||||
* resolvers/<lang>.ts | Resolver file (if non-standard) | (only if resolveStandard insufficient)
|
||||
* named-binding-extraction.ts | Extractor (if named imports) | (only if language has named imports)
|
||||
*
|
||||
* 4. Also check these files for language-specific if-checks (no compile-time guard):
|
||||
* - mro-processor.ts (MRO strategy selection)
|
||||
* - heritage-processor.ts (extends/implements handling)
|
||||
* - parse-worker.ts (AST edge cases)
|
||||
* - parsing-processor.ts (node label normalization)
|
||||
*
|
||||
* 5. Add tree-sitter-<lang> to package.json dependencies
|
||||
* 6. Add file extension mapping in utils.ts getLanguageFromFilename()
|
||||
* 7. Run full test suite
|
||||
*/
|
||||
export enum SupportedLanguages {
|
||||
JavaScript = 'javascript',
|
||||
TypeScript = 'typescript',
|
||||
@@ -7,9 +37,9 @@ export enum SupportedLanguages {
|
||||
CPlusPlus = 'cpp',
|
||||
CSharp = 'csharp',
|
||||
Go = 'go',
|
||||
Ruby = 'ruby',
|
||||
Rust = 'rust',
|
||||
PHP = 'php',
|
||||
Kotlin = 'kotlin',
|
||||
// Ruby = 'ruby',
|
||||
Swift = 'swift',
|
||||
}
|
||||
@@ -24,7 +24,7 @@ import { listRegisteredRepos } from '../../storage/repo-manager.js';
|
||||
async function findRepoForCwd(cwd: string): Promise<{
|
||||
name: string;
|
||||
storagePath: string;
|
||||
kuzuPath: string;
|
||||
lbugPath: string;
|
||||
} | null> {
|
||||
try {
|
||||
const entries = await listRegisteredRepos({ validate: true });
|
||||
@@ -66,7 +66,7 @@ async function findRepoForCwd(cwd: string): Promise<{
|
||||
return {
|
||||
name: bestMatch.name,
|
||||
storagePath: bestMatch.storagePath,
|
||||
kuzuPath: path.join(bestMatch.storagePath, 'kuzu'),
|
||||
lbugPath: path.join(bestMatch.storagePath, 'lbug'),
|
||||
};
|
||||
} catch {
|
||||
return null;
|
||||
@@ -92,19 +92,19 @@ export async function augment(pattern: string, cwd?: string): Promise<string> {
|
||||
const repo = await findRepoForCwd(workDir);
|
||||
if (!repo) return '';
|
||||
|
||||
// Lazy-load kuzu adapter (skip unnecessary init)
|
||||
const { initKuzu, executeQuery, isKuzuReady } = await import('../../mcp/core/kuzu-adapter.js');
|
||||
const { searchFTSFromKuzu } = await import('../search/bm25-index.js');
|
||||
|
||||
// Lazy-load lbug adapter (skip unnecessary init)
|
||||
const { initLbug, executeQuery, isLbugReady } = await import('../../mcp/core/lbug-adapter.js');
|
||||
const { searchFTSFromLbug } = await import('../search/bm25-index.js');
|
||||
|
||||
const repoId = repo.name.toLowerCase();
|
||||
|
||||
// Init KuzuDB if not already
|
||||
if (!isKuzuReady(repoId)) {
|
||||
await initKuzu(repoId, repo.kuzuPath);
|
||||
|
||||
// Init LadybugDB if not already
|
||||
if (!isLbugReady(repoId)) {
|
||||
await initLbug(repoId, repo.lbugPath);
|
||||
}
|
||||
|
||||
|
||||
// Step 1: BM25 search (fast, no embeddings)
|
||||
const bm25Results = await searchFTSFromKuzu(pattern, 10, repoId);
|
||||
const bm25Results = await searchFTSFromLbug(pattern, 10, repoId);
|
||||
|
||||
if (bm25Results.length === 0) return '';
|
||||
|
||||
@@ -140,8 +140,90 @@ export async function augment(pattern: string, cwd?: string): Promise<string> {
|
||||
|
||||
if (symbolMatches.length === 0) return '';
|
||||
|
||||
// Step 3: For top matches, fetch callers/callees/processes
|
||||
// Also get cluster cohesion internally for ranking
|
||||
// Step 3: Batch-fetch callers/callees/processes/cohesion for top matches
|
||||
// Uses batched WHERE n.id IN [...] queries instead of per-symbol queries
|
||||
const uniqueSymbols = symbolMatches.slice(0, 5).filter((sym, i, arr) =>
|
||||
arr.findIndex(s => s.nodeId === sym.nodeId) === i
|
||||
);
|
||||
|
||||
if (uniqueSymbols.length === 0) return '';
|
||||
|
||||
const idList = uniqueSymbols.map(s => `'${s.nodeId.replace(/'/g, "''")}'`).join(', ');
|
||||
|
||||
// Batch fetch callers
|
||||
const callersMap = new Map<string, string[]>();
|
||||
try {
|
||||
const rows = await executeQuery(repoId, `
|
||||
MATCH (caller)-[:CodeRelation {type: 'CALLS'}]->(n)
|
||||
WHERE n.id IN [${idList}]
|
||||
RETURN n.id AS targetId, caller.name AS name
|
||||
LIMIT 15
|
||||
`);
|
||||
for (const r of rows) {
|
||||
const tid = r.targetId || r[0];
|
||||
const name = r.name || r[1];
|
||||
if (tid && name) {
|
||||
if (!callersMap.has(tid)) callersMap.set(tid, []);
|
||||
callersMap.get(tid)!.push(name);
|
||||
}
|
||||
}
|
||||
} catch { /* skip */ }
|
||||
|
||||
// Batch fetch callees
|
||||
const calleesMap = new Map<string, string[]>();
|
||||
try {
|
||||
const rows = await executeQuery(repoId, `
|
||||
MATCH (n)-[:CodeRelation {type: 'CALLS'}]->(callee)
|
||||
WHERE n.id IN [${idList}]
|
||||
RETURN n.id AS sourceId, callee.name AS name
|
||||
LIMIT 15
|
||||
`);
|
||||
for (const r of rows) {
|
||||
const sid = r.sourceId || r[0];
|
||||
const name = r.name || r[1];
|
||||
if (sid && name) {
|
||||
if (!calleesMap.has(sid)) calleesMap.set(sid, []);
|
||||
calleesMap.get(sid)!.push(name);
|
||||
}
|
||||
}
|
||||
} catch { /* skip */ }
|
||||
|
||||
// Batch fetch processes
|
||||
const processesMap = new Map<string, string[]>();
|
||||
try {
|
||||
const rows = await executeQuery(repoId, `
|
||||
MATCH (n)-[r:CodeRelation {type: 'STEP_IN_PROCESS'}]->(p:Process)
|
||||
WHERE n.id IN [${idList}]
|
||||
RETURN n.id AS nodeId, p.heuristicLabel AS label, r.step AS step, p.stepCount AS stepCount
|
||||
`);
|
||||
for (const r of rows) {
|
||||
const nid = r.nodeId || r[0];
|
||||
const label = r.label || r[1];
|
||||
const step = r.step || r[2];
|
||||
const stepCount = r.stepCount || r[3];
|
||||
if (nid && label) {
|
||||
if (!processesMap.has(nid)) processesMap.set(nid, []);
|
||||
processesMap.get(nid)!.push(`${label} (step ${step}/${stepCount})`);
|
||||
}
|
||||
}
|
||||
} catch { /* skip */ }
|
||||
|
||||
// Batch fetch cohesion
|
||||
const cohesionMap = new Map<string, number>();
|
||||
try {
|
||||
const rows = await executeQuery(repoId, `
|
||||
MATCH (n)-[:CodeRelation {type: 'MEMBER_OF'}]->(c:Community)
|
||||
WHERE n.id IN [${idList}]
|
||||
RETURN n.id AS nodeId, c.cohesion AS cohesion
|
||||
`);
|
||||
for (const r of rows) {
|
||||
const nid = r.nodeId || r[0];
|
||||
const coh = r.cohesion ?? r[1] ?? 0;
|
||||
if (nid) cohesionMap.set(nid, coh);
|
||||
}
|
||||
} catch { /* skip */ }
|
||||
|
||||
// Assemble enriched results
|
||||
const enriched: Array<{
|
||||
name: string;
|
||||
filePath: string;
|
||||
@@ -150,72 +232,15 @@ export async function augment(pattern: string, cwd?: string): Promise<string> {
|
||||
processes: string[];
|
||||
cohesion: number;
|
||||
}> = [];
|
||||
|
||||
const seen = new Set<string>();
|
||||
|
||||
for (const sym of symbolMatches.slice(0, 5)) {
|
||||
if (seen.has(sym.nodeId)) continue;
|
||||
seen.add(sym.nodeId);
|
||||
|
||||
const escaped = sym.nodeId.replace(/'/g, "''");
|
||||
|
||||
// Callers
|
||||
let callers: string[] = [];
|
||||
try {
|
||||
const rows = await executeQuery(repoId, `
|
||||
MATCH (caller)-[:CodeRelation {type: 'CALLS'}]->(n {id: '${escaped}'})
|
||||
RETURN caller.name AS name
|
||||
LIMIT 3
|
||||
`);
|
||||
callers = rows.map((r: any) => r.name || r[0]).filter(Boolean);
|
||||
} catch { /* skip */ }
|
||||
|
||||
// Callees
|
||||
let callees: string[] = [];
|
||||
try {
|
||||
const rows = await executeQuery(repoId, `
|
||||
MATCH (n {id: '${escaped}'})-[:CodeRelation {type: 'CALLS'}]->(callee)
|
||||
RETURN callee.name AS name
|
||||
LIMIT 3
|
||||
`);
|
||||
callees = rows.map((r: any) => r.name || r[0]).filter(Boolean);
|
||||
} catch { /* skip */ }
|
||||
|
||||
// Processes
|
||||
let processes: string[] = [];
|
||||
try {
|
||||
const rows = await executeQuery(repoId, `
|
||||
MATCH (n {id: '${escaped}'})-[r:CodeRelation {type: 'STEP_IN_PROCESS'}]->(p:Process)
|
||||
RETURN p.heuristicLabel AS label, r.step AS step, p.stepCount AS stepCount
|
||||
`);
|
||||
processes = rows.map((r: any) => {
|
||||
const label = r.label || r[0];
|
||||
const step = r.step || r[1];
|
||||
const stepCount = r.stepCount || r[2];
|
||||
return `${label} (step ${step}/${stepCount})`;
|
||||
}).filter(Boolean);
|
||||
} catch { /* skip */ }
|
||||
|
||||
// Cluster cohesion (internal ranking signal)
|
||||
let cohesion = 0;
|
||||
try {
|
||||
const rows = await executeQuery(repoId, `
|
||||
MATCH (n {id: '${escaped}'})-[:CodeRelation {type: 'MEMBER_OF'}]->(c:Community)
|
||||
RETURN c.cohesion AS cohesion
|
||||
LIMIT 1
|
||||
`);
|
||||
if (rows.length > 0) {
|
||||
cohesion = (rows[0].cohesion ?? rows[0][0]) || 0;
|
||||
}
|
||||
} catch { /* skip */ }
|
||||
|
||||
|
||||
for (const sym of uniqueSymbols) {
|
||||
enriched.push({
|
||||
name: sym.name,
|
||||
filePath: sym.filePath,
|
||||
callers,
|
||||
callees,
|
||||
processes,
|
||||
cohesion,
|
||||
callers: (callersMap.get(sym.nodeId) || []).slice(0, 3),
|
||||
callees: (calleesMap.get(sym.nodeId) || []).slice(0, 3),
|
||||
processes: processesMap.get(sym.nodeId) || [],
|
||||
cohesion: cohesionMap.get(sym.nodeId) || 0,
|
||||
});
|
||||
}
|
||||
|
||||
|
||||
@@ -17,18 +17,54 @@ if (!process.env.ORT_LOG_LEVEL) {
|
||||
import { pipeline, env, type FeatureExtractionPipeline } from '@huggingface/transformers';
|
||||
import { existsSync } from 'fs';
|
||||
import { execFileSync } from 'child_process';
|
||||
import { join } from 'path';
|
||||
import { join, dirname } from 'path';
|
||||
import { createRequire } from 'module';
|
||||
import { DEFAULT_EMBEDDING_CONFIG, type EmbeddingConfig, type ModelProgress } from './types.js';
|
||||
import { isHttpMode, getHttpDimensions, httpEmbed } from './http-client.js';
|
||||
|
||||
/**
|
||||
* Check whether the onnxruntime-node package that @huggingface/transformers
|
||||
* will actually load at runtime ships the CUDA execution provider.
|
||||
*
|
||||
* Critical: we resolve from transformers' own module scope, NOT from ours.
|
||||
* npm may install two copies — a top-level 1.24.x (our dep) and a nested
|
||||
* 1.21.0 (transformers' pinned dep). The guard must inspect whichever copy
|
||||
* transformers.js will dlopen, otherwise the check is meaningless.
|
||||
*/
|
||||
function hasOrtCudaProvider(): boolean {
|
||||
try {
|
||||
const require = createRequire(import.meta.url);
|
||||
// Resolve from @huggingface/transformers' scope so we find the same
|
||||
// onnxruntime-node binary that transformers.js will use at runtime
|
||||
const transformersDir = dirname(require.resolve('@huggingface/transformers/package.json'));
|
||||
const ortRequire = createRequire(join(transformersDir, 'package.json'));
|
||||
const ortPath = dirname(ortRequire.resolve('onnxruntime-node/package.json'));
|
||||
// ORT 1.24.x only ships CUDA binaries for linux/x64 (downloaded from NuGet
|
||||
// at postinstall). arm64 will correctly return false here until ORT adds support.
|
||||
const arch = process.arch;
|
||||
return existsSync(join(ortPath, 'bin', 'napi-v6', 'linux', arch, 'libonnxruntime_providers_cuda.so'));
|
||||
} catch {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Check whether CUDA libraries are actually available on this system.
|
||||
* ONNX Runtime's native layer crashes (uncatchable) if we attempt CUDA
|
||||
* without the required shared libraries, so we probe first.
|
||||
*
|
||||
* Checks the dynamic linker cache (ldconfig) which covers all architectures
|
||||
* and install paths, then falls back to CUDA_PATH / LD_LIBRARY_PATH env vars.
|
||||
* Checks both:
|
||||
* 1. That system CUDA libraries (libcublasLt) are present
|
||||
* 2. That onnxruntime-node ships the CUDA execution provider binary
|
||||
*
|
||||
* Both conditions must be true — system CUDA libs alone are not enough
|
||||
* if onnxruntime-node is a CPU-only build (versions < 1.24.0).
|
||||
*/
|
||||
function isCudaAvailable(): boolean {
|
||||
// First, verify onnxruntime-node has the CUDA provider binary.
|
||||
// Without this, requesting CUDA causes an uncatchable native crash.
|
||||
if (!hasOrtCudaProvider()) return false;
|
||||
|
||||
// Primary: query the dynamic linker cache — covers all architectures,
|
||||
// distro layouts, and custom install paths registered with ldconfig
|
||||
try {
|
||||
@@ -83,6 +119,13 @@ export const initEmbedder = async (
|
||||
config: Partial<EmbeddingConfig> = {},
|
||||
forceDevice?: 'dml' | 'cuda' | 'cpu' | 'wasm'
|
||||
): Promise<FeatureExtractionPipeline> => {
|
||||
if (isHttpMode()) {
|
||||
throw new Error(
|
||||
'initEmbedder() should not be called in HTTP mode. ' +
|
||||
'Use embedText()/embedBatch() which handle HTTP transparently.'
|
||||
);
|
||||
}
|
||||
|
||||
// Return existing instance if available
|
||||
if (embedderInstance) {
|
||||
return embedderInstance;
|
||||
@@ -195,13 +238,27 @@ export const initEmbedder = async (
|
||||
* Check if the embedder is initialized and ready
|
||||
*/
|
||||
export const isEmbedderReady = (): boolean => {
|
||||
return embedderInstance !== null;
|
||||
return isHttpMode() || embedderInstance !== null;
|
||||
};
|
||||
|
||||
/**
|
||||
* Get the effective embedding dimensions.
|
||||
* In HTTP mode, uses GITNEXUS_EMBEDDING_DIMS if set, otherwise the default.
|
||||
*/
|
||||
export const getEmbeddingDimensions = (): number => {
|
||||
if (isHttpMode()) {
|
||||
return getHttpDimensions() ?? DEFAULT_EMBEDDING_CONFIG.dimensions;
|
||||
}
|
||||
return DEFAULT_EMBEDDING_CONFIG.dimensions;
|
||||
};
|
||||
|
||||
/**
|
||||
* Get the embedder instance (throws if not initialized)
|
||||
*/
|
||||
export const getEmbedder = (): FeatureExtractionPipeline => {
|
||||
if (isHttpMode()) {
|
||||
throw new Error('getEmbedder() is not available in HTTP embedding mode. Use embedText()/embedBatch() instead.');
|
||||
}
|
||||
if (!embedderInstance) {
|
||||
throw new Error('Embedder not initialized. Call initEmbedder() first.');
|
||||
}
|
||||
@@ -212,9 +269,14 @@ export const getEmbedder = (): FeatureExtractionPipeline => {
|
||||
* Embed a single text string
|
||||
*
|
||||
* @param text - Text to embed
|
||||
* @returns Float32Array of embedding vector (384 dimensions)
|
||||
* @returns Float32Array of embedding vector
|
||||
*/
|
||||
export const embedText = async (text: string): Promise<Float32Array> => {
|
||||
if (isHttpMode()) {
|
||||
const [vec] = await httpEmbed([text]);
|
||||
return vec;
|
||||
}
|
||||
|
||||
const embedder = getEmbedder();
|
||||
|
||||
const result = await embedder(text, {
|
||||
@@ -238,6 +300,10 @@ export const embedBatch = async (texts: string[]): Promise<Float32Array[]> => {
|
||||
return [];
|
||||
}
|
||||
|
||||
if (isHttpMode()) {
|
||||
return httpEmbed(texts);
|
||||
}
|
||||
|
||||
const embedder = getEmbedder();
|
||||
|
||||
// Process batch
|
||||
@@ -262,7 +328,7 @@ export const embedBatch = async (texts: string[]): Promise<Float32Array[]> => {
|
||||
};
|
||||
|
||||
/**
|
||||
* Convert Float32Array to regular number array (for KuzuDB storage)
|
||||
* Convert Float32Array to regular number array (for LadybugDB storage)
|
||||
*/
|
||||
export const embeddingToArray = (embedding: Float32Array): number[] => {
|
||||
return Array.from(embedding);
|
||||
|
||||
@@ -2,10 +2,10 @@
|
||||
* Embedding Pipeline Module
|
||||
*
|
||||
* Orchestrates the background embedding process:
|
||||
* 1. Query embeddable nodes from KuzuDB
|
||||
* 1. Query embeddable nodes from LadybugDB
|
||||
* 2. Generate text representations
|
||||
* 3. Batch embed using transformers.js
|
||||
* 4. Update KuzuDB with embeddings
|
||||
* 4. Update LadybugDB with embeddings
|
||||
* 5. Create vector index for semantic search
|
||||
*/
|
||||
|
||||
@@ -29,7 +29,7 @@ const isDev = process.env.NODE_ENV === 'development';
|
||||
export type EmbeddingProgressCallback = (progress: EmbeddingProgress) => void;
|
||||
|
||||
/**
|
||||
* Query all embeddable nodes from KuzuDB
|
||||
* Query all embeddable nodes from LadybugDB
|
||||
* Uses table-specific queries (File has different schema than code elements)
|
||||
*/
|
||||
const queryEmbeddableNodes = async (
|
||||
@@ -104,9 +104,23 @@ const batchInsertEmbeddings = async (
|
||||
* Create the vector index for semantic search
|
||||
* Now indexes the separate CodeEmbedding table
|
||||
*/
|
||||
let vectorExtensionLoaded = false;
|
||||
|
||||
const createVectorIndex = async (
|
||||
executeQuery: (cypher: string) => Promise<any[]>
|
||||
): Promise<void> => {
|
||||
// LadybugDB v0.15+ requires explicit VECTOR extension loading (once per session)
|
||||
if (!vectorExtensionLoaded) {
|
||||
try {
|
||||
await executeQuery('INSTALL VECTOR');
|
||||
await executeQuery('LOAD EXTENSION VECTOR');
|
||||
vectorExtensionLoaded = true;
|
||||
} catch {
|
||||
// Extension may already be loaded — CREATE_VECTOR_INDEX will fail clearly if not
|
||||
vectorExtensionLoaded = true;
|
||||
}
|
||||
}
|
||||
|
||||
const cypher = `
|
||||
CALL CREATE_VECTOR_INDEX('CodeEmbedding', 'code_embedding_idx', 'embedding', metric := 'cosine')
|
||||
`;
|
||||
@@ -124,7 +138,7 @@ const createVectorIndex = async (
|
||||
/**
|
||||
* Run the embedding pipeline
|
||||
*
|
||||
* @param executeQuery - Function to execute Cypher queries against KuzuDB
|
||||
* @param executeQuery - Function to execute Cypher queries against LadybugDB
|
||||
* @param executeWithReusedStatement - Function to execute with reused prepared statement
|
||||
* @param onProgress - Callback for progress updates
|
||||
* @param config - Optional configuration override
|
||||
@@ -147,14 +161,16 @@ export const runEmbeddingPipeline = async (
|
||||
modelDownloadPercent: 0,
|
||||
});
|
||||
|
||||
await initEmbedder((modelProgress: ModelProgress) => {
|
||||
const downloadPercent = modelProgress.progress ?? 0;
|
||||
onProgress({
|
||||
phase: 'loading-model',
|
||||
percent: Math.round(downloadPercent * 0.2),
|
||||
modelDownloadPercent: downloadPercent,
|
||||
});
|
||||
}, finalConfig);
|
||||
if (!isEmbedderReady()) {
|
||||
await initEmbedder((modelProgress: ModelProgress) => {
|
||||
const downloadPercent = modelProgress.progress ?? 0;
|
||||
onProgress({
|
||||
phase: 'loading-model',
|
||||
percent: Math.round(downloadPercent * 0.2),
|
||||
modelDownloadPercent: downloadPercent,
|
||||
});
|
||||
}, finalConfig);
|
||||
}
|
||||
|
||||
onProgress({
|
||||
phase: 'loading-model',
|
||||
@@ -219,7 +235,7 @@ export const runEmbeddingPipeline = async (
|
||||
// Embed the batch
|
||||
const embeddings = await embedBatch(texts);
|
||||
|
||||
// Update KuzuDB with embeddings
|
||||
// Update LadybugDB with embeddings
|
||||
const updates = batch.map((node, i) => ({
|
||||
id: node.id,
|
||||
embedding: embeddingToArray(embeddings[i]),
|
||||
@@ -312,7 +328,7 @@ export const semanticSearch = async (
|
||||
// Query the vector index on CodeEmbedding to get nodeIds and distances
|
||||
const vectorQuery = `
|
||||
CALL QUERY_VECTOR_INDEX('CodeEmbedding', 'code_embedding_idx',
|
||||
CAST(${queryVecStr} AS FLOAT[384]), ${k})
|
||||
CAST(${queryVecStr} AS FLOAT[${queryVec.length}]), ${k})
|
||||
YIELD node AS emb, distance
|
||||
WITH emb, distance
|
||||
WHERE distance < ${maxDistance}
|
||||
@@ -326,51 +342,64 @@ export const semanticSearch = async (
|
||||
return [];
|
||||
}
|
||||
|
||||
// Get metadata for each result by querying each node table
|
||||
const results: SemanticSearchResult[] = [];
|
||||
|
||||
// Group results by label for batched metadata queries
|
||||
const byLabel = new Map<string, Array<{ nodeId: string; distance: number }>>();
|
||||
for (const embRow of embResults) {
|
||||
const nodeId = embRow.nodeId ?? embRow[0];
|
||||
const distance = embRow.distance ?? embRow[1];
|
||||
|
||||
// Extract label from node ID (format: Label:path:name)
|
||||
const labelEndIdx = nodeId.indexOf(':');
|
||||
const label = labelEndIdx > 0 ? nodeId.substring(0, labelEndIdx) : 'Unknown';
|
||||
|
||||
// Query the specific table for this node
|
||||
// File nodes don't have startLine/endLine
|
||||
if (!byLabel.has(label)) byLabel.set(label, []);
|
||||
byLabel.get(label)!.push({ nodeId, distance });
|
||||
}
|
||||
|
||||
// Batch-fetch metadata per label
|
||||
const results: SemanticSearchResult[] = [];
|
||||
|
||||
for (const [label, items] of byLabel) {
|
||||
const idList = items.map(i => `'${i.nodeId.replace(/'/g, "''")}'`).join(', ');
|
||||
try {
|
||||
let nodeQuery: string;
|
||||
if (label === 'File') {
|
||||
nodeQuery = `
|
||||
MATCH (n:File {id: '${nodeId.replace(/'/g, "''")}'})
|
||||
RETURN n.name AS name, n.filePath AS filePath
|
||||
MATCH (n:File) WHERE n.id IN [${idList}]
|
||||
RETURN n.id AS id, n.name AS name, n.filePath AS filePath
|
||||
`;
|
||||
} else {
|
||||
nodeQuery = `
|
||||
MATCH (n:${label} {id: '${nodeId.replace(/'/g, "''")}'})
|
||||
RETURN n.name AS name, n.filePath AS filePath,
|
||||
MATCH (n:${label}) WHERE n.id IN [${idList}]
|
||||
RETURN n.id AS id, n.name AS name, n.filePath AS filePath,
|
||||
n.startLine AS startLine, n.endLine AS endLine
|
||||
`;
|
||||
}
|
||||
const nodeRows = await executeQuery(nodeQuery);
|
||||
if (nodeRows.length > 0) {
|
||||
const nodeRow = nodeRows[0];
|
||||
results.push({
|
||||
nodeId,
|
||||
name: nodeRow.name ?? nodeRow[0] ?? '',
|
||||
label,
|
||||
filePath: nodeRow.filePath ?? nodeRow[1] ?? '',
|
||||
distance,
|
||||
startLine: label !== 'File' ? (nodeRow.startLine ?? nodeRow[2]) : undefined,
|
||||
endLine: label !== 'File' ? (nodeRow.endLine ?? nodeRow[3]) : undefined,
|
||||
});
|
||||
const rowMap = new Map<string, any>();
|
||||
for (const row of nodeRows) {
|
||||
const id = row.id ?? row[0];
|
||||
rowMap.set(id, row);
|
||||
}
|
||||
for (const item of items) {
|
||||
const nodeRow = rowMap.get(item.nodeId);
|
||||
if (nodeRow) {
|
||||
results.push({
|
||||
nodeId: item.nodeId,
|
||||
name: nodeRow.name ?? nodeRow[1] ?? '',
|
||||
label,
|
||||
filePath: nodeRow.filePath ?? nodeRow[2] ?? '',
|
||||
distance: item.distance,
|
||||
startLine: label !== 'File' ? (nodeRow.startLine ?? nodeRow[3]) : undefined,
|
||||
endLine: label !== 'File' ? (nodeRow.endLine ?? nodeRow[4]) : undefined,
|
||||
});
|
||||
}
|
||||
}
|
||||
} catch {
|
||||
// Table might not exist, skip
|
||||
}
|
||||
}
|
||||
|
||||
// Re-sort by distance since batch queries may have mixed order
|
||||
results.sort((a, b) => a.distance - b.distance);
|
||||
|
||||
return results;
|
||||
};
|
||||
|
||||
|
||||
@@ -0,0 +1,226 @@
|
||||
/**
|
||||
* HTTP Embedding Client
|
||||
*
|
||||
* Shared fetch+retry logic for OpenAI-compatible /v1/embeddings endpoints.
|
||||
* Imported by both the core embedder (batch) and MCP embedder (query).
|
||||
*/
|
||||
|
||||
const HTTP_TIMEOUT_MS = 30_000;
|
||||
const HTTP_MAX_RETRIES = 2;
|
||||
const HTTP_RETRY_BACKOFF_MS = 1_000;
|
||||
const HTTP_BATCH_SIZE = 64;
|
||||
const DEFAULT_DIMS = 384;
|
||||
|
||||
interface HttpConfig {
|
||||
baseUrl: string;
|
||||
model: string;
|
||||
apiKey: string;
|
||||
dimensions?: number;
|
||||
}
|
||||
|
||||
/**
|
||||
* Build config from the current process.env snapshot.
|
||||
* Returns null when GITNEXUS_EMBEDDING_URL + GITNEXUS_EMBEDDING_MODEL are unset.
|
||||
* Not cached — env vars are read fresh so late configuration takes effect.
|
||||
*/
|
||||
const readConfig = (): HttpConfig | null => {
|
||||
const baseUrl = process.env.GITNEXUS_EMBEDDING_URL;
|
||||
const model = process.env.GITNEXUS_EMBEDDING_MODEL;
|
||||
if (!baseUrl || !model) return null;
|
||||
|
||||
const rawDims = process.env.GITNEXUS_EMBEDDING_DIMS;
|
||||
let dimensions: number | undefined;
|
||||
if (rawDims !== undefined) {
|
||||
const parsed = parseInt(rawDims, 10);
|
||||
if (Number.isNaN(parsed) || parsed <= 0) {
|
||||
throw new Error(
|
||||
`GITNEXUS_EMBEDDING_DIMS must be a positive integer, got "${rawDims}"`,
|
||||
);
|
||||
}
|
||||
dimensions = parsed;
|
||||
}
|
||||
|
||||
return {
|
||||
baseUrl: baseUrl.replace(/\/+$/, ''),
|
||||
model,
|
||||
apiKey: process.env.GITNEXUS_EMBEDDING_API_KEY ?? 'unused',
|
||||
dimensions,
|
||||
};
|
||||
};
|
||||
|
||||
/**
|
||||
* Check whether HTTP embedding mode is active (env vars are set).
|
||||
*/
|
||||
export const isHttpMode = (): boolean => readConfig() !== null;
|
||||
|
||||
/**
|
||||
* Return the configured embedding dimensions for HTTP mode, or undefined
|
||||
* if HTTP mode is not active or no explicit dimensions are set.
|
||||
*/
|
||||
export const getHttpDimensions = (): number | undefined => readConfig()?.dimensions;
|
||||
|
||||
/**
|
||||
* Return a safe representation of a URL for error messages.
|
||||
* Strips query string (may contain tokens) and userinfo.
|
||||
*/
|
||||
const safeUrl = (url: string): string => {
|
||||
try {
|
||||
const u = new URL(url);
|
||||
return `${u.protocol}//${u.host}${u.pathname}`;
|
||||
} catch {
|
||||
return '<invalid-url>';
|
||||
}
|
||||
};
|
||||
|
||||
interface EmbeddingItem {
|
||||
embedding: number[];
|
||||
}
|
||||
|
||||
/**
|
||||
* Send a single batch of texts to the embedding endpoint with retry.
|
||||
*
|
||||
* @param url - Full endpoint URL (e.g. https://host/v1/embeddings)
|
||||
* @param batch - Texts to embed
|
||||
* @param model - Model name for the request body
|
||||
* @param apiKey - Bearer token (only used in Authorization header)
|
||||
* @param batchIndex - Logical batch number (for error context)
|
||||
* @param attempt - Current retry attempt (internal)
|
||||
*/
|
||||
const httpEmbedBatch = async (
|
||||
url: string,
|
||||
batch: string[],
|
||||
model: string,
|
||||
apiKey: string,
|
||||
batchIndex = 0,
|
||||
attempt = 0,
|
||||
): Promise<EmbeddingItem[]> => {
|
||||
let resp: Response;
|
||||
try {
|
||||
resp = await fetch(url, {
|
||||
method: 'POST',
|
||||
signal: AbortSignal.timeout(HTTP_TIMEOUT_MS),
|
||||
headers: {
|
||||
'Content-Type': 'application/json',
|
||||
'Authorization': `Bearer ${apiKey}`,
|
||||
},
|
||||
body: JSON.stringify({ input: batch, model }),
|
||||
});
|
||||
} catch (err) {
|
||||
// Timeouts should not be retried — the server is unresponsive.
|
||||
// AbortSignal.timeout() throws DOMException with name 'TimeoutError'.
|
||||
const isTimeout = err instanceof DOMException && err.name === 'TimeoutError';
|
||||
if (isTimeout) {
|
||||
throw new Error(
|
||||
`Embedding request timed out after ${HTTP_TIMEOUT_MS}ms (${safeUrl(url)}, batch ${batchIndex})`,
|
||||
);
|
||||
}
|
||||
// DNS, connection errors — retry with backoff
|
||||
if (attempt < HTTP_MAX_RETRIES) {
|
||||
const delay = HTTP_RETRY_BACKOFF_MS * (attempt + 1);
|
||||
await new Promise(r => setTimeout(r, delay));
|
||||
return httpEmbedBatch(url, batch, model, apiKey, batchIndex, attempt + 1);
|
||||
}
|
||||
const reason = err instanceof Error ? err.message : String(err);
|
||||
throw new Error(
|
||||
`Embedding request failed (${safeUrl(url)}, batch ${batchIndex}): ${reason}`,
|
||||
);
|
||||
}
|
||||
|
||||
if (!resp.ok) {
|
||||
const status = resp.status;
|
||||
if ((status === 429 || status >= 500) && attempt < HTTP_MAX_RETRIES) {
|
||||
const delay = HTTP_RETRY_BACKOFF_MS * (attempt + 1);
|
||||
await new Promise(r => setTimeout(r, delay));
|
||||
return httpEmbedBatch(url, batch, model, apiKey, batchIndex, attempt + 1);
|
||||
}
|
||||
throw new Error(
|
||||
`Embedding endpoint returned ${status} (${safeUrl(url)}, batch ${batchIndex})`,
|
||||
);
|
||||
}
|
||||
|
||||
const data = (await resp.json()) as { data: EmbeddingItem[] };
|
||||
return data.data;
|
||||
};
|
||||
|
||||
/**
|
||||
* Embed texts via the HTTP backend, splitting into batches.
|
||||
* Reads config from env vars on every call.
|
||||
*
|
||||
* @param texts - Array of texts to embed
|
||||
* @returns Array of Float32Array embedding vectors
|
||||
*/
|
||||
export const httpEmbed = async (texts: string[]): Promise<Float32Array[]> => {
|
||||
if (texts.length === 0) return [];
|
||||
|
||||
const config = readConfig();
|
||||
if (!config) throw new Error('HTTP embedding not configured');
|
||||
|
||||
const url = `${config.baseUrl}/embeddings`;
|
||||
const allVectors: Float32Array[] = [];
|
||||
|
||||
for (let i = 0; i < texts.length; i += HTTP_BATCH_SIZE) {
|
||||
const batch = texts.slice(i, i + HTTP_BATCH_SIZE);
|
||||
const batchIndex = Math.floor(i / HTTP_BATCH_SIZE);
|
||||
const items = await httpEmbedBatch(url, batch, config.model, config.apiKey, batchIndex);
|
||||
|
||||
if (items.length !== batch.length) {
|
||||
throw new Error(
|
||||
`Embedding endpoint returned ${items.length} vectors for ${batch.length} texts ` +
|
||||
`(${safeUrl(url)}, batch ${batchIndex})`,
|
||||
);
|
||||
}
|
||||
|
||||
for (const item of items) {
|
||||
const vec = new Float32Array(item.embedding);
|
||||
// Fail fast on dimension mismatch rather than inserting bad vectors
|
||||
// into the FLOAT[N] column which would cause a cryptic Kuzu error.
|
||||
const expected = config.dimensions ?? DEFAULT_DIMS;
|
||||
if (vec.length !== expected) {
|
||||
const hint = config.dimensions
|
||||
? 'Update GITNEXUS_EMBEDDING_DIMS to match your model output.'
|
||||
: `Set GITNEXUS_EMBEDDING_DIMS=${vec.length} to match your model output.`;
|
||||
throw new Error(
|
||||
`Embedding dimension mismatch: endpoint returned ${vec.length}d vector, ` +
|
||||
`but expected ${expected}d. ${hint}`,
|
||||
);
|
||||
}
|
||||
|
||||
allVectors.push(vec);
|
||||
}
|
||||
}
|
||||
|
||||
return allVectors;
|
||||
};
|
||||
|
||||
/**
|
||||
* Embed a single query text via the HTTP backend.
|
||||
* Convenience for MCP search where only one vector is needed.
|
||||
*
|
||||
* @param text - Query text to embed
|
||||
* @returns Embedding vector as number array
|
||||
*/
|
||||
export const httpEmbedQuery = async (text: string): Promise<number[]> => {
|
||||
const config = readConfig();
|
||||
if (!config) throw new Error('HTTP embedding not configured');
|
||||
|
||||
const url = `${config.baseUrl}/embeddings`;
|
||||
const items = await httpEmbedBatch(url, [text], config.model, config.apiKey);
|
||||
if (!items.length) {
|
||||
throw new Error(`Embedding endpoint returned empty response (${safeUrl(url)})`);
|
||||
}
|
||||
|
||||
const embedding = items[0].embedding;
|
||||
// Same dimension checks as httpEmbed — catch mismatches before they
|
||||
// reach the Kuzu FLOAT[N] cast in search queries.
|
||||
const expected = config.dimensions ?? DEFAULT_DIMS;
|
||||
if (embedding.length !== expected) {
|
||||
const hint = config.dimensions
|
||||
? 'Update GITNEXUS_EMBEDDING_DIMS to match your model output.'
|
||||
: `Set GITNEXUS_EMBEDDING_DIMS=${embedding.length} to match your model output.`;
|
||||
throw new Error(
|
||||
`Embedding dimension mismatch: endpoint returned ${embedding.length}d vector, ` +
|
||||
`but expected ${expected}d. ${hint}`,
|
||||
);
|
||||
}
|
||||
return embedding;
|
||||
};
|
||||
@@ -5,6 +5,7 @@
|
||||
*/
|
||||
|
||||
export * from './types.js';
|
||||
export * from './http-client.js';
|
||||
export * from './embedder.js';
|
||||
export * from './text-generator.js';
|
||||
export * from './embedding-pipeline.js';
|
||||
|
||||
@@ -53,7 +53,7 @@ export interface EmbeddingProgress {
|
||||
* Configuration for the embedding pipeline
|
||||
*/
|
||||
export interface EmbeddingConfig {
|
||||
/** Model identifier for transformers.js */
|
||||
/** Model identifier for transformers.js (local) or the HTTP endpoint model name */
|
||||
modelId: string;
|
||||
/** Number of nodes to embed in each batch */
|
||||
batchSize: number;
|
||||
@@ -65,6 +65,7 @@ export interface EmbeddingConfig {
|
||||
maxSnippetLength: number;
|
||||
}
|
||||
|
||||
|
||||
/**
|
||||
* Default embedding configuration
|
||||
* Uses snowflake-arctic-embed-xs for browser efficiency
|
||||
@@ -92,7 +93,7 @@ export interface SemanticSearchResult {
|
||||
}
|
||||
|
||||
/**
|
||||
* Node data for embedding (minimal structure from KuzuDB query)
|
||||
* Node data for embedding (minimal structure from LadybugDB query)
|
||||
*/
|
||||
export interface EmbeddableNode {
|
||||
id: string;
|
||||
|
||||
@@ -32,7 +32,8 @@ export type NodeLabel =
|
||||
| 'Delegate'
|
||||
| 'Annotation'
|
||||
| 'Constructor'
|
||||
| 'Template';
|
||||
| 'Template'
|
||||
| 'Section';
|
||||
|
||||
|
||||
import { SupportedLanguages } from '../../config/supported-languages.js';
|
||||
@@ -65,6 +66,8 @@ export type NodeProperties = {
|
||||
entryPointReason?: string,
|
||||
// Method signature (for MRO disambiguation)
|
||||
parameterCount?: number,
|
||||
// Section-specific (markdown heading level, 1-6)
|
||||
level?: number,
|
||||
returnType?: string,
|
||||
}
|
||||
|
||||
@@ -80,6 +83,8 @@ export type RelationshipType =
|
||||
| 'IMPLEMENTS'
|
||||
| 'EXTENDS'
|
||||
| 'HAS_METHOD'
|
||||
| 'HAS_PROPERTY'
|
||||
| 'ACCESSES'
|
||||
| 'MEMBER_OF'
|
||||
| 'STEP_IN_PROCESS'
|
||||
|
||||
@@ -96,7 +101,7 @@ export interface GraphRelationship {
|
||||
type: RelationshipType,
|
||||
/** Confidence score 0-1 (1.0 = certain, lower = uncertain resolution) */
|
||||
confidence: number,
|
||||
/** Resolution reason: 'import-resolved', 'same-file', 'fuzzy-global', or empty for non-CALLS */
|
||||
/** Semantics are edge-type-dependent: CALLS uses resolution tier, ACCESSES uses 'read'/'write', OVERRIDES uses MRO reason */
|
||||
reason: string,
|
||||
/** Step number for STEP_IN_PROCESS relationships (1-indexed) */
|
||||
step?: number,
|
||||
|
||||
@@ -0,0 +1,710 @@
|
||||
import type Parser from 'tree-sitter';
|
||||
import { SupportedLanguages } from '../../config/supported-languages.js';
|
||||
import type { NodeLabel } from '../graph/types.js';
|
||||
import { generateId } from '../../lib/utils.js';
|
||||
import { extractSimpleTypeName } from './type-extractors/shared.js';
|
||||
|
||||
/** Tree-sitter AST node. Re-exported for use across ingestion modules. */
|
||||
export type SyntaxNode = Parser.SyntaxNode;
|
||||
|
||||
/**
|
||||
* Ordered list of definition capture keys for tree-sitter query matches.
|
||||
* Used to extract the definition node from a capture map.
|
||||
*/
|
||||
export const DEFINITION_CAPTURE_KEYS = [
|
||||
'definition.function',
|
||||
'definition.class',
|
||||
'definition.interface',
|
||||
'definition.method',
|
||||
'definition.struct',
|
||||
'definition.enum',
|
||||
'definition.namespace',
|
||||
'definition.module',
|
||||
'definition.trait',
|
||||
'definition.impl',
|
||||
'definition.type',
|
||||
'definition.const',
|
||||
'definition.static',
|
||||
'definition.typedef',
|
||||
'definition.macro',
|
||||
'definition.union',
|
||||
'definition.property',
|
||||
'definition.record',
|
||||
'definition.delegate',
|
||||
'definition.annotation',
|
||||
'definition.constructor',
|
||||
'definition.template',
|
||||
] as const;
|
||||
|
||||
/** Extract the definition node from a tree-sitter query capture map. */
|
||||
export const getDefinitionNodeFromCaptures = (captureMap: Record<string, any>): SyntaxNode | null => {
|
||||
for (const key of DEFINITION_CAPTURE_KEYS) {
|
||||
if (captureMap[key]) return captureMap[key];
|
||||
}
|
||||
return null;
|
||||
};
|
||||
|
||||
/**
|
||||
* Node types that represent function/method definitions across languages.
|
||||
* Used to find the enclosing function for a call site.
|
||||
*/
|
||||
export const FUNCTION_NODE_TYPES = new Set([
|
||||
// TypeScript/JavaScript
|
||||
'function_declaration',
|
||||
'arrow_function',
|
||||
'function_expression',
|
||||
'method_definition',
|
||||
'generator_function_declaration',
|
||||
// Python
|
||||
'function_definition',
|
||||
// Common async variants
|
||||
'async_function_declaration',
|
||||
'async_arrow_function',
|
||||
// Java
|
||||
'method_declaration',
|
||||
'constructor_declaration',
|
||||
// C/C++
|
||||
// 'function_definition' already included above
|
||||
// Go
|
||||
// 'method_declaration' already included from Java
|
||||
// C#
|
||||
'local_function_statement',
|
||||
// Rust
|
||||
'function_item',
|
||||
'impl_item', // Methods inside impl blocks
|
||||
// PHP
|
||||
'anonymous_function',
|
||||
// Kotlin
|
||||
'lambda_literal',
|
||||
// Swift
|
||||
'init_declaration',
|
||||
'deinit_declaration',
|
||||
// Ruby
|
||||
'method', // def foo
|
||||
'singleton_method', // def self.foo
|
||||
]);
|
||||
|
||||
/**
|
||||
* Node types for standard function declarations that need C/C++ declarator handling.
|
||||
* Used by extractFunctionName to determine how to extract the function name.
|
||||
*/
|
||||
export const FUNCTION_DECLARATION_TYPES = new Set([
|
||||
'function_declaration',
|
||||
'function_definition',
|
||||
'async_function_declaration',
|
||||
'generator_function_declaration',
|
||||
'function_item',
|
||||
]);
|
||||
|
||||
/** AST node types that represent a class-like container (for HAS_METHOD edge extraction) */
|
||||
export const CLASS_CONTAINER_TYPES = new Set([
|
||||
'class_declaration', 'abstract_class_declaration',
|
||||
'interface_declaration', 'struct_declaration', 'record_declaration',
|
||||
'class_specifier', 'struct_specifier',
|
||||
'impl_item', 'trait_item', 'struct_item', 'enum_item',
|
||||
'class_definition',
|
||||
'trait_declaration',
|
||||
'protocol_declaration',
|
||||
// Ruby
|
||||
'class',
|
||||
'module',
|
||||
// Kotlin
|
||||
'object_declaration',
|
||||
'companion_object',
|
||||
]);
|
||||
|
||||
export const CONTAINER_TYPE_TO_LABEL: Record<string, string> = {
|
||||
class_declaration: 'Class',
|
||||
abstract_class_declaration: 'Class',
|
||||
interface_declaration: 'Interface',
|
||||
struct_declaration: 'Struct',
|
||||
struct_specifier: 'Struct',
|
||||
class_specifier: 'Class',
|
||||
class_definition: 'Class',
|
||||
impl_item: 'Impl',
|
||||
trait_item: 'Trait',
|
||||
struct_item: 'Struct',
|
||||
enum_item: 'Enum',
|
||||
trait_declaration: 'Trait',
|
||||
record_declaration: 'Record',
|
||||
protocol_declaration: 'Interface',
|
||||
class: 'Class',
|
||||
module: 'Module',
|
||||
object_declaration: 'Class',
|
||||
companion_object: 'Class',
|
||||
};
|
||||
|
||||
/** Check if a Kotlin function_declaration capture is inside a class_body (i.e., a method).
|
||||
* Kotlin grammar uses function_declaration for both top-level functions and class methods.
|
||||
* Returns true when the captured definition node has a class_body ancestor. */
|
||||
export function isKotlinClassMethod(captureNode: { parent?: any } | null | undefined): boolean {
|
||||
let ancestor = captureNode?.parent;
|
||||
while (ancestor) {
|
||||
if (ancestor.type === 'class_body') return true;
|
||||
ancestor = ancestor.parent;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
/**
|
||||
* C/C++: check if a Function capture is inside a class/struct body.
|
||||
* If true, the function is already captured by @definition.method and should be skipped
|
||||
* to prevent double-indexing in globalIndex.
|
||||
*/
|
||||
export function isCppDuplicateClassFunction(
|
||||
functionNode: { parent?: any } | null | undefined,
|
||||
nodeLabel: string,
|
||||
language: SupportedLanguages,
|
||||
): boolean {
|
||||
if (nodeLabel !== 'Function') return false;
|
||||
if (language !== SupportedLanguages.CPlusPlus && language !== SupportedLanguages.C) return false;
|
||||
let ancestor = functionNode?.parent;
|
||||
while (ancestor) {
|
||||
if (ancestor.type === 'class_specifier' || ancestor.type === 'struct_specifier') return true;
|
||||
ancestor = ancestor.parent;
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
/**
|
||||
* Determine the graph node label from a tree-sitter capture map.
|
||||
* Handles language-specific reclassification (C/C++ duplicate skipping, Kotlin Method promotion).
|
||||
* Returns null if the capture should be skipped (import, call, C/C++ duplicate, missing name).
|
||||
*/
|
||||
export function getLabelFromCaptures(
|
||||
captureMap: Record<string, any>,
|
||||
language: SupportedLanguages,
|
||||
): NodeLabel | null {
|
||||
if (captureMap['import'] || captureMap['call']) return null;
|
||||
if (!captureMap['name'] && !captureMap['definition.constructor']) return null;
|
||||
|
||||
if (captureMap['definition.function']) {
|
||||
if (isCppDuplicateClassFunction(captureMap['definition.function'], 'Function', language)) return null;
|
||||
if (language === SupportedLanguages.Kotlin && isKotlinClassMethod(captureMap['definition.function'])) return 'Method';
|
||||
return 'Function';
|
||||
}
|
||||
if (captureMap['definition.class']) return 'Class';
|
||||
if (captureMap['definition.interface']) return 'Interface';
|
||||
if (captureMap['definition.method']) return 'Method';
|
||||
if (captureMap['definition.struct']) return 'Struct';
|
||||
if (captureMap['definition.enum']) return 'Enum';
|
||||
if (captureMap['definition.namespace']) return 'Namespace';
|
||||
if (captureMap['definition.module']) return 'Module';
|
||||
if (captureMap['definition.trait']) return 'Trait';
|
||||
if (captureMap['definition.impl']) return 'Impl';
|
||||
if (captureMap['definition.type']) return 'TypeAlias';
|
||||
if (captureMap['definition.const']) return 'Const';
|
||||
if (captureMap['definition.static']) return 'Static';
|
||||
if (captureMap['definition.typedef']) return 'Typedef';
|
||||
if (captureMap['definition.macro']) return 'Macro';
|
||||
if (captureMap['definition.union']) return 'Union';
|
||||
if (captureMap['definition.property']) return 'Property';
|
||||
if (captureMap['definition.record']) return 'Record';
|
||||
if (captureMap['definition.delegate']) return 'Delegate';
|
||||
if (captureMap['definition.annotation']) return 'Annotation';
|
||||
if (captureMap['definition.constructor']) return 'Constructor';
|
||||
if (captureMap['definition.template']) return 'Template';
|
||||
return 'CodeElement';
|
||||
}
|
||||
|
||||
/** Walk up AST to find enclosing class/struct/interface/impl, return its generateId or null.
|
||||
* For Go method_declaration nodes, extracts receiver type (e.g. `func (u *User) Save()` → User struct). */
|
||||
export const findEnclosingClassId = (node: any, filePath: string): string | null => {
|
||||
let current = node.parent;
|
||||
while (current) {
|
||||
// Go: method_declaration has a receiver parameter with the struct type
|
||||
if (current.type === 'method_declaration') {
|
||||
const receiver = current.childForFieldName?.('receiver');
|
||||
if (receiver) {
|
||||
// receiver is a parameter_list: (u *User) or (u User)
|
||||
const paramDecl = receiver.namedChildren?.find?.((c: any) => c.type === 'parameter_declaration');
|
||||
if (paramDecl) {
|
||||
const typeNode = paramDecl.childForFieldName?.('type');
|
||||
if (typeNode) {
|
||||
// Unwrap pointer_type (*User → User)
|
||||
const inner = typeNode.type === 'pointer_type' ? typeNode.firstNamedChild : typeNode;
|
||||
if (inner && (inner.type === 'type_identifier' || inner.type === 'identifier')) {
|
||||
return generateId('Struct', `${filePath}:${inner.text}`);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
// Go: type_declaration wrapping a struct_type (type User struct { ... })
|
||||
// field_declaration → field_declaration_list → struct_type → type_spec → type_declaration
|
||||
if (current.type === 'type_declaration') {
|
||||
const typeSpec = current.children?.find((c: any) => c.type === 'type_spec');
|
||||
if (typeSpec) {
|
||||
const typeBody = typeSpec.childForFieldName?.('type');
|
||||
if (typeBody?.type === 'struct_type' || typeBody?.type === 'interface_type') {
|
||||
const nameNode = typeSpec.childForFieldName?.('name');
|
||||
if (nameNode) {
|
||||
const label = typeBody.type === 'struct_type' ? 'Struct' : 'Interface';
|
||||
return generateId(label, `${filePath}:${nameNode.text}`);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
if (CLASS_CONTAINER_TYPES.has(current.type)) {
|
||||
// Rust impl_item: for `impl Trait for Struct {}`, pick the type after `for`
|
||||
if (current.type === 'impl_item') {
|
||||
const children = current.children ?? [];
|
||||
const forIdx = children.findIndex((c: any) => c.text === 'for');
|
||||
if (forIdx !== -1) {
|
||||
const nameNode = children.slice(forIdx + 1).find((c: any) =>
|
||||
c.type === 'type_identifier' || c.type === 'identifier'
|
||||
);
|
||||
if (nameNode) {
|
||||
return generateId('Impl', `${filePath}:${nameNode.text}`);
|
||||
}
|
||||
}
|
||||
// Fall through: plain `impl Struct {}` — use first type_identifier below
|
||||
}
|
||||
const nameNode = current.childForFieldName?.('name')
|
||||
?? current.children?.find((c: any) =>
|
||||
c.type === 'type_identifier' || c.type === 'identifier' || c.type === 'name' || c.type === 'constant'
|
||||
);
|
||||
if (nameNode) {
|
||||
const label = CONTAINER_TYPE_TO_LABEL[current.type] || 'Class';
|
||||
return generateId(label, `${filePath}:${nameNode.text}`);
|
||||
}
|
||||
}
|
||||
current = current.parent;
|
||||
}
|
||||
return null;
|
||||
};
|
||||
|
||||
/**
|
||||
* Find a child of `childType` within a sibling node of `siblingType`.
|
||||
* Used for Kotlin AST traversal where visibility_modifier lives inside a modifiers sibling.
|
||||
*/
|
||||
export const findSiblingChild = (parent: any, siblingType: string, childType: string): any | null => {
|
||||
for (let i = 0; i < parent.childCount; i++) {
|
||||
const sibling = parent.child(i);
|
||||
if (sibling?.type === siblingType) {
|
||||
for (let j = 0; j < sibling.childCount; j++) {
|
||||
const child = sibling.child(j);
|
||||
if (child?.type === childType) return child;
|
||||
}
|
||||
}
|
||||
}
|
||||
return null;
|
||||
};
|
||||
|
||||
/**
|
||||
* Extract function name and label from a function_definition or similar AST node.
|
||||
* Handles C/C++ qualified_identifier (ClassName::MethodName) and other language patterns.
|
||||
*/
|
||||
export const extractFunctionName = (node: SyntaxNode): { funcName: string | null; label: string } => {
|
||||
let funcName: string | null = null;
|
||||
let label = 'Function';
|
||||
|
||||
// Swift init/deinit
|
||||
if (node.type === 'init_declaration' || node.type === 'deinit_declaration') {
|
||||
return {
|
||||
funcName: node.type === 'init_declaration' ? 'init' : 'deinit',
|
||||
label: 'Constructor',
|
||||
};
|
||||
}
|
||||
|
||||
if (FUNCTION_DECLARATION_TYPES.has(node.type)) {
|
||||
// C/C++: function_definition -> [pointer_declarator ->] function_declarator -> qualified_identifier/identifier
|
||||
// Unwrap pointer_declarator / reference_declarator wrappers to reach function_declarator
|
||||
let declarator = node.childForFieldName?.('declarator');
|
||||
if (!declarator) {
|
||||
for (let i = 0; i < node.childCount; i++) {
|
||||
const c = node.child(i);
|
||||
if (c?.type === 'function_declarator') { declarator = c; break; }
|
||||
}
|
||||
}
|
||||
while (declarator && (declarator.type === 'pointer_declarator' || declarator.type === 'reference_declarator')) {
|
||||
let nextDeclarator = declarator.childForFieldName?.('declarator');
|
||||
if (!nextDeclarator) {
|
||||
for (let i = 0; i < declarator.childCount; i++) {
|
||||
const c = declarator.child(i);
|
||||
if (c?.type === 'function_declarator' || c?.type === 'pointer_declarator' || c?.type === 'reference_declarator') { nextDeclarator = c; break; }
|
||||
}
|
||||
}
|
||||
declarator = nextDeclarator;
|
||||
}
|
||||
if (declarator) {
|
||||
let innerDeclarator = declarator.childForFieldName?.('declarator');
|
||||
if (!innerDeclarator) {
|
||||
for (let i = 0; i < declarator.childCount; i++) {
|
||||
const c = declarator.child(i);
|
||||
if (c?.type === 'qualified_identifier' || c?.type === 'identifier'
|
||||
|| c?.type === 'field_identifier' || c?.type === 'parenthesized_declarator') { innerDeclarator = c; break; }
|
||||
}
|
||||
}
|
||||
|
||||
if (innerDeclarator?.type === 'qualified_identifier') {
|
||||
let nameNode = innerDeclarator.childForFieldName?.('name');
|
||||
if (!nameNode) {
|
||||
for (let i = 0; i < innerDeclarator.childCount; i++) {
|
||||
const c = innerDeclarator.child(i);
|
||||
if (c?.type === 'identifier') { nameNode = c; break; }
|
||||
}
|
||||
}
|
||||
if (nameNode?.text) {
|
||||
funcName = nameNode.text;
|
||||
label = 'Method';
|
||||
}
|
||||
} else if (innerDeclarator?.type === 'identifier' || innerDeclarator?.type === 'field_identifier') {
|
||||
// field_identifier is used for method names inside C++ class bodies
|
||||
funcName = innerDeclarator.text;
|
||||
if (innerDeclarator.type === 'field_identifier') label = 'Method';
|
||||
} else if (innerDeclarator?.type === 'parenthesized_declarator') {
|
||||
let nestedId: SyntaxNode | null = null;
|
||||
for (let i = 0; i < innerDeclarator.childCount; i++) {
|
||||
const c = innerDeclarator.child(i);
|
||||
if (c?.type === 'qualified_identifier' || c?.type === 'identifier') { nestedId = c; break; }
|
||||
}
|
||||
if (nestedId?.type === 'qualified_identifier') {
|
||||
let nameNode = nestedId.childForFieldName?.('name');
|
||||
if (!nameNode) {
|
||||
for (let i = 0; i < nestedId.childCount; i++) {
|
||||
const c = nestedId.child(i);
|
||||
if (c?.type === 'identifier') { nameNode = c; break; }
|
||||
}
|
||||
}
|
||||
if (nameNode?.text) {
|
||||
funcName = nameNode.text;
|
||||
label = 'Method';
|
||||
}
|
||||
} else if (nestedId?.type === 'identifier') {
|
||||
funcName = nestedId.text;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Fallback for other languages (Kotlin uses simple_identifier, Swift uses simple_identifier)
|
||||
if (!funcName) {
|
||||
let nameNode = node.childForFieldName?.('name');
|
||||
if (!nameNode) {
|
||||
for (let i = 0; i < node.childCount; i++) {
|
||||
const c = node.child(i);
|
||||
if (c?.type === 'identifier' || c?.type === 'property_identifier' || c?.type === 'simple_identifier') { nameNode = c; break; }
|
||||
}
|
||||
}
|
||||
funcName = nameNode?.text;
|
||||
|
||||
// Kotlin: function_declaration inside a class_body is a method, not a top-level function.
|
||||
// Must match the label assigned in parse-worker.ts for consistent generateId() output.
|
||||
if (funcName && node.type === 'function_declaration' && isKotlinClassMethod(node)) {
|
||||
label = 'Method';
|
||||
}
|
||||
}
|
||||
} else if (node.type === 'impl_item') {
|
||||
let funcItem: SyntaxNode | null = null;
|
||||
for (let i = 0; i < node.childCount; i++) {
|
||||
const c = node.child(i);
|
||||
if (c?.type === 'function_item') { funcItem = c; break; }
|
||||
}
|
||||
if (funcItem) {
|
||||
let nameNode = funcItem.childForFieldName?.('name');
|
||||
if (!nameNode) {
|
||||
for (let i = 0; i < funcItem.childCount; i++) {
|
||||
const c = funcItem.child(i);
|
||||
if (c?.type === 'identifier') { nameNode = c; break; }
|
||||
}
|
||||
}
|
||||
funcName = nameNode?.text;
|
||||
label = 'Method';
|
||||
}
|
||||
} else if (node.type === 'method_definition') {
|
||||
let nameNode = node.childForFieldName?.('name');
|
||||
if (!nameNode) {
|
||||
for (let i = 0; i < node.childCount; i++) {
|
||||
const c = node.child(i);
|
||||
if (c?.type === 'property_identifier') { nameNode = c; break; }
|
||||
}
|
||||
}
|
||||
funcName = nameNode?.text;
|
||||
label = 'Method';
|
||||
} else if (node.type === 'method_declaration' || node.type === 'constructor_declaration') {
|
||||
let nameNode = node.childForFieldName?.('name');
|
||||
if (!nameNode) {
|
||||
for (let i = 0; i < node.childCount; i++) {
|
||||
const c = node.child(i);
|
||||
if (c?.type === 'identifier') { nameNode = c; break; }
|
||||
}
|
||||
}
|
||||
funcName = nameNode?.text;
|
||||
label = 'Method';
|
||||
} else if (node.type === 'arrow_function' || node.type === 'function_expression') {
|
||||
const parent = node.parent;
|
||||
if (parent?.type === 'variable_declarator') {
|
||||
let nameNode = parent.childForFieldName?.('name');
|
||||
if (!nameNode) {
|
||||
for (let i = 0; i < parent.childCount; i++) {
|
||||
const c = parent.child(i);
|
||||
if (c?.type === 'identifier') { nameNode = c; break; }
|
||||
}
|
||||
}
|
||||
funcName = nameNode?.text;
|
||||
}
|
||||
} else if (node.type === 'method' || node.type === 'singleton_method') {
|
||||
let nameNode = node.childForFieldName?.('name');
|
||||
if (!nameNode) {
|
||||
for (let i = 0; i < node.childCount; i++) {
|
||||
const c = node.child(i);
|
||||
if (c?.type === 'identifier') { nameNode = c; break; }
|
||||
}
|
||||
}
|
||||
funcName = nameNode?.text;
|
||||
label = 'Method';
|
||||
}
|
||||
|
||||
return { funcName, label };
|
||||
};
|
||||
|
||||
export interface MethodSignature {
|
||||
parameterCount: number | undefined;
|
||||
/** Number of required (non-optional, non-default) parameters.
|
||||
* Only set when fewer than parameterCount — enables range-based arity filtering.
|
||||
* undefined means all parameters are required (or metadata unavailable). */
|
||||
requiredParameterCount: number | undefined;
|
||||
/** Per-parameter type names extracted via extractSimpleTypeName.
|
||||
* Only populated for languages with method overloading (Java, Kotlin, C#, C++).
|
||||
* undefined (not []) when no types are extractable — avoids empty array allocations. */
|
||||
parameterTypes: string[] | undefined;
|
||||
returnType: string | undefined;
|
||||
}
|
||||
|
||||
/** Argument list node types shared between extractMethodSignature and countCallArguments. */
|
||||
export const CALL_ARGUMENT_LIST_TYPES = new Set([
|
||||
'arguments',
|
||||
'argument_list',
|
||||
'value_arguments',
|
||||
]);
|
||||
|
||||
/**
|
||||
* Extract parameter count and return type text from an AST method/function node.
|
||||
* Works across languages by looking for common AST patterns.
|
||||
*/
|
||||
export const extractMethodSignature = (node: SyntaxNode | null | undefined): MethodSignature => {
|
||||
let parameterCount: number | undefined = 0;
|
||||
let requiredCount = 0;
|
||||
let returnType: string | undefined;
|
||||
let isVariadic = false;
|
||||
const paramTypes: string[] = [];
|
||||
|
||||
if (!node) return { parameterCount, requiredParameterCount: undefined, parameterTypes: undefined, returnType };
|
||||
|
||||
const paramListTypes = new Set([
|
||||
'formal_parameters', 'parameters', 'parameter_list',
|
||||
'function_parameters', 'method_parameters', 'function_value_parameters',
|
||||
]);
|
||||
|
||||
// Node types that indicate variadic/rest parameters
|
||||
const VARIADIC_PARAM_TYPES = new Set([
|
||||
'variadic_parameter_declaration', // Go: ...string
|
||||
'variadic_parameter', // Rust: extern "C" fn(...)
|
||||
'spread_parameter', // Java: Object... args
|
||||
'list_splat_pattern', // Python: *args
|
||||
'dictionary_splat_pattern', // Python: **kwargs
|
||||
]);
|
||||
|
||||
/** AST node types that represent parameters with default values. */
|
||||
const OPTIONAL_PARAM_TYPES = new Set([
|
||||
'optional_parameter', // TypeScript, Ruby: (x?: number), (x: number = 5), def f(x = 5)
|
||||
'default_parameter', // Python: def f(x=5)
|
||||
'typed_default_parameter', // Python: def f(x: int = 5)
|
||||
'optional_parameter_declaration', // C++: void f(int x = 5)
|
||||
]);
|
||||
|
||||
/** Check if a parameter node has a default value (handles Kotlin, C#, Swift, PHP
|
||||
* where defaults are expressed as child nodes rather than distinct node types). */
|
||||
const hasDefaultValue = (paramNode: SyntaxNode): boolean => {
|
||||
if (OPTIONAL_PARAM_TYPES.has(paramNode.type)) return true;
|
||||
// C#, Swift, PHP: check for '=' token or equals_value_clause child
|
||||
for (let i = 0; i < paramNode.childCount; i++) {
|
||||
const c = paramNode.child(i);
|
||||
if (!c) continue;
|
||||
if (c.type === '=' || c.type === 'equals_value_clause') return true;
|
||||
}
|
||||
// Kotlin: default values are siblings of the parameter node, not children.
|
||||
// The AST is: parameter, =, <literal> — all at function_value_parameters level.
|
||||
// Check if the immediately following sibling is '=' (default value separator).
|
||||
const sib = paramNode.nextSibling;
|
||||
if (sib && sib.type === '=') return true;
|
||||
return false;
|
||||
};
|
||||
|
||||
const findParameterList = (current: SyntaxNode): SyntaxNode | null => {
|
||||
for (const child of current.children) {
|
||||
if (paramListTypes.has(child.type)) return child;
|
||||
}
|
||||
for (const child of current.children) {
|
||||
const nested = findParameterList(child);
|
||||
if (nested) return nested;
|
||||
}
|
||||
return null;
|
||||
};
|
||||
|
||||
const parameterList = (
|
||||
paramListTypes.has(node.type) ? node // node itself IS the parameter list (e.g. C# primary constructors)
|
||||
: node.childForFieldName?.('parameters')
|
||||
?? findParameterList(node)
|
||||
);
|
||||
|
||||
if (parameterList && paramListTypes.has(parameterList.type)) {
|
||||
for (const param of parameterList.namedChildren) {
|
||||
if (param.type === 'comment') continue;
|
||||
if (param.text === 'self' || param.text === '&self' || param.text === '&mut self' ||
|
||||
param.type === 'self_parameter') {
|
||||
continue;
|
||||
}
|
||||
// Kotlin: default values are siblings of the parameter node inside
|
||||
// function_value_parameters, so they appear as named children (e.g.
|
||||
// string_literal, integer_literal, boolean_literal, call_expression).
|
||||
// Skip any named child that isn't a parameter-like or modifier node.
|
||||
if (param.type.endsWith('_literal') || param.type === 'call_expression'
|
||||
|| param.type === 'navigation_expression' || param.type === 'prefix_expression'
|
||||
|| param.type === 'parenthesized_expression') {
|
||||
continue;
|
||||
}
|
||||
// Check for variadic parameter types
|
||||
if (VARIADIC_PARAM_TYPES.has(param.type)) {
|
||||
isVariadic = true;
|
||||
continue;
|
||||
}
|
||||
// TypeScript/JavaScript: rest parameter — required_parameter containing rest_pattern
|
||||
if (param.type === 'required_parameter' || param.type === 'optional_parameter') {
|
||||
for (const child of param.children) {
|
||||
if (child.type === 'rest_pattern') {
|
||||
isVariadic = true;
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (isVariadic) continue;
|
||||
}
|
||||
// Kotlin: vararg modifier on a regular parameter
|
||||
if (param.type === 'parameter' || param.type === 'formal_parameter') {
|
||||
const prev = param.previousSibling;
|
||||
if (prev?.type === 'parameter_modifiers' && prev.text.includes('vararg')) {
|
||||
isVariadic = true;
|
||||
}
|
||||
}
|
||||
// Extract parameter type name for overload disambiguation.
|
||||
// Works for Java (formal_parameter), Kotlin (parameter), C# (parameter),
|
||||
// C++ (parameter_declaration). Uses childForFieldName('type') which is the
|
||||
// standard tree-sitter field for typed parameters across these languages.
|
||||
// Kotlin uses positional children instead of 'type' field — fall back to
|
||||
// searching for user_type/nullable_type/predefined_type children.
|
||||
const paramTypeNode = param.childForFieldName('type');
|
||||
if (paramTypeNode) {
|
||||
const typeName = extractSimpleTypeName(paramTypeNode);
|
||||
paramTypes.push(typeName ?? 'unknown');
|
||||
} else {
|
||||
// Kotlin: parameter → [simple_identifier, user_type|nullable_type]
|
||||
let found = false;
|
||||
for (const child of param.namedChildren) {
|
||||
if (child.type === 'user_type' || child.type === 'nullable_type'
|
||||
|| child.type === 'type_identifier' || child.type === 'predefined_type') {
|
||||
const typeName = extractSimpleTypeName(child);
|
||||
paramTypes.push(typeName ?? 'unknown');
|
||||
found = true;
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (!found) paramTypes.push('unknown');
|
||||
}
|
||||
if (!hasDefaultValue(param)) requiredCount++;
|
||||
parameterCount++;
|
||||
}
|
||||
// C/C++: bare `...` token in parameter list (not a named child — check all children)
|
||||
if (!isVariadic) {
|
||||
for (const child of parameterList.children) {
|
||||
if (!child.isNamed && child.text === '...') {
|
||||
isVariadic = true;
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Return type extraction — language-specific field names
|
||||
// Go: 'result' field is either a type_identifier or parameter_list (multi-return)
|
||||
const goResult = node.childForFieldName?.('result');
|
||||
if (goResult) {
|
||||
if (goResult.type === 'parameter_list') {
|
||||
// Multi-return: extract first parameter's type only (e.g. (*User, error) → *User)
|
||||
const firstParam = goResult.firstNamedChild;
|
||||
if (firstParam?.type === 'parameter_declaration') {
|
||||
const typeNode = firstParam.childForFieldName('type');
|
||||
if (typeNode) returnType = typeNode.text;
|
||||
} else if (firstParam) {
|
||||
// Unnamed return types: (string, error) — first child is a bare type node
|
||||
returnType = firstParam.text;
|
||||
}
|
||||
} else {
|
||||
returnType = goResult.text;
|
||||
}
|
||||
}
|
||||
|
||||
// Rust: 'return_type' field — the value IS the type node (e.g. primitive_type, type_identifier).
|
||||
// Skip if the node is a type_annotation (TS/Python), which is handled by the generic loop below.
|
||||
if (!returnType) {
|
||||
const rustReturn = node.childForFieldName?.('return_type');
|
||||
if (rustReturn && rustReturn.type !== 'type_annotation') {
|
||||
returnType = rustReturn.text;
|
||||
}
|
||||
}
|
||||
|
||||
// C/C++: 'type' field on function_definition
|
||||
if (!returnType) {
|
||||
const cppType = node.childForFieldName?.('type');
|
||||
if (cppType && cppType.text !== 'void') {
|
||||
returnType = cppType.text;
|
||||
}
|
||||
}
|
||||
|
||||
// C#: 'returns' field on method_declaration
|
||||
if (!returnType) {
|
||||
const csReturn = node.childForFieldName?.('returns');
|
||||
if (csReturn && csReturn.text !== 'void') {
|
||||
returnType = csReturn.text;
|
||||
}
|
||||
}
|
||||
|
||||
// TS/Rust/Python/C#/Kotlin: type_annotation or return_type child
|
||||
if (!returnType) {
|
||||
for (const child of node.children) {
|
||||
if (child.type === 'type_annotation' || child.type === 'return_type') {
|
||||
const typeNode = child.children.find((c) => c.isNamed);
|
||||
if (typeNode) returnType = typeNode.text;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
// Kotlin: fun getUser(): User — return type is a bare user_type child of
|
||||
// function_declaration. The Kotlin grammar does NOT wrap it in type_annotation
|
||||
// or return_type; it appears as a direct child after function_value_parameters.
|
||||
// Note: Kotlin uses function_value_parameters (not a field), so we find it by type.
|
||||
if (!returnType) {
|
||||
let paramsEnd = -1;
|
||||
for (let i = 0; i < node.childCount; i++) {
|
||||
const child = node.child(i);
|
||||
if (!child) continue;
|
||||
if (child.type === 'function_value_parameters' || child.type === 'value_parameters') {
|
||||
paramsEnd = child.endIndex;
|
||||
}
|
||||
if (paramsEnd >= 0 && child.type === 'user_type' && child.startIndex > paramsEnd) {
|
||||
returnType = child.text;
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (isVariadic) parameterCount = undefined;
|
||||
|
||||
// Only include parameterTypes when at least one type was successfully extracted.
|
||||
// Use undefined (not []) to avoid empty array allocations for untyped parameters.
|
||||
const hasTypes = paramTypes.length > 0 && paramTypes.some(t => t !== 'unknown');
|
||||
// Only set requiredParameterCount when it differs from total — saves memory on the common case.
|
||||
const requiredParameterCount = (!isVariadic && requiredCount < (parameterCount ?? 0))
|
||||
? requiredCount : undefined;
|
||||
return { parameterCount, requiredParameterCount, parameterTypes: hasTypes ? paramTypes : undefined, returnType };
|
||||
};
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user