Compare commits

...
6 Commits
Author SHA1 Message Date
Gergo Magyar 3768e3bfd0 fix(server): log skip-embedding count and table-not-found swallow path
Addresses review feedback on PR #823:
- Log count of already-embedded nodes when skipNodeIds is populated
  (aids debugging if Kuzu driver row shape changes).
- Log when the 'table does not exist' swallow path fires so ops can
  catch it if Kuzu ever changes error wording.
- Document the {} config positional argument with an inline comment
  referencing the runEmbeddingPipeline signature.
2026-04-15 07:40:58 +01:00
Gergo Magyar 3384575ac6 style: prettier format gitnexus/src/server/api.ts 2026-04-15 07:38:16 +01:00
jonasvanderhaegen-xve dd194d56b1 fix(server): narrow catch to table-not-exist errors only in POST /api/embed
Bare catch{} would silently swallow connection errors and proceed to
re-embed all nodes, hiding infrastructure issues. Now only swallows
errors where the CodeEmbedding table does not yet exist.
2026-04-14 14:27:03 +02:00
jonasvanderhaegen-xve 80a6fde2ba fix(server): skip already-embedded nodes in POST /api/embed to avoid vector-index SET error
Kuzu/LadybugDB forbids SET on a property that is part of a vector index.
The /api/embed endpoint was calling runEmbeddingPipeline without skipNodeIds,
causing it to attempt MERGE+SET on every node including those already embedded.

Fix: query existing CodeEmbedding nodeIds before running the pipeline and pass
them as skipNodeIds so only new (unembedded) nodes are processed.
2026-04-14 13:33:48 +02:00
jonasvanderhaegen-xve 8d38cc99fa fix(embeddings): use MERGE instead of CREATE for CodeEmbedding inserts
CREATE fails with duplicate PK when a CodeEmbedding node already exists,
which happens when:
- A PostToolUse hook triggers a concurrent gitnexus analyze during an
  active analyze run (git commits fire the hook)
- A partial prior run left some embeddings in the DB before a crash

Switching to MERGE makes the insert idempotent: existing embeddings are
updated in place, new ones are created, no PK violations.

Fixes: #822
2026-04-14 12:59:14 +02:00
jonasvanderhaegen-xve 41844edf88 fix(csv-generator): deduplicate all node types, not just File nodes
The pipeline can produce duplicate node IDs across all symbol types
(Class, Method, Function, etc.). Only File nodes were guarded by a
seenFileIds Set, leaving every other type unprotected. When the CSV
was COPY'd into LadybugDB, duplicate PKs caused mass "Batch execution
error: Found duplicated primary key value" warnings on gitnexus serve.

Replace the per-type seenFileIds with a single seenNodeIds Set checked
at the top of the iteration loop, before the switch, so every label is
covered by the same O(1) deduplication guard.

Fixes: #822
2026-04-14 12:32:08 +02:00
4 changed files with 57 additions and 25 deletions
@@ -100,8 +100,8 @@ const batchInsertEmbeddings = async (
) => Promise<void>,
updates: Array<{ id: string; embedding: number[] }>,
): Promise<void> => {
// INSERT into separate embedding table - much more memory efficient!
const cypher = `CREATE (e:CodeEmbedding {nodeId: $nodeId, embedding: $embedding})`;
// MERGE instead of CREATE — idempotent, handles concurrent analyzes and partial prior runs
const cypher = `MERGE (e:CodeEmbedding {nodeId: $nodeId}) SET e.embedding = $embedding`;
const paramsList = updates.map((u) => ({ nodeId: u.id, embedding: u.embedding }));
await executeWithReusedStatement(cypher, paramsList);
};
+7 -3
View File
@@ -315,14 +315,18 @@ export const streamAllCSVsToDisk = async (
CodeElement: codeElemWriter,
};
const seenFileIds = new Set<string>();
// Deduplicate all node types — the pipeline can produce duplicate IDs across
// all symbol types (Class, Method, Function, etc.), not just File nodes.
// A single Set covering every label prevents PK violations on COPY.
const seenNodeIds = new Set<string>();
// --- SINGLE PASS over all nodes ---
for (const node of graph.iterNodes()) {
if (seenNodeIds.has(node.id)) continue;
seenNodeIds.add(node.id);
switch (node.label) {
case 'File': {
if (seenFileIds.has(node.id)) break;
seenFileIds.add(node.id);
const content = await extractContent(node, contentCache);
await fileWriter.addRow(
[
+1 -1
View File
@@ -222,7 +222,7 @@ export async function runFullAnalysis(
const paramsList = batch.map((e) => ({ nodeId: e.nodeId, embedding: e.embedding }));
try {
await executeWithReusedStatement(
`CREATE (e:CodeEmbedding {nodeId: $nodeId, embedding: $embedding})`,
`MERGE (e:CodeEmbedding {nodeId: $nodeId}) SET e.embedding = $embedding`,
paramsList,
);
} catch {
+47 -19
View File
@@ -1449,25 +1449,53 @@ export const createServer = async (port: number, host: string = '127.0.0.1') =>
await withLbugDb(lbugPath, async () => {
const { runEmbeddingPipeline } =
await import('../core/embeddings/embedding-pipeline.js');
await runEmbeddingPipeline(executeQuery, executeWithReusedStatement, (p) => {
embedJobManager.updateJob(job.id, {
progress: {
phase:
p.phase === 'ready' ? 'complete' : p.phase === 'error' ? 'failed' : p.phase,
percent: p.percent,
message:
p.phase === 'loading-model'
? 'Loading embedding model...'
: p.phase === 'embedding'
? `Embedding nodes (${p.percent}%)...`
: p.phase === 'indexing'
? 'Creating vector index...'
: p.phase === 'ready'
? 'Embeddings complete'
: `${p.phase} (${p.percent}%)`,
},
});
});
// Skip nodes that already have embeddings — Kuzu forbids SET on vector-indexed properties.
let skipNodeIds: Set<string> | undefined;
try {
const rows = await executeQuery('MATCH (e:CodeEmbedding) RETURN e.nodeId AS nodeId');
if (rows && rows.length > 0) {
skipNodeIds = new Set(rows.map((r: any) => r.nodeId ?? r[0]).filter(Boolean));
console.log(
`[embed] ${skipNodeIds.size} nodes already embedded — skipping in incremental run`,
);
}
} catch (err: any) {
// Swallow only "table does not exist" — let real connection errors propagate.
// Log so ops can see this path fire if Kuzu ever changes error wording.
const msg = err?.message ?? '';
if (msg.includes('does not exist') || msg.includes('not found')) {
console.log(
`[embed] CodeEmbedding table not yet present — full embedding run (${msg})`,
);
} else {
throw err;
}
}
await runEmbeddingPipeline(
executeQuery,
executeWithReusedStatement,
(p) => {
embedJobManager.updateJob(job.id, {
progress: {
phase:
p.phase === 'ready' ? 'complete' : p.phase === 'error' ? 'failed' : p.phase,
percent: p.percent,
message:
p.phase === 'loading-model'
? 'Loading embedding model...'
: p.phase === 'embedding'
? `Embedding nodes (${p.percent}%)...`
: p.phase === 'indexing'
? 'Creating vector index...'
: p.phase === 'ready'
? 'Embeddings complete'
: `${p.phase} (${p.percent}%)`,
},
});
},
{}, // config: use defaults (runEmbeddingPipeline signature: executeQuery, executeWithReusedStatement, onProgress, config, skipNodeIds)
skipNodeIds,
);
});
clearTimeout(embedTimeout);