Compare commits
284
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
1d7782e4e1 | ||
|
|
7ce1371ad8 | ||
|
|
ae437dc30e | ||
|
|
832789f288 | ||
|
|
41b0d8dfad | ||
|
|
9760f966cb | ||
|
|
88c89c42e6 | ||
|
|
4af677e637 | ||
|
|
7999b6ba7b | ||
|
|
f0540b33fb | ||
|
|
fec47cbc32 | ||
|
|
a9d43d5680 | ||
|
|
e6aea1c82b | ||
|
|
3961ad01cf | ||
|
|
c437acf6bb | ||
|
|
2ede01dbff | ||
|
|
fd61cd2990 | ||
|
|
83ec256cec | ||
|
|
29795bb86e | ||
|
|
c8cd733815 | ||
|
|
5616f71fd2 | ||
|
|
2bd04fe2c4 | ||
|
|
4199c5e4f3 | ||
|
|
0d9a2ca6b6 | ||
|
|
88fa1a4e09 | ||
|
|
8956da61bb | ||
|
|
9ba2b22ac6 | ||
|
|
db6b302fee | ||
|
|
fcd2c3ff38 | ||
|
|
f4bb78c03a | ||
|
|
47f9b6c48d | ||
|
|
433f403fe2 | ||
|
|
519dc3eccc | ||
|
|
fcf8fb9bdf | ||
|
|
47fdad14ed | ||
|
|
ffabe857a3 | ||
|
|
1f4c4e77ab | ||
|
|
956dfd0bb4 | ||
|
|
57e087de49 | ||
|
|
fc6c114570 | ||
|
|
43d866c802 | ||
|
|
6b0c566392 | ||
|
|
96181fba05 | ||
|
|
208ade9b11 | ||
|
|
9f14edc226 | ||
|
|
cb1293b718 | ||
|
|
6c9b6eb0f6 | ||
|
|
3558cb8a7f | ||
|
|
9f2d1780d5 | ||
|
|
a57550815f | ||
|
|
ee78ebe64a | ||
|
|
141f864181 | ||
|
|
01fa5bf98e | ||
|
|
bbd95457df | ||
|
|
d43cc691f0 | ||
|
|
f90aabf9a8 | ||
|
|
beb5574d38 | ||
|
|
d48903cce6 | ||
|
|
a3fac2f672 | ||
|
|
9954f6fdfd | ||
|
|
90ca43f851 | ||
|
|
0a200a51cb | ||
|
|
c87cb81892 | ||
|
|
884b4acf84 | ||
|
|
b8ab23a838 | ||
|
|
e70a6d80c0 | ||
|
|
cd1c0ff7dc | ||
|
|
babf0f90d3 | ||
|
|
0a3cdce00e | ||
|
|
16b1a63134 | ||
|
|
99f0aaaea5 | ||
|
|
53b576776a | ||
|
|
712598a055 | ||
|
|
90c9153ccc | ||
|
|
dfe83f333e | ||
|
|
561a54a154 | ||
|
|
65bc99c448 | ||
|
|
0c8ec952ee | ||
|
|
907440cf0b | ||
|
|
843c561e9b | ||
|
|
4c8be50cb1 | ||
|
|
d07a69c3b1 | ||
|
|
d206bf6772 | ||
|
|
03156935cb | ||
|
|
22b5fce19e | ||
|
|
09e3609376 | ||
|
|
f0f384aab7 | ||
|
|
ef8b252bcc | ||
|
|
92a1086697 | ||
|
|
b272c6864c | ||
|
|
4031562f6b | ||
|
|
4dffd81b12 | ||
|
|
b75e380bd0 | ||
|
|
24421c2db1 | ||
|
|
e6eaf08382 | ||
|
|
f409fb7525 | ||
|
|
cc2b13e332 | ||
|
|
b909bbff70 | ||
|
|
56356b71db | ||
|
|
2c9e887a80 | ||
|
|
8fbfd09081 | ||
|
|
6f4281c946 | ||
|
|
60265c1d0d | ||
|
|
fa765beec1 | ||
|
|
61f2f6d954 | ||
|
|
2428e72bd2 | ||
|
|
76ed0fa53b | ||
|
|
2c17a4642c | ||
|
|
4f2e4ac24d | ||
|
|
1af9c028c5 | ||
|
|
fb20a3c752 | ||
|
|
f1fbe643df | ||
|
|
89a24d866e | ||
|
|
9baef90ae2 | ||
|
|
11575cf6c8 | ||
|
|
c2c694ac42 | ||
|
|
01ff2a7540 | ||
|
|
13b22e9879 | ||
|
|
c8480f899d | ||
|
|
1f7764c49b | ||
|
|
00758b102a | ||
|
|
abeb52e5e5 | ||
|
|
6c972079e0 | ||
|
|
cbfdae0303 | ||
|
|
66c1ffa370 | ||
|
|
88e0034771 | ||
|
|
0736bb23bc | ||
|
|
2ff4d93314 | ||
|
|
fff716dd92 | ||
|
|
56bc226a1c | ||
|
|
a6a1004e82 | ||
|
|
5b012c3351 | ||
|
|
862fdf9185 | ||
|
|
228c993bb7 | ||
|
|
c3a2815186 | ||
|
|
893f77ae89 | ||
|
|
d49c76ddc5 | ||
|
|
7dcafb647b | ||
|
|
5dcc567870 | ||
|
|
999fbf5b11 | ||
|
|
c79c717790 | ||
|
|
9dc00e47a7 | ||
|
|
c14a78a341 | ||
|
|
7a0a83e45c | ||
|
|
f2edbb4f82 | ||
|
|
1d27ad09a2 | ||
|
|
68d4c48aba | ||
|
|
bc771574d8 | ||
|
|
5d6e15eea3 | ||
|
|
eb74eb8590 | ||
|
|
700c9d16e4 | ||
|
|
19ff84fa31 | ||
|
|
2fe03d2a21 | ||
|
|
3e29f4e4b9 | ||
|
|
71353512a1 | ||
|
|
00c5126b24 | ||
|
|
06994e474a | ||
|
|
c507b4b197 | ||
|
|
b6947b0c02 | ||
|
|
e9ccec1a52 | ||
|
|
0122d9e694 | ||
|
|
00e2476eca | ||
|
|
217efcf015 | ||
|
|
7c72cefd8d | ||
|
|
58f67d07f7 | ||
|
|
a4863605e1 | ||
|
|
fc58c415f7 | ||
|
|
790d1d5b0f | ||
|
|
8273324f3c | ||
|
|
7b71b64427 | ||
|
|
e6b8edc1ac | ||
|
|
a7b8c302d4 | ||
|
|
5769872b70 | ||
|
|
60c93d7d4a | ||
|
|
b0f25e216d | ||
|
|
84f07e83ee | ||
|
|
1e19986ef3 | ||
|
|
973c7bfbf0 | ||
|
|
c0b4098c4e | ||
|
|
11a3d0515c | ||
|
|
e0a6c40b45 | ||
|
|
aa1bab597b | ||
|
|
60ede20a11 | ||
|
|
fb5270c260 | ||
|
|
604b575e4b | ||
|
|
02dfab578c | ||
|
|
1326490a5b | ||
|
|
b48cfe9894 | ||
|
|
c1703fc0a9 | ||
|
|
480fae933b | ||
|
|
3879490817 | ||
|
|
50dbd03779 | ||
|
|
1003d8b6a5 | ||
|
|
74b9701509 | ||
|
|
f0132c1077 | ||
|
|
f6b92d4f13 | ||
|
|
64b7ff0061 | ||
|
|
f2d3df48f6 | ||
|
|
5fa73bafdf | ||
|
|
fbff6d08c0 | ||
|
|
f83ef56ccb | ||
|
|
530b4be9ee | ||
|
|
6c18ae08f7 | ||
|
|
5a5850832c | ||
|
|
62242d5f44 | ||
|
|
6e38db879e | ||
|
|
0999595444 | ||
|
|
649ad80dbb | ||
|
|
3dbe08fab6 | ||
|
|
1afe9166aa | ||
|
|
03bfa3c4d9 | ||
|
|
74c0e462c3 | ||
|
|
7376e92063 | ||
|
|
1be910f54a | ||
|
|
fa9ba8925c | ||
|
|
8efc272609 | ||
|
|
c990d7e6c6 | ||
|
|
b4fbf33bd6 | ||
|
|
892e1d6088 | ||
|
|
c2bd8667a3 | ||
|
|
1952c2c346 | ||
|
|
c4eaf45ab1 | ||
|
|
0796e1e68c | ||
|
|
9d5ec5d19a | ||
|
|
4de40e4011 | ||
|
|
20e8c52028 | ||
|
|
2868da5ddb | ||
|
|
3db47f7ee5 | ||
|
|
821871cec1 | ||
|
|
f9a54cd588 | ||
|
|
8c6b064d18 | ||
|
|
84ef6524bc | ||
|
|
5674b2201d | ||
|
|
6915a9350b | ||
|
|
76e0e5a35a | ||
|
|
8e7d976c2a | ||
|
|
46b4b7e157 | ||
|
|
cbeb0e231a | ||
|
|
3431edcea0 | ||
|
|
40cb863cb4 | ||
|
|
ee95808478 | ||
|
|
3e3ea86ce4 | ||
|
|
1a52d05131 | ||
|
|
3d64e26f8f | ||
|
|
3576802574 | ||
|
|
20ebd6b781 | ||
|
|
8a100a76d3 | ||
|
|
c129e71ee7 | ||
|
|
80eff73459 | ||
|
|
48c8e6fe57 | ||
|
|
2eca3e0da3 | ||
|
|
e046bf734d | ||
|
|
b30248f969 | ||
|
|
eb48c7352e | ||
|
|
29db66c304 | ||
|
|
b7c582de76 | ||
|
|
508402fd4a | ||
|
|
da63281a5a | ||
|
|
c758f4eaf0 | ||
|
|
fd507a19ae | ||
|
|
799de20172 | ||
|
|
2be88ae1f8 | ||
|
|
2915e60630 | ||
|
|
019ed3ff85 | ||
|
|
43f525d056 | ||
|
|
e2a8bfa5ab | ||
|
|
ee6753bf05 | ||
|
|
1b8c3c77af | ||
|
|
a7fc9d2f88 | ||
|
|
0074fd71ff | ||
|
|
de935a4f4c | ||
|
|
8c41970631 | ||
|
|
5c3a32d0c6 | ||
|
|
a8b3c6b23f | ||
|
|
15caf1e014 | ||
|
|
1ed34a0007 | ||
|
|
7a4bc9a260 | ||
|
|
3872a73875 | ||
|
|
f557716998 | ||
|
|
ae8a76511d | ||
|
|
e803e7e9d6 | ||
|
|
c37b63ae8b | ||
|
|
984316d260 | ||
|
|
73590b2862 |
@@ -0,0 +1,82 @@
|
||||
---
|
||||
name: gitnexus-cli
|
||||
description: "Use when the user needs to run GitNexus CLI commands like analyze/index a repo, check status, clean the index, generate a wiki, or list indexed repos. Examples: \"Index this repo\", \"Reanalyze the codebase\", \"Generate a wiki\""
|
||||
---
|
||||
|
||||
# GitNexus CLI Commands
|
||||
|
||||
All commands work via `npx` — no global install required.
|
||||
|
||||
## Commands
|
||||
|
||||
### analyze — Build or refresh the index
|
||||
|
||||
```bash
|
||||
npx gitnexus analyze
|
||||
```
|
||||
|
||||
Run from the project root. This parses all source files, builds the knowledge graph, writes it to `.gitnexus/`, and generates CLAUDE.md / AGENTS.md context files.
|
||||
|
||||
| Flag | Effect |
|
||||
| -------------- | ---------------------------------------------------------------- |
|
||||
| `--force` | Force full re-index even if up to date |
|
||||
| `--embeddings` | Enable embedding generation for semantic search (off by default) |
|
||||
|
||||
**When to run:** First time in a project, after major code changes, or when `gitnexus://repo/{name}/context` reports the index is stale. In Claude Code, a PostToolUse hook runs `analyze` automatically after `git commit` and `git merge`, preserving embeddings if previously generated.
|
||||
|
||||
### status — Check index freshness
|
||||
|
||||
```bash
|
||||
npx gitnexus status
|
||||
```
|
||||
|
||||
Shows whether the current repo has a GitNexus index, when it was last updated, and symbol/relationship counts. Use this to check if re-indexing is needed.
|
||||
|
||||
### clean — Delete the index
|
||||
|
||||
```bash
|
||||
npx gitnexus clean
|
||||
```
|
||||
|
||||
Deletes the `.gitnexus/` directory and unregisters the repo from the global registry. Use before re-indexing if the index is corrupt or after removing GitNexus from a project.
|
||||
|
||||
| Flag | Effect |
|
||||
| --------- | ------------------------------------------------- |
|
||||
| `--force` | Skip confirmation prompt |
|
||||
| `--all` | Clean all indexed repos, not just the current one |
|
||||
|
||||
### wiki — Generate documentation from the graph
|
||||
|
||||
```bash
|
||||
npx gitnexus wiki
|
||||
```
|
||||
|
||||
Generates repository documentation from the knowledge graph using an LLM. Requires an API key (saved to `~/.gitnexus/config.json` on first use).
|
||||
|
||||
| Flag | Effect |
|
||||
| ------------------- | ----------------------------------------- |
|
||||
| `--force` | Force full regeneration |
|
||||
| `--model <model>` | LLM model (default: minimax/minimax-m2.5) |
|
||||
| `--base-url <url>` | LLM API base URL |
|
||||
| `--api-key <key>` | LLM API key |
|
||||
| `--concurrency <n>` | Parallel LLM calls (default: 3) |
|
||||
| `--gist` | Publish wiki as a public GitHub Gist |
|
||||
|
||||
### list — Show all indexed repos
|
||||
|
||||
```bash
|
||||
npx gitnexus list
|
||||
```
|
||||
|
||||
Lists all repositories registered in `~/.gitnexus/registry.json`. The MCP `list_repos` tool provides the same information.
|
||||
|
||||
## After Indexing
|
||||
|
||||
1. **Read `gitnexus://repo/{name}/context`** to verify the index loaded
|
||||
2. Use the other GitNexus skills (`exploring`, `debugging`, `impact-analysis`, `refactoring`) for your task
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
- **"Not inside a git repository"**: Run from a directory inside a git repo
|
||||
- **Index is stale after re-analyzing**: Restart Claude Code to reload the MCP server
|
||||
- **Embeddings slow**: Omit `--embeddings` (it's off by default) or set `OPENAI_API_KEY` for faster API-based embedding
|
||||
@@ -1,85 +1,89 @@
|
||||
---
|
||||
name: gitnexus-debugging
|
||||
description: "Use when the user is debugging a bug, tracing an error, or asking why something fails. Examples: \"Why is X failing?\", \"Where does this error come from?\", \"Trace this bug\""
|
||||
---
|
||||
|
||||
# Debugging with GitNexus
|
||||
|
||||
## When to Use
|
||||
- "Why is this function failing?"
|
||||
- "Trace where this error comes from"
|
||||
- "Who calls this method?"
|
||||
- "This endpoint returns 500"
|
||||
- Investigating bugs, errors, or unexpected behavior
|
||||
|
||||
## Workflow
|
||||
|
||||
```
|
||||
1. gitnexus_query({query: "<error or symptom>"}) → Find related execution flows
|
||||
2. gitnexus_context({name: "<suspect>"}) → See callers/callees/processes
|
||||
3. READ gitnexus://repo/{name}/process/{name} → Trace execution flow
|
||||
4. gitnexus_cypher({query: "MATCH path..."}) → Custom traces if needed
|
||||
```
|
||||
|
||||
> If "Index is stale" → run `npx gitnexus analyze` in terminal.
|
||||
|
||||
## Checklist
|
||||
|
||||
```
|
||||
- [ ] Understand the symptom (error message, unexpected behavior)
|
||||
- [ ] gitnexus_query for error text or related code
|
||||
- [ ] Identify the suspect function from returned processes
|
||||
- [ ] gitnexus_context to see callers and callees
|
||||
- [ ] Trace execution flow via process resource if applicable
|
||||
- [ ] gitnexus_cypher for custom call chain traces if needed
|
||||
- [ ] Read source files to confirm root cause
|
||||
```
|
||||
|
||||
## Debugging Patterns
|
||||
|
||||
| Symptom | GitNexus Approach |
|
||||
|---------|-------------------|
|
||||
| Error message | `gitnexus_query` for error text → `context` on throw sites |
|
||||
| Wrong return value | `context` on the function → trace callees for data flow |
|
||||
| Intermittent failure | `context` → look for external calls, async deps |
|
||||
| Performance issue | `context` → find symbols with many callers (hot paths) |
|
||||
| Recent regression | `detect_changes` to see what your changes affect |
|
||||
|
||||
## Tools
|
||||
|
||||
**gitnexus_query** — find code related to error:
|
||||
```
|
||||
gitnexus_query({query: "payment validation error"})
|
||||
→ Processes: CheckoutFlow, ErrorHandling
|
||||
→ Symbols: validatePayment, handlePaymentError, PaymentException
|
||||
```
|
||||
|
||||
**gitnexus_context** — full context for a suspect:
|
||||
```
|
||||
gitnexus_context({name: "validatePayment"})
|
||||
→ Incoming calls: processCheckout, webhookHandler
|
||||
→ Outgoing calls: verifyCard, fetchRates (external API!)
|
||||
→ Processes: CheckoutFlow (step 3/7)
|
||||
```
|
||||
|
||||
**gitnexus_cypher** — custom call chain traces:
|
||||
```cypher
|
||||
MATCH path = (a)-[:CodeRelation {type: 'CALLS'}*1..2]->(b:Function {name: "validatePayment"})
|
||||
RETURN [n IN nodes(path) | n.name] AS chain
|
||||
```
|
||||
|
||||
## Example: "Payment endpoint returns 500 intermittently"
|
||||
|
||||
```
|
||||
1. gitnexus_query({query: "payment error handling"})
|
||||
→ Processes: CheckoutFlow, ErrorHandling
|
||||
→ Symbols: validatePayment, handlePaymentError
|
||||
|
||||
2. gitnexus_context({name: "validatePayment"})
|
||||
→ Outgoing calls: verifyCard, fetchRates (external API!)
|
||||
|
||||
3. READ gitnexus://repo/my-app/process/CheckoutFlow
|
||||
→ Step 3: validatePayment → calls fetchRates (external)
|
||||
|
||||
4. Root cause: fetchRates calls external API without proper timeout
|
||||
```
|
||||
---
|
||||
name: gitnexus-debugging
|
||||
description: "Use when the user is debugging a bug, tracing an error, or asking why something fails. Examples: \"Why is X failing?\", \"Where does this error come from?\", \"Trace this bug\""
|
||||
---
|
||||
|
||||
# Debugging with GitNexus
|
||||
|
||||
## When to Use
|
||||
|
||||
- "Why is this function failing?"
|
||||
- "Trace where this error comes from"
|
||||
- "Who calls this method?"
|
||||
- "This endpoint returns 500"
|
||||
- Investigating bugs, errors, or unexpected behavior
|
||||
|
||||
## Workflow
|
||||
|
||||
```
|
||||
1. gitnexus_query({query: "<error or symptom>"}) → Find related execution flows
|
||||
2. gitnexus_context({name: "<suspect>"}) → See callers/callees/processes
|
||||
3. READ gitnexus://repo/{name}/process/{name} → Trace execution flow
|
||||
4. gitnexus_cypher({query: "MATCH path..."}) → Custom traces if needed
|
||||
```
|
||||
|
||||
> If "Index is stale" → run `npx gitnexus analyze` in terminal.
|
||||
|
||||
## Checklist
|
||||
|
||||
```
|
||||
- [ ] Understand the symptom (error message, unexpected behavior)
|
||||
- [ ] gitnexus_query for error text or related code
|
||||
- [ ] Identify the suspect function from returned processes
|
||||
- [ ] gitnexus_context to see callers and callees
|
||||
- [ ] Trace execution flow via process resource if applicable
|
||||
- [ ] gitnexus_cypher for custom call chain traces if needed
|
||||
- [ ] Read source files to confirm root cause
|
||||
```
|
||||
|
||||
## Debugging Patterns
|
||||
|
||||
| Symptom | GitNexus Approach |
|
||||
| -------------------- | ---------------------------------------------------------- |
|
||||
| Error message | `gitnexus_query` for error text → `context` on throw sites |
|
||||
| Wrong return value | `context` on the function → trace callees for data flow |
|
||||
| Intermittent failure | `context` → look for external calls, async deps |
|
||||
| Performance issue | `context` → find symbols with many callers (hot paths) |
|
||||
| Recent regression | `detect_changes` to see what your changes affect |
|
||||
|
||||
## Tools
|
||||
|
||||
**gitnexus_query** — find code related to error:
|
||||
|
||||
```
|
||||
gitnexus_query({query: "payment validation error"})
|
||||
→ Processes: CheckoutFlow, ErrorHandling
|
||||
→ Symbols: validatePayment, handlePaymentError, PaymentException
|
||||
```
|
||||
|
||||
**gitnexus_context** — full context for a suspect:
|
||||
|
||||
```
|
||||
gitnexus_context({name: "validatePayment"})
|
||||
→ Incoming calls: processCheckout, webhookHandler
|
||||
→ Outgoing calls: verifyCard, fetchRates (external API!)
|
||||
→ Processes: CheckoutFlow (step 3/7)
|
||||
```
|
||||
|
||||
**gitnexus_cypher** — custom call chain traces:
|
||||
|
||||
```cypher
|
||||
MATCH path = (a)-[:CodeRelation {type: 'CALLS'}*1..2]->(b:Function {name: "validatePayment"})
|
||||
RETURN [n IN nodes(path) | n.name] AS chain
|
||||
```
|
||||
|
||||
## Example: "Payment endpoint returns 500 intermittently"
|
||||
|
||||
```
|
||||
1. gitnexus_query({query: "payment error handling"})
|
||||
→ Processes: CheckoutFlow, ErrorHandling
|
||||
→ Symbols: validatePayment, handlePaymentError
|
||||
|
||||
2. gitnexus_context({name: "validatePayment"})
|
||||
→ Outgoing calls: verifyCard, fetchRates (external API!)
|
||||
|
||||
3. READ gitnexus://repo/my-app/process/CheckoutFlow
|
||||
→ Step 3: validatePayment → calls fetchRates (external)
|
||||
|
||||
4. Root cause: fetchRates calls external API without proper timeout
|
||||
```
|
||||
|
||||
@@ -1,75 +1,78 @@
|
||||
---
|
||||
name: gitnexus-exploring
|
||||
description: "Use when the user asks how code works, wants to understand architecture, trace execution flows, or explore unfamiliar parts of the codebase. Examples: \"How does X work?\", \"What calls this function?\", \"Show me the auth flow\""
|
||||
---
|
||||
|
||||
# Exploring Codebases with GitNexus
|
||||
|
||||
## When to Use
|
||||
- "How does authentication work?"
|
||||
- "What's the project structure?"
|
||||
- "Show me the main components"
|
||||
- "Where is the database logic?"
|
||||
- Understanding code you haven't seen before
|
||||
|
||||
## Workflow
|
||||
|
||||
```
|
||||
1. READ gitnexus://repos → Discover indexed repos
|
||||
2. READ gitnexus://repo/{name}/context → Codebase overview, check staleness
|
||||
3. gitnexus_query({query: "<what you want to understand>"}) → Find related execution flows
|
||||
4. gitnexus_context({name: "<symbol>"}) → Deep dive on specific symbol
|
||||
5. READ gitnexus://repo/{name}/process/{name} → Trace full execution flow
|
||||
```
|
||||
|
||||
> If step 2 says "Index is stale" → run `npx gitnexus analyze` in terminal.
|
||||
|
||||
## Checklist
|
||||
|
||||
```
|
||||
- [ ] READ gitnexus://repo/{name}/context
|
||||
- [ ] gitnexus_query for the concept you want to understand
|
||||
- [ ] Review returned processes (execution flows)
|
||||
- [ ] gitnexus_context on key symbols for callers/callees
|
||||
- [ ] READ process resource for full execution traces
|
||||
- [ ] Read source files for implementation details
|
||||
```
|
||||
|
||||
## Resources
|
||||
|
||||
| Resource | What you get |
|
||||
|----------|-------------|
|
||||
| `gitnexus://repo/{name}/context` | Stats, staleness warning (~150 tokens) |
|
||||
| `gitnexus://repo/{name}/clusters` | All functional areas with cohesion scores (~300 tokens) |
|
||||
| `gitnexus://repo/{name}/cluster/{name}` | Area members with file paths (~500 tokens) |
|
||||
| `gitnexus://repo/{name}/process/{name}` | Step-by-step execution trace (~200 tokens) |
|
||||
|
||||
## Tools
|
||||
|
||||
**gitnexus_query** — find execution flows related to a concept:
|
||||
```
|
||||
gitnexus_query({query: "payment processing"})
|
||||
→ Processes: CheckoutFlow, RefundFlow, WebhookHandler
|
||||
→ Symbols grouped by flow with file locations
|
||||
```
|
||||
|
||||
**gitnexus_context** — 360-degree view of a symbol:
|
||||
```
|
||||
gitnexus_context({name: "validateUser"})
|
||||
→ Incoming calls: loginHandler, apiMiddleware
|
||||
→ Outgoing calls: checkToken, getUserById
|
||||
→ Processes: LoginFlow (step 2/5), TokenRefresh (step 1/3)
|
||||
```
|
||||
|
||||
## Example: "How does payment processing work?"
|
||||
|
||||
```
|
||||
1. READ gitnexus://repo/my-app/context → 918 symbols, 45 processes
|
||||
2. gitnexus_query({query: "payment processing"})
|
||||
→ CheckoutFlow: processPayment → validateCard → chargeStripe
|
||||
→ RefundFlow: initiateRefund → calculateRefund → processRefund
|
||||
3. gitnexus_context({name: "processPayment"})
|
||||
→ Incoming: checkoutHandler, webhookHandler
|
||||
→ Outgoing: validateCard, chargeStripe, saveTransaction
|
||||
4. Read src/payments/processor.ts for implementation details
|
||||
```
|
||||
---
|
||||
name: gitnexus-exploring
|
||||
description: "Use when the user asks how code works, wants to understand architecture, trace execution flows, or explore unfamiliar parts of the codebase. Examples: \"How does X work?\", \"What calls this function?\", \"Show me the auth flow\""
|
||||
---
|
||||
|
||||
# Exploring Codebases with GitNexus
|
||||
|
||||
## When to Use
|
||||
|
||||
- "How does authentication work?"
|
||||
- "What's the project structure?"
|
||||
- "Show me the main components"
|
||||
- "Where is the database logic?"
|
||||
- Understanding code you haven't seen before
|
||||
|
||||
## Workflow
|
||||
|
||||
```
|
||||
1. READ gitnexus://repos → Discover indexed repos
|
||||
2. READ gitnexus://repo/{name}/context → Codebase overview, check staleness
|
||||
3. gitnexus_query({query: "<what you want to understand>"}) → Find related execution flows
|
||||
4. gitnexus_context({name: "<symbol>"}) → Deep dive on specific symbol
|
||||
5. READ gitnexus://repo/{name}/process/{name} → Trace full execution flow
|
||||
```
|
||||
|
||||
> If step 2 says "Index is stale" → run `npx gitnexus analyze` in terminal.
|
||||
|
||||
## Checklist
|
||||
|
||||
```
|
||||
- [ ] READ gitnexus://repo/{name}/context
|
||||
- [ ] gitnexus_query for the concept you want to understand
|
||||
- [ ] Review returned processes (execution flows)
|
||||
- [ ] gitnexus_context on key symbols for callers/callees
|
||||
- [ ] READ process resource for full execution traces
|
||||
- [ ] Read source files for implementation details
|
||||
```
|
||||
|
||||
## Resources
|
||||
|
||||
| Resource | What you get |
|
||||
| --------------------------------------- | ------------------------------------------------------- |
|
||||
| `gitnexus://repo/{name}/context` | Stats, staleness warning (~150 tokens) |
|
||||
| `gitnexus://repo/{name}/clusters` | All functional areas with cohesion scores (~300 tokens) |
|
||||
| `gitnexus://repo/{name}/cluster/{name}` | Area members with file paths (~500 tokens) |
|
||||
| `gitnexus://repo/{name}/process/{name}` | Step-by-step execution trace (~200 tokens) |
|
||||
|
||||
## Tools
|
||||
|
||||
**gitnexus_query** — find execution flows related to a concept:
|
||||
|
||||
```
|
||||
gitnexus_query({query: "payment processing"})
|
||||
→ Processes: CheckoutFlow, RefundFlow, WebhookHandler
|
||||
→ Symbols grouped by flow with file locations
|
||||
```
|
||||
|
||||
**gitnexus_context** — 360-degree view of a symbol:
|
||||
|
||||
```
|
||||
gitnexus_context({name: "validateUser"})
|
||||
→ Incoming calls: loginHandler, apiMiddleware
|
||||
→ Outgoing calls: checkToken, getUserById
|
||||
→ Processes: LoginFlow (step 2/5), TokenRefresh (step 1/3)
|
||||
```
|
||||
|
||||
## Example: "How does payment processing work?"
|
||||
|
||||
```
|
||||
1. READ gitnexus://repo/my-app/context → 918 symbols, 45 processes
|
||||
2. gitnexus_query({query: "payment processing"})
|
||||
→ CheckoutFlow: processPayment → validateCard → chargeStripe
|
||||
→ RefundFlow: initiateRefund → calculateRefund → processRefund
|
||||
3. gitnexus_context({name: "processPayment"})
|
||||
→ Incoming: checkoutHandler, webhookHandler
|
||||
→ Outgoing: validateCard, chargeStripe, saveTransaction
|
||||
4. Read src/payments/processor.ts for implementation details
|
||||
```
|
||||
|
||||
@@ -0,0 +1,64 @@
|
||||
---
|
||||
name: gitnexus-guide
|
||||
description: "Use when the user asks about GitNexus itself — available tools, how to query the knowledge graph, MCP resources, graph schema, or workflow reference. Examples: \"What GitNexus tools are available?\", \"How do I use GitNexus?\""
|
||||
---
|
||||
|
||||
# GitNexus Guide
|
||||
|
||||
Quick reference for all GitNexus MCP tools, resources, and the knowledge graph schema.
|
||||
|
||||
## Always Start Here
|
||||
|
||||
For any task involving code understanding, debugging, impact analysis, or refactoring:
|
||||
|
||||
1. **Read `gitnexus://repo/{name}/context`** — codebase overview + check index freshness
|
||||
2. **Match your task to a skill below** and **read that skill file**
|
||||
3. **Follow the skill's workflow and checklist**
|
||||
|
||||
> If step 1 warns the index is stale, run `npx gitnexus analyze` in the terminal first.
|
||||
|
||||
## Skills
|
||||
|
||||
| Task | Skill to read |
|
||||
| -------------------------------------------- | ------------------- |
|
||||
| Understand architecture / "How does X work?" | `gitnexus-exploring` |
|
||||
| Blast radius / "What breaks if I change X?" | `gitnexus-impact-analysis` |
|
||||
| Trace bugs / "Why is X failing?" | `gitnexus-debugging` |
|
||||
| Rename / extract / split / refactor | `gitnexus-refactoring` |
|
||||
| Tools, resources, schema reference | `gitnexus-guide` (this file) |
|
||||
| Index, status, clean, wiki CLI commands | `gitnexus-cli` |
|
||||
|
||||
## Tools Reference
|
||||
|
||||
| Tool | What it gives you |
|
||||
| ---------------- | ------------------------------------------------------------------------ |
|
||||
| `query` | Process-grouped code intelligence — execution flows related to a concept |
|
||||
| `context` | 360-degree symbol view — categorized refs, processes it participates in |
|
||||
| `impact` | Symbol blast radius — what breaks at depth 1/2/3 with confidence |
|
||||
| `detect_changes` | Git-diff impact — what do your current changes affect |
|
||||
| `rename` | Multi-file coordinated rename with confidence-tagged edits |
|
||||
| `cypher` | Raw graph queries (read `gitnexus://repo/{name}/schema` first) |
|
||||
| `list_repos` | Discover indexed repos |
|
||||
|
||||
## Resources Reference
|
||||
|
||||
Lightweight reads (~100-500 tokens) for navigation:
|
||||
|
||||
| Resource | Content |
|
||||
| ---------------------------------------------- | ----------------------------------------- |
|
||||
| `gitnexus://repo/{name}/context` | Stats, staleness check |
|
||||
| `gitnexus://repo/{name}/clusters` | All functional areas with cohesion scores |
|
||||
| `gitnexus://repo/{name}/cluster/{clusterName}` | Area members |
|
||||
| `gitnexus://repo/{name}/processes` | All execution flows |
|
||||
| `gitnexus://repo/{name}/process/{processName}` | Step-by-step trace |
|
||||
| `gitnexus://repo/{name}/schema` | Graph schema for Cypher |
|
||||
|
||||
## Graph Schema
|
||||
|
||||
**Nodes:** File, Function, Class, Interface, Method, Community, Process
|
||||
**Edges (via CodeRelation.type):** CALLS, IMPORTS, EXTENDS, IMPLEMENTS, DEFINES, MEMBER_OF, STEP_IN_PROCESS
|
||||
|
||||
```cypher
|
||||
MATCH (caller)-[:CodeRelation {type: 'CALLS'}]->(f:Function {name: "myFunc"})
|
||||
RETURN caller.name, caller.filePath
|
||||
```
|
||||
@@ -1,94 +1,97 @@
|
||||
---
|
||||
name: gitnexus-impact-analysis
|
||||
description: "Use when the user wants to know what will break if they change something, or needs safety analysis before editing code. Examples: \"Is it safe to change X?\", \"What depends on this?\", \"What will break?\""
|
||||
---
|
||||
|
||||
# Impact Analysis with GitNexus
|
||||
|
||||
## When to Use
|
||||
- "Is it safe to change this function?"
|
||||
- "What will break if I modify X?"
|
||||
- "Show me the blast radius"
|
||||
- "Who uses this code?"
|
||||
- Before making non-trivial code changes
|
||||
- Before committing — to understand what your changes affect
|
||||
|
||||
## Workflow
|
||||
|
||||
```
|
||||
1. gitnexus_impact({target: "X", direction: "upstream"}) → What depends on this
|
||||
2. READ gitnexus://repo/{name}/processes → Check affected execution flows
|
||||
3. gitnexus_detect_changes() → Map current git changes to affected flows
|
||||
4. Assess risk and report to user
|
||||
```
|
||||
|
||||
> If "Index is stale" → run `npx gitnexus analyze` in terminal.
|
||||
|
||||
## Checklist
|
||||
|
||||
```
|
||||
- [ ] gitnexus_impact({target, direction: "upstream"}) to find dependents
|
||||
- [ ] Review d=1 items first (these WILL BREAK)
|
||||
- [ ] Check high-confidence (>0.8) dependencies
|
||||
- [ ] READ processes to check affected execution flows
|
||||
- [ ] gitnexus_detect_changes() for pre-commit check
|
||||
- [ ] Assess risk level and report to user
|
||||
```
|
||||
|
||||
## Understanding Output
|
||||
|
||||
| Depth | Risk Level | Meaning |
|
||||
|-------|-----------|---------|
|
||||
| d=1 | **WILL BREAK** | Direct callers/importers |
|
||||
| d=2 | LIKELY AFFECTED | Indirect dependencies |
|
||||
| d=3 | MAY NEED TESTING | Transitive effects |
|
||||
|
||||
## Risk Assessment
|
||||
|
||||
| Affected | Risk |
|
||||
|----------|------|
|
||||
| <5 symbols, few processes | LOW |
|
||||
| 5-15 symbols, 2-5 processes | MEDIUM |
|
||||
| >15 symbols or many processes | HIGH |
|
||||
| Critical path (auth, payments) | CRITICAL |
|
||||
|
||||
## Tools
|
||||
|
||||
**gitnexus_impact** — the primary tool for symbol blast radius:
|
||||
```
|
||||
gitnexus_impact({
|
||||
target: "validateUser",
|
||||
direction: "upstream",
|
||||
minConfidence: 0.8,
|
||||
maxDepth: 3
|
||||
})
|
||||
|
||||
→ d=1 (WILL BREAK):
|
||||
- loginHandler (src/auth/login.ts:42) [CALLS, 100%]
|
||||
- apiMiddleware (src/api/middleware.ts:15) [CALLS, 100%]
|
||||
|
||||
→ d=2 (LIKELY AFFECTED):
|
||||
- authRouter (src/routes/auth.ts:22) [CALLS, 95%]
|
||||
```
|
||||
|
||||
**gitnexus_detect_changes** — git-diff based impact analysis:
|
||||
```
|
||||
gitnexus_detect_changes({scope: "staged"})
|
||||
|
||||
→ Changed: 5 symbols in 3 files
|
||||
→ Affected: LoginFlow, TokenRefresh, APIMiddlewarePipeline
|
||||
→ Risk: MEDIUM
|
||||
```
|
||||
|
||||
## Example: "What breaks if I change validateUser?"
|
||||
|
||||
```
|
||||
1. gitnexus_impact({target: "validateUser", direction: "upstream"})
|
||||
→ d=1: loginHandler, apiMiddleware (WILL BREAK)
|
||||
→ d=2: authRouter, sessionManager (LIKELY AFFECTED)
|
||||
|
||||
2. READ gitnexus://repo/my-app/processes
|
||||
→ LoginFlow and TokenRefresh touch validateUser
|
||||
|
||||
3. Risk: 2 direct callers, 2 processes = MEDIUM
|
||||
```
|
||||
---
|
||||
name: gitnexus-impact-analysis
|
||||
description: "Use when the user wants to know what will break if they change something, or needs safety analysis before editing code. Examples: \"Is it safe to change X?\", \"What depends on this?\", \"What will break?\""
|
||||
---
|
||||
|
||||
# Impact Analysis with GitNexus
|
||||
|
||||
## When to Use
|
||||
|
||||
- "Is it safe to change this function?"
|
||||
- "What will break if I modify X?"
|
||||
- "Show me the blast radius"
|
||||
- "Who uses this code?"
|
||||
- Before making non-trivial code changes
|
||||
- Before committing — to understand what your changes affect
|
||||
|
||||
## Workflow
|
||||
|
||||
```
|
||||
1. gitnexus_impact({target: "X", direction: "upstream"}) → What depends on this
|
||||
2. READ gitnexus://repo/{name}/processes → Check affected execution flows
|
||||
3. gitnexus_detect_changes() → Map current git changes to affected flows
|
||||
4. Assess risk and report to user
|
||||
```
|
||||
|
||||
> If "Index is stale" → run `npx gitnexus analyze` in terminal.
|
||||
|
||||
## Checklist
|
||||
|
||||
```
|
||||
- [ ] gitnexus_impact({target, direction: "upstream"}) to find dependents
|
||||
- [ ] Review d=1 items first (these WILL BREAK)
|
||||
- [ ] Check high-confidence (>0.8) dependencies
|
||||
- [ ] READ processes to check affected execution flows
|
||||
- [ ] gitnexus_detect_changes() for pre-commit check
|
||||
- [ ] Assess risk level and report to user
|
||||
```
|
||||
|
||||
## Understanding Output
|
||||
|
||||
| Depth | Risk Level | Meaning |
|
||||
| ----- | ---------------- | ------------------------ |
|
||||
| d=1 | **WILL BREAK** | Direct callers/importers |
|
||||
| d=2 | LIKELY AFFECTED | Indirect dependencies |
|
||||
| d=3 | MAY NEED TESTING | Transitive effects |
|
||||
|
||||
## Risk Assessment
|
||||
|
||||
| Affected | Risk |
|
||||
| ------------------------------ | -------- |
|
||||
| <5 symbols, few processes | LOW |
|
||||
| 5-15 symbols, 2-5 processes | MEDIUM |
|
||||
| >15 symbols or many processes | HIGH |
|
||||
| Critical path (auth, payments) | CRITICAL |
|
||||
|
||||
## Tools
|
||||
|
||||
**gitnexus_impact** — the primary tool for symbol blast radius:
|
||||
|
||||
```
|
||||
gitnexus_impact({
|
||||
target: "validateUser",
|
||||
direction: "upstream",
|
||||
minConfidence: 0.8,
|
||||
maxDepth: 3
|
||||
})
|
||||
|
||||
→ d=1 (WILL BREAK):
|
||||
- loginHandler (src/auth/login.ts:42) [CALLS, 100%]
|
||||
- apiMiddleware (src/api/middleware.ts:15) [CALLS, 100%]
|
||||
|
||||
→ d=2 (LIKELY AFFECTED):
|
||||
- authRouter (src/routes/auth.ts:22) [CALLS, 95%]
|
||||
```
|
||||
|
||||
**gitnexus_detect_changes** — git-diff based impact analysis:
|
||||
|
||||
```
|
||||
gitnexus_detect_changes({scope: "staged"})
|
||||
|
||||
→ Changed: 5 symbols in 3 files
|
||||
→ Affected: LoginFlow, TokenRefresh, APIMiddlewarePipeline
|
||||
→ Risk: MEDIUM
|
||||
```
|
||||
|
||||
## Example: "What breaks if I change validateUser?"
|
||||
|
||||
```
|
||||
1. gitnexus_impact({target: "validateUser", direction: "upstream"})
|
||||
→ d=1: loginHandler, apiMiddleware (WILL BREAK)
|
||||
→ d=2: authRouter, sessionManager (LIKELY AFFECTED)
|
||||
|
||||
2. READ gitnexus://repo/my-app/processes
|
||||
→ LoginFlow and TokenRefresh touch validateUser
|
||||
|
||||
3. Risk: 2 direct callers, 2 processes = MEDIUM
|
||||
```
|
||||
|
||||
@@ -0,0 +1,163 @@
|
||||
---
|
||||
name: gitnexus-pr-review
|
||||
description: "Use when the user wants to review a pull request, understand what a PR changes, assess risk of merging, or check for missing test coverage. Examples: \"Review this PR\", \"What does PR #42 change?\", \"Is this PR safe to merge?\""
|
||||
---
|
||||
|
||||
# PR Review with GitNexus
|
||||
|
||||
## When to Use
|
||||
|
||||
- "Review this PR"
|
||||
- "What does PR #42 change?"
|
||||
- "Is this safe to merge?"
|
||||
- "What's the blast radius of this PR?"
|
||||
- "Are there missing tests for this PR?"
|
||||
- Reviewing someone else's code changes before merge
|
||||
|
||||
## Workflow
|
||||
|
||||
```
|
||||
1. gh pr diff <number> → Get the raw diff
|
||||
2. gitnexus_detect_changes({scope: "compare", base_ref: "main"}) → Map diff to affected flows
|
||||
3. For each changed symbol:
|
||||
gitnexus_impact({target: "<symbol>", direction: "upstream"}) → Blast radius per change
|
||||
4. gitnexus_context({name: "<key symbol>"}) → Understand callers/callees
|
||||
5. READ gitnexus://repo/{name}/processes → Check affected execution flows
|
||||
6. Summarize findings with risk assessment
|
||||
```
|
||||
|
||||
> If "Index is stale" → run `npx gitnexus analyze` in terminal before reviewing.
|
||||
|
||||
## Checklist
|
||||
|
||||
```
|
||||
- [ ] Fetch PR diff (gh pr diff or git diff base...head)
|
||||
- [ ] gitnexus_detect_changes to map changes to affected execution flows
|
||||
- [ ] gitnexus_impact on each non-trivial changed symbol
|
||||
- [ ] Review d=1 items (WILL BREAK) — are callers updated?
|
||||
- [ ] gitnexus_context on key changed symbols to understand full picture
|
||||
- [ ] Check if affected processes have test coverage
|
||||
- [ ] Assess overall risk level
|
||||
- [ ] Write review summary with findings
|
||||
```
|
||||
|
||||
## Review Dimensions
|
||||
|
||||
| Dimension | How GitNexus Helps |
|
||||
| --- | --- |
|
||||
| **Correctness** | `context` shows callers — are they all compatible with the change? |
|
||||
| **Blast radius** | `impact` shows d=1/d=2/d=3 dependents — anything missed? |
|
||||
| **Completeness** | `detect_changes` shows all affected flows — are they all handled? |
|
||||
| **Test coverage** | `impact({includeTests: true})` shows which tests touch changed code |
|
||||
| **Breaking changes** | d=1 upstream items that aren't updated in the PR = potential breakage |
|
||||
|
||||
## Risk Assessment
|
||||
|
||||
| Signal | Risk |
|
||||
| --- | --- |
|
||||
| Changes touch <3 symbols, 0-1 processes | LOW |
|
||||
| Changes touch 3-10 symbols, 2-5 processes | MEDIUM |
|
||||
| Changes touch >10 symbols or many processes | HIGH |
|
||||
| Changes touch auth, payments, or data integrity code | CRITICAL |
|
||||
| d=1 callers exist outside the PR diff | Potential breakage — flag it |
|
||||
|
||||
## Tools
|
||||
|
||||
**gitnexus_detect_changes** — map PR diff to affected execution flows:
|
||||
|
||||
```
|
||||
gitnexus_detect_changes({scope: "compare", base_ref: "main"})
|
||||
|
||||
→ Changed: 8 symbols in 4 files
|
||||
→ Affected processes: CheckoutFlow, RefundFlow, WebhookHandler
|
||||
→ Risk: MEDIUM
|
||||
```
|
||||
|
||||
**gitnexus_impact** — blast radius per changed symbol:
|
||||
|
||||
```
|
||||
gitnexus_impact({target: "validatePayment", direction: "upstream"})
|
||||
|
||||
→ d=1 (WILL BREAK):
|
||||
- processCheckout (src/checkout.ts:42) [CALLS, 100%]
|
||||
- webhookHandler (src/webhooks.ts:15) [CALLS, 100%]
|
||||
|
||||
→ d=2 (LIKELY AFFECTED):
|
||||
- checkoutRouter (src/routes/checkout.ts:22) [CALLS, 95%]
|
||||
```
|
||||
|
||||
**gitnexus_impact with tests** — check test coverage:
|
||||
|
||||
```
|
||||
gitnexus_impact({target: "validatePayment", direction: "upstream", includeTests: true})
|
||||
|
||||
→ Tests that cover this symbol:
|
||||
- validatePayment.test.ts [direct]
|
||||
- checkout.integration.test.ts [via processCheckout]
|
||||
```
|
||||
|
||||
**gitnexus_context** — understand a changed symbol's role:
|
||||
|
||||
```
|
||||
gitnexus_context({name: "validatePayment"})
|
||||
|
||||
→ Incoming calls: processCheckout, webhookHandler
|
||||
→ Outgoing calls: verifyCard, fetchRates
|
||||
→ Processes: CheckoutFlow (step 3/7), RefundFlow (step 1/5)
|
||||
```
|
||||
|
||||
## Example: "Review PR #42"
|
||||
|
||||
```
|
||||
1. gh pr diff 42 > /tmp/pr42.diff
|
||||
→ 4 files changed: payments.ts, checkout.ts, types.ts, utils.ts
|
||||
|
||||
2. gitnexus_detect_changes({scope: "compare", base_ref: "main"})
|
||||
→ Changed symbols: validatePayment, PaymentInput, formatAmount
|
||||
→ Affected processes: CheckoutFlow, RefundFlow
|
||||
→ Risk: MEDIUM
|
||||
|
||||
3. gitnexus_impact({target: "validatePayment", direction: "upstream"})
|
||||
→ d=1: processCheckout, webhookHandler (WILL BREAK)
|
||||
→ webhookHandler is NOT in the PR diff — potential breakage!
|
||||
|
||||
4. gitnexus_impact({target: "PaymentInput", direction: "upstream"})
|
||||
→ d=1: validatePayment (in PR), createPayment (NOT in PR)
|
||||
→ createPayment uses the old PaymentInput shape — breaking change!
|
||||
|
||||
5. gitnexus_context({name: "formatAmount"})
|
||||
→ Called by 12 functions — but change is backwards-compatible (added optional param)
|
||||
|
||||
6. Review summary:
|
||||
- MEDIUM risk — 3 changed symbols affect 2 execution flows
|
||||
- BUG: webhookHandler calls validatePayment but isn't updated for new signature
|
||||
- BUG: createPayment depends on PaymentInput type which changed
|
||||
- OK: formatAmount change is backwards-compatible
|
||||
- Tests: checkout.test.ts covers processCheckout path, but no webhook test
|
||||
```
|
||||
|
||||
## Review Output Format
|
||||
|
||||
Structure your review as:
|
||||
|
||||
```markdown
|
||||
## PR Review: <title>
|
||||
|
||||
**Risk: LOW / MEDIUM / HIGH / CRITICAL**
|
||||
|
||||
### Changes Summary
|
||||
- <N> symbols changed across <M> files
|
||||
- <P> execution flows affected
|
||||
|
||||
### Findings
|
||||
1. **[severity]** Description of finding
|
||||
- Evidence from GitNexus tools
|
||||
- Affected callers/flows
|
||||
|
||||
### Missing Coverage
|
||||
- Callers not updated in PR: ...
|
||||
- Untested flows: ...
|
||||
|
||||
### Recommendation
|
||||
APPROVE / REQUEST CHANGES / NEEDS DISCUSSION
|
||||
```
|
||||
@@ -1,113 +1,121 @@
|
||||
---
|
||||
name: gitnexus-refactoring
|
||||
description: "Use when the user wants to rename, extract, split, move, or restructure code safely. Examples: \"Rename this function\", \"Extract this into a module\", \"Refactor this class\", \"Move this to a separate file\""
|
||||
---
|
||||
|
||||
# Refactoring with GitNexus
|
||||
|
||||
## When to Use
|
||||
- "Rename this function safely"
|
||||
- "Extract this into a module"
|
||||
- "Split this service"
|
||||
- "Move this to a new file"
|
||||
- Any task involving renaming, extracting, splitting, or restructuring code
|
||||
|
||||
## Workflow
|
||||
|
||||
```
|
||||
1. gitnexus_impact({target: "X", direction: "upstream"}) → Map all dependents
|
||||
2. gitnexus_query({query: "X"}) → Find execution flows involving X
|
||||
3. gitnexus_context({name: "X"}) → See all incoming/outgoing refs
|
||||
4. Plan update order: interfaces → implementations → callers → tests
|
||||
```
|
||||
|
||||
> If "Index is stale" → run `npx gitnexus analyze` in terminal.
|
||||
|
||||
## Checklists
|
||||
|
||||
### Rename Symbol
|
||||
```
|
||||
- [ ] gitnexus_rename({symbol_name: "oldName", new_name: "newName", dry_run: true}) — preview all edits
|
||||
- [ ] Review graph edits (high confidence) and ast_search edits (review carefully)
|
||||
- [ ] If satisfied: gitnexus_rename({..., dry_run: false}) — apply edits
|
||||
- [ ] gitnexus_detect_changes() — verify only expected files changed
|
||||
- [ ] Run tests for affected processes
|
||||
```
|
||||
|
||||
### Extract Module
|
||||
```
|
||||
- [ ] gitnexus_context({name: target}) — see all incoming/outgoing refs
|
||||
- [ ] gitnexus_impact({target, direction: "upstream"}) — find all external callers
|
||||
- [ ] Define new module interface
|
||||
- [ ] Extract code, update imports
|
||||
- [ ] gitnexus_detect_changes() — verify affected scope
|
||||
- [ ] Run tests for affected processes
|
||||
```
|
||||
|
||||
### Split Function/Service
|
||||
```
|
||||
- [ ] gitnexus_context({name: target}) — understand all callees
|
||||
- [ ] Group callees by responsibility
|
||||
- [ ] gitnexus_impact({target, direction: "upstream"}) — map callers to update
|
||||
- [ ] Create new functions/services
|
||||
- [ ] Update callers
|
||||
- [ ] gitnexus_detect_changes() — verify affected scope
|
||||
- [ ] Run tests for affected processes
|
||||
```
|
||||
|
||||
## Tools
|
||||
|
||||
**gitnexus_rename** — automated multi-file rename:
|
||||
```
|
||||
gitnexus_rename({symbol_name: "validateUser", new_name: "authenticateUser", dry_run: true})
|
||||
→ 12 edits across 8 files
|
||||
→ 10 graph edits (high confidence), 2 ast_search edits (review)
|
||||
→ Changes: [{file_path, edits: [{line, old_text, new_text, confidence}]}]
|
||||
```
|
||||
|
||||
**gitnexus_impact** — map all dependents first:
|
||||
```
|
||||
gitnexus_impact({target: "validateUser", direction: "upstream"})
|
||||
→ d=1: loginHandler, apiMiddleware, testUtils
|
||||
→ Affected Processes: LoginFlow, TokenRefresh
|
||||
```
|
||||
|
||||
**gitnexus_detect_changes** — verify your changes after refactoring:
|
||||
```
|
||||
gitnexus_detect_changes({scope: "all"})
|
||||
→ Changed: 8 files, 12 symbols
|
||||
→ Affected processes: LoginFlow, TokenRefresh
|
||||
→ Risk: MEDIUM
|
||||
```
|
||||
|
||||
**gitnexus_cypher** — custom reference queries:
|
||||
```cypher
|
||||
MATCH (caller)-[:CodeRelation {type: 'CALLS'}]->(f:Function {name: "validateUser"})
|
||||
RETURN caller.name, caller.filePath ORDER BY caller.filePath
|
||||
```
|
||||
|
||||
## Risk Rules
|
||||
|
||||
| Risk Factor | Mitigation |
|
||||
|-------------|------------|
|
||||
| Many callers (>5) | Use gitnexus_rename for automated updates |
|
||||
| Cross-area refs | Use detect_changes after to verify scope |
|
||||
| String/dynamic refs | gitnexus_query to find them |
|
||||
| External/public API | Version and deprecate properly |
|
||||
|
||||
## Example: Rename `validateUser` to `authenticateUser`
|
||||
|
||||
```
|
||||
1. gitnexus_rename({symbol_name: "validateUser", new_name: "authenticateUser", dry_run: true})
|
||||
→ 12 edits: 10 graph (safe), 2 ast_search (review)
|
||||
→ Files: validator.ts, login.ts, middleware.ts, config.json...
|
||||
|
||||
2. Review ast_search edits (config.json: dynamic reference!)
|
||||
|
||||
3. gitnexus_rename({symbol_name: "validateUser", new_name: "authenticateUser", dry_run: false})
|
||||
→ Applied 12 edits across 8 files
|
||||
|
||||
4. gitnexus_detect_changes({scope: "all"})
|
||||
→ Affected: LoginFlow, TokenRefresh
|
||||
→ Risk: MEDIUM — run tests for these flows
|
||||
```
|
||||
---
|
||||
name: gitnexus-refactoring
|
||||
description: "Use when the user wants to rename, extract, split, move, or restructure code safely. Examples: \"Rename this function\", \"Extract this into a module\", \"Refactor this class\", \"Move this to a separate file\""
|
||||
---
|
||||
|
||||
# Refactoring with GitNexus
|
||||
|
||||
## When to Use
|
||||
|
||||
- "Rename this function safely"
|
||||
- "Extract this into a module"
|
||||
- "Split this service"
|
||||
- "Move this to a new file"
|
||||
- Any task involving renaming, extracting, splitting, or restructuring code
|
||||
|
||||
## Workflow
|
||||
|
||||
```
|
||||
1. gitnexus_impact({target: "X", direction: "upstream"}) → Map all dependents
|
||||
2. gitnexus_query({query: "X"}) → Find execution flows involving X
|
||||
3. gitnexus_context({name: "X"}) → See all incoming/outgoing refs
|
||||
4. Plan update order: interfaces → implementations → callers → tests
|
||||
```
|
||||
|
||||
> If "Index is stale" → run `npx gitnexus analyze` in terminal.
|
||||
|
||||
## Checklists
|
||||
|
||||
### Rename Symbol
|
||||
|
||||
```
|
||||
- [ ] gitnexus_rename({symbol_name: "oldName", new_name: "newName", dry_run: true}) — preview all edits
|
||||
- [ ] Review graph edits (high confidence) and ast_search edits (review carefully)
|
||||
- [ ] If satisfied: gitnexus_rename({..., dry_run: false}) — apply edits
|
||||
- [ ] gitnexus_detect_changes() — verify only expected files changed
|
||||
- [ ] Run tests for affected processes
|
||||
```
|
||||
|
||||
### Extract Module
|
||||
|
||||
```
|
||||
- [ ] gitnexus_context({name: target}) — see all incoming/outgoing refs
|
||||
- [ ] gitnexus_impact({target, direction: "upstream"}) — find all external callers
|
||||
- [ ] Define new module interface
|
||||
- [ ] Extract code, update imports
|
||||
- [ ] gitnexus_detect_changes() — verify affected scope
|
||||
- [ ] Run tests for affected processes
|
||||
```
|
||||
|
||||
### Split Function/Service
|
||||
|
||||
```
|
||||
- [ ] gitnexus_context({name: target}) — understand all callees
|
||||
- [ ] Group callees by responsibility
|
||||
- [ ] gitnexus_impact({target, direction: "upstream"}) — map callers to update
|
||||
- [ ] Create new functions/services
|
||||
- [ ] Update callers
|
||||
- [ ] gitnexus_detect_changes() — verify affected scope
|
||||
- [ ] Run tests for affected processes
|
||||
```
|
||||
|
||||
## Tools
|
||||
|
||||
**gitnexus_rename** — automated multi-file rename:
|
||||
|
||||
```
|
||||
gitnexus_rename({symbol_name: "validateUser", new_name: "authenticateUser", dry_run: true})
|
||||
→ 12 edits across 8 files
|
||||
→ 10 graph edits (high confidence), 2 ast_search edits (review)
|
||||
→ Changes: [{file_path, edits: [{line, old_text, new_text, confidence}]}]
|
||||
```
|
||||
|
||||
**gitnexus_impact** — map all dependents first:
|
||||
|
||||
```
|
||||
gitnexus_impact({target: "validateUser", direction: "upstream"})
|
||||
→ d=1: loginHandler, apiMiddleware, testUtils
|
||||
→ Affected Processes: LoginFlow, TokenRefresh
|
||||
```
|
||||
|
||||
**gitnexus_detect_changes** — verify your changes after refactoring:
|
||||
|
||||
```
|
||||
gitnexus_detect_changes({scope: "all"})
|
||||
→ Changed: 8 files, 12 symbols
|
||||
→ Affected processes: LoginFlow, TokenRefresh
|
||||
→ Risk: MEDIUM
|
||||
```
|
||||
|
||||
**gitnexus_cypher** — custom reference queries:
|
||||
|
||||
```cypher
|
||||
MATCH (caller)-[:CodeRelation {type: 'CALLS'}]->(f:Function {name: "validateUser"})
|
||||
RETURN caller.name, caller.filePath ORDER BY caller.filePath
|
||||
```
|
||||
|
||||
## Risk Rules
|
||||
|
||||
| Risk Factor | Mitigation |
|
||||
| ------------------- | ----------------------------------------- |
|
||||
| Many callers (>5) | Use gitnexus_rename for automated updates |
|
||||
| Cross-area refs | Use detect_changes after to verify scope |
|
||||
| String/dynamic refs | gitnexus_query to find them |
|
||||
| External/public API | Version and deprecate properly |
|
||||
|
||||
## Example: Rename `validateUser` to `authenticateUser`
|
||||
|
||||
```
|
||||
1. gitnexus_rename({symbol_name: "validateUser", new_name: "authenticateUser", dry_run: true})
|
||||
→ 12 edits: 10 graph (safe), 2 ast_search (review)
|
||||
→ Files: validator.ts, login.ts, middleware.ts, config.json...
|
||||
|
||||
2. Review ast_search edits (config.json: dynamic reference!)
|
||||
|
||||
3. gitnexus_rename({symbol_name: "validateUser", new_name: "authenticateUser", dry_run: false})
|
||||
→ Applied 12 edits across 8 files
|
||||
|
||||
4. gitnexus_detect_changes({scope: "all"})
|
||||
→ Affected: LoginFlow, TokenRefresh
|
||||
→ Risk: MEDIUM — run tests for these flows
|
||||
```
|
||||
|
||||
@@ -0,0 +1,28 @@
|
||||
name: Setup GitNexus
|
||||
description: Setup Node.js 20, install dependencies, and optionally build
|
||||
|
||||
inputs:
|
||||
build:
|
||||
description: Whether to run npm run build after install
|
||||
required: false
|
||||
default: 'false'
|
||||
|
||||
runs:
|
||||
using: composite
|
||||
steps:
|
||||
- uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4
|
||||
with:
|
||||
node-version: 20
|
||||
cache: npm
|
||||
cache-dependency-path: gitnexus/package-lock.json
|
||||
|
||||
- name: Install dependencies
|
||||
run: npm ci
|
||||
shell: bash
|
||||
working-directory: gitnexus
|
||||
|
||||
- name: Build
|
||||
if: ${{ inputs.build == 'true' }}
|
||||
run: npm run build
|
||||
shell: bash
|
||||
working-directory: gitnexus
|
||||
@@ -0,0 +1,45 @@
|
||||
changelog:
|
||||
exclude:
|
||||
labels:
|
||||
- chore
|
||||
authors:
|
||||
- dependabot
|
||||
- dependabot[bot]
|
||||
categories:
|
||||
- title: "\U0001F6A8 Security"
|
||||
labels:
|
||||
- security
|
||||
- title: "\U0001F4A5 Breaking Changes"
|
||||
labels:
|
||||
- breaking
|
||||
- title: "\U0001F680 Features"
|
||||
labels:
|
||||
- enhancement
|
||||
- title: "\U0001F41B Bug Fixes"
|
||||
labels:
|
||||
- bug
|
||||
- title: "\U0001F3CE\uFE0F Performance"
|
||||
labels:
|
||||
- performance
|
||||
- title: "\U0001F9EA Tests"
|
||||
labels:
|
||||
- test
|
||||
- title: "\U0001F504 Refactoring"
|
||||
labels:
|
||||
- refactor
|
||||
- title: "\U0001F477 CI/CD"
|
||||
labels:
|
||||
- ci
|
||||
- title: "\U0001F4E6 Dependencies"
|
||||
labels:
|
||||
- dependencies
|
||||
- title: "\U0001F4DD Other Changes"
|
||||
labels:
|
||||
- "*"
|
||||
exclude:
|
||||
labels:
|
||||
- dependencies
|
||||
- ci
|
||||
- test
|
||||
- refactor
|
||||
- chore
|
||||
@@ -0,0 +1,257 @@
|
||||
"""Pure math utilities for triage sweep embedding analysis.
|
||||
|
||||
All functions are stateless and perform no I/O (except model loading by FastEmbed).
|
||||
Each function operates on numpy arrays and returns numpy arrays or plain Python types.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import numpy as np
|
||||
from numpy.typing import NDArray
|
||||
from fastembed import TextEmbedding
|
||||
from sklearn.decomposition import PCA
|
||||
from sklearn.covariance import EllipticEnvelope
|
||||
from sklearn.metrics.pairwise import cosine_similarity
|
||||
|
||||
# FastEmbed model — BAAI/bge-small-en-v1.5 produces 384-dimensional embeddings.
|
||||
# ~46MB quantized ONNX, runs on CPU in ~0.5s per batch of 32.
|
||||
EMBEDDING_MODEL: str = "BAAI/bge-small-en-v1.5"
|
||||
|
||||
# Embedding dimensionality (determined by model choice).
|
||||
EMBEDDING_DIM: int = 384
|
||||
|
||||
# Batch size for FastEmbed. 32 balances memory and throughput on
|
||||
# a 2-vCPU GitHub Actions runner with ~7GB RAM.
|
||||
EMBEDDING_BATCH_SIZE: int = 32
|
||||
|
||||
|
||||
def embed_texts(texts: list[str]) -> NDArray[np.float32]:
|
||||
"""Embed a list of texts into dense vectors using FastEmbed.
|
||||
|
||||
Returns an array of shape (len(texts), 384) with dtype float32.
|
||||
Empty input returns a (0, 384) array.
|
||||
"""
|
||||
if not texts:
|
||||
return np.empty((0, EMBEDDING_DIM), dtype=np.float32)
|
||||
|
||||
model = TextEmbedding(model_name=EMBEDDING_MODEL)
|
||||
vectors = list(model.embed(texts, batch_size=EMBEDDING_BATCH_SIZE))
|
||||
return np.vstack(vectors).astype(np.float32)
|
||||
|
||||
|
||||
def normalize_rows(matrix: NDArray[np.float32]) -> NDArray[np.float32]:
|
||||
"""L2-normalize each row to unit length.
|
||||
|
||||
Zero-norm rows (e.g. from empty text) remain zero vectors.
|
||||
Uses eps=1e-10 in the denominator to avoid division by zero.
|
||||
"""
|
||||
if matrix.shape[0] == 0:
|
||||
return matrix
|
||||
|
||||
norms = np.linalg.norm(matrix, axis=1, keepdims=True)
|
||||
return matrix / (norms + 1e-10)
|
||||
|
||||
|
||||
def reduce_dimensions(
|
||||
matrix: NDArray[np.float32],
|
||||
max_components: int,
|
||||
) -> NDArray[np.float32]:
|
||||
"""Reduce dimensionality via PCA.
|
||||
|
||||
Computes n_components = min(max_components, n-1, d). If n_components < 1,
|
||||
returns the matrix unchanged. Logs explained variance for observability.
|
||||
"""
|
||||
n, d = matrix.shape
|
||||
if n <= 1:
|
||||
return matrix
|
||||
|
||||
n_components = min(max_components, n - 1, d)
|
||||
if n_components < 1:
|
||||
return matrix
|
||||
|
||||
pca = PCA(n_components=n_components)
|
||||
reduced = pca.fit_transform(matrix)
|
||||
explained = pca.explained_variance_ratio_.sum()
|
||||
print(f"PCA: {d}d -> {n_components}d, explained variance: {explained:.3f}")
|
||||
return reduced.astype(np.float32)
|
||||
|
||||
|
||||
def detect_outliers(
|
||||
matrix: NDArray[np.float32],
|
||||
contamination: float = 0.1,
|
||||
iqr_multiplier: float = 3.0,
|
||||
max_outlier_pct: float = 0.05,
|
||||
) -> list[tuple[int, float]]:
|
||||
"""Flag items whose Mahalanobis distance exceeds an IQR-based cutoff.
|
||||
|
||||
Uses EllipticEnvelope (robust covariance via MCD) to estimate the
|
||||
multivariate Gaussian, then computes sqrt(squared Mahalanobis distance)
|
||||
for each sample. The cutoff is Q75 + iqr_multiplier * IQR, which
|
||||
adapts to the actual distribution of distances.
|
||||
|
||||
A hard cap ensures no more than max_outlier_pct * n items are flagged;
|
||||
when the cap is hit, only the most extreme items (sorted by distance
|
||||
descending) are kept.
|
||||
|
||||
Returns (index, distance) tuples sorted by index ascending, along with
|
||||
the cutoff value stored as an attribute on the returned list.
|
||||
"""
|
||||
n = matrix.shape[0]
|
||||
if n < 2:
|
||||
return []
|
||||
|
||||
envelope = EllipticEnvelope(contamination=contamination, random_state=42)
|
||||
envelope.fit(matrix)
|
||||
|
||||
# .mahalanobis() returns squared Mahalanobis distances
|
||||
distances = np.sqrt(envelope.mahalanobis(matrix))
|
||||
|
||||
# IQR-based cutoff
|
||||
q25, q75 = np.percentile(distances, [25, 75])
|
||||
iqr = q75 - q25
|
||||
cutoff = q75 + iqr_multiplier * iqr
|
||||
|
||||
outlier_mask = distances > cutoff
|
||||
indices = np.where(outlier_mask)[0]
|
||||
|
||||
# Hard cap: keep at most max_outlier_pct * n items
|
||||
max_count = max(1, int(max_outlier_pct * n))
|
||||
if len(indices) > max_count:
|
||||
# Sort by distance descending, take the most extreme
|
||||
sorted_by_dist = sorted(indices, key=lambda i: distances[i], reverse=True)
|
||||
indices = np.array(sorted_by_dist[:max_count])
|
||||
|
||||
# Sort by index ascending for stable output
|
||||
indices = np.sort(indices)
|
||||
result = [(int(idx), float(distances[idx])) for idx in indices]
|
||||
|
||||
# Attach cutoff as metadata so the report can use it
|
||||
result = _OutlierResult(result) # type: ignore[assignment]
|
||||
result.cutoff = float(cutoff) # type: ignore[attr-defined]
|
||||
return result # type: ignore[return-value]
|
||||
|
||||
|
||||
class _OutlierResult(list):
|
||||
"""A list subclass that carries metadata (cutoff) from outlier detection."""
|
||||
cutoff: float = 0.0
|
||||
|
||||
|
||||
def find_duplicate_pairs(
|
||||
matrix: NDArray[np.float32],
|
||||
threshold: float,
|
||||
) -> list[tuple[int, int, float]]:
|
||||
"""Find pairs of items with cosine similarity above threshold.
|
||||
|
||||
Returns (i, j, similarity) tuples where i < j. The input should be
|
||||
L2-normalized embeddings (full dimensionality, not PCA-reduced) so
|
||||
cosine similarity equals the dot product.
|
||||
"""
|
||||
n = matrix.shape[0]
|
||||
if n <= 1:
|
||||
return []
|
||||
|
||||
sim_matrix = cosine_similarity(matrix)
|
||||
# Upper triangle indices (i < j), excluding diagonal
|
||||
rows, cols = np.triu_indices(n, k=1)
|
||||
similarities = sim_matrix[rows, cols]
|
||||
|
||||
mask = similarities > threshold
|
||||
pairs: list[tuple[int, int, float]] = []
|
||||
for idx in np.where(mask)[0]:
|
||||
pairs.append((int(rows[idx]), int(cols[idx]), float(similarities[idx])))
|
||||
|
||||
return pairs
|
||||
|
||||
|
||||
# ── Label suggestion via z-score normalized embedding similarity ──────
|
||||
|
||||
# Z-score threshold: a label must be this many standard deviations above
|
||||
# the column mean to be considered a match.
|
||||
LABEL_Z_THRESHOLD: float = 1.5
|
||||
|
||||
# Margin gate: the top-1 label must beat the second-best by this many
|
||||
# z-score units to be accepted (subsequent labels don't need a margin).
|
||||
LABEL_Z_MARGIN: float = 0.5
|
||||
|
||||
# Floor for per-column standard deviation to avoid division by near-zero.
|
||||
LABEL_Z_STD_FLOOR: float = 0.01
|
||||
|
||||
# Minimum raw cosine similarity required even if z-score is high.
|
||||
# Prevents suggesting labels that are "relatively best" but still poor.
|
||||
MIN_RAW_SIMILARITY: float = 0.3
|
||||
|
||||
# Maximum number of labels to suggest per item.
|
||||
MAX_LABELS_PER_ITEM: int = 3
|
||||
|
||||
|
||||
def suggest_labels(
|
||||
item_embeddings: NDArray[np.float32],
|
||||
label_embeddings: NDArray[np.float32],
|
||||
label_names: list[str],
|
||||
z_threshold: float = LABEL_Z_THRESHOLD,
|
||||
z_margin: float = LABEL_Z_MARGIN,
|
||||
std_floor: float = LABEL_Z_STD_FLOOR,
|
||||
min_raw_sim: float = MIN_RAW_SIMILARITY,
|
||||
max_per_item: int = MAX_LABELS_PER_ITEM,
|
||||
) -> list[list[tuple[str, float]]]:
|
||||
"""Suggest labels for each item using z-score normalized similarity.
|
||||
|
||||
1. Compute raw cosine similarity matrix (n items x m labels).
|
||||
2. Column-wise z-score: for each label j, normalize across all items.
|
||||
3. For each item, rank labels by z-score descending.
|
||||
4. Accept a label only if z >= z_threshold AND raw_sim >= min_raw_sim.
|
||||
5. Margin gate: the top-1 label must beat #2 by z_margin; subsequent
|
||||
labels don't need a margin.
|
||||
6. Cap at max_per_item.
|
||||
|
||||
Returns a list of length n, where each element is a list of
|
||||
(label_name, raw_similarity) tuples. Empty list if nothing qualifies.
|
||||
"""
|
||||
n = item_embeddings.shape[0]
|
||||
m = label_embeddings.shape[0]
|
||||
if n == 0 or m == 0:
|
||||
return [[] for _ in range(n)]
|
||||
|
||||
# (n, m) raw similarity matrix
|
||||
sim_matrix = cosine_similarity(item_embeddings, label_embeddings)
|
||||
|
||||
# Column-wise z-score normalization
|
||||
col_means = sim_matrix.mean(axis=0) # shape (m,)
|
||||
col_stds = sim_matrix.std(axis=0) # shape (m,)
|
||||
col_stds = np.maximum(col_stds, std_floor)
|
||||
z_matrix = (sim_matrix - col_means) / col_stds
|
||||
|
||||
suggestions: list[list[tuple[str, float]]] = []
|
||||
for i in range(n):
|
||||
z_row = z_matrix[i]
|
||||
raw_row = sim_matrix[i]
|
||||
|
||||
# Rank labels by z-score descending
|
||||
ranked = np.argsort(z_row)[::-1]
|
||||
|
||||
item_labels: list[tuple[str, float]] = []
|
||||
|
||||
# Margin gate: top-1 z-score must beat #2 by z_margin.
|
||||
# If not, the assignment is ambiguous — skip this item entirely.
|
||||
if len(ranked) > 1:
|
||||
top1_z = float(z_row[ranked[0]])
|
||||
top2_z = float(z_row[ranked[1]])
|
||||
if top1_z - top2_z < z_margin:
|
||||
suggestions.append(item_labels)
|
||||
continue
|
||||
|
||||
for rank_pos, idx in enumerate(ranked):
|
||||
if len(item_labels) >= max_per_item:
|
||||
break
|
||||
|
||||
z_val = float(z_row[idx])
|
||||
raw_val = float(raw_row[idx])
|
||||
|
||||
# Must pass both z-threshold and raw similarity floor
|
||||
if z_val < z_threshold or raw_val < min_raw_sim:
|
||||
continue
|
||||
|
||||
item_labels.append((label_names[idx], raw_val))
|
||||
|
||||
suggestions.append(item_labels)
|
||||
|
||||
return suggestions
|
||||
@@ -0,0 +1,4 @@
|
||||
fastembed>=0.5.0
|
||||
numpy>=1.26.0
|
||||
scikit-learn>=1.4.0
|
||||
scipy>=1.10.0
|
||||
@@ -0,0 +1,600 @@
|
||||
"""Triage sweep: fetch open issues/PRs, detect outliers and duplicates, generate a report.
|
||||
|
||||
Entrypoint script for the triage-sweep workflow. Fetches all open items via
|
||||
the GitHub REST API, delegates embedding and analysis to embedding_utils,
|
||||
generates a markdown report, and optionally creates a report issue.
|
||||
"""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import os
|
||||
import sys
|
||||
import urllib.request
|
||||
import urllib.parse
|
||||
from typing import TypedDict
|
||||
from datetime import datetime, timezone
|
||||
|
||||
from embedding_utils import (
|
||||
embed_texts,
|
||||
normalize_rows,
|
||||
reduce_dimensions,
|
||||
detect_outliers,
|
||||
find_duplicate_pairs,
|
||||
suggest_labels,
|
||||
LABEL_Z_THRESHOLD,
|
||||
LABEL_Z_MARGIN,
|
||||
LABEL_Z_STD_FLOOR,
|
||||
MIN_RAW_SIMILARITY,
|
||||
MAX_LABELS_PER_ITEM,
|
||||
)
|
||||
|
||||
# ── Thresholds (overridable via workflow_dispatch inputs) ──────────────
|
||||
|
||||
# IQR multiplier for outlier cutoff: cutoff = Q75 + IQR_MULTIPLIER * IQR.
|
||||
IQR_MULTIPLIER: float = float(os.environ.get("INPUT_IQR_MULTIPLIER", "3.0"))
|
||||
|
||||
# Hard cap: at most this fraction of items can be flagged as outliers.
|
||||
MAX_OUTLIER_PCT: float = float(os.environ.get("INPUT_MAX_OUTLIER_PCT", "0.05"))
|
||||
|
||||
# EllipticEnvelope contamination: expected fraction of outliers in the data.
|
||||
# Governs how aggressively the robust covariance downweights extreme points.
|
||||
CONTAMINATION: float = float(os.environ.get("INPUT_CONTAMINATION", "0.1"))
|
||||
|
||||
# Cosine similarity above which two items are flagged as duplicates.
|
||||
# 0.92 catches near-identical issues while tolerating paraphrasing.
|
||||
COSINE_THRESHOLD: float = float(os.environ.get("INPUT_COSINE_THRESHOLD", "0.92"))
|
||||
|
||||
# Hard cap on items to process. Prevents runaway costs on very large repos.
|
||||
MAX_ITEMS: int = int(os.environ.get("INPUT_MAX_ITEMS", "500"))
|
||||
|
||||
# When true, print report to stdout/file but do not create a GitHub issue.
|
||||
DRY_RUN: bool = os.environ.get("INPUT_DRY_RUN", "false").lower() == "true"
|
||||
|
||||
# ── Fixed constants (not user-configurable) ───────────────────────────
|
||||
|
||||
# Minimum number of samples required for EllipticEnvelope to fit
|
||||
# a Gaussian reliably. Must be >= 3 * PCA_MAX_COMPONENTS so the
|
||||
# covariance matrix is estimated from enough data points.
|
||||
PCA_MAX_COMPONENTS: int = 20
|
||||
MIN_SAMPLES_FOR_OUTLIER_DETECTION: int = 100
|
||||
|
||||
# Max character length for embedding input text. bge-small-en-v1.5 has a
|
||||
# 512-token context window (~4 chars/token). We keep title + body under
|
||||
# this limit so the model sees the full text instead of silently truncating.
|
||||
MAX_EMBED_CHARS: int = 2000
|
||||
|
||||
# GitHub REST API page size (max allowed is 100).
|
||||
API_PAGE_SIZE: int = 100
|
||||
|
||||
# Report issue label.
|
||||
REPORT_LABEL: str = "triage-report"
|
||||
|
||||
# Report file path (written for the summary step to pick up).
|
||||
REPORT_FILE: str = "/tmp/triage-report.md"
|
||||
|
||||
|
||||
class TriageItem(TypedDict):
|
||||
"""One open issue or PR, with only the fields we need."""
|
||||
number: int
|
||||
title: str
|
||||
html_url: str
|
||||
is_pr: bool
|
||||
labels: list[str]
|
||||
created_at: str
|
||||
# title + body concatenated, used as embedding input
|
||||
text: str
|
||||
|
||||
|
||||
def github_api_get(path: str) -> list[dict]:
|
||||
"""Make a single authenticated GET request to the GitHub REST API.
|
||||
|
||||
Reads GITHUB_TOKEN and GITHUB_REPOSITORY from env. Raises SystemExit
|
||||
with the HTTP status and response body on any non-2xx response.
|
||||
"""
|
||||
token = os.environ["GITHUB_TOKEN"]
|
||||
repo = os.environ["GITHUB_REPOSITORY"]
|
||||
url = f"https://api.github.com/repos/{repo}{path}"
|
||||
|
||||
req = urllib.request.Request(url)
|
||||
req.add_header("Accept", "application/vnd.github+json")
|
||||
req.add_header("Authorization", f"Bearer {token}")
|
||||
req.add_header("X-GitHub-Api-Version", "2022-11-28")
|
||||
|
||||
try:
|
||||
with urllib.request.urlopen(req, timeout=30) as resp:
|
||||
return json.loads(resp.read().decode("utf-8"))
|
||||
except urllib.error.HTTPError as e:
|
||||
body = e.read().decode("utf-8", errors="replace")
|
||||
print(f"::error::GitHub API {e.code}: {body}")
|
||||
sys.exit(1)
|
||||
|
||||
|
||||
def fetch_all_open_items() -> list[TriageItem]:
|
||||
"""Paginate through all open issues and PRs.
|
||||
|
||||
Returns up to MAX_ITEMS TriageItem dicts. Items with a pull_request
|
||||
key are marked is_pr=True. The text field is title + body concatenated.
|
||||
"""
|
||||
items: list[TriageItem] = []
|
||||
page = 1
|
||||
|
||||
while len(items) < MAX_ITEMS:
|
||||
path = (
|
||||
f"/issues?state=open&per_page={API_PAGE_SIZE}"
|
||||
f"&sort=created&direction=desc&page={page}"
|
||||
)
|
||||
data = github_api_get(path)
|
||||
|
||||
if not data:
|
||||
break
|
||||
|
||||
for raw in data:
|
||||
if len(items) >= MAX_ITEMS:
|
||||
break
|
||||
|
||||
body = raw.get("body", "") or ""
|
||||
full_text = f"{raw['title']}\n\n{body}"
|
||||
# Truncate to fit the embedding model's token window.
|
||||
# Title is always preserved; body gets clipped if needed.
|
||||
if len(full_text) > MAX_EMBED_CHARS:
|
||||
full_text = full_text[:MAX_EMBED_CHARS]
|
||||
items.append(TriageItem(
|
||||
number=raw["number"],
|
||||
title=raw["title"],
|
||||
html_url=raw["html_url"],
|
||||
is_pr="pull_request" in raw,
|
||||
labels=[lbl["name"] for lbl in raw.get("labels", [])],
|
||||
created_at=raw["created_at"],
|
||||
text=full_text,
|
||||
))
|
||||
|
||||
if len(data) < API_PAGE_SIZE:
|
||||
break
|
||||
|
||||
page += 1
|
||||
|
||||
return items
|
||||
|
||||
|
||||
class RepoLabel(TypedDict):
|
||||
"""A label from the repo with its embedding text."""
|
||||
name: str
|
||||
description: str
|
||||
# "name: description" concatenated for embedding
|
||||
text: str
|
||||
|
||||
|
||||
def fetch_repo_labels() -> list[RepoLabel]:
|
||||
"""Fetch all labels from the repository, paginating if needed.
|
||||
|
||||
Returns labels with name, description, and a text field suitable
|
||||
for embedding ("name: description"). Labels with no description
|
||||
use just the name.
|
||||
"""
|
||||
labels: list[RepoLabel] = []
|
||||
page = 1
|
||||
|
||||
while True:
|
||||
data = github_api_get(f"/labels?per_page={API_PAGE_SIZE}&page={page}")
|
||||
for raw in data:
|
||||
name = raw["name"]
|
||||
desc = raw.get("description", "") or ""
|
||||
text = f"{name}: {desc}" if desc else name
|
||||
labels.append(RepoLabel(name=name, description=desc, text=text))
|
||||
|
||||
if len(data) < API_PAGE_SIZE:
|
||||
break
|
||||
page += 1
|
||||
|
||||
return labels
|
||||
|
||||
|
||||
def apply_labels_to_item(item_number: int, labels: list[str]) -> None:
|
||||
"""Add labels to a single issue/PR via the GitHub API.
|
||||
|
||||
Skips silently if labels list is empty. Uses POST which adds labels
|
||||
without removing existing ones.
|
||||
"""
|
||||
if not labels:
|
||||
return
|
||||
|
||||
token = os.environ["GITHUB_TOKEN"]
|
||||
repo = os.environ["GITHUB_REPOSITORY"]
|
||||
url = f"https://api.github.com/repos/{repo}/issues/{item_number}/labels"
|
||||
|
||||
payload = json.dumps({"labels": labels}).encode("utf-8")
|
||||
req = urllib.request.Request(url, data=payload, method="POST")
|
||||
req.add_header("Accept", "application/vnd.github+json")
|
||||
req.add_header("Authorization", f"Bearer {token}")
|
||||
req.add_header("X-GitHub-Api-Version", "2022-11-28")
|
||||
req.add_header("Content-Type", "application/json")
|
||||
|
||||
try:
|
||||
with urllib.request.urlopen(req, timeout=30) as resp:
|
||||
resp.read()
|
||||
except urllib.error.HTTPError as e:
|
||||
body = e.read().decode("utf-8", errors="replace")
|
||||
# Non-fatal: log warning but don't abort the sweep
|
||||
print(f"::warning::Failed to label #{item_number}: {e.code} {body}")
|
||||
|
||||
|
||||
def _item_age(created_at: str) -> str:
|
||||
"""Compute a human-readable age string from an ISO 8601 created_at timestamp."""
|
||||
try:
|
||||
created = datetime.fromisoformat(created_at.replace("Z", "+00:00"))
|
||||
delta = datetime.now(timezone.utc) - created
|
||||
days = delta.days
|
||||
if days < 1:
|
||||
return "<1d"
|
||||
if days < 30:
|
||||
return f"{days}d"
|
||||
if days < 365:
|
||||
return f"{days // 30}mo"
|
||||
return f"{days // 365}y"
|
||||
except (ValueError, TypeError):
|
||||
return "?"
|
||||
|
||||
|
||||
def _suggested_action(a: TriageItem, b: TriageItem) -> str:
|
||||
"""Determine a suggested action for a duplicate pair based on types and age."""
|
||||
if a["is_pr"] and b["is_pr"]:
|
||||
return "Review for overlap"
|
||||
if not a["is_pr"] and not b["is_pr"]:
|
||||
# Both issues — close the newer one
|
||||
try:
|
||||
a_dt = datetime.fromisoformat(a["created_at"].replace("Z", "+00:00"))
|
||||
b_dt = datetime.fromisoformat(b["created_at"].replace("Z", "+00:00"))
|
||||
newer = b if b_dt > a_dt else a
|
||||
except (ValueError, TypeError):
|
||||
newer = b
|
||||
return f"Close #{newer['number']} as duplicate"
|
||||
# One issue, one PR
|
||||
return "Link PR to issue"
|
||||
|
||||
|
||||
def generate_report(
|
||||
items: list[TriageItem],
|
||||
outlier_results: list[tuple[int, float]],
|
||||
duplicate_pairs: list[tuple[int, int, float]],
|
||||
label_suggestions: list[list[tuple[str, float]]] | None = None,
|
||||
) -> str:
|
||||
"""Generate a structured markdown triage report."""
|
||||
now = datetime.now(timezone.utc).strftime("%Y-%m-%d %H:%M:%S")
|
||||
repo = os.environ.get("GITHUB_REPOSITORY", "unknown/repo")
|
||||
|
||||
# Compute label suggestion counts early for the health table
|
||||
outlier_set = {idx for idx, _ in outlier_results}
|
||||
suggested_count = 0
|
||||
if label_suggestions is not None:
|
||||
suggested_count = sum(
|
||||
1 for i, s in enumerate(label_suggestions)
|
||||
if s and not items[i]["labels"] and i not in outlier_set
|
||||
)
|
||||
|
||||
# ── Health summary table at the top ──────────────────────────────
|
||||
lines: list[str] = [
|
||||
"## Triage Sweep Report",
|
||||
"",
|
||||
f"**Run:** {now} UTC",
|
||||
f"**Items analyzed:** {len(items)}",
|
||||
f"**Thresholds:** IQR multiplier {IQR_MULTIPLIER}, Cosine > {COSINE_THRESHOLD}",
|
||||
"",
|
||||
"### Health Summary",
|
||||
"",
|
||||
"| Metric | Value |",
|
||||
"|--------|-------|",
|
||||
f"| Items analyzed | {len(items)} |",
|
||||
f"| Outliers flagged | {len(outlier_results)} |",
|
||||
f"| Duplicate pairs | {len(duplicate_pairs)} |",
|
||||
f"| Label suggestions | {suggested_count} |",
|
||||
"",
|
||||
]
|
||||
|
||||
# ── Outlier section ──────────────────────────────────────────────
|
||||
# Determine cutoff for high-confidence split
|
||||
cutoff = getattr(outlier_results, "cutoff", 0.0)
|
||||
high_conf_cutoff = 2 * cutoff if cutoff > 0 else float("inf")
|
||||
|
||||
high_conf = [(idx, d) for idx, d in outlier_results if d > high_conf_cutoff]
|
||||
borderline = [(idx, d) for idx, d in outlier_results if d <= high_conf_cutoff]
|
||||
|
||||
lines.extend([
|
||||
f"### Potential Outliers / Spam ({len(outlier_results)})",
|
||||
"",
|
||||
"Items with unusually high Mahalanobis distance from the distribution center.",
|
||||
"These may be spam, off-topic, or poorly described.",
|
||||
"",
|
||||
])
|
||||
|
||||
if high_conf:
|
||||
lines.append(f"**High Confidence** ({len(high_conf)} items, distance > 2x cutoff)")
|
||||
lines.append("")
|
||||
lines.append("| # | Type | Title | Distance | Age |")
|
||||
lines.append("|---|------|-------|----------|-----|")
|
||||
for idx, distance in high_conf:
|
||||
item = items[idx]
|
||||
kind = "PR" if item["is_pr"] else "Issue"
|
||||
age = _item_age(item["created_at"])
|
||||
title = item["title"][:80] + ("..." if len(item["title"]) > 80 else "")
|
||||
lines.append(
|
||||
f"| [#{item['number']}]({item['html_url']}) "
|
||||
f"| {kind} | {title} | {distance:.2f} | {age} |"
|
||||
)
|
||||
lines.append("")
|
||||
|
||||
if borderline:
|
||||
lines.append("<details>")
|
||||
lines.append(f"<summary>Borderline ({len(borderline)} items)</summary>")
|
||||
lines.append("")
|
||||
lines.append("| # | Type | Title | Distance | Age |")
|
||||
lines.append("|---|------|-------|----------|-----|")
|
||||
for idx, distance in borderline:
|
||||
item = items[idx]
|
||||
kind = "PR" if item["is_pr"] else "Issue"
|
||||
age = _item_age(item["created_at"])
|
||||
title = item["title"][:80] + ("..." if len(item["title"]) > 80 else "")
|
||||
lines.append(
|
||||
f"| [#{item['number']}]({item['html_url']}) "
|
||||
f"| {kind} | {title} | {distance:.2f} | {age} |"
|
||||
)
|
||||
lines.append("")
|
||||
lines.append("</details>")
|
||||
lines.append("")
|
||||
|
||||
if not outlier_results:
|
||||
lines.append("None found.")
|
||||
|
||||
# ── Duplicate pairs section ──────────────────────────────────────
|
||||
lines.extend([
|
||||
"",
|
||||
f"### Potential Duplicates ({len(duplicate_pairs)} pairs)",
|
||||
"",
|
||||
"Pairs of items with cosine similarity above the threshold.",
|
||||
"",
|
||||
])
|
||||
|
||||
if duplicate_pairs:
|
||||
lines.append("| Item A | Item B | Similarity | Suggested Action |")
|
||||
lines.append("|--------|--------|------------|------------------|")
|
||||
for i, j, sim in duplicate_pairs:
|
||||
a = items[i]
|
||||
b = items[j]
|
||||
kind_a = "PR" if a["is_pr"] else "Issue"
|
||||
kind_b = "PR" if b["is_pr"] else "Issue"
|
||||
action = _suggested_action(a, b)
|
||||
lines.append(
|
||||
f"| [#{a['number']}]({a['html_url']}) {kind_a}: {a['title']} "
|
||||
f"| [#{b['number']}]({b['html_url']}) {kind_b}: {b['title']} "
|
||||
f"| {sim:.3f} | {action} |"
|
||||
)
|
||||
else:
|
||||
lines.append("None found.")
|
||||
|
||||
# ── Label suggestions section ────────────────────────────────────
|
||||
if label_suggestions is not None:
|
||||
# High confidence: top-1 label with raw_sim >= 0.5
|
||||
# Low confidence: top-1 label with raw_sim < 0.5
|
||||
high_conf_labels: list[tuple[int, list[tuple[str, float]]]] = []
|
||||
low_conf_labels: list[tuple[int, list[tuple[str, float]]]] = []
|
||||
for i, sugs in enumerate(label_suggestions):
|
||||
if sugs and not items[i]["labels"] and i not in outlier_set:
|
||||
top1 = sugs[:1]
|
||||
if top1[0][1] >= 0.5:
|
||||
high_conf_labels.append((i, top1))
|
||||
else:
|
||||
low_conf_labels.append((i, top1))
|
||||
|
||||
total_suggestions = len(high_conf_labels) + len(low_conf_labels)
|
||||
lines.extend([
|
||||
"",
|
||||
f"### Suggested Labels ({total_suggestions} unlabeled items)",
|
||||
"",
|
||||
"Labels suggested by z-score normalized embedding similarity against repo label descriptions.",
|
||||
"Only shown for unlabeled items that were not flagged as outliers.",
|
||||
"",
|
||||
])
|
||||
|
||||
# Label concentration warning
|
||||
if total_suggestions > 0:
|
||||
label_counts: dict[str, int] = {}
|
||||
for _, sugs in high_conf_labels + low_conf_labels:
|
||||
for name, _ in sugs:
|
||||
label_counts[name] = label_counts.get(name, 0) + 1
|
||||
for name, count in label_counts.items():
|
||||
if count > total_suggestions * 0.5:
|
||||
lines.append(
|
||||
f"> **Warning:** Label `{name}` accounts for "
|
||||
f"{count}/{total_suggestions} suggestions "
|
||||
f"({count * 100 // total_suggestions}%). "
|
||||
f"Consider reviewing label descriptions for specificity."
|
||||
)
|
||||
lines.append("")
|
||||
|
||||
if high_conf_labels:
|
||||
lines.append("| # | Type | Title | Suggested Label |")
|
||||
lines.append("|---|------|-------|--------------------|")
|
||||
for idx, sugs in high_conf_labels:
|
||||
item = items[idx]
|
||||
kind = "PR" if item["is_pr"] else "Issue"
|
||||
label_strs = [f"`{name}` ({score:.2f})" for name, score in sugs]
|
||||
lines.append(
|
||||
f"| [#{item['number']}]({item['html_url']}) "
|
||||
f"| {kind} | {item['title']} | {', '.join(label_strs)} |"
|
||||
)
|
||||
|
||||
if low_conf_labels:
|
||||
lines.append("")
|
||||
lines.append("<details>")
|
||||
lines.append(f"<summary>Low-confidence suggestions ({len(low_conf_labels)} items)</summary>")
|
||||
lines.append("")
|
||||
lines.append("| # | Type | Title | Suggested Label |")
|
||||
lines.append("|---|------|-------|--------------------|")
|
||||
for idx, sugs in low_conf_labels:
|
||||
item = items[idx]
|
||||
kind = "PR" if item["is_pr"] else "Issue"
|
||||
label_strs = [f"`{name}` ({score:.2f})" for name, score in sugs]
|
||||
lines.append(
|
||||
f"| [#{item['number']}]({item['html_url']}) "
|
||||
f"| {kind} | {item['title']} | {', '.join(label_strs)} |"
|
||||
)
|
||||
lines.append("")
|
||||
lines.append("</details>")
|
||||
|
||||
if not high_conf_labels and not low_conf_labels:
|
||||
lines.append("No unlabeled items need suggestions.")
|
||||
|
||||
lines.extend([
|
||||
"",
|
||||
"### Summary",
|
||||
"",
|
||||
f"- {len(outlier_results)} outliers flagged for review",
|
||||
f"- {len(duplicate_pairs)} duplicate pairs found",
|
||||
f"- {len(items)} items analyzed in total",
|
||||
])
|
||||
|
||||
if label_suggestions is not None:
|
||||
lines.append(f"- {suggested_count} items suggested for labeling")
|
||||
|
||||
lines.extend([
|
||||
"",
|
||||
"---",
|
||||
f"*Generated by [triage-sweep](https://github.com/{repo}/actions) — no LLM was used.*",
|
||||
])
|
||||
|
||||
return "\n".join(lines)
|
||||
|
||||
|
||||
def create_report_issue(report_body: str) -> None:
|
||||
"""Create a GitHub issue with the triage report.
|
||||
|
||||
Posts to the issues API with the triage-report label.
|
||||
Raises SystemExit on non-201 response.
|
||||
"""
|
||||
token = os.environ["GITHUB_TOKEN"]
|
||||
repo = os.environ["GITHUB_REPOSITORY"]
|
||||
url = f"https://api.github.com/repos/{repo}/issues"
|
||||
|
||||
today = datetime.now(timezone.utc).strftime("%Y-%m-%d")
|
||||
payload = json.dumps({
|
||||
"title": f"Triage Sweep Report — {today}",
|
||||
"body": report_body,
|
||||
"labels": [REPORT_LABEL],
|
||||
}).encode("utf-8")
|
||||
|
||||
req = urllib.request.Request(url, data=payload, method="POST")
|
||||
req.add_header("Accept", "application/vnd.github+json")
|
||||
req.add_header("Authorization", f"Bearer {token}")
|
||||
req.add_header("X-GitHub-Api-Version", "2022-11-28")
|
||||
req.add_header("Content-Type", "application/json")
|
||||
|
||||
try:
|
||||
with urllib.request.urlopen(req, timeout=30) as resp:
|
||||
resp_body = resp.read().decode("utf-8")
|
||||
if resp.status != 201:
|
||||
print(f"::error::Failed to create issue: {resp.status} {resp_body}")
|
||||
sys.exit(1)
|
||||
result = json.loads(resp_body)
|
||||
print(f"Created issue: {result.get('html_url', 'unknown')}")
|
||||
except urllib.error.HTTPError as e:
|
||||
body = e.read().decode("utf-8", errors="replace")
|
||||
print(f"::error::Failed to create issue: {e.code} {body}")
|
||||
sys.exit(1)
|
||||
|
||||
|
||||
def write_report(report: str) -> None:
|
||||
"""Write the report to the file system for the summary step."""
|
||||
with open(REPORT_FILE, "w", encoding="utf-8") as f:
|
||||
f.write(report)
|
||||
|
||||
|
||||
def main() -> None:
|
||||
"""Orchestrate the full triage sweep."""
|
||||
# 1. Validate environment
|
||||
for var in ("GITHUB_TOKEN", "GITHUB_REPOSITORY"):
|
||||
if not os.environ.get(var):
|
||||
print(f"::error::Missing required environment variable: {var}")
|
||||
sys.exit(1)
|
||||
|
||||
# 2. Fetch all open issues + PRs
|
||||
items = fetch_all_open_items()
|
||||
print(f"Fetched {len(items)} open items")
|
||||
|
||||
if len(items) == 0:
|
||||
report = "## Triage Sweep Report\n\nNo open issues or PRs found."
|
||||
write_report(report)
|
||||
print("No items to analyze.")
|
||||
return
|
||||
|
||||
# 3. Extract texts for embedding
|
||||
texts: list[str] = [item["text"] for item in items]
|
||||
|
||||
# 4. Embed all texts (returns numpy float32 array of shape [n, 384])
|
||||
embeddings = embed_texts(texts)
|
||||
|
||||
# 5. L2-normalize
|
||||
embeddings = normalize_rows(embeddings)
|
||||
|
||||
# 6. Outlier detection (Mahalanobis via EllipticEnvelope)
|
||||
outlier_results: list[tuple[int, float]] = []
|
||||
if len(items) >= MIN_SAMPLES_FOR_OUTLIER_DETECTION:
|
||||
reduced = reduce_dimensions(embeddings, PCA_MAX_COMPONENTS)
|
||||
outlier_results = detect_outliers(
|
||||
reduced,
|
||||
contamination=CONTAMINATION,
|
||||
iqr_multiplier=IQR_MULTIPLIER,
|
||||
max_outlier_pct=MAX_OUTLIER_PCT,
|
||||
)
|
||||
else:
|
||||
print(
|
||||
f"Skipping outlier detection: {len(items)} items < "
|
||||
f"{MIN_SAMPLES_FOR_OUTLIER_DETECTION} minimum"
|
||||
)
|
||||
|
||||
# 7. Duplicate detection (pairwise cosine similarity)
|
||||
duplicate_pairs = find_duplicate_pairs(embeddings, COSINE_THRESHOLD)
|
||||
|
||||
# 8. Label suggestion via embedding similarity
|
||||
label_suggestions: list[list[tuple[str, float]]] | None = None
|
||||
repo_labels = fetch_repo_labels()
|
||||
if repo_labels:
|
||||
label_texts = [lbl["text"] for lbl in repo_labels]
|
||||
label_names = [lbl["name"] for lbl in repo_labels]
|
||||
label_embeddings = embed_texts(label_texts)
|
||||
label_embeddings = normalize_rows(label_embeddings)
|
||||
label_suggestions = suggest_labels(embeddings, label_embeddings, label_names)
|
||||
print(f"Computed label suggestions against {len(repo_labels)} repo labels")
|
||||
|
||||
# NOTE: Auto-labeling is disabled. The report shows suggestions for
|
||||
# human review. To re-enable, uncomment the block below.
|
||||
#
|
||||
# # Apply top label to unlabeled items (unless dry run)
|
||||
# # Skip outliers — flagged items shouldn't get categorized
|
||||
# outlier_set = {idx for idx, _ in outlier_results}
|
||||
# if not DRY_RUN:
|
||||
# applied_count = 0
|
||||
# for i, sugs in enumerate(label_suggestions):
|
||||
# if sugs and not items[i]["labels"] and i not in outlier_set:
|
||||
# # Apply only the top-1 label (highest confidence)
|
||||
# apply_labels_to_item(items[i]["number"], [sugs[0][0]])
|
||||
# applied_count += 1
|
||||
# print(f"Applied labels to {applied_count} unlabeled items")
|
||||
else:
|
||||
print("No repo labels found — skipping label suggestions")
|
||||
|
||||
# 9. Generate report
|
||||
report = generate_report(items, outlier_results, duplicate_pairs, label_suggestions)
|
||||
|
||||
# 10. Write report to file (for summary step)
|
||||
write_report(report)
|
||||
|
||||
# 11. Create report issue (unless dry run)
|
||||
if DRY_RUN:
|
||||
print("Dry run — skipping issue creation and label application.")
|
||||
print(report)
|
||||
else:
|
||||
create_report_issue(report)
|
||||
print("Report issue created.")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
@@ -0,0 +1,468 @@
|
||||
"""Tests for embedding_utils.py — all embedding model calls are mocked."""
|
||||
from __future__ import annotations
|
||||
|
||||
import sys
|
||||
from unittest.mock import patch, MagicMock
|
||||
import numpy as np
|
||||
import pytest
|
||||
|
||||
# Mock fastembed before importing the module under test (persistent)
|
||||
if "fastembed" not in sys.modules:
|
||||
sys.modules["fastembed"] = MagicMock()
|
||||
|
||||
from embedding_utils import (
|
||||
embed_texts,
|
||||
normalize_rows,
|
||||
reduce_dimensions,
|
||||
detect_outliers,
|
||||
find_duplicate_pairs,
|
||||
suggest_labels,
|
||||
EMBEDDING_DIM,
|
||||
EMBEDDING_MODEL,
|
||||
EMBEDDING_BATCH_SIZE,
|
||||
LABEL_Z_THRESHOLD,
|
||||
LABEL_Z_MARGIN,
|
||||
LABEL_Z_STD_FLOOR,
|
||||
MIN_RAW_SIMILARITY,
|
||||
MAX_LABELS_PER_ITEM,
|
||||
)
|
||||
|
||||
|
||||
class TestEmbedTexts:
|
||||
"""Tests for the embed_texts function."""
|
||||
|
||||
def test_empty_list_returns_empty_array(self):
|
||||
result = embed_texts([])
|
||||
assert result.shape == (0, EMBEDDING_DIM)
|
||||
assert result.dtype == np.float32
|
||||
|
||||
@patch("embedding_utils.TextEmbedding")
|
||||
def test_single_text(self, mock_cls):
|
||||
mock_model = MagicMock()
|
||||
mock_cls.return_value = mock_model
|
||||
vec = np.random.randn(EMBEDDING_DIM).astype(np.float32)
|
||||
mock_model.embed.return_value = iter([vec])
|
||||
|
||||
result = embed_texts(["hello world"])
|
||||
|
||||
mock_cls.assert_called_once_with(model_name=EMBEDDING_MODEL)
|
||||
mock_model.embed.assert_called_once_with(
|
||||
["hello world"], batch_size=EMBEDDING_BATCH_SIZE
|
||||
)
|
||||
assert result.shape == (1, EMBEDDING_DIM)
|
||||
assert result.dtype == np.float32
|
||||
np.testing.assert_array_almost_equal(result[0], vec)
|
||||
|
||||
@patch("embedding_utils.TextEmbedding")
|
||||
def test_multiple_texts(self, mock_cls):
|
||||
mock_model = MagicMock()
|
||||
mock_cls.return_value = mock_model
|
||||
vecs = [
|
||||
np.random.randn(EMBEDDING_DIM).astype(np.float32)
|
||||
for _ in range(5)
|
||||
]
|
||||
mock_model.embed.return_value = iter(vecs)
|
||||
|
||||
result = embed_texts(["a", "b", "c", "d", "e"])
|
||||
assert result.shape == (5, EMBEDDING_DIM)
|
||||
assert result.dtype == np.float32
|
||||
|
||||
|
||||
class TestNormalizeRows:
|
||||
"""Tests for L2 row normalization."""
|
||||
|
||||
def test_empty_matrix(self):
|
||||
m = np.empty((0, 10), dtype=np.float32)
|
||||
result = normalize_rows(m)
|
||||
assert result.shape == (0, 10)
|
||||
|
||||
def test_single_row(self):
|
||||
m = np.array([[3.0, 4.0]], dtype=np.float32)
|
||||
result = normalize_rows(m)
|
||||
# Norm should be ~1.0
|
||||
norm = np.linalg.norm(result[0])
|
||||
assert abs(norm - 1.0) < 1e-5
|
||||
|
||||
def test_multiple_rows(self):
|
||||
rng = np.random.default_rng(42)
|
||||
m = rng.standard_normal((10, 50)).astype(np.float32)
|
||||
result = normalize_rows(m)
|
||||
norms = np.linalg.norm(result, axis=1)
|
||||
np.testing.assert_allclose(norms, 1.0, atol=1e-5)
|
||||
|
||||
def test_zero_row_stays_near_zero(self):
|
||||
m = np.array([[0.0, 0.0, 0.0], [1.0, 0.0, 0.0]], dtype=np.float32)
|
||||
result = normalize_rows(m)
|
||||
# Zero row divided by eps -> very small values
|
||||
assert np.linalg.norm(result[0]) < 1e-3
|
||||
# Non-zero row should be unit norm
|
||||
assert abs(np.linalg.norm(result[1]) - 1.0) < 1e-5
|
||||
|
||||
def test_preserves_direction(self):
|
||||
m = np.array([[2.0, 0.0], [0.0, 3.0]], dtype=np.float32)
|
||||
result = normalize_rows(m)
|
||||
np.testing.assert_allclose(result[0], [1.0, 0.0], atol=1e-5)
|
||||
np.testing.assert_allclose(result[1], [0.0, 1.0], atol=1e-5)
|
||||
|
||||
|
||||
class TestReduceDimensions:
|
||||
"""Tests for PCA dimensionality reduction."""
|
||||
|
||||
def test_single_sample_returns_unchanged(self):
|
||||
m = np.random.randn(1, 50).astype(np.float32)
|
||||
result = reduce_dimensions(m, 10)
|
||||
np.testing.assert_array_equal(result, m)
|
||||
|
||||
def test_reduces_dimensions(self):
|
||||
rng = np.random.default_rng(42)
|
||||
m = rng.standard_normal((100, 50)).astype(np.float32)
|
||||
result = reduce_dimensions(m, 10)
|
||||
assert result.shape == (100, 10)
|
||||
assert result.dtype == np.float32
|
||||
|
||||
def test_caps_at_n_minus_1(self):
|
||||
rng = np.random.default_rng(42)
|
||||
# 5 samples, 20 features -> max components = 4 (n-1)
|
||||
m = rng.standard_normal((5, 20)).astype(np.float32)
|
||||
result = reduce_dimensions(m, 50)
|
||||
assert result.shape == (5, 4)
|
||||
|
||||
def test_caps_at_d(self):
|
||||
rng = np.random.default_rng(42)
|
||||
# 100 samples, 3 features -> max components = 3
|
||||
m = rng.standard_normal((100, 3)).astype(np.float32)
|
||||
result = reduce_dimensions(m, 50)
|
||||
assert result.shape == (100, 3)
|
||||
|
||||
def test_max_components_respected(self):
|
||||
rng = np.random.default_rng(42)
|
||||
m = rng.standard_normal((50, 30)).astype(np.float32)
|
||||
result = reduce_dimensions(m, 5)
|
||||
assert result.shape[1] == 5
|
||||
|
||||
|
||||
class TestDetectOutliers:
|
||||
"""Tests for IQR-based outlier detection."""
|
||||
|
||||
def test_single_sample_returns_empty(self):
|
||||
m = np.random.randn(1, 5).astype(np.float32)
|
||||
result = detect_outliers(m)
|
||||
assert result == []
|
||||
|
||||
def test_empty_returns_empty(self):
|
||||
# n < 2 case
|
||||
m = np.empty((0, 5), dtype=np.float32)
|
||||
result = detect_outliers(m)
|
||||
assert result == []
|
||||
|
||||
def test_finds_outliers_in_synthetic_data(self):
|
||||
rng = np.random.default_rng(42)
|
||||
# Create a tight cluster with one obvious outlier
|
||||
cluster = rng.standard_normal((50, 3)).astype(np.float32) * 0.1
|
||||
outlier = np.array([[100.0, 100.0, 100.0]], dtype=np.float32)
|
||||
m = np.vstack([cluster, outlier])
|
||||
result = detect_outliers(m)
|
||||
# The outlier (index 50) should be detected
|
||||
outlier_indices = [idx for idx, _ in result]
|
||||
assert 50 in outlier_indices
|
||||
|
||||
def test_returns_list_of_index_distance_tuples(self):
|
||||
rng = np.random.default_rng(42)
|
||||
# Tight cluster + outlier to guarantee at least one result
|
||||
cluster = rng.standard_normal((20, 3)).astype(np.float32) * 0.1
|
||||
far_point = np.array([[50.0, 50.0, 50.0]], dtype=np.float32)
|
||||
m = np.vstack([cluster, far_point])
|
||||
result = detect_outliers(m)
|
||||
assert isinstance(result, list)
|
||||
for item in result:
|
||||
assert isinstance(item, tuple)
|
||||
assert len(item) == 2
|
||||
idx, dist = item
|
||||
assert isinstance(idx, int)
|
||||
assert isinstance(dist, float)
|
||||
assert dist > 0
|
||||
|
||||
def test_iqr_cutoff_behavior(self):
|
||||
"""Lower IQR multiplier should flag more items than higher multiplier."""
|
||||
rng = np.random.default_rng(42)
|
||||
m = rng.standard_normal((100, 3)).astype(np.float32)
|
||||
low = detect_outliers(m, iqr_multiplier=1.0, max_outlier_pct=0.5)
|
||||
high = detect_outliers(m, iqr_multiplier=5.0, max_outlier_pct=0.5)
|
||||
assert len(low) >= len(high)
|
||||
|
||||
def test_dimension_aware_no_mass_flagging(self):
|
||||
"""High-dimensional clean Gaussian data should not flag everything."""
|
||||
rng = np.random.default_rng(42)
|
||||
# 500 samples, 10 dims — well-conditioned for robust covariance
|
||||
m = rng.standard_normal((500, 10)).astype(np.float32)
|
||||
result = detect_outliers(m)
|
||||
# With IQR-based cutoff on clean Gaussian data,
|
||||
# only a small fraction should be flagged (well under 50%)
|
||||
assert len(result) < 250
|
||||
|
||||
def test_contamination_parameter(self):
|
||||
rng = np.random.default_rng(42)
|
||||
m = rng.standard_normal((50, 3)).astype(np.float32)
|
||||
# Should not raise with different contamination values
|
||||
result = detect_outliers(m, contamination=0.05)
|
||||
assert isinstance(result, list)
|
||||
|
||||
def test_max_outlier_pct_hard_cap(self):
|
||||
"""The hard cap should limit outlier count to max_outlier_pct * n."""
|
||||
rng = np.random.default_rng(42)
|
||||
# Create data with many potential outliers (bimodal)
|
||||
cluster = rng.standard_normal((80, 3)).astype(np.float32) * 0.1
|
||||
outliers = rng.standard_normal((20, 3)).astype(np.float32) * 50.0
|
||||
m = np.vstack([cluster, outliers])
|
||||
# Very low IQR multiplier to flag a lot, but cap at 5%
|
||||
result = detect_outliers(m, iqr_multiplier=0.5, max_outlier_pct=0.05)
|
||||
max_allowed = max(1, int(0.05 * 100)) # 5
|
||||
assert len(result) <= max_allowed
|
||||
|
||||
def test_hard_cap_keeps_most_extreme(self):
|
||||
"""When capped, the most extreme items (highest distance) should be kept."""
|
||||
rng = np.random.default_rng(42)
|
||||
cluster = rng.standard_normal((90, 3)).astype(np.float32) * 0.1
|
||||
# Create outliers with increasing extremity
|
||||
outliers = np.array([
|
||||
[10.0, 10.0, 10.0],
|
||||
[20.0, 20.0, 20.0],
|
||||
[50.0, 50.0, 50.0],
|
||||
], dtype=np.float32)
|
||||
m = np.vstack([cluster, outliers])
|
||||
# Cap at ~1 item (0.01 * 93 = 0, but min is 1)
|
||||
result = detect_outliers(m, iqr_multiplier=0.5, max_outlier_pct=0.02)
|
||||
if len(result) > 0:
|
||||
# The most extreme (index 92, distance for [50,50,50]) should be kept
|
||||
indices = [idx for idx, _ in result]
|
||||
assert 92 in indices
|
||||
|
||||
def test_cutoff_attribute(self):
|
||||
"""Returned result should carry a cutoff attribute."""
|
||||
rng = np.random.default_rng(42)
|
||||
m = rng.standard_normal((50, 3)).astype(np.float32)
|
||||
result = detect_outliers(m)
|
||||
assert hasattr(result, "cutoff")
|
||||
assert isinstance(result.cutoff, float)
|
||||
assert result.cutoff > 0
|
||||
|
||||
|
||||
class TestFindDuplicatePairs:
|
||||
"""Tests for cosine similarity duplicate detection."""
|
||||
|
||||
def test_single_item_returns_empty(self):
|
||||
m = np.random.randn(1, 10).astype(np.float32)
|
||||
result = find_duplicate_pairs(m, 0.9)
|
||||
assert result == []
|
||||
|
||||
def test_empty_returns_empty(self):
|
||||
m = np.empty((0, 10), dtype=np.float32)
|
||||
result = find_duplicate_pairs(m, 0.9)
|
||||
assert result == []
|
||||
|
||||
def test_identical_vectors_detected(self):
|
||||
vec = np.random.randn(10).astype(np.float32)
|
||||
vec = vec / np.linalg.norm(vec)
|
||||
m = np.vstack([vec, vec, np.random.randn(10).astype(np.float32)])
|
||||
result = find_duplicate_pairs(m, 0.99)
|
||||
# Items 0 and 1 are identical, should be found
|
||||
assert any(i == 0 and j == 1 for i, j, _ in result)
|
||||
|
||||
def test_orthogonal_vectors_not_detected(self):
|
||||
m = np.eye(5, dtype=np.float32)
|
||||
result = find_duplicate_pairs(m, 0.5)
|
||||
assert result == []
|
||||
|
||||
def test_returns_correct_format(self):
|
||||
vec = np.random.randn(10).astype(np.float32)
|
||||
vec = vec / np.linalg.norm(vec)
|
||||
m = np.vstack([vec, vec])
|
||||
result = find_duplicate_pairs(m, 0.5)
|
||||
assert len(result) >= 1
|
||||
for item in result:
|
||||
assert len(item) == 3
|
||||
i, j, sim = item
|
||||
assert isinstance(i, int)
|
||||
assert isinstance(j, int)
|
||||
assert isinstance(sim, float)
|
||||
assert i < j
|
||||
|
||||
def test_i_less_than_j(self):
|
||||
rng = np.random.default_rng(42)
|
||||
# Create some similar vectors
|
||||
base = rng.standard_normal(10).astype(np.float32)
|
||||
m = np.vstack([base + rng.standard_normal(10) * 0.01 for _ in range(5)])
|
||||
result = find_duplicate_pairs(m, 0.5)
|
||||
for i, j, _ in result:
|
||||
assert i < j
|
||||
|
||||
def test_high_threshold_fewer_pairs(self):
|
||||
rng = np.random.default_rng(42)
|
||||
m = rng.standard_normal((10, 20)).astype(np.float32)
|
||||
# Normalize for meaningful cosine similarities
|
||||
norms = np.linalg.norm(m, axis=1, keepdims=True)
|
||||
m = m / norms
|
||||
low = find_duplicate_pairs(m, 0.3)
|
||||
high = find_duplicate_pairs(m, 0.9)
|
||||
assert len(low) >= len(high)
|
||||
|
||||
|
||||
class TestSuggestLabels:
|
||||
"""Tests for z-score normalized label suggestion."""
|
||||
|
||||
def test_empty_items_returns_empty_lists(self):
|
||||
items = np.empty((0, 10), dtype=np.float32)
|
||||
labels = np.random.randn(3, 10).astype(np.float32)
|
||||
result = suggest_labels(items, labels, ["a", "b", "c"])
|
||||
assert result == []
|
||||
|
||||
def test_empty_labels_returns_empty_per_item(self):
|
||||
items = np.random.randn(5, 10).astype(np.float32)
|
||||
labels = np.empty((0, 10), dtype=np.float32)
|
||||
result = suggest_labels(items, labels, [])
|
||||
assert len(result) == 5
|
||||
assert all(s == [] for s in result)
|
||||
|
||||
def test_identical_embedding_gets_that_label(self):
|
||||
"""If an item embedding strongly matches one label, z-score should highlight it."""
|
||||
# Create multiple items so z-score normalization is meaningful
|
||||
rng = np.random.default_rng(42)
|
||||
# 10 random items + 1 item that matches label "bug" exactly
|
||||
random_items = rng.standard_normal((10, 3)).astype(np.float32)
|
||||
bug_vec = np.array([[1.0, 0.0, 0.0]], dtype=np.float32)
|
||||
items = np.vstack([random_items, bug_vec])
|
||||
labels = np.array([[1.0, 0.0, 0.0], [0.0, 1.0, 0.0], [0.0, 0.0, 1.0]], dtype=np.float32)
|
||||
result = suggest_labels(
|
||||
items, labels, ["bug", "feature", "docs"],
|
||||
z_threshold=0.5, z_margin=0.0, min_raw_sim=0.1,
|
||||
)
|
||||
# The last item (matching bug_vec) should get "bug" as top suggestion
|
||||
last_item_sugs = result[-1]
|
||||
if last_item_sugs:
|
||||
assert last_item_sugs[0][0] == "bug"
|
||||
|
||||
def test_z_score_suppresses_dominant_label(self):
|
||||
"""When all items are similar to one label, z-scores should be low
|
||||
(none stands out) and that label should not be blindly suggested."""
|
||||
# All items identical — z-score for every item on every label is 0
|
||||
items = np.ones((10, 3), dtype=np.float32)
|
||||
labels = np.array([[1.0, 1.0, 1.0], [0.0, 1.0, 0.0]], dtype=np.float32)
|
||||
result = suggest_labels(
|
||||
items, labels, ["catch-all", "specific"],
|
||||
z_threshold=1.5, min_raw_sim=0.3,
|
||||
)
|
||||
# With identical items, std=0 -> z-scores are all 0 -> nothing passes z_threshold
|
||||
for sugs in result:
|
||||
assert sugs == []
|
||||
|
||||
def test_margin_gate_blocks_top1(self):
|
||||
"""Top-1 label must beat #2 by z_margin to be accepted as position 0."""
|
||||
rng = np.random.default_rng(99)
|
||||
# 20 items, each slightly different, 2 labels
|
||||
items = rng.standard_normal((20, 5)).astype(np.float32)
|
||||
# Two labels that are nearly identical -> margin gate should block top-1
|
||||
labels = np.array([[1.0, 0.5, 0.0, 0.0, 0.0],
|
||||
[1.0, 0.5, 0.01, 0.0, 0.0]], dtype=np.float32)
|
||||
result = suggest_labels(
|
||||
items, labels, ["label-a", "label-b"],
|
||||
z_threshold=0.0, z_margin=10.0, min_raw_sim=0.0, max_per_item=1,
|
||||
)
|
||||
# With a huge margin requirement and max_per_item=1, nothing should pass
|
||||
# because the only candidate (top-1) is blocked by margin gate,
|
||||
# and max_per_item=1 prevents falling through to position 2
|
||||
for sugs in result:
|
||||
assert sugs == []
|
||||
|
||||
def test_margin_gate_passes_when_clear_winner(self):
|
||||
"""When top-1 clearly beats #2, it should pass the margin gate."""
|
||||
# Create items where one strongly matches label 0 vs label 1
|
||||
items = np.array([
|
||||
[1.0, 0.0, 0.0, 0.0, 0.0], # strongly matches label-a
|
||||
[0.0, 0.0, 0.0, 0.0, 1.0], # matches neither well
|
||||
] * 5, dtype=np.float32) # 10 items for stable z-scores
|
||||
labels = np.array([
|
||||
[1.0, 0.0, 0.0, 0.0, 0.0], # label-a
|
||||
[0.0, 1.0, 0.0, 0.0, 0.0], # label-b (orthogonal)
|
||||
], dtype=np.float32)
|
||||
result = suggest_labels(
|
||||
items, labels, ["label-a", "label-b"],
|
||||
z_threshold=0.5, z_margin=0.3, min_raw_sim=0.1,
|
||||
)
|
||||
# Items matching label-a should get it suggested (clear z-score advantage)
|
||||
got_label_a = sum(1 for sugs in result if sugs and sugs[0][0] == "label-a")
|
||||
assert got_label_a > 0
|
||||
|
||||
def test_min_raw_similarity_filter(self):
|
||||
"""Even with high z-score, low raw similarity should be filtered out."""
|
||||
# Items are orthogonal to all labels -> raw similarity near 0
|
||||
items = np.array([[1.0, 0.0, 0.0]], dtype=np.float32)
|
||||
labels = np.array([[0.0, 0.0, 1.0]], dtype=np.float32)
|
||||
result = suggest_labels(
|
||||
items, labels, ["irrelevant"],
|
||||
z_threshold=0.0, z_margin=0.0, min_raw_sim=0.9,
|
||||
)
|
||||
# Raw similarity is ~0, which is below min_raw_sim=0.9
|
||||
assert result[0] == []
|
||||
|
||||
def test_max_per_item_respected(self):
|
||||
"""Even if many labels qualify, max_per_item caps the results."""
|
||||
rng = np.random.default_rng(42)
|
||||
# Create items with some variance so z-scores differentiate
|
||||
items = rng.standard_normal((20, 10)).astype(np.float32)
|
||||
base = items[0]
|
||||
# All labels very similar to item 0
|
||||
labels = np.array([base + rng.standard_normal(10) * 0.01 for _ in range(10)])
|
||||
names = [f"label-{i}" for i in range(10)]
|
||||
result = suggest_labels(
|
||||
items, labels, names,
|
||||
z_threshold=0.0, z_margin=0.0, min_raw_sim=0.0, max_per_item=2,
|
||||
)
|
||||
for sugs in result:
|
||||
assert len(sugs) <= 2
|
||||
|
||||
def test_returns_raw_similarity_not_z_score(self):
|
||||
"""Returned scores should be raw cosine similarity, not z-scores."""
|
||||
rng = np.random.default_rng(42)
|
||||
items = rng.standard_normal((15, 5)).astype(np.float32)
|
||||
labels = rng.standard_normal((3, 5)).astype(np.float32)
|
||||
names = ["bug", "feature", "docs"]
|
||||
result = suggest_labels(
|
||||
items, labels, names,
|
||||
z_threshold=0.0, z_margin=0.0, min_raw_sim=-1.0,
|
||||
)
|
||||
# Raw cosine similarity should be in [-1, 1] range
|
||||
for sugs in result:
|
||||
for name, score in sugs:
|
||||
assert -1.0 <= score <= 1.0 + 1e-5
|
||||
assert isinstance(name, str)
|
||||
assert isinstance(score, float)
|
||||
|
||||
def test_returns_correct_format(self):
|
||||
rng = np.random.default_rng(42)
|
||||
items = rng.standard_normal((3, 10)).astype(np.float32)
|
||||
labels = rng.standard_normal((5, 10)).astype(np.float32)
|
||||
names = ["bug", "feature", "docs", "ci", "test"]
|
||||
result = suggest_labels(
|
||||
items, labels, names,
|
||||
z_threshold=0.0, z_margin=0.0, min_raw_sim=-1.0,
|
||||
)
|
||||
assert len(result) == 3
|
||||
for sugs in result:
|
||||
for name, score in sugs:
|
||||
assert isinstance(name, str)
|
||||
assert isinstance(score, float)
|
||||
assert name in names
|
||||
|
||||
def test_text_truncation_in_labels(self):
|
||||
"""Label names should be returned as-is even when very long."""
|
||||
rng = np.random.default_rng(42)
|
||||
items = rng.standard_normal((10, 5)).astype(np.float32)
|
||||
long_name = "a" * 200
|
||||
labels = rng.standard_normal((1, 5)).astype(np.float32)
|
||||
result = suggest_labels(
|
||||
items, labels, [long_name],
|
||||
z_threshold=0.0, z_margin=0.0, min_raw_sim=-1.0,
|
||||
)
|
||||
for sugs in result:
|
||||
if sugs:
|
||||
assert sugs[0][0] == long_name
|
||||
@@ -0,0 +1,873 @@
|
||||
"""Tests for sweep.py — all external calls (API, embedding) are mocked."""
|
||||
from __future__ import annotations
|
||||
|
||||
import json
|
||||
import os
|
||||
import sys
|
||||
from io import BytesIO
|
||||
from unittest.mock import patch, MagicMock, mock_open
|
||||
from urllib.error import HTTPError
|
||||
|
||||
import numpy as np
|
||||
import pytest
|
||||
|
||||
# Mock fastembed before importing sweep (which imports embedding_utils)
|
||||
sys.modules["fastembed"] = MagicMock()
|
||||
|
||||
# Set required env vars before importing sweep (module-level constants read env)
|
||||
os.environ.setdefault("GITHUB_TOKEN", "test-token")
|
||||
os.environ.setdefault("GITHUB_REPOSITORY", "owner/repo")
|
||||
|
||||
from sweep import (
|
||||
github_api_get,
|
||||
fetch_all_open_items,
|
||||
fetch_repo_labels,
|
||||
apply_labels_to_item,
|
||||
generate_report,
|
||||
create_report_issue,
|
||||
write_report,
|
||||
main,
|
||||
TriageItem,
|
||||
RepoLabel,
|
||||
REPORT_FILE,
|
||||
REPORT_LABEL,
|
||||
API_PAGE_SIZE,
|
||||
MIN_SAMPLES_FOR_OUTLIER_DETECTION,
|
||||
PCA_MAX_COMPONENTS,
|
||||
MAX_EMBED_CHARS,
|
||||
IQR_MULTIPLIER,
|
||||
MAX_OUTLIER_PCT,
|
||||
_item_age,
|
||||
_suggested_action,
|
||||
)
|
||||
|
||||
|
||||
def _make_api_issue(number: int, title: str = "Test issue", is_pr: bool = False,
|
||||
body: str = "Issue body", labels: list[str] | None = None,
|
||||
created_at: str = "2026-03-21T00:00:00Z") -> dict:
|
||||
"""Helper to build a mock GitHub API issue response object."""
|
||||
result: dict = {
|
||||
"number": number,
|
||||
"title": title,
|
||||
"html_url": f"https://github.com/owner/repo/issues/{number}",
|
||||
"body": body,
|
||||
"created_at": created_at,
|
||||
"labels": [{"name": lbl} for lbl in (labels or [])],
|
||||
}
|
||||
if is_pr:
|
||||
result["pull_request"] = {"url": "..."}
|
||||
return result
|
||||
|
||||
|
||||
class TestGithubApiGet:
|
||||
"""Tests for the github_api_get function."""
|
||||
|
||||
@patch("sweep.urllib.request.urlopen")
|
||||
def test_successful_request(self, mock_urlopen):
|
||||
mock_resp = MagicMock()
|
||||
mock_resp.read.return_value = json.dumps([{"id": 1}]).encode()
|
||||
mock_resp.__enter__ = lambda s: s
|
||||
mock_resp.__exit__ = MagicMock(return_value=False)
|
||||
mock_urlopen.return_value = mock_resp
|
||||
|
||||
result = github_api_get("/issues?state=open")
|
||||
assert result == [{"id": 1}]
|
||||
|
||||
@patch("sweep.urllib.request.urlopen")
|
||||
def test_http_error_exits(self, mock_urlopen):
|
||||
error = HTTPError(
|
||||
url="https://api.github.com/repos/owner/repo/issues",
|
||||
code=403,
|
||||
msg="Forbidden",
|
||||
hdrs=None, # type: ignore[arg-type]
|
||||
fp=BytesIO(b'{"message": "rate limited"}'),
|
||||
)
|
||||
mock_urlopen.side_effect = error
|
||||
|
||||
with pytest.raises(SystemExit) as exc_info:
|
||||
github_api_get("/issues")
|
||||
assert exc_info.value.code == 1
|
||||
|
||||
|
||||
class TestConstants:
|
||||
"""Tests for module-level constants."""
|
||||
|
||||
def test_min_samples_is_at_least_3x_pca_max(self):
|
||||
"""MIN_SAMPLES must be >= 3 * PCA_MAX_COMPONENTS for reliable covariance."""
|
||||
assert MIN_SAMPLES_FOR_OUTLIER_DETECTION >= 3 * PCA_MAX_COMPONENTS
|
||||
|
||||
def test_min_samples_is_100(self):
|
||||
assert MIN_SAMPLES_FOR_OUTLIER_DETECTION == 100
|
||||
|
||||
def test_pca_max_components_is_20(self):
|
||||
assert PCA_MAX_COMPONENTS == 20
|
||||
|
||||
def test_iqr_multiplier_default(self):
|
||||
assert IQR_MULTIPLIER == 3.0
|
||||
|
||||
def test_max_outlier_pct_default(self):
|
||||
assert MAX_OUTLIER_PCT == 0.05
|
||||
|
||||
|
||||
class TestFetchAllOpenItems:
|
||||
"""Tests for fetch_all_open_items."""
|
||||
|
||||
@patch("sweep.github_api_get")
|
||||
def test_empty_repo(self, mock_get):
|
||||
mock_get.return_value = []
|
||||
items = fetch_all_open_items()
|
||||
assert items == []
|
||||
|
||||
@patch("sweep.github_api_get")
|
||||
def test_single_page(self, mock_get):
|
||||
mock_get.return_value = [
|
||||
_make_api_issue(1, "Bug report"),
|
||||
_make_api_issue(2, "Feature request", is_pr=True),
|
||||
]
|
||||
items = fetch_all_open_items()
|
||||
assert len(items) == 2
|
||||
assert items[0]["number"] == 1
|
||||
assert items[0]["is_pr"] is False
|
||||
assert items[1]["is_pr"] is True
|
||||
|
||||
@patch("sweep.github_api_get")
|
||||
def test_text_field_constructed(self, mock_get):
|
||||
mock_get.return_value = [
|
||||
_make_api_issue(1, "My Title", body="My Body"),
|
||||
]
|
||||
items = fetch_all_open_items()
|
||||
assert items[0]["text"] == "My Title\n\nMy Body"
|
||||
|
||||
@patch("sweep.github_api_get")
|
||||
def test_long_body_truncated(self, mock_get):
|
||||
"""Bodies exceeding MAX_EMBED_CHARS are truncated to fit the token window."""
|
||||
long_body = "x" * (MAX_EMBED_CHARS + 500)
|
||||
mock_get.return_value = [
|
||||
_make_api_issue(1, "Title", body=long_body),
|
||||
]
|
||||
items = fetch_all_open_items()
|
||||
assert len(items[0]["text"]) == MAX_EMBED_CHARS
|
||||
|
||||
@patch("sweep.github_api_get")
|
||||
def test_short_body_not_truncated(self, mock_get):
|
||||
"""Bodies under the limit are left intact."""
|
||||
mock_get.return_value = [
|
||||
_make_api_issue(1, "Title", body="Short body"),
|
||||
]
|
||||
items = fetch_all_open_items()
|
||||
assert items[0]["text"] == "Title\n\nShort body"
|
||||
|
||||
@patch("sweep.github_api_get")
|
||||
def test_null_body_handled(self, mock_get):
|
||||
issue = _make_api_issue(1, "No body")
|
||||
issue["body"] = None
|
||||
mock_get.return_value = [issue]
|
||||
items = fetch_all_open_items()
|
||||
assert items[0]["text"] == "No body\n\n"
|
||||
|
||||
@patch("sweep.github_api_get")
|
||||
def test_labels_extracted(self, mock_get):
|
||||
mock_get.return_value = [
|
||||
_make_api_issue(1, "Labeled", labels=["bug", "high-priority"]),
|
||||
]
|
||||
items = fetch_all_open_items()
|
||||
assert items[0]["labels"] == ["bug", "high-priority"]
|
||||
|
||||
@patch("sweep.MAX_ITEMS", 3)
|
||||
@patch("sweep.github_api_get")
|
||||
def test_max_items_cap(self, mock_get):
|
||||
mock_get.return_value = [_make_api_issue(i) for i in range(100)]
|
||||
items = fetch_all_open_items()
|
||||
assert len(items) == 3
|
||||
|
||||
@patch("sweep.API_PAGE_SIZE", 2)
|
||||
@patch("sweep.github_api_get")
|
||||
def test_pagination(self, mock_get):
|
||||
# First page: 2 items (full page), second page: 1 item (partial -> stop)
|
||||
mock_get.side_effect = [
|
||||
[_make_api_issue(1), _make_api_issue(2)],
|
||||
[_make_api_issue(3)],
|
||||
]
|
||||
items = fetch_all_open_items()
|
||||
assert len(items) == 3
|
||||
assert mock_get.call_count == 2
|
||||
|
||||
|
||||
class TestItemAge:
|
||||
"""Tests for _item_age helper."""
|
||||
|
||||
def test_recent_item(self):
|
||||
from datetime import datetime, timezone, timedelta
|
||||
recent = (datetime.now(timezone.utc) - timedelta(hours=12)).isoformat()
|
||||
assert _item_age(recent) == "<1d"
|
||||
|
||||
def test_days_old(self):
|
||||
from datetime import datetime, timezone, timedelta
|
||||
old = (datetime.now(timezone.utc) - timedelta(days=15)).isoformat()
|
||||
assert _item_age(old) == "15d"
|
||||
|
||||
def test_months_old(self):
|
||||
from datetime import datetime, timezone, timedelta
|
||||
old = (datetime.now(timezone.utc) - timedelta(days=90)).isoformat()
|
||||
assert _item_age(old) == "3mo"
|
||||
|
||||
def test_years_old(self):
|
||||
from datetime import datetime, timezone, timedelta
|
||||
old = (datetime.now(timezone.utc) - timedelta(days=400)).isoformat()
|
||||
assert _item_age(old) == "1y"
|
||||
|
||||
def test_invalid_date(self):
|
||||
assert _item_age("not-a-date") == "?"
|
||||
|
||||
|
||||
class TestSuggestedAction:
|
||||
"""Tests for _suggested_action helper."""
|
||||
|
||||
def test_both_issues_close_newer(self):
|
||||
a = TriageItem(
|
||||
number=1, title="A", html_url="u", is_pr=False, labels=[],
|
||||
created_at="2026-01-01T00:00:00Z", text="t",
|
||||
)
|
||||
b = TriageItem(
|
||||
number=2, title="B", html_url="u", is_pr=False, labels=[],
|
||||
created_at="2026-02-01T00:00:00Z", text="t",
|
||||
)
|
||||
result = _suggested_action(a, b)
|
||||
assert "Close #2 as duplicate" in result
|
||||
|
||||
def test_both_prs_review(self):
|
||||
a = TriageItem(
|
||||
number=1, title="A", html_url="u", is_pr=True, labels=[],
|
||||
created_at="2026-01-01T00:00:00Z", text="t",
|
||||
)
|
||||
b = TriageItem(
|
||||
number=2, title="B", html_url="u", is_pr=True, labels=[],
|
||||
created_at="2026-01-01T00:00:00Z", text="t",
|
||||
)
|
||||
assert _suggested_action(a, b) == "Review for overlap"
|
||||
|
||||
def test_issue_pr_link(self):
|
||||
a = TriageItem(
|
||||
number=1, title="A", html_url="u", is_pr=False, labels=[],
|
||||
created_at="2026-01-01T00:00:00Z", text="t",
|
||||
)
|
||||
b = TriageItem(
|
||||
number=2, title="B", html_url="u", is_pr=True, labels=[],
|
||||
created_at="2026-01-01T00:00:00Z", text="t",
|
||||
)
|
||||
assert _suggested_action(a, b) == "Link PR to issue"
|
||||
|
||||
|
||||
class TestGenerateReport:
|
||||
"""Tests for the markdown report generator."""
|
||||
|
||||
def test_no_findings(self):
|
||||
items = [
|
||||
TriageItem(
|
||||
number=1, title="Test", html_url="https://example.com/1",
|
||||
is_pr=False, labels=[], created_at="2026-01-01T00:00:00Z", text="Test",
|
||||
),
|
||||
]
|
||||
report = generate_report(items, [], [])
|
||||
assert "## Triage Sweep Report" in report
|
||||
assert "Items analyzed:** 1" in report
|
||||
assert "None found." in report
|
||||
assert "0 outliers flagged" in report
|
||||
assert "0 duplicate pairs found" in report
|
||||
|
||||
def test_health_summary_table(self):
|
||||
items = [
|
||||
TriageItem(
|
||||
number=1, title="Test", html_url="https://example.com/1",
|
||||
is_pr=False, labels=[], created_at="2026-01-01T00:00:00Z", text="Test",
|
||||
),
|
||||
]
|
||||
report = generate_report(items, [], [])
|
||||
assert "### Health Summary" in report
|
||||
assert "| Metric | Value |" in report
|
||||
assert "| Items analyzed | 1 |" in report
|
||||
|
||||
def test_iqr_multiplier_in_thresholds(self):
|
||||
"""Report should show IQR multiplier, not percentile."""
|
||||
items = [
|
||||
TriageItem(
|
||||
number=1, title="Test", html_url="https://example.com/1",
|
||||
is_pr=False, labels=[], created_at="2026-01-01T00:00:00Z", text="Test",
|
||||
),
|
||||
]
|
||||
report = generate_report(items, [], [])
|
||||
assert "IQR multiplier" in report
|
||||
assert "percentile" not in report.lower().split("thresholds")[0] # not in thresholds line
|
||||
|
||||
def test_with_outliers_shows_distance_and_age(self):
|
||||
items = [
|
||||
TriageItem(
|
||||
number=10, title="Spam Issue", html_url="https://example.com/10",
|
||||
is_pr=False, labels=[], created_at="2026-01-01T00:00:00Z", text="spam",
|
||||
),
|
||||
TriageItem(
|
||||
number=20, title="Good Issue", html_url="https://example.com/20",
|
||||
is_pr=False, labels=[], created_at="2026-01-01T00:00:00Z", text="good",
|
||||
),
|
||||
]
|
||||
report = generate_report(items, [(0, 12.34)], [])
|
||||
assert "#10" in report
|
||||
assert "Spam Issue" in report
|
||||
assert "12.34" in report
|
||||
assert "1 outliers flagged" in report
|
||||
# Age column should be present
|
||||
assert "| Age |" in report
|
||||
|
||||
def test_outlier_borderline_in_details(self):
|
||||
"""Borderline outliers should be in a <details> section."""
|
||||
from embedding_utils import _OutlierResult
|
||||
items = [
|
||||
TriageItem(
|
||||
number=10, title="Borderline", html_url="https://example.com/10",
|
||||
is_pr=False, labels=[], created_at="2026-01-01T00:00:00Z", text="spam",
|
||||
),
|
||||
]
|
||||
# Create outlier results with cutoff=10.0, distance=12.0 (< 2*cutoff=20)
|
||||
outlier_results = _OutlierResult([(0, 12.0)])
|
||||
outlier_results.cutoff = 10.0
|
||||
report = generate_report(items, outlier_results, [])
|
||||
assert "<details>" in report
|
||||
assert "Borderline" in report
|
||||
|
||||
def test_outlier_high_confidence(self):
|
||||
"""Items with distance > 2x cutoff should be in high confidence section."""
|
||||
from embedding_utils import _OutlierResult
|
||||
items = [
|
||||
TriageItem(
|
||||
number=10, title="Definite Spam", html_url="https://example.com/10",
|
||||
is_pr=False, labels=[], created_at="2026-01-01T00:00:00Z", text="spam",
|
||||
),
|
||||
]
|
||||
outlier_results = _OutlierResult([(0, 25.0)])
|
||||
outlier_results.cutoff = 10.0
|
||||
report = generate_report(items, outlier_results, [])
|
||||
assert "High Confidence" in report
|
||||
|
||||
def test_with_duplicates_suggested_action(self):
|
||||
items = [
|
||||
TriageItem(
|
||||
number=1, title="First", html_url="https://example.com/1",
|
||||
is_pr=False, labels=[], created_at="2026-01-01T00:00:00Z", text="a",
|
||||
),
|
||||
TriageItem(
|
||||
number=2, title="Second", html_url="https://example.com/2",
|
||||
is_pr=True, labels=[], created_at="2026-02-01T00:00:00Z", text="b",
|
||||
),
|
||||
]
|
||||
report = generate_report(items, [], [(0, 1, 0.954)])
|
||||
assert "#1" in report
|
||||
assert "#2" in report
|
||||
assert "0.954" in report
|
||||
assert "1 duplicate pairs found" in report
|
||||
assert "Suggested Action" in report
|
||||
assert "Link PR to issue" in report
|
||||
|
||||
def test_duplicate_both_issues_close_newer(self):
|
||||
items = [
|
||||
TriageItem(
|
||||
number=1, title="First", html_url="https://example.com/1",
|
||||
is_pr=False, labels=[], created_at="2026-01-01T00:00:00Z", text="a",
|
||||
),
|
||||
TriageItem(
|
||||
number=2, title="Second", html_url="https://example.com/2",
|
||||
is_pr=False, labels=[], created_at="2026-02-01T00:00:00Z", text="b",
|
||||
),
|
||||
]
|
||||
report = generate_report(items, [], [(0, 1, 0.95)])
|
||||
assert "Close #2 as duplicate" in report
|
||||
|
||||
def test_duplicate_both_prs_review(self):
|
||||
items = [
|
||||
TriageItem(
|
||||
number=1, title="PR A", html_url="https://example.com/1",
|
||||
is_pr=True, labels=[], created_at="2026-01-01T00:00:00Z", text="a",
|
||||
),
|
||||
TriageItem(
|
||||
number=2, title="PR B", html_url="https://example.com/2",
|
||||
is_pr=True, labels=[], created_at="2026-01-01T00:00:00Z", text="b",
|
||||
),
|
||||
]
|
||||
report = generate_report(items, [], [(0, 1, 0.95)])
|
||||
assert "Review for overlap" in report
|
||||
|
||||
def test_pr_type_label(self):
|
||||
items = [
|
||||
TriageItem(
|
||||
number=5, title="PR Title", html_url="https://example.com/5",
|
||||
is_pr=True, labels=[], created_at="2026-01-01T00:00:00Z", text="pr",
|
||||
),
|
||||
]
|
||||
report = generate_report(items, [(0, 8.5)], [])
|
||||
assert "| PR |" in report
|
||||
|
||||
def test_footer_present(self):
|
||||
items = [
|
||||
TriageItem(
|
||||
number=1, title="T", html_url="u",
|
||||
is_pr=False, labels=[], created_at="2026-01-01T00:00:00Z", text="t",
|
||||
),
|
||||
]
|
||||
report = generate_report(items, [], [])
|
||||
assert "no LLM was used" in report
|
||||
|
||||
|
||||
class TestCreateReportIssue:
|
||||
"""Tests for creating the report GitHub issue."""
|
||||
|
||||
@patch("sweep.urllib.request.urlopen")
|
||||
def test_successful_creation(self, mock_urlopen):
|
||||
mock_resp = MagicMock()
|
||||
mock_resp.status = 201
|
||||
mock_resp.read.return_value = json.dumps({
|
||||
"html_url": "https://github.com/owner/repo/issues/99",
|
||||
}).encode()
|
||||
mock_resp.__enter__ = lambda s: s
|
||||
mock_resp.__exit__ = MagicMock(return_value=False)
|
||||
mock_urlopen.return_value = mock_resp
|
||||
|
||||
# Should not raise
|
||||
create_report_issue("# Test Report")
|
||||
|
||||
@patch("sweep.urllib.request.urlopen")
|
||||
def test_http_error_exits(self, mock_urlopen):
|
||||
error = HTTPError(
|
||||
url="https://api.github.com/repos/owner/repo/issues",
|
||||
code=422,
|
||||
msg="Unprocessable",
|
||||
hdrs=None, # type: ignore[arg-type]
|
||||
fp=BytesIO(b'{"message": "validation failed"}'),
|
||||
)
|
||||
mock_urlopen.side_effect = error
|
||||
|
||||
with pytest.raises(SystemExit) as exc_info:
|
||||
create_report_issue("# Test Report")
|
||||
assert exc_info.value.code == 1
|
||||
|
||||
|
||||
class TestWriteReport:
|
||||
"""Tests for the write_report helper."""
|
||||
|
||||
@patch("builtins.open", mock_open())
|
||||
def test_writes_to_file(self):
|
||||
write_report("# Report Content")
|
||||
from builtins import open as builtin_open # noqa
|
||||
# Verify open was called with the right path
|
||||
from unittest.mock import call
|
||||
open_mock = open # The patched version
|
||||
open_mock.assert_called_once_with(REPORT_FILE, "w", encoding="utf-8") # type: ignore[attr-defined]
|
||||
open_mock().write.assert_called_once_with("# Report Content") # type: ignore[attr-defined]
|
||||
|
||||
|
||||
class TestFetchRepoLabels:
|
||||
"""Tests for fetch_repo_labels."""
|
||||
|
||||
@patch("sweep.github_api_get")
|
||||
def test_fetches_and_constructs_labels(self, mock_get):
|
||||
mock_get.return_value = [
|
||||
{"name": "bug", "description": "Something isn't working"},
|
||||
{"name": "enhancement", "description": "New feature or request"},
|
||||
{"name": "docs", "description": ""},
|
||||
]
|
||||
labels = fetch_repo_labels()
|
||||
assert len(labels) == 3
|
||||
assert labels[0]["name"] == "bug"
|
||||
assert labels[0]["text"] == "bug: Something isn't working"
|
||||
assert labels[2]["text"] == "docs" # no description, just name
|
||||
|
||||
@patch("sweep.github_api_get")
|
||||
def test_empty_repo_labels(self, mock_get):
|
||||
mock_get.return_value = []
|
||||
labels = fetch_repo_labels()
|
||||
assert labels == []
|
||||
|
||||
@patch("sweep.github_api_get")
|
||||
def test_null_description_handled(self, mock_get):
|
||||
mock_get.return_value = [
|
||||
{"name": "wontfix", "description": None},
|
||||
]
|
||||
labels = fetch_repo_labels()
|
||||
assert labels[0]["text"] == "wontfix"
|
||||
|
||||
@patch("sweep.API_PAGE_SIZE", 2)
|
||||
@patch("sweep.github_api_get")
|
||||
def test_label_pagination(self, mock_get):
|
||||
"""Repos with more labels than one page should fetch all pages."""
|
||||
mock_get.side_effect = [
|
||||
# First page: full (2 items = API_PAGE_SIZE)
|
||||
[
|
||||
{"name": "bug", "description": "Broken"},
|
||||
{"name": "feature", "description": "New"},
|
||||
],
|
||||
# Second page: partial (1 item < API_PAGE_SIZE) -> stop
|
||||
[
|
||||
{"name": "docs", "description": "Documentation"},
|
||||
],
|
||||
]
|
||||
labels = fetch_repo_labels()
|
||||
assert len(labels) == 3
|
||||
assert mock_get.call_count == 2
|
||||
assert labels[0]["name"] == "bug"
|
||||
assert labels[2]["name"] == "docs"
|
||||
|
||||
|
||||
class TestApplyLabelsToItem:
|
||||
"""Tests for apply_labels_to_item."""
|
||||
|
||||
def test_empty_labels_skips(self):
|
||||
# Should not make any API call
|
||||
apply_labels_to_item(1, [])
|
||||
|
||||
@patch("sweep.urllib.request.urlopen")
|
||||
def test_successful_label_application(self, mock_urlopen):
|
||||
mock_resp = MagicMock()
|
||||
mock_resp.read.return_value = b'[{"name": "bug"}]'
|
||||
mock_resp.__enter__ = lambda s: s
|
||||
mock_resp.__exit__ = MagicMock(return_value=False)
|
||||
mock_urlopen.return_value = mock_resp
|
||||
|
||||
# Should not raise
|
||||
apply_labels_to_item(42, ["bug", "enhancement"])
|
||||
|
||||
@patch("sweep.urllib.request.urlopen")
|
||||
def test_http_error_is_non_fatal(self, mock_urlopen):
|
||||
error = HTTPError(
|
||||
url="https://api.github.com/repos/owner/repo/issues/1/labels",
|
||||
code=404,
|
||||
msg="Not Found",
|
||||
hdrs=None, # type: ignore[arg-type]
|
||||
fp=BytesIO(b'{"message": "not found"}'),
|
||||
)
|
||||
mock_urlopen.side_effect = error
|
||||
|
||||
# Should NOT raise — labeling failures are warnings, not fatal
|
||||
apply_labels_to_item(1, ["bug"])
|
||||
|
||||
|
||||
class TestGenerateReportWithLabels:
|
||||
"""Tests for label suggestions in the report."""
|
||||
|
||||
def test_report_includes_label_section_high_confidence(self):
|
||||
"""High-confidence label (raw_sim >= 0.5) should appear in main table."""
|
||||
items = [
|
||||
TriageItem(
|
||||
number=1, title="Fix crash", html_url="https://example.com/1",
|
||||
is_pr=False, labels=[], created_at="2026-01-01T00:00:00Z", text="crash",
|
||||
),
|
||||
]
|
||||
suggestions = [[("bug", 0.85)]]
|
||||
report = generate_report(items, [], [], label_suggestions=suggestions)
|
||||
assert "Suggested Labels" in report
|
||||
assert "`bug` (0.85)" in report
|
||||
assert "1 items suggested for labeling" in report
|
||||
|
||||
def test_report_low_confidence_in_details(self):
|
||||
"""Low-confidence label (raw_sim < 0.5) should be in <details> section."""
|
||||
items = [
|
||||
TriageItem(
|
||||
number=1, title="Something", html_url="https://example.com/1",
|
||||
is_pr=False, labels=[], created_at="2026-01-01T00:00:00Z", text="something",
|
||||
),
|
||||
]
|
||||
suggestions = [[("maybe-bug", 0.35)]]
|
||||
report = generate_report(items, [], [], label_suggestions=suggestions)
|
||||
assert "Low-confidence suggestions" in report
|
||||
assert "<details>" in report
|
||||
assert "`maybe-bug` (0.35)" in report
|
||||
|
||||
def test_report_skips_already_labeled_items(self):
|
||||
items = [
|
||||
TriageItem(
|
||||
number=1, title="Already labeled", html_url="https://example.com/1",
|
||||
is_pr=False, labels=["bug"], created_at="2026-01-01T00:00:00Z", text="bug",
|
||||
),
|
||||
]
|
||||
suggestions = [[("bug", 0.95)]]
|
||||
report = generate_report(items, [], [], label_suggestions=suggestions)
|
||||
assert "0 items suggested for labeling" in report
|
||||
assert "No unlabeled items" in report
|
||||
|
||||
def test_report_excludes_outliers_from_suggestions(self):
|
||||
items = [
|
||||
TriageItem(
|
||||
number=1, title="Spam garbage", html_url="https://example.com/1",
|
||||
is_pr=False, labels=[], created_at="2026-01-01T00:00:00Z", text="spam",
|
||||
),
|
||||
TriageItem(
|
||||
number=2, title="Real bug", html_url="https://example.com/2",
|
||||
is_pr=False, labels=[], created_at="2026-01-01T00:00:00Z", text="bug",
|
||||
),
|
||||
]
|
||||
suggestions = [[("bug", 0.85)], [("bug", 0.90)]]
|
||||
# Item 0 is an outlier (with distance) — should be excluded from label suggestions
|
||||
report = generate_report(items, [(0, 15.2)], [], label_suggestions=suggestions)
|
||||
assert "1 unlabeled items" in report # only item 2
|
||||
assert "#2" in report
|
||||
# Item 0 (outlier) should NOT be in the suggestions table
|
||||
assert "Spam garbage" not in report.split("Suggested Labels")[1]
|
||||
|
||||
def test_report_without_label_suggestions(self):
|
||||
items = [
|
||||
TriageItem(
|
||||
number=1, title="T", html_url="u",
|
||||
is_pr=False, labels=[], created_at="2026-01-01T00:00:00Z", text="t",
|
||||
),
|
||||
]
|
||||
report = generate_report(items, [], [], label_suggestions=None)
|
||||
assert "Suggested Labels" not in report
|
||||
|
||||
def test_label_concentration_warning(self):
|
||||
"""When >50% of suggestions point to the same label, a warning should appear."""
|
||||
items = [
|
||||
TriageItem(
|
||||
number=i, title=f"Item {i}", html_url=f"https://example.com/{i}",
|
||||
is_pr=False, labels=[], created_at="2026-01-01T00:00:00Z", text=f"text {i}",
|
||||
)
|
||||
for i in range(4)
|
||||
]
|
||||
# 3 out of 4 items get "bug" label -> 75% concentration
|
||||
suggestions = [
|
||||
[("bug", 0.85)],
|
||||
[("bug", 0.80)],
|
||||
[("bug", 0.75)],
|
||||
[("enhancement", 0.90)],
|
||||
]
|
||||
report = generate_report(items, [], [], label_suggestions=suggestions)
|
||||
assert "Warning" in report
|
||||
assert "`bug`" in report
|
||||
assert "3/4" in report
|
||||
|
||||
|
||||
class TestMain:
|
||||
"""Tests for the main orchestration function."""
|
||||
|
||||
@patch.dict(os.environ, {"GITHUB_TOKEN": "", "GITHUB_REPOSITORY": "owner/repo"})
|
||||
def test_missing_token_exits(self):
|
||||
with pytest.raises(SystemExit) as exc_info:
|
||||
main()
|
||||
assert exc_info.value.code == 1
|
||||
|
||||
@patch.dict(os.environ, {"GITHUB_TOKEN": "tok", "GITHUB_REPOSITORY": ""})
|
||||
def test_missing_repo_exits(self):
|
||||
with pytest.raises(SystemExit) as exc_info:
|
||||
main()
|
||||
assert exc_info.value.code == 1
|
||||
|
||||
@patch("sweep.write_report")
|
||||
@patch("sweep.fetch_all_open_items", return_value=[])
|
||||
def test_no_items(self, mock_fetch, mock_write):
|
||||
main()
|
||||
mock_write.assert_called_once()
|
||||
report = mock_write.call_args[0][0]
|
||||
assert "No open issues or PRs found" in report
|
||||
|
||||
@patch("sweep.create_report_issue")
|
||||
@patch("sweep.write_report")
|
||||
@patch("sweep.suggest_labels", return_value=[])
|
||||
@patch("sweep.find_duplicate_pairs", return_value=[])
|
||||
@patch("sweep.detect_outliers", return_value=[])
|
||||
@patch("sweep.reduce_dimensions")
|
||||
@patch("sweep.normalize_rows")
|
||||
@patch("sweep.embed_texts")
|
||||
@patch("sweep.fetch_repo_labels")
|
||||
@patch("sweep.fetch_all_open_items")
|
||||
def test_full_flow_with_enough_items(
|
||||
self, mock_fetch, mock_labels, mock_embed, mock_norm, mock_reduce,
|
||||
mock_outliers, mock_dupes, mock_suggest, mock_write, mock_create,
|
||||
):
|
||||
"""Test the full flow with >= MIN_SAMPLES items (outlier detection runs)."""
|
||||
n = MIN_SAMPLES_FOR_OUTLIER_DETECTION
|
||||
items = [
|
||||
TriageItem(
|
||||
number=i, title=f"Item {i}", html_url=f"https://example.com/{i}",
|
||||
is_pr=False, labels=[], created_at="2026-01-01T00:00:00Z", text=f"text {i}",
|
||||
)
|
||||
for i in range(n)
|
||||
]
|
||||
mock_fetch.return_value = items
|
||||
mock_labels.return_value = [
|
||||
RepoLabel(name="bug", description="Something broken", text="bug: Something broken"),
|
||||
]
|
||||
|
||||
embeddings = np.random.randn(n, 384).astype(np.float32)
|
||||
mock_embed.return_value = embeddings
|
||||
mock_norm.return_value = embeddings
|
||||
mock_reduce.return_value = np.random.randn(n, 10).astype(np.float32)
|
||||
|
||||
main()
|
||||
|
||||
mock_fetch.assert_called_once()
|
||||
mock_labels.assert_called_once()
|
||||
# embed_texts called twice: once for items, once for labels
|
||||
assert mock_embed.call_count == 2
|
||||
mock_norm.assert_called()
|
||||
mock_reduce.assert_called_once()
|
||||
mock_outliers.assert_called_once()
|
||||
mock_dupes.assert_called_once()
|
||||
mock_suggest.assert_called_once()
|
||||
mock_write.assert_called_once()
|
||||
mock_create.assert_called_once()
|
||||
|
||||
@patch("sweep.create_report_issue")
|
||||
@patch("sweep.write_report")
|
||||
@patch("sweep.suggest_labels", return_value=[])
|
||||
@patch("sweep.find_duplicate_pairs", return_value=[])
|
||||
@patch("sweep.detect_outliers")
|
||||
@patch("sweep.reduce_dimensions")
|
||||
@patch("sweep.normalize_rows")
|
||||
@patch("sweep.embed_texts")
|
||||
@patch("sweep.fetch_repo_labels", return_value=[])
|
||||
@patch("sweep.fetch_all_open_items")
|
||||
def test_skips_outlier_detection_for_few_items(
|
||||
self, mock_fetch, mock_labels, mock_embed, mock_norm, mock_reduce,
|
||||
mock_outliers, mock_dupes, mock_suggest, mock_write, mock_create,
|
||||
):
|
||||
"""With < MIN_SAMPLES items, outlier detection should be skipped."""
|
||||
n = MIN_SAMPLES_FOR_OUTLIER_DETECTION - 1
|
||||
items = [
|
||||
TriageItem(
|
||||
number=i, title=f"Item {i}", html_url=f"https://example.com/{i}",
|
||||
is_pr=False, labels=[], created_at="2026-01-01T00:00:00Z", text=f"text {i}",
|
||||
)
|
||||
for i in range(n)
|
||||
]
|
||||
mock_fetch.return_value = items
|
||||
|
||||
embeddings = np.random.randn(n, 384).astype(np.float32)
|
||||
mock_embed.return_value = embeddings
|
||||
mock_norm.return_value = embeddings
|
||||
|
||||
main()
|
||||
|
||||
# Outlier detection should not have been called
|
||||
mock_reduce.assert_not_called()
|
||||
mock_outliers.assert_not_called()
|
||||
# But duplicates should still be checked
|
||||
mock_dupes.assert_called_once()
|
||||
|
||||
@patch.dict(os.environ, {"INPUT_DRY_RUN": "true"})
|
||||
@patch("sweep.DRY_RUN", True)
|
||||
@patch("sweep.write_report")
|
||||
@patch("sweep.create_report_issue")
|
||||
@patch("sweep.apply_labels_to_item")
|
||||
@patch("sweep.suggest_labels", return_value=[[("bug", 0.85)]])
|
||||
@patch("sweep.find_duplicate_pairs", return_value=[])
|
||||
@patch("sweep.normalize_rows")
|
||||
@patch("sweep.embed_texts")
|
||||
@patch("sweep.fetch_repo_labels")
|
||||
@patch("sweep.fetch_all_open_items")
|
||||
def test_dry_run_skips_issue_creation_and_labeling(
|
||||
self, mock_fetch, mock_labels, mock_embed, mock_norm,
|
||||
mock_dupes, mock_suggest, mock_apply, mock_create, mock_write,
|
||||
):
|
||||
items = [
|
||||
TriageItem(
|
||||
number=1, title="Item", html_url="https://example.com/1",
|
||||
is_pr=False, labels=[], created_at="2026-01-01T00:00:00Z", text="text",
|
||||
)
|
||||
]
|
||||
mock_fetch.return_value = items
|
||||
mock_labels.return_value = [
|
||||
RepoLabel(name="bug", description="Broken", text="bug: Broken"),
|
||||
]
|
||||
embeddings = np.random.randn(1, 384).astype(np.float32)
|
||||
mock_embed.return_value = embeddings
|
||||
mock_norm.return_value = embeddings
|
||||
|
||||
main()
|
||||
|
||||
mock_create.assert_not_called()
|
||||
mock_apply.assert_not_called()
|
||||
mock_write.assert_called_once()
|
||||
|
||||
@patch("sweep.create_report_issue")
|
||||
@patch("sweep.write_report")
|
||||
@patch("sweep.apply_labels_to_item")
|
||||
@patch("sweep.suggest_labels")
|
||||
@patch("sweep.find_duplicate_pairs", return_value=[])
|
||||
@patch("sweep.normalize_rows")
|
||||
@patch("sweep.embed_texts")
|
||||
@patch("sweep.fetch_repo_labels")
|
||||
@patch("sweep.fetch_all_open_items")
|
||||
def test_labels_not_auto_applied(
|
||||
self, mock_fetch, mock_labels, mock_embed, mock_norm,
|
||||
mock_dupes, mock_suggest, mock_apply, mock_write, mock_create,
|
||||
):
|
||||
"""Auto-labeling is disabled; labels should appear in report only."""
|
||||
items = [
|
||||
TriageItem(
|
||||
number=1, title="Crash bug", html_url="https://example.com/1",
|
||||
is_pr=False, labels=[], created_at="2026-01-01T00:00:00Z", text="crash",
|
||||
),
|
||||
TriageItem(
|
||||
number=2, title="Already labeled", html_url="https://example.com/2",
|
||||
is_pr=False, labels=["enhancement"], created_at="2026-01-01T00:00:00Z", text="feat",
|
||||
),
|
||||
]
|
||||
mock_fetch.return_value = items
|
||||
mock_labels.return_value = [
|
||||
RepoLabel(name="bug", description="Broken", text="bug: Broken"),
|
||||
]
|
||||
mock_suggest.return_value = [
|
||||
[("bug", 0.90)],
|
||||
[("bug", 0.45)],
|
||||
]
|
||||
|
||||
embeddings = np.random.randn(2, 384).astype(np.float32)
|
||||
mock_embed.return_value = embeddings
|
||||
mock_norm.return_value = embeddings
|
||||
|
||||
main()
|
||||
|
||||
# Auto-labeling is disabled — apply_labels_to_item should never be called
|
||||
mock_apply.assert_not_called()
|
||||
|
||||
@patch("sweep.create_report_issue")
|
||||
@patch("sweep.write_report")
|
||||
@patch("sweep.apply_labels_to_item")
|
||||
@patch("sweep.suggest_labels")
|
||||
@patch("sweep.find_duplicate_pairs", return_value=[])
|
||||
@patch("sweep.detect_outliers")
|
||||
@patch("sweep.reduce_dimensions")
|
||||
@patch("sweep.normalize_rows")
|
||||
@patch("sweep.embed_texts")
|
||||
@patch("sweep.fetch_repo_labels")
|
||||
@patch("sweep.fetch_all_open_items")
|
||||
def test_outliers_excluded_from_report_suggestions(
|
||||
self, mock_fetch, mock_labels, mock_embed, mock_norm, mock_reduce,
|
||||
mock_outliers, mock_dupes, mock_suggest, mock_apply, mock_write, mock_create,
|
||||
):
|
||||
"""Items flagged as outliers should not appear in report label suggestions."""
|
||||
n = MIN_SAMPLES_FOR_OUTLIER_DETECTION
|
||||
items = [
|
||||
TriageItem(
|
||||
number=i, title=f"Item {i}", html_url=f"https://example.com/{i}",
|
||||
is_pr=False, labels=[], created_at="2026-01-01T00:00:00Z", text=f"text {i}",
|
||||
)
|
||||
for i in range(n)
|
||||
]
|
||||
mock_fetch.return_value = items
|
||||
mock_labels.return_value = [
|
||||
RepoLabel(name="bug", description="Broken", text="bug: Broken"),
|
||||
]
|
||||
mock_outliers.return_value = [(0, 12.5), (5, 15.3)]
|
||||
mock_suggest.return_value = [[("bug", 0.85)] for _ in range(n)]
|
||||
|
||||
embeddings = np.random.randn(n, 384).astype(np.float32)
|
||||
mock_embed.return_value = embeddings
|
||||
mock_norm.return_value = embeddings
|
||||
mock_reduce.return_value = np.random.randn(n, 10).astype(np.float32)
|
||||
|
||||
main()
|
||||
|
||||
# Auto-labeling is disabled
|
||||
mock_apply.assert_not_called()
|
||||
# Report should still be generated (outliers excluded from suggestions in report)
|
||||
mock_write.assert_called_once()
|
||||
report = mock_write.call_args[0][0]
|
||||
# Outlier items 0 and 5 should not appear in the label suggestions section
|
||||
assert "Item 0" not in report.split("Suggested Labels")[1] if "Suggested Labels" in report else True
|
||||
@@ -0,0 +1,91 @@
|
||||
name: E2E Tests
|
||||
|
||||
on:
|
||||
workflow_call:
|
||||
|
||||
jobs:
|
||||
check-changes:
|
||||
name: Check web module changes
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 5
|
||||
outputs:
|
||||
web_changed: ${{ steps.filter.outputs.web }}
|
||||
steps:
|
||||
- uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4
|
||||
- uses: dorny/paths-filter@de90cc6fb38fc0963ad72b210f1f284cd68cea36 # v3
|
||||
id: filter
|
||||
with:
|
||||
filters: |
|
||||
web:
|
||||
- 'gitnexus-web/**'
|
||||
|
||||
e2e:
|
||||
name: e2e (chromium)
|
||||
needs: check-changes
|
||||
if: needs.check-changes.result == 'success' && needs.check-changes.outputs.web_changed == 'true'
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 20
|
||||
steps:
|
||||
- uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4
|
||||
|
||||
- uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4
|
||||
with:
|
||||
node-version: 20
|
||||
cache: npm
|
||||
cache-dependency-path: gitnexus-web/package-lock.json
|
||||
|
||||
- name: Install frontend dependencies
|
||||
run: npm ci
|
||||
working-directory: gitnexus-web
|
||||
|
||||
- name: Install Playwright browsers
|
||||
run: npx playwright install --with-deps chromium
|
||||
working-directory: gitnexus-web
|
||||
|
||||
- name: Install backend dependencies
|
||||
run: npm ci
|
||||
working-directory: gitnexus
|
||||
|
||||
- name: Build backend
|
||||
run: npm run build
|
||||
working-directory: gitnexus
|
||||
|
||||
- name: Analyze repository (index for backend)
|
||||
run: |
|
||||
node gitnexus/dist/cli/index.js analyze || true
|
||||
if [ ! -d ".gitnexus" ]; then
|
||||
echo "::error::No .gitnexus index created"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
- name: Start backend server
|
||||
run: node dist/cli/index.js serve &
|
||||
working-directory: gitnexus
|
||||
|
||||
- name: Wait for backend readiness
|
||||
run: npx wait-on http://localhost:4747/api/repos --timeout 30000
|
||||
working-directory: gitnexus-web
|
||||
|
||||
- name: Start Vite dev server
|
||||
run: npm run dev &
|
||||
working-directory: gitnexus-web
|
||||
|
||||
- name: Wait for Vite dev server
|
||||
run: npx wait-on http://localhost:5173 --timeout 30000
|
||||
working-directory: gitnexus-web
|
||||
|
||||
- name: Run E2E tests
|
||||
run: npx playwright test
|
||||
working-directory: gitnexus-web
|
||||
env:
|
||||
E2E: '1'
|
||||
|
||||
- name: Upload test results
|
||||
if: always()
|
||||
uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4
|
||||
with:
|
||||
name: e2e-results
|
||||
path: |
|
||||
gitnexus-web/test-results/
|
||||
gitnexus-web/playwright-report/
|
||||
retention-days: 5
|
||||
@@ -0,0 +1,29 @@
|
||||
name: Quality Checks
|
||||
|
||||
on:
|
||||
workflow_call:
|
||||
|
||||
jobs:
|
||||
typecheck:
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 10
|
||||
steps:
|
||||
- uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4
|
||||
- uses: ./.github/actions/setup-gitnexus
|
||||
- run: npx tsc --noEmit
|
||||
working-directory: gitnexus
|
||||
|
||||
typecheck-web:
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 10
|
||||
steps:
|
||||
- uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4
|
||||
- uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4
|
||||
with:
|
||||
node-version: 20
|
||||
cache: npm
|
||||
cache-dependency-path: gitnexus-web/package-lock.json
|
||||
- run: npm ci
|
||||
working-directory: gitnexus-web
|
||||
- run: npx tsc -b --noEmit
|
||||
working-directory: gitnexus-web
|
||||
@@ -0,0 +1,413 @@
|
||||
name: CI Report
|
||||
|
||||
# Triggered after the CI workflow completes. Because workflow_run
|
||||
# always runs code from the *default branch*, it receives a read/write
|
||||
# GITHUB_TOKEN — even when the triggering PR comes from a fork.
|
||||
|
||||
on:
|
||||
workflow_run:
|
||||
workflows: ["CI"]
|
||||
types: [completed]
|
||||
|
||||
permissions:
|
||||
actions: read # needed to list/download workflow run artifacts
|
||||
contents: read # needed for sparse checkout of vitest.config.ts
|
||||
pull-requests: write # needed to post sticky PR comment
|
||||
|
||||
jobs:
|
||||
pr-report:
|
||||
name: PR Report
|
||||
# Only run for pull-request CI runs
|
||||
if: >-
|
||||
github.event.workflow_run.event == 'pull_request' &&
|
||||
github.event.workflow_run.conclusion != 'cancelled'
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 5
|
||||
steps:
|
||||
# ── Download artifacts from the CI run ────────────────────────
|
||||
- name: Download artifacts
|
||||
uses: actions/github-script@60a0d83039c74a4aee543508d2ffcb1c3799cdea # v7
|
||||
with:
|
||||
script: |
|
||||
const fs = require('fs');
|
||||
const path = require('path');
|
||||
const runId = context.payload.workflow_run.id;
|
||||
|
||||
const allArtifacts = await github.rest.actions.listWorkflowRunArtifacts({
|
||||
owner: context.repo.owner,
|
||||
repo: context.repo.repo,
|
||||
run_id: runId,
|
||||
});
|
||||
|
||||
async function downloadArtifact(name, dest) {
|
||||
const match = allArtifacts.data.artifacts.find(a => a.name === name);
|
||||
if (!match) {
|
||||
core.warning(`Artifact "${name}" not found`);
|
||||
return false;
|
||||
}
|
||||
const zip = await github.rest.actions.downloadArtifact({
|
||||
owner: context.repo.owner,
|
||||
repo: context.repo.repo,
|
||||
artifact_id: match.id,
|
||||
archive_format: 'zip',
|
||||
});
|
||||
fs.mkdirSync(dest, { recursive: true });
|
||||
fs.writeFileSync(path.join(dest, `${name}.zip`), Buffer.from(zip.data));
|
||||
return true;
|
||||
}
|
||||
|
||||
const temp = process.env.RUNNER_TEMP;
|
||||
await downloadArtifact('pr-meta', path.join(temp, 'dl'));
|
||||
await downloadArtifact('test-reports', path.join(temp, 'dl'));
|
||||
|
||||
- name: Extract artifacts
|
||||
shell: bash
|
||||
run: |
|
||||
cd "$RUNNER_TEMP/dl"
|
||||
# Extract each artifact into its own directory to avoid filename collisions
|
||||
for z in *.zip; do
|
||||
[ -f "$z" ] || continue
|
||||
name="${z%.zip}"
|
||||
mkdir -p "$RUNNER_TEMP/artifacts/$name"
|
||||
unzip -o "$z" -d "$RUNNER_TEMP/artifacts/$name"
|
||||
done
|
||||
|
||||
- name: Read PR metadata
|
||||
id: meta
|
||||
shell: bash
|
||||
run: |
|
||||
DIR="$RUNNER_TEMP/artifacts/pr-meta"
|
||||
if [ ! -f "$DIR/pr_number" ]; then
|
||||
echo "skip=true" >> "$GITHUB_OUTPUT"
|
||||
echo "::warning::pr_number artifact missing — skipping report"
|
||||
exit 0
|
||||
fi
|
||||
|
||||
# Validate PR number is a positive integer (artifact comes from
|
||||
# untrusted fork code, so treat contents defensively).
|
||||
PR_NUM=$(cat "$DIR/pr_number" | tr -d '[:space:]')
|
||||
if ! [[ "$PR_NUM" =~ ^[0-9]+$ ]]; then
|
||||
echo "skip=true" >> "$GITHUB_OUTPUT"
|
||||
echo "::error::Invalid PR number in artifact: '$PR_NUM'"
|
||||
exit 0
|
||||
fi
|
||||
|
||||
echo "skip=false" >> "$GITHUB_OUTPUT"
|
||||
echo "pr_number=$PR_NUM" >> "$GITHUB_OUTPUT"
|
||||
# Validate job-result strings against known GitHub Actions values.
|
||||
# Artifact contents come from the PR workflow (potentially untrusted
|
||||
# fork code), so we whitelist to prevent newline injection into
|
||||
# GITHUB_OUTPUT.
|
||||
validate_result() {
|
||||
local val
|
||||
val=$(cat "$1" | tr -d '[:space:]')
|
||||
case "$val" in
|
||||
success|failure|cancelled|skipped) echo "$val" ;;
|
||||
*) echo "unknown" ;;
|
||||
esac
|
||||
}
|
||||
|
||||
echo "quality=$(validate_result "$DIR/quality_result")" >> "$GITHUB_OUTPUT"
|
||||
echo "tests=$(validate_result "$DIR/tests_result")" >> "$GITHUB_OUTPUT"
|
||||
echo "e2e=$(validate_result "$DIR/e2e_result")" >> "$GITHUB_OUTPUT"
|
||||
|
||||
- name: Checkout (for vitest config)
|
||||
if: steps.meta.outputs.skip != 'true'
|
||||
uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4
|
||||
with:
|
||||
sparse-checkout: gitnexus/vitest.config.ts
|
||||
sparse-checkout-cone-mode: false
|
||||
|
||||
# ── Fetch base branch coverage for delta reporting ───────────
|
||||
- name: Fetch base branch coverage
|
||||
if: steps.meta.outputs.skip != 'true'
|
||||
id: base-coverage
|
||||
uses: actions/github-script@60a0d83039c74a4aee543508d2ffcb1c3799cdea # v7
|
||||
with:
|
||||
script: |
|
||||
const fs = require('fs');
|
||||
const path = require('path');
|
||||
|
||||
// Find the latest successful CI run on main
|
||||
const runs = await github.rest.actions.listWorkflowRuns({
|
||||
owner: context.repo.owner,
|
||||
repo: context.repo.repo,
|
||||
workflow_id: 'ci.yml',
|
||||
branch: 'main',
|
||||
status: 'success',
|
||||
per_page: 1,
|
||||
});
|
||||
|
||||
if (runs.data.workflow_runs.length === 0) {
|
||||
core.setOutput('found', 'false');
|
||||
core.info('No successful main branch CI runs found');
|
||||
return;
|
||||
}
|
||||
|
||||
const mainRunId = runs.data.workflow_runs[0].id;
|
||||
const artifacts = await github.rest.actions.listWorkflowRunArtifacts({
|
||||
owner: context.repo.owner,
|
||||
repo: context.repo.repo,
|
||||
run_id: mainRunId,
|
||||
});
|
||||
|
||||
const testReports = artifacts.data.artifacts.find(a => a.name === 'test-reports');
|
||||
if (!testReports) {
|
||||
core.setOutput('found', 'false');
|
||||
core.info('No test-reports artifact on main branch');
|
||||
return;
|
||||
}
|
||||
|
||||
const zip = await github.rest.actions.downloadArtifact({
|
||||
owner: context.repo.owner,
|
||||
repo: context.repo.repo,
|
||||
artifact_id: testReports.id,
|
||||
archive_format: 'zip',
|
||||
});
|
||||
|
||||
const dest = path.join(process.env.RUNNER_TEMP, 'base-coverage');
|
||||
fs.mkdirSync(dest, { recursive: true });
|
||||
fs.writeFileSync(path.join(dest, 'base.zip'), Buffer.from(zip.data));
|
||||
core.setOutput('found', 'true');
|
||||
core.setOutput('dir', dest);
|
||||
|
||||
- name: Extract base coverage
|
||||
if: steps.meta.outputs.skip != 'true' && steps.base-coverage.outputs.found == 'true'
|
||||
shell: bash
|
||||
run: |
|
||||
cd "${{ steps.base-coverage.outputs.dir }}"
|
||||
mkdir -p base
|
||||
unzip -o base.zip -d base
|
||||
|
||||
- name: Build report
|
||||
if: steps.meta.outputs.skip != 'true'
|
||||
id: report
|
||||
shell: bash
|
||||
env:
|
||||
QUALITY: ${{ steps.meta.outputs.quality }}
|
||||
TESTS: ${{ steps.meta.outputs.tests }}
|
||||
E2E: ${{ steps.meta.outputs.e2e }}
|
||||
BASE_FOUND: ${{ steps.base-coverage.outputs.found }}
|
||||
BASE_DIR: ${{ steps.base-coverage.outputs.dir }}
|
||||
RUN_URL: ${{ github.event.workflow_run.html_url }}
|
||||
run: |
|
||||
DIR="$RUNNER_TEMP/artifacts"
|
||||
|
||||
# ── Helper: read coverage summary into prefixed vars ──
|
||||
read_cov() {
|
||||
local prefix=$1 file=$2
|
||||
if [ -n "$file" ] && [ -f "$file" ]; then
|
||||
local val
|
||||
val=$(jq -r '.total.statements.pct // "N/A"' "$file" 2>/dev/null) || val="N/A"
|
||||
printf -v "${prefix}_STMTS" '%s' "$val"
|
||||
val=$(jq -r '.total.branches.pct // "N/A"' "$file" 2>/dev/null) || val="N/A"
|
||||
printf -v "${prefix}_BRANCH" '%s' "$val"
|
||||
val=$(jq -r '.total.functions.pct // "N/A"' "$file" 2>/dev/null) || val="N/A"
|
||||
printf -v "${prefix}_FUNCS" '%s' "$val"
|
||||
val=$(jq -r '.total.lines.pct // "N/A"' "$file" 2>/dev/null) || val="N/A"
|
||||
printf -v "${prefix}_LINES" '%s' "$val"
|
||||
val=$(jq -r '"\(.total.statements.covered)/\(.total.statements.total)"' "$file" 2>/dev/null) || val=""
|
||||
printf -v "${prefix}_STMTS_COV" '%s' "$val"
|
||||
val=$(jq -r '"\(.total.branches.covered)/\(.total.branches.total)"' "$file" 2>/dev/null) || val=""
|
||||
printf -v "${prefix}_BRANCH_COV" '%s' "$val"
|
||||
val=$(jq -r '"\(.total.functions.covered)/\(.total.functions.total)"' "$file" 2>/dev/null) || val=""
|
||||
printf -v "${prefix}_FUNCS_COV" '%s' "$val"
|
||||
val=$(jq -r '"\(.total.lines.covered)/\(.total.lines.total)"' "$file" 2>/dev/null) || val=""
|
||||
printf -v "${prefix}_LINES_COV" '%s' "$val"
|
||||
return 0
|
||||
else
|
||||
printf -v "${prefix}_STMTS" '%s' "N/A"
|
||||
printf -v "${prefix}_BRANCH" '%s' "N/A"
|
||||
printf -v "${prefix}_FUNCS" '%s' "N/A"
|
||||
printf -v "${prefix}_LINES" '%s' "N/A"
|
||||
printf -v "${prefix}_STMTS_COV" '%s' ""
|
||||
printf -v "${prefix}_BRANCH_COV" '%s' ""
|
||||
printf -v "${prefix}_FUNCS_COV" '%s' ""
|
||||
printf -v "${prefix}_LINES_COV" '%s' ""
|
||||
return 1
|
||||
fi
|
||||
}
|
||||
|
||||
# ── Read coverage reports ──
|
||||
UNIT_SUMMARY=$(find "$DIR/test-reports" -name "coverage-summary.json" -type f 2>/dev/null | head -1)
|
||||
|
||||
read_cov "U" "$UNIT_SUMMARY"
|
||||
|
||||
# ── Read base branch coverage (main) ──
|
||||
BASE_SUMMARY=""
|
||||
if [ "$BASE_FOUND" = "true" ] && [ -n "$BASE_DIR" ]; then
|
||||
BASE_SUMMARY=$(find "$BASE_DIR/base" -name "coverage-summary.json" -type f 2>/dev/null | head -1)
|
||||
fi
|
||||
read_cov "B" "$BASE_SUMMARY"
|
||||
|
||||
# ── Locate test results ──
|
||||
RESULTS_FILE=$(find "$DIR/test-reports" -name "test-results.json" -type f 2>/dev/null | head -1)
|
||||
WEB_RESULTS_FILE=$(find "$DIR/test-reports" -name "web-test-results.json" -type f 2>/dev/null | head -1)
|
||||
|
||||
sum_results() {
|
||||
local file=$1
|
||||
if [ -n "$file" ] && [ -f "$file" ]; then
|
||||
jq -r '"\(.numTotalTests) \(.numPassedTests) \(.numFailedTests) \(.numPendingTests) \(.numTotalTestSuites) \(((.testResults | map(.endTime) | max) - (.startTime)) / 1000 | floor)"' "$file" 2>/dev/null || echo "0 0 0 0 0 0"
|
||||
else
|
||||
echo "0 0 0 0 0 0"
|
||||
fi
|
||||
}
|
||||
|
||||
read CLI_T CLI_P CLI_F CLI_S CLI_SU CLI_D <<< "$(sum_results "$RESULTS_FILE")"
|
||||
read WEB_T WEB_P WEB_F WEB_S WEB_SU WEB_D <<< "$(sum_results "$WEB_RESULTS_FILE")"
|
||||
|
||||
TOTAL=$((CLI_T + WEB_T))
|
||||
PASSED=$((CLI_P + WEB_P))
|
||||
FAILED=$((CLI_F + WEB_F))
|
||||
SKIPPED=$((CLI_S + WEB_S))
|
||||
SUITES=$((CLI_SU + WEB_SU))
|
||||
DURATION=$((CLI_D > WEB_D ? CLI_D : WEB_D))
|
||||
|
||||
# ── Status helpers ──
|
||||
status_icon() {
|
||||
case "$1" in
|
||||
success) echo "✅" ;;
|
||||
failure) echo "❌" ;;
|
||||
cancelled) echo "⏭️" ;;
|
||||
*) echo "❓" ;;
|
||||
esac
|
||||
}
|
||||
|
||||
# Validate a value looks like a number (integer or decimal, optional
|
||||
# leading minus). Returns 1 for anything else — guards against awk
|
||||
# injection when artifact values come from untrusted fork code.
|
||||
is_numeric() { [[ "$1" =~ ^-?[0-9]+(\.[0-9]+)?$ ]]; }
|
||||
|
||||
cov_delta() {
|
||||
local pct=$1 base=$2
|
||||
if [ "$pct" = "N/A" ] || [ "$base" = "N/A" ]; then echo "—"; return; fi
|
||||
if ! is_numeric "$pct" || ! is_numeric "$base"; then echo "—"; return; fi
|
||||
local diff
|
||||
diff=$(awk -v p="$pct" -v b="$base" 'BEGIN { printf "%.1f", p - b }')
|
||||
if [ "$(awk -v p="$pct" -v b="$base" 'BEGIN { print (p > b) ? 1 : 0 }')" = "1" ]; then
|
||||
echo "📈 +${diff}"
|
||||
elif [ "$(awk -v p="$pct" -v b="$base" 'BEGIN { print (p < b) ? 1 : 0 }')" = "1" ]; then
|
||||
echo "📉 ${diff}"
|
||||
else
|
||||
echo "= ${diff}"
|
||||
fi
|
||||
}
|
||||
|
||||
cov_bar() {
|
||||
local pct=$1 base=$2
|
||||
if [ "$pct" = "N/A" ] || ! is_numeric "$pct"; then echo "—"; return; fi
|
||||
local filled
|
||||
filled=$(awk -v p="$pct" 'BEGIN { printf "%d", p / 5 }')
|
||||
(( filled < 0 )) && filled=0
|
||||
(( filled > 20 )) && filled=20
|
||||
local empty=$((20 - filled))
|
||||
local bar=""
|
||||
for ((i=0; i<filled; i++)); do bar+="█"; done
|
||||
for ((i=0; i<empty; i++)); do bar+="░"; done
|
||||
# Green if >= base (or base unavailable), red if dropped
|
||||
if [ "$base" = "N/A" ] || ! is_numeric "$base" || [ "$(awk -v p="$pct" -v b="$base" 'BEGIN { print (p >= b) ? 1 : 0 }')" = "1" ]; then
|
||||
echo "🟢 ${bar}"
|
||||
else
|
||||
echo "🔴 ${bar}"
|
||||
fi
|
||||
}
|
||||
|
||||
# ── Overall status ──
|
||||
if [[ "$QUALITY" == "success" && "$TESTS" == "success" && ("$E2E" == "success" || "$E2E" == "skipped") ]]; then
|
||||
OVERALL="✅ **All checks passed**"
|
||||
else
|
||||
OVERALL="❌ **Some checks failed**"
|
||||
fi
|
||||
|
||||
# ── Build markdown ──
|
||||
{
|
||||
echo "body<<GITNEXUS_CI_REPORT_EOF_7f3a"
|
||||
echo "## CI Report"
|
||||
echo ""
|
||||
echo "${OVERALL}"
|
||||
echo ""
|
||||
echo "### Pipeline Status"
|
||||
echo ""
|
||||
echo "| Stage | Status | Details |"
|
||||
echo "|-------|--------|---------|"
|
||||
echo "| $(status_icon "$QUALITY") Typecheck | \`${QUALITY}\` | tsc --noEmit |"
|
||||
echo "| $(status_icon "$TESTS") Tests | \`${TESTS}\` | unit tests, 3 platforms |"
|
||||
echo "| $(status_icon "$E2E") E2E | \`${E2E}\` | gitnexus-web changes only |"
|
||||
echo ""
|
||||
|
||||
if [ "$TOTAL" -gt 0 ] 2>/dev/null; then
|
||||
echo "### Test Results"
|
||||
echo ""
|
||||
echo "| Tests | Passed | Failed | Skipped | Duration |"
|
||||
echo "|-------|--------|--------|---------|----------|"
|
||||
echo "| ${TOTAL} | ${PASSED} | ${FAILED} | ${SKIPPED} | ${DURATION}s |"
|
||||
echo ""
|
||||
|
||||
if [ "$FAILED" = "0" ]; then
|
||||
echo "✅ All **${PASSED}** tests passed"
|
||||
else
|
||||
echo "❌ **${FAILED}** failed / **${PASSED}** passed"
|
||||
fi
|
||||
if [ "$SKIPPED" != "0" ]; then
|
||||
echo ""
|
||||
echo "<details>"
|
||||
echo "<summary>${SKIPPED} test(s) skipped — expand for details</summary>"
|
||||
echo ""
|
||||
for rf in "$RESULTS_FILE" "$WEB_RESULTS_FILE"; do
|
||||
if [ -n "$rf" ] && [ -f "$rf" ]; then
|
||||
jq -r '
|
||||
.testResults[]
|
||||
| .assertionResults[]?
|
||||
| select(.status == "pending" or .status == "skipped")
|
||||
| "- \(.ancestorTitles | join(" > ")) > \(.title)"
|
||||
' "$rf" 2>/dev/null || true
|
||||
fi
|
||||
done
|
||||
echo ""
|
||||
echo "</details>"
|
||||
fi
|
||||
echo ""
|
||||
fi
|
||||
|
||||
# ── Coverage table helper ──
|
||||
cov_table() {
|
||||
local label=$1 s=$2 b=$3 f=$4 l=$5 sc=$6 bc=$7 fc=$8 lc=$9
|
||||
shift 9
|
||||
local bs=$1 bb=$2 bf=$3 bl=$4
|
||||
echo "#### ${label}"
|
||||
echo ""
|
||||
echo "| Metric | Coverage | Covered | Base | Delta | Status |"
|
||||
echo "|--------|----------|---------|------|-------|--------|"
|
||||
echo "| Statements | **${s}%** | ${sc} | ${bs}% | $(cov_delta "$s" "$bs") | $(cov_bar "$s" "$bs") |"
|
||||
echo "| Branches | **${b}%** | ${bc} | ${bb}% | $(cov_delta "$b" "$bb") | $(cov_bar "$b" "$bb") |"
|
||||
echo "| Functions | **${f}%** | ${fc} | ${bf}% | $(cov_delta "$f" "$bf") | $(cov_bar "$f" "$bf") |"
|
||||
echo "| Lines | **${l}%** | ${lc} | ${bl}% | $(cov_delta "$l" "$bl") | $(cov_bar "$l" "$bl") |"
|
||||
echo ""
|
||||
}
|
||||
|
||||
if [ "$U_STMTS" != "N/A" ]; then
|
||||
echo "### Code Coverage"
|
||||
echo ""
|
||||
cov_table "Tests" \
|
||||
"$U_STMTS" "$U_BRANCH" "$U_FUNCS" "$U_LINES" \
|
||||
"$U_STMTS_COV" "$U_BRANCH_COV" "$U_FUNCS_COV" "$U_LINES_COV" \
|
||||
"$B_STMTS" "$B_BRANCH" "$B_FUNCS" "$B_LINES"
|
||||
else
|
||||
echo "### Code Coverage"
|
||||
echo ""
|
||||
echo "⚠️ Coverage data unavailable - check the [unit test job](${RUN_URL}) for details."
|
||||
echo ""
|
||||
fi
|
||||
|
||||
echo "---"
|
||||
echo "<sub>📋 [View full run](${RUN_URL}) · Generated by CI</sub>"
|
||||
echo "GITNEXUS_CI_REPORT_EOF_7f3a"
|
||||
} >> "$GITHUB_OUTPUT"
|
||||
|
||||
- name: Comment on PR
|
||||
if: steps.meta.outputs.skip != 'true'
|
||||
uses: marocchino/sticky-pull-request-comment@773744901bac0e8cbb5a0dc842800d45e9b2b405 # v2
|
||||
with:
|
||||
header: ci-report
|
||||
number: ${{ steps.meta.outputs.pr_number }}
|
||||
message: ${{ steps.report.outputs.body }}
|
||||
@@ -0,0 +1,70 @@
|
||||
name: Tests
|
||||
|
||||
on:
|
||||
workflow_call:
|
||||
|
||||
jobs:
|
||||
tests:
|
||||
name: ubuntu / coverage
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 25
|
||||
steps:
|
||||
- uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4
|
||||
- uses: ./.github/actions/setup-gitnexus
|
||||
with:
|
||||
build: 'true'
|
||||
|
||||
- name: Run all tests with coverage
|
||||
run: >-
|
||||
npx vitest run
|
||||
--reporter=default
|
||||
--reporter=json
|
||||
--outputFile=test-results.json
|
||||
--coverage
|
||||
--coverage.reporter=json-summary
|
||||
--coverage.reporter=json
|
||||
--coverage.reporter=text
|
||||
--coverage.thresholdAutoUpdate=false
|
||||
--coverage.reportOnFailure=true
|
||||
working-directory: gitnexus
|
||||
|
||||
- name: Install gitnexus-web dependencies
|
||||
run: npm ci
|
||||
working-directory: gitnexus-web
|
||||
|
||||
- name: Run gitnexus-web unit tests
|
||||
run: >-
|
||||
npx vitest run
|
||||
--reporter=default
|
||||
--reporter=json
|
||||
--outputFile=web-test-results.json
|
||||
working-directory: gitnexus-web
|
||||
|
||||
- name: Upload test reports
|
||||
if: always()
|
||||
uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4
|
||||
with:
|
||||
name: test-reports
|
||||
path: |
|
||||
gitnexus/coverage/coverage-summary.json
|
||||
gitnexus/coverage/coverage-final.json
|
||||
gitnexus/test-results.json
|
||||
gitnexus-web/web-test-results.json
|
||||
retention-days: 5
|
||||
|
||||
cross-platform:
|
||||
name: ${{ matrix.os }}
|
||||
strategy:
|
||||
fail-fast: false
|
||||
matrix:
|
||||
# Ubuntu already covered by the coverage job above
|
||||
os: [windows-latest, macos-latest]
|
||||
runs-on: ${{ matrix.os }}
|
||||
timeout-minutes: 25
|
||||
steps:
|
||||
- uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4
|
||||
- uses: ./.github/actions/setup-gitnexus
|
||||
with:
|
||||
build: 'true'
|
||||
- run: npx vitest run
|
||||
working-directory: gitnexus
|
||||
+100
-10
@@ -1,20 +1,110 @@
|
||||
name: CI
|
||||
|
||||
on:
|
||||
push:
|
||||
branches: [main]
|
||||
paths-ignore: ['**.md', 'docs/**', 'LICENSE']
|
||||
pull_request:
|
||||
branches: [main]
|
||||
paths-ignore: ['**.md', 'docs/**', 'LICENSE']
|
||||
workflow_call:
|
||||
|
||||
concurrency:
|
||||
group: ci-${{ github.ref }}
|
||||
cancel-in-progress: true
|
||||
|
||||
# ── Reusable workflow orchestration ─────────────────────────────────
|
||||
# Each concern lives in its own workflow file for maintainability:
|
||||
# ci-quality.yml — typecheck (tsc --noEmit)
|
||||
# ci-tests.yml — unit + integration tests with coverage + cross-platform
|
||||
# ci-e2e.yml — E2E tests (only when gitnexus-web/ changes)
|
||||
#
|
||||
# Shared setup is DRY via .github/actions/setup-gitnexus composite action.
|
||||
|
||||
jobs:
|
||||
typecheck:
|
||||
quality:
|
||||
uses: ./.github/workflows/ci-quality.yml
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
tests:
|
||||
uses: ./.github/workflows/ci-tests.yml
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
e2e:
|
||||
uses: ./.github/workflows/ci-e2e.yml
|
||||
permissions:
|
||||
contents: read
|
||||
|
||||
# ── Save PR metadata for the reporting workflow ─────────────────
|
||||
# The ci-report.yml workflow (triggered by workflow_run) needs the
|
||||
# PR number and job results to post a comment. We save them as an
|
||||
# artifact because workflow_run context doesn't reliably carry PR
|
||||
# info for fork PRs.
|
||||
save-pr-meta:
|
||||
name: Save PR Metadata
|
||||
if: always() && github.event_name == 'pull_request'
|
||||
needs: [quality, tests, e2e]
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 5
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: actions/setup-node@v4
|
||||
- name: Write metadata
|
||||
shell: bash
|
||||
env:
|
||||
PR_NUMBER: ${{ github.event.number }}
|
||||
QUALITY: ${{ needs.quality.result }}
|
||||
TESTS: ${{ needs.tests.result }}
|
||||
E2E: ${{ needs.e2e.result }}
|
||||
run: |
|
||||
mkdir -p pr-meta
|
||||
echo "$PR_NUMBER" > pr-meta/pr_number
|
||||
echo "$QUALITY" > pr-meta/quality_result
|
||||
echo "$TESTS" > pr-meta/tests_result
|
||||
echo "$E2E" > pr-meta/e2e_result
|
||||
# TODO(post-merge): remove backward-compat copies once ci-report.yml
|
||||
# on main reads underscore names.
|
||||
# Backward-compat: ci-report.yml on main still reads hyphenated
|
||||
# names. workflow_run always executes from the default branch, so
|
||||
# the main-branch reader won't find the underscore variants until
|
||||
# this PR is merged. Write both until then.
|
||||
cp pr-meta/pr_number pr-meta/pr-number
|
||||
cp pr-meta/quality_result pr-meta/quality-result
|
||||
cp pr-meta/tests_result pr-meta/tests-result
|
||||
cp pr-meta/e2e_result pr-meta/e2e-result
|
||||
|
||||
- name: Upload PR metadata
|
||||
uses: actions/upload-artifact@ea165f8d65b6e75b540449e92b4886f43607fa02 # v4
|
||||
with:
|
||||
node-version: 20
|
||||
cache: npm
|
||||
cache-dependency-path: gitnexus/package-lock.json
|
||||
- run: npm ci
|
||||
working-directory: gitnexus
|
||||
- run: npx tsc --noEmit
|
||||
working-directory: gitnexus
|
||||
name: pr-meta
|
||||
path: pr-meta/
|
||||
retention-days: 1
|
||||
|
||||
# ── Unified CI gate ──────────────────────────────────────────────
|
||||
# Single required check for branch protection.
|
||||
ci-status:
|
||||
name: CI Gate
|
||||
needs: [quality, tests, e2e]
|
||||
if: always()
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 5
|
||||
steps:
|
||||
- name: Check all jobs passed
|
||||
shell: bash
|
||||
env:
|
||||
QUALITY: ${{ needs.quality.result }}
|
||||
TESTS: ${{ needs.tests.result }}
|
||||
E2E: ${{ needs.e2e.result }}
|
||||
run: |
|
||||
echo "Quality: $QUALITY"
|
||||
echo "Tests: $TESTS"
|
||||
echo "E2E: $E2E"
|
||||
if [[ "$QUALITY" != "success" ]] ||
|
||||
[[ "$TESTS" != "success" ]]; then
|
||||
echo "::error::Quality or test jobs failed"
|
||||
exit 1
|
||||
fi
|
||||
if [[ "$E2E" != "success" && "$E2E" != "skipped" ]]; then
|
||||
echo "::error::E2E job failed"
|
||||
exit 1
|
||||
fi
|
||||
|
||||
@@ -1,44 +1,95 @@
|
||||
name: Claude Code Review
|
||||
|
||||
# Uses pull_request_target so the workflow runs as defined on the default branch,
|
||||
# which allows access to secrets for posting review comments on fork PRs.
|
||||
# SECURITY: The checkout pins the fork's HEAD SHA (not the branch name) to
|
||||
# prevent TOCTOU races (force-push between trigger and checkout). The
|
||||
# claude-code-action sandboxes execution — it does NOT run arbitrary code
|
||||
# from the checked-out source.
|
||||
|
||||
on:
|
||||
pull_request:
|
||||
types: [opened, synchronize, ready_for_review, reopened]
|
||||
# Optional: Only run on specific file changes
|
||||
# paths:
|
||||
# - "src/**/*.ts"
|
||||
# - "src/**/*.tsx"
|
||||
# - "src/**/*.js"
|
||||
# - "src/**/*.jsx"
|
||||
# Trigger only when explicitly requested:
|
||||
# - Add the "claude-review" label to a PR, OR
|
||||
# - Comment "@claude" or "/review" on a PR
|
||||
pull_request_target:
|
||||
types: [labeled]
|
||||
issue_comment:
|
||||
types: [created]
|
||||
|
||||
# Serialize per-PR to avoid racing review comments.
|
||||
concurrency:
|
||||
group: claude-review-${{ github.event.issue.number || github.event.pull_request.number }}
|
||||
cancel-in-progress: false
|
||||
|
||||
jobs:
|
||||
claude-review:
|
||||
# Optional: Filter by PR author
|
||||
# if: |
|
||||
# github.event.pull_request.user.login == 'external-contributor' ||
|
||||
# github.event.pull_request.user.login == 'new-developer' ||
|
||||
# github.event.pull_request.author_association == 'FIRST_TIME_CONTRIBUTOR'
|
||||
|
||||
# Run only when:
|
||||
# 1. The "claude-review" label is added to a non-draft PR by a trusted contributor, OR
|
||||
# 2. A trusted contributor comments "@claude" or "/review" on a PR
|
||||
if: |
|
||||
(
|
||||
github.event_name == 'pull_request_target' &&
|
||||
github.event.label.name == 'claude-review' &&
|
||||
github.event.pull_request.draft == false &&
|
||||
(github.event.pull_request.author_association == 'OWNER' ||
|
||||
github.event.pull_request.author_association == 'MEMBER' ||
|
||||
github.event.pull_request.author_association == 'COLLABORATOR')
|
||||
) ||
|
||||
(
|
||||
github.event_name == 'issue_comment' &&
|
||||
github.event.issue.pull_request &&
|
||||
(contains(github.event.comment.body, '@claude') ||
|
||||
contains(github.event.comment.body, '/review')) &&
|
||||
(github.event.comment.author_association == 'OWNER' ||
|
||||
github.event.comment.author_association == 'MEMBER' ||
|
||||
github.event.comment.author_association == 'COLLABORATOR')
|
||||
)
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 30
|
||||
permissions:
|
||||
contents: read
|
||||
pull-requests: read
|
||||
pull-requests: write
|
||||
issues: read
|
||||
id-token: write
|
||||
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@v4
|
||||
# For issue_comment triggers, resolve the PR number, head SHA, and fork repo
|
||||
- name: Resolve PR context
|
||||
id: pr
|
||||
uses: actions/github-script@60a0d83039c74a4aee543508d2ffcb1c3799cdea # v7
|
||||
with:
|
||||
script: |
|
||||
let pr;
|
||||
if (context.eventName === 'issue_comment') {
|
||||
const resp = await github.rest.pulls.get({
|
||||
owner: context.repo.owner,
|
||||
repo: context.repo.repo,
|
||||
pull_number: context.payload.issue.number,
|
||||
});
|
||||
pr = resp.data;
|
||||
} else {
|
||||
pr = context.payload.pull_request;
|
||||
}
|
||||
core.setOutput('number', pr.number);
|
||||
core.setOutput('sha', pr.head.sha);
|
||||
core.setOutput('repo', pr.head.repo.full_name);
|
||||
core.setOutput('branch', pr.head.ref);
|
||||
|
||||
- name: Checkout PR head
|
||||
uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4
|
||||
with:
|
||||
repository: ${{ steps.pr.outputs.repo }}
|
||||
ref: ${{ steps.pr.outputs.sha }}
|
||||
fetch-depth: 1
|
||||
|
||||
- name: Run Claude Code Review
|
||||
id: claude-review
|
||||
uses: anthropics/claude-code-action@v1
|
||||
uses: anthropics/claude-code-action@9469d113c6afd29550c402740f22d1a97dd1209b # v1
|
||||
with:
|
||||
claude_code_oauth_token: ${{ secrets.CLAUDE_CODE_OAUTH_TOKEN }}
|
||||
github_token: ${{ secrets.GITHUB_TOKEN }}
|
||||
allowed_non_write_users: '*'
|
||||
show_full_output: true
|
||||
plugin_marketplaces: 'https://github.com/anthropics/claude-code.git'
|
||||
plugins: 'code-review@claude-code-plugins'
|
||||
prompt: '/code-review:code-review ${{ github.repository }}/pull/${{ github.event.pull_request.number }}'
|
||||
# See https://github.com/anthropics/claude-code-action/blob/main/docs/usage.md
|
||||
# or https://code.claude.com/docs/en/cli-reference for available options
|
||||
|
||||
prompt: '/code-review:code-review ${{ github.repository }}/pull/${{ steps.pr.outputs.number }}'
|
||||
|
||||
@@ -10,41 +10,101 @@ on:
|
||||
pull_request_review:
|
||||
types: [submitted]
|
||||
|
||||
# Serialize per-PR/issue to avoid racing comments.
|
||||
concurrency:
|
||||
group: claude-code-${{ github.event.issue.number || github.event.pull_request.number || github.event.issue.id }}
|
||||
cancel-in-progress: false
|
||||
|
||||
jobs:
|
||||
claude:
|
||||
if: |
|
||||
(github.event_name == 'issue_comment' && contains(github.event.comment.body, '@claude')) ||
|
||||
(github.event_name == 'pull_request_review_comment' && contains(github.event.comment.body, '@claude')) ||
|
||||
(github.event_name == 'pull_request_review' && contains(github.event.review.body, '@claude')) ||
|
||||
(github.event_name == 'issues' && (contains(github.event.issue.body, '@claude') || contains(github.event.issue.title, '@claude')))
|
||||
(
|
||||
github.event_name == 'issue_comment' &&
|
||||
contains(github.event.comment.body, '@claude') &&
|
||||
(github.event.comment.author_association == 'OWNER' ||
|
||||
github.event.comment.author_association == 'MEMBER' ||
|
||||
github.event.comment.author_association == 'COLLABORATOR')
|
||||
) ||
|
||||
(
|
||||
github.event_name == 'pull_request_review_comment' &&
|
||||
contains(github.event.comment.body, '@claude') &&
|
||||
(github.event.comment.author_association == 'OWNER' ||
|
||||
github.event.comment.author_association == 'MEMBER' ||
|
||||
github.event.comment.author_association == 'COLLABORATOR')
|
||||
) ||
|
||||
(
|
||||
github.event_name == 'pull_request_review' &&
|
||||
contains(github.event.review.body, '@claude') &&
|
||||
(github.event.review.author_association == 'OWNER' ||
|
||||
github.event.review.author_association == 'MEMBER' ||
|
||||
github.event.review.author_association == 'COLLABORATOR')
|
||||
) ||
|
||||
(
|
||||
github.event_name == 'issues' &&
|
||||
(contains(github.event.issue.body, '@claude') || contains(github.event.issue.title, '@claude')) &&
|
||||
(github.event.issue.author_association == 'OWNER' ||
|
||||
github.event.issue.author_association == 'MEMBER' ||
|
||||
github.event.issue.author_association == 'COLLABORATOR')
|
||||
)
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 30
|
||||
permissions:
|
||||
contents: read
|
||||
pull-requests: read
|
||||
issues: read
|
||||
pull-requests: write
|
||||
issues: write
|
||||
id-token: write
|
||||
actions: read # Required for Claude to read CI results on PRs
|
||||
actions: read # required for Claude to read CI results on PRs
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@v4
|
||||
# For PR-related triggers, resolve the fork repo so we can checkout correctly.
|
||||
- name: Resolve PR context
|
||||
id: pr
|
||||
uses: actions/github-script@60a0d83039c74a4aee543508d2ffcb1c3799cdea # v7
|
||||
with:
|
||||
script: |
|
||||
// Determine if this event is PR-related
|
||||
let prNumber = null;
|
||||
if (context.eventName === 'issue_comment' && context.payload.issue.pull_request) {
|
||||
prNumber = context.payload.issue.number;
|
||||
} else if (context.eventName === 'pull_request_review_comment') {
|
||||
prNumber = context.payload.pull_request.number;
|
||||
} else if (context.eventName === 'pull_request_review') {
|
||||
prNumber = context.payload.pull_request.number;
|
||||
}
|
||||
|
||||
if (!prNumber) {
|
||||
core.setOutput('is_pr', 'false');
|
||||
return;
|
||||
}
|
||||
|
||||
const resp = await github.rest.pulls.get({
|
||||
owner: context.repo.owner,
|
||||
repo: context.repo.repo,
|
||||
pull_number: prNumber,
|
||||
});
|
||||
const pr = resp.data;
|
||||
|
||||
core.setOutput('is_pr', 'true');
|
||||
core.setOutput('number', String(prNumber));
|
||||
core.setOutput('sha', pr.head.sha);
|
||||
core.setOutput('repo', pr.head.repo.full_name);
|
||||
core.setOutput('branch', pr.head.ref);
|
||||
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4
|
||||
with:
|
||||
repository: ${{ steps.pr.outputs.is_pr == 'true' && steps.pr.outputs.repo || github.repository }}
|
||||
ref: ${{ steps.pr.outputs.is_pr == 'true' && steps.pr.outputs.sha || '' }}
|
||||
fetch-depth: 1
|
||||
|
||||
- name: Run Claude Code
|
||||
id: claude
|
||||
uses: anthropics/claude-code-action@v1
|
||||
uses: anthropics/claude-code-action@9469d113c6afd29550c402740f22d1a97dd1209b # v1
|
||||
with:
|
||||
claude_code_oauth_token: ${{ secrets.CLAUDE_CODE_OAUTH_TOKEN }}
|
||||
github_token: ${{ secrets.GITHUB_TOKEN }}
|
||||
allowed_non_write_users: '*'
|
||||
show_full_output: true
|
||||
|
||||
# This is an optional setting that allows Claude to read CI results on PRs
|
||||
additional_permissions: |
|
||||
actions: read
|
||||
|
||||
# Optional: Give a custom prompt to Claude. If this is not specified, Claude will perform the instructions specified in the comment that tagged it.
|
||||
# prompt: 'Update the pull request description to include a summary of changes.'
|
||||
|
||||
# Optional: Add claude_args to customize behavior and configuration
|
||||
# See https://github.com/anthropics/claude-code-action/blob/main/docs/usage.md
|
||||
# or https://code.claude.com/docs/en/cli-reference for available options
|
||||
# claude_args: '--allowed-tools Bash(gh pr:*)'
|
||||
|
||||
|
||||
@@ -0,0 +1,94 @@
|
||||
name: PR Description Check
|
||||
|
||||
on:
|
||||
pull_request:
|
||||
types: [opened, edited, reopened]
|
||||
branches: [main]
|
||||
|
||||
permissions:
|
||||
pull-requests: write
|
||||
|
||||
concurrency:
|
||||
group: pr-desc-${{ github.event.pull_request.number }}
|
||||
cancel-in-progress: true
|
||||
|
||||
jobs:
|
||||
check-description:
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 5
|
||||
steps:
|
||||
- name: Check PR description quality
|
||||
uses: actions/github-script@ed597411d8f924073f98dfc5c65a23a2325f34cd # v8
|
||||
with:
|
||||
script: |
|
||||
const MIN_BODY_LENGTH = 50;
|
||||
const LABEL = 'needs-description';
|
||||
|
||||
const pr = context.payload.pull_request;
|
||||
const body = (pr.body || '').trim();
|
||||
const owner = context.repo.owner;
|
||||
const repo = context.repo.repo;
|
||||
const number = pr.number;
|
||||
|
||||
const hasLabel = pr.labels.some(l => l.name === LABEL);
|
||||
|
||||
if (body.length < MIN_BODY_LENGTH) {
|
||||
// Add label if not already present
|
||||
if (!hasLabel) {
|
||||
await github.rest.issues.addLabels({
|
||||
owner, repo, issue_number: number,
|
||||
labels: [LABEL],
|
||||
});
|
||||
}
|
||||
|
||||
// Post or update a comment
|
||||
const marker = '<!-- pr-desc-check -->';
|
||||
const message = [
|
||||
marker,
|
||||
`### PR description is too short`,
|
||||
'',
|
||||
`This PR's description is **${body.length}** characters, ` +
|
||||
`but the minimum is **${MIN_BODY_LENGTH}**.`,
|
||||
'',
|
||||
'Please update the PR description to explain:',
|
||||
'- **What** this PR changes',
|
||||
'- **Why** the change is needed',
|
||||
'',
|
||||
'Use the PR template as a guide. This check will re-run when you edit the description.',
|
||||
].join('\n');
|
||||
|
||||
// Find existing bot comment to update (avoid spam)
|
||||
const comments = await github.rest.issues.listComments({
|
||||
owner, repo, issue_number: number,
|
||||
});
|
||||
const existing = comments.data.find(c =>
|
||||
c.body && c.body.includes(marker)
|
||||
);
|
||||
|
||||
if (existing) {
|
||||
await github.rest.issues.updateComment({
|
||||
owner, repo, comment_id: existing.id,
|
||||
body: message,
|
||||
});
|
||||
} else {
|
||||
await github.rest.issues.createComment({
|
||||
owner, repo, issue_number: number,
|
||||
body: message,
|
||||
});
|
||||
}
|
||||
|
||||
core.setFailed(
|
||||
`PR description is ${body.length} chars (minimum: ${MIN_BODY_LENGTH})`
|
||||
);
|
||||
} else {
|
||||
// Description is acceptable — remove the label if present
|
||||
if (hasLabel) {
|
||||
await github.rest.issues.removeLabel({
|
||||
owner, repo, issue_number: number,
|
||||
name: LABEL,
|
||||
}).catch(() => {});
|
||||
// .catch: label may have been removed manually
|
||||
}
|
||||
|
||||
core.info(`PR description OK (${body.length} chars)`);
|
||||
}
|
||||
@@ -5,14 +5,26 @@ on:
|
||||
tags:
|
||||
- 'v*'
|
||||
|
||||
# No workflow-level permissions — scoped per job below.
|
||||
|
||||
jobs:
|
||||
publish:
|
||||
runs-on: ubuntu-latest
|
||||
ci:
|
||||
uses: ./.github/workflows/ci.yml
|
||||
permissions:
|
||||
contents: read
|
||||
actions: read
|
||||
pull-requests: write
|
||||
|
||||
publish:
|
||||
needs: ci
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 15
|
||||
permissions:
|
||||
contents: write
|
||||
id-token: write
|
||||
steps:
|
||||
- uses: actions/checkout@v4
|
||||
- uses: actions/setup-node@v4
|
||||
- uses: actions/checkout@34e114876b0b11c390a56381ad16ebd13914f8d5 # v4
|
||||
- uses: actions/setup-node@49933ea5288caeca8642d1e84afbd3f7d6820020 # v4
|
||||
with:
|
||||
node-version: 20
|
||||
registry-url: https://registry.npmjs.org
|
||||
@@ -20,9 +32,38 @@ jobs:
|
||||
cache-dependency-path: gitnexus/package-lock.json
|
||||
- run: npm ci
|
||||
working-directory: gitnexus
|
||||
- run: npx tsc --noEmit
|
||||
|
||||
- name: Verify version consistency
|
||||
shell: bash
|
||||
run: |
|
||||
TAG_VERSION="${GITHUB_REF#refs/tags/v}"
|
||||
if ! [[ "$TAG_VERSION" =~ ^[0-9]+\.[0-9]+\.[0-9]+(-[a-zA-Z0-9.]+)?$ ]]; then
|
||||
echo "::error::Tag does not follow semver: v$TAG_VERSION"
|
||||
exit 1
|
||||
fi
|
||||
PKG_VERSION=$(node -p "require('./package.json').version")
|
||||
if [ "$TAG_VERSION" != "$PKG_VERSION" ]; then
|
||||
echo "::error::Tag version (v$TAG_VERSION) does not match package.json version ($PKG_VERSION)"
|
||||
exit 1
|
||||
fi
|
||||
echo "Version verified: $PKG_VERSION"
|
||||
working-directory: gitnexus
|
||||
- run: npm publish
|
||||
|
||||
- name: Build
|
||||
run: npm run build
|
||||
working-directory: gitnexus
|
||||
|
||||
- name: Dry-run publish
|
||||
run: npm publish --dry-run
|
||||
working-directory: gitnexus
|
||||
|
||||
- name: Publish to npm
|
||||
run: npm publish --provenance --access public
|
||||
working-directory: gitnexus
|
||||
env:
|
||||
NODE_AUTH_TOKEN: ${{ secrets.NPM_TOKEN }}
|
||||
|
||||
- name: Create GitHub Release
|
||||
uses: softprops/action-gh-release@a06a81a03ee405af7f2048a818ed3f03bbf83c7b # v2
|
||||
with:
|
||||
generate_release_notes: true
|
||||
|
||||
@@ -0,0 +1,103 @@
|
||||
name: Triage Sweep
|
||||
|
||||
on:
|
||||
workflow_dispatch:
|
||||
inputs:
|
||||
iqr_multiplier:
|
||||
description: >-
|
||||
IQR multiplier for outlier cutoff.
|
||||
cutoff = Q75 + multiplier * IQR.
|
||||
Higher = fewer outliers flagged.
|
||||
type: number
|
||||
default: 3.0
|
||||
max_outlier_pct:
|
||||
description: >-
|
||||
Maximum fraction of items that can be flagged as outliers (0-1).
|
||||
Hard cap to prevent over-flagging.
|
||||
type: number
|
||||
default: 0.05
|
||||
contamination:
|
||||
description: >-
|
||||
Expected fraction of outliers in the data (0-0.5).
|
||||
Controls how aggressively EllipticEnvelope downweights extremes.
|
||||
type: number
|
||||
default: 0.1
|
||||
cosine_threshold:
|
||||
description: >-
|
||||
Cosine similarity threshold for duplicate detection.
|
||||
Pairs with similarity above this are flagged as potential duplicates.
|
||||
Higher = only very similar pairs flagged.
|
||||
type: number
|
||||
default: 0.92
|
||||
max_items:
|
||||
description: >-
|
||||
Maximum number of open issues + PRs to process.
|
||||
Hard cap to prevent runaway costs on very large repos.
|
||||
type: number
|
||||
default: 500
|
||||
dry_run:
|
||||
description: >-
|
||||
Check this to only log results to the workflow summary.
|
||||
Uncheck to create a GitHub issue with the report and apply labels.
|
||||
type: boolean
|
||||
default: false
|
||||
|
||||
permissions:
|
||||
contents: read
|
||||
issues: write
|
||||
pull-requests: write
|
||||
|
||||
concurrency:
|
||||
group: triage-sweep
|
||||
cancel-in-progress: true
|
||||
|
||||
jobs:
|
||||
sweep:
|
||||
runs-on: ubuntu-latest
|
||||
timeout-minutes: 30
|
||||
steps:
|
||||
- name: Checkout repository
|
||||
uses: actions/checkout@de0fac2e4500dabe0009e67214ff5f5447ce83dd # v6
|
||||
with:
|
||||
sparse-checkout: .github/scripts/triage
|
||||
sparse-checkout-cone-mode: false
|
||||
fetch-depth: 1
|
||||
|
||||
- name: Set up Python
|
||||
uses: actions/setup-python@a309ff8b426b58ec0e2a45f0f869d46889d02405 # v6
|
||||
with:
|
||||
python-version: '3.12'
|
||||
cache: pip
|
||||
cache-dependency-path: .github/scripts/triage/requirements.txt
|
||||
|
||||
- name: Install dependencies
|
||||
run: pip install -r .github/scripts/triage/requirements.txt
|
||||
|
||||
- name: Cache FastEmbed model weights
|
||||
uses: actions/cache@668228422ae6a00e4ad889ee87cd7109ec5666a7 # v5
|
||||
with:
|
||||
path: ${{ github.workspace }}/.fastembed_cache
|
||||
key: fastembed-bge-small-en-v1.5
|
||||
|
||||
- name: Run triage sweep
|
||||
id: sweep
|
||||
env:
|
||||
GITHUB_TOKEN: ${{ secrets.GITHUB_TOKEN }}
|
||||
GITHUB_REPOSITORY: ${{ github.repository }}
|
||||
FASTEMBED_CACHE_PATH: ${{ github.workspace }}/.fastembed_cache
|
||||
INPUT_IQR_MULTIPLIER: ${{ inputs.iqr_multiplier }}
|
||||
INPUT_MAX_OUTLIER_PCT: ${{ inputs.max_outlier_pct }}
|
||||
INPUT_CONTAMINATION: ${{ inputs.contamination }}
|
||||
INPUT_COSINE_THRESHOLD: ${{ inputs.cosine_threshold }}
|
||||
INPUT_MAX_ITEMS: ${{ inputs.max_items }}
|
||||
INPUT_DRY_RUN: ${{ inputs.dry_run }}
|
||||
run: python .github/scripts/triage/sweep.py
|
||||
|
||||
- name: Post summary
|
||||
if: always()
|
||||
run: |
|
||||
if [ -f /tmp/triage-report.md ]; then
|
||||
cat /tmp/triage-report.md >> "$GITHUB_STEP_SUMMARY"
|
||||
else
|
||||
echo "No report generated." >> "$GITHUB_STEP_SUMMARY"
|
||||
fi
|
||||
+38
@@ -33,6 +33,8 @@ coverage/
|
||||
|
||||
# Misc
|
||||
*.local
|
||||
HANDOFF.md
|
||||
HANDOFF*.md
|
||||
|
||||
.vercel
|
||||
|
||||
@@ -48,11 +50,47 @@ coverage/
|
||||
# Claude Code worktrees
|
||||
.claude/worktrees/
|
||||
|
||||
# Claude code skills
|
||||
.claude/skills/generated/
|
||||
|
||||
# Assets (screenshots, images)
|
||||
assets/
|
||||
|
||||
# Generated files (should not be indexed)
|
||||
repomix-output*
|
||||
|
||||
# Playwright artifacts
|
||||
gitnexus-web/playwright-report/
|
||||
gitnexus-web/test-results/
|
||||
|
||||
# Python test artifacts
|
||||
eval/.coverage
|
||||
eval/.hypothesis/
|
||||
|
||||
# Design docs (local only)
|
||||
docs/plans/
|
||||
|
||||
gitnexus/test/fixtures/mini-repo/*.md
|
||||
gitnexus/test/fixtures/mini-repo/.claude
|
||||
gitnexus/test/fixtures/mini-repo/.gitignore
|
||||
|
||||
# Ignore csharp generated obj and bin folders
|
||||
gitnexus/test/fixtures/lang-resolution/**/obj
|
||||
gitnexus/test/fixtures/lang-resolution/**/bin
|
||||
GitNexus.sln
|
||||
# Git worktrees
|
||||
.worktrees/
|
||||
|
||||
/github/scripts/triage/__pycache__/
|
||||
|
||||
.claude-flow/
|
||||
|
||||
.claude/agents/
|
||||
.claude/commands/
|
||||
.claude/helpers
|
||||
.claude/skills/
|
||||
!.claude/skills/gitnexus/
|
||||
|
||||
.history/
|
||||
|
||||
.swarm/
|
||||
@@ -0,0 +1,33 @@
|
||||
import { defineConfig } from 'vitest/config';
|
||||
|
||||
export default defineConfig({
|
||||
test: {
|
||||
globalSetup: ['test/global-setup.ts'],
|
||||
include: ['test/**/*.test.ts'],
|
||||
testTimeout: 30000,
|
||||
hookTimeout: 120000,
|
||||
pool: 'forks',
|
||||
globals: true,
|
||||
setupFiles: ['test/setup.ts'],
|
||||
teardownTimeout: 3000,
|
||||
dangerouslyIgnoreUnhandledErrors: true, // LadybugDB N-API destructor segfaults on fork exit — not a test failure
|
||||
coverage: {
|
||||
provider: 'v8',
|
||||
include: ['src/**/*.ts'],
|
||||
exclude: [
|
||||
'src/cli/index.ts', // CLI entry point (commander wiring)
|
||||
'src/server/**', // HTTP server (requires network)
|
||||
'src/core/wiki/**', // Wiki generation (requires LLM)
|
||||
],
|
||||
// Auto-ratchet: vitest bumps thresholds when coverage exceeds them.
|
||||
// CI will fail if a PR drops below these floors.
|
||||
thresholds: {
|
||||
statements: 26,
|
||||
branches: 23,
|
||||
functions: 28,
|
||||
lines: 27,
|
||||
autoUpdate: true,
|
||||
},
|
||||
},
|
||||
},
|
||||
});
|
||||
Executable
+33
@@ -0,0 +1,33 @@
|
||||
#!/usr/bin/env bash
|
||||
# Pre-commit hook (husky): typecheck + unit tests for both packages.
|
||||
# Mirrors CI checks from ci-quality.yml and ci-tests.yml.
|
||||
# Skip with: git commit --no-verify
|
||||
#
|
||||
# CI coverage:
|
||||
# quality / typecheck → tsc --noEmit in gitnexus/
|
||||
# quality / typecheck-web → tsc -b --noEmit in gitnexus-web/
|
||||
# tests / ubuntu+coverage → vitest run in gitnexus/ (all projects)
|
||||
# e2e / chromium → playwright (requires servers — skipped)
|
||||
|
||||
ROOT="$(git rev-parse --show-toplevel)"
|
||||
|
||||
WEB_CHANGED=$(git diff --cached --name-only -- 'gitnexus-web/' | head -1)
|
||||
CLI_CHANGED=$(git diff --cached --name-only -- 'gitnexus/' | head -1)
|
||||
|
||||
if [ -n "$WEB_CHANGED" ]; then
|
||||
echo "pre-commit: typechecking gitnexus-web (tsc -b)..."
|
||||
cd "$ROOT/gitnexus-web" && npx tsc -b --noEmit
|
||||
|
||||
echo "pre-commit: running gitnexus-web unit tests..."
|
||||
npx vitest run --reporter=dot
|
||||
fi
|
||||
|
||||
if [ -n "$CLI_CHANGED" ]; then
|
||||
echo "pre-commit: typechecking gitnexus..."
|
||||
cd "$ROOT/gitnexus" && npx tsc --noEmit
|
||||
|
||||
echo "pre-commit: running gitnexus unit tests (default project)..."
|
||||
npx vitest run --project default --reporter=dot
|
||||
fi
|
||||
|
||||
echo "pre-commit: all checks passed"
|
||||
@@ -1,21 +1,93 @@
|
||||
<!-- gitnexus:start -->
|
||||
# GitNexus MCP
|
||||
# GitNexus — Code Intelligence
|
||||
|
||||
This project is indexed by GitNexus as **GitnexusV2** (1348 symbols, 3469 relationships, 104 execution flows).
|
||||
This project is indexed by GitNexus as **GitNexus** (2487 symbols, 6056 relationships, 188 execution flows). Use the GitNexus MCP tools to understand code, assess impact, and navigate safely.
|
||||
|
||||
GitNexus provides a knowledge graph over this codebase — call chains, blast radius, execution flows, and semantic search.
|
||||
> If any GitNexus tool warns the index is stale, run `npx gitnexus analyze` in terminal first.
|
||||
|
||||
## Always Start Here
|
||||
## Always Do
|
||||
|
||||
For any task involving code understanding, debugging, impact analysis, or refactoring, you must:
|
||||
- **MUST run impact analysis before editing any symbol.** Before modifying a function, class, or method, run `gitnexus_impact({target: "symbolName", direction: "upstream"})` and report the blast radius (direct callers, affected processes, risk level) to the user.
|
||||
- **MUST run `gitnexus_detect_changes()` before committing** to verify your changes only affect expected symbols and execution flows.
|
||||
- **MUST warn the user** if impact analysis returns HIGH or CRITICAL risk before proceeding with edits.
|
||||
- When exploring unfamiliar code, use `gitnexus_query({query: "concept"})` to find execution flows instead of grepping. It returns process-grouped results ranked by relevance.
|
||||
- When you need full context on a specific symbol — callers, callees, which execution flows it participates in — use `gitnexus_context({name: "symbolName"})`.
|
||||
|
||||
1. **Read `gitnexus://repo/{name}/context`** — codebase overview + check index freshness
|
||||
2. **Match your task to a skill below** and **read that skill file**
|
||||
3. **Follow the skill's workflow and checklist**
|
||||
## When Debugging
|
||||
|
||||
> If step 1 warns the index is stale, run `npx gitnexus analyze` in the terminal first.
|
||||
1. `gitnexus_query({query: "<error or symptom>"})` — find execution flows related to the issue
|
||||
2. `gitnexus_context({name: "<suspect function>"})` — see all callers, callees, and process participation
|
||||
3. `READ gitnexus://repo/GitNexus/process/{processName}` — trace the full execution flow step by step
|
||||
4. For regressions: `gitnexus_detect_changes({scope: "compare", base_ref: "main"})` — see what your branch changed
|
||||
|
||||
## Skills
|
||||
## When Refactoring
|
||||
|
||||
- **Renaming**: MUST use `gitnexus_rename({symbol_name: "old", new_name: "new", dry_run: true})` first. Review the preview — graph edits are safe, text_search edits need manual review. Then run with `dry_run: false`.
|
||||
- **Extracting/Splitting**: MUST run `gitnexus_context({name: "target"})` to see all incoming/outgoing refs, then `gitnexus_impact({target: "target", direction: "upstream"})` to find all external callers before moving code.
|
||||
- After any refactor: run `gitnexus_detect_changes({scope: "all"})` to verify only expected files changed.
|
||||
|
||||
## Never Do
|
||||
|
||||
- NEVER edit a function, class, or method without first running `gitnexus_impact` on it.
|
||||
- NEVER ignore HIGH or CRITICAL risk warnings from impact analysis.
|
||||
- NEVER rename symbols with find-and-replace — use `gitnexus_rename` which understands the call graph.
|
||||
- NEVER commit changes without running `gitnexus_detect_changes()` to check affected scope.
|
||||
|
||||
## Tools Quick Reference
|
||||
|
||||
| Tool | When to use | Command |
|
||||
|------|-------------|---------|
|
||||
| `query` | Find code by concept | `gitnexus_query({query: "auth validation"})` |
|
||||
| `context` | 360-degree view of one symbol | `gitnexus_context({name: "validateUser"})` |
|
||||
| `impact` | Blast radius before editing | `gitnexus_impact({target: "X", direction: "upstream"})` |
|
||||
| `detect_changes` | Pre-commit scope check | `gitnexus_detect_changes({scope: "staged"})` |
|
||||
| `rename` | Safe multi-file rename | `gitnexus_rename({symbol_name: "old", new_name: "new", dry_run: true})` |
|
||||
| `cypher` | Custom graph queries | `gitnexus_cypher({query: "MATCH ..."})` |
|
||||
|
||||
## Impact Risk Levels
|
||||
|
||||
| Depth | Meaning | Action |
|
||||
|-------|---------|--------|
|
||||
| d=1 | WILL BREAK — direct callers/importers | MUST update these |
|
||||
| d=2 | LIKELY AFFECTED — indirect deps | Should test |
|
||||
| d=3 | MAY NEED TESTING — transitive | Test if critical path |
|
||||
|
||||
## Resources
|
||||
|
||||
| Resource | Use for |
|
||||
|----------|---------|
|
||||
| `gitnexus://repo/GitNexus/context` | Codebase overview, check index freshness |
|
||||
| `gitnexus://repo/GitNexus/clusters` | All functional areas |
|
||||
| `gitnexus://repo/GitNexus/processes` | All execution flows |
|
||||
| `gitnexus://repo/GitNexus/process/{name}` | Step-by-step execution trace |
|
||||
|
||||
## Self-Check Before Finishing
|
||||
|
||||
Before completing any code modification task, verify:
|
||||
1. `gitnexus_impact` was run for all modified symbols
|
||||
2. No HIGH/CRITICAL risk warnings were ignored
|
||||
3. `gitnexus_detect_changes()` confirms changes match expected scope
|
||||
4. All d=1 (WILL BREAK) dependents were updated
|
||||
|
||||
## Keeping the Index Fresh
|
||||
|
||||
After committing code changes, the GitNexus index becomes stale. Re-run analyze to update it:
|
||||
|
||||
```bash
|
||||
npx gitnexus analyze
|
||||
```
|
||||
|
||||
If the index previously included embeddings, preserve them by adding `--embeddings`:
|
||||
|
||||
```bash
|
||||
npx gitnexus analyze --embeddings
|
||||
```
|
||||
|
||||
To check whether embeddings exist, inspect `.gitnexus/meta.json` — the `stats.embeddings` field shows the count (0 means no embeddings). **Running analyze without `--embeddings` will delete any previously generated embeddings.**
|
||||
|
||||
> Claude Code users: A PostToolUse hook handles this automatically after `git commit` and `git merge`.
|
||||
|
||||
## CLI
|
||||
|
||||
| Task | Read this skill file |
|
||||
|------|---------------------|
|
||||
@@ -23,40 +95,7 @@ For any task involving code understanding, debugging, impact analysis, or refact
|
||||
| Blast radius / "What breaks if I change X?" | `.claude/skills/gitnexus/gitnexus-impact-analysis/SKILL.md` |
|
||||
| Trace bugs / "Why is X failing?" | `.claude/skills/gitnexus/gitnexus-debugging/SKILL.md` |
|
||||
| Rename / extract / split / refactor | `.claude/skills/gitnexus/gitnexus-refactoring/SKILL.md` |
|
||||
| Tools, resources, schema reference | `.claude/skills/gitnexus/gitnexus-guide/SKILL.md` |
|
||||
| Index, status, clean, wiki CLI commands | `.claude/skills/gitnexus/gitnexus-cli/SKILL.md` |
|
||||
|
||||
## Tools Reference
|
||||
|
||||
| Tool | What it gives you |
|
||||
|------|-------------------|
|
||||
| `query` | Process-grouped code intelligence — execution flows related to a concept |
|
||||
| `context` | 360-degree symbol view — categorized refs, processes it participates in |
|
||||
| `impact` | Symbol blast radius — what breaks at depth 1/2/3 with confidence |
|
||||
| `detect_changes` | Git-diff impact — what do your current changes affect |
|
||||
| `rename` | Multi-file coordinated rename with confidence-tagged edits |
|
||||
| `cypher` | Raw graph queries (read `gitnexus://repo/{name}/schema` first) |
|
||||
| `list_repos` | Discover indexed repos |
|
||||
|
||||
## Resources Reference
|
||||
|
||||
Lightweight reads (~100-500 tokens) for navigation:
|
||||
|
||||
| Resource | Content |
|
||||
|----------|---------|
|
||||
| `gitnexus://repo/{name}/context` | Stats, staleness check |
|
||||
| `gitnexus://repo/{name}/clusters` | All functional areas with cohesion scores |
|
||||
| `gitnexus://repo/{name}/cluster/{clusterName}` | Area members |
|
||||
| `gitnexus://repo/{name}/processes` | All execution flows |
|
||||
| `gitnexus://repo/{name}/process/{processName}` | Step-by-step trace |
|
||||
| `gitnexus://repo/{name}/schema` | Graph schema for Cypher |
|
||||
|
||||
## Graph Schema
|
||||
|
||||
**Nodes:** File, Function, Class, Interface, Method, Community, Process
|
||||
**Edges (via CodeRelation.type):** CALLS, IMPORTS, EXTENDS, IMPLEMENTS, DEFINES, MEMBER_OF, STEP_IN_PROCESS
|
||||
|
||||
```cypher
|
||||
MATCH (caller)-[:CodeRelation {type: 'CALLS'}]->(f:Function {name: "myFunc"})
|
||||
RETURN caller.name, caller.filePath
|
||||
```
|
||||
|
||||
<!-- gitnexus:end -->
|
||||
<!-- gitnexus:end -->
|
||||
|
||||
+109
@@ -0,0 +1,109 @@
|
||||
# Changelog
|
||||
|
||||
All notable changes to GitNexus will be documented in this file.
|
||||
|
||||
## [Unreleased]
|
||||
|
||||
### Changed
|
||||
- Migrated from KuzuDB to LadybugDB v0.15 (`@ladybugdb/core`, `@ladybugdb/wasm-core`)
|
||||
- Renamed all internal paths from `kuzu` to `lbug` (storage: `.gitnexus/kuzu` → `.gitnexus/lbug`)
|
||||
- Added automatic cleanup of stale KuzuDB index files
|
||||
- LadybugDB v0.15 requires explicit VECTOR extension loading for semantic search
|
||||
|
||||
## [1.4.0] - 2026-03-13
|
||||
|
||||
### Added
|
||||
|
||||
- **Language-aware symbol resolution engine** with 3-tier resolver: exact FQN → scope-walk → guarded fuzzy fallback that refuses ambiguous matches (#238) — @magyargergo
|
||||
- **Method Resolution Order (MRO)** with 5 language-specific strategies: C++ leftmost-base, C#/Java class-over-interface, Python C3 linearization, Rust qualified syntax, default BFS (#238) — @magyargergo
|
||||
- **Constructor & struct literal resolution** across all languages — `new Foo()`, `User{...}`, C# primary constructors, target-typed new (#238) — @magyargergo
|
||||
- **Receiver-constrained resolution** using per-file TypeEnv — disambiguates `user.save()` vs `repo.save()` via `ownerId` matching (#238) — @magyargergo
|
||||
- **Heritage & ownership edges** — HAS_METHOD, OVERRIDES, Go struct embedding, Swift extension heritage, method signatures (`parameterCount`, `returnType`) (#238) — @magyargergo
|
||||
- **Language-specific resolver directory** (`resolvers/`) — extracted JVM, Go, C#, PHP, Rust resolvers from monolithic import-processor (#238) — @magyargergo
|
||||
- **Type extractor directory** (`type-extractors/`) — per-language type binding extraction with `Record<SupportedLanguages, Handler>` + `satisfies` dispatch (#238) — @magyargergo
|
||||
- **Export detection dispatch table** — compile-time exhaustive `Record` + `satisfies` pattern replacing switch/if chains (#238) — @magyargergo
|
||||
- **Language config module** (`language-config.ts`) — centralized tsconfig, go.mod, composer.json, .csproj, Swift package config loaders (#238) — @magyargergo
|
||||
- **Optional skill generation** via `npx gitnexus analyze --skills` — generates AI agent skills from KuzuDB knowledge graph (#171) — @zander-raycraft
|
||||
- **First-class C# support** — sibling-based modifier scanning, record/delegate/property/field/event declaration types (#163, #170, #178 via #237) — @Alice523, @benny-yamagata, @jnMetaCode
|
||||
- **C/C++ support fixes** — `.h` → C++ mapping, static-linkage export detection, qualified/parenthesized declarators, 48 entry point patterns (#163, #227 via #237) — @Alice523, @bitgineer
|
||||
- **Rust support fixes** — sibling-based `visibility_modifier` scanning for `pub` detection (#227 via #237) — @bitgineer
|
||||
- **Adaptive tree-sitter buffer sizing** — `Math.min(Math.max(contentLength * 2, 512KB), 32MB)` (#216 via #237) — @JasonOA888
|
||||
- **Call expression matching** in tree-sitter queries (#234 via #237) — @ex-nihilo-jg
|
||||
- **DeepSeek model configurations** (#217) — @JasonOA888
|
||||
- 282+ new unit tests, 178 integration resolver tests across 9 languages, 53 test files, 1146 total tests passing
|
||||
|
||||
### Fixed
|
||||
|
||||
- Skip unavailable native Swift parsers in sequential ingestion (#188) — @Gujiassh
|
||||
- Heritage heuristic language-gated — no longer applies class/interface rules to wrong languages (#238) — @magyargergo
|
||||
- C# `base_list` distinguishes EXTENDS vs IMPLEMENTS via symbol table + `I[A-Z]` heuristic (#238) — @magyargergo
|
||||
- Go `qualified_type` (`models.User`) correctly unwrapped in TypeEnv (#238) — @magyargergo
|
||||
- Global tier no longer blocks resolution when kind/arity filtering can narrow to 1 candidate (#238) — @magyargergo
|
||||
|
||||
### Changed
|
||||
|
||||
- `import-processor.ts` reduced from 1412 → 711 lines (50% reduction) via resolver and config extraction (#238) — @magyargergo
|
||||
- `type-env.ts` reduced from 635 → ~125 lines via type-extractor extraction (#238) — @magyargergo
|
||||
- CI/CD workflows hardened with security fixes and fork PR support (#222, #225) — @magyargergo
|
||||
|
||||
## [1.3.11] - 2026-03-08
|
||||
|
||||
### Security
|
||||
|
||||
- Fix FTS Cypher injection by escaping backslashes in search queries (#209) — @magyargergo
|
||||
|
||||
### Added
|
||||
|
||||
- Auto-reindex hook that runs `gitnexus analyze` after commits and merges, with automatic embeddings preservation (#205) — @L1nusB
|
||||
- 968 integration tests (up from ~840) covering unhappy paths across search, enrichment, CLI, pipeline, worker pool, and KuzuDB (#209) — @magyargergo
|
||||
- Coverage auto-ratcheting so thresholds bump automatically on CI (#209) — @magyargergo
|
||||
- Rich CI PR report with coverage bars, test counts, and threshold tracking (#209) — @magyargergo
|
||||
- Modular CI workflow architecture with separate unit-test, integration-test, and orchestrator jobs (#209) — @magyargergo
|
||||
|
||||
### Fixed
|
||||
|
||||
- KuzuDB native addon crashes on Linux/macOS by running integration tests in isolated vitest processes with `--pool=forks` (#209) — @magyargergo
|
||||
- Worker pool `MODULE_NOT_FOUND` crash when script path is invalid (#209) — @magyargergo
|
||||
|
||||
### Changed
|
||||
|
||||
- Added macOS to the cross-platform CI test matrix (#208) — @magyargergo
|
||||
|
||||
## [1.3.10] - 2026-03-07
|
||||
|
||||
### Security
|
||||
|
||||
- **MCP transport buffer cap**: Added 10 MB `MAX_BUFFER_SIZE` limit to prevent out-of-memory attacks via oversized `Content-Length` headers or unbounded newline-delimited input
|
||||
- **Content-Length validation**: Reject `Content-Length` values exceeding the buffer cap before allocating memory
|
||||
- **Stack overflow prevention**: Replaced recursive `readNewlineMessage` with iterative loop to prevent stack overflow from consecutive empty lines
|
||||
- **Ambiguous prefix hardening**: Tightened `looksLikeContentLength` to require 14+ bytes before matching, preventing false framing detection on short input
|
||||
- **Closed transport guard**: `send()` now rejects with a clear error when called after `close()`, with proper write-error propagation
|
||||
|
||||
### Added
|
||||
|
||||
- **Dual-framing MCP transport** (`CompatibleStdioServerTransport`): Auto-detects Content-Length (Codex/OpenCode) and newline-delimited JSON (Cursor/Claude Code) framing on the first message, responds in the same format (#207)
|
||||
- **Lazy CLI module loading**: All CLI subcommands now use `createLazyAction()` to defer heavy imports (tree-sitter, ONNX, KuzuDB) until invocation, significantly improving `gitnexus mcp` startup time (#207)
|
||||
- **Type-safe lazy actions**: `createLazyAction` uses constrained generics to validate export names against module types at compile time
|
||||
- **Regression test suite**: 13 unit tests covering transport framing, security hardening, buffer limits, and lazy action loading
|
||||
|
||||
### Fixed
|
||||
|
||||
- **CALLS edge sourceId alignment**: `findEnclosingFunctionId` now generates IDs with `:startLine` suffix matching node creation format, fixing process detector finding 0 entry points (#194)
|
||||
- **LRU cache zero maxSize crash**: Guard `createASTCache` against `maxSize=0` when repos have no parseable files (#144)
|
||||
|
||||
### Changed
|
||||
|
||||
- Transport constructor accepts `NodeJS.ReadableStream` / `NodeJS.WritableStream` (widened from concrete `ReadStream`/`WriteStream`)
|
||||
- `processReadBuffer` simplified to break on first error instead of stale-buffer retry loop
|
||||
|
||||
## [1.3.9] - 2026-03-06
|
||||
|
||||
### Fixed
|
||||
|
||||
- Aligned CALLS edge sourceId with node ID format in parse worker (#194)
|
||||
|
||||
## [1.3.8] - 2026-03-05
|
||||
|
||||
### Fixed
|
||||
|
||||
- Force-exit after analyze to prevent KuzuDB native cleanup hang (#192)
|
||||
@@ -1,21 +1,93 @@
|
||||
<!-- gitnexus:start -->
|
||||
# GitNexus MCP
|
||||
# GitNexus — Code Intelligence
|
||||
|
||||
This project is indexed by GitNexus as **GitnexusV2** (1348 symbols, 3469 relationships, 104 execution flows).
|
||||
This project is indexed by GitNexus as **GitNexus** (2487 symbols, 6056 relationships, 188 execution flows). Use the GitNexus MCP tools to understand code, assess impact, and navigate safely.
|
||||
|
||||
GitNexus provides a knowledge graph over this codebase — call chains, blast radius, execution flows, and semantic search.
|
||||
> If any GitNexus tool warns the index is stale, run `npx gitnexus analyze` in terminal first.
|
||||
|
||||
## Always Start Here
|
||||
## Always Do
|
||||
|
||||
For any task involving code understanding, debugging, impact analysis, or refactoring, you must:
|
||||
- **MUST run impact analysis before editing any symbol.** Before modifying a function, class, or method, run `gitnexus_impact({target: "symbolName", direction: "upstream"})` and report the blast radius (direct callers, affected processes, risk level) to the user.
|
||||
- **MUST run `gitnexus_detect_changes()` before committing** to verify your changes only affect expected symbols and execution flows.
|
||||
- **MUST warn the user** if impact analysis returns HIGH or CRITICAL risk before proceeding with edits.
|
||||
- When exploring unfamiliar code, use `gitnexus_query({query: "concept"})` to find execution flows instead of grepping. It returns process-grouped results ranked by relevance.
|
||||
- When you need full context on a specific symbol — callers, callees, which execution flows it participates in — use `gitnexus_context({name: "symbolName"})`.
|
||||
|
||||
1. **Read `gitnexus://repo/{name}/context`** — codebase overview + check index freshness
|
||||
2. **Match your task to a skill below** and **read that skill file**
|
||||
3. **Follow the skill's workflow and checklist**
|
||||
## When Debugging
|
||||
|
||||
> If step 1 warns the index is stale, run `npx gitnexus analyze` in the terminal first.
|
||||
1. `gitnexus_query({query: "<error or symptom>"})` — find execution flows related to the issue
|
||||
2. `gitnexus_context({name: "<suspect function>"})` — see all callers, callees, and process participation
|
||||
3. `READ gitnexus://repo/GitNexus/process/{processName}` — trace the full execution flow step by step
|
||||
4. For regressions: `gitnexus_detect_changes({scope: "compare", base_ref: "main"})` — see what your branch changed
|
||||
|
||||
## Skills
|
||||
## When Refactoring
|
||||
|
||||
- **Renaming**: MUST use `gitnexus_rename({symbol_name: "old", new_name: "new", dry_run: true})` first. Review the preview — graph edits are safe, text_search edits need manual review. Then run with `dry_run: false`.
|
||||
- **Extracting/Splitting**: MUST run `gitnexus_context({name: "target"})` to see all incoming/outgoing refs, then `gitnexus_impact({target: "target", direction: "upstream"})` to find all external callers before moving code.
|
||||
- After any refactor: run `gitnexus_detect_changes({scope: "all"})` to verify only expected files changed.
|
||||
|
||||
## Never Do
|
||||
|
||||
- NEVER edit a function, class, or method without first running `gitnexus_impact` on it.
|
||||
- NEVER ignore HIGH or CRITICAL risk warnings from impact analysis.
|
||||
- NEVER rename symbols with find-and-replace — use `gitnexus_rename` which understands the call graph.
|
||||
- NEVER commit changes without running `gitnexus_detect_changes()` to check affected scope.
|
||||
|
||||
## Tools Quick Reference
|
||||
|
||||
| Tool | When to use | Command |
|
||||
|------|-------------|---------|
|
||||
| `query` | Find code by concept | `gitnexus_query({query: "auth validation"})` |
|
||||
| `context` | 360-degree view of one symbol | `gitnexus_context({name: "validateUser"})` |
|
||||
| `impact` | Blast radius before editing | `gitnexus_impact({target: "X", direction: "upstream"})` |
|
||||
| `detect_changes` | Pre-commit scope check | `gitnexus_detect_changes({scope: "staged"})` |
|
||||
| `rename` | Safe multi-file rename | `gitnexus_rename({symbol_name: "old", new_name: "new", dry_run: true})` |
|
||||
| `cypher` | Custom graph queries | `gitnexus_cypher({query: "MATCH ..."})` |
|
||||
|
||||
## Impact Risk Levels
|
||||
|
||||
| Depth | Meaning | Action |
|
||||
|-------|---------|--------|
|
||||
| d=1 | WILL BREAK — direct callers/importers | MUST update these |
|
||||
| d=2 | LIKELY AFFECTED — indirect deps | Should test |
|
||||
| d=3 | MAY NEED TESTING — transitive | Test if critical path |
|
||||
|
||||
## Resources
|
||||
|
||||
| Resource | Use for |
|
||||
|----------|---------|
|
||||
| `gitnexus://repo/GitNexus/context` | Codebase overview, check index freshness |
|
||||
| `gitnexus://repo/GitNexus/clusters` | All functional areas |
|
||||
| `gitnexus://repo/GitNexus/processes` | All execution flows |
|
||||
| `gitnexus://repo/GitNexus/process/{name}` | Step-by-step execution trace |
|
||||
|
||||
## Self-Check Before Finishing
|
||||
|
||||
Before completing any code modification task, verify:
|
||||
1. `gitnexus_impact` was run for all modified symbols
|
||||
2. No HIGH/CRITICAL risk warnings were ignored
|
||||
3. `gitnexus_detect_changes()` confirms changes match expected scope
|
||||
4. All d=1 (WILL BREAK) dependents were updated
|
||||
|
||||
## Keeping the Index Fresh
|
||||
|
||||
After committing code changes, the GitNexus index becomes stale. Re-run analyze to update it:
|
||||
|
||||
```bash
|
||||
npx gitnexus analyze
|
||||
```
|
||||
|
||||
If the index previously included embeddings, preserve them by adding `--embeddings`:
|
||||
|
||||
```bash
|
||||
npx gitnexus analyze --embeddings
|
||||
```
|
||||
|
||||
To check whether embeddings exist, inspect `.gitnexus/meta.json` — the `stats.embeddings` field shows the count (0 means no embeddings). **Running analyze without `--embeddings` will delete any previously generated embeddings.**
|
||||
|
||||
> Claude Code users: A PostToolUse hook handles this automatically after `git commit` and `git merge`.
|
||||
|
||||
## CLI
|
||||
|
||||
| Task | Read this skill file |
|
||||
|------|---------------------|
|
||||
@@ -23,40 +95,7 @@ For any task involving code understanding, debugging, impact analysis, or refact
|
||||
| Blast radius / "What breaks if I change X?" | `.claude/skills/gitnexus/gitnexus-impact-analysis/SKILL.md` |
|
||||
| Trace bugs / "Why is X failing?" | `.claude/skills/gitnexus/gitnexus-debugging/SKILL.md` |
|
||||
| Rename / extract / split / refactor | `.claude/skills/gitnexus/gitnexus-refactoring/SKILL.md` |
|
||||
| Tools, resources, schema reference | `.claude/skills/gitnexus/gitnexus-guide/SKILL.md` |
|
||||
| Index, status, clean, wiki CLI commands | `.claude/skills/gitnexus/gitnexus-cli/SKILL.md` |
|
||||
|
||||
## Tools Reference
|
||||
|
||||
| Tool | What it gives you |
|
||||
|------|-------------------|
|
||||
| `query` | Process-grouped code intelligence — execution flows related to a concept |
|
||||
| `context` | 360-degree symbol view — categorized refs, processes it participates in |
|
||||
| `impact` | Symbol blast radius — what breaks at depth 1/2/3 with confidence |
|
||||
| `detect_changes` | Git-diff impact — what do your current changes affect |
|
||||
| `rename` | Multi-file coordinated rename with confidence-tagged edits |
|
||||
| `cypher` | Raw graph queries (read `gitnexus://repo/{name}/schema` first) |
|
||||
| `list_repos` | Discover indexed repos |
|
||||
|
||||
## Resources Reference
|
||||
|
||||
Lightweight reads (~100-500 tokens) for navigation:
|
||||
|
||||
| Resource | Content |
|
||||
|----------|---------|
|
||||
| `gitnexus://repo/{name}/context` | Stats, staleness check |
|
||||
| `gitnexus://repo/{name}/clusters` | All functional areas with cohesion scores |
|
||||
| `gitnexus://repo/{name}/cluster/{clusterName}` | Area members |
|
||||
| `gitnexus://repo/{name}/processes` | All execution flows |
|
||||
| `gitnexus://repo/{name}/process/{processName}` | Step-by-step trace |
|
||||
| `gitnexus://repo/{name}/schema` | Graph schema for Cypher |
|
||||
|
||||
## Graph Schema
|
||||
|
||||
**Nodes:** File, Function, Class, Interface, Method, Community, Process
|
||||
**Edges (via CodeRelation.type):** CALLS, IMPORTS, EXTENDS, IMPLEMENTS, DEFINES, MEMBER_OF, STEP_IN_PROCESS
|
||||
|
||||
```cypher
|
||||
MATCH (caller)-[:CodeRelation {type: 'CALLS'}]->(f:Function {name: "myFunc"})
|
||||
RETURN caller.name, caller.filePath
|
||||
```
|
||||
|
||||
<!-- gitnexus:end -->
|
||||
<!-- gitnexus:end -->
|
||||
|
||||
@@ -1,13 +1,30 @@
|
||||
# GitNexus
|
||||
⚠️ Important Notice:** GitNexus has NO official cryptocurrency, token, or coin. Any token/coin using the GitNexus name on Pump.fun or any other platform is **not affiliated with, endorsed by, or created by** this project or its maintainers. Do not purchase any cryptocurrency claiming association with GitNexus.
|
||||
|
||||
<a href="https://trendshift.io/repositories/19809" target="_blank"><img src="https://trendshift.io/api/badge/repositories/19809" alt="abhigyanpatwari%2FGitNexus | Trendshift" style="width: 250px; height: 55px;" width="250" height="55"/></a>
|
||||
<div align="center">
|
||||
|
||||
**Building git for agent context.**
|
||||
<a href="https://trendshift.io/repositories/19809" target="_blank">
|
||||
<img src="https://trendshift.io/api/badge/repositories/19809" alt="abhigyanpatwari%2FGitNexus | Trendshift" style="width: 250px; height: 55px;" width="250" height="55"/>
|
||||
</a>
|
||||
|
||||
<h2>Join the official Discord to discuss ideas, issues etc!</h2>
|
||||
|
||||
<a href="https://discord.gg/AAsRVT6fGb">
|
||||
<img src="https://img.shields.io/discord/1477255801545429032?color=5865F2&logo=discord&logoColor=white" alt="Discord"/>
|
||||
</a>
|
||||
<a href="https://www.npmjs.com/package/gitnexus">
|
||||
<img src="https://img.shields.io/npm/v/gitnexus.svg" alt="npm version"/>
|
||||
</a>
|
||||
<a href="https://polyformproject.org/licenses/noncommercial/1.0.0/">
|
||||
<img src="https://img.shields.io/badge/License-PolyForm%20Noncommercial-blue.svg" alt="License: PolyForm Noncommercial"/>
|
||||
</a>
|
||||
|
||||
</div>
|
||||
|
||||
**Building nervous system for agent context.**
|
||||
|
||||
Indexes any codebase into a knowledge graph — every dependency, call chain, cluster, and execution flow — then exposes it through smart tools so AI agents never miss code.
|
||||
|
||||
[](https://www.npmjs.com/package/gitnexus)
|
||||
[](https://polyformproject.org/licenses/noncommercial/1.0.0/)
|
||||
|
||||
|
||||
|
||||
@@ -17,7 +34,7 @@ https://github.com/user-attachments/assets/172685ba-8e54-4ea7-9ad1-e31a3398da72
|
||||
|
||||
> *Like DeepWiki, but deeper.* DeepWiki helps you *understand* code. GitNexus lets you *analyze* it — because a knowledge graph tracks every relationship, not just descriptions.
|
||||
|
||||
**TL;DR:** The **Web UI** is a quick way to chat with any repo. The **CLI + MCP** is how you make your AI agent actually reliable — it gives Cursor, Claude Code, and friends a deep architectural view of your codebase so they stop missing dependencies, breaking call chains, and shipping blind edits. Even smaller models get full architectural clarity, making it compete with goliath models.
|
||||
**TL;DR:** The **Web UI** is a quick way to chat with any repo. The **CLI + MCP** is how you make your AI agent actually reliable — it gives Cursor, Claude Code, Codex, and friends a deep architectural view of your codebase so they stop missing dependencies, breaking call chains, and shipping blind edits. Even smaller models get full architectural clarity, making it compete with goliath models.
|
||||
|
||||
---
|
||||
|
||||
@@ -31,10 +48,10 @@ https://github.com/user-attachments/assets/172685ba-8e54-4ea7-9ad1-e31a3398da72
|
||||
| | **CLI + MCP** | **Web UI** |
|
||||
| ----------------- | -------------------------------------------------------------- | ------------------------------------------------------------ |
|
||||
| **What** | Index repos locally, connect AI agents via MCP | Visual graph explorer + AI chat in browser |
|
||||
| **For** | Daily development with Cursor, Claude Code, Windsurf, OpenCode | Quick exploration, demos, one-off analysis |
|
||||
| **For** | Daily development with Cursor, Claude Code, Codex, Windsurf, OpenCode | Quick exploration, demos, one-off analysis |
|
||||
| **Scale** | Full repos, any size | Limited by browser memory (~5k files), or unlimited via backend mode |
|
||||
| **Install** | `npm install -g gitnexus` | No install —[gitnexus.vercel.app](https://gitnexus.vercel.app) |
|
||||
| **Storage** | KuzuDB native (fast, persistent) | KuzuDB WASM (in-memory, per session) |
|
||||
| **Storage** | LadybugDB native (fast, persistent) | LadybugDB WASM (in-memory, per session) |
|
||||
| **Parsing** | Tree-sitter native bindings | Tree-sitter WASM |
|
||||
| **Privacy** | Everything local, no network | Everything in-browser, no server |
|
||||
|
||||
@@ -65,25 +82,42 @@ To configure MCP for your editor, run `npx gitnexus setup` once — or set it up
|
||||
|
||||
| Editor | MCP | Skills | Hooks (auto-augment) | Support |
|
||||
| --------------------- | --- | ------ | -------------------- | -------------- |
|
||||
| **Claude Code** | Yes | Yes | Yes (PreToolUse) | **Full** |
|
||||
| **Claude Code** | Yes | Yes | Yes (PreToolUse + PostToolUse) | **Full** |
|
||||
| **Cursor** | Yes | Yes | — | MCP + Skills |
|
||||
| **Codex** | Yes | Yes | — | MCP + Skills |
|
||||
| **Windsurf** | Yes | — | — | MCP |
|
||||
| **OpenCode** | Yes | Yes | — | MCP + Skills |
|
||||
| **Codex** | Yes | — | — | MCP |
|
||||
|
||||
> **Claude Code** gets the deepest integration: MCP tools + agent skills + PreToolUse hooks that automatically enrich grep/glob/bash calls with knowledge graph context.
|
||||
> **Claude Code** gets the deepest integration: MCP tools + agent skills + PreToolUse hooks that enrich searches with graph context + PostToolUse hooks that auto-reindex after commits.
|
||||
|
||||
### Community Integrations
|
||||
## Community Integrations
|
||||
|
||||
| Agent | Install | Source |
|
||||
|-------|---------|--------|
|
||||
| [pi](https://pi.dev) | `pi install npm:pi-gitnexus` | [pi-gitnexus](https://github.com/tintinweb/pi-gitnexus) |
|
||||
Built by the community — not officially maintained, but worth checking out.
|
||||
|
||||
| Project | Author | Description |
|
||||
|---------|--------|-------------|
|
||||
| [pi-gitnexus](https://github.com/tintinweb/pi-gitnexus) | [@tintinweb](https://github.com/tintinweb) | GitNexus plugin for [pi](https://pi.dev) — `pi install npm:pi-gitnexus` |
|
||||
| [gitnexus-stable-ops](https://github.com/ShunsukeHayashi/gitnexus-stable-ops) | [@ShunsukeHayashi](https://github.com/ShunsukeHayashi) | Stable ops & deployment workflows (Miyabi ecosystem) |
|
||||
|
||||
> Have a project built on GitNexus? Open a PR to add it here!
|
||||
|
||||
If you prefer manual configuration:
|
||||
|
||||
**Claude Code** (full support — MCP + skills + hooks):
|
||||
|
||||
```bash
|
||||
# macOS / Linux
|
||||
claude mcp add gitnexus -- npx -y gitnexus@latest mcp
|
||||
|
||||
# Windows
|
||||
claude mcp add gitnexus -- cmd /c npx -y gitnexus@latest mcp
|
||||
```
|
||||
|
||||
**Codex** (full support — MCP + skills):
|
||||
|
||||
```bash
|
||||
codex mcp add gitnexus -- npx -y gitnexus@latest mcp
|
||||
```
|
||||
|
||||
**Cursor** (`~/.cursor/mcp.json` — global, works for all projects):
|
||||
@@ -112,13 +146,24 @@ claude mcp add gitnexus -- npx -y gitnexus@latest mcp
|
||||
}
|
||||
```
|
||||
|
||||
**Codex** (`~/.codex/config.toml` for system scope, or `.codex/config.toml` for project scope):
|
||||
|
||||
```toml
|
||||
[mcp_servers.gitnexus]
|
||||
command = "npx"
|
||||
args = ["-y", "gitnexus@latest", "mcp"]
|
||||
```
|
||||
|
||||
### CLI Commands
|
||||
|
||||
```bash
|
||||
gitnexus setup # Configure MCP for your editors (one-time)
|
||||
gitnexus analyze [path] # Index a repository (or update stale index)
|
||||
gitnexus analyze --force # Force full re-index
|
||||
gitnexus analyze --skills # Generate repo-specific skill files from detected communities
|
||||
gitnexus analyze --skip-embeddings # Skip embedding generation (faster)
|
||||
gitnexus analyze --embeddings # Enable embedding generation (slower, better search)
|
||||
gitnexus analyze --verbose # Log skipped files when parsers are unavailable
|
||||
gitnexus mcp # Start MCP server (stdio) — serves all indexed repos
|
||||
gitnexus serve # Start local HTTP server (multi-repo) for web UI connection
|
||||
gitnexus list # List all indexed repositories
|
||||
@@ -172,6 +217,10 @@ gitnexus wiki --base-url <url> # Wiki with custom LLM API base URL
|
||||
- **Impact Analysis** — Analyze blast radius before changes
|
||||
- **Refactoring** — Plan safe refactors using dependency mapping
|
||||
|
||||
**Repo-specific skills** generated with `--skills`:
|
||||
|
||||
When you run `gitnexus analyze --skills`, GitNexus detects the functional areas of your codebase (via Leiden community detection) and generates a `SKILL.md` file for each one under `.claude/skills/generated/`. Each skill describes a module's key files, entry points, execution flows, and cross-area connections — so your AI agent gets targeted context for the exact area of code you're working in. Skills are regenerated on each `--skills` run to stay current with the codebase.
|
||||
|
||||
---
|
||||
|
||||
## Multi-Repo MCP Architecture
|
||||
@@ -200,8 +249,8 @@ flowchart TD
|
||||
Server["server.ts"]
|
||||
Backend["LocalBackend"]
|
||||
Pool["Connection Pool"]
|
||||
ConnA["KuzuDB conn A"]
|
||||
ConnB["KuzuDB conn B"]
|
||||
ConnA["LadybugDB conn A"]
|
||||
ConnB["LadybugDB conn B"]
|
||||
end
|
||||
|
||||
Setup -->|"writes global MCP config"| CursorConfig["~/.cursor/mcp.json"]
|
||||
@@ -218,7 +267,7 @@ flowchart TD
|
||||
ConnB -->|"queries"| RepoB
|
||||
```
|
||||
|
||||
**How it works:** Each `gitnexus analyze` stores the index in `.gitnexus/` inside the repo (portable, gitignored) and registers a pointer in `~/.gitnexus/registry.json`. When an AI agent starts, the MCP server reads the registry and can serve any indexed repo. KuzuDB connections are opened lazily on first query and evicted after 5 minutes of inactivity (max 5 concurrent). If only one repo is indexed, the `repo` parameter is optional on all tools — agents don't need to change anything.
|
||||
**How it works:** Each `gitnexus analyze` stores the index in `.gitnexus/` inside the repo (portable, gitignored) and registers a pointer in `~/.gitnexus/registry.json`. When an AI agent starts, the MCP server reads the registry and can serve any indexed repo. LadybugDB connections are opened lazily on first query and evicted after 5 minutes of inactivity (max 5 concurrent). If only one repo is indexed, the `repo` parameter is optional on all tools — agents don't need to change anything.
|
||||
|
||||
---
|
||||
|
||||
@@ -239,7 +288,7 @@ npm install
|
||||
npm run dev
|
||||
```
|
||||
|
||||
The web UI uses the same indexing pipeline as the CLI but runs entirely in WebAssembly (Tree-sitter WASM, KuzuDB WASM, in-browser embeddings). It's great for quick exploration but limited by browser memory for larger repos.
|
||||
The web UI uses the same indexing pipeline as the CLI but runs entirely in WebAssembly (Tree-sitter WASM, LadybugDB WASM, in-browser embeddings). It's great for quick exploration but limited by browser memory for larger repos.
|
||||
|
||||
**Local Backend Mode:** Run `gitnexus serve` and open the web UI locally — it auto-detects the server and shows all your indexed repos, with full AI chat support. No need to re-upload or re-index. The agent's tools (Cypher queries, search, code navigation) route through the backend HTTP API automatically.
|
||||
|
||||
@@ -247,7 +296,7 @@ The web UI uses the same indexing pipeline as the CLI but runs entirely in WebAs
|
||||
|
||||
## The Problem GitNexus Solves
|
||||
|
||||
Tools like **Cursor**, **Claude Code**, **Cline**, **Roo Code**, and **Windsurf** are powerful — but they don't truly know your codebase structure.
|
||||
Tools like **Cursor**, **Claude Code**, **Codex**, **Cline**, **Roo Code**, and **Windsurf** are powerful — but they don't truly know your codebase structure.
|
||||
|
||||
**What happens:**
|
||||
|
||||
@@ -296,14 +345,30 @@ GitNexus builds a complete knowledge graph of your codebase through a multi-phas
|
||||
|
||||
1. **Structure** — Walks the file tree and maps folder/file relationships
|
||||
2. **Parsing** — Extracts functions, classes, methods, and interfaces using Tree-sitter ASTs
|
||||
3. **Resolution** — Resolves imports and function calls across files with language-aware logic
|
||||
3. **Resolution** — Resolves imports, function calls, heritage, constructor inference, and `self`/`this` receiver types across files with language-aware logic
|
||||
4. **Clustering** — Groups related symbols into functional communities
|
||||
5. **Processes** — Traces execution flows from entry points through call chains
|
||||
6. **Search** — Builds hybrid search indexes for fast retrieval
|
||||
|
||||
### Supported Languages
|
||||
|
||||
TypeScript, JavaScript, Python, Java, C, C++, C#, Go, Rust
|
||||
| Language | Imports | Named Bindings | Exports | Heritage | Type Annotations | Constructor Inference | Config | Frameworks | Entry Points |
|
||||
|----------|---------|----------------|---------|----------|-----------------|---------------------|--------|------------|-------------|
|
||||
| TypeScript | ✓ | ✓ | ✓ | ✓ | ✓ | ✓ | ✓ | ✓ | ✓ |
|
||||
| JavaScript | ✓ | ✓ | ✓ | ✓ | — | ✓ | ✓ | ✓ | ✓ |
|
||||
| Python | ✓ | ✓ | ✓ | ✓ | ✓ | ✓ | ✓ | ✓ | ✓ |
|
||||
| Java | ✓ | ✓ | ✓ | ✓ | ✓ | ✓ | — | ✓ | ✓ |
|
||||
| Kotlin | ✓ | ✓ | ✓ | ✓ | ✓ | ✓ | — | ✓ | ✓ |
|
||||
| C# | ✓ | ✓ | ✓ | ✓ | ✓ | ✓ | ✓ | ✓ | ✓ |
|
||||
| Go | ✓ | — | ✓ | ✓ | ✓ | ✓ | ✓ | ✓ | ✓ |
|
||||
| Rust | ✓ | ✓ | ✓ | ✓ | ✓ | ✓ | — | ✓ | ✓ |
|
||||
| PHP | ✓ | ✓ | ✓ | — | ✓ | ✓ | ✓ | ✓ | ✓ |
|
||||
| Ruby | ✓ | — | ✓ | ✓ | — | ✓ | — | ✓ | ✓ |
|
||||
| Swift | — | — | ✓ | ✓ | ✓ | ✓ | ✓ | ✓ | ✓ |
|
||||
| C | — | — | ✓ | — | ✓ | ✓ | — | ✓ | ✓ |
|
||||
| C++ | — | — | ✓ | ✓ | ✓ | ✓ | — | ✓ | ✓ |
|
||||
|
||||
**Imports** — cross-file import resolution · **Named Bindings** — `import { X as Y }` / re-export tracking · **Exports** — public/exported symbol detection · **Heritage** — class inheritance, interfaces, mixins · **Type Annotations** — explicit type extraction for receiver resolution · **Constructor Inference** — infer receiver type from constructor calls (`self`/`this` resolution included for all languages) · **Config** — language toolchain config parsing (tsconfig, go.mod, etc.) · **Frameworks** — AST-based framework pattern detection · **Entry Points** — entry point scoring heuristics
|
||||
|
||||
---
|
||||
|
||||
@@ -442,7 +507,7 @@ The wiki generator reads the indexed graph structure, groups files into modules
|
||||
| ------------------------- | ------------------------------------- | --------------------------------------- |
|
||||
| **Runtime** | Node.js (native) | Browser (WASM) |
|
||||
| **Parsing** | Tree-sitter native bindings | Tree-sitter WASM |
|
||||
| **Database** | KuzuDB native | KuzuDB WASM |
|
||||
| **Database** | LadybugDB native | LadybugDB WASM |
|
||||
| **Embeddings** | HuggingFace transformers.js (GPU/CPU) | transformers.js (WebGPU/WASM) |
|
||||
| **Search** | BM25 + semantic + RRF | BM25 + semantic + RRF |
|
||||
| **Agent Interface** | MCP (stdio) | LangChain ReAct agent |
|
||||
@@ -463,9 +528,10 @@ The wiki generator reads the indexed graph structure, groups files into modules
|
||||
|
||||
### Recently Completed
|
||||
|
||||
- [X] Constructor-Inferred Type Resolution, `self`/`this` Receiver Mapping
|
||||
- [X] Wiki Generation, Multi-File Rename, Git-Diff Impact Analysis
|
||||
- [X] Process-Grouped Search, 360-Degree Context, Claude Code Hooks
|
||||
- [X] Multi-Repo MCP, Zero-Config Setup, 9 Language Support
|
||||
- [X] Multi-Repo MCP, Zero-Config Setup, 13 Language Support
|
||||
- [X] Community Detection, Process Detection, Confidence Scoring
|
||||
- [X] Hybrid Search, Vector Index
|
||||
|
||||
@@ -482,7 +548,7 @@ The wiki generator reads the indexed graph structure, groups files into modules
|
||||
## Acknowledgments
|
||||
|
||||
- [Tree-sitter](https://tree-sitter.github.io/) — AST parsing
|
||||
- [KuzuDB](https://kuzudb.com/) — Embedded graph database with vector support
|
||||
- [LadybugDB](https://ladybugdb.com/) — Embedded graph database with vector support (formerly KuzuDB)
|
||||
- [Sigma.js](https://www.sigmajs.org/) — WebGL graph rendering
|
||||
- [transformers.js](https://huggingface.co/docs/transformers.js) — Browser ML
|
||||
- [Graphology](https://graphology.github.io/) — Graph data structures
|
||||
|
||||
@@ -0,0 +1,57 @@
|
||||
---
|
||||
review_agents: [kieran-typescript-reviewer, pattern-recognition-specialist, architecture-strategist, data-integrity-guardian, security-sentinel, performance-oracle, code-simplicity-reviewer]
|
||||
plan_review_agents: [kieran-typescript-reviewer, architecture-strategist, code-simplicity-reviewer]
|
||||
voltagent_agents: [voltagent-lang:typescript-pro, voltagent-qa-sec:security-auditor, voltagent-data-ai:database-optimizer]
|
||||
---
|
||||
|
||||
# Review Context
|
||||
|
||||
## Project Overview
|
||||
GitNexus is a code intelligence tool that builds a knowledge graph from source code using tree-sitter AST parsing across 12 languages and KuzuDB for graph storage. Two packages: `gitnexus/` (CLI/MCP, TypeScript) and `gitnexus-web/` (browser).
|
||||
|
||||
## Cross-Language Pattern Consistency (pattern-recognition-specialist)
|
||||
- 12 language-specific type extractors in `gitnexus/src/core/ingestion/type-extractors/` must follow identical patterns for: async unwrapping, constructor binding, namespace handling, nullable type stripping, for-loop element typing.
|
||||
- Past bugs: C#/Rust missing `await_expression` unwrapping that TypeScript handled correctly; PHP backslash namespace splitting inconsistent with other languages' `::` / `.` splitting.
|
||||
- When reviewing type extractor changes, verify the same pattern exists in ALL applicable language files — asymmetry is the #1 source of bugs.
|
||||
|
||||
## Data Integrity (data-integrity-guardian)
|
||||
- KuzuDB graph operations: schema in `gitnexus/src/core/kuzu/schema.ts`, adapter in `kuzu-adapter.ts`.
|
||||
- The ingestion pipeline writes symbols and relationships to the graph — changes to node/relation schemas or the ingestion pipeline can corrupt the index.
|
||||
- Known issue: KuzuDB `close()` hangs on Linux due to C++ destructor — use `detachKuzu()` pattern.
|
||||
- `lbug-adapter.ts` fallback path needs quote/newline escaping for Cypher injection prevention.
|
||||
|
||||
## Security (security-sentinel)
|
||||
- Cypher query construction in `lbug-adapter.ts` and `kuzu-adapter.ts` — watch for injection via unescaped user-provided symbol names.
|
||||
- CLI accepts `--repo` parameter and file paths — validate against path traversal.
|
||||
- MCP server exposes tools to external AI agents — all tool inputs are untrusted.
|
||||
|
||||
## Performance (performance-oracle)
|
||||
- Tree-sitter buffer size is adaptive (512KB–32MB) via `getTreeSitterBufferSize()` in `constants.ts`.
|
||||
- The ingestion pipeline processes entire repositories — O(n) per file with potential O(n²) in cross-file resolution.
|
||||
- KuzuDB batch inserts vs individual inserts matter for large repos.
|
||||
|
||||
## Architecture (architecture-strategist)
|
||||
- Ingestion pipeline phases: structure → parsing → imports → calls → heritage → processes → type resolution.
|
||||
- Shared modules: `export-detection.ts`, `constants.ts`, `utils.ts` — changes here have wide blast radius.
|
||||
- `gitnexus-web` package drifts behind CLI — flag if a change should be mirrored.
|
||||
|
||||
## Voltagent Supplementary Agents
|
||||
|
||||
Invoke these via the Agent tool alongside `/ce:review` for deeper specialist analysis. These cover gaps that compound-engineering agents don't:
|
||||
|
||||
### voltagent-lang:typescript-pro
|
||||
**When:** Changes touch type-resolution logic, generics, conditional types, or complex type-level programming in `type-env.ts`, `type-extractors/*.ts`, or `types.ts`.
|
||||
**Why:** The type resolution system uses advanced TypeScript patterns (discriminated unions, mapped types, recursive generics) that benefit from deep TS type-system review beyond what kieran-typescript-reviewer covers.
|
||||
|
||||
### voltagent-qa-sec:security-auditor
|
||||
**When:** Changes touch MCP tool handlers, Cypher query construction, CLI argument parsing, or any code that processes external input.
|
||||
**Why:** GitNexus is an MCP server — all tool inputs come from untrusted AI agents. Systematic OWASP-level audit catches injection vectors that spot-checking misses. Past finding: `lbug-adapter.ts` fallback path had unescaped newlines in Cypher queries.
|
||||
|
||||
### voltagent-data-ai:database-optimizer
|
||||
**When:** Changes touch `kuzu-adapter.ts`, `schema.ts`, `lbug-adapter.ts`, or any Cypher query construction/execution.
|
||||
**Why:** No CE agent specializes in graph database optimization. KuzuDB batch insert patterns, index usage, and query planning directly affect analysis speed on large repos.
|
||||
|
||||
## Review Tooling
|
||||
- Use `gitnexus_impact()` before approving changes to any symbol — check d=1 (WILL BREAK) callers.
|
||||
- Use `gitnexus_detect_changes({scope: "compare", base_ref: "main"})` to map PR diffs to affected execution flows.
|
||||
- Use claude-mem to surface past architectural decisions relevant to the code under review.
|
||||
@@ -0,0 +1,100 @@
|
||||
# COBOL Code Indexing
|
||||
|
||||
GitNexus indexes COBOL codebases using a **regex-only extraction** strategy, bypassing tree-sitter entirely. This document explains why, how the pipeline works, and links to detailed sub-documents.
|
||||
|
||||
## Why Regex-Only?
|
||||
|
||||
The tree-sitter-cobol grammar (v0.0.1) has three critical limitations that make it unusable for production indexing:
|
||||
|
||||
| Issue | Impact | Severity |
|
||||
|-------|--------|----------|
|
||||
| External scanner hangs on ~5% of files | No timeout mechanism exists for the C scanner; the process blocks indefinitely | **Blocking** |
|
||||
| Only ~15% of paragraph headers detected | Most procedure-division paragraphs are invisible to the grammar | High |
|
||||
| Patch markers in cols 1-6 cause parse errors | Enterprise COBOL uses non-standard sequence area content (e.g., `mzADD`, `estero`, `#FIX`) | High |
|
||||
|
||||
Because the external scanner hang cannot be interrupted (there is no `setTimeoutMicros` equivalent for tree-sitter), using tree-sitter-cobol would hang the indexing pipeline on a non-trivial fraction of real-world files.
|
||||
|
||||
The regex-only approach provides:
|
||||
|
||||
- **Speed**: ~1ms per file average extraction time
|
||||
- **Reliability**: zero hangs, zero crashes across 13,000+ files
|
||||
- **Coverage**: captures all critical symbols -- program name, paragraphs, sections, CALL, PERFORM, COPY, data items (01-77, 88-level), file declarations, FD entries, EXEC SQL/CICS blocks, ENTRY points, and MOVE statements
|
||||
|
||||
## Architecture
|
||||
|
||||
```mermaid
|
||||
flowchart TD
|
||||
A[Repository Scan] --> B{File Detection}
|
||||
B -->|Extension match| C[COBOL file]
|
||||
B -->|GITNEXUS_COBOL_DIRS match| C
|
||||
B -->|No match| Z[Skip]
|
||||
|
||||
C --> D{Copybook?}
|
||||
D -->|Yes| E[Add to Copybook Map]
|
||||
D -->|No| F[Source Program]
|
||||
|
||||
E --> G[COPY Expansion Engine]
|
||||
F --> G
|
||||
|
||||
G -->|Inline copybook content| H[Expanded Source]
|
||||
H --> I[Patch Marker Cleanup]
|
||||
I --> J[Regex State Machine]
|
||||
|
||||
J --> K[Extracted Symbols]
|
||||
K --> L[Graph Model Builder]
|
||||
L --> M[Knowledge Graph]
|
||||
|
||||
subgraph "Per-Chunk Processing"
|
||||
G
|
||||
H
|
||||
I
|
||||
J
|
||||
K
|
||||
L
|
||||
end
|
||||
|
||||
subgraph "Post-Processing"
|
||||
M --> N[Community Detection]
|
||||
M --> O[Process Detection]
|
||||
M --> P[Contract Detection]
|
||||
end
|
||||
|
||||
style J fill:#e8f5e9,stroke:#2e7d32
|
||||
style G fill:#e3f2fd,stroke:#1565c0
|
||||
```
|
||||
|
||||
## COBOL vs Tree-Sitter Languages
|
||||
|
||||
| Feature | COBOL (Regex) | Tree-Sitter Languages |
|
||||
|---------|--------------|----------------------|
|
||||
| Parser | Single-pass regex state machine | tree-sitter grammar + queries |
|
||||
| Speed | ~1ms/file | ~5ms/file |
|
||||
| AST available | No | Yes |
|
||||
| COPY expansion | Yes (pre-processing step) | N/A |
|
||||
| Deep indexing | Data items, SQL, CICS, FD, ENTRY | Type annotations, generics, etc. |
|
||||
| Call extraction | PERFORM (intra-file) + CALL (cross-program) | AST-based call site detection |
|
||||
| Import extraction | COPY statements | `import`/`require`/`use`/`#include` |
|
||||
| Coverage | All critical symbols | Language-dependent query coverage |
|
||||
| Failure mode | Never hangs | External scanner can hang (COBOL only) |
|
||||
|
||||
## Sub-Documents
|
||||
|
||||
| Document | Description |
|
||||
|----------|-------------|
|
||||
| [File Detection](./file-detection.md) | Extension mapping, `GITNEXUS_COBOL_DIRS`, copybook classification |
|
||||
| [COPY Expansion](./copy-expansion.md) | Copybook inlining, REPLACING transformations, cycle detection |
|
||||
| [Regex Extraction](./regex-extraction.md) | State machine, regex patterns, line processing |
|
||||
| [Deep Indexing](./deep-indexing.md) | Data items, EXEC SQL/CICS, file declarations, FD, ENTRY, MOVE |
|
||||
| [Graph Model](./graph-model.md) | COBOL-specific node types, edge types, full annotated example |
|
||||
| [Performance](./performance.md) | Benchmarks, worker pool tuning, caps, troubleshooting |
|
||||
|
||||
## Key Source Files
|
||||
|
||||
| File | Purpose |
|
||||
|------|---------|
|
||||
| `gitnexus/src/core/ingestion/cobol-preprocessor.ts` | Patch marker cleanup + regex extraction engine |
|
||||
| `gitnexus/src/core/ingestion/cobol-copy-expander.ts` | COPY statement expansion with REPLACING |
|
||||
| `gitnexus/src/core/ingestion/utils.ts` | `getLanguageFromPath`, `getLanguageFromFilename` |
|
||||
| `gitnexus/src/core/ingestion/pipeline.ts` | `isCobolCopybook`, `expandCobolCopies`, `detectCrossProgamContracts` |
|
||||
| `gitnexus/src/core/ingestion/workers/parse-worker.ts` | `processCobolRegexOnly` -- graph model builder |
|
||||
| `gitnexus/src/core/ingestion/workers/worker-pool.ts` | Configurable sub-batch size for COBOL |
|
||||
@@ -0,0 +1,145 @@
|
||||
# COBOL COPY Expansion
|
||||
|
||||
The COPY statement is COBOL's include mechanism -- analogous to `#include` in C or `import` in modern languages. GitNexus expands COPY statements **before** regex extraction so that symbols defined inside copybooks (data items, paragraphs, etc.) are visible in the program's extracted graph.
|
||||
|
||||
## Supported Syntax
|
||||
|
||||
### Basic COPY
|
||||
|
||||
```cobol
|
||||
COPY CPSESP.
|
||||
COPY "WORKGRID.CPY".
|
||||
```
|
||||
|
||||
Inlines the content of the named copybook, replacing the COPY line(s).
|
||||
|
||||
### COPY with REPLACING
|
||||
|
||||
```cobol
|
||||
COPY CPSESP REPLACING "ANAZI-KEY" BY "LK-KEY".
|
||||
COPY CPSESP REPLACING LEADING "ESP-" BY "LK-ESP-"
|
||||
LEADING "KPSESPL" BY "LK-KPSESPL".
|
||||
COPY LINKAGE REPLACING TRAILING "-IN" BY "-OUT".
|
||||
```
|
||||
|
||||
Three REPLACING types are supported:
|
||||
|
||||
| Type | Syntax | Behavior | Example |
|
||||
| ------------ | ------------------------------------ | --------------------------------------- | -------------------------------- |
|
||||
| **EXACT** | `REPLACING "OLD" BY "NEW"` | Replace exact identifier matches | `ANAZI-KEY` becomes `LK-KEY` |
|
||||
| **LEADING** | `REPLACING LEADING "PFX-" BY "NEW-"` | Replace prefix on all COBOL identifiers | `ESP-NAME` becomes `LK-ESP-NAME` |
|
||||
| **TRAILING** | `REPLACING TRAILING "-IN" BY "-OUT"` | Replace suffix on all COBOL identifiers | `DATA-IN` becomes `DATA-OUT` |
|
||||
|
||||
Multiple REPLACING clauses can appear in a single COPY statement. They are applied in order to each COBOL identifier in the copybook content.
|
||||
|
||||
### Multi-Line COPY
|
||||
|
||||
COPY statements can span multiple lines (standard COBOL continuation rules apply):
|
||||
|
||||
```cobol
|
||||
COPY CPSESP REPLACING
|
||||
- LEADING "ESP-" BY "LK-ESP-"
|
||||
- LEADING "KPSESPL" BY "LK-KPSESPL".
|
||||
```
|
||||
|
||||
Continuation lines (indicator `-` in column 7) are merged before COPY statement scanning.
|
||||
|
||||
## Expansion Flow
|
||||
|
||||
```mermaid
|
||||
sequenceDiagram
|
||||
participant Pipeline
|
||||
participant Expander as COPY Expander
|
||||
participant Resolver
|
||||
participant Reader
|
||||
|
||||
Pipeline->>Pipeline: Identify all COBOL files
|
||||
Pipeline->>Pipeline: Classify copybooks vs programs
|
||||
Pipeline->>Reader: Read all copybook content upfront
|
||||
Reader-->>Pipeline: Copybook content map (name -> content)
|
||||
|
||||
loop For each source file in chunk
|
||||
Pipeline->>Expander: expandCopies(content, filePath, resolveFile, readFile)
|
||||
Expander->>Expander: Merge continuation lines
|
||||
Expander->>Expander: Detect COPY statements via regex
|
||||
|
||||
loop For each COPY statement (reverse order)
|
||||
Expander->>Resolver: resolveFile(copyTarget)
|
||||
Resolver-->>Expander: Copybook key or null
|
||||
|
||||
alt Resolved successfully
|
||||
Expander->>Reader: readFile(resolvedKey)
|
||||
Reader-->>Expander: Copybook content
|
||||
|
||||
Expander->>Expander: Apply REPLACING transformations
|
||||
Expander->>Expander: Recurse for nested COPYs (depth + 1)
|
||||
Expander->>Expander: Splice expanded content into output
|
||||
else Not resolved
|
||||
Expander->>Expander: Keep original COPY line
|
||||
end
|
||||
end
|
||||
|
||||
Expander-->>Pipeline: Expanded content + resolution metadata
|
||||
Pipeline->>Pipeline: Replace file content with expanded content
|
||||
end
|
||||
```
|
||||
|
||||
## Cycle Detection
|
||||
|
||||
Circular COPY references (e.g., copybook A includes copybook B which includes copybook A) are detected and handled:
|
||||
|
||||
1. Each expansion chain maintains a `visited` set of resolved copybook paths
|
||||
2. If a copybook path is already in the visited set, the expansion is skipped
|
||||
3. A `warnedCircular` set (shared across all files in a chunk) deduplicates warning messages
|
||||
|
||||
Known circular copybooks in PROJECT-NAME: `ANAZI`, `ANDIP`, `QDIPE` (self-referential includes).
|
||||
|
||||
## Max Depth
|
||||
|
||||
Nested COPY expansion is limited to **10 levels** (`DEFAULT_MAX_DEPTH`). If a COPY chain exceeds this depth, a warning is logged and the remaining COPY statements are left unexpanded.
|
||||
|
||||
## REPLACING Application Detail
|
||||
|
||||
The REPLACING engine works by scanning all COBOL identifiers (matching `\b[A-Z][A-Z0-9-]*\b`) in the copybook content and applying each replacement rule:
|
||||
|
||||
```
|
||||
Original copybook content:
|
||||
05 ESP-NAME PIC X(30).
|
||||
05 ESP-CODE PIC X(10).
|
||||
05 KPSESPL-FLAG PIC X(01).
|
||||
|
||||
After REPLACING LEADING "ESP-" BY "LK-ESP-" LEADING "KPSESPL" BY "LK-KPSESPL":
|
||||
05 LK-ESP-NAME PIC X(30).
|
||||
05 LK-ESP-CODE PIC X(10).
|
||||
05 LK-KPSESPL-FLAG PIC X(01).
|
||||
```
|
||||
|
||||
For LEADING replacements, the engine checks if each identifier starts with the `from` prefix (case-insensitive) and replaces only the prefix portion, preserving the rest of the identifier.
|
||||
|
||||
For TRAILING replacements, the same logic applies to suffixes.
|
||||
|
||||
For EXACT replacements, only identifiers that match the `from` value exactly (case-insensitive) are replaced.
|
||||
|
||||
## Copybook Resolution
|
||||
|
||||
The resolver tries multiple strategies to match a COPY target name to a copybook file:
|
||||
|
||||
1. **Exact match**: `COPY CPSESP` resolves to copybook named `CPSESP`
|
||||
2. **Strip extension**: `COPY WORKGRID.CPY` strips `.CPY` and resolves to `WORKGRID`
|
||||
3. **Add extension**: `COPY CPSESP` tries `CPSESP.CPY` and `CPSESP.COPY`
|
||||
|
||||
If no match is found, the COPY statement is left in place (unexpanded) and a resolution record with `resolvedPath: null` is created.
|
||||
|
||||
## Pipeline Integration
|
||||
|
||||
The expansion runs **per chunk**, after file content is read but before dispatch to worker threads:
|
||||
|
||||
1. All copybook files are read upfront (they are typically small, collectively under 100MB)
|
||||
2. Per chunk, the copybook map is merged with chunk content (in case a chunk contains copybooks)
|
||||
3. Only programs (not copybooks themselves) undergo expansion
|
||||
4. The expanded content replaces the original content in-place before worker dispatch
|
||||
|
||||
## Source Files
|
||||
|
||||
- `gitnexus/src/core/ingestion/cobol-copy-expander.ts` -- `expandCopies()`, `parseReplacingClause()`, `applyReplacing()`
|
||||
- `gitnexus/src/core/ingestion/pipeline.ts` -- `expandCobolCopies()`, copybook map construction, chunk integration
|
||||
@@ -0,0 +1,265 @@
|
||||
# COBOL Deep Indexing
|
||||
|
||||
Beyond basic symbol extraction (program name, paragraphs, CALL, PERFORM, COPY), GitNexus performs deep indexing of COBOL-specific constructs: data items, EXEC SQL/CICS blocks, file declarations, FD entries, ENTRY points, and MOVE statements.
|
||||
|
||||
## Data Items
|
||||
|
||||
### Level Numbers
|
||||
|
||||
| Level Range | Meaning | Graph Node Type |
|
||||
|-------------|---------|-----------------|
|
||||
| 01 | Record (group item) | `Record` |
|
||||
| 02-49 | Elementary/group items | `Property` |
|
||||
| 66 | RENAMES | `Property` |
|
||||
| 77 | Independent item | `Property` |
|
||||
| 88 | Condition name | `Const` |
|
||||
|
||||
FILLER items are skipped (no useful name for the graph).
|
||||
|
||||
### Clauses Parsed
|
||||
|
||||
The `parseDataItemClauses()` function extracts these clauses from the trailing text of a data item declaration:
|
||||
|
||||
| Clause | Pattern | Example |
|
||||
|--------|---------|---------|
|
||||
| `PIC` / `PICTURE` | `\bPIC(?:TURE)?\s+(?:IS\s+)?(\S+)` | `PIC X(30)`, `PICTURE IS 9(5)V99` |
|
||||
| `USAGE` | `\bUSAGE\s+(?:IS\s+)?(COMP\|BINARY\|...)` | `USAGE IS COMP-3`, `BINARY` |
|
||||
| `REDEFINES` | `\bREDEFINES\s+([A-Z][A-Z0-9-]+)` | `REDEFINES WK-DATE-NUM` |
|
||||
| `OCCURS` | `\bOCCURS\s+(\d+)` | `OCCURS 12 TIMES` |
|
||||
|
||||
Standalone COMP variants (without the `USAGE` keyword) are also detected: `COMP`, `COMP-1` through `COMP-6`, `COMP-X`, `BINARY`, `PACKED-DECIMAL`.
|
||||
|
||||
### Data Hierarchy
|
||||
|
||||
Data items form a hierarchical structure based on level numbers. The extractor uses a **stack algorithm**:
|
||||
|
||||
```
|
||||
Processing order:
|
||||
01 WK-RECORD -> push {01, WK-RECORD} -> parent: Module
|
||||
05 WK-NAME -> push {05, WK-NAME} -> parent: WK-RECORD (01 < 05)
|
||||
10 WK-FIRST -> push {10, WK-FIRST} -> parent: WK-NAME (05 < 10)
|
||||
10 WK-LAST -> pop WK-FIRST, push -> parent: WK-NAME (05 < 10)
|
||||
05 WK-CODE -> pop WK-LAST, WK-NAME -> parent: WK-RECORD (01 < 05)
|
||||
88 WK-ACTIVE -> (88 handled separately) -> parent: WK-CODE
|
||||
```
|
||||
|
||||
The stack maintains items where each entry's level is strictly less than the next. When a new item arrives with a level <= the top of stack, items are popped until the stack top has a smaller level. A `CONTAINS` edge is created from the stack top to the new item.
|
||||
|
||||
For 88-level condition names, the parent is the immediately preceding non-88 data item (found by scanning backwards).
|
||||
|
||||
### Annotated Example
|
||||
|
||||
```cobol
|
||||
01 WK-EMPLOYEE.
|
||||
05 WK-EMP-ID PIC 9(6).
|
||||
05 WK-EMP-NAME PIC X(30).
|
||||
05 WK-EMP-STATUS PIC X(01).
|
||||
88 WK-ACTIVE VALUE "A".
|
||||
88 WK-INACTIVE VALUE "I".
|
||||
05 WK-SALARY PIC 9(7)V99 COMP-3.
|
||||
05 WK-DEPT PIC X(04) OCCURS 3 TIMES.
|
||||
```
|
||||
|
||||
Produces:
|
||||
- `Record` node: `WK-EMPLOYEE` (level 01, section: working-storage)
|
||||
- `Property` nodes: `WK-EMP-ID`, `WK-EMP-NAME`, `WK-EMP-STATUS`, `WK-SALARY`, `WK-DEPT`
|
||||
- `Const` nodes: `WK-ACTIVE` (values: `A`), `WK-INACTIVE` (values: `I`)
|
||||
- `CONTAINS` edges: `WK-EMPLOYEE -> WK-EMP-ID`, `WK-EMPLOYEE -> WK-EMP-NAME`, etc.
|
||||
- `CONTAINS` edges: `WK-EMP-STATUS -> WK-ACTIVE`, `WK-EMP-STATUS -> WK-INACTIVE`
|
||||
|
||||
### Data Item Cap
|
||||
|
||||
A maximum of **500 data items per file** (`MAX_DATA_ITEMS_PER_FILE`) are processed. Some COBOL programs (especially after COPY expansion) can have 10,000+ data items, which would cause graph bloat and push the V8 relationship Map past its 16.7M entry limit across thousands of files.
|
||||
|
||||
The cap applies after extraction: the first 500 items in source order are kept. Since 01-level records appear first, critical top-level structure is preserved.
|
||||
|
||||
## EXEC SQL
|
||||
|
||||
EXEC SQL blocks are accumulated across lines between `EXEC SQL` and `END-EXEC`, then parsed as a unit.
|
||||
|
||||
### Operation Classification
|
||||
|
||||
The first SQL keyword determines the operation:
|
||||
|
||||
| First Keyword | Operation |
|
||||
|---------------|-----------|
|
||||
| `SELECT` | SELECT |
|
||||
| `INSERT` | INSERT |
|
||||
| `UPDATE` | UPDATE |
|
||||
| `DELETE` | DELETE |
|
||||
| `DECLARE` | DECLARE |
|
||||
| `OPEN` | OPEN |
|
||||
| `CLOSE` | CLOSE |
|
||||
| `FETCH` | FETCH |
|
||||
| *(anything else)* | OTHER |
|
||||
|
||||
### Table Extraction
|
||||
|
||||
Tables are extracted from SQL clauses:
|
||||
|
||||
| Clause Pattern | Example |
|
||||
|----------------|---------|
|
||||
| `FROM <table>` | `SELECT * FROM EMPLOYEES` |
|
||||
| `INTO <table>` | `INSERT INTO EMPLOYEES` |
|
||||
| `UPDATE <table>` | `UPDATE EMPLOYEES SET ...` |
|
||||
| `JOIN <table>` | `LEFT JOIN DEPARTMENTS ON ...` |
|
||||
|
||||
### Cursor Detection
|
||||
|
||||
```cobol
|
||||
EXEC SQL
|
||||
DECLARE C-EMPLOYEES CURSOR FOR
|
||||
SELECT EMP-ID, EMP-NAME FROM EMPLOYEES
|
||||
WHERE DEPT = :WK-DEPT
|
||||
END-EXEC
|
||||
```
|
||||
|
||||
Extracts: cursor `C-EMPLOYEES`, table `EMPLOYEES`, host variable `WK-DEPT`.
|
||||
|
||||
### Host Variables
|
||||
|
||||
Host variables are COBOL variables referenced in SQL with a `:` prefix. The colon is stripped:
|
||||
|
||||
```sql
|
||||
WHERE EMP-ID = :WK-EMP-ID AND DEPT = :WK-DEPT
|
||||
```
|
||||
|
||||
Extracts: `WK-EMP-ID`, `WK-DEPT`.
|
||||
|
||||
### Graph Output
|
||||
|
||||
- `CodeElement` node per table, with description `sql-table op:{OP}`
|
||||
- `CodeElement` node per cursor, with description `sql-cursor`
|
||||
- `ACCESSES` edge from Module to each CodeElement
|
||||
- Deduplication: if the same table appears in multiple SQL blocks, only one node is created
|
||||
|
||||
## EXEC CICS
|
||||
|
||||
EXEC CICS blocks are accumulated and parsed similarly to SQL blocks.
|
||||
|
||||
### Command Detection
|
||||
|
||||
Two-word commands are detected first (matched against the block start):
|
||||
|
||||
```
|
||||
SEND MAP, RECEIVE MAP, SEND TEXT, SEND CONTROL, READ NEXT, READ PREV
|
||||
```
|
||||
|
||||
If no two-word command matches, the first word is used (e.g., `LINK`, `XCTL`, `RETURN`, `READ`, `WRITE`).
|
||||
|
||||
### Extraction
|
||||
|
||||
| Element | Pattern | Example |
|
||||
|---------|---------|---------|
|
||||
| MAP name | `MAP('name')` or `MAP("name")` | `EXEC CICS SEND MAP('EMPMENU')` |
|
||||
| PROGRAM name | `PROGRAM('name')` or `PROGRAM("name")` | `EXEC CICS LINK PROGRAM('BGTABUP')` |
|
||||
| TRANSID | `TRANSID('name')` or `TRANSID("name")` | `EXEC CICS START TRANSID('EMP1')` |
|
||||
|
||||
### Graph Output
|
||||
|
||||
- MAP: `CodeElement` node with description `cics-map cmd:{CMD}` + `ACCESSES` edge from Module
|
||||
- PROGRAM: `CALLS` edge (cross-program call via CICS LINK/XCTL)
|
||||
- TRANSID: `CodeElement` node with description `cics-transid cmd:{CMD}` + `ACCESSES` edge from Module
|
||||
|
||||
### Annotated Example
|
||||
|
||||
```cobol
|
||||
EXEC CICS
|
||||
SEND MAP('EMPMENU')
|
||||
MAPSET('EMPSET')
|
||||
FROM(WK-MAP-DATA)
|
||||
ERASE
|
||||
END-EXEC
|
||||
```
|
||||
|
||||
Produces:
|
||||
- `CodeElement` node: `EMPMENU` (description: `cics-map cmd:SEND MAP`)
|
||||
- `ACCESSES` edge: Module -> `EMPMENU`
|
||||
|
||||
## File Declarations
|
||||
|
||||
SELECT statements in the INPUT-OUTPUT SECTION are accumulated across multiple lines (until a period terminator) and parsed for:
|
||||
|
||||
| Clause | Pattern | Example |
|
||||
|--------|---------|---------|
|
||||
| SELECT | `SELECT <name>` | `SELECT MASTER-FILE` |
|
||||
| ASSIGN | `ASSIGN TO <file>` | `ASSIGN TO "MASTER.DAT"` |
|
||||
| ORGANIZATION | `ORGANIZATION IS <type>` | `ORGANIZATION IS INDEXED` |
|
||||
| ACCESS | `ACCESS MODE IS <mode>` | `ACCESS MODE IS DYNAMIC` |
|
||||
| RECORD KEY | `RECORD KEY IS <field>` | `RECORD KEY IS WK-EMP-ID` |
|
||||
| FILE STATUS | `FILE STATUS IS <field>` | `FILE STATUS IS WK-FILE-STATUS` |
|
||||
|
||||
### Graph Output
|
||||
|
||||
- `CodeElement` node with description containing all parsed clauses (e.g., `select org:INDEXED access:DYNAMIC key:WK-EMP-ID status:WK-FILE-STATUS assign:MASTER.DAT`)
|
||||
- `RECORD_KEY_OF` edge: from Property node to CodeElement (confidence 0.8)
|
||||
- `FILE_STATUS_OF` edge: from Property node to CodeElement (confidence 0.8)
|
||||
|
||||
## FD Entries
|
||||
|
||||
FD (File Description) entries associate a file name with its record layout:
|
||||
|
||||
```cobol
|
||||
FD MASTER-FILE.
|
||||
01 MASTER-RECORD.
|
||||
05 MR-EMP-ID PIC 9(6).
|
||||
05 MR-EMP-NAME PIC X(30).
|
||||
```
|
||||
|
||||
The extractor tracks `pendingFdName` state: when an `FD` line is seen, the next 01-level data item becomes its record.
|
||||
|
||||
### Graph Output
|
||||
|
||||
- `CodeElement` node with description `fd record:{recordName}`
|
||||
- `CONTAINS` edge: FD CodeElement -> Record node
|
||||
- `CONTAINS` edge: SELECT CodeElement -> FD CodeElement (linking file declaration to file description)
|
||||
|
||||
## ENTRY Points
|
||||
|
||||
The `ENTRY` statement defines additional entry points into a COBOL program (in addition to the main program entry):
|
||||
|
||||
```cobol
|
||||
ENTRY "SUBPROG" USING WK-PARAM-1 WK-PARAM-2.
|
||||
```
|
||||
|
||||
### Graph Output
|
||||
|
||||
- `Constructor` node with description `entry params:{param1},{param2}` (or just `entry` if no parameters)
|
||||
- `CONTAINS` edge: Module -> Constructor
|
||||
- Symbol table entry (so the entry point is discoverable by name)
|
||||
|
||||
## PROCEDURE DIVISION USING
|
||||
|
||||
```cobol
|
||||
PROCEDURE DIVISION USING WK-INPUT-REC WK-OUTPUT-REC.
|
||||
```
|
||||
|
||||
The USING clause identifies parameters received by the program from its caller.
|
||||
|
||||
### Graph Output
|
||||
|
||||
- `RECEIVES` edge: Module -> Property (for each parameter name, confidence 0.8)
|
||||
|
||||
## MOVE Statements
|
||||
|
||||
MOVE statements are extracted but currently only stored in the regex results (not emitted as graph edges):
|
||||
|
||||
```cobol
|
||||
MOVE WK-NAME TO OUT-NAME.
|
||||
MOVE CORRESPONDING WK-INPUT TO WK-OUTPUT.
|
||||
```
|
||||
|
||||
### Extraction Details
|
||||
|
||||
- Source and target identifiers are captured
|
||||
- `CORRESPONDING` keyword is tracked (bulk field-by-field move)
|
||||
- Figurative constants (SPACES, ZEROS, LOW-VALUES, HIGH-VALUES, QUOTES, ALL) are skipped
|
||||
- The enclosing paragraph (`caller`) is tracked for context
|
||||
|
||||
DATA_FLOW edges from MOVE statements are reserved for a future release.
|
||||
|
||||
## Source Files
|
||||
|
||||
- `gitnexus/src/core/ingestion/cobol-preprocessor.ts` -- All extraction logic, clause parsers, EXEC block parsers
|
||||
- `gitnexus/src/core/ingestion/workers/parse-worker.ts` -- `processCobolRegexOnly()`, graph node/edge emission
|
||||
- `gitnexus/src/core/ingestion/parsing-processor.ts` -- Sequential fallback with same `MAX_DATA_ITEMS_PER_FILE` cap
|
||||
@@ -0,0 +1,126 @@
|
||||
# COBOL File Detection
|
||||
|
||||
GitNexus detects COBOL files through two mechanisms: extension-based mapping and directory-based override for extensionless files. This document covers both, plus the copybook/program classification logic.
|
||||
|
||||
## Extension Mapping
|
||||
|
||||
### Program Extensions
|
||||
|
||||
| Extension | Type |
|
||||
|-----------|------|
|
||||
| `.cbl` | COBOL program |
|
||||
| `.cob` | COBOL program |
|
||||
| `.cobol` | COBOL program |
|
||||
|
||||
### Copybook Extensions
|
||||
|
||||
| Extension | Type | Notes |
|
||||
|-----------|------|-------|
|
||||
| `.cpy` | Copybook | Standard |
|
||||
| `.copy` | Copybook | Standard |
|
||||
| `.gnm` / `.GNM` | Copybook | Enterprise (GnuCOBOL naming) |
|
||||
| `.fd` / `.FD` | Copybook | File Description fragment |
|
||||
| `.wrk` / `.WRK` | Copybook | Working-Storage fragment |
|
||||
| `.sel` / `.SEL` | Copybook | SELECT clause fragment |
|
||||
| `.open` / `.OPEN` | Copybook | File OPEN fragment |
|
||||
| `.close` / `.CLOSE` | Copybook | File CLOSE fragment |
|
||||
| `.ini` / `.INI` | Copybook | Initialization fragment |
|
||||
| `.def` / `.DEF` | Copybook | Definition fragment |
|
||||
|
||||
All extension matching is case-sensitive in `getLanguageFromFilename` (the extensions above are matched as written, including uppercase variants like `.GNM`).
|
||||
|
||||
## Extensionless File Detection: `GITNEXUS_COBOL_DIRS`
|
||||
|
||||
Many enterprise COBOL repositories use extensionless files -- the filename alone identifies the program (e.g., `s/BGTABFL` is the source for program `BGTABFL`). GitNexus handles this via the `GITNEXUS_COBOL_DIRS` environment variable.
|
||||
|
||||
### Configuration
|
||||
|
||||
Set `GITNEXUS_COBOL_DIRS` to a comma-separated list of directory names:
|
||||
|
||||
```bash
|
||||
# Files in s/, c/, and wfproc/ directories (at any depth) are treated as COBOL
|
||||
export GITNEXUS_COBOL_DIRS=s,c,wfproc
|
||||
```
|
||||
|
||||
The matching is **case-insensitive** and checks all path segments:
|
||||
|
||||
- `/repo/s/BGTABFL` -- matches segment `s` -- COBOL
|
||||
- `/repo/src/c/CPSESP` -- matches segment `c` -- COBOL
|
||||
- `/repo/wfproc/WF001` -- matches segment `wfproc` -- COBOL
|
||||
- `/repo/docs/README` -- no matching segment -- skipped
|
||||
|
||||
### Decision Tree
|
||||
|
||||
```mermaid
|
||||
flowchart TD
|
||||
A[getLanguageFromPath] --> B[getLanguageFromFilename]
|
||||
B --> C{Known extension?}
|
||||
C -->|Yes .cbl/.cob/.cobol/.cpy/...| D[Return COBOL]
|
||||
C -->|Yes .ts/.py/.java/...| E[Return other language]
|
||||
C -->|No match| F{Has extension?}
|
||||
|
||||
F -->|"Has dot in basename"| G[Return null]
|
||||
F -->|"No dot = extensionless"| H{GITNEXUS_COBOL_DIRS set?}
|
||||
|
||||
H -->|No| G
|
||||
H -->|Yes| I{Any path segment<br/>matches a configured dir?}
|
||||
|
||||
I -->|Yes| D
|
||||
I -->|No| G
|
||||
|
||||
style D fill:#e8f5e9,stroke:#2e7d32
|
||||
style G fill:#ffebee,stroke:#c62828
|
||||
```
|
||||
|
||||
### Implementation Detail
|
||||
|
||||
The `GITNEXUS_COBOL_DIRS` value is parsed once (on first call) and cached in a `Set<string>`:
|
||||
|
||||
```typescript
|
||||
// From gitnexus/src/core/ingestion/utils.ts
|
||||
const getCobolDirs = (): Set<string> => {
|
||||
if (_cobolDirs) return _cobolDirs;
|
||||
const raw = process.env.GITNEXUS_COBOL_DIRS;
|
||||
_cobolDirs = raw
|
||||
? new Set(raw.split(',').map(d => d.trim().toLowerCase()))
|
||||
: new Set();
|
||||
return _cobolDirs;
|
||||
};
|
||||
```
|
||||
|
||||
The path segment check splits the full path on `/` and tests each segment against the cached set.
|
||||
|
||||
## Copybook vs Program Classification
|
||||
|
||||
After a file is identified as COBOL, it must be classified as either a **program** (to be parsed for symbols) or a **copybook** (to be loaded into the copybook map for COPY expansion).
|
||||
|
||||
### Classification Rules
|
||||
|
||||
A COBOL file is classified as a **copybook** if ANY of these conditions is true:
|
||||
|
||||
1. It has a recognized copybook extension (`.cpy`, `.copy`, `.gnm`, `.fd`, `.wrk`, `.sel`, `.open`, `.close`, `.ini`, `.def`)
|
||||
2. It is an extensionless file whose path contains a directory segment matching one of: `c`, `copy`, `copybooks`, `copylib`, `cpy`
|
||||
|
||||
A file is classified as a **program** if:
|
||||
|
||||
1. It has a program extension (`.cbl`, `.cob`, `.cobol`), OR
|
||||
2. It is extensionless and does NOT match any copybook directory pattern
|
||||
|
||||
### Copybook Name Resolution
|
||||
|
||||
Copybook names are derived from the filename:
|
||||
|
||||
- Strip the extension (if any)
|
||||
- Convert to uppercase
|
||||
|
||||
Examples:
|
||||
- `c/CPSESP` -- name: `CPSESP`
|
||||
- `copy/workgrid.cpy` -- name: `WORKGRID`
|
||||
- `c/ANAZI.GNM` -- name: `ANAZI`
|
||||
|
||||
This name is used to resolve `COPY CPSESP.` statements during expansion.
|
||||
|
||||
## Source Files
|
||||
|
||||
- `gitnexus/src/core/ingestion/utils.ts` -- `getLanguageFromPath()`, `getLanguageFromFilename()`, `getCobolDirs()`
|
||||
- `gitnexus/src/core/ingestion/pipeline.ts` -- `isCobolCopybook()`, `getCopybookName()`, `COPYBOOK_EXTENSIONS`, `COBOL_PROGRAM_EXTENSIONS`
|
||||
@@ -0,0 +1,193 @@
|
||||
# COBOL Graph Model
|
||||
|
||||
This document describes the graph nodes and edges that GitNexus creates for COBOL codebases. The COBOL graph model is richer than most tree-sitter languages because it captures domain-specific constructs: file declarations, FD entries, data hierarchies, SQL tables, CICS maps, and cross-program contracts.
|
||||
|
||||
## Entity-Relationship Diagram
|
||||
|
||||
```mermaid
|
||||
erDiagram
|
||||
File ||--o{ Module : DEFINES
|
||||
File ||--o{ Function : DEFINES
|
||||
File ||--o{ Namespace : DEFINES
|
||||
File ||--o{ Record : DEFINES
|
||||
File ||--o{ Property : DEFINES
|
||||
File ||--o{ Const : DEFINES
|
||||
File ||--o{ CodeElement : DEFINES
|
||||
File ||--o{ Constructor : DEFINES
|
||||
File }o--o{ File : IMPORTS
|
||||
|
||||
Module ||--o{ Record : CONTAINS
|
||||
Module ||--o{ Constructor : CONTAINS
|
||||
Module }o--o{ CodeElement : ACCESSES
|
||||
Module }o--o{ Module : CALLS
|
||||
Module }o--o{ Module : CONTRACTS
|
||||
Module }o--o{ Property : RECEIVES
|
||||
|
||||
Record ||--o{ Property : CONTAINS
|
||||
Record ||--o{ Const : CONTAINS
|
||||
Record }o--o{ Record : REDEFINES
|
||||
|
||||
Property ||--o{ Property : CONTAINS
|
||||
Property ||--o{ Const : CONTAINS
|
||||
Property }o--o{ Property : REDEFINES
|
||||
Property }o--o{ CodeElement : RECORD_KEY_OF
|
||||
Property }o--o{ CodeElement : FILE_STATUS_OF
|
||||
|
||||
CodeElement ||--o{ CodeElement : CONTAINS
|
||||
CodeElement ||--o{ Record : CONTAINS
|
||||
|
||||
Function }o--o{ Function : CALLS
|
||||
```
|
||||
|
||||
## Node Types
|
||||
|
||||
| Node Type | COBOL Concept | Created From | Example |
|
||||
|-----------|--------------|--------------|---------|
|
||||
| `Module` | PROGRAM-ID | `PROGRAM-ID. BGTABFL` | Name: `BGTABFL`, description may include author and date |
|
||||
| `Function` | Paragraph | `PROCESS-RECORD.` at column 8 | Name: `PROCESS-RECORD` |
|
||||
| `Namespace` | Procedure section | `MAIN-LOGIC SECTION.` at column 8 | Name: `MAIN-LOGIC` |
|
||||
| `Record` | 01-level data item | `01 WK-EMPLOYEE.` | Description: `level:01 section:working-storage` |
|
||||
| `Property` | 02-49/66/77 data item | `05 WK-NAME PIC X(30).` | Description: `level:05 pic:X(30) section:working-storage` |
|
||||
| `Const` | 88-level condition | `88 WK-ACTIVE VALUE "A".` | Description: `level:88 values:A` |
|
||||
| `CodeElement` | SELECT, FD, SQL table, CICS map, cursor, transid | Various | Description varies by subtype |
|
||||
| `Constructor` | ENTRY point | `ENTRY "SUBPROG" USING WK-DATA` | Description: `entry params:WK-DATA` |
|
||||
|
||||
### CodeElement Subtypes
|
||||
|
||||
CodeElement is used for multiple COBOL constructs, distinguished by their description prefix:
|
||||
|
||||
| Subtype | ID Pattern | Description Format | Example |
|
||||
|---------|-----------|-------------------|---------|
|
||||
| File SELECT | `CodeElement:{path}:SELECT:{name}` | `select org:INDEXED access:DYNAMIC ...` | `SELECT MASTER-FILE` |
|
||||
| FD entry | `CodeElement:{path}:FD:{name}` | `fd record:{recordName}` | `FD MASTER-FILE` |
|
||||
| SQL table | `CodeElement:{path}:sql-table:{name}` | `sql-table op:SELECT` | Table `EMPLOYEES` |
|
||||
| SQL cursor | `CodeElement:{path}:sql-cursor:{name}` | `sql-cursor` | Cursor `C-EMPLOYEES` |
|
||||
| CICS map | `CodeElement:{path}:cics-map:{name}` | `cics-map cmd:SEND MAP` | Map `EMPMENU` |
|
||||
| CICS transid | `CodeElement:{path}:cics-transid:{name}` | `cics-transid cmd:START` | Transid `EMP1` |
|
||||
|
||||
## Edge Types
|
||||
|
||||
| Edge Type | Source | Target | Created By | Confidence | Example |
|
||||
|-----------|--------|--------|-----------|------------|---------|
|
||||
| `DEFINES` | File | any node | File defines its symbols | 1.0 | File -> Module `BGTABFL` |
|
||||
| `CALLS` | Function | Function | `PERFORM X [THRU Y]` | (via call-processor) | `PROCESS-RECORD` -> `CALC-TAX` |
|
||||
| `CALLS` | Module | Module | `CALL "BGTABUP"` | (via call-processor) | `BGTABFL` -> `BGTABUP` |
|
||||
| `CALLS` | Module | Module | `EXEC CICS LINK PROGRAM('X')` | (via call-processor) | `BGTABFL` -> `BGTABUP` |
|
||||
| `IMPORTS` | File | File | `COPY copybook` | (via import-processor) | Source file -> Copybook file |
|
||||
| `CONTAINS` | Module | Record | Data hierarchy root | 1.0 | `BGTABFL` -> `WK-EMPLOYEE` |
|
||||
| `CONTAINS` | Record | Property | Data hierarchy | 1.0 | `WK-EMPLOYEE` -> `WK-NAME` |
|
||||
| `CONTAINS` | Property | Property | Nested data items | 1.0 | `WK-ADDRESS` -> `WK-CITY` |
|
||||
| `CONTAINS` | Record/Property | Const | 88-level parent | 1.0 | `WK-STATUS` -> `WK-ACTIVE` |
|
||||
| `CONTAINS` | CodeElement (FD) | Record | FD record link | 1.0 | `FD:MASTER-FILE` -> `MASTER-RECORD` |
|
||||
| `CONTAINS` | CodeElement (SELECT) | CodeElement (FD) | SELECT-FD link | 0.9 | `SELECT:MASTER-FILE` -> `FD:MASTER-FILE` |
|
||||
| `CONTAINS` | Module | Constructor | ENTRY in module | 1.0 | `BGTABFL` -> `SUBPROG` |
|
||||
| `REDEFINES` | Record | Record | `01 X REDEFINES Y` | 1.0 | `WK-DATE-NUM` -> `WK-DATE-ALPHA` |
|
||||
| `REDEFINES` | Property | Property | `05 X REDEFINES Y` | 1.0 | `WK-CODE-NUM` -> `WK-CODE-ALPHA` |
|
||||
| `RECORD_KEY_OF` | Property | CodeElement (SELECT) | `RECORD KEY IS field` | 0.8 | `WK-EMP-ID` -> `SELECT:MASTER-FILE` |
|
||||
| `FILE_STATUS_OF` | Property | CodeElement (SELECT) | `FILE STATUS IS field` | 0.8 | `WK-FS` -> `SELECT:MASTER-FILE` |
|
||||
| `ACCESSES` | Module | CodeElement | EXEC SQL/CICS | 0.9 | `BGTABFL` -> `sql-table:EMPLOYEES` |
|
||||
| `RECEIVES` | Module | Property | `PROCEDURE USING` | 0.8 | `BGTABFL` -> `WK-INPUT-REC` |
|
||||
| `CONTRACTS` | Module | Module | Shared copybook detection | 0.9 | `BGTABFL` -> `BGTABUP` (via `CPSESP`) |
|
||||
|
||||
## Full Annotated Example
|
||||
|
||||
Given this COBOL program:
|
||||
|
||||
```cobol
|
||||
IDENTIFICATION DIVISION.
|
||||
PROGRAM-ID. EMPMAINT.
|
||||
AUTHOR. Development Team.
|
||||
|
||||
ENVIRONMENT DIVISION.
|
||||
INPUT-OUTPUT SECTION.
|
||||
FILE-CONTROL.
|
||||
SELECT EMP-FILE
|
||||
ASSIGN TO "EMPLOYEE.DAT"
|
||||
ORGANIZATION IS INDEXED
|
||||
ACCESS MODE IS DYNAMIC
|
||||
RECORD KEY IS EMP-ID
|
||||
FILE STATUS IS WS-FILE-STATUS.
|
||||
|
||||
DATA DIVISION.
|
||||
FILE SECTION.
|
||||
FD EMP-FILE.
|
||||
01 EMP-RECORD.
|
||||
05 EMP-ID PIC 9(6).
|
||||
05 EMP-NAME PIC X(30).
|
||||
|
||||
WORKING-STORAGE SECTION.
|
||||
01 WS-FLAGS.
|
||||
05 WS-FILE-STATUS PIC X(02).
|
||||
05 WS-EOF-FLAG PIC X(01).
|
||||
88 WS-EOF VALUE "Y".
|
||||
|
||||
LINKAGE SECTION.
|
||||
01 LK-SEARCH-KEY PIC 9(6).
|
||||
|
||||
PROCEDURE DIVISION USING LK-SEARCH-KEY.
|
||||
MAIN-LOGIC SECTION.
|
||||
MAIN-START.
|
||||
PERFORM OPEN-FILE
|
||||
PERFORM PROCESS-RECORDS
|
||||
PERFORM CLOSE-FILE
|
||||
STOP RUN.
|
||||
|
||||
OPEN-FILE.
|
||||
OPEN I-O EMP-FILE.
|
||||
|
||||
PROCESS-RECORDS.
|
||||
MOVE LK-SEARCH-KEY TO EMP-ID
|
||||
EXEC SQL
|
||||
SELECT EMP_SALARY INTO :WS-SALARY
|
||||
FROM EMPLOYEES
|
||||
WHERE EMP_ID = :EMP-ID
|
||||
END-EXEC
|
||||
CALL "EMPREPORT".
|
||||
|
||||
CLOSE-FILE.
|
||||
CLOSE EMP-FILE.
|
||||
```
|
||||
|
||||
The graph produced contains:
|
||||
|
||||
**Nodes:**
|
||||
- `Module`: EMPMAINT (description: `author:Development Team`)
|
||||
- `Namespace`: MAIN-LOGIC
|
||||
- `Function`: MAIN-START, OPEN-FILE, PROCESS-RECORDS, CLOSE-FILE
|
||||
- `Record`: EMP-RECORD, WS-FLAGS, LK-SEARCH-KEY
|
||||
- `Property`: EMP-ID, EMP-NAME, WS-FILE-STATUS, WS-EOF-FLAG
|
||||
- `Const`: WS-EOF (values: Y)
|
||||
- `CodeElement`: SELECT:EMP-FILE, FD:EMP-FILE, sql-table:EMPLOYEES
|
||||
- (COPY imports, if any, would produce File IMPORTS edges)
|
||||
|
||||
**Edges:**
|
||||
- `DEFINES`: File -> all nodes
|
||||
- `CONTAINS`: EMPMAINT -> EMP-RECORD, EMPMAINT -> WS-FLAGS, EMPMAINT -> LK-SEARCH-KEY
|
||||
- `CONTAINS`: EMP-RECORD -> EMP-ID, EMP-RECORD -> EMP-NAME
|
||||
- `CONTAINS`: WS-FLAGS -> WS-FILE-STATUS, WS-FLAGS -> WS-EOF-FLAG
|
||||
- `CONTAINS`: WS-EOF-FLAG -> WS-EOF
|
||||
- `CONTAINS`: FD:EMP-FILE -> EMP-RECORD
|
||||
- `CONTAINS`: SELECT:EMP-FILE -> FD:EMP-FILE
|
||||
- `CALLS`: MAIN-START -> OPEN-FILE, MAIN-START -> PROCESS-RECORDS, MAIN-START -> CLOSE-FILE
|
||||
- `CALLS`: EMPMAINT -> EMPREPORT (external CALL)
|
||||
- `ACCESSES`: EMPMAINT -> sql-table:EMPLOYEES
|
||||
- `RECEIVES`: EMPMAINT -> LK-SEARCH-KEY (PROCEDURE USING)
|
||||
- `RECORD_KEY_OF`: EMP-ID -> SELECT:EMP-FILE
|
||||
- `FILE_STATUS_OF`: WS-FILE-STATUS -> SELECT:EMP-FILE
|
||||
|
||||
## How COBOL Differs from Tree-Sitter Languages
|
||||
|
||||
| Aspect | COBOL | Tree-Sitter Languages |
|
||||
|--------|-------|----------------------|
|
||||
| Node variety | 8 types (Module, Function, Namespace, Record, Property, Const, CodeElement, Constructor) | Typically 4-6 (Function, Class, Method, Interface, Module, Const) |
|
||||
| Domain edges | RECORD_KEY_OF, FILE_STATUS_OF, ACCESSES, RECEIVES, CONTRACTS, REDEFINES | Primarily CALLS, IMPORTS, EXTENDS, IMPLEMENTS |
|
||||
| Data hierarchy | Deep CONTAINS chains (01 -> 05 -> 10 -> 88) | Flat class members |
|
||||
| Cross-program calls | CALL "name" + CICS LINK PROGRAM | Import-based resolution |
|
||||
| Contract detection | Shared COPY copybook between caller/callee | Not applicable |
|
||||
| Metadata | AUTHOR, DATE-WRITTEN on Module | JSDoc/docstring (not indexed) |
|
||||
|
||||
## Source Files
|
||||
|
||||
- `gitnexus/src/core/ingestion/workers/parse-worker.ts` -- `processCobolRegexOnly()`, node/edge emission logic
|
||||
- `gitnexus/src/core/ingestion/pipeline.ts` -- `detectCrossProgamContracts()` for CONTRACTS edges
|
||||
- `gitnexus/src/core/ingestion/cobol-preprocessor.ts` -- `CobolRegexResults` interface (all extracted data)
|
||||
@@ -0,0 +1,232 @@
|
||||
# COBOL Performance and Tuning
|
||||
|
||||
This document covers real-world benchmarks, worker pool configuration, memory management, known limitations, and troubleshooting for COBOL indexing.
|
||||
|
||||
## PROJECT-NAME Benchmark
|
||||
|
||||
The PROJECT-NAME project is a large Italian payroll system written in COBOL. It serves as the primary benchmark for COBOL indexing performance.
|
||||
|
||||
### Input
|
||||
|
||||
| Metric | Value |
|
||||
| --------------------------- | ---------------------------------------------------------------------------- |
|
||||
| Paths scanned | 14,217 |
|
||||
| Parseable files | 13,129 |
|
||||
| Total source size | 224 MB |
|
||||
| Chunks | 12 (at 20 MB budget) |
|
||||
| Copybooks loaded | 2,976 |
|
||||
| Copybooks used in expansion | 2,955 |
|
||||
| Key directories | `s/` (7773 programs), `c/` (3036 copybooks), `wfproc/` (1973 workflow files) |
|
||||
|
||||
### Output
|
||||
|
||||
| Metric | Value |
|
||||
| ---------------------- | ------ |
|
||||
| Graph nodes | 2.79M |
|
||||
| Graph edges | 5.67M |
|
||||
| Clusters (communities) | 16,679 |
|
||||
| Execution flows | 300 |
|
||||
|
||||
### Timing
|
||||
|
||||
| Phase | Duration |
|
||||
| ------------------------------- | ----------------- |
|
||||
| Total | ~251s |
|
||||
| KuzuDB write | 132s |
|
||||
| Full-text search indexing | 6.7s |
|
||||
| Regex extraction (avg per file) | ~1ms |
|
||||
| COPY expansion + deep indexing | Remainder (~112s) |
|
||||
|
||||
### Indexing Command
|
||||
|
||||
```bash
|
||||
cd /path/to/PROJECT-NAME
|
||||
GITNEXUS_COBOL_DIRS=s,c,wfproc GITNEXUS_VERBOSE=1 node --max-old-space-size=8192 \
|
||||
/path/to/gitnexus/dist/cli/index.js analyze --force
|
||||
```
|
||||
|
||||
## Worker Pool Tuning
|
||||
|
||||
### Sub-Batch Size
|
||||
|
||||
The worker pool splits each worker's chunk into sub-batches to bound peak memory per `postMessage` serialization. COBOL repos use a smaller sub-batch size than the default:
|
||||
|
||||
| Parameter | Default | COBOL Mode |
|
||||
| --------------------- | ----------- | ------------------- |
|
||||
| Sub-batch size | 1,500 files | 200 files |
|
||||
| Per sub-batch timeout | 120s | 120s (configurable) |
|
||||
|
||||
**Why 200?** COBOL regex extraction + preprocessing takes ~1ms per file on average, but with COPY expansion and deep indexing the effective time is ~150ms per file. At sub-batch size 1500, that would be ~225s per sub-batch, exceeding the 120s timeout.
|
||||
|
||||
COBOL mode is activated automatically when `GITNEXUS_COBOL_DIRS` is set:
|
||||
|
||||
```typescript
|
||||
// From pipeline.ts
|
||||
const cobolSubBatch = process.env.GITNEXUS_COBOL_DIRS ? 200 : undefined;
|
||||
workerPool = createWorkerPool(workerUrl, undefined, cobolSubBatch);
|
||||
```
|
||||
|
||||
### Worker Count
|
||||
|
||||
Workers default to `min(8, cpus - 1)`. For COBOL repos, this is usually sufficient since regex extraction is CPU-bound but fast. The bottleneck is typically KuzuDB write, not extraction.
|
||||
|
||||
### Timeout Configuration
|
||||
|
||||
| Environment Variable | Default | Purpose |
|
||||
| ------------------------------------ | --------------- | --------------------------------------------------- |
|
||||
| `GITNEXUS_WORKER_TIMEOUT_MS` | 120,000 (2 min) | Per sub-batch processing timeout |
|
||||
| `GITNEXUS_WORKER_STARTUP_TIMEOUT_MS` | 60,000 (1 min) | Worker initialization timeout (tree-sitter loading) |
|
||||
|
||||
For COBOL-only repos, worker startup is faster because tree-sitter native modules are loaded lazily (skipped entirely if only COBOL files are present).
|
||||
|
||||
## Data Item Cap
|
||||
|
||||
### Configuration
|
||||
|
||||
```typescript
|
||||
const MAX_DATA_ITEMS_PER_FILE = 500;
|
||||
```
|
||||
|
||||
This constant appears in both `parse-worker.ts` (worker path) and `parsing-processor.ts` (sequential fallback).
|
||||
|
||||
### Rationale
|
||||
|
||||
Some COBOL programs, especially after COPY expansion, can have 10,000+ data items. At that scale:
|
||||
|
||||
- The in-memory relationship Map (for CONTAINS, REDEFINES, etc.) approaches the V8 16.7M entry limit across thousands of files
|
||||
- KuzuDB write time increases linearly with edge count
|
||||
- Most deep-nested items (level 20+) are rarely queried individually
|
||||
|
||||
### Impact
|
||||
|
||||
The cap truncates data items beyond the 500th in source order. Since 01-level Records appear first in COBOL source, the cap preserves:
|
||||
|
||||
- All 01-level record definitions
|
||||
- The most important 02-49 level items (those closest to the record root)
|
||||
- 88-level conditions associated with early items
|
||||
|
||||
To increase the cap for specific needs, modify the `MAX_DATA_ITEMS_PER_FILE` constant in both files.
|
||||
|
||||
## Memory Management
|
||||
|
||||
### COPY Expansion Memory
|
||||
|
||||
All copybook content is loaded upfront into a Map before chunk processing begins. For PROJECT-NAME:
|
||||
|
||||
- 2,976 copybooks, typically under 100MB total
|
||||
- The Map is shared (read-only) across chunk iterations
|
||||
- Per-chunk, the copybook map is merged with chunk file content (in case a chunk contains copybooks not in the pre-loaded set)
|
||||
- After all chunks are processed, the copybook map is freed (`cobolCopybookContents = undefined`)
|
||||
|
||||
### Chunk Budget
|
||||
|
||||
Source files are grouped into chunks of max 20MB (`CHUNK_BYTE_BUDGET`). Each chunk's lifecycle:
|
||||
|
||||
1. Read file content into memory
|
||||
2. Expand COPY statements (mutates content in-place)
|
||||
3. Dispatch to workers for extraction
|
||||
4. Workers return serialized results
|
||||
5. Merge results into graph
|
||||
6. Chunk content goes out of scope (GC reclaims)
|
||||
|
||||
This ensures only ~20MB of source + ~200-400MB of working memory (ASTs, extracted records, serialization) is active at any time.
|
||||
|
||||
### Shared Warning Deduplication
|
||||
|
||||
The `warnedCircular` set (used by the COPY expansion engine) is shared across all files in a chunk. This prevents the same circular copybook warning (e.g., `ANAZI includes itself`) from being logged thousands of times.
|
||||
|
||||
## Known Limitations
|
||||
|
||||
| Limitation | Impact | Workaround |
|
||||
| ---------------------------------------- | --------------------------------------------------------------------------------------------------------------- | ---------------------------------------------------------- |
|
||||
| tree-sitter-cobol hangs on ~5% of files | Cannot use tree-sitter for COBOL | Regex-only extraction (current approach) |
|
||||
| Data item cap (500/file) | May miss deeply nested items in large programs | Increase `MAX_DATA_ITEMS_PER_FILE` in source |
|
||||
| Circular copybooks (ANAZI, ANDIP, QDIPE) | Self-referential includes cannot be expanded | Detected and skipped with warning |
|
||||
| wfproc/ files may not be pure COBOL | Workflow files may produce extraction noise | Exclude `wfproc` from `GITNEXUS_COBOL_DIRS` if problematic |
|
||||
| No MOVE DATA_FLOW edges yet | Data flow between variables not in graph | Reserved for future release |
|
||||
| Continuation line handling | Some complex multi-line continuations (especially in string literals spanning 3+ lines) may not merge correctly | Known edge case; affects <0.1% of lines |
|
||||
| Single-line EXEC blocks | `EXEC SQL SELECT ... END-EXEC` on one line is handled, but pathological nesting is not | Extremely rare in practice |
|
||||
| Extension case sensitivity | `.GNM` and `.gnm` are matched differently | Use the exact case from the codebase |
|
||||
|
||||
## Troubleshooting
|
||||
|
||||
### "COPY expansion failed"
|
||||
|
||||
```
|
||||
[pipeline] COPY expansion failed for s/BGTABFL: Cannot read properties of null
|
||||
```
|
||||
|
||||
**Cause:** A copybook referenced by a COPY statement cannot be found.
|
||||
|
||||
**Fix:**
|
||||
|
||||
1. Verify `GITNEXUS_COBOL_DIRS` includes the directory containing copybooks (typically `c`)
|
||||
2. Check that copybook filenames match the COPY target (case-insensitive, after stripping extensions)
|
||||
3. Ensure copybook files are not in `.gitignore`
|
||||
|
||||
### Worker sub-batch timeout
|
||||
|
||||
```
|
||||
Worker 3 sub-batch timed out after 120s (chunk: 200 items)
|
||||
```
|
||||
|
||||
**Cause:** A sub-batch took longer than the timeout. Typically happens when one file is extremely large (50,000+ lines after COPY expansion).
|
||||
|
||||
**Fix:** Increase the timeout:
|
||||
|
||||
```bash
|
||||
GITNEXUS_WORKER_TIMEOUT_MS=300000 gitnexus analyze
|
||||
```
|
||||
|
||||
### Memory errors (heap out of memory)
|
||||
|
||||
```
|
||||
FATAL ERROR: CALL_AND_RETRY_LAST Allocation failed - JavaScript heap out of memory
|
||||
```
|
||||
|
||||
**Fix:** Increase Node.js heap size:
|
||||
|
||||
```bash
|
||||
node --max-old-space-size=16384 /path/to/gitnexus/dist/cli/index.js analyze
|
||||
```
|
||||
|
||||
For very large repos (>500MB source), consider `--max-old-space-size=32768`.
|
||||
|
||||
### Concurrent analyze corruption
|
||||
|
||||
**Rule:** Only ONE `gitnexus analyze` process should run at a time per repository. Concurrent writes to KuzuDB corrupt the database.
|
||||
|
||||
If corruption occurs:
|
||||
|
||||
```bash
|
||||
# Remove the KuzuDB directory and re-index
|
||||
rm -rf .gitnexus/kuzu
|
||||
gitnexus analyze --force
|
||||
```
|
||||
|
||||
### Slow KuzuDB write phase
|
||||
|
||||
The KuzuDB write phase (132s for PROJECT-NAME) is the bottleneck for large COBOL repos. This is proportional to the number of nodes and edges being written. Reducing `MAX_DATA_ITEMS_PER_FILE` or excluding non-essential directories from `GITNEXUS_COBOL_DIRS` can help.
|
||||
|
||||
### Verbose output
|
||||
|
||||
Enable verbose logging to see per-phase timing and statistics:
|
||||
|
||||
```bash
|
||||
GITNEXUS_VERBOSE=1 gitnexus analyze
|
||||
```
|
||||
|
||||
This outputs:
|
||||
|
||||
- Scan statistics (paths, parseable files, chunk count)
|
||||
- Worker pool configuration (worker count, sub-batch size)
|
||||
- COPY expansion statistics (copybooks loaded, files expanded)
|
||||
- Community and process detection results
|
||||
- Contract detection results
|
||||
|
||||
## Source Files
|
||||
|
||||
- `gitnexus/src/core/ingestion/workers/worker-pool.ts` -- `DEFAULT_SUB_BATCH_SIZE`, `SUB_BATCH_TIMEOUT_MS`, `WORKER_STARTUP_TIMEOUT_MS`
|
||||
- `gitnexus/src/core/ingestion/pipeline.ts` -- `CHUNK_BYTE_BUDGET`, COBOL sub-batch configuration, chunk lifecycle
|
||||
- `gitnexus/src/core/ingestion/workers/parse-worker.ts` -- `MAX_DATA_ITEMS_PER_FILE`, `processCobolRegexOnly()`
|
||||
- `gitnexus/src/core/ingestion/parsing-processor.ts` -- Sequential fallback `MAX_DATA_ITEMS_PER_FILE`
|
||||
@@ -0,0 +1,186 @@
|
||||
# COBOL Regex Extraction
|
||||
|
||||
The `extractCobolSymbolsWithRegex()` function in `cobol-preprocessor.ts` performs single-pass, state-machine-driven extraction of all COBOL symbols. This document describes the state machine, line processing flow, and every regex pattern used.
|
||||
|
||||
## State Machine: Division Tracking
|
||||
|
||||
The extractor tracks which COBOL division is currently being processed. Division transitions are detected by the `RE_DIVISION` pattern.
|
||||
|
||||
```mermaid
|
||||
stateDiagram-v2
|
||||
[*] --> null : Start of file
|
||||
null --> identification : IDENTIFICATION DIVISION
|
||||
identification --> environment : ENVIRONMENT DIVISION
|
||||
environment --> data : DATA DIVISION
|
||||
data --> procedure : PROCEDURE DIVISION
|
||||
|
||||
note right of identification
|
||||
Extracts: PROGRAM-ID, AUTHOR, DATE-WRITTEN
|
||||
end note
|
||||
note right of environment
|
||||
Extracts: SELECT ... ASSIGN ... (file declarations)
|
||||
end note
|
||||
note right of data
|
||||
Extracts: FD entries, data items (01-77, 88), COPY
|
||||
end note
|
||||
note right of procedure
|
||||
Extracts: paragraphs, sections, PERFORM, CALL,
|
||||
ENTRY, MOVE, EXEC SQL/CICS
|
||||
end note
|
||||
```
|
||||
|
||||
## State Machine: Data Section Tracking
|
||||
|
||||
Within the DATA DIVISION, a secondary state machine tracks the current section to tag data items with their origin.
|
||||
|
||||
```mermaid
|
||||
stateDiagram-v2
|
||||
[*] --> unknown : DATA DIVISION entered
|
||||
unknown --> working_storage : WORKING-STORAGE SECTION
|
||||
unknown --> linkage : LINKAGE SECTION
|
||||
unknown --> file : FILE SECTION
|
||||
unknown --> local_storage : LOCAL-STORAGE SECTION
|
||||
working_storage --> linkage : LINKAGE SECTION
|
||||
working_storage --> file : FILE SECTION
|
||||
linkage --> working_storage : WORKING-STORAGE SECTION
|
||||
file --> working_storage : WORKING-STORAGE SECTION
|
||||
file --> linkage : LINKAGE SECTION
|
||||
local_storage --> working_storage : WORKING-STORAGE SECTION
|
||||
```
|
||||
|
||||
Within the ENVIRONMENT DIVISION, the `currentEnvSection` tracks whether we are in `INPUT-OUTPUT` or `CONFIGURATION` section. SELECT statement accumulation only occurs in `INPUT-OUTPUT`.
|
||||
|
||||
## Line Processing Flow
|
||||
|
||||
Each raw source line goes through this pipeline:
|
||||
|
||||
```
|
||||
Raw line
|
||||
|
|
||||
v
|
||||
Length < 7? ---------> Skip (flush pending if any)
|
||||
|
|
||||
v
|
||||
Indicator col 7
|
||||
|
|
||||
+-- '*' or '/' -----> Comment: skip entirely
|
||||
|
|
||||
+-- '-' ------------> Continuation: append to pending line
|
||||
|
|
||||
+-- other ----------> Normal: flush pending, strip inline comments (|),
|
||||
buffer as new pending logical line
|
||||
```
|
||||
|
||||
After all lines are processed, the final pending line is flushed, along with any accumulated SELECT statement.
|
||||
|
||||
### Inline Comment Stripping
|
||||
|
||||
Enterprise COBOL (particularly Italian dialect) uses the pipe character `|` as an inline comment marker. Everything from `|` to end of line is stripped before processing.
|
||||
|
||||
### Patch Marker Handling
|
||||
|
||||
The `preprocessCobolSource()` function (run before extraction in the worker) replaces non-standard content in columns 1-6. Standard COBOL expects spaces or digit sequence numbers in this area. If any letter or `#` character is found, the entire sequence area is replaced with 6 spaces:
|
||||
|
||||
```
|
||||
Before: mzADD MOVE WK-AMT TO WK-TOTAL
|
||||
After: MOVE WK-AMT TO WK-TOTAL
|
||||
```
|
||||
|
||||
This preserves exact line count for position mapping.
|
||||
|
||||
## Regex Pattern Reference
|
||||
|
||||
All patterns are compiled once as module-level constants and reused across calls.
|
||||
|
||||
### Division and Section Detection
|
||||
|
||||
| Constant | Pattern | Purpose | Example Match |
|
||||
|----------|---------|---------|---------------|
|
||||
| `RE_DIVISION` | `\b(IDENTIFICATION\|ENVIRONMENT\|DATA\|PROCEDURE)\s+DIVISION\b` | Division boundary | `PROCEDURE DIVISION` |
|
||||
| `RE_SECTION` | `\b(WORKING-STORAGE\|LINKAGE\|FILE\|LOCAL-STORAGE\|INPUT-OUTPUT\|CONFIGURATION)\s+SECTION\b` | Section boundary | `WORKING-STORAGE SECTION` |
|
||||
|
||||
### IDENTIFICATION DIVISION
|
||||
|
||||
| Constant | Pattern | Purpose | Example Match |
|
||||
|----------|---------|---------|---------------|
|
||||
| `RE_PROGRAM_ID` | `\bPROGRAM-ID\.\s*([A-Z][A-Z0-9-]*)` | Program name | `PROGRAM-ID. BGTABFL` |
|
||||
| `RE_AUTHOR` | `^\s+AUTHOR\.\s*(.+)` | Author metadata | `AUTHOR. D. Smith` |
|
||||
| `RE_DATE_WRITTEN` | `^\s+DATE-WRITTEN\.\s*(.+)` | Date metadata | `DATE-WRITTEN. 2024-01-15` |
|
||||
|
||||
### ENVIRONMENT DIVISION
|
||||
|
||||
| Constant | Pattern | Purpose | Example Match |
|
||||
|----------|---------|---------|---------------|
|
||||
| `RE_SELECT_START` | `\bSELECT\s+([A-Z][A-Z0-9-]+)` | File SELECT start | `SELECT MASTER-FILE` |
|
||||
|
||||
SELECT statements are accumulated across multiple lines until a period terminator is found, then parsed for ASSIGN, ORGANIZATION, ACCESS, RECORD KEY, and FILE STATUS clauses.
|
||||
|
||||
### DATA DIVISION
|
||||
|
||||
| Constant | Pattern | Purpose | Example Match |
|
||||
|----------|---------|---------|---------------|
|
||||
| `RE_FD` | `^\s+FD\s+([A-Z][A-Z0-9-]+)` | File description | `FD MASTER-FILE` |
|
||||
| `RE_DATA_ITEM` | `^\s+(\d{1,2})\s+([A-Z][A-Z0-9-]+)\s*(.*)` | Data item (01-77) | `05 WK-NAME PIC X(30)` |
|
||||
| `RE_ANONYMOUS_REDEFINES` | `^\s+(\d{1,2})\s+REDEFINES\s+([A-Z][A-Z0-9-]+)` | Anonymous REDEFINES | `01 REDEFINES WK-REC` |
|
||||
| `RE_88_LEVEL` | `^\s+88\s+([A-Z][A-Z0-9-]+)\s+VALUES?\s+(?:ARE\s+)?(.+)` | Condition name | `88 WK-ACTIVE VALUE "Y"` |
|
||||
|
||||
The trailing clauses of `RE_DATA_ITEM` are parsed by `parseDataItemClauses()` for PIC, USAGE, OCCURS, and REDEFINES.
|
||||
|
||||
### PROCEDURE DIVISION
|
||||
|
||||
| Constant | Pattern | Purpose | Example Match |
|
||||
|----------|---------|---------|---------------|
|
||||
| `RE_PROC_SECTION` | `^ ([A-Z][A-Z0-9-]+)\s+SECTION\.\s*$` | Procedure section header | ` MAIN-LOGIC SECTION.` |
|
||||
| `RE_PROC_PARAGRAPH` | `^ ([A-Z][A-Z0-9-]+)\.\s*$` | Paragraph header | ` PROCESS-RECORD.` |
|
||||
| `RE_PERFORM` | `\bPERFORM\s+([A-Z][A-Z0-9-]+)(?:\s+THRU\s+([A-Z][A-Z0-9-]+))?` | PERFORM call | `PERFORM CALC-TAX THRU CALC-TAX-EXIT` |
|
||||
| `RE_PROC_USING` | `\bPROCEDURE\s+DIVISION\s+USING\s+([\s\S]*?)(?:\.\|$)` | USING parameters | `PROCEDURE DIVISION USING WK-PARAM` |
|
||||
| `RE_ENTRY` | `\bENTRY\s+"([^"]+)"(?:\s+USING\s+([\s\S]*?))?(?:\.\|$)` | ENTRY point | `ENTRY "SUBPROG" USING WK-DATA` |
|
||||
| `RE_MOVE` | `\bMOVE\s+(CORRESPONDING\s+)?([A-Z][A-Z0-9-]+)\s+TO\s+([A-Z][A-Z0-9-]+)` | MOVE statement | `MOVE WK-NAME TO OUT-NAME` |
|
||||
|
||||
Note: `RE_PROC_SECTION` and `RE_PROC_PARAGRAPH` require exactly 7 spaces of leading indentation (COBOL area A starting at column 8). This is the standard COBOL paragraph indentation.
|
||||
|
||||
### All-Division Patterns
|
||||
|
||||
These patterns are checked regardless of current division:
|
||||
|
||||
| Constant | Pattern | Purpose | Example Match |
|
||||
|----------|---------|---------|---------------|
|
||||
| `RE_CALL` | `\bCALL\s+"([^"]+)"` | External program call | `CALL "BGTABUP"` |
|
||||
| `RE_COPY_UNQUOTED` | `\bCOPY\s+([A-Z][A-Z0-9-]+)(?:\s\|\.)` | COPY (unquoted) | `COPY CPSESP.` |
|
||||
| `RE_COPY_QUOTED` | `\bCOPY\s+"([^"]+)"(?:\s\|\.)` | COPY (quoted) | `COPY "WORKGRID.CPY".` |
|
||||
|
||||
### EXEC Block Patterns
|
||||
|
||||
| Constant | Pattern | Purpose | Example Match |
|
||||
|----------|---------|---------|---------------|
|
||||
| `RE_EXEC_SQL_START` | `\bEXEC\s+SQL\b` | Start of EXEC SQL block | `EXEC SQL` |
|
||||
| `RE_EXEC_CICS_START` | `\bEXEC\s+CICS\b` | Start of EXEC CICS block | `EXEC CICS` |
|
||||
| `RE_END_EXEC` | `\bEND-EXEC\b` | End of EXEC block | `END-EXEC` |
|
||||
|
||||
EXEC blocks accumulate all lines between `EXEC SQL/CICS` and `END-EXEC`, then delegate to `parseExecSqlBlock()` or `parseExecCicsBlock()` for detailed extraction.
|
||||
|
||||
## Excluded Paragraph Names
|
||||
|
||||
The following names are excluded from paragraph detection to avoid false positives from division/section headers:
|
||||
|
||||
```
|
||||
DECLARATIVES, END, PROCEDURE, IDENTIFICATION,
|
||||
ENVIRONMENT, DATA, WORKING-STORAGE, LINKAGE,
|
||||
FILE, LOCAL-STORAGE, COMMUNICATION, REPORT,
|
||||
SCREEN, INPUT-OUTPUT, CONFIGURATION
|
||||
```
|
||||
|
||||
Additionally, paragraph candidates containing `DIVISION` or `SECTION` as substrings are excluded.
|
||||
|
||||
## MOVE Skip List (Figurative Constants)
|
||||
|
||||
MOVE statements where the source is a figurative constant are skipped:
|
||||
|
||||
```
|
||||
SPACES, ZEROS, ZEROES, LOW-VALUES, LOW-VALUE,
|
||||
HIGH-VALUES, HIGH-VALUE, QUOTES, QUOTE, ALL
|
||||
```
|
||||
|
||||
## Source Files
|
||||
|
||||
- `gitnexus/src/core/ingestion/cobol-preprocessor.ts` -- `preprocessCobolSource()`, `extractCobolSymbolsWithRegex()`, all regex constants
|
||||
+2
-2
@@ -148,7 +148,7 @@ Each mode has a `system_{mode}.jinja` + `instance_{mode}.jinja` pair. The agent
|
||||
|
||||
1. Docker container starts with SWE-bench instance (repo at specific commit)
|
||||
2. **GitNexus setup**: Node.js + gitnexus installed, `gitnexus analyze` runs (or restores from cache)
|
||||
3. **Eval-server starts**: `gitnexus eval-server` daemon (persistent HTTP server, keeps KuzuDB warm)
|
||||
3. **Eval-server starts**: `gitnexus eval-server` daemon (persistent HTTP server, keeps LadybugDB warm)
|
||||
4. **Standalone tool scripts installed** in `/usr/local/bin/` — works with `subprocess.run` (no `.bashrc` needed)
|
||||
5. Agent runs with the configured model + system prompt + GitNexus tools
|
||||
6. Agent's patch is extracted as a git diff
|
||||
@@ -167,7 +167,7 @@ Each tool script in `/usr/local/bin/` is standalone — no sourcing, no env inhe
|
||||
### Eval-server
|
||||
|
||||
The eval-server is a lightweight HTTP daemon that:
|
||||
- Keeps KuzuDB warm in memory (no cold start per tool call)
|
||||
- Keeps LadybugDB warm in memory (no cold start per tool call)
|
||||
- Returns LLM-friendly text (not raw JSON — saves tokens)
|
||||
- Includes next-step hints to guide tool chaining (query → context → impact → fix)
|
||||
- Auto-shuts down after idle timeout
|
||||
|
||||
@@ -2,7 +2,7 @@
|
||||
MCP Bridge for GitNexus
|
||||
|
||||
Starts the GitNexus MCP server as a subprocess and provides a Python interface
|
||||
to call MCP tools. Used by the bash wrapper scripts and the augmentation layer.
|
||||
to call MCP tools. Used by the bash wrapper scripts and the augmentation layer..
|
||||
|
||||
The bridge communicates with the MCP server via stdio using the JSON-RPC protocol.
|
||||
"""
|
||||
|
||||
@@ -0,0 +1,13 @@
|
||||
model: deepseek-ai/deepseek-chat
|
||||
provider: openrouter
|
||||
cost:
|
||||
input: 0.14 # per 1M tokens
|
||||
output: 0.28 # per 1M tokens
|
||||
|
||||
# Native DeepSeek API (direct)
|
||||
api_key: null
|
||||
base_url: null
|
||||
|
||||
# For OpenRouter, uncomment below and comment out direct config above
|
||||
# api_key: \${OPENROUTER_API_KEY}
|
||||
# base_url: https://openrouter.ai/api/v1
|
||||
@@ -0,0 +1,15 @@
|
||||
model: deepseek-ai/DeepSeek-V3
|
||||
provider: openrouter
|
||||
cost:
|
||||
input: 0.27 # per 1M tokens
|
||||
output: 1.10 # per 1M tokens
|
||||
|
||||
# Native DeepSeek API (direct)
|
||||
# Get your API key at: https://platform.deepseek.com/
|
||||
# Or use OpenRouter with: OPENROUTER_API_KEY
|
||||
api_key: null
|
||||
base_url: null
|
||||
|
||||
# For OpenRouter, uncomment below and comment out direct config above
|
||||
# api_key: \${OPENROUTER_API_KEY}
|
||||
# base_url: https://openrouter.ai/api/v1
|
||||
@@ -1,11 +1,11 @@
|
||||
{
|
||||
"name": "gitnexus",
|
||||
"description": "Code intelligence powered by a knowledge graph. Provides execution flow tracing, blast radius analysis, and augmented search across your codebase.",
|
||||
"version": "1.3.3",
|
||||
"version": "1.3.6",
|
||||
"author": {
|
||||
"name": "GitNexus"
|
||||
},
|
||||
"homepage": "https://github.com/nicosxt/gitnexus",
|
||||
"repository": "https://github.com/nicosxt/gitnexus",
|
||||
"homepage": "https://github.com/abhigyanpatwari/GitNexus",
|
||||
"repository": "https://github.com/abhigyanpatwari/GitNexus",
|
||||
"keywords": ["code-intelligence", "knowledge-graph", "mcp", "static-analysis"]
|
||||
}
|
||||
|
||||
@@ -2,8 +2,10 @@
|
||||
/**
|
||||
* GitNexus Claude Code Plugin Hook
|
||||
*
|
||||
* PreToolUse handler — intercepts Grep/Glob/Bash searches
|
||||
* and augments with graph context from the GitNexus index.
|
||||
* PreToolUse — intercepts Grep/Glob/Bash searches and augments
|
||||
* with graph context from the GitNexus index.
|
||||
* PostToolUse — detects stale index after git mutations and notifies
|
||||
* the agent to reindex.
|
||||
*
|
||||
* NOTE: SessionStart hooks are broken on Windows (Claude Code bug #23576).
|
||||
* Session context is injected via CLAUDE.md / skills instead.
|
||||
@@ -26,19 +28,19 @@ function readInput() {
|
||||
}
|
||||
|
||||
/**
|
||||
* Check if a directory (or ancestor) has a .gitnexus index.
|
||||
* Find the .gitnexus directory by walking up from startDir.
|
||||
* Returns the path to .gitnexus/ or null if not found.
|
||||
*/
|
||||
function findGitNexusIndex(startDir) {
|
||||
function findGitNexusDir(startDir) {
|
||||
let dir = startDir || process.cwd();
|
||||
for (let i = 0; i < 5; i++) {
|
||||
if (fs.existsSync(path.join(dir, '.gitnexus'))) {
|
||||
return true;
|
||||
}
|
||||
const candidate = path.join(dir, '.gitnexus');
|
||||
if (fs.existsSync(candidate)) return candidate;
|
||||
const parent = path.dirname(dir);
|
||||
if (parent === dir) break;
|
||||
dir = parent;
|
||||
}
|
||||
return false;
|
||||
return null;
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -83,64 +85,146 @@ function extractPattern(toolName, toolInput) {
|
||||
return null;
|
||||
}
|
||||
|
||||
/**
|
||||
* Spawn a gitnexus CLI command synchronously.
|
||||
* Detects binary on PATH once, then runs exactly once.
|
||||
*
|
||||
* SECURITY: Never use shell: true with user-controlled arguments.
|
||||
* On Windows, invoke gitnexus.cmd directly (no shell needed).
|
||||
*/
|
||||
function runGitNexusCli(args, cwd, timeout) {
|
||||
const isWin = process.platform === 'win32';
|
||||
|
||||
// Detect whether 'gitnexus' is on PATH (cheap check, no execution)
|
||||
let useDirectBinary = false;
|
||||
try {
|
||||
const which = spawnSync(
|
||||
isWin ? 'where' : 'which', ['gitnexus'],
|
||||
{ encoding: 'utf-8', timeout: 3000, stdio: ['pipe', 'pipe', 'pipe'] }
|
||||
);
|
||||
useDirectBinary = which.status === 0;
|
||||
} catch { /* not on PATH */ }
|
||||
|
||||
if (useDirectBinary) {
|
||||
return spawnSync(
|
||||
isWin ? 'gitnexus.cmd' : 'gitnexus', args,
|
||||
{ encoding: 'utf-8', timeout, cwd, stdio: ['pipe', 'pipe', 'pipe'] }
|
||||
);
|
||||
}
|
||||
// npx fallback needs shell on Windows since npx is a .cmd script
|
||||
return spawnSync(
|
||||
isWin ? 'npx.cmd' : 'npx', ['-y', 'gitnexus', ...args],
|
||||
{ encoding: 'utf-8', timeout: timeout + 5000, cwd, stdio: ['pipe', 'pipe', 'pipe'] }
|
||||
);
|
||||
}
|
||||
|
||||
/**
|
||||
* Emit a hook response with additional context for the agent.
|
||||
*/
|
||||
function sendHookResponse(hookEventName, message) {
|
||||
console.log(JSON.stringify({
|
||||
hookSpecificOutput: { hookEventName, additionalContext: message }
|
||||
}));
|
||||
}
|
||||
|
||||
/**
|
||||
* PreToolUse handler — augment searches with graph context.
|
||||
*/
|
||||
function handlePreToolUse(input) {
|
||||
const cwd = input.cwd || process.cwd();
|
||||
if (!path.isAbsolute(cwd)) return;
|
||||
if (!findGitNexusDir(cwd)) return;
|
||||
|
||||
const toolName = input.tool_name || '';
|
||||
const toolInput = input.tool_input || {};
|
||||
|
||||
if (toolName !== 'Grep' && toolName !== 'Glob' && toolName !== 'Bash') return;
|
||||
|
||||
const pattern = extractPattern(toolName, toolInput);
|
||||
if (!pattern || pattern.length < 3) return;
|
||||
|
||||
let result = '';
|
||||
try {
|
||||
const child = runGitNexusCli(['augment', '--', pattern], cwd, 7000);
|
||||
if (!child.error && child.status === 0) {
|
||||
result = child.stderr || '';
|
||||
}
|
||||
} catch { /* graceful failure */ }
|
||||
|
||||
if (result && result.trim()) {
|
||||
sendHookResponse('PreToolUse', result.trim());
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* PostToolUse handler — detect index staleness after git mutations.
|
||||
*
|
||||
* Instead of spawning a full `gitnexus analyze` synchronously (which blocks
|
||||
* the agent for up to 120s and risks LadybugDB corruption on timeout), we do a
|
||||
* lightweight staleness check: compare `git rev-parse HEAD` against the
|
||||
* lastCommit stored in `.gitnexus/meta.json`. If they differ, notify the
|
||||
* agent so it can decide when to reindex.
|
||||
*/
|
||||
function handlePostToolUse(input) {
|
||||
const toolName = input.tool_name || '';
|
||||
if (toolName !== 'Bash') return;
|
||||
|
||||
const command = (input.tool_input || {}).command || '';
|
||||
if (!/\bgit\s+(commit|merge|rebase|cherry-pick|pull)(\s|$)/.test(command)) return;
|
||||
|
||||
// Only proceed if the command succeeded
|
||||
const toolOutput = input.tool_output || {};
|
||||
if (toolOutput.exit_code !== undefined && toolOutput.exit_code !== 0) return;
|
||||
|
||||
const cwd = input.cwd || process.cwd();
|
||||
if (!path.isAbsolute(cwd)) return;
|
||||
const gitNexusDir = findGitNexusDir(cwd);
|
||||
if (!gitNexusDir) return;
|
||||
|
||||
// Compare HEAD against last indexed commit — skip if unchanged
|
||||
let currentHead = '';
|
||||
try {
|
||||
const headResult = spawnSync('git', ['rev-parse', 'HEAD'], {
|
||||
encoding: 'utf-8', timeout: 3000, cwd, stdio: ['pipe', 'pipe', 'pipe'],
|
||||
});
|
||||
currentHead = (headResult.stdout || '').trim();
|
||||
} catch { return; }
|
||||
|
||||
if (!currentHead) return;
|
||||
|
||||
let lastCommit = '';
|
||||
let hadEmbeddings = false;
|
||||
try {
|
||||
const meta = JSON.parse(fs.readFileSync(path.join(gitNexusDir, 'meta.json'), 'utf-8'));
|
||||
lastCommit = meta.lastCommit || '';
|
||||
hadEmbeddings = (meta.stats && meta.stats.embeddings > 0);
|
||||
} catch { /* no meta — treat as stale */ }
|
||||
|
||||
// If HEAD matches last indexed commit, no reindex needed
|
||||
if (currentHead && currentHead === lastCommit) return;
|
||||
|
||||
const analyzeCmd = `npx gitnexus analyze${hadEmbeddings ? ' --embeddings' : ''}`;
|
||||
sendHookResponse('PostToolUse',
|
||||
`GitNexus index is stale (last indexed: ${lastCommit ? lastCommit.slice(0, 7) : 'never'}). ` +
|
||||
`Run \`${analyzeCmd}\` to update the knowledge graph.`
|
||||
);
|
||||
}
|
||||
|
||||
// Dispatch map for hook events
|
||||
const handlers = {
|
||||
PreToolUse: handlePreToolUse,
|
||||
PostToolUse: handlePostToolUse,
|
||||
};
|
||||
|
||||
function main() {
|
||||
try {
|
||||
const input = readInput();
|
||||
const hookEvent = input.hook_event_name || '';
|
||||
|
||||
if (hookEvent !== 'PreToolUse') return;
|
||||
|
||||
const cwd = input.cwd || process.cwd();
|
||||
if (!findGitNexusIndex(cwd)) return;
|
||||
|
||||
const toolName = input.tool_name || '';
|
||||
const toolInput = input.tool_input || {};
|
||||
|
||||
if (toolName !== 'Grep' && toolName !== 'Glob' && toolName !== 'Bash') return;
|
||||
|
||||
const pattern = extractPattern(toolName, toolInput);
|
||||
if (!pattern || pattern.length < 3) return;
|
||||
|
||||
// augment CLI writes result to stderr (KuzuDB's native module captures
|
||||
// stdout fd at OS level, making it unusable in subprocess contexts).
|
||||
let result = '';
|
||||
|
||||
// Try direct gitnexus binary first (faster if globally installed)
|
||||
try {
|
||||
const child = spawnSync(
|
||||
'gitnexus',
|
||||
['augment', pattern],
|
||||
{ encoding: 'utf-8', timeout: 8000, cwd, stdio: ['pipe', 'pipe', 'pipe'] }
|
||||
);
|
||||
if (child.status === 0 && child.stderr && child.stderr.trim()) {
|
||||
result = child.stderr;
|
||||
}
|
||||
} catch { /* not on PATH */ }
|
||||
|
||||
// Fallback to npx if direct binary didn't produce output
|
||||
if (!result || !result.trim()) {
|
||||
try {
|
||||
const child = spawnSync(
|
||||
'npx',
|
||||
['-y', 'gitnexus', 'augment', pattern],
|
||||
{ encoding: 'utf-8', timeout: 15000, cwd, stdio: ['pipe', 'pipe', 'pipe'] }
|
||||
);
|
||||
if (child.status === 0 && child.stderr && child.stderr.trim()) {
|
||||
result = child.stderr;
|
||||
}
|
||||
} catch { /* graceful failure */ }
|
||||
const handler = handlers[input.hook_event_name || ''];
|
||||
if (handler) handler(input);
|
||||
} catch (err) {
|
||||
if (process.env.GITNEXUS_DEBUG) {
|
||||
console.error('GitNexus hook error:', (err.message || '').slice(0, 200));
|
||||
}
|
||||
|
||||
if (result && result.trim()) {
|
||||
console.log(JSON.stringify({
|
||||
hookSpecificOutput: {
|
||||
hookEventName: 'PreToolUse',
|
||||
additionalContext: result.trim()
|
||||
}
|
||||
}));
|
||||
}
|
||||
} catch {
|
||||
// Graceful failure
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -12,6 +12,19 @@
|
||||
}
|
||||
]
|
||||
}
|
||||
],
|
||||
"PostToolUse": [
|
||||
{
|
||||
"matcher": "Bash",
|
||||
"hooks": [
|
||||
{
|
||||
"type": "command",
|
||||
"command": "node ${CLAUDE_PLUGIN_ROOT}/hooks/gitnexus-hook.js",
|
||||
"timeout": 10,
|
||||
"statusMessage": "Checking GitNexus index freshness..."
|
||||
}
|
||||
]
|
||||
}
|
||||
]
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,163 @@
|
||||
---
|
||||
name: gitnexus-pr-review
|
||||
description: "Use when the user wants to review a pull request, understand what a PR changes, assess risk of merging, or check for missing test coverage. Examples: \"Review this PR\", \"What does PR #42 change?\", \"Is this PR safe to merge?\""
|
||||
---
|
||||
|
||||
# PR Review with GitNexus
|
||||
|
||||
## When to Use
|
||||
|
||||
- "Review this PR"
|
||||
- "What does PR #42 change?"
|
||||
- "Is this safe to merge?"
|
||||
- "What's the blast radius of this PR?"
|
||||
- "Are there missing tests for this PR?"
|
||||
- Reviewing someone else's code changes before merge
|
||||
|
||||
## Workflow
|
||||
|
||||
```
|
||||
1. gh pr diff <number> → Get the raw diff
|
||||
2. gitnexus_detect_changes({scope: "compare", base_ref: "main"}) → Map diff to affected flows
|
||||
3. For each changed symbol:
|
||||
gitnexus_impact({target: "<symbol>", direction: "upstream"}) → Blast radius per change
|
||||
4. gitnexus_context({name: "<key symbol>"}) → Understand callers/callees
|
||||
5. READ gitnexus://repo/{name}/processes → Check affected execution flows
|
||||
6. Summarize findings with risk assessment
|
||||
```
|
||||
|
||||
> If "Index is stale" → run `npx gitnexus analyze` in terminal before reviewing.
|
||||
|
||||
## Checklist
|
||||
|
||||
```
|
||||
- [ ] Fetch PR diff (gh pr diff or git diff base...head)
|
||||
- [ ] gitnexus_detect_changes to map changes to affected execution flows
|
||||
- [ ] gitnexus_impact on each non-trivial changed symbol
|
||||
- [ ] Review d=1 items (WILL BREAK) — are callers updated?
|
||||
- [ ] gitnexus_context on key changed symbols to understand full picture
|
||||
- [ ] Check if affected processes have test coverage
|
||||
- [ ] Assess overall risk level
|
||||
- [ ] Write review summary with findings
|
||||
```
|
||||
|
||||
## Review Dimensions
|
||||
|
||||
| Dimension | How GitNexus Helps |
|
||||
| --- | --- |
|
||||
| **Correctness** | `context` shows callers — are they all compatible with the change? |
|
||||
| **Blast radius** | `impact` shows d=1/d=2/d=3 dependents — anything missed? |
|
||||
| **Completeness** | `detect_changes` shows all affected flows — are they all handled? |
|
||||
| **Test coverage** | `impact({includeTests: true})` shows which tests touch changed code |
|
||||
| **Breaking changes** | d=1 upstream items that aren't updated in the PR = potential breakage |
|
||||
|
||||
## Risk Assessment
|
||||
|
||||
| Signal | Risk |
|
||||
| --- | --- |
|
||||
| Changes touch <3 symbols, 0-1 processes | LOW |
|
||||
| Changes touch 3-10 symbols, 2-5 processes | MEDIUM |
|
||||
| Changes touch >10 symbols or many processes | HIGH |
|
||||
| Changes touch auth, payments, or data integrity code | CRITICAL |
|
||||
| d=1 callers exist outside the PR diff | Potential breakage — flag it |
|
||||
|
||||
## Tools
|
||||
|
||||
**gitnexus_detect_changes** — map PR diff to affected execution flows:
|
||||
|
||||
```
|
||||
gitnexus_detect_changes({scope: "compare", base_ref: "main"})
|
||||
|
||||
→ Changed: 8 symbols in 4 files
|
||||
→ Affected processes: CheckoutFlow, RefundFlow, WebhookHandler
|
||||
→ Risk: MEDIUM
|
||||
```
|
||||
|
||||
**gitnexus_impact** — blast radius per changed symbol:
|
||||
|
||||
```
|
||||
gitnexus_impact({target: "validatePayment", direction: "upstream"})
|
||||
|
||||
→ d=1 (WILL BREAK):
|
||||
- processCheckout (src/checkout.ts:42) [CALLS, 100%]
|
||||
- webhookHandler (src/webhooks.ts:15) [CALLS, 100%]
|
||||
|
||||
→ d=2 (LIKELY AFFECTED):
|
||||
- checkoutRouter (src/routes/checkout.ts:22) [CALLS, 95%]
|
||||
```
|
||||
|
||||
**gitnexus_impact with tests** — check test coverage:
|
||||
|
||||
```
|
||||
gitnexus_impact({target: "validatePayment", direction: "upstream", includeTests: true})
|
||||
|
||||
→ Tests that cover this symbol:
|
||||
- validatePayment.test.ts [direct]
|
||||
- checkout.integration.test.ts [via processCheckout]
|
||||
```
|
||||
|
||||
**gitnexus_context** — understand a changed symbol's role:
|
||||
|
||||
```
|
||||
gitnexus_context({name: "validatePayment"})
|
||||
|
||||
→ Incoming calls: processCheckout, webhookHandler
|
||||
→ Outgoing calls: verifyCard, fetchRates
|
||||
→ Processes: CheckoutFlow (step 3/7), RefundFlow (step 1/5)
|
||||
```
|
||||
|
||||
## Example: "Review PR #42"
|
||||
|
||||
```
|
||||
1. gh pr diff 42 > /tmp/pr42.diff
|
||||
→ 4 files changed: payments.ts, checkout.ts, types.ts, utils.ts
|
||||
|
||||
2. gitnexus_detect_changes({scope: "compare", base_ref: "main"})
|
||||
→ Changed symbols: validatePayment, PaymentInput, formatAmount
|
||||
→ Affected processes: CheckoutFlow, RefundFlow
|
||||
→ Risk: MEDIUM
|
||||
|
||||
3. gitnexus_impact({target: "validatePayment", direction: "upstream"})
|
||||
→ d=1: processCheckout, webhookHandler (WILL BREAK)
|
||||
→ webhookHandler is NOT in the PR diff — potential breakage!
|
||||
|
||||
4. gitnexus_impact({target: "PaymentInput", direction: "upstream"})
|
||||
→ d=1: validatePayment (in PR), createPayment (NOT in PR)
|
||||
→ createPayment uses the old PaymentInput shape — breaking change!
|
||||
|
||||
5. gitnexus_context({name: "formatAmount"})
|
||||
→ Called by 12 functions — but change is backwards-compatible (added optional param)
|
||||
|
||||
6. Review summary:
|
||||
- MEDIUM risk — 3 changed symbols affect 2 execution flows
|
||||
- BUG: webhookHandler calls validatePayment but isn't updated for new signature
|
||||
- BUG: createPayment depends on PaymentInput type which changed
|
||||
- OK: formatAmount change is backwards-compatible
|
||||
- Tests: checkout.test.ts covers processCheckout path, but no webhook test
|
||||
```
|
||||
|
||||
## Review Output Format
|
||||
|
||||
Structure your review as:
|
||||
|
||||
```markdown
|
||||
## PR Review: <title>
|
||||
|
||||
**Risk: LOW / MEDIUM / HIGH / CRITICAL**
|
||||
|
||||
### Changes Summary
|
||||
- <N> symbols changed across <M> files
|
||||
- <P> execution flows affected
|
||||
|
||||
### Findings
|
||||
1. **[severity]** Description of finding
|
||||
- Evidence from GitNexus tools
|
||||
- Affected callers/flows
|
||||
|
||||
### Missing Coverage
|
||||
- Callers not updated in PR: ...
|
||||
- Untested flows: ...
|
||||
|
||||
### Recommendation
|
||||
APPROVE / REQUEST CHANGES / NEEDS DISCUSSION
|
||||
```
|
||||
@@ -0,0 +1,163 @@
|
||||
---
|
||||
name: gitnexus-pr-review
|
||||
description: "Use when the user wants to review a pull request, understand what a PR changes, assess risk of merging, or check for missing test coverage. Examples: \"Review this PR\", \"What does PR #42 change?\", \"Is this PR safe to merge?\""
|
||||
---
|
||||
|
||||
# PR Review with GitNexus
|
||||
|
||||
## When to Use
|
||||
|
||||
- "Review this PR"
|
||||
- "What does PR #42 change?"
|
||||
- "Is this safe to merge?"
|
||||
- "What's the blast radius of this PR?"
|
||||
- "Are there missing tests for this PR?"
|
||||
- Reviewing someone else's code changes before merge
|
||||
|
||||
## Workflow
|
||||
|
||||
```
|
||||
1. gh pr diff <number> → Get the raw diff
|
||||
2. gitnexus_detect_changes({scope: "compare", base_ref: "main"}) → Map diff to affected flows
|
||||
3. For each changed symbol:
|
||||
gitnexus_impact({target: "<symbol>", direction: "upstream"}) → Blast radius per change
|
||||
4. gitnexus_context({name: "<key symbol>"}) → Understand callers/callees
|
||||
5. READ gitnexus://repo/{name}/processes → Check affected execution flows
|
||||
6. Summarize findings with risk assessment
|
||||
```
|
||||
|
||||
> If "Index is stale" → run `npx gitnexus analyze` in terminal before reviewing.
|
||||
|
||||
## Checklist
|
||||
|
||||
```
|
||||
- [ ] Fetch PR diff (gh pr diff or git diff base...head)
|
||||
- [ ] gitnexus_detect_changes to map changes to affected execution flows
|
||||
- [ ] gitnexus_impact on each non-trivial changed symbol
|
||||
- [ ] Review d=1 items (WILL BREAK) — are callers updated?
|
||||
- [ ] gitnexus_context on key changed symbols to understand full picture
|
||||
- [ ] Check if affected processes have test coverage
|
||||
- [ ] Assess overall risk level
|
||||
- [ ] Write review summary with findings
|
||||
```
|
||||
|
||||
## Review Dimensions
|
||||
|
||||
| Dimension | How GitNexus Helps |
|
||||
| --- | --- |
|
||||
| **Correctness** | `context` shows callers — are they all compatible with the change? |
|
||||
| **Blast radius** | `impact` shows d=1/d=2/d=3 dependents — anything missed? |
|
||||
| **Completeness** | `detect_changes` shows all affected flows — are they all handled? |
|
||||
| **Test coverage** | `impact({includeTests: true})` shows which tests touch changed code |
|
||||
| **Breaking changes** | d=1 upstream items that aren't updated in the PR = potential breakage |
|
||||
|
||||
## Risk Assessment
|
||||
|
||||
| Signal | Risk |
|
||||
| --- | --- |
|
||||
| Changes touch <3 symbols, 0-1 processes | LOW |
|
||||
| Changes touch 3-10 symbols, 2-5 processes | MEDIUM |
|
||||
| Changes touch >10 symbols or many processes | HIGH |
|
||||
| Changes touch auth, payments, or data integrity code | CRITICAL |
|
||||
| d=1 callers exist outside the PR diff | Potential breakage — flag it |
|
||||
|
||||
## Tools
|
||||
|
||||
**gitnexus_detect_changes** — map PR diff to affected execution flows:
|
||||
|
||||
```
|
||||
gitnexus_detect_changes({scope: "compare", base_ref: "main"})
|
||||
|
||||
→ Changed: 8 symbols in 4 files
|
||||
→ Affected processes: CheckoutFlow, RefundFlow, WebhookHandler
|
||||
→ Risk: MEDIUM
|
||||
```
|
||||
|
||||
**gitnexus_impact** — blast radius per changed symbol:
|
||||
|
||||
```
|
||||
gitnexus_impact({target: "validatePayment", direction: "upstream"})
|
||||
|
||||
→ d=1 (WILL BREAK):
|
||||
- processCheckout (src/checkout.ts:42) [CALLS, 100%]
|
||||
- webhookHandler (src/webhooks.ts:15) [CALLS, 100%]
|
||||
|
||||
→ d=2 (LIKELY AFFECTED):
|
||||
- checkoutRouter (src/routes/checkout.ts:22) [CALLS, 95%]
|
||||
```
|
||||
|
||||
**gitnexus_impact with tests** — check test coverage:
|
||||
|
||||
```
|
||||
gitnexus_impact({target: "validatePayment", direction: "upstream", includeTests: true})
|
||||
|
||||
→ Tests that cover this symbol:
|
||||
- validatePayment.test.ts [direct]
|
||||
- checkout.integration.test.ts [via processCheckout]
|
||||
```
|
||||
|
||||
**gitnexus_context** — understand a changed symbol's role:
|
||||
|
||||
```
|
||||
gitnexus_context({name: "validatePayment"})
|
||||
|
||||
→ Incoming calls: processCheckout, webhookHandler
|
||||
→ Outgoing calls: verifyCard, fetchRates
|
||||
→ Processes: CheckoutFlow (step 3/7), RefundFlow (step 1/5)
|
||||
```
|
||||
|
||||
## Example: "Review PR #42"
|
||||
|
||||
```
|
||||
1. gh pr diff 42 > /tmp/pr42.diff
|
||||
→ 4 files changed: payments.ts, checkout.ts, types.ts, utils.ts
|
||||
|
||||
2. gitnexus_detect_changes({scope: "compare", base_ref: "main"})
|
||||
→ Changed symbols: validatePayment, PaymentInput, formatAmount
|
||||
→ Affected processes: CheckoutFlow, RefundFlow
|
||||
→ Risk: MEDIUM
|
||||
|
||||
3. gitnexus_impact({target: "validatePayment", direction: "upstream"})
|
||||
→ d=1: processCheckout, webhookHandler (WILL BREAK)
|
||||
→ webhookHandler is NOT in the PR diff — potential breakage!
|
||||
|
||||
4. gitnexus_impact({target: "PaymentInput", direction: "upstream"})
|
||||
→ d=1: validatePayment (in PR), createPayment (NOT in PR)
|
||||
→ createPayment uses the old PaymentInput shape — breaking change!
|
||||
|
||||
5. gitnexus_context({name: "formatAmount"})
|
||||
→ Called by 12 functions — but change is backwards-compatible (added optional param)
|
||||
|
||||
6. Review summary:
|
||||
- MEDIUM risk — 3 changed symbols affect 2 execution flows
|
||||
- BUG: webhookHandler calls validatePayment but isn't updated for new signature
|
||||
- BUG: createPayment depends on PaymentInput type which changed
|
||||
- OK: formatAmount change is backwards-compatible
|
||||
- Tests: checkout.test.ts covers processCheckout path, but no webhook test
|
||||
```
|
||||
|
||||
## Review Output Format
|
||||
|
||||
Structure your review as:
|
||||
|
||||
```markdown
|
||||
## PR Review: <title>
|
||||
|
||||
**Risk: LOW / MEDIUM / HIGH / CRITICAL**
|
||||
|
||||
### Changes Summary
|
||||
- <N> symbols changed across <M> files
|
||||
- <P> execution flows affected
|
||||
|
||||
### Findings
|
||||
1. **[severity]** Description of finding
|
||||
- Evidence from GitNexus tools
|
||||
- Affected callers/flows
|
||||
|
||||
### Missing Coverage
|
||||
- Callers not updated in PR: ...
|
||||
- Untested flows: ...
|
||||
|
||||
### Recommendation
|
||||
APPROVE / REQUEST CHANGES / NEEDS DISCUSSION
|
||||
```
|
||||
@@ -0,0 +1,127 @@
|
||||
import { test, expect, type TestInfo } from '@playwright/test';
|
||||
|
||||
/**
|
||||
* Debug harnesses for investigating specific UI issues.
|
||||
* Excluded from `npm run test:e2e` via testIgnore in playwright.config.ts.
|
||||
* Run directly: DEBUG_E2E=1 npx playwright test e2e/debug-issues.spec.ts
|
||||
*/
|
||||
const BACKEND_URL = process.env.BACKEND_URL ?? 'http://localhost:4747';
|
||||
const debugTest = process.env.DEBUG_E2E ? test : test.skip;
|
||||
|
||||
async function connectToServer(page: import('@playwright/test').Page) {
|
||||
page.on('console', msg => {
|
||||
if (msg.type() === 'error') console.log(`[error] ${msg.text()}`);
|
||||
});
|
||||
|
||||
await page.goto('/');
|
||||
await page.getByText('Server').click();
|
||||
const serverInput = page.locator('input[name="server-url-input"]');
|
||||
await serverInput.fill(BACKEND_URL);
|
||||
await page.getByRole('button', { name: /Connect/ }).click();
|
||||
await expect(page.locator('[data-testid="status-ready"]')).toBeVisible({ timeout: 30_000 });
|
||||
// Wait for LadybugDB to finish loading — poll isDatabaseReady via process list visibility
|
||||
await expect(page.locator('[data-testid="status-ready"]')).toBeVisible({ timeout: 30_000 });
|
||||
}
|
||||
|
||||
debugTest('debug: process view Reset View button', async ({ page }, testInfo) => {
|
||||
await connectToServer(page);
|
||||
|
||||
// Open Processes tab
|
||||
await page.getByRole('button', { name: 'Nexus AI' }).click();
|
||||
await page.getByText('Processes').click();
|
||||
await expect(page.getByText(/\d+ processes detected/)).toBeVisible({ timeout: 10_000 });
|
||||
|
||||
// Click View on the first Cross-Community process
|
||||
// The View button has opacity-0 by default, use JS click to bypass
|
||||
const viewButtons = page.locator('button:has-text("View")');
|
||||
const count = await viewButtons.count();
|
||||
console.log(`Found ${count} View buttons`);
|
||||
|
||||
// Use evaluate to click the first one regardless of visibility
|
||||
await page.evaluate(() => {
|
||||
const btns = document.querySelectorAll('button');
|
||||
for (const btn of btns) {
|
||||
if (btn.textContent?.trim() === 'View') {
|
||||
btn.click();
|
||||
return;
|
||||
}
|
||||
}
|
||||
});
|
||||
|
||||
// Wait for modal to appear
|
||||
const modal = page.locator('[data-testid="process-modal"]');
|
||||
await expect(modal).toBeVisible({ timeout: 5_000 });
|
||||
|
||||
// Screenshot: modal should be open with flowchart
|
||||
await page.screenshot({ path: testInfo.outputPath('debug-modal-open.png'), fullPage: true });
|
||||
|
||||
// Get the diagram's current transform
|
||||
const diagramDiv = modal.locator('[style*="transform"]');
|
||||
const transformBefore = await diagramDiv.getAttribute('style');
|
||||
console.log('Transform BEFORE zoom:', transformBefore);
|
||||
|
||||
// Zoom in using the + button
|
||||
const zoomInBtn = modal.getByRole('button', { name: /Zoom in/ });
|
||||
await zoomInBtn.click();
|
||||
await zoomInBtn.click();
|
||||
await zoomInBtn.click();
|
||||
// Wait for zoom animation to settle
|
||||
await expect(async () => {
|
||||
const t = await diagramDiv.getAttribute('style');
|
||||
expect(t).not.toBe(transformBefore);
|
||||
}).toPass({ timeout: 2_000 });
|
||||
|
||||
const transformAfterZoom = await diagramDiv.getAttribute('style');
|
||||
console.log('Transform AFTER zoom:', transformAfterZoom);
|
||||
await page.screenshot({ path: testInfo.outputPath('debug-modal-zoomed.png'), fullPage: true });
|
||||
|
||||
// Click Reset View
|
||||
const resetBtn = modal.getByRole('button', { name: 'Reset View' });
|
||||
await resetBtn.click();
|
||||
// Wait for reset animation to settle
|
||||
await expect(async () => {
|
||||
const t = await diagramDiv.getAttribute('style');
|
||||
expect(t).toBe(transformBefore);
|
||||
}).toPass({ timeout: 2_000 });
|
||||
|
||||
const transformAfterReset = await diagramDiv.getAttribute('style');
|
||||
console.log('Transform AFTER reset:', transformAfterReset);
|
||||
await page.screenshot({ path: testInfo.outputPath('debug-modal-after-reset.png'), fullPage: true });
|
||||
|
||||
// Verify transform actually changed back
|
||||
expect(transformAfterZoom).not.toBe(transformBefore);
|
||||
expect(transformAfterReset).toBe(transformBefore);
|
||||
});
|
||||
|
||||
debugTest('debug: lightbulb clears node selection dimming', async ({ page }, testInfo) => {
|
||||
await connectToServer(page);
|
||||
|
||||
// Wait for graph canvas to render
|
||||
await expect(page.locator('canvas').first()).toBeVisible({ timeout: 10_000 });
|
||||
|
||||
await page.screenshot({ path: testInfo.outputPath('debug-before-select.png'), fullPage: true });
|
||||
|
||||
// Click a file in the tree to select a node (causes dimming)
|
||||
const fileItem = page.getByText('start.sh');
|
||||
await fileItem.click();
|
||||
await page.waitForTimeout(500);
|
||||
await page.screenshot({ path: testInfo.outputPath('debug-node-selected.png'), fullPage: true });
|
||||
|
||||
// Check the lightbulb button state
|
||||
const lightbulbBtn = page.locator('button[title*="Turn off"], button[title*="Turn on"]');
|
||||
const title = await lightbulbBtn.getAttribute('title');
|
||||
console.log('Lightbulb title before click:', title);
|
||||
|
||||
// Click the lightbulb
|
||||
await lightbulbBtn.click();
|
||||
await page.waitForTimeout(500);
|
||||
|
||||
const titleAfter = await lightbulbBtn.getAttribute('title');
|
||||
console.log('Lightbulb title after click:', titleAfter);
|
||||
await page.screenshot({ path: testInfo.outputPath('debug-after-lightbulb.png'), fullPage: true });
|
||||
|
||||
// Click it again to toggle back on
|
||||
await lightbulbBtn.click();
|
||||
await page.waitForTimeout(500);
|
||||
await page.screenshot({ path: testInfo.outputPath('debug-after-lightbulb-toggle-back.png'), fullPage: true });
|
||||
});
|
||||
@@ -0,0 +1,28 @@
|
||||
import { test } from '@playwright/test';
|
||||
|
||||
/**
|
||||
* Manual recording session for interactive debugging.
|
||||
* Opens the app and pauses so you can interact with the UI.
|
||||
* Trace, video, and screenshots are saved automatically on close.
|
||||
*
|
||||
* Run with: npx playwright test e2e/manual-record.spec.ts --headed --timeout=0
|
||||
*
|
||||
* Excluded from `npm run test:e2e` via testIgnore in playwright.config.ts.
|
||||
* Also skipped when PWDEBUG is not set or in CI, as a safety net.
|
||||
*/
|
||||
test.skip(
|
||||
!!process.env.CI || process.env.PWDEBUG !== '1',
|
||||
'Manual recording requires --headed and PWDEBUG=1. Run: PWDEBUG=1 npx playwright test e2e/manual-record.spec.ts --headed --timeout=0'
|
||||
);
|
||||
|
||||
test('manual recording session', async ({ page }) => {
|
||||
page.on('console', msg => {
|
||||
if (msg.type() === 'error' || msg.type() === 'warning') {
|
||||
console.log(`[${msg.type()}] ${msg.text()}`);
|
||||
}
|
||||
});
|
||||
page.on('pageerror', err => console.log(`[crash] ${err.message}`));
|
||||
|
||||
await page.goto('http://localhost:5173');
|
||||
await page.pause();
|
||||
});
|
||||
@@ -0,0 +1,165 @@
|
||||
import { test, expect, type TestInfo } from '@playwright/test';
|
||||
|
||||
/**
|
||||
* E2E tests for the GitNexus web UI.
|
||||
* Requires:
|
||||
* - gitnexus serve running on localhost:4747
|
||||
* - gitnexus-web dev server running on localhost:5173
|
||||
*
|
||||
* Skipped when servers aren't available (CI without services, etc.).
|
||||
* Set E2E=1 to force-run even without the availability check.
|
||||
*/
|
||||
|
||||
const BACKEND_URL = process.env.BACKEND_URL ?? 'http://localhost:4747';
|
||||
const FRONTEND_URL = process.env.FRONTEND_URL ?? 'http://localhost:5173';
|
||||
// Skip all tests if the gitnexus server or Vite dev server isn't reachable
|
||||
test.beforeAll(async () => {
|
||||
if (process.env.E2E) return; // force-run
|
||||
try {
|
||||
const [backendRes, frontendRes] = await Promise.allSettled([
|
||||
fetch(`${BACKEND_URL}/api/repos`),
|
||||
fetch(FRONTEND_URL),
|
||||
]);
|
||||
if (backendRes.status === 'rejected' || (backendRes.status === 'fulfilled' && !backendRes.value.ok)) {
|
||||
test.skip(true, 'gitnexus serve not available on :4747');
|
||||
return;
|
||||
}
|
||||
if (frontendRes.status === 'rejected' || (frontendRes.status === 'fulfilled' && !frontendRes.value.ok)) {
|
||||
test.skip(true, 'Vite dev server not available on :5173');
|
||||
return;
|
||||
}
|
||||
} catch {
|
||||
test.skip(true, 'servers not available');
|
||||
}
|
||||
});
|
||||
|
||||
/** Shared helper: connect to the local server and wait for the graph to load */
|
||||
async function connectAndWaitForGraph(page: import('@playwright/test').Page, testInfo: TestInfo) {
|
||||
// Signal to the app that we are running under Playwright (used to skip heavy Ladybug loads).
|
||||
await page.addInitScript(() => {
|
||||
(window as unknown as { __PLAYWRIGHT_TEST__?: boolean }).__PLAYWRIGHT_TEST__ = true;
|
||||
});
|
||||
|
||||
await page.goto('/');
|
||||
|
||||
// Wait for the app to fully render before interacting
|
||||
const serverTab = page.getByRole('button', { name: 'Server' });
|
||||
await expect(serverTab).toBeVisible({ timeout: 15_000 });
|
||||
await page.screenshot({ path: testInfo.outputPath('step-1-landing.png') });
|
||||
|
||||
// Click "Server" tab and wait for the input to appear
|
||||
await serverTab.click();
|
||||
const serverInput = page.locator('input[name="server-url-input"]');
|
||||
await expect(serverInput).toBeVisible({ timeout: 15_000 });
|
||||
await page.screenshot({ path: testInfo.outputPath('step-2-server-tab.png') });
|
||||
await serverInput.fill(BACKEND_URL);
|
||||
await page.screenshot({ path: testInfo.outputPath('step-3-url-filled.png') });
|
||||
|
||||
await page.getByRole('button', { name: /Connect/ }).click();
|
||||
|
||||
// Wait for graph to load — status bar shows "Ready"
|
||||
await expect(page.locator('[data-testid="status-ready"]')).toBeVisible({ timeout: 30_000 });
|
||||
await expect(page.getByText(/\d+ nodes/).first()).toBeVisible();
|
||||
await page.screenshot({ path: testInfo.outputPath('step-4-graph-loaded.png') });
|
||||
}
|
||||
|
||||
test.describe('Server Connection & Graph Loading', () => {
|
||||
test('connects to server and loads graph', async ({ page }, testInfo) => {
|
||||
await connectAndWaitForGraph(page, testInfo);
|
||||
await page.screenshot({ path: testInfo.outputPath('graph-loaded.png'), fullPage: true });
|
||||
});
|
||||
});
|
||||
|
||||
test.describe('Nexus AI', () => {
|
||||
test('panel opens and agent initializes without error', async ({ page }, testInfo) => {
|
||||
await connectAndWaitForGraph(page, testInfo);
|
||||
|
||||
// Click Nexus AI button to open the panel
|
||||
await page.getByRole('button', { name: 'Nexus AI' }).click();
|
||||
|
||||
// Should see the Nexus AI tab content
|
||||
await expect(page.getByText('Ask me anything')).toBeVisible({ timeout: 15_000 });
|
||||
|
||||
await page.screenshot({ path: testInfo.outputPath('nexus-ai-panel.png'), fullPage: true });
|
||||
|
||||
// "Database not ready" should NOT be visible
|
||||
const errorBanner = page.getByText('Database not ready');
|
||||
expect(await errorBanner.isVisible().catch(() => false)).toBe(false);
|
||||
});
|
||||
});
|
||||
|
||||
test.describe('Processes Panel', () => {
|
||||
test('shows process list and View button works', async ({ page }, testInfo) => {
|
||||
await connectAndWaitForGraph(page, testInfo);
|
||||
|
||||
// Open Nexus AI panel, switch to Processes tab
|
||||
await page.getByRole('button', { name: 'Nexus AI' }).click();
|
||||
await page.getByText('Processes').click();
|
||||
|
||||
// Should show process count — wait for data-testid instead of fixed timeout
|
||||
await expect(page.locator('[data-testid="process-list-loaded"]')).toBeVisible({ timeout: 15_000 });
|
||||
await page.screenshot({ path: testInfo.outputPath('processes-panel.png'), fullPage: true });
|
||||
|
||||
// Hover first process item to reveal View button, then click it
|
||||
const processRow = page.locator('[data-testid="process-row"]').first();
|
||||
await expect(processRow).toBeVisible({ timeout: 10_000 });
|
||||
await processRow.hover();
|
||||
|
||||
const viewBtn = processRow.locator('[data-testid="process-view-button"]');
|
||||
await viewBtn.waitFor({ state: 'visible', timeout: 5_000 });
|
||||
await viewBtn.click();
|
||||
// Wait for modal to appear
|
||||
await expect(page.locator('[data-testid="process-modal"]')).toBeVisible({ timeout: 5_000 });
|
||||
await page.screenshot({ path: testInfo.outputPath('process-view-clicked.png'), fullPage: true });
|
||||
});
|
||||
|
||||
test('lightbulb highlights nodes in graph', async ({ page }, testInfo) => {
|
||||
await connectAndWaitForGraph(page, testInfo);
|
||||
|
||||
await page.getByRole('button', { name: 'Nexus AI' }).click();
|
||||
await page.getByText('Processes').click();
|
||||
await expect(page.locator('[data-testid="process-list-loaded"]')).toBeVisible({ timeout: 15_000 });
|
||||
|
||||
await page.screenshot({ path: testInfo.outputPath('before-highlight.png'), fullPage: true });
|
||||
|
||||
// Hover first process to reveal lightbulb
|
||||
const processRow = page.locator('[data-testid="process-row"]').first();
|
||||
await expect(processRow).toBeVisible({ timeout: 10_000 });
|
||||
await processRow.hover();
|
||||
|
||||
const lightbulb = processRow.locator('[data-testid="process-highlight-button"]');
|
||||
await lightbulb.waitFor({ state: 'visible', timeout: 5_000 });
|
||||
await lightbulb.click();
|
||||
// Wait for highlight to apply — the process row gets amber styling when focused
|
||||
await expect(processRow).toHaveClass(/bg-amber-950/, { timeout: 5_000 });
|
||||
await page.screenshot({ path: testInfo.outputPath('after-highlight.png'), fullPage: true });
|
||||
});
|
||||
});
|
||||
|
||||
test.describe('Turn Off All Highlights', () => {
|
||||
test('selecting a node dims others, button clears it', async ({ page }, testInfo) => {
|
||||
await connectAndWaitForGraph(page, testInfo);
|
||||
|
||||
// Wait for graph to fully render by checking for canvas element
|
||||
await expect(page.locator('canvas').first()).toBeVisible({ timeout: 10_000 });
|
||||
|
||||
await page.screenshot({ path: testInfo.outputPath('before-select.png'), fullPage: true });
|
||||
|
||||
// Click a file in the file tree to select a node
|
||||
const fileItem = page.getByText('package.json').first();
|
||||
await expect(fileItem).toBeVisible({ timeout: 10_000 });
|
||||
await fileItem.click();
|
||||
|
||||
// Wait for highlight toggle to show "Turn off" (indicates highlights are active)
|
||||
const highlightToggle = page.locator('[data-testid="ai-highlights-toggle"]');
|
||||
await expect(highlightToggle).toHaveAttribute('title', 'Turn off all highlights', { timeout: 5_000 });
|
||||
await page.screenshot({ path: testInfo.outputPath('node-selected.png'), fullPage: true });
|
||||
|
||||
// Click the toggle to clear all highlights
|
||||
await highlightToggle.click();
|
||||
|
||||
// Verify highlights are now off — button title changes to "Turn on"
|
||||
await expect(highlightToggle).toHaveAttribute('title', 'Turn on AI highlights', { timeout: 5_000 });
|
||||
await page.screenshot({ path: testInfo.outputPath('highlights-cleared.png'), fullPage: true });
|
||||
});
|
||||
});
|
||||
Generated
+2154
-40
File diff suppressed because it is too large
Load Diff
@@ -2,15 +2,25 @@
|
||||
"name": "gitnexus",
|
||||
"private": true,
|
||||
"version": "0.0.0",
|
||||
"engines": {
|
||||
"node": ">=20.0.0"
|
||||
},
|
||||
"type": "module",
|
||||
"scripts": {
|
||||
"dev": "vite",
|
||||
"build": "tsc -b && vite build",
|
||||
"preview": "vite preview"
|
||||
"preview": "vite preview",
|
||||
"test": "vitest run",
|
||||
"test:watch": "vitest",
|
||||
"test:coverage": "vitest run --coverage",
|
||||
"test:e2e": "playwright test",
|
||||
"test:e2e:ui": "playwright test --ui",
|
||||
"test:e2e:report": "playwright show-report"
|
||||
},
|
||||
"dependencies": {
|
||||
"@huggingface/transformers": "^3.0.0",
|
||||
"@isomorphic-git/lightning-fs": "^4.6.2",
|
||||
"@ladybugdb/wasm-core": "^0.15.1",
|
||||
"@langchain/anthropic": "^1.3.10",
|
||||
"@langchain/core": "^1.1.15",
|
||||
"@langchain/google-genai": "^2.1.10",
|
||||
@@ -23,22 +33,22 @@
|
||||
"buffer": "^6.0.3",
|
||||
"comlink": "^4.4.2",
|
||||
"d3": "^7.9.0",
|
||||
"dompurify": "^3.3.3",
|
||||
"graphology": "^0.26.0",
|
||||
"graphology-indices": "^0.17.0",
|
||||
"graphology-utils": "^2.3.0",
|
||||
"mnemonist": "^0.39.0",
|
||||
"pandemonium": "^2.4.0",
|
||||
"graphology-layout-force": "^0.2.4",
|
||||
"graphology-layout-forceatlas2": "^0.10.1",
|
||||
"graphology-layout-noverlap": "^0.4.2",
|
||||
"graphology-utils": "^2.3.0",
|
||||
"isomorphic-git": "^1.36.1",
|
||||
"jszip": "^3.10.1",
|
||||
"kuzu-wasm": "^0.11.1",
|
||||
"langchain": "^1.2.10",
|
||||
"lru-cache": "^11.2.4",
|
||||
"lucide-react": "^0.562.0",
|
||||
"mermaid": "^11.12.2",
|
||||
"minisearch": "^7.2.0",
|
||||
"mnemonist": "^0.39.0",
|
||||
"pandemonium": "^2.4.0",
|
||||
"react": "^18.3.1",
|
||||
"react-dom": "^18.3.1",
|
||||
"react-markdown": "^10.1.0",
|
||||
@@ -55,6 +65,11 @@
|
||||
},
|
||||
"devDependencies": {
|
||||
"@babel/types": "^7.28.5",
|
||||
"@playwright/test": "^1.58.2",
|
||||
"@testing-library/jest-dom": "^6.9.1",
|
||||
"@testing-library/react": "^16.3.2",
|
||||
"@testing-library/user-event": "^14.6.1",
|
||||
"@types/dompurify": "^3.0.5",
|
||||
"@types/jszip": "^3.4.0",
|
||||
"@types/node": "^24.10.1",
|
||||
"@types/react": "^18.3.5",
|
||||
@@ -62,9 +77,13 @@
|
||||
"@types/react-syntax-highlighter": "^15.5.13",
|
||||
"@vercel/node": "^5.5.16",
|
||||
"@vitejs/plugin-react": "^5.1.0",
|
||||
"@vitest/coverage-v8": "^3.2.4",
|
||||
"jsdom": "^29.0.0",
|
||||
"tree-sitter-wasms": "^0.1.13",
|
||||
"typescript": "^5.4.5",
|
||||
"vite": "^5.2.0",
|
||||
"vite-plugin-static-copy": "^3.1.4"
|
||||
"vite-plugin-static-copy": "^3.1.4",
|
||||
"vitest": "^3.2.4",
|
||||
"wait-on": "^8.0.5"
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,48 @@
|
||||
import { defineConfig } from '@playwright/test';
|
||||
|
||||
// Enable insecure browser config (disabled security + CSP bypass) only when explicitly requested.
|
||||
// Example: PLAYWRIGHT_INSECURE=1 npx playwright test
|
||||
const insecureE2E = process.env.PLAYWRIGHT_INSECURE === '1';
|
||||
|
||||
// Base launch args: always enable software WebGL for sigma.js graph rendering in headless mode.
|
||||
const launchArgs = [
|
||||
'--use-gl=angle',
|
||||
'--use-angle=swiftshader',
|
||||
'--enable-webgl',
|
||||
'--enable-unsafe-swiftshader',
|
||||
];
|
||||
|
||||
if (insecureE2E) {
|
||||
// Allow cross-origin requests to gitnexus serve on a different port when explicitly enabled.
|
||||
launchArgs.unshift('--disable-web-security', '--disable-site-isolation-trials');
|
||||
}
|
||||
|
||||
export default defineConfig({
|
||||
testDir: './e2e',
|
||||
testIgnore: ['**/manual-record.spec.ts', '**/debug-issues.spec.ts'],
|
||||
timeout: 60_000,
|
||||
retries: process.env.CI ? 1 : 0,
|
||||
use: {
|
||||
baseURL: 'http://localhost:5173',
|
||||
trace: 'retain-on-failure',
|
||||
screenshot: 'retain-on-failure',
|
||||
video: 'retain-on-failure',
|
||||
launchOptions: {
|
||||
args: launchArgs,
|
||||
},
|
||||
// Vite dev server sets COEP require-corp for SharedArrayBuffer (LadybugDB WASM).
|
||||
// Only bypass CSP when explicitly running in insecure E2E mode.
|
||||
bypassCSP: insecureE2E,
|
||||
},
|
||||
projects: [
|
||||
{
|
||||
name: 'chromium',
|
||||
use: { browserName: 'chromium' },
|
||||
},
|
||||
],
|
||||
reporter: [
|
||||
['list'],
|
||||
['html', { open: 'never', outputFolder: 'playwright-report' }],
|
||||
],
|
||||
outputDir: 'test-results',
|
||||
});
|
||||
Binary file not shown.
Binary file not shown.
BIN
Binary file not shown.
+47
-55
@@ -13,6 +13,7 @@ import { FileEntry } from './services/zip';
|
||||
import { getActiveProviderConfig } from './core/llm/settings-service';
|
||||
import { createKnowledgeGraph } from './core/graph/graph';
|
||||
import { connectToServer, fetchRepos, normalizeServerUrl, type ConnectToServerResult } from './services/server-connection';
|
||||
import { ERROR_RESET_DELAY_MS } from './config/ui-constants';
|
||||
|
||||
const AppContent = () => {
|
||||
const {
|
||||
@@ -30,7 +31,7 @@ const AppContent = () => {
|
||||
setSettingsPanelOpen,
|
||||
refreshLLMSettings,
|
||||
initializeAgent,
|
||||
startEmbeddings,
|
||||
startEmbeddingsWithFallback,
|
||||
embeddingStatus,
|
||||
codeReferences,
|
||||
selectedNode,
|
||||
@@ -40,6 +41,7 @@ const AppContent = () => {
|
||||
availableRepos,
|
||||
setAvailableRepos,
|
||||
switchRepo,
|
||||
loadServerGraph,
|
||||
} = useAppState();
|
||||
|
||||
const graphCanvasRef = useRef<GraphCanvasHandle>(null);
|
||||
@@ -67,13 +69,7 @@ const AppContent = () => {
|
||||
|
||||
// Auto-start embeddings pipeline in background
|
||||
// Uses WebGPU if available, falls back to WASM
|
||||
startEmbeddings().catch((err) => {
|
||||
if (err?.name === 'WebGPUNotAvailableError' || err?.message?.includes('WebGPU')) {
|
||||
startEmbeddings('wasm').catch(console.warn);
|
||||
} else {
|
||||
console.warn('Embeddings auto-start failed:', err);
|
||||
}
|
||||
});
|
||||
startEmbeddingsWithFallback();
|
||||
} catch (error) {
|
||||
console.error('Pipeline error:', error);
|
||||
setProgress({
|
||||
@@ -85,13 +81,16 @@ const AppContent = () => {
|
||||
setTimeout(() => {
|
||||
setViewMode('onboarding');
|
||||
setProgress(null);
|
||||
}, 3000);
|
||||
}, ERROR_RESET_DELAY_MS);
|
||||
}
|
||||
}, [setViewMode, setGraph, setFileContents, setProgress, setProjectName, runPipeline, startEmbeddings, initializeAgent]);
|
||||
}, [setViewMode, setGraph, setFileContents, setProgress, setProjectName, runPipeline, startEmbeddingsWithFallback, initializeAgent]);
|
||||
|
||||
const handleGitClone = useCallback(async (files: FileEntry[]) => {
|
||||
const firstPath = files[0]?.path || 'repository';
|
||||
const projectName = firstPath.split('/')[0].replace(/-\d+$/, '') || 'repository';
|
||||
const handleGitClone = useCallback(async (files: FileEntry[], repoName?: string) => {
|
||||
let projectName = repoName;
|
||||
if (!projectName) {
|
||||
const firstPath = files[0]?.path || 'repository';
|
||||
projectName = firstPath.split('/')[0].replace(/-\d+$/, '') || 'repository';
|
||||
}
|
||||
|
||||
setProjectName(projectName);
|
||||
setProgress({ phase: 'extracting', percent: 0, message: 'Starting...', detail: 'Preparing to process files' });
|
||||
@@ -110,13 +109,7 @@ const AppContent = () => {
|
||||
initializeAgent(projectName);
|
||||
}
|
||||
|
||||
startEmbeddings().catch((err) => {
|
||||
if (err?.name === 'WebGPUNotAvailableError' || err?.message?.includes('WebGPU')) {
|
||||
startEmbeddings('wasm').catch(console.warn);
|
||||
} else {
|
||||
console.warn('Embeddings auto-start failed:', err);
|
||||
}
|
||||
});
|
||||
startEmbeddingsWithFallback();
|
||||
} catch (error) {
|
||||
console.error('Pipeline error:', error);
|
||||
setProgress({
|
||||
@@ -128,17 +121,18 @@ const AppContent = () => {
|
||||
setTimeout(() => {
|
||||
setViewMode('onboarding');
|
||||
setProgress(null);
|
||||
}, 3000);
|
||||
}, ERROR_RESET_DELAY_MS);
|
||||
}
|
||||
}, [setViewMode, setGraph, setFileContents, setProgress, setProjectName, runPipelineFromFiles, startEmbeddings, initializeAgent]);
|
||||
}, [setViewMode, setGraph, setFileContents, setProgress, setProjectName, runPipelineFromFiles, startEmbeddingsWithFallback, initializeAgent]);
|
||||
|
||||
const handleServerConnect = useCallback((result: ConnectToServerResult) => {
|
||||
const handleServerConnect = useCallback((result: ConnectToServerResult): Promise<void> => {
|
||||
// Extract project name from repoPath
|
||||
const repoPath = result.repoInfo.repoPath;
|
||||
const projectName = repoPath.split('/').pop() || 'server-project';
|
||||
const parts = repoPath.split('/').filter(p => p && !p.startsWith('.'));
|
||||
const projectName = parts[parts.length - 1] || parts[0] || 'server-project';
|
||||
setProjectName(projectName);
|
||||
|
||||
// Build KnowledgeGraph from server data (bypasses WASM pipeline entirely)
|
||||
// Build KnowledgeGraph from server data for visualization
|
||||
const graph = createKnowledgeGraph();
|
||||
for (const node of result.nodes) {
|
||||
graph.addNode(node);
|
||||
@@ -158,20 +152,24 @@ const AppContent = () => {
|
||||
// Transition directly to exploring view
|
||||
setViewMode('exploring');
|
||||
|
||||
// Initialize agent if LLM is configured
|
||||
if (getActiveProviderConfig()) {
|
||||
initializeAgent(projectName);
|
||||
}
|
||||
// Load graph into LadybugDB (in-browser WASM database) for Nexus AI queries,
|
||||
// then initialize agent once the database is ready
|
||||
const loadGraphPromise = loadServerGraph(result.nodes, result.relationships, result.fileContents)
|
||||
.then(() => {
|
||||
if (getActiveProviderConfig()) {
|
||||
return initializeAgent(projectName);
|
||||
}
|
||||
})
|
||||
.then(() => {
|
||||
startEmbeddingsWithFallback();
|
||||
})
|
||||
.catch((err) => {
|
||||
console.warn('Failed to load graph into LadybugDB:', err);
|
||||
// Agent won't work but graph visualization still does
|
||||
});
|
||||
|
||||
// Auto-start embeddings
|
||||
startEmbeddings().catch((err) => {
|
||||
if (err?.name === 'WebGPUNotAvailableError' || err?.message?.includes('WebGPU')) {
|
||||
startEmbeddings('wasm').catch(console.warn);
|
||||
} else {
|
||||
console.warn('Embeddings auto-start failed:', err);
|
||||
}
|
||||
});
|
||||
}, [setViewMode, setGraph, setFileContents, setProjectName, initializeAgent, startEmbeddings]);
|
||||
return loadGraphPromise;
|
||||
}, [setViewMode, setGraph, setFileContents, setProjectName, loadServerGraph, initializeAgent, startEmbeddingsWithFallback]);
|
||||
|
||||
// Auto-connect when ?server query param is present (bookmarkable shortcut)
|
||||
const autoConnectRan = useRef(false);
|
||||
@@ -203,16 +201,12 @@ const AppContent = () => {
|
||||
setProgress({ phase: 'extracting', percent: 97, message: 'Processing...', detail: 'Extracting file contents' });
|
||||
}
|
||||
}).then(async (result) => {
|
||||
handleServerConnect(result);
|
||||
|
||||
// Store server URL and fetch available repos for the repo switcher
|
||||
await handleServerConnect(result);
|
||||
setProgress(null);
|
||||
setServerBaseUrl(baseUrl);
|
||||
try {
|
||||
const repos = await fetchRepos(baseUrl);
|
||||
setAvailableRepos(repos);
|
||||
} catch (e) {
|
||||
console.warn('Failed to fetch repo list:', e);
|
||||
}
|
||||
fetchRepos(baseUrl)
|
||||
.then((repos) => setAvailableRepos(repos))
|
||||
.catch((e) => console.warn('Failed to fetch repo list:', e));
|
||||
}).catch((err) => {
|
||||
console.error('Auto-connect failed:', err);
|
||||
setProgress({
|
||||
@@ -224,7 +218,7 @@ const AppContent = () => {
|
||||
setTimeout(() => {
|
||||
setViewMode('onboarding');
|
||||
setProgress(null);
|
||||
}, 3000);
|
||||
}, ERROR_RESET_DELAY_MS);
|
||||
});
|
||||
}, [handleServerConnect, setProgress, setViewMode, setServerBaseUrl, setAvailableRepos]);
|
||||
|
||||
@@ -246,16 +240,14 @@ const AppContent = () => {
|
||||
onFileSelect={handleFileSelect}
|
||||
onGitClone={handleGitClone}
|
||||
onServerConnect={async (result, serverUrl) => {
|
||||
handleServerConnect(result);
|
||||
await handleServerConnect(result);
|
||||
setProgress(null);
|
||||
if (serverUrl) {
|
||||
const baseUrl = normalizeServerUrl(serverUrl);
|
||||
setServerBaseUrl(baseUrl);
|
||||
try {
|
||||
const repos = await fetchRepos(baseUrl);
|
||||
setAvailableRepos(repos);
|
||||
} catch (e) {
|
||||
console.warn('Failed to fetch repo list:', e);
|
||||
}
|
||||
fetchRepos(baseUrl)
|
||||
.then((repos) => setAvailableRepos(repos))
|
||||
.catch((e) => console.warn('Failed to fetch repo list:', e));
|
||||
}
|
||||
}}
|
||||
/>
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
import { Server, ArrowRight } from 'lucide-react';
|
||||
import { Server, ArrowRight } from '@/lib/lucide-icons';
|
||||
import { BackendRepo } from '../services/backend';
|
||||
|
||||
interface BackendRepoSelectorProps {
|
||||
|
||||
@@ -1,10 +1,47 @@
|
||||
import { useCallback, useEffect, useMemo, useRef, useState } from 'react';
|
||||
import { Code, PanelLeftClose, PanelLeft, Trash2, X, Target, FileCode, Sparkles, MousePointerClick } from 'lucide-react';
|
||||
import { Code, PanelLeftClose, PanelLeft, Trash2, X, Target, FileCode, Sparkles, MousePointerClick } from '@/lib/lucide-icons';
|
||||
import { Prism as SyntaxHighlighter } from 'react-syntax-highlighter';
|
||||
import { vscDarkPlus } from 'react-syntax-highlighter/dist/esm/styles/prism';
|
||||
import { useAppState } from '../hooks/useAppState';
|
||||
import type { GraphNode } from '../core/graph/types';
|
||||
import { NODE_COLORS } from '../lib/constants';
|
||||
|
||||
/** Map file extension to Prism syntax highlighter language identifier */
|
||||
const getSyntaxLanguage = (filePath: string | undefined): string => {
|
||||
if (!filePath) return 'text';
|
||||
const ext = filePath.split('.').pop()?.toLowerCase();
|
||||
switch (ext) {
|
||||
case 'js': case 'jsx': case 'mjs': case 'cjs': return 'javascript';
|
||||
case 'ts': case 'tsx': case 'mts': case 'cts': return 'typescript';
|
||||
case 'py': case 'pyw': return 'python';
|
||||
case 'rb': case 'rake': case 'gemspec': return 'ruby';
|
||||
case 'java': return 'java';
|
||||
case 'go': return 'go';
|
||||
case 'rs': return 'rust';
|
||||
case 'c': case 'h': return 'c';
|
||||
case 'cpp': case 'cc': case 'cxx': case 'hpp': case 'hxx': case 'hh': return 'cpp';
|
||||
case 'cs': return 'csharp';
|
||||
case 'php': return 'php';
|
||||
case 'kt': case 'kts': return 'kotlin';
|
||||
case 'swift': return 'swift';
|
||||
case 'json': return 'json';
|
||||
case 'yaml': case 'yml': return 'yaml';
|
||||
case 'md': case 'mdx': return 'markdown';
|
||||
case 'html': case 'htm': case 'erb': return 'markup';
|
||||
case 'css': case 'scss': case 'sass': return 'css';
|
||||
case 'sh': case 'bash': case 'zsh': return 'bash';
|
||||
case 'sql': return 'sql';
|
||||
case 'xml': return 'xml';
|
||||
default: break;
|
||||
}
|
||||
// Handle extensionless Ruby files
|
||||
const basename = filePath.split('/').pop() || '';
|
||||
if (['Rakefile', 'Gemfile', 'Guardfile', 'Vagrantfile', 'Brewfile'].includes(basename)) return 'ruby';
|
||||
if (['Makefile'].includes(basename)) return 'makefile';
|
||||
if (['Dockerfile'].includes(basename)) return 'docker';
|
||||
return 'text';
|
||||
};
|
||||
|
||||
// Match the code theme used elsewhere in the app
|
||||
const customTheme = {
|
||||
...vscDarkPlus,
|
||||
@@ -39,6 +76,11 @@ export const CodeReferencesPanel = ({ onFocusNode }: CodeReferencesPanelProps) =
|
||||
codeReferenceFocus,
|
||||
} = useAppState();
|
||||
|
||||
const nodeById = useMemo(() => {
|
||||
if (!graph) return new Map<string, GraphNode>();
|
||||
return new Map(graph.nodes.map(n => [n.id, n]));
|
||||
}, [graph]);
|
||||
|
||||
const [isCollapsed, setIsCollapsed] = useState(false);
|
||||
const [glowRefId, setGlowRefId] = useState<string | null>(null);
|
||||
const panelRef = useRef<HTMLElement | null>(null);
|
||||
@@ -125,8 +167,9 @@ export const CodeReferencesPanel = ({ onFocusNode }: CodeReferencesPanelProps) =
|
||||
if (!target) return;
|
||||
|
||||
// Double rAF: wait for collapse state + list DOM to render.
|
||||
requestAnimationFrame(() => {
|
||||
requestAnimationFrame(() => {
|
||||
const rafIds: number[] = [];
|
||||
const outerRafId = requestAnimationFrame(() => {
|
||||
const innerRafId = requestAnimationFrame(() => {
|
||||
const el = refCardEls.current.get(target.id);
|
||||
if (!el) return;
|
||||
|
||||
@@ -141,7 +184,13 @@ export const CodeReferencesPanel = ({ onFocusNode }: CodeReferencesPanelProps) =
|
||||
glowTimerRef.current = null;
|
||||
}, 1200);
|
||||
});
|
||||
rafIds.push(innerRafId);
|
||||
});
|
||||
rafIds.push(outerRafId);
|
||||
|
||||
return () => {
|
||||
rafIds.forEach(id => cancelAnimationFrame(id));
|
||||
};
|
||||
}, [codeReferenceFocus?.ts, aiReferences]);
|
||||
|
||||
const refsWithSnippets = useMemo(() => {
|
||||
@@ -267,12 +316,7 @@ export const CodeReferencesPanel = ({ onFocusNode }: CodeReferencesPanelProps) =
|
||||
<div className="flex-1 min-h-0 overflow-auto scrollbar-thin">
|
||||
{selectedFileContent ? (
|
||||
<SyntaxHighlighter
|
||||
language={
|
||||
selectedFilePath?.endsWith('.py') ? 'python' :
|
||||
selectedFilePath?.endsWith('.js') || selectedFilePath?.endsWith('.jsx') ? 'javascript' :
|
||||
selectedFilePath?.endsWith('.ts') || selectedFilePath?.endsWith('.tsx') ? 'typescript' :
|
||||
'text'
|
||||
}
|
||||
language={getSyntaxLanguage(selectedFilePath)}
|
||||
style={customTheme as any}
|
||||
showLineNumbers
|
||||
startingLineNumber={1}
|
||||
@@ -339,11 +383,7 @@ export const CodeReferencesPanel = ({ onFocusNode }: CodeReferencesPanelProps) =
|
||||
const hasRange = typeof ref.startLine === 'number';
|
||||
const startDisplay = hasRange ? (ref.startLine ?? 0) + 1 : undefined;
|
||||
const endDisplay = hasRange ? (ref.endLine ?? ref.startLine ?? 0) + 1 : undefined;
|
||||
const language =
|
||||
ref.filePath.endsWith('.py') ? 'python' :
|
||||
ref.filePath.endsWith('.js') || ref.filePath.endsWith('.jsx') ? 'javascript' :
|
||||
ref.filePath.endsWith('.ts') || ref.filePath.endsWith('.tsx') ? 'typescript' :
|
||||
'text';
|
||||
const language = getSyntaxLanguage(ref.filePath);
|
||||
|
||||
const isGlowing = glowRefId === ref.id;
|
||||
|
||||
@@ -387,7 +427,7 @@ export const CodeReferencesPanel = ({ onFocusNode }: CodeReferencesPanelProps) =
|
||||
const nodeId = ref.nodeId!;
|
||||
// Sync selection + focus graph
|
||||
if (graph) {
|
||||
const node = graph.nodes.find((n) => n.id === nodeId);
|
||||
const node = nodeById.get(nodeId);
|
||||
if (node) setSelectedNode(node);
|
||||
}
|
||||
onFocusNode(nodeId);
|
||||
|
||||
@@ -1,12 +1,12 @@
|
||||
import { useState, useCallback, useRef, DragEvent } from 'react';
|
||||
import { Upload, FileArchive, Github, Loader2, ArrowRight, Key, Eye, EyeOff, Globe, X } from 'lucide-react';
|
||||
import { Upload, FileArchive, Github, Loader2, ArrowRight, Key, Eye, EyeOff, Globe, X } from '@/lib/lucide-icons';
|
||||
import { cloneRepository, parseGitHubUrl } from '../services/git-clone';
|
||||
import { connectToServer, type ConnectToServerResult } from '../services/server-connection';
|
||||
import { FileEntry } from '../services/zip';
|
||||
|
||||
interface DropZoneProps {
|
||||
onFileSelect: (file: File) => void;
|
||||
onGitClone?: (files: FileEntry[]) => void;
|
||||
onGitClone?: (files: FileEntry[], repoName?: string) => void;
|
||||
onServerConnect?: (result: ConnectToServerResult, serverUrl?: string) => void;
|
||||
}
|
||||
|
||||
@@ -27,9 +27,13 @@ export const DropZone = ({ onFileSelect, onGitClone, onServerConnect }: DropZone
|
||||
const [error, setError] = useState<string | null>(null);
|
||||
|
||||
// Server tab state
|
||||
const [serverUrl, setServerUrl] = useState(() =>
|
||||
localStorage.getItem('gitnexus-server-url') || ''
|
||||
);
|
||||
const [serverUrl, setServerUrl] = useState(() => {
|
||||
try {
|
||||
return localStorage.getItem('gitnexus-server-url') || '';
|
||||
} catch {
|
||||
return '';
|
||||
}
|
||||
});
|
||||
const [isConnecting, setIsConnecting] = useState(false);
|
||||
const [serverProgress, setServerProgress] = useState<{
|
||||
phase: string;
|
||||
@@ -104,7 +108,7 @@ export const DropZone = ({ onFileSelect, onGitClone, onServerConnect }: DropZone
|
||||
setGithubToken('');
|
||||
|
||||
if (onGitClone) {
|
||||
onGitClone(files);
|
||||
onGitClone(files, parsed.repo);
|
||||
}
|
||||
} catch (err) {
|
||||
console.error('Clone failed:', err);
|
||||
@@ -133,7 +137,11 @@ export const DropZone = ({ onFileSelect, onGitClone, onServerConnect }: DropZone
|
||||
}
|
||||
|
||||
// Persist URL to localStorage
|
||||
localStorage.setItem('gitnexus-server-url', serverUrl);
|
||||
try {
|
||||
localStorage.setItem('gitnexus-server-url', serverUrl);
|
||||
} catch {
|
||||
// localStorage may be unavailable (e.g. private browsing, quota exceeded)
|
||||
}
|
||||
|
||||
setError(null);
|
||||
setIsConnecting(true);
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
import { Brain, Loader2, Check, AlertCircle, Zap, FlaskConical } from 'lucide-react';
|
||||
import { Brain, Loader2, Check, AlertCircle, Zap, FlaskConical } from '@/lib/lucide-icons';
|
||||
import { useAppState } from '../hooks/useAppState';
|
||||
import { useState } from 'react';
|
||||
import { WebGPUFallbackDialog } from './WebGPUFallbackDialog';
|
||||
@@ -83,7 +83,7 @@ export const EmbeddingStatus = () => {
|
||||
<button
|
||||
onClick={handleTestArrayParams}
|
||||
className="flex items-center gap-1 px-2 py-1.5 bg-surface border border-border-subtle rounded-lg text-xs text-text-muted hover:bg-hover hover:text-text-secondary transition-all"
|
||||
title="Test if KuzuDB supports array params"
|
||||
title="Test if LadybugDB supports array params"
|
||||
>
|
||||
<FlaskConical className="w-3 h-3" />
|
||||
{testResult || 'Test'}
|
||||
|
||||
@@ -14,7 +14,7 @@ import {
|
||||
Variable,
|
||||
Hash,
|
||||
Target,
|
||||
} from 'lucide-react';
|
||||
} from '@/lib/lucide-icons';
|
||||
import { useAppState } from '../hooks/useAppState';
|
||||
import { FILTERABLE_LABELS, NODE_COLORS, ALL_EDGE_TYPES, EDGE_INFO, type EdgeType } from '../lib/constants';
|
||||
import { GraphNode, NodeLabel } from '../core/graph/types';
|
||||
@@ -98,13 +98,15 @@ const TreeItem = ({
|
||||
const isSelected = selectedPath === node.path;
|
||||
const hasChildren = node.children.length > 0;
|
||||
|
||||
// Filter children based on search
|
||||
// Filter children based on search (recursive)
|
||||
const filteredChildren = useMemo(() => {
|
||||
if (!searchQuery) return node.children;
|
||||
return node.children.filter(child =>
|
||||
child.name.toLowerCase().includes(searchQuery.toLowerCase()) ||
|
||||
child.children.some(c => c.name.toLowerCase().includes(searchQuery.toLowerCase()))
|
||||
);
|
||||
const searchLower = searchQuery.toLowerCase();
|
||||
const matchesSearch = (node: TreeNode, query: string): boolean => {
|
||||
if (node.name.toLowerCase().includes(query)) return true;
|
||||
return node.children?.some(child => matchesSearch(child, query)) ?? false;
|
||||
};
|
||||
return node.children.filter(child => matchesSearch(child, searchLower));
|
||||
}, [node.children, searchQuery]);
|
||||
|
||||
// Check if this node matches search
|
||||
|
||||
@@ -1,8 +1,9 @@
|
||||
import { useEffect, useCallback, useMemo, useState, forwardRef, useImperativeHandle } from 'react';
|
||||
import { ZoomIn, ZoomOut, Maximize2, Focus, RotateCcw, Play, Pause, Lightbulb, LightbulbOff } from 'lucide-react';
|
||||
import { ZoomIn, ZoomOut, Maximize2, Focus, RotateCcw, Play, Pause, Lightbulb, LightbulbOff } from '@/lib/lucide-icons';
|
||||
import { useSigma } from '../hooks/useSigma';
|
||||
import { useAppState } from '../hooks/useAppState';
|
||||
import { knowledgeGraphToGraphology, filterGraphByDepth, SigmaNodeAttributes, SigmaEdgeAttributes } from '../lib/graph-adapter';
|
||||
import type { GraphNode } from '../core/graph/types';
|
||||
import { QueryFAB } from './QueryFAB';
|
||||
import Graph from 'graphology';
|
||||
|
||||
@@ -26,6 +27,9 @@ export const GraphCanvas = forwardRef<GraphCanvasHandle>((_, ref) => {
|
||||
blastRadiusNodeIds,
|
||||
isAIHighlightsEnabled,
|
||||
toggleAIHighlights,
|
||||
clearAIToolHighlights,
|
||||
clearAICitationHighlights,
|
||||
clearBlastRadius,
|
||||
animatedNodes,
|
||||
} = useAppState();
|
||||
const [hoveredNodeName, setHoveredNodeName] = useState<string | null>(null);
|
||||
@@ -51,30 +55,44 @@ export const GraphCanvas = forwardRef<GraphCanvasHandle>((_, ref) => {
|
||||
return animatedNodes;
|
||||
}, [animatedNodes, isAIHighlightsEnabled]);
|
||||
|
||||
const nodeById = useMemo(() => {
|
||||
if (!graph) return new Map<string, GraphNode>();
|
||||
return new Map(graph.nodes.map(n => [n.id, n]));
|
||||
}, [graph]);
|
||||
|
||||
const handleNodeClick = useCallback((nodeId: string) => {
|
||||
if (!graph) return;
|
||||
const node = graph.nodes.find(n => n.id === nodeId);
|
||||
const node = nodeById.get(nodeId);
|
||||
if (node) {
|
||||
setSelectedNode(node);
|
||||
openCodePanel();
|
||||
}
|
||||
}, [graph, setSelectedNode, openCodePanel]);
|
||||
}, [graph, nodeById, setSelectedNode, openCodePanel]);
|
||||
|
||||
const handleNodeHover = useCallback((nodeId: string | null) => {
|
||||
if (!nodeId || !graph) {
|
||||
setHoveredNodeName(null);
|
||||
return;
|
||||
}
|
||||
const node = graph.nodes.find(n => n.id === nodeId);
|
||||
if (node) {
|
||||
setHoveredNodeName(node.properties.name);
|
||||
}
|
||||
}, [graph]);
|
||||
const node = nodeById.get(nodeId);
|
||||
setHoveredNodeName(node ? node.properties.name : null);
|
||||
}, [graph, nodeById]);
|
||||
|
||||
const handleStageClick = useCallback(() => {
|
||||
setSelectedNode(null);
|
||||
}, [setSelectedNode]);
|
||||
|
||||
const handleToggleAIHighlights = useCallback(() => {
|
||||
if (isAIHighlightsEnabled) {
|
||||
clearAIToolHighlights();
|
||||
clearAICitationHighlights();
|
||||
clearBlastRadius();
|
||||
setSelectedNode(null);
|
||||
setSigmaSelectedNode(null);
|
||||
}
|
||||
toggleAIHighlights();
|
||||
}, [isAIHighlightsEnabled, clearAIToolHighlights, clearAICitationHighlights, clearBlastRadius, setSelectedNode, toggleAIHighlights]);
|
||||
|
||||
const {
|
||||
containerRef,
|
||||
sigmaRef,
|
||||
@@ -103,7 +121,7 @@ export const GraphCanvas = forwardRef<GraphCanvasHandle>((_, ref) => {
|
||||
focusNode: (nodeId: string) => {
|
||||
// Also update app state so the selection syncs properly
|
||||
if (graph) {
|
||||
const node = graph.nodes.find(n => n.id === nodeId);
|
||||
const node = nodeById.get(nodeId);
|
||||
if (node) {
|
||||
setSelectedNode(node);
|
||||
openCodePanel();
|
||||
@@ -111,7 +129,7 @@ export const GraphCanvas = forwardRef<GraphCanvasHandle>((_, ref) => {
|
||||
}
|
||||
focusNode(nodeId);
|
||||
}
|
||||
}), [focusNode, graph, setSelectedNode, openCodePanel]);
|
||||
}), [focusNode, graph, nodeById, setSelectedNode, openCodePanel]);
|
||||
|
||||
// Update Sigma graph when KnowledgeGraph changes
|
||||
useEffect(() => {
|
||||
@@ -123,10 +141,11 @@ export const GraphCanvas = forwardRef<GraphCanvasHandle>((_, ref) => {
|
||||
graph.relationships.forEach(rel => {
|
||||
if (rel.type === 'MEMBER_OF') {
|
||||
// Find the community node to get its index
|
||||
const communityNode = graph.nodes.find(n => n.id === rel.targetId && n.label === 'Community');
|
||||
if (communityNode) {
|
||||
const communityNode = nodeById.get(rel.targetId);
|
||||
if (communityNode && communityNode.label === 'Community') {
|
||||
// Extract community index from id (e.g., "comm_5" -> 5)
|
||||
const communityIdx = parseInt(rel.targetId.replace('comm_', ''), 10) || 0;
|
||||
const numericPart = rel.targetId.replace('comm_', '');
|
||||
const communityIdx = /^\d+$/.test(numericPart) ? parseInt(numericPart, 10) : 0;
|
||||
communityMemberships.set(rel.sourceId, communityIdx);
|
||||
}
|
||||
}
|
||||
@@ -134,7 +153,7 @@ export const GraphCanvas = forwardRef<GraphCanvasHandle>((_, ref) => {
|
||||
|
||||
const sigmaGraph = knowledgeGraphToGraphology(graph, communityMemberships);
|
||||
setSigmaGraph(sigmaGraph);
|
||||
}, [graph, setSigmaGraph]);
|
||||
}, [graph, nodeById, setSigmaGraph]);
|
||||
|
||||
// Update node visibility when filters change
|
||||
useEffect(() => {
|
||||
@@ -146,7 +165,8 @@ export const GraphCanvas = forwardRef<GraphCanvasHandle>((_, ref) => {
|
||||
|
||||
filterGraphByDepth(sigmaGraph, appSelectedNode?.id || null, depthFilter, visibleLabels);
|
||||
sigma.refresh();
|
||||
}, [visibleLabels, depthFilter, appSelectedNode, sigmaRef]);
|
||||
// eslint-disable-next-line react-hooks/exhaustive-deps -- sigmaRef identity never changes
|
||||
}, [visibleLabels, depthFilter, appSelectedNode]);
|
||||
|
||||
// Sync app selected node with sigma
|
||||
useEffect(() => {
|
||||
@@ -304,19 +324,14 @@ export const GraphCanvas = forwardRef<GraphCanvasHandle>((_, ref) => {
|
||||
{/* AI Highlights toggle - Top Right */}
|
||||
<div className="absolute top-4 right-4 z-20">
|
||||
<button
|
||||
onClick={() => {
|
||||
// If turning off, also clear process highlights
|
||||
if (isAIHighlightsEnabled) {
|
||||
setHighlightedNodeIds(new Set());
|
||||
}
|
||||
toggleAIHighlights();
|
||||
}}
|
||||
onClick={handleToggleAIHighlights}
|
||||
className={
|
||||
isAIHighlightsEnabled
|
||||
? 'w-10 h-10 flex items-center justify-center bg-cyan-500/15 border border-cyan-400/40 rounded-lg text-cyan-200 hover:bg-cyan-500/20 hover:border-cyan-300/60 transition-colors'
|
||||
: 'w-10 h-10 flex items-center justify-center bg-elevated border border-border-subtle rounded-lg text-text-muted hover:bg-hover hover:text-text-primary transition-colors'
|
||||
}
|
||||
title={isAIHighlightsEnabled ? 'Turn off all highlights' : 'Turn on AI highlights'}
|
||||
data-testid="ai-highlights-toggle"
|
||||
>
|
||||
{isAIHighlightsEnabled ? <Lightbulb className="w-4 h-4" /> : <LightbulbOff className="w-4 h-4" />}
|
||||
</button>
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
import { Search, Settings, HelpCircle, Sparkles, Github, Star, ChevronDown } from 'lucide-react';
|
||||
import { Search, Settings, HelpCircle, Sparkles, Github, Star, ChevronDown } from '@/lib/lucide-icons';
|
||||
import { useAppState } from '../hooks/useAppState';
|
||||
import type { RepoSummary } from '../services/server-connection';
|
||||
import { useState, useMemo, useRef, useEffect, useCallback } from 'react';
|
||||
@@ -32,6 +32,7 @@ export const Header = ({ onFocusNode, availableRepos = [], onSwitchRepo }: Heade
|
||||
isRightPanelOpen,
|
||||
rightPanelTab,
|
||||
setSettingsPanelOpen,
|
||||
setHelpDialogBoxOpen
|
||||
} = useAppState();
|
||||
const [isRepoDropdownOpen, setIsRepoDropdownOpen] = useState(false);
|
||||
const repoDropdownRef = useRef<HTMLDivElement>(null);
|
||||
@@ -266,10 +267,13 @@ export const Header = ({ onFocusNode, availableRepos = [], onSwitchRepo }: Heade
|
||||
className="w-9 h-9 flex items-center justify-center rounded-md text-text-secondary hover:bg-hover hover:text-text-primary transition-colors"
|
||||
title="AI Settings"
|
||||
>
|
||||
<Settings className="w-[18px] h-[18px]" />
|
||||
<Settings className="w-4.5 h-4.5" />
|
||||
</button>
|
||||
<button className="w-9 h-9 flex items-center justify-center rounded-md text-text-secondary hover:bg-hover hover:text-text-primary transition-colors">
|
||||
<HelpCircle className="w-[18px] h-[18px]" />
|
||||
<button
|
||||
title="Help"
|
||||
onClick={() => setHelpDialogBoxOpen(true)}
|
||||
className="w-9 h-9 flex items-center justify-center rounded-md text-text-secondary hover:bg-hover hover:text-text-primary transition-colors">
|
||||
<HelpCircle className="w-4.5 h-4.5" />
|
||||
</button>
|
||||
|
||||
{/* AI Button */}
|
||||
|
||||
@@ -0,0 +1,390 @@
|
||||
import React, { useState } from 'react';
|
||||
import { X, GitBranch, Search, Filter, Zap, Keyboard, BarChart2, HelpCircle } from 'lucide-react';
|
||||
|
||||
interface HelpPanelProps {
|
||||
isOpen: boolean;
|
||||
onClose: () => void;
|
||||
nodeCount: number;
|
||||
edgeCount: number;
|
||||
}
|
||||
|
||||
type TabId = 'overview' | 'graph' | 'search' | 'ai' | 'shortcuts' | 'status';
|
||||
|
||||
interface Tab {
|
||||
id: TabId;
|
||||
label: string;
|
||||
icon: React.ReactNode;
|
||||
}
|
||||
|
||||
const tabs: Tab[] = [
|
||||
{ id: 'overview', label: 'Overview', icon: <HelpCircle className="w-4 h-4" /> },
|
||||
{ id: 'graph', label: 'Graph & nodes', icon: <GitBranch className="w-4 h-4" /> },
|
||||
{ id: 'search', label: 'Search & filter', icon: <Search className="w-4 h-4" /> },
|
||||
{ id: 'ai', label: 'Nexus AI', icon: <Zap className="w-4 h-4" /> },
|
||||
{ id: 'shortcuts', label: 'Shortcuts', icon: <Keyboard className="w-4 h-4" /> },
|
||||
{ id: 'status', label: 'Status bar', icon: <BarChart2 className="w-4 h-4" /> },
|
||||
];
|
||||
|
||||
const shortcuts = [
|
||||
{ label: 'Search nodes', mac: '⌘ K', win: 'Ctrl K' },
|
||||
{ label: 'Deselect / close', mac: 'Esc', win: 'Esc' },
|
||||
];
|
||||
|
||||
const nodeColors = [
|
||||
{ color: '#10b981', label: 'Function', desc: 'Function declarations' },
|
||||
{ color: '#3b82f6', label: 'File', desc: 'Source files' },
|
||||
{ color: '#f59e0b', label: 'Class', desc: 'Class declarations' },
|
||||
{ color: '#14b8a6', label: 'Method', desc: 'Class methods' },
|
||||
{ color: '#ec4899', label: 'Interface', desc: 'TypeScript interfaces' },
|
||||
{ color: '#6366f1', label: 'Folder', desc: 'Directory nodes' },
|
||||
];
|
||||
|
||||
const getStatusItems = (nodeCount: number, edgeCount: number) => [
|
||||
{ badge: <span style={{ width: 8, height: 8, borderRadius: '50%', background: '#34d399', display: 'inline-block', flexShrink: 0 }} />, title: 'Ready', desc: 'Graph is fully loaded and interactive' },
|
||||
{ badge: <span style={{ fontSize: 12, fontWeight: 500, color: '#a78bfa', flexShrink: 0 }}>{nodeCount}</span>, title: 'Nodes count', desc: 'Total files and symbols in the graph' },
|
||||
{ badge: <span style={{ fontSize: 12, fontWeight: 500, color: '#60a5fa', flexShrink: 0 }}>{edgeCount}</span>, title: 'Edges count', desc: 'Import / dependency connections' },
|
||||
{ badge: <span style={{ fontSize: 11, fontWeight: 500, color: '#34d399', flexShrink: 0, whiteSpace: 'nowrap' }}>Semantic Ready</span>, title: 'AI index status', desc: 'Repo is fully indexed for AI queries' },
|
||||
// { badge: <span style={{ fontSize: 11, fontWeight: 500, color: '#9ca3af', flexShrink: 0 }}>typescript</span>, title: 'Language', desc: 'Primary language detected in the repo' },
|
||||
];
|
||||
|
||||
const kbdStyle: React.CSSProperties = {
|
||||
fontSize: 11,
|
||||
background: 'rgba(255,255,255,0.08)',
|
||||
borderRadius: 4,
|
||||
padding: '2px 8px',
|
||||
color: '#e2e2e8',
|
||||
fontFamily: 'monospace',
|
||||
border: '0.5px solid rgba(255,255,255,0.12)',
|
||||
whiteSpace: 'nowrap',
|
||||
};
|
||||
|
||||
const kbdWinStyle: React.CSSProperties = {
|
||||
...kbdStyle,
|
||||
color: '#93c5fd',
|
||||
};
|
||||
|
||||
function TabContent({ active, nodeCount, edgeCount }: {
|
||||
active: TabId;
|
||||
nodeCount: number;
|
||||
edgeCount: number;
|
||||
}) {
|
||||
if (active === 'overview') return (
|
||||
<div style={{ display: 'flex', flexDirection: 'column', gap: 10 }}>
|
||||
<p style={{ fontSize: 11, color: '#6b7280', margin: '0 0 4px', textTransform: 'uppercase', letterSpacing: '0.08em' }}>Getting started</p>
|
||||
|
||||
<div style={{ background: 'rgba(255,255,255,0.04)', borderRadius: 10, padding: '12px 14px', borderLeft: '2px solid #a78bfa' }}>
|
||||
<p style={{ fontSize: 13, fontWeight: 500, color: '#e2e2e8', margin: '0 0 4px' }}>What is GitNexus?</p>
|
||||
<p style={{ fontSize: 12, color: '#9ca3af', margin: 0, lineHeight: 1.6 }}>An interactive graph explorer for your codebase. Every file, function, and import becomes a node you can explore, query, and navigate visually.</p>
|
||||
</div>
|
||||
|
||||
<div style={{ background: 'rgba(255,255,255,0.04)', borderRadius: 10, padding: '12px 14px', borderLeft: '2px solid #34d399' }}>
|
||||
<p style={{ fontSize: 13, fontWeight: 500, color: '#e2e2e8', margin: '0 0 4px' }}>Your current repo</p>
|
||||
<p style={{ fontSize: 12, color: '#9ca3af', margin: 0, lineHeight: 1.6 }}>
|
||||
Loaded: <span style={{ color: '#a78bfa', fontFamily: 'monospace' }}></span> {nodeCount} nodes · {edgeCount} edges
|
||||
</p>
|
||||
</div>
|
||||
|
||||
<div style={{ background: 'rgba(255,255,255,0.04)', borderRadius: 10, padding: '12px 14px', borderLeft: '2px solid #60a5fa' }}>
|
||||
<p style={{ fontSize: 13, fontWeight: 500, color: '#e2e2e8', margin: '0 0 4px' }}>Three ways to explore</p>
|
||||
<p style={{ fontSize: 12, color: '#9ca3af', margin: 0, lineHeight: 1.6 }}>
|
||||
<strong style={{ color: '#e2e2e8', fontWeight: 500 }}>1.</strong> Click nodes to inspect
|
||||
<br/>
|
||||
<strong style={{ color: '#e2e2e8', fontWeight: 500 }}>2.</strong> Search by name or type
|
||||
<br/>
|
||||
<strong style={{ color: '#e2e2e8', fontWeight: 500 }}>3.</strong> Ask Nexus AI a natural language question
|
||||
</p>
|
||||
</div>
|
||||
|
||||
<div style={{ background: 'rgba(255,255,255,0.04)', borderRadius: 10, padding: '12px 14px', borderLeft: '2px solid #fbbf24' }}>
|
||||
<p style={{ fontSize: 13, fontWeight: 500, color: '#e2e2e8', margin: '0 0 4px' }}>Navigation</p>
|
||||
<p style={{ fontSize: 12, color: '#9ca3af', margin: 0, lineHeight: 1.6 }}>
|
||||
· Scroll to zoom <br/>
|
||||
· Click and drag to pan <br/>
|
||||
· Double-click a node to focus its subgraph
|
||||
</p>
|
||||
</div>
|
||||
</div>
|
||||
);
|
||||
|
||||
if (active === 'graph') return (
|
||||
<div style={{ display: 'flex', flexDirection: 'column', gap: 12 }}>
|
||||
<p style={{ fontSize: 11, color: '#6b7280', margin: '0 0 4px', textTransform: 'uppercase', letterSpacing: '0.08em' }}>Node color legend</p>
|
||||
|
||||
{nodeColors.map(({ color, label, desc }) => (
|
||||
<div key={label} style={{ display: 'flex', gap: 10, alignItems: 'flex-start' }}>
|
||||
<span style={{ width: 12, height: 12, borderRadius: '50%', background: color, flexShrink: 0, marginTop: 2 }} />
|
||||
<div>
|
||||
<p style={{ fontSize: 12, fontWeight: 500, color: '#e2e2e8', margin: '0 0 2px' }}>{label} nodes</p>
|
||||
<p style={{ fontSize: 12, color: '#9ca3af', margin: 0 }}>{desc}</p>
|
||||
</div>
|
||||
</div>
|
||||
))}
|
||||
|
||||
<div style={{ borderTop: '0.5px solid rgba(255,255,255,0.08)', margin: '4px 0' }} />
|
||||
|
||||
<p style={{ fontSize: 12, color: '#9ca3af', margin: 0, lineHeight: 1.6 }}>
|
||||
Node <strong style={{ color: '#e2e2e8', fontWeight: 500 }}>size</strong> reflects connection count — larger nodes are depended on by more files. Edges point from importer → imported.
|
||||
</p>
|
||||
|
||||
<div style={{ background: 'rgba(255,255,255,0.04)', borderRadius: 10, padding: '10px 14px' }}>
|
||||
<p style={{ fontSize: 12, color: '#9ca3af', margin: 0, lineHeight: 1.6 }}>
|
||||
Click any node to open its detail panel — showing imports, exports, and reverse dependencies.
|
||||
</p>
|
||||
</div>
|
||||
</div>
|
||||
);
|
||||
|
||||
if (active === 'search') return (
|
||||
<div style={{ display: 'flex', flexDirection: 'column', gap: 10 }}>
|
||||
<p style={{ fontSize: 11, color: '#6b7280', margin: '0 0 4px', textTransform: 'uppercase', letterSpacing: '0.08em' }}>Search & filter</p>
|
||||
|
||||
<div style={{ background: 'rgba(255,255,255,0.04)', borderRadius: 10, padding: '12px 14px' }}>
|
||||
<div style={{ display: 'flex', alignItems: 'center', gap: 8, marginBottom: 6 }}>
|
||||
<kbd style={kbdStyle}>⌘K</kbd>/
|
||||
<kbd style={kbdStyle}>Ctrl K</kbd>
|
||||
<p style={{ fontSize: 12, fontWeight: 500, color: '#e2e2e8', margin: 0 }}>Search nodes</p>
|
||||
</div>
|
||||
<p style={{ fontSize: 12, color: '#9ca3af', margin: 0, lineHeight: 1.6 }}>
|
||||
Search by filename, function name, or import path. Matching nodes are highlighted live in the graph.
|
||||
</p>
|
||||
</div>
|
||||
|
||||
<div style={{ background: 'rgba(255,255,255,0.04)', borderRadius: 10, padding: '12px 14px' }}>
|
||||
<div style={{ display: 'flex', alignItems: 'center', gap: 8, marginBottom: 6 }}>
|
||||
<Filter style={{ width: 14, height: 14, color: '#a78bfa', flexShrink: 0 }} />
|
||||
<p style={{ fontSize: 12, fontWeight: 500, color: '#e2e2e8', margin: 0 }}>Filter panel</p>
|
||||
</div>
|
||||
<p style={{ fontSize: 12, color: '#9ca3af', margin: 0, lineHeight: 1.6 }}>
|
||||
Use the filter icon in the left sidebar to isolate specific node types, hide leaf nodes, or focus on a depth range from a selected root.
|
||||
</p>
|
||||
</div>
|
||||
|
||||
<div style={{ background: 'rgba(255,255,255,0.04)', borderRadius: 10, padding: '12px 14px' }}>
|
||||
<p style={{ fontSize: 12, fontWeight: 500, color: '#e2e2e8', margin: '0 0 6px' }}>Search syntax</p>
|
||||
{[
|
||||
{ query: 'auth', hint: 'match by name fragment' },
|
||||
{ query: './utils/', hint: 'match by path prefix' },
|
||||
{ query: 'type:config', hint: 'filter by node type' },
|
||||
].map(({ query, hint }) => (
|
||||
<div key={query} style={{ display: 'flex', alignItems: 'baseline', gap: 8, marginBottom: 4 }}>
|
||||
<code style={{ fontSize: 11, color: '#a78bfa', background: 'rgba(167,139,250,0.1)', borderRadius: 4, padding: '1px 6px', fontFamily: 'monospace', flexShrink: 0 }}>{query}</code>
|
||||
<span style={{ fontSize: 12, color: '#6b7280' }}>{hint}</span>
|
||||
</div>
|
||||
))}
|
||||
</div>
|
||||
</div>
|
||||
);
|
||||
|
||||
if (active === 'ai') return (
|
||||
<div style={{ display: 'flex', flexDirection: 'column', gap: 10 }}>
|
||||
<p style={{ fontSize: 11, color: '#6b7280', margin: '0 0 4px', textTransform: 'uppercase', letterSpacing: '0.08em' }}>Nexus AI</p>
|
||||
|
||||
<div style={{ background: 'rgba(167,139,250,0.08)', border: '0.5px solid rgba(167,139,250,0.25)', borderRadius: 10, padding: '12px 14px' }}>
|
||||
<p style={{ fontSize: 12, fontWeight: 500, color: '#a78bfa', margin: '0 0 4px' }}>✓ Semantic Ready</p>
|
||||
<p style={{ fontSize: 12, color: '#9ca3af', margin: 0, lineHeight: 1.6 }}>
|
||||
Your repo is indexed and ready for semantic queries. Nexus AI understands code structure and relationships, not just file names.
|
||||
</p>
|
||||
</div>
|
||||
|
||||
<p style={{ fontSize: 12, color: '#9ca3af', margin: '4px 0 2px' }}>Try asking:</p>
|
||||
{[
|
||||
'"Which files depend on the auth module?"',
|
||||
'"Find circular dependencies in this repo"',
|
||||
'"What are the most connected components?"',
|
||||
'"Show me all files that import useEffect"',
|
||||
].map(q => (
|
||||
<div key={q} style={{ background: 'rgba(255,255,255,0.04)', borderRadius: 8, padding: '8px 12px', fontSize: 12, color: '#e2e2e8', fontStyle: 'italic' }}>{q}</div>
|
||||
))}
|
||||
|
||||
<div style={{ borderTop: '0.5px solid rgba(255,255,255,0.08)', margin: '4px 0' }} />
|
||||
|
||||
<p style={{ fontSize: 12, color: '#6b7280', margin: 0, lineHeight: 1.6 }}>
|
||||
Open the prompt via the{' '}
|
||||
<span style={{ color: '#e2e2e8' }}>Nexus AI</span> button (top-right).
|
||||
</p>
|
||||
</div>
|
||||
);
|
||||
|
||||
if (active === 'shortcuts') return (
|
||||
<div style={{ display: 'flex', flexDirection: 'column', gap: 0 }}>
|
||||
{/* Column headers */}
|
||||
<div style={{
|
||||
display: 'grid',
|
||||
gridTemplateColumns: '1fr 80px 88px',
|
||||
gap: 8,
|
||||
padding: '0 0 8px',
|
||||
borderBottom: '0.5px solid rgba(255,255,255,0.08)',
|
||||
marginBottom: 4,
|
||||
}}>
|
||||
<span style={{ fontSize: 11, color: '#6b7280', textTransform: 'uppercase', letterSpacing: '0.08em' }}>Action</span>
|
||||
<span style={{ fontSize: 11, color: '#6b7280', textTransform: 'uppercase', letterSpacing: '0.08em', textAlign: 'center' }}>Mac</span>
|
||||
<span style={{ fontSize: 11, color: '#93c5fd', textTransform: 'uppercase', letterSpacing: '0.08em', textAlign: 'center' }}>Windows</span>
|
||||
</div>
|
||||
|
||||
{shortcuts.map(({ label, mac, win }, i) => (
|
||||
<div
|
||||
key={label}
|
||||
style={{
|
||||
display: 'grid',
|
||||
gridTemplateColumns: '1fr 80px 88px',
|
||||
gap: 8,
|
||||
alignItems: 'center',
|
||||
padding: '8px 0',
|
||||
borderBottom: i < shortcuts.length - 1 ? '0.5px solid rgba(255,255,255,0.05)' : 'none',
|
||||
}}
|
||||
>
|
||||
<span style={{ fontSize: 12, color: '#9ca3af' }}>{label}</span>
|
||||
<span style={{ display: 'flex', justifyContent: 'center' }}>
|
||||
<kbd style={kbdStyle}>{mac}</kbd>
|
||||
</span>
|
||||
<span style={{ display: 'flex', justifyContent: 'center' }}>
|
||||
<kbd style={kbdWinStyle}>{win}</kbd>
|
||||
</span>
|
||||
</div>
|
||||
))}
|
||||
</div>
|
||||
);
|
||||
|
||||
if (active === 'status') return (
|
||||
<div style={{ display: 'flex', flexDirection: 'column', gap: 8 }}>
|
||||
<p style={{ fontSize: 11, color: '#6b7280', margin: '0 0 4px', textTransform: 'uppercase', letterSpacing: '0.08em' }}>Status bar explained</p>
|
||||
{getStatusItems(nodeCount, edgeCount).map(({ badge, title, desc }) => (
|
||||
<div key={title} style={{ background: 'rgba(255,255,255,0.04)', borderRadius: 10, padding: '10px 14px', display: 'flex', gap: 12, alignItems: 'center' }}>
|
||||
{badge}
|
||||
<div>
|
||||
<p style={{ fontSize: 12, fontWeight: 500, color: '#e2e2e8', margin: '0 0 2px' }}>{title}</p>
|
||||
<p style={{ fontSize: 12, color: '#9ca3af', margin: 0 }}>{desc}</p>
|
||||
</div>
|
||||
</div>
|
||||
))}
|
||||
</div>
|
||||
);
|
||||
|
||||
return null;
|
||||
}
|
||||
|
||||
export const HelpPanel = ({ isOpen, onClose, nodeCount, edgeCount }: HelpPanelProps) => {
|
||||
const [active, setActive] = useState<TabId>('overview');
|
||||
|
||||
if (!isOpen) return null;
|
||||
|
||||
return (
|
||||
<div style={{ position: 'fixed', inset: 0, zIndex: 50, display: 'flex', alignItems: 'center', justifyContent: 'center' }}>
|
||||
{/* Backdrop */}
|
||||
<div
|
||||
style={{ position: 'absolute', inset: 0, background: 'rgba(0,0,0,0.6)', backdropFilter: 'blur(4px)' }}
|
||||
onClick={onClose}
|
||||
/>
|
||||
|
||||
{/* Panel */}
|
||||
<div style={{
|
||||
position: 'relative',
|
||||
background: '#12121a',
|
||||
border: '0.5px solid rgba(255,255,255,0.12)',
|
||||
borderRadius: 16,
|
||||
boxShadow: '0 25px 60px rgba(0,0,0,0.7)',
|
||||
width: '100%',
|
||||
maxWidth: 680,
|
||||
margin: '0 16px',
|
||||
height: '60vh',
|
||||
display: 'flex',
|
||||
flexDirection: 'column',
|
||||
overflow: 'hidden',
|
||||
fontFamily: 'var(--font-mono, monospace)',
|
||||
}}>
|
||||
|
||||
{/* Header */}
|
||||
<div style={{
|
||||
display: 'flex',
|
||||
alignItems: 'center',
|
||||
justifyContent: 'space-between',
|
||||
padding: '16px 20px',
|
||||
borderBottom: '0.5px solid rgba(255,255,255,0.08)',
|
||||
background: 'rgba(255,255,255,0.02)',
|
||||
}}>
|
||||
<div style={{ display: 'flex', alignItems: 'center', gap: 12 }}>
|
||||
<div style={{ width: 40, height: 40, display: 'flex', alignItems: 'center', justifyContent: 'center', background: 'rgba(167,139,250,0.15)', borderRadius: 12 }}>
|
||||
<HelpCircle style={{ width: 20, height: 20, color: '#a78bfa' }} />
|
||||
</div>
|
||||
<div>
|
||||
<h2 style={{ fontSize: 16, fontWeight: 600, color: '#e2e2e8', margin: 0 }}>Help & Reference</h2>
|
||||
<p style={{ fontSize: 12, color: '#6b7280', margin: 0 }}>GitNexus — graph explorer</p>
|
||||
</div>
|
||||
</div>
|
||||
<button
|
||||
onClick={onClose}
|
||||
style={{ padding: 8, color: '#6b7280', background: 'transparent', border: 'none', borderRadius: 8, cursor: 'pointer', display: 'flex', alignItems: 'center', justifyContent: 'center', transition: 'color 0.15s' }}
|
||||
onMouseEnter={e => (e.currentTarget.style.color = '#e2e2e8')}
|
||||
onMouseLeave={e => (e.currentTarget.style.color = '#6b7280')}
|
||||
>
|
||||
<X style={{ width: 20, height: 20 }} />
|
||||
</button>
|
||||
</div>
|
||||
|
||||
{/* Body: sidebar + content */}
|
||||
<div style={{ display: 'grid', gridTemplateColumns: '168px 1fr', flex: 1, overflow: 'hidden' }}>
|
||||
|
||||
{/* Sidebar nav */}
|
||||
<div style={{ borderRight: '0.5px solid rgba(255,255,255,0.08)', padding: '12px 8px', display: 'flex', flexDirection: 'column', gap: 2 }}>
|
||||
{tabs.map(({ id, label, icon }) => {
|
||||
const isActive = active === id;
|
||||
return (
|
||||
<button
|
||||
key={id}
|
||||
onClick={() => setActive(id)}
|
||||
style={{
|
||||
display: 'flex',
|
||||
alignItems: 'center',
|
||||
gap: 8,
|
||||
textAlign: 'left',
|
||||
background: isActive ? 'rgba(167,139,250,0.12)' : 'transparent',
|
||||
border: 'none',
|
||||
borderRadius: 8,
|
||||
padding: '8px 10px',
|
||||
fontSize: 12,
|
||||
fontFamily: 'inherit',
|
||||
color: isActive ? '#a78bfa' : '#9ca3af',
|
||||
cursor: 'pointer',
|
||||
transition: 'all 0.15s',
|
||||
width: '100%',
|
||||
}}
|
||||
onMouseEnter={e => { if (!isActive) { e.currentTarget.style.color = '#e2e2e8'; e.currentTarget.style.background = 'rgba(255,255,255,0.04)'; } }}
|
||||
onMouseLeave={e => { if (!isActive) { e.currentTarget.style.color = '#9ca3af'; e.currentTarget.style.background = 'transparent'; } }}
|
||||
>
|
||||
<span style={{ color: isActive ? '#a78bfa' : '#6b7280', display: 'flex', flexShrink: 0 }}>{icon}</span>
|
||||
{label}
|
||||
</button>
|
||||
);
|
||||
})}
|
||||
</div>
|
||||
|
||||
{/* Content pane */}
|
||||
<div style={{ padding: '20px', overflowY: 'auto' }}>
|
||||
<TabContent active={active} nodeCount={nodeCount} edgeCount={edgeCount} />
|
||||
</div>
|
||||
</div>
|
||||
|
||||
{/* Footer */}
|
||||
<div style={{
|
||||
display: 'flex',
|
||||
alignItems: 'center',
|
||||
justifyContent: 'space-between',
|
||||
padding: '10px 20px',
|
||||
borderTop: '0.5px solid rgba(255,255,255,0.08)',
|
||||
background: 'rgba(255,255,255,0.01)',
|
||||
}}>
|
||||
<span style={{ fontSize: 11, color: '#4b5563' }}>GitNexus — open source codebase graph explorer</span>
|
||||
<a
|
||||
href="https://github.com/abhigyanpatwari/GitNexus"
|
||||
target="_blank"
|
||||
rel="noopener noreferrer"
|
||||
style={{ fontSize: 11, color: '#a78bfa', textDecoration: 'none' }}
|
||||
>
|
||||
Docs & GitHub ↗
|
||||
</a>
|
||||
</div>
|
||||
|
||||
</div>
|
||||
</div>
|
||||
);
|
||||
};
|
||||
@@ -1,11 +1,11 @@
|
||||
import React, { useState } from 'react';
|
||||
import React, { useState, useRef, useEffect } from 'react';
|
||||
import ReactMarkdown from 'react-markdown';
|
||||
import remarkGfm from 'remark-gfm';
|
||||
import { Prism as SyntaxHighlighter } from 'react-syntax-highlighter';
|
||||
import { vscDarkPlus } from 'react-syntax-highlighter/dist/esm/styles/prism';
|
||||
import { MermaidDiagram } from './MermaidDiagram';
|
||||
import { ToolCallCard } from './ToolCallCard';
|
||||
import { Copy, Check } from 'lucide-react';
|
||||
import { Copy, Check } from '@/lib/lucide-icons';
|
||||
|
||||
// Custom syntax theme
|
||||
const customTheme = {
|
||||
@@ -39,12 +39,24 @@ export const MarkdownRenderer: React.FC<MarkdownRendererProps> = ({
|
||||
showCopyButton = false
|
||||
}) => {
|
||||
const [copied, setCopied] = useState(false);
|
||||
const copyTimerRef = useRef<ReturnType<typeof setTimeout>>(undefined);
|
||||
|
||||
useEffect(() => {
|
||||
return () => {
|
||||
if (copyTimerRef.current) {
|
||||
clearTimeout(copyTimerRef.current);
|
||||
}
|
||||
};
|
||||
}, []);
|
||||
|
||||
const handleCopy = async () => {
|
||||
try {
|
||||
await navigator.clipboard.writeText(content);
|
||||
setCopied(true);
|
||||
setTimeout(() => setCopied(false), 2000);
|
||||
if (copyTimerRef.current) {
|
||||
clearTimeout(copyTimerRef.current);
|
||||
}
|
||||
copyTimerRef.current = setTimeout(() => setCopied(false), 2000);
|
||||
} catch (err) {
|
||||
console.error('Failed to copy:', err);
|
||||
}
|
||||
@@ -78,13 +90,13 @@ export const MarkdownRenderer: React.FC<MarkdownRendererProps> = ({
|
||||
return parts.join('```');
|
||||
};
|
||||
|
||||
const handleLinkClick = (e: React.MouseEvent<HTMLAnchorElement>, href: string) => {
|
||||
const handleLinkClick = React.useCallback((e: React.MouseEvent<HTMLAnchorElement>, href: string) => {
|
||||
if (href.startsWith('code-ref:') || href.startsWith('node-ref:')) {
|
||||
e.preventDefault();
|
||||
onLinkClick?.(href);
|
||||
}
|
||||
// External links open in new tab (default behavior)
|
||||
};
|
||||
}, [onLinkClick]);
|
||||
|
||||
const formattedContent = React.useMemo(() => formatMarkdownForDisplay(content), [content]);
|
||||
|
||||
@@ -164,7 +176,7 @@ export const MarkdownRenderer: React.FC<MarkdownRendererProps> = ({
|
||||
);
|
||||
},
|
||||
pre: ({ children }: any) => <>{children}</>,
|
||||
}), [onLinkClick]); // Removed handleLinkClick dependency as it is defined inside component but depends on onLinkClick
|
||||
}), [handleLinkClick]);
|
||||
|
||||
return (
|
||||
<div className="text-text-primary text-sm">
|
||||
|
||||
@@ -1,9 +1,13 @@
|
||||
import { useEffect, useRef, useState } from 'react';
|
||||
import { Suspense, useEffect, useRef, useState, lazy } from 'react';
|
||||
import mermaid from 'mermaid';
|
||||
import { AlertTriangle, Maximize2 } from 'lucide-react';
|
||||
import { ProcessFlowModal } from './ProcessFlowModal';
|
||||
import DOMPurify from 'dompurify';
|
||||
import { AlertTriangle, Maximize2 } from '@/lib/lucide-icons';
|
||||
import type { ProcessData } from '../lib/mermaid-generator';
|
||||
|
||||
const ProcessFlowModal = lazy(() =>
|
||||
import('./ProcessFlowModal').then((m) => ({ default: m.ProcessFlowModal })),
|
||||
);
|
||||
|
||||
// Initialize mermaid with cyan theme matching ProcessFlowModal
|
||||
mermaid.initialize({
|
||||
startOnLoad: false,
|
||||
@@ -67,7 +71,8 @@ export const MermaidDiagram = ({ code }: MermaidDiagramProps) => {
|
||||
|
||||
// Render the diagram
|
||||
const { svg: renderedSvg } = await mermaid.render(id, code.trim());
|
||||
setSvg(renderedSvg);
|
||||
const sanitizedSvg = DOMPurify.sanitize(renderedSvg, { USE_PROFILES: { svg: true, svgFilters: true }, ADD_TAGS: ['foreignObject'] });
|
||||
setSvg(sanitizedSvg);
|
||||
setError(null);
|
||||
} catch (err) {
|
||||
// Silent catch for streaming:
|
||||
@@ -140,17 +145,23 @@ export const MermaidDiagram = ({ code }: MermaidDiagramProps) => {
|
||||
<div
|
||||
ref={containerRef}
|
||||
className="flex items-center justify-center p-4 overflow-auto max-h-[400px]"
|
||||
dangerouslySetInnerHTML={{ __html: svg }}
|
||||
dangerouslySetInnerHTML={{ __html: DOMPurify.sanitize(svg, { USE_PROFILES: { svg: true, svgFilters: true }, ADD_TAGS: ['foreignObject'] }) }}
|
||||
/>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
{/* Use ProcessFlowModal for expansion */}
|
||||
{showModal && processData && (
|
||||
<ProcessFlowModal
|
||||
process={processData}
|
||||
onClose={() => setShowModal(false)}
|
||||
/>
|
||||
<Suspense
|
||||
fallback={
|
||||
<div className="p-4 text-sm text-text-muted">Loading diagram…</div>
|
||||
}
|
||||
>
|
||||
<ProcessFlowModal
|
||||
process={processData}
|
||||
onClose={() => setShowModal(false)}
|
||||
/>
|
||||
</Suspense>
|
||||
)}
|
||||
</>
|
||||
);
|
||||
|
||||
@@ -7,6 +7,7 @@
|
||||
import { useEffect, useRef, useCallback, useState } from 'react';
|
||||
import { X, GitBranch, Copy, Focus, Layers, ZoomIn, ZoomOut } from 'lucide-react';
|
||||
import mermaid from 'mermaid';
|
||||
import DOMPurify from 'dompurify';
|
||||
import { ProcessData, generateProcessMermaid } from '../lib/mermaid-generator';
|
||||
|
||||
interface ProcessFlowModalProps {
|
||||
@@ -90,6 +91,7 @@ export const ProcessFlowModal = ({ process, onClose, onFocusInGraph, isFullScree
|
||||
// Handle keyboard zoom
|
||||
useEffect(() => {
|
||||
const handleKeyDown = (e: KeyboardEvent) => {
|
||||
if (e.target instanceof HTMLInputElement || e.target instanceof HTMLTextAreaElement) return;
|
||||
if (e.key === '+' || e.key === '=') {
|
||||
setZoom(prev => Math.min(prev + 0.2, maxZoom));
|
||||
} else if (e.key === '-' || e.key === '_') {
|
||||
@@ -136,8 +138,8 @@ export const ProcessFlowModal = ({ process, onClose, onFocusInGraph, isFullScree
|
||||
const renderDiagram = async () => {
|
||||
try {
|
||||
// Check if we have raw mermaid code (from AI chat) or need to generate it
|
||||
const mermaidCode = (process as any).rawMermaid
|
||||
? (process as any).rawMermaid
|
||||
const mermaidCode = process.rawMermaid
|
||||
? process.rawMermaid
|
||||
: generateProcessMermaid(process);
|
||||
const id = `mermaid-${Date.now()}`;
|
||||
|
||||
@@ -145,7 +147,8 @@ export const ProcessFlowModal = ({ process, onClose, onFocusInGraph, isFullScree
|
||||
diagramRef.current!.innerHTML = '';
|
||||
|
||||
const { svg } = await mermaid.render(id, mermaidCode);
|
||||
diagramRef.current!.innerHTML = svg;
|
||||
if (!diagramRef.current) return;
|
||||
diagramRef.current!.innerHTML = DOMPurify.sanitize(svg, { USE_PROFILES: { svg: true, svgFilters: true }, ADD_TAGS: ['foreignObject'] });
|
||||
} catch (error) {
|
||||
console.error('Mermaid render error:', error);
|
||||
const errorMessage = error instanceof Error ? error.message : String(error);
|
||||
@@ -208,6 +211,7 @@ export const ProcessFlowModal = ({ process, onClose, onFocusInGraph, isFullScree
|
||||
ref={containerRef}
|
||||
className="fixed inset-0 z-50 flex items-center justify-center bg-black/20 animate-fade-in"
|
||||
onClick={handleBackdropClick}
|
||||
data-testid="process-modal"
|
||||
>
|
||||
{/* Glassmorphism Modal */}
|
||||
<div className={`bg-slate-900/60 backdrop-blur-2xl border border-white/10 rounded-3xl shadow-2xl shadow-cyan-500/10 flex flex-col animate-scale-in overflow-hidden relative ${isFullScreen
|
||||
|
||||
@@ -11,6 +11,9 @@ import { useAppState } from '../hooks/useAppState';
|
||||
import { ProcessFlowModal } from './ProcessFlowModal';
|
||||
import type { ProcessData, ProcessStep } from '../lib/mermaid-generator';
|
||||
|
||||
/** Validate that an ID contains only expected node identifier characters (no Cypher metacharacters or spaces) */
|
||||
const isSafeId = (id: string): boolean => /^[a-zA-Z0-9_:.\-/@]+$/.test(id);
|
||||
|
||||
export const ProcessesPanel = () => {
|
||||
const { graph, runQuery, setHighlightedNodeIds, highlightedNodeIds } = useAppState();
|
||||
const [searchQuery, setSearchQuery] = useState('');
|
||||
@@ -79,7 +82,7 @@ export const ProcessesPanel = () => {
|
||||
setLoadingProcess('all');
|
||||
|
||||
try {
|
||||
const allProcessIds = [...processes.cross, ...processes.intra].map(p => p.id);
|
||||
const allProcessIds = [...processes.cross, ...processes.intra].map(p => p.id).filter(isSafeId);
|
||||
|
||||
if (allProcessIds.length === 0) return;
|
||||
|
||||
@@ -110,7 +113,7 @@ export const ProcessesPanel = () => {
|
||||
}
|
||||
|
||||
const allSteps = Array.from(allStepsMap.values());
|
||||
const stepIds = allSteps.map(s => s.id);
|
||||
const stepIds = allSteps.map(s => s.id).filter(isSafeId);
|
||||
|
||||
// Query for all CALLS edges between the combined steps
|
||||
if (stepIds.length > 0) {
|
||||
@@ -155,6 +158,7 @@ export const ProcessesPanel = () => {
|
||||
|
||||
// Load process steps and open modal
|
||||
const handleViewProcess = useCallback(async (processId: string, label: string, processType: string) => {
|
||||
if (!isSafeId(processId)) return;
|
||||
setLoadingProcess(processId);
|
||||
|
||||
try {
|
||||
@@ -175,7 +179,7 @@ export const ProcessesPanel = () => {
|
||||
}));
|
||||
|
||||
// Get step IDs for edge query
|
||||
const stepIds = steps.map(s => s.id);
|
||||
const stepIds = steps.map(s => s.id).filter(isSafeId);
|
||||
|
||||
// Query for CALLS edges between the steps in this process
|
||||
let edges: Array<{ from: string; to: string; type: string }> = [];
|
||||
@@ -228,6 +232,7 @@ export const ProcessesPanel = () => {
|
||||
|
||||
// Toggle focus for any process - loads steps on demand
|
||||
const handleToggleFocusForProcess = useCallback(async (processId: string) => {
|
||||
if (!isSafeId(processId)) return;
|
||||
// If already focused on this process, turn off
|
||||
if (focusedProcessId === processId) {
|
||||
setHighlightedNodeIds(new Set());
|
||||
@@ -321,7 +326,7 @@ export const ProcessesPanel = () => {
|
||||
/>
|
||||
</div>
|
||||
</div>
|
||||
<div className="flex items-center gap-2 text-xs text-text-muted">
|
||||
<div className="flex items-center gap-2 text-xs text-text-muted" data-testid="process-list-loaded">
|
||||
<span>{totalCount} processes detected</span>
|
||||
</div>
|
||||
</div>
|
||||
@@ -457,7 +462,7 @@ const ProcessItem = ({ process, isLoading, isSelected, isFocused, onView, onTogg
|
||||
: '';
|
||||
|
||||
return (
|
||||
<div className={`flex items-center gap-2 px-4 py-2 mx-2 rounded-lg hover:bg-hover group transition-all ${rowClass}`}>
|
||||
<div data-testid="process-row" className={`flex items-center gap-2 px-4 py-2 mx-2 rounded-lg hover:bg-hover group transition-all ${rowClass}`}>
|
||||
<GitBranch className="w-4 h-4 text-text-muted flex-shrink-0" />
|
||||
<div className="flex-1 min-w-0">
|
||||
<div className="text-sm text-text-primary truncate">{process.label}</div>
|
||||
@@ -479,12 +484,14 @@ const ProcessItem = ({ process, isLoading, isSelected, isFocused, onView, onTogg
|
||||
: 'text-text-muted hover:text-cyan-400 bg-white/5 hover:bg-cyan-500/20 border border-white/10 hover:border-cyan-400/40 opacity-0 group-hover:opacity-100'
|
||||
}`}
|
||||
title={isFocused ? 'Click to remove highlight from graph' : 'Click to highlight in graph'}
|
||||
data-testid="process-highlight-button"
|
||||
>
|
||||
<Lightbulb className="w-4 h-4" />
|
||||
</button>
|
||||
<button
|
||||
onClick={onView}
|
||||
disabled={isLoading}
|
||||
data-testid="process-view-button"
|
||||
className={`flex items-center gap-1.5 px-2.5 py-1.5 text-xs font-medium rounded-md transition-all disabled:opacity-50 shadow-sm ${isSelected
|
||||
? 'text-cyan-300 bg-cyan-900/60 border border-cyan-400/60 opacity-100'
|
||||
: 'text-cyan-400 hover:text-cyan-300 bg-cyan-950/30 hover:bg-cyan-900/50 border border-cyan-500/30 hover:border-cyan-400/50 opacity-0 group-hover:opacity-100 shadow-cyan-900/20'
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
import { useState, useRef, useEffect, useCallback } from 'react';
|
||||
import { Terminal, Play, X, ChevronDown, ChevronUp, Loader2, Sparkles, Table } from 'lucide-react';
|
||||
import { Terminal, Play, X, ChevronDown, ChevronUp, Loader2, Sparkles, Table } from '@/lib/lucide-icons';
|
||||
import { useAppState } from '../hooks/useAppState';
|
||||
|
||||
const EXAMPLE_QUERIES = [
|
||||
|
||||
@@ -2,7 +2,7 @@ import { useState, useRef, useEffect, useCallback } from 'react';
|
||||
import {
|
||||
Send, Square, Sparkles, User,
|
||||
PanelRightClose, Loader2, AlertTriangle, GitBranch
|
||||
} from 'lucide-react';
|
||||
} from '@/lib/lucide-icons';
|
||||
import { useAppState } from '../hooks/useAppState';
|
||||
import { ToolCallCard } from './ToolCallCard';
|
||||
import { isProviderConfigured } from '../core/llm/settings-service';
|
||||
|
||||
@@ -1,12 +1,15 @@
|
||||
import { useState, useEffect, useCallback, useRef, useMemo } from 'react';
|
||||
import { X, Key, Server, Brain, Check, AlertCircle, Eye, EyeOff, RefreshCw, ChevronDown, Loader2, Search } from 'lucide-react';
|
||||
import { X, Key, Server, Brain, Check, AlertCircle, Eye, EyeOff, RefreshCw, ChevronDown, Loader2, Search } from '@/lib/lucide-icons';
|
||||
import {
|
||||
loadSettings,
|
||||
saveSettings,
|
||||
getProviderDisplayName,
|
||||
getAvailableModels,
|
||||
fetchOpenRouterModels,
|
||||
} from '../core/llm/settings-service';
|
||||
import type { LLMSettings, LLMProvider } from '../core/llm/types';
|
||||
import { DEFAULT_OLLAMA_BASE_URL } from '../config/ui-constants';
|
||||
import { ProviderConfigCard } from './settings/ProviderConfigCard';
|
||||
|
||||
interface SettingsPanelProps {
|
||||
isOpen: boolean;
|
||||
@@ -216,6 +219,7 @@ export const SettingsPanel = ({ isOpen, onClose, onSettingsSaved, backendUrl, is
|
||||
const [settings, setSettings] = useState<LLMSettings>(loadSettings);
|
||||
const [showApiKey, setShowApiKey] = useState<Record<string, boolean>>({});
|
||||
const [saveStatus, setSaveStatus] = useState<'idle' | 'saved' | 'error'>('idle');
|
||||
const saveTimerRef = useRef<ReturnType<typeof setTimeout>>(undefined);
|
||||
// Ollama connection state
|
||||
const [ollamaError, setOllamaError] = useState<string | null>(null);
|
||||
const [isCheckingOllama, setIsCheckingOllama] = useState(false);
|
||||
@@ -223,6 +227,15 @@ export const SettingsPanel = ({ isOpen, onClose, onSettingsSaved, backendUrl, is
|
||||
const [openRouterModels, setOpenRouterModels] = useState<Array<{ id: string; name: string }>>([]);
|
||||
const [isLoadingModels, setIsLoadingModels] = useState(false);
|
||||
|
||||
// Clean up save timer on unmount
|
||||
useEffect(() => {
|
||||
return () => {
|
||||
if (saveTimerRef.current) {
|
||||
clearTimeout(saveTimerRef.current);
|
||||
}
|
||||
};
|
||||
}, []);
|
||||
|
||||
// Load settings when panel opens
|
||||
useEffect(() => {
|
||||
if (isOpen) {
|
||||
@@ -252,7 +265,7 @@ export const SettingsPanel = ({ isOpen, onClose, onSettingsSaved, backendUrl, is
|
||||
|
||||
useEffect(() => {
|
||||
if (settings.activeProvider === 'ollama') {
|
||||
const baseUrl = settings.ollama?.baseUrl ?? 'http://localhost:11434';
|
||||
const baseUrl = settings.ollama?.baseUrl ?? DEFAULT_OLLAMA_BASE_URL;
|
||||
const timer = setTimeout(() => {
|
||||
checkOllamaConnection(baseUrl);
|
||||
}, 300);
|
||||
@@ -269,7 +282,10 @@ export const SettingsPanel = ({ isOpen, onClose, onSettingsSaved, backendUrl, is
|
||||
saveSettings(settings);
|
||||
setSaveStatus('saved');
|
||||
onSettingsSaved?.();
|
||||
setTimeout(() => setSaveStatus('idle'), 2000);
|
||||
if (saveTimerRef.current) {
|
||||
clearTimeout(saveTimerRef.current);
|
||||
}
|
||||
saveTimerRef.current = setTimeout(() => setSaveStatus('idle'), 2000);
|
||||
} catch {
|
||||
setSaveStatus('error');
|
||||
}
|
||||
@@ -281,7 +297,7 @@ export const SettingsPanel = ({ isOpen, onClose, onSettingsSaved, backendUrl, is
|
||||
|
||||
if (!isOpen) return null;
|
||||
|
||||
const providers: LLMProvider[] = ['openai', 'gemini', 'anthropic', 'azure-openai', 'ollama', 'openrouter'];
|
||||
const providers: LLMProvider[] = ['openai', 'gemini', 'anthropic', 'azure-openai', 'ollama', 'openrouter', 'minimax', 'glm'];
|
||||
|
||||
|
||||
return (
|
||||
@@ -366,7 +382,7 @@ export const SettingsPanel = ({ isOpen, onClose, onSettingsSaved, backendUrl, is
|
||||
w-8 h-8 rounded-lg flex items-center justify-center text-lg
|
||||
${settings.activeProvider === provider ? 'bg-accent/20' : 'bg-surface'}
|
||||
`}>
|
||||
{provider === 'openai' ? '🤖' : provider === 'gemini' ? '💎' : provider === 'anthropic' ? '🧠' : provider === 'ollama' ? '🦙' : provider === 'openrouter' ? '🌐' : '☁️'}
|
||||
{provider === 'openai' ? '🤖' : provider === 'gemini' ? '💎' : provider === 'anthropic' ? '🧠' : provider === 'ollama' ? '🦙' : provider === 'openrouter' ? '🌐' : provider === 'minimax' ? '⚡' : provider === 'glm' ? '🔮' : '☁️'}
|
||||
</div>
|
||||
<span className="font-medium">{getProviderDisplayName(provider)}</span>
|
||||
</button>
|
||||
@@ -374,60 +390,36 @@ export const SettingsPanel = ({ isOpen, onClose, onSettingsSaved, backendUrl, is
|
||||
</div>
|
||||
</div>
|
||||
|
||||
<div className="p-3 bg-amber-500/10 border border-amber-500/30 rounded-xl text-xs text-amber-200">
|
||||
API keys are stored in session storage and will be cleared when you close this tab.
|
||||
</div>
|
||||
|
||||
{/* OpenAI Settings */}
|
||||
{settings.activeProvider === 'openai' && (
|
||||
<div className="space-y-4 animate-fade-in">
|
||||
<div className="space-y-2">
|
||||
<label className="flex items-center gap-2 text-sm font-medium text-text-secondary">
|
||||
<Key className="w-4 h-4" />
|
||||
API Key
|
||||
</label>
|
||||
<div className="relative">
|
||||
<input
|
||||
type={showApiKey['openai'] ? 'text' : 'password'}
|
||||
value={settings.openai?.apiKey ?? ''}
|
||||
onChange={e => setSettings(prev => ({
|
||||
...prev,
|
||||
openai: { ...prev.openai!, apiKey: e.target.value }
|
||||
}))}
|
||||
placeholder="Enter your OpenAI API key"
|
||||
className="w-full px-4 py-3 pr-12 bg-elevated border border-border-subtle rounded-xl text-text-primary placeholder:text-text-muted focus:border-accent focus:ring-2 focus:ring-accent/20 outline-none transition-all"
|
||||
/>
|
||||
<button
|
||||
type="button"
|
||||
onClick={() => toggleApiKeyVisibility('openai')}
|
||||
className="absolute right-3 top-1/2 -translate-y-1/2 p-1 text-text-muted hover:text-text-primary transition-colors"
|
||||
>
|
||||
{showApiKey['openai'] ? <EyeOff className="w-4 h-4" /> : <Eye className="w-4 h-4" />}
|
||||
</button>
|
||||
</div>
|
||||
<p className="text-xs text-text-muted">
|
||||
Get your API key from{' '}
|
||||
<a
|
||||
href="https://platform.openai.com/api-keys"
|
||||
target="_blank"
|
||||
rel="noopener noreferrer"
|
||||
className="text-accent hover:underline"
|
||||
>
|
||||
OpenAI Platform
|
||||
</a>
|
||||
</p>
|
||||
</div>
|
||||
|
||||
<div className="space-y-2">
|
||||
<label className="text-sm font-medium text-text-secondary">Model</label>
|
||||
<input
|
||||
type="text"
|
||||
value={settings.openai?.model ?? 'gpt-5.2-chat'}
|
||||
onChange={e => setSettings(prev => ({
|
||||
...prev,
|
||||
openai: { ...prev.openai!, model: e.target.value }
|
||||
}))}
|
||||
placeholder="e.g., gpt-4o, gpt-4-turbo, gpt-3.5-turbo"
|
||||
className="w-full px-4 py-3 bg-elevated border border-border-subtle rounded-xl text-text-primary placeholder:text-text-muted focus:border-accent focus:ring-2 focus:ring-accent/20 outline-none transition-all font-mono text-sm"
|
||||
/>
|
||||
</div>
|
||||
|
||||
<ProviderConfigCard
|
||||
title="OpenAI"
|
||||
apiKey={{
|
||||
value: settings.openai?.apiKey ?? '',
|
||||
placeholder: 'Enter your OpenAI API key',
|
||||
helperText: 'Get your API key from',
|
||||
helperLink: 'https://platform.openai.com/api-keys',
|
||||
helperLinkLabel: 'OpenAI Platform',
|
||||
isVisible: !!showApiKey['openai'],
|
||||
onChange: (value) => setSettings(prev => ({
|
||||
...prev,
|
||||
openai: { ...prev.openai!, apiKey: value }
|
||||
})),
|
||||
onToggleVisibility: () => toggleApiKeyVisibility('openai'),
|
||||
}}
|
||||
model={{
|
||||
value: settings.openai?.model ?? 'gpt-5.2-chat',
|
||||
placeholder: 'e.g., gpt-4o, gpt-4-turbo, gpt-3.5-turbo',
|
||||
onChange: (value) => setSettings(prev => ({
|
||||
...prev,
|
||||
openai: { ...prev.openai!, model: value }
|
||||
})),
|
||||
}}
|
||||
>
|
||||
<div className="space-y-2">
|
||||
<label className="flex items-center gap-2 text-sm font-medium text-text-secondary">
|
||||
<Server className="w-4 h-4" />
|
||||
@@ -447,119 +439,63 @@ export const SettingsPanel = ({ isOpen, onClose, onSettingsSaved, backendUrl, is
|
||||
Leave empty to use the default OpenAI API. Set a custom URL for proxies or compatible APIs.
|
||||
</p>
|
||||
</div>
|
||||
</div>
|
||||
</ProviderConfigCard>
|
||||
)}
|
||||
|
||||
{/* Gemini Settings */}
|
||||
{settings.activeProvider === 'gemini' && (
|
||||
<div className="space-y-4 animate-fade-in">
|
||||
<div className="space-y-2">
|
||||
<label className="flex items-center gap-2 text-sm font-medium text-text-secondary">
|
||||
<Key className="w-4 h-4" />
|
||||
API Key
|
||||
</label>
|
||||
<div className="relative">
|
||||
<input
|
||||
type={showApiKey['gemini'] ? 'text' : 'password'}
|
||||
value={settings.gemini?.apiKey ?? ''}
|
||||
onChange={e => setSettings(prev => ({
|
||||
...prev,
|
||||
gemini: { ...prev.gemini!, apiKey: e.target.value }
|
||||
}))}
|
||||
placeholder="Enter your Google AI API key"
|
||||
className="w-full px-4 py-3 pr-12 bg-elevated border border-border-subtle rounded-xl text-text-primary placeholder:text-text-muted focus:border-accent focus:ring-2 focus:ring-accent/20 outline-none transition-all"
|
||||
/>
|
||||
<button
|
||||
type="button"
|
||||
onClick={() => toggleApiKeyVisibility('gemini')}
|
||||
className="absolute right-3 top-1/2 -translate-y-1/2 p-1 text-text-muted hover:text-text-primary transition-colors"
|
||||
>
|
||||
{showApiKey['gemini'] ? <EyeOff className="w-4 h-4" /> : <Eye className="w-4 h-4" />}
|
||||
</button>
|
||||
</div>
|
||||
<p className="text-xs text-text-muted">
|
||||
Get your API key from{' '}
|
||||
<a
|
||||
href="https://aistudio.google.com/app/apikey"
|
||||
target="_blank"
|
||||
rel="noopener noreferrer"
|
||||
className="text-accent hover:underline"
|
||||
>
|
||||
Google AI Studio
|
||||
</a>
|
||||
</p>
|
||||
</div>
|
||||
|
||||
<div className="space-y-2">
|
||||
<label className="text-sm font-medium text-text-secondary">Model</label>
|
||||
<input
|
||||
type="text"
|
||||
value={settings.gemini?.model ?? 'gemini-2.0-flash'}
|
||||
onChange={e => setSettings(prev => ({
|
||||
...prev,
|
||||
gemini: { ...prev.gemini!, model: e.target.value }
|
||||
}))}
|
||||
placeholder="e.g., gemini-2.0-flash, gemini-1.5-pro"
|
||||
className="w-full px-4 py-3 bg-elevated border border-border-subtle rounded-xl text-text-primary placeholder:text-text-muted focus:border-accent focus:ring-2 focus:ring-accent/20 outline-none transition-all font-mono text-sm"
|
||||
/>
|
||||
</div>
|
||||
</div>
|
||||
<ProviderConfigCard
|
||||
title="Google Gemini"
|
||||
apiKey={{
|
||||
value: settings.gemini?.apiKey ?? '',
|
||||
placeholder: 'Enter your Google AI API key',
|
||||
helperText: 'Get your API key from',
|
||||
helperLink: 'https://aistudio.google.com/app/apikey',
|
||||
helperLinkLabel: 'Google AI Studio',
|
||||
isVisible: !!showApiKey['gemini'],
|
||||
onChange: (value) => setSettings(prev => ({
|
||||
...prev,
|
||||
gemini: { ...prev.gemini!, apiKey: value }
|
||||
})),
|
||||
onToggleVisibility: () => toggleApiKeyVisibility('gemini'),
|
||||
}}
|
||||
model={{
|
||||
value: settings.gemini?.model ?? 'gemini-2.0-flash',
|
||||
placeholder: 'e.g., gemini-2.0-flash, gemini-1.5-pro',
|
||||
onChange: (value) => setSettings(prev => ({
|
||||
...prev,
|
||||
gemini: { ...prev.gemini!, model: value }
|
||||
})),
|
||||
}}
|
||||
/>
|
||||
)}
|
||||
|
||||
{/* Anthropic Settings */}
|
||||
{settings.activeProvider === 'anthropic' && (
|
||||
<div className="space-y-4 animate-fade-in">
|
||||
<div className="space-y-2">
|
||||
<label className="flex items-center gap-2 text-sm font-medium text-text-secondary">
|
||||
<Key className="w-4 h-4" />
|
||||
API Key
|
||||
</label>
|
||||
<div className="relative">
|
||||
<input
|
||||
type={showApiKey['anthropic'] ? 'text' : 'password'}
|
||||
value={settings.anthropic?.apiKey ?? ''}
|
||||
onChange={e => setSettings(prev => ({
|
||||
...prev,
|
||||
anthropic: { ...prev.anthropic!, apiKey: e.target.value }
|
||||
}))}
|
||||
placeholder="Enter your Anthropic API key"
|
||||
className="w-full px-4 py-3 pr-12 bg-elevated border border-border-subtle rounded-xl text-text-primary placeholder:text-text-muted focus:border-accent focus:ring-2 focus:ring-accent/20 outline-none transition-all"
|
||||
/>
|
||||
<button
|
||||
type="button"
|
||||
onClick={() => toggleApiKeyVisibility('anthropic')}
|
||||
className="absolute right-3 top-1/2 -translate-y-1/2 p-1 text-text-muted hover:text-text-primary transition-colors"
|
||||
>
|
||||
{showApiKey['anthropic'] ? <EyeOff className="w-4 h-4" /> : <Eye className="w-4 h-4" />}
|
||||
</button>
|
||||
</div>
|
||||
<p className="text-xs text-text-muted">
|
||||
Get your API key from{' '}
|
||||
<a
|
||||
href="https://console.anthropic.com/settings/keys"
|
||||
target="_blank"
|
||||
rel="noopener noreferrer"
|
||||
className="text-accent hover:underline"
|
||||
>
|
||||
Anthropic Console
|
||||
</a>
|
||||
</p>
|
||||
</div>
|
||||
|
||||
<div className="space-y-2">
|
||||
<label className="text-sm font-medium text-text-secondary">Model</label>
|
||||
<input
|
||||
type="text"
|
||||
value={settings.anthropic?.model ?? 'claude-sonnet-4-20250514'}
|
||||
onChange={e => setSettings(prev => ({
|
||||
...prev,
|
||||
anthropic: { ...prev.anthropic!, model: e.target.value }
|
||||
}))}
|
||||
placeholder="e.g., claude-sonnet-4-20250514, claude-3-opus"
|
||||
className="w-full px-4 py-3 bg-elevated border border-border-subtle rounded-xl text-text-primary placeholder:text-text-muted focus:border-accent focus:ring-2 focus:ring-accent/20 outline-none transition-all font-mono text-sm"
|
||||
/>
|
||||
</div>
|
||||
</div>
|
||||
<ProviderConfigCard
|
||||
title="Anthropic"
|
||||
apiKey={{
|
||||
value: settings.anthropic?.apiKey ?? '',
|
||||
placeholder: 'Enter your Anthropic API key',
|
||||
helperText: 'Get your API key from',
|
||||
helperLink: 'https://console.anthropic.com/settings/keys',
|
||||
helperLinkLabel: 'Anthropic Console',
|
||||
isVisible: !!showApiKey['anthropic'],
|
||||
onChange: (value) => setSettings(prev => ({
|
||||
...prev,
|
||||
anthropic: { ...prev.anthropic!, apiKey: value }
|
||||
})),
|
||||
onToggleVisibility: () => toggleApiKeyVisibility('anthropic'),
|
||||
}}
|
||||
model={{
|
||||
value: settings.anthropic?.model ?? 'claude-sonnet-4-20250514',
|
||||
placeholder: 'e.g., claude-sonnet-4-20250514, claude-3-opus',
|
||||
onChange: (value) => setSettings(prev => ({
|
||||
...prev,
|
||||
anthropic: { ...prev.anthropic!, model: value }
|
||||
})),
|
||||
}}
|
||||
/>
|
||||
)}
|
||||
|
||||
{/* Azure OpenAI Settings */}
|
||||
@@ -695,17 +631,17 @@ export const SettingsPanel = ({ isOpen, onClose, onSettingsSaved, backendUrl, is
|
||||
<div className="flex gap-2">
|
||||
<input
|
||||
type="url"
|
||||
value={settings.ollama?.baseUrl ?? 'http://localhost:11434'}
|
||||
value={settings.ollama?.baseUrl ?? DEFAULT_OLLAMA_BASE_URL}
|
||||
onChange={e => setSettings(prev => ({
|
||||
...prev,
|
||||
ollama: { ...prev.ollama!, baseUrl: e.target.value }
|
||||
}))}
|
||||
placeholder="http://localhost:11434"
|
||||
placeholder={DEFAULT_OLLAMA_BASE_URL}
|
||||
className="flex-1 px-4 py-3 bg-elevated border border-border-subtle rounded-xl text-text-primary placeholder:text-text-muted focus:border-accent focus:ring-2 focus:ring-accent/20 outline-none transition-all font-mono text-sm"
|
||||
/>
|
||||
<button
|
||||
type="button"
|
||||
onClick={() => checkOllamaConnection(settings.ollama?.baseUrl ?? 'http://localhost:11434')}
|
||||
onClick={() => checkOllamaConnection(settings.ollama?.baseUrl ?? DEFAULT_OLLAMA_BASE_URL)}
|
||||
disabled={isCheckingOllama}
|
||||
className="px-3 py-3 bg-elevated border border-border-subtle rounded-xl text-text-secondary hover:text-text-primary hover:border-accent/50 transition-colors disabled:opacity-50"
|
||||
title="Check connection"
|
||||
@@ -749,44 +685,22 @@ export const SettingsPanel = ({ isOpen, onClose, onSettingsSaved, backendUrl, is
|
||||
|
||||
{/* OpenRouter Settings */}
|
||||
{settings.activeProvider === 'openrouter' && (
|
||||
<div className="space-y-4 animate-fade-in">
|
||||
<div className="space-y-2">
|
||||
<label className="flex items-center gap-2 text-sm font-medium text-text-secondary">
|
||||
<Key className="w-4 h-4" />
|
||||
API Key
|
||||
</label>
|
||||
<div className="relative">
|
||||
<input
|
||||
type={showApiKey['openrouter'] ? 'text' : 'password'}
|
||||
value={settings.openrouter?.apiKey ?? ''}
|
||||
onChange={e => setSettings(prev => ({
|
||||
...prev,
|
||||
openrouter: { ...prev.openrouter!, apiKey: e.target.value }
|
||||
}))}
|
||||
placeholder="Enter your OpenRouter API key"
|
||||
className="w-full px-4 py-3 pr-12 bg-elevated border border-border-subtle rounded-xl text-text-primary placeholder:text-text-muted focus:border-accent focus:ring-2 focus:ring-accent/20 outline-none transition-all"
|
||||
/>
|
||||
<button
|
||||
type="button"
|
||||
onClick={() => toggleApiKeyVisibility('openrouter')}
|
||||
className="absolute right-3 top-1/2 -translate-y-1/2 p-1 text-text-muted hover:text-text-primary transition-colors"
|
||||
>
|
||||
{showApiKey['openrouter'] ? <EyeOff className="w-4 h-4" /> : <Eye className="w-4 h-4" />}
|
||||
</button>
|
||||
</div>
|
||||
<p className="text-xs text-text-muted">
|
||||
Get your API key from{' '}
|
||||
<a
|
||||
href="https://openrouter.ai/keys"
|
||||
target="_blank"
|
||||
rel="noopener noreferrer"
|
||||
className="text-accent hover:underline"
|
||||
>
|
||||
OpenRouter Keys
|
||||
</a>
|
||||
</p>
|
||||
</div>
|
||||
|
||||
<ProviderConfigCard
|
||||
title="OpenRouter"
|
||||
apiKey={{
|
||||
value: settings.openrouter?.apiKey ?? '',
|
||||
placeholder: 'Enter your OpenRouter API key',
|
||||
helperText: 'Get your API key from',
|
||||
helperLink: 'https://openrouter.ai/keys',
|
||||
helperLinkLabel: 'OpenRouter Keys',
|
||||
isVisible: !!showApiKey['openrouter'],
|
||||
onChange: (value) => setSettings(prev => ({
|
||||
...prev,
|
||||
openrouter: { ...prev.openrouter!, apiKey: value }
|
||||
})),
|
||||
onToggleVisibility: () => toggleApiKeyVisibility('openrouter'),
|
||||
}}
|
||||
>
|
||||
<div className="space-y-2">
|
||||
<label className="text-sm font-medium text-text-secondary">Model</label>
|
||||
<OpenRouterModelCombobox
|
||||
@@ -811,10 +725,112 @@ export const SettingsPanel = ({ isOpen, onClose, onSettingsSaved, backendUrl, is
|
||||
</a>
|
||||
</p>
|
||||
</div>
|
||||
</div>
|
||||
</ProviderConfigCard>
|
||||
)}
|
||||
|
||||
{/* MiniMax Settings */}
|
||||
{settings.activeProvider === 'minimax' && (
|
||||
<ProviderConfigCard
|
||||
title="MiniMax"
|
||||
apiKey={{
|
||||
value: settings.minimax?.apiKey ?? '',
|
||||
placeholder: 'Enter your MiniMax API key',
|
||||
helperText: 'Get your API key from',
|
||||
helperLink: 'https://platform.minimax.io',
|
||||
helperLinkLabel: 'MiniMax Platform',
|
||||
isVisible: !!showApiKey['minimax'],
|
||||
onChange: (value) => setSettings(prev => ({
|
||||
...prev,
|
||||
minimax: { ...prev.minimax!, apiKey: value }
|
||||
})),
|
||||
onToggleVisibility: () => toggleApiKeyVisibility('minimax'),
|
||||
}}
|
||||
model={{
|
||||
value: settings.minimax?.model ?? 'MiniMax-M2.5',
|
||||
placeholder: 'e.g., MiniMax-M2.5, MiniMax-M2.5-highspeed',
|
||||
onChange: (value) => setSettings(prev => ({
|
||||
...prev,
|
||||
minimax: { ...prev.minimax!, model: value }
|
||||
})),
|
||||
helperText: 'Available: MiniMax-M2.5 (default), MiniMax-M2.5-highspeed (faster)',
|
||||
}}
|
||||
/>
|
||||
)}
|
||||
|
||||
{/* GLM Settings */}
|
||||
{settings.activeProvider === 'glm' && (
|
||||
<div className="space-y-4 animate-fade-in">
|
||||
<div className="space-y-2">
|
||||
<label className="flex items-center gap-2 text-sm font-medium text-text-secondary">
|
||||
<Key className="w-4 h-4" />
|
||||
API Key
|
||||
</label>
|
||||
<div className="relative">
|
||||
<input
|
||||
type={showApiKey['glm'] ? 'text' : 'password'}
|
||||
value={settings.glm?.apiKey ?? ''}
|
||||
onChange={e => setSettings(prev => ({
|
||||
...prev,
|
||||
glm: { ...prev.glm!, apiKey: e.target.value }
|
||||
}))}
|
||||
placeholder="Enter your Z.AI API key"
|
||||
className="w-full px-4 py-3 pr-12 bg-elevated border border-border-subtle rounded-xl text-text-primary placeholder:text-text-muted focus:border-accent focus:ring-2 focus:ring-accent/20 outline-none transition-all"
|
||||
/>
|
||||
<button
|
||||
type="button"
|
||||
onClick={() => toggleApiKeyVisibility('glm')}
|
||||
className="absolute right-3 top-1/2 -translate-y-1/2 p-1 text-text-muted hover:text-text-primary transition-colors"
|
||||
>
|
||||
{showApiKey['glm'] ? <EyeOff className="w-4 h-4" /> : <Eye className="w-4 h-4" />}
|
||||
</button>
|
||||
</div>
|
||||
<p className="text-xs text-text-muted">
|
||||
Get your API key from{' '}
|
||||
<a
|
||||
href="https://docs.z.ai"
|
||||
target="_blank"
|
||||
rel="noopener noreferrer"
|
||||
className="text-accent hover:underline"
|
||||
>
|
||||
Z.AI Platform
|
||||
</a>
|
||||
</p>
|
||||
</div>
|
||||
|
||||
<div className="space-y-2">
|
||||
<label className="text-sm font-medium text-text-secondary">Model</label>
|
||||
<select
|
||||
value={settings.glm?.model ?? 'GLM-5'}
|
||||
onChange={e => setSettings(prev => ({
|
||||
...prev,
|
||||
glm: { ...prev.glm!, model: e.target.value }
|
||||
}))}
|
||||
className="w-full px-4 py-3 bg-elevated border border-border-subtle rounded-xl text-text-primary focus:border-accent focus:ring-2 focus:ring-accent/20 outline-none transition-all font-mono text-sm"
|
||||
>
|
||||
{getAvailableModels('glm').map(model => (
|
||||
<option key={model} value={model}>{model}</option>
|
||||
))}
|
||||
</select>
|
||||
</div>
|
||||
|
||||
<div className="space-y-2">
|
||||
<label className="text-sm font-medium text-text-secondary">Base URL</label>
|
||||
<input
|
||||
type="text"
|
||||
value={settings.glm?.baseUrl ?? 'https://api.z.ai/api/coding/paas/v4'}
|
||||
onChange={e => setSettings(prev => ({
|
||||
...prev,
|
||||
glm: { ...prev.glm!, baseUrl: e.target.value }
|
||||
}))}
|
||||
placeholder="https://api.z.ai/api/coding/paas/v4"
|
||||
className="w-full px-4 py-3 bg-elevated border border-border-subtle rounded-xl text-text-primary placeholder:text-text-muted focus:border-accent focus:ring-2 focus:ring-accent/20 outline-none transition-all font-mono text-sm"
|
||||
/>
|
||||
<p className="text-xs text-text-muted">
|
||||
Coding API (default). Use https://api.z.ai/api/paas/v4 for the general API.
|
||||
</p>
|
||||
</div>
|
||||
</div>
|
||||
)}
|
||||
|
||||
{/* Privacy Note */}
|
||||
<div className="p-4 bg-elevated/50 border border-border-subtle rounded-xl">
|
||||
@@ -823,8 +839,7 @@ export const SettingsPanel = ({ isOpen, onClose, onSettingsSaved, backendUrl, is
|
||||
🔒
|
||||
</div>
|
||||
<div className="text-xs text-text-muted leading-relaxed">
|
||||
<span className="text-text-secondary font-medium">Privacy:</span> Your API keys are stored only in your browser's local storage.
|
||||
They're sent directly to the LLM provider when you chat. Your code never leaves your machine.
|
||||
<span className="text-text-secondary font-medium">Privacy:</span> Your API keys are stored only in your browser's session storage and are cleared when the tab closes. They're sent directly to the LLM provider when you chat. Your code never leaves your machine.
|
||||
</div>
|
||||
</div>
|
||||
</div>
|
||||
|
||||
@@ -1,4 +1,5 @@
|
||||
import { Heart } from 'lucide-react';
|
||||
import { useMemo } from 'react';
|
||||
import { Heart } from '@/lib/lucide-icons';
|
||||
import { useAppState } from '../hooks/useAppState';
|
||||
|
||||
export const StatusBar = () => {
|
||||
@@ -8,7 +9,7 @@ export const StatusBar = () => {
|
||||
const edgeCount = graph?.relationships.length ?? 0;
|
||||
|
||||
// Detect primary language
|
||||
const primaryLanguage = (() => {
|
||||
const primaryLanguage = useMemo(() => {
|
||||
if (!graph) return null;
|
||||
const languages = graph.nodes
|
||||
.map(n => n.properties.language)
|
||||
@@ -21,7 +22,7 @@ export const StatusBar = () => {
|
||||
}, {} as Record<string, number>);
|
||||
|
||||
return Object.entries(counts).sort((a, b) => b[1] - a[1])[0]?.[0];
|
||||
})();
|
||||
}, [graph]);
|
||||
|
||||
return (
|
||||
<footer className="flex items-center justify-between px-5 py-2 bg-deep border-t border-dashed border-border-subtle text-[11px] text-text-muted">
|
||||
@@ -38,7 +39,7 @@ export const StatusBar = () => {
|
||||
<span>{progress.message}</span>
|
||||
</>
|
||||
) : (
|
||||
<div className="flex items-center gap-1.5">
|
||||
<div className="flex items-center gap-1.5" data-testid="status-ready">
|
||||
<span className="w-1.5 h-1.5 bg-node-function rounded-full" />
|
||||
<span>Ready</span>
|
||||
</div>
|
||||
|
||||
@@ -6,7 +6,7 @@
|
||||
*/
|
||||
|
||||
import { useState } from 'react';
|
||||
import { ChevronDown, ChevronRight, Sparkles, Check, Loader2, AlertCircle } from 'lucide-react';
|
||||
import { ChevronDown, ChevronRight, Sparkles, Check, Loader2, AlertCircle } from '@/lib/lucide-icons';
|
||||
import type { ToolCallInfo } from '../core/llm/types';
|
||||
|
||||
interface ToolCallCardProps {
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
import { useState, useEffect } from 'react';
|
||||
import { X, Snail, Rocket, SkipForward } from 'lucide-react';
|
||||
import { X, Snail, Rocket, SkipForward } from '@/lib/lucide-icons';
|
||||
|
||||
interface WebGPUFallbackDialogProps {
|
||||
isOpen: boolean;
|
||||
|
||||
@@ -0,0 +1,110 @@
|
||||
import { ReactNode } from 'react';
|
||||
import { Eye, EyeOff, Key } from '@/lib/lucide-icons';
|
||||
|
||||
type ApiKeyField = {
|
||||
value: string;
|
||||
placeholder: string;
|
||||
helperText?: string;
|
||||
helperLink?: string;
|
||||
helperLinkLabel?: string;
|
||||
isVisible: boolean;
|
||||
onChange: (value: string) => void;
|
||||
onToggleVisibility: () => void;
|
||||
};
|
||||
|
||||
type ModelField = {
|
||||
value: string;
|
||||
placeholder: string;
|
||||
label?: string;
|
||||
helperText?: string;
|
||||
onChange: (value: string) => void;
|
||||
};
|
||||
|
||||
interface ProviderConfigCardProps {
|
||||
title: string;
|
||||
description?: string;
|
||||
apiKey?: ApiKeyField;
|
||||
model?: ModelField;
|
||||
children?: ReactNode;
|
||||
}
|
||||
|
||||
export const ProviderConfigCard = ({
|
||||
title,
|
||||
description,
|
||||
apiKey,
|
||||
model,
|
||||
children,
|
||||
}: ProviderConfigCardProps) => {
|
||||
return (
|
||||
<div className="space-y-4 animate-fade-in">
|
||||
<div className="flex items-center justify-between">
|
||||
<div>
|
||||
<h3 className="text-sm font-semibold text-text-primary">{title}</h3>
|
||||
{description ? (
|
||||
<p className="text-xs text-text-muted">{description}</p>
|
||||
) : null}
|
||||
</div>
|
||||
</div>
|
||||
|
||||
{apiKey && (
|
||||
<div className="space-y-2">
|
||||
<label className="flex items-center gap-2 text-sm font-medium text-text-secondary">
|
||||
<Key className="w-4 h-4" />
|
||||
API Key
|
||||
</label>
|
||||
<div className="relative">
|
||||
<input
|
||||
type={apiKey.isVisible ? 'text' : 'password'}
|
||||
value={apiKey.value}
|
||||
onChange={e => apiKey.onChange(e.target.value)}
|
||||
placeholder={apiKey.placeholder}
|
||||
className="w-full px-4 py-3 pr-12 bg-elevated border border-border-subtle rounded-xl text-text-primary placeholder:text-text-muted focus:border-accent focus:ring-2 focus:ring-accent/20 outline-none transition-all"
|
||||
/>
|
||||
<button
|
||||
type="button"
|
||||
onClick={apiKey.onToggleVisibility}
|
||||
className="absolute right-3 top-1/2 -translate-y-1/2 p-1 text-text-muted hover:text-text-primary transition-colors"
|
||||
>
|
||||
{apiKey.isVisible ? <EyeOff className="w-4 h-4" /> : <Eye className="w-4 h-4" />}
|
||||
</button>
|
||||
</div>
|
||||
{apiKey.helperText && (
|
||||
<p className="text-xs text-text-muted">
|
||||
{apiKey.helperText}{' '}
|
||||
{apiKey.helperLink ? (
|
||||
<a
|
||||
href={apiKey.helperLink}
|
||||
target="_blank"
|
||||
rel="noopener noreferrer"
|
||||
className="text-accent hover:underline"
|
||||
>
|
||||
{apiKey.helperLinkLabel ?? 'Learn more'}
|
||||
</a>
|
||||
) : null}
|
||||
</p>
|
||||
)}
|
||||
</div>
|
||||
)}
|
||||
|
||||
{model && (
|
||||
<div className="space-y-2">
|
||||
<label className="text-sm font-medium text-text-secondary">
|
||||
{model.label ?? 'Model'}
|
||||
</label>
|
||||
<input
|
||||
type="text"
|
||||
value={model.value}
|
||||
onChange={e => model.onChange(e.target.value)}
|
||||
placeholder={model.placeholder}
|
||||
className="w-full px-4 py-3 bg-elevated border border-border-subtle rounded-xl text-text-primary placeholder:text-text-muted focus:border-accent focus:ring-2 focus:ring-accent/20 outline-none transition-all font-mono text-sm"
|
||||
/>
|
||||
{model.helperText ? (
|
||||
<p className="text-xs text-text-muted">{model.helperText}</p>
|
||||
) : null}
|
||||
</div>
|
||||
)}
|
||||
|
||||
{children}
|
||||
</div>
|
||||
);
|
||||
};
|
||||
@@ -9,6 +9,7 @@ export enum SupportedLanguages {
|
||||
Go = 'go',
|
||||
Rust = 'rust',
|
||||
PHP = 'php',
|
||||
// Ruby = 'ruby',
|
||||
// Swift = 'swift',
|
||||
Ruby = 'ruby',
|
||||
Kotlin = 'kotlin',
|
||||
Swift = 'swift',
|
||||
}
|
||||
@@ -0,0 +1,7 @@
|
||||
// Centralized UI and provider defaults to reduce magic numbers and duplicated URLs.
|
||||
export const ERROR_RESET_DELAY_MS = 3000;
|
||||
export const BACKEND_URL_DEBOUNCE_MS = 500;
|
||||
|
||||
export const DEFAULT_BACKEND_URL = 'http://localhost:4747';
|
||||
export const DEFAULT_OLLAMA_BASE_URL = 'http://localhost:11434';
|
||||
export const DEFAULT_OPENROUTER_BASE_URL = 'https://openrouter.ai/api/v1';
|
||||
@@ -275,7 +275,7 @@ export const embedBatch = async (texts: string[]): Promise<Float32Array[]> => {
|
||||
};
|
||||
|
||||
/**
|
||||
* Convert Float32Array to regular number array (for KuzuDB storage)
|
||||
* Convert Float32Array to regular number array (for LadybugDB storage)
|
||||
*/
|
||||
export const embeddingToArray = (embedding: Float32Array): number[] => {
|
||||
return Array.from(embedding);
|
||||
|
||||
@@ -2,10 +2,10 @@
|
||||
* Embedding Pipeline Module
|
||||
*
|
||||
* Orchestrates the background embedding process:
|
||||
* 1. Query embeddable nodes from KuzuDB
|
||||
* 1. Query embeddable nodes from LadybugDB
|
||||
* 2. Generate text representations
|
||||
* 3. Batch embed using transformers.js
|
||||
* 4. Update KuzuDB with embeddings
|
||||
* 4. Update LadybugDB with embeddings
|
||||
* 5. Create vector index for semantic search
|
||||
*/
|
||||
|
||||
@@ -27,7 +27,7 @@ import {
|
||||
export type EmbeddingProgressCallback = (progress: EmbeddingProgress) => void;
|
||||
|
||||
/**
|
||||
* Query all embeddable nodes from KuzuDB
|
||||
* Query all embeddable nodes from LadybugDB
|
||||
* Uses table-specific queries (File has different schema than code elements)
|
||||
*/
|
||||
const queryEmbeddableNodes = async (
|
||||
@@ -102,9 +102,23 @@ const batchInsertEmbeddings = async (
|
||||
* Create the vector index for semantic search
|
||||
* Now indexes the separate CodeEmbedding table
|
||||
*/
|
||||
let vectorExtensionLoaded = false;
|
||||
|
||||
const createVectorIndex = async (
|
||||
executeQuery: (cypher: string) => Promise<any[]>
|
||||
): Promise<void> => {
|
||||
// LadybugDB v0.15+ requires explicit VECTOR extension loading (once per session)
|
||||
if (!vectorExtensionLoaded) {
|
||||
try {
|
||||
await executeQuery('INSTALL VECTOR');
|
||||
await executeQuery('LOAD EXTENSION VECTOR');
|
||||
vectorExtensionLoaded = true;
|
||||
} catch {
|
||||
// Extension may already be loaded — CREATE_VECTOR_INDEX will fail clearly if not
|
||||
vectorExtensionLoaded = true;
|
||||
}
|
||||
}
|
||||
|
||||
const cypher = `
|
||||
CALL CREATE_VECTOR_INDEX('CodeEmbedding', 'code_embedding_idx', 'embedding', metric := 'cosine')
|
||||
`;
|
||||
@@ -122,7 +136,7 @@ const createVectorIndex = async (
|
||||
/**
|
||||
* Run the embedding pipeline
|
||||
*
|
||||
* @param executeQuery - Function to execute Cypher queries against KuzuDB
|
||||
* @param executeQuery - Function to execute Cypher queries against LadybugDB
|
||||
* @param executeWithReusedStatement - Function to execute with reused prepared statement
|
||||
* @param onProgress - Callback for progress updates
|
||||
* @param config - Optional configuration override
|
||||
@@ -206,7 +220,7 @@ export const runEmbeddingPipeline = async (
|
||||
// Embed the batch
|
||||
const embeddings = await embedBatch(texts);
|
||||
|
||||
// Update KuzuDB with embeddings
|
||||
// Update LadybugDB with embeddings
|
||||
const updates = batch.map((node, i) => ({
|
||||
id: node.id,
|
||||
embedding: embeddingToArray(embeddings[i]),
|
||||
@@ -313,51 +327,64 @@ export const semanticSearch = async (
|
||||
return [];
|
||||
}
|
||||
|
||||
// Get metadata for each result by querying each node table
|
||||
const results: SemanticSearchResult[] = [];
|
||||
|
||||
// Group results by label for batched metadata queries
|
||||
const byLabel = new Map<string, Array<{ nodeId: string; distance: number }>>();
|
||||
for (const embRow of embResults) {
|
||||
const nodeId = embRow.nodeId ?? embRow[0];
|
||||
const distance = embRow.distance ?? embRow[1];
|
||||
|
||||
// Extract label from node ID (format: Label:path:name)
|
||||
const labelEndIdx = nodeId.indexOf(':');
|
||||
const label = labelEndIdx > 0 ? nodeId.substring(0, labelEndIdx) : 'Unknown';
|
||||
|
||||
// Query the specific table for this node
|
||||
// File nodes don't have startLine/endLine
|
||||
if (!byLabel.has(label)) byLabel.set(label, []);
|
||||
byLabel.get(label)!.push({ nodeId, distance });
|
||||
}
|
||||
|
||||
// Batch-fetch metadata per label
|
||||
const results: SemanticSearchResult[] = [];
|
||||
|
||||
for (const [label, items] of byLabel) {
|
||||
const idList = items.map(i => `'${i.nodeId.replace(/'/g, "''")}'`).join(', ');
|
||||
try {
|
||||
let nodeQuery: string;
|
||||
if (label === 'File') {
|
||||
nodeQuery = `
|
||||
MATCH (n:File {id: '${nodeId.replace(/'/g, "''")}'})
|
||||
RETURN n.name AS name, n.filePath AS filePath
|
||||
MATCH (n:File) WHERE n.id IN [${idList}]
|
||||
RETURN n.id AS id, n.name AS name, n.filePath AS filePath
|
||||
`;
|
||||
} else {
|
||||
nodeQuery = `
|
||||
MATCH (n:${label} {id: '${nodeId.replace(/'/g, "''")}'})
|
||||
RETURN n.name AS name, n.filePath AS filePath,
|
||||
MATCH (n:${label}) WHERE n.id IN [${idList}]
|
||||
RETURN n.id AS id, n.name AS name, n.filePath AS filePath,
|
||||
n.startLine AS startLine, n.endLine AS endLine
|
||||
`;
|
||||
}
|
||||
const nodeRows = await executeQuery(nodeQuery);
|
||||
if (nodeRows.length > 0) {
|
||||
const nodeRow = nodeRows[0];
|
||||
results.push({
|
||||
nodeId,
|
||||
name: nodeRow.name ?? nodeRow[0] ?? '',
|
||||
label,
|
||||
filePath: nodeRow.filePath ?? nodeRow[1] ?? '',
|
||||
distance,
|
||||
startLine: label !== 'File' ? (nodeRow.startLine ?? nodeRow[2]) : undefined,
|
||||
endLine: label !== 'File' ? (nodeRow.endLine ?? nodeRow[3]) : undefined,
|
||||
});
|
||||
const rowMap = new Map<string, any>();
|
||||
for (const row of nodeRows) {
|
||||
const id = row.id ?? row[0];
|
||||
rowMap.set(id, row);
|
||||
}
|
||||
for (const item of items) {
|
||||
const nodeRow = rowMap.get(item.nodeId);
|
||||
if (nodeRow) {
|
||||
results.push({
|
||||
nodeId: item.nodeId,
|
||||
name: nodeRow.name ?? nodeRow[1] ?? '',
|
||||
label,
|
||||
filePath: nodeRow.filePath ?? nodeRow[2] ?? '',
|
||||
distance: item.distance,
|
||||
startLine: label !== 'File' ? (nodeRow.startLine ?? nodeRow[3]) : undefined,
|
||||
endLine: label !== 'File' ? (nodeRow.endLine ?? nodeRow[4]) : undefined,
|
||||
});
|
||||
}
|
||||
}
|
||||
} catch {
|
||||
// Table might not exist, skip
|
||||
}
|
||||
}
|
||||
|
||||
// Re-sort by distance since batch queries may have mixed order
|
||||
results.sort((a, b) => a.distance - b.distance);
|
||||
|
||||
return results;
|
||||
};
|
||||
|
||||
|
||||
@@ -92,7 +92,7 @@ export interface SemanticSearchResult {
|
||||
}
|
||||
|
||||
/**
|
||||
* Node data for embedding (minimal structure from KuzuDB query)
|
||||
* Node data for embedding (minimal structure from LadybugDB query)
|
||||
*/
|
||||
export interface EmbeddableNode {
|
||||
id: string;
|
||||
|
||||
@@ -15,7 +15,24 @@ export type NodeLabel =
|
||||
| 'Type'
|
||||
| 'CodeElement'
|
||||
| 'Community'
|
||||
| 'Process';
|
||||
| 'Process'
|
||||
| 'Section'
|
||||
| 'Struct'
|
||||
| 'Trait'
|
||||
| 'Impl'
|
||||
| 'TypeAlias'
|
||||
| 'Const'
|
||||
| 'Static'
|
||||
| 'Namespace'
|
||||
| 'Union'
|
||||
| 'Typedef'
|
||||
| 'Macro'
|
||||
| 'Property'
|
||||
| 'Record'
|
||||
| 'Delegate'
|
||||
| 'Annotation'
|
||||
| 'Constructor'
|
||||
| 'Template';
|
||||
|
||||
|
||||
export type NodeProperties = {
|
||||
@@ -54,6 +71,9 @@ export type RelationshipType =
|
||||
| 'DECORATES'
|
||||
| 'IMPLEMENTS'
|
||||
| 'EXTENDS'
|
||||
| 'HAS_METHOD'
|
||||
| 'HAS_PROPERTY'
|
||||
| 'ACCESSES'
|
||||
| 'MEMBER_OF'
|
||||
| 'STEP_IN_PROCESS'
|
||||
|
||||
|
||||
@@ -6,6 +6,7 @@ import { loadParser, loadLanguage } from '../tree-sitter/parser-loader';
|
||||
import { LANGUAGE_QUERIES } from './tree-sitter-queries';
|
||||
import { generateId } from '../../lib/utils';
|
||||
import { getLanguageFromFilename } from './utils';
|
||||
import { callRouters } from './call-routing';
|
||||
|
||||
/**
|
||||
* Node types that represent function/method definitions across languages.
|
||||
@@ -35,6 +36,9 @@ const FUNCTION_NODE_TYPES = new Set([
|
||||
// Rust
|
||||
'function_item',
|
||||
'impl_item', // Methods inside impl blocks
|
||||
// Ruby
|
||||
'method', // def foo
|
||||
'singleton_method', // def self.foo
|
||||
]);
|
||||
|
||||
/**
|
||||
@@ -92,6 +96,18 @@ const findEnclosingFunction = (
|
||||
current.children?.find((c: any) => c.type === 'identifier');
|
||||
funcName = nameNode?.text;
|
||||
label = 'Method'; // Treat constructors as methods for process detection
|
||||
} else if (current.type === 'method') {
|
||||
// Ruby instance method: def foo
|
||||
const nameNode = current.childForFieldName?.('name') ||
|
||||
current.children?.find((c: any) => c.type === 'identifier');
|
||||
funcName = nameNode?.text;
|
||||
label = 'Method';
|
||||
} else if (current.type === 'singleton_method') {
|
||||
// Ruby class method: def self.foo
|
||||
const nameNode = current.childForFieldName?.('name') ||
|
||||
current.children?.find((c: any) => c.type === 'identifier');
|
||||
funcName = nameNode?.text;
|
||||
label = 'Method';
|
||||
} else if (current.type === 'arrow_function' || current.type === 'function_expression') {
|
||||
// Arrow/expression: const foo = () => {} - check parent variable declarator
|
||||
const parent = current.parent;
|
||||
@@ -126,6 +142,47 @@ const findEnclosingFunction = (
|
||||
return null; // Top-level call (not inside any function)
|
||||
};
|
||||
|
||||
/** AST node types that represent a class-like container */
|
||||
const CLASS_CONTAINER_TYPES = new Set([
|
||||
'class_declaration', 'abstract_class_declaration',
|
||||
'interface_declaration', 'struct_declaration', 'record_declaration',
|
||||
'class_specifier', 'struct_specifier',
|
||||
'impl_item', 'trait_item',
|
||||
'class_definition',
|
||||
'trait_declaration',
|
||||
'protocol_declaration',
|
||||
'class', 'module', // Ruby
|
||||
]);
|
||||
|
||||
const CONTAINER_TYPE_TO_LABEL: Record<string, string> = {
|
||||
class_declaration: 'Class', abstract_class_declaration: 'Class',
|
||||
interface_declaration: 'Interface',
|
||||
struct_declaration: 'Struct', struct_specifier: 'Struct',
|
||||
class_specifier: 'Class', class_definition: 'Class',
|
||||
impl_item: 'Impl', trait_item: 'Trait', trait_declaration: 'Trait',
|
||||
record_declaration: 'Record', protocol_declaration: 'Interface',
|
||||
class: 'Class', module: 'Module',
|
||||
};
|
||||
|
||||
/** Walk up AST to find enclosing class/struct/interface, return its generateId or null. */
|
||||
const findEnclosingClassId = (node: any, filePath: string): string | null => {
|
||||
let current = node.parent;
|
||||
while (current) {
|
||||
if (CLASS_CONTAINER_TYPES.has(current.type)) {
|
||||
const nameNode = current.childForFieldName?.('name')
|
||||
?? current.children?.find((c: any) =>
|
||||
c.type === 'type_identifier' || c.type === 'identifier' || c.type === 'name' || c.type === 'constant'
|
||||
);
|
||||
if (nameNode) {
|
||||
const label = CONTAINER_TYPE_TO_LABEL[current.type] || 'Class';
|
||||
return generateId(label, `${filePath}:${nameNode.text}`);
|
||||
}
|
||||
}
|
||||
current = current.parent;
|
||||
}
|
||||
return null;
|
||||
};
|
||||
|
||||
export const processCalls = async (
|
||||
graph: KnowledgeGraph,
|
||||
files: { path: string; content: string }[],
|
||||
@@ -171,6 +228,8 @@ export const processCalls = async (
|
||||
continue;
|
||||
}
|
||||
|
||||
const callRouter = callRouters[language];
|
||||
|
||||
// 3. Process each call match
|
||||
matches.forEach(match => {
|
||||
const captureMap: Record<string, any> = {};
|
||||
@@ -184,6 +243,68 @@ export const processCalls = async (
|
||||
|
||||
const calledName = nameNode.text;
|
||||
|
||||
// Dispatch: route language-specific calls (heritage, properties, imports)
|
||||
const routed = callRouter(calledName, captureMap['call']);
|
||||
if (routed) {
|
||||
switch (routed.kind) {
|
||||
case 'skip':
|
||||
case 'import': // handled by import-processor
|
||||
return;
|
||||
|
||||
case 'heritage':
|
||||
for (const item of routed.items) {
|
||||
const childId = symbolTable.lookupExact(file.path, item.enclosingClass) ||
|
||||
symbolTable.lookupFuzzy(item.enclosingClass)[0]?.nodeId ||
|
||||
generateId('Class', `${file.path}:${item.enclosingClass}`);
|
||||
const parentId = symbolTable.lookupFuzzy(item.mixinName)[0]?.nodeId ||
|
||||
generateId('Module', `${item.mixinName}`);
|
||||
if (childId && parentId) {
|
||||
const relId = generateId('IMPLEMENTS', `${childId}->${parentId}:${item.heritageKind}`);
|
||||
graph.addRelationship({
|
||||
id: relId, sourceId: childId, targetId: parentId,
|
||||
type: 'IMPLEMENTS', confidence: 1.0, reason: item.heritageKind,
|
||||
});
|
||||
}
|
||||
}
|
||||
return;
|
||||
|
||||
case 'properties': {
|
||||
const fileId = generateId('File', file.path);
|
||||
const propEnclosingClassId = findEnclosingClassId(captureMap['call'], file.path);
|
||||
for (const item of routed.items) {
|
||||
const nodeId = generateId('Property', `${file.path}:${item.propName}`);
|
||||
graph.addNode({
|
||||
id: nodeId,
|
||||
label: 'Property' as any, // TODO: add 'Property' to graph node label union
|
||||
properties: {
|
||||
name: item.propName, filePath: file.path,
|
||||
startLine: item.startLine, endLine: item.endLine,
|
||||
language, isExported: true,
|
||||
description: item.accessorType,
|
||||
},
|
||||
});
|
||||
symbolTable.add(file.path, item.propName, nodeId, 'Property');
|
||||
const relId = generateId('DEFINES', `${fileId}->${nodeId}`);
|
||||
graph.addRelationship({
|
||||
id: relId, sourceId: fileId, targetId: nodeId,
|
||||
type: 'DEFINES', confidence: 1.0, reason: '',
|
||||
});
|
||||
if (propEnclosingClassId) {
|
||||
graph.addRelationship({
|
||||
id: generateId('HAS_METHOD', `${propEnclosingClassId}->${nodeId}`),
|
||||
sourceId: propEnclosingClassId, targetId: nodeId,
|
||||
type: 'HAS_METHOD', confidence: 1.0, reason: '',
|
||||
});
|
||||
}
|
||||
}
|
||||
return;
|
||||
}
|
||||
|
||||
case 'call':
|
||||
break; // fall through to normal call processing below
|
||||
}
|
||||
}
|
||||
|
||||
// Skip common built-ins and noise
|
||||
if (isBuiltInOrNoise(calledName)) return;
|
||||
|
||||
@@ -200,10 +321,10 @@ export const processCalls = async (
|
||||
// 5. Find the enclosing function (caller)
|
||||
const callNode = captureMap['call'];
|
||||
const enclosingFuncId = findEnclosingFunction(callNode, file.path, symbolTable);
|
||||
|
||||
|
||||
// Use enclosing function as source, fallback to file for top-level calls
|
||||
const sourceId = enclosingFuncId || generateId('File', file.path);
|
||||
|
||||
|
||||
const relId = generateId('CALLS', `${sourceId}:${calledName}->${resolved.nodeId}`);
|
||||
|
||||
graph.addRelationship({
|
||||
@@ -216,6 +337,58 @@ export const processCalls = async (
|
||||
});
|
||||
});
|
||||
|
||||
// Extract Laravel routes from route files via procedural AST walk
|
||||
if (language === 'php' && (file.path.includes('/routes/') || file.path.startsWith('routes/')) && file.path.endsWith('.php')) {
|
||||
const extractedRoutes = extractLaravelRoutes(tree, file.path);
|
||||
for (const route of extractedRoutes) {
|
||||
if (!route.controllerName || !route.methodName) continue;
|
||||
|
||||
const controllerDefs = symbolTable.lookupFuzzy(route.controllerName);
|
||||
if (controllerDefs.length === 0) continue;
|
||||
|
||||
const routeImportedFiles = importMap.get(route.filePath);
|
||||
let controllerDef = controllerDefs[0];
|
||||
let conf = controllerDefs.length === 1 ? 0.7 : 0.5;
|
||||
|
||||
if (routeImportedFiles) {
|
||||
for (const def of controllerDefs) {
|
||||
if (routeImportedFiles.has(def.filePath)) {
|
||||
controllerDef = def;
|
||||
conf = 0.9;
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
const methodId = symbolTable.lookupExact(controllerDef.filePath, route.methodName);
|
||||
const routeSourceId = generateId('File', route.filePath);
|
||||
|
||||
if (!methodId) {
|
||||
const guessedId = generateId('Method', `${controllerDef.filePath}:${route.methodName}`);
|
||||
const routeRelId = generateId('CALLS', `${routeSourceId}:route->${guessedId}`);
|
||||
graph.addRelationship({
|
||||
id: routeRelId,
|
||||
sourceId: routeSourceId,
|
||||
targetId: guessedId,
|
||||
type: 'CALLS',
|
||||
confidence: conf * 0.8,
|
||||
reason: 'laravel-route',
|
||||
});
|
||||
continue;
|
||||
}
|
||||
|
||||
const routeRelId = generateId('CALLS', `${routeSourceId}:route->${methodId}`);
|
||||
graph.addRelationship({
|
||||
id: routeRelId,
|
||||
sourceId: routeSourceId,
|
||||
targetId: methodId,
|
||||
type: 'CALLS',
|
||||
confidence: conf,
|
||||
reason: 'laravel-route',
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
// Cleanup if re-parsed
|
||||
if (wasReparsed) {
|
||||
tree.delete();
|
||||
@@ -223,6 +396,387 @@ export const processCalls = async (
|
||||
}
|
||||
};
|
||||
|
||||
// ============================================================================
|
||||
// Laravel Route Extraction (procedural AST walk)
|
||||
// ============================================================================
|
||||
|
||||
interface ExtractedRoute {
|
||||
filePath: string;
|
||||
httpMethod: string;
|
||||
routePath: string | null;
|
||||
controllerName: string | null;
|
||||
methodName: string | null;
|
||||
middleware: string[];
|
||||
prefix: string | null;
|
||||
lineNumber: number;
|
||||
}
|
||||
|
||||
interface RouteGroupContext {
|
||||
middleware: string[];
|
||||
prefix: string | null;
|
||||
controller: string | null;
|
||||
}
|
||||
|
||||
const ROUTE_HTTP_METHODS = new Set([
|
||||
'get', 'post', 'put', 'patch', 'delete', 'options', 'any', 'match',
|
||||
]);
|
||||
|
||||
const ROUTE_RESOURCE_METHODS = new Set(['resource', 'apiResource']);
|
||||
|
||||
const RESOURCE_ACTIONS = ['index', 'create', 'store', 'show', 'edit', 'update', 'destroy'];
|
||||
const API_RESOURCE_ACTIONS = ['index', 'store', 'show', 'update', 'destroy'];
|
||||
|
||||
function isRouteStaticCall(node: any): boolean {
|
||||
if (node.type !== 'scoped_call_expression') return false;
|
||||
const obj = node.childForFieldName?.('object') ?? node.children?.[0];
|
||||
return obj?.text === 'Route';
|
||||
}
|
||||
|
||||
function getCallMethodName(node: any): string | null {
|
||||
const nameNode = node.childForFieldName?.('name') ??
|
||||
node.children?.find((c: any) => c.type === 'name');
|
||||
return nameNode?.text ?? null;
|
||||
}
|
||||
|
||||
function getArguments(node: any): any {
|
||||
return node.children?.find((c: any) => c.type === 'arguments') ?? null;
|
||||
}
|
||||
|
||||
function findClosureBody(argsNode: any): any | null {
|
||||
if (!argsNode) return null;
|
||||
for (const child of argsNode.children ?? []) {
|
||||
if (child.type === 'argument') {
|
||||
for (const inner of child.children ?? []) {
|
||||
if (inner.type === 'anonymous_function' ||
|
||||
inner.type === 'arrow_function') {
|
||||
return inner.childForFieldName?.('body') ??
|
||||
inner.children?.find((c: any) => c.type === 'compound_statement');
|
||||
}
|
||||
}
|
||||
}
|
||||
if (child.type === 'anonymous_function' ||
|
||||
child.type === 'arrow_function') {
|
||||
return child.childForFieldName?.('body') ??
|
||||
child.children?.find((c: any) => c.type === 'compound_statement');
|
||||
}
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
function findDescendant(node: any, type: string): any {
|
||||
if (node.type === type) return node;
|
||||
for (const child of (node.children ?? [])) {
|
||||
const found = findDescendant(child, type);
|
||||
if (found) return found;
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
function extractStringContent(node: any): string | null {
|
||||
if (!node) return null;
|
||||
const content = node.children?.find((c: any) => c.type === 'string_content');
|
||||
if (content) return content.text;
|
||||
if (node.type === 'string_content') return node.text;
|
||||
return null;
|
||||
}
|
||||
|
||||
function extractFirstStringArg(argsNode: any): string | null {
|
||||
if (!argsNode) return null;
|
||||
for (const child of argsNode.children ?? []) {
|
||||
const target = child.type === 'argument' ? child.children?.[0] : child;
|
||||
if (!target) continue;
|
||||
if (target.type === 'string' || target.type === 'encapsed_string') {
|
||||
return extractStringContent(target);
|
||||
}
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
function extractMiddlewareArg(argsNode: any): string[] {
|
||||
if (!argsNode) return [];
|
||||
for (const child of argsNode.children ?? []) {
|
||||
const target = child.type === 'argument' ? child.children?.[0] : child;
|
||||
if (!target) continue;
|
||||
if (target.type === 'string' || target.type === 'encapsed_string') {
|
||||
const val = extractStringContent(target);
|
||||
return val ? [val] : [];
|
||||
}
|
||||
if (target.type === 'array_creation_expression') {
|
||||
const items: string[] = [];
|
||||
for (const el of target.children ?? []) {
|
||||
if (el.type === 'array_element_initializer') {
|
||||
const str = el.children?.find((c: any) => c.type === 'string' || c.type === 'encapsed_string');
|
||||
const val = str ? extractStringContent(str) : null;
|
||||
if (val) items.push(val);
|
||||
}
|
||||
}
|
||||
return items;
|
||||
}
|
||||
}
|
||||
return [];
|
||||
}
|
||||
|
||||
function extractClassArg(argsNode: any): string | null {
|
||||
if (!argsNode) return null;
|
||||
for (const child of argsNode.children ?? []) {
|
||||
const target = child.type === 'argument' ? child.children?.[0] : child;
|
||||
if (target?.type === 'class_constant_access_expression') {
|
||||
return target.children?.find((c: any) => c.type === 'name')?.text ?? null;
|
||||
}
|
||||
}
|
||||
return null;
|
||||
}
|
||||
|
||||
function extractControllerTarget(argsNode: any): { controller: string | null; method: string | null } {
|
||||
if (!argsNode) return { controller: null, method: null };
|
||||
|
||||
const args: any[] = [];
|
||||
for (const child of argsNode.children ?? []) {
|
||||
if (child.type === 'argument') args.push(child.children?.[0]);
|
||||
else if (child.type !== '(' && child.type !== ')' && child.type !== ',') args.push(child);
|
||||
}
|
||||
|
||||
const handlerNode = args[1];
|
||||
if (!handlerNode) return { controller: null, method: null };
|
||||
|
||||
if (handlerNode.type === 'array_creation_expression') {
|
||||
let controller: string | null = null;
|
||||
let method: string | null = null;
|
||||
const elements: any[] = [];
|
||||
for (const el of handlerNode.children ?? []) {
|
||||
if (el.type === 'array_element_initializer') elements.push(el);
|
||||
}
|
||||
if (elements[0]) {
|
||||
const classAccess = findDescendant(elements[0], 'class_constant_access_expression');
|
||||
if (classAccess) {
|
||||
controller = classAccess.children?.find((c: any) => c.type === 'name')?.text ?? null;
|
||||
}
|
||||
}
|
||||
if (elements[1]) {
|
||||
const str = findDescendant(elements[1], 'string');
|
||||
method = str ? extractStringContent(str) : null;
|
||||
}
|
||||
return { controller, method };
|
||||
}
|
||||
|
||||
if (handlerNode.type === 'string' || handlerNode.type === 'encapsed_string') {
|
||||
const text = extractStringContent(handlerNode);
|
||||
if (text?.includes('@')) {
|
||||
const [controller, method] = text.split('@');
|
||||
return { controller, method };
|
||||
}
|
||||
}
|
||||
|
||||
if (handlerNode.type === 'class_constant_access_expression') {
|
||||
const controller = handlerNode.children?.find((c: any) => c.type === 'name')?.text ?? null;
|
||||
return { controller, method: '__invoke' };
|
||||
}
|
||||
|
||||
return { controller: null, method: null };
|
||||
}
|
||||
|
||||
interface ChainedRouteCall {
|
||||
isRouteFacade: boolean;
|
||||
terminalMethod: string;
|
||||
attributes: { method: string; argsNode: any }[];
|
||||
terminalArgs: any;
|
||||
node: any;
|
||||
}
|
||||
|
||||
function unwrapRouteChain(node: any): ChainedRouteCall | null {
|
||||
if (node.type !== 'member_call_expression') return null;
|
||||
|
||||
const terminalMethod = getCallMethodName(node);
|
||||
if (!terminalMethod) return null;
|
||||
|
||||
const terminalArgs = getArguments(node);
|
||||
const attributes: { method: string; argsNode: any }[] = [];
|
||||
|
||||
let current = node.children?.[0];
|
||||
|
||||
while (current) {
|
||||
if (current.type === 'member_call_expression') {
|
||||
const method = getCallMethodName(current);
|
||||
const args = getArguments(current);
|
||||
if (method) attributes.unshift({ method, argsNode: args });
|
||||
current = current.children?.[0];
|
||||
} else if (current.type === 'scoped_call_expression') {
|
||||
const obj = current.childForFieldName?.('object') ?? current.children?.[0];
|
||||
if (obj?.text !== 'Route') return null;
|
||||
|
||||
const method = getCallMethodName(current);
|
||||
const args = getArguments(current);
|
||||
if (method) attributes.unshift({ method, argsNode: args });
|
||||
|
||||
return { isRouteFacade: true, terminalMethod, attributes, terminalArgs, node };
|
||||
} else {
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
return null;
|
||||
}
|
||||
|
||||
function parseArrayGroupArgs(argsNode: any): RouteGroupContext {
|
||||
const ctx: RouteGroupContext = { middleware: [], prefix: null, controller: null };
|
||||
if (!argsNode) return ctx;
|
||||
|
||||
for (const child of argsNode.children ?? []) {
|
||||
const target = child.type === 'argument' ? child.children?.[0] : child;
|
||||
if (target?.type === 'array_creation_expression') {
|
||||
for (const el of target.children ?? []) {
|
||||
if (el.type !== 'array_element_initializer') continue;
|
||||
const children = el.children ?? [];
|
||||
const arrowIdx = children.findIndex((c: any) => c.type === '=>');
|
||||
if (arrowIdx === -1) continue;
|
||||
const key = extractStringContent(children[arrowIdx - 1]);
|
||||
const val = children[arrowIdx + 1];
|
||||
if (key === 'middleware') {
|
||||
if (val?.type === 'string') {
|
||||
const s = extractStringContent(val);
|
||||
if (s) ctx.middleware.push(s);
|
||||
} else if (val?.type === 'array_creation_expression') {
|
||||
for (const item of val.children ?? []) {
|
||||
if (item.type === 'array_element_initializer') {
|
||||
const str = item.children?.find((c: any) => c.type === 'string');
|
||||
const s = str ? extractStringContent(str) : null;
|
||||
if (s) ctx.middleware.push(s);
|
||||
}
|
||||
}
|
||||
}
|
||||
} else if (key === 'prefix') {
|
||||
ctx.prefix = extractStringContent(val) ?? null;
|
||||
} else if (key === 'controller') {
|
||||
if (val?.type === 'class_constant_access_expression') {
|
||||
ctx.controller = val.children?.find((c: any) => c.type === 'name')?.text ?? null;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
return ctx;
|
||||
}
|
||||
|
||||
function extractLaravelRoutes(tree: any, filePath: string): ExtractedRoute[] {
|
||||
const routes: ExtractedRoute[] = [];
|
||||
|
||||
function resolveStack(stack: RouteGroupContext[]): { middleware: string[]; prefix: string | null; controller: string | null } {
|
||||
const middleware: string[] = [];
|
||||
let prefix: string | null = null;
|
||||
let controller: string | null = null;
|
||||
for (const ctx of stack) {
|
||||
middleware.push(...ctx.middleware);
|
||||
if (ctx.prefix) prefix = prefix ? `${prefix}/${ctx.prefix}`.replace(/\/+/g, '/') : ctx.prefix;
|
||||
if (ctx.controller) controller = ctx.controller;
|
||||
}
|
||||
return { middleware, prefix, controller };
|
||||
}
|
||||
|
||||
function emitRoute(
|
||||
httpMethod: string,
|
||||
argsNode: any,
|
||||
lineNumber: number,
|
||||
groupStack: RouteGroupContext[],
|
||||
chainAttrs: { method: string; argsNode: any }[],
|
||||
) {
|
||||
const effective = resolveStack(groupStack);
|
||||
|
||||
for (const attr of chainAttrs) {
|
||||
if (attr.method === 'middleware') effective.middleware.push(...extractMiddlewareArg(attr.argsNode));
|
||||
if (attr.method === 'prefix') {
|
||||
const p = extractFirstStringArg(attr.argsNode);
|
||||
if (p) effective.prefix = effective.prefix ? `${effective.prefix}/${p}` : p;
|
||||
}
|
||||
if (attr.method === 'controller') {
|
||||
const cls = extractClassArg(attr.argsNode);
|
||||
if (cls) effective.controller = cls;
|
||||
}
|
||||
}
|
||||
|
||||
const routePath = extractFirstStringArg(argsNode);
|
||||
|
||||
if (ROUTE_RESOURCE_METHODS.has(httpMethod)) {
|
||||
const target = extractControllerTarget(argsNode);
|
||||
const actions = httpMethod === 'apiResource' ? API_RESOURCE_ACTIONS : RESOURCE_ACTIONS;
|
||||
for (const action of actions) {
|
||||
routes.push({
|
||||
filePath, httpMethod, routePath,
|
||||
controllerName: target.controller ?? effective.controller,
|
||||
methodName: action,
|
||||
middleware: [...effective.middleware],
|
||||
prefix: effective.prefix,
|
||||
lineNumber,
|
||||
});
|
||||
}
|
||||
} else {
|
||||
const target = extractControllerTarget(argsNode);
|
||||
routes.push({
|
||||
filePath, httpMethod, routePath,
|
||||
controllerName: target.controller ?? effective.controller,
|
||||
methodName: target.method,
|
||||
middleware: [...effective.middleware],
|
||||
prefix: effective.prefix,
|
||||
lineNumber,
|
||||
});
|
||||
}
|
||||
}
|
||||
|
||||
function walk(node: any, groupStack: RouteGroupContext[]) {
|
||||
if (isRouteStaticCall(node)) {
|
||||
const method = getCallMethodName(node);
|
||||
if (method && (ROUTE_HTTP_METHODS.has(method) || ROUTE_RESOURCE_METHODS.has(method))) {
|
||||
emitRoute(method, getArguments(node), node.startPosition.row, groupStack, []);
|
||||
return;
|
||||
}
|
||||
if (method === 'group') {
|
||||
const argsNode = getArguments(node);
|
||||
const groupCtx = parseArrayGroupArgs(argsNode);
|
||||
const body = findClosureBody(argsNode);
|
||||
if (body) {
|
||||
groupStack.push(groupCtx);
|
||||
walkChildren(body, groupStack);
|
||||
groupStack.pop();
|
||||
}
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
const chain = unwrapRouteChain(node);
|
||||
if (chain) {
|
||||
if (chain.terminalMethod === 'group') {
|
||||
const groupCtx: RouteGroupContext = { middleware: [], prefix: null, controller: null };
|
||||
for (const attr of chain.attributes) {
|
||||
if (attr.method === 'middleware') groupCtx.middleware.push(...extractMiddlewareArg(attr.argsNode));
|
||||
if (attr.method === 'prefix') groupCtx.prefix = extractFirstStringArg(attr.argsNode);
|
||||
if (attr.method === 'controller') groupCtx.controller = extractClassArg(attr.argsNode);
|
||||
}
|
||||
const body = findClosureBody(chain.terminalArgs);
|
||||
if (body) {
|
||||
groupStack.push(groupCtx);
|
||||
walkChildren(body, groupStack);
|
||||
groupStack.pop();
|
||||
}
|
||||
return;
|
||||
}
|
||||
if (ROUTE_HTTP_METHODS.has(chain.terminalMethod) || ROUTE_RESOURCE_METHODS.has(chain.terminalMethod)) {
|
||||
emitRoute(chain.terminalMethod, chain.terminalArgs, node.startPosition.row, groupStack, chain.attributes);
|
||||
return;
|
||||
}
|
||||
}
|
||||
|
||||
walkChildren(node, groupStack);
|
||||
}
|
||||
|
||||
function walkChildren(node: any, groupStack: RouteGroupContext[]) {
|
||||
for (const child of node.children ?? []) {
|
||||
walk(child, groupStack);
|
||||
}
|
||||
}
|
||||
|
||||
walk(tree.rootNode, []);
|
||||
return routes;
|
||||
}
|
||||
|
||||
/**
|
||||
* Resolution result with confidence scoring
|
||||
*/
|
||||
@@ -278,37 +832,72 @@ const resolveCallTarget = (
|
||||
* Filter out common built-in functions and noise
|
||||
* that shouldn't be tracked as calls
|
||||
*/
|
||||
const isBuiltInOrNoise = (name: string): boolean => {
|
||||
const builtIns = new Set([
|
||||
// JavaScript/TypeScript built-ins
|
||||
'console', 'log', 'warn', 'error', 'info', 'debug',
|
||||
'setTimeout', 'setInterval', 'clearTimeout', 'clearInterval',
|
||||
'parseInt', 'parseFloat', 'isNaN', 'isFinite',
|
||||
'encodeURI', 'decodeURI', 'encodeURIComponent', 'decodeURIComponent',
|
||||
'JSON', 'parse', 'stringify',
|
||||
'Object', 'Array', 'String', 'Number', 'Boolean', 'Symbol', 'BigInt',
|
||||
'Map', 'Set', 'WeakMap', 'WeakSet',
|
||||
'Promise', 'resolve', 'reject', 'then', 'catch', 'finally',
|
||||
'Math', 'Date', 'RegExp', 'Error',
|
||||
'require', 'import', 'export',
|
||||
'fetch', 'Response', 'Request',
|
||||
// React hooks and common functions
|
||||
'useState', 'useEffect', 'useCallback', 'useMemo', 'useRef', 'useContext',
|
||||
'useReducer', 'useLayoutEffect', 'useImperativeHandle', 'useDebugValue',
|
||||
'createElement', 'createContext', 'createRef', 'forwardRef', 'memo', 'lazy',
|
||||
// Common array/object methods
|
||||
'map', 'filter', 'reduce', 'forEach', 'find', 'findIndex', 'some', 'every',
|
||||
'includes', 'indexOf', 'slice', 'splice', 'concat', 'join', 'split',
|
||||
'push', 'pop', 'shift', 'unshift', 'sort', 'reverse',
|
||||
'keys', 'values', 'entries', 'assign', 'freeze', 'seal',
|
||||
'hasOwnProperty', 'toString', 'valueOf',
|
||||
// Python built-ins
|
||||
'print', 'len', 'range', 'str', 'int', 'float', 'list', 'dict', 'set', 'tuple',
|
||||
'open', 'read', 'write', 'close', 'append', 'extend', 'update',
|
||||
'super', 'type', 'isinstance', 'issubclass', 'getattr', 'setattr', 'hasattr',
|
||||
'enumerate', 'zip', 'sorted', 'reversed', 'min', 'max', 'sum', 'abs',
|
||||
]);
|
||||
/** Pre-built set (module-level singleton) to avoid re-creating per call */
|
||||
const BUILT_IN_NAMES = new Set([
|
||||
// JavaScript/TypeScript built-ins
|
||||
'console', 'log', 'warn', 'error', 'info', 'debug',
|
||||
'setTimeout', 'setInterval', 'clearTimeout', 'clearInterval',
|
||||
'parseInt', 'parseFloat', 'isNaN', 'isFinite',
|
||||
'encodeURI', 'decodeURI', 'encodeURIComponent', 'decodeURIComponent',
|
||||
'JSON', 'parse', 'stringify',
|
||||
'Object', 'Array', 'String', 'Number', 'Boolean', 'Symbol', 'BigInt',
|
||||
'Map', 'Set', 'WeakMap', 'WeakSet',
|
||||
'Promise', 'resolve', 'reject', 'then', 'catch', 'finally',
|
||||
'Math', 'Date', 'RegExp', 'Error',
|
||||
'require', 'import', 'export',
|
||||
'fetch', 'Response', 'Request',
|
||||
// React hooks and common functions
|
||||
'useState', 'useEffect', 'useCallback', 'useMemo', 'useRef', 'useContext',
|
||||
'useReducer', 'useLayoutEffect', 'useImperativeHandle', 'useDebugValue',
|
||||
'createElement', 'createContext', 'createRef', 'forwardRef', 'memo', 'lazy',
|
||||
// Common array/object methods
|
||||
'map', 'filter', 'reduce', 'forEach', 'find', 'findIndex', 'some', 'every',
|
||||
'includes', 'indexOf', 'slice', 'splice', 'concat', 'join', 'split',
|
||||
'push', 'pop', 'shift', 'unshift', 'sort', 'reverse',
|
||||
'keys', 'values', 'entries', 'assign', 'freeze', 'seal',
|
||||
'hasOwnProperty', 'toString', 'valueOf',
|
||||
// Python built-ins
|
||||
'print', 'len', 'range', 'str', 'int', 'float', 'list', 'dict', 'set', 'tuple',
|
||||
'open', 'read', 'write', 'close', 'append', 'extend', 'update',
|
||||
'super', 'type', 'isinstance', 'issubclass', 'getattr', 'setattr', 'hasattr',
|
||||
'enumerate', 'zip', 'sorted', 'reversed', 'min', 'max', 'sum', 'abs',
|
||||
// C/C++ standard library and common kernel helpers
|
||||
'printf', 'fprintf', 'sprintf', 'snprintf', 'vprintf', 'vfprintf', 'vsprintf', 'vsnprintf',
|
||||
'scanf', 'fscanf', 'sscanf',
|
||||
'malloc', 'calloc', 'realloc', 'free', 'memcpy', 'memmove', 'memset', 'memcmp',
|
||||
'strlen', 'strcpy', 'strncpy', 'strcat', 'strncat', 'strcmp', 'strncmp', 'strstr', 'strchr', 'strrchr',
|
||||
'atoi', 'atol', 'atof', 'strtol', 'strtoul', 'strtoll', 'strtoull', 'strtod',
|
||||
'sizeof', 'offsetof', 'typeof',
|
||||
'assert', 'abort', 'exit', '_exit',
|
||||
'fopen', 'fclose', 'fread', 'fwrite', 'fseek', 'ftell', 'rewind', 'fflush', 'fgets', 'fputs',
|
||||
// Linux kernel common macros/helpers (not real call targets)
|
||||
'likely', 'unlikely', 'BUG', 'BUG_ON', 'WARN', 'WARN_ON', 'WARN_ONCE',
|
||||
'IS_ERR', 'PTR_ERR', 'ERR_PTR', 'IS_ERR_OR_NULL',
|
||||
'ARRAY_SIZE', 'container_of', 'list_for_each_entry', 'list_for_each_entry_safe',
|
||||
'min', 'max', 'clamp', 'abs', 'swap',
|
||||
'pr_info', 'pr_warn', 'pr_err', 'pr_debug', 'pr_notice', 'pr_crit', 'pr_emerg',
|
||||
'printk', 'dev_info', 'dev_warn', 'dev_err', 'dev_dbg',
|
||||
'GFP_KERNEL', 'GFP_ATOMIC',
|
||||
'spin_lock', 'spin_unlock', 'spin_lock_irqsave', 'spin_unlock_irqrestore',
|
||||
'mutex_lock', 'mutex_unlock', 'mutex_init',
|
||||
'kfree', 'kmalloc', 'kzalloc', 'kcalloc', 'krealloc', 'kvmalloc', 'kvfree',
|
||||
'get', 'put',
|
||||
// Ruby built-ins and Kernel methods
|
||||
'puts', 'print', 'p', 'pp', 'warn', 'raise', 'fail',
|
||||
'require', 'require_relative', 'load', 'autoload',
|
||||
'include', 'extend', 'prepend',
|
||||
'attr_accessor', 'attr_reader', 'attr_writer',
|
||||
'public', 'private', 'protected', 'module_function',
|
||||
'lambda', 'proc', 'block_given?',
|
||||
'nil?', 'is_a?', 'kind_of?', 'instance_of?', 'respond_to?',
|
||||
'freeze', 'frozen?', 'dup', 'clone', 'tap', 'then', 'yield_self',
|
||||
// Ruby enumerables
|
||||
'each', 'map', 'select', 'reject', 'find', 'detect', 'collect',
|
||||
'inject', 'reduce', 'flat_map', 'each_with_object', 'each_with_index',
|
||||
'any?', 'all?', 'none?', 'count', 'first', 'last',
|
||||
'sort', 'sort_by', 'min', 'max', 'min_by', 'max_by',
|
||||
'group_by', 'partition', 'zip', 'compact', 'flatten', 'uniq',
|
||||
]);
|
||||
|
||||
return builtIns.has(name);
|
||||
};
|
||||
const isBuiltInOrNoise = (name: string): boolean => BUILT_IN_NAMES.has(name);
|
||||
|
||||
|
||||
@@ -0,0 +1,149 @@
|
||||
/**
|
||||
* Shared Ruby call routing logic.
|
||||
*
|
||||
* Ruby expresses imports, heritage (mixins), and property definitions as
|
||||
* method calls rather than syntax-level constructs. This module provides a
|
||||
* routing function used by the CLI call-processor, CLI parse-worker, and
|
||||
* the web call-processor so that the classification logic lives in one place.
|
||||
*
|
||||
* NOTE: This file is intentionally duplicated in gitnexus-web/ because the
|
||||
* two packages have separate build targets (Node native vs WASM/browser).
|
||||
* Keep both copies in sync until a shared package is introduced.
|
||||
*/
|
||||
|
||||
import { SupportedLanguages } from '../../config/supported-languages';
|
||||
|
||||
// ── Call routing dispatch table ─────────────────────────────────────────────
|
||||
|
||||
/** null = this call was not routed; fall through to default call handling */
|
||||
export type CallRoutingResult = RubyCallRouting | null;
|
||||
|
||||
export type CallRouter = (
|
||||
calledName: string,
|
||||
callNode: any,
|
||||
) => CallRoutingResult;
|
||||
|
||||
/** No-op router: returns null for every call (passthrough to normal processing) */
|
||||
const noRouting: CallRouter = () => null;
|
||||
|
||||
/** Per-language call routing. noRouting = no special routing (normal call processing) */
|
||||
export const callRouters = {
|
||||
[SupportedLanguages.JavaScript]: noRouting,
|
||||
[SupportedLanguages.TypeScript]: noRouting,
|
||||
[SupportedLanguages.Python]: noRouting,
|
||||
[SupportedLanguages.Java]: noRouting,
|
||||
[SupportedLanguages.Go]: noRouting,
|
||||
[SupportedLanguages.Rust]: noRouting,
|
||||
[SupportedLanguages.CSharp]: noRouting,
|
||||
[SupportedLanguages.PHP]: noRouting,
|
||||
[SupportedLanguages.Swift]: noRouting,
|
||||
[SupportedLanguages.CPlusPlus]: noRouting,
|
||||
[SupportedLanguages.C]: noRouting,
|
||||
[SupportedLanguages.Ruby]: routeRubyCall,
|
||||
[SupportedLanguages.Kotlin]: noRouting,
|
||||
} satisfies Record<SupportedLanguages, CallRouter>;
|
||||
|
||||
// ── Result types ────────────────────────────────────────────────────────────
|
||||
|
||||
export type RubyCallRouting =
|
||||
| { kind: 'import'; importPath: string; isRelative: boolean }
|
||||
| { kind: 'heritage'; items: RubyHeritageItem[] }
|
||||
| { kind: 'properties'; items: RubyPropertyItem[] }
|
||||
| { kind: 'call' }
|
||||
| { kind: 'skip' };
|
||||
|
||||
export interface RubyHeritageItem {
|
||||
enclosingClass: string;
|
||||
mixinName: string;
|
||||
heritageKind: 'include' | 'extend' | 'prepend';
|
||||
}
|
||||
|
||||
export type RubyAccessorType = 'attr_accessor' | 'attr_reader' | 'attr_writer';
|
||||
|
||||
export interface RubyPropertyItem {
|
||||
propName: string;
|
||||
accessorType: RubyAccessorType;
|
||||
startLine: number;
|
||||
endLine: number;
|
||||
}
|
||||
|
||||
// ── Pre-allocated singletons for common return values ────────────────────────
|
||||
const CALL_RESULT: RubyCallRouting = { kind: 'call' };
|
||||
const SKIP_RESULT: RubyCallRouting = { kind: 'skip' };
|
||||
|
||||
/** Max depth for parent-walking loops to prevent pathological AST traversals */
|
||||
const MAX_PARENT_DEPTH = 50;
|
||||
|
||||
// ── Routing function ────────────────────────────────────────────────────────
|
||||
|
||||
/**
|
||||
* Classify a Ruby call node and extract its semantic payload.
|
||||
*
|
||||
* @param calledName - The method name (e.g. 'require', 'include', 'attr_accessor')
|
||||
* @param callNode - The tree-sitter `call` AST node
|
||||
* @returns A discriminated union describing the call's semantic role
|
||||
*/
|
||||
export function routeRubyCall(calledName: string, callNode: any): RubyCallRouting {
|
||||
// ── require / require_relative → import ─────────────────────────────────
|
||||
if (calledName === 'require' || calledName === 'require_relative') {
|
||||
const argList = callNode.childForFieldName?.('arguments');
|
||||
const stringNode = argList?.children?.find((c: any) => c.type === 'string');
|
||||
const contentNode = stringNode?.children?.find((c: any) => c.type === 'string_content');
|
||||
if (!contentNode) return SKIP_RESULT;
|
||||
|
||||
let importPath: string = contentNode.text;
|
||||
// Validate: reject null bytes, control chars, excessively long paths
|
||||
if (!importPath || importPath.length > 1024 || /[\x00-\x1f]/.test(importPath)) {
|
||||
return SKIP_RESULT;
|
||||
}
|
||||
const isRelative = calledName === 'require_relative';
|
||||
if (isRelative && !importPath.startsWith('.')) {
|
||||
importPath = './' + importPath;
|
||||
}
|
||||
return { kind: 'import', importPath, isRelative };
|
||||
}
|
||||
|
||||
// ── include / extend / prepend → heritage (mixin) ──────────────────────
|
||||
if (calledName === 'include' || calledName === 'extend' || calledName === 'prepend') {
|
||||
let enclosingClass: string | null = null;
|
||||
let current = callNode.parent;
|
||||
let depth = 0;
|
||||
while (current && ++depth <= MAX_PARENT_DEPTH) {
|
||||
if (current.type === 'class' || current.type === 'module') {
|
||||
const nameNode = current.childForFieldName?.('name');
|
||||
if (nameNode) { enclosingClass = nameNode.text; break; }
|
||||
}
|
||||
current = current.parent;
|
||||
}
|
||||
if (!enclosingClass) return SKIP_RESULT;
|
||||
|
||||
const items: RubyHeritageItem[] = [];
|
||||
const argList = callNode.childForFieldName?.('arguments');
|
||||
for (const arg of (argList?.children ?? [])) {
|
||||
if (arg.type === 'constant' || arg.type === 'scope_resolution') {
|
||||
items.push({ enclosingClass, mixinName: arg.text, heritageKind: calledName as 'include' | 'extend' | 'prepend' });
|
||||
}
|
||||
}
|
||||
return items.length > 0 ? { kind: 'heritage', items } : SKIP_RESULT;
|
||||
}
|
||||
|
||||
// ── attr_accessor / attr_reader / attr_writer → property definitions ───
|
||||
if (calledName === 'attr_accessor' || calledName === 'attr_reader' || calledName === 'attr_writer') {
|
||||
const items: RubyPropertyItem[] = [];
|
||||
const argList = callNode.childForFieldName?.('arguments');
|
||||
for (const arg of (argList?.children ?? [])) {
|
||||
if (arg.type === 'simple_symbol') {
|
||||
items.push({
|
||||
propName: arg.text.startsWith(':') ? arg.text.slice(1) : arg.text,
|
||||
accessorType: calledName as RubyAccessorType,
|
||||
startLine: arg.startPosition.row,
|
||||
endLine: arg.endPosition.row,
|
||||
});
|
||||
}
|
||||
}
|
||||
return items.length > 0 ? { kind: 'properties', items } : SKIP_RESULT;
|
||||
}
|
||||
|
||||
// ── Everything else → regular call ─────────────────────────────────────
|
||||
return CALL_RESULT;
|
||||
}
|
||||
@@ -330,25 +330,20 @@ const calculateCohesion = (memberIds: string[], graph: Graph): number => {
|
||||
|
||||
const memberSet = new Set(memberIds);
|
||||
let internalEdges = 0;
|
||||
|
||||
// Count edges within the community
|
||||
let totalEdges = 0;
|
||||
|
||||
// Count internal vs total edges for community members
|
||||
memberIds.forEach(nodeId => {
|
||||
if (graph.hasNode(nodeId)) {
|
||||
graph.forEachNeighbor(nodeId, neighbor => {
|
||||
totalEdges++;
|
||||
if (memberSet.has(neighbor)) {
|
||||
internalEdges++;
|
||||
}
|
||||
});
|
||||
}
|
||||
});
|
||||
|
||||
// Each edge is counted twice (once from each end), so divide by 2
|
||||
internalEdges = internalEdges / 2;
|
||||
|
||||
// Maximum possible internal edges for n nodes: n*(n-1)/2
|
||||
const maxPossibleEdges = (memberIds.length * (memberIds.length - 1)) / 2;
|
||||
|
||||
if (maxPossibleEdges === 0) return 1.0;
|
||||
|
||||
return Math.min(1.0, internalEdges / maxPossibleEdges);
|
||||
|
||||
if (totalEdges === 0) return 1.0;
|
||||
return Math.min(1.0, internalEdges / totalEdges);
|
||||
};
|
||||
|
||||
@@ -13,7 +13,7 @@
|
||||
import { detectFrameworkFromPath } from './framework-detection';
|
||||
|
||||
// ============================================================================
|
||||
// NAME PATTERNS - All 9 supported languages
|
||||
// NAME PATTERNS - All 11 supported languages
|
||||
// ============================================================================
|
||||
|
||||
/**
|
||||
@@ -103,6 +103,26 @@ const ENTRY_POINT_PATTERNS: Record<string, RegExp[]> = {
|
||||
/^Start$/, // Start methods
|
||||
],
|
||||
|
||||
// Swift / iOS
|
||||
'swift': [
|
||||
/^viewDidLoad$/, // UIKit lifecycle
|
||||
/^viewWillAppear$/, // UIKit lifecycle
|
||||
/^viewDidAppear$/, // UIKit lifecycle
|
||||
/^viewWillDisappear$/, // UIKit lifecycle
|
||||
/^viewDidDisappear$/, // UIKit lifecycle
|
||||
/^application\(/, // AppDelegate methods
|
||||
/^scene\(/, // SceneDelegate methods
|
||||
/^body$/, // SwiftUI View.body
|
||||
/Coordinator$/, // Coordinator pattern
|
||||
/^sceneDidBecomeActive$/, // SceneDelegate lifecycle
|
||||
/^sceneWillResignActive$/, // SceneDelegate lifecycle
|
||||
/^didFinishLaunchingWithOptions$/, // AppDelegate
|
||||
/ViewController$/, // ViewController classes
|
||||
/^configure[A-Z]/, // Configuration methods
|
||||
/^setup[A-Z]/, // Setup methods
|
||||
/^makeBody$/, // SwiftUI ViewModifier
|
||||
],
|
||||
|
||||
// PHP / Laravel
|
||||
'php': [
|
||||
/Controller$/, // UserController (class name convention)
|
||||
@@ -123,6 +143,13 @@ const ENTRY_POINT_PATTERNS: Record<string, RegExp[]> = {
|
||||
/^save$/, // Repository::save()
|
||||
/^delete$/, // Repository::delete()
|
||||
],
|
||||
|
||||
// Ruby
|
||||
'ruby': [
|
||||
/^call$/, // Service objects (MyService.call)
|
||||
/^perform$/, // Background jobs (Sidekiq, ActiveJob)
|
||||
/^execute$/, // Command pattern
|
||||
],
|
||||
};
|
||||
|
||||
// ============================================================================
|
||||
@@ -271,6 +298,10 @@ export function isTestFile(filePath: string): boolean {
|
||||
p.includes('/src/test/') ||
|
||||
// Rust test patterns (inline tests are different, but test files)
|
||||
p.includes('/tests/') ||
|
||||
// Swift/iOS test patterns
|
||||
p.endsWith('tests.swift') ||
|
||||
p.endsWith('test.swift') ||
|
||||
p.includes('uitests/') ||
|
||||
// C# test patterns
|
||||
p.includes('.tests/') ||
|
||||
p.includes('tests.cs') ||
|
||||
@@ -278,7 +309,12 @@ export function isTestFile(filePath: string): boolean {
|
||||
p.endsWith('test.php') ||
|
||||
p.endsWith('spec.php') ||
|
||||
p.includes('/tests/feature/') ||
|
||||
p.includes('/tests/unit/')
|
||||
p.includes('/tests/unit/') ||
|
||||
// Ruby test patterns
|
||||
p.endsWith('_spec.rb') ||
|
||||
p.endsWith('_test.rb') ||
|
||||
p.includes('/spec/') ||
|
||||
p.includes('/test/fixtures/')
|
||||
);
|
||||
}
|
||||
|
||||
|
||||
@@ -257,22 +257,81 @@ export function detectFrameworkFromPath(filePath: string): FrameworkHint | null
|
||||
return { framework: 'laravel', entryPointMultiplier: 1.5, reason: 'laravel-repository' };
|
||||
}
|
||||
|
||||
// ========== RUBY ==========
|
||||
|
||||
// Ruby: bin/ or exe/ (CLI entry points)
|
||||
if ((p.includes('/bin/') || p.includes('/exe/')) && p.endsWith('.rb')) {
|
||||
return { framework: 'ruby', entryPointMultiplier: 2.5, reason: 'ruby-executable' };
|
||||
}
|
||||
|
||||
// Ruby: Rakefile or *.rake (task definitions)
|
||||
if (p.endsWith('/rakefile') || p.endsWith('.rake')) {
|
||||
return { framework: 'ruby', entryPointMultiplier: 1.5, reason: 'ruby-rake' };
|
||||
}
|
||||
// ========== SWIFT / iOS ==========
|
||||
|
||||
// iOS App entry points (highest priority)
|
||||
if (p.endsWith('/appdelegate.swift') || p.endsWith('/scenedelegate.swift') || p.endsWith('/app.swift')) {
|
||||
return { framework: 'ios', entryPointMultiplier: 3.0, reason: 'ios-app-entry' };
|
||||
}
|
||||
|
||||
// SwiftUI App entry (@main)
|
||||
if (p.endsWith('app.swift') && p.includes('/sources/')) {
|
||||
return { framework: 'swiftui', entryPointMultiplier: 3.0, reason: 'swiftui-app' };
|
||||
}
|
||||
|
||||
// UIKit ViewControllers (high priority - screen entry points)
|
||||
if ((p.includes('/viewcontrollers/') || p.includes('/controllers/') || p.includes('/screens/')) && p.endsWith('.swift')) {
|
||||
return { framework: 'uikit', entryPointMultiplier: 2.5, reason: 'uikit-viewcontroller' };
|
||||
}
|
||||
|
||||
// ViewController by filename convention
|
||||
if (p.endsWith('viewcontroller.swift') || p.endsWith('vc.swift')) {
|
||||
return { framework: 'uikit', entryPointMultiplier: 2.5, reason: 'uikit-viewcontroller-file' };
|
||||
}
|
||||
|
||||
// Coordinator pattern (navigation entry points)
|
||||
if (p.includes('/coordinators/') && p.endsWith('.swift')) {
|
||||
return { framework: 'ios-coordinator', entryPointMultiplier: 2.5, reason: 'ios-coordinator' };
|
||||
}
|
||||
|
||||
// Coordinator by filename
|
||||
if (p.endsWith('coordinator.swift')) {
|
||||
return { framework: 'ios-coordinator', entryPointMultiplier: 2.5, reason: 'ios-coordinator-file' };
|
||||
}
|
||||
|
||||
// SwiftUI Views (moderate - reusable components)
|
||||
if ((p.includes('/views/') || p.includes('/scenes/')) && p.endsWith('.swift')) {
|
||||
return { framework: 'swiftui', entryPointMultiplier: 1.8, reason: 'swiftui-view' };
|
||||
}
|
||||
|
||||
// Service layer
|
||||
if (p.includes('/services/') && p.endsWith('.swift')) {
|
||||
return { framework: 'ios-service', entryPointMultiplier: 1.8, reason: 'ios-service' };
|
||||
}
|
||||
|
||||
// Router / navigation
|
||||
if (p.includes('/router/') && p.endsWith('.swift')) {
|
||||
return { framework: 'ios-router', entryPointMultiplier: 2.0, reason: 'ios-router' };
|
||||
}
|
||||
|
||||
// ========== GENERIC PATTERNS ==========
|
||||
|
||||
// Any language: index files in API folders
|
||||
if (p.includes('/api/') && (
|
||||
p.endsWith('/index.ts') || p.endsWith('/index.js') ||
|
||||
p.endsWith('/index.ts') || p.endsWith('/index.js') ||
|
||||
p.endsWith('/__init__.py')
|
||||
)) {
|
||||
return { framework: 'api', entryPointMultiplier: 1.8, reason: 'api-index' };
|
||||
}
|
||||
|
||||
|
||||
// No framework detected - return null for graceful fallback (1.0 multiplier)
|
||||
return null;
|
||||
}
|
||||
|
||||
// ============================================================================
|
||||
// FUTURE: AST-BASED PATTERNS (for Phase 3)
|
||||
// PARTIALLY IMPLEMENTED: Route::* detection via procedural AST walk in parse-worker/call-processor
|
||||
// Remaining: NestJS, Express, FastAPI, Flask, Spring, etc.
|
||||
// ============================================================================
|
||||
|
||||
/**
|
||||
@@ -306,4 +365,9 @@ export const FRAMEWORK_AST_PATTERNS = {
|
||||
'actix': ['#[get', '#[post', '#[put', '#[delete'],
|
||||
'axum': ['Router::new'],
|
||||
'rocket': ['#[get', '#[post'],
|
||||
|
||||
// Swift/iOS
|
||||
'uikit': ['viewDidLoad', 'viewWillAppear', 'viewDidAppear', 'UIViewController'],
|
||||
'swiftui': ['@main', 'WindowGroup', 'ContentView', '@StateObject', '@ObservedObject'],
|
||||
'combine': ['sink', 'assign', 'Publisher', 'Subscriber'],
|
||||
};
|
||||
|
||||
@@ -4,6 +4,7 @@ import { loadParser, loadLanguage } from '../tree-sitter/parser-loader';
|
||||
import { LANGUAGE_QUERIES } from './tree-sitter-queries';
|
||||
import { generateId } from '../../lib/utils';
|
||||
import { getLanguageFromFilename } from './utils';
|
||||
import { callRouters } from './call-routing';
|
||||
|
||||
// Type: Map<FilePath, Set<ResolvedFilePath>>
|
||||
// Stores all files that a given file imports from
|
||||
@@ -53,7 +54,9 @@ const resolveImportPath = (
|
||||
// Go
|
||||
'.go',
|
||||
// Rust
|
||||
'.rs', '/mod.rs'
|
||||
'.rs', '/mod.rs',
|
||||
// Ruby
|
||||
'.rb', '.rake',
|
||||
];
|
||||
|
||||
if (importPath.startsWith('.')) {
|
||||
@@ -220,6 +223,35 @@ export const processImports = async (
|
||||
importMap.get(file.path)!.add(resolvedPath);
|
||||
}
|
||||
}
|
||||
|
||||
// ---- Language-specific call-as-import routing (Ruby require, etc.) ----
|
||||
if (captureMap['call']) {
|
||||
const callNameNode = captureMap['call.name'];
|
||||
if (callNameNode) {
|
||||
const callRouter = callRouters[language];
|
||||
const routed = callRouter(callNameNode.text, captureMap['call']);
|
||||
if (routed && routed.kind === 'import') {
|
||||
totalImportsFound++;
|
||||
const resolvedPath = resolveImportPath(
|
||||
file.path, routed.importPath, allFilePaths, allFileList, resolveCache
|
||||
);
|
||||
if (resolvedPath) {
|
||||
const sourceId = generateId('File', file.path);
|
||||
const targetId = generateId('File', resolvedPath);
|
||||
const relId = generateId('IMPORTS', `${file.path}->${resolvedPath}`);
|
||||
totalImportsResolved++;
|
||||
graph.addRelationship({
|
||||
id: relId, sourceId, targetId,
|
||||
type: 'IMPORTS', confidence: 1.0, reason: '',
|
||||
});
|
||||
if (!importMap.has(file.path)) {
|
||||
importMap.set(file.path, new Set());
|
||||
}
|
||||
importMap.get(file.path)!.add(resolvedPath);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
});
|
||||
|
||||
// If re-parsed just for this, delete the tree to save memory
|
||||
|
||||
@@ -14,7 +14,7 @@ export type FileProgressCallback = (current: number, total: number, filePath: st
|
||||
|
||||
/**
|
||||
* Check if a symbol (function, class, etc.) is exported/public
|
||||
* Handles all 9 supported languages with explicit logic
|
||||
* Handles all 11 supported languages with explicit logic
|
||||
*
|
||||
* @param node - The AST node for the symbol name
|
||||
* @param name - The symbol name
|
||||
@@ -104,7 +104,11 @@ const isNodeExported = (node: any, name: string, language: string): boolean => {
|
||||
case 'c':
|
||||
case 'cpp':
|
||||
return false;
|
||||
|
||||
|
||||
// Ruby: All top-level definitions are public by default
|
||||
case 'ruby':
|
||||
return true;
|
||||
|
||||
default:
|
||||
return false;
|
||||
}
|
||||
|
||||
@@ -396,6 +396,93 @@ export const PHP_QUERIES = `
|
||||
[(name) (qualified_name)] @heritage.trait))) @heritage
|
||||
`;
|
||||
|
||||
// Ruby queries - works with tree-sitter-ruby
|
||||
// NOTE: Ruby uses `call` for require, include, extend, prepend, attr_* etc.
|
||||
// These are all captured as @call and routed in JS post-processing:
|
||||
// - require/require_relative → import extraction
|
||||
// - include/extend/prepend → heritage (mixin) extraction
|
||||
// - attr_accessor/attr_reader/attr_writer → property definition extraction
|
||||
// - everything else → regular call extraction
|
||||
export const RUBY_QUERIES = `
|
||||
; ── Modules ──────────────────────────────────────────────────────────────────
|
||||
(module
|
||||
name: (constant) @name) @definition.module
|
||||
|
||||
; ── Classes ──────────────────────────────────────────────────────────────────
|
||||
(class
|
||||
name: (constant) @name) @definition.class
|
||||
|
||||
; ── Instance methods ─────────────────────────────────────────────────────────
|
||||
(method
|
||||
name: (identifier) @name) @definition.method
|
||||
|
||||
; ── Singleton (class-level) methods ──────────────────────────────────────────
|
||||
(singleton_method
|
||||
name: (identifier) @name) @definition.function
|
||||
|
||||
; ── All calls (require, include, attr_*, and regular calls routed in JS) ─────
|
||||
(call
|
||||
method: (identifier) @call.name) @call
|
||||
|
||||
; ── Heritage: class < SuperClass ─────────────────────────────────────────────
|
||||
(class
|
||||
name: (constant) @heritage.class
|
||||
superclass: (superclass
|
||||
(constant) @heritage.extends)) @heritage`;
|
||||
|
||||
// Swift queries - works with tree-sitter-swift
|
||||
export const SWIFT_QUERIES = `
|
||||
; Classes
|
||||
(class_declaration "class" name: (type_identifier) @name) @definition.class
|
||||
|
||||
; Structs
|
||||
(class_declaration "struct" name: (type_identifier) @name) @definition.struct
|
||||
|
||||
; Enums
|
||||
(class_declaration "enum" name: (type_identifier) @name) @definition.enum
|
||||
|
||||
; Extensions (mapped to class — no dedicated label in schema)
|
||||
(class_declaration "extension" name: (user_type (type_identifier) @name)) @definition.class
|
||||
|
||||
; Actors
|
||||
(class_declaration "actor" name: (type_identifier) @name) @definition.class
|
||||
|
||||
; Protocols (mapped to interface)
|
||||
(protocol_declaration name: (type_identifier) @name) @definition.interface
|
||||
|
||||
; Type aliases
|
||||
(typealias_declaration name: (type_identifier) @name) @definition.type
|
||||
|
||||
; Functions (top-level and methods)
|
||||
(function_declaration name: (simple_identifier) @name) @definition.function
|
||||
|
||||
; Protocol method declarations
|
||||
(protocol_function_declaration name: (simple_identifier) @name) @definition.method
|
||||
|
||||
; Initializers
|
||||
(init_declaration) @definition.constructor
|
||||
|
||||
; Properties (stored and computed)
|
||||
(property_declaration (pattern (simple_identifier) @name)) @definition.property
|
||||
|
||||
; Imports
|
||||
(import_declaration (identifier (simple_identifier) @import.source)) @import
|
||||
|
||||
; Calls - direct function calls
|
||||
(call_expression (simple_identifier) @call.name) @call
|
||||
|
||||
; Calls - member/navigation calls (obj.method())
|
||||
(call_expression (navigation_expression (navigation_suffix (simple_identifier) @call.name))) @call
|
||||
|
||||
; Heritage - class/struct/enum inheritance and protocol conformance
|
||||
(class_declaration name: (type_identifier) @heritage.class
|
||||
(inheritance_specifier inherits_from: (user_type (type_identifier) @heritage.extends))) @heritage
|
||||
|
||||
; Heritage - protocol inheritance
|
||||
(protocol_declaration name: (type_identifier) @heritage.class
|
||||
(inheritance_specifier inherits_from: (user_type (type_identifier) @heritage.extends))) @heritage
|
||||
`;
|
||||
|
||||
export const LANGUAGE_QUERIES: Record<SupportedLanguages, string> = {
|
||||
[SupportedLanguages.TypeScript]: TYPESCRIPT_QUERIES,
|
||||
[SupportedLanguages.JavaScript]: JAVASCRIPT_QUERIES,
|
||||
@@ -407,5 +494,8 @@ export const LANGUAGE_QUERIES: Record<SupportedLanguages, string> = {
|
||||
[SupportedLanguages.CSharp]: CSHARP_QUERIES,
|
||||
[SupportedLanguages.Rust]: RUST_QUERIES,
|
||||
[SupportedLanguages.PHP]: PHP_QUERIES,
|
||||
[SupportedLanguages.Ruby]: RUBY_QUERIES,
|
||||
[SupportedLanguages.Kotlin]: '', // Kotlin WASM parser not yet available for web
|
||||
[SupportedLanguages.Swift]: SWIFT_QUERIES,
|
||||
};
|
||||
|
||||
@@ -1,5 +1,8 @@
|
||||
import { SupportedLanguages } from '../../config/supported-languages';
|
||||
|
||||
/** Ruby extensionless filenames recognised as Ruby source */
|
||||
const RUBY_EXTENSIONLESS_FILES = new Set(['Rakefile', 'Gemfile', 'Guardfile', 'Vagrantfile', 'Brewfile']);
|
||||
|
||||
/**
|
||||
* Map file extension to SupportedLanguage enum
|
||||
*/
|
||||
@@ -31,6 +34,17 @@ export const getLanguageFromFilename = (filename: string): SupportedLanguages |
|
||||
filename.endsWith('.php5') || filename.endsWith('.php8')) {
|
||||
return SupportedLanguages.PHP;
|
||||
}
|
||||
// Ruby (extensions)
|
||||
if (filename.endsWith('.rb') || filename.endsWith('.rake') || filename.endsWith('.gemspec')) {
|
||||
return SupportedLanguages.Ruby;
|
||||
}
|
||||
// Ruby (extensionless files)
|
||||
const basename = filename.split('/').pop() || filename;
|
||||
if (RUBY_EXTENSIONLESS_FILES.has(basename)) {
|
||||
return SupportedLanguages.Ruby;
|
||||
}
|
||||
// Swift
|
||||
if (filename.endsWith('.swift')) return SupportedLanguages.Swift;
|
||||
return null;
|
||||
};
|
||||
|
||||
|
||||
@@ -1,527 +0,0 @@
|
||||
/**
|
||||
* KuzuDB Adapter
|
||||
*
|
||||
* Manages the KuzuDB WASM instance for client-side graph database operations.
|
||||
* Uses the "Snapshot / Bulk Load" pattern with COPY FROM for performance.
|
||||
*
|
||||
* Multi-table schema: separate tables for File, Function, Class, etc.
|
||||
*/
|
||||
|
||||
import { KnowledgeGraph } from '../graph/types';
|
||||
import {
|
||||
NODE_TABLES,
|
||||
REL_TABLE_NAME,
|
||||
SCHEMA_QUERIES,
|
||||
EMBEDDING_TABLE_NAME,
|
||||
NodeTableName,
|
||||
} from './schema';
|
||||
import { generateAllCSVs } from './csv-generator';
|
||||
|
||||
// Holds the reference to the dynamically loaded module
|
||||
let kuzu: any = null;
|
||||
let db: any = null;
|
||||
let conn: any = null;
|
||||
|
||||
/**
|
||||
* Initialize KuzuDB WASM module and create in-memory database
|
||||
*/
|
||||
export const initKuzu = async () => {
|
||||
if (conn) return { db, conn, kuzu };
|
||||
|
||||
try {
|
||||
if (import.meta.env.DEV) console.log('🚀 Initializing KuzuDB...');
|
||||
|
||||
// 1. Dynamic Import (Fixes the "not a function" bundler issue)
|
||||
const kuzuModule = await import('kuzu-wasm');
|
||||
|
||||
// 2. Handle Vite/Webpack "default" wrapping
|
||||
kuzu = kuzuModule.default || kuzuModule;
|
||||
|
||||
// 3. Initialize WASM
|
||||
await kuzu.init();
|
||||
|
||||
// 4. Create Database with 512MB buffer pool
|
||||
const BUFFER_POOL_SIZE = 512 * 1024 * 1024; // 512MB
|
||||
db = new kuzu.Database(':memory:', BUFFER_POOL_SIZE);
|
||||
conn = new kuzu.Connection(db);
|
||||
|
||||
if (import.meta.env.DEV) console.log('✅ KuzuDB WASM Initialized');
|
||||
|
||||
// 5. Initialize Schema (all node tables, then rel tables, then embedding table)
|
||||
for (const schemaQuery of SCHEMA_QUERIES) {
|
||||
try {
|
||||
await conn.query(schemaQuery);
|
||||
} catch (e) {
|
||||
// Schema might already exist, skip
|
||||
if (import.meta.env.DEV) {
|
||||
console.warn('Schema creation skipped (may already exist):', e);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (import.meta.env.DEV) console.log('✅ KuzuDB Multi-Table Schema Created');
|
||||
|
||||
return { db, conn, kuzu };
|
||||
} catch (error) {
|
||||
if (import.meta.env.DEV) console.error('❌ KuzuDB Initialization Failed:', error);
|
||||
throw error;
|
||||
}
|
||||
};
|
||||
|
||||
/**
|
||||
* Load a KnowledgeGraph into KuzuDB using COPY FROM (bulk load)
|
||||
* Uses batched CSV writes and COPY statements for optimal performance
|
||||
*/
|
||||
export const loadGraphToKuzu = async (
|
||||
graph: KnowledgeGraph,
|
||||
fileContents: Map<string, string>
|
||||
) => {
|
||||
const { conn, kuzu } = await initKuzu();
|
||||
|
||||
try {
|
||||
if (import.meta.env.DEV) console.log(`KuzuDB: Generating CSVs for ${graph.nodeCount} nodes...`);
|
||||
|
||||
// 1. Generate all CSVs (per-table)
|
||||
const csvData = generateAllCSVs(graph, fileContents);
|
||||
|
||||
const fs = kuzu.FS;
|
||||
|
||||
// 2. Write all node CSVs to virtual filesystem
|
||||
const nodeFiles: Array<{ table: NodeTableName; path: string }> = [];
|
||||
for (const [tableName, csv] of csvData.nodes.entries()) {
|
||||
// Skip empty CSVs (only header row)
|
||||
if (csv.split('\n').length <= 1) continue;
|
||||
|
||||
const path = `/${tableName.toLowerCase()}.csv`;
|
||||
try { await fs.unlink(path); } catch {}
|
||||
await fs.writeFile(path, csv);
|
||||
nodeFiles.push({ table: tableName, path });
|
||||
}
|
||||
|
||||
// 3. Parse relation CSV and prepare for INSERT (COPY FROM doesn't work with multi-pair tables)
|
||||
const relLines = csvData.relCSV.split('\n').slice(1).filter(line => line.trim());
|
||||
const relCount = relLines.length;
|
||||
|
||||
if (import.meta.env.DEV) {
|
||||
console.log(`KuzuDB: Wrote ${nodeFiles.length} node CSVs, ${relCount} relations to insert`);
|
||||
}
|
||||
|
||||
// 4. COPY all node tables (must complete before rels due to FK constraints)
|
||||
for (const { table, path } of nodeFiles) {
|
||||
const copyQuery = getCopyQuery(table, path);
|
||||
await conn.query(copyQuery);
|
||||
}
|
||||
|
||||
// 5. INSERT relations one by one (COPY doesn't work with multi-pair REL tables)
|
||||
// Build a set of valid table names for fast lookup
|
||||
const validTables = new Set<string>(NODE_TABLES as readonly string[]);
|
||||
|
||||
const getNodeLabel = (nodeId: string): string => {
|
||||
if (nodeId.startsWith('comm_')) return 'Community';
|
||||
if (nodeId.startsWith('proc_')) return 'Process';
|
||||
return nodeId.split(':')[0];
|
||||
};
|
||||
|
||||
// All multi-language tables are created with backticks - must always reference them with backticks
|
||||
const escapeLabel = (label: string): string => {
|
||||
return BACKTICK_TABLES.has(label) ? `\`${label}\`` : label;
|
||||
};
|
||||
|
||||
let insertedRels = 0;
|
||||
let skippedRels = 0;
|
||||
const skippedRelStats = new Map<string, number>();
|
||||
for (const line of relLines) {
|
||||
try {
|
||||
// Format: "from","to","type",confidence,"reason",step
|
||||
const match = line.match(/"([^"]*)","([^"]*)","([^"]*)",([0-9.]+),"([^"]*)",([0-9-]+)/);
|
||||
if (!match) continue;
|
||||
|
||||
const [, fromId, toId, relType, confidenceStr, reason, stepStr] = match;
|
||||
|
||||
const fromLabel = getNodeLabel(fromId);
|
||||
const toLabel = getNodeLabel(toId);
|
||||
|
||||
// Skip relationships where either node's label doesn't have a table in KuzuDB
|
||||
// Querying a non-existent table causes a fatal native crash
|
||||
if (!validTables.has(fromLabel) || !validTables.has(toLabel)) {
|
||||
skippedRels++;
|
||||
continue;
|
||||
}
|
||||
|
||||
const confidence = parseFloat(confidenceStr) || 1.0;
|
||||
const step = parseInt(stepStr) || 0;
|
||||
|
||||
const insertQuery = `
|
||||
MATCH (a:${escapeLabel(fromLabel)} {id: '${fromId.replace(/'/g, "''")}'}),
|
||||
(b:${escapeLabel(toLabel)} {id: '${toId.replace(/'/g, "''")}'})
|
||||
CREATE (a)-[:${REL_TABLE_NAME} {type: '${relType}', confidence: ${confidence}, reason: '${reason.replace(/'/g, "''")}', step: ${step}}]->(b)
|
||||
`;
|
||||
await conn.query(insertQuery);
|
||||
insertedRels++;
|
||||
} catch (err) {
|
||||
skippedRels++;
|
||||
const match = line.match(/"([^"]*)","([^"]*)","([^"]*)",([0-9.]+),"([^"]*)"/);
|
||||
if (match) {
|
||||
const [, fromId, toId, relType] = match;
|
||||
const fromLabel = getNodeLabel(fromId);
|
||||
const toLabel = getNodeLabel(toId);
|
||||
const key = `${relType}:${fromLabel}->` + toLabel;
|
||||
skippedRelStats.set(key, (skippedRelStats.get(key) || 0) + 1);
|
||||
|
||||
if (import.meta.env.DEV) {
|
||||
console.warn(`⚠️ Skipped: ${key} | "${fromId}" → "${toId}" | ${err instanceof Error ? err.message : String(err)}`);
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (import.meta.env.DEV) {
|
||||
console.log(`KuzuDB: Inserted ${insertedRels}/${relCount} relations`);
|
||||
if (skippedRels > 0) {
|
||||
const topSkipped = Array.from(skippedRelStats.entries())
|
||||
.sort((a, b) => b[1] - a[1])
|
||||
.slice(0, 10);
|
||||
console.warn(`KuzuDB: Skipped ${skippedRels}/${relCount} relations (top by kind/pair):`, topSkipped);
|
||||
}
|
||||
}
|
||||
|
||||
// 6. Verify results
|
||||
let totalNodes = 0;
|
||||
for (const tableName of NODE_TABLES) {
|
||||
try {
|
||||
const countRes = await conn.query(`MATCH (n:${tableName}) RETURN count(n) AS cnt`);
|
||||
const countRow = await countRes.getNext();
|
||||
const count = countRow ? (countRow.cnt ?? countRow[0] ?? 0) : 0;
|
||||
totalNodes += Number(count);
|
||||
} catch {
|
||||
// Table might be empty, skip
|
||||
}
|
||||
}
|
||||
|
||||
if (import.meta.env.DEV) console.log(`✅ KuzuDB Bulk Load Complete. Total nodes: ${totalNodes}, edges: ${insertedRels}`);
|
||||
|
||||
// 7. Cleanup CSV files
|
||||
for (const { path } of nodeFiles) {
|
||||
try { await fs.unlink(path); } catch {}
|
||||
}
|
||||
|
||||
return { success: true, count: totalNodes };
|
||||
|
||||
} catch (error) {
|
||||
if (import.meta.env.DEV) console.error('❌ KuzuDB Bulk Load Failed:', error);
|
||||
return { success: false, count: 0 };
|
||||
}
|
||||
};
|
||||
|
||||
// KuzuDB default ESCAPE is '\' (backslash), but our CSV uses RFC 4180 escaping ("" for literal quotes).
|
||||
// Source code content is full of backslashes which confuse the auto-detection.
|
||||
// We MUST explicitly set ESCAPE='"' and disable auto_detect.
|
||||
const COPY_CSV_OPTS = `(HEADER=true, ESCAPE='"', DELIM=',', QUOTE='"', PARALLEL=false, auto_detect=false)`;
|
||||
|
||||
// Multi-language table names created with backticks in CODE_ELEMENT_BASE
|
||||
const BACKTICK_TABLES = new Set([
|
||||
'Struct', 'Enum', 'Macro', 'Typedef', 'Union', 'Namespace', 'Trait', 'Impl',
|
||||
'TypeAlias', 'Const', 'Static', 'Property', 'Record', 'Delegate', 'Annotation',
|
||||
'Constructor', 'Template', 'Module',
|
||||
]);
|
||||
|
||||
const escapeTableName = (table: string): string => {
|
||||
return BACKTICK_TABLES.has(table) ? `\`${table}\`` : table;
|
||||
};
|
||||
|
||||
/**
|
||||
* Get the COPY query for a node table with correct column mapping
|
||||
*/
|
||||
const getCopyQuery = (table: NodeTableName, path: string): string => {
|
||||
const t = escapeTableName(table);
|
||||
if (table === 'File') {
|
||||
return `COPY ${t}(id, name, filePath, content) FROM "${path}" ${COPY_CSV_OPTS}`;
|
||||
}
|
||||
if (table === 'Folder') {
|
||||
return `COPY ${t}(id, name, filePath) FROM "${path}" ${COPY_CSV_OPTS}`;
|
||||
}
|
||||
if (table === 'Community') {
|
||||
return `COPY ${t}(id, label, heuristicLabel, keywords, description, enrichedBy, cohesion, symbolCount) FROM "${path}" ${COPY_CSV_OPTS}`;
|
||||
}
|
||||
if (table === 'Process') {
|
||||
return `COPY ${t}(id, label, heuristicLabel, processType, stepCount, communities, entryPointId, terminalId) FROM "${path}" ${COPY_CSV_OPTS}`;
|
||||
}
|
||||
// Code element tables (Function, Class, Interface, Method, CodeElement, and multi-language)
|
||||
return `COPY ${t}(id, name, filePath, startLine, endLine, isExported, content) FROM "${path}" ${COPY_CSV_OPTS}`;
|
||||
};
|
||||
|
||||
/**
|
||||
* Execute a Cypher query against the database
|
||||
* Returns results as named objects (not tuples) for better usability
|
||||
*/
|
||||
export const executeQuery = async (cypher: string): Promise<any[]> => {
|
||||
if (!conn) {
|
||||
await initKuzu();
|
||||
}
|
||||
|
||||
try {
|
||||
const result = await conn.query(cypher);
|
||||
|
||||
// Extract column names from RETURN clause
|
||||
const returnMatch = cypher.match(/RETURN\s+(.+?)(?:\s+ORDER|\s+LIMIT|\s+SKIP|\s*$)/is);
|
||||
let columnNames: string[] = [];
|
||||
if (returnMatch) {
|
||||
// Parse RETURN clause to get column names/aliases
|
||||
// Handles: "a.name, b.filePath AS path, count(x) AS cnt"
|
||||
const returnClause = returnMatch[1];
|
||||
columnNames = returnClause.split(',').map(col => {
|
||||
col = col.trim();
|
||||
// Check for AS alias
|
||||
const asMatch = col.match(/\s+AS\s+(\w+)\s*$/i);
|
||||
if (asMatch) return asMatch[1];
|
||||
// Check for property access like n.name
|
||||
const propMatch = col.match(/\.(\w+)\s*$/);
|
||||
if (propMatch) return propMatch[1];
|
||||
// Check for function call like count(x)
|
||||
const funcMatch = col.match(/^(\w+)\s*\(/);
|
||||
if (funcMatch) return funcMatch[1];
|
||||
// Just use as-is if simple identifier
|
||||
return col.replace(/[^a-zA-Z0-9_]/g, '_');
|
||||
});
|
||||
}
|
||||
|
||||
// Collect all rows
|
||||
const rows: any[] = [];
|
||||
while (await result.hasNext()) {
|
||||
const row = await result.getNext();
|
||||
|
||||
// Convert tuple to named object if we have column names and row is array
|
||||
if (Array.isArray(row) && columnNames.length === row.length) {
|
||||
const namedRow: Record<string, any> = {};
|
||||
for (let i = 0; i < row.length; i++) {
|
||||
namedRow[columnNames[i]] = row[i];
|
||||
}
|
||||
rows.push(namedRow);
|
||||
} else {
|
||||
// Already an object or column count doesn't match
|
||||
rows.push(row);
|
||||
}
|
||||
}
|
||||
|
||||
return rows;
|
||||
} catch (error) {
|
||||
if (import.meta.env.DEV) console.error('Query execution failed:', error);
|
||||
throw error;
|
||||
}
|
||||
};
|
||||
|
||||
/**
|
||||
* Get database statistics
|
||||
*/
|
||||
export const getKuzuStats = async (): Promise<{ nodes: number; edges: number }> => {
|
||||
if (!conn) {
|
||||
return { nodes: 0, edges: 0 };
|
||||
}
|
||||
|
||||
try {
|
||||
// Count nodes across all tables
|
||||
let totalNodes = 0;
|
||||
for (const tableName of NODE_TABLES) {
|
||||
try {
|
||||
const nodeResult = await conn.query(`MATCH (n:${tableName}) RETURN count(n) AS cnt`);
|
||||
const nodeRow = await nodeResult.getNext();
|
||||
totalNodes += Number(nodeRow?.cnt ?? nodeRow?.[0] ?? 0);
|
||||
} catch {
|
||||
// Table might not exist or be empty
|
||||
}
|
||||
}
|
||||
|
||||
// Count edges from single relation table
|
||||
let totalEdges = 0;
|
||||
try {
|
||||
const edgeResult = await conn.query(`MATCH ()-[r:${REL_TABLE_NAME}]->() RETURN count(r) AS cnt`);
|
||||
const edgeRow = await edgeResult.getNext();
|
||||
totalEdges = Number(edgeRow?.cnt ?? edgeRow?.[0] ?? 0);
|
||||
} catch {
|
||||
// Table might not exist or be empty
|
||||
}
|
||||
|
||||
return { nodes: totalNodes, edges: totalEdges };
|
||||
} catch (error) {
|
||||
if (import.meta.env.DEV) {
|
||||
console.warn('Failed to get Kuzu stats:', error);
|
||||
}
|
||||
return { nodes: 0, edges: 0 };
|
||||
}
|
||||
};
|
||||
|
||||
/**
|
||||
* Check if KuzuDB is initialized and has data
|
||||
*/
|
||||
export const isKuzuReady = (): boolean => {
|
||||
return conn !== null && db !== null;
|
||||
};
|
||||
|
||||
/**
|
||||
* Close the database connection (cleanup)
|
||||
*/
|
||||
export const closeKuzu = async (): Promise<void> => {
|
||||
if (conn) {
|
||||
try {
|
||||
await conn.close();
|
||||
} catch {}
|
||||
conn = null;
|
||||
}
|
||||
if (db) {
|
||||
try {
|
||||
await db.close();
|
||||
} catch {}
|
||||
db = null;
|
||||
}
|
||||
kuzu = null;
|
||||
};
|
||||
|
||||
/**
|
||||
* Execute a prepared statement with parameters
|
||||
* @param cypher - Cypher query with $param placeholders
|
||||
* @param params - Object mapping param names to values
|
||||
* @returns Query results
|
||||
*/
|
||||
export const executePrepared = async (
|
||||
cypher: string,
|
||||
params: Record<string, any>
|
||||
): Promise<any[]> => {
|
||||
if (!conn) {
|
||||
await initKuzu();
|
||||
}
|
||||
|
||||
try {
|
||||
const stmt = await conn.prepare(cypher);
|
||||
if (!stmt.isSuccess()) {
|
||||
const errMsg = await stmt.getErrorMessage();
|
||||
throw new Error(`Prepare failed: ${errMsg}`);
|
||||
}
|
||||
|
||||
const result = await conn.execute(stmt, params);
|
||||
|
||||
const rows: any[] = [];
|
||||
while (await result.hasNext()) {
|
||||
const row = await result.getNext();
|
||||
rows.push(row);
|
||||
}
|
||||
|
||||
await stmt.close();
|
||||
return rows;
|
||||
} catch (error) {
|
||||
if (import.meta.env.DEV) console.error('Prepared query failed:', error);
|
||||
throw error;
|
||||
}
|
||||
};
|
||||
|
||||
/**
|
||||
* Execute a prepared statement with multiple parameter sets in small sub-batches
|
||||
*/
|
||||
export const executeWithReusedStatement = async (
|
||||
cypher: string,
|
||||
paramsList: Array<Record<string, any>>
|
||||
): Promise<void> => {
|
||||
if (!conn) {
|
||||
await initKuzu();
|
||||
}
|
||||
|
||||
if (paramsList.length === 0) return;
|
||||
|
||||
const SUB_BATCH_SIZE = 4;
|
||||
|
||||
for (let i = 0; i < paramsList.length; i += SUB_BATCH_SIZE) {
|
||||
const subBatch = paramsList.slice(i, i + SUB_BATCH_SIZE);
|
||||
|
||||
const stmt = await conn.prepare(cypher);
|
||||
if (!stmt.isSuccess()) {
|
||||
const errMsg = await stmt.getErrorMessage();
|
||||
throw new Error(`Prepare failed: ${errMsg}`);
|
||||
}
|
||||
|
||||
try {
|
||||
for (const params of subBatch) {
|
||||
await conn.execute(stmt, params);
|
||||
}
|
||||
} finally {
|
||||
await stmt.close();
|
||||
}
|
||||
|
||||
if (i + SUB_BATCH_SIZE < paramsList.length) {
|
||||
await new Promise(r => setTimeout(r, 0));
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
/**
|
||||
* Test if array parameters work with prepared statements
|
||||
*/
|
||||
export const testArrayParams = async (): Promise<{ success: boolean; error?: string }> => {
|
||||
if (!conn) {
|
||||
await initKuzu();
|
||||
}
|
||||
|
||||
try {
|
||||
const testEmbedding = new Array(384).fill(0).map((_, i) => i / 384);
|
||||
|
||||
// Get any node ID to test with (try File first, then others)
|
||||
let testNodeId: string | null = null;
|
||||
for (const tableName of NODE_TABLES) {
|
||||
try {
|
||||
const nodeResult = await conn.query(`MATCH (n:${tableName}) RETURN n.id AS id LIMIT 1`);
|
||||
const nodeRow = await nodeResult.getNext();
|
||||
if (nodeRow) {
|
||||
testNodeId = nodeRow.id ?? nodeRow[0];
|
||||
break;
|
||||
}
|
||||
} catch {}
|
||||
}
|
||||
|
||||
if (!testNodeId) {
|
||||
return { success: false, error: 'No nodes found to test with' };
|
||||
}
|
||||
|
||||
if (import.meta.env.DEV) {
|
||||
console.log('🧪 Testing array params with node:', testNodeId);
|
||||
}
|
||||
|
||||
// First create an embedding entry
|
||||
const createQuery = `CREATE (e:${EMBEDDING_TABLE_NAME} {nodeId: $nodeId, embedding: $embedding})`;
|
||||
const stmt = await conn.prepare(createQuery);
|
||||
|
||||
if (!stmt.isSuccess()) {
|
||||
const errMsg = await stmt.getErrorMessage();
|
||||
return { success: false, error: `Prepare failed: ${errMsg}` };
|
||||
}
|
||||
|
||||
await conn.execute(stmt, {
|
||||
nodeId: testNodeId,
|
||||
embedding: testEmbedding,
|
||||
});
|
||||
|
||||
await stmt.close();
|
||||
|
||||
// Verify it was stored
|
||||
const verifyResult = await conn.query(
|
||||
`MATCH (e:${EMBEDDING_TABLE_NAME} {nodeId: '${testNodeId}'}) RETURN e.embedding AS emb`
|
||||
);
|
||||
const verifyRow = await verifyResult.getNext();
|
||||
const storedEmb = verifyRow?.emb ?? verifyRow?.[0];
|
||||
|
||||
if (storedEmb && Array.isArray(storedEmb) && storedEmb.length === 384) {
|
||||
if (import.meta.env.DEV) {
|
||||
console.log('✅ Array params WORK! Stored embedding length:', storedEmb.length);
|
||||
}
|
||||
return { success: true };
|
||||
} else {
|
||||
return {
|
||||
success: false,
|
||||
error: `Embedding not stored correctly. Got: ${typeof storedEmb}, length: ${storedEmb?.length}`
|
||||
};
|
||||
}
|
||||
} catch (error) {
|
||||
const errorMsg = error instanceof Error ? error.message : String(error);
|
||||
if (import.meta.env.DEV) {
|
||||
console.error('❌ Array params test failed:', errorMsg);
|
||||
}
|
||||
return { success: false, error: errorMsg };
|
||||
}
|
||||
};
|
||||
+45
-6
@@ -1,5 +1,5 @@
|
||||
/**
|
||||
* CSV Generator for KuzuDB Hybrid Schema
|
||||
* CSV Generator for LadybugDB Hybrid Schema
|
||||
*
|
||||
* Generates separate CSV files for each node table and one relation CSV.
|
||||
* This enables efficient bulk loading via COPY FROM for hybrid schema.
|
||||
@@ -18,10 +18,10 @@ import { NODE_TABLES, NodeTableName } from './schema';
|
||||
// ============================================================================
|
||||
|
||||
/**
|
||||
* Sanitize string to ensure valid UTF-8 and safe CSV content for KuzuDB
|
||||
* Sanitize string to ensure valid UTF-8 and safe CSV content for LadybugDB
|
||||
* Removes or replaces invalid characters that would break CSV parsing.
|
||||
*
|
||||
* Critical: KuzuDB's CSV parser can misinterpret \r\n inside quoted fields.
|
||||
* Critical: LadybugDB's CSV parser can misinterpret \r\n inside quoted fields.
|
||||
* We normalize all line endings to \n only.
|
||||
*/
|
||||
const sanitizeUTF8 = (str: string): string => {
|
||||
@@ -202,6 +202,35 @@ const generateCodeElementCSV = (
|
||||
return rows.join('\n');
|
||||
};
|
||||
|
||||
/**
|
||||
* Generate CSV for multi-language code element nodes (Struct, Enum, Macro, etc.)
|
||||
* These do NOT have isExported column.
|
||||
* Headers: id,name,filePath,startLine,endLine,content
|
||||
*/
|
||||
const generateMultiLangCSV = (
|
||||
nodes: GraphNode[],
|
||||
label: NodeLabel,
|
||||
fileContents: Map<string, string>
|
||||
): string => {
|
||||
const headers = ['id', 'name', 'filePath', 'startLine', 'endLine', 'content'];
|
||||
const rows: string[] = [headers.join(',')];
|
||||
|
||||
for (const node of nodes) {
|
||||
if (node.label !== label) continue;
|
||||
const content = extractContent(node, fileContents);
|
||||
rows.push([
|
||||
escapeCSVField(node.id),
|
||||
escapeCSVField(node.properties.name || ''),
|
||||
escapeCSVField(node.properties.filePath || ''),
|
||||
escapeCSVNumber(node.properties.startLine, -1),
|
||||
escapeCSVNumber(node.properties.endLine, -1),
|
||||
escapeCSVField(content),
|
||||
].join(','));
|
||||
}
|
||||
|
||||
return rows.join('\n');
|
||||
};
|
||||
|
||||
/**
|
||||
* Generate CSV for Community nodes (from Leiden algorithm)
|
||||
* Headers: id,label,heuristicLabel,keywords,description,enrichedBy,cohesion,symbolCount
|
||||
@@ -213,7 +242,7 @@ const generateCommunityCSV = (nodes: GraphNode[]): string => {
|
||||
for (const node of nodes) {
|
||||
if (node.label !== 'Community') continue;
|
||||
|
||||
// Handle keywords array - convert to KuzuDB array format
|
||||
// Handle keywords array - convert to LadybugDB array format
|
||||
const keywords = (node.properties as any).keywords || [];
|
||||
const keywordsStr = `[${keywords.map((k: string) => `'${k.replace(/'/g, "''")}'`).join(',')}]`;
|
||||
|
||||
@@ -221,7 +250,7 @@ const generateCommunityCSV = (nodes: GraphNode[]): string => {
|
||||
escapeCSVField(node.id),
|
||||
escapeCSVField(node.properties.name || ''), // label is stored in name
|
||||
escapeCSVField(node.properties.heuristicLabel || ''),
|
||||
keywordsStr, // Array format for KuzuDB
|
||||
escapeCSVField(keywordsStr), // Array format for LadybugDB, needs CSV escaping for commas
|
||||
escapeCSVField((node.properties as any).description || ''),
|
||||
escapeCSVField((node.properties as any).enrichedBy || 'heuristic'),
|
||||
escapeCSVNumber(node.properties.cohesion, 0),
|
||||
@@ -312,7 +341,17 @@ export const generateAllCSVs = (
|
||||
nodeCSVs.set('CodeElement', generateCodeElementCSV(nodes, 'CodeElement', fileContents));
|
||||
nodeCSVs.set('Community', generateCommunityCSV(nodes));
|
||||
nodeCSVs.set('Process', generateProcessCSV(nodes));
|
||||
|
||||
|
||||
// Generate CSVs for remaining multi-language tables (no isExported column)
|
||||
const handledTables = new Set<NodeTableName>([
|
||||
'File', 'Folder', 'Function', 'Class', 'Interface', 'Method', 'CodeElement', 'Community', 'Process',
|
||||
]);
|
||||
for (const table of NODE_TABLES) {
|
||||
if (!handledTables.has(table)) {
|
||||
nodeCSVs.set(table, generateMultiLangCSV(nodes, table as NodeLabel, fileContents));
|
||||
}
|
||||
}
|
||||
|
||||
// Generate single relation CSV
|
||||
const relCSV = generateRelationCSV(graph);
|
||||
|
||||
@@ -0,0 +1,662 @@
|
||||
/**
|
||||
* LadybugDB Adapter
|
||||
*
|
||||
* Manages the LadybugDB WASM instance for client-side graph database operations.
|
||||
* Uses the "Snapshot / Bulk Load" pattern with COPY FROM for performance.
|
||||
*
|
||||
* Multi-table schema: separate tables for File, Function, Class, etc.
|
||||
*/
|
||||
|
||||
import { KnowledgeGraph } from '../graph/types';
|
||||
import {
|
||||
NODE_TABLES,
|
||||
REL_TABLE_NAME,
|
||||
SCHEMA_QUERIES,
|
||||
EMBEDDING_TABLE_NAME,
|
||||
NodeTableName,
|
||||
} from './schema';
|
||||
import { generateAllCSVs } from './csv-generator';
|
||||
|
||||
// Holds the reference to the dynamically loaded module
|
||||
let lbug: any = null;
|
||||
let db: any = null;
|
||||
let conn: any = null;
|
||||
let initPromise: Promise<{ db: any; conn: any; lbug: any }> | null = null;
|
||||
|
||||
/**
|
||||
* Initialize LadybugDB WASM module and create in-memory database
|
||||
*/
|
||||
export const initLbug = async () => {
|
||||
if (initPromise) return initPromise;
|
||||
initPromise = (async () => {
|
||||
try {
|
||||
if (import.meta.env.DEV) console.log('🚀 Initializing LadybugDB...');
|
||||
|
||||
// 1. Dynamic Import (Fixes the "not a function" bundler issue)
|
||||
const lbugModule = await import('@ladybugdb/wasm-core');
|
||||
|
||||
// 2. Handle Vite/Webpack "default" wrapping
|
||||
lbug = lbugModule.default || lbugModule;
|
||||
|
||||
// 3. Initialize WASM
|
||||
await lbug.init();
|
||||
|
||||
// 4. Create Database with 512MB buffer manager
|
||||
const BUFFER_POOL_SIZE = 512 * 1024 * 1024; // 512MB
|
||||
db = new lbug.Database(':memory:', BUFFER_POOL_SIZE);
|
||||
conn = new lbug.Connection(db);
|
||||
|
||||
if (import.meta.env.DEV) console.log('✅ LadybugDB WASM Initialized');
|
||||
|
||||
// 5. Initialize Schema (all node tables, then rel tables, then embedding table)
|
||||
for (let i = 0; i < SCHEMA_QUERIES.length; i++) {
|
||||
try {
|
||||
await conn.query(SCHEMA_QUERIES[i]);
|
||||
} catch (e) {
|
||||
// Schema might already exist, skip
|
||||
if (import.meta.env.DEV) {
|
||||
console.warn(`Schema query ${i + 1}/${SCHEMA_QUERIES.length} skipped (may already exist):`, e);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
if (import.meta.env.DEV) console.log('✅ LadybugDB Multi-Table Schema Created');
|
||||
|
||||
return { db, conn, lbug };
|
||||
} catch (error) {
|
||||
if (import.meta.env.DEV) console.error('❌ LadybugDB Initialization Failed:', error);
|
||||
throw error;
|
||||
}
|
||||
})();
|
||||
try {
|
||||
return await initPromise;
|
||||
} catch (error) {
|
||||
initPromise = null; // Reset on failure so retry is possible
|
||||
throw error;
|
||||
}
|
||||
};
|
||||
|
||||
/**
|
||||
* Load a KnowledgeGraph into LadybugDB using COPY FROM (bulk load)
|
||||
* Uses batched CSV writes and COPY statements for optimal performance
|
||||
*/
|
||||
const isTestEnv = () => {
|
||||
// Browser-friendly check: Vite only exposes VITE_* vars at runtime; fall back to a window flag if injected by tests.
|
||||
if (typeof import.meta !== 'undefined' && typeof import.meta.env !== 'undefined') {
|
||||
if (import.meta.env.VITE_PLAYWRIGHT_TEST || import.meta.env.MODE === 'test') return true;
|
||||
}
|
||||
if (typeof window !== 'undefined' && (window as unknown as { __PLAYWRIGHT_TEST__?: boolean }).__PLAYWRIGHT_TEST__) {
|
||||
return true;
|
||||
}
|
||||
if (typeof navigator !== 'undefined' && navigator.webdriver) {
|
||||
return true;
|
||||
}
|
||||
return typeof process !== 'undefined' && (process.env.PLAYWRIGHT_TEST || process.env.NODE_ENV === 'test');
|
||||
};
|
||||
|
||||
export const loadGraphToLbug = async (
|
||||
graph: KnowledgeGraph,
|
||||
fileContents: Map<string, string>
|
||||
) => {
|
||||
// In headless Playwright, skip heavy bulk load to avoid hangs; UI still functions with empty DB.
|
||||
if (isTestEnv()) {
|
||||
if (import.meta.env.DEV) console.log('🧪 Skipping LadybugDB bulk load in test mode');
|
||||
await initLbug(); // ensure module initialized for downstream calls
|
||||
return { success: true, count: 0 };
|
||||
}
|
||||
const { lbug: lbugModule } = await initLbug();
|
||||
|
||||
// Close previous connection/database to avoid leaking WASM resources across repo switches
|
||||
if (conn) {
|
||||
try { await conn.close(); } catch {}
|
||||
conn = null;
|
||||
}
|
||||
if (db) {
|
||||
try { await db.close(); } catch {}
|
||||
db = null;
|
||||
}
|
||||
|
||||
// Recreate a fresh in-memory DB each load to avoid cleanup/quoting issues with reserved names
|
||||
const BUFFER_POOL_SIZE = 512 * 1024 * 1024; // 512MB (mirror init)
|
||||
db = new lbugModule.Database(':memory:', BUFFER_POOL_SIZE);
|
||||
conn = new lbugModule.Connection(db);
|
||||
|
||||
// Update initPromise so subsequent initLbug() calls return the fresh db/conn
|
||||
initPromise = Promise.resolve({ db, conn, lbug: lbugModule });
|
||||
|
||||
// Re-run schema creation
|
||||
for (let i = 0; i < SCHEMA_QUERIES.length; i++) {
|
||||
try {
|
||||
await conn.query(SCHEMA_QUERIES[i]);
|
||||
} catch (e) {
|
||||
if (import.meta.env.DEV) {
|
||||
console.warn(`Schema query ${i + 1}/${SCHEMA_QUERIES.length} skipped (may already exist):`, e);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
try {
|
||||
if (import.meta.env.DEV) console.log(`LadybugDB: Generating CSVs for ${graph.nodeCount} nodes...`);
|
||||
|
||||
// 1. Generate all CSVs (per-table)
|
||||
const csvData = generateAllCSVs(graph, fileContents);
|
||||
|
||||
const fs = lbug.FS;
|
||||
|
||||
// 2. Write all node CSVs to virtual filesystem
|
||||
const nodeFiles: Array<{ table: NodeTableName; path: string }> = [];
|
||||
for (const [tableName, csv] of csvData.nodes.entries()) {
|
||||
// Skip empty CSVs (only header row)
|
||||
if (csv.split('\n').length <= 1) continue;
|
||||
|
||||
const path = `/${tableName.toLowerCase()}.csv`;
|
||||
try { await fs.unlink(path); } catch {}
|
||||
await fs.writeFile(path, csv);
|
||||
nodeFiles.push({ table: tableName, path });
|
||||
}
|
||||
|
||||
// 3. Parse relation CSV and prepare for INSERT (COPY FROM doesn't work with multi-pair tables)
|
||||
const relLines = csvData.relCSV.split('\n').slice(1).filter(line => line.trim());
|
||||
const relCount = relLines.length;
|
||||
|
||||
if (import.meta.env.DEV) {
|
||||
console.log(`LadybugDB: Wrote ${nodeFiles.length} node CSVs, ${relCount} relations to insert`);
|
||||
}
|
||||
|
||||
// 4. COPY all node tables (must complete before rels due to FK constraints)
|
||||
for (const { table, path } of nodeFiles) {
|
||||
const copyQuery = getCopyQuery(table, path);
|
||||
await conn.query(copyQuery);
|
||||
}
|
||||
|
||||
// 5. INSERT relations one by one (COPY doesn't work with multi-pair REL tables)
|
||||
// Build a set of valid table names for fast lookup
|
||||
const validTables = new Set<string>(NODE_TABLES as readonly string[]);
|
||||
|
||||
const getNodeLabel = (nodeId: string): string => {
|
||||
if (nodeId.startsWith('comm_')) return 'Community';
|
||||
if (nodeId.startsWith('proc_')) return 'Process';
|
||||
return nodeId.split(':')[0];
|
||||
};
|
||||
|
||||
// All multi-language tables are created with backticks - must always reference them with backticks
|
||||
const escapeLabel = (label: string): string => {
|
||||
return BACKTICK_TABLES.has(label) ? `\`${label}\`` : label;
|
||||
};
|
||||
|
||||
let insertedRels = 0;
|
||||
let skippedRels = 0;
|
||||
const skippedRelStats = new Map<string, number>();
|
||||
|
||||
// Group relations by (fromLabel, toLabel) pair for prepared statement reuse
|
||||
const relsByLabelPair = new Map<string, Array<{ fromId: string; toId: string; relType: string; confidence: number; reason: string; step: number }>>();
|
||||
// RFC 4180 regex: handles doubled quotes ("") inside quoted fields
|
||||
const csvRegex = /"((?:[^"]|"")*)","((?:[^"]|"")*)","((?:[^"]|"")*)",([0-9.]+),"((?:[^"]|"")*)",([0-9-]+)/;
|
||||
|
||||
for (const line of relLines) {
|
||||
const match = line.match(csvRegex);
|
||||
if (!match) continue;
|
||||
|
||||
// Unescape RFC 4180 doubled quotes
|
||||
const fromId = match[1].replace(/""/g, '"');
|
||||
const toId = match[2].replace(/""/g, '"');
|
||||
const relType = match[3].replace(/""/g, '"');
|
||||
const reason = match[5].replace(/""/g, '"');
|
||||
|
||||
const fromLabel = getNodeLabel(fromId);
|
||||
const toLabel = getNodeLabel(toId);
|
||||
|
||||
// Skip relationships where either node's label doesn't have a table in LadybugDB
|
||||
// Querying a non-existent table causes a fatal native crash
|
||||
if (!validTables.has(fromLabel) || !validTables.has(toLabel)) {
|
||||
skippedRels++;
|
||||
continue;
|
||||
}
|
||||
|
||||
const key = `${fromLabel}:${toLabel}`;
|
||||
if (!relsByLabelPair.has(key)) relsByLabelPair.set(key, []);
|
||||
relsByLabelPair.get(key)!.push({
|
||||
fromId,
|
||||
toId,
|
||||
relType,
|
||||
confidence: parseFloat(match[4]) || 1.0,
|
||||
reason,
|
||||
step: parseInt(match[6]) || 0,
|
||||
});
|
||||
}
|
||||
|
||||
// Execute batched prepared statements per label pair
|
||||
// Prepare once per (fromLabel, toLabel) pair and reuse across all rows
|
||||
for (const [key, rels] of relsByLabelPair) {
|
||||
const [fromLabel, toLabel] = key.split(':');
|
||||
const cypher = `
|
||||
MATCH (a:${escapeLabel(fromLabel)} {id: $fromId}),
|
||||
(b:${escapeLabel(toLabel)} {id: $toId})
|
||||
CREATE (a)-[:${REL_TABLE_NAME} {type: $relType, confidence: $confidence, reason: $reason, step: $step}]->(b)
|
||||
`;
|
||||
|
||||
const stmt = await conn.prepare(cypher);
|
||||
if (!stmt.isSuccess()) {
|
||||
const errMsg = await stmt.getErrorMessage();
|
||||
if (import.meta.env.DEV) console.warn(`Prepare failed for ${key}: ${errMsg}`);
|
||||
skippedRels += rels.length;
|
||||
await stmt.close();
|
||||
continue;
|
||||
}
|
||||
|
||||
try {
|
||||
for (let i = 0; i < rels.length; i++) {
|
||||
try {
|
||||
await conn.execute(stmt, rels[i]);
|
||||
insertedRels++;
|
||||
} catch (err) {
|
||||
skippedRels++;
|
||||
const r = rels[i];
|
||||
const statKey = `${r.relType}:${fromLabel}->${toLabel}`;
|
||||
skippedRelStats.set(statKey, (skippedRelStats.get(statKey) || 0) + 1);
|
||||
if (import.meta.env.DEV) {
|
||||
console.warn(`⚠️ Skipped: ${statKey} | "${r.fromId}" → "${r.toId}" | ${err instanceof Error ? err.message : String(err)}`);
|
||||
}
|
||||
}
|
||||
|
||||
// Yield to event loop every 500 relations
|
||||
if (i > 0 && i % 500 === 0) {
|
||||
await new Promise(r => setTimeout(r, 0));
|
||||
}
|
||||
}
|
||||
} finally {
|
||||
await stmt.close();
|
||||
}
|
||||
}
|
||||
|
||||
if (import.meta.env.DEV) {
|
||||
console.log(`LadybugDB: Inserted ${insertedRels}/${relCount} relations`);
|
||||
if (skippedRels > 0) {
|
||||
const topSkipped = Array.from(skippedRelStats.entries())
|
||||
.sort((a, b) => b[1] - a[1])
|
||||
.slice(0, 10);
|
||||
console.warn(`LadybugDB: Skipped ${skippedRels}/${relCount} relations (top by kind/pair):`, topSkipped);
|
||||
}
|
||||
}
|
||||
|
||||
// 6. Verify results
|
||||
let totalNodes = 0;
|
||||
for (const tableName of NODE_TABLES) {
|
||||
try {
|
||||
const countRes = await conn.query(`MATCH (n:${escapeTableName(tableName)}) RETURN count(n) AS cnt`);
|
||||
const countRows = await countRes.getAllRows();
|
||||
const countRow = countRows[0];
|
||||
const count = countRow ? (countRow.cnt ?? countRow[0] ?? 0) : 0;
|
||||
totalNodes += Number(count);
|
||||
} catch {
|
||||
// Table might be empty, skip
|
||||
}
|
||||
}
|
||||
|
||||
if (import.meta.env.DEV) console.log(`✅ LadybugDB Bulk Load Complete. Total nodes: ${totalNodes}, edges: ${insertedRels}`);
|
||||
|
||||
// 7. Cleanup CSV files
|
||||
for (const { path } of nodeFiles) {
|
||||
try { await fs.unlink(path); } catch {}
|
||||
}
|
||||
|
||||
return { success: true, count: totalNodes };
|
||||
|
||||
} catch (error) {
|
||||
if (import.meta.env.DEV) console.error('❌ LadybugDB Bulk Load Failed:', error);
|
||||
return { success: false, count: 0 };
|
||||
}
|
||||
};
|
||||
|
||||
// LadybugDB default ESCAPE is '\' (backslash), but our CSV uses RFC 4180 escaping ("" for literal quotes).
|
||||
// Source code content is full of backslashes which confuse the auto-detection.
|
||||
// We MUST explicitly set ESCAPE='"' and disable auto_detect.
|
||||
const COPY_CSV_OPTS = `(HEADER=true, ESCAPE='"', DELIM=',', QUOTE='"', PARALLEL=false, auto_detect=false)`;
|
||||
|
||||
// Multi-language table names created with backticks in CODE_ELEMENT_BASE
|
||||
const BACKTICK_TABLES = new Set([
|
||||
'Struct', 'Enum', 'Macro', 'Typedef', 'Union', 'Namespace', 'Trait', 'Impl',
|
||||
'TypeAlias', 'Const', 'Static', 'Property', 'Record', 'Delegate', 'Annotation',
|
||||
'Constructor', 'Template', 'Module',
|
||||
// Reserved/ambiguous identifiers that need quoting
|
||||
'File',
|
||||
]);
|
||||
|
||||
const escapeTableName = (table: string): string => {
|
||||
return BACKTICK_TABLES.has(table) ? `\`${table}\`` : table;
|
||||
};
|
||||
|
||||
// LadybugDB DELETE needs standard quoted identifiers for reserved names (e.g., File)
|
||||
const escapeTableForDelete = (table: string): string => {
|
||||
if (table === 'File') return `"${table}"`;
|
||||
return escapeTableName(table);
|
||||
};
|
||||
|
||||
/** Tables with isExported column (TypeScript/JS-native types) */
|
||||
const TABLES_WITH_EXPORTED = new Set<string>(['Function', 'Class', 'Interface', 'Method', 'CodeElement']);
|
||||
|
||||
/**
|
||||
* Get the COPY query for a node table with correct column mapping
|
||||
*/
|
||||
const getCopyQuery = (table: NodeTableName, path: string): string => {
|
||||
const t = escapeTableName(table);
|
||||
if (table === 'File') {
|
||||
return `COPY ${t}(id, name, filePath, content) FROM "${path}" ${COPY_CSV_OPTS}`;
|
||||
}
|
||||
if (table === 'Folder') {
|
||||
return `COPY ${t}(id, name, filePath) FROM "${path}" ${COPY_CSV_OPTS}`;
|
||||
}
|
||||
if (table === 'Community') {
|
||||
return `COPY ${t}(id, label, heuristicLabel, keywords, description, enrichedBy, cohesion, symbolCount) FROM "${path}" ${COPY_CSV_OPTS}`;
|
||||
}
|
||||
if (table === 'Process') {
|
||||
return `COPY ${t}(id, label, heuristicLabel, processType, stepCount, communities, entryPointId, terminalId) FROM "${path}" ${COPY_CSV_OPTS}`;
|
||||
}
|
||||
// TypeScript/JS code element tables have isExported; multi-language tables do not
|
||||
if (TABLES_WITH_EXPORTED.has(table)) {
|
||||
return `COPY ${t}(id, name, filePath, startLine, endLine, isExported, content) FROM "${path}" ${COPY_CSV_OPTS}`;
|
||||
}
|
||||
// Multi-language tables (Struct, Impl, Trait, Macro, etc.)
|
||||
return `COPY ${t}(id, name, filePath, startLine, endLine, content) FROM "${path}" ${COPY_CSV_OPTS}`;
|
||||
};
|
||||
|
||||
/**
|
||||
* Execute a Cypher query against the database
|
||||
* Returns results as named objects (not tuples) for better usability
|
||||
*/
|
||||
export const executeQuery = async (cypher: string, readOnly = true): Promise<any[]> => {
|
||||
if (!conn) {
|
||||
await initLbug();
|
||||
}
|
||||
|
||||
if (readOnly) {
|
||||
// Strip quoted strings before checking for write keywords, so that
|
||||
// queries like WHERE n.name CONTAINS "delete" are not blocked.
|
||||
const stripped = cypher.replace(/'[^']*'|"[^"]*"/g, '').toUpperCase();
|
||||
if (/\b(CREATE|DELETE|SET|MERGE|REMOVE|DROP|DETACH)\b/.test(stripped)) {
|
||||
throw new Error('Read-only query attempted a write operation');
|
||||
}
|
||||
}
|
||||
|
||||
try {
|
||||
const result = await conn.query(cypher);
|
||||
|
||||
// Extract column names from RETURN clause
|
||||
const returnMatch = cypher.match(/RETURN\s+(.+?)(?:\s+ORDER|\s+LIMIT|\s+SKIP|\s*$)/is);
|
||||
let columnNames: string[] = [];
|
||||
if (returnMatch) {
|
||||
// Parse RETURN clause to get column names/aliases
|
||||
// Handles: "a.name, b.filePath AS path, count(x) AS cnt"
|
||||
const returnClause = returnMatch[1];
|
||||
columnNames = returnClause.split(',').map(col => {
|
||||
col = col.trim();
|
||||
// Check for AS alias
|
||||
const asMatch = col.match(/\s+AS\s+(\w+)\s*$/i);
|
||||
if (asMatch) return asMatch[1];
|
||||
// Check for property access like n.name
|
||||
const propMatch = col.match(/\.(\w+)\s*$/);
|
||||
if (propMatch) return propMatch[1];
|
||||
// Check for function call like count(x)
|
||||
const funcMatch = col.match(/^(\w+)\s*\(/);
|
||||
if (funcMatch) return funcMatch[1];
|
||||
// Just use as-is if simple identifier
|
||||
return col.replace(/[^a-zA-Z0-9_]/g, '_');
|
||||
});
|
||||
}
|
||||
|
||||
// Collect all rows
|
||||
const allRows = await result.getAllRows();
|
||||
const rows: any[] = [];
|
||||
for (const row of allRows) {
|
||||
// Convert tuple to named object if we have column names and row is array
|
||||
if (Array.isArray(row) && columnNames.length === row.length) {
|
||||
const namedRow: Record<string, any> = {};
|
||||
for (let i = 0; i < row.length; i++) {
|
||||
namedRow[columnNames[i]] = row[i];
|
||||
}
|
||||
rows.push(namedRow);
|
||||
} else {
|
||||
// Already an object or column count doesn't match
|
||||
rows.push(row);
|
||||
}
|
||||
}
|
||||
|
||||
return rows;
|
||||
} catch (error) {
|
||||
if (import.meta.env.DEV) console.error('Query execution failed:', error);
|
||||
throw error;
|
||||
}
|
||||
};
|
||||
|
||||
/**
|
||||
* Get database statistics
|
||||
*/
|
||||
export const getLbugStats = async (): Promise<{ nodes: number; edges: number }> => {
|
||||
if (!conn) {
|
||||
return { nodes: 0, edges: 0 };
|
||||
}
|
||||
|
||||
try {
|
||||
// Count nodes across all tables
|
||||
let totalNodes = 0;
|
||||
for (const tableName of NODE_TABLES) {
|
||||
try {
|
||||
const nodeResult = await conn.query(`MATCH (n:${escapeTableName(tableName)}) RETURN count(n) AS cnt`);
|
||||
const nodeRows = await nodeResult.getAllRows();
|
||||
const nodeRow = nodeRows[0];
|
||||
totalNodes += Number(nodeRow?.cnt ?? nodeRow?.[0] ?? 0);
|
||||
} catch {
|
||||
// Table might not exist or be empty
|
||||
}
|
||||
}
|
||||
|
||||
// Count edges from single relation table
|
||||
let totalEdges = 0;
|
||||
try {
|
||||
const edgeResult = await conn.query(`MATCH ()-[r:${REL_TABLE_NAME}]->() RETURN count(r) AS cnt`);
|
||||
const edgeRows = await edgeResult.getAllRows();
|
||||
const edgeRow = edgeRows[0];
|
||||
totalEdges = Number(edgeRow?.cnt ?? edgeRow?.[0] ?? 0);
|
||||
} catch {
|
||||
// Table might not exist or be empty
|
||||
}
|
||||
|
||||
return { nodes: totalNodes, edges: totalEdges };
|
||||
} catch (error) {
|
||||
if (import.meta.env.DEV) {
|
||||
console.warn('Failed to get LadybugDB stats:', error);
|
||||
}
|
||||
return { nodes: 0, edges: 0 };
|
||||
}
|
||||
};
|
||||
|
||||
/**
|
||||
* Check if LadybugDB is initialized and has data
|
||||
*/
|
||||
export const isLbugReady = (): boolean => {
|
||||
return conn !== null && db !== null;
|
||||
};
|
||||
|
||||
/**
|
||||
* Close the database connection (cleanup)
|
||||
*/
|
||||
export const closeLbug = async (): Promise<void> => {
|
||||
if (conn) {
|
||||
try {
|
||||
await conn.close();
|
||||
} catch {}
|
||||
conn = null;
|
||||
}
|
||||
if (db) {
|
||||
try {
|
||||
await db.close();
|
||||
} catch {}
|
||||
db = null;
|
||||
}
|
||||
lbug = null;
|
||||
initPromise = null;
|
||||
};
|
||||
|
||||
/**
|
||||
* Execute a prepared statement with parameters
|
||||
* @param cypher - Cypher query with $param placeholders
|
||||
* @param params - Object mapping param names to values
|
||||
* @returns Query results
|
||||
*/
|
||||
export const executePrepared = async (
|
||||
cypher: string,
|
||||
params: Record<string, any>
|
||||
): Promise<any[]> => {
|
||||
if (!conn) {
|
||||
await initLbug();
|
||||
}
|
||||
|
||||
try {
|
||||
const stmt = await conn.prepare(cypher);
|
||||
try {
|
||||
if (!stmt.isSuccess()) {
|
||||
const errMsg = await stmt.getErrorMessage();
|
||||
throw new Error(`Prepare failed: ${errMsg}`);
|
||||
}
|
||||
|
||||
const result = await conn.execute(stmt, params);
|
||||
const rows = await result.getAllRows();
|
||||
return rows;
|
||||
} finally {
|
||||
await stmt.close();
|
||||
}
|
||||
} catch (error) {
|
||||
if (import.meta.env.DEV) console.error('Prepared query failed:', error);
|
||||
throw error;
|
||||
}
|
||||
};
|
||||
|
||||
/**
|
||||
* Execute a prepared statement with multiple parameter sets in small sub-batches
|
||||
*/
|
||||
export const executeWithReusedStatement = async (
|
||||
cypher: string,
|
||||
paramsList: Array<Record<string, any>>
|
||||
): Promise<void> => {
|
||||
if (!conn) {
|
||||
await initLbug();
|
||||
}
|
||||
|
||||
if (paramsList.length === 0) return;
|
||||
|
||||
const SUB_BATCH_SIZE = 4;
|
||||
|
||||
for (let i = 0; i < paramsList.length; i += SUB_BATCH_SIZE) {
|
||||
const subBatch = paramsList.slice(i, i + SUB_BATCH_SIZE);
|
||||
|
||||
const stmt = await conn.prepare(cypher);
|
||||
if (!stmt.isSuccess()) {
|
||||
const errMsg = await stmt.getErrorMessage();
|
||||
throw new Error(`Prepare failed: ${errMsg}`);
|
||||
}
|
||||
|
||||
try {
|
||||
for (const params of subBatch) {
|
||||
await conn.execute(stmt, params);
|
||||
}
|
||||
} finally {
|
||||
await stmt.close();
|
||||
}
|
||||
|
||||
if (i + SUB_BATCH_SIZE < paramsList.length) {
|
||||
await new Promise(r => setTimeout(r, 0));
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
/**
|
||||
* Test if array parameters work with prepared statements
|
||||
*/
|
||||
export const testArrayParams = async (): Promise<{ success: boolean; error?: string }> => {
|
||||
if (!conn) {
|
||||
await initLbug();
|
||||
}
|
||||
|
||||
try {
|
||||
const testEmbedding = new Array(384).fill(0).map((_, i) => i / 384);
|
||||
|
||||
// Get any node ID to test with (try File first, then others)
|
||||
let testNodeId: string | null = null;
|
||||
for (const tableName of NODE_TABLES) {
|
||||
try {
|
||||
const nodeResult = await conn.query(`MATCH (n:${escapeTableName(tableName)}) RETURN n.id AS id LIMIT 1`);
|
||||
const nodeRows = await nodeResult.getAllRows();
|
||||
const nodeRow = nodeRows[0];
|
||||
if (nodeRow) {
|
||||
testNodeId = nodeRow.id ?? nodeRow[0];
|
||||
break;
|
||||
}
|
||||
} catch {}
|
||||
}
|
||||
|
||||
if (!testNodeId) {
|
||||
return { success: false, error: 'No nodes found to test with' };
|
||||
}
|
||||
|
||||
if (import.meta.env.DEV) {
|
||||
console.log('🧪 Testing array params with node:', testNodeId);
|
||||
}
|
||||
|
||||
// First create an embedding entry
|
||||
const createQuery = `CREATE (e:${EMBEDDING_TABLE_NAME} {nodeId: $nodeId, embedding: $embedding})`;
|
||||
const stmt = await conn.prepare(createQuery);
|
||||
|
||||
if (!stmt.isSuccess()) {
|
||||
const errMsg = await stmt.getErrorMessage();
|
||||
return { success: false, error: `Prepare failed: ${errMsg}` };
|
||||
}
|
||||
|
||||
await conn.execute(stmt, {
|
||||
nodeId: testNodeId,
|
||||
embedding: testEmbedding,
|
||||
});
|
||||
|
||||
await stmt.close();
|
||||
|
||||
// Verify it was stored (using prepared statement to avoid injection)
|
||||
const verifyStmt = await conn.prepare(
|
||||
`MATCH (e:${EMBEDDING_TABLE_NAME} {nodeId: $nodeId}) RETURN e.embedding AS emb`
|
||||
);
|
||||
try {
|
||||
if (!verifyStmt.isSuccess()) {
|
||||
const errMsg = await verifyStmt.getErrorMessage();
|
||||
return { success: false, error: `Verify prepare failed: ${errMsg}` };
|
||||
}
|
||||
const verifyResult = await conn.execute(verifyStmt, { nodeId: testNodeId });
|
||||
const verifyRows = await verifyResult.getAllRows();
|
||||
const verifyRow = verifyRows[0];
|
||||
const storedEmb = verifyRow?.emb ?? verifyRow?.[0];
|
||||
|
||||
// Clean up test embedding
|
||||
try {
|
||||
const cleanupStmt = await conn.prepare(`MATCH (e:${EMBEDDING_TABLE_NAME} {nodeId: $nodeId}) DELETE e`);
|
||||
try { await conn.execute(cleanupStmt, { nodeId: testNodeId }); } finally { await cleanupStmt.close(); }
|
||||
} catch {}
|
||||
|
||||
if (storedEmb && Array.isArray(storedEmb) && storedEmb.length === 384) {
|
||||
if (import.meta.env.DEV) {
|
||||
console.log('✅ Array params WORK! Stored embedding length:', storedEmb.length);
|
||||
}
|
||||
return { success: true };
|
||||
} else {
|
||||
return {
|
||||
success: false,
|
||||
error: `Embedding not stored correctly. Got: ${typeof storedEmb}, length: ${storedEmb?.length}`
|
||||
};
|
||||
}
|
||||
} finally {
|
||||
await verifyStmt.close();
|
||||
}
|
||||
} catch (error) {
|
||||
const errorMsg = error instanceof Error ? error.message : String(error);
|
||||
if (import.meta.env.DEV) {
|
||||
console.error('❌ Array params test failed:', errorMsg);
|
||||
}
|
||||
return { success: false, error: errorMsg };
|
||||
}
|
||||
};
|
||||
@@ -0,0 +1,41 @@
|
||||
import { describe, expect, it, vi } from 'vitest';
|
||||
|
||||
import { getQueryRows } from './query-result';
|
||||
|
||||
describe('getQueryRows', () => {
|
||||
it('prefers getAllObjects when available', async () => {
|
||||
const rows = [{ name: 'foo' }];
|
||||
const getAllObjects = vi.fn().mockResolvedValue(rows);
|
||||
const getAllRows = vi.fn().mockResolvedValue([{ name: 'bar' }]);
|
||||
const getAll = vi.fn().mockResolvedValue([['baz']]);
|
||||
|
||||
await expect(getQueryRows({ getAllObjects, getAllRows, getAll })).resolves.toEqual(rows);
|
||||
|
||||
expect(getAllObjects).toHaveBeenCalledTimes(1);
|
||||
expect(getAllRows).not.toHaveBeenCalled();
|
||||
expect(getAll).not.toHaveBeenCalled();
|
||||
});
|
||||
|
||||
it('falls back to getAllRows when getAllObjects is absent', async () => {
|
||||
const rows = [{ name: 'bar' }];
|
||||
const getAllRows = vi.fn().mockResolvedValue(rows);
|
||||
const getAll = vi.fn().mockResolvedValue([['baz']]);
|
||||
|
||||
await expect(getQueryRows({ getAllRows, getAll })).resolves.toEqual(rows);
|
||||
|
||||
expect(getAllRows).toHaveBeenCalledTimes(1);
|
||||
expect(getAll).not.toHaveBeenCalled();
|
||||
});
|
||||
|
||||
it('falls back to getAll as a final fallback', async () => {
|
||||
const rows = [['baz']];
|
||||
const getAll = vi.fn().mockResolvedValue(rows);
|
||||
|
||||
await expect(getQueryRows({ getAll })).resolves.toEqual(rows);
|
||||
expect(getAll).toHaveBeenCalledTimes(1);
|
||||
});
|
||||
|
||||
it('throws when no supported query API is exposed', async () => {
|
||||
await expect(getQueryRows({})).rejects.toThrow('Unsupported LadybugDB QueryResult shape');
|
||||
});
|
||||
});
|
||||
@@ -0,0 +1,21 @@
|
||||
export const getQueryRows = async (result: unknown): Promise<any[]> => {
|
||||
if (!result || typeof result !== 'object') return [];
|
||||
|
||||
const queryResult = result as {
|
||||
getAllObjects?: () => Promise<any[]>;
|
||||
getAllRows?: () => Promise<any[]>;
|
||||
getAll?: () => Promise<any[]>;
|
||||
};
|
||||
|
||||
if (typeof queryResult.getAllObjects === 'function') {
|
||||
return await queryResult.getAllObjects();
|
||||
}
|
||||
if (typeof queryResult.getAllRows === 'function') {
|
||||
return await queryResult.getAllRows();
|
||||
}
|
||||
if (typeof queryResult.getAll === 'function') {
|
||||
return await queryResult.getAll();
|
||||
}
|
||||
|
||||
throw new Error('Unsupported LadybugDB QueryResult shape');
|
||||
};
|
||||
@@ -1,5 +1,5 @@
|
||||
/**
|
||||
* KuzuDB Schema Definitions
|
||||
* LadybugDB Schema Definitions
|
||||
*
|
||||
* Hybrid Schema:
|
||||
* - Separate node tables for each code element type (File, Function, Class, etc.)
|
||||
@@ -26,7 +26,7 @@ export type NodeTableName = typeof NODE_TABLES[number];
|
||||
export const REL_TABLE_NAME = 'CodeRelation';
|
||||
|
||||
// Valid relation types
|
||||
export const REL_TYPES = ['CONTAINS', 'DEFINES', 'IMPORTS', 'CALLS', 'EXTENDS', 'IMPLEMENTS', 'MEMBER_OF', 'STEP_IN_PROCESS'] as const;
|
||||
export const REL_TYPES = ['CONTAINS', 'DEFINES', 'IMPORTS', 'CALLS', 'EXTENDS', 'IMPLEMENTS', 'MEMBER_OF', 'STEP_IN_PROCESS', 'HAS_METHOD', 'HAS_PROPERTY', 'OVERRIDES', 'ACCESSES', 'INHERITS', 'USES', 'DECORATES'] as const;
|
||||
export type RelType = typeof REL_TYPES[number];
|
||||
|
||||
// ============================================================================
|
||||
@@ -13,20 +13,23 @@ import { ChatAnthropic } from '@langchain/anthropic';
|
||||
import { ChatOllama } from '@langchain/ollama';
|
||||
import type { BaseChatModel } from '@langchain/core/language_models/chat_models';
|
||||
import { createGraphRAGTools } from './tools';
|
||||
import type {
|
||||
ProviderConfig,
|
||||
import type {
|
||||
ProviderConfig,
|
||||
OpenAIConfig,
|
||||
AzureOpenAIConfig,
|
||||
AzureOpenAIConfig,
|
||||
GeminiConfig,
|
||||
AnthropicConfig,
|
||||
OllamaConfig,
|
||||
OpenRouterConfig,
|
||||
MiniMaxConfig,
|
||||
GLMConfig,
|
||||
AgentStreamChunk,
|
||||
} from './types';
|
||||
import {
|
||||
type CodebaseContext,
|
||||
buildDynamicSystemPrompt,
|
||||
} from './context-builder';
|
||||
import { DEFAULT_OLLAMA_BASE_URL, DEFAULT_OPENROUTER_BASE_URL } from '../../config/ui-constants';
|
||||
|
||||
/**
|
||||
* System prompt for the Graph RAG agent
|
||||
@@ -183,7 +186,7 @@ export const createChatModel = (config: ProviderConfig): BaseChatModel => {
|
||||
case 'ollama': {
|
||||
const ollamaConfig = config as OllamaConfig;
|
||||
return new ChatOllama({
|
||||
baseUrl: ollamaConfig.baseUrl ?? 'http://localhost:11434',
|
||||
baseUrl: ollamaConfig.baseUrl ?? DEFAULT_OLLAMA_BASE_URL,
|
||||
model: ollamaConfig.model,
|
||||
temperature: ollamaConfig.temperature ?? 0.1,
|
||||
streaming: true,
|
||||
@@ -197,21 +200,20 @@ export const createChatModel = (config: ProviderConfig): BaseChatModel => {
|
||||
|
||||
case 'openrouter': {
|
||||
const openRouterConfig = config as OpenRouterConfig;
|
||||
|
||||
|
||||
// Debug logging
|
||||
if (import.meta.env.DEV) {
|
||||
console.log('🌐 OpenRouter config:', {
|
||||
hasApiKey: !!openRouterConfig.apiKey,
|
||||
apiKeyLength: openRouterConfig.apiKey?.length || 0,
|
||||
model: openRouterConfig.model,
|
||||
baseUrl: openRouterConfig.baseUrl,
|
||||
});
|
||||
}
|
||||
|
||||
|
||||
if (!openRouterConfig.apiKey || openRouterConfig.apiKey.trim() === '') {
|
||||
throw new Error('OpenRouter API key is required but was not provided');
|
||||
}
|
||||
|
||||
|
||||
return new ChatOpenAI({
|
||||
openAIApiKey: openRouterConfig.apiKey,
|
||||
apiKey: openRouterConfig.apiKey, // Fallback for some versions
|
||||
@@ -220,12 +222,51 @@ export const createChatModel = (config: ProviderConfig): BaseChatModel => {
|
||||
maxTokens: openRouterConfig.maxTokens,
|
||||
configuration: {
|
||||
apiKey: openRouterConfig.apiKey, // Ensure client receives it
|
||||
baseURL: openRouterConfig.baseUrl ?? 'https://openrouter.ai/api/v1',
|
||||
baseURL: openRouterConfig.baseUrl ?? DEFAULT_OPENROUTER_BASE_URL,
|
||||
},
|
||||
streaming: true,
|
||||
});
|
||||
}
|
||||
|
||||
|
||||
case 'minimax': {
|
||||
const minimaxConfig = config as MiniMaxConfig;
|
||||
|
||||
if (!minimaxConfig.apiKey || minimaxConfig.apiKey.trim() === '') {
|
||||
throw new Error('MiniMax API key is required but was not provided');
|
||||
}
|
||||
|
||||
return new ChatAnthropic({
|
||||
anthropicApiKey: minimaxConfig.apiKey,
|
||||
model: minimaxConfig.model,
|
||||
temperature: minimaxConfig.temperature ?? 0.1,
|
||||
maxTokens: minimaxConfig.maxTokens ?? 8192,
|
||||
streaming: true,
|
||||
clientOptions: {
|
||||
baseURL: 'https://api.minimax.io/anthropic',
|
||||
},
|
||||
});
|
||||
}
|
||||
|
||||
case 'glm': {
|
||||
const glmConfig = config as GLMConfig;
|
||||
|
||||
if (!glmConfig.apiKey || glmConfig.apiKey.trim() === '') {
|
||||
throw new Error('GLM API key is required but was not provided');
|
||||
}
|
||||
|
||||
return new ChatOpenAI({
|
||||
apiKey: glmConfig.apiKey,
|
||||
modelName: glmConfig.model,
|
||||
temperature: glmConfig.temperature ?? 0.1,
|
||||
maxTokens: glmConfig.maxTokens,
|
||||
configuration: {
|
||||
apiKey: glmConfig.apiKey,
|
||||
baseURL: glmConfig.baseUrl ?? 'https://api.z.ai/api/coding/paas/v4',
|
||||
},
|
||||
streaming: true,
|
||||
});
|
||||
}
|
||||
|
||||
default:
|
||||
throw new Error(`Unsupported provider: ${(config as any).provider}`);
|
||||
}
|
||||
@@ -335,8 +376,8 @@ export async function* streamAgentResponse(
|
||||
const yieldedToolCalls = new Set<string>();
|
||||
const yieldedToolResults = new Set<string>();
|
||||
let lastProcessedMsgCount = formattedMessages.length;
|
||||
// Track if all tools are done (for distinguishing reasoning vs final content)
|
||||
let allToolsDone = true;
|
||||
// Track pending tool calls (for distinguishing reasoning vs final content)
|
||||
let pendingToolCalls = 0;
|
||||
// Track if we've seen any tool calls in this response turn.
|
||||
// Anything before the first tool call should be treated as "reasoning/narration"
|
||||
// so the UI can show the Cursor-like loop: plan → tool → update → tool → answer.
|
||||
@@ -400,7 +441,7 @@ export async function* streamAgentResponse(
|
||||
const isReasoning =
|
||||
!hasSeenToolCallThisTurn ||
|
||||
toolCalls.length > 0 ||
|
||||
!allToolsDone;
|
||||
pendingToolCalls > 0;
|
||||
yield {
|
||||
type: isReasoning ? 'reasoning' : 'content',
|
||||
[isReasoning ? 'reasoning' : 'content']: content,
|
||||
@@ -410,17 +451,23 @@ export async function* streamAgentResponse(
|
||||
// Track tool calls from message chunks
|
||||
if (toolCalls.length > 0) {
|
||||
hasSeenToolCallThisTurn = true;
|
||||
allToolsDone = false;
|
||||
pendingToolCalls += toolCalls.length;
|
||||
for (const tc of toolCalls) {
|
||||
const toolId = tc.id || `tool-${Date.now()}-${Math.random().toString(36).slice(2)}`;
|
||||
if (!yieldedToolCalls.has(toolId)) {
|
||||
yieldedToolCalls.add(toolId);
|
||||
let parsedArgs: Record<string, any>;
|
||||
try {
|
||||
parsedArgs = tc.function?.arguments ? JSON.parse(tc.function.arguments) : {};
|
||||
} catch {
|
||||
parsedArgs = {};
|
||||
}
|
||||
yield {
|
||||
type: 'tool_call',
|
||||
toolCall: {
|
||||
id: toolId,
|
||||
name: tc.name || tc.function?.name || 'unknown',
|
||||
args: tc.args || (tc.function?.arguments ? JSON.parse(tc.function.arguments) : {}),
|
||||
args: tc.args || parsedArgs,
|
||||
status: 'running',
|
||||
},
|
||||
};
|
||||
@@ -445,8 +492,8 @@ export async function* streamAgentResponse(
|
||||
status: 'completed',
|
||||
},
|
||||
};
|
||||
// After tool result, next AI content could be reasoning or final
|
||||
allToolsDone = true;
|
||||
// After tool result, decrement pending count
|
||||
pendingToolCalls = Math.max(0, pendingToolCalls - 1);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -466,7 +513,7 @@ export async function* streamAgentResponse(
|
||||
for (const tc of toolCalls) {
|
||||
const toolId = tc.id || `tool-${Date.now()}`;
|
||||
if (!yieldedToolCalls.has(toolId)) {
|
||||
allToolsDone = false;
|
||||
pendingToolCalls++;
|
||||
yieldedToolCalls.add(toolId);
|
||||
yield {
|
||||
type: 'tool_call',
|
||||
@@ -497,7 +544,7 @@ export async function* streamAgentResponse(
|
||||
status: 'completed',
|
||||
},
|
||||
};
|
||||
allToolsDone = true;
|
||||
pendingToolCalls = Math.max(0, pendingToolCalls - 1);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -5,9 +5,9 @@
|
||||
* All API keys are stored locally - never sent to any server except the LLM provider.
|
||||
*/
|
||||
|
||||
import {
|
||||
LLMSettings,
|
||||
DEFAULT_LLM_SETTINGS,
|
||||
import {
|
||||
LLMSettings,
|
||||
DEFAULT_LLM_SETTINGS,
|
||||
LLMProvider,
|
||||
OpenAIConfig,
|
||||
AzureOpenAIConfig,
|
||||
@@ -15,52 +15,93 @@ import {
|
||||
AnthropicConfig,
|
||||
OllamaConfig,
|
||||
OpenRouterConfig,
|
||||
MiniMaxConfig,
|
||||
GLMConfig,
|
||||
ProviderConfig,
|
||||
} from './types';
|
||||
import { DEFAULT_OPENROUTER_BASE_URL, DEFAULT_OLLAMA_BASE_URL } from '../../config/ui-constants';
|
||||
|
||||
const STORAGE_KEY = 'gitnexus-llm-settings';
|
||||
|
||||
const mergeWithDefaults = (parsed?: Partial<LLMSettings> | null): LLMSettings => ({
|
||||
...DEFAULT_LLM_SETTINGS,
|
||||
...parsed,
|
||||
openai: {
|
||||
...DEFAULT_LLM_SETTINGS.openai,
|
||||
...parsed?.openai,
|
||||
},
|
||||
azureOpenAI: {
|
||||
...DEFAULT_LLM_SETTINGS.azureOpenAI,
|
||||
...parsed?.azureOpenAI,
|
||||
},
|
||||
gemini: {
|
||||
...DEFAULT_LLM_SETTINGS.gemini,
|
||||
...parsed?.gemini,
|
||||
},
|
||||
anthropic: {
|
||||
...DEFAULT_LLM_SETTINGS.anthropic,
|
||||
...parsed?.anthropic,
|
||||
},
|
||||
ollama: {
|
||||
...DEFAULT_LLM_SETTINGS.ollama,
|
||||
...parsed?.ollama,
|
||||
},
|
||||
openrouter: {
|
||||
...DEFAULT_LLM_SETTINGS.openrouter,
|
||||
...parsed?.openrouter,
|
||||
},
|
||||
minimax: {
|
||||
...DEFAULT_LLM_SETTINGS.minimax,
|
||||
...parsed?.minimax,
|
||||
},
|
||||
glm: {
|
||||
...DEFAULT_LLM_SETTINGS.glm,
|
||||
...parsed?.glm,
|
||||
},
|
||||
});
|
||||
|
||||
const readSettings = (storage: Storage): Partial<LLMSettings> | null => {
|
||||
const raw = storage.getItem(STORAGE_KEY);
|
||||
if (!raw) return null;
|
||||
try {
|
||||
return JSON.parse(raw) as Partial<LLMSettings>;
|
||||
} catch (error) {
|
||||
console.warn('Failed to parse LLM settings:', error);
|
||||
return null;
|
||||
}
|
||||
};
|
||||
|
||||
const writeSettings = (storage: Storage, settings: LLMSettings): void => {
|
||||
storage.setItem(STORAGE_KEY, JSON.stringify(settings));
|
||||
};
|
||||
|
||||
/**
|
||||
* Load settings from localStorage
|
||||
* Load settings from sessionStorage (migrates legacy localStorage once).
|
||||
*/
|
||||
export const loadSettings = (): LLMSettings => {
|
||||
try {
|
||||
const stored = localStorage.getItem(STORAGE_KEY);
|
||||
if (!stored) {
|
||||
return DEFAULT_LLM_SETTINGS;
|
||||
const sessionData = typeof sessionStorage !== 'undefined' ? readSettings(sessionStorage) : null;
|
||||
if (sessionData) {
|
||||
return mergeWithDefaults(sessionData);
|
||||
}
|
||||
|
||||
const parsed = JSON.parse(stored) as Partial<LLMSettings>;
|
||||
|
||||
// Merge with defaults to handle new fields
|
||||
return {
|
||||
...DEFAULT_LLM_SETTINGS,
|
||||
...parsed,
|
||||
openai: {
|
||||
...DEFAULT_LLM_SETTINGS.openai,
|
||||
...parsed.openai,
|
||||
},
|
||||
azureOpenAI: {
|
||||
...DEFAULT_LLM_SETTINGS.azureOpenAI,
|
||||
...parsed.azureOpenAI,
|
||||
},
|
||||
gemini: {
|
||||
...DEFAULT_LLM_SETTINGS.gemini,
|
||||
...parsed.gemini,
|
||||
},
|
||||
anthropic: {
|
||||
...DEFAULT_LLM_SETTINGS.anthropic,
|
||||
...parsed.anthropic,
|
||||
},
|
||||
ollama: {
|
||||
...DEFAULT_LLM_SETTINGS.ollama,
|
||||
...parsed.ollama,
|
||||
},
|
||||
openrouter: {
|
||||
...DEFAULT_LLM_SETTINGS.openrouter,
|
||||
...parsed.openrouter,
|
||||
},
|
||||
};
|
||||
|
||||
const legacyData = typeof localStorage !== 'undefined' ? readSettings(localStorage) : null;
|
||||
if (legacyData) {
|
||||
const merged = mergeWithDefaults(legacyData);
|
||||
try {
|
||||
if (typeof sessionStorage !== 'undefined') {
|
||||
writeSettings(sessionStorage, merged);
|
||||
}
|
||||
if (typeof localStorage !== 'undefined') {
|
||||
localStorage.removeItem(STORAGE_KEY);
|
||||
}
|
||||
} catch (error) {
|
||||
console.warn('Failed to migrate legacy LLM settings to sessionStorage:', error);
|
||||
}
|
||||
return merged;
|
||||
}
|
||||
|
||||
return DEFAULT_LLM_SETTINGS;
|
||||
} catch (error) {
|
||||
console.warn('Failed to load LLM settings:', error);
|
||||
return DEFAULT_LLM_SETTINGS;
|
||||
@@ -68,11 +109,13 @@ export const loadSettings = (): LLMSettings => {
|
||||
};
|
||||
|
||||
/**
|
||||
* Save settings to localStorage
|
||||
* Save settings to sessionStorage
|
||||
*/
|
||||
export const saveSettings = (settings: LLMSettings): void => {
|
||||
try {
|
||||
localStorage.setItem(STORAGE_KEY, JSON.stringify(settings));
|
||||
if (typeof sessionStorage !== 'undefined') {
|
||||
writeSettings(sessionStorage, settings);
|
||||
}
|
||||
} catch (error) {
|
||||
console.error('Failed to save LLM settings:', error);
|
||||
}
|
||||
@@ -89,6 +132,9 @@ export const updateProviderSettings = <T extends LLMProvider>(
|
||||
T extends 'gemini' ? Partial<Omit<GeminiConfig, 'provider'>> :
|
||||
T extends 'anthropic' ? Partial<Omit<AnthropicConfig, 'provider'>> :
|
||||
T extends 'ollama' ? Partial<Omit<OllamaConfig, 'provider'>> :
|
||||
T extends 'openrouter' ? Partial<Omit<OpenRouterConfig, 'provider'>> :
|
||||
T extends 'minimax' ? Partial<Omit<MiniMaxConfig, 'provider'>> :
|
||||
T extends 'glm' ? Partial<Omit<GLMConfig, 'provider'>> :
|
||||
never
|
||||
>
|
||||
): LLMSettings => {
|
||||
@@ -162,6 +208,28 @@ export const updateProviderSettings = <T extends LLMProvider>(
|
||||
saveSettings(updated);
|
||||
return updated;
|
||||
}
|
||||
case 'minimax': {
|
||||
const updated: LLMSettings = {
|
||||
...current,
|
||||
minimax: {
|
||||
...(current.minimax ?? {}),
|
||||
...(updates as Partial<Omit<MiniMaxConfig, 'provider'>>),
|
||||
},
|
||||
};
|
||||
saveSettings(updated);
|
||||
return updated;
|
||||
}
|
||||
case 'glm': {
|
||||
const updated: LLMSettings = {
|
||||
...current,
|
||||
glm: {
|
||||
...(current.glm ?? {}),
|
||||
...(updates as Partial<Omit<GLMConfig, 'provider'>>),
|
||||
},
|
||||
};
|
||||
saveSettings(updated);
|
||||
return updated;
|
||||
}
|
||||
default: {
|
||||
// Should be unreachable due to T extends LLMProvider, but keep a safe fallback
|
||||
const updated: LLMSettings = { ...current };
|
||||
@@ -187,68 +255,64 @@ export const setActiveProvider = (provider: LLMProvider): LLMSettings => {
|
||||
/**
|
||||
* Get the current provider configuration
|
||||
*/
|
||||
type ProviderBuilder = (settings: LLMSettings) => ProviderConfig | null;
|
||||
|
||||
const providerBuilders: Record<LLMProvider, ProviderBuilder> = {
|
||||
openai: (settings) => {
|
||||
if (!settings.openai?.apiKey) return null;
|
||||
return { provider: 'openai', ...settings.openai } as OpenAIConfig;
|
||||
},
|
||||
'azure-openai': (settings) => {
|
||||
if (!settings.azureOpenAI?.apiKey || !settings.azureOpenAI?.endpoint) return null;
|
||||
return { provider: 'azure-openai', ...settings.azureOpenAI } as AzureOpenAIConfig;
|
||||
},
|
||||
gemini: (settings) => {
|
||||
if (!settings.gemini?.apiKey) return null;
|
||||
return { provider: 'gemini', ...settings.gemini } as GeminiConfig;
|
||||
},
|
||||
anthropic: (settings) => {
|
||||
if (!settings.anthropic?.apiKey) return null;
|
||||
return { provider: 'anthropic', ...settings.anthropic } as AnthropicConfig;
|
||||
},
|
||||
ollama: (settings) => {
|
||||
return {
|
||||
provider: 'ollama',
|
||||
...settings.ollama,
|
||||
baseUrl: settings.ollama?.baseUrl ?? DEFAULT_OLLAMA_BASE_URL,
|
||||
} as OllamaConfig;
|
||||
},
|
||||
openrouter: (settings) => {
|
||||
if (!settings.openrouter?.apiKey || settings.openrouter.apiKey.trim() === '') return null;
|
||||
return {
|
||||
provider: 'openrouter',
|
||||
apiKey: settings.openrouter.apiKey,
|
||||
model: settings.openrouter.model || '',
|
||||
baseUrl: settings.openrouter.baseUrl || DEFAULT_OPENROUTER_BASE_URL,
|
||||
temperature: settings.openrouter.temperature,
|
||||
maxTokens: settings.openrouter.maxTokens,
|
||||
} as OpenRouterConfig;
|
||||
},
|
||||
minimax: (settings) => {
|
||||
if (!settings.minimax?.apiKey) return null;
|
||||
return { provider: 'minimax', ...settings.minimax } as MiniMaxConfig;
|
||||
},
|
||||
glm: (settings) => {
|
||||
if (!settings.glm?.apiKey) return null;
|
||||
return {
|
||||
provider: 'glm',
|
||||
apiKey: settings.glm.apiKey,
|
||||
model: settings.glm.model || 'GLM-5',
|
||||
baseUrl: settings.glm.baseUrl || 'https://api.z.ai/api/coding/paas/v4',
|
||||
temperature: settings.glm.temperature,
|
||||
maxTokens: settings.glm.maxTokens,
|
||||
} as GLMConfig;
|
||||
},
|
||||
};
|
||||
|
||||
export const getActiveProviderConfig = (): ProviderConfig | null => {
|
||||
const settings = loadSettings();
|
||||
|
||||
switch (settings.activeProvider) {
|
||||
case 'openai':
|
||||
if (!settings.openai?.apiKey) {
|
||||
return null;
|
||||
}
|
||||
return {
|
||||
provider: 'openai',
|
||||
...settings.openai,
|
||||
} as OpenAIConfig;
|
||||
|
||||
case 'azure-openai':
|
||||
if (!settings.azureOpenAI?.apiKey || !settings.azureOpenAI?.endpoint) {
|
||||
return null;
|
||||
}
|
||||
return {
|
||||
provider: 'azure-openai',
|
||||
...settings.azureOpenAI,
|
||||
} as AzureOpenAIConfig;
|
||||
|
||||
case 'gemini':
|
||||
if (!settings.gemini?.apiKey) {
|
||||
return null;
|
||||
}
|
||||
return {
|
||||
provider: 'gemini',
|
||||
...settings.gemini,
|
||||
} as GeminiConfig;
|
||||
|
||||
case 'anthropic':
|
||||
if (!settings.anthropic?.apiKey) {
|
||||
return null;
|
||||
}
|
||||
return {
|
||||
provider: 'anthropic',
|
||||
...settings.anthropic,
|
||||
} as AnthropicConfig;
|
||||
|
||||
case 'ollama':
|
||||
return {
|
||||
provider: 'ollama',
|
||||
...settings.ollama,
|
||||
} as OllamaConfig;
|
||||
|
||||
case 'openrouter':
|
||||
if (!settings.openrouter?.apiKey || settings.openrouter.apiKey.trim() === '') {
|
||||
return null;
|
||||
}
|
||||
return {
|
||||
provider: 'openrouter',
|
||||
apiKey: settings.openrouter.apiKey,
|
||||
model: settings.openrouter.model || '',
|
||||
baseUrl: settings.openrouter.baseUrl || 'https://openrouter.ai/api/v1',
|
||||
temperature: settings.openrouter.temperature,
|
||||
maxTokens: settings.openrouter.maxTokens,
|
||||
} as OpenRouterConfig;
|
||||
|
||||
default:
|
||||
return null;
|
||||
}
|
||||
const builder = providerBuilders[settings.activeProvider];
|
||||
return builder ? builder(settings) : null;
|
||||
};
|
||||
|
||||
/**
|
||||
@@ -262,7 +326,16 @@ export const isProviderConfigured = (): boolean => {
|
||||
* Clear all settings (reset to defaults)
|
||||
*/
|
||||
export const clearSettings = (): void => {
|
||||
localStorage.removeItem(STORAGE_KEY);
|
||||
try {
|
||||
if (typeof sessionStorage !== 'undefined') {
|
||||
sessionStorage.removeItem(STORAGE_KEY);
|
||||
}
|
||||
if (typeof localStorage !== 'undefined') {
|
||||
localStorage.removeItem(STORAGE_KEY);
|
||||
}
|
||||
} catch (error) {
|
||||
console.warn('Failed to clear LLM settings:', error);
|
||||
}
|
||||
};
|
||||
|
||||
/**
|
||||
@@ -282,6 +355,10 @@ export const getProviderDisplayName = (provider: LLMProvider): string => {
|
||||
return 'Ollama (Local)';
|
||||
case 'openrouter':
|
||||
return 'OpenRouter';
|
||||
case 'minimax':
|
||||
return 'MiniMax';
|
||||
case 'glm':
|
||||
return 'GLM (Z.AI)';
|
||||
default:
|
||||
return provider;
|
||||
}
|
||||
@@ -303,6 +380,10 @@ export const getAvailableModels = (provider: LLMProvider): string[] => {
|
||||
return ['claude-sonnet-4-20250514', 'claude-3-5-sonnet-20241022', 'claude-3-5-haiku-20241022', 'claude-3-opus-20240229'];
|
||||
case 'ollama':
|
||||
return ['llama3.2', 'llama3.1', 'mistral', 'codellama', 'deepseek-coder'];
|
||||
case 'minimax':
|
||||
return ['MiniMax-M2.5', 'MiniMax-M2.5-highspeed'];
|
||||
case 'glm':
|
||||
return ['GLM-5', 'GLM-5-Turbo', 'GLM-4.7', 'GLM-4.5'];
|
||||
default:
|
||||
return [];
|
||||
}
|
||||
@@ -313,7 +394,7 @@ export const getAvailableModels = (provider: LLMProvider): string[] => {
|
||||
*/
|
||||
export const fetchOpenRouterModels = async (): Promise<Array<{ id: string; name: string }>> => {
|
||||
try {
|
||||
const response = await fetch('https://openrouter.ai/api/v1/models');
|
||||
const response = await fetch(`${DEFAULT_OPENROUTER_BASE_URL}/models`);
|
||||
if (!response.ok) throw new Error('Failed to fetch models');
|
||||
const data = await response.json();
|
||||
return data.data.map((model: any) => ({
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user