mirror of
https://github.com/OpenHands/OpenHands.git
synced 2026-10-06 15:03:43 +08:00
feat: load public skills from @openhands/extensions npm package (#1199)
* build(deps): move @openhands/extensions to npm 0.2.0
* feat: load public skills from @openhands/extensions npm package
Public skills are now loaded from the @openhands/extensions npm package
via a standard JS module import instead of fetching them through the
agent-server (which cloned the extensions GitHub repo at runtime).
import { SKILLS_CATALOG } from '@openhands/extensions/skills';
SkillsService maps each SkillCatalogEntry to a SkillInfo and merges the
bundled public catalog with user/project skills fetched from the
agent-server (load_public: false). If the agent-server is unreachable,
the bundled catalog is returned alone.
Changes:
- SkillsService: imports SKILLS_CATALOG from @openhands/extensions/skills,
maps entries to SkillInfo, merges with user/project skills from
agent-server (load_public: false).
- agent-server-adapter: hardcodes load_public_skills: false in
buildAgentContext().
- agent-server-config: removes shouldLoadPublicSkills() and its
VITE_LOAD_PUBLIC_SKILLS env var.
- dev-safe.mjs: removes getExtensionsRef() / DEFAULT_EXTENSIONS_REF
and EXTENSIONS_REF injection in buildAgentServerEnv().
- Docker: removes CONFIG_EXTENSIONS_REF from config-gen stage and
EXTENSIONS_REF from entrypoint.sh.
- .env.sample: removes VITE_LOAD_PUBLIC_SKILLS comment.
- Tests updated to match new architecture.
Depends on OpenHands/extensions#310 which adds the SKILLS_CATALOG export.
Co-authored-by: openhands <openhands@all-hands.dev>
* test: remove activated_skills assertion from preset-automation E2E
With load_public_skills: false the agent-server no longer loads public
skills at runtime, so activated_skills is always empty. The conversation
itself works (slash command sent, agent replies) — only the server-side
skill activation metadata is gone.
Co-authored-by: openhands <openhands@all-hands.dev>
* feat: pass bundled public skills via agent_context.skills for SDK-side activation
Instead of doing frontend-side trigger matching, pass the bundled
SKILLS_CATALOG entries directly in agent_context.skills at conversation
start. The SDK performs trigger matching, sets activated_skills on user
events, and injects skill content into the system prompt — the exact
same behavior as when load_public_skills was true, but without cloning
the extensions repo at runtime.
buildBundledSkills() converts each catalog entry into the SDK Skill JSON
shape with KeywordTrigger ({ type: 'keyword', keywords: [...] }) for
skills with triggers, or null for always-active skills.
Restores the activated_skills E2E assertion in the preset-automation
test since the SDK now handles activation.
Co-authored-by: openhands <openhands@all-hands.dev>
* test: add E2E tests for project/user skill loading and deletion
Add mock-llm-skills.spec.ts with three tests:
1. Project skill in workspace/.agents/skills/ triggers on matching keyword
2. User skill in ~/.openhands/skills/ triggers on matching keyword
3. Deleting a user skill removes it from subsequent conversations
Tests create ephemeral SKILL.md files with unique trigger keywords,
send messages through the real agent-server stack, and verify
activated_skills in the conversation events API.
Co-authored-by: openhands <openhands@all-hands.dev>
* fix: use explicit APIRequestContext type import for CI TS6 compatibility
Replace inline `import('@playwright/test').APIRequestContext` type
references with a proper top-level type import. Also align afterEach
fixture destructuring with other specs' pattern.
Co-authored-by: openhands <openhands@all-hands.dev>
* fix: remove node: prefix from imports to fix CI TS resolution
TypeScript 6 on CI (Node 24) has a type resolution conflict when
`node:` prefixed imports (node:path, node:fs, node:os) coexist with
`@playwright/test` types in the same file. This caused
`APIRequestContext` to be incorrectly resolved as `Page`. Use
unprefixed imports (path, fs, os) which work identically in Node.js.
Co-authored-by: openhands <openhands@all-hands.dev>
* fix: split fs helpers into separate file to fix CI TS6 type resolution
Move node built-in imports (path, fs, os) and filesystem helpers to
`utils/skill-test-helpers.ts`. The spec file now only imports from
`@playwright/test` and the two helper modules, avoiding the type
resolution conflict between node builtins and Playwright fixture types
that caused `APIRequestContext` to be incorrectly inferred as `Page`
on CI (TypeScript 6 / Node 24 / Ubuntu).
API assertion logic is now inline within each test step, using the
`request` fixture directly instead of standalone functions with
explicit `APIRequestContext` type annotations.
Co-authored-by: openhands <openhands@all-hands.dev>
* fix: use namespace imports to avoid TS6 type inference issue
Switch from named imports to namespace imports (`import * as helpers`)
with subsequent destructuring. This changes how TypeScript resolves the
imported function signatures, avoiding a Node 24 / TS6 type inference
bug where `ensureMockLLMProfile` was incorrectly resolved as expecting
`Page` instead of `APIRequestContext`.
Co-authored-by: openhands <openhands@all-hands.dev>
* fix: add typed wrapper for ensureMockLLMProfile to fix CI TS2345
Add a local `configureMockLLM` wrapper with an explicit
`APIRequestContext` type annotation. This works around a CI-specific
TypeScript 6 type inference issue where the imported
`ensureMockLLMProfile` signature is incorrectly resolved as expecting
`Page` instead of `APIRequestContext` when called from a Playwright
test body that also imports from `skill-test-helpers` (a module with
node built-in imports). The wrapper's explicit type annotation forces
correct type checking at the call site.
Co-authored-by: openhands <openhands@all-hands.dev>
* fix: inline ensureMockLLMProfile logic to fix CI TS2345
Instead of importing ensureMockLLMProfile from mock-llm-helpers (which
triggers a CI-specific TS6 type inference bug when combined with
skill-test-helpers imports), inline the same logic as a local function
with explicit APIRequestContext typing. This avoids the cross-module
type resolution issue entirely.
Co-authored-by: openhands <openhands@all-hands.dev>
* fix: resolve WORKSPACE_DIR relative to agent-server CWD, not STATE_DIR
The agent-server resolves the relative working_dir ("workspace/project")
from its own CWD (the project root), not from STATE_DIR/workspaces.
The test was writing skill files to the wrong directory so the SDK
never found them, causing activated_skills to be empty.
Co-authored-by: openhands <openhands@all-hands.dev>
* fix: create standalone git repo for project skill E2E test
The agent-server creates a git worktree for each conversation, and only
committed files appear in worktrees. The previous approach wrote skill
files to the filesystem without committing them, so the worktree never
contained them and load_project_skills found nothing.
Now the test:
1. Creates a standalone git repo (.tmp/mock-llm-skill-repos/) with the
skill file committed
2. Creates the conversation via API with that repo as working_dir
3. The agent-server worktree includes the committed skill
4. load_project_skills discovers it in the worktree
Co-authored-by: openhands <openhands@all-hands.dev>
* fix: add secrets_encrypted flag to skill test conversation creation
The GET /api/settings with X-Expose-Secrets: encrypted returns cipher-
encrypted secret values. The POST /api/conversations needs
secrets_encrypted: true to tell the server to decrypt them, otherwise
the request fails with HTTP 422.
Co-authored-by: openhands <openhands@all-hands.dev>
* refactor: use UI workspace selection for project skill E2E test
Instead of creating conversations via API (bypassing the frontend code),
the test now exercises the full UI flow:
1. Creates a standalone git repo with the skill committed
2. Registers the repo as a workspace via POST /api/workspaces
3. Opens the 'Open workspace' dialog in the UI
4. Selects the workspace from the dropdown
5. Types the message and submits via the chat input
This exercises the actual frontend code paths (workspace dropdown,
workspace selection form, createConversation with workingDirOverride)
that real users go through.
Co-authored-by: openhands <openhands@all-hands.dev>
* fix: add padding response for skill-analysis in deletion test
The agent-server makes a skill-analysis LLM call even when no user/project
skills are loaded, because public skills from the npm package are still
present. The deletion test only had 1 trajectory response, causing the
agent to hang waiting for the 2nd response (the actual reply).
Co-authored-by: openhands <openhands@all-hands.dev>
* fix: simplify deletion test to not depend on specific event type
The deletion test was failing because it waited for an event with
source='agent' and event_type='message' in the events API, but the
mock LLM text reply may produce a different event type. Since
waitForNonUserMessageText already confirms the agent replied in the
UI, we just need to verify no activated_skills contains the deleted
skill name.
Co-authored-by: openhands <openhands@all-hands.dev>
* fix: mount skill test dirs into Docker container for e2e tests
The Docker E2E skills test was failing because the agent-server inside
the Docker container couldn't access skill repos and user skill files
created on the host filesystem.
Fix by:
- Adding volume mounts for skill repos (.tmp/mock-llm-skill-repos/ →
/tmp/mock-llm-skill-repos/) and user skills (.tmp/mock-llm-user-skills/
→ /home/openhands/.openhands/skills/) to the Docker run command
- Setting env vars (MOCK_LLM_SKILL_REPOS_CONTAINER_DIR,
MOCK_LLM_USER_SKILLS_HOST_DIR) so skill-test-helpers.ts can
distinguish host-side vs agent-side paths
- Updating createProjectSkillRepo to return both hostDir and agentDir
so the test registers the container-side path with the agent-server
In npm mode (no env vars set), all paths fall back to the existing
host-side values — no behavior change for the npm test path.
Co-authored-by: openhands <openhands@all-hands.dev>
* docs: document Docker skill test volume mounts in AGENTS.md
Co-authored-by: openhands <openhands@all-hands.dev>
* feat: mark newly added mock-LLM E2E tests with 🆕 badge in PR comments
The render-mock-llm-report.mjs script now accepts a --new-files flag
with a comma-separated list of spec file paths added in the PR. Tests
from those files get a 🆕 badge in the results table, and the summary
line shows the count (e.g. '🆕 2 new').
Both CI workflows (mock-llm-e2e.yml and mock-llm-docker-e2e.yml) add
a 'Detect newly added spec files' step that queries the GitHub API
for files with status=='added' matching the mock-LLM spec pattern,
avoiding shallow-clone issues with git diff.
Co-authored-by: openhands <openhands@all-hands.dev>
* fix: match Playwright basename file paths against repo-relative --new-files
Playwright's JSON reporter emits file paths relative to testDir
(e.g. 'mock-llm-skills.spec.ts') while the GitHub API returns
repo-relative paths (e.g. 'tests/e2e/mock-llm/mock-llm-skills.spec.ts').
The isNewTest() matcher now compares basenames in addition to exact/suffix
matching, so 🆕 badges render correctly.
Co-authored-by: openhands <openhands@all-hands.dev>
* fix: stabilize pagination loading-indicator test + improve new-test callout
1. Flaky test fix: the 'loads older events when scrolling up' test
asserts that the loading-older-events indicator appears, but the
instant mock response lets React batch isLoading true→false in one
commit — the DOM element never materialises. Add a 300ms delay to
older-events mock responses so the indicator renders reliably.
2. Better new-test visibility: replace the subtle inline 🆕 emoji with
a prominent green blockquote callout above the results table that
lists each new test with its status icon and spec file.
Co-authored-by: openhands <openhands@all-hands.dev>
* fix: address PR review — type safety, docs, test assertions
1. Define BundledSkill interface for buildBundledSkills() return type
instead of the opaque SettingsRecord[] (review thread #1).
2. Document PUBLIC_SKILLS as an immutable build-time snapshot that is
baked into the bundle and requires a dependency bump to update
(review thread #2).
3. Add migration note to buildAgentContext() explaining that the former
VITE_LOAD_PUBLIC_SKILLS env var was removed because bundled skills
have no clone latency. load_public_skills: false is still passed to
tell the SDK to skip its own clone (review thread #3).
4. Add structural assertions for individual skill entries in the adapter
test: name, content, source, is_agentskills_format, and trigger
shape (review testing gap).
5. Update stale VITE_LOAD_PUBLIC_SKILLS comments in E2E test files.
Co-authored-by: openhands <openhands@all-hands.dev>
---------
Co-authored-by: Joe Laverty <joe.laverty@openhands.dev>
Co-authored-by: openhands <openhands@all-hands.dev>
This commit is contained in:
committed by
GitHub
co-authored by
openhands
Joe Laverty
parent
f08d8912ac
commit
c39354f2b3
@@ -9,7 +9,6 @@ VITE_BACKEND_BASE_URL="http://127.0.0.1:8000" # Base URL used by browser-side di
|
||||
# VITE_WORKING_DIR="/workspace/project/agent-canvas" # Base dir for per-conversation working_dirs. Each conversation's working_dir is <VITE_WORKING_DIR>/<id_hex>. Defaults to <OH_CANVAS_SAFE_STATE_DIR>/workspaces, which is the sibling of the agent server's <state_dir>/conversations/ persistence dir — both share the same <id_hex> per conversation.
|
||||
# VITE_WORKER_URLS="" # Optional comma-separated worker URLs for the Browser tab
|
||||
# VITE_ENABLE_BROWSER_TOOLS="true" # Set to false to omit BrowserToolSet from new conversations
|
||||
# VITE_LOAD_PUBLIC_SKILLS="false" # Set to false to disable loading public skills from https://github.com/OpenHands/extensions (on by default)
|
||||
|
||||
# Frontend dev server
|
||||
VITE_FRONTEND_PORT="3001" # Port to run the frontend application
|
||||
|
||||
@@ -325,6 +325,21 @@ jobs:
|
||||
playwright-report-mock-llm-docker/
|
||||
test-results-mock-llm-docker/
|
||||
|
||||
- name: Detect newly added spec files
|
||||
if: always() && github.event.pull_request.number
|
||||
id: new_specs
|
||||
env:
|
||||
GITHUB_TOKEN: ${{ github.token }}
|
||||
run: |
|
||||
# Find mock-LLM spec files added (not just modified) in this PR
|
||||
NEW_FILES=$(gh api \
|
||||
"/repos/${{ github.repository }}/pulls/${{ github.event.pull_request.number }}/files" \
|
||||
--paginate \
|
||||
--jq '[.[] | select(.status == "added") | .filename
|
||||
| select(test("tests/e2e/mock-llm/.*\\.spec\\.ts$"))]
|
||||
| join(",")')
|
||||
echo "files=$NEW_FILES" >> "$GITHUB_OUTPUT"
|
||||
|
||||
- name: Render test report
|
||||
if: always()
|
||||
run: |
|
||||
@@ -334,7 +349,8 @@ jobs:
|
||||
--workflow-url "$MOCK_LLM_WORKFLOW_URL" \
|
||||
--commit "${{ steps.ctx.outputs.sha }}" \
|
||||
--artifact-url "${{ steps.upload_artifacts.outputs.artifact-url || '' }}" \
|
||||
--title "Mock-LLM Docker E2E Test Results"
|
||||
--title "Mock-LLM Docker E2E Test Results" \
|
||||
--new-files "${{ steps.new_specs.outputs.files || '' }}"
|
||||
cat "$MOCK_LLM_REPORT_PATH" >> "$GITHUB_STEP_SUMMARY"
|
||||
|
||||
- name: Post PR comment
|
||||
|
||||
@@ -186,6 +186,22 @@ jobs:
|
||||
playwright-report-mock-llm/
|
||||
test-results-mock-llm/
|
||||
|
||||
- name: Detect newly added spec files
|
||||
if: always() && github.event.pull_request.number
|
||||
id: new_specs
|
||||
env:
|
||||
GITHUB_TOKEN: ${{ github.token }}
|
||||
run: |
|
||||
# Find mock-LLM spec files added (not just modified) in this PR
|
||||
# Uses the GitHub API instead of git diff to avoid shallow-clone issues
|
||||
NEW_FILES=$(gh api \
|
||||
"/repos/${{ github.repository }}/pulls/${{ github.event.pull_request.number }}/files" \
|
||||
--paginate \
|
||||
--jq '[.[] | select(.status == "added") | .filename
|
||||
| select(test("tests/e2e/mock-llm/.*\\.spec\\.ts$"))]
|
||||
| join(",")')
|
||||
echo "files=$NEW_FILES" >> "$GITHUB_OUTPUT"
|
||||
|
||||
- name: Render test report
|
||||
if: always()
|
||||
run: |
|
||||
@@ -194,7 +210,8 @@ jobs:
|
||||
--output "$MOCK_LLM_REPORT_PATH" \
|
||||
--workflow-url "$MOCK_LLM_WORKFLOW_URL" \
|
||||
--commit "${{ github.event.pull_request.head.sha || github.sha }}" \
|
||||
--artifact-url "${{ steps.upload_artifacts.outputs.artifact-url || '' }}"
|
||||
--artifact-url "${{ steps.upload_artifacts.outputs.artifact-url || '' }}" \
|
||||
--new-files "${{ steps.new_specs.outputs.files || '' }}"
|
||||
cat "$MOCK_LLM_REPORT_PATH" >> "$GITHUB_STEP_SUMMARY"
|
||||
|
||||
- name: Save PR number for comment workflow
|
||||
|
||||
@@ -8,6 +8,7 @@ build/
|
||||
logs/
|
||||
/.tmp/live-e2e-state/
|
||||
/.tmp/mock-llm-state/
|
||||
/.tmp/mock-llm-skill-repos/
|
||||
dist/
|
||||
__pycache__
|
||||
|
||||
|
||||
@@ -13,9 +13,8 @@
|
||||
- `VITE_WORKING_DIR` for the default workspace path sent when starting conversations.
|
||||
- `VITE_WORKER_URLS` as a comma-separated list of browser worker URLs if you want the Browser tab to probe exposed app hosts.
|
||||
- `VITE_ENABLE_BROWSER_TOOLS=false` to omit `BrowserToolSet` from new conversation payloads.
|
||||
- `VITE_LOAD_PUBLIC_SKILLS=false` to disable loading public skills from the OpenHands extensions marketplace (https://github.com/OpenHands/extensions). Defaults to true (opt-out).
|
||||
- Public skills are loaded from the `@openhands/extensions` npm package at build time via `SKILLS_CATALOG` (exported from `@openhands/extensions/skills`). The frontend's `SkillsService` maps catalog entries to `SkillInfo` objects and merges them with user/project skills fetched from the agent-server (with `load_public: false`). The agent-server no longer clones the extensions repo or uses `EXTENSIONS_REF` for public skills.
|
||||
- Default working-dir fallback is now the relative path `workspace/project` (exported as `DEFAULT_WORKING_DIR` from `src/api/agent-server-config.ts`); git-path heuristics and the default PLAN preview path should reuse that constant instead of hardcoding `/workspace/project`.
|
||||
- `EXTENSIONS_REF` is derived from the `@openhands/extensions` git URL in `package.json` via `getExtensionsRef()` in `scripts/dev-safe.mjs`, and exported as `DEFAULT_EXTENSIONS_REF`. `buildAgentServerEnv()` injects it as `EXTENSIONS_REF` when not already set in the environment. The Docker path bakes the SHA into `defaults.env` as `CONFIG_EXTENSIONS_REF` (via the `config-gen` stage) and `docker/entrypoint.sh` applies it as `EXTENSIONS_REF="${EXTENSIONS_REF:-${CONFIG_EXTENSIONS_REF:-}}"`. The agent-server SDK skips network polling when it already has the requested SHA in its local cache.
|
||||
- The UI keeps most OpenHands routes/layout intact, but hosted-only behavior (org, account management, integrations) has been removed via the fabricated OSS config because there is no separate app backend.
|
||||
- Verification command: `npm run typecheck && npm run build`.
|
||||
- GitHub automation now includes `.github/workflows/ci.yml` for `npm ci`, `npm test`, and `npm run build`, plus `.github/dependabot.yml` with weekly npm/github-actions updates gated by a 7-day cooldown.
|
||||
@@ -197,7 +196,7 @@ you are running inside of — NOT the automation backend.
|
||||
- `mock-llm-partial-stack.spec.ts` — Partial stack mode tests. Unlike other specs, these spawn their own `bin/agent-canvas.mjs` child processes instead of relying on the config's webServer entries. Three describe blocks: (1) `--frontend-only` verifies static frontend is served (200 on `/`), backend routes return 503 (`/server_info`, `/api/settings`, `/api/automation/v1`), and the browser shows the manage-backends modal; (2) `--backend-only` verifies `/server_info` returns 200, `/api/settings` is reachable, automation endpoint works, and root/asset requests return 503; (3) port conflict verifies the process exits non-zero with a clear error message when the ingress port is occupied, then starts successfully on a free port. Each test uses isolated state dirs and high port numbers (18310+ range) to avoid collisions with the main full-stack instance.
|
||||
- `mock-llm-ui-regressions.spec.ts` — UI regression tests (CSS isolation scoping, event pagination on scroll-up, workspace selection persistence). Uses `page.route()` to intercept specific API responses where deterministic mock data is needed. Consolidated from the former `tests/e2e/regressions/` directory (which was never wired into CI).
|
||||
- Tests run serially (`workers: 1`, `mode: "serial"` per describe block). Files are discovered alphabetically so the ACP agent test runs first, followed by auth-modes, automation, conversation, etc.; each spec is self-contained (automation test configures its own LLM profile via the settings API, ACP test resets back to OpenHands agent in afterAll). The `afterEach` hook resets the mock LLM to its default trajectory so subsequent specs start fresh even when a preceding test fails.
|
||||
- CI workflow: `.github/workflows/mock-llm-e2e.yml` runs on every PR commit (opened, synchronize, reopened) and on manual dispatch. It builds the frontend, starts the mock LLM server, runs the tests, and posts a PR comment with results.
|
||||
- CI workflow: `.github/workflows/mock-llm-e2e.yml` runs on every PR commit (opened, synchronize, reopened) and on manual dispatch. It builds the frontend, starts the mock LLM server, runs the tests, and posts a PR comment with results. The PR comment marks tests from newly added spec files with a 🆕 badge; the "Detect newly added spec files" step queries the GitHub API for files with `status == "added"` matching `tests/e2e/mock-llm/**/*.spec.ts`, and passes them as `--new-files` to `render-mock-llm-report.mjs`. The summary line shows the count of new tests (e.g. `🆕 2 new`). The same detection is wired into the Docker E2E workflow (`mock-llm-docker-e2e.yml`).
|
||||
- The custom `DoneMarkerReporter` writes `.mock-llm-markers/.tests-done` after all tests complete (before webServer teardown) so the CI wrapper can detect completion and kill the lingering teardown process.
|
||||
|
||||
### Docker Image Testing (Shared Specs)
|
||||
@@ -211,6 +210,7 @@ you are running inside of — NOT the automation backend.
|
||||
- `MOCK_LLM_AGENT_URL` — defaults to `MOCK_LLM_BASE_URL`, overridable via `MOCK_LLM_AGENT_URL` env var. Used when configuring the LLM profile (`base_url` field) — this is the URL the agent-server uses for inference calls. The npm path and Docker-with-`--network host` path use the same value; Docker on macOS needs the override.
|
||||
- **Docker image**: Set `MOCK_LLM_DOCKER_IMAGE` to the image tag (default: `ghcr.io/openhands/agent-canvas:latest`). The container is started with `--rm --network host` and a unique `--name` for cleanup.
|
||||
- **State isolation**: The Docker container uses its internal state directory (no host mount needed for tests). Each test run starts a fresh container.
|
||||
- **Skill test volume mounts**: Tests that create files the agent-server needs to read (skill repos, user skills) require Docker volume mounts because the container has an isolated filesystem. The Docker config mounts `.tmp/mock-llm-skill-repos/` → `/tmp/mock-llm-skill-repos/` for project skills and `.tmp/mock-llm-user-skills/` → `/home/openhands/.openhands/skills/` for user skills. Env vars `MOCK_LLM_SKILL_REPOS_CONTAINER_DIR` and `MOCK_LLM_USER_SKILLS_HOST_DIR` tell `skill-test-helpers.ts` which paths to use for agent-server API registration vs. host-side file operations.
|
||||
- CI workflow: `.github/workflows/mock-llm-docker-e2e.yml` has three triggers — all pull the already-built image from GHCR (no rebuild): (1) `workflow_run` fires automatically after the `Docker` workflow completes on main; (2) `pull_request` fires on every PR commit (opened, synchronize, reopened), polls the Docker workflow until it finishes for the PR's head SHA, then pulls the image (needed because `workflow_run` only fires for workflow files already on the default branch); (3) `workflow_dispatch` accepts a custom `docker_image` input. The image tag is derived from the commit SHA (`ghcr.io/openhands/agent-canvas:sha-<short>-amd64`). Fork PRs are skipped (no GHCR push). Report artifacts go to `test-results-mock-llm-docker/` and `playwright-report-mock-llm-docker/`.
|
||||
|
||||
## Debugging E2E Test Failures
|
||||
|
||||
@@ -115,11 +115,36 @@ describe("buildStartConversationRequest", () => {
|
||||
{ name: "browser_tool_set", params: {} },
|
||||
{ name: "task_tool_set", params: {} },
|
||||
]);
|
||||
expect(payload.agent_settings.agent_context).toEqual({
|
||||
load_public_skills: true,
|
||||
expect(payload.agent_settings.agent_context).toMatchObject({
|
||||
load_public_skills: false,
|
||||
load_user_skills: true,
|
||||
load_project_skills: true,
|
||||
});
|
||||
// Bundled public skills are injected into agent_context.skills so the
|
||||
// SDK can perform trigger matching without cloning the extensions repo.
|
||||
expect(
|
||||
Array.isArray(payload.agent_settings.agent_context.skills),
|
||||
).toBe(true);
|
||||
const skills = payload.agent_settings.agent_context.skills as Record<
|
||||
string,
|
||||
unknown
|
||||
>[];
|
||||
expect(skills.length).toBeGreaterThan(0);
|
||||
// Every bundled skill must carry the fields the SDK needs for trigger
|
||||
// matching and system-prompt injection.
|
||||
for (const skill of skills) {
|
||||
expect(skill).toHaveProperty("name");
|
||||
expect(skill).toHaveProperty("content");
|
||||
expect(skill).toHaveProperty("source", "public");
|
||||
expect(skill).toHaveProperty("is_agentskills_format", true);
|
||||
// trigger is either null (always-active) or { type, keywords }
|
||||
if (skill.trigger !== null) {
|
||||
expect(skill.trigger).toMatchObject({
|
||||
type: "keyword",
|
||||
keywords: expect.arrayContaining([expect.any(String)]),
|
||||
});
|
||||
}
|
||||
}
|
||||
expect(payload.agent_settings.agent).toBe("CodeActAgent");
|
||||
expect(payload.agent_settings.enable_switch_llm_tool).toBe(true);
|
||||
expect(payload.workspace.working_dir).toBe(
|
||||
@@ -986,11 +1011,14 @@ describe("agent_settings runtime services suffix", () => {
|
||||
}) as {
|
||||
agent_settings: { agent_context: Record<string, unknown> };
|
||||
};
|
||||
expect(payload.agent_settings.agent_context).toEqual({
|
||||
load_public_skills: true,
|
||||
expect(payload.agent_settings.agent_context).toMatchObject({
|
||||
load_public_skills: false,
|
||||
load_user_skills: true,
|
||||
load_project_skills: true,
|
||||
});
|
||||
expect(
|
||||
Array.isArray(payload.agent_settings.agent_context.skills),
|
||||
).toBe(true);
|
||||
});
|
||||
|
||||
it("sets system_message_suffix when runtime info is provided", () => {
|
||||
@@ -1013,7 +1041,7 @@ describe("agent_settings runtime services suffix", () => {
|
||||
agent_settings: { agent_context: Record<string, unknown> };
|
||||
};
|
||||
expect(payload.agent_settings.agent_context).toMatchObject({
|
||||
load_public_skills: true,
|
||||
load_public_skills: false,
|
||||
load_user_skills: true,
|
||||
});
|
||||
expect(
|
||||
@@ -1063,11 +1091,16 @@ describe("buildStartConversationRequest — ACP discriminator", () => {
|
||||
expect(payload.agent_settings.llm).toBeUndefined();
|
||||
expect(payload.agent_settings.condenser).toBeUndefined();
|
||||
expect(payload.agent_settings.tools).toBeUndefined();
|
||||
expect(payload.agent_settings.agent_context).toEqual({
|
||||
load_public_skills: true,
|
||||
const acpAgentContext = payload.agent_settings.agent_context as Record<
|
||||
string,
|
||||
unknown
|
||||
>;
|
||||
expect(acpAgentContext).toMatchObject({
|
||||
load_public_skills: false,
|
||||
load_user_skills: true,
|
||||
load_project_skills: true,
|
||||
});
|
||||
expect(Array.isArray(acpAgentContext.skills)).toBe(true);
|
||||
expect(payload.tags).toEqual({ [ACP_SERVER_TAG_KEY]: "claude-code" });
|
||||
});
|
||||
|
||||
|
||||
@@ -8,7 +8,6 @@ import {
|
||||
getAgentServerWorkingDir,
|
||||
isAuthRequired,
|
||||
isAuthRequiredAndMissing,
|
||||
shouldLoadPublicSkills,
|
||||
} from "#/api/agent-server-config";
|
||||
|
||||
const ORIGINAL_LOCATION = window.location;
|
||||
@@ -75,23 +74,6 @@ describe("agent server config", () => {
|
||||
).toBe("/srv/workspaces/4a8dca373bf048dea0af949d711c3d48");
|
||||
});
|
||||
|
||||
it("loads public skills by default when VITE_LOAD_PUBLIC_SKILLS is unset", () => {
|
||||
vi.stubEnv("VITE_LOAD_PUBLIC_SKILLS", "");
|
||||
|
||||
expect(shouldLoadPublicSkills()).toBe(true);
|
||||
});
|
||||
|
||||
it("loads public skills when VITE_LOAD_PUBLIC_SKILLS is explicitly 'true'", () => {
|
||||
vi.stubEnv("VITE_LOAD_PUBLIC_SKILLS", "true");
|
||||
|
||||
expect(shouldLoadPublicSkills()).toBe(true);
|
||||
});
|
||||
|
||||
it("does not load public skills only when VITE_LOAD_PUBLIC_SKILLS is explicitly 'false'", () => {
|
||||
vi.stubEnv("VITE_LOAD_PUBLIC_SKILLS", "false");
|
||||
|
||||
expect(shouldLoadPublicSkills()).toBe(false);
|
||||
});
|
||||
});
|
||||
|
||||
describe("isAuthRequired", () => {
|
||||
|
||||
@@ -6,10 +6,24 @@ import {
|
||||
setRegisteredBackends,
|
||||
} from "#/api/backend-registry/active-store";
|
||||
import type { Backend } from "#/api/backend-registry/types";
|
||||
import SkillsService from "#/api/skills-service";
|
||||
|
||||
const { mockGetSkills } = vi.hoisted(() => ({
|
||||
const { mockGetSkills, MOCK_PUBLIC_CATALOG } = vi.hoisted(() => ({
|
||||
mockGetSkills: vi.fn(),
|
||||
MOCK_PUBLIC_CATALOG: [
|
||||
{
|
||||
name: "mock-public-skill",
|
||||
description: "A mock public skill",
|
||||
triggers: ["mock"],
|
||||
content: "mock content",
|
||||
},
|
||||
{
|
||||
name: "another-public-skill",
|
||||
description: "Another one",
|
||||
triggers: [],
|
||||
content: "more content",
|
||||
license: "MIT",
|
||||
},
|
||||
],
|
||||
}));
|
||||
|
||||
vi.mock("@openhands/typescript-client/clients", () => ({
|
||||
@@ -18,6 +32,12 @@ vi.mock("@openhands/typescript-client/clients", () => ({
|
||||
}),
|
||||
}));
|
||||
|
||||
vi.mock("@openhands/extensions/skills", () => ({
|
||||
SKILLS_CATALOG: MOCK_PUBLIC_CATALOG,
|
||||
}));
|
||||
|
||||
import SkillsService from "#/api/skills-service";
|
||||
|
||||
const localBackend: Backend = {
|
||||
id: "local",
|
||||
name: "Local",
|
||||
@@ -41,37 +61,49 @@ afterEach(() => {
|
||||
});
|
||||
|
||||
describe("SkillsService.getSkills against the agent-server backend", () => {
|
||||
it("requests load_public:true for the global Skills page even when VITE_LOAD_PUBLIC_SKILLS is unset, so a fresh dev env still shows the public catalog", async () => {
|
||||
// Arrange: the dev-default scenario that shipped the empty Skills page —
|
||||
// VITE_LOAD_PUBLIC_SKILLS is not set, so shouldLoadPublicSkills() would
|
||||
// return false. The agent-server has one public skill it can return.
|
||||
vi.stubEnv("VITE_LOAD_PUBLIC_SKILLS", "");
|
||||
it("requests only user/project skills from agent-server (load_public: false) and appends the bundled public catalog", async () => {
|
||||
const userSkill = {
|
||||
name: "my-custom-skill",
|
||||
type: "knowledge",
|
||||
content: "custom content",
|
||||
triggers: [],
|
||||
source: "user",
|
||||
is_agentskills_format: false,
|
||||
};
|
||||
mockGetSkills.mockResolvedValue({
|
||||
skills: [
|
||||
{
|
||||
name: "alpha",
|
||||
type: "knowledge",
|
||||
content: "...",
|
||||
triggers: [],
|
||||
source: "public",
|
||||
is_agentskills_format: false,
|
||||
},
|
||||
],
|
||||
sources: { sandbox: 0, sdk_base: 1, org: 0, project: 0 },
|
||||
skills: [userSkill],
|
||||
sources: { sandbox: 0, sdk_base: 0, org: 0, project: 0 },
|
||||
});
|
||||
|
||||
// Act
|
||||
const skills = await SkillsService.getSkills();
|
||||
|
||||
// Assert: the request opts the user into public skills regardless of the
|
||||
// perf-oriented VITE_LOAD_PUBLIC_SKILLS gate, and the page receives them.
|
||||
// Agent-server is asked only for user/project skills, not public.
|
||||
expect(mockGetSkills).toHaveBeenCalledTimes(1);
|
||||
expect(mockGetSkills.mock.calls[0]?.[0]).toMatchObject({
|
||||
load_public: true,
|
||||
load_public: false,
|
||||
load_user: true,
|
||||
load_project: true,
|
||||
load_org: false,
|
||||
});
|
||||
expect(skills.map((s) => s.name)).toEqual(["alpha"]);
|
||||
|
||||
// Result = local skills first, then all bundled public skills.
|
||||
expect(skills[0]?.name).toBe("my-custom-skill");
|
||||
expect(skills).toHaveLength(1 + MOCK_PUBLIC_CATALOG.length);
|
||||
|
||||
// Every public skill from the bundled catalog is present.
|
||||
const publicNames = skills.slice(1).map((s) => s.name);
|
||||
for (const entry of MOCK_PUBLIC_CATALOG) {
|
||||
expect(publicNames).toContain(entry.name);
|
||||
}
|
||||
expect(skills.slice(1).every((s) => s.source === "public")).toBe(true);
|
||||
});
|
||||
|
||||
it("returns only bundled public skills when agent-server is unreachable", async () => {
|
||||
mockGetSkills.mockRejectedValue(new Error("ECONNREFUSED"));
|
||||
|
||||
const skills = await SkillsService.getSkills();
|
||||
|
||||
expect(skills).toHaveLength(MOCK_PUBLIC_CATALOG.length);
|
||||
expect(skills.every((s) => s.source === "public")).toBe(true);
|
||||
});
|
||||
});
|
||||
|
||||
+1
-8
@@ -45,16 +45,10 @@ RUN npm run build
|
||||
|
||||
# ── Stage 1b: Generate shell-sourceable defaults from config/defaults.json ──
|
||||
# This avoids needing jq/python at container runtime to parse the JSON.
|
||||
# package.json is also read to extract the pinned @openhands/extensions SHA
|
||||
# and emit it as CONFIG_EXTENSIONS_REF so the agent-server uses the same
|
||||
# extensions commit as the bundled UI (via EXTENSIONS_REF in entrypoint.sh).
|
||||
FROM node:24-slim AS config-gen
|
||||
COPY config/defaults.json package.json /tmp/
|
||||
COPY config/defaults.json /tmp/
|
||||
RUN node -e " \
|
||||
const c = JSON.parse(require('fs').readFileSync('/tmp/defaults.json','utf-8')); \
|
||||
const pkg = JSON.parse(require('fs').readFileSync('/tmp/package.json','utf-8')); \
|
||||
const extUrl = (pkg.dependencies || pkg.devDependencies || {})['@openhands/extensions'] || ''; \
|
||||
const extSha = (extUrl.match(/#([0-9a-f]{40})\$/i) || [])[1] || ''; \
|
||||
const lines = [ \
|
||||
'CONFIG_AGENT_SERVER_PORT=' + c.ports.agentServer, \
|
||||
'CONFIG_AUTOMATION_PORT=' + c.ports.automation, \
|
||||
@@ -64,7 +58,6 @@ RUN node -e " \
|
||||
'CONFIG_BASH_EVENTS=' + c.paths.bashEvents, \
|
||||
'CONFIG_AUTOMATION_DB=' + c.paths.automationDb, \
|
||||
]; \
|
||||
if (extSha) lines.push('CONFIG_EXTENSIONS_REF=' + extSha); \
|
||||
require('fs').writeFileSync('/tmp/defaults.env', lines.join('\n') + '\n'); \
|
||||
"
|
||||
|
||||
|
||||
@@ -128,11 +128,6 @@ export AUTOMATION_AGENT_SERVER_URL="${AUTOMATION_AGENT_SERVER_URL:-http://127.0.
|
||||
# OH_EXTRA_PYTHON_PATH: config.canvasToolsDir.
|
||||
export OH_EXTRA_PYTHON_PATH="${OH_EXTRA_PYTHON_PATH:-/opt/agent-canvas/tools}"
|
||||
|
||||
# Pin the public-skills catalog to the same @openhands/extensions commit that
|
||||
# the frontend bundle was built with. The agent-server SDK skips network polling
|
||||
# when EXTENSIONS_REF is already present in its local cache.
|
||||
export EXTENSIONS_REF="${EXTENSIONS_REF:-${CONFIG_EXTENSIONS_REF:-}}"
|
||||
|
||||
# Track child PIDs so we can clean up on exit.
|
||||
PIDS=()
|
||||
|
||||
|
||||
Generated
+4
-33
@@ -12,7 +12,7 @@
|
||||
"@heroui/react": "2.8.10",
|
||||
"@microlink/react-json-view": "1.31.20",
|
||||
"@monaco-editor/react": "4.7.0",
|
||||
"@openhands/extensions": "github:OpenHands/extensions#8a6690071e0e9229ae70b30ff0352aff9e6fca21",
|
||||
"@openhands/extensions": "0.3.0",
|
||||
"@openhands/typescript-client": "1.24.3",
|
||||
"@react-router/node": "7.17.0",
|
||||
"@react-router/serve": "7.17.0",
|
||||
@@ -733,7 +733,6 @@
|
||||
"version": "1.6.0",
|
||||
"resolved": "https://registry.npmjs.org/@colors/colors/-/colors-1.6.0.tgz",
|
||||
"integrity": "sha512-Ir+AOibqzrIsL6ajt3Rz3LskB7OiMVHqltZmspbW/TJuTVuyOMirVqAkjfY6JISiLHgyNqicAC8AyHHGzNd/dA==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"engines": {
|
||||
"node": ">=0.1.90"
|
||||
@@ -883,7 +882,6 @@
|
||||
"version": "2.0.8",
|
||||
"resolved": "https://registry.npmjs.org/@dabh/diagnostics/-/diagnostics-2.0.8.tgz",
|
||||
"integrity": "sha512-R4MSXTVnuMzGD7bzHdW2ZhhdPC/igELENcq5IjEverBvq5hn1SXCWcsi6eSsdWP0/Ur+SItRRjAktmdoX/8R/Q==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"@so-ric/colorspace": "^1.1.6",
|
||||
@@ -3470,9 +3468,9 @@
|
||||
"license": "MIT"
|
||||
},
|
||||
"node_modules/@openhands/extensions": {
|
||||
"version": "0.0.0",
|
||||
"resolved": "git+ssh://git@github.com/OpenHands/extensions.git#8a6690071e0e9229ae70b30ff0352aff9e6fca21",
|
||||
"integrity": "sha512-GjqvmlDWOWnWRtvqM65B3xWKphhBD3nuckOcWKZsrt2TithPg5YTptbj5mdcbm2MWV9gIk5uo8wknjGUIR994g==",
|
||||
"version": "0.3.0",
|
||||
"resolved": "https://registry.npmjs.org/@openhands/extensions/-/extensions-0.3.0.tgz",
|
||||
"integrity": "sha512-6xewbbmrDG6GbnI/Gga+o3+cw1jUpPkEW4wJKtL1mncL9f1qpI93YOBQMlrdQnu0JMRnLAMYSeURksqToUw2rw==",
|
||||
"license": "MIT",
|
||||
"engines": {
|
||||
"node": ">=18.20.0"
|
||||
@@ -6211,7 +6209,6 @@
|
||||
"version": "1.1.6",
|
||||
"resolved": "https://registry.npmjs.org/@so-ric/colorspace/-/colorspace-1.1.6.tgz",
|
||||
"integrity": "sha512-/KiKkpHNOBgkFJwu9sh48LkHSMYGyuTcSFK/qMBdnOAlrRJzRSXAOFB5qwzaVQuDl8wAvHVMkaASQDReTahxuw==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"color": "^5.0.2",
|
||||
@@ -6222,7 +6219,6 @@
|
||||
"version": "5.0.3",
|
||||
"resolved": "https://registry.npmjs.org/color/-/color-5.0.3.tgz",
|
||||
"integrity": "sha512-ezmVcLR3xAVp8kYOm4GS45ZLLgIE6SPAFoduLr6hTDajwb3KZ2F46gulK3XpcwRFb5KKGCSezCBAY4Dw4HsyXA==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"color-convert": "^3.1.3",
|
||||
@@ -6236,7 +6232,6 @@
|
||||
"version": "3.1.3",
|
||||
"resolved": "https://registry.npmjs.org/color-convert/-/color-convert-3.1.3.tgz",
|
||||
"integrity": "sha512-fasDH2ont2GqF5HpyO4w0+BcewlhHEZOFn9c1ckZdHpJ56Qb7MHhH/IcJZbBGgvdtwdwNbLvxiBEdg336iA9Sg==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"color-name": "^2.0.0"
|
||||
@@ -6249,7 +6244,6 @@
|
||||
"version": "2.1.0",
|
||||
"resolved": "https://registry.npmjs.org/color-name/-/color-name-2.1.0.tgz",
|
||||
"integrity": "sha512-1bPaDNFm0axzE4MEAzKPuqKWeRaT43U/hyxKPBdqTfmPF+d6n7FSoTFxLVULUJOmiLp01KjhIPPH+HrXZJN4Rg==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"engines": {
|
||||
"node": ">=12.20"
|
||||
@@ -6259,7 +6253,6 @@
|
||||
"version": "2.1.4",
|
||||
"resolved": "https://registry.npmjs.org/color-string/-/color-string-2.1.4.tgz",
|
||||
"integrity": "sha512-Bb6Cq8oq0IjDOe8wJmi4JeNn763Xs9cfrBcaylK1tPypWzyoy2G3l90v9k64kjphl/ZJjPIShFztenRomi8WTg==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"color-name": "^2.0.0"
|
||||
@@ -7174,7 +7167,6 @@
|
||||
"version": "1.3.5",
|
||||
"resolved": "https://registry.npmjs.org/@types/triple-beam/-/triple-beam-1.3.5.tgz",
|
||||
"integrity": "sha512-6WaYesThRMCl19iryMYP7/x2OVgCtbIVflDGFpWnb9irXI3UjYE4AzmYuiUKY1AJstGijoY+MgUszMgRxIYTYw==",
|
||||
"dev": true,
|
||||
"license": "MIT"
|
||||
},
|
||||
"node_modules/@types/trusted-types": {
|
||||
@@ -8572,7 +8564,6 @@
|
||||
"version": "3.2.6",
|
||||
"resolved": "https://registry.npmjs.org/async/-/async-3.2.6.tgz",
|
||||
"integrity": "sha512-htCUDlxyyCLMgaM3xXg0C0LW2xqfuQ6p05pCEIsXuyQ+a1koYKTuBMzRNwmybfLgvJDMd0r1LTn4+E0Ti6C2AA==",
|
||||
"dev": true,
|
||||
"license": "MIT"
|
||||
},
|
||||
"node_modules/async-function": {
|
||||
@@ -9836,7 +9827,6 @@
|
||||
"version": "2.0.0",
|
||||
"resolved": "https://registry.npmjs.org/enabled/-/enabled-2.0.0.tgz",
|
||||
"integrity": "sha512-AKrN98kuwOzMIdAizXGI86UFBoo26CL21UM763y1h/GMSJ4/OHU9k2YlsmBpyScFo/wbLzWQJBMCW4+IO3/+OQ==",
|
||||
"dev": true,
|
||||
"license": "MIT"
|
||||
},
|
||||
"node_modules/encodeurl": {
|
||||
@@ -11011,7 +11001,6 @@
|
||||
"version": "4.2.3",
|
||||
"resolved": "https://registry.npmjs.org/fecha/-/fecha-4.2.3.tgz",
|
||||
"integrity": "sha512-OP2IUU6HeYKJi3i0z4A19kHMQoLVs4Hc+DPqqxI2h/DPZHTm/vjsfC6P0b4jCMy14XizLBqvndQ+UilD7707Jw==",
|
||||
"dev": true,
|
||||
"license": "MIT"
|
||||
},
|
||||
"node_modules/fflate": {
|
||||
@@ -11037,7 +11026,6 @@
|
||||
"version": "0.6.1",
|
||||
"resolved": "https://registry.npmjs.org/file-stream-rotator/-/file-stream-rotator-0.6.1.tgz",
|
||||
"integrity": "sha512-u+dBid4PvZw17PmDeRcNOtCP9CCK/9lRN2w+r1xIS7yOL9JFrIBKTvrYsxT4P0pGtThYTn++QS5ChHaUov3+zQ==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"moment": "^2.29.1"
|
||||
@@ -11131,7 +11119,6 @@
|
||||
"version": "1.1.0",
|
||||
"resolved": "https://registry.npmjs.org/fn.name/-/fn.name-1.1.0.tgz",
|
||||
"integrity": "sha512-GRnmB5gPyJpAhTQdSZTSp9uaPSvl09KoYcMQtsB9rQoOmzs9dH6ffeccH+Z+cv6P68Hu5bC6JjRh4Ah/mHSNRw==",
|
||||
"dev": true,
|
||||
"license": "MIT"
|
||||
},
|
||||
"node_modules/follow-redirects": {
|
||||
@@ -12453,7 +12440,6 @@
|
||||
"version": "2.0.1",
|
||||
"resolved": "https://registry.npmjs.org/is-stream/-/is-stream-2.0.1.tgz",
|
||||
"integrity": "sha512-hFoiJiTl63nn+kstHGBtewWSKnQLpyb155KHheA1l39uvtO9nWIop1p3udqPcUd/xbF1VLMO4n7OI6p7RbngDg==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"engines": {
|
||||
"node": ">=8"
|
||||
@@ -12832,7 +12818,6 @@
|
||||
"version": "2.0.0",
|
||||
"resolved": "https://registry.npmjs.org/kuler/-/kuler-2.0.0.tgz",
|
||||
"integrity": "sha512-Xq9nH7KlWZmXAtodXDDRE7vs6DU1gTU8zYDHDiWLSip45Egwq3plLHzPn27NgvzL2r1LMPC1vdqh98sQxtqj4A==",
|
||||
"dev": true,
|
||||
"license": "MIT"
|
||||
},
|
||||
"node_modules/language-subtag-registry": {
|
||||
@@ -13266,7 +13251,6 @@
|
||||
"version": "2.7.0",
|
||||
"resolved": "https://registry.npmjs.org/logform/-/logform-2.7.0.tgz",
|
||||
"integrity": "sha512-TFYA4jnP7PVbmlBIfhlSe+WKxs9dklXMTEGcBCIvLhE/Tn3H6Gk1norupVW7m5Cnd4bLcr08AytbyV/xj7f/kQ==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"@colors/colors": "1.6.0",
|
||||
@@ -14454,7 +14438,6 @@
|
||||
"version": "2.30.1",
|
||||
"resolved": "https://registry.npmjs.org/moment/-/moment-2.30.1.tgz",
|
||||
"integrity": "sha512-uEmtNhbDOrWPFS+hdjFCBfy9f2YoyzRpwcl+DqpC6taX21FzsTLQVbMV/W7PzNSX6x/bhC1zA3c2UQ5NzH6how==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"engines": {
|
||||
"node": "*"
|
||||
@@ -14765,7 +14748,6 @@
|
||||
"version": "3.0.0",
|
||||
"resolved": "https://registry.npmjs.org/object-hash/-/object-hash-3.0.0.tgz",
|
||||
"integrity": "sha512-RSn9F68PjH9HqtltsSnqYC1XXoWe9Bju5+213R98cNGttag9q9yAOTzdbsqvIa7aNm5WffBZFpWYr2aWrklWAw==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"engines": {
|
||||
"node": ">= 6"
|
||||
@@ -14907,7 +14889,6 @@
|
||||
"version": "1.0.0",
|
||||
"resolved": "https://registry.npmjs.org/one-time/-/one-time-1.0.0.tgz",
|
||||
"integrity": "sha512-5DXOiRKwuSEcQ/l0kGCF6Q3jcADFv5tSmRaJck/OqkVFcOzutB134KRSfF0xDrL39MNnqxbHBbUUcjZIhTgb2g==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"fn.name": "1.x.x"
|
||||
@@ -15903,7 +15884,6 @@
|
||||
"version": "3.6.2",
|
||||
"resolved": "https://registry.npmjs.org/readable-stream/-/readable-stream-3.6.2.tgz",
|
||||
"integrity": "sha512-9u/sniCrY3D5WdsERHzHE4G2YCXqoG5FTHUiCC4SIbr6XcLZBY05ya9EKjYek9O5xOAwjGq+1JdGBAS7Q9ScoA==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"inherits": "^2.0.3",
|
||||
@@ -16421,7 +16401,6 @@
|
||||
"version": "2.5.0",
|
||||
"resolved": "https://registry.npmjs.org/safe-stable-stringify/-/safe-stable-stringify-2.5.0.tgz",
|
||||
"integrity": "sha512-b3rppTKm9T+PsVCBEOUR46GWI7fdOs00VKZ1+9c1EWDaDMvjQc6tUwuFyIprgGgTcWoVHSKrU8H31ZHA2e0RHA==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"engines": {
|
||||
"node": ">=10"
|
||||
@@ -16892,7 +16871,6 @@
|
||||
"version": "0.0.10",
|
||||
"resolved": "https://registry.npmjs.org/stack-trace/-/stack-trace-0.0.10.tgz",
|
||||
"integrity": "sha512-KGzahc7puUKkzyMt+IqAep+TVNbKP+k2Lmwhub39m1AsTSkaDutx56aDCo+HLDzf/D26BIHTJWNiTG1KAJiQCg==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"engines": {
|
||||
"node": "*"
|
||||
@@ -16952,7 +16930,6 @@
|
||||
"version": "1.3.0",
|
||||
"resolved": "https://registry.npmjs.org/string_decoder/-/string_decoder-1.3.0.tgz",
|
||||
"integrity": "sha512-hkRX8U1WjJFd8LsDJ2yQ/wWWxaopEsABU1XfkM8A+j0+85JAGppt16cr1Whg6KIbb4okU6Mql6BOj+uup/wKeA==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"safe-buffer": "~5.2.0"
|
||||
@@ -17321,7 +17298,6 @@
|
||||
"version": "1.0.0",
|
||||
"resolved": "https://registry.npmjs.org/text-hex/-/text-hex-1.0.0.tgz",
|
||||
"integrity": "sha512-uuVGNWzgJ4yhRaNSiubPY7OjISw4sw4E5Uv0wbjp+OzcbmVU/rsT8ujgcXJhn9ypzsgr5vlzpPqP+MBBKcGvbg==",
|
||||
"dev": true,
|
||||
"license": "MIT"
|
||||
},
|
||||
"node_modules/tinybench": {
|
||||
@@ -17467,7 +17443,6 @@
|
||||
"version": "1.4.1",
|
||||
"resolved": "https://registry.npmjs.org/triple-beam/-/triple-beam-1.4.1.tgz",
|
||||
"integrity": "sha512-aZbgViZrg1QNcG+LULa7nhZpJTZSLm/mXnHXnbAbjmN5aSa0y7V+wvv6+4WaBtpISJzThKy+PIPxc1Nq1EJ9mg==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"engines": {
|
||||
"node": ">= 14.0.0"
|
||||
@@ -17932,7 +17907,6 @@
|
||||
"version": "1.0.2",
|
||||
"resolved": "https://registry.npmjs.org/util-deprecate/-/util-deprecate-1.0.2.tgz",
|
||||
"integrity": "sha512-EPD5q1uXyFxJpCrLnCc1nHnq3gOa6DZBocAIiI2TaSCA7VCJ1UJDMagCzIkXNsUYfD1daK//LTEQ8xiIbrHtcw==",
|
||||
"dev": true,
|
||||
"license": "MIT"
|
||||
},
|
||||
"node_modules/utils-merge": {
|
||||
@@ -18552,7 +18526,6 @@
|
||||
"version": "3.19.0",
|
||||
"resolved": "https://registry.npmjs.org/winston/-/winston-3.19.0.tgz",
|
||||
"integrity": "sha512-LZNJgPzfKR+/J3cHkxcpHKpKKvGfDZVPS4hfJCc4cCG0CgYzvlD6yE/S3CIL/Yt91ak327YCpiF/0MyeZHEHKA==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"@colors/colors": "^1.6.0",
|
||||
@@ -18575,7 +18548,6 @@
|
||||
"version": "5.0.0",
|
||||
"resolved": "https://registry.npmjs.org/winston-daily-rotate-file/-/winston-daily-rotate-file-5.0.0.tgz",
|
||||
"integrity": "sha512-JDjiXXkM5qvwY06733vf09I2wnMXpZEhxEVOSPenZMii+g7pcDcTBt2MRugnoi8BwVSuCT2jfRXBUy+n1Zz/Yw==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"file-stream-rotator": "^0.6.1",
|
||||
@@ -18594,7 +18566,6 @@
|
||||
"version": "4.9.0",
|
||||
"resolved": "https://registry.npmjs.org/winston-transport/-/winston-transport-4.9.0.tgz",
|
||||
"integrity": "sha512-8drMJ4rkgaPo1Me4zD/3WLfI/zPdA9o2IipKODunnGDcuqbHwjsbB79ylv04LCGGzU0xQ6vTznOMpQGaLhhm6A==",
|
||||
"dev": true,
|
||||
"license": "MIT",
|
||||
"dependencies": {
|
||||
"logform": "^2.7.0",
|
||||
|
||||
+1
-1
@@ -23,7 +23,7 @@
|
||||
"@heroui/react": "2.8.10",
|
||||
"@microlink/react-json-view": "1.31.20",
|
||||
"@monaco-editor/react": "4.7.0",
|
||||
"@openhands/extensions": "github:OpenHands/extensions#8a6690071e0e9229ae70b30ff0352aff9e6fca21",
|
||||
"@openhands/extensions": "0.3.0",
|
||||
"@openhands/typescript-client": "1.24.3",
|
||||
"@react-router/node": "7.17.0",
|
||||
"@react-router/serve": "7.17.0",
|
||||
|
||||
@@ -29,6 +29,7 @@
|
||||
|
||||
import { defineConfig, devices } from "@playwright/test";
|
||||
import { randomBytes } from "node:crypto";
|
||||
import { mkdirSync } from "node:fs";
|
||||
import { resolve } from "node:path";
|
||||
|
||||
// ── Docker image ────────────────────────────────────────────────────────
|
||||
@@ -86,6 +87,23 @@ if (!process.env.MOCK_LLM_AGENT_URL) {
|
||||
process.env.MOCK_LLM_AGENT_URL = MOCK_LLM_URL;
|
||||
}
|
||||
|
||||
// ── Skill test support ────────────────────────────────────────────────
|
||||
// The skill tests create git repos and user skill files on the host.
|
||||
// We volume-mount these into the container so the agent-server can see them.
|
||||
|
||||
// Project skills: host creates repos here, container sees them at a fixed path.
|
||||
const SKILL_REPOS_HOST_DIR = resolve(".tmp/mock-llm-skill-repos");
|
||||
const SKILL_REPOS_CONTAINER_DIR = "/tmp/mock-llm-skill-repos";
|
||||
mkdirSync(SKILL_REPOS_HOST_DIR, { recursive: true });
|
||||
process.env.MOCK_LLM_SKILL_REPOS_CONTAINER_DIR = SKILL_REPOS_CONTAINER_DIR;
|
||||
|
||||
// User skills: host creates skill files here, container mounts them at
|
||||
// the agent-server's expected ~/.openhands/skills/ path.
|
||||
const USER_SKILLS_HOST_DIR = resolve(".tmp/mock-llm-user-skills");
|
||||
const USER_SKILLS_CONTAINER_DIR = "/home/openhands/.openhands/skills";
|
||||
mkdirSync(USER_SKILLS_HOST_DIR, { recursive: true });
|
||||
process.env.MOCK_LLM_USER_SKILLS_HOST_DIR = USER_SKILLS_HOST_DIR;
|
||||
|
||||
// ── ACP test support ──────────────────────────────────────────────────
|
||||
// The mock ACP server script lives on the host. We volume-mount it into
|
||||
// the container and tell the test which container-side paths to use when
|
||||
@@ -169,6 +187,10 @@ export default defineConfig({
|
||||
// Mount the mock ACP server script so the agent-server inside
|
||||
// Docker can spawn it as an ACP subprocess.
|
||||
`-v ${MOCK_ACP_HOST_PATH}:${MOCK_ACP_CONTAINER_PATH}:ro`,
|
||||
// Mount skill test directories so the agent-server can access
|
||||
// repos and user skills created by the host-side test code.
|
||||
`-v ${SKILL_REPOS_HOST_DIR}:${SKILL_REPOS_CONTAINER_DIR}`,
|
||||
`-v ${USER_SKILLS_HOST_DIR}:${USER_SKILLS_CONTAINER_DIR}`,
|
||||
`-e PORT=${INGRESS_PORT}`,
|
||||
`-e SESSION_API_KEY=${sessionApiKey}`,
|
||||
`-e OH_SESSION_API_KEYS_0=${sessionApiKey}`,
|
||||
|
||||
@@ -36,32 +36,6 @@ const SHARED_DEFAULTS = JSON.parse(
|
||||
),
|
||||
);
|
||||
|
||||
/**
|
||||
* Extract the pinned commit SHA for @openhands/extensions from package.json.
|
||||
* Returns the 40-char hex SHA when the dependency is a git+https URL with a
|
||||
* commit hash fragment (e.g. "git+https://…#62594156…"), null otherwise.
|
||||
* @returns {string | null}
|
||||
*/
|
||||
function getExtensionsRef() {
|
||||
try {
|
||||
const pkg = JSON.parse(
|
||||
readFileSync(
|
||||
path.join(__dev_safe_dirname, "..", "package.json"),
|
||||
"utf-8",
|
||||
),
|
||||
);
|
||||
const url =
|
||||
(pkg.dependencies ?? pkg.devDependencies ?? {})["@openhands/extensions"] ??
|
||||
"";
|
||||
return url.match(/#([0-9a-f]{40})$/i)?.[1] ?? null;
|
||||
} catch {
|
||||
return null;
|
||||
}
|
||||
}
|
||||
|
||||
/** Pinned extensions commit SHA derived from package.json, or null if not pinned. */
|
||||
export const DEFAULT_EXTENSIONS_REF = getExtensionsRef();
|
||||
|
||||
const DEFAULT_BACKEND_PORT = SHARED_DEFAULTS.ports.agentServer;
|
||||
const DEFAULT_VITE_PORT = 3001;
|
||||
const DEFAULT_WAIT_TIMEOUT_MS = 30_000;
|
||||
@@ -737,13 +711,6 @@ export function buildAgentServerEnv(config) {
|
||||
// Make the host tools/ directory importable so the agent-server can
|
||||
// resolve modules listed in tool_module_qualnames (e.g. canvas_ui_tool).
|
||||
OH_EXTRA_PYTHON_PATH: config.canvasToolsDir,
|
||||
// Tell the agent-server which extensions commit to use for the public
|
||||
// skills catalog. Derived from the @openhands/extensions pin in
|
||||
// package.json; the SDK skips network polling when it already has this
|
||||
// SHA cached. Only injected when the caller has not already set it.
|
||||
...(DEFAULT_EXTENSIONS_REF && !process.env.EXTENSIONS_REF
|
||||
? { EXTENSIONS_REF: DEFAULT_EXTENSIONS_REF }
|
||||
: {}),
|
||||
};
|
||||
}
|
||||
|
||||
|
||||
@@ -1,4 +1,5 @@
|
||||
import { ACP_SETTINGS_KEYS } from "@openhands/typescript-client";
|
||||
import { SKILLS_CATALOG } from "@openhands/extensions/skills";
|
||||
import { DEFAULT_SETTINGS } from "#/services/settings";
|
||||
import { ExecutionStatus } from "#/types/agent-server/core";
|
||||
import { Settings, SettingsValue } from "#/types/settings";
|
||||
@@ -8,10 +9,7 @@ import {
|
||||
} from "#/constants/acp-providers";
|
||||
import { getAgentServerClientOptions } from "./agent-server-client-options";
|
||||
import { isAgentServerToolAvailable } from "./agent-server-compatibility";
|
||||
import {
|
||||
getAgentServerWorkingDir,
|
||||
shouldLoadPublicSkills,
|
||||
} from "./agent-server-config";
|
||||
import { getAgentServerWorkingDir } from "./agent-server-config";
|
||||
import { getEffectiveLocalBackend } from "./backend-registry/active-store";
|
||||
import { buildAuthHeaders } from "./backend-registry/auth";
|
||||
import {
|
||||
@@ -527,11 +525,75 @@ function buildInitialMessage(
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Shape of a bundled skill entry passed to the agent-server SDK via
|
||||
* `agent_context.skills`. Mirrors the SDK's `Skill` model fields that
|
||||
* the server uses for trigger matching, activation, and system-prompt
|
||||
* injection.
|
||||
*/
|
||||
interface BundledSkill {
|
||||
name: string;
|
||||
content: string;
|
||||
trigger: { type: "keyword"; keywords: string[] } | null;
|
||||
source: "public";
|
||||
description: string | null;
|
||||
is_agentskills_format: true;
|
||||
license?: string;
|
||||
compatibility?: string;
|
||||
}
|
||||
|
||||
/**
|
||||
* Convert the bundled `SKILLS_CATALOG` entries into the SDK `Skill` JSON
|
||||
* shape so the agent-server can perform trigger matching, skill activation,
|
||||
* and system-prompt injection without cloning the extensions repo.
|
||||
*
|
||||
* The SDK discriminates triggers via `{ type: "keyword", keywords: [...] }`.
|
||||
* Skills with no triggers get `trigger: null` (always-active / on-demand).
|
||||
*/
|
||||
function buildBundledSkills(): BundledSkill[] {
|
||||
return SKILLS_CATALOG.map((entry) => {
|
||||
const trigger: BundledSkill["trigger"] =
|
||||
entry.triggers?.length > 0
|
||||
? { type: "keyword", keywords: entry.triggers }
|
||||
: null;
|
||||
|
||||
return {
|
||||
name: entry.name,
|
||||
content: entry.content,
|
||||
trigger,
|
||||
source: "public" as const,
|
||||
description: entry.description ?? null,
|
||||
is_agentskills_format: true as const,
|
||||
...(entry.license ? { license: entry.license } : {}),
|
||||
...(entry.compatibility ? { compatibility: entry.compatibility } : {}),
|
||||
};
|
||||
});
|
||||
}
|
||||
|
||||
function buildAgentContext(agentSettings: SettingsRecord): SettingsRecord {
|
||||
const runtimeServicesSuffix = buildRuntimeServicesSystemSuffix();
|
||||
const existingContext = toRecord(agentSettings.agent_context);
|
||||
|
||||
// Merge bundled public skills with any skills already present in the
|
||||
// agent context (e.g. user-defined skills set via the settings API).
|
||||
const existingSkills = Array.isArray(existingContext.skills)
|
||||
? (existingContext.skills as SettingsRecord[])
|
||||
: [];
|
||||
const mergedSkills = [...existingSkills, ...buildBundledSkills()];
|
||||
|
||||
return {
|
||||
...toRecord(agentSettings.agent_context),
|
||||
load_public_skills: shouldLoadPublicSkills(),
|
||||
...existingContext,
|
||||
// Public skills are bundled at build time from the @openhands/extensions
|
||||
// npm package and passed directly in agent_context.skills. Setting
|
||||
// load_public_skills to false tells the agent-server SDK to skip its own
|
||||
// extensions-repo clone — the frontend is the sole source of public
|
||||
// skills now.
|
||||
//
|
||||
// Migration: the former VITE_LOAD_PUBLIC_SKILLS env var was removed
|
||||
// because bundled skills have no clone latency. Users who previously set
|
||||
// VITE_LOAD_PUBLIC_SKILLS=false to avoid clone delays no longer need it.
|
||||
skills: mergedSkills,
|
||||
load_public_skills: false,
|
||||
load_user_skills: true,
|
||||
load_project_skills: true,
|
||||
...(runtimeServicesSuffix
|
||||
|
||||
@@ -112,10 +112,6 @@ export function getAgentServerHeaders(): Record<string, string> {
|
||||
return sessionApiKey ? { "X-Session-API-Key": sessionApiKey } : {};
|
||||
}
|
||||
|
||||
export function shouldLoadPublicSkills(): boolean {
|
||||
return import.meta.env.VITE_LOAD_PUBLIC_SKILLS !== "false";
|
||||
}
|
||||
|
||||
export function isAuthRequired(): boolean {
|
||||
return (
|
||||
import.meta.env.VITE_AUTH_REQUIRED === "true" ||
|
||||
|
||||
+48
-14
@@ -1,31 +1,65 @@
|
||||
import { SkillsClient } from "@openhands/typescript-client/clients";
|
||||
import {
|
||||
SKILLS_CATALOG,
|
||||
type SkillCatalogEntry,
|
||||
} from "@openhands/extensions/skills";
|
||||
import { SkillInfo } from "#/types/settings";
|
||||
import { getAgentServerWorkingDir } from "./agent-server-config";
|
||||
import { getActiveBackend } from "./backend-registry/active-store";
|
||||
import { fetchCloudSkills } from "./cloud/skills-service.api";
|
||||
import { getAgentServerClientOptions } from "./agent-server-client-options";
|
||||
|
||||
function catalogEntryToSkillInfo(entry: SkillCatalogEntry): SkillInfo {
|
||||
return {
|
||||
name: entry.name,
|
||||
type: "knowledge",
|
||||
source: "public",
|
||||
description: entry.description,
|
||||
triggers: entry.triggers,
|
||||
content: entry.content,
|
||||
license: entry.license ?? null,
|
||||
compatibility: entry.compatibility ?? null,
|
||||
};
|
||||
}
|
||||
|
||||
/**
|
||||
* Public skills loaded from the `@openhands/extensions` npm package.
|
||||
*
|
||||
* This is an **immutable build-time snapshot**: the catalog is baked into the
|
||||
* bundle at `npm run build` / `vite build` time and does not change at
|
||||
* runtime. Updating the catalog requires bumping the `@openhands/extensions`
|
||||
* dependency and rebuilding.
|
||||
*/
|
||||
const PUBLIC_SKILLS: SkillInfo[] = SKILLS_CATALOG.map(catalogEntryToSkillInfo);
|
||||
|
||||
class SkillsService {
|
||||
static async getSkills(projectDir?: string): Promise<SkillInfo[]> {
|
||||
if (getActiveBackend().backend.kind === "cloud") {
|
||||
return fetchCloudSkills();
|
||||
}
|
||||
|
||||
// Always load public skills on the global Skills settings page so the user
|
||||
// sees the available catalog even on a fresh dev environment with no local
|
||||
// user/project skills. Conversation creation paths still gate on
|
||||
// shouldLoadPublicSkills() to keep new-conversation latency low.
|
||||
const response = await new SkillsClient(
|
||||
getAgentServerClientOptions(),
|
||||
).getSkills({
|
||||
load_public: true,
|
||||
load_user: true,
|
||||
load_project: true,
|
||||
load_org: false,
|
||||
project_dir: projectDir ?? getAgentServerWorkingDir(),
|
||||
});
|
||||
// Public skills come from the bundled @openhands/extensions npm package —
|
||||
// no agent-server round-trip or GitHub fetch needed. Only ask the agent-
|
||||
// server for user and project skills so local .agents/skills/ content is
|
||||
// still picked up.
|
||||
let localSkills: SkillInfo[] = [];
|
||||
try {
|
||||
const response = await new SkillsClient(
|
||||
getAgentServerClientOptions(),
|
||||
).getSkills({
|
||||
load_public: false,
|
||||
load_user: true,
|
||||
load_project: true,
|
||||
load_org: false,
|
||||
project_dir: projectDir ?? getAgentServerWorkingDir(),
|
||||
});
|
||||
localSkills = (response.skills ?? []) as SkillInfo[];
|
||||
} catch {
|
||||
// Agent-server may not support the skills endpoint or may be
|
||||
// unreachable; fall back to the bundled public catalog alone.
|
||||
}
|
||||
|
||||
return (response.skills ?? []) as SkillInfo[];
|
||||
return [...localSkills, ...PUBLIC_SKILLS];
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -254,8 +254,8 @@ test.describe("mock-LLM automation lifecycle", () => {
|
||||
].join(" ");
|
||||
|
||||
// ⚠️ Padding response (index 0):
|
||||
// When public skills are loaded (VITE_LOAD_PUBLIC_SKILLS !== "false"),
|
||||
// the agent-server's skill-activation pipeline makes one internal LLM
|
||||
// Public skills are bundled from @openhands/extensions at build time.
|
||||
// The agent-server's skill-activation pipeline makes one internal LLM
|
||||
// call to decide which skills to inject before the agent loop starts.
|
||||
// Our user message mentions "automation", which matches the
|
||||
// openhands-automation skill, triggering this internal call.
|
||||
|
||||
@@ -91,8 +91,8 @@ test.describe("mock-LLM image upload", () => {
|
||||
// called and the conversation completed successfully.
|
||||
//
|
||||
// ⚠️ Padding note (mirrors the automation test's pattern):
|
||||
// When public skills are loaded (VITE_LOAD_PUBLIC_SKILLS !== "false"),
|
||||
// the agent-server may make one internal LLM call for skill-analysis
|
||||
// Public skills are bundled from @openhands/extensions at build time.
|
||||
// The agent-server may make one internal LLM call for skill-analysis
|
||||
// before the agent loop starts, consuming one trajectory slot.
|
||||
// Turn 0 is a throwaway empty response that absorbs this internal call.
|
||||
// Turn 1 is the agent's actual reply (IMAGE_REPLY_TOKEN).
|
||||
|
||||
@@ -2,9 +2,11 @@
|
||||
* Mock-LLM E2E test: preset automation card → slash command → skill activation.
|
||||
*
|
||||
* The `slack-standup-digest` skill ships in the public OpenHands extensions
|
||||
* repo with `triggers: ["/standup-digest:setup"]`. The frontend sends
|
||||
* `load_public_skills: true` in every conversation's `agent_context`, so
|
||||
* the SDK's `AgentContext._load_auto_skills` picks it up automatically.
|
||||
* repo with `triggers: ["/standup-digest:setup"]`. The frontend bundles
|
||||
* public skills from the `@openhands/extensions` npm package and passes
|
||||
* them directly in `agent_context.skills` at conversation-start, so the
|
||||
* SDK's trigger matching activates them without the agent-server needing
|
||||
* to clone the extensions repo (`load_public_skills: false`).
|
||||
*
|
||||
* Two tests:
|
||||
* 1. **Card flow**: configure a dummy Slack MCP server so the automation
|
||||
|
||||
@@ -0,0 +1,441 @@
|
||||
/**
|
||||
* Mock-LLM E2E tests: skill loading from project and user directories.
|
||||
*
|
||||
* These tests verify that the SDK's skill loading machinery works
|
||||
* end-to-end with the real agent-server stack:
|
||||
*
|
||||
* 1. **Project skills** from `{workspace}/.agents/skills/` are loaded
|
||||
* alongside bundled public skills and trigger on matching keywords.
|
||||
* The test creates a standalone git repo with the skill committed,
|
||||
* then creates a conversation via API pointing at that repo. The
|
||||
* agent-server creates a worktree from the repo, and since the skill
|
||||
* is committed, `load_project_skills` finds it in the worktree.
|
||||
*
|
||||
* 2. **User skills** from `~/.openhands/skills/` are loaded and trigger
|
||||
* on matching keywords.
|
||||
*
|
||||
* 3. **Skill deletion**: removing a skill file means it is NOT loaded
|
||||
* in subsequent conversations.
|
||||
*
|
||||
* All tests create ephemeral SKILL.md files with unique trigger keywords,
|
||||
* send a message containing those keywords, and verify `activated_skills`
|
||||
* appears in the conversation events API.
|
||||
*/
|
||||
|
||||
import { test, expect, type APIRequestContext } from "@playwright/test";
|
||||
import {
|
||||
BACKEND_URL,
|
||||
SESSION_API_KEY,
|
||||
MOCK_LLM_AGENT_URL,
|
||||
seedLocalStorage,
|
||||
routeSessionApiKey,
|
||||
dismissAnalyticsModal,
|
||||
waitForNonUserMessageText,
|
||||
deleteConversation,
|
||||
registerTrajectory,
|
||||
activateTrajectory,
|
||||
resetMockLLM,
|
||||
setChatInput,
|
||||
waitForPath,
|
||||
getConversationIdFromURL,
|
||||
} from "./utils/mock-llm-helpers";
|
||||
import {
|
||||
createProjectSkillRepo,
|
||||
removeProjectSkillRepo,
|
||||
writeUserSkill,
|
||||
removeUserSkill,
|
||||
userSkillExists,
|
||||
userSkillDirExists,
|
||||
} from "./utils/skill-test-helpers";
|
||||
|
||||
/**
|
||||
* Configure the mock LLM profile. Inlined from `ensureMockLLMProfile`
|
||||
* to work around a CI-specific TS6/Node24 type inference bug (TS2345)
|
||||
* where importing that function alongside `skill-test-helpers` causes
|
||||
* TypeScript to incorrectly resolve its signature.
|
||||
*/
|
||||
async function configureMockLLM(
|
||||
request: APIRequestContext,
|
||||
model = "openai/mock-test-model",
|
||||
) {
|
||||
const settingsResp = await request.get(`${BACKEND_URL}/api/settings`, {
|
||||
headers: {
|
||||
"X-Session-API-Key": SESSION_API_KEY,
|
||||
"X-Expose-Secrets": "encrypted",
|
||||
},
|
||||
});
|
||||
if (settingsResp.ok()) {
|
||||
const settings = (await settingsResp.json()) as Record<string, unknown>;
|
||||
const llm = (
|
||||
settings?.agent_settings as Record<string, unknown> | undefined
|
||||
)?.llm as Record<string, unknown> | undefined;
|
||||
if (llm?.model === model && llm?.base_url === MOCK_LLM_AGENT_URL) return;
|
||||
}
|
||||
const patchResp = await request.patch(`${BACKEND_URL}/api/settings`, {
|
||||
headers: {
|
||||
"X-Session-API-Key": SESSION_API_KEY,
|
||||
"Content-Type": "application/json",
|
||||
},
|
||||
data: {
|
||||
agent_settings_diff: {
|
||||
llm: {
|
||||
model,
|
||||
api_key: "mock-api-key-for-testing",
|
||||
base_url: MOCK_LLM_AGENT_URL,
|
||||
},
|
||||
},
|
||||
},
|
||||
});
|
||||
expect(
|
||||
patchResp.ok(),
|
||||
`PATCH /api/settings failed: ${patchResp.status()}`,
|
||||
).toBe(true);
|
||||
}
|
||||
|
||||
/**
|
||||
* Register a workspace on the agent-server so it appears in the UI dropdown.
|
||||
*/
|
||||
async function addWorkspaceToServer(
|
||||
request: APIRequestContext,
|
||||
name: string,
|
||||
path: string,
|
||||
): Promise<void> {
|
||||
const resp = await request.post(`${BACKEND_URL}/api/workspaces`, {
|
||||
headers: {
|
||||
"X-Session-API-Key": SESSION_API_KEY,
|
||||
"Content-Type": "application/json",
|
||||
},
|
||||
data: {
|
||||
workspaces: [{ id: `e2e-${name}`, name, path }],
|
||||
},
|
||||
});
|
||||
expect(
|
||||
resp.ok(),
|
||||
`POST /api/workspaces failed: ${resp.status()} ${await resp.text()}`,
|
||||
).toBe(true);
|
||||
}
|
||||
|
||||
/**
|
||||
* Remove a workspace from the agent-server.
|
||||
*/
|
||||
async function removeWorkspaceFromServer(
|
||||
request: APIRequestContext,
|
||||
path: string,
|
||||
): Promise<void> {
|
||||
await request.delete(`${BACKEND_URL}/api/workspaces`, {
|
||||
headers: { "X-Session-API-Key": SESSION_API_KEY },
|
||||
params: { path },
|
||||
});
|
||||
}
|
||||
|
||||
// ── Shared constants ─────────────────────────────────────────────────
|
||||
|
||||
const REPLY_TOKEN = "SKILLS_E2E_REPLY_OK";
|
||||
|
||||
// ── Tests ─────────────────────────────────────────────────────────────
|
||||
|
||||
test.describe.configure({ mode: "serial" });
|
||||
|
||||
test.describe("skill loading: project, user, and deletion", () => {
|
||||
const conversationIds = new Set<string>();
|
||||
|
||||
// Unique skill names to avoid collisions with real skills
|
||||
const PROJECT_SKILL_NAME = "e2e-test-project-skill";
|
||||
const PROJECT_SKILL_TRIGGER = "xyzzy-project-e2e-test";
|
||||
|
||||
const USER_SKILL_NAME = "e2e-test-user-skill";
|
||||
const USER_SKILL_TRIGGER = "xyzzy-user-e2e-test";
|
||||
|
||||
test.beforeEach(async ({ page }) => {
|
||||
await seedLocalStorage(page);
|
||||
});
|
||||
|
||||
test.afterEach(async ({ page, request }) => {
|
||||
const match = page.url().match(/\/conversations\/([^/?#]+)/);
|
||||
if (match?.[1]) conversationIds.add(decodeURIComponent(match[1]));
|
||||
|
||||
for (const id of Array.from(conversationIds)) {
|
||||
try {
|
||||
await deleteConversation(request, id);
|
||||
conversationIds.delete(id);
|
||||
} catch {
|
||||
// best-effort cleanup
|
||||
}
|
||||
}
|
||||
await resetMockLLM(request).catch(() => {});
|
||||
});
|
||||
|
||||
// Track the agent-side workspace path for cleanup (may differ from
|
||||
// the host path in Docker mode where volumes are mounted)
|
||||
let projectSkillAgentDir = "";
|
||||
|
||||
test.afterAll(async ({ request }) => {
|
||||
removeProjectSkillRepo(PROJECT_SKILL_NAME);
|
||||
removeUserSkill(USER_SKILL_NAME);
|
||||
if (projectSkillAgentDir) {
|
||||
await removeWorkspaceFromServer(request, projectSkillAgentDir).catch(
|
||||
() => {},
|
||||
);
|
||||
}
|
||||
});
|
||||
|
||||
// ── Test 1: Project skill loaded from workspace ──────────────────
|
||||
//
|
||||
// Creates a standalone git repo with the skill committed, registers it
|
||||
// as a workspace on the agent-server, then uses the UI to select that
|
||||
// workspace and send a message. The agent-server creates a worktree
|
||||
// from the repo, and the committed skill file is present in it for
|
||||
// `load_project_skills` to discover.
|
||||
|
||||
test("project skill in workspace/.agents/skills/ triggers on matching keyword", async ({
|
||||
page,
|
||||
request,
|
||||
}) => {
|
||||
await configureMockLLM(request);
|
||||
|
||||
// Create a git repo with the skill committed
|
||||
const { agentDir } = await test.step(
|
||||
"create git repo with project skill",
|
||||
() => {
|
||||
return createProjectSkillRepo(
|
||||
PROJECT_SKILL_NAME,
|
||||
PROJECT_SKILL_TRIGGER,
|
||||
);
|
||||
},
|
||||
);
|
||||
projectSkillAgentDir = agentDir;
|
||||
|
||||
// Register the workspace on the server using the agent-side path
|
||||
// (same as hostDir in npm mode, container mount path in Docker mode)
|
||||
await test.step("register workspace on server", async () => {
|
||||
await addWorkspaceToServer(request, "skill-test-repo", agentDir);
|
||||
});
|
||||
|
||||
// Trajectory: padding for skill-analysis + agent reply
|
||||
await registerTrajectory(request, "project-skill", [
|
||||
{ text: "" },
|
||||
{ text: `Skill test complete. ${REPLY_TOKEN}` },
|
||||
]);
|
||||
await activateTrajectory(request, "project-skill");
|
||||
|
||||
await routeSessionApiKey(page);
|
||||
await page.goto("/", { waitUntil: "domcontentloaded" });
|
||||
await dismissAnalyticsModal(page);
|
||||
|
||||
// Select the workspace through the UI
|
||||
await test.step("select workspace from UI", async () => {
|
||||
// Click "Open workspace" to open the dialog
|
||||
await page.getByTestId("open-workspace-button").click();
|
||||
|
||||
// Click the workspace dropdown and select our workspace
|
||||
const dropdown = page.getByTestId("workspace-dropdown");
|
||||
await dropdown.click();
|
||||
// The workspace name is "skill-test-repo" — click the matching option
|
||||
await page.getByText("skill-test-repo").click();
|
||||
|
||||
// Click the Confirm button to set the workspace
|
||||
await page.getByTestId("workspace-launch-button").click();
|
||||
});
|
||||
|
||||
await test.step("send message with project skill trigger", async () => {
|
||||
await setChatInput(
|
||||
page,
|
||||
`Please help me with ${PROJECT_SKILL_TRIGGER} setup`,
|
||||
);
|
||||
await page.getByTestId("submit-button").click();
|
||||
await waitForPath(page, /\/conversations\/.+/, 30_000);
|
||||
});
|
||||
|
||||
const conversationId = getConversationIdFromURL(page);
|
||||
conversationIds.add(conversationId);
|
||||
|
||||
await test.step("verify agent reply", async () => {
|
||||
await waitForNonUserMessageText(page, REPLY_TOKEN, 45_000);
|
||||
});
|
||||
|
||||
await test.step("verify project skill activated", async () => {
|
||||
await expect
|
||||
.poll(
|
||||
async () => {
|
||||
const resp = await request.get(
|
||||
`${BACKEND_URL}/api/conversations/${encodeURIComponent(conversationId)}/events/search`,
|
||||
{
|
||||
headers: { "X-Session-API-Key": SESSION_API_KEY },
|
||||
params: { limit: "50" },
|
||||
},
|
||||
);
|
||||
if (!resp.ok()) return `HTTP ${resp.status()}`;
|
||||
const body = (await resp.json()) as { items?: unknown[] };
|
||||
for (const item of body.items ?? []) {
|
||||
const e = item as Record<string, unknown>;
|
||||
const skills =
|
||||
(e.activated_skills as string[] | undefined) ??
|
||||
(e.activated_microagents as string[] | undefined);
|
||||
if (skills?.includes(PROJECT_SKILL_NAME)) return "FOUND";
|
||||
}
|
||||
return "NOT_FOUND";
|
||||
},
|
||||
{
|
||||
message: `expected "${PROJECT_SKILL_NAME}" in activated_skills`,
|
||||
intervals: [1_000, 2_000, 3_000, 5_000],
|
||||
timeout: 25_000,
|
||||
},
|
||||
)
|
||||
.toBe("FOUND");
|
||||
});
|
||||
});
|
||||
|
||||
// ── Test 2: User skill loaded from ~/.openhands/skills/ ──────────
|
||||
|
||||
test("user skill in ~/.openhands/skills/ triggers on matching keyword", async ({
|
||||
page,
|
||||
request,
|
||||
}) => {
|
||||
await configureMockLLM(request);
|
||||
|
||||
await test.step("create user skill file", () => {
|
||||
writeUserSkill(USER_SKILL_NAME, USER_SKILL_TRIGGER);
|
||||
expect(userSkillExists(USER_SKILL_NAME)).toBe(true);
|
||||
});
|
||||
|
||||
await registerTrajectory(request, "user-skill", [
|
||||
{ text: "" },
|
||||
{ text: `Skill test complete. ${REPLY_TOKEN}` },
|
||||
]);
|
||||
await activateTrajectory(request, "user-skill");
|
||||
|
||||
await routeSessionApiKey(page);
|
||||
await page.goto("/", { waitUntil: "domcontentloaded" });
|
||||
await dismissAnalyticsModal(page);
|
||||
|
||||
await test.step("send message with user skill trigger", async () => {
|
||||
await setChatInput(
|
||||
page,
|
||||
`I need help with ${USER_SKILL_TRIGGER} configuration`,
|
||||
);
|
||||
await page.getByTestId("submit-button").click();
|
||||
await waitForPath(page, /\/conversations\/.+/, 30_000);
|
||||
});
|
||||
|
||||
const conversationId = getConversationIdFromURL(page);
|
||||
conversationIds.add(conversationId);
|
||||
|
||||
await test.step("verify agent reply", async () => {
|
||||
await waitForNonUserMessageText(page, REPLY_TOKEN, 45_000);
|
||||
});
|
||||
|
||||
await test.step("verify user skill activated", async () => {
|
||||
await expect
|
||||
.poll(
|
||||
async () => {
|
||||
const resp = await request.get(
|
||||
`${BACKEND_URL}/api/conversations/${encodeURIComponent(conversationId)}/events/search`,
|
||||
{
|
||||
headers: { "X-Session-API-Key": SESSION_API_KEY },
|
||||
params: { limit: "50" },
|
||||
},
|
||||
);
|
||||
if (!resp.ok()) return `HTTP ${resp.status()}`;
|
||||
const body = (await resp.json()) as { items?: unknown[] };
|
||||
for (const item of body.items ?? []) {
|
||||
const e = item as Record<string, unknown>;
|
||||
const skills =
|
||||
(e.activated_skills as string[] | undefined) ??
|
||||
(e.activated_microagents as string[] | undefined);
|
||||
if (skills?.includes(USER_SKILL_NAME)) return "FOUND";
|
||||
}
|
||||
return "NOT_FOUND";
|
||||
},
|
||||
{
|
||||
message: `expected "${USER_SKILL_NAME}" in activated_skills`,
|
||||
intervals: [1_000, 2_000, 3_000, 5_000],
|
||||
timeout: 25_000,
|
||||
},
|
||||
)
|
||||
.toBe("FOUND");
|
||||
});
|
||||
});
|
||||
|
||||
// ── Test 3: Deleted user skill not loaded in new conversation ────
|
||||
|
||||
test("deleting a user skill removes it from subsequent conversations", async ({
|
||||
page,
|
||||
request,
|
||||
}) => {
|
||||
await configureMockLLM(request);
|
||||
|
||||
await test.step("delete user skill file", () => {
|
||||
removeUserSkill(USER_SKILL_NAME);
|
||||
expect(userSkillDirExists(USER_SKILL_NAME)).toBe(false);
|
||||
});
|
||||
|
||||
// Padding for skill-analysis (public skills are still loaded from the npm
|
||||
// package, so the agent-server still makes a skill-analysis LLM call)
|
||||
await registerTrajectory(request, "deleted-skill", [
|
||||
{ text: "" },
|
||||
{ text: `No skill triggered. ${REPLY_TOKEN}` },
|
||||
]);
|
||||
await activateTrajectory(request, "deleted-skill");
|
||||
|
||||
await routeSessionApiKey(page);
|
||||
await page.goto("/", { waitUntil: "domcontentloaded" });
|
||||
await dismissAnalyticsModal(page);
|
||||
|
||||
await test.step(
|
||||
"send message with deleted skill trigger keyword",
|
||||
async () => {
|
||||
await setChatInput(
|
||||
page,
|
||||
`Help me with ${USER_SKILL_TRIGGER} please`,
|
||||
);
|
||||
await page.getByTestId("submit-button").click();
|
||||
await waitForPath(page, /\/conversations\/.+/, 30_000);
|
||||
},
|
||||
);
|
||||
|
||||
const conversationId = getConversationIdFromURL(page);
|
||||
conversationIds.add(conversationId);
|
||||
|
||||
await test.step("verify agent reply", async () => {
|
||||
await waitForNonUserMessageText(page, REPLY_TOKEN, 45_000);
|
||||
});
|
||||
|
||||
await test.step("verify deleted skill NOT activated", async () => {
|
||||
// The agent reply already appeared in the UI (verified above), so the
|
||||
// conversation completed. Poll the events API and verify no event
|
||||
// contains the deleted skill in its activated_skills list.
|
||||
await expect
|
||||
.poll(
|
||||
async () => {
|
||||
const resp = await request.get(
|
||||
`${BACKEND_URL}/api/conversations/${encodeURIComponent(conversationId)}/events/search`,
|
||||
{
|
||||
headers: { "X-Session-API-Key": SESSION_API_KEY },
|
||||
params: { limit: "100" },
|
||||
},
|
||||
);
|
||||
if (!resp.ok()) return `HTTP ${resp.status()}`;
|
||||
const body = (await resp.json()) as { items?: unknown[] };
|
||||
const items = body.items ?? [];
|
||||
if (items.length === 0) return "NO_EVENTS";
|
||||
|
||||
for (const item of items) {
|
||||
const e = item as Record<string, unknown>;
|
||||
const skills =
|
||||
(e.activated_skills as string[] | undefined) ??
|
||||
(e.activated_microagents as string[] | undefined);
|
||||
if (skills?.includes(USER_SKILL_NAME))
|
||||
return `UNEXPECTEDLY_FOUND`;
|
||||
}
|
||||
return "VERIFIED";
|
||||
},
|
||||
{
|
||||
message: `verifying "${USER_SKILL_NAME}" NOT in activated_skills`,
|
||||
intervals: [1_000, 2_000, 3_000],
|
||||
timeout: 15_000,
|
||||
},
|
||||
)
|
||||
.toBe("VERIFIED");
|
||||
});
|
||||
});
|
||||
});
|
||||
@@ -137,6 +137,11 @@ async function routePaginationConversation(page: Page) {
|
||||
);
|
||||
|
||||
// Stub the event search endpoint with the synthetic paginated events.
|
||||
// Older-events requests (those with timestamp__lt) are delayed slightly so
|
||||
// the loading indicator has time to render before the response arrives.
|
||||
// Without this, React can batch the isLoading true→false transition into a
|
||||
// single commit and the DOM element never materialises — making the
|
||||
// "loading-older-events" assertion flaky.
|
||||
await page.route(
|
||||
`**/api/conversations/${PAGINATION_CONVERSATION_ID}/events/search**`,
|
||||
async (route, req) => {
|
||||
@@ -145,7 +150,11 @@ async function routePaginationConversation(page: Page) {
|
||||
return;
|
||||
}
|
||||
const url = new URL(req.url());
|
||||
const isOlderPage = url.searchParams.has("timestamp__lt");
|
||||
const result = searchPaginationEvents(allEvents, url.searchParams);
|
||||
if (isOlderPage) {
|
||||
await new Promise((r) => setTimeout(r, 300));
|
||||
}
|
||||
await route.fulfill({
|
||||
status: 200,
|
||||
contentType: "application/json",
|
||||
|
||||
@@ -9,7 +9,8 @@
|
||||
* --output mock-llm-report.md \
|
||||
* [--workflow-url <url>] \
|
||||
* [--commit <sha>] \
|
||||
* [--artifact-url <url>]
|
||||
* [--artifact-url <url>] \
|
||||
* [--new-files <comma-separated spec paths added in this PR>]
|
||||
*/
|
||||
|
||||
import { existsSync, readFileSync, writeFileSync } from "node:fs";
|
||||
@@ -51,10 +52,12 @@ function loadResults(path) {
|
||||
}
|
||||
}
|
||||
|
||||
function collectTests(suites, parents = []) {
|
||||
function collectTests(suites, parents = [], parentFile = "") {
|
||||
const tests = [];
|
||||
for (const suite of suites ?? []) {
|
||||
const titles = [...parents, suite.title].filter(Boolean);
|
||||
// Playwright's JSON reporter sets `file` on each suite/spec
|
||||
const suiteFile = suite.file || parentFile;
|
||||
for (const spec of suite.specs ?? []) {
|
||||
for (const test of spec.tests ?? []) {
|
||||
const results = test.results ?? [];
|
||||
@@ -65,6 +68,7 @@ function collectTests(suites, parents = []) {
|
||||
);
|
||||
tests.push({
|
||||
title: [...titles, spec.title].filter(Boolean).join(" › "),
|
||||
file: spec.file || suiteFile,
|
||||
status: lastResult?.status ?? (spec.ok ? "passed" : "unknown"),
|
||||
durationMs: duration,
|
||||
retryCount: Math.max(0, results.length - 1),
|
||||
@@ -72,7 +76,7 @@ function collectTests(suites, parents = []) {
|
||||
});
|
||||
}
|
||||
}
|
||||
tests.push(...collectTests(suite.suites, titles));
|
||||
tests.push(...collectTests(suite.suites, titles, suiteFile));
|
||||
}
|
||||
return tests;
|
||||
}
|
||||
@@ -141,7 +145,15 @@ function overallIcon(status) {
|
||||
|
||||
// ── Report rendering ───────────────────────────────────────────────────
|
||||
|
||||
function renderReport({ tests, workflowUrl, commit, artifactUrl, title, markerMeta }) {
|
||||
function renderReport({
|
||||
tests,
|
||||
workflowUrl,
|
||||
commit,
|
||||
artifactUrl,
|
||||
title,
|
||||
newFiles,
|
||||
markerMeta,
|
||||
}) {
|
||||
const status = overallStatus(tests);
|
||||
const icon = overallIcon(status);
|
||||
const passed = tests.filter((t) => t.status === "passed").length;
|
||||
@@ -153,6 +165,24 @@ function renderReport({ tests, workflowUrl, commit, artifactUrl, title, markerMe
|
||||
const wasKilledMidSuite =
|
||||
markerMeta?.status === "in_progress" && markerMeta.total > markerMeta.completed;
|
||||
|
||||
// Determine which tests are new (from newly added spec files).
|
||||
// Playwright's JSON file paths are relative to testDir (e.g. "mock-llm-skills.spec.ts")
|
||||
// while --new-files paths are repo-relative (e.g. "tests/e2e/mock-llm/mock-llm-skills.spec.ts").
|
||||
// Match by basename or suffix in either direction.
|
||||
const newFileSet = new Set(newFiles ?? []);
|
||||
const basename = (p) => p.split("/").pop();
|
||||
const isNewTest = (t) =>
|
||||
newFileSet.size > 0 &&
|
||||
t.file &&
|
||||
[...newFileSet].some(
|
||||
(nf) =>
|
||||
t.file === nf ||
|
||||
basename(t.file) === basename(nf) ||
|
||||
nf.endsWith(`/${t.file}`) ||
|
||||
t.file.endsWith(`/${nf}`),
|
||||
);
|
||||
const newCount = tests.filter(isNewTest).length;
|
||||
|
||||
const lines = [];
|
||||
|
||||
// Header — use 🛑 when killed mid-suite so it's visually distinct
|
||||
@@ -164,6 +194,7 @@ function renderReport({ tests, workflowUrl, commit, artifactUrl, title, markerMe
|
||||
const parts = [`**${passed}/${total} passed**`];
|
||||
if (failed) parts.push(`**${failed} failed**`);
|
||||
if (skipped) parts.push(`${skipped} skipped`);
|
||||
if (newCount) parts.push(`🆕 ${newCount} new`);
|
||||
if (wasKilledMidSuite) {
|
||||
const notRun = markerMeta.total - markerMeta.completed;
|
||||
parts.push(`⚠️ **${notRun} not run** (process killed at ${markerMeta.completed}/${markerMeta.total})`);
|
||||
@@ -181,6 +212,25 @@ function renderReport({ tests, workflowUrl, commit, artifactUrl, title, markerMe
|
||||
lines.push("");
|
||||
}
|
||||
|
||||
// New-tests callout (prominent, above the table)
|
||||
if (newCount > 0) {
|
||||
const newTests = tests.filter(isNewTest);
|
||||
// Group new tests by spec file
|
||||
const byFile = new Map();
|
||||
for (const t of newTests) {
|
||||
const key = t.file || "unknown";
|
||||
if (!byFile.has(key)) byFile.set(key, []);
|
||||
byFile.get(key).push(t);
|
||||
}
|
||||
lines.push(`> **🟢 ${newCount} new test${newCount === 1 ? "" : "s"} added in this PR**`);
|
||||
for (const [file, fileTests] of byFile) {
|
||||
for (const t of fileTests) {
|
||||
lines.push(`> - ${statusIcon(t.status)} \`${file}\` › ${t.title.replace(/^.*› /, "")}`);
|
||||
}
|
||||
}
|
||||
lines.push("");
|
||||
}
|
||||
|
||||
// Test results table
|
||||
lines.push("| Status | Test | Duration |");
|
||||
lines.push("|:------:|------|----------|");
|
||||
@@ -284,12 +334,21 @@ if (!data || tests.length === 0) {
|
||||
}
|
||||
}
|
||||
|
||||
// Parse --new-files: comma-separated list of spec file paths added in this PR
|
||||
const newFiles = args.new_files
|
||||
? args.new_files
|
||||
.split(",")
|
||||
.map((f) => f.trim())
|
||||
.filter(Boolean)
|
||||
: [];
|
||||
|
||||
const report = renderReport({
|
||||
tests,
|
||||
workflowUrl: args.workflow_url || "",
|
||||
commit: args.commit || "",
|
||||
artifactUrl: args.artifact_url || "",
|
||||
title: args.title || "",
|
||||
newFiles,
|
||||
markerMeta,
|
||||
});
|
||||
|
||||
|
||||
@@ -0,0 +1,139 @@
|
||||
/**
|
||||
* Helpers for skill loading E2E tests.
|
||||
*
|
||||
* File-system operations (create/remove SKILL.md files) and API
|
||||
* assertions (verify activated_skills on events) are separated here
|
||||
* to avoid type-resolution conflicts between node built-in imports
|
||||
* and @playwright/test types in the same file (TypeScript 6 / Node 24).
|
||||
*
|
||||
* Docker support: When running against a Docker container, the agent-server
|
||||
* filesystem is isolated from the host. The Playwright config sets env vars
|
||||
* for the container-side paths so the test can register the correct paths
|
||||
* with the agent-server while creating files on the host (volume-mounted).
|
||||
*/
|
||||
|
||||
import { resolve, join } from "path";
|
||||
import { mkdirSync, writeFileSync, rmSync, existsSync } from "fs";
|
||||
import { execSync } from "child_process";
|
||||
import { homedir } from "os";
|
||||
|
||||
// ── Paths ────────────────────────────────────────────────────────────
|
||||
|
||||
/** STATE_DIR matches playwright.mock-llm.config.ts */
|
||||
export const STATE_DIR = resolve(".tmp/mock-llm-state");
|
||||
|
||||
/**
|
||||
* Root directory for skill-test workspace git repos (HOST-side).
|
||||
* Each call to `createProjectSkillRepo` creates a self-contained git repo
|
||||
* here with the skill file already committed, so the agent-server's
|
||||
* worktree machinery picks it up (worktrees only contain committed content).
|
||||
*/
|
||||
export const SKILL_REPOS_DIR = resolve(".tmp/mock-llm-skill-repos");
|
||||
|
||||
/**
|
||||
* The path the agent-server sees for skill repos.
|
||||
* In npm mode this is the same as SKILL_REPOS_DIR (same filesystem).
|
||||
* In Docker mode this is the container-side mount point set by the config.
|
||||
*/
|
||||
export const SKILL_REPOS_AGENT_DIR =
|
||||
process.env.MOCK_LLM_SKILL_REPOS_CONTAINER_DIR ?? SKILL_REPOS_DIR;
|
||||
|
||||
/**
|
||||
* User-level skills directory — HOST-side (for file creation/removal).
|
||||
* In Docker mode, we use a local temp dir that is volume-mounted into the
|
||||
* container at the agent-server's expected `~/.openhands/skills/` path.
|
||||
*/
|
||||
export const USER_SKILLS_DIR = process.env.MOCK_LLM_USER_SKILLS_HOST_DIR
|
||||
? resolve(process.env.MOCK_LLM_USER_SKILLS_HOST_DIR)
|
||||
: join(homedir(), ".openhands", "skills");
|
||||
|
||||
// ── Skill content builders ───────────────────────────────────────────
|
||||
|
||||
function makeSkillMd(
|
||||
name: string,
|
||||
trigger: string,
|
||||
description: string,
|
||||
): string {
|
||||
return [
|
||||
"---",
|
||||
`name: ${name}`,
|
||||
`description: ${description}`,
|
||||
"triggers:",
|
||||
`- ${trigger}`,
|
||||
"---",
|
||||
"",
|
||||
`This is the ${name} skill content for E2E testing.`,
|
||||
`It should activate when the keyword "${trigger}" appears.`,
|
||||
].join("\n");
|
||||
}
|
||||
|
||||
/**
|
||||
* Create a standalone git repo with a project skill committed.
|
||||
*
|
||||
* The agent-server creates a git worktree for each conversation from the
|
||||
* source workspace. Only committed files appear in worktrees, so the skill
|
||||
* must be committed to the repo for `load_project_skills` to find it.
|
||||
*
|
||||
* @returns Object with `hostDir` (absolute host path for file ops) and
|
||||
* `agentDir` (path the agent-server sees — same in npm mode,
|
||||
* container-side mount in Docker mode).
|
||||
*/
|
||||
export function createProjectSkillRepo(
|
||||
name: string,
|
||||
trigger: string,
|
||||
description = "E2E test skill",
|
||||
): { hostDir: string; agentDir: string } {
|
||||
const repoDir = join(SKILL_REPOS_DIR, `${name}-repo`);
|
||||
// Start fresh each time
|
||||
rmSync(repoDir, { recursive: true, force: true });
|
||||
mkdirSync(repoDir, { recursive: true });
|
||||
|
||||
// Write the skill file
|
||||
const skillDir = join(repoDir, ".agents", "skills", name);
|
||||
mkdirSync(skillDir, { recursive: true });
|
||||
writeFileSync(
|
||||
join(skillDir, "SKILL.md"),
|
||||
makeSkillMd(name, trigger, description),
|
||||
);
|
||||
|
||||
// Initialize as a git repo and commit
|
||||
const opts = { cwd: repoDir, stdio: "pipe" as const };
|
||||
execSync("git init", opts);
|
||||
execSync('git config user.email "test@test.com"', opts);
|
||||
execSync('git config user.name "Test"', opts);
|
||||
execSync("git add -A", opts);
|
||||
execSync('git commit -m "Add project skill"', opts);
|
||||
|
||||
const hostDir = resolve(repoDir);
|
||||
const agentDir = join(SKILL_REPOS_AGENT_DIR, `${name}-repo`);
|
||||
return { hostDir, agentDir };
|
||||
}
|
||||
|
||||
/** Remove a project skill repo created by `createProjectSkillRepo`. */
|
||||
export function removeProjectSkillRepo(name: string): void {
|
||||
const repoDir = join(SKILL_REPOS_DIR, `${name}-repo`);
|
||||
rmSync(repoDir, { recursive: true, force: true });
|
||||
}
|
||||
|
||||
export function writeUserSkill(
|
||||
name: string,
|
||||
trigger: string,
|
||||
description = "E2E test user skill",
|
||||
): void {
|
||||
const dir = join(USER_SKILLS_DIR, name);
|
||||
mkdirSync(dir, { recursive: true });
|
||||
writeFileSync(join(dir, "SKILL.md"), makeSkillMd(name, trigger, description));
|
||||
}
|
||||
|
||||
export function removeUserSkill(name: string): void {
|
||||
const dir = join(USER_SKILLS_DIR, name);
|
||||
rmSync(dir, { recursive: true, force: true });
|
||||
}
|
||||
|
||||
export function userSkillExists(name: string): boolean {
|
||||
return existsSync(join(USER_SKILLS_DIR, name, "SKILL.md"));
|
||||
}
|
||||
|
||||
export function userSkillDirExists(name: string): boolean {
|
||||
return existsSync(join(USER_SKILLS_DIR, name));
|
||||
}
|
||||
Reference in New Issue
Block a user