diff --git a/.env.sample b/.env.sample index 00a1d81a5a..85d123192c 100644 --- a/.env.sample +++ b/.env.sample @@ -9,7 +9,6 @@ VITE_BACKEND_BASE_URL="http://127.0.0.1:8000" # Base URL used by browser-side di # VITE_WORKING_DIR="/workspace/project/agent-canvas" # Base dir for per-conversation working_dirs. Each conversation's working_dir is /. Defaults to /workspaces, which is the sibling of the agent server's /conversations/ persistence dir β€” both share the same per conversation. # VITE_WORKER_URLS="" # Optional comma-separated worker URLs for the Browser tab # VITE_ENABLE_BROWSER_TOOLS="true" # Set to false to omit BrowserToolSet from new conversations -# VITE_LOAD_PUBLIC_SKILLS="false" # Set to false to disable loading public skills from https://github.com/OpenHands/extensions (on by default) # Frontend dev server VITE_FRONTEND_PORT="3001" # Port to run the frontend application diff --git a/.github/workflows/mock-llm-docker-e2e.yml b/.github/workflows/mock-llm-docker-e2e.yml index eab35b9734..1e9dbd415e 100644 --- a/.github/workflows/mock-llm-docker-e2e.yml +++ b/.github/workflows/mock-llm-docker-e2e.yml @@ -325,6 +325,21 @@ jobs: playwright-report-mock-llm-docker/ test-results-mock-llm-docker/ + - name: Detect newly added spec files + if: always() && github.event.pull_request.number + id: new_specs + env: + GITHUB_TOKEN: ${{ github.token }} + run: | + # Find mock-LLM spec files added (not just modified) in this PR + NEW_FILES=$(gh api \ + "/repos/${{ github.repository }}/pulls/${{ github.event.pull_request.number }}/files" \ + --paginate \ + --jq '[.[] | select(.status == "added") | .filename + | select(test("tests/e2e/mock-llm/.*\\.spec\\.ts$"))] + | join(",")') + echo "files=$NEW_FILES" >> "$GITHUB_OUTPUT" + - name: Render test report if: always() run: | @@ -334,7 +349,8 @@ jobs: --workflow-url "$MOCK_LLM_WORKFLOW_URL" \ --commit "${{ steps.ctx.outputs.sha }}" \ --artifact-url "${{ steps.upload_artifacts.outputs.artifact-url || '' }}" \ - --title "Mock-LLM Docker E2E Test Results" + --title "Mock-LLM Docker E2E Test Results" \ + --new-files "${{ steps.new_specs.outputs.files || '' }}" cat "$MOCK_LLM_REPORT_PATH" >> "$GITHUB_STEP_SUMMARY" - name: Post PR comment diff --git a/.github/workflows/mock-llm-e2e.yml b/.github/workflows/mock-llm-e2e.yml index afda220607..11c2dd2447 100644 --- a/.github/workflows/mock-llm-e2e.yml +++ b/.github/workflows/mock-llm-e2e.yml @@ -186,6 +186,22 @@ jobs: playwright-report-mock-llm/ test-results-mock-llm/ + - name: Detect newly added spec files + if: always() && github.event.pull_request.number + id: new_specs + env: + GITHUB_TOKEN: ${{ github.token }} + run: | + # Find mock-LLM spec files added (not just modified) in this PR + # Uses the GitHub API instead of git diff to avoid shallow-clone issues + NEW_FILES=$(gh api \ + "/repos/${{ github.repository }}/pulls/${{ github.event.pull_request.number }}/files" \ + --paginate \ + --jq '[.[] | select(.status == "added") | .filename + | select(test("tests/e2e/mock-llm/.*\\.spec\\.ts$"))] + | join(",")') + echo "files=$NEW_FILES" >> "$GITHUB_OUTPUT" + - name: Render test report if: always() run: | @@ -194,7 +210,8 @@ jobs: --output "$MOCK_LLM_REPORT_PATH" \ --workflow-url "$MOCK_LLM_WORKFLOW_URL" \ --commit "${{ github.event.pull_request.head.sha || github.sha }}" \ - --artifact-url "${{ steps.upload_artifacts.outputs.artifact-url || '' }}" + --artifact-url "${{ steps.upload_artifacts.outputs.artifact-url || '' }}" \ + --new-files "${{ steps.new_specs.outputs.files || '' }}" cat "$MOCK_LLM_REPORT_PATH" >> "$GITHUB_STEP_SUMMARY" - name: Save PR number for comment workflow diff --git a/.gitignore b/.gitignore index ea10421add..6714f35767 100644 --- a/.gitignore +++ b/.gitignore @@ -8,6 +8,7 @@ build/ logs/ /.tmp/live-e2e-state/ /.tmp/mock-llm-state/ +/.tmp/mock-llm-skill-repos/ dist/ __pycache__ diff --git a/AGENTS.md b/AGENTS.md index 9d46346256..9bc9753f50 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -13,9 +13,8 @@ - `VITE_WORKING_DIR` for the default workspace path sent when starting conversations. - `VITE_WORKER_URLS` as a comma-separated list of browser worker URLs if you want the Browser tab to probe exposed app hosts. - `VITE_ENABLE_BROWSER_TOOLS=false` to omit `BrowserToolSet` from new conversation payloads. - - `VITE_LOAD_PUBLIC_SKILLS=false` to disable loading public skills from the OpenHands extensions marketplace (https://github.com/OpenHands/extensions). Defaults to true (opt-out). +- Public skills are loaded from the `@openhands/extensions` npm package at build time via `SKILLS_CATALOG` (exported from `@openhands/extensions/skills`). The frontend's `SkillsService` maps catalog entries to `SkillInfo` objects and merges them with user/project skills fetched from the agent-server (with `load_public: false`). The agent-server no longer clones the extensions repo or uses `EXTENSIONS_REF` for public skills. - Default working-dir fallback is now the relative path `workspace/project` (exported as `DEFAULT_WORKING_DIR` from `src/api/agent-server-config.ts`); git-path heuristics and the default PLAN preview path should reuse that constant instead of hardcoding `/workspace/project`. -- `EXTENSIONS_REF` is derived from the `@openhands/extensions` git URL in `package.json` via `getExtensionsRef()` in `scripts/dev-safe.mjs`, and exported as `DEFAULT_EXTENSIONS_REF`. `buildAgentServerEnv()` injects it as `EXTENSIONS_REF` when not already set in the environment. The Docker path bakes the SHA into `defaults.env` as `CONFIG_EXTENSIONS_REF` (via the `config-gen` stage) and `docker/entrypoint.sh` applies it as `EXTENSIONS_REF="${EXTENSIONS_REF:-${CONFIG_EXTENSIONS_REF:-}}"`. The agent-server SDK skips network polling when it already has the requested SHA in its local cache. - The UI keeps most OpenHands routes/layout intact, but hosted-only behavior (org, account management, integrations) has been removed via the fabricated OSS config because there is no separate app backend. - Verification command: `npm run typecheck && npm run build`. - GitHub automation now includes `.github/workflows/ci.yml` for `npm ci`, `npm test`, and `npm run build`, plus `.github/dependabot.yml` with weekly npm/github-actions updates gated by a 7-day cooldown. @@ -197,7 +196,7 @@ you are running inside of β€” NOT the automation backend. - `mock-llm-partial-stack.spec.ts` β€” Partial stack mode tests. Unlike other specs, these spawn their own `bin/agent-canvas.mjs` child processes instead of relying on the config's webServer entries. Three describe blocks: (1) `--frontend-only` verifies static frontend is served (200 on `/`), backend routes return 503 (`/server_info`, `/api/settings`, `/api/automation/v1`), and the browser shows the manage-backends modal; (2) `--backend-only` verifies `/server_info` returns 200, `/api/settings` is reachable, automation endpoint works, and root/asset requests return 503; (3) port conflict verifies the process exits non-zero with a clear error message when the ingress port is occupied, then starts successfully on a free port. Each test uses isolated state dirs and high port numbers (18310+ range) to avoid collisions with the main full-stack instance. - `mock-llm-ui-regressions.spec.ts` β€” UI regression tests (CSS isolation scoping, event pagination on scroll-up, workspace selection persistence). Uses `page.route()` to intercept specific API responses where deterministic mock data is needed. Consolidated from the former `tests/e2e/regressions/` directory (which was never wired into CI). - Tests run serially (`workers: 1`, `mode: "serial"` per describe block). Files are discovered alphabetically so the ACP agent test runs first, followed by auth-modes, automation, conversation, etc.; each spec is self-contained (automation test configures its own LLM profile via the settings API, ACP test resets back to OpenHands agent in afterAll). The `afterEach` hook resets the mock LLM to its default trajectory so subsequent specs start fresh even when a preceding test fails. -- CI workflow: `.github/workflows/mock-llm-e2e.yml` runs on every PR commit (opened, synchronize, reopened) and on manual dispatch. It builds the frontend, starts the mock LLM server, runs the tests, and posts a PR comment with results. +- CI workflow: `.github/workflows/mock-llm-e2e.yml` runs on every PR commit (opened, synchronize, reopened) and on manual dispatch. It builds the frontend, starts the mock LLM server, runs the tests, and posts a PR comment with results. The PR comment marks tests from newly added spec files with a πŸ†• badge; the "Detect newly added spec files" step queries the GitHub API for files with `status == "added"` matching `tests/e2e/mock-llm/**/*.spec.ts`, and passes them as `--new-files` to `render-mock-llm-report.mjs`. The summary line shows the count of new tests (e.g. `πŸ†• 2 new`). The same detection is wired into the Docker E2E workflow (`mock-llm-docker-e2e.yml`). - The custom `DoneMarkerReporter` writes `.mock-llm-markers/.tests-done` after all tests complete (before webServer teardown) so the CI wrapper can detect completion and kill the lingering teardown process. ### Docker Image Testing (Shared Specs) @@ -211,6 +210,7 @@ you are running inside of β€” NOT the automation backend. - `MOCK_LLM_AGENT_URL` β€” defaults to `MOCK_LLM_BASE_URL`, overridable via `MOCK_LLM_AGENT_URL` env var. Used when configuring the LLM profile (`base_url` field) β€” this is the URL the agent-server uses for inference calls. The npm path and Docker-with-`--network host` path use the same value; Docker on macOS needs the override. - **Docker image**: Set `MOCK_LLM_DOCKER_IMAGE` to the image tag (default: `ghcr.io/openhands/agent-canvas:latest`). The container is started with `--rm --network host` and a unique `--name` for cleanup. - **State isolation**: The Docker container uses its internal state directory (no host mount needed for tests). Each test run starts a fresh container. +- **Skill test volume mounts**: Tests that create files the agent-server needs to read (skill repos, user skills) require Docker volume mounts because the container has an isolated filesystem. The Docker config mounts `.tmp/mock-llm-skill-repos/` β†’ `/tmp/mock-llm-skill-repos/` for project skills and `.tmp/mock-llm-user-skills/` β†’ `/home/openhands/.openhands/skills/` for user skills. Env vars `MOCK_LLM_SKILL_REPOS_CONTAINER_DIR` and `MOCK_LLM_USER_SKILLS_HOST_DIR` tell `skill-test-helpers.ts` which paths to use for agent-server API registration vs. host-side file operations. - CI workflow: `.github/workflows/mock-llm-docker-e2e.yml` has three triggers β€” all pull the already-built image from GHCR (no rebuild): (1) `workflow_run` fires automatically after the `Docker` workflow completes on main; (2) `pull_request` fires on every PR commit (opened, synchronize, reopened), polls the Docker workflow until it finishes for the PR's head SHA, then pulls the image (needed because `workflow_run` only fires for workflow files already on the default branch); (3) `workflow_dispatch` accepts a custom `docker_image` input. The image tag is derived from the commit SHA (`ghcr.io/openhands/agent-canvas:sha--amd64`). Fork PRs are skipped (no GHCR push). Report artifacts go to `test-results-mock-llm-docker/` and `playwright-report-mock-llm-docker/`. ## Debugging E2E Test Failures diff --git a/__tests__/api/agent-server-adapter.test.ts b/__tests__/api/agent-server-adapter.test.ts index 8ef3d3096a..b55a828529 100644 --- a/__tests__/api/agent-server-adapter.test.ts +++ b/__tests__/api/agent-server-adapter.test.ts @@ -115,11 +115,36 @@ describe("buildStartConversationRequest", () => { { name: "browser_tool_set", params: {} }, { name: "task_tool_set", params: {} }, ]); - expect(payload.agent_settings.agent_context).toEqual({ - load_public_skills: true, + expect(payload.agent_settings.agent_context).toMatchObject({ + load_public_skills: false, load_user_skills: true, load_project_skills: true, }); + // Bundled public skills are injected into agent_context.skills so the + // SDK can perform trigger matching without cloning the extensions repo. + expect( + Array.isArray(payload.agent_settings.agent_context.skills), + ).toBe(true); + const skills = payload.agent_settings.agent_context.skills as Record< + string, + unknown + >[]; + expect(skills.length).toBeGreaterThan(0); + // Every bundled skill must carry the fields the SDK needs for trigger + // matching and system-prompt injection. + for (const skill of skills) { + expect(skill).toHaveProperty("name"); + expect(skill).toHaveProperty("content"); + expect(skill).toHaveProperty("source", "public"); + expect(skill).toHaveProperty("is_agentskills_format", true); + // trigger is either null (always-active) or { type, keywords } + if (skill.trigger !== null) { + expect(skill.trigger).toMatchObject({ + type: "keyword", + keywords: expect.arrayContaining([expect.any(String)]), + }); + } + } expect(payload.agent_settings.agent).toBe("CodeActAgent"); expect(payload.agent_settings.enable_switch_llm_tool).toBe(true); expect(payload.workspace.working_dir).toBe( @@ -986,11 +1011,14 @@ describe("agent_settings runtime services suffix", () => { }) as { agent_settings: { agent_context: Record }; }; - expect(payload.agent_settings.agent_context).toEqual({ - load_public_skills: true, + expect(payload.agent_settings.agent_context).toMatchObject({ + load_public_skills: false, load_user_skills: true, load_project_skills: true, }); + expect( + Array.isArray(payload.agent_settings.agent_context.skills), + ).toBe(true); }); it("sets system_message_suffix when runtime info is provided", () => { @@ -1013,7 +1041,7 @@ describe("agent_settings runtime services suffix", () => { agent_settings: { agent_context: Record }; }; expect(payload.agent_settings.agent_context).toMatchObject({ - load_public_skills: true, + load_public_skills: false, load_user_skills: true, }); expect( @@ -1063,11 +1091,16 @@ describe("buildStartConversationRequest β€” ACP discriminator", () => { expect(payload.agent_settings.llm).toBeUndefined(); expect(payload.agent_settings.condenser).toBeUndefined(); expect(payload.agent_settings.tools).toBeUndefined(); - expect(payload.agent_settings.agent_context).toEqual({ - load_public_skills: true, + const acpAgentContext = payload.agent_settings.agent_context as Record< + string, + unknown + >; + expect(acpAgentContext).toMatchObject({ + load_public_skills: false, load_user_skills: true, load_project_skills: true, }); + expect(Array.isArray(acpAgentContext.skills)).toBe(true); expect(payload.tags).toEqual({ [ACP_SERVER_TAG_KEY]: "claude-code" }); }); diff --git a/__tests__/api/agent-server-config.test.ts b/__tests__/api/agent-server-config.test.ts index c895261446..79cc877c6d 100644 --- a/__tests__/api/agent-server-config.test.ts +++ b/__tests__/api/agent-server-config.test.ts @@ -8,7 +8,6 @@ import { getAgentServerWorkingDir, isAuthRequired, isAuthRequiredAndMissing, - shouldLoadPublicSkills, } from "#/api/agent-server-config"; const ORIGINAL_LOCATION = window.location; @@ -75,23 +74,6 @@ describe("agent server config", () => { ).toBe("/srv/workspaces/4a8dca373bf048dea0af949d711c3d48"); }); - it("loads public skills by default when VITE_LOAD_PUBLIC_SKILLS is unset", () => { - vi.stubEnv("VITE_LOAD_PUBLIC_SKILLS", ""); - - expect(shouldLoadPublicSkills()).toBe(true); - }); - - it("loads public skills when VITE_LOAD_PUBLIC_SKILLS is explicitly 'true'", () => { - vi.stubEnv("VITE_LOAD_PUBLIC_SKILLS", "true"); - - expect(shouldLoadPublicSkills()).toBe(true); - }); - - it("does not load public skills only when VITE_LOAD_PUBLIC_SKILLS is explicitly 'false'", () => { - vi.stubEnv("VITE_LOAD_PUBLIC_SKILLS", "false"); - - expect(shouldLoadPublicSkills()).toBe(false); - }); }); describe("isAuthRequired", () => { diff --git a/__tests__/api/skills-service.test.ts b/__tests__/api/skills-service.test.ts index d778b577c3..644b0ea640 100644 --- a/__tests__/api/skills-service.test.ts +++ b/__tests__/api/skills-service.test.ts @@ -6,10 +6,24 @@ import { setRegisteredBackends, } from "#/api/backend-registry/active-store"; import type { Backend } from "#/api/backend-registry/types"; -import SkillsService from "#/api/skills-service"; -const { mockGetSkills } = vi.hoisted(() => ({ +const { mockGetSkills, MOCK_PUBLIC_CATALOG } = vi.hoisted(() => ({ mockGetSkills: vi.fn(), + MOCK_PUBLIC_CATALOG: [ + { + name: "mock-public-skill", + description: "A mock public skill", + triggers: ["mock"], + content: "mock content", + }, + { + name: "another-public-skill", + description: "Another one", + triggers: [], + content: "more content", + license: "MIT", + }, + ], })); vi.mock("@openhands/typescript-client/clients", () => ({ @@ -18,6 +32,12 @@ vi.mock("@openhands/typescript-client/clients", () => ({ }), })); +vi.mock("@openhands/extensions/skills", () => ({ + SKILLS_CATALOG: MOCK_PUBLIC_CATALOG, +})); + +import SkillsService from "#/api/skills-service"; + const localBackend: Backend = { id: "local", name: "Local", @@ -41,37 +61,49 @@ afterEach(() => { }); describe("SkillsService.getSkills against the agent-server backend", () => { - it("requests load_public:true for the global Skills page even when VITE_LOAD_PUBLIC_SKILLS is unset, so a fresh dev env still shows the public catalog", async () => { - // Arrange: the dev-default scenario that shipped the empty Skills page β€” - // VITE_LOAD_PUBLIC_SKILLS is not set, so shouldLoadPublicSkills() would - // return false. The agent-server has one public skill it can return. - vi.stubEnv("VITE_LOAD_PUBLIC_SKILLS", ""); + it("requests only user/project skills from agent-server (load_public: false) and appends the bundled public catalog", async () => { + const userSkill = { + name: "my-custom-skill", + type: "knowledge", + content: "custom content", + triggers: [], + source: "user", + is_agentskills_format: false, + }; mockGetSkills.mockResolvedValue({ - skills: [ - { - name: "alpha", - type: "knowledge", - content: "...", - triggers: [], - source: "public", - is_agentskills_format: false, - }, - ], - sources: { sandbox: 0, sdk_base: 1, org: 0, project: 0 }, + skills: [userSkill], + sources: { sandbox: 0, sdk_base: 0, org: 0, project: 0 }, }); - // Act const skills = await SkillsService.getSkills(); - // Assert: the request opts the user into public skills regardless of the - // perf-oriented VITE_LOAD_PUBLIC_SKILLS gate, and the page receives them. + // Agent-server is asked only for user/project skills, not public. expect(mockGetSkills).toHaveBeenCalledTimes(1); expect(mockGetSkills.mock.calls[0]?.[0]).toMatchObject({ - load_public: true, + load_public: false, load_user: true, load_project: true, load_org: false, }); - expect(skills.map((s) => s.name)).toEqual(["alpha"]); + + // Result = local skills first, then all bundled public skills. + expect(skills[0]?.name).toBe("my-custom-skill"); + expect(skills).toHaveLength(1 + MOCK_PUBLIC_CATALOG.length); + + // Every public skill from the bundled catalog is present. + const publicNames = skills.slice(1).map((s) => s.name); + for (const entry of MOCK_PUBLIC_CATALOG) { + expect(publicNames).toContain(entry.name); + } + expect(skills.slice(1).every((s) => s.source === "public")).toBe(true); + }); + + it("returns only bundled public skills when agent-server is unreachable", async () => { + mockGetSkills.mockRejectedValue(new Error("ECONNREFUSED")); + + const skills = await SkillsService.getSkills(); + + expect(skills).toHaveLength(MOCK_PUBLIC_CATALOG.length); + expect(skills.every((s) => s.source === "public")).toBe(true); }); }); diff --git a/docker/Dockerfile b/docker/Dockerfile index 3755b3a93a..28b088a87d 100644 --- a/docker/Dockerfile +++ b/docker/Dockerfile @@ -45,16 +45,10 @@ RUN npm run build # ── Stage 1b: Generate shell-sourceable defaults from config/defaults.json ── # This avoids needing jq/python at container runtime to parse the JSON. -# package.json is also read to extract the pinned @openhands/extensions SHA -# and emit it as CONFIG_EXTENSIONS_REF so the agent-server uses the same -# extensions commit as the bundled UI (via EXTENSIONS_REF in entrypoint.sh). FROM node:24-slim AS config-gen -COPY config/defaults.json package.json /tmp/ +COPY config/defaults.json /tmp/ RUN node -e " \ const c = JSON.parse(require('fs').readFileSync('/tmp/defaults.json','utf-8')); \ - const pkg = JSON.parse(require('fs').readFileSync('/tmp/package.json','utf-8')); \ - const extUrl = (pkg.dependencies || pkg.devDependencies || {})['@openhands/extensions'] || ''; \ - const extSha = (extUrl.match(/#([0-9a-f]{40})\$/i) || [])[1] || ''; \ const lines = [ \ 'CONFIG_AGENT_SERVER_PORT=' + c.ports.agentServer, \ 'CONFIG_AUTOMATION_PORT=' + c.ports.automation, \ @@ -64,7 +58,6 @@ RUN node -e " \ 'CONFIG_BASH_EVENTS=' + c.paths.bashEvents, \ 'CONFIG_AUTOMATION_DB=' + c.paths.automationDb, \ ]; \ - if (extSha) lines.push('CONFIG_EXTENSIONS_REF=' + extSha); \ require('fs').writeFileSync('/tmp/defaults.env', lines.join('\n') + '\n'); \ " diff --git a/docker/entrypoint.sh b/docker/entrypoint.sh index c05935106e..a48f0e2ace 100644 --- a/docker/entrypoint.sh +++ b/docker/entrypoint.sh @@ -128,11 +128,6 @@ export AUTOMATION_AGENT_SERVER_URL="${AUTOMATION_AGENT_SERVER_URL:-http://127.0. # OH_EXTRA_PYTHON_PATH: config.canvasToolsDir. export OH_EXTRA_PYTHON_PATH="${OH_EXTRA_PYTHON_PATH:-/opt/agent-canvas/tools}" -# Pin the public-skills catalog to the same @openhands/extensions commit that -# the frontend bundle was built with. The agent-server SDK skips network polling -# when EXTENSIONS_REF is already present in its local cache. -export EXTENSIONS_REF="${EXTENSIONS_REF:-${CONFIG_EXTENSIONS_REF:-}}" - # Track child PIDs so we can clean up on exit. PIDS=() diff --git a/package-lock.json b/package-lock.json index 4c4a1d9c30..31e38ba260 100644 --- a/package-lock.json +++ b/package-lock.json @@ -12,7 +12,7 @@ "@heroui/react": "2.8.10", "@microlink/react-json-view": "1.31.20", "@monaco-editor/react": "4.7.0", - "@openhands/extensions": "github:OpenHands/extensions#8a6690071e0e9229ae70b30ff0352aff9e6fca21", + "@openhands/extensions": "0.3.0", "@openhands/typescript-client": "1.24.3", "@react-router/node": "7.17.0", "@react-router/serve": "7.17.0", @@ -733,7 +733,6 @@ "version": "1.6.0", "resolved": "https://registry.npmjs.org/@colors/colors/-/colors-1.6.0.tgz", "integrity": "sha512-Ir+AOibqzrIsL6ajt3Rz3LskB7OiMVHqltZmspbW/TJuTVuyOMirVqAkjfY6JISiLHgyNqicAC8AyHHGzNd/dA==", - "dev": true, "license": "MIT", "engines": { "node": ">=0.1.90" @@ -883,7 +882,6 @@ "version": "2.0.8", "resolved": "https://registry.npmjs.org/@dabh/diagnostics/-/diagnostics-2.0.8.tgz", "integrity": "sha512-R4MSXTVnuMzGD7bzHdW2ZhhdPC/igELENcq5IjEverBvq5hn1SXCWcsi6eSsdWP0/Ur+SItRRjAktmdoX/8R/Q==", - "dev": true, "license": "MIT", "dependencies": { "@so-ric/colorspace": "^1.1.6", @@ -3470,9 +3468,9 @@ "license": "MIT" }, "node_modules/@openhands/extensions": { - "version": "0.0.0", - "resolved": "git+ssh://git@github.com/OpenHands/extensions.git#8a6690071e0e9229ae70b30ff0352aff9e6fca21", - "integrity": "sha512-GjqvmlDWOWnWRtvqM65B3xWKphhBD3nuckOcWKZsrt2TithPg5YTptbj5mdcbm2MWV9gIk5uo8wknjGUIR994g==", + "version": "0.3.0", + "resolved": "https://registry.npmjs.org/@openhands/extensions/-/extensions-0.3.0.tgz", + "integrity": "sha512-6xewbbmrDG6GbnI/Gga+o3+cw1jUpPkEW4wJKtL1mncL9f1qpI93YOBQMlrdQnu0JMRnLAMYSeURksqToUw2rw==", "license": "MIT", "engines": { "node": ">=18.20.0" @@ -6211,7 +6209,6 @@ "version": "1.1.6", "resolved": "https://registry.npmjs.org/@so-ric/colorspace/-/colorspace-1.1.6.tgz", "integrity": "sha512-/KiKkpHNOBgkFJwu9sh48LkHSMYGyuTcSFK/qMBdnOAlrRJzRSXAOFB5qwzaVQuDl8wAvHVMkaASQDReTahxuw==", - "dev": true, "license": "MIT", "dependencies": { "color": "^5.0.2", @@ -6222,7 +6219,6 @@ "version": "5.0.3", "resolved": "https://registry.npmjs.org/color/-/color-5.0.3.tgz", "integrity": "sha512-ezmVcLR3xAVp8kYOm4GS45ZLLgIE6SPAFoduLr6hTDajwb3KZ2F46gulK3XpcwRFb5KKGCSezCBAY4Dw4HsyXA==", - "dev": true, "license": "MIT", "dependencies": { "color-convert": "^3.1.3", @@ -6236,7 +6232,6 @@ "version": "3.1.3", "resolved": "https://registry.npmjs.org/color-convert/-/color-convert-3.1.3.tgz", "integrity": "sha512-fasDH2ont2GqF5HpyO4w0+BcewlhHEZOFn9c1ckZdHpJ56Qb7MHhH/IcJZbBGgvdtwdwNbLvxiBEdg336iA9Sg==", - "dev": true, "license": "MIT", "dependencies": { "color-name": "^2.0.0" @@ -6249,7 +6244,6 @@ "version": "2.1.0", "resolved": "https://registry.npmjs.org/color-name/-/color-name-2.1.0.tgz", "integrity": "sha512-1bPaDNFm0axzE4MEAzKPuqKWeRaT43U/hyxKPBdqTfmPF+d6n7FSoTFxLVULUJOmiLp01KjhIPPH+HrXZJN4Rg==", - "dev": true, "license": "MIT", "engines": { "node": ">=12.20" @@ -6259,7 +6253,6 @@ "version": "2.1.4", "resolved": "https://registry.npmjs.org/color-string/-/color-string-2.1.4.tgz", "integrity": "sha512-Bb6Cq8oq0IjDOe8wJmi4JeNn763Xs9cfrBcaylK1tPypWzyoy2G3l90v9k64kjphl/ZJjPIShFztenRomi8WTg==", - "dev": true, "license": "MIT", "dependencies": { "color-name": "^2.0.0" @@ -7174,7 +7167,6 @@ "version": "1.3.5", "resolved": "https://registry.npmjs.org/@types/triple-beam/-/triple-beam-1.3.5.tgz", "integrity": "sha512-6WaYesThRMCl19iryMYP7/x2OVgCtbIVflDGFpWnb9irXI3UjYE4AzmYuiUKY1AJstGijoY+MgUszMgRxIYTYw==", - "dev": true, "license": "MIT" }, "node_modules/@types/trusted-types": { @@ -8572,7 +8564,6 @@ "version": "3.2.6", "resolved": "https://registry.npmjs.org/async/-/async-3.2.6.tgz", "integrity": "sha512-htCUDlxyyCLMgaM3xXg0C0LW2xqfuQ6p05pCEIsXuyQ+a1koYKTuBMzRNwmybfLgvJDMd0r1LTn4+E0Ti6C2AA==", - "dev": true, "license": "MIT" }, "node_modules/async-function": { @@ -9836,7 +9827,6 @@ "version": "2.0.0", "resolved": "https://registry.npmjs.org/enabled/-/enabled-2.0.0.tgz", "integrity": "sha512-AKrN98kuwOzMIdAizXGI86UFBoo26CL21UM763y1h/GMSJ4/OHU9k2YlsmBpyScFo/wbLzWQJBMCW4+IO3/+OQ==", - "dev": true, "license": "MIT" }, "node_modules/encodeurl": { @@ -11011,7 +11001,6 @@ "version": "4.2.3", "resolved": "https://registry.npmjs.org/fecha/-/fecha-4.2.3.tgz", "integrity": "sha512-OP2IUU6HeYKJi3i0z4A19kHMQoLVs4Hc+DPqqxI2h/DPZHTm/vjsfC6P0b4jCMy14XizLBqvndQ+UilD7707Jw==", - "dev": true, "license": "MIT" }, "node_modules/fflate": { @@ -11037,7 +11026,6 @@ "version": "0.6.1", "resolved": "https://registry.npmjs.org/file-stream-rotator/-/file-stream-rotator-0.6.1.tgz", "integrity": "sha512-u+dBid4PvZw17PmDeRcNOtCP9CCK/9lRN2w+r1xIS7yOL9JFrIBKTvrYsxT4P0pGtThYTn++QS5ChHaUov3+zQ==", - "dev": true, "license": "MIT", "dependencies": { "moment": "^2.29.1" @@ -11131,7 +11119,6 @@ "version": "1.1.0", "resolved": "https://registry.npmjs.org/fn.name/-/fn.name-1.1.0.tgz", "integrity": "sha512-GRnmB5gPyJpAhTQdSZTSp9uaPSvl09KoYcMQtsB9rQoOmzs9dH6ffeccH+Z+cv6P68Hu5bC6JjRh4Ah/mHSNRw==", - "dev": true, "license": "MIT" }, "node_modules/follow-redirects": { @@ -12453,7 +12440,6 @@ "version": "2.0.1", "resolved": "https://registry.npmjs.org/is-stream/-/is-stream-2.0.1.tgz", "integrity": "sha512-hFoiJiTl63nn+kstHGBtewWSKnQLpyb155KHheA1l39uvtO9nWIop1p3udqPcUd/xbF1VLMO4n7OI6p7RbngDg==", - "dev": true, "license": "MIT", "engines": { "node": ">=8" @@ -12832,7 +12818,6 @@ "version": "2.0.0", "resolved": "https://registry.npmjs.org/kuler/-/kuler-2.0.0.tgz", "integrity": "sha512-Xq9nH7KlWZmXAtodXDDRE7vs6DU1gTU8zYDHDiWLSip45Egwq3plLHzPn27NgvzL2r1LMPC1vdqh98sQxtqj4A==", - "dev": true, "license": "MIT" }, "node_modules/language-subtag-registry": { @@ -13266,7 +13251,6 @@ "version": "2.7.0", "resolved": "https://registry.npmjs.org/logform/-/logform-2.7.0.tgz", "integrity": "sha512-TFYA4jnP7PVbmlBIfhlSe+WKxs9dklXMTEGcBCIvLhE/Tn3H6Gk1norupVW7m5Cnd4bLcr08AytbyV/xj7f/kQ==", - "dev": true, "license": "MIT", "dependencies": { "@colors/colors": "1.6.0", @@ -14454,7 +14438,6 @@ "version": "2.30.1", "resolved": "https://registry.npmjs.org/moment/-/moment-2.30.1.tgz", "integrity": "sha512-uEmtNhbDOrWPFS+hdjFCBfy9f2YoyzRpwcl+DqpC6taX21FzsTLQVbMV/W7PzNSX6x/bhC1zA3c2UQ5NzH6how==", - "dev": true, "license": "MIT", "engines": { "node": "*" @@ -14765,7 +14748,6 @@ "version": "3.0.0", "resolved": "https://registry.npmjs.org/object-hash/-/object-hash-3.0.0.tgz", "integrity": "sha512-RSn9F68PjH9HqtltsSnqYC1XXoWe9Bju5+213R98cNGttag9q9yAOTzdbsqvIa7aNm5WffBZFpWYr2aWrklWAw==", - "dev": true, "license": "MIT", "engines": { "node": ">= 6" @@ -14907,7 +14889,6 @@ "version": "1.0.0", "resolved": "https://registry.npmjs.org/one-time/-/one-time-1.0.0.tgz", "integrity": "sha512-5DXOiRKwuSEcQ/l0kGCF6Q3jcADFv5tSmRaJck/OqkVFcOzutB134KRSfF0xDrL39MNnqxbHBbUUcjZIhTgb2g==", - "dev": true, "license": "MIT", "dependencies": { "fn.name": "1.x.x" @@ -15903,7 +15884,6 @@ "version": "3.6.2", "resolved": "https://registry.npmjs.org/readable-stream/-/readable-stream-3.6.2.tgz", "integrity": "sha512-9u/sniCrY3D5WdsERHzHE4G2YCXqoG5FTHUiCC4SIbr6XcLZBY05ya9EKjYek9O5xOAwjGq+1JdGBAS7Q9ScoA==", - "dev": true, "license": "MIT", "dependencies": { "inherits": "^2.0.3", @@ -16421,7 +16401,6 @@ "version": "2.5.0", "resolved": "https://registry.npmjs.org/safe-stable-stringify/-/safe-stable-stringify-2.5.0.tgz", "integrity": "sha512-b3rppTKm9T+PsVCBEOUR46GWI7fdOs00VKZ1+9c1EWDaDMvjQc6tUwuFyIprgGgTcWoVHSKrU8H31ZHA2e0RHA==", - "dev": true, "license": "MIT", "engines": { "node": ">=10" @@ -16892,7 +16871,6 @@ "version": "0.0.10", "resolved": "https://registry.npmjs.org/stack-trace/-/stack-trace-0.0.10.tgz", "integrity": "sha512-KGzahc7puUKkzyMt+IqAep+TVNbKP+k2Lmwhub39m1AsTSkaDutx56aDCo+HLDzf/D26BIHTJWNiTG1KAJiQCg==", - "dev": true, "license": "MIT", "engines": { "node": "*" @@ -16952,7 +16930,6 @@ "version": "1.3.0", "resolved": "https://registry.npmjs.org/string_decoder/-/string_decoder-1.3.0.tgz", "integrity": "sha512-hkRX8U1WjJFd8LsDJ2yQ/wWWxaopEsABU1XfkM8A+j0+85JAGppt16cr1Whg6KIbb4okU6Mql6BOj+uup/wKeA==", - "dev": true, "license": "MIT", "dependencies": { "safe-buffer": "~5.2.0" @@ -17321,7 +17298,6 @@ "version": "1.0.0", "resolved": "https://registry.npmjs.org/text-hex/-/text-hex-1.0.0.tgz", "integrity": "sha512-uuVGNWzgJ4yhRaNSiubPY7OjISw4sw4E5Uv0wbjp+OzcbmVU/rsT8ujgcXJhn9ypzsgr5vlzpPqP+MBBKcGvbg==", - "dev": true, "license": "MIT" }, "node_modules/tinybench": { @@ -17467,7 +17443,6 @@ "version": "1.4.1", "resolved": "https://registry.npmjs.org/triple-beam/-/triple-beam-1.4.1.tgz", "integrity": "sha512-aZbgViZrg1QNcG+LULa7nhZpJTZSLm/mXnHXnbAbjmN5aSa0y7V+wvv6+4WaBtpISJzThKy+PIPxc1Nq1EJ9mg==", - "dev": true, "license": "MIT", "engines": { "node": ">= 14.0.0" @@ -17932,7 +17907,6 @@ "version": "1.0.2", "resolved": "https://registry.npmjs.org/util-deprecate/-/util-deprecate-1.0.2.tgz", "integrity": "sha512-EPD5q1uXyFxJpCrLnCc1nHnq3gOa6DZBocAIiI2TaSCA7VCJ1UJDMagCzIkXNsUYfD1daK//LTEQ8xiIbrHtcw==", - "dev": true, "license": "MIT" }, "node_modules/utils-merge": { @@ -18552,7 +18526,6 @@ "version": "3.19.0", "resolved": "https://registry.npmjs.org/winston/-/winston-3.19.0.tgz", "integrity": "sha512-LZNJgPzfKR+/J3cHkxcpHKpKKvGfDZVPS4hfJCc4cCG0CgYzvlD6yE/S3CIL/Yt91ak327YCpiF/0MyeZHEHKA==", - "dev": true, "license": "MIT", "dependencies": { "@colors/colors": "^1.6.0", @@ -18575,7 +18548,6 @@ "version": "5.0.0", "resolved": "https://registry.npmjs.org/winston-daily-rotate-file/-/winston-daily-rotate-file-5.0.0.tgz", "integrity": "sha512-JDjiXXkM5qvwY06733vf09I2wnMXpZEhxEVOSPenZMii+g7pcDcTBt2MRugnoi8BwVSuCT2jfRXBUy+n1Zz/Yw==", - "dev": true, "license": "MIT", "dependencies": { "file-stream-rotator": "^0.6.1", @@ -18594,7 +18566,6 @@ "version": "4.9.0", "resolved": "https://registry.npmjs.org/winston-transport/-/winston-transport-4.9.0.tgz", "integrity": "sha512-8drMJ4rkgaPo1Me4zD/3WLfI/zPdA9o2IipKODunnGDcuqbHwjsbB79ylv04LCGGzU0xQ6vTznOMpQGaLhhm6A==", - "dev": true, "license": "MIT", "dependencies": { "logform": "^2.7.0", diff --git a/package.json b/package.json index 717774f2a8..197203563f 100644 --- a/package.json +++ b/package.json @@ -23,7 +23,7 @@ "@heroui/react": "2.8.10", "@microlink/react-json-view": "1.31.20", "@monaco-editor/react": "4.7.0", - "@openhands/extensions": "github:OpenHands/extensions#8a6690071e0e9229ae70b30ff0352aff9e6fca21", + "@openhands/extensions": "0.3.0", "@openhands/typescript-client": "1.24.3", "@react-router/node": "7.17.0", "@react-router/serve": "7.17.0", diff --git a/playwright.mock-llm-docker.config.ts b/playwright.mock-llm-docker.config.ts index 0af78594d1..8cebe62cff 100644 --- a/playwright.mock-llm-docker.config.ts +++ b/playwright.mock-llm-docker.config.ts @@ -29,6 +29,7 @@ import { defineConfig, devices } from "@playwright/test"; import { randomBytes } from "node:crypto"; +import { mkdirSync } from "node:fs"; import { resolve } from "node:path"; // ── Docker image ──────────────────────────────────────────────────────── @@ -86,6 +87,23 @@ if (!process.env.MOCK_LLM_AGENT_URL) { process.env.MOCK_LLM_AGENT_URL = MOCK_LLM_URL; } +// ── Skill test support ──────────────────────────────────────────────── +// The skill tests create git repos and user skill files on the host. +// We volume-mount these into the container so the agent-server can see them. + +// Project skills: host creates repos here, container sees them at a fixed path. +const SKILL_REPOS_HOST_DIR = resolve(".tmp/mock-llm-skill-repos"); +const SKILL_REPOS_CONTAINER_DIR = "/tmp/mock-llm-skill-repos"; +mkdirSync(SKILL_REPOS_HOST_DIR, { recursive: true }); +process.env.MOCK_LLM_SKILL_REPOS_CONTAINER_DIR = SKILL_REPOS_CONTAINER_DIR; + +// User skills: host creates skill files here, container mounts them at +// the agent-server's expected ~/.openhands/skills/ path. +const USER_SKILLS_HOST_DIR = resolve(".tmp/mock-llm-user-skills"); +const USER_SKILLS_CONTAINER_DIR = "/home/openhands/.openhands/skills"; +mkdirSync(USER_SKILLS_HOST_DIR, { recursive: true }); +process.env.MOCK_LLM_USER_SKILLS_HOST_DIR = USER_SKILLS_HOST_DIR; + // ── ACP test support ────────────────────────────────────────────────── // The mock ACP server script lives on the host. We volume-mount it into // the container and tell the test which container-side paths to use when @@ -169,6 +187,10 @@ export default defineConfig({ // Mount the mock ACP server script so the agent-server inside // Docker can spawn it as an ACP subprocess. `-v ${MOCK_ACP_HOST_PATH}:${MOCK_ACP_CONTAINER_PATH}:ro`, + // Mount skill test directories so the agent-server can access + // repos and user skills created by the host-side test code. + `-v ${SKILL_REPOS_HOST_DIR}:${SKILL_REPOS_CONTAINER_DIR}`, + `-v ${USER_SKILLS_HOST_DIR}:${USER_SKILLS_CONTAINER_DIR}`, `-e PORT=${INGRESS_PORT}`, `-e SESSION_API_KEY=${sessionApiKey}`, `-e OH_SESSION_API_KEYS_0=${sessionApiKey}`, diff --git a/scripts/dev-safe.mjs b/scripts/dev-safe.mjs index 223db70068..910dafe3f5 100644 --- a/scripts/dev-safe.mjs +++ b/scripts/dev-safe.mjs @@ -36,32 +36,6 @@ const SHARED_DEFAULTS = JSON.parse( ), ); -/** - * Extract the pinned commit SHA for @openhands/extensions from package.json. - * Returns the 40-char hex SHA when the dependency is a git+https URL with a - * commit hash fragment (e.g. "git+https://…#62594156…"), null otherwise. - * @returns {string | null} - */ -function getExtensionsRef() { - try { - const pkg = JSON.parse( - readFileSync( - path.join(__dev_safe_dirname, "..", "package.json"), - "utf-8", - ), - ); - const url = - (pkg.dependencies ?? pkg.devDependencies ?? {})["@openhands/extensions"] ?? - ""; - return url.match(/#([0-9a-f]{40})$/i)?.[1] ?? null; - } catch { - return null; - } -} - -/** Pinned extensions commit SHA derived from package.json, or null if not pinned. */ -export const DEFAULT_EXTENSIONS_REF = getExtensionsRef(); - const DEFAULT_BACKEND_PORT = SHARED_DEFAULTS.ports.agentServer; const DEFAULT_VITE_PORT = 3001; const DEFAULT_WAIT_TIMEOUT_MS = 30_000; @@ -737,13 +711,6 @@ export function buildAgentServerEnv(config) { // Make the host tools/ directory importable so the agent-server can // resolve modules listed in tool_module_qualnames (e.g. canvas_ui_tool). OH_EXTRA_PYTHON_PATH: config.canvasToolsDir, - // Tell the agent-server which extensions commit to use for the public - // skills catalog. Derived from the @openhands/extensions pin in - // package.json; the SDK skips network polling when it already has this - // SHA cached. Only injected when the caller has not already set it. - ...(DEFAULT_EXTENSIONS_REF && !process.env.EXTENSIONS_REF - ? { EXTENSIONS_REF: DEFAULT_EXTENSIONS_REF } - : {}), }; } diff --git a/src/api/agent-server-adapter.ts b/src/api/agent-server-adapter.ts index bab96893a1..b08b782e41 100644 --- a/src/api/agent-server-adapter.ts +++ b/src/api/agent-server-adapter.ts @@ -1,4 +1,5 @@ import { ACP_SETTINGS_KEYS } from "@openhands/typescript-client"; +import { SKILLS_CATALOG } from "@openhands/extensions/skills"; import { DEFAULT_SETTINGS } from "#/services/settings"; import { ExecutionStatus } from "#/types/agent-server/core"; import { Settings, SettingsValue } from "#/types/settings"; @@ -8,10 +9,7 @@ import { } from "#/constants/acp-providers"; import { getAgentServerClientOptions } from "./agent-server-client-options"; import { isAgentServerToolAvailable } from "./agent-server-compatibility"; -import { - getAgentServerWorkingDir, - shouldLoadPublicSkills, -} from "./agent-server-config"; +import { getAgentServerWorkingDir } from "./agent-server-config"; import { getEffectiveLocalBackend } from "./backend-registry/active-store"; import { buildAuthHeaders } from "./backend-registry/auth"; import { @@ -527,11 +525,75 @@ function buildInitialMessage( }; } +/** + * Shape of a bundled skill entry passed to the agent-server SDK via + * `agent_context.skills`. Mirrors the SDK's `Skill` model fields that + * the server uses for trigger matching, activation, and system-prompt + * injection. + */ +interface BundledSkill { + name: string; + content: string; + trigger: { type: "keyword"; keywords: string[] } | null; + source: "public"; + description: string | null; + is_agentskills_format: true; + license?: string; + compatibility?: string; +} + +/** + * Convert the bundled `SKILLS_CATALOG` entries into the SDK `Skill` JSON + * shape so the agent-server can perform trigger matching, skill activation, + * and system-prompt injection without cloning the extensions repo. + * + * The SDK discriminates triggers via `{ type: "keyword", keywords: [...] }`. + * Skills with no triggers get `trigger: null` (always-active / on-demand). + */ +function buildBundledSkills(): BundledSkill[] { + return SKILLS_CATALOG.map((entry) => { + const trigger: BundledSkill["trigger"] = + entry.triggers?.length > 0 + ? { type: "keyword", keywords: entry.triggers } + : null; + + return { + name: entry.name, + content: entry.content, + trigger, + source: "public" as const, + description: entry.description ?? null, + is_agentskills_format: true as const, + ...(entry.license ? { license: entry.license } : {}), + ...(entry.compatibility ? { compatibility: entry.compatibility } : {}), + }; + }); +} + function buildAgentContext(agentSettings: SettingsRecord): SettingsRecord { const runtimeServicesSuffix = buildRuntimeServicesSystemSuffix(); + const existingContext = toRecord(agentSettings.agent_context); + + // Merge bundled public skills with any skills already present in the + // agent context (e.g. user-defined skills set via the settings API). + const existingSkills = Array.isArray(existingContext.skills) + ? (existingContext.skills as SettingsRecord[]) + : []; + const mergedSkills = [...existingSkills, ...buildBundledSkills()]; + return { - ...toRecord(agentSettings.agent_context), - load_public_skills: shouldLoadPublicSkills(), + ...existingContext, + // Public skills are bundled at build time from the @openhands/extensions + // npm package and passed directly in agent_context.skills. Setting + // load_public_skills to false tells the agent-server SDK to skip its own + // extensions-repo clone β€” the frontend is the sole source of public + // skills now. + // + // Migration: the former VITE_LOAD_PUBLIC_SKILLS env var was removed + // because bundled skills have no clone latency. Users who previously set + // VITE_LOAD_PUBLIC_SKILLS=false to avoid clone delays no longer need it. + skills: mergedSkills, + load_public_skills: false, load_user_skills: true, load_project_skills: true, ...(runtimeServicesSuffix diff --git a/src/api/agent-server-config.ts b/src/api/agent-server-config.ts index 06c33d8216..a161a34924 100644 --- a/src/api/agent-server-config.ts +++ b/src/api/agent-server-config.ts @@ -112,10 +112,6 @@ export function getAgentServerHeaders(): Record { return sessionApiKey ? { "X-Session-API-Key": sessionApiKey } : {}; } -export function shouldLoadPublicSkills(): boolean { - return import.meta.env.VITE_LOAD_PUBLIC_SKILLS !== "false"; -} - export function isAuthRequired(): boolean { return ( import.meta.env.VITE_AUTH_REQUIRED === "true" || diff --git a/src/api/skills-service.ts b/src/api/skills-service.ts index baa1b258c4..f2b9fa7298 100644 --- a/src/api/skills-service.ts +++ b/src/api/skills-service.ts @@ -1,31 +1,65 @@ import { SkillsClient } from "@openhands/typescript-client/clients"; +import { + SKILLS_CATALOG, + type SkillCatalogEntry, +} from "@openhands/extensions/skills"; import { SkillInfo } from "#/types/settings"; import { getAgentServerWorkingDir } from "./agent-server-config"; import { getActiveBackend } from "./backend-registry/active-store"; import { fetchCloudSkills } from "./cloud/skills-service.api"; import { getAgentServerClientOptions } from "./agent-server-client-options"; +function catalogEntryToSkillInfo(entry: SkillCatalogEntry): SkillInfo { + return { + name: entry.name, + type: "knowledge", + source: "public", + description: entry.description, + triggers: entry.triggers, + content: entry.content, + license: entry.license ?? null, + compatibility: entry.compatibility ?? null, + }; +} + +/** + * Public skills loaded from the `@openhands/extensions` npm package. + * + * This is an **immutable build-time snapshot**: the catalog is baked into the + * bundle at `npm run build` / `vite build` time and does not change at + * runtime. Updating the catalog requires bumping the `@openhands/extensions` + * dependency and rebuilding. + */ +const PUBLIC_SKILLS: SkillInfo[] = SKILLS_CATALOG.map(catalogEntryToSkillInfo); + class SkillsService { static async getSkills(projectDir?: string): Promise { if (getActiveBackend().backend.kind === "cloud") { return fetchCloudSkills(); } - // Always load public skills on the global Skills settings page so the user - // sees the available catalog even on a fresh dev environment with no local - // user/project skills. Conversation creation paths still gate on - // shouldLoadPublicSkills() to keep new-conversation latency low. - const response = await new SkillsClient( - getAgentServerClientOptions(), - ).getSkills({ - load_public: true, - load_user: true, - load_project: true, - load_org: false, - project_dir: projectDir ?? getAgentServerWorkingDir(), - }); + // Public skills come from the bundled @openhands/extensions npm package β€” + // no agent-server round-trip or GitHub fetch needed. Only ask the agent- + // server for user and project skills so local .agents/skills/ content is + // still picked up. + let localSkills: SkillInfo[] = []; + try { + const response = await new SkillsClient( + getAgentServerClientOptions(), + ).getSkills({ + load_public: false, + load_user: true, + load_project: true, + load_org: false, + project_dir: projectDir ?? getAgentServerWorkingDir(), + }); + localSkills = (response.skills ?? []) as SkillInfo[]; + } catch { + // Agent-server may not support the skills endpoint or may be + // unreachable; fall back to the bundled public catalog alone. + } - return (response.skills ?? []) as SkillInfo[]; + return [...localSkills, ...PUBLIC_SKILLS]; } } diff --git a/tests/e2e/mock-llm/mock-llm-automation.spec.ts b/tests/e2e/mock-llm/mock-llm-automation.spec.ts index 7555ae9759..e1cb2651c1 100644 --- a/tests/e2e/mock-llm/mock-llm-automation.spec.ts +++ b/tests/e2e/mock-llm/mock-llm-automation.spec.ts @@ -254,8 +254,8 @@ test.describe("mock-LLM automation lifecycle", () => { ].join(" "); // ⚠️ Padding response (index 0): - // When public skills are loaded (VITE_LOAD_PUBLIC_SKILLS !== "false"), - // the agent-server's skill-activation pipeline makes one internal LLM + // Public skills are bundled from @openhands/extensions at build time. + // The agent-server's skill-activation pipeline makes one internal LLM // call to decide which skills to inject before the agent loop starts. // Our user message mentions "automation", which matches the // openhands-automation skill, triggering this internal call. diff --git a/tests/e2e/mock-llm/mock-llm-image-upload.spec.ts b/tests/e2e/mock-llm/mock-llm-image-upload.spec.ts index fa1cc9a1df..29fae1e472 100644 --- a/tests/e2e/mock-llm/mock-llm-image-upload.spec.ts +++ b/tests/e2e/mock-llm/mock-llm-image-upload.spec.ts @@ -91,8 +91,8 @@ test.describe("mock-LLM image upload", () => { // called and the conversation completed successfully. // // ⚠️ Padding note (mirrors the automation test's pattern): - // When public skills are loaded (VITE_LOAD_PUBLIC_SKILLS !== "false"), - // the agent-server may make one internal LLM call for skill-analysis + // Public skills are bundled from @openhands/extensions at build time. + // The agent-server may make one internal LLM call for skill-analysis // before the agent loop starts, consuming one trajectory slot. // Turn 0 is a throwaway empty response that absorbs this internal call. // Turn 1 is the agent's actual reply (IMAGE_REPLY_TOKEN). diff --git a/tests/e2e/mock-llm/mock-llm-preset-automation.spec.ts b/tests/e2e/mock-llm/mock-llm-preset-automation.spec.ts index 9d1be1d39d..ae14efa438 100644 --- a/tests/e2e/mock-llm/mock-llm-preset-automation.spec.ts +++ b/tests/e2e/mock-llm/mock-llm-preset-automation.spec.ts @@ -2,9 +2,11 @@ * Mock-LLM E2E test: preset automation card β†’ slash command β†’ skill activation. * * The `slack-standup-digest` skill ships in the public OpenHands extensions - * repo with `triggers: ["/standup-digest:setup"]`. The frontend sends - * `load_public_skills: true` in every conversation's `agent_context`, so - * the SDK's `AgentContext._load_auto_skills` picks it up automatically. + * repo with `triggers: ["/standup-digest:setup"]`. The frontend bundles + * public skills from the `@openhands/extensions` npm package and passes + * them directly in `agent_context.skills` at conversation-start, so the + * SDK's trigger matching activates them without the agent-server needing + * to clone the extensions repo (`load_public_skills: false`). * * Two tests: * 1. **Card flow**: configure a dummy Slack MCP server so the automation diff --git a/tests/e2e/mock-llm/mock-llm-skills.spec.ts b/tests/e2e/mock-llm/mock-llm-skills.spec.ts new file mode 100644 index 0000000000..23420c84ad --- /dev/null +++ b/tests/e2e/mock-llm/mock-llm-skills.spec.ts @@ -0,0 +1,441 @@ +/** + * Mock-LLM E2E tests: skill loading from project and user directories. + * + * These tests verify that the SDK's skill loading machinery works + * end-to-end with the real agent-server stack: + * + * 1. **Project skills** from `{workspace}/.agents/skills/` are loaded + * alongside bundled public skills and trigger on matching keywords. + * The test creates a standalone git repo with the skill committed, + * then creates a conversation via API pointing at that repo. The + * agent-server creates a worktree from the repo, and since the skill + * is committed, `load_project_skills` finds it in the worktree. + * + * 2. **User skills** from `~/.openhands/skills/` are loaded and trigger + * on matching keywords. + * + * 3. **Skill deletion**: removing a skill file means it is NOT loaded + * in subsequent conversations. + * + * All tests create ephemeral SKILL.md files with unique trigger keywords, + * send a message containing those keywords, and verify `activated_skills` + * appears in the conversation events API. + */ + +import { test, expect, type APIRequestContext } from "@playwright/test"; +import { + BACKEND_URL, + SESSION_API_KEY, + MOCK_LLM_AGENT_URL, + seedLocalStorage, + routeSessionApiKey, + dismissAnalyticsModal, + waitForNonUserMessageText, + deleteConversation, + registerTrajectory, + activateTrajectory, + resetMockLLM, + setChatInput, + waitForPath, + getConversationIdFromURL, +} from "./utils/mock-llm-helpers"; +import { + createProjectSkillRepo, + removeProjectSkillRepo, + writeUserSkill, + removeUserSkill, + userSkillExists, + userSkillDirExists, +} from "./utils/skill-test-helpers"; + +/** + * Configure the mock LLM profile. Inlined from `ensureMockLLMProfile` + * to work around a CI-specific TS6/Node24 type inference bug (TS2345) + * where importing that function alongside `skill-test-helpers` causes + * TypeScript to incorrectly resolve its signature. + */ +async function configureMockLLM( + request: APIRequestContext, + model = "openai/mock-test-model", +) { + const settingsResp = await request.get(`${BACKEND_URL}/api/settings`, { + headers: { + "X-Session-API-Key": SESSION_API_KEY, + "X-Expose-Secrets": "encrypted", + }, + }); + if (settingsResp.ok()) { + const settings = (await settingsResp.json()) as Record; + const llm = ( + settings?.agent_settings as Record | undefined + )?.llm as Record | undefined; + if (llm?.model === model && llm?.base_url === MOCK_LLM_AGENT_URL) return; + } + const patchResp = await request.patch(`${BACKEND_URL}/api/settings`, { + headers: { + "X-Session-API-Key": SESSION_API_KEY, + "Content-Type": "application/json", + }, + data: { + agent_settings_diff: { + llm: { + model, + api_key: "mock-api-key-for-testing", + base_url: MOCK_LLM_AGENT_URL, + }, + }, + }, + }); + expect( + patchResp.ok(), + `PATCH /api/settings failed: ${patchResp.status()}`, + ).toBe(true); +} + +/** + * Register a workspace on the agent-server so it appears in the UI dropdown. + */ +async function addWorkspaceToServer( + request: APIRequestContext, + name: string, + path: string, +): Promise { + const resp = await request.post(`${BACKEND_URL}/api/workspaces`, { + headers: { + "X-Session-API-Key": SESSION_API_KEY, + "Content-Type": "application/json", + }, + data: { + workspaces: [{ id: `e2e-${name}`, name, path }], + }, + }); + expect( + resp.ok(), + `POST /api/workspaces failed: ${resp.status()} ${await resp.text()}`, + ).toBe(true); +} + +/** + * Remove a workspace from the agent-server. + */ +async function removeWorkspaceFromServer( + request: APIRequestContext, + path: string, +): Promise { + await request.delete(`${BACKEND_URL}/api/workspaces`, { + headers: { "X-Session-API-Key": SESSION_API_KEY }, + params: { path }, + }); +} + +// ── Shared constants ───────────────────────────────────────────────── + +const REPLY_TOKEN = "SKILLS_E2E_REPLY_OK"; + +// ── Tests ───────────────────────────────────────────────────────────── + +test.describe.configure({ mode: "serial" }); + +test.describe("skill loading: project, user, and deletion", () => { + const conversationIds = new Set(); + + // Unique skill names to avoid collisions with real skills + const PROJECT_SKILL_NAME = "e2e-test-project-skill"; + const PROJECT_SKILL_TRIGGER = "xyzzy-project-e2e-test"; + + const USER_SKILL_NAME = "e2e-test-user-skill"; + const USER_SKILL_TRIGGER = "xyzzy-user-e2e-test"; + + test.beforeEach(async ({ page }) => { + await seedLocalStorage(page); + }); + + test.afterEach(async ({ page, request }) => { + const match = page.url().match(/\/conversations\/([^/?#]+)/); + if (match?.[1]) conversationIds.add(decodeURIComponent(match[1])); + + for (const id of Array.from(conversationIds)) { + try { + await deleteConversation(request, id); + conversationIds.delete(id); + } catch { + // best-effort cleanup + } + } + await resetMockLLM(request).catch(() => {}); + }); + + // Track the agent-side workspace path for cleanup (may differ from + // the host path in Docker mode where volumes are mounted) + let projectSkillAgentDir = ""; + + test.afterAll(async ({ request }) => { + removeProjectSkillRepo(PROJECT_SKILL_NAME); + removeUserSkill(USER_SKILL_NAME); + if (projectSkillAgentDir) { + await removeWorkspaceFromServer(request, projectSkillAgentDir).catch( + () => {}, + ); + } + }); + + // ── Test 1: Project skill loaded from workspace ────────────────── + // + // Creates a standalone git repo with the skill committed, registers it + // as a workspace on the agent-server, then uses the UI to select that + // workspace and send a message. The agent-server creates a worktree + // from the repo, and the committed skill file is present in it for + // `load_project_skills` to discover. + + test("project skill in workspace/.agents/skills/ triggers on matching keyword", async ({ + page, + request, + }) => { + await configureMockLLM(request); + + // Create a git repo with the skill committed + const { agentDir } = await test.step( + "create git repo with project skill", + () => { + return createProjectSkillRepo( + PROJECT_SKILL_NAME, + PROJECT_SKILL_TRIGGER, + ); + }, + ); + projectSkillAgentDir = agentDir; + + // Register the workspace on the server using the agent-side path + // (same as hostDir in npm mode, container mount path in Docker mode) + await test.step("register workspace on server", async () => { + await addWorkspaceToServer(request, "skill-test-repo", agentDir); + }); + + // Trajectory: padding for skill-analysis + agent reply + await registerTrajectory(request, "project-skill", [ + { text: "" }, + { text: `Skill test complete. ${REPLY_TOKEN}` }, + ]); + await activateTrajectory(request, "project-skill"); + + await routeSessionApiKey(page); + await page.goto("/", { waitUntil: "domcontentloaded" }); + await dismissAnalyticsModal(page); + + // Select the workspace through the UI + await test.step("select workspace from UI", async () => { + // Click "Open workspace" to open the dialog + await page.getByTestId("open-workspace-button").click(); + + // Click the workspace dropdown and select our workspace + const dropdown = page.getByTestId("workspace-dropdown"); + await dropdown.click(); + // The workspace name is "skill-test-repo" β€” click the matching option + await page.getByText("skill-test-repo").click(); + + // Click the Confirm button to set the workspace + await page.getByTestId("workspace-launch-button").click(); + }); + + await test.step("send message with project skill trigger", async () => { + await setChatInput( + page, + `Please help me with ${PROJECT_SKILL_TRIGGER} setup`, + ); + await page.getByTestId("submit-button").click(); + await waitForPath(page, /\/conversations\/.+/, 30_000); + }); + + const conversationId = getConversationIdFromURL(page); + conversationIds.add(conversationId); + + await test.step("verify agent reply", async () => { + await waitForNonUserMessageText(page, REPLY_TOKEN, 45_000); + }); + + await test.step("verify project skill activated", async () => { + await expect + .poll( + async () => { + const resp = await request.get( + `${BACKEND_URL}/api/conversations/${encodeURIComponent(conversationId)}/events/search`, + { + headers: { "X-Session-API-Key": SESSION_API_KEY }, + params: { limit: "50" }, + }, + ); + if (!resp.ok()) return `HTTP ${resp.status()}`; + const body = (await resp.json()) as { items?: unknown[] }; + for (const item of body.items ?? []) { + const e = item as Record; + const skills = + (e.activated_skills as string[] | undefined) ?? + (e.activated_microagents as string[] | undefined); + if (skills?.includes(PROJECT_SKILL_NAME)) return "FOUND"; + } + return "NOT_FOUND"; + }, + { + message: `expected "${PROJECT_SKILL_NAME}" in activated_skills`, + intervals: [1_000, 2_000, 3_000, 5_000], + timeout: 25_000, + }, + ) + .toBe("FOUND"); + }); + }); + + // ── Test 2: User skill loaded from ~/.openhands/skills/ ────────── + + test("user skill in ~/.openhands/skills/ triggers on matching keyword", async ({ + page, + request, + }) => { + await configureMockLLM(request); + + await test.step("create user skill file", () => { + writeUserSkill(USER_SKILL_NAME, USER_SKILL_TRIGGER); + expect(userSkillExists(USER_SKILL_NAME)).toBe(true); + }); + + await registerTrajectory(request, "user-skill", [ + { text: "" }, + { text: `Skill test complete. ${REPLY_TOKEN}` }, + ]); + await activateTrajectory(request, "user-skill"); + + await routeSessionApiKey(page); + await page.goto("/", { waitUntil: "domcontentloaded" }); + await dismissAnalyticsModal(page); + + await test.step("send message with user skill trigger", async () => { + await setChatInput( + page, + `I need help with ${USER_SKILL_TRIGGER} configuration`, + ); + await page.getByTestId("submit-button").click(); + await waitForPath(page, /\/conversations\/.+/, 30_000); + }); + + const conversationId = getConversationIdFromURL(page); + conversationIds.add(conversationId); + + await test.step("verify agent reply", async () => { + await waitForNonUserMessageText(page, REPLY_TOKEN, 45_000); + }); + + await test.step("verify user skill activated", async () => { + await expect + .poll( + async () => { + const resp = await request.get( + `${BACKEND_URL}/api/conversations/${encodeURIComponent(conversationId)}/events/search`, + { + headers: { "X-Session-API-Key": SESSION_API_KEY }, + params: { limit: "50" }, + }, + ); + if (!resp.ok()) return `HTTP ${resp.status()}`; + const body = (await resp.json()) as { items?: unknown[] }; + for (const item of body.items ?? []) { + const e = item as Record; + const skills = + (e.activated_skills as string[] | undefined) ?? + (e.activated_microagents as string[] | undefined); + if (skills?.includes(USER_SKILL_NAME)) return "FOUND"; + } + return "NOT_FOUND"; + }, + { + message: `expected "${USER_SKILL_NAME}" in activated_skills`, + intervals: [1_000, 2_000, 3_000, 5_000], + timeout: 25_000, + }, + ) + .toBe("FOUND"); + }); + }); + + // ── Test 3: Deleted user skill not loaded in new conversation ──── + + test("deleting a user skill removes it from subsequent conversations", async ({ + page, + request, + }) => { + await configureMockLLM(request); + + await test.step("delete user skill file", () => { + removeUserSkill(USER_SKILL_NAME); + expect(userSkillDirExists(USER_SKILL_NAME)).toBe(false); + }); + + // Padding for skill-analysis (public skills are still loaded from the npm + // package, so the agent-server still makes a skill-analysis LLM call) + await registerTrajectory(request, "deleted-skill", [ + { text: "" }, + { text: `No skill triggered. ${REPLY_TOKEN}` }, + ]); + await activateTrajectory(request, "deleted-skill"); + + await routeSessionApiKey(page); + await page.goto("/", { waitUntil: "domcontentloaded" }); + await dismissAnalyticsModal(page); + + await test.step( + "send message with deleted skill trigger keyword", + async () => { + await setChatInput( + page, + `Help me with ${USER_SKILL_TRIGGER} please`, + ); + await page.getByTestId("submit-button").click(); + await waitForPath(page, /\/conversations\/.+/, 30_000); + }, + ); + + const conversationId = getConversationIdFromURL(page); + conversationIds.add(conversationId); + + await test.step("verify agent reply", async () => { + await waitForNonUserMessageText(page, REPLY_TOKEN, 45_000); + }); + + await test.step("verify deleted skill NOT activated", async () => { + // The agent reply already appeared in the UI (verified above), so the + // conversation completed. Poll the events API and verify no event + // contains the deleted skill in its activated_skills list. + await expect + .poll( + async () => { + const resp = await request.get( + `${BACKEND_URL}/api/conversations/${encodeURIComponent(conversationId)}/events/search`, + { + headers: { "X-Session-API-Key": SESSION_API_KEY }, + params: { limit: "100" }, + }, + ); + if (!resp.ok()) return `HTTP ${resp.status()}`; + const body = (await resp.json()) as { items?: unknown[] }; + const items = body.items ?? []; + if (items.length === 0) return "NO_EVENTS"; + + for (const item of items) { + const e = item as Record; + const skills = + (e.activated_skills as string[] | undefined) ?? + (e.activated_microagents as string[] | undefined); + if (skills?.includes(USER_SKILL_NAME)) + return `UNEXPECTEDLY_FOUND`; + } + return "VERIFIED"; + }, + { + message: `verifying "${USER_SKILL_NAME}" NOT in activated_skills`, + intervals: [1_000, 2_000, 3_000], + timeout: 15_000, + }, + ) + .toBe("VERIFIED"); + }); + }); +}); diff --git a/tests/e2e/mock-llm/mock-llm-ui-regressions.spec.ts b/tests/e2e/mock-llm/mock-llm-ui-regressions.spec.ts index 759c4fa3c9..7042363c76 100644 --- a/tests/e2e/mock-llm/mock-llm-ui-regressions.spec.ts +++ b/tests/e2e/mock-llm/mock-llm-ui-regressions.spec.ts @@ -137,6 +137,11 @@ async function routePaginationConversation(page: Page) { ); // Stub the event search endpoint with the synthetic paginated events. + // Older-events requests (those with timestamp__lt) are delayed slightly so + // the loading indicator has time to render before the response arrives. + // Without this, React can batch the isLoading trueβ†’false transition into a + // single commit and the DOM element never materialises β€” making the + // "loading-older-events" assertion flaky. await page.route( `**/api/conversations/${PAGINATION_CONVERSATION_ID}/events/search**`, async (route, req) => { @@ -145,7 +150,11 @@ async function routePaginationConversation(page: Page) { return; } const url = new URL(req.url()); + const isOlderPage = url.searchParams.has("timestamp__lt"); const result = searchPaginationEvents(allEvents, url.searchParams); + if (isOlderPage) { + await new Promise((r) => setTimeout(r, 300)); + } await route.fulfill({ status: 200, contentType: "application/json", diff --git a/tests/e2e/mock-llm/scripts/render-mock-llm-report.mjs b/tests/e2e/mock-llm/scripts/render-mock-llm-report.mjs index 31eb8fd85b..fbba4ff46c 100644 --- a/tests/e2e/mock-llm/scripts/render-mock-llm-report.mjs +++ b/tests/e2e/mock-llm/scripts/render-mock-llm-report.mjs @@ -9,7 +9,8 @@ * --output mock-llm-report.md \ * [--workflow-url ] \ * [--commit ] \ - * [--artifact-url ] + * [--artifact-url ] \ + * [--new-files ] */ import { existsSync, readFileSync, writeFileSync } from "node:fs"; @@ -51,10 +52,12 @@ function loadResults(path) { } } -function collectTests(suites, parents = []) { +function collectTests(suites, parents = [], parentFile = "") { const tests = []; for (const suite of suites ?? []) { const titles = [...parents, suite.title].filter(Boolean); + // Playwright's JSON reporter sets `file` on each suite/spec + const suiteFile = suite.file || parentFile; for (const spec of suite.specs ?? []) { for (const test of spec.tests ?? []) { const results = test.results ?? []; @@ -65,6 +68,7 @@ function collectTests(suites, parents = []) { ); tests.push({ title: [...titles, spec.title].filter(Boolean).join(" β€Ί "), + file: spec.file || suiteFile, status: lastResult?.status ?? (spec.ok ? "passed" : "unknown"), durationMs: duration, retryCount: Math.max(0, results.length - 1), @@ -72,7 +76,7 @@ function collectTests(suites, parents = []) { }); } } - tests.push(...collectTests(suite.suites, titles)); + tests.push(...collectTests(suite.suites, titles, suiteFile)); } return tests; } @@ -141,7 +145,15 @@ function overallIcon(status) { // ── Report rendering ─────────────────────────────────────────────────── -function renderReport({ tests, workflowUrl, commit, artifactUrl, title, markerMeta }) { +function renderReport({ + tests, + workflowUrl, + commit, + artifactUrl, + title, + newFiles, + markerMeta, +}) { const status = overallStatus(tests); const icon = overallIcon(status); const passed = tests.filter((t) => t.status === "passed").length; @@ -153,6 +165,24 @@ function renderReport({ tests, workflowUrl, commit, artifactUrl, title, markerMe const wasKilledMidSuite = markerMeta?.status === "in_progress" && markerMeta.total > markerMeta.completed; + // Determine which tests are new (from newly added spec files). + // Playwright's JSON file paths are relative to testDir (e.g. "mock-llm-skills.spec.ts") + // while --new-files paths are repo-relative (e.g. "tests/e2e/mock-llm/mock-llm-skills.spec.ts"). + // Match by basename or suffix in either direction. + const newFileSet = new Set(newFiles ?? []); + const basename = (p) => p.split("/").pop(); + const isNewTest = (t) => + newFileSet.size > 0 && + t.file && + [...newFileSet].some( + (nf) => + t.file === nf || + basename(t.file) === basename(nf) || + nf.endsWith(`/${t.file}`) || + t.file.endsWith(`/${nf}`), + ); + const newCount = tests.filter(isNewTest).length; + const lines = []; // Header β€” use πŸ›‘ when killed mid-suite so it's visually distinct @@ -164,6 +194,7 @@ function renderReport({ tests, workflowUrl, commit, artifactUrl, title, markerMe const parts = [`**${passed}/${total} passed**`]; if (failed) parts.push(`**${failed} failed**`); if (skipped) parts.push(`${skipped} skipped`); + if (newCount) parts.push(`πŸ†• ${newCount} new`); if (wasKilledMidSuite) { const notRun = markerMeta.total - markerMeta.completed; parts.push(`⚠️ **${notRun} not run** (process killed at ${markerMeta.completed}/${markerMeta.total})`); @@ -181,6 +212,25 @@ function renderReport({ tests, workflowUrl, commit, artifactUrl, title, markerMe lines.push(""); } + // New-tests callout (prominent, above the table) + if (newCount > 0) { + const newTests = tests.filter(isNewTest); + // Group new tests by spec file + const byFile = new Map(); + for (const t of newTests) { + const key = t.file || "unknown"; + if (!byFile.has(key)) byFile.set(key, []); + byFile.get(key).push(t); + } + lines.push(`> **🟒 ${newCount} new test${newCount === 1 ? "" : "s"} added in this PR**`); + for (const [file, fileTests] of byFile) { + for (const t of fileTests) { + lines.push(`> - ${statusIcon(t.status)} \`${file}\` β€Ί ${t.title.replace(/^.*β€Ί /, "")}`); + } + } + lines.push(""); + } + // Test results table lines.push("| Status | Test | Duration |"); lines.push("|:------:|------|----------|"); @@ -284,12 +334,21 @@ if (!data || tests.length === 0) { } } +// Parse --new-files: comma-separated list of spec file paths added in this PR +const newFiles = args.new_files + ? args.new_files + .split(",") + .map((f) => f.trim()) + .filter(Boolean) + : []; + const report = renderReport({ tests, workflowUrl: args.workflow_url || "", commit: args.commit || "", artifactUrl: args.artifact_url || "", title: args.title || "", + newFiles, markerMeta, }); diff --git a/tests/e2e/mock-llm/utils/skill-test-helpers.ts b/tests/e2e/mock-llm/utils/skill-test-helpers.ts new file mode 100644 index 0000000000..6a8d1d96cf --- /dev/null +++ b/tests/e2e/mock-llm/utils/skill-test-helpers.ts @@ -0,0 +1,139 @@ +/** + * Helpers for skill loading E2E tests. + * + * File-system operations (create/remove SKILL.md files) and API + * assertions (verify activated_skills on events) are separated here + * to avoid type-resolution conflicts between node built-in imports + * and @playwright/test types in the same file (TypeScript 6 / Node 24). + * + * Docker support: When running against a Docker container, the agent-server + * filesystem is isolated from the host. The Playwright config sets env vars + * for the container-side paths so the test can register the correct paths + * with the agent-server while creating files on the host (volume-mounted). + */ + +import { resolve, join } from "path"; +import { mkdirSync, writeFileSync, rmSync, existsSync } from "fs"; +import { execSync } from "child_process"; +import { homedir } from "os"; + +// ── Paths ──────────────────────────────────────────────────────────── + +/** STATE_DIR matches playwright.mock-llm.config.ts */ +export const STATE_DIR = resolve(".tmp/mock-llm-state"); + +/** + * Root directory for skill-test workspace git repos (HOST-side). + * Each call to `createProjectSkillRepo` creates a self-contained git repo + * here with the skill file already committed, so the agent-server's + * worktree machinery picks it up (worktrees only contain committed content). + */ +export const SKILL_REPOS_DIR = resolve(".tmp/mock-llm-skill-repos"); + +/** + * The path the agent-server sees for skill repos. + * In npm mode this is the same as SKILL_REPOS_DIR (same filesystem). + * In Docker mode this is the container-side mount point set by the config. + */ +export const SKILL_REPOS_AGENT_DIR = + process.env.MOCK_LLM_SKILL_REPOS_CONTAINER_DIR ?? SKILL_REPOS_DIR; + +/** + * User-level skills directory β€” HOST-side (for file creation/removal). + * In Docker mode, we use a local temp dir that is volume-mounted into the + * container at the agent-server's expected `~/.openhands/skills/` path. + */ +export const USER_SKILLS_DIR = process.env.MOCK_LLM_USER_SKILLS_HOST_DIR + ? resolve(process.env.MOCK_LLM_USER_SKILLS_HOST_DIR) + : join(homedir(), ".openhands", "skills"); + +// ── Skill content builders ─────────────────────────────────────────── + +function makeSkillMd( + name: string, + trigger: string, + description: string, +): string { + return [ + "---", + `name: ${name}`, + `description: ${description}`, + "triggers:", + `- ${trigger}`, + "---", + "", + `This is the ${name} skill content for E2E testing.`, + `It should activate when the keyword "${trigger}" appears.`, + ].join("\n"); +} + +/** + * Create a standalone git repo with a project skill committed. + * + * The agent-server creates a git worktree for each conversation from the + * source workspace. Only committed files appear in worktrees, so the skill + * must be committed to the repo for `load_project_skills` to find it. + * + * @returns Object with `hostDir` (absolute host path for file ops) and + * `agentDir` (path the agent-server sees β€” same in npm mode, + * container-side mount in Docker mode). + */ +export function createProjectSkillRepo( + name: string, + trigger: string, + description = "E2E test skill", +): { hostDir: string; agentDir: string } { + const repoDir = join(SKILL_REPOS_DIR, `${name}-repo`); + // Start fresh each time + rmSync(repoDir, { recursive: true, force: true }); + mkdirSync(repoDir, { recursive: true }); + + // Write the skill file + const skillDir = join(repoDir, ".agents", "skills", name); + mkdirSync(skillDir, { recursive: true }); + writeFileSync( + join(skillDir, "SKILL.md"), + makeSkillMd(name, trigger, description), + ); + + // Initialize as a git repo and commit + const opts = { cwd: repoDir, stdio: "pipe" as const }; + execSync("git init", opts); + execSync('git config user.email "test@test.com"', opts); + execSync('git config user.name "Test"', opts); + execSync("git add -A", opts); + execSync('git commit -m "Add project skill"', opts); + + const hostDir = resolve(repoDir); + const agentDir = join(SKILL_REPOS_AGENT_DIR, `${name}-repo`); + return { hostDir, agentDir }; +} + +/** Remove a project skill repo created by `createProjectSkillRepo`. */ +export function removeProjectSkillRepo(name: string): void { + const repoDir = join(SKILL_REPOS_DIR, `${name}-repo`); + rmSync(repoDir, { recursive: true, force: true }); +} + +export function writeUserSkill( + name: string, + trigger: string, + description = "E2E test user skill", +): void { + const dir = join(USER_SKILLS_DIR, name); + mkdirSync(dir, { recursive: true }); + writeFileSync(join(dir, "SKILL.md"), makeSkillMd(name, trigger, description)); +} + +export function removeUserSkill(name: string): void { + const dir = join(USER_SKILLS_DIR, name); + rmSync(dir, { recursive: true, force: true }); +} + +export function userSkillExists(name: string): boolean { + return existsSync(join(USER_SKILLS_DIR, name, "SKILL.md")); +} + +export function userSkillDirExists(name: string): boolean { + return existsSync(join(USER_SKILLS_DIR, name)); +}