From 732f866ed66a66df7d09f44d7e7fd3e1daed4544 Mon Sep 17 00:00:00 2001 From: Hiep Le <69354317+hieptl@users.noreply.github.com> Date: Fri, 28 Aug 2026 00:40:12 +0700 Subject: [PATCH] chore: bump SDK deps (software-agent-sdk 1.44.0, automation 1.9.0, extensions 0.19.0) (#16968) Co-authored-by: openhands --- AGENTS.md | 6 +- .../recommended-automations-rail.test.tsx | 10 +++ .../recommended-automations.test.tsx | 8 ++- __tests__/scripts/dev-safe.test.ts | 41 +++++++++-- .../utils/recommended-automation-rail.test.ts | 12 ++++ config/defaults.json | 4 +- docker/entrypoint.sh | 10 ++- package-lock.json | 16 ++--- package.json | 4 +- scripts/check-sdk-version-sync.mjs | 12 ++-- scripts/dev-safe.mjs | 16 ++++- .../automations/mock-llm-automation.spec.ts | 41 +++++++++-- tests/e2e/mock-llm/scripts/mock-llm-server.py | 69 +++++++++++++++++++ .../mock-llm-profile-management.spec.ts | 15 ++++ tools/canvas_ui_tool.py | 19 +++++ 15 files changed, 249 insertions(+), 34 deletions(-) diff --git a/AGENTS.md b/AGENTS.md index 1f06aeb66d..dfbca3be37 100644 --- a/AGENTS.md +++ b/AGENTS.md @@ -249,6 +249,7 @@ you are running inside of — NOT the automation backend. - `POST /admin/trajectory/register` — register a named trajectory (JSON body: `{name, turns}` where each turn is `{tool_call: {name, arguments}}` or `{text: "..."}`) - `POST /admin/trajectory/activate` — activate a previously registered trajectory - `GET /admin/requests` — return the list of all `/v1/chat/completions` request bodies captured since the last reset (used by the image-upload test to verify the image was forwarded to the LLM) + - Profile pre-flight ping: agent-server ≥ 1.43 sends a 1-token `ping` completion whenever an LLM profile is saved (`POST /api/profiles/{name}/validate`, 30 s budget in the canvas). The mock answers it with a canned `pong` instead of feeding it to `TestLLM`, so it neither consumes a scripted turn nor 500s-and-retries past the canvas timeout when the trajectory is exhausted; it is also left out of the `/admin/requests` history. - **Real automation backend**: The automation test uses the production automation backend (started by `bin/agent-canvas.mjs`), NOT a mock server. Terminal `curl` commands from the agent hit the automation API through the ingress proxy at the test's `BACKEND_URL` (default `http://localhost:18300`). Auth uses the `X-Session-API-Key` header matching the stack's session key. - **Test helpers** (`tests/e2e/mock-llm/utils/mock-llm-helpers.ts`): Exports `registerTrajectory()`, `activateTrajectory()`, `resetMockLLM()`, `ensureMockLLMProfile()`, `getMockLLMRequests()` (fetches captured completion bodies from `GET /admin/requests`), `IMAGE_REPLY_TOKEN` + `MINIMAL_PNG_BASE64` (constants for the image-upload spec), ACP helpers (`configureAcpAgent()`, `verifyAcpAgentSettings()`, `resetToOpenHandsAgent()`, `ACP_REPLY_TOKEN`, `MOCK_ACP_SERVER_PATH`), and more. - **Padding response for internal LLM call**: The agent-server makes an internal LLM call (condenser/skill-analysis) before the agent's main loop starts when skills are activated. This consumes one trajectory response. Automation tests prepend a throwaway `{ text: "" }` response as padding. The conversation test does NOT need this because its user message doesn't trigger skill activation. @@ -578,13 +579,14 @@ When adding code that needs a new string, decide up front which rule it falls un - `scripts/dev-safe.mjs` uses `uvx` for temporary agent-server installation — no permanent `uv tool install` needed. Environment variables (highest precedence first): - `OH_AGENT_SERVER_LOCAL_PATH` — absolute path to a local `software-agent-sdk` checkout. Runs the local checkout via `uvx` with `--with-editable` for `openhands-sdk`/`openhands-tools`/`openhands-workspace` and `--reinstall` for `openhands-agent-server`, so SDK edits are picked up on restart. Highest precedence. - `OH_AGENT_SERVER_GIT_REF` — git commit SHA or branch name (takes precedence over version) - - `OH_AGENT_SERVER_VERSION` — specific PyPI version (e.g., "1.42.1") + - `OH_AGENT_SERVER_VERSION` — specific PyPI version (e.g., "1.44.0") - `OH_SECRET_KEY` — secret key for settings encryption; auto-generated and persisted to `~/.openhands/agent-canvas/secret-key.txt` on first run (same file Docker uses), ensuring dev mode and Docker share the same key when both mount the same `~/.openhands` directory. Override with the env var to pin a specific key. - `SESSION_API_KEY` / `OH_SESSION_API_KEYS_0` / `VITE_SESSION_API_KEY` — session API key for agent-server authentication; auto-generated using `crypto.randomBytes(32)` if not set, passed to both agent-server (`OH_SESSION_API_KEYS_0`) and frontend (`VITE_SESSION_API_KEY`) - - Default: released PyPI version `1.42.1` for agent-server SDK libraries + - Default: released PyPI version `1.44.0` for agent-server SDK libraries - Security: launchers generate and persist a 64-character session API key at `~/.openhands/agent-canvas/session-api-key.txt` unless overridden. The agent-server and automation backend share that session key. `OH_SECRET_KEY` protects settings encryption and is persisted separately at `~/.openhands/agent-canvas/secret-key.txt`. - `scripts/dev-safe.mjs` should fail fast if `uvx` cannot be spawned (for example missing PATH entries). +- `tools/` holds Python modules the agent-server can import: `buildAgentServerEnv` exposes the directory through `OH_EXTRA_PYTHON_PATH` (Docker: `/opt/agent-canvas/tools`, see `docker/entrypoint.sh`). `tools/canvas_ui_tool.py` is imported at startup via `--import-modules canvas_ui_tool` (appended by `buildAgentServerCommand` and by both launch lines in `docker/entrypoint.sh`), not only lazily from persisted conversation metadata: besides the legacy `canvas_ui` registration it registers the SDK's builtin `FinishTool`, so `openhands-automation` ≥ 1.9.0 presets — which dispatch remote conversations with `finish_tool_response_schema=TaskOutcome` and advertise the tool as the non-self-registering `openhands.sdk.tool.builtins.finish` — do not fail every run with `ToolDefinition 'FinishTool' is not registered`. Remove that registration once the SDK registers builtins for remote conversations. - `npm run dev` runs the full local stack via `uvx` (agent-server + automation backend + Vite dev server + ingress proxy) with no Docker dependency. `npm run dev:static` does the same but serves a production build of the frontend instead of the Vite dev server. - `scripts/dev-with-automation.mjs` runs the full stack: agent-server, automation backend (both via uvx), frontend server, and ingress proxy. It defaults to Vite when run directly, supports `--static` for an existing build, and supports `--dynamic` so wrappers that default static can opt back into Vite. Uses a standalone ingress proxy (`scripts/ingress.mjs`) to route traffic: - Keep `SIGINT`, `SIGTERM`, and `SIGHUP` wired through the coordinated shutdown handler. Services run in detached process groups on POSIX, so cleanup must use `signalProcessTree()` rather than signaling only the direct child; regression coverage lives in `__tests__/scripts/dev-with-automation.test.ts`. diff --git a/__tests__/components/automations/recommended-automations-rail.test.tsx b/__tests__/components/automations/recommended-automations-rail.test.tsx index 62d4342124..ffdebdf9bf 100644 --- a/__tests__/components/automations/recommended-automations-rail.test.tsx +++ b/__tests__/components/automations/recommended-automations-rail.test.tsx @@ -71,8 +71,13 @@ describe("RecommendedAutomationsRail", () => { "news-digest", "slack-standup-digest", "linear-triage-assistant", + "linear-issue-to-github-pr", + "linear-issue-to-gitlab-mr", + "linear-issue-to-bitbucket-pr", "jira-issue-to-pr", + "jira-issue-to-gitlab-mr", "research-brief-writer", + "jira-issue-to-bitbucket-pr", ]); expect( screen.getByText(I18nKey.RECOMMENDED_AUTOMATIONS$SECTION_LABEL), @@ -125,6 +130,11 @@ describe("RecommendedAutomationsRail", () => { { name: "Linear issue triage assistant" }, { name: "Jira issue to GitHub PR" }, { name: "Research brief writer" }, + { name: "Linear issue to GitHub PR" }, + { name: "Linear issue to GitLab MR" }, + { name: "Linear issue to Bitbucket PR" }, + { name: "Jira issue to GitLab MR" }, + { name: "Jira issue to Bitbucket PR" }, ]} onSelect={vi.fn()} />, diff --git a/__tests__/components/automations/recommended-automations.test.tsx b/__tests__/components/automations/recommended-automations.test.tsx index ab6016ebca..929a519436 100644 --- a/__tests__/components/automations/recommended-automations.test.tsx +++ b/__tests__/components/automations/recommended-automations.test.tsx @@ -213,8 +213,14 @@ describe("recommended automations", () => { "github-repo-monitor", "slack-standup-digest", "linear-triage-assistant", + "linear-issue-to-github-pr", + "linear-issue-to-gitlab-mr", + "linear-issue-to-bitbucket-pr", "jira-issue-to-pr", + "qa-changes", + "jira-issue-to-gitlab-mr", "research-brief-writer", + "jira-issue-to-bitbucket-pr", "upstream-fork-sync", "incident-retrospective-drafter", ]); @@ -240,7 +246,7 @@ describe("recommended automations", () => { expect(betaHeading).toHaveTextContent( I18nKey.RECOMMENDED_AUTOMATIONS$BETA_LABEL, ); - expect(within(betaHeading).getByText("7")).toBeInTheDocument(); + expect(within(betaHeading).getByText("13")).toBeInTheDocument(); const betaSection = screen.getByTestId( "recommended-automations-beta-section", diff --git a/__tests__/scripts/dev-safe.test.ts b/__tests__/scripts/dev-safe.test.ts index fec2a17c44..d2f56872a3 100644 --- a/__tests__/scripts/dev-safe.test.ts +++ b/__tests__/scripts/dev-safe.test.ts @@ -468,20 +468,22 @@ describe("buildAgentServerCommand", () => { // Defaults to the released PyPI version with all SDK packages pinned to same version expect(cmd.args).toEqual([ "--from", - "openhands-agent-server==1.42.1", + "openhands-agent-server==1.44.0", "--with", - "openhands-sdk==1.42.1", + "openhands-sdk==1.44.0", "--with", - "openhands-tools==1.42.1", + "openhands-tools==1.44.0", "--with", - "openhands-workspace==1.42.1", + "openhands-workspace==1.44.0", "--with", "agent-client-protocol<0.11", "--with", "posthog>=6,<7", "agent-server", + "--import-modules", + "canvas_ui_tool", ]); - expect(cmd.source).toBe("PyPI (1.42.1, default)"); + expect(cmd.source).toBe("PyPI (1.44.0, default)"); }); it("uses specific PyPI version when OH_AGENT_SERVER_VERSION is set with all packages pinned", () => { @@ -504,6 +506,8 @@ describe("buildAgentServerCommand", () => { "--with", "posthog>=6,<7", "agent-server", + "--import-modules", + "canvas_ui_tool", ]); expect(cmd.source).toBe("PyPI (1.18.0)"); }); @@ -527,6 +531,8 @@ describe("buildAgentServerCommand", () => { "--with", "posthog>=6,<7", "agent-server", + "--import-modules", + "canvas_ui_tool", ]); expect(cmd.source).toBe("git (feature-branch)"); }); @@ -548,6 +554,8 @@ describe("buildAgentServerCommand", () => { "--with", "posthog>=6,<7", "agent-server", + "--import-modules", + "canvas_ui_tool", ]); expect(cmd.source).toBe("git (abc1234)"); }); @@ -584,6 +592,8 @@ describe("buildAgentServerCommand", () => { "--with", "posthog>=6,<7", "agent-server", + "--import-modules", + "canvas_ui_tool", ]); expect(cmd.source).toBe(`local (${sdk})`); }); @@ -604,6 +614,27 @@ describe("buildAgentServerCommand", () => { expect(cmd.args).not.toContain("openhands-agent-server==1.18.0"); }); + it("passes --import-modules to the agent-server, after the executable, in every source mode", () => { + // The flag must sit after "agent-server" so uvx hands it to the server + // instead of parsing it itself. tools/canvas_ui_tool.py documents why the + // module has to be imported before any conversation is created. + const variants = [ + {}, + { OH_AGENT_SERVER_VERSION: "1.18.0" }, + { OH_AGENT_SERVER_GIT_REF: "feature-branch" }, + { OH_AGENT_SERVER_LOCAL_PATH: "/abs/path/to/software-agent-sdk" }, + ]; + for (const env of variants) { + const { args } = buildAgentServerCommand(env); + const executable = args.indexOf("agent-server"); + expect(executable).toBeGreaterThan(-1); + expect(args.slice(executable + 1)).toEqual([ + "--import-modules", + "canvas_ui_tool", + ]); + } + }); + it("rejects relative OH_AGENT_SERVER_LOCAL_PATH", () => { expect(() => buildAgentServerCommand({ diff --git a/__tests__/utils/recommended-automation-rail.test.ts b/__tests__/utils/recommended-automation-rail.test.ts index c1eb73a792..800fa7f224 100644 --- a/__tests__/utils/recommended-automation-rail.test.ts +++ b/__tests__/utils/recommended-automation-rail.test.ts @@ -70,14 +70,21 @@ describe("recommended automation rail", () => { expect(conversationIds).toEqual([ "slack-standup-digest", "linear-triage-assistant", + "linear-issue-to-github-pr", + "linear-issue-to-gitlab-mr", + "linear-issue-to-bitbucket-pr", "jira-issue-to-pr", + "jira-issue-to-gitlab-mr", "research-brief-writer", + "jira-issue-to-bitbucket-pr", ]); expect(conversationIds).not.toContain("upstream-fork-sync"); expect(conversationIds).not.toContain("incident-retrospective-drafter"); // Beta as of extensions 0.18.0, and set up by a host form rather than a // conversation, so it is in neither group. expect(conversationIds).not.toContain("github-repo-monitor"); + // Same for `qa-changes` (extensions 0.19.0): host form, so neither group. + expect(conversationIds).not.toContain("qa-changes"); expect(isConversationLaunchAutomation(slackStandup)).toBe(true); expect(isConversationLaunchAutomation(upstreamFork)).toBe(false); expect(SETUP_REGISTRY.findById(upstreamFork.id)).not.toBeNull(); @@ -94,6 +101,11 @@ describe("recommended automation rail", () => { { name: "Linear issue triage assistant" }, { name: "Jira issue to GitHub PR" }, { name: "Research brief writer" }, + { name: "Linear issue to GitHub PR" }, + { name: "Linear issue to GitLab MR" }, + { name: "Linear issue to Bitbucket PR" }, + { name: "Jira issue to GitLab MR" }, + { name: "Jira issue to Bitbucket PR" }, ]); expect(flattenRecommendedRailGroups(groups)).toEqual([]); diff --git a/config/defaults.json b/config/defaults.json index 18514780db..ee3063b07e 100644 --- a/config/defaults.json +++ b/config/defaults.json @@ -1,9 +1,9 @@ { "_comment": "Single source of truth for version pins, ports, paths, and defaults shared across the npm and Docker install paths. Read by scripts/dev-safe.mjs, scripts/dev-with-automation.mjs, docker/entrypoint.sh (via generated defaults.env), and .github/workflows/docker.yml.", "versions": { - "agentServer": "1.42.1", + "agentServer": "1.44.0", "agentCanvas": "1.15.0", - "automation": "1.8.0" + "automation": "1.9.0" }, "compatibility": { "minimumAgentServer": "1.28.0" diff --git a/docker/entrypoint.sh b/docker/entrypoint.sh index 2f2a190736..36da7ea00e 100644 --- a/docker/entrypoint.sh +++ b/docker/entrypoint.sh @@ -268,7 +268,11 @@ export AUTOMATION_AGENT_SERVER_URL="${AUTOMATION_AGENT_SERVER_URL:-http://127.0. # Keep the legacy canvas_ui_tool module importable when the agent-server restores # conversations whose persisted metadata still references its module qualname. +# It is also imported at startup below (--import-modules) so its builtin +# FinishTool registration lets automation runs resolve the tool on their +# remote conversations (see the note at the bottom of tools/canvas_ui_tool.py). export OH_EXTRA_PYTHON_PATH="${OH_EXTRA_PYTHON_PATH:-/opt/agent-canvas/tools}" +AGENT_SERVER_IMPORT_MODULES="canvas_ui_tool" # Track child PIDs so we can clean up on exit. PIDS=() @@ -288,10 +292,12 @@ log "Starting agent-server on port $AGENT_SERVER_PORT..." if command -v openhands-agent-server >/dev/null 2>&1; then # Binary build (production image) - openhands-agent-server --port "$AGENT_SERVER_PORT" & + openhands-agent-server --port "$AGENT_SERVER_PORT" \ + --import-modules "$AGENT_SERVER_IMPORT_MODULES" & elif [ -x /agent-server/.venv/bin/python ]; then # Source build (development image) - /agent-server/.venv/bin/python -m openhands.agent_server --port "$AGENT_SERVER_PORT" & + /agent-server/.venv/bin/python -m openhands.agent_server --port "$AGENT_SERVER_PORT" \ + --import-modules "$AGENT_SERVER_IMPORT_MODULES" & else log_error "Cannot find agent-server binary or source venv." exit 1 diff --git a/package-lock.json b/package-lock.json index ef6aeeefa2..b46f47242b 100644 --- a/package-lock.json +++ b/package-lock.json @@ -13,8 +13,8 @@ "@heroui/react": "2.8.10", "@microlink/react-json-view": "1.31.25", "@monaco-editor/react": "4.7.0", - "@openhands/extensions": "0.18.0", - "@openhands/typescript-client": "1.38.1", + "@openhands/extensions": "0.19.0", + "@openhands/typescript-client": "1.39.0", "@react-router/node": "7.18.2", "@react-router/serve": "7.18.2", "@tailwindcss/vite": "4.3.3", @@ -4164,9 +4164,9 @@ "license": "MIT" }, "node_modules/@openhands/extensions": { - "version": "0.18.0", - "resolved": "https://registry.npmjs.org/@openhands/extensions/-/extensions-0.18.0.tgz", - "integrity": "sha512-0cbVueh8wk5OLq+kbEgI5k1sFUaNgahmxXwvOR5+4fpxwc2YnVivdyDzxjZ+SiagUmYi/EgADFVa/5X38e9BzQ==", + "version": "0.19.0", + "resolved": "https://registry.npmjs.org/@openhands/extensions/-/extensions-0.19.0.tgz", + "integrity": "sha512-nT0jTxvVUmiCcKjg2lLYAUUAYFFFbMivZuCjj8qxOIXs6yePVANEpt8HkLe5ViCEweRsbIokfDwcnjQDzVhQ6w==", "license": "MIT", "engines": { "node": ">=18.20.0" @@ -4178,9 +4178,9 @@ } }, "node_modules/@openhands/typescript-client": { - "version": "1.38.1", - "resolved": "https://registry.npmjs.org/@openhands/typescript-client/-/typescript-client-1.38.1.tgz", - "integrity": "sha512-KYpcmgL85uKZ0pvEQAxuwfj0M/8Qq+c6Q2bhaPxpwGayXkDeVWoUxIxomly4vdAqhYseK2pUJiHVoCc81gcf7Q==", + "version": "1.39.0", + "resolved": "https://registry.npmjs.org/@openhands/typescript-client/-/typescript-client-1.39.0.tgz", + "integrity": "sha512-AXRR9bXYYsT87NvkJCB2sLBxAjoGwt6MIQED4g5EufpXaUvY21BDyuDjNpsVtGl6WFnjbvTK1kFytkbxAe0rvw==", "license": "MIT", "dependencies": { "@openrouter/sdk": "^1.2.11", diff --git a/package.json b/package.json index c02318e79a..933a033bf6 100644 --- a/package.json +++ b/package.json @@ -23,8 +23,8 @@ "@heroui/react": "2.8.10", "@microlink/react-json-view": "1.31.25", "@monaco-editor/react": "4.7.0", - "@openhands/extensions": "0.18.0", - "@openhands/typescript-client": "1.38.1", + "@openhands/extensions": "0.19.0", + "@openhands/typescript-client": "1.39.0", "@react-router/node": "7.18.2", "@react-router/serve": "7.18.2", "@tailwindcss/vite": "4.3.3", diff --git a/scripts/check-sdk-version-sync.mjs b/scripts/check-sdk-version-sync.mjs index 7617d553cc..80c67c9c1f 100644 --- a/scripts/check-sdk-version-sync.mjs +++ b/scripts/check-sdk-version-sync.mjs @@ -19,7 +19,7 @@ * * Usage: * node scripts/check-sdk-version-sync.mjs - * EXPECTED_SDK_VERSION=1.42.1 node scripts/check-sdk-version-sync.mjs + * EXPECTED_SDK_VERSION=1.44.0 node scripts/check-sdk-version-sync.mjs * node scripts/check-sdk-version-sync.mjs --check-pypi * * Environment variables: @@ -78,7 +78,7 @@ Triggering from other repos: -H "Authorization: token \$GITHUB_TOKEN" \\ -H "Accept: application/vnd.github.v3+json" \\ https://api.github.com/repos/OpenHands/OpenHands/dispatches \\ - -d '{"event_type": "sdk-version-check", "client_payload": {"version": "1.42.1"}}' + -d '{"event_type": "sdk-version-check", "client_payload": {"version": "1.44.0"}}' `); process.exit(0); } @@ -260,9 +260,9 @@ async function fetchPyPIDependencies(packageName, version) { * Parse PyPI requires_dist array and extract SDK package versions * * PyPI returns dependencies in PEP 508 format like: - * "openhands-sdk>=1.42.1,<2.0.0" - * "openhands-tools==1.42.1" - * "openhands-workspace (>=1.42.1)" + * "openhands-sdk>=1.44.0,<2.0.0" + * "openhands-tools==1.44.0" + * "openhands-workspace (>=1.44.0)" */ function parseSdkVersionsFromRequiresDist(requiresDist) { const versions = {}; @@ -276,7 +276,7 @@ function parseSdkVersionsFromRequiresDist(requiresDist) { } // Extract the version number - look for patterns like: - // ">=1.42.1", "==1.42.1", "(>=1.42.1)", "~=1.42.1" + // ">=1.44.0", "==1.44.0", "(>=1.44.0)", "~=1.44.0" // After the package name and before any comma or closing paren const versionPattern = /[><=~!]+\s*([0-9]+(?:\.[0-9]+)*)/; const match = dep.match(versionPattern); diff --git a/scripts/dev-safe.mjs b/scripts/dev-safe.mjs index 4134b1fa69..e411ca3184 100644 --- a/scripts/dev-safe.mjs +++ b/scripts/dev-safe.mjs @@ -401,6 +401,16 @@ export function validateFrontendDependencies( } } +/** + * Modules the agent-server imports at startup (`--import-modules`). They are + * resolved from `tools/`, which `buildAgentServerEnv` exposes through + * OH_EXTRA_PYTHON_PATH. Importing `canvas_ui_tool` eagerly registers the SDK's + * builtin FinishTool so automation presets (openhands-automation >= 1.9.0) can + * resolve it on the remote conversations they dispatch — see the note at the + * bottom of tools/canvas_ui_tool.py. + */ +export const AGENT_SERVER_IMPORT_MODULES = "canvas_ui_tool"; + /** * Build the uvx command and arguments for running agent-server. * @@ -411,7 +421,7 @@ export function validateFrontendDependencies( * edits are picked up without a manual reinstall. The agent-server itself * is rebuilt from local source on each invocation (--reinstall). * - OH_AGENT_SERVER_GIT_REF: Git commit SHA or branch name - * - OH_AGENT_SERVER_VERSION: Specific PyPI version (e.g., "1.42.1") + * - OH_AGENT_SERVER_VERSION: Specific PyPI version (e.g., "1.44.0") * * If none are set, defaults to the released version specified by * DEFAULT_AGENT_SERVER_VERSION. Set OH_AGENT_SERVER_GIT_REF to use a @@ -516,6 +526,10 @@ export function buildAgentServerCommand(env = process.env) { source = `PyPI (${DEFAULT_AGENT_SERVER_VERSION}, default)`; } + // Everything after the executable name is an agent-server CLI argument. + // Import the registration module before any conversation is created. + uvxArgs.push("--import-modules", AGENT_SERVER_IMPORT_MODULES); + return { command: "uvx", args: uvxArgs, diff --git a/tests/e2e/mock-llm/automations/mock-llm-automation.spec.ts b/tests/e2e/mock-llm/automations/mock-llm-automation.spec.ts index e032614cef..7dad262a52 100644 --- a/tests/e2e/mock-llm/automations/mock-llm-automation.spec.ts +++ b/tests/e2e/mock-llm/automations/mock-llm-automation.spec.ts @@ -156,23 +156,54 @@ async function listAutomationRuns( return resp.json(); } +interface AutomationRunRecord { + id: string; + status: string; + conversation_id: string | null; + error_detail?: string | null; +} + +/** Statuses a run never leaves (see RunStatus in openhands-automation). */ +const TERMINAL_RUN_STATUSES = new Set([ + "COMPLETED", + "FAILED", + "CANCELLED", + "SKIPPED", +]); + /** * Poll until a run reaches the expected status or times out. + * + * Fails fast, quoting the backend's `error_detail`, once every run has ended + * in some other terminal status — polling on would only surface a timeout + * and hide why the run actually stopped. */ async function waitForRunStatus( request: import("@playwright/test").APIRequestContext, automationId: string, expectedStatus: string, timeoutMs = 30_000, -) { +): Promise { const deadline = Date.now() + timeoutMs; while (Date.now() < deadline) { const data = await listAutomationRuns(request, automationId); - const runs = data.runs ?? data.items ?? []; - const match = runs.find( - (r: { status: string }) => r.status === expectedStatus, - ); + const runs: AutomationRunRecord[] = data.runs ?? data.items ?? []; + const match = runs.find((r) => r.status === expectedStatus); if (match) return match; + if ( + runs.length > 0 && + runs.every((r) => TERMINAL_RUN_STATUSES.has(r.status)) + ) { + const summary = runs + .map( + (r) => + `${r.id}=${r.status}${r.error_detail ? ` (${r.error_detail})` : ""}`, + ) + .join(", "); + throw new Error( + `No run reached "${expectedStatus}"; every run already ended: ${summary}`, + ); + } await new Promise((r) => setTimeout(r, 1_000)); } throw new Error( diff --git a/tests/e2e/mock-llm/scripts/mock-llm-server.py b/tests/e2e/mock-llm/scripts/mock-llm-server.py index b53298bc6f..39f0f5df85 100644 --- a/tests/e2e/mock-llm/scripts/mock-llm-server.py +++ b/tests/e2e/mock-llm/scripts/mock-llm-server.py @@ -33,6 +33,11 @@ from openhands.sdk.testing import TestLLM, TestLLMExhaustedError BASH_TOKEN = "MOCK_LLM_E2E_BASH_OK" REPLY_TOKEN = "MOCK_LLM_E2E_REPLY_OK" +# The user turn the agent-server sends when it pre-flights a saved LLM profile +# (POST /api/profiles/{name}/validate, agent-server >= 1.43). Matched by +# _is_preflight_ping() so the check never touches the scripted trajectory. +PREFLIGHT_PING_TEXT = "ping" + # SDK exception → (HTTP status, OpenAI error type) ERROR_MAP: dict[type, tuple[int, str]] = { LLMAuthenticationError: (401, "invalid_api_key"), @@ -162,6 +167,23 @@ class MockLLMHandler(BaseHTTPRequestHandler): length = int(self.headers.get("Content-Length", 0)) body = json.loads(self.rfile.read(length)) if length else {} + # ── Profile pre-flight ping ── + # Saving a profile against agent-server >= 1.43 fires a 1-token "ping" + # completion through the submitted config, and the canvas waits at + # most 30 s for the verdict. Answer it here instead of feeding it to + # TestLLM: it would otherwise consume a scripted turn, and once the + # trajectory is exhausted the 500 below makes the SDK retry with + # backoff until the canvas gives up — leaving the profile editor stuck + # on "Validating...". It stays out of the request history, which tests + # read for the conversation's own completions. + if _is_preflight_ping(body): + raw = _preflight_pong(body.get("model")) + if body.get("stream"): + self._send_streaming(raw) + else: + self._send_json(200, raw) + return + # Append to request history for test verification. # Tests can GET /admin/requests to confirm image content was included. with self._lock: @@ -338,6 +360,53 @@ class MockLLMHandler(BaseHTTPRequestHandler): print(f"[mock-llm] {args[0]}", file=sys.stderr, flush=True) +def _is_preflight_ping(body: dict) -> bool: + """Match the agent-server's profile pre-flight check. + + It is a single user message saying "ping" with ``max_tokens=1`` (see + ``profiles_router.validate_profile`` in openhands-agent-server). The text + arrives either as a plain string or as OpenAI content parts. + """ + if body.get("max_tokens") != 1: + return False + messages = body.get("messages") + if not isinstance(messages, list) or len(messages) != 1: + return False + message = messages[0] + if not isinstance(message, dict) or message.get("role") != "user": + return False + content = message.get("content") + if isinstance(content, list): + content = "".join( + part.get("text", "") + for part in content + if isinstance(part, dict) and part.get("type") == "text" + ) + return isinstance(content, str) and content.strip() == PREFLIGHT_PING_TEXT + + +def _preflight_pong(model: str | None) -> dict: + """Minimal OpenAI-style completion for the pre-flight ping. + + Uses the raw shape ``_send_streaming`` also understands, so the reply works + whether or not the caller asked for streaming. + """ + return { + "id": "chatcmpl-mock-preflight", + "object": "chat.completion", + "created": int(time.time()), + "model": model or "mock-preflight", + "choices": [ + { + "index": 0, + "message": {"role": "assistant", "content": "pong"}, + "finish_reason": "stop", + } + ], + "usage": {"prompt_tokens": 1, "completion_tokens": 1, "total_tokens": 2}, + } + + def _parse_trajectory_turns(raw_turns: list[dict]) -> list[Message | Exception]: """Convert JSON turn descriptors into Message objects. diff --git a/tests/e2e/mock-llm/settings/mock-llm-profile-management.spec.ts b/tests/e2e/mock-llm/settings/mock-llm-profile-management.spec.ts index 24add6cea8..8297d22f06 100644 --- a/tests/e2e/mock-llm/settings/mock-llm-profile-management.spec.ts +++ b/tests/e2e/mock-llm/settings/mock-llm-profile-management.spec.ts @@ -410,6 +410,21 @@ test.describe("OpenHands provider hidden base_url preservation", () => { // Advanced view. The value becomes hidden after switching to Basic, but it // is still part of the profile unless the model changes. ── await routeSessionApiKey(page); + // agent-server >= 1.43 pre-flights every profile save with a 1-token + // completion through the submitted config. CUSTOM_BASE_URL is a + // placeholder host by design — this test is about the value surviving a + // Basic-view re-save, not about reaching it — so answer the check from + // the browser instead of letting the SDK retry an unreachable host past + // the canvas's 30 s validation budget. Registered after + // routeSessionApiKey(): Playwright routes are LIFO, and that handler + // `continue()`s every backend request it sees first. + await page.route("**/api/profiles/*/validate", (route) => + route.fulfill({ + status: 200, + contentType: "application/json", + body: JSON.stringify({ valid: true, error: null }), + }), + ); await ensureMockLLMAgentProfile(page.request); await page.goto("/settings/llm", { waitUntil: "domcontentloaded" }); await dismissAnalyticsModal(page); diff --git a/tools/canvas_ui_tool.py b/tools/canvas_ui_tool.py index 1fe91b1842..228ee2026b 100644 --- a/tools/canvas_ui_tool.py +++ b/tools/canvas_ui_tool.py @@ -13,6 +13,10 @@ The server-side executor is a no-op that returns an acknowledgment. The actual UI effect happens client-side: the frontend watches the WebSocket stream for legacy ``canvas_ui`` and current ``canvas_ui_control`` ActionEvents and dispatches the command. + +The launchers also import this module at agent-server startup +(``--import-modules canvas_ui_tool``) so the builtin ``FinishTool`` registration +at the bottom runs before any conversation is created. """ from collections.abc import Sequence @@ -22,8 +26,10 @@ from pydantic import Field from openhands.sdk import Action, Observation, ToolDefinition from openhands.sdk.tool import ( + FinishTool, ToolAnnotations, ToolExecutor, + list_registered_tools, register_tool, ) @@ -137,3 +143,16 @@ class CanvasUITool(ToolDefinition[CanvasUIAction, CanvasUIObservation]): # restoring their agent and events. Keep the registration until those records # have a server-side migration path. register_tool("canvas_ui", CanvasUITool) + + +# openhands-automation >= 1.9.0 preset entrypoints build their agent with +# get_default_agent(finish_tool_response_schema=TaskOutcome). That registers the +# SDK's builtin FinishTool only inside the entrypoint's own process and +# advertises it to the agent-server as `openhands.sdk.tool.builtins.finish` — a +# module that does not self-register — so the remote conversation every Agent +# Canvas automation run dispatches fails with "ToolDefinition 'FinishTool' is +# not registered". Registering the plain builtin here is enough: resolve_tool() +# strips `response_schema` before FinishTool.create() and re-applies it. Drop +# this once the SDK registers its builtins for remote conversations. +if FinishTool.__name__ not in list_registered_tools(): + register_tool(FinishTool.__name__, FinishTool)