chore: bump SDK deps (software-agent-sdk 1.44.0, automation 1.9.0, extensions 0.19.0) (#16968)

Co-authored-by: openhands <openhands@all-hands.dev>
This commit is contained in:
Hiep Le
2026-08-27 17:40:12 +00:00
committed by GitHub
co-authored by openhands
parent d104ffdc33
commit 732f866ed6
15 changed files with 249 additions and 34 deletions
@@ -156,23 +156,54 @@ async function listAutomationRuns(
return resp.json();
}
interface AutomationRunRecord {
id: string;
status: string;
conversation_id: string | null;
error_detail?: string | null;
}
/** Statuses a run never leaves (see RunStatus in openhands-automation). */
const TERMINAL_RUN_STATUSES = new Set([
"COMPLETED",
"FAILED",
"CANCELLED",
"SKIPPED",
]);
/**
* Poll until a run reaches the expected status or times out.
*
* Fails fast, quoting the backend's `error_detail`, once every run has ended
* in some other terminal status — polling on would only surface a timeout
* and hide why the run actually stopped.
*/
async function waitForRunStatus(
request: import("@playwright/test").APIRequestContext,
automationId: string,
expectedStatus: string,
timeoutMs = 30_000,
) {
): Promise<AutomationRunRecord> {
const deadline = Date.now() + timeoutMs;
while (Date.now() < deadline) {
const data = await listAutomationRuns(request, automationId);
const runs = data.runs ?? data.items ?? [];
const match = runs.find(
(r: { status: string }) => r.status === expectedStatus,
);
const runs: AutomationRunRecord[] = data.runs ?? data.items ?? [];
const match = runs.find((r) => r.status === expectedStatus);
if (match) return match;
if (
runs.length > 0 &&
runs.every((r) => TERMINAL_RUN_STATUSES.has(r.status))
) {
const summary = runs
.map(
(r) =>
`${r.id}=${r.status}${r.error_detail ? ` (${r.error_detail})` : ""}`,
)
.join(", ");
throw new Error(
`No run reached "${expectedStatus}"; every run already ended: ${summary}`,
);
}
await new Promise((r) => setTimeout(r, 1_000));
}
throw new Error(
@@ -33,6 +33,11 @@ from openhands.sdk.testing import TestLLM, TestLLMExhaustedError
BASH_TOKEN = "MOCK_LLM_E2E_BASH_OK"
REPLY_TOKEN = "MOCK_LLM_E2E_REPLY_OK"
# The user turn the agent-server sends when it pre-flights a saved LLM profile
# (POST /api/profiles/{name}/validate, agent-server >= 1.43). Matched by
# _is_preflight_ping() so the check never touches the scripted trajectory.
PREFLIGHT_PING_TEXT = "ping"
# SDK exception → (HTTP status, OpenAI error type)
ERROR_MAP: dict[type, tuple[int, str]] = {
LLMAuthenticationError: (401, "invalid_api_key"),
@@ -162,6 +167,23 @@ class MockLLMHandler(BaseHTTPRequestHandler):
length = int(self.headers.get("Content-Length", 0))
body = json.loads(self.rfile.read(length)) if length else {}
# ── Profile pre-flight ping ──
# Saving a profile against agent-server >= 1.43 fires a 1-token "ping"
# completion through the submitted config, and the canvas waits at
# most 30 s for the verdict. Answer it here instead of feeding it to
# TestLLM: it would otherwise consume a scripted turn, and once the
# trajectory is exhausted the 500 below makes the SDK retry with
# backoff until the canvas gives up — leaving the profile editor stuck
# on "Validating...". It stays out of the request history, which tests
# read for the conversation's own completions.
if _is_preflight_ping(body):
raw = _preflight_pong(body.get("model"))
if body.get("stream"):
self._send_streaming(raw)
else:
self._send_json(200, raw)
return
# Append to request history for test verification.
# Tests can GET /admin/requests to confirm image content was included.
with self._lock:
@@ -338,6 +360,53 @@ class MockLLMHandler(BaseHTTPRequestHandler):
print(f"[mock-llm] {args[0]}", file=sys.stderr, flush=True)
def _is_preflight_ping(body: dict) -> bool:
"""Match the agent-server's profile pre-flight check.
It is a single user message saying "ping" with ``max_tokens=1`` (see
``profiles_router.validate_profile`` in openhands-agent-server). The text
arrives either as a plain string or as OpenAI content parts.
"""
if body.get("max_tokens") != 1:
return False
messages = body.get("messages")
if not isinstance(messages, list) or len(messages) != 1:
return False
message = messages[0]
if not isinstance(message, dict) or message.get("role") != "user":
return False
content = message.get("content")
if isinstance(content, list):
content = "".join(
part.get("text", "")
for part in content
if isinstance(part, dict) and part.get("type") == "text"
)
return isinstance(content, str) and content.strip() == PREFLIGHT_PING_TEXT
def _preflight_pong(model: str | None) -> dict:
"""Minimal OpenAI-style completion for the pre-flight ping.
Uses the raw shape ``_send_streaming`` also understands, so the reply works
whether or not the caller asked for streaming.
"""
return {
"id": "chatcmpl-mock-preflight",
"object": "chat.completion",
"created": int(time.time()),
"model": model or "mock-preflight",
"choices": [
{
"index": 0,
"message": {"role": "assistant", "content": "pong"},
"finish_reason": "stop",
}
],
"usage": {"prompt_tokens": 1, "completion_tokens": 1, "total_tokens": 2},
}
def _parse_trajectory_turns(raw_turns: list[dict]) -> list[Message | Exception]:
"""Convert JSON turn descriptors into Message objects.
@@ -410,6 +410,21 @@ test.describe("OpenHands provider hidden base_url preservation", () => {
// Advanced view. The value becomes hidden after switching to Basic, but it
// is still part of the profile unless the model changes. ──
await routeSessionApiKey(page);
// agent-server >= 1.43 pre-flights every profile save with a 1-token
// completion through the submitted config. CUSTOM_BASE_URL is a
// placeholder host by design — this test is about the value surviving a
// Basic-view re-save, not about reaching it — so answer the check from
// the browser instead of letting the SDK retry an unreachable host past
// the canvas's 30 s validation budget. Registered after
// routeSessionApiKey(): Playwright routes are LIFO, and that handler
// `continue()`s every backend request it sees first.
await page.route("**/api/profiles/*/validate", (route) =>
route.fulfill({
status: 200,
contentType: "application/json",
body: JSON.stringify({ valid: true, error: null }),
}),
);
await ensureMockLLMAgentProfile(page.request);
await page.goto("/settings/llm", { waitUntil: "domcontentloaded" });
await dismissAnalyticsModal(page);