test: capture Docker E2E container logs (#16660)

Co-authored-by: neubig <neubig@users.noreply.github.com>
Co-authored-by: openhands <openhands@all-hands.dev>
Co-authored-by: Vasco Schiavo <115561717+VascoSch92@users.noreply.github.com>
This commit is contained in:
Graham Neubig
2026-08-17 19:05:21 +00:00
committed by GitHub
co-authored by neubig openhands Vasco Schiavo
parent c367546043
commit d66157e689
3 changed files with 65 additions and 35 deletions
+13 -3
View File
@@ -393,12 +393,20 @@ jobs:
pw_exit=0
fi
# Clean up the Docker container (belt-and-suspenders)
docker ps -q --filter "name=agent-canvas-mock-llm" | xargs -r docker stop 2>/dev/null || true
echo "exit_code=$pw_exit" >> "$GITHUB_OUTPUT"
exit 0
- name: Capture Docker container logs
if: always()
run: |
docker ps -a --filter "name=agent-canvas-mock-llm" --format '{{.Names}}\t{{.Status}}' | tee docker-container-status.txt || true
: > docker-container-logs.txt
for container in $(docker ps -a --filter "name=agent-canvas-mock-llm" --format '{{.Names}}'); do
echo "=== $container ===" >> docker-container-logs.txt
docker logs "$container" >> docker-container-logs.txt 2>&1 || true
done
docker ps -aq --filter "name=agent-canvas-mock-llm" | xargs -r docker rm -f 2>/dev/null || true
# ── Reporting ──────────────────────────────────────────────────────
- name: Upload test artifacts
id: upload_artifacts
@@ -411,6 +419,8 @@ jobs:
path: |
playwright-report-mock-llm-docker/
test-results-mock-llm-docker/
docker-container-status.txt
docker-container-logs.txt
- name: Detect newly added spec files
if: always() && github.event.pull_request.number
+1 -1
View File
@@ -202,7 +202,7 @@ export default defineConfig({
// Stop any leftover container from a previous failed run
`docker rm -f ${CONTAINER_NAME} 2>/dev/null;`,
"exec docker run",
"--rm",
// Keep the container until the workflow captures its logs.
`--name ${CONTAINER_NAME}`,
"--network host",
// Mount the mock ACP server script so the agent-server inside
@@ -87,27 +87,46 @@ async function listAutomations(
);
}
/**
* Wait for a newly-created automation to become visible through the list API.
* Creation and listing are separate backend operations, so the list can lag
* briefly after the create request succeeds.
*/
async function waitForAutomation(
/** Poll the main conversation for the automation ID returned by the create command. */
async function waitForCreatedAutomationId(
request: import("@playwright/test").APIRequestContext,
name: string,
conversationId: string,
timeoutMs = 30_000,
) {
const deadline = Date.now() + timeoutMs;
while (Date.now() < deadline) {
const data = await listAutomations(request, 1);
const automations = data.automations ?? data.items ?? [];
const automation = automations.find(
(candidate: { name: string }) => candidate.name === name,
const response = await request.get(
`${BACKEND_URL}/api/conversations/${encodeURIComponent(conversationId)}/events/search`,
{
headers: { "X-Session-API-Key": SESSION_API_KEY },
params: { limit: "100", sort_order: "TIMESTAMP_DESC" },
},
);
if (automation) return automation;
await new Promise((resolve) => setTimeout(resolve, 1_000));
if (response.ok()) {
const body = (await response.json()) as { items?: unknown[] };
const text = JSON.stringify(body.items ?? []);
const match = text.match(/automation_id\\?":\\?"([0-9a-f-]{36})/i);
if (match) return match[1];
}
await new Promise((resolve) => setTimeout(resolve, 500));
}
throw new Error(`Automation "${name}" was not visible after ${timeoutMs}ms`);
throw new Error(
`Created automation ID was not reported after ${timeoutMs}ms`,
);
}
async function getAutomation(
request: import("@playwright/test").APIRequestContext,
automationId: string,
) {
const response = await request.get(
`${AUTOMATION_API_BASE}/${encodeURIComponent(automationId)}`,
{ headers: { "X-Session-API-Key": SESSION_API_KEY } },
);
expect(response.ok(), `GET automation returned ${response.status()}`).toBe(
true,
);
return response.json();
}
/**
@@ -197,7 +216,8 @@ test.describe.configure({ mode: "serial" });
test.describe("mock-LLM automation lifecycle", () => {
const conversationIds = new Set<string>();
const automationIds = new Set<string>();
/** conversation_id from the completed automation run (set in step 2, verified in step 3) */
/** IDs carried between the serial lifecycle steps. */
let createdAutomationId: string | null = null;
let runConversationId: string | null = null;
test.beforeEach(async ({ page }) => {
@@ -416,8 +436,13 @@ test.describe("mock-LLM automation lifecycle", () => {
// ── Verify: automation was created in the real automation backend ──
await test.step("verify automation was created", async () => {
const created = await waitForAutomation(request, AUTOMATION_NAME);
createdAutomationId = await waitForCreatedAutomationId(
request,
conversationId,
);
const created = await getAutomation(request, createdAutomationId);
automationIds.add(created.id);
expect(created.name).toBe(AUTOMATION_NAME);
expect(created.trigger?.schedule).toBe(CRON_SCHEDULE);
expect(created.enabled).toBe(true);
});
@@ -425,7 +450,8 @@ test.describe("mock-LLM automation lifecycle", () => {
// ── Verify: run completed successfully with a conversation link ──
await test.step("verify run completed with conversation link", async () => {
const automation = await waitForAutomation(request, AUTOMATION_NAME);
expect(createdAutomationId).toBeTruthy();
const automation = await getAutomation(request, createdAutomationId!);
automationIds.add(automation.id);
// Wait for the run to reach COMPLETED. The trajectory includes extra
@@ -514,26 +540,20 @@ test.describe("mock-LLM automation lifecycle", () => {
await test.step("automation card visible on list page", async () => {
await waitForTestId(page, "automations-add-automation", 15_000);
expect(createdAutomationId).toBeTruthy();
// Scope to the automation card (data-testid `automation-card-<id>`) so
// the locator only matches the list entry, not the global sidebar's
// conversation cards. The automation run creates a conversation named
// ``<AUTOMATION_NAME> — <timestamp>`` that the sidebar conversation list
// surfaces, so an unscoped ``getByText(AUTOMATION_NAME)`` resolves to
// multiple elements (the card + every run conversation) and trips
// Playwright strict mode.
const automationCard = page
.locator('[data-testid^="automation-card-"]')
.filter({ hasText: AUTOMATION_NAME });
const automationCard = page.locator(
`[data-testid="automation-card-${createdAutomationId}"]`,
);
await expect(automationCard).toBeVisible({ timeout: 15_000 });
await expect(automationCard).toContainText(AUTOMATION_NAME);
});
await test.step("click through to automation detail page", async () => {
// The automation card is a link — clicking it navigates to /automations/:id.
const automationCard = page
.locator('[data-testid^="automation-card-"]')
.filter({ hasText: AUTOMATION_NAME });
const automationCard = page.locator(
`[data-testid="automation-card-${createdAutomationId}"]`,
);
await automationCard.click();
await waitForPath(page, /\/automations\/.+/, 10_000);