diff --git a/__tests__/api/agent-server-adapter.test.ts b/__tests__/api/agent-server-adapter.test.ts index a403cde75a..aa43678562 100644 --- a/__tests__/api/agent-server-adapter.test.ts +++ b/__tests__/api/agent-server-adapter.test.ts @@ -918,6 +918,93 @@ describe("toAppConversation", () => { updated_at: "2026-01-01T00:00:00Z", }; + it("combines stats.usage_to_metrics into metrics when the backend doesn't set metrics directly (#16480)", () => { + const result = toAppConversation({ + ...baseInfo, + stats: { + usage_to_metrics: { + agent: { + model_name: "agent-model", + accumulated_cost: 1.5, + max_budget_per_task: 10, + accumulated_token_usage: { + prompt_tokens: 100, + completion_tokens: 20, + cache_read_tokens: 5, + cache_write_tokens: 1, + context_window: 8000, + per_turn_token: 120, + }, + costs: [], + response_latencies: [], + token_usages: [], + }, + condenser: { + model_name: "condenser-model", + accumulated_cost: 0.5, + max_budget_per_task: null, + accumulated_token_usage: { + prompt_tokens: 40, + completion_tokens: 10, + cache_read_tokens: 0, + cache_write_tokens: 0, + context_window: 4000, + per_turn_token: 50, + }, + costs: [], + response_latencies: [], + token_usages: [], + }, + }, + }, + }); + + expect(result.metrics).toEqual({ + accumulated_cost: 2, + max_budget_per_task: 10, + accumulated_token_usage: { + prompt_tokens: 140, + completion_tokens: 30, + cache_read_tokens: 5, + cache_write_tokens: 1, + context_window: 8000, + per_turn_token: 120, + }, + }); + }); + + it("prefers backend-provided metrics over stats.usage_to_metrics when both are present", () => { + const result = toAppConversation({ + ...baseInfo, + metrics: { accumulated_cost: 3, max_budget_per_task: null }, + stats: { + usage_to_metrics: { + agent: { + model_name: "agent-model", + accumulated_cost: 999, + max_budget_per_task: null, + accumulated_token_usage: null, + costs: [], + response_latencies: [], + token_usages: [], + }, + }, + }, + }); + + expect(result.metrics?.accumulated_cost).toBe(3); + }); + + it("defaults metrics to a zero-cost snapshot when neither metrics nor stats are present", () => { + const result = toAppConversation({ ...baseInfo }); + + expect(result.metrics).toEqual({ + accumulated_cost: 0, + max_budget_per_task: null, + accumulated_token_usage: null, + }); + }); + it("falls back to the default title when the backend returns null", () => { const result = toAppConversation({ ...baseInfo, title: null }); expect(result.title).toBe("Conversation 372eb"); diff --git a/__tests__/api/agent-server-conversation-service.test.ts b/__tests__/api/agent-server-conversation-service.test.ts index 0dbebf1a81..413f90731c 100644 --- a/__tests__/api/agent-server-conversation-service.test.ts +++ b/__tests__/api/agent-server-conversation-service.test.ts @@ -636,6 +636,50 @@ describe("AgentServerConversationService", () => { expect(result.items[0]?.sandbox_status).toBe("PAUSED"); }); + it("falls back to stats.usage_to_metrics when searchConversations omits metrics (#16480)", async () => { + const searchSpy = vi.fn().mockResolvedValue({ + items: [ + { + id: "conv-stats-only", + created_at: "2024-01-01", + updated_at: "2024-01-01", + stats: { + usage_to_metrics: { + default: { + model_name: "test-model", + accumulated_cost: 1.25, + max_budget_per_task: null, + accumulated_token_usage: { + prompt_tokens: 100, + completion_tokens: 50, + cache_read_tokens: 0, + cache_write_tokens: 0, + context_window: 8000, + per_turn_token: 150, + }, + costs: [], + response_latencies: [], + token_usages: [], + }, + }, + }, + }, + ], + next_page_id: null, + }); + mockConversationClient.mockReturnValue({ + searchConversations: searchSpy, + }); + + const result = + await AgentServerConversationService.searchConversations(10); + + expect(result.items[0]?.metrics?.accumulated_cost).toBe(1.25); + expect( + result.items[0]?.metrics?.accumulated_token_usage?.prompt_tokens, + ).toBe(100); + }); + it("preserves the launched Agent Profile through the wire normalizer", async () => { mockHttpGet.mockResolvedValue({ data: [ diff --git a/src/api/agent-server-adapter.ts b/src/api/agent-server-adapter.ts index ade417018a..af25083398 100644 --- a/src/api/agent-server-adapter.ts +++ b/src/api/agent-server-adapter.ts @@ -22,8 +22,10 @@ import { PluginSpec, AppConversation, AppConversationPage, + RuntimeConversationStats, SandboxStatus, } from "./conversation-service/agent-server-conversation-service.types"; +import { combineUsageMetrics } from "#/utils/conversation-metrics"; import SettingsService from "./settings-service/settings-service.api"; import { getStoredConversationMetadata } from "./conversation-metadata-store"; import LLMSubscriptionService from "./llm-subscription-service"; @@ -63,6 +65,12 @@ export interface DirectConversationInfo { per_turn_token?: number; } | null; } | null; + /** + * Raw per-usage-id LLM stats from the agent-server. Search/list responses + * often carry real usage here even when `metrics` above comes back unset; + * {@link toAppConversation} combines this as a fallback in that case. + */ + stats?: RuntimeConversationStats | null; agent?: { /** * Pydantic discriminator from the SDK union: ``"ACPAgent"`` for ACP CLI @@ -375,7 +383,7 @@ export function toAppConversation( } : null, } - : null, + : combineUsageMetrics(info.stats), created_at: info.created_at, updated_at: info.updated_at, execution_status: diff --git a/src/api/conversation-service/agent-server-conversation-service.api.ts b/src/api/conversation-service/agent-server-conversation-service.api.ts index 507769cd4b..40c29ff86c 100644 --- a/src/api/conversation-service/agent-server-conversation-service.api.ts +++ b/src/api/conversation-service/agent-server-conversation-service.api.ts @@ -66,6 +66,7 @@ import type { AppConversationStartTask, MetricsSnapshot, RuntimeConversationInfo, + RuntimeConversationStats, SendMessageRequest, SendMessageResponse, } from "./agent-server-conversation-service.types"; @@ -135,6 +136,16 @@ function normalizeMetrics(value: unknown): MetricsSnapshot | null { }; } +// Shallow check only (matches the trust level `getRuntimeConversation` used +// before this field was threaded through `DirectConversationInfo`): the +// per-usage-id entries are consumed via `combineUsageMetrics`, which already +// tolerates missing/malformed fields, so there's no need to validate them here. +function normalizeStats(value: unknown): RuntimeConversationStats | null { + return isRecord(value) + ? (value as unknown as RuntimeConversationStats) + : null; +} + function normalizeAgent(value: unknown): DirectConversationInfo["agent"] { if (!isRecord(value)) return null; const llm = isRecord(value.llm) @@ -244,6 +255,7 @@ function requireDirectConversationInfo(item: unknown): DirectConversationInfo { execution_status: stringOrNull(item.execution_status), sandbox_status: stringOrNull(item.sandbox_status), metrics: normalizeMetrics(item.metrics), + stats: normalizeStats(item.stats), agent: normalizeAgent(item.agent), workspace: normalizeWorkspace(item.workspace), tags: normalizeTags(item.tags), @@ -658,19 +670,14 @@ class AgentServerConversationService { conversationUrl: string | null | undefined, sessionApiKey?: string | null, ): Promise { - type RawRuntime = DirectConversationInfo & { - stats?: RuntimeConversationInfo["stats"]; - }; - // Fetch directly from the per-conversation runtime agent-server at conversationUrl. const response = await new ConversationClient( getAgentServerClientOptions({ conversationUrl, sessionApiKey, }), - ).getConversation(conversationId); + ).getConversation(conversationId); const data = requireDirectConversationInfo(response); - const stats = isRecord(response) ? response.stats : null; return { id: data.id, @@ -681,7 +688,7 @@ class AgentServerConversationService { created_at: data.created_at, updated_at: data.updated_at, status: toRuntimeStatus(data.execution_status), - stats: isRecord(stats) ? stats : { usage_to_metrics: {} }, + stats: data.stats ?? { usage_to_metrics: {} }, }; } diff --git a/src/utils/conversation-metrics.ts b/src/utils/conversation-metrics.ts index e6f2ede440..71adc57f2f 100644 --- a/src/utils/conversation-metrics.ts +++ b/src/utils/conversation-metrics.ts @@ -1,18 +1,17 @@ import type { MetricsSnapshot, RuntimeConversationInfo, + RuntimeConversationStats, TokenUsage, } from "#/api/conversation-service/agent-server-conversation-service.types"; /** * TypeScript equivalent of the get_combined_metrics method from the Python SDK - * Combines metrics from all LLM usage IDs in the conversation stats + * Combines metrics from all LLM usage IDs in a conversation's stats */ -export function getCombinedMetrics( - conversationInfo: RuntimeConversationInfo, +export function combineUsageMetrics( + stats: RuntimeConversationStats | null | undefined, ): MetricsSnapshot { - const { stats } = conversationInfo; - if (!stats?.usage_to_metrics) { return { accumulated_cost: 0, @@ -72,3 +71,9 @@ export function getCombinedMetrics( accumulated_token_usage: combinedTokenUsage, }; } + +export function getCombinedMetrics( + conversationInfo: RuntimeConversationInfo, +): MetricsSnapshot { + return combineUsageMetrics(conversationInfo.stats); +}