mirror of
https://github.com/OpenHands/OpenHands.git
synced 2026-10-06 15:03:43 +08:00
fix: combine stats.usage_to_metrics into AppConversation.metrics when metrics is unset (#16510)
This commit is contained in:
@@ -918,6 +918,93 @@ describe("toAppConversation", () => {
|
||||
updated_at: "2026-01-01T00:00:00Z",
|
||||
};
|
||||
|
||||
it("combines stats.usage_to_metrics into metrics when the backend doesn't set metrics directly (#16480)", () => {
|
||||
const result = toAppConversation({
|
||||
...baseInfo,
|
||||
stats: {
|
||||
usage_to_metrics: {
|
||||
agent: {
|
||||
model_name: "agent-model",
|
||||
accumulated_cost: 1.5,
|
||||
max_budget_per_task: 10,
|
||||
accumulated_token_usage: {
|
||||
prompt_tokens: 100,
|
||||
completion_tokens: 20,
|
||||
cache_read_tokens: 5,
|
||||
cache_write_tokens: 1,
|
||||
context_window: 8000,
|
||||
per_turn_token: 120,
|
||||
},
|
||||
costs: [],
|
||||
response_latencies: [],
|
||||
token_usages: [],
|
||||
},
|
||||
condenser: {
|
||||
model_name: "condenser-model",
|
||||
accumulated_cost: 0.5,
|
||||
max_budget_per_task: null,
|
||||
accumulated_token_usage: {
|
||||
prompt_tokens: 40,
|
||||
completion_tokens: 10,
|
||||
cache_read_tokens: 0,
|
||||
cache_write_tokens: 0,
|
||||
context_window: 4000,
|
||||
per_turn_token: 50,
|
||||
},
|
||||
costs: [],
|
||||
response_latencies: [],
|
||||
token_usages: [],
|
||||
},
|
||||
},
|
||||
},
|
||||
});
|
||||
|
||||
expect(result.metrics).toEqual({
|
||||
accumulated_cost: 2,
|
||||
max_budget_per_task: 10,
|
||||
accumulated_token_usage: {
|
||||
prompt_tokens: 140,
|
||||
completion_tokens: 30,
|
||||
cache_read_tokens: 5,
|
||||
cache_write_tokens: 1,
|
||||
context_window: 8000,
|
||||
per_turn_token: 120,
|
||||
},
|
||||
});
|
||||
});
|
||||
|
||||
it("prefers backend-provided metrics over stats.usage_to_metrics when both are present", () => {
|
||||
const result = toAppConversation({
|
||||
...baseInfo,
|
||||
metrics: { accumulated_cost: 3, max_budget_per_task: null },
|
||||
stats: {
|
||||
usage_to_metrics: {
|
||||
agent: {
|
||||
model_name: "agent-model",
|
||||
accumulated_cost: 999,
|
||||
max_budget_per_task: null,
|
||||
accumulated_token_usage: null,
|
||||
costs: [],
|
||||
response_latencies: [],
|
||||
token_usages: [],
|
||||
},
|
||||
},
|
||||
},
|
||||
});
|
||||
|
||||
expect(result.metrics?.accumulated_cost).toBe(3);
|
||||
});
|
||||
|
||||
it("defaults metrics to a zero-cost snapshot when neither metrics nor stats are present", () => {
|
||||
const result = toAppConversation({ ...baseInfo });
|
||||
|
||||
expect(result.metrics).toEqual({
|
||||
accumulated_cost: 0,
|
||||
max_budget_per_task: null,
|
||||
accumulated_token_usage: null,
|
||||
});
|
||||
});
|
||||
|
||||
it("falls back to the default title when the backend returns null", () => {
|
||||
const result = toAppConversation({ ...baseInfo, title: null });
|
||||
expect(result.title).toBe("Conversation 372eb");
|
||||
|
||||
@@ -636,6 +636,50 @@ describe("AgentServerConversationService", () => {
|
||||
expect(result.items[0]?.sandbox_status).toBe("PAUSED");
|
||||
});
|
||||
|
||||
it("falls back to stats.usage_to_metrics when searchConversations omits metrics (#16480)", async () => {
|
||||
const searchSpy = vi.fn().mockResolvedValue({
|
||||
items: [
|
||||
{
|
||||
id: "conv-stats-only",
|
||||
created_at: "2024-01-01",
|
||||
updated_at: "2024-01-01",
|
||||
stats: {
|
||||
usage_to_metrics: {
|
||||
default: {
|
||||
model_name: "test-model",
|
||||
accumulated_cost: 1.25,
|
||||
max_budget_per_task: null,
|
||||
accumulated_token_usage: {
|
||||
prompt_tokens: 100,
|
||||
completion_tokens: 50,
|
||||
cache_read_tokens: 0,
|
||||
cache_write_tokens: 0,
|
||||
context_window: 8000,
|
||||
per_turn_token: 150,
|
||||
},
|
||||
costs: [],
|
||||
response_latencies: [],
|
||||
token_usages: [],
|
||||
},
|
||||
},
|
||||
},
|
||||
},
|
||||
],
|
||||
next_page_id: null,
|
||||
});
|
||||
mockConversationClient.mockReturnValue({
|
||||
searchConversations: searchSpy,
|
||||
});
|
||||
|
||||
const result =
|
||||
await AgentServerConversationService.searchConversations(10);
|
||||
|
||||
expect(result.items[0]?.metrics?.accumulated_cost).toBe(1.25);
|
||||
expect(
|
||||
result.items[0]?.metrics?.accumulated_token_usage?.prompt_tokens,
|
||||
).toBe(100);
|
||||
});
|
||||
|
||||
it("preserves the launched Agent Profile through the wire normalizer", async () => {
|
||||
mockHttpGet.mockResolvedValue({
|
||||
data: [
|
||||
|
||||
@@ -22,8 +22,10 @@ import {
|
||||
PluginSpec,
|
||||
AppConversation,
|
||||
AppConversationPage,
|
||||
RuntimeConversationStats,
|
||||
SandboxStatus,
|
||||
} from "./conversation-service/agent-server-conversation-service.types";
|
||||
import { combineUsageMetrics } from "#/utils/conversation-metrics";
|
||||
import SettingsService from "./settings-service/settings-service.api";
|
||||
import { getStoredConversationMetadata } from "./conversation-metadata-store";
|
||||
import LLMSubscriptionService from "./llm-subscription-service";
|
||||
@@ -63,6 +65,12 @@ export interface DirectConversationInfo {
|
||||
per_turn_token?: number;
|
||||
} | null;
|
||||
} | null;
|
||||
/**
|
||||
* Raw per-usage-id LLM stats from the agent-server. Search/list responses
|
||||
* often carry real usage here even when `metrics` above comes back unset;
|
||||
* {@link toAppConversation} combines this as a fallback in that case.
|
||||
*/
|
||||
stats?: RuntimeConversationStats | null;
|
||||
agent?: {
|
||||
/**
|
||||
* Pydantic discriminator from the SDK union: ``"ACPAgent"`` for ACP CLI
|
||||
@@ -375,7 +383,7 @@ export function toAppConversation(
|
||||
}
|
||||
: null,
|
||||
}
|
||||
: null,
|
||||
: combineUsageMetrics(info.stats),
|
||||
created_at: info.created_at,
|
||||
updated_at: info.updated_at,
|
||||
execution_status:
|
||||
|
||||
@@ -66,6 +66,7 @@ import type {
|
||||
AppConversationStartTask,
|
||||
MetricsSnapshot,
|
||||
RuntimeConversationInfo,
|
||||
RuntimeConversationStats,
|
||||
SendMessageRequest,
|
||||
SendMessageResponse,
|
||||
} from "./agent-server-conversation-service.types";
|
||||
@@ -135,6 +136,16 @@ function normalizeMetrics(value: unknown): MetricsSnapshot | null {
|
||||
};
|
||||
}
|
||||
|
||||
// Shallow check only (matches the trust level `getRuntimeConversation` used
|
||||
// before this field was threaded through `DirectConversationInfo`): the
|
||||
// per-usage-id entries are consumed via `combineUsageMetrics`, which already
|
||||
// tolerates missing/malformed fields, so there's no need to validate them here.
|
||||
function normalizeStats(value: unknown): RuntimeConversationStats | null {
|
||||
return isRecord(value)
|
||||
? (value as unknown as RuntimeConversationStats)
|
||||
: null;
|
||||
}
|
||||
|
||||
function normalizeAgent(value: unknown): DirectConversationInfo["agent"] {
|
||||
if (!isRecord(value)) return null;
|
||||
const llm = isRecord(value.llm)
|
||||
@@ -244,6 +255,7 @@ function requireDirectConversationInfo(item: unknown): DirectConversationInfo {
|
||||
execution_status: stringOrNull(item.execution_status),
|
||||
sandbox_status: stringOrNull(item.sandbox_status),
|
||||
metrics: normalizeMetrics(item.metrics),
|
||||
stats: normalizeStats(item.stats),
|
||||
agent: normalizeAgent(item.agent),
|
||||
workspace: normalizeWorkspace(item.workspace),
|
||||
tags: normalizeTags(item.tags),
|
||||
@@ -658,19 +670,14 @@ class AgentServerConversationService {
|
||||
conversationUrl: string | null | undefined,
|
||||
sessionApiKey?: string | null,
|
||||
): Promise<RuntimeConversationInfo> {
|
||||
type RawRuntime = DirectConversationInfo & {
|
||||
stats?: RuntimeConversationInfo["stats"];
|
||||
};
|
||||
|
||||
// Fetch directly from the per-conversation runtime agent-server at conversationUrl.
|
||||
const response = await new ConversationClient(
|
||||
getAgentServerClientOptions({
|
||||
conversationUrl,
|
||||
sessionApiKey,
|
||||
}),
|
||||
).getConversation<RawRuntime>(conversationId);
|
||||
).getConversation<DirectConversationInfo>(conversationId);
|
||||
const data = requireDirectConversationInfo(response);
|
||||
const stats = isRecord(response) ? response.stats : null;
|
||||
|
||||
return {
|
||||
id: data.id,
|
||||
@@ -681,7 +688,7 @@ class AgentServerConversationService {
|
||||
created_at: data.created_at,
|
||||
updated_at: data.updated_at,
|
||||
status: toRuntimeStatus(data.execution_status),
|
||||
stats: isRecord(stats) ? stats : { usage_to_metrics: {} },
|
||||
stats: data.stats ?? { usage_to_metrics: {} },
|
||||
};
|
||||
}
|
||||
|
||||
|
||||
@@ -1,18 +1,17 @@
|
||||
import type {
|
||||
MetricsSnapshot,
|
||||
RuntimeConversationInfo,
|
||||
RuntimeConversationStats,
|
||||
TokenUsage,
|
||||
} from "#/api/conversation-service/agent-server-conversation-service.types";
|
||||
|
||||
/**
|
||||
* TypeScript equivalent of the get_combined_metrics method from the Python SDK
|
||||
* Combines metrics from all LLM usage IDs in the conversation stats
|
||||
* Combines metrics from all LLM usage IDs in a conversation's stats
|
||||
*/
|
||||
export function getCombinedMetrics(
|
||||
conversationInfo: RuntimeConversationInfo,
|
||||
export function combineUsageMetrics(
|
||||
stats: RuntimeConversationStats | null | undefined,
|
||||
): MetricsSnapshot {
|
||||
const { stats } = conversationInfo;
|
||||
|
||||
if (!stats?.usage_to_metrics) {
|
||||
return {
|
||||
accumulated_cost: 0,
|
||||
@@ -72,3 +71,9 @@ export function getCombinedMetrics(
|
||||
accumulated_token_usage: combinedTokenUsage,
|
||||
};
|
||||
}
|
||||
|
||||
export function getCombinedMetrics(
|
||||
conversationInfo: RuntimeConversationInfo,
|
||||
): MetricsSnapshot {
|
||||
return combineUsageMetrics(conversationInfo.stats);
|
||||
}
|
||||
|
||||
Reference in New Issue
Block a user