mirror of
https://github.com/Wei-Shaw/sub2api.git
synced 2026-10-07 13:28:39 +08:00
fix(openai): 探测响应未跑完时不再落标为「上游不支持 Responses」
ProbeOpenAIAPIKeyResponsesSupport 的 2xx 分支靠「output 里有没有 function_call」 判定上游工具能力,但这只在响应真的跑完时成立。以下两种响应会被误判并落标: - status=incomplete 且 incomplete_details.reason=max_output_tokens:探测请求自己 只给了 512 的输出预算,推理型模型可能把预算全烧在 reasoning 上,还没轮到 function_call 就被截断。 - status=failed:HTTP 200 携带的失败响应(上游瞬时故障)。 这个标记的代价很高:探测只在账号创建/更新时跑一次,写进 accounts.extra.openai_responses_supported 后不会自动重探,网关会长期把该账号的 /v1/responses 请求改走 /v1/chat/completions。对 Codex 客户端意味着 prompt 缓存 前缀被协议转换打散,issue #5371 实测缓存命中率从 ~85% 掉到 ~17%、成本涨 3.5 倍, 持续 32 小时无人察觉,最后是账号余额耗尽被停调度才自己弹回去。 同一个函数对非 2xx 早已写明「上游偶发故障不应误判为不支持」并保守返回 true, 2xx 分支缺的正是同一份保守。本次新增 responsesProbeVerdictIsConclusive:判据不 成立时保持 unknown 不写标记,与网络层失败、响应体读取失败的处理一致;网关按 「现状即证据」继续走 Responses。 status=completed 却只回 reasoning 的上游(火山方舟 coding/v3 × kimi-k2.6,即探测 最初要抓的目标)不受影响,仍落标为不支持。 另:落标为不支持时补一条 slog.Warn,让这次会长期改变成本结构的降级可被运维看到。 Fixes #5371
This commit is contained in:
@@ -5,6 +5,7 @@ import (
|
||||
"context"
|
||||
"encoding/json"
|
||||
"io"
|
||||
"log/slog"
|
||||
"net/http"
|
||||
"sort"
|
||||
"strings"
|
||||
@@ -24,6 +25,11 @@ const openaiResponsesProbeTimeout = 15 * time.Second
|
||||
// responsesProbeMaxBodyBytes 限制读取探测响应体的字节数,够判定 output 项类型即可。
|
||||
const responsesProbeMaxBodyBytes = 256 * 1024
|
||||
|
||||
// openaiResponsesProbeMaxOutputTokens 是探测请求的输出预算。
|
||||
// 推理型模型可能把预算全烧在 reasoning 上,还没轮到 function_call 就被截断——
|
||||
// 那种响应不能用来判定工具能力,见 responsesProbeVerdictIsConclusive。
|
||||
const openaiResponsesProbeMaxOutputTokens = 512
|
||||
|
||||
// openaiResponsesProbePayload 构造探测用的 Responses 请求体。
|
||||
//
|
||||
// 关键设计:请求携带一个工具并以 tool_choice=required 强制模型调用它。这样
|
||||
@@ -61,7 +67,7 @@ func openaiResponsesProbePayload(modelID string) []byte {
|
||||
},
|
||||
},
|
||||
"tool_choice": "required",
|
||||
"max_output_tokens": 512,
|
||||
"max_output_tokens": openaiResponsesProbeMaxOutputTokens,
|
||||
"stream": false,
|
||||
})
|
||||
return body
|
||||
@@ -175,6 +181,20 @@ func (s *AccountTestService) ProbeOpenAIAPIKeyResponsesSupport(ctx context.Conte
|
||||
return
|
||||
}
|
||||
|
||||
// 本次响应不足以下结论时保持 unknown,与网络层失败、响应体读取失败一致:
|
||||
// 标记一旦写成 false 就会一直粘住(只有下次账号创建/更新才重探),网关会静默
|
||||
// 改走 /v1/chat/completions —— 对 Codex 客户端意味着 prompt 缓存前缀被打散。
|
||||
// 宁可不写,让请求继续走既有的 Responses 路径。
|
||||
if !responsesProbeVerdictIsConclusive(resp.StatusCode, bodyBytes) {
|
||||
logger.LegacyPrintf("service.openai_probe",
|
||||
"probe_inconclusive_keep_unknown: account_id=%d base_url=%s probe_model=%s status=%d response_status=%s reason=%s",
|
||||
accountID, normalizedBaseURL, probeModel, resp.StatusCode,
|
||||
gjson.GetBytes(bodyBytes, "status").String(),
|
||||
gjson.GetBytes(bodyBytes, "incomplete_details.reason").String(),
|
||||
)
|
||||
return
|
||||
}
|
||||
|
||||
supported := decideResponsesProbeSupport(resp.StatusCode, bodyBytes)
|
||||
|
||||
if err := s.accountRepo.UpdateExtra(ctx, accountID, map[string]any{
|
||||
@@ -184,12 +204,55 @@ func (s *AccountTestService) ProbeOpenAIAPIKeyResponsesSupport(ctx context.Conte
|
||||
return
|
||||
}
|
||||
|
||||
if !supported {
|
||||
// 落标为不支持等于把该账号长期钉在 /v1/chat/completions 上,成本与缓存命中率
|
||||
// 都会变化,且不会自动恢复。这条必须能被运维看到(#5371)。
|
||||
slog.Warn(
|
||||
"openai_responses_probe_marked_unsupported",
|
||||
"account_id", accountID,
|
||||
"account_name", account.Name,
|
||||
"base_url", normalizedBaseURL,
|
||||
"probe_model", probeModel,
|
||||
"upstream_status", resp.StatusCode,
|
||||
)
|
||||
}
|
||||
|
||||
logger.LegacyPrintf("service.openai_probe",
|
||||
"probe_done: account_id=%d base_url=%s probe_model=%s status=%d supported=%v",
|
||||
accountID, normalizedBaseURL, probeModel, resp.StatusCode, supported,
|
||||
)
|
||||
}
|
||||
|
||||
// responsesProbeVerdictIsConclusive 判断本次探测响应是否足以对「上游是否支持带工具的
|
||||
// Responses 调用」下结论。
|
||||
//
|
||||
// 2xx 分支靠「output 里有没有 function_call」下结论,但这只在响应真的跑完时成立:
|
||||
//
|
||||
// - status=incomplete 且 incomplete_details.reason=max_output_tokens:探测请求自己
|
||||
// 只给了 openaiResponsesProbeMaxOutputTokens 的预算,推理型模型可能把预算全烧在
|
||||
// reasoning 上,还没轮到 function_call 就被截断。此时「没有 function_call」是探测
|
||||
// 预算不足造成的,不是上游能力缺失。
|
||||
// - status=failed:HTTP 200 携带的失败响应(上游瞬时故障)同样不构成能力证据。
|
||||
//
|
||||
// 其余 2xx 一律可下结论——尤其 status=completed 却只回 reasoning 的上游(火山方舟
|
||||
// coding/v3 × kimi-k2.6),仍按原逻辑判为不支持。
|
||||
//
|
||||
// 非 2xx 的结论只看状态码、不依赖响应内容,恒可下结论。
|
||||
// 缺少 status 字段的响应体(含非 JSON)也按可下结论处理,保持既有行为。
|
||||
func responsesProbeVerdictIsConclusive(status int, body []byte) bool {
|
||||
if status < 200 || status >= 300 {
|
||||
return true
|
||||
}
|
||||
switch strings.TrimSpace(gjson.GetBytes(body, "status").String()) {
|
||||
case "failed":
|
||||
return false
|
||||
case "incomplete":
|
||||
return strings.TrimSpace(gjson.GetBytes(body, "incomplete_details.reason").String()) != "max_output_tokens"
|
||||
default:
|
||||
return true
|
||||
}
|
||||
}
|
||||
|
||||
// isResponsesEndpointSupportedByStatus 根据探测响应的 HTTP 状态码判定上游
|
||||
// 是否暴露 /v1/responses 端点。
|
||||
//
|
||||
|
||||
@@ -0,0 +1,176 @@
|
||||
package service
|
||||
|
||||
import (
|
||||
"context"
|
||||
"io"
|
||||
"net/http"
|
||||
"strings"
|
||||
"testing"
|
||||
|
||||
"github.com/Wei-Shaw/sub2api/internal/config"
|
||||
"github.com/Wei-Shaw/sub2api/internal/pkg/openai_compat"
|
||||
"github.com/stretchr/testify/require"
|
||||
)
|
||||
|
||||
func newResponsesProbeAccount(id int64) Account {
|
||||
return Account{
|
||||
ID: id,
|
||||
Platform: PlatformOpenAI,
|
||||
Type: AccountTypeAPIKey,
|
||||
Name: "compat-upstream",
|
||||
Concurrency: 1,
|
||||
Credentials: map[string]any{
|
||||
"api_key": "sk-test",
|
||||
"base_url": "https://compat-upstream.example/v1",
|
||||
},
|
||||
}
|
||||
}
|
||||
|
||||
// runResponsesProbe 跑一次探测,返回落库的 extra 更新;未落库时返回 nil。
|
||||
func runResponsesProbe(t *testing.T, status int, body string) map[string]any {
|
||||
t.Helper()
|
||||
account := newResponsesProbeAccount(4200)
|
||||
// 带缓冲且不阻塞:探测决定不落标时通道应保持为空。
|
||||
updateCalls := make(chan map[string]any, 1)
|
||||
repo := &snapshotUpdateAccountRepo{
|
||||
stubOpenAIAccountRepo: stubOpenAIAccountRepo{accounts: []Account{account}},
|
||||
updateExtraCalls: updateCalls,
|
||||
}
|
||||
svc := &AccountTestService{
|
||||
accountRepo: repo,
|
||||
httpUpstream: &httpUpstreamRecorder{resp: &http.Response{
|
||||
StatusCode: status,
|
||||
Header: make(http.Header),
|
||||
Body: io.NopCloser(strings.NewReader(body)),
|
||||
}},
|
||||
cfg: &config.Config{Security: config.SecurityConfig{URLAllowlist: config.URLAllowlistConfig{Enabled: false}}},
|
||||
}
|
||||
|
||||
svc.ProbeOpenAIAPIKeyResponsesSupport(context.Background(), account.ID)
|
||||
|
||||
select {
|
||||
case updates := <-updateCalls:
|
||||
return updates
|
||||
default:
|
||||
return nil
|
||||
}
|
||||
}
|
||||
|
||||
// issue #5371:账号一旦被落标为「不支持 Responses」,网关就长期改走
|
||||
// /v1/chat/completions,Codex 的 prompt 缓存前缀被打散;而探测只在账号创建/更新时
|
||||
// 跑一次,标记不会自动恢复。因此判据不成立的响应绝不能落标。
|
||||
func TestProbeOpenAIAPIKeyResponsesSupport_InconclusiveResponseKeepsUnknown(t *testing.T) {
|
||||
cases := []struct {
|
||||
name string
|
||||
body string
|
||||
}{
|
||||
{
|
||||
// 探测请求自带 openaiResponsesProbeMaxOutputTokens 预算,推理型模型可能
|
||||
// 把预算烧在 reasoning 上就被截断——没有 function_call 是预算不足所致。
|
||||
name: "incomplete_max_output_tokens",
|
||||
body: `{"status":"incomplete","incomplete_details":{"reason":"max_output_tokens"},` +
|
||||
`"output":[{"type":"reasoning","summary":[]}]}`,
|
||||
},
|
||||
{
|
||||
// HTTP 200 携带的失败响应是上游瞬时故障,不构成工具能力证据。
|
||||
name: "failed_status_on_http_200",
|
||||
body: `{"status":"failed","error":{"code":"server_error","message":"upstream hiccup"},"output":[]}`,
|
||||
},
|
||||
}
|
||||
|
||||
for _, tc := range cases {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
require.Nil(t, runResponsesProbe(t, http.StatusOK, tc.body),
|
||||
"判据不成立时必须保持 unknown,不得落标")
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
// 对照不变式:能下结论的响应仍要落标,否则「不落标」写宽了就等于把整个探测废掉。
|
||||
func TestProbeOpenAIAPIKeyResponsesSupport_ConclusiveResponsesStillPersist(t *testing.T) {
|
||||
cases := []struct {
|
||||
name string
|
||||
status int
|
||||
body string
|
||||
want bool
|
||||
}{
|
||||
{
|
||||
name: "completed_with_function_call",
|
||||
status: http.StatusOK,
|
||||
body: `{"status":"completed","output":[{"type":"function_call","name":"probe_ping"}]}`,
|
||||
want: true,
|
||||
},
|
||||
{
|
||||
// 火山方舟 coding/v3 × kimi-k2.6:端点在、跑完了、就是不产出 function_call。
|
||||
// 这正是探测要抓的目标,必须继续落标为不支持。
|
||||
name: "completed_reasoning_only",
|
||||
status: http.StatusOK,
|
||||
body: `{"status":"completed","output":[{"type":"reasoning"}]}`,
|
||||
want: false,
|
||||
},
|
||||
{
|
||||
// 非 max_output_tokens 的截断(如内容过滤)不在放行范围内,维持原判定。
|
||||
name: "incomplete_other_reason",
|
||||
status: http.StatusOK,
|
||||
body: `{"status":"incomplete","incomplete_details":{"reason":"content_filter"},"output":[]}`,
|
||||
want: false,
|
||||
},
|
||||
{
|
||||
// 响应体没有 status 字段(第三方兼容上游常见)时维持既有行为。
|
||||
name: "no_status_field",
|
||||
status: http.StatusOK,
|
||||
body: `{"output":[{"type":"function_call","name":"probe_ping"}]}`,
|
||||
want: true,
|
||||
},
|
||||
{
|
||||
name: "endpoint_absent_404",
|
||||
status: http.StatusNotFound,
|
||||
body: `{"error":{"message":"Not Found"}}`,
|
||||
want: false,
|
||||
},
|
||||
{
|
||||
// 非 2xx 的结论只看状态码:body 里的 status=failed 不该让它变成"不下结论"。
|
||||
name: "server_error_stays_conservative_true",
|
||||
status: http.StatusInternalServerError,
|
||||
body: `{"status":"failed"}`,
|
||||
want: true,
|
||||
},
|
||||
}
|
||||
|
||||
for _, tc := range cases {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
updates := runResponsesProbe(t, tc.status, tc.body)
|
||||
require.NotNil(t, updates, "能下结论的响应必须落标")
|
||||
require.Equal(t, tc.want, updates[openai_compat.ExtraKeyResponsesSupported])
|
||||
})
|
||||
}
|
||||
}
|
||||
|
||||
func TestResponsesProbeVerdictIsConclusive(t *testing.T) {
|
||||
cases := []struct {
|
||||
name string
|
||||
status int
|
||||
body string
|
||||
want bool
|
||||
}{
|
||||
{"200_completed", 200, `{"status":"completed","output":[]}`, true},
|
||||
{"200_incomplete_max_output_tokens", 200,
|
||||
`{"status":"incomplete","incomplete_details":{"reason":"max_output_tokens"}}`, false},
|
||||
{"200_incomplete_content_filter", 200,
|
||||
`{"status":"incomplete","incomplete_details":{"reason":"content_filter"}}`, true},
|
||||
{"200_incomplete_without_reason", 200, `{"status":"incomplete"}`, true},
|
||||
{"200_failed", 200, `{"status":"failed"}`, false},
|
||||
{"200_no_status_field", 200, `{"output":[]}`, true},
|
||||
{"200_non_json", 200, `not-json`, true},
|
||||
{"200_empty_body", 200, ``, true},
|
||||
// 非 2xx 只看状态码,不读 body。
|
||||
{"404_ignores_body_status", 404, `{"status":"failed"}`, true},
|
||||
{"500_ignores_body_status", 500, `{"status":"incomplete","incomplete_details":{"reason":"max_output_tokens"}}`, true},
|
||||
}
|
||||
|
||||
for _, tc := range cases {
|
||||
t.Run(tc.name, func(t *testing.T) {
|
||||
require.Equal(t, tc.want, responsesProbeVerdictIsConclusive(tc.status, []byte(tc.body)))
|
||||
})
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user