Files
CLIProxyAPI/internal/runtime/executor/helps/response_model.go
Viggo95 25f40d8cf8 feat(codex): record upstream response model and warn on silent model substitution
Codex upstreams can silently serve a different model than the one requested
(HTTP 200, with response.model naming the substitute). The proxy kept no record
of it: nothing logged, nothing reported, only the pass-through response body.

- Add Record.ResponseModel to sdk/cliproxy/usage, aligned with the existing
  ResponseServiceTier field, and emit it from the redis usage queue as the
  optional response_model payload field alongside response_service_tier. Only
  the record for the requested model carries it: additional-model records
  (image generation tool usage) describe a side model the upstream response
  never refers to, and would otherwise look like a substitution downstream.
- Add internal/runtime/executor/helps/response_model.go with
  extractCodexResponseModelEvent (SSE frames and raw JSON, restricted to the events
  that embed the authoritative response object, rejecting non-string and
  oversized upstream model names) and IsCodexModelSubstituted (both sides
  trimmed, lower-cased and stripped of thinking suffixes, dated aliases such as
  gpt-5.6-terra-2026-05-13 accepted in either direction).
- UsageReporter records the served model on the event path and emits the WARN
  when the attempt publishes its usage record, so no logging work happens
  before the first event is forwarded. Repeats are throttled per
  (auth id, requested model, served model) with a 10 minute window, because on
  an affected credential every request is substituted and an unthrottled
  warning would mirror the whole request volume into the logs. The credential
  is labelled auth_index=<index> only: codex credential file names embed the
  account e-mail, which must not be written to the logs at request rate.

Coverage, by entry point. The served model is observed on the HTTP streaming
path (both the bootstrap-buffered handshake and the streaming goroutine), the
HTTP non-streaming Execute loop, the websocket streaming and non-streaming
paths, and the two /responses-shaped image entry points. The remaining codex
entry points cannot report it and are therefore left alone: executeCompact
(/responses/compact answers with a compaction object that has no event type and
no response.model), the two direct image endpoints (/images/generations and
/images/edits answer in the Images API shape and stream image_generation.*
events), and CountTokens (counts locally with tiktoken, never reaching an
upstream).

TokenAccountingSchemaVersion is not bumped: it versions the token accounting
contract (token breakdown semantics), and this change only adds an optional
non-token field that leaves existing consumers and all token math untouched.

Note: response_model ships with the usage record and is the counting source;
the WARN is a throttled alerting signal and must not be used to count
substitutions.

Tests: table-driven unit tests for both helpers over real model ids, reporter
tests covering the published record, the single throttled warning, the absence
of account identifiers in it, concurrent observation and publishing under
-race, the throttle window and its entry bound, an executor-level guard for the
observeCodexTokenEvent wiring and the per-model records, plus a redisqueue
payload assertion for response_model. gofmt, go vet, go test -race on the
touched packages and go test ./... are clean.
2026-09-18 12:31:52 +08:00

182 lines
5.7 KiB
Go

package helps
import (
"strings"
"sync"
"time"
"github.com/router-for-me/CLIProxyAPI/v7/internal/thinking"
"github.com/tidwall/gjson"
)
const (
// maxCodexResponseModelLength is a defensive bound on an upstream-controlled string
// reaching logs and usage records; known codex model ids stay under ~30 bytes.
maxCodexResponseModelLength = 128
// codexModelSubstitutionWarnWindow bounds how often one credential and model pair
// warns: on an affected credential every request is substituted.
codexModelSubstitutionWarnWindow = 10 * time.Minute
// codexModelSubstitutionWarnMaxEntries caps the throttle state, naturally bounded by
// credentials times models; memory safety wins over perfect throttling.
codexModelSubstitutionWarnMaxEntries = 1024
)
// extractCodexResponseModelEvent returns the model a codex upstream reports serving, read
// from a raw JSON frame or an SSE line, and whether the event terminates the response.
func extractCodexResponseModelEvent(payload []byte) (model string, terminal bool) {
data := jsonPayload(payload)
if len(data) == 0 {
return "", false
}
// The event type is checked before the payload is validated, so the hot path
// (output deltas) stays a single cheap lookup.
carriesModel, terminal := codexResponseModelEventKind(gjson.GetBytes(data, "type").String())
if !carriesModel {
return "", false
}
if !gjson.ValidBytes(data) {
return "", false
}
// The value is upstream-controlled: reject non-string and oversized names
// rather than propagating them into logs and usage records.
modelResult := gjson.GetBytes(data, "response.model")
if modelResult.Type != gjson.String {
return "", terminal
}
model = strings.TrimSpace(modelResult.String())
if len(model) > maxCodexResponseModelLength {
return "", terminal
}
return model, terminal
}
// codexResponseModelEventKind reports whether a codex event embeds the authoritative
// response object, and whether that event terminates the response.
func codexResponseModelEventKind(eventType string) (carriesModel bool, terminal bool) {
switch strings.TrimSpace(eventType) {
case "response.created", "response.in_progress":
return true, false
case "response.completed", "response.incomplete", "response.done":
return true, true
default:
return false, false
}
}
// normalizeCodexModelName lower-cases a model id and drops its thinking suffix,
// which never reaches the upstream request body.
func normalizeCodexModelName(model string) string {
return strings.TrimSpace(thinking.ParseSuffix(strings.ToLower(strings.TrimSpace(model))).ModelName)
}
// IsCodexModelSubstituted reports whether the upstream served a model other than the
// requested one; a dated alias pins a snapshot of the same model and is accepted.
func IsCodexModelSubstituted(requested, served string) bool {
servedModel := normalizeCodexModelName(served)
if servedModel == "" {
return false
}
requestedModel := normalizeCodexModelName(requested)
if requestedModel == "" {
return false
}
if requestedModel == servedModel {
return false
}
return !isCodexDatedModelAlias(requestedModel, servedModel) &&
!isCodexDatedModelAlias(servedModel, requestedModel)
}
// isCodexDatedModelAlias reports whether dated is base plus a release date suffix,
// which upstreams use to pin the exact snapshot of the same model.
func isCodexDatedModelAlias(base, dated string) bool {
prefix := base + "-"
if !strings.HasPrefix(dated, prefix) {
return false
}
return isCodexModelDateSuffix(dated[len(prefix):])
}
// isCodexModelDateSuffix reports whether suffix is a YYYY-MM-DD or YYYYMMDD date.
func isCodexModelDateSuffix(suffix string) bool {
switch len(suffix) {
case len("YYYY-MM-DD"):
if suffix[4] != '-' || suffix[7] != '-' {
return false
}
return isCodexModelDigits(suffix[:4]) && isCodexModelDigits(suffix[5:7]) && isCodexModelDigits(suffix[8:])
case len("YYYYMMDD"):
return isCodexModelDigits(suffix)
default:
return false
}
}
// isCodexModelDigits reports whether value is a non-empty run of ASCII digits.
func isCodexModelDigits(value string) bool {
if value == "" {
return false
}
for i := 0; i < len(value); i++ {
if value[i] < '0' || value[i] > '9' {
return false
}
}
return true
}
type codexModelSubstitutionKey struct {
authID string
requested string
served string
}
// codexModelSubstitutionThrottle records the last warning per key; nowFunc is
// injectable so tests can advance the window without sleeping.
type codexModelSubstitutionThrottle struct {
mu sync.Mutex
nowFunc func() time.Time
lastWarn map[codexModelSubstitutionKey]time.Time
}
func newCodexModelSubstitutionThrottle(nowFunc func() time.Time) *codexModelSubstitutionThrottle {
if nowFunc == nil {
nowFunc = time.Now
}
return &codexModelSubstitutionThrottle{
nowFunc: nowFunc,
lastWarn: make(map[codexModelSubstitutionKey]time.Time),
}
}
// allow reports whether the key may emit a warning now and records the decision.
// Suppressed repeats stay silent instead of moving to a lower level.
func (t *codexModelSubstitutionThrottle) allow(key codexModelSubstitutionKey) bool {
if t == nil {
return true
}
t.mu.Lock()
defer t.mu.Unlock()
now := t.nowFunc()
if last, ok := t.lastWarn[key]; ok && now.Sub(last) < codexModelSubstitutionWarnWindow {
return false
}
if len(t.lastWarn) >= codexModelSubstitutionWarnMaxEntries {
for storedKey, storedAt := range t.lastWarn {
if now.Sub(storedAt) >= codexModelSubstitutionWarnWindow {
delete(t.lastWarn, storedKey)
}
}
if len(t.lastWarn) >= codexModelSubstitutionWarnMaxEntries {
clear(t.lastWarn)
}
}
t.lastWarn[key] = now
return true
}
// codexModelSubstitutionWarns throttles substitution warnings process-wide.
var codexModelSubstitutionWarns = newCodexModelSubstitutionThrottle(time.Now)