mirror of
https://github.com/router-for-me/CLIProxyAPI.git
synced 2026-09-03 06:35:00 +08:00
* feat(codex): add opt-in stream bootstrap buffering
The upstream smuggles capacity rejections into an HTTP 200 stream. The
handshake events arrive normally and only a later event carries
{"error":{"type":"service_unavailable_error","code":
"server_is_overloaded"}}. By then the executor has already handed the
first chunk downstream, the response is committed, and the conductor can
no longer retry on another credential, so the request fails even though
other credentials were available.
When codex.stream-bootstrap-buffering is enabled the executor holds back
the handshake events until it can tell whether the stream carries real
output or a rejection. An overload rejection then fails the attempt
before any chunk is delivered, letting the conductor retry on another
credential; every other terminal failure is flushed in order and
delivered in-stream exactly as before.
Detection uses an event-type allow-list rather than a fixed count. On the
websocket transport codex.rate_limits and codex.response.metadata arrive
before response.created, making the first generated event the fifth
frame, so a small counter would release the stream before the rejection
is visible. Buffering is bounded and hitting the bound degrades to the
original unbuffered behaviour.
Two details are load-bearing. The error must be returned synchronously:
delivering it as the first stream chunk makes ExecuteStream downgrade it
into a committed 200 and the status is lost. And the websocket path must
not signal an upstream disconnect for a rejection it intends to retry,
because the downstream handler closes the client connection on that
signal and the retry would have nowhere to deliver.
The 503 status is produced only on this path rather than in the shared
codexTerminalFailureStatus mapping, so disabling the feature restores the
previous behaviour exactly, including cooldown classification and
retry-after parsing.
Defaults to false: response headers are withheld until generation
starts, which can trip client or reverse-proxy read timeouts.
* test(codex): pin bootstrap overload failover through the conductor
Executor-level tests cannot show what the client finally receives. These
exercise ExecuteStream end to end to pin three properties that are easy
to regress:
- consecutive overloaded credentials are skipped until one serves the
request, and retries are capped by max-retry-credentials rather than
multiplying with request-retry
- exhausting the pool surfaces the upstream status instead of a
committed 200 stream
- with buffering disabled the rejection stays an in-stream error on a
committed stream, which is the behaviour the feature must preserve
The third case also documents why the executor returns its error
synchronously: an error arriving as the first stream chunk is wrapped and
downgraded into a committed 200, silently losing the status.
839 lines
47 KiB
YAML
839 lines
47 KiB
YAML
# Server host/interface to bind to. Default is empty ("") to bind all interfaces (IPv4 + IPv6).
|
|
# Use "127.0.0.1" or "localhost" to restrict access to local machine only.
|
|
host: ""
|
|
|
|
# Server port
|
|
port: 8317
|
|
|
|
# TLS settings for HTTPS. When enabled, the server listens with the provided certificate and key.
|
|
tls:
|
|
enable: false
|
|
cert: ""
|
|
key: ""
|
|
|
|
# Management API settings
|
|
remote-management:
|
|
# Whether to allow remote (non-localhost) management access.
|
|
# When false, only localhost can access management endpoints (a key is still required).
|
|
allow-remote: false
|
|
|
|
# Management key. If a plaintext value is provided here, it will be hashed on startup.
|
|
# All management requests (even from localhost) require this key.
|
|
# Leave empty to disable the Management API entirely (404 for all /v0/management routes).
|
|
secret-key: ""
|
|
|
|
# Disable the bundled management control panel asset download and HTTP route when true.
|
|
disable-control-panel: false
|
|
|
|
# Disable automatic periodic background updates of the management panel from GitHub (default: false).
|
|
# When enabled, the panel is only downloaded on first access if missing, and never auto-updated afterward.
|
|
# disable-auto-update-panel: false
|
|
|
|
# GitHub repository for the management control panel. Accepts a repository URL or releases API URL.
|
|
panel-github-repository: "https://github.com/router-for-me/Cli-Proxy-API-Management-Center"
|
|
|
|
# Authentication directory (supports ~ for home directory)
|
|
auth-dir: "~/.cli-proxy-api"
|
|
|
|
# API keys for authentication
|
|
api-keys:
|
|
- "your-api-key-1"
|
|
- "your-api-key-2"
|
|
- "your-api-key-3"
|
|
|
|
# Enable debug logging
|
|
debug: false
|
|
|
|
# Enable pprof HTTP debug server (host:port). Keep it bound to localhost for safety.
|
|
pprof:
|
|
enable: false
|
|
addr: "127.0.0.1:8316"
|
|
|
|
# Credential concurrency is configured by Home in Home mode. The synthesized Home config is
|
|
# authoritative and local values, including the values below, are ignored. Do not use local
|
|
# configuration to override a Home concurrency policy.
|
|
# credential-concurrency:
|
|
# lifecycle-config-revision: 1
|
|
# observation-barrier-revision: 0
|
|
# cpa-heartbeat-timeout: "3s"
|
|
# cpa-cancel-bound: "5s"
|
|
# reclaim-grace: "5s"
|
|
# cleanup-interval: "5s"
|
|
# release-flush-interval: 250ms
|
|
# release-max-backoff: 2s
|
|
# busy-retry-min: 250ms
|
|
# busy-retry-max: 1s
|
|
# max-limit: 1000000
|
|
|
|
# Credential in-flight observation snapshot contract.
|
|
# credential-in-flight:
|
|
# snapshot-interval: 2s
|
|
# stale-after: 10s
|
|
# max-part-bytes: 262144
|
|
# max-part-count: 64
|
|
# max-revision-bytes: 16777216
|
|
# max-aggregate-groups: 100000
|
|
# max-details: 10000
|
|
# max-string-bytes: 256
|
|
# staging-retention: 1m
|
|
|
|
# Standard dynamic library plugins are trusted in-process code. They are disabled by default.
|
|
# Build Go examples with go build -buildmode=c-shared for the target GOOS/GOARCH.
|
|
# Other languages can implement the same C ABI and JSON method protocol.
|
|
# Plugin executors require a matching auth record with the same provider key.
|
|
# If the same provider is configured as OpenAI-compatible, the native executor wins.
|
|
# Plugin command-line flags and Management API routes are optional capabilities.
|
|
# Existing native flags/routes and higher-priority plugin flags/routes cannot be replaced.
|
|
# Plugin list Management API reads Logo and ConfigFields from plugin metadata for management UI display.
|
|
# Per-plugin enabled only controls plugins.configs.<pluginID>.enabled and does not implicitly change global plugins.enabled.
|
|
plugins:
|
|
enabled: false
|
|
dir: "plugins"
|
|
# Additional plugin store registries. The built-in official registry is always included.
|
|
# store-sources:
|
|
# - "https://example.com/cliproxy-plugins/registry.json"
|
|
# Optional plugin store auth rules. Values are read from environment variables;
|
|
# tokens are not written into plugin manifests or node status.
|
|
# store-auth:
|
|
# - match: "https://example.com/cliproxy-plugins/"
|
|
# apply-to: ["registry", "artifact"]
|
|
# type: bearer
|
|
# token-env: "CLIPROXY_PLUGIN_STORE_TOKEN"
|
|
configs:
|
|
example:
|
|
enabled: true
|
|
priority: 1
|
|
config1: true
|
|
config2: "string"
|
|
config3: 3
|
|
mode: "safe" # enum example: safe, fast
|
|
|
|
# When true, disable high-overhead request logging and HTTP middleware features to reduce per-request memory usage under high concurrency.
|
|
commercial-mode: false
|
|
|
|
# When true, write application logs to rotating files instead of stdout
|
|
logging-to-file: false
|
|
|
|
# Maximum total size (MB) of log files under the logs directory. When exceeded, the oldest log
|
|
# files are deleted until within the limit. Set to 0 to disable.
|
|
logs-max-total-size-mb: 0
|
|
|
|
# Maximum number of error log files retained when request logging is disabled.
|
|
# When exceeded, the oldest error log files are deleted. Default is 10. Set to 0 to disable cleanup.
|
|
error-logs-max-files: 10
|
|
|
|
# When false, disable in-memory usage statistics aggregation
|
|
usage-statistics-enabled: false
|
|
|
|
# How long (in seconds) usage queue items are retained in memory for the Management API.
|
|
# The local Redis RESP usage output is disabled.
|
|
# Default: 60. Max: 3600.
|
|
redis-usage-queue-retention-seconds: 60
|
|
|
|
# Proxy URL. Supports socks5/http/https protocols. Example: socks5://user:pass@192.168.1.1:1080/
|
|
# Per-entry proxy-url also supports "direct" or "none" to bypass both the global proxy-url and environment proxies explicitly.
|
|
proxy-url: ""
|
|
|
|
# When true, unprefixed model requests only use credentials without a prefix (except when prefix == model name).
|
|
force-model-prefix: false
|
|
|
|
# When true, forward filtered upstream response headers to downstream clients.
|
|
# Default is false (disabled).
|
|
passthrough-headers: false
|
|
|
|
# Number of times to retry a request. Retries will occur if the HTTP response code is 403, 408, 500, 502, 503, or 504.
|
|
# Individual API-key entries can override this via request-retry. OAuth/token JSON files can override via request_retry.
|
|
# 0 disables retries for that credential; omit the field or set a negative value to keep this global setting.
|
|
request-retry: 3
|
|
|
|
# Maximum number of different credentials to try for one failed request.
|
|
# Set to 0 to keep legacy behavior (try all available credentials).
|
|
max-retry-credentials: 0
|
|
|
|
# Maximum wait time in seconds for a cooled-down credential before triggering a retry.
|
|
max-retry-interval: 30
|
|
|
|
# When true, disable auth/model cooldown scheduling globally (prevents blackout windows after failure states).
|
|
# A credential/provider disable-cooling value, when present, overrides this global value.
|
|
disable-cooling: false
|
|
|
|
# When true, persist per-auth cooldown status as .cds files next to auth files.
|
|
# Default is false; when false, cooldown status is kept in memory only.
|
|
save-cooldown-status: false
|
|
|
|
# Cooldown duration in seconds for transient upstream errors (408/500/502/503/504).
|
|
# Set to 0 to keep the legacy 60-second cooldown; set to -1 to disable transient error cooldowns.
|
|
transient-error-cooldown-seconds: 0
|
|
|
|
# When true, globally disable Claude request cloaking (the Claude Code CLI disguise and
|
|
# system prompt replacement), so the original system prompt is passed through to Claude as-is.
|
|
# Individual credentials can still override this: a claude-api-key entry via its "cloak.mode",
|
|
# or a Claude OAuth/token file via a "cloak_mode" value. Default false keeps the per-client
|
|
# "auto" behavior (cloak only non-Claude-Code clients).
|
|
disable-claude-cloak-mode: false
|
|
|
|
# Claude Code compatibility settings.
|
|
claude-code:
|
|
# When true, return original model IDs in Anthropic model list responses instead of cloaked IDs.
|
|
disable-cloaking-model-list: false
|
|
|
|
# disable-image-generation supports: false (default), true, "chat", or "passthrough".
|
|
# - true: disable image_generation everywhere (also returns 404 for /v1/images/generations and /v1/images/edits).
|
|
# - "chat": disable image_generation injection on non-images endpoints, but keep /v1/images/generations and /v1/images/edits enabled.
|
|
# - "passthrough": never inject or strip image_generation on non-images endpoints (forward the client payload unchanged); behaves like "chat" on /v1/images/* endpoints.
|
|
disable-image-generation: false
|
|
|
|
# Base model used by the legacy hosted image_generation tool path when a Codex image request is not proxied directly through the Image API.
|
|
# Must start with "gpt-" (case-insensitive). If unset or invalid, defaults to "gpt-5.4-mini".
|
|
# gpt-image-2-base-model: "gpt-5.4-mini"
|
|
|
|
# How long video IDs returned by /openai/v1/videos and xAI video creation stay bound
|
|
# to the credential that created them. Default: 3h.
|
|
video-result-auth-cache-ttl: "3h"
|
|
|
|
# Core auth auto-refresh worker pool size (OAuth/file-based auth token refresh).
|
|
# When > 0, overrides the default worker count (16).
|
|
# auth-auto-refresh-workers: 16
|
|
|
|
# Quota exceeded behavior
|
|
quota-exceeded:
|
|
switch-project: true # Whether to automatically switch to another project when a quota is exceeded
|
|
switch-preview-model: true # Whether to automatically switch to a preview model when a quota is exceeded
|
|
antigravity-credits: true # Whether to use credits as last-resort fallback when all free-tier auths are exhausted for Claude models
|
|
|
|
# Routing strategy for selecting credentials when multiple match.
|
|
routing:
|
|
strategy: "round-robin" # round-robin (default), weighted-round-robin, fill-first
|
|
# weighted-round-robin uses each credential's integer weight (default 1, maximum 1,000,000).
|
|
# Non-positive weights exclude the credential while this strategy is active.
|
|
# For OAuth/file credentials, add a top-level numeric "weight" field to the auth JSON.
|
|
# Enable universal session-sticky routing for all clients.
|
|
# Explicit Claude Code, Codex, OpenCode, and pi session headers are preferred,
|
|
# followed by prompt_cache_key, Responses conversation IDs, legacy body IDs,
|
|
# execution or derived session identity, and the existing first-message hash fallback.
|
|
# Automatic failover is always enabled when bound auth becomes unavailable.
|
|
# An established binding outranks credential priority: once a session is bound, that
|
|
# credential is kept even if a higher-priority credential recovers. Credential priority
|
|
# still decides cold bindings, requests without a session, and post-failover rebinding.
|
|
session-affinity: false # default: false
|
|
# How long session-to-auth bindings are retained. Default: 1h
|
|
session-affinity-ttl: "1h"
|
|
|
|
# Codex provider behavior.
|
|
codex:
|
|
# When true, and routing.strategy is fill-first or routing.session-affinity is true,
|
|
# remap Codex prompt_cache_key and installation identity per selected auth.
|
|
# Some superstitious users believe request tracking identifiers can be used
|
|
# as evidence for TOS enforcement bans; this option only satisfies those odd concerns.
|
|
identity-confuse: false
|
|
# Disable forcing the official Codex User-Agent and Originator headers on HTTP/SSE and WebSocket requests.
|
|
disable-codex-cloaking: false
|
|
# Hold back the initial handshake events (response.created, response.in_progress and the
|
|
# websocket metadata frames) until the upstream emits its first generated event.
|
|
# Why: the upstream smuggles `server_is_overloaded` rejections *inside* an HTTP 200 stream,
|
|
# right after those handshake events, instead of returning 503 on the wire. Buffering them
|
|
# keeps the downstream response headers uncommitted long enough to transparently retry on
|
|
# another credential. Only overload/rate-limit rejections trigger failover; every other
|
|
# terminal failure is still delivered in-stream exactly as before.
|
|
# Trade-off: response headers are delayed until generation starts, which can trip client or
|
|
# reverse-proxy read timeouts (e.g. nginx proxy_read_timeout) on long reasoning requests.
|
|
# Default: false
|
|
stream-bootstrap-buffering: false
|
|
# When true, optimize Codex Desktop, codex-tui, and codex_cli_rs requests for multi-agent v2.
|
|
# This refreshes Codex spawn_agent model details, removes message parameter encryption,
|
|
# normalizes encrypted agent_message content for Codex, and converts agent_message input
|
|
# into standard user messages for non-Codex upstream protocols.
|
|
optimize-multi-agent-v2: false
|
|
# Terminate and relay Codex Live WebRTC audio and DataChannel traffic in this process.
|
|
# This requires inbound UDP reachability. Keep disabled to preserve direct media behavior.
|
|
live-media-relay:
|
|
enabled: false
|
|
# Maximum concurrent media sessions. Zero uses the default of 32.
|
|
max-sessions: 32
|
|
# Reject downstream SDP candidates that target private, loopback, link-local, or unspecified IPs.
|
|
# Keep false for local or trusted-network Codex Desktop connections.
|
|
disable-private-remote-ips: false
|
|
# Public IPv4 or IPv6 address advertised when CPA is behind 1:1 NAT.
|
|
public-ip: ""
|
|
# Optional UDP allocation range. Both values must be set together and provide at least two ports per session.
|
|
udp-port-min: 0
|
|
udp-port-max: 0
|
|
# Optional STUN/TURN servers. TURN credentials are never returned by the JSON config API.
|
|
# Without a concrete global/per-auth proxy-url, WebRTC uses normal direct ICE/STUN/TURN connectivity.
|
|
# With http, https, socks5, or socks5h proxy-url, the OpenAI-facing leg is forced through
|
|
# authenticated ICE-TCP over that proxy and never falls back to UDP or a direct connection.
|
|
# The Codex Desktop-facing leg remains direct, and configured ICE servers still apply to it.
|
|
# ice-servers:
|
|
# - urls:
|
|
# - "stun:stun.example.com:3478"
|
|
# - urls:
|
|
# - "turn:turn.example.com:3478?transport=udp"
|
|
# username: "user"
|
|
# credential: "secret"
|
|
|
|
# Antigravity provider behavior.
|
|
# antigravity:
|
|
# sensitive-words: # optional: words to obfuscate with zero-width characters in system instructions
|
|
# - "API"
|
|
# - "proxy"
|
|
|
|
# xAI provider behavior.
|
|
xai:
|
|
# When true, inject the native x_search tool when the request does not declare it.
|
|
# The injected tool is also added to tool_choice.allowed_tools when applicable.
|
|
inject-x-search: false
|
|
|
|
# When true, enable authentication for the WebSocket API (/v1/ws).
|
|
ws-auth: true
|
|
|
|
# When > 0, emit blank lines every N seconds for non-streaming responses to prevent idle timeouts.
|
|
nonstream-keepalive-interval: 0
|
|
# Streaming behavior (SSE keep-alives + safe bootstrap retries).
|
|
# streaming:
|
|
# keepalive-seconds: 15 # Default: 0 (disabled). <= 0 disables keep-alives.
|
|
# bootstrap-retries: 1 # Default: 0 (disabled). Retries before first byte is sent.
|
|
|
|
# Signature cache validation for thinking blocks (Antigravity/Claude).
|
|
# When true (default), cached signatures are preferred and validated.
|
|
# When false, client signatures are used directly after normalization (bypass mode for testing).
|
|
# antigravity-signature-cache-enabled: true
|
|
|
|
# Bypass mode signature validation strictness (only applies when signature cache is disabled).
|
|
# When true, validates full Claude protobuf tree (Field 2 -> Field 1 structure).
|
|
# When false (default), only checks R/E prefix + base64 + first byte 0x12.
|
|
# antigravity-signature-bypass-strict: false
|
|
|
|
# Gemini API keys
|
|
# gemini-api-key:
|
|
# - api-key: "AIzaSy...01"
|
|
# weight: 5 # optional: weighted-round-robin share; omitted defaults to 1; maximum 1,000,000
|
|
# prefix: "test" # optional: require calls like "test/gemini-3-pro-preview" to target this credential
|
|
# disable-cooling: false # optional override: true disables cooling, false enables it; omit to inherit global
|
|
# request-retry: 3 # optional: per-auth override of the global request-retry; 0 disables retries; omit or set < 0 to use the global value
|
|
# request-scoped-errors: # optional: custom rules to classify upstream errors by status and body patterns
|
|
# - status: 400 # HTTP status code to match
|
|
# match: # optional: string contains matching
|
|
# - "maximum_context_length"
|
|
# - "context_length_exceeded"
|
|
# match-regexr: # optional: regular expression matching
|
|
# - "maximum_context_length$"
|
|
# - "^context_length_exceeded"
|
|
# action: "stop" # "stop" (return error, no cooling), "stop-and-cooldown" (return error and cool down),
|
|
# # "continue" (try next credential, no cooling), "continue-and-cooldown" (try next credential and cool down)
|
|
# base-url: "https://generativelanguage.googleapis.com"
|
|
# headers:
|
|
# X-Custom-Header: "custom-value"
|
|
# # Values starting with "$" dynamically copy the header value from downstream client requests.
|
|
# # If the client did not send the specified header, the header is omitted.
|
|
# # X-Claude-Code-Session-Id: "$ABC" # copies client's "ABC" header
|
|
# proxy-url: "socks5://proxy.example.com:1080"
|
|
# # proxy-url: "direct" # optional: explicit direct connect for this credential
|
|
# models:
|
|
# - name: "gemini-2.5-flash" # upstream model name
|
|
# alias: "gemini-flash" # client alias mapped to the upstream model
|
|
# display-name: "Gemini Flash" # optional catalog display name
|
|
# max-context-length: 1048576 # optional: override Codex client context window metadata
|
|
# is-compat: false # optional: preserve thinking blocks with empty signatures for compatible upstreams
|
|
# thinking: # optional: exact thinking capability for this configured model
|
|
# levels: ["high", "medium", "low", "none", "auto"]
|
|
# excluded-models:
|
|
# - "gemini-2.5-pro" # exclude specific models from this provider (exact match)
|
|
# - "gemini-2.5-*" # wildcard matching prefix (e.g. gemini-2.5-flash, gemini-2.5-pro)
|
|
# - "*-preview" # wildcard matching suffix (e.g. gemini-3-pro-preview)
|
|
# - "*flash*" # wildcard matching substring (e.g. gemini-2.5-flash-lite)
|
|
# - api-key: "AIzaSy...02"
|
|
|
|
# Native Interactions API keys
|
|
# These keys are used only for direct /v1beta/interactions execution. Regular gemini-api-key entries still
|
|
# send Gemini generateContent/streamGenerateContent requests when the client enters through the interactions API.
|
|
# interactions-api-key:
|
|
# - api-key: "AIzaSy...03"
|
|
# weight: 5 # optional: weighted-round-robin share; omitted defaults to 1; maximum 1,000,000
|
|
# prefix: "native" # optional: require calls like "native/gemini-3-pro-preview" to target this credential
|
|
# disable-cooling: false # optional override: true disables cooling, false enables it; omit to inherit global
|
|
# request-retry: 3 # optional: per-auth override of the global request-retry; 0 disables retries; omit or set < 0 to use the global value
|
|
# request-scoped-errors: # optional: custom rules to classify upstream errors by status and body patterns
|
|
# - status: 400
|
|
# match:
|
|
# - "invalid_argument"
|
|
# action: "continue"
|
|
# base-url: "https://generativelanguage.googleapis.com"
|
|
# headers:
|
|
# X-Custom-Header: "custom-value"
|
|
# # Values starting with "$" dynamically copy the header value from downstream client requests.
|
|
# # If the client did not send the specified header, the header is omitted.
|
|
# # X-Claude-Code-Session-Id: "$ABC" # copies client's "ABC" header
|
|
# proxy-url: "socks5://proxy.example.com:1080"
|
|
# # proxy-url: "direct" # optional: explicit direct connect for this credential
|
|
# models:
|
|
# - name: "gemini-2.5-flash" # upstream model name
|
|
# alias: "native-gemini-flash" # client alias mapped to the upstream model
|
|
# max-context-length: 1048576 # optional: override Codex client context window metadata
|
|
# is-compat: false # optional: preserve thinking blocks with empty signatures for compatible upstreams
|
|
# thinking: # optional: exact thinking capability for this configured model
|
|
# levels: ["high", "medium", "low", "none", "auto"]
|
|
# excluded-models:
|
|
# - "gemini-2.5-pro"
|
|
|
|
# Codex API keys
|
|
# codex-api-key:
|
|
# - api-key: "sk-atSM..."
|
|
# weight: 5 # optional: weighted-round-robin share; omitted defaults to 1; maximum 1,000,000
|
|
# prefix: "test" # optional: require calls like "test/gpt-5-codex" to target this credential
|
|
# disable-cooling: false # optional override: true disables cooling, false enables it; omit to inherit global
|
|
# request-retry: 3 # optional: per-auth override of the global request-retry; 0 disables retries; omit or set < 0 to use the global value
|
|
# request-scoped-errors: # optional: custom rules to classify upstream errors by status and body patterns
|
|
# - status: 400
|
|
# match:
|
|
# - "context_window_exceeded"
|
|
# action: "stop-and-cooldown"
|
|
# base-url: "https://www.example.com" # use the custom codex API endpoint
|
|
# alpha-search: false # optional: allow this key to serve /v1/alpha/search via base-url + /alpha/search
|
|
# headers:
|
|
# X-Custom-Header: "custom-value"
|
|
# # Values starting with "$" dynamically copy the header value from downstream client requests.
|
|
# # If the client did not send the specified header, the header is omitted.
|
|
# # X-Claude-Code-Session-Id: "$ABC" # copies client's "ABC" header
|
|
# proxy-url: "socks5://proxy.example.com:1080" # optional: per-key proxy override
|
|
# # proxy-url: "direct" # optional: explicit direct connect for this credential
|
|
# models:
|
|
# - name: "gpt-5-codex" # upstream model name
|
|
# alias: "codex-latest" # client alias mapped to the upstream model
|
|
# display-name: "Codex Latest" # optional catalog display name
|
|
# max-context-length: 1048576 # optional: override Codex client context window metadata
|
|
# force-mapping: true # optional: rewrite response model fields back to the alias
|
|
# # When true and codex.optimize-multi-agent-v2 is also true, convert Codex
|
|
# # MultiAgentV2 agent_message items into portable Responses message/user input
|
|
# # for third-party Responses-compatible endpoints that reject agent_message.
|
|
# # Default false keeps agent_message unchanged for native OpenAI/Codex endpoints.
|
|
# # It also preserves thinking blocks with empty signatures for compatible upstreams.
|
|
# is-compat: false
|
|
# thinking: # optional: exact thinking capability for this configured model
|
|
# levels: ["xhigh", "high", "medium", "low"]
|
|
# excluded-models:
|
|
# - "gpt-5.1" # exclude specific models (exact match)
|
|
# - "gpt-5-*" # wildcard matching prefix (e.g. gpt-5-medium, gpt-5-codex)
|
|
# - "*-mini" # wildcard matching suffix (e.g. gpt-5-codex-mini)
|
|
# - "*codex*" # wildcard matching substring (e.g. gpt-5-codex-low)
|
|
|
|
# xAI API keys
|
|
# Uses the native xAI executor, including its Responses namespace-tool handling.
|
|
# xai-api-key:
|
|
# - api-key: "xai-..."
|
|
# weight: 5 # optional: weighted-round-robin share; omitted defaults to 1; maximum 1,000,000
|
|
# prefix: "xai" # optional: require calls like "xai/grok-4.5" to target this credential
|
|
# disable-cooling: false # optional override: true disables cooling, false enables it; omit to inherit global
|
|
# request-retry: 3 # optional: per-auth override of the global request-retry; 0 disables retries; omit or set < 0 to use the global value
|
|
# request-scoped-errors: # optional: custom rules to classify upstream errors by status and body patterns
|
|
# - status: 400
|
|
# match:
|
|
# - "rate_limit_exceeded"
|
|
# action: "continue-and-cooldown"
|
|
# base-url: "https://api.x.ai/v1" # xAI-compatible Responses API endpoint
|
|
# websockets: true # optional: use the xAI upstream websocket transport for downstream websocket requests
|
|
# headers:
|
|
# X-Custom-Header: "custom-value"
|
|
# # Values starting with "$" dynamically copy the header value from downstream client requests.
|
|
# # If the client did not send the specified header, the header is omitted.
|
|
# # X-Claude-Code-Session-Id: "$ABC" # copies client's "ABC" header
|
|
# proxy-url: "socks5://proxy.example.com:1080" # optional: per-key proxy override
|
|
# # proxy-url: "direct" # optional: explicit direct connect for this credential
|
|
# models:
|
|
# - name: "grok-4.5" # upstream model name
|
|
# alias: "grok-latest" # client alias mapped to the upstream model
|
|
# display-name: "Grok Latest" # optional catalog display name
|
|
# max-context-length: 1048576 # optional: override Codex client context window metadata
|
|
# force-mapping: true # optional: rewrite response model fields back to the alias
|
|
# is-compat: false # optional: preserve thinking blocks with empty signatures for compatible upstreams
|
|
# thinking: # optional: exact thinking capability for this configured model
|
|
# levels: ["xhigh", "high", "medium", "low"]
|
|
# excluded-models:
|
|
# - "grok-4.1" # exclude specific models (exact match)
|
|
# - "grok-3-*" # wildcard matching prefix
|
|
|
|
# Claude API keys
|
|
# claude-api-key:
|
|
# - api-key: "sk-atSM..." # use the official claude API key, no need to set the base url
|
|
# - api-key: "sk-atSM..."
|
|
# weight: 5 # optional: weighted-round-robin share; omitted defaults to 1; maximum 1,000,000
|
|
# prefix: "test" # optional: require calls like "test/claude-sonnet-latest" to target this credential
|
|
# disable-cooling: false # optional override: true disables cooling, false enables it; omit to inherit global
|
|
# request-retry: 3 # optional: per-auth override of the global request-retry; 0 disables retries; omit or set < 0 to use the global value
|
|
# request-scoped-errors: # optional: custom rules to classify upstream errors by status and body patterns
|
|
# - status: 400
|
|
# match:
|
|
# - "prompt is too long"
|
|
# action: "stop"
|
|
# base-url: "https://www.example.com" # use the custom claude API endpoint
|
|
# headers:
|
|
# X-Custom-Header: "custom-value"
|
|
# # Values starting with "$" dynamically copy the header value from downstream client requests.
|
|
# # If the client did not send the specified header, the header is omitted.
|
|
# # X-Claude-Code-Session-Id: "$ABC" # copies client's "ABC" header
|
|
# proxy-url: "socks5://proxy.example.com:1080" # optional: per-key proxy override
|
|
# # proxy-url: "direct" # optional: explicit direct connect for this credential
|
|
# models:
|
|
# - name: "claude-3-5-sonnet-20241022" # upstream model name
|
|
# alias: "claude-sonnet-latest" # client alias mapped to the upstream model
|
|
# display-name: "Claude Sonnet" # optional catalog display name
|
|
# max-context-length: 1048576 # optional: override Codex client context window metadata
|
|
# force-mapping: true # optional: rewrite response model fields back to the alias
|
|
# is-compat: false # optional: preserve thinking blocks with empty signatures for compatible upstreams
|
|
# thinking: # optional: exact thinking capability for this configured model
|
|
# levels: ["max", "xhigh", "high", "medium", "low", "minimal", "none", "auto"]
|
|
# excluded-models:
|
|
# - "claude-opus-4-5-20251101" # exclude specific models (exact match)
|
|
# - "claude-3-*" # wildcard matching prefix (e.g. claude-3-7-sonnet-20250219)
|
|
# - "*-thinking" # wildcard matching suffix (e.g. claude-opus-4-5-thinking)
|
|
# - "*haiku*" # wildcard matching substring (e.g. claude-3-5-haiku-20241022)
|
|
# rebuild-mid-system-message: false # optional: default is false; when true, move messages with role "system" into the top-level Claude system field
|
|
# cloak: # optional: explicitly enable request cloaking for non-Claude-Code clients
|
|
# mode: "auto" # "auto" (default inside this block): cloak only when client is not Claude Code
|
|
# # "always": cloak every unconfirmed client; confirmed native Claude Code still passes through
|
|
# # "never": never apply cloaking
|
|
# # This "cloak" block applies to this claude-api-key entry only. For Claude OAuth
|
|
# # credentials, set the same options in the auth/token JSON file via "cloak_mode" /
|
|
# # "cloak_strict_mode" / "cloak_sensitive_words" / "cloak_cache_user_id". The top-level
|
|
# # "disable-claude-cloak-mode: true" disables cloaking for all Claude credentials at once.
|
|
# strict-mode: false # false (default): legacy-model whitelist uses a user system-reminder;
|
|
# # all other and future models use messages[].role=system
|
|
# # true: strip caller prompts and keep only Claude Code billing and identity blocks
|
|
# sensitive-words: # optional: words to obfuscate with zero-width characters
|
|
# - "API"
|
|
# - "proxy"
|
|
# cache-user-id: true # optional: default is false; set true to reuse cached user_id per API key instead of generating a random one each request
|
|
# # Every custom tool on a cloaked OAuth request automatically uses a caller-stable opaque mcp__<server>__<tool> alias.
|
|
#
|
|
# # fingerprint-profile (optional, top-level on this claude-api-key entry; not a cloak sub-field):
|
|
# # OAuth and API-key fingerprints are different contracts.
|
|
# # - Real Claude OAuth stays on the strict Claude Code CLI wire fingerprint.
|
|
# # - API keys (official Anthropic, custom gateways, Kimi) stay loose and
|
|
# # caller-owned unless this field is set.
|
|
# #
|
|
# # Default (omit / empty): keep the caller request fingerprint and headers.
|
|
# # Official api.anthropic.com API keys do not add extra CLI betas/identity unless
|
|
# # this field is set. Custom gateways and delegated providers are the same.
|
|
# #
|
|
# # Controls request fingerprint only on /v1/messages (and related Claude executor paths).
|
|
# # Auth scheme stays API key (x-api-key on api.anthropic.com; Bearer on custom base-url).
|
|
# # Does NOT enable OAuth refresh, profile fetch, or OAuth-cancellation semantics.
|
|
# #
|
|
# # Values:
|
|
# # omit / empty = caller-owned API-key fingerprint (respects caller)
|
|
# # "claude-code-cli" = same Messages fingerprint as Claude Code OAuth CLI,
|
|
# # including official Anthropic API keys: OAuth Anthropic-Beta
|
|
# # set, CCH signing on api.anthropic.com, stable CLI
|
|
# # metadata.user_id / session_id / device identity.
|
|
# # API keys seed identity from the key;
|
|
# # delegated OAuth providers use stable auth ID instead of
|
|
# # rotating access tokens. "oauth-cli" is a legacy alias.
|
|
# #
|
|
# # count_tokens keeps the native model/messages/tools shape for every origin, including
|
|
# # Kimi opt-in. It does not send billing/CCH, currentDate, metadata, or diagnostics.
|
|
# #
|
|
# # CCH: the billing block may carry a per-request cch hash. CPA emits it exactly where
|
|
# # Claude Code does, which is api.anthropic.com (first-party) and Vertex only. An opt-in
|
|
# # on any other gateway (including Kimi) still sends the billing block, but without cch,
|
|
# # so a per-request hash cannot bust that gateway's prompt cache. api.anthropic.com
|
|
# # strips the block itself (0 tokens, no cache impact). Kimi drops the whole block by
|
|
# # default and keeps it, unsigned, after an explicit fingerprint-profile opt-in.
|
|
# # A real Claude OAuth credential always signs, on every upstream: a downstream Claude
|
|
# # Code pointed at CPA cannot produce that value itself.
|
|
# #
|
|
# # Example (official Anthropic or a custom Messages gateway):
|
|
# # - api-key: "your-key"
|
|
# # # base-url: "https://gateway.example" # omit for api.anthropic.com
|
|
# # fingerprint-profile: "claude-code-cli"
|
|
# # cloak:
|
|
# # mode: "always" # recommended when upstream rejects non-CLI clients
|
|
# #
|
|
# # Delegated Anthropic Messages OAuth files (Kimi, etc.) use the same field in the
|
|
# # auth JSON. Refresh keeps it. Example:
|
|
# # {
|
|
# # "type": "kimi",
|
|
# # "access_token": "...",
|
|
# # "refresh_token": "...",
|
|
# # "fingerprint-profile": "claude-code-cli"
|
|
# # }
|
|
# # fingerprint-profile: "claude-code-cli" # optional: default is empty (caller-owned); uncomment to opt into Claude Code CLI Messages fingerprinting
|
|
# experimental-cch-signing: false # deprecated compatibility field; CCH is generated automatically
|
|
# # for real Claude OAuth on any upstream, and for claude-code-cli profiles
|
|
# # only on api.anthropic.com; Vertex keeps provider-native signing
|
|
|
|
# Anthropic-Beta is assembled per request rather than sent as a fixed list, matching
|
|
# Claude Code 2.1.220: context-1m sits right after claude-code, mid-conversation-system
|
|
# is added only for models that accept a role=system turn, advanced-tool-use only when
|
|
# the request declares tools, and server-side-fallback / fallback-credit /
|
|
# structured-outputs trail effort. On direct api.anthropic.com a caller may only ask for
|
|
# betas real Claude Code also sends, and they are placed at their observed positions;
|
|
# anything else is dropped so the outgoing set stays one a real client could produce.
|
|
# Other Anthropic-compatible upstreams still forward caller betas verbatim.
|
|
#
|
|
# Default headers for Claude API requests. Update only after measuring a new Claude Code release.
|
|
# Unconfirmed clients use this CLI baseline. Verified native Claude Code CLI, sdk-cli,
|
|
# and VSCode requests preserve their measured entrypoint and software shape only when the
|
|
# Claude Code version, package version, and runtime version exactly match this configured
|
|
# baseline; unmeasured versions fall back to it. In legacy mode, timeout is a fallback and
|
|
# verified native OS/arch values remain client-supplied. When stabilize-device-profile is
|
|
# enabled, OS/arch are pinned to the values below and cached profiles remain constrained to
|
|
# the same exact software baseline rather than learning newer client versions.
|
|
# claude-header-defaults:
|
|
# user-agent: "claude-cli/2.1.220 (external, cli)"
|
|
# package-version: "0.94.0"
|
|
# runtime-version: "v26.3.0"
|
|
# os: "MacOS"
|
|
# arch: "arm64"
|
|
# timeout: "600"
|
|
# timezone: "Asia/Singapore" # fallback IANA timezone for cloaked currentDate; a credential JSON "timezone" takes priority
|
|
# stabilize-device-profile: false # optional, default false; set true to enable per-auth/API-key fingerprint pinning
|
|
|
|
# Default headers for Codex OAuth model requests.
|
|
# These are used only for file-backed/OAuth Codex requests when the client
|
|
# does not send the header. `user-agent` applies to HTTP and websocket requests;
|
|
# `beta-features` only applies to websocket requests. They do not apply to codex-api-key entries.
|
|
# codex-header-defaults:
|
|
# user-agent: "codex_cli_rs/0.114.0 (Mac OS 14.2.0; x86_64) vscode/1.111.0"
|
|
# beta-features: "multi_agent"
|
|
|
|
# OpenAI compatibility providers
|
|
# openai-compatibility:
|
|
# - name: "openrouter" # The name of the provider; it will be used in the user agent and other places.
|
|
# disabled: false # optional: set to true to disable this provider without removing it
|
|
# prefix: "test" # optional: require calls like "test/kimi-k2" to target this provider's credentials
|
|
# base-url: "https://openrouter.ai/api/v1" # The base URL of the provider.
|
|
# support-prompt-cache-key: false # optional: derive prompt_cache_key for requests from all input protocols
|
|
# disable-cooling: false # optional provider override: true disables cooling, false enables it; omit to inherit global
|
|
# request-retry: 3 # optional: per-provider override of the global request-retry; 0 disables retries; omit or set < 0 to use the global value
|
|
# request-scoped-errors: # optional: custom rules to classify upstream errors by status and body patterns
|
|
# - status: 400
|
|
# match:
|
|
# - "maximum_context_length"
|
|
# - "context_length_exceeded"
|
|
# match-regexr:
|
|
# - "maximum_context_length$"
|
|
# - "^context_length_exceeded"
|
|
# action: "stop" # "stop", "stop-and-cooldown", "continue", "continue-and-cooldown"
|
|
# headers:
|
|
# X-Custom-Header: "custom-value"
|
|
# # Values starting with "$" dynamically copy the header value from downstream client requests.
|
|
# # If the client did not send the specified header, the header is omitted.
|
|
# # X-Claude-Code-Session-Id: "$ABC" # copies client's "ABC" header
|
|
# api-key-entries:
|
|
# - api-key: "sk-or-v1-...b780"
|
|
# weight: 5 # optional: weighted-round-robin share; omitted defaults to 1; maximum 1,000,000
|
|
# proxy-url: "socks5://proxy.example.com:1080" # optional: per-key proxy override
|
|
# # proxy-url: "direct" # optional: explicit direct connect for this credential
|
|
# - api-key: "sk-or-v1-...b781" # without proxy-url
|
|
# models: # The models supported by the provider.
|
|
# - name: "moonshotai/kimi-k2:free" # The actual model name.
|
|
# alias: "kimi-k2" # The alias used in the API.
|
|
# display-name: "Kimi K2" # optional catalog display name
|
|
# max-context-length: 1048576 # optional: override Codex client context window metadata
|
|
# image: false # optional: set true to allow this model on /v1/images/generations and /v1/images/edits (not chat/responses image input)
|
|
# input-modalities: [text, image] # optional: declare /v1/chat/completions and /v1/responses multimodal input for Codex clients. Use [text] for upstreams that reject multimodal tool result content.
|
|
# output-modalities: [text] # optional: declare output modalities when known
|
|
# is-compat: false # optional: preserve Claude thinking blocks for compatible upstreams
|
|
# thinking: # optional: omit to default to levels ["low","medium","high"]
|
|
# levels: ["low", "medium", "high"]
|
|
# # You may repeat the same alias to build an internal model pool.
|
|
# # The client still sees only one alias in the model list.
|
|
# # Requests to that alias will round-robin across the upstream names below,
|
|
# # and if the chosen upstream fails before producing output, the request will
|
|
# # continue with the next upstream model in the same alias pool.
|
|
# - name: "deepseek-v3.1"
|
|
# alias: "claude-opus-4.66"
|
|
# - name: "glm-5"
|
|
# alias: "claude-opus-4.66"
|
|
# - name: "kimi-k2.5"
|
|
# alias: "claude-opus-4.66"
|
|
|
|
# Vertex API keys (Vertex-compatible endpoints, base-url is optional)
|
|
# vertex-api-key:
|
|
# - api-key: "vk-123..." # x-goog-api-key header
|
|
# weight: 5 # optional: weighted-round-robin share; omitted defaults to 1; maximum 1,000,000
|
|
# prefix: "test" # optional: require calls like "test/vertex-pro" to target this credential
|
|
# disable-cooling: false # optional override: true disables cooling, false enables it; omit to inherit global
|
|
# request-retry: 3 # optional: per-auth override of the global request-retry; 0 disables retries; omit or set < 0 to use the global value
|
|
# base-url: "https://example.com/api" # optional, e.g. https://zenmux.ai/api; falls back to Google Vertex when omitted
|
|
# proxy-url: "socks5://proxy.example.com:1080" # optional per-key proxy override
|
|
# # proxy-url: "direct" # optional: explicit direct connect for this credential
|
|
# headers:
|
|
# X-Custom-Header: "custom-value"
|
|
# # Values starting with "$" dynamically copy the header value from downstream client requests.
|
|
# # If the client did not send the specified header, the header is omitted.
|
|
# # X-Claude-Code-Session-Id: "$ABC" # copies client's "ABC" header
|
|
# models: # optional: map aliases to upstream model names
|
|
# - name: "gemini-2.5-flash" # upstream model name
|
|
# alias: "vertex-flash" # client-visible alias
|
|
# display-name: "Vertex Flash" # optional catalog display name
|
|
# thinking: # optional: exact thinking capability for this configured model
|
|
# levels: ["high", "medium", "low", "none", "auto"]
|
|
# - name: "gemini-2.5-pro"
|
|
# alias: "vertex-pro"
|
|
# excluded-models: # optional: models to exclude from listing
|
|
# - "imagen-3.0-generate-002"
|
|
# - "imagen-*"
|
|
|
|
# Global OAuth model name aliases (per channel)
|
|
# These aliases rename model IDs for both model listing and request routing.
|
|
# Supported channels: vertex, aistudio, antigravity, claude, codex, kimi, xai.
|
|
# NOTE: Aliases do not apply to gemini-api-key, interactions-api-key, codex-api-key, xai-api-key, claude-api-key, openai-compatibility, or vertex-api-key.
|
|
# NOTE: Because aliases affect the merged /v1 model list and merged request routing, overlapping
|
|
# client-visible names can become ambiguous across providers. For strict backend pinning, use
|
|
# unique aliases/prefixes or avoid overlapping names.
|
|
# You can repeat the same name with different aliases to expose multiple client model names.
|
|
# Optional per-entry fields:
|
|
# fork: true # keep the upstream model and also expose the alias as a separate client-visible model
|
|
# display-name: "Model Name" # override the human-readable name shown in model catalogs
|
|
# force-mapping: true # rewrite upstream response model fields back to the client-visible alias (example below uses antigravity only)
|
|
# Per-auth OAuth aliases can also be stored in an OAuth auth JSON file as "model-aliases".
|
|
# They apply only to that selected auth and take precedence over global aliases for the same client-visible alias.
|
|
# Example auth JSON:
|
|
# {
|
|
# "type": "codex",
|
|
# "email": "user@example.com",
|
|
# "model-aliases": [
|
|
# {"name": "gpt-5.3-codex-spark", "alias": "gpt-5.5"},
|
|
# {"name": "gpt-5.3-codex-spark", "alias": "gpt-5.4"}
|
|
# ]
|
|
# }
|
|
# oauth-model-alias:
|
|
# vertex:
|
|
# - name: "gemini-2.5-pro"
|
|
# alias: "g2.5p"
|
|
# aistudio:
|
|
# - name: "gemini-2.5-pro"
|
|
# alias: "g2.5p"
|
|
# antigravity:
|
|
# - name: "gemini-pro-agent" # upstream Antigravity model id
|
|
# alias: "gemini-3.1-pro-preview" # client-visible id (Gemini 3.1 Pro Preview)
|
|
# display-name: "Antigravity Gemini 3.1 Pro" # optional catalog display name
|
|
# fork: true
|
|
# force-mapping: true
|
|
# claude:
|
|
# - name: "claude-sonnet-4-5-20250929"
|
|
# alias: "cs4.5"
|
|
# codex:
|
|
# - name: "gpt-5"
|
|
# alias: "g5"
|
|
# kimi:
|
|
# - name: "kimi-k2.5"
|
|
# alias: "k2.5"
|
|
# xai:
|
|
# - name: "grok-4.3"
|
|
# alias: "grok-latest"
|
|
# sample-provider: # plugin provider keys are supported for OAuth plugins
|
|
# - name: "sample-model-latest"
|
|
# alias: "sample-latest"
|
|
|
|
# OAuth provider excluded models
|
|
# oauth-excluded-models:
|
|
# vertex:
|
|
# - "gemini-3-pro-preview"
|
|
# aistudio:
|
|
# - "gemini-3-pro-preview"
|
|
# antigravity:
|
|
# - "gemini-3-pro-preview"
|
|
# claude:
|
|
# - "claude-3-5-haiku-20241022"
|
|
# codex:
|
|
# - "gpt-5-codex-mini"
|
|
# kimi:
|
|
# - "kimi-k2-thinking"
|
|
# xai:
|
|
# - "grok-3-mini"
|
|
|
|
# OAuth provider request-scoped error rules (custom error classification for OAuth credentials)
|
|
# oauth-request-scoped-errors:
|
|
# vertex:
|
|
# - status: 400
|
|
# match:
|
|
# - "maximum_context_length"
|
|
# - "context_length_exceeded"
|
|
# match-regexr:
|
|
# - "maximum_context_length$"
|
|
# - "^context_length_exceeded"
|
|
# action: "stop" # options: "stop", "stop-and-cooldown", "continue", "continue-and-cooldown"
|
|
# aistudio:
|
|
# - status: 400
|
|
# match:
|
|
# - "invalid_argument"
|
|
# action: "stop"
|
|
# antigravity:
|
|
# - status: 500
|
|
# match:
|
|
# - "internal_server_error"
|
|
# action: "stop-and-cooldown"
|
|
# claude:
|
|
# - status: 400
|
|
# match:
|
|
# - "prompt is too long"
|
|
# action: "stop"
|
|
# codex:
|
|
# - status: 400
|
|
# match:
|
|
# - "context_window_exceeded"
|
|
# action: "stop"
|
|
# kimi:
|
|
# - status: 400
|
|
# match:
|
|
# - "length_limit"
|
|
# action: "stop"
|
|
# xai:
|
|
# - status: 400
|
|
# match:
|
|
# - "max_tokens_exceeded"
|
|
# action: "stop"
|
|
|
|
# Optional payload configuration
|
|
# payload:
|
|
# default: # Default rules only set parameters when they are missing in the payload.
|
|
# - models:
|
|
# - name: "gemini-2.5-pro" # Supports wildcards (e.g., "gemini-*")
|
|
# protocol: "gemini" # restricts the rule to a specific protocol, options: openai, gemini, claude, codex, antigravity
|
|
# from-protocol: "responses" # restricts the rule to the source protocol, options: openai, responses, gemini, claude
|
|
# headers: # all configured request headers must match; values support "*" wildcards
|
|
# X-Client-Tier: "tenant-*-region-*"
|
|
# match: # all payload JSON paths must equal the configured values
|
|
# - "metadata.client": "codex"
|
|
# not-match: # payload JSON paths must not equal the configured values
|
|
# - "metadata.mode": "dev"
|
|
# exist: # all payload JSON paths must exist and not be null
|
|
# - "tools.#(type==\"web_search\").type"
|
|
# not-exist: # all payload JSON paths must be missing or null
|
|
# - "metadata.disable_payload"
|
|
# params: # JSON path (gjson/sjson syntax) -> value
|
|
# "generationConfig.thinkingConfig.thinkingBudget": 32768
|
|
# default-raw: # Default raw rules set parameters using raw JSON when missing (must be valid JSON).
|
|
# - models:
|
|
# - name: "gemini-2.5-pro" # Supports wildcards (e.g., "gemini-*")
|
|
# protocol: "gemini" # restricts the rule to a specific protocol, options: openai, gemini, claude, codex, antigravity
|
|
# params: # JSON path (gjson/sjson syntax) -> raw JSON value (strings are used as-is, must be valid JSON)
|
|
# "generationConfig.responseJsonSchema": "{\"type\":\"object\",\"properties\":{\"answer\":{\"type\":\"string\"}}}"
|
|
# override: # Override rules always set parameters, overwriting any existing values.
|
|
# - models:
|
|
# - name: "gpt-5.4-fast"
|
|
# protocol: "codex"
|
|
# - name: "gpt-5.5-fast"
|
|
# protocol: "codex"
|
|
# params:
|
|
# service_tier: priority
|
|
# - models:
|
|
# - name: "gpt-*" # Supports wildcards (e.g., "gpt-*")
|
|
# protocol: "codex" # restricts the rule to a specific protocol, options: openai, gemini, claude, codex, antigravity
|
|
# params: # JSON path (gjson/sjson syntax) -> value
|
|
# "reasoning.effort": "high"
|
|
# override-raw: # Override raw rules always set parameters using raw JSON (must be valid JSON).
|
|
# - models:
|
|
# - name: "gpt-*" # Supports wildcards (e.g., "gpt-*")
|
|
# protocol: "codex" # restricts the rule to a specific protocol, options: openai, gemini, claude, codex, antigravity
|
|
# params: # JSON path (gjson/sjson syntax) -> raw JSON value (strings are used as-is, must be valid JSON)
|
|
# "response_format": "{\"type\":\"json_schema\",\"json_schema\":{\"name\":\"answer\",\"schema\":{\"type\":\"object\"}}}"
|
|
# filter: # Filter rules remove specified parameters from the payload.
|
|
# - models:
|
|
# - name: "gemini-2.5-pro" # Supports wildcards (e.g., "gemini-*")
|
|
# protocol: "gemini" # restricts the rule to a specific protocol, options: openai, gemini, claude, codex, antigravity
|
|
# params: # JSON paths (gjson/sjson syntax) to remove from the payload
|
|
# - "generationConfig.thinkingConfig.thinkingBudget"
|
|
# - "generationConfig.responseJsonSchema"
|