{"openapi":"3.0.1","info":{"title":"Edgee Gateway API","version":"2026-09-24","description":"HTTP API of the Edgee Agent Gateway. Route model requests using Chat Completions, Anthropic Messages or OpenAI Responses formats; list models, count tokens, and compress request payloads. Provider-specific features depend on the destination model and whether a request is forwarded natively or converted between formats. This reference covers the public managed API, not the CLI subscription/relay protocol.","x-gateway-revision":"4e21a9c","x-gateway-source":"edgee-ai/gateway/core/src/routing/mod.rs"},"servers":[{"url":"https://edgee.io","description":"Edgee AI Gateway"}],"security":[{"bearerAuth":[]}],"paths":{"/v1/chat/completions":{"post":{"operationId":"createChatCompletion","summary":"Create chat completion","description":"Create a chat completion, optionally streamed as SSE. Native dispatch preserves provider-specific JSON fields when the destination supports the caller’s format. Cross-format routing converts through the Gateway’s shared representation: provider-only features are not guaranteed to survive. Support for reasoning, images, built-in tools and structured outputs depends on the destination model.","tags":["Chat"],"parameters":[{"name":"X-Edgee-Tags","in":"header","schema":{"type":"string"},"description":"Comma-separated analytics tags. Trimmed and merged with body tags."},{"name":"X-Edgee-Debug","in":"header","schema":{"type":"string","enum":["true","1"]},"description":"Enable debug capture for this request. Does not add a debug object to the API response."},{"name":"X-Edgee-Compression-Tool-Result-Trimming","in":"header","required":false,"schema":{"type":"string","enum":["true","false","1","0","on","off"]},"description":"Turn tool-result trimming on or off. Overrides the API key setting for this request only. Values are case-insensitive, and a missing or invalid value keeps the key's setting."},{"name":"X-Edgee-Compression-Tool-Surface-Reduction","in":"header","required":false,"schema":{"type":"string","enum":["true","false","1","0","on","off"]},"description":"Turn MCP tool surface reduction on or off. The threshold and resolver mode still come from the API key. Overrides the API key setting for this request only. Values are case-insensitive, and a missing or invalid value keeps the key's setting."},{"name":"X-Edgee-Compression-Brevity","in":"header","required":false,"schema":{"type":"string","enum":["true","false","1","0","on","off"]},"description":"Turn output brevity on or off. Overrides the API key setting for this request only. Values are case-insensitive, and a missing or invalid value keeps the key's setting."},{"name":"X-Edgee-Session-Id","in":"header","schema":{"type":"string"},"description":"Session identifier used to group request usage."}],"requestBody":{"required":true,"content":{"application/json":{"schema":{"$ref":"#/components/schemas/ChatCompletionRequest"},"example":{"model":"openai/gpt-5.2","messages":[{"role":"user","content":"Hello!"}],"max_completion_tokens":256}}}},"responses":{"200":{"description":"Chat completion created successfully","headers":{"X-Edgee-Provider":{"description":"Provider selected for this request when known before response headers are sent. May be absent for streaming/native paths.","schema":{"type":"string"}},"X-Edgee-Fallback-Used":{"description":"Present as 1 when a fallback was used and the decision is known before headers are sent. A strategy reroute alone is not a fallback.","schema":{"type":"string","enum":["1"]}}},"content":{"application/json":{"schema":{"$ref":"#/components/schemas/ChatCompletionResponse"},"example":{"id":"chatcmpl-123","object":"chat.completion","created":1677652288,"model":"openai/gpt-5.2","choices":[{"index":0,"message":{"role":"assistant","content":"Hello! How can I assist you today?"},"finish_reason":"stop"}],"usage":{"prompt_tokens":10,"completion_tokens":10,"total_tokens":20,"input_tokens_details":{"cached_tokens":0},"output_tokens_details":{"reasoning_tokens":0}},"compression":{"saved_tokens":450,"cost_savings":27000,"reduction":48.99884991374353,"time_ms":12}}},"text/event-stream":{"schema":{"type":"string","format":"binary","description":"Server-Sent Events stream. Each event is a JSON object prefixed with 'data: ' and followed by two newlines. The stream consists of multiple `ChatCompletionChunk` objects, and optionally a final chunk with usage statistics if `stream_options.include_usage` is true."},"examples":{"contentChunk":{"value":"data: {\"id\":\"chatcmpl-123\",\"object\":\"chat.completion.chunk\",\"created\":1677652288,\"model\":\"openai/gpt-5.2\",\"choices\":[{\"index\":0,\"delta\":{\"content\":\"Hello\"},\"finish_reason\":null}]}\n\n"},"roleChunk":{"value":"data: {\"id\":\"chatcmpl-123\",\"object\":\"chat.completion.chunk\",\"created\":1677652288,\"model\":\"openai/gpt-5.2\",\"choices\":[{\"index\":0,\"delta\":{\"role\":\"assistant\"},\"finish_reason\":null}]}\n\n"},"finalChunk":{"value":"data: {\"id\":\"chatcmpl-123\",\"object\":\"chat.completion.chunk\",\"created\":1677652288,\"model\":\"openai/gpt-5.2\",\"choices\":[{\"index\":0,\"delta\":{},\"finish_reason\":\"stop\"}],\"usage\":{\"prompt_tokens\":10,\"completion_tokens\":10,\"total_tokens\":20,\"input_tokens_details\":{\"cached_tokens\":0},\"output_tokens_details\":{\"reasoning_tokens\":0}}}\n\n"}}}}},"400":{"description":"Malformed request, invalid/disabled model or provider, or unsupported conversion.","content":{"application/json":{"schema":{"$ref":"#/components/schemas/ErrorResponse"}}}},"401":{"description":"Missing or invalid API credentials.","content":{"application/json":{"schema":{"$ref":"#/components/schemas/ErrorResponse"}}}},"403":{"description":"Key inactive, expired, model restriction or other permission failure.","content":{"application/json":{"schema":{"$ref":"#/components/schemas/ErrorResponse"}}}},"429":{"description":"Usage or routing-strategy budget exhausted, or provider rate limit.","content":{"application/json":{"schema":{"$ref":"#/components/schemas/ErrorResponse"}}}},"500":{"description":"Internal processing error.","content":{"application/json":{"schema":{"$ref":"#/components/schemas/ErrorResponse"}}}},"502":{"description":"Upstream provider failure.","content":{"application/json":{"schema":{"$ref":"#/components/schemas/ErrorResponse"}}}},"503":{"description":"Usage accounting or upstream service temporarily unavailable.","content":{"application/json":{"schema":{"$ref":"#/components/schemas/ErrorResponse"}}}},"529":{"description":"Upstream provider overloaded (when preserved by the request path).","content":{"application/json":{"schema":{"$ref":"#/components/schemas/ErrorResponse"}}}}},"x-gateway-source":"core/src/routing/completions/mod.rs"}},"/v1/messages":{"post":{"operationId":"createMessage","summary":"Create message (Anthropic format)","description":"Create a response in Anthropic Messages format. Supports native Anthropic dispatch and conversion to other supported providers; it is not restricted to the Anthropic provider. Native dispatch preserves provider-specific JSON fields when the destination supports the caller’s format. Cross-format routing converts through the Gateway’s shared representation: provider-only features are not guaranteed to survive. Support for reasoning, images, built-in tools and structured outputs depends on the destination model.","tags":["Messages"],"parameters":[{"name":"X-Edgee-Tags","in":"header","schema":{"type":"string"},"description":"Comma-separated analytics tags. Trimmed and merged with body tags."},{"name":"X-Edgee-Debug","in":"header","schema":{"type":"string","enum":["true","1"]},"description":"Enable debug capture for this request. Does not add a debug object to the API response."},{"name":"X-Edgee-Compression-Tool-Result-Trimming","in":"header","required":false,"schema":{"type":"string","enum":["true","false","1","0","on","off"]},"description":"Turn tool-result trimming on or off. Overrides the API key setting for this request only. Values are case-insensitive, and a missing or invalid value keeps the key's setting."},{"name":"X-Edgee-Compression-Tool-Surface-Reduction","in":"header","required":false,"schema":{"type":"string","enum":["true","false","1","0","on","off"]},"description":"Turn MCP tool surface reduction on or off. The threshold and resolver mode still come from the API key. Overrides the API key setting for this request only. Values are case-insensitive, and a missing or invalid value keeps the key's setting."},{"name":"X-Edgee-Compression-Brevity","in":"header","required":false,"schema":{"type":"string","enum":["true","false","1","0","on","off"]},"description":"Turn output brevity on or off. Overrides the API key setting for this request only. Values are case-insensitive, and a missing or invalid value keeps the key's setting."},{"name":"X-Edgee-Session-Id","in":"header","schema":{"type":"string"},"description":"Session identifier used to group request usage."},{"name":"anthropic-version","in":"header","schema":{"type":"string","example":"2023-06-01"},"description":"Anthropic API version header. Forwarded on the native Anthropic path; not a cross-provider capability switch."},{"name":"anthropic-beta","in":"header","schema":{"type":"string"},"description":"Provider beta feature flags for native Anthropic requests. Availability depends on the upstream provider."}],"security":[{"bearerAuth":[]},{"apiKeyAuth":[]}],"requestBody":{"required":true,"content":{"application/json":{"schema":{"$ref":"#/components/schemas/CreateMessageRequest"},"example":{"model":"anthropic/claude-sonnet-4-6","max_tokens":256,"messages":[{"role":"user","content":"Hello!"}]}}}},"responses":{"200":{"description":"Message created successfully","headers":{"X-Edgee-Provider":{"description":"Provider selected for this request when known before response headers are sent. May be absent for streaming/native paths.","schema":{"type":"string"}},"X-Edgee-Fallback-Used":{"description":"Present as 1 when a fallback was used and the decision is known before headers are sent. A strategy reroute alone is not a fallback.","schema":{"type":"string","enum":["1"]}}},"content":{"application/json":{"schema":{"$ref":"#/components/schemas/CreateMessageResponse"}},"text/event-stream":{"schema":{"type":"string","format":"binary","description":"Anthropic-style SSE: message_start, content_block_start/delta/stop, message_delta and message_stop; reasoning and tool argument deltas use their corresponding block types. Inspect error events even if HTTP status is 200."}}}},"400":{"description":"Malformed request, invalid/disabled model or provider, or unsupported conversion.","content":{"application/json":{"schema":{"anyOf":[{"$ref":"#/components/schemas/AnthropicErrorResponse"},{"$ref":"#/components/schemas/ErrorResponse"}]}}}},"401":{"description":"Missing or invalid API credentials.","content":{"application/json":{"schema":{"anyOf":[{"$ref":"#/components/schemas/AnthropicErrorResponse"},{"$ref":"#/components/schemas/ErrorResponse"}]}}}},"403":{"description":"Key inactive, expired, model restriction or other permission failure.","content":{"application/json":{"schema":{"anyOf":[{"$ref":"#/components/schemas/AnthropicErrorResponse"},{"$ref":"#/components/schemas/ErrorResponse"}]}}}},"429":{"description":"Usage or routing-strategy budget exhausted, or provider rate limit.","content":{"application/json":{"schema":{"anyOf":[{"$ref":"#/components/schemas/AnthropicErrorResponse"},{"$ref":"#/components/schemas/ErrorResponse"}]}}}},"500":{"description":"Internal processing error.","content":{"application/json":{"schema":{"anyOf":[{"$ref":"#/components/schemas/AnthropicErrorResponse"},{"$ref":"#/components/schemas/ErrorResponse"}]}}}},"502":{"description":"Upstream provider failure.","content":{"application/json":{"schema":{"anyOf":[{"$ref":"#/components/schemas/AnthropicErrorResponse"},{"$ref":"#/components/schemas/ErrorResponse"}]}}}},"503":{"description":"Usage accounting or upstream service temporarily unavailable.","content":{"application/json":{"schema":{"anyOf":[{"$ref":"#/components/schemas/AnthropicErrorResponse"},{"$ref":"#/components/schemas/ErrorResponse"}]}}}},"529":{"description":"Upstream provider overloaded (when preserved by the request path).","content":{"application/json":{"schema":{"anyOf":[{"$ref":"#/components/schemas/AnthropicErrorResponse"},{"$ref":"#/components/schemas/ErrorResponse"}]}}}}},"x-gateway-source":"core/src/routing/messages/mod.rs"}},"/v1/responses":{"post":{"operationId":"createResponse","summary":"Create response (OpenAI Responses format)","description":"Create a response in OpenAI Responses format. Input can be text or typed items; tools use the flat Responses format. Only POST /v1/responses is exposed: the Gateway does not expose response retrieval, deletion or conversation-management routes. Native dispatch preserves provider-specific JSON fields when the destination supports the caller’s format. Cross-format routing converts through the Gateway’s shared representation: provider-only features are not guaranteed to survive. Support for reasoning, images, built-in tools and structured outputs depends on the destination model.","tags":["Responses"],"parameters":[{"name":"X-Edgee-Tags","in":"header","schema":{"type":"string"},"description":"Comma-separated analytics tags. Trimmed and merged with body tags."},{"name":"X-Edgee-Debug","in":"header","schema":{"type":"string","enum":["true","1"]},"description":"Enable debug capture for this request. Does not add a debug object to the API response."},{"name":"X-Edgee-Compression-Tool-Result-Trimming","in":"header","required":false,"schema":{"type":"string","enum":["true","false","1","0","on","off"]},"description":"Turn tool-result trimming on or off. Overrides the API key setting for this request only. Values are case-insensitive, and a missing or invalid value keeps the key's setting."},{"name":"X-Edgee-Compression-Tool-Surface-Reduction","in":"header","required":false,"schema":{"type":"string","enum":["true","false","1","0","on","off"]},"description":"Turn MCP tool surface reduction on or off. The threshold and resolver mode still come from the API key. Overrides the API key setting for this request only. Values are case-insensitive, and a missing or invalid value keeps the key's setting."},{"name":"X-Edgee-Compression-Brevity","in":"header","required":false,"schema":{"type":"string","enum":["true","false","1","0","on","off"]},"description":"Turn output brevity on or off. Overrides the API key setting for this request only. Values are case-insensitive, and a missing or invalid value keeps the key's setting."},{"name":"X-Edgee-Session-Id","in":"header","schema":{"type":"string"},"description":"Session identifier used to group request usage."}],"requestBody":{"required":true,"content":{"application/json":{"schema":{"$ref":"#/components/schemas/ResponsesRequest"},"examples":{"basic":{"summary":"Basic request","value":{"model":"openai/gpt-5.2","input":"Hello!","max_output_tokens":256}},"basicText":{"summary":"Basic text input","value":{"model":"openai/gpt-5.2","input":"What is the capital of France?"}},"withInstructions":{"summary":"Instructions and message array","value":{"model":"openai/gpt-5.2","instructions":"You are a helpful assistant that responds concisely.","input":[{"role":"user","content":"Summarize the water cycle in one sentence."}]}},"streaming":{"summary":"Streaming","value":{"model":"openai/gpt-5.2","stream":true,"input":"Write a short poem about the ocean."}},"withTools":{"summary":"With tools","value":{"model":"openai/gpt-5.2","input":"What is the weather in Paris?","tools":[{"type":"function","name":"get_weather","description":"Get current weather for a location","parameters":{"type":"object","properties":{"location":{"type":"string","description":"City name"}},"required":["location"]}}]}},"multiTurnTools":{"summary":"Multi-turn with tool results","value":{"model":"openai/gpt-5.2","input":[{"role":"user","content":"What is the weather in Paris?"},{"type":"function_call","call_id":"call_abc123","name":"get_weather","arguments":"{\"location\": \"Paris\"}"},{"type":"function_call_output","call_id":"call_abc123","output":"{\"temperature\": 22, \"condition\": \"sunny\"}"}],"tools":[{"type":"function","name":"get_weather","description":"Get current weather for a location","parameters":{"type":"object","properties":{"location":{"type":"string"}},"required":["location"]}}]}}}}}},"responses":{"200":{"description":"Response created successfully","headers":{"X-Edgee-Provider":{"description":"Provider selected for this request when known before response headers are sent. May be absent for streaming/native paths.","schema":{"type":"string"}},"X-Edgee-Fallback-Used":{"description":"Present as 1 when a fallback was used and the decision is known before headers are sent. A strategy reroute alone is not a fallback.","schema":{"type":"string","enum":["1"]}}},"content":{"application/json":{"schema":{"$ref":"#/components/schemas/ResponsesResponse"},"example":{"id":"resp_abc123","object":"response","status":"completed","created_at":1677652288,"model":"openai/gpt-5.2","output":[{"id":"msg_1","type":"message","status":"completed","role":"assistant","content":[{"type":"output_text","text":"Paris"}]}],"usage":{"input_tokens":8,"output_tokens":1,"total_tokens":9}}},"text/event-stream":{"schema":{"type":"string","format":"binary","description":"Server-Sent Events stream. Each event is a JSON object prefixed with 'data: ' and followed by two newlines. Each event has a `type` field. Text responses produce: `response.created`, `response.output_item.added`, `response.content_part.added`, `response.output_text.delta`, `response.output_text.done`, `response.content_part.done`, `response.output_item.done`, `response.completed`. Tool calls additionally produce: `response.function_call_arguments.delta`, `response.function_call_arguments.done`. Reasoning also emits response.reasoning_summary_part.added/done and response.reasoning_summary_text.delta/done. Freeform tools emit response.custom_tool_call_input.delta/done. There is no [DONE] sentinel. Failures after headers are sent are reported inside the stream."},"examples":{"createdEvent":{"value":"data: {\"type\":\"response.created\",\"response\":{\"id\":\"resp_abc123\",\"object\":\"response\",\"status\":\"in_progress\",\"created_at\":1677652288.0,\"model\":\"openai/gpt-5.2\",\"output\":[]}}\n\n"},"outputTextDeltaEvent":{"value":"data: {\"type\":\"response.output_text.delta\",\"item_id\":\"msg_1\",\"output_index\":0,\"content_index\":0,\"delta\":\"Hello\"}\n\n"},"completedEvent":{"value":"data: {\"type\":\"response.completed\",\"response\":{\"id\":\"resp_abc123\",\"object\":\"response\",\"status\":\"completed\",\"created_at\":1677652288.0,\"model\":\"openai/gpt-5.2\",\"output\":[{\"id\":\"msg_1\",\"type\":\"message\",\"status\":\"completed\",\"role\":\"assistant\",\"content\":[{\"type\":\"output_text\",\"text\":\"Hello\"}]}],\"usage\":{\"input_tokens\":8,\"output_tokens\":1,\"total_tokens\":9}}}\n\n"}}}}},"400":{"description":"Malformed request, invalid/disabled model or provider, or unsupported conversion.","content":{"application/json":{"schema":{"$ref":"#/components/schemas/ErrorResponse"}}}},"401":{"description":"Missing or invalid API credentials.","content":{"application/json":{"schema":{"$ref":"#/components/schemas/ErrorResponse"}}}},"403":{"description":"Key inactive, expired, model restriction or other permission failure.","content":{"application/json":{"schema":{"$ref":"#/components/schemas/ErrorResponse"}}}},"429":{"description":"Usage or routing-strategy budget exhausted, or provider rate limit.","content":{"application/json":{"schema":{"$ref":"#/components/schemas/ErrorResponse"}}}},"500":{"description":"Internal processing error.","content":{"application/json":{"schema":{"$ref":"#/components/schemas/ErrorResponse"}}}},"502":{"description":"Upstream provider failure.","content":{"application/json":{"schema":{"$ref":"#/components/schemas/ErrorResponse"}}}},"503":{"description":"Usage accounting or upstream service temporarily unavailable.","content":{"application/json":{"schema":{"$ref":"#/components/schemas/ErrorResponse"}}}},"529":{"description":"Upstream provider overloaded (when preserved by the request path).","content":{"application/json":{"schema":{"$ref":"#/components/schemas/ErrorResponse"}}}}},"x-gateway-source":"core/src/routing/responses/mod.rs"}},"/v1/models":{"get":{"operationId":"listModels","summary":"List models","description":"List active models. Authentication is optional. With a valid API key, organization model/provider blacklists and BYOK-only availability filter the list. An invalid supplied key returns 401. No provider query filter is implemented; results are not a guarantee that every model is allowed by the key’s model restrictions.","tags":["Models"],"parameters":[],"responses":{"200":{"description":"List of available models","content":{"application/json":{"schema":{"$ref":"#/components/schemas/ModelsResponse"},"example":{"object":"list","data":[{"id":"openai/gpt-5.2","object":"model","created":1677610602,"owned_by":"openai"},{"id":"anthropic/claude-opus-4-6","object":"model","created":1677610602,"owned_by":"anthropic"}]}}}},"401":{"description":"Missing or invalid API credentials.","content":{"application/json":{"schema":{"$ref":"#/components/schemas/ErrorResponse"}}}},"500":{"description":"Internal processing error.","content":{"application/json":{"schema":{"$ref":"#/components/schemas/ErrorResponse"}}}}},"security":[{},{"bearerAuth":[]},{"apiKeyAuth":[]}],"x-gateway-source":"core/src/routing/models.rs"}},"/v1/messages/count_tokens":{"post":{"operationId":"countTokensMessages","summary":"Count tokens (Anthropic format)","description":"Count tokens using Anthropic’s provider tokenizer. Requires a model with an Anthropic provider route and provider credentials. Unlike /v1/count_tokens, this sends the request to Anthropic and includes supported tool/system input. max_tokens is not required. Subscription passthrough is a separate CLI flow.","tags":["Tokens"],"security":[{"bearerAuth":[]},{"apiKeyAuth":[]}],"requestBody":{"required":true,"content":{"application/json":{"schema":{"$ref":"#/components/schemas/MessageCountTokensRequest"},"example":{"model":"anthropic/claude-sonnet-4-6","messages":[{"role":"user","content":"Hello!"}]}}}},"responses":{"200":{"description":"Token count estimated successfully","content":{"application/json":{"schema":{"$ref":"#/components/schemas/CountTokensResponse"},"example":{"input_tokens":14}}}},"400":{"description":"Malformed request, invalid/disabled model or provider, or unsupported conversion.","content":{"application/json":{"schema":{"anyOf":[{"$ref":"#/components/schemas/AnthropicErrorResponse"},{"$ref":"#/components/schemas/ErrorResponse"}]}}}},"401":{"description":"Missing or invalid API credentials.","content":{"application/json":{"schema":{"anyOf":[{"$ref":"#/components/schemas/AnthropicErrorResponse"},{"$ref":"#/components/schemas/ErrorResponse"}]}}}},"403":{"description":"Key inactive, expired, model restriction or other permission failure.","content":{"application/json":{"schema":{"anyOf":[{"$ref":"#/components/schemas/AnthropicErrorResponse"},{"$ref":"#/components/schemas/ErrorResponse"}]}}}},"500":{"description":"Internal processing error.","content":{"application/json":{"schema":{"anyOf":[{"$ref":"#/components/schemas/AnthropicErrorResponse"},{"$ref":"#/components/schemas/ErrorResponse"}]}}}}},"x-gateway-source":"core/src/routing/messages/mod.rs"}},"/v1/compress":{"post":{"operationId":"compress","summary":"Compress request","description":"Compresses an LLM request payload and returns it with the `messages`, `input`, `system`, and `tools` fields replaced by their compressed versions. Accepts OpenAI Chat Completions, Anthropic Messages, and OpenAI Responses API formats — the wire format is auto-detected from the request body.\n\nNo LLM provider is called. This is a pure pre-processing step intended for teams running their own LLM gateways who want Edgee token compression without routing requests through Edgee.","tags":["Compress"],"security":[{"bearerAuth":[]},{"apiKeyAuth":[]}],"requestBody":{"required":true,"content":{"application/json":{"schema":{"$ref":"#/components/schemas/CompressRequest"},"examples":{"basic":{"summary":"Basic request","value":{"model":"openai/gpt-5.2","messages":[{"role":"user","content":"Hello!"}]}},"openaiChat":{"summary":"OpenAI Chat Completions format","value":{"model":"openai/gpt-5.2","messages":[{"role":"system","content":"You are a helpful assistant."},{"role":"user","content":"What is the capital of France?"},{"role":"assistant","content":"Paris."},{"role":"tool","tool_call_id":"call_abc","content":"<large tool result>"}],"tools":[{"type":"function","function":{"name":"get_weather","description":"Get weather for a location","parameters":{"type":"object","properties":{"location":{"type":"string"}}}}}]}},"anthropic":{"summary":"Anthropic Messages format","value":{"model":"anthropic/claude-sonnet-4-6","max_tokens":1024,"system":"You are a helpful assistant.","messages":[{"role":"user","content":"What is the capital of France?"},{"role":"assistant","content":"Paris."}]}},"responsesApi":{"summary":"OpenAI Responses API format","value":{"model":"openai/gpt-5.2","instructions":"You are a helpful assistant.","input":[{"role":"user","content":"What is the capital of France?"},{"type":"function_call","call_id":"call_abc","name":"get_weather","arguments":"{\"location\":\"Paris\"}"},{"type":"function_call_output","call_id":"call_abc","output":"<large tool result>"}]}}}}}},"responses":{"200":{"description":"Request compressed successfully. The response mirrors the input format with the `messages`, `input`, `system`, and `tools` fields replaced by their compressed versions. All other fields pass through unchanged. A `compression` metadata object is appended when trimming runs. When the tool-result-trimming header disables compression, the original payload is returned without added compression metadata.","content":{"application/json":{"schema":{"$ref":"#/components/schemas/CompressResponse"},"example":{"model":"openai/gpt-5.2","messages":[{"role":"system","content":"You are a helpful assistant."},{"role":"user","content":"What is the capital of France?"},{"role":"assistant","content":"Paris."},{"role":"tool","tool_call_id":"call_abc","content":"<trimmed>"}],"tools":[{"type":"function","function":{"name":"get_weather","description":"Get weather for a location","parameters":{"type":"object","properties":{"location":{"type":"string"}}}}}],"compression":{"technique":"tool","applied_strategies":["tool_result_trimming"],"compression_rate":0.81,"uncompressed_input_tokens":1000,"compressed_input_tokens":810,"compression_time_ms":12}}}}},"400":{"description":"Malformed request, invalid/disabled model or provider, or unsupported conversion.","content":{"application/json":{"schema":{"$ref":"#/components/schemas/ErrorResponse"}}}},"401":{"description":"Missing or invalid API credentials.","content":{"application/json":{"schema":{"$ref":"#/components/schemas/ErrorResponse"}}}},"500":{"description":"Internal processing error.","content":{"application/json":{"schema":{"$ref":"#/components/schemas/ErrorResponse"}}}}},"x-gateway-source":"core/src/routing/compress.rs","parameters":[{"name":"X-Edgee-Compression-Tool-Result-Trimming","in":"header","required":false,"schema":{"type":"string","enum":["true","false","1","0","on","off"]},"description":"Tool-result trimming is on by default on this endpoint. Set to `false`, `0` or `off` to get the payload back unchanged, without the `compression` field. Values are case-insensitive, and an invalid value leaves trimming on."}]}},"/v1/count_tokens":{"post":{"operationId":"countTokens","summary":"Count tokens","description":"Estimate text tokens locally without a provider call. Requires model, accepts optional messages and system. Counts plain content strings and type:text blocks only. Images, tool definitions, non-text blocks and message framing overhead are not counted; this is not a provider billing estimate. Tokenizer selection uses the configured provider model name unless explicitly overridden.","tags":["Tokens"],"requestBody":{"required":true,"content":{"application/json":{"schema":{"$ref":"#/components/schemas/CountTokensRequest"},"example":{"model":"openai/gpt-5.2","messages":[{"role":"system","content":"You are a helpful assistant."},{"role":"user","content":"What is the capital of France?"}]}}}},"responses":{"200":{"description":"Token count estimated successfully","content":{"application/json":{"schema":{"$ref":"#/components/schemas/CountTokensResponse"},"example":{"input_tokens":42}}}},"400":{"description":"Malformed request, invalid/disabled model or provider, or unsupported conversion.","content":{"application/json":{"schema":{"$ref":"#/components/schemas/ErrorResponse"}}}},"401":{"description":"Missing or invalid API credentials.","content":{"application/json":{"schema":{"$ref":"#/components/schemas/ErrorResponse"}}}},"403":{"description":"Key inactive, expired, model restriction or other permission failure.","content":{"application/json":{"schema":{"$ref":"#/components/schemas/ErrorResponse"}}}},"500":{"description":"Internal processing error.","content":{"application/json":{"schema":{"$ref":"#/components/schemas/ErrorResponse"}}}}},"security":[{"bearerAuth":[]},{"apiKeyAuth":[]}],"x-gateway-source":"core/src/routing/count_tokens.rs"}}},"components":{"schemas":{"CountTokensRequest":{"type":"object","required":["model"],"properties":{"model":{"type":"string","description":"Model identifier from GET /v1/models, in author/model format, or a configured global alias. Append :provider to request a specific provider route. Available models and providers depend on your key and organization policies.","example":"openai/gpt-5.2"},"messages":{"type":"array","description":"Optional array of message objects to count tokens for. Accepts both OpenAI chat format (with `system`, `user`, `assistant` roles) and Anthropic Messages format; the format is auto-detected from the message structure. Defaults to an empty array.","items":{"type":"object","additionalProperties":true,"description":"Content may be a string or Anthropic text blocks. Other content contributes zero tokens."},"default":[]},"system":{"description":"Optional system prompt. Accepts a plain string or an array of Anthropic content blocks. Used when counting tokens for an Anthropic-style request.","oneOf":[{"type":"string"},{"type":"array","items":{"type":"object"}}]},"tokenizer":{"type":"string","enum":["cl100k_base","o200k_base"],"description":"Explicit tokenizer override. When omitted, the gateway picks one based on `model`."}}},"CountTokensResponse":{"type":"object","required":["input_tokens"],"properties":{"input_tokens":{"type":"integer","description":"Estimated number of input tokens for the provided messages. This is an approximation, counts may differ from provider-native tokenizers. Use for estimation and budgeting, not exact billing.","minimum":0,"example":42}}},"ChatCompletionRequest":{"type":"object","required":["model","messages"],"properties":{"model":{"type":"string","description":"Model identifier from GET /v1/models, in author/model format, or a configured global alias. Append :provider to request a specific provider route. Available models and providers depend on your key and organization policies.","example":"openai/gpt-5.2"},"messages":{"type":"array","description":"A list of messages comprising the conversation so far. The shared typed converter also accepts input as an alias, but messages is the portable Chat Completions field for native providers.","items":{"$ref":"#/components/schemas/Message"},"minItems":1},"max_tokens":{"type":"integer","description":"The maximum number of tokens that can be generated in the chat completion.","minimum":1},"stream":{"type":"boolean","description":"If set, partial message deltas will be sent, as in OpenAI. Streamed chunks are sent as Server-Sent Events (SSE).","default":false},"stream_options":{"type":"object","description":"Options for streaming response.","properties":{"include_usage":{"type":"boolean","description":"Request usage in the final usage chunk before the [DONE] sentinel. It does not add a second [DONE] event."}}},"tools":{"type":"array","description":"Function tools and provider-specific tools. Native paths preserve provider tool types; cross-format conversion supports only representable tool semantics.","items":{"$ref":"#/components/schemas/Tool"}},"tool_choice":{"oneOf":[{"type":"string","enum":["none","auto","required","any"],"description":"none declines tools; auto lets the model choose; required/any requests tool use. Provider support varies."},{"$ref":"#/components/schemas/ToolChoiceTypedMode"},{"$ref":"#/components/schemas/ToolChoiceSpecific"}],"description":"Tool selection. required/any request tool use; support for none and forced functions depends on provider and conversion path."},"tags":{"type":"array","items":{"type":"string"},"description":"Optional tags to categorize and label the request. Useful for filtering and grouping requests in analytics and logs. Can also be sent via the `x-edgee-tags` header as a comma-separated string."},"enable_debug":{"type":"boolean","description":"Enable request debug capture. Gateway-only field; does not add debug information to the response."},"max_completion_tokens":{"type":"integer","minimum":1,"description":"Alias of max_tokens. Use one spelling, not both."},"temperature":{"type":"number","description":"Sampling temperature; supported range depends on the destination model."},"top_p":{"type":"number","description":"Nucleus sampling probability; support depends on the destination model."},"reasoning_effort":{"type":"string","description":"Requested effort; common values include none, minimal, low, medium, high, xhigh and max. The supported vocabulary is model-specific. Cross-format routing maps supported values and can omit unrepresentable levels.","example":"high"},"reasoning_summary":{"type":"string","description":"Gateway reasoning-display override. Common values: auto, concise, detailed. For Chat Completions, positive reasoning effort normally exposes reasoning_content unless overridden."},"reasoning_enabled":{"type":"boolean","description":"Gateway override for reasoning intent on cross-format requests; false explicitly declines reasoning."},"reasoning_budget_tokens":{"type":"integer","minimum":0,"description":"Legacy Anthropic thinking budget, used only when translating to a compatible destination."}},"additionalProperties":true,"description":"Native dispatch preserves provider-specific JSON fields when the destination supports the caller’s format. Cross-format routing converts through the Gateway’s shared representation: provider-only features are not guaranteed to survive. Support for reasoning, images, built-in tools and structured outputs depends on the destination model."},"ToolChoiceTypedMode":{"type":"object","description":"Object form of a tool selection mode.","required":["type"],"properties":{"type":{"type":"string","enum":["auto","none","any","required"]}}},"Message":{"type":"object","required":["role"],"properties":{"role":{"type":"string","enum":["system","user","assistant","tool","developer"],"description":"The role of the message author. Required properties vary by role:\n- `system`, `user`, `developer`: requires `content`\n- `assistant`: `content` is optional (can be empty if `tool_calls` is present)\n- `tool`: requires `content` and `tool_call_id`"},"content":{"description":"Text or multimodal parts. Required for system, developer, user and tool turns; nullable/optional for assistant tool calls.","oneOf":[{"type":"string","nullable":true},{"type":"array","items":{"$ref":"#/components/schemas/ChatContentPart"}}]},"name":{"type":"string","description":"An optional name for the participant. Provides the model information to differentiate between participants of the same role. Used for `system`, `user`, `assistant`, and `developer` roles."},"tool_call_id":{"type":"string","description":"Required for tool messages; matches an assistant tool call ID."},"refusal":{"type":"string","description":"The refusal message from the model, if any. Used for `assistant` role only."},"tool_calls":{"type":"array","description":"The tool calls made by the assistant. Used for `assistant` role only.","items":{"$ref":"#/components/schemas/ToolCall"}},"cache_control":{"type":"object","description":"Anthropic cache breakpoint metadata when translating supported system, user or tool content. Cache behavior depends on the destination provider.","additionalProperties":true,"example":{"type":"ephemeral"}},"reasoning_content":{"type":"string","description":"Reasoning text from the provider, when exposed by the requested reasoning policy."}}},"Tool":{"type":"object","required":["type","function"],"properties":{"type":{"type":"string","enum":["function"],"description":"Discriminator for the standard function tool variant."},"function":{"$ref":"#/components/schemas/FunctionDefinition"}},"description":"Standard function tool. Provider-native tool objects are also accepted by native Chat-compatible dispatch."},"FunctionDefinition":{"type":"object","required":["name"],"properties":{"name":{"type":"string","description":"The name of the function to be called. Must be a-z, A-Z, 0-9, or contain underscores and dashes, with a maximum length of 64."},"description":{"type":"string","description":"A description of what the function does, used by the model to choose when and how to call the function."},"parameters":{"type":"object","description":"The parameters the functions accepts, described as a JSON Schema object. See the guide for examples, and the JSON Schema reference for documentation about the format.","additionalProperties":true}}},"ToolChoiceSpecific":{"type":"object","required":["type","function"],"properties":{"type":{"type":"string","enum":["function"],"description":"The type of the tool."},"function":{"$ref":"#/components/schemas/ToolChoiceFunction"}}},"ToolChoiceFunction":{"type":"object","required":["name"],"properties":{"name":{"type":"string","description":"The name of the function to call."}}},"ToolCall":{"type":"object","required":["id","type","function"],"properties":{"id":{"type":"string","description":"The ID of the tool call."},"type":{"type":"string","enum":["function"],"description":"The type of the tool call."},"function":{"$ref":"#/components/schemas/FunctionCall"},"extra_content":{"type":"object","properties":{"google":{"type":"object","properties":{"thought_signature":{"type":"string","description":"Opaque provider thought signature to preserve when replaying tool calls."}}}}}}},"FunctionCall":{"type":"object","required":["name","arguments"],"properties":{"name":{"type":"string","description":"The name of the function to call."},"arguments":{"type":"string","description":"The arguments to call the function with, as JSON."}}},"ChatCompletionResponse":{"type":"object","required":["id","object","created","model","choices","usage"],"properties":{"id":{"type":"string","description":"A unique identifier for the chat completion.","example":"chatcmpl-123"},"object":{"type":"string","enum":["chat.completion"],"description":"The object type, which is always `chat.completion`."},"created":{"type":"integer","description":"The Unix timestamp (in seconds) of when the chat completion was created.","example":1677652288},"model":{"type":"string","description":"The model used for the chat completion.","example":"openai/gpt-5.2"},"choices":{"type":"array","description":"Model completion choices. Number of choices depends on the provider and dispatch path.","items":{"$ref":"#/components/schemas/ChatCompletionChoice"}},"usage":{"$ref":"#/components/schemas/Usage"},"compression":{"$ref":"#/components/schemas/CompressionInfo"}}},"ChatCompletionChoice":{"type":"object","required":["index","message"],"properties":{"index":{"type":"integer","description":"The index of the choice in the list of choices.","minimum":0},"message":{"$ref":"#/components/schemas/Message"},"finish_reason":{"type":"string","enum":["stop","length","content_filter","tool_calls"],"description":"The reason the model stopped generating tokens. This will be `stop` if the model hit a natural stop point or a provided stop sequence, `length` if the maximum number of tokens specified in the request was reached, `content_filter` if content was omitted due to a flag from our content filters, or `tool_calls` if the model called a tool.","nullable":true}}},"Usage":{"type":"object","description":"Usage on shared converted Chat responses. Native responses preserve the provider’s usage representation, which may use prompt_tokens_details/completion_tokens_details rather than these Gateway detail fields.","required":["prompt_tokens","completion_tokens","total_tokens"],"properties":{"prompt_tokens":{"type":"integer","description":"Number of tokens in the prompt.","minimum":0},"completion_tokens":{"type":"integer","description":"Number of tokens in the generated completion.","minimum":0},"total_tokens":{"type":"integer","description":"Total number of tokens used in the request (prompt + completion).","minimum":0},"input_tokens_details":{"$ref":"#/components/schemas/InputTokenDetails"},"output_tokens_details":{"$ref":"#/components/schemas/OutputTokenDetails"}},"additionalProperties":true},"InputTokenDetails":{"type":"object","description":"Additional details about input tokens.","properties":{"cached_tokens":{"type":"integer","description":"Number of cached tokens read from the prompt cache.","minimum":0},"cache_creation_tokens":{"type":"integer","description":"Number of tokens written to the prompt cache (Anthropic-style cache creation).","minimum":0},"cache_creation_5m_tokens":{"type":"integer","minimum":0,"description":"Cache-write tokens for this TTL, omitted when zero."},"cache_creation_1h_tokens":{"type":"integer","minimum":0,"description":"Cache-write tokens for this TTL, omitted when zero."}}},"OutputTokenDetails":{"type":"object","description":"Additional details about output tokens.","properties":{"reasoning_tokens":{"type":"integer","description":"Number of reasoning tokens in the output.","minimum":0}}},"ModelsResponse":{"type":"object","required":["object","data"],"properties":{"object":{"type":"string","enum":["list"],"description":"The object type, which is always `list`."},"data":{"type":"array","description":"The list of models.","items":{"$ref":"#/components/schemas/Model"}}}},"Model":{"type":"object","required":["id","object","created","owned_by"],"properties":{"id":{"type":"string","description":"The model identifier, which can be referenced in the API. Format: `{author_id}/{model_id}`.","example":"openai/gpt-5.2"},"object":{"type":"string","enum":["model"],"description":"The object type, which is always `model`."},"created":{"type":"integer","description":"The Unix timestamp (in seconds) when the model was created.","example":1677610602},"owned_by":{"type":"string","description":"The organization that owns the model.","example":"openai"}}},"ChatCompletionChunk":{"type":"object","required":["id","object","created","model","choices"],"description":"A streaming chunk in the chat completion response. Used when `stream: true` in the request.","properties":{"id":{"type":"string","description":"A unique identifier for the chat completion chunk.","example":"chatcmpl-123"},"object":{"type":"string","enum":["chat.completion.chunk"],"description":"The object type, which is always `chat.completion.chunk` for streaming responses."},"created":{"type":"integer","description":"The Unix timestamp (in seconds) of when the chat completion was created.","example":1677652288},"model":{"type":"string","description":"The model used for the chat completion.","example":"openai/gpt-5.2"},"choices":{"type":"array","description":"A list of chat completion choices for this chunk.","items":{"$ref":"#/components/schemas/ChatCompletionChunkChoice"}},"usage":{"$ref":"#/components/schemas/Usage"}}},"ChatCompletionChunkChoice":{"type":"object","required":["index","delta"],"description":"A choice in a streaming chat completion chunk.","properties":{"index":{"type":"integer","description":"The index of the choice in the list of choices.","minimum":0},"delta":{"$ref":"#/components/schemas/Delta","description":"A delta representing the change in the message content. The first chunk typically contains `role`, subsequent chunks contain `content`."},"finish_reason":{"type":"string","enum":["stop","length","content_filter","tool_calls"],"description":"The reason the model stopped generating tokens. This will be `null` for all chunks except the final one. This will be `stop` if the model hit a natural stop point or a provided stop sequence, `length` if the maximum number of tokens specified in the request was reached, `content_filter` if content was omitted due to a flag from our content filters, or `tool_calls` if the model called a tool.","nullable":true}}},"Delta":{"type":"object","description":"Represents a change in message content. The first chunk typically contains `role`, subsequent chunks contain `content`.","properties":{"role":{"type":"string","description":"The role of the message author. Typically present only in the first chunk.","example":"assistant"},"content":{"type":"string","description":"The content of the message delta. Present in content chunks.","example":"Hello"},"reasoning_content":{"type":"string"},"tool_calls":{"type":"array","items":{"type":"object","properties":{"index":{"type":"integer","minimum":0},"id":{"type":"string"},"type":{"type":"string"},"function":{"type":"object","properties":{"name":{"type":"string"},"arguments":{"type":"string"}}}}}}}},"ErrorResponse":{"type":"object","required":["error"],"description":"OpenAI-shaped Gateway error. Native upstream errors can retain provider-specific details.","properties":{"error":{"type":"object","required":["message","type"],"properties":{"message":{"type":"string","description":"A human-readable error message."},"type":{"type":"string","description":"High-level error category."},"code":{"type":"string","nullable":true,"description":"Machine-readable error code, for example unauthorized, invalid_json, bad_model_id, model_disabled, provider_disabled, usage_limit_exceeded, usage_unavailable, provider_error or internal_error. Not an exhaustive enum.","example":"bad_model_id"},"param":{"type":"string","nullable":true,"description":"Name of the request parameter that caused the error, when applicable."}}}}},"CreateMessageRequest":{"type":"object","required":["model","max_tokens","messages"],"properties":{"model":{"type":"string","description":"Model identifier from GET /v1/models, in author/model format, or a configured global alias. Append :provider to request a specific provider route. Available models and providers depend on your key and organization policies.","example":"anthropic/claude-sonnet-4-6"},"max_tokens":{"type":"integer","description":"Maximum number of tokens to generate","minimum":1,"example":1024},"messages":{"type":"array","description":"Array of message objects","items":{"$ref":"#/components/schemas/MessageParam"},"minItems":1},"system":{"oneOf":[{"type":"string","description":"System prompt as a string"},{"type":"array","description":"System prompt as content blocks","items":{"$ref":"#/components/schemas/ContentBlock"}}]},"stream":{"type":"boolean","description":"Enable streaming responses","default":false},"tools":{"type":"array","description":"Tool definitions","items":{"$ref":"#/components/schemas/AnthropicTool"}},"tool_choice":{"$ref":"#/components/schemas/ToolChoice"},"enable_debug":{"type":"boolean","description":"Enable request debug capture. Gateway-only field; does not add debug information to the response."},"tags":{"type":"array","items":{"type":"string"},"description":"Optional tags to categorize and label the request. Useful for filtering and grouping requests in analytics and logs. Can also be sent via the `x-edgee-tags` header as a comma-separated string."},"temperature":{"type":"number"},"top_p":{"type":"number"},"thinking":{"type":"object","required":["type"],"properties":{"type":{"type":"string","description":"adaptive, disabled, or legacy enabled; supported modes depend on the model."},"display":{"type":"string","description":"summarized or omitted. Default display depends on the model."},"budget_tokens":{"type":"integer","minimum":0,"description":"Required for legacy enabled thinking; adaptive models do not use this budget."}}},"output_config":{"type":"object","description":"Only effort is portable through the shared representation. Other fields such as format and task_budget are provider-native and are not preserved by every cross-format path.","properties":{"effort":{"type":"string","description":"Requested effort; common values include none, minimal, low, medium, high, xhigh and max. The supported vocabulary is model-specific. Cross-format routing maps supported values and can omit unrepresentable levels.","example":"high"}}}},"additionalProperties":true,"description":"Native dispatch preserves provider-specific JSON fields when the destination supports the caller’s format. Cross-format routing converts through the Gateway’s shared representation: provider-only features are not guaranteed to survive. Support for reasoning, images, built-in tools and structured outputs depends on the destination model."},"MessageParam":{"type":"object","required":["role","content"],"properties":{"role":{"type":"string","enum":["user","assistant"],"description":"The role of the message"},"content":{"oneOf":[{"type":"string","description":"Simple text content"},{"type":"array","description":"Array of content blocks","items":{"$ref":"#/components/schemas/ContentBlock"}}]}}},"ContentBlock":{"type":"object","required":["type"],"discriminator":{"propertyName":"type"},"oneOf":[{"type":"object","required":["type","text"],"properties":{"type":{"type":"string","enum":["text"]},"text":{"type":"string"},"cache_control":{"$ref":"#/components/schemas/CacheControl"}}},{"type":"object","required":["type","id","name","input"],"properties":{"type":{"type":"string","enum":["tool_use"]},"id":{"type":"string"},"name":{"type":"string"},"input":{"type":"object"}}},{"type":"object","required":["type","tool_use_id","content"],"properties":{"type":{"type":"string","enum":["tool_result"]},"tool_use_id":{"type":"string"},"content":{"oneOf":[{"type":"string"},{"type":"array","items":{"oneOf":[{"type":"object","required":["type","text"],"properties":{"type":{"type":"string","enum":["text"]},"text":{"type":"string"},"cache_control":{"$ref":"#/components/schemas/CacheControl"}}},{"type":"object","required":["type","source"],"properties":{"type":{"type":"string","enum":["image"]},"source":{"oneOf":[{"type":"object","required":["type","url"],"properties":{"type":{"type":"string","enum":["url"]},"url":{"type":"string"}}},{"type":"object","required":["type","media_type","data"],"properties":{"type":{"type":"string","enum":["base64"]},"media_type":{"type":"string"},"data":{"type":"string"}}}]}}}]}}]},"is_error":{"type":"boolean","default":false},"cache_control":{"$ref":"#/components/schemas/CacheControl"}}},{"type":"object","required":["type","source"],"properties":{"type":{"type":"string","enum":["image"]},"source":{"oneOf":[{"type":"object","required":["type","url"],"properties":{"type":{"type":"string","enum":["url"]},"url":{"type":"string"}}},{"type":"object","required":["type","media_type","data"],"properties":{"type":{"type":"string","enum":["base64"]},"media_type":{"type":"string"},"data":{"type":"string"}}}]},"cache_control":{"$ref":"#/components/schemas/CacheControl"}}},{"type":"object","required":["type","thinking"],"properties":{"type":{"type":"string","enum":["thinking"]},"thinking":{"type":"string"},"signature":{"type":"string","description":"Preserve provider-issued signatures when replaying history. Unsigned thinking blocks are stripped before upstream Anthropic calls."}}},{"type":"object","required":["type","data"],"properties":{"type":{"type":"string","enum":["redacted_thinking"]},"data":{"type":"string"}}}],"description":"Common Anthropic content blocks. Native dispatch can carry additional provider-specific blocks; conversion paths only preserve supported content. Cross-provider reasoning is emitted as text, never as forged signed thinking blocks."},"AnthropicTool":{"type":"object","required":["name"],"properties":{"name":{"type":"string","description":"The name of the tool"},"description":{"type":"string","description":"Description of what the tool does"},"input_schema":{"type":"object","description":"JSON Schema describing the tool's input parameters","default":{"type":"object","properties":{}}},"cache_control":{"$ref":"#/components/schemas/CacheControl"}}},"ToolChoice":{"type":"object","required":["type"],"discriminator":{"propertyName":"type"},"oneOf":[{"type":"object","required":["type"],"properties":{"type":{"type":"string","enum":["auto"],"description":"Model decides whether to use tools"}}},{"type":"object","required":["type"],"properties":{"type":{"type":"string","enum":["any"],"description":"Model must use one of the provided tools"}}},{"type":"object","required":["type","name"],"properties":{"type":{"type":"string","enum":["tool"]},"name":{"type":"string","description":"Name of the specific tool to use"}}}]},"CreateMessageResponse":{"type":"object","required":["id","type","role","model","content","usage"],"properties":{"id":{"type":"string","description":"Unique identifier for this message"},"model":{"type":"string","description":"The model that generated the response"},"content":{"type":"array","description":"Array of content blocks","items":{"$ref":"#/components/schemas/ContentBlock"}},"usage":{"$ref":"#/components/schemas/AnthropicUsage"},"stop_reason":{"type":"string","description":"Stop reason. Native provider responses may include additional reasons.","nullable":true},"type":{"type":"string","enum":["message"]},"role":{"type":"string","enum":["assistant"]},"stop_sequence":{"type":"string","nullable":true}}},"CompressionInfo":{"type":"object","description":"Token compression metrics. Present in the response when token compression was applied to the request. The `usage.prompt_tokens` field reflects the compressed token count actually billed by the provider.","required":["saved_tokens","cost_savings","reduction","time_ms"],"properties":{"saved_tokens":{"type":"integer","description":"Estimated saved input tokens, calibrated to the provider token count from the ratio of local counts before and after compression.","minimum":0,"example":450},"cost_savings":{"type":"integer","description":"Estimated savings in nanodollars. Divide by 1,000,000,000 to convert to USD: 27,000,000 nanodollars = $0.027.","minimum":0,"example":27000000},"reduction":{"type":"number","description":"Percentage reduction: 100 × estimated saved tokens / estimated pre-compression tokens, calibrated to provider usage.","minimum":0,"maximum":100,"example":48.99884991374353},"time_ms":{"type":"integer","description":"Time taken to perform compression, in milliseconds.","minimum":0,"example":12}}},"AnthropicUsage":{"type":"object","required":["input_tokens","output_tokens"],"properties":{"input_tokens":{"type":"integer","description":"Number of input tokens.","minimum":0},"output_tokens":{"type":"integer","description":"Number of output tokens.","minimum":0},"cache_read_input_tokens":{"type":"integer","description":"Number of input tokens read from the prompt cache.","minimum":0},"cache_creation_input_tokens":{"type":"integer","description":"Number of input tokens written to the prompt cache.","minimum":0},"cache_creation":{"type":"object","properties":{"ephemeral_5m_input_tokens":{"type":"integer","minimum":0},"ephemeral_1h_input_tokens":{"type":"integer","minimum":0}}},"output_tokens_details":{"type":"object","properties":{"thinking_tokens":{"type":"integer","minimum":0}}}}},"ResponsesRequest":{"type":"object","required":["model","input"],"properties":{"model":{"type":"string","description":"Model identifier from GET /v1/models, in author/model format, or a configured global alias. Append :provider to request a specific provider route. Available models and providers depend on your key and organization policies.","example":"openai/gpt-5.2"},"input":{"description":"The input to the model. Either a plain string (treated as a single user message) or a flat array of typed input items (messages, function calls, function call outputs).","oneOf":[{"type":"string"},{"type":"array","items":{"$ref":"#/components/schemas/ResponsesInputItem"}}]},"instructions":{"type":"string","description":"System-level instruction prepended to the conversation. An alternative to including a `system` role message in the `input` array."},"stream":{"type":"boolean","description":"If set, the response is streamed as Server-Sent Events (SSE).","default":false},"max_output_tokens":{"type":"integer","description":"Maximum number of tokens to generate.","minimum":1},"tools":{"type":"array","description":"Tools available to the model. Uses the Responses API flat format (no nested `function` key).","items":{"$ref":"#/components/schemas/ResponsesTool"}},"tool_choice":{"description":"Tool selection on native Responses paths. Cross-format conversion supports modes and the shared nested function form; flat forced-function names are not preserved on every conversion path.","oneOf":[{"type":"string","enum":["auto","none","required","any"],"description":"Bare-string mode. `auto` lets the model decide; `none` forbids tool calls."},{"$ref":"#/components/schemas/ToolChoiceTypedMode"},{"$ref":"#/components/schemas/ResponsesToolChoiceFunction"}]},"temperature":{"type":"number","description":"Sampling temperature. Range and support depend on the destination model."},"top_p":{"type":"number","description":"Nucleus sampling probability. Alternative to temperature."},"tags":{"type":"array","items":{"type":"string"},"description":"Optional tags to categorize and label the request. Useful for filtering and grouping requests in analytics and logs. Can also be sent via the `x-edgee-tags` header as a comma-separated string."},"enable_debug":{"type":"boolean","description":"Enable request debug capture. Gateway-only field; does not add debug information to the response."},"reasoning":{"type":"object","properties":{"effort":{"type":"string","description":"Requested effort; common values include none, minimal, low, medium, high, xhigh and max. The supported vocabulary is model-specific. Cross-format routing maps supported values and can omit unrepresentable levels.","example":"high"},"summary":{"type":"string","description":"Requested reasoning display: auto, concise or detailed. May be omitted on destinations that cannot represent it."}}}},"additionalProperties":true,"description":"Native dispatch preserves provider-specific JSON fields when the destination supports the caller’s format. Cross-format routing converts through the Gateway’s shared representation: provider-only features are not guaranteed to survive. Support for reasoning, images, built-in tools and structured outputs depends on the destination model."},"ResponsesInputItem":{"description":"Typed input items understood by the shared converter. additional_tools supplies inline tool declarations. Native Responses accepts further item types. On cross-format conversion unrecognized items (including replayed reasoning/computer items) are skipped; do not assume provider-only conversation state is portable.","oneOf":[{"type":"object","title":"Message","required":["role"],"properties":{"role":{"type":"string","enum":["user","assistant","system","developer"],"description":"The role of the message author."},"content":{"description":"Message content. Either a plain string or an array of typed content parts.","oneOf":[{"type":"string","nullable":true},{"type":"array","items":{"$ref":"#/components/schemas/ResponsesContentPart"}}]}}},{"type":"object","title":"Function call","required":["type","call_id","name"],"properties":{"type":{"type":"string","enum":["function_call"]},"call_id":{"type":"string","description":"Identifier linking this call to its `function_call_output`.","example":"call_abc123"},"name":{"type":"string","description":"Name of the function the assistant is calling.","example":"get_weather"},"arguments":{"type":"string","description":"Arguments passed to the function, encoded as a JSON string.","example":"{\"location\": \"Paris\"}","default":"{}"}}},{"type":"object","title":"Function call output","required":["type","call_id","output"],"properties":{"type":{"type":"string","enum":["function_call_output"]},"call_id":{"type":"string","description":"Identifier matching the `function_call` item this is responding to.","example":"call_abc123"},"output":{"oneOf":[{"type":"string"},{"type":"array","items":{"$ref":"#/components/schemas/ResponsesContentPart"}}],"description":"Tool result as text or multimodal content parts."}}},{"type":"object","required":["type"],"properties":{"type":{"type":"string","enum":["additional_tools"]},"role":{"type":"string","enum":["developer"]},"tools":{"type":"array","items":{"$ref":"#/components/schemas/ResponsesTool"}}}}]},"ResponsesContentPart":{"anyOf":[{"type":"object","required":["type","text"],"properties":{"type":{"type":"string","enum":["input_text"]},"text":{"type":"string"}}},{"type":"object","required":["type","text"],"properties":{"type":{"type":"string","enum":["output_text"]},"text":{"type":"string"}}},{"type":"object","required":["type","text"],"properties":{"type":{"type":"string","enum":["text"]},"text":{"type":"string"}}},{"type":"object","required":["type","image_url"],"properties":{"type":{"type":"string","enum":["input_image"]},"image_url":{"type":"string","description":"Image URL or data URL."},"detail":{"type":"string"}}},{"type":"object","required":["type","image_url"],"properties":{"type":{"type":"string","enum":["image_url"]},"image_url":{"type":"object","required":["url"],"properties":{"url":{"type":"string"},"detail":{"type":"string"}}}}},{"type":"object","additionalProperties":true,"description":"Provider-native parts are not portable across all formats."}]},"ResponsesTool":{"oneOf":[{"type":"object","required":["type","name"],"properties":{"type":{"type":"string","enum":["function"]},"name":{"type":"string"},"description":{"type":"string"},"parameters":{"type":"object","additionalProperties":true}}},{"type":"object","required":["type","name"],"properties":{"type":{"type":"string","enum":["custom"]},"name":{"type":"string"},"description":{"type":"string"},"format":{"type":"object","description":"Native grammar constraint. Cross-format conversion uses an input string parameter and does not enforce the grammar."}}},{"type":"object","required":["type","name"],"properties":{"type":{"type":"string","enum":["namespace"]},"name":{"type":"string"},"tools":{"type":"array","items":{"$ref":"#/components/schemas/ResponsesTool"}}}}],"description":"Function, freeform and namespaced tools are understood by the converter. Namespaces are flattened for dispatch and restored on returned function calls. Other provider-native types can pass through natively but are dropped during typed conversion."},"ResponsesToolChoiceFunction":{"type":"object","description":"Native Responses forced-function form. The shared typed tool-choice representation uses {type:function,function:{name:...}}; the flat name is not preserved on every cross-format path.","required":["type","name"],"properties":{"type":{"type":"string","enum":["function"]},"name":{"type":"string","description":"Name of the function to call."}}},"ResponsesResponse":{"type":"object","required":["id","object","status","created_at","model","output","usage"],"properties":{"id":{"type":"string","description":"Unique identifier for the response, prefixed with `resp_`.","example":"resp_abc123"},"object":{"type":"string","enum":["response"]},"status":{"type":"string","description":"Shared converted responses use completed or in_progress; native responses can also report provider statuses such as incomplete or failed."},"created_at":{"type":"number","description":"Unix timestamp (as a float) of when the response was created.","example":1677652288},"model":{"type":"string","description":"The model used to generate the response.","example":"openai/gpt-5.2"},"output":{"type":"array","description":"Array of output items produced by the model.","items":{"$ref":"#/components/schemas/ResponsesOutputItem"}},"usage":{"$ref":"#/components/schemas/ResponsesUsage"}}},"ResponsesOutputItem":{"oneOf":[{"type":"object","required":["type","id","status","role","content"],"properties":{"type":{"type":"string","enum":["message"]},"id":{"type":"string"},"status":{"type":"string"},"role":{"type":"string","enum":["assistant"]},"content":{"type":"array","items":{"$ref":"#/components/schemas/ResponsesOutputContent"}}}},{"type":"object","required":["type","id","call_id","name","arguments","status"],"properties":{"type":{"type":"string","enum":["function_call"]},"id":{"type":"string"},"call_id":{"type":"string"},"name":{"type":"string"},"arguments":{"type":"string"},"status":{"type":"string"},"namespace":{"type":"string"}}},{"type":"object","required":["type","id","call_id","name","input","status"],"properties":{"type":{"type":"string","enum":["custom_tool_call"]},"id":{"type":"string"},"call_id":{"type":"string"},"name":{"type":"string"},"input":{"type":"string"},"status":{"type":"string"}}},{"type":"object","required":["type","id","summary"],"properties":{"type":{"type":"string","enum":["reasoning"]},"id":{"type":"string"},"summary":{"type":"array","items":{"type":"object","required":["type","text"],"properties":{"type":{"type":"string","enum":["summary_text"]},"text":{"type":"string"}}}}}}],"description":"Each message, function call, custom tool call and reasoning item is a separate top-level output item. Native providers may return additional item types."},"ResponsesOutputContent":{"type":"object","required":["type","text"],"properties":{"type":{"type":"string","enum":["output_text"]},"text":{"type":"string"}}},"ResponsesUsage":{"type":"object","description":"Token usage statistics for the response.","required":["input_tokens","output_tokens","total_tokens"],"properties":{"input_tokens":{"type":"integer","description":"Tokens in the input.","minimum":0},"output_tokens":{"type":"integer","description":"Tokens in the output.","minimum":0},"total_tokens":{"type":"integer","description":"Total tokens used.","minimum":0},"cached_input_tokens":{"type":"integer","minimum":0,"description":"Cached input tokens on converted responses; omitted when zero. Native responses preserve provider-specific usage fields."}},"additionalProperties":true},"CompressRequest":{"description":"An LLM request payload to compress. Accepts any of three wire formats — the format is auto-detected from the request body: a top-level `\"system\"` key indicates Anthropic Messages format; a top-level `\"input\"` key indicates OpenAI Responses API format; otherwise, OpenAI Chat Completions format is assumed. Include a top-level system field (even an empty string) to select Anthropic parsing when compressing a Messages payload. Format conversion can normalize message/tool content; other top-level fields are retained.","anyOf":[{"$ref":"#/components/schemas/ChatCompletionRequest"},{"$ref":"#/components/schemas/CreateMessageRequest"},{"$ref":"#/components/schemas/ResponsesRequest"}]},"CompressMetadata":{"type":"object","description":"Token compression metrics appended when `/v1/compress` runs tool-result trimming.","required":["technique","applied_strategies","compression_rate","uncompressed_input_tokens","compressed_input_tokens","compression_time_ms"],"properties":{"technique":{"type":"string","enum":["tool"],"description":"The compression technique applied. Always `tool` (tool-result trimming).","example":"tool"},"applied_strategies":{"type":"array","description":"Names of the compression strategies that were applied to the request.","items":{"type":"string"},"example":["tool_result_trimming"]},"compression_rate":{"type":"number","description":"Remaining token ratio: compressed_input_tokens / uncompressed_input_tokens. 0.81 means 19% saved; 1 means no reduction (also used for empty input).","minimum":0,"maximum":1,"example":0.81},"uncompressed_input_tokens":{"type":"integer","description":"Token count of the original, uncompressed input.","minimum":0,"example":1000},"compressed_input_tokens":{"type":"integer","description":"Local token count after compression. No provider is called and this count is not a provider billing measurement.","minimum":0,"example":810},"compression_time_ms":{"type":"integer","description":"Wall-clock time taken to perform compression, in milliseconds.","minimum":0,"example":12},"error":{"type":"string","description":"Human-readable error message if compression partially failed. Present only when an error occurred during compression."},"tool_stats":{"type":"object","description":"Per-tool compression statistics. Present only when `tool_result_trimming` was applied.","additionalProperties":{"type":"object","required":["before","after"],"properties":{"before":{"type":"integer","minimum":0},"after":{"type":"integer","minimum":0}}}}}},"CompressResponse":{"type":"object","description":"The original request payload with compressed content fields replaced. All fields not touched by compression (`model`, `temperature`, `top_p`, `stop_sequences`, etc.) pass through unchanged. A `compression` object is appended when tool-result trimming runs. When trimming is disabled by the request header, the original payload is returned without added compression metadata.","properties":{"compression":{"$ref":"#/components/schemas/CompressMetadata"}},"additionalProperties":true},"CacheControl":{"type":"object","required":["type"],"properties":{"type":{"type":"string","enum":["ephemeral"]},"ttl":{"type":"string","enum":["5m","1h"]}}},"ChatContentPart":{"anyOf":[{"type":"object","required":["type","text"],"properties":{"type":{"type":"string","enum":["text"]},"text":{"type":"string"},"cache_control":{"$ref":"#/components/schemas/CacheControl"}}},{"type":"object","required":["type","image_url"],"properties":{"type":{"type":"string","enum":["image_url"]},"image_url":{"type":"object","required":["url"],"properties":{"url":{"type":"string","description":"HTTPS URL or data URL."},"detail":{"type":"string"}}}}},{"type":"object","required":["type","image_url"],"properties":{"type":{"type":"string","enum":["input_image"]},"image_url":{"type":"string"},"detail":{"type":"string"}}},{"type":"object","required":["type","source"],"properties":{"type":{"type":"string","enum":["image"]},"source":{"oneOf":[{"type":"object","required":["type","url"],"properties":{"type":{"type":"string","enum":["url"]},"url":{"type":"string"}}},{"type":"object","required":["type","media_type","data"],"properties":{"type":{"type":"string","enum":["base64"]},"media_type":{"type":"string"},"data":{"type":"string"}}}]}}},{"type":"object","additionalProperties":true,"description":"Provider-native content such as audio; not guaranteed to survive cross-format conversion."}]},"AnthropicErrorResponse":{"type":"object","required":["type","error"],"properties":{"type":{"type":"string","enum":["error"]},"error":{"type":"object","required":["type","message"],"properties":{"type":{"type":"string"},"message":{"type":"string"}}},"request_id":{"type":"string"}}},"MessageCountTokensRequest":{"type":"object","description":"Anthropic provider token counting. No max_tokens is required. Requires an available Anthropic provider route for the selected model; this is not the local estimator.","required":["model","messages"],"properties":{"model":{"type":"string","description":"Model identifier from GET /v1/models, in author/model format, or a configured global alias. Append :provider to request a specific provider route. Available models and providers depend on your key and organization policies.","example":"anthropic/claude-sonnet-4-6"},"messages":{"type":"array","description":"Array of message objects","items":{"$ref":"#/components/schemas/MessageParam"},"minItems":1},"system":{"oneOf":[{"type":"string","description":"System prompt as a string"},{"type":"array","description":"System prompt as content blocks","items":{"$ref":"#/components/schemas/ContentBlock"}}]},"tools":{"type":"array","description":"Tool definitions","items":{"$ref":"#/components/schemas/AnthropicTool"}},"tool_choice":{"$ref":"#/components/schemas/ToolChoice"}}}},"securitySchemes":{"bearerAuth":{"type":"http","scheme":"bearer","bearerFormat":"JWT","description":"Bearer authentication header of the form `Bearer <token>`, where `<token>` is your API key. More info [here](/docs/api-reference/authentication)"},"apiKeyAuth":{"type":"apiKey","in":"header","name":"x-api-key","description":"Anthropic-style API key authentication using the x-api-key header"}}},"tags":[{"name":"Chat","description":"Chat completion endpoints (OpenAI format)"},{"name":"Messages","description":"Messages endpoints (Anthropic format)"},{"name":"Responses","description":"Responses endpoints (OpenAI Responses API format)"},{"name":"Models","description":"Model management endpoints"},{"name":"Tokens","description":"Token estimation endpoints"},{"name":"Compress","description":"Standalone token compression endpoint"}]}