{"openapi":"3.1.0","info":{"title":"bInference API","version":"1.0.0","summary":"Every AI model on one key, paid by your agent's trading fees.","description":"OpenAI and Anthropic compatible. Model bodies are passed on as sent, so every field of each format works beyond the ones listed here.","contact":{"url":"https://binference.io"}},"externalDocs":{"url":"https://docs.binference.io"},"servers":[{"url":"https://binference.io/api/v1"}],"components":{"securitySchemes":{"bearer":{"type":"http","scheme":"bearer","description":"Your binf_ key."},"apiKey":{"type":"apiKey","in":"header","name":"x-api-key","description":"Your binf_ key."}},"schemas":{"OpenAIError":{"type":"object","required":["error"],"properties":{"error":{"type":"object","required":["message","type","code"],"properties":{"message":{"type":"string"},"type":{"type":"string"},"code":{"type":"string","enum":["invalid_api_key","agent_inactive","agent_suspended","invalid_request","unknown_model","model_not_served","tool_not_served","request_too_large","request_over_capacity","not_found","insufficient_balance","key_limit","rate_limited","too_many_running_calls","daily_share_used","model_busy","gateway_busy","gateway_capacity","gateway_unavailable","catalog_unavailable","service_unavailable","gateway_error","upstream_timeout","upstream_unreachable","provider_error","provider_unavailable","no_provider","provider_refused","unprocessable","timeout","upstream_error"]},"param":{"type":["string","null"]}}}}},"AnthropicError":{"type":"object","required":["type","error"],"properties":{"type":{"const":"error"},"error":{"type":"object","required":["type","message"],"properties":{"type":{"type":"string"},"message":{"type":"string"},"code":{"type":"string","enum":["invalid_api_key","agent_inactive","agent_suspended","invalid_request","unknown_model","model_not_served","tool_not_served","request_too_large","request_over_capacity","not_found","insufficient_balance","key_limit","rate_limited","too_many_running_calls","daily_share_used","model_busy","gateway_busy","gateway_capacity","gateway_unavailable","catalog_unavailable","service_unavailable","gateway_error","upstream_timeout","upstream_unreachable","provider_error","provider_unavailable","no_provider","provider_refused","unprocessable","timeout","upstream_error"]}}}}}}},"paths":{"/chat/completions":{"post":{"operationId":"chat","summary":"Chat Completions","description":"The OpenAI chat format. Works with the OpenAI SDKs, the Vercel AI SDK, LangChain, OpenClaw and Hermes.","externalDocs":{"url":"https://docs.binference.io/api/chat-completions"},"security":[{"bearer":[]},{"apiKey":[]}],"requestBody":{"required":true,"content":{"application/json":{"schema":{"type":"object","additionalProperties":true,"properties":{"model":{"type":"string","description":"A model id from GET /models, such as anthropic/claude-sonnet-5.5. Add :online for web search, or :nitro, :floor or :exacto to steer which provider runs it."},"messages":{"type":"array","description":"The conversation so far, oldest first.","items":{"type":"object","properties":{"role":{"type":"string","enum":["system","user","assistant","tool"],"description":"Who wrote the message."},"content":{"anyOf":[{"type":"string"},{"type":"array","items":{"type":"object"}}],"description":"Text, or parts: text and image URLs. Send images and files as URLs, not inline data: a request body is at most 4 MB."}},"required":["role","content"]}},"max_tokens":{"type":"integer","description":"The longest answer, in tokens. Set it: a call reserves its longest possible answer while it runs, and without a limit that is the model's whole output. max_completion_tokens works too."},"stream":{"type":"boolean","description":"Send the answer as it is written, as server-sent events. Recommended for anything longer than a sentence: streams may run 30 minutes, other calls 13."},"stream_options":{"type":"object","description":"{ \"include_usage\": true } adds a last chunk with usage, including what the call cost."},"temperature":{"type":"number","description":"Randomness, 0 to 2. Lower is more predictable."},"tools":{"type":"array","items":{"type":"object"},"description":"Functions the model may call, in the format's own shape. Your code runs them and sends the results back. Web search and web fetch are also served."},"tool_choice":{"anyOf":[{"type":"string"},{"type":"object"}],"description":"\"auto\", \"none\", \"required\" or one function by name."},"response_format":{"type":"object","description":"Structured output: { \"type\": \"json_schema\", \"json_schema\": { ... } } makes the answer match your schema."},"reasoning":{"type":"object","description":"For reasoning models: { \"effort\": \"low\" | \"medium\" | \"high\" }. Reasoning tokens are billed as output."},"models":{"type":"array","items":{"type":"string"},"description":"Fallback models, tried in order when the first is busy or down. The reserve covers the dearest of them."}},"required":["model","messages"]},"example":{"model":"anthropic/claude-sonnet-5.5","messages":[{"role":"system","content":"You are a concise trading assistant."},{"role":"user","content":"Summarize the last 24 hours of trades in one line."}],"max_tokens":256}}}},"responses":{"200":{"description":"OK","headers":{"x-binference-call-id":{"description":"This call's id on your Calls page. Quote it when you contact us.","schema":{"type":"string"}},"x-generation-id":{"description":"The model provider's id for the answer.","schema":{"type":"string"}},"retry-after":{"description":"On a 429 or 503: whole seconds to wait before sending again.","schema":{"type":"integer"}},"retry-after-ms":{"description":"The same wait in milliseconds. The OpenAI and Anthropic SDKs read it.","schema":{"type":"integer"}}},"content":{"application/json":{"schema":{"type":"object","additionalProperties":true,"properties":{"id":{"type":"string","description":"The answer's id."},"choices":{"type":"array","items":{"type":"object"},"description":"The answer: message.content, any message.tool_calls, and finish_reason."},"usage":{"type":"object","properties":{"prompt_tokens":{"type":"integer","description":"Tokens read."},"completion_tokens":{"type":"integer","description":"Tokens written, reasoning included."},"cost":{"type":"number","description":"What this call is charged, in dollars. Exactly what leaves the balance."}},"required":[],"description":"Tokens and what the call cost."}}},"example":{"id":"gen-1790755652-kQ3hTz","object":"chat.completion","created":1790755652,"model":"anthropic/claude-sonnet-5.5","choices":[{"index":0,"message":{"role":"assistant","content":"12 trades today: 9 wins, net +3.4 BNB, biggest move on $NOVA."},"finish_reason":"stop"}],"usage":{"prompt_tokens":31,"completion_tokens":22,"total_tokens":53,"cost":0.000338}}},"text/event-stream":{"schema":{"type":"string","description":"Server-sent events, with stream: true."},"example":"data: {\"id\":\"gen-1790755652-kQ3hTz\",\"object\":\"chat.completion.chunk\",\"choices\":[{\"index\":0,\"delta\":{\"role\":\"assistant\",\"content\":\"12 trades\"}}]}\n\ndata: {\"id\":\"gen-1790755652-kQ3hTz\",\"object\":\"chat.completion.chunk\",\"choices\":[{\"index\":0,\"delta\":{\"content\":\" today: 9 wins\"}}]}\n\ndata: {\"id\":\"gen-1790755652-kQ3hTz\",\"object\":\"chat.completion.chunk\",\"choices\":[{\"index\":0,\"delta\":{},\"finish_reason\":\"stop\"}],\"usage\":{\"prompt_tokens\":31,\"completion_tokens\":22,\"total_tokens\":53,\"cost\":0.000338}}\n\ndata: [DONE]"}}},"400":{"description":"invalid_request, unknown_model, model_not_served, tool_not_served, request_over_capacity","content":{"application/json":{"schema":{"$ref":"#/components/schemas/OpenAIError"},"examples":{"invalid_request":{"summary":"Bad request body","value":{"error":{"message":"The request body is not valid JSON.","type":"invalid_request_error","code":"invalid_request","param":null}}},"unknown_model":{"summary":"Unknown model","value":{"error":{"message":"Unknown model `openai/gpt-9`. Every model we serve is listed at /api/v1/models.","type":"invalid_request_error","code":"unknown_model","param":null}}},"model_not_served":{"summary":"Model not served","value":{"error":{"message":"`meta-llama/llama-5:free` is a free model, and free models are not served here. Every model we serve is listed at /api/v1/models.","type":"invalid_request_error","code":"model_not_served","param":null}}},"tool_not_served":{"summary":"Tool not served","value":{"error":{"message":"X search bills for every post it returns, with no limit a request can set, so its cost cannot be reserved up front. Remove `x_search` to search the web only.","type":"invalid_request_error","code":"tool_not_served","param":null}}},"request_over_capacity":{"summary":"Output limit too high right now","value":{"error":{"message":"This request asks for more output than the gateway can take on right now. Lower max_tokens (max_output_tokens on Responses) and retry.","type":"invalid_request_error","code":"request_over_capacity","param":null}}}}}}},"401":{"description":"invalid_api_key","content":{"application/json":{"schema":{"$ref":"#/components/schemas/OpenAIError"},"examples":{"invalid_api_key":{"summary":"Bad or missing key","value":{"error":{"message":"Missing or malformed API key. Send your binf_ key as `Authorization: Bearer <key>` or `x-api-key`.","type":"authentication_error","code":"invalid_api_key","param":null}}}}}}},"402":{"description":"insufficient_balance, key_limit","content":{"application/json":{"schema":{"$ref":"#/components/schemas/OpenAIError"},"examples":{"insufficient_balance":{"summary":"Balance too low","value":{"error":{"message":"Balance $0.004210 ($0.001000 held by calls running or being priced) is below this request's reserve of $0.019660. Lower the output limit or add credit.","type":"billing_error","code":"insufficient_balance","param":null}}},"key_limit":{"summary":"Key's spending limit reached","value":{"error":{"message":"This key's limit of $5.000000 a day is used up: $4.990000 spent and $0.000000 held by calls running or being priced, and this request reserves $0.019660. It resets at 2026-10-01T00:00:00.000Z. Lower the output limit, or ask the agent's owner to raise the key's limit.","type":"billing_error","code":"key_limit","param":null}}}}}}},"403":{"description":"agent_inactive, agent_suspended, provider_refused","content":{"application/json":{"schema":{"$ref":"#/components/schemas/OpenAIError"},"examples":{"agent_inactive":{"summary":"Agent not active yet","value":{"error":{"message":"This agent is not active yet. Activate it on its page to use the gateway.","type":"permission_error","code":"agent_inactive","param":null}}},"agent_suspended":{"summary":"Agent suspended","value":{"error":{"message":"This agent is suspended.","type":"permission_error","code":"agent_suspended","param":null}}},"provider_refused":{"summary":"Provider refused","value":{"error":{"message":"The model's provider refused this request for this model.","type":"permission_error","code":"provider_refused","param":null}}}}}}},"404":{"description":"not_found, no_provider","content":{"application/json":{"schema":{"$ref":"#/components/schemas/OpenAIError"},"examples":{"not_found":{"summary":"Path not served","value":{"error":{"message":"`POST /v1/embeddings` is not served here. The models are served at /v1/chat/completions, /v1/responses and /v1/messages.","type":"not_found_error","code":"not_found","param":null}}},"no_provider":{"summary":"No provider for this request","value":{"error":{"message":"No provider can serve this request with this model. Try another model, or fewer options such as tools or images.","type":"not_found_error","code":"no_provider","param":null}}}}}}},"408":{"description":"timeout","content":{"application/json":{"schema":{"$ref":"#/components/schemas/OpenAIError"},"examples":{"timeout":{"summary":"Provider timed out","value":{"error":{"message":"The model took too long to answer. Try again.","type":"server_error","code":"timeout","param":null}}}}}}},"413":{"description":"request_too_large","content":{"application/json":{"schema":{"$ref":"#/components/schemas/OpenAIError"},"examples":{"request_too_large":{"summary":"Request too large","value":{"error":{"message":"The request body is larger than 4 MB. Send images and files as URLs instead of inline data.","type":"invalid_request_error","code":"request_too_large","param":null}}}}}}},"422":{"description":"unprocessable","content":{"application/json":{"schema":{"$ref":"#/components/schemas/OpenAIError"},"examples":{"unprocessable":{"summary":"Provider couldn't process it","value":{"error":{"message":"The model's provider could not process this request.","type":"server_error","code":"unprocessable","param":null}}}}}}},"429":{"description":"rate_limited, too_many_running_calls, daily_share_used","headers":{"x-binference-call-id":{"description":"This call's id on your Calls page. Quote it when you contact us.","schema":{"type":"string"}},"x-generation-id":{"description":"The model provider's id for the answer.","schema":{"type":"string"}},"retry-after":{"description":"On a 429 or 503: whole seconds to wait before sending again.","schema":{"type":"integer"}},"retry-after-ms":{"description":"The same wait in milliseconds. The OpenAI and Anthropic SDKs read it.","schema":{"type":"integer"}}},"content":{"application/json":{"schema":{"$ref":"#/components/schemas/OpenAIError"},"examples":{"rate_limited":{"summary":"Too fast","value":{"error":{"message":"This agent is starting calls faster than 600 a minute. Retry in 1 second.","type":"rate_limit_error","code":"rate_limited","param":null}}},"too_many_running_calls":{"summary":"8 calls already running","value":{"error":{"message":"This agent already has 8 calls running. Retry when one finishes.","type":"rate_limit_error","code":"too_many_running_calls","param":null}}},"daily_share_used":{"summary":"Fair share of a busy day used","value":{"error":{"message":"Today's shared AI capacity is running low, and this agent has used its fair share of it ($12.000000 of $12.000000). It can start calls again at 00:00 UTC, or sooner if capacity is added.","type":"rate_limit_error","code":"daily_share_used","param":null}}}}}}},"500":{"description":"gateway_error","content":{"application/json":{"schema":{"$ref":"#/components/schemas/OpenAIError"},"examples":{"gateway_error":{"summary":"Gateway failed","value":{"error":{"message":"The gateway failed on this call. Retry it.","type":"server_error","code":"gateway_error","param":null}}}}}}},"502":{"description":"upstream_unreachable, provider_error, upstream_error","content":{"application/json":{"schema":{"$ref":"#/components/schemas/OpenAIError"},"examples":{"upstream_unreachable":{"summary":"Provider unreachable","value":{"error":{"message":"Could not reach the model provider. Try again.","type":"server_error","code":"upstream_unreachable","param":null}}},"provider_error":{"summary":"Provider failed","value":{"error":{"message":"The model's provider failed to answer. Try again in a moment.","type":"server_error","code":"provider_error","param":null}}},"upstream_error":{"summary":"Other provider error","value":{"error":{"message":"The request could not be completed. Try again in a moment.","type":"server_error","code":"upstream_error","param":null}}}}}}},"503":{"description":"model_busy, gateway_busy, gateway_capacity, gateway_unavailable, catalog_unavailable, service_unavailable, provider_unavailable","headers":{"x-binference-call-id":{"description":"This call's id on your Calls page. Quote it when you contact us.","schema":{"type":"string"}},"x-generation-id":{"description":"The model provider's id for the answer.","schema":{"type":"string"}},"retry-after":{"description":"On a 429 or 503: whole seconds to wait before sending again.","schema":{"type":"integer"}},"retry-after-ms":{"description":"The same wait in milliseconds. The OpenAI and Anthropic SDKs read it.","schema":{"type":"integer"}}},"content":{"application/json":{"schema":{"$ref":"#/components/schemas/OpenAIError"},"examples":{"model_busy":{"summary":"Model busy","value":{"error":{"message":"The model `openai/gpt-6.1` is busy at its providers right now. Retry in 12 seconds.","type":"server_error","code":"model_busy","param":null}}},"gateway_busy":{"summary":"Gateway busy","value":{"error":{"message":"The gateway is busy right now. Retry in 8 seconds.","type":"server_error","code":"gateway_busy","param":null}}},"gateway_capacity":{"summary":"Out of capacity briefly","value":{"error":{"message":"The gateway is briefly out of capacity. Try again soon.","type":"server_error","code":"gateway_capacity","param":null}}},"gateway_unavailable":{"summary":"Gateway unavailable briefly","value":{"error":{"message":"The gateway is briefly unavailable. Try again soon.","type":"server_error","code":"gateway_unavailable","param":null}}},"catalog_unavailable":{"summary":"Prices unavailable","value":{"error":{"message":"Model prices are unavailable. Try again shortly.","type":"server_error","code":"catalog_unavailable","param":null}}},"service_unavailable":{"summary":"Database hiccup","value":{"error":{"message":"The balance is briefly unavailable. Try again in a moment.","type":"server_error","code":"service_unavailable","param":null}}},"provider_unavailable":{"summary":"Model unavailable","value":{"error":{"message":"The model is unavailable right now. Try again in a moment.","type":"server_error","code":"provider_unavailable","param":null}}}}}}},"504":{"description":"upstream_timeout","content":{"application/json":{"schema":{"$ref":"#/components/schemas/OpenAIError"},"examples":{"upstream_timeout":{"summary":"No answer in time","value":{"error":{"message":"The model did not answer in time.","type":"server_error","code":"upstream_timeout","param":null}}}}}}}}}},"/responses":{"post":{"operationId":"responses","summary":"Responses","description":"The OpenAI Responses format, which Codex speaks. Stateless: nothing is stored.","externalDocs":{"url":"https://docs.binference.io/api/responses"},"security":[{"bearer":[]},{"apiKey":[]}],"requestBody":{"required":true,"content":{"application/json":{"schema":{"type":"object","additionalProperties":true,"properties":{"model":{"type":"string","description":"A model id from GET /models, such as anthropic/claude-sonnet-5.5. Add :online for web search, or :nitro, :floor or :exacto to steer which provider runs it."},"input":{"anyOf":[{"type":"string"},{"type":"array","items":{"type":"object"}}],"description":"A prompt, or the conversation as input items."},"instructions":{"type":"string","description":"The system prompt."},"max_output_tokens":{"type":"integer","description":"The longest answer, in tokens. Set it: the call reserves its longest possible answer while it runs."},"stream":{"type":"boolean","description":"Send the answer as it is written, as server-sent events. Recommended for anything longer than a sentence: streams may run 30 minutes, other calls 13."},"temperature":{"type":"number","description":"Randomness, 0 to 2. Lower is more predictable."},"tools":{"type":"array","items":{"type":"object"},"description":"Functions the model may call, in the format's own shape. Your code runs them and sends the results back. Web search and web fetch are also served."},"max_tool_calls":{"type":"integer","description":"The most tool steps in one call. Web search loops reserve for every step, so a lower number reserves less."},"reasoning":{"type":"object","description":"For reasoning models: { \"effort\": \"low\" | \"medium\" | \"high\" }."},"models":{"type":"array","items":{"type":"string"},"description":"Fallback models, tried in order when the first is busy or down. The reserve covers the dearest of them."}},"required":["model","input"]},"example":{"model":"openai/gpt-6.1-sol","instructions":"You are a concise trading assistant.","input":"Summarize the last 24 hours of trades in one line.","max_output_tokens":256}}}},"responses":{"200":{"description":"OK","headers":{"x-binference-call-id":{"description":"This call's id on your Calls page. Quote it when you contact us.","schema":{"type":"string"}},"x-generation-id":{"description":"The model provider's id for the answer.","schema":{"type":"string"}},"retry-after":{"description":"On a 429 or 503: whole seconds to wait before sending again.","schema":{"type":"integer"}},"retry-after-ms":{"description":"The same wait in milliseconds. The OpenAI and Anthropic SDKs read it.","schema":{"type":"integer"}}},"content":{"application/json":{"schema":{"type":"object","additionalProperties":true,"properties":{"id":{"type":"string","description":"The response's id."},"output":{"type":"array","items":{"type":"object"},"description":"Output items: messages, reasoning and tool calls. output_text joins the text."},"usage":{"type":"object","description":"input_tokens, output_tokens and cost: what this call is charged, in dollars."}}},"example":{"id":"resp_7Yq2c1","object":"response","status":"completed","model":"openai/gpt-6.1-sol","output":[{"type":"message","role":"assistant","content":[{"type":"output_text","text":"12 trades today: 9 wins, net +3.4 BNB, biggest move on $NOVA."}]}],"usage":{"input_tokens":29,"output_tokens":21,"total_tokens":50,"cost":0.000322}}},"text/event-stream":{"schema":{"type":"string","description":"Server-sent events, with stream: true."},"example":"event: response.created\ndata: {\"type\":\"response.created\",\"response\":{\"id\":\"resp_7Yq2c1\",\"status\":\"in_progress\"}}\n\nevent: response.output_text.delta\ndata: {\"type\":\"response.output_text.delta\",\"delta\":\"12 trades today\"}\n\nevent: response.completed\ndata: {\"type\":\"response.completed\",\"response\":{\"id\":\"resp_7Yq2c1\",\"status\":\"completed\",\"usage\":{\"input_tokens\":29,\"output_tokens\":21,\"total_tokens\":50,\"cost\":0.000322}}}"}}},"400":{"description":"invalid_request, unknown_model, model_not_served, tool_not_served, request_over_capacity","content":{"application/json":{"schema":{"$ref":"#/components/schemas/OpenAIError"},"examples":{"invalid_request":{"summary":"Bad request body","value":{"error":{"message":"The request body is not valid JSON.","type":"invalid_request_error","code":"invalid_request","param":null}}},"unknown_model":{"summary":"Unknown model","value":{"error":{"message":"Unknown model `openai/gpt-9`. Every model we serve is listed at /api/v1/models.","type":"invalid_request_error","code":"unknown_model","param":null}}},"model_not_served":{"summary":"Model not served","value":{"error":{"message":"`meta-llama/llama-5:free` is a free model, and free models are not served here. Every model we serve is listed at /api/v1/models.","type":"invalid_request_error","code":"model_not_served","param":null}}},"tool_not_served":{"summary":"Tool not served","value":{"error":{"message":"X search bills for every post it returns, with no limit a request can set, so its cost cannot be reserved up front. Remove `x_search` to search the web only.","type":"invalid_request_error","code":"tool_not_served","param":null}}},"request_over_capacity":{"summary":"Output limit too high right now","value":{"error":{"message":"This request asks for more output than the gateway can take on right now. Lower max_tokens (max_output_tokens on Responses) and retry.","type":"invalid_request_error","code":"request_over_capacity","param":null}}}}}}},"401":{"description":"invalid_api_key","content":{"application/json":{"schema":{"$ref":"#/components/schemas/OpenAIError"},"examples":{"invalid_api_key":{"summary":"Bad or missing key","value":{"error":{"message":"Missing or malformed API key. Send your binf_ key as `Authorization: Bearer <key>` or `x-api-key`.","type":"authentication_error","code":"invalid_api_key","param":null}}}}}}},"402":{"description":"insufficient_balance, key_limit","content":{"application/json":{"schema":{"$ref":"#/components/schemas/OpenAIError"},"examples":{"insufficient_balance":{"summary":"Balance too low","value":{"error":{"message":"Balance $0.004210 ($0.001000 held by calls running or being priced) is below this request's reserve of $0.019660. Lower the output limit or add credit.","type":"billing_error","code":"insufficient_balance","param":null}}},"key_limit":{"summary":"Key's spending limit reached","value":{"error":{"message":"This key's limit of $5.000000 a day is used up: $4.990000 spent and $0.000000 held by calls running or being priced, and this request reserves $0.019660. It resets at 2026-10-01T00:00:00.000Z. Lower the output limit, or ask the agent's owner to raise the key's limit.","type":"billing_error","code":"key_limit","param":null}}}}}}},"403":{"description":"agent_inactive, agent_suspended, provider_refused","content":{"application/json":{"schema":{"$ref":"#/components/schemas/OpenAIError"},"examples":{"agent_inactive":{"summary":"Agent not active yet","value":{"error":{"message":"This agent is not active yet. Activate it on its page to use the gateway.","type":"permission_error","code":"agent_inactive","param":null}}},"agent_suspended":{"summary":"Agent suspended","value":{"error":{"message":"This agent is suspended.","type":"permission_error","code":"agent_suspended","param":null}}},"provider_refused":{"summary":"Provider refused","value":{"error":{"message":"The model's provider refused this request for this model.","type":"permission_error","code":"provider_refused","param":null}}}}}}},"404":{"description":"not_found, no_provider","content":{"application/json":{"schema":{"$ref":"#/components/schemas/OpenAIError"},"examples":{"not_found":{"summary":"Path not served","value":{"error":{"message":"`POST /v1/embeddings` is not served here. The models are served at /v1/chat/completions, /v1/responses and /v1/messages.","type":"not_found_error","code":"not_found","param":null}}},"no_provider":{"summary":"No provider for this request","value":{"error":{"message":"No provider can serve this request with this model. Try another model, or fewer options such as tools or images.","type":"not_found_error","code":"no_provider","param":null}}}}}}},"408":{"description":"timeout","content":{"application/json":{"schema":{"$ref":"#/components/schemas/OpenAIError"},"examples":{"timeout":{"summary":"Provider timed out","value":{"error":{"message":"The model took too long to answer. Try again.","type":"server_error","code":"timeout","param":null}}}}}}},"413":{"description":"request_too_large","content":{"application/json":{"schema":{"$ref":"#/components/schemas/OpenAIError"},"examples":{"request_too_large":{"summary":"Request too large","value":{"error":{"message":"The request body is larger than 4 MB. Send images and files as URLs instead of inline data.","type":"invalid_request_error","code":"request_too_large","param":null}}}}}}},"422":{"description":"unprocessable","content":{"application/json":{"schema":{"$ref":"#/components/schemas/OpenAIError"},"examples":{"unprocessable":{"summary":"Provider couldn't process it","value":{"error":{"message":"The model's provider could not process this request.","type":"server_error","code":"unprocessable","param":null}}}}}}},"429":{"description":"rate_limited, too_many_running_calls, daily_share_used","headers":{"x-binference-call-id":{"description":"This call's id on your Calls page. Quote it when you contact us.","schema":{"type":"string"}},"x-generation-id":{"description":"The model provider's id for the answer.","schema":{"type":"string"}},"retry-after":{"description":"On a 429 or 503: whole seconds to wait before sending again.","schema":{"type":"integer"}},"retry-after-ms":{"description":"The same wait in milliseconds. The OpenAI and Anthropic SDKs read it.","schema":{"type":"integer"}}},"content":{"application/json":{"schema":{"$ref":"#/components/schemas/OpenAIError"},"examples":{"rate_limited":{"summary":"Too fast","value":{"error":{"message":"This agent is starting calls faster than 600 a minute. Retry in 1 second.","type":"rate_limit_error","code":"rate_limited","param":null}}},"too_many_running_calls":{"summary":"8 calls already running","value":{"error":{"message":"This agent already has 8 calls running. Retry when one finishes.","type":"rate_limit_error","code":"too_many_running_calls","param":null}}},"daily_share_used":{"summary":"Fair share of a busy day used","value":{"error":{"message":"Today's shared AI capacity is running low, and this agent has used its fair share of it ($12.000000 of $12.000000). It can start calls again at 00:00 UTC, or sooner if capacity is added.","type":"rate_limit_error","code":"daily_share_used","param":null}}}}}}},"500":{"description":"gateway_error","content":{"application/json":{"schema":{"$ref":"#/components/schemas/OpenAIError"},"examples":{"gateway_error":{"summary":"Gateway failed","value":{"error":{"message":"The gateway failed on this call. Retry it.","type":"server_error","code":"gateway_error","param":null}}}}}}},"502":{"description":"upstream_unreachable, provider_error, upstream_error","content":{"application/json":{"schema":{"$ref":"#/components/schemas/OpenAIError"},"examples":{"upstream_unreachable":{"summary":"Provider unreachable","value":{"error":{"message":"Could not reach the model provider. Try again.","type":"server_error","code":"upstream_unreachable","param":null}}},"provider_error":{"summary":"Provider failed","value":{"error":{"message":"The model's provider failed to answer. Try again in a moment.","type":"server_error","code":"provider_error","param":null}}},"upstream_error":{"summary":"Other provider error","value":{"error":{"message":"The request could not be completed. Try again in a moment.","type":"server_error","code":"upstream_error","param":null}}}}}}},"503":{"description":"model_busy, gateway_busy, gateway_capacity, gateway_unavailable, catalog_unavailable, service_unavailable, provider_unavailable","headers":{"x-binference-call-id":{"description":"This call's id on your Calls page. Quote it when you contact us.","schema":{"type":"string"}},"x-generation-id":{"description":"The model provider's id for the answer.","schema":{"type":"string"}},"retry-after":{"description":"On a 429 or 503: whole seconds to wait before sending again.","schema":{"type":"integer"}},"retry-after-ms":{"description":"The same wait in milliseconds. The OpenAI and Anthropic SDKs read it.","schema":{"type":"integer"}}},"content":{"application/json":{"schema":{"$ref":"#/components/schemas/OpenAIError"},"examples":{"model_busy":{"summary":"Model busy","value":{"error":{"message":"The model `openai/gpt-6.1` is busy at its providers right now. Retry in 12 seconds.","type":"server_error","code":"model_busy","param":null}}},"gateway_busy":{"summary":"Gateway busy","value":{"error":{"message":"The gateway is busy right now. Retry in 8 seconds.","type":"server_error","code":"gateway_busy","param":null}}},"gateway_capacity":{"summary":"Out of capacity briefly","value":{"error":{"message":"The gateway is briefly out of capacity. Try again soon.","type":"server_error","code":"gateway_capacity","param":null}}},"gateway_unavailable":{"summary":"Gateway unavailable briefly","value":{"error":{"message":"The gateway is briefly unavailable. Try again soon.","type":"server_error","code":"gateway_unavailable","param":null}}},"catalog_unavailable":{"summary":"Prices unavailable","value":{"error":{"message":"Model prices are unavailable. Try again shortly.","type":"server_error","code":"catalog_unavailable","param":null}}},"service_unavailable":{"summary":"Database hiccup","value":{"error":{"message":"The balance is briefly unavailable. Try again in a moment.","type":"server_error","code":"service_unavailable","param":null}}},"provider_unavailable":{"summary":"Model unavailable","value":{"error":{"message":"The model is unavailable right now. Try again in a moment.","type":"server_error","code":"provider_unavailable","param":null}}}}}}},"504":{"description":"upstream_timeout","content":{"application/json":{"schema":{"$ref":"#/components/schemas/OpenAIError"},"examples":{"upstream_timeout":{"summary":"No answer in time","value":{"error":{"message":"The model did not answer in time.","type":"server_error","code":"upstream_timeout","param":null}}}}}}}}}},"/messages":{"post":{"operationId":"messages","summary":"Messages","description":"The Anthropic Messages format, which Claude Code and the Anthropic SDKs speak.","externalDocs":{"url":"https://docs.binference.io/api/messages"},"security":[{"bearer":[]},{"apiKey":[]}],"parameters":[{"name":"anthropic-version","in":"header","required":false,"description":"Passed on as sent. The Anthropic SDKs set it.","schema":{"type":"string"}},{"name":"anthropic-beta","in":"header","required":false,"description":"Passed on as sent, for beta features.","schema":{"type":"string"}}],"requestBody":{"required":true,"content":{"application/json":{"schema":{"type":"object","additionalProperties":true,"properties":{"model":{"type":"string","description":"A model id from GET /models. Claude clients' own names work too: claude-sonnet-5-5, dated names and the [1m] suffix map to the listed model."},"messages":{"type":"array","items":{"type":"object"},"description":"The conversation so far, alternating user and assistant."},"max_tokens":{"type":"integer","description":"The longest answer, in tokens. The call reserves this much output while it runs."},"system":{"anyOf":[{"type":"string"},{"type":"array","items":{"type":"object"}}],"description":"The system prompt."},"stream":{"type":"boolean","description":"Send the answer as it is written, as server-sent events. Recommended for anything longer than a sentence: streams may run 30 minutes, other calls 13."},"temperature":{"type":"number","description":"Randomness, 0 to 2. Lower is more predictable."},"tools":{"type":"array","items":{"type":"object"},"description":"Functions the model may call, in the format's own shape. Your code runs them and sends the results back. Web search and web fetch are also served."},"thinking":{"type":"object","description":"{ \"type\": \"enabled\", \"budget_tokens\": 2048 } for extended thinking. Thinking is billed as output."}},"required":["model","messages","max_tokens"]},"example":{"model":"anthropic/claude-sonnet-5.5","max_tokens":256,"system":"You are a concise trading assistant.","messages":[{"role":"user","content":"Summarize the last 24 hours of trades in one line."}]}}}},"responses":{"200":{"description":"OK","headers":{"x-binference-call-id":{"description":"This call's id on your Calls page. Quote it when you contact us.","schema":{"type":"string"}},"x-generation-id":{"description":"The model provider's id for the answer.","schema":{"type":"string"}},"retry-after":{"description":"On a 429 or 503: whole seconds to wait before sending again.","schema":{"type":"integer"}},"retry-after-ms":{"description":"The same wait in milliseconds. The OpenAI and Anthropic SDKs read it.","schema":{"type":"integer"}}},"content":{"application/json":{"schema":{"type":"object","additionalProperties":true,"properties":{"id":{"type":"string","description":"The message's id."},"content":{"type":"array","items":{"type":"object"},"description":"Blocks: text, thinking and tool_use."},"stop_reason":{"type":"string","description":"Why the answer ended."},"usage":{"type":"object","description":"input_tokens and output_tokens. The Messages format carries no cost: find each call's charge on your Calls page within a minute."}}},"example":{"id":"msg_01XFDUDYJgAACzvnptvVoYEL","type":"message","role":"assistant","model":"anthropic/claude-sonnet-5.5","content":[{"type":"text","text":"12 trades today: 9 wins, net +3.4 BNB, biggest move on $NOVA."}],"stop_reason":"end_turn","usage":{"input_tokens":27,"output_tokens":22}}},"text/event-stream":{"schema":{"type":"string","description":"Server-sent events, with stream: true."},"example":"event: message_start\ndata: {\"type\":\"message_start\",\"message\":{\"id\":\"msg_01XFDUDYJgAACzvnptvVoYEL\",\"role\":\"assistant\",\"usage\":{\"input_tokens\":27,\"output_tokens\":1}}}\n\nevent: content_block_delta\ndata: {\"type\":\"content_block_delta\",\"index\":0,\"delta\":{\"type\":\"text_delta\",\"text\":\"12 trades today\"}}\n\nevent: message_delta\ndata: {\"type\":\"message_delta\",\"delta\":{\"stop_reason\":\"end_turn\"},\"usage\":{\"output_tokens\":22}}\n\nevent: message_stop\ndata: {\"type\":\"message_stop\"}"}}},"400":{"description":"invalid_request, unknown_model, model_not_served, tool_not_served, request_over_capacity","content":{"application/json":{"schema":{"$ref":"#/components/schemas/AnthropicError"},"examples":{"invalid_request":{"summary":"Bad request body","value":{"type":"error","error":{"type":"invalid_request_error","message":"The request body is not valid JSON.","code":"invalid_request"}}},"unknown_model":{"summary":"Unknown model","value":{"type":"error","error":{"type":"invalid_request_error","message":"Unknown model `openai/gpt-9`. Every model we serve is listed at /api/v1/models.","code":"unknown_model"}}},"model_not_served":{"summary":"Model not served","value":{"type":"error","error":{"type":"invalid_request_error","message":"`meta-llama/llama-5:free` is a free model, and free models are not served here. Every model we serve is listed at /api/v1/models.","code":"model_not_served"}}},"tool_not_served":{"summary":"Tool not served","value":{"type":"error","error":{"type":"invalid_request_error","message":"X search bills for every post it returns, with no limit a request can set, so its cost cannot be reserved up front. Remove `x_search` to search the web only.","code":"tool_not_served"}}},"request_over_capacity":{"summary":"Output limit too high right now","value":{"type":"error","error":{"type":"invalid_request_error","message":"This request asks for more output than the gateway can take on right now. Lower max_tokens (max_output_tokens on Responses) and retry.","code":"request_over_capacity"}}}}}}},"401":{"description":"invalid_api_key","content":{"application/json":{"schema":{"$ref":"#/components/schemas/AnthropicError"},"examples":{"invalid_api_key":{"summary":"Bad or missing key","value":{"type":"error","error":{"type":"authentication_error","message":"Missing or malformed API key. Send your binf_ key as `Authorization: Bearer <key>` or `x-api-key`.","code":"invalid_api_key"}}}}}}},"402":{"description":"insufficient_balance, key_limit","content":{"application/json":{"schema":{"$ref":"#/components/schemas/AnthropicError"},"examples":{"insufficient_balance":{"summary":"Balance too low","value":{"type":"error","error":{"type":"billing_error","message":"Balance $0.004210 ($0.001000 held by calls running or being priced) is below this request's reserve of $0.019660. Lower the output limit or add credit.","code":"insufficient_balance"}}},"key_limit":{"summary":"Key's spending limit reached","value":{"type":"error","error":{"type":"billing_error","message":"This key's limit of $5.000000 a day is used up: $4.990000 spent and $0.000000 held by calls running or being priced, and this request reserves $0.019660. It resets at 2026-10-01T00:00:00.000Z. Lower the output limit, or ask the agent's owner to raise the key's limit.","code":"key_limit"}}}}}}},"403":{"description":"agent_inactive, agent_suspended, provider_refused","content":{"application/json":{"schema":{"$ref":"#/components/schemas/AnthropicError"},"examples":{"agent_inactive":{"summary":"Agent not active yet","value":{"type":"error","error":{"type":"permission_error","message":"This agent is not active yet. Activate it on its page to use the gateway.","code":"agent_inactive"}}},"agent_suspended":{"summary":"Agent suspended","value":{"type":"error","error":{"type":"permission_error","message":"This agent is suspended.","code":"agent_suspended"}}},"provider_refused":{"summary":"Provider refused","value":{"type":"error","error":{"type":"permission_error","message":"The model's provider refused this request for this model.","code":"provider_refused"}}}}}}},"404":{"description":"not_found, no_provider","content":{"application/json":{"schema":{"$ref":"#/components/schemas/AnthropicError"},"examples":{"not_found":{"summary":"Path not served","value":{"type":"error","error":{"type":"not_found_error","message":"`POST /v1/embeddings` is not served here. The models are served at /v1/chat/completions, /v1/responses and /v1/messages.","code":"not_found"}}},"no_provider":{"summary":"No provider for this request","value":{"type":"error","error":{"type":"not_found_error","message":"No provider can serve this request with this model. Try another model, or fewer options such as tools or images.","code":"no_provider"}}}}}}},"408":{"description":"timeout","content":{"application/json":{"schema":{"$ref":"#/components/schemas/AnthropicError"},"examples":{"timeout":{"summary":"Provider timed out","value":{"type":"error","error":{"type":"api_error","message":"The model took too long to answer. Try again.","code":"timeout"}}}}}}},"413":{"description":"request_too_large","content":{"application/json":{"schema":{"$ref":"#/components/schemas/AnthropicError"},"examples":{"request_too_large":{"summary":"Request too large","value":{"type":"error","error":{"type":"request_too_large","message":"The request body is larger than 4 MB. Send images and files as URLs instead of inline data.","code":"request_too_large"}}}}}}},"422":{"description":"unprocessable","content":{"application/json":{"schema":{"$ref":"#/components/schemas/AnthropicError"},"examples":{"unprocessable":{"summary":"Provider couldn't process it","value":{"type":"error","error":{"type":"api_error","message":"The model's provider could not process this request.","code":"unprocessable"}}}}}}},"429":{"description":"rate_limited, too_many_running_calls, daily_share_used","headers":{"x-binference-call-id":{"description":"This call's id on your Calls page. Quote it when you contact us.","schema":{"type":"string"}},"x-generation-id":{"description":"The model provider's id for the answer.","schema":{"type":"string"}},"retry-after":{"description":"On a 429 or 503: whole seconds to wait before sending again.","schema":{"type":"integer"}},"retry-after-ms":{"description":"The same wait in milliseconds. The OpenAI and Anthropic SDKs read it.","schema":{"type":"integer"}}},"content":{"application/json":{"schema":{"$ref":"#/components/schemas/AnthropicError"},"examples":{"rate_limited":{"summary":"Too fast","value":{"type":"error","error":{"type":"rate_limit_error","message":"This agent is starting calls faster than 600 a minute. Retry in 1 second.","code":"rate_limited"}}},"too_many_running_calls":{"summary":"8 calls already running","value":{"type":"error","error":{"type":"rate_limit_error","message":"This agent already has 8 calls running. Retry when one finishes.","code":"too_many_running_calls"}}},"daily_share_used":{"summary":"Fair share of a busy day used","value":{"type":"error","error":{"type":"rate_limit_error","message":"Today's shared AI capacity is running low, and this agent has used its fair share of it ($12.000000 of $12.000000). It can start calls again at 00:00 UTC, or sooner if capacity is added.","code":"daily_share_used"}}}}}}},"500":{"description":"gateway_error","content":{"application/json":{"schema":{"$ref":"#/components/schemas/AnthropicError"},"examples":{"gateway_error":{"summary":"Gateway failed","value":{"type":"error","error":{"type":"api_error","message":"The gateway failed on this call. Retry it.","code":"gateway_error"}}}}}}},"502":{"description":"upstream_unreachable, provider_error, upstream_error","content":{"application/json":{"schema":{"$ref":"#/components/schemas/AnthropicError"},"examples":{"upstream_unreachable":{"summary":"Provider unreachable","value":{"type":"error","error":{"type":"api_error","message":"Could not reach the model provider. Try again.","code":"upstream_unreachable"}}},"provider_error":{"summary":"Provider failed","value":{"type":"error","error":{"type":"api_error","message":"The model's provider failed to answer. Try again in a moment.","code":"provider_error"}}},"upstream_error":{"summary":"Other provider error","value":{"type":"error","error":{"type":"api_error","message":"The request could not be completed. Try again in a moment.","code":"upstream_error"}}}}}}},"503":{"description":"gateway_busy, gateway_capacity, gateway_unavailable, catalog_unavailable, service_unavailable, provider_unavailable","headers":{"x-binference-call-id":{"description":"This call's id on your Calls page. Quote it when you contact us.","schema":{"type":"string"}},"x-generation-id":{"description":"The model provider's id for the answer.","schema":{"type":"string"}},"retry-after":{"description":"On a 429 or 503: whole seconds to wait before sending again.","schema":{"type":"integer"}},"retry-after-ms":{"description":"The same wait in milliseconds. The OpenAI and Anthropic SDKs read it.","schema":{"type":"integer"}}},"content":{"application/json":{"schema":{"$ref":"#/components/schemas/AnthropicError"},"examples":{"gateway_busy":{"summary":"Gateway busy","value":{"type":"error","error":{"type":"overloaded_error","message":"The gateway is busy right now. Retry in 8 seconds.","code":"gateway_busy"}}},"gateway_capacity":{"summary":"Out of capacity briefly","value":{"type":"error","error":{"type":"overloaded_error","message":"The gateway is briefly out of capacity. Try again soon.","code":"gateway_capacity"}}},"gateway_unavailable":{"summary":"Gateway unavailable briefly","value":{"type":"error","error":{"type":"overloaded_error","message":"The gateway is briefly unavailable. Try again soon.","code":"gateway_unavailable"}}},"catalog_unavailable":{"summary":"Prices unavailable","value":{"type":"error","error":{"type":"overloaded_error","message":"Model prices are unavailable. Try again shortly.","code":"catalog_unavailable"}}},"service_unavailable":{"summary":"Database hiccup","value":{"type":"error","error":{"type":"overloaded_error","message":"The balance is briefly unavailable. Try again in a moment.","code":"service_unavailable"}}},"provider_unavailable":{"summary":"Model unavailable","value":{"type":"error","error":{"type":"overloaded_error","message":"The model is unavailable right now. Try again in a moment.","code":"provider_unavailable"}}}}}}},"504":{"description":"upstream_timeout","content":{"application/json":{"schema":{"$ref":"#/components/schemas/AnthropicError"},"examples":{"upstream_timeout":{"summary":"No answer in time","value":{"type":"error","error":{"type":"api_error","message":"The model did not answer in time.","code":"upstream_timeout"}}}}}}},"529":{"description":"model_busy","headers":{"x-binference-call-id":{"description":"This call's id on your Calls page. Quote it when you contact us.","schema":{"type":"string"}},"x-generation-id":{"description":"The model provider's id for the answer.","schema":{"type":"string"}},"retry-after":{"description":"On a 429 or 503: whole seconds to wait before sending again.","schema":{"type":"integer"}},"retry-after-ms":{"description":"The same wait in milliseconds. The OpenAI and Anthropic SDKs read it.","schema":{"type":"integer"}}},"content":{"application/json":{"schema":{"$ref":"#/components/schemas/AnthropicError"},"examples":{"model_busy":{"summary":"Model busy","value":{"type":"error","error":{"type":"overloaded_error","message":"The model `openai/gpt-6.1` is busy at its providers right now. Retry in 12 seconds.","code":"model_busy"}}}}}}}}}},"/models":{"get":{"operationId":"models","summary":"List models","description":"Every model you can call, with its price per token. No key needed.","externalDocs":{"url":"https://docs.binference.io/api/models"},"security":[],"responses":{"200":{"description":"OK","content":{"application/json":{"schema":{"type":"object","additionalProperties":true,"properties":{"data":{"type":"array","description":"One entry per model.","items":{"type":"object","properties":{"id":{"type":"string","description":"The id to send as model."},"name":{"type":"string","description":"Its display name."},"context_length":{"type":"integer","description":"The longest prompt plus answer, in tokens."},"max_output_tokens":{"anyOf":[{"type":"integer"},{"type":"null"}],"description":"The longest answer it can write."},"pricing":{"type":"object","description":"What you pay, in dollars as decimal strings: prompt, completion and reasoning per token, request per call, image per image, web_search per search."}},"required":[]}}}},"example":{"object":"list","data":[{"id":"anthropic/claude-sonnet-5.5","object":"model","owned_by":"anthropic","name":"Anthropic: Claude Sonnet 5.5","context_length":1000000,"max_output_tokens":128000,"pricing":{"prompt":"0.0000024","completion":"0.000012","reasoning":"0.000012","request":"0","image":"0","web_search":"0.012"}}]}}}},"402":{"description":"insufficient_balance, key_limit","content":{"application/json":{"schema":{"$ref":"#/components/schemas/OpenAIError"},"examples":{"insufficient_balance":{"summary":"Balance too low","value":{"error":{"message":"Balance $0.004210 ($0.001000 held by calls running or being priced) is below this request's reserve of $0.019660. Lower the output limit or add credit.","type":"billing_error","code":"insufficient_balance","param":null}}},"key_limit":{"summary":"Key's spending limit reached","value":{"error":{"message":"This key's limit of $5.000000 a day is used up: $4.990000 spent and $0.000000 held by calls running or being priced, and this request reserves $0.019660. It resets at 2026-10-01T00:00:00.000Z. Lower the output limit, or ask the agent's owner to raise the key's limit.","type":"billing_error","code":"key_limit","param":null}}}}}}},"429":{"description":"rate_limited, too_many_running_calls, daily_share_used","headers":{"x-binference-call-id":{"description":"This call's id on your Calls page. Quote it when you contact us.","schema":{"type":"string"}},"x-generation-id":{"description":"The model provider's id for the answer.","schema":{"type":"string"}},"retry-after":{"description":"On a 429 or 503: whole seconds to wait before sending again.","schema":{"type":"integer"}},"retry-after-ms":{"description":"The same wait in milliseconds. The OpenAI and Anthropic SDKs read it.","schema":{"type":"integer"}}},"content":{"application/json":{"schema":{"$ref":"#/components/schemas/OpenAIError"},"examples":{"rate_limited":{"summary":"Too fast","value":{"error":{"message":"This agent is starting calls faster than 600 a minute. Retry in 1 second.","type":"rate_limit_error","code":"rate_limited","param":null}}},"too_many_running_calls":{"summary":"8 calls already running","value":{"error":{"message":"This agent already has 8 calls running. Retry when one finishes.","type":"rate_limit_error","code":"too_many_running_calls","param":null}}},"daily_share_used":{"summary":"Fair share of a busy day used","value":{"error":{"message":"Today's shared AI capacity is running low, and this agent has used its fair share of it ($12.000000 of $12.000000). It can start calls again at 00:00 UTC, or sooner if capacity is added.","type":"rate_limit_error","code":"daily_share_used","param":null}}}}}}},"500":{"description":"gateway_error","content":{"application/json":{"schema":{"$ref":"#/components/schemas/OpenAIError"},"examples":{"gateway_error":{"summary":"Gateway failed","value":{"error":{"message":"The gateway failed on this call. Retry it.","type":"server_error","code":"gateway_error","param":null}}}}}}},"503":{"description":"model_busy, gateway_busy, gateway_capacity, gateway_unavailable, catalog_unavailable, service_unavailable","headers":{"x-binference-call-id":{"description":"This call's id on your Calls page. Quote it when you contact us.","schema":{"type":"string"}},"x-generation-id":{"description":"The model provider's id for the answer.","schema":{"type":"string"}},"retry-after":{"description":"On a 429 or 503: whole seconds to wait before sending again.","schema":{"type":"integer"}},"retry-after-ms":{"description":"The same wait in milliseconds. The OpenAI and Anthropic SDKs read it.","schema":{"type":"integer"}}},"content":{"application/json":{"schema":{"$ref":"#/components/schemas/OpenAIError"},"examples":{"model_busy":{"summary":"Model busy","value":{"error":{"message":"The model `openai/gpt-6.1` is busy at its providers right now. Retry in 12 seconds.","type":"server_error","code":"model_busy","param":null}}},"gateway_busy":{"summary":"Gateway busy","value":{"error":{"message":"The gateway is busy right now. Retry in 8 seconds.","type":"server_error","code":"gateway_busy","param":null}}},"gateway_capacity":{"summary":"Out of capacity briefly","value":{"error":{"message":"The gateway is briefly out of capacity. Try again soon.","type":"server_error","code":"gateway_capacity","param":null}}},"gateway_unavailable":{"summary":"Gateway unavailable briefly","value":{"error":{"message":"The gateway is briefly unavailable. Try again soon.","type":"server_error","code":"gateway_unavailable","param":null}}},"catalog_unavailable":{"summary":"Prices unavailable","value":{"error":{"message":"Model prices are unavailable. Try again shortly.","type":"server_error","code":"catalog_unavailable","param":null}}},"service_unavailable":{"summary":"Database hiccup","value":{"error":{"message":"The balance is briefly unavailable. Try again in a moment.","type":"server_error","code":"service_unavailable","param":null}}}}}}}}}},"/balance":{"get":{"operationId":"balance","summary":"Get balance","description":"What the key's agent or account can spend now, what is expiring and the key's own limit.","externalDocs":{"url":"https://docs.binference.io/api/balance"},"security":[{"bearer":[]},{"apiKey":[]}],"responses":{"200":{"description":"OK","content":{"application/json":{"schema":{"type":"object","additionalProperties":true,"properties":{"agent":{"type":"object","description":"The agent or account the key belongs to: id, name, token and status."},"spendable_usd":{"type":"string","description":"What new calls can reserve right now: the balance less what running calls hold."},"balance_usd":{"type":"string","description":"The whole balance."},"reserved_usd":{"type":"string","description":"What running calls hold."},"running_calls":{"type":"integer","description":"Calls running now, at most 8."},"expiring":{"type":"array","items":{"type":"object"},"description":"What is left of each day's credit and when it expires (at, usd), soonest first."},"key_limit":{"anyOf":[{"type":"object"},{"type":"null"}],"description":"This key's spending limit, if it has one: limit_usd, reset (daily, weekly or monthly), spent_usd, reserved_usd, remaining_usd and resets_at."}}},"example":{"object":"balance","agent":{"id":42,"name":"Nova","token":"0x9f3c7e2b8a41d05c6e1f9a8b7c2d3e4f50617777","status":"active"},"spendable_usd":"18.402110","balance_usd":"18.422110","reserved_usd":"0.020000","running_calls":1,"expiring":[{"at":"2026-10-02T00:00:00.000Z","usd":"2.108400"},{"at":"2026-10-03T00:00:00.000Z","usd":"4.920000"}],"key_limit":{"limit_usd":"5.000000","reset":"daily","spent_usd":"1.200000","reserved_usd":"0.020000","remaining_usd":"3.780000","resets_at":"2026-10-01T00:00:00.000Z"}}}}},"401":{"description":"invalid_api_key","content":{"application/json":{"schema":{"$ref":"#/components/schemas/OpenAIError"},"examples":{"invalid_api_key":{"summary":"Bad or missing key","value":{"error":{"message":"Missing or malformed API key. Send your binf_ key as `Authorization: Bearer <key>` or `x-api-key`.","type":"authentication_error","code":"invalid_api_key","param":null}}}}}}},"402":{"description":"insufficient_balance, key_limit","content":{"application/json":{"schema":{"$ref":"#/components/schemas/OpenAIError"},"examples":{"insufficient_balance":{"summary":"Balance too low","value":{"error":{"message":"Balance $0.004210 ($0.001000 held by calls running or being priced) is below this request's reserve of $0.019660. Lower the output limit or add credit.","type":"billing_error","code":"insufficient_balance","param":null}}},"key_limit":{"summary":"Key's spending limit reached","value":{"error":{"message":"This key's limit of $5.000000 a day is used up: $4.990000 spent and $0.000000 held by calls running or being priced, and this request reserves $0.019660. It resets at 2026-10-01T00:00:00.000Z. Lower the output limit, or ask the agent's owner to raise the key's limit.","type":"billing_error","code":"key_limit","param":null}}}}}}},"403":{"description":"agent_inactive, agent_suspended","content":{"application/json":{"schema":{"$ref":"#/components/schemas/OpenAIError"},"examples":{"agent_inactive":{"summary":"Agent not active yet","value":{"error":{"message":"This agent is not active yet. Activate it on its page to use the gateway.","type":"permission_error","code":"agent_inactive","param":null}}},"agent_suspended":{"summary":"Agent suspended","value":{"error":{"message":"This agent is suspended.","type":"permission_error","code":"agent_suspended","param":null}}}}}}},"429":{"description":"rate_limited, too_many_running_calls, daily_share_used","headers":{"x-binference-call-id":{"description":"This call's id on your Calls page. Quote it when you contact us.","schema":{"type":"string"}},"x-generation-id":{"description":"The model provider's id for the answer.","schema":{"type":"string"}},"retry-after":{"description":"On a 429 or 503: whole seconds to wait before sending again.","schema":{"type":"integer"}},"retry-after-ms":{"description":"The same wait in milliseconds. The OpenAI and Anthropic SDKs read it.","schema":{"type":"integer"}}},"content":{"application/json":{"schema":{"$ref":"#/components/schemas/OpenAIError"},"examples":{"rate_limited":{"summary":"Too fast","value":{"error":{"message":"This agent is starting calls faster than 600 a minute. Retry in 1 second.","type":"rate_limit_error","code":"rate_limited","param":null}}},"too_many_running_calls":{"summary":"8 calls already running","value":{"error":{"message":"This agent already has 8 calls running. Retry when one finishes.","type":"rate_limit_error","code":"too_many_running_calls","param":null}}},"daily_share_used":{"summary":"Fair share of a busy day used","value":{"error":{"message":"Today's shared AI capacity is running low, and this agent has used its fair share of it ($12.000000 of $12.000000). It can start calls again at 00:00 UTC, or sooner if capacity is added.","type":"rate_limit_error","code":"daily_share_used","param":null}}}}}}},"500":{"description":"gateway_error","content":{"application/json":{"schema":{"$ref":"#/components/schemas/OpenAIError"},"examples":{"gateway_error":{"summary":"Gateway failed","value":{"error":{"message":"The gateway failed on this call. Retry it.","type":"server_error","code":"gateway_error","param":null}}}}}}},"503":{"description":"model_busy, gateway_busy, gateway_capacity, gateway_unavailable, catalog_unavailable, service_unavailable","headers":{"x-binference-call-id":{"description":"This call's id on your Calls page. Quote it when you contact us.","schema":{"type":"string"}},"x-generation-id":{"description":"The model provider's id for the answer.","schema":{"type":"string"}},"retry-after":{"description":"On a 429 or 503: whole seconds to wait before sending again.","schema":{"type":"integer"}},"retry-after-ms":{"description":"The same wait in milliseconds. The OpenAI and Anthropic SDKs read it.","schema":{"type":"integer"}}},"content":{"application/json":{"schema":{"$ref":"#/components/schemas/OpenAIError"},"examples":{"model_busy":{"summary":"Model busy","value":{"error":{"message":"The model `openai/gpt-6.1` is busy at its providers right now. Retry in 12 seconds.","type":"server_error","code":"model_busy","param":null}}},"gateway_busy":{"summary":"Gateway busy","value":{"error":{"message":"The gateway is busy right now. Retry in 8 seconds.","type":"server_error","code":"gateway_busy","param":null}}},"gateway_capacity":{"summary":"Out of capacity briefly","value":{"error":{"message":"The gateway is briefly out of capacity. Try again soon.","type":"server_error","code":"gateway_capacity","param":null}}},"gateway_unavailable":{"summary":"Gateway unavailable briefly","value":{"error":{"message":"The gateway is briefly unavailable. Try again soon.","type":"server_error","code":"gateway_unavailable","param":null}}},"catalog_unavailable":{"summary":"Prices unavailable","value":{"error":{"message":"Model prices are unavailable. Try again shortly.","type":"server_error","code":"catalog_unavailable","param":null}}},"service_unavailable":{"summary":"Database hiccup","value":{"error":{"message":"The balance is briefly unavailable. Try again in a moment.","type":"server_error","code":"service_unavailable","param":null}}}}}}}}}}}}