openapi: 3.1.0
info:
title: Inflection Inference Chat API
version: 1.0.0
tags:
- name: Chat
paths:
/v1/chat/completions:
post:
summary: Chat Completions
description: 'Chat completions endpoint for Inflection hosted models.
⚠️ This endpoint is under active development to enable all Inflection model and config types.'
operationId: chat_completions_v1_chat_completions_post
security:
- OAuth2PasswordBearer: []
requestBody:
required: true
content:
application/json:
schema:
$ref: '#/components/schemas/ChatCompletionRequest'
example:
model: inflection_3_pi
messages:
- role: system
content: You are a helpful assistant.
- role: user
content: What is the weather like today?
max_tokens: 150
temperature: 0.7
responses:
'200':
description: Returns either the full response (`application/json`) or streams multiple `ChatCompletionStreamResponse` objects via `text/event-stream`. Each event in the stream is a JSON object matching the ChatCompletionStreamResponse schema, and each event is prefixed by `data:` (typical for Server-Sent Events).
content:
application/json:
schema:
$ref: '#/components/schemas/ChatCompletionResponse'
example:
id: cmpl-5y3yYz9d7Gv0yWw8
object: chat.completion
created: 1627290155
model: inflection_3_pi
choices:
- message:
role: assistant
content: The weather today is sunny, with a high of 25°C.
finish_reason: stop
usage:
prompt_tokens: 13
completion_tokens: 10
total_tokens: 23
text/event-stream:
schema:
$ref: '#/components/schemas/ChatCompletionStreamResponse'
example: "data: {\n \"id\": \"chatcmpl-abc123\",\n \"object\": \"chat.completion.chunk\",\n \"created\": 1627290155,\n \"model\": \"inflection_3_pi\",\n \"choices\": [\n {\n \"delta\": {\n \"role\": \"assistant\",\n \"content\": \"The weather today is sunny,\"\n },\n \"finish_reason\": null\n }\n ],\n}\n\ndata: {\n \"id\": \"chatcmpl-abc123\",\n \"object\": \"chat.completion.chunk\",\n \"created\": 1627290155,\n \"model\": \"inflection_3_pi\",\n \"choices\": [\n {\n \"delta\": {\n \"role\": \"assistant\",\n \"content\": \" with a high of 25°C.\"\n },\n \"finish_reason\": \"stop\"\n }\n ],\n}\n"
'422':
description: Validation Error
content:
application/json:
schema:
$ref: '#/components/schemas/HTTPValidationError'
x-code-samples:
- lang: curl Example
source: "curl --url https://api.inflection.ai/v1/chat/completions \\\n -X POST \\\n -H \"Content-Type: application/json\" \\\n -H \"Authorization: Bearer YOUR_API_KEY\" \\\n -d '{\n \"model\": \"inflection_3_pi\",\n \"messages\": [\n {\"role\": \"system\", \"content\": \"You are a helpful assistant.\"},\n {\"role\": \"user\", \"content\": \"What is the weather like today?\"}\n ],\n \"max_tokens\": 150,\n \"temperature\": 0.7\n }'"
- lang: curl Example Streaming
source: "curl --url https://api.inflection.ai/v1/chat/completions \\\n -X POST \\\n -H \"Content-Type: application/json\" \\\n -H \"Authorization: Bearer YOUR_API_KEY\" \\\n -d '{\n \"model\": \"inflection_3_pi\",\n \"messages\": [\n {\"role\": \"system\", \"content\": \"You are a helpful assistant.\"},\n {\"role\": \"user\", \"content\": \"What is the weather like today?\"}\n ],\n \"stream\": true\n }' --no-buffer"
tags:
- Chat
/v1/chat/attributes:
post:
summary: Chat Attributes
description: 'Get attribute scores for the most recent assistant message in a chat completions style request.
⚠️ Experimental endpoint for evaluating message attributes.'
operationId: chat_attributes_v1_chat_attributes_post
security:
- OAuth2PasswordBearer: []
requestBody:
required: true
content:
application/json:
schema:
$ref: '#/components/schemas/AttributesRequest'
example:
model: Pi-3.1
messages:
- role: system
content: You are a helpful assistant.
- role: user
content: Tell me about quantum computing.
- role: assistant
content: No way.
attributes:
- name: Helpful
definition: The assistant is helpful and provides useful information.
categories:
- Assistant
- name: Rude
definition: The assistant uses mean or offensive language or is otherwise rude.
categories:
- Assistant
filter_categories:
- Assistant
description: '
Follows the same schema as the chat completions endpoint, but with the addition of:
- `attributes`: An optional list of attributes to score. If no attributes are provided, a set of default attributes is used.
- `filter_categories`: An optional list of categories to filter the attributes by. If no categories are provided, all attributes are used.
- `disable_attributes_prompt`: An optional boolean to disable the attributes default prompt. If `true`, the attributes default prompt will not be added to the system prompt.
'
responses:
'200':
description: '
Returns attribute scores for the final assistant message. If no attributes are provided, a set of default attributes is used.
Scores are provided in JSON format as `{"AttributeName": {"score": float, "positive": bool}, ...}` in the first Choice message content.
'
content:
application/json:
schema:
$ref: '#/components/schemas/ChatCompletionResponse'
example:
id: chatcmpl-xyz789
object: chat.completion
created: 1627290155
model: Pi-3.1
choices:
- message:
role: assistant
content: '{"Helpful": {"score": 0.1, "positive": true}, "Rude": {"score": 0.9, "positive": false}}'
finish_reason: stop
usage:
prompt_tokens: 135
completion_tokens: 20
total_tokens: 155
'422':
description: Validation Error
content:
application/json:
schema:
$ref: '#/components/schemas/HTTPValidationError'
x-code-samples:
- lang: curl Example
source: "curl --url https://api.inflection.ai/v1/chat/attributes \\\n -X POST \\\n -H \"Content-Type: application/json\" \\\n -H \"Authorization: Bearer YOUR_API_KEY\" \\\n -d '{\n \"model\": \"Pi-3.1\",\n \"messages\": [\n {\"role\": \"system\", \"content\": \"You are a helpful assistant.\"},\n {\"role\": \"user\", \"content\": \"Tell me about quantum computing.\"},\n {\"role\": \"assistant\", \"content\": \"No way.\"}\n ],\n \"attributes\": [\n {\n \"name\": \"Helpful\",\n \"definition\": \"The assistant is helpful and provides useful information.\",\n \"categories\": [\"Assistant\"]\n },\n {\n \"name\": \"Rude\",\n \"definition\": \"The assistant uses mean or offensive language or is otherwise rude.\",\n \"categories\": [\"Assistant\"]\n }\n ],\n \"filter_categories\": [\"Assistant\"]\n }'"
tags:
- Chat
components:
schemas:
ChatCompletionContentPartInputAudioParam:
properties:
input_audio:
$ref: '#/components/schemas/InputAudio'
type:
type: string
enum:
- input_audio
const: input_audio
title: Type
type: object
required:
- input_audio
- type
title: ChatCompletionContentPartInputAudioParam
PromptTokenUsageInfo:
properties:
cached_tokens:
anyOf:
- type: integer
- type: 'null'
title: Cached Tokens
additionalProperties: true
type: object
title: PromptTokenUsageInfo
Audio:
properties:
id:
type: string
title: Id
type: object
required:
- id
title: Audio
File:
properties:
file:
$ref: '#/components/schemas/FileFile'
type:
type: string
enum:
- file
const: file
title: Type
type: object
required:
- file
- type
title: File
ChatCompletionResponse:
properties:
id:
type: string
title: Id
object:
type: string
enum:
- chat.completion
const: chat.completion
title: Object
default: chat.completion
created:
type: integer
title: Created
model:
type: string
title: Model
choices:
items:
$ref: '#/components/schemas/ChatCompletionResponseChoice'
type: array
title: Choices
usage:
$ref: '#/components/schemas/UsageInfo'
prompt_logprobs:
anyOf:
- items:
anyOf:
- additionalProperties:
$ref: '#/components/schemas/Logprob'
type: object
- type: 'null'
type: array
- type: 'null'
title: Prompt Logprobs
kv_transfer_params:
anyOf:
- type: object
- type: 'null'
title: Kv Transfer Params
description: KVTransfer parameters.
system_fingerprint:
anyOf:
- type: string
- type: 'null'
title: System Fingerprint
time_info:
anyOf:
- type: object
- type: 'null'
title: Time Info
additionalProperties: true
type: object
required:
- model
- choices
- usage
title: ChatCompletionResponse
VideoURL:
properties:
url:
type: string
title: Url
type: object
required:
- url
title: VideoURL
ChatCompletionNamedToolChoiceParam:
properties:
function:
$ref: '#/components/schemas/ChatCompletionNamedFunction'
type:
type: string
enum:
- function
const: function
title: Type
default: function
additionalProperties: true
type: object
required:
- function
title: ChatCompletionNamedToolChoiceParam
ChatCompletionContentPartImageParam:
properties:
image_url:
$ref: '#/components/schemas/ImageURL'
type:
type: string
enum:
- image_url
const: image_url
title: Type
type: object
required:
- image_url
- type
title: ChatCompletionContentPartImageParam
AttributesRequest:
properties:
messages:
items:
anyOf:
- $ref: '#/components/schemas/ChatCompletionDeveloperMessageParam'
- $ref: '#/components/schemas/ChatCompletionSystemMessageParam'
- $ref: '#/components/schemas/ChatCompletionUserMessageParam'
- $ref: '#/components/schemas/ChatCompletionAssistantMessageParam'
- $ref: '#/components/schemas/ChatCompletionToolMessageParam'
- $ref: '#/components/schemas/ChatCompletionFunctionMessageParam'
- $ref: '#/components/schemas/CustomChatCompletionMessageParam'
type: array
title: Messages
model:
anyOf:
- type: string
- type: 'null'
title: Model
frequency_penalty:
anyOf:
- type: number
- type: 'null'
title: Frequency Penalty
default: 0.0
logit_bias:
anyOf:
- additionalProperties:
type: number
type: object
- type: 'null'
title: Logit Bias
logprobs:
anyOf:
- type: boolean
- type: 'null'
title: Logprobs
default: false
top_logprobs:
anyOf:
- type: integer
- type: 'null'
title: Top Logprobs
default: 0
max_tokens:
anyOf:
- type: integer
- type: 'null'
title: Max Tokens
deprecated: true
max_completion_tokens:
anyOf:
- type: integer
- type: 'null'
title: Max Completion Tokens
n:
anyOf:
- type: integer
- type: 'null'
title: N
default: 1
presence_penalty:
anyOf:
- type: number
- type: 'null'
title: Presence Penalty
default: 0.0
response_format:
anyOf:
- $ref: '#/components/schemas/ResponseFormat'
- $ref: '#/components/schemas/StructuralTagResponseFormat'
- type: 'null'
title: Response Format
seed:
anyOf:
- type: integer
maximum: 4294967295.0
minimum: 0.0
- type: 'null'
title: Seed
stop:
anyOf:
- type: string
- items:
type: string
type: array
- type: 'null'
title: Stop
default: []
stream:
anyOf:
- type: boolean
- type: 'null'
title: Stream
default: false
stream_options:
anyOf:
- $ref: '#/components/schemas/StreamOptions'
- type: 'null'
temperature:
anyOf:
- type: number
- type: 'null'
title: Temperature
top_p:
anyOf:
- type: number
- type: 'null'
title: Top P
tools:
anyOf:
- items:
$ref: '#/components/schemas/ChatCompletionToolsParam'
type: array
- type: 'null'
title: Tools
tool_choice:
anyOf:
- type: string
enum:
- none
const: none
- type: string
enum:
- auto
const: auto
- type: string
enum:
- required
const: required
- $ref: '#/components/schemas/ChatCompletionNamedToolChoiceParam'
- type: 'null'
title: Tool Choice
default: none
parallel_tool_calls:
anyOf:
- type: boolean
- type: 'null'
title: Parallel Tool Calls
default: false
user:
anyOf:
- type: string
- type: 'null'
title: User
best_of:
anyOf:
- type: integer
- type: 'null'
title: Best Of
use_beam_search:
type: boolean
title: Use Beam Search
default: false
top_k:
anyOf:
- type: integer
- type: 'null'
title: Top K
min_p:
anyOf:
- type: number
- type: 'null'
title: Min P
repetition_penalty:
anyOf:
- type: number
- type: 'null'
title: Repetition Penalty
length_penalty:
type: number
title: Length Penalty
default: 1.0
stop_token_ids:
anyOf:
- items:
type: integer
type: array
- type: 'null'
title: Stop Token Ids
default: []
include_stop_str_in_output:
type: boolean
title: Include Stop Str In Output
default: false
ignore_eos:
type: boolean
title: Ignore Eos
default: false
min_tokens:
type: integer
title: Min Tokens
default: 0
skip_special_tokens:
type: boolean
title: Skip Special Tokens
default: true
spaces_between_special_tokens:
type: boolean
title: Spaces Between Special Tokens
default: true
truncate_prompt_tokens:
anyOf:
- type: integer
minimum: 1.0
- type: 'null'
title: Truncate Prompt Tokens
prompt_logprobs:
anyOf:
- type: integer
- type: 'null'
title: Prompt Logprobs
allowed_token_ids:
anyOf:
- items:
type: integer
type: array
- type: 'null'
title: Allowed Token Ids
bad_words:
items:
type: string
type: array
title: Bad Words
echo:
type: boolean
title: Echo
description: If true, the new message will be prepended with the last message if they belong to the same role.
default: false
add_generation_prompt:
type: boolean
title: Add Generation Prompt
description: If true, the generation prompt will be added to the chat template. This is a parameter used by chat template in tokenizer config of the model.
default: true
continue_final_message:
type: boolean
title: Continue Final Message
description: If this is set, the chat will be formatted so that the final message in the chat is open-ended, without any EOS tokens. The model will continue this message rather than starting a new one. This allows you to "prefill" part of the model's response for it. Cannot be used at the same time as `add_generation_prompt`.
default: false
add_special_tokens:
type: boolean
title: Add Special Tokens
description: If true, special tokens (e.g. BOS) will be added to the prompt on top of what is added by the chat template. For most models, the chat template takes care of adding the special tokens so this should be set to false (as is the default).
default: false
documents:
anyOf:
- items:
additionalProperties:
type: string
type: object
type: array
- type: 'null'
title: Documents
description: A list of dicts representing documents that will be accessible to the model if it is performing RAG (retrieval-augmented generation). If the template does not support RAG, this argument will have no effect. We recommend that each document should be a dict containing "title" and "text" keys.
chat_template:
anyOf:
- type: string
- type: 'null'
title: Chat Template
description: A Jinja template to use for this conversion. As of transformers v4.44, default chat template is no longer allowed, so you must provide a chat template if the tokenizer does not define one.
chat_template_kwargs:
anyOf:
- type: object
- type: 'null'
title: Chat Template Kwargs
description: Additional keyword args to pass to the template renderer. Will be accessible by the chat template.
mm_processor_kwargs:
anyOf:
- type: object
- type: 'null'
title: Mm Processor Kwargs
description: Additional kwargs to pass to the HF processor.
guided_json:
anyOf:
- type: string
- type: object
- $ref: '#/components/schemas/BaseModel'
- type: 'null'
title: Guided Json
description: If specified, the output will follow the JSON schema.
guided_regex:
anyOf:
- type: string
- type: 'null'
title: Guided Regex
description: If specified, the output will follow the regex pattern.
guided_choice:
anyOf:
- items:
type: string
type: array
- type: 'null'
title: Guided Choice
description: If specified, the output will be exactly one of the choices.
guided_grammar:
anyOf:
- type: string
- type: 'null'
title: Guided Grammar
description: If specified, the output will follow the context free grammar.
structural_tag:
anyOf:
- type: string
- type: 'null'
title: Structural Tag
description: If specified, the output will follow the structural tag schema.
guided_decoding_backend:
anyOf:
- type: string
- type: 'null'
title: Guided Decoding Backend
description: If specified, will override the default guided decoding backend of the server for this specific request. If set, must be either 'outlines' / 'lm-format-enforcer'
guided_whitespace_pattern:
anyOf:
- type: string
- type: 'null'
title: Guided Whitespace Pattern
description: If specified, will override the default whitespace pattern for guided json decoding.
guided_meta:
anyOf:
- type: object
- type: 'null'
title: Guided Meta
description: Specify custom arguments for your backend
priority:
type: integer
title: Priority
description: 'The priority of the request (lower means earlier handling; default: 0). Any priority other than 0 will raise an error if the served model does not use priority scheduling.'
default: 0
request_id:
type: string
title: Request Id
description: The request_id related to this request. If the caller does not set it, a random_uuid will be generated. This id is used through out the inference process and return in response.
logits_processors:
anyOf:
- items:
anyOf:
- type: string
- $ref: '#/components/schemas/LogitsProcessorConstructor'
type: array
- type: 'null'
title: Logits Processors
description: 'A list of either qualified names of logits processors, or constructor objects, to apply when sampling. A constructor is a JSON object with a required ''qualname'' field specifying the qualified name of the processor class/factory, and optional ''args'' and ''kwargs'' fields containing positional and keyword arguments. For example: {''qualname'': ''my_module.MyLogitsProcessor'', ''args'': [1, 2], ''kwargs'': {''param'': ''value''}}.'
return_tokens_as_token_ids:
anyOf:
- type: boolean
- type: 'null'
title: Return Tokens As Token Ids
description: If specified with 'logprobs', tokens are represented as strings of the form 'token_id:{token_id}' so that tokens that are not JSON-encodable can be identified.
cache_salt:
anyOf:
- type: string
- type: 'null'
title: Cache Salt
description: If specified, the prefix cache will be salted with the provided string to prevent an attacker to guess prompts in multi-user environments. The salt should be random, protected from access by 3rd parties, and long enough to be unpredictable (e.g., 43 characters base64-encoded, corresponding to 256 bit). Not supported by vLLM engine V0.
kv_transfer_params:
anyOf:
- type: object
- type: 'null'
title: Kv Transfer Params
description: KVTransfer parameters used for disaggregated serving.
vllm_xargs:
anyOf:
- additionalProperties:
anyOf:
- type: string
- type: integer
- type: number
type: object
- type: 'null'
title: Vllm Xargs
description: Additional request parameters with string or numeric values, used by custom extensions.
timeout_sec:
anyOf:
- type: number
- type: 'null'
title: Timeout Sec
attributes:
anyOf:
- items:
$ref: '#/components/schemas/Attribute'
type: array
- type: 'null'
title: Attributes
filter_categories:
anyOf:
- items:
$ref: '#/components/schemas/Category'
type: array
- type: 'null'
title: Filter Categories
disable_attributes_prompt:
anyOf:
- type: boolean
- type: 'null'
title: Disable Attributes Prompt
default: false
additionalProperties: true
type: object
required:
- messages
title: AttributesRequest
StreamOptions:
properties:
include_usage:
anyOf:
- type: boolean
- type: 'null'
title: Include Usage
default: true
continuous_usage_stats:
anyOf:
- type: boolean
- type: 'null'
title: Continuous Usage Stats
default: false
additionalProperties: true
type: object
title: StreamOptions
StructuralTag:
properties:
begin:
type: string
title: Begin
schema:
anyOf:
- type: object
- type: 'null'
title: Schema
end:
type: string
title: End
additionalProperties: true
type: object
required:
- begin
- end
title: StructuralTag
ChatCompletionToolMessageParam:
properties:
content:
anyOf:
- type: string
- items:
$ref: '#/components/schemas/ChatCompletionContentPartTextParam'
type: array
title: Content
role:
type: string
enum:
- tool
const: tool
title: Role
tool_call_id:
type: string
title: Tool Call Id
type: object
required:
- content
- role
- tool_call_id
title: ChatCompletionToolMessageParam
ChatCompletionResponseStreamChoice:
properties:
index:
type: integer
title: Index
delta:
$ref: '#/components/schemas/DeltaMessage'
logprobs:
anyOf:
- $ref: '#/components/schemas/ChatCompletionLogProbs'
- type: 'null'
finish_reason:
anyOf:
- type: string
- type: 'null'
title: Finish Reason
stop_reason:
anyOf:
- type: integer
- type: string
- type: 'null'
title: Stop Reason
additionalProperties: true
type: object
required:
- index
- delta
title: ChatCompletionResponseStreamChoice
ChatCompletionFunctionMessageParam:
properties:
content:
anyOf:
- type: string
- type: 'null'
title: Content
name:
type: string
title: Name
role:
type: string
enum:
- function
const: function
title: Role
type: object
required:
- content
- name
- role
title: ChatCompletionFunctionMessageParam
ChatCompletionSystemMessageParam:
properties:
content:
anyOf:
- type: string
- items:
$ref: '#/components/schemas/ChatCompletionContentPartTextParam'
type: array
title: Content
role:
type: string
enum:
- system
const: system
title: Role
name:
type: string
title: Name
type: object
required:
- content
- role
title: ChatCompletionSystemMessageParam
Custom:
properties:
input:
type: string
title: Input
name:
type: string
title: Name
type: object
required:
- input
- name
title: Custom
DeltaMessage:
properties:
role:
anyOf:
- type: string
- type: 'null'
title: Role
content:
anyOf:
- type: string
- type: 'null'
title: Content
reasoning_content:
anyOf:
- type: string
- type: 'null'
title: Reasoning Content
tool_calls:
items:
$ref: '#/components/schemas/DeltaToolCall'
type: array
title: Tool Calls
additionalProperties: true
type: object
title: DeltaMessage
ChatMessage:
properties:
role:
type: string
title: Role
reasoning_content:
anyOf:
- type: string
- type: 'null'
title: Reasoning Content
content:
anyOf:
- type: string
- type: 'null'
title: Content
tool_calls:
items:
$ref: '#/components/schemas/ToolCall'
type: array
title: Tool Calls
additionalProperties: true
type: object
required:
- role
title: ChatMessage
ChatCompletionContentPartTextParam:
properties:
text:
type: string
title: Text
type:
type: string
enum:
- text
const: text
title: Type
type: object
required:
- text
- type
title: ChatCompletionContentPartTextParam
HTTPValidationError:
properties:
detail:
items:
$ref: '#/components/schemas/ValidationError'
type: array
title: Detail
type: object
title: HTTPValidationError
AudioURL:
properties:
url:
type: string
title: Url
type: object
required:
- url
title: AudioURL
ResponseFormat:
properties:
type:
type: string
enum:
- text
- json_object
- json_schema
title: Type
json_schema:
anyOf:
- $ref: '#/components/schemas/JsonSchemaResponseFormat'
- type: 'null'
additionalProperties: true
type: object
required:
- type
title: ResponseFormat
ChatCompletionNamedFunction:
properties:
name:
type: string
title: Name
additionalProperties: true
type: object
required:
- name
title: ChatCompletionNamedFunction
ChatCompletionUserMessageParam:
properties:
content:
anyOf:
- type: string
- items:
anyOf:
- $ref: '#/components/schemas/ChatCompletionContentPartTextParam'
- $ref: '#/components/schemas/ChatCompletionContentPartImageParam'
- $ref: '#/components/schemas/ChatCompletionContentPartInputAudioParam'
- $ref: '#/components/schemas/File'
type: array
title: Content
role:
type: string
enum:
- user
const: user
title: Role
name:
type: string
title: Name
type: object
required:
- content
- role
title: ChatCompletionUserMessageParam
CustomChatCompletionMessageParam:
properties:
role:
type: string
title: Role
content:
anyOf:
- type: string
- items:
anyOf:
- $ref: '#/components/schemas/ChatCompletionContentPartTextParam'
- $ref: '#/components/schemas/ChatCompletionContentPartImageParam'
- $ref: '#/components/schemas/ChatCompletionContentPartInputAudioParam'
- $ref: '#/components/schemas
# --- truncated at 32 KB (62 KB total) ---
# Full source: https://raw.githubusercontent.com/api-evangelist/inflection/refs/heads/main/openapi/inflection-chat-api-openapi.yml