Sail Research Responses API API

OpenAI-compatible Responses API endpoints.

OpenAPI Specification

sail-research-responses-api-api-openapi.yml Raw ↑
openapi: 3.1.0
info:
  title: Sail Batches API Responses API API
  version: '2026-02-18'
  description: Sail provides OpenAI-compatible Responses and Chat Completions endpoints, plus an Anthropic-compatible Messages endpoint. This reference documents the currently supported subset of fields.
servers:
- url: https://api.sailresearch.com/v1
security:
- BearerAuth: []
tags:
- name: Responses API
  description: OpenAI-compatible Responses API endpoints.
paths:
  /responses:
    post:
      operationId: createResponse
      tags:
      - Responses API
      summary: Create a response
      description: Creates an OpenAI Responses API task. Returns 202 when background=true, otherwise returns 200 after completion. Foreground stream=true requests return OpenAI Responses Server-Sent Events.
      parameters:
      - $ref: '#/components/parameters/IdempotencyKey'
      requestBody:
        required: true
        content:
          application/json:
            schema:
              $ref: '#/components/schemas/CreateResponseRequest'
            examples:
              basic:
                value:
                  model: zai-org/GLM-5.2-FP8
                  input: Explain the key ideas behind transformer architectures.
              background:
                value:
                  model: zai-org/GLM-5.2-FP8
                  input: Summarize this document.
                  background: true
                  metadata:
                    completion_window: standard
      responses:
        '200':
          description: Response completed and returned synchronously. When stream=true, returns Server-Sent Events.
          content:
            application/json:
              schema:
                $ref: '#/components/schemas/ResponseObject'
            text/event-stream:
              schema:
                type: object
                additionalProperties: true
        '202':
          description: Response accepted for asynchronous processing.
          content:
            application/json:
              schema:
                $ref: '#/components/schemas/ResponseObject'
        '400':
          description: Invalid request or unsupported feature.
          content:
            application/json:
              schema:
                $ref: '#/components/schemas/ErrorResponse'
        '401':
          description: Authentication error.
          content:
            application/json:
              schema:
                $ref: '#/components/schemas/ErrorResponse'
        '500':
          description: Server error.
          content:
            application/json:
              schema:
                $ref: '#/components/schemas/ErrorResponse'
        '502':
          description: Upstream proxy error.
          content:
            application/json:
              schema:
                $ref: '#/components/schemas/ErrorResponse'
        '504':
          description: Timed out waiting for completion.
          content:
            application/json:
              schema:
                $ref: '#/components/schemas/ErrorResponse'
  /responses/{response_id}:
    get:
      operationId: getResponse
      tags:
      - Responses API
      summary: Retrieve a response
      parameters:
      - name: response_id
        in: path
        required: true
        schema:
          type: string
      responses:
        '200':
          description: The response object.
          content:
            application/json:
              schema:
                $ref: '#/components/schemas/ResponseObject'
        '401':
          description: Authentication error.
          content:
            application/json:
              schema:
                $ref: '#/components/schemas/ErrorResponse'
        '404':
          description: Response not found.
          content:
            application/json:
              schema:
                $ref: '#/components/schemas/ErrorResponse'
        '500':
          description: Server error.
          content:
            application/json:
              schema:
                $ref: '#/components/schemas/ErrorResponse'
components:
  schemas:
    ResponseInputMessage:
      type: object
      required:
      - role
      - content
      properties:
        role:
          type: string
          enum:
          - system
          - user
          - assistant
          - tool
          - function
        content:
          oneOf:
          - type: string
          - type: array
            minItems: 1
            items:
              oneOf:
              - $ref: '#/components/schemas/ResponseInputTextPart'
              - $ref: '#/components/schemas/ResponseInputImagePart'
        name:
          type: string
        tool_calls:
          type: array
          items:
            type: object
            additionalProperties: true
        tool_call_id:
          type: string
        function_call:
          type: object
          additionalProperties: true
      additionalProperties: false
    ErrorObject:
      type: object
      required:
      - message
      - type
      properties:
        message:
          type: string
        type:
          type: string
        param:
          oneOf:
          - type: string
          - type: 'null'
        code:
          oneOf:
          - type: string
          - type: integer
          - type: 'null'
      additionalProperties: true
    ResponseInputTextPart:
      type: object
      required:
      - type
      - text
      properties:
        type:
          type: string
          enum:
          - input_text
        text:
          type: string
      additionalProperties: false
    ResponseObject:
      type: object
      required:
      - id
      - object
      - created_at
      - status
      - model
      - metadata
      - usage
      properties:
        id:
          type: string
        object:
          type: string
          enum:
          - response
        created_at:
          type: integer
        status:
          type: string
          enum:
          - pending
          - running
          - failed
          - completed
          - cancelled
        model:
          type: string
        input:
          $ref: '#/components/schemas/ResponseInput'
        output:
          oneOf:
          - type: string
          - type: array
            items:
              type: object
              additionalProperties: true
          - type: object
            additionalProperties: true
          - type: 'null'
        error:
          oneOf:
          - type: object
            additionalProperties: true
          - type: 'null'
        incomplete_details:
          oneOf:
          - type: object
            additionalProperties: true
          - type: 'null'
        max_output_tokens:
          oneOf:
          - type: integer
          - type: 'null'
        reasoning:
          type: object
          additionalProperties: true
        text:
          $ref: '#/components/schemas/ResponseTextConfiguration'
        store:
          type: boolean
        temperature:
          type: number
        top_p:
          type: number
        parallel_tool_calls:
          type: boolean
        tool_choice:
          oneOf:
          - type: string
          - type: object
            additionalProperties: true
        tools:
          type: array
          items:
            type: object
            additionalProperties: true
        truncation:
          oneOf:
          - type: string
          - type: object
            additionalProperties: true
        usage:
          $ref: '#/components/schemas/ResponseUsage'
        user:
          oneOf:
          - type: string
          - type: 'null'
        metadata:
          type: object
          additionalProperties: true
      additionalProperties: true
    ResponseTextFormat:
      type: object
      required:
      - type
      properties:
        type:
          type: string
          enum:
          - text
      additionalProperties: false
    ResponseUsage:
      type: object
      required:
      - input_tokens
      - input_tokens_details
      - output_tokens
      - output_tokens_details
      - total_tokens
      properties:
        input_tokens:
          type: integer
        input_tokens_details:
          $ref: '#/components/schemas/ResponseUsageDetails'
        output_tokens:
          type: integer
        output_tokens_details:
          $ref: '#/components/schemas/ResponseUsageDetails'
        total_tokens:
          type: integer
        prompt_tokens:
          type: integer
        completion_tokens:
          type: integer
      additionalProperties: false
    ResponseInput:
      description: Text input, plus image input (input_image) on multimodal models. Audio, files, and item references are not currently supported.
      oneOf:
      - type: string
        minLength: 1
      - type: array
        minItems: 1
        items:
          $ref: '#/components/schemas/ResponseInputMessage'
      - $ref: '#/components/schemas/ResponseInputObject'
    ResponseReasoningConfiguration:
      type: object
      properties:
        effort:
          type: string
          enum:
          - none
          - minimal
          - low
          - medium
          - high
          - xhigh
        generate_summary:
          type: string
          enum:
          - auto
          - concise
          - detailed
      additionalProperties: true
    RequestMetadata:
      type: object
      description: Optional string metadata. completion_window controls scheduling; completion_webhook/webhook_token configure completion webhooks.
      properties:
        completion_window:
          type: string
          description: '[Completion window](/completion-windows) (i.e. latency tier) for the request. Support matrix is available on the [Pricing](/pricing) page. When omitted, defaults to `standard` if that window is supported for the model; otherwise `flex` for async requests when the model supports it, and `asap` in all other cases. See [default behavior](/completion-windows#default-behavior).'
          enum:
          - asap
          - priority
          - standard
          - flex
        completion_webhook:
          type: string
          format: uri
        webhook_token:
          type: string
      additionalProperties:
        type: string
    ResponseTextConfiguration:
      type: object
      required:
      - format
      properties:
        format:
          oneOf:
          - $ref: '#/components/schemas/ResponseTextFormat'
          - $ref: '#/components/schemas/ResponseJsonSchemaFormat'
      additionalProperties: false
    CreateResponseRequest:
      type: object
      required:
      - model
      - input
      properties:
        model:
          type: string
        input:
          $ref: '#/components/schemas/ResponseInput'
        raw_prompt_tokens:
          type: array
          items:
            type: integer
            minimum: 0
            maximum: 2147483647
          minItems: 1
          maxItems: 2000000
          description: 'Sail extension: pre-tokenized prompt as an array of token ids, forwarded to the model verbatim (no chat templating or tokenization). When set, `input` is still required but ignored — pass a placeholder string.'
        max_output_tokens:
          oneOf:
          - type: integer
            minimum: 1
          - type: 'null'
        temperature:
          oneOf:
          - type: number
            minimum: 0
            maximum: 2
          - type: 'null'
        top_p:
          oneOf:
          - type: number
            minimum: 0
            maximum: 1
          - type: 'null'
        text:
          $ref: '#/components/schemas/ResponseTextConfiguration'
        reasoning:
          $ref: '#/components/schemas/ResponseReasoningConfiguration'
        instructions:
          type: string
          description: Prepended to the input as a system message.
        parallel_tool_calls:
          type: boolean
          description: Accepted for OpenAI compatibility. Models decide their own tool-call cadence, so this field has no effect.
        include:
          type: array
          items:
            type: string
          description: Additional data to include. reasoning.encrypted_content is accepted for OpenAI-client compatibility, but reasoning items are returned without encrypted content.
        background:
          type: boolean
        prompt_cache_key:
          type: string
          description: Optional routing hint for prompt-prefix cache locality. Requests with the same key are preferentially routed to maximize cache hit rates.
        store:
          type: boolean
          description: Accepted for OpenAI compatibility. false does not change Sail's normal temporary request/response storage for processing, retries, polling, and idempotency; Customer Data remains governed by Sail's DPA retention and deletion terms.
        truncation:
          type: string
          enum:
          - disabled
        stream:
          type: boolean
          description: When true on a foreground request, returns OpenAI Responses Server-Sent Events. background=true requests cannot be streamed.
        user:
          type: string
          maxLength: 256
        metadata:
          $ref: '#/components/schemas/RequestMetadata'
      additionalProperties: true
    ErrorResponse:
      type: object
      required:
      - error
      properties:
        error:
          $ref: '#/components/schemas/ErrorObject'
      additionalProperties: false
    ResponseJsonSchemaFormat:
      type: object
      required:
      - type
      - name
      properties:
        type:
          type: string
          enum:
          - json_schema
        name:
          type: string
        description:
          type: string
        schema:
          type: object
          additionalProperties: true
        strict:
          type: boolean
      additionalProperties: false
    ResponseInputImagePart:
      type: object
      description: Image content part. Image input is supported only on multimodal models; see the Models page.
      required:
      - type
      - image_url
      properties:
        type:
          type: string
          enum:
          - input_image
        image_url:
          type: string
          description: Public http(s) URL or a base64 data URI (data:<media-type>;base64,<data>).
        detail:
          type: string
          enum:
          - auto
          - low
          - high
      additionalProperties: false
    ResponseUsageDetails:
      type: object
      required:
      - cached_tokens
      - reasoning_tokens
      properties:
        cached_tokens:
          type: integer
        reasoning_tokens:
          type: integer
      additionalProperties: false
    ResponseInputObject:
      type: object
      required:
      - messages
      properties:
        messages:
          type: array
          minItems: 1
          items:
            $ref: '#/components/schemas/ResponseInputMessage'
      additionalProperties: false
  parameters:
    IdempotencyKey:
      name: Idempotency-Key
      in: header
      required: false
      schema:
        type: string
        maxLength: 255
      description: Makes the request retry-safe. Sail stores a reservation keyed by (organization, API key, Idempotency-Key); retrying with the same value returns the previously stored response instead of re-running inference. Keys are capped at 255 characters. See [Idempotent Requests](/idempotency) for full semantics.
  securitySchemes:
    BearerAuth:
      type: http
      scheme: bearer
      bearerFormat: API Key