OpenAPI Specification
openapi: 3.1.0
info:
title: vLLM OpenAI-Compatible Server Audio Tokenize API
version: '1'
description: 'vLLM is a high-throughput open-source inference and serving engine for
LLMs. Running `vllm serve` exposes an OpenAI-compatible REST API plus
vLLM-specific endpoints. Authentication is via a server-startup
`--api-key` flag; clients supply it as a Bearer token in the
Authorization header (matching the OpenAI Python client).
'
contact:
name: vLLM Documentation
url: https://docs.vllm.ai/en/latest/serving/openai_compatible_server.html
license:
name: Apache-2.0
url: https://www.apache.org/licenses/LICENSE-2.0
servers:
- url: http://{host}:{port}
description: Local vLLM server
variables:
host:
default: localhost
port:
default: '8000'
security:
- bearerAuth: []
tags:
- name: Tokenize
description: vLLM-specific tokenize/detokenize utilities
paths:
/tokenize:
post:
tags:
- Tokenize
summary: Encode text to tokens
operationId: tokenize
requestBody:
required: true
content:
application/json:
schema:
type: object
properties:
model:
type: string
prompt:
type: string
responses:
'200':
description: Token IDs
/detokenize:
post:
tags:
- Tokenize
summary: Decode tokens to text
operationId: detokenize
requestBody:
required: true
content:
application/json:
schema:
type: object
properties:
model:
type: string
tokens:
type: array
items:
type: integer
responses:
'200':
description: Decoded text
components:
securitySchemes:
bearerAuth:
type: http
scheme: bearer
description: API key supplied at server startup via --api-key