Portkey Model Pricing API

Model pricing configurations for 2300+ LLMs across 40+ providers

OpenAPI Specification

portkey-model-pricing-api-openapi.yml Raw ↑
openapi: 3.0.0
info:
  title: Portkey Analytics > Graphs Model Pricing API
  description: The Portkey REST API. Please see https://portkey.ai/docs/api-reference for more details.
  version: 2.0.0
  termsOfService: https://portkey.ai/terms
  contact:
    name: Portkey Developer Forum
    url: https://portkey.wiki/community
  license:
    name: MIT
    url: https://github.com/Portkey-AI/portkey-openapi/blob/master/LICENSE
servers:
- url: https://api.portkey.ai/v1
  description: Portkey API Public Endpoint
security:
- Portkey-Key: []
tags:
- name: Model Pricing
  description: Model pricing configurations for 2300+ LLMs across 40+ providers
paths:
  /model-configs/pricing/{provider}/{model}:
    servers:
    - url: https://api.portkey.ai
      description: Portkey Public API (no auth required)
    get:
      summary: Get Model Pricing
      security: []
      description: "Returns pricing configuration for a specific model.\n\n**Note:** Prices are in USD cents per token.\n\n## Supported Providers\n\nopenai, anthropic, google, azure-openai, bedrock, mistral-ai, cohere, \ntogether-ai, groq, deepseek, fireworks-ai, perplexity-ai, anyscale, \ndeepinfra, cerebras, x-ai, and 25+ more.\n\n## Example Response Fields\n\n| Field | Description | Unit |\n|-------|-------------|------|\n| `request_token.price` | Input token cost | cents/token |\n| `response_token.price` | Output token cost | cents/token |\n| `cache_write_input_token.price` | Cache write cost | cents/token |\n| `cache_read_input_token.price` | Cache read cost | cents/token |\n| `additional_units.*` | Provider-specific features | cents/unit |\n"
      operationId: getModelPricing
      tags:
      - Model Pricing
      parameters:
      - name: provider
        in: path
        required: true
        description: 'Provider identifier. Use lowercase with hyphens.


          Examples: `openai`, `anthropic`, `google`, `azure-openai`, `bedrock`, `x-ai`

          '
        schema:
          type: string
          example: openai
      - name: model
        in: path
        required: true
        description: 'Model identifier. Use the exact model name as specified by the provider.


          Examples: `gpt-5`, `gpt-5.2`, `claude-opus-4-5-20251101`, `gemini-3.0-pro`

          '
        schema:
          type: string
          example: gpt-5
      responses:
        '200':
          description: Pricing configuration for the specified model
          content:
            application/json:
              schema:
                $ref: '#/components/schemas/ModelPricingConfig'
              examples:
                openai-gpt4:
                  summary: OpenAI GPT-4
                  value:
                    pay_as_you_go:
                      request_token:
                        price: 0.003
                      response_token:
                        price: 0.006
                    calculate:
                      request:
                        operation: sum
                        operands:
                        - operation: multiply
                          operands:
                          - value: input_tokens
                          - value: rates.request_token
                        - operation: multiply
                          operands:
                          - value: cache_write_tokens
                          - value: rates.cache_write_input_token
                        - operation: multiply
                          operands:
                          - value: cache_read_tokens
                          - value: rates.cache_read_input_token
                      response:
                        operation: multiply
                        operands:
                        - value: output_tokens
                        - value: rates.response_token
                    currency: USD
                openai-gpt4o-with-tools:
                  summary: OpenAI GPT-4o (with additional units)
                  value:
                    pay_as_you_go:
                      request_token:
                        price: 0.00025
                      response_token:
                        price: 0.001
                      cache_write_input_token:
                        price: 0
                      cache_read_input_token:
                        price: 0.000125
                      additional_units:
                        web_search:
                          price: 1
                        file_search:
                          price: 0.25
                    calculate:
                      request:
                        operation: sum
                        operands:
                        - operation: multiply
                          operands:
                          - value: input_tokens
                          - value: rates.request_token
                      response:
                        operation: multiply
                        operands:
                        - value: output_tokens
                        - value: rates.response_token
                    currency: USD
                anthropic-claude:
                  summary: Anthropic Claude 3.5 Sonnet
                  value:
                    pay_as_you_go:
                      request_token:
                        price: 0.0003
                      response_token:
                        price: 0.0015
                      cache_read_input_token:
                        price: 3.0e-05
                      cache_write_input_token:
                        price: 0.000375
                    currency: USD
                google-gemini:
                  summary: Google Gemini 2.5 Pro (with thinking tokens)
                  value:
                    pay_as_you_go:
                      request_token:
                        price: 0.000125
                      response_token:
                        price: 0.001
                      additional_units:
                        thinking_token:
                          price: 0.001
                        web_search:
                          price: 3.5
                        search:
                          price: 3.5
                    currency: USD
        '404':
          description: Model or provider not found
          content:
            application/json:
              schema:
                type: object
                properties:
                  error:
                    type: string
                    example: Model not found
components:
  schemas:
    ModelPricingConfig:
      type: object
      description: Complete pricing configuration for a model
      properties:
        pay_as_you_go:
          $ref: '#/components/schemas/ModelPayAsYouGo'
        calculate:
          $ref: '#/components/schemas/ModelCalculateConfig'
        currency:
          type: string
          enum:
          - USD
          description: Currency code (always USD)
        finetune_config:
          $ref: '#/components/schemas/ModelFinetuneConfig'
    ModelPayAsYouGo:
      type: object
      description: Token-based pricing (all prices in USD cents)
      properties:
        request_token:
          $ref: '#/components/schemas/ModelTokenPrice'
        response_token:
          $ref: '#/components/schemas/ModelTokenPrice'
        cache_write_input_token:
          $ref: '#/components/schemas/ModelTokenPrice'
        cache_read_input_token:
          $ref: '#/components/schemas/ModelTokenPrice'
        request_audio_token:
          $ref: '#/components/schemas/ModelTokenPrice'
        response_audio_token:
          $ref: '#/components/schemas/ModelTokenPrice'
        cache_read_audio_input_token:
          $ref: '#/components/schemas/ModelTokenPrice'
        additional_units:
          type: object
          description: 'Provider-specific additional pricing units.


            Common additional units:

            - `web_search`: Web search tool usage

            - `file_search`: File search tool usage

            - `thinking_token`: Chain-of-thought reasoning tokens (Google)

            - `image_token`: Image generation tokens

            - `video_duration_seconds_*`: Video generation (OpenAI Sora)

            '
          additionalProperties:
            $ref: '#/components/schemas/ModelTokenPrice'
        image:
          $ref: '#/components/schemas/ModelImagePricing'
    ModelCalculateOperation:
      type: object
      description: Mathematical operation for cost calculation
      properties:
        operation:
          type: string
          enum:
          - sum
          - multiply
          description: Operation type
        operands:
          type: array
          description: Operands for the operation
          items:
            oneOf:
            - $ref: '#/components/schemas/ModelCalculateOperation'
            - $ref: '#/components/schemas/ModelValueReference'
    ModelValueReference:
      type: object
      properties:
        value:
          type: string
          description: 'Reference to a value or rate.


            Examples:

            - `input_tokens`: Number of input tokens

            - `output_tokens`: Number of output tokens

            - `rates.request_token`: Request token rate

            - `rates.response_token`: Response token rate

            '
    ModelTokenPrice:
      type: object
      description: Price object (value is in USD cents)
      properties:
        price:
          type: number
          description: 'Price in USD cents per token/unit.


            Example: `0.003` = 0.003 cents/token = $0.03 per 1K tokens

            '
    ModelCalculateConfig:
      type: object
      description: Cost calculation formulas
      properties:
        request:
          $ref: '#/components/schemas/ModelCalculateOperation'
        response:
          $ref: '#/components/schemas/ModelCalculateOperation'
    ModelFinetuneConfig:
      type: object
      description: Fine-tuning pricing configuration
      properties:
        pay_per_token:
          $ref: '#/components/schemas/ModelTokenPrice'
        pay_per_hour:
          $ref: '#/components/schemas/ModelTokenPrice'
    ModelImagePricing:
      type: object
      description: Image generation pricing by quality and size
      additionalProperties:
        type: object
        additionalProperties:
          $ref: '#/components/schemas/ModelTokenPrice'
      example:
        standard:
          1024x1024:
            price: 4
          1024x1792:
            price: 8
        hd:
          1024x1024:
            price: 8
          1024x1792:
            price: 12
  securitySchemes:
    Portkey-Key:
      type: apiKey
      in: header
      name: x-portkey-api-key
    Virtual-Key:
      type: apiKey
      in: header
      name: x-portkey-virtual-key
    Provider-Auth:
      type: http
      scheme: bearer
    Provider-Name:
      type: apiKey
      in: header
      name: x-portkey-provider
    Config:
      type: apiKey
      in: header
      name: x-portkey-config
    Custom-Host:
      type: apiKey
      in: header
      name: x-portkey-custom-host
x-server-groups:
  ControlPlaneServers:
  - url: https://api.portkey.ai/v1
    description: Portkey API Public Endpoint
  - url: SELF_HOSTED_CONTROL_PLANE_URL
    description: Self-Hosted Control Plane URL
  DataPlaneServers:
  - url: https://api.portkey.ai/v1
    description: Portkey API Public Endpoint
  - url: SELF_HOSTED_GATEWAY_URL
    description: Self-Hosted Gateway URL
  PublicServers:
  - url: https://api.portkey.ai
    description: Portkey Public API (no auth required)
x-mint:
  mcp:
    enabled: true
    name: Portkey MCP
    description: Official MCP Server for Portkey Docs & APIs
x-code-samples:
  navigationGroups:
  - id: endpoints
    title: Endpoints
  - id: assistants
    title: Assistants
  - id: legacy
    title: Legacy
  groups:
  - id: audio
    title: Audio
    description: 'Learn how to turn audio into text or text into audio.


      Related guide: [Speech to text](https://platform.openai.com/docs/guides/speech-to-text)

      '
    navigationGroup: endpoints
    sections:
    - type: endpoint
      key: createSpeech
      path: createSpeech
    - type: endpoint
      key: createTranscription
      path: createTranscription
    - type: endpoint
      key: createTranslation
      path: createTranslation
    - type: object
      key: CreateTranscriptionResponseJson
      path: json-object
    - type: object
      key: CreateTranscriptionResponseVerboseJson
      path: verbose-json-object
  - id: chat
    title: Chat
    description: 'Given a list of messages comprising a conversation, the model will return a response.


      Related guide: [Chat Completions](https://platform.openai.com/docs/guides/text-generation)

      '
    navigationGroup: endpoints
    sections:
    - type: endpoint
      key: createChatCompletion
      path: create
    - type: object
      key: CreateChatCompletionResponse
      path: object
    - type: object
      key: CreateChatCompletionStreamResponse
      path: streaming
  - id: realtime
    title: Realtime
    description: 'WebSocket proxy for provider Realtime APIs (`GET` upgrade). Use `wss://` with the same `/v1` data-plane base as other gateway routes.


      Related guide: [OpenAI Realtime API](https://platform.openai.com/docs/guides/realtime)

      '
    navigationGroup: endpoints
    sections:
    - type: endpoint
      key: connectRealtime
      path: connect
  - id: embeddings
    title: Embeddings
    description: 'Get a vector representation of a given input that can be easily consumed by machine learning models and algorithms.


      Related guide: [Embeddings](https://platform.openai.com/docs/guides/embeddings)

      '
    navigationGroup: endpoints
    sections:
    - type: endpoint
      key: createEmbedding
      path: create
    - type: object
      key: Embedding
      path: object
  - id: rerank
    title: Rerank
    description: 'Rerank a list of documents based on their relevance to a query. Reranking improves search results by scoring documents based on semantic relevance rather than keyword matching.


      Supported providers: Cohere, Voyage, Jina, Pinecone, Bedrock, Azure AI.

      '
    navigationGroup: endpoints
    sections:
    - type: endpoint
      key: createRerank
      path: create
    - type: object
      key: CreateRerankResponse
      path: object
  - id: fine-tuning
    title: Fine-tuning
    description: 'Manage fine-tuning jobs to tailor a model to your specific training data.


      Related guide: [Fine-tune models](https://platform.openai.com/docs/guides/fine-tuning)

      '
    navigationGroup: endpoints
    sections:
    - type: endpoint
      key: createFineTuningJob
      path: create
    - type: endpoint
      key: listPaginatedFineTuningJobs
      path: list
    - type: endpoint
      key: listFineTuningEvents
      path: list-events
    - type: endpoint
      key: listFineTuningJobCheckpoints
      path: list-checkpoints
    - type: endpoint
      key: retrieveFineTuningJob
      path: retrieve
    - type: endpoint
      key: cancelFineTuningJob
      path: cancel
    - type: object
      key: FinetuneChatRequestInput
      path: chat-input
    - type: object
      key: FinetuneCompletionRequestInput
      path: completions-input
    - type: object
      key: FineTuningJob
      path: object
    - type: object
      key: FineTuningJobEvent
      path: event-object
    - type: object
      key: FineTuningJobCheckpoint
      path: checkpoint-object
  - id: batch
    title: Batch
    description: 'Create large batches of API requests for asynchronous processing. The Batch API returns completions within 24 hours for a 50% discount.


      Related guide: [Batch](https://platform.openai.com/docs/guides/batch)

      '
    navigationGroup: endpoints
    sections:
    - type: endpoint
      key: createBatch
      path: create
    - type: endpoint
      key: retrieveBatch
      path: retrieve
    - type: endpoint
      key: cancelBatch
      path: cancel
    - type: endpoint
      key: listBatches
      path: list
    - type: object
      key: Batch
      path: object
    - type: object
      key: BatchRequestInput
      path: request-input
    - type: object
      key: BatchRequestOutput
      path: request-output
  - id: files
    title: Files
    description: 'Files are used to upload documents that can be used with features like [Assistants](https://platform.openai.com/docs/api-reference/assistants), [Fine-tuning](https://platform.openai.com/docs/api-reference/fine-tuning), and [Batch API](https://platform.openai.com/docs/guides/batch).

      '
    navigationGroup: endpoints
    sections:
    - type: endpoint
      key: createFile
      path: create
    - type: endpoint
      key: listFiles
      path: list
    - type: endpoint
      key: retrieveFile
      path: retrieve
    - type: endpoint
      key: deleteFile
      path: delete
    - type: endpoint
      key: downloadFile
      path: retrieve-contents
    - type: object
      key: OpenAIFile
      path: object
  - id: images
    title: Images
    description: 'Given a prompt and/or an input image, the model will generate a new image.


      Related guide: [Image generation](https://platform.openai.com/docs/guides/images)

      '
    navigationGroup: endpoints
    sections:
    - type: endpoint
      key: createImage
      path: create
    - type: endpoint
      key: createImageEdit
      path: createEdit
    - type: endpoint
      key: createImageVariation
      path: createVariation
    - type: object
      key: Image
      path: object
  - id: models
    title: Models
    description: 'List and describe the various models available in the API. You can refer to the [Models](https://platform.openai.com/docs/models) documentation to understand what models are available and the differences between them.

      '
    navigationGroup: endpoints
    sections:
    - type: endpoint
      key: listModels
      path: list
    - type: endpoint
      key: retrieveModel
      path: retrieve
    - type: endpoint
      key: deleteModel
      path: delete
    - type: object
      key: Model
      path: object
  - id: moderations
    title: Moderations
    description: 'Given some input text, outputs if the model classifies it as potentially harmful across several categories.


      Related guide: [Moderations](https://platform.openai.com/docs/guides/moderation)

      '
    navigationGroup: endpoints
    sections:
    - type: endpoint
      key: createModeration
      path: create
    - type: object
      key: CreateModerationResponse
      path: object
  - id: assistants
    title: Assistants
    beta: true
    description: 'Build assistants that can call models and use tools to perform tasks.


      [Get started with the Assistants API](https://platform.openai.com/docs/assistants)

      '
    navigationGroup: assistants
    sections:
    - type: endpoint
      key: createAssistant
      path: createAssistant
    - type: endpoint
      key: listAssistants
      path: listAssistants
    - type: endpoint
      key: getAssistant
      path: getAssistant
    - type: endpoint
      key: modifyAssistant
      path: modifyAssistant
    - type: endpoint
      key: deleteAssistant
      path: deleteAssistant
    - type: object
      key: AssistantObject
      path: object
  - id: threads
    title: Threads
    beta: true
    description: 'Create threads that assistants can interact with.


      Related guide: [Assistants](https://platform.openai.com/docs/assistants/overview)

      '
    navigationGroup: assistants
    sections:
    - type: endpoint
      key: createThread
      path: createThread
    - type: endpoint
      key: getThread
      path: getThread
    - type: endpoint
      key: modifyThread
      path: modifyThread
    - type: endpoint
      key: deleteThread
      path: deleteThread
    - type: object
      key: ThreadObject
      path: object
  - id: messages
    title: Messages
    beta: true
    description: 'Create messages within threads


      Related guide: [Assistants](https://platform.openai.com/docs/assistants/overview)

      '
    navigationGroup: assistants
    sections:
    - type: endpoint
      key: createMessage
      path: createMessage
    - type: endpoint
      key: listMessages
      path: listMessages
    - type: endpoint
      key: getMessage
      path: getMessage
    - type: endpoint
      key: modifyMessage
      path: modifyMessage
    - type: endpoint
      key: deleteMessage
      path: deleteMessage
    - type: object
      key: MessageObject
      path: object
  - id: runs
    title: Runs
    beta: true
    description: 'Represents an execution run on a thread.


      Related guide: [Assistants](https://platform.openai.com/docs/assistants/overview)

      '
    navigationGroup: assistants
    sections:
    - type: endpoint
      key: createRun
      path: createRun
    - type: endpoint
      key: createThreadAndRun
      path: createThreadAndRun
    - type: endpoint
      key: listRuns
      path: listRuns
    - type: endpoint
      key: getRun
      path: getRun
    - type: endpoint
      key: modifyRun
      path: modifyRun
    - type: endpoint
      key: submitToolOuputsToRun
      path: submitToolOutputs
    - type: endpoint
      key: cancelRun
      path: cancelRun
    - type: object
      key: RunObject
      path: object
  - id: run-steps
    title: Run Steps
    beta: true
    description: 'Represents the steps (model and tool calls) taken during the run.


      Related guide: [Assistants](https://platform.openai.com/docs/assistants/overview)

      '
    navigationGroup: assistants
    sections:
    - type: endpoint
      key: listRunSteps
      path: listRunSteps
    - type: endpoint
      key: getRunStep
      path: getRunStep
    - type: object
      key: RunStepObject
      path: step-object
  - id: vector-stores
    title: Vector Stores
    beta: true
    description: 'Vector stores are used to store files for use by the `file_search` tool.


      Related guide: [File Search](https://platform.openai.com/docs/assistants/tools/file-search)

      '
    navigationGroup: assistants
    sections:
    - type: endpoint
      key: createVectorStore
      path: create
    - type: endpoint
      key: listVectorStores
      path: list
    - type: endpoint
      key: getVectorStore
      path: retrieve
    - type: endpoint
      key: modifyVectorStore
      path: modify
    - type: endpoint
      key: deleteVectorStore
      path: delete
    - type: object
      key: VectorStoreObject
      path: object
  - id: vector-stores-files
    title: Vector Store Files
    beta: true
    description: 'Vector store files represent files inside a vector store.


      Related guide: [File Search](https://platform.openai.com/docs/assistants/tools/file-search)

      '
    navigationGroup: assistants
    sections:
    - type: endpoint
      key: createVectorStoreFile
      path: createFile
    - type: endpoint
      key: listVectorStoreFiles
      path: listFiles
    - type: endpoint
      key: getVectorStoreFile
      path: getFile
    - type: endpoint
      key: deleteVectorStoreFile
      path: deleteFile
    - type: object
      key: VectorStoreFileObject
      path: file-object
  - id: vector-stores-file-batches
    title: Vector Store File Batches
    beta: true
    description: 'Vector store file batches represent operations to add multiple files to a vector store.


      Related guide: [File Search](https://platform.openai.com/docs/assistants/tools/file-search)

      '
    navigationGroup: assistants
    sections:
    - type: endpoint
      key: createVectorStoreFileBatch
      path: createBatch
    - type: endpoint
      key: getVectorStoreFileBatch
      path: getBatch
    - type: endpoint
      key: cancelVectorStoreFileBatch
      path: cancelBatch
    - type: endpoint
      key: listFilesInVectorStoreBatch
      path: listBatchFiles
    - type: object
      key: VectorStoreFileBatchObject
      path: batch-object
  - id: assistants-streaming
    title: Streaming
    beta: true
    description: 'Stream the result of executing a Run or resuming a Run after submitting tool outputs.


      You can stream events from the [Create Thread and Run](https://platform.openai.com/docs/api-reference/runs/createThreadAndRun),

      [Create Run](https://platform.openai.com/docs/api-reference/runs/createRun), and [Submit Tool Outputs](https://platform.openai.com/docs/api-reference/runs/submitToolOutputs)

      endpoints by passing `"stream": true`. The response will be a [Server-Sent events](https://html.spec.whatwg.org/multipage/server-sent-events.html#server-sent-events) stream.


      Our Node and Python SDKs provide helpful utilities to make streaming easy. Reference the

      [Assistants API quickstart](https://platform.openai.com/docs/assistants/overview) to learn more.

      '
    navigationGroup: assistants
    sections:
    - type: object
      key: MessageDeltaObject
      path: message-delta-object
    - type: object
      key: RunStepDeltaObject
      path: run-step-delta-object
    - type: object
      key: AssistantStreamEvent
      path: events
  - id: completions
    title: Completions
    legacy: true
    navigationGroup: legacy
    description: 'Given a prompt, the model will return one or more predicted completions along with the probabilities of alternative tokens at each position. Most developer should use our [Chat Completions API](https://platform.openai.com/docs/guides/text-generation/text-generation-models) to leverage our best and newest models.

      '
    sections:
    - type: endpoint
      key: createCompletion
      path: create
    - type: object
      key: CreateCompletionResponse
      path: object