openapi: 3.2.0
info:
version: latest
description: '# Introduction
The NVIDIA Run:ai Control-Plane API reference is a guide that provides an easy-to-use programming interface for adding various tasks to your application, including workload submission, resource management, and administrative operations.
NVIDIA Run:ai APIs are accessed using *bearer tokens*. To obtain a token, you need to create a **Service account** through the NVIDIA Run:ai user interface.
To create a service account, in your UI, go to Access → Service Accounts (for organization-level service accounts) or User settings → Access Keys (for user access keys), and create a new one.
After you have created a new service account, you will need to assign it access rules.
To assign access rules to the service account, see [Create access rules](https://run-ai-docs.nvidia.com/saas/infrastructure-setup/authentication/accessrules#create-or-delete-rules).
Make sure you assign the correct rules to your service account. Use the [Roles](https://run-ai-docs.nvidia.com/saas/infrastructure-setup/authentication/roles) to assign the correct access rules.
To get your access token, follow the instructions in [Request a token](https://run-ai-docs.nvidia.com/saas/reference/api/rest-auth/#request-an-api-token).
'
title: NVIDIA Run:ai Access Keys NVIDIA NIM API
x-logo:
url: https://api.redocly.com/registry/raw/runai-xq8/saas/latest/public/runai-logo-api.png
altText: NVIDIA Run:ai
href: https://run.ai
license:
name: NVIDIA Run:ai
url: https://www.nvidia.com/en-us/agreements/enterprise-software/nvidia-software-license-agreement/
servers:
- url: https://app.run.ai
security:
- bearerAuth: []
tags:
- name: NVIDIA NIM
description: 'The NVIDIA NIM API provides endpoints to create and manage workloads that deploy NVIDIA Inference Microservices (NIM) through the NIM Operator. These workloads package optimized NVIDIA model servers and run as managed services on the NVIDIA Run:ai platform.
Each request includes NVIDIA Run:ai scheduling metadata (for example, project, priority, and category) and a NIM service specification that defines the container image, compute resources, environment variables, storage, and networking configuration. Once submitted, NVIDIA Run:ai handles scheduling, orchestration, and lifecycle management of the NIM service to ensure reliable and efficient model serving.
'
paths:
/api/v2/workloads/nim-services:
post:
summary: Create a NVIDIA NIM service. [Experimental]
description: Create a NVIDIA NIM service
operationId: create_nim_service
tags:
- NVIDIA NIM
requestBody:
required: true
content:
application/json:
schema:
$ref: '#/components/schemas/NimServiceCreateRequest'
responses:
'202':
description: Workload creation accepted
content:
application/json:
schema:
$ref: '#/components/schemas/NimServiceResponse'
'400':
$ref: '#/components/responses/400SubmissionErrorV2'
'401':
$ref: '#/components/responses/401Unauthorized'
'403':
$ref: '#/components/responses/403Forbidden'
'409':
$ref: '#/components/responses/409Conflict'
'500':
$ref: '#/components/responses/500InternalServerError'
'503':
$ref: '#/components/responses/503ServiceUnavailable'
/api/v2/workloads/nim-services/{WorkloadV2Id}:
get:
summary: Get a NVIDIA NIM service. [Experimental]
description: Retrieve details of a specific NVIDIA NIM service, by id
operationId: get_nim_service_by_id
tags:
- NVIDIA NIM
parameters:
- $ref: '#/components/parameters/WorkloadV2Id'
responses:
'200':
description: Successfully retrieved the workload
content:
application/json:
schema:
$ref: '#/components/schemas/NimServiceResponse'
'401':
$ref: '#/components/responses/401Unauthorized'
'403':
$ref: '#/components/responses/403Forbidden'
'404':
$ref: '#/components/responses/404NotFound'
'500':
$ref: '#/components/responses/500InternalServerError'
'503':
$ref: '#/components/responses/503ServiceUnavailable'
patch:
summary: Update NVIDIA NIM service spec. [Experimental]
operationId: update_nim_service_spec
description: Update the specification of an existing NVIDIA NIM service.
tags:
- NVIDIA NIM
parameters:
- $ref: '#/components/parameters/WorkloadV2Id'
requestBody:
required: true
content:
application/json:
schema:
$ref: '#/components/schemas/NimServiceUpdateRequest'
responses:
'202':
description: Workload update request accepted
content:
application/json:
schema:
$ref: '#/components/schemas/NimServiceResponse'
'401':
$ref: '#/components/responses/401Unauthorized'
'403':
$ref: '#/components/responses/403Forbidden'
'404':
$ref: '#/components/responses/404NotFound'
'500':
$ref: '#/components/responses/500InternalServerError'
'503':
$ref: '#/components/responses/503ServiceUnavailable'
components:
schemas:
NimServiceNgcAuthSecret:
description: The name of a Kubernetes secret containing the NGC access credentials. The secret must contain a key named NGC_API_KEY with the API key as the value.
type:
- string
- 'null'
pattern: ^[a-z0-9]([a-z0-9-]*[a-z0-9])?$
ServingPortMetricsPort:
description: The port where metrics are exposed, required only if it's different than the main port.
type:
- integer
- 'null'
format: int32
minimum: 1
maximum: 65535
example: 8002
WorkloadV2MetadataResponse:
type: object
required:
- name
- projectId
properties:
name:
$ref: '#/components/schemas/WorkloadName'
projectId:
$ref: '#/components/schemas/ProjectId'
priority:
$ref: '#/components/schemas/PriorityClass'
category:
$ref: '#/components/schemas/Category'
preemptibility:
$ref: '#/components/schemas/Preemptibility'
Tolerations:
description: Set of tolerations to apply to the workload.
type:
- array
- 'null'
items:
$ref: '#/components/schemas/Toleration'
ImagePullPolicy:
description: Image pull policy. Defaults to `Always` if `:latest` tag is specified, otherwise it is `IfNotPresent`.
type:
- string
- 'null'
minLength: 1
enum:
- Always
- Never
- IfNotPresent
Label:
description: Label details to be populated into the container.
properties:
name:
description: The name of the label (mandatory)
type:
- string
- 'null'
minLength: 1
maxLength: 63
example: stage
pattern: .*
value:
description: The value of the label.
type:
- string
- 'null'
example: initial-research
pattern: .*
exclude:
description: Use 'true' in case the label is defined in defaults of the policy, and you wish to exclude it from the workload.
type:
- boolean
- 'null'
example: false
type:
- object
- 'null'
ProbeHandler:
description: The action taken to determine the health of the container. (mandatory)
type:
- object
- 'null'
properties:
httpGet:
description: An action based on HTTP Get requests.
type: object
properties:
path:
description: Path to access on the HTTP server, defaults to /.
type:
- string
- 'null'
pattern: ^(\x2F[a-zA-Z0-9\-_.\x2F]*)?$
example: /
port:
description: Number of the port to access on the container.
type:
- integer
- 'null'
format: int32
minimum: 1
maximum: 65535
host:
description: Host name to connect to, defaults to the pod IP.
type:
- string
- 'null'
format: hostname
example: example.com
pattern: .*
scheme:
$ref: '#/components/schemas/ProbeHandlerScheme'
NIMServiceMetadataCreateParams:
type: object
required:
- name
- projectId
properties:
name:
$ref: '#/components/schemas/WorkloadName'
useGivenNameAsPrefix:
description: When true, the requested name will be treated as a prefix. The final name of the workload will be composed of the name followed by a random set of characters.
type: boolean
example: true
default: false
projectId:
$ref: '#/components/schemas/ProjectId'
EnvironmentVariablePodFieldReference:
description: Details of the field-reference and key use to populate the environment variable
properties:
path:
description: The field path resource. (mandatory)
type:
- string
- 'null'
minLength: 1
example: metadata.name
pattern: .*
type:
- object
- 'null'
Labels:
description: Set of labels to populate into the container running the workload.
type:
- array
- 'null'
items:
$ref: '#/components/schemas/Label'
ImagePullSecrets:
description: A list of references to Kubernetes secrets in the same namespace used for pulling container images.
type:
- array
- 'null'
items:
$ref: '#/components/schemas/ImagePullSecret'
CpuMemoryLimit:
description: Limitations on the CPU memory to allocate for this workload (1G, 20M, .etc). The system guarantees that this workload will not be able to consume more than this amount of memory. The workload will receive an error when trying to allocate more memory than this limit.
type:
- string
- 'null'
pattern: ^([+]?[0-9.]+)([eEinumkKMGTP]*[-+]?[0-9]*)$
example: 30M
WorkloadV2MetadataAutoFill:
type: object
required:
- id
- gvk
- projectName
- clusterId
- tenantId
- departmentId
- departmentName
- createdAt
- createdBy
- updatedAt
- updatedBy
properties:
id:
$ref: '#/components/schemas/WorkloadId3'
gvk:
$ref: '#/components/schemas/GVK'
projectName:
$ref: '#/components/schemas/ProjectName2'
clusterId:
$ref: '#/components/schemas/ClusterId'
tenantId:
$ref: '#/components/schemas/TenantId'
departmentId:
$ref: '#/components/schemas/DepartmentId3'
departmentName:
$ref: '#/components/schemas/DepartmentName1'
createdAt:
type: string
format: date-time
description: The timestamp for when the workload was created.
example: '2024-01-15T10:30:00Z'
createdBy:
type: string
description: Identifier of the user who created the workload.
format: .*
example: user@run.ai
updatedAt:
type: string
format: date-time
description: The timestamp for the last time the workload was updated.
example: '2024-01-15T10:35:00Z'
updatedBy:
type: string
description: Identifier of the user who last updated the workload.
format: .*
example: user@run.ai
deletedAt:
type:
- string
- 'null'
format: date-time
description: The timestamp indicating when the workload was deleted.
example: '2024-01-15T10:35:00Z'
deletedBy:
type:
- string
- 'null'
format: .*
description: Identifier of the user who deleted the workload.
example: user@run.ai
AutoScalingMaxReplicas:
description: The maximum number of replicas for autoscaling. Defaults to minReplicas. Must be no less than minReplicas.
type:
- integer
- 'null'
format: int32
minimum: 1
EnvironmentVariableUserCredential:
description: Defines a reference to a user-created credential and a specific key within that credential whose value will populate the environment variable. User credentials can only be accessed by the user who created them.
properties:
name:
description: The name of the user credential. (mandatory)
type:
- string
- 'null'
minLength: 1
example: my_postgres_user_and_password
key:
description: The key in the user credential resource. (mandatory)
type:
- string
- 'null'
minLength: 1
example: POSTGRES_PASSWORD
type:
- object
- 'null'
WorkloadV2Metadata:
allOf:
- $ref: '#/components/schemas/WorkloadV2MetadataResponse'
- $ref: '#/components/schemas/WorkloadV2MetadataAutoFill'
ComplianceIssuesV2:
properties:
complianceIssues:
type: array
items:
type: object
required:
- details
- field
properties:
field:
type: string
example: compute.gpuDevicesRequest
details:
type: string
example: value must be no less than 3
rule:
$ref: '#/components/schemas/PolicyRuleEnum'
type:
- object
- 'null'
Annotations:
description: Set of annotations to populate into the container running the workload.
type:
- array
- 'null'
items:
$ref: '#/components/schemas/Annotation'
Annotation:
description: Annotation details to be populated into the container.
properties:
name:
description: The name of the annotation (mandatory)
type:
- string
- 'null'
minLength: 1
maxLength: 63
example: billing
pattern: .*
value:
description: The value of the annotation.
type:
- string
- 'null'
example: my-billing-unit
pattern: .*
exclude:
description: Use 'true' in case the annotation is defined in defaults of the policy, and you wish to exclude it from the workload.
type:
- boolean
- 'null'
default: false
example: false
type:
- object
- 'null'
AutoScalingMetricThreshold:
description: The threshold to use with the specified metric for autoscaling (mandatory).
type:
- integer
- 'null'
format: int32
RunAsUid:
description: The user id to run the entrypoint of the container which executes the workspace. Default to the value specified in the environment asset `runAsUid` field (optional). Use only when the source uid/gid of the environment asset is not `fromTheImage`, and `overrideUidGidInWorkspace` is enabled.
type:
- integer
- 'null'
format: int64
example: 500
PvcVolumeMode:
description: Default volume mode for the PVC. Choose between Filesystem (default) or Block.
type:
- string
- 'null'
enum:
- Filesystem
- Block
NimServiceCreateRequest:
type: object
required:
- metadata
- spec
properties:
metadata:
$ref: '#/components/schemas/NIMServiceMetadataCreateParams'
spec:
$ref: '#/components/schemas/NimServiceSpec'
NodePools:
description: A prioritized list of node pools for the scheduler to run the workload on. The scheduler will always try to use the first node pool before moving to the next one if the first is not available.
type:
- array
- 'null'
items:
type: string
pattern: .*
example:
- my-node-pool-a
- my-node-pool-b
PvcClaimSize:
description: 'Requested size for the PVC. Mandatory when existingPvc is false. Recommended sizes: TB/GB/MB/TIB/GIB/MIB'
type:
- string
- 'null'
pattern: ^([+]?[0-9.]+)([eEinumkKMGTP]*[-+]?[0-9]*)$
example: 1G
NimServiceWorkers:
description: Specifies the number of worker nodes to use when running the NIM service in multi-node.
type:
- integer
- 'null'
format: int32
minimum: 1
maximum: 1000
example: 3
ServingPortPort:
description: The port that the container running the inference service exposes (mandatory).
type:
- integer
- 'null'
format: int32
minimum: 1
maximum: 65535
example: 8000
EnvironmentVariables:
description: Set of environment variables to populate into the container running the workload.
type:
- array
- 'null'
items:
$ref: '#/components/schemas/EnvironmentVariable'
GpuPortionLimit:
description: Limitations on the portion consumed by the workload, per GPU device. The system guarantees The gpuPotionLimit must be no less than the gpuPortionRequest.
type:
- number
- 'null'
format: double
example: 0.5
minimum: 0
ServingPortExposedUrl:
description: The custom URL to use for the serving port. If empty (default), an autogenerated URL will be used.
type:
- string
- 'null'
pattern: .*
NimServiceResponse:
type: object
required:
- spec
- metadata
- desiredPhase
properties:
metadata:
$ref: '#/components/schemas/WorkloadV2Metadata'
desiredPhase:
$ref: '#/components/schemas/DesiredPhase'
spec:
$ref: '#/components/schemas/NimServiceSpec'
EnvironmentVariableConfigMap:
description: Details of the configMap and key use to populate the environment variable
properties:
name:
description: The name of the config-map resource. (mandatory)
type:
- string
- 'null'
minLength: 1
example: my-config-map
pattern: ^[a-z0-9]([a-z0-9-]*[a-z0-9])?$
key:
description: The key in the config-map resource. (mandatory)
type:
- string
- 'null'
minLength: 1
example: MY_POSTGRES_SCHEMA
pattern: .*
type:
- object
- 'null'
PolicyRuleEnum:
description: Indicates which validation rule (e.g., min, max, step, options, required, canEdit, canAdd) conflicted with policy restrictions, causing the asset or template to be rejected.
type:
- string
- 'null'
enum:
- min
- max
- step
- options
- required
- canEdit
- canAdd
- locked
AutoScalingMetricNim:
description: The metric to use for autoscaling (mandatory).
type:
- string
- 'null'
pattern: ^[a-zA-Z_:][a-zA-Z0-9_:]*$
example: http_requests_total
ClusterId:
description: The id of the cluster.
type: string
format: uuid
example: 71f69d83-ba66-4822-adf5-55ce55efd210
CpuMemoryRequest:
description: The amount of CPU memory to allocate for this workload (1G, 20M, .etc). The workload will receive at least this amount of memory. Note that the workload will not be scheduled unless the system can guarantee this amount of memory to the workload
type:
- string
- 'null'
pattern: ^([+]?[0-9.]+)([eEinumkKMGTP]*[-+]?[0-9]*)$
example: 20M
PvcAddedAttrValue:
type: object
required:
- key
properties:
key:
type: string
minLength: 1
maxLength: 63
pattern: ^([A-Za-z0-9][-A-Za-z0-9_.]*)?[A-Za-z0-9]$
example: dnsname
value:
type: string
example: my.dns.com
pattern: .*
GVK:
type: object
description: Specifies the Group, Version, and Kind (GVK) of the Kubernetes resource that defines the workload.
required:
- group
- version
- kind
properties:
group:
description: The API group of the Kubernetes resource.
type: string
example: apps
version:
description: The API version of the resource within the specified group.
type: string
example: v1
kind:
description: The type of Kubernetes resource being referenced.
type: string
example: Deployment
GpuDevicesRequest:
description: Requested number of GPU devices. Currently if more than one device is requested, it is not possible to provide values for gpuMemory or gpuPortion.
type:
- integer
- 'null'
format: int32
example: 1
minimum: 0
Image:
description: Docker image name. For more information, see [Images](https://kubernetes.io/docs/concepts/containers/images). The image name is mandatory for creating a workload.
type:
- string
- 'null'
minLength: 1
example: python:3.8
pattern: .*
GpuRequestType:
description: Sets the unit type for GPU resources requests. Stated in terms of portion or memory. Sets the unit type for other GPU request fields. If `gpuDevicesRequest > 1`, only `portion` is supported. If `gpuDeviceRequest = 1`, the request type can be stated as `portion` or `memory`.
type:
- string
- 'null'
minLength: 1
enum:
- portion
- memory
ProjectName2:
type: string
description: The name of the project
example: project-a
minLength: 1
pattern: ^[a-z0-9]([a-z0-9-]*[a-z0-9])?$
ProbeHandlerScheme:
description: Scheme to use for connecting to the host, defaults to HTTP.
type:
- string
- 'null'
enum:
- HTTP
- HTTPS
GpuPortionRequest:
description: Required if and only if gpuRequestType is portion. States the portion of the GPU to allocate for the created workload, per GPU device, between 0 and 1. The default is no allocated GPUs.
type:
- number
- 'null'
format: double
example: 0.5
minimum: 0
DepartmentId3:
description: The id of the department.
type: string
minLength: 1
example: 2
pattern: .*
GpuMemoryLimit:
description: Limitation on the memory consumed by the workload, per GPU device. The system guarantees The gpuMemoryLimit must be no less than gpuMemoryRequest.
type:
- string
- 'null'
pattern: ^([+]?[0-9.]+)([eEinumkKMGTP]*[-+]?[0-9]*)$
example: 10M
ClaimInfo:
description: Claim information for the newly created PVC. The information should not be provided when attempting to use existing PVC.
properties:
size:
$ref: '#/components/schemas/PvcClaimSize'
storageClass:
description: Storage class name to associate with the PVC. This parameter may be omitted if there is a single storage class in the system, or you are using the default storage class. For more information, see [Storage class](https://kubernetes.io/docs/concepts/storage/storage-classes).
type:
- string
- 'null'
minLength: 1
example: my-storage-class
pattern: ^[a-z0-9]([a-z0-9-]*[a-z0-9])?$
accessModes:
$ref: '#/components/schemas/PvcAccessModes'
volumeMode:
$ref: '#/components/schemas/PvcVolumeMode'
addedAttrValues:
$ref: '#/components/schemas/PvcAddedAttrValues'
type:
- object
- 'null'
NimServicePvcFields:
properties:
existingPvc:
description: Verify existing PVC. PVC is assumed to exist when set to `true`. If set to `false`, the PVC will be created, if it does not exist.
type:
- boolean
- 'null'
default: false
claimName:
description: Name for the PVC. Allow referencing it across workloads. If not provided, a name based on the workload name and scope will be auto-generated.
type:
- string
- 'null'
minLength: 1
maxLength: 63
example: my-claim
pattern: .*
readOnly:
description: Permit only read access to PVC.
type:
- boolean
- 'null'
default: false
claimInfo:
$ref: '#/components/schemas/ClaimInfo'
type:
- object
- 'null'
ServingPortGrpcPort:
description: The GRPC port that the container running the inference service exposes.
type:
- integer
- 'null'
format: int32
minimum: 1
maximum: 65535
example: 8001
ServingPortExposedProtocol:
description: The protocol to use for the exposed URL. If grpcPort is set, this defaults to grpc. Otherwise, it defaults to http.
type:
- string
- 'null'
enum:
- http
- grpc
PvcAccessModes:
description: Default access mode(s) applied to newly created PVCs unless explicitly overridden.
properties:
readWriteOnce:
description: Mount the volume as read/write by a single node.
type:
- boolean
- 'null'
default: true
readOnlyMany:
description: Mount the volume as read-only by many nodes.
type:
- boolean
- 'null'
default: false
readWriteMany:
description: Mount the volume as read/write by many nodes.
type:
- boolean
- 'null'
default: false
type:
- object
- 'null'
CpuCoreRequest:
description: CPU units to allocate for the created workload (0.5, 1, .etc). The workload will receive at least this amount of CPU. Note that the workload will not be scheduled unless the system can guarantee this amount of CPUs to the workload.
format: double
type:
- number
- 'null'
example: 0.5
minimum: 0
Category:
description: Specify the workload category assigned to the workload. Categories are used to classify and monitor different types of workloads within the NVIDIA Run:ai platform.
type:
- string
- 'null'
pattern: .*
PvcAddedAttrValues:
description: an optional array of key-values pairs that are written as annotations on the created PVC. the allowed attributes are determined according to the storage class configuration (see k8s-objects-tracker for further info).
type: array
items:
$ref: '#/components/schemas/PvcAddedAttrValue'
DepartmentName1:
type: string
description: The name of the department
example: default
minLength: 1
pattern: ^[a-z0-9]([a-z0-9-]*[a-z0-9])?$
Toleration:
description: Toleration details.
properties:
name:
description: The name of the toleration.
type:
- string
- 'null'
minLength: 1
pattern: .*
operator:
$ref: '#/components/schemas/TolerationOperator'
key:
description: The taint key that the toleration applies to. (mandatory)
type:
- string
- 'null'
pattern: .*
value:
description: The taint value the toleration matches to. Mandatory if operator is Exists, forbidden otherwise.
type:
- string
- 'null'
pattern: .*
effect:
$ref: '#/components/schemas/TolerationEffect'
seconds:
description: The period of time the toleration tolerates the taint. Valid only if effect is NoExecute. taint.
type:
- integer
- 'null'
minimum: 1
exclude:
description: Use 'true' in case the label is defined in defaults of the policy, and you wish to exclude it from the workload.
type:
- boolean
- 'null'
example: false
type:
- object
- 'null'
CpuCoreLimit:
description: Limitations on the number of CPUs consumed by the workload (0.5, 1, .etc). The system guarantees that this workload will not be able to consume more than this amount of CPUs.
format: double
type:
- number
- 'null'
example: 2
minimum: 0
ImagePullSecret:
description: A reference to a secret in the same namespace used to pull container images.
properties:
name:
type: string
description: The name of the Kubernetes secret containing the image pull credentials.
pattern: ^[a-z0-9]([a-z0-9-]*[a-z0-9])?$
userCredential:
type:
- boolean
- 'null'
description: Indicates whether the secret is a user credential. Set to true if the secret was created by the user and is only accessible by them.
exclude:
description: Use 'true' in case the secret is defined in defaults of the policy, and you wish to exclude it from the workload.
type:
- boolean
- 'null'
default: false
example: false
type:
- object
- 'null'
WorkloadId3:
description: A unique ID of the workload.
type: string
format: uuid
NimServiceSpec:
allOf:
- properties:
annotations:
$ref: '#/components/schemas/Annotations'
autoscaling:
properties:
maxReplicas:
$ref: '#/components/schemas/AutoScalingMaxReplicas'
metric:
$ref: '#/components/schemas/AutoScalingMetricNim'
metricThreshold:
$ref: '#/components/schemas/AutoScalingMetricThreshold'
minReplicas:
$ref: '#/components/schemas/AutoScalingMinReplicas'
scaleWindowSeconds:
$ref: '#/components/schemas/AutoScalingScaleWindowSeconds'
type:
- object
- 'null'
category:
$ref: '#/components/schemas/Category'
compute:
properties:
cpuCoreLimit:
$ref: '#/components/schemas/CpuCoreLimit'
cpuCoreRequest:
$ref: '#/components/schemas/CpuCoreRequest'
cpuMemoryLimit:
$ref: '#/components/schemas/CpuMemoryLimit'
cpuMemoryRequest:
$ref: '#/components/schemas/CpuMemoryRequest'
gpuDevicesRequest:
$ref: '#/components/schemas/GpuDevicesRequest'
gpuMemoryLimit:
$ref: '#/components/schemas/GpuMemoryLimit'
gpuMemoryRequest:
$ref: '#/components/schemas/GpuMemoryRequest'
gpuPortionLimit:
$ref: '#/components/schemas/GpuPortionLimit'
gpuPortionRequest:
$ref: '#/components/schemas/GpuPortionRequest'
gpuRequestType:
$ref: '#/components/schemas/GpuRequestType'
type:
- object
- 'null'
environmentVariables:
$ref: '#/components/schemas/EnvironmentVariables'
image:
$ref: '#/components/schemas/Image'
imagePullPolicy:
$ref: '#/components/schemas/ImagePullPolicy'
imagePullSecrets:
$ref: '#/components/schemas/ImagePullSecrets'
labels:
$ref: '#/components/schemas/Labels'
modelStore:
properties:
nimCache:
$ref: '#/components/schemas/NimCache'
pvc:
$ref: '#/components/schemas/NimServicePvcFields'
type:
- object
- 'null'
multiNode:
$ref: '#/components/schemas/NimServiceMultiNode'
# --- truncated at 32 KB (46 KB total) ---
# Full source: https://raw.githubusercontent.com/api-evangelist/runai/refs/heads/main/openapi/runai-nvidia-nim-api-openapi.yml