v32

latestOpenAPI 3.1.0raw.githubusercontent.com2026-05-151,1412,2144.5 MB
Everywhere Inference

Get inference deployment

get/cloud/v3/inference/{project_id}/deployments/{deployment_name}

Path parameters

project_idinteger required

Project ID

Example:1

Project ID

deployment_namestring required

Inference instance name.

Example:my-instance

Inference instance name.

Response

OK

addressstring uri nullable required

Address of the inference instance

api_keysstring[] nullable

List of API keys for the inference instance

auth_enabledboolean required

true if instance uses API key authentication. "Authorization": "Bearer *****" or "X-Api-Key": "*****" header is required for the requests to the instance if enabled.

commandstring nullable required

Command to be executed when running a container from an image.

created_atstring nullable required

Inference instance creation date in ISO 8601 format.

credentials_namestring required

Registry credentials name

descriptionstring required

Inference instance description.

envsobject nullable required

Environment variables for the inference instance

flavor_namestring required

Flavor name for the inference instance

imagestring required

Docker image for the inference instance. This field should contain the image name and tag in the format 'name:tag', e.g., 'nginx:latest'. It defaults to Docker Hub as the image registry, but any accessible Docker image URL can be specified.

listening_portinteger required

Listening port for the inference instance.

namestring required

Inference instance name.

project_idinteger required

Project ID. If not provided, your default project ID will be used.

status'ACTIVE' | 'DELETING' | 'DEPLOYING' | 'DISABLED' | 'PARTIALLYDEPLOYED' | 'PENDING' required
timeoutinteger nullable required

Specifies the duration in seconds without any requests after which the containers will be downscaled to their minimum scale value as defined by scale.min. If set, this helps in optimizing resource usage by reducing the number of container instances during periods of inactivity.

Example response

{
  "address": "https://example.com",
  "api_keys": [
    "key1",
    "key2"
  ],
  "containers": [
    {
      "address": "https://example.com",
      "deploy_status": {
        "ready": 1,
        "total": 2
      },
      "error_message": "Failed to pull image",
      "region_id": 1,
      "scale": {
        "cooldown_period": 60,
        "max": 3,
        "min": 1,
        "polling_interval": 30,
        "triggers": {
          "cpu": {
            "threshold": 75
          },
          "gpu_memory": {
            "threshold": 75
          },
          "gpu_utilization": {
            "threshold": 75
          },
          "http": {
            "rate": 1,
            "window": 60
          },
          "memory": {
            "threshold": 75
          },
          "sqs": {
            "activation_queue_length": 5,
            "aws_region": "us-east-1",
            "queue_length": 10,
            "queue_url": "https://sqs.us-east-1.amazonaws.com/123456789012/MyQueue",
            "scale_on_delayed": true,
            "scale_on_flight": true
          }
        }
      }
    }
  ],
  "created_at": "2023-08-22T11:21:00Z",
  "credentials_name": "dockerhub",
  "description": "My first instance",
  "envs": {
    "DEBUG_MODE": "False",
    "KEY": "12345"
  },
  "flavor_name": "inference-16vcpu-232gib-1xh100-80gb",
  "image": "nginx:latest",
  "ingress_opts": {
    "disable_response_buffering": true
  },
  "listening_port": 8080,
  "logging": {
    "destination_region_id": 1,
    "enabled": true,
    "retention_policy": {
      "period": 45
    },
    "topic_name": "my-log-name"
  },
  "name": "my-instance",
  "object_references": [
    {
      "name": "my-inference-app"
    }
  ],
  "probes": {
    "liveness_probe": {
      "enabled": true,
      "probe": {
        "exec": {
          "command": [
            "ls",
            "-l"
          ]
        },
        "failure_threshold": 3,
        "http_get": {
          "headers": {
            "Authorization": "Bearer token 123"
          },
          "host": "127.0.0.1",
          "path": "/healthz",
          "port": 80,
          "schema": "HTTP"
        },
        "period_seconds": 5,
        "success_threshold": 1,
        "tcp_socket": {
          "port": 80
        },
        "timeout_seconds": 1
      }
    },
    "readiness_probe": {
      "enabled": true,
      "probe": {
        "exec": {
          "command": [
            "ls",
            "-l"
          ]
        },
        "failure_threshold": 3,
        "http_get": {
          "headers": {
            "Authorization": "Bearer token 123"
          },
          "host": "127.0.0.1",
          "path": "/healthz",
          "port": 80,
          "schema": "HTTP"
        },
        "period_seconds": 5,
        "success_threshold": 1,
        "tcp_socket": {
          "port": 80
        },
        "timeout_seconds": 1
      }
    },
    "startup_probe": {
      "enabled": true,
      "probe": {
        "exec": {
          "command": [
            "ls",
            "-l"
          ]
        },
        "failure_threshold": 3,
        "http_get": {
          "headers": {
            "Authorization": "Bearer token 123"
          },
          "host": "127.0.0.1",
          "path": "/healthz",
          "port": 80,
          "schema": "HTTP"
        },
        "period_seconds": 5,
        "success_threshold": 1,
        "tcp_socket": {
          "port": 80
        },
        "timeout_seconds": 1
      }
    }
  },
  "project_id": 1,
  "timeout": 120
}