Create a provider-managed deployment
POST
/
api
/
v1
/
deployments
Create a provider-managed deployment
curl --request POST \
--url http://127.0.0.1:18000/api/v1/deployments \
--header 'Authorization: Bearer <token>' \
--header 'Content-Type: application/json' \
--header 'Idempotency-Key: <idempotency-key>' \
--data '
{
"cloud": "<string>",
"gpu": "<string>",
"model": "<string>",
"name": "<string>",
"billing_mode": "provider_account",
"endpoint_name": "<string>",
"gpu_count": 1,
"managed_runtime_seconds": 43650,
"max_replicas": 1,
"min_replicas": 1,
"model_revision": "<string>",
"model_secret_reference_id": "<string>",
"port": 32768,
"provider_adapter": "<string>",
"region": "<string>",
"runtime": "vllm",
"runtime_args": [
"<string>"
],
"runtime_version": "<string>",
"serving": {
"autoscaling": {
"max": 5000,
"min": 5000
},
"backend": "dynamo",
"cache": {
"configuration_ref": "<string>",
"disk_gib": 1,
"host_gib": 1,
"memory_gib": 1,
"metrics": true,
"storage_claim": "<string>"
},
"decode": {
"replicas": 5000,
"tensor_parallelism": 512
},
"prefill": {
"replicas": 5000,
"tensor_parallelism": 512
},
"schema_version": "infercrane.serving/v1",
"worker": {
"replicas": 5000,
"tensor_parallelism": 512
}
}
}
'import requests
url = "http://127.0.0.1:18000/api/v1/deployments"
payload = {
"cloud": "<string>",
"gpu": "<string>",
"model": "<string>",
"name": "<string>",
"billing_mode": "provider_account",
"endpoint_name": "<string>",
"gpu_count": 1,
"managed_runtime_seconds": 43650,
"max_replicas": 1,
"min_replicas": 1,
"model_revision": "<string>",
"model_secret_reference_id": "<string>",
"port": 32768,
"provider_adapter": "<string>",
"region": "<string>",
"runtime": "vllm",
"runtime_args": ["<string>"],
"runtime_version": "<string>",
"serving": {
"autoscaling": {
"max": 5000,
"min": 5000
},
"backend": "dynamo",
"cache": {
"configuration_ref": "<string>",
"disk_gib": 1,
"host_gib": 1,
"memory_gib": 1,
"metrics": True,
"storage_claim": "<string>"
},
"decode": {
"replicas": 5000,
"tensor_parallelism": 512
},
"prefill": {
"replicas": 5000,
"tensor_parallelism": 512
},
"schema_version": "infercrane.serving/v1",
"worker": {
"replicas": 5000,
"tensor_parallelism": 512
}
}
}
headers = {
"Idempotency-Key": "<idempotency-key>",
"Authorization": "Bearer <token>",
"Content-Type": "application/json"
}
response = requests.post(url, json=payload, headers=headers)
print(response.text)const options = {
method: 'POST',
headers: {
'Idempotency-Key': '<idempotency-key>',
Authorization: 'Bearer <token>',
'Content-Type': 'application/json'
},
body: JSON.stringify({
cloud: '<string>',
gpu: '<string>',
model: '<string>',
name: '<string>',
billing_mode: 'provider_account',
endpoint_name: '<string>',
gpu_count: 1,
managed_runtime_seconds: 43650,
max_replicas: 1,
min_replicas: 1,
model_revision: '<string>',
model_secret_reference_id: '<string>',
port: 32768,
provider_adapter: '<string>',
region: '<string>',
runtime: 'vllm',
runtime_args: ['<string>'],
runtime_version: '<string>',
serving: {
autoscaling: {max: 5000, min: 5000},
backend: 'dynamo',
cache: {
configuration_ref: '<string>',
disk_gib: 1,
host_gib: 1,
memory_gib: 1,
metrics: true,
storage_claim: '<string>'
},
decode: {replicas: 5000, tensor_parallelism: 512},
prefill: {replicas: 5000, tensor_parallelism: 512},
schema_version: 'infercrane.serving/v1',
worker: {replicas: 5000, tensor_parallelism: 512}
}
})
};
fetch('http://127.0.0.1:18000/api/v1/deployments', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));package main
import (
"fmt"
"strings"
"net/http"
"io"
)
func main() {
url := "http://127.0.0.1:18000/api/v1/deployments"
payload := strings.NewReader("{\n \"cloud\": \"<string>\",\n \"gpu\": \"<string>\",\n \"model\": \"<string>\",\n \"name\": \"<string>\",\n \"billing_mode\": \"provider_account\",\n \"endpoint_name\": \"<string>\",\n \"gpu_count\": 1,\n \"managed_runtime_seconds\": 43650,\n \"max_replicas\": 1,\n \"min_replicas\": 1,\n \"model_revision\": \"<string>\",\n \"model_secret_reference_id\": \"<string>\",\n \"port\": 32768,\n \"provider_adapter\": \"<string>\",\n \"region\": \"<string>\",\n \"runtime\": \"vllm\",\n \"runtime_args\": [\n \"<string>\"\n ],\n \"runtime_version\": \"<string>\",\n \"serving\": {\n \"autoscaling\": {\n \"max\": 5000,\n \"min\": 5000\n },\n \"backend\": \"dynamo\",\n \"cache\": {\n \"configuration_ref\": \"<string>\",\n \"disk_gib\": 1,\n \"host_gib\": 1,\n \"memory_gib\": 1,\n \"metrics\": true,\n \"storage_claim\": \"<string>\"\n },\n \"decode\": {\n \"replicas\": 5000,\n \"tensor_parallelism\": 512\n },\n \"prefill\": {\n \"replicas\": 5000,\n \"tensor_parallelism\": 512\n },\n \"schema_version\": \"infercrane.serving/v1\",\n \"worker\": {\n \"replicas\": 5000,\n \"tensor_parallelism\": 512\n }\n }\n}")
req, _ := http.NewRequest("POST", url, payload)
req.Header.Add("Idempotency-Key", "<idempotency-key>")
req.Header.Add("Authorization", "Bearer <token>")
req.Header.Add("Content-Type", "application/json")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}{
"operation": {
"id": "<string>",
"kind": "<string>",
"progress": 50,
"status": "pending",
"attempt": 123,
"cancel_requested": true,
"created_at": "2023-11-07T05:31:56Z",
"current_step": "<string>",
"current_step_status": "pending",
"error_code": "<string>",
"max_attempts": 123,
"message": "<string>",
"resource_name": "<string>",
"resource_type": "<string>",
"result": {},
"retryable": true,
"updated_at": "2023-11-07T05:31:56Z"
},
"created": true,
"deployment": {
"model": "<string>",
"name": "<string>",
"runtime": "<string>",
"active_revision_id": "<string>",
"candidate_revision_id": "<string>",
"endpoint_names": [
"<string>"
],
"id": "<string>",
"max_replicas": 123,
"min_replicas": 123,
"observed_state": "<string>"
}
}{
"error": {
"code": "<string>",
"message": "<string>",
"category": "<string>",
"remediation": "<string>",
"request_id": "<string>",
"retryable": true
}
}Authorizations
Bearer authentication header of the form Bearer <token>, where <token> is your auth token.
Headers
Stable key used to adopt the original durable operation after retries or disconnects.
Maximum string length:
128Body
application/json
Available options:
provider_account, customer_wallet Available options:
elastic, serverless Stable application endpoint alias. Defaults to the deployment name.
Pattern:
^[a-z0-9](?:[a-z0-9._-]{0,62}[a-z0-9])?$Accelerators allocated to each runtime replica.
Required range:
1 <= x <= 1024Required range:
900 <= x <= 86400Must be a multiple of 60Required range:
x >= 0Required range:
x >= 0Tenant-scoped server-side credential reference for private or gated model resolution.
Required range:
1 <= x <= 65535Exact provider profile for this immutable revision; omit only when the cloud/runtime default is unambiguous.
Available options:
vllm, sglang, custom-oci Show child attributes
Show child attributes
Show child attributes
Show child attributes
⌘I
Create a provider-managed deployment
curl --request POST \
--url http://127.0.0.1:18000/api/v1/deployments \
--header 'Authorization: Bearer <token>' \
--header 'Content-Type: application/json' \
--header 'Idempotency-Key: <idempotency-key>' \
--data '
{
"cloud": "<string>",
"gpu": "<string>",
"model": "<string>",
"name": "<string>",
"billing_mode": "provider_account",
"endpoint_name": "<string>",
"gpu_count": 1,
"managed_runtime_seconds": 43650,
"max_replicas": 1,
"min_replicas": 1,
"model_revision": "<string>",
"model_secret_reference_id": "<string>",
"port": 32768,
"provider_adapter": "<string>",
"region": "<string>",
"runtime": "vllm",
"runtime_args": [
"<string>"
],
"runtime_version": "<string>",
"serving": {
"autoscaling": {
"max": 5000,
"min": 5000
},
"backend": "dynamo",
"cache": {
"configuration_ref": "<string>",
"disk_gib": 1,
"host_gib": 1,
"memory_gib": 1,
"metrics": true,
"storage_claim": "<string>"
},
"decode": {
"replicas": 5000,
"tensor_parallelism": 512
},
"prefill": {
"replicas": 5000,
"tensor_parallelism": 512
},
"schema_version": "infercrane.serving/v1",
"worker": {
"replicas": 5000,
"tensor_parallelism": 512
}
}
}
'import requests
url = "http://127.0.0.1:18000/api/v1/deployments"
payload = {
"cloud": "<string>",
"gpu": "<string>",
"model": "<string>",
"name": "<string>",
"billing_mode": "provider_account",
"endpoint_name": "<string>",
"gpu_count": 1,
"managed_runtime_seconds": 43650,
"max_replicas": 1,
"min_replicas": 1,
"model_revision": "<string>",
"model_secret_reference_id": "<string>",
"port": 32768,
"provider_adapter": "<string>",
"region": "<string>",
"runtime": "vllm",
"runtime_args": ["<string>"],
"runtime_version": "<string>",
"serving": {
"autoscaling": {
"max": 5000,
"min": 5000
},
"backend": "dynamo",
"cache": {
"configuration_ref": "<string>",
"disk_gib": 1,
"host_gib": 1,
"memory_gib": 1,
"metrics": True,
"storage_claim": "<string>"
},
"decode": {
"replicas": 5000,
"tensor_parallelism": 512
},
"prefill": {
"replicas": 5000,
"tensor_parallelism": 512
},
"schema_version": "infercrane.serving/v1",
"worker": {
"replicas": 5000,
"tensor_parallelism": 512
}
}
}
headers = {
"Idempotency-Key": "<idempotency-key>",
"Authorization": "Bearer <token>",
"Content-Type": "application/json"
}
response = requests.post(url, json=payload, headers=headers)
print(response.text)const options = {
method: 'POST',
headers: {
'Idempotency-Key': '<idempotency-key>',
Authorization: 'Bearer <token>',
'Content-Type': 'application/json'
},
body: JSON.stringify({
cloud: '<string>',
gpu: '<string>',
model: '<string>',
name: '<string>',
billing_mode: 'provider_account',
endpoint_name: '<string>',
gpu_count: 1,
managed_runtime_seconds: 43650,
max_replicas: 1,
min_replicas: 1,
model_revision: '<string>',
model_secret_reference_id: '<string>',
port: 32768,
provider_adapter: '<string>',
region: '<string>',
runtime: 'vllm',
runtime_args: ['<string>'],
runtime_version: '<string>',
serving: {
autoscaling: {max: 5000, min: 5000},
backend: 'dynamo',
cache: {
configuration_ref: '<string>',
disk_gib: 1,
host_gib: 1,
memory_gib: 1,
metrics: true,
storage_claim: '<string>'
},
decode: {replicas: 5000, tensor_parallelism: 512},
prefill: {replicas: 5000, tensor_parallelism: 512},
schema_version: 'infercrane.serving/v1',
worker: {replicas: 5000, tensor_parallelism: 512}
}
})
};
fetch('http://127.0.0.1:18000/api/v1/deployments', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));package main
import (
"fmt"
"strings"
"net/http"
"io"
)
func main() {
url := "http://127.0.0.1:18000/api/v1/deployments"
payload := strings.NewReader("{\n \"cloud\": \"<string>\",\n \"gpu\": \"<string>\",\n \"model\": \"<string>\",\n \"name\": \"<string>\",\n \"billing_mode\": \"provider_account\",\n \"endpoint_name\": \"<string>\",\n \"gpu_count\": 1,\n \"managed_runtime_seconds\": 43650,\n \"max_replicas\": 1,\n \"min_replicas\": 1,\n \"model_revision\": \"<string>\",\n \"model_secret_reference_id\": \"<string>\",\n \"port\": 32768,\n \"provider_adapter\": \"<string>\",\n \"region\": \"<string>\",\n \"runtime\": \"vllm\",\n \"runtime_args\": [\n \"<string>\"\n ],\n \"runtime_version\": \"<string>\",\n \"serving\": {\n \"autoscaling\": {\n \"max\": 5000,\n \"min\": 5000\n },\n \"backend\": \"dynamo\",\n \"cache\": {\n \"configuration_ref\": \"<string>\",\n \"disk_gib\": 1,\n \"host_gib\": 1,\n \"memory_gib\": 1,\n \"metrics\": true,\n \"storage_claim\": \"<string>\"\n },\n \"decode\": {\n \"replicas\": 5000,\n \"tensor_parallelism\": 512\n },\n \"prefill\": {\n \"replicas\": 5000,\n \"tensor_parallelism\": 512\n },\n \"schema_version\": \"infercrane.serving/v1\",\n \"worker\": {\n \"replicas\": 5000,\n \"tensor_parallelism\": 512\n }\n }\n}")
req, _ := http.NewRequest("POST", url, payload)
req.Header.Add("Idempotency-Key", "<idempotency-key>")
req.Header.Add("Authorization", "Bearer <token>")
req.Header.Add("Content-Type", "application/json")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}{
"operation": {
"id": "<string>",
"kind": "<string>",
"progress": 50,
"status": "pending",
"attempt": 123,
"cancel_requested": true,
"created_at": "2023-11-07T05:31:56Z",
"current_step": "<string>",
"current_step_status": "pending",
"error_code": "<string>",
"max_attempts": 123,
"message": "<string>",
"resource_name": "<string>",
"resource_type": "<string>",
"result": {},
"retryable": true,
"updated_at": "2023-11-07T05:31:56Z"
},
"created": true,
"deployment": {
"model": "<string>",
"name": "<string>",
"runtime": "<string>",
"active_revision_id": "<string>",
"candidate_revision_id": "<string>",
"endpoint_names": [
"<string>"
],
"id": "<string>",
"max_replicas": 123,
"min_replicas": 123,
"observed_state": "<string>"
}
}{
"error": {
"code": "<string>",
"message": "<string>",
"category": "<string>",
"remediation": "<string>",
"request_id": "<string>",
"retryable": true
}
}