Deploy a fine-tuned model with vLLM
curl --request POST \
--url https://api.example.com/api/vllm/deployments \
--header 'Authorization: Bearer <token>' \
--header 'Content-Type: application/json' \
--data '
{
"name": "<string>",
"finetune_job_id": "3c90c3cc-0d44-4b50-8888-8dd25736052a",
"served_model_name": "<string>",
"gpu_count": 32,
"min_replicas": 1,
"max_replicas": 2,
"max_model_len": 2,
"dtype": "<string>",
"quality_score": 123
}
'import requests
url = "https://api.example.com/api/vllm/deployments"
payload = {
"name": "<string>",
"finetune_job_id": "3c90c3cc-0d44-4b50-8888-8dd25736052a",
"served_model_name": "<string>",
"gpu_count": 32,
"min_replicas": 1,
"max_replicas": 2,
"max_model_len": 2,
"dtype": "<string>",
"quality_score": 123
}
headers = {
"Authorization": "Bearer <token>",
"Content-Type": "application/json"
}
response = requests.post(url, json=payload, headers=headers)
print(response.text)const options = {
method: 'POST',
headers: {Authorization: 'Bearer <token>', 'Content-Type': 'application/json'},
body: JSON.stringify({
name: '<string>',
finetune_job_id: '3c90c3cc-0d44-4b50-8888-8dd25736052a',
served_model_name: '<string>',
gpu_count: 32,
min_replicas: 1,
max_replicas: 2,
max_model_len: 2,
dtype: '<string>',
quality_score: 123
})
};
fetch('https://api.example.com/api/vllm/deployments', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://api.example.com/api/vllm/deployments",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "POST",
CURLOPT_POSTFIELDS => json_encode([
'name' => '<string>',
'finetune_job_id' => '3c90c3cc-0d44-4b50-8888-8dd25736052a',
'served_model_name' => '<string>',
'gpu_count' => 32,
'min_replicas' => 1,
'max_replicas' => 2,
'max_model_len' => 2,
'dtype' => '<string>',
'quality_score' => 123
]),
CURLOPT_HTTPHEADER => [
"Authorization: Bearer <token>",
"Content-Type: application/json"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"strings"
"net/http"
"io"
)
func main() {
url := "https://api.example.com/api/vllm/deployments"
payload := strings.NewReader("{\n \"name\": \"<string>\",\n \"finetune_job_id\": \"3c90c3cc-0d44-4b50-8888-8dd25736052a\",\n \"served_model_name\": \"<string>\",\n \"gpu_count\": 32,\n \"min_replicas\": 1,\n \"max_replicas\": 2,\n \"max_model_len\": 2,\n \"dtype\": \"<string>\",\n \"quality_score\": 123\n}")
req, _ := http.NewRequest("POST", url, payload)
req.Header.Add("Authorization", "Bearer <token>")
req.Header.Add("Content-Type", "application/json")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.post("https://api.example.com/api/vllm/deployments")
.header("Authorization", "Bearer <token>")
.header("Content-Type", "application/json")
.body("{\n \"name\": \"<string>\",\n \"finetune_job_id\": \"3c90c3cc-0d44-4b50-8888-8dd25736052a\",\n \"served_model_name\": \"<string>\",\n \"gpu_count\": 32,\n \"min_replicas\": 1,\n \"max_replicas\": 2,\n \"max_model_len\": 2,\n \"dtype\": \"<string>\",\n \"quality_score\": 123\n}")
.asString();require 'uri'
require 'net/http'
url = URI("https://api.example.com/api/vllm/deployments")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Post.new(url)
request["Authorization"] = 'Bearer <token>'
request["Content-Type"] = 'application/json'
request.body = "{\n \"name\": \"<string>\",\n \"finetune_job_id\": \"3c90c3cc-0d44-4b50-8888-8dd25736052a\",\n \"served_model_name\": \"<string>\",\n \"gpu_count\": 32,\n \"min_replicas\": 1,\n \"max_replicas\": 2,\n \"max_model_len\": 2,\n \"dtype\": \"<string>\",\n \"quality_score\": 123\n}"
response = http.request(request)
puts response.read_body{
"deployment_id": "3c90c3cc-0d44-4b50-8888-8dd25736052a",
"name": "<string>",
"status": "pending",
"domain": "<string>",
"owner": "<string>",
"finetune_job_id": "3c90c3cc-0d44-4b50-8888-8dd25736052a",
"base_model": "<string>",
"registered_model_name": "<string>",
"served_model_name": "<string>",
"created_at": "2023-11-07T05:31:56Z",
"updated_at": "2023-11-07T05:31:56Z",
"model_version": "<string>",
"service_url": "<string>",
"replica_count": 123,
"ready_replicas": 123,
"router_endpoint_name": "<string>",
"router_model_name": "<string>",
"gpu_resources": {},
"error_message": "<string>",
"stopped_at": "2023-11-07T05:31:56Z"
}{
"error": "<string>",
"code": 500
}{
"error": "<string>",
"code": 500
}{
"error": "<string>",
"code": 500
}{
"detail": [
{
"loc": [
"<string>"
],
"msg": "<string>",
"type": "<string>",
"input": "<unknown>",
"ctx": {}
}
]
}{
"error": "<string>",
"code": 500
}{
"error": "<string>",
"code": 500
}Inferencing
Deploy a fine-tuned model with vLLM
Render + apply the vLLM manifests and register the model on readiness.
POST
/
api
/
vllm
/
deployments
Deploy a fine-tuned model with vLLM
curl --request POST \
--url https://api.example.com/api/vllm/deployments \
--header 'Authorization: Bearer <token>' \
--header 'Content-Type: application/json' \
--data '
{
"name": "<string>",
"finetune_job_id": "3c90c3cc-0d44-4b50-8888-8dd25736052a",
"served_model_name": "<string>",
"gpu_count": 32,
"min_replicas": 1,
"max_replicas": 2,
"max_model_len": 2,
"dtype": "<string>",
"quality_score": 123
}
'import requests
url = "https://api.example.com/api/vllm/deployments"
payload = {
"name": "<string>",
"finetune_job_id": "3c90c3cc-0d44-4b50-8888-8dd25736052a",
"served_model_name": "<string>",
"gpu_count": 32,
"min_replicas": 1,
"max_replicas": 2,
"max_model_len": 2,
"dtype": "<string>",
"quality_score": 123
}
headers = {
"Authorization": "Bearer <token>",
"Content-Type": "application/json"
}
response = requests.post(url, json=payload, headers=headers)
print(response.text)const options = {
method: 'POST',
headers: {Authorization: 'Bearer <token>', 'Content-Type': 'application/json'},
body: JSON.stringify({
name: '<string>',
finetune_job_id: '3c90c3cc-0d44-4b50-8888-8dd25736052a',
served_model_name: '<string>',
gpu_count: 32,
min_replicas: 1,
max_replicas: 2,
max_model_len: 2,
dtype: '<string>',
quality_score: 123
})
};
fetch('https://api.example.com/api/vllm/deployments', options)
.then(res => res.json())
.then(res => console.log(res))
.catch(err => console.error(err));<?php
$curl = curl_init();
curl_setopt_array($curl, [
CURLOPT_URL => "https://api.example.com/api/vllm/deployments",
CURLOPT_RETURNTRANSFER => true,
CURLOPT_ENCODING => "",
CURLOPT_MAXREDIRS => 10,
CURLOPT_TIMEOUT => 30,
CURLOPT_HTTP_VERSION => CURL_HTTP_VERSION_1_1,
CURLOPT_CUSTOMREQUEST => "POST",
CURLOPT_POSTFIELDS => json_encode([
'name' => '<string>',
'finetune_job_id' => '3c90c3cc-0d44-4b50-8888-8dd25736052a',
'served_model_name' => '<string>',
'gpu_count' => 32,
'min_replicas' => 1,
'max_replicas' => 2,
'max_model_len' => 2,
'dtype' => '<string>',
'quality_score' => 123
]),
CURLOPT_HTTPHEADER => [
"Authorization: Bearer <token>",
"Content-Type: application/json"
],
]);
$response = curl_exec($curl);
$err = curl_error($curl);
curl_close($curl);
if ($err) {
echo "cURL Error #:" . $err;
} else {
echo $response;
}package main
import (
"fmt"
"strings"
"net/http"
"io"
)
func main() {
url := "https://api.example.com/api/vllm/deployments"
payload := strings.NewReader("{\n \"name\": \"<string>\",\n \"finetune_job_id\": \"3c90c3cc-0d44-4b50-8888-8dd25736052a\",\n \"served_model_name\": \"<string>\",\n \"gpu_count\": 32,\n \"min_replicas\": 1,\n \"max_replicas\": 2,\n \"max_model_len\": 2,\n \"dtype\": \"<string>\",\n \"quality_score\": 123\n}")
req, _ := http.NewRequest("POST", url, payload)
req.Header.Add("Authorization", "Bearer <token>")
req.Header.Add("Content-Type", "application/json")
res, _ := http.DefaultClient.Do(req)
defer res.Body.Close()
body, _ := io.ReadAll(res.Body)
fmt.Println(string(body))
}HttpResponse<String> response = Unirest.post("https://api.example.com/api/vllm/deployments")
.header("Authorization", "Bearer <token>")
.header("Content-Type", "application/json")
.body("{\n \"name\": \"<string>\",\n \"finetune_job_id\": \"3c90c3cc-0d44-4b50-8888-8dd25736052a\",\n \"served_model_name\": \"<string>\",\n \"gpu_count\": 32,\n \"min_replicas\": 1,\n \"max_replicas\": 2,\n \"max_model_len\": 2,\n \"dtype\": \"<string>\",\n \"quality_score\": 123\n}")
.asString();require 'uri'
require 'net/http'
url = URI("https://api.example.com/api/vllm/deployments")
http = Net::HTTP.new(url.host, url.port)
http.use_ssl = true
request = Net::HTTP::Post.new(url)
request["Authorization"] = 'Bearer <token>'
request["Content-Type"] = 'application/json'
request.body = "{\n \"name\": \"<string>\",\n \"finetune_job_id\": \"3c90c3cc-0d44-4b50-8888-8dd25736052a\",\n \"served_model_name\": \"<string>\",\n \"gpu_count\": 32,\n \"min_replicas\": 1,\n \"max_replicas\": 2,\n \"max_model_len\": 2,\n \"dtype\": \"<string>\",\n \"quality_score\": 123\n}"
response = http.request(request)
puts response.read_body{
"deployment_id": "3c90c3cc-0d44-4b50-8888-8dd25736052a",
"name": "<string>",
"status": "pending",
"domain": "<string>",
"owner": "<string>",
"finetune_job_id": "3c90c3cc-0d44-4b50-8888-8dd25736052a",
"base_model": "<string>",
"registered_model_name": "<string>",
"served_model_name": "<string>",
"created_at": "2023-11-07T05:31:56Z",
"updated_at": "2023-11-07T05:31:56Z",
"model_version": "<string>",
"service_url": "<string>",
"replica_count": 123,
"ready_replicas": 123,
"router_endpoint_name": "<string>",
"router_model_name": "<string>",
"gpu_resources": {},
"error_message": "<string>",
"stopped_at": "2023-11-07T05:31:56Z"
}{
"error": "<string>",
"code": 500
}{
"error": "<string>",
"code": 500
}{
"error": "<string>",
"code": 500
}{
"detail": [
{
"loc": [
"<string>"
],
"msg": "<string>",
"type": "<string>",
"input": "<unknown>",
"ctx": {}
}
]
}{
"error": "<string>",
"code": 500
}{
"error": "<string>",
"code": 500
}Authorizations
OAuth2AuthorizationCodeBearerAPIKeyHeader
The access token received from the authorization server in the OAuth 2.0 flow.
Body
application/json
Deploy a fine-tuned model behind vLLM.
Deployment name (DNS-1123 label). Used for the K8s resources (serve-vllm-) and the router endpoint. Unique per served model.
A COMPLETE fine-tune training job whose adapter to serve.
Model id exposed via the router. Defaults to the job's registered model name.
Maximum string length:
200Override GPU count for the serving pod.
Required range:
1 <= x <= 64KEDA min replicas (0 enables scale-to-zero).
Required range:
x >= 0Required range:
x >= 1vLLM --max-model-len override.
Required range:
x >= 1vLLM --dtype (e.g. 'bfloat16', 'auto').
Router quality score for this model.
Response
Deployment created; auto-registered into the router when ready.
Available options:
pending, deploying, active, failed, stopped, archived Was this page helpful?

