> ## Documentation Index
> Fetch the complete documentation index at: https://docs.nx1cloud.com/llms.txt
> Use this file to discover all available pages before exploring further.

# Deploy a fine-tuned model with vLLM

> Render + apply the vLLM manifests and register the model on readiness.



## OpenAPI

````yaml post /api/vllm/deployments
openapi: 3.1.0
info:
  title: Nx1 AI API
  description: |

    AI API for Nx1 Data Platform Management and Automated Data Tasks.

    Authentication is required via PSK in Authorization header.

    Default PSK is | [ask a friend] |
  version: 0.10.2
servers: []
security: []
paths:
  /api/vllm/deployments:
    post:
      tags:
        - Inferencing
      summary: Deploy a fine-tuned model with vLLM
      description: Render + apply the vLLM manifests and register the model on readiness.
      operationId: create_deployment_api_vllm_deployments_post
      requestBody:
        required: true
        content:
          application/json:
            schema:
              $ref: '#/components/schemas/VllmDeploymentRequest'
      responses:
        '201':
          description: Deployment created; auto-registered into the router when ready.
          content:
            application/json:
              schema:
                $ref: '#/components/schemas/VllmDeploymentResponse'
        '400':
          content:
            application/json:
              schema:
                $ref: '#/components/schemas/ErrorResponse'
          description: Bad Request
        '403':
          content:
            application/json:
              schema:
                $ref: '#/components/schemas/ErrorResponse'
          description: Forbidden
        '409':
          description: Name in use or job not COMPLETE.
          content:
            application/json:
              schema:
                $ref: '#/components/schemas/ErrorResponse'
        '422':
          description: Validation Error
          content:
            application/json:
              schema:
                $ref: '#/components/schemas/HTTPValidationError'
        '502':
          description: K8s apply failed.
          content:
            application/json:
              schema:
                $ref: '#/components/schemas/ErrorResponse'
        '503':
          description: Serving/puller image not configured.
          content:
            application/json:
              schema:
                $ref: '#/components/schemas/ErrorResponse'
      security:
        - OAuth2AuthorizationCodeBearer: []
        - APIKeyHeader: []
components:
  schemas:
    VllmDeploymentRequest:
      properties:
        name:
          type: string
          title: Name
          description: >-
            Deployment name (DNS-1123 label). Used for the K8s resources
            (serve-vllm-<name>) and the router endpoint. Unique per served
            model.
        finetune_job_id:
          type: string
          format: uuid
          title: Finetune Job Id
          description: A COMPLETE fine-tune training job whose adapter to serve.
        served_model_name:
          anyOf:
            - type: string
              maxLength: 200
            - type: 'null'
          title: Served Model Name
          description: >-
            Model id exposed via the router. Defaults to the job's registered
            model name.
        gpu_count:
          anyOf:
            - type: integer
              maximum: 64
              minimum: 1
            - type: 'null'
          title: Gpu Count
          description: Override GPU count for the serving pod.
        min_replicas:
          anyOf:
            - type: integer
              minimum: 0
            - type: 'null'
          title: Min Replicas
          description: KEDA min replicas (0 enables scale-to-zero).
        max_replicas:
          anyOf:
            - type: integer
              minimum: 1
            - type: 'null'
          title: Max Replicas
        max_model_len:
          anyOf:
            - type: integer
              minimum: 1
            - type: 'null'
          title: Max Model Len
          description: vLLM --max-model-len override.
        dtype:
          anyOf:
            - type: string
            - type: 'null'
          title: Dtype
          description: vLLM --dtype (e.g. 'bfloat16', 'auto').
        quality_score:
          anyOf:
            - type: number
            - type: 'null'
          title: Quality Score
          description: Router quality score for this model.
      type: object
      required:
        - name
        - finetune_job_id
      title: VllmDeploymentRequest
      description: Deploy a fine-tuned model behind vLLM.
    VllmDeploymentResponse:
      properties:
        deployment_id:
          type: string
          format: uuid
          title: Deployment Id
        name:
          type: string
          title: Name
        status:
          $ref: '#/components/schemas/VllmDeploymentStatus'
        domain:
          type: string
          title: Domain
        owner:
          type: string
          title: Owner
        finetune_job_id:
          type: string
          format: uuid
          title: Finetune Job Id
        base_model:
          type: string
          title: Base Model
        registered_model_name:
          type: string
          title: Registered Model Name
        model_version:
          anyOf:
            - type: string
            - type: 'null'
          title: Model Version
        served_model_name:
          type: string
          title: Served Model Name
        service_url:
          anyOf:
            - type: string
            - type: 'null'
          title: Service Url
        replica_count:
          anyOf:
            - type: integer
            - type: 'null'
          title: Replica Count
        ready_replicas:
          anyOf:
            - type: integer
            - type: 'null'
          title: Ready Replicas
        router_endpoint_name:
          anyOf:
            - type: string
            - type: 'null'
          title: Router Endpoint Name
        router_model_name:
          anyOf:
            - type: string
            - type: 'null'
          title: Router Model Name
        gpu_resources:
          anyOf:
            - additionalProperties: true
              type: object
            - type: 'null'
          title: Gpu Resources
        error_message:
          anyOf:
            - type: string
            - type: 'null'
          title: Error Message
        created_at:
          type: string
          format: date-time
          title: Created At
        updated_at:
          type: string
          format: date-time
          title: Updated At
        stopped_at:
          anyOf:
            - type: string
              format: date-time
            - type: 'null'
          title: Stopped At
      type: object
      required:
        - deployment_id
        - name
        - status
        - domain
        - owner
        - finetune_job_id
        - base_model
        - registered_model_name
        - served_model_name
        - created_at
        - updated_at
      title: VllmDeploymentResponse
    ErrorResponse:
      properties:
        error:
          type: string
          title: Error
          description: A brief description of the error that occurred.
        code:
          type: integer
          title: Code
          description: The HTTP status code associated with the error.
          default: 500
      type: object
      required:
        - error
      title: ErrorResponse
    HTTPValidationError:
      properties:
        detail:
          items:
            $ref: '#/components/schemas/ValidationError'
          type: array
          title: Detail
      type: object
      title: HTTPValidationError
    VllmDeploymentStatus:
      type: string
      enum:
        - pending
        - deploying
        - active
        - failed
        - stopped
        - archived
      title: VllmDeploymentStatus
    ValidationError:
      properties:
        loc:
          items:
            anyOf:
              - type: string
              - type: integer
          type: array
          title: Location
        msg:
          type: string
          title: Message
        type:
          type: string
          title: Error Type
        input:
          title: Input
        ctx:
          type: object
          title: Context
      type: object
      required:
        - loc
        - msg
        - type
      title: ValidationError
  securitySchemes:
    OAuth2AuthorizationCodeBearer:
      type: oauth2
      flows:
        authorizationCode:
          scopes: {}
          authorizationUrl: >-
            https://sso-rapid.rapid.nx1cloud.com/realms/rapid/protocol/openid-connect/auth
          tokenUrl: >-
            https://sso-rapid.rapid.nx1cloud.com/realms/rapid/protocol/openid-connect/token
    APIKeyHeader:
      type: apiKey
      in: header
      name: Authorization-PSK

````