> ## Documentation Index
> Fetch the complete documentation index at: https://docs.cubeuaeai.com/llms.txt
> Use this file to discover all available pages before exploring further.

# Chat Completion

> Generate a model response for the given chat conversation.



## OpenAPI

````yaml en/api-reference/endpoints/openapi/openai/openapi.yaml POST /v1/chat/completions
openapi: 3.1.0
info:
  title: OpenAI-compatible endpoint
  description: >-
    Call the OpenAI-compatible chat completions API. All chat models are
    supported.
  version: 1.0.0
servers:
  - url: https://ai.cubeuaeai.com
security:
  - bearerAuth: []
paths:
  /v1/chat/completions:
    post:
      summary: Chat completion
      description: Generate a model response from the conversation context.
      operationId: createChatCompletionCompatible
      requestBody:
        required: true
        content:
          application/json:
            schema:
              type: object
              required:
                - model
                - messages
              properties:
                model:
                  type: string
                  enum:
                    - claude-sonnet-4-6
                    - claude-sonnet-4-5-20250929
                    - claude-sonnet-4-20250514
                    - claude-opus-4-6
                    - claude-opus-4-5-20251101
                    - claude-opus-4-1-20250805
                    - claude-opus-4-20250514
                    - claude-haiku-4-5-20251001
                    - claude-3-7-sonnet-20250219
                    - claude-3-5-sonnet-20241022
                    - gpt-5-4
                    - gpt-5-3-codex
                    - gpt-5-2
                    - gpt-5-2-chat
                    - gpt-5-2-chat-latest
                    - gpt-5-2-codex
                    - gpt-5-1
                    - gpt-5-1-chat
                    - gpt-5-1-chat-latest
                    - gpt-5-1-codex-mini
                    - gpt-5
                    - gpt-5-chat-latest
                    - gpt-5-pro
                    - gpt-5-codex
                    - gpt-5-codex-high
                    - gpt-5-codex-low
                    - gpt-5-mini
                    - gpt-5-nano
                    - gpt-4o
                    - gpt-4o-mini
                    - gpt-4-1
                    - gpt-4
                    - deepseek-v3-2-speciale
                    - deepseek-v3-2
                    - deepseek-v3-2-exp
                    - deepseek-v3-2-251201
                    - deepseek-v3-1-terminus
                    - deepseek-v3-1
                    - deepseek-v3
                    - qwen3.5-397b-a17b
                    - qwen3-coder-next
                    - qwen3-coder
                    - qwen3-235b-a22b
                    - qwen3-32b
                    - qwen3-14b
                    - qwen2.5-72b-instruct
                    - doubao-seed-2.0-pro
                    - doubao-seed-2.0-code
                    - doubao-seed-2.0-lite
                    - doubao-seed-2.0-mini
                    - doubao-seed-1-8-251228
                    - doubao-seed-1-6-flash-250828
                    - doubao-seed-1-6-251015
                    - doubao-seed-1-6-lite-251015
                    - doubao-seed-1-6-vision-250815
                    - minimax-m2.5
                    - minimax-m2.1
                    - kimi-k2.5
                    - grok-3
                    - grok-3-mini
                    - grok-4-fast-reasoning
                    - grok-4-fast-non-reasoning
                    - grok-4-1-fast-reasoning
                    - grok-4-1-fast-non-reasoning
                    - gemini-3.1-pro-preview
                    - gemini-3.1-flash-image-preview
                    - gemini-3-pro-preview
                    - gemini-3-flash-preview
                    - gemini-2.5-pro
                    - gemini-2.5-flash
                    - gemini-2.0-flash
                    - glm-4.7
                  description: The model ID used for completion.
                  example: claude-sonnet-4-6
                messages:
                  type: array
                  minItems: 1
                  description: The list of messages that make up the current conversation.
                  items:
                    type: object
                    required:
                      - role
                      - content
                    properties:
                      role:
                        type: string
                        enum:
                          - system
                          - user
                          - assistant
                      content:
                        oneOf:
                          - type: string
                          - type: array
                            items:
                              type: object
                  example:
                    - role: user
                      content: Hello!
                max_tokens:
                  type: integer
                  minimum: 1
                  description: Maximum number of tokens to generate in the chat completion.
                temperature:
                  type: number
                  minimum: 0
                  maximum: 2
                  default: 1
                  description: >-
                    Sampling temperature, from 0 to 2. Higher values make the
                    output more random; lower values make it more deterministic.
                  example: 1
                top_p:
                  type: number
                  minimum: 0
                  maximum: 1
                  description: >-
                    Nucleus sampling threshold, as an alternative to
                    temperature.
                frequency_penalty:
                  type: number
                  minimum: -2
                  maximum: 2
                  default: 0
                  description: >-
                    Penalize new tokens based on how often they already appear
                    in the text.
                presence_penalty:
                  type: number
                  minimum: -2
                  maximum: 2
                  default: 0
                  description: >-
                    Penalize new tokens if they have already appeared in the
                    text.
                stream:
                  type: boolean
                  default: false
                  description: >-
                    If true, partial messages are streamed via server-sent
                    events (SSE).
                stop:
                  oneOf:
                    - type: string
                    - type: array
                      items:
                        type: string
                  description: >-
                    Up to 4 sequences. The API stops generating further tokens
                    when these sequences appear.
                'n':
                  type: integer
                  minimum: 1
                  default: 1
                  description: >-
                    How many completion choices to generate for each input
                    message.
                response_format:
                  type: object
                  properties:
                    type:
                      type: string
                      enum:
                        - text
                        - json_object
                  description: >-
                    Specify the format the model must output. Set {"type":
                    "json_object"} to enable JSON mode.
                tools:
                  type: array
                  items:
                    type: object
                  description: >-
                    A list of tools the model can call. Currently only functions
                    are supported as tools.
                tool_choice:
                  oneOf:
                    - type: string
                    - type: object
                  description: Controls which tool the model calls, if any.
                user:
                  type: string
                  description: A unique identifier representing the end user.
      responses:
        '200':
          description: Completion generated successfully
          content:
            application/json:
              schema:
                $ref: '#/components/schemas/ChatCompletion'
        '400':
          description: Invalid request
          content:
            application/json:
              schema:
                $ref: '#/components/schemas/Error'
        '401':
          description: Unauthorized
          content:
            application/json:
              schema:
                $ref: '#/components/schemas/Error'
        '429':
          description: Rate limit exceeded
          content:
            application/json:
              schema:
                $ref: '#/components/schemas/Error'
components:
  schemas:
    ChatCompletion:
      type: object
      required:
        - id
        - object
        - created
        - model
        - choices
      properties:
        id:
          type: string
          example: chatcmpl-abc123
        object:
          type: string
          example: chat.completion
        created:
          type: integer
          description: Unix timestamp when the completion was created.
        model:
          type: string
          example: claude-sonnet-4-6
        choices:
          type: array
          items:
            type: object
            properties:
              index:
                type: integer
              message:
                type: object
                properties:
                  role:
                    type: string
                    example: assistant
                  content:
                    type: string
                    example: Hello! How can I help you?
              finish_reason:
                type: string
                enum:
                  - stop
                  - length
                  - content_filter
                  - tool_calls
                  - null
        usage:
          type: object
          properties:
            prompt_tokens:
              type: integer
            completion_tokens:
              type: integer
            total_tokens:
              type: integer
    Error:
      type: object
      properties:
        error:
          type: object
          properties:
            code:
              type: string
            message:
              type: string
  securitySchemes:
    bearerAuth:
      type: http
      scheme: bearer

````