> ## Documentation Index
> Fetch the complete documentation index at: https://docs.muna.ai/llms.txt
> Use this file to discover all available pages before exploring further.

# Create a Chat Completion

Creates a chat completion. Pass `stream: true` to stream the completion as server-sent events.

<RequestExample>
  ```py Python theme={null}
  from openai import OpenAI
  import os

  # 💥 Create an OpenAI client, pointed at Muna
  openai = OpenAI(
    api_key=os.environ["MUNA_API_KEY"],
    base_url="https://inference.muna.ai/v1"
  )

  # 🔥 Create a chat completion
  completion = openai.chat.completions.create(
    model="@qwen/qwen-3.8-27b",
    messages=[{ "role": "user", "content": "What is the capital of France?" }]
  )

  # 🚀 Print the result
  print(completion.choices[0].message)
  ```

  ```ts JavaScript theme={null}
  import OpenAI from "openai"

  // 💥 Create an OpenAI client, pointed at Muna
  const openai = new OpenAI({
    apiKey: process.env.MUNA_API_KEY,
    baseURL: "https://inference.muna.ai/v1"
  });

  // 🔥 Create a chat completion
  const completion = await openai.chat.completions.create({
    model: "@qwen/qwen-3.8-27b",
    messages: [{ role: "user", content: "What is the capital of France?" }]
  });

  // 🚀 Print the result
  console.log(completion.choices[0]);
  ```

  ```py Python (streaming) theme={null}
  # 🔥 Stream a chat completion
  stream = openai.chat.completions.create(
    model="@qwen/qwen-3.8-27b",
    messages=[{ "role": "user", "content": "What is life?" }],
    stream=True
  )

  # 🚀 Use completion chunks
  for chunk in stream:
    ...
  ```

  ```bash curl theme={null}
  curl https://inference.muna.ai/v1/chat/completions \
    -H "Authorization: Bearer $MUNA_API_KEY" \
    -H "Content-Type: application/json" \
    -d '{
      "model": "@qwen/qwen-3.8-27b",
      "messages": [{ "role": "user", "content": "What is the capital of France?" }]
    }'
  ```
</RequestExample>

<ResponseExample>
  ```json Response theme={null}
  {
    "id": "chatcmpl-6f1c2b9e-4d3a-4b8e-9c1f-2a7d5e8b0c34",
    "object": "chat.completion",
    "created": 1791516000,
    "model": "@qwen/qwen-3.8-27b",
    "choices": [
      {
        "index": 0,
        "message": {
          "role": "assistant",
          "content": "The capital of France is Paris."
        },
        "finish_reason": "stop"
      }
    ],
    "usage": {
      "prompt_tokens": 16,
      "completion_tokens": 8,
      "total_tokens": 24,
      "prompt_tokens_details": { "cached_tokens": 0 }
    }
  }
  ```
</ResponseExample>

### Body

<ParamField body="model" type="string" required>
  Model tag.
</ParamField>

<ParamField body="messages" type="Message[]" required>
  Messages comprising the conversation so far.

  <Expandable title="properties">
    <ParamField body="role" type="string" required>
      Message role: `system`, `user`, `assistant`, or `tool`.
    </ParamField>

    <ParamField body="content" type="string | ContentPart[]">
      Message content. Content parts can be `text` or `image_url`, for models that accept images.
    </ParamField>

    <ParamField body="tool_calls" type="ToolCall[]">
      Tool calls made by the model, on `assistant` messages.
    </ParamField>

    <ParamField body="tool_call_id" type="string">
      Tool call that this message responds to, on `tool` messages.
    </ParamField>
  </Expandable>
</ParamField>

<ParamField body="stream" type="boolean">
  Whether to stream the completion as server-sent events. Defaults to `false`.
</ParamField>

<ParamField body="tools" type="Tool[]">
  Function tools the model may call.
</ParamField>

<ParamField body="tool_choice" type="string">
  Tool choice mode: `auto` or `none`. Defaults to `auto`.
</ParamField>

<ParamField body="reasoning_effort" type="string">
  Reasoning effort for reasoning models: `none`, `minimal`, `low`, `medium`, `high`, or `xhigh`.
</ParamField>

<ParamField body="max_completion_tokens" type="integer">
  Maximum number of tokens to generate. Also accepted as `max_tokens`.
</ParamField>

<ParamField body="temperature" type="number">
  Sampling temperature.
</ParamField>

<ParamField body="top_p" type="number">
  Nucleus sampling coefficient.
</ParamField>

<ParamField body="seed" type="integer">
  Sampling seed for reproducible outputs.
</ParamField>

### Response

<ResponseField name="id" type="string" required>
  Chat completion identifier.
</ResponseField>

<ResponseField name="object" type="string" required>
  Object type, always `chat.completion`.
</ResponseField>

<ResponseField name="created" type="integer" required>
  Unix timestamp, in seconds, when the completion was created.
</ResponseField>

<ResponseField name="model" type="string" required>
  Model tag.
</ResponseField>

<ResponseField name="choices" type="Choice[]" required>
  Generated completion choices.

  <Expandable title="properties">
    <ResponseField name="index" type="integer" required>
      Choice index.
    </ResponseField>

    <ResponseField name="message" type="Message" required>
      Generated message.

      <Expandable title="properties">
        <ResponseField name="role" type="string" required>
          Message role, always `assistant`.
        </ResponseField>

        <ResponseField name="content" type="string">
          Message content.
        </ResponseField>

        <ResponseField name="reasoning_content" type="string">
          Model reasoning, for reasoning models.
        </ResponseField>

        <ResponseField name="tool_calls" type="ToolCall[]">
          Tool calls made by the model.
        </ResponseField>
      </Expandable>
    </ResponseField>

    <ResponseField name="finish_reason" type="string">
      Reason the model stopped generating: `stop`, `length`, or `tool_calls`.
    </ResponseField>
  </Expandable>
</ResponseField>

<ResponseField name="usage" type="Usage">
  Token usage.

  <Expandable title="properties">
    <ResponseField name="prompt_tokens" type="integer" required>
      Number of tokens in the prompt.
    </ResponseField>

    <ResponseField name="completion_tokens" type="integer" required>
      Number of tokens in the completion.
    </ResponseField>

    <ResponseField name="total_tokens" type="integer" required>
      Total number of tokens.
    </ResponseField>

    <ResponseField name="prompt_tokens_details.cached_tokens" type="integer">
      Number of prompt tokens served from cache, billed at the cached input rate.
    </ResponseField>

    <ResponseField name="completion_tokens_details.reasoning_tokens" type="integer">
      Number of completion tokens spent on reasoning.
    </ResponseField>
  </Expandable>
</ResponseField>


This documentation is built and hosted on [Mintlify](https://mintlify.com), a developer documentation platform.