> ## Documentation Index
> Fetch the complete documentation index at: https://docs.muna.ai/llms.txt
> Use this file to discover all available pages before exploring further.

# Create a Message

Creates a message. Pass `stream: true` to stream the message as server-sent events.

<RequestExample>
  ```py Python theme={null}
  from anthropic import Anthropic
  import os

  # 💥 Create an Anthropic client, pointed at Muna
  anthropic = Anthropic(
    api_key=os.environ["MUNA_API_KEY"],
    base_url="https://inference.muna.ai"
  )

  # 🔥 Create a message
  message = anthropic.messages.create(
    model="@qwen/qwen-3.8-27b",
    max_tokens=1024,
    messages=[{ "role": "user", "content": "What is the capital of France?" }]
  )

  # 🚀 Print the result
  print(message.content[0].text)
  ```

  ```ts JavaScript theme={null}
  import Anthropic from "@anthropic-ai/sdk"

  // 💥 Create an Anthropic client, pointed at Muna
  const anthropic = new Anthropic({
    apiKey: process.env.MUNA_API_KEY,
    baseURL: "https://inference.muna.ai"
  });

  // 🔥 Create a message
  const message = await anthropic.messages.create({
    model: "@qwen/qwen-3.8-27b",
    max_tokens: 1024,
    messages: [{ role: "user", content: "What is the capital of France?" }]
  });

  // 🚀 Print the result
  console.log(message.content[0].text);
  ```

  ```py Python (streaming) theme={null}
  # 🔥 Stream a message
  with anthropic.messages.stream(
    model="@qwen/qwen-3.8-27b",
    max_tokens=1024,
    messages=[{ "role": "user", "content": "What is life?" }]
  ) as stream:
    # 🚀 Use text deltas
    for text in stream.text_stream:
      ...
  ```

  ```bash curl theme={null}
  curl https://inference.muna.ai/v1/messages \
    -H "x-api-key: $MUNA_API_KEY" \
    -H "anthropic-version: 2023-06-01" \
    -H "Content-Type: application/json" \
    -d '{
      "model": "@qwen/qwen-3.8-27b",
      "max_tokens": 1024,
      "messages": [{ "role": "user", "content": "What is the capital of France?" }]
    }'
  ```
</RequestExample>

<ResponseExample>
  ```json Response theme={null}
  {
    "id": "msg_6f1c2b9e4d3a4b8e9c1f2a7d",
    "type": "message",
    "role": "assistant",
    "model": "@qwen/qwen-3.8-27b",
    "content": [
      { "type": "text", "text": "The capital of France is Paris." }
    ],
    "stop_reason": "end_turn",
    "stop_sequence": null,
    "usage": {
      "input_tokens": 16,
      "output_tokens": 8,
      "cache_read_input_tokens": 0
    }
  }
  ```
</ResponseExample>

### Body

<ParamField body="model" type="string" required>
  Model tag.
</ParamField>

<ParamField body="max_tokens" type="integer" required>
  Maximum number of tokens to generate.
</ParamField>

<ParamField body="messages" type="MessageParam[]" required>
  Input messages. Content blocks can be `text`, `image`, `tool_use`, or `tool_result`.
</ParamField>

<ParamField body="system" type="string | TextBlock[]">
  System prompt.
</ParamField>

<ParamField body="stream" type="boolean">
  Whether to stream the message as server-sent events. Defaults to `false`.
</ParamField>

<ParamField body="tools" type="Tool[]">
  Tools the model may call.
</ParamField>

<ParamField body="thinking" type="ThinkingConfig">
  Extended thinking configuration: `{ "type": "disabled" }`, `{ "type": "enabled", "budget_tokens": N }`,
  or `{ "type": "adaptive" }`. Mapped onto the model's reasoning effort.
</ParamField>

<ParamField body="output_config.effort" type="string">
  Reasoning effort: `low`, `medium`, `high`, `xhigh`, or `max`.
</ParamField>

<ParamField body="temperature" type="number">
  Sampling temperature.
</ParamField>

<ParamField body="top_p" type="number">
  Nucleus sampling coefficient.
</ParamField>

<ParamField body="top_k" type="integer">
  Only sample from the top K options for each token. Ignored by models that do not support it.
</ParamField>

<ParamField body="stop_sequences" type="string[]">
  Custom text sequences that stop generation. Ignored by models that do not support it.
</ParamField>

### Response

<ResponseField name="id" type="string" required>
  Message identifier.
</ResponseField>

<ResponseField name="type" type="string" required>
  Object type, always `message`.
</ResponseField>

<ResponseField name="role" type="string" required>
  Message role, always `assistant`.
</ResponseField>

<ResponseField name="model" type="string" required>
  Model tag.
</ResponseField>

<ResponseField name="content" type="ContentBlock[]" required>
  Generated content. Blocks can be `text`, `thinking`, or `tool_use`.
</ResponseField>

<ResponseField name="stop_reason" type="string">
  Reason the model stopped generating: `end_turn`, `max_tokens`, `stop_sequence`, or `tool_use`.
</ResponseField>

<ResponseField name="stop_sequence" type="string">
  Custom stop sequence that was generated, if any.
</ResponseField>

<ResponseField name="usage" type="Usage" required>
  Token usage.

  <Expandable title="properties">
    <ResponseField name="input_tokens" type="integer">
      Number of input tokens, excluding cached tokens.
    </ResponseField>

    <ResponseField name="output_tokens" type="integer" required>
      Number of output tokens.
    </ResponseField>

    <ResponseField name="cache_read_input_tokens" type="integer">
      Number of input tokens served from cache, billed at the cached input rate.
    </ResponseField>

    <ResponseField name="cache_creation_input_tokens" type="integer">
      Number of input tokens written to cache.
    </ResponseField>
  </Expandable>
</ResponseField>


This documentation is built and hosted on [Mintlify](https://mintlify.com), a developer documentation platform.