from openai import OpenAI
import os
# 💥 Create an OpenAI client, pointed at Muna
openai = OpenAI(
api_key=os.environ["MUNA_API_KEY"],
base_url="https://inference.muna.ai/v1"
)
# 🔥 Create a chat completion
completion = openai.chat.completions.create(
model="@qwen/qwen-3.8-27b",
messages=[{ "role": "user", "content": "What is the capital of France?" }]
)
# 🚀 Print the result
print(completion.choices[0].message)
import OpenAI from "openai"
// 💥 Create an OpenAI client, pointed at Muna
const openai = new OpenAI({
apiKey: process.env.MUNA_API_KEY,
baseURL: "https://inference.muna.ai/v1"
});
// 🔥 Create a chat completion
const completion = await openai.chat.completions.create({
model: "@qwen/qwen-3.8-27b",
messages: [{ role: "user", content: "What is the capital of France?" }]
});
// 🚀 Print the result
console.log(completion.choices[0]);
# 🔥 Stream a chat completion
stream = openai.chat.completions.create(
model="@qwen/qwen-3.8-27b",
messages=[{ "role": "user", "content": "What is life?" }],
stream=True
)
# 🚀 Use completion chunks
for chunk in stream:
...
curl https://inference.muna.ai/v1/chat/completions \
-H "Authorization: Bearer $MUNA_API_KEY" \
-H "Content-Type: application/json" \
-d '{
"model": "@qwen/qwen-3.8-27b",
"messages": [{ "role": "user", "content": "What is the capital of France?" }]
}'
{
"id": "chatcmpl-6f1c2b9e-4d3a-4b8e-9c1f-2a7d5e8b0c34",
"object": "chat.completion",
"created": 1791516000,
"model": "@qwen/qwen-3.8-27b",
"choices": [
{
"index": 0,
"message": {
"role": "assistant",
"content": "The capital of France is Paris."
},
"finish_reason": "stop"
}
],
"usage": {
"prompt_tokens": 16,
"completion_tokens": 8,
"total_tokens": 24,
"prompt_tokens_details": { "cached_tokens": 0 }
}
}
OpenAI
Create a Chat Completion
POST
/
v1
/
chat
/
completions
from openai import OpenAI
import os
# 💥 Create an OpenAI client, pointed at Muna
openai = OpenAI(
api_key=os.environ["MUNA_API_KEY"],
base_url="https://inference.muna.ai/v1"
)
# 🔥 Create a chat completion
completion = openai.chat.completions.create(
model="@qwen/qwen-3.8-27b",
messages=[{ "role": "user", "content": "What is the capital of France?" }]
)
# 🚀 Print the result
print(completion.choices[0].message)
import OpenAI from "openai"
// 💥 Create an OpenAI client, pointed at Muna
const openai = new OpenAI({
apiKey: process.env.MUNA_API_KEY,
baseURL: "https://inference.muna.ai/v1"
});
// 🔥 Create a chat completion
const completion = await openai.chat.completions.create({
model: "@qwen/qwen-3.8-27b",
messages: [{ role: "user", content: "What is the capital of France?" }]
});
// 🚀 Print the result
console.log(completion.choices[0]);
# 🔥 Stream a chat completion
stream = openai.chat.completions.create(
model="@qwen/qwen-3.8-27b",
messages=[{ "role": "user", "content": "What is life?" }],
stream=True
)
# 🚀 Use completion chunks
for chunk in stream:
...
curl https://inference.muna.ai/v1/chat/completions \
-H "Authorization: Bearer $MUNA_API_KEY" \
-H "Content-Type: application/json" \
-d '{
"model": "@qwen/qwen-3.8-27b",
"messages": [{ "role": "user", "content": "What is the capital of France?" }]
}'
{
"id": "chatcmpl-6f1c2b9e-4d3a-4b8e-9c1f-2a7d5e8b0c34",
"object": "chat.completion",
"created": 1791516000,
"model": "@qwen/qwen-3.8-27b",
"choices": [
{
"index": 0,
"message": {
"role": "assistant",
"content": "The capital of France is Paris."
},
"finish_reason": "stop"
}
],
"usage": {
"prompt_tokens": 16,
"completion_tokens": 8,
"total_tokens": 24,
"prompt_tokens_details": { "cached_tokens": 0 }
}
}
Creates a chat completion. Pass
stream: true to stream the completion as server-sent events.
from openai import OpenAI
import os
# 💥 Create an OpenAI client, pointed at Muna
openai = OpenAI(
api_key=os.environ["MUNA_API_KEY"],
base_url="https://inference.muna.ai/v1"
)
# 🔥 Create a chat completion
completion = openai.chat.completions.create(
model="@qwen/qwen-3.8-27b",
messages=[{ "role": "user", "content": "What is the capital of France?" }]
)
# 🚀 Print the result
print(completion.choices[0].message)
import OpenAI from "openai"
// 💥 Create an OpenAI client, pointed at Muna
const openai = new OpenAI({
apiKey: process.env.MUNA_API_KEY,
baseURL: "https://inference.muna.ai/v1"
});
// 🔥 Create a chat completion
const completion = await openai.chat.completions.create({
model: "@qwen/qwen-3.8-27b",
messages: [{ role: "user", content: "What is the capital of France?" }]
});
// 🚀 Print the result
console.log(completion.choices[0]);
# 🔥 Stream a chat completion
stream = openai.chat.completions.create(
model="@qwen/qwen-3.8-27b",
messages=[{ "role": "user", "content": "What is life?" }],
stream=True
)
# 🚀 Use completion chunks
for chunk in stream:
...
curl https://inference.muna.ai/v1/chat/completions \
-H "Authorization: Bearer $MUNA_API_KEY" \
-H "Content-Type: application/json" \
-d '{
"model": "@qwen/qwen-3.8-27b",
"messages": [{ "role": "user", "content": "What is the capital of France?" }]
}'
{
"id": "chatcmpl-6f1c2b9e-4d3a-4b8e-9c1f-2a7d5e8b0c34",
"object": "chat.completion",
"created": 1791516000,
"model": "@qwen/qwen-3.8-27b",
"choices": [
{
"index": 0,
"message": {
"role": "assistant",
"content": "The capital of France is Paris."
},
"finish_reason": "stop"
}
],
"usage": {
"prompt_tokens": 16,
"completion_tokens": 8,
"total_tokens": 24,
"prompt_tokens_details": { "cached_tokens": 0 }
}
}
Body
string
required
Model tag.
Message[]
required
Messages comprising the conversation so far.
Show properties
Show properties
string
required
Message role:
system, user, assistant, or tool.string | ContentPart[]
Message content. Content parts can be
text or image_url, for models that accept images.ToolCall[]
Tool calls made by the model, on
assistant messages.string
Tool call that this message responds to, on
tool messages.boolean
Whether to stream the completion as server-sent events. Defaults to
false.Tool[]
Function tools the model may call.
string
Tool choice mode:
auto or none. Defaults to auto.string
Reasoning effort for reasoning models:
none, minimal, low, medium, high, or xhigh.integer
Maximum number of tokens to generate. Also accepted as
max_tokens.number
Sampling temperature.
number
Nucleus sampling coefficient.
integer
Sampling seed for reproducible outputs.
Response
string
required
Chat completion identifier.
string
required
Object type, always
chat.completion.integer
required
Unix timestamp, in seconds, when the completion was created.
string
required
Model tag.
Choice[]
required
Generated completion choices.
Show properties
Show properties
integer
required
Choice index.
Message
required
string
Reason the model stopped generating:
stop, length, or tool_calls.Usage
Token usage.