DocumentationStreaming
Build
Streaming
The response appears as the model writes it. People read the first words instead of waiting for the end.
On this page
Turn on streaming
Set stream to true. The response arrives as server-sent events, each carrying a chunk of the text.
stream = client.chat.completions.create(
model="flash",
messages=[
{"role": "user", "content": "Draft a payment reminder."}
],
stream=True,
)
for chunk in stream:
if chunk.choices and chunk.choices[0].delta.content:
print(
chunk.choices[0].delta.content, end="", flush=True
)
if chunk.usage:
print("\n", chunk.usage.total_tokens, "tokens")const stream = await client.chat.completions.create({
model: "flash",
messages: [
{ role: "user", content: "Draft a payment reminder." },
],
stream: true,
})
for await (const chunk of stream) {
process.stdout.write(chunk.choices[0]?.delta?.content ?? "")
if (chunk.usage)
console.log("\n", chunk.usage.total_tokens, "tokens")
}import type { ChatCompletionChunk } from "openai/resources"
const stream: AsyncIterable<ChatCompletionChunk> =
await client.chat.completions.create({
model: "flash",
messages: [
{ role: "user", content: "Draft a payment reminder." },
],
stream: true,
})
for await (const chunk of stream) {
process.stdout.write(chunk.choices[0]?.delta?.content ?? "")
if (chunk.usage)
console.log("\n", chunk.usage.total_tokens, "tokens")
}curl -N https://api.learnya.ai/v1/chat/completions \
-H "Authorization: Bearer $LEARNYA_API_KEY" \
-H "Content-Type: application/json" \
-d '{
"model": "flash",
"stream": true,
"messages": [
{"role": "user", "content": "Draft a payment reminder."}
]
}'Response
data: {"id":"chatcmpl-91a7","object":"chat.completion.chunk","choices":[{"index":0,"delta":{"role":"assistant","content":""}}]}
data: {"id":"chatcmpl-91a7","object":"chat.completion.chunk","choices":[{"index":0,"delta":{"content":"Dear Ms Keller,"}}]}
data: {"id":"chatcmpl-91a7","object":"chat.completion.chunk","choices":[{"index":0,"delta":{},"finish_reason":"stop"}]}
data: {"id":"chatcmpl-91a7","object":"chat.completion.chunk","choices":[],"usage":{"prompt_tokens":17,"completion_tokens":96,"total_tokens":113}}
data: [DONE]Format
| Event | What it carries |
|---|---|
delta.content | The next chunk of text |
delta.reasoning_content | The reasoning, when you asked for it |
finish_reason | Why the response stopped, on the last chunk of text |
usage | The token count, in a final event with no choices |
[DONE] | The end of the stream |
Timeouts
- A stream that stays silent for two minutes is closed, with a final error event.
- A response lasts at most ten minutes.
- To stop a response, close the connection.