Core Concepts
Streaming allows you to receive tokens as soon as they are emitted by Cortiqa LPU inference nodes. Instead of waiting for full generation, chunks stream incrementally over Server-Sent Events (SSE) with sub-second time-to-first-token.
When streaming is enabled (stream: true), Cortiqa returns an event-stream containing incremental text delta chunks. All official Cortiqa SDKs feature built-in async iterators to handle streaming automatically.
openai/gpt-oss-120b. Streaming delivers the fastest interactive experience for your users.Using the official cortiqa Python library:
1import cortiqa23client = cortiqa.Client()45stream = client.chat.create(6 model="openai/gpt-oss-120b",7 messages=[8 {"role": "system", "content": "You are a helpful AI assistant powered by Cortiqa."},9 {"role": "user", "content": "Explain quantum entanglement in simple terms."}10 ],11 stream=True12)1314for chunk in stream:15 if chunk.choices and chunk.choices[0].delta.content:16 print(chunk.choices[0].delta.content, end="", flush=True)Using the official @cortiqa/sdk:
1import { Cortiqa } from "@cortiqa/sdk";23const cortiqa = new Cortiqa();45async function main() {6 const stream = await cortiqa.chat.createStream({7 model: "openai/gpt-oss-120b",8 messages: [9 { role: "user", content: "Write a short poem about distributed computing." }10 ],11 });1213 for await (const chunk of stream) {14 process.stdout.write(chunk.content || "");15 }16}1718main();1package main23import (4 "context"5 "fmt"6 "os"78 "github.com/cortiqa-ai/cortiqa-go"9)1011func main() {12 client := cortiqa.NewClient(cortiqa.WithAPIKey(os.Getenv("CORTIQA_API_KEY")))1314 stream, err := client.Chat.CreateStream(context.Background(), &cortiqa.ChatRequest{15 Model: "openai/gpt-oss-120b",16 Messages: []cortiqa.Message{17 {Role: "user", Content: "Give me 3 tips for cloud security."},18 },19 })20 if err != nil {21 panic(err)22 }2324 for chunk := range stream {25 fmt.Print(chunk.Delta)26 }27}Direct HTTP request sending "stream": true to https://api.cortiqa.co/api/v1/chat/completions:
curl https://api.cortiqa.co/api/v1/chat/completions \
-H "Authorization: Bearer sk-cortiqa-YOUR_API_KEY" \
-H "Content-Type: application/json" \
-d '{
"model": "openai/gpt-oss-120b",
"messages": [{"role": "user", "content": "Hi"}],
"stream": true
}'Raw response chunks arrive as Server-Sent Events:
data: {"id":"chatcmpl-91a...","object":"chat.completion.chunk","model":"openai/gpt-oss-120b","choices":[{"index":0,"delta":{"role":"assistant","content":""},"finish_reason":null}]}
data: {"id":"chatcmpl-91a...","object":"chat.completion.chunk","model":"openai/gpt-oss-120b","choices":[{"index":0,"delta":{"content":"Hello"},"finish_reason":null}]}
data: {"id":"chatcmpl-91a...","object":"chat.completion.chunk","model":"openai/gpt-oss-120b","choices":[{"index":0,"delta":{"content":"! How"},"finish_reason":null}]}
data: {"id":"chatcmpl-91a...","object":"chat.completion.chunk","model":"openai/gpt-oss-120b","choices":[{"index":0,"delta":{"content":" can I help you today?"},"finish_reason":"stop"}]}
data: [DONE]