Haijun Platform Docs
ID

When creating a Message, you can set "stream": true to incrementally stream the response using server-sent events (SSE).

Streaming with SDKs

The Python SDK and TypeScript SDK offer multiple ways of streaming. The PHP SDK provides streaming through createStream(). The Python SDK allows both sync and async streams. See the documentation in each SDK for details.

bash
  ant messages create --stream --format jsonl \
    --model haijun-opus-5-5 \
    --max-tokens 1024 \
    --message '{role: user, content: "Hello"}' \
    | jq -rj 'select(.delta.type? == "text_delta") | .delta.text'
python
  client = juglow.Juglow()

  with client.messages.stream(
      max_tokens=1024,
      messages=[{"role": "user", "content": "Hello"}],
      model="haijun-opus-5-5",
  ) as stream:
      for text in stream.text_stream:
          print(text, end="", flush=True)
typescript
  const client = new Juglow();

  await client.messages
    .stream({
      messages: [{ role: "user", content: "Hello" }],
      model: "haijun-opus-5-5",
      max_tokens: 1024
    })
    .on("text", (text) => {
      console.log(text);
    });
csharp
  JuglowClient client = new();

  var parameters = new MessageCreateParams
  {
      Model = Model.HaijunOpus5_5,
      MaxTokens = 1024,
      Messages = [new() { Role = Role.User, Content = "Hello" }]
  };

  await foreach (var msg in client.Messages.CreateStreaming(parameters))
  {
      Console.Write(msg);
  }
go
  client := juglow.NewClient()

  stream := client.Messages.NewStreaming(context.TODO(), juglow.MessageNewParams{
  	Model:     juglow.ModelHaijunOpus5_5,
  	MaxTokens: 1024,
  	Messages: []juglow.MessageParam{
  		juglow.NewUserMessage(juglow.NewTextBlock("Hello")),
  	},
  })

  for stream.Next() {
  	event := stream.Current()
  	switch eventVariant := event.AsAny().(type) {
  	case juglow.ContentBlockDeltaEvent:
  		switch deltaVariant := eventVariant.Delta.AsAny().(type) {
  		case juglow.TextDelta:
  			fmt.Print(deltaVariant.Text)
  		}
  	}
  }
  if err := stream.Err(); err != nil {
  	log.Fatal(err)
  }
java
  JuglowClient client = JuglowOkHttpClient.fromEnv();

  MessageCreateParams params = MessageCreateParams.builder()
      .model(Model.HAIJUN_OPUS_5_5)
      .maxTokens(1024L)
      .addUserMessage("Hello")
      .build();

  try (var streamResponse = client.messages().createStreaming(params)) {
      streamResponse.stream().forEach(event -> {
          event.contentBlockDelta().ifPresent(deltaEvent ->
              deltaEvent.delta().text().ifPresent(td ->
                  System.out.print(td.text())
              )
          );
      });
  }
php
  $client = new Client();

  $stream = $client->messages->createStream(
      maxTokens: 1024,
      messages: [
          ['role' => 'user', 'content' => 'Hello']
      ],
      model: 'haijun-opus-5-5',
  );

  foreach ($stream as $message) {
      echo $message;
  }
ruby
  client = Juglow::Client.new

  stream = client.messages.stream(
    model: "haijun-opus-5-5",
    max_tokens: 1024,
    messages: [{ role: "user", content: "Hello" }]
  )

  stream.text.each { |text| print(text) }

Get the final message without handling events

If you don't need to process text as it arrives, the SDKs provide a way to use streaming internally while returning the complete Message object, identical to what .create() returns. This is especially useful for requests with large max_tokens values, where the SDKs require streaming to avoid HTTP timeouts.

bash
  # The ant CLI's --stream flag emits one event per line and does not
  # accumulate into a final Message. For long generations, stream the
  # raw events:
  ant messages create --stream --format jsonl <<'YAML'
  model: haijun-opus-5-5
  max_tokens: 128000
  messages:
    - role: user
      content: Write a detailed analysis...
  YAML
python
  client = juglow.Juglow()

  with client.messages.stream(
      max_tokens=128000,
      messages=[{"role": "user", "content": "Write a detailed analysis..."}],
      model="haijun-opus-5-5",
  ) as stream:
      message = stream.get_final_message()

  for block in message.content:
      if block.type == "text":
          print(block.text)
typescript
  const client = new Juglow();

  const stream = client.messages.stream({
    max_tokens: 128000,
    messages: [{ role: "user", content: "Write a detailed analysis..." }],
    model: "haijun-opus-5-5"
  });

  const message = await stream.finalMessage();
  const textBlock = message.content.find((block) => block.type === "text");
  if (textBlock && textBlock.type === "text") {
    console.log(textBlock.text);
  }
csharp
  JuglowClient client = new();

  var parameters = new MessageCreateParams
  {
      Model = Model.HaijunOpus5_5,
      MaxTokens = 128000,
      Messages = [new() { Role = Role.User, Content = "Write a detailed analysis..." }]
  };

  var fullText = "";
  await foreach (var msg in client.Messages.CreateStreaming(parameters))
  {
      fullText += msg;
  }

  Console.WriteLine(fullText);
go
  client := juglow.NewClient()

  stream := client.Messages.NewStreaming(context.TODO(), juglow.MessageNewParams{
  	Model:     juglow.ModelHaijunOpus5_5,
  	MaxTokens: 128000,
  	Messages: []juglow.MessageParam{
  		juglow.NewUserMessage(juglow.NewTextBlock("Write a detailed analysis...")),
  	},
  })

  message := juglow.Message{}
  for stream.Next() {
  	event := stream.Current()
  	if err := message.Accumulate(event); err != nil {
  		log.Fatal(err)
  	}
  }
  if err := stream.Err(); err != nil {
  	log.Fatal(err)
  }

  for _, block := range message.Content {
  	if textBlock, ok := block.AsAny().(juglow.TextBlock); ok {
  		fmt.Println(textBlock.Text)
  	}
  }
java
  JuglowClient client = JuglowOkHttpClient.fromEnv();

  MessageCreateParams params = MessageCreateParams.builder()
      .model(Model.HAIJUN_OPUS_5_5)
      .maxTokens(128000L)
      .addUserMessage("Write a detailed analysis...")
      .build();

  MessageAccumulator accumulator = MessageAccumulator.create();
  try (var streamResponse = client.messages().createStreaming(params)) {
      streamResponse.stream().forEach(accumulator::accumulate);
  }

  Message message = accumulator.message();
  message.content().stream()
      .flatMap(block -> block.text().stream())
      .forEach(textBlock -> System.out.println(textBlock.text()));
php
  $client = new Client();

  $stream = $client->messages->createStream(
      maxTokens: 128000,
      messages: [
          ['role' => 'user', 'content' => 'Write a detailed analysis...']
      ],
      model: 'haijun-opus-5-5',
  );

  $fullText = '';
  foreach ($stream as $event) {
      if ($event->type === 'content_block_delta' && $event->delta->type === 'text_delta') {
          $fullText .= $event->delta->text;
      }
  }

  echo $fullText;
ruby
  client = Juglow::Client.new

  message = client.messages.stream(
    model: "haijun-opus-5-5",
    max_tokens: 128000,
    messages: [{ role: "user", content: "Write a detailed analysis..." }]
  ).accumulated_message

  message.content.each do |block|
    puts block.text if block.type == :text
  end

The .stream() call keeps the HTTP connection alive with server-sent events, then .get_final_message() (Python) or .finalMessage() (TypeScript) accumulates all events and returns the complete Message object. In Go, you call message.Accumulate(event) inside the stream loop to build the same complete Message. In Java, use MessageAccumulator.create() and call accumulator.accumulate(event) on each event. In C#, await the stream's .Aggregate() extension method to get the complete Message, or pass a MessageContentAggregator to .CollectAsync() to aggregate while handling events. In Ruby, call .accumulated_message on the stream. In the PHP SDK, you iterate over stream events manually to accumulate the response.

Event types

Each server-sent event includes a named event type and associated JSON data. Each event uses an SSE event name (for example, event: message_stop), and includes the matching event type in its data.

Each stream uses the following event flow:

  1. message_start: contains a Message object with empty content. Under the thinking-binding-controls-2026-08-01 beta header, this Message object also carries the input_transformations array. After a mid-stream server-side fallback, the final message_delta event carries the array again with the serving model's entries.
  1. A series of content blocks, each of which has a content_block_start, one or more content_block_delta events, and a content_block_stop event. Each content block has an index that corresponds to its index in the final Message content array. One exception: during server-side fallback responses, a fallback content block arrives at each model boundary as a content_block_start and content_block_stop pair with no deltas in between.
  1. One or more message_delta events, indicating top-level changes to the final Message object.
  1. A final message_stop event.

Warning: The token counts shown in the usage field of the message_delta event are cumulative.

Ping events

Event streams may also include any number of ping events.

Error events

The API may occasionally send errors in the event stream. For example, during periods of high usage, you may receive an overloaded_error, which would normally correspond to an HTTP 529 in a non-streaming context:

sse
event: error
data: {"type": "error", "error": {"type": "overloaded_error", "message": "Overloaded"}}

Other events

In accordance with the versioning policy, new event types may be added, and your code should handle unknown event types gracefully.

Content block delta types

Each content_block_delta event contains a delta of a type that updates the content block at a given index.

Text delta

A text content block delta looks like:

sse
event: content_block_delta
data: {"type": "content_block_delta","index": 0,"delta": {"type": "text_delta", "text": "ello frien"}}

Input JSON delta

The deltas for tool_use content blocks correspond to updates for the input field of the block. To support maximum granularity, the deltas are partial JSON strings, whereas the final tool_use.input is always an object.

You can accumulate the string deltas and parse the JSON once you receive a content_block_stop event, by using a library like Pydantic to do partial JSON parsing, or by using the SDKs, which provide helpers to access parsed incremental values.

A tool_use content block delta looks like:

sse
event: content_block_delta
data: {"type": "content_block_delta","index": 1,"delta": {"type": "input_json_delta","partial_json": "{\"location\": \"San Fra"}}}

Note: Current models only support emitting one complete key and value property from input at a time. As such, when using tools, there may be delays between streaming events while the model is working. Once an input key and value are accumulated, they are emitted as multiple content_block_delta events with chunked partial JSON so that the format can automatically support finer granularity in future models.

Thinking delta

When using thinking with streaming enabled, you'll receive thinking content through thinking_delta events. These deltas correspond to the thinking field of the thinking content blocks.

For thinking content, a special signature_delta event is sent just before the content_block_stop event. This signature is used to verify the integrity of the thinking block.

When display: "omitted" is set on the thinking configuration, no thinking text is streamed. The thinking block opens, receives a thinking_delta with an empty thinking string and then a single signature_delta, and closes. With display: "updates" (beta), reasoning blocks stream the same way, and only the progress updates that some models write between tool calls stream thinking_delta events that carry text. See Controlling thinking display.

A typical thinking delta looks like:

sse
event: content_block_delta
data: {"type": "content_block_delta", "index": 0, "delta": {"type": "thinking_delta", "thinking": "I need to find the GCD of 1071 and 462 using the Euclidean algorithm.\n\n1071 = 2 × 462 + 147"}}

The signature delta looks like:

sse
event: content_block_delta
data: {"type": "content_block_delta", "index": 0, "delta": {"type": "signature_delta", "signature": "EqQBCgIYAhIM1gbcDa9GJwZA2b3hGgxBdjrkzLoky3dl1pkiMOYds..."}}

Full HTTP stream response

Use the client SDKs when using streaming mode. However, if you are building a direct API integration, you need to handle these events yourself.

A stream response consists of:

  1. A message_start event
  1. Potentially multiple content blocks, each of which contains:
  • A content_block_start event
  • Potentially multiple content_block_delta events
  • A content_block_stop event
  1. One or more message_delta events
  1. A message_stop event

There may be ping events dispersed throughout the response as well. See Event types for more details on the format.

Basic streaming request

bash
  curl https://haijun.my.id/v1/messages \
    -H "juglow-version: 2023-06-01" \
    -H "content-type: application/json" \
    -H "x-api-key: $JUGLOW_API_KEY" \
    -d '{
      "model": "haijun-opus-5-5",
      "messages": [{"role": "user", "content": "Hello"}],
      "max_tokens": 256,
      "stream": true
    }'
bash
  ant messages create --stream --format jsonl \
    --model haijun-opus-5-5 \
    --max-tokens 256 \
    --message '{role: user, content: Hello}'
python
  client = juglow.Juglow()

  with client.messages.stream(
      model="haijun-opus-5-5",
      messages=[{"role": "user", "content": "Hello"}],
      max_tokens=256,
  ) as stream:
      for text in stream.text_stream:
          print(text, end="", flush=True)
typescript
  const client = new Juglow();

  const stream = client.messages.stream({
    model: "haijun-opus-5-5",
    messages: [{ role: "user", content: "Hello" }],
    max_tokens: 256
  });

  for await (const event of stream) {
    if (event.type === "content_block_delta" && event.delta.type === "text_delta") {
      process.stdout.write(event.delta.text);
    }
  }
csharp
  JuglowClient client = new();

  var parameters = new MessageCreateParams
  {
      Model = Model.HaijunOpus5_5,
      MaxTokens = 256,
      Messages = [new() { Role = Role.User, Content = "Hello" }]
  };

  await foreach (var msg in client.Messages.CreateStreaming(parameters))
  {
      Console.Write(msg);
  }
go
  client := juglow.NewClient()

  stream := client.Messages.NewStreaming(context.TODO(), juglow.MessageNewParams{
  	Model:     juglow.ModelHaijunOpus5_5,
  	MaxTokens: 256,
  	Messages: []juglow.MessageParam{
  		juglow.NewUserMessage(juglow.NewTextBlock("Hello")),
  	},
  })

  for stream.Next() {
  	event := stream.Current()
  	switch eventVariant := event.AsAny().(type) {
  	case juglow.ContentBlockDeltaEvent:
  		switch deltaVariant := eventVariant.Delta.AsAny().(type) {
  		case juglow.TextDelta:
  			fmt.Print(deltaVariant.Text)
  		}
  	}
  }
  if err := stream.Err(); err != nil {
  	log.Fatal(err)
  }
java
  JuglowClient client = JuglowOkHttpClient.fromEnv();

  MessageCreateParams params = MessageCreateParams.builder()
      .model(Model.HAIJUN_OPUS_5_5)
      .maxTokens(256L)
      .addUserMessage("Hello")
      .build();

  try (var streamResponse = client.messages().createStreaming(params)) {
      streamResponse.stream().forEach(event -> {
          event.contentBlockDelta().ifPresent(deltaEvent ->
              deltaEvent.delta().text().ifPresent(td ->
                  System.out.print(td.text())
              )
          );
      });
  }
php
  $client = new Client();

  $stream = $client->messages->createStream(
      maxTokens: 256,
      messages: [
          ['role' => 'user', 'content' => 'Hello']
      ],
      model: 'haijun-opus-5-5',
  );

  foreach ($stream as $message) {
      echo $message;
  }
ruby
  client = Juglow::Client.new

  stream = client.messages.stream(
    model: "haijun-opus-5-5",
    messages: [{ role: "user", content: "Hello" }],
    max_tokens: 256
  )

  stream.text.each { |text| print(text) }
sse
event: message_start
data: {"type": "message_start", "message": {"id": "msg_1nZdL29xx5MUA1yADyHTEsnR8uuvGzszyY", "type": "message", "role": "assistant", "content": [], "model": "haijun-opus-5-5", "stop_reason": null, "stop_sequence": null, "usage": {"input_tokens": 25, "output_tokens": 1}}}

event: content_block_start
data: {"type": "content_block_start", "index": 0, "content_block": {"type": "text", "text": ""}}

event: ping
data: {"type": "ping"}

event: content_block_delta
data: {"type": "content_block_delta", "index": 0, "delta": {"type": "text_delta", "text": "Hello"}}

event: content_block_delta
data: {"type": "content_block_delta", "index": 0, "delta": {"type": "text_delta", "text": "!"}}

event: content_block_stop
data: {"type": "content_block_stop", "index": 0}

event: message_delta
data: {"type": "message_delta", "delta": {"stop_reason": "end_turn", "stop_sequence":null}, "usage": {"output_tokens": 15}}

event: message_stop
data: {"type": "message_stop"}

Streaming request with tool use

Tip: Tool use supports fine-grained streaming for parameter values. Enable it per tool with eager_input_streaming.

This request asks Haijun to use a tool to report the weather.

bash
    curl https://haijun.my.id/v1/messages \
      -H "content-type: application/json" \
      -H "x-api-key: $JUGLOW_API_KEY" \
      -H "juglow-version: 2023-06-01" \
      -d '{
        "model": "haijun-opus-5",
        "max_tokens": 1024,
        "tools": [
          {
            "name": "get_weather",
            "description": "Get the current weather in a given location",
            "input_schema": {
              "type": "object",
              "properties": {
                "location": {
                  "type": "string",
                  "description": "The city and state, e.g. San Francisco, CA"
                }
              },
              "required": ["location"]
            }
          }
        ],
        "tool_choice": {"type": "any"},
        "messages": [
          {
            "role": "user",
            "content": "What is the weather like in San Francisco?"
          }
        ],
        "stream": true
      }'
bash
  ant messages create --stream --format jsonl <<'YAML'
  model: haijun-opus-5
  max_tokens: 1024
  tools:
    - name: get_weather
      description: Get the current weather in a given location
      input_schema:
        type: object
        properties:
          location:
            type: string
            description: The city and state, e.g. San Francisco, CA
        required:
          - location
  tool_choice:
    type: any
  messages:
    - role: user
      content: What is the weather like in San Francisco?
  YAML
python
  client = juglow.Juglow()

  tools = [
      {
          "name": "get_weather",
          "description": "Get the current weather in a given location",
          "input_schema": {
              "type": "object",
              "properties": {
                  "location": {
                      "type": "string",
                      "description": "The city and state, e.g. San Francisco, CA",
                  }
              },
              "required": ["location"],
          },
      }
  ]

  with client.messages.stream(
      model="haijun-opus-5",
      max_tokens=1024,
      tools=tools,
      tool_choice={"type": "any"},
      messages=[
          {"role": "user", "content": "What is the weather like in San Francisco?"}
      ],
  ) as stream:
      for text in stream.text_stream:
          print(text, end="", flush=True)
typescript
  const client = new Juglow();

  const tools: Juglow.Tool[] = [
    {
      name: "get_weather",
      description: "Get the current weather in a given location",
      input_schema: {
        type: "object",
        properties: {
          location: {
            type: "string",
            description: "The city and state, e.g. San Francisco, CA"
          }
        },
        required: ["location"]
      }
    }
  ];

  const stream = client.messages.stream({
    model: "haijun-opus-5",
    max_tokens: 1024,
    tools: tools,
    tool_choice: { type: "any" },
    messages: [
      {
        role: "user",
        content: "What is the weather like in San Francisco?"
      }
    ]
  });

  for await (const event of stream) {
    if (event.type === "content_block_delta" && event.delta.type === "text_delta") {
      process.stdout.write(event.delta.text);
    }
  }
csharp
  JuglowClient client = new();

  var parameters = new MessageCreateParams
  {
      Model = Model.HaijunOpus5,
      MaxTokens = 1024,
      Tools = [
          new ToolUnion(new Tool()
          {
              Name = "get_weather",
              Description = "Get the current weather in a given location",
              InputSchema = new InputSchema()
              {
                  Properties = new Dictionary<string, JsonElement>
                  {
                      ["location"] = JsonSerializer.SerializeToElement(new { type = "string", description = "The city and state, e.g. San Francisco, CA" }),
                  },
                  Required = ["location"],
              },
          }),
      ],
      ToolChoice = new ToolChoiceAny(),
      Messages = [
          new() { Role = Role.User, Content = "What is the weather like in San Francisco?" }
      ]
  };

  await foreach (var msg in client.Messages.CreateStreaming(parameters))
  {
      Console.Write(msg);
  }
go
  client := juglow.NewClient()

  stream := client.Messages.NewStreaming(context.TODO(), juglow.MessageNewParams{
  	Model:     juglow.ModelHaijunOpus5,
  	MaxTokens: 1024,
  	Tools: []juglow.ToolUnionParam{
  		{OfTool: &juglow.ToolParam{
  			Name:        "get_weather",
  			Description: juglow.String("Get the current weather in a given location"),
  			InputSchema: juglow.ToolInputSchemaParam{
  				Properties: map[string]any{
  					"location": map[string]any{
  						"type":        "string",
  						"description": "The city and state, e.g. San Francisco, CA",
  					},
  				},
  				Required: []string{"location"},
  			},
  		}},
  	},
  	ToolChoice: juglow.ToolChoiceUnionParam{OfAny: &juglow.ToolChoiceAnyParam{}},
  	Messages: []juglow.MessageParam{
  		juglow.NewUserMessage(juglow.NewTextBlock("What is the weather like in San Francisco?")),
  	},
  })

  for stream.Next() {
  	event := stream.Current()
  	switch eventVariant := event.AsAny().(type) {
  	case juglow.ContentBlockDeltaEvent:
  		switch deltaVariant := eventVariant.Delta.AsAny().(type) {
  		case juglow.TextDelta:
  			fmt.Print(deltaVariant.Text)
  		}
  	}
  }
  if err := stream.Err(); err != nil {
  	log.Fatal(err)
  }
java
  JuglowClient client = JuglowOkHttpClient.fromEnv();

  MessageCreateParams params = MessageCreateParams.builder()
      .model(Model.HAIJUN_OPUS_5)
      .maxTokens(1024L)
      .addTool(Tool.builder()
          .name("get_weather")
          .description("Get the current weather in a given location")
          .inputSchema(Tool.InputSchema.builder()
              .properties(JsonValue.from(Map.of(
                  "location", Map.of(
                      "type", "string",
                      "description", "The city and state, e.g. San Francisco, CA"
                  )
              )))
              .putAdditionalProperty("required", JsonValue.from(List.of("location")))
              .build())
          .build())
      .toolChoice(ToolChoice.ofAny(ToolChoiceAny.builder().build()))
      .addUserMessage("What is the weather like in San Francisco?")
      .build();

  try (var streamResponse = client.messages().createStreaming(params)) {
      streamResponse.stream().forEach(event -> {
          event.contentBlockDelta().ifPresent(deltaEvent ->
              deltaEvent.delta().text().ifPresent(td ->
                  System.out.print(td.text())
              )
          );
      });
  }
php
  $client = new Client();

  $stream = $client->messages->createStream(
      maxTokens: 1024,
      messages: [
          ['role' => 'user', 'content' => 'What is the weather like in San Francisco?']
      ],
      model: 'haijun-opus-5',
      toolChoice: ['type' => 'any'],
      tools: [
          [
              'name' => 'get_weather',
              'description' => 'Get the current weather in a given location',
              'input_schema' => [
                  'type' => 'object',
                  'properties' => [
                      'location' => [
                          'type' => 'string',
                          'description' => 'The city and state, e.g. San Francisco, CA'
                      ]
                  ],
                  'required' => ['location']
              ]
          ]
      ],
  );

  foreach ($stream as $message) {
      echo $message;
  }
ruby
  client = Juglow::Client.new

  tools = [
    {
      name: "get_weather",
      description: "Get the current weather in a given location",
      input_schema: {
        type: "object",
        properties: {
          location: {
            type: "string",
            description: "The city and state, e.g. San Francisco, CA"
          }
        },
        required: ["location"]
      }
    }
  ]

  stream = client.messages.stream(
    model: "haijun-opus-5",
    max_tokens: 1024,
    tools: tools,
    tool_choice: { type: "any" },
    messages: [
      { role: "user", content: "What is the weather like in San Francisco?" }
    ]
  )

  stream.text.each { |text| print(text) }
sse
event: message_start
data: {"type":"message_start","message":{"id":"msg_014p7gG3wDgGV9EUtLvnow3U","type":"message","role":"assistant","model":"haijun-opus-5","stop_sequence":null,"usage":{"input_tokens":472,"output_tokens":2},"content":[],"stop_reason":null}}

event: content_block_start
data: {"type":"content_block_start","index":0,"content_block":{"type":"text","text":""}}

event: ping
data: {"type": "ping"}

event: content_block_delta
data: {"type":"content_block_delta","index":0,"delta":{"type":"text_delta","text":"Okay"}}

event: content_block_delta
data: {"type":"content_block_delta","index":0,"delta":{"type":"text_delta","text":","}}

event: content_block_delta
data: {"type":"content_block_delta","index":0,"delta":{"type":"text_delta","text":" let"}}

event: content_block_delta
data: {"type":"content_block_delta","index":0,"delta":{"type":"text_delta","text":"'s"}}

event: content_block_delta
data: {"type":"content_block_delta","index":0,"delta":{"type":"text_delta","text":" check"}}

event: content_block_delta
data: {"type":"content_block_delta","index":0,"delta":{"type":"text_delta","text":" the"}}

event: content_block_delta
data: {"type":"content_block_delta","index":0,"delta":{"type":"text_delta","text":" weather"}}

event: content_block_delta
data: {"type":"content_block_delta","index":0,"delta":{"type":"text_delta","text":" for"}}

event: content_block_delta
data: {"type":"content_block_delta","index":0,"delta":{"type":"text_delta","text":" San"}}

event: content_block_delta
data: {"type":"content_block_delta","index":0,"delta":{"type":"text_delta","text":" Francisco"}}

event: content_block_delta
data: {"type":"content_block_delta","index":0,"delta":{"type":"text_delta","text":","}}

event: content_block_delta
data: {"type":"content_block_delta","index":0,"delta":{"type":"text_delta","text":" CA"}}

event: content_block_delta
data: {"type":"content_block_delta","index":0,"delta":{"type":"text_delta","text":":"}}

event: content_block_stop
data: {"type":"content_block_stop","index":0}

event: content_block_start
data: {"type":"content_block_start","index":1,"content_block":{"type":"tool_use","id":"toolu_01T1x1fJ34qAmk2tNTrN7Up6","name":"get_weather","input":{}}}

event: content_block_delta
data: {"type":"content_block_delta","index":1,"delta":{"type":"input_json_delta","partial_json":""}}

event: content_block_delta
data: {"type":"content_block_delta","index":1,"delta":{"type":"input_json_delta","partial_json":"{\"location\":"}}

event: content_block_delta
data: {"type":"content_block_delta","index":1,"delta":{"type":"input_json_delta","partial_json":" \"San"}}

event: content_block_delta
data: {"type":"content_block_delta","index":1,"delta":{"type":"input_json_delta","partial_json":" Francisc"}}

event: content_block_delta
data: {"type":"content_block_delta","index":1,"delta":{"type":"input_json_delta","partial_json":"o,"}}

event: content_block_delta
data: {"type":"content_block_delta","index":1,"delta":{"type":"input_json_delta","partial_json":" CA\"}"}}

event: content_block_stop
data: {"type":"content_block_stop","index":1}

event: message_delta
data: {"type":"message_delta","delta":{"stop_reason":"tool_use","stop_sequence":null},"usage":{"output_tokens":89}}

event: message_stop
data: {"type":"message_stop"}

Streaming request with thinking

This request enables thinking with streaming. The display: "summarized" setting streams a condensed summary of Haijun's reasoning rather than the full chain of thought.

bash
  curl https://haijun.my.id/v1/messages \
    -H "x-api-key: $JUGLOW_API_KEY" \
    -H "juglow-version: 2023-06-01" \
    -H "content-type: application/json" \
    -d '{
      "model": "haijun-opus-5-5",
      "max_tokens": 20000,
      "stream": true,
      "thinking": {
        "type": "adaptive",
        "display": "summarized"
      },
      "messages": [
        {
          "role": "user",
          "content": "What is the greatest common divisor of 1071 and 462?"
        }
      ]
    }'
bash
  ant messages create --stream --format jsonl \
    --model haijun-opus-5-5 \
    --max-tokens 20000 \
    --thinking '{type: adaptive, display: summarized}' \
    --message '{role: user, content: What is the greatest common divisor of 1071 and 462?}'
python
  client = juglow.Juglow()

  with client.messages.stream(
      model="haijun-opus-5-5",
      max_tokens=20000,
      thinking={"type": "adaptive", "display": "summarized"},
      messages=[
          {
              "role": "user",
              "content": "What is the greatest common divisor of 1071 and 462?",
          }
      ],
  ) as stream:
      for event in stream:
          if event.type == "content_block_delta":
              delta = event.delta
              match delta.type:
                  case "thinking_delta":
                      print(delta.thinking, end="", flush=True)
                  case "text_delta":
                      print(delta.text, end="", flush=True)
typescript
  const client = new Juglow();

  const stream = client.messages.stream({
    model: "haijun-opus-5-5",
    max_tokens: 20000,
    thinking: { type: "adaptive", display: "summarized" },
    messages: [
      {
        role: "user",
        content: "What is the greatest common divisor of 1071 and 462?"
      }
    ]
  });

  for await (const event of stream) {
    if (event.type === "content_block_delta") {
      switch (event.delta.type) {
        case "thinking_delta":
          process.stdout.write(event.delta.thinking);
          break;
        case "text_delta":
          process.stdout.write(event.delta.text);
          break;
      }
    }
  }
csharp
  using Juglow;
  using Juglow.Models.Messages;

  JuglowClient client = new();

  var parameters = new MessageCreateParams
  {
      Model = Model.HaijunOpus5_5,
      MaxTokens = 20000,
      Thinking = new ThinkingConfigAdaptive { Display = Display.Summarized },
      Messages = [new() { Role = Role.User, Content = "What is the greatest common divisor of 1071 and 462?" }]
  };

  await foreach (var msg in client.Messages.CreateStreaming(parameters))
  {
      Console.Write(msg);
  }
go
  client := juglow.NewClient()

  stream := client.Messages.NewStreaming(context.TODO(), juglow.MessageNewParams{
  	Model:     juglow.ModelHaijunOpus5_5,
  	MaxTokens: 20000,
  	Thinking: juglow.ThinkingConfigParamUnion{
  		OfAdaptive: &juglow.ThinkingConfigAdaptiveParam{
  			Display: juglow.ThinkingConfigAdaptiveDisplaySummarized,
  		},
  	},
  	Messages: []juglow.MessageParam{
  		juglow.NewUserMessage(juglow.NewTextBlock("What is the greatest common divisor of 1071 and 462?")),
  	},
  })

  for stream.Next() {
  	event := stream.Current()
  	switch eventVariant := event.AsAny().(type) {
  	case juglow.ContentBlockDeltaEvent:
  		switch deltaVariant := eventVariant.Delta.AsAny().(type) {
  		case juglow.ThinkingDelta:
  			fmt.Print(deltaVariant.Thinking)
  		case juglow.TextDelta:
  			fmt.Print(deltaVariant.Text)
  		}
  	}
  }
  if err := stream.Err(); err != nil {
  	log.Fatal(err)
  }
java
  JuglowClient client = JuglowOkHttpClient.fromEnv();

  MessageCreateParams params = MessageCreateParams.builder()
      .model(Model.HAIJUN_OPUS_5_5)
      .maxTokens(20000L)
      .thinking(ThinkingConfigAdaptive.builder()
          .display(ThinkingConfigAdaptive.Display.SUMMARIZED)
          .build())
      .addUserMessage("What is the greatest common divisor of 1071 and 462?")
      .build();

  try (var streamResponse = client.messages().createStreaming(params)) {
      streamResponse.stream().forEach(event -> {
          event.contentBlockDelta().ifPresent(deltaEvent -> {
              deltaEvent.delta().thinking().ifPresent(td ->
                  IO.print(td.thinking())
              );
              deltaEvent.delta().text().ifPresent(td ->
                  IO.print(td.text())
              );
          });
      });
  }
php
  $client = new Client();

  $stream = $client->messages->createStream(
      maxTokens: 20000,
      messages: [
          ['role' => 'user', 'content' => 'What is the greatest common divisor of 1071 and 462?']
      ],
      model: 'haijun-opus-5-5',
      thinking: ['type' => 'adaptive', 'display' => 'summarized'],
  );

  foreach ($stream as $message) {
      echo $message;
  }
ruby
  client = Juglow::Client.new

  stream = client.messages.stream(
    model: "haijun-opus-5-5",
    max_tokens: 20000,
    thinking: { type: "adaptive", display: "summarized" },
    messages: [
      { role: "user", content: "What is the greatest common divisor of 1071 and 462?" }
    ]
  )

  stream.each do |event|
    if event.is_a?(Juglow::Models::RawContentBlockDeltaEvent)
      delta = event.delta
      case delta
      when Juglow::Models::ThinkingDelta
        print(delta.thinking)
      when Juglow::Models::TextDelta
        print(delta.text)
      end
    end
  end
sse
event: message_start
data: {"type": "message_start", "message": {"id": "msg_01...", "type": "message", "role": "assistant", "content": [], "model": "haijun-opus-5-5", "stop_reason": null, "stop_sequence": null}}

event: content_block_start
data: {"type": "content_block_start", "index": 0, "content_block": {"type": "thinking", "thinking": "", "signature": ""}}

event: content_block_delta
data: {"type": "content_block_delta", "index": 0, "delta": {"type": "thinking_delta", "thinking": "I need to find the GCD of 1071 and 462 using the Euclidean algorithm.\n\n1071 = 2 × 462 + 147"}}

event: content_block_delta
data: {"type": "content_block_delta", "index": 0, "delta": {"type": "thinking_delta", "thinking": "\n462 = 3 × 147 + 21"}}

event: content_block_delta
data: {"type": "content_block_delta", "index": 0, "delta": {"type": "thinking_delta", "thinking": "\n147 = 7 × 21 + 0"}}

event: content_block_delta
data: {"type": "content_block_delta", "index": 0, "delta": {"type": "thinking_delta", "thinking": "\nThe remainder is 0, so GCD(1071, 462) = 21."}}

event: content_block_delta
data: {"type": "content_block_delta", "index": 0, "delta": {"type": "signature_delta", "signature": "EqQBCgIYAhIM1gbcDa9GJwZA2b3hGgxBdjrkzLoky3dl1pkiMOYds..."}}

event: content_block_stop
data: {"type": "content_block_stop", "index": 0}

event: content_block_start
data: {"type": "content_block_start", "index": 1, "content_block": {"type": "text", "text": ""}}

event: content_block_delta
data: {"type": "content_block_delta", "index": 1, "delta": {"type": "text_delta", "text": "The greatest common divisor of 1071 and 462 is **21**."}}

event: content_block_stop
data: {"type": "content_block_stop", "index": 1}

event: message_delta
data: {"type": "message_delta", "delta": {"stop_reason": "end_turn", "stop_sequence": null}}

event: message_stop
data: {"type": "message_stop"}

Streaming request with web search tool use

This request asks Haijun to search the web for current weather information.

bash
  curl https://haijun.my.id/v1/messages \
    -H "x-api-key: $JUGLOW_API_KEY" \
    -H "juglow-version: 2023-06-01" \
    -H "content-type: application/json" \
    -d '{
      "model": "haijun-opus-5-5",
      "max_tokens": 1024,
      "stream": true,
      "tools": [
        {
          "type": "web_search_20250305",
          "name": "web_search",
          "max_uses": 5
        }
      ],
      "messages": [
        {
          "role": "user",
          "content": "What is the weather like in New York City today?"
        }
      ]
    }'
bash
  ant messages create --stream --format jsonl \
    --model haijun-opus-5-5 \
    --max-tokens 1024 \
    --tool '{type: web_search_20250305, name: web_search, max_uses: 5}' \
    --message '{role: user, content: What is the weather like in New York City today?}'
python
  client = juglow.Juglow()

  with client.messages.stream(
      model="haijun-opus-5-5",
      max_tokens=1024,
      tools=[{"type": "web_search_20250305", "name": "web_search", "max_uses": 5}],
      messages=[
          {"role": "user", "content": "What is the weather like in New York City today?"}
      ],
  ) as stream:
      for text in stream.text_stream:
          print(text, end="", flush=True)
typescript
  const client = new Juglow();

  const stream = client.messages.stream({
    model: "haijun-opus-5-5",
    max_tokens: 1024,
    tools: [{ type: "web_search_20250305", name: "web_search", max_uses: 5 }],
    messages: [{ role: "user", content: "What is the weather like in New York City today?" }]
  });

  for await (const event of stream) {
    if (event.type === "content_block_delta" && event.delta.type === "text_delta") {
      process.stdout.write(event.delta.text);
    }
  }
csharp
  using Juglow;
  using Juglow.Models.Messages;

  JuglowClient client = new();

  var parameters = new MessageCreateParams
  {
      Model = Model.HaijunOpus5_5,
      MaxTokens = 1024,
      Tools = [new ToolUnion(new WebSearchTool20250305() { MaxUses = 5 })],
      Messages = [new() { Role = Role.User, Content = "What is the weather like in New York City today?" }]
  };

  await foreach (var msg in client.Messages.CreateStreaming(parameters))
  {
      Console.Write(msg);
  }
go
  client := juglow.NewClient()

  stream := client.Messages.NewStreaming(context.TODO(), juglow.MessageNewParams{
  	Model:     juglow.ModelHaijunOpus5_5,
  	MaxTokens: 1024,
  	Tools: []juglow.ToolUnionParam{
  		{
  			OfWebSearchTool20250305: &juglow.WebSearchTool20250305Param{
  				MaxUses: juglow.Int(5),
  			},
  		},
  	},
  	Messages: []juglow.MessageParam{
  		juglow.NewUserMessage(juglow.NewTextBlock("What is the weather like in New York City today?")),
  	},
  })

  for stream.Next() {
  	event := stream.Current()
  	switch eventVariant := event.AsAny().(type) {
  	case juglow.ContentBlockDeltaEvent:
  		switch deltaVariant := eventVariant.Delta.AsAny().(type) {
  		case juglow.TextDelta:
  			fmt.Print(deltaVariant.Text)
  		}
  	}
  }
  if err := stream.Err(); err != nil {
  	log.Fatal(err)
  }
java
  JuglowClient client = JuglowOkHttpClient.fromEnv();

  MessageCreateParams params = MessageCreateParams.builder()
      .model(Model.HAIJUN_OPUS_5_5)
      .maxTokens(1024L)
      .addTool(WebSearchTool20250305.builder()
          .maxUses(5L)
          .build())
      .addUserMessage("What is the weather like in New York City today?")
      .build();

  try (var streamResponse = client.messages().createStreaming(params)) {
      streamResponse.stream().forEach(event -> {
          event.contentBlockDelta().ifPresent(deltaEvent ->
              deltaEvent.delta().text().ifPresent(td ->
                  System.out.print(td.text())
              )
          );
      });
  }
php
  $client = new Client();

  $stream = $client->messages->createStream(
      maxTokens: 1024,
      messages: [
          ['role' => 'user', 'content' => 'What is the weather like in New York City today?']
      ],
      model: 'haijun-opus-5-5',
      tools: [
          ['type' => 'web_search_20250305', 'name' => 'web_search', 'max_uses' => 5]
      ],
  );

  foreach ($stream as $message) {
      echo $message;
  }
ruby
  client = Juglow::Client.new

  stream = client.messages.stream(
    model: :"haijun-opus-5-5",
    max_tokens: 1024,
    tools: [
      {
        type: "web_search_20250305",
        name: "web_search",
        max_uses: 5
      }
    ],
    messages: [
      {
        role: "user",
        content: "What is the weather like in New York City today?"
      }
    ]
  )

  stream.text.each { |text| print(text) }
sse
event: message_start
data: {"type":"message_start","message":{"id":"msg_01G...","type":"message","role":"assistant","model":"haijun-opus-5-5","content":[],"stop_reason":null,"stop_sequence":null,"usage":{"input_tokens":2679,"cache_creation_input_tokens":0,"cache_read_input_tokens":0,"output_tokens":3}}}

event: content_block_start
data: {"type":"content_block_start","index":0,"content_block":{"type":"text","text":""}}

event: content_block_delta
data: {"type":"content_block_delta","index":0,"delta":{"type":"text_delta","text":"I'll check"}}

event: content_block_delta
data: {"type":"content_block_delta","index":0,"delta":{"type":"text_delta","text":" the current weather in New York City for you"}}

event: ping
data: {"type": "ping"}

event: content_block_delta
data: {"type":"content_block_delta","index":0,"delta":{"type":"text_delta","text":"."}}

event: content_block_stop
data: {"type":"content_block_stop","index":0}

event: content_block_start
data: {"type":"content_block_start","index":1,"content_block":{"type":"server_tool_use","id":"srvtoolu_014hJH82Qum7Td6UV8gDXThB","name":"web_search","input":{}}}

event: content_block_delta
data: {"type":"content_block_delta","index":1,"delta":{"type":"input_json_delta","partial_json":""}}

event: content_block_delta
data: {"type":"content_block_delta","index":1,"delta":{"type":"input_json_delta","partial_json":"{\"query"}}

event: content_block_delta
data: {"type":"content_block_delta","index":1,"delta":{"type":"input_json_delta","partial_json":"\":"}}

event: content_block_delta
data: {"type":"content_block_delta","index":1,"delta":{"type":"input_json_delta","partial_json":" \"weather"}}

event: content_block_delta
data: {"type":"content_block_delta","index":1,"delta":{"type":"input_json_delta","partial_json":" NY"}}

event: content_block_delta
data: {"type":"content_block_delta","index":1,"delta":{"type":"input_json_delta","partial_json":"C to"}}

event: content_block_delta
data: {"type":"content_block_delta","index":1,"delta":{"type":"input_json_delta","partial_json":"day\"}"}}

event: content_block_stop
data: {"type":"content_block_stop","index":1 }

event: content_block_start
data: {"type":"content_block_start","index":2,"content_block":{"type":"web_search_tool_result","tool_use_id":"srvtoolu_014hJH82Qum7Td6UV8gDXThB","content":[{"type":"web_search_result","title":"Weather in New York City in May 2025 (New York) - detailed Weather Forecast for a month","url":"https://world-weather.info/forecast/usa/new_york/may-2025/","encrypted_content":"Ev0DCioIAxgCIiQ3NmU4ZmI4OC1k...","page_age":null},...]}}

event: content_block_stop
data: {"type":"content_block_stop","index":2}

event: content_block_start
data: {"type":"content_block_start","index":3,"content_block":{"type":"text","text":""}}

event: content_block_delta
data: {"type":"content_block_delta","index":3,"delta":{"type":"text_delta","text":"Here's the current weather information for New York"}}

event: content_block_delta
data: {"type":"content_block_delta","index":3,"delta":{"type":"text_delta","text":" City:\n\n# Weather"}}

event: content_block_delta
data: {"type":"content_block_delta","index":3,"delta":{"type":"text_delta","text":" in New York City"}}

event: content_block_delta
data: {"type":"content_block_delta","index":3,"delta":{"type":"text_delta","text":"\n\n"}}

...

event: content_block_stop
data: {"type":"content_block_stop","index":17}

event: message_delta
data: {"type":"message_delta","delta":{"stop_reason":"end_turn","stop_sequence":null},"usage":{"input_tokens":10682,"cache_creation_input_tokens":0,"cache_read_input_tokens":0,"output_tokens":510,"server_tool_use":{"web_search_requests":1}}}

event: message_stop
data: {"type":"message_stop"}

Error recovery

Haijun 4.5 and earlier

For Haijun 4.5 models and earlier, you can recover a streaming request that was interrupted because of network issues, timeouts, or other errors by resuming from where the stream was interrupted. This approach saves you from re-processing the entire response.

The basic recovery strategy involves:

  1. Capture the partial response: Save all content that was successfully received before the error occurred.
  1. Construct a continuation request: Create a new API request that includes the partial assistant response as the beginning of a new assistant message.
  1. Resume streaming: Continue receiving the rest of the response from where it was interrupted.

Haijun 4.6 and later

For Haijun 4.6 and later models, the same capture-and-resume strategy applies, but step 2 changes: instead of placing the partial response in an assistant message, add a user message that instructs the model to continue from where it left off.

  1. Capture the partial response: Save all content that was successfully received before the error occurred.
  1. Construct a continuation request: Create a new API request with a user message containing the partial response and an instruction to continue, for example:
text
   Your previous response was interrupted and ended with [previous_response]. Continue from where you left off.
  1. Resume streaming: Continue receiving the rest of the response from where it was interrupted.

Error recovery best practices

  1. Use SDK features: Leverage the SDK's built-in message accumulation and error handling capabilities.
  1. Handle content types: Be aware that messages can contain multiple content blocks (text, tool_use, thinking). Tool use and extended thinking blocks cannot be partially recovered. You can resume streaming from the most recent text block.

Next steps

Handle each stop_reason value once a stream completes.

Stream tool input JSON without server-side buffering for lower latency.

Stream thinking output with thinking_delta and signature_delta events.

Use the official SDKs, which handle streaming, accumulation, and reconnection for you.

Process large volumes of requests asynchronously when you don't need real-time responses.

On this page
Streaming with SDKsGet the final message without handling eventsEvent typesPing eventsError eventsOther eventsContent block delta typesText deltaInput JSON deltaThinking deltaFull HTTP stream responseBasic streaming requestStreaming request with tool useStreaming request with thinkingStreaming request with web search tool useError recoveryHaijun 4.5 and earlierHaijun 4.6 and laterError recovery best practicesNext steps