Introduction
When a detail in an image is too small to make out, you lean in and look closer. Haijun can't do that on its own: it sees an image once, at a fixed effective resolution, and when a chart label or a pair of closely spaced lines is too small at that resolution, no amount of prompting recovers it.
r.
By the end of this cookbook, you'll be able to:
relative to the image: dense documents, UI screenshots, schematics, scanned forms, photos of labels.
; haijun-sonnet-5 from 13% to 44% (+31 points). The figure note lists mean cost per question: haijun-fable-5 0.08 without the tool and 0.95 with it; haijun-sonnet-5 0.02 and 0.64.](/cookbook/images/notebooks/multimodal-crop-tool/results_accuracy.png
import matplotlib.pyplot as plt
import numpy as np
rng = np.random.default_rng(12)
months = np.arange(60)
fig, ax = plt.subplots(figsize=(16, 9), dpi=100)
series = {}
for name, base, amp in [
("Widget-A", 5200, 420),
("Widget-B", 4800, 380),
("Widget-C", 5600, 450),
("Widget-D", 4400, 300),
("Widget-E", 5014, 340),
("Widget-F", 4600, 410),
]:
y = base + amp * np.sin(months / 7 + base % 7) + rng.normal(0, 90, months.size).cumsum() / 4
series[name] = y
ax.plot(months, y, lw=1.0, label=name)
The detail our question hinges on: a tiny annotation at Widget-C's peak
peak_x = int(np.argmax(series["Widget-C"]))
peak_y = float(series["Widget-C"][peak_x])
ax.scatter([peak_x], [peak_y], s=12, color="crimson", zorder=3)
ax.annotate(
f"peak: {peak_y:,.0f} units",
(peak_x, peak_y),
textcoords="offset points",
xytext=(5, 4),
fontsize=5,
)
ax.set_title("Monthly production by product line")
ax.set_xlabel("Month")
ax.set_ylabel("Units")
ax.grid(True, lw=0.3, alpha=0.6)
ax.legend(fontsize=7, ncols=3)
Hand the figure to the rest of the notebook as a PIL image
buffer = BytesIO()
fig.savefig(buffer, format="PNG", bbox_inches="tight")
plt.close(fig)
buffer.seek(0)
chart_image = PILImage.open(buffer)
question = "What exact value is annotated at the peak of the Widget-C line?"
answer = f"peak: {peak_y:,.0f} units"
print(f"Chart size: {chart_image.size}")
print(f"Question: {question}")
_buf = BytesIO()
chart_image.save(_buf, format="PNG")
display(Image(data=_buf.getvalue(), width=560))
Chart size: (1325, 777) Question: What exact value is annotated at the peak of the Widget-C line? ,
"input_schema": {
"type": "object",
"properties": {
"x1": {
"type": "integer",
"description": "Left edge of the region, in pixels (the origin is the image's top-left corner)",
},
"y1": {
"type": "integer",
"description": "Top edge of the region, in pixels",
},
"x2": {
"type": "integer",
"description": "Right edge of the region, in pixels (must be greater than x1)",
},
"y2": {
"type": "integer",
"description": "Bottom edge of the region, in pixels (must be greater than y1)",
},
"image_index": {
"type": "integer",
"description": "Which image to zoom into, counting images in the conversation from 0. Omit when there is only one image.",
},
},
"required": ["x1", "y1", "x2", "y2"],
},
}
def handle_zoom(
original: PILImage.Image, view_size: tuple[int, int], x1: int, y1: int, x2: int, y2: int
) -> list[TextBlockParam | ImageBlockParam]:
"""Execute a zoom call and return the tool result content for Haijun.
original is the full-resolution image; view_size is the size of the copy
Haijun saw (the output of prepare_image()), which is the coordinate space
of x1/y1/x2/y2.
"""
view_w, view_h = view_size
x1, x2 = max(0, min(x1, view_w)), max(0, min(x2, view_w))
y1, y2 = max(0, min(y1, view_h)), max(0, min(y2, view_h))
if x1 >= x2 or y1 >= y2:
return [
{
"type": "text",
"text": "Error: invalid region (need x1 < x2 and y1 < y2, within the image)",
}
]
Map view coordinates onto the original. The resize preserved the aspect
ratio, so one scale factor covers both axes.
scale = original.width / view_w
box = (round(x1 * scale), round(y1 * scale), round(x2 * scale), round(y2 * scale))
cropped = original.crop(box)
zoomed = cropped.resize(zoom_size(cropped.width, cropped.height), PILImage.Resampling.LANCZOS)
Zoomed crops go back as JPEG: every tool result stays in the conversation,
and a few full-budget PNGs of detailed charts can exceed the API's 32 MB
request size limit. JPEG keeps each zoom to a fraction of that.
return [
{
"type": "text",
"text": (
f"Zoomed into ({x1},{y1})-({x2},{y2}) of the image you see at "
f"{view_w}x{view_h}px: a {cropped.width}x{cropped.height}px region "
f"of the original, returned magnified to {zoomed.width}x{zoomed.height}px."
),
},
{
"type": "image",
"source": {
"type": "base64",
"media_type": "image/jpeg",
"data": pil_to_base64(zoomed, "JPEG"),
},
},
]
Let's test the zoom tool manually before handing it to Haijun — zooming into the upper region of the chart where the peak annotation sits:
if response.stop_reason == "pause_turn":
The model paused a long turn; hand it back as-is and let it continue.
messages.append({"role": "assistant", "content": response.content})
continue
tool_calls = [block for block in response.content if block.type == "tool_use"]
if tool_calls:
Execute the tool calls and continue — keyed on the presence of the
calls rather than on stop_reason, so a turn that emitted a complete
call and then hit the token cap still gets its tool result.
messages.append({"role": "assistant", "content": response.content})
tool_results: list[ToolResultBlockParam] = []
for block in tool_calls:
if block.name != "zoom":
Dispatch by name: when you add more tools, route each name
to its handler here. Unknown names get an error result.
tool_results.append(
{
"type": "tool_result",
"tool_use_id": block.id,
"content": f"Unknown tool: {block.name}",
"is_error": True,
}
)
continue
block.input is typed loosely; the tool call's arguments arrive as a dict.
inputs = dict(cast(Mapping[str, Any], block.input))
image_index = inputs.pop("image_index", 0)
if not 0 <= image_index < len(originals):
result: list[TextBlockParam | ImageBlockParam] = [
{
"type": "text",
"text": f"Error: image_index {image_index} is out of range",
}
]
else:
try:
result = handle_zoom(
originals[image_index], views[image_index].size, **inputs
)
except TypeError as e:
A malformed call (e.g. a missing coordinate, possible when a
turn is cut off mid-call) gets an error result back, so the
model can correct itself instead of crashing the loop.
result = [{"type": "text", "text": f"Error: invalid zoom arguments ({e})"}]
if verbose:
show_tool_result(result, show_text=False)
tool_results.append(
{"type": "tool_result", "tool_use_id": block.id, "content": result}
)
messages.append({"role": "user", "content": tool_results})
continue
if (
response.stop_reason == "max_tokens"
and not any(block.type == "text" for block in response.content)
and response.content
and not nudged
):
The turn hit the token cap mid-thought: no answer text, no tool call.
Hand the partial turn back and ask for a concise wrap-up (once).
nudged = True
messages.append({"role": "assistant", "content": response.content})
messages.append(
{
"role": "user",
"content": "You ran out of output tokens. State your final answer concisely now.",
}
)
continue
Join every text block of the final turn: models that think can emit
several, and the answer is not always in the last one.
return "\n".join(block.text for block in response.content if block.type == "text")