When tool use + thinking + structured output are used together, Claude Sonnet 4.6 hallucinates a JSON response right away instead of returning tool calls first.
I reproduced this on a Google Vertex deployment of Claude Sonnet 4.6 + adaptive thinking. It does not reproduce with:
--- turn 1 ---
stop_reason: tool_use
→ get_weather({"city": "Paris", "unit": "celsius"})
← {"city": "Paris", "temperature": 22, "unit": "celsius", "condition": "partly cloudy", "humidity_pct": 65, "wind_speed_kmh": 15}
→ get_city_info({"city": "Paris"})
← {"city": "Paris", "population": 2161000, "timezone": "Europe/Paris", "country": "France"}
--- turn 2 ---
stop_reason: end_turn
=== final JSON response ===
{
"city": "Paris",
"temperature_celsius": 22,
"condition": "partly cloudy",
"humidity_pct": 65,
"population": 2161000,
"timezone": "Europe/Paris",
"summary": "Paris, France is currently experiencing partly cloudy skies with a pleasant temperature of 22\u00b0C and humidity at 65%. Home to over 2.1 million residents, the city operates in the Europe/Paris timezone."
}
Input tokens: 1234
Output tokens: 89
The model returns a JSON in turn 1 without calling any tools.
--- turn 1 ---
stop_reason: end_turn
[thinking] Let me fetch both simultaneously.
=== final JSON response ===
{
"city": "Paris",
"temperature_celsius": 18.5,
"condition": "Partly Cloudy",
"humidity_pct": 62,
"population": 2161000,
"timezone": "Europe/Paris",
"summary": "Paris is currently experiencing partly cloudy skies with a comfortable temperature of 18.5\u00b0C and moderate humidity at 62%. The city, home to over 2.1 million residents, operates in the Europe/Paris timezone (CET/CEST)."
}
request-id: req_vrtx_011CfNZwTJDzAh8yCrtZkaqa
x-vertex-ai-internal-prediction-backend: harpoon
# /// script
# requires-python = ">=3.11"
# dependencies = [
# "anthropic[vertex]",
# ]
# ///
import os
import json
import argparse
from anthropic import AnthropicVertex
from google.oauth2 import service_account
parser = argparse.ArgumentParser()
parser.add_argument("credentials_file")
parser.add_argument("--region", default=os.environ.get("GOOGLE_CLOUD_REGION", "global"))
parser.add_argument("--model", default="claude-sonnet-4-6")
args = parser.parse_args()
CREDENTIALS_FILE = args.credentials_file
REGION = args.region
MODEL = args.model
credentials = service_account.Credentials.from_service_account_file(
CREDENTIALS_FILE,
scopes=["https://www.googleapis.com/auth/cloud-platform"],
)
client = AnthropicVertex(
project_id=credentials.project_id,
region=REGION,
credentials=credentials,
)
TOOLS = [
{
"name": "get_weather",
"description": "Get the current weather for a city.",
"input_schema": {
"type": "object",
"properties": {
"city": {"type": "string", "description": "City name"},
"unit": {
"type": "string",
"enum": ["celsius", "fahrenheit"],
"description": "Temperature unit (default: celsius)",
},
},
"required": ["city"],
},
},
{
"name": "get_city_info",
"description": "Get general information about a city: population, timezone, country.",
"input_schema": {
"type": "object",
"properties": {
"city": {"type": "string", "description": "City name"},
},
"required": ["city"],
},
},
]
def simulate_tool(name: str, args: dict) -> dict:
if name == "get_weather":
unit = args.get("unit", "celsius")
return {
"city": args["city"],
"temperature": 22 if unit == "celsius" else 72,
"unit": unit,
"condition": "partly cloudy",
"humidity_pct": 65,
"wind_speed_kmh": 15,
}
if name == "get_city_info":
return {
"city": args["city"],
"population": 2_161_000,
"timezone": "Europe/Paris",
"country": "France",
}
return {"error": f"unknown tool: {name}"}
def print_headers(headers):
print(" [response headers]")
for key, value in headers.items():
print(f" {key}: {value}")
def send_message(messages: list):
raw = client.messages.with_raw_response.create(
model=MODEL,
max_tokens=1024,
messages=messages,
tools=TOOLS,
thinking={"type": "adaptive", "display": "summarized"},
output_config={
"effort": "low",
"format": {
"type": "json_schema",
"schema": {
"type": "object",
"properties": {
"city": {"type": "string"},
"temperature_celsius": {"type": "number"},
"condition": {"type": "string"},
"humidity_pct": {"type": "integer"},
"population": {"type": "integer"},
"timezone": {"type": "string"},
"summary": {"type": "string"},
},
"required": [
"city", "temperature_celsius", "condition",
"humidity_pct", "population", "timezone", "summary",
],
"additionalProperties": False,
},
}
},
)
return raw.parse(), raw.headers
messages = [
{
"role": "user",
"content": (
"Fetch the weather and city info for Paris, then return a JSON object with these keys: "
"city, temperature_celsius, condition, humidity_pct, population, timezone, summary."
),
}
]
iteration = 0
while True:
iteration += 1
print(f"\n--- turn {iteration} ---")
response, headers = send_message(messages)
stop_reason = response.stop_reason
content = response.content
print(f"stop_reason: {stop_reason}")
for block in content:
if block.type == "thinking" and block.thinking:
print(f" [thinking] {block.thinking}")
# Always append the full assistant content block list to preserve tool_use ids
messages.append({"role": "assistant", "content": content})
if stop_reason == "tool_use":
tool_results = []
for block in content:
if block.type == "tool_use":
name, args, tid = block.name, block.input, block.id
print(f" → {name}({json.dumps(args)})")
result = simulate_tool(name, args)
print(f" ← {json.dumps(result)}")
tool_results.append({
"type": "tool_result",
"tool_use_id": tid,
"content": json.dumps(result),
})
messages.append({"role": "user", "content": tool_results})
print_headers(headers)
elif stop_reason == "end_turn":
print("\n=== final JSON response ===")
for block in content:
if block.type == "text":
try:
print(json.dumps(json.loads(block.text), indent=2))
except json.JSONDecodeError:
print("(response is not valid JSON)")
print(block.text)
usage = response.usage
print(f"\nInput tokens: {usage.input_tokens}")
print(f"Output tokens: {usage.output_tokens}")
print()
print_headers(headers)
break
else:
print(f"Unexpected stop_reason: {stop_reason!r} — aborting")
break
When tool use + thinking + structured output are used together, Claude Sonnet 4.6 hallucinates a JSON response right away instead of returning tool calls first.
I reproduced this on a Google Vertex deployment of Claude Sonnet 4.6 + adaptive thinking. It does not reproduce with:
I gave the model two tools (
get_weatherandget_city_info), setoutput_config.formatto a JSON schema, and then asked it to fetch weather & information of Paris.This issue could be related to #1204
Expected behavior
The model request tool calls for
get_weatherandget_city_infoin the first turn, and then return the JSON response in the second turn.The expected behavior was observed when I disabled thinking (i.e. commenting line 99 in the Python snippet below).
Actual behavior
The model returns a JSON in turn 1 without calling any tools.
Relevant Vertex response headers:
Snippet to reproduce
Usage:
send_request.py [--region REGION] --model claude-sonnet-4.6 {path_to_service_account_json}