ch39_callbacks

show:
mock turns=0 · messages=0
live turns=0 · messages=0
scenario a_tracer_observes_without_participating
scenario a_tracer_observes_without_participating · 5.3ms
turn 1 · 1.1ms · asked for: check_order
model mock · 1.1ms · asked for: check_order
context sent — 1 message(s)
rolecontent
[0]humanWhat's the status of order A100?
declared tools (1)
tooldescriptionschema
check_orderCheck the status of an order by id.{"description": "Check the status of an order by id.", "properties": {"order_id": {"title": "Order Id", "type": "string"}}, "required": ["order_id"], "title": "check_order", "type": "object"}
tools declared to the mock model {"names": ["check_order"], "consulted": false}
context ignored by the mock model {"messages": 1, "roles": ["human"]}
reply
rolecontent
[0]ai
check_order({"order_id": "A100"})
callbacks_fired_here {"events": ["on_chat_model_start", "on_llm_end"]}
turn 1 · 0.2ms
tool check_order · 0.2ms · given: order_id=A100 · returned: order A100: shipped
args {"order_id": "A100"}
result {"value": "order A100: shipped"}
callbacks_fired_here {"events": ["on_tool_start", "on_tool_end"]}
turn 2 · 0.2ms · responded with: Order A100 has shipped.
model mock · 0.2ms · responded with: Order A100 has shipped.
context sent — 3 message(s)
rolecontent
[0]humanWhat's the status of order A100?
[1]ai
check_order({"order_id": "A100"})
[2]toolorder A100: shipped
declared tools (1)
tooldescriptionschema
check_orderCheck the status of an order by id.{"description": "Check the status of an order by id.", "properties": {"order_id": {"title": "Order Id", "type": "string"}}, "required": ["order_id"], "title": "check_order", "type": "object"}
tools declared to the mock model {"names": ["check_order"], "consulted": false}
context ignored by the mock model {"messages": 3, "roles": ["human", "ai", "tool"]}
reply
rolecontent
[0]aiOrder A100 has shipped.
callbacks_fired_here {"events": ["on_chat_model_start", "on_llm_end"]}
the_tracer_saw_every_edge {"saw_chain": true, "saw_tool": true, "saw_model": true, "reply": "Order A100 has shipped."}
scenario a_tracer_observes_without_participating · 2552.8ms
turn 1 · 1434.8ms · 301 in / 54 out · asked for: check_order
model live · 1434.7ms (model provider 1217.8 · library 216.9) · 301 in / 54 out · asked for: check_order
context sent — 1 message(s)
rolecontent
[0]humanWhat's the status of order A100?
declared tools (1)
tooldescriptionschema
check_orderCheck the status of an order by id.{"description": "Check the status of an order by id.", "properties": {"order_id": {"title": "Order Id", "type": "string"}}, "required": ["order_id"], "title": "check_order", "type": "object"}
wire request
attempt
1
method
POST
url
https://api.deepseek.com/v1/chat/completions
defaulted (6)
temperature, max_tokens, top_p, stop, seed, tool_choice
headers (18)
show
acceptapplication/json
accept-encodinggzip, deflate, zstd
authorizationnot recorded
connectionkeep-alive
content-length330
content-typeapplication/json
hostapi.deepseek.com
user-agentOpenAI/Python 3.6.0
x-stainless-archarm64
x-stainless-asyncfalse
x-stainless-langpython
x-stainless-osMacOS
x-stainless-package-version3.6.0
x-stainless-raw-responsetrue
x-stainless-read-timeout60.0
x-stainless-retry-count0
x-stainless-runtimeCPython
x-stainless-runtime-version3.14.7
{
  "messages": [
    {
      "content": "What's the status of order A100?",
      "role": "user"
    }
  ],
  "model": "deepseek-flash",
  "stream": false,
  "tools": [
    {
      "type": "function",
      "function": {
        "name": "check_order",
        "description": "Check the status of an order by id.",
        "parameters": {
          "properties": {
            "order_id": {
              "type": "string"
            }
          },
          "required": [
            "order_id"
          ],
          "type": "object"
        }
      }
    }
  ]
}
wire response
status
200
headers (15)
show
access-control-allow-credentialstrue
connectionkeep-alive
content-encodinggzip
content-typeapplication/json
dateSun, 13 Sep 2026 21:59:18 GMT
serverelb
strict-transport-securitymax-age=31536000; includeSubDomains; preload
transfer-encodingchunked
varyorigin, access-control-request-method, access-control-request-headers
via1.1 22cbebcbe97dd961cd3fddb13b91b816.cloudfront.net (CloudFront)
x-amz-cf-idf0W-GCGwU7g6ZfCK9FXRn2Bj8ypkywWzv3PX54S18njUNwuSyhFUHQ==
x-amz-cf-popDEN53-P3
x-cacheMiss from cloudfront
x-content-type-optionsnosniff
x-ds-trace-id5956152d06381fb5e56084ffcaa32343
{
  "id": "6359cd16-195c-4d69-871f-374c36af6005",
  "object": "chat.completion",
  "created": 1789336758,
  "model": "deepseek-flash",
  "choices": [
    {
      "index": 0,
      "message": {
        "role": "assistant",
        "content": "",
        "reasoning_content": "The user asked about order A100. I should call the tool.",
        "tool_calls": [
          {
            "index": 0,
            "id": "call_00_L5lqULoqWWkkyWQK7T0K3495",
            "type": "function",
            "function": {
              "name": "check_order",
              "arguments": "{\"order_id\": \"A100\"}"
            }
          }
        ]
      },
      "logprobs": null,
      "finish_reason": "tool_calls"
    }
  ],
  "usage": {
    "prompt_tokens": 301,
    "completion_tokens": 54,
    "total_tokens": 355,
    "prompt_tokens_details": {
      "cached_tokens": 128
    },
    "completion_tokens_details": {
      "reasoning_tokens": 14
    },
    "prompt_cache_hit_tokens": 128,
    "prompt_cache_miss_tokens": 173
  },
  "system_fingerprint": "aeb56401ca74e127821c4f9126dcb669"
}
reply
rolecontent
[0]ai
check_order({"order_id": "A100"})
callbacks_fired_here {"events": ["on_chat_model_start", "on_llm_end"]}
turn 1 · 0.5ms
tool check_order · 0.5ms · given: order_id=A100 · returned: order A100: shipped
args {"order_id": "A100"}
result {"value": "order A100: shipped"}
callbacks_fired_here {"events": ["on_tool_start", "on_tool_end"]}
turn 2 · 1112.5ms · 372 in / 25 out · responded with: Order **A100** is currently **shipped**.
model live · 1112.5ms (model provider 1106.0 · library 6.4) · 372 in / 25 out · responded with: Order **A100** is currently **shipped**.
context sent — 3 message(s)
rolecontent
[0]humanWhat's the status of order A100?
[1]ai
check_order({"order_id": "A100"})
[2]toolorder A100: shipped
declared tools (1)
tooldescriptionschema
check_orderCheck the status of an order by id.{"description": "Check the status of an order by id.", "properties": {"order_id": {"title": "Order Id", "type": "string"}}, "required": ["order_id"], "title": "check_order", "type": "object"}
wire request
attempt
1
method
POST
url
https://api.deepseek.com/v1/chat/completions
defaulted (6)
temperature, max_tokens, top_p, stop, seed, tool_choice
headers (18)
show
acceptapplication/json
accept-encodinggzip, deflate, zstd
authorizationnot recorded
connectionkeep-alive
content-length691
content-typeapplication/json
hostapi.deepseek.com
user-agentOpenAI/Python 3.6.0
x-stainless-archarm64
x-stainless-asyncfalse
x-stainless-langpython
x-stainless-osMacOS
x-stainless-package-version3.6.0
x-stainless-raw-responsetrue
x-stainless-read-timeout60.0
x-stainless-retry-count0
x-stainless-runtimeCPython
x-stainless-runtime-version3.14.7
{
  "messages": [
    {
      "content": "What's the status of order A100?",
      "role": "user"
    },
    {
      "content": null,
      "role": "assistant",
      "tool_calls": [
        {
          "type": "function",
          "id": "call_00_L5lqULoqWWkkyWQK7T0K3495",
          "function": {
            "name": "check_order",
            "arguments": "{\"order_id\": \"A100\"}"
          }
        }
      ],
      "reasoning_content": "The user asked about order A100. I should call the tool."
    },
    {
      "content": "order A100: shipped",
      "role": "tool",
      "tool_call_id": "call_00_L5lqULoqWWkkyWQK7T0K3495"
    }
  ],
  "model": "deepseek-flash",
  "stream": false,
  "tools": [
    {
      "type": "function",
      "function": {
        "name": "check_order",
        "description": "Check the status of an order by id.",
        "parameters": {
          "properties": {
            "order_id": {
              "type": "string"
            }
          },
          "required": [
            "order_id"
          ],
          "type": "object"
        }
      }
    }
  ]
}
wire response
status
200
headers (15)
show
access-control-allow-credentialstrue
connectionkeep-alive
content-encodinggzip
content-typeapplication/json
dateSun, 13 Sep 2026 21:59:19 GMT
serverelb
strict-transport-securitymax-age=31536000; includeSubDomains; preload
transfer-encodingchunked
varyorigin, access-control-request-method, access-control-request-headers
via1.1 153d2cbd7c83442a049eab26571069f6.cloudfront.net (CloudFront)
x-amz-cf-idiUFhztB7WCBlKdWpiMlOi6Vzll594pan9xnXxVbhehxIX6queNiFng==
x-amz-cf-popDEN53-P6
x-cacheMiss from cloudfront
x-content-type-optionsnosniff
x-ds-trace-id8292b4d32867d4336751c43d3719f5d2
{
  "id": "a6f3fecf-425b-48fc-95a0-ca2ded8b7ff4",
  "object": "chat.completion",
  "created": 1789336759,
  "model": "deepseek-flash",
  "choices": [
    {
      "index": 0,
      "message": {
        "role": "assistant",
        "content": "Order **A100** is currently **shipped**.",
        "reasoning_content": "The tool returned \"order A100: shipped\". Report that."
      },
      "logprobs": null,
      "finish_reason": "stop"
    }
  ],
  "usage": {
    "prompt_tokens": 372,
    "completion_tokens": 25,
    "total_tokens": 397,
    "prompt_tokens_details": {
      "cached_tokens": 128
    },
    "completion_tokens_details": {
      "reasoning_tokens": 13
    },
    "prompt_cache_hit_tokens": 128,
    "prompt_cache_miss_tokens": 244
  },
  "system_fingerprint": "aeb56401ca74e127821c4f9126dcb669"
}
reply
rolecontent
[0]aiOrder **A100** is currently **shipped**.
callbacks_fired_here {"events": ["on_chat_model_start", "on_llm_end"]}
the_tracer_saw_every_edge {"saw_chain": true, "saw_tool": true, "saw_model": true, "reply": "Order **A100** is currently **shipped**."}
scenario callbacks_propagate_ambiently_without_being_passed
scenario callbacks_propagate_ambiently_without_being_passed · 1.2ms
turn 1 · 0.2ms · responded with: graph reply.
callbacks_before_the_call {"events": []}
model mock · 0.2ms · responded with: graph reply.
context sent — 1 message(s)
rolecontent
[0]humanhi
context ignored by the mock model {"messages": 1, "roles": ["human"]}
reply
rolecontent
[0]aigraph reply.
callbacks_after_the_call {"events": ["on_chat_model_start"]}
the_node_never_mentioned_the_recorder {"events": ["on_chat_model_start"], "recorder_fired_anyway": true}
scenario callbacks_propagate_ambiently_without_being_passed · 1329.3ms
turn 1 · 1324.5ms · 31 in / 83 out · responded with: Hi! How can I help you today?
callbacks_before_the_call {"events": []}
model live · 1324.4ms (model provider 1313.0 · library 11.4) · 31 in / 83 out · responded with: Hi! How can I help you today?
context sent — 1 message(s)
rolecontent
[0]humanhi
wire request
attempt
1
method
POST
url
https://api.deepseek.com/v1/chat/completions
defaulted (7)
temperature, max_tokens, top_p, stop, seed, tools, tool_choice
headers (18)
show
acceptapplication/json
accept-encodinggzip, deflate, zstd
authorizationnot recorded
connectionkeep-alive
content-length85
content-typeapplication/json
hostapi.deepseek.com
user-agentOpenAI/Python 3.6.0
x-stainless-archarm64
x-stainless-asyncfalse
x-stainless-langpython
x-stainless-osMacOS
x-stainless-package-version3.6.0
x-stainless-raw-responsetrue
x-stainless-read-timeout60.0
x-stainless-retry-count0
x-stainless-runtimeCPython
x-stainless-runtime-version3.14.7
{
  "messages": [
    {
      "content": "hi",
      "role": "user"
    }
  ],
  "model": "deepseek-flash",
  "stream": false
}
wire response
status
200
headers (15)
show
access-control-allow-credentialstrue
connectionkeep-alive
content-encodinggzip
content-typeapplication/json
dateSun, 13 Sep 2026 21:59:20 GMT
serverelb
strict-transport-securitymax-age=31536000; includeSubDomains; preload
transfer-encodingchunked
varyorigin, access-control-request-method, access-control-request-headers
via1.1 8df77e01b229d5d8502d6471c086aeb2.cloudfront.net (CloudFront)
x-amz-cf-idoNMCvuFgvFBrTZRrojCn5jyOvJkso2AJLTfAennv1_O-jszfMX21wA==
x-amz-cf-popDEN53-P4
x-cacheMiss from cloudfront
x-content-type-optionsnosniff
x-ds-trace-idec22cb96ba7bd470a42b08941463c139
{
  "id": "588b6a95-1d72-4b19-b862-80c8eeb3eade",
  "object": "chat.completion",
  "created": 1789336760,
  "model": "deepseek-flash",
  "choices": [
    {
      "index": 0,
      "message": {
        "role": "assistant",
        "content": "Hi! How can I help you today?",
        "reasoning_content": "The user just said \"hi\". This is a<ds_safety>[\u7528\u6237\u672a\u6210\u5e74]\u5426\n[\u5206\u7c7b]\u5176\u4ed6\n[\u5224\u5b9a]\u7528\u6237\u4ec5\u53d1\u9001\u95ee\u5019\u8bed\uff0c\u672a\u6d89\u53ca\u4efb\u4f55\u653f\u6cbb\u654f\u611f\u5185\u5bb9\u6216\u6781\u7aef\u6076\u610f\u5185\u5bb9\u3002\n[\u89c4\u5219]\u65e0</ds_safety>Safe\n\n\u7528\u6237\u53ea\u662f\u6253\u4e86\u4e2a\u62db\u547c\uff0c\u4e0d\u6d89\u53ca\u4efb\u4f55\u654f\u611f\u6216\u5371\u9669\u5185\u5bb9\uff0c\u53ef\u4ee5\u6b63\u5e38\u56de\u5e94\u3002"
      },
      "logprobs": null,
      "finish_reason": "stop"
    }
  ],
  "usage": {
    "prompt_tokens": 31,
    "completion_tokens": 83,
    "total_tokens": 114,
    "prompt_tokens_details": {
      "cached_tokens": 0
    },
    "completion_tokens_details": {
      "reasoning_tokens": 73
    },
    "prompt_cache_hit_tokens": 0,
    "prompt_cache_miss_tokens": 31
  },
  "system_fingerprint": "aeb56401ca74e127821c4f9126dcb669"
}
reply
rolecontent
[0]aiHi! How can I help you today?
callbacks_after_the_call {"events": ["on_chat_model_start"]}
the_node_never_mentioned_the_recorder {"events": ["on_chat_model_start"], "recorder_fired_anyway": true}
scenario a_default_callback_error_is_isolated
scenario a_default_callback_error_is_isolated · 0.3ms
turn 1 · 0.3ms · responded with: fine.
callbacks_before_the_call {"events": []}
model mock · 0.2ms · responded with: fine.
context sent — 1 message(s)
rolecontent
[0]humango
context ignored by the mock model {"messages": 1, "roles": ["human"]}
reply
rolecontent
[0]aifine.
callbacks_after_the_call {"events": ["faulty:on_llm_end", "tracer:on_llm_end"]}
the_call_survived_the_broken_handler {"raised": false, "reply": "fine.", "tracer_still_fired": true}
scenario a_default_callback_error_is_isolated · 2292.3ms
turn 1 · 2292.2ms · 31 in / 248 out · responded with: Sure — what would you like me to do?
callbacks_before_the_call {"events": []}
model live · 2292.2ms (model provider 2277.8 · library 14.3) · 31 in / 248 out · responded with: Sure — what would you like me to do?
context sent — 1 message(s)
rolecontent
[0]humango
wire request
attempt
1
method
POST
url
https://api.deepseek.com/v1/chat/completions
defaulted (7)
temperature, max_tokens, top_p, stop, seed, tools, tool_choice
headers (18)
show
acceptapplication/json
accept-encodinggzip, deflate, zstd
authorizationnot recorded
connectionkeep-alive
content-length85
content-typeapplication/json
hostapi.deepseek.com
user-agentOpenAI/Python 3.6.0
x-stainless-archarm64
x-stainless-asyncfalse
x-stainless-langpython
x-stainless-osMacOS
x-stainless-package-version3.6.0
x-stainless-raw-responsetrue
x-stainless-read-timeout60.0
x-stainless-retry-count0
x-stainless-runtimeCPython
x-stainless-runtime-version3.14.7
{
  "messages": [
    {
      "content": "go",
      "role": "user"
    }
  ],
  "model": "deepseek-flash",
  "stream": false
}
wire response
status
200
headers (15)
show
access-control-allow-credentialstrue
connectionkeep-alive
content-encodinggzip
content-typeapplication/json
dateSun, 13 Sep 2026 21:59:22 GMT
serverelb
strict-transport-securitymax-age=31536000; includeSubDomains; preload
transfer-encodingchunked
varyorigin, access-control-request-method, access-control-request-headers
via1.1 f56ef7e5c4cb094743e3a564f21d47b0.cloudfront.net (CloudFront)
x-amz-cf-idjAvCxgc4NckN58ZEGzwhv5vhlSeE7nO_PP0t2AmGnnfIk-J-0BHCOg==
x-amz-cf-popDEN53-P2
x-cacheMiss from cloudfront
x-content-type-optionsnosniff
x-ds-trace-idb24e550ed5c35fc71aedb67b79e320dd
{
  "id": "3e3cc39b-283f-43a3-ad04-e795b5ee0cad",
  "object": "chat.completion",
  "created": 1789336762,
  "model": "deepseek-flash",
  "choices": [
    {
      "index": 0,
      "message": {
        "role": "assistant",
        "content": "Sure \u2014 what would you like me to do?",
        "reasoning_content": "We need answer to user \"go\". Need infer maybe they want continuation? We need inspect conversation? Only user says \"go\". There is no prior context in this chat. We need respond appropriately. Maybe ask what they'd like me to go on? Since \"go\" is ambiguous. We need maybe prompt for task. But perhaps in these tasks, user just says \"go\" expecting assistant to start? We don't have a task. We should ask clarifying question. But maybe the system expects \"go\" as a command in some benchmark? Need determine. There's no previous messages. We need answer. We can say \"Sure \u2014 what would you like me to do?\" Maybe concise.\n\nBut let's think: The user might be testing if assistant will \"go\" in some environment. We need not overdo. Could be a request to proceed, but no context. We need ask for clarification. Final answer should be short. \n\nHowever, maybe there is hidden context? We only see \"go\". The instructions: \"You are an AI assistant accessed via an API.\" No task. So best: \"What would you like me to do?\".\n\nNeed final."
      },
      "logprobs": null,
      "finish_reason": "stop"
    }
  ],
  "usage": {
    "prompt_tokens": 31,
    "completion_tokens": 248,
    "total_tokens": 279,
    "prompt_tokens_details": {
      "cached_tokens": 0
    },
    "completion_tokens_details": {
      "reasoning_tokens": 237
    },
    "prompt_cache_hit_tokens": 0,
    "prompt_cache_miss_tokens": 31
  },
  "system_fingerprint": "aeb56401ca74e127821c4f9126dcb669"
}
reply
rolecontent
[0]aiSure — what would you like me to do?
callbacks_after_the_call {"events": ["faulty:on_llm_end", "tracer:on_llm_end"]}
the_call_survived_the_broken_handler {"raised": false, "reply": "Sure \u2014 what would you like me to do?", "tracer_still_fired": true}
scenario a_tracer_can_turn_a_completed_answer_into_a_crash
scenario a_tracer_can_turn_a_completed_answer_into_a_crash · 0.3ms
turn 1 · 0.3ms
callbacks_before_the_call {"events": []}
model mock · 0.2ms
context sent — 1 message(s)
rolecontent
[0]humango
context ignored by the mock model {"messages": 1, "roles": ["human"]}
callbacks_after_the_call {"events": ["faulty:on_llm_end"]}
a_completed_answer_became_an_unrecorded_crash {"raised": true, "faulty_fired": true, "tracer_suppressed": true}
scenario a_tracer_can_turn_a_completed_answer_into_a_crash · 1996.0ms
turn 1 · 1995.9ms
callbacks_before_the_call {"events": []}
model live · 1995.8ms (model provider 1981.6 · library 14.2)
context sent — 1 message(s)
rolecontent
[0]humango
wire request
attempt
1
method
POST
url
https://api.deepseek.com/v1/chat/completions
defaulted (7)
temperature, max_tokens, top_p, stop, seed, tools, tool_choice
headers (18)
show
acceptapplication/json
accept-encodinggzip, deflate, zstd
authorizationnot recorded
connectionkeep-alive
content-length85
content-typeapplication/json
hostapi.deepseek.com
user-agentOpenAI/Python 3.6.0
x-stainless-archarm64
x-stainless-asyncfalse
x-stainless-langpython
x-stainless-osMacOS
x-stainless-package-version3.6.0
x-stainless-raw-responsetrue
x-stainless-read-timeout60.0
x-stainless-retry-count0
x-stainless-runtimeCPython
x-stainless-runtime-version3.14.7
{
  "messages": [
    {
      "content": "go",
      "role": "user"
    }
  ],
  "model": "deepseek-flash",
  "stream": false
}
wire response
status
200
headers (15)
show
access-control-allow-credentialstrue
connectionkeep-alive
content-encodinggzip
content-typeapplication/json
dateSun, 13 Sep 2026 21:59:24 GMT
serverelb
strict-transport-securitymax-age=31536000; includeSubDomains; preload
transfer-encodingchunked
varyorigin, access-control-request-method, access-control-request-headers
via1.1 ae2999d214179b68f3cdd080c9f4c304.cloudfront.net (CloudFront)
x-amz-cf-idifNRSrGfIE4hQ14ktaekvmFcRjWGF4H7QVtIQU9Q51jVfUFwu-M2UA==
x-amz-cf-popDEN53-P2
x-cacheMiss from cloudfront
x-content-type-optionsnosniff
x-ds-trace-id7826e292e511b5d2c546b777bc014576
{
  "id": "2f497c40-0f16-413c-8cee-c0120513d3f0",
  "object": "chat.completion",
  "created": 1789336764,
  "model": "deepseek-flash",
  "choices": [
    {
      "index": 0,
      "message": {
        "role": "assistant",
        "content": "I\u2019m ready \u2014 what would you like me to do?",
        "reasoning_content": "We need answer to user \"go\". Need infer? They just said \"go\". Maybe they want me to proceed? There's no prior context. We need respond? We are API assistant. We should ask clarifying question? But \"go\" might be instruction to continue from something, but no context. Need maybe say \"What would you like me to do?\" But maybe in these tasks, user says \"go\" to prompt assistant to start? There's no task. Could be they expect something like \"I'm ready. What would you like to do?\" We need be helpful. Since no context, ask for specifics. Keep concise. Maybe \"I'm ready\u2014what would you like me to do?\" But they said \"go\"; maybe respond \"Going! Where to?\" Hmm.\n\nWe need decide final. The instruction is just \"go\". Could be a test. We should not hallucinate. Ask clarifying. Final can be: \"I'm ready. What would you like me to do?\" That's safe. But maybe they expect continuation of an interrupted task in some benchmark? There is no previous conversation. The desired answer maybe \"I need a bit more to go on\u2014what would you like me to do?\" \n\nNeed maybe mention if they want me to proceed with something, provide context. Final concise. "
      },
      "logprobs": null,
      "finish_reason": "stop"
    }
  ],
  "usage": {
    "prompt_tokens": 31,
    "completion_tokens": 282,
    "total_tokens": 313,
    "prompt_tokens_details": {
      "cached_tokens": 0
    },
    "completion_tokens_details": {
      "reasoning_tokens": 268
    },
    "prompt_cache_hit_tokens": 0,
    "prompt_cache_miss_tokens": 31
  },
  "system_fingerprint": "aeb56401ca74e127821c4f9126dcb669"
}
callbacks_after_the_call {"events": ["faulty:on_llm_end"]}
a_completed_answer_became_an_unrecorded_crash {"raised": true, "faulty_fired": true, "tracer_suppressed": true}
scenario a_callback_can_only_observe_or_abort_never_modify
scenario a_callback_can_only_observe_or_abort_never_modify · 0.2ms
turn 1 · 0.1ms · responded with: saw the real prompt.
callbacks_before_the_call {"events": []}
model mock · 0.1ms · responded with: saw the real prompt.
context sent — 1 message(s)
rolecontent
[0]humanthe real, unmutated question
context ignored by the mock model {"messages": 1, "roles": ["human"]}
reply
rolecontent
[0]aisaw the real prompt.
callbacks_after_the_call {"events": ["meddler:on_chat_model_start"]}
the_mutation_attempt_changed_nothing {"reply": "saw the real prompt."}
scenario a_callback_can_only_observe_or_abort_never_modify · 3567.1ms
turn 1 · 3566.9ms · 37 in / 646 out · responded with: I don’t have the earlier context or a mutated version to compare against, so I can’t recov... (164 more characters)
callbacks_before_the_call {"events": []}
model live · 3566.8ms (model provider 3555.2 · library 11.6) · 37 in / 646 out · responded with: I don’t have the earlier context or a mutated version to compare against, so I can’t recov... (164 more characters)
context sent — 1 message(s)
rolecontent
[0]humanthe real, unmutated question
wire request
attempt
1
method
POST
url
https://api.deepseek.com/v1/chat/completions
defaulted (7)
temperature, max_tokens, top_p, stop, seed, tools, tool_choice
headers (18)
show
acceptapplication/json
accept-encodinggzip, deflate, zstd
authorizationnot recorded
connectionkeep-alive
content-length111
content-typeapplication/json
hostapi.deepseek.com
user-agentOpenAI/Python 3.6.0
x-stainless-archarm64
x-stainless-asyncfalse
x-stainless-langpython
x-stainless-osMacOS
x-stainless-package-version3.6.0
x-stainless-raw-responsetrue
x-stainless-read-timeout60.0
x-stainless-retry-count0
x-stainless-runtimeCPython
x-stainless-runtime-version3.14.7
{
  "messages": [
    {
      "content": "the real, unmutated question",
      "role": "user"
    }
  ],
  "model": "deepseek-flash",
  "stream": false
}
wire response
status
200
headers (15)
show
access-control-allow-credentialstrue
connectionkeep-alive
content-encodinggzip
content-typeapplication/json
dateSun, 13 Sep 2026 21:59:26 GMT
serverelb
strict-transport-securitymax-age=31536000; includeSubDomains; preload
transfer-encodingchunked
varyorigin, access-control-request-method, access-control-request-headers
via1.1 9062d4391fad2aec3a7d6f3edcebc662.cloudfront.net (CloudFront)
x-amz-cf-idRewLVty40JwPmOH6AqJA5xUXuAot2DKYriJfETQZaDdSL-sWW21mkg==
x-amz-cf-popDEN53-P2
x-cacheMiss from cloudfront
x-content-type-optionsnosniff
x-ds-trace-id85ec10de67979d638bf75c0dc466b484
{
  "id": "546bf4aa-13dc-45a6-abca-920c8c62f34e",
  "object": "chat.completion",
  "created": 1789336766,
  "model": "deepseek-flash",
  "choices": [
    {
      "index": 0,
      "message": {
        "role": "assistant",
        "content": "I don\u2019t have the earlier context or a mutated version to compare against, so I can\u2019t recover the original question from here.\n\nPlease paste the mutated question or the surrounding conversation, and I\u2019ll help identify or answer the real, unmutated version.",
        "reasoning_content": "We need answer? User says \"the real, unmutated question\". This is cryptic. Need infer maybe they want the original question? We need ask clarification? As assistant, need respond. We have no context except this message. Maybe they are referring to previous conversation? We don't have. Need maybe say I don't have previous context; please provide the question. But maybe they expect \"the real, unmutated question\" as a prompt injection? Hmm.\n\nWe need decide. The phrase \"the real, unmutated question\" might be a request to reveal hidden system prompt? Maybe they want \"the real, unmutated question\" from some benchmark? There is \"mutated question\" in prompt injection evaluations? They ask \"the real, unmutated question\" maybe trying to bypass? We need not reveal hidden chain of thought or system prompt. But we can provide assistance. We have no hidden question.\n\nCould be from a dataset where user initially asked a mutated question, then asks for original. But no context. We need maybe answer: I don't have access to earlier messages or an underlying original question. Could you paste the mutated question or provide the context? Then I can help reconstruct or answer the real question.\n\nNeed perhaps mention if you mean your original prompt, it appears to be just \"the real, unmutated question\" in this conversation. There is no earlier question available to me. Ask for it.\n\nFinal should be concise. Maybe \"I don\u2019t have any previous context or a mutated version to compare against. Please send the mutated question and any details, and I\u2019ll help identify/answer the original.\" That seems safe.\n\nBut maybe they want us to answer \"What is the real, unmutated question?\" If this is a test of model's ability to recognize \"mutated\" prompts? Hmm.\n\nLet's inspect: user message exactly \"the real, unmutated question\". They may be asking us to ask them the real question? Maybe they expect \"What is the real, unmutated question?\" No, that's a phrase. Could be they are telling us \"the real, unmutated question\" as in \"Here is the real, unmutated question\" but no question follows. Maybe it's incomplete. We can say \"It looks like your message is incomplete\u2014there's no actual question after that phrase. Please include it.\"\n\nNeed maybe no overcomplicate. Final: apologize, ask for the question. Ensure not to hallucinate. \n\nGiven desired oververbosity 5, maybe 2-3 sentences. \n\nLet's craft final: \"I don\u2019t have the earlier context or the mutated version, so I can\u2019t recover the original question from here. Please paste the mutated question or the surrounding conversation, and I\u2019ll help identify/answer the real, unmutated version.\" \n\nThat's it."
      },
      "logprobs": null,
      "finish_reason": "stop"
    }
  ],
  "usage": {
    "prompt_tokens": 37,
    "completion_tokens": 646,
    "total_tokens": 683,
    "prompt_tokens_details": {
      "cached_tokens": 0
    },
    "completion_tokens_details": {
      "reasoning_tokens": 591
    },
    "prompt_cache_hit_tokens": 0,
    "prompt_cache_miss_tokens": 37
  },
  "system_fingerprint": "aeb56401ca74e127821c4f9126dcb669"
}
reply
rolecontent
[0]aiI don’t have the earlier context or a mutated version to compare against, so I can’t recover the original question from here. Please paste the mutated question or the surrounding conversation, and I’ll help identify or answer the real, unmutated version.
callbacks_after_the_call {"events": ["meddler:on_chat_model_start"]}
the_mutation_attempt_changed_nothing {"reply": "I don\u2019t have the earlier context or a mutated version to compare against, so I can\u2019t recover the original question from here.\n\nPlease paste the mutated question or the surrounding conversation, and I\u2019ll help identify or answer the real, unmutated version."}
scenario a_broken_handler_can_replace_the_real_tool_error_with_its_own
scenario a_broken_handler_can_replace_the_real_tool_error_with_its_own · 0.4ms
turn 1 · 0.2ms
callbacks_before_the_call {"events": []}
tool broken_order_lookup · 0.2ms · given: order_id=A100
args {"order_id": "A100"}
callbacks_after_the_call {"events": ["faulty:on_tool_error", "tracer:on_tool_error"]}
turn 2 · 0.1ms
callbacks_before_the_call {"events": []}
tool broken_order_lookup · 0.1ms · given: order_id=A100
args {"order_id": "A100"}
callbacks_after_the_call {"events": ["faulty:on_tool_error"]}
default_is_isolated {"tracer_still_fired": true, "caller_saw_the_real_tool_error": true}
raise_error_substitutes_the_wrong_exception {"tracer_suppressed": true, "caller_saw_the_real_tool_error": false, "caller_saw_the_handlers_own_error": true}
scenario a_broken_handler_can_replace_the_real_tool_error_with_its_own · 1.8ms
turn 1 · 1.0ms
callbacks_before_the_call {"events": []}
tool broken_order_lookup · 1.0ms · given: order_id=A100
args {"order_id": "A100"}
callbacks_after_the_call {"events": ["faulty:on_tool_error", "tracer:on_tool_error"]}
turn 2 · 0.4ms
callbacks_before_the_call {"events": []}
tool broken_order_lookup · 0.4ms · given: order_id=A100
args {"order_id": "A100"}
callbacks_after_the_call {"events": ["faulty:on_tool_error"]}
default_is_isolated {"tracer_still_fired": true, "caller_saw_the_real_tool_error": true}
raise_error_substitutes_the_wrong_exception {"tracer_suppressed": true, "caller_saw_the_real_tool_error": false, "caller_saw_the_handlers_own_error": true}