From b03261caada890105f0c9c3dafb9c6b01604887e Mon Sep 17 00:00:00 2001 From: Yichen Jiang Date: Wed, 2 Sep 2026 10:33:35 +0800 Subject: [PATCH] fix(llm): narrow the fix to identity acceptance MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit Refusing a response whose tool call never receives an identity needed a new failure code, a change to the default retryable set, and a `[DONE]` gate that overrode the finish reason a provider had already sent — turning a safe `max-tokens` truncation into up to five retries. The lenient wire it guarded against is hypothetical: no report describes a stream that omits identity entirely, and the pre-existing test for it is labelled as such. Only `acceptIdentity` and the widened wire types remain. They close the reported erasure and cannot reach a worse outcome than the previous assignment, because the set of inputs that assign only narrows. --- ...-21-bounded-llm-request-recovery.i18n.yaml | 4 +- ...2026-06-21-bounded-llm-request-recovery.md | 2 +- ...6-06-21-bounded-llm-request-recovery.zh.md | 2 +- ...9-01-streamed-tool-call-identity.i18n.yaml | 4 +- .../2026-09-01-streamed-tool-call-identity.md | 20 +++---- ...26-09-01-streamed-tool-call-identity.zh.md | 20 +++---- apps/web/tests/shipped-composition.e2e.ts | 2 - packages/llm/llm-deepseek/README.i18n.yaml | 4 +- packages/llm/llm-deepseek/README.md | 2 +- packages/llm/llm-deepseek/README.zh.md | 2 +- packages/llm/llm-deepseek/src/translate.ts | 57 ++++--------------- .../llm/llm-deepseek/tests/translate.spec.ts | 45 +++------------ packages/llm/llm-retry/README.i18n.yaml | 4 +- packages/llm/llm-retry/README.md | 2 +- packages/llm/llm-retry/README.zh.md | 2 +- packages/llm/llm/src/assembler.ts | 5 -- packages/llm/llm/src/error.ts | 12 ---- packages/llm/llm/src/retry-policy.ts | 3 +- packages/llm/llm/tests/retry-policy.spec.ts | 2 +- .../empty-response-retry/session.jsonl | 2 +- .../malformed-tool-call-retry/session.jsonl | 27 --------- .../malformed-tool-call-retry/snapshot.yml | 9 --- 22 files changed, 50 insertions(+), 182 deletions(-) delete mode 100644 snapshots/session/malformed-tool-call-retry/session.jsonl delete mode 100644 snapshots/session/malformed-tool-call-retry/snapshot.yml diff --git a/.agents/notes/implemented/architecture/2026-06-21-bounded-llm-request-recovery.i18n.yaml b/.agents/notes/implemented/architecture/2026-06-21-bounded-llm-request-recovery.i18n.yaml index 37d36945cb..a0af3cb0e6 100644 --- a/.agents/notes/implemented/architecture/2026-06-21-bounded-llm-request-recovery.i18n.yaml +++ b/.agents/notes/implemented/architecture/2026-06-21-bounded-llm-request-recovery.i18n.yaml @@ -2,5 +2,5 @@ # side as of the last confirmed-consistent state. Both languages carry equal authority; # after editing either side, bring the other along and re-record with: # pnpm run verify-translation-pairing --write .agents/notes/implemented/architecture/2026-06-21-bounded-llm-request-recovery.md -2026-06-21-bounded-llm-request-recovery.md: 3c9cb4f58d7b1c0090216ac5bf753667680a3ff2 -2026-06-21-bounded-llm-request-recovery.zh.md: 1760951bdc23bad7dffb90edd7c61d2545a3bd28 +2026-06-21-bounded-llm-request-recovery.md: 42bf460e52133b2a5471479fa3d7647e70092b48 +2026-06-21-bounded-llm-request-recovery.zh.md: 2a13f0a740348a5f74bd3d90120a148b25f2e870 diff --git a/.agents/notes/implemented/architecture/2026-06-21-bounded-llm-request-recovery.md b/.agents/notes/implemented/architecture/2026-06-21-bounded-llm-request-recovery.md index 3c9cb4f58d..42bf460e52 100644 --- a/.agents/notes/implemented/architecture/2026-06-21-bounded-llm-request-recovery.md +++ b/.agents/notes/implemented/architecture/2026-06-21-bounded-llm-request-recovery.md @@ -46,7 +46,7 @@ The agent loop passes the terminal finish's `LlmFailure` to `agent/request-error Adapters extract structured facts before falling back to message inspection. They validate HTTP status, parse `Retry-After` seconds or dates into a positive finite millisecond delay, brand the provider request id when exposed, and distinguish their own timeout from the caller's abort. Provider-specific codes and messages may refine a mapping, but no recovery listener parses them. -The shared transient-code set is intentionally small: adapter mappings for `RATE_LIMIT` and `SERVER`, explicit `TIMEOUT` and `TRANSPORT` codes for remote failures, `EMPTY_RESPONSE` for a completed provider response with no content blocks, and `MALFORMED_TOOL_CALL` for a streamed tool call the provider never identified. Both adapters classify the empty response as an error finish; see [empty model responses are retryable](../bug-fix/2026-07-24-empty-model-response-is-retryable.md). The DeepSeek adapter classifies the unidentified tool call the same way; see [streamed tool-call identity](2026-09-01-streamed-tool-call-identity.md). Authentication, quota, invalid request, context overflow, protocol, abort, and unknown failures keep distinct stable codes and are not transient by default. Adding a code requires adapter fixtures and a documented policy decision; it does not require expanding a second failure-class enum. +The shared transient-code set is intentionally small: adapter mappings for `RATE_LIMIT` and `SERVER`, explicit `TIMEOUT` and `TRANSPORT` codes for remote failures, and `EMPTY_RESPONSE` for a completed provider response with no content blocks. Both adapters classify the last case as an error finish; see [empty model responses are retryable](../bug-fix/2026-07-24-empty-model-response-is-retryable.md). Authentication, quota, invalid request, context overflow, protocol, abort, and unknown failures keep distinct stable codes and are not transient by default. Adding a code requires adapter fixtures and a documented policy decision; it does not require expanding a second failure-class enum. ### Put retry policy on the existing failed-step extension point diff --git a/.agents/notes/implemented/architecture/2026-06-21-bounded-llm-request-recovery.zh.md b/.agents/notes/implemented/architecture/2026-06-21-bounded-llm-request-recovery.zh.md index 1760951bdc..2a13f0a740 100644 --- a/.agents/notes/implemented/architecture/2026-06-21-bounded-llm-request-recovery.zh.md +++ b/.agents/notes/implemented/architecture/2026-06-21-bounded-llm-request-recovery.zh.md @@ -46,7 +46,7 @@ agent loop(智能体循环)会将终止 finish 的 `LlmFailure` 传给 `agen 适配器会先提取结构化事实,再回退到消息检查。它们会验证 HTTP 状态,将 `Retry-After` 的秒数或日期解析为正的有限毫秒延迟,在提供方公开请求 id 时将其品牌化,并区分自身超时与调用方中止。提供方专用 code 和消息可以细化映射,但恢复监听器不会解析它们。 -共享的暂时性 code 集有意保持很小:适配器针对 `RATE_LIMIT` 和 `SERVER` 的映射,远程失败使用的显式 `TIMEOUT` 和 `TRANSPORT` code,提供方响应已完成却没有内容块时使用的 `EMPTY_RESPONSE`,以及提供方始终未给出身份的流式工具调用使用的 `MALFORMED_TOOL_CALL`。两个适配器都会把空响应归类为错误 finish;详见[空模型响应可重试](../bug-fix/2026-07-24-empty-model-response-is-retryable.zh.md)。DeepSeek 适配器以同样方式归类无身份的工具调用;详见[流式工具调用身份](2026-09-01-streamed-tool-call-identity.zh.md)。身份验证、配额、无效请求、上下文溢出、协议、中止和未知失败都保留不同的稳定 code,且默认不属于暂时性失败。新增 code 需要适配器 fixture(测试前置数据)和已记录的策略决策;无需扩展第二个失败类枚举。 +共享的暂时性 code 集有意保持很小:适配器针对 `RATE_LIMIT` 和 `SERVER` 的映射,远程失败使用的显式 `TIMEOUT` 和 `TRANSPORT` code,以及提供方响应已完成却没有内容块时使用的 `EMPTY_RESPONSE`。两个适配器都会把最后一种情况归类为错误 finish;详见[空模型响应可重试](../bug-fix/2026-07-24-empty-model-response-is-retryable.zh.md)。身份验证、配额、无效请求、上下文溢出、协议、中止和未知失败都保留不同的稳定 code,且默认不属于暂时性失败。新增 code 需要适配器 fixture(测试前置数据)和已记录的策略决策;无需扩展第二个失败类枚举。 ### 将重试策略放在现有失败步骤扩展点上 diff --git a/.agents/notes/implemented/architecture/2026-09-01-streamed-tool-call-identity.i18n.yaml b/.agents/notes/implemented/architecture/2026-09-01-streamed-tool-call-identity.i18n.yaml index 11b7087b8b..aa070add66 100644 --- a/.agents/notes/implemented/architecture/2026-09-01-streamed-tool-call-identity.i18n.yaml +++ b/.agents/notes/implemented/architecture/2026-09-01-streamed-tool-call-identity.i18n.yaml @@ -2,5 +2,5 @@ # side as of the last confirmed-consistent state. Both languages carry equal authority; # after editing either side, bring the other along and re-record with: # pnpm run verify-translation-pairing --write .agents/notes/implemented/architecture/2026-09-01-streamed-tool-call-identity.md -2026-09-01-streamed-tool-call-identity.md: d940b0eb376928a4217eb69d1fb8f4aafa1796a0 -2026-09-01-streamed-tool-call-identity.zh.md: 153d9d4fc87cd5d440668b1b5e5cb6995628e2df +2026-09-01-streamed-tool-call-identity.md: c52f39b735199270d65ed30a388333a217003237 +2026-09-01-streamed-tool-call-identity.zh.md: 9a7ffb6343870a08e06507408e007ad94fbad178 diff --git a/.agents/notes/implemented/architecture/2026-09-01-streamed-tool-call-identity.md b/.agents/notes/implemented/architecture/2026-09-01-streamed-tool-call-identity.md index d940b0eb37..c52f39b735 100644 --- a/.agents/notes/implemented/architecture/2026-09-01-streamed-tool-call-identity.md +++ b/.agents/notes/implemented/architecture/2026-09-01-streamed-tool-call-identity.md @@ -6,32 +6,28 @@ English | [中文](2026-09-01-streamed-tool-call-identity.zh.md) ## Problem -The DeepSeek SSE translator assigned `id` and `name` on every tool-call delta that carried the field, so a continuation delta repeating either as an empty string erased the identity established by the call's first delta. The assembled block reached the loop with an empty name, which the tool registry refuses as `unknown tool ""`. Gateways that fill those fields with `null` erased the identity the same way, and `WireToolCallDelta` declared both as `string | undefined`, keeping the observed `null` out of the compiler's reach. +The DeepSeek SSE translator assigned `id` and `name` on every tool-call delta that carried the field, so a continuation delta repeating either as an empty string erased the identity established by the call's first delta. The assembled block reached the loop with an empty name, which the tool registry refuses as `unknown tool ""`, leaving the affected models unable to run any tool. Gateways that fill those fields with `null` erased the identity the same way, and `WireToolCallDelta` declared both as `string | undefined`, keeping the observed `null` out of the compiler's reach. -The empty identity outlived the turn. `appendToolCall` and `appendToolResult` write the block's id verbatim and no write path validates it, while `adoptSessionEvent` refuses a `tool/result` whose `callId` is empty. A session that recorded one such call was writable and no longer loadable: the persistence coordinator wrapped that refusal in `SessionPersistenceCorruptionError`. +The empty identity outlived the turn. `appendToolCall` and `appendToolResult` write the block's id verbatim and no write path validates it, while `adoptSessionEvent` refuses a `tool/result` whose `callId` is empty, so the persistence coordinator wrapped that refusal in `SessionPersistenceCorruptionError`. A session that recorded one such call was writable and no longer loadable. ## Decision -`acceptIdentity` accepts only a non-empty string for a tool call's `id` and `name`; `undefined`, `null`, `''`, and any non-string leave the established value in place. `WireToolCallDelta` widens `id`, `function.name`, and `function.arguments` to admit `null`, so the values gateways actually send are in the type system and the runtime guard is load-bearing rather than speculative. - -A tool call still lacking `id` or `name` when the stream reaches `[DONE]` is not closed. The translator reports any pending usage, then ends the response with an error finish carrying the new `MALFORMED_TOOL_CALL` code, and emits no `block-end` at all. `closeBlock` returns which field is missing instead of substituting an empty string, so no path can assemble an unidentified tool call. - -`MALFORMED_TOOL_CALL` joins the default retryable codes described in [bounded LLM request recovery](2026-06-21-bounded-llm-request-recovery.md); retry reaches the attempt either way, because `LlmRuntime.adapterStream` normalizes a thrown `LlmError` into the same error `finish` the loop routes to `agent/request-error`. The refusal is yielded rather than thrown so the attempt's already-billed usage still reaches the loop before the failure, and so the refusal follows the same chunk protocol as the `EMPTY_RESPONSE` case beside it. +`acceptIdentity` accepts only a non-empty string for a tool call's `id` and `name`; `undefined`, `null`, `''`, and any non-string leave the established value in place. The assignment set only narrows, so no input reaches a worse outcome than before. `WireToolCallDelta` widens `id`, `function.name`, and `function.arguments` to admit `null`, putting the values gateways actually send into the type system and making the runtime guard load-bearing rather than speculative. ## Alternatives considered **Concatenate `id` and `name` across deltas.** Rejected: they are identity, not accumulation. Concatenation produces `Globnull` against a gateway that sends `null`, and a doubled name against one that repeats a non-empty value. -**Refuse a conflicting non-empty identity mid-stream.** Deferred: a gateway that fragments a long tool name would be refused for it, and the `[DONE]` check already keeps an unusable call away from the loop. +**Refuse a conflicting non-empty identity mid-stream.** Deferred: a gateway that fragments a long tool name would be refused for it, and no observed provider re-sends a different non-empty identity within one call index. + +**Refuse a response whose tool call never receives an identity.** Deferred. It requires a new failure code, a change to the default retryable set, and a `[DONE]` gate that must not override the finish reason a provider already sent — cost and risk that the reported defect does not carry. The lenient wire it guards against is hypothetical: no report describes a stream that omits identity entirely. **Relax the session reader's empty-`callId` refusal.** Rejected: an empty `callId` cannot be paired back to the provider on the next request, so accepting it moves the failure into the model request. That refusal is the durable-boundary gate; the producer was the defect. -**Refuse as soon as a call's first delta carries no `id`.** Rejected: "first delta only" is a claim about the remote encoder, so a gateway that sends `id` one delta later would be refused for nothing, and the `[DONE]` check covers every case an early check would. - ## Consequences -A continuation delta repeating identity empty or null is inert, so a call keeps the identity its first delta established. A provider that never identifies a call costs a retry instead of an `unknown tool ""` result and a session that cannot be reopened. Sessions that already recorded an empty `callId` stay unreadable; recovering them is outside this change. +A continuation delta repeating identity empty or null is inert, so a call keeps the identity its first delta established, and the reported path to `unknown tool ""` and an unreadable session is closed. A stream that never carries identity at all still assembles an empty one, exactly as before; that path and the recovery of sessions already holding an empty `callId` are outside this change. ## Testing -`translate.spec.ts` covers empty and null continuation deltas, a repeated identical identity, parallel calls holding separate identities under empty continuations, and refusal when `id` or `name` never arrives — including that usage is reported before the failure and that no `block-end` precedes it. The two cases that documented lenient empty-identity output now assert refusal. `retry-policy.spec.ts` pins the new default retryable set. +`translate.spec.ts` covers empty and null continuation deltas, a repeated identical identity, and parallel calls holding separate identities under empty continuations. The existing cases for a wire that omits identity entirely keep their recorded empty-identity output. diff --git a/.agents/notes/implemented/architecture/2026-09-01-streamed-tool-call-identity.zh.md b/.agents/notes/implemented/architecture/2026-09-01-streamed-tool-call-identity.zh.md index 153d9d4fc8..9a7ffb6343 100644 --- a/.agents/notes/implemented/architecture/2026-09-01-streamed-tool-call-identity.zh.md +++ b/.agents/notes/implemented/architecture/2026-09-01-streamed-tool-call-identity.zh.md @@ -6,32 +6,28 @@ Status: implemented ## 问题 -DeepSeek SSE 翻译器对每个携带该字段的工具调用分片都直接赋值 `id` 与 `name`,因此续传分片把其中任一字段重复发送为空串时,会抹掉该调用首个分片已建立的身份。组装出的块带着空名字进入循环,工具注册表以 `unknown tool ""` 拒绝它。把这些字段填成 `null` 的网关会造成同样的抹除,而 `WireToolCallDelta` 把两者都声明为 `string | undefined`,让实际观察到的 `null` 落在编译器视野之外。 +DeepSeek SSE 翻译器对每个携带该字段的工具调用分片都直接赋值 `id` 与 `name`,因此续传分片把其中任一字段重复发送为空串时,会抹掉该调用首个分片已建立的身份。组装出的块带着空名字进入循环,工具注册表以 `unknown tool ""` 拒绝它,受影响的模型上任何工具都跑不起来。把这些字段填成 `null` 的网关会造成同样的抹除,而 `WireToolCallDelta` 把两者都声明为 `string | undefined`,让实际观察到的 `null` 落在编译器视野之外。 -空身份还会活过本轮。`appendToolCall` 与 `appendToolResult` 原样写入块的 id 且没有任何写入路径校验它,而 `adoptSessionEvent` 拒绝 `callId` 为空的 `tool/result`。记录过一次这种调用的会话可写但不再可读:持久化协调器把该拒绝包装成 `SessionPersistenceCorruptionError`。 +空身份还会活过本轮。`appendToolCall` 与 `appendToolResult` 原样写入块的 id 且没有任何写入路径校验它,而 `adoptSessionEvent` 拒绝 `callId` 为空的 `tool/result`,持久化协调器于是把该拒绝包装成 `SessionPersistenceCorruptionError`。记录过一次这种调用的会话可写但不再可读。 ## 决定 -`acceptIdentity` 对工具调用的 `id` 与 `name` 只接受非空字符串;`undefined`、`null`、`''` 以及任何非字符串都保留已建立的值。`WireToolCallDelta` 把 `id`、`function.name` 与 `function.arguments` 放宽到允许 `null`,使网关实际发送的值进入类型系统,运行时守卫因此是承重的而非臆测的。 - -流到达 `[DONE]` 时仍缺少 `id` 或 `name` 的工具调用不会被闭合。翻译器先报告待发的用量,再以携带新增 `MALFORMED_TOOL_CALL` code 的错误 finish 结束响应,并且完全不发出 `block-end`。`closeBlock` 返回缺失的是哪个字段,而不是替换成空串,因此没有任何路径能组装出无身份的工具调用。 - -`MALFORMED_TOOL_CALL` 加入[有界 LLM 请求恢复](2026-06-21-bounded-llm-request-recovery.zh.md)所述的默认可重试 code 集;两种投递方式都能让重试覆盖该尝试,因为 `LlmRuntime.adapterStream` 会把抛出的 `LlmError` 归一化为同一个错误 `finish`,再由循环路由到 `agent/request-error`。这里以 yield 而非抛出给出拒绝,是为了让该尝试已计费的用量仍先抵达循环,并让拒绝与紧邻的 `EMPTY_RESPONSE` 走同一套 chunk 协议。 +`acceptIdentity` 对工具调用的 `id` 与 `name` 只接受非空字符串;`undefined`、`null`、`''` 以及任何非字符串都保留已建立的值。会触发赋值的输入集合只减不增,因此没有任何输入会比改动前更差。`WireToolCallDelta` 把 `id`、`function.name` 与 `function.arguments` 放宽到允许 `null`,使网关实际发送的值进入类型系统,运行时守卫因此是承重的而非臆测的。 ## 考虑过的替代方案 **跨分片拼接 `id` 与 `name`。** 否决:它们是身份而非累积。面对发送 `null` 的网关,拼接产生 `Globnull`;面对重复发送非空值的网关,产生重复的名字。 -**流中途拒绝冲突的非空身份。** 推迟:分片发送长工具名的网关会因此被拒,而 `[DONE]` 处的检查已经能让不可用的调用到不了循环。 +**流中途拒绝冲突的非空身份。** 推迟:分片发送长工具名的网关会因此被拒,且没有观察到任何提供方在同一个调用 index 内改发不同的非空身份。 + +**拒绝始终未获得身份的响应。** 推迟。它需要新增失败 code、改动默认可重试集,还需要一个不得覆盖提供方已给出终止原因的 `[DONE]` 闸门——这些代价与风险,已报告的缺陷并不需要承担。它所防的宽松线上格式是假想的:没有任何报告描述过完全不发送身份的流。 **放宽会话读取端对空 `callId` 的拒绝。** 否决:空 `callId` 无法在下一次请求中与提供方配对,接受它只是把失败推进模型请求。该拒绝是持久化边界的闸门;缺陷在生产方。 -**在调用的首个分片不带 `id` 时立即拒绝。** 否决:"仅首个分片"是对远端编码器的声明,晚一个分片才发送 `id` 的网关会被无谓拒绝,而 `[DONE]` 处的检查覆盖了早检查能覆盖的全部情况。 - ## 后果 -重复发送空或 null 身份的续传分片不产生作用,调用因此保有其首个分片建立的身份。始终不给出调用身份的提供方现在的代价是一次重试,而不是一个 `unknown tool ""` 结果加一个无法重新打开的会话。已经记录了空 `callId` 的会话仍不可读;恢复它们不在本次改动范围内。 +重复发送空或 null 身份的续传分片不产生作用,调用因此保有其首个分片建立的身份,通往 `unknown tool ""` 与不可读会话的已报告路径就此切断。完全不携带身份的流仍会组装出空身份,与改动前一致;该路径以及已经写入空 `callId` 的会话恢复都不在本次改动范围内。 ## 测试 -`translate.spec.ts` 覆盖空与 null 续传分片、重复的相同身份、空续传下并行调用各自保有身份,以及 `id` 或 `name` 始终未抵达时的拒绝——包括失败前先报告用量、且其前没有 `block-end`。原先记录空身份输出的两个用例现在断言拒绝。`retry-policy.spec.ts` 钉住新的默认可重试集。 +`translate.spec.ts` 覆盖空与 null 续传分片、重复的相同身份,以及空续传下并行调用各自保有身份。原有那些描述完全不发送身份的线上格式的用例,保留其记录的空身份输出。 diff --git a/apps/web/tests/shipped-composition.e2e.ts b/apps/web/tests/shipped-composition.e2e.ts index b877b1b485..ac882e2cd6 100644 --- a/apps/web/tests/shipped-composition.e2e.ts +++ b/apps/web/tests/shipped-composition.e2e.ts @@ -93,7 +93,6 @@ it('assembles the shipped Web transport, catalog, guidance, and defaults', async "mode": "normal", "retryableCodes": [ "EMPTY_RESPONSE", - "MALFORMED_TOOL_CALL", "RATE_LIMIT", "SERVER", "TIMEOUT", @@ -127,7 +126,6 @@ it('assembles the shipped Web transport, catalog, guidance, and defaults', async "mode": "normal", "retryableCodes": [ "EMPTY_RESPONSE", - "MALFORMED_TOOL_CALL", "RATE_LIMIT", "SERVER", "TIMEOUT", diff --git a/packages/llm/llm-deepseek/README.i18n.yaml b/packages/llm/llm-deepseek/README.i18n.yaml index ec4537805f..3943c96eca 100644 --- a/packages/llm/llm-deepseek/README.i18n.yaml +++ b/packages/llm/llm-deepseek/README.i18n.yaml @@ -2,5 +2,5 @@ # side as of the last confirmed-consistent state. Both languages carry equal authority; # after editing either side, bring the other along and re-record with: # pnpm run verify-translation-pairing --write packages/llm/llm-deepseek/README.md -README.md: 9ec80aa2c0f2190fa691bdf28a6e0ed07c805c8a -README.zh.md: fe2b39059760854c1df16390233158e3432b65d6 +README.md: 6c69083909c71631dde04b453b399cc6bf687110 +README.zh.md: 9b6d35864dee20320e3f16bed82d8eecb4f39ec1 diff --git a/packages/llm/llm-deepseek/README.md b/packages/llm/llm-deepseek/README.md index 9ec80aa2c0..6c69083909 100644 --- a/packages/llm/llm-deepseek/README.md +++ b/packages/llm/llm-deepseek/README.md @@ -92,7 +92,7 @@ When `ctx.deepseekLlmApiExtensions` is present, the adapter prepares its registe ### Failures and recovery -Non-2xx responses fail with stable codes: `AUTH` (401/403), `QUOTA`, `RATE_LIMIT`, `CONTEXT_WINDOW_EXCEEDED`, `INVALID_REQUEST`, `SERVER`, and `HTTP_` otherwise; pre-response transport failures throw `TRANSPORT`, caller aborts throw `ABORTED`, and stream-idle expiry throws `TIMEOUT`. Request-extension preparation, field collision, or post-2xx acceptance fails with `REQUEST_EXTENSION`. A normalized-image rejection names every plausible attachment and its durable position when the provider does not identify a file id. Stale-file rejection invalidates the named mappings (or every mapping used by the attempt) and permits one replacement chat attempt. Protocol violations throw `STREAM_CLOSED` or `MALFORMED_RESPONSE`, a terminal `stop` with no content blocks becomes `EMPTY_RESPONSE`, and a streamed tool call still missing its `id` or `name` when the stream ends becomes `MALFORMED_TOOL_CALL`; the default retry policy retries the last two. A request with no key anywhere fails with `MISSING_CREDENTIAL`, and a malformed credential fails with `INVALID_CREDENTIAL` naming the reference to fix — never any part of the key. +Non-2xx responses fail with stable codes: `AUTH` (401/403), `QUOTA`, `RATE_LIMIT`, `CONTEXT_WINDOW_EXCEEDED`, `INVALID_REQUEST`, `SERVER`, and `HTTP_` otherwise; pre-response transport failures throw `TRANSPORT`, caller aborts throw `ABORTED`, and stream-idle expiry throws `TIMEOUT`. Request-extension preparation, field collision, or post-2xx acceptance fails with `REQUEST_EXTENSION`. A normalized-image rejection names every plausible attachment and its durable position when the provider does not identify a file id. Stale-file rejection invalidates the named mappings (or every mapping used by the attempt) and permits one replacement chat attempt. Protocol violations throw `STREAM_CLOSED` or `MALFORMED_RESPONSE`, and a terminal `stop` with no content blocks becomes `EMPTY_RESPONSE`, which the default retry policy retries. A request with no key anywhere fails with `MISSING_CREDENTIAL`, and a malformed credential fails with `INVALID_CREDENTIAL` naming the reference to fix — never any part of the key. ----- diff --git a/packages/llm/llm-deepseek/README.zh.md b/packages/llm/llm-deepseek/README.zh.md index fe2b390597..9b6d35864d 100644 --- a/packages/llm/llm-deepseek/README.zh.md +++ b/packages/llm/llm-deepseek/README.zh.md @@ -92,7 +92,7 @@ Files 模式通过 `maxRequestFilesBytes` 与 `maxImagesPerRequest` 限制保留 ### 失败与恢复 -非 2xx 响应以稳定 code 失败:`AUTH`(401/403)、`QUOTA`、`RATE_LIMIT`、`CONTEXT_WINDOW_EXCEEDED`、`INVALID_REQUEST`、`SERVER` 以及其他情况的 `HTTP_`;响应前传输失败抛出 `TRANSPORT`,调用方中止抛出 `ABORTED`,流空闲超时抛出 `TIMEOUT`。请求扩展准备、字段冲突或 2xx 后接受失败使用 `REQUEST_EXTENSION`。当提供方未指出 file id 时,规范化图片拒绝会列出所有可能附件及其持久位置。陈旧文件拒绝会使点名映射(或该次尝试使用的全部映射)失效,并允许一次替换 chat 尝试。协议违规抛出 `STREAM_CLOSED` 或 `MALFORMED_RESPONSE`;不带内容块的终止 `stop` 变成 `EMPTY_RESPONSE`,流结束时仍缺少 `id` 或 `name` 的流式工具调用变成 `MALFORMED_TOOL_CALL`,默认重试策略会重试后两者。任何位置都没有密钥的请求以 `MISSING_CREDENTIAL` 失败;格式错误的凭据以 `INVALID_CREDENTIAL` 失败,并点名需要修复的引用——绝不包含密钥的任何部分。 +非 2xx 响应以稳定 code 失败:`AUTH`(401/403)、`QUOTA`、`RATE_LIMIT`、`CONTEXT_WINDOW_EXCEEDED`、`INVALID_REQUEST`、`SERVER` 以及其他情况的 `HTTP_`;响应前传输失败抛出 `TRANSPORT`,调用方中止抛出 `ABORTED`,流空闲超时抛出 `TIMEOUT`。请求扩展准备、字段冲突或 2xx 后接受失败使用 `REQUEST_EXTENSION`。当提供方未指出 file id 时,规范化图片拒绝会列出所有可能附件及其持久位置。陈旧文件拒绝会使点名映射(或该次尝试使用的全部映射)失效,并允许一次替换 chat 尝试。协议违规抛出 `STREAM_CLOSED` 或 `MALFORMED_RESPONSE`;不带内容块的终止 `stop` 变成 `EMPTY_RESPONSE`,默认重试策略会重试它。任何位置都没有密钥的请求以 `MISSING_CREDENTIAL` 失败;格式错误的凭据以 `INVALID_CREDENTIAL` 失败,并点名需要修复的引用——绝不包含密钥的任何部分。 ----- diff --git a/packages/llm/llm-deepseek/src/translate.ts b/packages/llm/llm-deepseek/src/translate.ts index 3d98c31f4b..8dd939e547 100644 --- a/packages/llm/llm-deepseek/src/translate.ts +++ b/packages/llm/llm-deepseek/src/translate.ts @@ -9,7 +9,7 @@ */ import { brandString } from '@deepseek-ai/dsh-brand' -import { EMPTY_RESPONSE_CODE, LlmError, MALFORMED_TOOL_CALL_CODE } from '@deepseek-ai/dsh-llm' +import { EMPTY_RESPONSE_CODE, LlmError } from '@deepseek-ai/dsh-llm' import type { ContentBlock, FinishReason, StreamChunk, TokenUsage, ToolCallId } from '@deepseek-ai/dsh-llm' import { DONE } from './sse.ts' import type { WireChunk, WireUsage } from './types.ts' @@ -86,28 +86,16 @@ function acceptIdentity(current: string | undefined, incoming: unknown): string return typeof incoming === 'string' && incoming.length > 0 ? incoming : current } -/** The identity field a streamed tool call never received. */ -interface Unidentified { - unidentified: 'id' | 'name' -} - -/** - * Assemble the final ContentBlock for one open block. - * @param block - the block to close. - * @returns the assembled block, or which identity field a tool call is missing. - * A tool call without both fields cannot be dispatched, and its result cannot - * be paired back to the provider, so the caller rejects the whole response - * rather than closing the block. - */ -function closeBlock(block: OpenBlock): ContentBlock | Unidentified { +/** Assemble the final ContentBlock for one open block. */ +function closeBlock(block: OpenBlock): ContentBlock { switch (block.kind) { case 'text': return { type: 'text', text: block.text } case 'reasoning': return { type: 'reasoning', text: block.text } - case 'tool-call': { - const { callId, name } = block - if (callId === undefined) return { unidentified: 'id' } - if (name === undefined) return { unidentified: 'name' } - return { type: 'tool-call', id: brandString(callId), name, arguments: block.text } + case 'tool-call': return { + type: 'tool-call', + id: brandString(block.callId ?? ''), + name: block.name ?? '', + arguments: block.text, } } } @@ -118,9 +106,7 @@ function closeBlock(block: OpenBlock): ContentBlock | Unidentified { * @param payloads - SSE data payloads from {@link parseSse}, `[DONE]`-terminated. * @returns deltas as they arrive; `block-end`s, `usage`, and `finish` are all deferred to the `[DONE]` sentinel. * A `stop` (or absent) finish with no opened blocks is a degenerate provider completion and maps to an - * `EMPTY_RESPONSE` error finish instead of a successful empty message. A tool call left without an `id` or - * `name` is unusable, so the response ends in a `MALFORMED_TOOL_CALL` error finish, after any usage and - * without a `block-end` for any block. + * `EMPTY_RESPONSE` error finish instead of a successful empty message. */ export async function* translate(payloads: AsyncIterable): AsyncGenerator { let nextIndex = 0 @@ -139,29 +125,9 @@ export async function* translate(payloads: AsyncIterable): AsyncGenerato for await (const payload of payloads) { if (payload === DONE) { - const ends: StreamChunk[] = [] for (const block of order) { - const closed = closeBlock(block) - if ('unidentified' in closed) { - // A rejected response commits no assistant message, tool call, or - // tool result; the streamed chunks before it stay in the log, so the - // usage the attempt already burned is reported before the failure. - if (pendingUsage) yield { type: 'usage', usage: pendingUsage } - yield { - type: 'finish', - reason: { - kind: 'error', - failure: { - message: `model streamed a tool call with no ${closed.unidentified}`, - code: MALFORMED_TOOL_CALL_CODE, - }, - }, - } - return - } - ends.push({ type: 'block-end', index: block.index, block: closed }) + yield { type: 'block-end', index: block.index, block: closeBlock(block) } } - yield* ends if (pendingUsage) yield { type: 'usage', usage: pendingUsage } const reason = pendingFinish ?? { kind: 'stop' as const } yield { @@ -219,9 +185,6 @@ export async function* translate(payloads: AsyncIterable): AsyncGenerato block.name = acceptIdentity(block.name, call.function?.name) const fragment = call.function?.arguments ?? '' block.text += fragment - // An in-flight delta may precede the call's id. The empty stand-in never - // reaches a ContentBlock: `[DONE]` either has the identity by then or - // refuses the whole response before closing the block. yield { type: 'tool-call-delta', index: block.index, diff --git a/packages/llm/llm-deepseek/tests/translate.spec.ts b/packages/llm/llm-deepseek/tests/translate.spec.ts index eb4d17dc8a..43ff59822f 100644 --- a/packages/llm/llm-deepseek/tests/translate.spec.ts +++ b/packages/llm/llm-deepseek/tests/translate.spec.ts @@ -1,5 +1,5 @@ import { describe, expect, it } from 'vitest' -import { BlockAssembler, EMPTY_RESPONSE_CODE, LlmError, MALFORMED_TOOL_CALL_CODE } from '@deepseek-ai/dsh-llm' +import { BlockAssembler, EMPTY_RESPONSE_CODE, LlmError } from '@deepseek-ai/dsh-llm' import type { StreamChunk } from '@deepseek-ai/dsh-llm' import { DONE } from '../src/sse.ts' import { mapFinishReason, mapUsage, translate } from '../src/translate.ts' @@ -328,24 +328,19 @@ describe('mapUsage', () => { }) describe('translate: defensive tool-call branches', () => { - it('rejects a stream whose tool call never carries an id, reporting usage first', async () => { + it('handles deltas that never carry id or name (empty-string fallbacks)', async () => { const chunks = await collect(translate(feed( firstChunk, + // Hypothetical lenient wire: argument fragments with no id/name at all. { choices: [{ delta: { tool_calls: [{ index: 0, function: { arguments: '{}' } }] } }] }, - { choices: [{ delta: {}, finish_reason: 'tool_calls' }], usage: { prompt_tokens: 12, completion_tokens: 4 } }, + { choices: [{ delta: {}, finish_reason: 'tool_calls' }] }, DONE, ))) expect(chunks).toEqual([ { type: 'block-start', index: 0, blockType: 'tool-call' }, { type: 'tool-call-delta', index: 0, id: '', argumentsDelta: '{}' }, - { type: 'usage', usage: { inputTokens: 12, outputTokens: 4, totalTokens: 16 } }, - { - type: 'finish', - reason: { - kind: 'error', - failure: { message: 'model streamed a tool call with no id', code: MALFORMED_TOOL_CALL_CODE }, - }, - }, + { type: 'block-end', index: 0, block: { type: 'tool-call', id: '', name: '', arguments: '{}' } }, + { type: 'finish', reason: { kind: 'tool-calls' } }, ]) }) @@ -359,25 +354,7 @@ describe('translate: defensive tool-call branches', () => { expect(chunks[1]).toEqual({ type: 'tool-call-delta', index: 0, id: 'c', name: 'f', argumentsDelta: '' }) }) - it('suppresses the block-end of an already closable block when a later tool call has no identity', async () => { - const chunks = await collect(translate(feed( - firstChunk, - { choices: [{ delta: { content: 'Checking.' } }] }, - { choices: [{ delta: { tool_calls: [{ index: 0, function: { arguments: '{}' } }] } }] }, - { choices: [{ delta: {}, finish_reason: 'tool_calls' }] }, - DONE, - ))) - expect(chunks.some(chunk => chunk.type === 'block-end')).toBe(false) - expect(chunks.at(-1)).toEqual({ - type: 'finish', - reason: { - kind: 'error', - failure: { message: 'model streamed a tool call with no id', code: MALFORMED_TOOL_CALL_CODE }, - }, - }) - }) - - it('rejects a stream whose tool call never carries a name', async () => { + it('handles tool_call deltas with no function object at all', async () => { const chunks = await collect(translate(feed( firstChunk, { choices: [{ delta: { tool_calls: [{ index: 0, id: 'c' }] } }] }, @@ -385,14 +362,6 @@ describe('translate: defensive tool-call branches', () => { DONE, ))) expect(chunks[1]).toEqual({ type: 'tool-call-delta', index: 0, id: 'c', argumentsDelta: '' }) - expect(chunks.some(chunk => chunk.type === 'block-end')).toBe(false) - expect(chunks.at(-1)).toEqual({ - type: 'finish', - reason: { - kind: 'error', - failure: { message: 'model streamed a tool call with no name', code: MALFORMED_TOOL_CALL_CODE }, - }, - }) }) }) diff --git a/packages/llm/llm-retry/README.i18n.yaml b/packages/llm/llm-retry/README.i18n.yaml index 2138717836..295b07d56e 100644 --- a/packages/llm/llm-retry/README.i18n.yaml +++ b/packages/llm/llm-retry/README.i18n.yaml @@ -2,5 +2,5 @@ # side as of the last confirmed-consistent state. Both languages carry equal authority; # after editing either side, bring the other along and re-record with: # pnpm run verify-translation-pairing --write packages/llm/llm-retry/README.md -README.md: 84e4a731e60619e8332d339f51276ac31d615661 -README.zh.md: b5f38e40321af0a9f04803fbef30a8b73cd27cfc +README.md: 0ea815130212a165582540e43aad7d59fad32d2c +README.zh.md: 43a0187c68d272e2764d34db35fa26e51a9d3a83 diff --git a/packages/llm/llm-retry/README.md b/packages/llm/llm-retry/README.md index 84e4a731e6..0ea8151302 100644 --- a/packages/llm/llm-retry/README.md +++ b/packages/llm/llm-retry/README.md @@ -47,7 +47,7 @@ Choose it when a composition runs the agent loop and wants durable request recov - name: '@deepseek-ai/dsh-llm-retry' ``` -Omission of `retryPolicy` uses normal mode: five retries for `EMPTY_RESPONSE`, `MALFORMED_TOOL_CALL`, `RATE_LIMIT`, `SERVER`, `TIMEOUT`, and `TRANSPORT`, with bounded exponential backoff from 500 ms to 10 seconds and 10 percent jitter. Normal mode can change its finite budget, eligible codes, and backoff; always mode asks downstream recovery first, then retries every model-request failure without an attempt limit, stopping only on success, cancellation, or plugin disposal. +Omission of `retryPolicy` uses normal mode: five retries for `EMPTY_RESPONSE`, `RATE_LIMIT`, `SERVER`, `TIMEOUT`, and `TRANSPORT`, with bounded exponential backoff from 500 ms to 10 seconds and 10 percent jitter. Normal mode can change its finite budget, eligible codes, and backoff; always mode asks downstream recovery first, then retries every model-request failure without an attempt limit, stopping only on success, cancellation, or plugin disposal. ### What you can observe diff --git a/packages/llm/llm-retry/README.zh.md b/packages/llm/llm-retry/README.zh.md index b5f38e4032..43a0187c68 100644 --- a/packages/llm/llm-retry/README.zh.md +++ b/packages/llm/llm-retry/README.zh.md @@ -47,7 +47,7 @@ kind: "package-reference" - name: '@deepseek-ai/dsh-llm-retry' ``` -省略 `retryPolicy` 时使用 normal mode:对 `EMPTY_RESPONSE`、`MALFORMED_TOOL_CALL`、`RATE_LIMIT`、`SERVER`、`TIMEOUT` 与 `TRANSPORT` 最多重试五次,退避从 500 毫秒到 10 秒、带 10% 抖动。normal mode 可以更改其有界预算、合格 code 与退避;always mode 先询问下游恢复,然后无尝试上限地重试每个模型请求失败,只在成功、取消或插件释放时停止。 +省略 `retryPolicy` 时使用 normal mode:对 `EMPTY_RESPONSE`、`RATE_LIMIT`、`SERVER`、`TIMEOUT` 与 `TRANSPORT` 最多重试五次,退避从 500 毫秒到 10 秒、带 10% 抖动。normal mode 可以更改其有界预算、合格 code 与退避;always mode 先询问下游恢复,然后无尝试上限地重试每个模型请求失败,只在成功、取消或插件释放时停止。 ### 你可以观察到什么 diff --git a/packages/llm/llm/src/assembler.ts b/packages/llm/llm/src/assembler.ts index 09e9f81a31..88ae77d80b 100644 --- a/packages/llm/llm/src/assembler.ts +++ b/packages/llm/llm/src/assembler.ts @@ -110,11 +110,6 @@ export class BlockAssembler { switch (partial.blockType) { case 'text': return { type: 'text', text: partial.text } case 'reasoning': return { type: 'reasoning', text: partial.text } - // TODO: the delta-only fallback still invents an empty name for a tool - // call that never carried one. No shipped adapter reaches it — the - // DeepSeek translator refuses such a response outright (see - // .agents/notes/implemented/architecture/2026-09-01-streamed-tool-call-identity.md) - // — but a future delta-only adapter would assemble an undispatchable call. case 'tool-call': return { type: 'tool-call', id: partial.toolCallId ?? brandString(`call-${index}`), diff --git a/packages/llm/llm/src/error.ts b/packages/llm/llm/src/error.ts index 90f3785315..4b3e25c75f 100644 --- a/packages/llm/llm/src/error.ts +++ b/packages/llm/llm/src/error.ts @@ -38,18 +38,6 @@ export const QUOTA_EXCEEDED_CODE = 'QUOTA' */ export const EMPTY_RESPONSE_CODE = 'EMPTY_RESPONSE' -/** - * Canonical provider-neutral code for a streamed tool call the provider never - * identified: its `id` or `name` was absent or empty by the end of the stream. - * Such a call cannot be dispatched, and its result cannot be paired back to the - * provider on the next request, so adapters classify it as this failure instead - * of emitting a tool call the loop would reject as unknown. No assistant - * message or tool result is written for the attempt — only the streamed - * `assistant/chunk` events it already produced — so retry policy treats it as - * safe to repeat. - */ -export const MALFORMED_TOOL_CALL_CODE = 'MALFORMED_TOOL_CALL' - /** * Canonical provider-neutral code for a credential that was supplied but * cannot be used — malformed rather than absent. Distinct from diff --git a/packages/llm/llm/src/retry-policy.ts b/packages/llm/llm/src/retry-policy.ts index ae9ccf6c2f..f6e6175cb9 100644 --- a/packages/llm/llm/src/retry-policy.ts +++ b/packages/llm/llm/src/retry-policy.ts @@ -9,7 +9,7 @@ import z from '@deepseek-ai/schemastery' import { MAX_TIMER_DELAY_MS } from '@deepseek-ai/dsh-timeout' -import { EMPTY_RESPONSE_CODE, MALFORMED_TOOL_CALL_CODE } from './error.ts' +import { EMPTY_RESPONSE_CODE } from './error.ts' const DEFAULT_MAX_RETRIES = 5 const DEFAULT_INITIAL_DELAY_MS = 500 @@ -17,7 +17,6 @@ const DEFAULT_MAX_DELAY_MS = 10_000 const DEFAULT_JITTER_RATIO = 0.1 const DEFAULT_RETRYABLE_CODES = Object.freeze([ EMPTY_RESPONSE_CODE, - MALFORMED_TOOL_CALL_CODE, 'RATE_LIMIT', 'SERVER', 'TIMEOUT', diff --git a/packages/llm/llm/tests/retry-policy.spec.ts b/packages/llm/llm/tests/retry-policy.spec.ts index 1cfa3c4c6d..cc7ebb8fa7 100644 --- a/packages/llm/llm/tests/retry-policy.spec.ts +++ b/packages/llm/llm/tests/retry-policy.spec.ts @@ -13,7 +13,7 @@ describe('provider retry policy', () => { expect(policy).toEqual({ mode: 'normal', maxRetries: 5, - retryableCodes: ['EMPTY_RESPONSE', 'MALFORMED_TOOL_CALL', 'RATE_LIMIT', 'SERVER', 'TIMEOUT', 'TRANSPORT'], + retryableCodes: ['EMPTY_RESPONSE', 'RATE_LIMIT', 'SERVER', 'TIMEOUT', 'TRANSPORT'], initialDelayMs: 500, maxDelayMs: 10_000, jitterRatio: 0.1, diff --git a/snapshots/session/empty-response-retry/session.jsonl b/snapshots/session/empty-response-retry/session.jsonl index 79360b4195..3302c2c540 100644 --- a/snapshots/session/empty-response-retry/session.jsonl +++ b/snapshots/session/empty-response-retry/session.jsonl @@ -13,7 +13,7 @@ {"type":"request/context","data":{"provider":"deepseek-official","model":"deepseek-v4-flash"}} {"type":"assistant/chunk","data":{"turn":1,"step":1,"chunk":{"type":"usage","usage":{"inputTokens":0,"outputTokens":0}}}} {"type":"assistant/chunk","data":{"turn":1,"step":1,"chunk":{"type":"finish","reason":{"kind":"error","failure":{"message":"model returned a completed response with no content","code":"EMPTY_RESPONSE"}}}}} -{"type":"llm/retry","data":{"retryId":"{{retry:1}}","turn":1,"step":1,"provider":"deepseek-official","mode":"normal","policyKey":"[\"normal\",2,[\"EMPTY_RESPONSE\",\"MALFORMED_TOOL_CALL\",\"RATE_LIMIT\",\"SERVER\",\"TIMEOUT\",\"TRANSPORT\"],1,1,0]","retry":1,"maxRetries":2,"delayMs":1,"failure":{"message":"model returned a completed response with no content","code":"EMPTY_RESPONSE"}}} +{"type":"llm/retry","data":{"retryId":"{{retry:1}}","turn":1,"step":1,"provider":"deepseek-official","mode":"normal","policyKey":"[\"normal\",2,[\"EMPTY_RESPONSE\",\"RATE_LIMIT\",\"SERVER\",\"TIMEOUT\",\"TRANSPORT\"],1,1,0]","retry":1,"maxRetries":2,"delayMs":1,"failure":{"message":"model returned a completed response with no content","code":"EMPTY_RESPONSE"}}} {"type":"llm/retry-started","data":{"retryId":"{{retry:1}}","turn":1,"step":1,"retry":1}} {"type":"assistant/chunk","data":{"turn":1,"step":1,"chunk":{"type":"block-start","index":0,"blockType":"text"}}} {"type":"assistant/chunk","data":{"turn":1,"step":1,"chunk":{"type":"text-delta","index":0,"text":"Recovered."}}} diff --git a/snapshots/session/malformed-tool-call-retry/session.jsonl b/snapshots/session/malformed-tool-call-retry/session.jsonl deleted file mode 100644 index 84592e46fc..0000000000 --- a/snapshots/session/malformed-tool-call-retry/session.jsonl +++ /dev/null @@ -1,27 +0,0 @@ -{"type":"session","version":0,"id":"{{session:1}}","createdAt":0,"cwd":"{{cwd}}","delegationDepth":0} -{"type":"permission/preset","data":{"preset":"danger-full-access"}} -{"type":"sandbox/mode","data":{"mode":"danger-full-access"}} -{"type":"approval/policy","data":{"policy":"never"}} -{"type":"agent/inbox/spliced","data":{"target":"next-turn","start":0,"inserted":[{"content":[{"type":"text","text":"This prompt first receives a tool call the model never named, then a retried reply."}],"source":{"kind":"user"},"role":"user","id":"{{message:1}}"}]}} -{"type":"turn/start","data":{"turn":1}} -{"type":"agent/inbox/spliced","data":{"target":"next-turn","start":0,"removedCount":1,"inserted":[]}} -{"type":"step/start","data":{"turn":1,"step":1}} -{"type":"user/message","data":{"content":[{"type":"text","text":"This prompt first receives a tool call the model never named, then a retried reply."}],"source":{"kind":"user"},"role":"user","id":"{{message:1}}"},"surfaceOp":"append"} -{"type":"user/message","data":{"content":[{"type":"text","text":"Current runtime context. This snapshot supersedes earlier runtime-context snapshots.\n\nCurrent DSH file policy: danger-full-access. The DSH file sandbox does not restrict file modifications by available operations.\n\nApproval prompts are disabled in this session: actions that require approval are rejected automatically — do not request sandbox escalation (do not set `sandbox_permissions`)."}],"source":{"kind":"plugin","plugin":"@deepseek-ai/dsh-system-prompt","form":"snapshot","sections":[{"name":"sandbox:policy","text":"Current DSH file policy: danger-full-access. The DSH file sandbox does not restrict file modifications by available operations."},{"name":"approval:policy","text":"Approval prompts are disabled in this session: actions that require approval are rejected automatically — do not request sandbox escalation (do not set `sandbox_permissions`)."}]},"role":"user","id":"{{message:2}}"},"surfaceOp":"append"} -{"type":"session/title","data":{"title":"This prompt first receives a","messageSeqs":[7],"source":{"kind":"fallback"}}} -{"type":"request/header","data":{"header":{"config":{"provider":"deepseek-official","model":"deepseek-v4-flash"},"system":"{{system}}","tools":"{{tools}}"},"reason":"initial"}} -{"type":"request/context","data":{"provider":"deepseek-official","model":"deepseek-v4-flash"}} -{"type":"assistant/chunk","data":{"turn":1,"step":1,"chunk":{"type":"block-start","index":0,"blockType":"tool-call"}}} -{"type":"assistant/chunk","data":{"turn":1,"step":1,"chunk":{"type":"tool-call-delta","index":0,"id":"unnamed-call","argumentsDelta":"{\"path\":\"README.md\"}"}}} -{"type":"assistant/chunk","data":{"turn":1,"step":1,"chunk":{"type":"usage","usage":{"inputTokens":14,"outputTokens":6}}}} -{"type":"assistant/chunk","data":{"turn":1,"step":1,"chunk":{"type":"finish","reason":{"kind":"error","failure":{"message":"model streamed a tool call with no name","code":"MALFORMED_TOOL_CALL"}}}}} -{"type":"llm/retry","data":{"retryId":"{{retry:1}}","turn":1,"step":1,"provider":"deepseek-official","mode":"normal","policyKey":"[\"normal\",2,[\"EMPTY_RESPONSE\",\"MALFORMED_TOOL_CALL\",\"RATE_LIMIT\",\"SERVER\",\"TIMEOUT\",\"TRANSPORT\"],1,1,0]","retry":1,"maxRetries":2,"delayMs":1,"failure":{"message":"model streamed a tool call with no name","code":"MALFORMED_TOOL_CALL"}}} -{"type":"llm/retry-started","data":{"retryId":"{{retry:1}}","turn":1,"step":1,"retry":1}} -{"type":"assistant/chunk","data":{"turn":1,"step":1,"chunk":{"type":"block-start","index":0,"blockType":"text"}}} -{"type":"assistant/chunk","data":{"turn":1,"step":1,"chunk":{"type":"text-delta","index":0,"text":"Recovered."}}} -{"type":"assistant/chunk","data":{"turn":1,"step":1,"chunk":{"type":"block-end","index":0,"block":{"type":"text","text":"Recovered."}}}} -{"type":"assistant/chunk","data":{"turn":1,"step":1,"chunk":{"type":"usage","usage":{"inputTokens":12,"outputTokens":3}}}} -{"type":"assistant/chunk","data":{"turn":1,"step":1,"chunk":{"type":"finish","reason":{"kind":"stop"}}}} -{"type":"assistant/message","data":{"turn":1,"step":1,"message":{"role":"assistant","content":[{"type":"text","text":"Recovered."}],"source":{"kind":"model","provider":"deepseek-official","model":"deepseek-v4-flash"},"id":"{{message:3}}"},"usage":{"inputTokens":12,"outputTokens":3}},"sourceEventSeqs":[18,19,20,21,22],"surfaceOp":"append"} -{"type":"step/end","data":{"turn":1,"step":1}} -{"type":"turn/end","data":{"turn":1,"reason":{"kind":"completed"}}} diff --git a/snapshots/session/malformed-tool-call-retry/snapshot.yml b/snapshots/session/malformed-tool-call-retry/snapshot.yml deleted file mode 100644 index 40bb505dd2..0000000000 --- a/snapshots/session/malformed-tool-call-retry/snapshot.yml +++ /dev/null @@ -1,9 +0,0 @@ -version: 1 -scenario: malformed-tool-call-retry -profile: headless -composition: retry -recording: authored -header: - class: retry - systemPromptSource: text-turn - toolSchemasSource: text-turn