fix(subagent): address Claude permission review findings
This commit is contained in:
parent
4d03472cd0
commit
62da706b64
18 changed files with 148 additions and 125 deletions
|
|
@ -2,5 +2,5 @@
|
|||
# side as of the last confirmed-consistent state. Both languages carry equal authority;
|
||||
# after editing either side, bring the other along and re-record with:
|
||||
# pnpm run verify-translation-pairing --write .agents/notes/implemented/feature/2026-08-15-product-subagent-noninteractive-permissions.md
|
||||
2026-08-15-product-subagent-noninteractive-permissions.md: f382bc7ad058fefd8001da6181824fc9b6f767d4
|
||||
2026-08-15-product-subagent-noninteractive-permissions.zh.md: 76cf53c7c9af791db6e54a8b779a7284187d0716
|
||||
2026-08-15-product-subagent-noninteractive-permissions.md: d4d29d982e5eb2a06f7cb710860ce72c506c4ade
|
||||
2026-08-15-product-subagent-noninteractive-permissions.zh.md: 3431465e6240e169dd8d240628d651348ac029b7
|
||||
|
|
|
|||
|
|
@ -19,12 +19,12 @@ The Claude Code Provider owns one Profile-level `permissionMode` value. It defau
|
|||
| `dontAsk` | Deny operations that are not already authorized instead of prompting. |
|
||||
| `acceptEdits` | Accept edits; deny any remaining permission prompt through the unattended callback. |
|
||||
| `auto` | Let Claude Code's native classifier allow or deny permission requests. |
|
||||
| `plan` | Use Claude Code's planning-only mode without tool execution. |
|
||||
| `plan` | Use planning mode, deny execution approval, and return the completed plan as the final answer. |
|
||||
| `bypassPermissions` | Set the SDK's explicit dangerous confirmation and bypass permission checks. |
|
||||
|
||||
The Provider fixes the resolved value for every run from that plugin instance. The subagent tool schema and `SubagentStartRequest` contain no permission field, so a model or individual delegation cannot change it. The Provider continues to omit `settingSources`: Claude Code remains the owner of user, project, and local settings, authentication, tools, and sandbox behavior outside the selected mode.
|
||||
|
||||
Every query disables `AskUserQuestion`. Non-bypass permission callbacks deny instead of returning the SDK's indefinitely blocking `null`; MCP elicitation is declined; the supported refusal dialog is cancelled; undeclared dialog kinds use the SDK's no-dialog failure behavior. A native `permission_denied` message records the same operation-local fact. These paths do not create an approval session, queue, cache, or retry loop.
|
||||
Every query disables `AskUserQuestion`. Non-bypass permission callbacks deny instead of returning the SDK's indefinitely blocking `null`; in plan mode, `ExitPlanMode` receives a fixed denial that tells the model to return the completed plan without executing it. MCP elicitation is declined; the supported refusal dialog is cancelled; undeclared dialog kinds use the SDK's no-dialog failure behavior. A native `permission_denied` message records the same operation-local fact. These paths do not create an approval session, queue, cache, or retry loop.
|
||||
|
||||
### Failure diagnostic
|
||||
|
||||
|
|
|
|||
|
|
@ -19,12 +19,12 @@ Claude Code 提供方拥有一个 Profile 级 `permissionMode` 值。它默认
|
|||
| `dontAsk` | 不弹出提示,直接拒绝尚未获授权的操作。 |
|
||||
| `acceptEdits` | 接受编辑;其余权限提示由无人值守回调拒绝。 |
|
||||
| `auto` | 由 Claude Code 原生分类器允许或拒绝权限请求。 |
|
||||
| `plan` | 使用 Claude Code 的仅规划模式,不执行工具。 |
|
||||
| `plan` | 使用规划模式,拒绝执行审批,并把完整计划作为最终答案返回。 |
|
||||
| `bypassPermissions` | 设置 SDK 的显式危险确认并跳过权限检查。 |
|
||||
|
||||
提供方会为该插件实例的每次运行固定已解析值。subagent 工具 schema 与 `SubagentStartRequest` 都不包含权限字段,因此模型或单次委派无法改变它。提供方继续省略 `settingSources`:除所选模式以外,用户、项目和本地设置、身份验证、工具与沙箱行为仍由 Claude Code 拥有。
|
||||
|
||||
每次 query 都禁用 `AskUserQuestion`。非 bypass 模式的权限回调会拒绝请求,而不会返回 SDK 中会无限阻塞的 `null`;MCP elicitation 会被拒绝;已支持的拒绝对话会被取消;未声明的对话类型使用 SDK 的无对话失败行为。原生 `permission_denied` 消息会记录同一份当前运行事实。这些路径不会创建审批会话、队列、缓存或重试循环。
|
||||
每次 query 都禁用 `AskUserQuestion`。非 bypass 模式的权限回调会拒绝请求,而不会返回 SDK 中会无限阻塞的 `null`;在 plan 模式下,`ExitPlanMode` 会收到一项固定拒绝,要求模型返回完整计划且不得执行。MCP elicitation 会被拒绝;已支持的拒绝对话会被取消;未声明的对话类型使用 SDK 的无对话失败行为。原生 `permission_denied` 消息会记录同一份当前运行事实。这些路径不会创建审批会话、队列、缓存或重试循环。
|
||||
|
||||
### 失败诊断
|
||||
|
||||
|
|
|
|||
|
|
@ -2,5 +2,5 @@
|
|||
# side as of the last confirmed-consistent state. Both languages carry equal authority;
|
||||
# after editing either side, bring the other along and re-record with:
|
||||
# pnpm run verify-translation-pairing --write docs/config-catalog.md
|
||||
config-catalog.md: 8294c2187f2b80fbf36787c784ad8b73a16206c1
|
||||
config-catalog.zh.md: f35392a5b005212067c9b593b7fa2818202466dd
|
||||
config-catalog.md: 1c78a854366e9bcd4633c56fcf31f0c6a55cefb1
|
||||
config-catalog.zh.md: cb70da0ede446aabfada3e93bc3d23c73ccb8271
|
||||
|
|
|
|||
|
|
@ -2088,19 +2088,19 @@ export interface Config {
|
|||
* credential-scrubbed parent environment.
|
||||
*/
|
||||
env?: Record<string, string>
|
||||
/** Native non-interactive permission mode fixed for this Provider instance. */
|
||||
/**
|
||||
* Native non-interactive mode fixed for this Provider instance. Defaults to
|
||||
* `dontAsk`; `acceptEdits` accepts edits, `auto` uses the native classifier,
|
||||
* `plan` returns a plan without approving execution, and
|
||||
* `bypassPermissions` explicitly skips permission checks.
|
||||
*/
|
||||
permissionMode?: ClaudeCodePermissionMode
|
||||
/** Grace in milliseconds for Claude Code process-tree termination. */
|
||||
disposeGraceMs?: number
|
||||
}
|
||||
|
||||
/** Profile-selectable non-interactive Claude Code permission mode. */
|
||||
export type ClaudeCodePermissionMode =
|
||||
| 'dontAsk'
|
||||
| 'acceptEdits'
|
||||
| 'auto'
|
||||
| 'plan'
|
||||
| 'bypassPermissions'
|
||||
export type ClaudeCodePermissionMode = typeof CLAUDE_CODE_PERMISSION_MODES[number]
|
||||
```
|
||||
|
||||
Source: [`packages/subagent/subagent-claude-code/src/index.ts:35`](../packages/subagent/subagent-claude-code/src/index.ts)
|
||||
|
|
|
|||
|
|
@ -2090,19 +2090,19 @@ export interface Config {
|
|||
* credential-scrubbed parent environment.
|
||||
*/
|
||||
env?: Record<string, string>
|
||||
/** Native non-interactive permission mode fixed for this Provider instance. */
|
||||
/**
|
||||
* Native non-interactive mode fixed for this Provider instance. Defaults to
|
||||
* `dontAsk`; `acceptEdits` accepts edits, `auto` uses the native classifier,
|
||||
* `plan` returns a plan without approving execution, and
|
||||
* `bypassPermissions` explicitly skips permission checks.
|
||||
*/
|
||||
permissionMode?: ClaudeCodePermissionMode
|
||||
/** Grace in milliseconds for Claude Code process-tree termination. */
|
||||
disposeGraceMs?: number
|
||||
}
|
||||
|
||||
/** Profile-selectable non-interactive Claude Code permission mode. */
|
||||
export type ClaudeCodePermissionMode =
|
||||
| 'dontAsk'
|
||||
| 'acceptEdits'
|
||||
| 'auto'
|
||||
| 'plan'
|
||||
| 'bypassPermissions'
|
||||
export type ClaudeCodePermissionMode = typeof CLAUDE_CODE_PERMISSION_MODES[number]
|
||||
```
|
||||
|
||||
来源:[`packages/subagent/subagent-claude-code/src/index.ts:35`](../packages/subagent/subagent-claude-code/src/index.ts)
|
||||
|
|
|
|||
|
|
@ -148,7 +148,7 @@ const SCENARIOS: Scenario[] = [
|
|||
hasModelTurn: true,
|
||||
recorded: false,
|
||||
pinsHeader: true,
|
||||
headerClass: 'product-subagent-result-diagnostic',
|
||||
headerClass: 'product-subagent-codex',
|
||||
configPath: PRODUCT_SUBAGENT_CODEX_CONFIG,
|
||||
},
|
||||
{
|
||||
|
|
@ -165,10 +165,7 @@ const SCENARIOS: Scenario[] = [
|
|||
hasModelTurn: true,
|
||||
recorded: false,
|
||||
overridden: true,
|
||||
pinsHeader: true,
|
||||
headerClass: 'product-subagent-codex',
|
||||
systemPromptSource: 'product-subagent-codex',
|
||||
toolSchemasSource: 'product-subagent-codex',
|
||||
configPath: PRODUCT_SUBAGENT_RESULT_DIAGNOSTIC_CONFIG,
|
||||
},
|
||||
{
|
||||
|
|
|
|||
|
|
@ -2,5 +2,5 @@
|
|||
# side as of the last confirmed-consistent state. Both languages carry equal authority;
|
||||
# after editing either side, bring the other along and re-record with:
|
||||
# pnpm run verify-translation-pairing --write packages/subagent/subagent-claude-code/README.md
|
||||
README.md: c74c092d58d7853cedee3c5f326467a5036c50fb
|
||||
README.zh.md: e87f03399e6ffd926d0ea25c6f84339ba8c94d6f
|
||||
README.md: e7c5debddfdc740802d7bc25c2a863c7de287d07
|
||||
README.zh.md: 9e68b5f3f3824ffc3913fdba159c95b5c94353c9
|
||||
|
|
|
|||
|
|
@ -16,7 +16,7 @@ Local cancellation wins the result race and maps to `aborted`. `dispose()` is id
|
|||
|
||||
The provider deliberately omits the SDK `settingSources` option. The official SDK therefore reads the host's normal user, project, and local Claude settings relative to the parent Session cwd, including native account state and product configuration. The provider neither copies nor filters those files and does not create or modify login state. The Profile-selected `permissionMode` is the one query-level override: Claude Code still owns its settings and sandbox, while the selected native mode decides how this unattended query handles permission checks.
|
||||
|
||||
Each query sets `persistSession: false` and disables `AskUserQuestion`. Except in bypass mode, `canUseTool` immediately denies requests that still require human approval. MCP elicitation is declined, the known refusal fallback dialog is cancelled, and undeclared dialog kinds use the SDK's no-dialog failure behavior. These decisions never wait for a user interface. A permission denial or unattended callback that contributes to a failed run produces an optional `SubagentResult.diagnostic` containing only the product, effective mode, request category, decision, and fixed safe reason; the shared result boundary limits the complete text to 4096 UTF-8 bytes. Successful and locally cancelled runs do not expose the captured failure detail.
|
||||
Each query sets `persistSession: false` and disables `AskUserQuestion`. Except in bypass mode, `canUseTool` immediately denies requests that still require human approval. In plan mode, the `ExitPlanMode` approval is denied with a fixed instruction to return the completed plan as the final answer without executing it. MCP elicitation is declined, the known refusal fallback dialog is cancelled, and undeclared dialog kinds use the SDK's no-dialog failure behavior. These decisions never wait for a user interface. A permission denial or unattended callback that contributes to a failed run produces an optional `SubagentResult.diagnostic` containing only the product, effective mode, request category, decision, and fixed safe reason; the shared result boundary limits the complete text to 4096 UTF-8 bytes. Successful and locally cancelled runs do not expose the captured failure detail.
|
||||
|
||||
## Capabilities and context
|
||||
|
||||
|
|
@ -35,7 +35,7 @@ The provider advertises no optional start-time capabilities and reports `inherit
|
|||
| `dontAsk` | Deny operations that are not already authorized instead of prompting. |
|
||||
| `acceptEdits` | Accept file edits; any remaining permission prompt is denied by the unattended callback. |
|
||||
| `auto` | Let Claude Code's native classifier allow or deny permission requests. |
|
||||
| `plan` | Run Claude Code in its native planning-only mode without tool execution. |
|
||||
| `plan` | Run in native planning mode, deny execution approval, and return the completed plan as the final answer. |
|
||||
| `bypassPermissions` | Explicitly set the SDK's dangerous confirmation and bypass permission checks. |
|
||||
|
||||
Production resolves `claude` from the subprocess execution world's credential-scrubbed `PATH`, with explicit `env` entries applied, and passes the resulting path to the SDK as `pathToClaudeCodeExecutable`. On Windows, a resolved `.cmd` or `.bat` path is carried as a quoted, per-spawn environment value that `cmd.exe /v:off` expands once, so valid path metacharacters remain data. The pinned SDK's fixed flags then occupy cmd's command tail and contain no cmd metacharacters; they are not ordinary Windows argv. Native settings and authentication remain authoritative. The plugin does not install another CLI, select a model, create a product home, log in, or probe an account. Credential-shaped ambient variables are removed before the explicit `env` overlay is applied, so an API key or token intended for the child must be supplied there. Non-credential endpoint variables such as `ANTHROPIC_BASE_URL`, along with ordinary ambient values such as `PATH` and `HOME`, remain inherited unless overridden.
|
||||
|
|
|
|||
|
|
@ -16,7 +16,7 @@ SDK 接收由文本块原样拼接成的任务。提供方会完整迭代 SDK
|
|||
|
||||
提供方故意省略 SDK 的 `settingSources` 选项。因此,官方 SDK 会相对于父会话 cwd 读取宿主机常规的用户、项目和本地 Claude 设置,包括原生账户状态与产品配置。提供方既不复制也不过滤这些文件,也不会创建或修改登录状态。Profile 选择的 `permissionMode` 是唯一的 query 级覆盖:Claude Code 仍拥有其设置与沙箱,而所选原生模式决定这个无人值守 query 如何处理权限检查。
|
||||
|
||||
每次 query 都设置 `persistSession: false` 并禁用 `AskUserQuestion`。除 bypass 模式外,`canUseTool` 会立即拒绝仍需人工审批的请求。MCP elicitation 会被拒绝,已知的拒绝回退对话会被取消,未声明的对话类型则使用 SDK 的无对话失败行为。这些决定都不会等待用户界面。若权限拒绝或无人值守回调参与了一次失败运行,提供方会生成可选的 `SubagentResult.diagnostic`,其中只包含产品、有效模式、请求类别、决定与固定的安全原因;共享结果边界会把完整文本限制在 4096 个 UTF-8 字节以内。成功运行与本地取消不会公开已捕获的失败说明。
|
||||
每次 query 都设置 `persistSession: false` 并禁用 `AskUserQuestion`。除 bypass 模式外,`canUseTool` 会立即拒绝仍需人工审批的请求。在 plan 模式下,`ExitPlanMode` 审批会被拒绝,同时用固定指令要求模型把完整计划作为最终答案返回且不得执行。MCP elicitation 会被拒绝,已知的拒绝回退对话会被取消,未声明的对话类型则使用 SDK 的无对话失败行为。这些决定都不会等待用户界面。若权限拒绝或无人值守回调参与了一次失败运行,提供方会生成可选的 `SubagentResult.diagnostic`,其中只包含产品、有效模式、请求类别、决定与固定的安全原因;共享结果边界会把完整文本限制在 4096 个 UTF-8 字节以内。成功运行与本地取消不会公开已捕获的失败说明。
|
||||
|
||||
## 能力与上下文
|
||||
|
||||
|
|
@ -35,7 +35,7 @@ SDK 接收由文本块原样拼接成的任务。提供方会完整迭代 SDK
|
|||
| `dontAsk` | 不弹出提示,直接拒绝尚未获授权的操作。 |
|
||||
| `acceptEdits` | 接受文件编辑;其余权限提示由无人值守回调拒绝。 |
|
||||
| `auto` | 由 Claude Code 原生分类器允许或拒绝权限请求。 |
|
||||
| `plan` | 使用 Claude Code 原生的仅规划模式,不执行工具。 |
|
||||
| `plan` | 使用原生规划模式,拒绝执行审批,并把完整计划作为最终答案返回。 |
|
||||
| `bypassPermissions` | 显式设置 SDK 的危险确认并跳过权限检查。 |
|
||||
|
||||
生产环境从子进程执行世界清除凭证后的 `PATH` 解析 `claude`,再应用显式 `env` 条目,并把所得路径作为 `pathToClaudeCodeExecutable` 交给 SDK。在 Windows 上,解析到的 `.cmd` 或 `.bat` 路径会作为带引号、仅供本次 spawn 使用的环境值交给 `cmd.exe /v:off` 展开一次,因此合法路径中的元字符仍只是数据。锁定版本的 SDK 随后把固定命令行选项放在 cmd 的命令尾部;这些选项不含 cmd 元字符,也并不是普通的 Windows argv。原生设置与身份验证继续是权威来源。本插件不安装另一份 CLI、不选择模型、不创建产品主目录、不执行登录,也不探测账户。具有凭证特征的环境变量会在显式 `env` 覆盖生效前被清除,因此供子进程使用的 API 密钥或 token 必须在该配置中显式提供。除非被覆盖,`ANTHROPIC_BASE_URL` 等非凭证端点变量以及 `PATH` 和 `HOME` 等普通环境变量仍会被继承。
|
||||
|
|
|
|||
|
|
@ -38,7 +38,12 @@ export interface Config {
|
|||
* credential-scrubbed parent environment.
|
||||
*/
|
||||
env?: Record<string, string>
|
||||
/** Native non-interactive permission mode fixed for this Provider instance. */
|
||||
/**
|
||||
* Native non-interactive mode fixed for this Provider instance. Defaults to
|
||||
* `dontAsk`; `acceptEdits` accepts edits, `auto` uses the native classifier,
|
||||
* `plan` returns a plan without approving execution, and
|
||||
* `bypassPermissions` explicitly skips permission checks.
|
||||
*/
|
||||
permissionMode?: ClaudeCodePermissionMode
|
||||
/** Grace in milliseconds for Claude Code process-tree termination. */
|
||||
disposeGraceMs?: number
|
||||
|
|
|
|||
|
|
@ -38,14 +38,6 @@ import {
|
|||
/** Default POSIX grace between subprocess termination tiers. */
|
||||
export const DEFAULT_DISPOSE_GRACE_MS = 3_000
|
||||
|
||||
/** Profile-selectable non-interactive Claude Code permission mode. */
|
||||
export type ClaudeCodePermissionMode =
|
||||
| 'dontAsk'
|
||||
| 'acceptEdits'
|
||||
| 'auto'
|
||||
| 'plan'
|
||||
| 'bypassPermissions'
|
||||
|
||||
/** Claude Code permission modes that cannot wait for a human response. */
|
||||
export const CLAUDE_CODE_PERMISSION_MODES = [
|
||||
'dontAsk',
|
||||
|
|
@ -53,16 +45,21 @@ export const CLAUDE_CODE_PERMISSION_MODES = [
|
|||
'auto',
|
||||
'plan',
|
||||
'bypassPermissions',
|
||||
] as const satisfies readonly ClaudeCodePermissionMode[]
|
||||
] as const satisfies readonly NonNullable<Options['permissionMode']>[]
|
||||
|
||||
/** Profile-selectable non-interactive Claude Code permission mode. */
|
||||
export type ClaudeCodePermissionMode = typeof CLAUDE_CODE_PERMISSION_MODES[number]
|
||||
|
||||
/** Safe default for unattended Claude Code runs. */
|
||||
export const DEFAULT_CLAUDE_CODE_PERMISSION_MODE: ClaudeCodePermissionMode = 'dontAsk'
|
||||
|
||||
const SUPPORTED_UNATTENDED_DIALOG_KINDS = ['refusal_fallback_prompt']
|
||||
const SUPPORTED_UNATTENDED_DIALOG_KINDS = [
|
||||
'refusal_fallback_prompt',
|
||||
] satisfies NonNullable<Options['supportedDialogKinds']>
|
||||
|
||||
function unattendedDiagnostic(
|
||||
mode: ClaudeCodePermissionMode,
|
||||
request: 'tool permission' | 'MCP elicitation' | 'user dialog',
|
||||
request: 'tool permission' | 'plan approval' | 'MCP elicitation' | 'user dialog',
|
||||
decision: 'denied' | 'declined' | 'cancelled',
|
||||
reason: string,
|
||||
): string {
|
||||
|
|
@ -231,7 +228,19 @@ export function claudeQueryOptions(
|
|||
...spec.permissionMode === 'bypassPermissions'
|
||||
? { allowDangerouslySkipPermissions: true }
|
||||
: {
|
||||
canUseTool: () => {
|
||||
canUseTool: (toolName) => {
|
||||
if (spec.permissionMode === 'plan' && toolName === 'ExitPlanMode') {
|
||||
captureDiagnostic(unattendedDiagnostic(
|
||||
spec.permissionMode,
|
||||
'plan approval',
|
||||
'denied',
|
||||
'the provider returns the plan without approving execution',
|
||||
))
|
||||
return Promise.resolve({
|
||||
behavior: 'deny' as const,
|
||||
message: 'Plan approval is unavailable in this unattended run. Return the completed plan in your final response without executing it.',
|
||||
})
|
||||
}
|
||||
captureDiagnostic(unattendedDiagnostic(
|
||||
spec.permissionMode,
|
||||
'tool permission',
|
||||
|
|
@ -344,6 +353,7 @@ export async function startClaudeCodeRun(
|
|||
)
|
||||
}
|
||||
}
|
||||
// oxlint-disable-next-line typescript/no-unnecessary-condition -- the request can abort while process cleanup is awaited.
|
||||
if (cancelledBeforeCleanup || request.signal.aborted) {
|
||||
throw new Error('subagent-claude-code: request was aborted before SDK startup')
|
||||
}
|
||||
|
|
|
|||
|
|
@ -361,6 +361,23 @@ describe('real Claude Agent SDK 0.3.220 and its distributed Claude Code 2.1.220
|
|||
await expectQuiescent(harness.handles)
|
||||
})
|
||||
|
||||
it('returns the completed plan without approving execution', async () => {
|
||||
const { harness, fixture } = await realHarness({
|
||||
kind: 'tool-use',
|
||||
toolName: 'ExitPlanMode',
|
||||
input: {},
|
||||
finalText: 'PLAN_ONLY_RESULT',
|
||||
}, 'plan')
|
||||
const run = await startRequest(harness, 'Design the fixture change without implementing it.')
|
||||
await expect(run.result).resolves.toEqual({
|
||||
output: [{ type: 'text', text: 'PLAN_ONLY_RESULT' }],
|
||||
stopReason: 'completed',
|
||||
})
|
||||
expect(fixture.requests).toHaveLength(2)
|
||||
await run.dispose()
|
||||
await expectQuiescent(harness.handles)
|
||||
})
|
||||
|
||||
it('settles cancellation and leaves the real SDK-spawned CLI tree quiescent', async () => {
|
||||
const { harness, fixture } = await realHarness({ kind: 'hold' })
|
||||
const controller = new AbortController()
|
||||
|
|
|
|||
|
|
@ -689,6 +689,34 @@ describe('query options and result mapping', () => {
|
|||
},
|
||||
)
|
||||
|
||||
it('returns a plan without approving ExitPlanMode execution', async () => {
|
||||
const child = fakeChild()
|
||||
const diagnostics: string[] = []
|
||||
const options = claudeQueryOptions({
|
||||
cwd: '/workspace',
|
||||
executable: '/native/claude',
|
||||
permissionMode: 'plan',
|
||||
env: {},
|
||||
disposeGraceMs: 17,
|
||||
spawn: () => child.handle,
|
||||
}, new AbortController(), () => {}, value => diagnostics.push(value))
|
||||
await expect(options.canUseTool!(
|
||||
'ExitPlanMode',
|
||||
{},
|
||||
{
|
||||
signal: new AbortController().signal,
|
||||
toolUseID: 'exit-plan',
|
||||
requestId: 'exit-plan-request',
|
||||
},
|
||||
)).resolves.toEqual({
|
||||
behavior: 'deny',
|
||||
message: 'Plan approval is unavailable in this unattended run. Return the completed plan in your final response without executing it.',
|
||||
})
|
||||
expect(diagnostics).toEqual([
|
||||
'Claude Code unattended decision (mode: plan; request: plan approval; decision: denied): the provider returns the plan without approving execution',
|
||||
])
|
||||
})
|
||||
|
||||
it('accepts only a non-error success with a non-blank final result', () => {
|
||||
expect(successfulResult(success('exact final'))).toBe('exact final')
|
||||
expect(() => successfulResult(success('answer', true)))
|
||||
|
|
|
|||
|
|
@ -17,7 +17,7 @@ import type { ContentBlock } from '@deepseek-ai/dsh-llm'
|
|||
import type { SubagentCapabilities, SubagentResult, SubagentRun, SubagentStopReason } from './types.ts'
|
||||
|
||||
/** Maximum UTF-8 size of {@link SubagentResult.diagnostic}. */
|
||||
export const MAX_SUBAGENT_DIAGNOSTIC_BYTES = 4_096
|
||||
const MAX_SUBAGENT_DIAGNOSTIC_BYTES = 4_096
|
||||
|
||||
const DIAGNOSTIC_TRUNCATION_SUFFIX = '\n[diagnostic truncated]'
|
||||
const utf8Encoder = new TextEncoder()
|
||||
|
|
@ -28,7 +28,7 @@ const utf8Decoder = new TextDecoder()
|
|||
* @param diagnostic - safe diagnostic text produced by the provider.
|
||||
* @returns the original text, or a visibly truncated value within the limit.
|
||||
*/
|
||||
export function limitSubagentDiagnostic(diagnostic: string): string {
|
||||
function limitSubagentDiagnostic(diagnostic: string): string {
|
||||
const bytes = utf8Encoder.encode(diagnostic)
|
||||
if (bytes.byteLength <= MAX_SUBAGENT_DIAGNOSTIC_BYTES) return diagnostic
|
||||
|
||||
|
|
@ -195,15 +195,10 @@ export async function settleRunResult(parts: RunResultSettlement): Promise<Subag
|
|||
} catch {
|
||||
// The diagnostic sink cannot reject the run result.
|
||||
}
|
||||
let diagnostic: string | undefined
|
||||
try {
|
||||
const collected = parts.collectDiagnostic?.()
|
||||
diagnostic = collected === undefined
|
||||
? undefined
|
||||
: limitSubagentDiagnostic(collected)
|
||||
} catch {
|
||||
// A provider diagnostic collector cannot reject the terminal result.
|
||||
}
|
||||
const collected = parts.collectDiagnostic?.()
|
||||
const diagnostic = collected === undefined
|
||||
? undefined
|
||||
: limitSubagentDiagnostic(collected)
|
||||
return {
|
||||
output: parts.collectOutput(),
|
||||
...diagnostic === undefined ? {} : { diagnostic },
|
||||
|
|
|
|||
|
|
@ -1,12 +1,12 @@
|
|||
import { describe, expect, it } from 'vitest'
|
||||
import { SessionId } from '@deepseek-ai/dsh-session'
|
||||
import {
|
||||
limitSubagentDiagnostic,
|
||||
MAX_SUBAGENT_DIAGNOSTIC_BYTES,
|
||||
settleRun,
|
||||
settleRunResult,
|
||||
} from '../src/index.ts'
|
||||
|
||||
const MAX_SUBAGENT_DIAGNOSTIC_BYTES = 4_096
|
||||
|
||||
describe('outcome mapping helpers', () => {
|
||||
it.each([
|
||||
['completed', { status: 'completed', output: 'partial' }],
|
||||
|
|
@ -86,16 +86,18 @@ describe('outcome mapping helpers', () => {
|
|||
|
||||
it('bounds multibyte diagnostics and marks truncation', async () => {
|
||||
const exact = 'x'.repeat(MAX_SUBAGENT_DIAGNOSTIC_BYTES)
|
||||
expect(limitSubagentDiagnostic(exact)).toBe(exact)
|
||||
|
||||
const oversized = '权限'.repeat(MAX_SUBAGENT_DIAGNOSTIC_BYTES)
|
||||
const limited = limitSubagentDiagnostic(oversized)
|
||||
expect(Buffer.byteLength(limited, 'utf8'))
|
||||
.toBeLessThanOrEqual(MAX_SUBAGENT_DIAGNOSTIC_BYTES)
|
||||
expect(limited.endsWith('[diagnostic truncated]')).toBe(true)
|
||||
expect(limited).not.toContain('\uFFFD')
|
||||
|
||||
const controller = new AbortController()
|
||||
const exactResult = await settleRunResult({
|
||||
attempt: async () => { throw new Error('provider failed') },
|
||||
collectOutput: () => [],
|
||||
collectDiagnostic: () => exact,
|
||||
cancelled: () => false,
|
||||
signal: controller.signal,
|
||||
onAbort: () => {},
|
||||
})
|
||||
expect(exactResult.diagnostic).toBe(exact)
|
||||
|
||||
const result = await settleRunResult({
|
||||
attempt: async () => { throw new Error('provider failed') },
|
||||
collectOutput: () => [],
|
||||
|
|
@ -104,19 +106,12 @@ describe('outcome mapping helpers', () => {
|
|||
signal: controller.signal,
|
||||
onAbort: () => {},
|
||||
})
|
||||
const limited = result.diagnostic ?? ''
|
||||
expect(Buffer.byteLength(limited, 'utf8'))
|
||||
.toBeLessThanOrEqual(MAX_SUBAGENT_DIAGNOSTIC_BYTES)
|
||||
expect(limited.endsWith('[diagnostic truncated]')).toBe(true)
|
||||
expect(limited).not.toContain('\uFFFD')
|
||||
expect(result.stopReason).toBe('error')
|
||||
expect(result.diagnostic).toBe(limited)
|
||||
|
||||
await expect(settleRunResult({
|
||||
attempt: async () => { throw new Error('provider failed') },
|
||||
collectOutput: () => [{ type: 'text', text: 'partial' }],
|
||||
collectDiagnostic: () => { throw new Error('collector failed') },
|
||||
cancelled: () => false,
|
||||
signal: controller.signal,
|
||||
onAbort: () => {},
|
||||
})).resolves.toEqual({
|
||||
output: [{ type: 'text', text: 'partial' }],
|
||||
stopReason: 'error',
|
||||
})
|
||||
})
|
||||
})
|
||||
|
|
|
|||
|
|
@ -27,6 +27,8 @@ export interface Config {
|
|||
reply?: string
|
||||
/** Terminal result reason. */
|
||||
stopReason?: SubagentStopReason
|
||||
/** Safe non-assistant detail for a non-completed result. */
|
||||
diagnostic?: string
|
||||
/** Start-time features advertised by the provider. */
|
||||
capabilities?: Partial<SubagentCapabilities>
|
||||
/** Whether tool descriptions say the child inherits completed turns. */
|
||||
|
|
@ -65,11 +67,17 @@ class ScriptedSubagentProvider implements SubagentProvider {
|
|||
throw new Error('scripted subagent start aborted before publication')
|
||||
}
|
||||
|
||||
const resultFor = (): SubagentResult => ({
|
||||
output,
|
||||
...wantsStructured ? { structured: this.config.structured ?? { reply } } : {},
|
||||
stopReason: state.cancelled ? 'aborted' : stopReason,
|
||||
})
|
||||
const resultFor = (): SubagentResult => {
|
||||
const terminal = state.cancelled ? 'aborted' : stopReason
|
||||
return {
|
||||
output,
|
||||
...wantsStructured ? { structured: this.config.structured ?? { reply } } : {},
|
||||
...this.config.diagnostic !== undefined && terminal !== 'completed'
|
||||
? { diagnostic: this.config.diagnostic }
|
||||
: {},
|
||||
stopReason: terminal,
|
||||
}
|
||||
}
|
||||
const gate = Promise.resolve(this.config.onStart?.(request))
|
||||
const result = gate.then(() => new Promise<SubagentResult>((resolve) => {
|
||||
setTimeout(() => { resolve(resultFor()) }, 0)
|
||||
|
|
|
|||
|
|
@ -186,26 +186,11 @@ describe('dsh-tool-subagent', () => {
|
|||
})
|
||||
|
||||
it('renders provider diagnostics before preserved partial assistant output', async () => {
|
||||
const ctx = new Context()
|
||||
await ctx.plugin(SystemPrompt)
|
||||
await ctx.plugin(ToolRuntime)
|
||||
await ctx.plugin(SubagentRuntime)
|
||||
ctx.subagents.registerProvider({
|
||||
name: 'diagnostic',
|
||||
capabilities: { outputSchema: false, depthLimit: false, toolFilter: false, persona: false },
|
||||
inheritsParentContext: false,
|
||||
start: async () => ({
|
||||
id: SessionId('diagnostic-child'),
|
||||
localAgent: undefined,
|
||||
result: Promise.resolve({
|
||||
output: [{ type: 'text', text: 'partial assistant text' }],
|
||||
diagnostic: 'Claude Code denied a tool request',
|
||||
stopReason: 'error',
|
||||
}),
|
||||
dispose: async () => {},
|
||||
}),
|
||||
const ctx = await setup({ provider: 'mock' }, {
|
||||
reply: 'partial assistant text',
|
||||
diagnostic: 'Claude Code denied a tool request',
|
||||
stopReason: 'error',
|
||||
})
|
||||
await ctx.plugin(tool, { provider: 'diagnostic', maxDepth: 'provider-managed' })
|
||||
|
||||
const result = await callSubagent(ctx, { description: 'd', prompt: 'p' })
|
||||
expect(result.isError).toBe(true)
|
||||
|
|
@ -884,34 +869,17 @@ describe('dsh-tool-subagent background mode', () => {
|
|||
})
|
||||
|
||||
it('preserves provider diagnostics in one-shot background failure detail', async () => {
|
||||
const ctx = await backgroundSetup({ provider: 'mock' })
|
||||
const ctx = await backgroundSetup({ provider: 'mock' }, {
|
||||
reply: 'not background output',
|
||||
diagnostic: 'Claude Code cancelled an unattended dialog',
|
||||
stopReason: 'error',
|
||||
})
|
||||
const parent = ownerAgent(ctx, 'sess-parent')
|
||||
ctx.subagents.registerProvider({
|
||||
name: 'diagnostic-background',
|
||||
capabilities: { outputSchema: false, depthLimit: false, toolFilter: false, persona: false },
|
||||
inheritsParentContext: false,
|
||||
start: async () => ({
|
||||
id: SessionId('diagnostic-background-child'),
|
||||
localAgent: undefined,
|
||||
result: Promise.resolve({
|
||||
output: [{ type: 'text', text: 'not background output' }],
|
||||
diagnostic: 'Claude Code cancelled an unattended dialog',
|
||||
stopReason: 'error',
|
||||
}),
|
||||
dispose: async () => {},
|
||||
}),
|
||||
})
|
||||
tool.apply(ctx, {
|
||||
provider: 'diagnostic-background',
|
||||
toolName: 'subagent_diagnostic_background',
|
||||
backgroundMode: 'one-shot',
|
||||
maxDepth: 'provider-managed',
|
||||
})
|
||||
|
||||
const started = await ctx.tools.execute({
|
||||
signal: testToolSignal,
|
||||
callId: CallId('diagnostic-background-start'),
|
||||
name: 'subagent_diagnostic_background',
|
||||
name: 'subagent',
|
||||
arguments: { description: 'd', prompt: 'p', run_in_background: true },
|
||||
agent: parent,
|
||||
})
|
||||
|
|
|
|||
Loading…
Add table
Reference in a new issue