deepseek-harness/packages/feedback/command-feedback/tests/command-feedback.spec.ts

247 lines
10 KiB
TypeScript
Raw Normal View History

import { beforeEach, describe, expect, it, vi } from 'vitest'
build(vendor): rescope the vendored Cordis packages into @deepseek-ai Machine-produced by `pnpm run rescope-vendor --apply` plus the regeneration it prints: `pnpm install` for the lockfile, `pnpm run gen-third-party-notices`, `verify-translation-pairing --write` for the touched bilingual pairs, `gen-doc-graphs`, and one typert snapshot whose ids embed character offsets. `pnpm run rescope-vendor --check` verifies the result. Renames nine vendored packages (cordis, cosmokit, schemastery and the six @cordisjs plugins) and every reference that resolves them: manifest names and dependency keys, module specifiers including declare-module merges, cordis.yml plugin names, tsconfig paths, every Markdown fence, and `docs/` prose. Directory names, upstream versions, and dependency ranges are unchanged, so vendor/README.md still reads as an upstream snapshot; its manifest table gains an upstream-name column so THIRD_PARTY_NOTICES keeps MIT attribution pointed at each fork's origin. The tutorial tier follows the rename end to end: its yaml fences named plugins the Loader can no longer resolve, its `ts ignore-check` fences disagreed with the compiled fences beside them, and its prose quoted both. The contracts that told readers to keep upstream names — the root convention and the vendoring cookbook's tree comment and manifest invariant — now say to rescope instead. Two rules read `@deepseek-ai/` as "another workspace plugin": the client bundle purity gate now names the vendored libraries a browser bundle inlines, and the files where a bare `cordis` is an agent-preset id keep that product data.
2026-08-10 22:04:06 +08:00
import { Context } from '@deepseek-ai/cordis'
import Loader from '@deepseek-ai/cordis-plugin-loader'
import AgentRegistry, { Inbox } from '@deepseek-ai/dsh-agent'
import type { Agent, AgentStatus } from '@deepseek-ai/dsh-agent'
import CommandRuntime from '@deepseek-ai/dsh-commands'
import SessionStore, { foldSurface, Session, SessionId } from '@deepseek-ai/dsh-session'
import { SessionTelemetryBackend, type SessionTelemetrySharingStatus } from '@deepseek-ai/dsh-session-telemetry'
import * as commandFeedback from '@deepseek-ai/dsh-command-feedback'
const { USER_ID, getOrCreateAnonymousUserId } = vi.hoisted(() => {
const USER_ID = '01234567-89ab-4cde-8f01-23456789abcd'
return { USER_ID, getOrCreateAnonymousUserId: vi.fn(() => USER_ID) }
})
vi.mock('@deepseek-ai/dsh-anonymous-user-id', () => ({
getOrCreateAnonymousUserId,
}))
beforeEach(() => getOrCreateAnonymousUserId.mockClear())
interface Harness {
readonly ctx: Context
readonly agent: Agent
readonly session: Session
readonly plugin: Awaited<ReturnType<Context['plugin']>>
}
/** Minimal mounted backend disclosing one sharing policy. */
class FakeTelemetry extends SessionTelemetryBackend {
override readonly sharing: SessionTelemetrySharingStatus
constructor(ctx: Context, config: { sharing: SessionTelemetrySharingStatus }) {
super(ctx)
this.sharing = config.sharing
}
emit(): void {}
async shutdown(): Promise<void> {}
}
/** Build a live idle agent over a store-owned session, as an app's spine does. */
function stubAgent(ctx: Context, id: string): { agent: Agent; session: Session } {
const session = ctx.sessions.create(SessionId(id))
const inbox = new Inbox(session, { inserted: () => {}, discarded: () => {}, claimed: () => {} })
let status: AgentStatus = 'idle'
const agent: Agent = {
id: session.id,
options: {},
session,
inbox,
ctx: new Context(),
get status() { return status },
send: () => {},
followup: () => {},
steer: () => {},
inject: () => {},
cancel() { status = 'idle' },
runMaintenance: task => task(new AbortController().signal),
whenIdle() { return Promise.resolve() },
}
return { agent, session }
}
/**
* Mount the real command registry, this producer, and optionally a telemetry
* backend disclosing one sharing policy. Without `sharing`, no telemetry
* service exists and the acknowledgement reports "not configured".
*/
async function harness(sharing?: SessionTelemetrySharingStatus): Promise<Harness> {
const ctx = new Context()
await ctx.plugin(CommandRuntime)
await ctx.plugin(AgentRegistry)
await ctx.plugin(SessionStore)
if (sharing !== undefined) await ctx.plugin(FakeTelemetry, { sharing })
const plugin = await ctx.plugin(commandFeedback)
const { agent, session } = stubAgent(ctx, `command-feedback-${Math.random()}`)
ctx.agents.register(agent)
return { ctx, agent, session, plugin }
}
/** Execute `/feedback` through the same registry boundary as a UI adapter. */
async function run(test: Harness, suffix = ''): Promise<{ kind: string; text?: string }> {
const settled = await test.ctx.commands.execute(
test.agent,
`/feedback${suffix}`,
feat(commands): route composer image attachments through slash commands A claimed slash command consumed only the text half of the composer submission: /goal with reference images executed, cleared the draft, and silently stranded the images in the rail. Model-visible attachment intent had no route through the command plane. The submission envelope is now modeled end to end. CommandDefinition input.images declares acceptance; the declaration rides the descriptor to every client, onto the minted CommandClaim, and into the input machine's claim snapshot. commands.execute carries the submission's base64 images and enforces the declaration in the executor: non-declaring commands, a missing attachment store, and exceeded batch limits settle as logged error results before the handler runs. Admission reuses the attachment package's new admitEncodedImages, extracted from api-proxy's prompt path so both wire endpoints share one limits/validation/commit sequence. Producers own model visibility: /goal submits one user followup (image blocks + a fixed reference line) after a successful create/edit so goal rounds read the images from session history; /plan folds them into its steered message. Grammar misfits (/goal pause, bare /plan, /plan off) return direct errors and the composer keeps the images. On the client, enter adjudication carries a SubmitEnvelope and every command route that cannot consume images throws a localized refusal that renders as one composer notice with draft and images retained; the claimed pre-gate applies the same copy. An accepting claim serializes the draft images, forwards them to commands.execute, and clears plus releases them only on a success outcome. The assembled web test roster gains the ui-input-trigger and ui-commands plugins, mirroring the shipped composition, so slash submissions exercise the command plane; a new keyless snapshot pins the refusal banner and the accepting /goal flow over the built client graph.
2026-08-17 18:57:55 +08:00
[],
new AbortController().signal,
)
if (settled === undefined) throw new Error('feedback command was not registered')
return settled.result
}
/** Authoritative feedback payloads in log order. */
function feedbackTexts(session: Session): string[] {
return session.events
.filter(event => event.type === 'feedback/record')
.map(event => event.data.text)
}
describe('@deepseek-ai/dsh-command-feedback registration', () => {
it('registers one global command with Loader-safe exports and disposes it', async () => {
const test = await harness()
expect(commandFeedback.name).toBe('command-feedback')
expect(commandFeedback.inject).toEqual(['commands'])
expect('default' in commandFeedback).toBe(false)
const loader = Object.create(Loader.prototype) as Loader
expect(loader.unwrapExports(commandFeedback)).toBe(commandFeedback)
expect(test.ctx.commands.list(test.agent)).toContainEqual({
name: 'feedback',
description: 'record feedback about this session',
input: { hint: '<text>' },
})
expect(test.ctx.commands.find(test.agent, 'feedback')).toMatchObject({ recordInput: false })
await test.plugin.dispose()
expect(test.ctx.commands.find(test.agent, 'feedback')).toBeUndefined()
})
})
describe('/feedback human command', () => {
it('acknowledges feedback and records its payload exactly once in the domain event', async () => {
const test = await harness()
await expect(run(test, ' the diff view is unreadable')).resolves.toEqual({
kind: 'success',
text: `Feedback recorded for session ${test.session.id}\nAnonymous user: ${USER_ID}. Session sharing is not configured.`,
})
expect(feedbackTexts(test.session)).toEqual(['the diff view is unreadable'])
const commandRun = test.session.events.find(event => event.type === 'command/run')
expect(commandRun?.type === 'command/run' && Object.hasOwn(commandRun.data, 'args')).toBe(false)
expect(JSON.stringify(test.session.events).match(/the diff view is unreadable/gu)).toHaveLength(1)
})
it('exports a command-independent feedback producer', async () => {
const test = await harness()
commandFeedback.recordFeedback(test.session, ' recorded outside a command ')
expect(test.session.events.map(event => event.type)).toEqual(['feedback/record'])
expect(feedbackTexts(test.session)).toEqual(['recorded outside a command'])
expect(() => { commandFeedback.recordFeedback(test.session, ' \n\t ') })
.toThrow('feedback text must not be empty')
expect(feedbackTexts(test.session)).toEqual(['recorded outside a command'])
})
it('keeps command bookkeeping around the authoritative feedback event', async () => {
const test = await harness()
await run(test, ' nothing else happens')
expect(test.session.events.map(event => event.type)).toEqual([
'command/run', 'feedback/record', 'command/done',
])
})
it('normalizes surrounding whitespace without parsing command-like content', async () => {
const test = await harness()
await run(test, ' /plan felt SLOW\n\ttwice today ')
expect(feedbackTexts(test.session)).toEqual(['/plan felt SLOW\n\ttwice today'])
})
it('records each entry separately without replacing earlier ones', async () => {
const test = await harness()
await run(test, ' first')
await run(test, ' second')
expect(feedbackTexts(test.session)).toEqual(['first', 'second'])
})
it('records concurrent submissions in dispatch order', async () => {
const test = await harness()
const signal = new AbortController().signal
// Command adapters may dispatch concurrent requests without awaiting one another.
const settled = await Promise.all([
feat(commands): route composer image attachments through slash commands A claimed slash command consumed only the text half of the composer submission: /goal with reference images executed, cleared the draft, and silently stranded the images in the rail. Model-visible attachment intent had no route through the command plane. The submission envelope is now modeled end to end. CommandDefinition input.images declares acceptance; the declaration rides the descriptor to every client, onto the minted CommandClaim, and into the input machine's claim snapshot. commands.execute carries the submission's base64 images and enforces the declaration in the executor: non-declaring commands, a missing attachment store, and exceeded batch limits settle as logged error results before the handler runs. Admission reuses the attachment package's new admitEncodedImages, extracted from api-proxy's prompt path so both wire endpoints share one limits/validation/commit sequence. Producers own model visibility: /goal submits one user followup (image blocks + a fixed reference line) after a successful create/edit so goal rounds read the images from session history; /plan folds them into its steered message. Grammar misfits (/goal pause, bare /plan, /plan off) return direct errors and the composer keeps the images. On the client, enter adjudication carries a SubmitEnvelope and every command route that cannot consume images throws a localized refusal that renders as one composer notice with draft and images retained; the claimed pre-gate applies the same copy. An accepting claim serializes the draft images, forwards them to commands.execute, and clears plus releases them only on a success outcome. The assembled web test roster gains the ui-input-trigger and ui-commands plugins, mirroring the shipped composition, so slash submissions exercise the command plane; a new keyless snapshot pins the refusal banner and the accepting /goal flow over the built client graph.
2026-08-17 18:57:55 +08:00
test.ctx.commands.execute(test.agent, '/feedback first', [], signal),
test.ctx.commands.execute(test.agent, '/feedback second', [], signal),
])
expect(settled.map(item => item?.result)).toEqual([
{ kind: 'success', text: `Feedback recorded for session ${test.session.id}\nAnonymous user: ${USER_ID}. Session sharing is not configured.` },
{ kind: 'success', text: `Feedback recorded for session ${test.session.id}\nAnonymous user: ${USER_ID}. Session sharing is not configured.` },
])
expect(feedbackTexts(test.session)).toEqual(['first', 'second'])
})
it('discloses full session sharing in the acknowledgement', async () => {
const test = await harness('full')
await expect(run(test, ' everything shared')).resolves.toEqual({
kind: 'success',
text: `Feedback recorded for session ${test.session.id}\nAnonymous user: ${USER_ID}. Session sharing is enabled.`,
})
expect(feedbackTexts(test.session)).toEqual(['everything shared'])
})
it('discloses feedback-gated session sharing in the acknowledgement', async () => {
const test = await harness('feedback-only')
await expect(run(test, ' gated sharing')).resolves.toEqual({
kind: 'success',
text: `Feedback recorded for session ${test.session.id}\nAnonymous user: ${USER_ID}. Session sharing is feedback-gated; recording feedback releases the session prefix for sharing.`,
})
expect(feedbackTexts(test.session)).toEqual(['gated sharing'])
})
it('discloses disabled session sharing in the acknowledgement', async () => {
const test = await harness('disabled')
await expect(run(test, ' local only')).resolves.toEqual({
kind: 'success',
text: `Feedback recorded for session ${test.session.id}\nAnonymous user: ${USER_ID}. Session sharing is disabled.`,
})
expect(feedbackTexts(test.session)).toEqual(['local only'])
})
it('keeps every recorded event out of model context and derived history', async () => {
const test = await harness()
await run(test, ' invisible to the model')
for (const event of test.session.events) {
expect('surfaceOp' in event).toBe(false)
expect(test.session.deriveEventMessage(event)).toBeNull()
}
expect(foldSurface(test.session.events).nodes).toEqual([])
expect(test.session.surface.nodes).toEqual([])
expect(test.session.deriveMessages()).toEqual([])
})
it('rejects empty and whitespace-only input as a failed command record', async () => {
const test = await harness()
const expected = {
kind: 'error',
text: 'Feedback text is required. Usage: /feedback <text>',
}
await expect(run(test)).resolves.toEqual(expected)
await expect(run(test, ' \n\t ')).resolves.toEqual(expected)
expect(getOrCreateAnonymousUserId).not.toHaveBeenCalled()
expect(feedbackTexts(test.session)).toEqual([])
const done = test.session.events.filter(event => event.type === 'command/done')
expect(done.map(event => event.data.kind)).toEqual(['error', 'error'])
for (const event of test.session.events) {
if (event.type === 'command/run') expect(Object.hasOwn(event.data, 'args')).toBe(false)
}
})
it('records nothing when dispatch rejects an already-cancelled request', async () => {
const test = await harness()
const controller = new AbortController()
controller.abort(new Error('user cancelled the command'))
feat(commands): route composer image attachments through slash commands A claimed slash command consumed only the text half of the composer submission: /goal with reference images executed, cleared the draft, and silently stranded the images in the rail. Model-visible attachment intent had no route through the command plane. The submission envelope is now modeled end to end. CommandDefinition input.images declares acceptance; the declaration rides the descriptor to every client, onto the minted CommandClaim, and into the input machine's claim snapshot. commands.execute carries the submission's base64 images and enforces the declaration in the executor: non-declaring commands, a missing attachment store, and exceeded batch limits settle as logged error results before the handler runs. Admission reuses the attachment package's new admitEncodedImages, extracted from api-proxy's prompt path so both wire endpoints share one limits/validation/commit sequence. Producers own model visibility: /goal submits one user followup (image blocks + a fixed reference line) after a successful create/edit so goal rounds read the images from session history; /plan folds them into its steered message. Grammar misfits (/goal pause, bare /plan, /plan off) return direct errors and the composer keeps the images. On the client, enter adjudication carries a SubmitEnvelope and every command route that cannot consume images throws a localized refusal that renders as one composer notice with draft and images retained; the claimed pre-gate applies the same copy. An accepting claim serializes the draft images, forwards them to commands.execute, and clears plus releases them only on a success outcome. The assembled web test roster gains the ui-input-trigger and ui-commands plugins, mirroring the shipped composition, so slash submissions exercise the command plane; a new keyless snapshot pins the refusal banner and the accepting /goal flow over the built client graph.
2026-08-17 18:57:55 +08:00
await expect(test.ctx.commands.execute(test.agent, '/feedback too late', [], controller.signal))
.rejects.toThrow('user cancelled the command')
expect(test.session.events).toEqual([])
})
})