diff --git a/packages/diagram-client/src/chat-panel-integrated.ts b/packages/diagram-client/src/chat-panel-integrated.ts index 83d395c..dbf8e16 100644 --- a/packages/diagram-client/src/chat-panel-integrated.ts +++ b/packages/diagram-client/src/chat-panel-integrated.ts @@ -193,6 +193,8 @@ export class ChatPanel implements IDiagramStartup, ISelectionListener { private isVisible = false; private isCompact = false; private currentMode: 'plan' | 'build' = 'build'; + /** A task the host asked for (`chat.startTask`), sent once its session exists. */ + private pendingTask: { name: string; mode: 'plan' | 'build'; prompt: string } | undefined; private currentSessionId: string | null = null; private timeline: TimelineItem[] = []; private sessions: SessionEntry[] = []; @@ -494,11 +496,40 @@ export class ChatPanel implements IDiagramStartup, ISelectionListener { this.pushMessage('system', `Created session: ${data.session.name ?? data.session.id}`); } this.update(); + // A task's session: send its message, as if typed, so it is in the + // transcript like any other. + if (data?.session?.id && this.pendingTask) { + const task = this.pendingTask; + this.pendingTask = undefined; + this.inputValue = task.prompt; + this.sendMessage(); + } + break; + + case 'chat.startTask': + // The host asks for a task in a new session: "Fix with AI" on a run + // that failed. Open the panel, start the session in the task's mode, + // and send its message once the session exists. + if (data && typeof data.prompt === 'string' && data.prompt.trim() !== '') { + const mode: 'plan' | 'build' = data.mode === 'build' ? 'build' : 'plan'; + this.pendingTask = { + name: typeof data.name === 'string' && data.name.trim() !== '' ? data.name : 'Task', + mode, + prompt: data.prompt + }; + this.currentMode = mode; + this.isLoadingSession = true; + this.loadingLabel = 'Creating session…'; + this.autoShow('task'); + this.update(); + this.sendToHost('chat.createSession', { mode, name: this.pendingTask.name }); + } break; case 'chat.sessionCreateAborted': // Name prompt cancelled — drop the "Creating session…" spinner. this.isLoadingSession = false; + this.pendingTask = undefined; this.update(); break; diff --git a/packages/diagram-client/test/chat-panel-start-task.test.ts b/packages/diagram-client/test/chat-panel-start-task.test.ts new file mode 100644 index 0000000..f3a5004 --- /dev/null +++ b/packages/diagram-client/test/chat-panel-start-task.test.ts @@ -0,0 +1,61 @@ +/** + * A task the host asks the chat to start ("Fix with AI" on a failed run): the + * panel opens, creates a session in the task's mode under the task's name, and + * once it exists sends the task's message as if typed. + */ +import 'reflect-metadata'; +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest'; +import { ChatPanel } from '../src/chat-panel-integrated'; + +function makePanel() { + const panel = new ChatPanel(); + const sent: Array<{ type: string; data: any }> = []; + (panel as any).channel = { sendToHost: (_method: string, env: any) => void sent.push(env) }; + const showSpy = vi.fn(() => { (panel as any).isVisible = true; }); + (panel as any).show = showSpy; + return { panel, sent, showSpy }; +} + +beforeEach(() => { + (globalThis as any).requestAnimationFrame = () => 1; + (globalThis as any).cancelAnimationFrame = () => undefined; +}); +afterEach(() => { + delete (globalThis as any).requestAnimationFrame; + delete (globalThis as any).cancelAnimationFrame; +}); + +const task = { name: 'Fix: x (in m › i)', mode: 'plan', prompt: 'The last run of workflow `top` failed. …' }; + +describe('a task the host starts in the chat', () => { + it('opens the panel and creates a session in the task’s mode and name', () => { + const { panel, sent, showSpy } = makePanel(); + + (panel as any).handleIncomingMessage('chat.startTask', task); + + expect(showSpy).toHaveBeenCalled(); + expect(sent.find(m => m.type === 'chat.createSession')?.data).toMatchObject({ mode: 'plan', name: task.name }); + expect(sent.some(m => m.type === 'chat.sendMessage')).toBe(false); + }); + + it('sends the task’s message once the session exists, in plan mode, once', () => { + const { panel, sent } = makePanel(); + (panel as any).handleIncomingMessage('chat.startTask', task); + + (panel as any).handleIncomingMessage('chat.sessionCreated', { session: { id: 's1', name: task.name } }); + (panel as any).handleIncomingMessage('chat.sessionCreated', { session: { id: 's2', name: 'later' } }); + + const messages = sent.filter(m => m.type === 'chat.sendMessage').map(m => m.data); + expect(messages).toHaveLength(1); + expect(messages[0]).toMatchObject({ text: task.prompt, sessionId: 's1', mode: 'plan' }); + }); + + it('drops the task when the session is not created', () => { + const { panel, sent } = makePanel(); + (panel as any).handleIncomingMessage('chat.startTask', task); + (panel as any).handleIncomingMessage('chat.sessionCreateAborted', {}); + (panel as any).handleIncomingMessage('chat.sessionCreated', { session: { id: 's1' } }); + + expect(sent.some(m => m.type === 'chat.sendMessage')).toBe(false); + }); +}); diff --git a/packages/extension-core/src/api.ts b/packages/extension-core/src/api.ts index 5b8020c..24cc865 100644 --- a/packages/extension-core/src/api.ts +++ b/packages/extension-core/src/api.ts @@ -227,6 +227,17 @@ export interface DiagramRunHost { ): void; output: vscode.OutputChannel; useLiveOverlaySignatureSource(source: DiagramLiveOverlaySource): void; + /** Starts a task in the chat on the diagram at `sourceUri`: a new session in + * `task.mode`, named `task.name`, whose first message is `task.prompt`. + * `false` when no chat can take it there. */ + startChatTask?(task: DiagramChatTask, sourceUri: string): Promise; +} + +/** A task the platform asks a diagram's chat to start (see `DiagramRunHost.startChatTask`). */ +export interface DiagramChatTask { + name: string; + mode: "plan" | "build"; + prompt: string; } /** diff --git a/packages/extension-core/src/extension/chat/chat-runtime.ts b/packages/extension-core/src/extension/chat/chat-runtime.ts index 426f171..76e8c10 100644 --- a/packages/extension-core/src/extension/chat/chat-runtime.ts +++ b/packages/extension-core/src/extension/chat/chat-runtime.ts @@ -881,6 +881,21 @@ export class ChatRuntime { }); } + /** + * Start a task in the chat panel open on `uri`: a new session named + * `task.name`, in `task.mode`, whose first message is `task.prompt` -- sent by + * the panel as if typed, so the person sees what was asked. `false` when no + * panel can be reached there. + */ + startTask(uri: string, task: { name: string; mode: "plan" | "build"; prompt: string }): boolean { + if (this.canReach && !this.canReach(uri)) { + this.logLine(`chat task "${task.name}": no chat panel on ${uri}`); + return false; + } + this.postToWebview(uri, { type: "chat.startTask", data: { ...task } }); + return true; + } + /** The last model the user explicitly chose (migrated from the legacy key). */ private getPreferredModel(): string | undefined { return readStateWithFallback( diff --git a/packages/extension-core/src/extension/diagram/glsp-activation.ts b/packages/extension-core/src/extension/diagram/glsp-activation.ts index f8db054..67ee7b1 100644 --- a/packages/extension-core/src/extension/diagram/glsp-activation.ts +++ b/packages/extension-core/src/extension/diagram/glsp-activation.ts @@ -37,7 +37,7 @@ import { executeViewerCommand, executeViewerOpen, executeViewerReveal } from './ import { decideDiagramOpen } from './diagram-open-decision'; import { readMcpServerUrl } from './mcp-server-url'; import { composeStorageRuntimeOptions } from './profile-storage-options'; -import { type DiagramProfile, type DiagramRunAnswer, type DiagramRunHost, type DiagramRunQuestion } from '../../api'; +import { type DiagramChatTask, type DiagramProfile, type DiagramRunAnswer, type DiagramRunHost, type DiagramRunQuestion } from '../../api'; // Define the diagram type constant locally to avoid import const WORKFLOW_DIAGRAM_TYPE = 'cal-network-diagram'; @@ -302,6 +302,8 @@ function serializedRangeToVscodeRange(range: SerializedRange): vscode.Range { interface GlspActivationState { /** See {@link GlspIntegrationHandle.setRunQuestionHandler}. */ runQuestionHandler?: RunQuestionHandler; + /** See {@link GlspIntegrationHandle.setChatTaskHandler}. */ + chatTaskHandler?: ChatTaskHandler; context: vscode.ExtensionContext; profile: DiagramProfile; // Transient cross-file drill-down handoff, scoped to this activation (per profile instance). @@ -341,9 +343,12 @@ export interface GlspIntegrationHandle extends vscode.Disposable { * the driver falls back to a VS Code prompt. */ setRunQuestionHandler(handler: RunQuestionHandler | undefined): void; + /** Where the run driver's chat tasks go ("Fix with AI" on a failed run). */ + setChatTaskHandler(handler: ChatTaskHandler | undefined): void; } export type RunQuestionHandler = (question: DiagramRunQuestion, sourceUri: string) => Promise; +export type ChatTaskHandler = (task: DiagramChatTask, sourceUri: string) => Promise; /** * Activate the GLSP integration for Workflow diagrams. @@ -981,6 +986,9 @@ export async function activateGlspIntegration( setRunQuestionHandler: (handler) => { state.runQuestionHandler = handler; }, + setChatTaskHandler: (handler) => { + state.chatTaskHandler = handler; + }, dispose: () => disposable.dispose() }; } @@ -1457,6 +1465,11 @@ function registerCalDiagramCommands( const host: DiagramRunHost = { overlay: executionOverlay, askUser: async (question, sourceUri) => state.runQuestionHandler?.(question, sourceUri), + // Only a profile with a chat can take a task; without one a failed + // run is reported and nothing is offered. + ...(profile.chat + ? { startChatTask: async (task: DiagramChatTask, sourceUri: string) => (await state.chatTaskHandler?.(task, sourceUri)) ?? false } + : {}), requestRefresh: requestRunRefresh, output: runOutput, useLiveOverlaySignatureSource: (source) => { diff --git a/packages/extension-core/src/extension/profile-runtime.ts b/packages/extension-core/src/extension/profile-runtime.ts index 5745588..6048dc8 100644 --- a/packages/extension-core/src/extension/profile-runtime.ts +++ b/packages/extension-core/src/extension/profile-runtime.ts @@ -72,6 +72,7 @@ export async function activateProfileRuntime( // chat panel open on the diagram (the chat is the run's viewer there). const runtime = chatRuntime; glsp.setRunQuestionHandler((question, uri) => runtime.askRunQuestion(uri, question)); + glsp.setChatTaskHandler(async (task, uri) => runtime.startTask(uri, task)); context.subscriptions.push(chatRuntime, { dispose: () => { transport?.dispose(); diff --git a/packages/sidecar-toolkit/src/cli-run-driver.ts b/packages/sidecar-toolkit/src/cli-run-driver.ts index 2e3bfc1..f295ffc 100644 --- a/packages/sidecar-toolkit/src/cli-run-driver.ts +++ b/packages/sidecar-toolkit/src/cli-run-driver.ts @@ -37,6 +37,7 @@ import * as fs from 'node:fs/promises'; import * as path from 'node:path'; import type { ExecutionOverlaySink } from '@dialogram/shared'; import { resumeStepFor } from './resume-at-step.js'; +import { appendTail, failureHeadline, fixTask, type FixTask, type RunFailure } from './run-failure.js'; import { requestWorkflowStop, spawnWorkflowProcess } from './process-control.js'; import { RunEventStreamClient, type RunStreamEvent } from './run-event-stream-client.js'; @@ -163,6 +164,10 @@ export interface CliRunDriverHost { requestRefresh(sourceUri: string, kind: 'full' | 'agentContextOnly', networkName?: string): void; /** Run output channel, owned by core. */ output: vscode.OutputChannel; + /** Starts a task in the chat on the diagram at `sourceUri` (a new session in + * `task.mode`, its first message `task.prompt`); `false` when no chat can. + * Without it a failed run is reported, and nothing is offered. */ + startChatTask?(task: FixTask, sourceUri: string): Promise; } /** Subscription handle returned by the driver's live-overlay APIs. */ @@ -611,7 +616,8 @@ export class CliRunDriver { return { cmd: cliCommand, argsPrefix: [], cwd: startDir }; } - private async findLatestQueueTracePath(opts: { + /** The directory of the latest run of this workflow that started after `startedAfterMs`. */ + private async findLatestRunDir(opts: { baseOutDir: string; sourcePath: string; startedAfterMs: number; @@ -660,11 +666,21 @@ export class CliRunDriver { } } - if (!latest?.outDir) { + return latest?.outDir; + } + + private async findLatestQueueTracePath(opts: { + baseOutDir: string; + sourcePath: string; + startedAfterMs: number; + workflowName?: string; + }): Promise { + const runDir = await this.findLatestRunDir(opts); + if (!runDir) { return undefined; } - const queueTracePath = path.join(latest.outDir, 'run.wf-queues.json'); + const queueTracePath = path.join(runDir, 'run.wf-queues.json'); if (!(await this.fileExists(queueTracePath))) { return undefined; } @@ -675,7 +691,8 @@ export class CliRunDriver { inv: { cmd: string; args: string[]; cwd: string }, output: vscode.OutputChannel, onSpawn?: (child: cp.ChildProcessWithoutNullStreams) => void, - wasStopRequested?: () => boolean + wasStopRequested?: () => boolean, + onStderr?: (text: string) => void ): Promise { return await new Promise((resolve) => { const child = spawnWorkflowProcess(inv); @@ -685,7 +702,11 @@ export class CliRunDriver { } child.stdout.on('data', (d) => output.append(d.toString())); - child.stderr.on('data', (d) => output.append(d.toString())); + child.stderr.on('data', (d) => { + const text = d.toString(); + output.append(text); + onStderr?.(text); + }); child.on('close', (code, signal) => { const exitSummary = `\n[wf-lang] Process exited (code=${code ?? 'null'}${signal ? `, signal=${signal}` : ''})\n`; output.append(exitSummary); @@ -707,6 +728,67 @@ export class CliRunDriver { }); } + /** The error a run recorded, if its record has one. */ + private async readRunFailure(runDir: string): Promise { + try { + const record = JSON.parse(await fs.readFile(path.join(runDir, 'run.wf-run.json'), 'utf-8')); + const error = record?.error; + if (!error || typeof error !== 'object') { + return undefined; + } + return { + ...(typeof error.message === 'string' ? { message: error.message } : {}), + ...(typeof error.entityInstanceName === 'string' ? { entityInstanceName: error.entityInstanceName } : {}), + ...(Array.isArray(error.entityInstancePath) + ? { entityInstancePath: error.entityInstancePath.filter((p: unknown): p is string => typeof p === 'string') } + : {}) + }; + } catch { + return undefined; + } + } + + /** + * Tell the person the run failed, and where. With a chat behind the + * diagram, offer to have the chat find the cause and propose a fix: a new + * session in plan mode, its first message the failure. + */ + private async reportFailure(opts: { + sourceUri: vscode.Uri; + workflowName?: string; + failure: RunFailure | undefined; + exitCode: number; + stderrTail: string; + runDir?: string; + output: vscode.OutputChannel; + }): Promise { + const headline = failureHeadline(opts.failure, opts.exitCode); + const FIX = 'Fix with AI'; + const OUTPUT = 'Show Output'; + const actions = this.host.startChatTask ? [FIX, OUTPUT] : [OUTPUT]; + const choice = await vscode.window.showErrorMessage(headline, ...actions); + if (choice === OUTPUT) { + opts.output.show(true); + return; + } + if (choice !== FIX || !this.host.startChatTask) { + return; + } + const task: FixTask = fixTask({ + workflowName: opts.workflowName, + sourceFile: opts.sourceUri.fsPath, + failure: opts.failure, + exitCode: opts.exitCode, + stderrTail: opts.stderrTail, + runDir: opts.runDir, + canResume: !!this.config.cliResumeArgs && !!opts.runDir + }); + const started = await this.host.startChatTask(task, opts.sourceUri.toString()); + if (!started) { + void vscode.window.showWarningMessage('Open the chat on this diagram to have the failure looked at.'); + } + } + private async stopWorkflow(): Promise<{ ok: boolean; message: string }> { const active = this.activeRun; if (!active) { @@ -980,6 +1062,8 @@ export class CliRunDriver { this.stopElicitSocket(); }; + // The end of what the run wrote to stderr -- its traceback, when it fails. + let stderrTail = ''; const exitCode = await vscode.window.withProgress( { location: vscode.ProgressLocation.Notification, @@ -1007,7 +1091,10 @@ export class CliRunDriver { stopRequested: false }; }, - () => this.activeRun?.stopRequested === true + () => this.activeRun?.stopRequested === true, + (text) => { + stderrTail = appendTail(stderrTail, text); + } ); } ); @@ -1026,9 +1113,18 @@ export class CliRunDriver { } if (exitCode !== 0) { - const msg = `Workflow run failed (exit code ${exitCode}). See 'wf-lang Run' output.`; - vscode.window.showErrorMessage(msg); - return { error: msg }; + // Show where it failed on the diagram, then say so -- and, with a + // chat behind the diagram, offer to have it looked at. + await refreshDiagram(true); + const runDir = await this.findLatestRunDir({ + baseOutDir: runOutDir, + sourcePath: sourceUri.fsPath, + startedAfterMs: runStartedAtMs, + workflowName + }); + const failure = runDir ? await this.readRunFailure(runDir) : undefined; + void this.reportFailure({ sourceUri, workflowName, failure, exitCode, stderrTail, runDir, output }); + return { error: failureHeadline(failure, exitCode) }; } // Refresh diagram so overlay decorations can pick up the new run artifacts. diff --git a/packages/sidecar-toolkit/src/run-failure.ts b/packages/sidecar-toolkit/src/run-failure.ts new file mode 100644 index 0000000..05f18e6 --- /dev/null +++ b/packages/sidecar-toolkit/src/run-failure.ts @@ -0,0 +1,97 @@ +/** + * A failed run, told to the person and to the agent that may fix it. + * + * When a run fails in a diagram with a chat behind it, the driver offers to + * have the chat look at it: a new session, in plan mode, whose first message is + * the failure -- where it happened, the error, the end of the run's output and + * where its files are. Plan mode, because the person asked for a diagnosis and + * a proposal, not for files to change under them. + */ + +/** What a run record says about a failure (all of it optional). */ +export interface RunFailure { + message?: string; + /** The failing node's instance path from the root, `['m', 'i', 'x']`. */ + entityInstancePath?: string[]; + entityInstanceName?: string; +} + +/** The task the chat is asked to start: its session's name, its mode, its first message. */ +export interface FixTask { + name: string; + mode: 'plan' | 'build'; + prompt: string; +} + +/** At most this much of the run's stderr goes to the agent: the end, where the traceback is. */ +export const STDERR_TAIL_CHARS = 8000; + +/** Keep the end of a growing stream of text, as the run writes it. */ +export function appendTail(tail: string, chunk: string, max = STDERR_TAIL_CHARS): string { + const next = tail + chunk; + return next.length > max ? next.slice(next.length - max) : next; +} + +/** Where it failed, for a person: `x (in m › i)`, or the node alone at the root. */ +export function failureWhere(failure: RunFailure | undefined): string | undefined { + const path = failure?.entityInstancePath?.filter(part => part.trim() !== '') ?? []; + if (path.length > 0) { + const node = path[path.length - 1]; + return path.length > 1 ? `${node} (in ${path.slice(0, -1).join(' › ')})` : node; + } + return failure?.entityInstanceName?.trim() || undefined; +} + +/** The first line of the error, short enough for a notification. */ +export function failureHeadline(failure: RunFailure | undefined, exitCode: number): string { + const first = failure?.message?.split(/\r?\n/).find(line => line.trim() !== '')?.trim(); + const error = first ? (first.length > 160 ? `${first.slice(0, 157)}…` : first) : `exit code ${exitCode}`; + const where = failureWhere(failure); + return where ? `Run failed in ${where}: ${error}` : `Run failed: ${error}`; +} + +/** The chat task that asks the agent for the cause and a fix. */ +export function fixTask(opts: { + workflowName?: string; + sourceFile: string; + failure: RunFailure | undefined; + exitCode: number; + stderrTail: string; + runDir?: string; + canResume: boolean; +}): FixTask { + const workflow = opts.workflowName ? `\`${opts.workflowName}\` (${opts.sourceFile})` : opts.sourceFile; + const path = opts.failure?.entityInstancePath?.filter(part => part.trim() !== '') ?? []; + const lines = [`The last run of workflow ${workflow} failed. Find the cause and propose a fix.`, '']; + if (path.length > 0) { + const node = path[path.length - 1]; + lines.push( + path.length > 1 + ? `- Where: node \`${node}\`, inside the nested workflows ${path.slice(0, -1).map(p => `\`${p}\``).join(' › ')} (instance path \`${path.join('/')}\`).` + : `- Where: node \`${node}\`.` + ); + } else if (opts.failure?.entityInstanceName) { + lines.push(`- Where: node \`${opts.failure.entityInstanceName}\`.`); + } + lines.push(`- Error: ${opts.failure?.message?.trim() || `the run exited with code ${opts.exitCode}`}`); + if (opts.runDir) { + lines.push(`- Run directory: ${opts.runDir} (its run record, logs and intermediate files).`); + } + const tail = opts.stderrTail.trim(); + if (tail) { + lines.push('', 'The end of the run\'s error output:', '', '```text', tail, '```'); + } + lines.push( + '', + 'Explain the cause first, then propose the change, and say which file it goes in.' + + (opts.canResume + ? ' Once it is fixed, the run can be resumed from where it failed rather than started over.' + : '') + ); + const where = failureWhere(opts.failure); + return { + name: `Fix: ${where ?? opts.workflowName ?? 'failed run'}`, + mode: 'plan', + prompt: lines.join('\n') + }; +} diff --git a/packages/sidecar-toolkit/src/sidecar-diagram-profile.ts b/packages/sidecar-toolkit/src/sidecar-diagram-profile.ts index 8f4a447..cc95749 100644 --- a/packages/sidecar-toolkit/src/sidecar-diagram-profile.ts +++ b/packages/sidecar-toolkit/src/sidecar-diagram-profile.ts @@ -363,7 +363,9 @@ export function createSidecarDiagramProfile(input: SidecarProfileInput) { output: host.output, // The human port: a running agent's question goes to the platform // (the chat panel on the diagram); without it the driver prompts. - askUser: host.askUser ? (question, sourceUri) => host.askUser!(question, sourceUri) : undefined + askUser: host.askUser ? (question, sourceUri) => host.askUser!(question, sourceUri) : undefined, + // "Fix with AI" on a failed run: a task for the diagram's chat. + startChatTask: host.startChatTask ? (task, sourceUri) => host.startChatTask!(task, sourceUri) : undefined }); driver.registerCommands(context); host.useLiveOverlaySignatureSource({ diff --git a/packages/sidecar-toolkit/test/cli-run-driver-fix-with-ai.test.ts b/packages/sidecar-toolkit/test/cli-run-driver-fix-with-ai.test.ts new file mode 100644 index 0000000..9189693 --- /dev/null +++ b/packages/sidecar-toolkit/test/cli-run-driver-fix-with-ai.test.ts @@ -0,0 +1,132 @@ +/** + * A run that fails: the driver says where, and -- with a chat behind the + * diagram -- offers "Fix with AI", which starts a plan-mode chat session whose + * first message is the failure: its path, error, traceback and run directory. + */ +import { EventEmitter } from 'node:events'; +import * as fs from 'node:fs'; +import * as os from 'node:os'; +import * as path from 'node:path'; +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest'; +import * as vscode from 'vscode'; +import { resetRegisteredCommands } from './vscode-mock'; + +vi.mock('../src/process-control.js', () => ({ + spawnWorkflowProcess: (invocation: { cmd: string; args: string[]; cwd: string }) => { + const child: any = new EventEmitter(); + child.stdout = new EventEmitter(); + child.stderr = new EventEmitter(); + // The run, as the runtime would leave it: a log line and a failed record. + const outDir = invocation.args[invocation.args.indexOf('--out-dir') + 1]; + const runDir = path.join(outDir, 'r1'); + fs.mkdirSync(runDir, { recursive: true }); + const source = invocation.args[invocation.args.indexOf('run') + 1]; + const now = new Date().toISOString(); + fs.writeFileSync(path.join(outDir, 'run-log.jsonl'), JSON.stringify({ + runId: 'r1', workflowName: 'top', sourcePath: source, startedAt: now, finishedAt: now, outDir: runDir + }) + '\n'); + fs.writeFileSync(path.join(runDir, 'run.wf-run.json'), JSON.stringify({ + error: { message: 'ValueError: deep down', entityInstanceName: 'x', entityInstancePath: ['m', 'i', 'x'] } + })); + setTimeout(() => { + child.stderr.emit('data', Buffer.from('Traceback (most recent call last):\nValueError: deep down\n')); + child.emit('close', 1, null); + }, 0); + return child; + }, + requestWorkflowStop: () => true +})); +vi.mock('../src/run-event-stream-client.js', () => ({ + RunEventStreamClient: class { constructor(_o: unknown) {} start(): void {} stop(): void {} } +})); + +import { CliRunDriver, type CliRunDriverConfig } from '../src/index'; + +const RUN_CMD = 'test.fix.runWorkflow'; + +function makeDriver(startChatTask?: (task: any, uri: string) => Promise) { + const config: CliRunDriverConfig = { + settingsNamespace: 'wfLang', customEditorViewType: 'workflow.networkDiagram', + cliCommandSettingKey: 'wfpyCommand', cliCommandDefault: 'fake-wfpy', cliPythonModule: undefined, + runOutputDirSettingKey: 'runOutputDir', liveExecutionGlowSettingKey: 'liveExecutionGlow', + agentToolsSettingKey: 'agentTools', agentToolAuthSettingKey: 'agentToolAuth', + agentToolPolicySettingKey: 'agentToolPolicy', agentToolTimeoutMsSettingKey: 'agentToolTimeoutMs', + agentToolRegistrySettingKey: 'agentToolRegistry', agentMcpBridgeCmdSettingKey: 'agentMcpBridgeCmd', + runWorkflowCommandId: RUN_CMD, stopWorkflowCommandId: 'test.fix.stop', + agentToolConfigCommands: { set: 'test.fix.set', get: 'test.fix.get' }, + overrideState: { get: () => undefined, update: () => Promise.resolve() }, + cliResumeArgs: (dir, step) => ['--resume-from', dir, '--at-step', String(step)], + elicitSocket: false + }; + const host: any = { + overlay: { emitEvents: () => {} }, + requestRefresh: () => {}, + output: { show: vi.fn(), append: () => {}, appendLine: () => {} }, + ...(startChatTask ? { startChatTask } : {}) + }; + const driver = new CliRunDriver(config, host); + driver.registerCommands({ subscriptions: [] } as any); + return { driver, host }; +} + +let original: any; +let shown: Array<{ message: string; actions: string[] }>; +let answer: string | undefined; + +beforeEach(() => { + resetRegisteredCommands(); + shown = []; + original = (vscode.window as any).showErrorMessage; + (vscode.window as any).showErrorMessage = async (message: string, ...actions: string[]) => { + shown.push({ message, actions }); + return answer; + }; +}); +afterEach(() => { + (vscode.window as any).showErrorMessage = original; +}); + +async function failRun(): Promise { + const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'fix-with-ai-')); + const file = path.join(dir, 'top.py'); + fs.writeFileSync(file, '# top\n'); + await vscode.commands.executeCommand(RUN_CMD, { sourceUri: `file://${file}`, workflowName: 'top' }); + // The offer is made without holding the run command up. + await new Promise(resolve => setTimeout(resolve, 20)); +} + +describe('a run that fails', () => { + it('says where it failed, and offers to fix it with the chat', async () => { + answer = undefined; + makeDriver(async () => true); + await failRun(); + + expect(shown).toHaveLength(1); + expect(shown[0].message).toBe('Run failed in x (in m › i): ValueError: deep down'); + expect(shown[0].actions).toEqual(['Fix with AI', 'Show Output']); + }); + + it('starts a plan-mode chat task with the failure when the offer is taken', async () => { + answer = 'Fix with AI'; + const startChatTask = vi.fn(async () => true); + makeDriver(startChatTask); + await failRun(); + + expect(startChatTask).toHaveBeenCalledTimes(1); + const [task, uri] = startChatTask.mock.calls[0] as any[]; + expect(uri).toMatch(/top\.py$/); + expect(task.mode).toBe('plan'); + expect(task.name).toBe('Fix: x (in m › i)'); + expect(task.prompt).toContain('instance path `m/i/x`'); + expect(task.prompt).toContain('Traceback (most recent call last):'); + expect(task.prompt).toContain('resumed from where it failed'); + }); + + it('offers no fix without a chat behind the diagram', async () => { + answer = undefined; + makeDriver(undefined); + await failRun(); + + expect(shown[0].actions).toEqual(['Show Output']); + }); +}); diff --git a/packages/sidecar-toolkit/test/run-failure.test.ts b/packages/sidecar-toolkit/test/run-failure.test.ts new file mode 100644 index 0000000..4e1165f --- /dev/null +++ b/packages/sidecar-toolkit/test/run-failure.test.ts @@ -0,0 +1,63 @@ +/** + * A failed run told to the person (one line) and to the agent asked to fix it + * (the error, where it happened, the traceback, where the run's files are). + */ +import { describe, expect, it } from 'vitest'; +import { appendTail, failureHeadline, failureWhere, fixTask } from '../src/run-failure'; + +const deep = { message: 'ValueError: deep down', entityInstancePath: ['m', 'i', 'x'], entityInstanceName: 'x' }; + +describe('where a run failed', () => { + it('names the node and the nested workflows it is in', () => { + expect(failureWhere(deep)).toBe('x (in m › i)'); + expect(failureWhere({ entityInstancePath: ['x'] })).toBe('x'); + expect(failureWhere({ entityInstanceName: 'y' })).toBe('y'); + expect(failureWhere(undefined)).toBeUndefined(); + }); + + it('fits in a notification: where, and the first line of the error', () => { + expect(failureHeadline(deep, 1)).toBe('Run failed in x (in m › i): ValueError: deep down'); + expect(failureHeadline(undefined, 3)).toBe('Run failed: exit code 3'); + expect(failureHeadline({ message: 'x'.repeat(400) }, 1).length).toBeLessThan(200); + }); +}); + +describe('the end of the run’s error output', () => { + it('keeps the end, where the traceback is', () => { + expect(appendTail('abc', 'def', 4)).toBe('cdef'); + expect(appendTail('', 'Traceback', 100)).toBe('Traceback'); + }); +}); + +describe('the task the chat is asked to start', () => { + const task = fixTask({ + workflowName: 'top', + sourceFile: '/w/top.py', + failure: deep, + exitCode: 1, + stderrTail: 'Traceback (most recent call last):\n ...\nValueError: deep down\n', + runDir: '/w/wf-out/r1', + canResume: true + }); + + it('is a plan-mode session named for where it failed', () => { + expect(task.mode).toBe('plan'); + expect(task.name).toBe('Fix: x (in m › i)'); + }); + + it('gives the agent what it needs to find the cause', () => { + expect(task.prompt).toContain('workflow `top` (/w/top.py) failed'); + expect(task.prompt).toContain('node `x`, inside the nested workflows `m` › `i` (instance path `m/i/x`)'); + expect(task.prompt).toContain('- Error: ValueError: deep down'); + expect(task.prompt).toContain('/w/wf-out/r1'); + expect(task.prompt).toContain('```text\nTraceback (most recent call last):'); + expect(task.prompt).toContain('resumed from where it failed'); + }); + + it('says nothing of resuming when the run cannot be resumed', () => { + const plain = fixTask({ sourceFile: '/w/top.py', failure: undefined, exitCode: 2, stderrTail: '', canResume: false }); + expect(plain.prompt).toContain('- Error: the run exited with code 2'); + expect(plain.prompt).not.toContain('resumed'); + expect(plain.prompt).not.toContain('```text'); + }); +});