From 189b26117339a31116759729d18b034db79da2ce Mon Sep 17 00:00:00 2001 From: Endri Bezati Date: Fri, 2 Oct 2026 16:04:21 +0200 Subject: [PATCH] A failed run offers "Fix with AI": a chat session that finds the cause and proposes a fix MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit A failed run said only "Workflow run failed (exit code N). See the run output." -- and did not refresh the diagram, so the error marks waited for something else to redraw it. Now the diagram is refreshed, and the notification says where it failed and what (`Run failed in x (in m › i): ValueError: …`, from the run record's error and instance path). With a chat behind the diagram it also offers "Fix with AI". Taking it starts a task in that diagram's chat: a new session in plan mode, named for where it failed, whose first message is the failure -- the node and the nested workflows it is in, the error, the end of the run's stderr (its traceback, kept as the run writes it), the run directory, and, when the product can resume runs, that the run can be resumed from the failure once fixed. Plan mode: the person asked for a diagnosis and a proposal, not for files to change under them. The panel sends the message as if typed, so it is in the transcript like any other. The path: the run driver's host gains `startChatTask`, set only for a profile with a chat; it reaches `ChatRuntime.startTask`, which posts `chat.startTask` to the panel on that diagram. Without a chat, a failed run is reported and nothing is offered. Claude-Session: https://claude.ai/code/session_015VK7fH1c4aKbexnq2QcuKU --- .../src/chat-panel-integrated.ts | 31 ++++ .../test/chat-panel-start-task.test.ts | 61 ++++++++ packages/extension-core/src/api.ts | 11 ++ .../src/extension/chat/chat-runtime.ts | 15 ++ .../src/extension/diagram/glsp-activation.ts | 15 +- .../src/extension/profile-runtime.ts | 1 + .../sidecar-toolkit/src/cli-run-driver.ts | 114 +++++++++++++-- packages/sidecar-toolkit/src/run-failure.ts | 97 +++++++++++++ .../src/sidecar-diagram-profile.ts | 4 +- .../test/cli-run-driver-fix-with-ai.test.ts | 132 ++++++++++++++++++ .../sidecar-toolkit/test/run-failure.test.ts | 63 +++++++++ 11 files changed, 533 insertions(+), 11 deletions(-) create mode 100644 packages/diagram-client/test/chat-panel-start-task.test.ts create mode 100644 packages/sidecar-toolkit/src/run-failure.ts create mode 100644 packages/sidecar-toolkit/test/cli-run-driver-fix-with-ai.test.ts create mode 100644 packages/sidecar-toolkit/test/run-failure.test.ts diff --git a/packages/diagram-client/src/chat-panel-integrated.ts b/packages/diagram-client/src/chat-panel-integrated.ts index 83d395c..dbf8e16 100644 --- a/packages/diagram-client/src/chat-panel-integrated.ts +++ b/packages/diagram-client/src/chat-panel-integrated.ts @@ -193,6 +193,8 @@ export class ChatPanel implements IDiagramStartup, ISelectionListener { private isVisible = false; private isCompact = false; private currentMode: 'plan' | 'build' = 'build'; + /** A task the host asked for (`chat.startTask`), sent once its session exists. */ + private pendingTask: { name: string; mode: 'plan' | 'build'; prompt: string } | undefined; private currentSessionId: string | null = null; private timeline: TimelineItem[] = []; private sessions: SessionEntry[] = []; @@ -494,11 +496,40 @@ export class ChatPanel implements IDiagramStartup, ISelectionListener { this.pushMessage('system', `Created session: ${data.session.name ?? data.session.id}`); } this.update(); + // A task's session: send its message, as if typed, so it is in the + // transcript like any other. + if (data?.session?.id && this.pendingTask) { + const task = this.pendingTask; + this.pendingTask = undefined; + this.inputValue = task.prompt; + this.sendMessage(); + } + break; + + case 'chat.startTask': + // The host asks for a task in a new session: "Fix with AI" on a run + // that failed. Open the panel, start the session in the task's mode, + // and send its message once the session exists. + if (data && typeof data.prompt === 'string' && data.prompt.trim() !== '') { + const mode: 'plan' | 'build' = data.mode === 'build' ? 'build' : 'plan'; + this.pendingTask = { + name: typeof data.name === 'string' && data.name.trim() !== '' ? data.name : 'Task', + mode, + prompt: data.prompt + }; + this.currentMode = mode; + this.isLoadingSession = true; + this.loadingLabel = 'Creating session…'; + this.autoShow('task'); + this.update(); + this.sendToHost('chat.createSession', { mode, name: this.pendingTask.name }); + } break; case 'chat.sessionCreateAborted': // Name prompt cancelled — drop the "Creating session…" spinner. this.isLoadingSession = false; + this.pendingTask = undefined; this.update(); break; diff --git a/packages/diagram-client/test/chat-panel-start-task.test.ts b/packages/diagram-client/test/chat-panel-start-task.test.ts new file mode 100644 index 0000000..f3a5004 --- /dev/null +++ b/packages/diagram-client/test/chat-panel-start-task.test.ts @@ -0,0 +1,61 @@ +/** + * A task the host asks the chat to start ("Fix with AI" on a failed run): the + * panel opens, creates a session in the task's mode under the task's name, and + * once it exists sends the task's message as if typed. + */ +import 'reflect-metadata'; +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest'; +import { ChatPanel } from '../src/chat-panel-integrated'; + +function makePanel() { + const panel = new ChatPanel(); + const sent: Array<{ type: string; data: any }> = []; + (panel as any).channel = { sendToHost: (_method: string, env: any) => void sent.push(env) }; + const showSpy = vi.fn(() => { (panel as any).isVisible = true; }); + (panel as any).show = showSpy; + return { panel, sent, showSpy }; +} + +beforeEach(() => { + (globalThis as any).requestAnimationFrame = () => 1; + (globalThis as any).cancelAnimationFrame = () => undefined; +}); +afterEach(() => { + delete (globalThis as any).requestAnimationFrame; + delete (globalThis as any).cancelAnimationFrame; +}); + +const task = { name: 'Fix: x (in m › i)', mode: 'plan', prompt: 'The last run of workflow `top` failed. …' }; + +describe('a task the host starts in the chat', () => { + it('opens the panel and creates a session in the task’s mode and name', () => { + const { panel, sent, showSpy } = makePanel(); + + (panel as any).handleIncomingMessage('chat.startTask', task); + + expect(showSpy).toHaveBeenCalled(); + expect(sent.find(m => m.type === 'chat.createSession')?.data).toMatchObject({ mode: 'plan', name: task.name }); + expect(sent.some(m => m.type === 'chat.sendMessage')).toBe(false); + }); + + it('sends the task’s message once the session exists, in plan mode, once', () => { + const { panel, sent } = makePanel(); + (panel as any).handleIncomingMessage('chat.startTask', task); + + (panel as any).handleIncomingMessage('chat.sessionCreated', { session: { id: 's1', name: task.name } }); + (panel as any).handleIncomingMessage('chat.sessionCreated', { session: { id: 's2', name: 'later' } }); + + const messages = sent.filter(m => m.type === 'chat.sendMessage').map(m => m.data); + expect(messages).toHaveLength(1); + expect(messages[0]).toMatchObject({ text: task.prompt, sessionId: 's1', mode: 'plan' }); + }); + + it('drops the task when the session is not created', () => { + const { panel, sent } = makePanel(); + (panel as any).handleIncomingMessage('chat.startTask', task); + (panel as any).handleIncomingMessage('chat.sessionCreateAborted', {}); + (panel as any).handleIncomingMessage('chat.sessionCreated', { session: { id: 's1' } }); + + expect(sent.some(m => m.type === 'chat.sendMessage')).toBe(false); + }); +}); diff --git a/packages/extension-core/src/api.ts b/packages/extension-core/src/api.ts index 5b8020c..24cc865 100644 --- a/packages/extension-core/src/api.ts +++ b/packages/extension-core/src/api.ts @@ -227,6 +227,17 @@ export interface DiagramRunHost { ): void; output: vscode.OutputChannel; useLiveOverlaySignatureSource(source: DiagramLiveOverlaySource): void; + /** Starts a task in the chat on the diagram at `sourceUri`: a new session in + * `task.mode`, named `task.name`, whose first message is `task.prompt`. + * `false` when no chat can take it there. */ + startChatTask?(task: DiagramChatTask, sourceUri: string): Promise; +} + +/** A task the platform asks a diagram's chat to start (see `DiagramRunHost.startChatTask`). */ +export interface DiagramChatTask { + name: string; + mode: "plan" | "build"; + prompt: string; } /** diff --git a/packages/extension-core/src/extension/chat/chat-runtime.ts b/packages/extension-core/src/extension/chat/chat-runtime.ts index 426f171..76e8c10 100644 --- a/packages/extension-core/src/extension/chat/chat-runtime.ts +++ b/packages/extension-core/src/extension/chat/chat-runtime.ts @@ -881,6 +881,21 @@ export class ChatRuntime { }); } + /** + * Start a task in the chat panel open on `uri`: a new session named + * `task.name`, in `task.mode`, whose first message is `task.prompt` -- sent by + * the panel as if typed, so the person sees what was asked. `false` when no + * panel can be reached there. + */ + startTask(uri: string, task: { name: string; mode: "plan" | "build"; prompt: string }): boolean { + if (this.canReach && !this.canReach(uri)) { + this.logLine(`chat task "${task.name}": no chat panel on ${uri}`); + return false; + } + this.postToWebview(uri, { type: "chat.startTask", data: { ...task } }); + return true; + } + /** The last model the user explicitly chose (migrated from the legacy key). */ private getPreferredModel(): string | undefined { return readStateWithFallback( diff --git a/packages/extension-core/src/extension/diagram/glsp-activation.ts b/packages/extension-core/src/extension/diagram/glsp-activation.ts index f8db054..67ee7b1 100644 --- a/packages/extension-core/src/extension/diagram/glsp-activation.ts +++ b/packages/extension-core/src/extension/diagram/glsp-activation.ts @@ -37,7 +37,7 @@ import { executeViewerCommand, executeViewerOpen, executeViewerReveal } from './ import { decideDiagramOpen } from './diagram-open-decision'; import { readMcpServerUrl } from './mcp-server-url'; import { composeStorageRuntimeOptions } from './profile-storage-options'; -import { type DiagramProfile, type DiagramRunAnswer, type DiagramRunHost, type DiagramRunQuestion } from '../../api'; +import { type DiagramChatTask, type DiagramProfile, type DiagramRunAnswer, type DiagramRunHost, type DiagramRunQuestion } from '../../api'; // Define the diagram type constant locally to avoid import const WORKFLOW_DIAGRAM_TYPE = 'cal-network-diagram'; @@ -302,6 +302,8 @@ function serializedRangeToVscodeRange(range: SerializedRange): vscode.Range { interface GlspActivationState { /** See {@link GlspIntegrationHandle.setRunQuestionHandler}. */ runQuestionHandler?: RunQuestionHandler; + /** See {@link GlspIntegrationHandle.setChatTaskHandler}. */ + chatTaskHandler?: ChatTaskHandler; context: vscode.ExtensionContext; profile: DiagramProfile; // Transient cross-file drill-down handoff, scoped to this activation (per profile instance). @@ -341,9 +343,12 @@ export interface GlspIntegrationHandle extends vscode.Disposable { * the driver falls back to a VS Code prompt. */ setRunQuestionHandler(handler: RunQuestionHandler | undefined): void; + /** Where the run driver's chat tasks go ("Fix with AI" on a failed run). */ + setChatTaskHandler(handler: ChatTaskHandler | undefined): void; } export type RunQuestionHandler = (question: DiagramRunQuestion, sourceUri: string) => Promise; +export type ChatTaskHandler = (task: DiagramChatTask, sourceUri: string) => Promise; /** * Activate the GLSP integration for Workflow diagrams. @@ -981,6 +986,9 @@ export async function activateGlspIntegration( setRunQuestionHandler: (handler) => { state.runQuestionHandler = handler; }, + setChatTaskHandler: (handler) => { + state.chatTaskHandler = handler; + }, dispose: () => disposable.dispose() }; } @@ -1457,6 +1465,11 @@ function registerCalDiagramCommands( const host: DiagramRunHost = { overlay: executionOverlay, askUser: async (question, sourceUri) => state.runQuestionHandler?.(question, sourceUri), + // Only a profile with a chat can take a task; without one a failed + // run is reported and nothing is offered. + ...(profile.chat + ? { startChatTask: async (task: DiagramChatTask, sourceUri: string) => (await state.chatTaskHandler?.(task, sourceUri)) ?? false } + : {}), requestRefresh: requestRunRefresh, output: runOutput, useLiveOverlaySignatureSource: (source) => { diff --git a/packages/extension-core/src/extension/profile-runtime.ts b/packages/extension-core/src/extension/profile-runtime.ts index 5745588..6048dc8 100644 --- a/packages/extension-core/src/extension/profile-runtime.ts +++ b/packages/extension-core/src/extension/profile-runtime.ts @@ -72,6 +72,7 @@ export async function activateProfileRuntime( // chat panel open on the diagram (the chat is the run's viewer there). const runtime = chatRuntime; glsp.setRunQuestionHandler((question, uri) => runtime.askRunQuestion(uri, question)); + glsp.setChatTaskHandler(async (task, uri) => runtime.startTask(uri, task)); context.subscriptions.push(chatRuntime, { dispose: () => { transport?.dispose(); diff --git a/packages/sidecar-toolkit/src/cli-run-driver.ts b/packages/sidecar-toolkit/src/cli-run-driver.ts index 2e3bfc1..f295ffc 100644 --- a/packages/sidecar-toolkit/src/cli-run-driver.ts +++ b/packages/sidecar-toolkit/src/cli-run-driver.ts @@ -37,6 +37,7 @@ import * as fs from 'node:fs/promises'; import * as path from 'node:path'; import type { ExecutionOverlaySink } from '@dialogram/shared'; import { resumeStepFor } from './resume-at-step.js'; +import { appendTail, failureHeadline, fixTask, type FixTask, type RunFailure } from './run-failure.js'; import { requestWorkflowStop, spawnWorkflowProcess } from './process-control.js'; import { RunEventStreamClient, type RunStreamEvent } from './run-event-stream-client.js'; @@ -163,6 +164,10 @@ export interface CliRunDriverHost { requestRefresh(sourceUri: string, kind: 'full' | 'agentContextOnly', networkName?: string): void; /** Run output channel, owned by core. */ output: vscode.OutputChannel; + /** Starts a task in the chat on the diagram at `sourceUri` (a new session in + * `task.mode`, its first message `task.prompt`); `false` when no chat can. + * Without it a failed run is reported, and nothing is offered. */ + startChatTask?(task: FixTask, sourceUri: string): Promise; } /** Subscription handle returned by the driver's live-overlay APIs. */ @@ -611,7 +616,8 @@ export class CliRunDriver { return { cmd: cliCommand, argsPrefix: [], cwd: startDir }; } - private async findLatestQueueTracePath(opts: { + /** The directory of the latest run of this workflow that started after `startedAfterMs`. */ + private async findLatestRunDir(opts: { baseOutDir: string; sourcePath: string; startedAfterMs: number; @@ -660,11 +666,21 @@ export class CliRunDriver { } } - if (!latest?.outDir) { + return latest?.outDir; + } + + private async findLatestQueueTracePath(opts: { + baseOutDir: string; + sourcePath: string; + startedAfterMs: number; + workflowName?: string; + }): Promise { + const runDir = await this.findLatestRunDir(opts); + if (!runDir) { return undefined; } - const queueTracePath = path.join(latest.outDir, 'run.wf-queues.json'); + const queueTracePath = path.join(runDir, 'run.wf-queues.json'); if (!(await this.fileExists(queueTracePath))) { return undefined; } @@ -675,7 +691,8 @@ export class CliRunDriver { inv: { cmd: string; args: string[]; cwd: string }, output: vscode.OutputChannel, onSpawn?: (child: cp.ChildProcessWithoutNullStreams) => void, - wasStopRequested?: () => boolean + wasStopRequested?: () => boolean, + onStderr?: (text: string) => void ): Promise { return await new Promise((resolve) => { const child = spawnWorkflowProcess(inv); @@ -685,7 +702,11 @@ export class CliRunDriver { } child.stdout.on('data', (d) => output.append(d.toString())); - child.stderr.on('data', (d) => output.append(d.toString())); + child.stderr.on('data', (d) => { + const text = d.toString(); + output.append(text); + onStderr?.(text); + }); child.on('close', (code, signal) => { const exitSummary = `\n[wf-lang] Process exited (code=${code ?? 'null'}${signal ? `, signal=${signal}` : ''})\n`; output.append(exitSummary); @@ -707,6 +728,67 @@ export class CliRunDriver { }); } + /** The error a run recorded, if its record has one. */ + private async readRunFailure(runDir: string): Promise { + try { + const record = JSON.parse(await fs.readFile(path.join(runDir, 'run.wf-run.json'), 'utf-8')); + const error = record?.error; + if (!error || typeof error !== 'object') { + return undefined; + } + return { + ...(typeof error.message === 'string' ? { message: error.message } : {}), + ...(typeof error.entityInstanceName === 'string' ? { entityInstanceName: error.entityInstanceName } : {}), + ...(Array.isArray(error.entityInstancePath) + ? { entityInstancePath: error.entityInstancePath.filter((p: unknown): p is string => typeof p === 'string') } + : {}) + }; + } catch { + return undefined; + } + } + + /** + * Tell the person the run failed, and where. With a chat behind the + * diagram, offer to have the chat find the cause and propose a fix: a new + * session in plan mode, its first message the failure. + */ + private async reportFailure(opts: { + sourceUri: vscode.Uri; + workflowName?: string; + failure: RunFailure | undefined; + exitCode: number; + stderrTail: string; + runDir?: string; + output: vscode.OutputChannel; + }): Promise { + const headline = failureHeadline(opts.failure, opts.exitCode); + const FIX = 'Fix with AI'; + const OUTPUT = 'Show Output'; + const actions = this.host.startChatTask ? [FIX, OUTPUT] : [OUTPUT]; + const choice = await vscode.window.showErrorMessage(headline, ...actions); + if (choice === OUTPUT) { + opts.output.show(true); + return; + } + if (choice !== FIX || !this.host.startChatTask) { + return; + } + const task: FixTask = fixTask({ + workflowName: opts.workflowName, + sourceFile: opts.sourceUri.fsPath, + failure: opts.failure, + exitCode: opts.exitCode, + stderrTail: opts.stderrTail, + runDir: opts.runDir, + canResume: !!this.config.cliResumeArgs && !!opts.runDir + }); + const started = await this.host.startChatTask(task, opts.sourceUri.toString()); + if (!started) { + void vscode.window.showWarningMessage('Open the chat on this diagram to have the failure looked at.'); + } + } + private async stopWorkflow(): Promise<{ ok: boolean; message: string }> { const active = this.activeRun; if (!active) { @@ -980,6 +1062,8 @@ export class CliRunDriver { this.stopElicitSocket(); }; + // The end of what the run wrote to stderr -- its traceback, when it fails. + let stderrTail = ''; const exitCode = await vscode.window.withProgress( { location: vscode.ProgressLocation.Notification, @@ -1007,7 +1091,10 @@ export class CliRunDriver { stopRequested: false }; }, - () => this.activeRun?.stopRequested === true + () => this.activeRun?.stopRequested === true, + (text) => { + stderrTail = appendTail(stderrTail, text); + } ); } ); @@ -1026,9 +1113,18 @@ export class CliRunDriver { } if (exitCode !== 0) { - const msg = `Workflow run failed (exit code ${exitCode}). See 'wf-lang Run' output.`; - vscode.window.showErrorMessage(msg); - return { error: msg }; + // Show where it failed on the diagram, then say so -- and, with a + // chat behind the diagram, offer to have it looked at. + await refreshDiagram(true); + const runDir = await this.findLatestRunDir({ + baseOutDir: runOutDir, + sourcePath: sourceUri.fsPath, + startedAfterMs: runStartedAtMs, + workflowName + }); + const failure = runDir ? await this.readRunFailure(runDir) : undefined; + void this.reportFailure({ sourceUri, workflowName, failure, exitCode, stderrTail, runDir, output }); + return { error: failureHeadline(failure, exitCode) }; } // Refresh diagram so overlay decorations can pick up the new run artifacts. diff --git a/packages/sidecar-toolkit/src/run-failure.ts b/packages/sidecar-toolkit/src/run-failure.ts new file mode 100644 index 0000000..05f18e6 --- /dev/null +++ b/packages/sidecar-toolkit/src/run-failure.ts @@ -0,0 +1,97 @@ +/** + * A failed run, told to the person and to the agent that may fix it. + * + * When a run fails in a diagram with a chat behind it, the driver offers to + * have the chat look at it: a new session, in plan mode, whose first message is + * the failure -- where it happened, the error, the end of the run's output and + * where its files are. Plan mode, because the person asked for a diagnosis and + * a proposal, not for files to change under them. + */ + +/** What a run record says about a failure (all of it optional). */ +export interface RunFailure { + message?: string; + /** The failing node's instance path from the root, `['m', 'i', 'x']`. */ + entityInstancePath?: string[]; + entityInstanceName?: string; +} + +/** The task the chat is asked to start: its session's name, its mode, its first message. */ +export interface FixTask { + name: string; + mode: 'plan' | 'build'; + prompt: string; +} + +/** At most this much of the run's stderr goes to the agent: the end, where the traceback is. */ +export const STDERR_TAIL_CHARS = 8000; + +/** Keep the end of a growing stream of text, as the run writes it. */ +export function appendTail(tail: string, chunk: string, max = STDERR_TAIL_CHARS): string { + const next = tail + chunk; + return next.length > max ? next.slice(next.length - max) : next; +} + +/** Where it failed, for a person: `x (in m › i)`, or the node alone at the root. */ +export function failureWhere(failure: RunFailure | undefined): string | undefined { + const path = failure?.entityInstancePath?.filter(part => part.trim() !== '') ?? []; + if (path.length > 0) { + const node = path[path.length - 1]; + return path.length > 1 ? `${node} (in ${path.slice(0, -1).join(' › ')})` : node; + } + return failure?.entityInstanceName?.trim() || undefined; +} + +/** The first line of the error, short enough for a notification. */ +export function failureHeadline(failure: RunFailure | undefined, exitCode: number): string { + const first = failure?.message?.split(/\r?\n/).find(line => line.trim() !== '')?.trim(); + const error = first ? (first.length > 160 ? `${first.slice(0, 157)}…` : first) : `exit code ${exitCode}`; + const where = failureWhere(failure); + return where ? `Run failed in ${where}: ${error}` : `Run failed: ${error}`; +} + +/** The chat task that asks the agent for the cause and a fix. */ +export function fixTask(opts: { + workflowName?: string; + sourceFile: string; + failure: RunFailure | undefined; + exitCode: number; + stderrTail: string; + runDir?: string; + canResume: boolean; +}): FixTask { + const workflow = opts.workflowName ? `\`${opts.workflowName}\` (${opts.sourceFile})` : opts.sourceFile; + const path = opts.failure?.entityInstancePath?.filter(part => part.trim() !== '') ?? []; + const lines = [`The last run of workflow ${workflow} failed. Find the cause and propose a fix.`, '']; + if (path.length > 0) { + const node = path[path.length - 1]; + lines.push( + path.length > 1 + ? `- Where: node \`${node}\`, inside the nested workflows ${path.slice(0, -1).map(p => `\`${p}\``).join(' › ')} (instance path \`${path.join('/')}\`).` + : `- Where: node \`${node}\`.` + ); + } else if (opts.failure?.entityInstanceName) { + lines.push(`- Where: node \`${opts.failure.entityInstanceName}\`.`); + } + lines.push(`- Error: ${opts.failure?.message?.trim() || `the run exited with code ${opts.exitCode}`}`); + if (opts.runDir) { + lines.push(`- Run directory: ${opts.runDir} (its run record, logs and intermediate files).`); + } + const tail = opts.stderrTail.trim(); + if (tail) { + lines.push('', 'The end of the run\'s error output:', '', '```text', tail, '```'); + } + lines.push( + '', + 'Explain the cause first, then propose the change, and say which file it goes in.' + + (opts.canResume + ? ' Once it is fixed, the run can be resumed from where it failed rather than started over.' + : '') + ); + const where = failureWhere(opts.failure); + return { + name: `Fix: ${where ?? opts.workflowName ?? 'failed run'}`, + mode: 'plan', + prompt: lines.join('\n') + }; +} diff --git a/packages/sidecar-toolkit/src/sidecar-diagram-profile.ts b/packages/sidecar-toolkit/src/sidecar-diagram-profile.ts index 8f4a447..cc95749 100644 --- a/packages/sidecar-toolkit/src/sidecar-diagram-profile.ts +++ b/packages/sidecar-toolkit/src/sidecar-diagram-profile.ts @@ -363,7 +363,9 @@ export function createSidecarDiagramProfile(input: SidecarProfileInput) { output: host.output, // The human port: a running agent's question goes to the platform // (the chat panel on the diagram); without it the driver prompts. - askUser: host.askUser ? (question, sourceUri) => host.askUser!(question, sourceUri) : undefined + askUser: host.askUser ? (question, sourceUri) => host.askUser!(question, sourceUri) : undefined, + // "Fix with AI" on a failed run: a task for the diagram's chat. + startChatTask: host.startChatTask ? (task, sourceUri) => host.startChatTask!(task, sourceUri) : undefined }); driver.registerCommands(context); host.useLiveOverlaySignatureSource({ diff --git a/packages/sidecar-toolkit/test/cli-run-driver-fix-with-ai.test.ts b/packages/sidecar-toolkit/test/cli-run-driver-fix-with-ai.test.ts new file mode 100644 index 0000000..9189693 --- /dev/null +++ b/packages/sidecar-toolkit/test/cli-run-driver-fix-with-ai.test.ts @@ -0,0 +1,132 @@ +/** + * A run that fails: the driver says where, and -- with a chat behind the + * diagram -- offers "Fix with AI", which starts a plan-mode chat session whose + * first message is the failure: its path, error, traceback and run directory. + */ +import { EventEmitter } from 'node:events'; +import * as fs from 'node:fs'; +import * as os from 'node:os'; +import * as path from 'node:path'; +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest'; +import * as vscode from 'vscode'; +import { resetRegisteredCommands } from './vscode-mock'; + +vi.mock('../src/process-control.js', () => ({ + spawnWorkflowProcess: (invocation: { cmd: string; args: string[]; cwd: string }) => { + const child: any = new EventEmitter(); + child.stdout = new EventEmitter(); + child.stderr = new EventEmitter(); + // The run, as the runtime would leave it: a log line and a failed record. + const outDir = invocation.args[invocation.args.indexOf('--out-dir') + 1]; + const runDir = path.join(outDir, 'r1'); + fs.mkdirSync(runDir, { recursive: true }); + const source = invocation.args[invocation.args.indexOf('run') + 1]; + const now = new Date().toISOString(); + fs.writeFileSync(path.join(outDir, 'run-log.jsonl'), JSON.stringify({ + runId: 'r1', workflowName: 'top', sourcePath: source, startedAt: now, finishedAt: now, outDir: runDir + }) + '\n'); + fs.writeFileSync(path.join(runDir, 'run.wf-run.json'), JSON.stringify({ + error: { message: 'ValueError: deep down', entityInstanceName: 'x', entityInstancePath: ['m', 'i', 'x'] } + })); + setTimeout(() => { + child.stderr.emit('data', Buffer.from('Traceback (most recent call last):\nValueError: deep down\n')); + child.emit('close', 1, null); + }, 0); + return child; + }, + requestWorkflowStop: () => true +})); +vi.mock('../src/run-event-stream-client.js', () => ({ + RunEventStreamClient: class { constructor(_o: unknown) {} start(): void {} stop(): void {} } +})); + +import { CliRunDriver, type CliRunDriverConfig } from '../src/index'; + +const RUN_CMD = 'test.fix.runWorkflow'; + +function makeDriver(startChatTask?: (task: any, uri: string) => Promise) { + const config: CliRunDriverConfig = { + settingsNamespace: 'wfLang', customEditorViewType: 'workflow.networkDiagram', + cliCommandSettingKey: 'wfpyCommand', cliCommandDefault: 'fake-wfpy', cliPythonModule: undefined, + runOutputDirSettingKey: 'runOutputDir', liveExecutionGlowSettingKey: 'liveExecutionGlow', + agentToolsSettingKey: 'agentTools', agentToolAuthSettingKey: 'agentToolAuth', + agentToolPolicySettingKey: 'agentToolPolicy', agentToolTimeoutMsSettingKey: 'agentToolTimeoutMs', + agentToolRegistrySettingKey: 'agentToolRegistry', agentMcpBridgeCmdSettingKey: 'agentMcpBridgeCmd', + runWorkflowCommandId: RUN_CMD, stopWorkflowCommandId: 'test.fix.stop', + agentToolConfigCommands: { set: 'test.fix.set', get: 'test.fix.get' }, + overrideState: { get: () => undefined, update: () => Promise.resolve() }, + cliResumeArgs: (dir, step) => ['--resume-from', dir, '--at-step', String(step)], + elicitSocket: false + }; + const host: any = { + overlay: { emitEvents: () => {} }, + requestRefresh: () => {}, + output: { show: vi.fn(), append: () => {}, appendLine: () => {} }, + ...(startChatTask ? { startChatTask } : {}) + }; + const driver = new CliRunDriver(config, host); + driver.registerCommands({ subscriptions: [] } as any); + return { driver, host }; +} + +let original: any; +let shown: Array<{ message: string; actions: string[] }>; +let answer: string | undefined; + +beforeEach(() => { + resetRegisteredCommands(); + shown = []; + original = (vscode.window as any).showErrorMessage; + (vscode.window as any).showErrorMessage = async (message: string, ...actions: string[]) => { + shown.push({ message, actions }); + return answer; + }; +}); +afterEach(() => { + (vscode.window as any).showErrorMessage = original; +}); + +async function failRun(): Promise { + const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'fix-with-ai-')); + const file = path.join(dir, 'top.py'); + fs.writeFileSync(file, '# top\n'); + await vscode.commands.executeCommand(RUN_CMD, { sourceUri: `file://${file}`, workflowName: 'top' }); + // The offer is made without holding the run command up. + await new Promise(resolve => setTimeout(resolve, 20)); +} + +describe('a run that fails', () => { + it('says where it failed, and offers to fix it with the chat', async () => { + answer = undefined; + makeDriver(async () => true); + await failRun(); + + expect(shown).toHaveLength(1); + expect(shown[0].message).toBe('Run failed in x (in m › i): ValueError: deep down'); + expect(shown[0].actions).toEqual(['Fix with AI', 'Show Output']); + }); + + it('starts a plan-mode chat task with the failure when the offer is taken', async () => { + answer = 'Fix with AI'; + const startChatTask = vi.fn(async () => true); + makeDriver(startChatTask); + await failRun(); + + expect(startChatTask).toHaveBeenCalledTimes(1); + const [task, uri] = startChatTask.mock.calls[0] as any[]; + expect(uri).toMatch(/top\.py$/); + expect(task.mode).toBe('plan'); + expect(task.name).toBe('Fix: x (in m › i)'); + expect(task.prompt).toContain('instance path `m/i/x`'); + expect(task.prompt).toContain('Traceback (most recent call last):'); + expect(task.prompt).toContain('resumed from where it failed'); + }); + + it('offers no fix without a chat behind the diagram', async () => { + answer = undefined; + makeDriver(undefined); + await failRun(); + + expect(shown[0].actions).toEqual(['Show Output']); + }); +}); diff --git a/packages/sidecar-toolkit/test/run-failure.test.ts b/packages/sidecar-toolkit/test/run-failure.test.ts new file mode 100644 index 0000000..4e1165f --- /dev/null +++ b/packages/sidecar-toolkit/test/run-failure.test.ts @@ -0,0 +1,63 @@ +/** + * A failed run told to the person (one line) and to the agent asked to fix it + * (the error, where it happened, the traceback, where the run's files are). + */ +import { describe, expect, it } from 'vitest'; +import { appendTail, failureHeadline, failureWhere, fixTask } from '../src/run-failure'; + +const deep = { message: 'ValueError: deep down', entityInstancePath: ['m', 'i', 'x'], entityInstanceName: 'x' }; + +describe('where a run failed', () => { + it('names the node and the nested workflows it is in', () => { + expect(failureWhere(deep)).toBe('x (in m › i)'); + expect(failureWhere({ entityInstancePath: ['x'] })).toBe('x'); + expect(failureWhere({ entityInstanceName: 'y' })).toBe('y'); + expect(failureWhere(undefined)).toBeUndefined(); + }); + + it('fits in a notification: where, and the first line of the error', () => { + expect(failureHeadline(deep, 1)).toBe('Run failed in x (in m › i): ValueError: deep down'); + expect(failureHeadline(undefined, 3)).toBe('Run failed: exit code 3'); + expect(failureHeadline({ message: 'x'.repeat(400) }, 1).length).toBeLessThan(200); + }); +}); + +describe('the end of the run’s error output', () => { + it('keeps the end, where the traceback is', () => { + expect(appendTail('abc', 'def', 4)).toBe('cdef'); + expect(appendTail('', 'Traceback', 100)).toBe('Traceback'); + }); +}); + +describe('the task the chat is asked to start', () => { + const task = fixTask({ + workflowName: 'top', + sourceFile: '/w/top.py', + failure: deep, + exitCode: 1, + stderrTail: 'Traceback (most recent call last):\n ...\nValueError: deep down\n', + runDir: '/w/wf-out/r1', + canResume: true + }); + + it('is a plan-mode session named for where it failed', () => { + expect(task.mode).toBe('plan'); + expect(task.name).toBe('Fix: x (in m › i)'); + }); + + it('gives the agent what it needs to find the cause', () => { + expect(task.prompt).toContain('workflow `top` (/w/top.py) failed'); + expect(task.prompt).toContain('node `x`, inside the nested workflows `m` › `i` (instance path `m/i/x`)'); + expect(task.prompt).toContain('- Error: ValueError: deep down'); + expect(task.prompt).toContain('/w/wf-out/r1'); + expect(task.prompt).toContain('```text\nTraceback (most recent call last):'); + expect(task.prompt).toContain('resumed from where it failed'); + }); + + it('says nothing of resuming when the run cannot be resumed', () => { + const plain = fixTask({ sourceFile: '/w/top.py', failure: undefined, exitCode: 2, stderrTail: '', canResume: false }); + expect(plain.prompt).toContain('- Error: the run exited with code 2'); + expect(plain.prompt).not.toContain('resumed'); + expect(plain.prompt).not.toContain('```text'); + }); +});