From 6eea20c7fdef21814a07e93bf15f49bb4d193b32 Mon Sep 17 00:00:00 2001 From: Endri Bezati Date: Fri, 2 Oct 2026 16:26:28 +0200 Subject: [PATCH] chat: once its fix is in, the agent offers to resume the failed run The fix task now tells the agent, when the runtime can resume, to call the `resume_failed_run` chat tool once its fix is applied. The tool asks the person (Resume / Not now) and, on Resume, reruns the workflow from the failed run's directory at its last completed step -- where it failed -- so what ran before is replayed from the journal rather than run again. The driver remembers the last failed run per workflow file (forgotten once that workflow runs clean); the tool falls back to the last failure, so a chat session opened on a nested workflow's file still finds it. Claude-Session: https://claude.ai/code/session_015VK7fH1c4aKbexnq2QcuKU --- .../sidecar-toolkit/src/cli-run-driver.ts | 71 ++++++- packages/sidecar-toolkit/src/index.ts | 1 + .../sidecar-toolkit/src/resume-at-step.ts | 6 + packages/sidecar-toolkit/src/run-failure.ts | 18 +- .../src/sidecar-diagram-profile.ts | 34 +++- .../test/cli-run-driver-fix-with-ai.test.ts | 2 +- .../test/cli-run-driver.test.ts | 6 +- .../test/resume-failed-run.test.ts | 180 ++++++++++++++++++ .../sidecar-toolkit/test/run-failure.test.ts | 5 +- 9 files changed, 308 insertions(+), 15 deletions(-) create mode 100644 packages/sidecar-toolkit/test/resume-failed-run.test.ts diff --git a/packages/sidecar-toolkit/src/cli-run-driver.ts b/packages/sidecar-toolkit/src/cli-run-driver.ts index f295ffc..752a8c4 100644 --- a/packages/sidecar-toolkit/src/cli-run-driver.ts +++ b/packages/sidecar-toolkit/src/cli-run-driver.ts @@ -37,7 +37,7 @@ import * as fs from 'node:fs/promises'; import * as path from 'node:path'; import type { ExecutionOverlaySink } from '@dialogram/shared'; import { resumeStepFor } from './resume-at-step.js'; -import { appendTail, failureHeadline, fixTask, type FixTask, type RunFailure } from './run-failure.js'; +import { appendTail, failureHeadline, failureWhere, fixTask, type FixTask, type RunFailure } from './run-failure.js'; import { requestWorkflowStop, spawnWorkflowProcess } from './process-control.js'; import { RunEventStreamClient, type RunStreamEvent } from './run-event-stream-client.js'; @@ -51,8 +51,15 @@ type RunWorkflowArgs = { atStep?: number; /** With `resumeFrom`: resume at the step before this node's last firing. */ resumeAtActor?: string; + /** With `resumeFrom`: resume at its last completed step -- where it failed. */ + resumeAtLastStep?: boolean; }; +/** The command the chat's `resume_failed_run` tool runs (see `CliRunDriver.resumeFailedRun`). */ +export function resumeFailedRunCommandId(runWorkflowCommandId: string): string { + return `${runWorkflowCommandId}.resumeFailed`; +} + /** * Per-entity agent tool-calling settings, keyed by entity instance name. * Persisted (via {@link CliRunDriverConfig.overrideState}) so it survives VS @@ -177,6 +184,9 @@ export interface LiveOverlaySubscription { export class CliRunDriver { private activeRun: ActiveWorkflowRun | undefined; + /** The last failed run of each workflow file (by fsPath), until it runs clean. */ + private readonly failedRuns = new Map(); + private lastFailedFile: string | undefined; /** Per-entity agent-tool overrides; mutated by the config commands and read * when building run args. Persisted through `config.overrideState`. */ private readonly agentToolOverrides = new Map(); @@ -211,6 +221,15 @@ export class CliRunDriver { vscode.commands.registerCommand(this.config.stopWorkflowCommandId, async () => this.stopWorkflow()) ); + // Command: resume the last failed run, at the user's word -- what the + // chat's `resume_failed_run` tool runs once the agent's fix is in. + context.subscriptions.push( + vscode.commands.registerCommand( + resumeFailedRunCommandId(this.config.runWorkflowCommandId), + async (args?: { file?: string }) => this.resumeFailedRun(args?.file) + ) + ); + // Command: Run workflow (invokes the configured runtime CLI) context.subscriptions.push( vscode.commands.registerCommand(this.config.runWorkflowCommandId, async (args?: RunWorkflowArgs) => this.runWorkflow(args)) @@ -728,6 +747,48 @@ export class CliRunDriver { }); } + /** + * Resume the last failed run of `file`'s workflow from where it failed, if + * the person agrees. Returns what happened, in words, for the agent that + * asked: it proposed this after fixing the failure. + * + * `file` is the chat session's file. Navigating a hierarchy in place that + * can be a nested workflow's file, not the root that ran; then the last run + * that failed is the one meant. + */ + async resumeFailedRun(file?: string): Promise { + if (!this.config.cliResumeArgs) { + return 'This runtime cannot resume a run; ask the user to run the workflow again.'; + } + const failed = (file ? this.failedRuns.get(file) : undefined) + ?? (this.lastFailedFile ? this.failedRuns.get(this.lastFailedFile) : undefined); + if (!failed) { + return 'There is no failed run to resume: the workflow has run clean since, or has not failed in this session.'; + } + if (this.activeRun) { + return 'A run is already in progress; the failed run can be resumed once it ends.'; + } + const RESUME = 'Resume'; + const where = failureWhere(failed.failure); + const choice = await vscode.window.showInformationMessage( + `The chat's fix is in. Resume the run${failed.workflowName ? ` of ${failed.workflowName}` : ''} from where it failed${where ? ` (${where})` : ''}? What ran before is replayed, not run again.`, + RESUME, + 'Not now' + ); + if (choice !== RESUME) { + return 'The user chose not to resume the run now.'; + } + // Not awaited: a run can outlast a tool call. Its outcome shows on the + // diagram, and a new failure is offered for fixing again. + void this.runWorkflow({ + sourceUri: failed.sourceUri.toString(), + workflowName: failed.workflowName, + resumeFrom: failed.runDir, + resumeAtLastStep: true + }); + return 'The run is resuming from where it failed. Its result will show on the diagram; if it fails again, the user is offered a new fix.'; + } + /** The error a run recorded, if its record has one. */ private async readRunFailure(runDir: string): Promise { try { @@ -862,6 +923,7 @@ export class CliRunDriver { } const resolved = await resumeStepFor(resumeFrom, { atStep: args?.atStep, + atLastStep: args?.resumeAtLastStep === true, actor: typeof args?.resumeAtActor === 'string' && args.resumeAtActor.trim() ? args.resumeAtActor.trim() : undefined }); if ('error' in resolved) { @@ -1123,12 +1185,19 @@ export class CliRunDriver { workflowName }); const failure = runDir ? await this.readRunFailure(runDir) : undefined; + if (runDir) { + // What the agent's fix resumes (`resume_failed_run`). + this.failedRuns.set(sourceUri.fsPath, { sourceUri, workflowName, runDir, failure }); + this.lastFailedFile = sourceUri.fsPath; + } void this.reportFailure({ sourceUri, workflowName, failure, exitCode, stderrTail, runDir, output }); return { error: failureHeadline(failure, exitCode) }; } // Refresh diagram so overlay decorations can pick up the new run artifacts. await refreshDiagram(true); + // It ran clean: nothing left to resume for it. + this.failedRuns.delete(sourceUri.fsPath); const queueTracePath = await this.findLatestQueueTracePath({ baseOutDir: runOutDir, diff --git a/packages/sidecar-toolkit/src/index.ts b/packages/sidecar-toolkit/src/index.ts index f786dd8..2ba3120 100644 --- a/packages/sidecar-toolkit/src/index.ts +++ b/packages/sidecar-toolkit/src/index.ts @@ -50,6 +50,7 @@ export { export { CliRunDriver, + resumeFailedRunCommandId, type CliRunDriverConfig, type CliRunDriverHost, type AgentToolEntitySettings, diff --git a/packages/sidecar-toolkit/src/resume-at-step.ts b/packages/sidecar-toolkit/src/resume-at-step.ts index 67d8073..f968e08 100644 --- a/packages/sidecar-toolkit/src/resume-at-step.ts +++ b/packages/sidecar-toolkit/src/resume-at-step.ts @@ -7,6 +7,8 @@ export interface ResumeRequest { atStep?: number; /** A node: resume at the step before its last firing, so that firing happens again. */ actor?: string; + /** The run's last completed step: where it stopped, every firing before replayed. */ + atLastStep?: boolean; } export type ResumeStep = { step: number } | { error: string }; @@ -34,6 +36,10 @@ export async function resumeStepFor(runDir: string, request: ResumeRequest): Pro return { error: `The run in ${runDir} was made by a runtime that cannot resume at a step.` }; } + if (request.atLastStep) { + return { step: steps.length }; + } + if (request.actor !== undefined) { // The last step at which it fired; resuming at the step before runs it again. for (let i = steps.length - 1; i >= 0; i--) { diff --git a/packages/sidecar-toolkit/src/run-failure.ts b/packages/sidecar-toolkit/src/run-failure.ts index 05f18e6..5d0e5c9 100644 --- a/packages/sidecar-toolkit/src/run-failure.ts +++ b/packages/sidecar-toolkit/src/run-failure.ts @@ -23,6 +23,9 @@ export interface FixTask { prompt: string; } +/** The chat tool that resumes the failed run once a fix is in. */ +export const RESUME_TOOL = 'resume_failed_run'; + /** At most this much of the run's stderr goes to the agent: the end, where the traceback is. */ export const STDERR_TAIL_CHARS = 8000; @@ -81,13 +84,14 @@ export function fixTask(opts: { if (tail) { lines.push('', 'The end of the run\'s error output:', '', '```text', tail, '```'); } - lines.push( - '', - 'Explain the cause first, then propose the change, and say which file it goes in.' - + (opts.canResume - ? ' Once it is fixed, the run can be resumed from where it failed rather than started over.' - : '') - ); + lines.push('', 'Explain the cause first, then propose the change, and say which file it goes in.'); + if (opts.canResume) { + lines.push( + '', + `Once the fix is applied, call the \`${RESUME_TOOL}\` tool: it offers the user to resume the run from where it failed, ` + + 'with what ran before replayed rather than run again.' + ); + } const where = failureWhere(opts.failure); return { name: `Fix: ${where ?? opts.workflowName ?? 'failed run'}`, diff --git a/packages/sidecar-toolkit/src/sidecar-diagram-profile.ts b/packages/sidecar-toolkit/src/sidecar-diagram-profile.ts index cc95749..6e885d8 100644 --- a/packages/sidecar-toolkit/src/sidecar-diagram-profile.ts +++ b/packages/sidecar-toolkit/src/sidecar-diagram-profile.ts @@ -14,7 +14,8 @@ import type { EntityPaletteItemSpec, NodeFamilySpec } from '@dialogram/shared'; import * as vscode from 'vscode'; import { invokeSidecarOp } from './sidecar-graph-export.js'; -import { createRegistryChatTools } from './registry-tools.js'; +import { createRegistryChatTools, type RegistryChatTool } from './registry-tools.js'; +import { RESUME_TOOL } from './run-failure.js'; import { getCliInvocation, getSidecarCommand, type SidecarRuntimeConfig, @@ -32,6 +33,7 @@ import { registerSidecarCommands } from './sidecar-commands.js'; import { registerNewSourceFileCommand } from './new-source-file-command.js'; import { CliRunDriver, + resumeFailedRunCommandId, type CliRunDriverConfig, type CliRunDriverHost, type AgentToolEntitySettings @@ -326,6 +328,12 @@ export function createSidecarDiagramProfile(input: SidecarProfileInput) { args ) }); + // A product that can resume runs gives the chat a way to offer it: the + // agent that fixed a failed run proposes resuming it, and the person says + // yes or no. + if (input.cliResumeArgs) { + chatTools.push(resumeFailedRunTool(input.commands.runWorkflow)); + } const runDriver = (context: vscode.ExtensionContext, host: RunHost): vscode.Disposable => { const config: CliRunDriverConfig = { @@ -462,3 +470,27 @@ export function createSidecarDiagramProfile(input: SidecarProfileInput) { }); return profile; } + + +/** + * The chat tool that resumes a failed run once the agent's fix is in. It asks + * the person first (the run driver's `resumeFailedRun`); the agent learns the + * answer, and the run's outcome shows on the diagram. + */ +export function resumeFailedRunTool(runWorkflowCommandId: string): RegistryChatTool { + return { + name: RESUME_TOOL, + description: + 'After fixing the cause of a failed workflow run, offer the user to resume that run from where it failed. ' + + 'The user is asked to confirm; what ran before the failure is replayed, not run again. ' + + 'Call it once the fix is applied to the files -- not before, and not to start a new run.', + inputSchema: { type: 'object', properties: {} }, + handler: async (file) => { + const result = await vscode.commands.executeCommand( + resumeFailedRunCommandId(runWorkflowCommandId), + { file } + ); + return typeof result === 'string' ? result : 'The resume could not be offered.'; + } + }; +} diff --git a/packages/sidecar-toolkit/test/cli-run-driver-fix-with-ai.test.ts b/packages/sidecar-toolkit/test/cli-run-driver-fix-with-ai.test.ts index 9189693..47f4e6e 100644 --- a/packages/sidecar-toolkit/test/cli-run-driver-fix-with-ai.test.ts +++ b/packages/sidecar-toolkit/test/cli-run-driver-fix-with-ai.test.ts @@ -119,7 +119,7 @@ describe('a run that fails', () => { expect(task.name).toBe('Fix: x (in m › i)'); expect(task.prompt).toContain('instance path `m/i/x`'); expect(task.prompt).toContain('Traceback (most recent call last):'); - expect(task.prompt).toContain('resumed from where it failed'); + expect(task.prompt).toContain('call the `resume_failed_run` tool'); }); it('offers no fix without a chat behind the diagram', async () => { diff --git a/packages/sidecar-toolkit/test/cli-run-driver.test.ts b/packages/sidecar-toolkit/test/cli-run-driver.test.ts index 5ad764d..32edba7 100644 --- a/packages/sidecar-toolkit/test/cli-run-driver.test.ts +++ b/packages/sidecar-toolkit/test/cli-run-driver.test.ts @@ -106,13 +106,13 @@ describe('CliRunDriver agent-tool overrides', () => { vi.restoreAllMocks(); }); - it('registers both config commands alongside run/stop', () => { + it('registers both config commands alongside run/stop/resume', () => { const overrideState = makeOverrideState(); const driver = new CliRunDriver(makeConfig(overrideState), makeHost() as any); const context = makeContext(); driver.registerCommands(context); - // run + stop + set + get - expect(context.subscriptions.length).toBe(4); + // run + stop + resume-failed + set + get + expect(context.subscriptions.length).toBe(5); }); it('set mutates + persists via the accessor, get reads current', async () => { diff --git a/packages/sidecar-toolkit/test/resume-failed-run.test.ts b/packages/sidecar-toolkit/test/resume-failed-run.test.ts new file mode 100644 index 0000000..5d31ac1 --- /dev/null +++ b/packages/sidecar-toolkit/test/resume-failed-run.test.ts @@ -0,0 +1,180 @@ +/** + * The agent that fixed a failed run proposes resuming it: its chat tool runs + * the driver's command, which asks the person, then resumes the run from its + * last completed step -- where it failed -- with what ran before replayed. + */ +import { EventEmitter } from 'node:events'; +import * as fs from 'node:fs'; +import * as os from 'node:os'; +import * as path from 'node:path'; +import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest'; +import * as vscode from 'vscode'; +import { resetRegisteredCommands } from './vscode-mock'; + +/** Each spawn: the first fails (leaving a record and a three-step trace), later ones succeed. */ +const spawns: string[][] = []; +vi.mock('../src/process-control.js', () => ({ + spawnWorkflowProcess: (invocation: { args: string[] }) => { + spawns.push(invocation.args); + const child: any = new EventEmitter(); + child.stdout = new EventEmitter(); + child.stderr = new EventEmitter(); + const fails = spawns.length === 1; + if (fails) { + const outDir = invocation.args[invocation.args.indexOf('--out-dir') + 1]; + const runDir = path.join(outDir, 'r1'); + fs.mkdirSync(runDir, { recursive: true }); + const now = new Date().toISOString(); + fs.writeFileSync(path.join(outDir, 'run-log.jsonl'), JSON.stringify({ + runId: 'r1', workflowName: 'top', sourcePath: invocation.args[invocation.args.indexOf('run') + 1], + startedAt: now, finishedAt: now, outDir: runDir + }) + '\n'); + fs.writeFileSync(path.join(runDir, 'run.wf-run.json'), JSON.stringify({ + error: { message: 'ValueError: no', entityInstancePath: ['m', 'x'] } + })); + fs.writeFileSync(path.join(runDir, 'run.wf-queues.json'), JSON.stringify({ + version: 1, + steps: [1, 2, 3].map(step => ({ step, actorInstanceName: 'a', journalSeq: step })) + })); + } + setTimeout(() => child.emit('close', fails ? 1 : 0, null), 0); + return child; + }, + requestWorkflowStop: () => true +})); +vi.mock('../src/run-event-stream-client.js', () => ({ + RunEventStreamClient: class { constructor(_o: unknown) {} start(): void {} stop(): void {} } +})); + +import { CliRunDriver, resumeFailedRunCommandId, type CliRunDriverConfig } from '../src/index'; +import { resumeFailedRunTool } from '../src/sidecar-diagram-profile'; + +const RUN_CMD = 'test.resume.runWorkflow'; + +function makeDriver(canResume = true) { + const config: CliRunDriverConfig = { + settingsNamespace: 'wfLang', customEditorViewType: 'workflow.networkDiagram', + cliCommandSettingKey: 'wfpyCommand', cliCommandDefault: 'fake-wfpy', cliPythonModule: undefined, + runOutputDirSettingKey: 'runOutputDir', liveExecutionGlowSettingKey: 'liveExecutionGlow', + agentToolsSettingKey: 'agentTools', agentToolAuthSettingKey: 'agentToolAuth', + agentToolPolicySettingKey: 'agentToolPolicy', agentToolTimeoutMsSettingKey: 'agentToolTimeoutMs', + agentToolRegistrySettingKey: 'agentToolRegistry', agentMcpBridgeCmdSettingKey: 'agentMcpBridgeCmd', + runWorkflowCommandId: RUN_CMD, stopWorkflowCommandId: 'test.resume.stop', + agentToolConfigCommands: { set: 'test.resume.set', get: 'test.resume.get' }, + overrideState: { get: () => undefined, update: () => Promise.resolve() }, + ...(canResume ? { cliResumeArgs: (dir: string, step: number) => ['--resume-from', dir, '--at-step', String(step)] } : {}), + elicitSocket: false + }; + const driver = new CliRunDriver(config, { + overlay: { emitEvents: () => {} }, + requestRefresh: () => {}, + output: { show: () => {}, append: () => {}, appendLine: () => {} } + } as any); + driver.registerCommands({ subscriptions: [] } as any); + return driver; +} + +let answer: string | undefined; +let asked: string[]; +const originals: any = {}; + +beforeEach(() => { + resetRegisteredCommands(); + spawns.length = 0; + asked = []; + originals.info = (vscode.window as any).showInformationMessage; + originals.error = (vscode.window as any).showErrorMessage; + (vscode.window as any).showInformationMessage = async (message: string) => { + asked.push(message); + return message.startsWith("The chat's fix") ? answer : undefined; + }; + (vscode.window as any).showErrorMessage = async () => undefined; +}); +afterEach(() => { + (vscode.window as any).showInformationMessage = originals.info; + (vscode.window as any).showErrorMessage = originals.error; +}); + +async function failRun(): Promise { + const dir = fs.mkdtempSync(path.join(os.tmpdir(), 'resume-failed-')); + const file = path.join(dir, 'top.py'); + fs.writeFileSync(file, '# top\n'); + await vscode.commands.executeCommand(RUN_CMD, { sourceUri: `file://${file}`, workflowName: 'top' }); + return file; +} + +const settle = () => new Promise(resolve => setTimeout(resolve, 30)); + +describe('resuming a failed run when the agent proposes it', () => { + it('asks the person, then resumes from the last completed step', async () => { + makeDriver(); + const file = await failRun(); + answer = 'Resume'; + + const result = await vscode.commands.executeCommand(resumeFailedRunCommandId(RUN_CMD), { file }); + await settle(); + + expect(asked.some(q => q.includes('Resume the run of top from where it failed (x (in m))'))).toBe(true); + expect(result).toContain('resuming from where it failed'); + const resumed = spawns[1]; + expect(resumed.slice(resumed.indexOf('--resume-from'))).toEqual( + expect.arrayContaining(['--resume-from', '--at-step', '3']) + ); + }); + + it('does not resume when the person says not now', async () => { + makeDriver(); + const file = await failRun(); + answer = 'Not now'; + + const result = await vscode.commands.executeCommand(resumeFailedRunCommandId(RUN_CMD), { file }); + await settle(); + + expect(result).toContain('chose not to resume'); + expect(spawns).toHaveLength(1); + }); + + it('finds the failed run from a nested workflow’s file too', async () => { + makeDriver(); + await failRun(); + answer = 'Resume'; + + await vscode.commands.executeCommand(resumeFailedRunCommandId(RUN_CMD), { file: '/w/layers/nested.py' }); + await settle(); + + expect(spawns).toHaveLength(2); + }); + + it('has nothing to resume once the workflow has run clean', async () => { + makeDriver(); + const file = await failRun(); + answer = 'Resume'; + await vscode.commands.executeCommand(resumeFailedRunCommandId(RUN_CMD), { file }); + await settle(); // the resumed run succeeds + + const again = await vscode.commands.executeCommand(resumeFailedRunCommandId(RUN_CMD), { file }); + expect(again).toContain('no failed run to resume'); + }); + + it('says so when the runtime cannot resume', async () => { + makeDriver(false); + const file = await failRun(); + const result = await vscode.commands.executeCommand(resumeFailedRunCommandId(RUN_CMD), { file }); + expect(result).toContain('cannot resume'); + }); +}); + +describe('the chat tool', () => { + it('runs the driver’s command with the session’s file and returns its answer', async () => { + const tool = resumeFailedRunTool(RUN_CMD); + const calls: unknown[] = []; + vscode.commands.registerCommand(resumeFailedRunCommandId(RUN_CMD), async (args: unknown) => { + calls.push(args); + return 'The run is resuming from where it failed.'; + }); + + expect(tool.name).toBe('resume_failed_run'); + expect(await tool.handler('/w/top.py', {})).toBe('The run is resuming from where it failed.'); + expect(calls).toEqual([{ file: '/w/top.py' }]); + }); +}); diff --git a/packages/sidecar-toolkit/test/run-failure.test.ts b/packages/sidecar-toolkit/test/run-failure.test.ts index 4e1165f..7764028 100644 --- a/packages/sidecar-toolkit/test/run-failure.test.ts +++ b/packages/sidecar-toolkit/test/run-failure.test.ts @@ -51,13 +51,14 @@ describe('the task the chat is asked to start', () => { expect(task.prompt).toContain('- Error: ValueError: deep down'); expect(task.prompt).toContain('/w/wf-out/r1'); expect(task.prompt).toContain('```text\nTraceback (most recent call last):'); - expect(task.prompt).toContain('resumed from where it failed'); + expect(task.prompt).toContain('Once the fix is applied, call the `resume_failed_run` tool'); + expect(task.prompt).toContain('resume the run from where it failed'); }); it('says nothing of resuming when the run cannot be resumed', () => { const plain = fixTask({ sourceFile: '/w/top.py', failure: undefined, exitCode: 2, stderrTail: '', canResume: false }); expect(plain.prompt).toContain('- Error: the run exited with code 2'); - expect(plain.prompt).not.toContain('resumed'); + expect(plain.prompt).not.toContain('resume_failed_run'); expect(plain.prompt).not.toContain('```text'); }); });