diff --git a/docs/repo-config.md b/docs/repo-config.md index 3cad17a..5c3c2ea 100644 --- a/docs/repo-config.md +++ b/docs/repo-config.md @@ -54,7 +54,7 @@ Preamble passed to [Open Code Review](https://alibaba.github.io/open-code-review ## `.localagent-box/setup.sh` — Setup Script -A committed shell script the host runs **instead of** any `environment.json` source before the agent starts. The host runs it from the workspace root as `bash .localagent-box/setup.sh`; anything it exits non-zero fails the agent start the same way `setup.failOnError: true` does (a timeout also fails the agent). The script must be committed in the repo — like `.localagent-box/environment.json`, it is read straight from the fresh clone before agent commits are ignored. +A committed shell script the host runs **instead of** any `environment.json` source before the agent starts. The host runs it from the workspace root as `bash .localagent-box/setup.sh`; a non-zero exit is recorded as a failed bootstrap and the error is fed to the agent's first prompt so it can attempt a fix — unless `setup.failOnError: true` is set in `environment.json`, in which case the agent start fails (a timeout behaves the same way). The script must be committed in the repo — like `.localagent-box/environment.json`, it is read straight from the fresh clone before agent commits are ignored. A committed `setup.sh` **always wins** over `setup.command`, `profiles`, and lockfile auto-detect — commit an `environment.json` without touching the script and the script still runs. @@ -103,15 +103,18 @@ When present, declares the host-run setup step. #### `setup.command` *(string, required)* -The shell command run in the workspace root before the agent starts. Workspaces are fresh-cloned before every agent run; when the [dependency cache](#dependency-cache-`cacheKey`) is enabled the host restores the cached dependencies before running this command. A non-zero exit fails the agent start by default (see `setup.failOnError`). +The shell command run in the workspace root before the agent starts. Workspaces are fresh-cloned before every agent run; when the [dependency cache](#dependency-cache-`cacheKey`) is enabled the host restores the cached dependencies before running this command. A non-zero exit is recorded as a failed bootstrap and fed to the agent (see `setup.failOnError`). #### `setup.timeoutMs` *(number, optional)* Timeout in milliseconds before the shell is killed. Must be a positive integer no greater than `1800000` (30 min). Defaults to `600000` (10 min). -#### `setup.failOnError` *(boolean, optional, default true)* +#### `setup.failOnError` *(boolean, optional, default false)* -When `true` (default), a non-zero exit code (or timeout) from `setup.command` **fails the whole agent** — OpenCode never starts. Set to `false` to log the failure and continue anyway. +Controls what happens when the setup step (or `verifyCommand`) fails: + +- `false` (default) — **non-blocking**: the failure is recorded on the agent's `bootstrap` record (`status: 'failed'` with the exit code and output tail) and the run continues; the host prepends a failure block to the agent's first prompt carrying the command, exit code, and output tail so the agent can diagnose and attempt to fix the workspace environment itself. +- `true` — **fail-hard**: a non-zero exit code (or timeout) fails the whole agent — OpenCode never starts. #### `setup.runOnModes` *(array, optional)* @@ -119,7 +122,7 @@ Agent modes for which the setup runs: any of `batch`, `interactive`, `loop`, `re ### `verifyCommand` *(string, optional)* -Post-setup smoke test run **after** a successful setup command (and before the agent starts). A non-zero exit **always fails the bootstrap** — there is no `failOnError` opt-out for verify, so a broken environment never reaches the agent even when the agent's own checks would be more forgiving. A success records `verifyCommand` and its exit code on the agent's `bootstrap` record. +Post-setup smoke test run **after** a successful setup command (and before the agent starts). A non-zero exit is handled exactly like a failed setup command under `setup.failOnError`: by default it is recorded (`verifyCommand` and its exit code land on the agent's `bootstrap` record), the run continues, and the error is fed to the agent's prompt for a fix attempt; with `setup.failOnError: true` the agent start fails. ### `verifyTimeoutMs` *(number, optional)* @@ -127,7 +130,7 @@ Timeout in milliseconds for `verifyCommand`. Same constraints as `setup.timeoutM ### Post-setup summary -After the setup command (and, if set, the `verifyCommand`) finishes successfully, the host prepends a short workspace-ready block to the agent's first prompt so the model doesn't waste turns rediscovering how to build or test the repo. The block is only shown when setup **completed** — a skipped or failed bootstrap leaves the prompt untouched — and reports the resolved command, the runtime profiles, the setup duration, and any dependency-cache hit. +After the setup command (and, if set, the `verifyCommand`) finishes, the host prepends a short bootstrap block to the agent's first prompt so the model doesn't waste turns rediscovering how to build or test the repo. A **successful** bootstrap reports the resolved command, the runtime profiles, the setup duration, and any dependency-cache hit. A **failed** bootstrap (default `setup.failOnError: false`) reports the failing command, the exit code, the error, and the output tail, instructing the agent to diagnose and fix the environment as part of the task. Only a skipped bootstrap (nothing to run) leaves the prompt untouched. ### `profiles` *(array, optional)* diff --git a/src/domains/agents/worker/environment-detect.ts b/src/domains/agents/worker/environment-detect.ts index c827755..5c31688 100644 --- a/src/domains/agents/worker/environment-detect.ts +++ b/src/domains/agents/worker/environment-detect.ts @@ -31,12 +31,13 @@ export const setupScriptCommand = `bash ${setupScriptRelative}`; * Resolution order (P4-T1): the script wins over `environment.json` * `setup.command`, `profiles`, and lockfile auto-detect. * - * For a bare script (no `environment.json` in the repo), the caller runs it - * with fail-hard semantics: a non-zero exit always fails the agent start - * (the `setup.failOnError: false` opt-out and `setup.verifyCommand` are - * `environment.json` settings that do not exist here), and `timeoutMs`, - * `runOnModes`, and `cacheKey` have no effect. The global - * `globalSetupTimeoutMs` and the lockfile-derived dep cache still apply. + * For a bare script (no `environment.json` in the repo), failures follow the + * default non-blocking path: they are recorded on the agent record and fed to + * the agent's prompt for a fix attempt (`setup.failOnError: true` / + * `setup.verifyCommand` are `environment.json` settings that do not exist + * here), and `timeoutMs`, `runOnModes`, and `cacheKey` have no effect. The + * global `globalSetupTimeoutMs` and the lockfile-derived dep cache still + * apply. */ export function detectSetupScript(workspaceDir: string): string | null { return fs.existsSync(path.join(workspaceDir, setupScriptRelative)) diff --git a/src/domains/agents/worker/loop-run-flow.ts b/src/domains/agents/worker/loop-run-flow.ts index 994b521..b499468 100644 --- a/src/domains/agents/worker/loop-run-flow.ts +++ b/src/domains/agents/worker/loop-run-flow.ts @@ -88,7 +88,9 @@ interface RunLoopStepParams { loopState: AgentLoopState; /** * P4-T5: host-run bootstrap state from the agent record. Injected into the - * first INITIAL_PLAN kickoff prompt when the bootstrap completed. + * first INITIAL_PLAN kickoff prompt when the bootstrap completed — and, on + * a failed bootstrap, feeds the failure block so the agent can attempt a + * fix. */ bootstrap?: AgentBootstrapState | null; /** Agent data directory for loop handoff state (plan + loop-state.json). */ @@ -181,8 +183,8 @@ async function runLoopStep(params: RunLoopStepParams): Promise<{ const conversationParts: (string | null)[] = [interpolated]; // P4-T5: the INITIAL_PLAN kickoff (fresh session, step 0) carries the - // host-run bootstrap summary (completed only) so the model doesn't spend a - // turn rediscovering the environment. + // host-run bootstrap summary — completed, or failed with the error tail so + // the agent can attempt a fix. if (stepIndex === 0) { conversationParts.push(formatBootstrapSummaryBlock(params.bootstrap)); } diff --git a/src/domains/agents/worker/workspace-bootstrap.test.ts b/src/domains/agents/worker/workspace-bootstrap.test.ts index e711a53..5d432eb 100644 --- a/src/domains/agents/worker/workspace-bootstrap.test.ts +++ b/src/domains/agents/worker/workspace-bootstrap.test.ts @@ -333,10 +333,33 @@ describe('runWorkspaceBootstrap', () => { assert.match(log, /Running workspace bootstrap/); }); - it('treats a failing setup.sh like any other failed setup and throws', async () => { + it('treats a failing setup.sh like any other failed setup — recorded, non-blocking by default', async () => { const h = makeHarness(); writeSetupScript(h.workspaceDir, '#!/usr/bin/env bash\nexit 1\n'); + const state = await runBootstrap(h, { + runCommand: fakeRunCommand(h, { + command: 'bash .localagent-box/setup.sh', + exitCode: 1, + outputTail: 'failure in setup.sh', + timedOut: false, + success: false, + }), + }); + + assert.equal(state.status, 'failed'); + assert.equal(state.command, 'bash .localagent-box/setup.sh'); + assert.equal(state.exitCode, 1); + assert.match(state.error ?? '', /^Bootstrap failed: `bash \.localagent-box\/setup\.sh` exited 1/); + assert.equal(h.agent.bootstrap?.status, 'failed'); + assert.equal(h.agent.bootstrap?.source, 'script'); + }); + + it('throws on a failing setup.sh when environment.json sets failOnError=true', async () => { + const h = makeHarness(); + writeSetupScript(h.workspaceDir, '#!/usr/bin/env bash\nexit 1\n'); + writeConfig(h.workspaceDir, JSON.stringify({ version: 1, setup: { command: 'npm ci', failOnError: true } })); + let caught: unknown; try { await runBootstrap(h, { @@ -458,7 +481,7 @@ describe('runWorkspaceBootstrap', () => { assert.deepEqual(before?.profiles, []); }); - it('throws when the setup command fails with the default failOnError', async () => { + it('does not throw on failure by default and records the failure state', async () => { const h = makeHarness(); writeConfig(h.workspaceDir, JSON.stringify({ version: 1, setup: { command: 'npm ci' } })); const opts = { @@ -470,6 +493,34 @@ describe('runWorkspaceBootstrap', () => { success: false, }), }; + + const state = await runBootstrap(h, opts); + + assert.equal(state.status, 'failed'); + assert.equal(state.exitCode, 1); + assert.equal(state.error, 'Bootstrap failed: `npm ci` exited 1'); + assert.equal(state.outputTail, 'npm ERR! Missing script: "prepare"'); + assert.equal(h.agent.bootstrap?.status, 'failed'); + const log = fs.readFileSync(h.logPath, 'utf8'); + assert.match(log, /Workspace bootstrap failed with exit code 1/); + assert.match(log, /continuing \(the agent will attempt to fix it\)/); + }); + + it('throws when the setup command fails with explicit failOnError=true', async () => { + const h = makeHarness(); + writeConfig( + h.workspaceDir, + JSON.stringify({ version: 1, setup: { command: 'npm ci', failOnError: true } }), + ); + const opts = { + runCommand: fakeRunCommand(h, { + command: 'npm ci', + exitCode: 1, + outputTail: 'npm ERR! Missing script: "prepare"', + timedOut: false, + success: false, + }), + }; let caught: unknown; try { await runBootstrap(h, opts); @@ -483,40 +534,46 @@ describe('runWorkspaceBootstrap', () => { assert.equal(h.agent.bootstrap?.status, 'failed'); const log = fs.readFileSync(h.logPath, 'utf8'); assert.match(log, /Workspace bootstrap failed with exit code 1/); + assert.match(log, /failOnError=true/); }); - it('does not throw when the setup command fails with failOnError=false', async () => { + it('treats a timed-out setup as a failure, records it, and does not throw by default', async () => { const h = makeHarness(); writeConfig( h.workspaceDir, - JSON.stringify({ version: 1, setup: { command: 'npm ci', failOnError: false } }), + JSON.stringify({ + version: 1, + setup: { command: 'npm ci', timeoutMs: 60_000 }, + }), ); - const state = await runBootstrap(h, { + const opts = { runCommand: fakeRunCommand(h, { command: 'npm ci', - exitCode: 1, - outputTail: 'npm ERR! missing', - timedOut: false, + exitCode: 124, + outputTail: '', + timedOut: true, success: false, }), - }); + }; + const state = await runBootstrap(h, opts); assert.equal(state.status, 'failed'); - assert.equal(state.exitCode, 1); - assert.equal(state.error, 'Bootstrap failed: `npm ci` exited 1'); + assert.equal(state.exitCode, 124); + assert.match(state.error ?? '', /Bootstrap timed out/); assert.equal(h.agent.bootstrap?.status, 'failed'); + assert.equal(h.agent.bootstrap?.exitCode, 124); const log = fs.readFileSync(h.logPath, 'utf8'); - assert.match(log, /failOnError=false/); + assert.match(log, /Workspace bootstrap timed out/); }); - it('treats a timed-out setup as a failure and throws', async () => { + it('treats a timed-out setup as a failure and throws with explicit failOnError=true', async () => { const h = makeHarness(); writeConfig( h.workspaceDir, JSON.stringify({ version: 1, - setup: { command: 'npm ci', timeoutMs: 60_000 }, + setup: { command: 'npm ci', timeoutMs: 60_000, failOnError: true }, }), ); @@ -958,36 +1015,27 @@ describe('runWorkspaceBootstrap', () => { assert.equal(h.runCalls[1].timeoutMs, 45_000); }); - it('fails the bootstrap (throw) when verify fails, regardless of failOnError=false', async () => { + it('feeds a verify failure to the agent instead of throwing by default', async () => { const h = makeHarness(); writeConfig( h.workspaceDir, JSON.stringify({ version: 1, - setup: { command: 'npm ci', failOnError: false }, + setup: { command: 'npm ci' }, verifyCommand: 'npm test', }), ); - let caught: unknown; - try { - await runBootstrap(h, { - runCommand: setupAndVerifyCommand( - h, - success('npm ci', 'ok'), - { command: 'npm test', exitCode: 1, outputTail: 'npm ERR! test failed', timedOut: false, success: false }, - ), - }); - } catch (err) { - caught = err; - } + const state = await runBootstrap(h, { + runCommand: setupAndVerifyCommand( + h, + success('npm ci', 'ok'), + { command: 'npm test', exitCode: 1, outputTail: 'npm ERR! test failed', timedOut: false, success: false }, + ), + }); - assert.ok(caught instanceof Error); - assert.match( - (caught as { message: string }).message, - /^Bootstrap verify failed: `npm test` exited 1/, - ); - assert.match((caught as { message: string }).message, /npm ERR! test failed/); + assert.equal(state.status, 'failed'); + assert.equal(state.error, 'Bootstrap verify failed: `npm test` exited 1'); assert.equal(h.agent.bootstrap?.status, 'failed'); assert.equal(h.agent.bootstrap?.command, 'npm ci'); assert.equal(h.agent.bootstrap?.verifyCommand, 'npm test'); @@ -997,19 +1045,16 @@ describe('runWorkspaceBootstrap', () => { const log = fs.readFileSync(h.logPath, 'utf8'); assert.match(log, /Workspace bootstrap verify failed/); - assert.ok( - !log.includes('failOnError=false'), - 'verify failure must not honor failOnError=false', - ); + assert.match(log, /verify failed — continuing/); }); - it('treats a verify timeout as a failure and throws', async () => { + it('throws on a verify failure when setup.failOnError=true', async () => { const h = makeHarness(); writeConfig( h.workspaceDir, JSON.stringify({ version: 1, - setup: { command: 'npm ci' }, + setup: { command: 'npm ci', failOnError: true }, verifyCommand: 'npm test', }), ); @@ -1020,7 +1065,7 @@ describe('runWorkspaceBootstrap', () => { runCommand: setupAndVerifyCommand( h, success('npm ci', 'ok'), - { command: 'npm test', exitCode: 124, outputTail: '', timedOut: true, success: false }, + { command: 'npm test', exitCode: 1, outputTail: 'npm ERR! test failed', timedOut: false, success: false }, ), }); } catch (err) { @@ -1028,7 +1073,36 @@ describe('runWorkspaceBootstrap', () => { } assert.ok(caught instanceof Error); - assert.match((caught as { message: string }).message, /Bootstrap verify timed out/); + assert.match( + (caught as { message: string }).message, + /^Bootstrap verify failed: `npm test` exited 1/, + ); + assert.match((caught as { message: string }).message, /npm ERR! test failed/); + assert.equal(h.agent.bootstrap?.status, 'failed'); + assert.equal(h.agent.bootstrap?.verifyExitCode, 1); + }); + + it('treats a verify timeout as a failure and does not throw by default', async () => { + const h = makeHarness(); + writeConfig( + h.workspaceDir, + JSON.stringify({ + version: 1, + setup: { command: 'npm ci' }, + verifyCommand: 'npm test', + }), + ); + + const state = await runBootstrap(h, { + runCommand: setupAndVerifyCommand( + h, + success('npm ci', 'ok'), + { command: 'npm test', exitCode: 124, outputTail: '', timedOut: true, success: false }, + ), + }); + + assert.equal(state.status, 'failed'); + assert.match(state.error ?? '', /Bootstrap verify timed out/); assert.equal(h.agent.bootstrap?.verifyExitCode, 124); const log = fs.readFileSync(h.logPath, 'utf8'); assert.match(log, /Workspace bootstrap verify failed/); diff --git a/src/domains/agents/worker/workspace-bootstrap.ts b/src/domains/agents/worker/workspace-bootstrap.ts index bff4042..77d03bb 100644 --- a/src/domains/agents/worker/workspace-bootstrap.ts +++ b/src/domains/agents/worker/workspace-bootstrap.ts @@ -63,9 +63,10 @@ export interface RunWorkspaceBootstrapOptions { * Resolution: * - `.localagent-box/setup.sh` committed in the repo → run via `bash` * (P4-T1), before any other source is considered. Effective semantics - * for a bare script (no `environment.json`): it always fail-hards — - * a non-zero exit throws and fails the agent start — no `verifyCommand` - * runs, and `setup.runOnModes` and `cacheKey` have no effect. With an + * for a bare script (no `environment.json`): failures are non-blocking — + * they are recorded on the agent record and surfaced to the model's first + * prompt so it can attempt a fix — no `verifyCommand` runs, and + * `setup.runOnModes` and `cacheKey` have no effect. With an * `environment.json` alongside, `setup.failOnError`, `setup.timeoutMs`, * `setup.runOnModes`, `verifyCommand`, and `cacheKey` still apply. * - No `.localagent-box/environment.json` → skipped, unless the server config @@ -83,11 +84,14 @@ export interface RunWorkspaceBootstrapOptions { * * Post-setup verification (P4-T2): when the config sets `verifyCommand`, it * is run after a successful setup with the same timeout as the setup - * command (or `verifyTimeoutMs`); a failure always fails the bootstrap - * (there is no `failOnError` opt-out for verify). + * command (or `verifyTimeoutMs`); a failure is handled per `failOnError` + * (recorded + fed to the agent by default, throw when `true`). * - * Throws when the setup command fails with `failOnError` left at its - * default (`true`); caller is expected to fail the agent start. + * Non-blocking on failure: a failed setup/verify is recorded + * (`status: 'failed'`, `error`, `outputTail`) and returned so the run + * continues and the error is fed to the agent's prompt for a fix attempt. + * Only an explicit `setup.failOnError: true` still throws and fails the + * agent start. */ export async function runWorkspaceBootstrap( options: RunWorkspaceBootstrapOptions, @@ -246,10 +250,10 @@ export async function runWorkspaceBootstrap( const verifyCommand = envConfig?.verifyCommand; if (verifyCommand !== undefined) { return await runVerifyCommand( + workspaceDir, agentsStore, agentId, logPath, - workspaceDir, command, profiles, source, @@ -258,6 +262,7 @@ export async function runWorkspaceBootstrap( cacheHit, verifyCommand, envConfig?.verifyTimeoutMs ?? timeoutMsToUse, + failOnError, runCommand, ); } @@ -300,26 +305,32 @@ export async function runWorkspaceBootstrap( cacheHit, }; - if (failOnError === false) { - appendLog(logPath, 'Workspace bootstrap failed but failOnError=false — continuing'); + if (failOnError === true) { + appendLog(logPath, 'Workspace bootstrap failed with failOnError=true — failing the agent start'); updateAgentRecord(agentsStore, agentId, { bootstrap: failedState }); - return failedState; + throw new Error(`${error}\n${outputTail}`); } + // Default: non-blocking. The failure is recorded (status 'failed') and the + // run continues; the error is fed to the agent's first prompt so it can + // attempt to fix the workspace itself. + appendLog(logPath, 'Workspace bootstrap failed — continuing (the agent will attempt to fix it)'); updateAgentRecord(agentsStore, agentId, { bootstrap: failedState }); - throw new Error(`${error}\n${outputTail}`); + return failedState; } /** - * Run a successful setup's post-setup smoke test (P4-T2). A failure always - * fails the bootstrap and throws (no `failOnError` opt-out for verify), so - * a broken environment never reaches the agent. + * Run a successful setup's post-setup smoke test (P4-T2). A failure is + * recorded (`status: 'failed'`) and returned so the run continues and the + * error reaches the agent's prompt for a fix attempt — `setup.failOnError: + * true` is honored for verify by the caller in `prepareWorkspace` (the + * returned failure state's `verifyCommand` marks it). */ async function runVerifyCommand( + workspaceDir: string, agentsStore: JsonStore<{ agents: Agent[] }>, agentId: string, logPath: string, - workspaceDir: string, setupCommand: string, profiles: string[], source: AgentBootstrapState['source'], @@ -328,6 +339,7 @@ async function runVerifyCommand( cacheHit: boolean, verifyCommand: string, timeoutMs: number, + failOnError: boolean | undefined, runCommand: typeof runWorkspaceCommand, ): Promise { updateAgentRecord(agentsStore, agentId, { @@ -388,6 +400,13 @@ async function runVerifyCommand( error, cacheHit, }; + + if (failOnError === true) { + updateAgentRecord(agentsStore, agentId, { bootstrap: failedState }); + throw new Error(`${error}\n${outputTail}`); + } + + appendLog(logPath, 'Workspace bootstrap verify failed — continuing (the agent will attempt to fix it)'); updateAgentRecord(agentsStore, agentId, { bootstrap: failedState }); - throw new Error(`${error}\n${outputTail}`); -} \ No newline at end of file + return failedState; +} diff --git a/src/domains/agents/worker/workspace-setup.test.ts b/src/domains/agents/worker/workspace-setup.test.ts index 55e63c2..aedf336 100644 --- a/src/domains/agents/worker/workspace-setup.test.ts +++ b/src/domains/agents/worker/workspace-setup.test.ts @@ -206,23 +206,41 @@ describe('prepareWorkspace — bootstrap wiring', () => { } }); - it('propagates a failing bootstrap so the worker never reaches OpenCode', async () => { + it('records a failing bootstrap on the agent without blocking the run', async () => { const h = makeHarness(); fs.writeFileSync(path.join(h.fixtureDir, 'package.json'), '{}', 'utf8'); writeEnvironmentJson(h, JSON.stringify({ version: 1, setup: { command: 'exit 1' } })); + try { + await withNoCodegraph(() => prepareWorkspace(h.buildContext())); + + const agent = h.agentsStore.load().agents[0]; + assert.equal(agent.bootstrap?.status, 'failed'); + assert.equal(agent.bootstrap?.exitCode, 1); + assert.match(agent.bootstrap?.error ?? '', /Bootstrap failed: `exit 1` exited 1/); + + const log = fs.readFileSync(h.job.logPath, 'utf8'); + assert.match(log, /Running workspace bootstrap/); + assert.match(log, /Workspace bootstrap failed with exit code 1/); + } finally { + fs.rmSync(h.dataDir, { recursive: true, force: true }); + } + }); + + it('still throws (fails the agent start) when setup.failOnError is true', async () => { + const h = makeHarness(); + fs.writeFileSync(path.join(h.fixtureDir, 'package.json'), '{}', 'utf8'); + writeEnvironmentJson( + h, + JSON.stringify({ version: 1, setup: { command: 'exit 1', failOnError: true } }), + ); + let caught: unknown; try { try { await withNoCodegraph(async () => { await prepareWorkspace(h.buildContext()); }); - const agent = h.agentsStore.load().agents[0]; - assert.equal(agent.bootstrap?.status, 'failed'); - assert.equal(agent.bootstrap?.exitCode, 1); - const log = fs.readFileSync(h.job.logPath, 'utf8'); - assert.match(log, /Running workspace bootstrap/); - assert.match(log, /Workspace bootstrap failed with exit code 1/); } catch (err) { caught = err; } diff --git a/src/integrations/opencode/runner.test.ts b/src/integrations/opencode/runner.test.ts index d10745d..f28c5b6 100644 --- a/src/integrations/opencode/runner.test.ts +++ b/src/integrations/opencode/runner.test.ts @@ -1,6 +1,7 @@ import assert from 'node:assert/strict'; import { describe, it } from 'node:test'; import { + formatBootstrapSummaryBlock, isAgentRunTimedOut, parseTimeoutMs, resolveAgentRunStartedAtMs, @@ -31,6 +32,66 @@ describe('resolveAgentRunStartedAtMs', () => { }); }); +describe('formatBootstrapSummaryBlock', () => { + it('returns null for skipped or missing bootstrap states', () => { + assert.equal(formatBootstrapSummaryBlock(null), null); + assert.equal(formatBootstrapSummaryBlock(undefined), null); + assert.equal(formatBootstrapSummaryBlock({ status: 'skipped' }), null); + }); + + it('renders the workspace-ready block for a completed bootstrap', () => { + const block = formatBootstrapSummaryBlock({ + status: 'completed', + command: 'npm ci', + profiles: ['nodejs'], + source: 'detect', + durationMs: 38_000, + exitCode: 0, + cacheHit: true, + }); + assert.ok(block); + assert.match(block, /^## Workspace ready \(host\)$/m); + assert.match(block, /- Setup: npm ci \(completed in 38s, cache hit\)/); + assert.match(block, /- Profiles: nodejs/); + }); + + it('renders a failure block with command, error, and output tail for a failed bootstrap', () => { + const block = formatBootstrapSummaryBlock({ + status: 'failed', + command: 'npm ci', + source: 'explicit', + durationMs: 5_000, + exitCode: 1, + outputTail: 'npm ERR! Missing script: "prepare"', + error: 'Bootstrap failed: `npm ci` exited 1', + }); + assert.ok(block); + assert.match(block, /^## Workspace bootstrap failed \(host\)$/m); + assert.match(block, /failed with exit code 1/); + assert.match(block, /- Setup: npm ci/); + assert.match(block, /- Error: Bootstrap failed: `npm ci` exited 1/); + assert.match(block, /npm ERR! Missing script: "prepare"/); + assert.match(block, /attempt to fix the workspace environment/); + assert.doesNotMatch(block, /- Verify:/); + }); + + it('renders the failed verify command in the failure block', () => { + const block = formatBootstrapSummaryBlock({ + status: 'failed', + command: 'npm ci', + verifyCommand: 'npm test', + source: 'explicit', + exitCode: 0, + verifyExitCode: 2, + outputTail: 'tests broke', + error: 'Bootstrap verify failed: `npm test` exited 2', + }); + assert.ok(block); + assert.match(block, /- Verify: failed \(`npm test` exited 2\)/); + assert.match(block, /tests broke/); + }); +}); + describe('isAgentRunTimedOut', () => { const timeoutMs = 3600_000; diff --git a/src/integrations/opencode/runner.ts b/src/integrations/opencode/runner.ts index a80f41c..b72f7da 100644 --- a/src/integrations/opencode/runner.ts +++ b/src/integrations/opencode/runner.ts @@ -184,16 +184,40 @@ function formatBootstrapDuration(durationMs: number | undefined): string | null } /** - * P4-T5 — deterministic host-generated summary of a successful workspace - * bootstrap (setup + verify), injected into the first prompt so the model - * doesn't spend turns discovering how to check the environment. - * Returns null unless the bootstrap completed (the block is first-prompt-only, - * so a null here is simply omitted). + * P4-T5 — deterministic host-generated summary of the workspace bootstrap + * (setup + verify), injected into the first prompt so the model doesn't spend + * turns discovering how to check the environment. Completed runs get the + * workspace-ready block; failed runs get a failure block carrying the + * command, exit code, and output tail so the agent can attempt to fix the + * environment itself. Returns null for skipped/missing bootstrap states + * (nothing happened worth reporting; the block is first-prompt-only). */ export function formatBootstrapSummaryBlock(bootstrap: AgentBootstrapState | null | undefined): string | null { - if (!bootstrap || bootstrap.status !== 'completed') { + if (!bootstrap || bootstrap.status === 'skipped') { return null; } + if (bootstrap.status === 'failed') { + const lines = [`## Workspace bootstrap failed (host)`]; + lines.push( + `The host setup step failed with exit code ${bootstrap.exitCode ?? 'unknown'}; you must attempt to fix the workspace environment as part of the task.`, + ); + lines.push(`- Setup: ${bootstrap.command ?? 'unknown'}`); + if (bootstrap.verifyCommand) { + lines.push( + `- Verify: failed (\`${bootstrap.verifyCommand}\` exited ${bootstrap.verifyExitCode ?? 'unknown'})`, + ); + } + if (bootstrap.error) { + lines.push(`- Error: ${bootstrap.error}`); + } + if (bootstrap.outputTail?.trim()) { + lines.push('- Output tail:', '```', bootstrap.outputTail.trimEnd(), '```'); + } + lines.push( + 'Diagnose the failure and repair the environment (e.g. install missing dependencies, fix the broken step), then continue with the task. Do not modify `.localagent-box/` config files.', + ); + return lines.join('\n'); + } const duration = formatBootstrapDuration(bootstrap.durationMs); const lines = [`## Workspace ready (host)`]; const setupParts: string[] = []; diff --git a/src/integrations/opencode/session-orchestrator.ts b/src/integrations/opencode/session-orchestrator.ts index 747ca0f..dec723d 100644 --- a/src/integrations/opencode/session-orchestrator.ts +++ b/src/integrations/opencode/session-orchestrator.ts @@ -557,9 +557,10 @@ export async function runSessionOrchestrator( batchPromptBusySeen = false; } - // P4-T5: successful host bootstrap is summarized for the model so it - // doesn't spend turns rediscovering the environment (skipped/failed - // bootstrap produces no block). + // P4-T5: the host bootstrap outcome is summarized for the model so + // it doesn't spend turns rediscovering the environment, and a + // failed bootstrap feeds the error tail so the agent can attempt a + // fix (skipped bootstrap produces no block). const bootstrapSummary = formatBootstrapSummaryBlock( readAgentRecord(agentsStore, job.agentId)?.bootstrap, ); diff --git a/src/types/index.ts b/src/types/index.ts index 1a3377e..3a33045 100644 --- a/src/types/index.ts +++ b/src/types/index.ts @@ -386,7 +386,10 @@ export interface AgentTokenUsage { export interface RepoEnvironmentSetupConfig { command: string; timeoutMs?: number; - /** Default true — a non-zero exit fails the agent (applied at runtime) */ + /** + * Default false — a failed setup/verify is recorded and fed to the agent's + * prompt for a fix attempt; `true` fails the agent start (applied at runtime) + */ failOnError?: boolean; /** * Modes for which the setup runs (P4-T3). When set and the agent's mode is @@ -415,7 +418,8 @@ export interface RepoEnvironmentConfig { cacheKey?: string; /** * Post-setup smoke test run after a successful setup command (P4-T2). - * A failure always fails the bootstrap, regardless of `setup.failOnError`. + * A failure is handled per `setup.failOnError`: recorded + fed to the + * agent by default, or fails the bootstrap when `true`. */ verifyCommand?: string; /**