From 726d2ad3a2f282648e129abde50d618d18d18695 Mon Sep 17 00:00:00 2001 From: jarvis Date: Tue, 18 Aug 2026 05:52:45 +0000 Subject: [PATCH] fix(ri-050): forge fails closed without providers; explicit typed simulation (#1275) (#1278) --- eslint.config.mjs | 1 + packages/forge/PLAN.md | 40 +++ packages/forge/__tests__/fail-closed.test.ts | 319 +++++++++++++++++ .../forge/__tests__/pipeline-runner.test.ts | 195 +++++++++-- packages/forge/src/board-tasks.ts | 17 +- packages/forge/src/cli.spec.ts | 97 +++++- packages/forge/src/cli.ts | 170 ++++++--- packages/forge/src/constants.ts | 84 ++++- packages/forge/src/errors.ts | 46 +++ packages/forge/src/index.ts | 26 ++ packages/forge/src/outcomes.ts | 147 ++++++++ packages/forge/src/pipeline-runner.ts | 326 ++++++++++++------ packages/forge/src/simulated-executor.ts | 32 ++ packages/forge/src/types.ts | 95 ++++- 14 files changed, 1392 insertions(+), 203 deletions(-) create mode 100644 packages/forge/__tests__/fail-closed.test.ts create mode 100644 packages/forge/src/errors.ts create mode 100644 packages/forge/src/outcomes.ts create mode 100644 packages/forge/src/simulated-executor.ts diff --git a/eslint.config.mjs b/eslint.config.mjs index bcfe1995..acfed05f 100644 --- a/eslint.config.mjs +++ b/eslint.config.mjs @@ -34,6 +34,7 @@ export default tseslint.config( 'packages/storage/vitest.config.ts', 'packages/mosaic/vitest.config.ts', 'packages/mosaic/__tests__/*.ts', + 'packages/forge/__tests__/*.ts', 'tools/federation-harness/*.ts', ], }, diff --git a/packages/forge/PLAN.md b/packages/forge/PLAN.md index f8b4bc81..3b918111 100644 --- a/packages/forge/PLAN.md +++ b/packages/forge/PLAN.md @@ -539,3 +539,43 @@ Not every brief needs full Board of Directors review. The classification system ### Backward compatibility Existing briefs without a `class` field are auto-classified. The default (no matching keywords) is `strategic`, so all existing runs get the full pipeline unless keywords trigger `technical`. + +--- + +## Fail-Closed Execution & Explicit Simulation (SDLC-D-035) + +**Added:** 2026-08-17 + +Forge fails closed when a required capability is missing. It never runs a +pipeline with a stub executor and reports success. + +### Normal mode (default) + +- No task executor wired → the CLI exits nonzero with the typed capability + error `FORGE_NO_EXECUTOR`. No run is created. +- A stage whose gate is approval-based (board approval, planning approvals, + remediation re-review, discovery/analysis attestations) records a typed + `waiting-for-authority` stage result and raises `FORGE_AUTHORITY_REQUIRED`. + It never passes vacuously. +- A stage whose gate requires an unwired provider (AI reviewer, CI pipeline) + records a typed `blocked` stage result and raises `FORGE_NO_REVIEWER` / + `FORGE_NO_CI_PIPELINE`. The synthetic echo-review approval in `06-review` + and all vacuous `true` gates were removed. + +### Explicit simulation (`--simulate`) + +Opts into stub/synthetic execution. Every stage result, every gate result, and +the run manifest carry the distinct typed status `simulated` (manifest also +records `mode: "simulated"`). `simulated` is a non-satisfying outcome: +`isSatisfyingOutcome()` and all completion/gate consumers treat only `passed` +as satisfying. The CLI exits 0 for a simulated run only because the caller +explicitly passed `--simulate`, and prints a loud SIMULATED banner. + +### Typed outcome model + +Every gate/task outcome is one of the closed set +`passed | failed | blocked | error | waiting-for-authority | simulated | +not-applicable`, with the reason recorded on the stage status and each gate +result in `manifest.json`. Missing implementations, missing gate evidence, +unknown stages, process errors, and timeouts map to fail-closed members — +never to `passed`. diff --git a/packages/forge/__tests__/fail-closed.test.ts b/packages/forge/__tests__/fail-closed.test.ts new file mode 100644 index 00000000..c707a3be --- /dev/null +++ b/packages/forge/__tests__/fail-closed.test.ts @@ -0,0 +1,319 @@ +import fs from 'node:fs'; +import os from 'node:os'; +import path from 'node:path'; +import { describe, it, expect, beforeEach, afterEach } from 'vitest'; + +import { generateBoardTasks } from '../src/board-tasks.js'; +import { STAGE_SPECS } from '../src/constants.js'; +import { ForgeCapabilityError } from '../src/errors.js'; +import { + evaluateStageGates, + gateLabel, + isCommandGate, + isSatisfyingOutcome, +} from '../src/outcomes.js'; +import { loadManifest, runPipeline } from '../src/pipeline-runner.js'; +import type { ForgeTask, ForgeTaskResult, TaskExecutor } from '../src/types.js'; + +/** + * Mock real executor that returns typed results. + * + * Command gates are "verified" by the mock so normal-mode runs can pass + * mechanically gated stages; authority/provider gates are never reported + * because they have no mechanical implementation. + */ +function createTypedExecutor(options?: { + failStage?: string; + gateOutcomes?: Record; +}): TaskExecutor & { submittedTasks: ForgeTask[] } { + const submittedTasks: ForgeTask[] = []; + return { + submittedTasks, + async submitTask(task: ForgeTask) { + submittedTasks.push(task); + }, + async waitForCompletion(taskId: string): Promise { + const task = submittedTasks.find((t) => t.id === taskId); + const stageName = task?.metadata?.['stageName'] as string | undefined; + + if (options?.failStage && stageName === options.failStage) { + return { + task_id: taskId, + outcome: 'failed', + reason: 'mock task failure', + completed_at: new Date().toISOString(), + exit_code: 1, + gate_results: [], + }; + } + + const gateResults = (task?.qualityGates ?? []) + .filter((gate) => isCommandGate(gate)) + .map((gate) => { + const label = gateLabel(gate); + const outcome = options?.gateOutcomes?.[label] ?? 'passed'; + return { + gate: label, + outcome, + reason: outcome === 'passed' ? 'mock verified' : `mock gate outcome: ${outcome}`, + }; + }); + + return { + task_id: taskId, + outcome: 'passed', + reason: 'mock verified', + completed_at: new Date().toISOString(), + exit_code: 0, + gate_results: gateResults, + }; + }, + async getTaskStatus() { + return 'completed' as const; + }, + }; +} + +describe('fail-closed: no executor wired', () => { + let tmpDir: string; + let briefPath: string; + + beforeEach(() => { + tmpDir = fs.mkdtempSync(path.join(os.tmpdir(), 'forge-failclosed-')); + briefPath = path.join(tmpDir, 'brief.md'); + fs.writeFileSync(briefPath, '# Fix bug\n\nA bugfix for lint cleanup.'); + }); + + afterEach(() => { + fs.rmSync(tmpDir, { recursive: true, force: true }); + }); + + it('throws a typed FORGE_NO_EXECUTOR capability error without --simulate', async () => { + await expect( + runPipeline(briefPath, tmpDir, { + // no executor, no simulate — must fail closed, never run with a stub + stages: ['00-intake'], + }), + ).rejects.toMatchObject({ + name: 'ForgeCapabilityError', + code: 'FORGE_NO_EXECUTOR', + capability: 'task-executor', + }); + }); + + it('does not create a run directory when failing closed on a missing executor', async () => { + try { + await runPipeline(briefPath, tmpDir, { stages: ['00-intake'] }); + } catch { + // expected + } + expect(fs.existsSync(path.join(tmpDir, '.forge', 'runs'))).toBe(false); + }); + + it('completes with every result typed simulated when simulate is set', async () => { + const result = await runPipeline(briefPath, tmpDir, { + simulate: true, + stages: ['00-intake', '00b-discovery', '02-planning-1', '06-review'], + }); + + expect(result.manifest.mode).toBe('simulated'); + expect(result.manifest.status).toBe('simulated'); + + for (const stage of result.stages) { + const stageStatus = result.manifest.stages[stage]; + expect(stageStatus?.status, `stage ${stage}`).toBe('simulated'); + expect(stageStatus?.status, `stage ${stage}`).not.toBe('passed'); + expect(stageStatus?.reason, `stage ${stage}`).toBeTruthy(); + for (const gateResult of stageStatus?.gateResults ?? []) { + expect(gateResult.outcome, `gate ${gateResult.gate} of ${stage}`).toBe('simulated'); + expect(gateResult.outcome, `gate ${gateResult.gate} of ${stage}`).not.toBe('passed'); + } + } + + // The persisted manifest agrees. + const persisted = loadManifest(result.runDir); + expect(persisted.mode).toBe('simulated'); + expect(persisted.status).toBe('simulated'); + expect(persisted.stages['02-planning-1']?.status).toBe('simulated'); + }); +}); + +describe('fail-closed: typed outcome model', () => { + it('only passed satisfies the gate/dependency predicate', () => { + expect(isSatisfyingOutcome('passed')).toBe(true); + expect(isSatisfyingOutcome('failed')).toBe(false); + expect(isSatisfyingOutcome('blocked')).toBe(false); + expect(isSatisfyingOutcome('error')).toBe(false); + expect(isSatisfyingOutcome('waiting-for-authority')).toBe(false); + expect(isSatisfyingOutcome('simulated')).toBe(false); + expect(isSatisfyingOutcome('not-applicable')).toBe(false); + }); + + it('a simulated gate result cannot satisfy the stage gate evaluation', () => { + const evaluation = evaluateStageGates('05-coding', STAGE_SPECS['05-coding']!.qualityGates, { + task_id: 'FORGE-x-05', + outcome: 'passed', + reason: 'executor claims success', + completed_at: new Date().toISOString(), + exit_code: 0, + gate_results: [{ gate: 'pnpm lint', outcome: 'simulated', reason: 'simulated gate' }], + }); + expect(isSatisfyingOutcome(evaluation.outcome)).toBe(false); + expect(evaluation.outcome).toBe('error'); + }); + + it('a simulated task outcome cannot satisfy evaluation in normal mode', () => { + const evaluation = evaluateStageGates('00-intake', [], { + task_id: 'FORGE-x-00', + outcome: 'simulated', + reason: 'executor reported simulated', + completed_at: new Date().toISOString(), + exit_code: 0, + gate_results: [], + }); + expect(isSatisfyingOutcome(evaluation.outcome)).toBe(false); + }); + + it('a missing gate result blocks the stage instead of passing vacuously', () => { + const evaluation = evaluateStageGates('05-coding', STAGE_SPECS['05-coding']!.qualityGates, { + task_id: 'FORGE-x-05', + outcome: 'passed', + reason: 'executor claims success', + completed_at: new Date().toISOString(), + exit_code: 0, + gate_results: [], + }); + expect(evaluation.outcome).toBe('blocked'); + }); +}); + +describe('fail-closed: authority and provider gates', () => { + let tmpDir: string; + let briefPath: string; + + beforeEach(() => { + tmpDir = fs.mkdtempSync(path.join(os.tmpdir(), 'forge-authority-')); + briefPath = path.join(tmpDir, 'brief.md'); + fs.writeFileSync(briefPath, '# Fix bug\n\nA bugfix for lint cleanup.'); + }); + + afterEach(() => { + fs.rmSync(tmpDir, { recursive: true, force: true }); + }); + + it.each(['02-planning-1', '03-planning-2', '04-planning-3', '07-remediate'])( + 'planning/remediation stage %s yields waiting-for-authority (not passed) in normal mode', + async (stage) => { + const executor = createTypedExecutor(); + let runDir: string | undefined; + + try { + await runPipeline(briefPath, tmpDir, { + executor, + stages: [stage as string], + }); + expect.unreachable('runPipeline should have failed closed'); + } catch (err) { + expect(err).toBeInstanceOf(ForgeCapabilityError); + expect((err as ForgeCapabilityError).code).toBe('FORGE_AUTHORITY_REQUIRED'); + runDir = path.join(tmpDir, '.forge', 'runs'); + } + + const runIds = fs.readdirSync(runDir!); + expect(runIds).toHaveLength(1); + const manifest = loadManifest(path.join(runDir!, runIds[0]!)); + expect(manifest.stages[stage]?.status).toBe('waiting-for-authority'); + expect(manifest.stages[stage]?.status).not.toBe('passed'); + expect(manifest.status).toBe('waiting-for-authority'); + }, + ); + + it('review stage fails closed with a typed FORGE_NO_REVIEWER error in normal mode', async () => { + const executor = createTypedExecutor(); + + try { + await runPipeline(briefPath, tmpDir, { + executor, + stages: ['06-review'], + }); + expect.unreachable('runPipeline should have failed closed'); + } catch (err) { + expect(err).toBeInstanceOf(ForgeCapabilityError); + expect((err as ForgeCapabilityError).code).toBe('FORGE_NO_REVIEWER'); + expect((err as ForgeCapabilityError).capability).toBe('reviewer'); + } + + const runsDir = path.join(tmpDir, '.forge', 'runs'); + const runIds = fs.readdirSync(runsDir); + const manifest = loadManifest(path.join(runsDir, runIds[0]!)); + expect(manifest.stages['06-review']?.status).toBe('blocked'); + expect(manifest.stages['06-review']?.status).not.toBe('passed'); + expect(manifest.status).toBe('failed'); + }); + + it('review stage produces simulated results under --simulate', async () => { + const result = await runPipeline(briefPath, tmpDir, { + simulate: true, + stages: ['06-review'], + }); + + expect(result.manifest.mode).toBe('simulated'); + expect(result.manifest.stages['06-review']?.status).toBe('simulated'); + for (const gateResult of result.manifest.stages['06-review']?.gateResults ?? []) { + expect(gateResult.outcome).toBe('simulated'); + } + }); + + it('deploy stage fails closed without a wired ci-pipeline provider in normal mode', async () => { + const executor = createTypedExecutor(); + + await expect( + runPipeline(briefPath, tmpDir, { + executor, + stages: ['09-deploy'], + }), + ).rejects.toMatchObject({ + name: 'ForgeCapabilityError', + code: 'FORGE_NO_CI_PIPELINE', + }); + }); +}); + +describe('fail-closed: no vacuous gate commands remain', () => { + it('stage constants contain no echo/synthetic-approval, vacuous true, or empty gate commands', () => { + for (const [stageName, spec] of Object.entries(STAGE_SPECS)) { + for (const gate of spec.qualityGates) { + const serialized = JSON.stringify(gate); + // The echo-review synthetic approval must be gone. + expect(serialized, `stage ${stageName} gate ${serialized}`).not.toContain('echo'); + expect(serialized, `stage ${stageName} gate ${serialized}`).not.toMatch(/"verdict"\s*:/); + expect(serialized, `stage ${stageName} gate ${serialized}`).not.toMatch( + /"summary"\s*:\s*"review-pass"/, + ); + // No vacuous literal `true` gate. + expect(gate, `stage ${stageName}`).not.toBe('true'); + // Command gates must carry a real, non-empty command. + if (isCommandGate(gate)) { + const command = typeof gate === 'string' ? gate : gate.command; + expect(command.trim().length, `stage ${stageName} gate ${serialized}`).toBeGreaterThan(0); + } + } + } + }); + + it('board tasks contain no vacuous true gates', () => { + const tmpDir = fs.mkdtempSync(path.join(os.tmpdir(), 'forge-board-gates-')); + try { + const tasks = generateBoardTasks('# Brief', [], tmpDir, 'BOARD-TEST'); + for (const task of tasks) { + for (const gate of task.qualityGates) { + expect(gate, `task ${task.id}`).not.toBe('true'); + const serialized = JSON.stringify(gate); + expect(serialized, `task ${task.id} gate ${serialized}`).not.toContain('echo'); + } + } + } finally { + fs.rmSync(tmpDir, { recursive: true, force: true }); + } + }); +}); diff --git a/packages/forge/__tests__/pipeline-runner.test.ts b/packages/forge/__tests__/pipeline-runner.test.ts index 398b8997..0baeeed3 100644 --- a/packages/forge/__tests__/pipeline-runner.test.ts +++ b/packages/forge/__tests__/pipeline-runner.test.ts @@ -12,10 +12,10 @@ import { resumePipeline, getPipelineStatus, } from '../src/pipeline-runner.js'; -import type { ForgeTask, RunManifest, TaskExecutor } from '../src/types.js'; -import type { TaskResult } from '@mosaicstack/macp'; +import type { ForgeTask, ForgeTaskResult, RunManifest, TaskExecutor } from '../src/types.js'; +import { gateLabel, isCommandGate } from '../src/outcomes.js'; -/** Mock TaskExecutor that records submitted tasks and returns success. */ +/** Mock TaskExecutor that records submitted tasks and returns typed results. */ function createMockExecutor(options?: { failStage?: string; }): TaskExecutor & { submittedTasks: ForgeTask[] } { @@ -25,7 +25,7 @@ function createMockExecutor(options?: { async submitTask(task: ForgeTask) { submittedTasks.push(task); }, - async waitForCompletion(taskId: string): Promise { + async waitForCompletion(taskId: string): Promise { const failStage = options?.failStage; const task = submittedTasks.find((t) => t.id === taskId); const stageName = task?.metadata?.['stageName'] as string | undefined; @@ -33,7 +33,8 @@ function createMockExecutor(options?: { if (failStage && stageName === failStage) { return { task_id: taskId, - status: 'failed', + outcome: 'failed', + reason: 'mock task failure', completed_at: new Date().toISOString(), exit_code: 1, gate_results: [], @@ -41,10 +42,17 @@ function createMockExecutor(options?: { } return { task_id: taskId, - status: 'completed', + outcome: 'passed', + reason: 'mock verified', completed_at: new Date().toISOString(), exit_code: 0, - gate_results: [], + gate_results: (task?.qualityGates ?? []) + .filter((gate) => isCommandGate(gate)) + .map((gate) => ({ + gate: gateLabel(gate), + outcome: 'passed' as const, + reason: 'mock verified', + })), }; }, async getTaskStatus() { @@ -156,12 +164,13 @@ describe('runPipeline', () => { const executor = createMockExecutor(); const result = await runPipeline(briefPath, tmpDir, { executor, - stages: ['00-intake', '00b-discovery'], + stages: ['00-intake', '05-coding'], }); expect(result.runId).toMatch(/^\d{8}-\d{6}$/); - expect(result.stages).toEqual(['00-intake', '00b-discovery']); + expect(result.stages).toEqual(['00-intake', '05-coding']); expect(result.manifest.status).toBe('completed'); + expect(result.manifest.mode).toBe('normal'); expect(executor.submittedTasks).toHaveLength(2); }); @@ -180,12 +189,17 @@ describe('runPipeline', () => { const executor = createMockExecutor(); const result = await runPipeline(briefPath, tmpDir, { executor, - stages: ['00-intake', '00b-discovery'], + stages: ['00-intake', '05-coding'], }); const manifest = loadManifest(result.runDir); expect(manifest.stages['00-intake']?.status).toBe('passed'); - expect(manifest.stages['00b-discovery']?.status).toBe('passed'); + expect(manifest.stages['05-coding']?.status).toBe('passed'); + expect(manifest.stages['05-coding']?.gateResults?.map((g) => g.outcome)).toEqual([ + 'passed', + 'passed', + 'passed', + ]); }); it('respects CLI class override', async () => { @@ -215,7 +229,7 @@ describe('runPipeline', () => { const executor = createMockExecutor(); await runPipeline(briefPath, tmpDir, { executor, - stages: ['00-intake', '00b-discovery', '02-planning-1'], + stages: ['00-intake', '05-coding', '08-test'], }); expect(executor.submittedTasks[0]!.dependsOn).toBeUndefined(); @@ -224,14 +238,14 @@ describe('runPipeline', () => { }); it('handles stage failure', async () => { - const executor = createMockExecutor({ failStage: '00b-discovery' }); + const executor = createMockExecutor({ failStage: '05-coding' }); await expect( runPipeline(briefPath, tmpDir, { executor, - stages: ['00-intake', '00b-discovery'], + stages: ['00-intake', '05-coding'], }), - ).rejects.toThrow('Stage 00b-discovery failed'); + ).rejects.toThrow('Stage 05-coding failed'); }); it('marks manifest as failed on stage failure', async () => { @@ -270,30 +284,143 @@ describe('resumePipeline', () => { fs.rmSync(tmpDir, { recursive: true, force: true }); }); - it('resumes from first incomplete stage', async () => { - // First run fails on discovery - const executor1 = createMockExecutor({ failStage: '00b-discovery' }); - let runDir: string; + it('resumes from first incomplete stage and fails closed at the next provider gate', async () => { + // Simulate a run whose authority stages were approved out-of-band + // (recorded as passed) and whose coding stage failed mechanically. + const runId = '20260101-000000'; + const runDir = path.join(tmpDir, '.forge', 'runs', runId); + fs.mkdirSync(runDir, { recursive: true }); + const passed = { status: 'passed' as const, startedAt: '2026-01-01T00:00:00Z' }; + saveManifest(runDir, { + runId, + brief: briefPath, + codebase: tmpDir, + briefClass: 'hotfix', + classSource: 'frontmatter', + forceBoard: false, + mode: 'normal', + createdAt: '2026-01-01T00:00:00Z', + updatedAt: '2026-01-01T00:00:00Z', + currentStage: '05-coding', + status: 'failed', + stages: { + '00-intake': passed, + '00b-discovery': passed, + '02-planning-1': passed, + '03-planning-2': passed, + '04-planning-3': passed, + '05-coding': { status: 'failed', reason: 'gate failed' }, + }, + }); - try { - await runPipeline(briefPath, tmpDir, { - executor: executor1, - stages: ['00-intake', '00b-discovery', '02-planning-1'], - }); - } catch { - // expected + // Resume re-runs 05-coding (the first non-passed stage), then fails + // closed at 06-review because no reviewer provider is wired. + const executor = createMockExecutor(); + await expect(resumePipeline(runDir, executor)).rejects.toMatchObject({ + name: 'ForgeCapabilityError', + code: 'FORGE_NO_REVIEWER', + }); + + const manifest = loadManifest(runDir); + expect(manifest.stages['05-coding']?.status).toBe('passed'); + expect(manifest.stages['06-review']?.status).toBe('blocked'); + expect(manifest.status).toBe('failed'); + }); + + it('resumes to completion as simulated under explicit simulate', async () => { + const runId = '20260101-000003'; + const runDir = path.join(tmpDir, '.forge', 'runs', runId); + fs.mkdirSync(runDir, { recursive: true }); + const passed = { status: 'passed' as const, startedAt: '2026-01-01T00:00:00Z' }; + saveManifest(runDir, { + runId, + brief: briefPath, + codebase: tmpDir, + briefClass: 'hotfix', + classSource: 'frontmatter', + forceBoard: false, + mode: 'normal', + createdAt: '2026-01-01T00:00:00Z', + updatedAt: '2026-01-01T00:00:00Z', + currentStage: '05-coding', + status: 'failed', + stages: { + '00-intake': passed, + '00b-discovery': passed, + '02-planning-1': passed, + '03-planning-2': passed, + '04-planning-3': passed, + '05-coding': { status: 'failed', reason: 'gate failed' }, + }, + }); + + const result = await resumePipeline(runDir, undefined, { simulate: true }); + + expect(result.manifest.status).toBe('simulated'); + expect(result.manifest.mode).toBe('simulated'); + expect(result.stages[0]).toBe('05-coding'); + for (const stage of result.stages) { + expect(result.manifest.stages[stage]?.status).toBe('simulated'); } + }); - const runsDir = path.join(tmpDir, '.forge', 'runs'); - runDir = path.join(runsDir, fs.readdirSync(runsDir)[0]!); + it('fails closed on resume when the next stage needs authority sign-off', async () => { + const runId = '20260101-000001'; + const runDir = path.join(tmpDir, '.forge', 'runs', runId); + fs.mkdirSync(runDir, { recursive: true }); + saveManifest(runDir, { + runId, + brief: briefPath, + codebase: tmpDir, + briefClass: 'hotfix', + classSource: 'frontmatter', + forceBoard: false, + mode: 'normal', + createdAt: '2026-01-01T00:00:00Z', + updatedAt: '2026-01-01T00:00:00Z', + currentStage: '00-intake', + status: 'in_progress', + stages: { + '00-intake': { status: 'passed' }, + }, + }); - // Resume should pick up from 00b-discovery - const executor2 = createMockExecutor(); - const result = await resumePipeline(runDir, executor2); + const executor = createMockExecutor(); + await expect(resumePipeline(runDir, executor)).rejects.toMatchObject({ + name: 'ForgeCapabilityError', + code: 'FORGE_AUTHORITY_REQUIRED', + }); - expect(result.manifest.status).toBe('completed'); - // Should have re-run from 00b-discovery onward - expect(result.stages[0]).toBe('00b-discovery'); + const manifest = loadManifest(runDir); + expect(manifest.stages['00b-discovery']?.status).toBe('waiting-for-authority'); + expect(manifest.status).toBe('waiting-for-authority'); + }); + + it('fails closed on resume without an executor or --simulate', async () => { + const runId = '20260101-000002'; + const runDir = path.join(tmpDir, '.forge', 'runs', runId); + fs.mkdirSync(runDir, { recursive: true }); + saveManifest(runDir, { + runId, + brief: briefPath, + codebase: tmpDir, + briefClass: 'hotfix', + classSource: 'frontmatter', + forceBoard: false, + mode: 'normal', + createdAt: '2026-01-01T00:00:00Z', + updatedAt: '2026-01-01T00:00:00Z', + currentStage: '00-intake', + status: 'in_progress', + stages: { + '00-intake': { status: 'passed' }, + }, + }); + + await expect(resumePipeline(runDir)).rejects.toMatchObject({ + name: 'ForgeCapabilityError', + code: 'FORGE_NO_EXECUTOR', + }); }); }); diff --git a/packages/forge/src/board-tasks.ts b/packages/forge/src/board-tasks.ts index 701ec2b3..6112389c 100644 --- a/packages/forge/src/board-tasks.ts +++ b/packages/forge/src/board-tasks.ts @@ -95,7 +95,14 @@ export function generateBoardTasks( briefPath, resultPath: resultRelPath, timeoutSeconds: 120, - qualityGates: ['true'], + qualityGates: [ + { + kind: 'authority', + capability: 'board-approval', + reason: + 'persona evaluation is judged by board synthesis (authority review); no mechanical gate exists', + }, + ], metadata: { personaName: persona.name, personaSlug: persona.slug, @@ -121,7 +128,13 @@ export function generateBoardTasks( timeoutSeconds: 120, dependsOn: personaTaskIds, dependsOnPolicy: 'all_terminal', - qualityGates: ['true'], + qualityGates: [ + { + kind: 'authority', + capability: 'board-approval', + reason: 'board synthesis is an authority decision; no mechanical gate exists', + }, + ], metadata: { resultOutputPath: synthesisResult, inputResultPaths: personaResultPaths, diff --git a/packages/forge/src/cli.spec.ts b/packages/forge/src/cli.spec.ts index d2fe881e..ff7eb367 100644 --- a/packages/forge/src/cli.spec.ts +++ b/packages/forge/src/cli.spec.ts @@ -1,7 +1,11 @@ +import fs from 'node:fs'; +import os from 'node:os'; +import path from 'node:path'; import { Command } from 'commander'; -import { describe, expect, it } from 'vitest'; +import { describe, expect, it, vi, beforeEach, afterEach } from 'vitest'; import { registerForgeCommand } from './cli.js'; +import { loadManifest } from './pipeline-runner.js'; describe('registerForgeCommand', () => { it('registers a "forge" command on the parent program', () => { @@ -55,3 +59,94 @@ describe('registerForgeCommand', () => { }).not.toThrow(); }); }); + +describe('forge run fail-closed behavior (SDLC-D-035)', () => { + let tmpDir: string; + let briefPath: string; + let errSpy: ReturnType; + let logSpy: ReturnType; + let prevExitCode: string | number | null | undefined; + + const parse = (args: string[]) => { + const program = new Command(); + registerForgeCommand(program); + return program.parseAsync(['forge', ...args], { from: 'user' }); + }; + + beforeEach(() => { + tmpDir = fs.mkdtempSync(path.join(os.tmpdir(), 'forge-cli-failclosed-')); + briefPath = path.join(tmpDir, 'brief.md'); + fs.writeFileSync(briefPath, '# Fix bug\n\nA bugfix for lint cleanup.'); + errSpy = vi.spyOn(console, 'error').mockImplementation(() => {}); + logSpy = vi.spyOn(console, 'log').mockImplementation(() => {}); + prevExitCode = process.exitCode; + }); + + afterEach(() => { + errSpy.mockRestore(); + logSpy.mockRestore(); + process.exitCode = prevExitCode; + fs.rmSync(tmpDir, { recursive: true, force: true }); + }); + + it('exits nonzero with a typed FORGE_NO_EXECUTOR error when no executor is wired and --simulate is absent', async () => { + await parse(['run', '--brief', briefPath, '--codebase', tmpDir]); + + expect(process.exitCode).toBe(1); + const errText = errSpy.mock.calls.map((c) => c.join(' ')).join('\n'); + expect(errText).toContain('FORGE_NO_EXECUTOR'); + // It must never run the pipeline with a stub and report success. + expect(fs.existsSync(path.join(tmpDir, '.forge', 'runs'))).toBe(false); + }); + + it('completes with typed simulated results and exit 0 under explicit --simulate', async () => { + await parse(['run', '--brief', briefPath, '--codebase', tmpDir, '--simulate']); + + expect(process.exitCode).toBeUndefined(); + + // Loud simulated-mode summary. + const logText = logSpy.mock.calls.map((c) => c.join(' ')).join('\n'); + expect(logText).toContain('SIMULATED'); + + // Manifest records the mode and simulated per-result statuses. + const runsDir = path.join(tmpDir, '.forge', 'runs'); + const runIds = fs.readdirSync(runsDir); + expect(runIds).toHaveLength(1); + const manifest = loadManifest(path.join(runsDir, runIds[0]!)); + expect(manifest.mode).toBe('simulated'); + expect(manifest.status).toBe('simulated'); + for (const stageStatus of Object.values(manifest.stages)) { + expect(stageStatus?.status).toBe('simulated'); + for (const gateResult of stageStatus?.gateResults ?? []) { + expect(gateResult.outcome).toBe('simulated'); + } + } + }); + + it('resume exits nonzero with a typed FORGE_NO_EXECUTOR error without --simulate', async () => { + const runDir = path.join(tmpDir, '.forge', 'runs', '20260101-000000'); + fs.mkdirSync(runDir, { recursive: true }); + fs.writeFileSync( + path.join(runDir, 'manifest.json'), + JSON.stringify({ + runId: '20260101-000000', + brief: briefPath, + codebase: tmpDir, + briefClass: 'hotfix', + classSource: 'frontmatter', + forceBoard: false, + createdAt: '2026-01-01T00:00:00Z', + updatedAt: '2026-01-01T00:00:00Z', + currentStage: '00-intake', + status: 'in_progress', + stages: { '00-intake': { status: 'passed' } }, + }), + ); + + await parse(['resume', '20260101-000000', '--project', tmpDir]); + + expect(process.exitCode).toBe(1); + const errText = errSpy.mock.calls.map((c) => c.join(' ')).join('\n'); + expect(errText).toContain('FORGE_NO_EXECUTOR'); + }); +}); diff --git a/packages/forge/src/cli.ts b/packages/forge/src/cli.ts index 618150a8..175df78b 100644 --- a/packages/forge/src/cli.ts +++ b/packages/forge/src/cli.ts @@ -5,37 +5,47 @@ import type { Command } from 'commander'; import { classifyBrief } from './brief-classifier.js'; import { STAGE_LABELS, STAGE_SEQUENCE } from './constants.js'; +import { ForgeCapabilityError } from './errors.js'; import { getEffectivePersonas, loadBoardPersonas } from './persona-loader.js'; import { generateRunId, getPipelineStatus, loadManifest, runPipeline } from './pipeline-runner.js'; -import type { PipelineOptions, RunManifest, TaskExecutor } from './types.js'; - -// --------------------------------------------------------------------------- -// Stub executor — used when no real executor is wired at CLI invocation time. -// --------------------------------------------------------------------------- - -const stubExecutor: TaskExecutor = { - async submitTask(task) { - console.log(` [forge] stage submitted: ${task.id} (${task.title})`); - }, - async waitForCompletion(taskId, _timeoutMs) { - console.log(` [forge] stage complete: ${taskId}`); - return { - task_id: taskId, - status: 'completed' as const, - completed_at: new Date().toISOString(), - exit_code: 0, - gate_results: [], - }; - }, - async getTaskStatus(_taskId) { - return 'completed' as const; - }, -}; +import { createSimulatedExecutor } from './simulated-executor.js'; +import type { PipelineOptions, RunManifest, RunMode } from './types.js'; // --------------------------------------------------------------------------- // Helpers // --------------------------------------------------------------------------- +/** Resolve a run's effective mode, defaulting legacy manifests to normal. */ +function runModeOf(manifest: RunManifest): RunMode { + return manifest.mode ?? 'normal'; +} + +/** Print a loud banner so a simulated run can never be misread as verified. */ +function printSimulatedBanner(): void { + console.log(''); + console.log('[forge] ==============================================================='); + console.log('[forge] MODE: SIMULATED — no stage or gate was really executed.'); + console.log('[forge] All results are synthetic and MUST NOT be read as verified'); + console.log('[forge] success. Wire a real executor/providers and re-run to verify.'); + console.log('[forge] ==============================================================='); +} + +/** Print a typed error line for fail-closed capability errors. */ +function printCapabilityError(err: ForgeCapabilityError): void { + console.error(`[forge] error ${err.code}: ${err.message}`); + console.error(`[forge] missing capability: ${err.capability}`); +} + +/** Handle a pipeline error uniformly: typed capability errors get their code. */ +function handlePipelineError(err: unknown): void { + if (err instanceof ForgeCapabilityError) { + printCapabilityError(err); + } else { + console.error(`[forge] pipeline failed: ${err instanceof Error ? err.message : String(err)}`); + } + process.exitCode = 1; +} + function formatDuration(startedAt?: string, completedAt?: string): string { if (!startedAt || !completedAt) return '-'; const ms = new Date(completedAt).getTime() - new Date(startedAt).getTime(); @@ -44,19 +54,24 @@ function formatDuration(startedAt?: string, completedAt?: string): string { } function printManifestTable(manifest: RunManifest): void { + const mode = runModeOf(manifest); console.log(`\nRun ID : ${manifest.runId}`); console.log(`Status : ${manifest.status}`); + console.log(`Mode : ${mode}`); + if (mode === 'simulated') { + console.log('WARNING: SIMULATED RUN — results are synthetic, not verified success.'); + } console.log(`Brief : ${manifest.brief}`); console.log(`Class : ${manifest.briefClass} (${manifest.classSource})`); console.log(`Updated: ${manifest.updatedAt}`); console.log(''); - console.log('Stage'.padEnd(22) + 'Status'.padEnd(14) + 'Duration'); - console.log('-'.repeat(50)); + console.log('Stage'.padEnd(22) + 'Status'.padEnd(24) + 'Duration'); + console.log('-'.repeat(60)); for (const stage of STAGE_SEQUENCE) { const s = manifest.stages[stage]; if (!s) continue; const label = (STAGE_LABELS[stage] ?? stage).padEnd(22); - const status = s.status.padEnd(14); + const status = s.status.padEnd(24); const dur = formatDuration(s.startedAt, s.completedAt); console.log(`${label}${status}${dur}`); } @@ -90,23 +105,58 @@ function listRecentRuns(projectRoot?: string): void { } console.log('\nRecent runs:'); - console.log('Run ID'.padEnd(22) + 'Status'.padEnd(14) + 'Brief'); - console.log('-'.repeat(70)); + console.log('Run ID'.padEnd(22) + 'Status'.padEnd(24) + 'Mode'.padEnd(12) + 'Brief'); + console.log('-'.repeat(80)); for (const runId of entries) { const runDir = path.join(runsDir, runId); try { const manifest = loadManifest(runDir); - const status = manifest.status.padEnd(14); + const status = manifest.status.padEnd(24); + const mode = runModeOf(manifest).padEnd(12); const brief = path.basename(manifest.brief); - console.log(`${runId.padEnd(22)}${status}${brief}`); + console.log(`${runId.padEnd(22)}${status}${mode}${brief}`); } catch { - console.log(`${runId.padEnd(22)}${'(unreadable)'.padEnd(14)}`); + console.log(`${runId.padEnd(22)}${'(unreadable)'.padEnd(24)}`); } } console.log(''); } +/** + * Apply the exit-code policy for a finished pipeline run (SDLC-D-035): + * + * - exit 0 only for a verified `completed` normal run, or for an overall + * `simulated` run when the caller explicitly passed --simulate; + * - anything else exits nonzero so it can never be read as success. + */ +function applyRunExitPolicy(result: { manifest: RunManifest; runDir: string }, simulate: boolean) { + const { manifest } = result; + + if (runModeOf(manifest) === 'simulated') { + if (!simulate || manifest.status !== 'simulated') { + console.error( + '[forge] error FORGE_MODE_MISMATCH: run reports simulated results without an explicit, ' + + 'consistent --simulate request; refusing to report success.', + ); + process.exitCode = 1; + return; + } + printSimulatedBanner(); + console.log(`[forge] run directory: ${result.runDir}`); + return; // exit 0 — the caller explicitly opted into simulation + } + + if (manifest.status !== 'completed') { + console.error(`[forge] run did not complete: terminal status '${manifest.status}'`); + process.exitCode = 1; + return; + } + + console.log(`[forge] pipeline complete (mode: normal): ${manifest.runId}`); + console.log(`[forge] run directory: ${result.runDir}`); +} + // --------------------------------------------------------------------------- // Register function // --------------------------------------------------------------------------- @@ -129,6 +179,11 @@ export function registerForgeCommand(parent: Command): void { .option('--config ', 'Path to forge config file (.forge/config.yaml)') .option('--codebase ', 'Codebase root to pass to the pipeline', process.cwd()) .option('--dry-run', 'Print planned stages without executing', false) + .option( + '--simulate', + 'Simulate execution without real providers (every result is typed simulated, never verified)', + false, + ) .action( async (opts: { brief: string; @@ -137,6 +192,7 @@ export function registerForgeCommand(parent: Command): void { config?: string; codebase: string; dryRun: boolean; + simulate: boolean; }) => { const briefPath = path.resolve(opts.brief); @@ -149,14 +205,22 @@ export function registerForgeCommand(parent: Command): void { const briefContent = fs.readFileSync(briefPath, 'utf-8'); const briefClass = classifyBrief(briefContent); const projectRoot = opts.codebase; + // A real executor is never wired at CLI invocation time today, so the + // only executor we may construct is the explicitly-requested simulated + // one. Normal mode fails closed with FORGE_NO_EXECUTOR. + const executor = opts.simulate ? createSimulatedExecutor() : undefined; if (opts.resume) { const runId = opts.runId ?? generateRunId(); const runDir = resolveRunDir(runId, projectRoot); console.log(`[forge] resuming run: ${runId}`); - const { resumePipeline } = await import('./pipeline-runner.js'); - const result = await resumePipeline(runDir, stubExecutor); - console.log(`[forge] pipeline complete: ${result.runId}`); + try { + const { resumePipeline } = await import('./pipeline-runner.js'); + const result = await resumePipeline(runDir, executor, { simulate: opts.simulate }); + applyRunExitPolicy(result, opts.simulate); + } catch (err) { + handlePipelineError(err); + } return; } @@ -164,7 +228,8 @@ export function registerForgeCommand(parent: Command): void { briefClass, codebase: projectRoot, dryRun: opts.dryRun, - executor: stubExecutor, + executor, + simulate: opts.simulate, }; if (opts.dryRun) { @@ -180,16 +245,15 @@ export function registerForgeCommand(parent: Command): void { console.log(`[forge] starting pipeline for brief: ${briefPath}`); console.log(`[forge] classified as: ${briefClass}`); + if (opts.simulate) { + console.log('[forge] mode: SIMULATED (explicit --simulate)'); + } try { const result = await runPipeline(briefPath, projectRoot, pipelineOptions); - console.log(`[forge] pipeline complete: ${result.runId}`); - console.log(`[forge] run directory: ${result.runDir}`); + applyRunExitPolicy(result, opts.simulate); } catch (err) { - console.error( - `[forge] pipeline failed: ${err instanceof Error ? err.message : String(err)}`, - ); - process.exitCode = 1; + handlePipelineError(err); } }, ); @@ -224,7 +288,12 @@ export function registerForgeCommand(parent: Command): void { .command('resume ') .description('Resume a stopped or failed pipeline run') .option('--project ', 'Project root (defaults to cwd)', process.cwd()) - .action(async (runId: string, opts: { project: string }) => { + .option( + '--simulate', + 'Simulate execution without real providers (every result is typed simulated, never verified)', + false, + ) + .action(async (runId: string, opts: { project: string; simulate: boolean }) => { const runDir = resolveRunDir(runId, opts.project); if (!fs.existsSync(runDir)) { @@ -234,15 +303,20 @@ export function registerForgeCommand(parent: Command): void { } console.log(`[forge] resuming run: ${runId}`); + if (opts.simulate) { + console.log('[forge] mode: SIMULATED (explicit --simulate)'); + } + + // No real executor is wired at CLI invocation time; only the explicitly + // requested simulated executor may be constructed (fail closed otherwise). + const executor = opts.simulate ? createSimulatedExecutor() : undefined; try { const { resumePipeline } = await import('./pipeline-runner.js'); - const result = await resumePipeline(runDir, stubExecutor); - console.log(`[forge] pipeline complete: ${result.runId}`); - console.log(`[forge] run directory: ${result.runDir}`); + const result = await resumePipeline(runDir, executor, { simulate: opts.simulate }); + applyRunExitPolicy(result, opts.simulate); } catch (err) { - console.error(`[forge] resume failed: ${err instanceof Error ? err.message : String(err)}`); - process.exitCode = 1; + handlePipelineError(err); } }); diff --git a/packages/forge/src/constants.ts b/packages/forge/src/constants.ts index b5165f15..46fc3c31 100644 --- a/packages/forge/src/constants.ts +++ b/packages/forge/src/constants.ts @@ -9,7 +9,16 @@ export const PACKAGE_ROOT = path.resolve(path.dirname(fileURLToPath(import.meta. /** Pipeline asset directory (stages, agents, rails, gates, templates). */ export const PIPELINE_DIR = path.join(PACKAGE_ROOT, 'pipeline'); -/** Stage specifications — defines every pipeline stage. */ +/** Stage specifications — defines every pipeline stage. + *\n * Gate semantics (SDLC-D-035): every gate is one of + * - a real command string / GateEntry a mechanical runner can execute, + * - an `authority` gate (human/board sign-off; produces waiting-for-authority), + * - a `provider` gate (requires a wired provider such as a reviewer or CI pipeline). + * + * Vacuous gates (`true`, echo'd synthetic approvals, placeholder ci-pipeline + * commands) are forbidden: a stage whose gate has no real implementation + * fails closed instead of passing. + */ export const STAGE_SPECS: Record = { '00-intake': { number: '00', @@ -27,7 +36,13 @@ export const STAGE_SPECS: Record = { type: 'research', gate: 'discovery-complete', promptFile: '00b-discovery.md', - qualityGates: ['true'], + qualityGates: [ + { + kind: 'authority', + capability: 'discovery-complete', + reason: 'discovery completion is attested by an authority; no mechanical check exists', + }, + ], }, '01-board': { number: '01', @@ -36,7 +51,13 @@ export const STAGE_SPECS: Record = { type: 'review', gate: 'board-approval', promptFile: '01-board.md', - qualityGates: [{ type: 'ci-pipeline', command: 'board-approval (via board-tasks)' }], + qualityGates: [ + { + kind: 'authority', + capability: 'board-approval', + reason: 'board approval is a board/human decision; no mechanical gate exists', + }, + ], }, '01b-brief-analyzer': { number: '01b', @@ -45,7 +66,13 @@ export const STAGE_SPECS: Record = { type: 'research', gate: 'brief-analysis-complete', promptFile: '01-board.md', - qualityGates: ['true'], + qualityGates: [ + { + kind: 'authority', + capability: 'brief-analysis-complete', + reason: 'brief analysis completion is attested by an authority; no mechanical check exists', + }, + ], }, '02-planning-1': { number: '02', @@ -54,7 +81,13 @@ export const STAGE_SPECS: Record = { type: 'research', gate: 'architecture-approval', promptFile: '02-planning-1-architecture.md', - qualityGates: ['true'], + qualityGates: [ + { + kind: 'authority', + capability: 'architecture-approval', + reason: 'ADR approval requires authority sign-off; no mechanical check exists', + }, + ], }, '03-planning-2': { number: '03', @@ -63,7 +96,14 @@ export const STAGE_SPECS: Record = { type: 'research', gate: 'implementation-approval', promptFile: '03-planning-2-implementation.md', - qualityGates: ['true'], + qualityGates: [ + { + kind: 'authority', + capability: 'implementation-approval', + reason: + 'implementation spec approval requires authority sign-off; no mechanical check exists', + }, + ], }, '04-planning-3': { number: '04', @@ -72,7 +112,14 @@ export const STAGE_SPECS: Record = { type: 'research', gate: 'decomposition-approval', promptFile: '04-planning-3-decomposition.md', - qualityGates: ['true'], + qualityGates: [ + { + kind: 'authority', + capability: 'decomposition-approval', + reason: + 'task decomposition approval requires authority sign-off; no mechanical check exists', + }, + ], }, '05-coding': { number: '05', @@ -92,9 +139,10 @@ export const STAGE_SPECS: Record = { promptFile: '06-review.md', qualityGates: [ { - type: 'ai-review', - command: - 'echo \'{"summary":"review-pass","verdict":"approve","findings":[],"stats":{"blockers":0,"should_fix":0,"suggestions":0}}\'', + kind: 'provider', + capability: 'reviewer', + reason: + 'review verdicts require a wired reviewer provider; synthetic approvals are not permitted', }, ], }, @@ -105,7 +153,13 @@ export const STAGE_SPECS: Record = { type: 'coding', gate: 're-review', promptFile: '07-remediate.md', - qualityGates: ['true'], + qualityGates: [ + { + kind: 'authority', + capability: 're-review', + reason: 'remediation re-review is an approval-based gate; no mechanical check exists', + }, + ], }, '08-test': { number: '08', @@ -123,7 +177,13 @@ export const STAGE_SPECS: Record = { type: 'deploy', gate: 'deploy-verification', promptFile: '09-deploy.md', - qualityGates: [{ type: 'ci-pipeline', command: 'deploy-verification' }], + qualityGates: [ + { + kind: 'provider', + capability: 'ci-pipeline', + reason: 'deploy verification requires a wired CI pipeline provider', + }, + ], }, }; diff --git a/packages/forge/src/errors.ts b/packages/forge/src/errors.ts new file mode 100644 index 00000000..afe88355 --- /dev/null +++ b/packages/forge/src/errors.ts @@ -0,0 +1,46 @@ +/** + * Typed fail-closed capability errors (SDLC-D-035). + * + * A Forge run must fail closed when a required capability (executor, reviewer + * provider, CI pipeline, authority sign-off) is missing. These typed errors + * name the missing capability so callers can distinguish "not wired" from + * ordinary execution failures. + */ + +/** Closed set of typed Forge capability error codes. */ +export const FORGE_ERROR_CODES = [ + 'FORGE_NO_EXECUTOR', + 'FORGE_NO_REVIEWER', + 'FORGE_NO_CI_PIPELINE', + 'FORGE_NO_PROVIDER', + 'FORGE_AUTHORITY_REQUIRED', +] as const; + +export type ForgeErrorCode = (typeof FORGE_ERROR_CODES)[number]; + +/** Raised when a required capability is missing and the pipeline must fail closed. */ +export class ForgeCapabilityError extends Error { + /** Typed error code from the closed FORGE_ERROR_CODES set. */ + readonly code: ForgeErrorCode; + /** The missing capability, e.g. `task-executor`, `reviewer`, `board-approval`. */ + readonly capability: string; + + constructor(code: ForgeErrorCode, capability: string, message: string) { + super(message); + this.name = 'ForgeCapabilityError'; + this.code = code; + this.capability = capability; + } +} + +/** Map a provider gate capability to its typed error code. */ +export function providerErrorCode(capability: string): ForgeErrorCode { + switch (capability) { + case 'reviewer': + return 'FORGE_NO_REVIEWER'; + case 'ci-pipeline': + return 'FORGE_NO_CI_PIPELINE'; + default: + return 'FORGE_NO_PROVIDER'; + } +} diff --git a/packages/forge/src/index.ts b/packages/forge/src/index.ts index 62c765a1..86a1a239 100644 --- a/packages/forge/src/index.ts +++ b/packages/forge/src/index.ts @@ -5,6 +5,13 @@ export type { StageSpec, BriefClass, ClassSource, + ForgeOutcome, + AuthorityGate, + ProviderGate, + ForgeGate, + ForgeGateResult, + ForgeTaskResult, + RunMode, StageStatus, RunManifest, ForgeTaskStatus, @@ -81,5 +88,24 @@ export { getPipelineStatus, } from './pipeline-runner.js'; +// Fail-closed errors and typed outcome model (SDLC-D-035) +export { FORGE_ERROR_CODES, ForgeCapabilityError, providerErrorCode } from './errors.js'; +export type { ForgeErrorCode } from './errors.js'; +export { + isSatisfyingOutcome, + isCapabilityGate, + isCommandGate, + gateLabel, + uniformGateResults, + simulatedGateResults, + waitingGateResults, + blockedGateResults, + evaluateStageGates, +} from './outcomes.js'; +export type { StageEvaluation } from './outcomes.js'; + +// Simulated executor (explicit --simulate only) +export { createSimulatedExecutor } from './simulated-executor.js'; + // CLI export { registerForgeCommand } from './cli.js'; diff --git a/packages/forge/src/outcomes.ts b/packages/forge/src/outcomes.ts new file mode 100644 index 00000000..07b3fc5e --- /dev/null +++ b/packages/forge/src/outcomes.ts @@ -0,0 +1,147 @@ +import type { GateEntry } from '@mosaicstack/macp'; + +import type { + AuthorityGate, + ForgeGate, + ForgeGateResult, + ForgeOutcome, + ForgeTaskResult, + ProviderGate, +} from './types.js'; + +/** + * Gate and dependency satisfaction predicate (SDLC-D-035). + * + * ONLY a verified `passed` outcome satisfies. Every other member of the closed + * outcome set — including `simulated` — is non-satisfying, so a simulated or + * authority-blocked result can never be read as success-by-verification. + */ +export function isSatisfyingOutcome(outcome: ForgeOutcome): boolean { + return outcome === 'passed'; +} + +/** Whether a gate is an authority or provider gate (capability-based, command-less). */ +export function isCapabilityGate(gate: ForgeGate): gate is AuthorityGate | ProviderGate { + if (typeof gate !== 'object' || gate === null) return false; + const kind = (gate as Record)['kind']; + return kind === 'authority' || kind === 'provider'; +} + +/** Whether a gate definition carries a real command a mechanical runner can execute. */ +export function isCommandGate(gate: ForgeGate): gate is string | GateEntry { + if (typeof gate === 'string') { + return gate.trim().length > 0; + } + if (isCapabilityGate(gate)) { + // Authority and provider gates are satisfied by a capability, not a command. + return false; + } + return typeof gate.command === 'string' && gate.command.trim().length > 0; +} + +/** Typed label identifying a gate in results and logs. */ +export function gateLabel(gate: ForgeGate): string { + if (typeof gate === 'string') return gate; + if (isCapabilityGate(gate)) return `${gate.kind}:${gate.capability}`; + return gate.command || gate.type || 'unnamed-gate'; +} + +/** Reason string stamped on every simulated gate result. */ +export const SIMULATED_GATE_REASON = + 'simulated execution (--simulate): gate was not evaluated by a real implementation'; + +/** Build typed gate results with a uniform outcome for a stage's declared gates. */ +export function uniformGateResults( + gates: ForgeGate[], + outcome: ForgeOutcome, + reason: string, +): ForgeGateResult[] { + return gates.map((gate) => ({ gate: gateLabel(gate), outcome, reason })); +} + +/** Typed simulated gate results — used exclusively in `--simulate` runs. */ +export function simulatedGateResults(gates: ForgeGate[]): ForgeGateResult[] { + return uniformGateResults(gates, 'simulated', SIMULATED_GATE_REASON); +} + +/** Typed waiting-for-authority gate results for approval-based stages. */ +export function waitingGateResults(gates: ForgeGate[], reason: string): ForgeGateResult[] { + return uniformGateResults(gates, 'waiting-for-authority', reason); +} + +/** Typed blocked gate results for stages whose provider capability is not wired. */ +export function blockedGateResults(gates: ForgeGate[], reason: string): ForgeGateResult[] { + return uniformGateResults(gates, 'blocked', reason); +} + +/** Outcome of evaluating a completed stage in normal mode. */ +export interface StageEvaluation { + outcome: ForgeOutcome; + reason: string; + gateResults: ForgeGateResult[]; +} + +/** + * Evaluate a stage's declared gates against the executor's typed result. + * + * Fail-closed mapping: + * - a `simulated` task or gate outcome in normal mode maps to `error` + * - a missing gate result for a required command gate maps to `blocked` + * - a non-passing task outcome propagates as the stage outcome + * - only verified `passed` task and gate outcomes yield a `passed` stage + */ +export function evaluateStageGates( + stageName: string, + gates: ForgeGate[], + result: ForgeTaskResult, +): StageEvaluation { + const gateResults = result.gate_results ?? []; + + if (result.outcome === 'simulated') { + return { + outcome: 'error', + reason: `executor reported a simulated outcome for stage '${stageName}' in normal mode — refusing to treat simulated results as verified`, + gateResults, + }; + } + + if (!isSatisfyingOutcome(result.outcome)) { + return { + outcome: result.outcome, + reason: `task outcome is '${result.outcome}': ${result.reason}`, + gateResults, + }; + } + + for (const gate of gates) { + // Authority and provider gates are pre-flighted before execution; they have + // no mechanical result to verify here. + if (!isCommandGate(gate)) continue; + + const label = gateLabel(gate); + const gateResult = gateResults.find((r) => r.gate === label); + if (!gateResult) { + return { + outcome: 'blocked', + reason: `no gate result was reported for required gate '${label}' (stage '${stageName}')`, + gateResults, + }; + } + if (!isSatisfyingOutcome(gateResult.outcome)) { + return { + outcome: gateResult.outcome === 'simulated' ? 'error' : gateResult.outcome, + reason: `gate '${label}' outcome is '${gateResult.outcome}': ${gateResult.reason}`, + gateResults, + }; + } + } + + return { + outcome: 'passed', + reason: + gates.length === 0 + ? "stage declares no gates; task outcome 'passed' accepted" + : 'all declared gates verified passed', + gateResults, + }; +} diff --git a/packages/forge/src/pipeline-runner.ts b/packages/forge/src/pipeline-runner.ts index e43381df..a6d46a80 100644 --- a/packages/forge/src/pipeline-runner.ts +++ b/packages/forge/src/pipeline-runner.ts @@ -1,18 +1,33 @@ import fs from 'node:fs'; import path from 'node:path'; -import { STAGE_SEQUENCE } from './constants.js'; +import { STAGE_SEQUENCE, STAGE_SPECS } from './constants.js'; import { determineBriefClass, stagesForClass } from './brief-classifier.js'; +import { ForgeCapabilityError, providerErrorCode } from './errors.js'; +import { + blockedGateResults, + evaluateStageGates, + isCapabilityGate, + simulatedGateResults, + waitingGateResults, +} from './outcomes.js'; import { mapStageToTask } from './stage-adapter.js'; +import { createSimulatedExecutor } from './simulated-executor.js'; import type { ForgeTask, + ForgeTaskResult, PipelineOptions, PipelineResult, RunManifest, + RunMode, StageStatus, TaskExecutor, } from './types.js'; +/** Reason stamped on stages that complete under explicit simulation. */ +const SIMULATED_STAGE_REASON = + 'simulated execution (--simulate): stage was not executed by a real executor'; + /** * Generate a timestamp-based run ID. */ @@ -47,6 +62,7 @@ function createManifest(opts: { briefClass: RunManifest['briefClass']; classSource: RunManifest['classSource']; forceBoard: boolean; + mode: RunMode; runDir: string; }): RunManifest { const ts = nowISO(); @@ -57,6 +73,7 @@ function createManifest(opts: { briefClass: opts.briefClass, classSource: opts.classSource, forceBoard: opts.forceBoard, + mode: opts.mode, createdAt: ts, updatedAt: ts, currentStage: '', @@ -108,20 +125,199 @@ export function selectStages(stages?: string[], skipTo?: string): string[] { return selected.slice(skipIndex); } +/** + * Fail closed when the required executor capability is missing (SDLC-D-035). + */ +function requireExecutor(executor: TaskExecutor | undefined, simulate: boolean): TaskExecutor { + if (executor) return executor; + if (simulate) return createSimulatedExecutor({ log: false }); + throw new ForgeCapabilityError( + 'FORGE_NO_EXECUTOR', + 'task-executor', + 'no task executor is wired; refusing to run the pipeline with a stub executor (fail closed). ' + + 'Pass --simulate to opt into explicitly simulated execution.', + ); +} + +/** + * Pre-flight a stage's gates in normal mode (fail closed, SDLC-D-035). + * + * - authority gates: record a typed `waiting-for-authority` stage result and + * raise FORGE_AUTHORITY_REQUIRED — approval-based gates never pass vacuously. + * - provider gates: record a typed `blocked` stage result and raise the typed + * capability error for the missing provider. + * + * Returns the stage status to record when the pre-flight blocks, or undefined + * when the stage may proceed. + */ +function preflightStageGates( + stageName: string, + manifest: RunManifest, +): { status: StageStatus; error: ForgeCapabilityError } | undefined { + const spec = STAGE_SPECS[stageName]; + if (!spec) throw new Error(`Unknown Forge stage: ${stageName}`); + + for (const gate of spec.qualityGates) { + if (!isCapabilityGate(gate)) continue; + + const startedAt = manifest.stages[stageName]?.startedAt; + const completedAt = nowISO(); + + if (gate.kind === 'authority') { + const reason = `gate '${gate.capability}' requires authority sign-off; no mechanical implementation exists (${gate.reason})`; + return { + status: { + status: 'waiting-for-authority', + reason, + startedAt, + completedAt, + gateResults: waitingGateResults(spec.qualityGates, reason), + }, + error: new ForgeCapabilityError( + 'FORGE_AUTHORITY_REQUIRED', + gate.capability, + `stage '${stageName}' is blocked on authority gate '${gate.capability}': ${gate.reason}. ` + + 'The pipeline fails closed instead of passing vacuously. Record the approval out-of-band ' + + 'or run with --simulate for explicitly simulated execution.', + ), + }; + } + + const reason = `gate '${gate.capability}' requires provider '${gate.capability}' and none is wired (${gate.reason})`; + return { + status: { + status: 'blocked', + reason, + startedAt, + completedAt, + gateResults: blockedGateResults(spec.qualityGates, reason), + }, + error: new ForgeCapabilityError( + providerErrorCode(gate.capability), + gate.capability, + `stage '${stageName}' requires provider '${gate.capability}' which is not wired: ${gate.reason}. ` + + 'The pipeline fails closed instead of passing vacuously.', + ), + }; + } + + return undefined; +} + +/** + * Execute the given stage tasks sequentially, updating the manifest. + * + * Normal mode requires a real executor and evaluates every declared command + * gate through the typed outcome model; any non-verified result fails closed. + * Simulate mode types every stage and gate result as `simulated`. + */ +async function executeStages(opts: { + manifest: RunManifest; + runDir: string; + tasks: ForgeTask[]; + stageNames: string[]; + executor: TaskExecutor; + simulate: boolean; +}): Promise { + const { manifest, runDir, tasks, stageNames, executor, simulate } = opts; + + for (let i = 0; i < tasks.length; i++) { + const task = tasks[i]!; + const stageName = stageNames[i]!; + const spec = STAGE_SPECS[stageName]; + if (!spec) throw new Error(`Unknown Forge stage: ${stageName}`); + + // Update manifest: stage in progress + manifest.currentStage = stageName; + manifest.stages[stageName] = { + status: 'in_progress', + startedAt: nowISO(), + }; + saveManifest(runDir, manifest); + + // Fail-closed pre-flight (normal mode only): authority/provider gates have + // no mechanical implementation and must never pass vacuously. + if (!simulate) { + const blocked = preflightStageGates(stageName, manifest); + if (blocked) { + manifest.stages[stageName] = blocked.status; + manifest.status = + blocked.status.status === 'waiting-for-authority' ? 'waiting-for-authority' : 'failed'; + saveManifest(runDir, manifest); + throw blocked.error; + } + } + + let result: ForgeTaskResult; + try { + await executor.submitTask(task); + result = await executor.waitForCompletion(task.id, task.timeoutSeconds * 1000); + } catch (error) { + // Process errors (including timeouts) map to the fail-closed `error` outcome. + const reason = error instanceof Error ? error.message : String(error); + manifest.stages[stageName] = { + status: 'error', + reason: `executor error: ${reason}`, + startedAt: manifest.stages[stageName]?.startedAt, + completedAt: nowISO(), + gateResults: [], + }; + manifest.status = 'failed'; + saveManifest(runDir, manifest); + throw error instanceof Error ? error : new Error(reason); + } + + if (simulate) { + manifest.stages[stageName] = { + status: 'simulated', + reason: SIMULATED_STAGE_REASON, + startedAt: manifest.stages[stageName]?.startedAt, + completedAt: nowISO(), + gateResults: simulatedGateResults(spec.qualityGates), + }; + saveManifest(runDir, manifest); + continue; + } + + const evaluation = evaluateStageGates(stageName, spec.qualityGates, result); + manifest.stages[stageName] = { + status: evaluation.outcome, + reason: evaluation.reason, + startedAt: manifest.stages[stageName]?.startedAt, + completedAt: nowISO(), + gateResults: evaluation.gateResults, + }; + + if (evaluation.outcome !== 'passed') { + manifest.status = + evaluation.outcome === 'waiting-for-authority' ? 'waiting-for-authority' : 'failed'; + saveManifest(runDir, manifest); + throw new Error(`Stage ${stageName} ${evaluation.outcome}: ${evaluation.reason}`); + } + + saveManifest(runDir, manifest); + } +} + /** * Run the Forge pipeline. * - * 1. Classify the brief - * 2. Generate a run ID and create run directory - * 3. Map stages to tasks and submit to TaskExecutor - * 4. Track manifest with stage statuses - * 5. Return pipeline result + * 1. Fail closed unless a real executor is wired or simulation is explicit + * 2. Classify the brief + * 3. Generate a run ID and create run directory + * 4. Map stages to tasks and submit to TaskExecutor + * 5. Track manifest with typed stage outcomes + * 6. Return pipeline result */ export async function runPipeline( briefPath: string, projectRoot: string, options: PipelineOptions, ): Promise { + const simulate = options.simulate ?? false; + const executor = requireExecutor(options.executor, simulate); + const mode: RunMode = simulate ? 'simulated' : 'normal'; + const resolvedRoot = path.resolve(projectRoot); const resolvedBrief = path.resolve(briefPath); const briefContent = fs.readFileSync(resolvedBrief, 'utf-8'); @@ -146,6 +342,7 @@ export async function runPipeline( briefClass, classSource, forceBoard: options.forceBoard ?? false, + mode, runDir, }); @@ -172,54 +369,10 @@ export async function runPipeline( } // Execute stages - const { executor } = options; - for (let i = 0; i < tasks.length; i++) { - const task = tasks[i]!; - const stageName = selectedStages[i]!; + await executeStages({ manifest, runDir, tasks, stageNames: selectedStages, executor, simulate }); - // Update manifest: stage in progress - manifest.currentStage = stageName; - manifest.stages[stageName] = { - status: 'in_progress', - startedAt: nowISO(), - }; - saveManifest(runDir, manifest); - - try { - await executor.submitTask(task); - const result = await executor.waitForCompletion(task.id, task.timeoutSeconds * 1000); - - // Update manifest: stage completed or failed - const stageStatus: StageStatus = { - status: result.status === 'completed' ? 'passed' : 'failed', - startedAt: manifest.stages[stageName]!.startedAt, - completedAt: nowISO(), - }; - manifest.stages[stageName] = stageStatus; - - if (result.status !== 'completed') { - manifest.status = 'failed'; - saveManifest(runDir, manifest); - throw new Error(`Stage ${stageName} failed with status: ${result.status}`); - } - - saveManifest(runDir, manifest); - } catch (error) { - if (!manifest.stages[stageName]?.completedAt) { - manifest.stages[stageName] = { - status: 'failed', - startedAt: manifest.stages[stageName]?.startedAt, - completedAt: nowISO(), - }; - } - manifest.status = 'failed'; - saveManifest(runDir, manifest); - throw error; - } - } - - // All stages passed - manifest.status = 'completed'; + // All stages reached a terminal state for this mode + manifest.status = simulate ? 'simulated' : 'completed'; saveManifest(runDir, manifest); return { @@ -234,22 +387,30 @@ export async function runPipeline( } /** - * Resume a pipeline from the last incomplete stage. + * Resume a pipeline from the last non-passed stage. */ export async function resumePipeline( runDir: string, - executor: TaskExecutor, + executor?: TaskExecutor, + options?: { simulate?: boolean }, ): Promise { + const simulate = options?.simulate ?? false; + const wiredExecutor = requireExecutor(executor, simulate); + const mode: RunMode = simulate ? 'simulated' : 'normal'; + const manifest = loadManifest(runDir); const resolvedRoot = path.dirname(path.dirname(path.dirname(runDir))); // .forge/runs/{id} → project root const briefContent = fs.readFileSync(manifest.brief, 'utf-8'); const allStages = stagesForClass(manifest.briefClass, manifest.forceBoard); - // Find first non-passed stage + manifest.mode = mode; + + // Find first non-satisfying stage (only a verified `passed` counts as done; + // simulated and waiting-for-authority stages are re-run). const resumeFrom = allStages.find((s) => manifest.stages[s]?.status !== 'passed'); if (!resumeFrom) { - manifest.status = 'completed'; + manifest.status = mode === 'simulated' ? 'simulated' : 'completed'; saveManifest(runDir, manifest); return { runId: manifest.runId, @@ -284,49 +445,16 @@ export async function resumePipeline( tasks.push(task); } - for (let i = 0; i < tasks.length; i++) { - const task = tasks[i]!; - const stageName = remainingStages[i]!; + await executeStages({ + manifest, + runDir, + tasks, + stageNames: remainingStages, + executor: wiredExecutor, + simulate, + }); - manifest.currentStage = stageName; - manifest.stages[stageName] = { - status: 'in_progress', - startedAt: nowISO(), - }; - saveManifest(runDir, manifest); - - try { - await executor.submitTask(task); - const result = await executor.waitForCompletion(task.id, task.timeoutSeconds * 1000); - - manifest.stages[stageName] = { - status: result.status === 'completed' ? 'passed' : 'failed', - startedAt: manifest.stages[stageName]!.startedAt, - completedAt: nowISO(), - }; - - if (result.status !== 'completed') { - manifest.status = 'failed'; - saveManifest(runDir, manifest); - throw new Error(`Stage ${stageName} failed with status: ${result.status}`); - } - - saveManifest(runDir, manifest); - } catch (error) { - if (!manifest.stages[stageName]?.completedAt) { - manifest.stages[stageName] = { - status: 'failed', - startedAt: manifest.stages[stageName]?.startedAt, - completedAt: nowISO(), - }; - } - manifest.status = 'failed'; - saveManifest(runDir, manifest); - throw error; - } - } - - manifest.status = 'completed'; + manifest.status = simulate ? 'simulated' : 'completed'; saveManifest(runDir, manifest); return { diff --git a/packages/forge/src/simulated-executor.ts b/packages/forge/src/simulated-executor.ts new file mode 100644 index 00000000..759c4385 --- /dev/null +++ b/packages/forge/src/simulated-executor.ts @@ -0,0 +1,32 @@ +import type { ForgeTask, ForgeTaskResult, TaskExecutor } from './types.js'; + +/** + * Simulated executor — used ONLY when the caller explicitly passes --simulate. + * + * It submits no real work and returns typed `simulated` results so a simulated + * run can never be confused with a verified one. In normal mode (no --simulate) + * the CLI refuses to run at all with FORGE_NO_EXECUTOR instead of wiring this + * stub (fail closed, SDLC-D-035). + */ +export function createSimulatedExecutor(options?: { log?: boolean }): TaskExecutor { + const log = options?.log ?? true; + return { + async submitTask(task: ForgeTask) { + if (log) console.log(` [forge:simulated] stage submitted: ${task.id} (${task.title})`); + }, + async waitForCompletion(taskId: string): Promise { + if (log) console.log(` [forge:simulated] stage complete: ${taskId}`); + return { + task_id: taskId, + outcome: 'simulated', + reason: 'no executor wired; simulated execution requested via --simulate', + completed_at: new Date().toISOString(), + exit_code: 0, + gate_results: [], + }; + }, + async getTaskStatus() { + return 'completed' as const; + }, + }; +} diff --git a/packages/forge/src/types.ts b/packages/forge/src/types.ts index f339d1de..4d2c8727 100644 --- a/packages/forge/src/types.ts +++ b/packages/forge/src/types.ts @@ -1,4 +1,4 @@ -import type { GateEntry, TaskResult } from '@mosaicstack/macp'; +import type { GateEntry } from '@mosaicstack/macp'; /** Stage dispatch mode. */ export type StageDispatch = 'exec' | 'yolo' | 'pi'; @@ -6,6 +6,58 @@ export type StageDispatch = 'exec' | 'yolo' | 'pi'; /** Stage type — determines agent selection and gate requirements. */ export type StageType = 'research' | 'review' | 'coding' | 'deploy'; +/** + * Typed outcome for every gate and stage evaluation — closed set (SDLC-D-035). + * + * Only `passed` means "verified by a real implementation". `simulated` is + * produced exclusively in explicit `--simulate` runs and is never satisfying. + */ +export type ForgeOutcome = + | 'passed' + | 'failed' + | 'blocked' + | 'error' + | 'waiting-for-authority' + | 'simulated' + | 'not-applicable'; + +/** A gate that requires authority (human/board) sign-off; no mechanical command can satisfy it. */ +export interface AuthorityGate { + kind: 'authority'; + capability: string; + reason: string; +} + +/** A gate that requires a wired provider (e.g. an AI reviewer, CI pipeline) to evaluate. */ +export interface ProviderGate { + kind: 'provider'; + capability: string; + reason: string; +} + +/** Forge quality gate: a real command, an authority sign-off, or a provider-backed check. */ +export type ForgeGate = string | GateEntry | AuthorityGate | ProviderGate; + +/** Typed result of evaluating a single quality gate. */ +export interface ForgeGateResult { + gate: string; + outcome: ForgeOutcome; + reason: string; + exitCode?: number; + output?: string; + timedOut?: boolean; +} + +/** Typed result of a task/stage execution returned by a TaskExecutor. */ +export interface ForgeTaskResult { + task_id: string; + outcome: ForgeOutcome; + reason: string; + completed_at: string; + exit_code: number; + gate_results: ForgeGateResult[]; +} + /** Stage specification — defines a single pipeline stage. */ export interface StageSpec { number: string; @@ -14,7 +66,7 @@ export interface StageSpec { type: StageType; gate: string; promptFile: string; - qualityGates: (string | GateEntry)[]; + qualityGates: ForgeGate[]; } /** Brief classification. */ @@ -25,11 +77,18 @@ export type ClassSource = 'cli' | 'frontmatter' | 'auto'; /** Per-stage status within a run manifest. */ export interface StageStatus { - status: 'pending' | 'in_progress' | 'passed' | 'failed'; + status: 'pending' | 'in_progress' | ForgeOutcome; + /** Why the stage reached its current (terminal) outcome, when applicable. */ + reason?: string; startedAt?: string; completedAt?: string; + /** Typed per-gate results recorded alongside the stage outcome. */ + gateResults?: ForgeGateResult[]; } +/** Execution mode of a run. */ +export type RunMode = 'normal' | 'simulated'; + /** Run manifest — persisted to disk as manifest.json. */ export interface RunManifest { runId: string; @@ -38,10 +97,23 @@ export interface RunManifest { briefClass: BriefClass; classSource: ClassSource; forceBoard: boolean; + /** + * Execution mode. `simulated` runs stub execution; their results are typed + * `simulated` and must never be read as verified success. Optional because + * manifests written before this field existed default to `normal`. + */ + mode?: RunMode; createdAt: string; updatedAt: string; currentStage: string; - status: 'in_progress' | 'completed' | 'failed' | 'interrupted' | 'rejected'; + status: + | 'in_progress' + | 'completed' + | 'failed' + | 'interrupted' + | 'rejected' + | 'simulated' + | 'waiting-for-authority'; stages: Record; } @@ -65,7 +137,7 @@ export interface ForgeTask { briefPath: string; resultPath: string; timeoutSeconds: number; - qualityGates: (string | GateEntry)[]; + qualityGates: ForgeGate[]; worktree?: string; command?: string; dependsOn?: string[]; @@ -76,7 +148,7 @@ export interface ForgeTask { /** Abstract task executor — decouples from packages/coord. */ export interface TaskExecutor { submitTask(task: ForgeTask): Promise; - waitForCompletion(taskId: string, timeoutMs: number): Promise; + waitForCompletion(taskId: string, timeoutMs: number): Promise; getTaskStatus(taskId: string): Promise; } @@ -122,7 +194,16 @@ export interface PipelineOptions { stages?: string[]; skipTo?: string; dryRun?: boolean; - executor: TaskExecutor; + /** + * Real task executor. Required in normal mode: the pipeline fails closed + * with FORGE_NO_EXECUTOR when it is absent. + */ + executor?: TaskExecutor; + /** + * Explicit opt-in to simulated execution. Every stage and gate result is + * typed `simulated` and is never satisfying. + */ + simulate?: boolean; } /** Pipeline run result. */