Compare commits
1
Commits
| Author | SHA1 | Date | |
|---|---|---|---|
|
|
b09589f02c |
@@ -0,0 +1,253 @@
|
|||||||
|
import { mkdirSync, readFileSync, rmSync } from 'node:fs';
|
||||||
|
import { join } from 'node:path';
|
||||||
|
import { tmpdir } from 'node:os';
|
||||||
|
import { randomUUID } from 'node:crypto';
|
||||||
|
import { describe, it, expect, beforeEach, afterEach } from 'vitest';
|
||||||
|
import { normalizeGate, countAIFindings, runGate, runGates } from '../src/gate-runner.js';
|
||||||
|
|
||||||
|
function makeTmpDir(): string {
|
||||||
|
const dir = join(tmpdir(), `macp-gate-${randomUUID()}`);
|
||||||
|
mkdirSync(dir, { recursive: true });
|
||||||
|
return dir;
|
||||||
|
}
|
||||||
|
|
||||||
|
describe('normalizeGate', () => {
|
||||||
|
it('normalizes a string to mechanical gate', () => {
|
||||||
|
expect(normalizeGate('echo test')).toEqual({
|
||||||
|
command: 'echo test',
|
||||||
|
type: 'mechanical',
|
||||||
|
fail_on: 'blocker',
|
||||||
|
});
|
||||||
|
});
|
||||||
|
|
||||||
|
it('normalizes an object gate with defaults', () => {
|
||||||
|
expect(normalizeGate({ command: 'lint' })).toEqual({
|
||||||
|
command: 'lint',
|
||||||
|
type: 'mechanical',
|
||||||
|
fail_on: 'blocker',
|
||||||
|
});
|
||||||
|
});
|
||||||
|
|
||||||
|
it('preserves explicit type and fail_on', () => {
|
||||||
|
expect(normalizeGate({ command: 'review', type: 'ai-review', fail_on: 'any' })).toEqual({
|
||||||
|
command: 'review',
|
||||||
|
type: 'ai-review',
|
||||||
|
fail_on: 'any',
|
||||||
|
});
|
||||||
|
});
|
||||||
|
|
||||||
|
it('handles non-string/non-object input', () => {
|
||||||
|
expect(normalizeGate(42)).toEqual({ command: '', type: 'mechanical', fail_on: 'blocker' });
|
||||||
|
expect(normalizeGate(null)).toEqual({ command: '', type: 'mechanical', fail_on: 'blocker' });
|
||||||
|
});
|
||||||
|
});
|
||||||
|
|
||||||
|
describe('countAIFindings', () => {
|
||||||
|
it('returns zeros for non-object', () => {
|
||||||
|
expect(countAIFindings(null)).toEqual({ blockers: 0, total: 0 });
|
||||||
|
expect(countAIFindings('string')).toEqual({ blockers: 0, total: 0 });
|
||||||
|
expect(countAIFindings([])).toEqual({ blockers: 0, total: 0 });
|
||||||
|
});
|
||||||
|
|
||||||
|
it('counts from stats block', () => {
|
||||||
|
const output = { stats: { blockers: 2, should_fix: 3, suggestions: 1 } };
|
||||||
|
expect(countAIFindings(output)).toEqual({ blockers: 2, total: 6 });
|
||||||
|
});
|
||||||
|
|
||||||
|
it('counts from findings array when stats has no blockers', () => {
|
||||||
|
const output = {
|
||||||
|
stats: { blockers: 0 },
|
||||||
|
findings: [{ severity: 'blocker' }, { severity: 'warning' }, { severity: 'blocker' }],
|
||||||
|
};
|
||||||
|
expect(countAIFindings(output)).toEqual({ blockers: 2, total: 3 });
|
||||||
|
});
|
||||||
|
|
||||||
|
it('uses stats blockers over findings array when stats has blockers', () => {
|
||||||
|
const output = {
|
||||||
|
stats: { blockers: 5 },
|
||||||
|
findings: [{ severity: 'blocker' }, { severity: 'warning' }],
|
||||||
|
};
|
||||||
|
// stats.blockers = 5, total from stats = 5+0+0 = 5, findings not used for total since stats total is non-zero
|
||||||
|
expect(countAIFindings(output)).toEqual({ blockers: 5, total: 5 });
|
||||||
|
});
|
||||||
|
|
||||||
|
it('counts findings length as total when stats has zero total', () => {
|
||||||
|
const output = {
|
||||||
|
findings: [{ severity: 'warning' }, { severity: 'info' }],
|
||||||
|
};
|
||||||
|
expect(countAIFindings(output)).toEqual({ blockers: 0, total: 2 });
|
||||||
|
});
|
||||||
|
});
|
||||||
|
|
||||||
|
describe('runGate', () => {
|
||||||
|
let tmp: string;
|
||||||
|
let logPath: string;
|
||||||
|
|
||||||
|
beforeEach(() => {
|
||||||
|
tmp = makeTmpDir();
|
||||||
|
logPath = join(tmp, 'gate.log');
|
||||||
|
});
|
||||||
|
|
||||||
|
afterEach(() => {
|
||||||
|
rmSync(tmp, { recursive: true, force: true });
|
||||||
|
});
|
||||||
|
|
||||||
|
it('passes mechanical gate on exit 0', () => {
|
||||||
|
const result = runGate('echo hello', tmp, logPath, 30);
|
||||||
|
expect(result.passed).toBe(true);
|
||||||
|
expect(result.exit_code).toBe(0);
|
||||||
|
expect(result.type).toBe('mechanical');
|
||||||
|
expect(result.output).toContain('hello');
|
||||||
|
});
|
||||||
|
|
||||||
|
it('fails mechanical gate on non-zero exit', () => {
|
||||||
|
const result = runGate('exit 1', tmp, logPath, 30);
|
||||||
|
expect(result.passed).toBe(false);
|
||||||
|
expect(result.exit_code).toBe(1);
|
||||||
|
});
|
||||||
|
|
||||||
|
it('ci-pipeline always passes', () => {
|
||||||
|
const result = runGate({ command: 'anything', type: 'ci-pipeline' }, tmp, logPath, 30);
|
||||||
|
expect(result.passed).toBe(true);
|
||||||
|
expect(result.type).toBe('ci-pipeline');
|
||||||
|
expect(result.output).toBe('CI pipeline gate placeholder');
|
||||||
|
});
|
||||||
|
|
||||||
|
it('empty command passes', () => {
|
||||||
|
const result = runGate({ command: '' }, tmp, logPath, 30);
|
||||||
|
expect(result.passed).toBe(true);
|
||||||
|
});
|
||||||
|
|
||||||
|
it('ai-review gate parses JSON output', () => {
|
||||||
|
const json = JSON.stringify({ stats: { blockers: 0, should_fix: 1 } });
|
||||||
|
const result = runGate({ command: `echo '${json}'`, type: 'ai-review' }, tmp, logPath, 30);
|
||||||
|
expect(result.passed).toBe(true);
|
||||||
|
expect(result.blockers).toBe(0);
|
||||||
|
expect(result.findings).toBe(1);
|
||||||
|
});
|
||||||
|
|
||||||
|
it('ai-review gate fails on blockers', () => {
|
||||||
|
const json = JSON.stringify({ stats: { blockers: 2 } });
|
||||||
|
const result = runGate({ command: `echo '${json}'`, type: 'ai-review' }, tmp, logPath, 30);
|
||||||
|
expect(result.passed).toBe(false);
|
||||||
|
expect(result.blockers).toBe(2);
|
||||||
|
});
|
||||||
|
|
||||||
|
it('ai-review gate with fail_on=any fails on any findings', () => {
|
||||||
|
const json = JSON.stringify({ stats: { blockers: 0, should_fix: 1 } });
|
||||||
|
const result = runGate(
|
||||||
|
{ command: `echo '${json}'`, type: 'ai-review', fail_on: 'any' },
|
||||||
|
tmp,
|
||||||
|
logPath,
|
||||||
|
30,
|
||||||
|
);
|
||||||
|
expect(result.passed).toBe(false);
|
||||||
|
expect(result.fail_on).toBe('any');
|
||||||
|
});
|
||||||
|
|
||||||
|
it('ai-review gate fails on invalid JSON output', () => {
|
||||||
|
const result = runGate({ command: 'echo "not json"', type: 'ai-review' }, tmp, logPath, 30);
|
||||||
|
expect(result.passed).toBe(false);
|
||||||
|
expect(result.parse_error).toBeDefined();
|
||||||
|
});
|
||||||
|
|
||||||
|
it('writes to log file', () => {
|
||||||
|
runGate('echo logged', tmp, logPath, 30);
|
||||||
|
const log = readFileSync(logPath, 'utf-8');
|
||||||
|
expect(log).toContain('COMMAND: echo logged');
|
||||||
|
expect(log).toContain('logged');
|
||||||
|
expect(log).toContain('EXIT:');
|
||||||
|
});
|
||||||
|
});
|
||||||
|
|
||||||
|
describe('runGates', () => {
|
||||||
|
let tmp: string;
|
||||||
|
let logPath: string;
|
||||||
|
let eventsPath: string;
|
||||||
|
|
||||||
|
beforeEach(() => {
|
||||||
|
tmp = makeTmpDir();
|
||||||
|
logPath = join(tmp, 'gates.log');
|
||||||
|
eventsPath = join(tmp, 'events.ndjson');
|
||||||
|
});
|
||||||
|
|
||||||
|
afterEach(() => {
|
||||||
|
rmSync(tmp, { recursive: true, force: true });
|
||||||
|
});
|
||||||
|
|
||||||
|
it('runs multiple gates and returns results', () => {
|
||||||
|
const { allPassed, gateResults } = runGates(
|
||||||
|
['echo one', 'echo two'],
|
||||||
|
tmp,
|
||||||
|
logPath,
|
||||||
|
30,
|
||||||
|
eventsPath,
|
||||||
|
'task-1',
|
||||||
|
);
|
||||||
|
expect(allPassed).toBe(true);
|
||||||
|
expect(gateResults).toHaveLength(2);
|
||||||
|
});
|
||||||
|
|
||||||
|
it('reports failure when any gate fails', () => {
|
||||||
|
const { allPassed, gateResults } = runGates(
|
||||||
|
['echo ok', 'exit 1'],
|
||||||
|
tmp,
|
||||||
|
logPath,
|
||||||
|
30,
|
||||||
|
eventsPath,
|
||||||
|
'task-2',
|
||||||
|
);
|
||||||
|
expect(allPassed).toBe(false);
|
||||||
|
expect(gateResults[0]!.passed).toBe(true);
|
||||||
|
expect(gateResults[1]!.passed).toBe(false);
|
||||||
|
});
|
||||||
|
|
||||||
|
it('emits events for each gate', () => {
|
||||||
|
runGates(['echo test'], tmp, logPath, 30, eventsPath, 'task-3');
|
||||||
|
const events = readFileSync(eventsPath, 'utf-8')
|
||||||
|
.trim()
|
||||||
|
.split('\n')
|
||||||
|
.map((l) => JSON.parse(l));
|
||||||
|
expect(events).toHaveLength(2); // started + passed
|
||||||
|
expect(events[0].event_type).toBe('rail.check.started');
|
||||||
|
expect(events[1].event_type).toBe('rail.check.passed');
|
||||||
|
});
|
||||||
|
|
||||||
|
it('skips gates with empty command (non ci-pipeline)', () => {
|
||||||
|
const { gateResults } = runGates(
|
||||||
|
[{ command: '', type: 'mechanical' }, 'echo real'],
|
||||||
|
tmp,
|
||||||
|
logPath,
|
||||||
|
30,
|
||||||
|
eventsPath,
|
||||||
|
'task-4',
|
||||||
|
);
|
||||||
|
expect(gateResults).toHaveLength(1);
|
||||||
|
});
|
||||||
|
|
||||||
|
it('does not skip ci-pipeline even with empty command', () => {
|
||||||
|
const { gateResults } = runGates(
|
||||||
|
[{ command: '', type: 'ci-pipeline' }],
|
||||||
|
tmp,
|
||||||
|
logPath,
|
||||||
|
30,
|
||||||
|
eventsPath,
|
||||||
|
'task-5',
|
||||||
|
);
|
||||||
|
expect(gateResults).toHaveLength(1);
|
||||||
|
expect(gateResults[0]!.passed).toBe(true);
|
||||||
|
});
|
||||||
|
|
||||||
|
it('emits failed event with correct message', () => {
|
||||||
|
runGates(['exit 42'], tmp, logPath, 30, eventsPath, 'task-6');
|
||||||
|
const events = readFileSync(eventsPath, 'utf-8')
|
||||||
|
.trim()
|
||||||
|
.split('\n')
|
||||||
|
.map((l) => JSON.parse(l));
|
||||||
|
const failEvent = events.find(
|
||||||
|
(e: Record<string, unknown>) => e.event_type === 'rail.check.failed',
|
||||||
|
);
|
||||||
|
expect(failEvent).toBeDefined();
|
||||||
|
expect(failEvent.message).toContain('Gate failed (');
|
||||||
|
});
|
||||||
|
});
|
||||||
@@ -1,8 +1,5 @@
|
|||||||
import { describe, it, expect, afterEach, beforeEach, vi } from 'vitest';
|
import { describe, it, expect } from 'vitest';
|
||||||
import { Command } from 'commander';
|
import { Command } from 'commander';
|
||||||
import fs from 'node:fs';
|
|
||||||
import os from 'node:os';
|
|
||||||
import path from 'node:path';
|
|
||||||
import { registerMacpCommand } from './cli.js';
|
import { registerMacpCommand } from './cli.js';
|
||||||
|
|
||||||
describe('registerMacpCommand', () => {
|
describe('registerMacpCommand', () => {
|
||||||
@@ -78,162 +75,3 @@ describe('registerMacpCommand', () => {
|
|||||||
expect(topLevel).toContain('events');
|
expect(topLevel).toContain('events');
|
||||||
});
|
});
|
||||||
});
|
});
|
||||||
|
|
||||||
/**
|
|
||||||
* RI-N2 fail-closed CLI behavior: an unimplemented capability is a failure,
|
|
||||||
* never a success. Every stub exits nonzero with a typed message, and the
|
|
||||||
* implemented `macp gate` mirrors the typed gate-runner states.
|
|
||||||
*/
|
|
||||||
describe('registerMacpCommand fail-closed (RI-N2)', () => {
|
|
||||||
let tmpDir: string;
|
|
||||||
|
|
||||||
function buildProgram(): Command {
|
|
||||||
const program = new Command();
|
|
||||||
program.exitOverride();
|
|
||||||
program.configureOutput({ writeErr: () => {} });
|
|
||||||
registerMacpCommand(program);
|
|
||||||
return program;
|
|
||||||
}
|
|
||||||
|
|
||||||
beforeEach(() => {
|
|
||||||
tmpDir = fs.mkdtempSync(path.join(os.tmpdir(), 'macp-cli-failclosed-'));
|
|
||||||
process.exitCode = 0;
|
|
||||||
});
|
|
||||||
|
|
||||||
afterEach(() => {
|
|
||||||
process.exitCode = 0;
|
|
||||||
fs.rmSync(tmpDir, { recursive: true, force: true });
|
|
||||||
});
|
|
||||||
|
|
||||||
it('macp tasks list exits nonzero (unimplemented capability)', async () => {
|
|
||||||
const program = buildProgram();
|
|
||||||
await program.parseAsync(['macp', 'tasks', 'list'], { from: 'user' });
|
|
||||||
expect(process.exitCode).not.toBe(0);
|
|
||||||
});
|
|
||||||
|
|
||||||
it('macp submit exits nonzero with a typed MACP_NOT_IMPLEMENTED message', async () => {
|
|
||||||
const program = buildProgram();
|
|
||||||
const errSpy = vi.spyOn(console, 'error').mockImplementation(() => {});
|
|
||||||
try {
|
|
||||||
await program.parseAsync(['macp', 'submit', 'spec.json'], { from: 'user' });
|
|
||||||
expect(process.exitCode).not.toBe(0);
|
|
||||||
const errText = errSpy.mock.calls.map((c) => String(c[0])).join('\n');
|
|
||||||
expect(errText).toContain('MACP_NOT_IMPLEMENTED');
|
|
||||||
} finally {
|
|
||||||
errSpy.mockRestore();
|
|
||||||
}
|
|
||||||
});
|
|
||||||
|
|
||||||
it('macp events tail exits nonzero (unimplemented capability)', async () => {
|
|
||||||
const program = buildProgram();
|
|
||||||
await program.parseAsync(['macp', 'events', 'tail'], { from: 'user' });
|
|
||||||
expect(process.exitCode).not.toBe(0);
|
|
||||||
});
|
|
||||||
|
|
||||||
it('macp gate runs a green inline command and exits 0', async () => {
|
|
||||||
const program = buildProgram();
|
|
||||||
await program.parseAsync(
|
|
||||||
[
|
|
||||||
'macp',
|
|
||||||
'gate',
|
|
||||||
'exit 0',
|
|
||||||
'--cwd',
|
|
||||||
tmpDir,
|
|
||||||
'--log',
|
|
||||||
path.join(tmpDir, 'g.log'),
|
|
||||||
'--timeout',
|
|
||||||
'10',
|
|
||||||
],
|
|
||||||
{ from: 'user' },
|
|
||||||
);
|
|
||||||
expect(process.exitCode).toBe(0);
|
|
||||||
});
|
|
||||||
|
|
||||||
it('macp gate exits nonzero on a failing command', async () => {
|
|
||||||
const program = buildProgram();
|
|
||||||
await program.parseAsync(
|
|
||||||
[
|
|
||||||
'macp',
|
|
||||||
'gate',
|
|
||||||
'exit 9',
|
|
||||||
'--cwd',
|
|
||||||
tmpDir,
|
|
||||||
'--log',
|
|
||||||
path.join(tmpDir, 'g.log'),
|
|
||||||
'--timeout',
|
|
||||||
'10',
|
|
||||||
],
|
|
||||||
{ from: 'user' },
|
|
||||||
);
|
|
||||||
expect(process.exitCode).not.toBe(0);
|
|
||||||
});
|
|
||||||
|
|
||||||
it('macp gate with an unimplemented ci-pipeline capability exits nonzero', async () => {
|
|
||||||
const program = buildProgram();
|
|
||||||
const specPath = path.join(tmpDir, 'gates.json');
|
|
||||||
fs.writeFileSync(specPath, JSON.stringify([{ type: 'ci-pipeline' }]));
|
|
||||||
await program.parseAsync(
|
|
||||||
[
|
|
||||||
'macp',
|
|
||||||
'gate',
|
|
||||||
specPath,
|
|
||||||
'--cwd',
|
|
||||||
tmpDir,
|
|
||||||
'--log',
|
|
||||||
path.join(tmpDir, 'g.log'),
|
|
||||||
'--timeout',
|
|
||||||
'10',
|
|
||||||
],
|
|
||||||
{ from: 'user' },
|
|
||||||
);
|
|
||||||
expect(process.exitCode).not.toBe(0);
|
|
||||||
});
|
|
||||||
|
|
||||||
it('macp gate --simulate completes (exit 0) but reports simulated results', async () => {
|
|
||||||
const program = buildProgram();
|
|
||||||
const logSpy = vi.spyOn(console, 'log').mockImplementation(() => {});
|
|
||||||
try {
|
|
||||||
await program.parseAsync(
|
|
||||||
[
|
|
||||||
'macp',
|
|
||||||
'gate',
|
|
||||||
'exit 0',
|
|
||||||
'--simulate',
|
|
||||||
'--cwd',
|
|
||||||
tmpDir,
|
|
||||||
'--log',
|
|
||||||
path.join(tmpDir, 'g.log'),
|
|
||||||
'--timeout',
|
|
||||||
'10',
|
|
||||||
],
|
|
||||||
{ from: 'user' },
|
|
||||||
);
|
|
||||||
// completes only because the caller explicitly asked to simulate
|
|
||||||
expect(process.exitCode).toBe(0);
|
|
||||||
const outText = logSpy.mock.calls.map((c) => String(c[0])).join('\n');
|
|
||||||
expect(outText).toContain('simulated');
|
|
||||||
expect(outText).toContain('SIMULATED');
|
|
||||||
} finally {
|
|
||||||
logSpy.mockRestore();
|
|
||||||
}
|
|
||||||
});
|
|
||||||
|
|
||||||
it('macp gate with an empty spec exits nonzero with a typed error', async () => {
|
|
||||||
const program = buildProgram();
|
|
||||||
await program.parseAsync(
|
|
||||||
[
|
|
||||||
'macp',
|
|
||||||
'gate',
|
|
||||||
' ',
|
|
||||||
'--cwd',
|
|
||||||
tmpDir,
|
|
||||||
'--log',
|
|
||||||
path.join(tmpDir, 'g.log'),
|
|
||||||
'--timeout',
|
|
||||||
'10',
|
|
||||||
],
|
|
||||||
{ from: 'user' },
|
|
||||||
);
|
|
||||||
expect(process.exitCode).not.toBe(0);
|
|
||||||
});
|
|
||||||
});
|
|
||||||
|
|||||||
+19
-129
@@ -1,73 +1,5 @@
|
|||||||
import { existsSync, readFileSync } from 'node:fs';
|
|
||||||
|
|
||||||
import type { Command } from 'commander';
|
import type { Command } from 'commander';
|
||||||
|
|
||||||
import { runGates } from './gate-runner.js';
|
|
||||||
import { MACPCapabilityError, type MacpErrorCode } from './errors.js';
|
|
||||||
|
|
||||||
/**
|
|
||||||
* Load gates from a spec: an existing file (JSON gates array, a JSON object
|
|
||||||
* with `quality_gates`, a JSON gate object, or one command per line) or an
|
|
||||||
* inline command string. Fails closed with a typed capability error when the
|
|
||||||
* spec contains no executable gate definition.
|
|
||||||
*/
|
|
||||||
function loadGateSpec(spec: string): unknown[] {
|
|
||||||
if (existsSync(spec)) {
|
|
||||||
const raw = readFileSync(spec, 'utf-8');
|
|
||||||
try {
|
|
||||||
const parsed = JSON.parse(raw) as unknown;
|
|
||||||
if (Array.isArray(parsed)) {
|
|
||||||
if (parsed.length === 0) {
|
|
||||||
throw new MACPCapabilityError(
|
|
||||||
'MACP_NO_COMMAND',
|
|
||||||
'gate-spec',
|
|
||||||
`gate spec file '${spec}' contains an empty gates array`,
|
|
||||||
);
|
|
||||||
}
|
|
||||||
return parsed;
|
|
||||||
}
|
|
||||||
if (typeof parsed === 'object' && parsed !== null) {
|
|
||||||
const obj = parsed as Record<string, unknown>;
|
|
||||||
if (Array.isArray(obj['quality_gates'])) {
|
|
||||||
return obj['quality_gates'];
|
|
||||||
}
|
|
||||||
return [parsed];
|
|
||||||
}
|
|
||||||
throw new MACPCapabilityError(
|
|
||||||
'MACP_NO_COMMAND',
|
|
||||||
'gate-spec',
|
|
||||||
`gate spec file '${spec}' parsed to ${typeof parsed} — expected a gates array, a task with quality_gates, or a gate object`,
|
|
||||||
);
|
|
||||||
} catch (exc) {
|
|
||||||
if (exc instanceof MACPCapabilityError) throw exc;
|
|
||||||
// Not JSON — treat each non-empty line as a command gate.
|
|
||||||
const lines = raw
|
|
||||||
.split('\n')
|
|
||||||
.map((l) => l.trim())
|
|
||||||
.filter((l) => l.length > 0);
|
|
||||||
if (lines.length > 0) return lines;
|
|
||||||
throw new MACPCapabilityError(
|
|
||||||
'MACP_NO_COMMAND',
|
|
||||||
'gate-spec',
|
|
||||||
`gate spec file '${spec}' contains no gates`,
|
|
||||||
);
|
|
||||||
}
|
|
||||||
}
|
|
||||||
if (spec.trim().length > 0) return [spec];
|
|
||||||
throw new MACPCapabilityError('MACP_NO_COMMAND', 'gate-spec', 'gate spec is empty');
|
|
||||||
}
|
|
||||||
|
|
||||||
/** Print a typed not-implemented failure and exit nonzero (RI-N2 fail-closed). */
|
|
||||||
function notImplemented(subcommand: string, capability: string, hint: string): void {
|
|
||||||
const err = new MACPCapabilityError(
|
|
||||||
'MACP_NOT_IMPLEMENTED',
|
|
||||||
capability,
|
|
||||||
`${subcommand} is not implemented in @mosaicstack/macp yet (${capability} capability absent) — ${hint}`,
|
|
||||||
);
|
|
||||||
console.error(`[macp] ${subcommand}: ${err.message} [${err.code}]`);
|
|
||||||
process.exitCode = 1;
|
|
||||||
}
|
|
||||||
|
|
||||||
/**
|
/**
|
||||||
* Register macp subcommands on an existing Commander program.
|
* Register macp subcommands on an existing Commander program.
|
||||||
* This avoids cross-package Commander version mismatches by using the
|
* This avoids cross-package Commander version mismatches by using the
|
||||||
@@ -92,14 +24,15 @@ export function registerMacpCommand(parent: Command): void {
|
|||||||
'Filter by task type (coding|deploy|research|review|documentation|infrastructure)',
|
'Filter by task type (coding|deploy|research|review|documentation|infrastructure)',
|
||||||
)
|
)
|
||||||
.action((opts: { status?: string; type?: string }) => {
|
.action((opts: { status?: string; type?: string }) => {
|
||||||
// unimplemented capability — a failure, never a success (RI-N2)
|
// not yet wired — task persistence layer is not present in @mosaicstack/macp
|
||||||
|
console.log('[macp] tasks list: not yet wired — use macp package programmatically');
|
||||||
if (opts.status) {
|
if (opts.status) {
|
||||||
console.log(` status filter: ${opts.status}`);
|
console.log(` status filter: ${opts.status}`);
|
||||||
}
|
}
|
||||||
if (opts.type) {
|
if (opts.type) {
|
||||||
console.log(` type filter: ${opts.type}`);
|
console.log(` type filter: ${opts.type}`);
|
||||||
}
|
}
|
||||||
notImplemented('tasks list', 'task-persistence', 'use the macp package programmatically');
|
process.exitCode = 0;
|
||||||
});
|
});
|
||||||
|
|
||||||
// ─── submit ──────────────────────────────────────────────────────────────
|
// ─── submit ──────────────────────────────────────────────────────────────
|
||||||
@@ -108,11 +41,12 @@ export function registerMacpCommand(parent: Command): void {
|
|||||||
.command('submit <path>')
|
.command('submit <path>')
|
||||||
.description('Submit a task from a JSON/YAML spec file')
|
.description('Submit a task from a JSON/YAML spec file')
|
||||||
.action((specPath: string) => {
|
.action((specPath: string) => {
|
||||||
// unimplemented capability — a failure, never a success (RI-N2)
|
// not yet wired — task submission requires a running MACP server
|
||||||
|
console.log('[macp] submit: not yet wired — use macp package programmatically');
|
||||||
console.log(` spec path: ${specPath}`);
|
console.log(` spec path: ${specPath}`);
|
||||||
console.log(' task id: (unavailable — no MACP server connected)');
|
console.log(' task id: (unavailable — no MACP server connected)');
|
||||||
console.log(' status: (unavailable — no MACP server connected)');
|
console.log(' status: (unavailable — no MACP server connected)');
|
||||||
notImplemented('submit', 'macp-server', 'use the macp package programmatically');
|
process.exitCode = 0;
|
||||||
});
|
});
|
||||||
|
|
||||||
// ─── gate ────────────────────────────────────────────────────────────────
|
// ─── gate ────────────────────────────────────────────────────────────────
|
||||||
@@ -124,58 +58,16 @@ export function registerMacpCommand(parent: Command): void {
|
|||||||
.option('--cwd <path>', 'Working directory for gate execution', process.cwd())
|
.option('--cwd <path>', 'Working directory for gate execution', process.cwd())
|
||||||
.option('--log <path>', 'Path to write gate log output', '/tmp/macp-gate.log')
|
.option('--log <path>', 'Path to write gate log output', '/tmp/macp-gate.log')
|
||||||
.option('--timeout <seconds>', 'Gate timeout in seconds', '60')
|
.option('--timeout <seconds>', 'Gate timeout in seconds', '60')
|
||||||
.option(
|
.action((spec: string, opts: { failOn: string; cwd: string; log: string; timeout: string }) => {
|
||||||
'--simulate',
|
// not yet wired — gate execution requires a task context and event sink
|
||||||
'Simulate gates instead of executing them; results are typed simulated and never satisfy a check',
|
console.log('[macp] gate: not yet wired — use macp package programmatically');
|
||||||
)
|
console.log(` spec: ${spec}`);
|
||||||
.action(
|
console.log(` fail-on: ${opts.failOn}`);
|
||||||
(
|
console.log(` cwd: ${opts.cwd}`);
|
||||||
spec: string,
|
console.log(` log: ${opts.log}`);
|
||||||
opts: { failOn: string; cwd: string; log: string; timeout: string; simulate?: boolean },
|
console.log(` timeout: ${opts.timeout}s`);
|
||||||
) => {
|
process.exitCode = 0;
|
||||||
let gates: unknown[];
|
});
|
||||||
try {
|
|
||||||
gates = loadGateSpec(spec);
|
|
||||||
} catch (exc) {
|
|
||||||
if (exc instanceof MACPCapabilityError) {
|
|
||||||
console.error(`[macp] gate: ${exc.message} [${exc.code}]`);
|
|
||||||
} else {
|
|
||||||
console.error(`[macp] gate: ${String(exc)}`);
|
|
||||||
}
|
|
||||||
process.exitCode = 1;
|
|
||||||
return;
|
|
||||||
}
|
|
||||||
|
|
||||||
const timeoutSec = Number.parseInt(opts.timeout, 10) || 60;
|
|
||||||
const eventsPath = `${opts.log}.events.ndjson`;
|
|
||||||
const { state, gateResults } = runGates(
|
|
||||||
gates,
|
|
||||||
opts.cwd,
|
|
||||||
opts.log,
|
|
||||||
timeoutSec,
|
|
||||||
eventsPath,
|
|
||||||
'macp-cli-gate',
|
|
||||||
{
|
|
||||||
simulate: opts.simulate,
|
|
||||||
},
|
|
||||||
);
|
|
||||||
|
|
||||||
for (const r of gateResults) {
|
|
||||||
const label = r.command || r.type;
|
|
||||||
const reason = r.reason ? ` — ${r.reason}` : '';
|
|
||||||
console.log(`[macp] gate ${r.status}: ${label}${reason}`);
|
|
||||||
}
|
|
||||||
if (opts.simulate) {
|
|
||||||
console.log(
|
|
||||||
'[macp] SIMULATED run — every result is typed simulated and can never satisfy a gate, dependency, or release check',
|
|
||||||
);
|
|
||||||
}
|
|
||||||
|
|
||||||
// Simulated runs may complete (exit 0) only because the caller
|
|
||||||
// explicitly passed --simulate; the typed state stays 'simulated'.
|
|
||||||
process.exitCode = state === 'passed' || state === 'simulated' ? 0 : 1;
|
|
||||||
},
|
|
||||||
);
|
|
||||||
|
|
||||||
// ─── events ──────────────────────────────────────────────────────────────
|
// ─── events ──────────────────────────────────────────────────────────────
|
||||||
|
|
||||||
@@ -187,16 +79,14 @@ export function registerMacpCommand(parent: Command): void {
|
|||||||
.option('--file <path>', 'Path to the MACP events NDJSON file')
|
.option('--file <path>', 'Path to the MACP events NDJSON file')
|
||||||
.option('--follow', 'Follow the file for new events (like tail -f)')
|
.option('--follow', 'Follow the file for new events (like tail -f)')
|
||||||
.action((opts: { file?: string; follow?: boolean }) => {
|
.action((opts: { file?: string; follow?: boolean }) => {
|
||||||
// unimplemented capability — a failure, never a success (RI-N2)
|
// not yet wired — event streaming requires a live event source
|
||||||
|
console.log('[macp] events tail: not yet wired — use macp package programmatically');
|
||||||
if (opts.file) {
|
if (opts.file) {
|
||||||
console.log(` file: ${opts.file}`);
|
console.log(` file: ${opts.file}`);
|
||||||
}
|
}
|
||||||
if (opts.follow) {
|
if (opts.follow) {
|
||||||
console.log(' mode: follow');
|
console.log(' mode: follow');
|
||||||
}
|
}
|
||||||
notImplemented('events tail', 'event-source', 'use the macp package programmatically');
|
process.exitCode = 0;
|
||||||
});
|
});
|
||||||
}
|
}
|
||||||
|
|
||||||
// Re-export so CLI consumers can surface typed capability codes.
|
|
||||||
export type { MacpErrorCode };
|
|
||||||
|
|||||||
@@ -1,35 +0,0 @@
|
|||||||
/** Typed error code from the closed MACP_ERROR_CODES set. */
|
|
||||||
export type MacpErrorCode = (typeof MACP_ERROR_CODES)[number];
|
|
||||||
/**
|
|
||||||
* Typed fail-closed capability errors (RI-N2, SDLC-D-035).
|
|
||||||
*
|
|
||||||
* MACP must fail closed when a required capability (executor, reviewer,
|
|
||||||
* command, CI provider, human authority) is absent. These typed codes mirror
|
|
||||||
* the Forge failure vocabulary (FORGE_NO_*) so both packages speak the same
|
|
||||||
* language: an unimplemented capability is a failure, never a stub success.
|
|
||||||
*/
|
|
||||||
|
|
||||||
/** Closed set of typed MACP capability error codes. */
|
|
||||||
export const MACP_ERROR_CODES = [
|
|
||||||
'MACP_NOT_IMPLEMENTED',
|
|
||||||
'MACP_NO_COMMAND',
|
|
||||||
'MACP_NO_REVIEWER',
|
|
||||||
'MACP_NO_CI_PIPELINE',
|
|
||||||
'MACP_NO_PROVIDER',
|
|
||||||
'MACP_AUTHORITY_REQUIRED',
|
|
||||||
] as const;
|
|
||||||
|
|
||||||
/** Raised when a required capability is missing and execution must fail closed. */
|
|
||||||
export class MACPCapabilityError extends Error {
|
|
||||||
/** Typed error code from the closed MACP_ERROR_CODES set. */
|
|
||||||
readonly code: MacpErrorCode;
|
|
||||||
/** The missing capability, e.g. `ci-provider`, `task-persistence`, `command`. */
|
|
||||||
readonly capability: string;
|
|
||||||
|
|
||||||
constructor(code: MacpErrorCode, capability: string, message: string) {
|
|
||||||
super(message);
|
|
||||||
this.name = 'MACPCapabilityError';
|
|
||||||
this.code = code;
|
|
||||||
this.capability = capability;
|
|
||||||
}
|
|
||||||
}
|
|
||||||
@@ -1,429 +0,0 @@
|
|||||||
import fs from 'node:fs';
|
|
||||||
import os from 'node:os';
|
|
||||||
import path from 'node:path';
|
|
||||||
import { afterEach, beforeEach, describe, expect, it } from 'vitest';
|
|
||||||
|
|
||||||
import { countAIFindings, normalizeGate, runGate, runGates } from './gate-runner.js';
|
|
||||||
|
|
||||||
function makeTmpDir(): string {
|
|
||||||
return fs.mkdtempSync(path.join(os.tmpdir(), 'macp-gate-'));
|
|
||||||
}
|
|
||||||
|
|
||||||
describe('normalizeGate', () => {
|
|
||||||
it('normalizes a string to mechanical gate', () => {
|
|
||||||
expect(normalizeGate('echo test')).toEqual({
|
|
||||||
command: 'echo test',
|
|
||||||
type: 'mechanical',
|
|
||||||
fail_on: 'blocker',
|
|
||||||
});
|
|
||||||
});
|
|
||||||
|
|
||||||
it('normalizes an object gate with defaults', () => {
|
|
||||||
expect(normalizeGate({ command: 'lint' })).toEqual({
|
|
||||||
command: 'lint',
|
|
||||||
type: 'mechanical',
|
|
||||||
fail_on: 'blocker',
|
|
||||||
});
|
|
||||||
});
|
|
||||||
|
|
||||||
it('preserves explicit type and fail_on', () => {
|
|
||||||
expect(normalizeGate({ command: 'review', type: 'ai-review', fail_on: 'any' })).toEqual({
|
|
||||||
command: 'review',
|
|
||||||
type: 'ai-review',
|
|
||||||
fail_on: 'any',
|
|
||||||
});
|
|
||||||
});
|
|
||||||
|
|
||||||
it('handles non-string/non-object input', () => {
|
|
||||||
expect(normalizeGate(42)).toEqual({ command: '', type: 'mechanical', fail_on: 'blocker' });
|
|
||||||
expect(normalizeGate(null)).toEqual({ command: '', type: 'mechanical', fail_on: 'blocker' });
|
|
||||||
});
|
|
||||||
});
|
|
||||||
|
|
||||||
describe('countAIFindings', () => {
|
|
||||||
it('returns zeros for non-object', () => {
|
|
||||||
expect(countAIFindings(null)).toEqual({ blockers: 0, total: 0 });
|
|
||||||
expect(countAIFindings('string')).toEqual({ blockers: 0, total: 0 });
|
|
||||||
expect(countAIFindings([])).toEqual({ blockers: 0, total: 0 });
|
|
||||||
});
|
|
||||||
|
|
||||||
it('counts from stats block', () => {
|
|
||||||
const output = { stats: { blockers: 2, should_fix: 3, suggestions: 1 } };
|
|
||||||
expect(countAIFindings(output)).toEqual({ blockers: 2, total: 6 });
|
|
||||||
});
|
|
||||||
|
|
||||||
it('counts from findings array when stats has no blockers', () => {
|
|
||||||
const output = {
|
|
||||||
stats: { blockers: 0 },
|
|
||||||
findings: [{ severity: 'blocker' }, { severity: 'warning' }, { severity: 'blocker' }],
|
|
||||||
};
|
|
||||||
expect(countAIFindings(output)).toEqual({ blockers: 2, total: 3 });
|
|
||||||
});
|
|
||||||
|
|
||||||
it('uses stats blockers over findings array when stats has blockers', () => {
|
|
||||||
const output = {
|
|
||||||
stats: { blockers: 5 },
|
|
||||||
findings: [{ severity: 'blocker' }, { severity: 'warning' }],
|
|
||||||
};
|
|
||||||
// stats.blockers = 5, total from stats = 5+0+0 = 5, findings not used for total since stats total is non-zero
|
|
||||||
expect(countAIFindings(output)).toEqual({ blockers: 5, total: 5 });
|
|
||||||
});
|
|
||||||
|
|
||||||
it('counts findings length as total when stats has zero total', () => {
|
|
||||||
const output = {
|
|
||||||
findings: [{ severity: 'warning' }, { severity: 'info' }],
|
|
||||||
};
|
|
||||||
expect(countAIFindings(output)).toEqual({ blockers: 0, total: 2 });
|
|
||||||
});
|
|
||||||
});
|
|
||||||
|
|
||||||
describe('runGate', () => {
|
|
||||||
let tmp: string;
|
|
||||||
let logPath: string;
|
|
||||||
|
|
||||||
beforeEach(() => {
|
|
||||||
tmp = makeTmpDir();
|
|
||||||
logPath = path.join(tmp, 'gate.log');
|
|
||||||
});
|
|
||||||
|
|
||||||
afterEach(() => {
|
|
||||||
fs.rmSync(tmp, { recursive: true, force: true });
|
|
||||||
});
|
|
||||||
|
|
||||||
it('passes mechanical gate on exit 0', () => {
|
|
||||||
const result = runGate('echo hello', tmp, logPath, 30);
|
|
||||||
expect(result.passed).toBe(true);
|
|
||||||
expect(result.exit_code).toBe(0);
|
|
||||||
expect(result.type).toBe('mechanical');
|
|
||||||
expect(result.output).toContain('hello');
|
|
||||||
});
|
|
||||||
|
|
||||||
it('fails mechanical gate on non-zero exit', () => {
|
|
||||||
const result = runGate('exit 1', tmp, logPath, 30);
|
|
||||||
expect(result.passed).toBe(false);
|
|
||||||
expect(result.exit_code).toBe(1);
|
|
||||||
});
|
|
||||||
|
|
||||||
it('ci-pipeline fails closed without a CI provider (no placeholder pass)', () => {
|
|
||||||
const result = runGate({ command: 'anything', type: 'ci-pipeline' }, tmp, logPath, 30);
|
|
||||||
expect(result.passed).toBe(false);
|
|
||||||
expect(result.status).toBe('capability_failure');
|
|
||||||
expect(result.capability_code).toBe('MACP_NO_CI_PIPELINE');
|
|
||||||
expect(result.type).toBe('ci-pipeline');
|
|
||||||
expect(result.output).not.toBe('CI pipeline gate placeholder');
|
|
||||||
});
|
|
||||||
|
|
||||||
it('empty command is a typed capability failure, never a pass', () => {
|
|
||||||
const result = runGate({ command: '' }, tmp, logPath, 30);
|
|
||||||
expect(result.passed).toBe(false);
|
|
||||||
expect(result.status).toBe('capability_failure');
|
|
||||||
expect(result.capability_code).toBe('MACP_NO_COMMAND');
|
|
||||||
});
|
|
||||||
|
|
||||||
it('ai-review gate parses JSON output', () => {
|
|
||||||
const json = JSON.stringify({ stats: { blockers: 0, should_fix: 1 } });
|
|
||||||
const result = runGate({ command: `echo '${json}'`, type: 'ai-review' }, tmp, logPath, 30);
|
|
||||||
expect(result.passed).toBe(true);
|
|
||||||
expect(result.blockers).toBe(0);
|
|
||||||
expect(result.findings).toBe(1);
|
|
||||||
});
|
|
||||||
|
|
||||||
it('ai-review gate fails on blockers', () => {
|
|
||||||
const json = JSON.stringify({ stats: { blockers: 2 } });
|
|
||||||
const result = runGate({ command: `echo '${json}'`, type: 'ai-review' }, tmp, logPath, 30);
|
|
||||||
expect(result.passed).toBe(false);
|
|
||||||
expect(result.blockers).toBe(2);
|
|
||||||
});
|
|
||||||
|
|
||||||
it('ai-review gate with fail_on=any fails on any findings', () => {
|
|
||||||
const json = JSON.stringify({ stats: { blockers: 0, should_fix: 1 } });
|
|
||||||
const result = runGate(
|
|
||||||
{ command: `echo '${json}'`, type: 'ai-review', fail_on: 'any' },
|
|
||||||
tmp,
|
|
||||||
logPath,
|
|
||||||
30,
|
|
||||||
);
|
|
||||||
expect(result.passed).toBe(false);
|
|
||||||
expect(result.fail_on).toBe('any');
|
|
||||||
});
|
|
||||||
|
|
||||||
it('ai-review gate fails on invalid JSON output', () => {
|
|
||||||
const result = runGate({ command: 'echo "not json"', type: 'ai-review' }, tmp, logPath, 30);
|
|
||||||
expect(result.passed).toBe(false);
|
|
||||||
expect(result.parse_error).toBeDefined();
|
|
||||||
});
|
|
||||||
|
|
||||||
it('writes to log file', () => {
|
|
||||||
runGate('echo logged', tmp, logPath, 30);
|
|
||||||
const log = fs.readFileSync(logPath, 'utf-8');
|
|
||||||
expect(log).toContain('COMMAND: echo logged');
|
|
||||||
expect(log).toContain('logged');
|
|
||||||
expect(log).toContain('EXIT:');
|
|
||||||
});
|
|
||||||
});
|
|
||||||
|
|
||||||
describe('runGates', () => {
|
|
||||||
let tmp: string;
|
|
||||||
let logPath: string;
|
|
||||||
let eventsPath: string;
|
|
||||||
|
|
||||||
beforeEach(() => {
|
|
||||||
tmp = makeTmpDir();
|
|
||||||
logPath = path.join(tmp, 'gates.log');
|
|
||||||
eventsPath = path.join(tmp, 'events.ndjson');
|
|
||||||
});
|
|
||||||
|
|
||||||
afterEach(() => {
|
|
||||||
fs.rmSync(tmp, { recursive: true, force: true });
|
|
||||||
});
|
|
||||||
|
|
||||||
it('runs multiple gates and returns results', () => {
|
|
||||||
const { allPassed, gateResults } = runGates(
|
|
||||||
['echo one', 'echo two'],
|
|
||||||
tmp,
|
|
||||||
logPath,
|
|
||||||
30,
|
|
||||||
eventsPath,
|
|
||||||
'task-1',
|
|
||||||
);
|
|
||||||
expect(allPassed).toBe(true);
|
|
||||||
expect(gateResults).toHaveLength(2);
|
|
||||||
});
|
|
||||||
|
|
||||||
it('reports failure when any gate fails', () => {
|
|
||||||
const { allPassed, gateResults } = runGates(
|
|
||||||
['echo ok', 'exit 1'],
|
|
||||||
tmp,
|
|
||||||
logPath,
|
|
||||||
30,
|
|
||||||
eventsPath,
|
|
||||||
'task-2',
|
|
||||||
);
|
|
||||||
expect(allPassed).toBe(false);
|
|
||||||
expect(gateResults[0]!.passed).toBe(true);
|
|
||||||
expect(gateResults[1]!.passed).toBe(false);
|
|
||||||
});
|
|
||||||
|
|
||||||
it('emits events for each gate', () => {
|
|
||||||
runGates(['echo test'], tmp, logPath, 30, eventsPath, 'task-3');
|
|
||||||
const events = fs
|
|
||||||
.readFileSync(eventsPath, 'utf-8')
|
|
||||||
.trim()
|
|
||||||
.split('\n')
|
|
||||||
.map((l) => JSON.parse(l));
|
|
||||||
expect(events).toHaveLength(2); // started + passed
|
|
||||||
expect(events[0].event_type).toBe('rail.check.started');
|
|
||||||
expect(events[1].event_type).toBe('rail.check.passed');
|
|
||||||
});
|
|
||||||
|
|
||||||
it('does not silently skip gates with empty command — they become capability failures', () => {
|
|
||||||
const { gateResults, allPassed, state } = runGates(
|
|
||||||
[{ command: '', type: 'mechanical' }, 'echo real'],
|
|
||||||
tmp,
|
|
||||||
logPath,
|
|
||||||
30,
|
|
||||||
eventsPath,
|
|
||||||
'task-4',
|
|
||||||
);
|
|
||||||
expect(gateResults).toHaveLength(2);
|
|
||||||
expect(gateResults[0]!.status).toBe('capability_failure');
|
|
||||||
expect(gateResults[1]!.status).toBe('passed');
|
|
||||||
expect(allPassed).toBe(false);
|
|
||||||
expect(state).toBe('capability_failure');
|
|
||||||
});
|
|
||||||
|
|
||||||
it('does not skip ci-pipeline even with empty command — typed capability failure', () => {
|
|
||||||
const { gateResults, allPassed, state } = runGates(
|
|
||||||
[{ command: '', type: 'ci-pipeline' }],
|
|
||||||
tmp,
|
|
||||||
logPath,
|
|
||||||
30,
|
|
||||||
eventsPath,
|
|
||||||
'task-5',
|
|
||||||
);
|
|
||||||
expect(gateResults).toHaveLength(1);
|
|
||||||
expect(gateResults[0]!.passed).toBe(false);
|
|
||||||
expect(gateResults[0]!.status).toBe('capability_failure');
|
|
||||||
expect(allPassed).toBe(false);
|
|
||||||
expect(state).toBe('capability_failure');
|
|
||||||
});
|
|
||||||
|
|
||||||
it('emits failed event with correct message', () => {
|
|
||||||
runGates(['exit 42'], tmp, logPath, 30, eventsPath, 'task-6');
|
|
||||||
const events = fs
|
|
||||||
.readFileSync(eventsPath, 'utf-8')
|
|
||||||
.trim()
|
|
||||||
.split('\n')
|
|
||||||
.map((l) => JSON.parse(l));
|
|
||||||
const failEvent = events.find(
|
|
||||||
(e: Record<string, unknown>) => e.event_type === 'rail.check.failed',
|
|
||||||
);
|
|
||||||
expect(failEvent).toBeDefined();
|
|
||||||
expect(failEvent.message).toContain('Gate failed (');
|
|
||||||
});
|
|
||||||
});
|
|
||||||
|
|
||||||
/**
|
|
||||||
* RI-N2 / SDLC-D-035 fail-closed controls for the MACP gate runner.
|
|
||||||
*
|
|
||||||
* Invariant under test: `passed: true` occurs ONLY when a gate really executed
|
|
||||||
* and really exited green (`status === 'passed'`). Absent capabilities,
|
|
||||||
* manual sign-offs, and simulated runs are typed distinctly and can never
|
|
||||||
* make the aggregate `passed`.
|
|
||||||
*/
|
|
||||||
describe('gate-runner fail-closed (RI-N2)', () => {
|
|
||||||
let tmpDir: string;
|
|
||||||
let logPath: string;
|
|
||||||
let eventsPath: string;
|
|
||||||
|
|
||||||
beforeEach(() => {
|
|
||||||
tmpDir = makeTmpDir();
|
|
||||||
logPath = path.join(tmpDir, 'gate.log');
|
|
||||||
eventsPath = path.join(tmpDir, 'events.ndjson');
|
|
||||||
});
|
|
||||||
|
|
||||||
afterEach(() => {
|
|
||||||
fs.rmSync(tmpDir, { recursive: true, force: true });
|
|
||||||
});
|
|
||||||
|
|
||||||
function run(gates: unknown[], options?: { simulate?: boolean }) {
|
|
||||||
return runGates(gates, tmpDir, logPath, 10, eventsPath, 'spec-task', options);
|
|
||||||
}
|
|
||||||
|
|
||||||
// ─── positive controls ───────────────────────────────────────────────────
|
|
||||||
|
|
||||||
it('a really-executed green command gate still passes', () => {
|
|
||||||
const result = run([{ command: 'exit 0', type: 'mechanical' }]);
|
|
||||||
expect(result.gateResults[0]!.status).toBe('passed');
|
|
||||||
expect(result.gateResults[0]!.passed).toBe(true);
|
|
||||||
expect(result.allPassed).toBe(true);
|
|
||||||
expect(result.state).toBe('passed');
|
|
||||||
});
|
|
||||||
|
|
||||||
it('explicit simulate completes and types every result simulated', () => {
|
|
||||||
const result = run([{ command: 'exit 0', type: 'mechanical' }, 'echo hello'], {
|
|
||||||
simulate: true,
|
|
||||||
});
|
|
||||||
expect(result.gateResults).toHaveLength(2);
|
|
||||||
for (const gate of result.gateResults) {
|
|
||||||
expect(gate.status).toBe('simulated');
|
|
||||||
expect(gate.passed).toBe(false);
|
|
||||||
}
|
|
||||||
expect(result.state).toBe('simulated');
|
|
||||||
});
|
|
||||||
|
|
||||||
it('a really-executed red command gate fails with typed status failed', () => {
|
|
||||||
const result = run([{ command: 'exit 3', type: 'mechanical' }]);
|
|
||||||
expect(result.gateResults[0]!.status).toBe('failed');
|
|
||||||
expect(result.gateResults[0]!.passed).toBe(false);
|
|
||||||
expect(result.allPassed).toBe(false);
|
|
||||||
expect(result.state).toBe('failed');
|
|
||||||
});
|
|
||||||
|
|
||||||
// ─── negative controls — each asserts typed status AND aggregate not passed ──
|
|
||||||
|
|
||||||
it('an empty-command gate is a capability_failure, not skipped and not passed', () => {
|
|
||||||
const result = run([{ command: '', type: 'mechanical' }]);
|
|
||||||
// runGates must not silently skip it — it produces a typed result
|
|
||||||
expect(result.gateResults).toHaveLength(1);
|
|
||||||
const gate = result.gateResults[0]!;
|
|
||||||
expect(gate.status).toBe('capability_failure');
|
|
||||||
expect(gate.capability_code).toBe('MACP_NO_COMMAND');
|
|
||||||
expect(gate.passed).toBe(false);
|
|
||||||
// aggregate is not passed
|
|
||||||
expect(result.allPassed).toBe(false);
|
|
||||||
expect(result.state).toBe('capability_failure');
|
|
||||||
expect(result.state).not.toBe('passed');
|
|
||||||
});
|
|
||||||
|
|
||||||
it('a commandless ai-review gate is a typed MACP_NO_REVIEWER capability_failure', () => {
|
|
||||||
const result = run([{ command: '', type: 'ai-review' }]);
|
|
||||||
expect(result.gateResults[0]!.status).toBe('capability_failure');
|
|
||||||
expect(result.gateResults[0]!.capability_code).toBe('MACP_NO_REVIEWER');
|
|
||||||
expect(result.allPassed).toBe(false);
|
|
||||||
expect(result.state).not.toBe('passed');
|
|
||||||
});
|
|
||||||
|
|
||||||
it('a ci-pipeline gate without a provider implementation is a capability_failure, never a placeholder pass', () => {
|
|
||||||
const result = run([{ command: '', type: 'ci-pipeline' }]);
|
|
||||||
const gate = result.gateResults[0]!;
|
|
||||||
expect(gate.status).toBe('capability_failure');
|
|
||||||
expect(gate.capability_code).toBe('MACP_NO_CI_PIPELINE');
|
|
||||||
expect(gate.passed).toBe(false);
|
|
||||||
// the old false-success placeholder must be gone
|
|
||||||
expect(gate.output).not.toBe('CI pipeline gate placeholder');
|
|
||||||
expect(result.allPassed).toBe(false);
|
|
||||||
expect(result.state).not.toBe('passed');
|
|
||||||
});
|
|
||||||
|
|
||||||
it('a ci-pipeline gate fails closed even alongside an otherwise green run', () => {
|
|
||||||
const result = run(['exit 0', { type: 'ci-pipeline', command: 'fake-ci' }]);
|
|
||||||
expect(result.gateResults[1]!.status).toBe('capability_failure');
|
|
||||||
expect(result.gateResults[0]!.status).toBe('passed');
|
|
||||||
expect(result.allPassed).toBe(false);
|
|
||||||
expect(result.state).toBe('capability_failure');
|
|
||||||
});
|
|
||||||
|
|
||||||
it('a manual gate with no automation enters typed waiting — neither pass nor fail', () => {
|
|
||||||
const result = run([{ type: 'manual' }]);
|
|
||||||
const gate = result.gateResults[0]!;
|
|
||||||
expect(gate.status).toBe('waiting');
|
|
||||||
expect(gate.passed).toBe(false);
|
|
||||||
expect(gate.exit_code).toBe(0);
|
|
||||||
// aggregate is not passed while any gate is waiting
|
|
||||||
expect(result.allPassed).toBe(false);
|
|
||||||
expect(result.state).toBe('waiting');
|
|
||||||
expect(result.state).not.toBe('passed');
|
|
||||||
});
|
|
||||||
|
|
||||||
it('a simulated result can never make the aggregate passed', () => {
|
|
||||||
const result = run(['exit 0', 'exit 0'], { simulate: true });
|
|
||||||
expect(result.gateResults.every((g) => g.status === 'simulated')).toBe(true);
|
|
||||||
expect(result.allPassed).toBe(false);
|
|
||||||
expect(result.state).toBe('simulated');
|
|
||||||
expect(result.state).not.toBe('passed');
|
|
||||||
});
|
|
||||||
|
|
||||||
it('waiting dominates an otherwise green aggregate', () => {
|
|
||||||
const result = run(['exit 0', { type: 'manual' }]);
|
|
||||||
expect(result.allPassed).toBe(false);
|
|
||||||
expect(result.state).toBe('waiting');
|
|
||||||
});
|
|
||||||
});
|
|
||||||
|
|
||||||
describe('runGate fail-closed (RI-N2)', () => {
|
|
||||||
let tmpDir: string;
|
|
||||||
let logPath: string;
|
|
||||||
|
|
||||||
beforeEach(() => {
|
|
||||||
tmpDir = makeTmpDir();
|
|
||||||
logPath = path.join(tmpDir, 'gate.log');
|
|
||||||
});
|
|
||||||
|
|
||||||
afterEach(() => {
|
|
||||||
fs.rmSync(tmpDir, { recursive: true, force: true });
|
|
||||||
});
|
|
||||||
|
|
||||||
it('simulate: true returns a typed simulated result without executing', () => {
|
|
||||||
const result = runGate('this-command-does-not-exist-xyz', tmpDir, logPath, 10, {
|
|
||||||
simulate: true,
|
|
||||||
});
|
|
||||||
expect(result.status).toBe('simulated');
|
|
||||||
expect(result.passed).toBe(false);
|
|
||||||
expect(result.exit_code).toBe(0);
|
|
||||||
});
|
|
||||||
|
|
||||||
it('normal mode executes for real and types a green gate passed', () => {
|
|
||||||
const result = runGate('echo ok', tmpDir, logPath, 10);
|
|
||||||
expect(result.status).toBe('passed');
|
|
||||||
expect(result.passed).toBe(true);
|
|
||||||
expect(result.output).toContain('ok');
|
|
||||||
});
|
|
||||||
|
|
||||||
it('a bare string gate normalizes to mechanical and executes', () => {
|
|
||||||
const result = runGate('exit 7', tmpDir, logPath, 10);
|
|
||||||
expect(result.type).toBe('mechanical');
|
|
||||||
expect(result.status).toBe('failed');
|
|
||||||
expect(result.passed).toBe(false);
|
|
||||||
});
|
|
||||||
});
|
|
||||||
@@ -4,20 +4,7 @@ import { dirname } from 'node:path';
|
|||||||
|
|
||||||
import { emitEvent } from './event-emitter.js';
|
import { emitEvent } from './event-emitter.js';
|
||||||
import { nowISO } from './event-emitter.js';
|
import { nowISO } from './event-emitter.js';
|
||||||
import type { GateResult, GateStatus, RunGatesResult } from './types.js';
|
import type { GateResult } from './types.js';
|
||||||
|
|
||||||
/** Typed reason stamped on every simulated gate result. */
|
|
||||||
export const SIMULATED_GATE_REASON =
|
|
||||||
'simulated execution (explicit simulate opt-in): gate was not evaluated by a real implementation';
|
|
||||||
|
|
||||||
/** Options for gate execution (RI-N2 fail-closed / explicit simulation). */
|
|
||||||
export interface RunGateOptions {
|
|
||||||
/**
|
|
||||||
* Explicit caller opt-in to simulation. Simulated gates are NOT executed;
|
|
||||||
* every result is typed `simulated` and never satisfies anything.
|
|
||||||
*/
|
|
||||||
simulate?: boolean;
|
|
||||||
}
|
|
||||||
|
|
||||||
export interface NormalizedGate {
|
export interface NormalizedGate {
|
||||||
command: string;
|
command: string;
|
||||||
@@ -116,91 +103,36 @@ export function countAIFindings(parsedOutput: unknown): { blockers: number; tota
|
|||||||
return { blockers, total };
|
return { blockers, total };
|
||||||
}
|
}
|
||||||
|
|
||||||
function simulatedResult(gateEntry: NormalizedGate): GateResult {
|
|
||||||
return {
|
|
||||||
command: gateEntry.command,
|
|
||||||
exit_code: 0,
|
|
||||||
type: gateEntry.type,
|
|
||||||
output: SIMULATED_GATE_REASON,
|
|
||||||
timed_out: false,
|
|
||||||
passed: false,
|
|
||||||
status: 'simulated',
|
|
||||||
reason: SIMULATED_GATE_REASON,
|
|
||||||
};
|
|
||||||
}
|
|
||||||
|
|
||||||
function capabilityFailureResult(
|
|
||||||
gateEntry: NormalizedGate,
|
|
||||||
code: GateResult['capability_code'],
|
|
||||||
reason: string,
|
|
||||||
): GateResult {
|
|
||||||
return {
|
|
||||||
command: gateEntry.command,
|
|
||||||
exit_code: 1,
|
|
||||||
type: gateEntry.type,
|
|
||||||
output: '',
|
|
||||||
timed_out: false,
|
|
||||||
passed: false,
|
|
||||||
status: 'capability_failure',
|
|
||||||
capability_code: code,
|
|
||||||
reason,
|
|
||||||
};
|
|
||||||
}
|
|
||||||
|
|
||||||
function waitingResult(gateEntry: NormalizedGate, reason: string): GateResult {
|
|
||||||
return {
|
|
||||||
command: gateEntry.command,
|
|
||||||
exit_code: 0,
|
|
||||||
type: gateEntry.type,
|
|
||||||
output: '',
|
|
||||||
timed_out: false,
|
|
||||||
passed: false,
|
|
||||||
status: 'waiting',
|
|
||||||
capability_code: 'MACP_AUTHORITY_REQUIRED',
|
|
||||||
reason,
|
|
||||||
};
|
|
||||||
}
|
|
||||||
|
|
||||||
export function runGate(
|
export function runGate(
|
||||||
gate: unknown,
|
gate: unknown,
|
||||||
cwd: string,
|
cwd: string,
|
||||||
logPath: string,
|
logPath: string,
|
||||||
timeoutSec: number,
|
timeoutSec: number,
|
||||||
options: RunGateOptions = {},
|
|
||||||
): GateResult {
|
): GateResult {
|
||||||
const gateEntry = normalizeGate(gate);
|
const gateEntry = normalizeGate(gate);
|
||||||
const gateType = gateEntry.type;
|
const gateType = gateEntry.type;
|
||||||
const command = gateEntry.command;
|
const command = gateEntry.command;
|
||||||
|
|
||||||
// Explicit simulation only: never executes, typed simulated, never satisfying.
|
|
||||||
if (options.simulate) {
|
|
||||||
return simulatedResult(gateEntry);
|
|
||||||
}
|
|
||||||
|
|
||||||
// Fail closed: no CI provider implementation exists in @mosaicstack/macp,
|
|
||||||
// so a ci-pipeline gate is an absent capability — never a placeholder pass.
|
|
||||||
if (gateType === 'ci-pipeline') {
|
if (gateType === 'ci-pipeline') {
|
||||||
return capabilityFailureResult(
|
return {
|
||||||
gateEntry,
|
command,
|
||||||
'MACP_NO_CI_PIPELINE',
|
exit_code: 0,
|
||||||
`ci-pipeline gate '${gateEntry.command || gateType}' has no CI provider implementation wired — refusing placeholder pass`,
|
type: gateType,
|
||||||
);
|
output: 'CI pipeline gate placeholder',
|
||||||
|
timed_out: false,
|
||||||
|
passed: true,
|
||||||
|
};
|
||||||
}
|
}
|
||||||
|
|
||||||
if (!command) {
|
if (!command) {
|
||||||
// A manual gate with no automation waits for human sign-off: not pass, not fail.
|
return {
|
||||||
if (gateType === 'manual') {
|
command: '',
|
||||||
return waitingResult(
|
exit_code: 0,
|
||||||
gateEntry,
|
type: gateType,
|
||||||
`manual gate has no automation — waiting for human sign-off (type: ${gateType})`,
|
output: '',
|
||||||
);
|
timed_out: false,
|
||||||
}
|
passed: true,
|
||||||
// Any other commandless gate is an absent capability — never a vacuous pass.
|
};
|
||||||
return capabilityFailureResult(
|
|
||||||
gateEntry,
|
|
||||||
gateType === 'ai-review' ? 'MACP_NO_REVIEWER' : 'MACP_NO_COMMAND',
|
|
||||||
`gate of type '${gateType}' has no command to execute — refusing empty-command pass`,
|
|
||||||
);
|
|
||||||
}
|
}
|
||||||
|
|
||||||
const { exitCode, output, timedOut } = runShell(command, cwd, logPath, timeoutSec);
|
const { exitCode, output, timedOut } = runShell(command, cwd, logPath, timeoutSec);
|
||||||
@@ -211,12 +143,10 @@ export function runGate(
|
|||||||
output,
|
output,
|
||||||
timed_out: timedOut,
|
timed_out: timedOut,
|
||||||
passed: false,
|
passed: false,
|
||||||
status: 'failed',
|
|
||||||
};
|
};
|
||||||
|
|
||||||
if (gateType !== 'ai-review') {
|
if (gateType !== 'ai-review') {
|
||||||
result.passed = exitCode === 0;
|
result.passed = exitCode === 0;
|
||||||
result.status = result.passed ? 'passed' : 'failed';
|
|
||||||
return result;
|
return result;
|
||||||
}
|
}
|
||||||
|
|
||||||
@@ -240,7 +170,6 @@ export function runGate(
|
|||||||
} else {
|
} else {
|
||||||
result.passed = exitCode === 0 && blockers === 0 && !timedOut && parseError === undefined;
|
result.passed = exitCode === 0 && blockers === 0 && !timedOut && parseError === undefined;
|
||||||
}
|
}
|
||||||
result.status = result.passed ? 'passed' : 'failed';
|
|
||||||
|
|
||||||
result.fail_on = failOn;
|
result.fail_on = failOn;
|
||||||
result.blockers = blockers;
|
result.blockers = blockers;
|
||||||
@@ -262,19 +191,16 @@ export function runGates(
|
|||||||
timeoutSec: number,
|
timeoutSec: number,
|
||||||
eventsPath: string,
|
eventsPath: string,
|
||||||
taskId: string,
|
taskId: string,
|
||||||
options: RunGateOptions = {},
|
): { allPassed: boolean; gateResults: GateResult[] } {
|
||||||
): RunGatesResult {
|
let allPassed = true;
|
||||||
const gateResults: GateResult[] = [];
|
const gateResults: GateResult[] = [];
|
||||||
let hasCapabilityFailure = false;
|
|
||||||
let hasSimulated = false;
|
|
||||||
let hasFailed = false;
|
|
||||||
let hasWaiting = false;
|
|
||||||
|
|
||||||
for (const gate of gates) {
|
for (const gate of gates) {
|
||||||
const gateEntry = normalizeGate(gate);
|
const gateEntry = normalizeGate(gate);
|
||||||
const gateCmd = gateEntry.command;
|
const gateCmd = gateEntry.command;
|
||||||
|
if (!gateCmd && gateEntry.type !== 'ci-pipeline') continue;
|
||||||
|
|
||||||
const label = gateCmd || gateEntry.type;
|
const label = gateCmd || gateEntry.type;
|
||||||
// NOTE: no silent skip — every gate produces a typed result (RI-N2).
|
|
||||||
emitEvent(
|
emitEvent(
|
||||||
eventsPath,
|
eventsPath,
|
||||||
'rail.check.started',
|
'rail.check.started',
|
||||||
@@ -283,10 +209,10 @@ export function runGates(
|
|||||||
'quality-gate',
|
'quality-gate',
|
||||||
`Running gate: ${label}`,
|
`Running gate: ${label}`,
|
||||||
);
|
);
|
||||||
const result = runGate(gate, cwd, logPath, timeoutSec, options);
|
const result = runGate(gate, cwd, logPath, timeoutSec);
|
||||||
gateResults.push(result);
|
gateResults.push(result);
|
||||||
|
|
||||||
if (result.status === 'passed') {
|
if (result.passed) {
|
||||||
emitEvent(
|
emitEvent(
|
||||||
eventsPath,
|
eventsPath,
|
||||||
'rail.check.passed',
|
'rail.check.passed',
|
||||||
@@ -298,46 +224,7 @@ export function runGates(
|
|||||||
continue;
|
continue;
|
||||||
}
|
}
|
||||||
|
|
||||||
if (result.status === 'waiting') {
|
allPassed = false;
|
||||||
hasWaiting = true;
|
|
||||||
emitEvent(
|
|
||||||
eventsPath,
|
|
||||||
'rail.check.waiting',
|
|
||||||
taskId,
|
|
||||||
'gated',
|
|
||||||
'quality-gate',
|
|
||||||
`Gate waiting: ${label} — ${result.reason ?? 'manual gate awaits sign-off'}`,
|
|
||||||
);
|
|
||||||
continue;
|
|
||||||
}
|
|
||||||
|
|
||||||
if (result.status === 'simulated') {
|
|
||||||
hasSimulated = true;
|
|
||||||
emitEvent(
|
|
||||||
eventsPath,
|
|
||||||
'rail.check.simulated',
|
|
||||||
taskId,
|
|
||||||
'gated',
|
|
||||||
'quality-gate',
|
|
||||||
`Gate simulated (non-satisfying): ${label}`,
|
|
||||||
);
|
|
||||||
continue;
|
|
||||||
}
|
|
||||||
|
|
||||||
if (result.status === 'capability_failure') {
|
|
||||||
hasCapabilityFailure = true;
|
|
||||||
emitEvent(
|
|
||||||
eventsPath,
|
|
||||||
'rail.check.failed',
|
|
||||||
taskId,
|
|
||||||
'gated',
|
|
||||||
'quality-gate',
|
|
||||||
`Gate capability failure (${result.capability_code ?? 'MACP_NO_PROVIDER'}): ${label} — ${result.reason ?? 'required capability is absent'}`,
|
|
||||||
);
|
|
||||||
continue;
|
|
||||||
}
|
|
||||||
|
|
||||||
hasFailed = true;
|
|
||||||
let message: string;
|
let message: string;
|
||||||
if (result.timed_out) {
|
if (result.timed_out) {
|
||||||
message = `Gate timed out after ${timeoutSec}s: ${label}`;
|
message = `Gate timed out after ${timeoutSec}s: ${label}`;
|
||||||
@@ -349,15 +236,5 @@ export function runGates(
|
|||||||
emitEvent(eventsPath, 'rail.check.failed', taskId, 'gated', 'quality-gate', message);
|
emitEvent(eventsPath, 'rail.check.failed', taskId, 'gated', 'quality-gate', message);
|
||||||
}
|
}
|
||||||
|
|
||||||
const state: GateStatus = hasCapabilityFailure
|
return { allPassed, gateResults };
|
||||||
? 'capability_failure'
|
|
||||||
: hasSimulated
|
|
||||||
? 'simulated'
|
|
||||||
: hasFailed
|
|
||||||
? 'failed'
|
|
||||||
: hasWaiting
|
|
||||||
? 'waiting'
|
|
||||||
: 'passed';
|
|
||||||
|
|
||||||
return { allPassed: state === 'passed', gateResults, state };
|
|
||||||
}
|
}
|
||||||
|
|||||||
@@ -6,13 +6,11 @@ export type {
|
|||||||
DependsOnPolicy,
|
DependsOnPolicy,
|
||||||
GateType,
|
GateType,
|
||||||
GateFailOn,
|
GateFailOn,
|
||||||
GateStatus,
|
|
||||||
GateEntry,
|
GateEntry,
|
||||||
Task,
|
Task,
|
||||||
EventType,
|
EventType,
|
||||||
MACPEvent,
|
MACPEvent,
|
||||||
GateResult,
|
GateResult,
|
||||||
RunGatesResult,
|
|
||||||
TaskResult,
|
TaskResult,
|
||||||
ProviderMeta,
|
ProviderMeta,
|
||||||
ProviderRegistry,
|
ProviderRegistry,
|
||||||
@@ -20,11 +18,6 @@ export type {
|
|||||||
|
|
||||||
export { CredentialError } from './types.js';
|
export { CredentialError } from './types.js';
|
||||||
|
|
||||||
// Typed fail-closed capability errors (RI-N2, SDLC-D-035)
|
|
||||||
export { MACP_ERROR_CODES, MACPCapabilityError } from './errors.js';
|
|
||||||
|
|
||||||
export type { MacpErrorCode } from './errors.js';
|
|
||||||
|
|
||||||
// Credential resolver
|
// Credential resolver
|
||||||
export {
|
export {
|
||||||
DEFAULT_CREDENTIALS_DIR,
|
DEFAULT_CREDENTIALS_DIR,
|
||||||
@@ -42,16 +35,9 @@ export {
|
|||||||
export type { ResolveCredentialsOptions } from './credential-resolver.js';
|
export type { ResolveCredentialsOptions } from './credential-resolver.js';
|
||||||
|
|
||||||
// Gate runner
|
// Gate runner
|
||||||
export {
|
export { normalizeGate, runShell, countAIFindings, runGate, runGates } from './gate-runner.js';
|
||||||
normalizeGate,
|
|
||||||
runShell,
|
|
||||||
countAIFindings,
|
|
||||||
runGate,
|
|
||||||
runGates,
|
|
||||||
SIMULATED_GATE_REASON,
|
|
||||||
} from './gate-runner.js';
|
|
||||||
|
|
||||||
export type { NormalizedGate, RunGateOptions } from './gate-runner.js';
|
export type { NormalizedGate } from './gate-runner.js';
|
||||||
|
|
||||||
// Risk-floor (agent reflection loop — diff review classifier)
|
// Risk-floor (agent reflection loop — diff review classifier)
|
||||||
export { evaluateRiskFloor, DEFAULT_RISK_THRESHOLD } from './risk-floor.js';
|
export { evaluateRiskFloor, DEFAULT_RISK_THRESHOLD } from './risk-floor.js';
|
||||||
|
|||||||
@@ -1,5 +1,3 @@
|
|||||||
import type { MacpErrorCode } from './errors.js';
|
|
||||||
|
|
||||||
/** Task status values. */
|
/** Task status values. */
|
||||||
export type TaskStatus = 'pending' | 'running' | 'gated' | 'completed' | 'failed' | 'escalated';
|
export type TaskStatus = 'pending' | 'running' | 'gated' | 'completed' | 'failed' | 'escalated';
|
||||||
|
|
||||||
@@ -19,17 +17,7 @@ export type DispatchMode = 'yolo' | 'acp' | 'exec';
|
|||||||
export type DependsOnPolicy = 'all' | 'any' | 'all_terminal';
|
export type DependsOnPolicy = 'all' | 'any' | 'all_terminal';
|
||||||
|
|
||||||
/** Quality gate type. */
|
/** Quality gate type. */
|
||||||
export type GateType = 'mechanical' | 'ai-review' | 'ci-pipeline' | 'manual';
|
export type GateType = 'mechanical' | 'ai-review' | 'ci-pipeline';
|
||||||
|
|
||||||
/**
|
|
||||||
* Typed execution state of a gate — closed set (RI-N2, SDLC-D-035).
|
|
||||||
*
|
|
||||||
* Only `passed` means "really executed and green". `simulated` is produced
|
|
||||||
* exclusively under an explicit simulate opt-in and never satisfies anything.
|
|
||||||
* `capability_failure` means a required executor/provider/command was absent.
|
|
||||||
* `waiting` means a manual gate awaits human sign-off (neither pass nor fail).
|
|
||||||
*/
|
|
||||||
export type GateStatus = 'passed' | 'failed' | 'simulated' | 'waiting' | 'capability_failure';
|
|
||||||
|
|
||||||
/** Gate fail_on mode. */
|
/** Gate fail_on mode. */
|
||||||
export type GateFailOn = 'blocker' | 'any';
|
export type GateFailOn = 'blocker' | 'any';
|
||||||
@@ -79,9 +67,7 @@ export type EventType =
|
|||||||
| 'task.retry.scheduled'
|
| 'task.retry.scheduled'
|
||||||
| 'rail.check.started'
|
| 'rail.check.started'
|
||||||
| 'rail.check.passed'
|
| 'rail.check.passed'
|
||||||
| 'rail.check.failed'
|
| 'rail.check.failed';
|
||||||
| 'rail.check.waiting'
|
|
||||||
| 'rail.check.simulated';
|
|
||||||
|
|
||||||
/** Structured event record. */
|
/** Structured event record. */
|
||||||
export interface MACPEvent {
|
export interface MACPEvent {
|
||||||
@@ -102,14 +88,7 @@ export interface GateResult {
|
|||||||
type: string;
|
type: string;
|
||||||
output: string;
|
output: string;
|
||||||
timed_out: boolean;
|
timed_out: boolean;
|
||||||
/** Back-compat boolean view — true ONLY when `status === 'passed'`. */
|
|
||||||
passed: boolean;
|
passed: boolean;
|
||||||
/** Typed discriminator — the authoritative gate outcome (RI-N2). */
|
|
||||||
status: GateStatus;
|
|
||||||
/** Typed capability error code, set when `status === 'capability_failure'`. */
|
|
||||||
capability_code?: MacpErrorCode;
|
|
||||||
/** Why a non-executed state (simulated/waiting/capability_failure) was reached. */
|
|
||||||
reason?: string;
|
|
||||||
fail_on?: string;
|
fail_on?: string;
|
||||||
blockers?: number;
|
blockers?: number;
|
||||||
findings?: number;
|
findings?: number;
|
||||||
@@ -117,22 +96,6 @@ export interface GateResult {
|
|||||||
parse_error?: string;
|
parse_error?: string;
|
||||||
}
|
}
|
||||||
|
|
||||||
/**
|
|
||||||
* Aggregate outcome of `runGates` (RI-N2).
|
|
||||||
*
|
|
||||||
* `state` is the typed aggregate: it is `passed` only when every gate really
|
|
||||||
* executed green. A `simulated` result makes the aggregate `simulated` (never
|
|
||||||
* `passed`); a `waiting` manual gate keeps the aggregate `waiting`; a missing
|
|
||||||
* capability makes it `capability_failure`. `allPassed` is exactly
|
|
||||||
* `state === 'passed'`, so a simulated or waiting result can never satisfy a
|
|
||||||
* dependency, acceptance criterion, gate, merge, or release check.
|
|
||||||
*/
|
|
||||||
export interface RunGatesResult {
|
|
||||||
allPassed: boolean;
|
|
||||||
gateResults: GateResult[];
|
|
||||||
state: GateStatus;
|
|
||||||
}
|
|
||||||
|
|
||||||
/** Result from a completed task. */
|
/** Result from a completed task. */
|
||||||
export interface TaskResult {
|
export interface TaskResult {
|
||||||
task_id: string;
|
task_id: string;
|
||||||
|
|||||||
@@ -39,20 +39,3 @@ packages/mosaic/framework/tools/tmux/test-send-message-verdict.sh | requires rea
|
|||||||
# recorded judgement. These lines ARE that judgement, signed.)
|
# recorded judgement. These lines ARE that judgement, signed.)
|
||||||
packages/mosaic/framework/tools/orchestrator/smoke-test.sh | behavior smoke checks for coord continue/run workflows, run manually by orchestrator seats; unmeasured in CI; #1017 burndown
|
packages/mosaic/framework/tools/orchestrator/smoke-test.sh | behavior smoke checks for coord continue/run workflows, run manually by orchestrator seats; unmeasured in CI; #1017 burndown
|
||||||
packages/mosaic/framework/tools/wake/validate-973/microtest-wake-assert.sh | #973 instrument self-test, run as a precondition of the validate-973 evidence procedure rather than as a standing CI suite; #1017 burndown candidate
|
packages/mosaic/framework/tools/wake/validate-973/microtest-wake-assert.sh | #973 instrument self-test, run as a precondition of the validate-973 evidence procedure rather than as a standing CI suite; #1017 burndown candidate
|
||||||
|
|
||||||
# --- tools/fleet: precondition is unsatisfiable in the CI image (#1271) ---
|
|
||||||
# Signed by fred (sb-it-1-dt, 2026-08-16) at origin/next 476db12.
|
|
||||||
# This suite asserts the launcher's behaviour when `mosaic` and `pi` are MISSING.
|
|
||||||
# It shims fakes into $FAKE_BIN, but the constructed PANE_PATH always ends in the
|
|
||||||
# real system path, so on a host that installs those binaries the missing-binary
|
|
||||||
# cases cannot be measured at all. The suite's own guard (line 103) says so and
|
|
||||||
# fails rather than reporting a pass it cannot back. That guard is correct.
|
|
||||||
# The error was wiring the suite into CI: #1017 (c56483eb) enumerated it and
|
|
||||||
# dropped this exclusion, and the CI image provides `pi` in the system path, so
|
|
||||||
# it has failed on every pipeline since. Measured 2026-08-16 across pipelines
|
|
||||||
# 2444 (#1256), 2438 (#1240) and 2441 (#1017-quality): exactly one FAIL line in
|
|
||||||
# each full log, identical, this assertion; control `zzz-not-present-zzz` -> 0.
|
|
||||||
# Burn-down and the full measurement are tracked in #1271; unwired by PR #1270.
|
|
||||||
# Because test:framework-shell is one && chain and this sat at position 44 of 48,
|
|
||||||
# the four suites after it had not run at all since the merge.
|
|
||||||
packages/mosaic/framework/tools/fleet/test-start-agent-session.sh | precondition unsatisfiable in the CI image: asserts missing-binary behaviour, but PANE_PATH always ends in the system path and the image provides `pi` there; guard at line 103 fails by design rather than passing unmeasured. Burn down by controlling the tail of PANE_PATH inside the test. NOT by removing `pi` from the image: the CI image installs @earendil-works/[email protected] deliberately (measured in pipeline 2444's test-step log), and other suites depend on that pin. Burn-down tracked in #1271
|
|
||||||
|
|||||||
@@ -97,13 +97,34 @@ printf '%s' "$MSG" | "${tmux_cmd[@]}" load-buffer -b "$BUF" -
|
|||||||
# would otherwise accumulate forever.
|
# would otherwise accumulate forever.
|
||||||
sleep 0.5
|
sleep 0.5
|
||||||
|
|
||||||
# 2) Submit, then POSITIVELY confirm submission; flush with another Enter if it is
|
# 2) Submit, then POSITIVELY confirm submission by DRAFT TRANSITION, not by prompt
|
||||||
# still a draft. Success requires positive evidence — the queued banner, OR the
|
# glyph. The historical bug was treating ABSENCE of a draft as delivery; the
|
||||||
# REPL input box located AND clear of our message tail. The historical bug was
|
# 2026-08 fix over-corrected to glyph inference (grep '❯|^>|│ >'), which locates
|
||||||
# treating ABSENCE of a draft as delivery: if the prompt glyph was never matched
|
# only Claude Code's box and false-NEGATIVES every glyphless REPL (pi renders a
|
||||||
# (wrong pane / prompt-glyph drift), an unsubmitted message read as "delivered"
|
# U+2500 rule, no glyph) — a delivered message reported "UNDELIVERED", driving a
|
||||||
# and worker->lead relays stalled silently. We now default to UNCONFIRMED and only
|
# retry that duplicates it. Runtime-agnostic evidence: our message tail sits on
|
||||||
# upgrade to delivered on positive evidence; anything we cannot confirm fails loud.
|
# the INPUT line (located by the cursor row, not a glyph) BEFORE Enter, and has
|
||||||
|
# LEFT it AFTER — that transition is positive proof of submission and needs no
|
||||||
|
# glyph. Absence alone still never means delivered: if we never saw our draft on
|
||||||
|
# the input line we stay UNCONFIRMED (wrong/dead pane), and a draft that never
|
||||||
|
# leaves the input line stays a DRAFT (exit 2), preserving both historical guards.
|
||||||
|
_cursor_line() { # echo the pane's current input (cursor) line, glyph-free
|
||||||
|
local cy line
|
||||||
|
cy=$("${tmux_cmd[@]}" display-message -p -t "$EFFECTIVE_TARGET" -F '#{cursor_y}' 2>/dev/null) || return 1
|
||||||
|
[ -n "$cy" ] || return 1
|
||||||
|
"${tmux_cmd[@]}" capture-pane -t "$EFFECTIVE_TARGET" -p 2>/dev/null | sed -n "$((cy + 1))p"
|
||||||
|
}
|
||||||
|
_draft_on_input() { # true iff our message tail is sitting on the input line now
|
||||||
|
[ -n "$snippet" ] || return 1
|
||||||
|
printf '%s' "$(_cursor_line)" | grep -qF "$snippet"
|
||||||
|
}
|
||||||
|
|
||||||
|
# Baseline: after the paste, our draft must be on the input line. This is positive
|
||||||
|
# proof we are on the right pane and the paste landed — the anchor the transition
|
||||||
|
# check measures against.
|
||||||
|
saw_draft=0
|
||||||
|
_draft_on_input && saw_draft=1
|
||||||
|
|
||||||
status="unconfirmed"
|
status="unconfirmed"
|
||||||
for attempt in $(seq 1 $((RETRIES + 1))); do
|
for attempt in $(seq 1 $((RETRIES + 1))); do
|
||||||
"${tmux_cmd[@]}" send-keys -t "$EFFECTIVE_TARGET" Enter
|
"${tmux_cmd[@]}" send-keys -t "$EFFECTIVE_TARGET" Enter
|
||||||
@@ -113,20 +134,26 @@ for attempt in $(seq 1 $((RETRIES + 1))); do
|
|||||||
if printf '%s' "$pane" | grep -qF "$QUEUED_RE"; then
|
if printf '%s' "$pane" | grep -qF "$QUEUED_RE"; then
|
||||||
status="queued"; break
|
status="queued"; break
|
||||||
fi
|
fi
|
||||||
# Locate the REPL input box (prompt glyph). If we cannot see it, we have NO
|
# POSITIVE draft evidence from a located prompt box, when one exists. This is the
|
||||||
# evidence of submission state — stay UNCONFIRMED and retry; never infer delivery.
|
# cursor-row check's blind spot: a pane in COOKED mode (a plain shell whose
|
||||||
|
# foreground process never reads stdin) echoes our paste via the kernel line
|
||||||
|
# discipline and moves the cursor off it on Enter, which is indistinguishable from
|
||||||
|
# a real submit by cursor row alone. If a prompt box IS locatable and still carries
|
||||||
|
# our tail, that is affirmative proof the message was not consumed. Absence of a
|
||||||
|
# glyph is still never used for anything — that inference is the original E7 bug.
|
||||||
promptline=$(printf '%s' "$pane" | grep -E '❯|^>|│ >' | tail -1)
|
promptline=$(printf '%s' "$pane" | grep -E '❯|^>|│ >' | tail -1)
|
||||||
if [ -z "$promptline" ]; then
|
if [ -n "$promptline" ] && [ -n "$snippet" ] && printf '%s' "$promptline" | grep -qF "$snippet"; then
|
||||||
status="unconfirmed"; continue
|
|
||||||
fi
|
|
||||||
# Input box located AND still carrying our tail => unsubmitted draft. Flush + retry.
|
|
||||||
# (Submitted messages scroll up into history; a draft stays on the ❯ line.)
|
|
||||||
if [ -n "$snippet" ] && printf '%s' "$promptline" | grep -qF "$snippet"; then
|
|
||||||
status="draft"; continue
|
status="draft"; continue
|
||||||
fi
|
fi
|
||||||
# Input box located AND clear of our tail => positively submitted. This is the
|
if [ "$saw_draft" = 1 ]; then
|
||||||
# only path to success besides the queued banner.
|
if _draft_on_input; then
|
||||||
status="delivered"; break
|
status="draft"; continue # still on the input line => not submitted; flush + retry
|
||||||
|
fi
|
||||||
|
status="delivered"; break # left the input line => positively submitted
|
||||||
|
fi
|
||||||
|
# No confirmed baseline yet: try to (re)acquire it; never infer delivery from absence.
|
||||||
|
if _draft_on_input; then saw_draft=1; status="draft"; continue; fi
|
||||||
|
status="unconfirmed"; continue
|
||||||
done
|
done
|
||||||
|
|
||||||
[ "$VERBOSE" = 1 ] && { echo "--- pane tail ($TARGET) ---"; printf '%s\n' "$pane" | tail -4; echo "---"; }
|
[ "$VERBOSE" = 1 ] && { echo "--- pane tail ($TARGET) ---"; printf '%s\n' "$pane" | tail -4; echo "---"; }
|
||||||
|
|||||||
@@ -0,0 +1,97 @@
|
|||||||
|
#!/usr/bin/env bash
|
||||||
|
# Red-first regression test for E7 (#1017 task 2): the confirm-check must bind
|
||||||
|
# "delivered" to WHETHER THE MESSAGE WAS SUBMITTED, not to which runtime's prompt
|
||||||
|
# glyph is present. A pi seat renders a U+2500 rule input box with no ❯/^>/│ >
|
||||||
|
# glyph; send-message.sh:118 locates the box only by glyph, so a genuinely
|
||||||
|
# delivered message on a glyphless REPL falsely reports exit 2 "may be UNDELIVERED",
|
||||||
|
# and the operator's rc=2-driven retry duplicates it.
|
||||||
|
#
|
||||||
|
# Parameterized on $SEND: RED against the shipping blob (B and D fail), GREEN
|
||||||
|
# against a candidate patch. No pi; no fake HOME; hermetic throwaway socket.
|
||||||
|
#
|
||||||
|
# Submission counting is EXACT and terminal-echo-independent: the fixture message
|
||||||
|
# is `echo <tok> >>SINK`; each real submission appends one line. wc -l SINK ==
|
||||||
|
# number of times the REPL actually executed the send. This does not depend on how
|
||||||
|
# many times the marker string is painted on screen.
|
||||||
|
set -u
|
||||||
|
SEND="${SEND:?set SEND=/path/to/send-message.sh}"
|
||||||
|
SOCKET="glyphagnostic-$$"
|
||||||
|
TMP="$(mktemp -d)"
|
||||||
|
tmux() { command tmux -L "$SOCKET" "$@"; }
|
||||||
|
cleanup() { command tmux -L "$SOCKET" kill-server 2>/dev/null; rm -rf "$TMP"; }
|
||||||
|
trap cleanup EXIT
|
||||||
|
pass=0; fail=0
|
||||||
|
ok() { printf 'ok %s\n' "$1"; pass=$((pass+1)); }
|
||||||
|
no() { printf 'FAIL %s -- %s\n' "$1" "$2"; fail=$((fail+1)); }
|
||||||
|
|
||||||
|
mk() { tmux new-session -d -s "$1" -x 120 -y 40 -c "$TMP" "PS1='$2' exec bash --noprofile --norc -i"; sleep 0.5; }
|
||||||
|
subs() { [ -f "$1" ] && wc -l <"$1" | tr -d ' ' || echo 0; } # exact submission count
|
||||||
|
|
||||||
|
echo "SEND=$SEND tmux $(command tmux -V | awk '{print $2}')"
|
||||||
|
|
||||||
|
# --- A (control): glyph box (❯) that submits => exit 0, exactly one submission.
|
||||||
|
mk ctl '❯ '
|
||||||
|
SINK="$TMP/sink.ctl"
|
||||||
|
out=$("$SEND" -L "$SOCKET" -t ctl -m "echo x >>'$SINK'" 2>"$TMP/e.ctl"); rc=$?; sleep 0.4
|
||||||
|
if [ "$rc" = 0 ] && [ "$(subs "$SINK")" = 1 ]; then
|
||||||
|
ok "control: ❯-box submits => exit 0, exactly one submission"
|
||||||
|
else no "control: ❯-box submits => exit 0, one submission" "rc=$rc subs=$(subs "$SINK") err=[$(cat "$TMP/e.ctl")]"; fi
|
||||||
|
|
||||||
|
# --- B (THE false-rc regression): glyphless U+2500 box that SUBMITS. Message lands
|
||||||
|
# (subs==1) yet shipping reports exit 2. Must be exit 0.
|
||||||
|
mk sub $'──────── \n'
|
||||||
|
SINK="$TMP/sink.sub"
|
||||||
|
out=$("$SEND" -L "$SOCKET" -t sub -m "echo x >>'$SINK'" 2>"$TMP/e.sub"); rc=$?; sleep 0.4
|
||||||
|
if [ "$rc" = 0 ] && [ "$(subs "$SINK")" = 1 ]; then
|
||||||
|
ok "glyphless: U+2500 box that submits => exit 0 (delivered, not 'UNDELIVERED')"
|
||||||
|
else no "glyphless: U+2500 box that submits => exit 0" \
|
||||||
|
"rc=$rc subs=$(subs "$SINK")(delivered=$([ "$(subs "$SINK")" -ge 1 ] && echo yes||echo no)) err=[$(cat "$TMP/e.sub")]"; fi
|
||||||
|
|
||||||
|
# --- D (duplicate arm): operator follows the rc=2 stderr and retries once. On the
|
||||||
|
# glyphless box, shipping => two submissions (the reported duplicate). The
|
||||||
|
# property: one logical send => exactly one submission. Same fix closes it.
|
||||||
|
mk dup $'──────── \n'
|
||||||
|
SINK="$TMP/sink.dup"
|
||||||
|
tries=0
|
||||||
|
for attempt in 1 2; do
|
||||||
|
tries=$((tries+1))
|
||||||
|
out=$("$SEND" -L "$SOCKET" -t dup -m "echo x >>'$SINK'" 2>/dev/null); rc=$?
|
||||||
|
sleep 0.4
|
||||||
|
[ "$rc" = 0 ] && break # operator stops retrying only when told delivered
|
||||||
|
done
|
||||||
|
if [ "$(subs "$SINK")" = 1 ]; then
|
||||||
|
ok "duplicate: one logical send (rc-driven retry) => exactly one submission (tries=$tries)"
|
||||||
|
else no "duplicate: one logical send => exactly one submission" "submissions=$(subs "$SINK") tries=$tries"; fi
|
||||||
|
|
||||||
|
# --- E (faithful hung managed TUI, NOT a cooked shell): raw/no-echo, paints nothing.
|
||||||
|
# A cooked `sleep infinity` echoes the paste via the kernel line discipline and
|
||||||
|
# false-passes a cursor-row fix that is correct on real seats (measured). So: raw.
|
||||||
|
mk_rawstuck() { tmux new-session -d -s "$1" -x 120 -y 40 -c "$TMP" \
|
||||||
|
"bash --noprofile --norc -c 'stty -echo -icanon min 1 time 0 2>/dev/null; exec sleep infinity'"; sleep 0.5; }
|
||||||
|
mk_rawstuck estuck
|
||||||
|
SINK="$TMP/sink.estuck"
|
||||||
|
out=$("$SEND" -L "$SOCKET" -t estuck -r 1 -m "this stuck draft was never submitted" 2>/dev/null); rc=$?
|
||||||
|
sleep 0.3
|
||||||
|
if [ "$rc" != 0 ] && [ "$(subs "$SINK")" = 0 ]; then
|
||||||
|
ok "raw/no-echo stuck TUI (not submitted) => non-zero (no false delivered)"
|
||||||
|
else no "raw stuck TUI must NOT report delivered" "rc=$rc subs=$(subs "$SINK")"; fi
|
||||||
|
|
||||||
|
# --- F (busy/queued branch, your BUSY-not-runtime finding): glyphless pane rendering the
|
||||||
|
# queued banner, never consuming. QUEUED_RE :113 fires before the glyph grep => rc=0.
|
||||||
|
mk_busy() { tmux new-session -d -s "$1" -x 120 -y 40 -c "$TMP" \
|
||||||
|
"bash --noprofile --norc -c 'printf \"Press up to edit queued messages\n\"; exec sleep infinity'"; sleep 0.5; }
|
||||||
|
mk_busy ebusy
|
||||||
|
SINK="$TMP/sink.ebusy"
|
||||||
|
out=$("$SEND" -L "$SOCKET" -t ebusy -m "echo x >>'$SINK'" 2>/dev/null); rc=$?; sleep 0.3
|
||||||
|
if [ "$rc" = 0 ]; then
|
||||||
|
ok "busy/queued-banner glyphless => exit 0 (queued is delivery; runtime owns custody)"
|
||||||
|
else no "busy/queued-banner must report delivered" "rc=$rc"; fi
|
||||||
|
|
||||||
|
# --- C (historical-bug guard): unresolvable target. No pane ever carried our draft
|
||||||
|
# => must fail, never infer delivered from absence of a glyph/snippet.
|
||||||
|
if out=$("$SEND" -L "$SOCKET" -t "nonexistent-$$" -m "echo x >>'$TMP/sink.wrong'" 2>/dev/null); then
|
||||||
|
no "wrong-pane: unresolvable target must NOT report success" "expected non-zero, got 0"
|
||||||
|
else ok "wrong-pane: unresolvable target => non-zero (no false delivered)"; fi
|
||||||
|
|
||||||
|
echo "---"; echo "pass=$pass fail=$fail"
|
||||||
|
[ "$fail" = 0 ]
|
||||||
@@ -4,10 +4,13 @@
|
|||||||
#
|
#
|
||||||
# 1. DELIVERED — a REPL that renders a `❯ ` input box and submits on Enter
|
# 1. DELIVERED — a REPL that renders a `❯ ` input box and submits on Enter
|
||||||
# (text scrolls to history, box clears) => exit 0 "✓ delivered".
|
# (text scrolls to history, box clears) => exit 0 "✓ delivered".
|
||||||
# 2. UNCONFIRMED — a pane with NO locatable prompt glyph. This is the exact
|
# 2. DELIVERED — a pane with NO prompt glyph that DOES submit => exit 0. A pi
|
||||||
# historical FALSE POSITIVE: pre-patch it printed "✓ delivered"
|
# seat is this fixture (U+2500 rule, no glyph). Reshaped for
|
||||||
# exit 0; post-patch it MUST fail loud (exit 2, stderr
|
# #1257; see the note at the fixture for why the old exit-2
|
||||||
# "could not confirm submission").
|
# assertion was wrong.
|
||||||
|
# 2b. UNCONFIRMED— a glyphless pane that never submits (raw/no-echo hung TUI)
|
||||||
|
# => must fail loud. This carries the historical
|
||||||
|
# false-positive guard that fixture 2 used to be credited with.
|
||||||
# 3. DRAFT — a `❯ `-prompt pane that never submits (message stays on the
|
# 3. DRAFT — a `❯ `-prompt pane that never submits (message stays on the
|
||||||
# input line) => exit 2, stderr "unsubmitted draft".
|
# input line) => exit 2, stderr "unsubmitted draft".
|
||||||
set -uo pipefail
|
set -uo pipefail
|
||||||
@@ -37,19 +40,44 @@ else
|
|||||||
no "delivered: ❯-prompt REPL that submits => exit 0 ✓ delivered" "rc=$rc out=[$out] err=[$(cat "$TMP/e1")]"
|
no "delivered: ❯-prompt REPL that submits => exit 0 ✓ delivered" "rc=$rc out=[$out] err=[$(cat "$TMP/e1")]"
|
||||||
fi
|
fi
|
||||||
|
|
||||||
# --- Fixture 2: NO prompt glyph (default bash PS1). THE regression: pre-patch this
|
# --- Fixture 2: NO prompt glyph, and the pane DOES submit (interactive bash).
|
||||||
# was a silent false-positive "delivered"; post-patch it must be unconfirmed→exit 2.
|
# RESHAPED 2026-08-16 (#1257), deliberately. This fixture previously asserted
|
||||||
|
# exit 2 here and was labelled "false-positive FIXED". That assertion was wrong,
|
||||||
|
# and locking it in is what kept E7 alive: the pane submits, so "delivered" is
|
||||||
|
# the truth, and a pi seat — whose input box is a bare U+2500 rule with no glyph
|
||||||
|
# — IS this fixture. Reporting exit 2 for it told operators a delivered message
|
||||||
|
# may be undelivered, and the retry that advice invites is the duplicate.
|
||||||
|
#
|
||||||
|
# The guard this fixture was reaching for is real and is NOT dropped: "never
|
||||||
|
# infer delivered from absence" is now enforced positively by fixture 2b below
|
||||||
|
# (glyphless AND not submitting => must fail) and by fixture 3 (locatable box
|
||||||
|
# still carrying our tail => draft). Absence alone decides nothing either way.
|
||||||
tmux -L "$SOCKET" new-session -d -s noglyph -c "$TMP" \
|
tmux -L "$SOCKET" new-session -d -s noglyph -c "$TMP" \
|
||||||
'PS1="sh-noglyph$ " exec bash --noprofile --norc -i'
|
'PS1="sh-noglyph$ " exec bash --noprofile --norc -i'
|
||||||
sleep 0.3
|
sleep 0.3
|
||||||
if out=$("$SEND" -L "$SOCKET" -t "=noglyph" -m "verdict fixture two must fail loud" 2>"$TMP/e2"); then
|
out=$("$SEND" -L "$SOCKET" -t "=noglyph" -m "verdict fixture two must fail loud" 2>"$TMP/e2"); rc=$?
|
||||||
no "unconfirmed: glyphless pane must NOT report success" "expected exit 2, got 0 (out=[$out])"
|
if [ "$rc" -eq 0 ] && printf '%s' "$out" | grep -qF "✓ delivered"; then
|
||||||
|
ok "delivered: glyphless pane that submits => exit 0 (runtime-agnostic, E7 FIXED)"
|
||||||
|
else
|
||||||
|
no "delivered: glyphless pane that submits => exit 0" "rc=$rc out=[$out] err=[$(cat "$TMP/e2")]"
|
||||||
|
fi
|
||||||
|
|
||||||
|
# --- Fixture 2b: NO prompt glyph AND never submits — a hung managed TUI holding the
|
||||||
|
# terminal in raw/no-echo, which is what a stuck agent seat actually is (measured
|
||||||
|
# on live pi: stty -echo -icanon). Nothing is echoed, nothing is consumed, so
|
||||||
|
# there is no positive evidence of submission and the tool MUST fail loud. This
|
||||||
|
# is the historical false-positive guard, kept as a positive test.
|
||||||
|
tmux -L "$SOCKET" new-session -d -s rawstuck -c "$TMP" \
|
||||||
|
'bash --noprofile --norc -c "stty -echo -icanon min 1 time 0 2>/dev/null; exec sleep infinity"'
|
||||||
|
sleep 0.3
|
||||||
|
if out=$("$SEND" -L "$SOCKET" -t "=rawstuck" -r 1 -m "verdict fixture two-b never submitted" 2>"$TMP/e2b"); then
|
||||||
|
no "unconfirmed: glyphless hung TUI must NOT report success" "expected non-zero, got 0 (out=[$out])"
|
||||||
else
|
else
|
||||||
rc=$?
|
rc=$?
|
||||||
if [ "$rc" -eq 2 ] && grep -qF "could not confirm submission" "$TMP/e2"; then
|
if [ "$rc" -ne 0 ] && grep -qF "could not confirm submission" "$TMP/e2b"; then
|
||||||
ok "unconfirmed: glyphless pane => exit 2 + 'could not confirm submission' (false-positive FIXED)"
|
ok "unconfirmed: glyphless hung TUI (raw/no-echo) => non-zero + 'could not confirm submission'"
|
||||||
else
|
else
|
||||||
no "unconfirmed: glyphless pane => exit 2 + stderr" "rc=$rc err=[$(cat "$TMP/e2")]"
|
no "unconfirmed: glyphless hung TUI => non-zero + stderr" "rc=$rc err=[$(cat "$TMP/e2b")]"
|
||||||
fi
|
fi
|
||||||
fi
|
fi
|
||||||
|
|
||||||
|
|||||||
@@ -25,7 +25,7 @@
|
|||||||
"lint": "eslint src",
|
"lint": "eslint src",
|
||||||
"typecheck": "tsc --noEmit",
|
"typecheck": "tsc --noEmit",
|
||||||
"test": "vitest run --passWithNoTests && pnpm run test:framework-shell",
|
"test": "vitest run --passWithNoTests && pnpm run test:framework-shell",
|
||||||
"test:framework-shell": "bash framework/tools/quality/scripts/check-test-enumeration.sh && bash framework/tools/quality/scripts/test-check-test-enumeration.sh && python3 src/lease-broker/daemon_deadline_unittest.py && python3 src/lease-broker/normative_fragments_unittest.py && python3 src/lease-broker/promotion_binding_unittest.py && python3 src/lease-broker/promotion_trigger_unittest.py && python3 src/lease-broker/receipt_challenge_unittest.py && python3 src/lease-broker/context_recovery_unittest.py && python3 src/lease-broker/recovery_runtime_unittest.py && python3 src/lease-broker/recovery_b1_adversarial_unittest.py && python3 src/lease-broker/receipt_observer_client_unittest.py && python3 src/lease-broker/invariant_r_unittest.py && python3 src/lease-broker/framework_skill_portability_unittest.py && python3 src/mutator-gate/runtime_tools_unittest.py && python3 src/mutator-gate/runtime_launch_guard_unittest.py && python3 src/mutator-gate/version_coupling_unittest.py && python3 framework/tools/lease-broker/check-runtime-launches.py --root ../.. && bash framework/tools/codex/test-pr-diff-context.sh && bash framework/tools/qa/test-deps-preflight.sh && bash framework/tools/git/test-pr-review-gitea-comment.sh && bash framework/tools/git/test-pr-review-repo-host-override.sh && bash framework/tools/git/test-ci-queue-wait-branch-absent.sh && bash framework/tools/git/test-ci-queue-wait-tristate.sh && bash framework/tools/git/test-ci-queue-wait-github-checks.sh && bash framework/tools/git/test-pr-merge-queue-branch.sh && bash framework/tools/git/test-pr-merge-head-pin.sh && bash framework/tools/git/test-pr-merge-message-field.sh && bash framework/tools/git/test-git-credential-mosaic.sh && bash framework/tools/git/test-gitea-token-identity.sh && bash framework/tools/woodpecker/test-terminal-green-contract.sh && bash framework/tools/_scripts/test-install-ordering-guard.sh && bash framework/tools/_scripts/test-mosaic-init-rce.sh && bash framework/tools/tmux/agent-send.test.sh && bash framework/tools/wake/test-wake-store-ack.sh && bash framework/tools/wake/test-wake-store-enqueue-race.sh && bash framework/tools/wake/test-wake-digest-hmac.sh && bash framework/tools/wake/test-wake-digest-quarantine.sh && bash framework/tools/wake/test-wake-detector.sh && bash framework/tools/wake/test-wake-fn-oracle.sh && bash framework/tools/wake/test-wake-reconcile.sh && bash framework/tools/wake/test-wake-beacon.sh && bash framework/tools/wake/test-wake-preimage.sh && bash framework/tools/wake/test-wake-install.sh && bash framework/tools/glpi/test-list-http-status.sh && bash framework/tools/orchestrator/test-board-roll.sh && bash framework/tools/woodpecker/test-ci-wait-exit-matrix.sh && bash framework/tools/_scripts/test-fleet-transport-check.sh"
|
"test:framework-shell": "bash framework/tools/quality/scripts/check-test-enumeration.sh && bash framework/tools/quality/scripts/test-check-test-enumeration.sh && python3 src/lease-broker/daemon_deadline_unittest.py && python3 src/lease-broker/normative_fragments_unittest.py && python3 src/lease-broker/promotion_binding_unittest.py && python3 src/lease-broker/promotion_trigger_unittest.py && python3 src/lease-broker/receipt_challenge_unittest.py && python3 src/lease-broker/context_recovery_unittest.py && python3 src/lease-broker/recovery_runtime_unittest.py && python3 src/lease-broker/recovery_b1_adversarial_unittest.py && python3 src/lease-broker/receipt_observer_client_unittest.py && python3 src/lease-broker/invariant_r_unittest.py && python3 src/lease-broker/framework_skill_portability_unittest.py && python3 src/mutator-gate/runtime_tools_unittest.py && python3 src/mutator-gate/runtime_launch_guard_unittest.py && python3 src/mutator-gate/version_coupling_unittest.py && python3 framework/tools/lease-broker/check-runtime-launches.py --root ../.. && bash framework/tools/codex/test-pr-diff-context.sh && bash framework/tools/qa/test-deps-preflight.sh && bash framework/tools/git/test-pr-review-gitea-comment.sh && bash framework/tools/git/test-pr-review-repo-host-override.sh && bash framework/tools/git/test-ci-queue-wait-branch-absent.sh && bash framework/tools/git/test-ci-queue-wait-tristate.sh && bash framework/tools/git/test-ci-queue-wait-github-checks.sh && bash framework/tools/git/test-pr-merge-queue-branch.sh && bash framework/tools/git/test-pr-merge-head-pin.sh && bash framework/tools/git/test-pr-merge-message-field.sh && bash framework/tools/git/test-git-credential-mosaic.sh && bash framework/tools/git/test-gitea-token-identity.sh && bash framework/tools/woodpecker/test-terminal-green-contract.sh && bash framework/tools/_scripts/test-install-ordering-guard.sh && bash framework/tools/_scripts/test-mosaic-init-rce.sh && bash framework/tools/tmux/agent-send.test.sh && bash framework/tools/wake/test-wake-store-ack.sh && bash framework/tools/wake/test-wake-store-enqueue-race.sh && bash framework/tools/wake/test-wake-digest-hmac.sh && bash framework/tools/wake/test-wake-digest-quarantine.sh && bash framework/tools/wake/test-wake-detector.sh && bash framework/tools/wake/test-wake-fn-oracle.sh && bash framework/tools/wake/test-wake-reconcile.sh && bash framework/tools/wake/test-wake-beacon.sh && bash framework/tools/wake/test-wake-preimage.sh && bash framework/tools/wake/test-wake-install.sh && bash framework/tools/fleet/test-start-agent-session.sh && bash framework/tools/glpi/test-list-http-status.sh && bash framework/tools/orchestrator/test-board-roll.sh && bash framework/tools/woodpecker/test-ci-wait-exit-matrix.sh && bash framework/tools/_scripts/test-fleet-transport-check.sh"
|
||||||
},
|
},
|
||||||
"dependencies": {
|
"dependencies": {
|
||||||
"@mosaicstack/brain": "workspace:*",
|
"@mosaicstack/brain": "workspace:*",
|
||||||
|
|||||||
Reference in New Issue
Block a user