Compare commits

..
128 changed files with 277 additions and 16574 deletions
-10
View File
@@ -109,16 +109,6 @@ steps:
# `apk add` guarantees openssl is present on PR pipelines too (and is a
# fast no-op once the rebuilt image already ships it).
- apk add --no-cache openssl
# Pi runtime (Invariant R): invariant_r_unittest.py hard-requires an
# installed `pi` binary at exactly this measured version — the test
# boots Pi's real tool registry to prove the read-only carve-out
# resolves to real, unshadowed builtins, and fails loud (by design)
# when the runtime is absent or drifts. The canonical Pi is
# @earendil-works/[email protected] exactly (@mariozechner/* is
# embedded-legacy). Step-level install because ci-base image publishes
# are currently blocked on registry auth; fold into Dockerfile.ci once
# that is fixed, keeping this as a fast no-op guard.
- npm install -g @earendil-works/[email protected]
# postgresql-client (pg_isready) is baked into ci-base.
# Wait up to 60s for CI postgres to be ready; fail fast if it never comes up.
- |
+3 -7
View File
@@ -48,13 +48,9 @@ mosaic wizard # Full guided setup (gateway install → verify)
### Requirements
- Node.js ≥ 22
- Node.js ≥ 20
- npm (for global @mosaicstack/mosaic install)
- One or more runtimes:
- [Claude Code](https://docs.anthropic.com/en/docs/claude-code)
- [Codex](https://github.com/openai/codex)
- [OpenCode](https://opencode.ai)
- [Pi](https://pi.dev)
- One or more runtimes: [Claude Code](https://docs.anthropic.com/en/docs/claude-code), [Codex](https://github.com/openai/codex), [OpenCode](https://opencode.ai), or [Pi](https://github.com/mariozechner/pi-coding-agent)
## Usage
@@ -204,7 +200,7 @@ Consent state is persisted in config. Remote upload is a no-op until you run `mo
### Prerequisites
- Node.js ≥ 22
- Node.js ≥ 20
- pnpm 10.6+
- Docker & Docker Compose
@@ -8,7 +8,6 @@
* to avoid real I/O — they verify the complete classify → match → decide path.
*/
import { describe, it, expect, vi } from 'vitest';
import type { ProviderHealthStatus } from '@mosaicstack/types';
import { RoutingEngineService } from './routing-engine.service.js';
import { DEFAULT_ROUTING_RULES } from '../routing/default-rules.js';
import type { RoutingRule } from './routing.types.js';
@@ -18,7 +17,7 @@ import type { RoutingRule } from './routing.types.js';
/** Build a RoutingEngineService backed by the given rule set and health map. */
function makeService(
rules: RoutingRule[],
healthMap: Record<string, { status: ProviderHealthStatus }>,
healthMap: Record<string, { status: string }>,
): RoutingEngineService {
const mockDb = {
select: vi.fn().mockReturnValue({
@@ -68,11 +67,11 @@ function defaultRules(): RoutingRule[] {
}
/** A health map where anthropic, openai, and zai are all healthy. */
const allHealthy: Record<string, { status: ProviderHealthStatus }> = {
anthropic: { status: 'healthy' },
openai: { status: 'healthy' },
zai: { status: 'healthy' },
ollama: { status: 'healthy' },
const allHealthy: Record<string, { status: string }> = {
anthropic: { status: 'up' },
openai: { status: 'up' },
zai: { status: 'up' },
ollama: { status: 'up' },
};
// ─── M4-013 E2E tests ─────────────────────────────────────────────────────────
@@ -213,10 +212,10 @@ describe('M4-013: routing end-to-end pipeline', () => {
// Let's use a simple coding message to target Simple coding → Codex (openai)
const message = 'implement a sort function';
const unhealthyHealth: Record<string, { status: ProviderHealthStatus }> = {
const unhealthyHealth = {
anthropic: { status: 'down' },
openai: { status: 'healthy' },
zai: { status: 'healthy' },
openai: { status: 'up' },
zai: { status: 'up' },
ollama: { status: 'down' },
};
@@ -1,6 +1,5 @@
import { Inject, Injectable, Logger } from '@nestjs/common';
import { routingRules, type Db, and, asc, eq, or } from '@mosaicstack/db';
import type { ProviderHealthStatus } from '@mosaicstack/types';
import { DB } from '../../database/database.module.js';
import { ProviderService } from '../provider.service.js';
import { classifyTask } from './task-classifier.js';
@@ -50,7 +49,7 @@ export class RoutingEngineService {
async resolve(
message: string,
userId?: string,
availableProviders?: Record<string, { status: ProviderHealthStatus }>,
availableProviders?: Record<string, { status: string }>,
): Promise<RoutingDecision> {
const classification = classifyTask(message);
this.logger.debug(
@@ -70,8 +69,9 @@ export class RoutingEngineService {
if (!this.matchConditions(rule, classification)) continue;
const providerStatus = health[rule.action.provider]?.status;
const isHealthy = providerStatus === 'up' || providerStatus === 'ok';
if (!this.isRoutable(providerStatus)) {
if (!isHealthy) {
this.logger.debug(
`Rule "${rule.name}" matched but provider "${rule.action.provider}" is unhealthy (status: ${providerStatus ?? 'unknown'})`,
);
@@ -111,10 +111,6 @@ export class RoutingEngineService {
// ─── Private helpers ───────────────────────────────────────────────────────
private isRoutable(status: ProviderHealthStatus | undefined): boolean {
return status === 'healthy' || status === 'degraded';
}
private evaluateCondition(
condition: RoutingCondition,
classification: TaskClassification,
@@ -190,12 +186,11 @@ export class RoutingEngineService {
* Walk the fallback chain and return the first healthy provider/model pair.
* If none are healthy, return the first entry unconditionally (last resort).
*/
private applyFallbackChain(
health: Record<string, { status: ProviderHealthStatus }>,
): RoutingDecision {
private applyFallbackChain(health: Record<string, { status: string }>): RoutingDecision {
for (const candidate of FALLBACK_CHAIN) {
const providerStatus = health[candidate.provider]?.status;
if (this.isRoutable(providerStatus)) {
const isHealthy = providerStatus === 'up' || providerStatus === 'ok';
if (isHealthy) {
this.logger.debug(`Fallback resolved: ${candidate.provider}/${candidate.model}`);
return {
provider: candidate.provider,
@@ -1,5 +1,4 @@
import { describe, it, expect, vi, beforeEach } from 'vitest';
import type { ProviderHealthStatus } from '@mosaicstack/types';
import { RoutingEngineService } from './routing-engine.service.js';
import type { RoutingRule, TaskClassification } from './routing.types.js';
@@ -30,7 +29,7 @@ function makeClassification(overrides: Partial<TaskClassification> = {}): TaskCl
/** Build a minimal RoutingEngineService with mocked DB and ProviderService. */
function makeService(
rules: RoutingRule[] = [],
healthMap: Record<string, { status: ProviderHealthStatus }> = {},
healthMap: Record<string, { status: string }> = {},
): RoutingEngineService {
const mockDb = {
select: vi.fn().mockReturnValue({
@@ -218,10 +217,7 @@ describe('RoutingEngineService.resolve — priority ordering', () => {
}),
];
const service = makeService(rules, {
anthropic: { status: 'healthy' },
openai: { status: 'healthy' },
});
const service = makeService(rules, { anthropic: { status: 'up' }, openai: { status: 'up' } });
const decision = await service.resolve('implement a function');
expect(decision.ruleName).toBe('high priority');
@@ -245,10 +241,7 @@ describe('RoutingEngineService.resolve — priority ordering', () => {
}),
];
const service = makeService(rules, {
anthropic: { status: 'healthy' },
openai: { status: 'healthy' },
});
const service = makeService(rules, { anthropic: { status: 'up' }, openai: { status: 'up' } });
const decision = await service.resolve('implement a function');
expect(decision.ruleName).toBe('coding rule');
@@ -277,7 +270,7 @@ describe('RoutingEngineService.resolve — unhealthy provider handling', () => {
const service = makeService(rules, {
anthropic: { status: 'down' }, // primary is unhealthy
openai: { status: 'healthy' },
openai: { status: 'up' },
});
const decision = await service.resolve('implement a function');
@@ -297,7 +290,7 @@ describe('RoutingEngineService.resolve — unhealthy provider handling', () => {
];
const service2 = makeService(unhealthyRules, {
anthropic: { status: 'healthy' },
anthropic: { status: 'up' },
openai: { status: 'down' },
});
@@ -313,7 +306,7 @@ describe('RoutingEngineService.resolve — unhealthy provider handling', () => {
const service = makeService(rules, {
anthropic: { status: 'down' }, // Sonnet is on anthropic — down
ollama: { status: 'healthy' }, // Haiku is also on anthropic — use Ollama as next
ollama: { status: 'up' }, // Haiku is also on anthropic — use Ollama as next
});
const decision = await service.resolve('hello there');
@@ -352,7 +345,7 @@ describe('RoutingEngineService.resolve — empty conditions (fallback rule)', ()
}),
];
const service = makeService(rules, { anthropic: { status: 'healthy' } });
const service = makeService(rules, { anthropic: { status: 'up' } });
const decision = await service.resolve('completely unrelated message xyz');
expect(decision.ruleName).toBe('catch-all');
@@ -376,7 +369,7 @@ describe('RoutingEngineService.resolve — empty conditions (fallback rule)', ()
}),
];
const service = makeService(rules, { anthropic: { status: 'healthy' } });
const service = makeService(rules, { anthropic: { status: 'up' } });
const codingDecision = await service.resolve('implement a function');
expect(codingDecision.ruleName).toBe('specific coding rule');
@@ -408,7 +401,7 @@ describe('RoutingEngineService.resolve — disabled rules', () => {
}),
];
const service = makeService(rules, { anthropic: { status: 'healthy' } });
const service = makeService(rules, { anthropic: { status: 'up' } });
const decision = await service.resolve('implement a function');
expect(decision.ruleName).toBe('enabled fallback');
@@ -459,45 +452,9 @@ describe('RoutingEngineService.resolve — availableProviders override', () => {
ps: unknown,
) => RoutingEngineService)(mockDb, mockProviderService);
const preSupplied: Record<string, { status: ProviderHealthStatus }> = {
anthropic: { status: 'healthy' },
};
const preSupplied = { anthropic: { status: 'up' } };
await service.resolve('implement a function', undefined, preSupplied);
expect(mockHealthCheckAll).not.toHaveBeenCalled();
});
});
// ─── resolve — canonical ProviderHealthStatus values ──────────────────────────
describe('RoutingEngineService.resolve — canonical health status routing', () => {
it('routes healthy and degraded providers by rule, and falls through to fallback when down', async () => {
const codingRule = makeRule({
name: 'coding rule',
priority: 1,
conditions: [{ field: 'taskType', operator: 'eq', value: 'coding' }],
action: { provider: 'openai', model: 'gpt-4o' },
});
// healthy → selected by its own rule, not the fallback chain
const healthyService = makeService([codingRule], { openai: { status: 'healthy' } });
const healthyDecision = await healthyService.resolve('implement a function');
expect(healthyDecision.ruleName).toBe('coding rule');
expect(healthyDecision.provider).toBe('openai');
// down → rule is skipped as unroutable, falls through to the fallback chain
const downService = makeService([codingRule], {
openai: { status: 'down' },
anthropic: { status: 'healthy' },
});
const downDecision = await downService.resolve('implement a function');
expect(downDecision.ruleName).toBe('fallback');
expect(downDecision.provider).toBe('anthropic');
// degraded → still routable, selected by its own rule, not the fallback chain
const degradedService = makeService([codingRule], { openai: { status: 'degraded' } });
const degradedDecision = await degradedService.resolve('implement a function');
expect(degradedDecision.ruleName).toBe('coding rule');
expect(degradedDecision.provider).toBe('openai');
});
});
-2
View File
@@ -21,7 +21,6 @@ import { AdminModule } from './admin/admin.module.js';
import { CommandsModule } from './commands/commands.module.js';
import { PreferencesModule } from './preferences/preferences.module.js';
import { GCModule } from './gc/gc.module.js';
import { HarnessModule } from './harness/harness.module.js';
import { ReloadModule } from './reload/reload.module.js';
import { WorkspaceModule } from './workspace/workspace.module.js';
import { QueueModule } from './queue/queue.module.js';
@@ -61,7 +60,6 @@ const federationEnabled = loadConfig(resolveGatewayConfigPath()).tier === 'feder
PreferencesModule,
CommandsModule,
GCModule,
HarnessModule,
QueueModule,
ReloadModule,
WorkspaceModule,
@@ -1,4 +1,3 @@
import { Logger } from '@nestjs/common';
import { describe, it, expect, vi, beforeEach } from 'vitest';
import { CommandExecutorService } from './command-executor.service.js';
import type { SlashCommandPayload } from '@mosaicstack/types';
@@ -13,7 +12,6 @@ const mockRegistry = {
{ name: 'agent', aliases: ['a'], scope: 'agent', execution: 'socket', available: true },
{ name: 'prdy', aliases: [], scope: 'agent', execution: 'socket', available: true },
{ name: 'tools', aliases: [], scope: 'agent', execution: 'socket', available: true },
{ name: 'mcp', aliases: [], scope: 'agent', execution: 'socket', available: true },
],
skills: [],
})),
@@ -74,20 +72,7 @@ const mockChatGateway = {
broadcastSessionInfo: vi.fn(),
};
const mockMcpClient = {
reconnectServer: vi.fn().mockResolvedValue(undefined),
getServerStatuses: vi.fn(() => []),
getToolDefinitions: vi.fn(() => []),
};
function buildService(
redis: typeof mockRedis | null = mockRedis,
mcpClient: {
reconnectServer: ReturnType<typeof vi.fn>;
getServerStatuses: ReturnType<typeof vi.fn>;
getToolDefinitions: ReturnType<typeof vi.fn>;
} = mockMcpClient,
): CommandExecutorService {
function buildService(redis: typeof mockRedis | null = mockRedis): CommandExecutorService {
return new CommandExecutorService(
mockRegistry as never,
mockAgentService as never,
@@ -97,7 +82,7 @@ function buildService(
mockBrain as never,
null,
mockChatGateway as never,
mcpClient as never,
null,
);
}
@@ -273,124 +258,4 @@ describe('CommandExecutorService — P8-012 commands', () => {
expect(result.command).toBe('tools');
expect(result.message).toContain('tools');
});
// Top-level catch sanitization (P3-4 re-review finding #1): a rejected
// Redis `set` inside /provider login is the only reachable path into the
// top-level catch in `execute()`. The raw exception must be logged
// server-side but never handed back to the socket client.
it('sanitizes the top-level command catch, logging the raw exception but never returning it to the client', async () => {
const distinctiveRawFailure = 'ECONNREFUSED distinctive-raw-redis-failure-token-9f31';
const rawError = new Error(distinctiveRawFailure);
const failingRedis = {
set: vi.fn().mockRejectedValue(rawError),
get: vi.fn(),
del: vi.fn(),
};
const failingService = buildService(failingRedis as unknown as typeof mockRedis);
const loggerErrorSpy = vi.spyOn(Logger.prototype, 'error').mockImplementation(() => undefined);
const payload: SlashCommandPayload = {
command: 'provider',
args: 'login anthropic',
conversationId,
};
const result = await failingService.execute(payload, userScope);
expect(result.success).toBe(false);
expect(result.command).toBe('provider');
expect(result.message).toBe('Command failed due to an internal error.');
expect(result.message).not.toContain(distinctiveRawFailure);
expect(result.message).not.toContain('ECONNREFUSED');
// The real exception is still logged server-side, as the raw Error
// object itself (not stringified/interpolated into the log message).
expect(loggerErrorSpy).toHaveBeenCalled();
const loggedRawError = loggerErrorSpy.mock.calls.some((call) => call.includes(rawError));
expect(loggedRawError).toBe(true);
loggerErrorSpy.mockRestore();
});
// Inner catch sanitization (P3-5 operator ruling): every catch in
// command-executor.service.ts that returns a SlashCommandResultPayload
// must sanitize the client-facing message the same way the top-level
// catch does, while still logging the raw exception server-side.
it('/agent new sanitizes agent-creation failures, logging the raw exception but never returning it to the client', async () => {
const marker = new Error('distinctive-agent-create-failure-token-A17f');
mockBrain.agents.create.mockRejectedValueOnce(marker);
const loggerErrorSpy = vi.spyOn(Logger.prototype, 'error').mockImplementation(() => undefined);
const payload: SlashCommandPayload = {
command: 'agent',
args: 'new my-new-agent',
conversationId,
};
const result = await service.execute(payload, userScope);
expect(result.success).toBe(false);
expect(result.command).toBe('agent');
expect(result.message).toBe('Failed to create agent due to an internal error.');
expect(result.message).not.toContain('distinctive-agent-create-failure-token-A17f');
expect(loggerErrorSpy).toHaveBeenCalled();
const loggedRawError = loggerErrorSpy.mock.calls.some((call) => call.includes(marker));
expect(loggedRawError).toBe(true);
loggerErrorSpy.mockRestore();
});
it('/agent <name> switch sanitizes agent-lookup failures, logging the raw exception but never returning it to the client', async () => {
const marker = new Error('distinctive-agent-switch-failure-token-B29c');
mockBrain.agents.findByName.mockRejectedValueOnce(marker);
const loggerErrorSpy = vi.spyOn(Logger.prototype, 'error').mockImplementation(() => undefined);
const payload: SlashCommandPayload = {
command: 'agent',
args: 'some-other-agent',
conversationId,
};
const result = await service.execute(payload, userScope);
expect(result.success).toBe(false);
expect(result.command).toBe('agent');
expect(result.message).toBe('Failed to switch agent due to an internal error.');
expect(result.message).not.toContain('distinctive-agent-switch-failure-token-B29c');
expect(loggerErrorSpy).toHaveBeenCalled();
const loggedRawError = loggerErrorSpy.mock.calls.some((call) => call.includes(marker));
expect(loggedRawError).toBe(true);
loggerErrorSpy.mockRestore();
});
it('/mcp reconnect sanitizes MCP client failures, logging the raw exception but never returning it to the client', async () => {
const marker = new Error('distinctive-mcp-reconnect-failure-token-C33e');
const mockMcpClient = {
reconnectServer: vi.fn().mockRejectedValue(marker),
getServerStatuses: vi.fn(() => []),
getToolDefinitions: vi.fn(() => []),
};
const mcpService = buildService(mockRedis, mockMcpClient);
const loggerErrorSpy = vi.spyOn(Logger.prototype, 'error').mockImplementation(() => undefined);
const payload: SlashCommandPayload = {
command: 'mcp',
args: 'reconnect my-server',
conversationId,
};
const result = await mcpService.execute(payload, userScope);
expect(result.success).toBe(false);
expect(result.command).toBe('mcp');
expect(result.message).toBe(
'Failed to reconnect MCP server "my-server" due to an internal error.',
);
expect(result.message).not.toContain('distinctive-mcp-reconnect-failure-token-C33e');
expect(loggerErrorSpy).toHaveBeenCalled();
const loggedRawError = loggerErrorSpy.mock.calls.some((call) => call.includes(marker));
expect(loggedRawError).toBe(true);
loggerErrorSpy.mockRestore();
});
});
@@ -36,12 +36,6 @@ const authorization = {
),
};
const mockMcpClient = {
getServerStatuses: vi.fn(() => []),
getToolDefinitions: vi.fn(() => []),
reconnectServer: vi.fn().mockResolvedValue(undefined),
};
function buildExecutor(authorizationService: unknown = authorization): CommandExecutorService {
return new CommandExecutorService(
registry as never,
@@ -52,7 +46,7 @@ function buildExecutor(authorizationService: unknown = authorization): CommandEx
{ agents: {} } as never,
null,
null,
mockMcpClient as never,
null,
authorizationService as never,
);
}
@@ -34,7 +34,9 @@ export class CommandExecutorService {
@Optional()
@Inject(forwardRef(() => ChatGateway))
private readonly chatGateway: ChatGateway | null,
@Inject(McpClientService) private readonly mcpClient: McpClientService,
@Optional()
@Inject(McpClientService)
private readonly mcpClient: McpClientService | null,
@Optional()
@Inject(CommandAuthorizationService)
private readonly authorization: CommandAuthorizationService | null = null,
@@ -157,13 +159,8 @@ export class CommandExecutorService {
};
}
} catch (err) {
this.logger.error(`Command /${command} failed`, err);
return {
command,
conversationId,
success: false,
message: 'Command failed due to an internal error.',
};
this.logger.error(`Command /${command} failed: ${err}`);
return { command, conversationId, success: false, message: String(err) };
}
}
@@ -339,11 +336,11 @@ export class CommandExecutorService {
data: { agentId: newAgent.id, agentName: newAgent.name },
};
} catch (err) {
this.logger.error(`Failed to create agent "${namePart}" for user ${userId}`, err);
this.logger.error(`Failed to create agent: ${err}`);
return {
command: 'agent',
success: false,
message: 'Failed to create agent due to an internal error.',
message: `Failed to create agent: ${String(err)}`,
conversationId,
};
}
@@ -394,11 +391,11 @@ export class CommandExecutorService {
data: { agentId: agentConfig.id, agentName: agentConfig.name, model: agentConfig.model },
};
} catch (err) {
this.logger.error(`Failed to switch agent "${agentName}"`, err);
this.logger.error(`Failed to switch agent "${agentName}": ${err}`);
return {
command: 'agent',
success: false,
message: 'Failed to switch agent due to an internal error.',
message: `Failed to switch agent: ${String(err)}`,
conversationId,
};
}
@@ -546,6 +543,15 @@ export class CommandExecutorService {
args: string | null,
conversationId: string,
): Promise<SlashCommandResultPayload> {
if (!this.mcpClient) {
return {
command: 'mcp',
conversationId,
success: false,
message: 'MCP client service is not available.',
};
}
const action = args?.trim().split(/\s+/)[0] ?? 'status';
switch (action) {
@@ -602,12 +608,11 @@ export class CommandExecutorService {
message: `MCP server "${serverName}" reconnected successfully.`,
};
} catch (err) {
this.logger.error(`Failed to reconnect MCP server "${serverName}"`, err);
return {
command: 'mcp',
conversationId,
success: false,
message: `Failed to reconnect MCP server "${serverName}" due to an internal error.`,
message: `Failed to reconnect MCP server "${serverName}": ${err instanceof Error ? err.message : String(err)}`,
};
}
}
@@ -11,8 +11,6 @@
* - Unknown command returns descriptive error
*/
import { describe, it, expect, vi, beforeEach } from 'vitest';
import { CommandsModule } from './commands.module.js';
import { McpClientModule } from '../mcp-client/mcp-client.module.js';
import { CommandRegistryService } from './command-registry.service.js';
import { CommandExecutorService } from './command-executor.service.js';
import type { SlashCommandPayload } from '@mosaicstack/types';
@@ -49,12 +47,6 @@ const mockBrain = {
},
};
const mockMcpClient = {
getServerStatuses: vi.fn(() => []),
getToolDefinitions: vi.fn(() => []),
reconnectServer: vi.fn().mockResolvedValue(undefined),
};
// ─── Helpers ─────────────────────────────────────────────────────────────────
function buildRegistry(): CommandRegistryService {
@@ -73,7 +65,7 @@ function buildExecutor(registry: CommandRegistryService): CommandExecutorService
mockBrain as never,
null, // reloadService (optional)
null, // chatGateway (optional)
mockMcpClient as never,
null, // mcpClient (optional)
);
}
@@ -161,15 +153,6 @@ describe('CommandRegistryService — integration', () => {
}
});
// ─── Module Wiring Tests ──────────────────────────────────────────────────────
describe('CommandsModule — Nest wiring', () => {
it('CommandsModule imports McpClientModule in its Nest metadata', () => {
const imports = Reflect.getMetadata('imports', CommandsModule) ?? [];
expect(imports).toContain(McpClientModule);
});
});
// ─── Executor Tests ───────────────────────────────────────────────────────────
describe('CommandExecutorService — integration', () => {
@@ -276,14 +259,4 @@ describe('CommandExecutorService — integration', () => {
expect(result.command).toBe(cmd);
});
}
// /mcp status reaches the required McpClientService and never reports it unavailable
it('/mcp status calls the wired McpClientService and reports the no-servers message', async () => {
const payload: SlashCommandPayload = { command: 'mcp', conversationId };
const result = await executor.execute(payload, userScope);
expect(mockMcpClient.getServerStatuses).toHaveBeenCalledOnce();
expect(result.success).toBe(true);
expect(result.message).toContain('No MCP servers configured.');
expect(result.message).not.toBe('MCP client service is not available.');
});
});
+1 -7
View File
@@ -4,7 +4,6 @@ import type { MosaicConfig } from '@mosaicstack/config';
import { MOSAIC_CONFIG } from '../config/config.module.js';
import { ChatModule } from '../chat/chat.module.js';
import { GCModule } from '../gc/gc.module.js';
import { McpClientModule } from '../mcp-client/mcp-client.module.js';
import { ReloadModule } from '../reload/reload.module.js';
import { CommandAuthorizationService } from './command-authorization.service.js';
import { CommandExecutorService } from './command-executor.service.js';
@@ -15,12 +14,7 @@ import { COMMANDS_REDIS } from './commands.tokens.js';
const COMMANDS_QUEUE_HANDLE = 'COMMANDS_QUEUE_HANDLE';
@Module({
imports: [
GCModule,
McpClientModule,
forwardRef(() => ReloadModule),
forwardRef(() => ChatModule),
],
imports: [GCModule, forwardRef(() => ReloadModule), forwardRef(() => ChatModule)],
providers: [
{
provide: COMMANDS_QUEUE_HANDLE,
@@ -1,19 +0,0 @@
import 'reflect-metadata';
import { Test } from '@nestjs/testing';
import { describe, expect, it } from 'vitest';
import { CoordModule } from './coord.module.js';
import { InteractionCoordinationService } from './interaction-coordination.service.js';
import { AuthGuard } from '../auth/auth.guard.js';
describe('CoordModule DI (compiled-metadata boot)', () => {
it('resolves InteractionCoordinationService through Nest DI', async () => {
const moduleRef = await Test.createTestingModule({ imports: [CoordModule] })
.overrideGuard(AuthGuard)
.useValue({ canActivate: (): boolean => true })
.compile();
expect(moduleRef.get(InteractionCoordinationService)).toBeInstanceOf(
InteractionCoordinationService,
);
await moduleRef.close();
});
});
@@ -1,4 +1,4 @@
import { Inject, Injectable, Optional } from '@nestjs/common';
import { Inject, Injectable } from '@nestjs/common';
import {
InteractionCoordinationClient,
type CoordinationObservation,
@@ -13,7 +13,6 @@ import type { CreateHandoffDto } from './interaction-coordination.dto.js';
export const COORDINATION_PORT = Symbol('COORDINATION_PORT');
export const COORDINATION_CONFIG = Symbol('COORDINATION_CONFIG');
export const HANDOFF_ID_FACTORY = Symbol('HANDOFF_ID_FACTORY');
const HANDOFF_TRACKING_TTL_MS = 60 * 60 * 1_000;
const MAX_TRACKED_HANDOFFS = 1_000;
@@ -61,8 +60,6 @@ export class InteractionCoordinationService {
constructor(
@Inject(COORDINATION_PORT) private readonly port: InteractionCoordinationPort,
@Inject(COORDINATION_CONFIG) private readonly config: InteractionCoordinationConfig,
@Optional()
@Inject(HANDOFF_ID_FACTORY)
private readonly handoffIdFactory: () => string = (): string => crypto.randomUUID(),
) {}
@@ -1,164 +0,0 @@
import 'reflect-metadata';
import {
type CanActivate,
type ExecutionContext,
type INestApplication,
ValidationPipe,
} from '@nestjs/common';
import { FastifyAdapter, type NestFastifyApplication } from '@nestjs/platform-fastify';
import { Test } from '@nestjs/testing';
import request from 'supertest';
import { afterAll, beforeAll, beforeEach, describe, expect, it } from 'vitest';
import { AuthGuard } from '../auth/auth.guard.js';
import { HarnessRegistry } from './harness.registry.js';
import { HARNESS_REGISTRY } from './harness.tokens.js';
import { HarnessSelectionRepository } from './harness-selection.repository.js';
import { FakeHarnessAdapter } from './testing/fake-harness.adapter.js';
// Import the REAL module (not a hand-listed controllers+mocks list) so an
// unresolved provider fails at app.init() — the #1145-class DI-boot guard.
import { HarnessModule } from './harness.module.js';
// A known-available tuple from the fake adapter's default catalog.
const VALID = { harnessId: 'fake', providerId: 'fake-openai', modelId: 'fake-mini' };
// A tuple whose provider/model are not in any catalog.
const UNKNOWN = { harnessId: 'fake', providerId: 'ghost-provider', modelId: 'ghost-model' };
// A tuple that is known in the catalog but flagged unavailable.
const UNAVAILABLE = { harnessId: 'fake', providerId: 'fake-openai', modelId: 'fake-legacy' };
const authGuard: CanActivate = {
canActivate(context: ExecutionContext): boolean {
const requestContext = context.switchToHttp().getRequest<{ user?: { id: string } }>();
requestContext.user = { id: 'user-1' };
return true;
},
};
function registryWithFake(): HarnessRegistry {
const registry = new HarnessRegistry();
registry.register(new FakeHarnessAdapter({ id: 'fake' }));
return registry;
}
describe('Harness selection HTTP surface', () => {
let app: INestApplication;
let repository: HarnessSelectionRepository;
beforeAll(async () => {
const moduleRef = await Test.createTestingModule({
imports: [HarnessModule],
})
.overrideGuard(AuthGuard)
.useValue(authGuard)
.overrideProvider(HARNESS_REGISTRY)
.useValue(registryWithFake())
.compile();
// Real in-memory repository from the module graph — proves the module wired it.
repository = moduleRef.get(HarnessSelectionRepository);
app = moduleRef.createNestApplication<NestFastifyApplication>(new FastifyAdapter());
app.useGlobalPipes(
new ValidationPipe({ whitelist: true, forbidNonWhitelisted: true, transform: true }),
);
await app.init();
await app.getHttpAdapter().getInstance().ready();
});
beforeEach(() => {
// Reset owner-scoped state between tests via the public API surface.
repository.set({ userId: 'user-1', tenantId: 'user-1' }, VALID);
});
afterAll(async () => {
await app.close();
});
it('GET selection is server-scoped and ignores caller-supplied scope in the query', async () => {
const response = await request(app.getHttpServer())
.get('/api/chat/preferences/selection')
.query({ userId: 'attacker', tenantId: 'attacker-tenant', seatId: 'attacker-seat' });
expect(response.status).toBe(200);
// The returned selection is user-1's (guard-derived scope), not the query's.
expect(response.body.selection).toEqual(VALID);
});
it('PUT with a valid structured tuple persists and round-trips via GET', async () => {
const next = { harnessId: 'fake', providerId: 'fake-openai', modelId: 'fake-pro' };
const put = await request(app.getHttpServer())
.put('/api/chat/preferences/selection')
.send(next)
.set('Content-Type', 'application/json');
expect(put.status).toBe(200);
expect(put.body.selection).toEqual(next);
const get = await request(app.getHttpServer()).get('/api/chat/preferences/selection');
expect(get.status).toBe(200);
expect(get.body.selection).toEqual(next);
});
it('PUT with FREE TEXT is rejected 400 and does not mutate the stored selection', async () => {
const response = await request(app.getHttpServer())
.put('/api/chat/preferences/selection')
.send({ selection: 'gpt-4o' })
.set('Content-Type', 'application/json');
expect(response.status).toBe(400);
const get = await request(app.getHttpServer()).get('/api/chat/preferences/selection');
expect(get.body.selection).toEqual(VALID);
});
it.each([
['seatId', { ...VALID, seatId: 'attacker-seat' }],
['tenantId', { ...VALID, tenantId: 'attacker-tenant' }],
['userId', { ...VALID, userId: 'attacker' }],
['nativeSessionPath', { ...VALID, nativeSessionPath: '/var/native/x.jsonl' }],
['executable', { ...VALID, executable: '/usr/bin/evil' }],
['home', { ...VALID, home: '/home/attacker' }],
['cwd', { ...VALID, cwd: '/tmp/attacker' }],
])(
'PUT with an extra authority-bearing field (%s) is rejected 400 and does not mutate stored selection',
async (_name, body) => {
const response = await request(app.getHttpServer())
.put('/api/chat/preferences/selection')
.send(body)
.set('Content-Type', 'application/json');
expect(response.status).toBe(400);
const get = await request(app.getHttpServer()).get('/api/chat/preferences/selection');
expect(get.body.selection).toEqual(VALID);
},
);
it('PUT with an UNKNOWN tuple returns selection_invalid, unchanged and echoed unchanged (no fallback)', async () => {
const response = await request(app.getHttpServer())
.put('/api/chat/preferences/selection')
.send(UNKNOWN)
.set('Content-Type', 'application/json');
expect(response.status).toBe(422);
expect(response.body.code).toBe('selection_invalid');
// Echoed back unchanged: no first-row / first-provider substitution.
expect(response.body.selection).toEqual(UNKNOWN);
const get = await request(app.getHttpServer()).get('/api/chat/preferences/selection');
expect(get.body.selection).toEqual(VALID);
});
it('PUT with a KNOWN-but-UNAVAILABLE tuple returns model_unavailable, unchanged (distinct from selection_invalid)', async () => {
const response = await request(app.getHttpServer())
.put('/api/chat/preferences/selection')
.send(UNAVAILABLE)
.set('Content-Type', 'application/json');
expect(response.status).toBe(422);
expect(response.body.code).toBe('model_unavailable');
expect(response.body.selection).toEqual(UNAVAILABLE);
const get = await request(app.getHttpServer()).get('/api/chat/preferences/selection');
expect(get.body.selection).toEqual(VALID);
});
});
@@ -1,46 +0,0 @@
import { Body, Controller, Get, HttpException, HttpStatus, Put, UseGuards } from '@nestjs/common';
import { AuthGuard } from '../auth/auth.guard.js';
import { CurrentUser } from '../auth/current-user.decorator.js';
import { scopeFromUser, type AuthenticatedUserLike } from '../auth/session-scope.js';
import { HarnessOperationError } from './harness.registry.js';
import { HarnessSelectionService } from './harness-selection.service.js';
import { HarnessSelectionInputDto, type SelectionResponseDto } from './harness.dto.js';
/**
* Chat-preferences selection surface. The scope is ALWAYS derived on the server
* from the authenticated user (`scopeFromUser(CurrentUser)`); the request body and
* query string can never name another user, tenant, or seat. A typed selection
* failure (unknown tuple → `selection_invalid`, known-but-unavailable →
* `model_unavailable`) is returned as 422 with the requested tuple echoed back
* unchanged, and never mutates the stored selection.
*/
@Controller('api/chat/preferences/selection')
@UseGuards(AuthGuard)
export class HarnessSelectionController {
constructor(private readonly selection: HarnessSelectionService) {}
@Get()
get(@CurrentUser() user: AuthenticatedUserLike): SelectionResponseDto {
return { selection: this.selection.getSelection(scopeFromUser(user)) };
}
@Put()
async put(
@CurrentUser() user: AuthenticatedUserLike,
@Body() dto: HarnessSelectionInputDto,
): Promise<SelectionResponseDto> {
try {
const stored = await this.selection.setSelection(scopeFromUser(user), {
harnessId: dto.harnessId,
providerId: dto.providerId,
modelId: dto.modelId,
});
return { selection: stored };
} catch (error) {
if (error instanceof HarnessOperationError) {
throw new HttpException(error.dto, HttpStatus.UNPROCESSABLE_ENTITY);
}
throw error;
}
}
}
@@ -1,90 +0,0 @@
import { randomUUID } from 'node:crypto';
import { Inject, Injectable } from '@nestjs/common';
import type { HarnessSelection } from '@mosaicstack/types';
import type { ActorTenantScope } from '../auth/session-scope.js';
import {
HarnessAdapterUnavailableError,
HarnessRegistry,
operationError,
} from './harness.registry.js';
import { HARNESS_REGISTRY } from './harness.tokens.js';
import { readContextFromScope } from './harness.dto.js';
import { HarnessSelectionRepository } from './harness-selection.repository.js';
/**
* Selection logic for the Slice-Zero chat-preferences surface. It validates the
* requested harness/provider/model tuple against the live catalog with NO
* fallback substitution, then persists it owner-scoped. The stored selection is
* only ever mutated when the tuple is valid AND available.
*/
@Injectable()
export class HarnessSelectionService {
constructor(
@Inject(HARNESS_REGISTRY) private readonly registry: HarnessRegistry,
private readonly repository: HarnessSelectionRepository,
) {}
getSelection(scope: ActorTenantScope): HarnessSelection | null {
return this.repository.get(scope);
}
async setSelection(
scope: ActorTenantScope,
selection: HarnessSelection,
): Promise<HarnessSelection> {
// Throws HarnessOperationError (selection_invalid / model_unavailable) with the
// requested tuple echoed back unchanged. The store is untouched on any throw.
await this.assertSelectionAvailable(scope, selection);
return this.repository.set(scope, selection);
}
private async assertSelectionAvailable(
scope: ActorTenantScope,
selection: HarnessSelection,
): Promise<void> {
const correlationId = randomUUID();
let adapter;
try {
adapter = this.registry.get(selection.harnessId);
} catch (error) {
if (error instanceof HarnessAdapterUnavailableError) {
// An unknown harness makes the whole tuple invalid — no fallback adapter.
throw operationError(
'selection_invalid',
'The requested harness/provider/model tuple is not in the catalog.',
selection,
correlationId,
);
}
throw error;
}
const catalog = await adapter.catalog(readContextFromScope(scope));
const entry = catalog.models.find(
(candidate) =>
candidate.harnessId === selection.harnessId &&
candidate.providerId === selection.providerId &&
candidate.modelId === selection.modelId,
);
if (!entry) {
// No first-row / first-provider fallback: reject the requested tuple unchanged.
throw operationError(
'selection_invalid',
'The requested harness/provider/model tuple is not in the catalog.',
selection,
correlationId,
);
}
if (entry.availability === 'unavailable') {
throw operationError(
'model_unavailable',
'The requested model is currently unavailable.',
selection,
correlationId,
true,
);
}
}
}
@@ -1,138 +0,0 @@
import 'reflect-metadata';
import {
type CanActivate,
type ExecutionContext,
type INestApplication,
ValidationPipe,
} from '@nestjs/common';
import { FastifyAdapter, type NestFastifyApplication } from '@nestjs/platform-fastify';
import { Test } from '@nestjs/testing';
import request from 'supertest';
import { afterAll, beforeAll, describe, expect, it } from 'vitest';
import { AuthGuard } from '../auth/auth.guard.js';
import { HarnessRegistry } from './harness.registry.js';
import { HARNESS_REGISTRY } from './harness.tokens.js';
import { FakeHarnessAdapter } from './testing/fake-harness.adapter.js';
// The real module under test — importing it (not a hand-listed controllers/mocks
// list) is what makes an unresolved provider fail loudly at app.init() (#1145 guard).
import { HarnessModule } from './harness.module.js';
// Fields that must NEVER surface on a browser-facing catalog/list response.
const FORBIDDEN_KEYS = [
'executable',
'executablePath',
'home',
'homeDir',
'cwd',
'workingDir',
'workingDirectory',
'nativeSessionPath',
'sessionPath',
'env',
'secret',
'secrets',
'token',
'apiKey',
];
function assertNoForbiddenLeak(payload: unknown): void {
const serialized = JSON.stringify(payload).toLowerCase();
for (const key of FORBIDDEN_KEYS) {
expect(serialized).not.toContain(key.toLowerCase());
}
}
const authGuard: CanActivate = {
canActivate(context: ExecutionContext): boolean {
const requestContext = context.switchToHttp().getRequest<{ user?: { id: string } }>();
requestContext.user = { id: 'user-1' };
return true;
},
};
function registryWithFake(): HarnessRegistry {
const registry = new HarnessRegistry();
registry.register(new FakeHarnessAdapter({ id: 'fake' }));
return registry;
}
describe('Harness catalog HTTP surface', () => {
let app: INestApplication;
beforeAll(async () => {
const moduleRef = await Test.createTestingModule({
imports: [HarnessModule],
})
.overrideGuard(AuthGuard)
.useValue(authGuard)
.overrideProvider(HARNESS_REGISTRY)
.useValue(registryWithFake())
.compile();
app = moduleRef.createNestApplication<NestFastifyApplication>(new FastifyAdapter());
app.useGlobalPipes(
new ValidationPipe({ whitelist: true, forbidNonWhitelisted: true, transform: true }),
);
await app.init();
await app.getHttpAdapter().getInstance().ready();
});
afterAll(async () => {
await app.close();
});
it('boots the real HarnessModule so all providers resolve at app.init()', () => {
// If HarnessModule failed to resolve a provider, beforeAll's app.init() would
// have thrown and this suite would never reach here.
expect(app).toBeDefined();
});
it('GET /api/harnesses returns 200 with safe fields only', async () => {
const response = await request(app.getHttpServer()).get('/api/harnesses');
expect(response.status).toBe(200);
expect(Array.isArray(response.body)).toBe(true);
expect(response.body.length).toBeGreaterThan(0);
const summary = response.body[0];
expect(Object.keys(summary).sort()).toEqual(['capabilities', 'displayName', 'id']);
expect(summary.id).toBe('fake');
expect(typeof summary.displayName).toBe('string');
expect(Array.isArray(summary.capabilities)).toBe(true);
assertNoForbiddenLeak(response.body);
});
it('GET /api/harnesses/:harnessId/catalog returns 200 with safe catalog fields only', async () => {
const response = await request(app.getHttpServer()).get('/api/harnesses/fake/catalog');
expect(response.status).toBe(200);
expect(response.body.harnessId).toBe('fake');
expect(typeof response.body.version).toBe('string');
expect(typeof response.body.fingerprint).toBe('string');
expect(Array.isArray(response.body.models)).toBe(true);
expect(response.body.models.length).toBeGreaterThan(0);
const entry = response.body.models[0];
// Whitelisted catalog-entry fields only (no executables/paths/secrets).
expect(Object.keys(entry).sort()).toEqual(
[
'authState',
'availability',
'displayName',
'harnessId',
'inputTypes',
'modelId',
'providerId',
'reasoningCapability',
].sort(),
);
assertNoForbiddenLeak(response.body);
});
it('GET catalog for an unknown harnessId returns a typed adapter_unavailable error, never a fallback catalog', async () => {
const response = await request(app.getHttpServer()).get('/api/harnesses/ghost-harness/catalog');
expect(response.status).toBe(404);
expect(response.body.code).toBe('adapter_unavailable');
// A fallback catalog would carry a models array; a typed error must not.
expect(response.body.models).toBeUndefined();
});
});
@@ -1,65 +0,0 @@
import {
Controller,
Get,
HttpException,
HttpStatus,
Inject,
Param,
UseGuards,
} from '@nestjs/common';
import { AuthGuard } from '../auth/auth.guard.js';
import { CurrentUser } from '../auth/current-user.decorator.js';
import { scopeFromUser, type AuthenticatedUserLike } from '../auth/session-scope.js';
import { HarnessAdapterUnavailableError, HarnessRegistry } from './harness.registry.js';
import { HARNESS_REGISTRY } from './harness.tokens.js';
import {
readContextFromScope,
toHarnessSummary,
toSafeCatalog,
type HarnessCatalogDto,
type HarnessSummaryDto,
} from './harness.dto.js';
/**
* Generic harness catalog surface. It exposes only harness-neutral, browser-safe
* fields (identity, capabilities, provider/model catalog) — never executables,
* native paths, home/cwd, env, or secrets. There is NO provider-probe route here;
* `/api/providers` and `POST /api/providers/test` are intentionally out of scope.
*/
@Controller('api/harnesses')
@UseGuards(AuthGuard)
export class HarnessController {
constructor(@Inject(HARNESS_REGISTRY) private readonly registry: HarnessRegistry) {}
@Get()
async list(@CurrentUser() user: AuthenticatedUserLike): Promise<HarnessSummaryDto[]> {
const context = readContextFromScope(scopeFromUser(user));
const summaries: HarnessSummaryDto[] = [];
for (const adapter of this.registry.list()) {
summaries.push(toHarnessSummary(await adapter.describe(context)));
}
return summaries;
}
@Get(':harnessId/catalog')
async catalog(
@CurrentUser() user: AuthenticatedUserLike,
@Param('harnessId') harnessId: string,
): Promise<HarnessCatalogDto> {
const context = readContextFromScope(scopeFromUser(user));
let adapter;
try {
adapter = this.registry.get(harnessId);
} catch (error) {
if (error instanceof HarnessAdapterUnavailableError) {
// Typed failure — NEVER a fallback catalog for an unknown harness id.
throw new HttpException(
{ code: error.code, message: error.message, harnessId },
HttpStatus.NOT_FOUND,
);
}
throw error;
}
return toSafeCatalog(await adapter.catalog(context));
}
}
-116
View File
@@ -1,116 +0,0 @@
import { randomUUID } from 'node:crypto';
import { IsNotEmpty, IsString } from 'class-validator';
import type {
HarnessActorContext,
HarnessAuthState,
HarnessCapability,
HarnessCatalog,
HarnessCatalogEntry,
HarnessDescriptor,
HarnessInputType,
HarnessModelAvailability,
HarnessSelection,
} from '@mosaicstack/types';
import type { ActorTenantScope } from '../auth/session-scope.js';
/**
* Structured selection tuple accepted on `PUT /api/chat/preferences/selection`.
*
* The body is a STRUCTURED tuple (harness + provider + model), never a free-text
* model string. With `ValidationPipe({ whitelist: true, forbidNonWhitelisted: true })`
* any extra property — including smuggled server-authority fields such as
* `seatId`, `tenantId`, `userId`, `nativeSessionPath`, `executable`, `home`, `cwd` —
* is rejected with 400. There is deliberately no field through which a caller can
* name a scope; scope is derived on the server from the authenticated session.
*/
export class HarnessSelectionInputDto {
@IsString()
@IsNotEmpty()
harnessId!: string;
@IsString()
@IsNotEmpty()
providerId!: string;
@IsString()
@IsNotEmpty()
modelId!: string;
}
/** Browser-safe harness summary — identity and capabilities only. */
export interface HarnessSummaryDto {
readonly id: string;
readonly displayName: string;
readonly capabilities: readonly HarnessCapability[];
}
/** Browser-safe catalog entry — no executables, paths, secrets, or env. */
export interface HarnessCatalogEntryDto {
readonly harnessId: string;
readonly providerId: string;
readonly modelId: string;
readonly displayName: string;
readonly reasoningCapability: boolean;
readonly inputTypes: readonly HarnessInputType[];
readonly authState: HarnessAuthState;
readonly availability: HarnessModelAvailability;
}
/** Browser-safe catalog envelope. */
export interface HarnessCatalogDto {
readonly harnessId: string;
readonly version: string;
readonly fingerprint: string;
readonly models: readonly HarnessCatalogEntryDto[];
}
/** Response envelope for the caller's current selection (null when unset). */
export interface SelectionResponseDto {
readonly selection: HarnessSelection | null;
}
/**
* Derive a server-trusted {@link HarnessActorContext} for read operations from the
* session-derived {@link ActorTenantScope}. All authority originates on the server;
* nothing here is caller-supplied. A fresh correlation id is minted per call.
*/
export function readContextFromScope(scope: ActorTenantScope): HarnessActorContext {
return {
actorId: scope.userId,
tenantId: scope.tenantId,
seatId: scope.userId,
correlationId: randomUUID(),
};
}
/** Project a descriptor onto the browser-safe summary shape (whitelist by construction). */
export function toHarnessSummary(descriptor: HarnessDescriptor): HarnessSummaryDto {
return {
id: descriptor.id,
displayName: descriptor.displayName,
capabilities: [...descriptor.capabilities],
};
}
/** Project a catalog onto the browser-safe shape (whitelist by construction). */
export function toSafeCatalog(catalog: HarnessCatalog): HarnessCatalogDto {
return {
harnessId: catalog.harnessId,
version: catalog.version,
fingerprint: catalog.fingerprint,
models: catalog.models.map(toSafeCatalogEntry),
};
}
function toSafeCatalogEntry(entry: HarnessCatalogEntry): HarnessCatalogEntryDto {
return {
harnessId: entry.harnessId,
providerId: entry.providerId,
modelId: entry.modelId,
displayName: entry.displayName,
reasoningCapability: entry.reasoningCapability,
inputTypes: [...entry.inputTypes],
authState: entry.authState,
availability: entry.availability,
};
}
@@ -1,28 +0,0 @@
import { Module } from '@nestjs/common';
import { HarnessRegistry } from './harness.registry.js';
import { HarnessService } from './harness.service.js';
import { HARNESS_REGISTRY, HARNESS_SERVICE } from './harness.tokens.js';
import { HarnessController } from './harness.controller.js';
import { HarnessSelectionController } from './harness-selection.controller.js';
import { HarnessSelectionService } from './harness-selection.service.js';
import { HarnessSelectionRepository } from './harness-selection.repository.js';
/**
* Wires the harness-neutral registry/service (Task Two) together with the
* Slice-Zero catalog and selection HTTP surfaces (Task Three).
*
* The registry is provided empty here; real harness adapters are registered in a
* later task. Because the controllers/services resolve their collaborators through
* this real module graph, an unresolved provider fails loudly at `app.init()`.
*/
@Module({
controllers: [HarnessController, HarnessSelectionController],
providers: [
{ provide: HARNESS_REGISTRY, useFactory: () => new HarnessRegistry() },
{ provide: HARNESS_SERVICE, useClass: HarnessService },
HarnessSelectionRepository,
HarnessSelectionService,
],
exports: [HARNESS_REGISTRY, HARNESS_SERVICE],
})
export class HarnessModule {}
@@ -1,69 +0,0 @@
import { describe, expect, it } from 'vitest';
import {
HarnessAdapterUnavailableError,
HarnessRegistrationError,
HarnessRegistry,
} from './harness.registry.js';
import { FakeHarnessAdapter } from './testing/fake-harness.adapter.js';
describe('HarnessRegistry', () => {
it('registers and looks up an adapter by harness id', () => {
const registry = new HarnessRegistry();
const adapter = new FakeHarnessAdapter({ id: 'fake' });
registry.register(adapter);
expect(registry.get('fake')).toBe(adapter);
expect(registry.has('fake')).toBe(true);
expect(registry.list().map((entry) => entry.id)).toEqual(['fake']);
});
it('rejects a blank adapter id', () => {
const registry = new HarnessRegistry();
let error: unknown;
try {
registry.register(new FakeHarnessAdapter({ id: ' ' }));
} catch (caught) {
error = caught;
}
expect(error).toBeInstanceOf(HarnessRegistrationError);
expect((error as HarnessRegistrationError).reason).toBe('blank_id');
expect(registry.list()).toEqual([]);
});
it('rejects a duplicate adapter id', () => {
const registry = new HarnessRegistry();
registry.register(new FakeHarnessAdapter({ id: 'fake' }));
let error: unknown;
try {
registry.register(new FakeHarnessAdapter({ id: 'fake' }));
} catch (caught) {
error = caught;
}
expect(error).toBeInstanceOf(HarnessRegistrationError);
expect((error as HarnessRegistrationError).reason).toBe('duplicate_id');
expect((error as HarnessRegistrationError).harnessId).toBe('fake');
// The original registration is untouched.
expect(registry.list()).toHaveLength(1);
});
it('returns adapter_unavailable for an unknown harness id', () => {
const registry = new HarnessRegistry();
let error: unknown;
try {
registry.get('missing');
} catch (caught) {
error = caught;
}
expect(error).toBeInstanceOf(HarnessAdapterUnavailableError);
expect((error as HarnessAdapterUnavailableError).code).toBe('adapter_unavailable');
expect((error as HarnessAdapterUnavailableError).harnessId).toBe('missing');
expect(registry.has('missing')).toBe(false);
});
});
@@ -1,100 +0,0 @@
import { Injectable } from '@nestjs/common';
import type {
HarnessAdapter,
HarnessErrorCode,
HarnessErrorDto,
HarnessSelection,
} from '@mosaicstack/types';
/**
* A typed harness operation failure that carries a fully-formed, browser-safe
* {@link HarnessErrorDto}. The DTO's `selection` is always the exact requested
* tuple — there is no field through which a substituted "effective" selection
* could ever be reported.
*/
export class HarnessOperationError extends Error {
readonly code: HarnessErrorCode;
readonly dto: HarnessErrorDto;
constructor(dto: HarnessErrorDto) {
super(dto.message);
this.name = 'HarnessOperationError';
this.code = dto.code;
this.dto = dto;
}
}
/** Build a {@link HarnessOperationError} that echoes the requested selection unchanged. */
export function operationError(
code: HarnessErrorCode,
message: string,
selection: HarnessSelection,
correlationId: string,
retryable = false,
): HarnessOperationError {
return new HarnessOperationError({ code, message, retryable, correlationId, selection });
}
/** Raised when an unknown harness id is looked up. Discriminated by `code`. */
export class HarnessAdapterUnavailableError extends Error {
readonly code = 'adapter_unavailable' as const satisfies HarnessErrorCode;
constructor(readonly harnessId: string) {
super(`No harness adapter is registered for id "${harnessId}".`);
this.name = 'HarnessAdapterUnavailableError';
}
}
export type HarnessRegistrationFailure = 'blank_id' | 'duplicate_id';
/** Raised when an adapter cannot be registered (blank or duplicate id). */
export class HarnessRegistrationError extends Error {
constructor(
readonly reason: HarnessRegistrationFailure,
readonly harnessId: string,
) {
super(
reason === 'blank_id'
? 'A harness adapter id must be a non-empty string.'
: `A harness adapter is already registered for id "${harnessId}".`,
);
this.name = 'HarnessRegistrationError';
}
}
/**
* Harness-neutral adapter registry. Adapters are keyed by their harness id.
* Registration rejects blank and duplicate ids; lookup of an unknown id fails
* with {@link HarnessAdapterUnavailableError} (`adapter_unavailable`).
*/
@Injectable()
export class HarnessRegistry {
private readonly adapters = new Map<string, HarnessAdapter>();
register(adapter: HarnessAdapter): void {
const id = adapter.id;
if (typeof id !== 'string' || id.trim().length === 0) {
throw new HarnessRegistrationError('blank_id', id ?? '');
}
if (this.adapters.has(id)) {
throw new HarnessRegistrationError('duplicate_id', id);
}
this.adapters.set(id, adapter);
}
get(harnessId: string): HarnessAdapter {
const adapter = this.adapters.get(harnessId);
if (!adapter) {
throw new HarnessAdapterUnavailableError(harnessId);
}
return adapter;
}
has(harnessId: string): boolean {
return this.adapters.has(harnessId);
}
list(): readonly HarnessAdapter[] {
return [...this.adapters.values()];
}
}
@@ -1,227 +0,0 @@
import { describe, expect, it } from 'vitest';
import type { HarnessActorContext, HarnessCapability, HarnessSelection } from '@mosaicstack/types';
import { HARNESS_CAPABILITIES } from '@mosaicstack/types';
import { HarnessOperationError, HarnessRegistry } from './harness.registry.js';
import {
HarnessScopeViolationError,
HarnessService,
type TrustedGatewayScope,
} from './harness.service.js';
import { FakeHarnessAdapter } from './testing/fake-harness.adapter.js';
const SCOPE: TrustedGatewayScope = {
actorId: 'actor-trusted',
tenantId: 'tenant-trusted',
seatId: 'seat-trusted',
correlationId: 'correlation-trusted',
};
const READ_CONTEXT: HarnessActorContext = {
actorId: SCOPE.actorId,
tenantId: SCOPE.tenantId,
seatId: SCOPE.seatId,
correlationId: SCOPE.correlationId,
};
function setup(capabilities?: readonly HarnessCapability[]) {
const registry = new HarnessRegistry();
const adapter = new FakeHarnessAdapter({ id: 'fake', capabilities });
registry.register(adapter);
const service = new HarnessService(registry);
return { registry, adapter, service };
}
async function availableSelection(adapter: FakeHarnessAdapter): Promise<HarnessSelection> {
const catalog = await adapter.catalog(READ_CONTEXT);
const entry = catalog.models.find((model) => model.availability === 'available');
if (!entry) {
throw new Error('fixture requires an available model');
}
return { harnessId: entry.harnessId, providerId: entry.providerId, modelId: entry.modelId };
}
describe('HarnessService', () => {
it('derives the actor context from trusted scope on create', async () => {
const { service, adapter } = setup();
const selection = await availableSelection(adapter);
const snapshot = await service.createSession(SCOPE, {
conversationId: 'conversation-1',
selection,
});
expect(snapshot.seatId).toBe(SCOPE.seatId);
expect(snapshot.state).toBe('idle');
expect(snapshot.selection).toEqual(selection);
expect(snapshot.nativeSessionId).toBeTruthy();
});
it('rejects server-authority fields supplied by an external caller', async () => {
const { service, adapter } = setup();
const selection = await availableSelection(adapter);
const hostile = {
conversationId: 'conversation-1',
selection,
seatId: 'attacker-seat',
executablePath: '/usr/bin/evil',
home: '/home/attacker',
cwd: '/tmp/attacker',
nativeSessionPath: '/var/native/attacker.jsonl',
} as unknown as Parameters<HarnessService['createSession']>[1];
let error: unknown;
try {
await service.createSession(SCOPE, hostile);
} catch (caught) {
error = caught;
}
expect(error).toBeInstanceOf(HarnessScopeViolationError);
expect((error as HarnessScopeViolationError).field).toBe('seatId');
});
it('returns adapter_unavailable for an unknown harness id, echoing the requested tuple', async () => {
const { service } = setup();
const selection: HarnessSelection = {
harnessId: 'ghost-harness',
providerId: 'p',
modelId: 'm',
};
let error: unknown;
try {
await service.createSession(SCOPE, { conversationId: 'conversation-1', selection });
} catch (caught) {
error = caught;
}
expect(error).toBeInstanceOf(HarnessOperationError);
const dto = (error as HarnessOperationError).dto;
expect(dto.code).toBe('adapter_unavailable');
expect(dto.selection).toEqual(selection);
expect(dto.correlationId).toBe(SCOPE.correlationId);
});
it('returns selection_invalid for an unknown provider/model tuple, unchanged', async () => {
const { service } = setup();
const selection: HarnessSelection = {
harnessId: 'fake',
providerId: 'ghost-provider',
modelId: 'ghost-model',
};
let error: unknown;
try {
await service.createSession(SCOPE, { conversationId: 'conversation-1', selection });
} catch (caught) {
error = caught;
}
expect(error).toBeInstanceOf(HarnessOperationError);
const dto = (error as HarnessOperationError).dto;
expect(dto.code).toBe('selection_invalid');
expect(dto.selection).toEqual(selection);
});
it('returns model_unavailable without falling back for a known unavailable model', async () => {
const { service, adapter } = setup();
const catalog = await adapter.catalog(READ_CONTEXT);
const unavailable = catalog.models.find((entry) => entry.availability === 'unavailable');
expect(unavailable).toBeDefined();
const selection: HarnessSelection = {
harnessId: unavailable!.harnessId,
providerId: unavailable!.providerId,
modelId: unavailable!.modelId,
};
let error: unknown;
try {
await service.createSession(SCOPE, { conversationId: 'conversation-1', selection });
} catch (caught) {
error = caught;
}
expect(error).toBeInstanceOf(HarnessOperationError);
const dto = (error as HarnessOperationError).dto;
expect(dto.code).toBe('model_unavailable');
// No substitution: the DTO tuple is exactly what was requested.
expect(dto.selection).toEqual(selection);
});
it('gives create, resume, detach, evict, and end distinct observable effects', async () => {
const { service, adapter } = setup();
const selection = await availableSelection(adapter);
const created = await service.createSession(SCOPE, {
conversationId: 'conversation-create',
selection,
});
expect(created.state).toBe('idle');
expect(created.processId).toBeTruthy();
expect(created.attachedClientIds).toEqual([]);
const resumed = await service.resumeSession(SCOPE, {
conversationId: 'conversation-resume',
nativeSessionId: 'native-preexisting-123',
selection,
});
// Resume binds the supplied native session; create mints a fresh one.
expect(resumed.nativeSessionId).toBe('native-preexisting-123');
expect(resumed.nativeSessionId).not.toBe(created.nativeSessionId);
await service.attach(SCOPE, {
conversationId: 'conversation-create',
clientId: 'browser-1',
});
const afterAttach = await service.snapshot(SCOPE, 'conversation-create');
expect(afterAttach.attachedClientIds).toEqual(['browser-1']);
const afterDetach = await service.detach(SCOPE, {
conversationId: 'conversation-create',
clientId: 'browser-1',
});
// Detach removes the browser attachment only; the process stays alive.
expect(afterDetach.attachedClientIds).toEqual([]);
expect(afterDetach.state).toBe('idle');
expect(afterDetach.processId).toBeTruthy();
const afterEvict = await service.evict(SCOPE, {
conversationId: 'conversation-create',
reason: 'idle_timeout',
});
// Evict stops the process but retains the resumable native session.
expect(afterEvict.state).toBe('evicted');
expect(afterEvict.processId).toBeUndefined();
expect(afterEvict.nativeSessionId).toBe(created.nativeSessionId);
const afterEnd = await service.end(SCOPE, {
conversationId: 'conversation-create',
reason: 'session_ended',
});
// End destructively terminates the native session.
expect(afterEnd.state).toBe('ended');
});
it('fails typed when an unsupported capability is exercised', async () => {
const withoutExtensionUi = HARNESS_CAPABILITIES.filter(
(capability) => capability !== 'extensionUi',
);
const { service, adapter } = setup(withoutExtensionUi);
const selection = await availableSelection(adapter);
await service.createSession(SCOPE, { conversationId: 'conversation-1', selection });
let error: unknown;
try {
await service.respondInteraction(SCOPE, {
conversationId: 'conversation-1',
response: { requestId: 'interaction-1', type: 'confirm', accepted: true },
});
} catch (caught) {
error = caught;
}
expect(error).toBeInstanceOf(HarnessOperationError);
expect((error as HarnessOperationError).dto.code).toBe('interaction_unsupported');
});
});
-285
View File
@@ -1,285 +0,0 @@
import { Inject, Injectable } from '@nestjs/common';
import type {
HarnessActorContext,
HarnessAdapter,
HarnessCatalog,
HarnessCloseReason,
HarnessInteractionResponse,
HarnessSelection,
HarnessSessionHandle,
HarnessSessionSnapshot,
} from '@mosaicstack/types';
import {
HarnessAdapterUnavailableError,
HarnessRegistry,
operationError,
} from './harness.registry.js';
import { HARNESS_REGISTRY } from './harness.tokens.js';
/**
* Trusted, server-derived authority. In production this is produced by the
* Gateway from the authenticated session — never from a browser/caller DTO.
*/
export interface TrustedGatewayScope {
readonly actorId: string;
readonly tenantId: string;
readonly seatId: string;
readonly correlationId: string;
}
/** Server-authority fields that must never arrive from an external request DTO. */
const FORBIDDEN_REQUEST_FIELDS = [
'actorId',
'tenantId',
'correlationId',
'seatId',
'seat',
'executable',
'executablePath',
'home',
'homeDir',
'cwd',
'workingDir',
'workingDirectory',
'nativeSessionPath',
'sessionPath',
] as const;
/** Raised when an external request DTO smuggles a server-authority field. */
export class HarnessScopeViolationError extends Error {
constructor(readonly field: string) {
super(`External request supplied server-authority field "${field}".`);
this.name = 'HarnessScopeViolationError';
}
}
export interface CreateHarnessSessionRequest {
readonly conversationId: string;
readonly selection: HarnessSelection;
}
export interface ResumeHarnessSessionRequest {
readonly conversationId: string;
readonly nativeSessionId: string;
readonly selection: HarnessSelection;
}
export interface AttachClientRequest {
readonly conversationId: string;
readonly clientId: string;
}
export interface DetachClientRequest {
readonly conversationId: string;
readonly clientId: string;
}
export interface EvictSessionRequest {
readonly conversationId: string;
readonly reason: HarnessCloseReason;
}
export interface EndSessionRequest {
readonly conversationId: string;
readonly reason: HarnessCloseReason;
}
export interface RespondInteractionRequest {
readonly conversationId: string;
readonly response: HarnessInteractionResponse;
}
interface ActiveSession {
readonly harnessId: string;
readonly handle: HarnessSessionHandle;
readonly correlationId: string;
}
/**
* Harness-neutral service. It derives the {@link HarnessActorContext} strictly
* from trusted Gateway scope, validates the selected provider/model tuple with
* NO fallback substitution, and exposes distinct create/resume/detach/evict/end
* lifecycle operations.
*/
@Injectable()
export class HarnessService {
private readonly sessions = new Map<string, ActiveSession>();
constructor(@Inject(HARNESS_REGISTRY) private readonly registry: HarnessRegistry) {}
async createSession(
scope: TrustedGatewayScope,
request: CreateHarnessSessionRequest,
): Promise<HarnessSessionSnapshot> {
assertTrustedRequest(request);
const { conversationId, selection } = request;
const adapter = this.resolveAdapter(scope, selection);
const context = deriveActorContext(scope);
await this.assertSelectionAvailable(scope, adapter.catalog(context), selection);
const handle = await adapter.create({ context, conversationId, selection });
this.sessions.set(conversationId, {
harnessId: selection.harnessId,
handle,
correlationId: scope.correlationId,
});
return handle.snapshot();
}
async resumeSession(
scope: TrustedGatewayScope,
request: ResumeHarnessSessionRequest,
): Promise<HarnessSessionSnapshot> {
assertTrustedRequest(request);
const { conversationId, nativeSessionId, selection } = request;
const adapter = this.resolveAdapter(scope, selection);
const context = deriveActorContext(scope);
await this.assertSelectionAvailable(scope, adapter.catalog(context), selection);
const handle = await adapter.resume({ context, conversationId, nativeSessionId, selection });
this.sessions.set(conversationId, {
harnessId: selection.harnessId,
handle,
correlationId: scope.correlationId,
});
return handle.snapshot();
}
async attach(
scope: TrustedGatewayScope,
request: AttachClientRequest,
): Promise<HarnessSessionSnapshot> {
assertTrustedRequest(request);
const handle = this.requireHandle(scope, request.conversationId);
await handle.attach({ clientId: request.clientId });
return handle.snapshot();
}
async detach(
scope: TrustedGatewayScope,
request: DetachClientRequest,
): Promise<HarnessSessionSnapshot> {
assertTrustedRequest(request);
const handle = this.requireHandle(scope, request.conversationId);
await handle.detach(request.clientId);
return handle.snapshot();
}
async evict(
scope: TrustedGatewayScope,
request: EvictSessionRequest,
): Promise<HarnessSessionSnapshot> {
assertTrustedRequest(request);
const handle = this.requireHandle(scope, request.conversationId);
await handle.evictProcess(request.reason);
return handle.snapshot();
}
async end(
scope: TrustedGatewayScope,
request: EndSessionRequest,
): Promise<HarnessSessionSnapshot> {
assertTrustedRequest(request);
const handle = this.requireHandle(scope, request.conversationId);
await handle.endSession(request.reason);
const snapshot = await handle.snapshot();
this.sessions.delete(request.conversationId);
return snapshot;
}
async respondInteraction(
scope: TrustedGatewayScope,
request: RespondInteractionRequest,
): Promise<void> {
assertTrustedRequest(request);
const handle = this.requireHandle(scope, request.conversationId);
await handle.respondInteraction(request.response);
}
async snapshot(
scope: TrustedGatewayScope,
conversationId: string,
): Promise<HarnessSessionSnapshot> {
const handle = this.requireHandle(scope, conversationId);
return handle.snapshot();
}
private resolveAdapter(scope: TrustedGatewayScope, selection: HarnessSelection): HarnessAdapter {
try {
return this.registry.get(selection.harnessId);
} catch (error) {
if (error instanceof HarnessAdapterUnavailableError) {
throw operationError('adapter_unavailable', error.message, selection, scope.correlationId);
}
throw error;
}
}
private async assertSelectionAvailable(
scope: TrustedGatewayScope,
catalogPromise: Promise<HarnessCatalog>,
selection: HarnessSelection,
): Promise<void> {
const catalog = await catalogPromise;
const entry = catalog.models.find(
(candidate) =>
candidate.harnessId === selection.harnessId &&
candidate.providerId === selection.providerId &&
candidate.modelId === selection.modelId,
);
if (!entry) {
// No first-row fallback: reject the requested tuple unchanged.
throw operationError(
'selection_invalid',
'The requested harness/provider/model tuple is not in the catalog.',
selection,
scope.correlationId,
);
}
if (entry.availability === 'unavailable') {
throw operationError(
'model_unavailable',
'The requested model is currently unavailable.',
selection,
scope.correlationId,
true,
);
}
}
private requireHandle(scope: TrustedGatewayScope, conversationId: string): HarnessSessionHandle {
const active = this.sessions.get(conversationId);
if (!active) {
throw operationError(
'session_not_found',
`No active harness session for conversation "${conversationId}".`,
{ harnessId: '', providerId: '', modelId: '' },
scope.correlationId,
);
}
return active.handle;
}
}
/** Build the actor context strictly from trusted scope. No caller data leaks in. */
export function deriveActorContext(scope: TrustedGatewayScope): HarnessActorContext {
return {
actorId: scope.actorId,
tenantId: scope.tenantId,
seatId: scope.seatId,
correlationId: scope.correlationId,
};
}
/** Reject any request object that carries a server-authority field. */
function assertTrustedRequest(request: object): void {
for (const field of FORBIDDEN_REQUEST_FIELDS) {
if (Object.prototype.hasOwnProperty.call(request, field)) {
throw new HarnessScopeViolationError(field);
}
}
}
// Re-export the typed operation error so callers importing from the service
// have the discriminated failure type without reaching into the registry.
export { HarnessOperationError } from './harness.registry.js';
@@ -1,11 +0,0 @@
/**
* Nest dependency-injection tokens for the harness-neutral registry and service.
*
* String tokens follow the existing Gateway convention (see `memory/memory.tokens.ts`)
* and remain valid Nest `InjectionToken`s for `@Inject(...)`.
*/
export const HARNESS_REGISTRY = 'HARNESS_REGISTRY' as const;
export const HARNESS_SERVICE = 'HARNESS_SERVICE' as const;
export type HarnessRegistryToken = typeof HARNESS_REGISTRY;
export type HarnessServiceToken = typeof HARNESS_SERVICE;
@@ -1,107 +0,0 @@
import { describe, expect, it } from 'vitest';
import type { HarnessActorContext, HarnessSelection } from '@mosaicstack/types';
import { HarnessOperationError } from '../harness.registry.js';
import { FakeHarnessAdapter } from './fake-harness.adapter.js';
import { runHarnessAdapterContract } from './harness-adapter.contract.js';
const CONTEXT: HarnessActorContext = {
actorId: 'actor-1',
tenantId: 'tenant-1',
seatId: 'seat-1',
correlationId: 'correlation-1',
};
// The reusable conformance suite. Task 13 re-runs it against the native Pi adapter.
runHarnessAdapterContract('FakeHarnessAdapter', () => new FakeHarnessAdapter({ id: 'fake' }));
describe('FakeHarnessAdapter no-substitution', () => {
it('never substitutes the first catalog row when a bogus selection is requested', async () => {
const adapter = new FakeHarnessAdapter({ id: 'fake' });
const catalog = await adapter.catalog(CONTEXT);
const firstRow = catalog.models[0];
if (!firstRow) {
throw new Error('fixture requires a catalog model');
}
const available = catalog.models.find(
(entry) => entry.availability === 'available' && entry.modelId !== firstRow.modelId,
);
expect(available).toBeDefined();
const selected: HarnessSelection = {
harnessId: available!.harnessId,
providerId: available!.providerId,
modelId: available!.modelId,
};
const handle = await adapter.create({
context: CONTEXT,
conversationId: 'conversation-1',
selection: selected,
});
const bogus: HarnessSelection = {
harnessId: 'fake',
providerId: 'ghost-provider',
modelId: 'ghost-model',
};
let error: unknown;
try {
await handle.setModel(bogus);
} catch (caught) {
error = caught;
}
expect(error).toBeInstanceOf(HarnessOperationError);
const dto = (error as HarnessOperationError).dto;
expect(dto.code).toBe('selection_invalid');
// The DTO echoes the exact requested tuple, unchanged.
expect(dto.selection).toEqual(bogus);
// No substitution to the first catalog row.
expect(dto.selection).not.toEqual({
harnessId: firstRow.harnessId,
providerId: firstRow.providerId,
modelId: firstRow.modelId,
});
// The active selection is untouched by the rejected request.
expect((await handle.snapshot()).selection).toEqual(selected);
});
it('reports model_unavailable with the unchanged tuple for a known but unavailable model', async () => {
const adapter = new FakeHarnessAdapter({ id: 'fake' });
const catalog = await adapter.catalog(CONTEXT);
const unavailable = catalog.models.find((entry) => entry.availability === 'unavailable');
const available = catalog.models.find((entry) => entry.availability === 'available');
expect(unavailable).toBeDefined();
expect(available).toBeDefined();
const startingSelection: HarnessSelection = {
harnessId: available!.harnessId,
providerId: available!.providerId,
modelId: available!.modelId,
};
const handle = await adapter.create({
context: CONTEXT,
conversationId: 'conversation-2',
selection: startingSelection,
});
const requested: HarnessSelection = {
harnessId: unavailable!.harnessId,
providerId: unavailable!.providerId,
modelId: unavailable!.modelId,
};
let error: unknown;
try {
await handle.setModel(requested);
} catch (caught) {
error = caught;
}
expect(error).toBeInstanceOf(HarnessOperationError);
const dto = (error as HarnessOperationError).dto;
expect(dto.code).toBe('model_unavailable');
expect(dto.selection).toEqual(requested);
expect((await handle.snapshot()).selection).toEqual(startingSelection);
});
});
@@ -1,248 +0,0 @@
import type {
AttachClient,
CreateHarnessSession,
HarnessAdapter,
HarnessActorContext,
HarnessCapability,
HarnessCatalog,
HarnessCatalogEntry,
HarnessCloseReason,
HarnessDescriptor,
HarnessEvent,
HarnessInteractionResponse,
HarnessPrompt,
HarnessPromptReceipt,
HarnessSelection,
HarnessSessionHandle,
HarnessSessionSnapshot,
HarnessSessionState,
ResumeHarnessSession,
} from '@mosaicstack/types';
import { HARNESS_CAPABILITIES } from '@mosaicstack/types';
import { operationError } from '../harness.registry.js';
export interface FakeHarnessAdapterOptions {
readonly id: string;
readonly capabilities?: readonly HarnessCapability[];
readonly catalog?: readonly HarnessCatalogEntry[];
}
const FAKE_PROVIDER = 'fake-openai';
function defaultCatalog(harnessId: string): readonly HarnessCatalogEntry[] {
return [
{
harnessId,
providerId: FAKE_PROVIDER,
modelId: 'fake-mini',
displayName: 'Fake Mini',
reasoningCapability: false,
inputTypes: ['text'],
authState: 'ready',
availability: 'available',
},
{
harnessId,
providerId: FAKE_PROVIDER,
modelId: 'fake-pro',
displayName: 'Fake Pro',
reasoningCapability: true,
inputTypes: ['text', 'image'],
authState: 'ready',
availability: 'available',
},
{
harnessId,
providerId: FAKE_PROVIDER,
modelId: 'fake-legacy',
displayName: 'Fake Legacy',
reasoningCapability: false,
inputTypes: ['text'],
authState: 'unavailable',
availability: 'unavailable',
},
];
}
function matches(entry: HarnessCatalogEntry, selection: HarnessSelection): boolean {
return (
entry.harnessId === selection.harnessId &&
entry.providerId === selection.providerId &&
entry.modelId === selection.modelId
);
}
/**
* In-memory harness session handle used by the fake adapter and by the shared
* conformance suite. It enforces the two invariants the real adapters must also
* honor: model selection is validated against the catalog and is NEVER
* substituted, and unsupported capabilities fail with a typed error.
*/
export class FakeHarnessSessionHandle implements HarnessSessionHandle {
private state: HarnessSessionState = 'idle';
private processId: string | undefined;
private readonly attachedClientIds = new Set<string>();
private readonly listeners = new Set<(event: HarnessEvent) => void>();
constructor(
private readonly conversationId: string,
private readonly nativeSessionId: string,
private readonly seatId: string,
private selection: HarnessSelection,
private readonly correlationId: string,
private readonly capabilities: readonly HarnessCapability[],
private readonly catalog: readonly HarnessCatalogEntry[],
) {
this.processId = `process-${nativeSessionId}`;
}
async snapshot(): Promise<HarnessSessionSnapshot> {
return {
conversationId: this.conversationId,
nativeSessionId: this.nativeSessionId,
processId: this.processId,
seatId: this.seatId,
selection: this.selection,
state: this.state,
attachedClientIds: [...this.attachedClientIds],
};
}
async attach(input: AttachClient): Promise<void> {
this.attachedClientIds.add(input.clientId);
}
async detach(clientId: string): Promise<void> {
// Removes the browser attachment only; the process and native session persist.
this.attachedClientIds.delete(clientId);
}
async prompt(input: HarnessPrompt & { idempotencyKey: string }): Promise<HarnessPromptReceipt> {
return {
conversationId: this.conversationId,
turnId: input.turnId,
correlationId: input.correlationId,
state: 'accepted',
selection: this.selection,
};
}
async setModel(selection: HarnessSelection): Promise<HarnessSelection> {
const entry = this.catalog.find((candidate) => matches(candidate, selection));
if (!entry) {
// No fallback to the first catalog row: reject with the requested tuple, unchanged.
throw operationError(
'selection_invalid',
'The requested harness/provider/model tuple is not in the catalog.',
selection,
this.correlationId,
);
}
if (entry.availability === 'unavailable') {
throw operationError(
'model_unavailable',
'The requested model is currently unavailable.',
selection,
this.correlationId,
true,
);
}
this.selection = selection;
return this.selection;
}
async abort(_turnId: string): Promise<void> {
// No active turn machinery in the fake; abort is a no-op acknowledgement.
}
async respondInteraction(_input: HarnessInteractionResponse): Promise<void> {
if (!this.capabilities.includes('extensionUi')) {
throw operationError(
'interaction_unsupported',
'This harness does not support interactive responses.',
this.selection,
this.correlationId,
);
}
}
events(listener: (event: HarnessEvent) => void): () => void {
this.listeners.add(listener);
return () => {
this.listeners.delete(listener);
};
}
async evictProcess(_reason: HarnessCloseReason): Promise<void> {
// Stop the process but keep the resumable native session.
this.processId = undefined;
this.state = 'evicted';
}
async endSession(_reason: HarnessCloseReason): Promise<void> {
// Destructively end the native session.
this.processId = undefined;
this.state = 'ended';
}
}
/**
* Minimal in-memory {@link HarnessAdapter} for Slice Zero. It mints a fresh
* native session id on `create` and binds the supplied one on `resume`, so the
* two paths are observably distinct.
*/
export class FakeHarnessAdapter implements HarnessAdapter {
readonly id: string;
private readonly capabilities: readonly HarnessCapability[];
private readonly catalogEntries: readonly HarnessCatalogEntry[];
private createdCount = 0;
constructor(options: FakeHarnessAdapterOptions) {
this.id = options.id;
this.capabilities = options.capabilities ?? [...HARNESS_CAPABILITIES];
this.catalogEntries = options.catalog ?? defaultCatalog(options.id);
}
async describe(_context: HarnessActorContext): Promise<HarnessDescriptor> {
return {
id: this.id,
displayName: `Fake harness (${this.id})`,
capabilities: this.capabilities,
};
}
async catalog(_context: HarnessActorContext): Promise<HarnessCatalog> {
return {
harnessId: this.id,
version: '1.0.0',
fingerprint: `fake-${this.id}-${this.catalogEntries.length}`,
models: this.catalogEntries,
};
}
async create(input: CreateHarnessSession): Promise<HarnessSessionHandle> {
this.createdCount += 1;
const nativeSessionId = `native-${input.conversationId}-${this.createdCount}`;
return new FakeHarnessSessionHandle(
input.conversationId,
nativeSessionId,
input.context.seatId,
input.selection,
input.context.correlationId,
this.capabilities,
this.catalogEntries,
);
}
async resume(input: ResumeHarnessSession): Promise<HarnessSessionHandle> {
return new FakeHarnessSessionHandle(
input.conversationId,
input.nativeSessionId,
input.context.seatId,
input.selection,
input.context.correlationId,
this.capabilities,
this.catalogEntries,
);
}
}
@@ -1,157 +0,0 @@
import { describe, expect, it } from 'vitest';
import type {
HarnessActorContext,
HarnessAdapter,
HarnessCatalogEntry,
HarnessSelection,
} from '@mosaicstack/types';
import { HarnessOperationError } from '../harness.registry.js';
const CONTEXT: HarnessActorContext = {
actorId: 'contract-actor',
tenantId: 'contract-tenant',
seatId: 'contract-seat',
correlationId: 'contract-correlation',
};
function toSelection(entry: HarnessCatalogEntry): HarnessSelection {
return { harnessId: entry.harnessId, providerId: entry.providerId, modelId: entry.modelId };
}
function pickAvailable(models: readonly HarnessCatalogEntry[]): HarnessCatalogEntry {
const entry = models.find((candidate) => candidate.availability === 'available') ?? models[0];
if (!entry) {
throw new Error('contract fixture requires at least one catalog model');
}
return entry;
}
async function captureError(run: () => Promise<unknown>): Promise<unknown> {
try {
await run();
return undefined;
} catch (caught) {
return caught;
}
}
/**
* Shared conformance suite every {@link HarnessAdapter} must pass. Slice Zero
* runs it against the fake adapter; Task 13 re-runs the identical suite against
* the native Pi adapter so both share one behavioral contract.
*/
export function runHarnessAdapterContract(
label: string,
createAdapter: () => HarnessAdapter,
): void {
describe(`harness adapter contract: ${label}`, () => {
it('mints a fresh native session on create and binds the supplied one on resume', async () => {
const adapter = createAdapter();
const catalog = await adapter.catalog(CONTEXT);
const selection = toSelection(pickAvailable(catalog.models));
const created = await (
await adapter.create({ context: CONTEXT, conversationId: 'conv-create', selection })
).snapshot();
const resumed = await (
await adapter.resume({
context: CONTEXT,
conversationId: 'conv-resume',
nativeSessionId: 'native-supplied-1',
selection,
})
).snapshot();
expect(created.nativeSessionId).toBeTruthy();
expect(resumed.nativeSessionId).toBe('native-supplied-1');
expect(created.nativeSessionId).not.toBe(resumed.nativeSessionId);
expect(created.seatId).toBe(CONTEXT.seatId);
});
it('gives detach, evict, and end distinct effects (not aliases)', async () => {
const adapter = createAdapter();
const catalog = await adapter.catalog(CONTEXT);
const selection = toSelection(pickAvailable(catalog.models));
const handle = await adapter.create({
context: CONTEXT,
conversationId: 'conv-lifecycle',
selection,
});
await handle.attach({ clientId: 'browser-1' });
await handle.detach('browser-1');
const afterDetach = await handle.snapshot();
expect(afterDetach.attachedClientIds).toEqual([]);
expect(afterDetach.state).not.toBe('evicted');
expect(afterDetach.state).not.toBe('ended');
await handle.evictProcess('idle_timeout');
const afterEvict = await handle.snapshot();
expect(afterEvict.state).toBe('evicted');
// The native session survives eviction (resumable); the process does not.
expect(afterEvict.nativeSessionId).toBe(afterDetach.nativeSessionId);
expect(afterEvict.processId).toBeUndefined();
await handle.endSession('session_ended');
const afterEnd = await handle.snapshot();
expect(afterEnd.state).toBe('ended');
// End is not an alias of evict.
expect(afterEnd.state).not.toBe(afterEvict.state);
});
it('never substitutes the first catalog row for an unknown selection', async () => {
const adapter = createAdapter();
const catalog = await adapter.catalog(CONTEXT);
const firstRow = catalog.models[0];
if (!firstRow) {
throw new Error('contract fixture requires a catalog model');
}
const start = toSelection(pickAvailable(catalog.models));
const handle = await adapter.create({
context: CONTEXT,
conversationId: 'conv-nosub',
selection: start,
});
const bogus: HarnessSelection = {
harnessId: adapter.id,
providerId: 'contract-ghost-provider',
modelId: 'contract-ghost-model',
};
const error = await captureError(() => handle.setModel(bogus));
expect(error).toBeInstanceOf(HarnessOperationError);
const dto = (error as HarnessOperationError).dto;
expect(dto.code).toBe('selection_invalid');
expect(dto.selection).toEqual(bogus);
expect(dto.selection).not.toEqual(toSelection(firstRow));
expect((await handle.snapshot()).selection).toEqual(start);
});
it('validates capability-gated interactions with a typed error, not a silent no-op', async () => {
const adapter = createAdapter();
const descriptor = await adapter.describe(CONTEXT);
const catalog = await adapter.catalog(CONTEXT);
const selection = toSelection(pickAvailable(catalog.models));
const handle = await adapter.create({
context: CONTEXT,
conversationId: 'conv-interaction',
selection,
});
const response = {
requestId: 'interaction-1',
type: 'confirm',
accepted: true,
} as const;
if (descriptor.capabilities.includes('extensionUi')) {
await expect(handle.respondInteraction(response)).resolves.toBeUndefined();
} else {
const error = await captureError(() => handle.respondInteraction(response));
expect(error).toBeInstanceOf(HarnessOperationError);
expect((error as HarnessOperationError).dto.code).toBe('interaction_unsupported');
}
});
});
}
@@ -1,44 +0,0 @@
import { Logger } from '@nestjs/common';
import { Client } from '@modelcontextprotocol/sdk/client/index.js';
import { afterEach, beforeEach, describe, expect, it, vi } from 'vitest';
import { McpClientService } from './mcp-client.service.js';
const MCP_LEAK_MARKER = 'MCP_LEAK_MARKER /srv/secret';
describe('McpClientService — failed connect error sanitization', () => {
const originalMcpServers = process.env['MCP_SERVERS'];
beforeEach(() => {
process.env['MCP_SERVERS'] = JSON.stringify([
{ name: 'leaky-server', url: 'http://localhost:9999/mcp' },
]);
});
afterEach(() => {
vi.restoreAllMocks();
if (originalMcpServers === undefined) {
delete process.env['MCP_SERVERS'];
} else {
process.env['MCP_SERVERS'] = originalMcpServers;
}
});
it('stores a generic serverEntry.error while logging the raw exception server-side', async () => {
vi.spyOn(Client.prototype, 'connect').mockRejectedValue(new Error(MCP_LEAK_MARKER));
const errorSpy = vi.spyOn(Logger.prototype, 'error').mockImplementation(() => undefined);
const service = new McpClientService();
await service.onModuleInit();
const statuses = service.getServerStatuses();
expect(statuses).toHaveLength(1);
expect(statuses[0]?.connected).toBe(false);
expect(statuses[0]?.error).toBe('Connection failed (see server logs).');
expect(statuses[0]?.error).not.toContain(MCP_LEAK_MARKER);
const loggedRawMarker = errorSpy.mock.calls.some((call) =>
call.some((arg) => typeof arg === 'string' && arg.includes(MCP_LEAK_MARKER)),
);
expect(loggedRawMarker).toBe(true);
});
});
@@ -189,7 +189,7 @@ export class McpClientService implements OnModuleInit, OnModuleDestroy {
);
} catch (err) {
const message = err instanceof Error ? err.message : String(err);
serverEntry.error = 'Connection failed (see server logs).';
serverEntry.error = message;
serverEntry.connected = false;
this.logger.error(`Failed to connect to MCP server "${config.name}": ${message}`);
}
@@ -1,8 +1,5 @@
import { Logger } from '@nestjs/common';
import { describe, expect, it, vi } from 'vitest';
import type { SlashCommandPayload, SystemReloadPayload } from '@mosaicstack/types';
import { ReloadService } from './reload.service.js';
import { CommandExecutorService } from '../commands/command-executor.service.js';
function createMockCommandRegistry() {
return {
@@ -107,85 +104,3 @@ describe('ReloadService', () => {
expect(() => service.registerPlugin('my-plugin', {})).not.toThrow();
});
});
describe('ReloadService — /reload command sanitizes plugin errors', () => {
it('generic per-plugin errors reach the chat surface while raw markers stay server-side only', async () => {
const registry = {
getManifest: vi.fn().mockReturnValue({
version: 1,
commands: [
{ name: 'reload', aliases: [], scope: 'core', execution: 'socket', available: true },
],
skills: [],
}),
};
const reloadService = new ReloadService(registry as never);
const RELOAD_LOAD_LEAK_MARKER = 'RELOAD_LOAD_LEAK_MARKER /srv/load-secret';
const RELOAD_UNLOAD_LEAK_MARKER = 'RELOAD_UNLOAD_LEAK_MARKER /srv/unload-secret';
reloadService.registerPlugin('unload-fails', {
pluginName: 'unload-fails',
onLoad: vi.fn().mockResolvedValue(undefined),
onUnload: vi.fn().mockRejectedValue(new Error(RELOAD_UNLOAD_LEAK_MARKER)),
});
reloadService.registerPlugin('load-fails', {
pluginName: 'load-fails',
onLoad: vi.fn().mockRejectedValue(new Error(RELOAD_LOAD_LEAK_MARKER)),
onUnload: vi.fn().mockResolvedValue(undefined),
});
const errorSpy = vi.spyOn(Logger.prototype, 'error').mockImplementation(() => undefined);
const broadcastReload = vi.fn();
const mockChatGateway = { broadcastReload };
const mockAgentService = { getSession: vi.fn(), applyAgentConfig: vi.fn() };
const mockSystemOverride = { set: vi.fn(), get: vi.fn(), clear: vi.fn() };
const mockSessionGC = { sweepOrphans: vi.fn() };
const mockBrain = { agents: { findByName: vi.fn(), findById: vi.fn(), create: vi.fn() } };
const mockMcpClient = {
getServerStatuses: vi.fn(() => []),
getToolDefinitions: vi.fn(() => []),
reconnectServer: vi.fn().mockResolvedValue(undefined),
};
const executor = new CommandExecutorService(
registry as never,
mockAgentService as never,
mockSystemOverride as never,
mockSessionGC as never,
null,
mockBrain as never,
reloadService,
mockChatGateway as never,
mockMcpClient as never,
);
const payload: SlashCommandPayload = { command: 'reload', conversationId: 'conv-1' };
const result = await executor.execute(payload, { userId: 'user-1', tenantId: 'user-1' });
expect(result.success).toBe(true);
expect(result.message).toContain('unload-fails: unload failed (internal error)');
expect(result.message).toContain('load-fails: load failed (internal error)');
expect(result.message).not.toContain(RELOAD_UNLOAD_LEAK_MARKER);
expect(result.message).not.toContain(RELOAD_LOAD_LEAK_MARKER);
expect(broadcastReload).toHaveBeenCalledOnce();
const broadcastPayload = broadcastReload.mock.calls[0]?.[0] as SystemReloadPayload;
expect(broadcastPayload.message).toContain('unload-fails: unload failed (internal error)');
expect(broadcastPayload.message).toContain('load-fails: load failed (internal error)');
expect(broadcastPayload.message).not.toContain(RELOAD_UNLOAD_LEAK_MARKER);
expect(broadcastPayload.message).not.toContain(RELOAD_LOAD_LEAK_MARKER);
const loggedUnloadMarker = errorSpy.mock.calls.some((call) =>
call.some((arg) => typeof arg === 'string' && arg.includes(RELOAD_UNLOAD_LEAK_MARKER)),
);
const loggedLoadMarker = errorSpy.mock.calls.some((call) =>
call.some((arg) => typeof arg === 'string' && arg.includes(RELOAD_LOAD_LEAK_MARKER)),
);
expect(loggedUnloadMarker).toBe(true);
expect(loggedLoadMarker).toBe(true);
errorSpy.mockRestore();
});
});
+2 -4
View File
@@ -58,8 +58,7 @@ export class ReloadService implements OnApplicationBootstrap, OnApplicationShutd
await plugin.onUnload();
reloaded.push(name);
} catch (err) {
this.logger.error(`Plugin "${name}" failed during onUnload: ${err}`);
errors.push(`${name}: unload failed (internal error)`);
errors.push(`${name}: unload failed — ${err}`);
}
}
}
@@ -70,8 +69,7 @@ export class ReloadService implements OnApplicationBootstrap, OnApplicationShutd
try {
await plugin.onLoad();
} catch (err) {
this.logger.error(`Plugin "${name}" failed during onLoad: ${err}`);
errors.push(`${name}: load failed (internal error)`);
errors.push(`${name}: load failed ${err}`);
}
}
}
+2 -3
View File
@@ -5,17 +5,16 @@
"scripts": {
"build": "node ../../scripts/build-web.mjs",
"build:vite": "vite build",
"dev": "next dev -p 3101",
"dev": "next dev",
"dev:vite": "vite",
"lint": "eslint src",
"typecheck": "tsc --noEmit",
"test": "vitest run --passWithNoTests",
"test:e2e": "playwright test",
"start": "next start -p 3101"
"start": "next start"
},
"dependencies": {
"@mosaicstack/design-tokens": "workspace:^",
"@mosaicstack/types": "workspace:^",
"better-auth": "^1.5.5",
"clsx": "^2.1.0",
"next": "^16.0.0",
-61
View File
@@ -1,61 +0,0 @@
// Centralizes the type-only import of the shared `/chat` Socket.IO contract from
// the public `@mosaicstack/types` package. `import type` is erased at compile
// time, so this introduces no runtime dependency — it only reuses the exact
// payload shapes instead of redeclaring them.
import type { Socket } from 'socket.io-client';
import type {
AbortPayload,
AgentEndPayload,
AgentStartPayload,
AgentTextPayload,
AgentThinkingPayload,
ChatMessagePayload,
ClientToServerEvents,
CommandDef,
CommandManifest,
CommandManifestPayload,
ErrorPayload,
MessageAckPayload,
RoutingDecisionInfo,
ServerToClientEvents,
SessionInfoPayload,
SessionUsagePayload,
SetThinkingPayload,
SkillCommandDef,
SlashCommandApprovalResultPayload,
SlashCommandPayload,
SlashCommandResultPayload,
SystemReloadPayload,
ToolEndPayload,
ToolStartPayload,
} from '@mosaicstack/types';
export type {
AbortPayload,
AgentEndPayload,
AgentStartPayload,
AgentTextPayload,
AgentThinkingPayload,
ChatMessagePayload,
ClientToServerEvents,
CommandDef,
CommandManifest,
CommandManifestPayload,
ErrorPayload,
MessageAckPayload,
RoutingDecisionInfo,
ServerToClientEvents,
SessionInfoPayload,
SessionUsagePayload,
SetThinkingPayload,
SkillCommandDef,
SlashCommandApprovalResultPayload,
SlashCommandPayload,
SlashCommandResultPayload,
SystemReloadPayload,
ToolEndPayload,
ToolStartPayload,
};
/** The `/chat` namespace socket, narrowed to the exact typed event contract. */
export type ChatSocket = Socket<ServerToClientEvents, ClientToServerEvents>;
+17 -63
View File
@@ -10,52 +10,30 @@ vi.mock('socket.io-client', () => ({
import { destroySocket, getSocket } from './socket';
interface MockChatSocket {
on: ReturnType<typeof vi.fn>;
offAny: ReturnType<typeof vi.fn>;
disconnect: ReturnType<typeof vi.fn>;
/** Test-only helper: fires every handler registered for `event` via
* `.on`, mirroring how a real socket.io-client instance invokes its own
* listeners (e.g. calling the registered `disconnect` handler(s) on a
* real transient disconnect). */
trigger(event: string): void;
}
function createMockSocket(): MockChatSocket {
const handlers = new Map<string, Set<() => void>>();
const mockSocket: MockChatSocket = {
on: vi.fn((event: string, handler: () => void) => {
if (!handlers.has(event)) handlers.set(event, new Set());
handlers.get(event)?.add(handler);
return mockSocket;
}),
offAny: vi.fn(() => mockSocket),
disconnect: vi.fn(() => mockSocket),
trigger(event: string): void {
for (const handler of handlers.get(event) ?? []) handler();
},
};
return mockSocket;
}
let currentMock!: MockChatSocket;
describe('chat socket', () => {
let disconnectHandler: (() => void) | undefined;
beforeEach(() => {
disconnectHandler = undefined;
ioMock.mockReset();
// A fresh object per io() call so identity assertions (same singleton vs.
// a genuinely new instance) are meaningful.
ioMock.mockImplementation(() => {
currentMock = createMockSocket();
return currentMock;
});
const mockSocket = {
on: vi.fn((event: string, handler: () => void) => {
if (event === 'disconnect') disconnectHandler = handler;
return mockSocket;
}),
offAny: vi.fn(() => mockSocket),
disconnect: vi.fn(() => mockSocket),
};
ioMock.mockReturnValue(mockSocket);
});
afterEach(() => {
destroySocket();
});
it('creates one same-origin /chat namespace socket', () => {
it('creates one same-origin /chat namespace socket until it disconnects', () => {
const first = getSocket();
const second = getSocket();
@@ -66,33 +44,9 @@ describe('chat socket', () => {
autoConnect: false,
transports: ['websocket', 'polling'],
});
});
it('keeps the same singleton instance across a transient disconnect', () => {
const first = getSocket();
// socket.ts must not react to a real socket's `disconnect` event by
// nulling the singleton — it registers no such handler at all now.
// Actually fire every handler registered via `.on('disconnect', ...)`
// (mirroring a real socket.io-client reconnect) instead of merely
// calling getSocket() again: this is what makes the test fail if
// production reintroduces `socket.on('disconnect', () => { socket =
// null; })`, since that handler would run here and null the singleton
// before the next getSocket() call.
currentMock.trigger('disconnect');
const second = getSocket();
expect(second).toBe(first);
expect(ioMock).toHaveBeenCalledOnce();
});
it('only creates a new singleton after an explicit destroySocket()', () => {
const first = getSocket();
destroySocket();
const second = getSocket();
expect(second).not.toBe(first);
disconnectHandler?.();
getSocket();
expect(ioMock).toHaveBeenCalledTimes(2);
});
});
+10 -16
View File
@@ -1,27 +1,21 @@
import { io } from 'socket.io-client';
import type { ChatSocket } from './chat-contract';
import { io, type Socket } from 'socket.io-client';
let socket: ChatSocket | null = null;
let socket: Socket | null = null;
export function getSocket(): ChatSocket {
export function getSocket(): Socket {
if (!socket) {
// socket.io-client 4.8.3's `io()` factory declaration always returns the
// default unparameterized Socket (it accepts no <ListenEvents, EmitEvents>
// generics), so this one cast is the unavoidable boundary between that and the
// typed `/chat` contract. Every other call site uses the resulting ChatSocket
// with no further assertions.
socket = io('/chat', {
withCredentials: true,
autoConnect: false,
transports: ['websocket', 'polling'],
}) as unknown as ChatSocket;
});
// A transient `disconnect` (network blip, server restart) must NOT null
// the singleton: socket.io-client auto-reconnects this same instance,
// and its listeners stay registered across that reconnect. Nulling here
// previously orphaned those listeners on the next getSocket() call by
// handing back a brand-new, unconnected instance. Only destroySocket()
// (an explicit, intentional teardown) may reset the singleton.
// Reset singleton reference when socket is fully closed so the next
// getSocket() call creates a fresh instance instead of returning a
// closed/dead socket.
socket.on('disconnect', () => {
socket = null;
});
}
return socket;
}
-39
View File
@@ -1,42 +1,3 @@
import type {
HarnessAuthState,
HarnessModelAvailability,
HarnessSelection,
} from '@mosaicstack/types';
// The exact harness/provider/model tuple and its closed enum companions are the
// shared domain types — re-exported here so web consumers (and the runtime
// guards) import one shape, never a divergent local redefinition.
export type { HarnessSelection, HarnessAuthState, HarnessModelAvailability };
/** Harness summary row from `GET /api/harnesses` (the `HarnessSummaryDto`). The
* harness id is kept distinct from any provider id — they are never merged. */
export interface HarnessSummary {
id: string;
displayName: string;
capabilities: string[];
}
/** One selectable model in a harness catalog. Extends the `{harnessId,
* providerId, modelId}` tuple with the display/availability metadata the UI
* needs; `inputTypes` is kept as a plain `string[]` on the client boundary
* because it arrives from untrusted JSON and is only ever displayed. */
export interface HarnessCatalogEntry extends HarnessSelection {
displayName: string;
reasoningCapability: boolean;
inputTypes: string[];
authState: HarnessAuthState;
availability: HarnessModelAvailability;
}
/** Harness-scoped catalog from `GET /api/harnesses/:harnessId/catalog`. */
export interface HarnessCatalog {
harnessId: string;
version: string;
fingerprint: string;
models: HarnessCatalogEntry[];
}
/** Conversation returned by the gateway API. */
export interface Conversation {
id: string;
+4 -22
View File
@@ -3,16 +3,6 @@ import { createBrowserRouter, Navigate, Outlet, type RouteObject } from 'react-r
import { LoginPage } from '@/spa/pages/login';
import { RegisterPage } from '@/spa/pages/register';
import { SsoCallbackPage } from '@/spa/pages/sso-callback';
import { ChatPage } from '@/spa/pages/chat';
import { ChatRouteErrorBoundary } from '@/spa/pages/chat-error-boundary';
import { ProjectDetailPage } from '@/spa/pages/project-detail';
import { ProjectsPage } from '@/spa/pages/projects';
import {
ProjectDetailRouteErrorBoundary,
ProjectsRouteErrorBoundary,
TasksRouteErrorBoundary,
} from '@/spa/pages/resource-route-error-boundaries';
import { TasksPage } from '@/spa/pages/tasks';
import { AuthGuard, GuestGuard } from '@/spa/guards';
import { Placeholder } from '@/spa/placeholder';
@@ -44,18 +34,10 @@ export const routes: RouteObject[] = [
element: <AuthGuard />,
children: [
{ path: '/', element: <Navigate to="/chat" replace /> },
{ path: '/chat', element: <ChatPage />, errorElement: <ChatRouteErrorBoundary /> },
{
path: '/projects',
element: <ProjectsPage />,
errorElement: <ProjectsRouteErrorBoundary />,
},
{
path: '/projects/:id',
element: <ProjectDetailPage />,
errorElement: <ProjectDetailRouteErrorBoundary />,
},
{ path: '/tasks', element: <TasksPage />, errorElement: <TasksRouteErrorBoundary /> },
{ path: '/chat', element: <Placeholder title="Chat" /> },
{ path: '/projects', element: <Placeholder title="Projects" /> },
{ path: '/projects/:id', element: <Placeholder title="Project" /> },
{ path: '/tasks', element: <Placeholder title="Tasks" /> },
{ path: '/settings', element: <Placeholder title="Settings" /> },
{ path: '/admin', element: <Placeholder title="Admin" /> },
],
-195
View File
@@ -1,195 +0,0 @@
import { afterEach, describe, expect, it, vi } from 'vitest';
import {
fetchCatalog,
fetchHarnesses,
fetchPersistedSelection,
persistSelection,
} from './chat-api';
function json(body: unknown, status = 200): Response {
return new Response(JSON.stringify(body), {
status,
headers: { 'Content-Type': 'application/json' },
});
}
function stubFetch(): ReturnType<typeof vi.fn> {
const fetchMock = vi.fn();
vi.stubGlobal('fetch', fetchMock);
return fetchMock;
}
/** Every URL the client actually requested, across all calls. */
function requestedUrls(fetchMock: ReturnType<typeof vi.fn>): string[] {
return fetchMock.mock.calls.map((call) => String(call[0]));
}
describe('chat-api', () => {
afterEach(() => {
vi.unstubAllGlobals();
});
it('fetchHarnesses GETs /api/harnesses and returns typed summaries (harness id separate from provider)', async () => {
const fetchMock = stubFetch();
fetchMock.mockResolvedValue(
json([
{ id: 'pi', displayName: 'Pi', capabilities: ['chat', 'tools'] },
{ id: 'openai', displayName: 'OpenAI', capabilities: ['chat'] },
]),
);
const harnesses = await fetchHarnesses();
expect(fetchMock).toHaveBeenCalledOnce();
expect(String(fetchMock.mock.calls[0]?.[0])).toBe('/api/harnesses');
expect(harnesses).toEqual([
{ id: 'pi', displayName: 'Pi', capabilities: ['chat', 'tools'] },
{ id: 'openai', displayName: 'OpenAI', capabilities: ['chat'] },
]);
});
it('fetchCatalog GETs the harness-scoped catalog and returns only its model entries', async () => {
const fetchMock = stubFetch();
fetchMock.mockResolvedValue(
json({
harnessId: 'pi',
version: '2026-08-11',
fingerprint: 'abc123',
models: [
{
harnessId: 'pi',
providerId: 'openai',
modelId: 'gpt-5',
displayName: 'GPT-5',
reasoningCapability: true,
inputTypes: ['text'],
authState: 'ready',
availability: 'available',
},
],
}),
);
const result = await fetchCatalog('pi');
expect(String(fetchMock.mock.calls[0]?.[0])).toBe('/api/harnesses/pi/catalog');
expect(result.ok).toBe(true);
if (!result.ok) throw new Error('expected ok catalog');
expect(result.catalog.harnessId).toBe('pi');
expect(result.catalog.models).toHaveLength(1);
expect(result.catalog.models[0]).toMatchObject({
harnessId: 'pi',
providerId: 'openai',
modelId: 'gpt-5',
availability: 'available',
});
});
it('normalizes a catalog 404 into a typed catalog_unavailable result without surfacing the raw body', async () => {
const fetchMock = stubFetch();
fetchMock.mockResolvedValue(
json(
{
code: 'adapter_unavailable',
message: 'raw gateway detail that must not leak verbatim',
harnessId: 'attacker-echo',
extra: { hostile: 'blob' },
},
404,
),
);
const result = await fetchCatalog('ghost');
expect(result.ok).toBe(false);
if (result.ok) throw new Error('expected unavailable result');
expect(result.code).toBe('catalog_unavailable');
// harnessId comes from the request, never the (untrusted) response body.
expect(result.harnessId).toBe('ghost');
expect(typeof result.message).toBe('string');
// The raw response body is never rendered/returned verbatim.
expect(JSON.stringify(result)).not.toContain('hostile');
expect(JSON.stringify(result)).not.toContain('attacker-echo');
});
it('fetchPersistedSelection returns the stored tuple, or null when unset', async () => {
const fetchMock = stubFetch();
fetchMock.mockResolvedValueOnce(
json({ selection: { harnessId: 'pi', providerId: 'openai', modelId: 'gpt-5' } }),
);
await expect(fetchPersistedSelection()).resolves.toEqual({
harnessId: 'pi',
providerId: 'openai',
modelId: 'gpt-5',
});
expect(String(fetchMock.mock.calls[0]?.[0])).toBe('/api/chat/preferences/selection');
fetchMock.mockResolvedValueOnce(json({ selection: null }));
await expect(fetchPersistedSelection()).resolves.toBeNull();
});
it('persistSelection PUTs the structured tuple (not free text) and returns the confirmed selection', async () => {
const fetchMock = stubFetch();
fetchMock.mockResolvedValue(
json({ selection: { harnessId: 'pi', providerId: 'openai', modelId: 'gpt-5' } }),
);
const result = await persistSelection({
harnessId: 'pi',
providerId: 'openai',
modelId: 'gpt-5',
});
expect(result.ok).toBe(true);
const call = fetchMock.mock.calls[0];
expect(String(call?.[0])).toBe('/api/chat/preferences/selection');
const init = call?.[1] as RequestInit;
expect(String(init.method).toUpperCase()).toBe('PUT');
// The body is exactly the structured tuple — harness/provider/model kept distinct.
expect(JSON.parse(String(init.body))).toEqual({
harnessId: 'pi',
providerId: 'openai',
modelId: 'gpt-5',
});
});
it('normalizes a selection 422 into a typed error preserving the requested tuple exactly', async () => {
const fetchMock = stubFetch();
fetchMock.mockResolvedValue(
json(
{
code: 'model_unavailable',
message: 'raw detail that must not leak',
selection: { harnessId: 'x', providerId: 'y', modelId: 'z' },
},
422,
),
);
const requested = { harnessId: 'pi', providerId: 'openai', modelId: 'gpt-5' };
const result = await persistSelection(requested);
expect(result.ok).toBe(false);
if (result.ok) throw new Error('expected failed persist');
expect(['selection_invalid', 'model_unavailable']).toContain(result.code);
// The requested tuple is preserved unchanged — not replaced by the body's echo.
expect(result.requested).toEqual(requested);
expect(JSON.stringify(result)).not.toContain('raw detail');
});
it('never requests any /api/providers* endpoint', async () => {
const fetchMock = stubFetch();
fetchMock.mockResolvedValue(json([]));
await fetchHarnesses();
fetchMock.mockResolvedValue(
json({ harnessId: 'pi', version: '1', fingerprint: 'f', models: [] }),
);
await fetchCatalog('pi');
fetchMock.mockResolvedValue(json({ selection: null }));
await fetchPersistedSelection();
for (const url of requestedUrls(fetchMock)) {
expect(url).not.toContain('/api/providers');
}
});
});
-132
View File
@@ -1,132 +0,0 @@
/**
* Typed fetch wrappers for the Task-3 harness HTTP contract the chat selection
* UI depends on. Every response body is untrusted and is normalized through the
* runtime guards before it reaches state — a 404 (catalog) and a 422 (selection)
* are mapped to typed, body-free error results so a raw gateway body is never
* rendered, and the caller's requested tuple is preserved verbatim on failure.
*
* This module talks ONLY to the harness/chat-preferences endpoints. It never
* calls `/api/providers*` — provider identity lives inside the harness catalog.
*/
import { asHarnessCatalog, asHarnessSelection, asHarnessSummaries } from './runtime-guards';
import type { HarnessCatalog, HarnessSelection, HarnessSummary } from '@/lib/types';
/** A catalog fetch either yields the typed catalog or a typed unavailability —
* never a thrown raw body. */
export type CatalogResult =
| { ok: true; catalog: HarnessCatalog }
| { ok: false; code: 'catalog_unavailable'; harnessId: string; message: string };
export type SelectionErrorCode = 'selection_invalid' | 'model_unavailable';
/** A persist either confirms the stored tuple or reports a typed domain failure
* that echoes back the exact tuple the caller requested. */
export type SelectionPersistResult =
| { ok: true; selection: HarnessSelection }
| { ok: false; code: SelectionErrorCode; message: string; requested: HarnessSelection };
/** A safe, generic message for an unavailable catalog — the raw 404 body is
* never surfaced. */
const CATALOG_UNAVAILABLE_MESSAGE = 'This harness catalog is currently unavailable.';
/** A safe, generic message for a rejected selection. The untrusted 422 body's
* own `message` is deliberately NEVER surfaced — only this fixed copy — so a
* raw gateway detail can never leak into the UI. Only the closed `code` enum is
* read from the body. */
const SELECTION_REJECTED_MESSAGE = 'This selection was rejected.';
async function readJson(response: Response): Promise<unknown> {
return response.json().catch(() => null);
}
function safeSelectionCode(body: unknown): SelectionErrorCode {
if (typeof body === 'object' && body !== null && 'code' in body) {
const code = (body as { code: unknown }).code;
if (code === 'selection_invalid' || code === 'model_unavailable') return code;
}
// Default to the more conservative "invalid" classification for anything
// unrecognized rather than guessing "model_unavailable".
return 'selection_invalid';
}
/** `GET /api/harnesses` → the list of harness summaries. A non-OK response
* normalizes to an empty list (the UI then has no harness to select). */
export async function fetchHarnesses(): Promise<HarnessSummary[]> {
const response = await fetch('/api/harnesses', {
credentials: 'include',
headers: { Accept: 'application/json' },
});
if (!response.ok) return [];
return asHarnessSummaries(await readJson(response));
}
/** `GET /api/harnesses/:harnessId/catalog` → the harness-scoped catalog. A 404
* (or any non-OK) becomes a typed `catalog_unavailable` result rather than a
* fallback catalog or a rendered raw body. */
export async function fetchCatalog(harnessId: string): Promise<CatalogResult> {
const response = await fetch(`/api/harnesses/${encodeURIComponent(harnessId)}/catalog`, {
credentials: 'include',
headers: { Accept: 'application/json' },
});
if (!response.ok) {
return {
ok: false,
code: 'catalog_unavailable',
// Scoped to the requested harness id, never the untrusted body's echo.
harnessId,
message: CATALOG_UNAVAILABLE_MESSAGE,
};
}
return { ok: true, catalog: asHarnessCatalog(await readJson(response), harnessId) };
}
/** `GET /api/chat/preferences/selection` → the persisted tuple, or null when
* unset or malformed. */
export async function fetchPersistedSelection(): Promise<HarnessSelection | null> {
const response = await fetch('/api/chat/preferences/selection', {
credentials: 'include',
headers: { Accept: 'application/json' },
});
if (!response.ok) return null;
const body = await readJson(response);
if (typeof body !== 'object' || body === null) return null;
return asHarnessSelection((body as { selection?: unknown }).selection);
}
/** `PUT /api/chat/preferences/selection` with the structured tuple as the body.
* On success returns the confirmed selection; on a typed domain failure (422)
* or validation error, returns a typed result carrying the EXACT requested
* tuple — never the body's echo — and never the raw body text. */
export async function persistSelection(
selection: HarnessSelection,
): Promise<SelectionPersistResult> {
const requested: HarnessSelection = {
harnessId: selection.harnessId,
providerId: selection.providerId,
modelId: selection.modelId,
};
const response = await fetch('/api/chat/preferences/selection', {
method: 'PUT',
credentials: 'include',
headers: { Accept: 'application/json', 'Content-Type': 'application/json' },
body: JSON.stringify(requested),
});
if (!response.ok) {
const body = await readJson(response);
return {
ok: false,
code: safeSelectionCode(body),
// Fixed copy only — the untrusted body's message is never surfaced.
message: SELECTION_REJECTED_MESSAGE,
requested,
};
}
const body = await readJson(response);
const confirmed =
typeof body === 'object' && body !== null
? asHarnessSelection((body as { selection?: unknown }).selection)
: null;
// A malformed 2xx body is treated as a confirmation of exactly what we sent —
// the server accepted the tuple, so the requested tuple is the source of truth.
return { ok: true, selection: confirmed ?? requested };
}
@@ -1,280 +0,0 @@
import { act } from 'react';
import { createRoot, type Root } from 'react-dom/client';
import { afterAll, afterEach, beforeAll, describe, expect, it, vi } from 'vitest';
import { CommandsPanel } from './commands-panel';
beforeAll(() => {
Object.defineProperty(globalThis, 'IS_REACT_ACT_ENVIRONMENT', {
configurable: true,
value: true,
});
});
afterAll(() => {
Reflect.deleteProperty(globalThis, 'IS_REACT_ACT_ENVIRONMENT');
});
let root: Root | null;
let container: HTMLElement | null;
async function render(node: Parameters<Root['render']>[0]): Promise<void> {
container = document.createElement('div');
document.body.append(container);
root = createRoot(container);
await act(async () => {
root?.render(node);
});
}
afterEach(async () => {
await act(async () => {
root?.unmount();
});
document.body.replaceChildren();
root = null;
container = null;
});
describe('CommandsPanel', () => {
it('shows the frozen local pendingApproval args in the confirmation area, regardless of misleading server message text', async () => {
await render(
<CommandsPanel
manifest={null}
results={[]}
approval={{
conversationId: 'c1',
command: 'deploy', // matches pendingApproval — this is a legitimately approved request
success: true,
approvalId: 'ap1',
expiresAt: '2026-01-01T00:00:00.000Z',
// Free-text server message claims a different, less alarming target
// than what will actually be sent — the UI must not rely on this.
message: 'This will only affect the staging environment.',
}}
pendingApproval={{ command: 'deploy', args: 'prod' }}
hasConversation
onExecute={vi.fn()}
onApprove={vi.fn()}
onRunApproved={vi.fn()}
/>,
);
// The exact frozen combined action is visible...
expect(container?.textContent).toContain('/deploy');
expect(container?.textContent).toContain('prod');
// ...and the misleading server free-text is never shown next to it.
expect(container?.textContent).not.toContain('staging environment');
});
it('does not throw when a manifest commands entry is null', async () => {
const manifest = {
commands: [
null,
{
name: 'model',
aliases: [],
description: 'Change the active model',
scope: 'core',
execution: 'socket',
available: true,
},
],
skills: [null],
version: 1,
} as unknown as Parameters<typeof CommandsPanel>[0]['manifest'];
await expect(
render(
<CommandsPanel
manifest={manifest}
results={[]}
approval={null}
pendingApproval={null}
hasConversation={false}
onExecute={vi.fn()}
onApprove={vi.fn()}
onRunApproved={vi.fn()}
/>,
),
).resolves.not.toThrow();
expect(container?.textContent).toContain('model');
});
it('shows an explicit no-args fallback when the frozen pendingApproval has no args', async () => {
await render(
<CommandsPanel
manifest={null}
results={[]}
approval={{
conversationId: 'c1',
command: 'deploy',
success: true,
approvalId: 'ap1',
expiresAt: '2026-01-01T00:00:00.000Z',
}}
pendingApproval={{ command: 'deploy' }}
hasConversation
onExecute={vi.fn()}
onApprove={vi.fn()}
onRunApproved={vi.fn()}
/>,
);
expect(container?.textContent?.toLowerCase()).toContain('no args');
});
it('renders skills from a skills-only manifest', async () => {
await render(
<CommandsPanel
manifest={{
commands: [],
skills: [{ name: 'brave-search', description: 'Search the web', available: true }],
version: 1,
}}
results={[]}
approval={null}
pendingApproval={null}
hasConversation={false}
onExecute={vi.fn()}
onApprove={vi.fn()}
onRunApproved={vi.fn()}
/>,
);
expect(container?.textContent).toContain('brave-search');
expect(container?.textContent).toContain('Search the web');
});
it('does not show the Run affordance when approval.success/approvalId are objects, even though command matches pendingApproval', async () => {
const approval = {
conversationId: 'c1',
command: 'deploy',
success: { truthy: 'object' },
approvalId: { also: 'object' },
} as unknown as Parameters<typeof CommandsPanel>[0]['approval'];
await render(
<CommandsPanel
manifest={null}
results={[]}
approval={approval}
pendingApproval={{ command: 'deploy', args: 'prod' }}
hasConversation
onExecute={vi.fn()}
onApprove={vi.fn()}
onRunApproved={vi.fn()}
/>,
);
expect(
[...(container?.querySelectorAll('button') ?? [])].some((button) =>
button.textContent?.includes('Run approved command'),
),
).toBe(false);
});
it('shows the guarded server-provided denial reason for a denied approval', async () => {
await render(
<CommandsPanel
manifest={null}
results={[]}
approval={{
conversationId: 'c1',
command: 'deploy',
success: false,
message: 'Not authorized',
}}
pendingApproval={{ command: 'deploy', args: 'prod' }}
hasConversation
onExecute={vi.fn()}
onApprove={vi.fn()}
onRunApproved={vi.fn()}
/>,
);
expect(container?.textContent).toContain('Not authorized');
});
it('falls back to a stable "Denied." copy when a denial has no usable message', async () => {
await render(
<CommandsPanel
manifest={null}
results={[]}
approval={{ conversationId: 'c1', command: 'deploy', success: false }}
pendingApproval={{ command: 'deploy', args: 'prod' }}
hasConversation
onExecute={vi.fn()}
onApprove={vi.fn()}
onRunApproved={vi.fn()}
/>,
);
expect(container?.textContent).toContain('Denied.');
});
it('shows the guarded contract-provided reason for a failed command result, falling back to a stable copy only when absent', async () => {
await render(
<CommandsPanel
manifest={null}
results={[
{ conversationId: 'c1', command: 'model', success: false, message: 'Unknown model' },
{ conversationId: 'c1', command: 'deploy', success: false },
]}
approval={null}
pendingApproval={null}
hasConversation={false}
onExecute={vi.fn()}
onApprove={vi.fn()}
onRunApproved={vi.fn()}
/>,
);
expect(container?.textContent).toContain('Unknown model');
expect(container?.textContent).toContain('Command failed.');
});
it('bounds an oversized command result message at the render site as defense-in-depth', async () => {
const hostileMessage = 'y'.repeat(50_000);
await render(
<CommandsPanel
manifest={null}
results={[
{ conversationId: 'c1', command: 'model', success: false, message: hostileMessage },
]}
approval={null}
pendingApproval={null}
hasConversation={false}
onExecute={vi.fn()}
onApprove={vi.fn()}
onRunApproved={vi.fn()}
/>,
);
const text = container?.textContent ?? '';
expect(text.length).toBeLessThan(hostileMessage.length);
});
it('does not throw when the manifest fields are malformed (non-array commands/skills)', async () => {
const manifest = {
commands: 'not-an-array',
skills: null,
version: 1,
} as unknown as Parameters<typeof CommandsPanel>[0]['manifest'];
await expect(
render(
<CommandsPanel
manifest={manifest}
results={[]}
approval={null}
pendingApproval={null}
hasConversation={false}
onExecute={vi.fn()}
onApprove={vi.fn()}
onRunApproved={vi.fn()}
/>,
),
).resolves.not.toThrow();
});
});
-164
View File
@@ -1,164 +0,0 @@
import { useState, type ReactElement } from 'react';
import type { PendingApproval } from './use-chat-connection';
import { MAX_COMMAND_MESSAGE_CHARS } from './limits';
import { asNonEmptyString, asString } from './runtime-guards';
import type {
CommandManifest,
SlashCommandApprovalResultPayload,
SlashCommandResultPayload,
} from '@/lib/chat-contract';
/** Stable fallback copy shown for a failed command only when the server's
* own guarded, non-empty `message` (e.g. "Unknown model") is absent or
* malformed — the structured contract reason itself is otherwise shown
* directly, never a raw thrown exception, stack trace, or object value. */
const COMMAND_FAILURE_COPY = 'Command failed.';
/** Render-site defense-in-depth: `use-chat-connection.ts` already bounds a
* stored command:result message at ingestion, but this component must never
* assume every caller went through that path — bounding again here means a
* hostile/oversized message can never force an unbounded render. */
function boundMessage(value: string): string {
return value.length > MAX_COMMAND_MESSAGE_CHARS
? value.slice(0, MAX_COMMAND_MESSAGE_CHARS)
: value;
}
interface CommandsPanelProps {
manifest: CommandManifest | null;
results: SlashCommandResultPayload[];
approval: SlashCommandApprovalResultPayload | null;
pendingApproval: PendingApproval | null;
hasConversation: boolean;
onExecute: (input: { command: string; args?: string }) => void;
onApprove: (input: { command: string; args?: string }) => void;
onRunApproved: () => void;
}
export function CommandsPanel({
manifest,
results,
approval,
pendingApproval,
hasConversation,
onExecute,
onApprove,
onRunApproved,
}: CommandsPanelProps): ReactElement {
const [command, setCommand] = useState('');
const [args, setArgs] = useState('');
// Defense-in-depth: the reducer already normalizes success/approvalId
// before storing `approval`, but a matching command string alone must
// never be trusted here either — require the literal boolean `true` and a
// non-empty string approvalId, not merely truthy values.
const canRunApproved =
approval?.success === true &&
typeof approval.approvalId === 'string' &&
approval.approvalId.length > 0 &&
!!pendingApproval &&
pendingApproval.command === approval.command;
// A manifest arrives from the server as untyped JSON at runtime — guard
// both collections before mapping so a malformed manifest cannot throw.
const commands = Array.isArray(manifest?.commands) ? manifest.commands : [];
const skills = Array.isArray(manifest?.skills) ? manifest.skills : [];
return (
<section aria-label="Commands" className="flex flex-col gap-2 border-b px-4 py-3 text-xs">
{commands.length > 0 ? (
<ul aria-label="Available commands" className="flex flex-col gap-1">
{commands.map((cmd, index) => (
<li key={asString(cmd?.name) || `cmd-${index}`}>
<strong>/{asString(cmd?.name)}</strong> {asString(cmd?.description)}
</li>
))}
</ul>
) : null}
{skills.length > 0 ? (
<ul aria-label="Available skills" className="flex flex-col gap-1">
{skills.map((skill, index) => (
<li key={asString(skill?.name) || `skill-${index}`}>
<strong>/skill:{asString(skill?.name)}</strong> {asString(skill?.description)}
</li>
))}
</ul>
) : null}
<div className="flex flex-wrap items-center gap-2">
<input
aria-label="Command name"
value={command}
onChange={(event) => setCommand(event.target.value)}
placeholder="command"
/>
<input
aria-label="Command arguments"
value={args}
onChange={(event) => setArgs(event.target.value)}
placeholder="args (optional)"
/>
<button
type="button"
disabled={!hasConversation || !command.trim()}
onClick={() => onExecute({ command: command.trim(), args: args.trim() || undefined })}
>
Run command
</button>
<button
type="button"
disabled={!hasConversation || !command.trim()}
onClick={() => onApprove({ command: command.trim(), args: args.trim() || undefined })}
>
Request approval
</button>
</div>
{approval ? (
<div role={approval.success ? 'status' : 'alert'} className="flex items-center gap-2">
{/* A successful approval shows stable client copy only — never
the server-controlled approval.message or echoed
approval.command as the primary confirmation. The frozen local
pendingApproval below (not this line) is the sole authoritative
statement of what will run. A denial, by contrast, is not an
execution authority and safely surfaces the guarded structured
reason the server gave (e.g. "Not authorized"), falling back to
a stable copy only when absent/malformed. */}
<span>
{approval.success ? 'Approved.' : asNonEmptyString(approval.message, 'Denied.')}
</span>
{canRunApproved && pendingApproval ? (
<>
{/* Authoritative frozen local command+args — what the click below
will actually emit. The server's `approval` above is display-only
and must never be trusted to represent the executed payload. */}
<span>
Will run: /{pendingApproval.command}{' '}
{pendingApproval.args ? pendingApproval.args : '(no args)'}
</span>
<button type="button" onClick={onRunApproved}>
Run approved command
</button>
</>
) : null}
</div>
) : null}
{results.length > 0 ? (
<ul aria-label="Command results" className="flex flex-col gap-1">
{results.map((result, index) => (
<li key={`${result.command}-${index}`} role={result.success ? 'status' : 'alert'}>
/{asString(result.command)}: {result.success ? 'success' : 'failed'}
{result.success
? typeof result.message === 'string' && result.message
? `${boundMessage(result.message)}`
: ''
: `${boundMessage(asNonEmptyString(result.message, COMMAND_FAILURE_COPY))}`}
</li>
))}
</ul>
) : null}
</section>
);
}
-174
View File
@@ -1,174 +0,0 @@
import { useState, type KeyboardEvent, type ReactElement } from 'react';
import type { HarnessSelectionValue } from './use-harness-selection';
interface ComposerProps {
onSend: (input: { content: string; provider?: string; modelId?: string }) => void;
onStop: () => void;
streaming: boolean;
/** True from local send time through server turn startup/ack and
* throughout streaming — a superset of `streaming` that also covers the
* pre-ack window where a second send could otherwise slip through. */
sending: boolean;
hasConversation: boolean;
/** Structured harness/provider/model selection state. The composer never
* accepts free-text provider/model — every sendable tuple is a validated,
* persisted catalog entry, and the send projection is derived from it. */
harness: HarnessSelectionValue;
}
/** The distinct provider ids present in the current catalog, in first-seen
* order — the provider select is catalog-derived, never a hardcoded list. */
function providerOptions(harness: HarnessSelectionValue): string[] {
const seen = new Set<string>();
const out: string[] = [];
for (const model of harness.catalog?.models ?? []) {
if (seen.has(model.providerId)) continue;
seen.add(model.providerId);
out.push(model.providerId);
}
return out;
}
export function Composer({
onSend,
onStop,
streaming,
sending,
hasConversation,
harness,
}: ComposerProps): ReactElement {
const [content, setContent] = useState('');
const busy = streaming || sending;
function submit(): void {
if (busy) return;
// Send is gated on a validated, persisted catalog tuple — a draft or unset
// selection can never emit, so provider/model never travel as free text.
if (!harness.canSend) return;
const trimmed = content.trim();
if (!trimmed) return;
onSend({ content: trimmed, ...harness.projection });
setContent('');
}
function handleKeyDown(event: KeyboardEvent<HTMLTextAreaElement>): void {
if (event.key === 'Enter' && !event.shiftKey) {
event.preventDefault();
submit();
}
}
// Scope the model options to the intentionally selected provider. With no
// provider chosen (`providerId === ''`) nothing matches, so the model select
// offers only the placeholder — never a cross-provider row.
const models = (harness.catalog?.models ?? []).filter(
(model) => model.providerId === harness.providerId,
);
// A collision-safe composite option identity covering the full provider+model
// tuple. The controlled select mirrors the same identity so the exact catalog
// row highlights (a bare modelId would collide across providers).
const modelOptionValue = (model: { providerId: string; modelId: string }): string =>
`${model.providerId}:${model.modelId}`;
const selectedModelValue = harness.modelId ? `${harness.providerId}:${harness.modelId}` : '';
return (
<form
onSubmit={(event) => {
event.preventDefault();
submit();
}}
className="flex flex-col gap-2 border-t p-4"
>
<div className="flex flex-wrap gap-2">
<select
aria-label="Harness"
value={harness.harnessId}
onChange={(event) => harness.selectHarness(event.target.value)}
className="rounded border px-2 py-1 text-xs"
>
<option value="">Select a harness</option>
{harness.harnesses.map((item) => (
<option key={item.id} value={item.id}>
{item.displayName}
</option>
))}
</select>
<select
aria-label="Provider"
value={harness.providerId}
onChange={(event) => harness.selectProvider(event.target.value)}
disabled={harness.catalogUnavailable || providerOptions(harness).length === 0}
className="rounded border px-2 py-1 text-xs"
>
<option value="">Select a provider</option>
{providerOptions(harness).map((providerId) => (
<option key={providerId} value={providerId}>
{providerId}
</option>
))}
</select>
<select
aria-label="Model"
value={selectedModelValue}
onChange={(event) => {
// Resolve the composite option identity back to the exact catalog
// row and persist that row's own provider+model — never a bare id.
const selected = models.find((model) => modelOptionValue(model) === event.target.value);
if (selected) harness.selectModel(selected.providerId, selected.modelId);
}}
disabled={harness.catalogUnavailable || models.length === 0}
className="rounded border px-2 py-1 text-xs"
>
<option value="">Select a model</option>
{models.map((model) => (
<option key={modelOptionValue(model)} value={modelOptionValue(model)}>
{model.displayName}
</option>
))}
</select>
</div>
{harness.catalogUnavailable ? (
<p role="status" className="text-xs opacity-70">
This harness catalog is currently unavailable.
</p>
) : null}
{harness.isStale ? (
<p role="status" className="text-xs opacity-70">
The saved model is no longer available pick another to continue.
</p>
) : null}
{harness.persistError ? (
<p role="alert" className="text-xs">
{harness.persistError.message}
</p>
) : null}
<div className="flex items-end gap-2">
<textarea
aria-label="Message"
value={content}
onChange={(event) => setContent(event.target.value)}
onKeyDown={handleKeyDown}
rows={2}
placeholder="Message… (Enter to send, Shift+Enter for a new line)"
className="flex-1 resize-none rounded border px-3 py-2 text-sm"
/>
<button
type="submit"
disabled={!content.trim() || busy || !harness.canSend}
className="rounded px-3 py-2 text-sm font-medium"
>
Send
</button>
<button
type="button"
aria-label="Stop"
disabled={!hasConversation || !streaming}
onClick={onStop}
className="rounded px-3 py-2 text-sm font-medium"
>
Stop
</button>
</div>
</form>
);
}
-22
View File
@@ -1,22 +0,0 @@
/**
* Bounds on server-fed chat state. A hostile or malfunctioning gateway can
* flood any of these collections; caps keep memory/render cost flat instead
* of growing unboundedly for the lifetime of the connection.
*/
/** Max characters retained for the in-flight streamed text/thinking buffers. */
export const MAX_STREAM_CHARS = 20_000;
/** Max transcript turns retained (oldest dropped first). */
export const MAX_MESSAGES = 500;
/** Max tool-call entries (including anomaly entries) retained per turn history. */
export const MAX_TOOLS = 200;
/** Max slash-command results retained. */
export const MAX_COMMAND_RESULTS = 200;
/** Max commands/skills accepted from a single manifest push. */
export const MAX_MANIFEST_ITEMS = 500;
/** Max executed approval IDs remembered for single-flight dedup. */
export const MAX_EXECUTED_APPROVAL_IDS = 200;
/** Max characters retained for a single command:result message — a hostile
* or malfunctioning gateway must not be able to push an unbounded curated
* success/failure reason into state (or, defensively, onto the page). */
export const MAX_COMMAND_MESSAGE_CHARS = 1_000;
@@ -1,39 +0,0 @@
import type { ReactElement } from 'react';
import type { ChatTranscriptMessage } from './use-chat-connection';
interface MessageTranscriptProps {
messages: ChatTranscriptMessage[];
streaming: boolean;
text: string;
}
export function MessageTranscript({
messages,
streaming,
text,
}: MessageTranscriptProps): ReactElement {
return (
<div
role="log"
aria-live="polite"
aria-label="Conversation"
className="flex flex-1 flex-col gap-3 overflow-y-auto p-4"
>
{messages.map((message) => (
<div key={message.id} data-role={message.role} className="whitespace-pre-wrap text-sm">
<span className="font-medium">{message.role === 'user' ? 'You' : 'Assistant'}: </span>
<span>{message.text}</span>
{message.thinking ? (
<div className="pt-1 text-xs italic opacity-70">{message.thinking}</div>
) : null}
</div>
))}
{streaming ? (
<div data-role="assistant-streaming" className="whitespace-pre-wrap text-sm">
<span className="font-medium">Assistant: </span>
<span>{text || 'Thinking…'}</span>
</div>
) : null}
</div>
);
}
-150
View File
@@ -1,150 +0,0 @@
/**
* Socket.IO payloads are only statically typed at the call site — a
* misbehaving or compromised gateway can send anything at runtime. These
* guards protect the dereference sites that would otherwise throw (`.map` on
* a non-array, `.toFixed` on a non-number) or render an object as a React
* child.
*/
import type {
HarnessAuthState,
HarnessCatalog,
HarnessCatalogEntry,
HarnessModelAvailability,
HarnessSelection,
HarnessSummary,
} from '@/lib/types';
export function asString(value: unknown, fallback = ''): string {
return typeof value === 'string' ? value : fallback;
}
/** Like `asString`, but an empty string also falls back — used for guarded
* contract-provided reason strings (e.g. a denial or failure message) where
* an empty string is not a meaningful value to display in place of the
* stable fallback copy. */
export function asNonEmptyString(value: unknown, fallback: string): string {
return typeof value === 'string' && value.length > 0 ? value : fallback;
}
export function asFiniteNumber(value: unknown, fallback = 0): number {
return typeof value === 'number' && Number.isFinite(value) ? value : fallback;
}
/** Like `asFiniteNumber`, but returns `null` on failure instead of a numeric
* fallback — callers that must not fabricate a plausible-looking value (e.g.
* `0 tokens` / `$0.0000` for genuinely unknown usage) use this to render an
* honest "unavailable" label instead. */
export function asFiniteNumberOrNull(value: unknown): number | null {
return typeof value === 'number' && Number.isFinite(value) ? value : null;
}
export function asStringArray(value: unknown): string[] {
return Array.isArray(value) && value.every((item) => typeof item === 'string') ? value : [];
}
export function isRecord(value: unknown): value is Record<string, unknown> {
return typeof value === 'object' && value !== null;
}
/**
* The HTTP harness/catalog/selection JSON bodies are as untrusted as the socket
* payloads above — a misbehaving or compromised gateway can send anything. The
* guards below normalize those bodies into the typed client shapes without ever
* rendering a raw body, so a 404/422/malformed response can never inject an
* object into React or a non-tuple into the selection state.
*/
/** Normalizes an untrusted `authState` to the closed set, defaulting to the
* safest value (`unavailable`) for anything unrecognized. */
export function asHarnessAuthState(value: unknown): HarnessAuthState {
return value === 'ready' || value === 'auth_required' || value === 'unavailable'
? value
: 'unavailable';
}
/** Normalizes an untrusted `availability` to the closed set, defaulting to
* `unavailable` so a malformed row can never present as sendable. */
export function asHarnessAvailability(value: unknown): HarnessModelAvailability {
return value === 'available' ? 'available' : 'unavailable';
}
/** A tuple is valid only when all three ids are non-empty strings — a partial
* or malformed selection is rejected (null) rather than half-adopted. */
export function asHarnessSelection(value: unknown): HarnessSelection | null {
if (!isRecord(value)) return null;
const harnessId = value.harnessId;
const providerId = value.providerId;
const modelId = value.modelId;
if (
typeof harnessId !== 'string' ||
typeof providerId !== 'string' ||
typeof modelId !== 'string' ||
harnessId.length === 0 ||
providerId.length === 0 ||
modelId.length === 0
) {
return null;
}
return { harnessId, providerId, modelId };
}
/** Normalizes an untrusted array into typed harness summaries, dropping any row
* without a usable id. */
export function asHarnessSummaries(value: unknown): HarnessSummary[] {
if (!Array.isArray(value)) return [];
const out: HarnessSummary[] = [];
for (const item of value) {
if (!isRecord(item)) continue;
const id = asString(item.id);
if (id.length === 0) continue;
out.push({
id,
displayName: asNonEmptyString(item.displayName, id),
capabilities: asStringArray(item.capabilities),
});
}
return out;
}
function asHarnessCatalogEntry(value: unknown): HarnessCatalogEntry | null {
const selection = asHarnessSelection(value);
if (selection === null || !isRecord(value)) return null;
return {
...selection,
displayName: asNonEmptyString(value.displayName, selection.modelId),
reasoningCapability: value.reasoningCapability === true,
inputTypes: asStringArray(value.inputTypes),
authState: asHarnessAuthState(value.authState),
availability: asHarnessAvailability(value.availability),
};
}
/** Normalizes an untrusted catalog body into the typed client catalog. The
* caller supplies `harnessId` (from the request path) so the returned catalog
* is scoped to the harness that was actually requested, never a body-echoed id.
* Malformed model rows are dropped rather than invalidating the whole catalog. */
export function asHarnessCatalog(value: unknown, harnessId: string): HarnessCatalog {
const record = isRecord(value) ? value : {};
const rawModels = Array.isArray(record.models) ? record.models : [];
const models: HarnessCatalogEntry[] = [];
for (const row of rawModels) {
const entry = asHarnessCatalogEntry(row);
if (entry !== null) models.push(entry);
}
return {
harnessId,
version: asString(record.version),
fingerprint: asString(record.fingerprint),
models,
};
}
/** The single point of truth for what counts as a valid conversation ID
* anywhere a scoped server event may adopt one into state — a non-empty
* string, nothing else. Every site that establishes or compares
* `state.conversationId` against a raw socket payload must route through
* this guard so a malformed first frame (null/object/number/empty string)
* can never be adopted verbatim. */
export function asConversationId(value: unknown): string | null {
return typeof value === 'string' && value.length > 0 ? value : null;
}
-68
View File
@@ -1,68 +0,0 @@
import type { ReactElement } from 'react';
import type { SessionInfoPayload } from '@/lib/chat-contract';
import { MAX_MANIFEST_ITEMS } from './limits';
import { asString, asStringArray } from './runtime-guards';
interface SessionPanelProps {
sessionInfo: SessionInfoPayload | null;
onSetThinking: (level: string) => void;
}
const THINKING_LEVEL_UNAVAILABLE = '';
export function SessionPanel({
sessionInfo,
onSetThinking,
}: SessionPanelProps): ReactElement | null {
if (!sessionInfo) return null;
// The reducer already caps this before storing it, but the render site
// defends independently — a hostile payload must never be able to force
// this <select> to lay out an unbounded number of options.
const availableThinkingLevels = asStringArray(sessionInfo.availableThinkingLevels).slice(
0,
MAX_MANIFEST_ITEMS,
);
const hasThinkingLevels = availableThinkingLevels.length > 0;
return (
<section
aria-label="Session info"
className="flex flex-wrap items-center gap-3 border-b px-4 py-2 text-xs"
>
<span>{asString(sessionInfo.provider, 'unknown')}</span>
<span>{asString(sessionInfo.modelId, 'unknown')}</span>
<label className="flex items-center gap-2">
<span>Thinking level</span>
<select
aria-label="Thinking level"
value={
hasThinkingLevels ? asString(sessionInfo.thinkingLevel) : THINKING_LEVEL_UNAVAILABLE
}
onChange={(event) => {
// The placeholder option is not a real, settable level — a
// malformed availableThinkingLevels list must never let the
// client emit set:thinking for it.
if (!hasThinkingLevels) return;
onSetThinking(event.target.value);
}}
>
{hasThinkingLevels ? (
availableThinkingLevels.map((level) => (
<option key={level} value={level}>
{level}
</option>
))
) : (
<option value={THINKING_LEVEL_UNAVAILABLE}>Thinking level unavailable</option>
)}
</select>
</label>
{sessionInfo.routingDecision ? (
<span title={asString(sessionInfo.routingDecision.ruleName)}>
{asString(sessionInfo.routingDecision.reason)}
</span>
) : null}
</section>
);
}
@@ -1,125 +0,0 @@
import { vi } from 'vitest';
import type { ClientToServerEvents, ServerToClientEvents } from '@/lib/chat-contract';
type ServerEvent = keyof ServerToClientEvents;
type ClientEvent = keyof ClientToServerEvents;
type ServerHandler<K extends ServerEvent> = ServerToClientEvents[K];
type ClientPayload<K extends ClientEvent> = Parameters<ClientToServerEvents[K]>[0];
export interface EmittedEvent<K extends ClientEvent = ClientEvent> {
event: K;
payload: ClientPayload<K>;
}
/** The subset of a Socket.IO `ChatSocket` that `useChatConnection` drives. */
export interface FakeChatSocket {
connected: boolean;
connect(): FakeChatSocket;
on<K extends ServerEvent>(event: K, handler: ServerHandler<K>): FakeChatSocket;
off<K extends ServerEvent>(event: K, handler: ServerHandler<K>): FakeChatSocket;
emit<K extends ClientEvent>(event: K, payload: ClientPayload<K>): FakeChatSocket;
}
/**
* A typed in-memory stand-in for `getSocket()`. Unlike a bare
* `(event: string, payload: unknown) => void` mock, every public method here is
* checked against the real `/chat` contract — a typo'd event name or a payload
* missing a required field fails to compile instead of silently no-op'ing at
* runtime.
*/
/** Socket.IO's built-in connection-state events. Not part of the app-level
* ServerToClientEvents contract, but real sockets always support them and
* `useChatConnection` registers a `disconnect` handler on the real socket. */
type LifecycleEvent = 'connect' | 'disconnect';
export function createFakeChatSocket(): {
socket: FakeChatSocket;
listeners: Map<ServerEvent, Set<(payload: never) => void>>;
emitted: EmittedEvent[];
serverEmit<K extends ServerEvent>(
event: K,
payload: Parameters<ServerToClientEvents[K]>[0],
): void;
/** Escape hatch for malformed-payload tests: bypasses the compile-time
* payload contract to simulate a genuinely untrusted runtime value from the
* server, e.g. a `session:info` with a non-array `availableThinkingLevels`. */
serverEmitRaw(event: ServerEvent, payload: unknown): void;
/** Simulates a transient Socket.IO `disconnect` — fires any handler(s)
* registered via `socket.on('disconnect', ...)` without clearing any
* listeners, mirroring how a real reconnecting socket behaves. */
simulateDisconnect(): void;
/** Simulates socket.io-client's automatic reconnect of the *same*
* instance after a transient disconnect: marks the socket connected again
* and fires any handler(s) registered via `socket.on('connect', ...)`,
* without clearing or replacing any listeners. */
simulateReconnect(): void;
} {
const listeners = new Map<ServerEvent, Set<(payload: never) => void>>();
const emitted: EmittedEvent[] = [];
// Internal storage is intentionally keyed loosely (the per-event handler shape
// varies by K, which a single Map can't express); the generic signatures on the
// exported `socket`/`serverEmit` above and below are what keep test call sites
// type-checked against ServerToClientEvents/ClientToServerEvents.
const socket = {
connected: false,
connect: vi.fn(function connect(this: void) {
socket.connected = true;
return socket;
}),
on: vi.fn(function on(this: void, event: ServerEvent, handler: (payload: never) => void) {
if (!listeners.has(event)) listeners.set(event, new Set());
listeners.get(event)?.add(handler);
return socket;
}),
off: vi.fn(function off(this: void, event: ServerEvent, handler: (payload: never) => void) {
listeners.get(event)?.delete(handler);
return socket;
}),
emit: vi.fn(function emit(this: void, event: ClientEvent, payload: unknown) {
emitted.push({ event, payload } as EmittedEvent);
return socket;
}),
} as unknown as FakeChatSocket;
function serverEmit<K extends ServerEvent>(
event: K,
payload: Parameters<ServerToClientEvents[K]>[0],
): void {
for (const handler of listeners.get(event) ?? []) {
(handler as (payload: Parameters<ServerToClientEvents[K]>[0]) => void)(payload);
}
}
function serverEmitRaw(event: ServerEvent, payload: unknown): void {
for (const handler of listeners.get(event) ?? []) {
(handler as (payload: unknown) => void)(payload);
}
}
function simulateDisconnect(): void {
socket.connected = false;
const lifecycleKey = 'disconnect' satisfies LifecycleEvent as unknown as ServerEvent;
for (const handler of listeners.get(lifecycleKey) ?? []) {
(handler as () => void)();
}
}
function simulateReconnect(): void {
socket.connected = true;
const lifecycleKey = 'connect' satisfies LifecycleEvent as unknown as ServerEvent;
for (const handler of listeners.get(lifecycleKey) ?? []) {
(handler as () => void)();
}
}
return {
socket,
listeners,
emitted,
serverEmit,
serverEmitRaw,
simulateDisconnect,
simulateReconnect,
};
}
@@ -1,63 +0,0 @@
import { act } from 'react';
import { createRoot, type Root } from 'react-dom/client';
import { afterAll, afterEach, beforeAll, describe, expect, it, vi } from 'vitest';
import { ToolCallList } from './tool-call-list';
beforeAll(() => {
Object.defineProperty(globalThis, 'IS_REACT_ACT_ENVIRONMENT', {
configurable: true,
value: true,
});
});
afterAll(() => {
Reflect.deleteProperty(globalThis, 'IS_REACT_ACT_ENVIRONMENT');
});
let root: Root | null;
let container: HTMLElement | null;
async function render(node: Parameters<Root['render']>[0]): Promise<void> {
container = document.createElement('div');
document.body.append(container);
root = createRoot(container);
await act(async () => {
root?.render(node);
});
}
afterEach(async () => {
await act(async () => {
root?.unmount();
});
document.body.replaceChildren();
root = null;
container = null;
});
describe('ToolCallList', () => {
it('renders two entries independently, without a duplicate-key warning, when a valid toolCallId is shared', async () => {
const consoleError = vi.spyOn(console, 'error').mockImplementation(() => {});
await render(
<ToolCallList
tools={[
{ toolCallId: 'dup', toolName: 'search', status: 'success' },
{ toolCallId: 'dup', toolName: 'search', status: 'running' },
]}
/>,
);
const items = [...(container?.querySelectorAll('li') ?? [])];
expect(items).toHaveLength(2);
expect(items[0]?.textContent).toContain('success');
expect(items[1]?.textContent).toContain('running');
const duplicateKeyWarning = consoleError.mock.calls.some((args) =>
args.some((arg) => typeof arg === 'string' && arg.includes('same key')),
);
expect(duplicateKeyWarning).toBe(false);
consoleError.mockRestore();
});
});
-24
View File
@@ -1,24 +0,0 @@
import type { ReactElement } from 'react';
import type { ToolCallState } from './use-chat-connection';
export function ToolCallList({ tools }: { tools: ToolCallState[] }): ReactElement | null {
if (tools.length === 0) return null;
return (
<ul aria-label="Tool calls" className="flex flex-col gap-1 px-4 pb-2 text-xs">
{tools.map((tool, index) => (
<li
// A valid server-controlled toolCallId can legitimately repeat
// (e.g. two tool:start events sharing one id) — keying on it alone
// would give React two identical keys. Pairing it with its
// (stable, append-only) render index keeps every key unique.
key={`${tool.toolCallId}-${index}`}
role={tool.status === 'error' || tool.status === 'anomaly' ? 'alert' : 'status'}
>
{tool.toolName} {' '}
{tool.status === 'anomaly' ? 'unexpected end (unknown tool call)' : tool.status}
</li>
))}
</ul>
);
}
File diff suppressed because it is too large Load Diff
File diff suppressed because it is too large Load Diff
@@ -1,445 +0,0 @@
import { act, type ReactElement } from 'react';
import { createRoot, type Root } from 'react-dom/client';
import { afterAll, afterEach, beforeAll, beforeEach, describe, expect, it, vi } from 'vitest';
import { useHarnessSelection, type HarnessSelectionValue } from './use-harness-selection';
function json(body: unknown, status = 200): Response {
return new Response(JSON.stringify(body), {
status,
headers: { 'Content-Type': 'application/json' },
});
}
interface Scenario {
harnesses?: unknown;
catalog?: { body: unknown; status?: number };
selection?: unknown;
/** When set, the PUT resolves only when this is called (for race tests). */
deferPut?: boolean;
}
interface Deferred<T> {
promise: Promise<T>;
resolve: (value: T) => void;
}
function defer<T>(): Deferred<T> {
let resolve!: (value: T) => void;
const promise = new Promise<T>((r) => {
resolve = r;
});
return { promise, resolve };
}
let putBodies: unknown[] = [];
let putDeferred: Deferred<Response> | null = null;
function installFetch(scenario: Scenario): ReturnType<typeof vi.fn> {
putBodies = [];
putDeferred = scenario.deferPut ? defer<Response>() : null;
const fetchMock = vi.fn(async (input: unknown, init?: RequestInit) => {
const url = String(input);
const method = String(init?.method ?? 'GET').toUpperCase();
if (url === '/api/harnesses') return json(scenario.harnesses ?? []);
if (url.startsWith('/api/harnesses/') && url.endsWith('/catalog')) {
const spec = scenario.catalog ?? {
body: { harnessId: 'pi', version: '1', fingerprint: 'f', models: [] },
};
return json(spec.body, spec.status ?? 200);
}
if (url === '/api/chat/preferences/selection' && method === 'GET') {
return json({ selection: scenario.selection ?? null });
}
if (url === '/api/chat/preferences/selection' && method === 'PUT') {
putBodies.push(JSON.parse(String(init?.body)));
const ok = json({ selection: JSON.parse(String(init?.body)) });
if (putDeferred) return putDeferred.promise;
return ok;
}
return new Response('not found', { status: 404 });
});
vi.stubGlobal('fetch', fetchMock);
return fetchMock;
}
let latest: HarnessSelectionValue | null = null;
function Probe(): ReactElement | null {
latest = useHarnessSelection();
return null;
}
let root: Root | null;
let container: HTMLElement;
beforeAll(() => {
Object.defineProperty(globalThis, 'IS_REACT_ACT_ENVIRONMENT', {
configurable: true,
value: true,
});
});
afterAll(() => {
Reflect.deleteProperty(globalThis, 'IS_REACT_ACT_ENVIRONMENT');
});
beforeEach(() => {
latest = null;
container = document.createElement('div');
document.body.append(container);
root = createRoot(container);
});
afterEach(async () => {
await act(async () => {
root?.unmount();
});
document.body.replaceChildren();
vi.unstubAllGlobals();
});
async function mount(): Promise<void> {
await act(async () => {
root?.render(<Probe />);
});
await flush();
}
async function flush(times = 5): Promise<void> {
for (let i = 0; i < times; i += 1) {
await act(async () => {
await Promise.resolve();
});
}
}
function value(): HarnessSelectionValue {
if (!latest) throw new Error('hook value not captured');
return latest;
}
const PI_CATALOG = {
harnessId: 'pi',
version: '2026-08-11',
fingerprint: 'fp',
models: [
{
harnessId: 'pi',
providerId: 'openai',
modelId: 'gpt-5',
displayName: 'GPT-5',
reasoningCapability: true,
inputTypes: ['text'],
authState: 'ready',
availability: 'available',
},
{
harnessId: 'pi',
providerId: 'anthropic',
modelId: 'claude',
displayName: 'Claude',
reasoningCapability: true,
inputTypes: ['text'],
authState: 'ready',
availability: 'available',
},
],
};
describe('useHarnessSelection', () => {
it('loads harnesses and, once a harness is chosen, the model options come only from its catalog', async () => {
installFetch({
harnesses: [{ id: 'pi', displayName: 'Pi', capabilities: [] }],
catalog: { body: PI_CATALOG },
selection: null,
});
await mount();
expect(value().harnesses).toEqual([{ id: 'pi', displayName: 'Pi', capabilities: [] }]);
expect(value().catalog).toBeNull();
await act(async () => {
value().selectHarness('pi');
});
await flush();
expect(value().catalog?.harnessId).toBe('pi');
expect(value().catalog?.models.map((m) => m.modelId)).toEqual(['gpt-5', 'claude']);
});
it('does not auto-select any catalog row when there is no persisted selection (no first-row fallback)', async () => {
const fetchMock = installFetch({
harnesses: [{ id: 'pi', displayName: 'Pi', capabilities: [] }],
catalog: { body: PI_CATALOG },
selection: null,
});
await mount();
await act(async () => {
value().selectHarness('pi');
});
await flush();
expect(value().modelId).toBe('');
expect(value().persistedSelection).toBeNull();
expect(value().canSend).toBe(false);
// Nothing was persisted — no PUT fired for an unset selection.
const putCalls = fetchMock.mock.calls.filter(
(c) => String((c[1] as RequestInit)?.method).toUpperCase() === 'PUT',
);
expect(putCalls).toHaveLength(0);
});
it('persists the structured tuple and only enables send AFTER the PUT resolves (no race ahead of persistence)', async () => {
installFetch({
harnesses: [{ id: 'pi', displayName: 'Pi', capabilities: [] }],
catalog: { body: PI_CATALOG },
selection: null,
deferPut: true,
});
await mount();
await act(async () => {
value().selectHarness('pi');
});
await flush();
await act(async () => {
value().selectProvider('openai');
});
await act(async () => {
value().selectModel('openai', 'gpt-5');
});
await flush();
// PUT is in flight (deferred) — send MUST NOT be enabled yet.
expect(value().canSend).toBe(false);
await act(async () => {
putDeferred?.resolve(
json({ selection: { harnessId: 'pi', providerId: 'openai', modelId: 'gpt-5' } }),
);
});
await flush();
expect(putBodies).toContainEqual({ harnessId: 'pi', providerId: 'openai', modelId: 'gpt-5' });
expect(value().persistedSelection).toEqual({
harnessId: 'pi',
providerId: 'openai',
modelId: 'gpt-5',
});
expect(value().canSend).toBe(true);
expect(value().projection).toEqual({ provider: 'openai', modelId: 'gpt-5' });
});
it('keeps a stale/unavailable persisted selection visibly displayed rather than silently dropping it', async () => {
installFetch({
harnesses: [{ id: 'pi', displayName: 'Pi', capabilities: [] }],
catalog: { body: PI_CATALOG },
selection: { harnessId: 'pi', providerId: 'openai', modelId: 'retired-model' },
});
await mount();
// The persisted tuple is displayed even though its model is gone from the catalog.
expect(value().persistedSelection).toEqual({
harnessId: 'pi',
providerId: 'openai',
modelId: 'retired-model',
});
expect(value().modelId).toBe('retired-model');
expect(value().isStale).toBe(true);
// A stale model is not a valid catalog option, so send stays disabled.
expect(value().canSend).toBe(false);
});
it('disables send for an empty catalog (no viable model) and never fabricates one', async () => {
installFetch({
harnesses: [{ id: 'pi', displayName: 'Pi', capabilities: [] }],
catalog: { body: { harnessId: 'pi', version: '1', fingerprint: 'f', models: [] } },
selection: null,
});
await mount();
await act(async () => {
value().selectHarness('pi');
});
await flush();
expect(value().catalog?.models ?? []).toHaveLength(0);
expect(value().canSend).toBe(false);
});
it('marks the catalog unavailable and disables send when the catalog request 404s', async () => {
installFetch({
harnesses: [{ id: 'pi', displayName: 'Pi', capabilities: [] }],
catalog: {
body: { code: 'adapter_unavailable', message: 'x', harnessId: 'pi' },
status: 404,
},
selection: null,
});
await mount();
await act(async () => {
value().selectHarness('pi');
});
await flush();
expect(value().catalogUnavailable).toBe(true);
expect(value().canSend).toBe(false);
});
it('on a 422 persist, keeps the requested tuple visible, surfaces a typed error, and leaves send disabled', async () => {
installFetch({
harnesses: [{ id: 'pi', displayName: 'Pi', capabilities: [] }],
catalog: {
body: {
...PI_CATALOG,
models: [{ ...PI_CATALOG.models[0], availability: 'unavailable' }],
},
},
selection: null,
});
// Override PUT to 422.
const fetchMock = vi.fn(async (input: unknown, init?: RequestInit) => {
const url = String(input);
const method = String(init?.method ?? 'GET').toUpperCase();
if (url === '/api/harnesses')
return json([{ id: 'pi', displayName: 'Pi', capabilities: [] }]);
if (url.endsWith('/catalog')) return json(PI_CATALOG);
if (url === '/api/chat/preferences/selection' && method === 'GET')
return json({ selection: null });
if (url === '/api/chat/preferences/selection' && method === 'PUT') {
return json(
{
code: 'model_unavailable',
message: 'nope',
selection: { harnessId: 'a', providerId: 'b', modelId: 'c' },
},
422,
);
}
return new Response('nf', { status: 404 });
});
vi.stubGlobal('fetch', fetchMock);
await mount();
await act(async () => {
value().selectHarness('pi');
});
await flush();
await act(async () => {
value().selectProvider('openai');
});
await act(async () => {
value().selectModel('openai', 'gpt-5');
});
await flush();
expect(value().modelId).toBe('gpt-5');
expect(value().persistError?.code).toBe('model_unavailable');
expect(value().persistError?.requested).toEqual({
harnessId: 'pi',
providerId: 'openai',
modelId: 'gpt-5',
});
expect(value().persistedSelection).toBeNull();
expect(value().canSend).toBe(false);
});
it('invalidates the model on a provider change and keeps send disabled until the new tuple persists', async () => {
installFetch({
harnesses: [{ id: 'pi', displayName: 'Pi', capabilities: [] }],
catalog: { body: PI_CATALOG },
selection: null,
});
await mount();
await act(async () => {
value().selectHarness('pi');
});
await flush();
await act(async () => {
value().selectProvider('openai');
});
await act(async () => {
value().selectModel('openai', 'gpt-5');
});
await flush();
// A valid provider-A tuple has persisted.
expect(value().canSend).toBe(true);
expect(value().persistedSelection).toEqual({
harnessId: 'pi',
providerId: 'openai',
modelId: 'gpt-5',
});
// Switching provider clears the model that no longer belongs to it.
await act(async () => {
value().selectProvider('anthropic');
});
expect(value().modelId).toBe('');
expect(value().canSend).toBe(false);
// Send stays disabled until the new exact provider-B tuple persists.
await act(async () => {
value().selectModel('anthropic', 'claude');
});
await flush();
expect(value().canSend).toBe(true);
expect(value().persistedSelection).toEqual({
harnessId: 'pi',
providerId: 'anthropic',
modelId: 'claude',
});
expect(value().projection).toEqual({ provider: 'anthropic', modelId: 'claude' });
});
it('does not enable send on a model pick until the PUT for that exact new tuple resolves', async () => {
installFetch({
harnesses: [{ id: 'pi', displayName: 'Pi', capabilities: [] }],
catalog: { body: PI_CATALOG },
selection: { harnessId: 'pi', providerId: 'openai', modelId: 'gpt-5' },
deferPut: true,
});
await mount();
// The persisted, in-catalog tuple is sendable after mount (no PUT needed).
expect(value().canSend).toBe(true);
await act(async () => {
value().selectProvider('anthropic');
});
expect(value().modelId).toBe('');
expect(value().canSend).toBe(false);
await act(async () => {
value().selectModel('anthropic', 'claude');
});
await flush();
// PUT for the new tuple is still in flight — send MUST stay disabled.
expect(value().canSend).toBe(false);
await act(async () => {
putDeferred?.resolve(
json({ selection: { harnessId: 'pi', providerId: 'anthropic', modelId: 'claude' } }),
);
});
await flush();
expect(value().canSend).toBe(true);
expect(value().projection).toEqual({ provider: 'anthropic', modelId: 'claude' });
});
it('never requests any /api/providers* endpoint across the whole flow', async () => {
const fetchMock = installFetch({
harnesses: [{ id: 'pi', displayName: 'Pi', capabilities: [] }],
catalog: { body: PI_CATALOG },
selection: { harnessId: 'pi', providerId: 'openai', modelId: 'gpt-5' },
});
await mount();
await act(async () => {
value().selectProvider('anthropic');
});
await act(async () => {
value().selectModel('anthropic', 'claude');
});
await flush();
for (const call of fetchMock.mock.calls) {
expect(String(call[0])).not.toContain('/api/providers');
}
});
});
@@ -1,216 +0,0 @@
import { useCallback, useEffect, useRef, useState } from 'react';
import {
fetchCatalog,
fetchHarnesses,
fetchPersistedSelection,
persistSelection,
type SelectionErrorCode,
} from './chat-api';
import type { HarnessCatalog, HarnessSelection, HarnessSummary } from '@/lib/types';
export interface HarnessPersistError {
code: SelectionErrorCode;
message: string;
/** The exact tuple the user requested — preserved so the failed selection
* stays visible rather than being silently dropped. */
requested: HarnessSelection;
}
export interface HarnessSelectionValue {
harnesses: HarnessSummary[];
catalog: HarnessCatalog | null;
/** True when the selected harness has no usable catalog (404/error). */
catalogUnavailable: boolean;
/** The working (displayed) selection, kept as three distinct ids. Empty
* strings mean "not chosen yet" — there is deliberately no first-row default. */
harnessId: string;
providerId: string;
modelId: string;
/** The last tuple confirmed persisted by the server, or null. */
persistedSelection: HarnessSelection | null;
/** True when a persisted selection references a model no longer present as an
* available catalog entry — it stays visibly displayed rather than dropped. */
isStale: boolean;
/** True ONLY once a full tuple has been confirmed persisted AND it is a
* currently-available catalog entry. Send stays disabled otherwise, so a send
* can never race ahead of successful persistence. */
canSend: boolean;
persistError: HarnessPersistError | null;
selectHarness: (harnessId: string) => void;
selectProvider: (providerId: string) => void;
/** Persist the EXACT catalog row's `{providerId, modelId}` — the caller
* resolves the composite option identity to the real entry and passes both
* ids, so a bare model id is never combined with ambient provider state. */
selectModel: (providerId: string, modelId: string) => void;
/** The compatibility `{provider, modelId}` projection for the legacy socket
* send path — derived ONLY from the validated persisted tuple, never from any
* free-text or unpersisted draft. Empty when nothing is sendable. */
projection: { provider?: string; modelId?: string };
}
/** A tuple is a currently-usable catalog option only when the catalog holds a
* matching, available entry — the single gate that keeps a stale/unavailable
* model from ever counting as sendable. */
function isAvailableInCatalog(
selection: HarnessSelection | null,
catalog: HarnessCatalog | null,
): boolean {
if (selection === null || catalog === null) return false;
return catalog.models.some(
(model) =>
model.providerId === selection.providerId &&
model.modelId === selection.modelId &&
model.availability === 'available',
);
}
function tuplesEqual(a: HarnessSelection | null, b: HarnessSelection | null): boolean {
if (a === null || b === null) return a === b;
return a.harnessId === b.harnessId && a.providerId === b.providerId && a.modelId === b.modelId;
}
/**
* Owns the harness/catalog/selection state for the chat composer: loads the
* harness list and any persisted tuple on mount, loads a harness's catalog when
* chosen, and PUT-persists the full `{harnessId, providerId, modelId}` tuple
* when a model is picked. It never auto-selects a catalog row, keeps a
* stale/unavailable persisted tuple visible, and only reports `canSend` true
* once a full tuple has actually persisted as an available catalog entry.
*/
export function useHarnessSelection(): HarnessSelectionValue {
const [harnesses, setHarnesses] = useState<HarnessSummary[]>([]);
const [catalog, setCatalog] = useState<HarnessCatalog | null>(null);
const [catalogUnavailable, setCatalogUnavailable] = useState(false);
const [harnessId, setHarnessId] = useState('');
const [providerId, setProviderId] = useState('');
const [modelId, setModelId] = useState('');
const [persistedSelection, setPersistedSelection] = useState<HarnessSelection | null>(null);
const [persistError, setPersistError] = useState<HarnessPersistError | null>(null);
// Monotonic request ids so a slow in-flight catalog/persist response can never
// overwrite the result of a newer request the user has since triggered.
const catalogRequestRef = useRef(0);
const persistRequestRef = useRef(0);
const loadCatalog = useCallback(async (id: string): Promise<void> => {
const requestId = catalogRequestRef.current + 1;
catalogRequestRef.current = requestId;
setCatalog(null);
setCatalogUnavailable(false);
const result = await fetchCatalog(id);
if (catalogRequestRef.current !== requestId) return;
if (result.ok) {
setCatalog(result.catalog);
setCatalogUnavailable(false);
} else {
setCatalog(null);
setCatalogUnavailable(true);
}
}, []);
useEffect(() => {
let active = true;
void (async (): Promise<void> => {
const [list, persisted] = await Promise.all([fetchHarnesses(), fetchPersistedSelection()]);
if (!active) return;
setHarnesses(list);
if (persisted !== null) {
// Adopt the persisted tuple as the displayed selection and load its
// catalog. If the model has since been retired, it still shows (stale).
setHarnessId(persisted.harnessId);
setProviderId(persisted.providerId);
setModelId(persisted.modelId);
setPersistedSelection(persisted);
await loadCatalog(persisted.harnessId);
}
// No persisted selection → nothing is auto-selected; the user must choose.
})();
return () => {
active = false;
};
}, [loadCatalog]);
const selectHarness = useCallback(
(id: string): void => {
setHarnessId(id);
// Changing harness invalidates the provider/model draft — never carry a
// model across harnesses.
setProviderId('');
setModelId('');
setPersistError(null);
void loadCatalog(id);
},
[loadCatalog],
);
const selectProvider = useCallback((id: string): void => {
setProviderId(id);
// A new provider invalidates the chosen model — no cross-provider carryover.
setModelId('');
setPersistError(null);
}, []);
const selectModel = useCallback(
(selectedProviderId: string, selectedModelId: string): void => {
// Bind the model to the EXACT catalog row's provider — never to ambient
// provider state — so two providers exposing the same modelId can never
// collide or mis-resolve. Keep the displayed provider consistent with the
// resolved row.
setProviderId(selectedProviderId);
setModelId(selectedModelId);
setPersistError(null);
const requested: HarnessSelection = {
harnessId,
providerId: selectedProviderId,
modelId: selectedModelId,
};
const requestId = persistRequestRef.current + 1;
persistRequestRef.current = requestId;
void (async (): Promise<void> => {
const result = await persistSelection(requested);
if (persistRequestRef.current !== requestId) return;
if (result.ok) {
setPersistedSelection(result.selection);
setPersistError(null);
} else {
// Leave persistedSelection unchanged (send stays disabled) and surface
// the typed error carrying the exact requested tuple.
setPersistError({
code: result.code,
message: result.message,
requested: result.requested,
});
}
})();
},
[harnessId],
);
const draft: HarnessSelection = { harnessId, providerId, modelId };
const isStale = persistedSelection !== null && !isAvailableInCatalog(persistedSelection, catalog);
const canSend =
persistedSelection !== null &&
!catalogUnavailable &&
tuplesEqual(draft, persistedSelection) &&
isAvailableInCatalog(persistedSelection, catalog);
const projection: { provider?: string; modelId?: string } = canSend
? { provider: persistedSelection.providerId, modelId: persistedSelection.modelId }
: {};
return {
harnesses,
catalog,
catalogUnavailable,
harnessId,
providerId,
modelId,
persistedSelection,
isStale,
canSend,
persistError,
selectHarness,
selectProvider,
selectModel,
projection,
};
}
@@ -1,87 +0,0 @@
import { act } from 'react';
import { createRoot, type Root } from 'react-dom/client';
import { createMemoryRouter, RouterProvider, type RouteObject } from 'react-router-dom';
import { afterAll, afterEach, beforeAll, describe, expect, it, vi } from 'vitest';
const { useSessionMock } = vi.hoisted(() => ({
useSessionMock: vi.fn(),
}));
vi.mock('@/lib/auth-client', () => ({
useSession: useSessionMock,
}));
import { routes } from '@/routes';
beforeAll(() => {
Object.defineProperty(globalThis, 'IS_REACT_ACT_ENVIRONMENT', {
configurable: true,
value: true,
});
});
afterAll(() => {
Reflect.deleteProperty(globalThis, 'IS_REACT_ACT_ENVIRONMENT');
});
function Boom(): never {
throw new Error('render blew up');
}
/** Recursively clones the real exported route table, replacing only the
* `/chat` route's `element` with `<Boom />` — every other route (including
* the real `AuthGuard` nesting and the real `/chat` `errorElement`) is left
* exactly as exported. This is what makes the test fail if a future change
* removes the real route's `errorElement`, unlike a hand-built independent
* route tree that could drift from production undetected. */
function replaceChatElementWithBoom(nodes: RouteObject[]): RouteObject[] {
return nodes.map((node) => {
const cloned: RouteObject = { ...node };
if (cloned.path === '/chat') {
cloned.element = <Boom />;
}
if (cloned.children) {
cloned.children = replaceChatElementWithBoom(cloned.children);
}
return cloned;
});
}
let root: Root | null;
let container: HTMLElement;
afterEach(async () => {
await act(async () => {
root?.unmount();
});
document.body.replaceChildren();
root = null;
useSessionMock.mockReset();
});
describe('ChatRouteErrorBoundary', () => {
it('renders a recoverable, non-blank fallback when the /chat route element throws during render', async () => {
useSessionMock.mockReturnValue({ data: { user: { id: 'user-1' } }, isPending: false });
const routeObjects = replaceChatElementWithBoom(routes);
const router = createMemoryRouter(routeObjects, { initialEntries: ['/chat'] });
container = document.createElement('div');
document.body.append(container);
root = createRoot(container);
const consoleErrorSpy = vi.spyOn(console, 'error').mockImplementation(() => {});
try {
await act(async () => {
root?.render(<RouterProvider router={router} />);
});
expect(consoleErrorSpy).toHaveBeenCalled();
} finally {
consoleErrorSpy.mockRestore();
}
expect(container.textContent).not.toBe('');
expect(container.querySelector('[role="alert"]')).toBeTruthy();
});
});
@@ -1,22 +0,0 @@
import type { ReactElement } from 'react';
import { useRouteError } from 'react-router-dom';
/**
* `/chat` renders live, server-driven state (streamed text, tool calls,
* manifests) that can carry malformed payloads no compile-time contract can
* fully rule out at every dereference site. This is the last line of
* defense: if something still throws during render, show a recoverable
* alert instead of leaving the user on a blank/white screen.
*/
export function ChatRouteErrorBoundary(): ReactElement {
useRouteError();
return (
<div role="alert" className="flex min-h-screen flex-col items-center justify-center gap-3 p-8">
<p className="text-sm font-medium">Something went wrong loading chat.</p>
<a href="/chat" className="text-sm underline">
Reload chat
</a>
</div>
);
}
-914
View File
@@ -1,914 +0,0 @@
import { act } from 'react';
import { createRoot, type Root } from 'react-dom/client';
import { afterAll, afterEach, beforeAll, beforeEach, describe, expect, it, vi } from 'vitest';
import { createFakeChatSocket } from '@/spa/chat/test-support/fake-chat-socket';
import { MAX_MANIFEST_ITEMS } from '@/spa/chat/limits';
const { getSocketMock, destroySocketMock } = vi.hoisted(() => ({
getSocketMock: vi.fn(),
destroySocketMock: vi.fn(),
}));
vi.mock('@/lib/socket', () => ({
getSocket: getSocketMock,
destroySocket: destroySocketMock,
}));
import { ChatPage } from './chat';
function setValue(el: HTMLInputElement | HTMLTextAreaElement, value: string): void {
const proto =
el instanceof HTMLTextAreaElement ? HTMLTextAreaElement.prototype : HTMLInputElement.prototype;
const setter = Object.getOwnPropertyDescriptor(proto, 'value')?.set;
setter?.call(el, value);
el.dispatchEvent(new Event('input', { bubbles: true }));
}
function selectValue(el: HTMLSelectElement, value: string): void {
const setter = Object.getOwnPropertyDescriptor(HTMLSelectElement.prototype, 'value')?.set;
setter?.call(el, value);
el.dispatchEvent(new Event('change', { bubbles: true }));
}
function findButton(container: HTMLElement, text: string): HTMLButtonElement {
const button = [...container.querySelectorAll('button')].find((candidate) =>
candidate.textContent?.includes(text),
);
if (!button) throw new Error(`Button with text "${text}" not found`);
return button;
}
function jsonResponse(body: unknown, status = 200): Response {
return new Response(JSON.stringify(body), {
status,
headers: { 'Content-Type': 'application/json' },
});
}
const DEFAULT_CATALOG = {
harnessId: 'pi',
version: '2026-08-11',
fingerprint: 'fp',
models: [
{
harnessId: 'pi',
providerId: 'openai',
modelId: 'gpt-5',
displayName: 'GPT-5',
reasoningCapability: true,
inputTypes: ['text'],
authState: 'ready',
availability: 'available',
},
{
harnessId: 'pi',
providerId: 'anthropic',
modelId: 'claude',
displayName: 'Claude',
reasoningCapability: true,
inputTypes: ['text'],
authState: 'ready',
availability: 'available',
},
],
};
/** A harness/catalog/selection HTTP stub for the chat-api the selection hook
* drives. `selection` seeds the persisted tuple returned by the GET (a valid
* in-catalog tuple by default, so `canSend` settles true after mount). */
function harnessFetch(
selection: unknown = { harnessId: 'pi', providerId: 'openai', modelId: 'gpt-5' },
): typeof fetch {
return vi.fn(async (input: unknown, init?: RequestInit) => {
const url = String(input);
const method = String(init?.method ?? 'GET').toUpperCase();
if (url === '/api/harnesses') {
return jsonResponse([{ id: 'pi', displayName: 'Pi', capabilities: [] }]);
}
if (url.startsWith('/api/harnesses/') && url.endsWith('/catalog')) {
return jsonResponse(DEFAULT_CATALOG);
}
if (url === '/api/chat/preferences/selection' && method === 'GET') {
return jsonResponse({ selection });
}
if (url === '/api/chat/preferences/selection' && method === 'PUT') {
return jsonResponse({ selection: JSON.parse(String(init?.body)) });
}
return new Response('not found', { status: 404 });
}) as unknown as typeof fetch;
}
/** Drains the selection hook's chained mount fetches (harnesses → selection →
* catalog) and any pending PUT so derived `canSend` settles before assertions. */
async function flushAsync(times = 5): Promise<void> {
for (let i = 0; i < times; i += 1) {
await act(async () => {
await Promise.resolve();
});
}
}
let fake: ReturnType<typeof createFakeChatSocket>;
let root: Root | null;
let container: HTMLElement;
beforeAll(() => {
Object.defineProperty(globalThis, 'IS_REACT_ACT_ENVIRONMENT', {
configurable: true,
value: true,
});
});
afterAll(() => {
Reflect.deleteProperty(globalThis, 'IS_REACT_ACT_ENVIRONMENT');
});
beforeEach(async () => {
fake = createFakeChatSocket();
getSocketMock.mockReset().mockReturnValue(fake.socket);
destroySocketMock.mockReset();
vi.stubGlobal('fetch', harnessFetch());
container = document.createElement('div');
document.body.append(container);
root = createRoot(container);
await act(async () => {
root?.render(<ChatPage />);
});
// Settle the selection hook's mount fetches so the default in-catalog tuple
// persists and `canSend` is true for the existing send-path tests.
await flushAsync();
});
afterEach(async () => {
await act(async () => {
root?.unmount();
});
document.body.replaceChildren();
vi.unstubAllGlobals();
});
/** Re-mounts ChatPage against a custom fetch stub (e.g. an unset selection) for
* tests that need a non-default selection scenario. */
async function remountWithFetch(fetchImpl: typeof fetch): Promise<void> {
await act(async () => {
root?.unmount();
});
vi.stubGlobal('fetch', fetchImpl);
root = createRoot(container);
await act(async () => {
root?.render(<ChatPage />);
});
await flushAsync();
}
describe('ChatPage', () => {
it('streams agent:text and agent:thinking, shows tool status, and finalizes on agent:end with usage', async () => {
await act(async () => {
fake.serverEmit('message:ack', { conversationId: 'c1', messageId: 'm1' });
fake.serverEmit('agent:start', { conversationId: 'c1' });
});
await act(async () => {
fake.serverEmit('agent:thinking', { conversationId: 'c1', text: 'pondering…' });
fake.serverEmit('agent:text', { conversationId: 'c1', text: 'Hel' });
fake.serverEmit('agent:text', { conversationId: 'c1', text: 'lo!' });
fake.serverEmit('agent:tool:start', {
conversationId: 'c1',
toolCallId: 't1',
toolName: 'web_search',
});
});
expect(container.textContent).toContain('pondering…');
expect(container.textContent).toContain('Hello!');
expect(container.textContent).toContain('web_search');
expect(container.textContent).toMatch(/running/i);
await act(async () => {
fake.serverEmit('agent:tool:end', {
conversationId: 'c1',
toolCallId: 't1',
toolName: 'web_search',
isError: false,
});
fake.serverEmit('agent:end', {
conversationId: 'c1',
usage: {
provider: 'anthropic',
modelId: 'claude',
thinkingLevel: 'medium',
tokens: { input: 12, output: 34, cacheRead: 0, cacheWrite: 0, total: 46 },
cost: 0.02,
context: { percent: 3, window: 200000 },
},
});
});
expect(container.textContent).toMatch(/success/i);
expect(container.textContent).toContain('Hello!');
expect(container.textContent).toMatch(/46/);
});
it('renders the commands manifest and session info, and lets the user pick a thinking level', async () => {
await act(async () => {
fake.serverEmit('message:ack', { conversationId: 'c1', messageId: 'm1' });
fake.serverEmit('commands:manifest', {
manifest: {
commands: [
{
name: 'model',
aliases: ['m'],
description: 'Change the active model',
scope: 'core',
execution: 'socket',
available: true,
},
],
skills: [],
version: 1,
},
});
fake.serverEmit('session:info', {
conversationId: 'c1',
provider: 'anthropic',
modelId: 'claude',
thinkingLevel: 'medium',
availableThinkingLevels: ['low', 'medium', 'high'],
routingDecision: {
model: 'claude',
provider: 'anthropic',
ruleName: 'default',
reason: 'default routing',
},
});
});
expect(container.textContent).toContain('model');
expect(container.textContent).toContain('Change the active model');
expect(container.textContent).toContain('anthropic');
expect(container.textContent).toContain('default routing');
const select = container.querySelector(
'select[aria-label="Thinking level"]',
) as HTMLSelectElement;
expect(select).toBeTruthy();
expect([...select.options].map((o) => o.value)).toEqual(['low', 'medium', 'high']);
await act(async () => {
selectValue(select, 'high');
});
expect(fake.emitted).toContainEqual({
event: 'set:thinking',
payload: { conversationId: 'c1', level: 'high' },
});
});
it('executes and approves commands with exact payloads and surfaces the approval affordance', async () => {
await act(async () => {
fake.serverEmit('message:ack', { conversationId: 'c1', messageId: 'm1' });
});
const commandInput = container.querySelector(
'input[aria-label="Command name"]',
) as HTMLInputElement;
const argsInput = container.querySelector(
'input[aria-label="Command arguments"]',
) as HTMLInputElement;
await act(async () => {
setValue(commandInput, 'model');
setValue(argsInput, 'gpt-5');
});
await act(async () => {
findButton(container, 'Run command').click();
});
expect(fake.emitted).toContainEqual({
event: 'command:execute',
payload: { conversationId: 'c1', command: 'model', args: 'gpt-5' },
});
await act(async () => {
setValue(commandInput, 'deploy');
setValue(argsInput, 'prod');
});
await act(async () => {
findButton(container, 'Request approval').click();
});
expect(fake.emitted).toContainEqual({
event: 'command:approve',
payload: { conversationId: 'c1', command: 'deploy', args: 'prod' },
});
await act(async () => {
fake.serverEmit('command:approval', {
conversationId: 'c1',
command: 'deploy',
success: true,
approvalId: 'ap1',
expiresAt: '2026-01-01T00:00:00.000Z',
});
});
expect(container.textContent).toMatch(/approved/i);
await act(async () => {
findButton(container, 'Run approved command').click();
});
expect(fake.emitted).toContainEqual({
event: 'command:execute',
payload: { conversationId: 'c1', command: 'deploy', args: 'prod', approvalId: 'ap1' },
});
});
it('shows visible alert surfaces for a server error and the structured contract reason for a failed command result', async () => {
await act(async () => {
fake.serverEmit('message:ack', { conversationId: 'c1', messageId: 'm1' });
fake.serverEmit('error', { conversationId: 'c1', error: 'The model is unavailable' });
fake.serverEmit('command:result', {
conversationId: 'c1',
command: 'model',
success: false,
message: 'Unknown model',
});
});
const alerts = [...container.querySelectorAll('[role="alert"]')];
const alertText = alerts.map((node) => node.textContent).join(' ');
expect(alertText).toContain('The model is unavailable');
// The structured, contract-provided denial reason is visibly rendered.
expect(alertText).toContain('Unknown model');
});
it('falls back to a stable "Command failed." copy when a failed command result has no usable message', async () => {
await act(async () => {
fake.serverEmit('message:ack', { conversationId: 'c1', messageId: 'm1' });
fake.serverEmitRaw('command:result', {
conversationId: 'c1',
command: 'model',
success: false,
message: { bad: 'object' },
});
});
const alerts = [...container.querySelectorAll('[role="alert"]')];
const alertText = alerts.map((node) => node.textContent).join(' ');
expect(alertText).toContain('Command failed.');
});
it('caps availableThinkingLevels before storing and rendering a hostile session payload', async () => {
const hostileLevels = Array.from({ length: MAX_MANIFEST_ITEMS + 50 }, (_, i) => `level-${i}`);
await act(async () => {
fake.serverEmit('message:ack', { conversationId: 'c1', messageId: 'm1' });
fake.serverEmit('session:info', {
conversationId: 'c1',
provider: 'anthropic',
modelId: 'claude',
thinkingLevel: 'level-0',
availableThinkingLevels: hostileLevels,
});
});
const select = container.querySelector(
'select[aria-label="Thinking level"]',
) as HTMLSelectElement;
expect(select).toBeTruthy();
expect(select.options.length).toBeLessThanOrEqual(MAX_MANIFEST_ITEMS);
});
it('renders a safe fallback when session:info arrives with a malformed (non-array) availableThinkingLevels, without throwing', async () => {
await act(async () => {
fake.serverEmit('message:ack', { conversationId: 'c1', messageId: 'm1' });
fake.serverEmitRaw('session:info', {
conversationId: 'c1',
provider: 'anthropic',
modelId: 'claude',
thinkingLevel: 'medium',
availableThinkingLevels: null,
});
});
expect(container.querySelector('section[aria-label="Session info"]')).toBeTruthy();
const select = container.querySelector(
'select[aria-label="Thinking level"]',
) as HTMLSelectElement;
expect(select).toBeTruthy();
// A malformed level list still shows a visible, safe placeholder option
// rather than a silently empty select.
expect([...select.options]).toHaveLength(1);
expect(select.options[0]?.textContent).toMatch(/unavailable/i);
await act(async () => {
selectValue(select, '');
});
expect(fake.emitted.filter((e) => e.event === 'set:thinking')).toHaveLength(0);
});
it('renders honest unavailable labels — not fabricated zeros — when agent:end usage has malformed/missing numeric fields', async () => {
await act(async () => {
fake.serverEmit('message:ack', { conversationId: 'c1', messageId: 'm1' });
fake.serverEmit('agent:start', { conversationId: 'c1' });
fake.serverEmitRaw('agent:end', {
conversationId: 'c1',
usage: {
provider: { nested: 'object' },
modelId: undefined,
thinkingLevel: 'medium',
tokens: { total: 'not-a-number' },
cost: undefined,
context: { percent: null, window: 200000 },
},
});
});
const usage = container.querySelector('[aria-label="Usage"]');
expect(usage).toBeTruthy();
expect(usage?.textContent).toContain('tokens unavailable');
expect(usage?.textContent).toContain('cost unavailable');
expect(usage?.textContent).not.toContain('0 tokens');
expect(usage?.textContent).not.toContain('$0.0000');
expect(usage?.textContent).toContain('unknown/unknown');
});
it('renders a safe fallback for message:ack when messageId is a malformed non-string value, without throwing', async () => {
await expect(
act(async () => {
fake.serverEmitRaw('message:ack', { conversationId: 'c1', messageId: { bad: 'object' } });
}),
).resolves.not.toThrow();
const status = [...container.querySelectorAll('[role="status"]')].find((node) =>
node.textContent?.includes('Message accepted'),
);
expect(status).toBeTruthy();
// A malformed messageId gets a stable, visible fallback — never blank,
// never the raw object.
expect(status?.textContent).toContain('unknown');
});
it('renders safely and does not throw when system:reload.message is a malformed non-string value', async () => {
await expect(
act(async () => {
fake.serverEmitRaw('system:reload', {
commands: [],
skills: [],
providers: [],
message: { bad: 'object' },
});
}),
).resolves.not.toThrow();
const status = container.querySelector('[role="status"]');
expect(status).toBeTruthy();
// A malformed reload message renders a stable, visible fallback rather
// than a silently empty status line.
expect(status?.textContent).toContain('Commands reloaded.');
});
it('renders safely and does not throw when a scoped error carries a malformed non-string error value', async () => {
await expect(
act(async () => {
fake.serverEmit('message:ack', { conversationId: 'c1', messageId: 'm1' });
fake.serverEmitRaw('error', { conversationId: 'c1', error: ['not', 'a', 'string'] });
}),
).resolves.not.toThrow();
expect(container.querySelector('[role="alert"]')).toBeTruthy();
});
it('renders harness and provider as separate selects (not merged) and no free-text provider/model inputs', async () => {
// The old free-text inputs are gone.
expect(container.querySelector('input[aria-label="Provider"]')).toBeNull();
expect(container.querySelector('input[aria-label="Model"]')).toBeNull();
const harnessSelect = container.querySelector(
'select[aria-label="Harness"]',
) as HTMLSelectElement;
const providerSelect = container.querySelector(
'select[aria-label="Provider"]',
) as HTMLSelectElement;
const modelSelect = container.querySelector('select[aria-label="Model"]') as HTMLSelectElement;
expect(harnessSelect).toBeTruthy();
expect(providerSelect).toBeTruthy();
expect(modelSelect).toBeTruthy();
// Harness and provider are distinct controls carrying distinct identifiers.
expect(harnessSelect).not.toBe(providerSelect);
expect([...harnessSelect.options].map((o) => o.value)).toContain('pi');
expect([...providerSelect.options].map((o) => o.value)).toContain('openai');
expect([...providerSelect.options].map((o) => o.value)).toContain('anthropic');
// The model options are catalog-derived (not hardcoded) and scoped to the
// selected provider (openai, from the persisted tuple) using a collision-safe
// composite identity — the anthropic row is absent, not a bare 'claude'.
const modelValues = [...modelSelect.options].map((o) => o.value);
expect(modelValues).toContain('openai:gpt-5');
expect(modelValues).not.toContain('anthropic:claude');
expect(modelValues).not.toContain('claude');
});
it('sends provider/model derived from the persisted catalog tuple (never free text) and emits abort from Stop', async () => {
const textarea = container.querySelector(
'textarea[aria-label="Message"]',
) as HTMLTextAreaElement;
const stopButtonBefore = container.querySelector(
'button[aria-label="Stop"]',
) as HTMLButtonElement;
expect(stopButtonBefore.disabled).toBe(true);
// Choose a fresh tuple from the catalog and let it persist.
const providerSelect = container.querySelector(
'select[aria-label="Provider"]',
) as HTMLSelectElement;
await act(async () => {
selectValue(providerSelect, 'anthropic');
});
const modelSelect = container.querySelector('select[aria-label="Model"]') as HTMLSelectElement;
await act(async () => {
// Composite provider+model option identity (provider was switched to
// anthropic above); the bare 'claude' no longer identifies an option.
selectValue(modelSelect, 'anthropic:claude');
});
await flushAsync();
await act(async () => {
setValue(textarea, 'hello there');
});
await act(async () => {
textarea.dispatchEvent(
new KeyboardEvent('keydown', { key: 'Enter', bubbles: true, cancelable: true }),
);
});
// The projected provider/model come from the validated persisted tuple.
expect(fake.emitted).toContainEqual({
event: 'message',
payload: {
conversationId: undefined,
content: 'hello there',
provider: 'anthropic',
modelId: 'claude',
},
});
expect(container.textContent).toContain('hello there');
await act(async () => {
fake.serverEmit('message:ack', { conversationId: 'c1', messageId: 'm1' });
fake.serverEmit('agent:start', { conversationId: 'c1' });
});
const stopButtonDuring = container.querySelector(
'button[aria-label="Stop"]',
) as HTMLButtonElement;
expect(stopButtonDuring.disabled).toBe(false);
await act(async () => {
stopButtonDuring.click();
});
expect(fake.emitted).toContainEqual({ event: 'abort', payload: { conversationId: 'c1' } });
});
it('disables send until a selection has persisted — no send with an unset selection', async () => {
await remountWithFetch(harnessFetch(null));
const sendButton = findButton(container, 'Send');
const textarea = container.querySelector(
'textarea[aria-label="Message"]',
) as HTMLTextAreaElement;
await act(async () => {
setValue(textarea, 'should not send');
});
// Content present, but no selection persisted → Send stays disabled.
expect(sendButton.disabled).toBe(true);
await act(async () => {
textarea.dispatchEvent(
new KeyboardEvent('keydown', { key: 'Enter', bubbles: true, cancelable: true }),
);
});
expect(fake.emitted.filter((e) => e.event === 'message')).toHaveLength(0);
});
it('renders the session panel from a pre-ack session:info and keeps it visible after the later ack', async () => {
const textarea = container.querySelector(
'textarea[aria-label="Message"]',
) as HTMLTextAreaElement;
await act(async () => {
setValue(textarea, 'hello');
});
await act(async () => {
textarea.dispatchEvent(
new KeyboardEvent('keydown', { key: 'Enter', bubbles: true, cancelable: true }),
);
});
await act(async () => {
fake.serverEmit('session:info', {
conversationId: 'c1',
provider: 'anthropic',
modelId: 'claude',
thinkingLevel: 'medium',
availableThinkingLevels: ['low', 'medium', 'high'],
});
});
expect(container.querySelector('section[aria-label="Session info"]')).toBeTruthy();
expect(container.textContent).toContain('anthropic');
await act(async () => {
fake.serverEmit('message:ack', { conversationId: 'c1', messageId: 'm1' });
});
expect(container.querySelector('section[aria-label="Session info"]')).toBeTruthy();
expect(container.textContent).toContain('anthropic');
});
it('surfaces a pre-ack error as an alert without leaving the Stop control stuck active', async () => {
const textarea = container.querySelector(
'textarea[aria-label="Message"]',
) as HTMLTextAreaElement;
await act(async () => {
setValue(textarea, 'hello');
});
await act(async () => {
textarea.dispatchEvent(
new KeyboardEvent('keydown', { key: 'Enter', bubbles: true, cancelable: true }),
);
});
await act(async () => {
fake.serverEmit('error', {
conversationId: 'c1',
error: 'Failed to start agent session. Please try again.',
});
});
const alerts = [...container.querySelectorAll('[role="alert"]')];
expect(alerts.some((node) => node.textContent?.includes('Failed to start agent session'))).toBe(
true,
);
const stopButton = container.querySelector('button[aria-label="Stop"]') as HTMLButtonElement;
expect(stopButton.disabled).toBe(true);
});
it('shows an accessible status once the message is acknowledged', async () => {
await act(async () => {
fake.serverEmit('message:ack', { conversationId: 'c1', messageId: 'm1' });
});
const statuses = [...container.querySelectorAll('[role="status"]')];
expect(statuses.some((node) => node.textContent?.includes('m1'))).toBe(true);
});
it('renders finalized thinking text in the transcript after agent:end, not only while streaming', async () => {
await act(async () => {
fake.serverEmit('message:ack', { conversationId: 'c1', messageId: 'm1' });
fake.serverEmit('agent:start', { conversationId: 'c1' });
fake.serverEmit('agent:thinking', { conversationId: 'c1', text: 'reasoning about it' });
fake.serverEmit('agent:text', { conversationId: 'c1', text: 'Done.' });
});
expect(container.textContent).toContain('reasoning about it');
await act(async () => {
fake.serverEmit('agent:end', { conversationId: 'c1' });
});
expect(container.textContent).toContain('reasoning about it');
expect(container.textContent).toContain('Done.');
});
it('ignores a concurrent approval request and only executes the approved command once', async () => {
await act(async () => {
fake.serverEmit('message:ack', { conversationId: 'c1', messageId: 'm1' });
});
const commandInput = container.querySelector(
'input[aria-label="Command name"]',
) as HTMLInputElement;
const argsInput = container.querySelector(
'input[aria-label="Command arguments"]',
) as HTMLInputElement;
await act(async () => {
setValue(commandInput, 'deploy');
setValue(argsInput, 'prod');
});
await act(async () => {
findButton(container, 'Request approval').click();
});
await act(async () => {
setValue(argsInput, 'staging');
});
await act(async () => {
findButton(container, 'Request approval').click();
});
expect(fake.emitted.filter((e) => e.event === 'command:approve')).toHaveLength(1);
expect(fake.emitted).toContainEqual({
event: 'command:approve',
payload: { conversationId: 'c1', command: 'deploy', args: 'prod' },
});
await act(async () => {
fake.serverEmit('command:approval', {
conversationId: 'c1',
command: 'deploy',
success: true,
approvalId: 'ap1',
expiresAt: '2026-01-01T00:00:00.000Z',
});
});
await act(async () => {
findButton(container, 'Run approved command').click();
findButton(container, 'Run approved command').click();
});
expect(fake.emitted.filter((e) => e.event === 'command:execute')).toHaveLength(1);
expect(fake.emitted).toContainEqual({
event: 'command:execute',
payload: { conversationId: 'c1', command: 'deploy', args: 'prod', approvalId: 'ap1' },
});
});
it('disables sending a second message while a turn is streaming', async () => {
const textarea = container.querySelector(
'textarea[aria-label="Message"]',
) as HTMLTextAreaElement;
await act(async () => {
setValue(textarea, 'first');
});
await act(async () => {
textarea.dispatchEvent(
new KeyboardEvent('keydown', { key: 'Enter', bubbles: true, cancelable: true }),
);
});
await act(async () => {
fake.serverEmit('message:ack', { conversationId: 'c1', messageId: 'm1' });
fake.serverEmit('agent:start', { conversationId: 'c1' });
});
const sendButton = findButton(container, 'Send');
expect(sendButton.disabled).toBe(true);
await act(async () => {
setValue(textarea, 'second');
});
await act(async () => {
textarea.dispatchEvent(
new KeyboardEvent('keydown', { key: 'Enter', bubbles: true, cancelable: true }),
);
});
expect(fake.emitted.filter((e) => e.event === 'message')).toHaveLength(1);
});
it('scopes the model options to the intentionally selected provider (cross-provider models absent)', async () => {
await remountWithFetch(harnessFetch(null));
const harnessSelect = container.querySelector(
'select[aria-label="Harness"]',
) as HTMLSelectElement;
await act(async () => {
selectValue(harnessSelect, 'pi');
});
await flushAsync();
const providerSelect = container.querySelector(
'select[aria-label="Provider"]',
) as HTMLSelectElement;
await act(async () => {
selectValue(providerSelect, 'openai');
});
const modelSelect = container.querySelector('select[aria-label="Model"]') as HTMLSelectElement;
const optionValues = [...modelSelect.options].map((o) => o.value).filter((v) => v !== '');
// Only the selected provider's models are offered — provider B's model
// (anthropic:claude) is absent, so a user cannot pick across providers.
expect(optionValues).toEqual(['openai:gpt-5']);
expect(optionValues).not.toContain('anthropic:claude');
});
it('keeps identical modelIds under two providers distinct and resolves the pick to the exact tuple', async () => {
const COLLIDING_CATALOG = {
harnessId: 'pi',
version: '2026-08-11',
fingerprint: 'fp',
models: [
{
harnessId: 'pi',
providerId: 'alpha',
modelId: 'gpt-x',
displayName: 'Alpha GPT-X',
reasoningCapability: true,
inputTypes: ['text'],
authState: 'ready',
availability: 'available',
},
{
harnessId: 'pi',
providerId: 'beta',
modelId: 'gpt-x',
displayName: 'Beta GPT-X',
reasoningCapability: true,
inputTypes: ['text'],
authState: 'ready',
availability: 'available',
},
],
};
const collidingFetch = vi.fn(async (input: unknown, init?: RequestInit) => {
const url = String(input);
const method = String(init?.method ?? 'GET').toUpperCase();
if (url === '/api/harnesses') {
return jsonResponse([{ id: 'pi', displayName: 'Pi', capabilities: [] }]);
}
if (url.startsWith('/api/harnesses/') && url.endsWith('/catalog')) {
return jsonResponse(COLLIDING_CATALOG);
}
if (url === '/api/chat/preferences/selection' && method === 'GET') {
return jsonResponse({ selection: null });
}
if (url === '/api/chat/preferences/selection' && method === 'PUT') {
return jsonResponse({ selection: JSON.parse(String(init?.body)) });
}
return new Response('not found', { status: 404 });
}) as unknown as typeof fetch;
await remountWithFetch(collidingFetch);
const harnessSelect = container.querySelector(
'select[aria-label="Harness"]',
) as HTMLSelectElement;
await act(async () => {
selectValue(harnessSelect, 'pi');
});
await flushAsync();
const providerSelect = container.querySelector(
'select[aria-label="Provider"]',
) as HTMLSelectElement;
await act(async () => {
selectValue(providerSelect, 'alpha');
});
const modelSelect = container.querySelector('select[aria-label="Model"]') as HTMLSelectElement;
// The colliding modelId is provider-qualified in the option value, never a
// bare id, so the two providers' 'gpt-x' rows are uniquely identifiable.
const optionValues = [...modelSelect.options].map((o) => o.value).filter((v) => v !== '');
expect(optionValues).toEqual(['alpha:gpt-x']);
await act(async () => {
selectValue(modelSelect, 'alpha:gpt-x');
});
await flushAsync();
// The controlled select highlights the alpha row via the composite identity.
expect(modelSelect.value).toBe('alpha:gpt-x');
const textarea = container.querySelector(
'textarea[aria-label="Message"]',
) as HTMLTextAreaElement;
await act(async () => {
setValue(textarea, 'ping');
});
await act(async () => {
textarea.dispatchEvent(
new KeyboardEvent('keydown', { key: 'Enter', bubbles: true, cancelable: true }),
);
});
// The persisted/sent tuple resolves to provider alpha — NOT beta — even
// though the bare modelId 'gpt-x' exists under both providers.
expect(fake.emitted).toContainEqual({
event: 'message',
payload: {
conversationId: undefined,
content: 'ping',
provider: 'alpha',
modelId: 'gpt-x',
},
});
});
it('removes socket handlers and tears down the socket on unmount, with no network calls', async () => {
expect(fake.listeners.size).toBeGreaterThan(0);
await act(async () => {
root?.unmount();
});
root = null;
for (const [, handlers] of fake.listeners) {
expect(handlers.size).toBe(0);
}
expect(destroySocketMock).toHaveBeenCalledOnce();
});
});
-95
View File
@@ -1,95 +0,0 @@
import type { ReactElement } from 'react';
import { CommandsPanel } from '@/spa/chat/commands-panel';
import { Composer } from '@/spa/chat/composer';
import { MessageTranscript } from '@/spa/chat/message-transcript';
import { asFiniteNumberOrNull, asString } from '@/spa/chat/runtime-guards';
import { SessionPanel } from '@/spa/chat/session-panel';
import { ToolCallList } from '@/spa/chat/tool-call-list';
import { useChatConnection } from '@/spa/chat/use-chat-connection';
import { useHarnessSelection } from '@/spa/chat/use-harness-selection';
/** Renders a real value normally, but an honest "unavailable" label instead
* of a fabricated `0` for a missing/malformed count — a real `0 tokens` and
* an unknown token count must never look the same. */
function formatTokens(value: unknown): string {
const tokens = asFiniteNumberOrNull(value);
return tokens === null ? 'tokens unavailable' : `${tokens} tokens`;
}
/** Same honesty guarantee as `formatTokens`, for cost. */
function formatCost(value: unknown): string {
const cost = asFiniteNumberOrNull(value);
return cost === null ? 'cost unavailable' : `$${cost.toFixed(4)}`;
}
export function ChatPage(): ReactElement {
const { state, actions } = useChatConnection();
const harness = useHarnessSelection();
const hasConversation = state.conversationId !== null;
return (
<div className="flex h-[calc(100vh-3.5rem)] min-h-0 flex-col overflow-hidden md:h-screen">
<header className="border-b px-4 py-3">
<h1 className="text-lg font-semibold">Chat</h1>
</header>
{state.systemReload ? (
<div role="status" className="border-b px-4 py-2 text-sm">
{asString(state.systemReload.message)}
</div>
) : null}
{state.error ? (
<div role="alert" className="border-b px-4 py-2 text-sm">
{asString(state.error)}
</div>
) : null}
{state.ack ? (
<div role="status" className="border-b px-4 py-1 text-xs opacity-70">
Message accepted · conversation {asString(state.ack.conversationId, 'unknown')} · id{' '}
{asString(state.ack.messageId, 'unknown')}
</div>
) : null}
<SessionPanel sessionInfo={state.sessionInfo} onSetThinking={actions.setThinking} />
<MessageTranscript messages={state.messages} streaming={state.streaming} text={state.text} />
{state.thinking ? (
<section aria-label="Thinking" className="px-4 pb-2 text-xs italic opacity-80">
{state.thinking}
</section>
) : null}
<ToolCallList tools={state.tools} />
{state.usage ? (
<div aria-label="Usage" className="px-4 pb-2 text-xs opacity-80">
{formatTokens(state.usage.tokens?.total)} · {formatCost(state.usage.cost)} ·{' '}
{asString(state.usage.provider, 'unknown')}/{asString(state.usage.modelId, 'unknown')}
</div>
) : null}
<CommandsPanel
manifest={state.manifest}
results={state.commandResults}
approval={state.approval}
pendingApproval={state.pendingApproval}
hasConversation={hasConversation}
onExecute={actions.executeCommand}
onApprove={actions.approveCommand}
onRunApproved={actions.runApprovedCommand}
/>
<Composer
onSend={actions.sendMessage}
onStop={actions.abort}
streaming={state.streaming}
sending={state.sending}
hasConversation={hasConversation}
harness={harness}
/>
</div>
);
}
-7
View File
@@ -1,7 +0,0 @@
export function getErrorMessage(error: unknown, fallback: string): string {
if (error instanceof Error && error.message.trim().length > 0) {
return error.message;
}
return fallback;
}
-115
View File
@@ -1,115 +0,0 @@
import type { Mission, Project, Task } from '@/lib/types';
export const projectFixtures: Project[] = [
{
id: 'project-1',
name: 'Mosaic Stack',
description: 'Gateway and dashboard parity work',
status: 'active',
userId: 'user-1',
metadata: {
prd: '# Mosaic Stack PRD\n\n## Objective\n\nShip the SPA route parity pages.',
},
createdAt: '2026-08-01T12:00:00.000Z',
updatedAt: '2026-08-09T18:30:00.000Z',
},
{
id: 'project-2',
name: 'Agent Runtime',
description: 'Pi SDK integration',
status: 'paused',
userId: 'user-1',
metadata: null,
createdAt: '2026-08-02T08:00:00.000Z',
updatedAt: '2026-08-05T10:00:00.000Z',
},
];
export const missionFixtures: Mission[] = [
{
id: 'mission-1',
name: 'Ship web parity',
description: 'Port read-only routes into the SPA',
status: 'active',
projectId: 'project-1',
metadata: null,
createdAt: '2026-08-03T12:00:00.000Z',
updatedAt: '2026-08-09T12:00:00.000Z',
},
{
id: 'mission-2',
name: 'Unrelated mission',
description: 'Must be filtered out of the project detail view',
status: 'planning',
projectId: 'project-2',
metadata: null,
createdAt: '2026-08-04T12:00:00.000Z',
updatedAt: '2026-08-04T12:00:00.000Z',
},
];
export const taskFixtures: Task[] = [
{
id: 'task-1',
title: 'Route /projects',
description: 'Port the read-only projects listing into the SPA',
status: 'done',
priority: 'high',
projectId: 'project-1',
missionId: 'mission-1',
assignee: 'Jarvis',
tags: ['spa', 'projects'],
dueDate: '2026-08-12T00:00:00.000Z',
metadata: {
notes: 'Read-only modal content should remain intact.',
pr_links: [{ url: 'https://example.invalid/pr/1', label: 'PR #1' }],
},
createdAt: '2026-08-04T10:00:00.000Z',
updatedAt: '2026-08-09T15:00:00.000Z',
},
{
id: 'task-2',
title: 'Route /projects/:id',
description: 'Reuse overview, tasks, missions, and PRD tabs',
status: 'in-progress',
priority: 'critical',
projectId: 'project-1',
missionId: 'mission-1',
assignee: null,
tags: ['spa', 'detail'],
dueDate: null,
metadata: null,
createdAt: '2026-08-05T09:00:00.000Z',
updatedAt: '2026-08-10T08:00:00.000Z',
},
{
id: 'task-3',
title: 'Route /tasks',
description: 'Wire list and kanban modal interactions',
status: 'blocked',
priority: 'medium',
projectId: 'project-1',
missionId: null,
assignee: null,
tags: ['spa', 'tasks'],
dueDate: null,
metadata: null,
createdAt: '2026-08-06T09:00:00.000Z',
updatedAt: '2026-08-08T08:00:00.000Z',
},
{
id: 'task-4',
title: 'Other project task',
description: 'Used only to confirm mission filtering remains project-scoped',
status: 'not-started',
priority: 'low',
projectId: 'project-2',
missionId: 'mission-2',
assignee: null,
tags: null,
dueDate: null,
metadata: null,
createdAt: '2026-08-06T09:00:00.000Z',
updatedAt: '2026-08-06T09:00:00.000Z',
},
];
@@ -1,174 +0,0 @@
import { act } from 'react';
import { createRoot, type Root } from 'react-dom/client';
import { createMemoryRouter, RouterProvider, type RouteObject } from 'react-router-dom';
import { afterAll, afterEach, beforeAll, describe, expect, it, vi } from 'vitest';
import { missionFixtures, projectFixtures, taskFixtures } from './page-fixtures';
const { apiMock } = vi.hoisted(() => ({
apiMock: vi.fn(),
}));
vi.mock('@/lib/api', () => ({
api: apiMock,
}));
import { ProjectDetailPage } from './project-detail';
let root: Root | null = null;
let container: HTMLDivElement;
beforeAll(() => {
Object.defineProperty(globalThis, 'IS_REACT_ACT_ENVIRONMENT', {
configurable: true,
value: true,
});
});
afterAll(() => {
Reflect.deleteProperty(globalThis, 'IS_REACT_ACT_ENVIRONMENT');
});
afterEach(async () => {
await act(async () => {
root?.unmount();
});
document.body.replaceChildren();
root = null;
apiMock.mockReset();
});
async function renderProjectDetailPage(): Promise<ReturnType<typeof createMemoryRouter>> {
const routes: RouteObject[] = [
{ path: '/projects', element: <p>Projects index target</p> },
{ path: '/projects/:id', element: <ProjectDetailPage /> },
];
const router = createMemoryRouter(routes, { initialEntries: ['/projects/project-1'] });
container = document.createElement('div');
document.body.append(container);
root = createRoot(container);
await act(async () => {
root?.render(<RouterProvider router={router} />);
});
return router;
}
function clickButtonByText(text: string): void {
const button = [...container.querySelectorAll('button')].find((candidate) =>
candidate.textContent?.includes(text),
);
if (!button) {
throw new Error(`Button containing "${text}" not found`);
}
button.dispatchEvent(new MouseEvent('click', { bubbles: true }));
}
describe('ProjectDetailPage', () => {
it('loads the project, tasks, missions, and optional PRD content for the active project', async () => {
apiMock
.mockResolvedValueOnce(projectFixtures[0])
.mockResolvedValueOnce(missionFixtures)
.mockResolvedValueOnce(taskFixtures.filter((task) => task.projectId === 'project-1'));
await renderProjectDetailPage();
expect(apiMock.mock.calls).toEqual([
['/api/projects/project-1'],
['/api/missions'],
['/api/tasks?projectId=project-1'],
]);
expect(container.textContent).toContain('Mosaic Stack');
expect(container.textContent).toContain('Route /projects/:id');
expect(container.textContent).toContain('Tasks');
expect(container.textContent).toContain('Done');
expect(container.textContent).toContain('Blocked');
await act(async () => {
clickButtonByText('Missions (1)');
});
expect(container.textContent).toContain('Ship web parity');
expect(container.textContent).not.toContain('Unrelated mission');
await act(async () => {
clickButtonByText('PRD');
});
expect(container.textContent).toContain('Mosaic Stack PRD');
expect(container.textContent).toContain('Ship the SPA route parity pages.');
});
it('opens and closes the existing read-only task modal from the tasks tab', async () => {
apiMock
.mockResolvedValueOnce(projectFixtures[0])
.mockResolvedValueOnce(missionFixtures)
.mockResolvedValueOnce(taskFixtures.filter((task) => task.projectId === 'project-1'));
await renderProjectDetailPage();
await act(async () => {
clickButtonByText('Tasks (3)');
});
const row = [...container.querySelectorAll('tr')].find((candidate) =>
candidate.textContent?.includes('Route /projects'),
);
expect(row).toBeTruthy();
await act(async () => {
row?.dispatchEvent(new MouseEvent('click', { bubbles: true }));
});
expect(container.querySelector('[role="dialog"]')).toBeTruthy();
expect(container.textContent).toContain('Read-only modal content should remain intact.');
const closeButton = container.querySelector('button[aria-label="Close task details"]');
expect(closeButton).toBeTruthy();
await act(async () => {
closeButton?.dispatchEvent(new MouseEvent('click', { bubbles: true }));
});
expect(container.querySelector('[role="dialog"]')).toBeNull();
});
it('renders the project with an empty missions tab when the missions request fails', async () => {
apiMock
.mockResolvedValueOnce(projectFixtures[0])
.mockRejectedValueOnce(new Error('Missions request failed'))
.mockResolvedValueOnce(taskFixtures.filter((task) => task.projectId === 'project-1'));
await renderProjectDetailPage();
expect(container.textContent).toContain('Mosaic Stack');
expect(container.querySelector('[role="alert"]')).toBeNull();
await act(async () => {
clickButtonByText('Missions (0)');
});
expect(container.textContent).toContain('No missions for this project');
});
it('renders a visible alert when the project request fails and lets the user navigate back', async () => {
apiMock
.mockRejectedValueOnce(new Error('Project request failed'))
.mockResolvedValueOnce(missionFixtures)
.mockResolvedValueOnce(taskFixtures.filter((task) => task.projectId === 'project-1'));
const router = await renderProjectDetailPage();
const alert = container.querySelector('[role="alert"]');
expect(alert).toBeTruthy();
expect(alert?.textContent).toContain('Project request failed');
expect(container.textContent).not.toContain('Mosaic Stack');
await act(async () => {
clickButtonByText('Back to projects');
});
expect(router.state.location.pathname).toBe('/projects');
});
});
-344
View File
@@ -1,344 +0,0 @@
import { useEffect, useState, type ReactElement } from 'react';
import { useNavigate, useParams } from 'react-router-dom';
import { MissionTimeline } from '@/components/projects/mission-timeline';
import { PrdViewer } from '@/components/projects/prd-viewer';
import { TaskDetailModal } from '@/components/tasks/task-detail-modal';
import { TaskListView } from '@/components/tasks/task-list-view';
import { TaskStatusSummary } from '@/components/tasks/task-status-summary';
import { api } from '@/lib/api';
import { cn } from '@/lib/cn';
import type { Mission, Project, Task, TaskStatus } from '@/lib/types';
import { getErrorMessage } from './page-errors';
type Tab = 'overview' | 'tasks' | 'missions' | 'prd';
const projectStatusColors: Record<string, string> = {
active: 'bg-success/20 text-success',
paused: 'bg-warning/20 text-warning',
completed: 'bg-blue-600/20 text-blue-400',
archived: 'bg-gray-600/20 text-gray-400',
};
const taskStatusColors: Record<string, string> = {
'not-started': 'bg-gray-600/20 text-gray-300',
'in-progress': 'bg-blue-600/20 text-blue-400',
blocked: 'bg-error/20 text-error',
done: 'bg-success/20 text-success',
cancelled: 'bg-gray-600/20 text-gray-500',
};
interface TabButtonProps {
id: Tab;
label: string;
activeTab: Tab;
onClick: (tab: Tab) => void;
}
function TabButton({ id, label, activeTab, onClick }: TabButtonProps): ReactElement {
return (
<button
type="button"
onClick={() => onClick(id)}
className={cn(
'border-b-2 px-4 py-2 text-sm transition-colors',
activeTab === id
? 'border-text-primary text-text-primary'
: 'border-transparent text-text-muted hover:text-text-secondary',
)}
>
{label}
</button>
);
}
export function ProjectDetailPage(): ReactElement {
const { id = '' } = useParams();
const navigate = useNavigate();
const [project, setProject] = useState<Project | null>(null);
const [missions, setMissions] = useState<Mission[]>([]);
const [tasks, setTasks] = useState<Task[]>([]);
const [loading, setLoading] = useState(true);
const [error, setError] = useState<string | null>(null);
const [activeTab, setActiveTab] = useState<Tab>('overview');
const [taskFilter, setTaskFilter] = useState<TaskStatus | 'all'>('all');
const [selectedTask, setSelectedTask] = useState<Task | null>(null);
useEffect(() => {
if (!id) {
setError('Project id is missing.');
setLoading(false);
return;
}
let cancelled = false;
setLoading(true);
setError(null);
void Promise.all([
api<Project>('/api/projects/' + id),
api<Mission[]>('/api/missions').catch(() => [] as Mission[]),
api<Task[]>('/api/tasks?projectId=' + id).catch(() => [] as Task[]),
])
.then(([loadedProject, allMissions, loadedTasks]) => {
if (cancelled) return;
setProject(loadedProject);
setMissions(allMissions.filter((mission) => mission.projectId === id));
setTasks(loadedTasks);
})
.catch((caught: unknown) => {
if (cancelled) return;
setError(getErrorMessage(caught, 'Failed to load project.'));
})
.finally(() => {
if (cancelled) return;
setLoading(false);
});
return () => {
cancelled = true;
};
}, [id]);
if (loading) {
return (
<div className="flex min-h-screen flex-col px-4 py-6 sm:px-6">
<header className="mb-6 border-b px-1 pb-3">
<h1 className="text-2xl font-semibold">Project</h1>
</header>
<p className="py-16 text-center text-sm text-text-muted">Loading project...</p>
</div>
);
}
if (error || !project) {
return (
<div className="flex min-h-screen flex-col px-4 py-6 sm:px-6">
<header className="mb-6 border-b px-1 pb-3">
<h1 className="text-2xl font-semibold">Project</h1>
</header>
<div role="alert" className="rounded-lg border border-error/40 px-4 py-3 text-sm">
{error ?? 'Project not found.'}
</div>
<button
type="button"
onClick={() => navigate('/projects')}
className="mt-4 w-fit text-sm underline"
>
Back to projects
</button>
</div>
);
}
const filteredTasks =
taskFilter === 'all' ? tasks : tasks.filter((task) => task.status === taskFilter);
const prdContent = getPrdContent(project);
const tabs: Array<{ id: Tab; label: string }> = [
{ id: 'overview', label: 'Overview' },
{ id: 'tasks', label: `Tasks (${tasks.length})` },
{ id: 'missions', label: `Missions (${missions.length})` },
...(prdContent ? [{ id: 'prd' as const, label: 'PRD' }] : []),
];
return (
<div className="flex min-h-screen flex-col px-4 py-6 sm:px-6">
<header className="mb-6 border-b px-1 pb-3">
<nav className="mb-4 flex items-center gap-2 text-sm text-text-muted">
<button
type="button"
onClick={() => navigate('/projects')}
className="hover:text-text-secondary"
>
Projects
</button>
<span>/</span>
<span className="text-text-primary">{project.name}</span>
</nav>
<div className="flex items-start justify-between gap-4">
<div>
<div className="flex items-center gap-3">
<h1 className="text-2xl font-semibold text-text-primary">{project.name}</h1>
<span
className={cn(
'rounded-full px-2 py-0.5 text-xs',
projectStatusColors[project.status] ?? 'bg-gray-600/20 text-gray-400',
)}
>
{project.status}
</span>
</div>
{project.description ? (
<p className="mt-1 text-sm text-text-muted">{project.description}</p>
) : null}
<p className="mt-2 text-xs text-text-muted">
Created {new Date(project.createdAt).toLocaleDateString()} · Updated{' '}
{new Date(project.updatedAt).toLocaleDateString()}
</p>
</div>
</div>
</header>
<div className="mb-6 grid grid-cols-2 gap-3 sm:grid-cols-4">
<StatCard label="Tasks" value={String(tasks.length)} />
<StatCard
label="Done"
value={String(tasks.filter((task) => task.status === 'done').length)}
valueClass="text-success"
/>
<StatCard
label="In Progress"
value={String(tasks.filter((task) => task.status === 'in-progress').length)}
valueClass="text-blue-400"
/>
<StatCard
label="Blocked"
value={String(tasks.filter((task) => task.status === 'blocked').length)}
valueClass={tasks.some((task) => task.status === 'blocked') ? 'text-error' : undefined}
/>
</div>
<div className="mb-6 flex gap-0 border-b border-surface-border">
{tabs.map((tab) => (
<TabButton
key={tab.id}
id={tab.id}
label={tab.label}
activeTab={activeTab}
onClick={setActiveTab}
/>
))}
</div>
{activeTab === 'overview' ? (
<OverviewTab project={project} missions={missions} tasks={tasks} />
) : null}
{activeTab === 'tasks' ? (
<div>
<div className="mb-4">
<TaskStatusSummary
tasks={tasks}
activeFilter={taskFilter}
onFilterChange={setTaskFilter}
/>
</div>
<TaskListView tasks={filteredTasks} onTaskClick={setSelectedTask} />
</div>
) : null}
{activeTab === 'missions' ? <MissionTimeline missions={missions} /> : null}
{activeTab === 'prd' && prdContent ? (
<div className="rounded-lg border border-surface-border bg-surface-card p-6">
<PrdViewer content={prdContent} />
</div>
) : null}
{selectedTask ? (
<TaskDetailModal task={selectedTask} onClose={() => setSelectedTask(null)} />
) : null}
</div>
);
}
function OverviewTab({
project,
missions,
tasks,
}: {
project: Project;
missions: Mission[];
tasks: Task[];
}): ReactElement {
const recentTasks = [...tasks]
.sort((left, right) => new Date(right.updatedAt).getTime() - new Date(left.updatedAt).getTime())
.slice(0, 5);
return (
<div className="grid gap-6 lg:grid-cols-2">
<section>
<h2 className="mb-3 text-sm font-semibold text-text-secondary">Recent Tasks</h2>
{recentTasks.length === 0 ? (
<div className="rounded-lg border border-surface-border bg-surface-card p-4 text-center">
<p className="text-sm text-text-muted">No tasks yet</p>
</div>
) : (
<div className="space-y-2">
{recentTasks.map((task) => (
<div
key={task.id}
className="flex items-center justify-between gap-2 rounded-lg border border-surface-border bg-surface-card px-3 py-2"
>
<span className="truncate text-sm text-text-primary">{task.title}</span>
<span
className={cn(
'shrink-0 rounded-full px-2 py-0.5 text-xs',
taskStatusColors[task.status] ?? 'bg-gray-600/20 text-gray-400',
)}
>
{task.status}
</span>
</div>
))}
</div>
)}
</section>
<section>
<h2 className="mb-3 text-sm font-semibold text-text-secondary">Missions</h2>
{missions.length === 0 ? (
<div className="rounded-lg border border-surface-border bg-surface-card p-4 text-center">
<p className="text-sm text-text-muted">No missions yet</p>
</div>
) : (
<MissionTimeline missions={missions.slice(0, 4)} />
)}
</section>
{project.metadata && Object.keys(project.metadata).length > 0 ? (
<section className="lg:col-span-2">
<h2 className="mb-3 text-sm font-semibold text-text-secondary">Project Metadata</h2>
<div className="rounded-lg border border-surface-border bg-surface-card p-4">
<pre className="overflow-x-auto text-xs text-text-muted">
{JSON.stringify(project.metadata, null, 2)}
</pre>
</div>
</section>
) : null}
</div>
);
}
function StatCard({
label,
value,
valueClass,
}: {
label: string;
value: string;
valueClass?: string;
}): ReactElement {
return (
<div className="rounded-lg border border-surface-border bg-surface-card p-3">
<p className="text-xs text-text-muted">{label}</p>
<p className={cn('mt-1 text-lg font-semibold', valueClass ?? 'text-text-primary')}>{value}</p>
</div>
);
}
function getPrdContent(project: Project): string | null {
if (!project.metadata) return null;
const prd = project.metadata['prd'];
if (typeof prd === 'string' && prd.trim().length > 0) {
return prd;
}
const prdContent = project.metadata['prdContent'];
if (typeof prdContent === 'string' && prdContent.trim().length > 0) {
return prdContent;
}
return null;
}
-131
View File
@@ -1,131 +0,0 @@
import { act } from 'react';
import { createRoot, type Root } from 'react-dom/client';
import { createMemoryRouter, RouterProvider, type RouteObject } from 'react-router-dom';
import { afterAll, afterEach, beforeAll, describe, expect, it, vi } from 'vitest';
import { projectFixtures } from './page-fixtures';
const { apiMock } = vi.hoisted(() => ({
apiMock: vi.fn(),
}));
vi.mock('@/lib/api', () => ({
api: apiMock,
}));
import { ProjectsPage } from './projects';
interface Deferred<T> {
promise: Promise<T>;
resolve: (value: T) => void;
reject: (reason?: unknown) => void;
}
function createDeferred<T>(): Deferred<T> {
let resolve!: (value: T) => void;
let reject!: (reason?: unknown) => void;
const promise = new Promise<T>((res, rej) => {
resolve = res;
reject = rej;
});
return { promise, resolve, reject };
}
let root: Root | null = null;
let container: HTMLDivElement;
beforeAll(() => {
Object.defineProperty(globalThis, 'IS_REACT_ACT_ENVIRONMENT', {
configurable: true,
value: true,
});
});
afterAll(() => {
Reflect.deleteProperty(globalThis, 'IS_REACT_ACT_ENVIRONMENT');
});
afterEach(async () => {
await act(async () => {
root?.unmount();
});
document.body.replaceChildren();
root = null;
apiMock.mockReset();
});
async function renderProjectsPage(): Promise<ReturnType<typeof createMemoryRouter>> {
const routes: RouteObject[] = [
{ path: '/projects', element: <ProjectsPage /> },
{ path: '/projects/:id', element: <p>Project detail target</p> },
];
const router = createMemoryRouter(routes, { initialEntries: ['/projects'] });
container = document.createElement('div');
document.body.append(container);
root = createRoot(container);
await act(async () => {
root?.render(<RouterProvider router={router} />);
});
return router;
}
describe('ProjectsPage', () => {
it('shows a visible loading state while the project request is in flight', async () => {
const deferred = createDeferred<typeof projectFixtures>();
apiMock.mockReturnValueOnce(deferred.promise);
await renderProjectsPage();
expect(container.textContent).toContain('Loading projects...');
await act(async () => {
deferred.resolve(projectFixtures);
await deferred.promise;
});
});
it('renders project cards from the API and navigates to a project detail route on click', async () => {
apiMock.mockResolvedValueOnce(projectFixtures);
const router = await renderProjectsPage();
expect(apiMock).toHaveBeenCalledWith('/api/projects');
expect(container.textContent).toContain('Mosaic Stack');
expect(container.textContent).toContain('Agent Runtime');
const button = [...container.querySelectorAll('button')].find((candidate) =>
candidate.textContent?.includes('Mosaic Stack'),
);
expect(button).toBeTruthy();
await act(async () => {
button?.dispatchEvent(new MouseEvent('click', { bubbles: true }));
});
expect(router.state.location.pathname).toBe('/projects/project-1');
expect(container.textContent).toContain('Project detail target');
});
it('renders the empty state when the API returns no projects', async () => {
apiMock.mockResolvedValueOnce([]);
await renderProjectsPage();
expect(container.textContent).toContain('No projects yet');
expect(container.textContent).toContain(
'Projects will appear here when created via the gateway API',
);
});
it('renders a visible alert when the projects request fails', async () => {
apiMock.mockRejectedValueOnce(new Error('Projects are unavailable'));
await renderProjectsPage();
const alert = container.querySelector('[role="alert"]');
expect(alert).toBeTruthy();
expect(alert?.textContent).toContain('Projects are unavailable');
});
});
-70
View File
@@ -1,70 +0,0 @@
import { useEffect, useState, type ReactElement } from 'react';
import { useNavigate } from 'react-router-dom';
import { ProjectCard } from '@/components/projects/project-card';
import { api } from '@/lib/api';
import type { Project } from '@/lib/types';
import { getErrorMessage } from './page-errors';
export function ProjectsPage(): ReactElement {
const navigate = useNavigate();
const [projects, setProjects] = useState<Project[]>([]);
const [loading, setLoading] = useState(true);
const [error, setError] = useState<string | null>(null);
useEffect(() => {
let cancelled = false;
void api<Project[]>('/api/projects')
.then((response) => {
if (cancelled) return;
setProjects(response);
})
.catch((caught: unknown) => {
if (cancelled) return;
setError(getErrorMessage(caught, 'Failed to load projects.'));
})
.finally(() => {
if (cancelled) return;
setLoading(false);
});
return () => {
cancelled = true;
};
}, []);
return (
<div className="flex min-h-screen flex-col px-4 py-6 sm:px-6">
<header className="mb-6 border-b px-1 pb-3">
<h1 className="text-2xl font-semibold">Projects</h1>
</header>
{error ? (
<div role="alert" className="mb-6 rounded-lg border border-error/40 px-4 py-3 text-sm">
{error}
</div>
) : null}
{loading ? (
<p className="py-8 text-center text-sm text-text-muted">Loading projects...</p>
) : projects.length === 0 ? (
<div className="py-12 text-center">
<h2 className="text-lg font-medium text-text-secondary">No projects yet</h2>
<p className="mt-1 text-sm text-text-muted">
Projects will appear here when created via the gateway API
</p>
</div>
) : (
<div className="grid gap-4 sm:grid-cols-2 lg:grid-cols-3">
{projects.map((project) => (
<ProjectCard
key={project.id}
project={project}
onClick={(selectedProject) => navigate(`/projects/${selectedProject.id}`)}
/>
))}
</div>
)}
</div>
);
}
@@ -1,86 +0,0 @@
import { act } from 'react';
import { createRoot, type Root } from 'react-dom/client';
import { createMemoryRouter, RouterProvider, type RouteObject } from 'react-router-dom';
import { afterAll, afterEach, beforeAll, describe, expect, it, vi } from 'vitest';
const { useSessionMock } = vi.hoisted(() => ({
useSessionMock: vi.fn(),
}));
vi.mock('@/lib/auth-client', () => ({
useSession: useSessionMock,
}));
import { routes } from '@/routes';
function Boom(): never {
throw new Error('resource route render blew up');
}
function replaceRouteElementWithBoom(nodes: RouteObject[], path: string): RouteObject[] {
return nodes.map((node) => {
const cloned: RouteObject = { ...node };
if (cloned.path === path) {
cloned.element = <Boom />;
}
if (cloned.children) {
cloned.children = replaceRouteElementWithBoom(cloned.children, path);
}
return cloned;
});
}
let root: Root | null = null;
let container: HTMLDivElement;
beforeAll(() => {
Object.defineProperty(globalThis, 'IS_REACT_ACT_ENVIRONMENT', {
configurable: true,
value: true,
});
});
afterAll(() => {
Reflect.deleteProperty(globalThis, 'IS_REACT_ACT_ENVIRONMENT');
});
afterEach(async () => {
await act(async () => {
root?.unmount();
});
document.body.replaceChildren();
root = null;
useSessionMock.mockReset();
});
describe('resource route error boundaries', () => {
it.each(['/projects', '/projects/:id', '/tasks'])(
'renders a recoverable fallback when %s throws during route render',
async (path) => {
useSessionMock.mockReturnValue({ data: { user: { id: 'user-1' } }, isPending: false });
const initialEntry = path === '/projects/:id' ? '/projects/project-1' : path;
const router = createMemoryRouter(replaceRouteElementWithBoom(routes, path), {
initialEntries: [initialEntry],
});
container = document.createElement('div');
document.body.append(container);
root = createRoot(container);
const consoleErrorSpy = vi.spyOn(console, 'error').mockImplementation(() => {});
try {
await act(async () => {
root?.render(<RouterProvider router={router} />);
});
expect(consoleErrorSpy).toHaveBeenCalled();
} finally {
consoleErrorSpy.mockRestore();
}
expect(container.textContent).not.toBe('');
expect(container.querySelector('[role="alert"]')).toBeTruthy();
},
);
});
@@ -1,55 +0,0 @@
import type { ReactElement } from 'react';
import { useRouteError } from 'react-router-dom';
interface ResourceRouteErrorBoundaryProps {
message: string;
href: string;
linkLabel: string;
}
function ResourceRouteErrorBoundary({
message,
href,
linkLabel,
}: ResourceRouteErrorBoundaryProps): ReactElement {
useRouteError();
return (
<div role="alert" className="flex min-h-screen flex-col items-center justify-center gap-3 p-8">
<p className="text-sm font-medium">{message}</p>
<a href={href} className="text-sm underline">
{linkLabel}
</a>
</div>
);
}
export function ProjectsRouteErrorBoundary(): ReactElement {
return (
<ResourceRouteErrorBoundary
message="Something went wrong loading projects."
href="/projects"
linkLabel="Reload projects"
/>
);
}
export function ProjectDetailRouteErrorBoundary(): ReactElement {
return (
<ResourceRouteErrorBoundary
message="Something went wrong loading this project."
href="/projects"
linkLabel="Back to projects"
/>
);
}
export function TasksRouteErrorBoundary(): ReactElement {
return (
<ResourceRouteErrorBoundary
message="Something went wrong loading tasks."
href="/tasks"
linkLabel="Reload tasks"
/>
);
}
-144
View File
@@ -1,144 +0,0 @@
import { act } from 'react';
import { createRoot, type Root } from 'react-dom/client';
import { createMemoryRouter, RouterProvider, type RouteObject } from 'react-router-dom';
import { afterAll, afterEach, beforeAll, describe, expect, it, vi } from 'vitest';
import { taskFixtures } from './page-fixtures';
const { apiMock } = vi.hoisted(() => ({
apiMock: vi.fn(),
}));
vi.mock('@/lib/api', () => ({
api: apiMock,
}));
import { TasksPage } from './tasks';
interface Deferred<T> {
promise: Promise<T>;
resolve: (value: T) => void;
}
function createDeferred<T>(): Deferred<T> {
let resolve!: (value: T) => void;
const promise = new Promise<T>((res) => {
resolve = res;
});
return { promise, resolve };
}
let root: Root | null = null;
let container: HTMLDivElement;
beforeAll(() => {
Object.defineProperty(globalThis, 'IS_REACT_ACT_ENVIRONMENT', {
configurable: true,
value: true,
});
});
afterAll(() => {
Reflect.deleteProperty(globalThis, 'IS_REACT_ACT_ENVIRONMENT');
});
afterEach(async () => {
await act(async () => {
root?.unmount();
});
document.body.replaceChildren();
root = null;
apiMock.mockReset();
});
async function renderTasksPage(): Promise<void> {
const routes: RouteObject[] = [{ path: '/tasks', element: <TasksPage /> }];
const router = createMemoryRouter(routes, { initialEntries: ['/tasks'] });
container = document.createElement('div');
document.body.append(container);
root = createRoot(container);
await act(async () => {
root?.render(<RouterProvider router={router} />);
});
}
function clickButtonByText(text: string): void {
const button = [...container.querySelectorAll('button')].find((candidate) =>
candidate.textContent?.includes(text),
);
if (!button) {
throw new Error(`Button containing "${text}" not found`);
}
button.dispatchEvent(new MouseEvent('click', { bubbles: true }));
}
describe('TasksPage', () => {
it('shows a visible loading state before the tasks request settles', async () => {
const deferred = createDeferred<typeof taskFixtures>();
apiMock.mockReturnValueOnce(deferred.promise);
await renderTasksPage();
expect(container.textContent).toContain('Loading tasks...');
await act(async () => {
deferred.resolve(taskFixtures);
await deferred.promise;
});
});
it('starts in kanban view, toggles to list view, and opens the read-only modal from cards and rows', async () => {
apiMock.mockResolvedValueOnce(taskFixtures);
await renderTasksPage();
expect(container.textContent).toContain('Not Started');
expect(container.textContent).toContain('In Progress');
expect(container.textContent).toContain('Blocked');
const kanbanCard = [...container.querySelectorAll('button')].find((candidate) =>
candidate.textContent?.includes('Route /projects/:id'),
);
expect(kanbanCard).toBeTruthy();
await act(async () => {
kanbanCard?.dispatchEvent(new MouseEvent('click', { bubbles: true }));
});
expect(container.querySelector('[role="dialog"]')).toBeTruthy();
await act(async () => {
container
.querySelector('button[aria-label="Close task details"]')
?.dispatchEvent(new MouseEvent('click', { bubbles: true }));
});
expect(container.querySelector('[role="dialog"]')).toBeNull();
await act(async () => {
clickButtonByText('List');
});
const row = [...container.querySelectorAll('tr')].find((candidate) =>
candidate.textContent?.includes('Route /tasks'),
);
expect(row).toBeTruthy();
await act(async () => {
row?.dispatchEvent(new MouseEvent('click', { bubbles: true }));
});
expect(container.querySelector('[role="dialog"]')).toBeTruthy();
expect(container.textContent).toContain('Wire list and kanban modal interactions');
});
it('renders a visible alert when the tasks request fails', async () => {
apiMock.mockRejectedValueOnce(new Error('Tasks request failed'));
await renderTasksPage();
const alert = container.querySelector('[role="alert"]');
expect(alert).toBeTruthy();
expect(alert?.textContent).toContain('Tasks request failed');
});
});
-92
View File
@@ -1,92 +0,0 @@
import { useEffect, useState, type ReactElement } from 'react';
import { KanbanBoard } from '@/components/tasks/kanban-board';
import { TaskDetailModal } from '@/components/tasks/task-detail-modal';
import { TaskListView } from '@/components/tasks/task-list-view';
import { api } from '@/lib/api';
import { cn } from '@/lib/cn';
import type { Task } from '@/lib/types';
import { getErrorMessage } from './page-errors';
type ViewMode = 'list' | 'kanban';
export function TasksPage(): ReactElement {
const [tasks, setTasks] = useState<Task[]>([]);
const [view, setView] = useState<ViewMode>('kanban');
const [loading, setLoading] = useState(true);
const [error, setError] = useState<string | null>(null);
const [selectedTask, setSelectedTask] = useState<Task | null>(null);
useEffect(() => {
let cancelled = false;
void api<Task[]>('/api/tasks')
.then((response) => {
if (cancelled) return;
setTasks(response);
})
.catch((caught: unknown) => {
if (cancelled) return;
setError(getErrorMessage(caught, 'Failed to load tasks.'));
})
.finally(() => {
if (cancelled) return;
setLoading(false);
});
return () => {
cancelled = true;
};
}, []);
return (
<div className="flex min-h-screen flex-col px-4 py-6 sm:px-6">
<header className="mb-6 flex items-center justify-between gap-4 border-b px-1 pb-3">
<h1 className="text-2xl font-semibold">Tasks</h1>
<div className="flex rounded-lg border border-surface-border">
<button
type="button"
onClick={() => setView('list')}
className={cn(
'px-3 py-1.5 text-xs transition-colors',
view === 'list'
? 'bg-surface-elevated text-text-primary'
: 'text-text-muted hover:text-text-secondary',
)}
>
List
</button>
<button
type="button"
onClick={() => setView('kanban')}
className={cn(
'px-3 py-1.5 text-xs transition-colors',
view === 'kanban'
? 'bg-surface-elevated text-text-primary'
: 'text-text-muted hover:text-text-secondary',
)}
>
Kanban
</button>
</div>
</header>
{error ? (
<div role="alert" className="mb-6 rounded-lg border border-error/40 px-4 py-3 text-sm">
{error}
</div>
) : null}
{loading ? (
<p className="py-8 text-center text-sm text-text-muted">Loading tasks...</p>
) : view === 'kanban' ? (
<KanbanBoard tasks={tasks} onTaskClick={setSelectedTask} />
) : (
<TaskListView tasks={tasks} onTaskClick={setSelectedTask} />
)}
{selectedTask ? (
<TaskDetailModal task={selectedTask} onClose={() => setSelectedTask(null)} />
) : null}
</div>
);
}
-39
View File
@@ -3,10 +3,6 @@ import { describe, expect, it } from 'vitest';
import type { RouteObject } from 'react-router-dom';
import { routes } from '@/routes';
import { Placeholder } from '@/spa/placeholder';
import { ChatPage } from '@/spa/pages/chat';
import { ProjectDetailPage } from '@/spa/pages/project-detail';
import { ProjectsPage } from '@/spa/pages/projects';
import { TasksPage } from '@/spa/pages/tasks';
function collectPaths(routeObjects: RouteObject[]): string[] {
return routeObjects.flatMap((route) => [
@@ -59,39 +55,4 @@ describe('SPA route table', () => {
expect(element.type).not.toBe(Placeholder);
},
);
it('renders the real chat page instead of the P1 placeholder at /chat, inside the authenticated group', () => {
const authPaths = collectPaths(routes.at(1)?.children ?? []);
expect(authPaths).toContain('/chat');
const element = findRoute(routes, '/chat')?.element;
expect(isValidElement(element)).toBe(true);
if (!isValidElement(element)) throw new Error('Missing route element for /chat');
expect(element.type).not.toBe(Placeholder);
expect(element.type).toBe(ChatPage);
});
it.each([
['/projects', ProjectsPage],
['/projects/:id', ProjectDetailPage],
['/tasks', TasksPage],
])(
'renders a real authenticated page instead of the P1 placeholder at %s',
(path, expectedType) => {
const element = findRoute(routes, path)?.element;
expect(isValidElement(element)).toBe(true);
if (!isValidElement(element)) throw new Error(`Missing route element for ${path}`);
expect(element.type).not.toBe(Placeholder);
expect(element.type).toBe(expectedType);
},
);
it.each(['/chat', '/projects', '/projects/:id', '/tasks'])(
'defines an error boundary for %s',
(path) => {
const route = findRoute(routes, path);
expect(route?.errorElement).toBeTruthy();
expect(isValidElement(route?.errorElement)).toBe(true);
},
);
});
-1
View File
@@ -16,7 +16,6 @@ export default defineConfig({
},
server: {
port: 3100,
strictPort: true,
proxy: {
'/api': gatewayTarget,
'/socket.io': { target: gatewayTarget, ws: true },
@@ -1,71 +0,0 @@
# #1019 — Zero-timeout queue-guard harness race
- **Issue:** #1019 (parent status remains `believed-fixed, pending jarvis validation`; do not close)
- **Branch:** `fix/1019-ci-queue-timeout-harness`
- **Owner:** `be-coder-08`
- **Base:** `origin/main` at `5916aeefd6ed12bcac086c6834c7f6c4ae38e1bc`
- **Charter:** `/home/hermes/agent-work/tl-mosaic/CHARTER-1019-HARNESS-FIX.md`
## Objective
Make `test-ci-queue-wait-tristate.sh` deterministic without changing any asserted outcome. Remove the indiscriminate zero-timeout race, require every status-classification case to prove the provider was observed, and prove the harness-controlled virtual clock is active.
## Scope
- In scope: `packages/mosaic/framework/tools/git/test-ci-queue-wait-tristate.sh` only, plus this evidence scratchpad.
- Out of scope: guard parsers, D2/D3 behavior, installer/reseed staleness, PR #1060, and issue closure.
## Acceptance criteria
1. RED deterministically reproduces deadline pre-emption before the provider call.
2. Every case that intends status classification positively proves provider observation.
3. Pending observes `pending` before deterministic virtual-time expiration.
4. The virtual clock has a positive interception control; a broken-clock mutant makes the suite red.
5. The exact CI-base image passes the final harness repeatedly with zero failures.
6. Baseline gates, independent code/security review, exact-head CI, and coordinator-authorized squash merge pass.
## Plan
1. Add deterministic RED instrumentation for the known merge/provider-unreachable pre-emption.
2. Replace global `-t 0` with a nonzero timeout interpreted under an event-driven virtual clock; stub sleep without wall waiting.
3. Add provider-observation and virtual-clock positive controls without changing outcome assertions.
4. Run focused shell checks, repeat in exact CI-base image, baseline gates, and independent reviews.
5. Commit with both identity layers, queue-guard plus direct Woodpecker terminal-state verification, push, self-post PR, verify poster/head/CI, obtain coordinator merge authorization, then squash merge without closing #1019.
## Budget
- No explicit token cap supplied. Keep scope to one harness file and one scratchpad; stop/report at the charter's 60% context gate.
## Evidence
- RED, deterministic pre-provider expiry: `evidence/1019-harness-fix/red-pre-provider-expiry.log` — rc 1; merge/provider-unreachable got rc 124 instead of 75, omitted CANNOT_ASSERT, did not observe the status provider, and wrote no additional audit record (four named failures).
- GREEN host focused harness: `evidence/1019-harness-fix/green-host.log` — rc 0, all outcome classes passed.
- Load-bearing clock negative control: a temporary same-directory mutant replaced the virtual `date` body with `/bin/date`; `evidence/1019-harness-fix/red-clock-not-intercepted.log` — rc 1 with named `virtual clock interception did not run` failures. The mutant file was removed after the run.
- Exact CI-base repeat: `git.mosaicstack.dev/mosaicstack/stack/ci-base:latest`, repository mounted read-only, harness work under container `/tmp`; `evidence/1019-harness-fix/ci-image-repeat/summary.log`**100 pass / 0 fail / 100 total**.
- Synchronization design: provider-status observation creates the event marker; virtual time is 1000 before the event and 1002 afterward. Pending alone reaches the stubbed no-op sleep and a post-observation deadline check. `-t 1` is uniquely load-bearing because removing it restores the 900-second default deadline at virtual time 1900, which 1002 does not cross. The numeric timeout is subject semantics under virtual time, not a wall-clock synchronization duration.
## Review remediation — semantic timeout vs. liveness bound
Security review found that virtual time remained at 1000 forever before provider observation and stubbed sleep never waited. A regression looping before the status endpoint—or blocking in the first provider call—therefore could prevent `run_guard` from returning, so the post-return provider assertion could never fire.
**General rule:** A timeout usually serves two purposes: semantics and liveness. Removing wall time from semantic synchronization can silently remove the only independent hang bound. Preserve deterministic virtual time for subject semantics, but provide a separately implemented real-clock liveness watchdog and prove that watchdog fires.
Remediation:
- Every guard subject invocation is launched by absolute `/usr/bin/python3` in a new session. Python's internal monotonic `wait(timeout=...)` provides real-clock liveness independently of PATH; expiry kills the entire isolated process group, so neither PATH-front shims nor a blocked provider descendant can retain the capture pipe.
- Watchdog expiry returns distinct harness rc 90 plus `FAIL HANG watchdog`, separate from subject timeout rc 124.
- A first attempt using absolute `/usr/bin/timeout -s KILL` passed on GNU coreutils but failed in the exact Alpine CI-base image: BusyBox killed the immediate wrapper while the guard/provider descendants survived and retained the command-substitution pipe. The process-group kill is therefore required behavior, not portability polish.
- A committed positive control hangs the branch-provider stub before the status endpoint. It must terminate through the watchdog, emit the hang-specific diagnostic, return rc 90, and prove the status provider was never reached.
- RED before remediation: a temporary ordinary-success mutant hung before provider observation; only an external control could kill the suite (rc 137), and there was no internal hang-specific diagnostic (`red-watchdog-absent.log`).
- The watchdog mutant/control is load-bearing: removing the internal watchdog leaves the control unable to produce its required rc 90 and diagnostic.
Post-review evidence:
- Host focused harness with process-group watchdog: rc 0 (`green-watchdog-process-group-host.log`).
- Exact Alpine CI-base focused harness with process-group watchdog: rc 0 (`green-watchdog-ci-image.log`).
- Hanging ordinary-success mutant: suite rc 1; success returned rc 90, emitted `FAIL HANG watchdog`, and loudly reported that provider/clock observation did not occur (`red-watchdog-fires.log`).
- Removed-`-t 1` mutant: suite rc 1; pending was terminated by the watchdog instead of producing `ASSERTED_NOT_READY`, proving the explicit timeout is load-bearing (`red-timeout-argument-removed.log`).
## 60% context hold
Stopped before baseline/review/commit as required by the charter. Remaining: inspect final diff, shell/static/baseline gates, independent code/security review, remediation if any, identity-bound commit/trailer verification, mandatory queue guard plus direct terminal Woodpecker `mosaic` enumeration, push, self-posted PR/provider poster read-back, exact-head terminal-green CI, coordinator merge authorization, squash merge, main CI verification, and leave #1019 unclosed as `believed-fixed, pending jarvis validation`.
@@ -1,93 +0,0 @@
# W-B — Measure Pi's real tool registry
- **Task / internal ref:** W-B from the lease-remediation orchestrator brief (no matching `docs/TASKS.md` row; workers do not modify that file)
- **Objective:** identify the exact tool names emitted as `event.toolName` by the installed Pi runtime and compare them with the broker's Pi read-only carve-out.
- **Scope:** measurement and report only; no broker or runtime source changes. W-C is out of scope.
- **Budget:** no explicit token cap; constrained to this scratchpad and one local commit.
- **Installed runtime:** `@earendil-works/pi-coding-agent` / `pi` `0.84.1`.
## Method
I created a throwaway extension at `/tmp/measure-pi-tool-registry.ts` (not in the worktree). On `session_start` it recorded `pi.getAllTools()` and `pi.getActiveTools()`; on every `tool_call` it appended the exact `event.toolName`. I then launched an isolated, ephemeral Pi session with all built-ins explicitly selected:
```text
PI_OFFLINE=1 pi --mode print --no-session --no-approve \
--no-context-files --no-skills --no-prompt-templates --no-extensions \
-e /tmp/measure-pi-tool-registry.ts \
--tools read,bash,edit,write,grep,find,ls <deterministic probe prompt>
```
The prompt exercised file read, content search, file search, directory listing, shell execution, file write, and file edit. Pi exited `0`; every selected tool produced one `tool_call`. The write/edit control artifact ended with exact content `after`, proving the mutating calls executed in order.
This runtime observation was cross-checked against the installed distribution's canonical registry at `dist/core/tools/index.js:17`, which declares the same seven names. The gate consumes the measured field directly at `packages/mosaic/framework/runtime/pi/mosaic-extension.ts:368`.
## Exact distinct built-in set
The installed Pi built-in registry is exactly:
```text
{bash, edit, find, grep, ls, read, write}
```
| Tool | Runtime registry observation | `tool_call` observation | Installed definition |
| --- | --- | --- | --- |
| `read` | `<builtin:read>` | observed once | `dist/core/tools/read.js:138` |
| `bash` | `<builtin:bash>` | observed once | `dist/core/tools/bash.js:231` |
| `edit` | `<builtin:edit>` | observed once | `dist/core/tools/edit.js:170` |
| `write` | `<builtin:write>` | observed once | `dist/core/tools/write.js:138` |
| `grep` | `<builtin:grep>` | observed once | `dist/core/tools/grep.js:79` |
| `find` | `<builtin:find>` | observed once | `dist/core/tools/find.js:79` |
| `ls` | `<builtin:ls>` | observed once | `dist/core/tools/ls.js:61` |
The raw distinct `event.toolName` result was:
```json
["bash", "edit", "find", "grep", "ls", "read", "write"]
```
Pi registers all seven, but its default active set is only `read`, `bash`, `edit`, and `write` (`dist/core/sdk.js:132`). The probe explicitly activated all seven so the three search/list tools could be observed at the hook.
## Positive control
The known `read` tool was the control. The method surfaced it twice:
1. `pi.getAllTools()` returned `read` with source path `<builtin:read>`.
2. Reading `/tmp/pi-registry-probe/seed.txt`, which contained `CONTROL_TOKEN`, produced one hook record with `event.toolName === "read"`.
The control was therefore positive; the seven-name result is measured, not an empty-probe inference.
## Carve-out comparison and collision result
The broker currently declares `{"read", "grep", "find", "ls"}` at `packages/mosaic/framework/tools/lease-broker/daemon.py:54`.
- `read`: real built-in.
- `grep`: real built-in.
- `find`: real built-in.
- `ls`: real built-in.
All four carve-out names are exact, case-sensitive Pi tool names.
The general execution/writing tool names are `bash`, `edit`, and `write`. Their intersection with the carve-out is empty:
```text
{bash, edit, write} ∩ {read, grep, find, ls} = ∅
```
Therefore no general shell-exec or file-mutating Pi tool shares a name with a carve-out entry. `grep` and `find` may invoke constrained search helpers internally, but neither exposes an arbitrary command interface; the arbitrary command tool is distinctly named `bash`.
The Mosaic extension separately registers the non-built-in custom tool `mosaic_context_recover` at `packages/mosaic/framework/runtime/pi/mosaic-extension.ts:379`; the broker handles that identity through its dedicated recovery exemption rather than the read-only set (`daemon.py:722`). Unknown or third-party custom tools are not part of Pi's built-in seven-name registry and remain outside the carve-out.
## Verification evidence
- `pi --version``0.84.1`.
- Isolated probe exit → `0`.
- Runtime `getAllTools()` count → `7`, all with `sourceInfo.source === "builtin"`.
- Distinct hook names → `bash`, `edit`, `find`, `grep`, `ls`, `read`, `write`.
- Hook counts → exactly one call for each of the seven names.
- Mutation artifact after `write` then `edit` → exact content `after`.
- Installed registry source → `allToolNames = new Set(["read", "bash", "edit", "write", "grep", "find", "ls"])`.
## Risks / limitations
- The probe deliberately disabled all other extensions, so extension-defined third-party tools were excluded from the built-in registry measurement. The production gate still receives those names and treats names outside the broker carve-out as mutating/fail-closed.
- Explicit `--tools` activation was required to exercise `grep`, `find`, and `ls`; this does not imply they are active in Pi's default four-tool configuration.
@@ -1,99 +0,0 @@
# PR merge squash message field
- **Charter:** `/home/hermes/agent-work/CHARTER-PRMERGE-MESSAGE-FIELD.md`
- **Owner:** `be-coder-08`
- **Branch:** `fix/pr-merge-message-field`
- **Base:** remote `main` / local `origin/main` at `85d2108e4ed15c744ad3b87a5b629e7b2d39405a`
- **Estate:** HOMELAB tooling shared by HOMELAB and USC
## Objective
Add an optional, identity-checked Gitea squash message to `pr-merge.sh` so genuine multi-author PRs retain non-poster branch authors without weakening hardcoded squash behavior.
## Binding requirements
1. `Do` remains hardcoded to `squash`; no provider/repository default may select merge style.
2. A verified trailer uses a PR commit's linked `author.login` and that same commit's author email. No `/users/{login}` primary-email lookup occurs. Recorded rationale: this asks only what the provider can answer.
3. A commit with `author.login` null blocks before merge, prints both the null provider fact and commit email fact, and names the escalation principal.
4. The BLOCK arm must be observed firing; a normal canonical single-author API payload remains explicit squash plus its reviewed `head_commit_id`.
5. Every provider mutation is read back from the provider; no real PR is merged during tests.
## Derived interface decisions
- Add `--co-author-trailers` rather than accepting arbitrary message text. The wrapper enumerates PR commits and constructs trailers, making an unchecked `Co-authored-by` line unexpressible.
- Require `--escalate-to PRINCIPAL` with `--co-author-trailers`, so the BLOCK diagnostic always names a principal rather than a generic role.
- Do not expose `MergeTitleField` separately. When trailers exist, set it from the provider PR title and set `MergeMessageField` only to construction-generated trailers. This preserves one provider source for the title and avoids an unrelated caller-controlled degree of freedom.
- Preserve first-commit order and emit one trailer per distinct non-poster `author.login`, using that first linked commit's own email.
## Canonical delivery plan
1. Port the capability into the installed source of truth, `packages/mosaic/framework/tools/git/pr-merge.sh`; do not retain `infra/fleet/tools/git` as a second copy.
2. Preserve canonical `--expect-head`, exact head branch/repository/SHA queue inspection, Gitea atomic head pinning, GitHub `--match-head-commit`, and delete-after-merge semantics.
3. Do not port the deployed-only `--skip-queue-guard` bypass. Add the focused harness to the canonical framework-shell suite and re-establish RED/GREEN on the packaged baseline.
4. Deliver through a reviewed package release followed by `mosaic update` with its default framework reseed. The installer snapshots, manifest-syncs framework-owned `tools/**`, and rolls back on failure.
5. Before either estate relies on the change, require installed/package hash equality, `MergeMessageField` presence, and a green focused harness. Release/reseed ownership is currently unassigned and blocks activation after source merge.
## Evidence
- RED against the byte-identical deployed baseline (`sha256 08a65e8584c5…`): rc 1 with eight named failures. The wrapper rejected `--co-author-trailers`; the null-login path emitted none of the required BLOCK facts/principal; and both verified/ordinary API paths failed the stdin-config credential assertion (ordinary path exposed the fixture token through curl argv). Log: `/home/hermes/agent-work/be-coder-08/evidence/prmerge-message-field-red.log`.
- GREEN on the deployed-baseline candidate: verified linked multi-author payload, null-login BLOCK, required named principal, explicit squash, stdin-config token transport, and absence of `/users` lookup all passed. Log: `/home/hermes/agent-work/be-coder-08/evidence/prmerge-message-field-green.log`.
- RED against canonical packaged baseline `c581ef48…`: rc 1 with 32 assertions. It rejects the new option, and the first harness version did not satisfy canonical head branch/repository/SHA metadata. Log: `/home/hermes/agent-work/be-coder-08/evidence/prmerge-packaged-baseline-red.log`. The port adapts the fixture rather than weakening canonical head controls.
- Provider capability probe against `git.mosaicstack.dev`: authenticated `be-coder-08` POST to deliberately nonexistent PR `2147483647` with both message fields returned JSON HTTP 404; the unauthenticated same request returned JSON HTTP 401 (not the charter's predicted 403). The authenticated-vs-unauthenticated differential proves write authorization resolved while no mergeable subject existed. `tl-mosaic` ruled the literal non-load-bearing: preserve the observed 404/401 pair and do not manufacture a 403 case. No cause was inferred and no real PR was targeted.
- Provider-generated trailer behavior is not treated as exclusive or absent. The wrapper's VERIFIED/BLOCK decision binds each requested non-poster trailer to commit `author.login` plus that commit's email; it does not assume `MergeMessageField` is the squash's only trailer source. The poster is omitted from the constructed list because the resulting squash author already records the poster; any additional provider-generated trailer is outside this change's unmeasured mechanism.
- An early candidate SHA-256 `5de32876990e4f26920448cb3220cc7f1146d558b4dd2bc1ee1a2abee2f2cbe6` passed the initial harness, then author-side review found credential-fallback and argv-exposure defects. The live deployed wrapper was atomically restored to baseline SHA-256 `08a65e8584c52c6d41ea1c686f8b95585c21e4b37320a2447eba09359a0e02c1`; the remediated candidate remains only in the worktree.
## Remediation and current review state
1. Token and Basic Auth now use stdin curl configuration, not argv. PR title, contributor email, and the JSON payload also remain out of child argv.
2. Each credential attempt binds commit inspection and merge. A token failure during either inspection or mutation causes Basic fallback to repeat inspection before mutation; the payload pins the inspected `head_commit_id`.
3. Focused tests cover token-resolution fail-closed behavior, both HTTP-401 fallback seams, metadata/credential argv absence, null-login BLOCK, explicit squash, canonical reviewed-head binding, unchanged ordinary payload, and retained log-safe provider diagnostics. Token-resolution RED: `/home/hermes/agent-work/be-coder-08/evidence/prmerge-token-resolution-red.log`.
4. Codex review rounds 35 requested retained provider error text, log-safe provider diagnostics, fail-closed credential fallback, stable value-option parsing, and PR-title trailer-injection prevention. These are remediated with regression assertions. A post-remediation independent review is still required.
5. **Accepted linkage limitation:** `author.login` resolution proves that the commit address maps to a registered provider account. It does not prove that the named principal authored the commit because Git author metadata is self-asserted. This gate checks attribution linkage, not authorship; commit signing is out of scope and currently unadopted. Coordinators explicitly ruled that this does not add a third state.
6. Codex's sandbox could not execute the harness because its checkout was read-only; that environmental limitation is recorded separately from host-side test results.
## Disposable provider fixture acceptance
- Use a retained scratch repository only, with two branch authors and `author != committer` on at least one commit.
- Arm A supplies a message-field trailer for one non-poster; record whether that value lands without forcing the partial-pair result into under-specified `APPENDS`/`REPLACES` labels. Demonstrate an absence control.
- Arm B includes a registered trailer for a different non-poster on a branch commit; record whether it survives or drops. Verify identity through an existing commit whose `author.login` resolves and demonstrate an absence control.
- Parse landed trailers key-agnostically with `^[A-Za-z-]+-[Bb]y:` and record generated poster pair presence/absence plus resulting poster attribution.
- Record `/users/<login>` status and raw email only as non-gating estate telemetry. Never read `active`, `visibility`, or any profile field as an identity gate.
- Use distinct principals: poster `be-coder-08`, merger `Mos`, Arm A `be-coder-07`, and Arm B `be-coder-06`. Capture every trailer-shaped line verbatim and in order. Zero trailer lines means the generator did not fire and the run is `VOID`, not evidence that either arm dropped.
- Report the same read-back evidence to `mos-claude` on socket `default` and `tl-mosaic` on socket `mosaic-fleet`. Report values rather than mechanism inferences and stop on any poster-attribution regression.
## Fixture preflight
- Retained public repository: `mosaicstack/prmerge-trailer-fixture`; PR `#1`, posted by `be-coder-08` and reserved for merge by `Mos`.
- Existing `mosaicstack/stack` commits resolve `be-coder-07` and `be-coder-06` through `author.login`; exact addresses are `[email protected]` and `[email protected]`.
- Non-gating HOMELAB telemetry for authenticated reader `be-coder-08`: `/api/v1/users/be-coder-06` returned HTTP 200 with raw `email` value `[email protected]`.
- Provider preflight showed PR commit enumeration is newest-first. A new RED test proved that deriving `head_commit_id` from the final array element selected the wrong commit. The candidate now reads `.head.sha` from the authenticated PR endpoint before enumeration, verifies it appears in the commit set, and atomically pins that SHA in the explicit squash payload. RED: `/home/hermes/agent-work/be-coder-08/evidence/prmerge-head-order-red.log`.
- Fixture PR head `f6ba6e5105031fa21f5ff7bd8e4379d99c16e1de` has `author.login=be-coder-07`, `committer.login=be-coder-08`, and branch-message trailer `Co-authored-by: be-coder-06 <[email protected]>`.
## Fixture result
- `Mos` merged retained fixture PR `#1` through staged candidate SHA-256 `60e779a85fd13b729d859ea7c986d1e9b1641b97991611329226c1b3113ffb6e`; resulting squash commit: `3f550715d9bc716426fd355a65fe997b3a90fa7d` with one parent.
- Provider read-back: poster/commit author `be-coder-08`, committer/merger `Mos`. The run is non-void.
- Trailer-shaped lines, verbatim and in order:
1. `Co-authored-by: be-coder-07 <[email protected]>`
2. `Co-authored-by: be-coder-08 <[email protected]>`
- Arm A supplied field value (`be-coder-07`) landed. Arm B branch trailer (`be-coder-06`) dropped. Both fabricated absence controls remained absent. No `Co-committed-by:` line landed.
- The candidate payload construction explicitly excludes the poster and supplied only the Arm A `be-coder-07` line. Therefore the landed poster line was provider-generated, not candidate-composed. The raw result supports `FIELD LANDS`, `BRANCH DROPS`, and `POSTER GENERATED`; it does not support a claim that candidate code supplied the poster. Evidence: `/home/hermes/agent-work/be-coder-08/evidence/prmerge-fixture-readback.log` and the retained provider object.
- Retained fixture PR `#2` measured the N=2 shape needed by `#1030`: supplied `be-coder-07` then `be-coder-06`; both landed in that order, followed by the provider-generated poster line. No truncation or dedup occurred at N=2. Resulting squash: `39db9d13aed0…`.
## Current hold point
PR `mosaicstack/stack#1066` is open. Its first frozen head `f4b162fa…` was terminal-green in Woodpecker `mosaic` pipeline `#2225`, but that evidence becomes stale when the canonical port moves the head. The deployed wrapper remains baseline `08a65e85…`; no manual copy will occur. Canonical port tests, commit amendment, rebase, one guarded force-with-lease, exact-head CI, and new independent review remain. Even after source merge, activation remains blocked on an assigned package-release/reseed owner and installed-byte read-back.
## Security review 96 remediation
Exact reviewed predecessor head: `1ceb11058f64dd7f4a817ceb2124f980a1c4dd23`.
RED-first focused harness produced 10 named failures: all curl calls lacked size/time/connect bounds; raw ESC email reached mutation; oversized and stalled curl failures were discarded and reached mutation; nonempty Basic output with resolver rc 91 authorized mutation.
Security remediation:
- Removed the cross-principal HTTP-401 Basic fallback. Both inspection-401 and merge-401 paths now refuse without Basic resolution or mutation; `get_gitea_basic_auth` references in the merge subject are 0.
- Applied `--max-filesize`, `--max-time`, and `--connect-timeout` to all 3/3 provider curl sites and fail closed on curl transport rc at all 3/3 sites.
- Required linked email bytes to be ASCII and printable before constructing `MergeMessageField`; guarded construction sites 1/1.
GREEN: message-field, exact-head, empty-UID/API, queue branch/repository/SHA, bash syntax, ShellCheck, and diff check pass. R7 total-removal mutants went RED: email guard 3 rows; bound switches 1 row; transport-rc guards 4 rows; HTTP-401 refusal 3 rows. R7 bound: mutants prove total removal only; explicit denominators above prove site coverage.
@@ -1,99 +0,0 @@
# P3-R1 — Routing Health Enum + `/mcp` Wiring Scratchpad
**Task:** P3 hands-on acceptance blockers #1 and #5
**Mission:** `mvp-20260312` (active)
**Branch:** `feat/webui-p3r1-routing-mcp` from `origin/next`
**Required base:** `20718b5a273d243363a4f5cbef5bbf692a805bdb`
**Tracking ref:** Direct P3-R1 author brief; no provider issue supplied; no PR or merge authorized
**Started:** 2026-08-11T14:52:52-05:00
## Objective
Fix exactly two P3 acceptance blockers:
1. Make routing consume the canonical `ProviderHealthStatus` enum, with `healthy` and `degraded` routable and `down` non-routable, at both routing decision sites.
2. Wire `McpClientModule` into `CommandsModule`, make `McpClientService` required, remove the unreachable unavailable-service branch, and prove `/mcp status` reaches the client.
Explicitly excluded: provider registry/adapters, `provider.service.ts`, `agent.service.ts`, `chat.gateway.ts`, fallback membership, task classification, selector UI, conversation resume, reload UX, WS/origin/handshake behavior, dependencies/lockfile, and `apps/web/**` changes.
## Plan
1. Confirm exact base/branch and inspect every cited source/test anchor plus all direct `CommandExecutorService` construction sites.
2. Record baseline gateway and focused routing/commands/MCP test totals.
3. Add regression tests first and run focused tests to capture expected RED failures.
4. Apply only the typed routing helper/signature changes and required MCP module/constructor/guard changes; update impossible `up` fixtures.
5. Run focused tests, all user-required verification gates, lockfile/scope/diff checks, and record counts.
6. Obtain independent spec and code/security review; remediate any findings and repeat affected gates.
7. Commit conventionally, run the pre-push queue guard, and push only the feature branch (no PR or merge).
## Budget
No explicit token cap supplied. Working soft cap: **30K tokens**, based on two bounded gateway bug fixes, focused TDD, full gateway/root verification, independent review, and branch delivery. One coding worker will execute serially; reviews will be independent and serial to avoid worktree collisions.
## Startup Evidence
- `git fetch origin` rc=0.
- `origin/next` confirmed exactly `20718b5a273d243363a4f5cbef5bbf692a805bdb`.
- Local and remote `feat/webui-p3r1-routing-mcp` were absent before creation.
- Branch creation rc=0; HEAD equals the required base.
- Harness-owned `.mosaic/orchestrator/session.lock` was already dirty and remains excluded from staging/commit.
## Baseline Evidence
- Gateway full suite: rc=0; **64 files / 693 tests passed**, 7 files / 17 tests skipped (71 files / 710 tests total).
- Routing sub-suite (`src/agent/routing`): rc=0; **3 files / 105 tests passed**.
- Commands/MCP sub-suite (`src/commands`, `src/mcp-client`): rc=0; **6 files / 76 tests passed**.
## TDD and Implementation Evidence
- Routing RED: after canonical fixture/test changes but before the service fix, focused routing run returned rc=1 with **15 failed / 91 passed (106)** because the old `up`/`ok` gates rejected `healthy` and `degraded`.
- Routing GREEN: canonical `ProviderHealthStatus` map types, one `isRoutable` helper, and both comparison sites corrected; focused routing suite passed.
- MCP behavior test now drives `/mcp status` through a required mock client and asserts the client is called, success is returned for zero servers, and the former unavailable message is absent.
- MCP wiring mutation RED used the final `Reflect.getMetadata('imports', CommandsModule)` assertion with only the production `McpClientModule` import/registration temporarily removed: rc=1, exact failure `expected [GCModule, …] to include McpClientModule`.
- MCP wiring GREEN after byte-for-byte production restoration: rc=0. Temporary mutation did not remain.
- Every direct `new CommandExecutorService(...)` test construction now supplies a non-null MCP client mock.
- Initial Codex worker launch failed rc=1 from missing OpenAI bearer authentication. First Mosaic Claude launch failed rc=1 because Distrobox-local runtime contracts were absent; retry with the supported host `MOSAIC_HOME=/home/jwoltje/.config/mosaic` succeeded.
## Review Evidence
- Independent spec review: **approve**, 0 blockers, scope OK.
- Independent code/security review: **approve**, 0 blockers, 0 critical/high security findings. It suggested an actual module-wiring assertion, which was added with valid mutation RED/GREEN evidence.
- Independent final re-review after remediation: **approve**, 0 blockers, 0 critical/high security findings, no remaining findings.
- Optional missing-provider/`undefined` test suggestion was not adopted: the brief explicitly requires the three canonical statuses (`healthy`, `degraded`, `down`) and forbids scope expansion; runtime behavior for absent keys remains `undefined` → non-routable through the required helper signature.
## Documentation Assessment
- `docs/PRD.md` already requires provider fallback/routing and MCP capability; this increment restores implementation to those existing contracts.
- No public API endpoint, payload schema, auth/permission rule, navigation, deployment procedure, or new user workflow changes. OpenAPI, endpoint index, user/admin/developer guides, and sitemap are therefore N/A for this bounded repair.
- This append-only scratchpad is the implementation, TDD, review, and verification record. Canonical docs remain in-repo; no publishing action is in scope.
## Final Verification Evidence
All required and repository-situational gates completed with rc=0:
| Gate | Result |
| --- | --- |
| `pnpm install --frozen-lockfile` | rc=0 |
| Gateway typecheck | rc=0 |
| Gateway lint | rc=0 |
| Routing focused suite | rc=0; 3 files / 106 tests |
| Commands/MCP focused suite | rc=0; 6 files / 78 tests |
| Gateway full suite | rc=0; 64 files / 696 tests passed; 7 files / 17 tests skipped |
| Gateway build | rc=0 |
| Root typecheck | rc=0; 45/45 tasks |
| Web test | rc=0; 19 files / 154 tests |
| Root lint | rc=0; 25/25 tasks |
| Root format check | rc=0 |
| `git diff --check` | rc=0 |
- Gateway suite before→after: **693→696 passing tests**; skipped remained 17 (total 710→713).
- Routing focused before→after: **105→106 passing tests**.
- Commands/MCP focused before→after: **76→78 passing tests**.
- `pnpm-lock.yaml` SHA-256 before/after frozen install: `9acaa89d213b3281e757b6edf6fdb8727176570d725b78a0de234c61a7f3c332`; diff versus `origin/next` rc=0.
- Verified no changed path under `apps/web/**`, provider service/adapters, `agent.service.ts`, `chat.gateway.ts`, or lockfile.
- Verified both `McpClientModule` production wiring lines remain and no impossible `up`/`ok` routing status checks/fixtures remain.
- Verified code/test diff SHA-256: `27b41a855084b9dd85a7bc2a79fa3251d114ee226a4f1b835e65134c25a4f5a8`.
## Delivery State
Implementation, testing, documentation assessment, and independent review are complete. Remaining authorized actions: format this final scratchpad append, create one conventional commit, run the required push queue guard, and push only `feat/webui-p3r1-routing-mcp`; no PR or merge.
-271
View File
@@ -1,271 +0,0 @@
# WebUI Phase P — P4-1 Projects + Tasks SPA Scratchpad
**Task ID:** P4-1
**Tracking ref:** Phase P RFC §6.4 / §2.4 author brief; no provider issue supplied
**Mission context:** `mvp-20260312` (active)
**Branch:** `feat/webui-p4-1` from `origin/next`
**Started:** 2026-08-11
**Role:** orchestrator-controlled author worker; `docs/TASKS.md` remains orchestrator-only and is not part of this increment
## Original tasking
Implement the bounded P4-1 SPA parity slice exactly as briefed: real authenticated `/projects`, `/projects/:id`, and `/tasks` React Router pages ported from the existing Next baseline; reuse existing project/task components; preserve relative same-origin REST access; support only the existing project/task PATCH edit flows; add route error boundaries and page tests; pin Vite strict port 3100 and legacy Next dev/start to 3101; make no gateway, package, dependency, nav-shell, chat, settings, or admin changes; verify, commit, and push only `feat/webui-p4-1` (no PR or merge). If any required REST endpoint is absent, stop as `BLOCKED:` rather than inventing a workaround.
## Objective and acceptance map
- A: `/projects` list with loading, cards, empty, and surfaced API-error states; defer MissionStatus side panel.
- B: `/projects/:id` detail with overview/tasks/missions tabs, parallel project/mission/task load, bounded project and task PATCH edits.
- C: `/tasks` list/kanban with bounded task PATCH edits.
- D: replace the three route placeholders and add route error boundaries without changing chat.
- E: use only `@/lib/types` and `@/lib/api`; relative `/api/...` REST paths only.
- F: Vite `strictPort: true`; Next dev/start pinned to 3101; no proxy/dependency changes.
- G: fixture-backed Vitest coverage for lists, states, tabs/toggles, PATCH calls, and real route elements; all named verification gates pass.
## Plan
1. Verify the required gateway REST routes and inspect the exact `origin/next` SPA, Next baseline pages, shared components, types, API helper, and P3 test conventions.
2. Record the pre-change web test count and add required page/route tests first, observing expected RED failures.
3. Port the three pages and shared route error boundary, then wire routes and bounded PATCH flows.
4. Apply only the two dev-port pin changes.
5. Run focused tests, all user-required gates, lockfile and same-origin checks, and `git diff --check`.
6. Obtain independent spec/code/security review; remediate and repeat affected gates until clear.
7. Commit conventionally, run the required pre-push queue guard, push only the feature branch, and record exact evidence here.
## Testing strategy
TDD is applied because this adds user-visible data/edit behavior. Component tests mock only the existing API boundary and exercise rendered behavior and PATCH payloads. Primary situational evidence is the required page interaction suite plus route-resolution checks; baseline evidence is typecheck, lint, full web test, build, frozen install, formatting/diff hygiene, and same-origin grep.
## Budget
No explicit token cap was supplied. Working soft cap: **50K tokens**, derived from three coupled React pages, route/error wiring, interaction tests, port config, review/remediation, and full verification. One Codex implementation worker and independent review workers will be used serially to avoid worktree collisions.
## Base evidence
- `git fetch origin` rc=0.
- `origin/next` and branch start: `e00cc475a2b1e9866bd4e2f8df80aff640c3a543`.
- Branch created: `feat/webui-p4-1` tracking `origin/next`.
- Harness-owned `.mosaic/orchestrator/session.lock` is dirty and must remain unstaged/uncommitted.
## Progress / evidence
- [x] Loaded active mission manifest, latest scratchpad, top-level tasks, PRD, orchestration/delivery/frontend/QA/documentation/review/TypeScript guides, and matching implementation skills.
- [x] Confirmed exact base SHA and created the feature branch.
- [x] Required REST endpoints verified in Gateway source: project list/detail/PATCH, task list/filter/detail/PATCH, and mission list all exist.
- [x] Scope assumption check found a blocking contradiction before source implementation.
- [ ] Pre-change web test count recorded.
- [ ] RED tests observed.
- [ ] Implementation complete.
- [ ] Independent review clear.
- [x] Required verification gates run against the unchanged web baseline; exact leak-grep expectation is independently blocked by 12 pre-existing matches.
- [x] Blocker record committed as `5ede86a5` and feature branch pushed; no PR opened and no merge performed.
## Blocker — 2026-08-11
`P4-1` is blocked because the bounded edit UX asserted by the brief does not exist at the confirmed `origin/next` base (`e00cc475`):
- `apps/web/src/components/tasks/task-detail-modal.tsx` is read-only. Its props are only `task` and `onClose`; it contains no input/select/textarea, update callback, or `api()`/PATCH call.
- `apps/web/src/app/(dashboard)/projects/[id]/page.tsx` contains no project edit controls and no project/task PATCH call.
- `apps/web/src/app/(dashboard)/tasks/page.tsx` does not open `TaskDetailModal`; its click handler only logs that a future detail view will be added.
- A repository grep over those reference/component paths finds zero `method: 'PATCH'` calls and zero edit/save form controls.
- The required leak grep also disproves the brief's baseline claim that `apps/web/src` is clean: the exact command returns rc=0 with 12 pre-existing matches at `origin/next` (external font links, an avatar placeholder, an SVG namespace, and test fixtures). There are zero new matches in the worktree, but making the exact grep empty would require unrelated changes the brief forbids.
The brief simultaneously requires tests proving task/project PATCH edits and forbids inventing edit UX, fields, or flows beyond what those references expose. Adding controls or callback contracts would therefore be new design outside the authorized port/wiring scope. Per the explicit stop rule, no SPA source, route, config, package, lockfile, or test implementation was started.
## Verification evidence on unchanged web baseline
These gates were still run to distinguish the brief contradiction from a broken base:
- `pnpm --filter @mosaicstack/web typecheck` — rc=0.
- `pnpm --filter @mosaicstack/web lint` — rc=0.
- `pnpm --filter @mosaicstack/web test` — rc=0; 15 files, 133 tests passed. Before→after remains 133→133 because implementation was halted.
- `pnpm --filter @mosaicstack/web build` — rc=0; Next production build completed, 10/10 static pages generated.
- `pnpm install --frozen-lockfile` — rc=0; lock SHA-256 remained `9acaa89d213b3281e757b6edf6fdb8727176570d725b78a0de234c61a7f3c332`; `git diff` for the lockfile rc=0.
- `git grep -nE "http://|https://" -- apps/web/src` — rc=0 with 12 pre-existing matches; `origin/next` count=12, current HEAD count=12, new worktree additions=0.
- `pnpm format:check` — rc=0.
- `git diff --check` — rc=0.
## Delivery evidence
- Initial queue-guard invocation through shell `~` failed rc=127 because this Distrobox resolves `HOME` to `/home/jwoltje/distrobox-homes/mosaic-dev`, where the injected fleet status already reports the tools installation missing.
- Correct supported host-tool invocation `/home/jwoltje/.config/mosaic/tools/git/ci-queue-wait.sh --purpose push -B feat/webui-p4-1` — rc=0; branch absent remotely and queue clear.
- `git push -u origin feat/webui-p4-1` — rc=0. The push hook additionally ran repository preflight, typecheck (45/45 tasks), lint (25/25 tasks), and format check successfully.
- No PR was opened and no merge was attempted, per the brief.
## Risks / blockers
- The active harness mutates `.mosaic/orchestrator/session.lock`; it is excluded from staging.
- The Phase P §6.4 prose is absent from this checkout, so the brief-named route spec and existing Next pages are the bounded implementation anchors.
- The MissionStatus `/api/coord/status` panel is explicitly deferred to a follow-up and must not enter P4-1.
## REV 2 continuation — 2026-08-10T21:59:17-05:00
REV 2 supersedes the original tasking above. The independent check confirmed that the Next reference pages and `task-detail-modal.tsx` are read-only, so project/task editing is deliberately deferred to P4-1b. Do not re-litigate or implement the former PATCH requirements.
### Revised objective and acceptance map
- A: `/projects` read-only SPA list with loading, cards, empty, and surfaced API-error states; card navigation to `/projects/:id`; no MissionStatus panel.
- B: `/projects/:id` read-only SPA detail using `useParams`, `useNavigate`, and the specified three-request `Promise.all`; preserve reference overview/tasks/missions tabs and read-only task modal.
- C: `/tasks` read-only SPA list/kanban view with the existing read-only task modal.
- D: replace only the three route placeholders and add page-local route error boundaries without modifying chat.
- E: use only `@/lib/types` and `@/lib/api` with relative `/api/...` REST paths.
- F: add Vite `strictPort: true` and pin legacy Next dev/start to 3101 without dependency or proxy changes.
- G: add fixture-backed page tests and route-resolution assertions; run every user-specified verification gate and the narrowed new-code origin check.
### Revised plan
1. Reset `feat/webui-p4-1` to the exact `origin/next` base and independently verify the read-only reference/component/API assumptions.
2. Record the pre-change web suite count; write and run focused page/route tests first to observe expected RED failures.
3. Port the three read-only pages, add independent route error boundaries, wire routes, and apply only the two dev-port pins.
4. Run focused tests, full required web gates, frozen install/lockfile proof, origin check, format/diff hygiene, and accessibility/state-transition sanity checks.
5. Obtain independent spec/code/security review, remediate every blocker, and repeat affected gates.
6. Commit conventionally, run the supported pre-push queue guard, and push only `feat/webui-p4-1`; no PR and no merge.
### Revised budget and session state
- No explicit token cap was supplied. Working soft cap remains **50K tokens**.
- TDD is required by the user and frontend skill for the new SPA behavior; tests must fail for missing pages before production implementation.
- Documentation assessment: this is a parity port of existing read-only behavior, not a new public workflow or API contract; the task scratchpad is the required delivery record, with no user/developer/API documentation changes in this bounded increment.
- Exact base confirmed after fresh fetch/reset: `e00cc475a2b1e9866bd4e2f8df80aff640c3a543`.
- Remote `feat/webui-p4-1` is absent; the eventual push is a fresh branch creation.
- Harness-owned `.mosaic/orchestrator/session.lock` remains excluded from staging.
## REV 2 implementation pass — 2026-08-11T22:05:00Z
### Startup verification
- Loaded required startup files: `/home/jwoltje/.config/mosaic/CONSTITUTION.md`, `/home/jwoltje/.config/mosaic/SOUL.md`, project `AGENTS.md`, `/home/jwoltje/.config/mosaic/guides/E2E-DELIVERY.md`, `docs/PRD.md`, and this scratchpad.
- Loaded required skills: `test-driven-development`, `vitest`, `vite`, `next-best-practices`, and `verification-before-completion`.
- Loaded required runtime guide: `/home/jwoltje/.config/mosaic/runtime/codex/RUNTIME.md`.
- Structured reasoning tool availability verified through the harness before planning.
- `git rev-parse HEAD` confirmed exact required base: `e00cc475a2b1e9866bd4e2f8df80aff640c3a543`.
- `git status --short` at startup showed only the expected untracked append-only scratchpad; `.mosaic/orchestrator/session.lock` did not appear and must remain unstaged if it changes later.
### Current plan
1. Record baseline evidence.
2. Add focused failing SPA page/route specs first.
3. Run focused RED command and record the missing-page failure.
4. Implement the bounded read-only SPA pages, boundaries, route wiring, and the two dev-port pins.
5. Run the required verification matrix and inspect diff hygiene.
### Pre-change baseline
- Command: `pnpm --filter @mosaicstack/web test`
- rc=0
- File/test totals before changes: `15 files / 133 tests`
- Notes: baseline already includes `/chat` SPA route coverage and raw React `createRoot`/`act` page specs that this pass should mirror.
### RED evidence before production implementation
- Command: `pnpm --filter @mosaicstack/web test -- src/spa/pages/projects.spec.tsx src/spa/pages/project-detail.spec.tsx src/spa/pages/tasks.spec.tsx src/spa/pages/resource-route-boundaries.spec.tsx src/spa/routes.spec.tsx`
- rc=1
- Expected missing-feature reason confirmed:
- `src/spa/pages/projects.spec.tsx`, `src/spa/pages/project-detail.spec.tsx`, `src/spa/pages/tasks.spec.tsx`, and `src/spa/routes.spec.tsx` fail import resolution because the SPA page modules do not exist yet.
- `src/spa/pages/resource-route-boundaries.spec.tsx` fails because the current `/projects`, `/projects/:id`, and `/tasks` routes still render placeholders without route-level alert fallbacks.
### Implementation summary
- Added SPA pages:
- `apps/web/src/spa/pages/projects.tsx`
- `apps/web/src/spa/pages/project-detail.tsx`
- `apps/web/src/spa/pages/tasks.tsx`
- Added page-local route boundary components in `apps/web/src/spa/pages/resource-route-error-boundaries.tsx`.
- Wired `/projects`, `/projects/:id`, and `/tasks` in `apps/web/src/routes.tsx` with real elements and `errorElement`s.
- Added fixture-backed raw React Vitest coverage plus route assertions for the three pages and their boundaries.
- Applied only the requested dev topology pins:
- `apps/web/vite.config.ts`: `server.strictPort = true`
- `apps/web/package.json`: `next dev -p 3101`, `next start -p 3101`
### Post-implementation verification
- Focused changed specs:
- Command: `pnpm --filter @mosaicstack/web exec vitest run src/spa/pages/projects.spec.tsx src/spa/pages/project-detail.spec.tsx src/spa/pages/tasks.spec.tsx src/spa/pages/resource-route-boundaries.spec.tsx src/spa/routes.spec.tsx`
- rc=0
- Result: `5 files / 26 tests` passed.
- `pnpm --filter @mosaicstack/web typecheck` — rc=0.
- `pnpm --filter @mosaicstack/web lint` — rc=0.
- `pnpm --filter @mosaicstack/web test` — rc=0; post-change totals `19 files / 153 tests`.
- `pnpm --filter @mosaicstack/web build` — rc=1.
- Limitation: Next/Turbopack hit a sandbox/runtime failure while processing `apps/web/src/app/globals.css`: `creating new process`, `binding to a port`, `Operation not permitted (os error 1)`. This appears environmental, not route-code-specific.
- `pnpm --filter @mosaicstack/web build:vite` — rc=0.
- `pnpm install --frozen-lockfile` — rc=1.
- Limitation: repo `prepare` hook attempted to lock `/home/jwoltje/distrobox-homes/mosaic-dev/src/stack/.git/config`, which is read-only in this harness.
- `pnpm-lock.yaml` SHA-256 before/after install attempt: `9acaa89d213b3281e757b6edf6fdb8727176570d725b78a0de234c61a7f3c332`.
- `origin/next` `pnpm-lock.yaml` SHA-256: `9acaa89d213b3281e757b6edf6fdb8727176570d725b78a0de234c61a7f3c332`.
- `git diff -- pnpm-lock.yaml` — rc=0 (unchanged).
- `git diff -- apps/web/src | grep -nE '^\+.*(fetch|io|api)\(\s*[\x27\"]https?://'` — rc=1 (empty, as required).
- `pnpm format:check` — rc=0.
- `git diff --check` — rc=0.
### Final worktree check
- `git status --short` shows only the authorized web files plus this scratchpad.
- `.mosaic/orchestrator/session.lock` remains unstaged.
- Current changed file set:
- `apps/web/package.json`
- `apps/web/src/routes.tsx`
- `apps/web/src/spa/routes.spec.tsx`
- `apps/web/vite.config.ts`
- `apps/web/src/spa/pages/page-errors.ts`
- `apps/web/src/spa/pages/page-fixtures.ts`
- `apps/web/src/spa/pages/project-detail.spec.tsx`
- `apps/web/src/spa/pages/project-detail.tsx`
- `apps/web/src/spa/pages/projects.spec.tsx`
- `apps/web/src/spa/pages/projects.tsx`
- `apps/web/src/spa/pages/resource-route-boundaries.spec.tsx`
- `apps/web/src/spa/pages/resource-route-error-boundaries.tsx`
- `apps/web/src/spa/pages/tasks.spec.tsx`
- `apps/web/src/spa/pages/tasks.tsx`
- `docs/scratchpads/webui-p4-1.md`
## P4-1 REV 2 Final Delivery Verification — 2026-08-10T22:23:00Z
### Spec/Security Review Verdicts (Independent Reviews)
Both independent reviews cleared P4-1 REV 2 implementation without blockers:
- **Spec review:** approved; 0 blockers, 0 should-fix findings
- **Code/security review:** approved; 0 blockers, 0 critical/high findings, 0 should-fix recommendations
### Final Fresh Gate Verification (This Session)
All verification gates executed sequentially with rc=0 (except where noted):
1. `pnpm --filter @mosaicstack/web typecheck` — rc=0
2. `pnpm --filter @mosaicstack/web lint` — rc=0
3. `pnpm --filter @mosaicstack/web test` — rc=0; **19 files / 153 tests** (baseline: 15 files / 133 tests; added: 4 files / 20 tests)
4. `pnpm --filter @mosaicstack/web build` — rc=0; Next production build completed, 10/10 static pages generated
5. `pnpm --filter @mosaicstack/web build:vite` — rc=0; Vite production bundle generated
6. Lockfile integrity:
- Before install: SHA-256 `9acaa89d213b3281e757b6edf6fdb8727176570d725b78a0de234c61a7f3c332`
- After frozen install: SHA-256 `9acaa89d213b3281e757b6edf6fdb8727176570d725b78a0de234c61a7f3c332`
- Equality: ✓ verified
- `git diff --quiet origin/next -- pnpm-lock.yaml` rc=0 ✓
7. `pnpm format:check` — rc=0; all matched files use Prettier code style
8. `git diff --check` — rc=0; no trailing whitespace or merged conflict markers
9. New-code origin check: `git diff -- apps/web/src | grep -nE '^\+.*(fetch|io|api)\(\s*[\x27\"]https?://'` — rc=1 (empty result, as required)
10. `pnpm --filter @mosaicstack/web exec vitest run src/spa/pages/projects.spec.tsx src/spa/pages/project-detail.spec.tsx src/spa/pages/tasks.spec.tsx src/spa/pages/resource-route-boundaries.spec.tsx src/spa/routes.spec.tsx` — rc=0; 5 files / 26 tests passed
### Scope Audit
**Authorized in-scope changes:**
-`apps/web/package.json` — dev port pins only
-`apps/web/vite.config.ts` — strictPort flag only
-`apps/web/src/routes.tsx` — route wiring with errorElements
-`apps/web/src/spa/pages/projects.tsx` — read-only SPA list
-`apps/web/src/spa/pages/project-detail.tsx` — read-only SPA detail
-`apps/web/src/spa/pages/tasks.tsx` — read-only SPA list/kanban
-`apps/web/src/spa/pages/resource-route-error-boundaries.tsx` — page-local error components
-`apps/web/src/spa/pages/*.spec.tsx` — fixture-backed tests (4 new)
-`docs/scratchpads/webui-p4-1.md` — append-only record
**Out-of-scope verification:**
- `.mosaic/orchestrator/session.lock` — not modified/staged ✓
- `pnpm-lock.yaml` — not modified ✓
- `apps/gateway/**` — not modified ✓
- `packages/**` — not modified ✓
- Chat SPA — not modified ✓
- Settings, admin, navigation, or auth flow — not modified ✓
### Next Action
Commit, run authorized queue guard, and push only `feat/webui-p4-1` (no PR, no merge).
-115
View File
@@ -1,115 +0,0 @@
# WebUI Phase P — File / Folder Structure & Migration Map
> **Status:** living document — first pass. Structure and increment status are verified against
> `next` as of merge `8c27024d`. Details (per-surface component inventories, exact route tables,
> test matrices) are still being fleshed out; extend the stub sections below rather than rewriting
> the verified structure.
## 1. What Phase P is
Phase P migrates the Mosaic **web UI** (`apps/web`) from the legacy **Next.js App Router** app to a
**Vite + React Router single-page app (SPA)** that the **Gateway serves same-origin** on
`:14242`. The RFC splits the work into **six increments (P1P6)**; the P1 PR title records this as
"increment 1/6".
The migration is deliberately **incremental and non-destructive**: the new SPA is built up
_beside_ the existing Next app, sharing one `apps/web/src/lib` networking/auth layer, until the
final cutover (P5) removes the Next tree. At every point in between, **both app trees exist in the
same package** — this is intentional, not drift.
## 2. Current tree on `next` (dual-app, transitional)
```
apps/web/
├── next.config.ts # legacy Next.js config (removed at P5)
├── vite.config.ts # SPA build + DEV proxy config (canonical from P5)
├── package.json # dev/build default to NEXT today; :vite variants opt in
└── src/
├── main.tsx # ── SPA entry (Vite)
├── routes.tsx # ── SPA React Router route table
├── spa/ # ── NEW SPA surfaces
│ ├── guards.tsx # guest / authenticated route guards
│ ├── pages/ # login, register, sso-callback (P2); chat + error boundary (P3)
│ └── chat/ # P3 typed chat: use-chat-connection, commands-panel,
│ # session-panel, message-transcript, tool-call-list, composer
├── lib/ # ── SHARED by BOTH trees (origin-relative networking + auth)
│ ├── api.ts # fetch wrapper — relative /api/...
│ ├── socket.ts # Socket.IO singleton — relative /chat
│ ├── auth-client.ts # BetterAuth client — relative /api/auth/...
│ ├── auth-redirect.ts # post-auth redirect resolution (protocol-relative rejected)
│ ├── chat-contract.ts # P3 typed chat wire contract (runtime-guarded)
│ ├── sso.ts · types.ts · cn.ts
├── app/ # ══ LEGACY Next.js App Router (removed at P5)
│ ├── (auth)/{login,register}/
│ ├── (dashboard)/{admin,chat,projects,projects/[id],settings,tasks}/
│ ├── auth/provider/[provider]/
│ └── layout.tsx · page.tsx · globals.css
├── components/ # ══ LEGACY Next component library (auth, chat, layout,
│ # projects, settings, tasks, ui) — ported into spa/ across P3/P4
└── providers/ # ══ theme-provider (legacy; SPA equivalent under providers)
```
Legend: `──` new SPA (keep), `══` legacy Next (removed at P5), shared `lib/` in the middle.
## 3. Networking / serving model (why it's same-origin)
- The SPA speaks **origin-relative paths only**: `/api/...`, `/api/auth/...`, `/chat`. No
`NEXT_PUBLIC_*` / `VITE_*` origin var, no hard-coded `http://localhost:14242` under
`apps/web/src`.
- **Dev:** `vite.config.ts` runs a dev-only proxy that forwards those paths to the Gateway (so the
SPA on its dev port and the Gateway on `:14242` behave as one origin).
- **Prod (target):** the SPA is **same-origin with the Gateway** — the Gateway serves the built
static bundle and the API/WS on `:14242`, so no proxy and no CORS. _(The Gateway does not serve
the web `dist` yet — adding that is the core of P5; see §5.)_
## 4. Build scripts (`apps/web/package.json`)
| Script | Today | Notes |
| ----------------------------- | -------------------------------------------- | ------------------------------ |
| `dev` | `next dev` | legacy dev server |
| `dev:vite` | `vite` | SPA dev server (+ dev proxy) |
| `build` | `node ../../scripts/build-web.mjs` | currently a **Next** build |
| `build:vite` | `vite build` | SPA production build → `dist/` |
| `lint` / `typecheck` / `test` | `eslint src` / `tsc --noEmit` / `vitest run` | tree-agnostic |
At **P5** the `:vite` variants become the defaults (`dev`→vite, `build`→vite build) and the Next
build path is retired.
## 5. Increment map (P1P6)
| # | Increment | Branch | Status |
| ------ | ---------------------------------------------------------------------------------------------------------------------------------------------- | ----------------------------- | ------------------------- |
| **P1** | Vite + React Router skeleton beside Next (entry, router, guards, vitest) | `feat/webui-p1-vite-skeleton` | ✅ merged — PR **#1143** |
| **P2** | SPA data layer + same-origin auth (login/register/SSO pages, guards, relative api/socket/auth-client) | `feat/webui-p2-data-auth` | ✅ merged — PR **#1144** |
| **P3** | Typed SPA **chat** (`spa/chat/*`, `chat-contract.ts`, chat page + error boundary) | `feat/webui-p3-chat` | 🚧 in progress (unmerged) |
| **P4** | Port **projects / tasks / settings / admin** dashboard surfaces into the SPA | _tbd_ | ⏳ not started |
| **P5** | **Cutover**: Gateway serves the Vite `dist` on `:14242`; flip `dev`/`build` to vite; **remove** the legacy Next `app/` tree + `next.config.ts` | _tbd_ | ⏳ not started |
| **P6** | CI / images (trails): build the SPA in CI, ship images | _tbd_ | ⏳ trails |
Each increment follows the same delivery pipeline: brief traceable to the RFC → author →
**independent** integrator verification (build+test+typecheck+lint) → **independent** code + security
review (author ≠ reviewer) → author remediates → branch + PR to `next`**independent** merge-gate
merges. Author self-reports are not trusted; every gate is re-derived independently.
## 6. Known dependency / blocker
- **Issue #1145 — Gateway `dist` boot is broken** (DI failure on a defaulted constructor param);
the Gateway currently runs **dev-mode only**. This is a **hard precondition for P5**: the Gateway
cannot serve the SPA `dist` on `:14242` until `dist` boot works. P3/P4 remain on the dev-proxy
topology meanwhile.
## 7. Not part of Phase P (disambiguation)
`docs/plans/2026-08-09-webui-fleet-claude-bridge.md` and
`docs/scratchpads/webui-fleet-bridge-plan.md` describe a **separate** WebUI ↔ fleet/Claude bridge
effort. They are **not** the Phase P SPA migration and should not be conflated with the increments
above.
## 8. Where the detail lives (extend these)
- Per-increment working notes: `docs/scratchpads/webui-p*-*.md` (e.g. `webui-p2-data-auth.md`).
- _Stub — to flesh out:_ per-surface component inventory (which `components/*` port to which
`spa/*`), the full SPA route table, the P5 cutover checklist, and the P6 CI/image plan.
@@ -193,19 +193,16 @@ describe('Unified wizard (runWizard with default skipGateway)', () => {
'Your timezone': 'UTC',
});
await expect(
runWizard({
mosaicHome: tmpDir,
sourceDir: tmpDir,
prompter,
configService: createConfigService(tmpDir, tmpDir),
skipGatewayNpmInstall: true,
}),
).rejects.toThrow('Gateway configuration failed');
await runWizard({
mosaicHome: tmpDir,
sourceDir: tmpDir,
prompter,
configService: createConfigService(tmpDir, tmpDir),
skipGatewayNpmInstall: true,
});
const logs = prompter.getLogs();
expect(logs.some((line) => line.includes('Gateway did not become healthy'))).toBe(true);
expect(logs.some((line) => line.includes('Gateway configuration failed'))).toBe(true);
expect(logs.some((line) => line.includes('Installation Summary'))).toBe(false);
expect(logs.some((line) => line.includes('Mosaic is ready.'))).toBe(false);
expect(gatewayConfigMock).toHaveBeenCalledTimes(1);
@@ -39,7 +39,6 @@ overwritten on upgrade. (Layer model: `constitution/LAYER-MODEL.md`.)
| TypeScript strict typing | `guides/TYPESCRIPT.md` |
| QA / test strategy | `guides/QA-TESTING.md` |
| Documentation (any code/API/auth/infra change) | `guides/DOCUMENTATION.md` |
| Writing style (docs, comms, any prose) | `guides/WRITING-STYLE.md` |
| Secrets / vault usage | `guides/VAULT-SECRETS.md` |
| Tool/credential reference (service CLIs, wrappers) | `guides/TOOLS-REFERENCE.md` |
| Memory protocol (OpenBrain capture/recall) | `guides/MEMORY.md` |
@@ -27,14 +27,6 @@ Master/slave model:
- Do not perform destructive git/file actions without explicit instruction.
- Browser automation (Playwright, Cypress, Puppeteer) MUST run in headless mode. Never launch a visible browser — it collides with the user's display and active session.
### Output standards (writing + code)
- Technical documentation follows **MOS-STE** (Mosaic Simplified Technical English — an adapted ASD-STE100 profile): short sentences, one instruction per sentence, active voice, one word per meaning, one term per concept. Full rules: `~/.config/mosaic/guides/WRITING-STYLE.md`.
- Apply MOS-STE **hardest to verification artifacts** (acceptance criteria, witness predicates, gate/alarm conditions). There an ambiguous term produces a false green, not just a confused reader.
- Source code follows the **Google Style Guide** for the language.
- User-facing comms follow the user's declared `communicationStyle` in `USER.md` "Communication Preferences" (`direct` | `friendly` | `formal`, default `direct`); `guides/WRITING-STYLE.md` §5 maps each value to output. The documentation standard does not change with user preference.
- **Carve-out:** MOS-STE does NOT apply to content that must carry a specific human voice (letters, personal or marketing prose, voice-matched output). A declared voice profile wins.
### Secrets handling (HARD RULE)
- Vault is the canonical source-of-truth for every secret in every environment. No exceptions.
@@ -1,134 +0,0 @@
# Writing Style Standard — MOS-STE (MANDATORY)
This guide defines how agents write. It sets one style standard per output type.
It is written in the standard it defines, as a worked example.
**Adapted, not compliant.** MOS-STE (Mosaic Simplified Technical English) is an
adapted profile of ASD-STE100. Mosaic does not license or certify against
ASD-STE100. Mosaic uses the load-bearing rules and fits them to agent work. This
is the same stance Mosaic takes toward DO-178B/C: use the rigor, do not claim the
certification.
## Scope — which standard governs which output
| Output type | Standard |
| ------------------------------------------------------------------------------------------------------------- | --------------------------------------------------------------------------- |
| Technical documentation (READMEs, runbooks, PRDs, procedures, ADRs, guides, acceptance criteria, design docs) | **MOS-STE** (this guide) |
| Source code and code comments | **Google Style Guide** for the language (§4) |
| Inter-agent comms | MOS-STE by default (concise, structured) |
| User-facing comms | **Per-user style choice** — read `USER.md` "Communication Preferences" (§5) |
| End-user prose the user owns (marketing, letters, personal writing, voice-matched content) | The user's declared voice. MOS-STE does NOT apply. |
**The user-voice carve-out is absolute.** Do not apply MOS-STE to content that
must carry a specific human voice (for example a cover letter, a personal
message, or marketing copy). That content needs the user's voice. MOS-STE would
damage it. When a project declares a voice profile, that profile wins.
## 1. Why one standard
Agent documentation drifts across projects. Different agents use different terms,
sentence styles, and structures for the same concept. Readers lose time.
Assumptions hide in ambiguous prose. One standard gives agents a clear target. It
gives reviewers a clear test.
## 2. Where MOS-STE matters most — verification artifacts
Apply MOS-STE hardest to acceptance criteria, witness predicates, gate
definitions, and alarm conditions. In prose, an ambiguous term produces a
confused reader. In a verification artifact, an ambiguous term produces a false
green — a check that passes without testing the claim.
The one-term-one-concept rule (rule 9) is the guard. When one word names two
concepts in one predicate, the check can test the wrong concept and still pass.
**Worked failure.** A rename used a witness predicate with three clauses: ref A
present, ref B absent, tip committed from this host. Every clause tested the git
_ref_ (the channel). The claim under test was about a _field inside the payload_.
The word "beacon" named two concepts in one sentence. Deleting ref B was the next
scheduled step. That step flips the last clause green and certifies a state in
which the payload still names the wrong host. The predicate was one planned action
away from a false green on its normal path. The payload field was never tested.
Rule: when N failure modes share one observable, the observable is not a
diagnostic. In a verification artifact, that ambiguity does not confuse a reader —
it certifies the defect.
## 3. MOS-STE rules
### 3.1 Sentence rules
1. Keep sentences short. Use 20 words or fewer for a procedure. Use 25 words or
fewer for a description. (Reasoning and doctrine prose relaxes this limit —
see §3.4. A future lint enforces §3.1, not §3.4.)
2. Write one instruction per sentence. In a procedure, give one command per step.
3. Use the active voice. Write "Run the script." Do not write "The script should
be run."
4. Use the imperative for instructions. Start the sentence with the verb.
5. Use simple verb tenses. Prefer the present tense. Avoid the perfect and
progressive tenses when a simple tense works.
6. Do not use an `-ing` form when it makes the meaning unclear.
7. Write positive statements. State what to do, not only what to avoid.
### 3.2 Word rules
8. Use one word for one meaning. Do not use the same word in two senses.
9. Use one term for one concept. Do not use synonyms for variety. Example: choose
`secret`, `credential`, or `key` for each concept, and keep it.
10. Use articles (`a`, `the`). Do not drop words to save space.
11. Keep an approved-terms glossary per project. Add each domain noun and each
chosen verb. Technical names (for example `Vault`, `cgroup`, `systemd`) are
always allowed.
12. Define an abbreviation at its first use. Then use it consistently.
### 3.3 Structure rules
13. Use a list for parallel items or sequential steps. Do not put them in one long
sentence.
14. Use a table for data with more than two dimensions.
15. Use parallel structure in headings and steps.
16. Repeat the noun. Do not use a pronoun when the reference is unclear.
### 3.4 Adaptation notes (where MOS-STE deviates from ASD-STE100, and why)
- **No licensed dictionary.** ASD-STE100 ships a controlled dictionary under
copyright. MOS-STE uses per-project glossaries instead (rule 11).
- **Domain terms are allowed.** MOS-STE keeps every term the work needs.
- **Reasoning prose gets structure, not amputation.** Apply the sentence and word
rules to design and doctrine writing. Allow the length a subtle argument needs.
Readable-first beats rule-strict when the two conflict.
## 4. Code — Google Style Guide
Write source code to the Google Style Guide for the language (Python, TypeScript,
Shell, Go, and so on). Match the existing file when a local convention already
exists. Keep code comments to the MOS-STE sentence and word rules.
## 5. User-facing comms — a per-user choice
Mosaic is multi-user. Different users want different comms styles. The framework
already carries the selectable setting: `communicationStyle` (`direct` |
`friendly` | `formal`, default `direct`). `mosaic init` writes it, and the
builder renders it into the generated `USER.md` "Communication Preferences"
section. This guide adds the OUTPUT meaning of each value; do not invent new
values.
The builder renders the style as prose bullets, not the token name, so match on
the leading bullet the generated `USER.md` actually contains:
| `USER.md` leading bullet | Style | User-facing output |
| ----------------------------- | ------------------ | ---------------------------------------------------------------------- |
| "Direct and concise" | `direct` (default) | MOS-STE structure — short, active, defined terms, tables for overview. |
| "Warm and conversational" | `friendly` | Warmer register. Full sentences, explain reasoning, fewer tables. |
| "Professional and structured" | `formal` | Professional and structured. Thorough, with explicit recommendations. |
This setting governs **user-facing comms only**. It does not change the
documentation standard (§3), which is always MOS-STE regardless of the value.
## 6. Enforcement
- **Now:** human review only. **No mechanical prose check exists today.** The
pre-push gate runs typecheck, lint, build, and tests; it inspects no prose.
Reviewers check output against the scope table and the MOS-STE rules by hand.
- **Future:** an MOS-STE lint check (built from the §3.1 sentence rules) and a
Google-style linter in the pre-push gate. A future linter enforces §3.1, not
§3.4 — see the note at rule 1.
@@ -1 +0,0 @@
Mosaic lease promotion was processed mechanically; no action is needed.
@@ -32,18 +32,6 @@
]
}
],
"UserPromptSubmit": [
{
"matcher": "^/mosaic-promote$",
"hooks": [
{
"type": "command",
"command": "python3 ~/.config/mosaic/tools/lease-broker/promote-begin.py",
"timeout": 15
}
]
}
],
"PreToolUse": [
{
"matcher": ".*",
@@ -93,8 +81,8 @@
"hooks": [
{
"type": "command",
"command": "python3 ~/.config/mosaic/tools/lease-broker/receipt-observer-client.py --runtime claude --latest-entry; observer_status=$?; python3 ~/.config/mosaic/tools/lease-broker/promote-complete.py; exit $observer_status",
"timeout": 15
"command": "python3 ~/.config/mosaic/tools/lease-broker/receipt-observer-client.py --runtime claude --latest-entry",
"timeout": 3
},
{
"type": "command",
@@ -99,7 +99,7 @@ prompt_if_empty() {
if [[ $NON_INTERACTIVE -eq 1 ]]; then
if [[ -n "$default_value" ]]; then
printf -v "$var_name" %s "$default_value"
eval "$var_name=\"$default_value\""
return
fi
echo "[mosaic-init] ERROR: --$var_name is required in non-interactive mode" >&2
@@ -115,7 +115,7 @@ prompt_if_empty() {
if [[ -z "$value" && -n "$default_value" ]]; then
value="$default_value"
fi
printf -v "$var_name" %s "$value"
eval "$var_name=\"$value\""
}
prompt_multiline() {
@@ -129,7 +129,7 @@ prompt_multiline() {
fi
if [[ $NON_INTERACTIVE -eq 1 ]]; then
printf -v "$var_name" %s "$default_value"
eval "$var_name=\"$default_value\""
return
fi
@@ -139,7 +139,7 @@ prompt_multiline() {
if [[ -z "$value" ]]; then
value="$default_value"
fi
printf -v "$var_name" %s "$value"
eval "$var_name=\"$value\""
}
# ── Existing file detection ────────────────────────────────────
@@ -233,14 +233,6 @@ for runtime_file in \
copy_file_managed "$src" "$HOME/.claude/$runtime_file"
done
if [[ -d "$MOSAIC_HOME/runtime/claude/commands" ]]; then
mkdir -p "$HOME/.claude/commands"
for command_file in "$MOSAIC_HOME/runtime/claude/commands/"*; do
[[ -f "$command_file" ]] || continue
copy_file_managed "$command_file" "$HOME/.claude/commands/$(basename "$command_file")"
done
fi
# OpenCode runtime adapter (thin pointer to AGENTS.md)
opencode_adapter="$MOSAIC_HOME/runtime/opencode/AGENTS.md"
if [[ -f "$opencode_adapter" ]]; then
@@ -153,24 +153,7 @@ if [[ $link_only -eq 1 ]]; then
exit 0
fi
# Skills are linked into the MOSAIC-OWNED harness homes, never a base install.
# Paths mirror the config-dir env vars the launcher injects (HARNESS_HOME_ENV in
# commands/launch.js):
# claude CLAUDE_CONFIG_DIR -> <home>/skills
# pi PI_CODING_AGENT_DIR -> <home>/skills (replaces ~/.pi/agent)
# codex CODEX_HOME -> <home>/skills
# opencode XDG_CONFIG_HOME -> <home>/opencode/skills (XDG adds a level)
link_targets=(
"$MOSAIC_HOME/.claude/skills"
"$MOSAIC_HOME/.codex/skills"
"$MOSAIC_HOME/.opencode/opencode/skills"
"$MOSAIC_HOME/.pi/skills"
)
# Pre-isolation installs planted the same symlink farm directly in the operator's
# base installs. Those are now orphaned: the launcher no longer reads them, but
# they persist and make a "clean" base install look mosaic-managed.
legacy_link_targets=(
"$HOME/.claude/skills"
"$HOME/.codex/skills"
"$HOME/.config/opencode/skills"
@@ -262,72 +245,13 @@ prune_stale_links_in_target() {
# -m resolves lexical dangling targets too. If resolution fails, ownership
# is unproven and the link must be preserved.
resolved="$(readlink -m "$link_path" 2>/dev/null || true)"
# $canonical_real must be length-checked BEFORE use as a prefix: if it were
# ever empty, "$resolved" == "$canonical_real/"* collapses to == "/"* and
# matches every absolute path. Combined with the is_mosaic_skill_name skip
# above, that inverts the function precisely — it would delete exactly the
# FOREIGN symlinks and keep the mosaic ones. (#1087, reported by mos-claude.)
if [[ -n "$resolved" && -n "$canonical_real" && "$resolved" == "$canonical_real/"* ]]; then
if [[ -n "$resolved" && "$resolved" == "$canonical_real/"* ]]; then
rm -f "$link_path"
echo "[mosaic-skills] Removed stale retired skill link: $link_path"
fi
done < <(find "$target_dir" -mindepth 1 -maxdepth 1 -type l -print0)
}
# Remove mosaic-owned symlinks left in a base install by a pre-isolation sync.
#
# Ownership is proven by RESOLUTION, not by name: only links resolving inside the
# canonical or local skills dirs are removed. Anything else — a real directory, a
# link elsewhere, an unresolvable link — is left untouched. This mirrors the
# refusal in commands/skill.js ("only symlinks pointing inside the Mosaic skills
# directory are managed") and preserves e.g. codex's own `.system` dir.
#
# The directory itself is kept: mosaic-doctor warns when ~/.pi/agent/skills is
# missing, and an empty dir is the correct end state, not an absent one.
cleanup_legacy_target() {
local target_dir="$1"
local removed=0 kept=0
[[ -d "$target_dir" ]] || return 0
while IFS= read -r -d '' link_path; do
local resolved owned=0
resolved="$(readlink -m "$link_path" 2>/dev/null || true)"
# Guard the empty-prefix trap: an unset *_real would make "$resolved" == "/"*
# match every absolute path and delete foreign links.
if [[ -n "$resolved" ]]; then
if [[ -n "$canonical_real" && "$resolved" == "$canonical_real/"* ]]; then
owned=1
elif [[ -n "$local_real" && "$resolved" == "$local_real/"* ]]; then
owned=1
fi
fi
if [[ $owned -eq 1 ]]; then
rm -f "$link_path"
removed=$((removed + 1))
else
kept=$((kept + 1))
fi
done < <(find "$target_dir" -mindepth 1 -maxdepth 1 -type l -print0)
if [[ $removed -gt 0 ]]; then
echo "[mosaic-skills] Legacy cleanup: removed $removed mosaic symlink(s) from $target_dir (preserved $kept foreign)"
fi
}
for legacy in "${legacy_link_targets[@]}"; do
# Skip anything that is also a current target, so isolation can never
# self-destruct if the two lists ever overlap.
skip=0
for target in "${link_targets[@]}"; do
[[ "$legacy" == "$target" ]] && skip=1
done
[[ $skip -eq 1 ]] && continue
cleanup_legacy_target "$legacy"
done
for target in "${link_targets[@]}"; do
mkdir -p "$target"
@@ -1,21 +0,0 @@
#!/usr/bin/env bash
set -euo pipefail
# Source only the prompt helpers; executing mosaic-init itself requires templates.
source <(head -n 144 "$(dirname "$0")/mosaic-init")
rm -f /tmp/pwned
payload='literal "$(touch /tmp/pwned)"'
AGENT_NAME=""
prompt_if_empty AGENT_NAME "Agent name" <<<"$payload"
[[ "$AGENT_NAME" == "$payload" ]] || {
echo "FAIL: prompt answer did not round-trip literally" >&2
exit 1
}
[[ ! -e /tmp/pwned ]] || {
echo "FAIL: prompt answer executed code" >&2
rm -f /tmp/pwned
exit 1
}
echo "mosaic-init RCE regression: PASS"
+58 -468
View File
@@ -1,6 +1,6 @@
#!/bin/bash
# pr-merge.sh - Merge pull requests on Gitea or GitHub
# Usage: pr-merge.sh -n PR_NUMBER [-m squash] [-d] [--expect-head SHA] [--co-author-trailers --escalate-to PRINCIPAL]
# Usage: pr-merge.sh -n PR_NUMBER [-m squash] [-d]
set -euo pipefail
@@ -14,8 +14,6 @@ MERGE_METHOD="squash"
DELETE_BRANCH=false
DRY_RUN=false
EXPECT_HEAD=""
CO_AUTHOR_TRAILERS=false
ESCALATE_TO=""
usage() {
cat <<EOF
@@ -29,16 +27,12 @@ Options:
-d, --delete-branch Delete the head branch after merge
--dry-run Run metadata/login preflight without merging
--expect-head SHA Refuse unless the PR head matches this full commit SHA
--co-author-trailers Build verified trailers from linked PR commit authors
--escalate-to NAME Named principal for an unresolved-author BLOCK
-h, --help Show this help message
Examples:
$(basename "$0") -n 42 # Merge PR #42
$(basename "$0") -n 42 -m squash # Squash merge
$(basename "$0") -n 42 -d # Squash merge and delete branch
$(basename "$0") -n 42 --expect-head 0123456789abcdef0123456789abcdef01234567
$(basename "$0") -n 42 --co-author-trailers --escalate-to tl-mosaic
EOF
exit "${1:-1}"
}
@@ -63,25 +57,9 @@ while [[ $# -gt 0 ]]; do
shift
;;
--expect-head)
if [[ $# -lt 2 ]]; then
echo "Error: --expect-head requires one full commit SHA." >&2
exit 1
fi
EXPECT_HEAD="$2"
shift 2
;;
--co-author-trailers)
CO_AUTHOR_TRAILERS=true
shift
;;
--escalate-to)
if [[ $# -lt 2 ]]; then
echo "Error: --escalate-to requires one principal name." >&2
exit 1
fi
ESCALATE_TO="$2"
shift 2
;;
-h|--help)
usage 0
;;
@@ -110,30 +88,17 @@ if [[ -n "$EXPECT_HEAD" && ! "$EXPECT_HEAD" =~ ^[0-9a-fA-F]{40}$ ]]; then
echo "Error: --expect-head must be a full 40-character hexadecimal commit SHA." >&2
exit 1
fi
if [[ "$CO_AUTHOR_TRAILERS" == true && -z "$ESCALATE_TO" ]]; then
echo "Error: --co-author-trailers requires --escalate-to with a named principal." >&2
exit 1
fi
if [[ -n "$ESCALATE_TO" && ! "$ESCALATE_TO" =~ ^[A-Za-z0-9_.-]+$ ]]; then
echo "Error: --escalate-to must be one exact principal name." >&2
exit 1
fi
if [[ "$CO_AUTHOR_TRAILERS" != true && -n "$ESCALATE_TO" ]]; then
echo "Error: --escalate-to is valid only with --co-author-trailers." >&2
exit 1
fi
PR_METADATA="$("$SCRIPT_DIR/pr-metadata.sh" -n "$PR_NUMBER")"
BASE_BRANCH="$(printf '%s' "$PR_METADATA" | python3 -c 'import json, sys; print((json.load(sys.stdin).get("baseRefName") or "").strip())')"
HEAD_BRANCH="$(printf '%s' "$PR_METADATA" | python3 -c 'import json, sys; print((json.load(sys.stdin).get("headRefName") or "").strip())')"
HEAD_SHA="$(printf '%s' "$PR_METADATA" | python3 -c 'import json, sys; print((json.load(sys.stdin).get("headRefOid") or "").strip())')"
HEAD_REPO="$(printf '%s' "$PR_METADATA" | python3 -c 'import json, sys; value=json.load(sys.stdin).get("headRepository") or ""; print((value.get("nameWithOwner") or value.get("full_name") or "") if isinstance(value, dict) else str(value).strip())')"
PR_TITLE="$(printf '%s' "$PR_METADATA" | python3 -c 'import json, sys; print((json.load(sys.stdin).get("title") or "").strip())')"
PR_AUTHOR="$(printf '%s' "$PR_METADATA" | python3 -c 'import json, sys; value=json.load(sys.stdin).get("author") or ""; print((value.get("login") or "").strip() if isinstance(value, dict) else str(value).strip())')"
if [[ "$BASE_BRANCH" != "main" && "$BASE_BRANCH" != "next" ]]; then
echo "Error: Mosaic policy allows merges only for PRs targeting 'main' or 'next' (found '$BASE_BRANCH')." >&2
exit 1
fi
if [[ -z "$HEAD_BRANCH" || -z "$HEAD_REPO" || ! "$HEAD_SHA" =~ ^[0-9a-fA-F]{40}$ ]]; then
echo "Error: Could not resolve the PR head branch, repository, and full commit SHA for queue inspection." >&2
exit 1
@@ -157,442 +122,70 @@ PLATFORM=$(detect_platform)
OWNER=$(get_repo_owner)
REPO=$(get_repo_name)
write_curl_auth_config() {
local mode="$1" credential="$2"
printf '%s' "$credential" | python3 -c '
import sys
mode = sys.argv[1]
credential = sys.stdin.read()
if not credential or any(char in credential for char in "\r\n"):
raise SystemExit(1)
escaped = credential.replace("\\", "\\\\").replace("\"", "\\\"")
if mode == "token":
print(f"header = \"Authorization: token {escaped}\"")
elif mode == "basic":
print(f"user = \"{escaped}\"")
else:
raise SystemExit(1)
' "$mode"
}
LAST_GITEA_HTTP_CODE="000"
LAST_GITEA_ERROR=""
MERGE_TEMP_DIRS=()
GITEA_CURL_MAX_BYTES="${MOSAIC_GITEA_CURL_MAX_BYTES:-1048576}"
GITEA_CURL_MAX_TIME="${MOSAIC_GITEA_CURL_MAX_TIME_SEC:-30}"
GITEA_CURL_CONNECT_TIMEOUT="${MOSAIC_GITEA_CURL_CONNECT_TIMEOUT_SEC:-10}"
for bound in "$GITEA_CURL_MAX_BYTES" "$GITEA_CURL_MAX_TIME" "$GITEA_CURL_CONNECT_TIMEOUT"; do
if [[ ! "$bound" =~ ^[1-9][0-9]*$ ]]; then
echo "Error: Gitea curl bounds must be positive integers; refusing request." >&2
exit 1
fi
done
GITEA_CURL_BOUNDS=(
--max-filesize "$GITEA_CURL_MAX_BYTES"
--max-time "$GITEA_CURL_MAX_TIME"
--connect-timeout "$GITEA_CURL_CONNECT_TIMEOUT"
)
format_gitea_error_response() {
local response_file="$1"
python3 - "$response_file" <<'PY'
import json
import sys
with open(sys.argv[1], "rb") as handle:
raw = handle.read(65536)
try:
response = json.loads(raw.decode("utf-8", errors="replace"))
except (UnicodeDecodeError, json.JSONDecodeError):
message = "non-JSON response omitted"
else:
if isinstance(response, dict):
message = response.get("message") or response.get("error")
if not message and response.get("errors") is not None:
message = json.dumps(response["errors"], separators=(",", ":"))
else:
message = None
if not message:
message = "JSON response contained no error message"
message = str(message)
if len(message) > 500:
message = message[:500] + "..."
print(ascii(message))
PY
}
cleanup_merge_temp_dirs() {
local path
for path in "${MERGE_TEMP_DIRS[@]}"; do
[[ -n "$path" ]] && rm -rf -- "$path"
done
}
trap cleanup_merge_temp_dirs EXIT
trap 'exit 130' INT
trap 'exit 143' TERM
fetch_gitea_pr_head() {
local host="$1" auth_mode="$2" credential="$3" work_root="$4"
local response_file raw_code api_url auth_config curl_rc
response_file=$(mktemp "$work_root/pr-merge-pr.XXXXXX")
api_url="https://${host}/api/v1/repos/${OWNER}/${REPO}/pulls/${PR_NUMBER}"
if ! auth_config=$(write_curl_auth_config "$auth_mode" "$credential"); then
echo "Error: Could not construct Gitea authentication config; refusing request." >&2
rm -f "$response_file"
return 1
fi
raw_code=$(curl -sS -K - "${GITEA_CURL_BOUNDS[@]}" -w '%{http_code}' -o "$response_file" \
-H "User-Agent: curl/8" "$api_url" <<<"$auth_config")
curl_rc=$?
LAST_GITEA_HTTP_CODE="${raw_code:-000}"
if [[ "$curl_rc" -ne 0 ]]; then
LAST_GITEA_ERROR="curl transport failed (rc=$curl_rc)"
rm -f "$response_file"
return 1
fi
if [[ ! "$raw_code" =~ ^2 ]]; then
LAST_GITEA_ERROR=$(format_gitea_error_response "$response_file")
rm -f "$response_file"
return 1
fi
if ! python3 - "$response_file" <<'PY'
import json
import re
import sys
with open(sys.argv[1], encoding="utf-8") as handle:
pull = json.load(handle)
head = pull.get("head") if isinstance(pull, dict) else None
sha = str(head.get("sha") or "") if isinstance(head, dict) else ""
if not re.fullmatch(r"[0-9a-fA-F]{40}", sha):
raise SystemExit(1)
print(sha)
PY
then
echo "Error: Gitea PR response has no valid head SHA; refusing merge." >&2
rm -f "$response_file"
return 1
fi
rm -f "$response_file"
}
fetch_gitea_pr_commits() {
local host="$1" auth_mode="$2" credential="$3" work_root="$4"
local page page_file combined_file merged_file raw_code page_count api_url auth_config curl_rc
mkdir -p "$work_root"
if ! auth_config=$(write_curl_auth_config "$auth_mode" "$credential"); then
echo "Error: Could not construct Gitea authentication config; refusing request." >&2
return 1
fi
combined_file=$(mktemp "$work_root/pr-merge-commits.XXXXXX")
printf '[]' > "$combined_file"
page=1
while true; do
page_file=$(mktemp "$work_root/pr-merge-commits-page.XXXXXX")
api_url="https://${host}/api/v1/repos/${OWNER}/${REPO}/pulls/${PR_NUMBER}/commits?limit=50&page=${page}"
raw_code=$(curl -sS -K - "${GITEA_CURL_BOUNDS[@]}" -w '%{http_code}' -o "$page_file" \
-H "User-Agent: curl/8" "$api_url" <<<"$auth_config")
curl_rc=$?
LAST_GITEA_HTTP_CODE="${raw_code:-000}"
if [[ "$curl_rc" -ne 0 ]]; then
LAST_GITEA_ERROR="curl transport failed (rc=$curl_rc)"
rm -f "$page_file" "$combined_file"
return 1
fi
if [[ ! "$raw_code" =~ ^2 ]]; then
LAST_GITEA_ERROR=$(format_gitea_error_response "$page_file")
rm -f "$page_file" "$combined_file"
return 1
fi
if ! page_count=$(python3 - "$page_file" <<'PY'
import json
import sys
with open(sys.argv[1], encoding="utf-8") as handle:
page = json.load(handle)
if not isinstance(page, list):
raise SystemExit(1)
print(len(page))
PY
); then
echo "Error: Gitea PR commits response is not a JSON array; refusing merge." >&2
rm -f "$page_file" "$combined_file"
return 1
fi
merged_file=$(mktemp "$work_root/pr-merge-commits-merged.XXXXXX")
if ! python3 - "$combined_file" "$page_file" > "$merged_file" <<'PY'
import json
import sys
with open(sys.argv[1], encoding="utf-8") as handle:
combined = json.load(handle)
with open(sys.argv[2], encoding="utf-8") as handle:
page = json.load(handle)
json.dump(combined + page, sys.stdout, separators=(",", ":"))
PY
then
echo "Error: Could not combine paginated PR commit metadata; refusing merge." >&2
rm -f "$page_file" "$combined_file" "$merged_file"
return 1
fi
mv "$merged_file" "$combined_file"
rm -f "$page_file"
if [[ "$page_count" -lt 50 ]]; then
break
fi
page=$((page + 1))
if [[ "$page" -gt 1000 ]]; then
echo "Error: PR commit pagination exceeded 1000 pages; refusing merge." >&2
rm -f "$combined_file"
return 1
fi
done
cat "$combined_file"
rm -f "$combined_file"
}
# LIMITATION: author.login resolution proves the commit address maps to a registered account.
# It does NOT prove the named principal authored the commit — git author metadata is self-asserted.
# This gate checks ATTRIBUTION LINKAGE, not AUTHORSHIP. Commit signing is out of scope and unadopted.
build_coauthor_message_fields() {
local commits_file="$1" context_file="$2" head_file="$3"
python3 - "$commits_file" "$context_file" "$head_file" <<'PY'
import json
import re
import sys
commits_path, context_path, head_path = sys.argv[1:]
with open(commits_path, encoding="utf-8") as handle:
commits = json.load(handle)
head_sha = open(head_path, encoding="utf-8").read().strip()
context_parts = open(context_path, "rb").read().split(b"\0")
if len(context_parts) != 4 or context_parts[-1] != b"":
raise SystemExit(1)
poster, title, principal = (part.decode("utf-8") for part in context_parts[:3])
if not isinstance(commits, list) or not commits:
print(
f"BLOCK: provider returned no PR commits; author identity is unmeasurable. "
f"Refusing merge; escalate to named principal '{principal}'.",
file=sys.stderr,
)
raise SystemExit(75)
if not poster:
print(
f"BLOCK: PR poster login is empty; refusing merge; "
f"escalate to named principal '{principal}'.",
file=sys.stderr,
)
raise SystemExit(75)
if not re.fullmatch(r"[0-9a-fA-F]{40}", head_sha):
print(
f"BLOCK: inspected PR head SHA is invalid; refusing merge; "
f"escalate to named principal '{principal}'.",
file=sys.stderr,
)
raise SystemExit(75)
seen = set()
trailers = []
head_seen = False
for item in commits:
if not isinstance(item, dict):
print(f"BLOCK: malformed PR commit metadata; escalate to named principal '{principal}'.", file=sys.stderr)
raise SystemExit(75)
sha = str(item.get("sha") or "<unknown>")
if sha == head_sha:
head_seen = True
commit = item.get("commit") if isinstance(item.get("commit"), dict) else {}
commit_author = commit.get("author") if isinstance(commit.get("author"), dict) else {}
email = str(commit_author.get("email") or "").strip()
provider_author = item.get("author") if isinstance(item.get("author"), dict) else {}
login = str(provider_author.get("login") or "").strip()
if not login:
diagnostic_email = email or "<missing>"
print(
f"BLOCK: commit {sha!r} has author.login=NULL while "
f"commit.author.email={diagnostic_email!r}; refusing merge; "
f"escalate to named principal '{principal}'.",
file=sys.stderr,
)
raise SystemExit(75)
if (
not email.isascii()
or not email.isprintable()
or not re.fullmatch(r"[A-Za-z0-9_.-]+", login)
or not re.fullmatch(r"[^<>\s]+@[^<>\s]+", email)
):
print(
f"BLOCK: commit {sha!r} has unusable linked identity "
f"author.login={login!r}, commit.author.email={email!r}; refusing merge; "
f"escalate to named principal '{principal}'.",
file=sys.stderr,
)
raise SystemExit(75)
if login == poster or login in seen:
continue
seen.add(login)
trailers.append(f"Co-authored-by: {login} <{email}>")
if not head_seen:
print(
f"BLOCK: inspected PR head is absent from commit enumeration; refusing merge; "
f"escalate to named principal '{principal}'.",
file=sys.stderr,
)
raise SystemExit(75)
if not trailers:
print("{}")
raise SystemExit(0)
if not title:
print(
f"BLOCK: PR title is empty; refusing merge; escalate to named principal '{principal}'.",
file=sys.stderr,
)
raise SystemExit(75)
if not title.isprintable() or re.match(r"^[A-Za-z-]+-[Bb]y:", title):
print(
f"BLOCK: PR title is not one printable, non-trailer line; refusing merge; "
f"escalate to named principal '{principal}'.",
file=sys.stderr,
)
raise SystemExit(75)
print(json.dumps({
"MergeTitleField": title,
"MergeMessageField": "\n".join(trailers),
}, separators=(",", ":")))
PY
}
merge_gitea_api_attempt() {
local host="$1" auth_mode="$2" credential="$3"
local api_url attempt_dir body_file raw_code commits_file fields_file context_file head_file payload_file work_root attempt_rc auth_config curl_rc
LAST_GITEA_HTTP_CODE="000"
LAST_GITEA_ERROR=""
merge_gitea_with_api() {
local host="$1" api_url token basic_auth body_file raw_code payload
api_url="https://${host}/api/v1/repos/${OWNER}/${REPO}/pulls/${PR_NUMBER}/merge"
work_root="${AGENT_WORK_ROOT:-${HOME:-/tmp}/mosaic/agent-work}"
mkdir -p "$work_root"
attempt_dir=$(mktemp -d "$work_root/pr-merge-attempt.XXXXXX")
chmod 0700 "$attempt_dir"
MERGE_TEMP_DIRS+=("$attempt_dir")
body_file=$(mktemp "$attempt_dir/api-response.XXXXXX")
fields_file=$(mktemp "$attempt_dir/message-fields.XXXXXX")
payload_file=$(mktemp "$attempt_dir/payload.XXXXXX")
printf '{}' > "$fields_file"
if [[ "$CO_AUTHOR_TRAILERS" == true ]]; then
commits_file=$(mktemp "$attempt_dir/pr-merge-commits-input.XXXXXX")
context_file=$(mktemp "$attempt_dir/pr-merge-message-context.XXXXXX")
head_file=$(mktemp "$attempt_dir/pr-merge-head-input.XXXXXX")
printf '%s\0%s\0%s\0' "$PR_AUTHOR" "$PR_TITLE" "$ESCALATE_TO" > "$context_file"
if fetch_gitea_pr_head "$host" "$auth_mode" "$credential" "$attempt_dir" > "$head_file"; then
:
else
attempt_rc=$?
rm -f "$body_file" "$fields_file" "$payload_file" "$commits_file" "$context_file" "$head_file"
return "$attempt_rc"
fi
if [[ "$(<"$head_file")" != "$HEAD_SHA" ]]; then
echo "BLOCK: authenticated PR head moved from reviewed $HEAD_SHA to $(<"$head_file"); refusing merge; escalate to named principal '$ESCALATE_TO'." >&2
rm -f "$body_file" "$fields_file" "$payload_file" "$commits_file" "$context_file" "$head_file"
return 75
fi
if fetch_gitea_pr_commits "$host" "$auth_mode" "$credential" "$attempt_dir" > "$commits_file"; then
:
else
attempt_rc=$?
rm -f "$body_file" "$fields_file" "$payload_file" "$commits_file" "$context_file" "$head_file"
return "$attempt_rc"
fi
if build_coauthor_message_fields "$commits_file" "$context_file" "$head_file" > "$fields_file"; then
:
else
attempt_rc=$?
rm -f "$body_file" "$fields_file" "$payload_file" "$commits_file" "$context_file" "$head_file"
return "$attempt_rc"
fi
rm -f "$commits_file" "$context_file" "$head_file"
fi
if ! python3 - "$fields_file" "$HEAD_SHA" "$DELETE_BRANCH" > "$payload_file" <<'PY'
mkdir -p "${AGENT_WORK_ROOT:-${HOME:-/tmp}/mosaic/agent-work}"
body_file=$(mktemp "${AGENT_WORK_ROOT:-${HOME:-/tmp}/mosaic/agent-work}/pr-merge-api-response.XXXXXX")
payload=$(python3 - "$HEAD_SHA" "$DELETE_BRANCH" <<'PY'
import json
import sys
with open(sys.argv[1], encoding="utf-8") as handle:
fields = json.load(handle)
head_sha, delete_branch = sys.argv[2:]
head_sha, delete_branch = sys.argv[1:]
payload = {"Do": "squash", "head_commit_id": head_sha}
if delete_branch == "true":
payload["delete_branch_after_merge"] = True
payload.update(fields)
allowed = {"Do", "head_commit_id", "delete_branch_after_merge", "MergeTitleField", "MergeMessageField"}
if payload.get("Do") != "squash" or set(payload) - allowed:
raise SystemExit(1)
print(json.dumps(payload, separators=(",", ":")))
PY
then
rm -f "$body_file" "$fields_file" "$payload_file"
return 1
fi
rm -f "$fields_file"
)
if ! auth_config=$(write_curl_auth_config "$auth_mode" "$credential"); then
echo "Error: Could not construct Gitea authentication config; refusing request." >&2
rm -f "$body_file" "$payload_file"
return 1
token=$(get_gitea_token "$host" || true)
if [[ -n "$token" ]]; then
raw_code=$(curl -sS -w '%{http_code}' -o "$body_file" \
-X POST \
-H "User-Agent: curl/8" \
-H "Authorization: token $token" \
-H 'Content-Type: application/json' \
-d "$payload" \
"$api_url" || true)
if [[ "$raw_code" =~ ^2 ]]; then
rm -f "$body_file"
return 0
fi
fi
raw_code=$(curl -sS -K - "${GITEA_CURL_BOUNDS[@]}" -w '%{http_code}' -o "$body_file" \
-X POST -H "User-Agent: curl/8" \
-H 'Content-Type: application/json' \
--data-binary "@$payload_file" "$api_url" <<<"$auth_config")
curl_rc=$?
LAST_GITEA_HTTP_CODE="${raw_code:-000}"
if [[ "$curl_rc" -ne 0 ]]; then
LAST_GITEA_ERROR="curl transport failed (rc=$curl_rc)"
rm -f "$body_file" "$payload_file"
rm -rf -- "$attempt_dir"
return 1
fi
if [[ ! "$raw_code" =~ ^2 ]]; then
LAST_GITEA_ERROR=$(format_gitea_error_response "$body_file")
fi
rm -f "$body_file" "$payload_file"
rm -rf -- "$attempt_dir"
[[ "$raw_code" =~ ^2 ]]
}
merge_gitea_with_api() {
local host="$1" token attempt_rc
basic_auth=$(get_gitea_basic_auth "$host" || true)
if [[ -n "$basic_auth" ]]; then
raw_code=$(curl -sS -w '%{http_code}' -o "$body_file" \
-X POST \
-u "$basic_auth" \
-H "User-Agent: curl/8" \
-H 'Content-Type: application/json' \
-d "$payload" \
"$api_url" || true)
if [[ "$raw_code" =~ ^2 ]]; then
rm -f "$body_file"
return 0
fi
fi
if ! token=$(get_gitea_token "$host"); then
echo "Error: Could not resolve the required Gitea token; refusing merge without changing principals." >&2
return 1
fi
if [[ -z "$token" ]]; then
echo "Error: Required Gitea token resolved empty; refusing merge without changing principals." >&2
return 1
fi
if merge_gitea_api_attempt "$host" token "$token"; then
return 0
else
attempt_rc=$?
fi
if [[ "$attempt_rc" -eq 75 ]]; then
return 75
fi
if [[ "$LAST_GITEA_HTTP_CODE" != "401" ]]; then
echo "Error: Gitea API merge failed with the identity-bound token (HTTP ${LAST_GITEA_HTTP_CODE:-000}).${LAST_GITEA_ERROR:+ Provider response: $LAST_GITEA_ERROR}" >&2
return 1
fi
echo "Error: Gitea API rejected the identity-bound token with HTTP 401; refusing cross-principal credential fallback." >&2
python3 - "${raw_code:-000}" "$body_file" <<'PY' >&2
import json
import sys
code, path = sys.argv[1], sys.argv[2]
try:
with open(path, encoding="utf-8", errors="replace") as handle:
raw = handle.read(500)
data = json.loads(raw) if raw else {}
message = data.get("message") or data.get("error") or raw or "empty response"
except Exception:
try:
message = open(path, encoding="utf-8", errors="replace").read(500) or "empty response"
except Exception:
message = "unreadable response"
print(f"Error: Gitea API merge failed with HTTP {code}: {message}")
PY
rm -f "$body_file"
return 1
}
@@ -602,10 +195,11 @@ if [[ "$DRY_RUN" == true ]]; then
echo "Error: Cannot determine host from origin remote URL" >&2
exit 1
}
if [[ "$CO_AUTHOR_TRAILERS" == true ]]; then
echo "Dry run: would verify PR commit authors and merge PR #$PR_NUMBER on $HOST with authenticated Gitea API message fields (base=$BASE_BRANCH, method=squash)."
TEA_LOGIN="$(get_gitea_login_for_host "$HOST" || true)"
if [[ -n "$TEA_LOGIN" ]]; then
echo "Dry run: would merge PR #$PR_NUMBER on $HOST with tea login '$TEA_LOGIN' (base=$BASE_BRANCH, method=squash)."
else
echo "Dry run: would merge PR #$PR_NUMBER on $HOST with the authenticated exact-head Gitea API path (base=$BASE_BRANCH, method=squash)."
echo "Dry run: would merge PR #$PR_NUMBER on $HOST with authenticated Gitea API fallback (base=$BASE_BRANCH, method=squash)."
fi
else
echo "Dry run: would merge PR #$PR_NUMBER on $PLATFORM (base=$BASE_BRANCH, method=squash)."
@@ -615,10 +209,6 @@ fi
case "$PLATFORM" in
github)
if [[ "$CO_AUTHOR_TRAILERS" == true ]]; then
echo "Error: --co-author-trailers currently requires the Gitea REST message-field contract." >&2
exit 1
fi
cmd=(gh pr merge "$PR_NUMBER" --squash --match-head-commit "$HEAD_SHA")
[[ "$DELETE_BRANCH" == true ]] && cmd+=(--delete-branch)
"${cmd[@]}"
@@ -629,7 +219,7 @@ case "$PLATFORM" in
exit 1
}
# Gitea's API head_commit_id is an atomic compare-and-merge precondition.
# tea cannot express it, so every Gitea merge uses the authenticated API path.
# tea cannot express it, so exact-head merges use the authenticated API path.
merge_gitea_with_api "$HOST"
;;
*)
@@ -9,51 +9,10 @@ WORK_DIR="${MOSAIC_TEST_WORK_DIR:-$PWD/.mosaic-test-work/ci-queue-wait-tristate}
REPO_DIR="$WORK_DIR/repo"
STUB_DIR="$WORK_DIR/stubs"
AUDIT_LOG="$WORK_DIR/audit/ci-queue-wait.jsonl"
STATUS_OBSERVED="$WORK_DIR/status-observed"
CLOCK_LOG="$WORK_DIR/clock.log"
WATCHDOG_PYTHON="/usr/bin/python3"
WATCHDOG_SCRIPT="$WORK_DIR/real-clock-watchdog.py"
WATCHDOG_TIMEOUT_SEC=5
WATCHDOG_EXIT=90
FEATURE_BRANCH="fix/rm-03-fixture"
if [[ ! -x "$WATCHDOG_PYTHON" ]]; then
echo "FAIL setup: required real-clock watchdog runtime is unavailable at $WATCHDOG_PYTHON" >&2
exit 1
fi
rm -rf "$WORK_DIR"
mkdir -p "$REPO_DIR" "$STUB_DIR"
cat > "$WATCHDOG_SCRIPT" <<'PY'
import os
import signal
import subprocess
import sys
if len(sys.argv) < 3:
raise SystemExit(2)
timeout_seconds = float(sys.argv[1])
process = subprocess.Popen(sys.argv[2:], start_new_session=True)
try:
return_code = process.wait(timeout=timeout_seconds)
except subprocess.TimeoutExpired:
try:
os.killpg(process.pid, signal.SIGKILL)
except ProcessLookupError:
pass
process.wait()
print(
f"FAIL HANG watchdog: subject exceeded {timeout_seconds:g}s "
"before completing its intended path",
file=sys.stderr,
)
raise SystemExit(90)
if return_code < 0:
raise SystemExit(128 - return_code)
raise SystemExit(return_code)
PY
git -C "$REPO_DIR" init -q
git -C "$REPO_DIR" checkout -q -b "$FEATURE_BRANCH"
git -C "$REPO_DIR" remote add origin https://git.example.test/acme/widgets.git
@@ -74,9 +33,6 @@ printf '%s\n' "$url" >> "${MOSAIC_STUB_URL_LOG:?}"
case "$url" in
*/branches/*)
if [[ "${MOSAIC_STUB_BRANCH_MODE:-ok}" == "hang-before-provider" ]]; then
while :; do :; done
fi
if [[ "${MOSAIC_STUB_BRANCH_MODE:-ok}" == "unreachable" ]]; then
exit 7
fi
@@ -88,7 +44,6 @@ case "$url" in
fi
;;
*/status)
: > "${MOSAIC_STUB_STATUS_OBSERVED:?}"
case "${MOSAIC_STUB_STATUS_MODE:?}" in
success) printf '%s' '{"state":"success","statuses":[{"status":"success"}]}' ;;
pending) printf '%s' '{"state":"pending","statuses":[{"status":"pending","context":"ci/test"}]}' ;;
@@ -109,31 +64,7 @@ case "$url" in
*) echo "unexpected curl URL: $url" >&2; exit 2 ;;
esac
SH
cat > "$STUB_DIR/date" <<'SH'
#!/usr/bin/env bash
set -euo pipefail
if [[ "$#" -ne 1 || "$1" != "+%s" ]]; then
echo "unexpected date invocation: $*" >&2
exit 2
fi
if [[ -e "${MOSAIC_STUB_STATUS_OBSERVED:?}" ]]; then
printf 'date-phase=after-status\n' >> "${MOSAIC_STUB_CLOCK_LOG:?}"
printf '1002\n'
else
printf 'date-phase=before-status\n' >> "${MOSAIC_STUB_CLOCK_LOG:?}"
printf '1000\n'
fi
SH
cat > "$STUB_DIR/sleep" <<'SH'
#!/usr/bin/env bash
set -euo pipefail
printf 'sleep-after-status=%s\n' "$*" >> "${MOSAIC_STUB_CLOCK_LOG:?}"
SH
chmod +x "$STUB_DIR/curl" "$STUB_DIR/date" "$STUB_DIR/sleep"
chmod +x "$STUB_DIR/curl"
run_guard() {
local status_mode="$1"
@@ -153,46 +84,13 @@ run_guard() {
export GITEA_URL=https://git.example.test
export MOSAIC_STUB_STATUS_MODE="$status_mode"
fi
rm -f "$STATUS_OBSERVED" "$CLOCK_LOG"
export MOSAIC_STUB_URL_LOG="$WORK_DIR/urls.log"
export MOSAIC_STUB_STATUS_OBSERVED="$STATUS_OBSERVED"
export MOSAIC_STUB_CLOCK_LOG="$CLOCK_LOG"
export MOSAIC_CI_QUEUE_AUDIT_LOG="$audit_log"
# Provider observation is the synchronization event. The one-second
# timeout is subject semantics under virtual time, never a wall wait.
# The absolute Python runtime uses an internal monotonic wait and kills
# the subject's isolated process group. Neither operation can resolve
# to the virtual date/sleep stubs at the front of PATH.
local subject_rc
if "$WATCHDOG_PYTHON" "$WATCHDOG_SCRIPT" "$WATCHDOG_TIMEOUT_SEC" \
"$SCRIPT_DIR/ci-queue-wait.sh" --purpose "${MOSAIC_TEST_PURPOSE:-push}" -t 1 -i 1 "$@"; then
subject_rc=0
else
subject_rc=$?
fi
return "$subject_rc"
"$SCRIPT_DIR/ci-queue-wait.sh" --purpose "${MOSAIC_TEST_PURPOSE:-push}" -t 0 -i 0 "$@"
)
}
failures=0
assert_provider_observed() {
local name="$1" require_expiration="${2:-0}"
if [[ ! -e "$STATUS_OBSERVED" ]]; then
echo "FAIL $name: status provider was not observed" >&2
failures=$((failures + 1))
fi
if [[ ! -s "$CLOCK_LOG" ]] || ! grep -q '^date-phase=before-status$' "$CLOCK_LOG"; then
echo "FAIL $name: virtual clock interception did not run before provider observation" >&2
failures=$((failures + 1))
fi
if [[ "$require_expiration" -eq 1 ]]; then
if ! grep -q '^sleep-after-status=' "$CLOCK_LOG" || ! grep -q '^date-phase=after-status$' "$CLOCK_LOG"; then
echo "FAIL $name: pending path did not expire after provider observation" >&2
failures=$((failures + 1))
fi
fi
}
run_assertion() {
local name="$1" expected_rc="$2" status_mode="$3" required_text="$4"
local output rc
@@ -227,13 +125,6 @@ run_assertion() {
printf '%s\n' "$output" >&2
failures=$((failures + 1))
fi
if [[ "$status_mode" != "credential-unresolvable" ]]; then
if [[ "$status_mode" == "pending" ]]; then
assert_provider_observed "$name" 1
else
assert_provider_observed "$name"
fi
fi
}
set -e
@@ -259,27 +150,6 @@ MOSAIC_TEST_PURPOSE=merge run_assertion merge-failure nonzero failure 'ASSERTED_
MOSAIC_TEST_PURPOSE=merge run_assertion merge-no-status nonzero no-status 'ASSERTED_NOT_READY state=no-status'
MOSAIC_TEST_PURPOSE=merge run_assertion merge-unknown nonzero unknown 'ASSERTED_NOT_READY state=unknown'
# Positive liveness control: a subject mutant hangs before the branch lookup
# can reach the status provider. Only the independent real-clock watchdog may
# terminate it, and its failure must be distinct from subject timeout rc=124.
set +e
watchdog_output=$(MOSAIC_STUB_BRANCH_MODE=hang-before-provider run_guard success "$AUDIT_LOG" 2>&1)
watchdog_rc=$?
set -e
if [[ "$watchdog_rc" -ne "$WATCHDOG_EXIT" ]]; then
echo "FAIL watchdog-control: expected hang-specific rc=$WATCHDOG_EXIT, got rc=$watchdog_rc" >&2
failures=$((failures + 1))
fi
if [[ "$watchdog_output" != *"FAIL HANG watchdog:"* ]]; then
echo "FAIL watchdog-control: expected distinct hang-specific diagnostic" >&2
printf '%s\n' "$watchdog_output" >&2
failures=$((failures + 1))
fi
if [[ -e "$STATUS_OBSERVED" ]]; then
echo "FAIL watchdog-control: hanging mutant unexpectedly reached the status provider" >&2
failures=$((failures + 1))
fi
if [[ ! -s "$AUDIT_LOG" ]] || ! grep -q '"outcome":"CANNOT_ASSERT"' "$AUDIT_LOG"; then
echo "FAIL provider-unreachable-audit: expected durable CANNOT_ASSERT JSONL record" >&2
failures=$((failures + 1))
@@ -300,7 +170,6 @@ if [[ "$merge_unreachable_output" != *"CANNOT_ASSERT"* ]]; then
echo "FAIL merge-provider-unreachable: expected loud CANNOT_ASSERT diagnostic" >&2
failures=$((failures + 1))
fi
assert_provider_observed merge-provider-unreachable
merge_audit_lines_after=$(wc -l < "$AUDIT_LOG")
if [[ "$merge_audit_lines_after" -le "$merge_audit_lines_before" ]]; then
echo "FAIL merge-provider-unreachable: expected an additional audit record" >&2
@@ -364,7 +233,6 @@ if [[ "$audit_failure_output" != *"audit"* ]]; then
echo "FAIL audit-unavailable: expected loud audit failure diagnostic" >&2
failures=$((failures + 1))
fi
assert_provider_observed audit-unavailable
if [[ "$failures" -ne 0 ]]; then
echo "ci-queue-wait tri-state regression failed ($failures assertions)" >&2
@@ -51,23 +51,22 @@ for arg in "$@"; do
prev=""
continue
fi
if [[ "$prev" == "data" ]]; then
if [[ "$prev" == "-d" ]]; then
post_data="$arg"
[[ "$post_data" == @* ]] && post_data=$(<"${post_data#@}")
prev=""
continue
fi
if [[ "$prev" == "config" ]]; then
[[ "$arg" == "-" ]] && cat >/dev/null
prev=""
if [[ "$arg" == "-o" ]]; then
prev="-o"
continue
fi
case "$arg" in
-o) prev="-o" ;;
-d|--data|--data-binary) prev="data" ;;
-K|--config) prev="config" ;;
-w) write_code=true ;;
esac
if [[ "$arg" == "-d" ]]; then
prev="-d"
continue
fi
if [[ "$arg" == "-w" ]]; then
write_code=true
fi
done
emit_response() {
local body="$1"
@@ -37,30 +37,13 @@ cat > "$WORK_DIR/gitea/curl" <<'SH'
#!/usr/bin/env bash
set -euo pipefail
payload=""
out_file=""
while [[ $# -gt 0 ]]; do
case "$1" in
-d|--data|--data-binary)
payload="$2"
[[ "$payload" == @* ]] && payload=$(<"${payload#@}")
shift 2
;;
-o)
out_file="$2"
shift 2
;;
-K|--config)
[[ "$2" == "-" ]] && cat >/dev/null
shift 2
;;
-w|-X|-H)
shift 2
;;
*) shift ;;
esac
for ((i=1; i<=$#; i++)); do
if [[ "${!i}" == "-d" ]]; then
j=$((i + 1))
payload="${!j}"
fi
done
printf '%s' "$payload" > "${MOSAIC_MERGE_PAYLOAD_LOG:?}"
[[ -n "$out_file" ]] && printf '{}' > "$out_file"
printf '200'
SH
chmod +x "$WORK_DIR/gitea/curl"
@@ -1,541 +0,0 @@
#!/usr/bin/env bash
# Regression harness for the optional, identity-checked Gitea squash message.
set -u
SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
SUBJECT="${MOSAIC_TEST_SUBJECT:-$SCRIPT_DIR/pr-merge.sh}"
WORK_DIR="${MOSAIC_TEST_WORK_DIR:-$PWD/.mosaic-test-work/pr-merge-message-field}"
ORIG_PATH="$PATH"
failures=0
rm -rf "$WORK_DIR"
mkdir -p "$WORK_DIR"
fail() {
echo "FAIL $1" >&2
failures=$((failures + 1))
}
make_case() {
local name="$1" case_dir
case_dir="$WORK_DIR/$name"
mkdir -p "$case_dir/bin" "$case_dir/agent"
cp "$SUBJECT" "$case_dir/pr-merge.sh"
chmod +x "$case_dir/pr-merge.sh"
cat > "$case_dir/detect-platform.sh" <<'SH'
#!/usr/bin/env bash
detect_platform() { PLATFORM=gitea; printf 'gitea\n'; }
get_repo_owner() { printf 'acme\n'; }
get_repo_name() { printf 'widgets\n'; }
get_remote_host() { printf 'git.example.test\n'; }
get_gitea_token() {
printf 'resolved\n' >> "${MOSAIC_TEST_TOKEN_RESOLUTION_LOG:?}"
if [[ "${MOSAIC_TEST_TOKEN_AVAILABLE:-true}" != "true" ]]; then
return 1
fi
printf 'fixture-token\n'
}
get_gitea_basic_auth() {
printf 'resolved\n' >> "${MOSAIC_TEST_BASIC_RESOLUTION_LOG:?}"
if [[ "${MOSAIC_TEST_BASIC_AVAILABLE:-false}" == "true" ]]; then
printf 'fixture-user:fixture-password\n'
return "${MOSAIC_TEST_BASIC_RC:-0}"
fi
return 1
}
get_gitea_login_for_host() { return 1; }
SH
cat > "$case_dir/pr-metadata.sh" <<'SH'
#!/usr/bin/env bash
if [[ "${MOSAIC_TEST_TITLE_MODE:-safe}" == "injection" ]]; then
title='Preserve authors\n\nCo-authored-by: victim <[email protected]>'
else
title='Preserve both branch authors'
fi
case "${MOSAIC_TEST_COMMITS_MODE:?}" in
verified) head_sha=2222222222222222222222222222222222222222 ;;
null-login|unsafe-identity) head_sha=3333333333333333333333333333333333333333 ;;
single) head_sha=1111111111111111111111111111111111111111 ;;
*) echo "unknown commits mode" >&2; exit 2 ;;
esac
printf '{"number":42,"title":"%s","author":"poster","baseRefName":"main","headRefName":"feature/fixture","headRefOid":"%s","headRepository":"acme/widgets"}\n' "$title" "$head_sha"
SH
cat > "$case_dir/ci-queue-wait.sh" <<'SH'
#!/usr/bin/env bash
exit 0
SH
cat > "$case_dir/bin/python3" <<'SH'
#!/usr/bin/env bash
for arg in "$@"; do
case "$arg" in
*"Preserve both branch authors"*|*"[email protected]"*)
: > "${MOSAIC_TEST_METADATA_ARGV_MARKER:?}"
;;
esac
done
exec "${MOSAIC_TEST_REAL_PYTHON:?}" "$@"
SH
cat > "$case_dir/bin/curl" <<'SH'
#!/usr/bin/env bash
set -eu
for arg in "$@"; do
case "$arg" in
*"Preserve both branch authors"*|*"[email protected]"*)
: > "${MOSAIC_TEST_METADATA_ARGV_MARKER:?}"
;;
esac
done
url=""
method="GET"
out_file=""
data=""
config=""
auth_mode="none"
has_max_filesize=0
has_max_time=0
has_connect_timeout=0
while [[ $# -gt 0 ]]; do
case "$1" in
-o)
out_file="$2"
shift 2
;;
-w)
shift 2
;;
-X)
method="$2"
shift 2
;;
-d|--data|--data-binary)
data="$2"
if [[ "$data" == @* ]]; then
data=$(<"${data#@}")
fi
shift 2
;;
-K|--config)
if [[ "$2" == "-" ]]; then
config=$(cat)
fi
shift 2
;;
--max-filesize)
has_max_filesize=1
shift 2
;;
--max-time)
has_max_time=1
shift 2
;;
--connect-timeout)
has_connect_timeout=1
shift 2
;;
-H|--header|-u|--user)
if [[ "$2" == *"fixture-token"* ]]; then
: > "${MOSAIC_TEST_TOKEN_ARGV_MARKER:?}"
fi
if [[ "$2" == *"fixture-password"* ]]; then
: > "${MOSAIC_TEST_BASIC_ARGV_MARKER:?}"
fi
shift 2
;;
http://*|https://*)
url="$1"
shift
;;
*)
shift
;;
esac
done
if [[ "$config" == *"Authorization: token fixture-token"* ]]; then
auth_mode="token"
: > "${MOSAIC_TEST_AUTH_CONFIG_MARKER:?}"
elif [[ "$config" == *"user = \"fixture-user:fixture-password\""* ]]; then
auth_mode="basic"
: > "${MOSAIC_TEST_BASIC_CONFIG_MARKER:?}"
fi
printf '%s %s %s\n' "$method" "$auth_mode" "$url" >> "${MOSAIC_TEST_CURL_LOG:?}"
printf '%s:%s:%s\n' "$has_max_filesize" "$has_max_time" "$has_connect_timeout" >> "${MOSAIC_TEST_CURL_BOUNDS_LOG:?}"
case "$url" in
*/pulls/42)
case "${MOSAIC_TEST_COMMITS_MODE:?}" in
verified) head_sha=2222222222222222222222222222222222222222 ;;
null-login|unsafe-identity) head_sha=3333333333333333333333333333333333333333 ;;
single) head_sha=1111111111111111111111111111111111111111 ;;
*) echo "unknown commits mode" >&2; exit 2 ;;
esac
if [[ "${MOSAIC_TEST_HEAD_MODE:-stable}" == "moved" ]]; then
head_sha=4444444444444444444444444444444444444444
fi
body="{\"head\":{\"sha\":\"$head_sha\"}}"
code=200
if [[ "${MOSAIC_TEST_FALLBACK_MODE:-none}" == "inspection" && "$auth_mode" == "token" ]]; then
body='{"message":"token rejected"}'
code=401
fi
;;
*/pulls/42/commits*)
case "${MOSAIC_TEST_COMMITS_MODE:?}" in
verified)
if [[ "${MOSAIC_TEST_EMAIL_MODE:-safe}" == "escape" ]]; then
body='[{"sha":"2222222222222222222222222222222222222222","commit":{"author":{"name":"Alice","email":"alice+\u001b[[email protected]"}},"author":{"login":"alice"}},{"sha":"1111111111111111111111111111111111111111","commit":{"author":{"name":"Poster","email":"[email protected]"}},"author":{"login":"poster"}}]'
else
body='[{"sha":"2222222222222222222222222222222222222222","commit":{"author":{"name":"Alice","email":"[email protected]"}},"author":{"login":"alice"}},{"sha":"1111111111111111111111111111111111111111","commit":{"author":{"name":"Poster","email":"[email protected]"}},"author":{"login":"poster"}}]'
fi
;;
null-login)
body='[{"sha":"1111111111111111111111111111111111111111","commit":{"author":{"name":"Poster","email":"[email protected]"}},"author":{"login":"poster"}},{"sha":"3333333333333333333333333333333333333333","commit":{"author":{"name":"Unresolved Author","email":"[email protected]\n\u001b[31m"}},"author":null}]'
;;
unsafe-identity)
body='[{"sha":"unsafe\n\u001b[31m","commit":{"author":{"name":"Unsafe","email":"not-an-email"}},"author":{"login":"unsafe"}},{"sha":"3333333333333333333333333333333333333333","commit":{"author":{"name":"Poster","email":"[email protected]"}},"author":{"login":"poster"}}]'
;;
single)
body='[{"sha":"1111111111111111111111111111111111111111","commit":{"author":{"name":"Poster","email":"[email protected]"}},"author":{"login":"poster"}}]'
;;
*)
echo "unknown commits mode" >&2
exit 2
;;
esac
code=200
if [[ "${MOSAIC_TEST_FALLBACK_MODE:-none}" == "inspection" && "$auth_mode" == "token" ]]; then
body='{"message":"token rejected"}'
code=401
fi
;;
*/pulls/42/merge)
body='{}'
code=200
if [[ "${MOSAIC_TEST_FALLBACK_MODE:-none}" == "merge" && "$auth_mode" == "token" ]]; then
body='{"message":"token rejected"}'
code=401
elif [[ "${MOSAIC_TEST_FALLBACK_MODE:-none}" == "provider-error" ]]; then
body='{"message":"branch policy rejected\n\u001b[31m"}'
code=409
elif [[ "${MOSAIC_TEST_FALLBACK_MODE:-none}" == "forbidden" ]]; then
body='{"message":"permission denied"}'
code=403
else
printf '%s' "$data" > "${MOSAIC_TEST_MERGE_PAYLOAD:?}"
fi
;;
*/users/*)
body='{"message":"not found"}'
code=404
;;
*)
body='{"message":"unexpected URL"}'
code=500
;;
esac
if [[ -n "$out_file" ]]; then
printf '%s' "$body" > "$out_file"
else
printf '%s' "$body"
fi
printf '%s' "$code"
case "${MOSAIC_TEST_CURL_FAILURE:-none}" in
oversize) exit 63 ;;
stalled) exit 28 ;;
esac
SH
chmod +x "$case_dir/detect-platform.sh" "$case_dir/pr-metadata.sh" \
"$case_dir/ci-queue-wait.sh" "$case_dir/bin/curl" "$case_dir/bin/python3"
printf '%s\n' "$case_dir"
}
run_case() {
local case_dir="$1" mode="$2"
shift 2
MOSAIC_TEST_COMMITS_MODE="$mode" \
MOSAIC_TEST_CURL_LOG="$case_dir/curl.log" \
MOSAIC_TEST_CURL_BOUNDS_LOG="$case_dir/curl-bounds.log" \
MOSAIC_TEST_MERGE_PAYLOAD="$case_dir/merge-payload.json" \
MOSAIC_TEST_TOKEN_ARGV_MARKER="$case_dir/token-in-argv" \
MOSAIC_TEST_BASIC_ARGV_MARKER="$case_dir/basic-in-argv" \
MOSAIC_TEST_AUTH_CONFIG_MARKER="$case_dir/auth-via-config" \
MOSAIC_TEST_BASIC_CONFIG_MARKER="$case_dir/basic-via-config" \
MOSAIC_TEST_TOKEN_RESOLUTION_LOG="$case_dir/token-resolution.log" \
MOSAIC_TEST_BASIC_RESOLUTION_LOG="$case_dir/basic-resolution.log" \
MOSAIC_TEST_METADATA_ARGV_MARKER="$case_dir/metadata-in-argv" \
MOSAIC_TEST_REAL_PYTHON="$(command -v python3)" \
AGENT_WORK_ROOT="$case_dir/agent" \
PATH="$case_dir/bin:$ORIG_PATH" \
"$case_dir/pr-merge.sh" -n 42 "$@"
}
# Verified multi-author path: the non-poster trailer is built from one commit's
# linked author.login and that same commit's author email. No /users lookup.
verified_dir=$(make_case verified)
set +e
verified_output=$(run_case "$verified_dir" verified --co-author-trailers --escalate-to tl-mosaic 2>&1)
verified_rc=$?
set -e
if [[ "$verified_rc" -ne 0 ]]; then
fail "verified multi-author merge expected rc=0, got rc=$verified_rc: $verified_output"
elif [[ ! -s "$verified_dir/merge-payload.json" ]]; then
fail "verified multi-author merge did not reach the API payload"
else
python3 - "$verified_dir/merge-payload.json" <<'PY' || fail "verified payload did not preserve squash and exact message fields"
import json
import sys
payload = json.load(open(sys.argv[1], encoding="utf-8"))
assert payload == {
"Do": "squash",
"head_commit_id": "2222222222222222222222222222222222222222",
"MergeTitleField": "Preserve both branch authors",
"MergeMessageField": "Co-authored-by: alice <[email protected]>",
}, payload
PY
fi
[[ -e "$verified_dir/auth-via-config" ]] || fail "verified path did not authenticate curl through stdin config"
[[ ! -e "$verified_dir/token-in-argv" ]] || fail "verified path placed the Gitea token in curl argv"
[[ ! -e "$verified_dir/metadata-in-argv" ]] || fail "verified path placed PR title or contributor email in child argv"
[[ "$(wc -l < "$verified_dir/token-resolution.log")" -eq 1 ]] || fail "verified path did not bind inspection and merge to one credential resolution"
if grep -q '/users/' "$verified_dir/curl.log" 2>/dev/null; then
fail "verified path performed a forbidden second /users lookup"
fi
if grep -qv '^1:1:1$' "$verified_dir/curl-bounds.log"; then
fail "verified path did not apply size/max-time/connect-time bounds to every provider download"
fi
# A linked email containing a terminal escape must block before mutation.
escape_email_dir=$(make_case escape-email)
set +e
escape_email_output=$(MOSAIC_TEST_EMAIL_MODE=escape run_case "$escape_email_dir" verified --co-author-trailers --escalate-to tl-mosaic 2>&1)
escape_email_rc=$?
set -e
[[ "$escape_email_rc" -ne 0 ]] || fail "control-byte email unexpectedly passed"
[[ "$escape_email_output" == *"unusable linked identity"* ]] || fail "control-byte email refusal lost its diagnostic"
[[ ! -e "$escape_email_dir/merge-payload.json" ]] || fail "control-byte email reached the merge API"
# Curl transfer and duration failures must remain failures even with HTTP 200.
for failure_mode in oversize stalled; do
failure_dir=$(make_case "curl-$failure_mode")
set +e
failure_output=$(MOSAIC_TEST_CURL_FAILURE="$failure_mode" run_case "$failure_dir" verified --co-author-trailers --escalate-to tl-mosaic 2>&1)
failure_rc=$?
set -e
[[ "$failure_rc" -ne 0 ]] || fail "curl $failure_mode failure was discarded: $failure_output"
[[ ! -e "$failure_dir/merge-payload.json" ]] || fail "curl $failure_mode failure reached the merge API"
done
# The authenticated head is re-read under the mutation credential but cannot
# replace the canonical preflight/review head. A move blocks before enumeration
# or mutation even though the provider returned a valid new SHA.
moved_dir=$(make_case moved-head)
set +e
moved_output=$(MOSAIC_TEST_HEAD_MODE=moved \
run_case "$moved_dir" verified --co-author-trailers --escalate-to tl-mosaic 2>&1)
moved_rc=$?
set -e
[[ "$moved_rc" -ne 0 ]] || fail "moved authenticated head unexpectedly passed"
[[ "$moved_output" == *"authenticated PR head moved from reviewed"* ]] || fail "moved head refusal lost its diagnostic"
[[ "$moved_output" == *"tl-mosaic"* ]] || fail "moved head refusal omitted the named escalation principal"
[[ ! -e "$moved_dir/merge-payload.json" ]] || fail "moved head refusal reached the merge API"
moved_sequence=$(awk '{print $1 ":" $2}' "$moved_dir/curl.log" | paste -sd, -)
[[ "$moved_sequence" == "GET:token" ]] || fail "moved head refusal performed post-move inspection/mutation (calls=$moved_sequence)"
# Token resolution failure is not an authentication response. It must fail
# closed instead of borrowing a Basic credential under a different principal.
token_missing_dir=$(make_case token-missing)
set +e
token_missing_output=$(MOSAIC_TEST_TOKEN_AVAILABLE=false MOSAIC_TEST_BASIC_AVAILABLE=true \
run_case "$token_missing_dir" single 2>&1)
token_missing_rc=$?
set -e
[[ "$token_missing_rc" -ne 0 ]] || fail "missing token unexpectedly borrowed Basic Auth"
[[ "$token_missing_output" == *"required Gitea token"* ]] || fail "missing token refusal lost its diagnostic"
[[ ! -e "$token_missing_dir/basic-resolution.log" ]] || fail "missing token resolved Basic Auth after identity failure"
[[ ! -e "$token_missing_dir/curl.log" ]] || fail "missing token reached a provider request"
# A failed Basic resolver must never use its nonempty output or reach mutation.
basic_rc_dir=$(make_case basic-resolver-rc)
set +e
basic_rc_output=$(MOSAIC_TEST_BASIC_AVAILABLE=true MOSAIC_TEST_BASIC_RC=91 MOSAIC_TEST_FALLBACK_MODE=inspection \
run_case "$basic_rc_dir" verified --co-author-trailers --escalate-to tl-mosaic 2>&1)
basic_rc_rc=$?
set -e
[[ "$basic_rc_rc" -ne 0 ]] || fail "failed Basic resolver output unexpectedly authorized a merge: $basic_rc_output"
[[ ! -e "$basic_rc_dir/merge-payload.json" ]] || fail "failed Basic resolver reached the merge API"
# HTTP 401 never changes principals: inspection rejection fails closed without
# resolving or attempting Basic Auth.
fallback_inspect_dir=$(make_case fallback-inspection)
set +e
fallback_inspect_output=$(MOSAIC_TEST_BASIC_AVAILABLE=true MOSAIC_TEST_FALLBACK_MODE=inspection \
run_case "$fallback_inspect_dir" verified --co-author-trailers --escalate-to tl-mosaic 2>&1)
fallback_inspect_rc=$?
set -e
[[ "$fallback_inspect_rc" -ne 0 ]] || fail "inspection token rejection unexpectedly changed principals"
[[ "$fallback_inspect_output" == *"refusing cross-principal credential fallback"* ]] || fail "inspection token rejection lost its refusal diagnostic"
[[ ! -e "$fallback_inspect_dir/basic-resolution.log" ]] || fail "inspection token rejection resolved Basic Auth"
[[ ! -e "$fallback_inspect_dir/merge-payload.json" ]] || fail "inspection token rejection reached merge mutation"
inspect_sequence=$(awk '{print $1 ":" $2}' "$fallback_inspect_dir/curl.log" | paste -sd, -)
[[ "$inspect_sequence" == "GET:token" ]] || fail "inspection rejection made unexpected provider calls (calls=$inspect_sequence)"
# Token rejection at merge likewise fails closed without cross-principal retry.
fallback_merge_dir=$(make_case fallback-merge)
set +e
fallback_merge_output=$(MOSAIC_TEST_BASIC_AVAILABLE=true MOSAIC_TEST_FALLBACK_MODE=merge \
run_case "$fallback_merge_dir" verified --co-author-trailers --escalate-to tl-mosaic 2>&1)
fallback_merge_rc=$?
set -e
[[ "$fallback_merge_rc" -ne 0 ]] || fail "merge token rejection unexpectedly changed principals"
[[ "$fallback_merge_output" == *"refusing cross-principal credential fallback"* ]] || fail "merge token rejection lost its refusal diagnostic"
[[ ! -e "$fallback_merge_dir/basic-resolution.log" ]] || fail "merge token rejection resolved Basic Auth"
[[ ! -e "$fallback_merge_dir/merge-payload.json" ]] || fail "merge token rejection recorded a successful payload"
merge_sequence=$(awk '{print $1 ":" $2}' "$fallback_merge_dir/curl.log" | paste -sd, -)
[[ "$merge_sequence" == "GET:token,GET:token,POST:token" ]] || fail "merge rejection made unexpected provider calls (calls=$merge_sequence)"
# BLOCK path: a commit email exists but author.login is null. It must name both
# facts, name the escalation principal, and never reach the merge endpoint.
null_dir=$(make_case null-login)
set +e
null_output=$(run_case "$null_dir" null-login --co-author-trailers --escalate-to tl-mosaic 2>&1)
null_rc=$?
set -e
[[ "$null_rc" -ne 0 ]] || fail "null-login author expected a non-zero BLOCK"
[[ "$null_output" == *"BLOCK"* ]] || fail "null-login author omitted BLOCK diagnostic"
[[ "$null_output" == *"author.login=NULL"* ]] || fail "null-login author omitted the null provider fact"
[[ "$null_output" == *"[email protected]"* ]] || fail "null-login author omitted the commit email fact"
[[ "$null_output" == *'\n\x1b[31m'* ]] || fail "null-login author diagnostic did not escape control characters"
[[ "$null_output" != *$'\033'* ]] || fail "null-login author diagnostic emitted a raw terminal escape"
[[ "$(printf '%s\n' "$null_output" | wc -l)" -eq 1 ]] || fail "null-login author diagnostic permitted newline injection"
[[ "$null_output" == *"tl-mosaic"* ]] || fail "null-login author omitted the named escalation principal"
[[ ! -e "$null_dir/merge-payload.json" ]] || fail "null-login BLOCK still reached the merge API"
# Every provider-derived field in alternate BLOCK diagnostics is log-safe too,
# including an invalid non-head SHA that contains control characters.
unsafe_dir=$(make_case unsafe-identity)
set +e
unsafe_output=$(run_case "$unsafe_dir" unsafe-identity --co-author-trailers --escalate-to tl-mosaic 2>&1)
unsafe_rc=$?
set -e
[[ "$unsafe_rc" -ne 0 ]] || fail "unsafe identity expected a non-zero BLOCK"
[[ "$unsafe_output" == *"unusable linked identity"* ]] || fail "unsafe identity omitted its BLOCK reason"
[[ "$unsafe_output" == *'\n\x1b[31m'* ]] || fail "unsafe identity SHA did not escape control characters"
[[ "$unsafe_output" != *$'\033'* ]] || fail "unsafe identity diagnostic emitted a raw terminal escape"
[[ "$(printf '%s\n' "$unsafe_output" | wc -l)" -eq 1 ]] || fail "unsafe identity diagnostic permitted newline injection"
[[ ! -e "$unsafe_dir/merge-payload.json" ]] || fail "unsafe identity BLOCK still reached the merge API"
# The provider PR title cannot add an unchecked trailer outside the constructed
# message field: multi-line and trailer-shaped titles block before mutation.
title_dir=$(make_case title-injection)
set +e
title_output=$(MOSAIC_TEST_TITLE_MODE=injection \
run_case "$title_dir" verified --co-author-trailers --escalate-to tl-mosaic 2>&1)
title_rc=$?
set -e
[[ "$title_rc" -ne 0 ]] || fail "title trailer injection unexpectedly passed"
[[ "$title_output" == *"not one printable, non-trailer line"* ]] || fail "title injection refusal lost its diagnostic"
[[ ! -e "$title_dir/merge-payload.json" ]] || fail "title injection reached the merge API"
# Provider failures remain diagnosable after their temporary response file is
# removed, but provider-controlled control characters stay log-safe.
error_dir=$(make_case provider-error)
set +e
error_output=$(MOSAIC_TEST_BASIC_AVAILABLE=true MOSAIC_TEST_FALLBACK_MODE=provider-error \
run_case "$error_dir" single 2>&1)
error_rc=$?
set -e
[[ "$error_rc" -ne 0 ]] || fail "provider error unexpectedly passed"
[[ "$error_output" == *"HTTP 409"* ]] || fail "provider error omitted the HTTP status"
[[ "$error_output" == *"branch policy rejected"* ]] || fail "provider error response was discarded"
[[ "$error_output" == *'\n\x1b[31m'* ]] || fail "provider error response did not escape control characters"
[[ "$error_output" != *$'\033'* ]] || fail "provider error response emitted a raw terminal escape"
[[ "$error_output" != *"Basic Auth fallback"* ]] || fail "provider error advertised removed Basic Auth fallback"
[[ ! -e "$error_dir/basic-resolution.log" ]] || fail "HTTP 409 policy denial incorrectly triggered Basic Auth fallback"
# Authorization denials likewise fail closed instead of changing principals.
forbidden_dir=$(make_case forbidden)
set +e
forbidden_output=$(MOSAIC_TEST_BASIC_AVAILABLE=true MOSAIC_TEST_FALLBACK_MODE=forbidden \
run_case "$forbidden_dir" single 2>&1)
forbidden_rc=$?
set -e
[[ "$forbidden_rc" -ne 0 ]] || fail "HTTP 403 authorization denial unexpectedly passed"
[[ "$forbidden_output" == *"HTTP 403"* ]] || fail "authorization denial omitted the HTTP status"
[[ "$forbidden_output" != *"Basic Auth fallback"* ]] || fail "authorization denial advertised removed Basic Auth fallback"
[[ ! -e "$forbidden_dir/basic-resolution.log" ]] || fail "HTTP 403 authorization denial incorrectly triggered Basic Auth fallback"
# The BLOCK destination cannot be generic or inferred after failure: opting in
# without a named principal is refused before any provider operation.
principal_dir=$(make_case missing-principal)
set +e
principal_output=$(run_case "$principal_dir" verified --co-author-trailers 2>&1)
principal_rc=$?
set -e
[[ "$principal_rc" -ne 0 ]] || fail "co-author mode without a named principal unexpectedly passed"
[[ "$principal_output" == *"requires --escalate-to with a named principal"* ]] || fail "missing-principal refusal lost its diagnostic"
[[ ! -e "$principal_dir/merge-payload.json" ]] || fail "missing-principal refusal reached the merge API"
# A trailing value-taking option receives a stable CLI diagnostic instead of a
# set -u unbound-variable crash.
value_dir=$(make_case missing-principal-value)
set +e
value_output=$(run_case "$value_dir" verified --co-author-trailers --escalate-to 2>&1)
value_rc=$?
set -e
[[ "$value_rc" -ne 0 ]] || fail "missing --escalate-to value unexpectedly passed"
[[ "$value_output" == *"--escalate-to requires one principal name"* ]] || fail "missing --escalate-to value lost its diagnostic"
[[ "$value_output" != *"unbound variable"* ]] || fail "missing --escalate-to value crashed under set -u"
[[ ! -e "$value_dir/merge-payload.json" ]] || fail "missing --escalate-to value reached the merge API"
# Negative control: ordinary single-author merge remains byte-for-byte payload
# compatible and hardcoded to squash, with no optional message fields.
single_dir=$(make_case single)
set +e
single_output=$(run_case "$single_dir" single 2>&1)
single_rc=$?
set -e
if [[ "$single_rc" -ne 0 ]]; then
fail "ordinary single-author merge expected rc=0, got rc=$single_rc: $single_output"
elif [[ ! -s "$single_dir/merge-payload.json" ]]; then
fail "ordinary single-author merge did not reach the API payload"
else
python3 - "$single_dir/merge-payload.json" <<'PY' || fail "ordinary single-author payload changed"
import json
import sys
payload = json.load(open(sys.argv[1], encoding="utf-8"))
assert payload == {
"Do": "squash",
"head_commit_id": "1111111111111111111111111111111111111111",
}, payload
PY
fi
[[ -e "$single_dir/auth-via-config" ]] || fail "ordinary path did not authenticate curl through stdin config"
[[ ! -e "$single_dir/token-in-argv" ]] || fail "ordinary path placed the Gitea token in curl argv"
[[ "$(wc -l < "$single_dir/token-resolution.log")" -eq 1 ]] || fail "ordinary path did not use exactly one credential resolution"
# Squash is not defaultable: an explicit non-squash method must remain refused.
method_dir=$(make_case method-refusal)
set +e
method_output=$(run_case "$method_dir" single -m merge 2>&1)
method_rc=$?
set -e
[[ "$method_rc" -ne 0 ]] || fail "non-squash method unexpectedly passed"
[[ "$method_output" == *"enforces squash merge only"* ]] || fail "non-squash refusal lost its policy diagnostic"
[[ ! -e "$method_dir/merge-payload.json" ]] || fail "non-squash refusal reached the merge API"
if [[ "$failures" -ne 0 ]]; then
echo "pr-merge message-field regression failed ($failures assertions)" >&2
exit 1
fi
echo "pr-merge message-field regression passed (verified, BLOCK, and unchanged squash control)"
@@ -39,7 +39,7 @@ MAX_FRAME: Final = 64 * 1024
MAX_STATE: Final = 4 * 1024 * 1024
MAX_PENDING_TOKENS: Final = 256
MAX_IN_FLIGHT_CONNECTIONS: Final = 16
MAX_LEASE_TTL_SECONDS: Final = 3600
MAX_LEASE_TTL_SECONDS: Final = 300
STATE_VERSION: Final = 1
READ_DEADLINE_SECONDS: Final = 1.0
HANDLE_QUEUE_TIMEOUT_SECONDS: Final = 1.0
@@ -50,8 +50,8 @@ LEASE_PENDING: Final = "PENDING_VERIFICATION"
LEASE_PENDING_PROMOTION: Final = "PENDING_PROMOTION"
LEASE_VERIFIED: Final = "VERIFIED"
READ_ONLY_TOOLS: Final = {
"claude": frozenset({"Read", "Grep", "Glob"}),
"pi": frozenset({"read", "ls"}),
"claude": frozenset({"Read", "Grep", "Glob", "Ls", "Find"}),
"pi": frozenset({"read", "grep", "find", "ls"}),
}
RECOVERY_TOOL: Final = "mosaic_context_recover"
@@ -8,9 +8,7 @@ import json
import os
import socket
import sys
import time
from collections.abc import Callable, Mapping, Sequence
from datetime import datetime, timezone
from pathlib import Path
from typing import Final
@@ -55,48 +53,6 @@ def broker_request(socket_path: Path, request: dict[str, object]) -> dict[str, o
return value
def _self_starttime() -> str | None:
"""Field 22 of our own /proc stat — the anchor starttime the broker records.
Read past the comm field's parens, since a process name may contain them.
"""
try:
raw = Path(f"/proc/{os.getpid()}/stat").read_text()
return raw.rsplit(")", 1)[1].split()[19]
except (OSError, IndexError, ValueError):
return None
def _append_launch_record(environ: Mapping[str, str], record: dict[str, object]) -> None:
"""Append one NDJSON event to the #797 Runtime Session Ledger.
`fleet/run/sessions/` is operator-classified in framework-manifest.txt and is
already covered by test-upgrade-manifest-guard.sh, so an upgrade can neither
overwrite nor prune it. Files 0600 under a 0700 dir, matching what that guard
asserts.
Never raises: a launch must not be denied over bookkeeping. But it also never
fails silently a missing record is exactly the kind of gap that made the
2026-08-06 MUTATOR_UNVERIFIED investigation cost a day.
"""
try:
mosaic_home = environ.get("MOSAIC_HOME") or str(Path.home() / ".config" / "mosaic")
directory = Path(mosaic_home) / "fleet" / "run" / "sessions"
directory.mkdir(parents=True, exist_ok=True)
os.chmod(directory, 0o700)
framed = {
"seq": time.time_ns() // 1_000_000,
"ts": datetime.now(timezone.utc).isoformat(),
**record,
}
path = directory / "events.ndjson"
descriptor = os.open(path, os.O_WRONLY | os.O_CREAT | os.O_APPEND, 0o600)
with os.fdopen(descriptor, "w") as handle:
handle.write(json.dumps(framed, separators=(",", ":")) + "\n")
except (OSError, ValueError, TypeError) as error:
print(f"[mosaic] WARNING: launch record not written: {error}", file=sys.stderr)
def main(
argv: Sequence[str] | None = None,
*,
@@ -138,9 +94,8 @@ def main(
# silent pass and never folded into the generic registration-failure
# branch.
try:
activation_capability = probe_activation_capability(source_environment)
assert_activation_capability_matches(
activation_capability,
probe_activation_capability(source_environment),
expected_activation_capability,
)
except VersionCouplingError as version_error:
@@ -173,32 +128,6 @@ def main(
print("Mosaic lease broker registration failed; runtime launch denied.", file=sys.stderr)
return 1
# Immutable launch record, half two. `mosaic` wrote `session.launch` with the
# config/provenance it knows; only this process knows the broker session id
# and the activation capability it just asserted. os.execvpe preserves the
# PID, so this PID is BOTH the anchor pid and the join key back to that
# record. Never fatal — bookkeeping must not deny a launch — but never
# silent either.
_append_launch_record(
source_environment,
{
"kind": "lease.register",
# Joins back to `mosaic`'s session.launch record. NOT pid: execRuntime()
# spawns rather than execs, so this process is a CHILD of mosaic with a
# different pid. This pid IS the broker anchor pid (os.execvpe below
# preserves it), which is a separate and still-useful fact.
"launch_id": source_environment.get("MOSAIC_LAUNCH_ID"),
"pid": os.getpid(),
"runtime": arguments.runtime,
"session_id": session_id,
"runtime_generation": generation,
"generation_file": str(generation_file),
"anchor_starttime": _self_starttime(),
"activation_capability": activation_capability,
"command": Path(command[0]).name,
},
)
environment = dict(source_environment)
environment["MOSAIC_LEASE_SESSION_ID"] = session_id
environment["MOSAIC_RUNTIME_GENERATION"] = str(generation)
@@ -1,337 +0,0 @@
#!/usr/bin/env python3
"""Lease promotion client — the half the enforcement toolkit never shipped.
The enforcement half (``daemon.py`` + ``mutator-gate.py``) ships and denies. The
promotion half has no production caller anywhere in the package: as of 0.0.48,
0.0.49 and 0.0.50-next.2207, ``begin_verification`` / ``observe_receipt`` /
``promote_lease`` are invoked only by ``broker-test-client.ts``, the acceptance
spec, unit tests, and two probes under ``docs/``. Consequence: **no lease on any
host can reach VERIFIED**, so every mutator is denied ``MUTATOR_UNVERIFIED`` by a
gate nothing can satisfy.
THE PROTOCOL (``daemon.py:578-754``)
------------------------------------
1. ``begin_verification`` broker revokes, mints a challenge, and returns the
exact ``receipt`` text the MODEL must emit
2. *the model emits that text verbatim as its ENTIRE latest message*
3. the runtime adapter ships that message to the daemon-owned observer socket
4. ``observe_receipt`` -> ``PENDING_PROMOTION``
5. ``promote_lease`` -> ``VERIFIED``
THIS MODULE IMPLEMENTS 1, 4 AND 5 NEVER 2
-------------------------------------------
Step 2 is the security property, not a formality. ``is_verbatim_receipt`` uses
``hmac.compare_digest`` against the exact minted string explicitly "not a
transcript substring" (``receipt_challenge.py``). Promotion therefore requires a
live model that received the challenge in its context and echoed it exactly.
``receipt-observer-client.py`` will post ANY string as the latest assistant
message. A promotion client that posted its own receipt would satisfy the broker
while proving nothing a gate-disabler indistinguishable from a working fix
unless someone looks for it. **This module never posts a receipt.** Emitting it
belongs to the runtime adapter, where a real model turn happens.
The construction binds the exact normative source bytes. ``h_source`` /
``h_payload`` are derived by the framework's own
``normative_fragments.build_payload`` rather than reimplemented: the broker
derives them the same way and any divergence yields ``PAYLOAD_BINDING_MISMATCH``.
There must be exactly one implementation.
WHAT THE BINDING DOES *NOT* PROVE
---------------------------------
It is tempting to read a VERIFIED lease as "this agent is running THIS law".
**It does not mean that**, and writing it down that way is how the belief spread.
The broker holds no reference copy of any normative source and never opens one;
it recomputes ``h_source`` / ``h_payload`` from the fragment bytes THIS CLIENT
sent and compares them to the binding THIS CLIENT sent (``daemon.py:602-616``).
Both sides of that comparison originate here, so it detects corruption in
transit and nothing else. What the binding actually asserts is "the client
claims these bytes, self-consistently".
Making it mean the stronger thing requires the broker to re-read the on-disk
sources itself, against a manifest the agent cannot rewrite i.e. broker code
attestation under its own uid. Until then, do not cite a VERIFIED lease as
evidence of law integrity.
Usage
-----
lease_promote.py --begin # prints the receipt the MODEL must emit
lease_promote.py --complete <challenge> # after the adapter observed it
"""
from __future__ import annotations
import argparse
import base64
import hashlib
import json
import os
import socket
import sys
from pathlib import Path
from typing import Final
# Isolated (`python -I`) adapter invocations must still import co-located
# framework modules; never depend on the caller's PYTHONPATH.
_MODULE_DIRECTORY = str(Path(__file__).resolve().parent)
if _MODULE_DIRECTORY not in sys.path:
sys.path.insert(0, _MODULE_DIRECTORY)
from normative_fragments import NormativeFragment, build_payload # noqa: E402
MAX_FRAME: Final = 64 * 1024
BROKER_TIMEOUT_SECONDS: Final = 3.0
SCHEMA_VERSION: Final = 1
MANIFEST_VERSION: Final = 1
GENERATOR_VERSION: Final = "mosaic/lease_promote@1"
DEFAULT_TTL_SECONDS: Final = 3600
# Normative sources whose exact bytes bind the lease, in binding order. Order is
# load-bearing: ``h_source`` frames the resolved sequence, so reordering changes
# the derivation. Never fabricate a source that is not on disk.
FRAGMENT_SOURCES: Final = (
"CONSTITUTION.md",
"AGENTS.md",
"SOUL.md",
"USER.md",
"STANDARDS.md",
"TOOLS.md",
)
# Framework-owned sources, reconciled on every upgrade — `install.sh:76`
# FRAMEWORK_OWNED and `config/file-adapter.ts` FRAMEWORK_OWNED_FILES — plus the
# per-runtime contract shipped under `framework/runtime/<runtime>/`. A deployment
# missing one of these is broken, not minimal, so their absence is refused rather
# than silently dropped from the binding.
#
# SOUL.md and USER.md are deliberately excluded: install.sh does not seed them
# ("intentionally NOT seeded here — they are generated by `mosaic init`"), so a
# fresh install legitimately lacks both. TOOLS.md is user-seeded on first install
# only. Absence of those three is reported, not fatal.
REQUIRED_SOURCES: Final = frozenset({"CONSTITUTION.md", "AGENTS.md", "STANDARDS.md"})
class IncompleteBinding(RuntimeError):
"""A source that must bind this lease could not be read.
**Never downgrade this to a skip.** The broker recomputes the hashes from the
fragments it is sent, so an omitted fragment is internally consistent and
``PAYLOAD_BINDING_MISMATCH`` cannot fire a partial law promotes exactly like
a complete one, and nothing downstream can tell the difference. Dropping an
unreadable source therefore does not degrade the binding, it forges a smaller
one. Fail here, where the omission is still visible.
"""
def mosaic_home() -> Path:
return Path(os.environ.get("MOSAIC_HOME") or Path.home() / ".config" / "mosaic")
def broker_socket() -> Path:
value = os.environ.get("MOSAIC_LEASE_BROKER_SOCKET")
if value:
return Path(value)
runtime_dir = os.environ.get("XDG_RUNTIME_DIR")
if runtime_dir:
return Path(runtime_dir) / "mosaic-lease" / "broker.sock"
return Path(f"/run/user/{os.getuid()}/mosaic-lease/broker.sock")
def session_identity() -> tuple[str, int, str]:
"""Session id, CURRENT generation, runtime.
The generation file wins over the env var, matching ``lease_generation.py``.
Sending a generation HIGHER than the broker's would revoke this session's own
authority (``daemon.py:342-344``), so this never guesses.
"""
session_id = os.environ["MOSAIC_LEASE_SESSION_ID"]
runtime = os.environ["MOSAIC_LEASE_RUNTIME"]
state_file = os.environ.get("MOSAIC_LEASE_GENERATION_FILE")
if state_file:
try:
return session_id, int(Path(state_file).read_text().strip()), runtime
except (OSError, ValueError):
pass
return session_id, int(os.environ["MOSAIC_RUNTIME_GENERATION"]), runtime
def build_construction(runtime: str) -> tuple[dict[str, object], object]:
"""Assemble the wire construction and derive its hashes with the sole builder."""
runtime_contract = f"runtime/{runtime}/RUNTIME.md"
sources = list(FRAGMENT_SOURCES) + [runtime_contract]
required = REQUIRED_SOURCES | {runtime_contract}
wire_fragments: list[dict[str, str]] = []
objects: list[NormativeFragment] = []
absent: list[str] = []
for source_id in sources:
try:
content = (mosaic_home() / source_id).read_bytes()
except FileNotFoundError:
# Genuinely not on disk. Legitimate only for operator-owned sources.
if source_id in required:
raise IncompleteBinding(
f"required normative source is absent: {source_id}"
) from None
absent.append(source_id)
continue
except OSError as exc:
# The path resolves but will not read — EACCES, EIO, EISDIR, ELOOP.
# That is an anomaly for EVERY source, optional ones included: an
# unreadable file is not an un-configured one, and treating it as
# absent is what lets a permission change quietly shrink the law.
raise IncompleteBinding(
f"normative source is present but unreadable: {source_id} "
f"({type(exc).__name__})"
) from exc
digest = hashlib.sha256(content).hexdigest()
wire_fragments.append(
{
"source_id": source_id,
"content_base64": base64.b64encode(content).decode("ascii"),
"expected_sha256": digest,
}
)
objects.append(NormativeFragment(source_id, content, digest))
if not wire_fragments:
raise IncompleteBinding("no normative sources found — refusing an empty binding")
# Absence is legitimate here but never invisible. The omission is already
# baked into h_source (the framed source sequence differs), but nothing
# compares h_source to an expected value, so this line is the only place a
# human learns the binding was narrower than the full set.
if absent:
print(
f"lease_promote: binding omits absent operator sources: {', '.join(absent)}",
file=sys.stderr,
)
result = build_payload(
manifest_version=MANIFEST_VERSION,
generator_version=GENERATOR_VERSION,
fragments=objects,
)
if result.injectionDecision != "ACCEPTED" or not result.promotion:
raise RuntimeError(f"construction refused locally: {result.source_reason}")
return (
{
"manifest_version": MANIFEST_VERSION,
"generator_version": GENERATOR_VERSION,
"fragments": wire_fragments,
},
result,
)
def broker_request(payload: dict[str, object]) -> dict[str, object]:
raw = (json.dumps(payload, separators=(",", ":")) + "\n").encode()
if len(raw) > MAX_FRAME:
raise ValueError(
f"request too large ({len(raw)} bytes); broker frame cap is {MAX_FRAME}"
)
response = bytearray()
with socket.socket(socket.AF_UNIX, socket.SOCK_STREAM) as connection:
connection.settimeout(BROKER_TIMEOUT_SECONDS)
connection.connect(str(broker_socket()))
connection.sendall(raw)
connection.shutdown(socket.SHUT_WR)
while len(response) <= MAX_FRAME:
chunk = connection.recv(4096)
if not chunk:
break
response.extend(chunk)
if len(response) > MAX_FRAME or not response.endswith(b"\n"):
raise ValueError("invalid broker reply")
value = json.loads(response)
if not isinstance(value, dict):
raise ValueError("invalid broker reply")
return value
def begin(
ttl_seconds: int = DEFAULT_TTL_SECONDS,
compaction_epoch: int = 0,
request_epoch: int = 0,
) -> dict[str, object]:
"""Step 1. Returns the broker reply, including the exact ``receipt`` text."""
session_id, generation, runtime = session_identity()
construction, derived = build_construction(runtime)
return broker_request(
{
"action": "begin_verification",
"session_id": session_id,
"runtime_generation": generation,
"runtime": runtime,
"ttl_seconds": ttl_seconds,
"binding": {
"compaction_epoch": compaction_epoch,
"request_epoch": request_epoch,
"h_source": derived.h_source,
"h_payload": derived.h_payload,
"schema_version": SCHEMA_VERSION,
},
"construction": construction,
}
)
def complete(challenge: str) -> dict[str, object]:
"""Steps 4-5. Assumes the model already emitted the receipt and the adapter
shipped it to the observer socket."""
session_id, generation, _ = session_identity()
observed = broker_request(
{
"action": "observe_receipt",
"session_id": session_id,
"runtime_generation": generation,
"receipt_challenge": challenge,
}
)
if observed.get("ok") is not True or observed.get("state") != "PENDING_PROMOTION":
return {"stage": "observe_receipt", **observed}
promoted = broker_request(
{
"action": "promote_lease",
"session_id": session_id,
"runtime_generation": generation,
"receipt_challenge": challenge,
}
)
return {"stage": "promote_lease", **promoted}
def main(argv: list[str] | None = None) -> int:
parser = argparse.ArgumentParser(description="Mosaic lease promotion client.")
group = parser.add_mutually_exclusive_group(required=True)
group.add_argument(
"--begin",
action="store_true",
help="mint a challenge; prints the receipt the MODEL must emit verbatim",
)
group.add_argument(
"--complete",
metavar="CHALLENGE",
help="observe the emitted receipt and promote the lease",
)
parser.add_argument("--ttl-seconds", type=int, default=DEFAULT_TTL_SECONDS)
arguments = parser.parse_args(argv)
try:
if arguments.begin:
print(json.dumps(begin(ttl_seconds=arguments.ttl_seconds), indent=2))
else:
print(json.dumps(complete(arguments.complete), indent=2))
except KeyError as exc:
print(f"missing lease environment: {exc}; not a lease-gated session", file=sys.stderr)
return 2
except (OSError, ValueError, RuntimeError, json.JSONDecodeError) as exc:
print(f"{type(exc).__name__}: {exc}", file=sys.stderr)
return 2
return 0
if __name__ == "__main__":
raise SystemExit(main())
@@ -1,399 +0,0 @@
#!/usr/bin/env python3
"""Claude UserPromptSubmit hook for operator-triggered lease promotion."""
from __future__ import annotations
import fcntl
import importlib.util
import json
import os
import secrets
import stat
import subprocess
import sys
import time
from collections.abc import Callable, Mapping
from pathlib import Path
from typing import Final, TextIO
_MODULE_DIRECTORY = str(Path(__file__).resolve().parent)
if _MODULE_DIRECTORY not in sys.path:
sys.path.insert(0, _MODULE_DIRECTORY)
from receipt_challenge import receipt_for # noqa: E402
_observer_spec = importlib.util.spec_from_file_location(
"mosaic_receipt_observer_client", Path(__file__).resolve().with_name("receipt-observer-client.py")
)
if _observer_spec is None or _observer_spec.loader is None:
raise RuntimeError("unable to load receipt observer client")
_observer_module = importlib.util.module_from_spec(_observer_spec)
_observer_spec.loader.exec_module(_observer_module)
observer_request = _observer_module.observer_request
MAX_FRAME: Final = 64 * 1024
PENDING_MAX_AGE_SECONDS: Final = 60 * 60
PROMOTER_TIMEOUT_SECONDS: Final = 10.0
PROMOTION_PROMPT: Final = "/mosaic-promote"
PROMOTER: Final = Path(__file__).resolve().with_name("lease_promote.py")
PENDING_DIRECTORY: Final = "mosaic-lease"
AUTHORIZATION_DIRECTORY: Final = "authorizations"
AUTHORIZATION_TTL_SECONDS: Final = 60
LEASE_TTL_SECONDS: Final = 60 * 60
LOCK_FILE: Final = "promotion.lock"
RESULT_FILE: Final = "last-result.json"
EXPECTED_BEGIN_KEYS: Final = frozenset(
{"ok", "state", "receipt_challenge", "receipt", "binding"}
)
EXPECTED_BINDING_KEYS: Final = frozenset(
{
"compaction_epoch",
"request_epoch",
"h_source",
"h_payload",
"runtime_generation",
"schema_version",
}
)
class PromotionAlreadyInProgress(RuntimeError):
pass
def reject_duplicate_json_keys(pairs: list[tuple[str, object]]) -> dict[str, object]:
value: dict[str, object] = {}
for key, item in pairs:
if key in value:
raise ValueError("duplicate promoter JSON key")
value[key] = item
return value
def read_hook_input(stream: object) -> dict[str, object]:
raw = getattr(stream, "buffer", stream).read(MAX_FRAME + 1)
if not isinstance(raw, bytes) or len(raw) > MAX_FRAME:
raise ValueError("invalid UserPromptSubmit input")
value = json.loads(raw, object_pairs_hook=reject_duplicate_json_keys)
if not isinstance(value, dict):
raise ValueError("invalid UserPromptSubmit input")
return value
def emit_context(stream: TextIO, message: str) -> None:
json.dump(
{
"hookSpecificOutput": {
"hookEventName": "UserPromptSubmit",
"additionalContext": message,
}
},
stream,
separators=(",", ":"),
)
stream.write("\n")
def session_pending_name(environ: Mapping[str, str]) -> tuple[Path, str]:
runtime_dir = Path(environ["XDG_RUNTIME_DIR"])
session_id = environ["MOSAIC_LEASE_SESSION_ID"]
if not runtime_dir.is_absolute():
raise ValueError("XDG_RUNTIME_DIR must be absolute")
if len(session_id) != 64 or any(character not in "0123456789abcdef" for character in session_id):
raise ValueError("invalid lease session id")
return runtime_dir, f"pending-{session_id}"
def open_pending_directory(runtime_dir: Path) -> int:
directory_flags = (
os.O_RDONLY
| getattr(os, "O_CLOEXEC", 0)
| getattr(os, "O_DIRECTORY", 0)
| getattr(os, "O_NOFOLLOW", 0)
)
runtime_descriptor = os.open(runtime_dir, directory_flags)
try:
runtime_metadata = os.fstat(runtime_descriptor)
if (
not stat.S_ISDIR(runtime_metadata.st_mode)
or runtime_metadata.st_uid != os.getuid()
or stat.S_IMODE(runtime_metadata.st_mode) != 0o700
):
raise ValueError("unsafe XDG runtime directory")
try:
os.mkdir(PENDING_DIRECTORY, mode=0o700, dir_fd=runtime_descriptor)
except FileExistsError:
pass
descriptor = os.open(PENDING_DIRECTORY, directory_flags, dir_fd=runtime_descriptor)
finally:
os.close(runtime_descriptor)
metadata = os.fstat(descriptor)
if (
not stat.S_ISDIR(metadata.st_mode)
or metadata.st_uid != os.getuid()
or stat.S_IMODE(metadata.st_mode) != 0o700
):
os.close(descriptor)
raise ValueError("unsafe promotion pending directory")
return descriptor
def acquire_lock(directory_descriptor: int) -> int:
flags = (
os.O_RDWR
| os.O_CREAT
| getattr(os, "O_CLOEXEC", 0)
| getattr(os, "O_NOFOLLOW", 0)
)
descriptor = os.open(LOCK_FILE, flags, 0o600, dir_fd=directory_descriptor)
metadata = os.fstat(descriptor)
if (
not stat.S_ISREG(metadata.st_mode)
or metadata.st_uid != os.getuid()
or stat.S_IMODE(metadata.st_mode) != 0o600
):
os.close(descriptor)
raise ValueError("unsafe promotion lock file")
try:
fcntl.flock(descriptor, fcntl.LOCK_EX | fcntl.LOCK_NB)
except BlockingIOError as error:
os.close(descriptor)
raise PromotionAlreadyInProgress() from error
return descriptor
def sweep_stale_pending(directory_descriptor: int, current_time: float) -> None:
cutoff = current_time - PENDING_MAX_AGE_SECONDS
removed = False
with os.scandir(directory_descriptor) as entries:
for candidate in entries:
if not (
candidate.name.startswith("pending-")
or candidate.name.startswith(".pending-")
):
continue
try:
metadata = candidate.stat(follow_symlinks=False)
if metadata.st_mtime < cutoff and not stat.S_ISDIR(metadata.st_mode):
os.unlink(candidate.name, dir_fd=directory_descriptor)
removed = True
except FileNotFoundError:
continue
if removed:
os.fsync(directory_descriptor)
def consume_authorization(directory_descriptor: int, session_id: str, wall_clock: float) -> str | None:
flags = os.O_RDONLY | getattr(os, "O_CLOEXEC", 0) | getattr(os, "O_DIRECTORY", 0) | getattr(os, "O_NOFOLLOW", 0)
try:
authorization_descriptor = os.open(AUTHORIZATION_DIRECTORY, flags, dir_fd=directory_descriptor)
except FileNotFoundError:
return None
try:
metadata = os.fstat(authorization_descriptor)
if not stat.S_ISDIR(metadata.st_mode) or metadata.st_uid != os.getuid() or stat.S_IMODE(metadata.st_mode) != 0o700:
raise ValueError("unsafe promotion authorization directory")
name = f"{session_id}.auth"
try:
descriptor = os.open(name, os.O_RDONLY | getattr(os, "O_CLOEXEC", 0) | getattr(os, "O_NOFOLLOW", 0), dir_fd=authorization_descriptor)
except FileNotFoundError:
return None
try:
token_metadata = os.fstat(descriptor)
if not stat.S_ISREG(token_metadata.st_mode) or token_metadata.st_uid != os.getuid() or stat.S_IMODE(token_metadata.st_mode) != 0o600 or token_metadata.st_size <= 0 or token_metadata.st_size > MAX_FRAME:
raise ValueError("unsafe promotion authorization")
raw = os.read(descriptor, MAX_FRAME + 1)
finally:
os.close(descriptor)
os.unlink(name, dir_fd=authorization_descriptor)
os.fsync(authorization_descriptor)
token = json.loads(raw, object_pairs_hook=reject_duplicate_json_keys)
if not isinstance(token, dict) or set(token) != {"nonce", "seat", "session_id", "expires_at", "ts"}:
return None
nonce = token.get("nonce")
expires_at = token.get("expires_at")
issued_at = token.get("ts")
if token.get("session_id") != session_id or not isinstance(token.get("seat"), str) or not isinstance(nonce, str) or len(nonce) != 64 or any(char not in "0123456789abcdef" for char in nonce) or type(expires_at) not in (int, float) or type(issued_at) not in (int, float) or expires_at <= wall_clock or expires_at > issued_at + AUTHORIZATION_TTL_SECONDS:
return None
return nonce
finally:
os.close(authorization_descriptor)
def write_result(directory_descriptor: int, attempt_id: str, verified: bool, reason: str | None, session_id: str, wall_clock: float) -> None:
temporary = f".{RESULT_FILE}.tmp-{secrets.token_hex(8)}"
descriptor = os.open(temporary, os.O_WRONLY | os.O_CREAT | os.O_EXCL | getattr(os, "O_CLOEXEC", 0) | getattr(os, "O_NOFOLLOW", 0), 0o600, dir_fd=directory_descriptor)
try:
os.fchmod(descriptor, 0o600)
with os.fdopen(descriptor, "w", encoding="utf-8", closefd=False) as stream:
json.dump({"attempt_id": attempt_id, "expires_at_wallclock": wall_clock + LEASE_TTL_SECONDS if verified else None, "reason": reason, "session_id": session_id, "ts": wall_clock, "verified": verified}, stream, separators=(",", ":"), sort_keys=True)
stream.flush(); os.fsync(stream.fileno())
os.replace(temporary, RESULT_FILE, src_dir_fd=directory_descriptor, dst_dir_fd=directory_descriptor)
os.fsync(directory_descriptor)
finally:
os.close(descriptor)
def write_pending(directory_descriptor: int, name: str, challenge: str) -> None:
temporary = f".{name}.tmp-{secrets.token_hex(8)}"
flags = (
os.O_WRONLY
| os.O_CREAT
| os.O_EXCL
| getattr(os, "O_CLOEXEC", 0)
| getattr(os, "O_NOFOLLOW", 0)
)
descriptor = os.open(temporary, flags, 0o600, dir_fd=directory_descriptor)
try:
os.fchmod(descriptor, 0o600)
with os.fdopen(descriptor, "w", encoding="utf-8", closefd=False) as stream:
stream.write(challenge)
stream.flush()
os.fsync(stream.fileno())
os.replace(
temporary,
name,
src_dir_fd=directory_descriptor,
dst_dir_fd=directory_descriptor,
)
os.fsync(directory_descriptor)
except Exception:
try:
os.unlink(temporary, dir_fd=directory_descriptor)
except FileNotFoundError:
pass
raise
finally:
os.close(descriptor)
def parse_begin_reply(
completed: subprocess.CompletedProcess[str],
) -> tuple[str, dict[str, object] | None]:
if completed.returncode != 0:
return f"PROMOTER_EXIT_{completed.returncode}", None
try:
value = json.loads(
completed.stdout,
object_pairs_hook=reject_duplicate_json_keys,
)
except (json.JSONDecodeError, RecursionError, TypeError, ValueError):
return "INVALID_PROMOTER_REPLY", None
if not isinstance(value, dict):
return "INVALID_PROMOTER_REPLY", None
if value.get("ok") is False and set(value) == {"ok", "code"}:
code = value.get("code")
return code if isinstance(code, str) and code else "PROMOTION_BEGIN_REFUSED", value
if set(value) != EXPECTED_BEGIN_KEYS or value.get("ok") is not True:
return "INVALID_PROMOTER_REPLY", None
if value.get("state") != "PENDING_VERIFICATION":
return "INVALID_PROMOTER_REPLY", None
challenge = value.get("receipt_challenge")
receipt = value.get("receipt")
binding = value.get("binding")
if (
not isinstance(challenge, str)
or len(challenge) != 64
or any(character not in "0123456789abcdef" for character in challenge)
or not isinstance(receipt, str)
or not isinstance(binding, dict)
or set(binding) != EXPECTED_BINDING_KEYS
):
return "INVALID_PROMOTER_REPLY", None
integer_fields = (
"compaction_epoch",
"request_epoch",
"runtime_generation",
"schema_version",
)
if any(type(binding.get(field)) is not int or binding[field] < 0 for field in integer_fields):
return "INVALID_PROMOTER_REPLY", None
if not all(
isinstance(binding.get(field), str)
and len(binding[field]) == 64
and all(character in "0123456789abcdef" for character in binding[field])
for field in ("h_source", "h_payload")
):
return "INVALID_PROMOTER_REPLY", None
if not secrets.compare_digest(
receipt.encode("utf-8"),
receipt_for(challenge, binding).encode("utf-8"),
):
return "INVALID_PROMOTER_REPLY", None
return "", value
def main(
*,
environ: Mapping[str, str] | None = None,
stdin: object | None = None,
stdout: TextIO | None = None,
stderr: TextIO | None = None,
run: Callable[..., subprocess.CompletedProcess[str]] = subprocess.run,
now: Callable[[], float] = time.time,
) -> int:
source_environment = os.environ if environ is None else environ
input_stream = sys.stdin if stdin is None else stdin
output_stream = sys.stdout if stdout is None else stdout
error_stream = sys.stderr if stderr is None else stderr
try:
hook_input = read_hook_input(input_stream)
except (OSError, RecursionError, ValueError, json.JSONDecodeError) as error:
print(f"Mosaic promotion trigger ignored invalid hook input: {error}", file=error_stream)
return 0
if hook_input.get("prompt") != PROMOTION_PROMPT:
return 0
directory_descriptor: int | None = None
lock_descriptor: int | None = None
try:
runtime_dir, pending_name = session_pending_name(source_environment)
session_id = source_environment["MOSAIC_LEASE_SESSION_ID"]
directory_descriptor = open_pending_directory(runtime_dir)
lock_descriptor = acquire_lock(directory_descriptor)
wall_clock = now()
nonce = consume_authorization(directory_descriptor, session_id, wall_clock)
if nonce is None:
write_result(directory_descriptor, "0" * 64, False, "NOT_AUTHORIZED", session_id, wall_clock)
print("Mosaic promotion denied: NOT_AUTHORIZED.", file=error_stream)
return 0
sweep_stale_pending(directory_descriptor, wall_clock)
completed = run([sys.executable, "-I", "-S", "-B", str(PROMOTER), "--begin"], check=False, capture_output=True, text=True, env=dict(source_environment), timeout=PROMOTER_TIMEOUT_SECONDS)
code, reply = parse_begin_reply(completed)
if code or reply is None:
write_result(directory_descriptor, nonce, False, code or "PROMOTION_BEGIN_FAILED", session_id, now())
return 0
challenge = str(reply["receipt_challenge"])
observation = observer_request(
Path(source_environment["MOSAIC_RECEIPT_OBSERVER_SOCKET"]),
{"action": "record_runtime_observation", "session_id": session_id, "runtime_generation": int(source_environment["MOSAIC_RUNTIME_GENERATION"]), "runtime": "claude", "latest_assistant_message": reply["receipt"]},
)
if set(observation) != {"ok"} or observation.get("ok") is not True:
write_result(directory_descriptor, challenge, False, "OBSERVATION_REJECTED", session_id, now())
return 0
completion = run([sys.executable, "-I", "-S", "-B", str(PROMOTER), "--complete", challenge], check=False, capture_output=True, text=True, env=dict(source_environment), timeout=PROMOTER_TIMEOUT_SECONDS)
try:
outcome = json.loads(completion.stdout, object_pairs_hook=reject_duplicate_json_keys)
except (json.JSONDecodeError, ValueError):
outcome = None
if completion.returncode == 0 and isinstance(outcome, dict) and outcome.get("stage") == "promote_lease" and outcome.get("ok") is True and outcome.get("state") == "VERIFIED":
write_result(directory_descriptor, challenge, True, None, session_id, now())
else:
reason = outcome.get("code") if isinstance(outcome, dict) and isinstance(outcome.get("code"), str) else "PROMOTION_INCOMPLETE"
write_result(directory_descriptor, challenge, False, reason, session_id, now())
except PromotionAlreadyInProgress:
print("Mosaic promotion denied: PROMOTION_ALREADY_IN_PROGRESS.", file=error_stream)
except (KeyError, OSError, RecursionError, ValueError, subprocess.SubprocessError) as error:
print(f"Mosaic promotion begin failed: {type(error).__name__}: {error}", file=error_stream)
finally:
if lock_descriptor is not None:
os.close(lock_descriptor)
if directory_descriptor is not None:
os.close(directory_descriptor)
return 0
if __name__ == "__main__":
raise SystemExit(main())
@@ -1,361 +0,0 @@
#!/usr/bin/env python3
"""Claude Stop hook that completes a pending operator-triggered promotion."""
from __future__ import annotations
import fcntl
import json
import os
import secrets
import stat
import subprocess
import sys
import time
from collections.abc import Callable, Mapping
from pathlib import Path
from typing import Final, NamedTuple, TextIO
MAX_FRAME: Final = 64 * 1024
PROMOTER_TIMEOUT_SECONDS: Final = 10.0
LEASE_TTL_SECONDS: Final = 60 * 60
PROMOTER: Final = Path(__file__).resolve().with_name("lease_promote.py")
PENDING_DIRECTORY: Final = "mosaic-lease"
LOCK_FILE: Final = "promotion.lock"
RESULT_FILE: Final = "last-result.json"
TERMINAL_FAILURE_CODES: Final = frozenset(
{
"RECEIPT_REPLAY",
"RECEIPT_MISMATCH",
"INVALID_LEASE_TRANSITION",
"PROMOTION_TOKEN_INVALID",
}
)
class PendingChallenge(NamedTuple):
value: str
device: int
inode: int
def reject_duplicate_json_keys(pairs: list[tuple[str, object]]) -> dict[str, object]:
value: dict[str, object] = {}
for key, item in pairs:
if key in value:
raise ValueError("duplicate promoter JSON key")
value[key] = item
return value
def session_pending_name(environ: Mapping[str, str]) -> tuple[Path, str]:
runtime_dir = Path(environ["XDG_RUNTIME_DIR"])
session_id = environ["MOSAIC_LEASE_SESSION_ID"]
if not runtime_dir.is_absolute():
raise ValueError("XDG_RUNTIME_DIR must be absolute")
if len(session_id) != 64 or any(character not in "0123456789abcdef" for character in session_id):
raise ValueError("invalid lease session id")
return runtime_dir, f"pending-{session_id}"
def open_pending_directory(runtime_dir: Path) -> int | None:
directory_flags = (
os.O_RDONLY
| getattr(os, "O_CLOEXEC", 0)
| getattr(os, "O_DIRECTORY", 0)
| getattr(os, "O_NOFOLLOW", 0)
)
try:
runtime_descriptor = os.open(runtime_dir, directory_flags)
except FileNotFoundError:
return None
try:
runtime_metadata = os.fstat(runtime_descriptor)
if (
not stat.S_ISDIR(runtime_metadata.st_mode)
or runtime_metadata.st_uid != os.getuid()
or stat.S_IMODE(runtime_metadata.st_mode) != 0o700
):
raise ValueError("unsafe XDG runtime directory")
try:
descriptor = os.open(PENDING_DIRECTORY, directory_flags, dir_fd=runtime_descriptor)
except FileNotFoundError:
return None
finally:
os.close(runtime_descriptor)
metadata = os.fstat(descriptor)
if (
not stat.S_ISDIR(metadata.st_mode)
or metadata.st_uid != os.getuid()
or stat.S_IMODE(metadata.st_mode) != 0o700
):
os.close(descriptor)
raise ValueError("unsafe promotion pending directory")
return descriptor
def acquire_lock(directory_descriptor: int) -> int:
flags = (
os.O_RDWR
| os.O_CREAT
| getattr(os, "O_CLOEXEC", 0)
| getattr(os, "O_NOFOLLOW", 0)
)
descriptor = os.open(LOCK_FILE, flags, 0o600, dir_fd=directory_descriptor)
metadata = os.fstat(descriptor)
if (
not stat.S_ISREG(metadata.st_mode)
or metadata.st_uid != os.getuid()
or stat.S_IMODE(metadata.st_mode) != 0o600
):
os.close(descriptor)
raise ValueError("unsafe promotion lock file")
try:
fcntl.flock(descriptor, fcntl.LOCK_EX | fcntl.LOCK_NB)
except BlockingIOError:
os.close(descriptor)
raise
return descriptor
def read_pending(directory_descriptor: int, name: str) -> PendingChallenge | None:
flags = os.O_RDONLY | getattr(os, "O_CLOEXEC", 0) | getattr(os, "O_NOFOLLOW", 0)
try:
descriptor = os.open(name, flags, dir_fd=directory_descriptor)
except FileNotFoundError:
return None
try:
metadata = os.fstat(descriptor)
if (
not stat.S_ISREG(metadata.st_mode)
or metadata.st_uid != os.getuid()
or stat.S_IMODE(metadata.st_mode) != 0o600
or metadata.st_size <= 0
or metadata.st_size > MAX_FRAME
):
raise ValueError("unsafe promotion pending file")
raw = os.read(descriptor, MAX_FRAME + 1)
finally:
os.close(descriptor)
if len(raw) > MAX_FRAME:
raise ValueError("oversized promotion challenge")
challenge = raw.decode("utf-8")
if (
len(challenge) != 64
or any(character not in "0123456789abcdef" for character in challenge)
):
raise ValueError("invalid promotion challenge")
return PendingChallenge(challenge, metadata.st_dev, metadata.st_ino)
def write_result(
directory_descriptor: int,
attempt_id: str,
verified: bool,
reason: str | None,
session_id: str,
wall_clock: float,
) -> None:
result = {
"attempt_id": attempt_id,
"expires_at_wallclock": wall_clock + LEASE_TTL_SECONDS if verified else None,
"reason": reason,
"session_id": session_id,
"ts": wall_clock,
"verified": verified,
}
temporary = f".{RESULT_FILE}.tmp-{secrets.token_hex(8)}"
flags = (
os.O_WRONLY
| os.O_CREAT
| os.O_EXCL
| getattr(os, "O_CLOEXEC", 0)
| getattr(os, "O_NOFOLLOW", 0)
)
descriptor = os.open(temporary, flags, 0o600, dir_fd=directory_descriptor)
try:
os.fchmod(descriptor, 0o600)
with os.fdopen(descriptor, "w", encoding="utf-8", closefd=False) as stream:
json.dump(result, stream, separators=(",", ":"), sort_keys=True)
stream.flush()
os.fsync(stream.fileno())
os.replace(
temporary,
RESULT_FILE,
src_dir_fd=directory_descriptor,
dst_dir_fd=directory_descriptor,
)
os.fsync(directory_descriptor)
except Exception:
try:
os.unlink(temporary, dir_fd=directory_descriptor)
except FileNotFoundError:
pass
raise
finally:
os.close(descriptor)
def delete_pending_if_unchanged(
directory_descriptor: int,
name: str,
pending: PendingChallenge,
error_stream: TextIO,
) -> None:
quarantine = f".{name}.delete-{secrets.token_hex(8)}"
try:
os.rename(
name,
quarantine,
src_dir_fd=directory_descriptor,
dst_dir_fd=directory_descriptor,
)
except FileNotFoundError:
return
except OSError as error:
print(f"Mosaic promotion could not quarantine pending file: {error}", file=error_stream)
return
try:
moved = os.stat(
quarantine,
dir_fd=directory_descriptor,
follow_symlinks=False,
)
if (moved.st_dev, moved.st_ino) == (pending.device, pending.inode):
os.unlink(quarantine, dir_fd=directory_descriptor)
os.fsync(directory_descriptor)
return
print("Mosaic promotion pending file changed; preserving replacement.", file=error_stream)
try:
os.link(
quarantine,
name,
src_dir_fd=directory_descriptor,
dst_dir_fd=directory_descriptor,
follow_symlinks=False,
)
except FileExistsError:
print(
f"Mosaic promotion preserved replacement as {quarantine}.",
file=error_stream,
)
else:
os.unlink(quarantine, dir_fd=directory_descriptor)
os.fsync(directory_descriptor)
except OSError as error:
print(f"Mosaic promotion could not resolve pending file: {error}", file=error_stream)
def parse_reply(completed: subprocess.CompletedProcess[str]) -> dict[str, object] | None:
if completed.returncode != 0:
return None
try:
value = json.loads(
completed.stdout,
object_pairs_hook=reject_duplicate_json_keys,
)
except (json.JSONDecodeError, RecursionError, TypeError, ValueError):
return None
if not isinstance(value, dict):
return None
if set(value) == {"stage", "ok", "state"}:
if (
value.get("stage") == "promote_lease"
and value.get("ok") is True
and value.get("state") == "VERIFIED"
):
return value
return None
if set(value) == {"stage", "ok", "code"}:
if (
value.get("stage") in {"observe_receipt", "promote_lease"}
and value.get("ok") is False
and isinstance(value.get("code"), str)
and value.get("code")
):
return value
return None
def main(
*,
environ: Mapping[str, str] | None = None,
stderr: TextIO | None = None,
run: Callable[..., subprocess.CompletedProcess[str]] = subprocess.run,
now: Callable[[], float] = time.time,
) -> int:
source_environment = os.environ if environ is None else environ
error_stream = sys.stderr if stderr is None else stderr
directory_descriptor: int | None = None
lock_descriptor: int | None = None
try:
runtime_dir, pending_name = session_pending_name(source_environment)
session_id = source_environment["MOSAIC_LEASE_SESSION_ID"]
directory_descriptor = open_pending_directory(runtime_dir)
if directory_descriptor is None:
return 0
try:
lock_descriptor = acquire_lock(directory_descriptor)
except (BlockingIOError, FileNotFoundError):
print("Mosaic promotion completion deferred: promotion is in progress.", file=error_stream)
return 0
pending = read_pending(directory_descriptor, pending_name)
if pending is None:
return 0
completed = run(
[
sys.executable,
"-I",
"-S",
"-B",
str(PROMOTER),
"--complete",
pending.value,
],
check=False,
capture_output=True,
text=True,
env=dict(source_environment),
timeout=PROMOTER_TIMEOUT_SECONDS,
)
reply = parse_reply(completed)
if reply is not None and reply.get("ok") is True:
write_result(directory_descriptor, pending.value, True, None, session_id, now())
delete_pending_if_unchanged(
directory_descriptor,
pending_name,
pending,
error_stream,
)
print("Mosaic lease promotion completed.", file=error_stream)
return 0
if reply is not None:
code = str(reply["code"])
print(f"Mosaic promotion incomplete: {code}.", file=error_stream)
if code in TERMINAL_FAILURE_CODES:
write_result(directory_descriptor, pending.value, False, code, session_id, now())
delete_pending_if_unchanged(
directory_descriptor,
pending_name,
pending,
error_stream,
)
else:
diagnostic = completed.stderr.strip() or f"promoter exit {completed.returncode}"
print(f"Mosaic promotion retryable failure: {diagnostic}.", file=error_stream)
except (KeyError, OSError, RecursionError, UnicodeError, ValueError, subprocess.SubprocessError) as error:
print(f"Mosaic promotion completion deferred: {type(error).__name__}: {error}", file=error_stream)
finally:
if lock_descriptor is not None:
os.close(lock_descriptor)
if directory_descriptor is not None:
os.close(directory_descriptor)
return 0
if __name__ == "__main__":
raise SystemExit(main())

Some files were not shown because too many files have changed in this diff Show More