From 1679594d844036cc7fb437f66ba5654a7265139f Mon Sep 17 00:00:00 2001 From: Relayflow Date: Wed, 23 Sep 2026 05:34:21 +0000 Subject: [PATCH 01/13] fix(sdk): classify worker lease loss with typed errors and verb context --- packages/sdk/src/journal-client.ts | 12 +- packages/sdk/src/llm-worker.ts | 2 +- packages/sdk/src/worker-lease.ts | 37 +++++- packages/sdk/src/worker.ts | 2 +- packages/sdk/tests/worker-lease-lost.test.ts | 120 +++++++++++++++++++ 5 files changed, 165 insertions(+), 8 deletions(-) create mode 100644 packages/sdk/tests/worker-lease-lost.test.ts diff --git a/packages/sdk/src/journal-client.ts b/packages/sdk/src/journal-client.ts index 2b08178b2..378f722de 100644 --- a/packages/sdk/src/journal-client.ts +++ b/packages/sdk/src/journal-client.ts @@ -51,6 +51,7 @@ interface Pending { /** A structured rejection returned by relayflowd over the journal protocol. */ export class JournalProtocolError extends Error { readonly code: string; + verb?: string; constructor(code: string, message: string) { super(`${code}: ${message}`); @@ -370,6 +371,9 @@ export class JournalClient extends EventEmitter { step_id: stepId, attempt, lease_id: leaseId, + }).catch(error => { + if (error instanceof JournalProtocolError) error.verb = 'step.heartbeat'; + throw error; }); } @@ -401,7 +405,10 @@ export class JournalClient extends EventEmitter { idempotency_key: idempotencyKey, completionReason, ...extra, - }, null); + }, null).catch(error => { + if (error instanceof JournalProtocolError) error.verb = 'step.complete'; + throw error; + }); } /** Satisfy `wait.event`; a human response arrives here too. */ @@ -422,6 +429,9 @@ export class JournalClient extends EventEmitter { attempt, idempotency_key: idempotencyKey, ...wait, + }).catch(error => { + if (error instanceof JournalProtocolError) error.verb = 'step.wait'; + throw error; }); } diff --git a/packages/sdk/src/llm-worker.ts b/packages/sdk/src/llm-worker.ts index a6271460b..3890aab05 100644 --- a/packages/sdk/src/llm-worker.ts +++ b/packages/sdk/src/llm-worker.ts @@ -46,7 +46,7 @@ export class LlmWorker extends EventEmitter { private readonly onDispatch = (dispatch: StepDispatchEvent): void => { if (this.closing || dispatch.step_type !== 'llm') return; - const running = this.execute(dispatch).catch(error => { this.emit('error', error); }); + const running = this.execute(dispatch).catch(error => { this.emit('error', error, dispatch); }); this.inFlight.add(running); void running.finally(() => { this.inFlight.delete(running); }); }; diff --git a/packages/sdk/src/worker-lease.ts b/packages/sdk/src/worker-lease.ts index 5424b7e3a..9218e8982 100644 --- a/packages/sdk/src/worker-lease.ts +++ b/packages/sdk/src/worker-lease.ts @@ -1,6 +1,33 @@ -import type { JournalClient } from './journal-client.js'; +import { JournalProtocolError, type JournalClient } from './journal-client.js'; import type { StepDispatchEvent } from './protocol.js'; + +export class WorkerLeaseLostError extends Error { + constructor(readonly reason: 'already_expired' | 'renewal_expired' | 'completion_expired', message: string) { + super(message); + this.name = 'WorkerLeaseLostError'; + } +} + +/** No cause traversal: worker errors preserve their original identity. */ +export function isLeaseLost(error: unknown): boolean { + return error instanceof WorkerLeaseLostError || (error instanceof JournalProtocolError + && (error.code === 'lease_conflict' || (error.code === 'run_terminal' + && ['step.heartbeat', 'step.complete', 'step.wait'].includes(error.verb ?? '')))); +} + +export function onWorkerFailure(label: string, fatal: (error: unknown) => void) { + return (error: unknown, dispatch: StepDispatchEvent): void => { + if (!isLeaseLost(error)) { fatal(error); return; } + // The kernel owns this attempt's fate; its journal supplies the run outcome. + // stderr keeps this diagnostic out of structured reports on stdout. + process.emitWarning( + `${label}: run_id=${dispatch.run_id} step_id=${dispatch.step_id} attempt=${dispatch.attempt}: ${String(error)}`, + { code: 'FLOWS_WORKER_LEASE_LOST' }, + ); + }; +} + /** Hold the dispatched lease only while its subprocess is still ours to run. */ export async function withWorkerLease( client: JournalClient, @@ -17,11 +44,11 @@ export async function withWorkerLease( const armExpiry = (deadline: number): number => { const remaining = deadline - Date.now(); if (!Number.isFinite(remaining) || remaining <= 0) { - throw new Error(`Agent lease is already expired for ${dispatch.run_id}/${dispatch.step_id}.`); + throw new WorkerLeaseLostError('already_expired', `Agent lease is already expired for ${dispatch.run_id}/${dispatch.step_id}.`); } latestDeadline = deadline; if (expiryTimer !== undefined) clearTimeout(expiryTimer); - expiryTimer = setTimeout(() => fail(new Error( + expiryTimer = setTimeout(() => fail(new WorkerLeaseLostError('renewal_expired', `Agent lease expired before renewal for ${dispatch.run_id}/${dispatch.step_id}.`, )), remaining); return remaining; @@ -34,7 +61,7 @@ export async function withWorkerLease( // A response handled after local expiry cannot revive ownership, even // if its future deadline was issued before this event loop stalled. if (Date.now() >= latestDeadline) { - throw new Error(`Agent lease expired before renewal for ${dispatch.run_id}/${dispatch.step_id}.`); + throw new WorkerLeaseLostError('renewal_expired', `Agent lease expired before renewal for ${dispatch.run_id}/${dispatch.step_id}.`); } const remaining = armExpiry(result.lease_deadline_ms); if (!stopped) { @@ -57,7 +84,7 @@ export async function withWorkerLease( // Timer callbacks can be delayed behind a resolved subprocess promise. // Check the clock itself before permitting step.complete. if (Date.now() >= latestDeadline) { - throw new Error(`Agent lease expired before completion for ${dispatch.run_id}/${dispatch.step_id}.`); + throw new WorkerLeaseLostError('completion_expired', `Agent lease expired before completion for ${dispatch.run_id}/${dispatch.step_id}.`); } return result; } finally { diff --git a/packages/sdk/src/worker.ts b/packages/sdk/src/worker.ts index 663809880..38dc4633c 100644 --- a/packages/sdk/src/worker.ts +++ b/packages/sdk/src/worker.ts @@ -95,7 +95,7 @@ export class AgentWorker extends EventEmitter { if (this.closing) return; if (dispatch.step_type !== 'agent') return; const running: Promise = this.execute(dispatch).catch((error: unknown) => { - this.emit('error', error); + this.emit('error', error, dispatch); }); this.inFlight.add(running); void running.finally(() => { this.inFlight.delete(running); }); diff --git a/packages/sdk/tests/worker-lease-lost.test.ts b/packages/sdk/tests/worker-lease-lost.test.ts new file mode 100644 index 000000000..e146a248e --- /dev/null +++ b/packages/sdk/tests/worker-lease-lost.test.ts @@ -0,0 +1,120 @@ +import { EventEmitter } from 'node:events'; +import { afterEach, describe, expect, it, vi } from 'vitest'; +import { JournalClient, JournalProtocolError } from '../src/journal-client.js'; +import type { StepDispatchEvent } from '../src/protocol.js'; +import { AgentWorker } from '../src/worker.js'; +import { LlmWorker } from '../src/llm-worker.js'; +import { runAgentCli } from '../src/worker-cli.js'; +import { isLeaseLost, onWorkerFailure, withWorkerLease } from '../src/worker-lease.js'; + +vi.mock('../src/worker-cli.js', () => ({ runAgentCli: vi.fn() })); +afterEach(() => { vi.useRealTimers(); vi.restoreAllMocks(); vi.resetAllMocks(); }); + +function setup(type: 'agent' | 'llm') { + vi.useFakeTimers(); + const warning = vi.spyOn(process, 'emitWarning').mockImplementation(() => {}); + const client = Object.assign(new EventEmitter(), { + workerAttach: vi.fn(async () => ({})), + stepHeartbeat: vi.fn(async () => ({ lease_deadline_ms: Date.now() + 30_000 })), + stepComplete: vi.fn(async () => ({})), + close: vi.fn(), + }); + const worker = type === 'agent' + ? new AgentWorker(client as unknown as JournalClient, { workerId: 'test', pins: { workspace: [], streams: [] } }) + : new LlmWorker(client as unknown as JournalClient, 'test'); + const fatal = vi.fn((error: unknown) => client.close(error)); + worker.on('error', onWorkerFailure('test', fatal)); + const dispatch: StepDispatchEvent = { + type: 'step.dispatch', run_id: 'run', step_id: 'step', attempt: 1, + step_type: type, spec: { cli: 'claude', instruction: 'hello', prompt: 'hello' }, + lease_id: 'lease', lease_deadline_ms: Date.now() + 30_000, + idempotency_key: 'effect', pins: { workspace: [], streams: [] }, + }; + vi.mocked(runAgentCli).mockResolvedValue({ exit_code: 0, stdout_tail: 'hello', stderr_tail: '' }); + return { client, worker, fatal, warning, dispatch }; +} + +describe.each(['agent', 'llm'] as const)('%s stale lease subscriber', type => { + it('drops an expired dispatch and completes the kernel retry', async () => { + const { client, worker, fatal, warning, dispatch } = setup(type); + await worker.attach(); + client.emit('step.dispatch', { ...dispatch, lease_deadline_ms: Date.now() - 1 }); + await vi.advanceTimersByTimeAsync(0); + expect(runAgentCli).not.toHaveBeenCalled(); + expect(client.stepComplete).not.toHaveBeenCalled(); + expect(client.close).not.toHaveBeenCalled(); + expect(fatal).not.toHaveBeenCalled(); + expect(warning).toHaveBeenCalledWith(expect.stringContaining('run_id=run step_id=step attempt=1'), + { code: 'FLOWS_WORKER_LEASE_LOST' }); + client.emit('step.dispatch', { ...dispatch, attempt: 2, lease_id: 'retry' }); + await worker.close(); + expect(client.stepComplete).toHaveBeenCalledTimes(1); + expect(client.stepComplete.mock.calls[0]).toEqual(expect.arrayContaining(['run', 'step', 2, 'success'])); + }); + + it.each([ + ['stepHeartbeat', 'lease_conflict', 'step.heartbeat'], + ['stepComplete', 'lease_conflict', 'step.complete'], + ['stepComplete', 'run_terminal', 'step.complete'], + ] as const)('drops %s %s after journal success', async (method, code, verb) => { + const { client, worker, fatal, warning, dispatch } = setup(type); + const error = new JournalProtocolError(code, 'attempt has no active worker lease'); + error.verb = verb; + // The kernel already owns the completed outcome when the straggler arrives. + const journal = { completionReason: 'success' }; + client[method].mockRejectedValueOnce(error); + await worker.attach(); + client.emit('step.dispatch', dispatch); + await worker.close(); + expect(journal.completionReason).toBe('success'); + expect(client.close).not.toHaveBeenCalled(); + expect(fatal).not.toHaveBeenCalled(); + expect(warning).toHaveBeenCalledWith(expect.stringContaining(error.message), expect.anything()); + }); + + it('keeps a non-lease worker error fatal with its original identity', async () => { + const { client, worker, fatal, warning, dispatch } = setup(type); + const error = new Error('cli exploded'); + client.stepHeartbeat.mockRejectedValueOnce(error); + await worker.attach(); + client.emit('step.dispatch', dispatch); + await worker.close(); + expect(fatal).toHaveBeenCalledWith(error); + expect(client.close).toHaveBeenCalledWith(error); + expect(warning).not.toHaveBeenCalled(); + }); +}); + +it('preserves the heartbeat rejection through abort and finally', async () => { + const { client, dispatch } = setup('llm'); + const error = new JournalProtocolError('lease_conflict', 'lost'); + client.stepHeartbeat.mockResolvedValueOnce({ lease_deadline_ms: Date.now() + 30_000 }) + .mockRejectedValueOnce(error); + const result = withWorkerLease(client as unknown as JournalClient, dispatch, signal => + new Promise(resolve => signal.addEventListener('abort', () => resolve('aborted')))); + const assertion = expect(result).rejects.toBe(error); + await vi.advanceTimersByTimeAsync(10_000); + await assertion; +}); + +it.each(['step.heartbeat', 'step.complete', 'step.wait'])('tags %s refusals at the client wrapper', async verb => { + const client = new JournalClient('/unused'); + const error = new JournalProtocolError('run_terminal', 'finished'); + vi.spyOn(client, 'request').mockRejectedValueOnce(error); + const request = verb === 'step.heartbeat' ? client.stepHeartbeat('r', 's', 1, 'lease') + : verb === 'step.complete' ? client.stepComplete('r', 's', 1, 'key', 'success') + : client.stepWait('r', 's', 1, 'key', { wait_id: 'w', prompt: '?', requested_of: 'human' }); + await expect(request).rejects.toBe(error); + expect(error.verb).toBe(verb); + expect(isLeaseLost(error)).toBe(true); +}); + +it('does not classify messages, causes, or unrelated terminal refusals as lease loss', () => { + for (const error of [ + new Error('Agent lease is already expired'), + new Error('wrapped', { cause: new JournalProtocolError('lease_conflict', 'lost') }), + new JournalProtocolError('run_terminal', 'finished'), + Object.assign(new JournalProtocolError('run_terminal', 'finished'), { verb: 'run.resume' }), + new JournalProtocolError('journal_error', 'disk full'), + ]) expect(isLeaseLost(error)).toBe(false); +}); From 101394078ebf849374ca6f304876a046befffb0e Mon Sep 17 00:00:00 2001 From: Relayflow Date: Wed, 23 Sep 2026 05:34:21 +0000 Subject: [PATCH 02/13] fix(sdk): keep local workers connected after lease refusal --- packages/sdk/src/cli/direct-run.ts | 3 ++- packages/sdk/src/cli/hn-monitor.ts | 3 ++- packages/sdk/src/communication/local.ts | 3 ++- packages/sdk/src/local-agent.ts | 3 ++- 4 files changed, 8 insertions(+), 4 deletions(-) diff --git a/packages/sdk/src/cli/direct-run.ts b/packages/sdk/src/cli/direct-run.ts index 98743efd1..97bb4732d 100644 --- a/packages/sdk/src/cli/direct-run.ts +++ b/packages/sdk/src/cli/direct-run.ts @@ -1,3 +1,4 @@ +import { onWorkerFailure } from '../worker-lease.js'; import { McpStepError } from '../authored-mcp.js'; import { randomUUID } from 'node:crypto'; import { authoredLocalAgentStream } from '../authored-admission.js'; @@ -82,7 +83,7 @@ export async function runDirectFlow( await llmClient.connect(); await llmClient.hello('flows-local-llm'); localLlm = new LlmWorker(llmClient, `${localAgent.stream}-llm`, workerCapacity); - localLlm.on('error', error => { llmFailure = error; client.close(); }); + localLlm.on('error', onWorkerFailure('local-llm', error => { llmFailure = error; client.close(); })); await localLlm.attach(); } const result = await executeDurableAuthoredFlow( diff --git a/packages/sdk/src/cli/hn-monitor.ts b/packages/sdk/src/cli/hn-monitor.ts index 7f064ee97..466b8d6b2 100644 --- a/packages/sdk/src/cli/hn-monitor.ts +++ b/packages/sdk/src/cli/hn-monitor.ts @@ -1,3 +1,4 @@ +import { onWorkerFailure } from '../worker-lease.js'; /** * `flows hn-monitor start` — CLI-inlined proactive workload for gate 2. * @@ -169,7 +170,7 @@ async function defaultAttachWorker( // synchronously, bypasses the drain-aware close(), and crashes the // process. Subscribe BEFORE attach so an error during attach is not // lost. - worker.on('error', onWorkerError); + worker.on('error', onWorkerFailure('hn-monitor', onWorkerError)); await worker.attach(); return worker; } diff --git a/packages/sdk/src/communication/local.ts b/packages/sdk/src/communication/local.ts index 97b88690a..d382c3442 100644 --- a/packages/sdk/src/communication/local.ts +++ b/packages/sdk/src/communication/local.ts @@ -1,3 +1,4 @@ +import { onWorkerFailure } from '../worker-lease.js'; import { randomUUID } from 'node:crypto'; import { JournalClient } from '../journal-client.js'; import { AgentWorker } from '../worker.js'; @@ -31,7 +32,7 @@ export async function attachCommunicationWorkers(spec: KernelRunSpec, socketPath requiredStreams: [channelName(step.id, '$receipts')], pins: { workspace: [], streams: step.surfaces?.streams?.map(({ stream }) => ({ stream, read_offset: 0 })) ?? [] } }); workers.push({ client, worker }); - worker.on('error', error => { failure = error; client.close(); }); + worker.on('error', onWorkerFailure('communication', error => { failure = error; client.close(); })); await client.connect(); await client.hello('flows-communication'); await worker.attach(); diff --git a/packages/sdk/src/local-agent.ts b/packages/sdk/src/local-agent.ts index a836fecb6..33c827c52 100644 --- a/packages/sdk/src/local-agent.ts +++ b/packages/sdk/src/local-agent.ts @@ -1,3 +1,4 @@ +import { onWorkerFailure } from './worker-lease.js'; import { randomUUID } from 'node:crypto'; import type { JournalClient } from './journal-client.js'; import { AgentWorker } from './worker.js'; @@ -29,7 +30,7 @@ export async function attachLocalAgent( }); let failure: unknown; // The cause travels with the close, so the flow's next request names it. - worker.on('error', error => { failure = error; client.close(error); }); + worker.on('error', onWorkerFailure('local-agent', error => { failure = error; client.close(error); })); await worker.attach(); return { stream, From d5960c13f0bdc4cc6ef155ef0db209aba7cc8516 Mon Sep 17 00:00:00 2001 From: Relayflow Date: Wed, 23 Sep 2026 05:35:21 +0000 Subject: [PATCH 03/13] fix(sdk): handle authored resume LLM failures explicitly --- packages/sdk/src/cli/run.ts | 10 +++++- .../sdk/tests/resume-worker-lease.test.ts | 36 +++++++++++++++++++ 2 files changed, 45 insertions(+), 1 deletion(-) create mode 100644 packages/sdk/tests/resume-worker-lease.test.ts diff --git a/packages/sdk/src/cli/run.ts b/packages/sdk/src/cli/run.ts index 1f0aba7f3..3c64730e1 100644 --- a/packages/sdk/src/cli/run.ts +++ b/packages/sdk/src/cli/run.ts @@ -1,3 +1,4 @@ +import { onWorkerFailure } from '../worker-lease.js'; import { communicationInstruction } from '../communication/spec.js'; import { checkCommunicationEnvironment, CommunicationEnvironmentError } from '../communication/preflight.js'; import { parseHumanRecipient } from '../human-to.js'; @@ -206,6 +207,7 @@ export async function resumeFlow( let communicationWorkers: Awaited> | undefined; let authoredAgent: Awaited> | undefined; let authoredLlm: LlmWorker | undefined; + let llmFailure: unknown; let authoredLlmClient: JournalClient | undefined; const workerCapacity = options.agentCapacity ?? DEFAULT_LOCAL_AGENT_CAPACITY; try { @@ -225,6 +227,10 @@ export async function resumeFlow( await authoredLlmClient.connect(); await authoredLlmClient.hello('flows-authored-resume-llm'); authoredLlm = new LlmWorker(authoredLlmClient, `${authoredAgent.stream}-llm`, workerCapacity); + authoredLlm.on('error', onWorkerFailure('resume-llm', error => { + llmFailure = error; + client.close(); + })); await authoredLlm.attach(); } const result = await resumeDurableAuthoredFlow(runId, client, { @@ -283,7 +289,9 @@ export async function resumeFlow( return authoredHumanParked('resume', base, socketPath, error, { dataDir, localAgent: options.localAgent === true }); } if (!(error instanceof JournalProtocolError) || error.code !== 'run_not_found') { - return protocolFailure('resume', base, socketPath, error, runId); + const cause = error instanceof AuthoredFlowExecutionError + ? error : llmFailure ?? authoredAgent?.failure ?? error; + return protocolFailure('resume', base, socketPath, cause, runId); } return { exitCode: 2, diff --git a/packages/sdk/tests/resume-worker-lease.test.ts b/packages/sdk/tests/resume-worker-lease.test.ts new file mode 100644 index 000000000..dfb595070 --- /dev/null +++ b/packages/sdk/tests/resume-worker-lease.test.ts @@ -0,0 +1,36 @@ +import { afterEach, expect, it, vi } from 'vitest'; +import { JournalClient, JournalProtocolError } from '../src/journal-client.js'; +import { LlmWorker } from '../src/llm-worker.js'; +import { resumeFlow } from '../src/cli/run.js'; +import { readAuthoredRootMetadata, resumeDurableAuthoredFlow } from '../src/authored-root.js'; + +vi.mock('../src/daemon-lifecycle.js', () => ({ + ensureDaemon: async () => ({ kind: 'attached', socketPath: '/unused', connection: null }), +})); +vi.mock('../src/authored-root.js', () => ({ + readAuthoredRootMetadata: vi.fn(async () => ({ localAgentStream: 'stream' })), + resumeDurableAuthoredFlow: vi.fn(async () => { throw new Error('connection closed'); }), +})); +vi.mock('../src/local-agent.js', () => ({ + attachLocalAgent: async () => ({ stream: 'stream', close: async () => {} }), +})); +afterEach(() => { vi.restoreAllMocks(); vi.clearAllMocks(); }); + +it.each([false, true])('resume handles LLM errors with leaseLost=%s', async leaseLost => { + vi.spyOn(JournalClient.prototype, 'connect').mockResolvedValue(); + vi.spyOn(JournalClient.prototype, 'hello').mockResolvedValue({} as never); + const close = vi.spyOn(JournalClient.prototype, 'close').mockReturnValue(); + const warning = vi.spyOn(process, 'emitWarning').mockImplementation(() => {}); + const error = leaseLost ? new JournalProtocolError('lease_conflict', 'lost') : new Error('cli exploded'); + vi.spyOn(LlmWorker.prototype, 'attach').mockImplementation(async function () { + expect(this.listenerCount('error')).toBe(1); + this.emit('error', error, { run_id: 'run', step_id: 'llm', attempt: 1 }); + expect(close).toHaveBeenCalledTimes(leaseLost ? 0 : 1); + }); + const result = await resumeFlow('run', '/unused', { localAgent: true }); + expect(readAuthoredRootMetadata).toHaveBeenCalled(); + expect(resumeDurableAuthoredFlow).toHaveBeenCalled(); + expect(result.exitCode).toBe(1); + expect(JSON.stringify(result.report)).toContain(leaseLost ? 'connection closed' : 'cli exploded'); + expect(warning).toHaveBeenCalledTimes(leaseLost ? 1 : 0); +}); From 5ce971fd898c090e73e6df2eefc343f6fe48a86f Mon Sep 17 00:00:00 2001 From: Relayflow Date: Wed, 23 Sep 2026 05:37:34 +0000 Subject: [PATCH 04/13] test(sdk): preserve journal success after a late completion refusal --- .../sdk/tests/worker-lease-lost-live.test.ts | 49 +++++++++++++++++++ 1 file changed, 49 insertions(+) create mode 100644 packages/sdk/tests/worker-lease-lost-live.test.ts diff --git a/packages/sdk/tests/worker-lease-lost-live.test.ts b/packages/sdk/tests/worker-lease-lost-live.test.ts new file mode 100644 index 000000000..e760b6d05 --- /dev/null +++ b/packages/sdk/tests/worker-lease-lost-live.test.ts @@ -0,0 +1,49 @@ +import { afterEach, expect, it, vi } from 'vitest'; +import { flow } from '@relayflows/surface'; +import { executeAuthoredFlow } from '../src/authored-flow-executor.js'; +import { JournalProtocolError } from '../src/journal-client.js'; +import { LlmWorker } from '../src/llm-worker.js'; +import { onWorkerFailure } from '../src/worker-lease.js'; +import { chainFixture } from './flow-chain-fixture.js'; + +afterEach(() => vi.restoreAllMocks()); + +it.each(['lease_conflict', 'run_terminal'])('reports journal success after completion rejects with %s', async code => { + const fixture = chainFixture(); + const client = await fixture.connect(); + const worker = new LlmWorker(client, 'lease-race'); + const fatal = vi.fn((error: unknown) => client.close(error)); + const warning = vi.spyOn(process, 'emitWarning').mockImplementation(() => {}); + worker.on('error', onWorkerFailure('race', fatal)); + const complete = client.stepComplete.bind(client); + let completedRun: string | undefined; + vi.spyOn(client, 'stepComplete').mockImplementation(async (...args) => { + await complete(...args); + completedRun = args[0]; + const entries = (await client.journalRead(args[0], 1)).entries; + for (const entry_type of ['step.completed', 'run.completed']) { + expect(entries).toEqual(expect.arrayContaining([expect.objectContaining({ + entry_type, payload: expect.objectContaining({ completionReason: 'success' }), + })])); + } + const error = new JournalProtocolError(code, 'attempt has no active worker lease'); + error.verb = 'step.complete'; + throw error; + }); + try { + await worker.attach(); + const handle = flow('lease-race', async f => { + await f.llm('hello', { model: 'test-model', output: { type: 'object' } }); + f.done('success'); + }); + const result = await executeAuthoredFlow(handle, client, undefined, { flowPath: fixture.flowPath }); + await worker.close(); + expect(completedRun).toBeDefined(); + expect(result.completionReason).toBe('success'); + expect(fatal).not.toHaveBeenCalled(); + expect(warning).toHaveBeenCalledTimes(1); + } finally { + await worker.close(); + await fixture.close(); + } +}, 30_000); From 11d8b1c9a82a0d6787afd50a64d14119ee9b537f Mon Sep 17 00:00:00 2001 From: Relayflow Date: Wed, 23 Sep 2026 05:37:34 +0000 Subject: [PATCH 05/13] fix(sdk): allow kernel lease sweep grace before failing a running step --- packages/sdk/src/cli/run.ts | 5 ++- packages/sdk/tests/worker-lease-sweep.test.ts | 43 +++++++++++++++++++ 2 files changed, 47 insertions(+), 1 deletion(-) create mode 100644 packages/sdk/tests/worker-lease-sweep.test.ts diff --git a/packages/sdk/src/cli/run.ts b/packages/sdk/src/cli/run.ts index 3c64730e1..f71cb8183 100644 --- a/packages/sdk/src/cli/run.ts +++ b/packages/sdk/src/cli/run.ts @@ -786,6 +786,9 @@ async function inspectOutOfBandStep( }; } +// Match kernel/relayflowd/src/server/client.rs: allow the lease sweep to dispatch a retry. +const LEASE_SWEEP_GRACE_MS = 5_000; + async function waitForRunningStep( client: JournalClient, runId: string, @@ -804,7 +807,7 @@ async function waitForRunningStep( }); while (true) { throwIfCanceled(options.signal, runningStep.id); - const remainingMs = leaseDeadlineMs - Date.now(); + const remainingMs = leaseDeadlineMs + LEASE_SWEEP_GRACE_MS - Date.now(); if (remainingMs <= 0) { throw new Error( `worker lease for step "${runningStep.id}" expired at ${leaseDeadlineMs} without completion`, diff --git a/packages/sdk/tests/worker-lease-sweep.test.ts b/packages/sdk/tests/worker-lease-sweep.test.ts new file mode 100644 index 000000000..d169a3de3 --- /dev/null +++ b/packages/sdk/tests/worker-lease-sweep.test.ts @@ -0,0 +1,43 @@ +import { afterEach, expect, it, vi } from 'vitest'; +import { classifyOutcome, emptyReport } from '../src/cli/run.js'; +import type { JournalClient } from '../src/journal-client.js'; + +afterEach(() => vi.useRealTimers()); + +function fixture() { + vi.useFakeTimers(); + const deadline = Date.now() - 1; + const snapshot = (state: string, lease = deadline) => ({ + run_id: 'run', status: 'parked', + steps: { answer: { type: 'llm', state, lease_deadline_ms: lease } }, + budget: { tokens_in: 0, tokens_out: 0, dollars: '0' }, + }); + const parked = { run_id: 'run', status: 'parked' as const, completion_reason: null, completed_steps: 0 }; + const client = { + runGet: vi.fn().mockResolvedValue(snapshot('running')), + runResume: vi.fn().mockResolvedValue(parked), + }; + const run = () => classifyOutcome(client as unknown as JournalClient, 'run', parked, emptyReport('run'), '/unused', {}); + return { client, snapshot, run }; +} + +it('waits through an expired snapshot until the kernel retries and completes', async () => { + const { client, snapshot, run } = fixture(); + client.runGet.mockResolvedValueOnce(snapshot('running')) + .mockResolvedValueOnce(snapshot('runnable')) + .mockResolvedValueOnce(snapshot('running', Date.now() + 30_000)) + .mockResolvedValue(snapshot('completed')); + client.runResume.mockResolvedValueOnce({ run_id: 'run', status: 'parked', completion_reason: null, completed_steps: 0 }) + .mockResolvedValue({ run_id: 'run', status: 'completed', completion_reason: 'success', completed_steps: 1 }); + const execution = run(); + const assertion = expect(execution).resolves.toMatchObject({ exitCode: 0, report: { completionReason: 'success' } }); + await Promise.all([assertion, vi.advanceTimersByTimeAsync(100)]); + expect(client.runResume).toHaveBeenCalledTimes(2); +}); + +it('still fails when the kernel never resolves an expired lease after sweep grace', async () => { + const { run } = fixture(); + const assertion = expect(run()).rejects.toThrow('without completion'); + await vi.advanceTimersByTimeAsync(5_001); + await assertion; +}); From d82e97d034eba01dcb8a598f9369e9dc7ead428e Mon Sep 17 00:00:00 2001 From: Relayflow Date: Wed, 23 Sep 2026 05:37:34 +0000 Subject: [PATCH 06/13] test(sdk): typecheck lease regression fixtures --- packages/sdk/tests/resume-worker-lease.test.ts | 2 +- packages/sdk/tests/worker-lease-lost.test.ts | 9 +++------ 2 files changed, 4 insertions(+), 7 deletions(-) diff --git a/packages/sdk/tests/resume-worker-lease.test.ts b/packages/sdk/tests/resume-worker-lease.test.ts index dfb595070..91a5b304c 100644 --- a/packages/sdk/tests/resume-worker-lease.test.ts +++ b/packages/sdk/tests/resume-worker-lease.test.ts @@ -22,7 +22,7 @@ it.each([false, true])('resume handles LLM errors with leaseLost=%s', async leas const close = vi.spyOn(JournalClient.prototype, 'close').mockReturnValue(); const warning = vi.spyOn(process, 'emitWarning').mockImplementation(() => {}); const error = leaseLost ? new JournalProtocolError('lease_conflict', 'lost') : new Error('cli exploded'); - vi.spyOn(LlmWorker.prototype, 'attach').mockImplementation(async function () { + vi.spyOn(LlmWorker.prototype, 'attach').mockImplementation(async function (this: LlmWorker) { expect(this.listenerCount('error')).toBe(1); this.emit('error', error, { run_id: 'run', step_id: 'llm', attempt: 1 }); expect(close).toHaveBeenCalledTimes(leaseLost ? 0 : 1); diff --git a/packages/sdk/tests/worker-lease-lost.test.ts b/packages/sdk/tests/worker-lease-lost.test.ts index e146a248e..0daf3e5e1 100644 --- a/packages/sdk/tests/worker-lease-lost.test.ts +++ b/packages/sdk/tests/worker-lease-lost.test.ts @@ -25,7 +25,7 @@ function setup(type: 'agent' | 'llm') { const fatal = vi.fn((error: unknown) => client.close(error)); worker.on('error', onWorkerFailure('test', fatal)); const dispatch: StepDispatchEvent = { - type: 'step.dispatch', run_id: 'run', step_id: 'step', attempt: 1, + run_id: 'run', step_id: 'step', attempt: 1, step_type: type, spec: { cli: 'claude', instruction: 'hello', prompt: 'hello' }, lease_id: 'lease', lease_deadline_ms: Date.now() + 30_000, idempotency_key: 'effect', pins: { workspace: [], streams: [] }, @@ -56,17 +56,14 @@ describe.each(['agent', 'llm'] as const)('%s stale lease subscriber', type => { ['stepHeartbeat', 'lease_conflict', 'step.heartbeat'], ['stepComplete', 'lease_conflict', 'step.complete'], ['stepComplete', 'run_terminal', 'step.complete'], - ] as const)('drops %s %s after journal success', async (method, code, verb) => { + ] as const)('drops %s %s for a released attempt', async (method, code, verb) => { const { client, worker, fatal, warning, dispatch } = setup(type); const error = new JournalProtocolError(code, 'attempt has no active worker lease'); error.verb = verb; - // The kernel already owns the completed outcome when the straggler arrives. - const journal = { completionReason: 'success' }; client[method].mockRejectedValueOnce(error); await worker.attach(); client.emit('step.dispatch', dispatch); await worker.close(); - expect(journal.completionReason).toBe('success'); expect(client.close).not.toHaveBeenCalled(); expect(fatal).not.toHaveBeenCalled(); expect(warning).toHaveBeenCalledWith(expect.stringContaining(error.message), expect.anything()); @@ -100,7 +97,7 @@ it('preserves the heartbeat rejection through abort and finally', async () => { it.each(['step.heartbeat', 'step.complete', 'step.wait'])('tags %s refusals at the client wrapper', async verb => { const client = new JournalClient('/unused'); const error = new JournalProtocolError('run_terminal', 'finished'); - vi.spyOn(client, 'request').mockRejectedValueOnce(error); + vi.spyOn(client as unknown as { request: () => Promise }, 'request').mockRejectedValueOnce(error); const request = verb === 'step.heartbeat' ? client.stepHeartbeat('r', 's', 1, 'lease') : verb === 'step.complete' ? client.stepComplete('r', 's', 1, 'key', 'success') : client.stepWait('r', 's', 1, 'key', { wait_id: 'w', prompt: '?', requested_of: 'human' }); From 2d31cb14c25569f8ea7b3aaefcd58a7f9b200c01 Mon Sep 17 00:00:00 2001 From: Relayflow Date: Wed, 23 Sep 2026 05:38:29 +0000 Subject: [PATCH 07/13] test(sdk): pin direct-run fatal policy and malformed lease rejection --- packages/sdk/src/worker-lease.ts | 6 ++-- .../sdk/tests/direct-run-worker-lease.test.ts | 34 +++++++++++++++++++ packages/sdk/tests/worker-lease-lost.test.ts | 9 +++++ 3 files changed, 47 insertions(+), 2 deletions(-) create mode 100644 packages/sdk/tests/direct-run-worker-lease.test.ts diff --git a/packages/sdk/src/worker-lease.ts b/packages/sdk/src/worker-lease.ts index 9218e8982..184850c7c 100644 --- a/packages/sdk/src/worker-lease.ts +++ b/packages/sdk/src/worker-lease.ts @@ -1,7 +1,6 @@ import { JournalProtocolError, type JournalClient } from './journal-client.js'; import type { StepDispatchEvent } from './protocol.js'; - export class WorkerLeaseLostError extends Error { constructor(readonly reason: 'already_expired' | 'renewal_expired' | 'completion_expired', message: string) { super(message); @@ -43,7 +42,10 @@ export async function withWorkerLease( const fail = (error: unknown): void => { controller.abort(error); }; const armExpiry = (deadline: number): number => { const remaining = deadline - Date.now(); - if (!Number.isFinite(remaining) || remaining <= 0) { + if (!Number.isFinite(remaining)) { + throw new Error(`Agent lease is already expired for ${dispatch.run_id}/${dispatch.step_id}.`); + } + if (remaining <= 0) { throw new WorkerLeaseLostError('already_expired', `Agent lease is already expired for ${dispatch.run_id}/${dispatch.step_id}.`); } latestDeadline = deadline; diff --git a/packages/sdk/tests/direct-run-worker-lease.test.ts b/packages/sdk/tests/direct-run-worker-lease.test.ts new file mode 100644 index 000000000..2b1bbe763 --- /dev/null +++ b/packages/sdk/tests/direct-run-worker-lease.test.ts @@ -0,0 +1,34 @@ +import { afterEach, expect, it, vi } from 'vitest'; +import { runDirectFlow } from '../src/cli/direct-run.js'; +import { JournalClient, JournalProtocolError } from '../src/journal-client.js'; +import { LlmWorker } from '../src/llm-worker.js'; +import { executeDurableAuthoredFlow } from '../src/authored-root.js'; + +vi.mock('../src/cli/run.js', async importOriginal => ({ + ...await importOriginal(), connect: async () => undefined, +})); +vi.mock('../src/authored-flow-loader.js', async importOriginal => ({ + ...await importOriginal(), + loadAuthoredFlow: async () => ({ handle: {}, getDefinition: () => ({}) }), +})); +vi.mock('../src/authored-root.js', () => ({ executeDurableAuthoredFlow: vi.fn() })); +vi.mock('../src/local-agent.js', () => ({ attachLocalAgent: async () => ({ + stream: 'test', close: async () => {}, +}) })); +afterEach(() => { vi.restoreAllMocks(); vi.resetAllMocks(); }); + +it.each([false, true])('direct run handles LLM errors with leaseLost=%s', async leaseLost => { + vi.spyOn(JournalClient.prototype, 'connect').mockResolvedValue(); + vi.spyOn(JournalClient.prototype, 'hello').mockResolvedValue({} as never); + const close = vi.spyOn(JournalClient.prototype, 'close').mockReturnValue(); + vi.spyOn(process, 'emitWarning').mockImplementation(() => {}); + const error = leaseLost ? new JournalProtocolError('lease_conflict', 'lost') : new Error('cli exploded'); + vi.spyOn(LlmWorker.prototype, 'attach').mockImplementation(async function (this: LlmWorker) { + this.emit('error', error, { run_id: 'run', step_id: 'llm', attempt: 1 }); + expect(close).toHaveBeenCalledTimes(leaseLost ? 0 : 1); + }); + vi.mocked(executeDurableAuthoredFlow).mockRejectedValueOnce(new Error('connection closed')); + const result = await runDirectFlow('flow.ts', '{}', '/unused', { localAgent: true }); + expect(result.exitCode).toBe(1); + expect(JSON.stringify(result.report)).toContain(leaseLost ? 'connection closed' : 'cli exploded'); +}); diff --git a/packages/sdk/tests/worker-lease-lost.test.ts b/packages/sdk/tests/worker-lease-lost.test.ts index 0daf3e5e1..e023c098f 100644 --- a/packages/sdk/tests/worker-lease-lost.test.ts +++ b/packages/sdk/tests/worker-lease-lost.test.ts @@ -115,3 +115,12 @@ it('does not classify messages, causes, or unrelated terminal refusals as lease new JournalProtocolError('journal_error', 'disk full'), ]) expect(isLeaseLost(error)).toBe(false); }); + +it('keeps an invalid lease deadline fatal', async () => { + const { client, worker, fatal, dispatch } = setup('llm'); + await worker.attach(); + client.emit('step.dispatch', { ...dispatch, lease_deadline_ms: NaN }); + await worker.close(); + expect(fatal).toHaveBeenCalledTimes(1); + expect(client.close).toHaveBeenCalledTimes(1); +}); From 99c29ddfc523262d4613cb1d0aaae9dfa6fbf48c Mon Sep 17 00:00:00 2001 From: Relayflow Date: Wed, 23 Sep 2026 05:39:46 +0000 Subject: [PATCH 08/13] test(sdk): keep CLI expiry bound beyond kernel sweep grace --- packages/sdk/tests/cli.test.ts | 4 ++-- 1 file changed, 2 insertions(+), 2 deletions(-) diff --git a/packages/sdk/tests/cli.test.ts b/packages/sdk/tests/cli.test.ts index 9c36f10f2..ac82ace71 100644 --- a/packages/sdk/tests/cli.test.ts +++ b/packages/sdk/tests/cli.test.ts @@ -1048,9 +1048,9 @@ describe('flows run/resume CLI over the journal protocol', () => { expect(output.stderr.join('\n')).not.toContain('protocol_error'); }); - it('bounds a worker wait by its lease and reports what it is waiting for', async () => { + it('bounds a worker wait by its lease plus sweep grace and reports what it is waiting for', async () => { const dataDir = temporaryProject('flows-run-lease-'); - const leaseDeadlineMs = Date.now() + 150; + const leaseDeadlineMs = Date.now() - 5_001; let snapshots = 0; await startCliLoopback(dataDir, { hello: sendOk, From 29eaa4bb65a2fa806e8d5ef4690f91b3ca493eea Mon Sep 17 00:00:00 2001 From: Relayflow Date: Wed, 23 Sep 2026 05:42:19 +0000 Subject: [PATCH 09/13] test(sdk): cover heartbeat refusal after journaled success --- .../sdk/tests/worker-lease-lost-live.test.ts | 52 +++++++++++++++++++ 1 file changed, 52 insertions(+) diff --git a/packages/sdk/tests/worker-lease-lost-live.test.ts b/packages/sdk/tests/worker-lease-lost-live.test.ts index e760b6d05..ffec9336b 100644 --- a/packages/sdk/tests/worker-lease-lost-live.test.ts +++ b/packages/sdk/tests/worker-lease-lost-live.test.ts @@ -1,4 +1,6 @@ import { afterEach, expect, it, vi } from 'vitest'; +import { readFileSync, writeFileSync } from 'node:fs'; +import type { StepDispatchEvent } from '../src/protocol.js'; import { flow } from '@relayflows/surface'; import { executeAuthoredFlow } from '../src/authored-flow-executor.js'; import { JournalProtocolError } from '../src/journal-client.js'; @@ -47,3 +49,53 @@ it.each(['lease_conflict', 'run_terminal'])('reports journal success after compl await fixture.close(); } }, 30_000); + + +it('reports journal success when a renewal rejects after completion landed', async () => { + const fixture = chainFixture(); + const wrapper = readFileSync(fixture.wrapper, 'utf8'); + writeFileSync(fixture.wrapper, wrapper.replace(' process.stdout.write', + ' await new Promise(resolve => setTimeout(resolve, 2000));\n process.stdout.write')); + const client = await fixture.connect(); + const worker = new LlmWorker(client, 'heartbeat-race'); + const fatal = vi.fn((error: unknown) => client.close(error)); + const warning = vi.spyOn(process, 'emitWarning').mockImplementation(() => {}); + worker.on('error', onWorkerFailure('heartbeat-race', fatal)); + let dispatch: StepDispatchEvent; + client.on('step.dispatch', event => { dispatch = event; }); + const heartbeat = client.stepHeartbeat.bind(client); + const refusal = new JournalProtocolError('lease_conflict', 'attempt has no active worker lease'); + refusal.verb = 'step.heartbeat'; + vi.spyOn(client, 'stepHeartbeat') + .mockImplementationOnce(async (...args) => { + await heartbeat(...args); + return { lease_deadline_ms: Date.now() + 300 }; + }) + .mockImplementationOnce(async () => { + await client.stepComplete(dispatch.run_id, dispatch.step_id, dispatch.attempt, + dispatch.idempotency_key, 'success', { output: { message: 'already completed' } }); + const entries = (await client.journalRead(dispatch.run_id, 1)).entries; + for (const entry_type of ['step.completed', 'run.completed']) { + expect(entries).toEqual(expect.arrayContaining([expect.objectContaining({ + entry_type, payload: expect.objectContaining({ completionReason: 'success' }), + })])); + } + throw refusal; + }); + try { + await worker.attach(); + const handle = flow('heartbeat-race', async f => { + await f.llm('hello', { model: 'test-model', output: { type: 'object' } }); + f.done('success'); + }); + const result = await executeAuthoredFlow(handle, client, undefined, { flowPath: fixture.flowPath }); + await worker.close(); + expect(result.completionReason).toBe('success'); + expect(client.stepHeartbeat).toHaveBeenCalledTimes(2); + expect(fatal).not.toHaveBeenCalled(); + expect(warning).toHaveBeenCalledWith(expect.stringContaining(refusal.message), expect.anything()); + } finally { + await worker.close(); + await fixture.close(); + } +}, 30_000); From 7b021bae4e0ea58d72250b94381abfba61cc726f Mon Sep 17 00:00:00 2001 From: Relayflow Date: Wed, 23 Sep 2026 05:49:20 +0000 Subject: [PATCH 10/13] docs: capture lease regression mutations and full-suite limitations --- evidence/worker-lease-lost/README.md | 69 + evidence/worker-lease-lost/baseline-build.txt | 7 + .../worker-lease-lost/baseline-comparison.txt | 13 + .../baseline-live-kernel.txt | 200 +++ evidence/worker-lease-lost/baseline.py | 21 + evidence/worker-lease-lost/direct-mutant.txt | 33 + .../worker-lease-lost/direct-restored.txt | 13 + evidence/worker-lease-lost/fatal-mutant.txt | 53 + evidence/worker-lease-lost/fatal-restored.txt | 13 + evidence/worker-lease-lost/filter-mutant.txt | 204 +++ .../worker-lease-lost/filter-restored.txt | 13 + evidence/worker-lease-lost/live-first.txt | 14 + evidence/worker-lease-lost/live-mutant.txt | 92 ++ evidence/worker-lease-lost/live-restored.txt | 15 + evidence/worker-lease-lost/mutate.py | 52 + evidence/worker-lease-lost/mutations.txt | 49 + evidence/worker-lease-lost/new-test-types.txt | 3 + evidence/worker-lease-lost/npm-test-final.txt | 997 ++++++++++++ .../worker-lease-lost/npm-test-initial.txt | 1370 +++++++++++++++++ evidence/worker-lease-lost/restored-build.txt | 7 + evidence/worker-lease-lost/resume-mutant.txt | 32 + .../worker-lease-lost/resume-restored.txt | 13 + evidence/worker-lease-lost/sandbox-probe.txt | 4 + evidence/worker-lease-lost/sweep-before.txt | 70 + evidence/worker-lease-lost/sweep-mutant.txt | 34 + evidence/worker-lease-lost/sweep-restored.txt | 13 + .../worker-lease-lost/terminal-mutant.txt | 68 + .../worker-lease-lost/terminal-restored.txt | 13 + .../worker-lease-lost/tsconfig.tests.json | 17 + 29 files changed, 3502 insertions(+) create mode 100644 evidence/worker-lease-lost/README.md create mode 100644 evidence/worker-lease-lost/baseline-build.txt create mode 100644 evidence/worker-lease-lost/baseline-comparison.txt create mode 100644 evidence/worker-lease-lost/baseline-live-kernel.txt create mode 100644 evidence/worker-lease-lost/baseline.py create mode 100644 evidence/worker-lease-lost/direct-mutant.txt create mode 100644 evidence/worker-lease-lost/direct-restored.txt create mode 100644 evidence/worker-lease-lost/fatal-mutant.txt create mode 100644 evidence/worker-lease-lost/fatal-restored.txt create mode 100644 evidence/worker-lease-lost/filter-mutant.txt create mode 100644 evidence/worker-lease-lost/filter-restored.txt create mode 100644 evidence/worker-lease-lost/live-first.txt create mode 100644 evidence/worker-lease-lost/live-mutant.txt create mode 100644 evidence/worker-lease-lost/live-restored.txt create mode 100644 evidence/worker-lease-lost/mutate.py create mode 100644 evidence/worker-lease-lost/mutations.txt create mode 100644 evidence/worker-lease-lost/new-test-types.txt create mode 100644 evidence/worker-lease-lost/npm-test-final.txt create mode 100644 evidence/worker-lease-lost/npm-test-initial.txt create mode 100644 evidence/worker-lease-lost/restored-build.txt create mode 100644 evidence/worker-lease-lost/resume-mutant.txt create mode 100644 evidence/worker-lease-lost/resume-restored.txt create mode 100644 evidence/worker-lease-lost/sandbox-probe.txt create mode 100644 evidence/worker-lease-lost/sweep-before.txt create mode 100644 evidence/worker-lease-lost/sweep-mutant.txt create mode 100644 evidence/worker-lease-lost/sweep-restored.txt create mode 100644 evidence/worker-lease-lost/terminal-mutant.txt create mode 100644 evidence/worker-lease-lost/terminal-restored.txt create mode 100644 evidence/worker-lease-lost/tsconfig.tests.json diff --git a/evidence/worker-lease-lost/README.md b/evidence/worker-lease-lost/README.md new file mode 100644 index 000000000..e4a917660 --- /dev/null +++ b/evidence/worker-lease-lost/README.md @@ -0,0 +1,69 @@ +# Worker lease refusal evidence + +All commands run from the repository root unless the transcript starts with +`cd packages/sdk`. Each raw transcript includes the command and captured output. + +The SDK tests inject stale dispatches and refusals; they do not claim to reproduce +the separate concurrent-dispatch root cause or to verify kernel lease expiry. +The live tests use the repository-built daemon and assert actual `step.completed` +and `run.completed` success entries before injecting completion/heartbeat refusals. + +## Mutation checks + +Run `python3 evidence/worker-lease-lost/mutate.py` from the repository root. +The script saves each source file as bytes, applies the exact change in +[mutations.txt](mutations.txt), captures failure, restores those bytes in a +`finally` block, checks byte equality, and captures the passing rerun. +The ledger includes SHA256 values for the restored source. + +The fatal-policy check deliberately removes the retained fatal callback. It is +a negative-control mutation, not a claim that fatal behavior was newly added. +The terminal-only check removes terminal-refusal support separately from the +whole-filter reversion. + +| Check | Command (in packages/sdk) | Mutated output | Restored output | +|---|---|---|---| +| filter | `npx vitest run tests/worker-lease-lost.test.ts` | [filter-mutant.txt](filter-mutant.txt) | [filter-restored.txt](filter-restored.txt) | +| terminal | `npx vitest run tests/worker-lease-lost.test.ts -t run_terminal` | [terminal-mutant.txt](terminal-mutant.txt) | [terminal-restored.txt](terminal-restored.txt) | +| fatal | `npx vitest run tests/worker-lease-lost.test.ts -t 'non-lease worker error'` | [fatal-mutant.txt](fatal-mutant.txt) | [fatal-restored.txt](fatal-restored.txt) | +| live | `npx vitest run tests/worker-lease-lost-live.test.ts` | [live-mutant.txt](live-mutant.txt) | [live-restored.txt](live-restored.txt) | +| sweep | `npx vitest run tests/worker-lease-sweep.test.ts` | [sweep-mutant.txt](sweep-mutant.txt) | [sweep-restored.txt](sweep-restored.txt) | +| direct | `npx vitest run tests/direct-run-worker-lease.test.ts` | [direct-mutant.txt](direct-mutant.txt) | [direct-restored.txt](direct-restored.txt) | +| resume | `npx vitest run tests/resume-worker-lease.test.ts` | [resume-mutant.txt](resume-mutant.txt) | [resume-restored.txt](resume-restored.txt) | + +## Full checks + +- [Final npm test](npm-test-final.txt): final source, Rust on PATH, explicit + RELAYFLOWD_BIN pointing at this checkout's build. Full captured output. +- [New test typecheck](new-test-types.txt): `npx tsc -p ../../evidence/worker-lease-lost/tsconfig.tests.json` + from packages/sdk. The supplemental config includes every new regression file; + the repository's existing test typecheck only enumerates selected files. +- [Sandbox probe](sandbox-probe.txt): direct OS isolation probe. + +## Development transcripts + +- [Initial npm test](npm-test-initial.txt): superseded development run, without + RELAYFLOWD_BIN. This was started before the final test/source corrections; + it includes interim CLI deadline and malformed-deadline failures. + Use the final run for review. +- [Sweep before the fix](sweep-before.txt): reproduces immediate executor failure + on an expired running snapshot. +- [First live completion check](live-first.txt): the two completion-refusal + variants before adding the real-kernel heartbeat case. The restored live + transcript above includes all three. + +## Baseline comparison for live-kernel failures + +`python3 evidence/worker-lease-lost/baseline.py` replaces only changed SDK source +files with their original bytes from `f6ece41`, rebuilds, and runs the unchanged +live-kernel suite. It restores the implementation byte-for-byte in `finally`, +asserts equality, and rebuilds it. The script uses the same PATH and +RELAYFLOWD_BIN as the final full suite. + +- [Baseline build](baseline-build.txt) +- [Baseline live-kernel command and full output](baseline-live-kernel.txt) +- [Matching failed test names](baseline-comparison.txt) +- [Restored implementation build](restored-build.txt) + +The same eight live-kernel tests failed on the original SDK source. +This comparison is limited to that suite; it is not a baseline full-suite run. diff --git a/evidence/worker-lease-lost/baseline-build.txt b/evidence/worker-lease-lost/baseline-build.txt new file mode 100644 index 000000000..585c5975e --- /dev/null +++ b/evidence/worker-lease-lost/baseline-build.txt @@ -0,0 +1,7 @@ +$ cd packages/sdk && npm run build + +> @relayflows/sdk@2.0.29 build +> tsc && node scripts/make-cli-executable.mjs + + +Exit code: 0 diff --git a/evidence/worker-lease-lost/baseline-comparison.txt b/evidence/worker-lease-lost/baseline-comparison.txt new file mode 100644 index 000000000..11c00f691 --- /dev/null +++ b/evidence/worker-lease-lost/baseline-comparison.txt @@ -0,0 +1,13 @@ +Compared literal FAIL test names in npm-test-final.txt and baseline-live-kernel.txt. +Baseline: original SDK source at f6ece41, rebuilt before running the unchanged live-kernel suite. +Environment: PATH=/home/daytona/.cargo/bin:$PATH +RELAYFLOWD_BIN=/home/daytona/.relayflows-toolchain/target/2962130851/debug/relayflowd +Both runs failed the same eight test names: + FAIL tests/live-kernel.test.ts > built flows CLI against live relayflowd > runs hn-monitor analyze-story end-to-end via a stub agent CLI (gate 2 clause 2 demo) + FAIL tests/live-kernel.test.ts > built flows CLI against live relayflowd > hn-monitor analyze-story FAILS verification when the CLI omits required schema fields + FAIL tests/live-kernel.test.ts > built flows CLI against live relayflowd > agent step preserves the CliResult wrapper as output when the CLI emits non-JSON text + FAIL tests/live-kernel.test.ts > built flows CLI against live relayflowd > AgentWorker exposes wake_context to the CLI via RELAYFLOW_WAKE_CONTEXT env var (real analyzer prerequisite) + FAIL tests/live-kernel.test.ts > built flows CLI against live relayflowd > AgentWorker leaves RELAYFLOW_WAKE_CONTEXT UNSET when the run has no wake_context (undefined-vs-null pin) + FAIL tests/live-kernel.test.ts > built flows CLI against live relayflowd > AgentWorker leaves RELAYFLOW_MODEL UNSET when the step declares no model + FAIL tests/live-kernel.test.ts > built flows CLI against live relayflowd > hn-monitor analyze-story reaches done through the real Claude analyzer CLI + FAIL tests/live-kernel.test.ts > a relayflow can be scheduled: tick source against live relayflowd > a tick spawns a real run whose step reports the SCHEDULED instant diff --git a/evidence/worker-lease-lost/baseline-live-kernel.txt b/evidence/worker-lease-lost/baseline-live-kernel.txt new file mode 100644 index 000000000..bff460020 --- /dev/null +++ b/evidence/worker-lease-lost/baseline-live-kernel.txt @@ -0,0 +1,200 @@ +$ cd packages/sdk && npx vitest run tests/live-kernel.test.ts + + RUN v2.1.9 /home/daytona/.relayflow-v2-supervisor/durable/repository/packages/sdk + +stdout | tests/live-kernel.test.ts +LIVE_KERNEL relayflowd=/home/daytona/.relayflows-toolchain/target/2962130851/debug/relayflowd +LIVE_KERNEL flows=/home/daytona/.relayflow-v2-supervisor/durable/repository/packages/sdk/dist/cli.js + +stdout | tests/live-kernel.test.ts > surface resume after a real daemon kill > resumes a three-step run with each successful completion exactly once +LIVE_KERNEL kill -9 pid=48843 run=01M36CYFYEE1F30A8SJTMT2RJZ while step=two state=Running + + ❯ tests/live-kernel.test.ts (31 tests | 8 failed) 53521ms + ✓ built flows CLI against live relayflowd > twenty-six-step reuses 25 durable completions after editing the failed final step 2121ms + ✓ built flows CLI against live relayflowd > runs rung (a), parks rung (b), and keeps JSON report-shaped 2581ms + ✓ built flows CLI against live relayflowd > allows a deterministic run to exceed the bounded request timeout 32453ms + ✓ built flows CLI against live relayflowd > follows a live worker dispatch through flows run 624ms + ✓ built flows CLI against live relayflowd > runs an agent CLI end to end through the SDK worker 564ms + ✓ built flows CLI against live relayflowd > f.agent lowers to a real agent step and dispatches through a live worker 632ms + ✓ built flows CLI against live relayflowd > f.agent's default flowPath anchors on cwd, not cwd's parent 343ms + ✓ built flows CLI against live relayflowd > can always get a parked run to a late-attaching worker 5942ms + ✓ built flows CLI against live relayflowd > reports a real manual-recovery NeedsHuman state as parked 428ms + × built flows CLI against live relayflowd > runs hn-monitor analyze-story end-to-end via a stub agent CLI (gate 2 clause 2 demo) 433ms + → expected { …(12) } to match object { output: { …(3) }, …(1) } +(22 matching properties omitted from actual) + × built flows CLI against live relayflowd > hn-monitor analyze-story FAILS verification when the CLI omits required schema fields 566ms + → expected { …(12) } to match object { …(3) } +(21 matching properties omitted from actual) + × built flows CLI against live relayflowd > agent step preserves the CliResult wrapper as output when the CLI emits non-JSON text 402ms + → expected null not to be null + × built flows CLI against live relayflowd > AgentWorker exposes wake_context to the CLI via RELAYFLOW_WAKE_CONTEXT env var (real analyzer prerequisite) 514ms + → Cannot read properties of null (reading 'story_title') + × built flows CLI against live relayflowd > AgentWorker leaves RELAYFLOW_WAKE_CONTEXT UNSET when the run has no wake_context (undefined-vs-null pin) 401ms + → Cannot read properties of null (reading 'env_present') + ✓ built flows CLI against live relayflowd > AgentWorker passes a declared model to an identified wrapper as RELAYFLOW_MODEL 495ms + ✓ built flows CLI against live relayflowd > AgentWorker refuses a nonconforming journal-submitted wrapper before exposing RELAYFLOW_MODEL 486ms + ✓ built flows CLI against live relayflowd > AgentWorker executes the raw claude adapter with its real model flag 506ms + ✓ built flows CLI against live relayflowd > AgentWorker executes the raw codex adapter with its real model flag 387ms + × built flows CLI against live relayflowd > AgentWorker leaves RELAYFLOW_MODEL UNSET when the step declares no model 371ms + → Cannot read properties of null (reading 'story_title') + × built flows CLI against live relayflowd > hn-monitor analyze-story reaches done through the real Claude analyzer CLI 35ms + → LIVE_ANALYZER_UNAVAILABLE: "/home/daytona/.relayflow-v2-supervisor/durable/repository/testdata/preflight/analyze-story-claude-cli" does not identify as relayflows-agent-cli-v1 — failing because gate-2 acceptance requires the real analyzer to execute. Set RELAYFLOWS_ALLOW_ANALYZER_SKIP=1 only if this run is not gate evidence. + ✓ built flows CLI against live relayflowd > preflights before journaling and names an unreachable socket 839ms + ✓ built flows CLI against live relayflowd > starts exactly one daemon when two runs race for one empty data dir 483ms + ✓ surface resume after a real daemon kill > resumes a three-step run with each successful completion exactly once 915ms + × a relayflow can be scheduled: tick source against live relayflowd > a tick spawns a real run whose step reports the SCHEDULED instant 448ms + → expected null to deeply equal { schedule_id: 'heartbeat-1m', …(3) } + +⎯⎯⎯⎯⎯⎯⎯ Failed Tests 8 ⎯⎯⎯⎯⎯⎯⎯ + + FAIL tests/live-kernel.test.ts > built flows CLI against live relayflowd > runs hn-monitor analyze-story end-to-end via a stub agent CLI (gate 2 clause 2 demo) +AssertionError: expected { …(12) } to match object { output: { …(3) }, …(1) } +(22 matching properties omitted from actual) + +- Expected ++ Received + + Object { +- "output": Object { +- "reasoning": "stub agent runtime — deterministic output for gate-2 clause-2 demo", +- "relevance_score": 5, +- "story_title": "stub", +- }, ++ "output": null, + "verification": Object { +- "gate": "json_schema", +- "verdict": "pass", ++ "gate": "execution", ++ "verdict": "fail", + }, + } + + ❯ tests/live-kernel.test.ts:657:36 + 655| && (entry as { step_id?: string }).step_id === 'analyze-story', + 656| ) as { payload: { output: unknown; verification: unknown } } | und… + 657| expect(stepCompleted?.payload).toMatchObject({ + | ^ + 658| output: { + 659| story_title: 'stub', + +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯[1/8]⎯ + + FAIL tests/live-kernel.test.ts > built flows CLI against live relayflowd > hn-monitor analyze-story FAILS verification when the CLI omits required schema fields +AssertionError: expected { …(12) } to match object { …(3) } +(21 matching properties omitted from actual) + +- Expected ++ Received + + Object { +- "completionReason": "retries_exhausted", ++ "completionReason": "worker_error", + "output": null, + "verification": Object { +- "gate": "json_schema", ++ "gate": "execution", + "verdict": "fail", + }, + } + + ❯ tests/live-kernel.test.ts:752:36 + 750| // its verification record names the json_schema rejection. The re… + 751| // parsed value is nulled before the completion is persisted. + 752| expect(stepCompleted?.payload).toMatchObject({ + | ^ + 753| completionReason: 'retries_exhausted', + 754| output: null, + +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯[2/8]⎯ + + FAIL tests/live-kernel.test.ts > built flows CLI against live relayflowd > agent step preserves the CliResult wrapper as output when the CLI emits non-JSON text +AssertionError: expected null not to be null + ❯ tests/live-kernel.test.ts:823:24 + 821| // here (parseJsonOutput returned null on non-JSON stdout) and + 822| // these assertions would all fail. + 823| expect(output).not.toBeNull(); + | ^ + 824| expect(output.exit_code).toBe(0); + 825| expect(output.stdout_tail).toContain('looked at the story'); + +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯[3/8]⎯ + + FAIL tests/live-kernel.test.ts > built flows CLI against live relayflowd > AgentWorker exposes wake_context to the CLI via RELAYFLOW_WAKE_CONTEXT env var (real analyzer prerequisite) +TypeError: Cannot read properties of null (reading 'story_title') + ❯ tests/live-kernel.test.ts:891:42 + 889| ) as { payload: { output: { story_title: string; reasoning: string… + 890| expect(stepCompleted).toBeDefined(); + 891| expect(stepCompleted!.payload.output.story_title).toBe(`echoed:${s… + | ^ + 892| expect(stepCompleted!.payload.output.reasoning).toContain(String(s… + 893| + +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯[4/8]⎯ + + FAIL tests/live-kernel.test.ts > built flows CLI against live relayflowd > AgentWorker leaves RELAYFLOW_WAKE_CONTEXT UNSET when the run has no wake_context (undefined-vs-null pin) +TypeError: Cannot read properties of null (reading 'env_present') + ❯ tests/live-kernel.test.ts:958:38 + 956| ) as { payload: { output: { env_present: boolean } } } | undefined; + 957| expect(completed).toBeDefined(); + 958| expect(completed!.payload.output.env_present).toBe(false); + | ^ + 959| + 960| delete process.env.RELAYFLOW_WAKE_CONTEXT; + +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯[5/8]⎯ + + FAIL tests/live-kernel.test.ts > built flows CLI against live relayflowd > AgentWorker leaves RELAYFLOW_MODEL UNSET when the step declares no model +TypeError: Cannot read properties of null (reading 'story_title') + ❯ tests/live-kernel.test.ts:1194:38 + 1192| expect(completed).toBeDefined(); + 1193| // UNSET, not EMPTY and not the leaked parent value. + 1194| expect(completed!.payload.output.story_title).toBe('model:UNSET'); + | ^ + 1195| + 1196| delete process.env.RELAYFLOW_MODEL; + +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯[6/8]⎯ + + FAIL tests/live-kernel.test.ts > built flows CLI against live relayflowd > hn-monitor analyze-story reaches done through the real Claude analyzer CLI +Error: LIVE_ANALYZER_UNAVAILABLE: "/home/daytona/.relayflow-v2-supervisor/durable/repository/testdata/preflight/analyze-story-claude-cli" does not identify as relayflows-agent-cli-v1 — failing because gate-2 acceptance requires the real analyzer to execute. Set RELAYFLOWS_ALLOW_ANALYZER_SKIP=1 only if this run is not gate evidence. + ❯ tests/live-kernel.test.ts:1223:15 + 1221| const notice = `LIVE_ANALYZER_UNAVAILABLE: ${readiness.detail}`; + 1222| if (process.env['RELAYFLOWS_ALLOW_ANALYZER_SKIP'] !== '1') { + 1223| throw new Error( + | ^ + 1224| `${notice} — failing because gate-2 acceptance requires the … + 1225| + 'Set RELAYFLOWS_ALLOW_ANALYZER_SKIP=1 only if this run is … + +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯[7/8]⎯ + + FAIL tests/live-kernel.test.ts > a relayflow can be scheduled: tick source against live relayflowd > a tick spawns a real run whose step reports the SCHEDULED instant +AssertionError: expected null to deeply equal { schedule_id: 'heartbeat-1m', …(3) } + +- Expected: +Object { + "lag_ms": 43000, + "schedule_id": "heartbeat-1m", + "scheduled_for_ms": 1764000000000, + "slot": 29400000, +} + ++ Received: +null + + ❯ tests/live-kernel.test.ts:1665:39 + 1663| // The bound: the run reports the grid instant and its own lag, so… + 1664| // backfilled run can tell it is running for a slot from the past. + 1665| expect(completed!.payload.output).toEqual({ + | ^ + 1666| schedule_id: 'heartbeat-1m', + 1667| slot: 29_400_000, + +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯[8/8]⎯ + + Test Files 1 failed (1) + Tests 8 failed | 23 passed (31) + Start at 05:47:55 + Duration 54.86s (transform 697ms, setup 0ms, collect 1.18s, tests 53.52s, environment 0ms, prepare 42ms) + + +Exit code: 1 diff --git a/evidence/worker-lease-lost/baseline.py b/evidence/worker-lease-lost/baseline.py new file mode 100644 index 000000000..6df578c60 --- /dev/null +++ b/evidence/worker-lease-lost/baseline.py @@ -0,0 +1,21 @@ +from pathlib import Path +import subprocess,os +root=Path.cwd() +files=subprocess.check_output(['git','diff','--name-only','f6ece41','HEAD','--','packages/sdk/src'],text=True).splitlines() +originals={name:(root/name).read_bytes() for name in files} +env={**os.environ,'PATH':'/home/daytona/.cargo/bin:'+os.environ['PATH'],'RELAYFLOWD_BIN':'/home/daytona/.relayflows-toolchain/target/2962130851/debug/relayflowd'} +def run(cmd,name): + with (root/'evidence/worker-lease-lost'/name).open('w') as f: + f.write('$ cd packages/sdk && '+cmd+'\n');f.flush() + result=subprocess.run(cmd,shell=True,cwd=root/'packages/sdk',env=env,stdout=f,stderr=subprocess.STDOUT) + f.write('\nExit code: '+str(result.returncode)+'\n') + return result.returncode +try: + for name in files: + (root/name).write_bytes(subprocess.check_output(['git','show','f6ece41:'+name])) + assert run('npm run build','baseline-build.txt')==0 + run('npx vitest run tests/live-kernel.test.ts','baseline-live-kernel.txt') +finally: + for name,data in originals.items(): (root/name).write_bytes(data) + assert all((root/name).read_bytes()==data for name,data in originals.items()) + assert run('npm run build','restored-build.txt')==0 diff --git a/evidence/worker-lease-lost/direct-mutant.txt b/evidence/worker-lease-lost/direct-mutant.txt new file mode 100644 index 000000000..854f7db9e --- /dev/null +++ b/evidence/worker-lease-lost/direct-mutant.txt @@ -0,0 +1,33 @@ +$ cd packages/sdk && npx vitest run tests/direct-run-worker-lease.test.ts + + RUN v2.1.9 /home/daytona/.relayflow-v2-supervisor/durable/repository/packages/sdk + + ❯ tests/direct-run-worker-lease.test.ts (2 tests | 1 failed) 13ms + × direct run handles LLM errors with leaseLost=true 8ms + → expected '{"ok":false,"command":"run","resoluti…' to contain 'connection closed' + +⎯⎯⎯⎯⎯⎯⎯ Failed Tests 1 ⎯⎯⎯⎯⎯⎯⎯ + + FAIL tests/direct-run-worker-lease.test.ts > direct run handles LLM errors with leaseLost=true +AssertionError: expected '{"ok":false,"command":"run","resoluti…' to contain 'connection closed' + +Expected: "connection closed" +Received: "{"ok":false,"command":"run","resolutions":[],"diagnostics":[{"severity":"failure","kind":"protocol_error","message":"relayflowd could not complete the run request: lease_conflict: lost"}],"path":"flow.ts","socketPath":"/tmp/relayflowd-4d1d0e012c91.sock"}" + + ❯ tests/direct-run-worker-lease.test.ts:33:41 + 31| const result = await runDirectFlow('flow.ts', '{}', '/unused', { loc… + 32| expect(result.exitCode).toBe(1); + 33| expect(JSON.stringify(result.report)).toContain(leaseLost ? 'connect… + | ^ + 34| }); + 35| + +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯[1/1]⎯ + + Test Files 1 failed (1) + Tests 1 failed | 1 passed (2) + Start at 05:42:37 + Duration 1.33s (transform 677ms, setup 0ms, collect 1.15s, tests 13ms, environment 0ms, prepare 47ms) + + +Exit code: 1 diff --git a/evidence/worker-lease-lost/direct-restored.txt b/evidence/worker-lease-lost/direct-restored.txt new file mode 100644 index 000000000..166d5e2b0 --- /dev/null +++ b/evidence/worker-lease-lost/direct-restored.txt @@ -0,0 +1,13 @@ +$ cd packages/sdk && npx vitest run tests/direct-run-worker-lease.test.ts + + RUN v2.1.9 /home/daytona/.relayflow-v2-supervisor/durable/repository/packages/sdk + + ✓ tests/direct-run-worker-lease.test.ts (2 tests) 6ms + + Test Files 1 passed (1) + Tests 2 passed (2) + Start at 05:42:39 + Duration 1.29s (transform 646ms, setup 0ms, collect 1.12s, tests 6ms, environment 0ms, prepare 50ms) + + +Exit code: 0 diff --git a/evidence/worker-lease-lost/fatal-mutant.txt b/evidence/worker-lease-lost/fatal-mutant.txt new file mode 100644 index 000000000..fed43a85c --- /dev/null +++ b/evidence/worker-lease-lost/fatal-mutant.txt @@ -0,0 +1,53 @@ +$ cd packages/sdk && npx vitest run tests/worker-lease-lost.test.ts -t 'non-lease worker error' + + RUN v2.1.9 /home/daytona/.relayflow-v2-supervisor/durable/repository/packages/sdk + + ❯ tests/worker-lease-lost.test.ts (16 tests | 2 failed | 14 skipped) 9ms + × agent stale lease subscriber > keeps a non-lease worker error fatal with its original identity 7ms + → expected "spy" to be called with arguments: [ Error: cli exploded ] + +Received: + + + +Number of calls: 0 + + × llm stale lease subscriber > keeps a non-lease worker error fatal with its original identity 1ms + → expected "spy" to be called with arguments: [ Error: cli exploded ] + +Received: + + + +Number of calls: 0 + + +⎯⎯⎯⎯⎯⎯⎯ Failed Tests 2 ⎯⎯⎯⎯⎯⎯⎯ + + FAIL tests/worker-lease-lost.test.ts > agent stale lease subscriber > keeps a non-lease worker error fatal with its original identity + FAIL tests/worker-lease-lost.test.ts > llm stale lease subscriber > keeps a non-lease worker error fatal with its original identity +AssertionError: expected "spy" to be called with arguments: [ Error: cli exploded ] + +Received: + + + +Number of calls: 0 + + ❯ tests/worker-lease-lost.test.ts:79:19 + 77| client.emit('step.dispatch', dispatch); + 78| await worker.close(); + 79| expect(fatal).toHaveBeenCalledWith(error); + | ^ + 80| expect(client.close).toHaveBeenCalledWith(error); + 81| expect(warning).not.toHaveBeenCalled(); + +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯[1/2]⎯ + + Test Files 1 failed (1) + Tests 2 failed | 14 skipped (16) + Start at 05:42:25 + Duration 890ms (transform 351ms, setup 0ms, collect 716ms, tests 9ms, environment 0ms, prepare 48ms) + + +Exit code: 1 diff --git a/evidence/worker-lease-lost/fatal-restored.txt b/evidence/worker-lease-lost/fatal-restored.txt new file mode 100644 index 000000000..54ead3c69 --- /dev/null +++ b/evidence/worker-lease-lost/fatal-restored.txt @@ -0,0 +1,13 @@ +$ cd packages/sdk && npx vitest run tests/worker-lease-lost.test.ts -t 'non-lease worker error' + + RUN v2.1.9 /home/daytona/.relayflow-v2-supervisor/durable/repository/packages/sdk + + ✓ tests/worker-lease-lost.test.ts (16 tests | 14 skipped) 6ms + + Test Files 1 passed (1) + Tests 2 passed | 14 skipped (16) + Start at 05:42:26 + Duration 888ms (transform 346ms, setup 0ms, collect 713ms, tests 6ms, environment 0ms, prepare 47ms) + + +Exit code: 0 diff --git a/evidence/worker-lease-lost/filter-mutant.txt b/evidence/worker-lease-lost/filter-mutant.txt new file mode 100644 index 000000000..ab083eab0 --- /dev/null +++ b/evidence/worker-lease-lost/filter-mutant.txt @@ -0,0 +1,204 @@ +$ cd packages/sdk && npx vitest run tests/worker-lease-lost.test.ts + + RUN v2.1.9 /home/daytona/.relayflow-v2-supervisor/durable/repository/packages/sdk + + ❯ tests/worker-lease-lost.test.ts (16 tests | 8 failed) 17ms + × agent stale lease subscriber > drops an expired dispatch and completes the kernel retry 7ms + → expected "spy" to not be called at all, but actually been called 1 times + +Received: + + 1st spy call: + + Array [ + [WorkerLeaseLostError: Agent lease is already expired for run/step.], + ] + + +Number of calls: 1 + + × agent stale lease subscriber > drops stepHeartbeat lease_conflict for a released attempt 1ms + → expected "spy" to not be called at all, but actually been called 1 times + +Received: + + 1st spy call: + + Array [ + [JournalProtocolError: lease_conflict: attempt has no active worker lease], + ] + + +Number of calls: 1 + + × agent stale lease subscriber > drops stepComplete lease_conflict for a released attempt 1ms + → expected "spy" to not be called at all, but actually been called 1 times + +Received: + + 1st spy call: + + Array [ + [JournalProtocolError: lease_conflict: attempt has no active worker lease], + ] + + +Number of calls: 1 + + × agent stale lease subscriber > drops stepComplete run_terminal for a released attempt 1ms + → expected "spy" to not be called at all, but actually been called 1 times + +Received: + + 1st spy call: + + Array [ + [JournalProtocolError: run_terminal: attempt has no active worker lease], + ] + + +Number of calls: 1 + + × llm stale lease subscriber > drops an expired dispatch and completes the kernel retry 1ms + → expected "spy" to not be called at all, but actually been called 1 times + +Received: + + 1st spy call: + + Array [ + [WorkerLeaseLostError: Agent lease is already expired for run/step.], + ] + + +Number of calls: 1 + + × llm stale lease subscriber > drops stepHeartbeat lease_conflict for a released attempt 1ms + → expected "spy" to not be called at all, but actually been called 1 times + +Received: + + 1st spy call: + + Array [ + [JournalProtocolError: lease_conflict: attempt has no active worker lease], + ] + + +Number of calls: 1 + + × llm stale lease subscriber > drops stepComplete lease_conflict for a released attempt 1ms + → expected "spy" to not be called at all, but actually been called 1 times + +Received: + + 1st spy call: + + Array [ + [JournalProtocolError: lease_conflict: attempt has no active worker lease], + ] + + +Number of calls: 1 + + × llm stale lease subscriber > drops stepComplete run_terminal for a released attempt 1ms + → expected "spy" to not be called at all, but actually been called 1 times + +Received: + + 1st spy call: + + Array [ + [JournalProtocolError: run_terminal: attempt has no active worker lease], + ] + + +Number of calls: 1 + + +⎯⎯⎯⎯⎯⎯⎯ Failed Tests 8 ⎯⎯⎯⎯⎯⎯⎯ + + FAIL tests/worker-lease-lost.test.ts > agent stale lease subscriber > drops an expired dispatch and completes the kernel retry + FAIL tests/worker-lease-lost.test.ts > llm stale lease subscriber > drops an expired dispatch and completes the kernel retry +AssertionError: expected "spy" to not be called at all, but actually been called 1 times + +Received: + + 1st spy call: + + Array [ + [WorkerLeaseLostError: Agent lease is already expired for run/step.], + ] + + +Number of calls: 1 + + ❯ tests/worker-lease-lost.test.ts:45:30 + 43| expect(runAgentCli).not.toHaveBeenCalled(); + 44| expect(client.stepComplete).not.toHaveBeenCalled(); + 45| expect(client.close).not.toHaveBeenCalled(); + | ^ + 46| expect(fatal).not.toHaveBeenCalled(); + 47| expect(warning).toHaveBeenCalledWith(expect.stringContaining('run_… + +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯[1/8]⎯ + + FAIL tests/worker-lease-lost.test.ts > agent stale lease subscriber > drops stepHeartbeat lease_conflict for a released attempt + FAIL tests/worker-lease-lost.test.ts > agent stale lease subscriber > drops stepComplete lease_conflict for a released attempt + FAIL tests/worker-lease-lost.test.ts > llm stale lease subscriber > drops stepHeartbeat lease_conflict for a released attempt + FAIL tests/worker-lease-lost.test.ts > llm stale lease subscriber > drops stepComplete lease_conflict for a released attempt +AssertionError: expected "spy" to not be called at all, but actually been called 1 times + +Received: + + 1st spy call: + + Array [ + [JournalProtocolError: lease_conflict: attempt has no active worker lease], + ] + + +Number of calls: 1 + + ❯ tests/worker-lease-lost.test.ts:67:30 + 65| client.emit('step.dispatch', dispatch); + 66| await worker.close(); + 67| expect(client.close).not.toHaveBeenCalled(); + | ^ + 68| expect(fatal).not.toHaveBeenCalled(); + 69| expect(warning).toHaveBeenCalledWith(expect.stringContaining(error… + +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯[2/8]⎯ + + FAIL tests/worker-lease-lost.test.ts > agent stale lease subscriber > drops stepComplete run_terminal for a released attempt + FAIL tests/worker-lease-lost.test.ts > llm stale lease subscriber > drops stepComplete run_terminal for a released attempt +AssertionError: expected "spy" to not be called at all, but actually been called 1 times + +Received: + + 1st spy call: + + Array [ + [JournalProtocolError: run_terminal: attempt has no active worker lease], + ] + + +Number of calls: 1 + + ❯ tests/worker-lease-lost.test.ts:67:30 + 65| client.emit('step.dispatch', dispatch); + 66| await worker.close(); + 67| expect(client.close).not.toHaveBeenCalled(); + | ^ + 68| expect(fatal).not.toHaveBeenCalled(); + 69| expect(warning).toHaveBeenCalledWith(expect.stringContaining(error… + +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯[3/8]⎯ + + Test Files 1 failed (1) + Tests 8 failed | 8 passed (16) + Start at 05:42:19 + Duration 870ms (transform 338ms, setup 0ms, collect 694ms, tests 17ms, environment 0ms, prepare 44ms) + + +Exit code: 1 diff --git a/evidence/worker-lease-lost/filter-restored.txt b/evidence/worker-lease-lost/filter-restored.txt new file mode 100644 index 000000000..851288894 --- /dev/null +++ b/evidence/worker-lease-lost/filter-restored.txt @@ -0,0 +1,13 @@ +$ cd packages/sdk && npx vitest run tests/worker-lease-lost.test.ts + + RUN v2.1.9 /home/daytona/.relayflow-v2-supervisor/durable/repository/packages/sdk + + ✓ tests/worker-lease-lost.test.ts (16 tests) 15ms + + Test Files 1 passed (1) + Tests 16 passed (16) + Start at 05:42:21 + Duration 904ms (transform 356ms, setup 0ms, collect 733ms, tests 15ms, environment 0ms, prepare 43ms) + + +Exit code: 0 diff --git a/evidence/worker-lease-lost/live-first.txt b/evidence/worker-lease-lost/live-first.txt new file mode 100644 index 000000000..edab08b65 --- /dev/null +++ b/evidence/worker-lease-lost/live-first.txt @@ -0,0 +1,14 @@ +$ cd packages/sdk && PATH=/home/daytona/.cargo/bin:$PATH npx vitest run tests/worker-lease-lost-live.test.ts + + RUN v2.1.9 /home/daytona/.relayflow-v2-supervisor/durable/repository/packages/sdk + + ✓ tests/worker-lease-lost-live.test.ts (2 tests) 652ms + ✓ reports journal success after completion rejects with lease_conflict 383ms + + Test Files 1 passed (1) + Tests 2 passed (2) + Start at 05:37:16 + Duration 2.00s (transform 682ms, setup 0ms, collect 1.18s, tests 652ms, environment 0ms, prepare 48ms) + + +Exit code: 0 diff --git a/evidence/worker-lease-lost/live-mutant.txt b/evidence/worker-lease-lost/live-mutant.txt new file mode 100644 index 000000000..d7db4f128 --- /dev/null +++ b/evidence/worker-lease-lost/live-mutant.txt @@ -0,0 +1,92 @@ +$ cd packages/sdk && npx vitest run tests/worker-lease-lost-live.test.ts + + RUN v2.1.9 /home/daytona/.relayflow-v2-supervisor/durable/repository/packages/sdk + + ❯ tests/worker-lease-lost-live.test.ts (3 tests | 3 failed) 811ms + × reports journal success after completion rejects with lease_conflict 273ms + → journal client: closed after lease_conflict: attempt has no active worker lease + × reports journal success after completion rejects with run_terminal 244ms + → journal client: not connected (run.get): journal client: closed after run_terminal: attempt has no active worker lease + × reports journal success when a renewal rejects after completion landed 293ms + → journal client: closed after lease_conflict: attempt has no active worker lease + +⎯⎯⎯⎯⎯⎯⎯ Failed Tests 3 ⎯⎯⎯⎯⎯⎯⎯ + + FAIL tests/worker-lease-lost-live.test.ts > reports journal success after completion rejects with lease_conflict +Error: journal client: closed after lease_conflict: attempt has no active worker lease + ❯ JournalClient.close src/journal-client.ts:131:9 + 129| const closed = cause === undefined + 130| ? new Error('journal client: closed by caller') + 131| : new Error(`journal client: closed after ${cause instanceof Err… + | ^ + 132| this.disconnectCause ??= closed; + 133| this.failAll(closed); + ❯ tests/worker-lease-lost-live.test.ts:17:50 + ❯ LlmWorker. src/worker-lease.ts:20:17 + ❯ src/llm-worker.ts:49:66 + +Caused by: JournalProtocolError: lease_conflict: attempt has no active worker lease + ❯ JournalClient. tests/worker-lease-lost-live.test.ts:31:19 + +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯ +Serialized Error: { code: 'lease_conflict', verb: 'step.complete' } +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯[1/3]⎯ + + FAIL tests/worker-lease-lost-live.test.ts > reports journal success after completion rejects with run_terminal +Error: journal client: not connected (run.get): journal client: closed after run_terminal: attempt has no active worker lease + ❯ src/journal-client.ts:190:16 + 188| if (!this.socket || this.socket.destroyed) { + 189| const cause = this.disconnectCause; + 190| reject(new Error( + | ^ + 191| `journal client: not connected (${verb})${cause === undefine… + 192| cause === undefined ? undefined : { cause }, + ❯ JournalClient.request src/journal-client.ts:187:12 + ❯ JournalClient.runGet src/journal-client.ts:248:17 + ❯ waitForRunningStep src/cli/run.ts:781:35 + ❯ Module.classifyOutcome src/cli/run.ts:584:7 + ❯ consume src/authored-worker-step.ts:77:23 + ❯ Object.llm src/authored-worker-step.ts:212:22 + ❯ Module.observeStep src/progress.ts:48:20 + ❯ AuthoredFlowOperation.begin src/authored-flow-operation.ts:174:23 + +Caused by: Error: journal client: closed after run_terminal: attempt has no active worker lease + ❯ JournalClient.close src/journal-client.ts:131:9 + ❯ tests/worker-lease-lost-live.test.ts:17:50 + ❯ LlmWorker. src/worker-lease.ts:20:17 + ❯ src/llm-worker.ts:49:66 + +Caused by: JournalProtocolError: run_terminal: attempt has no active worker lease + ❯ JournalClient. tests/worker-lease-lost-live.test.ts:31:19 + +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯ +Serialized Error: { code: 'run_terminal', verb: 'step.complete' } +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯[2/3]⎯ + + FAIL tests/worker-lease-lost-live.test.ts > reports journal success when a renewal rejects after completion landed +Error: journal client: closed after lease_conflict: attempt has no active worker lease + ❯ JournalClient.close src/journal-client.ts:131:9 + 129| const closed = cause === undefined + 130| ? new Error('journal client: closed by caller') + 131| : new Error(`journal client: closed after ${cause instanceof Err… + | ^ + 132| this.disconnectCause ??= closed; + 133| this.failAll(closed); + ❯ tests/worker-lease-lost-live.test.ts:61:50 + ❯ LlmWorker. src/worker-lease.ts:20:17 + ❯ src/llm-worker.ts:49:66 + +Caused by: JournalProtocolError: lease_conflict: attempt has no active worker lease + ❯ tests/worker-lease-lost-live.test.ts:67:19 + +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯ +Serialized Error: { code: 'lease_conflict', verb: 'step.heartbeat' } +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯[3/3]⎯ + + Test Files 1 failed (1) + Tests 3 failed (3) + Start at 05:42:28 + Duration 2.13s (transform 672ms, setup 0ms, collect 1.15s, tests 811ms, environment 0ms, prepare 46ms) + + +Exit code: 1 diff --git a/evidence/worker-lease-lost/live-restored.txt b/evidence/worker-lease-lost/live-restored.txt new file mode 100644 index 000000000..982515818 --- /dev/null +++ b/evidence/worker-lease-lost/live-restored.txt @@ -0,0 +1,15 @@ +$ cd packages/sdk && npx vitest run tests/worker-lease-lost-live.test.ts + + RUN v2.1.9 /home/daytona/.relayflow-v2-supervisor/durable/repository/packages/sdk + + ✓ tests/worker-lease-lost-live.test.ts (3 tests) 884ms + ✓ reports journal success after completion rejects with lease_conflict 302ms + ✓ reports journal success when a renewal rejects after completion landed 310ms + + Test Files 1 passed (1) + Tests 3 passed (3) + Start at 05:42:30 + Duration 2.21s (transform 687ms, setup 0ms, collect 1.15s, tests 884ms, environment 0ms, prepare 46ms) + + +Exit code: 0 diff --git a/evidence/worker-lease-lost/mutate.py b/evidence/worker-lease-lost/mutate.py new file mode 100644 index 000000000..b45368f74 --- /dev/null +++ b/evidence/worker-lease-lost/mutate.py @@ -0,0 +1,52 @@ +from pathlib import Path +import subprocess, hashlib, os +root=Path.cwd() +sdk=root/'packages/sdk' +evidence=root/'evidence/worker-lease-lost' +env={**os.environ, 'PATH':'/home/daytona/.cargo/bin:'+os.environ['PATH']} +def run(name,command): + result=subprocess.run(command,cwd=sdk,env=env,shell=True,text=True,stdout=subprocess.PIPE,stderr=subprocess.STDOUT) + (evidence/(name+'.txt')).write_text('$ cd packages/sdk && '+command+'\n'+result.stdout+'\nExit code: '+str(result.returncode)+'\n') + return result.returncode +def mutation(name,path,old,new,command): + p=root/path + original=p.read_bytes() + assert old.encode() in original + try: + p.write_bytes(original.replace(old.encode(),new.encode(),1)) + failed=run(name+'-mutant',command) + finally: + p.write_bytes(original) + assert p.read_bytes()==original + passed=run(name+'-restored',command) + with (evidence/'mutations.txt').open('a') as f: + f.write(f'{name}\nFile: {path}\nReplace: {old!r}\nWith: {new!r}\nRestored SHA256: {hashlib.sha256(original).hexdigest()}\nMutant exit: {failed}; restored exit: {passed}\n\n') + assert failed != 0 and passed == 0,(name,failed,passed) +(evidence/'mutations.txt').write_text('') +mutation('filter','packages/sdk/src/worker-lease.ts', + 'if (!isLeaseLost(error))', 'if (true)', + 'npx vitest run tests/worker-lease-lost.test.ts') +mutation('terminal','packages/sdk/src/worker-lease.ts', + "error.code === 'run_terminal'", "error.code === 'never_drop_terminal'", + "npx vitest run tests/worker-lease-lost.test.ts -t run_terminal") +mutation('fatal','packages/sdk/src/worker-lease.ts', + 'if (!isLeaseLost(error)) { fatal(error); return; }', + 'if (!isLeaseLost(error)) { return; }', + "npx vitest run tests/worker-lease-lost.test.ts -t 'non-lease worker error'") +mutation('live','packages/sdk/src/worker-lease.ts', + 'if (!isLeaseLost(error))', 'if (true)', + 'npx vitest run tests/worker-lease-lost-live.test.ts') +mutation('sweep','packages/sdk/src/cli/run.ts', + 'leaseDeadlineMs + LEASE_SWEEP_GRACE_MS - Date.now()', 'leaseDeadlineMs - Date.now()', + 'npx vitest run tests/worker-lease-sweep.test.ts') +mutation('direct','packages/sdk/src/cli/direct-run.ts', + "localLlm.on('error', onWorkerFailure('local-llm', error => { llmFailure = error; client.close(); }));", + "localLlm.on('error', error => { llmFailure = error; client.close(); });", + 'npx vitest run tests/direct-run-worker-lease.test.ts') +mutation('resume','packages/sdk/src/cli/run.ts', + """ authoredLlm.on('error', onWorkerFailure('resume-llm', error => { + llmFailure = error; + client.close(); + })); +""", '', + 'npx vitest run tests/resume-worker-lease.test.ts') diff --git a/evidence/worker-lease-lost/mutations.txt b/evidence/worker-lease-lost/mutations.txt new file mode 100644 index 000000000..06b96a219 --- /dev/null +++ b/evidence/worker-lease-lost/mutations.txt @@ -0,0 +1,49 @@ +filter +File: packages/sdk/src/worker-lease.ts +Replace: 'if (!isLeaseLost(error))' +With: 'if (true)' +Restored SHA256: 4738712912b3d44be73d33469bd7b1e9eab0e8bae5a853bbd4a3f9235fd1440e +Mutant exit: 1; restored exit: 0 + +terminal +File: packages/sdk/src/worker-lease.ts +Replace: "error.code === 'run_terminal'" +With: "error.code === 'never_drop_terminal'" +Restored SHA256: 4738712912b3d44be73d33469bd7b1e9eab0e8bae5a853bbd4a3f9235fd1440e +Mutant exit: 1; restored exit: 0 + +fatal +File: packages/sdk/src/worker-lease.ts +Replace: 'if (!isLeaseLost(error)) { fatal(error); return; }' +With: 'if (!isLeaseLost(error)) { return; }' +Restored SHA256: 4738712912b3d44be73d33469bd7b1e9eab0e8bae5a853bbd4a3f9235fd1440e +Mutant exit: 1; restored exit: 0 + +live +File: packages/sdk/src/worker-lease.ts +Replace: 'if (!isLeaseLost(error))' +With: 'if (true)' +Restored SHA256: 4738712912b3d44be73d33469bd7b1e9eab0e8bae5a853bbd4a3f9235fd1440e +Mutant exit: 1; restored exit: 0 + +sweep +File: packages/sdk/src/cli/run.ts +Replace: 'leaseDeadlineMs + LEASE_SWEEP_GRACE_MS - Date.now()' +With: 'leaseDeadlineMs - Date.now()' +Restored SHA256: b12b6991b1d3d6850e3fdcb0d711438e23f2219fdc08ac46ea1048ea6861584c +Mutant exit: 1; restored exit: 0 + +direct +File: packages/sdk/src/cli/direct-run.ts +Replace: "localLlm.on('error', onWorkerFailure('local-llm', error => { llmFailure = error; client.close(); }));" +With: "localLlm.on('error', error => { llmFailure = error; client.close(); });" +Restored SHA256: 4c28be2f845789047c8ae39eca2003d2d7b0c1c572e92d0aee88edb3cda81341 +Mutant exit: 1; restored exit: 0 + +resume +File: packages/sdk/src/cli/run.ts +Replace: " authoredLlm.on('error', onWorkerFailure('resume-llm', error => {\n llmFailure = error;\n client.close();\n }));\n" +With: '' +Restored SHA256: b12b6991b1d3d6850e3fdcb0d711438e23f2219fdc08ac46ea1048ea6861584c +Mutant exit: 1; restored exit: 0 + diff --git a/evidence/worker-lease-lost/new-test-types.txt b/evidence/worker-lease-lost/new-test-types.txt new file mode 100644 index 000000000..f471084b1 --- /dev/null +++ b/evidence/worker-lease-lost/new-test-types.txt @@ -0,0 +1,3 @@ +$ cd packages/sdk && npx tsc -p ../../evidence/worker-lease-lost/tsconfig.tests.json + +Exit code: 0 diff --git a/evidence/worker-lease-lost/npm-test-final.txt b/evidence/worker-lease-lost/npm-test-final.txt new file mode 100644 index 000000000..83345938b --- /dev/null +++ b/evidence/worker-lease-lost/npm-test-final.txt @@ -0,0 +1,997 @@ +$ cd packages/sdk && PATH=/home/daytona/.cargo/bin:$PATH RELAYFLOWD_BIN=/home/daytona/.relayflows-toolchain/target/2962130851/debug/relayflowd npm test + +> @relayflows/sdk@2.0.29 test +> sh scripts/test.sh + + +> @relayflows/sdk@2.0.29 test:prep +> ( cd ../../kernel && sh ../ops/cargo.sh build ) && ( [ ! -d ../../testdata/preflight ] || find ../../testdata/preflight -name '*-cli' -type f -exec chmod +x {} + ) + + Finished `dev` profile [unoptimized + debuginfo] target(s) in 0.07s + +> @relayflows/sdk@2.0.29 typecheck +> tsc --noEmit && tsc -p tsconfig.type-tests.json + + +> @relayflows/sdk@2.0.29 build +> tsc && node scripts/make-cli-executable.mjs + + +> @relayflows/sdk@2.0.29 typecheck:tests +> tsc -p tsconfig.tests.json + + + RUN v2.1.9 /home/daytona/.relayflow-v2-supervisor/durable/repository/packages/sdk + +stdout | tests/live-kernel.test.ts +LIVE_KERNEL relayflowd=/home/daytona/.relayflows-toolchain/target/2962130851/debug/relayflowd +LIVE_KERNEL flows=/home/daytona/.relayflow-v2-supervisor/durable/repository/packages/sdk/dist/cli.js + + ✓ tests/preflight.test.ts (67 tests) 106ms + ✓ tests/cloud-read.test.ts (41 tests) 48ms + ✓ tests/cli.test.ts (70 tests) 1661ms + ✓ flows check CLI > binds a checked relative wrapper to the flow directory for worker execution 429ms + ✓ flows check CLI > resolves a bare PATH-resolved claude with no declared model, in an isolated PATH 411ms + ✓ tests/cloud-sync.test.ts (40 tests) 770ms + ✓ tests/plugin-extension.test.ts (90 tests) 438ms + ✓ tests/cloud-transcript-codex.test.ts (39 tests) 17ms + ✓ tests/observer-link.test.ts (39 tests) 135ms + ❯ tests/hosted-extension-isolation.test.ts (22 tests | 13 failed) 3960ms + × hosted extension capability isolation > executes the exact capability-only handler for a queued receipt 207ms + → Hosted extension sandbox exited without a valid completion (exit 1): bwrap: loopback: Failed RTM_NEWADDR: Operation not permitted + + × hosted extension capability isolation > executes the exact capability-only handler for a duplicate receipt 212ms + → Hosted extension sandbox exited without a valid completion (exit 1): bwrap: loopback: Failed RTM_NEWADDR: Operation not permitted + + × hosted extension capability isolation > launches through the captured process primitive after builtin export synchronization 221ms + → promise rejected "Error: Hosted extension sandbox exited wi… { code: '…' }" instead of resolving + × hosted extension capability isolation > ignores inherited launcher overrides and decodes manifests with the captured Buffer intrinsic 217ms + → promise rejected "Error: Hosted extension sandbox exited wi… { code: '…' }" instead of resolving + × hosted extension capability isolation > streams verified bytes when the live store is replaced and no writable staging path exists 239ms + → promise rejected "Error: Hosted extension sandbox exited wi… { code: '…' }" instead of resolving + × hosted extension capability isolation > mounts pinned private Surface bytes when the live package changes before launch 217ms + → promise rejected "Error: Hosted extension sandbox exited wi… { code: '…' }" instead of resolving + × hosted extension capability isolation > shields verified Surface files before async settlement 240ms + → promise rejected "Error: Hosted extension sandbox exited wi… { code: '…' }" instead of resolving + × hosted extension capability isolation > preserves a typed host refusal while disclosing only a fixed marker to the child 221ms + → expected Error: Hosted extension sandbox exited wi… { code: '…' } to be Error: private Cloud policy detail { code: '…' } // Object.is equality + × hosted extension capability isolation > denies ambient credentials, host files, writes, network, subprocesses, and undeclared context verbs 214ms + → Hosted extension sandbox exited without a valid completion (exit 1): bwrap: loopback: Failed RTM_NEWADDR: Operation not permitted + + × hosted extension capability isolation > enforces OS address-space and data bounds on native Buffer allocation 214ms + → expected Error: Hosted extension sandbox exited wi… { code: '…' } to match object { code: 'plugin_unsupported', …(1) } + × hosted extension capability isolation > blocks extra handler fields and authority-bearing receipt fields at the parent port 223ms + → expected Error: Hosted extension sandbox exited wi… { code: '…' } to match object { code: 'plugin_event_unroutable' } + × hosted extension capability isolation > constructs adapter authority with the captured freeze intrinsic 218ms + → Hosted extension sandbox exited without a valid completion (exit 1): bwrap: loopback: Failed RTM_NEWADDR: Operation not permitted + + × hosted extension capability isolation > writes the Surface manifest and protocol without inherited toJSON behavior 299ms + → promise rejected "Error: Hosted extension sandbox exited wi… { code: '…' }" instead of resolving + ✓ hosted extension capability isolation > fails closed when the handler omits or repeats the single capability call 690ms +(node:35128) ExperimentalWarning: SQLite is an experimental feature and might change at any time +(Use `node --trace-warnings ...` to show where the warning was created) + ✓ tests/authored-flow.test.ts (34 tests) 766ms + ✓ tests/agent-transcript.test.ts (29 tests) 255ms + ✓ tests/cli-status.test.ts (27 tests) 983ms + ✓ flows status > resolves the run with no arguments from inside a worker-spawned agent 769ms + ✓ tests/cloud-run.test.ts (58 tests) 731ms + ❯ tests/babysitter-native-extension.test.ts (41 tests | 1 failed) 1886ms + × native Babysitter extension > runs the exact published 2.0.26 native bytes in the isolated capability path 241ms + → promise rejected "Error: Hosted extension sandbox exited wi… { code: '…' }" instead of resolving + ✓ native Babysitter extension > checks the reviewed base pin with captured hash methods 308ms + ✓ tests/relay-cli-surface.test.ts (75 tests) 35ms + ✓ tests/worker-cli.test.ts (18 tests) 25258ms + ✓ registered CLI model defaults > passes the same priced Claude default to the real provider invocation 408ms + ✓ step discovery environment > names the run, step, attempt and an absolute data dir for a direct agent spawn 418ms + ✓ step discovery environment > exports none of the four without a data dir, even when the worker inherited them 443ms + ✓ wrapper discovery environment > sets the four names from the dispatch and still refuses ambient values and other secrets 417ms + ✓ wrapper discovery environment > exports none of the four to a wrapper without a data dir, even when the worker inherited them 430ms + ✓ custom wrapper execution identity > passes an explicit safe environment at identification and execution 364ms + ✓ custom wrapper execution identity > refuses a wrapper symlink retarget before delivering private values 383ms + ✓ custom wrapper execution identity > bounds wrapper execution after acknowledgement 393ms + ✓ custom wrapper execution identity > bounds captured wrapper output 403ms + ✓ custom wrapper execution identity > refuses a duplicate execute protocol frame 364ms + ✓ custom wrapper execution bounds are reader-owned > resolves when a conforming wrapper leaks a stdio pipe to a background helper 1897ms + ✓ custom wrapper execution bounds are reader-owned > resolves when the leaked helper inherits stderr only 1945ms + ✓ custom wrapper execution bounds are reader-owned > resolves when a wrapper leaks a stdio pipe and exits before identifying 3573ms + ✓ custom wrapper execution bounds are reader-owned > journals a completionReason at the default bound when a wrapper leaks a stdio pipe 11572ms + ✓ custom wrapper execution bounds are reader-owned > accepts an execute token and an over-8KiB payload flushed in one write 366ms + ✓ custom wrapper execution bounds are reader-owned > accepts the same over-8KiB payload whether or not it coalesces with the execute token 1153ms + ✓ custom wrapper execution bounds are reader-owned > still bounds an un-terminated handshake buffer and names the bound 372ms + ✓ delivers the journaled memory pack to the real wrapper and excludes its charge from completion usage 356ms + ✓ tests/step-failure-diagnostic.test.ts (25 tests) 69ms + ✓ tests/daemon-lifecycle.test.ts (42 tests) 46ms + ✓ tests/stop-process-group.test.ts (9 tests) 12850ms + ✓ every stop reaches the process group, not just the direct child > exits the run after an execution-timeout stop 789ms + ✓ every stop reaches the process group, not just the direct child > exits the run after a protocol terminate stop 418ms + ✓ every stop reaches the process group, not just the direct child > kills a SIGTERM-deaf grandchild after a protocol terminate stop 1668ms + ✓ every stop reaches the process group, not just the direct child > kills a SIGTERM-deaf grandchild after an execution-timeout stop 2050ms + ✓ every stop reaches the process group, not just the direct child > holds the loop open long enough for the escalation to run 1088ms + ✓ a wrapper that exits with no execution deadline still drains > reports the wrapper result and reaps a grandchild holding its pipes 662ms + ✓ a wrapper that exits with no execution deadline still drains > reaps a SIGTERM-deaf grandchild holding its pipes 1652ms + ✓ a wrapper that exits with no execution deadline still drains > settles on its own deadline when an escaped holder withholds close 4265ms + ✓ tests/run-state.test.ts (21 tests) 11ms + ✓ tests/cloud-deploy.test.ts (40 tests) 989ms +stdout | tests/live-kernel.test.ts > surface resume after a real daemon kill > resumes a three-step run with each successful completion exactly once +LIVE_KERNEL kill -9 pid=36640 run=01M36CNPCDT7TPM6F097RN7VQS while step=two state=Running + + ✓ tests/hosted-base-snapshot.test.ts (18 tests) 1960ms + ✓ hosted base private snapshot > stops streaming project entries at the shared count limit 1876ms + ❯ tests/live-kernel.test.ts (31 tests | 8 failed) 53101ms + ✓ built flows CLI against live relayflowd > twenty-six-step reuses 25 durable completions after editing the failed final step 2163ms + ✓ built flows CLI against live relayflowd > runs rung (a), parks rung (b), and keeps JSON report-shaped 2858ms + ✓ built flows CLI against live relayflowd > allows a deterministic run to exceed the bounded request timeout 32618ms + ✓ built flows CLI against live relayflowd > follows a live worker dispatch through flows run 590ms + ✓ built flows CLI against live relayflowd > runs an agent CLI end to end through the SDK worker 422ms + ✓ built flows CLI against live relayflowd > f.agent lowers to a real agent step and dispatches through a live worker 541ms + ✓ built flows CLI against live relayflowd > can always get a parked run to a late-attaching worker 5589ms + ✓ built flows CLI against live relayflowd > reports a real manual-recovery NeedsHuman state as parked 457ms + × built flows CLI against live relayflowd > runs hn-monitor analyze-story end-to-end via a stub agent CLI (gate 2 clause 2 demo) 434ms + → expected { …(12) } to match object { output: { …(3) }, …(1) } +(22 matching properties omitted from actual) + × built flows CLI against live relayflowd > hn-monitor analyze-story FAILS verification when the CLI omits required schema fields 392ms + → expected { …(12) } to match object { …(3) } +(21 matching properties omitted from actual) + × built flows CLI against live relayflowd > agent step preserves the CliResult wrapper as output when the CLI emits non-JSON text 416ms + → expected null not to be null + × built flows CLI against live relayflowd > AgentWorker exposes wake_context to the CLI via RELAYFLOW_WAKE_CONTEXT env var (real analyzer prerequisite) 703ms + → Cannot read properties of null (reading 'story_title') + × built flows CLI against live relayflowd > AgentWorker leaves RELAYFLOW_WAKE_CONTEXT UNSET when the run has no wake_context (undefined-vs-null pin) 398ms + → Cannot read properties of null (reading 'env_present') + ✓ built flows CLI against live relayflowd > AgentWorker passes a declared model to an identified wrapper as RELAYFLOW_MODEL 456ms + ✓ built flows CLI against live relayflowd > AgentWorker refuses a nonconforming journal-submitted wrapper before exposing RELAYFLOW_MODEL 439ms + ✓ built flows CLI against live relayflowd > AgentWorker executes the raw claude adapter with its real model flag 419ms + ✓ built flows CLI against live relayflowd > AgentWorker executes the raw codex adapter with its real model flag 379ms + × built flows CLI against live relayflowd > AgentWorker leaves RELAYFLOW_MODEL UNSET when the step declares no model 387ms + → Cannot read properties of null (reading 'story_title') + × built flows CLI against live relayflowd > hn-monitor analyze-story reaches done through the real Claude analyzer CLI 28ms + → LIVE_ANALYZER_UNAVAILABLE: "/home/daytona/.relayflow-v2-supervisor/durable/repository/testdata/preflight/analyze-story-claude-cli" does not identify as relayflows-agent-cli-v1 — failing because gate-2 acceptance requires the real analyzer to execute. Set RELAYFLOWS_ALLOW_ANALYZER_SKIP=1 only if this run is not gate evidence. + ✓ built flows CLI against live relayflowd > preflights before journaling and names an unreachable socket 826ms + ✓ built flows CLI against live relayflowd > starts exactly one daemon when two runs race for one empty data dir 508ms + ✓ surface resume after a real daemon kill > resumes a three-step run with each successful completion exactly once 902ms + × a relayflow can be scheduled: tick source against live relayflowd > a tick spawns a real run whose step reports the SCHEDULED instant 451ms + → expected null to deeply equal { schedule_id: 'heartbeat-1m', …(3) } + ❯ tests/hosted-extension-protocol.test.ts (24 tests | 8 failed) 41948ms + × hosted extension hostile protocol > uses captured JSON intrinsics for the complete parent boundary 221ms + → Hosted extension sandbox exited without a valid completion (exit 1): bwrap: loopback: Failed RTM_NEWADDR: Operation not permitted + + × hosted extension hostile protocol > rejects an import-time different PR frame with zero adapter calls 222ms + → expected Error: Hosted extension sandbox exited wi… { code: '…' } to match object { code: 'plugin_event_unroutable' } + × hosted extension hostile protocol > rejects an import-time different delivery frame with zero adapter calls 221ms + → expected Error: Hosted extension sandbox exited wi… { code: '…' } to match object { code: 'plugin_event_unroutable' } + × hosted extension hostile protocol > rejects an import-time different event frame with zero adapter calls 219ms + → expected Error: Hosted extension sandbox exited wi… { code: '…' } to match object { code: 'plugin_event_unroutable' } + × hosted extension hostile protocol > rejects two forged calls after the authoritative first outcome settles 10148ms + → hostile child did not invoke the adapter + × hosted extension hostile protocol > waits for a pending adapter to reject after a forged child error 10142ms + → hostile child did not invoke the adapter + × hosted extension hostile protocol > waits for a pending adapter to resolve after a forged child error 10144ms + → hostile child did not invoke the adapter + × hosted extension hostile protocol > returns a typed adapter rejection even when the hostile child hangs 10142ms + → hostile child did not invoke the adapter + ✓ tests/journal-client.test.ts (17 tests) 79ms + ✓ tests/authored-root.test.ts (13 tests) 154ms + ✓ tests/cloud-connect.test.ts (24 tests) 3051ms + ✓ hosted verbs connect before they submit > flows run --cloud submits once the prompt connected the integration 2128ms + ✓ tests/flow-extension-compose.test.ts (22 tests) 3765ms + ✓ composing flow extensions onto a base flow > composes two extensions in declaration order, and the order is the lockfile order 512ms + ✓ composing flow extensions onto a base flow > flows check reports the composition and keeps the composed triggers deliverable 757ms + ❯ tests/authored-node-runtime.test.ts (14 tests | 14 skipped) 12ms + ✓ tests/close-pr-flow.test.ts (28 tests) 318ms + ✓ tests/validate.test.ts (68 tests) 28ms + ✓ tests/verb-field-lint.test.ts (96 tests) 312ms + ✓ tests/bundle.test.ts (26 tests) 8708ms + ✓ immutable bundles > returns exit 2 naming a byte-flipped payload and refuses to reuse corruption 389ms + ✓ immutable bundles > verifies with --verify in any position and answers --json with one object 773ms + ✓ immutable bundles > refuses --out with --verify rather than ignoring the destination 395ms + ✓ immutable bundles > builds and verifies the canonical YAML fixture through the compiled CLI 1203ms + ✓ immutable bundles > emits the ephemeral warning on CLI stderr and uses the default output directory 827ms + ✓ immutable bundles > refuses build-provable CLI resolution errors without environment probes 390ms + ✓ immutable bundles > builds a standalone TS fixture twice with identical executable hashes 2288ms + ✓ immutable bundles > refuses to label installed dependency drift with lockfile pins 389ms + ✓ immutable bundles > refuses invalid CLI arguments %j 406ms + ✓ immutable bundles > refuses invalid CLI arguments "--out" 375ms + ✓ immutable bundles > refuses invalid CLI arguments "--verify" 388ms + ✓ immutable bundles > refuses invalid CLI arguments "--verify" 386ms + ✓ immutable bundles > refuses invalid CLI arguments "--out" 375ms + ✓ tests/tick-source.test.ts (33 tests) 27ms + ✓ tests/authored-step-graph.test.ts (25 tests) 794ms + ✓ tests/flow-executor-chain.test.ts (14 tests) 9500ms + ✓ flow executor LLM and output-binding chain > runs f.llm -> f.agent -> f.run with schema-verified journal output and the exact allowed model 810ms + ✓ flow executor LLM and output-binding chain > runs a dollar-budgeted authored Claude agent with the same default used by preflight 554ms + ✓ flow executor LLM and output-binding chain > runs the exact authored flagship f.llm -> f.agent -> f.run path through the durable CLI root 1492ms + ✓ flow executor LLM and output-binding chain > resumes an interrupted durable authored root without replaying completed flagship effects 3301ms + ✓ flow executor LLM and output-binding chain > passes a declarative verified value through an agent into a deterministic artifact 635ms + ✓ flow executor LLM and output-binding chain > flows run consumes YAML bindings and resume reuses the original journal output 1025ms + ✓ tests/agent-relay-transport.test.ts (16 tests) 2217ms + ✓ Relay completion at the journal boundary > does not complete at readiness and journals exact output, receipt, and priced accounting 1006ms + ✓ Relay completion at the journal boundary > aborts polling on rejected renewal and never writes a stale completion 1002ms + ✓ tests/authored-flow-lifecycle-executor.test.ts (27 tests) 579ms + ✓ tests/pr-review-post.test.ts (21 tests) 2036ms + ✓ tests/step-failure-excerpt.test.ts (42 tests) 194ms + ✓ tests/authored-flow-slack.test.ts (7 tests) 1694ms + ✓ authored Slack helper effects > replays after SIGKILL before confirm with the same token and one successful completion 561ms + ✓ authored Slack helper effects > replays after SIGKILL before complete with the same token and one successful completion 547ms + ✓ tests/mcp.test.ts (30 tests) 21251ms + ✓ MCP preflight and transports > flows check refuses an undeclared server with exit 2 and no daemon 567ms + ✓ MCP preflight and transports > flows check reports a refusing server and leaves no PID 626ms + ✓ MCP preflight and transports > kills a SIGTERM-resistant silent child after a parent-owned handshake deadline 1312ms + ✓ MCP preflight and transports > reaps a SIGTERM-resistant descendant with inherit stdio before cleanup finishes 1111ms + ✓ MCP preflight and transports > reaps a SIGTERM-resistant descendant with ignore stdio before cleanup finishes 2062ms + ✓ MCP preflight and transports > reports malformed connection configuration as config_invalid 587ms + ✓ authored MCP effects against the real kernel > reports a dropped tool connection as a failed CLI run 14082ms + ✓ tests/authored-step-index.test.ts (17 tests) 24ms + ✓ tests/tick-runner.test.ts (22 tests) 2398ms + ✓ CLI argument parsing refuses coercion rather than accepting it > refuses --interval-ms fractional as an invocation error 379ms + ✓ CLI argument parsing refuses coercion rather than accepting it > refuses --interval-ms exponent notation as an invocation error 381ms + ✓ CLI argument parsing refuses coercion rather than accepting it > refuses --interval-ms hex as an invocation error 378ms + ✓ CLI argument parsing refuses coercion rather than accepting it > refuses --interval-ms trailing text as an invocation error 404ms + ✓ CLI argument parsing refuses coercion rather than accepting it > refuses --interval-ms empty as an invocation error 381ms + ✓ CLI argument parsing refuses coercion rather than accepting it > accepts an exact integer and proceeds past parsing 391ms +(node:38832) ExperimentalWarning: SQLite is an experimental feature and might change at any time +(Use `node --trace-warnings ...` to show where the warning was created) + ✓ tests/gate-contract.test.ts (20 tests) 156ms + ✓ tests/authored-node-result.test.ts (39 tests) 19ms + ✓ tests/cli-replay.test.ts (37 tests) 1211ms + ✓ flows replay > --json is byte-identical across two CLI invocations (diff) 788ms + ✓ tests/authored-human.test.ts (13 tests) 83ms + ✓ tests/wrapper-execution-duration.test.ts (7 tests) 10824ms + ✓ keeps the handshake deadline independent of the removed execution deadline 10058ms + ✓ still lets a lease abort stop an unlimited wrapper before it produces output 509ms + ✓ tests/direct-input.test.ts (6 tests) 5572ms + ✓ direct .flow.ts input through the built CLI and live runtime > returns exit 3 for an authored human handoff and persists its outcome 631ms + ✓ direct .flow.ts input through the built CLI and live runtime > returns exit 1 for an authored step_failed verdict and persists its outcome 621ms + ✓ direct .flow.ts input through the built CLI and live runtime > executes inline and file JSON input through relayflowd 1888ms + ✓ direct .flow.ts input through the built CLI and live runtime > refuses missing and malformed input before contacting relayflowd 1512ms + ✓ direct .flow.ts input through the built CLI and live runtime > does not run the authored body before daemon availability 517ms + ✓ direct .flow.ts input through the built CLI and live runtime > refuses oversized file input before contacting relayflowd 402ms + ✓ tests/cloud-schedule.test.ts (17 tests) 5188ms + ✓ schedule lowering > marks a non-grid cron as Cloud-only rather than approximating it, with a silence budget from its own cadence 3027ms + ✓ flows check prints declared schedules > shows the lowering for a fixed interval and the Cloud-only note for a real cron 1654ms + ✓ tests/cli-hn-monitor.test.ts (16 tests) 95ms + ✓ tests/authored-run-failure-evidence.test.ts (9 tests) 901ms + ✓ the child index after the process that wrote it is gone > still names every child, with its own run id, after a daemon restart 519ms + ✓ tests/daemon-lifecycle-live.test.ts (9 tests) 6736ms + ✓ flows run against a data dir with no daemon (§6 test 7) > cold start spawns exactly one daemon, the run succeeds, and the daemon outlives the CLI 610ms + ✓ flows run against a data dir with no daemon (§6 test 7) > polls, bounded, for a daemon that holds the lock before it binds 1403ms + ✓ flows run against a data dir with no daemon (§6 test 7) > attaches to a serving daemon that has not published a connection file 463ms + ✓ flows run against a data dir with no daemon (§6 test 7) > a second run attaches to the daemon the first one started, spawning nothing 877ms + ✓ flows run against a data dir with no daemon (§6 test 7) > detects a stale connection file left by a hard kill and starts a fresh daemon 942ms + ✓ concurrent invocations against one empty data dir (§6 test 15) > ends with exactly one daemon owning the socket, and both runs succeed 1088ms + ✓ refusals from a spawn that cannot produce a daemon > names relayflowd_not_found rather than falling through to PATH 400ms + ✓ refusals from a spawn that cannot produce a daemon > names daemon_start_failed and quotes the daemon log when startup dies 507ms + ✓ refusals from a spawn that cannot produce a daemon > refuses a daemon speaking another protocol version instead of binding over it 444ms +(node:39850) Warning: Transcript tail for run-9/analyze attempt 1 (stdout) could not be written; the step continues without it: EACCES: permission denied, mkdir '/tmp/transcript-tail-Nz5vPK/runs/run-9/steps' +(Use `node --trace-warnings ...` to show where the warning was created) + ✓ tests/transcript-tail.test.ts (11 tests) 720ms + ✓ direct agent spawn > tees stdout and stderr into tail files that name the dispatch 313ms + ✓ direct agent spawn > completes the step when the tail directory cannot be created 325ms + ✓ tests/authored-agent-artifacts.test.ts (4 tests) 400ms + ✓ tests/authored-helpers.test.ts (6 tests) 3116ms + ✓ runs every available provider through the real kernel and resumes completed effects without a second write 1616ms + ✓ replays after SIGKILL before confirm with the same token and one successful completion 515ms + ✓ replays after SIGKILL before complete with the same token and one successful completion 525ms + ✓ tests/authored-flow-operation.test.ts (24 tests) 534ms + ✓ tests/flow-requirements.test.ts (14 tests) 515ms + ✓ flows check prints REQUIRES > names the helper, the harness and the mcp server of an authored flow 307ms + ✓ tests/backlog-picker.test.ts (14 tests) 41ms + ✓ tests/plugin-store-bounds.test.ts (11 tests) 76ms + ✓ tests/backlog-picker-flow.test.ts (6 tests) 265ms + ✓ tests/hosted-extension-protocol-intrinsics.test.ts (6 tests) 13ms + ✓ tests/preflight-permissions-unenforced.test.ts (17 tests) 256ms + ✓ tests/wrapper-exit-drain.test.ts (8 tests) 2604ms + ✓ reports a signalled wrapper death while a pipe is held, with its output intact 381ms + ✓ lets a lease abort outrank a successful exit still being drained 532ms + ✓ tests/stuck-run-triage.test.ts (22 tests) 3105ms + ✓ stuck-run-triage shell text > collects tails with no GNU timeout on PATH, as on a stock macOS 3066ms + ✓ tests/worker-transcript.test.ts (5 tests) 202ms + ✓ tests/hosted-extension-routing.test.ts (7 tests) 7ms + ✓ tests/artifact-gates.test.ts (7 tests) 174ms + ✓ tests/webhook.test.ts (9 tests) 422ms + ✓ webhook ingress > checks TS declarations against flows.json without invoking handlers 354ms + ✓ tests/webhook-live.test.ts (6 tests) 9818ms + ✓ executes and deduplicates 'app_mention' only for its provider and matching payload 1475ms + ✓ executes and deduplicates 'reaction_added' only for its provider and matching payload 1457ms + ✓ executes and deduplicates 'pull_request' only for its provider and matching payload 1444ms + ✓ flows serve-webhook writes JSON before the daemon starts, then journals and archives exactly once 1480ms + ✓ replays a dropped file after SIGKILL before spawn 476ms + ✓ resumes the same journal after SIGKILL after spawn and before acknowledgement 3486ms + ✓ tests/agent-transcript-live.test.ts (4 tests) 40535ms + ✓ the transcript digest through the built CLI, a real daemon and the local agent > preserves structured agent failure details and its completed root index 14559ms + ✓ the transcript digest through the built CLI, a real daemon and the local agent > preserves structured llm failure details and its completed root index 12506ms + ✓ the transcript digest through the built CLI, a real daemon and the local agent > journals the digest in trajectory_tail on a successful agent step and writes the file it points at 786ms + ✓ the transcript digest through the built CLI, a real daemon and the local agent > on a failed agent step, names the failure and the transcript in the terminal diagnostic, redacted 12684ms + ✓ tests/agent-artifacts-live.test.ts (5 tests) 43084ms + ✓ agent artifacts and gates through the built CLI, a real daemon and the local agent > journals the files the agent wrote, including under a dot-directory, and every artifact gate passes on that journal 1426ms + ✓ agent artifacts and gates through the built CLI, a real daemon and the local agent > fails the run when the artifact_exists gate names a file the agent did not write 12592ms + ✓ agent artifacts and gates through the built CLI, a real daemon and the local agent > fails the run with the author reason when a predicate gate returns false, journaling the verdict 13726ms + ✓ review follow-ups > applies a predicate gate on a helper step too, and journals its verdict 14639ms + ✓ review follow-ups > records predicate verdicts on the root run so a resume reuses them instead of re-running the closure 701ms + ✓ tests/authored-step-failed.test.ts (10 tests) 39ms + ✓ tests/human-live.test.ts (3 tests) 6549ms + ✓ f.human against a real daemon > parks with the question, refuses wrong answers, records one, and resumes to success 3914ms + ✓ f.human against a real daemon > a "no" is a value the body branches on: declined, exit 0, no effect 1630ms + ✓ f.human against a real daemon > refuses to answer a run the daemon does not know 1004ms + ✓ tests/budget-preflight.test.ts (25 tests) 18ms + ✓ tests/cli-watch.test.ts (10 tests) 16227ms + ✓ flows check --watch > rechecks syntax errors, clears once, and returns the last refusal on Ctrl-C 1339ms + ✓ flows check --watch > streams JSON lines without ANSI, recovers after atomic saves, and exits zero after repair 1866ms + ✓ flows check --watch > coalesces 20 concurrent saves into at most two rechecks 1833ms + ✓ flows check --watch > watches transitive relative use imports, cycles, and nearest config changes 2533ms + ✓ flows check --watch > refreshes the import graph and notices missing imports being created 2461ms + ✓ flows check --watch > reloads authored TypeScript instead of reusing the first imported definition 1532ms + ✓ flows check --watch > detects a nearer config appearing and falls back after it is deleted 1928ms + ✓ flows check --watch > keeps watching after the target is deleted and recreated 1959ms + ✓ flows check --watch > queues changes during a slow check without overlapping checks 773ms + ✓ tests/budget-unmetered-live.test.ts (3 tests) 965ms + ✓ unmetered budget spend through the live kernel > runs an unpriced step under a dollar budget without tripping it, journaling unknown dollars 484ms + ✓ tests/provider-trigger-contract.test.ts (7 tests) 533ms + ✓ provider trigger contract > fails `flows check` before deployment and passes once the event is real 346ms + ✓ tests/work-package-consumer.test.ts (13 tests) 115ms + ✓ tests/spec-parity.test.ts (31 tests) 351ms + ✓ tests/helpers-fanout.test.ts (96 tests) 154ms +(node:42697) ExperimentalWarning: SQLite is an experimental feature and might change at any time +(Use `node --trace-warnings ...` to show where the warning was created) + ✓ tests/authored-step-graph-live.test.ts (1 test) 666ms + ✓ the authored step DAG through the live kernel > carries labels and predecessors on every index record and journal step, ids unchanged 665ms + ✓ tests/generate-triggers.test.ts (7 tests) 1091ms + ✓ discovers new adapters, preserves exact event names, and prefers adapter-local mappings 347ms + ✓ tests/authored-parallel-agents.test.ts (8 tests) 11697ms + ✓ authored steps under local workers with capacity > runs more concurrent f.llm calls than the worker holds side by side, never more than its capacity 1223ms + ✓ authored steps under local workers with capacity > completes more concurrent f.agent calls than the worker holds: the overflow waits for a slot instead of parking 2531ms + ✓ authored steps under local workers with capacity > runs agents in distinct working directories side by side (the kernel carries cwd) 663ms + ✓ authored steps under local workers with capacity > serializes agents whose cwd is a symlink alias of the same directory 1068ms + ✓ authored steps under local workers with capacity > serializes agents whose cwd is a directory nested inside the other 1067ms + ✓ authored steps under local workers with capacity > never starts queued agents once the body has failed 2053ms + ✓ authored steps under local workers with capacity > never starts a queued agent when the agent holding the only slot fails 2074ms + ✓ authored steps under local workers with capacity > parks the overflow when the body is not told the capacity (the defect this closes) 1016ms + ✓ tests/webhook-hardening.test.ts (11 tests) 55ms + ✓ tests/human-to.test.ts (8 tests) 9ms + ✓ tests/plugin-loader.test.ts (9 tests) 193ms + ✓ tests/pty-sidechannel.test.ts (11 tests) 5689ms + ✓ view attach preserves worker completion and marks only drive 768ms + ✓ drive attach preserves worker completion and marks only drive 314ms + ✓ passthrough attach preserves worker completion and marks only drive 864ms + ✓ none attach preserves worker completion and marks only drive 785ms + ✓ none subscriber lets an unattended CLI read EOF 442ms + ✓ view subscriber lets an unattended CLI read EOF 404ms + ✓ passthrough subscriber lets an unattended CLI read EOF 403ms + ✓ incomplete subscriber lets an unattended CLI read EOF 397ms + ✓ rejects drive after EOF without marking human intervention 717ms + ✓ delivers all drive bytes in order across child stdin backpressure 592ms + ✓ tests/worker-lease.test.ts (7 tests) 9ms + ✓ tests/yaml-helpers.test.ts (33 tests) 70ms + ✓ tests/authored-agent-permissions.test.ts (26 tests) 739ms + ✓ tests/deploy.test.ts (11 tests) 5207ms + ✓ flows deploy file buckets > publishes the full signed layout byte-for-byte and redeploys as a noop 831ms + ✓ flows deploy file buckets > answers --json with one object per outcome 797ms + ✓ flows deploy file buckets > reports a refusal as JSON under --json 388ms + ✓ flows deploy file buckets > refuses a missing local bundle before creating the bucket 387ms + ✓ flows deploy file buckets > refuses an unreachable bucket before copying 401ms + ✓ flows deploy file buckets > refuses an unwritable bucket 417ms + ✓ flows deploy file buckets > refuses local tampering of spec.canonical.json 394ms + ✓ flows deploy file buckets > refuses local tampering of identity.json 381ms + ✓ flows deploy file buckets > refuses asset bundles instead of using daemon-relative files 383ms + ✓ flows deploy file buckets > never labels a corrupt existing deployment as a noop 807ms + ✓ tests/babysitter-catalog-export.test.ts (14 tests) 825ms + ✓ Babysitter catalog artifact export > CLI refuses an existing output and leaves no file on validation failure 688ms + ✓ tests/redact.test.ts (35 tests) 7ms + ✓ tests/communication.test.ts (10 tests) 13ms + ✓ tests/worker-lease-lost.test.ts (16 tests) 17ms + ✓ tests/typed-output.test.ts (14 tests) 199ms + ✓ tests/canonical-software-factory.test.ts (3 tests) 102ms + ✓ tests/budget-attribution.test.ts (5 tests) 7ms + ✓ tests/worker-cli-result-exit.test.ts (5 tests) 32889ms + ✓ a Claude agent step completes on its result, not only on process exit > settles a hung, successful run within the grace and stops its whole tree 31615ms + ✓ a Claude agent step completes on its result, not only on process exit > maps an error result on a hung run to a failed exit 31615ms + ✓ a Claude agent step completes on its result, not only on process exit > leaves a hang before any result to the existing stops 32012ms + ✓ an agent tree does not outlive the process that spawned it > kills the agent group when the run process is terminated by SIGTERM 795ms + ✓ tests/effect-channel.test.ts (5 tests) 347ms + ✓ tests/mcp-lifecycle.test.ts (4 tests) 14ms + ✓ tests/model-selection.test.ts (10 tests) 17ms + ✓ tests/json-schema-bound.test.ts (71 tests) 2357ms + ✓ JSON Schema termination bound > walks a deep schema with an explicit stack rather than recursion 1925ms + ✓ tests/relayflowd-path.test.ts (10 tests) 6ms + ✓ tests/agent-artifacts.test.ts (9 tests) 17ms + ✓ tests/f-memory.test.ts (7 tests) 869ms + ✓ tests/authored-plugin-effect.test.ts (6 tests) 61ms + ✓ tests/yaml-local-agent-live.test.ts (7 tests) 4049ms + ✓ YAML --local-agent through the built CLI and real daemon > runs with the checked step CLI and model and journals done 633ms + ✓ YAML --local-agent through the built CLI and real daemon > runs with the checked named CLI and model and journals done 609ms + ✓ YAML --local-agent through the built CLI and real daemon > runs with the checked flow CLI and model and journals done 598ms + ✓ YAML --local-agent through the built CLI and real daemon > runs with the checked project CLI and model and journals done 578ms + ✓ YAML --local-agent through the built CLI and real daemon > still parks without --local-agent 541ms + ✓ YAML --local-agent through the built CLI and real daemon > reports the agent process failure 564ms + ✓ YAML --local-agent through the built CLI and real daemon > preserves declared workspace surfaces that the local worker cannot pin 525ms + ✓ tests/worker-slots.test.ts (7 tests) 6ms + ✓ tests/local-dev-ux.test.ts (8 tests) 16ms + ✓ tests/relay-cli-surface-live.test.ts (3 tests) 387ms + ✓ tests/authored-declined.test.ts (13 tests) 50ms + ✓ tests/resume-failure.test.ts (2 tests) 7ms + ✓ tests/dependency-validation.test.ts (6 tests) 721ms + ✓ dependency validation > accepts a valid 10,000-step reverse chain through every direct public boundary 374ms + ✓ tests/worker-lease-lost-live.test.ts (3 tests) 826ms + ✓ tests/authored-hooks.test.ts (5 tests) 5ms + ✓ tests/input-binding.test.ts (12 tests) 216ms + ✓ tests/communication-review.test.ts (5 tests) 323ms + ✓ tests/yaml-helper-effect.test.ts (4 tests) 78ms + ✓ tests/deterministic-llm.test.ts (5 tests) 51ms + ✓ tests/scope-preflight.test.ts (6 tests) 11ms + ✓ tests/bin.test.ts (7 tests) 2473ms + ✓ built flows binary > refuses through a symlink to the built artifact 387ms + ✓ built flows binary > refuses through a symlinked directory component 392ms + ✓ built flows binary > classifies a signal-terminated auth probe as probe_failed 502ms + ✓ built flows binary > classifies an unavailable PATH resolver as probe_failed 389ms + ✓ built flows binary > does not describe a present non-executable CLI as missing 385ms + ✓ built flows binary > runs one auth probe for three steps sharing a flow CLI 416ms + ✓ tests/build-gate.test.ts (3 tests) 1153ms + ✓ flows build gates on flows check green (#318) > refuses a flow with an unresolvable named-agent CLI and leaves no artifacts 390ms + ✓ flows build gates on flows check green (#318) > --json emits one CheckReport object on stdout on refusal, exits 2, no artifacts 385ms + ✓ flows build gates on flows check green (#318) > builds the bundle on success (regression: gate must not block valid flows) 377ms + ✓ tests/scope-compiler.test.ts (25 tests) 10ms + ✓ tests/run-from-digest.test.ts (6 tests) 4507ms + ✓ flows run digest input > submits the sealed canonical spec through the normal journal path without checkout 453ms + ✓ flows run digest input > uses a verified cache hit even after the bucket is removed 403ms + ✓ flows run digest input > resolves deploy.bucket from flows.json and honors explicit override 1190ms + ✓ flows run digest input > refuses an unconfigured bucket 816ms + ✓ flows run digest input > refuses tampered spec.canonical.json before creating run data 835ms + ✓ flows run digest input > refuses tampered identity.json before creating run data 809ms + ✓ tests/communication-worker.test.ts (15 tests) 1491ms + ✓ tests/hn-poller.test.ts (6 tests) 6ms + ✓ tests/plugin-add.test.ts (7 tests) 1207ms + ✓ typechecks the augmented verb and rejects unknown namespaces 896ms + ✓ tests/authored-step-failed-exit.test.ts (3 tests) 10ms + ✓ tests/direct-run-failure.test.ts (8 tests) 12ms + ✓ tests/dir-watcher-poller.test.ts (6 tests) 6ms + ✓ tests/model-pricing.test.ts (10 tests) 5ms + ✓ tests/yaml-helper-live.test.ts (1 test) 947ms + ✓ runs compiled YAML helpers through the built CLI and kernel effect journal 946ms + ✓ tests/provider-trigger-executor.test.ts (4 tests) 200ms + ✓ tests/transcript-tail-close.test.ts (2 tests) 987ms + ✓ a stalled transcript-tail close > does not hold the spawn open past its bounded window 477ms + ✓ a stalled tail close beside a transcript that finished > still journals the transcript pointer 509ms + ✓ tests/wrapper-artifacts-cwd.test.ts (2 tests) 75ms + ✓ tests/hello-deterministic.test.ts (5 tests) 17ms + ✓ tests/transcript-exclusion-timeout.test.ts (1 test) 190ms + ✓ tests/cli-adapter.test.ts (4 tests) 5ms + ✓ tests/communication-mixed-resume.test.ts (1 test) 177ms + ✓ tests/work-package-validator.test.ts (7 tests) 5ms + ✓ tests/authored-use-loader.test.ts (5 tests) 656ms + ✓ tests/authored-declined-live.test.ts (1 test) 1681ms + ✓ runs an input guard and resumes its completed declined root without repeated effects 1680ms + ✓ tests/cli-answer.test.ts (15 tests) 13ms + ✓ tests/bundle-preflight.test.ts (4 tests) 941ms + ✓ bundle execution preflight > ignores surrounding cache configuration on a verified cache hit 457ms + ✓ bundle execution preflight > uses the built alias for a nameless flow even in a digest-only cache directory 467ms + ✓ tests/agent-relay-hardening.test.ts (12 tests) 16ms + ✓ tests/classify-outcome.test.ts (2 tests) 2163ms + ✓ classifyOutcome > gives up and reports when a running run never becomes classifiable 2009ms + ✓ tests/communication-preflight.test.ts (13 tests) 33ms + ↓ tests/real-cli-adapters.test.ts (3 tests | 3 skipped) + ✓ tests/memoization.test.ts (57 tests) 54ms + ✓ tests/fs-descriptor.test.ts (1 test) 5ms + ✓ tests/parse-json-output.test.ts (7 tests) 3ms + ✓ tests/journal-client-completion.test.ts (4 tests) 100ms + ✓ tests/worker-cli-abort.test.ts (2 tests) 2518ms + ✓ stops claude and its process group when lease ownership is lost 1252ms + ✓ stops wrapper.mjs and its process group when lease ownership is lost 1264ms + ✓ tests/communication-environment-preflight.test.ts (6 tests) 4ms + ✓ tests/budget-authored-live.test.ts (2 tests) 155ms + ✓ tests/slack-writeback.test.ts (1 test) 260ms + ✓ tests/authored-surface-authority.test.ts (2 tests) 15ms + ✓ tests/adapters/claude.test.ts (7 tests) 4ms + ✓ tests/worker-cli-cwd.test.ts (2 tests) 268ms + ✓ tests/adapters/codex.test.ts (7 tests) 4ms + ✓ tests/slack-block-kit.test.ts (5 tests) 369ms + ✓ tests/worker-lease-sweep.test.ts (2 tests) 7ms + ✓ tests/step-lease.test.ts (36 tests) 66399ms + ✓ f.run leases against the live kernel > enforces 10000 ms for 'sleep 5; printf ok' 5061ms + ✓ f.run leases against the live kernel > enforces 40000 ms for 'sleep 31; printf ok' 31057ms + ✓ f.run leases against the live kernel > enforces 30000 ms for 'sleep 31; printf ok' 30086ms + ✓ tests/communication-history.test.ts (1 test) 4ms + ✓ tests/adapters/registry.test.ts (4 tests) 5ms + ✓ tests/resume-worker-lease.test.ts (2 tests) 5ms + ✓ tests/direct-run-worker-lease.test.ts (2 tests) 6ms + ✓ tests/authored-declined-report.test.ts (6 tests) 7ms + ✓ tests/promise-ancestry.test.ts (2 tests) 290ms + ✓ tests/communication-refusal.test.ts (1 test) 12ms + ✓ tests/agent-cwd-validation.test.ts (2 tests) 411ms + ✓ declarative agent cwd > is refused by `flows check` on a YAML flow before anything runs 408ms + ✓ tests/catalog-plugins.test.ts (2 tests) 3ms + ✓ tests/check-command-cwd.test.ts (1 test) 21ms + ✓ tests/communication-lazy.test.ts (1 test) 4ms + ✓ tests/cli-progress-wait.test.ts (2 tests) 4ms + ✓ tests/bundle-transport.test.ts (20 tests) 2687ms + ✓ digest references > accepts and deploys the build output for hello 486ms + ✓ digest references > accepts and deploys the build output for Hello 476ms + ✓ digest references > accepts and deploys the build output for hello.world 444ms + ✓ digest references > accepts and deploys the build output for hello_world 432ms + ✓ digest references > accepts and deploys the build output for 123 404ms + ✓ digest references > accepts and deploys the build output for A_b.c-1 442ms + ✓ tests/placement.test.ts (54 tests) 19ms + ✓ tests/canonical-tree.test.ts (1 test) 2ms + ✓ tests/run-digest-live.test.ts (1 test) 927ms + ✓ executes a deployed digest on the real kernel after deleting the authoring tree 926ms + ✓ tests/communication-tools.test.ts (1 test) 76ms + ✓ tests/authored-admission.test.ts (2 tests) 3ms + ✓ tests/memory.test.ts (18 tests) 11ms + ✓ tests/worker-platform.test.ts (1 test) 3ms + ✓ tests/run-digest.test.ts (4 tests) 1666ms + ✓ digest run configuration refusals > reports config_invalid before fetching or starting a run for {invalid json 429ms + ✓ digest run configuration refusals > reports config_invalid before fetching or starting a run for {"deploy":{}} 414ms + ✓ digest run configuration refusals > reports config_invalid before fetching or starting a run for {"deploy":{"bucket":123}} 384ms + ✓ digest run configuration refusals > reports config_invalid before fetching or starting a run for {"deploy":{"bucket":""}} 438ms + ✓ tests/local-agent-live.test.ts (5 tests) 64242ms + ✓ built CLI local agent against a real daemon > dispatches through the wrapper and keeps --json stdout report-shaped 781ms + ✓ built CLI local agent against a real daemon > runs beyond the initial 30-second lease without a second invocation 35761ms + ✓ built CLI local agent against a real daemon > renders actual agent completion in text output 786ms + ✓ built CLI local agent against a real daemon > returns a failed run when the agent process fails 14863ms + ✓ built CLI local agent against a real daemon > refuses a workspace it cannot pin before invoking the agent 12050ms + +⎯⎯⎯⎯⎯⎯ Failed Suites 1 ⎯⎯⎯⎯⎯⎯⎯ + + FAIL tests/authored-node-runtime.test.ts [ tests/authored-node-runtime.test.ts ] +AssertionError: expected '1.3.6' to be '1.4.0' // Object.is equality + +Expected: "1.4.0" +Received: "1.3.6" + + ❯ tests/authored-node-runtime.test.ts:18:77 + 16| + 17| beforeAll(() => { + 18| expect(spawnSync(bun, ['--version'], { encoding: 'utf8' }).stdout.tr… + | ^ + 19| expect(existsSync(daemon), 'build the current kernel or set RELAYFLO… + 20| stage = mkdtempSync(join(tmpdir(), 'authored-standalone-build-')); + +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯[1/31]⎯ + +⎯⎯⎯⎯⎯⎯ Failed Tests 30 ⎯⎯⎯⎯⎯⎯⎯ + + FAIL tests/babysitter-native-extension.test.ts > native Babysitter extension > runs the exact published 2.0.26 native bytes in the isolated capability path +AssertionError: promise rejected "Error: Hosted extension sandbox exited wi… { code: '…' }" instead of resolving + ❯ tests/babysitter-native-extension.test.ts:153:7 + 151| return { receiptId: `bst_${'a'.repeat(64)}`, status: 'queued' … + 152| } }, + 153| })).resolves.toEqual({ completionReason: 'success', capabilityCall… + | ^ + 154| expect(calls).toEqual([{ delivery: { + 155| deliveryId: 'gh-delivery-7', provider: 'github', eventType: 'pul… + +Caused by: Error: Hosted extension sandbox exited without a valid completion (exit 1): bwrap: loopback: Failed RTM_NEWADDR: Operation not permitted + + ❯ refuse src/hosted-extension-protocol.ts:135:21 + ❯ ChildProcess. src/hosted-extension-protocol.ts:234:21 + +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯ +Serialized Error: { code: 'plugin_unsupported' } +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯[2/31]⎯ + + FAIL tests/hosted-extension-isolation.test.ts > hosted extension capability isolation > executes the exact capability-only handler for a queued receipt + FAIL tests/hosted-extension-protocol.test.ts > hosted extension hostile protocol > uses captured JSON intrinsics for the complete parent boundary +Error: Hosted extension sandbox exited without a valid completion (exit 1): bwrap: loopback: Failed RTM_NEWADDR: Operation not permitted + + ❯ refuse src/hosted-extension-protocol.ts:135:21 + 133| : new PluginError('plugin_unsupported', 'Hosted capability rejec… + 134| const refuse = (message: string) => { + 135| const error = new PluginError('plugin_unsupported', message); + | ^ + 136| CHILD_PROCESS_KILL(child, 'SIGKILL'); + 137| if (capabilityState === 'pending') { + ❯ ChildProcess. src/hosted-extension-protocol.ts:234:21 + +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯[3/31]⎯ + + FAIL tests/hosted-extension-isolation.test.ts > hosted extension capability isolation > executes the exact capability-only handler for a duplicate receipt + FAIL tests/hosted-extension-isolation.test.ts > hosted extension capability isolation > denies ambient credentials, host files, writes, network, subprocesses, and undeclared context verbs + FAIL tests/hosted-extension-isolation.test.ts > hosted extension capability isolation > constructs adapter authority with the captured freeze intrinsic +Error: Hosted extension sandbox exited without a valid completion (exit 1): bwrap: loopback: Failed RTM_NEWADDR: Operation not permitted + + ❯ refuse src/hosted-extension-protocol.ts:135:21 + 133| : new PluginError('plugin_unsupported', 'Hosted capability rejec… + 134| const refuse = (message: string) => { + 135| const error = new PluginError('plugin_unsupported', message); + | ^ + 136| CHILD_PROCESS_KILL(child, 'SIGKILL'); + 137| if (capabilityState === 'pending') { + ❯ ChildProcess. src/hosted-extension-protocol.ts:234:21 + +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯[4/31]⎯ + + FAIL tests/hosted-extension-isolation.test.ts > hosted extension capability isolation > launches through the captured process primitive after builtin export synchronization +AssertionError: promise rejected "Error: Hosted extension sandbox exited wi… { code: '…' }" instead of resolving + ❯ tests/hosted-extension-isolation.test.ts:283:11 + 281| input: descriptor(), + 282| babysitterTurn: { queue: async () => ({ receiptId: 'receipt-… + 283| })).resolves.toEqual({ completionReason: 'success', capability… + | ^ + 284| } finally { + 285| process.execPath = originalExecPath; + +Caused by: Error: Hosted extension sandbox exited without a valid completion (exit 1): bwrap: loopback: Failed RTM_NEWADDR: Operation not permitted + + ❯ refuse src/hosted-extension-protocol.ts:135:21 + ❯ ChildProcess. src/hosted-extension-protocol.ts:234:21 + +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯ +Serialized Error: { code: 'plugin_unsupported' } +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯[5/31]⎯ + + FAIL tests/hosted-extension-isolation.test.ts > hosted extension capability isolation > ignores inherited launcher overrides and decodes manifests with the captured Buffer intrinsic +AssertionError: promise rejected "Error: Hosted extension sandbox exited wi… { code: '…' }" instead of resolving + ❯ tests/hosted-extension-isolation.test.ts:373:11 + 371| input: descriptor('delivery-options'), + 372| babysitterTurn: { queue: async () => ({ receiptId: 'receipt-… + 373| })).resolves.toEqual({ completionReason: 'success', capability… + | ^ + 374| } finally { + 375| Buffer.prototype.toString = bufferToString; + +Caused by: Error: Hosted extension sandbox exited without a valid completion (exit 1): bwrap: loopback: Failed RTM_NEWADDR: Operation not permitted + + ❯ refuse src/hosted-extension-protocol.ts:135:21 + ❯ ChildProcess. src/hosted-extension-protocol.ts:234:21 + +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯ +Serialized Error: { code: 'plugin_unsupported' } +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯[6/31]⎯ + + FAIL tests/hosted-extension-isolation.test.ts > hosted extension capability isolation > streams verified bytes when the live store is replaced and no writable staging path exists +AssertionError: promise rejected "Error: Hosted extension sandbox exited wi… { code: '…' }" instead of resolving + ❯ tests/hosted-extension-isolation.test.ts:431:7 + 429| return { receiptId: 'receipt-1', status: 'queued' }; + 430| } }, + 431| })).resolves.toEqual({ completionReason: 'success', capabilityCall… + | ^ + 432| expect(calls).toBe(1); + 433| expect(readFileSync(join(installed.directory, 'babysitter.flow.ts'… + +Caused by: Error: Hosted extension sandbox exited without a valid completion (exit 1): bwrap: loopback: Failed RTM_NEWADDR: Operation not permitted + + ❯ refuse src/hosted-extension-protocol.ts:135:21 + ❯ ChildProcess. src/hosted-extension-protocol.ts:234:21 + +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯ +Serialized Error: { code: 'plugin_unsupported' } +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯[7/31]⎯ + + FAIL tests/hosted-extension-isolation.test.ts > hosted extension capability isolation > mounts pinned private Surface bytes when the live package changes before launch +AssertionError: promise rejected "Error: Hosted extension sandbox exited wi… { code: '…' }" instead of resolving + ❯ tests/hosted-extension-isolation.test.ts:456:7 + 454| return { receiptId: 'receipt-1', status: 'queued' }; + 455| } }, + 456| })).resolves.toEqual({ completionReason: 'success', capabilityCall… + | ^ + 457| expect(calls).toBe(1); + 458| expect(readFileSync(join(surfaceRoot, 'dist/flow.js'), 'utf8')).to… + +Caused by: Error: Hosted extension sandbox exited without a valid completion (exit 1): bwrap: loopback: Failed RTM_NEWADDR: Operation not permitted + + ❯ refuse src/hosted-extension-protocol.ts:135:21 + ❯ ChildProcess. src/hosted-extension-protocol.ts:234:21 + +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯ +Serialized Error: { code: 'plugin_unsupported' } +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯[8/31]⎯ + + FAIL tests/hosted-extension-isolation.test.ts > hosted extension capability isolation > shields verified Surface files before async settlement +AssertionError: promise rejected "Error: Hosted extension sandbox exited wi… { code: '…' }" instead of resolving + ❯ tests/hosted-extension-isolation.test.ts:532:9 + 530| surfaceRoot, + 531| babysitterTurn: { queue: async () => ({ receiptId: 'receipt-1'… + 532| })).resolves.toEqual({ completionReason: 'success', capabilityCa… + | ^ + 533| } finally { + 534| if (previous === undefined) delete (Array.prototype as { then?: … + +Caused by: Error: Hosted extension sandbox exited without a valid completion (exit 1): bwrap: loopback: Failed RTM_NEWADDR: Operation not permitted + + ❯ refuse src/hosted-extension-protocol.ts:135:21 + ❯ ChildProcess. src/hosted-extension-protocol.ts:234:21 + +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯ +Serialized Error: { code: 'plugin_unsupported' } +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯[9/31]⎯ + + FAIL tests/hosted-extension-isolation.test.ts > hosted extension capability isolation > preserves a typed host refusal while disclosing only a fixed marker to the child +AssertionError: expected Error: Hosted extension sandbox exited wi… { code: '…' } to be Error: private Cloud policy detail { code: '…' } // Object.is equality + +- Expected ++ Received + +- [Error: private Cloud policy detail] ++ [Error: Hosted extension sandbox exited without a valid completion (exit 1): bwrap: loopback: Failed RTM_NEWADDR: Operation not permitted ++ ] + + ❯ tests/hosted-extension-isolation.test.ts:560:5 + 558| provider: 'github', eventType: 'pull_request.labeled', deliveryI… + 559| }); + 560| await expect(runVerifiedNativeExtensionSandbox({ + | ^ + 561| artifact: await artifact(source), manifest: validateFlowExtensio… + 562| babysitterTurn: { queue: async () => { throw refusal; } }, + +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯[10/31]⎯ + + FAIL tests/hosted-extension-isolation.test.ts > hosted extension capability isolation > enforces OS address-space and data bounds on native Buffer allocation +AssertionError: expected Error: Hosted extension sandbox exited wi… { code: '…' } to match object { code: 'plugin_unsupported', …(1) } + +- Expected ++ Received + +- Object { ++ PluginError { + "code": "plugin_unsupported", +- "message": StringMatching /(?:Failed to allocate memory|Array buffer allocation failed)/u, + } + + ❯ tests/hosted-extension-isolation.test.ts:646:5 + 644| }); + 645| let calls = 0; + 646| await expect(runVerifiedNativeExtensionSandbox({ + | ^ + 647| artifact: installed, + 648| manifest: validateFlowExtensionManifest(manifest()), + +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯[11/31]⎯ + + FAIL tests/hosted-extension-isolation.test.ts > hosted extension capability isolation > blocks extra handler fields and authority-bearing receipt fields at the parent port +AssertionError: expected Error: Hosted extension sandbox exited wi… { code: '…' } to match object { code: 'plugin_event_unroutable' } + +- Expected ++ Received + +- Object { +- "code": "plugin_event_unroutable", ++ PluginError { ++ "code": "plugin_unsupported", + } + + ❯ tests/hosted-extension-isolation.test.ts:675:5 + 673| }); + 674| let calls = 0; + 675| await expect(runVerifiedNativeExtensionSandbox({ + | ^ + 676| artifact: await artifact(source), manifest: validateFlowExtensio… + 677| babysitterTurn: { queue: async () => { calls += 1; return { rece… + +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯[12/31]⎯ + + FAIL tests/hosted-extension-isolation.test.ts > hosted extension capability isolation > writes the Surface manifest and protocol without inherited toJSON behavior +AssertionError: promise rejected "Error: Hosted extension sandbox exited wi… { code: '…' }" instead of resolving + ❯ tests/hosted-extension-isolation.test.ts:788:11 + 786| babysitterTurn: { queue: async () => ({ receiptId: 'receipt-… + 787| timeoutMs: 3_000, + 788| })).resolves.toEqual({ completionReason: 'success', capability… + | ^ + 789| } finally { + 790| if (previous === undefined) delete (Object.prototype as { toJS… + +Caused by: Error: Hosted extension sandbox exited without a valid completion (exit 1): bwrap: loopback: Failed RTM_NEWADDR: Operation not permitted + + ❯ refuse src/hosted-extension-protocol.ts:135:21 + ❯ ChildProcess. src/hosted-extension-protocol.ts:234:21 + +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯ +Serialized Error: { code: 'plugin_unsupported' } +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯[13/31]⎯ + + FAIL tests/hosted-extension-protocol.test.ts > hosted extension hostile protocol > rejects an import-time different PR frame with zero adapter calls + FAIL tests/hosted-extension-protocol.test.ts > hosted extension hostile protocol > rejects an import-time different delivery frame with zero adapter calls + FAIL tests/hosted-extension-protocol.test.ts > hosted extension hostile protocol > rejects an import-time different event frame with zero adapter calls +AssertionError: expected Error: Hosted extension sandbox exited wi… { code: '…' } to match object { code: 'plugin_event_unroutable' } + +- Expected ++ Received + +- Object { +- "code": "plugin_event_unroutable", ++ PluginError { ++ "code": "plugin_unsupported", + } + + ❯ tests/hosted-extension-protocol.test.ts:430:5 + 428| ])('rejects an import-time %s frame with zero adapter calls', async … + 429| let calls = 0; + 430| await expect(runVerifiedNativeExtensionSandbox({ + | ^ + 431| artifact: await artifact(hostileImport([frame, { type: 'error', … + 432| manifest: validateFlowExtensionManifest(manifest()), dispatch: d… + +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯[14/31]⎯ + + FAIL tests/hosted-extension-protocol.test.ts > hosted extension hostile protocol > rejects two forged calls after the authoritative first outcome settles + FAIL tests/hosted-extension-protocol.test.ts > hosted extension hostile protocol > waits for a pending adapter to reject after a forged child error + FAIL tests/hosted-extension-protocol.test.ts > hosted extension hostile protocol > waits for a pending adapter to resolve after a forged child error + FAIL tests/hosted-extension-protocol.test.ts > hosted extension hostile protocol > returns a typed adapter rejection even when the hostile child hangs +Error: hostile child did not invoke the adapter + ❯ Timeout._onTimeout tests/hosted-extension-protocol.test.ts:118:45 + 116| async function waitForInvocation(invoked: Promise): Promise((resolve, reject) => { + 118| const timeout = setTimeout(() => reject(new Error('hostile child d… + | ^ + 119| void invoked.then(() => { clearTimeout(timeout); resolve(); }, rej… + 120| }); + +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯[15/31]⎯ + + FAIL tests/live-kernel.test.ts > built flows CLI against live relayflowd > runs hn-monitor analyze-story end-to-end via a stub agent CLI (gate 2 clause 2 demo) +AssertionError: expected { …(12) } to match object { output: { …(3) }, …(1) } +(22 matching properties omitted from actual) + +- Expected ++ Received + + Object { +- "output": Object { +- "reasoning": "stub agent runtime — deterministic output for gate-2 clause-2 demo", +- "relevance_score": 5, +- "story_title": "stub", +- }, ++ "output": null, + "verification": Object { +- "gate": "json_schema", +- "verdict": "pass", ++ "gate": "execution", ++ "verdict": "fail", + }, + } + + ❯ tests/live-kernel.test.ts:657:36 + 655| && (entry as { step_id?: string }).step_id === 'analyze-story', + 656| ) as { payload: { output: unknown; verification: unknown } } | und… + 657| expect(stepCompleted?.payload).toMatchObject({ + | ^ + 658| output: { + 659| story_title: 'stub', + +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯[16/31]⎯ + + FAIL tests/live-kernel.test.ts > built flows CLI against live relayflowd > hn-monitor analyze-story FAILS verification when the CLI omits required schema fields +AssertionError: expected { …(12) } to match object { …(3) } +(21 matching properties omitted from actual) + +- Expected ++ Received + + Object { +- "completionReason": "retries_exhausted", ++ "completionReason": "worker_error", + "output": null, + "verification": Object { +- "gate": "json_schema", ++ "gate": "execution", + "verdict": "fail", + }, + } + + ❯ tests/live-kernel.test.ts:752:36 + 750| // its verification record names the json_schema rejection. The re… + 751| // parsed value is nulled before the completion is persisted. + 752| expect(stepCompleted?.payload).toMatchObject({ + | ^ + 753| completionReason: 'retries_exhausted', + 754| output: null, + +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯[17/31]⎯ + + FAIL tests/live-kernel.test.ts > built flows CLI against live relayflowd > agent step preserves the CliResult wrapper as output when the CLI emits non-JSON text +AssertionError: expected null not to be null + ❯ tests/live-kernel.test.ts:823:24 + 821| // here (parseJsonOutput returned null on non-JSON stdout) and + 822| // these assertions would all fail. + 823| expect(output).not.toBeNull(); + | ^ + 824| expect(output.exit_code).toBe(0); + 825| expect(output.stdout_tail).toContain('looked at the story'); + +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯[18/31]⎯ + + FAIL tests/live-kernel.test.ts > built flows CLI against live relayflowd > AgentWorker exposes wake_context to the CLI via RELAYFLOW_WAKE_CONTEXT env var (real analyzer prerequisite) +TypeError: Cannot read properties of null (reading 'story_title') + ❯ tests/live-kernel.test.ts:891:42 + 889| ) as { payload: { output: { story_title: string; reasoning: string… + 890| expect(stepCompleted).toBeDefined(); + 891| expect(stepCompleted!.payload.output.story_title).toBe(`echoed:${s… + | ^ + 892| expect(stepCompleted!.payload.output.reasoning).toContain(String(s… + 893| + +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯[19/31]⎯ + + FAIL tests/live-kernel.test.ts > built flows CLI against live relayflowd > AgentWorker leaves RELAYFLOW_WAKE_CONTEXT UNSET when the run has no wake_context (undefined-vs-null pin) +TypeError: Cannot read properties of null (reading 'env_present') + ❯ tests/live-kernel.test.ts:958:38 + 956| ) as { payload: { output: { env_present: boolean } } } | undefined; + 957| expect(completed).toBeDefined(); + 958| expect(completed!.payload.output.env_present).toBe(false); + | ^ + 959| + 960| delete process.env.RELAYFLOW_WAKE_CONTEXT; + +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯[20/31]⎯ + + FAIL tests/live-kernel.test.ts > built flows CLI against live relayflowd > AgentWorker leaves RELAYFLOW_MODEL UNSET when the step declares no model +TypeError: Cannot read properties of null (reading 'story_title') + ❯ tests/live-kernel.test.ts:1194:38 + 1192| expect(completed).toBeDefined(); + 1193| // UNSET, not EMPTY and not the leaked parent value. + 1194| expect(completed!.payload.output.story_title).toBe('model:UNSET'); + | ^ + 1195| + 1196| delete process.env.RELAYFLOW_MODEL; + +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯[21/31]⎯ + + FAIL tests/live-kernel.test.ts > built flows CLI against live relayflowd > hn-monitor analyze-story reaches done through the real Claude analyzer CLI +Error: LIVE_ANALYZER_UNAVAILABLE: "/home/daytona/.relayflow-v2-supervisor/durable/repository/testdata/preflight/analyze-story-claude-cli" does not identify as relayflows-agent-cli-v1 — failing because gate-2 acceptance requires the real analyzer to execute. Set RELAYFLOWS_ALLOW_ANALYZER_SKIP=1 only if this run is not gate evidence. + ❯ tests/live-kernel.test.ts:1223:15 + 1221| const notice = `LIVE_ANALYZER_UNAVAILABLE: ${readiness.detail}`; + 1222| if (process.env['RELAYFLOWS_ALLOW_ANALYZER_SKIP'] !== '1') { + 1223| throw new Error( + | ^ + 1224| `${notice} — failing because gate-2 acceptance requires the … + 1225| + 'Set RELAYFLOWS_ALLOW_ANALYZER_SKIP=1 only if this run is … + +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯[22/31]⎯ + + FAIL tests/live-kernel.test.ts > a relayflow can be scheduled: tick source against live relayflowd > a tick spawns a real run whose step reports the SCHEDULED instant +AssertionError: expected null to deeply equal { schedule_id: 'heartbeat-1m', …(3) } + +- Expected: +Object { + "lag_ms": 43000, + "schedule_id": "heartbeat-1m", + "scheduled_for_ms": 1764000000000, + "slot": 29400000, +} + ++ Received: +null + + ❯ tests/live-kernel.test.ts:1665:39 + 1663| // The bound: the run reports the grid instant and its own lag, so… + 1664| // backfilled run can tell it is running for a slot from the past. + 1665| expect(completed!.payload.output).toEqual({ + | ^ + 1666| schedule_id: 'heartbeat-1m', + 1667| slot: 29_400_000, + +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯[23/31]⎯ + +⎯⎯⎯⎯⎯⎯ Unhandled Errors ⎯⎯⎯⎯⎯⎯ + +Vitest caught 1 unhandled error during the test run. +This might cause false positive tests. Resolve unhandled errors to make sure your tests are not affected. + +⎯⎯⎯⎯ Unhandled Rejection ⎯⎯⎯⎯⎯ +Error: Hosted extension sandbox exited without a valid completion (exit 1): bwrap: loopback: Failed RTM_NEWADDR: Operation not permitted + + ❯ refuse src/hosted-extension-protocol.ts:135:21 + 133| : new PluginError('plugin_unsupported', 'Hosted capability rejec… + 134| const refuse = (message: string) => { + 135| const error = new PluginError('plugin_unsupported', message); + | ^ + 136| CHILD_PROCESS_KILL(child, 'SIGKILL'); + 137| if (capabilityState === 'pending') { + ❯ ChildProcess. src/hosted-extension-protocol.ts:234:21 + ❯ ChildProcess.emit node:events:520:22 + ❯ maybeClose node:internal/child_process:1084:16 + ❯ Process.ChildProcess._handle.onexit node:internal/child_process:304:5 + +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯ +Serialized Error: { code: 'plugin_unsupported' } +This error originated in "tests/hosted-extension-protocol.test.ts" test file. It doesn't mean the error was thrown inside the file itself, but while it was running. +The latest test that might've caused the error is "rejects two forged calls after the authoritative first outcome settles". It might mean one of the following: +- The error was thrown, while Vitest was running this test. +- If the error occurred after the test had been completed, this was the last documented test before it was thrown. +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯ + + Test Files 5 failed | 181 passed | 1 skipped (187) + Tests 30 failed | 2866 passed | 17 skipped (2913) + Errors 1 error + Start at 05:43:08 + Duration 230.04s (transform 3.02s, setup 0ms, collect 49.95s, tests 596.27s, environment 24ms, prepare 7.90s) + + +Exit code: 1 diff --git a/evidence/worker-lease-lost/npm-test-initial.txt b/evidence/worker-lease-lost/npm-test-initial.txt new file mode 100644 index 000000000..080fe7bf0 --- /dev/null +++ b/evidence/worker-lease-lost/npm-test-initial.txt @@ -0,0 +1,1370 @@ +$ cd packages/sdk && PATH=/home/daytona/.cargo/bin:$PATH npm test + +> @relayflows/sdk@2.0.29 test +> sh scripts/test.sh + + +> @relayflows/sdk@2.0.29 test:prep +> ( cd ../../kernel && sh ../ops/cargo.sh build ) && ( [ ! -d ../../testdata/preflight ] || find ../../testdata/preflight -name '*-cli' -type f -exec chmod +x {} + ) + + Updating crates.io index + Downloading crates ... + Downloaded clap_lex v1.1.0 + Downloaded bit-set v0.8.0 + Downloaded anstyle-parse v1.0.0 + Downloaded anstyle-query v1.1.5 + Downloaded autocfg v1.5.1 + Downloaded ahash v0.8.12 + Downloaded cfg-if v1.0.4 + Downloaded colorchoice v1.0.5 + Downloaded cpufeatures v0.2.17 + Downloaded crypto-common v0.1.7 + Downloaded heck v0.5.0 + Downloaded block-buffer v0.10.4 + Downloaded fallible-streaming-iterator v0.1.9 + Downloaded generic-array v0.14.7 + Downloaded is_terminal_polyfill v1.70.2 + Downloaded anstream v1.0.0 + Downloaded borrow-or-share v0.2.4 + Downloaded num-cmp v0.1.0 + Downloaded zerofrom-derive v0.1.7 + Downloaded idna_adapter v1.2.2 + Downloaded lazy_static v1.5.0 + Downloaded bit-vec v0.8.0 + Downloaded digest v0.10.7 + Downloaded find-msvc-tools v0.1.11 + Downloaded foldhash v0.1.5 + Downloaded hashlink v0.10.0 + Downloaded utf8parse v0.2.2 + Downloaded thiserror-impl v2.0.20 + Downloaded bytecount v0.6.9 + Downloaded displaydoc v0.2.7 + Downloaded fallible-iterator v0.3.0 + Downloaded num-iter v0.1.46 + Downloaded uuid v1.26.0 + Downloaded zerofrom v0.1.8 + Downloaded stable_deref_trait v1.2.1 + Downloaded version_check v0.9.5 + Downloaded wait-timeout v0.2.1 + Downloaded yoke v0.8.3 + Downloaded uuid-simd v0.8.0 + Downloaded yoke-derive v0.8.2 + Downloaded zmij v1.0.23 + Downloaded vsimd v0.8.0 + Downloaded zerovec-derive v0.11.6 + Downloaded email_address v0.2.9 + Downloaded fluent-uri v0.3.2 + Downloaded itoa v1.0.18 + Downloaded num-integer v0.1.47 + Downloaded rand_core v0.9.5 + Downloaded writeable v0.6.4 + Downloaded anyhow v1.0.104 + Downloaded lock_api v0.4.14 + Downloaded percent-encoding v2.3.2 + Downloaded potential_utf v0.1.6 + Downloaded ref-cast v1.0.27 + Downloaded zerotrie v0.2.5 + Downloaded num v0.4.3 + Downloaded outref v0.5.2 + Downloaded strsim v0.11.1 + Downloaded num-rational v0.4.2 + Downloaded ref-cast-impl v1.0.27 + Downloaded scopeguard v1.2.0 + Downloaded sha2 v0.10.9 + Downloaded synstructure v0.13.2 + Downloaded tinystr v0.8.4 + Downloaded utf8_iter v1.0.4 + Downloaded zerovec v0.11.8 + Downloaded base64 v0.22.1 + Downloaded bitflags v2.13.1 + Downloaded clap v4.6.6 + Downloaded clap_derive v4.6.4 + Downloaded icu_locale_core v2.3.0 + Downloaded litemap v0.8.3 + Downloaded num-complex v0.4.6 + Downloaded once_cell v1.21.4 + Downloaded parking_lot v0.12.5 + Downloaded parking_lot_core v0.9.12 + Downloaded pkg-config v0.3.34 + Downloaded ppv-lite86 v0.2.21 + Downloaded quote v1.0.47 + Downloaded rand_chacha v0.9.0 + Downloaded referencing v0.33.0 + Downloaded shlex v2.0.1 + Downloaded smallvec v1.15.2 + Downloaded thiserror v2.0.20 + Downloaded ulid v1.2.1 + Downloaded getrandom v0.3.4 + Downloaded icu_normalizer_data v2.3.0 + Downloaded icu_provider v2.3.1 + Downloaded num-traits v0.2.19 + Downloaded serde_core v1.0.229 + Downloaded serde_derive v1.0.229 + Downloaded vcpkg v0.2.15 + Downloaded cc v1.4.4 + Downloaded icu_collections v2.3.0 + Downloaded icu_properties v2.3.0 + Downloaded memchr v2.8.3 + Downloaded proc-macro2 v1.0.107 + Downloaded serde v1.0.229 + Downloaded unicode-ident v1.0.24 + Downloaded fancy-regex v0.16.2 + Downloaded fraction v0.15.4 + Downloaded num-bigint v0.4.8 + Downloaded idna v1.1.0 + Downloaded rand v0.9.5 + Downloaded typenum v1.20.1 + Downloaded aho-corasick v1.1.5 + Downloaded icu_properties_data v2.3.0 + Downloaded serde_json v1.0.151 + Downloaded clap_builder v4.6.6 + Downloaded hashbrown v0.15.5 + Downloaded jsonschema v0.33.0 + Downloaded regex v1.13.1 + Downloaded rusqlite v0.37.0 + Downloaded zerocopy v0.8.56 + Downloaded syn v2.0.119 + Downloaded syn v3.0.4 + Downloaded regex-syntax v0.8.11 + Downloaded icu_normalizer v2.3.0 + Downloaded regex-automata v0.4.18 + Downloaded libc v0.2.189 + Downloaded libsqlite3-sys v0.35.0 + Downloaded anstyle v1.0.14 + Downloaded ryu-js v1.0.3 + Compiling proc-macro2 v1.0.107 + Compiling unicode-ident v1.0.24 + Compiling quote v1.0.47 + Compiling libc v0.2.189 + Compiling stable_deref_trait v1.2.1 + Compiling version_check v0.9.5 + Compiling cfg-if v1.0.4 + Compiling autocfg v1.5.1 + Compiling serde_core v1.0.229 + Compiling num-traits v0.2.19 + Compiling getrandom v0.3.4 + Compiling zerocopy v0.8.56 + Compiling syn v3.0.4 + Compiling syn v2.0.119 + Compiling smallvec v1.15.2 + Compiling serde v1.0.229 + Compiling synstructure v0.13.2 + Compiling zerofrom-derive v0.1.7 + Compiling yoke-derive v0.8.2 + Compiling zerofrom v0.1.8 + Compiling writeable v0.6.4 + Compiling yoke v0.8.3 + Compiling memchr v2.8.3 + Compiling litemap v0.8.3 + Compiling num-integer v0.1.47 + Compiling zerovec-derive v0.11.6 + Compiling displaydoc v0.2.7 + Compiling serde_derive v1.0.229 + Compiling generic-array v0.14.7 + Compiling utf8_iter v1.0.4 + Compiling icu_normalizer_data v2.3.0 + Compiling icu_properties_data v2.3.0 + Compiling zerotrie v0.2.5 + Compiling zmij v1.0.23 + Compiling zerovec v0.11.8 + Compiling ref-cast v1.0.27 + Compiling typenum v1.20.1 + Compiling parking_lot_core v0.9.12 + Compiling tinystr v0.8.4 + Compiling potential_utf v0.1.6 + Compiling icu_locale_core v2.3.0 + Compiling icu_collections v2.3.0 + Compiling aho-corasick v1.1.5 + Compiling num-bigint v0.4.8 + Compiling icu_provider v2.3.1 + Compiling ref-cast-impl v1.0.27 + Compiling ahash v0.8.12 + Compiling scopeguard v1.2.0 + Compiling find-msvc-tools v0.1.11 + Compiling serde_json v1.0.151 + Compiling regex-syntax v0.8.11 + Compiling shlex v2.0.1 + Compiling lock_api v0.4.14 + Compiling cc v1.4.4 + Compiling num-rational v0.4.2 + Compiling icu_normalizer v2.3.0 + Compiling icu_properties v2.3.0 + Compiling regex-automata v0.4.18 + Compiling ppv-lite86 v0.2.21 + Compiling num-iter v0.1.46 + Compiling rand_core v0.9.5 + Compiling num-complex v0.4.6 + Compiling bit-vec v0.8.0 + Compiling pkg-config v0.3.34 + Compiling once_cell v1.21.4 + Compiling vcpkg v0.2.15 + Compiling itoa v1.0.18 + Compiling borrow-or-share v0.2.4 + Compiling fluent-uri v0.3.2 + Compiling libsqlite3-sys v0.35.0 + Compiling num v0.4.3 + Compiling bit-set v0.8.0 + Compiling rand_chacha v0.9.0 + Compiling idna_adapter v1.2.2 + Compiling parking_lot v0.12.5 + Compiling block-buffer v0.10.4 + Compiling crypto-common v0.1.7 + Compiling utf8parse v0.2.2 + Compiling vsimd v0.8.0 + Compiling thiserror v2.0.20 + Compiling percent-encoding v2.3.2 + Compiling uuid v1.26.0 + Compiling lazy_static v1.5.0 + Compiling foldhash v0.1.5 + Compiling outref v0.5.2 + Compiling hashbrown v0.15.5 + Compiling uuid-simd v0.8.0 + Compiling fraction v0.15.4 + Compiling referencing v0.33.0 + Compiling regex v1.13.1 + Compiling fancy-regex v0.16.2 + Compiling anstyle-parse v1.0.0 + Compiling digest v0.10.7 + Compiling rand v0.9.5 + Compiling idna v1.1.0 + Compiling email_address v0.2.9 + Compiling thiserror-impl v2.0.20 + Compiling anstyle-query v1.1.5 + Compiling is_terminal_polyfill v1.70.2 + Compiling anstyle v1.0.14 + Compiling bytecount v0.6.9 + Compiling cpufeatures v0.2.17 + Compiling base64 v0.22.1 + Compiling num-cmp v0.1.0 + Compiling colorchoice v1.0.5 + Compiling jsonschema v0.33.0 + Compiling anstream v1.0.0 + Compiling sha2 v0.10.9 + Compiling ulid v1.2.1 + Compiling hashlink v0.10.0 + Compiling bitflags v2.13.1 + Compiling anyhow v1.0.104 + Compiling strsim v0.11.1 + Compiling fallible-streaming-iterator v0.1.9 + Compiling clap_lex v1.1.0 + Compiling ryu-js v1.0.3 + Compiling fallible-iterator v0.3.0 + Compiling heck v0.5.0 + Compiling clap_derive v4.6.4 + Compiling clap_builder v4.6.6 + Compiling relayflowd-core v0.1.0 (/home/daytona/.relayflow-v2-supervisor/durable/repository/kernel/relayflowd-core) + Compiling clap v4.6.6 + Compiling wait-timeout v0.2.1 + Compiling rusqlite v0.37.0 + Compiling relayflowd-journal v0.1.0 (/home/daytona/.relayflow-v2-supervisor/durable/repository/kernel/relayflowd-journal) + Compiling relayflowd v0.1.0 (/home/daytona/.relayflow-v2-supervisor/durable/repository/kernel/relayflowd) + Finished `dev` profile [unoptimized + debuginfo] target(s) in 19.46s + +> @relayflows/sdk@2.0.29 typecheck +> tsc --noEmit && tsc -p tsconfig.type-tests.json + + +> @relayflows/sdk@2.0.29 build +> tsc && node scripts/make-cli-executable.mjs + + +> @relayflows/sdk@2.0.29 typecheck:tests +> tsc -p tsconfig.tests.json + + + RUN v2.1.9 /home/daytona/.relayflow-v2-supervisor/durable/repository/packages/sdk + +stdout | tests/live-kernel.test.ts +LIVE_KERNEL relayflowd=/home/daytona/.relayflows-toolchain/target/2962130851/debug/relayflowd +LIVE_KERNEL flows=/home/daytona/.relayflow-v2-supervisor/durable/repository/packages/sdk/dist/cli.js + + ✓ tests/preflight.test.ts (67 tests) 111ms + ✓ tests/cloud-read.test.ts (41 tests) 43ms + ❯ tests/cli.test.ts (70 tests | 1 failed) 1885ms + ✓ flows check CLI > binds a checked relative wrapper to the flow directory for worker execution 389ms + ✓ flows check CLI > resolves a bare PATH-resolved claude with no declared model, in an isolated PATH 434ms + × flows run/resume CLI over the journal protocol > bounds a worker wait by its lease and reports what it is waiting for 250ms + → expected +0 to be 1 // Object.is equality + ✓ tests/cloud-sync.test.ts (40 tests) 745ms + ✓ tests/plugin-extension.test.ts (90 tests) 490ms + ✓ tests/cloud-transcript-codex.test.ts (39 tests) 19ms + ❯ tests/hosted-extension-isolation.test.ts (22 tests | 13 failed) 3682ms + × hosted extension capability isolation > executes the exact capability-only handler for a queued receipt 199ms + → Hosted extension sandbox exited without a valid completion (exit 1): bwrap: loopback: Failed RTM_NEWADDR: Operation not permitted + + × hosted extension capability isolation > executes the exact capability-only handler for a duplicate receipt 206ms + → Hosted extension sandbox exited without a valid completion (exit 1): bwrap: loopback: Failed RTM_NEWADDR: Operation not permitted + + × hosted extension capability isolation > launches through the captured process primitive after builtin export synchronization 204ms + → promise rejected "Error: Hosted extension sandbox exited wi… { code: '…' }" instead of resolving + × hosted extension capability isolation > ignores inherited launcher overrides and decodes manifests with the captured Buffer intrinsic 201ms + → promise rejected "Error: Hosted extension sandbox exited wi… { code: '…' }" instead of resolving + × hosted extension capability isolation > streams verified bytes when the live store is replaced and no writable staging path exists 235ms + → promise rejected "Error: Hosted extension sandbox exited wi… { code: '…' }" instead of resolving + × hosted extension capability isolation > mounts pinned private Surface bytes when the live package changes before launch 197ms + → promise rejected "Error: Hosted extension sandbox exited wi… { code: '…' }" instead of resolving + × hosted extension capability isolation > shields verified Surface files before async settlement 208ms + → promise rejected "Error: Hosted extension sandbox exited wi… { code: '…' }" instead of resolving + × hosted extension capability isolation > preserves a typed host refusal while disclosing only a fixed marker to the child 213ms + → expected Error: Hosted extension sandbox exited wi… { code: '…' } to be Error: private Cloud policy detail { code: '…' } // Object.is equality + × hosted extension capability isolation > denies ambient credentials, host files, writes, network, subprocesses, and undeclared context verbs 216ms + → Hosted extension sandbox exited without a valid completion (exit 1): bwrap: loopback: Failed RTM_NEWADDR: Operation not permitted + + × hosted extension capability isolation > enforces OS address-space and data bounds on native Buffer allocation 215ms + → expected Error: Hosted extension sandbox exited wi… { code: '…' } to match object { code: 'plugin_unsupported', …(1) } + × hosted extension capability isolation > blocks extra handler fields and authority-bearing receipt fields at the parent port 211ms + → expected Error: Hosted extension sandbox exited wi… { code: '…' } to match object { code: 'plugin_event_unroutable' } + × hosted extension capability isolation > constructs adapter authority with the captured freeze intrinsic 209ms + → Hosted extension sandbox exited without a valid completion (exit 1): bwrap: loopback: Failed RTM_NEWADDR: Operation not permitted + + × hosted extension capability isolation > writes the Surface manifest and protocol without inherited toJSON behavior 256ms + → promise rejected "Error: Hosted extension sandbox exited wi… { code: '…' }" instead of resolving + ✓ hosted extension capability isolation > fails closed when the handler omits or repeats the single capability call 597ms + ✓ tests/observer-link.test.ts (39 tests) 136ms +(node:17171) ExperimentalWarning: SQLite is an experimental feature and might change at any time +(Use `node --trace-warnings ...` to show where the warning was created) + ✓ tests/authored-flow.test.ts (34 tests) 774ms + ✓ tests/cli-status.test.ts (27 tests) 1010ms + ✓ flows status > resolves the run with no arguments from inside a worker-spawned agent 798ms + ✓ tests/agent-transcript.test.ts (29 tests) 257ms + ✓ tests/cloud-run.test.ts (58 tests) 589ms + ❯ tests/babysitter-native-extension.test.ts (41 tests | 1 failed) 1587ms + × native Babysitter extension > runs the exact published 2.0.26 native bytes in the isolated capability path 211ms + → promise rejected "Error: Hosted extension sandbox exited wi… { code: '…' }" instead of resolving + ✓ tests/relay-cli-surface.test.ts (75 tests) 38ms + ✓ tests/worker-cli.test.ts (18 tests) 24957ms + ✓ registered CLI model defaults > passes the same priced Claude default to the real provider invocation 342ms + ✓ step discovery environment > names the run, step, attempt and an absolute data dir for a direct agent spawn 365ms + ✓ step discovery environment > exports none of the four without a data dir, even when the worker inherited them 357ms + ✓ wrapper discovery environment > sets the four names from the dispatch and still refuses ambient values and other secrets 354ms + ✓ wrapper discovery environment > exports none of the four to a wrapper without a data dir, even when the worker inherited them 381ms + ✓ custom wrapper execution identity > passes an explicit safe environment at identification and execution 366ms + ✓ custom wrapper execution identity > refuses a wrapper symlink retarget before delivering private values 468ms + ✓ custom wrapper execution identity > bounds wrapper execution after acknowledgement 455ms + ✓ custom wrapper execution identity > bounds captured wrapper output 395ms + ✓ custom wrapper execution identity > refuses a duplicate execute protocol frame 404ms + ✓ custom wrapper execution bounds are reader-owned > resolves when a conforming wrapper leaks a stdio pipe to a background helper 1915ms + ✓ custom wrapper execution bounds are reader-owned > resolves when the leaked helper inherits stderr only 1876ms + ✓ custom wrapper execution bounds are reader-owned > resolves when a wrapper leaks a stdio pipe and exits before identifying 3564ms + ✓ custom wrapper execution bounds are reader-owned > journals a completionReason at the default bound when a wrapper leaks a stdio pipe 11590ms + ✓ custom wrapper execution bounds are reader-owned > accepts an execute token and an over-8KiB payload flushed in one write 458ms + ✓ custom wrapper execution bounds are reader-owned > accepts the same over-8KiB payload whether or not it coalesces with the execute token 1028ms + ✓ custom wrapper execution bounds are reader-owned > still bounds an un-terminated handshake buffer and names the bound 308ms + ✓ delivers the journaled memory pack to the real wrapper and excludes its charge from completion usage 329ms + ✓ tests/step-failure-diagnostic.test.ts (25 tests) 61ms + ✓ tests/daemon-lifecycle.test.ts (42 tests) 39ms + ✓ tests/stop-process-group.test.ts (9 tests) 12811ms + ✓ every stop reaches the process group, not just the direct child > exits the run after an execution-timeout stop 824ms + ✓ every stop reaches the process group, not just the direct child > exits the run after a protocol terminate stop 398ms + ✓ every stop reaches the process group, not just the direct child > kills a SIGTERM-deaf grandchild after a protocol terminate stop 1684ms + ✓ every stop reaches the process group, not just the direct child > kills a SIGTERM-deaf grandchild after an execution-timeout stop 1995ms + ✓ every stop reaches the process group, not just the direct child > holds the loop open long enough for the escalation to run 1087ms + ✓ a wrapper that exits with no execution deadline still drains > reports the wrapper result and reaps a grandchild holding its pipes 656ms + ✓ a wrapper that exits with no execution deadline still drains > reaps a SIGTERM-deaf grandchild holding its pipes 1659ms + ✓ a wrapper that exits with no execution deadline still drains > settles on its own deadline when an escaped holder withholds close 4241ms + ✓ tests/run-state.test.ts (21 tests) 12ms + ✓ tests/cloud-deploy.test.ts (40 tests) 909ms +stdout | tests/live-kernel.test.ts > surface resume after a real daemon kill > resumes a three-step run with each successful completion exactly once +LIVE_KERNEL kill -9 pid=18996 run=01M36CABNXG5QC8DQX88V2BTAQ while step=two state=Running + + ✓ tests/hosted-base-snapshot.test.ts (18 tests) 1872ms + ✓ hosted base private snapshot > stops streaming project entries at the shared count limit 1795ms + ❯ tests/live-kernel.test.ts (31 tests | 8 failed) 52698ms + ✓ built flows CLI against live relayflowd > twenty-six-step reuses 25 durable completions after editing the failed final step 2203ms + ✓ built flows CLI against live relayflowd > runs rung (a), parks rung (b), and keeps JSON report-shaped 2755ms + ✓ built flows CLI against live relayflowd > allows a deterministic run to exceed the bounded request timeout 32497ms + ✓ built flows CLI against live relayflowd > follows a live worker dispatch through flows run 627ms + ✓ built flows CLI against live relayflowd > runs an agent CLI end to end through the SDK worker 531ms + ✓ built flows CLI against live relayflowd > f.agent lowers to a real agent step and dispatches through a live worker 534ms + ✓ built flows CLI against live relayflowd > can always get a parked run to a late-attaching worker 5577ms + ✓ built flows CLI against live relayflowd > reports a real manual-recovery NeedsHuman state as parked 430ms + × built flows CLI against live relayflowd > runs hn-monitor analyze-story end-to-end via a stub agent CLI (gate 2 clause 2 demo) 450ms + → expected { …(12) } to match object { output: { …(3) }, …(1) } +(22 matching properties omitted from actual) + × built flows CLI against live relayflowd > hn-monitor analyze-story FAILS verification when the CLI omits required schema fields 447ms + → expected { …(12) } to match object { …(3) } +(21 matching properties omitted from actual) + × built flows CLI against live relayflowd > agent step preserves the CliResult wrapper as output when the CLI emits non-JSON text 403ms + → expected null not to be null + × built flows CLI against live relayflowd > AgentWorker exposes wake_context to the CLI via RELAYFLOW_WAKE_CONTEXT env var (real analyzer prerequisite) 447ms + → Cannot read properties of null (reading 'story_title') + × built flows CLI against live relayflowd > AgentWorker leaves RELAYFLOW_WAKE_CONTEXT UNSET when the run has no wake_context (undefined-vs-null pin) 402ms + → Cannot read properties of null (reading 'env_present') + ✓ built flows CLI against live relayflowd > AgentWorker passes a declared model to an identified wrapper as RELAYFLOW_MODEL 446ms + ✓ built flows CLI against live relayflowd > AgentWorker refuses a nonconforming journal-submitted wrapper before exposing RELAYFLOW_MODEL 402ms + ✓ built flows CLI against live relayflowd > AgentWorker executes the raw claude adapter with its real model flag 365ms + ✓ built flows CLI against live relayflowd > AgentWorker executes the raw codex adapter with its real model flag 359ms + × built flows CLI against live relayflowd > AgentWorker leaves RELAYFLOW_MODEL UNSET when the step declares no model 401ms + → Cannot read properties of null (reading 'story_title') + × built flows CLI against live relayflowd > hn-monitor analyze-story reaches done through the real Claude analyzer CLI 28ms + → LIVE_ANALYZER_UNAVAILABLE: "/home/daytona/.relayflow-v2-supervisor/durable/repository/testdata/preflight/analyze-story-claude-cli" does not identify as relayflows-agent-cli-v1 — failing because gate-2 acceptance requires the real analyzer to execute. Set RELAYFLOWS_ALLOW_ANALYZER_SKIP=1 only if this run is not gate evidence. + ✓ built flows CLI against live relayflowd > preflights before journaling and names an unreachable socket 821ms + ✓ built flows CLI against live relayflowd > starts exactly one daemon when two runs race for one empty data dir 505ms + ✓ surface resume after a real daemon kill > resumes a three-step run with each successful completion exactly once 909ms + × a relayflow can be scheduled: tick source against live relayflowd > a tick spawns a real run whose step reports the SCHEDULED instant 448ms + → expected null to deeply equal { schedule_id: 'heartbeat-1m', …(3) } + ❯ tests/hosted-extension-protocol.test.ts (24 tests | 8 failed) 41933ms + × hosted extension hostile protocol > uses captured JSON intrinsics for the complete parent boundary 240ms + → Hosted extension sandbox exited without a valid completion (exit 1): bwrap: loopback: Failed RTM_NEWADDR: Operation not permitted + + × hosted extension hostile protocol > rejects an import-time different PR frame with zero adapter calls 216ms + → expected Error: Hosted extension sandbox exited wi… { code: '…' } to match object { code: 'plugin_event_unroutable' } + × hosted extension hostile protocol > rejects an import-time different delivery frame with zero adapter calls 217ms + → expected Error: Hosted extension sandbox exited wi… { code: '…' } to match object { code: 'plugin_event_unroutable' } + × hosted extension hostile protocol > rejects an import-time different event frame with zero adapter calls 223ms + → expected Error: Hosted extension sandbox exited wi… { code: '…' } to match object { code: 'plugin_event_unroutable' } + × hosted extension hostile protocol > rejects two forged calls after the authoritative first outcome settles 10145ms + → hostile child did not invoke the adapter + × hosted extension hostile protocol > waits for a pending adapter to reject after a forged child error 10145ms + → hostile child did not invoke the adapter + × hosted extension hostile protocol > waits for a pending adapter to resolve after a forged child error 10139ms + → hostile child did not invoke the adapter + × hosted extension hostile protocol > returns a typed adapter rejection even when the hostile child hangs 10134ms + → hostile child did not invoke the adapter + ✓ tests/journal-client.test.ts (17 tests) 81ms + ✓ tests/authored-root.test.ts (13 tests) 153ms + ✓ tests/flow-extension-compose.test.ts (22 tests) 3618ms + ✓ composing flow extensions onto a base flow > composes two extensions in declaration order, and the order is the lockfile order 385ms + ✓ composing flow extensions onto a base flow > flows check reports the composition and keeps the composed triggers deliverable 759ms + ✓ tests/cloud-connect.test.ts (24 tests) 2988ms + ✓ hosted verbs connect before they submit > flows run --cloud submits once the prompt connected the integration 2114ms + ❯ tests/authored-node-runtime.test.ts (14 tests | 14 skipped) 11ms + ✓ tests/close-pr-flow.test.ts (28 tests) 316ms + ✓ tests/validate.test.ts (68 tests) 26ms + ✓ tests/verb-field-lint.test.ts (96 tests) 304ms + ✓ tests/bundle.test.ts (26 tests) 8599ms + ✓ immutable bundles > returns exit 2 naming a byte-flipped payload and refuses to reuse corruption 383ms + ✓ immutable bundles > verifies with --verify in any position and answers --json with one object 756ms + ✓ immutable bundles > refuses --out with --verify rather than ignoring the destination 373ms + ✓ immutable bundles > builds and verifies the canonical YAML fixture through the compiled CLI 1151ms + ✓ immutable bundles > emits the ephemeral warning on CLI stderr and uses the default output directory 792ms + ✓ immutable bundles > refuses build-provable CLI resolution errors without environment probes 387ms + ✓ immutable bundles > builds a standalone TS fixture twice with identical executable hashes 2358ms + ✓ immutable bundles > refuses to label installed dependency drift with lockfile pins 379ms + ✓ immutable bundles > refuses invalid CLI arguments %j 375ms + ✓ immutable bundles > refuses invalid CLI arguments "--out" 382ms + ✓ immutable bundles > refuses invalid CLI arguments "--verify" 379ms + ✓ immutable bundles > refuses invalid CLI arguments "--verify" 378ms + ✓ immutable bundles > refuses invalid CLI arguments "--out" 383ms + ✓ tests/tick-source.test.ts (33 tests) 24ms + ✓ tests/authored-step-graph.test.ts (25 tests) 816ms + ✓ the authored step DAG > does not walk a long-running step's own polling chain to find its dependents' edges 305ms + ✓ tests/flow-executor-chain.test.ts (14 tests) 9698ms + ✓ flow executor LLM and output-binding chain > runs f.llm -> f.agent -> f.run with schema-verified journal output and the exact allowed model 817ms + ✓ flow executor LLM and output-binding chain > runs a dollar-budgeted authored Claude agent with the same default used by preflight 569ms + ✓ flow executor LLM and output-binding chain > fails invalid LLM output before the next step: {"message":7} 306ms + ✓ flow executor LLM and output-binding chain > runs the exact authored flagship f.llm -> f.agent -> f.run path through the durable CLI root 1488ms + ✓ flow executor LLM and output-binding chain > resumes an interrupted durable authored root without replaying completed flagship effects 3258ms + ✓ flow executor LLM and output-binding chain > passes a declarative verified value through an agent into a deterministic artifact 652ms + ✓ flow executor LLM and output-binding chain > flows run consumes YAML bindings and resume reuses the original journal output 1054ms + ✓ tests/agent-relay-transport.test.ts (16 tests) 2217ms + ✓ Relay completion at the journal boundary > does not complete at readiness and journals exact output, receipt, and priced accounting 1006ms + ✓ Relay completion at the journal boundary > aborts polling on rejected renewal and never writes a stale completion 1002ms + ✓ tests/authored-flow-lifecycle-executor.test.ts (27 tests) 577ms + ✓ tests/step-failure-excerpt.test.ts (42 tests) 206ms + ✓ tests/pr-review-post.test.ts (21 tests) 2028ms + ✓ tests/authored-flow-slack.test.ts (7 tests) 1632ms + ✓ authored Slack helper effects > replays after SIGKILL before confirm with the same token and one successful completion 524ms + ✓ authored Slack helper effects > replays after SIGKILL before complete with the same token and one successful completion 538ms + ✓ tests/tick-runner.test.ts (22 tests) 2279ms + ✓ CLI argument parsing refuses coercion rather than accepting it > refuses --interval-ms fractional as an invocation error 375ms + ✓ CLI argument parsing refuses coercion rather than accepting it > refuses --interval-ms exponent notation as an invocation error 366ms + ✓ CLI argument parsing refuses coercion rather than accepting it > refuses --interval-ms hex as an invocation error 383ms + ✓ CLI argument parsing refuses coercion rather than accepting it > refuses --interval-ms trailing text as an invocation error 383ms + ✓ CLI argument parsing refuses coercion rather than accepting it > refuses --interval-ms empty as an invocation error 368ms + ✓ CLI argument parsing refuses coercion rather than accepting it > accepts an exact integer and proceeds past parsing 374ms + ✓ tests/mcp.test.ts (30 tests) 21472ms + ✓ MCP preflight and transports > flows check refuses an undeclared server with exit 2 and no daemon 568ms + ✓ MCP preflight and transports > flows check reports a refusing server and leaves no PID 594ms + ✓ MCP preflight and transports > kills a SIGTERM-resistant silent child after a parent-owned handshake deadline 1312ms + ✓ MCP preflight and transports > reaps a SIGTERM-resistant descendant with inherit stdio before cleanup finishes 1112ms + ✓ MCP preflight and transports > reaps a SIGTERM-resistant descendant with ignore stdio before cleanup finishes 2061ms + ✓ MCP preflight and transports > reports malformed connection configuration as config_invalid 579ms + ✓ authored MCP effects against the real kernel > reports a dropped tool connection as a failed CLI run 14069ms + ✓ authored MCP effects against the real kernel > journals drop/echo failure as worker_error with diagnostic mcp_disconnected 469ms + ✓ tests/authored-step-index.test.ts (17 tests) 18ms +(node:21315) ExperimentalWarning: SQLite is an experimental feature and might change at any time +(Use `node --trace-warnings ...` to show where the warning was created) + ✓ tests/gate-contract.test.ts (20 tests) 138ms + ✓ tests/authored-node-result.test.ts (39 tests) 18ms + ✓ tests/cli-replay.test.ts (37 tests) 1144ms + ✓ flows replay > --json is byte-identical across two CLI invocations (diff) 774ms + ✓ tests/authored-human.test.ts (13 tests) 138ms + ✓ tests/wrapper-execution-duration.test.ts (7 tests) 10824ms + ✓ keeps the handshake deadline independent of the removed execution deadline 10058ms + ✓ still lets a lease abort stop an unlimited wrapper before it produces output 511ms + ✓ tests/direct-input.test.ts (6 tests) 5876ms + ✓ direct .flow.ts input through the built CLI and live runtime > returns exit 3 for an authored human handoff and persists its outcome 755ms + ✓ direct .flow.ts input through the built CLI and live runtime > returns exit 1 for an authored step_failed verdict and persists its outcome 708ms + ✓ direct .flow.ts input through the built CLI and live runtime > executes inline and file JSON input through relayflowd 1893ms + ✓ direct .flow.ts input through the built CLI and live runtime > refuses missing and malformed input before contacting relayflowd 1650ms + ✓ direct .flow.ts input through the built CLI and live runtime > does not run the authored body before daemon availability 498ms + ✓ direct .flow.ts input through the built CLI and live runtime > refuses oversized file input before contacting relayflowd 372ms + ✓ tests/authored-run-failure-evidence.test.ts (9 tests) 1112ms + ✓ the child index after the process that wrote it is gone > still names every child, with its own run id, after a daemon restart 638ms + ✓ tests/cloud-schedule.test.ts (17 tests) 5404ms + ✓ schedule lowering > marks a non-grid cron as Cloud-only rather than approximating it, with a silence budget from its own cadence 3222ms + ✓ flows check prints declared schedules > shows the lowering for a fixed interval and the Cloud-only note for a real cron 1789ms + ✓ tests/cli-hn-monitor.test.ts (16 tests) 94ms + ✓ tests/daemon-lifecycle-live.test.ts (9 tests) 6497ms + ✓ flows run against a data dir with no daemon (§6 test 7) > cold start spawns exactly one daemon, the run succeeds, and the daemon outlives the CLI 492ms + ✓ flows run against a data dir with no daemon (§6 test 7) > polls, bounded, for a daemon that holds the lock before it binds 1366ms + ✓ flows run against a data dir with no daemon (§6 test 7) > attaches to a serving daemon that has not published a connection file 461ms + ✓ flows run against a data dir with no daemon (§6 test 7) > a second run attaches to the daemon the first one started, spawning nothing 887ms + ✓ flows run against a data dir with no daemon (§6 test 7) > detects a stale connection file left by a hard kill and starts a fresh daemon 934ms + ✓ concurrent invocations against one empty data dir (§6 test 15) > ends with exactly one daemon owning the socket, and both runs succeed 1065ms + ✓ refusals from a spawn that cannot produce a daemon > names relayflowd_not_found rather than falling through to PATH 395ms + ✓ refusals from a spawn that cannot produce a daemon > names daemon_start_failed and quotes the daemon log when startup dies 446ms + ✓ refusals from a spawn that cannot produce a daemon > refuses a daemon speaking another protocol version instead of binding over it 450ms +(node:22576) Warning: Transcript tail for run-9/analyze attempt 1 (stdout) could not be written; the step continues without it: EACCES: permission denied, mkdir '/tmp/transcript-tail-tdD5KP/runs/run-9/steps' +(Use `node --trace-warnings ...` to show where the warning was created) + ✓ tests/transcript-tail.test.ts (11 tests) 736ms + ✓ direct agent spawn > tees stdout and stderr into tail files that name the dispatch 304ms + ✓ direct agent spawn > completes the step when the tail directory cannot be created 347ms + ✓ tests/authored-agent-artifacts.test.ts (4 tests) 380ms + ✓ tests/authored-helpers.test.ts (6 tests) 2921ms + ✓ runs every available provider through the real kernel and resumes completed effects without a second write 1479ms + ✓ replays after SIGKILL before confirm with the same token and one successful completion 523ms + ✓ replays after SIGKILL before complete with the same token and one successful completion 504ms + ✓ tests/authored-flow-operation.test.ts (24 tests) 548ms + ✓ tests/flow-requirements.test.ts (14 tests) 530ms + ✓ flows check prints REQUIRES > names the helper, the harness and the mcp server of an authored flow 314ms + ✓ tests/backlog-picker.test.ts (14 tests) 45ms + ✓ tests/plugin-store-bounds.test.ts (11 tests) 87ms + ✓ tests/backlog-picker-flow.test.ts (6 tests) 260ms + ✓ tests/hosted-extension-protocol-intrinsics.test.ts (6 tests) 13ms + ✓ tests/preflight-permissions-unenforced.test.ts (17 tests) 234ms + ✓ tests/wrapper-exit-drain.test.ts (8 tests) 2607ms + ✓ reports a signalled wrapper death while a pipe is held, with its output intact 382ms + ✓ lets a lease abort outrank a successful exit still being drained 533ms + ✓ tests/stuck-run-triage.test.ts (22 tests) 3106ms + ✓ stuck-run-triage shell text > collects tails with no GNU timeout on PATH, as on a stock macOS 3068ms + ✓ tests/worker-transcript.test.ts (5 tests) 192ms + ✓ tests/hosted-extension-routing.test.ts (7 tests) 8ms + ✓ tests/artifact-gates.test.ts (7 tests) 181ms + ✓ tests/webhook.test.ts (9 tests) 452ms + ✓ webhook ingress > checks TS declarations against flows.json without invoking handlers 388ms + ✓ tests/agent-artifacts-live.test.ts (5 tests) 41830ms + ✓ agent artifacts and gates through the built CLI, a real daemon and the local agent > journals the files the agent wrote, including under a dot-directory, and every artifact gate passes on that journal 1260ms + ✓ agent artifacts and gates through the built CLI, a real daemon and the local agent > fails the run when the artifact_exists gate names a file the agent did not write 12089ms + ✓ agent artifacts and gates through the built CLI, a real daemon and the local agent > fails the run with the author reason when a predicate gate returns false, journaling the verdict 13615ms + ✓ review follow-ups > applies a predicate gate on a helper step too, and journals its verdict 14209ms + ✓ review follow-ups > records predicate verdicts on the root run so a resume reuses them instead of re-running the closure 656ms + ✓ tests/agent-transcript-live.test.ts (4 tests) 43604ms + ✓ the transcript digest through the built CLI, a real daemon and the local agent > preserves structured agent failure details and its completed root index 14553ms + ✓ the transcript digest through the built CLI, a real daemon and the local agent > preserves structured llm failure details and its completed root index 14292ms + ✓ the transcript digest through the built CLI, a real daemon and the local agent > journals the digest in trajectory_tail on a successful agent step and writes the file it points at 757ms + ✓ the transcript digest through the built CLI, a real daemon and the local agent > on a failed agent step, names the failure and the transcript in the terminal diagnostic, redacted 14002ms + ✓ tests/human-live.test.ts (3 tests) 6400ms + ✓ f.human against a real daemon > parks with the question, refuses wrong answers, records one, and resumes to success 3801ms + ✓ f.human against a real daemon > a "no" is a value the body branches on: declined, exit 0, no effect 1592ms + ✓ f.human against a real daemon > refuses to answer a run the daemon does not know 1007ms + ✓ tests/authored-step-failed.test.ts (10 tests) 36ms + ✓ tests/cli-watch.test.ts (10 tests) 16022ms + ✓ flows check --watch > rechecks syntax errors, clears once, and returns the last refusal on Ctrl-C 1338ms + ✓ flows check --watch > streams JSON lines without ANSI, recovers after atomic saves, and exits zero after repair 1882ms + ✓ flows check --watch > coalesces 20 concurrent saves into at most two rechecks 1852ms + ✓ flows check --watch > watches transitive relative use imports, cycles, and nearest config changes 2401ms + ✓ flows check --watch > refreshes the import graph and notices missing imports being created 2449ms + ✓ flows check --watch > reloads authored TypeScript instead of reusing the first imported definition 1556ms + ✓ flows check --watch > detects a nearer config appearing and falls back after it is deleted 1886ms + ✓ flows check --watch > keeps watching after the target is deleted and recreated 1883ms + ✓ flows check --watch > queues changes during a slow check without overlapping checks 773ms + ✓ tests/budget-preflight.test.ts (25 tests) 19ms + ✓ tests/authored-parallel-agents.test.ts (8 tests) 12199ms + ✓ authored steps under local workers with capacity > runs more concurrent f.llm calls than the worker holds side by side, never more than its capacity 1241ms + ✓ authored steps under local workers with capacity > completes more concurrent f.agent calls than the worker holds: the overflow waits for a slot instead of parking 2559ms + ✓ authored steps under local workers with capacity > runs agents in distinct working directories side by side (the kernel carries cwd) 701ms + ✓ authored steps under local workers with capacity > serializes agents whose cwd is a symlink alias of the same directory 1087ms + ✓ authored steps under local workers with capacity > serializes agents whose cwd is a directory nested inside the other 1100ms + ✓ authored steps under local workers with capacity > never starts queued agents once the body has failed 2032ms + ✓ authored steps under local workers with capacity > never starts a queued agent when the agent holding the only slot fails 2049ms + ✓ authored steps under local workers with capacity > parks the overflow when the body is not told the capacity (the defect this closes) 1429ms + ✓ tests/budget-unmetered-live.test.ts (3 tests) 1176ms + ✓ unmetered budget spend through the live kernel > runs an unpriced step under a dollar budget without tripping it, journaling unknown dollars 602ms + ✓ unmetered budget spend through the live kernel > still counts an unpriced step toward a token budget 309ms + ✓ tests/provider-trigger-contract.test.ts (7 tests) 461ms + ✓ tests/work-package-consumer.test.ts (13 tests) 111ms + ✓ tests/spec-parity.test.ts (31 tests) 358ms + ✓ tests/helpers-fanout.test.ts (96 tests) 202ms +(node:25864) ExperimentalWarning: SQLite is an experimental feature and might change at any time +(Use `node --trace-warnings ...` to show where the warning was created) + ✓ tests/authored-step-graph-live.test.ts (1 test) 760ms + ✓ the authored step DAG through the live kernel > carries labels and predecessors on every index record and journal step, ids unchanged 759ms + ✓ tests/generate-triggers.test.ts (7 tests) 1063ms + ✓ discovers new adapters, preserves exact event names, and prefers adapter-local mappings 340ms + ✓ tests/worker-cli-result-exit.test.ts (5 tests) 32885ms + ✓ a Claude agent step completes on its result, not only on process exit > settles a hung, successful run within the grace and stops its whole tree 31628ms + ✓ a Claude agent step completes on its result, not only on process exit > maps an error result on a hung run to a failed exit 31627ms + ✓ a Claude agent step completes on its result, not only on process exit > leaves a hang before any result to the existing stops 32010ms + ✓ an agent tree does not outlive the process that spawned it > kills the agent group when the run process is terminated by SIGTERM 793ms + ✓ tests/webhook-hardening.test.ts (11 tests) 62ms + ✓ tests/human-to.test.ts (8 tests) 11ms + ✓ tests/pty-sidechannel.test.ts (11 tests) 6023ms + ✓ view attach preserves worker completion and marks only drive 771ms + ✓ drive attach preserves worker completion and marks only drive 342ms + ✓ passthrough attach preserves worker completion and marks only drive 801ms + ✓ none attach preserves worker completion and marks only drive 811ms + ✓ none subscriber lets an unattended CLI read EOF 451ms + ✓ view subscriber lets an unattended CLI read EOF 492ms + ✓ passthrough subscriber lets an unattended CLI read EOF 485ms + ✓ incomplete subscriber lets an unattended CLI read EOF 438ms + ✓ rejects drive after EOF without marking human intervention 757ms + ✓ delivers all drive bytes in order across child stdin backpressure 672ms + ✓ tests/plugin-loader.test.ts (9 tests) 232ms + ✓ tests/worker-lease.test.ts (7 tests) 13ms + ✓ tests/yaml-helpers.test.ts (33 tests) 78ms + ❯ tests/webhook-live.test.ts (6 tests | 6 failed) 62555ms + × executes and deduplicates 'app_mention' only for its provider and matching payload 10452ms + → webhook integration timed out: spawn /home/daytona/.relayflow-v2-supervisor/durable/repository/kernel/target/debug/relayflowd ENOENT + × executes and deduplicates 'reaction_added' only for its provider and matching payload 10413ms + → webhook integration timed out: spawn /home/daytona/.relayflow-v2-supervisor/durable/repository/kernel/target/debug/relayflowd ENOENT + × executes and deduplicates 'pull_request' only for its provider and matching payload 10456ms + → webhook integration timed out: spawn /home/daytona/.relayflow-v2-supervisor/durable/repository/kernel/target/debug/relayflowd ENOENT + × flows serve-webhook writes JSON before the daemon starts, then journals and archives exactly once 10418ms + → webhook integration timed out: spawn /home/daytona/.relayflow-v2-supervisor/durable/repository/kernel/target/debug/relayflowd ENOENT + × replays a dropped file after SIGKILL before spawn 10404ms + → webhook integration timed out: spawn /home/daytona/.relayflow-v2-supervisor/durable/repository/kernel/target/debug/relayflowd ENOENT + × resumes the same journal after SIGKILL after spawn and before acknowledgement 10411ms + → webhook integration timed out: spawn /home/daytona/.relayflow-v2-supervisor/durable/repository/kernel/target/debug/relayflowd ENOENT + ✓ tests/authored-agent-permissions.test.ts (26 tests) 779ms + ✓ tests/babysitter-catalog-export.test.ts (14 tests) 790ms + ✓ Babysitter catalog artifact export > CLI refuses an existing output and leaves no file on validation failure 659ms + ✓ tests/redact.test.ts (35 tests) 8ms + ✓ tests/communication.test.ts (10 tests) 16ms + ✓ tests/typed-output.test.ts (14 tests) 223ms + ❯ tests/worker-lease-lost.test.ts (16 tests | 1 failed) 19ms + × keeps an invalid lease deadline fatal 3ms + → expected "spy" to be called 1 times, but got 0 times + ✓ tests/canonical-software-factory.test.ts (3 tests) 100ms + ✓ tests/budget-attribution.test.ts (5 tests) 9ms + ✓ tests/deploy.test.ts (11 tests) 5250ms + ✓ flows deploy file buckets > publishes the full signed layout byte-for-byte and redeploys as a noop 829ms + ✓ flows deploy file buckets > answers --json with one object per outcome 791ms + ✓ flows deploy file buckets > reports a refusal as JSON under --json 380ms + ✓ flows deploy file buckets > refuses a missing local bundle before creating the bucket 373ms + ✓ flows deploy file buckets > refuses an unreachable bucket before copying 400ms + ✓ flows deploy file buckets > refuses an unwritable bucket 404ms + ✓ flows deploy file buckets > refuses local tampering of spec.canonical.json 430ms + ✓ flows deploy file buckets > refuses local tampering of identity.json 385ms + ✓ flows deploy file buckets > refuses asset bundles instead of using daemon-relative files 409ms + ✓ flows deploy file buckets > never labels a corrupt existing deployment as a noop 825ms + ✓ tests/json-schema-bound.test.ts (71 tests) 2359ms + ✓ JSON Schema termination bound > walks a deep schema with an explicit stack rather than recursion 1896ms + ✓ tests/effect-channel.test.ts (5 tests) 361ms + ✓ tests/model-selection.test.ts (10 tests) 17ms + ✓ tests/mcp-lifecycle.test.ts (4 tests) 13ms + ✓ tests/relayflowd-path.test.ts (10 tests) 5ms + ✓ tests/agent-artifacts.test.ts (9 tests) 18ms + ✓ tests/f-memory.test.ts (7 tests) 818ms + ✓ tests/authored-plugin-effect.test.ts (6 tests) 67ms + ✓ tests/yaml-local-agent-live.test.ts (7 tests) 4181ms + ✓ YAML --local-agent through the built CLI and real daemon > runs with the checked step CLI and model and journals done 623ms + ✓ YAML --local-agent through the built CLI and real daemon > runs with the checked named CLI and model and journals done 597ms + ✓ YAML --local-agent through the built CLI and real daemon > runs with the checked flow CLI and model and journals done 617ms + ✓ YAML --local-agent through the built CLI and real daemon > runs with the checked project CLI and model and journals done 627ms + ✓ YAML --local-agent through the built CLI and real daemon > still parks without --local-agent 559ms + ✓ YAML --local-agent through the built CLI and real daemon > reports the agent process failure 596ms + ✓ YAML --local-agent through the built CLI and real daemon > preserves declared workspace surfaces that the local worker cannot pin 562ms + ✓ tests/worker-slots.test.ts (7 tests) 6ms + ✓ tests/local-dev-ux.test.ts (8 tests) 16ms + ↓ tests/relay-cli-surface-live.test.ts (3 tests | 3 skipped) + ✓ tests/authored-declined.test.ts (13 tests) 72ms + ✓ tests/resume-failure.test.ts (2 tests) 6ms + ✓ tests/dependency-validation.test.ts (6 tests) 697ms + ✓ dependency validation > accepts a valid 10,000-step reverse chain through every direct public boundary 369ms + ✓ tests/authored-hooks.test.ts (5 tests) 5ms + ✓ tests/input-binding.test.ts (12 tests) 211ms + ✓ tests/communication-review.test.ts (5 tests) 319ms + ✓ tests/yaml-helper-effect.test.ts (4 tests) 74ms + ✓ tests/deterministic-llm.test.ts (5 tests) 49ms + ✓ tests/scope-preflight.test.ts (6 tests) 8ms + ✓ tests/bin.test.ts (7 tests) 2843ms + ✓ built flows binary > refuses through a symlink to the built artifact 385ms + ✓ built flows binary > refuses through a symlinked directory component 399ms + ✓ built flows binary > classifies a signal-terminated auth probe as probe_failed 873ms + ✓ built flows binary > classifies an unavailable PATH resolver as probe_failed 396ms + ✓ built flows binary > does not describe a present non-executable CLI as missing 392ms + ✓ built flows binary > runs one auth probe for three steps sharing a flow CLI 395ms + ✓ tests/build-gate.test.ts (3 tests) 1216ms + ✓ flows build gates on flows check green (#318) > refuses a flow with an unresolvable named-agent CLI and leaves no artifacts 374ms + ✓ flows build gates on flows check green (#318) > --json emits one CheckReport object on stdout on refusal, exits 2, no artifacts 418ms + ✓ flows build gates on flows check green (#318) > builds the bundle on success (regression: gate must not block valid flows) 423ms + ✓ tests/scope-compiler.test.ts (25 tests) 12ms + ✓ tests/run-from-digest.test.ts (6 tests) 4568ms + ✓ flows run digest input > submits the sealed canonical spec through the normal journal path without checkout 446ms + ✓ flows run digest input > uses a verified cache hit even after the bucket is removed 465ms + ✓ flows run digest input > resolves deploy.bucket from flows.json and honors explicit override 1225ms + ✓ flows run digest input > refuses an unconfigured bucket 781ms + ✓ flows run digest input > refuses tampered spec.canonical.json before creating run data 863ms + ✓ flows run digest input > refuses tampered identity.json before creating run data 788ms + ✓ tests/communication-worker.test.ts (15 tests) 1492ms + ✓ tests/hn-poller.test.ts (6 tests) 6ms + ✓ tests/plugin-add.test.ts (7 tests) 1291ms + ✓ installs a real offline npm fixture and includes declarations 343ms + ✓ typechecks the augmented verb and rejects unknown namespaces 930ms + ✓ tests/authored-step-failed-exit.test.ts (3 tests) 8ms + ✓ tests/direct-run-failure.test.ts (8 tests) 12ms + ✓ tests/dir-watcher-poller.test.ts (6 tests) 4ms + ✓ tests/model-pricing.test.ts (10 tests) 5ms + ✓ tests/yaml-helper-live.test.ts (1 test) 1041ms + ✓ runs compiled YAML helpers through the built CLI and kernel effect journal 1040ms + ❯ tests/provider-trigger-executor.test.ts (4 tests | 3 failed) 18ms + × the kernel executes compiled 'app_mention' subscriptions with provider isolation and durable dedupe 9ms + → spawnSync /home/daytona/.relayflow-v2-supervisor/durable/repository/kernel/target/debug/relayflowd ENOENT + × the kernel executes compiled 'reaction_added' subscriptions with provider isolation and durable dedupe 3ms + → spawnSync /home/daytona/.relayflow-v2-supervisor/durable/repository/kernel/target/debug/relayflowd ENOENT + × the kernel executes compiled 'pull_request' subscriptions with provider isolation and durable dedupe 3ms + → spawnSync /home/daytona/.relayflow-v2-supervisor/durable/repository/kernel/target/debug/relayflowd ENOENT + ✓ tests/transcript-tail-close.test.ts (2 tests) 1016ms + ✓ a stalled transcript-tail close > does not hold the spawn open past its bounded window 475ms + ✓ a stalled tail close beside a transcript that finished > still journals the transcript pointer 540ms + ✓ tests/wrapper-artifacts-cwd.test.ts (2 tests) 75ms + ✓ tests/hello-deterministic.test.ts (5 tests) 17ms + ✓ tests/transcript-exclusion-timeout.test.ts (1 test) 188ms + ✓ tests/cli-adapter.test.ts (4 tests) 6ms + ✓ tests/communication-mixed-resume.test.ts (1 test) 165ms + ✓ tests/work-package-validator.test.ts (7 tests) 5ms + ✓ tests/authored-use-loader.test.ts (5 tests) 523ms + ✓ tests/authored-declined-live.test.ts (1 test) 1701ms + ✓ runs an input guard and resumes its completed declined root without repeated effects 1701ms + ✓ tests/cli-answer.test.ts (15 tests) 9ms + ✓ tests/bundle-preflight.test.ts (4 tests) 876ms + ✓ bundle execution preflight > ignores surrounding cache configuration on a verified cache hit 438ms + ✓ bundle execution preflight > uses the built alias for a nameless flow even in a digest-only cache directory 417ms + ✓ tests/agent-relay-hardening.test.ts (12 tests) 12ms + ✓ tests/classify-outcome.test.ts (2 tests) 2162ms + ✓ classifyOutcome > gives up and reports when a running run never becomes classifiable 2009ms + ✓ tests/communication-preflight.test.ts (13 tests) 30ms + ↓ tests/real-cli-adapters.test.ts (3 tests | 3 skipped) + ✓ tests/memoization.test.ts (57 tests) 58ms + ✓ tests/fs-descriptor.test.ts (1 test) 4ms + ✓ tests/parse-json-output.test.ts (7 tests) 3ms + ✓ tests/journal-client-completion.test.ts (4 tests) 100ms + ✓ tests/worker-cli-abort.test.ts (2 tests) 2535ms + ✓ stops claude and its process group when lease ownership is lost 1252ms + ✓ stops wrapper.mjs and its process group when lease ownership is lost 1282ms + ✓ tests/communication-environment-preflight.test.ts (6 tests) 4ms + ✓ tests/budget-authored-live.test.ts (2 tests) 202ms + ✓ tests/slack-writeback.test.ts (1 test) 262ms + ✓ tests/authored-surface-authority.test.ts (2 tests) 16ms + ✓ tests/adapters/claude.test.ts (7 tests) 4ms + ✓ tests/worker-cli-cwd.test.ts (2 tests) 296ms + ✓ tests/adapters/codex.test.ts (7 tests) 4ms + ✓ tests/worker-lease-lost-live.test.ts (2 tests) 586ms + ✓ reports journal success after completion rejects with lease_conflict 307ms + ✓ tests/slack-block-kit.test.ts (5 tests) 31ms + ✓ tests/worker-lease-sweep.test.ts (2 tests) 6ms + ✓ tests/communication-history.test.ts (1 test) 3ms + ✓ tests/adapters/registry.test.ts (4 tests) 4ms + ✓ tests/resume-worker-lease.test.ts (2 tests) 4ms + ✓ tests/authored-declined-report.test.ts (6 tests) 7ms + ✓ tests/promise-ancestry.test.ts (2 tests) 248ms + ✓ tests/communication-refusal.test.ts (1 test) 12ms + ✓ tests/agent-cwd-validation.test.ts (2 tests) 395ms + ✓ declarative agent cwd > is refused by `flows check` on a YAML flow before anything runs 392ms + ✓ tests/bundle-transport.test.ts (20 tests) 2555ms + ✓ digest references > accepts and deploys the build output for hello 446ms + ✓ digest references > accepts and deploys the build output for Hello 415ms + ✓ digest references > accepts and deploys the build output for hello.world 414ms + ✓ digest references > accepts and deploys the build output for hello_world 429ms + ✓ digest references > accepts and deploys the build output for 123 415ms + ✓ digest references > accepts and deploys the build output for A_b.c-1 433ms + ✓ tests/catalog-plugins.test.ts (2 tests) 3ms + ✓ tests/check-command-cwd.test.ts (1 test) 12ms + ✓ tests/communication-lazy.test.ts (1 test) 3ms + ✓ tests/cli-progress-wait.test.ts (2 tests) 3ms + ✓ tests/step-lease.test.ts (36 tests) 66639ms + ✓ f.run leases against the live kernel > enforces 10000 ms for 'sleep 5; printf ok' 5083ms + ✓ f.run leases against the live kernel > enforces 40000 ms for 'sleep 31; printf ok' 31231ms + ✓ f.run leases against the live kernel > enforces 30000 ms for 'sleep 31; printf ok' 30105ms + ↓ tests/run-digest-live.test.ts (1 test | 1 skipped) + ✓ tests/canonical-tree.test.ts (1 test) 2ms + ✓ tests/placement.test.ts (54 tests) 23ms + ✓ tests/communication-tools.test.ts (1 test) 71ms + ✓ tests/authored-admission.test.ts (2 tests) 3ms + ✓ tests/memory.test.ts (18 tests) 8ms + ✓ tests/worker-platform.test.ts (1 test) 3ms + ✓ tests/run-digest.test.ts (4 tests) 1615ms + ✓ digest run configuration refusals > reports config_invalid before fetching or starting a run for {invalid json 403ms + ✓ digest run configuration refusals > reports config_invalid before fetching or starting a run for {"deploy":{}} 396ms + ✓ digest run configuration refusals > reports config_invalid before fetching or starting a run for {"deploy":{"bucket":123}} 406ms + ✓ digest run configuration refusals > reports config_invalid before fetching or starting a run for {"deploy":{"bucket":""}} 410ms + ✓ tests/local-agent-live.test.ts (5 tests) 63796ms + ✓ built CLI local agent against a real daemon > dispatches through the wrapper and keeps --json stdout report-shaped 760ms + ✓ built CLI local agent against a real daemon > runs beyond the initial 30-second lease without a second invocation 35795ms + ✓ built CLI local agent against a real daemon > renders actual agent completion in text output 792ms + ✓ built CLI local agent against a real daemon > returns a failed run when the agent process fails 13531ms + ✓ built CLI local agent against a real daemon > refuses a workspace it cannot pin before invoking the agent 12917ms + +⎯⎯⎯⎯⎯⎯ Failed Suites 1 ⎯⎯⎯⎯⎯⎯⎯ + + FAIL tests/authored-node-runtime.test.ts [ tests/authored-node-runtime.test.ts ] +AssertionError: expected '1.3.6' to be '1.4.0' // Object.is equality + +Expected: "1.4.0" +Received: "1.3.6" + + ❯ tests/authored-node-runtime.test.ts:18:77 + 16| + 17| beforeAll(() => { + 18| expect(spawnSync(bun, ['--version'], { encoding: 'utf8' }).stdout.tr… + | ^ + 19| expect(existsSync(daemon), 'build the current kernel or set RELAYFLO… + 20| stage = mkdtempSync(join(tmpdir(), 'authored-standalone-build-')); + +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯[1/42]⎯ + +⎯⎯⎯⎯⎯⎯ Failed Tests 41 ⎯⎯⎯⎯⎯⎯⎯ + + FAIL tests/babysitter-native-extension.test.ts > native Babysitter extension > runs the exact published 2.0.26 native bytes in the isolated capability path +AssertionError: promise rejected "Error: Hosted extension sandbox exited wi… { code: '…' }" instead of resolving + ❯ tests/babysitter-native-extension.test.ts:153:7 + 151| return { receiptId: `bst_${'a'.repeat(64)}`, status: 'queued' … + 152| } }, + 153| })).resolves.toEqual({ completionReason: 'success', capabilityCall… + | ^ + 154| expect(calls).toEqual([{ delivery: { + 155| deliveryId: 'gh-delivery-7', provider: 'github', eventType: 'pul… + +Caused by: Error: Hosted extension sandbox exited without a valid completion (exit 1): bwrap: loopback: Failed RTM_NEWADDR: Operation not permitted + + ❯ refuse src/hosted-extension-protocol.ts:135:21 + ❯ ChildProcess. src/hosted-extension-protocol.ts:234:21 + +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯ +Serialized Error: { code: 'plugin_unsupported' } +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯[2/42]⎯ + + FAIL tests/cli.test.ts > flows run/resume CLI over the journal protocol > bounds a worker wait by its lease and reports what it is waiting for +AssertionError: expected +0 to be 1 // Object.is equality + +- Expected ++ Received + +- 1 ++ 0 + + ❯ tests/cli.test.ts:1089:18 + 1087| ], output.io); + 1088| + 1089| expect(code).toBe(1); + | ^ + 1090| expect(output.stderr.join('\n')).toContain('WAITING [worker_lease]… + 1091| expect(output.stderr.join('\n')).toContain(`until ${leaseDeadlineM… + +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯[3/42]⎯ + + FAIL tests/hosted-extension-isolation.test.ts > hosted extension capability isolation > executes the exact capability-only handler for a queued receipt + FAIL tests/hosted-extension-protocol.test.ts > hosted extension hostile protocol > uses captured JSON intrinsics for the complete parent boundary +Error: Hosted extension sandbox exited without a valid completion (exit 1): bwrap: loopback: Failed RTM_NEWADDR: Operation not permitted + + ❯ refuse src/hosted-extension-protocol.ts:135:21 + 133| : new PluginError('plugin_unsupported', 'Hosted capability rejec… + 134| const refuse = (message: string) => { + 135| const error = new PluginError('plugin_unsupported', message); + | ^ + 136| CHILD_PROCESS_KILL(child, 'SIGKILL'); + 137| if (capabilityState === 'pending') { + ❯ ChildProcess. src/hosted-extension-protocol.ts:234:21 + +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯[4/42]⎯ + + FAIL tests/hosted-extension-isolation.test.ts > hosted extension capability isolation > executes the exact capability-only handler for a duplicate receipt + FAIL tests/hosted-extension-isolation.test.ts > hosted extension capability isolation > denies ambient credentials, host files, writes, network, subprocesses, and undeclared context verbs + FAIL tests/hosted-extension-isolation.test.ts > hosted extension capability isolation > constructs adapter authority with the captured freeze intrinsic +Error: Hosted extension sandbox exited without a valid completion (exit 1): bwrap: loopback: Failed RTM_NEWADDR: Operation not permitted + + ❯ refuse src/hosted-extension-protocol.ts:135:21 + 133| : new PluginError('plugin_unsupported', 'Hosted capability rejec… + 134| const refuse = (message: string) => { + 135| const error = new PluginError('plugin_unsupported', message); + | ^ + 136| CHILD_PROCESS_KILL(child, 'SIGKILL'); + 137| if (capabilityState === 'pending') { + ❯ ChildProcess. src/hosted-extension-protocol.ts:234:21 + +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯[5/42]⎯ + + FAIL tests/hosted-extension-isolation.test.ts > hosted extension capability isolation > launches through the captured process primitive after builtin export synchronization +AssertionError: promise rejected "Error: Hosted extension sandbox exited wi… { code: '…' }" instead of resolving + ❯ tests/hosted-extension-isolation.test.ts:283:11 + 281| input: descriptor(), + 282| babysitterTurn: { queue: async () => ({ receiptId: 'receipt-… + 283| })).resolves.toEqual({ completionReason: 'success', capability… + | ^ + 284| } finally { + 285| process.execPath = originalExecPath; + +Caused by: Error: Hosted extension sandbox exited without a valid completion (exit 1): bwrap: loopback: Failed RTM_NEWADDR: Operation not permitted + + ❯ refuse src/hosted-extension-protocol.ts:135:21 + ❯ ChildProcess. src/hosted-extension-protocol.ts:234:21 + +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯ +Serialized Error: { code: 'plugin_unsupported' } +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯[6/42]⎯ + + FAIL tests/hosted-extension-isolation.test.ts > hosted extension capability isolation > ignores inherited launcher overrides and decodes manifests with the captured Buffer intrinsic +AssertionError: promise rejected "Error: Hosted extension sandbox exited wi… { code: '…' }" instead of resolving + ❯ tests/hosted-extension-isolation.test.ts:373:11 + 371| input: descriptor('delivery-options'), + 372| babysitterTurn: { queue: async () => ({ receiptId: 'receipt-… + 373| })).resolves.toEqual({ completionReason: 'success', capability… + | ^ + 374| } finally { + 375| Buffer.prototype.toString = bufferToString; + +Caused by: Error: Hosted extension sandbox exited without a valid completion (exit 1): bwrap: loopback: Failed RTM_NEWADDR: Operation not permitted + + ❯ refuse src/hosted-extension-protocol.ts:135:21 + ❯ ChildProcess. src/hosted-extension-protocol.ts:234:21 + +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯ +Serialized Error: { code: 'plugin_unsupported' } +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯[7/42]⎯ + + FAIL tests/hosted-extension-isolation.test.ts > hosted extension capability isolation > streams verified bytes when the live store is replaced and no writable staging path exists +AssertionError: promise rejected "Error: Hosted extension sandbox exited wi… { code: '…' }" instead of resolving + ❯ tests/hosted-extension-isolation.test.ts:431:7 + 429| return { receiptId: 'receipt-1', status: 'queued' }; + 430| } }, + 431| })).resolves.toEqual({ completionReason: 'success', capabilityCall… + | ^ + 432| expect(calls).toBe(1); + 433| expect(readFileSync(join(installed.directory, 'babysitter.flow.ts'… + +Caused by: Error: Hosted extension sandbox exited without a valid completion (exit 1): bwrap: loopback: Failed RTM_NEWADDR: Operation not permitted + + ❯ refuse src/hosted-extension-protocol.ts:135:21 + ❯ ChildProcess. src/hosted-extension-protocol.ts:234:21 + +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯ +Serialized Error: { code: 'plugin_unsupported' } +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯[8/42]⎯ + + FAIL tests/hosted-extension-isolation.test.ts > hosted extension capability isolation > mounts pinned private Surface bytes when the live package changes before launch +AssertionError: promise rejected "Error: Hosted extension sandbox exited wi… { code: '…' }" instead of resolving + ❯ tests/hosted-extension-isolation.test.ts:456:7 + 454| return { receiptId: 'receipt-1', status: 'queued' }; + 455| } }, + 456| })).resolves.toEqual({ completionReason: 'success', capabilityCall… + | ^ + 457| expect(calls).toBe(1); + 458| expect(readFileSync(join(surfaceRoot, 'dist/flow.js'), 'utf8')).to… + +Caused by: Error: Hosted extension sandbox exited without a valid completion (exit 1): bwrap: loopback: Failed RTM_NEWADDR: Operation not permitted + + ❯ refuse src/hosted-extension-protocol.ts:135:21 + ❯ ChildProcess. src/hosted-extension-protocol.ts:234:21 + +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯ +Serialized Error: { code: 'plugin_unsupported' } +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯[9/42]⎯ + + FAIL tests/hosted-extension-isolation.test.ts > hosted extension capability isolation > shields verified Surface files before async settlement +AssertionError: promise rejected "Error: Hosted extension sandbox exited wi… { code: '…' }" instead of resolving + ❯ tests/hosted-extension-isolation.test.ts:532:9 + 530| surfaceRoot, + 531| babysitterTurn: { queue: async () => ({ receiptId: 'receipt-1'… + 532| })).resolves.toEqual({ completionReason: 'success', capabilityCa… + | ^ + 533| } finally { + 534| if (previous === undefined) delete (Array.prototype as { then?: … + +Caused by: Error: Hosted extension sandbox exited without a valid completion (exit 1): bwrap: loopback: Failed RTM_NEWADDR: Operation not permitted + + ❯ refuse src/hosted-extension-protocol.ts:135:21 + ❯ ChildProcess. src/hosted-extension-protocol.ts:234:21 + +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯ +Serialized Error: { code: 'plugin_unsupported' } +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯[10/42]⎯ + + FAIL tests/hosted-extension-isolation.test.ts > hosted extension capability isolation > preserves a typed host refusal while disclosing only a fixed marker to the child +AssertionError: expected Error: Hosted extension sandbox exited wi… { code: '…' } to be Error: private Cloud policy detail { code: '…' } // Object.is equality + +- Expected ++ Received + +- [Error: private Cloud policy detail] ++ [Error: Hosted extension sandbox exited without a valid completion (exit 1): bwrap: loopback: Failed RTM_NEWADDR: Operation not permitted ++ ] + + ❯ tests/hosted-extension-isolation.test.ts:560:5 + 558| provider: 'github', eventType: 'pull_request.labeled', deliveryI… + 559| }); + 560| await expect(runVerifiedNativeExtensionSandbox({ + | ^ + 561| artifact: await artifact(source), manifest: validateFlowExtensio… + 562| babysitterTurn: { queue: async () => { throw refusal; } }, + +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯[11/42]⎯ + + FAIL tests/hosted-extension-isolation.test.ts > hosted extension capability isolation > enforces OS address-space and data bounds on native Buffer allocation +AssertionError: expected Error: Hosted extension sandbox exited wi… { code: '…' } to match object { code: 'plugin_unsupported', …(1) } + +- Expected ++ Received + +- Object { ++ PluginError { + "code": "plugin_unsupported", +- "message": StringMatching /(?:Failed to allocate memory|Array buffer allocation failed)/u, + } + + ❯ tests/hosted-extension-isolation.test.ts:646:5 + 644| }); + 645| let calls = 0; + 646| await expect(runVerifiedNativeExtensionSandbox({ + | ^ + 647| artifact: installed, + 648| manifest: validateFlowExtensionManifest(manifest()), + +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯[12/42]⎯ + + FAIL tests/hosted-extension-isolation.test.ts > hosted extension capability isolation > blocks extra handler fields and authority-bearing receipt fields at the parent port +AssertionError: expected Error: Hosted extension sandbox exited wi… { code: '…' } to match object { code: 'plugin_event_unroutable' } + +- Expected ++ Received + +- Object { +- "code": "plugin_event_unroutable", ++ PluginError { ++ "code": "plugin_unsupported", + } + + ❯ tests/hosted-extension-isolation.test.ts:675:5 + 673| }); + 674| let calls = 0; + 675| await expect(runVerifiedNativeExtensionSandbox({ + | ^ + 676| artifact: await artifact(source), manifest: validateFlowExtensio… + 677| babysitterTurn: { queue: async () => { calls += 1; return { rece… + +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯[13/42]⎯ + + FAIL tests/hosted-extension-isolation.test.ts > hosted extension capability isolation > writes the Surface manifest and protocol without inherited toJSON behavior +AssertionError: promise rejected "Error: Hosted extension sandbox exited wi… { code: '…' }" instead of resolving + ❯ tests/hosted-extension-isolation.test.ts:788:11 + 786| babysitterTurn: { queue: async () => ({ receiptId: 'receipt-… + 787| timeoutMs: 3_000, + 788| })).resolves.toEqual({ completionReason: 'success', capability… + | ^ + 789| } finally { + 790| if (previous === undefined) delete (Object.prototype as { toJS… + +Caused by: Error: Hosted extension sandbox exited without a valid completion (exit 1): bwrap: loopback: Failed RTM_NEWADDR: Operation not permitted + + ❯ refuse src/hosted-extension-protocol.ts:135:21 + ❯ ChildProcess. src/hosted-extension-protocol.ts:234:21 + +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯ +Serialized Error: { code: 'plugin_unsupported' } +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯[14/42]⎯ + + FAIL tests/hosted-extension-protocol.test.ts > hosted extension hostile protocol > rejects an import-time different PR frame with zero adapter calls + FAIL tests/hosted-extension-protocol.test.ts > hosted extension hostile protocol > rejects an import-time different delivery frame with zero adapter calls + FAIL tests/hosted-extension-protocol.test.ts > hosted extension hostile protocol > rejects an import-time different event frame with zero adapter calls +AssertionError: expected Error: Hosted extension sandbox exited wi… { code: '…' } to match object { code: 'plugin_event_unroutable' } + +- Expected ++ Received + +- Object { +- "code": "plugin_event_unroutable", ++ PluginError { ++ "code": "plugin_unsupported", + } + + ❯ tests/hosted-extension-protocol.test.ts:430:5 + 428| ])('rejects an import-time %s frame with zero adapter calls', async … + 429| let calls = 0; + 430| await expect(runVerifiedNativeExtensionSandbox({ + | ^ + 431| artifact: await artifact(hostileImport([frame, { type: 'error', … + 432| manifest: validateFlowExtensionManifest(manifest()), dispatch: d… + +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯[15/42]⎯ + + FAIL tests/hosted-extension-protocol.test.ts > hosted extension hostile protocol > rejects two forged calls after the authoritative first outcome settles + FAIL tests/hosted-extension-protocol.test.ts > hosted extension hostile protocol > waits for a pending adapter to reject after a forged child error + FAIL tests/hosted-extension-protocol.test.ts > hosted extension hostile protocol > waits for a pending adapter to resolve after a forged child error + FAIL tests/hosted-extension-protocol.test.ts > hosted extension hostile protocol > returns a typed adapter rejection even when the hostile child hangs +Error: hostile child did not invoke the adapter + ❯ Timeout._onTimeout tests/hosted-extension-protocol.test.ts:118:45 + 116| async function waitForInvocation(invoked: Promise): Promise((resolve, reject) => { + 118| const timeout = setTimeout(() => reject(new Error('hostile child d… + | ^ + 119| void invoked.then(() => { clearTimeout(timeout); resolve(); }, rej… + 120| }); + +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯[16/42]⎯ + + FAIL tests/live-kernel.test.ts > built flows CLI against live relayflowd > runs hn-monitor analyze-story end-to-end via a stub agent CLI (gate 2 clause 2 demo) +AssertionError: expected { …(12) } to match object { output: { …(3) }, …(1) } +(22 matching properties omitted from actual) + +- Expected ++ Received + + Object { +- "output": Object { +- "reasoning": "stub agent runtime — deterministic output for gate-2 clause-2 demo", +- "relevance_score": 5, +- "story_title": "stub", +- }, ++ "output": null, + "verification": Object { +- "gate": "json_schema", +- "verdict": "pass", ++ "gate": "execution", ++ "verdict": "fail", + }, + } + + ❯ tests/live-kernel.test.ts:657:36 + 655| && (entry as { step_id?: string }).step_id === 'analyze-story', + 656| ) as { payload: { output: unknown; verification: unknown } } | und… + 657| expect(stepCompleted?.payload).toMatchObject({ + | ^ + 658| output: { + 659| story_title: 'stub', + +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯[17/42]⎯ + + FAIL tests/live-kernel.test.ts > built flows CLI against live relayflowd > hn-monitor analyze-story FAILS verification when the CLI omits required schema fields +AssertionError: expected { …(12) } to match object { …(3) } +(21 matching properties omitted from actual) + +- Expected ++ Received + + Object { +- "completionReason": "retries_exhausted", ++ "completionReason": "worker_error", + "output": null, + "verification": Object { +- "gate": "json_schema", ++ "gate": "execution", + "verdict": "fail", + }, + } + + ❯ tests/live-kernel.test.ts:752:36 + 750| // its verification record names the json_schema rejection. The re… + 751| // parsed value is nulled before the completion is persisted. + 752| expect(stepCompleted?.payload).toMatchObject({ + | ^ + 753| completionReason: 'retries_exhausted', + 754| output: null, + +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯[18/42]⎯ + + FAIL tests/live-kernel.test.ts > built flows CLI against live relayflowd > agent step preserves the CliResult wrapper as output when the CLI emits non-JSON text +AssertionError: expected null not to be null + ❯ tests/live-kernel.test.ts:823:24 + 821| // here (parseJsonOutput returned null on non-JSON stdout) and + 822| // these assertions would all fail. + 823| expect(output).not.toBeNull(); + | ^ + 824| expect(output.exit_code).toBe(0); + 825| expect(output.stdout_tail).toContain('looked at the story'); + +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯[19/42]⎯ + + FAIL tests/live-kernel.test.ts > built flows CLI against live relayflowd > AgentWorker exposes wake_context to the CLI via RELAYFLOW_WAKE_CONTEXT env var (real analyzer prerequisite) +TypeError: Cannot read properties of null (reading 'story_title') + ❯ tests/live-kernel.test.ts:891:42 + 889| ) as { payload: { output: { story_title: string; reasoning: string… + 890| expect(stepCompleted).toBeDefined(); + 891| expect(stepCompleted!.payload.output.story_title).toBe(`echoed:${s… + | ^ + 892| expect(stepCompleted!.payload.output.reasoning).toContain(String(s… + 893| + +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯[20/42]⎯ + + FAIL tests/live-kernel.test.ts > built flows CLI against live relayflowd > AgentWorker leaves RELAYFLOW_WAKE_CONTEXT UNSET when the run has no wake_context (undefined-vs-null pin) +TypeError: Cannot read properties of null (reading 'env_present') + ❯ tests/live-kernel.test.ts:958:38 + 956| ) as { payload: { output: { env_present: boolean } } } | undefined; + 957| expect(completed).toBeDefined(); + 958| expect(completed!.payload.output.env_present).toBe(false); + | ^ + 959| + 960| delete process.env.RELAYFLOW_WAKE_CONTEXT; + +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯[21/42]⎯ + + FAIL tests/live-kernel.test.ts > built flows CLI against live relayflowd > AgentWorker leaves RELAYFLOW_MODEL UNSET when the step declares no model +TypeError: Cannot read properties of null (reading 'story_title') + ❯ tests/live-kernel.test.ts:1194:38 + 1192| expect(completed).toBeDefined(); + 1193| // UNSET, not EMPTY and not the leaked parent value. + 1194| expect(completed!.payload.output.story_title).toBe('model:UNSET'); + | ^ + 1195| + 1196| delete process.env.RELAYFLOW_MODEL; + +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯[22/42]⎯ + + FAIL tests/live-kernel.test.ts > built flows CLI against live relayflowd > hn-monitor analyze-story reaches done through the real Claude analyzer CLI +Error: LIVE_ANALYZER_UNAVAILABLE: "/home/daytona/.relayflow-v2-supervisor/durable/repository/testdata/preflight/analyze-story-claude-cli" does not identify as relayflows-agent-cli-v1 — failing because gate-2 acceptance requires the real analyzer to execute. Set RELAYFLOWS_ALLOW_ANALYZER_SKIP=1 only if this run is not gate evidence. + ❯ tests/live-kernel.test.ts:1223:15 + 1221| const notice = `LIVE_ANALYZER_UNAVAILABLE: ${readiness.detail}`; + 1222| if (process.env['RELAYFLOWS_ALLOW_ANALYZER_SKIP'] !== '1') { + 1223| throw new Error( + | ^ + 1224| `${notice} — failing because gate-2 acceptance requires the … + 1225| + 'Set RELAYFLOWS_ALLOW_ANALYZER_SKIP=1 only if this run is … + +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯[23/42]⎯ + + FAIL tests/live-kernel.test.ts > a relayflow can be scheduled: tick source against live relayflowd > a tick spawns a real run whose step reports the SCHEDULED instant +AssertionError: expected null to deeply equal { schedule_id: 'heartbeat-1m', …(3) } + +- Expected: +Object { + "lag_ms": 43000, + "schedule_id": "heartbeat-1m", + "scheduled_for_ms": 1764000000000, + "slot": 29400000, +} + ++ Received: +null + + ❯ tests/live-kernel.test.ts:1665:39 + 1663| // The bound: the run reports the grid instant and its own lag, so… + 1664| // backfilled run can tell it is running for a slot from the past. + 1665| expect(completed!.payload.output).toEqual({ + | ^ + 1666| schedule_id: 'heartbeat-1m', + 1667| slot: 29_400_000, + +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯[24/42]⎯ + + FAIL tests/provider-trigger-executor.test.ts > the kernel executes compiled 'app_mention' subscriptions with provider isolation and durable dedupe + FAIL tests/provider-trigger-executor.test.ts > the kernel executes compiled 'reaction_added' subscriptions with provider isolation and durable dedupe + FAIL tests/provider-trigger-executor.test.ts > the kernel executes compiled 'pull_request' subscriptions with provider isolation and durable dedupe +Error: spawnSync /home/daytona/.relayflow-v2-supervisor/durable/repository/kernel/target/debug/relayflowd ENOENT + ❯ submit tests/provider-trigger-executor.test.ts:43:89 + 41| steps: [{ id: 'effect', type: 'deterministic', command: `printf ac… + 42| })))); + 43| const submit = (envelope: unknown, key: string, executor = source.na… + | ^ + 44| '--data-dir', dir, 'run', spec, '--event', JSON.stringify({ type: … + 45| ], { encoding: 'utf8', stdio: 'pipe' })) as { matched: boolean; dedu… + ❯ tests/provider-trigger-executor.test.ts:50:12 + +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯[25/42]⎯ + + FAIL tests/webhook-live.test.ts > executes and deduplicates 'app_mention' only for its provider and matching payload + FAIL tests/webhook-live.test.ts > executes and deduplicates 'reaction_added' only for its provider and matching payload + FAIL tests/webhook-live.test.ts > executes and deduplicates 'pull_request' only for its provider and matching payload +Error: webhook integration timed out: spawn /home/daytona/.relayflow-v2-supervisor/durable/repository/kernel/target/debug/relayflowd ENOENT + ❯ until tests/webhook-live.test.ts:39:9 + 37| const deadline = Date.now() + 10_000; + 38| while (Date.now() < deadline) { if (await predicate()) return; await… + 39| throw new Error(`webhook integration timed out: ${detail()}`); + | ^ + 40| } + 41| async function daemon(dir: string): Promise { + ❯ daemon tests/webhook-live.test.ts:43:3 + ❯ tests/webhook-live.test.ts:100:3 + +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯[26/42]⎯ + + FAIL tests/webhook-live.test.ts > flows serve-webhook writes JSON before the daemon starts, then journals and archives exactly once +Error: webhook integration timed out: spawn /home/daytona/.relayflow-v2-supervisor/durable/repository/kernel/target/debug/relayflowd ENOENT + ❯ until tests/webhook-live.test.ts:39:9 + 37| const deadline = Date.now() + 10_000; + 38| while (Date.now() < deadline) { if (await predicate()) return; await… + 39| throw new Error(`webhook integration timed out: ${detail()}`); + | ^ + 40| } + 41| async function daemon(dir: string): Promise { + ❯ daemon tests/webhook-live.test.ts:43:3 + ❯ tests/webhook-live.test.ts:121:3 + +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯[27/42]⎯ + + FAIL tests/webhook-live.test.ts > replays a dropped file after SIGKILL before spawn +Error: webhook integration timed out: spawn /home/daytona/.relayflow-v2-supervisor/durable/repository/kernel/target/debug/relayflowd ENOENT + ❯ until tests/webhook-live.test.ts:39:9 + 37| const deadline = Date.now() + 10_000; + 38| while (Date.now() < deadline) { if (await predicate()) return; await… + 39| throw new Error(`webhook integration timed out: ${detail()}`); + | ^ + 40| } + 41| async function daemon(dir: string): Promise { + ❯ daemon tests/webhook-live.test.ts:43:3 + ❯ tests/webhook-live.test.ts:137:17 + +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯[28/42]⎯ + + FAIL tests/webhook-live.test.ts > resumes the same journal after SIGKILL after spawn and before acknowledgement +Error: webhook integration timed out: spawn /home/daytona/.relayflow-v2-supervisor/durable/repository/kernel/target/debug/relayflowd ENOENT + ❯ until tests/webhook-live.test.ts:39:9 + 37| const deadline = Date.now() + 10_000; + 38| while (Date.now() < deadline) { if (await predicate()) return; await… + 39| throw new Error(`webhook integration timed out: ${detail()}`); + | ^ + 40| } + 41| async function daemon(dir: string): Promise { + ❯ daemon tests/webhook-live.test.ts:43:3 + ❯ tests/webhook-live.test.ts:150:17 + +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯[29/42]⎯ + + FAIL tests/worker-lease-lost.test.ts > keeps an invalid lease deadline fatal +AssertionError: expected "spy" to be called 1 times, but got 0 times + ❯ tests/worker-lease-lost.test.ts:124:17 + 122| client.emit('step.dispatch', { ...dispatch, lease_deadline_ms: NaN }… + 123| await worker.close(); + 124| expect(fatal).toHaveBeenCalledTimes(1); + | ^ + 125| expect(client.close).toHaveBeenCalledTimes(1); + 126| }); + +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯[30/42]⎯ + +⎯⎯⎯⎯⎯⎯ Unhandled Errors ⎯⎯⎯⎯⎯⎯ + +Vitest caught 1 unhandled error during the test run. +This might cause false positive tests. Resolve unhandled errors to make sure your tests are not affected. + +⎯⎯⎯⎯ Unhandled Rejection ⎯⎯⎯⎯⎯ +Error: Hosted extension sandbox exited without a valid completion (exit 1): bwrap: loopback: Failed RTM_NEWADDR: Operation not permitted + + ❯ refuse src/hosted-extension-protocol.ts:135:21 + 133| : new PluginError('plugin_unsupported', 'Hosted capability rejec… + 134| const refuse = (message: string) => { + 135| const error = new PluginError('plugin_unsupported', message); + | ^ + 136| CHILD_PROCESS_KILL(child, 'SIGKILL'); + 137| if (capabilityState === 'pending') { + ❯ ChildProcess. src/hosted-extension-protocol.ts:234:21 + ❯ ChildProcess.emit node:events:520:22 + ❯ maybeClose node:internal/child_process:1084:16 + ❯ Process.ChildProcess._handle.onexit node:internal/child_process:304:5 + +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯ +Serialized Error: { code: 'plugin_unsupported' } +This error originated in "tests/hosted-extension-protocol.test.ts" test file. It doesn't mean the error was thrown inside the file itself, but while it was running. +The latest test that might've caused the error is "rejects two forged calls after the authoritative first outcome settles". It might mean one of the following: +- The error was thrown, while Vitest was running this test. +- If the error occurred after the test had been completed, this was the last documented test before it was thrown. +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯ + + Test Files 9 failed | 174 passed | 3 skipped (186) + Tests 41 failed | 2848 passed | 21 skipped (2910) + Errors 1 error + Start at 05:36:56 + Duration 244.33s (transform 2.94s, setup 0ms, collect 48.79s, tests 648.69s, environment 24ms, prepare 7.59s) + + +Exit code: 1 diff --git a/evidence/worker-lease-lost/restored-build.txt b/evidence/worker-lease-lost/restored-build.txt new file mode 100644 index 000000000..585c5975e --- /dev/null +++ b/evidence/worker-lease-lost/restored-build.txt @@ -0,0 +1,7 @@ +$ cd packages/sdk && npm run build + +> @relayflows/sdk@2.0.29 build +> tsc && node scripts/make-cli-executable.mjs + + +Exit code: 0 diff --git a/evidence/worker-lease-lost/resume-mutant.txt b/evidence/worker-lease-lost/resume-mutant.txt new file mode 100644 index 000000000..e0e5783b3 --- /dev/null +++ b/evidence/worker-lease-lost/resume-mutant.txt @@ -0,0 +1,32 @@ +$ cd packages/sdk && npx vitest run tests/resume-worker-lease.test.ts + + RUN v2.1.9 /home/daytona/.relayflow-v2-supervisor/durable/repository/packages/sdk + + ❯ tests/resume-worker-lease.test.ts (2 tests | 2 failed) 9ms + × resume handles LLM errors with leaseLost=false 7ms + → expected "spy" to be called at least once + × resume handles LLM errors with leaseLost=true 1ms + → expected "spy" to be called at least once + +⎯⎯⎯⎯⎯⎯⎯ Failed Tests 2 ⎯⎯⎯⎯⎯⎯⎯ + + FAIL tests/resume-worker-lease.test.ts > resume handles LLM errors with leaseLost=false + FAIL tests/resume-worker-lease.test.ts > resume handles LLM errors with leaseLost=true +AssertionError: expected "spy" to be called at least once + ❯ tests/resume-worker-lease.test.ts:32:37 + 30| const result = await resumeFlow('run', '/unused', { localAgent: true… + 31| expect(readAuthoredRootMetadata).toHaveBeenCalled(); + 32| expect(resumeDurableAuthoredFlow).toHaveBeenCalled(); + | ^ + 33| expect(result.exitCode).toBe(1); + 34| expect(JSON.stringify(result.report)).toContain(leaseLost ? 'connect… + +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯[1/2]⎯ + + Test Files 1 failed (1) + Tests 2 failed (2) + Start at 05:42:40 + Duration 1.16s (transform 542ms, setup 0ms, collect 985ms, tests 9ms, environment 0ms, prepare 45ms) + + +Exit code: 1 diff --git a/evidence/worker-lease-lost/resume-restored.txt b/evidence/worker-lease-lost/resume-restored.txt new file mode 100644 index 000000000..34a7f6be0 --- /dev/null +++ b/evidence/worker-lease-lost/resume-restored.txt @@ -0,0 +1,13 @@ +$ cd packages/sdk && npx vitest run tests/resume-worker-lease.test.ts + + RUN v2.1.9 /home/daytona/.relayflow-v2-supervisor/durable/repository/packages/sdk + + ✓ tests/resume-worker-lease.test.ts (2 tests) 4ms + + Test Files 1 passed (1) + Tests 2 passed (2) + Start at 05:42:42 + Duration 1.13s (transform 532ms, setup 0ms, collect 961ms, tests 4ms, environment 0ms, prepare 42ms) + + +Exit code: 0 diff --git a/evidence/worker-lease-lost/sandbox-probe.txt b/evidence/worker-lease-lost/sandbox-probe.txt new file mode 100644 index 000000000..cf633ab1d --- /dev/null +++ b/evidence/worker-lease-lost/sandbox-probe.txt @@ -0,0 +1,4 @@ +$ /usr/bin/bwrap --unshare-all --ro-bind / / -- /bin/true +bwrap: loopback: Failed RTM_NEWADDR: Operation not permitted + +Exit code: 1 diff --git a/evidence/worker-lease-lost/sweep-before.txt b/evidence/worker-lease-lost/sweep-before.txt new file mode 100644 index 000000000..e15f38050 --- /dev/null +++ b/evidence/worker-lease-lost/sweep-before.txt @@ -0,0 +1,70 @@ +$ cd packages/sdk && npx vitest run tests/worker-lease-sweep.test.ts + + RUN v2.1.9 /home/daytona/.relayflow-v2-supervisor/durable/repository/packages/sdk + +(node:12534) PromiseRejectionHandledWarning: Promise rejection was handled asynchronously (rejection id: 3) +(Use `node --trace-warnings ...` to show where the warning was created) + ❯ tests/worker-lease-sweep.test.ts (2 tests | 1 failed) 10ms + × waits through an expired snapshot until the kernel retries and completes 8ms + → promise rejected "Error: worker lease for step "answer" exp…" instead of resolving + +⎯⎯⎯⎯⎯⎯⎯ Failed Tests 1 ⎯⎯⎯⎯⎯⎯⎯ + + FAIL tests/worker-lease-sweep.test.ts > waits through an expired snapshot until the kernel retries and completes +AssertionError: promise rejected "Error: worker lease for step "answer" exp…" instead of resolving + ❯ tests/worker-lease-sweep.test.ts:33:37 + 31| .mockResolvedValue({ run_id: 'run', status: 'completed', completio… + 32| const execution = run(); + 33| const assertion = expect(execution).resolves.toMatchObject({ exitCod… + | ^ + 34| await vi.advanceTimersByTimeAsync(100); + 35| await assertion; + +Caused by: Error: worker lease for step "answer" expired at 1790141762175 without completion + ❯ waitForRunningStep src/cli/run.ts:773:13 + ❯ Module.classifyOutcome src/cli/run.ts:584:13 + +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯[1/1]⎯ + +⎯⎯⎯⎯⎯⎯ Unhandled Errors ⎯⎯⎯⎯⎯⎯ + +Vitest caught 1 unhandled error during the test run. +This might cause false positive tests. Resolve unhandled errors to make sure your tests are not affected. + +⎯⎯⎯⎯ Unhandled Rejection ⎯⎯⎯⎯⎯ +AssertionError: promise rejected "Error: worker lease for step "answer" exp…" instead of resolving + ❯ _Assertion.__VITEST_RESOLVES__ node_modules/@vitest/expect/dist/index.js:1746:21 + ❯ _Assertion.propertyGetter node_modules/chai/index.js:1557:27 + ❯ Object.proxyGetter [as get] node_modules/chai/index.js:1645:22 + ❯ tests/worker-lease-sweep.test.ts:33:37 + 31| .mockResolvedValue({ run_id: 'run', status: 'completed', completio… + 32| const execution = run(); + 33| const assertion = expect(execution).resolves.toMatchObject({ exitCod… + | ^ + 34| await vi.advanceTimersByTimeAsync(100); + 35| await assertion; + ❯ node_modules/@vitest/runner/dist/index.js:146:14 + ❯ node_modules/@vitest/runner/dist/index.js:533:11 + ❯ runWithTimeout node_modules/@vitest/runner/dist/index.js:39:7 + ❯ runTest node_modules/@vitest/runner/dist/index.js:1056:17 + ❯ processTicksAndRejections node:internal/process/task_queues:104:5 + +This error originated in "tests/worker-lease-sweep.test.ts" test file. It doesn't mean the error was thrown inside the file itself, but while it was running. +The latest test that might've caused the error is "waits through an expired snapshot until the kernel retries and completes". It might mean one of the following: +- The error was thrown, while Vitest was running this test. +- If the error occurred after the test had been completed, this was the last documented test before it was thrown. +Caused by: Error: worker lease for step "answer" expired at 1790141762175 without completion + ❯ waitForRunningStep src/cli/run.ts:773:13 + ❯ Module.classifyOutcome src/cli/run.ts:584:13 + ❯ processTicksAndRejections node:internal/process/task_queues:104:5 + +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯ + + Test Files 1 failed (1) + Tests 1 failed | 1 passed (2) + Errors 1 error + Start at 05:36:00 + Duration 1.28s (transform 652ms, setup 0ms, collect 1.11s, tests 10ms, environment 0ms, prepare 41ms) + + +Exit code: 1 diff --git a/evidence/worker-lease-lost/sweep-mutant.txt b/evidence/worker-lease-lost/sweep-mutant.txt new file mode 100644 index 000000000..7b4023b2e --- /dev/null +++ b/evidence/worker-lease-lost/sweep-mutant.txt @@ -0,0 +1,34 @@ +$ cd packages/sdk && npx vitest run tests/worker-lease-sweep.test.ts + + RUN v2.1.9 /home/daytona/.relayflow-v2-supervisor/durable/repository/packages/sdk + + ❯ tests/worker-lease-sweep.test.ts (2 tests | 1 failed) 9ms + × waits through an expired snapshot until the kernel retries and completes 7ms + → promise rejected "Error: worker lease for step "answer" exp…" instead of resolving + +⎯⎯⎯⎯⎯⎯⎯ Failed Tests 1 ⎯⎯⎯⎯⎯⎯⎯ + + FAIL tests/worker-lease-sweep.test.ts > waits through an expired snapshot until the kernel retries and completes +AssertionError: promise rejected "Error: worker lease for step "answer" exp…" instead of resolving + ❯ tests/worker-lease-sweep.test.ts:33:37 + 31| .mockResolvedValue({ run_id: 'run', status: 'completed', completio… + 32| const execution = run(); + 33| const assertion = expect(execution).resolves.toMatchObject({ exitCod… + | ^ + 34| await Promise.all([assertion, vi.advanceTimersByTimeAsync(100)]); + 35| expect(client.runResume).toHaveBeenCalledTimes(2); + +Caused by: Error: worker lease for step "answer" expired at 1790142154876 without completion + ❯ waitForRunningStep src/cli/run.ts:776:13 + ❯ Module.classifyOutcome src/cli/run.ts:584:13 + ❯ tests/worker-lease-sweep.test.ts:34:3 + +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯[1/1]⎯ + + Test Files 1 failed (1) + Tests 1 failed | 1 passed (2) + Start at 05:42:33 + Duration 1.29s (transform 656ms, setup 0ms, collect 1.12s, tests 9ms, environment 0ms, prepare 44ms) + + +Exit code: 1 diff --git a/evidence/worker-lease-lost/sweep-restored.txt b/evidence/worker-lease-lost/sweep-restored.txt new file mode 100644 index 000000000..0edea990f --- /dev/null +++ b/evidence/worker-lease-lost/sweep-restored.txt @@ -0,0 +1,13 @@ +$ cd packages/sdk && npx vitest run tests/worker-lease-sweep.test.ts + + RUN v2.1.9 /home/daytona/.relayflow-v2-supervisor/durable/repository/packages/sdk + + ✓ tests/worker-lease-sweep.test.ts (2 tests) 7ms + + Test Files 1 passed (1) + Tests 2 passed (2) + Start at 05:42:35 + Duration 1.38s (transform 685ms, setup 0ms, collect 1.21s, tests 7ms, environment 0ms, prepare 45ms) + + +Exit code: 0 diff --git a/evidence/worker-lease-lost/terminal-mutant.txt b/evidence/worker-lease-lost/terminal-mutant.txt new file mode 100644 index 000000000..e57a711ec --- /dev/null +++ b/evidence/worker-lease-lost/terminal-mutant.txt @@ -0,0 +1,68 @@ +$ cd packages/sdk && npx vitest run tests/worker-lease-lost.test.ts -t run_terminal + + RUN v2.1.9 /home/daytona/.relayflow-v2-supervisor/durable/repository/packages/sdk + + ❯ tests/worker-lease-lost.test.ts (16 tests | 2 failed | 14 skipped) 9ms + × agent stale lease subscriber > drops stepComplete run_terminal for a released attempt 7ms + → expected "spy" to not be called at all, but actually been called 1 times + +Received: + + 1st spy call: + + Array [ + [JournalProtocolError: run_terminal: attempt has no active worker lease], + ] + + +Number of calls: 1 + + × llm stale lease subscriber > drops stepComplete run_terminal for a released attempt 1ms + → expected "spy" to not be called at all, but actually been called 1 times + +Received: + + 1st spy call: + + Array [ + [JournalProtocolError: run_terminal: attempt has no active worker lease], + ] + + +Number of calls: 1 + + +⎯⎯⎯⎯⎯⎯⎯ Failed Tests 2 ⎯⎯⎯⎯⎯⎯⎯ + + FAIL tests/worker-lease-lost.test.ts > agent stale lease subscriber > drops stepComplete run_terminal for a released attempt + FAIL tests/worker-lease-lost.test.ts > llm stale lease subscriber > drops stepComplete run_terminal for a released attempt +AssertionError: expected "spy" to not be called at all, but actually been called 1 times + +Received: + + 1st spy call: + + Array [ + [JournalProtocolError: run_terminal: attempt has no active worker lease], + ] + + +Number of calls: 1 + + ❯ tests/worker-lease-lost.test.ts:67:30 + 65| client.emit('step.dispatch', dispatch); + 66| await worker.close(); + 67| expect(client.close).not.toHaveBeenCalled(); + | ^ + 68| expect(fatal).not.toHaveBeenCalled(); + 69| expect(warning).toHaveBeenCalledWith(expect.stringContaining(error… + +⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯[1/2]⎯ + + Test Files 1 failed (1) + Tests 2 failed | 14 skipped (16) + Start at 05:42:22 + Duration 895ms (transform 359ms, setup 0ms, collect 714ms, tests 9ms, environment 0ms, prepare 52ms) + + +Exit code: 1 diff --git a/evidence/worker-lease-lost/terminal-restored.txt b/evidence/worker-lease-lost/terminal-restored.txt new file mode 100644 index 000000000..2759cb2a0 --- /dev/null +++ b/evidence/worker-lease-lost/terminal-restored.txt @@ -0,0 +1,13 @@ +$ cd packages/sdk && npx vitest run tests/worker-lease-lost.test.ts -t run_terminal + + RUN v2.1.9 /home/daytona/.relayflow-v2-supervisor/durable/repository/packages/sdk + + ✓ tests/worker-lease-lost.test.ts (16 tests | 14 skipped) 7ms + + Test Files 1 passed (1) + Tests 2 passed | 14 skipped (16) + Start at 05:42:24 + Duration 890ms (transform 347ms, setup 0ms, collect 716ms, tests 7ms, environment 0ms, prepare 48ms) + + +Exit code: 0 diff --git a/evidence/worker-lease-lost/tsconfig.tests.json b/evidence/worker-lease-lost/tsconfig.tests.json new file mode 100644 index 000000000..559f56917 --- /dev/null +++ b/evidence/worker-lease-lost/tsconfig.tests.json @@ -0,0 +1,17 @@ +{ + "extends": "../../packages/sdk/tsconfig.tests.json", + "include": [ + "../../packages/sdk/src/**/*.ts", + "../../packages/sdk/tests/worker-lease-lost.test.ts", + "../../packages/sdk/tests/worker-lease-lost-live.test.ts", + "../../packages/sdk/tests/worker-lease-sweep.test.ts", + "../../packages/sdk/tests/resume-worker-lease.test.ts", + "../../packages/sdk/tests/direct-run-worker-lease.test.ts" + ], + "compilerOptions": { + "typeRoots": [ + "../../packages/sdk/node_modules/@types", + "../../packages/sdk/node_modules" + ] + } +} From cc52e1fb548c0fcf617c367157d2d6ac99abbe9b Mon Sep 17 00:00:00 2001 From: Relayflow Date: Wed, 23 Sep 2026 06:11:20 +0000 Subject: [PATCH 11/13] test(sdk): gate the lease regression files in typecheck:tests The five worker-lease regression files were not in tsconfig.tests.json, whose `include` names its files rather than globbing, so `npm run typecheck:tests` never saw them. Add them; the evidence directory's supplemental config existed only to cover that gap and goes with them. Also record what the live-kernel failures in this directory's transcripts actually were. They are not the branch: /home/daytona/package.json sits above the checkout declaring "type": "commonjs", which turns off Node's module-syntax detection for testdata/preflight's extensionless ESM agent-CLI fixtures. Each fixture then exits 0 having written nothing and every agent step it drives fails its execution gate. Declaring that one directory ESM restores the condition a GitHub runner has and the whole suite passes on this branch unchanged; the transcript carries the reproduction and the passing run. Co-Authored-By: Claude Opus 5 --- evidence/worker-lease-lost/README.md | 17 +++- .../live-kernel-module-type.txt | 80 +++++++++++++++++++ evidence/worker-lease-lost/new-test-types.txt | 7 +- .../worker-lease-lost/tsconfig.tests.json | 17 ---- packages/sdk/tsconfig.tests.json | 7 +- 5 files changed, 106 insertions(+), 22 deletions(-) create mode 100644 evidence/worker-lease-lost/live-kernel-module-type.txt delete mode 100644 evidence/worker-lease-lost/tsconfig.tests.json diff --git a/evidence/worker-lease-lost/README.md b/evidence/worker-lease-lost/README.md index e4a917660..24e4f2d77 100644 --- a/evidence/worker-lease-lost/README.md +++ b/evidence/worker-lease-lost/README.md @@ -35,9 +35,9 @@ whole-filter reversion. - [Final npm test](npm-test-final.txt): final source, Rust on PATH, explicit RELAYFLOWD_BIN pointing at this checkout's build. Full captured output. -- [New test typecheck](new-test-types.txt): `npx tsc -p ../../evidence/worker-lease-lost/tsconfig.tests.json` - from packages/sdk. The supplemental config includes every new regression file; - the repository's existing test typecheck only enumerates selected files. +- [New test typecheck](new-test-types.txt): `npm run typecheck:tests` from + packages/sdk. That config enumerates its files by name rather than globbing, + so every new regression file is listed in it; the gate covers them. - [Sandbox probe](sandbox-probe.txt): direct OS isolation probe. ## Development transcripts @@ -67,3 +67,14 @@ RELAYFLOWD_BIN as the final full suite. The same eight live-kernel tests failed on the original SDK source. This comparison is limited to that suite; it is not a baseline full-suite run. + +Those failures have since been traced to the machine rather than to either +source revision: `/home/daytona/package.json` sits ABOVE the checkout and +declares `"type": "commonjs"`, which turns off Node's module-syntax detection +for `testdata/preflight`'s extensionless ESM agent-CLI fixtures. Each fixture +then exits 0 having written nothing, and every agent step it drives fails its +execution gate with `output: null`. Declaring that one directory ESM restores +the condition a GitHub runner has, and the whole live-kernel suite passes on +this branch unchanged. + +- [Root cause, minimal reproduction and the passing suite](live-kernel-module-type.txt) diff --git a/evidence/worker-lease-lost/live-kernel-module-type.txt b/evidence/worker-lease-lost/live-kernel-module-type.txt new file mode 100644 index 000000000..7679a33ab --- /dev/null +++ b/evidence/worker-lease-lost/live-kernel-module-type.txt @@ -0,0 +1,80 @@ +Root cause of the live-kernel failures in npm-test-final.txt and +baseline-live-kernel.txt: this machine, not the branch. + +testdata/preflight's agent-CLI fixtures are extensionless files containing an +ESM `import`. Node runs them only because module-syntax detection applies when +no package.json governs the file. This machine has /home/daytona/package.json +declaring "type": "commonjs" ABOVE the checkout, which turns detection off. + +$ cat /home/daytona/package.json | grep '"type"' + "type": "commonjs", + +$ ls /home/daytona/.relayflow-v2-supervisor/durable/repository/package.json +ls: cannot access '/home/daytona/.relayflow-v2-supervisor/durable/repository/package.json': No such file or directory + +Minimal reproduction: the same extensionless ESM script under each "type". +$ mkdir -p /tmp/cjstest && printf 'export const x=5;\n' > /tmp/cjstest/zz-mod.mjs +$ printf '#!/usr/bin/env node\nimport {x} from "./zz-mod.mjs";\nprocess.stderr.write("RAN "+x+"\\n");\n' > /tmp/cjstest/zz2 + +$ printf '{"name":"x","type":"commonjs"}\n' > /tmp/cjstest/package.json && node /tmp/cjstest/zz2; echo "exit=$?" +exit=0 + (no output: the module body never runs, and the process exits 0) + +$ printf '{"name":"x","type":"module"}\n' > /tmp/cjstest/package.json && node /tmp/cjstest/zz2; echo "exit=$?" +RAN 5 +exit=0 + +Same shape for the real fixture: exit 0, zero bytes, so every agent step it +drives fails its execution gate with output null. +$ printf '{"protocol":"relayflows-agent-cli-v1","instruction":"hi"}' | ./testdata/preflight/analyze-story-stub-cli --relayflows-adapter-v1 | od -c +0000000 + +Declaring the fixture directory ESM restores the CI condition, and the whole +live-kernel suite passes on this branch unchanged. +$ printf '{"type":"module"}\n' > testdata/preflight/package.json +$ cd packages/sdk && RELAYFLOWS_ALLOW_ANALYZER_SKIP=1 RELAYFLOWD_BIN=/kernel/target/release/relayflowd npx vitest run tests/live-kernel.test.ts + + RUN v2.1.9 /home/daytona/.relayflow-v2-supervisor/durable/repository/packages/sdk + +stdout | tests/live-kernel.test.ts +LIVE_KERNEL relayflowd=/home/daytona/.relayflow-v2-supervisor/durable/repository/kernel/target/release/relayflowd +LIVE_KERNEL flows=/home/daytona/.relayflow-v2-supervisor/durable/repository/packages/sdk/dist/cli.js + +stderr | tests/live-kernel.test.ts > built flows CLI against live relayflowd > hn-monitor analyze-story reaches done through the real Claude analyzer CLI +LIVE_ANALYZER_UNAVAILABLE: "/home/daytona/.relayflow-v2-supervisor/durable/repository/testdata/preflight/analyze-story-claude-cli auth status" exited 1: analyze-story-claude-cli: "claude -p --model claude-haiku-4-5-20251001" exited 1: Ignoring 1 permissions.allow entry from .claude/settings.json: this workspace has not been trusted. Run Claude Code interactively here once and accept the trust dialog, or set projects["/home/daytona/.relayflow-v2-supervisor/durable/repository"].hasTrustDialogAccepted: true in /home/daytona/.claude.json. — SKIPPING. This skip is diagnostics, not gate-2 acceptance evidence. + +stdout | tests/live-kernel.test.ts > surface resume after a real daemon kill > resumes a three-step run with each successful completion exactly once +LIVE_KERNEL kill -9 pid=96106 run=01M36E3SCANHX97HGGW2WK0KNW while step=two state=Running + + ✓ tests/live-kernel.test.ts (31 tests | 1 skipped) 52437ms + ✓ built flows CLI against live relayflowd > twenty-six-step reuses 25 durable completions after editing the failed final step 1775ms + ✓ built flows CLI against live relayflowd > runs rung (a), parks rung (b), and keeps JSON report-shaped 2588ms + ✓ built flows CLI against live relayflowd > allows a deterministic run to exceed the bounded request timeout 32471ms + ✓ built flows CLI against live relayflowd > follows a live worker dispatch through flows run 557ms + ✓ built flows CLI against live relayflowd > runs an agent CLI end to end through the SDK worker 354ms + ✓ built flows CLI against live relayflowd > f.agent lowers to a real agent step and dispatches through a live worker 521ms + ✓ built flows CLI against live relayflowd > can always get a parked run to a late-attaching worker 5551ms + ✓ built flows CLI against live relayflowd > reports a real manual-recovery NeedsHuman state as parked 439ms + ✓ built flows CLI against live relayflowd > runs hn-monitor analyze-story end-to-end via a stub agent CLI (gate 2 clause 2 demo) 456ms + ✓ built flows CLI against live relayflowd > hn-monitor analyze-story FAILS verification when the CLI omits required schema fields 471ms + ✓ built flows CLI against live relayflowd > agent step preserves the CliResult wrapper as output when the CLI emits non-JSON text 448ms + ✓ built flows CLI against live relayflowd > AgentWorker exposes wake_context to the CLI via RELAYFLOW_WAKE_CONTEXT env var (real analyzer prerequisite) 435ms + ✓ built flows CLI against live relayflowd > AgentWorker leaves RELAYFLOW_WAKE_CONTEXT UNSET when the run has no wake_context (undefined-vs-null pin) 416ms + ✓ built flows CLI against live relayflowd > AgentWorker passes a declared model to an identified wrapper as RELAYFLOW_MODEL 456ms + ✓ built flows CLI against live relayflowd > AgentWorker refuses a nonconforming journal-submitted wrapper before exposing RELAYFLOW_MODEL 403ms + ✓ built flows CLI against live relayflowd > AgentWorker executes the raw claude adapter with its real model flag 357ms + ✓ built flows CLI against live relayflowd > AgentWorker executes the raw codex adapter with its real model flag 398ms + ✓ built flows CLI against live relayflowd > AgentWorker leaves RELAYFLOW_MODEL UNSET when the step declares no model 424ms + ✓ built flows CLI against live relayflowd > preflights before journaling and names an unreachable socket 846ms + ✓ built flows CLI against live relayflowd > starts exactly one daemon when two runs race for one empty data dir 467ms + ✓ surface resume after a real daemon kill > resumes a three-step run with each successful completion exactly once 994ms + ✓ a relayflow can be scheduled: tick source against live relayflowd > a tick spawns a real run whose step reports the SCHEDULED instant 394ms + + Test Files 1 passed (1) + Tests 30 passed | 1 skipped (31) + Start at 06:08:18 + Duration 53.86s (transform 748ms, setup 0ms, collect 1.26s, tests 52.44s, environment 0ms, prepare 42ms) + + +Exit code: 0 +$ rm -f testdata/preflight/package.json # left out of the commit diff --git a/evidence/worker-lease-lost/new-test-types.txt b/evidence/worker-lease-lost/new-test-types.txt index f471084b1..2ddd88f37 100644 --- a/evidence/worker-lease-lost/new-test-types.txt +++ b/evidence/worker-lease-lost/new-test-types.txt @@ -1,3 +1,8 @@ -$ cd packages/sdk && npx tsc -p ../../evidence/worker-lease-lost/tsconfig.tests.json +$ cd packages/sdk && npm run typecheck:tests + + +> @relayflows/sdk@2.0.29 typecheck:tests +> tsc -p tsconfig.tests.json + Exit code: 0 diff --git a/evidence/worker-lease-lost/tsconfig.tests.json b/evidence/worker-lease-lost/tsconfig.tests.json deleted file mode 100644 index 559f56917..000000000 --- a/evidence/worker-lease-lost/tsconfig.tests.json +++ /dev/null @@ -1,17 +0,0 @@ -{ - "extends": "../../packages/sdk/tsconfig.tests.json", - "include": [ - "../../packages/sdk/src/**/*.ts", - "../../packages/sdk/tests/worker-lease-lost.test.ts", - "../../packages/sdk/tests/worker-lease-lost-live.test.ts", - "../../packages/sdk/tests/worker-lease-sweep.test.ts", - "../../packages/sdk/tests/resume-worker-lease.test.ts", - "../../packages/sdk/tests/direct-run-worker-lease.test.ts" - ], - "compilerOptions": { - "typeRoots": [ - "../../packages/sdk/node_modules/@types", - "../../packages/sdk/node_modules" - ] - } -} diff --git a/packages/sdk/tsconfig.tests.json b/packages/sdk/tsconfig.tests.json index a58e1106a..8305cc33a 100644 --- a/packages/sdk/tsconfig.tests.json +++ b/packages/sdk/tsconfig.tests.json @@ -57,7 +57,12 @@ "tests/agent-transcript.test.ts", "tests/agent-transcript-live.test.ts", "tests/worker-transcript.test.ts", - "tests/step-failure-diagnostic.test.ts" + "tests/step-failure-diagnostic.test.ts", + "tests/worker-lease-lost.test.ts", + "tests/worker-lease-lost-live.test.ts", + "tests/worker-lease-sweep.test.ts", + "tests/resume-worker-lease.test.ts", + "tests/direct-run-worker-lease.test.ts" ], "exclude": ["node_modules", "dist"] } From 2dfb9519508ffc4792f3963c17f09e71452793fd Mon Sep 17 00:00:00 2001 From: Relayflow Lead Date: Wed, 23 Sep 2026 00:00:06 -0700 Subject: [PATCH 12/13] fix(sdk): fail closed when a worker error arrives without its dispatch onWorkerFailure downgrades a lease loss to a warning only when it can name the attempt the kernel now owns. Every emitter passes the dispatch today; this keeps a future one from turning a lease loss into a TypeError. Co-Authored-By: Claude Opus 5.5 (1M context) --- packages/sdk/src/worker-lease.ts | 5 +++-- 1 file changed, 3 insertions(+), 2 deletions(-) diff --git a/packages/sdk/src/worker-lease.ts b/packages/sdk/src/worker-lease.ts index 184850c7c..a1ef1c7db 100644 --- a/packages/sdk/src/worker-lease.ts +++ b/packages/sdk/src/worker-lease.ts @@ -16,8 +16,9 @@ export function isLeaseLost(error: unknown): boolean { } export function onWorkerFailure(label: string, fatal: (error: unknown) => void) { - return (error: unknown, dispatch: StepDispatchEvent): void => { - if (!isLeaseLost(error)) { fatal(error); return; } + return (error: unknown, dispatch?: StepDispatchEvent): void => { + // Without the dispatch there is no attempt to hand back to the kernel: fail closed. + if (!isLeaseLost(error) || dispatch === undefined) { fatal(error); return; } // The kernel owns this attempt's fate; its journal supplies the run outcome. // stderr keeps this diagnostic out of structured reports on stdout. process.emitWarning( From cebdc0adb60b9fcbafb2f7dd92faa3831c013f90 Mon Sep 17 00:00:00 2001 From: Relayflow Lead Date: Wed, 23 Sep 2026 00:16:19 -0700 Subject: [PATCH 13/13] test(sdk): regenerate lease-lost evidence on the rebased head mutate.py matches the fail-closed predicate; all seven mutations reproduce (mutant exit 1, restored exit 0). baseline.py compares against the rebased parent instead of a hard-coded commit and takes RELAYFLOWD_BIN from the caller: the live-kernel suite passes 31/31 at base 2e2043f and at this head. Co-Authored-By: Claude Opus 5.5 (1M context) --- evidence/worker-lease-lost/README.md | 49 ++-- evidence/worker-lease-lost/baseline-build.txt | 3 +- .../worker-lease-lost/baseline-comparison.txt | 13 - .../baseline-live-kernel.txt | 230 +++--------------- evidence/worker-lease-lost/baseline.py | 10 +- evidence/worker-lease-lost/direct-mutant.txt | 12 +- .../worker-lease-lost/direct-restored.txt | 8 +- evidence/worker-lease-lost/fatal-mutant.txt | 10 +- evidence/worker-lease-lost/fatal-restored.txt | 8 +- evidence/worker-lease-lost/filter-mutant.txt | 14 +- .../worker-lease-lost/filter-restored.txt | 8 +- .../worker-lease-lost/live-kernel-head.txt | 4 + evidence/worker-lease-lost/live-mutant.txt | 51 ++-- evidence/worker-lease-lost/live-restored.txt | 13 +- evidence/worker-lease-lost/mutate.py | 10 +- evidence/worker-lease-lost/mutations.txt | 20 +- evidence/worker-lease-lost/restored-build.txt | 3 +- evidence/worker-lease-lost/resume-mutant.txt | 10 +- .../worker-lease-lost/resume-restored.txt | 8 +- evidence/worker-lease-lost/sweep-mutant.txt | 16 +- evidence/worker-lease-lost/sweep-restored.txt | 8 +- .../worker-lease-lost/terminal-mutant.txt | 10 +- .../worker-lease-lost/terminal-restored.txt | 8 +- 23 files changed, 187 insertions(+), 339 deletions(-) delete mode 100644 evidence/worker-lease-lost/baseline-comparison.txt create mode 100644 evidence/worker-lease-lost/live-kernel-head.txt diff --git a/evidence/worker-lease-lost/README.md b/evidence/worker-lease-lost/README.md index 24e4f2d77..3e6bfe743 100644 --- a/evidence/worker-lease-lost/README.md +++ b/evidence/worker-lease-lost/README.md @@ -52,29 +52,26 @@ whole-filter reversion. variants before adding the real-kernel heartbeat case. The restored live transcript above includes all three. -## Baseline comparison for live-kernel failures - -`python3 evidence/worker-lease-lost/baseline.py` replaces only changed SDK source -files with their original bytes from `f6ece41`, rebuilds, and runs the unchanged -live-kernel suite. It restores the implementation byte-for-byte in `finally`, -asserts equality, and rebuilds it. The script uses the same PATH and -RELAYFLOWD_BIN as the final full suite. - -- [Baseline build](baseline-build.txt) -- [Baseline live-kernel command and full output](baseline-live-kernel.txt) -- [Matching failed test names](baseline-comparison.txt) -- [Restored implementation build](restored-build.txt) - -The same eight live-kernel tests failed on the original SDK source. -This comparison is limited to that suite; it is not a baseline full-suite run. - -Those failures have since been traced to the machine rather than to either -source revision: `/home/daytona/package.json` sits ABOVE the checkout and -declares `"type": "commonjs"`, which turns off Node's module-syntax detection -for `testdata/preflight`'s extensionless ESM agent-CLI fixtures. Each fixture -then exits 0 having written nothing, and every agent step it drives fails its -execution gate with `output: null`. Declaring that one directory ESM restores -the condition a GitHub runner has, and the whole live-kernel suite passes on -this branch unchanged. - -- [Root cause, minimal reproduction and the passing suite](live-kernel-module-type.txt) +## Baseline comparison for the live-kernel suite + +`python3 evidence/worker-lease-lost/baseline.py` replaces only the SDK source +files this change touches with their bytes at the **rebased parent** +(`git merge-base HEAD origin/main`; `BASE=` overrides), rebuilds, and +runs the unchanged live-kernel suite. It restores the implementation +byte-for-byte in `finally`, asserts equality, and rebuilds it. Each transcript +records the base it used. + +Rerun after the rebase onto `2e2043f` (macOS, with `packages/sdk` built against +the workspace `@relayflows/surface`, as CI does): + +- [Baseline build](baseline-build.txt): exit 0. +- [Baseline live-kernel](baseline-live-kernel.txt): `Tests 31 passed (31)` at + base `2e2043f`. +- [Live-kernel at this head](live-kernel-head.txt): `Tests 31 passed (31)`. +- [Restored implementation build](restored-build.txt): exit 0. + +The eight live-kernel failures the factory first reported came from its +sandbox, not from either revision. `/home/daytona/package.json` sits above the +checkout and declares `"type": "commonjs"`, which breaks +`testdata/preflight`'s extensionless ESM fixtures. On a clean machine the +suite passes before and after this change. diff --git a/evidence/worker-lease-lost/baseline-build.txt b/evidence/worker-lease-lost/baseline-build.txt index 585c5975e..721d8bbfb 100644 --- a/evidence/worker-lease-lost/baseline-build.txt +++ b/evidence/worker-lease-lost/baseline-build.txt @@ -1,6 +1,7 @@ +# base 2e2043f4c82f1a5c83b090d8a3a82c2d067a79fe $ cd packages/sdk && npm run build -> @relayflows/sdk@2.0.29 build +> @relayflows/sdk@2.0.30 build > tsc && node scripts/make-cli-executable.mjs diff --git a/evidence/worker-lease-lost/baseline-comparison.txt b/evidence/worker-lease-lost/baseline-comparison.txt deleted file mode 100644 index 11c00f691..000000000 --- a/evidence/worker-lease-lost/baseline-comparison.txt +++ /dev/null @@ -1,13 +0,0 @@ -Compared literal FAIL test names in npm-test-final.txt and baseline-live-kernel.txt. -Baseline: original SDK source at f6ece41, rebuilt before running the unchanged live-kernel suite. -Environment: PATH=/home/daytona/.cargo/bin:$PATH -RELAYFLOWD_BIN=/home/daytona/.relayflows-toolchain/target/2962130851/debug/relayflowd -Both runs failed the same eight test names: - FAIL tests/live-kernel.test.ts > built flows CLI against live relayflowd > runs hn-monitor analyze-story end-to-end via a stub agent CLI (gate 2 clause 2 demo) - FAIL tests/live-kernel.test.ts > built flows CLI against live relayflowd > hn-monitor analyze-story FAILS verification when the CLI omits required schema fields - FAIL tests/live-kernel.test.ts > built flows CLI against live relayflowd > agent step preserves the CliResult wrapper as output when the CLI emits non-JSON text - FAIL tests/live-kernel.test.ts > built flows CLI against live relayflowd > AgentWorker exposes wake_context to the CLI via RELAYFLOW_WAKE_CONTEXT env var (real analyzer prerequisite) - FAIL tests/live-kernel.test.ts > built flows CLI against live relayflowd > AgentWorker leaves RELAYFLOW_WAKE_CONTEXT UNSET when the run has no wake_context (undefined-vs-null pin) - FAIL tests/live-kernel.test.ts > built flows CLI against live relayflowd > AgentWorker leaves RELAYFLOW_MODEL UNSET when the step declares no model - FAIL tests/live-kernel.test.ts > built flows CLI against live relayflowd > hn-monitor analyze-story reaches done through the real Claude analyzer CLI - FAIL tests/live-kernel.test.ts > a relayflow can be scheduled: tick source against live relayflowd > a tick spawns a real run whose step reports the SCHEDULED instant diff --git a/evidence/worker-lease-lost/baseline-live-kernel.txt b/evidence/worker-lease-lost/baseline-live-kernel.txt index bff460020..e2837ab79 100644 --- a/evidence/worker-lease-lost/baseline-live-kernel.txt +++ b/evidence/worker-lease-lost/baseline-live-kernel.txt @@ -1,200 +1,44 @@ +# base 2e2043f4c82f1a5c83b090d8a3a82c2d067a79fe $ cd packages/sdk && npx vitest run tests/live-kernel.test.ts - RUN v2.1.9 /home/daytona/.relayflow-v2-supervisor/durable/repository/packages/sdk + RUN v2.1.9 /Users/khaliqgant/Projects/AgentWorkforce/flows/packages/sdk stdout | tests/live-kernel.test.ts -LIVE_KERNEL relayflowd=/home/daytona/.relayflows-toolchain/target/2962130851/debug/relayflowd -LIVE_KERNEL flows=/home/daytona/.relayflow-v2-supervisor/durable/repository/packages/sdk/dist/cli.js +LIVE_KERNEL relayflowd=/Users/khaliqgant/.relayflows-toolchain/target/1166253295/debug/relayflowd +LIVE_KERNEL flows=/Users/khaliqgant/Projects/AgentWorkforce/flows/packages/sdk/dist/cli.js -stdout | tests/live-kernel.test.ts > surface resume after a real daemon kill > resumes a three-step run with each successful completion exactly once -LIVE_KERNEL kill -9 pid=48843 run=01M36CYFYEE1F30A8SJTMT2RJZ while step=two state=Running - - ❯ tests/live-kernel.test.ts (31 tests | 8 failed) 53521ms - ✓ built flows CLI against live relayflowd > twenty-six-step reuses 25 durable completions after editing the failed final step 2121ms - ✓ built flows CLI against live relayflowd > runs rung (a), parks rung (b), and keeps JSON report-shaped 2581ms - ✓ built flows CLI against live relayflowd > allows a deterministic run to exceed the bounded request timeout 32453ms - ✓ built flows CLI against live relayflowd > follows a live worker dispatch through flows run 624ms - ✓ built flows CLI against live relayflowd > runs an agent CLI end to end through the SDK worker 564ms - ✓ built flows CLI against live relayflowd > f.agent lowers to a real agent step and dispatches through a live worker 632ms - ✓ built flows CLI against live relayflowd > f.agent's default flowPath anchors on cwd, not cwd's parent 343ms - ✓ built flows CLI against live relayflowd > can always get a parked run to a late-attaching worker 5942ms - ✓ built flows CLI against live relayflowd > reports a real manual-recovery NeedsHuman state as parked 428ms - × built flows CLI against live relayflowd > runs hn-monitor analyze-story end-to-end via a stub agent CLI (gate 2 clause 2 demo) 433ms - → expected { …(12) } to match object { output: { …(3) }, …(1) } -(22 matching properties omitted from actual) - × built flows CLI against live relayflowd > hn-monitor analyze-story FAILS verification when the CLI omits required schema fields 566ms - → expected { …(12) } to match object { …(3) } -(21 matching properties omitted from actual) - × built flows CLI against live relayflowd > agent step preserves the CliResult wrapper as output when the CLI emits non-JSON text 402ms - → expected null not to be null - × built flows CLI against live relayflowd > AgentWorker exposes wake_context to the CLI via RELAYFLOW_WAKE_CONTEXT env var (real analyzer prerequisite) 514ms - → Cannot read properties of null (reading 'story_title') - × built flows CLI against live relayflowd > AgentWorker leaves RELAYFLOW_WAKE_CONTEXT UNSET when the run has no wake_context (undefined-vs-null pin) 401ms - → Cannot read properties of null (reading 'env_present') - ✓ built flows CLI against live relayflowd > AgentWorker passes a declared model to an identified wrapper as RELAYFLOW_MODEL 495ms - ✓ built flows CLI against live relayflowd > AgentWorker refuses a nonconforming journal-submitted wrapper before exposing RELAYFLOW_MODEL 486ms - ✓ built flows CLI against live relayflowd > AgentWorker executes the raw claude adapter with its real model flag 506ms - ✓ built flows CLI against live relayflowd > AgentWorker executes the raw codex adapter with its real model flag 387ms - × built flows CLI against live relayflowd > AgentWorker leaves RELAYFLOW_MODEL UNSET when the step declares no model 371ms - → Cannot read properties of null (reading 'story_title') - × built flows CLI against live relayflowd > hn-monitor analyze-story reaches done through the real Claude analyzer CLI 35ms - → LIVE_ANALYZER_UNAVAILABLE: "/home/daytona/.relayflow-v2-supervisor/durable/repository/testdata/preflight/analyze-story-claude-cli" does not identify as relayflows-agent-cli-v1 — failing because gate-2 acceptance requires the real analyzer to execute. Set RELAYFLOWS_ALLOW_ANALYZER_SKIP=1 only if this run is not gate evidence. - ✓ built flows CLI against live relayflowd > preflights before journaling and names an unreachable socket 839ms - ✓ built flows CLI against live relayflowd > starts exactly one daemon when two runs race for one empty data dir 483ms - ✓ surface resume after a real daemon kill > resumes a three-step run with each successful completion exactly once 915ms - × a relayflow can be scheduled: tick source against live relayflowd > a tick spawns a real run whose step reports the SCHEDULED instant 448ms - → expected null to deeply equal { schedule_id: 'heartbeat-1m', …(3) } - -⎯⎯⎯⎯⎯⎯⎯ Failed Tests 8 ⎯⎯⎯⎯⎯⎯⎯ - - FAIL tests/live-kernel.test.ts > built flows CLI against live relayflowd > runs hn-monitor analyze-story end-to-end via a stub agent CLI (gate 2 clause 2 demo) -AssertionError: expected { …(12) } to match object { output: { …(3) }, …(1) } -(22 matching properties omitted from actual) - -- Expected -+ Received - - Object { -- "output": Object { -- "reasoning": "stub agent runtime — deterministic output for gate-2 clause-2 demo", -- "relevance_score": 5, -- "story_title": "stub", -- }, -+ "output": null, - "verification": Object { -- "gate": "json_schema", -- "verdict": "pass", -+ "gate": "execution", -+ "verdict": "fail", - }, - } - - ❯ tests/live-kernel.test.ts:657:36 - 655| && (entry as { step_id?: string }).step_id === 'analyze-story', - 656| ) as { payload: { output: unknown; verification: unknown } } | und… - 657| expect(stepCompleted?.payload).toMatchObject({ - | ^ - 658| output: { - 659| story_title: 'stub', - -⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯[1/8]⎯ - - FAIL tests/live-kernel.test.ts > built flows CLI against live relayflowd > hn-monitor analyze-story FAILS verification when the CLI omits required schema fields -AssertionError: expected { …(12) } to match object { …(3) } -(21 matching properties omitted from actual) - -- Expected -+ Received - - Object { -- "completionReason": "retries_exhausted", -+ "completionReason": "worker_error", - "output": null, - "verification": Object { -- "gate": "json_schema", -+ "gate": "execution", - "verdict": "fail", - }, - } - - ❯ tests/live-kernel.test.ts:752:36 - 750| // its verification record names the json_schema rejection. The re… - 751| // parsed value is nulled before the completion is persisted. - 752| expect(stepCompleted?.payload).toMatchObject({ - | ^ - 753| completionReason: 'retries_exhausted', - 754| output: null, - -⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯[2/8]⎯ - - FAIL tests/live-kernel.test.ts > built flows CLI against live relayflowd > agent step preserves the CliResult wrapper as output when the CLI emits non-JSON text -AssertionError: expected null not to be null - ❯ tests/live-kernel.test.ts:823:24 - 821| // here (parseJsonOutput returned null on non-JSON stdout) and - 822| // these assertions would all fail. - 823| expect(output).not.toBeNull(); - | ^ - 824| expect(output.exit_code).toBe(0); - 825| expect(output.stdout_tail).toContain('looked at the story'); - -⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯[3/8]⎯ +stdout | tests/live-kernel.test.ts > built flows CLI against live relayflowd > hn-monitor analyze-story reaches done through the real Claude analyzer CLI +LIVE_ANALYZER ready: claude -p --model claude-haiku-4-5-20251001 round-trip OK - FAIL tests/live-kernel.test.ts > built flows CLI against live relayflowd > AgentWorker exposes wake_context to the CLI via RELAYFLOW_WAKE_CONTEXT env var (real analyzer prerequisite) -TypeError: Cannot read properties of null (reading 'story_title') - ❯ tests/live-kernel.test.ts:891:42 - 889| ) as { payload: { output: { story_title: string; reasoning: string… - 890| expect(stepCompleted).toBeDefined(); - 891| expect(stepCompleted!.payload.output.story_title).toBe(`echoed:${s… - | ^ - 892| expect(stepCompleted!.payload.output.reasoning).toContain(String(s… - 893| +stdout | tests/live-kernel.test.ts > built flows CLI against live relayflowd > hn-monitor analyze-story reaches done through the real Claude analyzer CLI +LIVE_ANALYZER analysis: {"reasoning":"This story is directly about an AI agent performing sophisticated automation tasks—opening pull requests and conducting code reviews—which is core to agent-based development workflows and directly relevant to the Relayflow project's focus on agent automation and workflow orchestration.","relevance_score":9,"story_title":"Show HN: an agent that opens and reviews its own pull requests [wake-nonce-7f3a91c4]"} -⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯[4/8]⎯ - - FAIL tests/live-kernel.test.ts > built flows CLI against live relayflowd > AgentWorker leaves RELAYFLOW_WAKE_CONTEXT UNSET when the run has no wake_context (undefined-vs-null pin) -TypeError: Cannot read properties of null (reading 'env_present') - ❯ tests/live-kernel.test.ts:958:38 - 956| ) as { payload: { output: { env_present: boolean } } } | undefined; - 957| expect(completed).toBeDefined(); - 958| expect(completed!.payload.output.env_present).toBe(false); - | ^ - 959| - 960| delete process.env.RELAYFLOW_WAKE_CONTEXT; - -⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯[5/8]⎯ - - FAIL tests/live-kernel.test.ts > built flows CLI against live relayflowd > AgentWorker leaves RELAYFLOW_MODEL UNSET when the step declares no model -TypeError: Cannot read properties of null (reading 'story_title') - ❯ tests/live-kernel.test.ts:1194:38 - 1192| expect(completed).toBeDefined(); - 1193| // UNSET, not EMPTY and not the leaked parent value. - 1194| expect(completed!.payload.output.story_title).toBe('model:UNSET'); - | ^ - 1195| - 1196| delete process.env.RELAYFLOW_MODEL; - -⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯[6/8]⎯ - - FAIL tests/live-kernel.test.ts > built flows CLI against live relayflowd > hn-monitor analyze-story reaches done through the real Claude analyzer CLI -Error: LIVE_ANALYZER_UNAVAILABLE: "/home/daytona/.relayflow-v2-supervisor/durable/repository/testdata/preflight/analyze-story-claude-cli" does not identify as relayflows-agent-cli-v1 — failing because gate-2 acceptance requires the real analyzer to execute. Set RELAYFLOWS_ALLOW_ANALYZER_SKIP=1 only if this run is not gate evidence. - ❯ tests/live-kernel.test.ts:1223:15 - 1221| const notice = `LIVE_ANALYZER_UNAVAILABLE: ${readiness.detail}`; - 1222| if (process.env['RELAYFLOWS_ALLOW_ANALYZER_SKIP'] !== '1') { - 1223| throw new Error( - | ^ - 1224| `${notice} — failing because gate-2 acceptance requires the … - 1225| + 'Set RELAYFLOWS_ALLOW_ANALYZER_SKIP=1 only if this run is … - -⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯[7/8]⎯ - - FAIL tests/live-kernel.test.ts > a relayflow can be scheduled: tick source against live relayflowd > a tick spawns a real run whose step reports the SCHEDULED instant -AssertionError: expected null to deeply equal { schedule_id: 'heartbeat-1m', …(3) } - -- Expected: -Object { - "lag_ms": 43000, - "schedule_id": "heartbeat-1m", - "scheduled_for_ms": 1764000000000, - "slot": 29400000, -} - -+ Received: -null - - ❯ tests/live-kernel.test.ts:1665:39 - 1663| // The bound: the run reports the grid instant and its own lag, so… - 1664| // backfilled run can tell it is running for a slot from the past. - 1665| expect(completed!.payload.output).toEqual({ - | ^ - 1666| schedule_id: 'heartbeat-1m', - 1667| slot: 29_400_000, - -⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯[8/8]⎯ - - Test Files 1 failed (1) - Tests 8 failed | 23 passed (31) - Start at 05:47:55 - Duration 54.86s (transform 697ms, setup 0ms, collect 1.18s, tests 53.52s, environment 0ms, prepare 42ms) - - -Exit code: 1 +stdout | tests/live-kernel.test.ts > surface resume after a real daemon kill > resumes a three-step run with each successful completion exactly once +LIVE_KERNEL kill -9 pid=91432 run=01M36HY65T82B19WXM8827FP7X while step=two state=Running + + ✓ tests/live-kernel.test.ts (31 tests) 70036ms + ✓ built flows CLI against live relayflowd > twenty-six-step reuses 25 durable completions after editing the failed final step 1666ms + ✓ built flows CLI against live relayflowd > runs rung (a), parks rung (b), and keeps JSON report-shaped 4563ms + ✓ built flows CLI against live relayflowd > allows a deterministic run to exceed the bounded request timeout 32336ms + ✓ built flows CLI against live relayflowd > follows a live worker dispatch through flows run 2296ms + ✓ built flows CLI against live relayflowd > runs an agent CLI end to end through the SDK worker 1076ms + ✓ built flows CLI against live relayflowd > f.agent lowers to a real agent step and dispatches through a live worker 553ms + ✓ built flows CLI against live relayflowd > f.agent's default flowPath anchors on cwd, not cwd's parent 367ms + ✓ built flows CLI against live relayflowd > can always get a parked run to a late-attaching worker 5566ms + ✓ built flows CLI against live relayflowd > reports a real manual-recovery NeedsHuman state as parked 1388ms + ✓ built flows CLI against live relayflowd > AgentWorker passes a declared model to an identified wrapper as RELAYFLOW_MODEL 426ms + ✓ built flows CLI against live relayflowd > AgentWorker refuses a nonconforming journal-submitted wrapper before exposing RELAYFLOW_MODEL 422ms + ✓ built flows CLI against live relayflowd > AgentWorker executes the raw claude adapter with its real model flag 436ms + ✓ built flows CLI against live relayflowd > AgentWorker executes the raw codex adapter with its real model flag 415ms + ✓ built flows CLI against live relayflowd > hn-monitor analyze-story reaches done through the real Claude analyzer CLI 11511ms + ✓ built flows CLI against live relayflowd > preflights before journaling and names an unreachable socket 2199ms + ✓ built flows CLI against live relayflowd > starts exactly one daemon when two runs race for one empty data dir 937ms + ✓ surface resume after a real daemon kill > resumes a three-step run with each successful completion exactly once 1708ms + + Test Files 1 passed (1) + Tests 31 passed (31) + Start at 00:14:52 + Duration 71.50s (transform 549ms, setup 0ms, collect 1.25s, tests 70.04s, environment 0ms, prepare 52ms) + + +Exit code: 0 diff --git a/evidence/worker-lease-lost/baseline.py b/evidence/worker-lease-lost/baseline.py index 6df578c60..b604f4aba 100644 --- a/evidence/worker-lease-lost/baseline.py +++ b/evidence/worker-lease-lost/baseline.py @@ -1,18 +1,20 @@ from pathlib import Path import subprocess,os root=Path.cwd() -files=subprocess.check_output(['git','diff','--name-only','f6ece41','HEAD','--','packages/sdk/src'],text=True).splitlines() +# The rebased parent, not a hard-coded commit: BASE= overrides. +base=os.environ.get('BASE') or subprocess.check_output(['git','merge-base','HEAD','origin/main'],text=True).strip() +files=subprocess.check_output(['git','diff','--name-only',base,'HEAD','--','packages/sdk/src'],text=True).splitlines() originals={name:(root/name).read_bytes() for name in files} -env={**os.environ,'PATH':'/home/daytona/.cargo/bin:'+os.environ['PATH'],'RELAYFLOWD_BIN':'/home/daytona/.relayflows-toolchain/target/2962130851/debug/relayflowd'} +env={**os.environ,'PATH':os.path.expanduser('~/.cargo/bin')+':'+os.environ['PATH']} # RELAYFLOWD_BIN, if needed, comes from the caller def run(cmd,name): with (root/'evidence/worker-lease-lost'/name).open('w') as f: - f.write('$ cd packages/sdk && '+cmd+'\n');f.flush() + f.write('# base '+base+'\n$ cd packages/sdk && '+cmd+'\n');f.flush() result=subprocess.run(cmd,shell=True,cwd=root/'packages/sdk',env=env,stdout=f,stderr=subprocess.STDOUT) f.write('\nExit code: '+str(result.returncode)+'\n') return result.returncode try: for name in files: - (root/name).write_bytes(subprocess.check_output(['git','show','f6ece41:'+name])) + (root/name).write_bytes(subprocess.check_output(['git','show',base+':'+name])) assert run('npm run build','baseline-build.txt')==0 run('npx vitest run tests/live-kernel.test.ts','baseline-live-kernel.txt') finally: diff --git a/evidence/worker-lease-lost/direct-mutant.txt b/evidence/worker-lease-lost/direct-mutant.txt index 854f7db9e..182b133f3 100644 --- a/evidence/worker-lease-lost/direct-mutant.txt +++ b/evidence/worker-lease-lost/direct-mutant.txt @@ -1,9 +1,9 @@ $ cd packages/sdk && npx vitest run tests/direct-run-worker-lease.test.ts - RUN v2.1.9 /home/daytona/.relayflow-v2-supervisor/durable/repository/packages/sdk + RUN v2.1.9 /Users/khaliqgant/Projects/AgentWorkforce/flows/packages/sdk - ❯ tests/direct-run-worker-lease.test.ts (2 tests | 1 failed) 13ms - × direct run handles LLM errors with leaseLost=true 8ms + ❯ tests/direct-run-worker-lease.test.ts (2 tests | 1 failed) 7ms + × direct run handles LLM errors with leaseLost=true 4ms → expected '{"ok":false,"command":"run","resoluti…' to contain 'connection closed' ⎯⎯⎯⎯⎯⎯⎯ Failed Tests 1 ⎯⎯⎯⎯⎯⎯⎯ @@ -12,7 +12,7 @@ $ cd packages/sdk && npx vitest run tests/direct-run-worker-lease.test.ts AssertionError: expected '{"ok":false,"command":"run","resoluti…' to contain 'connection closed' Expected: "connection closed" -Received: "{"ok":false,"command":"run","resolutions":[],"diagnostics":[{"severity":"failure","kind":"protocol_error","message":"relayflowd could not complete the run request: lease_conflict: lost"}],"path":"flow.ts","socketPath":"/tmp/relayflowd-4d1d0e012c91.sock"}" +Received: "{"ok":false,"command":"run","resolutions":[],"diagnostics":[{"severity":"failure","kind":"protocol_error","message":"relayflowd could not complete the run request: lease_conflict: lost"}],"path":"flow.ts","socketPath":"/var/folders/6d/0x5fkt8d01gfmmjdzkxqzwnh0000gn/T/relayflowd-4d1d0e012c91.sock"}" ❯ tests/direct-run-worker-lease.test.ts:33:41 31| const result = await runDirectFlow('flow.ts', '{}', '/unused', { loc… @@ -26,8 +26,8 @@ Received: "{"ok":false,"command":"run","resolutions":[],"diagnostics":[{"severit Test Files 1 failed (1) Tests 1 failed | 1 passed (2) - Start at 05:42:37 - Duration 1.33s (transform 677ms, setup 0ms, collect 1.15s, tests 13ms, environment 0ms, prepare 47ms) + Start at 00:14:32 + Duration 707ms (transform 284ms, setup 0ms, collect 519ms, tests 7ms, environment 0ms, prepare 42ms) Exit code: 1 diff --git a/evidence/worker-lease-lost/direct-restored.txt b/evidence/worker-lease-lost/direct-restored.txt index 166d5e2b0..b351e04fc 100644 --- a/evidence/worker-lease-lost/direct-restored.txt +++ b/evidence/worker-lease-lost/direct-restored.txt @@ -1,13 +1,13 @@ $ cd packages/sdk && npx vitest run tests/direct-run-worker-lease.test.ts - RUN v2.1.9 /home/daytona/.relayflow-v2-supervisor/durable/repository/packages/sdk + RUN v2.1.9 /Users/khaliqgant/Projects/AgentWorkforce/flows/packages/sdk - ✓ tests/direct-run-worker-lease.test.ts (2 tests) 6ms + ✓ tests/direct-run-worker-lease.test.ts (2 tests) 4ms Test Files 1 passed (1) Tests 2 passed (2) - Start at 05:42:39 - Duration 1.29s (transform 646ms, setup 0ms, collect 1.12s, tests 6ms, environment 0ms, prepare 50ms) + Start at 00:14:33 + Duration 793ms (transform 282ms, setup 0ms, collect 599ms, tests 4ms, environment 0ms, prepare 26ms) Exit code: 0 diff --git a/evidence/worker-lease-lost/fatal-mutant.txt b/evidence/worker-lease-lost/fatal-mutant.txt index fed43a85c..4c2b75e55 100644 --- a/evidence/worker-lease-lost/fatal-mutant.txt +++ b/evidence/worker-lease-lost/fatal-mutant.txt @@ -1,9 +1,9 @@ $ cd packages/sdk && npx vitest run tests/worker-lease-lost.test.ts -t 'non-lease worker error' - RUN v2.1.9 /home/daytona/.relayflow-v2-supervisor/durable/repository/packages/sdk + RUN v2.1.9 /Users/khaliqgant/Projects/AgentWorkforce/flows/packages/sdk - ❯ tests/worker-lease-lost.test.ts (16 tests | 2 failed | 14 skipped) 9ms - × agent stale lease subscriber > keeps a non-lease worker error fatal with its original identity 7ms + ❯ tests/worker-lease-lost.test.ts (16 tests | 2 failed | 14 skipped) 6ms + × agent stale lease subscriber > keeps a non-lease worker error fatal with its original identity 5ms → expected "spy" to be called with arguments: [ Error: cli exploded ] Received: @@ -46,8 +46,8 @@ Number of calls: 0 Test Files 1 failed (1) Tests 2 failed | 14 skipped (16) - Start at 05:42:25 - Duration 890ms (transform 351ms, setup 0ms, collect 716ms, tests 9ms, environment 0ms, prepare 48ms) + Start at 00:14:22 + Duration 529ms (transform 150ms, setup 0ms, collect 334ms, tests 6ms, environment 0ms, prepare 43ms) Exit code: 1 diff --git a/evidence/worker-lease-lost/fatal-restored.txt b/evidence/worker-lease-lost/fatal-restored.txt index 54ead3c69..c49b3a5c4 100644 --- a/evidence/worker-lease-lost/fatal-restored.txt +++ b/evidence/worker-lease-lost/fatal-restored.txt @@ -1,13 +1,13 @@ $ cd packages/sdk && npx vitest run tests/worker-lease-lost.test.ts -t 'non-lease worker error' - RUN v2.1.9 /home/daytona/.relayflow-v2-supervisor/durable/repository/packages/sdk + RUN v2.1.9 /Users/khaliqgant/Projects/AgentWorkforce/flows/packages/sdk - ✓ tests/worker-lease-lost.test.ts (16 tests | 14 skipped) 6ms + ✓ tests/worker-lease-lost.test.ts (16 tests | 14 skipped) 4ms Test Files 1 passed (1) Tests 2 passed | 14 skipped (16) - Start at 05:42:26 - Duration 888ms (transform 346ms, setup 0ms, collect 713ms, tests 6ms, environment 0ms, prepare 47ms) + Start at 00:14:23 + Duration 545ms (transform 159ms, setup 0ms, collect 358ms, tests 4ms, environment 0ms, prepare 58ms) Exit code: 0 diff --git a/evidence/worker-lease-lost/filter-mutant.txt b/evidence/worker-lease-lost/filter-mutant.txt index ab083eab0..4ab4cd8ce 100644 --- a/evidence/worker-lease-lost/filter-mutant.txt +++ b/evidence/worker-lease-lost/filter-mutant.txt @@ -1,9 +1,9 @@ $ cd packages/sdk && npx vitest run tests/worker-lease-lost.test.ts - RUN v2.1.9 /home/daytona/.relayflow-v2-supervisor/durable/repository/packages/sdk + RUN v2.1.9 /Users/khaliqgant/Projects/AgentWorkforce/flows/packages/sdk - ❯ tests/worker-lease-lost.test.ts (16 tests | 8 failed) 17ms - × agent stale lease subscriber > drops an expired dispatch and completes the kernel retry 7ms + ❯ tests/worker-lease-lost.test.ts (16 tests | 8 failed) 15ms + × agent stale lease subscriber > drops an expired dispatch and completes the kernel retry 5ms → expected "spy" to not be called at all, but actually been called 1 times Received: @@ -73,7 +73,7 @@ Received: Number of calls: 1 - × llm stale lease subscriber > drops stepHeartbeat lease_conflict for a released attempt 1ms + × llm stale lease subscriber > drops stepHeartbeat lease_conflict for a released attempt 0ms → expected "spy" to not be called at all, but actually been called 1 times Received: @@ -87,7 +87,7 @@ Received: Number of calls: 1 - × llm stale lease subscriber > drops stepComplete lease_conflict for a released attempt 1ms + × llm stale lease subscriber > drops stepComplete lease_conflict for a released attempt 0ms → expected "spy" to not be called at all, but actually been called 1 times Received: @@ -197,8 +197,8 @@ Number of calls: 1 Test Files 1 failed (1) Tests 8 failed | 8 passed (16) - Start at 05:42:19 - Duration 870ms (transform 338ms, setup 0ms, collect 694ms, tests 17ms, environment 0ms, prepare 44ms) + Start at 00:14:19 + Duration 584ms (transform 158ms, setup 0ms, collect 365ms, tests 15ms, environment 0ms, prepare 48ms) Exit code: 1 diff --git a/evidence/worker-lease-lost/filter-restored.txt b/evidence/worker-lease-lost/filter-restored.txt index 851288894..b4cb79c0b 100644 --- a/evidence/worker-lease-lost/filter-restored.txt +++ b/evidence/worker-lease-lost/filter-restored.txt @@ -1,13 +1,13 @@ $ cd packages/sdk && npx vitest run tests/worker-lease-lost.test.ts - RUN v2.1.9 /home/daytona/.relayflow-v2-supervisor/durable/repository/packages/sdk + RUN v2.1.9 /Users/khaliqgant/Projects/AgentWorkforce/flows/packages/sdk - ✓ tests/worker-lease-lost.test.ts (16 tests) 15ms + ✓ tests/worker-lease-lost.test.ts (16 tests) 10ms Test Files 1 passed (1) Tests 16 passed (16) - Start at 05:42:21 - Duration 904ms (transform 356ms, setup 0ms, collect 733ms, tests 15ms, environment 0ms, prepare 43ms) + Start at 00:14:20 + Duration 592ms (transform 168ms, setup 0ms, collect 364ms, tests 10ms, environment 0ms, prepare 43ms) Exit code: 0 diff --git a/evidence/worker-lease-lost/live-kernel-head.txt b/evidence/worker-lease-lost/live-kernel-head.txt new file mode 100644 index 000000000..204c8083a --- /dev/null +++ b/evidence/worker-lease-lost/live-kernel-head.txt @@ -0,0 +1,4 @@ +$ cd packages/sdk && npx vitest run tests/live-kernel.test.ts # macOS, head 2dfb9519 + Test Files 1 passed (1) + Tests 31 passed (31) +exit=0 diff --git a/evidence/worker-lease-lost/live-mutant.txt b/evidence/worker-lease-lost/live-mutant.txt index d7db4f128..60d2574f3 100644 --- a/evidence/worker-lease-lost/live-mutant.txt +++ b/evidence/worker-lease-lost/live-mutant.txt @@ -1,28 +1,39 @@ $ cd packages/sdk && npx vitest run tests/worker-lease-lost-live.test.ts - RUN v2.1.9 /home/daytona/.relayflow-v2-supervisor/durable/repository/packages/sdk + RUN v2.1.9 /Users/khaliqgant/Projects/AgentWorkforce/flows/packages/sdk - ❯ tests/worker-lease-lost-live.test.ts (3 tests | 3 failed) 811ms - × reports journal success after completion rejects with lease_conflict 273ms - → journal client: closed after lease_conflict: attempt has no active worker lease - × reports journal success after completion rejects with run_terminal 244ms + ❯ tests/worker-lease-lost-live.test.ts (3 tests | 3 failed) 1839ms + × reports journal success after completion rejects with lease_conflict 810ms + → journal client: not connected (run.get): journal client: closed after lease_conflict: attempt has no active worker lease + × reports journal success after completion rejects with run_terminal 542ms → journal client: not connected (run.get): journal client: closed after run_terminal: attempt has no active worker lease - × reports journal success when a renewal rejects after completion landed 293ms + × reports journal success when a renewal rejects after completion landed 487ms → journal client: closed after lease_conflict: attempt has no active worker lease ⎯⎯⎯⎯⎯⎯⎯ Failed Tests 3 ⎯⎯⎯⎯⎯⎯⎯ FAIL tests/worker-lease-lost-live.test.ts > reports journal success after completion rejects with lease_conflict -Error: journal client: closed after lease_conflict: attempt has no active worker lease +Error: journal client: not connected (run.get): journal client: closed after lease_conflict: attempt has no active worker lease + ❯ src/journal-client.ts:190:16 + 188| if (!this.socket || this.socket.destroyed) { + 189| const cause = this.disconnectCause; + 190| reject(new Error( + | ^ + 191| `journal client: not connected (${verb})${cause === undefine… + 192| cause === undefined ? undefined : { cause }, + ❯ JournalClient.request src/journal-client.ts:187:12 + ❯ JournalClient.runGet src/journal-client.ts:248:17 + ❯ waitForRunningStep src/cli/run.ts:817:35 + ❯ Module.classifyOutcome src/cli/run.ts:620:7 + ❯ consume src/authored-worker-step.ts:77:23 + ❯ Object.llm src/authored-worker-step.ts:212:22 + ❯ Module.observeStep src/progress.ts:48:20 + ❯ AuthoredFlowOperation.begin src/authored-flow-operation.ts:174:23 + +Caused by: Error: journal client: closed after lease_conflict: attempt has no active worker lease ❯ JournalClient.close src/journal-client.ts:131:9 - 129| const closed = cause === undefined - 130| ? new Error('journal client: closed by caller') - 131| : new Error(`journal client: closed after ${cause instanceof Err… - | ^ - 132| this.disconnectCause ??= closed; - 133| this.failAll(closed); ❯ tests/worker-lease-lost-live.test.ts:17:50 - ❯ LlmWorker. src/worker-lease.ts:20:17 + ❯ LlmWorker. src/worker-lease.ts:21:17 ❯ src/llm-worker.ts:49:66 Caused by: JournalProtocolError: lease_conflict: attempt has no active worker lease @@ -43,8 +54,8 @@ Error: journal client: not connected (run.get): journal client: closed after run 192| cause === undefined ? undefined : { cause }, ❯ JournalClient.request src/journal-client.ts:187:12 ❯ JournalClient.runGet src/journal-client.ts:248:17 - ❯ waitForRunningStep src/cli/run.ts:781:35 - ❯ Module.classifyOutcome src/cli/run.ts:584:7 + ❯ waitForRunningStep src/cli/run.ts:817:35 + ❯ Module.classifyOutcome src/cli/run.ts:620:7 ❯ consume src/authored-worker-step.ts:77:23 ❯ Object.llm src/authored-worker-step.ts:212:22 ❯ Module.observeStep src/progress.ts:48:20 @@ -53,7 +64,7 @@ Error: journal client: not connected (run.get): journal client: closed after run Caused by: Error: journal client: closed after run_terminal: attempt has no active worker lease ❯ JournalClient.close src/journal-client.ts:131:9 ❯ tests/worker-lease-lost-live.test.ts:17:50 - ❯ LlmWorker. src/worker-lease.ts:20:17 + ❯ LlmWorker. src/worker-lease.ts:21:17 ❯ src/llm-worker.ts:49:66 Caused by: JournalProtocolError: run_terminal: attempt has no active worker lease @@ -73,7 +84,7 @@ Error: journal client: closed after lease_conflict: attempt has no active worker 132| this.disconnectCause ??= closed; 133| this.failAll(closed); ❯ tests/worker-lease-lost-live.test.ts:61:50 - ❯ LlmWorker. src/worker-lease.ts:20:17 + ❯ LlmWorker. src/worker-lease.ts:21:17 ❯ src/llm-worker.ts:49:66 Caused by: JournalProtocolError: lease_conflict: attempt has no active worker lease @@ -85,8 +96,8 @@ Serialized Error: { code: 'lease_conflict', verb: 'step.heartbeat' } Test Files 1 failed (1) Tests 3 failed (3) - Start at 05:42:28 - Duration 2.13s (transform 672ms, setup 0ms, collect 1.15s, tests 811ms, environment 0ms, prepare 46ms) + Start at 00:14:24 + Duration 2.60s (transform 289ms, setup 0ms, collect 561ms, tests 1.84s, environment 0ms, prepare 39ms) Exit code: 1 diff --git a/evidence/worker-lease-lost/live-restored.txt b/evidence/worker-lease-lost/live-restored.txt index 982515818..6e172b8bb 100644 --- a/evidence/worker-lease-lost/live-restored.txt +++ b/evidence/worker-lease-lost/live-restored.txt @@ -1,15 +1,16 @@ $ cd packages/sdk && npx vitest run tests/worker-lease-lost-live.test.ts - RUN v2.1.9 /home/daytona/.relayflow-v2-supervisor/durable/repository/packages/sdk + RUN v2.1.9 /Users/khaliqgant/Projects/AgentWorkforce/flows/packages/sdk - ✓ tests/worker-lease-lost-live.test.ts (3 tests) 884ms - ✓ reports journal success after completion rejects with lease_conflict 302ms - ✓ reports journal success when a renewal rejects after completion landed 310ms + ✓ tests/worker-lease-lost-live.test.ts (3 tests) 1366ms + ✓ reports journal success after completion rejects with lease_conflict 440ms + ✓ reports journal success after completion rejects with run_terminal 419ms + ✓ reports journal success when a renewal rejects after completion landed 507ms Test Files 1 passed (1) Tests 3 passed (3) - Start at 05:42:30 - Duration 2.21s (transform 687ms, setup 0ms, collect 1.15s, tests 884ms, environment 0ms, prepare 46ms) + Start at 00:14:27 + Duration 2.10s (transform 287ms, setup 0ms, collect 543ms, tests 1.37s, environment 0ms, prepare 27ms) Exit code: 0 diff --git a/evidence/worker-lease-lost/mutate.py b/evidence/worker-lease-lost/mutate.py index b45368f74..27cfe21f0 100644 --- a/evidence/worker-lease-lost/mutate.py +++ b/evidence/worker-lease-lost/mutate.py @@ -3,7 +3,7 @@ root=Path.cwd() sdk=root/'packages/sdk' evidence=root/'evidence/worker-lease-lost' -env={**os.environ, 'PATH':'/home/daytona/.cargo/bin:'+os.environ['PATH']} +env={**os.environ, 'PATH':os.path.expanduser('~/.cargo/bin')+':'+os.environ['PATH']} def run(name,command): result=subprocess.run(command,cwd=sdk,env=env,shell=True,text=True,stdout=subprocess.PIPE,stderr=subprocess.STDOUT) (evidence/(name+'.txt')).write_text('$ cd packages/sdk && '+command+'\n'+result.stdout+'\nExit code: '+str(result.returncode)+'\n') @@ -24,17 +24,17 @@ def mutation(name,path,old,new,command): assert failed != 0 and passed == 0,(name,failed,passed) (evidence/'mutations.txt').write_text('') mutation('filter','packages/sdk/src/worker-lease.ts', - 'if (!isLeaseLost(error))', 'if (true)', + 'if (!isLeaseLost(error) || dispatch === undefined)', 'if (true)', 'npx vitest run tests/worker-lease-lost.test.ts') mutation('terminal','packages/sdk/src/worker-lease.ts', "error.code === 'run_terminal'", "error.code === 'never_drop_terminal'", "npx vitest run tests/worker-lease-lost.test.ts -t run_terminal") mutation('fatal','packages/sdk/src/worker-lease.ts', - 'if (!isLeaseLost(error)) { fatal(error); return; }', - 'if (!isLeaseLost(error)) { return; }', + 'if (!isLeaseLost(error) || dispatch === undefined) { fatal(error); return; }', + 'if (!isLeaseLost(error) || dispatch === undefined) { return; }', "npx vitest run tests/worker-lease-lost.test.ts -t 'non-lease worker error'") mutation('live','packages/sdk/src/worker-lease.ts', - 'if (!isLeaseLost(error))', 'if (true)', + 'if (!isLeaseLost(error) || dispatch === undefined)', 'if (true)', 'npx vitest run tests/worker-lease-lost-live.test.ts') mutation('sweep','packages/sdk/src/cli/run.ts', 'leaseDeadlineMs + LEASE_SWEEP_GRACE_MS - Date.now()', 'leaseDeadlineMs - Date.now()', diff --git a/evidence/worker-lease-lost/mutations.txt b/evidence/worker-lease-lost/mutations.txt index 06b96a219..a61e7357e 100644 --- a/evidence/worker-lease-lost/mutations.txt +++ b/evidence/worker-lease-lost/mutations.txt @@ -1,36 +1,36 @@ filter File: packages/sdk/src/worker-lease.ts -Replace: 'if (!isLeaseLost(error))' +Replace: 'if (!isLeaseLost(error) || dispatch === undefined)' With: 'if (true)' -Restored SHA256: 4738712912b3d44be73d33469bd7b1e9eab0e8bae5a853bbd4a3f9235fd1440e +Restored SHA256: c8d0c19879d2e4eae4c4bb1cea4946700d86c5c2395ef2bcbc63940c5e7d2afb Mutant exit: 1; restored exit: 0 terminal File: packages/sdk/src/worker-lease.ts Replace: "error.code === 'run_terminal'" With: "error.code === 'never_drop_terminal'" -Restored SHA256: 4738712912b3d44be73d33469bd7b1e9eab0e8bae5a853bbd4a3f9235fd1440e +Restored SHA256: c8d0c19879d2e4eae4c4bb1cea4946700d86c5c2395ef2bcbc63940c5e7d2afb Mutant exit: 1; restored exit: 0 fatal File: packages/sdk/src/worker-lease.ts -Replace: 'if (!isLeaseLost(error)) { fatal(error); return; }' -With: 'if (!isLeaseLost(error)) { return; }' -Restored SHA256: 4738712912b3d44be73d33469bd7b1e9eab0e8bae5a853bbd4a3f9235fd1440e +Replace: 'if (!isLeaseLost(error) || dispatch === undefined) { fatal(error); return; }' +With: 'if (!isLeaseLost(error) || dispatch === undefined) { return; }' +Restored SHA256: c8d0c19879d2e4eae4c4bb1cea4946700d86c5c2395ef2bcbc63940c5e7d2afb Mutant exit: 1; restored exit: 0 live File: packages/sdk/src/worker-lease.ts -Replace: 'if (!isLeaseLost(error))' +Replace: 'if (!isLeaseLost(error) || dispatch === undefined)' With: 'if (true)' -Restored SHA256: 4738712912b3d44be73d33469bd7b1e9eab0e8bae5a853bbd4a3f9235fd1440e +Restored SHA256: c8d0c19879d2e4eae4c4bb1cea4946700d86c5c2395ef2bcbc63940c5e7d2afb Mutant exit: 1; restored exit: 0 sweep File: packages/sdk/src/cli/run.ts Replace: 'leaseDeadlineMs + LEASE_SWEEP_GRACE_MS - Date.now()' With: 'leaseDeadlineMs - Date.now()' -Restored SHA256: b12b6991b1d3d6850e3fdcb0d711438e23f2219fdc08ac46ea1048ea6861584c +Restored SHA256: 4190bc2dfdf2ba162498e438ae508e9a4b9dd408909c6c711e393a381009d3e6 Mutant exit: 1; restored exit: 0 direct @@ -44,6 +44,6 @@ resume File: packages/sdk/src/cli/run.ts Replace: " authoredLlm.on('error', onWorkerFailure('resume-llm', error => {\n llmFailure = error;\n client.close();\n }));\n" With: '' -Restored SHA256: b12b6991b1d3d6850e3fdcb0d711438e23f2219fdc08ac46ea1048ea6861584c +Restored SHA256: 4190bc2dfdf2ba162498e438ae508e9a4b9dd408909c6c711e393a381009d3e6 Mutant exit: 1; restored exit: 0 diff --git a/evidence/worker-lease-lost/restored-build.txt b/evidence/worker-lease-lost/restored-build.txt index 585c5975e..721d8bbfb 100644 --- a/evidence/worker-lease-lost/restored-build.txt +++ b/evidence/worker-lease-lost/restored-build.txt @@ -1,6 +1,7 @@ +# base 2e2043f4c82f1a5c83b090d8a3a82c2d067a79fe $ cd packages/sdk && npm run build -> @relayflows/sdk@2.0.29 build +> @relayflows/sdk@2.0.30 build > tsc && node scripts/make-cli-executable.mjs diff --git a/evidence/worker-lease-lost/resume-mutant.txt b/evidence/worker-lease-lost/resume-mutant.txt index e0e5783b3..777735292 100644 --- a/evidence/worker-lease-lost/resume-mutant.txt +++ b/evidence/worker-lease-lost/resume-mutant.txt @@ -1,9 +1,9 @@ $ cd packages/sdk && npx vitest run tests/resume-worker-lease.test.ts - RUN v2.1.9 /home/daytona/.relayflow-v2-supervisor/durable/repository/packages/sdk + RUN v2.1.9 /Users/khaliqgant/Projects/AgentWorkforce/flows/packages/sdk - ❯ tests/resume-worker-lease.test.ts (2 tests | 2 failed) 9ms - × resume handles LLM errors with leaseLost=false 7ms + ❯ tests/resume-worker-lease.test.ts (2 tests | 2 failed) 5ms + × resume handles LLM errors with leaseLost=false 4ms → expected "spy" to be called at least once × resume handles LLM errors with leaseLost=true 1ms → expected "spy" to be called at least once @@ -25,8 +25,8 @@ AssertionError: expected "spy" to be called at least once Test Files 1 failed (1) Tests 2 failed (2) - Start at 05:42:40 - Duration 1.16s (transform 542ms, setup 0ms, collect 985ms, tests 9ms, environment 0ms, prepare 45ms) + Start at 00:14:34 + Duration 614ms (transform 213ms, setup 0ms, collect 417ms, tests 5ms, environment 0ms, prepare 26ms) Exit code: 1 diff --git a/evidence/worker-lease-lost/resume-restored.txt b/evidence/worker-lease-lost/resume-restored.txt index 34a7f6be0..d1fb64b1c 100644 --- a/evidence/worker-lease-lost/resume-restored.txt +++ b/evidence/worker-lease-lost/resume-restored.txt @@ -1,13 +1,13 @@ $ cd packages/sdk && npx vitest run tests/resume-worker-lease.test.ts - RUN v2.1.9 /home/daytona/.relayflow-v2-supervisor/durable/repository/packages/sdk + RUN v2.1.9 /Users/khaliqgant/Projects/AgentWorkforce/flows/packages/sdk - ✓ tests/resume-worker-lease.test.ts (2 tests) 4ms + ✓ tests/resume-worker-lease.test.ts (2 tests) 2ms Test Files 1 passed (1) Tests 2 passed (2) - Start at 05:42:42 - Duration 1.13s (transform 532ms, setup 0ms, collect 961ms, tests 4ms, environment 0ms, prepare 42ms) + Start at 00:14:35 + Duration 604ms (transform 214ms, setup 0ms, collect 428ms, tests 2ms, environment 0ms, prepare 38ms) Exit code: 0 diff --git a/evidence/worker-lease-lost/sweep-mutant.txt b/evidence/worker-lease-lost/sweep-mutant.txt index 7b4023b2e..e26dd532e 100644 --- a/evidence/worker-lease-lost/sweep-mutant.txt +++ b/evidence/worker-lease-lost/sweep-mutant.txt @@ -1,9 +1,9 @@ $ cd packages/sdk && npx vitest run tests/worker-lease-sweep.test.ts - RUN v2.1.9 /home/daytona/.relayflow-v2-supervisor/durable/repository/packages/sdk + RUN v2.1.9 /Users/khaliqgant/Projects/AgentWorkforce/flows/packages/sdk - ❯ tests/worker-lease-sweep.test.ts (2 tests | 1 failed) 9ms - × waits through an expired snapshot until the kernel retries and completes 7ms + ❯ tests/worker-lease-sweep.test.ts (2 tests | 1 failed) 7ms + × waits through an expired snapshot until the kernel retries and completes 5ms → promise rejected "Error: worker lease for step "answer" exp…" instead of resolving ⎯⎯⎯⎯⎯⎯⎯ Failed Tests 1 ⎯⎯⎯⎯⎯⎯⎯ @@ -18,17 +18,17 @@ AssertionError: promise rejected "Error: worker lease for step "answer" exp…" 34| await Promise.all([assertion, vi.advanceTimersByTimeAsync(100)]); 35| expect(client.runResume).toHaveBeenCalledTimes(2); -Caused by: Error: worker lease for step "answer" expired at 1790142154876 without completion - ❯ waitForRunningStep src/cli/run.ts:776:13 - ❯ Module.classifyOutcome src/cli/run.ts:584:13 +Caused by: Error: worker lease for step "answer" expired at 1790147670919 without completion + ❯ waitForRunningStep src/cli/run.ts:812:13 + ❯ Module.classifyOutcome src/cli/run.ts:620:13 ❯ tests/worker-lease-sweep.test.ts:34:3 ⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯⎯[1/1]⎯ Test Files 1 failed (1) Tests 1 failed | 1 passed (2) - Start at 05:42:33 - Duration 1.29s (transform 656ms, setup 0ms, collect 1.12s, tests 9ms, environment 0ms, prepare 44ms) + Start at 00:14:30 + Duration 767ms (transform 305ms, setup 0ms, collect 559ms, tests 7ms, environment 0ms, prepare 43ms) Exit code: 1 diff --git a/evidence/worker-lease-lost/sweep-restored.txt b/evidence/worker-lease-lost/sweep-restored.txt index 0edea990f..8c2be2c18 100644 --- a/evidence/worker-lease-lost/sweep-restored.txt +++ b/evidence/worker-lease-lost/sweep-restored.txt @@ -1,13 +1,13 @@ $ cd packages/sdk && npx vitest run tests/worker-lease-sweep.test.ts - RUN v2.1.9 /home/daytona/.relayflow-v2-supervisor/durable/repository/packages/sdk + RUN v2.1.9 /Users/khaliqgant/Projects/AgentWorkforce/flows/packages/sdk - ✓ tests/worker-lease-sweep.test.ts (2 tests) 7ms + ✓ tests/worker-lease-sweep.test.ts (2 tests) 6ms Test Files 1 passed (1) Tests 2 passed (2) - Start at 05:42:35 - Duration 1.38s (transform 685ms, setup 0ms, collect 1.21s, tests 7ms, environment 0ms, prepare 45ms) + Start at 00:14:31 + Duration 728ms (transform 285ms, setup 0ms, collect 522ms, tests 6ms, environment 0ms, prepare 33ms) Exit code: 0 diff --git a/evidence/worker-lease-lost/terminal-mutant.txt b/evidence/worker-lease-lost/terminal-mutant.txt index e57a711ec..2c8cd916d 100644 --- a/evidence/worker-lease-lost/terminal-mutant.txt +++ b/evidence/worker-lease-lost/terminal-mutant.txt @@ -1,9 +1,9 @@ $ cd packages/sdk && npx vitest run tests/worker-lease-lost.test.ts -t run_terminal - RUN v2.1.9 /home/daytona/.relayflow-v2-supervisor/durable/repository/packages/sdk + RUN v2.1.9 /Users/khaliqgant/Projects/AgentWorkforce/flows/packages/sdk - ❯ tests/worker-lease-lost.test.ts (16 tests | 2 failed | 14 skipped) 9ms - × agent stale lease subscriber > drops stepComplete run_terminal for a released attempt 7ms + ❯ tests/worker-lease-lost.test.ts (16 tests | 2 failed | 14 skipped) 6ms + × agent stale lease subscriber > drops stepComplete run_terminal for a released attempt 5ms → expected "spy" to not be called at all, but actually been called 1 times Received: @@ -61,8 +61,8 @@ Number of calls: 1 Test Files 1 failed (1) Tests 2 failed | 14 skipped (16) - Start at 05:42:22 - Duration 895ms (transform 359ms, setup 0ms, collect 714ms, tests 9ms, environment 0ms, prepare 52ms) + Start at 00:14:21 + Duration 586ms (transform 163ms, setup 0ms, collect 372ms, tests 6ms, environment 0ms, prepare 35ms) Exit code: 1 diff --git a/evidence/worker-lease-lost/terminal-restored.txt b/evidence/worker-lease-lost/terminal-restored.txt index 2759cb2a0..6b3b05efe 100644 --- a/evidence/worker-lease-lost/terminal-restored.txt +++ b/evidence/worker-lease-lost/terminal-restored.txt @@ -1,13 +1,13 @@ $ cd packages/sdk && npx vitest run tests/worker-lease-lost.test.ts -t run_terminal - RUN v2.1.9 /home/daytona/.relayflow-v2-supervisor/durable/repository/packages/sdk + RUN v2.1.9 /Users/khaliqgant/Projects/AgentWorkforce/flows/packages/sdk - ✓ tests/worker-lease-lost.test.ts (16 tests | 14 skipped) 7ms + ✓ tests/worker-lease-lost.test.ts (16 tests | 14 skipped) 4ms Test Files 1 passed (1) Tests 2 passed | 14 skipped (16) - Start at 05:42:24 - Duration 890ms (transform 347ms, setup 0ms, collect 716ms, tests 7ms, environment 0ms, prepare 48ms) + Start at 00:14:22 + Duration 518ms (transform 149ms, setup 0ms, collect 331ms, tests 4ms, environment 0ms, prepare 28ms) Exit code: 0