diff --git a/apps/server/src/coil/http/testAuth.ts b/apps/server/src/coil/http/testAuth.ts index 0a3cf01bfcc3..a1a7d5163473 100644 --- a/apps/server/src/coil/http/testAuth.ts +++ b/apps/server/src/coil/http/testAuth.ts @@ -88,6 +88,24 @@ export const postJson = (pathname: string, body: unknown) => return yield* client.execute(request); }); +/** + * A POST whose body is **not** JSON, sent with a JSON content type. + * + * This is the shape a broken client, a proxy that rewrote a body, or a mistyped `curl` + * produces, and it is the one path that never reaches a route's own decoder — so it is also + * the one that quietly returns a bare, uncoded 400 unless the route handles it. + */ +export const postText = (pathname: string, body: string) => + Effect.gen(function* () { + const client = yield* HttpClient.HttpClient; + const request = HttpClientRequest.bodyText( + HttpClientRequest.post(yield* baseUrl(pathname)), + body, + "application/json", + ); + return yield* client.execute(request); + }); + /** Reads a JSON response body as a plain record, for field-by-field assertions. */ export const jsonBody = (response: HttpClientResponse.HttpClientResponse) => Effect.map(response.json, (value) => value as Record); diff --git a/apps/server/src/coil/index.ts b/apps/server/src/coil/index.ts index 609cd1bbcfd7..75138ea67048 100644 --- a/apps/server/src/coil/index.ts +++ b/apps/server/src/coil/index.ts @@ -22,6 +22,9 @@ import { ServerConfig } from "../config.ts"; import { autoResumeRouteLayer } from "./autoResume/http.ts"; import { AutoResumeReactorLive } from "./autoResume/Reactor.ts"; import { AutoResumeStore, makeAutoResumeStore } from "./autoResume/state.ts"; +import { loopRouteLayer } from "./loop/http.ts"; +import { LoopStoreLive } from "./loop/layer.ts"; +import { LoopReactorLive } from "./loop/Reactor.ts"; import { resolveConfig as resolveWebPushConfig } from "./webPush/config.ts"; import { webPushRouteLayer } from "./webPush/http.ts"; import { WebPushReactorLive } from "./webPush/Reactor.ts"; @@ -51,6 +54,20 @@ const PushSubscriptionStoreLive = Layer.effect( }), ); +/** + * The loops record is IMPORTED, not declared here. + * + * Same shared-identity rule as `AutoResumeStoreLive` above — Effect memoises layer + * construction by layer identity, so every consumer that `Layer.provide`s this one value gets + * one store over one file — but loops has a third consumer the other two do not: the MCP + * toolkit, which registers off `mcp/McpHttpServer.ts`, on the far side of the layer graph + * from this aggregator. It cannot import from here without a cycle, so the value lives in + * `loop/layer.ts` and all three import it. Re-declaring it here would hand the reactor and + * the routes a *different* instance from the toolkit's, and the failure is silent: the agent + * would call `loop_done`, the tool would report success, and the supervisor would keep + * checking in against a record that never saw the write. + */ + /** VAPID keypair (env override, else generated + persisted in the secret store). */ const WebPushVapidLive = Layer.effect(WebPushVapid, makeWebPushVapid(resolveWebPushConfig())); @@ -71,6 +88,10 @@ const WebPushDepsLive = Layer.mergeAll(PushSubscriptionStoreLive, WebPushVapidLi export const CoilLayerLive = Layer.mergeAll( AutoResumeReactorLive.pipe(Layer.provide(AutoResumeStoreLive)), WebPushReactorLive.pipe(Layer.provide(WebPushDepsLive)), + // The loop supervisor reads the auto-resume record for guard 9 (two reactors must never + // both nudge one thread), so it takes the SAME `AutoResumeStoreLive` value the auto-resume + // reactor and route already share. + LoopReactorLive.pipe(Layer.provide(Layer.merge(LoopStoreLive, AutoResumeStoreLive))), ); /** @@ -100,4 +121,5 @@ export const CoilLayerLive = Layer.mergeAll( export const CoilRoutesLive = Layer.mergeAll( autoResumeRouteLayer.pipe(Layer.provide(AutoResumeStoreLive)), webPushRouteLayer.pipe(Layer.provide(WebPushDepsLive)), + loopRouteLayer.pipe(Layer.provide(LoopStoreLive)), ); diff --git a/apps/server/src/coil/loop/Reactor.test.ts b/apps/server/src/coil/loop/Reactor.test.ts new file mode 100644 index 000000000000..60cde948b134 --- /dev/null +++ b/apps/server/src/coil/loop/Reactor.test.ts @@ -0,0 +1,1086 @@ +/** + * The loop supervisor's fibers, end to end. TESTS.md cases 87–107. + * + * Real reactor, real store, real decision table, real sentinel reads; a scripted projection + * and a recording engine. Everything here runs on `TestClock`, so an eight-hour overnight + * run costs milliseconds and no assertion depends on a timeout. + * + * Timing vocabulary used throughout, from the shipped defaults: the tick polls every + * **60 s**, the idle threshold is **15 min**, the busy threshold is **45 min**, and guard + * 11's floor between check-ins is **15 min**. `TestClock` starts at epoch 0, so a thread + * whose `updatedAt` is `msToIso(0)` is exactly `now` milliseconds idle and fires on the 15th + * tick. + * + * @module coil/loop/Reactor.test + */ + +// @effect-diagnostics nodeBuiltinImport:off -- the persistence assertions read the state file +// directly, which is the point: they check what survived a process, not what a Ref remembers. +import * as NodeFSP from "node:fs/promises"; + +import type { OrchestrationCommand } from "@t3tools/contracts"; +import * as NodeServices from "@effect/platform-node/NodeServices"; +import { assert, describe, it } from "@effect/vitest"; +import * as Effect from "effect/Effect"; +import type * as FileSystem from "effect/FileSystem"; +import * as Layer from "effect/Layer"; +import type * as Path from "effect/Path"; +import * as Option from "effect/Option"; +import * as Ref from "effect/Ref"; +import * as Schema from "effect/Schema"; +import type * as Scope from "effect/Scope"; +import * as TestClock from "effect/testing/TestClock"; + +import { loopHooksFor } from "./crons.ts"; +import { LOOP_ACTIVITY_KINDS, LoopReactorLive } from "./Reactor.ts"; +import { + activitiesOfKind, + advancePolls, + advanceUntil, + advanceWithoutReactor, + clearLoopEnv, + commandTypes, + harness, + HOUR, + isStopped, + LOOP_THREAD_ID, + MINUTE, + msToIso, + rateLimitEvent, + settleQuiet, + threadShell, + turnStarts, + turnStartsAtLeast, + untilAllReceipts, + untilReceipt, + userInputRequestedEvent, + writeSentinel, +} from "./reactorHarness.ts"; +import { LoopState, LoopStore, type LoopStoreShape } from "./state.ts"; + +clearLoopEnv(); + +/** Hoisted: both the schema literal and the compiled decoder would otherwise be rebuilt. */ +const decodeLoopState = Schema.decodeUnknownEffect(Schema.fromJsonString(LoopState)); + +/** The default armed run: six check-ins, eight hours, master toggle on. */ +const arm = ( + store: LoopStoreShape, + o: { + readonly threadId?: string; + readonly maxCheckIns?: number; + readonly deadlineAtMs?: number; + readonly armedAtMs?: number; + readonly pinnedByLoop?: boolean; + } = {}, +) => + Effect.gen(function* () { + yield* store.setGlobal({ enabled: true }); + yield* store.arm({ + threadId: o.threadId ?? LOOP_THREAD_ID, + armedAtMs: o.armedAtMs ?? 0, + deadlineAtMs: o.deadlineAtMs ?? 8 * HOUR, + maxCheckIns: o.maxCheckIns ?? 6, + ...(o.pinnedByLoop === undefined ? {} : { pinnedByLoop: o.pinnedByLoop }), + }); + }); + +/** A recurring wake scheduled past the deadline: pending for the bound, not deferred to. */ +const recurringCronPastDeadline = (store: LoopStoreShape, deadlineAtMs: number) => + store.setCrons(LOOP_THREAD_ID, { + recordedAtMs: 0, + entries: [ + { + id: "cron-1", + schedule: "*/20 * * * *", + recurring: true, + prompt: "keep going", + nextFireAtMs: deadlineAtMs + HOUR, + }, + ], + }); + +/** Boot the real reactor over the harness doubles for the duration of `body`. */ +const withReactor = ( + deps: Layer.Layer, + body: Effect.Effect, +) => body.pipe(Effect.provide(LoopReactorLive.pipe(Layer.provideMerge(deps)))); + +/** Everything the harness itself needs: a scope for the temp dir, real fs, and TestClock. */ +type OuterR = Scope.Scope | FileSystem.FileSystem | Path.Path; + +const scoped = (body: Effect.Effect) => + body.pipe(Effect.scoped, Effect.provide(Layer.mergeAll(NodeServices.layer, TestClock.layer()))); + +const record = (store: LoopStoreShape, threadId: string = LOOP_THREAD_ID) => + store.getThread(threadId); + +describe("LoopReactor — the fibers", () => { + // --- 87, 88: the cost floor ------------------------------------------------ + + it.effect("87: COIL_LOOP_ENABLED=0 forks no fiber at all", () => + Effect.gen(function* () { + const h = yield* harness({ shell: threadShell() }); + yield* arm(h.store); + process.env.COIL_LOOP_ENABLED = "0"; + try { + yield* withReactor( + h.deps, + Effect.gen(function* () { + // Three simulated hours past every threshold this thread has. Blind, because a + // switched-off supervisor never completes a tick and so never announces one. + yield* advanceWithoutReactor(180); + assert.strictEqual( + yield* Ref.get(h.shellCalls), + 0, + "a disabled reactor must not read the projection", + ); + assert.deepStrictEqual(yield* Ref.get(h.dispatched), []); + assert.strictEqual((yield* record(h.store)).checkInsUsed, 0); + }), + ); + } finally { + delete process.env.COIL_LOOP_ENABLED; + } + }).pipe(scoped), + ); + + it.effect("88: nothing armed issues zero projection queries per tick", () => + Effect.gen(function* () { + const h = yield* harness({ shell: threadShell() }); + yield* h.store.setGlobal({ enabled: true }); + yield* withReactor( + h.deps, + Effect.gen(function* () { + yield* advancePolls(30); + assert.strictEqual(yield* Ref.get(h.shellCalls), 0, "no thread shell reads"); + assert.strictEqual(yield* Ref.get(h.projectCalls), 0, "no project shell reads"); + }), + ); + }).pipe(scoped), + ); + + // --- 89–92: the nudge itself ---------------------------------------------- + + it.effect("89/91/92: one armed idle thread fires exactly one correctly-shaped turn", () => + Effect.gen(function* () { + const h = yield* harness({ + shell: threadShell({ runtimeMode: "read-only", interactionMode: "plan" }), + }); + yield* arm(h.store); + yield* withReactor( + h.deps, + Effect.gen(function* () { + yield* advanceUntil(turnStartsAtLeast(h.dispatched, 1), "the first check-in"); + const starts = turnStarts(yield* Ref.get(h.dispatched)); + assert.strictEqual(starts.length, 1); + const turn = starts[0]!; + // 91 — both ids carry the prefix, so a loop turn is identifiable in the event log. + assert.isTrue(turn.commandId.startsWith("coil-loop:"), turn.commandId); + assert.isTrue(turn.message.messageId.startsWith("coil-loop:"), turn.message.messageId); + // 92 — copied from the shell, never defaulted. A loop turn that silently promoted a + // read-only thread to full access would be a security regression. + assert.strictEqual(turn.runtimeMode, "read-only"); + assert.strictEqual(turn.interactionMode, "plan"); + assert.strictEqual(turn.threadId, LOOP_THREAD_ID); + assert.include(turn.message.text, "Loop check-in 1 of 6."); + assert.isFalse(turn.message.text.startsWith("/"), "never readable as a slash command"); + }), + ); + }).pipe(scoped), + ); + + it.effect("90: a second tick while still idle does not fire again (guard 11's floor)", () => + Effect.gen(function* () { + const h = yield* harness({ shell: threadShell() }); + yield* arm(h.store); + yield* withReactor( + h.deps, + Effect.gen(function* () { + yield* advanceUntil(turnStartsAtLeast(h.dispatched, 1), "the first check-in"); + yield* advancePolls(2); + assert.strictEqual(turnStarts(yield* Ref.get(h.dispatched)).length, 1); + assert.strictEqual((yield* record(h.store)).checkInsUsed, 1); + }), + ); + }).pipe(scoped), + ); + + // --- 93, 94: reserve before dispatch -------------------------------------- + + it.effect("93: a dispatch that fails still consumes the reservation, and does not retry", () => + Effect.gen(function* () { + const h = yield* harness({ shell: threadShell() }); + yield* arm(h.store); + yield* Ref.set(h.failCommandTypes, new Set(["thread.turn.start"])); + yield* withReactor( + h.deps, + Effect.gen(function* () { + yield* advanceUntil( + record(h.store).pipe(Effect.map((r) => r.checkInsUsed === 1)), + "the reservation", + ); + assert.notInclude(commandTypes(yield* Ref.get(h.dispatched)), "thread.turn.start"); + // The floor holds even though nothing was sent: a provider that cannot spawn burns + // budget rather than tight-looping. + yield* advancePolls(3); + assert.strictEqual((yield* record(h.store)).checkInsUsed, 1); + }), + ); + }).pipe(scoped), + ); + + it.effect("93b: a provider that can never spawn is bounded by attempts, not by the night", () => + Effect.gen(function* () { + // The design prices this path at "6 attempts, not 480 a night". In practice strikes + // bound it tighter: a dispatch that never lands never moves `updatedAt`, so two + // consecutive check-ins are judged unproductive and the run reports `stalled` at two. + // Both bounds are real; what matters is that neither is unbounded. + const h = yield* harness({ shell: threadShell() }); + yield* arm(h.store, { maxCheckIns: 6, deadlineAtMs: 8 * HOUR }); + yield* Ref.set(h.failCommandTypes, new Set(["thread.turn.start"])); + yield* withReactor( + h.deps, + Effect.gen(function* () { + yield* advanceUntil( + record(h.store).pipe(Effect.map((r) => r.stopped !== null)), + "the run to end itself", + 400, + ); + const after = yield* record(h.store); + assert.strictEqual(after.checkInsUsed, 2, "two attempts, not four hundred and eighty"); + assert.isAtMost(after.checkInsUsed, after.maxCheckIns); + assert.strictEqual(after.stopped?.reason, "stalled"); + }), + ); + }).pipe(scoped), + ); + + it.effect("93c: a responsive thread spends its whole budget and then reports spent", () => + Effect.gen(function* () { + // The budget bound on its own, with the strike bound held off: after each check-in the + // thread is scripted to move (three minutes, past `productiveMs`) and then go quiet + // again, which is what a working agent looks like. + const h = yield* harness({ shell: threadShell() }); + yield* arm(h.store, { maxCheckIns: 6, deadlineAtMs: 8 * HOUR }); + yield* withReactor( + h.deps, + Effect.gen(function* () { + for (let n = 1; n <= 6; n++) { + // Wait for the TURN, not the reservation. Moving the shell in between would land + // inside the pre-dispatch re-read and the check-in would abort — which is the + // wake race working, and would have made this test measure the wrong thing. + yield* advanceUntil(turnStartsAtLeast(h.dispatched, n), `check-in ${n}`, 60); + const current = yield* record(h.store); + yield* Ref.set( + h.shellRef, + threadShell({ updatedAt: msToIso(current.lastCheckIn!.firedAtMs + 3 * MINUTE) }), + ); + } + yield* advanceUntil( + record(h.store).pipe(Effect.map((r) => r.stopped !== null)), + "budget exhaustion", + 60, + ); + const after = yield* record(h.store); + assert.strictEqual(after.checkInsUsed, 6, "exactly the budget, not one more"); + assert.strictEqual(after.stopped?.reason, "spent"); + assert.strictEqual(after.strikes, 0, "a moving thread never accrues a strike"); + assert.strictEqual(turnStarts(yield* Ref.get(h.dispatched)).length, 6); + }), + ); + }).pipe(scoped), + ); + + it.effect("94: a defect thrown dispatching does not kill the tick fiber", () => + Effect.gen(function* () { + const h = yield* harness({ shell: threadShell() }); + yield* arm(h.store); + yield* Ref.set(h.dieOnCommandType, "thread.turn.start"); + yield* withReactor( + h.deps, + Effect.gen(function* () { + yield* advanceUntil( + record(h.store).pipe(Effect.map((r) => r.checkInsUsed === 1)), + "the first (defecting) attempt", + ); + yield* Ref.set(h.dieOnCommandType, null); + yield* advanceUntil( + turnStartsAtLeast(h.dispatched, 1), + "a later check-in after the defect", + ); + assert.strictEqual(turnStarts(yield* Ref.get(h.dispatched)).length, 1); + }), + ); + }).pipe(scoped), + ); + + it.effect("94b: a defect on one thread does not stop the others in the same pass", () => + Effect.gen(function* () { + const other = threadShell({ id: "thread-2" }); + const h = yield* harness({ shell: threadShell(), extraShells: [other] }); + yield* arm(h.store); + yield* arm(h.store, { threadId: "thread-2" }); + yield* Ref.set(h.dieOnThreadId, LOOP_THREAD_ID); + yield* withReactor( + h.deps, + Effect.gen(function* () { + yield* advanceUntil(turnStartsAtLeast(h.dispatched, 1), "thread-2's check-in"); + const starts = turnStarts(yield* Ref.get(h.dispatched)); + assert.deepStrictEqual( + starts.map((t) => t.threadId), + ["thread-2"], + "the healthy thread is nudged even though its neighbour defected", + ); + // Both spent their reservation; only one produced a turn. + assert.strictEqual((yield* record(h.store)).checkInsUsed, 1); + assert.strictEqual((yield* record(h.store, "thread-2")).checkInsUsed, 1); + }), + ); + }).pipe(scoped), + ); + + // --- 95: the wake race ---------------------------------------------------- + + it.effect("95: a thread that blocks between the guard block and dispatch is not nudged", () => + Effect.gen(function* () { + const h = yield* harness({ shell: threadShell() }); + yield* arm(h.store); + yield* withReactor( + h.deps, + Effect.gen(function* () { + // Stop one tick short of the 15-minute threshold, then arm the swap for the NEXT + // read — the guard block sees an idle thread, the pre-dispatch re-read sees a + // thread now waiting on a human. + yield* advancePolls(14); + assert.strictEqual(turnStarts(yield* Ref.get(h.dispatched)).length, 0); + const readsSoFar = yield* Ref.get(h.shellCalls); + yield* Ref.set(h.shellOverrideRef, { + afterCall: readsSoFar + 1, + shell: threadShell({ hasPendingUserInput: true }), + }); + const abortNoted = Ref.get(h.dispatched).pipe( + Effect.map((all) => + activitiesOfKind(all, LOOP_ACTIVITY_KINDS.skipped).some( + (a) => (a.payload as { reason?: string }).reason === "aborted_pre_dispatch", + ), + ), + ); + // A tight bound, with a little margin for pump scheduling under load: the abort + // must land within a few polls, not eventually. + yield* advanceUntil(abortNoted, "the aborted attempt", 6); + + assert.strictEqual( + turnStarts(yield* Ref.get(h.dispatched)).length, + 0, + "the nudge must not be sent", + ); + const aborted = activitiesOfKind( + yield* Ref.get(h.dispatched), + LOOP_ACTIVITY_KINDS.skipped, + ).filter((a) => (a.payload as { reason?: string }).reason === "aborted_pre_dispatch"); + assert.strictEqual(aborted.length, 1, "the aborted attempt is recorded"); + assert.strictEqual((yield* record(h.store)).checkInsUsed, 1, "the reservation stands"); + }), + ); + }).pipe(scoped), + ); + + // --- 96–98: the keep-active pin ------------------------------------------- + + it.effect("96: settledOverride 'active' is repaired with thread.unsettle after the turn", () => + Effect.gen(function* () { + const h = yield* harness({ shell: threadShell({ settledOverride: "active" }) }); + yield* arm(h.store); + yield* withReactor( + h.deps, + Effect.gen(function* () { + yield* advanceUntil( + Ref.get(h.dispatched).pipe( + Effect.map((all) => commandTypes(all).includes("thread.unsettle")), + ), + "the pin repair", + ); + const types = commandTypes(yield* Ref.get(h.dispatched)); + assert.include(types, "thread.unsettle"); + assert.isAbove( + types.indexOf("thread.unsettle"), + types.indexOf("thread.turn.start"), + "the repair follows the turn; issuing it first would be cleared by the decider", + ); + const unsettle = (yield* Ref.get(h.dispatched)).find( + (c): c is Extract => + c.type === "thread.unsettle", + ); + assert.strictEqual(unsettle?.reason, "user"); + }), + ); + }).pipe(scoped), + ); + + it.effect("97: no pin, no repair — the reactor can never create one", () => + Effect.gen(function* () { + const h = yield* harness({ shell: threadShell({ settledOverride: null }) }); + yield* arm(h.store); + yield* withReactor( + h.deps, + Effect.gen(function* () { + yield* advanceUntil(turnStartsAtLeast(h.dispatched, 1), "the check-in"); + yield* settleQuiet; + assert.notInclude(commandTypes(yield* Ref.get(h.dispatched)), "thread.unsettle"); + }), + ); + }).pipe(scoped), + ); + + it.effect("98: a failed pin repair is never silent — it posts an error-tone breadcrumb", () => + Effect.gen(function* () { + const h = yield* harness({ shell: threadShell({ settledOverride: "active" }) }); + yield* arm(h.store); + yield* Ref.set(h.failCommandTypes, new Set(["thread.unsettle"])); + yield* withReactor( + h.deps, + Effect.gen(function* () { + yield* advanceUntil( + Ref.get(h.dispatched).pipe( + Effect.map( + (all) => activitiesOfKind(all, LOOP_ACTIVITY_KINDS.pinRepairFailed).length > 0, + ), + ), + "the pin-repair breadcrumb", + ); + const notes = activitiesOfKind( + yield* Ref.get(h.dispatched), + LOOP_ACTIVITY_KINDS.pinRepairFailed, + ); + assert.strictEqual(notes.length, 1); + assert.strictEqual(notes[0]!.tone, "error"); + // The check-in itself still landed: the repair is a follow-up, not a precondition. + assert.strictEqual(turnStarts(yield* Ref.get(h.dispatched)).length, 1); + }), + ); + }).pipe(scoped), + ); + + // --- 99: the bound actually stops the agent ------------------------------- + + it.effect("99: budget exhaustion writes spent once, stops the session, then no-ops", () => + Effect.gen(function* () { + const deadlineAtMs = 8 * HOUR; + const h = yield* harness({ shell: threadShell() }); + yield* arm(h.store, { maxCheckIns: 1, deadlineAtMs }); + yield* recurringCronPastDeadline(h.store, deadlineAtMs); + yield* withReactor( + h.deps, + Effect.gen(function* () { + yield* advanceUntil( + Ref.get(h.stopSessions).pipe(Effect.map((all) => all.length > 0)), + "the spent terminal and its session stop", + 120, + ); + const stopped = (yield* record(h.store)).stopped; + assert.strictEqual(stopped?.reason, "spent"); + assert.strictEqual( + (yield* Ref.get(h.stopSessions)).length, + 1, + "recorded crons are still pending, so the session must be ended", + ); + + const before = activitiesOfKind( + yield* Ref.get(h.dispatched), + LOOP_ACTIVITY_KINDS.stopped, + ); + assert.strictEqual(before.length, 1); + // `spent` is never reported as `done`. + assert.strictEqual((before[0]!.payload as { reason: string }).reason, "spent"); + assert.notStrictEqual((before[0]!.payload as { reason: string }).reason, "done"); + + yield* advancePolls(20); + assert.strictEqual( + activitiesOfKind(yield* Ref.get(h.dispatched), LOOP_ACTIVITY_KINDS.stopped).length, + 1, + "the terminal is sticky and written exactly once", + ); + assert.strictEqual((yield* Ref.get(h.stopSessions)).length, 1); + }), + ); + }).pipe(scoped), + ); + + it.effect("99b: a passed deadline stops the run while the thread is busy", () => + Effect.gen(function* () { + // `running` selects the 45-minute fuse, so this thread never reaches an idle guard. + // With the stop sweep at 4b it is still stopped at its deadline; at the old position + // (guard 13) it would have walked through it indefinitely. + const h = yield* harness({ + shell: threadShell({ sessionStatus: "running", latestTurnState: "running" }), + }); + yield* arm(h.store, { deadlineAtMs: 5 * MINUTE }); + yield* withReactor( + h.deps, + Effect.gen(function* () { + yield* advanceUntil( + record(h.store).pipe(Effect.map((r) => r.stopped !== null)), + "the deadline stop", + 30, + ); + const stopped = (yield* record(h.store)).stopped; + assert.strictEqual(stopped?.reason, "spent"); + assert.include(stopped?.detail ?? "", "deadline"); + assert.strictEqual(turnStarts(yield* Ref.get(h.dispatched)).length, 0); + }), + ); + }).pipe(scoped), + ); + + // --- 100–102: the rate-limit tap ------------------------------------------ + + it.effect("100: a rejected verdict writes rateLimitedUntilMs durably", () => + Effect.gen(function* () { + const h = yield* harness({ + shell: threadShell(), + events: [rateLimitEvent({ status: "rejected", resetsAtSeconds: 3_600 })], + }); + yield* arm(h.store); + yield* withReactor( + h.deps, + Effect.gen(function* () { + // The tap is stream-driven, not clock-driven: it announces the write, so this + // waits for the announcement rather than advancing time in the hope of catching it. + yield* untilReceipt((r) => r.type === "rateLimit.recorded"); + assert.strictEqual((yield* record(h.store)).rateLimitedUntilMs, 3_600_000); + // Durable: a five-hour limit outlives the process that observed it. + const persisted = yield* Effect.promise(() => NodeFSP.readFile(h.statePath, "utf8")); + assert.include(persisted, '"rateLimitedUntilMs":3600000'); + }), + ); + }).pipe(scoped), + ); + + it.effect("101: a non-rejected verdict writes nothing", () => + Effect.gen(function* () { + const h = yield* harness({ + shell: threadShell(), + events: [rateLimitEvent({ status: "allowed_warning", resetsAtSeconds: 3_600 })], + }); + yield* arm(h.store); + yield* withReactor( + h.deps, + Effect.gen(function* () { + yield* advancePolls(3); + assert.strictEqual((yield* record(h.store)).rateLimitedUntilMs, 0); + }), + ); + }).pipe(scoped), + ); + + it.effect("102: two subscribers on one stream — neither consumes the other's events", () => + Effect.gen(function* () { + // The rate-limit tap and the user-input recorder each subscribe to `streamEvents` + // independently. If the reactor shared one subscription, one of these two records + // would be missing. (Upstream's PubSub semantics are upstream's contract; what this + // pins is that the fork forks two subscriptions rather than one.) + const h = yield* harness({ + shell: threadShell(), + events: [ + rateLimitEvent({ status: "rejected", resetsAtSeconds: 3_600 }), + userInputRequestedEvent({ requestId: "req-1", question: "which branch?" }), + ], + }); + yield* arm(h.store); + yield* withReactor( + h.deps, + Effect.gen(function* () { + // Both taps announce, and one drain collects both — waiting for them separately + // would throw away whichever arrived first. + yield* untilAllReceipts([ + (r) => r.type === "rateLimit.recorded", + (r) => r.type === "userInput.recorded", + ]); + const after = yield* record(h.store); + assert.strictEqual(after.rateLimitedUntilMs, 3_600_000); + assert.strictEqual(after.userInputs[0]?.requestId, "req-1"); + }), + ); + }).pipe(scoped), + ); + + // --- 103, 104: boot grace ------------------------------------------------- + + it.effect("103: a long-idle thread does not fire on the first post-restart tick", () => + Effect.gen(function* () { + // A day of idleness on the projection clock. `processStartedAtMs` floors it, so the + // reactor sees one minute of idleness, not twenty-four hours. + const h = yield* harness({ shell: threadShell({ updatedAt: msToIso(-24 * HOUR) }) }); + yield* arm(h.store); + yield* withReactor( + h.deps, + Effect.gen(function* () { + yield* advancePolls(1); + assert.strictEqual(turnStarts(yield* Ref.get(h.dispatched)).length, 0); + assert.strictEqual((yield* record(h.store)).checkInsUsed, 0); + }), + ); + }).pipe(scoped), + ); + + it.effect("104: it fires once the idle threshold has passed since process start", () => + Effect.gen(function* () { + const h = yield* harness({ shell: threadShell({ updatedAt: msToIso(-24 * HOUR) }) }); + yield* arm(h.store); + yield* withReactor( + h.deps, + Effect.gen(function* () { + yield* advancePolls(14); + assert.strictEqual(turnStarts(yield* Ref.get(h.dispatched)).length, 0, "not yet"); + // A tight bound: it must fire within a few polls of the threshold, not eventually. + yield* advanceUntil(turnStartsAtLeast(h.dispatched, 1), "the check-in", 6); + assert.strictEqual(turnStarts(yield* Ref.get(h.dispatched)).length, 1); + }), + ); + }).pipe(scoped), + ); + + // --- 105, 106: breadcrumbs ------------------------------------------------ + + it.effect("105: a skip repeated across ten ticks appends one activity, not ten", () => + Effect.gen(function* () { + const h = yield* harness({ shell: threadShell() }); + yield* arm(h.store); + yield* h.store.setRateLimitedUntil(LOOP_THREAD_ID, 8 * HOUR); + yield* withReactor( + h.deps, + Effect.gen(function* () { + yield* advancePolls(10); + const skips = activitiesOfKind(yield* Ref.get(h.dispatched), LOOP_ACTIVITY_KINDS.skipped); + assert.strictEqual( + skips.length, + 1, + "an activity per tick would bump updatedAt and reset the reactor's own idle clock", + ); + assert.strictEqual((skips[0]!.payload as { reason: string }).reason, "rate_limited"); + assert.strictEqual(turnStarts(yield* Ref.get(h.dispatched)).length, 0); + assert.strictEqual((yield* record(h.store)).checkInsUsed, 0, "a skip spends nothing"); + }), + ); + }).pipe(scoped), + ); + + it.effect("106: a breadcrumb failure does not abort the check-in", () => + Effect.gen(function* () { + const h = yield* harness({ shell: threadShell() }); + yield* arm(h.store); + yield* Ref.set(h.failCommandTypes, new Set(["thread.activity.append"])); + yield* withReactor( + h.deps, + Effect.gen(function* () { + yield* advanceUntil(turnStartsAtLeast(h.dispatched, 1), "the check-in"); + assert.strictEqual(turnStarts(yield* Ref.get(h.dispatched)).length, 1); + assert.strictEqual( + activitiesOfKind(yield* Ref.get(h.dispatched), LOOP_ACTIVITY_KINDS.checkedIn).length, + 0, + "the note really did fail", + ); + assert.strictEqual((yield* record(h.store)).checkInsUsed, 1); + }), + ); + }).pipe(scoped), + ); + + // --- 107: shutdown -------------------------------------------------------- + + it.effect("107: interrupting the fibers leaves the persisted record consistent", () => + Effect.gen(function* () { + const h = yield* harness({ shell: threadShell() }); + yield* arm(h.store); + yield* Effect.scoped( + withReactor( + h.deps, + // Advance to the firing tick and tear the scope down without waiting for quiet, so + // the interruption lands at an arbitrary point in the decision. + Effect.gen(function* () { + yield* advancePolls(14); + yield* TestClock.adjust(60_000); + }), + ), + ); + + const persisted = yield* Effect.promise(() => NodeFSP.readFile(h.statePath, "utf8")); + // Decoded with the real schema, not `JSON.parse`: a file that no longer decodes is the + // failure mode that collapses to EMPTY_STATE and silently disarms every loop. + const parsed = yield* decodeLoopState(persisted); + const row = parsed.threads[LOOP_THREAD_ID]!; + assert.isAtMost(row.checkInsUsed, 1, "never more reservations than ticks"); + // The invariant that matters: a reservation is never half-written. If the counter moved + // the ledger moved with it, because both live in one atomic file rewrite. + if (row.checkInsUsed === 1) { + assert.isNotNull(row.lastCheckIn); + assert.strictEqual(row.checkIns.length, 1); + } + }).pipe(scoped), + ); + + // --- the master toggle, and the reverse states ----------------------------- + + it.effect("toggle off: nothing fires, and nothing is disarmed or stopped", () => + Effect.gen(function* () { + const h = yield* harness({ shell: threadShell() }); + yield* arm(h.store); + yield* h.store.setGlobal({ enabled: false }); + yield* withReactor( + h.deps, + Effect.gen(function* () { + yield* advancePolls(60); + assert.strictEqual(turnStarts(yield* Ref.get(h.dispatched)).length, 0); + const after = yield* record(h.store); + assert.isTrue(after.armed, "the toggle stands loops down; it disarms nothing"); + assert.isNull(after.stopped, "and it manufactures no terminal nobody chose"); + assert.strictEqual(after.checkInsUsed, 0, "budgets stay intact"); + const skips = activitiesOfKind(yield* Ref.get(h.dispatched), LOOP_ACTIVITY_KINDS.skipped); + assert.strictEqual(skips.length, 1); + assert.strictEqual((skips[0]!.payload as { reason: string }).reason, "disabled"); + }), + ); + }).pipe(scoped), + ); + + it.effect("takeover stops the loop as handed-back without refunding the budget", () => + Effect.gen(function* () { + const h = yield* harness({ shell: threadShell() }); + yield* arm(h.store); + yield* withReactor( + h.deps, + Effect.gen(function* () { + yield* advanceUntil(turnStartsAtLeast(h.dispatched, 1), "the first check-in"); + const afterFire = yield* record(h.store); + assert.strictEqual(afterFire.checkInsUsed, 1); + // A human message strictly after our own minted `createdAt`. Equal would be our own + // nudge, and would disarm every loop on its own first check-in. + yield* Ref.set( + h.shellRef, + threadShell({ + latestUserMessageAt: msToIso(Date.parse(afterFire.lastCheckIn!.createdAtIso) + 1_000), + }), + ); + yield* advanceUntil( + record(h.store).pipe(Effect.map((r) => r.stopped !== null)), + "the handback", + 40, + ); + const after = yield* record(h.store); + assert.strictEqual(after.stopped?.reason, "handed-back"); + assert.strictEqual(after.checkInsUsed, 1, "takeover is not a budget reset"); + assert.isFalse(after.armed); + }), + ); + }).pipe(scoped), + ); + + it.effect("a deleted thread disarms, unpins what the loop pinned, and ends its session", () => + Effect.gen(function* () { + const deadlineAtMs = 8 * HOUR; + const h = yield* harness({ shell: null }); + yield* arm(h.store, { deadlineAtMs, pinnedByLoop: true }); + yield* recurringCronPastDeadline(h.store, deadlineAtMs); + yield* withReactor( + h.deps, + Effect.gen(function* () { + yield* advanceUntil( + Ref.get(h.stopSessions).pipe(Effect.map((all) => all.length > 0)), + "the disarm, unpin and session stop", + 10, + ); + const after = yield* record(h.store); + assert.isFalse(after.armed); + assert.isNull(after.stopped, "a disarm is not a terminal state"); + assert.strictEqual(after.checkInsUsed, 0); + assert.isFalse(after.pinnedByLoop, "the pin the loop created is removed"); + assert.include(commandTypes(yield* Ref.get(h.dispatched)), "thread.unpin"); + assert.deepStrictEqual( + yield* Ref.get(h.stopSessions), + [LOOP_THREAD_ID], + "wakes are still recorded, and nothing left will ever bound them", + ); + }), + ); + }).pipe(scoped), + ); + + it.effect("a projection read failure skips the tick rather than disarming the loop", () => + Effect.gen(function* () { + const h = yield* harness({ shell: threadShell() }); + yield* arm(h.store); + yield* Ref.set(h.shellReadFails, true); + yield* withReactor( + h.deps, + Effect.gen(function* () { + yield* advancePolls(20); + const after = yield* record(h.store); + assert.isTrue(after.armed, "a transient SQL error is not a deleted thread"); + assert.isNull(after.stopped); + assert.strictEqual(turnStarts(yield* Ref.get(h.dispatched)).length, 0); + }), + ); + }).pipe(scoped), + ); + + it.effect( + "a projection failure on the pre-dispatch re-read is not reported as a missing thread", + () => + Effect.gen(function* () { + const h = yield* harness({ shell: threadShell() }); + yield* arm(h.store); + // One strike already on the record, so "did this abort add one?" is answerable. + yield* h.store.update(LOOP_THREAD_ID, (current) => ({ ...current, strikes: 1 })); + yield* withReactor( + h.deps, + Effect.gen(function* () { + yield* advancePolls(14); + const readsSoFar = yield* Ref.get(h.shellCalls); + // The guard block's read succeeds; the pre-dispatch re-read fails. + yield* Ref.set(h.shellFailsAfterCall, readsSoFar + 1); + const abortNoted = Ref.get(h.dispatched).pipe( + Effect.map((all) => + activitiesOfKind(all, LOOP_ACTIVITY_KINDS.skipped).some( + (a) => (a.payload as { reason?: string }).reason === "aborted_pre_dispatch", + ), + ), + ); + yield* advanceUntil(abortNoted, "the aborted attempt", 6); + + const abort = activitiesOfKind( + yield* Ref.get(h.dispatched), + LOOP_ACTIVITY_KINDS.skipped, + ).find((a) => (a.payload as { reason?: string }).reason === "aborted_pre_dispatch"); + assert.strictEqual( + (abort!.payload as { verdict: string }).verdict, + "projection_unavailable", + "a projection that did not answer is not a thread that is gone", + ); + assert.strictEqual(turnStarts(yield* Ref.get(h.dispatched)).length, 0); + const after = yield* record(h.store); + assert.isTrue(after.armed, "a hiccup never disarms"); + assert.isNull(after.stopped); + assert.strictEqual( + after.strikes, + 1, + "nothing was sent, so nothing about the agent was demonstrated", + ); + }), + ); + }).pipe(scoped), + ); + + it.effect("a takeover seen by the pre-dispatch re-read is recorded on that same tick", () => + Effect.gen(function* () { + const h = yield* harness({ shell: threadShell() }); + yield* arm(h.store); + yield* withReactor( + h.deps, + Effect.gen(function* () { + yield* advancePolls(14); + const readsSoFar = yield* Ref.get(h.shellCalls); + yield* Ref.set(h.shellOverrideRef, { + afterCall: readsSoFar + 1, + // Typed while the tick was mid-flight: later than any `createdAt` this nudge + // could carry, so the compare reads it as a takeover rather than as our own turn. + shell: threadShell({ latestUserMessageAt: msToIso(HOUR) }), + }); + yield* advanceUntil(isStopped(h.store), "the handback terminal", 6); + + const after = yield* record(h.store); + assert.strictEqual( + after.stopped?.reason, + "handed-back", + "the console must not keep saying 'watching' about a run that is already over", + ); + assert.isFalse(after.armed); + assert.strictEqual(turnStarts(yield* Ref.get(h.dispatched)).length, 0, "no nudge"); + assert.strictEqual( + activitiesOfKind(yield* Ref.get(h.dispatched), LOOP_ACTIVITY_KINDS.stopped).length, + 1, + ); + }), + ); + }).pipe(scoped), + ); + + it.effect("a stop request banked by the console is serviced exactly once", () => + Effect.gen(function* () { + const h = yield* harness({ shell: threadShell() }); + // Exactly what `POST /api/coil/loop` leaves behind on a disarm with pending wakes: a + // stopped record, nothing armed, and one banked request. + yield* arm(h.store); + yield* h.store.stop(LOOP_THREAD_ID, { + reason: "handed-back", + atMs: MINUTE, + detail: "disarmed from the console", + }); + yield* h.store.requestSessionStop(LOOP_THREAD_ID, MINUTE); + yield* withReactor( + h.deps, + Effect.gen(function* () { + yield* advanceUntil( + Ref.get(h.stopSessions).pipe(Effect.map((all) => all.length > 0)), + "the banked stop", + 6, + ); + // Ten more polls: the request is cleared, so it cannot re-fire every minute for the + // rest of the night. + yield* advancePolls(10); + assert.deepStrictEqual(yield* Ref.get(h.stopSessions), [LOOP_THREAD_ID]); + assert.strictEqual((yield* record(h.store)).stopRequestedAtMs, 0); + }), + ); + }).pipe(scoped), + ); + + it.effect("the done-file ends the run as done, with budget left over", () => + Effect.gen(function* () { + const h = yield* harness({ shell: threadShell() }); + yield* arm(h.store); + // Written after arming, so it is a signal rather than a leftover. + yield* writeSentinel(h.workspaceRoot, 5 * MINUTE); + yield* withReactor( + h.deps, + Effect.gen(function* () { + yield* advanceUntil( + record(h.store).pipe(Effect.map((r) => r.stopped !== null)), + "the done terminal", + 30, + ); + yield* settleQuiet; + const after = yield* record(h.store); + assert.strictEqual(after.stopped?.reason, "done"); + assert.strictEqual(after.checkInsUsed, 0, "done costs no check-ins"); + assert.deepStrictEqual( + yield* Ref.get(h.stopSessions), + [], + "done never kills the session — the agent said it finished", + ); + }), + ); + }).pipe(scoped), + ); + + it.effect("loop_done from the MCP toolkit ends the run as done and unpins", () => + Effect.gen(function* () { + // The toolkit's whole contract with the supervisor is two fields on the record. It + // never dispatches and never stops anything itself, so this is the only place the two + // halves meet — and the file channel and the tool channel must be indistinguishable + // once written, because `enableAgentBrowserAccess` off removes the tool entirely. + const h = yield* harness({ shell: threadShell() }); + yield* arm(h.store, { pinnedByLoop: true }); + yield* h.store.update(LOOP_THREAD_ID, (current) => ({ + ...current, + loopDoneAtMs: 5 * MINUTE, + loopDoneReason: "shipped the migration and the tests are green", + })); + yield* withReactor( + h.deps, + Effect.gen(function* () { + yield* advanceUntil( + Ref.get(h.dispatched).pipe( + Effect.map((all) => activitiesOfKind(all, LOOP_ACTIVITY_KINDS.stopped).length > 0), + ), + "the done terminal", + 10, + ); + const after = yield* record(h.store); + assert.strictEqual(after.stopped?.reason, "done"); + assert.notStrictEqual(after.stopped?.reason, "spent", "done is never reported as spent"); + assert.strictEqual(after.checkInsUsed, 0, "loop_done costs no check-ins"); + assert.include( + after.stopped?.detail ?? "", + "shipped the migration", + "the agent's own words survive onto the terminal", + ); + + const notes = activitiesOfKind(yield* Ref.get(h.dispatched), LOOP_ACTIVITY_KINDS.stopped); + assert.strictEqual(notes.length, 1); + assert.strictEqual((notes[0]!.payload as { cause: string }).cause, "loop_done"); + + assert.include(commandTypes(yield* Ref.get(h.dispatched)), "thread.unpin"); + assert.isFalse((yield* record(h.store)).pinnedByLoop); + assert.strictEqual(turnStarts(yield* Ref.get(h.dispatched)).length, 0); + }), + ); + }).pipe(scoped), + ); + + it.effect("a pending auto-resume stands the loop down without spending budget", () => + Effect.gen(function* () { + const h = yield* harness({ shell: threadShell(), autoResumePending: true }); + yield* arm(h.store); + yield* withReactor( + h.deps, + Effect.gen(function* () { + yield* advancePolls(40); + assert.strictEqual(turnStarts(yield* Ref.get(h.dispatched)).length, 0); + const after = yield* record(h.store); + assert.strictEqual(after.checkInsUsed, 0); + assert.isTrue(after.armed); + const skips = activitiesOfKind(yield* Ref.get(h.dispatched), LOOP_ACTIVITY_KINDS.skipped); + assert.strictEqual( + (skips[0]!.payload as { reason: string }).reason, + "auto_resume_pending", + ); + }), + ); + }).pipe(scoped), + ); +}); + +/** + * The seam the whole feature hangs off: the Claude adapter's hooks. + * + * `server.ts` composes the supervisor with `Layer.provide`, not `provideMerge` — so + * `LoopStore` is discharged and does *not* reach the app context the adapter's fiber runs in. + * Every other test in this file uses `provideMerge` for convenience, which is exactly the + * shape that hid this: `Effect.serviceOption(LoopStore)` answered `Some` in the tests and + * `None` in production, and the hooks were never installed on any real machine. + */ +describe("LoopReactor — installing the adapter's hooks", () => { + it.effect("a server-shaped composition installs them, even with no LoopStore in context", () => + Effect.gen(function* () { + const h = yield* harness({ shell: threadShell() }); + // `Layer.provide`, exactly as `coil/index.ts` composes it. Nothing is re-exported. + const serverShaped = LoopReactorLive.pipe(Layer.provide(h.deps)); + yield* Effect.provide( + Effect.gen(function* () { + assert.isTrue( + Option.isNone(yield* Effect.serviceOption(LoopStore)), + "this is the adapter's world: the store is genuinely not in context", + ); + const hooks = yield* loopHooksFor(LOOP_THREAD_ID); + assert.isDefined(hooks, "the adapter must still get its hooks"); + assert.deepStrictEqual(Object.keys(hooks!).sort(), [ + "PostToolUse", + "Stop", + "SubagentStop", + ]); + }), + serverShaped, + ); + // Outside the supervisor's scope there is nothing to record into, and the holder says so. + assert.isUndefined(yield* loopHooksFor(LOOP_THREAD_ID)); + }).pipe(scoped), + ); + + it.effect("the kill switch still means no hooks at all", () => + Effect.gen(function* () { + const h = yield* harness({ shell: threadShell() }); + process.env.COIL_LOOP_ENABLED = "0"; + try { + yield* Effect.provide( + Effect.map(loopHooksFor(LOOP_THREAD_ID), (hooks) => { + assert.isUndefined(hooks); + }), + LoopReactorLive.pipe(Layer.provide(h.deps)), + ); + } finally { + delete process.env.COIL_LOOP_ENABLED; + } + }).pipe(scoped), + ); +}); diff --git a/apps/server/src/coil/loop/Reactor.ts b/apps/server/src/coil/loop/Reactor.ts new file mode 100644 index 000000000000..902ee705ab31 --- /dev/null +++ b/apps/server/src/coil/loop/Reactor.ts @@ -0,0 +1,828 @@ +/** + * LoopReactor — the supervisor that keeps an agent thread working unattended. + * + * Self-starts three scoped fibers at layer construction, exactly as + * `autoResume/Reactor.ts` does, so the only upstream seam the loops feature takes stays the + * three lines in `ClaudeAdapter.ts`: + * + * 1. **tick** — every `pollMs`, list the armed loops, resolve each one's shell, project + * shell and done-file, hand the lot to the pure `decide`, and execute the single action + * it returns. + * 2. **rate-limit tap** — subscribes to `providerService.streamEvents` and writes + * `rateLimitedUntilMs` durably on a rejected verdict. Necessary because auto-resume only + * arms when *its* per-thread switch is on: on a thread where the user turned it off a + * usage limit produces no pending resume, guard 9 passes, and the loop would otherwise + * nudge straight into a live five-hour limit. + * 3. **user inputs** — `recordUserInputs` from `userInputs.ts`, forked here rather than + * given its own layer so the whole feature is one `Layer.effectDiscard`. + * + * Everything that *decides* lives in `decide.ts` / `guards.ts` and is pure. This file only + * gathers facts, executes actions and writes breadcrumbs, which is why it has no branching + * on guard order at all. + * + * ## Never `getSnapshot()` + * + * `autoResume/Reactor.ts` reads the whole projection per pass; it has an open OOM follow-up + * for exactly that. This reactor uses `getThreadShellById` / `getProjectShellById`, and a + * tick with nothing armed issues **zero** queries of any kind — no SQL, no filesystem. + * + * ## The three firing disciplines (BACKEND §5) + * + * 1. **Reserve before dispatch.** `store.recordCheckIn` persists before `engine.dispatch`. + * This is the only unbounded path in the design: a provider that cannot spawn would + * otherwise tight-loop. It burns budget instead — six attempts, not four hundred and + * eighty a night. + * 2. **Re-read the shell after the guard block and again before dispatch.** The re-read is + * re-decided against the *pre-reserve* record, so the wake race closes without the + * reservation we just wrote voting on its own necessity. + * 3. **Repair the keep-active pin.** The decider clears `settledOverride` for any non-null + * value, so a nudge silently destroys a user's keep-active pin. It is restored with + * `thread.unsettle` **only** when the pre-dispatch shell already carried it, so this path + * can never create a pin. + * + * @module coil/loop/Reactor + */ + +import type { + OrchestrationProjectShell, + OrchestrationThreadActivityTone, + OrchestrationThreadShell, + ProviderRuntimeEvent, +} from "@t3tools/contracts"; +import { CommandId, EventId, MessageId, ProjectId, ThreadId } from "@t3tools/contracts"; +import * as Cause from "effect/Cause"; +import * as Clock from "effect/Clock"; +import * as Crypto from "effect/Crypto"; +import * as DateTime from "effect/DateTime"; +import * as Duration from "effect/Duration"; +import * as Effect from "effect/Effect"; +import * as FileSystem from "effect/FileSystem"; +import * as Layer from "effect/Layer"; +import * as Option from "effect/Option"; +import * as Stream from "effect/Stream"; + +import { OrchestrationEngineService } from "../../orchestration/Services/OrchestrationEngine.ts"; +import { ProjectionSnapshotQuery } from "../../orchestration/Services/ProjectionSnapshotQuery.ts"; +import { ProviderService } from "../../provider/Services/ProviderService.ts"; +import { classifyRateLimit } from "../autoResume/classifyRateLimit.ts"; +import { AutoResumeStore } from "../autoResume/state.ts"; +import { composeCheckInPrompt, resolveConfig } from "./config.ts"; +import { decide, resolveWake } from "./decide.ts"; +import { hasPendingCrons } from "./guards.ts"; +import { installLoopStore } from "./hooksRegistry.ts"; +import { receiptEmitter } from "./receipts.ts"; +import { readSentinel } from "./sentinel.ts"; +import { LoopStore, type LoopGlobalSettings, type LoopRecord, type StopRecord } from "./state.ts"; +import type { + DisarmAction, + FireAction, + LoopDecisionInput, + StandDownAction, + StandDownReason, + StopAction, +} from "./types.ts"; +import { recordUserInputs } from "./userInputs.ts"; + +/** + * The timeline vocabulary, exported so the console keys off these constants rather than + * re-typing the strings. `kind` is an open `TrimmedNonEmptyString` on the contract and + * `payload` is `Schema.Unknown`, so breadcrumbs cost zero contract edits. + */ +export const LOOP_ACTIVITY_KINDS = { + checkedIn: "coil.loop.checked-in", + skipped: "coil.loop.skipped", + wakeLost: "coil.loop.wake-lost", + stopped: "coil.loop.stopped", + disarmed: "coil.loop.disarmed", + pinRepairFailed: "coil.loop.pin-repair-failed", +} as const; + +/** + * Skip reasons worth a timeline entry. + * + * Deliberately not every `StandDownReason`. `not_idle` and `check_in_floor` are what a + * *healthy* supervised thread reports on almost every tick, and appending an activity bumps + * `updatedAt` — the very value the trigger measures. A breadcrumb on the healthy path would + * therefore reset the loop's own idle clock, which is the self-sustaining loop edge detection + * exists to prevent; making the common case silent removes the possibility rather than + * bounding it. The live reason is always available from `GET /api/coil/loop`, so nothing is + * hidden — only the *history* is limited to the reasons a human would want to see later. + * + * `not_armed` and `stopped` cannot occur here: `listArmed` filters on `armed`, and `stop` + * clears it. + */ +const NOTEWORTHY_SKIPS: ReadonlySet = new Set([ + "disabled", + "snoozed", + "pending_approval", + "pending_user_input", + "pending_plan", + "auto_resume_pending", + "rate_limited", + "self_pacing", + "ceiling", +]); + +/** + * How long to stand down for a rejected verdict that carries no `resetsAt`. + * + * The SDK documents `resetsAt` as optional. Writing nothing would leave the loop free to + * nudge into a live limit, which is the hole this fiber exists to close; holding forever + * would spend the run's deadline standing down. An hour is the compromise, and it + * self-corrects in the expensive direction only: the one nudge that follows an expired hold + * produces a fresh rejected event, which re-extends it. + */ +const RATE_LIMIT_FALLBACK_HOLD_MS = 60 * 60_000; + +/** Human-readable summary for a stop, kept in one place so the console reads one vocabulary. */ +const STOP_SUMMARY: Record = { + done: "Loop finished: the agent signalled done.", + spent: "Loop ended: budget or deadline exhausted.", + stalled: "Loop stopped: two consecutive check-ins made no progress.", + "handed-back": "Loop stopped: you took over.", +}; + +const makeSupervisor = Effect.gen(function* () { + const engine = yield* OrchestrationEngineService; + const providerService = yield* ProviderService; + const snapshotQuery = yield* ProjectionSnapshotQuery; + const store = yield* LoopStore; + const autoResumeStore = yield* AutoResumeStore; + const crypto = yield* Crypto.Crypto; + const fs = yield* FileSystem.FileSystem; + const config = resolveConfig(); + // Optional, and absent in every production graph. See `receipts.ts`. + const receipts = yield* receiptEmitter; + + /** + * The boot-grace floor, captured once. + * + * Without it every armed thread fires on the first post-restart tick, because a projection + * `updatedAt` from before the outage reads as hours of idleness. It also covers laptop + * sleep and upstream's restart-continuation window (#9167), where a continued thread is + * briefly `starting` with no `activeTurnId`. + */ + const processStartedAtMs = yield* Clock.currentTimeMillis; + + const isoNow = DateTime.now.pipe(Effect.map(DateTime.formatIso)); + + /** + * Last skip reason announced per thread, so a loop standing down for the same reason + * across ten ticks appends one activity rather than ten. In memory on purpose: a restart + * re-announcing once is correct (the console has no other record of *why* it is quiet), + * and persisting it would grow the record for a fact with a one-process lifetime. + */ + const lastSkipReason = new Map(); + + // --- timeline ------------------------------------------------------------- + + /** + * Best-effort breadcrumb. A timeline failure must never abort a check-in — the nudge is + * the product, the note about it is not. + */ + const appendActivity = ( + threadId: string, + tone: OrchestrationThreadActivityTone, + kind: string, + summary: string, + payload: Record = {}, + ) => + Effect.gen(function* () { + const commandUuid = yield* crypto.randomUUIDv4; + const eventUuid = yield* crypto.randomUUIDv4; + const createdAt = yield* isoNow; + yield* engine.dispatch({ + type: "thread.activity.append", + commandId: CommandId.make(`coil-loop-activity:${commandUuid}`), + threadId: ThreadId.make(threadId), + activity: { + id: EventId.make(`coil-loop:${eventUuid}`), + tone, + kind, + summary, + payload, + turnId: null, + createdAt, + }, + createdAt, + }); + return true; + }).pipe( + Effect.catchCause((cause) => + Effect.logDebug("coil loop: activity append failed", { + threadId, + kind, + cause: Cause.pretty(cause), + }).pipe(Effect.as(false)), + ), + ); + + // --- reads ---------------------------------------------------------------- + + /** + * One thread shell by id. + * + * A projection failure resolves to `null`, which `decide` reads as "the thread is gone" + * and answers with a **disarm**. That is the wrong answer for a transient SQL error, so + * the failure is distinguished here and the whole thread is skipped for this tick instead + * — a tick that reads nothing is always safe, a tick that disarms on a hiccup is not. + */ + const readShell = (threadId: string) => + snapshotQuery.getThreadShellById(ThreadId.make(threadId)).pipe( + Effect.map((option) => ({ ok: true, shell: Option.getOrNull(option) }) as const), + Effect.catch((cause) => + Effect.logWarning("coil loop: thread shell read failed", { threadId, cause }).pipe( + Effect.as({ ok: false, shell: null } as const), + ), + ), + ); + + /** The project row, for `workspaceRoot`. Every failure mode is "no workspace root". */ + const readProjectShell = ( + shell: OrchestrationThreadShell, + ): Effect.Effect => + snapshotQuery.getProjectShellById(ProjectId.make(shell.projectId)).pipe( + Effect.map(Option.getOrNull), + Effect.orElseSucceed(() => null), + ); + + /** + * Assemble the pure decision input. + * + * The two done channels arrive here side by side and `doneSignal` takes the newer of them: + * `sentinelAtMs` is the done-file's mtime, `loopDoneAtMs` is what the `loop_done` MCP tool + * wrote to the record. The file stays the primary contract because it works from a plain + * terminal with no MCP at all — `enableAgentBrowserAccess` off removes the toolkit + * entirely — so neither channel may depend on the other. + */ + const gatherInput = (record: LoopRecord, shell: OrchestrationThreadShell, nowMs: number) => + Effect.gen(function* () { + const project = yield* readProjectShell(shell); + const sentinel = yield* readSentinel( + fs, + { worktreePath: shell.worktreePath, workspaceRoot: project?.workspaceRoot ?? null }, + { armedAtMs: record.armedAtMs }, + ); + const autoResume = yield* autoResumeStore.getThread(shell.id); + return { + nowMs, + processStartedAtMs, + record, + shell, + // `stale` still carries its mtime: `doneSignal` owns the freshness compare, so the + // reactor never re-implements it and the two can never disagree. + sentinelAtMs: sentinel.kind === "absent" ? null : sentinel.mtimeMs, + loopDoneAtMs: record.loopDoneAtMs, + autoResumePending: autoResume.pending !== null, + config, + workspaceRoot: project?.workspaceRoot ?? null, + }; + }); + + // --- actions -------------------------------------------------------------- + + const onStandDown = (threadId: string, action: StandDownAction) => + Effect.gen(function* () { + if (!NOTEWORTHY_SKIPS.has(action.reason)) { + // Still an edge: leaving a noteworthy reason for a quiet one must let the next + // occurrence of the noteworthy reason announce itself again. + lastSkipReason.delete(threadId); + return; + } + if (lastSkipReason.get(threadId) === action.reason) return; + lastSkipReason.set(threadId, action.reason); + yield* appendActivity( + threadId, + "info", + LOOP_ACTIVITY_KINDS.skipped, + `Loop standing by: ${action.reason.replaceAll("_", " ")}.`, + action.untilMs === null + ? { reason: action.reason } + : { reason: action.reason, until: action.untilMs }, + ); + }); + + /** + * Remove the pin, but only one this feature created. + * + * `pinnedByLoop` is cleared only when the unpin actually landed, so a failed dispatch + * leaves the loop still responsible for a pin it made rather than orphaning it. + */ + const unpinIfOurs = (threadId: string, record: LoopRecord) => + Effect.gen(function* () { + if (!record.pinnedByLoop) return; + const uuid = yield* crypto.randomUUIDv4; + const unpinned = yield* engine + .dispatch({ + type: "thread.unpin", + commandId: CommandId.make(`coil-loop-unpin:${uuid}`), + threadId: ThreadId.make(threadId), + }) + .pipe( + Effect.as(true), + Effect.catchCause((cause) => + Effect.logWarning("coil loop: unpin failed", { + threadId, + cause: Cause.pretty(cause), + }).pipe(Effect.as(false)), + ), + ); + if (unpinned) { + yield* store.update(threadId, (current) => ({ ...current, pinnedByLoop: false })); + } + }); + + /** + * End the provider session. + * + * A bound that cannot stop the agent is not a bound: T3 has no write handle on the binary's + * cron table, so the only way to stop a self-paced run walking through its own deadline is + * to end the session the crons live in. Blunt on purpose — it takes any live background + * work in that session with it — and reached only where `decide` (or `hasPendingCrons` on + * the disarm path) says wakes are still pending. + */ + const stopProviderSession = (threadId: string) => + providerService.stopSession({ threadId: ThreadId.make(threadId) }).pipe( + Effect.catchCause((cause) => + Effect.logWarning("coil loop: stopSession failed", { + threadId, + cause: Cause.pretty(cause), + }), + ), + ); + + const onStop = (threadId: string, record: LoopRecord, action: StopAction, nowMs: number) => + Effect.gen(function* () { + // `loop_done(reason)` is the one channel that carries the agent's own words. The + // decision table never sees them — it is pure and takes a timestamp — so they are + // stitched on here, where the terminal is actually written and read. + const spoken = action.cause === "loop_done" ? record.loopDoneReason?.trim() : null; + const detail = spoken ? `${action.detail}: ${spoken}` : action.detail; + yield* store.stop(threadId, { reason: action.outcome, atMs: nowMs, detail }); + lastSkipReason.delete(threadId); + yield* unpinIfOurs(threadId, record); + if (action.stopSession) yield* stopProviderSession(threadId); + yield* appendActivity( + threadId, + // `spent` is a normal ending, not a fault — zinc, never red, and never `done`. + action.outcome === "stalled" ? "error" : "info", + LOOP_ACTIVITY_KINDS.stopped, + STOP_SUMMARY[action.outcome], + { + reason: action.outcome, + cause: action.cause, + detail, + checkInsUsed: record.checkInsUsed, + of: record.maxCheckIns, + }, + ); + yield* receipts.emit({ type: "stopped", threadId, outcome: action.outcome }); + }); + + /** + * Service the stop requests the console banked. + * + * `POST /api/coil/loop` cannot end a session itself — `ProviderService` is not already a + * requirement of upstream's `makeRoutesLayer`, and taking it would widen an upstream + * signature — so a disarm that left the agent's own wakes pending records the request and + * this is where it lands. The list arrives with the tick's own single read of the store, so + * a tick with nothing banked costs nothing at all: no second trip through the synchronized + * ref, and the "zero queries when nothing is armed" floor is untouched. + */ + const serviceStopRequests = (requested: ReadonlyArray<{ readonly threadId: string }>) => + Effect.forEach( + requested, + (entry) => + // Cleared first, so a `stopSession` that fails cannot leave a request that retries + // every minute for the rest of the night. + store + .clearSessionStopRequest(entry.threadId) + .pipe( + Effect.andThen(stopProviderSession(entry.threadId)), + Effect.andThen( + receipts.emit({ type: "stopRequest.serviced", threadId: entry.threadId }), + ), + ), + { discard: true }, + ); + + const onDisarm = (threadId: string, record: LoopRecord, action: DisarmAction, nowMs: number) => + Effect.gen(function* () { + yield* store.disarm(threadId); + lastSkipReason.delete(threadId); + yield* unpinIfOurs(threadId, record); + // The thread is gone or archived but its session may still hold scheduled wakes, and + // nothing left will ever bound them. + if (hasPendingCrons(record, nowMs)) yield* stopProviderSession(threadId); + yield* appendActivity( + threadId, + "info", + LOOP_ACTIVITY_KINDS.disarmed, + `Loop disarmed: ${action.reason.replaceAll("_", " ")}.`, + { reason: action.reason }, + ); + yield* receipts.emit({ type: "disarmed", threadId }); + }); + + /** + * The nudge. + * + * Byte-for-byte the shape a keystroke produces, with `runtimeMode` and `interactionMode` + * **copied from the shell** rather than defaulted — a loop turn that silently downgraded a + * thread's runtime mode would be a security regression, not a cosmetic one. + */ + const dispatchCheckIn = (shell: OrchestrationThreadShell, text: string, createdAt: string) => + Effect.gen(function* () { + const commandUuid = yield* crypto.randomUUIDv4; + const messageUuid = yield* crypto.randomUUIDv4; + yield* engine.dispatch({ + type: "thread.turn.start", + commandId: CommandId.make(`coil-loop:${commandUuid}`), + threadId: shell.id, + message: { + messageId: MessageId.make(`coil-loop:${messageUuid}`), + role: "user", + text, + attachments: [], + }, + runtimeMode: shell.runtimeMode, + interactionMode: shell.interactionMode, + createdAt, + }); + }); + + /** + * Restore the keep-active pin the decider just cleared. + * + * Issued only when the pre-dispatch shell carried `settledOverride === "active"`, so it can + * never create a pin the user did not have. A failure here logs **and** posts an + * error-tone breadcrumb: the user asked for this thread to stay in the active list, and a + * silent loss of that is exactly the class of bug this fork keeps paying for. + */ + const repairPin = (threadId: string) => + Effect.gen(function* () { + const uuid = yield* crypto.randomUUIDv4; + yield* engine.dispatch({ + type: "thread.unsettle", + commandId: CommandId.make(`coil-loop-unsettle:${uuid}`), + threadId: ThreadId.make(threadId), + reason: "user", + }); + }).pipe( + Effect.catchCause((cause) => + Effect.logWarning("coil loop: keep-active pin repair failed", { + threadId, + cause: Cause.pretty(cause), + }).pipe( + Effect.andThen( + appendActivity( + threadId, + "error", + LOOP_ACTIVITY_KINDS.pinRepairFailed, + "Could not restore this thread's keep-active pin after the loop check-in.", + { cause: Cause.pretty(cause) }, + ), + ), + ), + ), + ); + + /** + * Fire a check-in, in the order the three disciplines require. + * + * Reserve, then re-read, then dispatch. The re-read is re-decided against the **pre-reserve** + * record so the reservation we just wrote cannot vote on its own necessity — decided against + * the post-reserve record, guard 11's check-in floor would abort every single fire. + */ + const onFire = ( + input: LoopDecisionInput & { + readonly shell: OrchestrationThreadShell; + readonly workspaceRoot: string | null; + }, + action: FireAction, + ) => + Effect.gen(function* () { + const threadId = input.shell.id; + const record = input.record; + const createdAtIso = yield* isoNow; + + // 1. RESERVE. Before anything that can fail, so a provider that cannot spawn burns + // budget instead of tight-looping. + const reserved = yield* store.recordCheckIn({ + threadId, + firedAtMs: input.nowMs, + createdAtIso, + // Where the thread stood when we nudged. `updatedAt` is the cursor the ledger slices + // activities between later, and it is the same value the trigger measured. + activityCursor: input.shell.updatedAt, + }); + // Carry the strike projection and settle the *previous* row's verdict. Keyed on `n` + // rather than an index, so the ledger's 20-row cap cannot shift it onto a stranger. + yield* store.update(threadId, (current) => ({ + ...current, + strikes: action.checkIn.strikes, + checkIns: current.checkIns.map((row) => + row.n === reserved.n - 1 ? { ...row, outcome: action.checkIn.previousOutcome } : row, + ), + })); + lastSkipReason.delete(threadId); + yield* receipts.emit({ type: "checkIn.reserved", threadId, n: reserved.n }); + + // 2. RE-READ, and re-decide. Guard 2 comes back with a fresh `global` here too, so the + // master toggle is never one tick stale at the moment money is spent. + const fresh = yield* readShell(threadId); + if (!fresh.ok) { + // A projection that failed to answer is NOT a missing thread, and reporting it as one + // sends a reader hunting for a thread that is fine. The reservation still stands — the + // reserve-before-dispatch discipline is not conditional on why the dispatch did not + // happen — but the strike projection is rolled back to what the record carried before + // this tick: nothing was sent, so nothing about the agent was demonstrated, and two + // hiccups in a row must not read as a stalled run. + yield* store.update(threadId, (current) => ({ ...current, strikes: record.strikes })); + yield* appendActivity( + threadId, + "info", + LOOP_ACTIVITY_KINDS.skipped, + "Loop check-in aborted: the thread index did not answer in time.", + { reason: "aborted_pre_dispatch", verdict: "projection_unavailable", n: reserved.n }, + ); + yield* receipts.emit({ + type: "checkIn.aborted", + threadId, + reason: "projection_unavailable", + }); + return; + } + // A `null` shell here is a thread that really is gone; `decide` answers it with a + // disarm, and naming it in the guard is also what narrows it for the dispatch below. + const shell = fresh.shell; + const global = yield* store.getGlobal; + const verdict = decide({ ...input, record, global, shell, nowMs: input.nowMs }); + if (shell === null || verdict.type !== "fire") { + // The reservation stands. It is spent, and that is the discipline: an abort needs + // the thread to have moved inside one tick, which resets the idle clock, so this can + // never repeat tightly — and refunding here would make "reserve before dispatch" + // conditional on a judgement the fast path cannot make. + yield* appendActivity( + threadId, + "info", + LOOP_ACTIVITY_KINDS.skipped, + "Loop check-in aborted: the thread moved before the nudge was sent.", + { reason: "aborted_pre_dispatch", verdict: verdict.type, n: reserved.n }, + ); + // What the re-read saw is as real as what a tick sees: a takeover observed here is a + // handback, and a thread that vanished under us is a disarm. Leaving them for the next + // tick would keep the console saying "watching" about a run that is already over, and + // would leave a pin the loop owns in place for another poll interval. + if (verdict.type === "stop") yield* onStop(threadId, record, verdict, input.nowMs); + if (verdict.type === "disarm") yield* onDisarm(threadId, record, verdict, input.nowMs); + yield* receipts.emit({ type: "checkIn.aborted", threadId, reason: verdict.type }); + return; + } + + // 3. COMPOSE. Banked answers are marked delivered only after the dispatch, and only + // the ids actually included, so an answer that landed mid-composition is not lost. + const banked = yield* store.listUndeliveredAnswers(threadId); + const prompt = yield* composeCheckInPrompt({ + worktreePath: shell.worktreePath, + workspaceRoot: input.workspaceRoot, + overridePrompt: record.overridePrompt, + checkInNumber: reserved.n, + maxCheckIns: record.maxCheckIns, + deadlineAtMs: record.deadlineAtMs, + nowMs: input.nowMs, + goal: record.goal, + bankedAnswers: banked.map((entry) => ({ + id: entry.id, + question: entry.question, + answer: entry.answer ?? "", + })), + }); + + // 4. DISPATCH. `createdAtIso` is the same string `lastCheckIn` recorded, which is what + // makes the handback compare exact: the decider stamps the user message with the + // command's `createdAt`, so our own nudge can never read as a takeover. + const sent = yield* dispatchCheckIn(shell, prompt.text, createdAtIso).pipe( + Effect.as(true), + Effect.catchCause((cause) => + Effect.logWarning("coil loop: check-in dispatch failed", { + threadId, + cause: Cause.pretty(cause), + }).pipe(Effect.as(false)), + ), + ); + if (!sent) { + yield* receipts.emit({ type: "checkIn.aborted", threadId, reason: "dispatch_failed" }); + return; + } + + yield* store.markBlockersDelivered(threadId, prompt.deliveredBlockerIds); + yield* receipts.emit({ + type: "blockers.delivered", + threadId, + ids: prompt.deliveredBlockerIds, + }); + + // 5. REPAIR, only when the pin was already there. + if (shell.settledOverride === "active") yield* repairPin(threadId); + + // 6. BREADCRUMBS. A lost wake is the strongest signal in the design, so it gets its own + // row with the numbers needed to re-diagnose it. + if (action.degrade === "wake_lost") { + const wake = resolveWake(input); + // One field carries both degraded states; `gate_off` is the more actionable of the + // two (with the gate off no wake can ever fire) so it is never clobbered. + if (record.degraded === null) yield* store.setDegraded(threadId, "wake_lost"); + yield* appendActivity( + threadId, + "error", + LOOP_ACTIVITY_KINDS.wakeLost, + "A scheduled wake never landed. T3 covered it.", + wake === null + ? { cronId: null } + : { cronId: wake.cronId, expectedAtMs: wake.atMs, graceMs: wake.graceMs }, + ); + } + yield* appendActivity( + threadId, + "info", + LOOP_ACTIVITY_KINDS.checkedIn, + `Loop check-in ${reserved.n} of ${record.maxCheckIns}.`, + { n: reserved.n, of: record.maxCheckIns, firedAtMs: input.nowMs, source: prompt.source }, + ); + yield* receipts.emit({ type: "checkIn.dispatched", threadId }); + }); + + // --- the tick ------------------------------------------------------------- + + const evaluateOne = ( + entry: { readonly threadId: string; readonly record: LoopRecord }, + global: LoopGlobalSettings, + armedCount: number, + nowMs: number, + ) => + Effect.gen(function* () { + const { threadId, record } = entry; + const lookup = yield* readShell(threadId); + if (!lookup.ok) return; + if (lookup.shell === null) { + // No shell and no read error: the thread really is gone. `decide` answers disarm. + const action = decide({ + nowMs, + processStartedAtMs, + record, + global, + shell: null, + sentinelAtMs: null, + loopDoneAtMs: record.loopDoneAtMs, + autoResumePending: false, + armedCount, + config, + }); + if (action.type === "disarm") yield* onDisarm(threadId, record, action, nowMs); + return; + } + + const gathered = yield* gatherInput(record, lookup.shell, nowMs); + const input = { ...gathered, global, armedCount }; + const action = decide(input); + switch (action.type) { + case "stand_down": + return yield* onStandDown(threadId, action); + case "stop": + return yield* onStop(threadId, record, action, nowMs); + case "disarm": + return yield* onDisarm(threadId, record, action, nowMs); + case "fire": + return yield* onFire(input, action); + } + }).pipe( + // Per thread, so one thread's defect cannot stop the others in the same pass — the + // supervisor is a shared resource and a single bad record must not retire it. + Effect.catchCause((cause) => + Effect.logWarning("coil loop: thread evaluation failed", { + threadId: entry.threadId, + cause: Cause.pretty(cause), + }), + ), + ); + + /** + * The end-of-tick receipt. + * + * Assembled only when someone is listening: the empty-armed path has no `nowMs` in hand, + * and reading one there would put a clock read on every poll of every install for a + * subscriber that exists in tests alone. + */ + const emitTickCompleted = (armedCount: number, nowMs: number | null) => + receipts.enabled + ? Effect.gen(function* () { + const at = nowMs ?? (yield* Clock.currentTimeMillis); + yield* receipts.emit({ type: "tick.completed", armedCount, nowMs: at }); + }) + : Effect.void; + + const tick = Effect.gen(function* () { + // ONE read of the store per tick, for both work lists. Two reads would be one more + // scheduler turn every minute forever, for a stop-request list that is nearly always + // empty. + const { armed, stopRequested } = yield* store.listWork; + if (stopRequested.length > 0) yield* serviceStopRequests(stopRequested); + // Zero SQL and zero filesystem work when nothing is armed. This is the whole reason the + // by-id reads replaced `getSnapshot()`. + if (armed.length === 0) return yield* emitTickCompleted(0, null); + const global = yield* store.getGlobal; + const nowMs = yield* Clock.currentTimeMillis; + yield* Effect.forEach(armed, (entry) => evaluateOne(entry, global, armed.length, nowMs), { + discard: true, + }); + yield* emitTickCompleted(armed.length, nowMs); + }); + + // --- the rate-limit tap --------------------------------------------------- + + const onRuntimeEvent = (event: ProviderRuntimeEvent) => + Effect.gen(function* () { + if (event.type !== "account.rate-limits.updated") return; + // No provider gate: `classifyRateLimit` refuses anything that is not a recognisable + // rate-limit info object, which is the real discriminator, and any adapter forwarding + // that shape is reporting a genuine account limit. + const verdict = classifyRateLimit(event.payload.rateLimits); + if (!verdict?.rejected) return; + + // Only armed threads accrue a record. `coil-loop.json` is rewritten atomically on every + // mutation and rate limits land on threads that will never be supervised; recording + // them all would grow one shared file without bound for no reader. Same rule as + // `userInputs.ts`. + const record = yield* store.getThread(event.threadId); + if (!record.armed) return; + + const nowMs = yield* Clock.currentTimeMillis; + const untilMs = verdict.resetsAtMs ?? nowMs + RATE_LIMIT_FALLBACK_HOLD_MS; + // The longer of the two wins. Two limits can be live at once (a five-hour and a + // seven-day), and holding is a non-consuming skip: over-holding costs deference time, + // under-holding burns a check-in against a wall. + const heldUntilMs = Math.max(record.rateLimitedUntilMs, untilMs); + yield* store.setRateLimitedUntil(event.threadId, heldUntilMs); + yield* receipts.emit({ + type: "rateLimit.recorded", + threadId: event.threadId, + untilMs: heldUntilMs, + }); + }); + + // --- lifecycle ------------------------------------------------------------ + + if (!config.enabled) { + yield* Effect.logInfo("coil loop: disabled via COIL_LOOP_ENABLED"); + return; + } + + // The Claude adapter builds its hooks from `loopHooksFor`, and its fiber runs in upstream's + // layer graph where `LoopStore` has already been discharged — so it cannot reach the store + // through the context. Publish it for the life of this layer's scope. See `hooksRegistry.ts`. + yield* installLoopStore(store); + yield* receipts.emit({ type: "hooks.installed" }); + + yield* Effect.forkScoped( + Stream.runForEach(providerService.streamEvents, (event) => + onRuntimeEvent(event).pipe( + Effect.catchCause((cause) => + Effect.logWarning("coil loop: rate-limit tap failed", { + eventType: event.type, + cause: Cause.pretty(cause), + }), + ), + ), + ), + ); + + yield* Effect.forkScoped(recordUserInputs(store, providerService, receipts)); + + yield* Effect.forkScoped( + Effect.gen(function* () { + // Sleep FIRST. The projection still shows crashed turns as running until the + // `reconcile.interrupted-turns` startup phase runs, and `processStartedAtMs` only + // floors the idle clock — it does not stop the first pass reading a half-built world. + yield* Effect.sleep(Duration.millis(config.pollMs)); + return yield* Effect.gen(function* () { + yield* tick; + yield* Effect.sleep(Duration.millis(config.pollMs)); + }).pipe( + // Inside the loop, so a defect in one pass cannot kill the only tick fiber and + // silently retire supervision for every thread on the machine. + Effect.catchCause((cause) => + Effect.logWarning("coil loop: tick failed", { cause: Cause.pretty(cause) }), + ), + Effect.forever, + ); + }), + ); + + yield* Effect.logInfo("coil loop: supervisor started", { + pollMs: config.pollMs, + processStartedAtMs, + }); +}); + +export const LoopReactorLive = Layer.effectDiscard(makeSupervisor); diff --git a/apps/server/src/coil/loop/config.test.ts b/apps/server/src/coil/loop/config.test.ts new file mode 100644 index 000000000000..7eca76d53a9e --- /dev/null +++ b/apps/server/src/coil/loop/config.test.ts @@ -0,0 +1,384 @@ +// @effect-diagnostics nodeBuiltinImport:off +import * as NodeServices from "@effect/platform-node/NodeServices"; +import { assert, describe, it } from "@effect/vitest"; +import * as Effect from "effect/Effect"; +import * as FileSystem from "effect/FileSystem"; +import * as Path from "effect/Path"; +import * as NodePath from "node:path"; + +import { + type CheckInPromptInput, + composeCheckInPrompt, + DEFERENCE_LINE, + LOOP_DONE_RELATIVE_PATH, + LOOP_PROMPT_RELATIVE_PATH, + resolveConfig, + resolveLoopRoots, + wakeGraceMs, +} from "./config.ts"; + +describe("resolveConfig", () => { + it("119 — an empty env yields the documented defaults", () => { + const config = resolveConfig({}); + // The kill switch defaults ON: the fiber exists, and the thing that stops loops firing + // is `global.enabled` in the durable store, which defaults OFF. + assert.strictEqual(config.enabled, true); + assert.strictEqual(config.pollMs, 60_000); + assert.strictEqual(config.idleMs, 15 * 60_000); + assert.strictEqual(config.busyIdleMs, 45 * 60_000); + assert.strictEqual(config.productiveMs, 2 * 60_000); + assert.strictEqual(config.wakeGraceMinMs, 90_000); + assert.strictEqual(config.wakeGraceMaxMs, 15 * 60_000); + assert.strictEqual(config.wakeGraceFraction, 0.1); + }); + + it("119 — COIL_LOOP_* overrides are parsed and invalid values fall back", () => { + const config = resolveConfig({ + COIL_LOOP_ENABLED: "off", + COIL_LOOP_POLL_MS: "15000", + COIL_LOOP_IDLE_MS: "0", // invalid -> default + COIL_LOOP_BUSY_IDLE_MS: "-1", // invalid -> default + COIL_LOOP_PRODUCTIVE_MS: "30000", + COIL_LOOP_WAKE_GRACE_FRACTION: "0.25", + }); + assert.strictEqual(config.enabled, false); + assert.strictEqual(config.pollMs, 15_000); + assert.strictEqual(config.idleMs, 15 * 60_000); + assert.strictEqual(config.busyIdleMs, 45 * 60_000); + assert.strictEqual(config.productiveMs, 30_000); + assert.strictEqual(config.wakeGraceFraction, 0.25); + }); + + it("119 — a nonsense grace fraction falls back rather than disabling deference", () => { + assert.strictEqual( + resolveConfig({ COIL_LOOP_WAKE_GRACE_FRACTION: "9" }).wakeGraceFraction, + 0.1, + ); + assert.strictEqual( + resolveConfig({ COIL_LOOP_WAKE_GRACE_FRACTION: "0" }).wakeGraceFraction, + 0.1, + ); + assert.strictEqual( + resolveConfig({ COIL_LOOP_WAKE_GRACE_FRACTION: "nope" }).wakeGraceFraction, + 0.1, + ); + }); + + it("119 — the T3X_* names of the neighbouring features are not inherited", () => { + const config = resolveConfig({ T3X_LOOP_ENABLED: "false", T3X_AUTO_RESUME_POLL_MS: "1" }); + assert.strictEqual(config.enabled, true); + assert.strictEqual(config.pollMs, 60_000); + }); +}); + +describe("wakeGraceMs", () => { + const config = resolveConfig({}); + + it("119 — a one-shot wake gets the flat floor", () => { + assert.strictEqual(wakeGraceMs({ recurring: false, periodMs: 30 * 60_000 }, config), 90_000); + }); + + it("119 — a recurring wake's grace scales with its period", () => { + // 10% of 30 minutes is 3 minutes, comfortably above the floor and below the cap. + assert.strictEqual(wakeGraceMs({ recurring: true, periodMs: 30 * 60_000 }, config), 180_000); + }); + + it("119 — a short period never drops below the floor", () => { + // 10% of 5 minutes is 30s; a flat-90s reading is the FLOOR, not the whole rule. + assert.strictEqual(wakeGraceMs({ recurring: true, periodMs: 5 * 60_000 }, config), 90_000); + }); + + it("119 — a long period is capped, so a weekly wake is not deferred to for a day", () => { + assert.strictEqual( + wakeGraceMs({ recurring: true, periodMs: 7 * 24 * 3_600_000 }, config), + 15 * 60_000, + ); + }); + + it("119 — an unknown period falls back to the floor rather than deferring forever", () => { + assert.strictEqual(wakeGraceMs({ recurring: true, periodMs: null }, config), 90_000); + assert.strictEqual(wakeGraceMs({ recurring: true, periodMs: 0 }, config), 90_000); + }); +}); + +describe("resolveLoopRoots", () => { + it("120 — the worktree comes first, because that is the agent's real cwd", () => { + assert.deepStrictEqual( + [...resolveLoopRoots({ worktreePath: "/wt", workspaceRoot: "/repo" })], + ["/wt", "/repo"], + ); + }); + + it("120 — a thread with no worktree falls back to the workspace root", () => { + assert.deepStrictEqual( + [...resolveLoopRoots({ worktreePath: null, workspaceRoot: "/repo" })], + ["/repo"], + ); + }); + + it("120 — identical roots are not stat-ed twice", () => { + assert.deepStrictEqual( + [...resolveLoopRoots({ worktreePath: "/repo", workspaceRoot: "/repo" })], + ["/repo"], + ); + }); +}); + +const baseInput = (overrides: Partial = {}): CheckInPromptInput => ({ + worktreePath: null, + workspaceRoot: null, + overridePrompt: null, + checkInNumber: 2, + maxCheckIns: 6, + deadlineAtMs: Date.parse("2026-09-02T07:00:00.000Z"), + nowMs: Date.parse("2026-09-02T01:00:00.000Z"), + goal: null, + bankedAnswers: [], + ...overrides, +}); + +const compose = (input: CheckInPromptInput) => + composeCheckInPrompt(input).pipe(Effect.provide(NodeServices.layer), Effect.runPromise); + +const withRoot = ( + f: (root: string) => Effect.Effect, +) => + Effect.gen(function* () { + const fs = yield* FileSystem.FileSystem; + const root = yield* fs.makeTempDirectoryScoped({ prefix: "coil-loop-prompt-" }); + return yield* f(root); + }).pipe(Effect.scoped, Effect.orDie, Effect.provide(NodeServices.layer), Effect.runPromise); + +const writePromptFile = (root: string, contents: string) => + Effect.gen(function* () { + const fs = yield* FileSystem.FileSystem; + const promptPath = NodePath.join(root, LOOP_PROMPT_RELATIVE_PATH); + yield* fs.makeDirectory(NodePath.dirname(promptPath), { recursive: true }); + yield* fs.writeFileString(promptPath, contents); + }); + +describe("composeCheckInPrompt", () => { + it("119 — a per-thread override wins over everything", async () => { + const prompt = await withRoot((root) => + Effect.gen(function* () { + yield* writePromptFile(root, "from the project file"); + return yield* composeCheckInPrompt( + baseInput({ workspaceRoot: root, overridePrompt: " from the override " }), + ); + }), + ); + assert.strictEqual(prompt.source, "override"); + assert.ok(prompt.text.includes("from the override")); + assert.ok(!prompt.text.includes("from the project file")); + }); + + it("119 — the project file wins over the built-in", async () => { + const prompt = await withRoot((root) => + Effect.gen(function* () { + yield* writePromptFile(root, "read the latest iter log and continue\n"); + return yield* composeCheckInPrompt(baseInput({ workspaceRoot: root })); + }), + ); + assert.strictEqual(prompt.source, "project-file"); + assert.ok(prompt.text.includes("read the latest iter log and continue")); + }); + + it("119 — the built-in is used when there is neither", async () => { + const prompt = await compose(baseInput()); + assert.strictEqual(prompt.source, "built-in"); + }); + + it("119 — the worktree's prompt file wins over the workspace root's", async () => { + const prompt = await withRoot((root) => + Effect.gen(function* () { + const fs = yield* FileSystem.FileSystem; + const worktree = NodePath.join(root, "wt"); + const workspace = NodePath.join(root, "repo"); + yield* fs.makeDirectory(worktree, { recursive: true }); + yield* fs.makeDirectory(workspace, { recursive: true }); + yield* writePromptFile(worktree, "worktree body"); + yield* writePromptFile(workspace, "workspace body"); + return yield* composeCheckInPrompt( + baseInput({ worktreePath: worktree, workspaceRoot: workspace }), + ); + }), + ); + assert.ok(prompt.text.includes("worktree body")); + assert.ok(!prompt.text.includes("workspace body")); + }); + + it("120 — the done-file path is absolute and rooted at the worktree", async () => { + const prompt = await compose( + baseInput({ worktreePath: "/tmp/wt", workspaceRoot: "/tmp/repo" }), + ); + // The supervisor stats `worktreePath ?? workspaceRoot` first, so telling the agent the + // workspace path would have it write the sentinel where nothing looks for it. + assert.strictEqual(prompt.doneFilePath, NodePath.join("/tmp/wt", LOOP_DONE_RELATIVE_PATH)); + assert.ok(NodePath.isAbsolute(prompt.doneFilePath)); + assert.ok(prompt.text.includes(prompt.doneFilePath)); + }); + + it("120 — a thread with no worktree gets the workspace root", async () => { + const prompt = await compose(baseInput({ worktreePath: null, workspaceRoot: "/tmp/repo" })); + assert.strictEqual(prompt.doneFilePath, NodePath.join("/tmp/repo", LOOP_DONE_RELATIVE_PATH)); + }); + + it("121 — the check-in number and budget are interpolated at every position", async () => { + for (let n = 1; n <= 6; n++) { + const prompt = await compose(baseInput({ checkInNumber: n, maxCheckIns: 6 })); + assert.ok( + prompt.text.includes(`check-in ${n} of 6`), + `check-in ${n} of 6 missing from: ${prompt.text.slice(0, 60)}`, + ); + } + }); + + it("121 — the deadline is stated, with the time remaining", async () => { + const prompt = await compose(baseInput()); + assert.ok(prompt.text.includes("2026-09-02T07:00:00.000Z")); + assert.ok(prompt.text.includes("6h 0m left")); + }); + + it("121 — the do-not-restart instruction is restated in full every time", async () => { + // A six-check-in overnight run compacts; a contract taught once is gone by check-in four. + for (const input of [ + baseInput(), + baseInput({ overridePrompt: "custom" }), + baseInput({ goal: "land the sync" }), + ]) { + const prompt = await compose(input); + assert.ok(prompt.text.includes("Do not restart from the top")); + } + }); + + it("122 — answered-but-undelivered blockers are included and reported for marking", async () => { + const prompt = await compose( + baseInput({ + bankedAnswers: [ + { id: "b-1", question: "Migration or shim?", answer: "shim" }, + { id: "b-2", question: "Ship tonight?", answer: "hold" }, + ], + }), + ); + assert.ok(prompt.text.includes("Migration or shim?")); + assert.ok(prompt.text.includes("shim")); + assert.ok(prompt.text.includes("hold")); + assert.deepStrictEqual([...prompt.deliveredBlockerIds], ["b-1", "b-2"]); + }); + + it("123 — only the blockers actually included are reported, so a later answer is not lost", async () => { + // The caller flips `deliveredToAgent` for exactly these ids AFTER composing. A blocker + // answered while the prompt was being built is simply not in the list, so it is carried + // by the next check-in rather than being marked delivered and dropped. + const prompt = await compose( + baseInput({ + bankedAnswers: [ + { id: "b-1", question: "Migration or shim?", answer: "shim" }, + { id: "b-blank", question: "Not answered yet", answer: " " }, + ], + }), + ); + assert.deepStrictEqual([...prompt.deliveredBlockerIds], ["b-1"]); + assert.ok(!prompt.text.includes("Not answered yet")); + }); + + it("122 — no banked answers means no answers section and nothing to mark", async () => { + const prompt = await compose(baseInput()); + assert.ok(!prompt.text.includes("Answers to questions you raised")); + assert.deepStrictEqual([...prompt.deliveredBlockerIds], []); + }); + + it("124 — the prompt never begins with `/`, whatever the body is", async () => { + const prompts = await Promise.all([ + compose(baseInput()), + compose(baseInput({ overridePrompt: "/compact and keep going" })), + compose(baseInput({ goal: "/tmp is full" })), + ]); + for (const prompt of prompts) { + assert.ok(!prompt.text.startsWith("/"), `would be read as a slash command: ${prompt.text}`); + } + + const fromFile = await withRoot((root) => + Effect.gen(function* () { + yield* writePromptFile(root, "/clear\nthen continue"); + return yield* composeCheckInPrompt(baseInput({ workspaceRoot: root })); + }), + ); + assert.ok(!fromFile.text.startsWith("/")); + }); + + it("125 — the deference line is present verbatim in every resolution path", async () => { + const prompts = await Promise.all([ + compose(baseInput()), + compose(baseInput({ overridePrompt: "custom body" })), + compose(baseInput({ bankedAnswers: [{ id: "b-1", question: "q", answer: "a" }] })), + ]); + for (const prompt of prompts) { + // Composing with the user's own self-pacing skill rather than claiming the schedule is + // the whole reason the reactor can defer at all. + assert.ok(prompt.text.includes(DEFERENCE_LINE), prompt.text); + } + + const fromFile = await withRoot((root) => + Effect.gen(function* () { + yield* writePromptFile(root, "project body"); + return yield* composeCheckInPrompt(baseInput({ workspaceRoot: root })); + }), + ); + assert.ok(fromFile.text.includes(DEFERENCE_LINE)); + }); + + it("126 — an empty or whitespace prompt file falls through to the built-in", async () => { + for (const contents of ["", " \n\t\n"]) { + const prompt = await withRoot((root) => + Effect.gen(function* () { + yield* writePromptFile(root, contents); + return yield* composeCheckInPrompt(baseInput({ workspaceRoot: root })); + }), + ); + assert.strictEqual(prompt.source, "built-in"); + assert.ok(prompt.text.includes(DEFERENCE_LINE)); + } + }); + + it("126 — an empty worktree prompt file falls through to the workspace root's", async () => { + const prompt = await withRoot((root) => + Effect.gen(function* () { + const fs = yield* FileSystem.FileSystem; + const worktree = NodePath.join(root, "wt"); + const workspace = NodePath.join(root, "repo"); + yield* fs.makeDirectory(worktree, { recursive: true }); + yield* fs.makeDirectory(workspace, { recursive: true }); + yield* writePromptFile(worktree, "\n\n"); + yield* writePromptFile(workspace, "workspace body"); + return yield* composeCheckInPrompt( + baseInput({ worktreePath: worktree, workspaceRoot: workspace }), + ); + }), + ); + assert.strictEqual(prompt.source, "project-file"); + assert.ok(prompt.text.includes("workspace body")); + }); + + it("127 — an unreadable prompt file falls through rather than failing the check-in", async () => { + const prompt = await withRoot((root) => + Effect.gen(function* () { + const fs = yield* FileSystem.FileSystem; + // A directory at the prompt path exists but cannot be read as a file (EISDIR). + yield* fs.makeDirectory(NodePath.join(root, LOOP_PROMPT_RELATIVE_PATH), { + recursive: true, + }); + return yield* composeCheckInPrompt(baseInput({ workspaceRoot: root })); + }), + ); + assert.strictEqual(prompt.source, "built-in"); + assert.ok(prompt.text.includes(DEFERENCE_LINE)); + }); + + it("121 — the goal is carried when there is one, and omitted when there is not", async () => { + const withGoal = await compose(baseInput({ goal: " land the sync " })); + assert.ok(withGoal.text.includes("land the sync")); + const withoutGoal = await compose(baseInput({ goal: " " })); + assert.ok(!withoutGoal.text.includes("The goal for this run")); + }); +}); diff --git a/apps/server/src/coil/loop/config.ts b/apps/server/src/coil/loop/config.ts new file mode 100644 index 000000000000..93f43bb75dee --- /dev/null +++ b/apps/server/src/coil/loop/config.ts @@ -0,0 +1,291 @@ +// @effect-diagnostics globalDate:off - formats an explicit deadline timestamp; this reads no ambient clock. +/** + * Loop configuration and check-in prompt composition. + * + * Config is read from env with safe defaults so the feature works with zero setup, and + * `resolveConfig` is pure (env passed in) for testability. The env prefix is `COIL_LOOP_*`: + * the neighbouring fork features use `T3X_*`, which are legacy names kept only to avoid + * orphaning state on machines that already set them. + * + * `COIL_LOOP_ENABLED` is the deployment kill switch and the ONLY condition under which the + * supervisor fiber does not exist. It is not the user-facing toggle — that is + * `global.enabled` in the durable store, which is a guard rather than a lifecycle: flipping + * it off stands every loop down without disarming or stopping anything. + * + * @module coil/loop/config + */ + +import * as Effect from "effect/Effect"; +import * as FileSystem from "effect/FileSystem"; +import * as Path from "effect/Path"; + +export interface LoopConfig { + /** `COIL_LOOP_ENABLED`. False ⇒ no fiber is forked at all. */ + readonly enabled: boolean; + readonly pollMs: number; + /** + * Staleness threshold fallbacks. + * + * The per-thread values on the record (seeded from `global.default*` at arm time) are what + * the trigger reads; these are the deployment-level floor a record falls back to. + */ + readonly idleMs: number; + readonly busyIdleMs: number; + /** Movement below this between two check-ins counts as unproductive (a strike). */ + readonly productiveMs: number; + /** The floor, and the whole grace for a one-shot wake. */ + readonly wakeGraceMinMs: number; + /** The cap on a recurring wake's derived grace. */ + readonly wakeGraceMaxMs: number; + /** Fraction of a recurring wake's period allowed as lateness before it counts as lost. */ + readonly wakeGraceFraction: number; +} + +const DEFAULTS: LoopConfig = { + enabled: true, + pollMs: 60_000, + idleMs: 15 * 60_000, + busyIdleMs: 45 * 60_000, + productiveMs: 2 * 60_000, + wakeGraceMinMs: 90_000, + wakeGraceMaxMs: 15 * 60_000, + wakeGraceFraction: 0.1, +}; + +function parseBool(value: string | undefined, fallback: boolean): boolean { + if (value === undefined) return fallback; + const normalized = value.trim().toLowerCase(); + if (normalized === "false" || normalized === "0" || normalized === "no" || normalized === "off") { + return false; + } + if (normalized === "true" || normalized === "1" || normalized === "yes" || normalized === "on") { + return true; + } + return fallback; +} + +function parsePositiveInt(value: string | undefined, fallback: number): number { + if (value === undefined) return fallback; + const parsed = Number.parseInt(value.trim(), 10); + return Number.isFinite(parsed) && parsed > 0 ? parsed : fallback; +} + +function parseFraction(value: string | undefined, fallback: number): number { + if (value === undefined) return fallback; + const parsed = Number.parseFloat(value.trim()); + return Number.isFinite(parsed) && parsed > 0 && parsed <= 1 ? parsed : fallback; +} + +export function resolveConfig(env: Record = process.env): LoopConfig { + return { + enabled: parseBool(env.COIL_LOOP_ENABLED, DEFAULTS.enabled), + pollMs: parsePositiveInt(env.COIL_LOOP_POLL_MS, DEFAULTS.pollMs), + idleMs: parsePositiveInt(env.COIL_LOOP_IDLE_MS, DEFAULTS.idleMs), + busyIdleMs: parsePositiveInt(env.COIL_LOOP_BUSY_IDLE_MS, DEFAULTS.busyIdleMs), + productiveMs: parsePositiveInt(env.COIL_LOOP_PRODUCTIVE_MS, DEFAULTS.productiveMs), + wakeGraceMinMs: parsePositiveInt(env.COIL_LOOP_WAKE_GRACE_MIN_MS, DEFAULTS.wakeGraceMinMs), + wakeGraceMaxMs: parsePositiveInt(env.COIL_LOOP_WAKE_GRACE_MAX_MS, DEFAULTS.wakeGraceMaxMs), + wakeGraceFraction: parseFraction(env.COIL_LOOP_WAKE_GRACE_FRACTION, DEFAULTS.wakeGraceFraction), + }; +} + +/** + * How late a recorded wake may land before it counts as lost. + * + * Derived, not constant. The scheduler's own text is *"recurring tasks fire up to 10% of + * their period late (max 15 min); one-shot tasks landing on :00 or :30 fire up to 90 s + * early"*. The 90s is the **early** half and can never make a healthy wake look lost; the + * half that matters is **late**, and it scales with the period. A flat 90s would fire + * `wake_lost` — the strongest trigger in the design — on a merely jittered thread. + */ +export function wakeGraceMs( + entry: { readonly recurring: boolean; readonly periodMs: number | null }, + config: Pick, +): number { + const { periodMs } = entry; + if (!entry.recurring || periodMs === null || !Number.isFinite(periodMs) || periodMs <= 0) { + return config.wakeGraceMinMs; + } + return Math.max( + config.wakeGraceMinMs, + Math.min(periodMs * config.wakeGraceFraction, config.wakeGraceMaxMs), + ); +} + +/** The agent writes this to end a run early. T3 only ever *stats* it — never writes it. */ +export const LOOP_DONE_RELATIVE_PATH = ".coil/loop-done"; +/** A project-committed replacement for the built-in check-in body. */ +export const LOOP_PROMPT_RELATIVE_PATH = ".coil/loop-prompt.md"; + +/** + * The roots to stat, **worktree first**. + * + * `resolveThreadWorkspaceCwd` returns the worktree first, so that is the agent's real cwd. + * `autoResume/config.ts` resolves from `workspaceRoot` only, and copying it would put the + * supervisor and the agent in different directories on every worktree-backed thread — the + * done-file would be written where nothing looks for it. + */ +export function resolveLoopRoots(input: { + readonly worktreePath: string | null; + readonly workspaceRoot: string | null; +}): ReadonlyArray { + const roots = [input.worktreePath, input.workspaceRoot].filter( + (root): root is string => typeof root === "string" && root.trim().length > 0, + ); + return [...new Set(roots)]; +} + +/** Verbatim, in every resolution path. See `composeCheckInPrompt`. */ +export const DEFERENCE_LINE = + "T3 checked in because no wake of yours landed. Keep scheduling your own wake-ups as normal — " + + "T3 stands by while one is pending inside this run's deadline, covers any that are lost, and " + + "enforces the budget and deadline."; + +const BUILT_IN_BODY = + "Continue the work already in progress on this thread. Read your most recent output first so " + + "you resume mid-task rather than re-deriving what is already done."; + +export interface BankedAnswer { + readonly id: string; + readonly question: string; + readonly answer: string; +} + +export interface CheckInPromptInput { + readonly worktreePath: string | null; + readonly workspaceRoot: string | null; + /** The per-thread override from the durable record. */ + readonly overridePrompt: string | null; + /** 1-based. */ + readonly checkInNumber: number; + readonly maxCheckIns: number; + readonly deadlineAtMs: number; + readonly nowMs: number; + readonly goal: string | null; + /** Answered-but-undelivered blockers, from `store.listUndeliveredAnswers`. */ + readonly bankedAnswers: ReadonlyArray; +} + +export interface CheckInPrompt { + readonly text: string; + readonly source: "override" | "project-file" | "built-in"; + /** Absolute wherever a root is known — the agent cannot act on a relative instruction. */ + readonly doneFilePath: string; + /** + * Exactly the blockers whose answers this text carries. + * + * The caller flips `deliveredToAgent` for these ids AFTER the dispatch, which is what stops + * an answer that landed mid-composition being marked delivered and silently lost. + */ + readonly deliveredBlockerIds: ReadonlyArray; +} + +function formatRemaining(deadlineAtMs: number, nowMs: number): string { + const remainingMs = Math.max(0, deadlineAtMs - nowMs); + const hours = Math.floor(remainingMs / 3_600_000); + const minutes = Math.floor((remainingMs % 3_600_000) / 60_000); + return `${hours}h ${minutes}m`; +} + +function formatDeadline(deadlineAtMs: number): string { + const date = new Date(deadlineAtMs); + return Number.isNaN(date.getTime()) ? "unknown" : date.toISOString(); +} + +/** + * Reads the project check-in body, worktree first. Empty, whitespace-only, missing or + * unreadable all fall through to the next root and finally to the built-in; only an + * unreadable *existing* file logs, so an absent file is not noise. + */ +function readProjectBody( + roots: ReadonlyArray, +): Effect.Effect { + return Effect.gen(function* () { + const path = yield* Path.Path; + const fs = yield* FileSystem.FileSystem; + for (const root of roots) { + const promptPath = path.join(root, LOOP_PROMPT_RELATIVE_PATH); + const exists = yield* fs.exists(promptPath).pipe(Effect.orElseSucceed(() => false)); + if (!exists) continue; + const contents = yield* fs.readFileString(promptPath).pipe( + Effect.map((text): string | null => text), + Effect.orElseSucceed(() => null), + ); + if (contents === null) { + yield* Effect.logWarning("coil loop: check-in prompt file unreadable; using the built-in", { + promptPath, + }); + continue; + } + const trimmed = contents.trim(); + if (trimmed.length > 0) return trimmed; + } + return null; + }); +} + +/** + * Builds the check-in text. + * + * The contract is restated **in full every time**: a six-check-in overnight run compacts, and + * a contract taught once is gone by check-in four. Resolution order for the body is + * per-thread override → `/.coil/loop-prompt.md` → built-in, but the envelope — the + * check-in number and budget, the do-not-restart instruction, the absolute done-file path, + * the deadline, the banked answers and the deference line — is fork-owned and identical in + * all three, which is also what guarantees the text never begins with `/` and is therefore + * never read as a slash command. + */ +export function composeCheckInPrompt( + input: CheckInPromptInput, +): Effect.Effect { + return Effect.gen(function* () { + const path = yield* Path.Path; + const roots = resolveLoopRoots(input); + const doneFilePath = + roots.length > 0 ? path.join(roots[0]!, LOOP_DONE_RELATIVE_PATH) : LOOP_DONE_RELATIVE_PATH; + + const override = input.overridePrompt?.trim(); + let body = override && override.length > 0 ? override : null; + let source: CheckInPrompt["source"] = body === null ? "built-in" : "override"; + if (body === null) { + const projectBody = yield* readProjectBody(roots); + if (projectBody !== null) { + body = projectBody; + source = "project-file"; + } else { + body = BUILT_IN_BODY; + } + } + + const sections = [ + `Loop check-in ${input.checkInNumber} of ${input.maxCheckIns}.`, + body, + "Do not restart from the top. Pick up exactly where you left off and keep going.", + `This run ends at ${formatDeadline(input.deadlineAtMs)} (${formatRemaining(input.deadlineAtMs, input.nowMs)} left) or when the check-in budget runs out, whichever comes first.`, + `When the work is genuinely finished, write the file ${doneFilePath}. Its contents do not matter; T3 reads only its timestamp. That is how you end this run early.`, + ]; + + if (input.goal !== null && input.goal.trim().length > 0) { + sections.splice(1, 0, `The goal for this run: ${input.goal.trim()}`); + } + + const banked = input.bankedAnswers.filter((entry) => entry.answer.trim().length > 0); + if (banked.length > 0) { + sections.push( + [ + "Answers to questions you raised:", + ...banked.map((entry) => `- ${entry.question.trim()}\n ${entry.answer.trim()}`), + ].join("\n"), + ); + } + + sections.push(DEFERENCE_LINE); + + return { + text: sections.join("\n\n"), + source, + doneFilePath, + deliveredBlockerIds: banked.map((entry) => entry.id), + }; + }); +} diff --git a/apps/server/src/coil/loop/cron/parse.test.ts b/apps/server/src/coil/loop/cron/parse.test.ts new file mode 100644 index 000000000000..78c7a1b46b80 --- /dev/null +++ b/apps/server/src/coil/loop/cron/parse.test.ts @@ -0,0 +1,228 @@ +// @effect-diagnostics globalDate:off - the expectations are built from explicit calendar values. +import { describe, expect, it } from "vite-plus/test"; + +import { nextFireAtMs, periodMsAfter } from "./parse.ts"; + +const utc = (iso: string) => Date.parse(iso); +const NY = "America/New_York"; + +// Every case here is TESTS 70b: the SDK delivers no timestamp for a cron entry, so this +// parse is the whole of the deference risk. `null` always means "no deference from this +// entry" — never "defer forever" — so the rejection cases matter as much as the matches. +describe("nextFireAtMs — recurring forms (70b)", () => { + it("70b — a step expression yields the next match strictly after now", () => { + expect(nextFireAtMs("*/5 * * * *", utc("2026-03-01T10:02:30Z"), "UTC")).toBe( + utc("2026-03-01T10:05:00Z"), + ); + }); + + it("70b — at exactly a matching minute the NEXT match is returned, not now", () => { + expect(nextFireAtMs("*/5 * * * *", utc("2026-03-01T10:05:00.000Z"), "UTC")).toBe( + utc("2026-03-01T10:10:00Z"), + ); + }); + + it("70b — a daily expression rolls over to tomorrow once today's fire has passed", () => { + expect(nextFireAtMs("0 3 * * *", utc("2026-03-01T04:00:00Z"), "UTC")).toBe( + utc("2026-03-02T03:00:00Z"), + ); + }); + + it("70b — a day-of-week range skips the weekend", () => { + // 2026-03-06 is a Friday; the next weekday 09:00 is Monday the 9th. + expect(nextFireAtMs("0 9 * * 1-5", utc("2026-03-06T10:00:00Z"), "UTC")).toBe( + utc("2026-03-09T09:00:00Z"), + ); + }); + + it("70b — a comma list picks the earliest listed value", () => { + expect(nextFireAtMs("0 0,12 * * *", utc("2026-03-01T06:00:00Z"), "UTC")).toBe( + utc("2026-03-01T12:00:00Z"), + ); + }); + + it("70b — a stepped range expands to its members and no further", () => { + // 9-17/4 is 09:00, 13:00, 17:00 — 21:00 is outside the range. + expect(nextFireAtMs("0 9-17/4 * * *", utc("2026-03-01T13:30:00Z"), "UTC")).toBe( + utc("2026-03-01T17:00:00Z"), + ); + expect(nextFireAtMs("0 9-17/4 * * *", utc("2026-03-01T17:30:00Z"), "UTC")).toBe( + utc("2026-03-02T09:00:00Z"), + ); + }); + + it("70b — 7 is Sunday, the same day as 0", () => { + // 2026-03-01 is a Sunday. + expect(nextFireAtMs("0 6 * * 7", utc("2026-02-28T12:00:00Z"), "UTC")).toBe( + utc("2026-03-01T06:00:00Z"), + ); + }); + + it("70b — restricting BOTH day fields is a union, not an intersection", () => { + // "the 1st, and every Monday". 2026-03-02 is a Monday, 2026-04-01 a Wednesday. + expect(nextFireAtMs("0 9 1 * 1", utc("2026-03-01T12:00:00Z"), "UTC")).toBe( + utc("2026-03-02T09:00:00Z"), + ); + expect(nextFireAtMs("0 9 1 * 1", utc("2026-03-31T12:00:00Z"), "UTC")).toBe( + utc("2026-04-01T09:00:00Z"), + ); + }); +}); + +describe("nextFireAtMs — the one-shot form (70b)", () => { + it("70b — a one-shot's five fields resolve to exactly the instant they encode", () => { + // What `ScheduleWakeup` produces: every field concrete, day-of-week open. + expect(nextFireAtMs("35 2 15 9 *", utc("2026-09-15T00:10:00Z"), "UTC")).toBe( + utc("2026-09-15T02:35:00Z"), + ); + }); + + it("70b — a one-shot recorded seconds before it fires still resolves to that instant", () => { + expect(nextFireAtMs("35 2 15 9 *", utc("2026-09-15T02:34:59Z"), "UTC")).toBe( + utc("2026-09-15T02:35:00Z"), + ); + }); + + it("70b — a one-shot already in the past resolves a year out, which is why the value is computed at hook time", () => { + // Documented behaviour, not a bug: `crons.ts` computes and persists `nextFireAtMs` when + // the entry is observed. A past wake is detected from the PERSISTED value, and re-parsing + // one on a later tick would silently turn a lost wake into next year's commitment — well + // beyond any deadline, so guard 10b declines to defer to it either way. + expect(nextFireAtMs("35 2 15 9 *", utc("2026-09-15T03:00:00Z"), "UTC")).toBe( + utc("2027-09-15T02:35:00Z"), + ); + }); +}); + +describe("nextFireAtMs — outside the producer's grammar (70b)", () => { + const rejected = [ + ["four fields", "0 3 * *"], + ["six fields (a seconds column)", "0 0 3 * * *"], + ["a macro", "@daily"], + ["a month name", "0 3 * JAN *"], + ["a weekday name", "0 3 * * MON"], + ["a bare stepped value", "5/2 * * * *"], + ["a wrap-around range", "0 22-2 * * *"], + ["a minute out of range", "60 * * * *"], + ["a day-of-week out of range", "0 3 * * 8"], + ["a month out of range", "0 3 * 13 *"], + ["a zero step", "*/0 * * * *"], + ["an empty term", "0,, 3 * * *"], + ["a negative value", "-1 3 * * *"], + ["an empty expression", ""], + ["whitespace only", " "], + ["a last-Sunday extension", "0 3 * * 0L"], + ["a nearest-weekday extension", "0 3 15W * *"], + ] as const; + + for (const [label, schedule] of rejected) { + it(`70b — ${label} yields null, which means no deference rather than deferring forever`, () => { + expect(nextFireAtMs(schedule, utc("2026-03-01T00:00:00Z"), "UTC")).toBeNull(); + }); + } + + it("70b — an expression that can never match returns null instead of searching forever", () => { + expect(nextFireAtMs("0 0 30 2 *", utc("2026-03-01T00:00:00Z"), "UTC")).toBeNull(); + }); + + it("70b — an unknown timezone returns null and does not throw into the hook callback", () => { + expect(nextFireAtMs("0 3 * * *", utc("2026-03-01T00:00:00Z"), "Mars/Olympus_Mons")).toBeNull(); + }); + + it("70b — a nonsense `nowMs` returns null rather than throwing", () => { + expect(nextFireAtMs("0 3 * * *", Number.NaN, "UTC")).toBeNull(); + expect(nextFireAtMs("0 3 * * *", Number.POSITIVE_INFINITY, "UTC")).toBeNull(); + }); + + it("70b — hostile input returns null rather than throwing", () => { + expect(nextFireAtMs("*".repeat(10_000), utc("2026-03-01T00:00:00Z"), "UTC")).toBeNull(); + expect(nextFireAtMs("🕐 🕑 🕒 🕓 🕔", utc("2026-03-01T00:00:00Z"), "UTC")).toBeNull(); + }); +}); + +// A cron expression names a WALL CLOCK, so a correct parse keeps the local time fixed across +// a transition and lets the UTC offset move. These are the cases that separate a real parse +// from `now + period` arithmetic, which drifts by an hour twice a year and would fire +// `wake_lost` — the strongest trigger in the design — on a perfectly healthy thread. +describe("nextFireAtMs — daylight saving (70b)", () => { + it("70b — a daily wake keeps its local hour across spring forward", () => { + // 09:00 EST is 14:00Z on the Saturday; 09:00 EDT is 13:00Z on the Monday. + const saturday = nextFireAtMs("0 9 * * *", utc("2026-03-07T00:00:00Z"), NY); + expect(saturday).toBe(utc("2026-03-07T14:00:00Z")); + const monday = nextFireAtMs("0 9 * * *", utc("2026-03-08T20:00:00Z"), NY); + expect(monday).toBe(utc("2026-03-09T13:00:00Z")); + }); + + it("70b — a wake inside the spring-forward gap is skipped, not shifted", () => { + // On 2026-03-08 the clock jumps 01:59 -> 03:00, so 02:30 never happens that day; the + // next 02:30 is on the 9th (06:30Z, EDT). + expect(nextFireAtMs("30 2 * * *", utc("2026-03-07T12:00:00Z"), NY)).toBe( + utc("2026-03-09T06:30:00Z"), + ); + }); + + it("70b — a wake on the far side of the gap fires at the correct instant", () => { + expect(nextFireAtMs("0 3 * * *", utc("2026-03-08T05:00:00Z"), NY)).toBe( + utc("2026-03-08T07:00:00Z"), + ); + }); + + it("70b — an ambiguous wall clock on fall-back resolves to its LATER occurrence", () => { + // 2026-11-01 01:30 happens twice: 05:30Z (EDT) and 06:30Z (EST). Nothing in the payload + // says which one the binary meant, and the two errors are not symmetric: the wake grace is + // capped at fifteen minutes, so resolving early on an hour-wide ambiguity reports a + // perfectly healthy self-paced run as `wake_lost` and nudges it an hour before its own + // wake was due. Resolving late costs one extra hour of deference, one night a year. + expect(nextFireAtMs("30 1 * * *", utc("2026-10-31T12:00:00Z"), NY)).toBe( + utc("2026-11-01T06:30:00Z"), + ); + // And from between the two occurrences it is still the later one that is next. + expect(nextFireAtMs("30 1 * * *", utc("2026-11-01T06:00:00Z"), NY)).toBe( + utc("2026-11-01T06:30:00Z"), + ); + }); +}); + +describe("nextFireAtMs — calendar boundaries (70b)", () => { + it("70b — day 31 skips the months that do not have one", () => { + expect(nextFireAtMs("0 0 31 * *", utc("2026-01-31T12:00:00Z"), "UTC")).toBe( + utc("2026-03-31T00:00:00Z"), + ); + }); + + it("70b — 29 February resolves to the next leap year", () => { + expect(nextFireAtMs("0 12 29 2 *", utc("2026-03-01T00:00:00Z"), "UTC")).toBe( + utc("2028-02-29T12:00:00Z"), + ); + }); + + it("70b — the last minute of the year rolls into the next one", () => { + expect(nextFireAtMs("*/30 * * * *", utc("2026-12-31T23:45:00Z"), "UTC")).toBe( + utc("2027-01-01T00:00:00Z"), + ); + }); +}); + +describe("nextFireAtMs — the server's local zone (70b)", () => { + it("70b — omitting the timezone evaluates in local time, as the tool documents", () => { + // Built from the local calendar so the assertion holds under any machine TZ. + const now = new Date(2026, 5, 10, 14, 20, 0, 0).getTime(); + const expected = new Date(2026, 5, 10, 15, 0, 0, 0).getTime(); + expect(nextFireAtMs("0 * * * *", now, undefined)).toBe(expected); + }); +}); + +describe("periodMsAfter (70b)", () => { + it("70b — the period of a recurring wake is the gap between two successive fires", () => { + expect(periodMsAfter("*/15 * * * *", utc("2026-03-01T00:00:00Z"), "UTC")).toBe(15 * 60_000); + expect(periodMsAfter("0 3 * * *", utc("2026-03-01T00:00:00Z"), "UTC")).toBe(24 * 3_600_000); + }); + + it("70b — a daily period across spring forward is 23 hours, so the derived grace shrinks with it", () => { + expect(periodMsAfter("0 9 * * *", utc("2026-03-07T00:00:00Z"), NY)).toBe(23 * 3_600_000); + }); + + it("70b — an unparseable schedule has no period", () => { + expect(periodMsAfter("@daily", utc("2026-03-01T00:00:00Z"), "UTC")).toBeNull(); + }); +}); diff --git a/apps/server/src/coil/loop/cron/parse.ts b/apps/server/src/coil/loop/cron/parse.ts new file mode 100644 index 000000000000..010aef9f7ee2 --- /dev/null +++ b/apps/server/src/coil/loop/cron/parse.ts @@ -0,0 +1,368 @@ +// @effect-diagnostics globalDate:off - calendar arithmetic on an explicit `nowMs`; this reads no ambient clock. +/** + * Fork-owned 5-field cron parser. + * + * There is no cron parser anywhere in this repo and this must not add a dependency: a + * general parser handles a grammar the producer never emits. The producer's grammar is + * documented and narrow — *"Standard 5-field cron expression in local time: `M H DoM Mon + * DoW`"*, with the terms `*`, `N` and `A-B`, either of the first and last optionally + * followed by a `/S` step, plus comma-lists of those. **No seconds field, no `@daily` + * macros, no `JAN`/`MON` names.** Anything outside that yields `null`, which + * means **no deference from that entry** — an unparseable schedule must never stand + * supervision down, and it must never mean "defer forever". + * + * The one-shot form (`recurring: false`) is the same grammar: the binary encodes the single + * instant into the same five fields, all concrete, so "the next match strictly after now" is + * exactly that instant. This is why one call covers both forms. It also means the value is + * only correct if computed *when the entry is observed* — a one-shot re-parsed after it has + * fired resolves to next year's occurrence, which is why `crons.ts` computes `nextFireAtMs` + * at hook time and persists it rather than re-deriving it on each tick. + * + * Everything is evaluated in wall-clock time (the tool says "local time"), which is what + * makes it DST-correct rather than DST-ignorant: candidates are generated as calendar + * minutes and only then converted to an instant, so a `0 3 * * *` job is 03:00 local on both + * sides of a transition. A wall-clock minute that does not exist (the spring-forward gap) is + * skipped; one that occurs twice (the fall-back repeat) resolves to its **later** occurrence, + * because deferring an hour too long is cheap and deferring an hour too little reports a + * healthy run as a lost wake — see `timestampFor`. + * + * @module coil/loop/cron/parse + */ + +/** > 4 years, so a search that can only match on Feb 29 still terminates on a real date. */ +const MAX_SEARCH_DAYS = 1500; + +const INTEGER = /^\d+$/; + +interface WallClock { + readonly year: number; + readonly month: number; // 1-12 + readonly day: number; // 1-31 + readonly hour: number; // 0-23 + readonly minute: number; // 0-59 +} + +interface ParsedCron { + readonly minutes: ReadonlyArray; + readonly hours: ReadonlyArray; + readonly daysOfMonth: ReadonlyArray; + readonly months: ReadonlyArray; + readonly daysOfWeek: ReadonlyArray; + /** + * Whether the day-of-month / day-of-week fields were written as anything other than `*`. + * + * Cron's one genuinely surprising rule: when *both* day fields are restricted a day + * matches if *either* does (a union, not an intersection), so `0 9 1 * 1` is "the 1st and + * every Monday", not "Mondays that fall on the 1st". + */ + readonly dayOfMonthRestricted: boolean; + readonly dayOfWeekRestricted: boolean; +} + +/** + * Expands one comma-separated term of a field. + * + * A step with no range (`5/2`) is deliberately rejected: it is Vixie cron, not the + * producer's documented grammar, and guessing at it would defer supervision on an + * expression we cannot claim to understand. + */ +function parseTerm(term: string, min: number, max: number): ReadonlyArray | null { + const slash = term.indexOf("/"); + const rangeText = slash === -1 ? term : term.slice(0, slash); + + let step = 1; + if (slash !== -1) { + const stepText = term.slice(slash + 1); + if (!INTEGER.test(stepText)) return null; + step = Number(stepText); + if (step < 1 || step > max - min + 1) return null; + } + + let low: number; + let high: number; + if (rangeText === "*") { + low = min; + high = max; + } else { + const dash = rangeText.indexOf("-"); + if (dash === -1) { + if (slash !== -1) return null; + if (!INTEGER.test(rangeText)) return null; + low = Number(rangeText); + high = low; + } else { + const lowText = rangeText.slice(0, dash); + const highText = rangeText.slice(dash + 1); + if (!INTEGER.test(lowText) || !INTEGER.test(highText)) return null; + low = Number(lowText); + high = Number(highText); + // Wrap-around ranges (`22-2`) are not in the producer's grammar. + if (low > high) return null; + } + } + + if (low < min || high > max) return null; + + const values: Array = []; + for (let value = low; value <= high; value += step) values.push(value); + return values; +} + +function parseField(raw: string, min: number, max: number): ReadonlyArray | null { + const values = new Set(); + for (const term of raw.split(",")) { + const expanded = parseTerm(term, min, max); + if (expanded === null) return null; + for (const value of expanded) values.add(value); + } + if (values.size === 0) return null; + return [...values].sort((a, b) => a - b); +} + +function parseCronExpression(schedule: string): ParsedCron | null { + if (typeof schedule !== "string") return null; + const fields = schedule.trim().split(/\s+/); + if (fields.length !== 5) return null; + const [minuteField, hourField, dayOfMonthField, monthField, dayOfWeekField] = fields as [ + string, + string, + string, + string, + string, + ]; + + const minutes = parseField(minuteField, 0, 59); + const hours = parseField(hourField, 0, 23); + const daysOfMonth = parseField(dayOfMonthField, 1, 31); + const months = parseField(monthField, 1, 12); + // 7 is a second spelling of Sunday. + const rawDaysOfWeek = parseField(dayOfWeekField, 0, 7); + if (!minutes || !hours || !daysOfMonth || !months || !rawDaysOfWeek) return null; + + return { + minutes, + hours, + daysOfMonth, + months, + daysOfWeek: [...new Set(rawDaysOfWeek.map((d) => d % 7))].sort((a, b) => a - b), + dayOfMonthRestricted: dayOfMonthField !== "*", + dayOfWeekRestricted: dayOfWeekField !== "*", + }; +} + +const formatterCache = new Map(); + +function formatterFor(timeZone: string): Intl.DateTimeFormat { + const cached = formatterCache.get(timeZone); + if (cached) return cached; + // Throws RangeError on an unknown zone; `nextFireAtMs` turns that into `null`. + const formatter = new Intl.DateTimeFormat("en-US", { + timeZone, + hourCycle: "h23", + year: "numeric", + month: "2-digit", + day: "2-digit", + hour: "2-digit", + minute: "2-digit", + second: "2-digit", + }); + formatterCache.set(timeZone, formatter); + return formatter; +} + +/** Wall-clock parts of an instant, plus seconds (needed to invert the zone offset). */ +function partsAt( + timestampMs: number, + timeZone: string | undefined, +): (WallClock & { readonly second: number }) | null { + if (!Number.isFinite(timestampMs)) return null; + const date = new Date(timestampMs); + if (Number.isNaN(date.getTime())) return null; + + if (timeZone === undefined) { + return { + year: date.getFullYear(), + month: date.getMonth() + 1, + day: date.getDate(), + hour: date.getHours(), + minute: date.getMinutes(), + second: date.getSeconds(), + }; + } + + const parts = formatterFor(timeZone).formatToParts(date); + const read = (type: Intl.DateTimeFormatPartTypes): number => { + const part = parts.find((candidate) => candidate.type === type); + return part === undefined ? Number.NaN : Number(part.value); + }; + const wall = { + year: read("year"), + month: read("month"), + day: read("day"), + hour: read("hour"), + minute: read("minute"), + second: read("second"), + }; + return Object.values(wall).some((value) => Number.isNaN(value)) ? null : wall; +} + +/** The zone's UTC offset at an instant, derived by reading the wall clock back. */ +function offsetMsAt(timestampMs: number, timeZone: string | undefined): number | null { + if (timeZone === undefined) { + // `getTimezoneOffset` is minutes WEST of UTC, i.e. the negation of the offset. + const minutesWest = new Date(timestampMs).getTimezoneOffset(); + return Number.isFinite(minutesWest) ? -minutesWest * 60_000 : null; + } + return zoneOffsetMsAt(timestampMs, timeZone); +} + +function zoneOffsetMsAt(timestampMs: number, timeZone: string): number | null { + const parts = partsAt(timestampMs, timeZone); + if (parts === null) return null; + const asUtc = Date.UTC( + parts.year, + parts.month - 1, + parts.day, + parts.hour, + parts.minute, + parts.second, + ); + return asUtc - Math.floor(timestampMs / 1000) * 1000; +} + +/** A transition happens at most once a day, so probes a day out bracket it. */ +const OFFSET_PROBE_MS = 86_400_000; + +/** + * The instant a wall-clock minute names, or `null` when it names none. + * + * `null` is the spring-forward gap: 02:30 simply does not happen on that date, and a job + * scheduled for it does not run that day. + * + * ## An ambiguous minute resolves to the LATER instant + * + * On a fall-back day 01:30 happens twice, an hour apart, and nothing in the payload says + * which one the binary meant. This is not a "pick a convention" choice, because the two + * errors are not symmetric: T3 uses this value only to decide how long to keep deferring to + * a wake it did not schedule, and the wake grace is capped at fifteen minutes. Resolving + * early on an hour-wide ambiguity therefore reports a healthy self-paced run as `wake_lost` + * and nudges it, an hour before its own wake was even due. Resolving late costs at most one + * extra hour of deference on one night a year, and the deadline still bounds it. + * + * The offset is probed on both sides of the naive instant rather than re-derived once: away + * from a transition the probes agree, and across one they are exactly the two candidate + * offsets, so the ambiguous case falls out of taking the later surviving candidate. + */ +function timestampFor(wall: WallClock, timeZone: string | undefined): number | null { + const asUtc = Date.UTC(wall.year, wall.month - 1, wall.day, wall.hour, wall.minute, 0, 0); + if (Number.isNaN(asUtc)) return null; + + let latest: number | null = null; + for (const probe of [asUtc - OFFSET_PROBE_MS, asUtc, asUtc + OFFSET_PROBE_MS]) { + const offset = offsetMsAt(probe, timeZone); + if (offset === null) continue; + const candidate = asUtc - offset; + if (latest !== null && candidate <= latest) continue; + // Reading the wall clock back is what rejects the spring-forward gap: a minute that does + // not exist resolves to some other minute, and no candidate survives the compare. + const check = partsAt(candidate, timeZone); + if (check === null) continue; + if ( + check.year === wall.year && + check.month === wall.month && + check.day === wall.day && + check.hour === wall.hour && + check.minute === wall.minute + ) { + latest = candidate; + } + } + return latest; +} + +function dayOfWeek(year: number, month: number, day: number): number { + return new Date(Date.UTC(year, month - 1, day)).getUTCDay(); +} + +function nextCalendarDay(year: number, month: number, day: number): WallClock { + const next = new Date(Date.UTC(year, month - 1, day + 1)); + return { + year: next.getUTCFullYear(), + month: next.getUTCMonth() + 1, + day: next.getUTCDate(), + hour: 0, + minute: 0, + }; +} + +function dayMatches(parsed: ParsedCron, year: number, month: number, day: number): boolean { + if (!parsed.months.includes(month)) return false; + const dayOfMonthHit = parsed.daysOfMonth.includes(day); + const dayOfWeekHit = parsed.daysOfWeek.includes(dayOfWeek(year, month, day)); + if (parsed.dayOfMonthRestricted && parsed.dayOfWeekRestricted) { + return dayOfMonthHit || dayOfWeekHit; + } + if (parsed.dayOfMonthRestricted) return dayOfMonthHit; + if (parsed.dayOfWeekRestricted) return dayOfWeekHit; + return true; +} + +/** + * The next instant `schedule` fires, strictly after `nowMs`, or `null`. + * + * `null` covers every failure — an expression outside the producer's grammar, an unknown + * `timeZone`, a nonsensical `nowMs`, or no match inside the search horizon. It never + * throws, because this runs inside a provider hook callback where a thrown error would cost + * the user a turn. + * + * `timeZone` is an IANA zone name; omitted, the server's local zone is used, which is what + * the producer means by "local time" (the binary and the server share a process group). + * + * A recurring entry's period — needed for the derived wake grace — is the gap between two + * successive fires: `nextFireAtMs(s, nextFireAtMs(s, now)!)`. See `periodMsAfter`. + */ +export function nextFireAtMs(schedule: string, nowMs: number, timeZone?: string): number | null { + try { + if (!Number.isFinite(nowMs)) return null; + const parsed = parseCronExpression(schedule); + if (parsed === null) return null; + + const start = partsAt(nowMs, timeZone); + if (start === null) return null; + + let cursor: WallClock = start; + for (let dayIndex = 0; dayIndex < MAX_SEARCH_DAYS; dayIndex++) { + if (dayMatches(parsed, cursor.year, cursor.month, cursor.day)) { + const isStartDay = dayIndex === 0; + for (const hour of parsed.hours) { + if (isStartDay && hour < start.hour) continue; + for (const minute of parsed.minutes) { + if (isStartDay && hour === start.hour && minute < start.minute) continue; + const timestampMs = timestampFor( + { year: cursor.year, month: cursor.month, day: cursor.day, hour, minute }, + timeZone, + ); + // `> nowMs`, not `>=`: the match must be strictly in the future, and on a + // fall-back day the first candidate can resolve to an instant already past. + if (timestampMs !== null && timestampMs > nowMs) return timestampMs; + } + } + } + cursor = nextCalendarDay(cursor.year, cursor.month, cursor.day); + } + return null; + } catch { + return null; + } +} + +/** + * The gap to the fire after `fromMs`'s next one — the `periodMs` the derived wake grace + * scales with. `null` whenever either fire is unknown. + */ +export function periodMsAfter(schedule: string, fromMs: number, timeZone?: string): number | null { + const first = nextFireAtMs(schedule, fromMs, timeZone); + if (first === null) return null; + const second = nextFireAtMs(schedule, first, timeZone); + return second === null ? null : second - first; +} diff --git a/apps/server/src/coil/loop/crons.test.ts b/apps/server/src/coil/loop/crons.test.ts new file mode 100644 index 000000000000..e72704bb2b17 --- /dev/null +++ b/apps/server/src/coil/loop/crons.test.ts @@ -0,0 +1,561 @@ +// @effect-diagnostics nodeBuiltinImport:off +// @effect-diagnostics globalDateInEffect:off - reads back the wall clock of an explicit +// timestamp the hook already computed; no ambient time is sampled. +import * as NodePath from "node:path"; + +import type { + HookInput, + StopHookInput, + SubagentStopHookInput, +} from "@anthropic-ai/claude-agent-sdk"; +import * as NodeServices from "@effect/platform-node/NodeServices"; +import { assert, describe, it } from "@effect/vitest"; +import * as Effect from "effect/Effect"; +import * as FileSystem from "effect/FileSystem"; +import * as Path from "effect/Path"; + +import { loopHooksFor, type LoopHookRun, type LoopHooks, makeLoopHooks } from "./crons.ts"; +import { LoopStore, type LoopStoreShape, makeLoopStore } from "./state.ts"; + +const THREAD_ID = "thread-loop-1"; +const MINUTE_MS = 60_000; +const HOUR_MS = 60 * MINUTE_MS; + +// The SDK hook contract is a Promise, so the test runner has to be one too. In production +// this is the adapter's own runtime (`Effect.runPromiseWith`), which is what puts hook +// logs on the session that owns them. +const run: LoopHookRun = Effect.runPromise; + +const notAborted = { signal: new AbortController().signal }; + +const abortedOptions = () => { + const controller = new AbortController(); + controller.abort(); + return { signal: controller.signal }; +}; + +const baseInput = { + session_id: "session-1", + transcript_path: "/tmp/transcript.jsonl", + cwd: "/tmp/workspace", +}; + +/** + * `session_crons` is deliberately `unknown`: the whole point of these cases is what the fork + * does with a payload it did not author, including one that omits the field entirely. + */ +const stopInput = (sessionCrons?: unknown): HookInput => + ({ + ...baseInput, + hook_event_name: "Stop", + stop_hook_active: false, + ...(sessionCrons === undefined ? {} : { session_crons: sessionCrons }), + }) as unknown as StopHookInput; + +const subagentStopInput = (sessionCrons?: unknown): HookInput => + ({ + ...baseInput, + hook_event_name: "SubagentStop", + stop_hook_active: false, + agent_id: "agent-1", + agent_transcript_path: "/tmp/agent.jsonl", + agent_type: "general-purpose", + ...(sessionCrons === undefined ? {} : { session_crons: sessionCrons }), + }) as unknown as SubagentStopHookInput; + +const postToolUseInput = (toolName: string, toolResponse: unknown): HookInput => + ({ + ...baseInput, + hook_event_name: "PostToolUse", + tool_name: toolName, + tool_input: {}, + tool_response: toolResponse, + tool_use_id: "toolu_1", + }) as unknown as HookInput; + +const stopCallback = (hooks: LoopHooks) => hooks.Stop![0]!.hooks[0]!; +const subagentStopCallback = (hooks: LoopHooks) => hooks.SubagentStop![0]!.hooks[0]!; +const postToolUseMatcher = (hooks: LoopHooks) => hooks.PostToolUse![0]!; + +/** + * A real store over a temp file, plus the hooks wired to it. `storePath` is handed back so a + * case can reopen the same file with a second store and prove the record is durable. + * + * The thread is **armed** by default, because that is the only state in which these + * callbacks record anything: they run on the Stop of every Claude turn on the machine, and + * an unarmed thread must cost zero writes. `armed: false` opts out, for the cases that prove + * exactly that. + */ +const withHooks = ( + f: (context: { + readonly store: LoopStoreShape; + readonly hooks: LoopHooks; + readonly storePath: string; + readonly reopen: Effect.Effect; + }) => Effect.Effect, + makeStore: (store: LoopStoreShape) => LoopStoreShape = (store) => store, + options: { readonly armed?: boolean } = {}, +) => + Effect.gen(function* () { + const fs = yield* FileSystem.FileSystem; + const root = yield* fs.makeTempDirectoryScoped({ prefix: "coil-loop-crons-" }); + const storePath = NodePath.join(root, "coil-loop.json"); + const store = yield* makeLoopStore(storePath); + if (options.armed !== false) { + yield* store.arm({ + threadId: THREAD_ID, + armedAtMs: 0, + deadlineAtMs: 4_102_444_800_000, + maxCheckIns: 6, + }); + } + return yield* f({ + store, + hooks: makeLoopHooks({ store: makeStore(store), threadId: THREAD_ID, run }), + storePath, + reopen: makeLoopStore(storePath), + }); + }).pipe(Effect.scoped, Effect.orDie, Effect.provide(NodeServices.layer), Effect.runPromise); + +describe("coil loop crons hooks", () => { + it("70b: records populated session_crons, computing nextFireAtMs from each schedule", () => + withHooks(({ store, hooks }) => + Effect.gen(function* () { + const output = yield* Effect.promise(() => + stopCallback(hooks)( + stopInput([ + { id: "cron-1", schedule: "*/5 * * * *", recurring: true, prompt: "keep going" }, + { id: "cron-2", schedule: "30 3 * * *", recurring: false, prompt: "one shot" }, + ]), + undefined, + notAborted, + ), + ); + assert.deepStrictEqual(output, { continue: true }); + + const record = yield* store.getThread(THREAD_ID); + assert.isNotNull(record.crons); + const crons = record.crons!; + assert.strictEqual(crons.entries.length, 2); + + const recurring = crons.entries[0]!; + assert.strictEqual(recurring.id, "cron-1"); + assert.strictEqual(recurring.schedule, "*/5 * * * *"); + assert.strictEqual(recurring.recurring, true); + assert.isNotNull(recurring.nextFireAtMs); + // The next match of a five-minute step is strictly ahead and at most a period away. + assert.isAbove(recurring.nextFireAtMs!, crons.recordedAtMs); + assert.isAtMost(recurring.nextFireAtMs!, crons.recordedAtMs + 5 * MINUTE_MS); + assert.strictEqual(new Date(recurring.nextFireAtMs!).getMinutes() % 5, 0); + + // A one-shot's cron fields encode a single instant; the same parse resolves it. + const oneShot = crons.entries[1]!; + assert.strictEqual(oneShot.recurring, false); + assert.isNotNull(oneShot.nextFireAtMs); + assert.isAbove(oneShot.nextFireAtMs!, crons.recordedAtMs); + assert.isAtMost(oneShot.nextFireAtMs!, crons.recordedAtMs + 24 * HOUR_MS); + const fireAt = new Date(oneShot.nextFireAtMs!); + assert.strictEqual(fireAt.getHours(), 3); + assert.strictEqual(fireAt.getMinutes(), 30); + }), + )); + + it("70b: an unparseable schedule is still recorded, with nextFireAtMs null (no deference)", () => + withHooks(({ store, hooks }) => + Effect.gen(function* () { + yield* Effect.promise(() => + stopCallback(hooks)( + stopInput([ + { id: "cron-1", schedule: "@daily", recurring: true, prompt: "macro" }, + { id: "cron-2", schedule: "0 9 * * 1-5", recurring: true, prompt: "weekdays" }, + ]), + undefined, + notAborted, + ), + ); + + const crons = (yield* store.getThread(THREAD_ID)).crons!; + assert.strictEqual(crons.entries.length, 2); + assert.strictEqual(crons.entries[0]!.nextFireAtMs, null); + assert.isNotNull(crons.entries[1]!.nextFireAtMs); + }), + )); + + it("70c: an empty session_crons array clears the record", () => + withHooks(({ store, hooks }) => + Effect.gen(function* () { + yield* Effect.promise(() => + stopCallback(hooks)( + stopInput([{ id: "cron-1", schedule: "*/5 * * * *", recurring: true, prompt: "x" }]), + undefined, + notAborted, + ), + ); + assert.strictEqual((yield* store.getThread(THREAD_ID)).crons!.entries.length, 1); + + yield* Effect.promise(() => stopCallback(hooks)(stopInput([]), undefined, notAborted)); + + const crons = (yield* store.getThread(THREAD_ID)).crons; + // Observed-and-empty, NOT never-observed: the agent stopped self-pacing. + assert.isNotNull(crons); + assert.deepStrictEqual(crons!.entries, []); + }), + )); + + it("70d: an absent session_crons field leaves the record untouched", () => + withHooks(({ store, hooks }) => + Effect.gen(function* () { + yield* Effect.promise(() => + stopCallback(hooks)( + stopInput([{ id: "cron-1", schedule: "*/5 * * * *", recurring: true, prompt: "x" }]), + undefined, + notAborted, + ), + ); + const before = (yield* store.getThread(THREAD_ID)).crons; + + yield* Effect.promise(() => stopCallback(hooks)(stopInput(), undefined, notAborted)); + + assert.deepStrictEqual((yield* store.getThread(THREAD_ID)).crons, before); + }), + )); + + it("70d: a never-observed thread stays null when the field is absent", () => + withHooks(({ store, hooks }) => + Effect.gen(function* () { + yield* Effect.promise(() => stopCallback(hooks)(stopInput(), undefined, notAborted)); + assert.strictEqual((yield* store.getThread(THREAD_ID)).crons, null); + }), + )); + + it("70e: a malformed entry is dropped individually and the rest still record", () => + withHooks(({ store, hooks }) => + Effect.gen(function* () { + yield* Effect.promise(() => + stopCallback(hooks)( + stopInput([ + null, + 42, + { schedule: "*/5 * * * *", recurring: true, prompt: "no id" }, + { id: "", schedule: "*/5 * * * *", recurring: true, prompt: "empty id" }, + { id: "cron-blank", schedule: " ", recurring: true, prompt: "blank schedule" }, + { id: "cron-ok", schedule: "0 3 * * *", recurring: true, prompt: "keeps" }, + ]), + undefined, + notAborted, + ), + ); + + const crons = (yield* store.getThread(THREAD_ID)).crons!; + assert.deepStrictEqual( + crons.entries.map((entry) => entry.id), + ["cron-ok"], + ); + }), + )); + + it("70e: a non-array session_crons is treated as absent, not as empty", () => + withHooks(({ store, hooks }) => + Effect.gen(function* () { + yield* Effect.promise(() => + stopCallback(hooks)( + stopInput([{ id: "cron-1", schedule: "0 3 * * *", recurring: true, prompt: "x" }]), + undefined, + notAborted, + ), + ); + + yield* Effect.promise(() => + stopCallback(hooks)(stopInput({ nope: true }), undefined, notAborted), + ); + + assert.strictEqual((yield* store.getThread(THREAD_ID)).crons!.entries.length, 1); + }), + )); + + it("70f: a store that dies still returns continue and writes nothing", () => + withHooks( + ({ store, hooks }) => + Effect.gen(function* () { + const output = yield* Effect.promise(() => + stopCallback(hooks)( + stopInput([{ id: "cron-1", schedule: "0 3 * * *", recurring: true, prompt: "x" }]), + undefined, + notAborted, + ), + ); + // No `decision`, no `continue: false` — a Stop hook can halt a turn and this one + // never may, whatever the fork's own bookkeeping did. + assert.deepStrictEqual(output, { continue: true }); + assert.strictEqual((yield* store.getThread(THREAD_ID)).crons, null); + }), + (store) => ({ + ...store, + setCrons: () => + Effect.sync(() => { + throw new Error("simulated store defect"); + }), + }), + )); + + it("70f: a store that throws synchronously still returns continue", () => + withHooks( + ({ hooks }) => + Effect.gen(function* () { + const output = yield* Effect.promise(() => + stopCallback(hooks)( + stopInput([{ id: "cron-1", schedule: "0 3 * * *", recurring: true, prompt: "x" }]), + undefined, + notAborted, + ), + ); + assert.deepStrictEqual(output, { continue: true }); + }), + (store) => ({ + ...store, + setCrons: () => { + throw new Error("simulated synchronous throw"); + }, + }), + )); + + it("70f: an already-aborted signal returns continue and records nothing", () => + withHooks(({ store, hooks }) => + Effect.gen(function* () { + const output = yield* Effect.promise(() => + stopCallback(hooks)( + stopInput([{ id: "cron-1", schedule: "0 3 * * *", recurring: true, prompt: "x" }]), + undefined, + abortedOptions(), + ), + ); + assert.deepStrictEqual(output, { continue: true }); + assert.strictEqual((yield* store.getThread(THREAD_ID)).crons, null); + }), + )); + + it("records nothing at all for a thread with no armed loop", () => + withHooks( + ({ store, hooks }) => + Effect.gen(function* () { + const output = yield* Effect.promise(() => + stopCallback(hooks)( + stopInput([{ id: "cron-1", schedule: "0 3 * * *", recurring: true, prompt: "x" }]), + undefined, + notAborted, + ), + ); + assert.deepStrictEqual(output, { continue: true }); + // Every Claude turn on the machine ends in a Stop, and the state file is rewritten + // in full on every mutation: recording here would be a write per turn per thread + // for a table nothing supervises and nothing will read. + assert.strictEqual((yield* store.getThread(THREAD_ID)).crons, null); + }), + undefined, + { armed: false }, + )); + + it("does not set degraded on a thread with no armed loop", () => + withHooks( + ({ store, hooks }) => + Effect.gen(function* () { + yield* Effect.promise(() => + postToolUseMatcher(hooks).hooks[0]!( + postToolUseInput("ScheduleWakeup", { error: "gate_off" }), + undefined, + notAborted, + ), + ); + assert.strictEqual((yield* store.getThread(THREAD_ID)).degraded, null); + }), + undefined, + { armed: false }, + )); + + it("70f: every matcher carries a timeout so a wedged callback cannot stall a turn", () => + withHooks(({ hooks }) => + Effect.sync(() => { + for (const matchers of [hooks.Stop, hooks.SubagentStop, hooks.PostToolUse]) { + assert.isDefined(matchers); + for (const matcher of matchers!) { + assert.isAbove(matcher.timeout ?? 0, 0); + } + } + }), + )); + + it("70g: SubagentStop is handled identically to Stop", () => + withHooks(({ store, hooks }) => + Effect.gen(function* () { + const output = yield* Effect.promise(() => + subagentStopCallback(hooks)( + subagentStopInput([ + { id: "cron-sub", schedule: "*/10 * * * *", recurring: true, prompt: "sub" }, + ]), + undefined, + notAborted, + ), + ); + assert.deepStrictEqual(output, { continue: true }); + + const crons = (yield* store.getThread(THREAD_ID)).crons!; + assert.deepStrictEqual( + crons.entries.map((entry) => entry.id), + ["cron-sub"], + ); + assert.isNotNull(crons.entries[0]!.nextFireAtMs); + + // And SubagentStop clears on empty, exactly as Stop does. + yield* Effect.promise(() => + subagentStopCallback(hooks)(subagentStopInput([]), undefined, notAborted), + ); + assert.deepStrictEqual((yield* store.getThread(THREAD_ID)).crons!.entries, []); + }), + )); + + it("70h: the record survives a store round-trip", () => + withHooks(({ store, hooks, reopen }) => + Effect.gen(function* () { + yield* Effect.promise(() => + stopCallback(hooks)( + stopInput([{ id: "cron-1", schedule: "0 3 * * *", recurring: true, prompt: "night" }]), + undefined, + notAborted, + ), + ); + const written = (yield* store.getThread(THREAD_ID)).crons; + + const rehydrated = yield* reopen; + assert.deepStrictEqual((yield* rehydrated.getThread(THREAD_ID)).crons, written); + }), + )); + + it("70i: the binary's truncated prompt round-trips verbatim", () => + withHooks(({ store, hooks }) => + Effect.gen(function* () { + // Exactly what the binary sends: capped at 1000 chars with its own clip marker. + const truncated = `${"a".repeat(982)}… [+412 chars]`; + assert.strictEqual(truncated.length, 996); + + yield* Effect.promise(() => + stopCallback(hooks)( + stopInput([ + { id: "cron-1", schedule: "0 3 * * *", recurring: true, prompt: truncated }, + ]), + undefined, + notAborted, + ), + ); + + const entry = (yield* store.getThread(THREAD_ID)).crons!.entries[0]!; + assert.strictEqual(entry.prompt, truncated); + }), + )); + + it("70j: a ScheduleWakeup response containing the gate marker sets degraded", () => + withHooks(({ store, hooks }) => + Effect.gen(function* () { + const matcher = postToolUseMatcher(hooks); + assert.strictEqual(matcher.matcher, "ScheduleWakeup"); + + const output = yield* Effect.promise(() => + matcher.hooks[0]!( + postToolUseInput("ScheduleWakeup", { + status: "error", + detail: "scheduler unavailable: GATE_OFF", + }), + undefined, + notAborted, + ), + ); + assert.deepStrictEqual(output, { continue: true }); + assert.strictEqual((yield* store.getThread(THREAD_ID)).degraded, "gate_off"); + }), + )); + + it("70k: a response without the marker leaves degraded untouched", () => + withHooks(({ store, hooks }) => + Effect.gen(function* () { + yield* store.setDegraded(THREAD_ID, "wake_lost"); + + yield* Effect.promise(() => + postToolUseMatcher(hooks).hooks[0]!( + postToolUseInput("ScheduleWakeup", { status: "ok", wakeAt: "2026-09-02T03:00:00Z" }), + undefined, + notAborted, + ), + ); + + // A successful call must never clear an unrelated degraded state by accident. + assert.strictEqual((yield* store.getThread(THREAD_ID)).degraded, "wake_lost"); + }), + )); + + it("70k: an unserializable response finds nothing and behaves like no probe", () => + withHooks(({ store, hooks }) => + Effect.gen(function* () { + const circular: Record = {}; + circular.self = circular; + + const output = yield* Effect.promise(() => + postToolUseMatcher(hooks).hooks[0]!( + postToolUseInput("ScheduleWakeup", circular), + undefined, + notAborted, + ), + ); + assert.deepStrictEqual(output, { continue: true }); + assert.strictEqual((yield* store.getThread(THREAD_ID)).degraded, null); + }), + )); + + it("70k: a different tool reaching the callback is ignored", () => + withHooks(({ store, hooks }) => + Effect.gen(function* () { + yield* Effect.promise(() => + postToolUseMatcher(hooks).hooks[0]!( + postToolUseInput("Bash", { stdout: "gate_off" }), + undefined, + notAborted, + ), + ); + assert.strictEqual((yield* store.getThread(THREAD_ID)).degraded, null); + }), + )); +}); + +describe("loopHooksFor", () => { + it("builds nothing when no LoopStore is in context, so the adapter stays hook-free", () => + loopHooksFor("thread-1").pipe( + Effect.map((hooks) => { + assert.isUndefined(hooks); + }), + Effect.runPromise, + )); + + it("builds the three subscriptions when a LoopStore is available", () => + withHooks(({ store }) => + Effect.gen(function* () { + const hooks = yield* loopHooksFor(THREAD_ID).pipe(Effect.provideService(LoopStore, store)); + assert.isDefined(hooks); + assert.deepStrictEqual(Object.keys(hooks!).sort(), ["PostToolUse", "Stop", "SubagentStop"]); + }), + )); + + it("builds nothing when the deployment kill switch is off", () => + withHooks(({ store }) => + Effect.gen(function* () { + const previous = process.env.COIL_LOOP_ENABLED; + process.env.COIL_LOOP_ENABLED = "false"; + try { + const hooks = yield* loopHooksFor(THREAD_ID).pipe( + Effect.provideService(LoopStore, store), + ); + assert.isUndefined(hooks); + } finally { + if (previous === undefined) delete process.env.COIL_LOOP_ENABLED; + else process.env.COIL_LOOP_ENABLED = previous; + } + }), + )); +}); diff --git a/apps/server/src/coil/loop/crons.ts b/apps/server/src/coil/loop/crons.ts new file mode 100644 index 000000000000..ee51178a5ac9 --- /dev/null +++ b/apps/server/src/coil/loop/crons.ts @@ -0,0 +1,297 @@ +/** + * The Claude hook callbacks that record the agent's own scheduled wakes. + * + * `options.hooks` is set nowhere else in this repo — 40 events, zero subscriptions — so the + * whole subscription is built here and reaches the SDK through one additive spread in + * `ClaudeAdapter.ts`. Three events are subscribed: + * + * | Event | Matcher | Reads | + * | ------------- | ------------------ | ------------------- | + * | `Stop` | — | `session_crons` | + * | `SubagentStop`| — | `session_crons` | + * | `PostToolUse` | `"ScheduleWakeup"` | `tool_response` | + * + * ## Absent and empty are different facts + * + * `session_crons` is typed optional but the shipped binary always sends it — `[]` when + * nothing is scheduled. So `[]` **clears** the record (the agent stopped self-pacing) and + * **absent leaves it untouched** (an older or future build that does not report). Getting + * this backwards silently retires deference: every stop would look like "no wake pending" + * and the supervisor would nudge straight through a healthy self-paced run. + * + * ## Only armed threads accrue records + * + * Both callbacks read the record and return early unless the thread is armed. Every Claude + * turn on the machine ends in a `Stop`, and `coil-loop.json` is rewritten in full on every + * mutation, so recording unconditionally would mean a write per turn per thread for a table + * nothing will ever read. The `gate_off` probe follows the same rule for the same reason — + * a degraded state is a fact about a supervised run. + * + * ## A throwing hook must never break the turn + * + * `HookJSONOutput` carries `continue` and, for `Stop`, `decision: "block"` — **a Stop hook + * can halt a turn.** "Observability only" is therefore a property this code holds, not one + * the surface gives you. Every callback catches every failure *and every defect*, logs at + * debug, and always resolves to `{ continue: true }`: never `false`, never `"block"`, never + * `undefined`, never a rejected promise. Each matcher also carries a timeout so a wedged + * fork callback cannot stall a turn, and an already-aborted signal short-circuits. + * + * ## Why `nextFireAtMs` is computed here and persisted + * + * The SDK delivers a cron *expression*, never a timestamp. A one-shot entry encodes a single + * instant in those same five fields, so re-parsing it after it has fired resolves to next + * year's occurrence — the value is only correct when computed at the moment the entry is + * observed. That is also why the parse is logged beside the raw `schedule`: it is the + * biggest open assumption in the design, and this is the cheap place to check it. + * + * @module coil/loop/crons + */ + +import type { + HookCallback, + HookJSONOutput, + Options as ClaudeQueryOptions, +} from "@anthropic-ai/claude-agent-sdk"; +import * as Cause from "effect/Cause"; +import * as Clock from "effect/Clock"; +import * as Effect from "effect/Effect"; +import * as Option from "effect/Option"; + +import { resolveConfig } from "./config.ts"; +import { nextFireAtMs } from "./cron/parse.ts"; +import { installedLoopStore } from "./hooksRegistry.ts"; +import { type CronEntry, type CronRecord, LoopStore, type LoopStoreShape } from "./state.ts"; + +/** What `queryOptions.hooks` wants. Kept as an alias so the SDK owns the shape. */ +export type LoopHooks = NonNullable; + +/** + * Seconds. Bounds a wedged fork callback, not a slow one: everything these callbacks do is + * an in-memory mutation plus one atomic file write. + */ +const HOOK_TIMEOUT_SECONDS = 5; + +/** The tool whose response carries the gate status. See `gate_off` below. */ +const SCHEDULE_WAKEUP_TOOL = "ScheduleWakeup"; + +/** + * Substring probe, deliberately not a parse. + * + * The plumbing is verified — the matcher selects by tool name and `PostToolUseHookInput` + * carries `tool_response` — but the response *body* is not. So a response that stringifies + * to something containing this marker sets `degraded`, and anything else leaves `degraded` + * exactly as it was: a successful call must never clear an unrelated degraded state by + * accident, and a probe that finds nothing must behave exactly like no probe at all. + */ +const GATE_OFF_MARKER = "gate_off"; + +/** Always this, on every path. */ +const CONTINUE: HookJSONOutput = { continue: true }; + +/** + * Runs a hook body on the adapter's own runtime, so logs and spans land where the rest of + * the session's do. Supplied by `loopHooksFor`; injected here so the callbacks are testable + * without a provider. + */ +export type LoopHookRun = (effect: Effect.Effect) => Promise; + +export interface LoopHooksInput { + readonly store: LoopStoreShape; + readonly threadId: string; + readonly run: LoopHookRun; +} + +/** + * One `session_crons` entry, defensively. + * + * Returns `null` for anything unusable, and the caller drops it individually so one + * malformed entry cannot cost the rest of the array. `prompt` is stored exactly as it + * arrives — the binary already truncates it to 1000 chars with a `… [+N chars]` marker, and + * it is console display text, never the agent's prompt. + */ +function normalizeCronEntry(raw: unknown, nowMs: number): CronEntry | null { + if (typeof raw !== "object" || raw === null) return null; + const candidate = raw as { + readonly id?: unknown; + readonly schedule?: unknown; + readonly recurring?: unknown; + readonly prompt?: unknown; + }; + if (typeof candidate.id !== "string" || candidate.id.length === 0) return null; + if (typeof candidate.schedule !== "string" || candidate.schedule.trim().length === 0) return null; + return { + id: candidate.id, + schedule: candidate.schedule, + recurring: candidate.recurring === true, + prompt: typeof candidate.prompt === "string" ? candidate.prompt : "", + // `null` = did not parse = NO deference from this entry. Never "defer forever". + nextFireAtMs: nextFireAtMs(candidate.schedule, nowMs), + }; +} + +/** + * The snapshot to persist, or `null` when the payload says nothing. + * + * `null` is the "leave it alone" answer and is returned for exactly one reason: the field is + * absent. An empty array is a statement — the agent has no pending wake — and produces a + * record with no entries. + */ +export function normalizeCronSnapshot(sessionCrons: unknown, nowMs: number): CronRecord | null { + if (!Array.isArray(sessionCrons)) return null; + const entries = sessionCrons.flatMap((raw) => { + const entry = normalizeCronEntry(raw, nowMs); + return entry === null ? [] : [entry]; + }); + return { recordedAtMs: nowMs, entries }; +} + +function stringifyToolResponse(toolResponse: unknown): string | null { + if (typeof toolResponse === "string") return toolResponse; + try { + return JSON.stringify(toolResponse) ?? null; + } catch { + // Circular or otherwise unserializable: the probe simply finds nothing. + return null; + } +} + +const recordCronSnapshot = ( + store: LoopStoreShape, + threadId: string, + event: string, + sessionCrons: unknown, +): Effect.Effect => + Effect.gen(function* () { + // Armed only. These callbacks run on the Stop of EVERY Claude turn on the machine, and + // `coil-loop.json` is rewritten in full on every mutation — recording a cron table for + // threads nothing supervises would grow one shared file by a write per turn for no + // reader at all. Same rule as `userInputs.ts`. + if (!(yield* store.getThread(threadId)).armed) return; + const nowMs = yield* Clock.currentTimeMillis; + const snapshot = normalizeCronSnapshot(sessionCrons, nowMs); + if (snapshot === null) { + yield* Effect.logDebug("coil loop: stop hook reported no session_crons field", { + threadId, + event, + }); + return; + } + yield* store.setCrons(threadId, snapshot); + // The parse beside the raw expression: this log is how the biggest open assumption in + // the design gets checked against what actually fires. + yield* Effect.logDebug("coil loop: recorded the agent's scheduled wakes", { + threadId, + event, + entries: snapshot.entries.map((entry) => ({ + id: entry.id, + schedule: entry.schedule, + recurring: entry.recurring, + nextFireAtMs: entry.nextFireAtMs, + })), + }); + }); + +const probeGateOff = ( + store: LoopStoreShape, + threadId: string, + toolResponse: unknown, +): Effect.Effect => + Effect.gen(function* () { + const text = stringifyToolResponse(toolResponse); + if (text === null || !text.toLowerCase().includes(GATE_OFF_MARKER)) return; + // Armed only, for the same reason the cron snapshot is: `degraded` is a fact about a + // supervised run, and only a supervised run has anywhere to show it. + if (!(yield* store.getThread(threadId)).armed) return; + yield* store.setDegraded(threadId, "gate_off"); + yield* Effect.logDebug("coil loop: scheduler gate reported off", { threadId }); + }); + +/** + * The boundary. Nothing past this point can reach the turn: an aborted signal returns + * immediately, a failure or defect inside the body is logged and swallowed, a rejected + * promise is swallowed, and a synchronous throw while starting the effect is swallowed. + */ +function settle( + run: LoopHookRun, + signal: AbortSignal, + body: () => Effect.Effect, +): Promise { + if (signal.aborted) return Promise.resolve(CONTINUE); + try { + return run( + Effect.suspend(body).pipe( + Effect.catchCause((cause) => + Effect.logDebug("coil loop: hook callback failed", { cause: Cause.pretty(cause) }), + ), + ), + ).then( + () => CONTINUE, + () => CONTINUE, + ); + } catch { + return Promise.resolve(CONTINUE); + } +} + +/** + * The `hooks` object for one Claude session. + * + * Takes no `LoopConfig`: the deployment kill switch is read once in `loopHooksFor` (no hooks + * object is built at all when it is off), and nothing else in this module is tunable. + */ +export function makeLoopHooks(input: LoopHooksInput): LoopHooks { + const { store, threadId, run } = input; + + // `Stop` and `SubagentStop` are handled identically: both carry the same session-scoped + // cron table, and a subagent finishing is as good a moment to read it as the main loop. + const onStop: HookCallback = (hookInput, _toolUseId, options) => + settle(run, options.signal, () => + recordCronSnapshot( + store, + threadId, + hookInput.hook_event_name, + "session_crons" in hookInput ? hookInput.session_crons : undefined, + ), + ); + + const onScheduleWakeup: HookCallback = (hookInput, _toolUseId, options) => + settle(run, options.signal, () => + hookInput.hook_event_name === "PostToolUse" && hookInput.tool_name === SCHEDULE_WAKEUP_TOOL + ? probeGateOff(store, threadId, hookInput.tool_response) + : Effect.void, + ); + + return { + Stop: [{ hooks: [onStop], timeout: HOOK_TIMEOUT_SECONDS }], + SubagentStop: [{ hooks: [onStop], timeout: HOOK_TIMEOUT_SECONDS }], + PostToolUse: [ + { matcher: SCHEDULE_WAKEUP_TOOL, hooks: [onScheduleWakeup], timeout: HOOK_TIMEOUT_SECONDS }, + ], + }; +} + +/** + * The adapter's whole entry point: one `yield*` that resolves to the hooks object, or + * `undefined` when loops are not in play. + * + * `Effect.serviceOption` is what keeps this free: reading `LoopStore` optionally means the + * adapter's Layer requirements do not widen, so no second seam edit appears in `server.ts` + * or in upstream's adapter tests. But it is not what makes it *work* in production — the + * adapter's fiber runs in upstream's layer graph, where `CoilLayerLive` has already + * discharged `LoopStore` with `Layer.provide`, so the service is genuinely absent there. + * `hooksRegistry.ts` is the fallback the running supervisor installs; the context is still + * preferred when one carries the store, which is how every test that provides it stays + * honest. A server with no loop layer at all finds neither and gets `undefined`. + * + * The runtime context is captured from the caller either way, so hook logs and spans belong + * to the session that owns them. + */ +export const loopHooksFor = (threadId: string): Effect.Effect => + Effect.gen(function* () { + if (!resolveConfig().enabled) return undefined; + const fromContext = yield* Effect.serviceOption(LoopStore); + const store = Option.isSome(fromContext) ? fromContext.value : installedLoopStore(); + if (store === null) return undefined; + const context = yield* Effect.context(); + return makeLoopHooks({ store, threadId, run: Effect.runPromiseWith(context) }); + }); diff --git a/apps/server/src/coil/loop/decide.test.ts b/apps/server/src/coil/loop/decide.test.ts new file mode 100644 index 000000000000..187bea9dd2f6 --- /dev/null +++ b/apps/server/src/coil/loop/decide.test.ts @@ -0,0 +1,664 @@ +// @effect-diagnostics globalDate:off -- `iso` is a pure ms->ISO fixture helper anchored on a +// fixed constant, never a wall-clock reading, so DateTime's effectful now-semantics would add +// ceremony without adding correctness. +/** + * TESTS.md §1, cases 1–29 (including 11b–11l for deference and 15/15b/15c for the deadline + * while busy): the decision table. Case numbers lead each test name so a number quoted in + * the design still names the same test. + * + * Every fixture record is frozen, so a decision that mutated its input would throw rather + * than quietly spend a budget. + */ + +import { describe, expect, it } from "vite-plus/test"; + +import { resolveConfig, wakeGraceMs } from "./config.ts"; +import { decide, judgeProgress, resolveTrigger, resolveWake } from "./decide.ts"; +import { DEFAULT_GLOBAL_SETTINGS, EMPTY_RECORD, type CronEntry, type LoopRecord } from "./state.ts"; +import type { LoopAction, LoopDecisionInput, LoopThreadShell } from "./types.ts"; + +const config = resolveConfig({}); // idle 15m, busy 45m, productive 2m, grace 90s..15m @10% +const NOW = 1_800_000_000_000; // 2027-01-15T08:00:00Z +const MINUTE = 60_000; +const HOUR = 60 * MINUTE; +const iso = (ms: number) => new Date(ms).toISOString(); + +const globalSettings = { ...DEFAULT_GLOBAL_SETTINGS, enabled: true, maxArmedThreads: 3 }; + +const record = (o: Partial = {}): LoopRecord => + Object.freeze({ + ...EMPTY_RECORD, + armed: true, + armedAtMs: NOW - 2 * HOUR, + maxCheckIns: 6, + checkInsUsed: 1, + deadlineAtMs: NOW + 4 * HOUR, + ...o, + }); + +type ShellOverrides = { + updatedAt?: string; + sessionStatus?: string | null; + providerName?: string | null; + latestTurnState?: string | null; + backgroundLiveness?: "working" | "monitoring" | null; + latestUserMessageAt?: string | null; + settledOverride?: "settled" | "active" | null; +}; + +const shell = (o: ShellOverrides = {}): LoopThreadShell => + ({ + updatedAt: o.updatedAt ?? iso(NOW - 20 * MINUTE), + archivedAt: null, + settledOverride: o.settledOverride ?? null, + snoozedUntil: null, + session: + o.sessionStatus === null + ? null + : { + threadId: "thread-1", + status: o.sessionStatus ?? "ready", + providerName: o.providerName === undefined ? "claudeAgent" : o.providerName, + runtimeMode: "local", + activeTurnId: null, + lastError: null, + updatedAt: iso(NOW - 20 * MINUTE), + }, + latestTurn: + typeof o.latestTurnState === "string" + ? { + turnId: "turn-1", + state: o.latestTurnState, + requestedAt: iso(NOW - HOUR), + startedAt: iso(NOW - HOUR), + completedAt: null, + assistantMessageId: null, + } + : null, + latestUserMessageAt: o.latestUserMessageAt ?? null, + hasPendingApprovals: false, + hasPendingUserInput: false, + hasActionableProposedPlan: false, + backgroundLiveness: o.backgroundLiveness ?? null, + }) as unknown as LoopThreadShell; + +const entry = (o: Partial = {}): CronEntry => ({ + id: "cron-1", + schedule: "*/30 * * * *", + recurring: false, + prompt: "keep going", + nextFireAtMs: null, + ...o, +}); + +const crons = (entries: ReadonlyArray) => ({ recordedAtMs: NOW - HOUR, entries }); + +const input = (o: Partial = {}): LoopDecisionInput => ({ + nowMs: NOW, + processStartedAtMs: NOW - 24 * HOUR, + record: record(), + global: globalSettings, + shell: shell(), + sentinelAtMs: null, + loopDoneAtMs: null, + autoResumePending: false, + armedCount: 1, + config, + ...o, +}); + +const act = (o: Partial = {}): LoopAction => decide(input(o)); + +describe("decide — 1.1 trigger arithmetic", () => { + it("1. idle below the threshold keeps watching", () => { + expect(act({ shell: shell({ updatedAt: iso(NOW - 14 * MINUTE) }) })).toMatchObject({ + type: "stand_down", + reason: "not_idle", + phase: "watching", + }); + }); + + it("2. idle exactly at the threshold fires (the boundary is inclusive)", () => { + expect(act({ shell: shell({ updatedAt: iso(NOW - 15 * MINUTE) }) })).toMatchObject({ + type: "fire", + kind: "check_in", + }); + }); + + it("3. idle above the threshold fires", () => { + expect(act({ shell: shell({ updatedAt: iso(NOW - HOUR) }) })).toMatchObject({ type: "fire" }); + }); + + it("4. a busy turn uses busyIdleMs, not idleMs", () => { + const busy = shell({ sessionStatus: "running", updatedAt: iso(NOW - 46 * MINUTE) }); + expect(resolveTrigger(input({ shell: busy })).thresholdMs).toBe(45 * MINUTE); + expect(act({ shell: busy })).toMatchObject({ type: "fire" }); + }); + + it("5. a busy turn idle between the two thresholds keeps watching (the long tool call)", () => { + const busy = shell({ sessionStatus: "running", updatedAt: iso(NOW - 30 * MINUTE) }); + expect(act({ shell: busy })).toMatchObject({ type: "stand_down", reason: "not_idle" }); + }); + + it("6. `starting` counts as busy", () => { + expect(resolveTrigger(input({ shell: shell({ sessionStatus: "starting" }) })).busyTurn).toBe( + true, + ); + }); + + it("7. a running latest turn counts as busy even when the session is null", () => { + const facts = resolveTrigger( + input({ shell: shell({ sessionStatus: null, latestTurnState: "running" }) }), + ); + expect(facts.busyTurn).toBe(true); + expect(facts.thresholdMs).toBe(45 * MINUTE); + }); + + // The single most important line in the design: gating on `running` deadlocks the exact + // threads this feature is for, because a turn whose completion never arrives pins the + // status with nothing automated to clear it. + it("8. `running` alone never suppresses a fire — it only lengthens the fuse", () => { + expect( + act({ shell: shell({ sessionStatus: "running", updatedAt: iso(NOW - 3 * HOUR) }) }), + ).toMatchObject({ type: "fire" }); + }); + + it("8b. backgroundLiveness lengthens the fuse and is never a veto", () => { + const live = shell({ backgroundLiveness: "working", updatedAt: iso(NOW - 30 * MINUTE) }); + expect(resolveTrigger(input({ shell: live })).thresholdMs).toBe(45 * MINUTE); + expect(act({ shell: live })).toMatchObject({ type: "stand_down", reason: "not_idle" }); + // ...but past the longer threshold it still fires. An empty roster after a restart must + // never read as "nothing is running". + const stale = shell({ backgroundLiveness: "monitoring", updatedAt: iso(NOW - 3 * HOUR) }); + expect(act({ shell: stale })).toMatchObject({ type: "fire" }); + }); + + it("9. processStartedAtMs clamps the idle floor after a restart", () => { + const action = act({ + shell: shell({ updatedAt: iso(NOW - 6 * HOUR) }), + processStartedAtMs: NOW - 30_000, + }); + expect(action).toMatchObject({ type: "stand_down", reason: "not_idle" }); + }); + + it("10. an unparseable updatedAt yields no NaN and does not fire", () => { + const facts = resolveTrigger(input({ shell: shell({ updatedAt: "not-a-timestamp" }) })); + expect(facts.idleForMs).toBe(0); + expect(Number.isNaN(facts.idleForMs)).toBe(false); + expect(act({ shell: shell({ updatedAt: "not-a-timestamp" }) })).toMatchObject({ + type: "stand_down", + reason: "not_idle", + }); + }); + + it("11. an updatedAt in the future yields idle 0, not a negative", () => { + const facts = resolveTrigger(input({ shell: shell({ updatedAt: iso(NOW + HOUR) }) })); + expect(facts.idleForMs).toBe(0); + }); +}); + +describe("decide — 1.1b deference to the agent's own scheduler", () => { + const selfPaced = (entries: ReadonlyArray, o: Partial = {}) => + act({ record: record({ crons: crons(entries) }), ...o }); + + it("11b. a wake inside the threshold window stands the loop down, budget untouched", () => { + const rec = record({ crons: crons([entry({ nextFireAtMs: NOW + 5 * MINUTE })]) }); + expect(act({ record: rec })).toMatchObject({ + type: "stand_down", + reason: "self_pacing", + phase: "self_pacing", + untilMs: NOW + 5 * MINUTE, + }); + expect(rec.checkInsUsed).toBe(1); + }); + + it("11c. a wake beyond the window but inside the deadline still stands the loop down", () => { + // A thread waiting on a wake is not idle, whatever the wake's distance inside the run. + expect(selfPaced([entry({ nextFireAtMs: NOW + 3 * HOUR })])).toMatchObject({ + reason: "self_pacing", + }); + }); + + it("11d. a wake with updatedAt movement after it landed — detected without waiting out the grace", () => { + const action = selfPaced([entry({ nextFireAtMs: NOW - 30 * MINUTE })], { + shell: shell({ updatedAt: iso(NOW - 5 * MINUTE) }), + }); + expect(action).toMatchObject({ type: "stand_down", reason: "not_idle" }); + }); + + it("11e. a wake overdue by its grace fires as wake_lost, and the boundary is inclusive", () => { + const atMs = NOW - 90_000; + const rec = record({ crons: crons([entry({ nextFireAtMs: atMs })]) }); + const stale = shell({ updatedAt: iso(NOW - 3 * HOUR) }); + expect(decide(input({ record: rec, shell: stale }))).toMatchObject({ + type: "fire", + kind: "wake_lost", + degrade: "wake_lost", + }); + // One millisecond inside the grace, T3 still stands down. + expect(decide(input({ record: rec, shell: stale, nowMs: NOW - 1 }))).toMatchObject({ + reason: "self_pacing", + }); + }); + + it("11f. no cron record, or a non-Claude thread, falls back to pure staleness", () => { + expect(act()).toMatchObject({ type: "fire", kind: "check_in", degrade: null }); + // Every non-Claude adapter behaves exactly as it did before deference existed, even if + // a record somehow carries entries. + const rec = record({ crons: crons([entry({ nextFireAtMs: NOW + 5 * MINUTE })]) }); + expect(act({ record: rec, shell: shell({ providerName: "codex" }) })).toMatchObject({ + type: "fire", + }); + expect(resolveWake(input({ record: rec, shell: shell({ providerName: null }) }))).toBeNull(); + // An entry whose schedule did not parse contributes no wake: an unreadable schedule must + // never stand supervision down. + const unparseable = record({ + crons: crons([entry({ schedule: "every so often", nextFireAtMs: null })]), + }); + expect(resolveWake(input({ record: unparseable }))).toBeNull(); + expect(act({ record: unparseable })).toMatchObject({ type: "fire", kind: "check_in" }); + }); + + it("11i-b. with several live crons the earliest wake wins, whatever the recorded order", () => { + const early = entry({ id: "early", nextFireAtMs: NOW + 5 * MINUTE }); + const late = entry({ id: "late", nextFireAtMs: NOW + 90 * MINUTE }); + expect(resolveWake(input({ record: record({ crons: crons([late, early]) }) }))?.cronId).toBe( + "early", + ); + expect(resolveWake(input({ record: record({ crons: crons([early, late]) }) }))?.cronId).toBe( + "early", + ); + }); + + it("11i-c. a wake that already landed never masks a still-pending one", () => { + // The snapshot is a table, not a queue: it is only refreshed when a `Stop` hook lands, so + // the earliest entry is routinely one that already fired. Taking it anyway answered "the + // wake landed, nothing to defer to" while a second entry was still hours out — one dropped + // `Stop` (a timeout, a teardown, a restart) and T3 nudged a self-pacing thread. + const landed = entry({ id: "landed", nextFireAtMs: NOW - 30 * MINUTE }); + const pending = entry({ id: "pending", nextFireAtMs: NOW + 105 * MINUTE }); + const rec = record({ crons: crons([landed, pending]) }); + // `updatedAt` moved after the first wake, which is what makes it history. + const moved = shell({ updatedAt: iso(NOW - 29 * MINUTE) }); + expect(resolveWake(input({ record: rec, shell: moved }))?.cronId).toBe("pending"); + expect(decide(input({ record: rec, shell: moved }))).toMatchObject({ + type: "stand_down", + reason: "self_pacing", + untilMs: NOW + 105 * MINUTE, + }); + // With every recorded wake landed there is simply nothing left to defer to, exactly as + // before: no deference, and no `wake_lost` either. + const allLanded = record({ crons: crons([landed]) }); + expect(resolveWake(input({ record: allLanded, shell: moved }))).toBeNull(); + expect(decide(input({ record: allLanded, shell: moved }))).toMatchObject({ + type: "fire", + kind: "check_in", + degrade: null, + }); + }); + + it("11g. a stale record whose session is gone or stopped is not a live wake", () => { + const rec = record({ crons: crons([entry({ nextFireAtMs: NOW + 5 * MINUTE })]) }); + expect( + resolveWake(input({ record: rec, shell: shell({ sessionStatus: "stopped" }) })), + ).toBeNull(); + expect(resolveWake(input({ record: rec, shell: shell({ sessionStatus: null }) }))).toBeNull(); + expect(act({ record: rec, shell: shell({ sessionStatus: "stopped" }) })).toMatchObject({ + type: "fire", + }); + }); + + it("11h. a gate_off degradation is surfaced, never a reason to stand down", () => { + const rec = record({ degraded: "gate_off", crons: crons([]) }); + expect(act({ record: rec })).toMatchObject({ type: "fire", kind: "check_in" }); + expect(rec.degraded).toBe("gate_off"); + }); + + it("11i. two entries for the same cron: the newest wins, and there is one decision", () => { + const rec = record({ + crons: crons([ + entry({ id: "c", nextFireAtMs: NOW + 5 * MINUTE }), + entry({ id: "c", nextFireAtMs: NOW - 3 * HOUR }), + ]), + }); + const stale = shell({ updatedAt: iso(NOW - 4 * HOUR) }); + expect(resolveWake(input({ record: rec, shell: stale }))?.atMs).toBe(NOW - 3 * HOUR); + expect(decide(input({ record: rec, shell: stale }))).toMatchObject({ + type: "fire", + kind: "wake_lost", + }); + }); + + it("11j. a wake past the deadline is not deferred to — T3 paces on its own clock", () => { + // A recurring `0 9 * * *` recorded against an earlier deadline... + const recurring = record({ + deadlineAtMs: NOW + HOUR, + crons: crons([ + entry({ id: "c", schedule: "0 9 * * *", recurring: true, nextFireAtMs: NOW + 20 * HOUR }), + ]), + }); + expect(act({ record: recurring })).toMatchObject({ type: "fire", kind: "check_in" }); + // ...and a one-shot pinned days out. + const oneShot = record({ + deadlineAtMs: NOW + HOUR, + crons: crons([entry({ nextFireAtMs: NOW + 5 * 24 * HOUR })]), + }); + expect(act({ record: oneShot })).toMatchObject({ type: "fire", kind: "check_in" }); + expect(resolveWake(input({ record: oneShot }))?.deferrable).toBe(false); + }); + + it("11k. the grace is derived from the entry, so a jittered recurring wake is not lost", () => { + // A 30-minute wake two minutes late: tolerated, because 10% of 30 minutes is 3. + const late = record({ + crons: crons([ + entry({ schedule: "*/30 * * * *", recurring: true, nextFireAtMs: NOW - 2 * MINUTE }), + ]), + }); + expect(act({ record: late, shell: shell({ updatedAt: iso(NOW - 3 * HOUR) }) })).toMatchObject({ + reason: "self_pacing", + }); + + const graceFor = (schedule: string, recurring: boolean) => + resolveWake( + input({ + record: record({ + crons: crons([entry({ schedule, recurring, nextFireAtMs: NOW - MINUTE })]), + }), + }), + )?.graceMs; + expect(graceFor("*/30 * * * *", true)).toBe(3 * MINUTE); + expect(graceFor("*/20 * * * *", true)).toBe(2 * MINUTE); + expect(graceFor("0 */4 * * *", true)).toBe(15 * MINUTE); // the cap binds + expect(graceFor("*/10 * * * *", true)).toBe(90_000); // the floor binds + expect(graceFor("*/30 * * * *", false)).toBe(90_000); // one-shot: the flat floor + // An unparseable schedule still yields the floor rather than throwing. + expect(graceFor("every thirty minutes", true)).toBe(90_000); + expect(wakeGraceMs({ recurring: true, periodMs: 30 * MINUTE }, config)).toBe(3 * MINUTE); + }); + + it("11l. a record with the fail-closed deadline default never defers and never fires", () => { + const corrupted = record({ + deadlineAtMs: 0, + crons: crons([entry({ nextFireAtMs: NOW + MINUTE })]), + }); + const action = act({ record: corrupted }); + expect(action).toMatchObject({ type: "stop", outcome: "spent", cause: "deadline" }); + expect(action.type).not.toBe("stand_down"); + }); +}); + +describe("decide — 1.2 budget and deadline", () => { + it("12. budget remaining is allowed to fire", () => { + expect(act({ record: record({ checkInsUsed: 5, maxCheckIns: 6 }) })).toMatchObject({ + type: "fire", + checkIn: { n: 6, of: 6 }, + }); + }); + + it("13. an exhausted budget stops the run", () => { + expect(act({ record: record({ checkInsUsed: 6, maxCheckIns: 6 }) })).toMatchObject({ + type: "stop", + outcome: "spent", + cause: "budget", + }); + }); + + it("14. a passed deadline stops the run even with budget left", () => { + expect(act({ record: record({ deadlineAtMs: NOW - MINUTE, checkInsUsed: 0 }) })).toMatchObject({ + type: "stop", + outcome: "spent", + cause: "deadline", + }); + }); + + it("15. a deadline that did not survive a write means over, never unbounded", () => { + expect(act({ record: record({ deadlineAtMs: 0 }) })).toMatchObject({ + type: "stop", + outcome: "spent", + }); + }); + + it("15b. the deadline stops the loop while the thread is busy", () => { + // Guard 4b is swept before every skip for exactly this: a thread that never goes idle + // used to walk through its own deadline indefinitely. + const busy = shell({ sessionStatus: "running", updatedAt: iso(NOW - MINUTE) }); + expect(act({ shell: busy, record: record({ deadlineAtMs: NOW - MINUTE }) })).toMatchObject({ + type: "stop", + outcome: "spent", + cause: "deadline", + }); + }); + + it("15c. the sentinel is honoured while the thread is busy", () => { + const busy = shell({ sessionStatus: "running", updatedAt: iso(NOW - MINUTE) }); + expect(act({ shell: busy, sentinelAtMs: NOW - MINUTE })).toMatchObject({ + type: "stop", + outcome: "done", + cause: "sentinel", + }); + }); + + it("16. a deadline already past at arm time stops on the first evaluation", () => { + // The route rejects this with `400 deadline_required`/a past deadline; if one reaches + // the table anyway it is over, not unbounded. + const armedIntoThePast = record({ + armedAtMs: NOW, + deadlineAtMs: NOW - HOUR, + lastCheckIn: null, + }); + expect(act({ record: armedIntoThePast })).toMatchObject({ type: "stop", outcome: "spent" }); + }); + + it("17. `spent` is reported as `spent`, never as `done`", () => { + const action = act({ record: record({ checkInsUsed: 6, maxCheckIns: 6 }) }); + expect(action.type).toBe("stop"); + if (action.type !== "stop") return; + expect(action.outcome).toBe("spent"); + expect(action.outcome === "done").toBe(false); + }); + + it("a bound that cannot stop the agent is not a bound: spent ends the session when wakes are pending", () => { + const pending = crons([entry({ recurring: true, nextFireAtMs: NOW + 20 * MINUTE })]); + const spent = act({ record: record({ deadlineAtMs: NOW - 1, crons: pending }) }); + expect(spent).toMatchObject({ outcome: "spent", stopSession: true }); + // Nothing recorded means nothing to stop. + expect(act({ record: record({ deadlineAtMs: NOW - 1 }) })).toMatchObject({ + stopSession: false, + }); + // `done` leaves the session alone: the agent said it finished, and killing it would take + // any live background work with it. + expect(act({ record: record({ crons: pending }), sentinelAtMs: NOW - MINUTE })).toMatchObject({ + outcome: "done", + stopSession: false, + }); + // ...and so does a takeover, where the session is the turn the human just started. + expect( + act({ + record: record({ + crons: pending, + lastCheckIn: { firedAtMs: NOW - HOUR, createdAtIso: iso(NOW - HOUR) }, + }), + shell: shell({ latestUserMessageAt: iso(NOW - MINUTE) }), + }), + ).toMatchObject({ outcome: "handed-back", stopSession: false }); + }); +}); + +describe("decide — 1.3 strikes", () => { + const firedAtMs = NOW - HOUR; + const lastCheckIn = { firedAtMs, createdAtIso: iso(firedAtMs) }; + + it("18. movement at or above productiveMs resets strikes to zero", () => { + const action = act({ + record: record({ lastCheckIn, strikes: 1 }), + shell: shell({ updatedAt: iso(firedAtMs + config.productiveMs) }), + }); + expect(action).toMatchObject({ + type: "fire", + checkIn: { previousOutcome: "productive", strikes: 0 }, + }); + }); + + it("19. movement below productiveMs takes a strike", () => { + const action = act({ + record: record({ lastCheckIn, strikes: 0 }), + shell: shell({ updatedAt: iso(firedAtMs + 30_000) }), + }); + expect(action).toMatchObject({ + type: "fire", + checkIn: { previousOutcome: "unproductive", strikes: 1 }, + }); + }); + + it("20. two consecutive unproductive check-ins stop the run as stalled", () => { + expect( + act({ + record: record({ lastCheckIn, strikes: 1 }), + shell: shell({ updatedAt: iso(firedAtMs + 30_000) }), + }), + ).toMatchObject({ type: "stop", outcome: "stalled", cause: "strikes" }); + }); + + it("21. unproductive, productive, unproductive is still running (strikes are consecutive)", () => { + const afterProductive = act({ + record: record({ lastCheckIn, strikes: 1 }), + shell: shell({ updatedAt: iso(firedAtMs + 10 * MINUTE) }), + }); + expect(afterProductive).toMatchObject({ type: "fire", checkIn: { strikes: 0 } }); + const afterReset = act({ + record: record({ lastCheckIn, strikes: 0 }), + shell: shell({ updatedAt: iso(firedAtMs + 30_000) }), + }); + expect(afterReset).toMatchObject({ type: "fire", checkIn: { strikes: 1 } }); + }); + + it("22. strikes are only judged once a check-in exists to judge against", () => { + const action = act({ record: record({ lastCheckIn: null, strikes: 1 }) }); + expect(action).toMatchObject({ + type: "fire", + checkIn: { previousOutcome: "unknown", strikes: 1 }, + }); + // Nor can an unreadable updatedAt manufacture a strike. + expect( + judgeProgress( + input({ record: record({ lastCheckIn }), shell: shell({ updatedAt: "junk" }) }), + ), + ).toEqual({ outcome: "unknown", strikes: 0 }); + expect(judgeProgress(input({ record: record({ lastCheckIn }), shell: null }))).toEqual({ + outcome: "unknown", + strikes: 0, + }); + }); + + it("a durable two-strike record stops on the sweep, before anything can skip", () => { + expect( + act({ + record: record({ strikes: 2, rateLimitedUntilMs: NOW + HOUR }), + }), + ).toMatchObject({ type: "stop", outcome: "stalled", cause: "strikes", stopSession: false }); + }); +}); + +describe("decide — 1.4 terminal stickiness", () => { + const stopped = { reason: "done", atMs: NOW - HOUR, detail: "sentinel" } as const; + + it("23. a stopped record never fires, whatever else is true", () => { + expect( + act({ + record: record({ stopped, checkInsUsed: 0, deadlineAtMs: NOW + 8 * HOUR }), + shell: shell({ updatedAt: iso(NOW - 8 * HOUR) }), + }), + ).toMatchObject({ type: "stand_down", reason: "stopped", phase: "off" }); + }); + + it("24. a terminal state survives a decision pass unchanged", () => { + const rec = record({ stopped }); + decide(input({ record: rec })); + expect(rec.stopped).toEqual(stopped); + expect(rec.armed).toBe(true); + }); + + it("25. a re-armed record fires again: the table keeps no memory of the last run", () => { + const rearmed = record({ + stopped: null, + checkInsUsed: 0, + strikes: 0, + armedAtMs: NOW - MINUTE, + lastCheckIn: null, + }); + expect(act({ record: rearmed })).toMatchObject({ + type: "fire", + checkIn: { n: 1, of: 6, firedAtMs: NOW }, + }); + }); +}); + +describe("decide — 1.5 handback", () => { + const firedAtMs = NOW - HOUR; + const lastCheckIn = { firedAtMs, createdAtIso: iso(firedAtMs) }; + + it("26. a user message newer than our nudge hands the loop back", () => { + expect( + act({ + record: record({ lastCheckIn }), + shell: shell({ latestUserMessageAt: iso(NOW - MINUTE) }), + }), + ).toMatchObject({ type: "stop", outcome: "handed-back", cause: "takeover" }); + }); + + it("27. our own minted createdAt is not a takeover (exact string compare)", () => { + expect( + act({ + record: record({ lastCheckIn }), + shell: shell({ latestUserMessageAt: lastCheckIn.createdAtIso }), + }), + ).toMatchObject({ type: "fire" }); + }); + + it("28. handback does not reset the budget", () => { + const rec = record({ lastCheckIn, checkInsUsed: 3, strikes: 1 }); + const action = act({ + record: rec, + shell: shell({ latestUserMessageAt: iso(NOW - MINUTE) }), + }); + expect(action).toMatchObject({ type: "stop", outcome: "handed-back" }); + expect(rec.checkInsUsed).toBe(3); + expect(rec.strikes).toBe(1); + expect(action).not.toHaveProperty("checkIn"); + }); + + it("29. armed but never fired: armedAtMs is the baseline", () => { + const rec = record({ lastCheckIn: null, armedAtMs: NOW - HOUR }); + expect( + act({ record: rec, shell: shell({ latestUserMessageAt: iso(NOW - MINUTE) }) }), + ).toMatchObject({ outcome: "handed-back" }); + expect( + act({ record: rec, shell: shell({ latestUserMessageAt: iso(NOW - 2 * HOUR) }) }), + ).toMatchObject({ type: "fire" }); + }); +}); + +describe("decide — the remaining outcomes", () => { + it("passes a disarm straight through", () => { + expect(act({ shell: null })).toEqual({ type: "disarm", reason: "thread_gone" }); + }); + + it("resolveTrigger reports neutral facts for a thread that is gone", () => { + expect(resolveTrigger(input({ shell: null }))).toEqual({ + idleForMs: 0, + thresholdMs: 15 * MINUTE, + busyTurn: false, + wake: null, + }); + }); + + it("carries the keep-active pin repair and the reservation the reactor persists", () => { + expect(act({ shell: shell({ settledOverride: "active" }) })).toEqual({ + type: "fire", + kind: "check_in", + repairPin: true, + degrade: null, + checkIn: { n: 2, of: 6, firedAtMs: NOW, strikes: 0, previousOutcome: "unknown" }, + }); + }); +}); diff --git a/apps/server/src/coil/loop/decide.ts b/apps/server/src/coil/loop/decide.ts new file mode 100644 index 000000000000..455d7f0f23ee --- /dev/null +++ b/apps/server/src/coil/loop/decide.ts @@ -0,0 +1,285 @@ +/** + * The loop decision table, pure. + * + * `decide` takes the durable record, a thread shell, the clock as a number and a handful of + * facts the reactor already holds, and returns the single action to execute. Being pure is + * the most important structural choice in the feature: the whole table tests without a + * server, a clock or a provider, which is the only reason a design this full of orderings + * can be trusted. + * + * It reads. It never writes. Nothing here mutates the record — the reactor persists what the + * returned action asks for, and every stand-down asks for nothing, which is what makes + * "a skip never spends budget" a property rather than a promise. + * + * ## The trigger (§4) + * + * ``` + * idleForMs = now - max(Date.parse(shell.updatedAt), processStartedAtMs) + * busyTurn = session.status ∈ {running, starting} || latestTurn.state === "running" + * || backgroundLiveness != null + * threshold = busyTurn ? record.busyIdleMs : record.idleMs + * + * wake = earliest recorded nextFireAtMs + * deferrable = wake <= record.deadlineAtMs + * fire when idleForMs >= threshold && !(deferrable && now < wake + grace(entry)) + * ``` + * + * Both boundaries are inclusive: at exactly the threshold it fires, and at exactly + * `wake + grace` the wake counts as lost. + * + * **`session.status` never vetoes a fire — it only lengthens the fuse.** Gating on it + * deadlocks the exact threads this feature is for: the session reaper skips any binding whose + * thread still has an `activeTurnId`, so a turn whose completion never arrives pins `running` + * with nothing automated to clear it. `backgroundLiveness` is read the same way and for a + * stricter reason — it is in-memory and empty after a restart, which is the exact gap this + * feature closes, so a veto on it would fail silent precisely when supervision matters. + * + * On a healthy self-pacing thread this should almost never fire. How rarely it fires is the + * measure of a correct implementation. + * + * @module coil/loop/decide + */ + +import { wakeGraceMs } from "./config.ts"; +import { periodMsAfter } from "./cron/parse.ts"; +import { evaluateGuards, hasPendingCrons, isClaudeThread, isoMs, STRIKE_LIMIT } from "./guards.ts"; +import type { CronEntry, LoopRecord } from "./state.ts"; +import type { + CheckInOutcome, + LoopAction, + LoopDecisionInput, + LoopThreadShell, + ResolvedWake, + StopAction, + StopOutcome, + TriggerFacts, +} from "./types.ts"; + +/** + * When the thread last moved, from `updatedAt` alone. + * + * `updatedAt` and nothing else, because it is a SQL projection column rather than a hot + * stream: background subagent work lands there as activity appends, and it is the only + * signal in the field that survives a mid-loop server restart. + * + * Unparseable reads as *now* — idle 0, no fire. A timestamp we cannot read is not evidence + * the thread is stale, and `NaN` arithmetic would silently compare false in both directions. + */ +function movedAtMs(shell: LoopThreadShell, nowMs: number): number { + return isoMs(shell.updatedAt) ?? nowMs; +} + +/** `running` and `starting` lengthen the fuse. They never veto. */ +function isBusyTurn(shell: LoopThreadShell): boolean { + const status = shell.session?.status; + if (status === "running" || status === "starting") return true; + if (shell.latestTurn?.state === "running") return true; + return typeof shell.backgroundLiveness === "string"; +} + +/** + * The provider's cron table is in-process, so the recorded snapshot only describes a live + * session. Once the session is gone or stopped the entries describe wakes that can no longer + * fire, and deferring to them would stand supervision down forever. + */ +function cronsAreLive(shell: LoopThreadShell): boolean { + const session = shell.session; + if (session === null) return false; + return session.status !== "stopped"; +} + +/** + * Newest entry per id. + * + * A re-arm records the same cron twice in one snapshot; the later write is the current one, + * and without this the older copy could win the `earliest wake` comparison and defer to a + * fire time that was already superseded. + */ +function newestPerId(entries: ReadonlyArray): ReadonlyArray { + const byId = new Map(); + for (const entry of entries) byId.set(entry.id, entry); + return [...byId.values()]; +} + +/** + * The wake T3 defers to: the earliest recorded `nextFireAtMs`. + * + * `null` at every step means *no deference from that entry* — an unparseable schedule, a + * non-Claude thread, a dead session or a record that never observed the hook must never + * stand supervision down. Deference is bounded by the run's own deadline, because + * `CronCreate` takes an unbounded 5-field expression: a recorded `0 9 * * *` would otherwise + * stand supervision down for 24 hours, and a one-shot pinned days out indefinitely, all + * while the run is nominally armed. Past the deadline there is nothing left to defer to. + * + * ## A wake that already landed is not a candidate + * + * The snapshot describes a table, not a queue, and it is only refreshed when a `Stop` hook + * lands. So the earliest entry is routinely one that has already fired — and taking it + * anyway answers "the wake landed, nothing to defer to" while a *second*, still-pending + * entry sits behind it. One dropped `Stop` (a timeout, a teardown, a restart) and T3 would + * nudge a thread whose own next wake is hours away, which is precisely the deference this + * guard exists to give. Landed entries are therefore skipped in the selection, and the + * earliest of what remains is the wake T3 measures itself against. + */ +export function resolveWake(input: LoopDecisionInput): ResolvedWake | null { + const { record, shell } = input; + // Liveness before provenance: a dead session's entries describe wakes that can no longer + // fire whatever wrote them. + if (shell === null || !cronsAreLive(shell) || !isClaudeThread(shell)) return null; + const crons = record.crons; + if (crons === null) return null; + + // Raw `updatedAt`, deliberately NOT floored by `processStartedAtMs` — see `landed` below. + const movedAtMsValue = movedAtMs(shell, input.nowMs); + let best: { readonly entry: CronEntry; readonly atMs: number } | null = null; + for (const entry of newestPerId(crons.entries)) { + const atMs = entry.nextFireAtMs; + if (atMs === null) continue; + // Already landed: the thread moved after this wake, so it is history, not a commitment. + if (movedAtMsValue > atMs) continue; + if (best === null || atMs < best.atMs) best = { entry, atMs }; + } + if (best === null) return null; + + const { entry, atMs } = best; + // The period the derived grace scales with, measured from the wake itself so a schedule + // recorded hours ago still yields its own cadence rather than the cadence from `now`. + const periodMs = entry.recurring ? periodMsAfter(entry.schedule, atMs) : null; + return { + cronId: entry.id, + atMs, + graceMs: wakeGraceMs({ recurring: entry.recurring, periodMs }, input.config), + deferrable: atMs <= record.deadlineAtMs, + // False by construction now that the selection skips landed entries, and still computed + // rather than hardcoded so `wakeIsPending` / `wakeIsLost` keep reading the fact instead + // of an assumption. Raw `updatedAt`, deliberately NOT floored by `processStartedAtMs`: + // the boot clamp would read a restart as the wake having landed, which is the one case + // this signal exists to catch. + landed: movedAtMsValue > atMs, + }; +} + +/** The §4 arithmetic, computed once and handed to guards 10b, 11 and 12. */ +export function resolveTrigger(input: LoopDecisionInput): TriggerFacts { + const { record, shell, nowMs } = input; + if (shell === null) { + return { idleForMs: 0, thresholdMs: record.idleMs, busyTurn: false, wake: null }; + } + const busyTurn = isBusyTurn(shell); + // The boot-grace floor. Without it every armed thread fires simultaneously on the first + // post-restart tick; it also covers laptop sleep and the restart-continuation window. + const lastActivityMs = Math.max(movedAtMs(shell, nowMs), input.processStartedAtMs); + return { + // Clamped at 0 so clock skew reads as "just moved" rather than a negative idle. + idleForMs: Math.max(0, nowMs - lastActivityMs), + thresholdMs: busyTurn ? record.busyIdleMs : record.idleMs, + busyTurn, + wake: resolveWake(input), + }; +} + +/** + * How the previous check-in turned out, and the strike count that follows from it. + * + * Movement is `updatedAt` advancing past the moment we nudged, measured on the raw + * projection value rather than the boot-clamped one — after a restart the clamp would credit + * the agent with work it never did. Strikes are consecutive, not cumulative: any productive + * check-in resets them to zero, so unproductive → productive → unproductive is still running. + */ +export function judgeProgress(input: LoopDecisionInput): { + readonly outcome: CheckInOutcome; + readonly strikes: number; +} { + const { record, shell, config } = input; + const last = record.lastCheckIn; + const updatedAtMs = shell === null ? null : isoMs(shell.updatedAt); + if (last === null || updatedAtMs === null) { + return { outcome: "unknown", strikes: record.strikes }; + } + if (updatedAtMs - last.firedAtMs >= config.productiveMs) { + return { outcome: "productive", strikes: 0 }; + } + return { outcome: "unproductive", strikes: record.strikes + 1 }; +} + +/** + * Whether ending the run also has to end the provider session. + * + * Only when the run is over with the agent still live and its own wakes still pending: + * `spent` (a deadline or an exhausted budget) and `stalled`. Not `done` — the agent said it + * finished and killing the session would take any live background work with it — and + * emphatically not `handed-back`, where the human is at the keyboard and the session we would + * kill is the turn they just started. + */ +function stopsSession(outcome: StopOutcome, record: LoopRecord, nowMs: number): boolean { + if (outcome !== "spent" && outcome !== "stalled") return false; + return hasPendingCrons(record, nowMs); +} + +/** + * The one entry point. + * + * Guards decide *whether*; this decides *what*, and adds the two things a guard cannot see: + * the strike projection that turns a second consecutive dead check-in into `stalled` before + * the nudge rather than after it, and the reservation the reactor persists before it + * dispatches. + */ +export function decide(input: LoopDecisionInput): LoopAction { + const outcome = evaluateGuards({ ...input, trigger: resolveTrigger(input) }); + + switch (outcome.kind) { + case "stand_down": + return { + type: "stand_down", + guard: outcome.guard, + reason: outcome.reason, + phase: outcome.phase, + untilMs: outcome.untilMs, + }; + case "disarm": + return { type: "disarm", reason: outcome.reason }; + case "stop": + return toStop(outcome.outcome, outcome.cause, outcome.detail, input); + case "fire": { + const progress = judgeProgress(input); + if (progress.strikes >= STRIKE_LIMIT) { + return toStop( + "stalled", + "strikes", + `${progress.strikes} consecutive unproductive check-ins`, + input, + ); + } + return { + type: "fire", + kind: outcome.degrade === "wake_lost" ? "wake_lost" : "check_in", + repairPin: outcome.repairPin, + degrade: outcome.degrade, + checkIn: { + // Reserved before dispatch: a provider that cannot spawn burns budget instead of + // tight-looping. Six attempts, not four hundred and eighty a night. + n: input.record.checkInsUsed + 1, + of: input.record.maxCheckIns, + firedAtMs: input.nowMs, + strikes: progress.strikes, + previousOutcome: progress.outcome, + }, + }; + } + } +} + +function toStop( + outcome: StopOutcome, + cause: StopAction["cause"], + detail: string, + input: LoopDecisionInput, +): StopAction { + return { + type: "stop", + outcome, + cause, + detail, + stopSession: stopsSession(outcome, input.record, input.nowMs), + }; +} diff --git a/apps/server/src/coil/loop/guards.test.ts b/apps/server/src/coil/loop/guards.test.ts new file mode 100644 index 000000000000..bba640544c63 --- /dev/null +++ b/apps/server/src/coil/loop/guards.test.ts @@ -0,0 +1,671 @@ +// @effect-diagnostics globalDate:off -- `iso` is a pure ms->ISO fixture helper anchored on a +// fixed constant, never a wall-clock reading, so DateTime's effectful now-semantics would add +// ceremony without adding correctness. +/** + * TESTS.md §2, cases 30–46: the guard table's ordering and the *kind* of block each guard + * produces. Case numbers are cited in the test names so a number quoted anywhere in the + * design still names the same test. Retired guards (5, 13) keep their numbers reserved. + */ + +import { describe, expect, it } from "vite-plus/test"; + +import { resolveConfig } from "./config.ts"; +import { + atArmedCeiling, + blockingRequest, + checkInFloorMet, + doneSignal, + evaluateGuards, + hasPendingCrons, + idleThresholdMet, + isArchived, + isArmed, + isoMs, + isRateLimited, + isSnoozed, + isStopped, + masterToggleOff, + needsPinRepair, + stopCondition, + STRIKE_LIMIT, + tookOver, + wakeIsLost, + wakeIsPending, +} from "./guards.ts"; +import { DEFAULT_GLOBAL_SETTINGS, EMPTY_RECORD, type LoopRecord } from "./state.ts"; +import type { LoopGuardInput, LoopThreadShell, ResolvedWake, TriggerFacts } from "./types.ts"; + +const config = resolveConfig({}); // idle 15m, busy 45m, productive 2m, grace 90s..15m @10% +const NOW = 1_800_000_000_000; // 2027-01-15T08:00:00Z +const MINUTE = 60_000; +const HOUR = 60 * MINUTE; +const iso = (ms: number) => new Date(ms).toISOString(); + +const globalSettings = (o: Partial = {}) => ({ + ...DEFAULT_GLOBAL_SETTINGS, + enabled: true, + ...o, +}); + +/** An armed, healthy, mid-run record. Overrides are the point of each case. */ +const record = (o: Partial = {}): LoopRecord => + Object.freeze({ + ...EMPTY_RECORD, + armed: true, + armedAtMs: NOW - 2 * HOUR, + maxCheckIns: 6, + checkInsUsed: 1, + deadlineAtMs: NOW + 4 * HOUR, + ...o, + }); + +type ShellOverrides = { + updatedAt?: string; + archivedAt?: string | null; + settledOverride?: "settled" | "active" | null; + snoozedUntil?: string | null; + sessionStatus?: string | null; + providerName?: string | null; + latestUserMessageAt?: string | null; + hasPendingApprovals?: boolean; + hasPendingUserInput?: boolean; + hasActionableProposedPlan?: boolean; +}; + +const shell = (o: ShellOverrides = {}): LoopThreadShell => + ({ + updatedAt: o.updatedAt ?? iso(NOW - 20 * MINUTE), + archivedAt: o.archivedAt ?? null, + settledOverride: o.settledOverride ?? null, + snoozedUntil: o.snoozedUntil ?? null, + session: + o.sessionStatus === null + ? null + : { + threadId: "thread-1", + status: o.sessionStatus ?? "ready", + providerName: o.providerName === undefined ? "claudeAgent" : o.providerName, + runtimeMode: "local", + activeTurnId: null, + lastError: null, + updatedAt: iso(NOW - 20 * MINUTE), + }, + latestTurn: null, + latestUserMessageAt: o.latestUserMessageAt ?? null, + hasPendingApprovals: o.hasPendingApprovals ?? false, + hasPendingUserInput: o.hasPendingUserInput ?? false, + hasActionableProposedPlan: o.hasActionableProposedPlan ?? false, + }) as unknown as LoopThreadShell; + +/** Defaults to "the thread is stale enough to nudge", so each case perturbs one thing. */ +const trigger = (o: Partial = {}): TriggerFacts => ({ + idleForMs: 20 * MINUTE, + thresholdMs: 15 * MINUTE, + busyTurn: false, + wake: null, + ...o, +}); + +const wake = (o: Partial = {}): ResolvedWake => ({ + cronId: "cron-1", + atMs: NOW + 10 * MINUTE, + graceMs: 90_000, + deferrable: true, + landed: false, + ...o, +}); + +const guardInput = (o: Partial = {}): LoopGuardInput => ({ + nowMs: NOW, + processStartedAtMs: NOW - 24 * HOUR, + record: record(), + global: globalSettings(), + shell: shell(), + sentinelAtMs: null, + loopDoneAtMs: null, + autoResumePending: false, + armedCount: 1, + config, + trigger: trigger(), + ...o, +}); + +describe("evaluateGuards — the blocking guards", () => { + it("30. the master toggle stands the loop down without disarming or stopping it", () => { + const rec = record(); + const outcome = evaluateGuards( + guardInput({ record: rec, global: globalSettings({ enabled: false }) }), + ); + expect(outcome).toEqual({ + kind: "stand_down", + guard: "2", + reason: "disabled", + phase: "standing_down", + untilMs: null, + }); + expect(rec.armed).toBe(true); + expect(rec.stopped).toBeNull(); + expect(rec.checkInsUsed).toBe(1); + }); + + it("31. an unarmed record stands down, and a stopped one says so instead", () => { + expect(evaluateGuards(guardInput({ record: record({ armed: false }) }))).toMatchObject({ + kind: "stand_down", + guard: "3", + reason: "not_armed", + phase: "off", + }); + // Terminal states are sticky and report themselves, even on a record still flagged armed. + expect( + evaluateGuards( + guardInput({ + record: record({ stopped: { reason: "done", atMs: NOW - HOUR, detail: "" } }), + }), + ), + ).toMatchObject({ kind: "stand_down", guard: "3", reason: "stopped" }); + }); + + it("32. a thread that is gone disarms rather than skipping", () => { + expect(evaluateGuards(guardInput({ shell: null }))).toEqual({ + kind: "disarm", + guard: "4", + reason: "thread_gone", + }); + }); + + it("33. an archived thread disarms", () => { + expect(evaluateGuards(guardInput({ shell: shell({ archivedAt: iso(NOW - HOUR) }) }))).toEqual({ + kind: "disarm", + guard: "4", + reason: "archived", + }); + }); + + // Guard 5 is retired. `settledOverride` no longer distinguishes a human from a timer: + // ThreadSettlementReactor sweeps every minute and emits the same event with no provenance. + it("34. settledness never blocks a check-in (guard 5 is retired)", () => { + expect(evaluateGuards(guardInput({ shell: shell({ settledOverride: "settled" }) }))).toEqual({ + kind: "fire", + repairPin: false, + degrade: null, + }); + }); + + it("34b. a loop auto-settled mid-run still checks in", () => { + // The failure the retirement prevents: with the old guard this sat armed doing nothing + // until its deadline and then reported `spent` — a skip never stops, so it failed silent. + const settledByTimer = shell({ settledOverride: "settled", updatedAt: iso(NOW - 3 * HOUR) }); + const outcome = evaluateGuards( + guardInput({ shell: settledByTimer, trigger: trigger({ idleForMs: 3 * HOUR }) }), + ); + expect(outcome.kind).toBe("fire"); + }); + + it("35. a live snooze stands the loop down and reports when it lifts", () => { + const until = NOW + 30 * MINUTE; + expect(evaluateGuards(guardInput({ shell: shell({ snoozedUntil: iso(until) }) }))).toEqual({ + kind: "stand_down", + guard: "6", + reason: "snoozed", + // `held`, the same word `status.ts` and the route's `derived` use for it: a snooze is a + // bounded hold with an expiry, not a thread that is being watched. + phase: "held", + untilMs: until, + }); + }); + + it("36. an expired, absent or unreadable snooze passes", () => { + expect( + evaluateGuards(guardInput({ shell: shell({ snoozedUntil: iso(NOW - MINUTE) }) })).kind, + ).toBe("fire"); + // Exactly at the expiry the snooze is over. An inclusive boundary here would hold the loop + // for one more poll on the tick the user's snooze was supposed to end. + expect(evaluateGuards(guardInput({ shell: shell({ snoozedUntil: iso(NOW) }) })).kind).toBe( + "fire", + ); + expect(evaluateGuards(guardInput({ shell: shell({ snoozedUntil: null }) })).kind).toBe("fire"); + expect(evaluateGuards(guardInput({ shell: shell({ snoozedUntil: "not-a-date" }) })).kind).toBe( + "fire", + ); + }); + + it("37. a pending approval blocks", () => { + expect(evaluateGuards(guardInput({ shell: shell({ hasPendingApprovals: true }) }))).toEqual({ + kind: "stand_down", + guard: "8", + reason: "pending_approval", + phase: "blocked", + untilMs: null, + }); + }); + + it("38. a pending user-input request blocks", () => { + expect( + evaluateGuards(guardInput({ shell: shell({ hasPendingUserInput: true }) })), + ).toMatchObject({ guard: "8", reason: "pending_user_input" }); + }); + + // The clause every design in the original panel missed: Sidebar.logic.ts treats + // plan-ready as NOT pending-input, so a thread parked on an unapproved plan otherwise + // passes every blocking guard and gets pushed past the human's yes. + it("39. an actionable proposed plan blocks", () => { + expect( + evaluateGuards(guardInput({ shell: shell({ hasActionableProposedPlan: true }) })), + ).toMatchObject({ guard: "8", reason: "pending_plan" }); + }); + + it("40. none of the three blocking facts set: the guard passes", () => { + expect(blockingRequest(shell())).toBeNull(); + expect(evaluateGuards(guardInput()).kind).toBe("fire"); + }); + + it("41. an armed auto-resume owns the thread; the loop stands down", () => { + expect(evaluateGuards(guardInput({ autoResumePending: true }))).toEqual({ + kind: "stand_down", + guard: "9", + reason: "auto_resume_pending", + phase: "watching", + untilMs: null, + }); + }); + + it("42. a usage limit holds the loop, and the boundary is inclusive", () => { + expect( + evaluateGuards(guardInput({ record: record({ rateLimitedUntilMs: NOW + 5 * MINUTE }) })), + ).toEqual({ + kind: "stand_down", + guard: "10", + reason: "rate_limited", + phase: "held", + untilMs: NOW + 5 * MINUTE, + }); + // `now >= rateLimitedUntilMs` passes: the hold is over at exactly its own deadline. + expect(evaluateGuards(guardInput({ record: record({ rateLimitedUntilMs: NOW }) })).kind).toBe( + "fire", + ); + }); + + it("10b. a pending wake inside the deadline stands the loop down as self-pacing", () => { + const outcome = evaluateGuards(guardInput({ trigger: trigger({ wake: wake() }) })); + expect(outcome).toEqual({ + kind: "stand_down", + guard: "10b", + reason: "self_pacing", + phase: "self_pacing", + untilMs: NOW + 10 * MINUTE, + }); + }); + + it("10b. a wake past its grace fires and flags the run degraded", () => { + const lost = wake({ atMs: NOW - 10 * MINUTE, graceMs: 90_000 }); + expect(evaluateGuards(guardInput({ trigger: trigger({ wake: lost }) }))).toEqual({ + kind: "fire", + repairPin: false, + degrade: "wake_lost", + }); + }); + + it("43. the check-in floor holds even when the idle threshold appears met", () => { + const rec = record({ lastCheckIn: { firedAtMs: NOW - 5 * MINUTE, createdAtIso: iso(NOW) } }); + expect(evaluateGuards(guardInput({ record: rec }))).toEqual({ + kind: "stand_down", + guard: "11", + reason: "check_in_floor", + phase: "watching", + untilMs: NOW - 5 * MINUTE + config.idleMs, + }); + // Exactly at the floor it passes — the same inclusive boundary as every other threshold. + const atFloor = record({ + lastCheckIn: { firedAtMs: NOW - config.idleMs, createdAtIso: iso(NOW - config.idleMs) }, + }); + expect(evaluateGuards(guardInput({ record: atFloor })).kind).toBe("fire"); + }); + + it("12. below the staleness threshold the loop keeps watching", () => { + expect(evaluateGuards(guardInput({ trigger: trigger({ idleForMs: MINUTE }) }))).toEqual({ + kind: "stand_down", + guard: "12", + reason: "not_idle", + phase: "watching", + untilMs: null, + }); + }); + + it("44. the machine-wide ceiling is re-checked per tick, and counts the OTHER loops", () => { + // Four armed against a ceiling of three: this loop is the one over the line. + expect(evaluateGuards(guardInput({ armedCount: 4 }))).toEqual({ + kind: "stand_down", + guard: "14", + reason: "ceiling", + phase: "standing_down", + untilMs: null, + }); + // Exactly at the ceiling, every one of the three still fires. Counting yourself here made + // the ceiling refuse the very population it was sized for: with three armed and a limit of + // three, all three stood down and nothing was supervised at all. + expect(evaluateGuards(guardInput({ armedCount: 3 })).kind).toBe("fire"); + expect(evaluateGuards(guardInput({ armedCount: 2 })).kind).toBe("fire"); + }); + + it("7. the keep-active pin rides out on the fire so the reactor can repair it", () => { + expect(evaluateGuards(guardInput({ shell: shell({ settledOverride: "active" }) }))).toEqual({ + kind: "fire", + repairPin: true, + degrade: null, + }); + }); +}); + +describe("evaluateGuards — ordering", () => { + it("45. a record tripping several guards reports the first, because that string is rendered", () => { + const outcome = evaluateGuards( + guardInput({ + global: globalSettings({ enabled: false }), + record: record({ armed: false, rateLimitedUntilMs: NOW + HOUR }), + shell: shell({ archivedAt: iso(NOW), hasPendingApprovals: true }), + autoResumePending: true, + armedCount: 99, + }), + ); + expect(outcome).toMatchObject({ guard: "2", reason: "disabled" }); + }); + + it("45b. guard 4b is swept before every skip: past-deadline-and-held reports the stop", () => { + // "The loop is held" and "the loop is over" are different words on the console, and the + // wrong one hides a finished run behind a hold. + const outcome = evaluateGuards( + guardInput({ + record: record({ deadlineAtMs: NOW - MINUTE, rateLimitedUntilMs: NOW + HOUR }), + }), + ); + expect(outcome).toMatchObject({ + kind: "stop", + guard: "4b", + outcome: "spent", + cause: "deadline", + }); + }); + + it("45c. guard 4b runs after guard 4: a deleted thread disarms, it does not report spent", () => { + expect( + evaluateGuards(guardInput({ shell: null, record: record({ deadlineAtMs: NOW - MINUTE }) })), + ).toEqual({ kind: "disarm", guard: "4", reason: "thread_gone" }); + }); + + it("45d. guard 2 precedes 4b: the toggle stands loops down, it never manufactures a stop", () => { + expect( + evaluateGuards( + guardInput({ + global: globalSettings({ enabled: false }), + record: record({ deadlineAtMs: NOW - MINUTE }), + }), + ), + ).toMatchObject({ kind: "stand_down", guard: "2", reason: "disabled" }); + }); + + // A future guard added without the non-consuming property fails here rather than in + // production at 3am. + it("46. no stand-down touches the budget, across every skipping guard", () => { + const cases: ReadonlyArray<{ readonly name: string; readonly input: LoopGuardInput }> = [ + { name: "2 disabled", input: guardInput({ global: globalSettings({ enabled: false }) }) }, + { name: "3 not armed", input: guardInput({ record: record({ armed: false }) }) }, + { + name: "3 stopped", + input: guardInput({ + record: record({ stopped: { reason: "spent", atMs: NOW, detail: "" } }), + }), + }, + { + name: "6 snoozed", + input: guardInput({ shell: shell({ snoozedUntil: iso(NOW + HOUR) }) }), + }, + { + name: "8 approval", + input: guardInput({ shell: shell({ hasPendingApprovals: true }) }), + }, + { + name: "8 user input", + input: guardInput({ shell: shell({ hasPendingUserInput: true }) }), + }, + { + name: "8 plan", + input: guardInput({ shell: shell({ hasActionableProposedPlan: true }) }), + }, + { name: "9 auto-resume", input: guardInput({ autoResumePending: true }) }, + { + name: "10 rate limited", + input: guardInput({ record: record({ rateLimitedUntilMs: NOW + HOUR }) }), + }, + { name: "10b self-pacing", input: guardInput({ trigger: trigger({ wake: wake() }) }) }, + { + name: "11 check-in floor", + input: guardInput({ + record: record({ lastCheckIn: { firedAtMs: NOW - MINUTE, createdAtIso: iso(NOW) } }), + }), + }, + { name: "12 not idle", input: guardInput({ trigger: trigger({ idleForMs: 0 }) }) }, + { name: "14 ceiling", input: guardInput({ armedCount: 5 }) }, + ]; + + for (const { name, input } of cases) { + const before = { used: input.record.checkInsUsed, strikes: input.record.strikes }; + const outcome = evaluateGuards(input); + expect(outcome.kind, name).toBe("stand_down"); + // The record is frozen, so a mutating guard would have thrown above; this asserts the + // decision itself carries no budget change either. + expect({ used: input.record.checkInsUsed, strikes: input.record.strikes }, name).toEqual( + before, + ); + } + // Every reason in the union is exercised except the ones only `decide` can reach. + expect(cases).toHaveLength(13); + }); +}); + +describe("stopCondition — the 4b sweep", () => { + const sweep = (o: Partial = {}) => { + const input = guardInput(o); + return stopCondition({ ...input, shell: input.shell ?? shell() }); + }; + + it("reports nothing on a healthy mid-run record", () => { + expect(sweep()).toBeNull(); + }); + + it("14/17. a passed deadline is `spent`, never `done`, even with budget left", () => { + const stop = sweep({ record: record({ deadlineAtMs: NOW - 1, checkInsUsed: 0 }) }); + expect(stop?.outcome).toBe("spent"); + expect(stop?.cause).toBe("deadline"); + }); + + it("13. an exhausted budget is `spent`", () => { + expect(sweep({ record: record({ checkInsUsed: 6, maxCheckIns: 6 }) })).toMatchObject({ + outcome: "spent", + cause: "budget", + }); + }); + + it("a fresh done-file stops the run as done; a stale one is a leftover", () => { + expect(sweep({ sentinelAtMs: NOW - MINUTE })).toMatchObject({ + outcome: "done", + cause: "sentinel", + }); + // Older than armedAtMs: a file from a previous run is not this run's signal. + expect(sweep({ sentinelAtMs: NOW - 3 * HOUR })).toBeNull(); + }); + + it("the loop_done call is equivalent to the file, and the newer of the two wins", () => { + expect(sweep({ loopDoneAtMs: NOW - MINUTE })).toMatchObject({ cause: "loop_done" }); + expect(sweep({ sentinelAtMs: NOW - MINUTE, loopDoneAtMs: NOW - 2 * MINUTE })).toMatchObject({ + cause: "sentinel", + }); + expect(sweep({ sentinelAtMs: NOW - 2 * MINUTE, loopDoneAtMs: NOW - MINUTE })).toMatchObject({ + cause: "loop_done", + }); + }); + + it("a run that finishes on its LAST check-in reports done, not spent", () => { + // The budget is exhausted the moment the final check-in is reserved, so with the bounds + // swept first every successful run that used its whole budget was recorded as "out of + // rope" — and `spent` also ends the provider session, which `done` deliberately does not. + expect( + sweep({ + record: record({ checkInsUsed: 6, maxCheckIns: 6 }), + loopDoneAtMs: NOW - MINUTE, + }), + ).toMatchObject({ outcome: "done", cause: "loop_done" }); + expect( + sweep({ record: record({ checkInsUsed: 6, maxCheckIns: 6 }), sentinelAtMs: NOW - MINUTE }), + ).toMatchObject({ outcome: "done", cause: "sentinel" }); + // Same on the deadline: the agent said it finished, and it did. + expect( + sweep({ record: record({ deadlineAtMs: NOW - 1 }), sentinelAtMs: NOW - MINUTE }), + ).toMatchObject({ outcome: "done", cause: "sentinel" }); + // A STALE signal is still no signal: the bounds win, exactly as before. + expect( + sweep({ record: record({ checkInsUsed: 6, maxCheckIns: 6 }), sentinelAtMs: NOW - 3 * HOUR }), + ).toMatchObject({ outcome: "spent", cause: "budget" }); + }); + + it("two strikes on the durable record stop the run as stalled", () => { + expect(sweep({ record: record({ strikes: STRIKE_LIMIT }) })).toMatchObject({ + outcome: "stalled", + cause: "strikes", + }); + expect(sweep({ record: record({ strikes: STRIKE_LIMIT - 1 }) })).toBeNull(); + }); + + it("takeover is swept last, so a run already over reports why T3 ended it", () => { + const takenOver = shell({ latestUserMessageAt: iso(NOW - MINUTE) }); + expect(sweep({ shell: takenOver })).toMatchObject({ + outcome: "handed-back", + cause: "takeover", + }); + expect(sweep({ shell: takenOver, record: record({ deadlineAtMs: NOW - 1 }) })).toMatchObject({ + outcome: "spent", + cause: "deadline", + }); + }); +}); + +describe("guard predicates", () => { + it("isoMs reads a timestamp, or reports that it could not", () => { + expect(isoMs(iso(NOW))).toBe(NOW); + expect(isoMs("nonsense")).toBeNull(); + expect(isoMs(null)).toBeNull(); + expect(isoMs(undefined)).toBeNull(); + }); + + it("masterToggleOff, isStopped, isArmed and isArchived read one field each", () => { + expect(masterToggleOff(globalSettings({ enabled: false }))).toBe(true); + expect(masterToggleOff(globalSettings())).toBe(false); + expect(isStopped(record())).toBe(false); + expect(isStopped(record({ stopped: { reason: "done", atMs: NOW, detail: "" } }))).toBe(true); + expect(isArmed(record())).toBe(true); + expect(isArmed(record({ armed: false }))).toBe(false); + expect(isArchived(shell())).toBe(false); + expect(isArchived(shell({ archivedAt: iso(NOW) }))).toBe(true); + }); + + it("isSnoozed, needsPinRepair, isRateLimited and atArmedCeiling", () => { + expect(isSnoozed(shell({ snoozedUntil: iso(NOW + 1) }), NOW)).toBe(true); + // Exactly now is NOT snoozed: `snoozedUntil` is the instant the snooze lifts, so the + // boundary is exclusive and a snooze that has just expired stops holding the loop back. + expect(isSnoozed(shell({ snoozedUntil: iso(NOW) }), NOW)).toBe(false); + expect(isSnoozed(shell({ snoozedUntil: iso(NOW - 1) }), NOW)).toBe(false); + expect(needsPinRepair(shell({ settledOverride: "active" }))).toBe(true); + expect(needsPinRepair(shell({ settledOverride: "settled" }))).toBe(false); + expect(isRateLimited(record({ rateLimitedUntilMs: NOW + 1 }), NOW)).toBe(true); + expect(isRateLimited(record(), NOW)).toBe(false); + // `armedCount` includes the loop being evaluated, so the ceiling is measured against the + // others: three armed against a limit of three is fine, four is one too many. + expect(atArmedCeiling(4, globalSettings({ maxArmedThreads: 3 }))).toBe(true); + expect(atArmedCeiling(3, globalSettings({ maxArmedThreads: 3 }))).toBe(false); + expect(atArmedCeiling(2, globalSettings({ maxArmedThreads: 3 }))).toBe(false); + // The only armed loop on a machine with a ceiling of one is never over it. + expect(atArmedCeiling(1, globalSettings({ maxArmedThreads: 1 }))).toBe(false); + }); + + it("the wake predicates split pending from lost at an inclusive boundary", () => { + const atMs = NOW - 10 * MINUTE; + const graceMs = 5 * MINUTE; + const pending = wake({ atMs, graceMs }); + expect(wakeIsPending(trigger({ wake: pending }), atMs + graceMs - 1)).toBe(true); + expect(wakeIsPending(trigger({ wake: pending }), atMs + graceMs)).toBe(false); + expect(wakeIsLost(trigger({ wake: pending }), atMs + graceMs)).toBe(true); + expect(wakeIsLost(trigger({ wake: pending }), atMs + graceMs - 1)).toBe(false); + // No wake, a wake past the deadline, and a wake that landed all mean "no deference". + for (const candidate of [null, wake({ deferrable: false }), wake({ landed: true })]) { + const facts = trigger({ wake: candidate }); + expect(wakeIsPending(facts, NOW)).toBe(false); + expect(wakeIsLost(facts, NOW)).toBe(false); + } + }); + + it("checkInFloorMet and idleThresholdMet are both inclusive", () => { + expect(checkInFloorMet(record(), config, NOW)).toBe(true); + const last = { firedAtMs: NOW - config.idleMs, createdAtIso: iso(NOW - config.idleMs) }; + expect(checkInFloorMet(record({ lastCheckIn: last }), config, NOW)).toBe(true); + expect(checkInFloorMet(record({ lastCheckIn: last }), config, NOW - 1)).toBe(false); + expect(idleThresholdMet(trigger({ idleForMs: 15 * MINUTE, thresholdMs: 15 * MINUTE }))).toBe( + true, + ); + expect(idleThresholdMet(trigger({ idleForMs: 15 * MINUTE - 1 }))).toBe(false); + }); + + it("tookOver compares the exact minted createdAt, then falls back to armedAtMs", () => { + const last = { firedAtMs: NOW - HOUR, createdAtIso: iso(NOW - HOUR) }; + const rec = record({ lastCheckIn: last }); + expect(tookOver(rec, shell({ latestUserMessageAt: iso(NOW - MINUTE) }))).toBe(true); + // Our own nudge: equal is not later. This is the off-by-one that would disarm every + // loop on its own first check-in. + expect(tookOver(rec, shell({ latestUserMessageAt: last.createdAtIso }))).toBe(false); + expect(tookOver(rec, shell({ latestUserMessageAt: null }))).toBe(false); + // 29. Armed but never fired: the baseline is armedAtMs. + const fresh = record({ lastCheckIn: null, armedAtMs: NOW - HOUR }); + expect(tookOver(fresh, shell({ latestUserMessageAt: iso(NOW - MINUTE) }))).toBe(true); + expect(tookOver(fresh, shell({ latestUserMessageAt: iso(NOW - 2 * HOUR) }))).toBe(false); + expect(tookOver(fresh, shell({ latestUserMessageAt: "nonsense" }))).toBe(false); + // An EMPTY `createdAtIso` is `LastCheckIn`'s decoding default, not a baseline. Comparing + // against `""` — which every ISO string sorts after — ended an armed run as handed-back on + // the first tick after a partial write, the one direction a fail-closed default must not + // fail in. It falls back to `armedAtMs` exactly as a record with no check-in does. + const partial = record({ + lastCheckIn: { firedAtMs: NOW - HOUR, createdAtIso: "" }, + armedAtMs: NOW - HOUR, + }); + expect(tookOver(partial, shell({ latestUserMessageAt: iso(NOW - 2 * HOUR) }))).toBe(false); + expect(tookOver(partial, shell({ latestUserMessageAt: iso(NOW - MINUTE) }))).toBe(true); + }); + + it("doneSignal ignores anything older than the arm", () => { + const rec = record({ armedAtMs: NOW - HOUR }); + expect(doneSignal({ record: rec, sentinelAtMs: null, loopDoneAtMs: null })).toBeNull(); + expect( + doneSignal({ record: rec, sentinelAtMs: NOW - 2 * HOUR, loopDoneAtMs: NOW - 2 * HOUR }), + ).toBeNull(); + expect(doneSignal({ record: rec, sentinelAtMs: NOW, loopDoneAtMs: null })).toEqual({ + cause: "sentinel", + atMs: NOW, + }); + }); + + it("hasPendingCrons treats a recurring entry as pending whatever its next fire", () => { + expect(hasPendingCrons(record(), NOW)).toBe(false); + const entry = (o: Record) => ({ + id: "c1", + schedule: "*/30 * * * *", + recurring: false, + prompt: "", + nextFireAtMs: null, + ...o, + }); + const withCrons = (entries: ReadonlyArray>) => + record({ crons: { recordedAtMs: NOW - HOUR, entries } }); + expect(hasPendingCrons(withCrons([]), NOW)).toBe(false); + expect(hasPendingCrons(withCrons([entry({})]), NOW)).toBe(false); + expect(hasPendingCrons(withCrons([entry({ nextFireAtMs: NOW - 1 })]), NOW)).toBe(false); + expect(hasPendingCrons(withCrons([entry({ nextFireAtMs: NOW + 1 })]), NOW)).toBe(true); + expect(hasPendingCrons(withCrons([entry({ recurring: true })]), NOW)).toBe(true); + }); +}); diff --git a/apps/server/src/coil/loop/guards.ts b/apps/server/src/coil/loop/guards.ts new file mode 100644 index 000000000000..7dec8f77ceeb --- /dev/null +++ b/apps/server/src/coil/loop/guards.ts @@ -0,0 +1,390 @@ +/** + * The loop guard table, pure. + * + * One function per guard plus `evaluateGuards`, which runs them in the design's exact order + * and returns the *first* thing that blocks. That ordering is the product: the string the + * first blocking guard returns is what the console renders, so "the loop is held" and "the + * loop is over" can never be swapped. + * + * Every guard reads plain values — `nowMs`, the durable record, a thread shell, the global + * settings, the resolved trigger facts. No store, no clock, no filesystem, no provider. + * + * ## Two orderings that are load-bearing + * + * **Guard 4b (the stop sweep) runs before every non-consuming skip.** Stop conditions are + * facts about the *run*, not about the thread's current activity. With the sweep after the + * idle guards — where it used to sit, as guard 13 — a thread that never went idle never + * reached it, so a self-paced run strolled through its own deadline indefinitely and an + * agent that wrote `.coil/loop-done` while still working was not recorded as `done` until it + * happened to go quiet. + * + * **Guard 4b runs after guard 4, and guard 2 runs before both.** A deleted thread disarms + * rather than reporting `spent`, and the master toggle stands loops down without + * manufacturing terminal states nobody chose. + * + * ## Guard 5 is retired. Do not re-add it + * + * `settledOverride !== "settled"` was a skip meaning "the human is done here". It has not + * meant that since upstream #8600 moved settlement server-side: `ThreadSettlementReactor` + * sweeps every minute and dispatches `thread.auto-settle`, which shares `thread.settle`'s + * decider case and emits the same event with **no provenance marker**. There is no + * discriminator. `autoResume/guards.ts` already paid for this once — a timer destroyed an + * armed week-long resume on day 3 — and keeping it here would have been worse, because a + * skip never *stops*: an auto-settled loop would sit armed doing nothing until its deadline + * and then report `spent`. The user's opt-out is disarm. + * + * Snooze is not in that position and guard 6 stands: there is no auto-snooze command, so + * `snoozedUntil` still carries a human's intent. The general rule this fork keeps + * re-learning: before adding any guard, ask *could a server timer write this value?* + * + * @module coil/loop/guards + */ + +import { isClaudeThread } from "../autoResume/guards.ts"; +import type { LoopConfig } from "./config.ts"; +import type { LoopGlobalSettings, LoopRecord } from "./state.ts"; +import type { + GuardId, + GuardOutcome, + GuardStandDown, + GuardStop, + LoopGuardInput, + LoopPhase, + LoopThreadShell, + StandDownReason, + StopCause, + TriggerFacts, +} from "./types.ts"; + +export { isClaudeThread }; + +/** Two *consecutive* unproductive check-ins end the run. Not cumulative. */ +export const STRIKE_LIMIT = 2; + +/** Epoch ms for an ISO timestamp, or `null` when it is absent or unparseable. */ +export function isoMs(value: string | null | undefined): number | null { + if (typeof value !== "string") return null; + const ms = Date.parse(value); + return Number.isFinite(ms) ? ms : null; +} + +/** Guard 2 — the master toggle, re-read every tick and again pre-dispatch. */ +export function masterToggleOff(global: LoopGlobalSettings): boolean { + return global.enabled !== true; +} + +/** Guard 3, first half — terminal states are sticky; only a human re-arm clears them. */ +export function isStopped(record: LoopRecord): boolean { + return record.stopped !== null; +} + +/** Guard 3, second half — nothing is supervised implicitly. */ +export function isArmed(record: LoopRecord): boolean { + return record.armed === true; +} + +/** Guard 4 — the thread is archived, a destination a check-in should not reach. */ +export function isArchived(shell: LoopThreadShell): boolean { + return shell.archivedAt !== null; +} + +/** + * Guard 6 — a snooze is a human saying "not now", and honouring it is the only reason this + * guard exists. + * + * An unparseable timestamp reads as *not* snoozed. A snooze that cannot be read is not an + * instruction, and the failing-closed reading here would be a loop that never runs again + * with nothing on the console explaining why. + */ +export function isSnoozed(shell: LoopThreadShell, nowMs: number): boolean { + const until = isoMs(shell.snoozedUntil); + return until !== null && until > nowMs; +} + +/** + * Guard 7 — not a blocker. When the pre-dispatch shell carries the keep-active pin the + * reactor nudges and *then* repairs it, because the decider clears `settledOverride` for any + * non-null value. Reported only when the pin is already there, so it can never create one. + */ +export function needsPinRepair(shell: LoopThreadShell): boolean { + return shell.settledOverride === "active"; +} + +/** + * Guard 8 — blocked on a human. + * + * The third clause is the one every design in the original panel missed: `Sidebar.logic.ts` + * treats plan-ready as *not* pending-input, so a thread parked on an unapproved plan + * otherwise passes every blocking guard and gets pushed past the human's yes. + */ +export function blockingRequest(shell: LoopThreadShell): StandDownReason | null { + if (shell.hasPendingApprovals) return "pending_approval"; + if (shell.hasPendingUserInput) return "pending_user_input"; + if (shell.hasActionableProposedPlan) return "pending_plan"; + return null; +} + +/** Guard 10 — a usage limit, held durably so it survives a restart. */ +export function isRateLimited(record: LoopRecord, nowMs: number): boolean { + return nowMs < record.rateLimitedUntilMs; +} + +/** + * Guard 10b — the deference rule. + * + * While a recorded wake is pending inside the run's deadline the agent is pacing itself and + * T3 stands by: no dispatch, no budget spent. T3 wins the moment the wake is overdue by its + * grace with no `updatedAt` movement — an unmet commitment rather than an inference, and the + * strongest trigger in the design. + */ +export function wakeIsPending(trigger: TriggerFacts, nowMs: number): boolean { + const wake = trigger.wake; + if (wake === null || !wake.deferrable || wake.landed) return false; + return nowMs < wake.atMs + wake.graceMs; +} + +/** + * A recorded wake that never landed. The boundary is inclusive: at exactly + * `nextFireAtMs + graceMs` the wake counts as lost. + */ +export function wakeIsLost(trigger: TriggerFacts, nowMs: number): boolean { + const wake = trigger.wake; + if (wake === null || !wake.deferrable || wake.landed) return false; + return nowMs >= wake.atMs + wake.graceMs; +} + +/** + * Guard 11 — the structural anti-tight-loop floor. + * + * Reads the deployment-level `config.idleMs` rather than the per-thread override, and the + * clock rather than `updatedAt`, so a tight loop stays impossible even on a thread whose + * `updatedAt` never bumps. + */ +export function checkInFloorMet(record: LoopRecord, config: LoopConfig, nowMs: number): boolean { + const last = record.lastCheckIn; + if (last === null) return true; + return nowMs - last.firedAtMs >= config.idleMs; +} + +/** Guard 12 — the staleness threshold. `>=` is inclusive: at exactly the threshold it fires. */ +export function idleThresholdMet(trigger: TriggerFacts): boolean { + return trigger.idleForMs >= trigger.thresholdMs; +} + +/** + * Guard 14 — the machine-wide ceiling, re-checked here so hand-editing the file cannot + * bypass it. + * + * `armedCount` includes the loop being evaluated, and the count that matters is of the + * **others**: with exactly `maxArmedThreads` armed, counting yourself makes every loop on + * the machine report `ceiling` and stand down, which is the ceiling refusing the very + * population it was sized for. The arm route already applies the other-loops rule; this is + * the same predicate, so the two lenses cannot disagree. + */ +export function atArmedCeiling(armedCount: number, global: LoopGlobalSettings): boolean { + return Math.max(0, armedCount - 1) >= global.maxArmedThreads; +} + +/** + * The human took the wheel. + * + * Exact string compare against our own minted `createdAt`, which is the off-by-one that + * would otherwise disarm every loop on its own first check-in: `latestUserMessageAt` equal + * to the nudge we sent is the nudge, not a takeover. Before the first check-in the baseline + * is `armedAtMs`, so a message typed after arming still counts. + * + * An **empty** `createdAtIso` is not a baseline either. It is `LastCheckIn`'s decoding + * default, so a record written by an older build or truncated mid-write would compare every + * user message against `""` — which every ISO string sorts after — and end the run as + * handed-back on the first tick after a restart. Fail-closed defaults must not fail *loudly* + * in the one direction that destroys an armed run. + */ +export function tookOver(record: LoopRecord, shell: LoopThreadShell): boolean { + const latest = shell.latestUserMessageAt; + if (typeof latest !== "string") return false; + const baselineIso = record.lastCheckIn?.createdAtIso ?? ""; + if (baselineIso !== "") return latest > baselineIso; + const latestMs = isoMs(latest); + return latestMs !== null && latestMs > record.armedAtMs; +} + +/** + * The done signal, from either channel. + * + * Freshness is the recorded mtime against `armedAtMs`, so a done-file left over from a + * previous run is a leftover rather than a signal. Newest wins when both channels fired. + */ +export function doneSignal(input: { + readonly record: LoopRecord; + readonly sentinelAtMs: number | null; + readonly loopDoneAtMs: number | null; +}): { readonly cause: Extract; readonly atMs: number } | null { + const armedAtMs = input.record.armedAtMs; + const sentinel = input.sentinelAtMs; + const called = input.loopDoneAtMs; + const sentinelFresh = sentinel !== null && sentinel > armedAtMs; + const calledFresh = called !== null && called > armedAtMs; + if (sentinelFresh && calledFresh) { + return sentinel >= called + ? { cause: "sentinel", atMs: sentinel } + : { cause: "loop_done", atMs: called }; + } + if (sentinelFresh) return { cause: "sentinel", atMs: sentinel }; + if (calledFresh) return { cause: "loop_done", atMs: called }; + return null; +} + +/** + * Recorded wakes that would still fire if the session kept running. + * + * A recurring entry counts whatever its next fire time, because it reschedules itself. This + * is what decides whether ending a run also has to end the session: T3 has no write handle + * on the binary's cron table, and a bound that cannot stop the agent is not a bound. + */ +export function hasPendingCrons(record: LoopRecord, nowMs: number): boolean { + const crons = record.crons; + if (crons === null) return false; + return crons.entries.some( + (entry) => entry.recurring || (entry.nextFireAtMs !== null && entry.nextFireAtMs > nowMs), + ); +} + +const stop = (outcome: GuardStop["outcome"], cause: StopCause, detail: string): GuardStop => ({ + kind: "stop", + guard: "4b", + outcome, + cause, + detail, +}); + +/** + * Guard 4b — the stop sweep: **done first**, then deadline, budget, strikes, takeover. + * + * ## Why done outranks the bounds + * + * `checkInsUsed` reaches `maxCheckIns` on the *final* fire, so a run that succeeds on its + * last check-in is over on the budget before the tick that would read its done signal. With + * the bounds first, every loop that finished on its last check-in reported `spent` — "it ran + * out of rope" for a run that actually finished — and `stopsSession` fires for `spent` and + * not for `done`, so T3 would also kill a session whose agent had just declared it was done + * and might still have background work in it. Reading the signal first costs nothing: the + * bounds are still swept immediately after, and the run still ends on this same tick. + * + * `deadlineAtMs: 0` and `maxCheckIns: 0` are the fail-closed decoding defaults, and both + * land here on the first evaluation: a deadline that did not survive a write means "over", + * never "unbounded". `spent` covers both the deadline and the budget and is never reported + * as `done` — the two words mean opposite things to the person reading the console in the + * morning. + * + * Takeover is swept last, after the four conditions the design enumerates, so a run that was + * already over on T3's clock reports why T3 ended it rather than attributing it to a message + * typed afterwards. + */ +export function stopCondition(input: { + readonly nowMs: number; + readonly record: LoopRecord; + readonly shell: LoopThreadShell; + readonly sentinelAtMs: number | null; + readonly loopDoneAtMs: number | null; +}): GuardStop | null { + const { nowMs, record } = input; + const done = doneSignal(input); + if (done !== null) { + return stop("done", done.cause, `${done.cause} at ${done.atMs}`); + } + if (nowMs >= record.deadlineAtMs) { + return stop("spent", "deadline", `deadline passed at ${record.deadlineAtMs}`); + } + if (record.checkInsUsed >= record.maxCheckIns) { + return stop( + "spent", + "budget", + `used ${record.checkInsUsed} of ${record.maxCheckIns} check-ins`, + ); + } + if (record.strikes >= STRIKE_LIMIT) { + return stop("stalled", "strikes", `${record.strikes} consecutive unproductive check-ins`); + } + if (tookOver(record, input.shell)) { + return stop("handed-back", "takeover", `user message at ${input.shell.latestUserMessageAt}`); + } + return null; +} + +const standDown = ( + guard: GuardId, + reason: StandDownReason, + phase: LoopPhase, + untilMs: number | null = null, +): GuardStandDown => ({ kind: "stand_down", guard, reason, phase, untilMs }); + +/** + * Run the table and return the first thing that blocks, or `fire`. + * + * The order is: 2 master toggle, 3 terminal/armed, 4 shell, **4b stop sweep**, 6 snooze, + * 8 blocked on a human, 9 auto-resume, 10 rate limit, 10b deference, 11 check-in floor, + * 12 idle threshold, 14 ceiling. Guards 5 and 13 are retired and their numbers are not + * reused. Guard 7 is not a blocker — it rides out on the `fire` result as `repairPin`. + */ +export function evaluateGuards(input: LoopGuardInput): GuardOutcome { + const { record, global, shell, nowMs, config, trigger } = input; + + // 2 — nothing is disarmed and nothing is stopped; budgets stay intact. + if (masterToggleOff(global)) return standDown("2", "disabled", "standing_down"); + + // 3 — sticky terminal first, so a stopped loop says so rather than "not armed". + if (isStopped(record)) return standDown("3", "stopped", "off"); + if (!isArmed(record)) return standDown("3", "not_armed", "off"); + + // 4 — the thread itself. `null` is `Option.none`: the thread was deleted. + if (shell === null) return { kind: "disarm", guard: "4", reason: "thread_gone" }; + if (isArchived(shell)) return { kind: "disarm", guard: "4", reason: "archived" }; + + // 4b — before every skip, because a stop is a fact about the run. + const stopped = stopCondition({ ...input, shell }); + if (stopped !== null) return stopped; + + // 6 — `held`, not `watching`: a snooze is a bounded hold with an expiry, exactly like the + // rate limit, and `status.ts` reports the same fact for the same thread. Two phases for one + // state is how a console and a tool end up telling a user different things at 03:00. + if (isSnoozed(shell, nowMs)) { + return standDown("6", "snoozed", "held", isoMs(shell.snoozedUntil)); + } + + // 8 + const blocked = blockingRequest(shell); + if (blocked !== null) return standDown("8", blocked, "blocked"); + + // 9 — auto-resume owns this thread right now; two reactors must not both nudge it. + if (input.autoResumePending) return standDown("9", "auto_resume_pending", "watching"); + + // 10 + if (isRateLimited(record, nowMs)) { + return standDown("10", "rate_limited", "held", record.rateLimitedUntilMs); + } + + // 10b. `wake` is re-read here rather than through the predicate so the console gets the + // time it is standing down until without a second null check that could never be true. + const wake = trigger.wake; + if (wake !== null && wakeIsPending(trigger, nowMs)) { + return standDown("10b", "self_pacing", "self_pacing", wake.atMs); + } + + // 11. Same shape: with no previous check-in there is no floor to be under. + const last = record.lastCheckIn; + if (last !== null && !checkInFloorMet(record, config, nowMs)) { + return standDown("11", "check_in_floor", "watching", last.firedAtMs + config.idleMs); + } + + // 12 + if (!idleThresholdMet(trigger)) return standDown("12", "not_idle", "watching"); + + // 14 + if (atArmedCeiling(input.armedCount, global)) return standDown("14", "ceiling", "standing_down"); + + return { + kind: "fire", + repairPin: needsPinRepair(shell), + degrade: wakeIsLost(trigger, nowMs) ? "wake_lost" : null, + }; +} diff --git a/apps/server/src/coil/loop/hooksRegistry.ts b/apps/server/src/coil/loop/hooksRegistry.ts new file mode 100644 index 000000000000..1721f6dc15d2 --- /dev/null +++ b/apps/server/src/coil/loop/hooksRegistry.ts @@ -0,0 +1,50 @@ +/** + * The process-level handle on the running supervisor's `LoopStore`. + * + * This exists for exactly one caller: `loopHooksFor`, which the Claude adapter yields while + * building its query options. `Effect.serviceOption(LoopStore)` looked like it would be + * enough — it reads the service optionally, so the adapter's requirements never widen — but + * it resolves against the *fiber's* context, and in production that context is upstream's + * layer graph. `CoilLayerLive` builds the store with `Layer.provide`, which discharges the + * requirement rather than publishing it, so `LoopStore` is not in the app context, the + * adapter reads `None`, and the hooks are never installed. The tests that pass do so because + * they provide the store into the test fiber themselves. + * + * Publishing `LoopStore` out of `CoilLayerLive` instead would put a fork-only service in the + * type of upstream's layer graph — the thing every seam decision in this fork exists to + * avoid. So the supervisor installs itself here for the life of its scope, `loopHooksFor` + * prefers the context when one is present (which keeps every existing test honest) and falls + * back to this holder otherwise. + * + * A module-level mutable holder rather than a `Ref`: the reader is a synchronous SDK + * callback path, there is exactly one supervisor per process, and the install/uninstall pair + * is scoped so a torn-down layer cannot leave a stale store behind. + * + * @module coil/loop/hooksRegistry + */ + +import * as Effect from "effect/Effect"; + +import type { LoopStoreShape } from "./state.ts"; + +let installedStore: LoopStoreShape | null = null; + +/** The store the running supervisor installed, or `null` when no supervisor is up. */ +export const installedLoopStore = (): LoopStoreShape | null => installedStore; + +/** + * Install for the life of the calling scope. + * + * The release only clears the holder when it still points at *this* store, so a layer torn + * down after a second one was built cannot retire the live supervisor's hooks. + */ +export const installLoopStore = (store: LoopStoreShape) => + Effect.acquireRelease( + Effect.sync(() => { + installedStore = store; + }), + () => + Effect.sync(() => { + if (installedStore === store) installedStore = null; + }), + ); diff --git a/apps/server/src/coil/loop/http.test.ts b/apps/server/src/coil/loop/http.test.ts new file mode 100644 index 000000000000..6cf243ec40fd --- /dev/null +++ b/apps/server/src/coil/loop/http.test.ts @@ -0,0 +1,1225 @@ +// @effect-diagnostics nodeBuiltinImport:off +/** + * Route-level tests for the loop API — TESTS.md cases 71–86e. + * + * Serves ONLY the fork's route layer over a real HTTP server on an ephemeral port, with + * `EnvironmentAuth` mocked (`coil/http/testAuth.ts`), a **real** `LoopStore` on a temp file, + * and stub engine / projection layers that record what was dispatched. + * + * The store is real on purpose: half of these cases are assertions that a refusal left the + * durable record untouched, and a mocked store would assert that the handler did not call a + * method rather than that the file did not change. + * + * `POST /api/coil/loop/answer` (cases 81 and 83) covers the **blocker** half only. Case 82's + * native half is deliberately absent: a native `AskUserQuestion` is answered through upstream's + * own composer path, which the console points at rather than cloning — see the route's own note. + */ +import { assert, describe, it } from "@effect/vitest"; +import { + AuthOrchestrationOperateScope, + AuthOrchestrationReadScope, + ProjectId, + ProviderInstanceId, + ThreadId, + type OrchestrationCommand, + type OrchestrationThreadShell, +} from "@t3tools/contracts"; +import * as Effect from "effect/Effect"; +import * as FileSystem from "effect/FileSystem"; +import * as Layer from "effect/Layer"; +import * as Option from "effect/Option"; +import * as Schema from "effect/Schema"; +import type { HttpClient, HttpServer } from "effect/unstable/http"; +import * as NodePath from "node:path"; + +import * as EnvironmentAuth from "../../auth/EnvironmentAuth.ts"; +import { OrchestrationEngineService } from "../../orchestration/Services/OrchestrationEngine.ts"; +import { ProjectionSnapshotQuery } from "../../orchestration/Services/ProjectionSnapshotQuery.ts"; +import { + authFails, + authOk, + getJson, + jsonBody, + postJson, + postText, + runServed, + serveRoutes, +} from "../http/testAuth.ts"; +import { LoopSettingsView, LoopView, loopRouteLayer } from "./http.ts"; +import { LoopStore, makeLoopStore, type LoopStoreShape } from "./state.ts"; + +const PATH = "/api/coil/loop"; +const LOOPS_PATH = "/api/coil/loops"; +const SETTINGS_PATH = "/api/coil/loop/settings"; +const ANSWER_PATH = "/api/coil/loop/answer"; + +const THREAD_ID = "thread-a"; +const PROJECT_ID = ProjectId.make("project-a"); + +/** + * Fixed far-future timestamps, so nothing here reads a clock: the routes compare against + * the real `Clock`, and a literal in 2100 is unambiguously ahead of it. + */ +const FUTURE_DEADLINE = 4_102_444_800_000; // 2100-01-01T00:00:00Z +/** An hour before the deadline: deference applies. */ +const WAKE_INSIDE_DEADLINE = FUTURE_DEADLINE - 3_600_000; +const FUTURE_ISO = "2099-01-01T00:00:00.000Z"; + +const makeThread = ( + id: string, + overrides: Partial = {}, +): OrchestrationThreadShell => ({ + id: ThreadId.make(id), + projectId: PROJECT_ID, + title: id, + modelSelection: { instanceId: ProviderInstanceId.make("claude"), model: "sonnet" }, + runtimeMode: "full-access", + interactionMode: "default", + branch: null, + worktreePath: null, + latestTurn: null, + createdAt: "2026-08-01T00:00:00.000Z", + updatedAt: "2026-09-01T00:00:00.000Z", + archivedAt: null, + settledOverride: null, + settledAt: null, + session: null, + latestUserMessageAt: "2026-09-01T00:00:00.000Z", + hasPendingApprovals: false, + hasPendingUserInput: false, + hasActionableProposedPlan: false, + ...overrides, +}); + +interface Harness { + readonly store: LoopStoreShape; + /** Every command the routes dispatched, in order. */ + readonly dispatched: Array; +} + +const decodeLoopView = Schema.decodeUnknownEffect(LoopView); +const decodeSettingsView = Schema.decodeUnknownEffect(LoopSettingsView); + +const withRoutes = (options: { + readonly auth?: Layer.Layer; + readonly threads?: ReadonlyArray; + readonly body: ( + harness: Harness, + ) => Effect.Effect; +}) => + runServed( + Effect.gen(function* () { + const fs = yield* FileSystem.FileSystem; + const root = yield* fs.makeTempDirectoryScoped({ prefix: "coil-loop-http-" }); + const store = yield* makeLoopStore(NodePath.join(root, "coil-loop.json")); + const dispatched: Array = []; + const threads = new Map( + (options.threads ?? [makeThread(THREAD_ID)]).map((thread) => [thread.id as string, thread]), + ); + + const deps = Layer.mergeAll( + Layer.succeed(LoopStore, store), + Layer.mock(OrchestrationEngineService)({ + dispatch: (command) => { + dispatched.push(command); + return Effect.succeed({ sequence: dispatched.length }); + }, + }), + Layer.mock(ProjectionSnapshotQuery)({ + getThreadShellById: (threadId) => + Effect.succeed(Option.fromUndefinedOr(threads.get(threadId as string))), + }), + options.auth ?? authOk([AuthOrchestrationOperateScope]), + ); + + return yield* serveRoutes({ + routes: loopRouteLayer, + deps, + body: options.body({ store, dispatched }), + }); + }), + ); + +const armBody = (overrides: Record = {}) => ({ + threadId: THREAD_ID, + action: "arm", + deadlineAtMs: FUTURE_DEADLINE, + maxCheckIns: 6, + ...overrides, +}); + +const pinCommands = (dispatched: ReadonlyArray) => + dispatched.filter((command) => command.type === "thread.pin" || command.type === "thread.unpin"); + +describe("/api/coil/loop", () => { + // 71 + it("rejects a bad credential with a 401", () => + withRoutes({ + auth: authFails(new EnvironmentAuth.ServerAuthInvalidCredentialError({})), + body: () => + Effect.gen(function* () { + assert.strictEqual((yield* getJson(`${PATH}?threadId=${THREAD_ID}`)).status, 401); + assert.strictEqual((yield* postJson(PATH, armBody())).status, 401); + assert.strictEqual((yield* getJson(SETTINGS_PATH)).status, 401); + assert.strictEqual((yield* getJson(LOOPS_PATH)).status, 401); + }), + })); + + // 72 — every route is operate scope, including the reads: they describe scheduling. + it("rejects a read-scope-only session with a 403 on every route", () => + withRoutes({ + auth: authOk([AuthOrchestrationReadScope]), + body: () => + Effect.gen(function* () { + assert.strictEqual((yield* getJson(`${PATH}?threadId=${THREAD_ID}`)).status, 403); + assert.strictEqual((yield* postJson(PATH, armBody())).status, 403); + assert.strictEqual((yield* getJson(LOOPS_PATH)).status, 403); + assert.strictEqual((yield* getJson(SETTINGS_PATH)).status, 403); + assert.strictEqual((yield* postJson(SETTINGS_PATH, { enabled: true })).status, 403); + }), + })); + + // 73 — the console opens on every thread, so "no loop here" must not be an error path. + it("GET on an unknown thread returns the fail-closed off record, not a 404", () => + withRoutes({ + threads: [], + body: () => + Effect.gen(function* () { + const response = yield* getJson(`${PATH}?threadId=never-seen`); + assert.strictEqual(response.status, 200); + const view = yield* decodeLoopView(yield* response.json); + assert.strictEqual(view.record.armed, false); + assert.strictEqual(view.record.deadlineAtMs, 0); + assert.strictEqual(view.record.maxCheckIns, 0); + assert.strictEqual(view.derived.state, "off"); + assert.strictEqual(view.derived.threadKnown, false); + }), + })); + + it("GET without a threadId is a 400", () => + withRoutes({ + body: () => + Effect.gen(function* () { + const response = yield* getJson(PATH); + assert.strictEqual(response.status, 400); + assert.strictEqual((yield* jsonBody(response)).error, "missing_thread_id"); + }), + })); + + // 74 + it("POST arm arms the thread and seeds the thresholds from the global defaults", () => + withRoutes({ + body: ({ store }) => + Effect.gen(function* () { + const response = yield* postJson(PATH, armBody({ goal: "land the sync" })); + assert.strictEqual(response.status, 200); + const view = yield* decodeLoopView(yield* response.json); + assert.strictEqual(view.record.armed, true); + assert.strictEqual(view.record.maxCheckIns, 6); + assert.strictEqual(view.record.deadlineAtMs, FUTURE_DEADLINE); + assert.strictEqual(view.record.goal, "land the sync"); + assert.strictEqual(view.record.checkInsUsed, 0); + assert.ok(view.record.armedAtMs > 0, "armedAtMs is the sentinel freshness baseline"); + + const global = yield* store.getGlobal; + assert.strictEqual(view.record.idleMs, global.defaultIdleMs); + assert.strictEqual(view.record.busyIdleMs, global.defaultBusyIdleMs); + + // The master toggle ships off, so an armed loop reports standing_down until it is + // flipped on. Arming is still what the user asked for, and nothing is disarmed. + assert.strictEqual(view.derived.state, "standing_down"); + assert.strictEqual(view.derived.reason, "disabled"); + }), + })); + + it("POST arm honours explicit thresholds over the defaults", () => + withRoutes({ + body: () => + Effect.gen(function* () { + const response = yield* postJson( + PATH, + armBody({ idleMs: 5 * 60_000, busyIdleMs: 30 * 60_000 }), + ); + const view = yield* decodeLoopView(yield* response.json); + assert.strictEqual(view.record.idleMs, 5 * 60_000); + assert.strictEqual(view.record.busyIdleMs, 30 * 60_000); + }), + })); + + // 75 — a silent clamp hides a mistake in a feature that spends money unattended. + it("POST arm with a budget over 20 is a 400 and not a clamp", () => + withRoutes({ + body: ({ store }) => + Effect.gen(function* () { + const response = yield* postJson(PATH, armBody({ maxCheckIns: 21 })); + assert.strictEqual(response.status, 400); + assert.strictEqual((yield* jsonBody(response)).error, "budget_too_large"); + const record = yield* store.getThread(THREAD_ID); + assert.strictEqual(record.armed, false, "a 400 must not mutate the store"); + assert.strictEqual(record.maxCheckIns, 0, "and must certainly not clamp to 20"); + }), + })); + + // 76 + it("POST arm with a budget under 1 is a 400", () => + withRoutes({ + body: ({ store }) => + Effect.gen(function* () { + const response = yield* postJson(PATH, armBody({ maxCheckIns: 0 })); + assert.strictEqual(response.status, 400); + assert.strictEqual((yield* jsonBody(response)).error, "budget_too_small"); + assert.strictEqual((yield* store.getThread(THREAD_ID)).armed, false); + }), + })); + + it("POST arm with no budget is a 400 budget_required", () => + withRoutes({ + body: () => + Effect.gen(function* () { + const response = yield* postJson(PATH, { + threadId: THREAD_ID, + action: "arm", + deadlineAtMs: FUTURE_DEADLINE, + }); + assert.strictEqual(response.status, 400); + assert.strictEqual((yield* jsonBody(response)).error, "budget_required"); + }), + })); + + // 77 + it("POST arm with a deadline in the past is a 400", () => + withRoutes({ + body: ({ store }) => + Effect.gen(function* () { + const response = yield* postJson(PATH, armBody({ deadlineAtMs: 1_000 })); + assert.strictEqual(response.status, 400); + assert.strictEqual((yield* jsonBody(response)).error, "deadline_in_past"); + assert.strictEqual((yield* store.getThread(THREAD_ID)).armed, false); + }), + })); + + // 77b — the console words "pick an end time" differently from "that time has passed", + // so the two refusals must not collapse into one bare 400. + it("POST arm with no deadline is a 400 deadline_required", () => + withRoutes({ + body: ({ store }) => + Effect.gen(function* () { + const missing = yield* postJson(PATH, { + threadId: THREAD_ID, + action: "arm", + maxCheckIns: 6, + }); + assert.strictEqual(missing.status, 400); + assert.strictEqual((yield* jsonBody(missing)).error, "deadline_required"); + + const explicitNull = yield* postJson(PATH, armBody({ deadlineAtMs: null })); + assert.strictEqual(explicitNull.status, 400); + assert.strictEqual((yield* jsonBody(explicitNull)).error, "deadline_required"); + + assert.strictEqual((yield* store.getThread(THREAD_ID)).deadlineAtMs, 0); + }), + })); + + // 78 + it("POST arm at the armed ceiling is a 400 ceiling_reached", () => + withRoutes({ + threads: [makeThread("t1"), makeThread("t2"), makeThread("t3"), makeThread("t4")], + body: ({ store }) => + Effect.gen(function* () { + const global = yield* store.getGlobal; + for (const threadId of ["t1", "t2", "t3"]) { + const armed = yield* postJson(PATH, armBody({ threadId })); + assert.strictEqual(armed.status, 200, `arming ${threadId} within the ceiling`); + } + assert.strictEqual(global.maxArmedThreads, 3); + + const refused = yield* postJson(PATH, armBody({ threadId: "t4" })); + assert.strictEqual(refused.status, 400); + assert.strictEqual((yield* jsonBody(refused)).error, "ceiling_reached"); + assert.strictEqual((yield* store.getThread("t4")).armed, false); + + // Re-arming a thread that already holds one of the slots is not a new arm, so the + // ceiling must not refuse it. + const rearmed = yield* postJson(PATH, armBody({ threadId: "t2", action: "rearm" })); + assert.strictEqual(rearmed.status, 200); + }), + })); + + // 78b — `thread.pin` emits companion unsnoozed/unsettled events, so arming a snoozed + // thread would silently cancel a snooze a human set. Assert the *absence* of the dispatch. + it("POST arm on a snoozed thread is a 400 thread_snoozed and dispatches no pin", () => + withRoutes({ + threads: [makeThread(THREAD_ID, { snoozedUntil: FUTURE_ISO })], + body: ({ store, dispatched }) => + Effect.gen(function* () { + const response = yield* postJson(PATH, armBody()); + assert.strictEqual(response.status, 400); + assert.strictEqual((yield* jsonBody(response)).error, "thread_snoozed"); + assert.deepStrictEqual(pinCommands(dispatched), []); + assert.strictEqual((yield* store.getThread(THREAD_ID)).armed, false); + }), + })); + + it("POST arm on a thread whose snooze has already passed succeeds", () => + withRoutes({ + threads: [makeThread(THREAD_ID, { snoozedUntil: "2020-01-01T00:00:00.000Z" })], + body: () => + Effect.gen(function* () { + assert.strictEqual((yield* postJson(PATH, armBody())).status, 200); + }), + })); + + // 78c — arming a settled thread is a promotion the human asked for. + it("POST arm on a settled thread succeeds and pins it", () => + withRoutes({ + threads: [ + makeThread(THREAD_ID, { + settledOverride: "settled", + settledAt: "2026-09-01T01:00:00.000Z", + }), + ], + body: ({ store, dispatched }) => + Effect.gen(function* () { + assert.strictEqual((yield* postJson(PATH, armBody())).status, 200); + const pins = pinCommands(dispatched); + assert.strictEqual(pins.length, 1); + assert.strictEqual(pins[0]?.type, "thread.pin"); + assert.strictEqual((yield* store.getThread(THREAD_ID)).pinnedByLoop, true); + }), + })); + + // 78d — otherwise disarming removes a pin that was the user's, with nothing recording + // that it had ever been theirs. + it("does not unpin a thread the user pinned themselves", () => + withRoutes({ + threads: [makeThread(THREAD_ID, { pinnedAt: "2026-08-30T00:00:00.000Z" })], + body: ({ store, dispatched }) => + Effect.gen(function* () { + assert.strictEqual((yield* postJson(PATH, armBody())).status, 200); + assert.deepStrictEqual(pinCommands(dispatched), [], "already pinned: no pin needed"); + assert.strictEqual((yield* store.getThread(THREAD_ID)).pinnedByLoop, false); + + const disarmed = yield* postJson(PATH, { threadId: THREAD_ID, action: "disarm" }); + assert.strictEqual(disarmed.status, 200); + assert.deepStrictEqual(pinCommands(dispatched), [], "and no unpin on the way out"); + }), + })); + + it("derived reports the ceiling stand-down rather than claiming the loop is watching", () => + withRoutes({ + threads: [makeThread(THREAD_ID), makeThread("thread-b")], + body: ({ store }) => + Effect.gen(function* () { + yield* store.setGlobal({ enabled: true, maxArmedThreads: 3 }); + yield* postJson(PATH, armBody()); + yield* postJson(PATH, { ...armBody(), threadId: "thread-b" }); + + // Two armed against a ceiling of two: nobody is over it, and both keep watching. + yield* store.setGlobal({ maxArmedThreads: 2 }); + const fine = yield* decodeLoopView( + yield* jsonBody(yield* getJson(`${PATH}?threadId=${THREAD_ID}`)), + ); + assert.strictEqual(fine.derived.state, "watching"); + + // Lowering the ceiling stands the excess down — and the console has to say so. It + // was the one stand-down with no lens at all: the loop went quiet and the panel + // kept claiming it was running. + yield* store.setGlobal({ maxArmedThreads: 1 }); + const view = yield* decodeLoopView( + yield* jsonBody(yield* getJson(`${PATH}?threadId=${THREAD_ID}`)), + ); + assert.strictEqual(view.derived.state, "standing_down"); + assert.strictEqual(view.derived.reason, "ceiling"); + assert.isTrue(view.record.armed, "a ceiling never disarms anything"); + }), + })); + + it("a re-arm of a thread the loop itself pinned keeps owning that pin", () => + withRoutes({ + // Pinned when the arm route looks: exactly what the projection shows on a RE-arm, + // because the previous arm is what pinned it. + threads: [makeThread(THREAD_ID, { pinnedAt: "2026-09-01T02:00:00.000Z" })], + body: ({ store, dispatched }) => + Effect.gen(function* () { + yield* store.arm({ + threadId: THREAD_ID, + armedAtMs: 1_000, + deadlineAtMs: FUTURE_DEADLINE, + maxCheckIns: 6, + pinnedByLoop: true, + }); + + assert.strictEqual((yield* postJson(PATH, armBody())).status, 200); + // Reading "already pinned" as "the user's pin" orphaned it: the flag dropped to + // false and no disarm would ever remove a pin the loop had created. + assert.isTrue((yield* store.getThread(THREAD_ID)).pinnedByLoop); + yield* postJson(PATH, { threadId: THREAD_ID, action: "disarm" }); + assert.deepStrictEqual( + pinCommands(dispatched).map((command) => command.type), + ["thread.unpin"], + ); + }), + })); + + it("unpins on disarm exactly when the loop created the pin", () => + withRoutes({ + body: ({ dispatched }) => + Effect.gen(function* () { + yield* postJson(PATH, armBody()); + yield* postJson(PATH, { threadId: THREAD_ID, action: "disarm" }); + assert.deepStrictEqual( + pinCommands(dispatched).map((command) => command.type), + ["thread.pin", "thread.unpin"], + ); + }), + })); + + // 79 — takeover disarms and is not a budget reset: deliberately stopping a thread must + // not hand the next loop a fresh six. + it("POST disarm writes the handed-back terminal and keeps the budget spent", () => + withRoutes({ + body: ({ store }) => + Effect.gen(function* () { + yield* postJson(PATH, armBody()); + yield* store.recordCheckIn({ + threadId: THREAD_ID, + firedAtMs: 1_000, + createdAtIso: "2026-09-02T01:00:00.000Z", + activityCursor: "cursor-1", + }); + + const response = yield* postJson(PATH, { threadId: THREAD_ID, action: "disarm" }); + assert.strictEqual(response.status, 200); + const view = yield* decodeLoopView(yield* response.json); + assert.strictEqual(view.record.armed, false); + assert.strictEqual(view.record.stopped?.reason, "handed-back"); + assert.strictEqual(view.record.checkInsUsed, 1, "disarm is not a budget reset"); + assert.strictEqual(view.derived.state, "stopped"); + assert.strictEqual(view.derived.stoppedReason, "handed-back"); + }), + })); + + it("POST disarm still works when the thread is gone from the projection", () => + withRoutes({ + body: ({ store }) => + Effect.gen(function* () { + // Armed, and the projection has no such row any more: the way out must not depend on + // the thread still existing. + yield* store.arm({ + threadId: "never-seen", + armedAtMs: 1_000, + deadlineAtMs: FUTURE_DEADLINE, + maxCheckIns: 6, + }); + const response = yield* postJson(PATH, { threadId: "never-seen", action: "disarm" }); + assert.strictEqual(response.status, 200); + assert.strictEqual((yield* store.getThread("never-seen")).stopped?.reason, "handed-back"); + }), + })); + + it("POST disarm and edit are refused with 409 unless the loop is armed, and mutate nothing", () => + withRoutes({ + body: ({ store }) => + Effect.gen(function* () { + // A thread nobody ever armed: neither action may mint a record for it. + for (const action of ["disarm", "edit"] as const) { + const response = yield* postJson(PATH, { threadId: THREAD_ID, action }); + assert.strictEqual(response.status, 409); + assert.strictEqual((yield* jsonBody(response)).error, "not_armed"); + } + assert.deepStrictEqual(yield* store.listArmed, []); + assert.isNull((yield* store.getThread(THREAD_ID)).stopped); + + // A run that already ended: a stale tab must not rewrite how the night finished. + yield* postJson(PATH, armBody()); + yield* store.stop(THREAD_ID, { + reason: "done", + atMs: 2_000, + detail: "the agent said so", + }); + const stale = yield* postJson(PATH, { threadId: THREAD_ID, action: "disarm" }); + assert.strictEqual(stale.status, 409); + assert.strictEqual((yield* store.getThread(THREAD_ID)).stopped?.reason, "done"); + + const edited = yield* postJson(PATH, { + threadId: THREAD_ID, + action: "edit", + maxCheckIns: 20, + }); + assert.strictEqual(edited.status, 409); + assert.strictEqual((yield* store.getThread(THREAD_ID)).maxCheckIns, 6); + }), + })); + + it("POST clear removes a finished run, and is refused while one is armed", () => + withRoutes({ + body: ({ store }) => + Effect.gen(function* () { + yield* postJson(PATH, armBody()); + // Armed: clearing is not a way to end a live run silently. That is `disarm`, which + // records why. + const refused = yield* postJson(PATH, { threadId: THREAD_ID, action: "clear" }); + assert.strictEqual(refused.status, 409); + assert.strictEqual((yield* jsonBody(refused)).error, "armed"); + assert.isTrue((yield* store.getThread(THREAD_ID)).armed); + + yield* store.stop(THREAD_ID, { reason: "spent", atMs: 2_000, detail: "budget" }); + const response = yield* postJson(PATH, { threadId: THREAD_ID, action: "clear" }); + assert.strictEqual(response.status, 200); + // The stopped pill is gone, and the view says so rather than 404ing. + const view = yield* decodeLoopView(yield* jsonBody(response)); + assert.strictEqual(view.derived.state, "off"); + assert.isNull(view.record.stopped); + assert.isNull((yield* store.getThread(THREAD_ID)).stopped); + }), + })); + + it("POST disarm with the agent's own wakes still pending banks one session stop", () => + withRoutes({ + body: ({ store }) => + Effect.gen(function* () { + yield* postJson(PATH, armBody()); + // A recurring wake: it reschedules itself, so ending the record does not end it. + yield* store.setCrons(THREAD_ID, { + recordedAtMs: 1_000, + entries: [ + { + id: "cron-1", + schedule: "*/20 * * * *", + recurring: true, + prompt: "keep going", + nextFireAtMs: WAKE_INSIDE_DEADLINE, + }, + ], + }); + + const response = yield* postJson(PATH, { threadId: THREAD_ID, action: "disarm" }); + assert.strictEqual(response.status, 200); + // The route cannot end a session itself; it banks the request and the supervisor's + // next tick issues exactly one `stopSession`. `Reactor.test.ts` owns that half. + assert.isAbove((yield* store.getThread(THREAD_ID)).stopRequestedAtMs, 0); + }), + })); + + it("POST disarm with no pending wakes banks nothing", () => + withRoutes({ + body: ({ store }) => + Effect.gen(function* () { + yield* postJson(PATH, armBody()); + yield* postJson(PATH, { threadId: THREAD_ID, action: "disarm" }); + assert.strictEqual((yield* store.getThread(THREAD_ID)).stopRequestedAtMs, 0); + }), + })); + + // 80 + it("POST rearm after a terminal clears it and starts a fresh budget", () => + withRoutes({ + body: ({ store }) => + Effect.gen(function* () { + yield* postJson(PATH, armBody()); + yield* store.recordCheckIn({ + threadId: THREAD_ID, + firedAtMs: 1_000, + createdAtIso: "2026-09-02T01:00:00.000Z", + activityCursor: "cursor-1", + }); + yield* store.stop(THREAD_ID, { reason: "spent", atMs: 2_000, detail: "budget" }); + + const response = yield* postJson( + PATH, + armBody({ action: "rearm", maxCheckIns: 4, deadlineAtMs: FUTURE_DEADLINE + 1000 }), + ); + assert.strictEqual(response.status, 200); + const view = yield* decodeLoopView(yield* response.json); + assert.strictEqual(view.record.stopped, null); + assert.strictEqual(view.record.armed, true); + assert.strictEqual(view.record.checkInsUsed, 0); + assert.strictEqual(view.record.maxCheckIns, 4); + assert.deepStrictEqual(view.record.checkIns, []); + assert.strictEqual(view.record.strikes, 0); + }), + })); + + it("POST edit changes the bounds without touching the budget already spent", () => + withRoutes({ + body: ({ store }) => + Effect.gen(function* () { + yield* postJson(PATH, armBody()); + yield* store.recordCheckIn({ + threadId: THREAD_ID, + firedAtMs: 1_000, + createdAtIso: "2026-09-02T01:00:00.000Z", + activityCursor: "cursor-1", + }); + + const response = yield* postJson(PATH, { + threadId: THREAD_ID, + action: "edit", + maxCheckIns: 10, + deadlineAtMs: FUTURE_DEADLINE + 3_600_000, + }); + const view = yield* decodeLoopView(yield* response.json); + assert.strictEqual(view.record.maxCheckIns, 10); + assert.strictEqual(view.record.deadlineAtMs, FUTURE_DEADLINE + 3_600_000); + assert.strictEqual(view.record.checkInsUsed, 1, "edit is not a re-arm"); + assert.strictEqual(view.record.checkIns.length, 1); + }), + })); + + it("POST edit refuses an out-of-range budget with the same codes arming uses", () => + withRoutes({ + body: ({ store }) => + Effect.gen(function* () { + yield* postJson(PATH, armBody()); + const response = yield* postJson(PATH, { + threadId: THREAD_ID, + action: "edit", + maxCheckIns: 40, + }); + assert.strictEqual(response.status, 400); + assert.strictEqual((yield* jsonBody(response)).error, "budget_too_large"); + assert.strictEqual((yield* store.getThread(THREAD_ID)).maxCheckIns, 6); + }), + })); + + // 84 + it("POST with a malformed body is a 400 and mutates nothing", () => + withRoutes({ + body: ({ store, dispatched }) => + Effect.gen(function* () { + const malformed = [ + ["no fields at all", { nope: true }], + ["no action", { threadId: THREAD_ID }], + ["an action nobody defined", { threadId: THREAD_ID, action: "detonate" }], + [ + "an empty threadId", + { threadId: "", action: "arm", deadlineAtMs: FUTURE_DEADLINE, maxCheckIns: 3 }, + ], + ] as const; + for (const [label, body] of malformed) { + const response = yield* postJson(PATH, body); + assert.strictEqual(response.status, 400, label); + assert.strictEqual((yield* jsonBody(response)).error, "invalid_body", label); + } + assert.strictEqual((yield* store.getThread(THREAD_ID)).armed, false); + assert.deepStrictEqual(dispatched, []); + }), + })); + + it("POST with a body that is not JSON at all is still a coded invalid_body", () => + withRoutes({ + body: ({ store, dispatched }) => + Effect.gen(function* () { + // `request.json` FAILS on these rather than decoding to something unusable, and that + // failure is not one of the auth tags — so it escaped the handler and the client got + // a bare, empty 400 with no code for the console to word. + for (const raw of ["not json at all", "{", ""]) { + const response = yield* postText(PATH, raw); + assert.strictEqual(response.status, 400, raw); + assert.strictEqual((yield* jsonBody(response)).error, "invalid_body", raw); + } + for (const path of [SETTINGS_PATH, ANSWER_PATH]) { + const response = yield* postText(path, "not json at all"); + assert.strictEqual(response.status, 400, path); + assert.strictEqual((yield* jsonBody(response)).error, "invalid_body", path); + } + assert.strictEqual((yield* store.getThread(THREAD_ID)).armed, false); + assert.deepStrictEqual(dispatched, []); + }), + })); + + // A caller-controlled threadId reaches `Object.hasOwn` in the store. Pinned at the HTTP + // boundary as well as in the store, because this route is what makes it reachable. + it("handles a prototype-chain threadId as an ordinary unknown thread", () => + withRoutes({ + threads: [], + body: () => + Effect.gen(function* () { + for (const hostile of ["constructor", "__proto__", "toString"]) { + const response = yield* getJson(`${PATH}?threadId=${encodeURIComponent(hostile)}`); + assert.strictEqual(response.status, 200, `GET threadId=${hostile} must not 500`); + const view = yield* decodeLoopView(yield* response.json); + assert.strictEqual(view.record.armed, false); + } + }), + })); + + // 85 — the response shapes are the contract with the console; a drift here is a silent + // client break, since these routes deliberately cost zero `packages/contracts` edits. + it("every response decodes against its schema", () => + withRoutes({ + body: () => + Effect.gen(function* () { + yield* decodeLoopView(yield* jsonBody(yield* postJson(PATH, armBody()))); + yield* decodeLoopView(yield* jsonBody(yield* getJson(`${PATH}?threadId=${THREAD_ID}`))); + yield* decodeSettingsView(yield* jsonBody(yield* getJson(SETTINGS_PATH))); + yield* decodeSettingsView( + yield* jsonBody(yield* postJson(SETTINGS_PATH, { enabled: true })), + ); + const loops = yield* jsonBody(yield* getJson(LOOPS_PATH)); + assert.ok(Array.isArray(loops.loops)); + for (const loop of loops.loops as ReadonlyArray) { + yield* decodeLoopView(loop); + } + }), + })); + + // 86 + it("GET /api/coil/loops lists every armed loop in a deterministic order", () => + withRoutes({ + threads: [makeThread("zeta"), makeThread("alpha"), makeThread("mid")], + body: () => + Effect.gen(function* () { + // Armed out of alphabetical order: insertion order must not leak into the response. + for (const threadId of ["zeta", "alpha", "mid"]) { + yield* postJson(PATH, armBody({ threadId })); + } + yield* postJson(PATH, { threadId: "mid", action: "disarm" }); + + const response = yield* getJson(LOOPS_PATH); + assert.strictEqual(response.status, 200); + const body = yield* jsonBody(response); + const threadIds: Array = []; + for (const loop of body.loops as ReadonlyArray) { + threadIds.push((yield* decodeLoopView(loop)).threadId); + } + assert.deepStrictEqual( + threadIds, + ["alpha", "zeta"], + "disarmed loops drop out, and the rest are ordered by threadId", + ); + }), + })); +}); + +describe("/api/coil/loop/settings", () => { + // 86b — the route ships in phase 2 with the reactor. Shipping "default off behind the + // master toggle" while only the settings UI could flip it left phase 2 unswitchable. + it("GET returns the fail-closed global block on a fresh install", () => + withRoutes({ + body: () => + Effect.gen(function* () { + const response = yield* getJson(SETTINGS_PATH); + assert.strictEqual(response.status, 200); + const view = yield* decodeSettingsView(yield* response.json); + assert.strictEqual(view.enabled, false, "the master toggle defaults OFF"); + assert.strictEqual(view.maxArmedThreads, 3); + assert.strictEqual(view.defaultMaxCheckIns, 6); + assert.strictEqual(view.defaultRunMs, 8 * 3_600_000); + assert.strictEqual(view.defaultIdleMs, 15 * 60_000); + assert.strictEqual(view.defaultBusyIdleMs, 45 * 60_000); + assert.strictEqual(view.armedCount, 0); + }), + })); + + // 86c — durable, so the next tick observes it rather than a per-process copy. + it("POST writes the master toggle to the durable store", () => + withRoutes({ + body: ({ store }) => + Effect.gen(function* () { + const response = yield* postJson(SETTINGS_PATH, { enabled: true, defaultMaxCheckIns: 8 }); + assert.strictEqual(response.status, 200); + const view = yield* decodeSettingsView(yield* response.json); + assert.strictEqual(view.enabled, true); + assert.strictEqual(view.defaultMaxCheckIns, 8); + + // What the tick fiber would read. + const global = yield* store.getGlobal; + assert.strictEqual(global.enabled, true); + assert.strictEqual(global.defaultMaxCheckIns, 8); + // Untouched keys survive a partial patch. + assert.strictEqual(global.maxArmedThreads, 3); + + const reread = yield* decodeSettingsView(yield* jsonBody(yield* getJson(SETTINGS_PATH))); + assert.strictEqual(reread.enabled, true); + }), + })); + + // 86d — the toggle is a guard, not a lifecycle. + it("toggling off stands loops down without disarming or stopping anything", () => + withRoutes({ + body: ({ store }) => + Effect.gen(function* () { + yield* postJson(SETTINGS_PATH, { enabled: true }); + yield* postJson(PATH, armBody()); + yield* store.recordCheckIn({ + threadId: THREAD_ID, + firedAtMs: 1_000, + createdAtIso: "2026-09-02T01:00:00.000Z", + activityCursor: "cursor-1", + }); + const watching = yield* decodeLoopView( + yield* jsonBody(yield* getJson(`${PATH}?threadId=${THREAD_ID}`)), + ); + assert.strictEqual(watching.derived.state, "watching"); + + yield* postJson(SETTINGS_PATH, { enabled: false }); + const down = yield* decodeLoopView( + yield* jsonBody(yield* getJson(`${PATH}?threadId=${THREAD_ID}`)), + ); + assert.strictEqual(down.derived.state, "standing_down"); + assert.strictEqual(down.derived.reason, "disabled"); + assert.strictEqual(down.record.armed, true, "nothing is disarmed"); + assert.strictEqual(down.record.stopped, null, "nothing is stopped"); + assert.strictEqual(down.record.checkInsUsed, 1, "the budget is untouched"); + + yield* postJson(SETTINGS_PATH, { enabled: true }); + const resumed = yield* decodeLoopView( + yield* jsonBody(yield* getJson(`${PATH}?threadId=${THREAD_ID}`)), + ); + assert.strictEqual(resumed.derived.state, "watching"); + assert.strictEqual(resumed.record.checkInsUsed, 1, "and resumes with the same budget"); + }), + })); + + // 86e — same rule as 86d: lowering the ceiling stands the excess down at the next tick + // rather than manufacturing terminal states nobody chose. + it("accepts a maxArmedThreads below the current armed count without disarming anything", () => + withRoutes({ + threads: [makeThread("t1"), makeThread("t2"), makeThread("t3")], + body: ({ store }) => + Effect.gen(function* () { + for (const threadId of ["t1", "t2", "t3"]) { + yield* postJson(PATH, armBody({ threadId })); + } + const response = yield* postJson(SETTINGS_PATH, { maxArmedThreads: 1 }); + assert.strictEqual(response.status, 200); + const view = yield* decodeSettingsView(yield* response.json); + assert.strictEqual(view.maxArmedThreads, 1); + assert.strictEqual(view.armedCount, 3, "and reports the overhang honestly"); + + const armed = yield* store.listArmed; + assert.strictEqual(armed.length, 3, "nothing is disarmed by lowering the ceiling"); + for (const entry of armed) { + assert.strictEqual(entry.record.stopped, null); + } + }), + })); + + it("POST settings out of range is a 400 and writes nothing", () => + withRoutes({ + body: ({ store }) => + Effect.gen(function* () { + const refusals = [ + ["a ceiling of zero", { maxArmedThreads: 0 }, "out_of_range"], + ["a default budget over the cap", { defaultMaxCheckIns: 21 }, "out_of_range"], + ["a zero idle threshold", { defaultIdleMs: 0 }, "out_of_range"], + ["a fractional ceiling", { maxArmedThreads: 2.5 }, "invalid_body"], + ["a toggle that is not a boolean", { enabled: "yes" }, "invalid_body"], + ] as const; + for (const [label, body, code] of refusals) { + const response = yield* postJson(SETTINGS_PATH, body); + assert.strictEqual(response.status, 400, label); + assert.strictEqual((yield* jsonBody(response)).error, code, label); + } + assert.deepStrictEqual(yield* store.getGlobal, { + enabled: false, + maxArmedThreads: 3, + defaultMaxCheckIns: 6, + defaultRunMs: 8 * 3_600_000, + defaultIdleMs: 15 * 60_000, + defaultBusyIdleMs: 45 * 60_000, + }); + }), + })); +}); + +describe("/api/coil/loop derived state", () => { + const armedAndEnabled = (harness: Harness) => + Effect.gen(function* () { + yield* postJson(SETTINGS_PATH, { enabled: true }); + yield* postJson(PATH, armBody()); + return harness; + }); + + it("reports held while a usage limit is live", () => + withRoutes({ + body: (harness) => + Effect.gen(function* () { + yield* armedAndEnabled(harness); + yield* harness.store.setRateLimitedUntil(THREAD_ID, FUTURE_DEADLINE); + const view = yield* decodeLoopView( + yield* jsonBody(yield* getJson(`${PATH}?threadId=${THREAD_ID}`)), + ); + assert.strictEqual(view.derived.state, "held"); + assert.strictEqual(view.derived.reason, "rate_limited"); + }), + })); + + // Guard 8's third clause: a thread parked on an unapproved plan is waiting on a human even + // though `hasPendingUserInput` is false. + it("reports blocked on an actionable proposed plan, not just on a pending input", () => + withRoutes({ + threads: [makeThread(THREAD_ID, { hasActionableProposedPlan: true })], + body: (harness) => + Effect.gen(function* () { + yield* armedAndEnabled(harness); + const view = yield* decodeLoopView( + yield* jsonBody(yield* getJson(`${PATH}?threadId=${THREAD_ID}`)), + ); + assert.strictEqual(view.derived.state, "blocked"); + assert.strictEqual(view.derived.reason, "pending_input"); + }), + })); + + it("reports self_pacing while a recorded wake is still ahead and inside the deadline", () => + withRoutes({ + body: (harness) => + Effect.gen(function* () { + yield* armedAndEnabled(harness); + yield* harness.store.setCrons(THREAD_ID, { + recordedAtMs: 1_000, + entries: [ + { + id: "cron-1", + schedule: "*/30 * * * *", + recurring: true, + prompt: "keep going", + nextFireAtMs: WAKE_INSIDE_DEADLINE, + }, + ], + }); + const view = yield* decodeLoopView( + yield* jsonBody(yield* getJson(`${PATH}?threadId=${THREAD_ID}`)), + ); + assert.strictEqual(view.derived.state, "self_pacing"); + assert.strictEqual(view.derived.nextWakeAtMs, WAKE_INSIDE_DEADLINE); + }), + })); + + // A wake past the deadline is not deference: `CronCreate` is unbounded, so a run must not + // stand by for a wake that lands after it was supposed to have ended. + it("does not defer to a wake scheduled past the deadline", () => + withRoutes({ + body: (harness) => + Effect.gen(function* () { + yield* armedAndEnabled(harness); + yield* harness.store.setCrons(THREAD_ID, { + recordedAtMs: 1_000, + entries: [ + { + id: "cron-1", + schedule: "0 3 * * *", + recurring: false, + prompt: "tomorrow", + nextFireAtMs: FUTURE_DEADLINE + 3_600_000, + }, + ], + }); + const view = yield* decodeLoopView( + yield* jsonBody(yield* getJson(`${PATH}?threadId=${THREAD_ID}`)), + ); + assert.strictEqual(view.derived.state, "watching"); + }), + })); + + // An unparseable schedule means no deference from that entry, never "wake now". + it("treats an unparsed wake as no deference at all", () => + withRoutes({ + body: (harness) => + Effect.gen(function* () { + yield* armedAndEnabled(harness); + yield* harness.store.setCrons(THREAD_ID, { + recordedAtMs: 1_000, + entries: [ + { + id: "cron-1", + schedule: "not a cron", + recurring: false, + prompt: "?", + nextFireAtMs: null, + }, + ], + }); + const view = yield* decodeLoopView( + yield* jsonBody(yield* getJson(`${PATH}?threadId=${THREAD_ID}`)), + ); + assert.strictEqual(view.derived.state, "watching"); + assert.strictEqual(view.derived.nextWakeAtMs, null); + }), + })); + + it("keeps a terminal state visible over every live reading", () => + withRoutes({ + body: (harness) => + Effect.gen(function* () { + yield* armedAndEnabled(harness); + yield* harness.store.stop(THREAD_ID, { + reason: "spent", + atMs: 2_000, + detail: "budget exhausted", + }); + const view = yield* decodeLoopView( + yield* jsonBody(yield* getJson(`${PATH}?threadId=${THREAD_ID}`)), + ); + assert.strictEqual(view.derived.state, "stopped"); + assert.strictEqual(view.derived.stoppedReason, "spent", "spent is never done"); + }), + })); +}); + +describe("/api/coil/loop/answer", () => { + const blocker = (overrides: Record = {}) => ({ + id: "blocker-1", + raisedAtMs: 1_000, + question: "Migrate in place or backfill?", + options: [], + context: null, + answeredAtMs: null, + answer: null, + deliveredToAgent: false, + ...overrides, + }); + + it("rejects a bad credential and a read-only one, like every other route", () => + withRoutes({ + auth: authFails(new EnvironmentAuth.ServerAuthInvalidCredentialError({})), + body: () => + Effect.gen(function* () { + const response = yield* postJson(ANSWER_PATH, { + threadId: THREAD_ID, + blockerId: "blocker-1", + answer: "yes", + }); + assert.strictEqual(response.status, 401); + }), + })); + + it("refuses a read-scope credential — answering mutates scheduling", () => + withRoutes({ + auth: authOk([AuthOrchestrationReadScope]), + body: () => + Effect.gen(function* () { + const response = yield* postJson(ANSWER_PATH, { + threadId: THREAD_ID, + blockerId: "blocker-1", + answer: "yes", + }); + assert.strictEqual(response.status, 403); + }), + })); + + // 81 — the answer is recorded and stays UNDELIVERED: the thread is idle, so nothing has told + // the agent yet. `deliveredToAgent` is what stops the next check-in either losing it or + // restating it twice. + it("records the answer and leaves it undelivered", () => + withRoutes({ + body: ({ store }) => + Effect.gen(function* () { + yield* postJson(PATH, armBody()); + yield* store.addBlocker(THREAD_ID, blocker()); + + const response = yield* postJson(ANSWER_PATH, { + threadId: THREAD_ID, + blockerId: "blocker-1", + answer: "migrate in place", + }); + assert.strictEqual(response.status, 200); + assert.deepStrictEqual(yield* jsonBody(response), { ok: true }); + + const record = yield* store.getThread(THREAD_ID); + const stored = record.blockers[0]; + assert.strictEqual(stored?.answer, "migrate in place"); + assert.notStrictEqual(stored?.answeredAtMs, null); + assert.strictEqual(stored?.deliveredToAgent, false); + + // And it drops out of the console's actionable list. + const view = yield* decodeLoopView( + yield* jsonBody(yield* getJson(`${PATH}?threadId=${THREAD_ID}`)), + ); + assert.deepStrictEqual(view.blockers, []); + }), + })); + + it("is idempotent — a second answer keeps the first", () => + withRoutes({ + body: ({ store }) => + Effect.gen(function* () { + yield* postJson(PATH, armBody()); + yield* store.addBlocker(THREAD_ID, blocker()); + yield* postJson(ANSWER_PATH, { + threadId: THREAD_ID, + blockerId: "blocker-1", + answer: "first", + }); + const second = yield* postJson(ANSWER_PATH, { + threadId: THREAD_ID, + blockerId: "blocker-1", + answer: "second", + }); + + assert.strictEqual(second.status, 200); + const record = yield* store.getThread(THREAD_ID); + assert.strictEqual(record.blockers.length, 1, "not a second append"); + assert.strictEqual(record.blockers[0]?.answer, "first"); + }), + })); + + // 83 — an id nobody raised. A silent 200 would show the console an answer that was never stored. + it("404s an unknown blocker id, and an unknown thread", () => + withRoutes({ + body: ({ store }) => + Effect.gen(function* () { + yield* postJson(PATH, armBody()); + yield* store.addBlocker(THREAD_ID, blocker()); + + assert.strictEqual( + (yield* postJson(ANSWER_PATH, { + threadId: THREAD_ID, + blockerId: "never-raised", + answer: "yes", + })).status, + 404, + ); + assert.strictEqual( + (yield* postJson(ANSWER_PATH, { + threadId: "never-seen", + blockerId: "blocker-1", + answer: "yes", + })).status, + 404, + ); + }), + })); + + it("400s a malformed body and mutates nothing", () => + withRoutes({ + body: ({ store }) => + Effect.gen(function* () { + yield* postJson(PATH, armBody()); + yield* store.addBlocker(THREAD_ID, blocker()); + + const malformed = [ + ["no fields", {}], + ["no blockerId", { threadId: THREAD_ID, answer: "yes" }], + ["no answer", { threadId: THREAD_ID, blockerId: "blocker-1" }], + ["empty blockerId", { threadId: THREAD_ID, blockerId: " ", answer: "yes" }], + ["empty threadId", { threadId: "", blockerId: "blocker-1", answer: "yes" }], + ["a non-string answer", { threadId: THREAD_ID, blockerId: "blocker-1", answer: 3 }], + ] as const; + for (const [label, body] of malformed) { + const response = yield* postJson(ANSWER_PATH, body); + assert.strictEqual(response.status, 400, label); + assert.strictEqual((yield* jsonBody(response)).error, "invalid_body", label); + } + assert.strictEqual((yield* store.getThread(THREAD_ID)).blockers[0]?.answer, null); + }), + })); + + // A caller-controlled threadId reaches `Object.hasOwn` in the store on this route too. + it("handles a prototype-chain threadId as an ordinary unknown thread", () => + withRoutes({ + threads: [], + body: () => + Effect.gen(function* () { + for (const hostile of ["constructor", "__proto__", "toString"]) { + const response = yield* postJson(ANSWER_PATH, { + threadId: hostile, + blockerId: "blocker-1", + answer: "yes", + }); + assert.strictEqual(response.status, 404, `answer threadId=${hostile} must not 500`); + } + }), + })); +}); diff --git a/apps/server/src/coil/loop/http.ts b/apps/server/src/coil/loop/http.ts new file mode 100644 index 000000000000..1f9f3426a7d0 --- /dev/null +++ b/apps/server/src/coil/loop/http.ts @@ -0,0 +1,761 @@ +// @effect-diagnostics globalDate:off - parses explicit ISO strings off a thread shell; the clock is read through `Clock`. +/** + * Fork-owned raw HTTP routes backing the loop console and the loop settings panel. + * + * GET /api/coil/loop?threadId=… -> { threadId, record, derived, blockers, ledger } + * POST /api/coil/loop -> arm / rearm / edit / disarm, same shape back + * GET /api/coil/loops -> { loops: [...] }, deterministically ordered + * GET /api/coil/loop/settings -> the global block + armedCount + * POST /api/coil/loop/settings -> write the master toggle and the defaults + * POST /api/coil/loop/answer -> answer one deferred blocker + * + * Raw routes rather than WS-RPC for the reason `webPush/http.ts` states: an RPC would force + * edits to `@t3tools/contracts`, `ws.ts` and its scope map, where a raw route costs one + * additive line in a fork-owned file. Everything here is **operate** scope, including the + * reads: these endpoints describe and mutate unattended scheduling, and the read scope is + * for content. + * + * ## Two rules this module exists to enforce + * + * **Nothing is clamped.** `maxCheckIns` outside 1..20 and a deadline in the past are 400s + * with distinct codes, never a silently corrected value. This is a feature that spends money + * unattended overnight; a clamp turns a typo into a bill and hides it. Every validation runs + * before the first store write, so a 400 leaves the durable record byte-identical. + * + * **Arming pins, and only unpins what it pinned.** `thread.pin`'s decider case emits + * companion `thread.unsettled` and `thread.unsnoozed` events — it is a promotion, not a + * decoration. So arming a *snoozed* thread is `400 thread_snoozed` (cancelling a human's + * snooze is the human's decision, not a supervisor's), arming a *settled* thread is fine + * (that is what the human asked for), and `pinnedByLoop` — recorded true only when + * `pinnedAt` was null before the arm — gates the unpin so disarming never removes a pin the + * user set themselves. + * + * @module coil/loop/http + */ + +import type { OrchestrationThreadShell } from "@t3tools/contracts"; +import { AuthOrchestrationOperateScope, CommandId, ThreadId } from "@t3tools/contracts"; +import * as Clock from "effect/Clock"; +import * as Crypto from "effect/Crypto"; +import * as Effect from "effect/Effect"; +import * as Layer from "effect/Layer"; +import * as Option from "effect/Option"; +import * as Schema from "effect/Schema"; +import { HttpRouter, HttpServerRequest, HttpServerResponse } from "effect/unstable/http"; + +import { OrchestrationEngineService } from "../../orchestration/Services/OrchestrationEngine.ts"; +import type { OrchestrationEngineShape } from "../../orchestration/Services/OrchestrationEngine.ts"; +import { ProjectionSnapshotQuery } from "../../orchestration/Services/ProjectionSnapshotQuery.ts"; +import type { ProjectionSnapshotQueryShape } from "../../orchestration/Services/ProjectionSnapshotQuery.ts"; +import { authenticateWithScope, routeAuthErrorTags } from "../http/auth.ts"; +import { atArmedCeiling, hasPendingCrons } from "./guards.ts"; +import { + Blocker, + CheckInRow, + LoopGlobalSettings, + LoopRecord, + LoopStore, + type LoopStoreShape, +} from "./state.ts"; + +export const LOOP_ROUTE_PATH = "/api/coil/loop"; +export const LOOPS_ROUTE_PATH = "/api/coil/loops"; +export const LOOP_SETTINGS_ROUTE_PATH = "/api/coil/loop/settings"; +export const LOOP_ANSWER_ROUTE_PATH = "/api/coil/loop/answer"; + +/** The hard ceiling on a single run's check-ins. A request above it is a 400, never a clamp. */ +export const MAX_CHECK_INS = 20; + +/** Operate (not read) scope on every route: these describe and mutate scheduling. */ +const authenticate = authenticateWithScope(AuthOrchestrationOperateScope); + +// --- wire shapes ------------------------------------------------------------ + +/** + * The route's reading of the state machine, from the durable record plus one thread shell. + * + * Deliberately *not* the reactor's verdict. The reactor owns guard order, the wake-grace + * arithmetic and the stop sweep; this is the subset a request can state truthfully without + * a tick, so the console can render before the next poll. Where the two could disagree — + * a wake that is late but still inside its grace — this reports the conservative reading + * (`self_pacing` only while the wake is still in the future) and lets the reactor decide + * whether the wake was lost. + */ +export const LoopDerivedView = Schema.Struct({ + state: Schema.Literals([ + "off", + "watching", + "self_pacing", + "standing_down", + "held", + "blocked", + "stopped", + ]), + /** Why, when the state alone does not say: `disabled`, `rate_limited`, `pending_input`, … */ + reason: Schema.NullOr(Schema.String), + stoppedReason: Schema.NullOr(Schema.Literals(["done", "spent", "stalled", "handed-back"])), + checkInsUsed: Schema.Number, + maxCheckIns: Schema.Number, + deadlineAtMs: Schema.Number, + msUntilDeadline: Schema.Number, + rateLimitedUntilMs: Schema.Number, + /** Earliest recorded wake that parsed, past or future. `null` = no deference available. */ + nextWakeAtMs: Schema.NullOr(Schema.Number), + snoozedUntilMs: Schema.NullOr(Schema.Number), + /** False when the thread has no shell — deleted, archived away, or never existed. */ + threadKnown: Schema.Boolean, + globalEnabled: Schema.Boolean, + armedCount: Schema.Number, + maxArmedThreads: Schema.Number, +}); +export type LoopDerivedView = typeof LoopDerivedView.Type; + +/** + * `blockers` and `ledger` are projections of `record`, lifted to the top level because they + * are the console's two main sections: `blockers` is the *unanswered* subset (what is + * actionable now), `ledger` is the iteration history. + */ +export const LoopView = Schema.Struct({ + threadId: Schema.String, + record: LoopRecord, + derived: LoopDerivedView, + blockers: Schema.Array(Blocker), + ledger: Schema.Array(CheckInRow), +}); +export type LoopView = typeof LoopView.Type; + +export const LoopListView = Schema.Struct({ loops: Schema.Array(LoopView) }); +export type LoopListView = typeof LoopListView.Type; + +export const LoopSettingsView = Schema.Struct({ + ...LoopGlobalSettings.fields, + /** How many threads are armed right now, against `maxArmedThreads`. */ + armedCount: Schema.Number, +}); +export type LoopSettingsView = typeof LoopSettingsView.Type; + +const WriteBody = Schema.Struct({ + threadId: Schema.String, + action: Schema.Literals(["arm", "rearm", "edit", "disarm", "clear"]), + // Nullable *and* optional: `deadline_required` must be able to tell "absent" from a + // deliberate null, and both are the same refusal. + maxCheckIns: Schema.optionalKey(Schema.NullOr(Schema.Number)), + deadlineAtMs: Schema.optionalKey(Schema.NullOr(Schema.Number)), + goal: Schema.optionalKey(Schema.NullOr(Schema.String)), + idleMs: Schema.optionalKey(Schema.NullOr(Schema.Number)), + busyIdleMs: Schema.optionalKey(Schema.NullOr(Schema.Number)), + overridePrompt: Schema.optionalKey(Schema.NullOr(Schema.String)), +}); +type WriteBody = typeof WriteBody.Type; +const decodeWriteBody = Schema.decodeUnknownEffect(WriteBody); + +const SettingsBody = Schema.Struct({ + enabled: Schema.optionalKey(Schema.Boolean), + maxArmedThreads: Schema.optionalKey(Schema.Number), + defaultMaxCheckIns: Schema.optionalKey(Schema.Number), + defaultRunMs: Schema.optionalKey(Schema.Number), + defaultIdleMs: Schema.optionalKey(Schema.Number), + defaultBusyIdleMs: Schema.optionalKey(Schema.Number), +}); +type SettingsBody = typeof SettingsBody.Type; +const decodeSettingsBody = Schema.decodeUnknownEffect(SettingsBody); + +/** + * `POST /api/coil/loop/answer`. + * + * **The blocker half only, and the field is named `blockerId` rather than §9's `id`.** §9 sketched + * one polymorphic `id` routing two mechanisms — a native pending input to + * `thread.user-input.respond`, a deferred blocker to this store — behind a single console control. + * The native half is not built and this route does not stand in for it: a native + * `AskUserQuestion` is already rendered live and answerable by `ComposerPendingUserInputPanel` + * inside the composer, so the console names it and points there rather than cloning the control. + * That leaves upstream's answer path with exactly one instance of itself on the page, and it is + * also the honest reading of the §9 note that this route cannot build a correct `answers` map: + * upstream keys answers by *question* id while the phase-1 `UserInputRecord` stores one + * `requestId` and one question string. An explicit `blockerId` says which of the two mechanisms + * this route serves; a `requestId` field is what the native half would add if it is ever built. + */ +const AnswerBody = Schema.Struct({ + threadId: Schema.String, + blockerId: Schema.String, + answer: Schema.String, +}); +type AnswerBody = typeof AnswerBody.Type; +const decodeAnswerBody = Schema.decodeUnknownEffect(AnswerBody); + +// --- responses -------------------------------------------------------------- + +/** + * Every refusal carries a machine-readable code, because the console words them + * differently: "you must pick an end time" is a different sentence from "that end time has + * already passed", and a bare 400 collapses them. + */ +const fail = (status: number, error: string, detail?: Record) => + HttpServerResponse.jsonUnsafe({ error, ...detail }, { status }); + +/** + * The request body as JSON, or `unknown` that will never decode. + * + * `request.json` *fails* on a body that is not JSON at all, and that failure is not one of + * `routeAuthErrorTags` — so without this it escapes the handler and the client gets a bare + * empty 400 instead of the `invalid_body` code the console words. Every refusal on these + * routes carries a machine-readable code; a malformed body is not the exception. + */ +const readJsonBody = (request: HttpServerRequest.HttpServerRequest): Effect.Effect => + request.json.pipe(Effect.orElseSucceed(() => null as unknown)); + +// --- derivation ------------------------------------------------------------- + +const parseIsoMs = (value: string | null | undefined): number | null => { + if (typeof value !== "string" || value.length === 0) return null; + const parsed = Date.parse(value); + return Number.isFinite(parsed) ? parsed : null; +}; + +/** Earliest recorded wake that parsed. `null` entries mean "no deference", not "now". */ +const earliestWakeMs = (record: LoopRecord): number | null => { + let earliest: number | null = null; + for (const entry of record.crons?.entries ?? []) { + const next = entry.nextFireAtMs; + if (next === null) continue; + if (earliest === null || next < earliest) earliest = next; + } + return earliest; +}; + +const deriveView = (input: { + readonly record: LoopRecord; + readonly shell: OrchestrationThreadShell | null; + readonly global: LoopGlobalSettings; + readonly armedCount: number; + readonly nowMs: number; +}): LoopDerivedView => { + const { record, shell, global, nowMs } = input; + const snoozedUntilMs = parseIsoMs(shell?.snoozedUntil ?? null); + const nextWakeAtMs = earliestWakeMs(record); + + const base = { + reason: null as string | null, + stoppedReason: record.stopped?.reason ?? null, + checkInsUsed: record.checkInsUsed, + maxCheckIns: record.maxCheckIns, + deadlineAtMs: record.deadlineAtMs, + msUntilDeadline: Math.max(0, record.deadlineAtMs - nowMs), + rateLimitedUntilMs: record.rateLimitedUntilMs, + nextWakeAtMs, + snoozedUntilMs, + threadKnown: shell !== null, + globalEnabled: global.enabled, + armedCount: input.armedCount, + maxArmedThreads: global.maxArmedThreads, + }; + + // Terminal first: a stop is sticky and outranks every live reading, including the master + // toggle. Then `off`, so an unarmed thread never reports a guard's opinion of it. + if (record.stopped !== null) { + return { ...base, state: "stopped", reason: record.stopped.reason }; + } + if (!record.armed) return { ...base, state: "off" }; + // Guard 2: the toggle stands loops down; it disarms and stops nothing. + if (!global.enabled) return { ...base, state: "standing_down", reason: "disabled" }; + // `held`, matching guard 6's phase and `status.ts`: a snooze is a bounded hold with an + // expiry, not a question waiting on an answer. + if (snoozedUntilMs !== null && snoozedUntilMs > nowMs) { + return { ...base, state: "held", reason: "snoozed" }; + } + // Guard 8, including the plan clause: a thread parked on an unapproved plan is waiting on + // a human even though `hasPendingUserInput` is false. + if ( + shell !== null && + (shell.hasPendingApprovals || shell.hasPendingUserInput || shell.hasActionableProposedPlan) + ) { + return { ...base, state: "blocked", reason: "pending_input" }; + } + if (nowMs < record.rateLimitedUntilMs) { + return { ...base, state: "held", reason: "rate_limited" }; + } + // Guard 10b, conservatively: only a wake still *ahead of us* and inside the run's deadline + // is visible deference. A wake past due is the reactor's call, since whether it is merely + // late or genuinely lost depends on the derived grace. + if (nextWakeAtMs !== null && nextWakeAtMs > nowMs && nextWakeAtMs <= record.deadlineAtMs) { + return { ...base, state: "self_pacing", reason: null }; + } + // Guard 14, last as in the guard table, and through the guard's own predicate so the + // console can never say "Watching" about a loop the supervisor is standing down. Without + // it the ceiling is the one stand-down with no lens at all: the loop goes quiet and the + // panel keeps claiming it is running. + if (atArmedCeiling(input.armedCount, global)) { + return { ...base, state: "standing_down", reason: "ceiling" }; + } + return { ...base, state: "watching" }; +}; + +// --- validation ------------------------------------------------------------- + +/** A finite number the caller actually sent, as opposed to absent or null. */ +const provided = (value: number | null | undefined): value is number => + typeof value === "number" && Number.isFinite(value); + +interface ArmBounds { + readonly deadlineAtMs: number; + readonly maxCheckIns: number; +} + +/** + * The arm-time bounds check, in the order §9 lists the codes. + * + * Returns an error code rather than a response so `edit` can reuse the same rules on the + * fields it was actually given. + */ +const validateArmBounds = (body: WriteBody, nowMs: number): ArmBounds | string => { + if (body.deadlineAtMs === undefined || body.deadlineAtMs === null) return "deadline_required"; + if (!Number.isFinite(body.deadlineAtMs)) return "invalid_body"; + if (body.deadlineAtMs <= nowMs) return "deadline_in_past"; + if (body.maxCheckIns === undefined || body.maxCheckIns === null) return "budget_required"; + if (!Number.isInteger(body.maxCheckIns)) return "invalid_body"; + if (body.maxCheckIns > MAX_CHECK_INS) return "budget_too_large"; + if (body.maxCheckIns < 1) return "budget_too_small"; + return { deadlineAtMs: body.deadlineAtMs, maxCheckIns: body.maxCheckIns }; +}; + +/** The optional per-thread thresholds, shared by arm and edit. Never clamped either. */ +const validateThresholds = (body: WriteBody): string | null => { + for (const value of [body.idleMs, body.busyIdleMs]) { + if (value === undefined || value === null) continue; + if (!Number.isFinite(value) || value <= 0) return "invalid_body"; + } + return null; +}; + +const SETTINGS_BOUNDS = { + maxArmedThreads: { min: 1, max: 100, integer: true }, + defaultMaxCheckIns: { min: 1, max: MAX_CHECK_INS, integer: true }, + defaultRunMs: { min: 1, max: Number.MAX_SAFE_INTEGER, integer: false }, + defaultIdleMs: { min: 1, max: Number.MAX_SAFE_INTEGER, integer: false }, + defaultBusyIdleMs: { min: 1, max: Number.MAX_SAFE_INTEGER, integer: false }, +} as const; + +const validateSettings = (body: SettingsBody): string | null => { + for (const [key, bounds] of Object.entries(SETTINGS_BOUNDS)) { + const value = body[key as keyof typeof SETTINGS_BOUNDS]; + if (value === undefined) continue; + if (!Number.isFinite(value)) return "invalid_body"; + if (bounds.integer && !Number.isInteger(value)) return "invalid_body"; + if (value < bounds.min || value > bounds.max) return "out_of_range"; + } + return null; +}; + +// --- handlers --------------------------------------------------------------- + +interface RouteDeps { + readonly store: LoopStoreShape; + readonly engine: OrchestrationEngineShape; + readonly snapshotQuery: ProjectionSnapshotQueryShape; + readonly crypto: Crypto.Crypto; +} + +/** + * A shell read that distinguishes "no such thread" from "the projection is unavailable". + * + * Collapsing them would report a transient SQL failure as a deleted thread, and the console + * would offer to re-arm a loop that is fine. + */ +const readShell = (deps: RouteDeps, threadId: string) => + deps.snapshotQuery.getThreadShellById(ThreadId.make(threadId)).pipe( + Effect.map((option) => ({ ok: true, shell: Option.getOrNull(option) }) as const), + Effect.catch((cause) => + Effect.logWarning("coil loop: thread shell lookup failed", { threadId, cause }).pipe( + Effect.as({ ok: false, shell: null } as const), + ), + ), + ); + +/** + * Dispatches a pin or unpin, reporting whether it landed. + * + * Best-effort by design, but the *result* is load-bearing: `pinnedByLoop` records what + * actually happened, so a pin that failed can never authorise an unpin later. + */ +const dispatchPinCommand = (deps: RouteDeps, threadId: string, type: "thread.pin" | "unpin") => + Effect.gen(function* () { + const uuid = yield* deps.crypto.randomUUIDv4; + yield* deps.engine.dispatch( + type === "thread.pin" + ? { + type: "thread.pin", + commandId: CommandId.make(`coil-loop-pin:${uuid}`), + threadId: ThreadId.make(threadId), + } + : { + type: "thread.unpin", + commandId: CommandId.make(`coil-loop-unpin:${uuid}`), + threadId: ThreadId.make(threadId), + }, + ); + return true; + }).pipe( + Effect.catchCause((cause) => + Effect.logWarning("coil loop: pin command failed", { threadId, type, cause }).pipe( + Effect.as(false), + ), + ), + ); + +const buildView = (deps: RouteDeps, threadId: string, shell: OrchestrationThreadShell | null) => + Effect.gen(function* () { + const nowMs = yield* Clock.currentTimeMillis; + const record = yield* deps.store.getThread(threadId); + const global = yield* deps.store.getGlobal; + const armed = yield* deps.store.listArmed; + return { + threadId, + record, + derived: deriveView({ record, shell, global, armedCount: armed.length, nowMs }), + blockers: record.blockers.filter((entry) => entry.answeredAtMs === null), + ledger: record.checkIns, + } satisfies LoopView; + }); + +const respondWithView = ( + deps: RouteDeps, + threadId: string, + shell: OrchestrationThreadShell | null, +) => Effect.map(buildView(deps, threadId, shell), (view) => HttpServerResponse.jsonUnsafe(view)); + +const settingsView = (deps: RouteDeps) => + Effect.gen(function* () { + const global = yield* deps.store.getGlobal; + const armed = yield* deps.store.listArmed; + return { ...global, armedCount: armed.length } satisfies LoopSettingsView; + }); + +/** + * Arm or re-arm. + * + * One handler for both because a re-arm *is* a fresh run — the store clears the terminal + * state, the budget and the ledger — and the two actions differ only in what the console + * called the button. + */ +const handleArm = (deps: RouteDeps, body: WriteBody, nowMs: number) => + Effect.gen(function* () { + const bounds = validateArmBounds(body, nowMs); + if (typeof bounds === "string") return fail(400, bounds); + const thresholdError = validateThresholds(body); + if (thresholdError !== null) return fail(400, thresholdError); + + // Guard 14 at the route. The thread's own arm does not count against the ceiling, so + // re-arming an already-armed loop is never refused by it. + const global = yield* deps.store.getGlobal; + const armed = yield* deps.store.listArmed; + const others = armed.filter((entry) => entry.threadId !== body.threadId).length; + if (others >= global.maxArmedThreads) { + return fail(400, "ceiling_reached", { armedCount: others, max: global.maxArmedThreads }); + } + + const lookup = yield* readShell(deps, body.threadId); + if (!lookup.ok) return fail(503, "projection_unavailable"); + if (lookup.shell === null) return fail(404, "unknown_thread"); + + // `thread.pin` clears a snooze as a side effect of promotion, so refuse rather than + // silently cancel one. Checked before any dispatch AND before any store write. + const snoozedUntilMs = parseIsoMs(lookup.shell.snoozedUntil ?? null); + if (snoozedUntilMs !== null && snoozedUntilMs > nowMs) { + return fail(400, "thread_snoozed", { snoozedUntilMs }); + } + + // Only a pin this route created may ever be removed by this route — and a re-arm must + // not disown one it already owns. The thread is pinned on a re-arm precisely *because* + // the previous arm pinned it, so reading "already pinned" as "the user's pin" would + // orphan it: `pinnedByLoop` would drop to false and no disarm would ever unpin. + const previous = yield* deps.store.getThread(body.threadId); + const alreadyPinned = parseIsoMs(lookup.shell.pinnedAt ?? null) !== null; + const pinnedByLoop = alreadyPinned + ? previous.pinnedByLoop + : yield* dispatchPinCommand(deps, body.threadId, "thread.pin"); + + yield* deps.store.arm({ + threadId: body.threadId, + armedAtMs: nowMs, + deadlineAtMs: bounds.deadlineAtMs, + maxCheckIns: bounds.maxCheckIns, + goal: body.goal ?? null, + ...(provided(body.idleMs) ? { idleMs: body.idleMs } : {}), + ...(provided(body.busyIdleMs) ? { busyIdleMs: body.busyIdleMs } : {}), + ...(body.overridePrompt === undefined ? {} : { overridePrompt: body.overridePrompt }), + pinnedByLoop, + }); + + return yield* respondWithView(deps, body.threadId, lookup.shell); + }); + +/** + * Disarm — the human taking over. + * + * Writes the sticky `handed-back` terminal rather than merely clearing `armed`, because + * budget is deliberately *not* reset: deciding to stop a thread must not hand the next + * loop a fresh six. + * + * Needs no shell, and that is deliberate: disarming must keep working when the thread has + * been deleted from under the record. A one-way door is a bug. + * + * **Refused unless the loop is armed.** A stale tab holding a disarm button would otherwise + * overwrite a `done` or `spent` terminal with `handed-back` hours later, rewriting how the + * night ended, and a disarm of a thread that never had a loop would mint a terminal record + * for a run that never existed. + */ +const handleDisarm = (deps: RouteDeps, body: WriteBody, nowMs: number) => + Effect.gen(function* () { + const record = yield* deps.store.getThread(body.threadId); + if (!record.armed) return fail(409, "not_armed"); + // Cleared only when the unpin actually landed. A failed unpin leaves the flag set, so a + // later disarm can still remove the pin the loop is still responsible for. + const unpinned = + record.pinnedByLoop && (yield* dispatchPinCommand(deps, body.threadId, "unpin")); + + yield* deps.store.stop(body.threadId, { + reason: "handed-back", + atMs: nowMs, + detail: "disarmed from the console", + }); + if (unpinned) { + yield* deps.store.update(body.threadId, (current) => ({ ...current, pinnedByLoop: false })); + } + // The agent's own wakes outlive the record that bounded them, and this route cannot end a + // session itself. Bank the request; the supervisor's next tick issues the one `stopSession` + // — the same call its own disarm path makes, on the same code path. + if (hasPendingCrons(record, nowMs)) { + yield* deps.store.requestSessionStop(body.threadId, nowMs); + } + const lookup = yield* readShell(deps, body.threadId); + return yield* respondWithView(deps, body.threadId, lookup.shell); + }); + +/** + * Clear a finished run — the reverse of arming, for a loop that already ended. + * + * Without it the stopped pill and its bounds sit above the composer forever unless the thread + * is armed again, which is a one-way door in the other direction. Refused while the loop is + * armed, so this can never be a way to end a live run silently: that is `disarm`, which + * records why. + */ +const handleClear = (deps: RouteDeps, body: WriteBody) => + Effect.gen(function* () { + const record = yield* deps.store.getThread(body.threadId); + if (record.armed) return fail(409, "armed"); + yield* deps.store.clearThread(body.threadId); + const lookup = yield* readShell(deps, body.threadId); + return yield* respondWithView(deps, body.threadId, lookup.shell); + }); + +/** + * Edit the bounds of a run in flight, without touching its budget or its ledger. + * + * Every field is optional and every one that is present is validated with the same rules + * arming uses — extending a deadline into the past is as wrong here as it is there. + * + * **Refused unless the loop is armed**, for the same reason `disarm` is: editing the bounds + * of a run that is over would resurrect its numbers without resurrecting the run, and + * editing a thread that never had a loop would create a record out of a form submission. + */ +const handleEdit = (deps: RouteDeps, body: WriteBody, nowMs: number) => + Effect.gen(function* () { + if (!(yield* deps.store.getThread(body.threadId)).armed) return fail(409, "not_armed"); + if (body.deadlineAtMs !== undefined && body.deadlineAtMs !== null) { + if (!Number.isFinite(body.deadlineAtMs)) return fail(400, "invalid_body"); + if (body.deadlineAtMs <= nowMs) return fail(400, "deadline_in_past"); + } + if (body.maxCheckIns !== undefined && body.maxCheckIns !== null) { + if (!Number.isInteger(body.maxCheckIns)) return fail(400, "invalid_body"); + if (body.maxCheckIns > MAX_CHECK_INS) return fail(400, "budget_too_large"); + if (body.maxCheckIns < 1) return fail(400, "budget_too_small"); + } + const thresholdError = validateThresholds(body); + if (thresholdError !== null) return fail(400, thresholdError); + + yield* deps.store.update(body.threadId, (record) => ({ + ...record, + ...(provided(body.deadlineAtMs) ? { deadlineAtMs: body.deadlineAtMs } : {}), + ...(provided(body.maxCheckIns) ? { maxCheckIns: body.maxCheckIns } : {}), + ...(provided(body.idleMs) ? { idleMs: body.idleMs } : {}), + ...(provided(body.busyIdleMs) ? { busyIdleMs: body.busyIdleMs } : {}), + ...(body.goal === undefined ? {} : { goal: body.goal }), + ...(body.overridePrompt === undefined ? {} : { overridePrompt: body.overridePrompt }), + })); + + const lookup = yield* readShell(deps, body.threadId); + return yield* respondWithView(deps, body.threadId, lookup.shell); + }); + +// --- routes ----------------------------------------------------------------- + +const makeGetLoopRoute = (deps: RouteDeps) => + HttpRouter.add( + "GET", + LOOP_ROUTE_PATH, + Effect.gen(function* () { + yield* authenticate; + const request = yield* HttpServerRequest.HttpServerRequest; + const url = HttpServerRequest.toURL(request); + if (Option.isNone(url)) return fail(400, "invalid_request"); + const threadId = url.value.searchParams.get("threadId"); + if (threadId === null || threadId === "") return fail(400, "missing_thread_id"); + + // An unknown thread is a 200 with the fail-closed "off" record, NOT a 404: the console + // opens on every thread and must render "no loop here" without an error path. + const lookup = yield* readShell(deps, threadId); + return yield* respondWithView(deps, threadId, lookup.shell); + }).pipe(Effect.catchTags(routeAuthErrorTags)), + ); + +const makePostLoopRoute = (deps: RouteDeps) => + HttpRouter.add( + "POST", + LOOP_ROUTE_PATH, + Effect.gen(function* () { + yield* authenticate; + const request = yield* HttpServerRequest.HttpServerRequest; + const body = yield* readJsonBody(request).pipe( + Effect.flatMap(decodeWriteBody), + Effect.map((decoded): WriteBody | null => decoded), + Effect.orElseSucceed(() => null), + ); + if (body === null || body.threadId.trim() === "") return fail(400, "invalid_body"); + + const nowMs = yield* Clock.currentTimeMillis; + switch (body.action) { + case "arm": + case "rearm": + return yield* handleArm(deps, body, nowMs); + case "disarm": + return yield* handleDisarm(deps, body, nowMs); + case "edit": + return yield* handleEdit(deps, body, nowMs); + case "clear": + return yield* handleClear(deps, body); + } + }).pipe(Effect.catchTags(routeAuthErrorTags)), + ); + +const makeListLoopsRoute = (deps: RouteDeps) => + HttpRouter.add( + "GET", + LOOPS_ROUTE_PATH, + Effect.gen(function* () { + yield* authenticate; + const armed = yield* deps.store.listArmed; + // Sorted by threadId: `Object.entries` order is insertion order, which depends on the + // sequence of arms and would make this response unstable across restarts. + const ordered = [...armed].sort((a, b) => (a.threadId < b.threadId ? -1 : 1)); + const loops: Array = []; + for (const entry of ordered) { + const lookup = yield* readShell(deps, entry.threadId); + loops.push(yield* buildView(deps, entry.threadId, lookup.shell)); + } + return HttpServerResponse.jsonUnsafe({ loops } satisfies LoopListView); + }).pipe(Effect.catchTags(routeAuthErrorTags)), + ); + +const makeGetSettingsRoute = (deps: RouteDeps) => + HttpRouter.add( + "GET", + LOOP_SETTINGS_ROUTE_PATH, + Effect.gen(function* () { + yield* authenticate; + return HttpServerResponse.jsonUnsafe(yield* settingsView(deps)); + }).pipe(Effect.catchTags(routeAuthErrorTags)), + ); + +const makePostSettingsRoute = (deps: RouteDeps) => + HttpRouter.add( + "POST", + LOOP_SETTINGS_ROUTE_PATH, + Effect.gen(function* () { + yield* authenticate; + const request = yield* HttpServerRequest.HttpServerRequest; + const body = yield* readJsonBody(request).pipe( + Effect.flatMap(decodeSettingsBody), + Effect.map((decoded): SettingsBody | null => decoded), + Effect.orElseSucceed(() => null), + ); + if (body === null) return fail(400, "invalid_body"); + const error = validateSettings(body); + if (error !== null) return fail(400, error); + + // Lowering `maxArmedThreads` below the current armed count is accepted on purpose: + // the excess loops stand down at the next tick with their budgets intact. The toggle + // and its ceiling are guards, not a lifecycle, so neither ever disarms anything. + yield* deps.store.setGlobal(body); + return HttpServerResponse.jsonUnsafe(yield* settingsView(deps)); + }).pipe(Effect.catchTags(routeAuthErrorTags)), + ); + +/** + * Answer one deferred blocker. + * + * Idempotent by construction: `store.answerBlocker` keeps the first answer, so a double-click or + * a retried request is not a second append and never overwrites what was already banked. + * `deliveredToAgent` stays false — the answer is owed to the agent, and the next check-in prompt + * is what discharges it. + */ +const makeAnswerRoute = (deps: RouteDeps) => + HttpRouter.add( + "POST", + LOOP_ANSWER_ROUTE_PATH, + Effect.gen(function* () { + yield* authenticate; + const request = yield* HttpServerRequest.HttpServerRequest; + const body = yield* readJsonBody(request).pipe( + Effect.flatMap(decodeAnswerBody), + Effect.map((decoded): AnswerBody | null => decoded), + Effect.orElseSucceed(() => null), + ); + if (body === null || body.threadId.trim() === "" || body.blockerId.trim() === "") { + return fail(400, "invalid_body"); + } + + const nowMs = yield* Clock.currentTimeMillis; + const answered = yield* deps.store.answerBlocker( + body.threadId, + body.blockerId, + body.answer, + nowMs, + ); + // An id nobody raised is a 404, not a silent success: the console would otherwise show an + // answer as banked while nothing was recorded. + if (answered === null) return fail(404, "not_found"); + return HttpServerResponse.jsonUnsafe({ ok: true }); + }).pipe(Effect.catchTags(routeAuthErrorTags)), + ); + +/** + * Mounted from `coil/index.ts`, which discharges `LoopStore`. + * + * `Layer.unwrap` resolves the services once at *layer construction* and the handlers close + * over the values, so the handlers' own requirement stays `never`. That is what keeps + * `LoopStore` — a fork-only service — out of the type of upstream's `makeRoutesLayer`; a + * fork change must never widen an upstream signature. + * + * `OrchestrationEngineService`, `ProjectionSnapshotQuery` and `Crypto` are deliberately left + * open on the layer rather than provided here: all three are already requirements of + * `makeRoutesLayer` (upstream's own orchestration API layer needs the first two), so leaving + * them unsatisfied adds nothing to that signature and avoids constructing a second engine. + */ +export const loopRouteLayer = Layer.unwrap( + Effect.gen(function* () { + const deps: RouteDeps = { + store: yield* LoopStore, + engine: yield* OrchestrationEngineService, + snapshotQuery: yield* ProjectionSnapshotQuery, + crypto: yield* Crypto.Crypto, + }; + return Layer.mergeAll( + makeGetLoopRoute(deps), + makePostLoopRoute(deps), + makeListLoopsRoute(deps), + makeGetSettingsRoute(deps), + makePostSettingsRoute(deps), + makeAnswerRoute(deps), + ); + }), +); diff --git a/apps/server/src/coil/loop/integration.test.ts b/apps/server/src/coil/loop/integration.test.ts new file mode 100644 index 000000000000..8c38aa833a01 --- /dev/null +++ b/apps/server/src/coil/loop/integration.test.ts @@ -0,0 +1,613 @@ +/** + * The scenarios that motivated the feature. TESTS.md cases 128–137. + * + * Each one replays a real failure, or a real *non*-failure the design must not break. They + * are heavier than the unit cases on purpose: the reactor, the store, the decision table, + * the guards, the cron deference and the sentinel are all real, and the only doubles are the + * projection rows, the engine and the provider stream. + * + * The two headline acceptance tests are **136b** (a healthy self-pacing agent must be left + * completely alone) and **136c** (a wake lost to a restart must be covered exactly once). + * How rarely this reactor fires is the measure of a correct implementation, so a regression + * that makes it chattier shows up as a failure here rather than as a surprise at 03:00. + * + * @module coil/loop/integration.test + */ + +import * as NodeServices from "@effect/platform-node/NodeServices"; +import { assert, describe, it } from "@effect/vitest"; +import * as Effect from "effect/Effect"; +import type * as FileSystem from "effect/FileSystem"; +import * as Layer from "effect/Layer"; +import type * as Path from "effect/Path"; +import * as Ref from "effect/Ref"; +import type * as Scope from "effect/Scope"; +import * as TestClock from "effect/testing/TestClock"; + +import { LOOP_ACTIVITY_KINDS, LoopReactorLive } from "./Reactor.ts"; +import { + activitiesOfKind, + advancePolls, + advanceUntil, + clearLoopEnv, + harness, + HOUR, + LOOP_THREAD_ID, + MINUTE, + msToIso, + rateLimitEvent, + threadShell, + turnStarts, + turnStartsAtLeast, + untilReceipt, + writeSentinel, +} from "./reactorHarness.ts"; +import { makeLoopStore, type LoopStoreShape } from "./state.ts"; + +clearLoopEnv(); + +type OuterR = Scope.Scope | FileSystem.FileSystem | Path.Path; + +const scoped = (body: Effect.Effect) => + body.pipe(Effect.scoped, Effect.provide(Layer.mergeAll(NodeServices.layer, TestClock.layer()))); + +const withReactor = ( + deps: Layer.Layer, + body: Effect.Effect, +) => body.pipe(Effect.provide(LoopReactorLive.pipe(Layer.provideMerge(deps)))); + +const arm = ( + store: LoopStoreShape, + o: { + readonly threadId?: string; + readonly maxCheckIns?: number; + readonly deadlineAtMs?: number; + readonly armedAtMs?: number; + } = {}, +) => + Effect.gen(function* () { + yield* store.setGlobal({ enabled: true }); + yield* store.arm({ + threadId: o.threadId ?? LOOP_THREAD_ID, + armedAtMs: o.armedAtMs ?? 0, + deadlineAtMs: o.deadlineAtMs ?? 8 * HOUR, + maxCheckIns: o.maxCheckIns ?? 6, + }); + }); + +const record = (store: LoopStoreShape, threadId: string = LOOP_THREAD_ID) => + store.getThread(threadId); + +/** + * A recurring wake the agent scheduled for itself. + * + * `nextFireAtMs` is what the `Stop` hook recorded; `schedule` is the expression the fork + * parses to derive the grace. A 20-minute period gives a 2-minute grace (10% of the period, + * floored at 90 s), which is what makes "late but healthy" distinguishable from "lost". + */ +const scheduleWake = (store: LoopStoreShape, nextFireAtMs: number, recordedAtMs: number) => + store.setCrons(LOOP_THREAD_ID, { + recordedAtMs, + entries: [ + { + id: "cron-1", + schedule: "*/20 * * * *", + recurring: true, + prompt: "continue the build loop", + nextFireAtMs, + }, + ], + }); + +describe("Loops — the scenarios that motivated the feature", () => { + // --- 136b: the regression test for "T3 does not fight the agent" ----------- + + it.effect("136b: a healthy self-pacing agent is left completely alone for three hours", () => + Effect.gen(function* () { + const h = yield* harness({ shell: threadShell() }); + yield* arm(h.store, { deadlineAtMs: 8 * HOUR }); + yield* withReactor( + h.deps, + Effect.gen(function* () { + // Three hours of a well-behaved agent: it wakes itself every twenty minutes, and + // each wake moves `updatedAt`. The reactor should never once decide it is needed. + for (let wake = 1; wake <= 9; wake++) { + const wakeAtMs = wake * 20 * MINUTE; + yield* scheduleWake(h.store, wakeAtMs, wakeAtMs - 20 * MINUTE); + yield* advancePolls(20); + // The wake landed: the agent worked, so the row moved just past its own wake. + yield* Ref.set(h.shellRef, threadShell({ updatedAt: msToIso(wakeAtMs + MINUTE) })); + } + + assert.strictEqual( + turnStarts(yield* Ref.get(h.dispatched)).length, + 0, + "T3 must not nudge a thread that is pacing itself", + ); + const after = yield* record(h.store); + assert.strictEqual(after.checkInsUsed, 0, "and it must spend nothing doing so"); + assert.isTrue(after.armed, "while staying armed, so the deadline still applies"); + assert.isNull(after.degraded, "and nothing is reported as degraded"); + }), + ); + }).pipe(scoped), + ); + + // --- 136c: the durability gap, as a test ---------------------------------- + + it.effect("136c: a wake lost to a restart is noticed and covered exactly once", () => + Effect.gen(function* () { + // The agent scheduled a wake for +25 min. The provider's cron table is in-process + // (`cron_durable` is false), so a restart at +10 min destroys it and leaves no trace + // anywhere except T3's own record. This is the one thing T3 can supply that the binary + // cannot, and it is the strongest trigger in the design. + const wakeAtMs = 25 * MINUTE; + const h = yield* harness({ shell: threadShell({ updatedAt: msToIso(5 * MINUTE) }) }); + yield* arm(h.store); + yield* scheduleWake(h.store, wakeAtMs, 0); + + // The record survives the process that wrote it: a second store over the same file + // sees the armed run and the wake it was waiting on. + const rebuilt = yield* makeLoopStore(h.statePath); + const survived = yield* rebuilt.getThread(LOOP_THREAD_ID); + assert.isTrue(survived.armed, "the arm survives a restart"); + assert.strictEqual(survived.crons?.entries[0]?.nextFireAtMs, wakeAtMs); + + yield* withReactor( + h.deps, + Effect.gen(function* () { + // Up to the wake plus its 2-minute grace, T3 stands by: an overdue wake is not yet + // a lost one, and firing here would fight a merely jittered scheduler. + yield* advancePolls(26); + assert.strictEqual( + turnStarts(yield* Ref.get(h.dispatched)).length, + 0, + "a wake one minute late is late, not lost", + ); + + yield* advanceUntil( + Ref.get(h.dispatched).pipe( + Effect.map((all) => activitiesOfKind(all, LOOP_ACTIVITY_KINDS.wakeLost).length > 0), + ), + "the covered wake", + 20, + ); + assert.strictEqual( + turnStarts(yield* Ref.get(h.dispatched)).length, + 1, + "covered exactly once", + ); + const notes = activitiesOfKind( + yield* Ref.get(h.dispatched), + LOOP_ACTIVITY_KINDS.wakeLost, + ); + assert.strictEqual(notes.length, 1, "and said so, once"); + assert.strictEqual(notes[0]!.tone, "error"); + assert.deepStrictEqual((notes[0]!.payload as { cronId: string }).cronId, "cron-1"); + assert.strictEqual((yield* record(h.store)).degraded, "wake_lost"); + + // And it does not keep covering it. The check-in floor and the spent reservation + // together mean one lost wake costs one check-in. + yield* advancePolls(10); + assert.strictEqual(turnStarts(yield* Ref.get(h.dispatched)).length, 1); + }), + ); + }).pipe(scoped), + ); + + // --- 128: the original night ---------------------------------------------- + + it.effect("128: the original night — quiet only after the activity stream stops", () => + Effect.gen(function* () { + // The turn completed at 00:19 while `task.*` activities kept arriving until 00:52, then + // silence. Every one of those activities bumps `updatedAt`, so the trigger must read + // the thread as busy right through them and fire ~15 minutes after the last one. + const turnCompletedAtMs = 19 * MINUTE; + const lastActivityAtMs = 52 * MINUTE; + const h = yield* harness({ + shell: threadShell({ updatedAt: msToIso(turnCompletedAtMs), latestTurnState: "completed" }), + }); + yield* arm(h.store); + yield* withReactor( + h.deps, + Effect.gen(function* () { + // The turn runs to 00:19 and background subagent work keeps appending activities + // until 00:52. Both move `updatedAt`, and the fixture keeps it in step with the + // clock so the reactor sees a thread that never stops moving. + for (let minute = 1; minute <= 52; minute++) { + yield* Ref.set( + h.shellRef, + threadShell({ + updatedAt: msToIso(minute * MINUTE), + latestTurnState: minute * MINUTE < turnCompletedAtMs ? "running" : "completed", + }), + ); + yield* advancePolls(1); + } + assert.strictEqual( + turnStarts(yield* Ref.get(h.dispatched)).length, + 0, + "no fire while subagent activity is still arriving", + ); + + // Then silence. Nothing else changes; the row simply stops moving. + yield* advanceUntil(turnStartsAtLeast(h.dispatched, 1), "the check-in after silence", 25); + assert.strictEqual(turnStarts(yield* Ref.get(h.dispatched)).length, 1); + const firedAtMs = (yield* record(h.store)).lastCheckIn!.firedAtMs; + assert.isAtLeast( + firedAtMs, + lastActivityAtMs + 15 * MINUTE, + "never before the idle threshold has actually elapsed", + ); + assert.isBelow(firedAtMs, lastActivityAtMs + 20 * MINUTE, "and not much after it"); + }), + ); + }).pipe(scoped), + ); + + // --- 129, 130, 130b: restarts --------------------------------------------- + + it.effect("129: a restart mid-loop keeps the budget, deadline, strikes and armedAtMs", () => + Effect.gen(function* () { + const h = yield* harness({ shell: threadShell() }); + yield* arm(h.store, { armedAtMs: 1_000, deadlineAtMs: 8 * HOUR, maxCheckIns: 6 }); + yield* withReactor( + h.deps, + Effect.gen(function* () { + for (let n = 1; n <= 2; n++) { + yield* advanceUntil(turnStartsAtLeast(h.dispatched, n), `check-in ${n}`, 60); + const current = yield* record(h.store); + yield* Ref.set( + h.shellRef, + threadShell({ updatedAt: msToIso(current.lastCheckIn!.firedAtMs + 3 * MINUTE) }), + ); + } + }), + ); + + // Kill and rebuild from disk, exactly as a server restart does. + const rebuilt = yield* makeLoopStore(h.statePath); + const after = yield* rebuilt.getThread(LOOP_THREAD_ID); + assert.isTrue(after.armed); + assert.strictEqual(after.checkInsUsed, 2, "the loop continues from check-in 3"); + assert.strictEqual(after.maxCheckIns, 6); + assert.strictEqual(after.deadlineAtMs, 8 * HOUR); + assert.strictEqual(after.armedAtMs, 1_000); + assert.strictEqual(after.strikes, 0); + assert.strictEqual(after.checkIns.length, 2, "and the ledger survives with it"); + }).pipe(scoped), + ); + + it.effect("130: a reboot storm does not fire three long-idle threads at once", () => + Effect.gen(function* () { + const day = msToIso(-24 * HOUR); + const two = threadShell({ id: "thread-2", updatedAt: day }); + const three = threadShell({ id: "thread-3", updatedAt: day }); + const h = yield* harness({ + shell: threadShell({ updatedAt: day }), + extraShells: [two, three], + }); + yield* arm(h.store); + yield* arm(h.store, { threadId: "thread-2" }); + yield* arm(h.store, { threadId: "thread-3" }); + yield* withReactor( + h.deps, + Effect.gen(function* () { + // The first tick after a restart: every one of these looks a day idle on the + // projection, and without the boot-grace floor all three would fire together. + yield* advancePolls(1); + assert.strictEqual(turnStarts(yield* Ref.get(h.dispatched)).length, 0); + + // They stay quiet right up to the threshold measured from process start. + yield* advancePolls(13); + assert.strictEqual(turnStarts(yield* Ref.get(h.dispatched)).length, 0); + }), + ); + }).pipe(scoped), + ); + + it.effect("130b: upstream's restart continuation window is not mistaken for idleness", () => + Effect.gen(function* () { + // `5b7d72aad` (#9167) re-establishes a binding after a self-update and dispatches + // `session.status: "starting", activeTurnId: null` synchronously at startup, while the + // actual `sendTurn` waits on server activation. A continued thread therefore looks idle + // with no error for a real window. Two mechanisms cover it and NEITHER was added for + // this: `busyTurn` counts `"starting"`, so the fuse is the 45-minute one; and + // `processStartedAtMs` floors the idle clock at process start. + const h = yield* harness({ + shell: threadShell({ + updatedAt: msToIso(-2 * HOUR), + sessionStatus: "starting", + latestTurnState: null, + }), + }); + yield* arm(h.store); + yield* withReactor( + h.deps, + Effect.gen(function* () { + yield* advancePolls(40); + assert.strictEqual( + turnStarts(yield* Ref.get(h.dispatched)).length, + 0, + "a continued thread must not be nudged inside its activation window", + ); + assert.strictEqual((yield* record(h.store)).checkInsUsed, 0, "and spends nothing"); + + // The premise the whole feature rests on, asserted rather than assumed: a thread + // waiting on a scheduled wake has no live turn, so upstream never marks it for + // continuation. The durability gap is untouched, and T3's record is the only trace. + yield* Ref.set( + h.shellRef, + threadShell({ updatedAt: msToIso(0), sessionStatus: "ready", latestTurnState: null }), + ); + yield* advanceUntil(turnStartsAtLeast(h.dispatched, 1), "the covered thread", 60); + assert.strictEqual(turnStarts(yield* Ref.get(h.dispatched)).length, 1); + }), + ); + }).pipe(scoped), + ); + + // --- 131, 132: not fighting auto-resume ----------------------------------- + + it.effect("131: a usage limit holds the loop, which spends no budget waiting", () => + Effect.gen(function* () { + // Auto-resume is off for this thread, so a limit produces no pending resume and guard 9 + // passes. Without the rate-limit fiber the loop would nudge straight into a live limit. + const h = yield* harness({ + shell: threadShell(), + events: [rateLimitEvent({ status: "rejected", resetsAtSeconds: 3 * 3_600 })], + autoResumePending: false, + }); + yield* arm(h.store); + yield* withReactor( + h.deps, + Effect.gen(function* () { + // Stream-driven: the tap announces its write, so the hold is provably in place + // before a single poll is spent rather than probably in place after five. + yield* untilReceipt((r) => r.type === "rateLimit.recorded"); + yield* advancePolls(60); + assert.strictEqual( + turnStarts(yield* Ref.get(h.dispatched)).length, + 0, + "no nudge into a live limit", + ); + const held = yield* record(h.store); + assert.strictEqual(held.checkInsUsed, 0, "held is not spent"); + assert.strictEqual(held.rateLimitedUntilMs, 3 * HOUR); + + // Held ≠ stalled: once the window reopens the run carries on with a full budget. + yield* advanceUntil(turnStartsAtLeast(h.dispatched, 1), "the resumed check-in", 240); + assert.strictEqual((yield* record(h.store)).checkInsUsed, 1); + }), + ); + }).pipe(scoped), + ); + + it.effect("132: a pending auto-resume stands the loop down and is left intact", () => + Effect.gen(function* () { + const h = yield* harness({ shell: threadShell(), autoResumePending: true }); + yield* arm(h.store); + yield* withReactor( + h.deps, + Effect.gen(function* () { + yield* advancePolls(60); + assert.strictEqual( + turnStarts(yield* Ref.get(h.dispatched)).length, + 0, + "two reactors must never both nudge one thread", + ); + assert.strictEqual((yield* record(h.store)).checkInsUsed, 0); + + // Nothing the loop did touched auto-resume's arm; when it clears, the loop resumes. + yield* Ref.set(h.autoResumePending, false); + yield* advanceUntil(turnStartsAtLeast(h.dispatched, 1), "the check-in", 30); + assert.strictEqual(turnStarts(yield* Ref.get(h.dispatched)).length, 1); + }), + ); + }).pipe(scoped), + ); + + // --- 133, 134: the two question channels ---------------------------------- + + it.effect("133: a blocking question overnight parks the loop and spends nothing", () => + Effect.gen(function* () { + // `AskUserQuestion` at 01:00. Guard 8 skips: nudging past a pending input is worse than + // waiting, and the console keeps a blocking-since reading either way. + const h = yield* harness({ shell: threadShell({ hasPendingUserInput: true }) }); + yield* arm(h.store, { deadlineAtMs: 3 * HOUR }); + yield* withReactor( + h.deps, + Effect.gen(function* () { + yield* advancePolls(90); + assert.strictEqual(turnStarts(yield* Ref.get(h.dispatched)).length, 0); + const parked = yield* record(h.store); + assert.strictEqual(parked.checkInsUsed, 0, "a parked loop spends nothing"); + assert.isTrue(parked.armed); + const skips = activitiesOfKind(yield* Ref.get(h.dispatched), LOOP_ACTIVITY_KINDS.skipped); + assert.strictEqual(skips.length, 1, "one blocking-since note, not ninety"); + assert.strictEqual( + (skips[0]!.payload as { reason: string }).reason, + "pending_user_input", + ); + + // It runs out its deadline as `spent`, never `stalled`: nothing was tried and + // failed, the loop simply never got a turn. + yield* advanceUntil( + record(h.store).pipe(Effect.map((r) => r.stopped !== null)), + "the deadline", + 120, + ); + assert.strictEqual((yield* record(h.store)).stopped?.reason, "spent"); + }), + ); + }).pipe(scoped), + ); + + it.effect("134: a deferred blocker does not park the loop, and its answer is delivered", () => + Effect.gen(function* () { + // The other half of the channel split: `raise_blocker` records a question WITHOUT + // blocking the turn, so the loop keeps firing, and the answer rides the next prompt. + const h = yield* harness({ shell: threadShell() }); + yield* arm(h.store); + yield* h.store.addBlocker(LOOP_THREAD_ID, { + id: "blocker-1", + raisedAtMs: 60 * MINUTE, + question: "Should I bump the major version?", + options: [], + context: null, + answeredAtMs: null, + answer: null, + deliveredToAgent: false, + }); + yield* withReactor( + h.deps, + Effect.gen(function* () { + yield* advanceUntil(turnStartsAtLeast(h.dispatched, 1), "the first check-in", 30); + const first = turnStarts(yield* Ref.get(h.dispatched))[0]!; + assert.notInclude( + first.message.text, + "Should I bump the major version?", + "an unanswered blocker carries nothing to say", + ); + + // The human answers at 09:04 while the thread is idle. + yield* h.store.answerBlocker(LOOP_THREAD_ID, "blocker-1", "yes, go to 2.0", 70 * MINUTE); + const current = yield* record(h.store); + yield* Ref.set( + h.shellRef, + threadShell({ updatedAt: msToIso(current.lastCheckIn!.firedAtMs + 3 * MINUTE) }), + ); + + yield* advanceUntil(turnStartsAtLeast(h.dispatched, 2), "the second check-in", 60); + const second = turnStarts(yield* Ref.get(h.dispatched))[1]!; + assert.include(second.message.text, "Should I bump the major version?"); + assert.include(second.message.text, "yes, go to 2.0"); + // No settling: the wait above returned on a completed tick, and the delivery was + // marked inside it. This is the assertion CI failed on when the wait was a turn + // budget that could run out between the dispatch and the write that follows it. + assert.deepStrictEqual( + yield* h.store.listUndeliveredAnswers(LOOP_THREAD_ID), + [], + "and it is marked delivered exactly once it has actually been said", + ); + }), + ); + }).pipe(scoped), + ); + + // --- 135: the empty console ----------------------------------------------- + + it.effect("135: a run that ends spent with no blockers still reports why, and the budget", () => + Effect.gen(function* () { + // The console degrades to useful, not to silence: a model that never raised a blocker + // must still leave a stop reason and a budget behind. + const h = yield* harness({ shell: threadShell({ sessionStatus: "running" }) }); + yield* arm(h.store, { deadlineAtMs: 10 * MINUTE, maxCheckIns: 4 }); + yield* withReactor( + h.deps, + Effect.gen(function* () { + yield* advanceUntil( + record(h.store).pipe(Effect.map((r) => r.stopped !== null)), + "the spent terminal", + 30, + ); + const after = yield* record(h.store); + assert.strictEqual(after.stopped?.reason, "spent"); + assert.isNotEmpty(after.stopped?.detail ?? "", "with a reason a human can read"); + assert.strictEqual(after.maxCheckIns, 4, "and the budget it was given"); + assert.deepStrictEqual([...after.blockers], []); + const notes = activitiesOfKind(yield* Ref.get(h.dispatched), LOOP_ACTIVITY_KINDS.stopped); + assert.strictEqual(notes.length, 1); + const payload = notes[0]!.payload as { reason: string; of: number; checkInsUsed: number }; + assert.strictEqual(payload.reason, "spent"); + assert.strictEqual(payload.of, 4); + assert.strictEqual(payload.checkInsUsed, 0); + }), + ); + }).pipe(scoped), + ); + + // --- 136: takeover --------------------------------------------------------- + + it.effect("136: a human takeover at 04:00 disarms, and re-arming restores a full budget", () => + Effect.gen(function* () { + const h = yield* harness({ shell: threadShell() }); + yield* arm(h.store, { maxCheckIns: 6 }); + yield* withReactor( + h.deps, + Effect.gen(function* () { + yield* advanceUntil(turnStartsAtLeast(h.dispatched, 1), "the first check-in", 30); + const fired = yield* record(h.store); + yield* Ref.set( + h.shellRef, + threadShell({ + latestUserMessageAt: msToIso(Date.parse(fired.lastCheckIn!.createdAtIso) + 1_000), + }), + ); + yield* advanceUntil( + record(h.store).pipe(Effect.map((r) => r.stopped !== null)), + "the handback", + 40, + ); + const handedBack = yield* record(h.store); + assert.strictEqual(handedBack.stopped?.reason, "handed-back"); + assert.strictEqual(handedBack.checkInsUsed, 1, "takeover is not a budget reset"); + + // One tap to re-arm: a fresh run, not a repair of the old one. + yield* h.store.arm({ + threadId: LOOP_THREAD_ID, + armedAtMs: 5 * HOUR, + deadlineAtMs: 12 * HOUR, + maxCheckIns: 6, + }); + const rearmed = yield* record(h.store); + assert.isNull(rearmed.stopped, "the terminal clears only on a human re-arm"); + assert.strictEqual(rearmed.checkInsUsed, 0, "with a full budget"); + assert.strictEqual(rearmed.strikes, 0); + assert.deepStrictEqual([...rearmed.checkIns], []); + }), + ); + }).pipe(scoped), + ); + + // --- 137: the done-file ---------------------------------------------------- + + it.effect("137: the done-file at check-in 3 ends the run as done, with budget unspent", () => + Effect.gen(function* () { + const h = yield* harness({ shell: threadShell() }); + yield* arm(h.store, { maxCheckIns: 6 }); + yield* withReactor( + h.deps, + Effect.gen(function* () { + for (let n = 1; n <= 3; n++) { + yield* advanceUntil(turnStartsAtLeast(h.dispatched, n), `check-in ${n}`, 80); + const current = yield* record(h.store); + yield* Ref.set( + h.shellRef, + threadShell({ updatedAt: msToIso(current.lastCheckIn!.firedAtMs + 3 * MINUTE) }), + ); + } + // The agent writes the file after its third check-in. T3 only ever stats it. + const now = yield* Effect.clockWith((clock) => clock.currentTimeMillis); + yield* writeSentinel(h.workspaceRoot, now); + + yield* advanceUntil( + record(h.store).pipe(Effect.map((r) => r.stopped !== null)), + "the done terminal", + 40, + ); + const after = yield* record(h.store); + assert.strictEqual(after.stopped?.reason, "done"); + assert.notStrictEqual(after.stopped?.reason, "spent", "done is never reported as spent"); + assert.strictEqual(after.checkInsUsed, 3, "three check-ins left unused"); + assert.strictEqual(turnStarts(yield* Ref.get(h.dispatched)).length, 3); + assert.deepStrictEqual( + yield* Ref.get(h.stopSessions), + [], + "a finished agent's session is not killed under it", + ); + }), + ); + }).pipe(scoped), + ); +}); diff --git a/apps/server/src/coil/loop/layer.ts b/apps/server/src/coil/loop/layer.ts new file mode 100644 index 000000000000..590f62a33977 --- /dev/null +++ b/apps/server/src/coil/loop/layer.ts @@ -0,0 +1,36 @@ +/** + * The one `LoopStore` layer value, shared by everything that touches `coil-loop.json`. + * + * The reactor, the HTTP routes and the MCP toolkit all mutate the same single JSON file, so + * they must resolve to the same in-memory store. Effect memoises layer construction per + * build keyed on **layer identity**, so this module-level value — imported, never + * re-declared — is what makes that true. `coil/index.ts` uses the identical trick for + * `AutoResumeStoreLive` and `WebPushDepsLive`, and documents why: two copies over one file + * means the console's write is invisible to the supervisor's read. + * + * It lives here rather than in `coil/index.ts` because the MCP toolkit registration hangs + * off `mcp/McpHttpServer.ts`, which is on the other side of the layer graph from the coil + * aggregator. If the toolkit left `LoopStore` as an open requirement instead, the gap would + * surface in upstream's `makeRoutesLayer` signature — a second seam edit for nothing. + * + * @module coil/loop/layer + */ + +import * as Effect from "effect/Effect"; +import * as Layer from "effect/Layer"; +import * as Path from "effect/Path"; + +import { ServerConfig } from "../../config.ts"; +import { LoopStore, makeLoopStore } from "./state.ts"; + +export const LOOP_STATE_FILENAME = "coil-loop.json"; + +/** Wires the durable store to the server state directory. Import this; do not re-declare it. */ +export const LoopStoreLive = Layer.effect( + LoopStore, + Effect.gen(function* () { + const config = yield* ServerConfig; + const path = yield* Path.Path; + return yield* makeLoopStore(path.join(config.stateDir, LOOP_STATE_FILENAME)); + }), +); diff --git a/apps/server/src/coil/loop/reactorHarness.test.ts b/apps/server/src/coil/loop/reactorHarness.test.ts new file mode 100644 index 000000000000..99a9c07634d8 --- /dev/null +++ b/apps/server/src/coil/loop/reactorHarness.test.ts @@ -0,0 +1,149 @@ +/** + * The harness's own timing contract. + * + * `reactorHarness.ts` is test infrastructure, and normally that is not worth testing — except + * that forty-eight scenarios read their assertions out of a simulated clock it advances, so + * *when* it advances that clock is a load-bearing property rather than an implementation + * detail. It has been wrong twice, and both times the bill arrived as assertions about the + * product: a loop covering one wake twice, and a check-in landing five simulated minutes late, + * on a two-core CI runner where a wait that counted scheduler turns ran out of them. + * + * The waits are receipts now, so the properties below are the ones that matter: a poll waits + * for the whole tick however slow the machine is, and simulated time moves only where a + * scenario says it moves. + * + * The reactor here is a stand-in that publishes the same `tick.completed` the real one does, + * with deliberately slow work in front of it. Booting the real supervisor would test the + * supervisor; this tests the waiting. + * + * @module coil/loop/reactorHarness.test + */ + +import { assert, describe, it } from "@effect/vitest"; +import * as Clock from "effect/Clock"; +import * as Duration from "effect/Duration"; +import * as Effect from "effect/Effect"; +import * as Exit from "effect/Exit"; +import * as Layer from "effect/Layer"; +import * as Ref from "effect/Ref"; +import type * as Scope from "effect/Scope"; +import * as TestClock from "effect/testing/TestClock"; + +import { advancePolls, advanceUntil, POLL_MS, untilReceipt } from "./reactorHarness.ts"; +import { LoopReactorReceipts, LoopReactorReceiptsLive } from "./receipts.ts"; + +/** + * Work that finishes only after `turns` real event-loop turns. + * + * A stand-in for the reactor's store writes, which are real filesystem I/O: how many turns + * they need is a fact about the machine, and nothing about it may reach the clock. + */ +const afterTurns = (turns: number) => + Effect.promise(async () => { + for (let i = 0; i < turns; i += 1) { + await new Promise((resolve) => setImmediate(resolve)); + } + }); + +/** Far more turns than any turn-counting wait would have budgeted. */ +const SLOW_TURNS = 200; + +/** + * A reactor-shaped fiber: sleep a poll, do slow work, announce that the tick is over. + * + * `ticks` counts completed passes, so a test can assert the wait really waited rather than + * merely returned. + */ +const fakeReactor = Effect.gen(function* () { + const receipts = yield* LoopReactorReceipts; + const ticks = yield* Ref.make(0); + yield* Effect.forkScoped( + Effect.gen(function* () { + yield* Effect.sleep(Duration.millis(POLL_MS)); + yield* afterTurns(SLOW_TURNS); + const nowMs = yield* Clock.currentTimeMillis; + yield* Ref.update(ticks, (n) => n + 1); + yield* receipts.publish({ type: "tick.completed", armedCount: 0, nowMs }); + }).pipe(Effect.forever), + ); + return { ticks, receipts }; +}); + +const scoped = (body: Effect.Effect) => + body.pipe( + Effect.scoped, + Effect.provide(Layer.mergeAll(LoopReactorReceiptsLive, TestClock.layer())), + ); + +describe("reactorHarness — a poll is one whole tick, and time moves nowhere else", () => { + it.effect("a poll waits for the tick to finish, however slow the machine is", () => + Effect.gen(function* () { + const { ticks } = yield* fakeReactor; + + yield* advancePolls(3); + + assert.strictEqual(yield* Ref.get(ticks), 3, "three ticks ran to completion"); + assert.strictEqual( + yield* Clock.currentTimeMillis, + 3 * POLL_MS, + "three polls of movement, and not a millisecond of the machine's own", + ); + }).pipe(scoped), + ); + + it.effect("advanceUntil rests on the poll where the answer arrived", () => + Effect.gen(function* () { + const { ticks } = yield* fakeReactor; + // True on the fourth completed tick and no earlier, so the resting place is known. + const condition = Ref.get(ticks).pipe(Effect.map((n) => n >= 4)); + + yield* advanceUntil(condition, "the fourth tick", 20); + + assert.strictEqual(yield* Clock.currentTimeMillis, 4 * POLL_MS); + }).pipe(scoped), + ); + + it.effect("a condition that already holds costs nothing at all", () => + Effect.gen(function* () { + yield* fakeReactor; + yield* advanceUntil(Effect.succeed(true), "nothing to wait for", 20); + assert.strictEqual(yield* Clock.currentTimeMillis, 0); + }).pipe(scoped), + ); + + it.effect("even giving up costs only the polls it was given", () => + Effect.gen(function* () { + // The discriminating case, and the one CI hit: a wait that cannot be satisfied is where + // a harness that buys simulated time hoping the next round answers shows itself. By the + // time it gives up — or worse, succeeds — the clock is somewhere no test asked for, and + // in `136c` that was past the loop's check-in floor, which covered one wake twice. + yield* fakeReactor; + const exit = yield* Effect.exit( + advanceUntil(Effect.succeed(false), "something that never happens", 2), + ); + + assert.isTrue(Exit.isFailure(exit), "an unsatisfiable condition still fails the test"); + assert.strictEqual( + yield* Clock.currentTimeMillis, + 2 * POLL_MS, + "the two polls it was given, and not one minute more", + ); + }).pipe(scoped), + ); + + it.effect("waiting for a receipt spends no simulated time", () => + Effect.gen(function* () { + const { receipts } = yield* fakeReactor; + yield* Effect.forkScoped( + afterTurns(SLOW_TURNS).pipe( + Effect.andThen(receipts.publish({ type: "disarmed", threadId: "thread-1" })), + ), + ); + + const receipt = yield* untilReceipt((r) => r.type === "disarmed"); + + assert.strictEqual(receipt.type, "disarmed"); + assert.strictEqual(yield* Clock.currentTimeMillis, 0, "the clock never moved"); + }).pipe(scoped), + ); +}); diff --git a/apps/server/src/coil/loop/reactorHarness.ts b/apps/server/src/coil/loop/reactorHarness.ts new file mode 100644 index 000000000000..1f9a47a592eb --- /dev/null +++ b/apps/server/src/coil/loop/reactorHarness.ts @@ -0,0 +1,561 @@ +/** + * Test harness for the loop supervisor. + * + * Boots the **real** `LoopReactorLive` and the **real** durable `LoopStore` against a + * scripted projection and a recording orchestration engine, on Effect's `TestClock` so an + * eight-hour overnight run costs milliseconds. Only four things are doubles, and each is a + * deliberate cut: + * + * - **Orchestration engine** — records dispatched commands instead of running turns, and + * can be told to fail a chosen command type. Dispatch is the reactor's output. + * - **Projection snapshot** — two mutable shell rows the scenario scripts, read through + * the same `getThreadShellById` / `getProjectShellById` the reactor uses, with a call + * counter so "zero SQL when nothing is armed" is assertable. + * - **Provider** — `streamEvents` pre-loaded with the scenario's events (emit-then-block, + * so delivery does not depend on publish timing) and a recording `stopSession`. + * - **Auto-resume store** — one boolean, because guard 9 reads exactly one field. + * + * The store, the reactor, its config, its guards, its decision table, its sentinel reads and + * its persistence are all real. + * + * Nothing here waits by counting scheduler turns. The reactor publishes a receipt at every + * milestone (`receipts.ts`) and every wait below is an await on one, so a scenario's timing is + * a property of the scenario and not of the machine running it. + * + * @module coil/loop/reactorHarness + */ + +// @effect-diagnostics nodeBuiltinImport:off +// @effect-diagnostics globalErrorInEffectFailure:off -- the engine stub raises bare Errors on +// purpose, mirroring the arbitrary driver throws the reactor must survive. +// @effect-diagnostics globalDate:off -- msToIso is a pure ms->ISO helper for fixtures; the +// value it formats is a TestClock offset, never a wall-clock reading. +import * as NodeFSP from "node:fs/promises"; +import * as NodePath from "node:path"; + +import type { + OrchestrationCommand, + OrchestrationProjectShell, + OrchestrationThreadShell, + ProviderRuntimeEvent, +} from "@t3tools/contracts"; +import * as Crypto from "effect/Crypto"; +import * as Duration from "effect/Duration"; +import * as Effect from "effect/Effect"; +import * as FileSystem from "effect/FileSystem"; +import * as Layer from "effect/Layer"; +import * as Option from "effect/Option"; +import * as PubSub from "effect/PubSub"; +import * as Ref from "effect/Ref"; +import * as Stream from "effect/Stream"; +import * as TestClock from "effect/testing/TestClock"; + +import { OrchestrationEngineService } from "../../orchestration/Services/OrchestrationEngine.ts"; +import { ProjectionSnapshotQuery } from "../../orchestration/Services/ProjectionSnapshotQuery.ts"; +import { ProviderService } from "../../provider/Services/ProviderService.ts"; +import { AutoResumeStore } from "../autoResume/state.ts"; +import { LOOP_DONE_RELATIVE_PATH } from "./config.ts"; +import { + type LoopReactorReceipt, + LoopReactorReceipts, + LoopReactorReceiptsLive, +} from "./receipts.ts"; +import { LoopStore, makeLoopStore } from "./state.ts"; + +export const LOOP_THREAD_ID = "thread-1"; +export const LOOP_PROJECT_ID = "project-1"; + +/** Handy constants so a scenario reads in the units the design talks in. */ +export const MINUTE = 60_000; +export const HOUR = 60 * MINUTE; +/** The reactor's default poll cadence, and therefore the granularity of every scenario. */ +export const POLL_MS = MINUTE; + +/** + * Test-clock milliseconds as an ISO timestamp. + * + * `TestClock` begins at epoch 0, so a scenario's "now" is 1970. Every projection field the + * trigger parses (`updatedAt`, `latestUserMessageAt`, `snoozedUntil`) is compared against + * that clock, so a real 2026 timestamp in a fixture silently inverts the scenario. + */ +export const msToIso = (ms: number): string => new Date(ms).toISOString(); + +/** + * The reactor reads `COIL_LOOP_*` once, at layer construction. A developer with any of them + * exported would otherwise change the meaning of every timing assertion in these files. + */ +export const clearLoopEnv = (): void => { + for (const key of Object.keys(process.env)) { + if (key.startsWith("COIL_LOOP_")) delete process.env[key]; + } +}; + +export interface ThreadShellOverrides { + readonly id?: string; + readonly updatedAt?: string; + readonly archivedAt?: string | null; + readonly settledOverride?: "settled" | "active" | null; + readonly snoozedUntil?: string | null; + readonly sessionStatus?: string | null; + readonly providerName?: string; + readonly latestTurnState?: string | null; + readonly latestUserMessageAt?: string | null; + readonly hasPendingApprovals?: boolean; + readonly hasPendingUserInput?: boolean; + readonly hasActionableProposedPlan?: boolean; + readonly backgroundLiveness?: "working" | "monitoring" | null; + readonly worktreePath?: string | null; + readonly runtimeMode?: string; + readonly interactionMode?: string; +} + +/** + * A thread shell row. + * + * Cast once at the end because building every branded field (`ThreadId`, `IsoDateTime`, + * `ModelSelection`, …) is noise the reactor never reads — the fields it *does* read are all + * set explicitly above, which is what makes the cast safe rather than a hole. + */ +export const threadShell = (o: ThreadShellOverrides = {}): OrchestrationThreadShell => + ({ + id: o.id ?? LOOP_THREAD_ID, + projectId: LOOP_PROJECT_ID, + title: "a thread", + modelSelection: { providerName: "claudeAgent", modelId: "sonnet" }, + runtimeMode: o.runtimeMode ?? "full-access", + interactionMode: o.interactionMode ?? "default", + branch: null, + worktreePath: o.worktreePath ?? null, + latestTurn: + o.latestTurnState === undefined || o.latestTurnState === null + ? null + : { turnId: "turn-1", state: o.latestTurnState }, + createdAt: msToIso(0), + updatedAt: o.updatedAt ?? msToIso(0), + archivedAt: o.archivedAt ?? null, + settledOverride: o.settledOverride ?? null, + settledAt: null, + snoozedUntil: o.snoozedUntil ?? null, + pinnedAt: null, + session: + o.sessionStatus === null + ? null + : { + status: o.sessionStatus ?? "ready", + providerName: o.providerName ?? "claudeAgent", + }, + latestUserMessageAt: o.latestUserMessageAt ?? null, + hasPendingApprovals: o.hasPendingApprovals ?? false, + hasPendingUserInput: o.hasPendingUserInput ?? false, + hasActionableProposedPlan: o.hasActionableProposedPlan ?? false, + backgroundLiveness: o.backgroundLiveness ?? null, + }) as unknown as OrchestrationThreadShell; + +export const projectShell = (workspaceRoot: string): OrchestrationProjectShell => + ({ + id: LOOP_PROJECT_ID, + title: "a project", + workspaceRoot, + defaultModelSelection: null, + scripts: [], + createdAt: msToIso(0), + updatedAt: msToIso(0), + }) as unknown as OrchestrationProjectShell; + +/** A Claude `account.rate-limits.updated` runtime event. */ +export const rateLimitEvent = (o: { + readonly status: "allowed" | "allowed_warning" | "rejected"; + readonly resetsAtSeconds?: number; + readonly threadId?: string; +}): ProviderRuntimeEvent => + ({ + type: "account.rate-limits.updated", + eventId: `evt-rl-${o.status}-${o.resetsAtSeconds ?? 0}`, + provider: "claudeAgent", + threadId: o.threadId ?? LOOP_THREAD_ID, + createdAt: msToIso(0), + payload: { + rateLimits: { + type: "rate_limit_event", + rate_limit_info: { + status: o.status, + rateLimitType: "five_hour", + ...(o.resetsAtSeconds === undefined ? {} : { resetsAt: o.resetsAtSeconds }), + }, + }, + }, + }) as unknown as ProviderRuntimeEvent; + +/** A `user-input.requested` runtime event, for the two-subscriber proof. */ +export const userInputRequestedEvent = (o: { + readonly requestId: string; + readonly question: string; + readonly threadId?: string; +}): ProviderRuntimeEvent => + ({ + type: "user-input.requested", + eventId: `evt-ui-${o.requestId}`, + provider: "claudeAgent", + threadId: o.threadId ?? LOOP_THREAD_ID, + requestId: o.requestId, + createdAt: msToIso(0), + payload: { questions: [{ question: o.question, options: [] }] }, + }) as unknown as ProviderRuntimeEvent; + +export interface HarnessOptions { + /** The thread row the projection returns. `null` models a deleted thread. */ + readonly shell?: OrchestrationThreadShell | null; + /** Runtime events, pre-loaded emit-then-block. */ + readonly events?: ReadonlyArray; + /** Guard 9's one field. */ + readonly autoResumePending?: boolean; + /** Additional thread rows, keyed by id, for multi-thread scenarios. */ + readonly extraShells?: ReadonlyArray; +} + +export const harness = (options: HarnessOptions = {}) => + Effect.gen(function* () { + const fs = yield* FileSystem.FileSystem; + const root = yield* fs.makeTempDirectoryScoped({ prefix: "coil-loop-" }); + const workspaceRoot = NodePath.join(root, "workspace"); + yield* fs.makeDirectory(workspaceRoot, { recursive: true }); + const statePath = NodePath.join(root, "coil-loop.json"); + + const dispatched = yield* Ref.make([]); + /** Command types the engine should reject, so a scenario can break exactly one path. */ + const failCommandTypes = yield* Ref.make>(new Set()); + /** Defect (not failure) to raise from dispatch, for the "a defect cannot kill it" cases. */ + const dieOnCommandType = yield* Ref.make(null); + /** Same, scoped to one thread, so a multi-thread pass can prove the others still run. */ + const dieOnThreadId = yield* Ref.make(null); + + const EngineStub = Layer.succeed(OrchestrationEngineService, { + dispatch: (command: OrchestrationCommand) => + Effect.gen(function* () { + if ((yield* Ref.get(dieOnCommandType)) === command.type) { + throw new Error(`simulated defect dispatching ${command.type}`); + } + if ( + command.type === "thread.turn.start" && + (yield* Ref.get(dieOnThreadId)) === command.threadId + ) { + throw new Error(`simulated defect for ${command.threadId}`); + } + if ((yield* Ref.get(failCommandTypes)).has(command.type)) { + return yield* Effect.fail(new Error(`simulated failure: ${command.type}`)); + } + yield* Ref.update(dispatched, (all) => [...all, command]); + return { sequence: 0 }; + }), + streamDomainEvents: Stream.empty, + readEvents: () => Stream.empty, + latestSequence: Effect.succeed(0), + } as unknown as typeof OrchestrationEngineService.Service); + + const shellRef = yield* Ref.make( + options.shell === undefined ? threadShell() : options.shell, + ); + const extraShellsRef = yield* Ref.make>( + options.extraShells ?? [], + ); + /** + * Swaps in a different row once the tick has read the shell `afterCall` times, which is + * how the wake race is scripted: the guard block sees one world and the pre-dispatch + * re-read sees another. + */ + const shellOverrideRef = yield* Ref.make<{ + readonly afterCall: number; + readonly shell: OrchestrationThreadShell | null; + } | null>(null); + const shellCalls = yield* Ref.make(0); + const projectCalls = yield* Ref.make(0); + /** Set to make the projection read fail, distinct from "the thread is gone". */ + const shellReadFails = yield* Ref.make(false); + /** + * Fail reads only from call `n + 1` on, the same shape `shellOverrideRef` uses — which is + * how a scenario makes the *pre-dispatch* re-read fail while the guard block's read + * succeeded. The two are different worlds and the reactor must not conflate them. + */ + const shellFailsAfterCall = yield* Ref.make(null); + + const SnapshotStub = Layer.succeed(ProjectionSnapshotQuery, { + getThreadShellById: (threadId: string) => + Effect.gen(function* () { + const n = yield* Ref.updateAndGet(shellCalls, (c) => c + 1); + const failAfter = yield* Ref.get(shellFailsAfterCall); + if ((yield* Ref.get(shellReadFails)) || (failAfter !== null && n > failAfter)) { + return yield* Effect.fail(new Error("simulated projection failure")); + } + const extra = (yield* Ref.get(extraShellsRef)).find((s) => s.id === threadId); + if (extra) return Option.some(extra); + const override = yield* Ref.get(shellOverrideRef); + const shell = + override !== null && n > override.afterCall ? override.shell : yield* Ref.get(shellRef); + if (shell === null || shell.id !== threadId) return Option.none(); + return Option.some(shell); + }), + getProjectShellById: () => + Ref.update(projectCalls, (c) => c + 1).pipe( + Effect.as(Option.some(projectShell(workspaceRoot))), + ), + } as unknown as typeof ProjectionSnapshotQuery.Service); + + const stopSessions = yield* Ref.make([]); + const ProviderStub = Layer.succeed(ProviderService, { + get streamEvents() { + return Stream.concat(Stream.fromIterable(options.events ?? []), Stream.never); + }, + stopSession: (input: { readonly threadId: string }) => + Ref.update(stopSessions, (all) => [...all, input.threadId]), + } as unknown as typeof ProviderService.Service); + + const autoResumePending = yield* Ref.make(options.autoResumePending ?? false); + const AutoResumeStub = Layer.succeed(AutoResumeStore, { + getThread: (_threadId: string) => + Ref.get(autoResumePending).pipe( + Effect.map((pending) => ({ + enabled: true, + overridePrompt: null, + pending: pending ? { threadId: _threadId, resumeAtMs: 0 } : null, + fired: [], + })), + ), + } as unknown as typeof AutoResumeStore.Service); + + const store = yield* makeLoopStore(statePath); + const StoreLive = Layer.succeed(LoopStore, store); + + // Crypto.make derives randomUUIDv4 from randomBytes; a counter keeps bytes distinct so + // generated command/message ids differ across calls. + let seed = 1; + const CryptoStub = Layer.succeed( + Crypto.Crypto, + Crypto.make({ + randomBytes: (size) => { + const bytes = new Uint8Array(size); + for (let i = 0; i < size; i++) bytes[i] = (seed + i) & 0xff; + seed += size; + return bytes; + }, + digest: (_algorithm, data) => Effect.succeed(data), + }), + ); + + const deps = Layer.mergeAll( + EngineStub, + SnapshotStub, + ProviderStub, + AutoResumeStub, + CryptoStub, + StoreLive, + // Test-only, and the whole reason the waits below are exact. Production never provides + // it, so the reactor's emitter resolves to a no-op there. See `receipts.ts`. + LoopReactorReceiptsLive, + ); + + return { + deps, + store, + statePath, + workspaceRoot, + dispatched, + failCommandTypes, + dieOnCommandType, + dieOnThreadId, + shellRef, + shellOverrideRef, + extraShellsRef, + shellCalls, + projectCalls, + shellReadFails, + shellFailsAfterCall, + stopSessions, + autoResumePending, + }; + }); + +/** Write the done-file with an explicit mtime **on the test clock**. */ +export const writeSentinel = (workspaceRoot: string, mtimeMs: number): Effect.Effect => + Effect.promise(async () => { + const filePath = NodePath.join(workspaceRoot, LOOP_DONE_RELATIVE_PATH); + await NodeFSP.mkdir(NodePath.dirname(filePath), { recursive: true }); + await NodeFSP.writeFile(filePath, "finished\n"); + // `utimes` takes SECONDS, not milliseconds. Passing ms puts the mtime ~31 000 years out + // and every freshness compare silently answers "fresh". + await NodeFSP.utimes(filePath, mtimeMs / 1000, mtimeMs / 1000); + }); + +export const commandTypes = (commands: ReadonlyArray): string[] => + commands.map((c) => c.type); + +export const turnStarts = (commands: ReadonlyArray) => + commands.filter( + (c): c is Extract => + c.type === "thread.turn.start", + ); + +export interface AppendedActivity { + readonly kind: string; + readonly tone: string; + readonly summary: string; + readonly payload: unknown; + readonly threadId: string; +} + +export const activities = ( + commands: ReadonlyArray, +): ReadonlyArray => + commands + .filter((c) => c.type === "thread.activity.append") + .map((c) => { + const command = c as unknown as { + threadId: string; + activity: { kind: string; tone: string; summary: string; payload: unknown }; + }; + return { ...command.activity, threadId: command.threadId }; + }); + +export const activitiesOfKind = ( + commands: ReadonlyArray, + kind: string, +): ReadonlyArray => activities(commands).filter((a) => a.kind === kind); + +// --- waiting ----------------------------------------------------------------- +// +// Every wait below is an await on a receipt the reactor published. Nothing here counts +// scheduler turns, and nothing here moves the clock except the polls a scenario asks for. +// +// It used to count turns, and that is what three CI failures were: the store persists through +// `writeFileStringAtomically`, real filesystem I/O whose completion fires on the Node event +// loop and not on `TestClock`, so on a two-core runner a turn buys less real progress and a +// budget that is generous on a laptop runs out. The tests then read a half-finished tick as a +// finished one — and because the harness kept advancing, the failures came back as assertions +// about the product (a wake covered twice, a check-in five simulated minutes late) rather than +// as anything that looked like a timing bug. See `receipts.ts`. + +/** A real event-loop tick. */ +const realTick = Effect.promise(() => new Promise((resolve) => setImmediate(resolve))); + +/** One pump of both schedulers: the Effect fiber scheduler and the real Node event loop. */ +export const pump = Effect.gen(function* () { + yield* realTick; + for (let j = 0; j < 5; j++) yield* Effect.yieldNow; +}); + +/** + * A bounded spin, for asserting that something does NOT happen. + * + * The only wait left that is not a receipt, because you cannot await the absence of one. It + * moves no clock, so the worst a starved runner can do here is make the assertion weaker — + * never make it wrong. Anything waiting for something to APPEAR must await its receipt. + */ +export const settleQuiet = Effect.gen(function* () { + for (let i = 0; i < 10; i++) yield* pump; +}); + +/** The next receipt matching `matches`. Unbounded on purpose: the test timeout is the bound. */ +export const untilReceipt = (matches: (receipt: LoopReactorReceipt) => boolean) => + Effect.gen(function* () { + const { log } = yield* LoopReactorReceipts; + while (true) { + const receipt = yield* PubSub.take(log); + if (matches(receipt)) return receipt; + } + }); + +/** + * Wait until every matcher has seen a receipt, in any order. + * + * One drain serves all of them, because a second `untilReceipt` would have thrown away the + * receipt the first one was not looking for. + */ +export const untilAllReceipts = ( + matchers: ReadonlyArray<(receipt: LoopReactorReceipt) => boolean>, +) => + Effect.gen(function* () { + const { log } = yield* LoopReactorReceipts; + const outstanding = [...matchers]; + while (outstanding.length > 0) { + const receipt = yield* PubSub.take(log); + const index = outstanding.findIndex((matches) => matches(receipt)); + if (index >= 0) outstanding.splice(index, 1); + } + }); + +/** + * Advance one poll and wait for the tick it triggers to finish. + * + * `tick.completed` is published after every armed thread has been evaluated and every write + * is durable, so when this returns the world is exactly one whole tick further on — no more + * and no less, on any machine. Receipts from earlier in the tick are drained on the way past. + */ +const advanceOnePoll = (log: PubSub.Subscription) => + Effect.gen(function* () { + yield* TestClock.adjust(Duration.millis(POLL_MS)); + while (true) { + const receipt = yield* PubSub.take(log); + if (receipt.type === "tick.completed") return receipt; + } + }); + +/** + * One real turn before the first poll, so a tick fiber that was forked but has not started + * yet reaches its `Effect.sleep` before the clock steps over it. Turns, never time. + */ +const reactorSleeping = realTick; + +/** Advance the clock by whole poll intervals, waiting out each tick. */ +export const advancePolls = (polls: number) => + Effect.gen(function* () { + const { log } = yield* LoopReactorReceipts; + yield* reactorSleeping; + for (let i = 0; i < polls; i++) yield* advanceOnePoll(log); + }); + +/** + * Advance the clock in poll-sized steps until `condition` holds. + * + * The condition is evaluated once per *completed* tick, so it can never see a half-finished + * one and the clock's resting place is a property of the scenario alone. `maxPolls` bounds + * simulated time — how long the scenario is willing to wait — and nothing about the machine. + */ +export const advanceUntil = ( + condition: Effect.Effect, + description: string, + maxPolls = 600, +) => + Effect.gen(function* () { + const { log } = yield* LoopReactorReceipts; + yield* reactorSleeping; + for (let poll = 0; poll < maxPolls; poll++) { + if (yield* condition) return; + yield* advanceOnePoll(log); + } + if (yield* condition) return; + return yield* Effect.die( + new Error(`timed out waiting for ${description} after ${maxPolls} polls`), + ); + }); + +/** + * Move the clock with no reactor to wait for. + * + * Only correct where the supervisor is switched off and no tick will ever complete: awaiting + * a receipt that cannot come would hang instead of asserting. + */ +export const advanceWithoutReactor = (polls: number) => + Effect.gen(function* () { + for (let i = 0; i < polls; i++) { + yield* TestClock.adjust(Duration.millis(POLL_MS)); + yield* settleQuiet; + } + }); + +/** `true` once at least `n` turn starts have been dispatched. */ +export const turnStartsAtLeast = (dispatched: Ref.Ref, n: number) => + Ref.get(dispatched).pipe(Effect.map((all) => turnStarts(all).length >= n)); + +/** `true` once the record has a terminal state. */ +export const isStopped = (store: { + readonly getThread: (id: string) => Effect.Effect<{ stopped: unknown }>; +}) => store.getThread(LOOP_THREAD_ID).pipe(Effect.map((record) => record.stopped !== null)); diff --git a/apps/server/src/coil/loop/receipts.ts b/apps/server/src/coil/loop/receipts.ts new file mode 100644 index 000000000000..9789cb5f9a3e --- /dev/null +++ b/apps/server/src/coil/loop/receipts.ts @@ -0,0 +1,138 @@ +/** + * The loop supervisor's receipts. + * + * The reactor's milestones are all asynchronous and most of them end in real filesystem I/O, + * so a test that wants to assert about one has two choices: infer it from a store read after + * spinning the scheduler, or be told. Inference is what this fork kept paying for — CI failed + * three times on `Reactor.test.ts` and `integration.test.ts` with three different sets of + * tests, every one of them a wait that ran out of turns on a two-core runner rather than a + * defect in the product. AGENTS.md is explicit about the remedy: wait on receipts and worker + * drains, never on sleeps or polling. + * + * So the reactor announces. Every milestone a test currently infers is published here, and + * the harness waits on the announcement instead of guessing how many scheduler turns it + * should take. + * + * ## Nothing is paid for in production + * + * The service is **optional**. `receiptEmitter` resolves it with `Effect.serviceOption`, and + * when nobody has provided it — which is every production layer graph, since `CoilLayerLive` + * does not mention this module — `emit` is a constant `Effect.void` and `enabled` is false. + * The reactor uses `enabled` to skip even building a receipt whose fields would cost a read + * (`tick.completed` would otherwise read the clock on the empty-armed path, which is the one + * path that currently issues no queries of any kind). + * + * The buffer is bounded and **dropping**: a subscriber that stops draining loses receipts + * rather than blocking the tick. A supervisor that stalls because a test stopped listening + * would be a worse bug than the flake this replaces. + * + * @module coil/loop/receipts + */ + +import * as Context from "effect/Context"; +import * as Effect from "effect/Effect"; +import * as Layer from "effect/Layer"; +import * as Option from "effect/Option"; +import * as PubSub from "effect/PubSub"; +import * as Scope from "effect/Scope"; + +import type { StopRecord } from "./state.ts"; + +/** + * One announced milestone. + * + * Every variant names something a test used to infer. `tick.completed` is the important one: + * it is published at the **end** of every tick, including the early exit when nothing is + * armed, so "advance one poll and let the reactor finish" is a single exact await rather than + * a budget of scheduler turns. + */ +export type LoopReactorReceipt = + /** The Claude adapter's route to the store is open. Published once, at layer construction. */ + | { readonly type: "hooks.installed" } + /** A whole tick is over: every armed thread evaluated, every write durable. */ + | { readonly type: "tick.completed"; readonly armedCount: number; readonly nowMs: number } + /** Budget spent, before anything that can fail. */ + | { readonly type: "checkIn.reserved"; readonly threadId: string; readonly n: number } + /** The nudge landed in the engine. */ + | { readonly type: "checkIn.dispatched"; readonly threadId: string } + /** The nudge did not go out; `reason` is the verdict that stopped it. */ + | { readonly type: "checkIn.aborted"; readonly threadId: string; readonly reason: string } + /** Banked answers marked delivered, after the dispatch that carried them. */ + | { + readonly type: "blockers.delivered"; + readonly threadId: string; + readonly ids: ReadonlyArray; + } + /** A run reached a terminal state. */ + | { + readonly type: "stopped"; + readonly threadId: string; + readonly outcome: StopRecord["reason"]; + } + /** A record was disarmed without a terminal state (the thread went away). */ + | { readonly type: "disarmed"; readonly threadId: string } + /** A stop the console banked was cleared and the session ended. */ + | { readonly type: "stopRequest.serviced"; readonly threadId: string } + /** The rate-limit tap wrote a hold. */ + | { readonly type: "rateLimit.recorded"; readonly threadId: string; readonly untilMs: number } + /** A question the runtime raised was recorded against an armed thread. */ + | { readonly type: "userInput.recorded"; readonly threadId: string; readonly requestId: string }; + +export interface LoopReactorReceiptsShape { + readonly publish: (receipt: LoopReactorReceipt) => Effect.Effect; + /** + * Everything published for the life of the service, in order. + * + * Subscribed at construction rather than handed out per wait, because the service is built + * before the reactor is: the rate-limit tap can publish while the test body is still being + * assembled, and a subscription opened later would miss it and wait forever. + */ + readonly log: PubSub.Subscription; +} + +export class LoopReactorReceipts extends Context.Service< + LoopReactorReceipts, + LoopReactorReceiptsShape +>()("t3/coil/loop/receipts/LoopReactorReceipts") {} + +/** + * Room for every receipt a scenario can produce without draining. + * + * Scenarios run at most a few hundred simulated polls and drain the tick receipt on each one, + * so this is slack rather than a working limit — but it is a limit, because the alternative + * is a queue that grows with a stuck subscriber. + */ +const RECEIPT_BUFFER = 4096; + +export const makeLoopReactorReceipts: Effect.Effect = + Effect.gen(function* () { + const pubsub = yield* PubSub.dropping(RECEIPT_BUFFER); + const log = yield* PubSub.subscribe(pubsub); + return { + publish: (receipt) => PubSub.publish(pubsub, receipt).pipe(Effect.asVoid), + log, + }; + }); + +export const LoopReactorReceiptsLive = Layer.effect(LoopReactorReceipts, makeLoopReactorReceipts); + +const noEmit = (_receipt: LoopReactorReceipt): Effect.Effect => Effect.void; + +export interface LoopReceiptEmitter { + /** Whether anyone is listening. Lets a caller skip assembling a receipt, not just sending. */ + readonly enabled: boolean; + readonly emit: (receipt: LoopReactorReceipt) => Effect.Effect; +} + +/** + * Resolve the optional service into something the reactor can call unconditionally. + * + * `enabled` exists so a caller can skip *assembling* a receipt, not just publishing it: the + * empty-armed tick has no `nowMs` to hand and reading one where nobody is listening would + * spend a clock read on every poll of every T3 install forever. + */ +export const receiptEmitter: Effect.Effect = Effect.gen(function* () { + const service = yield* Effect.serviceOption(LoopReactorReceipts); + if (Option.isNone(service)) return { enabled: false, emit: noEmit }; + return { enabled: true, emit: service.value.publish }; +}); diff --git a/apps/server/src/coil/loop/sentinel.test.ts b/apps/server/src/coil/loop/sentinel.test.ts new file mode 100644 index 000000000000..95eabe44cc53 --- /dev/null +++ b/apps/server/src/coil/loop/sentinel.test.ts @@ -0,0 +1,446 @@ +// @effect-diagnostics nodeBuiltinImport:off +// @effect-diagnostics globalDate:off - `asDate` below converts fixture millis for `fs.utimes`; nothing reads a clock. +/** + * TESTS.md cases 47–58 — the done-file. + * + * Real `FileSystem` against a temp dir rather than a mock, because every interesting case + * here is a filesystem edge (mtime ordering, ENOTDIR, a symlink loop, a directory where a + * file should be) that a hand-written stub would model as whatever the author expected. + * Where a case is about *how* the module touches the filesystem rather than what it + * returns — call order (52) and never writing (56) — the real service is wrapped in a + * recording proxy. + */ +import * as NodeServices from "@effect/platform-node/NodeServices"; +import { assert, describe, it } from "@effect/vitest"; +import * as Effect from "effect/Effect"; +import * as FileSystem from "effect/FileSystem"; +import * as Path from "effect/Path"; +import * as NodePath from "node:path"; + +import { LOOP_DONE_RELATIVE_PATH } from "./config.ts"; +import { readSentinel } from "./sentinel.ts"; + +const ARMED_AT_MS = 1_700_000_000_000; + +/** + * Epoch millis as a `Date`, for `fs.utimes`. + * + * Node reads a bare number there as *seconds*, so passing `mtimeMs` straight through + * silently backdates every fixture by a factor of a thousand. + */ +const asDate = (ms: number) => new Date(ms); + +interface Probe { + readonly fs: FileSystem.FileSystem; + readonly stats: Array; + readonly mutations: Array; +} + +/** + * Wraps the real service, recording every `stat` and every mutating call. + * + * The mutators delegate rather than throw: the assertion that matters is that the array + * stays empty, and delegating keeps the wrapper honest about signatures instead of + * silently diverging from the interface it stands in for. + */ +const probe = (real: FileSystem.FileSystem): Probe => { + const stats: Array = []; + const mutations: Array = []; + const fs: FileSystem.FileSystem = { + ...real, + stat: (path) => { + stats.push(path); + return real.stat(path); + }, + writeFile: (path, data, options) => { + mutations.push(`writeFile:${path}`); + return real.writeFile(path, data, options); + }, + writeFileString: (path, data, options) => { + mutations.push(`writeFileString:${path}`); + return real.writeFileString(path, data, options); + }, + makeDirectory: (path, options) => { + mutations.push(`makeDirectory:${path}`); + return real.makeDirectory(path, options); + }, + remove: (path, options) => { + mutations.push(`remove:${path}`); + return real.remove(path, options); + }, + rename: (from, to) => { + mutations.push(`rename:${from}`); + return real.rename(from, to); + }, + utimes: (path, atime, mtime) => { + mutations.push(`utimes:${path}`); + return real.utimes(path, atime, mtime); + }, + open: (path, options) => { + mutations.push(`open:${path}`); + return real.open(path, options); + }, + }; + return { fs, stats, mutations }; +}; + +interface Fixture { + readonly worktreePath: string; + readonly workspaceRoot: string; + /** Writes `/.coil/loop-done` and forces its mtime. */ + readonly writeDone: ( + root: string, + options?: { readonly contents?: string; readonly mtimeMs?: number }, + ) => Effect.Effect; +} + +const withRoots = ( + f: (fixture: Fixture) => Effect.Effect, +) => + Effect.gen(function* () { + const fs = yield* FileSystem.FileSystem; + const base = yield* fs.makeTempDirectoryScoped({ prefix: "coil-loop-sentinel-" }); + const worktreePath = NodePath.join(base, "worktree"); + const workspaceRoot = NodePath.join(base, "workspace"); + yield* fs.makeDirectory(worktreePath, { recursive: true }); + yield* fs.makeDirectory(workspaceRoot, { recursive: true }); + + const writeDone: Fixture["writeDone"] = (root, options = {}) => + Effect.gen(function* () { + const donePath = NodePath.join(root, LOOP_DONE_RELATIVE_PATH); + yield* fs.makeDirectory(NodePath.dirname(donePath), { recursive: true }); + yield* fs.writeFileString(donePath, options.contents ?? "done\n"); + const mtimeMs = options.mtimeMs ?? ARMED_AT_MS + 60_000; + yield* fs.utimes(donePath, asDate(mtimeMs), asDate(mtimeMs)); + return donePath; + }).pipe(Effect.orDie); + + return yield* f({ worktreePath, workspaceRoot, writeDone }); + }).pipe(Effect.scoped, Effect.orDie, Effect.provide(NodeServices.layer), Effect.runPromise); + +describe("coil/loop/sentinel", () => { + // 47 + it("reports absent when neither root holds a done-file", () => + withRoots(({ worktreePath, workspaceRoot }) => + Effect.gen(function* () { + const fs = yield* FileSystem.FileSystem; + const sentinel = yield* readSentinel( + fs, + { worktreePath, workspaceRoot }, + { armedAtMs: ARMED_AT_MS }, + ); + assert.strictEqual(sentinel.kind, "absent"); + }), + )); + + // 48 + it("detects a done-file under the worktree", () => + withRoots(({ worktreePath, workspaceRoot, writeDone }) => + Effect.gen(function* () { + const fs = yield* FileSystem.FileSystem; + const donePath = yield* writeDone(worktreePath); + const sentinel = yield* readSentinel( + fs, + { worktreePath, workspaceRoot }, + { armedAtMs: ARMED_AT_MS }, + ); + assert.strictEqual(sentinel.kind, "done"); + assert.strictEqual(sentinel.kind === "done" ? sentinel.path : null, donePath); + }), + )); + + // 49 + it("detects a done-file under the workspace root", () => + withRoots(({ worktreePath, workspaceRoot, writeDone }) => + Effect.gen(function* () { + const fs = yield* FileSystem.FileSystem; + const donePath = yield* writeDone(workspaceRoot); + const sentinel = yield* readSentinel( + fs, + { worktreePath, workspaceRoot }, + { armedAtMs: ARMED_AT_MS }, + ); + assert.strictEqual(sentinel.kind, "done"); + assert.strictEqual(sentinel.kind === "done" ? sentinel.path : null, donePath); + }), + )); + + // 50 + it("takes the worktree copy when it is the newer of the two", () => + withRoots(({ worktreePath, workspaceRoot, writeDone }) => + Effect.gen(function* () { + const fs = yield* FileSystem.FileSystem; + yield* writeDone(workspaceRoot, { mtimeMs: ARMED_AT_MS + 60_000 }); + const newer = yield* writeDone(worktreePath, { mtimeMs: ARMED_AT_MS + 120_000 }); + const sentinel = yield* readSentinel( + fs, + { worktreePath, workspaceRoot }, + { armedAtMs: ARMED_AT_MS }, + ); + assert.strictEqual(sentinel.kind === "done" ? sentinel.path : null, newer); + }), + )); + + // 51 — newest mtime wins, NOT first found. Precedence orders the stats, not the verdict. + it("takes the workspace copy when it is the newer of the two", () => + withRoots(({ worktreePath, workspaceRoot, writeDone }) => + Effect.gen(function* () { + const fs = yield* FileSystem.FileSystem; + yield* writeDone(worktreePath, { mtimeMs: ARMED_AT_MS + 60_000 }); + const newer = yield* writeDone(workspaceRoot, { mtimeMs: ARMED_AT_MS + 120_000 }); + const sentinel = yield* readSentinel( + fs, + { worktreePath, workspaceRoot }, + { armedAtMs: ARMED_AT_MS }, + ); + assert.strictEqual(sentinel.kind === "done" ? sentinel.path : null, newer); + }), + )); + + // 52 — asserted by call order, not by outcome: an implementation that statted the + // workspace root first would still pass 48-51 while disagreeing with the agent's cwd. + it("stats the worktree before the workspace root", () => + withRoots(({ worktreePath, workspaceRoot, writeDone }) => + Effect.gen(function* () { + const real = yield* FileSystem.FileSystem; + const { fs, stats } = probe(real); + yield* writeDone(worktreePath); + yield* writeDone(workspaceRoot); + yield* readSentinel(fs, { worktreePath, workspaceRoot }, { armedAtMs: ARMED_AT_MS }); + assert.deepStrictEqual(stats, [ + NodePath.join(worktreePath, LOOP_DONE_RELATIVE_PATH), + NodePath.join(workspaceRoot, LOOP_DONE_RELATIVE_PATH), + ]); + }), + )); + + // 53 — a leftover from a previous run cannot end a new one. + it("ignores a done-file older than armedAtMs", () => + withRoots(({ worktreePath, workspaceRoot, writeDone }) => + Effect.gen(function* () { + const fs = yield* FileSystem.FileSystem; + yield* writeDone(worktreePath, { mtimeMs: ARMED_AT_MS - 60_000 }); + const sentinel = yield* readSentinel( + fs, + { worktreePath, workspaceRoot }, + { armedAtMs: ARMED_AT_MS }, + ); + assert.strictEqual(sentinel.kind, "stale"); + }), + )); + + // 53, the boundary: `mtime > armedAtMs` is strict, so an equal timestamp is stale. + it("treats a done-file written exactly at armedAtMs as stale", () => + withRoots(({ worktreePath, workspaceRoot, writeDone }) => + Effect.gen(function* () { + const fs = yield* FileSystem.FileSystem; + yield* writeDone(worktreePath, { mtimeMs: ARMED_AT_MS }); + const sentinel = yield* readSentinel( + fs, + { worktreePath, workspaceRoot }, + { armedAtMs: ARMED_AT_MS }, + ); + assert.strictEqual(sentinel.kind, "stale"); + }), + )); + + // 54 + it("honours a done-file newer than armedAtMs", () => + withRoots(({ worktreePath, workspaceRoot, writeDone }) => + Effect.gen(function* () { + const fs = yield* FileSystem.FileSystem; + yield* writeDone(worktreePath, { mtimeMs: ARMED_AT_MS + 1 }); + const sentinel = yield* readSentinel( + fs, + { worktreePath, workspaceRoot }, + { armedAtMs: ARMED_AT_MS }, + ); + assert.strictEqual(sentinel.kind, "done"); + assert.strictEqual(sentinel.kind === "done" ? sentinel.mtimeMs : 0, ARMED_AT_MS + 1); + }), + )); + + // 55 — a missing root directory. + it("reports absent when a root does not exist at all", () => + withRoots(({ worktreePath, workspaceRoot }) => + Effect.gen(function* () { + const fs = yield* FileSystem.FileSystem; + const sentinel = yield* readSentinel( + fs, + { + worktreePath: NodePath.join(worktreePath, "does", "not", "exist"), + workspaceRoot, + }, + { armedAtMs: ARMED_AT_MS }, + ); + assert.strictEqual(sentinel.kind, "absent"); + }), + )); + + // 55 — `.coil` is a regular file, so the candidate path is an ENOTDIR stat error. + it("reports absent when the .coil path is not a directory", () => + withRoots(({ worktreePath, workspaceRoot }) => + Effect.gen(function* () { + const fs = yield* FileSystem.FileSystem; + yield* fs.writeFileString(NodePath.join(worktreePath, ".coil"), "not a directory"); + const sentinel = yield* readSentinel( + fs, + { worktreePath, workspaceRoot }, + { armedAtMs: ARMED_AT_MS }, + ); + assert.strictEqual(sentinel.kind, "absent"); + }), + )); + + // 55 — a self-referential symlink: the stat loops and must not escape as a defect. + it("reports absent on a symlink loop", () => + withRoots(({ worktreePath, workspaceRoot }) => + Effect.gen(function* () { + const fs = yield* FileSystem.FileSystem; + const donePath = NodePath.join(worktreePath, LOOP_DONE_RELATIVE_PATH); + yield* fs.makeDirectory(NodePath.dirname(donePath), { recursive: true }); + yield* fs.symlink(donePath, donePath); + const sentinel = yield* readSentinel( + fs, + { worktreePath, workspaceRoot }, + { armedAtMs: ARMED_AT_MS }, + ); + assert.strictEqual(sentinel.kind, "absent"); + }), + )); + + // 55 — a directory named loop-done establishes no freshness, so it is not a signal. + it("reports absent when the done-file is a directory", () => + withRoots(({ worktreePath, workspaceRoot }) => + Effect.gen(function* () { + const fs = yield* FileSystem.FileSystem; + yield* fs.makeDirectory(NodePath.join(worktreePath, LOOP_DONE_RELATIVE_PATH), { + recursive: true, + }); + const sentinel = yield* readSentinel( + fs, + { worktreePath, workspaceRoot }, + { armedAtMs: ARMED_AT_MS }, + ); + assert.strictEqual(sentinel.kind, "absent"); + }), + )); + + // 56 — the supervisor is read-only. This is the property that makes it safe to point at + // an arbitrary repo, so it is asserted on the *hit* path, not just the absent one. + it("never writes to the user's tree", () => + withRoots(({ worktreePath, workspaceRoot, writeDone }) => + Effect.gen(function* () { + const real = yield* FileSystem.FileSystem; + yield* writeDone(worktreePath); + const { fs, mutations } = probe(real); + yield* readSentinel(fs, { worktreePath, workspaceRoot }, { armedAtMs: ARMED_AT_MS }); + yield* readSentinel( + fs, + { worktreePath: null, workspaceRoot: NodePath.join(workspaceRoot, "missing") }, + { armedAtMs: ARMED_AT_MS }, + ); + assert.deepStrictEqual(mutations, []); + }), + )); + + // 57 + it("captures the first line for display without letting contents decide", () => + withRoots(({ worktreePath, workspaceRoot, writeDone }) => + Effect.gen(function* () { + const fs = yield* FileSystem.FileSystem; + yield* writeDone(worktreePath, { + contents: " shipped the migration \nplus a second line nobody reads\n", + }); + const sentinel = yield* readSentinel( + fs, + { worktreePath, workspaceRoot }, + { armedAtMs: ARMED_AT_MS }, + ); + assert.strictEqual(sentinel.kind, "done"); + assert.strictEqual( + sentinel.kind === "done" ? sentinel.firstLine : null, + "shipped the migration", + ); + }), + )); + + // 57 — an empty file is still a done-file; only the display text is missing. + it("still reports done for an empty file", () => + withRoots(({ worktreePath, workspaceRoot, writeDone }) => + Effect.gen(function* () { + const fs = yield* FileSystem.FileSystem; + yield* writeDone(worktreePath, { contents: "" }); + const sentinel = yield* readSentinel( + fs, + { worktreePath, workspaceRoot }, + { armedAtMs: ARMED_AT_MS }, + ); + assert.strictEqual(sentinel.kind, "done"); + assert.strictEqual(sentinel.kind === "done" ? sentinel.firstLine : "unset", null); + }), + )); + + // 58 — freshness is mtimeMs only. A model that guesses the wall clock badly must not be + // able to make `done` unreachable (nor to reach it from a stale file). + it("never reads a timestamp written inside the file", () => + withRoots(({ worktreePath, workspaceRoot, writeDone }) => + Effect.gen(function* () { + const fs = yield* FileSystem.FileSystem; + // Contents claim 1999; the mtime is fresh, so this is done. + yield* writeDone(worktreePath, { + contents: `finished at 1999-01-01T00:00:00.000Z`, + mtimeMs: ARMED_AT_MS + 5_000, + }); + const fresh = yield* readSentinel( + fs, + { worktreePath, workspaceRoot }, + { armedAtMs: ARMED_AT_MS }, + ); + assert.strictEqual(fresh.kind, "done"); + + // Contents claim the future; the mtime is stale, so this is not. + yield* writeDone(workspaceRoot, { + contents: "finished at 2099-01-01T00:00:00.000Z", + mtimeMs: ARMED_AT_MS - 5_000, + }); + const stale = yield* readSentinel( + fs, + { worktreePath: null, workspaceRoot }, + { armedAtMs: ARMED_AT_MS }, + ); + assert.strictEqual(stale.kind, "stale"); + }), + )); + + // The roots collapse to one when a thread has no worktree, and a duplicate root is + // statted once — `resolveLoopRoots` dedupes, which keeps the ledger of stats honest. + it("stats a single root once when both inputs point at it", () => + withRoots(({ workspaceRoot, writeDone }) => + Effect.gen(function* () { + const real = yield* FileSystem.FileSystem; + const { fs, stats } = probe(real); + yield* writeDone(workspaceRoot); + const sentinel = yield* readSentinel( + fs, + { worktreePath: workspaceRoot, workspaceRoot }, + { armedAtMs: ARMED_AT_MS }, + ); + assert.strictEqual(sentinel.kind, "done"); + assert.strictEqual(stats.length, 1); + }), + )); + + it("reports absent when no root is known at all", () => + withRoots(() => + Effect.gen(function* () { + const fs = yield* FileSystem.FileSystem; + const sentinel = yield* readSentinel( + fs, + { worktreePath: null, workspaceRoot: null }, + { armedAtMs: ARMED_AT_MS }, + ); + assert.strictEqual(sentinel.kind, "absent"); + }), + )); +}); diff --git a/apps/server/src/coil/loop/sentinel.ts b/apps/server/src/coil/loop/sentinel.ts new file mode 100644 index 000000000000..80bbc12f6588 --- /dev/null +++ b/apps/server/src/coil/loop/sentinel.ts @@ -0,0 +1,147 @@ +/** + * The done-file sentinel — how an agent ends its own loop. + * + * The agent writes `/.coil/loop-done`; the supervisor only ever *stats* it. That + * asymmetry is the whole safety argument for pointing this feature at an arbitrary repo: + * there is no code path here that writes, creates or removes anything under the user's + * tree, and `readSentinel` takes its `FileSystem` as an argument partly so a test can hand + * it a proxy that fails on every mutating method. + * + * Three rules, each of which has a way of being quietly re-broken: + * + * 1. **Worktree first.** The roots are `worktreePath ?? workspaceRoot`, in that order, + * because `resolveThreadWorkspaceCwd` hands the *agent* the worktree. `autoResume/` + * resolves from `workspaceRoot` only; copying it would leave the agent writing the + * done-file where the supervisor never looks, on every worktree-backed thread. + * 2. **Freshness is `mtimeMs`, never a timestamp inside the file.** Models do not know the + * wall clock, so gating `done` on a model-authored timestamp makes `done` unreachable + * whenever it guesses wrong. Contents are read for *display* and never for detection. + * 3. **Every filesystem failure means "no sentinel".** EACCES, a symlink loop, a missing + * parent directory, a `.coil/loop-done` that is a directory, a stat with no mtime — all + * of them resolve to `absent` rather than a crash, because this runs inside the tick + * fiber and a defect there stops supervision for every thread on the machine. + * + * @module coil/loop/sentinel + */ + +import * as Effect from "effect/Effect"; +import type * as FileSystem from "effect/FileSystem"; +import * as Option from "effect/Option"; +import * as Path from "effect/Path"; + +import { LOOP_DONE_RELATIVE_PATH, resolveLoopRoots } from "./config.ts"; + +/** How much of the done-file's first line the console is offered. */ +const FIRST_LINE_MAX_CHARS = 200; + +export interface SentinelRoots { + readonly worktreePath: string | null; + readonly workspaceRoot: string | null; +} + +export interface SentinelOptions { + /** + * The run's `armedAtMs`. A done-file at or before it is a leftover from a previous run + * and must not end this one — which is the only reason `arm` takes a fresh `armedAtMs`. + */ + readonly armedAtMs: number; +} + +/** + * What the supervisor found. + * + * `firstLine` is `null` when the file could not be read or held nothing printable. That is + * deliberately *not* a fourth variant: an unreadable done-file is still a done-file, since + * detection is `mtimeMs` alone (rule 2). Modelling "unreadable" as its own state would make + * a permissions quirk on a file the agent already wrote look like the run never ended. + */ +export type LoopSentinel = + | { readonly kind: "absent" } + | { + /** Present, but not newer than `armedAtMs`: a leftover, not a signal. */ + readonly kind: "stale"; + readonly path: string; + readonly mtimeMs: number; + readonly firstLine: string | null; + } + | { + readonly kind: "done"; + readonly path: string; + readonly mtimeMs: number; + readonly firstLine: string | null; + }; + +export const SENTINEL_ABSENT: LoopSentinel = { kind: "absent" }; + +interface SentinelHit { + readonly path: string; + readonly mtimeMs: number; +} + +/** + * Stats one candidate path, resolving every failure mode to `null`. + * + * A non-`File` entry and a stat carrying no `mtime` are both rejected here rather than + * downstream: neither can establish freshness, and treating either as a hit would let a + * directory named `loop-done` end a run at the epoch. + */ +const statCandidate = ( + fs: FileSystem.FileSystem, + path: string, +): Effect.Effect => + fs.stat(path).pipe( + Effect.map((info): SentinelHit | null => { + if (info.type !== "File") return null; + const mtime = Option.getOrNull(info.mtime); + if (mtime === null) return null; + const mtimeMs = mtime.getTime(); + return Number.isFinite(mtimeMs) ? { path, mtimeMs } : null; + }), + Effect.orElseSucceed(() => null), + ); + +/** Display text only. An unreadable or blank file yields `null` and changes no verdict. */ +const readFirstLine = (fs: FileSystem.FileSystem, path: string): Effect.Effect => + fs.readFileString(path).pipe( + Effect.map((contents): string | null => { + const line = contents.split("\n", 1)[0]?.trim() ?? ""; + return line.length === 0 ? null : line.slice(0, FIRST_LINE_MAX_CHARS); + }), + Effect.orElseSucceed(() => null), + ); + +/** + * Reads the done-file across both roots. + * + * The roots are visited **worktree first** and every root is statted, because the newest + * `mtimeMs` wins rather than the first hit: a stale worktree copy must not mask a fresh one + * under the workspace root. Ties keep the earlier root, so the worktree stays authoritative + * when both files carry the same timestamp. + * + * `fs` is a parameter rather than a context read so the caller resolves it once per tick and + * so the read-only property is testable by substitution. + */ +export const readSentinel = ( + fs: FileSystem.FileSystem, + roots: SentinelRoots, + options: SentinelOptions, +): Effect.Effect => + Effect.gen(function* () { + const path = yield* Path.Path; + let best: SentinelHit | null = null; + for (const root of resolveLoopRoots(roots)) { + const hit = yield* statCandidate(fs, path.join(root, LOOP_DONE_RELATIVE_PATH)); + if (hit !== null && (best === null || hit.mtimeMs > best.mtimeMs)) { + best = hit; + } + } + if (best === null) return SENTINEL_ABSENT; + + const firstLine = yield* readFirstLine(fs, best.path); + return { + kind: best.mtimeMs > options.armedAtMs ? "done" : "stale", + path: best.path, + mtimeMs: best.mtimeMs, + firstLine, + }; + }); diff --git a/apps/server/src/coil/loop/sharing.test.ts b/apps/server/src/coil/loop/sharing.test.ts new file mode 100644 index 000000000000..57d94868d3dd --- /dev/null +++ b/apps/server/src/coil/loop/sharing.test.ts @@ -0,0 +1,100 @@ +/** + * Pins the layer-memoisation assumption the loop console depends on. + * + * `coil/index.ts` gives the supervisor (`CoilLayerLive`) and the HTTP routes + * (`CoilRoutesLive`) their record by `Layer.provide`-ing the *same* `LoopStoreLive` value to + * each, independently. That is deliberate — leaving `LoopStore` as an open requirement on the + * routes would widen upstream's `makeRoutesLayer` signature and break its tests. + * + * It only works because Effect memoises layer construction per build, so both consumers + * receive one instance. If that stopped holding, arming from the console would write one + * in-memory copy while the supervisor kept reading another: the UI would show an armed loop, + * `listArmed` would stay empty, and nothing would ever check in. The failure is completely + * silent, hence this test — and the stakes are higher here than for auto-resume, because both + * consumers *write*, so two copies would also race each other over one file. + * + * SCOPE, stated honestly: this builds its own store layer and two probe services rather than + * importing `CoilLayerLive` / `CoilRoutesLive`. It guards the *assumption* (one layer value + * provided to two independent consumers yields one instance) for the composition shape + * `coil/index.ts` uses — it does not exercise that file's actual graph, and would stay green + * if someone gave each consumer its own store layer. Importing the real layers here is not + * practical: both are `Layer.provide`d shut precisely so they leak no requirement. + */ +// @effect-diagnostics nodeBuiltinImport:off +import * as NodePath from "node:path"; + +import * as NodeServices from "@effect/platform-node/NodeServices"; +import { assert, describe, it } from "@effect/vitest"; +import * as Context from "effect/Context"; +import * as Effect from "effect/Effect"; +import * as FileSystem from "effect/FileSystem"; +import * as Layer from "effect/Layer"; + +import { LoopStore, type LoopStoreShape, makeLoopStore } from "./state.ts"; + +/** Two independent consumers, mirroring the supervisor and the routes. */ +class ProbeA extends Context.Service()( + "t3/coil/loop/sharing.test/ProbeA", +) {} +class ProbeB extends Context.Service()( + "t3/coil/loop/sharing.test/ProbeB", +) {} + +describe("LoopStore layer sharing", () => { + it("hands the same store instance to two consumers that each provide it independently", () => + Effect.gen(function* () { + const fs = yield* FileSystem.FileSystem; + const root = yield* fs.makeTempDirectoryScoped({ prefix: "coil-loop-share-" }); + const statePath = NodePath.join(root, "coil-loop.json"); + + // Exactly the shape used in coil/index.ts: ONE layer value, provided to each consumer. + const StoreLive = Layer.effect(LoopStore, makeLoopStore(statePath)); + + const ProbeALive = Layer.effect( + ProbeA, + Effect.gen(function* () { + return { store: yield* LoopStore }; + }), + ).pipe(Layer.provide(StoreLive)); + + const ProbeBLive = Layer.effect( + ProbeB, + Effect.gen(function* () { + return { store: yield* LoopStore }; + }), + ).pipe(Layer.provide(StoreLive)); + + yield* Effect.gen(function* () { + const a = yield* ProbeA; + const b = yield* ProbeB; + + assert.strictEqual(a.store, b.store, "both consumers must share one store instance"); + + // Behavioural proof, not just reference equality. This is the exact path the console + // takes: the route arms, and the supervisor's `listArmed` has to see it. + yield* a.store.setGlobal({ enabled: true }); + yield* a.store.arm({ + threadId: "thread-a", + armedAtMs: 1_000, + deadlineAtMs: 2_000_000, + maxCheckIns: 6, + }); + const armed = yield* b.store.listArmed; + assert.deepStrictEqual( + armed.map((entry) => entry.threadId), + ["thread-a"], + "a loop armed via one consumer must be visible to the other", + ); + assert.isTrue((yield* b.store.getGlobal).enabled, "and so must the master toggle"); + + // And back the other way: the supervisor's writes have to reach the console. + yield* b.store.recordCheckIn({ + threadId: "thread-a", + firedAtMs: 5_000, + createdAtIso: "1970-01-01T00:00:05.000Z", + activityCursor: "1970-01-01T00:00:00.000Z", + }); + assert.strictEqual((yield* a.store.getThread("thread-a")).checkInsUsed, 1); + }).pipe(Effect.provide(Layer.merge(ProbeALive, ProbeBLive))); + }).pipe(Effect.scoped, Effect.provide(NodeServices.layer), Effect.runPromise)); +}); diff --git a/apps/server/src/coil/loop/state.test.ts b/apps/server/src/coil/loop/state.test.ts new file mode 100644 index 000000000000..03ee7b48792f --- /dev/null +++ b/apps/server/src/coil/loop/state.test.ts @@ -0,0 +1,895 @@ +// @effect-diagnostics nodeBuiltinImport:off +import * as NodeServices from "@effect/platform-node/NodeServices"; +import { assert, describe, it } from "@effect/vitest"; +import * as Clock from "effect/Clock"; +import * as Effect from "effect/Effect"; +import * as FileSystem from "effect/FileSystem"; +import * as Path from "effect/Path"; +import * as NodePath from "node:path"; + +import { + type Blocker, + DEFAULT_GLOBAL_SETTINGS, + EMPTY_RECORD, + LoopGlobalSettings, + type LoopRecord, + LoopRecord as LoopRecordSchema, + type LoopStoreShape, + makeLoopStore, +} from "./state.ts"; + +const withStore = (f: (store: LoopStoreShape) => Effect.Effect) => + Effect.gen(function* () { + const fs = yield* FileSystem.FileSystem; + const root = yield* fs.makeTempDirectoryScoped({ prefix: "coil-loop-" }); + const store = yield* makeLoopStore(NodePath.join(root, "coil-loop.json")); + return yield* f(store); + }).pipe(Effect.scoped, Effect.orDie, Effect.provide(NodeServices.layer), Effect.runPromise); + +const withTempDir = ( + f: (root: string) => Effect.Effect, +) => + Effect.gen(function* () { + const fs = yield* FileSystem.FileSystem; + const root = yield* fs.makeTempDirectoryScoped({ prefix: "coil-loop-" }); + return yield* f(root); + }).pipe(Effect.scoped, Effect.orDie, Effect.provide(NodeServices.layer), Effect.runPromise); + +const arm = (threadId: string, overrides: Record = {}) => ({ + threadId, + armedAtMs: 1_000, + deadlineAtMs: 9_000_000, + maxCheckIns: 6, + ...overrides, +}); + +const blocker = (id: string, overrides: Partial = {}): Blocker => ({ + id, + raisedAtMs: 5_000, + question: "Should I take the migration or the shim?", + options: [{ label: "migration", description: "slower, correct" }], + context: "packages/contracts/src/settings.ts", + answeredAtMs: null, + answer: null, + deliveredToAgent: false, + ...overrides, +}); + +/** + * A record with every field set to something that is NOT its decoding default, so a test + * that drops one field can prove the rest survived rather than the whole file collapsing. + */ +const FULL_RECORD: LoopRecord = { + armed: true, + armedAtMs: 1_700_000_000_000, + goal: "land the sync", + maxCheckIns: 6, + checkInsUsed: 2, + deadlineAtMs: 1_700_028_800_000, + idleMs: 11 * 60_000, + busyIdleMs: 33 * 60_000, + crons: { + recordedAtMs: 1_700_000_100_000, + entries: [ + { + id: "cron-1", + schedule: "*/30 * * * *", + recurring: true, + prompt: "keep going", + nextFireAtMs: 1_700_001_800_000, + }, + ], + }, + degraded: "gate_off", + userInputs: [ + { + requestId: "req-1", + raisedAtMs: 1_700_000_200_000, + dialogKind: "resume_return", + question: "Resume this session?", + resolution: "voided", + resolvedAtMs: 1_700_000_300_000, + }, + ], + lastCheckIn: { firedAtMs: 1_700_000_400_000, createdAtIso: "2026-09-02T01:00:00.000Z" }, + checkIns: [ + { + n: 1, + firedAtMs: 1_700_000_400_000, + createdAtIso: "2026-09-02T01:00:00.000Z", + activityCursor: "act-42", + outcome: "productive", + }, + ], + strikes: 1, + rateLimitedUntilMs: 1_700_005_000_000, + pinnedByLoop: true, + stopRequestedAtMs: 1_700_005_500_000, + stopped: { reason: "stalled", atMs: 1_700_006_000_000, detail: "two quiet check-ins" }, + overridePrompt: "resume the migration", + blockers: [ + blocker("b-1", { answeredAtMs: 1_700_007_000_000, answer: "shim", deliveredToAgent: true }), + ], + loopDoneAtMs: 1_700_008_000_000, + loopDoneReason: "migration landed", +}; + +// These two pin the exact on-disk bytes rather than round-tripping through the encoder, so +// the cases below assert what a real file written by another build looks like. +const stateFileWith = (record: Record) => + JSON.stringify({ version: 1, threads: { "thread-a": record } }); + +const globalFileWith = (global: Record) => + JSON.stringify({ version: 1, global, threads: {} }); + +describe("LoopStore — rehydrate", () => { + it("59 — a missing state file reads as empty and never throws", () => + withStore((store) => + Effect.gen(function* () { + assert.deepStrictEqual(yield* store.listArmed, []); + assert.deepStrictEqual(yield* store.getGlobal, DEFAULT_GLOBAL_SETTINGS); + assert.deepStrictEqual(yield* store.getThread("never-seen"), EMPTY_RECORD); + }), + )); + + it("60 — a record round-trips through disk unchanged", () => + withTempDir((root) => + Effect.gen(function* () { + const path = NodePath.join(root, "coil-loop.json"); + const store1 = yield* makeLoopStore(path); + yield* store1.update("thread-a", () => FULL_RECORD); + + const store2 = yield* makeLoopStore(path); + assert.deepStrictEqual(yield* store2.getThread("thread-a"), FULL_RECORD); + }), + )); + + it("62 — a corrupt file reads as empty rather than throwing at boot", () => + withTempDir((root) => + Effect.gen(function* () { + const fs = yield* FileSystem.FileSystem; + const garbage = NodePath.join(root, "garbage.json"); + yield* fs.writeFileString(garbage, '{"version":1,"threads":{ truncated'); + const store = yield* makeLoopStore(garbage); + assert.deepStrictEqual(yield* store.listArmed, []); + }), + )); + + it("62 — a future schema version reads as empty rather than throwing at boot", () => + withTempDir((root) => + Effect.gen(function* () { + const fs = yield* FileSystem.FileSystem; + const path = NodePath.join(root, "v2.json"); + yield* fs.writeFileString(path, '{"version":2,"threads":{}}'); + const store = yield* makeLoopStore(path); + assert.deepStrictEqual(yield* store.listArmed, []); + }), + )); + + it("62 — a present-but-unreadable file is preserved and the session runs in memory", () => + withTempDir((root) => + Effect.gen(function* () { + const fs = yield* FileSystem.FileSystem; + // A directory at the state path is *present* but unreadable as a file (EISDIR). The + // store must not mistake that for a fresh store: overwriting it would be the + // transient-I/O-error path silently disarming every loop on the machine. + const dirPath = NodePath.join(root, "state-is-a-dir"); + yield* fs.makeDirectory(dirPath); + + const store = yield* makeLoopStore(dirPath); + yield* store.arm(arm("thread-a")); + assert.strictEqual((yield* store.listArmed).length, 1); + + assert.strictEqual((yield* fs.stat(dirPath)).type, "Directory"); + }), + )); + + it("63 — unknown keys from a newer build are tolerated", () => + withTempDir((root) => + Effect.gen(function* () { + const fs = yield* FileSystem.FileSystem; + const path = NodePath.join(root, "forward.json"); + // `workSource` lands with the maintainer loop (#44); a build that predates it must + // not choke on a file a newer build wrote. + yield* fs.writeFileString( + path, + stateFileWith({ ...FULL_RECORD, workSource: "issue-queue" }), + ); + const store = yield* makeLoopStore(path); + assert.strictEqual((yield* store.getThread("thread-a")).goal, "land the sync"); + }), + )); +}); + +// The highest-severity footgun in the module: a missing REQUIRED key fails the whole-file +// decode, the boot path turns that into EMPTY_STATE, and every armed loop on the machine is +// silently disarmed. This suite is schema-reflective on purpose — adding a field without a +// decoding default fails it without anyone remembering to write a case. +describe("LoopStore — fail-closed decoding defaults", () => { + for (const field of Object.keys(LoopRecordSchema.fields)) { + it(`61 — a record written without \`${field}\` still decodes, and the rest survives`, () => + withTempDir((root) => + Effect.gen(function* () { + const fs = yield* FileSystem.FileSystem; + const path = NodePath.join(root, `missing-${field}.json`); + const partial: Record = { ...FULL_RECORD }; + delete partial[field]; + yield* fs.writeFileString(path, stateFileWith(partial)); + + const store = yield* makeLoopStore(path); + const record = yield* store.getThread("thread-a"); + const witness = field === "armedAtMs" ? "maxCheckIns" : "armedAtMs"; + assert.deepStrictEqual( + record[witness], + FULL_RECORD[witness], + "the file must not have collapsed to EMPTY_STATE", + ); + }), + )); + } + + for (const field of Object.keys(LoopGlobalSettings.fields)) { + it(`61 — global settings written without \`${field}\` still decode`, () => + withTempDir((root) => + Effect.gen(function* () { + const fs = yield* FileSystem.FileSystem; + const path = NodePath.join(root, `missing-global-${field}.json`); + const partial: Record = { + ...DEFAULT_GLOBAL_SETTINGS, + maxArmedThreads: 7, + defaultMaxCheckIns: 9, + }; + delete partial[field]; + yield* fs.writeFileString(path, globalFileWith(partial)); + + const store = yield* makeLoopStore(path); + const global = yield* store.getGlobal; + // The witness is never the deleted field, so a surviving non-default value proves + // the whole file decoded rather than collapsing to EMPTY_STATE. + const witness = field === "maxArmedThreads" ? "defaultMaxCheckIns" : "maxArmedThreads"; + assert.strictEqual(global[witness], field === "maxArmedThreads" ? 9 : 7); + }), + )); + } + + it("61 — an entirely absent thread record decodes to every default", () => + withTempDir((root) => + Effect.gen(function* () { + const fs = yield* FileSystem.FileSystem; + const path = NodePath.join(root, "bare.json"); + yield* fs.writeFileString(path, stateFileWith({})); + const store = yield* makeLoopStore(path); + assert.deepStrictEqual(yield* store.getThread("thread-a"), EMPTY_RECORD); + }), + )); + + it("61b — every default is the fail-closed reading, not merely present", () => + withTempDir((root) => + Effect.gen(function* () { + const fs = yield* FileSystem.FileSystem; + const path = NodePath.join(root, "fail-closed.json"); + // The top level itself is missing `global` and the record is empty: exactly what a + // truncated write or an older build leaves behind. + yield* fs.writeFileString(path, stateFileWith({})); + const store = yield* makeLoopStore(path); + + const global = yield* store.getGlobal; + assert.strictEqual(global.enabled, false, "the master toggle must default OFF"); + + const record = yield* store.getThread("thread-a"); + assert.strictEqual(record.armed, false, "nothing is supervised implicitly"); + // 0 is always <= now, so the stop sweep ends the run on its first evaluation. A + // default meaning "unbounded" would turn one truncated write into an unbounded spend. + assert.strictEqual(record.deadlineAtMs, 0, "an unknown deadline is already spent"); + assert.strictEqual(record.maxCheckIns, 0, "an unknown budget is already spent"); + assert.strictEqual(record.crons, null, "never observed, which is not the same as empty"); + assert.strictEqual(record.pinnedByLoop, false, "never unpin a pin the loop did not make"); + assert.strictEqual(record.stopped, null); + assert.deepStrictEqual([...record.blockers], []); + }), + )); +}); + +describe("LoopStore — arming", () => { + it("59 — arming seeds the thresholds from the global settings", () => + withStore((store) => + Effect.gen(function* () { + yield* store.setGlobal({ defaultIdleMs: 7 * 60_000, defaultBusyIdleMs: 21 * 60_000 }); + const record = yield* store.arm(arm("thread-a", { goal: "ship it" })); + assert.strictEqual(record.armed, true); + assert.strictEqual(record.idleMs, 7 * 60_000); + assert.strictEqual(record.busyIdleMs, 21 * 60_000); + assert.strictEqual(record.goal, "ship it"); + assert.deepStrictEqual(yield* store.listArmed, [{ threadId: "thread-a", record }]); + }), + )); + + it("59 — an explicit threshold wins over the global default", () => + withStore((store) => + Effect.gen(function* () { + const record = yield* store.arm(arm("thread-a", { idleMs: 60_000 })); + assert.strictEqual(record.idleMs, 60_000); + }), + )); + + it("59 — disarm keeps the budget so a re-arm is a deliberate act, not a repair", () => + withStore((store) => + Effect.gen(function* () { + yield* store.arm(arm("thread-a")); + yield* store.recordCheckIn({ + threadId: "thread-a", + firedAtMs: 2_000, + createdAtIso: "2026-09-02T01:00:00.000Z", + activityCursor: "act-1", + }); + yield* store.disarm("thread-a"); + + const record = yield* store.getThread("thread-a"); + assert.strictEqual(record.armed, false); + assert.strictEqual(record.checkInsUsed, 1, "a takeover is not a budget reset"); + assert.strictEqual(record.stopped, null, "guard 4's disarm writes no terminal state"); + }), + )); + + it("59 — a terminal state is sticky and only a re-arm clears it", () => + withStore((store) => + Effect.gen(function* () { + yield* store.arm(arm("thread-a")); + yield* store.recordCheckIn({ + threadId: "thread-a", + firedAtMs: 2_000, + createdAtIso: "2026-09-02T01:00:00.000Z", + activityCursor: "act-1", + }); + yield* store.update("thread-a", (record) => ({ ...record, strikes: 2 })); + yield* store.stop("thread-a", { reason: "spent", atMs: 3_000, detail: "budget" }); + + const stopped = yield* store.getThread("thread-a"); + assert.strictEqual(stopped.armed, false); + assert.strictEqual(stopped.stopped?.reason, "spent"); + assert.deepStrictEqual(yield* store.listArmed, []); + + const rearmed = yield* store.arm(arm("thread-a", { armedAtMs: 5_000 })); + assert.strictEqual(rearmed.stopped, null); + assert.strictEqual(rearmed.armed, true); + assert.strictEqual(rearmed.armedAtMs, 5_000); + assert.strictEqual(rearmed.checkInsUsed, 0); + assert.strictEqual(rearmed.strikes, 0); + assert.deepStrictEqual([...rearmed.checkIns], []); + }), + )); + + it("59 — a re-arm keeps the facts that outlive a run", () => + withStore((store) => + Effect.gen(function* () { + yield* store.arm(arm("thread-a")); + yield* store.setRateLimitedUntil("thread-a", 8_000); + yield* store.setCrons("thread-a", { + recordedAtMs: 100, + entries: [ + { + id: "c1", + schedule: "0 * * * *", + recurring: true, + prompt: "p", + nextFireAtMs: 900, + }, + ], + }); + yield* store.addBlocker("thread-a", blocker("b-1")); + yield* store.recordUserInput("thread-a", { + requestId: "req-1", + raisedAtMs: 200, + dialogKind: null, + question: "q", + resolution: null, + resolvedAtMs: null, + }); + + const rearmed = yield* store.arm(arm("thread-a", { armedAtMs: 5_000 })); + // An account limit outlives the run, the provider's cron table belongs to the + // session, and an answer banked before the re-arm is still owed to the agent. + assert.strictEqual(rearmed.rateLimitedUntilMs, 8_000); + assert.strictEqual(rearmed.crons?.entries.length, 1); + assert.strictEqual(rearmed.blockers.length, 1); + assert.strictEqual(rearmed.userInputs.length, 1); + }), + )); + + it("59 — global settings round-trip and survive a restart", () => + withTempDir((root) => + Effect.gen(function* () { + const path = NodePath.join(root, "coil-loop.json"); + const store1 = yield* makeLoopStore(path); + const updated = yield* store1.setGlobal({ enabled: true, maxArmedThreads: 5 }); + assert.strictEqual(updated.enabled, true); + assert.strictEqual(updated.defaultMaxCheckIns, 6, "an unset key keeps its default"); + + const store2 = yield* makeLoopStore(path); + const global = yield* store2.getGlobal; + assert.strictEqual(global.enabled, true); + assert.strictEqual(global.maxArmedThreads, 5); + }), + )); +}); + +describe("LoopStore — check-ins and durability", () => { + it("67 — recordCheckIn is on disk before it returns, so a reservation precedes dispatch", () => + withTempDir((root) => + Effect.gen(function* () { + const fs = yield* FileSystem.FileSystem; + const path = NodePath.join(root, "coil-loop.json"); + const store = yield* makeLoopStore(path); + yield* store.arm(arm("thread-a")); + + const row = yield* store.recordCheckIn({ + threadId: "thread-a", + firedAtMs: 4_000, + createdAtIso: "2026-09-02T01:00:00.000Z", + activityCursor: "act-7", + }); + assert.strictEqual(row.n, 1); + assert.strictEqual(row.outcome, "unknown"); + + // Read the raw bytes: a provider that cannot spawn must burn budget, not tight-loop. + const contents = yield* fs.readFileString(path); + assert.ok(contents.includes('"checkInsUsed":1'), contents); + assert.ok(contents.includes('"activityCursor":"act-7"'), contents); + }), + )); + + it("67 — the ledger records the cursor and fire time at nudge time", () => + withStore((store) => + Effect.gen(function* () { + yield* store.arm(arm("thread-a")); + yield* store.recordCheckIn({ + threadId: "thread-a", + firedAtMs: 4_000, + createdAtIso: "2026-09-02T01:00:00.000Z", + activityCursor: "act-1", + }); + yield* store.recordCheckIn({ + threadId: "thread-a", + firedAtMs: 5_000, + createdAtIso: "2026-09-02T02:00:00.000Z", + activityCursor: "act-2", + }); + + const record = yield* store.getThread("thread-a"); + assert.strictEqual(record.checkInsUsed, 2); + assert.deepStrictEqual( + record.checkIns.map((row) => [row.n, row.activityCursor]), + [ + [1, "act-1"], + [2, "act-2"], + ], + ); + assert.deepStrictEqual(record.lastCheckIn, { + firedAtMs: 5_000, + createdAtIso: "2026-09-02T02:00:00.000Z", + }); + }), + )); + + it("60 — a rate limit is durable and survives a restart", () => + withTempDir((root) => + Effect.gen(function* () { + const path = NodePath.join(root, "coil-loop.json"); + const store1 = yield* makeLoopStore(path); + yield* store1.setRateLimitedUntil("thread-a", 1_700_000_000_000); + + const store2 = yield* makeLoopStore(path); + assert.strictEqual( + (yield* store2.getThread("thread-a")).rateLimitedUntilMs, + 1_700_000_000_000, + ); + }), + )); + + it("64 — concurrent mutations serialize with no lost update", () => + withStore((store) => + Effect.gen(function* () { + yield* store.arm(arm("thread-a")); + yield* Effect.all( + Array.from({ length: 64 }, () => + store.update("thread-a", (record) => ({ ...record, strikes: record.strikes + 1 })), + ), + { concurrency: "unbounded" }, + ); + assert.strictEqual((yield* store.getThread("thread-a")).strikes, 64); + }), + )); + + it("65 — the file is never observable half-written", () => + withTempDir((root) => + Effect.gen(function* () { + const fs = yield* FileSystem.FileSystem; + const path = NodePath.join(root, "coil-loop.json"); + const store = yield* makeLoopStore(path); + yield* store.arm(arm("thread-a")); + + // Read the file repeatedly while it is being rewritten. The write is a rename over a + // fully-written temp file, so every read that finds a file must find a whole one. + const reads: Array = []; + const reader = Effect.forEach( + Array.from({ length: 200 }, (_, index) => index), + () => + fs.readFileString(path).pipe( + Effect.map((contents) => { + reads.push(contents); + }), + Effect.orElseSucceed(() => undefined), + Effect.flatMap(() => Effect.yieldNow), + ), + { discard: true }, + ); + const writer = Effect.all( + Array.from({ length: 100 }, (_, index) => + store.update("thread-a", (record) => ({ ...record, strikes: index })), + ), + { concurrency: "unbounded" }, + ); + + yield* Effect.all([reader, writer], { concurrency: "unbounded" }); + + assert.ok(reads.length > 0, "the reader never observed the file at all"); + for (const contents of reads) { + assert.ok( + contents.endsWith("}\n") && contents.startsWith('{"version":1'), + `observed a partial document: ${contents.slice(0, 80)}`, + ); + } + }), + )); + + it("66 — a failed persist does not fail the mutation, and memory stays authoritative", () => + withTempDir((root) => + Effect.gen(function* () { + const fs = yield* FileSystem.FileSystem; + const path = NodePath.join(root, "nested", "coil-loop.json"); + const store = yield* makeLoopStore(path); + + // Put a regular file where the state directory needs to be, so every subsequent + // write fails at `makeDirectory`. + yield* fs.writeFileString(NodePath.join(root, "nested"), "not a directory"); + + yield* store.arm(arm("thread-a")); + const record = yield* store.getThread("thread-a"); + assert.strictEqual(record.armed, true, "the mutation must still succeed in memory"); + assert.strictEqual( + (yield* fs.stat(NodePath.join(root, "nested"))).type, + "File", + "nothing on disk was clobbered", + ); + // And the write really did fail, so this is not a vacuous assertion: a restart from + // the same path finds nothing, rather than a state file claiming a persist that + // never happened. + const restarted = yield* makeLoopStore(path); + assert.deepStrictEqual([...(yield* restarted.listArmed)], []); + }), + )); +}); + +describe("LoopStore — hostile thread ids", () => { + // `threads` is a plain object, so these ids resolve on Object.prototype and are truthy. A + // `?? EMPTY_RECORD` lookup would hand back a prototype method typed as a LoopRecord: reads + // blow up, and writes persist a record with no keys, which fails the whole-file decode on + // the next boot and disarms every loop. The HTTP route takes threadId straight from the + // caller, so this is reachable input. + for (const hostile of ["constructor", "toString", "valueOf", "hasOwnProperty", "__proto__"]) { + it(`68 — the prototype-chain threadId "${hostile}" reads as an absent record`, () => + withStore((store) => + Effect.gen(function* () { + assert.deepStrictEqual(yield* store.getThread(hostile), EMPTY_RECORD); + }), + )); + } + + it("68 — a write under a prototype-chain threadId does not corrupt other threads", () => + withTempDir((root) => + Effect.gen(function* () { + const path = NodePath.join(root, "coil-loop.json"); + const store1 = yield* makeLoopStore(path); + yield* store1.arm(arm("real-thread")); + yield* store1.arm(arm("__proto__", { armedAtMs: 42 })); + yield* store1.setOverridePrompt("constructor", "x"); + + const store2 = yield* makeLoopStore(path); + assert.strictEqual((yield* store2.listArmed).length, 2); + assert.strictEqual((yield* store2.getThread("__proto__")).armedAtMs, 42); + assert.strictEqual((yield* store2.getThread("constructor")).overridePrompt, "x"); + assert.strictEqual((yield* store2.getThread("real-thread")).armed, true); + }), + )); +}); + +describe("LoopStore — recorded user inputs", () => { + const requested = (requestId: string, dialogKind: string | null) => ({ + requestId, + raisedAtMs: 1_000, + dialogKind, + question: "Which branch?", + resolution: null, + resolvedAtMs: null, + }); + + it("60 — a requested input is recorded once, with its dialog kind", () => + withStore((store) => + Effect.gen(function* () { + yield* store.recordUserInput("thread-a", requested("req-1", "resume_return")); + yield* store.recordUserInput("thread-a", requested("req-1", "resume_return")); + const record = yield* store.getThread("thread-a"); + assert.strictEqual(record.userInputs.length, 1); + assert.strictEqual(record.userInputs[0]?.dialogKind, "resume_return"); + }), + )); + + it("60 — a voided input is distinguishable from an answered one", () => + withStore((store) => + Effect.gen(function* () { + yield* store.recordUserInput("thread-a", requested("req-1", null)); + yield* store.recordUserInput("thread-a", requested("req-2", null)); + yield* store.resolveUserInput("thread-a", "req-1", "answered", 2_000); + yield* store.resolveUserInput("thread-a", "req-2", "voided", 2_500); + + const record = yield* store.getThread("thread-a"); + assert.deepStrictEqual( + record.userInputs.map((entry) => [entry.requestId, entry.resolution]), + [ + ["req-1", "answered"], + ["req-2", "voided"], + ], + ); + }), + )); + + it("60 — a teardown void cannot overwrite a human's answer", () => + withStore((store) => + Effect.gen(function* () { + yield* store.recordUserInput("thread-a", requested("req-1", null)); + yield* store.resolveUserInput("thread-a", "req-1", "answered", 2_000); + yield* store.resolveUserInput("thread-a", "req-1", "voided", 3_000); + + const record = yield* store.getThread("thread-a"); + assert.strictEqual(record.userInputs[0]?.resolution, "answered"); + assert.strictEqual(record.userInputs[0]?.resolvedAtMs, 2_000); + }), + )); +}); + +describe("LoopStore — crons and degradation", () => { + it("60 — `null` and an empty entry list are different facts", () => + withStore((store) => + Effect.gen(function* () { + assert.strictEqual((yield* store.getThread("thread-a")).crons, null); + yield* store.setCrons("thread-a", { recordedAtMs: 10, entries: [] }); + const record = yield* store.getThread("thread-a"); + assert.notStrictEqual(record.crons, null, "observed-and-empty is not never-observed"); + assert.deepStrictEqual([...(record.crons?.entries ?? [])], []); + }), + )); + + it("60 — a degraded state is set and cleared only explicitly", () => + withStore((store) => + Effect.gen(function* () { + yield* store.setDegraded("thread-a", "gate_off"); + assert.strictEqual((yield* store.getThread("thread-a")).degraded, "gate_off"); + yield* store.setCrons("thread-a", { recordedAtMs: 10, entries: [] }); + assert.strictEqual( + (yield* store.getThread("thread-a")).degraded, + "gate_off", + "an unrelated write must not clear it", + ); + yield* store.setDegraded("thread-a", null); + assert.strictEqual((yield* store.getThread("thread-a")).degraded, null); + }), + )); +}); + +describe("LoopStore — blockers", () => { + it("69 — add, answer, list-unanswered and the delivered flip all persist", () => + withTempDir((root) => + Effect.gen(function* () { + const path = NodePath.join(root, "coil-loop.json"); + const store1 = yield* makeLoopStore(path); + yield* store1.addBlocker("thread-a", blocker("b-1")); + yield* store1.addBlocker("thread-a", blocker("b-2", { question: "Ship or hold?" })); + + assert.deepStrictEqual( + (yield* store1.listOpenBlockers("thread-a")).map((entry) => entry.id), + ["b-1", "b-2"], + ); + + yield* store1.answerBlocker("thread-a", "b-1", "take the shim", 6_000); + assert.deepStrictEqual( + (yield* store1.listOpenBlockers("thread-a")).map((entry) => entry.id), + ["b-2"], + ); + assert.deepStrictEqual( + (yield* store1.listUndeliveredAnswers("thread-a")).map((entry) => entry.id), + ["b-1"], + ); + + yield* store1.markBlockersDelivered("thread-a", ["b-1"]); + assert.deepStrictEqual([...(yield* store1.listUndeliveredAnswers("thread-a"))], []); + + const store2 = yield* makeLoopStore(path); + const record = yield* store2.getThread("thread-a"); + assert.strictEqual(record.blockers.length, 2); + assert.strictEqual(record.blockers[0]?.answer, "take the shim"); + assert.strictEqual(record.blockers[0]?.deliveredToAgent, true); + assert.strictEqual(record.blockers[1]?.answeredAtMs, null); + }), + )); + + it("69 — an answer that lands after composition is not marked delivered", () => + withStore((store) => + Effect.gen(function* () { + yield* store.addBlocker("thread-a", blocker("b-1")); + yield* store.addBlocker("thread-a", blocker("b-2")); + yield* store.answerBlocker("thread-a", "b-1", "yes", 6_000); + + // The prompt was composed with b-1 only; b-2 was answered while it was being built. + const composed = ["b-1"]; + yield* store.answerBlocker("thread-a", "b-2", "no", 6_500); + yield* store.markBlockersDelivered("thread-a", composed); + + assert.deepStrictEqual( + (yield* store.listUndeliveredAnswers("thread-a")).map((entry) => entry.id), + ["b-2"], + ); + }), + )); + + it("70 — answering an already-answered blocker keeps the first answer", () => + withStore((store) => + Effect.gen(function* () { + yield* store.addBlocker("thread-a", blocker("b-1")); + yield* store.answerBlocker("thread-a", "b-1", "first", 6_000); + const second = yield* store.answerBlocker("thread-a", "b-1", "second", 7_000); + + assert.strictEqual(second?.answer, "first"); + const record = yield* store.getThread("thread-a"); + assert.strictEqual(record.blockers.length, 1, "not a second append"); + assert.strictEqual(record.blockers[0]?.answer, "first"); + assert.strictEqual(record.blockers[0]?.answeredAtMs, 6_000); + }), + )); + + it("70 — answering an unknown blocker is a no-op, not a crash", () => + withStore((store) => + Effect.gen(function* () { + assert.strictEqual(yield* store.answerBlocker("thread-a", "nope", "x", 1), null); + assert.deepStrictEqual([...(yield* store.getThread("thread-a")).blockers], []); + }), + )); + + it("70 — adding a blocker twice under one id does not duplicate it", () => + withStore((store) => + Effect.gen(function* () { + yield* store.addBlocker("thread-a", blocker("b-1")); + yield* store.addBlocker("thread-a", blocker("b-1", { question: "different text" })); + const record = yield* store.getThread("thread-a"); + assert.strictEqual(record.blockers.length, 1); + assert.strictEqual(record.blockers[0]?.question, blocker("b-1").question); + }), + )); +}); + +/** + * One file is shared by every thread on the machine and rewritten in full on every mutation, + * so anything that only ever grows is a cost every other thread pays. + */ +describe("LoopStore — the file is bounded", () => { + it("caps blockers and recorded questions at fifty per thread, oldest first", () => + withStore((store) => + Effect.gen(function* () { + yield* store.arm(arm("thread-a")); + for (let n = 0; n < 60; n += 1) { + yield* store.addBlocker("thread-a", blocker(`b-${n}`)); + yield* store.recordUserInput("thread-a", { + requestId: `req-${n}`, + raisedAtMs: n, + dialogKind: null, + question: `question ${n}`, + resolution: null, + resolvedAtMs: null, + }); + } + const record = yield* store.getThread("thread-a"); + assert.lengthOf(record.blockers, 50); + assert.strictEqual(record.blockers[0]?.id, "b-10", "the oldest fall off"); + assert.strictEqual(record.blockers.at(-1)?.id, "b-59"); + assert.lengthOf(record.userInputs, 50); + assert.strictEqual(record.userInputs[0]?.requestId, "req-10"); + }), + )); + + it("drops a stopped record once it is older than the retention window", () => + withStore((store) => + Effect.gen(function* () { + yield* store.arm(arm("ancient")); + // Epoch: months in the past against the real clock these writes read. + yield* store.stop("ancient", { reason: "spent", atMs: 0, detail: "last winter" }); + yield* store.arm(arm("recent")); + yield* store.stop("recent", { + reason: "done", + atMs: yield* Clock.currentTimeMillis, + detail: "this morning", + }); + + // A write about a THIRD thing sweeps: the thread a write is about is exempt from its + // own prune, or a write followed by a read would look like the write never happened. + yield* store.setGlobal({ enabled: true }); + assert.deepStrictEqual(yield* store.getThread("ancient"), EMPTY_RECORD); + assert.strictEqual((yield* store.getThread("recent")).stopped?.reason, "done"); + }), + )); + + it("never drops an armed record, whatever its age", () => + withStore((store) => + Effect.gen(function* () { + yield* store.arm(arm("armed-forever", { armedAtMs: 0 })); + yield* store.setGlobal({ enabled: true }); + assert.isTrue((yield* store.getThread("armed-forever")).armed); + assert.lengthOf(yield* store.listArmed, 1); + }), + )); + + it("does not keep a record left carrying nothing", () => + withTempDir((root) => + Effect.gen(function* () { + const fs = yield* FileSystem.FileSystem; + const path = NodePath.join(root, "coil-loop.json"); + const store = yield* makeLoopStore(path); + // A resolution runs for every thread on the machine, armed or not — it is a no-op when + // nothing matches, and it must not leave an empty record behind for each one. + yield* store.resolveUserInput("passer-by", "req-1", "voided", 1_000); + yield* store.setGlobal({ enabled: true }); + // The file, not the reader: `getThread` answers `EMPTY_RECORD` either way, so only the + // bytes can say whether the row is gone. + assert.notInclude(yield* fs.readFileString(path), "passer-by"); + }), + )); + + it("clearThread forgets a finished run entirely", () => + withStore((store) => + Effect.gen(function* () { + yield* store.arm(arm("thread-a")); + yield* store.stop("thread-a", { reason: "done", atMs: 5_000, detail: "finished" }); + yield* store.clearThread("thread-a"); + assert.deepStrictEqual(yield* store.getThread("thread-a"), EMPTY_RECORD); + // And clearing something that was never there is a no-op, not a crash. + yield* store.clearThread("never-seen"); + }), + )); + + it("a banked session stop is listable and clearable, and a fresh arm cancels it", () => + withStore((store) => + Effect.gen(function* () { + yield* store.arm(arm("thread-a")); + yield* store.requestSessionStop("thread-a", 7_000); + // Both work lists come from ONE read: the tick runs every minute forever, and a + // second trip through the synchronized ref is a scheduler turn per tick for a list + // that is nearly always empty. + const work = yield* store.listWork; + assert.deepStrictEqual( + work.stopRequested.map((entry) => entry.threadId), + ["thread-a"], + ); + assert.deepStrictEqual( + work.armed.map((entry) => entry.threadId), + ["thread-a"], + ); + + yield* store.clearSessionStopRequest("thread-a"); + assert.deepStrictEqual((yield* store.listWork).stopRequested, []); + + // A disarm the human changed their mind about must never reach the session the + // re-arm is starting. + yield* store.requestSessionStop("thread-a", 8_000); + yield* store.arm(arm("thread-a")); + assert.deepStrictEqual((yield* store.listWork).stopRequested, []); + }), + )); +}); diff --git a/apps/server/src/coil/loop/state.ts b/apps/server/src/coil/loop/state.ts new file mode 100644 index 000000000000..c4341ddba5c8 --- /dev/null +++ b/apps/server/src/coil/loop/state.ts @@ -0,0 +1,842 @@ +/** + * Durable loop store. + * + * A single JSON file (`coil-loop.json`, in the server state dir) holding the master + * settings plus one supervision record per thread. Rehydrated on boot so an armed + * overnight run survives a server restart — T3's copy is the only durable record that a + * wake was ever armed, because the provider's own cron table is in-process. + * + * Mutations serialize through a `SynchronizedRef` and persist atomically inside the + * critical section, so memory and disk stay consistent under concurrent access from the + * tick fiber, the rate-limit fiber, the hook callbacks and the HTTP routes. + * + * Deliberately NOT a DB migration: the migration registry is upstream-owned, and adding to + * it would buy permanent conflict surface for what one JSON file does fine. Mirrors + * `coil/autoResume/state.ts`, which is proven in production. + * + * ## Every field carries a decoding default, and every default is fail-closed + * + * This is not style. A missing *required* key fails the whole-file decode, the boot path + * turns a decode failure into `EMPTY_STATE`, and that would silently disarm every loop on + * the machine. So each field is `Schema.withDecodingDefaultKey`, and each default is chosen + * so the reading you get when the field is absent is the one that *spends nothing*: + * `armed: false`, `deadlineAtMs: 0` and `maxCheckIns: 0` (both ⇒ immediately `spent` on the + * first stop sweep), `crons: null`, `pinnedByLoop: false`, `global.enabled: false`. A + * default meaning "unbounded" would turn one truncated write into an unbounded overnight + * spend, which is the single worst outcome this feature can produce. + * + * ## The file is bounded on every write + * + * One file is shared by every thread on the machine and is rewritten in full on each + * mutation, so anything that only ever grows is a cost every other thread pays. Three caps + * hold it: the ledger keeps its last 20 rows, blockers and recorded user-inputs their last + * 50 each, and `pruneThreads` drops records nothing will read again — a stopped run older + * than the retention window, and a record that is neither armed nor carrying anything. All + * three are applied inside the same critical section that persists, so memory and disk never + * disagree about what was dropped. + * + * @module coil/loop/state + */ + +import * as Clock from "effect/Clock"; +import * as Context from "effect/Context"; +import * as Effect from "effect/Effect"; +import * as FileSystem from "effect/FileSystem"; +import * as Path from "effect/Path"; +import * as Schema from "effect/Schema"; +import * as SynchronizedRef from "effect/SynchronizedRef"; + +import { writeFileStringAtomically } from "../../atomicWrite.ts"; + +/** + * Attaches the fail-closed decoding default described in the module doc. + * + * `withDecodingDefaultKey` defaults on an *absent key* (not on `undefined`), and the value + * it takes is on the Encoded side — which is why a struct whose own fields all carry + * defaults can itself default to `{}`. + */ +const withDefault = (schema: S, encodedDefault: S["Encoded"]) => + Schema.withDecodingDefaultKey(Effect.succeed(encodedDefault))(schema); + +/** + * One entry of the provider's in-process cron table, as reported by the `Stop` / + * `SubagentStop` hooks. + */ +export const CronEntry = Schema.Struct({ + id: withDefault(Schema.String, ""), + /** A 5-field cron expression, NOT a timestamp. */ + schedule: withDefault(Schema.String, ""), + recurring: withDefault(Schema.Boolean, false), + /** Truncated to 1000 chars by the binary: console display text, never the agent's prompt. */ + prompt: withDefault(Schema.String, ""), + /** Computed fork-side by `cron/parse.ts`. `null` = did not parse = NO deference. */ + nextFireAtMs: withDefault(Schema.NullOr(Schema.Number), null), +}); +export type CronEntry = typeof CronEntry.Type; + +/** + * The recorded cron snapshot. + * + * The *record* being `null` and its `entries` being `[]` are different facts and the + * trigger reads them differently: `null` = never observed, `{entries: []}` = observed and + * the agent is not self-pacing. + */ +export const CronRecord = Schema.Struct({ + recordedAtMs: withDefault(Schema.Number, 0), + entries: withDefault(Schema.Array(CronEntry), []), +}); +export type CronRecord = typeof CronRecord.Type; + +/** One row of the iteration ledger. */ +export const CheckInRow = Schema.Struct({ + n: withDefault(Schema.Number, 0), + firedAtMs: withDefault(Schema.Number, 0), + createdAtIso: withDefault(Schema.String, ""), + /** Where the thread's activity stream stood AT NUDGE TIME — the ledger is unreconstructable without it. */ + activityCursor: withDefault(Schema.String, ""), + outcome: withDefault(Schema.Literals(["productive", "unproductive", "unknown"]), "unknown"), +}); +export type CheckInRow = typeof CheckInRow.Type; + +/** Terminal state. Sticky: only a human re-arm clears it. */ +export const StopRecord = Schema.Struct({ + reason: withDefault(Schema.Literals(["done", "spent", "stalled", "handed-back"]), "spent"), + atMs: withDefault(Schema.Number, 0), + detail: withDefault(Schema.String, ""), +}); +export type StopRecord = typeof StopRecord.Type; + +/** + * A blocking question the runtime raised, recorded fork-side. + * + * Upstream settles every pending user-input as an empty answer during session teardown, so + * `hasPendingUserInput` reads false afterwards and a question nobody ever saw is + * indistinguishable from an answered one. This record is what makes `voided` visible. + */ +export const UserInputRecord = Schema.Struct({ + requestId: withDefault(Schema.String, ""), + raisedAtMs: withDefault(Schema.Number, 0), + /** `"resume_return"` for the session-resume dialog; `null` for `AskUserQuestion`. */ + dialogKind: withDefault(Schema.NullOr(Schema.String), null), + question: withDefault(Schema.String, ""), + resolution: withDefault(Schema.NullOr(Schema.Literals(["answered", "voided"])), null), + resolvedAtMs: withDefault(Schema.NullOr(Schema.Number), null), +}); +export type UserInputRecord = typeof UserInputRecord.Type; + +export const BlockerOption = Schema.Struct({ + label: withDefault(Schema.String, ""), + description: withDefault(Schema.String, ""), +}); +export type BlockerOption = typeof BlockerOption.Type; + +/** + * A deferred question raised through `raise_blocker`, which records and returns rather than + * parking the turn on a `Deferred`. + */ +export const Blocker = Schema.Struct({ + id: withDefault(Schema.String, ""), + raisedAtMs: withDefault(Schema.Number, 0), + question: withDefault(Schema.String, ""), + /** Empty = free text. */ + options: withDefault(Schema.Array(BlockerOption), []), + context: withDefault(Schema.NullOr(Schema.String), null), + answeredAtMs: withDefault(Schema.NullOr(Schema.Number), null), + answer: withDefault(Schema.NullOr(Schema.String), null), + /** Answering is asynchronous; this is what stops an answer being delivered twice or lost. */ + deliveredToAgent: withDefault(Schema.Boolean, false), +}); +export type Blocker = typeof Blocker.Type; + +export const LastCheckIn = Schema.Struct({ + firedAtMs: withDefault(Schema.Number, 0), + /** The `createdAt` we minted for the nudge, for the exact-string handback compare. */ + createdAtIso: withDefault(Schema.String, ""), +}); +export type LastCheckIn = typeof LastCheckIn.Type; + +/** Machine-wide settings. `enabled` is the master toggle (guard 2) and defaults OFF. */ +export const LoopGlobalSettings = Schema.Struct({ + enabled: withDefault(Schema.Boolean, false), + maxArmedThreads: withDefault(Schema.Number, 3), + defaultMaxCheckIns: withDefault(Schema.Number, 6), + /** Seeds the arm form only. NEVER a fallback deadline. */ + defaultRunMs: withDefault(Schema.Number, 8 * 3_600_000), + defaultIdleMs: withDefault(Schema.Number, 15 * 60_000), + defaultBusyIdleMs: withDefault(Schema.Number, 45 * 60_000), +}); +export type LoopGlobalSettings = typeof LoopGlobalSettings.Type; + +export const LoopRecord = Schema.Struct({ + /** Nothing is supervised implicitly. */ + armed: withDefault(Schema.Boolean, false), + /** Sentinel freshness baseline: a done-file older than this is a leftover, not a signal. */ + armedAtMs: withDefault(Schema.Number, 0), + goal: withDefault(Schema.NullOr(Schema.String), null), + + /** 1..20, enforced at the route with a 400. `0` ⇒ immediately `spent`. */ + maxCheckIns: withDefault(Schema.Number, 0), + checkInsUsed: withDefault(Schema.Number, 0), + /** Mandatory at arm time, never null. `0` ⇒ epoch ⇒ immediately `spent`. */ + deadlineAtMs: withDefault(Schema.Number, 0), + + idleMs: withDefault(Schema.Number, 15 * 60_000), + busyIdleMs: withDefault(Schema.Number, 45 * 60_000), + + crons: withDefault(Schema.NullOr(CronRecord), null), + degraded: withDefault(Schema.NullOr(Schema.Literals(["gate_off", "wake_lost"])), null), + userInputs: withDefault(Schema.Array(UserInputRecord), []), + + lastCheckIn: withDefault(Schema.NullOr(LastCheckIn), null), + checkIns: withDefault(Schema.Array(CheckInRow), []), + strikes: withDefault(Schema.Number, 0), + /** Durable, so a 5-hour usage limit survives a restart. */ + rateLimitedUntilMs: withDefault(Schema.Number, 0), + + /** Gates the unpin: `false` ⇒ the loop never created the pin ⇒ never remove it. */ + pinnedByLoop: withDefault(Schema.Boolean, false), + /** + * When the console asked for the provider session to be ended, or `0`. + * + * The route cannot end a session itself — `ProviderService` is not one of the services + * upstream's `makeRoutesLayer` already requires, and taking it would widen an upstream + * signature. So a disarm that leaves the agent's own wakes pending records the request + * here and the supervisor's next tick services it, which keeps `stopSession` on exactly + * one code path. + */ + stopRequestedAtMs: withDefault(Schema.Number, 0), + stopped: withDefault(Schema.NullOr(StopRecord), null), + overridePrompt: withDefault(Schema.NullOr(Schema.String), null), + blockers: withDefault(Schema.Array(Blocker), []), + + /** + * When the agent called the `loop_done` MCP tool, or `null`. + * + * The tool-shaped twin of the done-file sentinel, and read the same way: `doneSignal` + * honours it only when it is newer than `armedAtMs`, so a re-arm supersedes a previous + * run's call exactly as it supersedes a leftover `.coil/loop-done`. Nothing clears it, + * for the same reason the supervisor never deletes the file. + */ + loopDoneAtMs: withDefault(Schema.NullOr(Schema.Number), null), + /** The `reason` the agent gave, for the stop detail and the console. Display only. */ + loopDoneReason: withDefault(Schema.NullOr(Schema.String), null), +}); +export type LoopRecord = typeof LoopRecord.Type; + +export const LoopState = Schema.Struct({ + version: Schema.Literal(1), + global: withDefault(LoopGlobalSettings, {}), + threads: withDefault(Schema.Record(Schema.String, LoopRecord), {}), +}); +export type LoopState = typeof LoopState.Type; + +export const DEFAULT_GLOBAL_SETTINGS: LoopGlobalSettings = { + enabled: false, + maxArmedThreads: 3, + defaultMaxCheckIns: 6, + defaultRunMs: 8 * 3_600_000, + defaultIdleMs: 15 * 60_000, + defaultBusyIdleMs: 45 * 60_000, +}; + +/** The record a thread that has never been armed reads as. Every value is fail-closed. */ +export const EMPTY_RECORD: LoopRecord = { + armed: false, + armedAtMs: 0, + goal: null, + maxCheckIns: 0, + checkInsUsed: 0, + deadlineAtMs: 0, + idleMs: 15 * 60_000, + busyIdleMs: 45 * 60_000, + crons: null, + degraded: null, + userInputs: [], + lastCheckIn: null, + checkIns: [], + strikes: 0, + rateLimitedUntilMs: 0, + pinnedByLoop: false, + stopRequestedAtMs: 0, + stopped: null, + overridePrompt: null, + blockers: [], + loopDoneAtMs: null, + loopDoneReason: null, +}; + +const EMPTY_STATE: LoopState = { version: 1, global: DEFAULT_GLOBAL_SETTINGS, threads: {} }; + +/** + * Structural cap on the persisted ledger. + * + * The route caps `maxCheckIns` at 20 and guard 4b stops the loop at the budget, so this is + * never reached in normal operation — it bounds the file even for a hand-edited record. + */ +const LEDGER_MAX_ROWS = 20; + +/** + * Structural caps on the two append-only lists, applied on write. + * + * Both are appended by things outside a run's budget — `raise_blocker` is bounded per + * check-in window but not per run, and `user-input.requested` fires as often as the agent + * asks — so without a cap one long-lived thread grows a file every other thread on the + * machine shares. Fifty is far above what a night produces and far below what makes the file + * expensive; the oldest entries fall off, which is also the order a human stops caring about + * them in. + */ +const MAX_BLOCKERS = 50; +const MAX_USER_INPUTS = 50; + +/** + * How long a stopped record survives. + * + * The console explains last night's run from this record, so it cannot be dropped when the + * loop ends — but a month later nobody is reading it and it is pure weight. Armed records + * are never pruned at any age. + */ +const STOPPED_RETENTION_MS = 30 * 24 * 3_600_000; + +/** Anything a human or the supervisor might still want to read back. */ +const hasContent = (record: LoopRecord): boolean => + record.armed || + record.stopped !== null || + record.blockers.length > 0 || + record.userInputs.length > 0 || + record.checkIns.length > 0 || + record.crons !== null || + record.degraded !== null || + record.overridePrompt !== null || + record.loopDoneAtMs !== null || + record.rateLimitedUntilMs > 0 || + record.stopRequestedAtMs > 0; + +/** + * Drop the records nothing will ever read again, on every write. + * + * Two rules, both conservative: a stopped run older than the retention window, and a record + * that is neither armed nor carrying anything (which is what a thread that was armed and + * then cleared, or one touched by a write that turned out to be a no-op, decays to). + * `keepThreadId` is the thread the current write is about — pruning it here would make a + * write followed by a read look like the write never happened. + */ +const pruneThreads = (state: LoopState, nowMs: number, keepThreadId: string | null): LoopState => { + let dropped = false; + const threads: Record = {}; + for (const [threadId, record] of Object.entries(state.threads)) { + const expired = record.stopped !== null && nowMs - record.stopped.atMs > STOPPED_RETENTION_MS; + if (threadId !== keepThreadId && !record.armed && (expired || !hasContent(record))) { + dropped = true; + continue; + } + threads[threadId] = record; + } + return dropped ? { ...state, threads } : state; +}; + +const decodeState = Schema.decodeUnknownEffect(Schema.fromJsonString(LoopState)); + +export interface LoopStoreEntry { + readonly threadId: string; + readonly record: LoopRecord; +} + +/** Everything the arm route must supply. `deadlineAtMs` and `maxCheckIns` are mandatory. */ +export interface LoopArmInput { + readonly threadId: string; + readonly armedAtMs: number; + readonly deadlineAtMs: number; + readonly maxCheckIns: number; + readonly goal?: string | null; + /** Omitted ⇒ seeded from `global.defaultIdleMs`. */ + readonly idleMs?: number; + /** Omitted ⇒ seeded from `global.defaultBusyIdleMs`. */ + readonly busyIdleMs?: number; + readonly overridePrompt?: string | null; + /** True ONLY when the arm route itself created the pin. */ + readonly pinnedByLoop?: boolean; +} + +export interface RecordCheckInInput { + readonly threadId: string; + readonly firedAtMs: number; + readonly createdAtIso: string; + readonly activityCursor: string; +} + +export interface LoopStoreShape { + /** The machine-wide settings, including the master toggle. Re-read every tick. */ + readonly getGlobal: Effect.Effect; + /** Merge a partial settings patch; returns the settings as persisted. */ + readonly setGlobal: (patch: Partial) => Effect.Effect; + + /** Never fails and never 404s: an unknown thread reads as the fail-closed empty record. */ + readonly getThread: (threadId: string) => Effect.Effect; + /** The tick's work list, and the input to guard 14's ceiling check. */ + readonly listArmed: Effect.Effect>; + + /** + * Arm, or re-arm. One operation, because a human re-arm of a stopped loop is a *fresh + * run*: it clears `stopped`, resets `checkInsUsed` / `strikes` / the ledger, and takes a + * new `armedAtMs`. Deliberately preserved across a re-arm: `blockers` and `userInputs` + * (an answer banked before the re-arm is still owed to the agent), `crons` (the provider's + * table belongs to the session, and arming does not cancel it) and `rateLimitedUntilMs` + * (an account limit is real whether or not a loop is armed). + */ + readonly arm: (input: LoopArmInput) => Effect.Effect; + /** + * Stand a loop down without a terminal breadcrumb — guard 4's "the thread is gone" case. + * Budget, deadline and ledger stay intact so a re-arm is a deliberate act, not a repair. + */ + readonly disarm: (threadId: string) => Effect.Effect; + /** Write the sticky terminal state and disarm. Only `arm` clears it. */ + readonly stop: (threadId: string, stopped: StopRecord) => Effect.Effect; + /** + * Forget a thread entirely — the console's "clear" on a finished run. + * + * The reverse of `arm`: a stopped pill that cannot be dismissed is a one-way door. Refused + * at the route while the loop is armed, so this can never be the way a live run ends. + */ + readonly clearThread: (threadId: string) => Effect.Effect; + + /** + * Ask the supervisor to end this thread's provider session on its next tick. + * + * The console's disarm cannot call `stopSession` itself (see `stopRequestedAtMs`), so it + * records the request and the reactor services it — one code path, whichever side disarmed. + */ + readonly requestSessionStop: (threadId: string, atMs: number) => Effect.Effect; + /** + * Everything one tick needs, from **one** read of the in-memory state. + * + * Deliberately not two effects. The tick runs every `pollMs` forever, and a second read is + * a second trip through the synchronized ref — one more scheduler turn per tick, on every + * machine, for a list that is almost always empty. Reading both at once keeps a tick with + * nothing armed exactly as cheap as it was before stop requests existed. + */ + readonly listWork: Effect.Effect<{ + readonly armed: ReadonlyArray; + readonly stopRequested: ReadonlyArray; + }>; + readonly clearSessionStopRequest: (threadId: string) => Effect.Effect; + /** The escape hatch for reactor-owned bookkeeping (strikes, ledger outcomes, pins). */ + readonly update: ( + threadId: string, + f: (record: LoopRecord) => LoopRecord, + ) => Effect.Effect; + + /** + * Reserve a check-in. The reactor calls this BEFORE `engine.dispatch`, so a provider that + * cannot spawn burns budget (6 attempts) instead of tight-looping (480 a night). The + * write is persisted before this effect returns. + */ + readonly recordCheckIn: (input: RecordCheckInInput) => Effect.Effect; + /** Durable, because a usage limit outlives the process that observed it. */ + readonly setRateLimitedUntil: (threadId: string, untilMs: number) => Effect.Effect; + + /** + * Replace the cron snapshot. `null` and `{entries: []}` are different facts — the hook + * simply does not call this when `session_crons` is absent. + */ + readonly setCrons: (threadId: string, crons: CronRecord | null) => Effect.Effect; + /** Never inferred and never cleared by accident: a probe that finds nothing does not call this. */ + readonly setDegraded: ( + threadId: string, + degraded: "gate_off" | "wake_lost" | null, + ) => Effect.Effect; + readonly setOverridePrompt: ( + threadId: string, + overridePrompt: string | null, + ) => Effect.Effect; + + /** Record a `user-input.requested`. Idempotent on `requestId`. */ + readonly recordUserInput: (threadId: string, input: UserInputRecord) => Effect.Effect; + /** + * Resolve a recorded question. First resolution wins, so a teardown void cannot overwrite + * a human's answer. `"voided"` is the empty-answer teardown case. + */ + readonly resolveUserInput: ( + threadId: string, + requestId: string, + resolution: "answered" | "voided", + resolvedAtMs: number, + ) => Effect.Effect; + + readonly addBlocker: (threadId: string, blocker: Blocker) => Effect.Effect; + /** Idempotent: answering an already-answered blocker keeps the first answer. */ + readonly answerBlocker: ( + threadId: string, + blockerId: string, + answer: string, + answeredAtMs: number, + ) => Effect.Effect; + /** Unanswered blockers — what the console renders as blocking. */ + readonly listOpenBlockers: (threadId: string) => Effect.Effect>; + /** Answered but not yet told to the agent — what the next check-in prompt banks. */ + readonly listUndeliveredAnswers: (threadId: string) => Effect.Effect>; + /** + * Flip `deliveredToAgent` for exactly the listed ids. Called AFTER the prompt is composed + * and only for the blockers actually included, so an answer that landed mid-composition + * is not marked delivered and is not lost. + */ + readonly markBlockersDelivered: ( + threadId: string, + blockerIds: ReadonlyArray, + ) => Effect.Effect; +} + +export class LoopStore extends Context.Service()( + "t3/coil/loop/state/LoopStore", +) {} + +export const makeLoopStore = ( + stateFilePath: string, +): Effect.Effect => + Effect.gen(function* () { + const fs = yield* FileSystem.FileSystem; + const pathSvc = yield* Path.Path; + + // Rehydrate. Boot never fails, but the four cases must stay distinct so a *transient* + // read error on a file that still holds valid state is not mistaken for a fresh store — + // otherwise the first mutation persists an empty file over the still-valid one and + // silently disarms every loop on the machine: + // - no file (fresh install) -> empty, persistence ON + // - present + parses -> use it, persistence ON + // - present + unparseable (corrupt/vN) -> empty, persistence ON (safe to replace) + // - present + unreadable (I/O error) -> empty, persistence OFF (preserve the file) + const fileExists = yield* fs.exists(stateFilePath).pipe(Effect.orElseSucceed(() => false)); + let initial: LoopState = EMPTY_STATE; + let persistEnabled = true; + if (fileExists) { + const contents = yield* fs.readFileString(stateFilePath).pipe( + Effect.map((c): string | null => c), + Effect.orElseSucceed(() => null), + ); + if (contents === null) { + persistEnabled = false; + yield* Effect.logWarning( + "coil loop: state file present but unreadable; running in-memory only this session so the existing file is not overwritten", + { stateFilePath }, + ); + } else { + initial = yield* decodeState(contents).pipe( + Effect.tapCause((cause) => + Effect.logWarning("coil loop: state file did not decode; starting from empty", { + stateFilePath, + cause, + }), + ), + Effect.orElseSucceed(() => EMPTY_STATE), + ); + } + } + + const ref = yield* SynchronizedRef.make(initial); + + // FileSystem + Path are captured at construction and provided here so the store's public + // methods carry no context requirement (R = never). + const persist = (state: LoopState): Effect.Effect => { + if (!persistEnabled) return Effect.void; + return writeFileStringAtomically({ + filePath: stateFilePath, + contents: `${JSON.stringify(state)}\n`, + }).pipe( + Effect.provideService(FileSystem.FileSystem, fs), + Effect.provideService(Path.Path, pathSvc), + // A persistence failure must not crash the supervisor; log-and-continue keeps the + // in-memory state authoritative for this process. + Effect.catch((cause) => + Effect.logWarning("coil loop: failed to persist state", { stateFilePath, cause }), + ), + ); + }; + + /** + * Every write goes through here. Deliberately as thin as it was before pruning existed: + * one function call, one persist, and nothing that reads a clock or walks the table. + * + * The tick writes through this path several times per check-in, forever, and the reactor + * tests measure it — an extra generator frame and a clock read per write was enough to + * move the whole suite onto a cliff, where whether a scenario passed depended on how busy + * the machine was. Housekeeping does not belong on the path the product uses most. + */ + const modify = (f: (state: LoopState) => readonly [A, LoopState]) => + SynchronizedRef.modifyEffect(ref, (state) => { + const [value, next] = f(state); + return persist(next).pipe(Effect.as([value, next] as const)); + }); + + const mutate = (f: (state: LoopState) => LoopState) => + SynchronizedRef.updateEffect(ref, (state) => + Effect.suspend(() => { + const next = f(state); + return persist(next).pipe(Effect.as(next)); + }), + ); + + /** + * A write that also sweeps the table. + * + * Pruning runs where records *become* prunable — a run reaching its terminal state, a + * disarm, a clear, or a settings write — rather than on every write. Each of those sweeps + * the whole table, so a stopped record that aged out months ago is dropped by the next + * loop that ends or the next setting anyone changes; nothing accumulates, and the paths + * the supervisor runs every minute pay nothing for it. `keepThreadId` is the thread this + * write is about, exempt so that a write followed by a read cannot look like it never + * happened. + */ + const mutatePruned = (keepThreadId: string | null, f: (state: LoopState) => LoopState) => + SynchronizedRef.updateEffect(ref, (state) => + Effect.gen(function* () { + const next = f(state); + const nowMs = yield* Clock.currentTimeMillis; + const pruned = pruneThreads(next, nowMs, keepThreadId); + yield* persist(pruned); + return pruned; + }), + ); + + // `Object.hasOwn`, NOT `state.threads[threadId] ?? EMPTY_RECORD`. + // + // `threads` is a plain object, so a threadId of `constructor` / `toString` / `valueOf` / + // `__proto__` resolves on Object.prototype and is *truthy* — `??` never falls through and + // the caller gets a prototype method typed as a LoopRecord. Reads then dereference + // `record.armed` (undefined, not false) and writes spread a prototype object with no own + // enumerable properties, persisting a record missing every key. The next boot fails the + // WHOLE-file decode, collapses to EMPTY_STATE, and destroys every armed loop. The HTTP + // route takes threadId straight from the caller, so this is reachable input. + const recordFor = (state: LoopState, threadId: string): LoopRecord => + Object.hasOwn(state.threads, threadId) ? state.threads[threadId]! : EMPTY_RECORD; + + const withRecord = (state: LoopState, threadId: string, record: LoopRecord): LoopState => ({ + ...state, + threads: { ...state.threads, [threadId]: record }, + }); + + const updateRecord = (threadId: string, f: (record: LoopRecord) => LoopRecord) => + modify((state) => { + const next = f(recordFor(state, threadId)); + return [next, withRecord(state, threadId, next)] as const; + }); + + const readRecord = (threadId: string, f: (record: LoopRecord) => A) => + SynchronizedRef.get(ref).pipe(Effect.map((state) => f(recordFor(state, threadId)))); + + return { + getGlobal: SynchronizedRef.get(ref).pipe(Effect.map((state) => state.global)), + + // Pruned: a settings write is a human at the console, the cheapest possible moment to + // sweep a table that only grows between terminal states. + setGlobal: (patch) => + SynchronizedRef.modifyEffect(ref, (state) => + Effect.gen(function* () { + const global = { ...state.global, ...patch }; + const nowMs = yield* Clock.currentTimeMillis; + const pruned = pruneThreads({ ...state, global }, nowMs, null); + yield* persist(pruned); + return [global, pruned] as const; + }), + ), + + getThread: (threadId) => readRecord(threadId, (record) => record), + + listArmed: SynchronizedRef.get(ref).pipe( + Effect.map((state) => + Object.entries(state.threads).flatMap(([threadId, record]) => + record.armed ? [{ threadId, record }] : [], + ), + ), + ), + + listWork: SynchronizedRef.get(ref).pipe( + Effect.map((state) => { + const armed: Array = []; + const stopRequested: Array = []; + for (const [threadId, record] of Object.entries(state.threads)) { + if (record.armed) armed.push({ threadId, record }); + if (record.stopRequestedAtMs > 0) stopRequested.push({ threadId, record }); + } + return { armed, stopRequested }; + }), + ), + + arm: (input) => + modify((state) => { + const previous = recordFor(state, input.threadId); + const next: LoopRecord = { + ...previous, + armed: true, + armedAtMs: input.armedAtMs, + goal: input.goal ?? null, + maxCheckIns: input.maxCheckIns, + checkInsUsed: 0, + deadlineAtMs: input.deadlineAtMs, + idleMs: input.idleMs ?? state.global.defaultIdleMs, + busyIdleMs: input.busyIdleMs ?? state.global.defaultBusyIdleMs, + degraded: null, + lastCheckIn: null, + checkIns: [], + strikes: 0, + pinnedByLoop: input.pinnedByLoop ?? false, + // A stop request banked by a disarm the human then changed their mind about must + // never reach the session this arm is starting. + stopRequestedAtMs: 0, + stopped: null, + overridePrompt: input.overridePrompt ?? previous.overridePrompt, + }; + return [next, withRecord(state, input.threadId, next)] as const; + }), + + disarm: (threadId) => + mutatePruned(threadId, (state) => + withRecord(state, threadId, { ...recordFor(state, threadId), armed: false }), + ), + + stop: (threadId, stopped) => + mutatePruned(threadId, (state) => + withRecord(state, threadId, { ...recordFor(state, threadId), armed: false, stopped }), + ), + + clearThread: (threadId) => + mutatePruned(null, (state) => { + if (!Object.hasOwn(state.threads, threadId)) return state; + const { [threadId]: _removed, ...threads } = state.threads; + return { ...state, threads }; + }), + + requestSessionStop: (threadId, atMs) => + mutate((state) => + withRecord(state, threadId, { + ...recordFor(state, threadId), + stopRequestedAtMs: atMs, + }), + ), + + // `null`, not this thread: clearing the last thing a spent record carried is exactly + // when it becomes prunable. + clearSessionStopRequest: (threadId) => + mutatePruned(null, (state) => + withRecord(state, threadId, { ...recordFor(state, threadId), stopRequestedAtMs: 0 }), + ), + + update: (threadId, f) => updateRecord(threadId, f), + + recordCheckIn: (input) => + modify((state) => { + const previous = recordFor(state, input.threadId); + const n = previous.checkInsUsed + 1; + const row: CheckInRow = { + n, + firedAtMs: input.firedAtMs, + createdAtIso: input.createdAtIso, + activityCursor: input.activityCursor, + outcome: "unknown", + }; + const next: LoopRecord = { + ...previous, + checkInsUsed: n, + checkIns: [...previous.checkIns, row].slice(-LEDGER_MAX_ROWS), + lastCheckIn: { firedAtMs: input.firedAtMs, createdAtIso: input.createdAtIso }, + }; + return [row, withRecord(state, input.threadId, next)] as const; + }), + + setRateLimitedUntil: (threadId, untilMs) => + mutate((state) => + withRecord(state, threadId, { + ...recordFor(state, threadId), + rateLimitedUntilMs: untilMs, + }), + ), + + setCrons: (threadId, crons) => + mutate((state) => withRecord(state, threadId, { ...recordFor(state, threadId), crons })), + + setDegraded: (threadId, degraded) => + mutate((state) => withRecord(state, threadId, { ...recordFor(state, threadId), degraded })), + + setOverridePrompt: (threadId, overridePrompt) => + mutate((state) => + withRecord(state, threadId, { ...recordFor(state, threadId), overridePrompt }), + ), + + recordUserInput: (threadId, input) => + mutate((state) => { + const previous = recordFor(state, threadId); + if (previous.userInputs.some((entry) => entry.requestId === input.requestId)) { + return state; + } + return withRecord(state, threadId, { + ...previous, + userInputs: [...previous.userInputs, input].slice(-MAX_USER_INPUTS), + }); + }), + + resolveUserInput: (threadId, requestId, resolution, resolvedAtMs) => + mutate((state) => { + const previous = recordFor(state, threadId); + let changed = false; + const userInputs = previous.userInputs.map((entry) => { + if (entry.requestId !== requestId || entry.resolution !== null) return entry; + changed = true; + return { ...entry, resolution, resolvedAtMs }; + }); + return changed ? withRecord(state, threadId, { ...previous, userInputs }) : state; + }), + + addBlocker: (threadId, blocker) => + modify((state) => { + const previous = recordFor(state, threadId); + const existing = previous.blockers.find((entry) => entry.id === blocker.id); + if (existing) return [existing, state] as const; + return [ + blocker, + withRecord(state, threadId, { + ...previous, + blockers: [...previous.blockers, blocker].slice(-MAX_BLOCKERS), + }), + ] as const; + }), + + answerBlocker: (threadId, blockerId, answer, answeredAtMs) => + modify((state) => { + const previous = recordFor(state, threadId); + const target = previous.blockers.find((entry) => entry.id === blockerId); + if (!target) return [null, state] as const; + if (target.answeredAtMs !== null) return [target, state] as const; + const answered: Blocker = { ...target, answer, answeredAtMs, deliveredToAgent: false }; + return [ + answered, + withRecord(state, threadId, { + ...previous, + blockers: previous.blockers.map((entry) => + entry.id === blockerId ? answered : entry, + ), + }), + ] as const; + }), + + listOpenBlockers: (threadId) => + readRecord(threadId, (record) => + record.blockers.filter((entry) => entry.answeredAtMs === null), + ), + + listUndeliveredAnswers: (threadId) => + readRecord(threadId, (record) => + record.blockers.filter((entry) => entry.answeredAtMs !== null && !entry.deliveredToAgent), + ), + + markBlockersDelivered: (threadId, blockerIds) => + mutate((state) => { + const previous = recordFor(state, threadId); + const ids = new Set(blockerIds); + if (ids.size === 0) return state; + return withRecord(state, threadId, { + ...previous, + blockers: previous.blockers.map((entry) => + ids.has(entry.id) ? { ...entry, deliveredToAgent: true } : entry, + ), + }); + }), + } satisfies LoopStoreShape; + }); diff --git a/apps/server/src/coil/loop/status.test.ts b/apps/server/src/coil/loop/status.test.ts new file mode 100644 index 000000000000..09850408af9e --- /dev/null +++ b/apps/server/src/coil/loop/status.test.ts @@ -0,0 +1,200 @@ +// @effect-diagnostics globalDate:off -- `iso` is a pure ms->ISO fixture helper anchored on a +// fixed constant, never a wall-clock reading. +/** + * The shell-free status derivation `loop_status` answers from, and the property the tool + * depends on: every input produces a status, and a missing shell is skipped rather than + * reported as a state. + */ + +import { describe, expect, it } from "vite-plus/test"; + +import { DEFAULT_GLOBAL_SETTINGS, EMPTY_RECORD, type LoopRecord } from "./state.ts"; +import { deriveLoopStatus, earliestWakeMs } from "./status.ts"; +import type { LoopThreadShell } from "./types.ts"; + +const NOW = 1_800_000_000_000; +const MINUTE = 60_000; +const HOUR = 60 * MINUTE; +const iso = (ms: number) => new Date(ms).toISOString(); + +const globalSettings = (o: Partial = {}) => ({ + ...DEFAULT_GLOBAL_SETTINGS, + enabled: true, + ...o, +}); + +const record = (o: Partial = {}): LoopRecord => ({ + ...EMPTY_RECORD, + armed: true, + armedAtMs: NOW - 2 * HOUR, + maxCheckIns: 6, + checkInsUsed: 2, + deadlineAtMs: NOW + 4 * HOUR, + ...o, +}); + +const shell = (o: Partial = {}): LoopThreadShell => + ({ + updatedAt: iso(NOW - 20 * MINUTE), + archivedAt: null, + settledOverride: null, + snoozedUntil: null, + session: null, + latestTurn: null, + latestUserMessageAt: null, + hasPendingApprovals: false, + hasPendingUserInput: false, + hasActionableProposedPlan: false, + ...o, + }) as unknown as LoopThreadShell; + +const derive = (o: { + record?: LoopRecord; + global?: typeof DEFAULT_GLOBAL_SETTINGS; + shell?: LoopThreadShell | null; +}) => + deriveLoopStatus({ + nowMs: NOW, + record: o.record ?? record(), + global: o.global ?? globalSettings(), + shell: o.shell ?? null, + }); + +describe("deriveLoopStatus", () => { + it("reports the live budget and the time left", () => { + const status = derive({}); + expect(status).toMatchObject({ + armed: true, + state: "watching", + reason: null, + checkInsUsed: 2, + maxCheckIns: 6, + deadlineAtMs: NOW + 4 * HOUR, + msToDeadline: 4 * HOUR, + }); + }); + + it("clamps a passed deadline at zero rather than reporting negative time", () => { + expect(derive({ record: record({ deadlineAtMs: NOW - HOUR }) }).msToDeadline).toBe(0); + }); + + it("puts a terminal state ahead of every live reading, including the master toggle", () => { + const stopped = record({ + stopped: { reason: "spent", atMs: NOW - MINUTE, detail: "budget" }, + }); + expect(derive({ record: stopped, global: globalSettings({ enabled: false }) })).toMatchObject({ + armed: false, + state: "stopped", + reason: "spent", + }); + }); + + it("reports an unarmed thread as off with no-loop, never as a guard's opinion of it", () => { + expect(derive({ record: EMPTY_RECORD })).toMatchObject({ + armed: false, + state: "off", + reason: "no-loop", + }); + }); + + it("reads the master toggle as standing down, and as not supervised", () => { + // The record stays armed — the toggle disarms nothing — but nothing will nudge this + // thread, so the agent-facing answer to "am I being watched" is no. + expect(derive({ global: globalSettings({ enabled: false }) })).toMatchObject({ + armed: false, + state: "standing_down", + reason: "disabled", + }); + }); + + it("reports a durable rate limit as held", () => { + expect(derive({ record: record({ rateLimitedUntilMs: NOW + 30 * MINUTE }) })).toMatchObject({ + state: "held", + reason: "rate_limited", + }); + }); + + it("reports a wake still ahead of us and inside the deadline as self-pacing", () => { + const withWake = record({ + crons: { + recordedAtMs: NOW - HOUR, + entries: [ + { + id: "cron-1", + schedule: "*/30 * * * *", + recurring: true, + prompt: "keep going", + nextFireAtMs: NOW + 20 * MINUTE, + }, + ], + }, + }); + expect(derive({ record: withWake })).toMatchObject({ + state: "self_pacing", + nextWakeAtMs: NOW + 20 * MINUTE, + }); + }); + + it("leaves a wake past due to the reactor, since only it knows the grace", () => { + const late = record({ + crons: { + recordedAtMs: NOW - HOUR, + entries: [ + { + id: "cron-1", + schedule: "*/30 * * * *", + recurring: true, + prompt: "keep going", + nextFireAtMs: NOW - MINUTE, + }, + ], + }, + }); + expect(derive({ record: late }).state).toBe("watching"); + }); + + it("skips the shell clauses when there is no shell instead of guessing at them", () => { + // The MCP tool always passes null. Reading that as "thread gone" would have + // `loop_status` tell a running agent its own thread does not exist. + expect(derive({ shell: null }).state).toBe("watching"); + }); + + it("uses the shell clauses when a shell is supplied", () => { + // `held`, matching guard 6's phase and the route's `derived`: a snooze is a bounded hold + // with an expiry, not a question waiting on an answer. One fact, one word. + expect(derive({ shell: shell({ snoozedUntil: iso(NOW + HOUR) }) })).toMatchObject({ + state: "held", + reason: "snoozed", + }); + expect(derive({ shell: shell({ hasActionableProposedPlan: true }) })).toMatchObject({ + state: "blocked", + reason: "pending_plan", + }); + }); +}); + +describe("earliestWakeMs", () => { + it("ignores entries that did not parse — null means no deference, not now", () => { + const entry = (id: string, nextFireAtMs: number | null) => ({ + id, + schedule: "*/30 * * * *", + recurring: true, + prompt: "keep going", + nextFireAtMs, + }); + expect(earliestWakeMs(record({ crons: null }))).toBeNull(); + expect( + earliestWakeMs(record({ crons: { recordedAtMs: NOW, entries: [entry("a", null)] } })), + ).toBeNull(); + expect( + earliestWakeMs( + record({ + crons: { + recordedAtMs: NOW, + entries: [entry("a", NOW + HOUR), entry("b", null), entry("c", NOW + MINUTE)], + }, + }), + ), + ).toBe(NOW + MINUTE); + }); +}); diff --git a/apps/server/src/coil/loop/status.ts b/apps/server/src/coil/loop/status.ts new file mode 100644 index 000000000000..5831073a1dcf --- /dev/null +++ b/apps/server/src/coil/loop/status.ts @@ -0,0 +1,131 @@ +/** + * The loop's own budget readout, derived from the durable record alone. + * + * This is the shell-free half of `http.ts`'s `deriveView`: the subset of the state machine + * that can be stated truthfully from `{ record, global, nowMs }` plus, optionally, a thread + * shell. It exists because `loop_status` is answered inside an MCP tool call, which has a + * `LoopStore` and nothing else — no projection query, no thread shell, no tick. + * + * Two properties hold and are what the tool depends on: + * + * 1. **It never fails.** Every branch returns a status. An agent asking how much budget it + * has left must never be handed an error it then has to reason about mid-turn. + * 2. **A missing shell is not a state.** When `shell` is `null` the shell-derived clauses + * (snooze, blocked-on-a-human) are simply skipped rather than reported as `off` or + * `blocked`. `http.ts` distinguishes "the thread is gone" with its own `threadKnown`; + * conflating the two here would have `loop_status` tell a running agent its thread does + * not exist. + * + * The state literals are `LoopDerivedView["state"]`'s, deliberately, so the console and the + * tool cannot drift into two vocabularies for one machine. This module is the intended home + * of that derivation — `deriveView` can adopt it by passing its shell through. + * + * @module coil/loop/status + */ + +import { blockingRequest, isoMs, isRateLimited, isSnoozed } from "./guards.ts"; +import type { LoopGlobalSettings, LoopRecord } from "./state.ts"; +import type { LoopThreadShell } from "./types.ts"; + +/** The same seven literals `http.ts` renders, so one machine has one vocabulary. */ +export type LoopStatusState = + | "off" + | "watching" + | "self_pacing" + | "standing_down" + | "held" + | "blocked" + | "stopped"; + +export interface LoopStatusInput { + readonly nowMs: number; + readonly record: LoopRecord; + readonly global: LoopGlobalSettings; + /** `null` = no shell facts to hand. The shell clauses are skipped, never guessed. */ + readonly shell: LoopThreadShell | null; +} + +export interface LoopStatus { + /** + * Supervised right now — armed, not stopped, and the master toggle on. + * + * Deliberately *not* `record.armed`. An agent asking whether it is being watched wants + * the operational answer, and a loop standing down behind a switched-off master toggle + * is not being watched. The console reads `record.armed` for the raw fact. + */ + readonly armed: boolean; + readonly state: LoopStatusState; + /** Why, when the state alone does not say. `null` for `watching`. */ + readonly reason: string | null; + readonly checkInsUsed: number; + readonly maxCheckIns: number; + readonly deadlineAtMs: number; + /** Clamped at 0: a passed deadline reads as "no time left", never as negative time. */ + readonly msToDeadline: number; + /** Earliest recorded wake that parsed, past or future. `null` = no deference available. */ + readonly nextWakeAtMs: number | null; +} + +/** Earliest recorded wake that parsed. `null` entries mean "no deference", not "now". */ +export function earliestWakeMs(record: LoopRecord): number | null { + let earliest: number | null = null; + for (const entry of record.crons?.entries ?? []) { + const next = entry.nextFireAtMs; + if (next === null) continue; + if (earliest === null || next < earliest) earliest = next; + } + return earliest; +} + +/** + * The record's reading of the state machine, in the guard table's order. + * + * Terminal first — a stop is sticky and outranks every live reading, including the master + * toggle — then `off`, so an unarmed thread never reports a guard's opinion of it. + */ +export function deriveLoopStatus(input: LoopStatusInput): LoopStatus { + const { nowMs, record, global, shell } = input; + const nextWakeAtMs = earliestWakeMs(record); + const base = { + checkInsUsed: record.checkInsUsed, + maxCheckIns: record.maxCheckIns, + deadlineAtMs: record.deadlineAtMs, + msToDeadline: Math.max(0, record.deadlineAtMs - nowMs), + nextWakeAtMs, + }; + + if (record.stopped !== null) { + return { ...base, armed: false, state: "stopped", reason: record.stopped.reason }; + } + if (!record.armed) return { ...base, armed: false, state: "off", reason: "no-loop" }; + // Guard 2: the toggle stands loops down; it disarms and stops nothing, so `armed` stays + // true. The tool reports the toggle separately — this is a state, not a disarm. + if (!global.enabled) { + return { ...base, armed: false, state: "standing_down", reason: "disabled" }; + } + // `held`, matching guard 6's phase: a snooze is a bounded hold with an expiry, not a + // question waiting on an answer. One fact, one word, whichever lens asks. + if (shell !== null && isSnoozed(shell, nowMs)) { + return { + ...base, + armed: true, + state: "held", + reason: "snoozed", + nextWakeAtMs: isoMs(shell.snoozedUntil) ?? nextWakeAtMs, + }; + } + if (shell !== null) { + const blocked = blockingRequest(shell); + if (blocked !== null) return { ...base, armed: true, state: "blocked", reason: blocked }; + } + if (isRateLimited(record, nowMs)) { + return { ...base, armed: true, state: "held", reason: "rate_limited" }; + } + // Guard 10b, conservatively: only a wake still ahead of us and inside the run's deadline + // is visible deference. A wake past due is the reactor's call, since whether it is merely + // late or genuinely lost depends on the derived grace. + if (nextWakeAtMs !== null && nextWakeAtMs > nowMs && nextWakeAtMs <= record.deadlineAtMs) { + return { ...base, armed: true, state: "self_pacing", reason: null }; + } + return { ...base, armed: true, state: "watching", reason: null }; +} diff --git a/apps/server/src/coil/loop/types.ts b/apps/server/src/coil/loop/types.ts new file mode 100644 index 000000000000..e3659c221837 --- /dev/null +++ b/apps/server/src/coil/loop/types.ts @@ -0,0 +1,222 @@ +/** + * The pure input and output shapes the loop decision table is written against. + * + * Lives apart from `decide.ts` and `guards.ts` only so those two can import each other's + * vocabulary without a cycle: `decide.ts` computes the trigger facts, `guards.ts` consumes + * them, and both speak this file. Nothing here has a runtime body — a type-only module + * costs no branches and no bundle. + * + * Everything the decision reads is a plain value: a number for the clock, the durable + * record, a thread shell, and a handful of facts the reactor already has in hand. There is + * no `Effect`, no store and no filesystem, which is what makes the whole table testable + * without a server, a clock or a provider. + * + * @module coil/loop/types + */ + +import type { OrchestrationThreadShell } from "@t3tools/contracts"; + +import type { LoopConfig } from "./config.ts"; +import type { CheckInRow, LoopGlobalSettings, LoopRecord, StopRecord } from "./state.ts"; + +/** + * Exactly the shell fields the decision reads. + * + * A real `OrchestrationThreadShell` is structurally assignable, so the reactor passes its + * projection row straight through; a test builds the ten fields and nothing else. Narrowing + * it here is also the enforcement of the two refusals in the design: `deletedAt` is not on + * the shell at all (a deleted thread is an absent shell, i.e. `null`), and `pinnedAt` is + * absent because the pin is an arm-time fact recorded on the record, never re-derived here. + */ +export type LoopThreadShell = Pick< + OrchestrationThreadShell, + | "updatedAt" + | "archivedAt" + | "settledOverride" + | "snoozedUntil" + | "session" + | "latestTurn" + | "latestUserMessageAt" + | "hasPendingApprovals" + | "hasPendingUserInput" + | "hasActionableProposedPlan" + | "backgroundLiveness" +>; + +/** The user-visible state a decision puts the loop in. Derived, never stored. */ +export type LoopPhase = "off" | "watching" | "self_pacing" | "standing_down" | "held" | "blocked"; + +/** + * Why a tick did nothing. Every one of these is non-consuming: the budget is untouched and + * the loop stays armed, which is what distinguishes a stand-down from a stop. + */ +export type StandDownReason = + | "disabled" + | "stopped" + | "not_armed" + | "snoozed" + | "pending_approval" + | "pending_user_input" + | "pending_plan" + | "auto_resume_pending" + | "rate_limited" + | "self_pacing" + | "check_in_floor" + | "not_idle" + | "ceiling"; + +/** The thread is no longer a destination a check-in could reach. */ +export type DisarmReason = "thread_gone" | "archived"; + +/** The four terminal reasons, and they are `StopRecord`'s literals rather than a second set. */ +export type StopOutcome = StopRecord["reason"]; + +/** + * Which fact ended the run. + * + * Separate from `StopOutcome` because the record only stores four reasons while the console + * and the breadcrumb want to say *why*: a deadline and an exhausted budget both report + * `spent`, and that is deliberate — `spent` is never rendered as success. + */ +export type StopCause = "deadline" | "budget" | "sentinel" | "loop_done" | "strikes" | "takeover"; + +/** Retired numbers (5, 13) are not reused, so a case number cited anywhere keeps meaning. */ +export type GuardId = "2" | "3" | "4" | "4b" | "6" | "8" | "9" | "10" | "10b" | "11" | "12" | "14"; + +/** How the previous check-in is judged, in `CheckInRow`'s own vocabulary. */ +export type CheckInOutcome = CheckInRow["outcome"]; + +/** A recorded wake, resolved against the record and the shell. */ +export interface ResolvedWake { + readonly cronId: string; + readonly atMs: number; + /** + * How late this wake may land before it counts as lost — derived from the entry, never a + * constant. See `wakeGraceMs`. + */ + readonly graceMs: number; + /** + * At or before the run's deadline. Past the deadline there is nothing left to defer to, + * so an unbounded `CronCreate` expression cannot stand supervision down for a day. + */ + readonly deferrable: boolean; + /** `updatedAt` moved past the wake, so it landed and there is nothing to cover. */ + readonly landed: boolean; +} + +/** The §4 arithmetic, computed once per decision and read by guards 10b, 11 and 12. */ +export interface TriggerFacts { + /** Clamped at 0 and floored by `processStartedAtMs`; never `NaN`, never negative. */ + readonly idleForMs: number; + /** `busyIdleMs` or `idleMs`, chosen by `busyTurn`. */ + readonly thresholdMs: number; + /** Lengthens the fuse. It is never a veto — see `resolveTrigger`. */ + readonly busyTurn: boolean; + readonly wake: ResolvedWake | null; +} + +export interface LoopDecisionInput { + readonly nowMs: number; + /** Boot-grace floor. Without it every armed thread fires at once on the first tick. */ + readonly processStartedAtMs: number; + readonly record: LoopRecord; + readonly global: LoopGlobalSettings; + /** `null` is `Option.none` from `getThreadShellById`: the thread is gone. */ + readonly shell: LoopThreadShell | null; + /** Newest `.coil/loop-done` mtime across both roots, or `null`. Never file contents. */ + readonly sentinelAtMs: number | null; + /** When the agent called `loop_done`, or `null`. Equivalent to the file. */ + readonly loopDoneAtMs: number | null; + /** Auto-resume has a pending resume armed for this thread (guard 9). */ + readonly autoResumePending: boolean; + /** + * Armed loops machine-wide, **including this one**. Guard 14 measures the others against + * the ceiling, so the count it is handed must include the loop under evaluation. + */ + readonly armedCount: number; + readonly config: LoopConfig; +} + +export interface LoopGuardInput extends LoopDecisionInput { + readonly trigger: TriggerFacts; +} + +export interface GuardFire { + readonly kind: "fire"; + /** The pre-dispatch shell showed `settledOverride: "active"` — repair it after the turn. */ + readonly repairPin: boolean; + /** A recorded wake went past its grace with no activity. */ + readonly degrade: "wake_lost" | null; +} + +export interface GuardStandDown { + readonly kind: "stand_down"; + readonly guard: GuardId; + readonly reason: StandDownReason; + readonly phase: LoopPhase; + /** When the reason expires, for the console and the `{ reason, until? }` breadcrumb. */ + readonly untilMs: number | null; +} + +export interface GuardDisarm { + readonly kind: "disarm"; + readonly guard: "4"; + readonly reason: DisarmReason; +} + +export interface GuardStop { + readonly kind: "stop"; + readonly guard: "4b"; + readonly outcome: StopOutcome; + readonly cause: StopCause; + readonly detail: string; +} + +export type GuardOutcome = GuardFire | GuardStandDown | GuardDisarm | GuardStop; + +/** What the reactor reserves before it dispatches. */ +export interface LoopCheckIn { + /** 1-based. */ + readonly n: number; + readonly of: number; + readonly firedAtMs: number; + /** The strike count to persist with this check-in: 0 on movement, +1 without. */ + readonly strikes: number; + /** How the *previous* check-in turned out; `unknown` when there was none. */ + readonly previousOutcome: CheckInOutcome; +} + +export interface StandDownAction { + readonly type: "stand_down"; + readonly guard: GuardId; + readonly reason: StandDownReason; + readonly phase: LoopPhase; + readonly untilMs: number | null; +} + +export interface FireAction { + readonly type: "fire"; + readonly kind: "check_in" | "wake_lost"; + readonly repairPin: boolean; + readonly degrade: "wake_lost" | null; + readonly checkIn: LoopCheckIn; +} + +export interface StopAction { + readonly type: "stop"; + readonly outcome: StopOutcome; + readonly cause: StopCause; + readonly detail: string; + /** + * End the provider session as well, because T3 has no write handle on the binary's cron + * table and a bound that cannot stop the agent is not a bound. + */ + readonly stopSession: boolean; +} + +export interface DisarmAction { + readonly type: "disarm"; + readonly reason: DisarmReason; +} + +export type LoopAction = StandDownAction | FireAction | StopAction | DisarmAction; diff --git a/apps/server/src/coil/loop/userInputs.test.ts b/apps/server/src/coil/loop/userInputs.test.ts new file mode 100644 index 000000000000..1462de048145 --- /dev/null +++ b/apps/server/src/coil/loop/userInputs.test.ts @@ -0,0 +1,285 @@ +// @effect-diagnostics nodeBuiltinImport:off +import * as NodePath from "node:path"; + +import type { ProviderRuntimeEvent } from "@t3tools/contracts"; +import * as NodeServices from "@effect/platform-node/NodeServices"; +import { assert, describe, it } from "@effect/vitest"; +import * as Effect from "effect/Effect"; +import * as FileSystem from "effect/FileSystem"; +import * as Stream from "effect/Stream"; + +import type { ProviderServiceShape } from "../../provider/Services/ProviderService.ts"; +import { type LoopStoreShape, makeLoopStore } from "./state.ts"; +import { recordUserInputs } from "./userInputs.ts"; + +const THREAD_ID = "thread-loop-1"; + +/** + * The fields the recorder reads. Cast because building every branded field of a runtime + * event is noise for a bookkeeping tap. + */ +const requested = (o: { + requestId?: string; + question?: string; + threadId?: string; +}): ProviderRuntimeEvent => + ({ + type: "user-input.requested", + eventId: "evt-req", + provider: "claudeAgent", + threadId: o.threadId ?? THREAD_ID, + createdAt: "2026-09-02T01:04:00.000Z", + ...(o.requestId === undefined ? {} : { requestId: o.requestId }), + payload: { + questions: [ + { + id: o.question ?? "Which migration path?", + header: "Decision", + question: o.question ?? "Which migration path?", + options: [], + multiSelect: false, + }, + ], + }, + }) as unknown as ProviderRuntimeEvent; + +const resolved = (o: { + requestId: string; + answers: Record; + threadId?: string; +}): ProviderRuntimeEvent => + ({ + type: "user-input.resolved", + eventId: "evt-res", + provider: "claudeAgent", + threadId: o.threadId ?? THREAD_ID, + createdAt: "2026-09-02T01:09:00.000Z", + requestId: o.requestId, + payload: { answers: o.answers }, + }) as unknown as ProviderRuntimeEvent; + +/** An unrelated event that must be ignored without a store write. */ +const unrelated: ProviderRuntimeEvent = { + type: "turn.started", + eventId: "evt-turn", + provider: "claudeAgent", + threadId: THREAD_ID, + createdAt: "2026-09-02T01:00:00.000Z", +} as unknown as ProviderRuntimeEvent; + +/** + * A finite stub stream, so `recordUserInputs` completes and the assertions run after every + * event has been consumed — no sleeps, no polling. + */ +const providerStub = (events: ReadonlyArray): ProviderServiceShape => + ({ streamEvents: Stream.fromIterable(events) }) as unknown as ProviderServiceShape; + +const withStore = (f: (store: LoopStoreShape) => Effect.Effect) => + Effect.gen(function* () { + const fs = yield* FileSystem.FileSystem; + const root = yield* fs.makeTempDirectoryScoped({ prefix: "coil-loop-inputs-" }); + const store = yield* makeLoopStore(NodePath.join(root, "coil-loop.json")); + return yield* f(store); + }).pipe(Effect.scoped, Effect.orDie, Effect.provide(NodeServices.layer), Effect.runPromise); + +const armed = (store: LoopStoreShape) => + store.arm({ threadId: THREAD_ID, armedAtMs: 1_000, deadlineAtMs: 9_000_000, maxCheckIns: 6 }); + +describe("coil loop user-input recording", () => { + it("118b: a user-input.requested on an armed thread is recorded", () => + withStore((store) => + Effect.gen(function* () { + yield* armed(store); + yield* recordUserInputs( + store, + providerStub([unrelated, requested({ requestId: "req-1" })]), + ); + + const inputs = (yield* store.getThread(THREAD_ID)).userInputs; + assert.strictEqual(inputs.length, 1); + assert.strictEqual(inputs[0]!.requestId, "req-1"); + assert.strictEqual(inputs[0]!.question, "Which migration path?"); + assert.strictEqual(inputs[0]!.resolution, null); + assert.strictEqual(inputs[0]!.resolvedAtMs, null); + assert.isAbove(inputs[0]!.raisedAtMs, 0); + assert.strictEqual(inputs[0]!.dialogKind, null); + }), + )); + + it("118c: an empty resolution is recorded as voided, not answered", () => + withStore((store) => + Effect.gen(function* () { + yield* armed(store); + yield* recordUserInputs( + store, + providerStub([ + requested({ requestId: "req-1" }), + // Upstream #5127: teardown settles every pending input with `{}`. + resolved({ requestId: "req-1", answers: {} }), + ]), + ); + + const input = (yield* store.getThread(THREAD_ID)).userInputs[0]!; + assert.strictEqual(input.resolution, "voided"); + assert.isNotNull(input.resolvedAtMs); + }), + )); + + it("118d: a genuine human answer is recorded as answered", () => + withStore((store) => + Effect.gen(function* () { + yield* armed(store); + yield* recordUserInputs( + store, + providerStub([ + requested({ requestId: "req-1" }), + resolved({ + requestId: "req-1", + answers: { "Which migration path?": "Take the migration" }, + }), + ]), + ); + + assert.strictEqual( + (yield* store.getThread(THREAD_ID)).userInputs[0]!.resolution, + "answered", + ); + }), + )); + + it("118d: the first resolution wins, so a teardown void cannot overwrite an answer", () => + withStore((store) => + Effect.gen(function* () { + yield* armed(store); + yield* recordUserInputs( + store, + providerStub([ + requested({ requestId: "req-1" }), + resolved({ requestId: "req-1", answers: { q: "yes" } }), + resolved({ requestId: "req-1", answers: {} }), + ]), + ); + + assert.strictEqual( + (yield* store.getThread(THREAD_ID)).userInputs[0]!.resolution, + "answered", + ); + }), + )); + + it("118h: the resume-return dialog is recorded with its dialog kind", () => + withStore((store) => + Effect.gen(function* () { + yield* armed(store); + yield* recordUserInputs( + store, + providerStub([ + requested({ + requestId: "req-1", + // The exact copy upstream's resume dialog asks, recognised through the shared + // predicate the web client already uses. + question: + "This session is 2h 5m old and uses 132,000 tokens. Compact it before continuing?", + }), + ]), + ); + + assert.strictEqual( + (yield* store.getThread(THREAD_ID)).userInputs[0]!.dialogKind, + "resume_return", + ); + }), + )); + + it("recording is idempotent on requestId", () => + withStore((store) => + Effect.gen(function* () { + yield* armed(store); + yield* recordUserInputs( + store, + providerStub([requested({ requestId: "req-1" }), requested({ requestId: "req-1" })]), + ); + + assert.strictEqual((yield* store.getThread(THREAD_ID)).userInputs.length, 1); + }), + )); + + it("an unarmed thread accrues nothing, so the shared file cannot grow without a reader", () => + withStore((store) => + Effect.gen(function* () { + yield* recordUserInputs(store, providerStub([requested({ requestId: "req-1" })])); + + assert.deepStrictEqual((yield* store.getThread(THREAD_ID)).userInputs, []); + }), + )); + + it("a question raised under supervision still resolves after the loop stands down", () => + withStore((store) => + Effect.gen(function* () { + yield* armed(store); + yield* recordUserInputs(store, providerStub([requested({ requestId: "req-1" })])); + yield* store.disarm(THREAD_ID); + + yield* recordUserInputs( + store, + providerStub([resolved({ requestId: "req-1", answers: {} })]), + ); + + assert.strictEqual((yield* store.getThread(THREAD_ID)).userInputs[0]!.resolution, "voided"); + }), + )); + + it("a resolution for an unrecorded request creates nothing", () => + withStore((store) => + Effect.gen(function* () { + yield* armed(store); + yield* recordUserInputs( + store, + providerStub([resolved({ requestId: "req-unknown", answers: {} })]), + ); + + assert.deepStrictEqual((yield* store.getThread(THREAD_ID)).userInputs, []); + }), + )); + + it("an unkeyed request is skipped rather than recorded as permanently pending", () => + withStore((store) => + Effect.gen(function* () { + yield* armed(store); + yield* recordUserInputs(store, providerStub([requested({})])); + + assert.deepStrictEqual((yield* store.getThread(THREAD_ID)).userInputs, []); + }), + )); + + it("a bookkeeping failure on one event does not tear down the subscription", () => + withStore((store) => + Effect.gen(function* () { + yield* armed(store); + let firstCall = true; + const flaky: LoopStoreShape = { + ...store, + recordUserInput: (threadId, input) => { + if (firstCall) { + firstCall = false; + return Effect.sync(() => { + throw new Error("simulated store defect"); + }); + } + return store.recordUserInput(threadId, input); + }, + }; + + yield* recordUserInputs( + flaky, + providerStub([requested({ requestId: "req-1" }), requested({ requestId: "req-2" })]), + ); + + const inputs = (yield* store.getThread(THREAD_ID)).userInputs; + assert.deepStrictEqual( + inputs.map((entry) => entry.requestId), + ["req-2"], + ); + }), + )); +}); diff --git a/apps/server/src/coil/loop/userInputs.ts b/apps/server/src/coil/loop/userInputs.ts new file mode 100644 index 000000000000..bb250797ddc5 --- /dev/null +++ b/apps/server/src/coil/loop/userInputs.ts @@ -0,0 +1,131 @@ +/** + * Fork-side record of the questions the runtime raised, and how they really ended. + * + * Upstream #5127 made session teardown settle every pending user-input as an **empty + * answer** so the thread can settle. The tool call is denied, the session tears down, and + * `hasPendingUserInput` reads false afterwards — so a question nobody ever saw is + * indistinguishable from an answered one. The console cannot derive its blocking list from + * the projection alone; this is the record that makes `voided` visible. + * + * It also carries the **dialog kind**. Upstream #8144 added a second blocking dialog + * (`resume_return`) that routes through the same `AskUserQuestion` path and fires on session + * *resume* — so a check-in landing on a torn-down session can park the loop on a dialog the + * loop itself caused. With the kind recorded the console can say "waiting on a session-resume + * confirmation since 01:04" instead of showing an unexplained idle loop. The runtime event + * does not name the kind, so it is recognised through + * `isClaudeResumeCompactionQuestion` — the same predicate the web client already uses to + * recognise this dialog, rather than a fork-defined copy of the copy. + * + * ## Only armed threads accrue records + * + * `recordUserInput` appends, `coil-loop.json` is rewritten atomically on every mutation, and + * `AskUserQuestion` fires on threads that will never be supervised. Recording every question + * on the machine would grow one shared file without bound for no reader. Requests are + * therefore recorded only while the thread is armed. Resolutions are applied unconditionally + * — `resolveUserInput` is a no-op when nothing matches, so it cannot create a record — which + * is what keeps a question raised under supervision resolvable after the loop stands down. + * + * Exported as a plain `Effect` rather than a layer: the loop reactor forks it into its own + * fiber set beside the tick, the same way `autoResume/Reactor.ts` forks its detection tap. + * + * @module coil/loop/userInputs + */ + +import type { ProviderRuntimeEvent } from "@t3tools/contracts"; +import { isClaudeResumeCompactionQuestion } from "@t3tools/shared/claudeCompaction"; +import * as Cause from "effect/Cause"; +import * as Clock from "effect/Clock"; +import * as Effect from "effect/Effect"; +import * as Stream from "effect/Stream"; + +import type { ProviderServiceShape } from "../../provider/Services/ProviderService.ts"; +import type { LoopReceiptEmitter } from "./receipts.ts"; +import type { LoopStoreShape } from "./state.ts"; + +/** The upstream #8144 dialog, recorded as itself so the console can name it. */ +const RESUME_DIALOG_KIND = "resume_return"; + +/** No listener. The reactor always passes its own emitter; unit callers do not care. */ +const SILENT: LoopReceiptEmitter = { enabled: false, emit: () => Effect.void }; + +const onRequested = ( + store: LoopStoreShape, + event: Extract, + receipts: LoopReceiptEmitter, +): Effect.Effect => + Effect.gen(function* () { + const requestId = event.requestId; + // Unkeyed requests could never be resolved, so recording one would leave a question + // that reads as pending forever. + if (requestId === undefined) return; + const record = yield* store.getThread(event.threadId); + if (!record.armed) return; + + const questions = event.payload.questions; + // The stream is hot, so "now" is the moment it was raised to within a tick — and it + // avoids re-deriving an instant from the event's ISO string. + const raisedAtMs = yield* Clock.currentTimeMillis; + yield* store.recordUserInput(event.threadId, { + requestId, + raisedAtMs, + dialogKind: questions.some((question) => isClaudeResumeCompactionQuestion(question.question)) + ? RESUME_DIALOG_KIND + : null, + // Header text for the console. A multi-question ask is rare and the first question is + // what the dialog leads with. + question: questions.find((question) => question.question.trim().length > 0)?.question ?? "", + resolution: null, + resolvedAtMs: null, + }); + yield* receipts.emit({ type: "userInput.recorded", threadId: event.threadId, requestId }); + }); + +const onResolved = ( + store: LoopStoreShape, + event: Extract, +): Effect.Effect => + Effect.gen(function* () { + const requestId = event.requestId; + if (requestId === undefined) return; + // The whole point of the record: teardown and an interrupted turn both settle with `{}`, + // and neither is a human decision. Anything non-empty came from a person. + const answered = Object.keys(event.payload.answers).length > 0; + const resolvedAtMs = yield* Clock.currentTimeMillis; + yield* store.resolveUserInput( + event.threadId, + requestId, + answered ? "answered" : "voided", + resolvedAtMs, + ); + }); + +/** Exported for the reactor's tests; the stream tap below is the only production caller. */ +export const recordUserInputEvent = ( + store: LoopStoreShape, + event: ProviderRuntimeEvent, + receipts: LoopReceiptEmitter = SILENT, +): Effect.Effect => { + if (event.type === "user-input.requested") return onRequested(store, event, receipts); + if (event.type === "user-input.resolved") return onResolved(store, event); + return Effect.void; +}; + +/** + * Subscribes for the life of the fiber it is forked into. Never fails: a bookkeeping error + * on one event must not tear down the subscription and silently retire the record. + */ +export const recordUserInputs = ( + store: LoopStoreShape, + providerService: Pick, + receipts: LoopReceiptEmitter = SILENT, +): Effect.Effect => + Stream.runForEach(providerService.streamEvents, (event) => + recordUserInputEvent(store, event, receipts).pipe( + Effect.catchCause((cause) => + Effect.logDebug("coil loop: user-input recording failed", { + eventType: event.type, + cause: Cause.pretty(cause), + }), + ), + ), + ); diff --git a/apps/server/src/mcp/McpHttpServer.ts b/apps/server/src/mcp/McpHttpServer.ts index 87975a49de2c..4b0b36a64801 100644 --- a/apps/server/src/mcp/McpHttpServer.ts +++ b/apps/server/src/mcp/McpHttpServer.ts @@ -13,6 +13,7 @@ import packageJson from "../../package.json" with { type: "json" }; import * as McpInvocationContext from "./McpInvocationContext.ts"; import * as McpSessionRegistry from "./McpSessionRegistry.ts"; import * as PreviewAutomationBroker from "./PreviewAutomationBroker.ts"; +import { LoopToolkitRegistrationLive } from "./toolkits/loop/handlers.ts"; import { PreviewSnapshotToolkitHandlersLive, PreviewStandardToolkitHandlersLive, @@ -223,4 +224,7 @@ const McpTransportLive = McpServer.layerHttp({ protocols: [McpProtocol.v2025_06_18], }).pipe(Layer.provide(McpAuthMiddlewareLive)); -export const layer = PreviewToolkitRegistrationLive.pipe(Layer.provideMerge(McpTransportLive)); +export const layer = Layer.mergeAll( + PreviewToolkitRegistrationLive, + LoopToolkitRegistrationLive, // coil fork seam — fork-owned MCP toolkits register here +).pipe(Layer.provideMerge(McpTransportLive)); diff --git a/apps/server/src/mcp/toolkits/loop/handlers.test.ts b/apps/server/src/mcp/toolkits/loop/handlers.test.ts new file mode 100644 index 000000000000..b35296041bf1 --- /dev/null +++ b/apps/server/src/mcp/toolkits/loop/handlers.test.ts @@ -0,0 +1,619 @@ +// @effect-diagnostics nodeBuiltinImport:off +// @effect-diagnostics globalDate:off -- fixture timestamps anchored on a fixed constant, never +// a wall-clock reading. +/** + * TESTS.md §7, cases 108–118h: the question channel. + * + * The tools are driven through a real `McpServer` rather than by calling the handler + * functions, because half of what is under test is the registration and the attribution: a + * handler invoked directly would be handed an `McpInvocationContext` by the test, which is + * exactly the thing that must come from the credential instead. + */ + +import { EnvironmentId, ProviderInstanceId, ThreadId, UserInputQuestion } from "@t3tools/contracts"; +import * as NodeServices from "@effect/platform-node/NodeServices"; +import { assert, describe, expect, it } from "@effect/vitest"; +import * as Clock from "effect/Clock"; +import * as Deferred from "effect/Deferred"; +import * as Effect from "effect/Effect"; +import * as FileSystem from "effect/FileSystem"; +import * as Layer from "effect/Layer"; +import * as Schema from "effect/Schema"; +import * as TestClock from "effect/testing/TestClock"; +import { McpSchema, McpServer } from "effect/unstable/ai"; +import * as NodePath from "node:path"; + +import { resolveConfig } from "../../../coil/loop/config.ts"; +import { doneSignal, evaluateGuards, stopCondition } from "../../../coil/loop/guards.ts"; +import { + DEFAULT_GLOBAL_SETTINGS, + EMPTY_RECORD, + LoopStore, + type LoopStoreShape, + makeLoopStore, + type UserInputRecord, +} from "../../../coil/loop/state.ts"; +import type { LoopGuardInput, LoopThreadShell } from "../../../coil/loop/types.ts"; +import * as McpInvocationContext from "../../McpInvocationContext.ts"; +import { LoopToolkitHandlersLive, MAX_OPEN_BLOCKERS_PER_WINDOW } from "./handlers.ts"; +import { LoopToolkit } from "./tools.ts"; + +const decodeUserInputQuestion = Schema.decodeUnknownEffect(UserInputQuestion); + +const NOW = 1_800_000_000_000; // 2027-01-15T08:00:00Z +const MINUTE = 60_000; +const HOUR = 60 * MINUTE; +const THREAD = "thread-loop-mcp"; +const OTHER_THREAD = "thread-someone-else"; + +const invocation = (threadId: string) => ({ + environmentId: EnvironmentId.make("environment-loop-test"), + threadId: ThreadId.make(threadId), + providerSessionId: "provider-session-loop-test", + providerInstanceId: ProviderInstanceId.make("claudeAgent"), + // The loop tools are ungated by design: nothing consults `capabilities`, and the real + // gate is `global.enabled` plus the armed record. An empty set proves that. + capabilities: new Set(), + issuedAt: 1, +}); + +const client = McpSchema.McpServerClient.of({ + clientId: 1, + protocolVersion: "2025-06-18", + initializePayload: { + protocolVersion: "2025-06-18", + capabilities: {}, + clientInfo: { name: "loop-toolkit-test", version: "1.0.0" }, + }, + getClient: Effect.die("unused"), +}); + +/** Registers the toolkit against a store the test controls, on one shared `McpServer`. */ +const toolkitLayer = (store: LoopStoreShape) => + McpServer.toolkit(LoopToolkit).pipe( + Layer.provide(LoopToolkitHandlersLive), + Layer.provide(Layer.succeed(LoopStore, store)), + Layer.provideMerge(McpServer.McpServer.layer), + ); + +/** + * A tool call as the transport makes it: arguments in, invocation context provided + * separately by the authenticated middleware. + */ +const callTool = (name: string, args: Record, threadId = THREAD) => + McpServer.McpServer.pipe( + Effect.flatMap((server) => server.callTool({ name, arguments: args })), + Effect.provideService(McpInvocationContext.McpInvocationContext, invocation(threadId)), + Effect.provideService(McpSchema.McpServerClient, client), + ); + +const structured = (result: McpSchema.CallToolResult) => + result.structuredContent as Record; + +const armed = (store: LoopStoreShape, threadId = THREAD) => + store.arm({ + threadId, + armedAtMs: NOW - 2 * HOUR, + deadlineAtMs: NOW + 4 * HOUR, + maxCheckIns: 6, + }); + +/** The default world: loops on machine-wide, one armed thread, virtual clock at NOW. */ +const withTools = ( + body: (store: LoopStoreShape) => Effect.Effect, + seed: (store: LoopStoreShape) => Effect.Effect = (store) => + store.setGlobal({ enabled: true }).pipe(Effect.andThen(armed(store))), +) => + Effect.gen(function* () { + const fs = yield* FileSystem.FileSystem; + const root = yield* fs.makeTempDirectoryScoped({ prefix: "coil-loop-mcp-" }); + const store = yield* makeLoopStore(NodePath.join(root, "coil-loop.json")); + yield* seed(store); + yield* TestClock.setTime(NOW); + return yield* body(store).pipe(Effect.provide(toolkitLayer(store))); + }).pipe(Effect.scoped, Effect.provide(NodeServices.layer), Effect.orDie); + +// --- pure-side fixtures ----------------------------------------------------- + +const shell = (o: Partial = {}): LoopThreadShell => + ({ + updatedAt: new Date(NOW - 20 * MINUTE).toISOString(), + archivedAt: null, + settledOverride: null, + snoozedUntil: null, + session: null, + latestTurn: null, + latestUserMessageAt: null, + hasPendingApprovals: false, + hasPendingUserInput: false, + hasActionableProposedPlan: false, + ...o, + }) as unknown as LoopThreadShell; + +const guardInput = (o: Partial = {}): LoopGuardInput => ({ + nowMs: NOW, + processStartedAtMs: NOW - 24 * HOUR, + record: { + ...EMPTY_RECORD, + armed: true, + armedAtMs: NOW - 2 * HOUR, + maxCheckIns: 6, + checkInsUsed: 2, + deadlineAtMs: NOW + 4 * HOUR, + }, + global: { ...DEFAULT_GLOBAL_SETTINGS, enabled: true }, + shell: shell(), + sentinelAtMs: null, + loopDoneAtMs: null, + autoResumePending: false, + armedCount: 1, + config: resolveConfig({}), + trigger: { idleForMs: 20 * MINUTE, thresholdMs: 15 * MINUTE, busyTurn: false, wake: null }, + ...o, +}); + +describe("raise_blocker", () => { + it.effect("108. returns immediately — it awaits nothing but its own write", () => + withTools((store) => + Effect.gen(function* () { + // Standing in for the `Deferred` `AskUserQuestion` parks the turn on. If + // `raise_blocker` ever waits for an answer this test deadlocks and the suite goes + // red, which is the only assertion that actually protects the design. + const answer = yield* Deferred.make(); + + const before = yield* Clock.currentTimeMillis; + const result = yield* callTool("raise_blocker", { question: "Migration or shim?" }); + const after = yield* Clock.currentTimeMillis; + + assert.strictEqual(after - before, 0); + assert.isFalse(yield* Deferred.isDone(answer)); + expect(structured(result)).toMatchObject({ status: "recorded" }); + assert.lengthOf(yield* store.listOpenBlockers(THREAD), 1); + }), + ), + ); + + it.effect("109. attributes the blocker to the calling thread, never to an argument", () => + withTools( + (store) => + Effect.gen(function* () { + // A `threadId` in the payload is stripped by the parameter schema before a + // handler ever sees it, so it cannot redirect a question at another thread. + yield* callTool("raise_blocker", { + question: "Whose thread is this?", + threadId: OTHER_THREAD, + }); + + assert.lengthOf(yield* store.listOpenBlockers(THREAD), 1); + assert.lengthOf(yield* store.listOpenBlockers(OTHER_THREAD), 0); + }), + (store) => + store + .setGlobal({ enabled: true }) + .pipe(Effect.andThen(armed(store)), Effect.andThen(armed(store, OTHER_THREAD))), + ), + ); + + it.effect("110. records a free-text blocker when no options are given", () => + withTools((store) => + Effect.gen(function* () { + yield* callTool("raise_blocker", { question: " What should the default be? " }); + + const [blocker] = yield* store.listOpenBlockers(THREAD); + expect(blocker).toMatchObject({ + question: "What should the default be?", + options: [], + context: null, + answeredAtMs: null, + deliveredToAgent: false, + }); + }), + ), + ); + + it.effect("111. shapes options so the console renders them as a UserInputQuestion", () => + withTools((store) => + Effect.gen(function* () { + yield* callTool("raise_blocker", { + question: "Migration or shim?", + options: [ + { label: "migration", description: " slower, correct " }, + { label: " ", description: "unusable, dropped" }, + ], + context: "packages/contracts/src/settings.ts", + }); + + const [blocker] = yield* store.listOpenBlockers(THREAD); + assert.isDefined(blocker); + // The real proof: the recorded options decode as the contract's own question type, + // so the console can hand them to the native component rather than a second widget. + const question = yield* decodeUserInputQuestion({ + id: blocker.id, + header: "Blocker", + question: blocker.question, + options: blocker.options, + }); + expect(question.options).toEqual([{ label: "migration", description: "slower, correct" }]); + assert.strictEqual(blocker.context, "packages/contracts/src/settings.ts"); + }), + ), + ); + + it.effect("112. says so rather than banking a question from a thread with no armed loop", () => + withTools( + (store) => + Effect.gen(function* () { + const result = yield* callTool("raise_blocker", { question: "Still worth asking?" }); + + // `recorded` here was a promise nothing could keep: answers are delivered by a + // check-in prompt, and with no armed loop there will never be one — so the agent + // would park a branch of the work on an answer that could not arrive. That is the + // exact failure this tool exists to prevent, and a status is how it is reported. + expect(structured(result)).toMatchObject({ status: "unavailable", id: null }); + assert.include(String(structured(result).detail), "No loop is armed"); + assert.lengthOf(yield* store.listOpenBlockers(THREAD), 0); + }), + (store) => store.setGlobal({ enabled: true }), + ), + ); + + it.effect("112b. a stopped loop is no more deliverable than a thread that never had one", () => + withTools( + (store) => + Effect.gen(function* () { + const result = yield* callTool("raise_blocker", { question: "One more thing?" }); + expect(structured(result)).toMatchObject({ status: "unavailable" }); + assert.lengthOf(yield* store.listOpenBlockers(THREAD), 0); + }), + (store) => + store + .setGlobal({ enabled: true }) + .pipe( + Effect.andThen(armed(store)), + Effect.andThen(store.stop(THREAD, { reason: "spent", atMs: NOW, detail: "budget" })), + ), + ), + ); + + it.effect("113. reports the cap back to the agent rather than dropping the question", () => + withTools((store) => + Effect.gen(function* () { + for (let n = 0; n < MAX_OPEN_BLOCKERS_PER_WINDOW; n += 1) { + const accepted = yield* callTool("raise_blocker", { question: `question ${n}` }); + expect(structured(accepted)).toMatchObject({ status: "recorded" }); + } + + const capped = yield* callTool("raise_blocker", { question: "one too many" }); + + expect(structured(capped)).toMatchObject({ + status: "capped", + id: null, + openBlockers: MAX_OPEN_BLOCKERS_PER_WINDOW, + cap: MAX_OPEN_BLOCKERS_PER_WINDOW, + }); + assert.include(String(structured(capped).detail), "Not recorded"); + assert.lengthOf(yield* store.listOpenBlockers(THREAD), MAX_OPEN_BLOCKERS_PER_WINDOW); + }), + ), + ); + + it.effect("113b. answering frees cap budget, so the channel is never exhausted for good", () => + withTools((store) => + Effect.gen(function* () { + for (let n = 0; n < MAX_OPEN_BLOCKERS_PER_WINDOW; n += 1) { + yield* callTool("raise_blocker", { question: `question ${n}` }); + } + const open = yield* store.listOpenBlockers(THREAD); + yield* store.answerBlocker(THREAD, open[0]!.id, "take the migration", NOW + MINUTE); + + const result = yield* callTool("raise_blocker", { question: "room again" }); + + expect(structured(result)).toMatchObject({ status: "recorded" }); + }), + ), + ); + + it.effect("118. is unavailable, in-band, when the master toggle is off", () => + withTools( + (store) => + Effect.gen(function* () { + const result = yield* callTool("raise_blocker", { question: "anyone home?" }); + + expect(structured(result)).toMatchObject({ status: "unavailable", id: null }); + assert.lengthOf(yield* store.listOpenBlockers(THREAD), 0); + }), + (store) => store.setGlobal({ enabled: false }).pipe(Effect.andThen(armed(store))), + ), + ); + + it.effect("118g. cannot record anything without an invocation context", () => + withTools((store) => + Effect.gen(function* () { + // With `enableAgentBrowserAccess` off, `ProviderService.prepareMcpSession` revokes + // the credential and mints none, so `/mcp` 401s and no invocation context is ever + // provided. This is that state at the tool boundary: there is no unattributed path, + // so the tools simply vanish rather than writing against a guessed thread. + const exit = yield* McpServer.McpServer.pipe( + Effect.flatMap((server) => + server.callTool({ name: "raise_blocker", arguments: { question: "no credential" } }), + ), + Effect.provideService(McpSchema.McpServerClient, client), + Effect.exit, + ); + + // The server turns the missing service into an error result with no structured + // content: the agent is told the tool failed, and nothing is written against a + // guessed thread. That is the whole of "the tools vanish". + assert.strictEqual(exit._tag, "Success"); + const result = exit._tag === "Success" ? exit.value : null; + assert.isTrue(result?.isError); + assert.isUndefined(result?.structuredContent); + assert.lengthOf(yield* store.listOpenBlockers(THREAD), 0); + }), + ), + ); +}); + +describe("loop_status", () => { + it.effect("114. reports the true remaining budget and deadline", () => + withTools( + () => + Effect.gen(function* () { + const result = yield* callTool("loop_status", {}); + + expect(structured(result)).toEqual({ + armed: true, + reason: null, + state: "watching", + checkInsUsed: 2, + maxCheckIns: 6, + deadlineAtMs: NOW + 4 * HOUR, + msToDeadline: 4 * HOUR, + blockersOpen: 1, + }); + }), + (store) => + store.setGlobal({ enabled: true }).pipe( + Effect.andThen(armed(store)), + Effect.andThen( + store.recordCheckIn({ + threadId: THREAD, + firedAtMs: NOW - 90 * MINUTE, + createdAtIso: "2027-01-15T06:30:00.000Z", + activityCursor: "act-1", + }), + ), + Effect.andThen( + store.recordCheckIn({ + threadId: THREAD, + firedAtMs: NOW - 45 * MINUTE, + createdAtIso: "2027-01-15T07:15:00.000Z", + activityCursor: "act-2", + }), + ), + Effect.andThen( + store.addBlocker(THREAD, { + id: "b-1", + raisedAtMs: NOW - 40 * MINUTE, + question: "still open", + options: [], + context: null, + answeredAtMs: null, + answer: null, + deliveredToAgent: false, + }), + ), + ), + ), + ); + + it.effect("115. answers no-loop rather than failing on an unsupervised thread", () => + withTools( + () => + Effect.gen(function* () { + const result = yield* callTool("loop_status", {}); + + assert.isFalse(result.isError); + expect(structured(result)).toMatchObject({ + armed: false, + reason: "no-loop", + state: "off", + checkInsUsed: 0, + maxCheckIns: 0, + msToDeadline: 0, + blockersOpen: 0, + }); + }), + (store) => store.setGlobal({ enabled: true }), + ), + ); + + it.effect("118. answers rather than failing when the master toggle is off", () => + withTools( + () => + Effect.gen(function* () { + const result = yield* callTool("loop_status", {}); + + assert.isFalse(result.isError); + expect(structured(result)).toMatchObject({ + armed: false, + reason: "disabled", + state: "standing_down", + }); + }), + (store) => store.setGlobal({ enabled: false }).pipe(Effect.andThen(armed(store))), + ), + ); +}); + +describe("loop_done", () => { + it.effect("116. sets exactly the signal the reactor's stop sweep reads", () => + withTools((store) => + Effect.gen(function* () { + const result = yield* callTool("loop_done", { reason: "migration landed" }); + + expect(structured(result)).toMatchObject({ ok: true, status: "recorded" }); + const record = yield* store.getThread(THREAD); + assert.strictEqual(record.loopDoneAtMs, NOW); + assert.strictEqual(record.loopDoneReason, "migration landed"); + + // `doneSignal` is the predicate the reactor feeds from the record. Asserting on it + // rather than on a field name is what keeps this test honest if the plumbing moves. + expect( + doneSignal({ record, sentinelAtMs: null, loopDoneAtMs: record.loopDoneAtMs }), + ).toEqual({ cause: "loop_done", atMs: NOW }); + expect( + stopCondition({ + nowMs: NOW, + record, + shell: shell(), + sentinelAtMs: null, + loopDoneAtMs: record.loopDoneAtMs, + }), + ).toMatchObject({ kind: "stop", outcome: "done", cause: "loop_done" }); + }), + ), + ); + + it.effect("116b. is equivalent to the done-file: same outcome, same freshness rule", () => + withTools((store) => + Effect.gen(function* () { + yield* callTool("loop_done", { reason: "done here" }); + const record = yield* store.getThread(THREAD); + + const viaTool = stopCondition({ + nowMs: NOW, + record, + shell: shell(), + sentinelAtMs: null, + loopDoneAtMs: record.loopDoneAtMs, + }); + const viaFile = stopCondition({ + nowMs: NOW, + record: { ...record, loopDoneAtMs: null }, + shell: shell(), + sentinelAtMs: NOW, + loopDoneAtMs: null, + }); + assert.strictEqual(viaTool?.outcome, viaFile?.outcome); + assert.strictEqual(viaTool?.outcome, "done"); + + // Same freshness rule as a leftover `.coil/loop-done`: a call from a previous run + // does not end the next one, because a re-arm takes a newer `armedAtMs`. + const rearmed = { ...record, armedAtMs: NOW + MINUTE }; + expect( + doneSignal({ record: rearmed, sentinelAtMs: null, loopDoneAtMs: record.loopDoneAtMs }), + ).toBeNull(); + }), + ), + ); + + it.effect("116c. the whole guard table stops the run once the signal is set", () => + withTools((store) => + Effect.gen(function* () { + yield* callTool("loop_done", { reason: "all finished" }); + const record = yield* store.getThread(THREAD); + + expect(evaluateGuards(guardInput({ record, loopDoneAtMs: record.loopDoneAtMs }))).toEqual({ + kind: "stop", + guard: "4b", + outcome: "done", + cause: "loop_done", + detail: `loop_done at ${NOW}`, + }); + }), + ), + ); + + it.effect("117. is a no-op, not a crash, from a thread with no loop", () => + withTools( + (store) => + Effect.gen(function* () { + const result = yield* callTool("loop_done", { reason: "nothing to end" }); + + assert.isFalse(result.isError); + expect(structured(result)).toMatchObject({ ok: true, status: "no-loop" }); + assert.strictEqual((yield* store.getThread(THREAD)).loopDoneAtMs, null); + }), + (store) => store.setGlobal({ enabled: true }), + ), + ); + + it.effect("118. records the done signal even when the master toggle is off", () => + withTools( + (store) => + Effect.gen(function* () { + const result = yield* callTool("loop_done", { reason: "toggle is off" }); + + // The toggle gates what FIRES, never what is written down. Discarding the signal + // meant an agent that finished while loops were switched off had its `done` thrown + // away — and switching loops back on resumed check-ins against a finished run. + expect(structured(result)).toMatchObject({ ok: true, status: "disabled" }); + const record = yield* store.getThread(THREAD); + assert.isNotNull(record.loopDoneAtMs); + assert.strictEqual(record.loopDoneReason, "toggle is off"); + assert.isNull(record.stopped, "and it is still the supervisor that writes the stop"); + }), + (store) => store.setGlobal({ enabled: false }).pipe(Effect.andThen(armed(store))), + ), + ); +}); + +describe("118h. a session-resume dialog the loop caused itself", () => { + const resumeReturn: UserInputRecord = { + requestId: "req-resume-1", + raisedAtMs: NOW - 4 * HOUR, + dialogKind: "resume_return", + question: "Resume this session?", + resolution: null, + resolvedAtMs: null, + }; + + it.effect("keeps the non-blocking channel open while the turn is parked on the dialog", () => + withTools( + (store) => + Effect.gen(function* () { + // The point of `raise_blocker`: the blocking channel is occupied by a dialog the + // check-in itself triggered, and the agent can still bank a question and read its + // budget without waiting on anyone. + const raised = yield* callTool("raise_blocker", { question: "parked, still asking" }); + const status = yield* callTool("loop_status", {}); + + expect(structured(raised)).toMatchObject({ status: "recorded" }); + expect(structured(status)).toMatchObject({ armed: true, blockersOpen: 1 }); + const record = yield* store.getThread(THREAD); + assert.strictEqual(record.userInputs[0]?.dialogKind, "resume_return"); + }), + (store) => + store + .setGlobal({ enabled: true }) + .pipe( + Effect.andThen(armed(store)), + Effect.andThen(store.recordUserInput(THREAD, resumeReturn)), + ), + ), + ); + + it("guard 8 still skips, spends nothing, and the run ends spent rather than stalled", () => { + const parked = shell({ hasPendingUserInput: true }); + const record = { + ...guardInput().record, + checkInsUsed: 1, + strikes: 0, + userInputs: [resumeReturn], + }; + + // Parked: a non-consuming stand-down, so the budget is untouched however long it sits. + expect(evaluateGuards(guardInput({ record, shell: parked }))).toMatchObject({ + kind: "stand_down", + guard: "8", + reason: "pending_user_input", + }); + + // And when the deadline arrives it is `spent` — never `stalled`, which would blame the + // agent for a dialog it was never given the chance to answer. + expect( + evaluateGuards(guardInput({ record, shell: parked, nowMs: record.deadlineAtMs + MINUTE })), + ).toMatchObject({ kind: "stop", outcome: "spent", cause: "deadline" }); + }); +}); diff --git a/apps/server/src/mcp/toolkits/loop/handlers.ts b/apps/server/src/mcp/toolkits/loop/handlers.ts new file mode 100644 index 000000000000..ca176033c0be --- /dev/null +++ b/apps/server/src/mcp/toolkits/loop/handlers.ts @@ -0,0 +1,250 @@ +/** + * Handlers for the loop toolkit, plus the layer that registers it on the MCP server. + * + * Everything here obeys four rules that the tests pin: + * + * 1. **Attribution is read, never accepted.** `threadId` comes from + * `McpInvocationContext`, which the per-thread bearer credential resolves. There is no + * argument to spoof, and an extra `threadId` in the payload is stripped by the + * parameter schema before a handler ever sees it. + * 2. **`raise_blocker` awaits nothing but its own write.** It appends to the record and + * returns. It never constructs, reads or waits on a `Deferred`, which is the entire + * reason the tool exists. + * 3. **The cap reports itself.** A capped call comes back as `status: "capped"` with the + * numbers, so the agent knows the question was not banked. Silently dropping it would + * reproduce the failure `raise_blocker` was built to prevent. + * 4. **Nothing fails.** The gate (`global.enabled`), an unarmed thread and a thread that + * has never been supervised are all statuses. A tool that errors mid-turn costs the + * agent reasoning it should be spending on the work. + * + * @module mcp/toolkits/loop/handlers + */ + +import * as Clock from "effect/Clock"; +import * as Effect from "effect/Effect"; +import * as Layer from "effect/Layer"; +import * as Random from "effect/Random"; +import { McpServer } from "effect/unstable/ai"; + +import { LoopStoreLive } from "../../../coil/loop/layer.ts"; +import { type Blocker, type BlockerOption, LoopStore } from "../../../coil/loop/state.ts"; +import { deriveLoopStatus } from "../../../coil/loop/status.ts"; +import * as McpInvocationContext from "../../McpInvocationContext.ts"; +import { LoopToolkit } from "./tools.ts"; + +/** + * How many *unanswered* blockers one check-in window may hold. + * + * The brief calls this a per-turn cap. `McpInvocationContext` carries no turn identity — + * `{environmentId, threadId, providerSessionId, providerInstanceId, capabilities, issuedAt}` + * and nothing else — so the window is the check-in, which is the loop's unit of work + * anyway, and for a thread with no loop it degrades to "at most this many questions + * outstanding at once". Only unanswered blockers count, so a human answering always frees + * budget, and no agent can permanently exhaust its own channel. + * + * Deliberately a constant here rather than in `coil/loop/config.ts`: it bounds a tool, not + * the supervisor, and there is no operator story for tuning it yet. Promote it to + * `COIL_LOOP_MAX_BLOCKERS_PER_TURN` the first time someone actually hits it. + */ +export const MAX_OPEN_BLOCKERS_PER_WINDOW = 10; + +/** Bounds one blocker's contribution to the state file. Truncation is always reported. */ +const MAX_QUESTION_CHARS = 2_000; +const MAX_CONTEXT_CHARS = 2_000; +const MAX_OPTIONS = 10; + +const truncate = (value: string, max: number): { text: string; truncated: boolean } => + value.length <= max + ? { text: value, truncated: false } + : { text: value.slice(0, max), truncated: true }; + +/** + * Trims to `UserInputQuestionOption`'s shape and drops the unusable. + * + * An option with no label cannot be rendered or chosen, so it is dropped rather than + * carried as a blank row. An empty description is kept — the native component tolerates it, + * and inventing text would put words in the agent's mouth. + */ +export const normalizeBlockerOptions = ( + options: ReadonlyArray<{ readonly label: string; readonly description: string }> | undefined, +): ReadonlyArray => + (options ?? []) + .map((option) => ({ label: option.label.trim(), description: option.description.trim() })) + .filter((option) => option.label.length > 0) + .slice(0, MAX_OPTIONS); + +/** + * The start of the current cap window. + * + * The last check-in when there is one, otherwise the arm, otherwise the beginning of time — + * which is the right reading for a thread with no loop, where every open blocker counts. + */ +export const capWindowStartMs = (input: { + readonly armedAtMs: number; + readonly lastCheckInAtMs: number | null; +}): number => Math.max(input.armedAtMs, input.lastCheckInAtMs ?? 0); + +const handlers = { + raise_blocker: Effect.fn("LoopToolkit.raise_blocker")(function* (input) { + const { threadId } = yield* McpInvocationContext.McpInvocationContext; + const store = yield* LoopStore; + const global = yield* store.getGlobal; + const record = yield* store.getThread(threadId); + const open = record.blockers.filter((entry) => entry.answeredAtMs === null); + + if (!global.enabled) { + return { + status: "unavailable" as const, + id: null, + detail: + "Loops are switched off for this machine, so nothing would ever read this question. Ask directly instead.", + openBlockers: open.length, + cap: MAX_OPEN_BLOCKERS_PER_WINDOW, + }; + } + // No armed loop means no check-in will ever carry the answer back, and the console shows + // this channel only for a thread that has a loop. Recording here would report `recorded` + // for a question that is not merely late but undeliverable — the exact failure this tool + // exists to prevent, with the agent believing it had banked the decision. + if (!record.armed || record.stopped !== null) { + return { + status: "unavailable" as const, + id: null, + detail: + "No loop is armed on this thread, so no check-in will ever deliver an answer to this question. Ask directly instead.", + openBlockers: open.length, + cap: MAX_OPEN_BLOCKERS_PER_WINDOW, + }; + } + + const windowStartMs = capWindowStartMs({ + armedAtMs: record.armedAtMs, + lastCheckInAtMs: record.lastCheckIn?.firedAtMs ?? null, + }); + const openThisWindow = open.filter((entry) => entry.raisedAtMs >= windowStartMs); + if (openThisWindow.length >= MAX_OPEN_BLOCKERS_PER_WINDOW) { + return { + status: "capped" as const, + id: null, + detail: `Not recorded: ${openThisWindow.length} questions are already waiting on a human (cap ${MAX_OPEN_BLOCKERS_PER_WINDOW}). Carry on with work that does not depend on this one; the cap frees up as questions are answered.`, + openBlockers: open.length, + cap: MAX_OPEN_BLOCKERS_PER_WINDOW, + }; + } + + const nowMs = yield* Clock.currentTimeMillis; + const suffix = yield* Random.nextIntBetween(0, 0xff_ff_ff); + const question = truncate(input.question.trim(), MAX_QUESTION_CHARS); + const context = truncate((input.context ?? "").trim(), MAX_CONTEXT_CHARS); + const options = normalizeBlockerOptions(input.options); + const blocker: Blocker = { + id: `blocker-${nowMs.toString(36)}-${suffix.toString(36).padStart(5, "0")}`, + raisedAtMs: nowMs, + question: question.text, + options, + context: context.text.length > 0 ? context.text : null, + answeredAtMs: null, + answer: null, + deliveredToAgent: false, + }; + const recorded = yield* store.addBlocker(threadId, blocker); + const droppedOptions = (input.options?.length ?? 0) - options.length; + const notes = [ + question.truncated ? `question truncated to ${MAX_QUESTION_CHARS} characters` : null, + context.truncated ? `context truncated to ${MAX_CONTEXT_CHARS} characters` : null, + droppedOptions > 0 ? `${droppedOptions} unusable option(s) dropped` : null, + ].filter((note): note is string => note !== null); + return { + status: "recorded" as const, + id: recorded.id, + detail: + notes.length === 0 + ? "Recorded. Keep working; the answer arrives in a later check-in message." + : `Recorded (${notes.join("; ")}). Keep working; the answer arrives in a later check-in message.`, + openBlockers: open.length + 1, + cap: MAX_OPEN_BLOCKERS_PER_WINDOW, + }; + }), + + loop_status: Effect.fn("LoopToolkit.loop_status")(function* () { + const { threadId } = yield* McpInvocationContext.McpInvocationContext; + const store = yield* LoopStore; + const global = yield* store.getGlobal; + const record = yield* store.getThread(threadId); + const nowMs = yield* Clock.currentTimeMillis; + // No thread shell here: an MCP call has a store and nothing else. The shell-derived + // clauses (snooze, blocked-on-a-human) are skipped rather than guessed — and an agent + // running well enough to call this is not the thread those clauses describe. + const status = deriveLoopStatus({ nowMs, record, global, shell: null }); + return { + armed: status.armed, + reason: status.reason, + state: status.state, + checkInsUsed: status.checkInsUsed, + maxCheckIns: status.maxCheckIns, + deadlineAtMs: status.deadlineAtMs, + msToDeadline: status.msToDeadline, + blockersOpen: record.blockers.filter((entry) => entry.answeredAtMs === null).length, + }; + }), + + loop_done: Effect.fn("LoopToolkit.loop_done")(function* (input) { + const { threadId } = yield* McpInvocationContext.McpInvocationContext; + const store = yield* LoopStore; + const global = yield* store.getGlobal; + const record = yield* store.getThread(threadId); + + if (!record.armed || record.stopped !== null) { + return { + ok: true, + status: "no-loop" as const, + detail: "No loop is running on this thread, so there was nothing to end.", + }; + } + + // Recorded **before** the master toggle is consulted, and that ordering is the whole + // point: the toggle gates what *fires*, never what is written down. With the write behind + // it, an agent that finished while loops were switched off had its `done` thrown away — + // and switching loops back on resumed check-ins against a run that was over. + // + // Exactly what `guards.ts` `doneSignal` reads: a timestamp newer than `armedAtMs`. The + // record is never cleared, for the same reason the supervisor never deletes a + // `.coil/loop-done` file — a re-arm takes a fresh `armedAtMs` and supersedes both. + const nowMs = yield* Clock.currentTimeMillis; + const reason = truncate(input.reason.trim(), MAX_CONTEXT_CHARS).text; + yield* store.update(threadId, (current) => ({ + ...current, + loopDoneAtMs: nowMs, + loopDoneReason: reason.length > 0 ? reason : null, + })); + if (!global.enabled) { + return { + ok: true, + status: "disabled" as const, + detail: + "Loops are switched off for this machine, so nothing was checking in on you anyway. The run is recorded as done; switching loops back on will not resume it.", + }; + } + return { + ok: true, + status: "recorded" as const, + detail: "Run marked done. No further check-ins will be sent for it.", + }; + }), +} satisfies Parameters[0]; + +export const LoopToolkitHandlersLive = LoopToolkit.toLayer(handlers); + +/** + * Registers the three tools on the shared `McpServer`. + * + * `LoopStore` is provided here rather than left open, so the fork never widens the type of + * upstream's `makeRoutesLayer`: an unsatisfied requirement in `McpHttpServer.layer` would + * surface in `server.ts` and cost a second seam edit. `LoopStoreLive` is the same + * module-level layer value the reactor and the HTTP routes use, so all three share one + * store over one file. + */ +export const LoopToolkitRegistrationLive = McpServer.toolkit(LoopToolkit).pipe( + Layer.provide(LoopToolkitHandlersLive), + Layer.provide(LoopStoreLive), +); diff --git a/apps/server/src/mcp/toolkits/loop/tools.test.ts b/apps/server/src/mcp/toolkits/loop/tools.test.ts new file mode 100644 index 000000000000..b0806f1ebd3b --- /dev/null +++ b/apps/server/src/mcp/toolkits/loop/tools.test.ts @@ -0,0 +1,110 @@ +// @effect-diagnostics nodeBuiltinImport:off +/** + * TESTS.md §7, case 118 and the registration half of the phase: the three tools reach a + * provider through the same `McpServer` upstream's preview toolkit registers on, and their + * schemas are the shapes the console and the agent were promised. + */ + +import * as NodeServices from "@effect/platform-node/NodeServices"; +import { assert, expect, it } from "@effect/vitest"; +import * as Effect from "effect/Effect"; +import * as FileSystem from "effect/FileSystem"; +import * as Layer from "effect/Layer"; +import { McpServer, Tool } from "effect/unstable/ai"; + +import * as ServerConfig from "../../../config.ts"; +import * as McpHttpServer from "../../McpHttpServer.ts"; +import * as PreviewAutomationBroker from "../../PreviewAutomationBroker.ts"; +import { LoopToolkitRegistrationLive } from "./handlers.ts"; +import { LOOP_TOOL_NAMES, LoopToolkit } from "./tools.ts"; + +/** + * A described field may sit inside the `anyOf` an optional parameter compiles to, so the + * check recurses the same way upstream's preview toolkit test does. + */ +const schemaHasDescription = (schema: unknown): boolean => { + if (!schema || typeof schema !== "object") return false; + const record = schema as Record; + if (typeof record.description === "string" && record.description.length > 0) return true; + return [record.anyOf, record.oneOf, record.allOf] + .filter(Array.isArray) + .some((members) => members.some(schemaHasDescription)); +}; + +const toolNames = Effect.map(McpServer.McpServer, (server) => + server.tools.map((entry) => entry.tool.name), +); + +it.effect("registers all three tools on the shared MCP server", () => + Effect.gen(function* () { + const fs = yield* FileSystem.FileSystem; + const root = yield* fs.makeTempDirectoryScoped({ prefix: "coil-loop-registration-" }); + + const names = yield* toolNames.pipe( + Effect.provide( + LoopToolkitRegistrationLive.pipe( + Layer.provideMerge(McpServer.McpServer.layer), + Layer.provide(ServerConfig.layerTest(root, root)), + ), + ), + ); + + expect([...names].sort()).toEqual([...LOOP_TOOL_NAMES].sort()); + }).pipe(Effect.scoped, Effect.provide(NodeServices.layer), Effect.orDie), +); + +it.effect("registers alongside upstream's preview toolkit rather than replacing it", () => + Effect.gen(function* () { + const fs = yield* FileSystem.FileSystem; + const root = yield* fs.makeTempDirectoryScoped({ prefix: "coil-loop-registration-" }); + + // The same merge `McpHttpServer.layer` performs, minus the HTTP transport: if the two + // registrations ever stopped sharing one `McpServer` instance, one set would vanish. + const names = yield* toolNames.pipe( + Effect.provide( + Layer.mergeAll( + McpHttpServer.PreviewToolkitRegistrationLive, + LoopToolkitRegistrationLive, + ).pipe( + Layer.provideMerge(McpServer.McpServer.layer), + Layer.provide(ServerConfig.layerTest(root, root)), + Layer.provide(PreviewAutomationBroker.layer), + ), + ), + ); + + for (const name of LOOP_TOOL_NAMES) assert.include(names, name); + assert.include(names, "preview_snapshot"); + assert.include(names, "preview_status"); + }).pipe(Effect.scoped, Effect.provide(NodeServices.layer), Effect.orDie), +); + +it("exports described object schemas the agent can fill in without guessing", () => { + for (const tool of Object.values(LoopToolkit.tools)) { + const schema = Tool.getJsonSchema(tool) as { + readonly type?: unknown; + readonly properties?: Readonly>; + readonly anyOf?: unknown; + readonly oneOf?: unknown; + }; + expect( + tool.description?.length ?? 0, + `${tool.name} should have a useful description`, + ).toBeGreaterThan(40); + expect(schema.type, `${tool.name} must export a top-level object schema`).toBe("object"); + expect(schema.anyOf, `${tool.name} must not export a root anyOf`).toBeUndefined(); + expect(schema.oneOf, `${tool.name} must not export a root oneOf`).toBeUndefined(); + // Attribution is the credential's job. A thread id in the arguments would be a value + // the model can get wrong, and the handler would have to decide whether to trust it. + expect( + schema.properties?.threadId, + `${tool.name} must not accept a thread id from the model`, + ).toBeUndefined(); + for (const [field, fieldSchema] of Object.entries(schema.properties ?? {})) { + expect( + schemaHasDescription(fieldSchema), + `${tool.name}.${field} should explain what data the agent must pass`, + ).toBe(true); + } + } +}); diff --git a/apps/server/src/mcp/toolkits/loop/tools.ts b/apps/server/src/mcp/toolkits/loop/tools.ts new file mode 100644 index 000000000000..0a13463febea --- /dev/null +++ b/apps/server/src/mcp/toolkits/loop/tools.ts @@ -0,0 +1,157 @@ +/** + * The loop toolkit — a non-blocking question channel, a budget readout, and a done signal. + * + * `raise_blocker` is the reason this toolkit exists. `AskUserQuestion` parks the turn on a + * `Deferred`, so an agent that hits a real fork in the road at 01:00 and asks about it + * *correctly* stops working until someone wakes up, and the supervisor then correctly + * refuses to nudge a thread that is blocked on a human. No tuning resolves that; a second, + * non-blocking channel does. `raise_blocker` records the question and returns — the answer + * is delivered on a later check-in through the prompt, never by unblocking a `Deferred`. + * + * Three deliberate shapes: + * + * - **No `threadId` parameter, on any tool.** Attribution comes from + * `McpInvocationContext`, which the per-thread bearer credential resolves. A thread id in + * the arguments would be a value the model could get wrong or be talked into. + * - **`options` mirrors `UserInputQuestionOption` (`{label, description}`)** so the console + * renders a raised blocker with the same native component it uses for a real + * `AskUserQuestion`, rather than a second bespoke widget. + * - **Every result is a flat struct with a `status`, and no tool declares a failure.** A + * capped `raise_blocker`, a `loop_status` on a thread with no loop and a `loop_done` from + * an unsupervised thread are all *answers*, not errors. An error mid-turn is something + * the agent has to reason about; a status is something it can act on. + * + * @module mcp/toolkits/loop/tools + */ + +import * as Schema from "effect/Schema"; +import { Tool, Toolkit } from "effect/unstable/ai"; + +import { LoopStore } from "../../../coil/loop/state.ts"; +import * as McpInvocationContext from "../../McpInvocationContext.ts"; + +const dependencies = [McpInvocationContext.McpInvocationContext, LoopStore]; + +/** + * Shaped exactly like `UserInputQuestionOption`. + * + * Plain strings rather than the contract's trimmed-non-empty refinements: a refusal here + * reaches the agent as a schema error on a tool that is supposed to never fail, so the + * handler trims and drops empties instead. + */ +const BlockerOptionInput = Schema.Struct({ + label: Schema.String.annotate({ description: "Short label for this choice." }), + description: Schema.String.annotate({ + description: "One sentence on what choosing this option would mean.", + }), +}); + +const RaiseBlockerParams = Schema.Struct({ + question: Schema.String.annotate({ + description: "The decision you need a human to make. Self-contained: it is read hours later.", + }), + options: Schema.optional( + Schema.Array(BlockerOptionInput).annotate({ + description: "The choices you see. Omit for a free-text answer.", + }), + ), + context: Schema.optional( + Schema.String.annotate({ + description: "What you were doing — file, issue, branch. Shown beside the question.", + }), + ), +}); + +const RaiseBlockerResult = Schema.Struct({ + /** + * `recorded` — banked, keep working. `capped` — too many unanswered questions already; + * this one was NOT recorded. `unavailable` — loops are switched off machine-wide. + */ + status: Schema.Literals(["recorded", "capped", "unavailable"]), + /** The blocker id, or `null` when nothing was recorded. */ + id: Schema.NullOr(Schema.String), + /** Always a sentence the agent can act on, including on the refusal paths. */ + detail: Schema.String, + /** Unanswered blockers on this thread after the call. */ + openBlockers: Schema.Number, + /** The ceiling `openBlockers` is measured against. */ + cap: Schema.Number, +}); + +const LoopStatusResult = Schema.Struct({ + /** Supervised right now: armed, and not in a terminal state. */ + armed: Schema.Boolean, + /** `no-loop` and `disabled` are the two ways `armed` is false without anything failing. */ + reason: Schema.NullOr(Schema.String), + state: Schema.Literals([ + "off", + "watching", + "self_pacing", + "standing_down", + "held", + "blocked", + "stopped", + ]), + checkInsUsed: Schema.Number, + maxCheckIns: Schema.Number, + deadlineAtMs: Schema.Number, + /** Clamped at 0. A passed deadline reads as no time left, never as negative time. */ + msToDeadline: Schema.Number, + blockersOpen: Schema.Number, +}); + +const LoopDoneParams = Schema.Struct({ + reason: Schema.String.annotate({ + description: "One line on what you finished, or why you are stopping.", + }), +}); + +const LoopDoneResult = Schema.Struct({ + /** True on every path: ending a run that was never supervised is a no-op, not a failure. */ + ok: Schema.Boolean, + status: Schema.Literals(["recorded", "no-loop", "disabled"]), + detail: Schema.String, +}); + +export const RaiseBlockerTool = Tool.make("raise_blocker", { + description: + "Bank a question for a human and keep working. Records the question against this thread and returns immediately — it does NOT wait for an answer, and answers arrive later in a check-in message. Use this instead of asking directly whenever stopping to ask would cost you the rest of an unattended run. Park that branch of the work and continue with something else.", + parameters: RaiseBlockerParams, + success: RaiseBlockerResult, + dependencies, +}) + .annotate(Tool.Title, "Raise a blocker") + .annotate(Tool.Readonly, false) + .annotate(Tool.Destructive, false) + .annotate(Tool.Idempotent, false) + .annotate(Tool.OpenWorld, false); + +export const LoopStatusTool = Tool.make("loop_status", { + description: + "Report this thread's supervision budget: whether a loop is armed, check-ins used against the budget, time left before the deadline, and how many raised blockers are still unanswered. Use it to scope how much work to take on before the run ends. Returns a status rather than an error when no loop is armed.", + success: LoopStatusResult, + dependencies, +}) + .annotate(Tool.Title, "Get loop status") + .annotate(Tool.Readonly, true) + .annotate(Tool.Destructive, false) + .annotate(Tool.Idempotent, true) + .annotate(Tool.OpenWorld, false); + +export const LoopDoneTool = Tool.make("loop_done", { + description: + "Declare the supervised run finished, so no further check-ins are sent. Equivalent to writing .coil/loop-done in the working tree, which stays the primary contract because it works from a plain terminal. A no-op when no loop is armed.", + parameters: LoopDoneParams, + success: LoopDoneResult, + dependencies, +}) + .annotate(Tool.Title, "End the loop") + .annotate(Tool.Readonly, false) + .annotate(Tool.Destructive, false) + .annotate(Tool.Idempotent, true) + .annotate(Tool.OpenWorld, false); + +export const LoopToolkit = Toolkit.make(RaiseBlockerTool, LoopStatusTool, LoopDoneTool); + +/** The three names, for the registration test and for anything listing the fork's tools. */ +export const LOOP_TOOL_NAMES = ["raise_blocker", "loop_status", "loop_done"] as const; diff --git a/apps/server/src/provider/Layers/ClaudeAdapter.ts b/apps/server/src/provider/Layers/ClaudeAdapter.ts index 07739146b074..a1d38871863c 100644 --- a/apps/server/src/provider/Layers/ClaudeAdapter.ts +++ b/apps/server/src/provider/Layers/ClaudeAdapter.ts @@ -76,6 +76,7 @@ import * as Schema from "effect/Schema"; import * as Stream from "effect/Stream"; import { resolveAttachmentPath } from "../../attachmentStore.ts"; +import { loopHooksFor } from "../../coil/loop/crons.ts"; import { ServerConfig } from "../../config.ts"; import * as McpProviderSession from "../../mcp/McpProviderSession.ts"; import { resolveClaudeSdkExecutablePath } from "../Drivers/ClaudeExecutable.ts"; @@ -4349,6 +4350,7 @@ export const makeClaudeAdapter = Effect.fn("makeClaudeAdapter")(function* ( ...(input.cwd ? [input.cwd] : []), serverConfig.attachmentsDir, ]; + const loopHooks = yield* loopHooksFor(threadId); const queryOptions: ClaudeQueryOptions = { ...(input.cwd ? { cwd: input.cwd } : {}), ...(apiModelId ? { model: apiModelId } : {}), @@ -4389,6 +4391,7 @@ export const makeClaudeAdapter = Effect.fn("makeClaudeAdapter")(function* ( }, } : {}), + ...(loopHooks ? { hooks: loopHooks } : {}), }; yield* Effect.annotateCurrentSpan({ diff --git a/apps/web/src/coil/AutoResumeOverlay.tsx b/apps/web/src/coil/AutoResumeOverlay.tsx index 37c5e54fe313..c114a0c53e49 100644 --- a/apps/web/src/coil/AutoResumeOverlay.tsx +++ b/apps/web/src/coil/AutoResumeOverlay.tsx @@ -14,13 +14,8 @@ import { Textarea } from "~/components/ui/textarea"; import { Tooltip, TooltipPopup, TooltipTrigger } from "~/components/ui/tooltip"; import { cn } from "~/lib/utils"; -import { - type AnchorRect, - COMPOSER_FALLBACK_OFFSET_PX, - type ComposerAnchor, - resolveComposerAnchor, -} from "./autoResumeAnchor"; import { type AutoResumeThreadRef, httpAutoResumeClient } from "./autoResumeClient"; +import { useComposerAnchor } from "./composerAnchor"; import { createAutoResumeController } from "./autoResumeController"; import { describeAutoResumeTooltip, @@ -35,91 +30,6 @@ export { formatAutoResumeStatus } from "./autoResumePresentation"; const POLL_INTERVAL_MS = 30_000; const COUNTDOWN_TICK_MS = 1_000; -/** - * Read-only DOM dependency on upstream's composer overlay — the same element `ChatView` measures - * for its own `composerOverlayHeight`. The overlay is mounted as a sibling of `` in the - * route file, so it cannot receive that geometry as a prop without widening the seam. Degrades to - * `COMPOSER_FALLBACK_OFFSET_PX` if the attribute ever disappears. Recorded in docs/coil/SEAMS.md. - */ -const COMPOSER_OVERLAY_SELECTOR = '[data-chat-composer-overlay="true"]'; - -const UNMEASURED_ANCHOR: ComposerAnchor = { - visible: true, - bottom: COMPOSER_FALLBACK_OFFSET_PX, - left: null, - width: null, -}; - -function toAnchorRect(element: Element | null): AnchorRect | null { - if (!(element instanceof HTMLElement)) { - return null; - } - const { top, bottom, left, width, height } = element.getBoundingClientRect(); - return { top, bottom, left, width, height }; -} - -function sameAnchor(a: ComposerAnchor, b: ComposerAnchor): boolean { - return ( - a.visible === b.visible && a.bottom === b.bottom && a.left === b.left && a.width === b.width - ); -} - -/** - * Mirrors the composer overlay's box onto the capsule, in the capsule's own coordinate space. - * - * `anchorElement` is the capsule's positioned wrapper: it is what supplies the third rect the maths - * needs — its `offsetParent` (`SidebarInset`) is the box the returned `bottom`/`left` are resolved - * against, and it is the one box the panels do NOT resize. Reading it here rather than assuming it - * matches the composer's parent is the whole fix; see `resolveComposerAnchor`. - * - * Observes all three boxes because each panel perturbs a different one: the right panel changes the - * chat column's width, the terminal drawer changes its height, and collapsing the main sidebar - * changes `SidebarInset`'s width. A `ResizeObserver` fires per frame during those transitions and - * during a drag-resize, so the capsule travels with the composer rather than snapping after it. - */ -function useComposerAnchor(anchorElement: HTMLElement | null): ComposerAnchor { - const [anchor, setAnchor] = useState(UNMEASURED_ANCHOR); - - useEffect(() => { - if (anchorElement === null) { - return; - } - const composer = document.querySelector(COMPOSER_OVERLAY_SELECTOR); - if (!(composer instanceof HTMLElement)) { - setAnchor(UNMEASURED_ANCHOR); - return; - } - - const measure = () => { - const next = resolveComposerAnchor({ - composer: toAnchorRect(composer), - composerParent: toAnchorRect(composer.offsetParent), - anchorParent: toAnchorRect(anchorElement.offsetParent), - }); - // Identity-stable when nothing moved: a drag-resize of the drawer would otherwise re-render - // the capsule on every frame for no visible change. - setAnchor((previous) => (sameAnchor(previous, next) ? previous : next)); - }; - - measure(); - const observer = new ResizeObserver(measure); - observer.observe(composer); - if (composer.offsetParent instanceof HTMLElement) { - observer.observe(composer.offsetParent); - } - if (anchorElement.offsetParent instanceof HTMLElement) { - observer.observe(anchorElement.offsetParent); - } - window.addEventListener("resize", measure); - return () => { - observer.disconnect(); - window.removeEventListener("resize", measure); - }; - }, [anchorElement]); - - return anchor; -} - /** Re-renders once a second, but only while a resume is actually scheduled. */ function useCountdownTick(active: boolean): number { const [nowMs, setNowMs] = useState(() => Date.now()); diff --git a/apps/web/src/coil/ThreadCoilOverlay.tsx b/apps/web/src/coil/ThreadCoilOverlay.tsx new file mode 100644 index 000000000000..3eba9561a17a --- /dev/null +++ b/apps/web/src/coil/ThreadCoilOverlay.tsx @@ -0,0 +1,34 @@ +/** + * The fork's per-thread overlay surface, in one place. + * + * This exists so `routes/_chat.$environmentId.$threadId.tsx` — an upstream file with a seam row — + * names exactly one fork component forever. The row swapped one JSX element for one JSX element + * and one import for one import, so its `+10/−6` is unchanged and the seam delta of adding the + * loop console is **zero**. Every future per-thread fork surface now costs nothing at all: it is + * added here, not there. + * + * Both children position themselves against the docked composer through the shared + * `useComposerAnchor`, and each owns its own absolutely-positioned wrapper. That is deliberate + * rather than a shared wrapper: `resolveComposerAnchor` reads `anchorElement.offsetParent`, so + * introducing a positioned box between the overlays and `SidebarInset` would silently change the + * third rect the whole measurement is resolved against. They share the measurement, not the box — + * the auto-resume capsule holds the right of the composer card, the loop console the left. + * + * @module coil/ThreadCoilOverlay + */ + +import { AutoResumeOverlay, type AutoResumeThreadRef } from "./AutoResumeOverlay"; +import { LoopConsole } from "./loop/LoopConsole"; + +export interface ThreadCoilOverlayProps { + readonly threadRef: AutoResumeThreadRef; +} + +export function ThreadCoilOverlay({ threadRef }: ThreadCoilOverlayProps) { + return ( + <> + + + + ); +} diff --git a/apps/web/src/coil/composerAnchor.ts b/apps/web/src/coil/composerAnchor.ts new file mode 100644 index 000000000000..2a084b5d808e --- /dev/null +++ b/apps/web/src/coil/composerAnchor.ts @@ -0,0 +1,108 @@ +/** + * The shared "sit directly above the docked composer" hook. + * + * Extracted from `AutoResumeOverlay.tsx` unchanged when the loop console became a second + * per-thread fork surface with the same placement problem. Two independent copies of this + * measurement would be the fork's own known failure mode — a parallel path that drifts from the + * one that was actually debugged — so both overlays now read the same numbers from here. + * + * The maths itself lives in `autoResumeAnchor.ts`; this module is only the DOM half: find the + * composer, observe the three boxes that can move it, and hand `resolveComposerAnchor` three + * rects. + * + * @module coil/composerAnchor + */ + +import { useEffect, useState } from "react"; + +import { + type AnchorRect, + COMPOSER_FALLBACK_OFFSET_PX, + type ComposerAnchor, + resolveComposerAnchor, +} from "./autoResumeAnchor"; + +/** + * Read-only DOM dependency on upstream's composer overlay — the same element `ChatView` measures + * for its own `composerOverlayHeight`. The overlays are mounted as siblings of `` in the + * route file, so they cannot receive that geometry as a prop without widening the seam. Degrades to + * `COMPOSER_FALLBACK_OFFSET_PX` if the attribute ever disappears. Recorded in docs/coil/SEAMS.md. + */ +export const COMPOSER_OVERLAY_SELECTOR = '[data-chat-composer-overlay="true"]'; + +export const UNMEASURED_ANCHOR: ComposerAnchor = { + visible: true, + bottom: COMPOSER_FALLBACK_OFFSET_PX, + left: null, + width: null, +}; + +function toAnchorRect(element: Element | null): AnchorRect | null { + if (!(element instanceof HTMLElement)) { + return null; + } + const { top, bottom, left, width, height } = element.getBoundingClientRect(); + return { top, bottom, left, width, height }; +} + +function sameAnchor(a: ComposerAnchor, b: ComposerAnchor): boolean { + return ( + a.visible === b.visible && a.bottom === b.bottom && a.left === b.left && a.width === b.width + ); +} + +/** + * Mirrors the composer overlay's box onto an overlay, in that overlay's own coordinate space. + * + * `anchorElement` is the overlay's positioned wrapper: it is what supplies the third rect the maths + * needs — its `offsetParent` (`SidebarInset`) is the box the returned `bottom`/`left` are resolved + * against, and it is the one box the panels do NOT resize. Reading it here rather than assuming it + * matches the composer's parent is the whole fix; see `resolveComposerAnchor`. + * + * Observes all three boxes because each panel perturbs a different one: the right panel changes the + * chat column's width, the terminal drawer changes its height, and collapsing the main sidebar + * changes `SidebarInset`'s width. A `ResizeObserver` fires per frame during those transitions and + * during a drag-resize, so the overlay travels with the composer rather than snapping after it. + */ +export function useComposerAnchor(anchorElement: HTMLElement | null): ComposerAnchor { + const [anchor, setAnchor] = useState(UNMEASURED_ANCHOR); + + useEffect(() => { + if (anchorElement === null) { + return; + } + const composer = document.querySelector(COMPOSER_OVERLAY_SELECTOR); + if (!(composer instanceof HTMLElement)) { + setAnchor(UNMEASURED_ANCHOR); + return; + } + + const measure = () => { + const next = resolveComposerAnchor({ + composer: toAnchorRect(composer), + composerParent: toAnchorRect(composer.offsetParent), + anchorParent: toAnchorRect(anchorElement.offsetParent), + }); + // Identity-stable when nothing moved: a drag-resize of the drawer would otherwise re-render + // the overlay on every frame for no visible change. + setAnchor((previous) => (sameAnchor(previous, next) ? previous : next)); + }; + + measure(); + const observer = new ResizeObserver(measure); + observer.observe(composer); + if (composer.offsetParent instanceof HTMLElement) { + observer.observe(composer.offsetParent); + } + if (anchorElement.offsetParent instanceof HTMLElement) { + observer.observe(anchorElement.offsetParent); + } + window.addEventListener("resize", measure); + return () => { + observer.disconnect(); + window.removeEventListener("resize", measure); + }; + }, [anchorElement]); + + return anchor; +} diff --git a/apps/web/src/coil/loop/LoopArmForm.tsx b/apps/web/src/coil/loop/LoopArmForm.tsx new file mode 100644 index 000000000000..ac36c38bf3f7 --- /dev/null +++ b/apps/web/src/coil/loop/LoopArmForm.tsx @@ -0,0 +1,173 @@ +/** + * Arm, re-arm, edit and disarm — the four writes the console makes. + * + * **No validation lives here.** The bounds are the server's: `maxCheckIns` outside 1..20 and a + * deadline in the past are 400s with distinct codes, deliberately never clamped, because a silent + * clamp turns a typo into an overnight bill and hides it. This form therefore submits what was + * typed and renders the refusal it gets back, code and all. Duplicating the rules client-side + * would make the browser and the server two sources of truth for a money-spending bound. + * + * The deadline is picked as a wall-clock time in the reader's own timezone and sent as an absolute + * instant; the server never stores anything else. + * + * @module coil/loop/LoopArmForm + */ + +import { useState } from "react"; + +import { Button } from "~/components/ui/button"; +import { Input } from "~/components/ui/input"; + +import { LOOP_MAX_CHECK_INS, type LoopView, type LoopWriteBody } from "./loopClient"; +import { fromDateTimeLocalValue, seedArmDraft, toDateTimeLocalValue } from "./loopPresentation"; + +function Field({ + label, + hint, + children, +}: { + readonly label: string; + readonly hint?: string; + readonly children: React.ReactNode; +}) { + return ( + + ); +} + +export interface LoopArmFormProps { + readonly view: LoopView; + readonly defaultMaxCheckIns: number; + readonly defaultRunMs: number; + readonly busy: boolean; + readonly onSubmit: (body: Omit) => void; +} + +export function LoopArmForm({ + view, + defaultMaxCheckIns, + defaultRunMs, + busy, + onSubmit, +}: LoopArmFormProps) { + const armed = view.record.armed; + const stopped = view.record.stopped !== null; + // Seeded once. The caller keys this component on the threadId, so switching threads remounts it + // with fresh values while a 30s poll landing mid-edit can never rewrite what is being typed. + const [draft, setDraft] = useState(() => + seedArmDraft({ + settings: { + enabled: true, + maxArmedThreads: 0, + defaultMaxCheckIns, + defaultRunMs, + defaultIdleMs: 0, + defaultBusyIdleMs: 0, + armedCount: 0, + }, + record: view.record, + nowMs: Date.now(), + }), + ); + + const action = armed ? "edit" : stopped ? "rearm" : "arm"; + const submitLabel = armed ? "Save bounds" : stopped ? "Give it another run" : "Arm loop"; + + return ( +
{ + event.preventDefault(); + onSubmit({ + action, + goal: draft.goal.trim() === "" ? null : draft.goal.trim(), + maxCheckIns: draft.maxCheckIns, + // Sent even when it did not change: `edit` re-validates whatever it is given, and an + // omitted deadline on an `arm` is `deadline_required`, never a default. + deadlineAtMs: draft.deadlineAtMs, + }); + }} + > + + setDraft((previous) => ({ ...previous, goal: event.target.value }))} + placeholder="Finish the auth refactor" + size="sm" + value={draft.goal} + /> + +
+ + + setDraft((previous) => ({ + ...previous, + maxCheckIns: Number.parseInt(event.target.value, 10), + })) + } + size="sm" + type="number" + value={Number.isFinite(draft.maxCheckIns) ? String(draft.maxCheckIns) : ""} + /> + + + + setDraft((previous) => ({ + ...previous, + deadlineAtMs: fromDateTimeLocalValue(event.target.value) ?? Number.NaN, + })) + } + size="sm" + type="datetime-local" + value={ + Number.isFinite(draft.deadlineAtMs) ? toDateTimeLocalValue(draft.deadlineAtMs) : "" + } + /> + +
+
+ + {armed ? ( + // The way out, always present while armed. A one-way door is a bug. + + ) : null} + {stopped ? ( + // The other way out. Without it a finished run's pill and bounds sit above the + // composer forever, and the only way to dismiss them is to start another run — + // which is the one-way door pointing the other way. + + ) : null} +
+
+ ); +} diff --git a/apps/web/src/coil/loop/LoopConsole.tsx b/apps/web/src/coil/loop/LoopConsole.tsx new file mode 100644 index 000000000000..6f8f8488b4fd --- /dev/null +++ b/apps/web/src/coil/loop/LoopConsole.tsx @@ -0,0 +1,367 @@ +/** + * The loop console — the standing answer to "what do you need from me?". + * + * An **overlay on the thread route**, not a second default view: the transcript stays what you + * land on. That is a declared divergence from the original ask (PLAN §3, FINDINGS §F1) and the + * reason is that a second default view has nothing to toggle back from — it needs a sticky + * per-thread toggle that survives reload, agrees across two windows, and is discoverable when it + * is wrong, and choosing the view means owning the thread route's render decision, which is + * `ChatView.tsx` territory rather than the delta-zero overlay row. + * + * Sits above the docked composer, mirroring its box through the shared `useComposerAnchor`, and + * left-aligned so it never collides with the auto-resume capsule on the right. + * + * ## Two rules the rendering obeys + * + * **No spinner, and no repainting clock.** Deadlines are absolute (`ends 07:00`), ages are + * computed once per poll rather than once per second, and the panel says when it last heard from + * the server instead of pretending to be busy. A spinner over a run that has not moved in eight + * hours is a lying one, and a per-second repaint on a high-refresh display is a GPU cost for + * nothing. + * + * **`spent` is never green.** The tone comes from `describeLoopState`, where the distinction + * between "it finished" and "it ran out of rope" is pinned by tests. + * + * @module coil/loop/LoopConsole + */ + +import { ChevronDownIcon } from "lucide-react"; +import { useCallback, useEffect, useMemo, useState } from "react"; + +import { Collapsible, CollapsiblePanel, CollapsibleTrigger } from "~/components/ui/collapsible"; +import { usePrimarySettings, usePrimarySettingsAvailable } from "~/hooks/useSettings"; +import { cn } from "~/lib/utils"; +import { usePrimaryEnvironmentId } from "~/state/environments"; + +import { useComposerAnchor } from "../composerAnchor"; +import { LoopArmForm } from "./LoopArmForm"; +import { LoopQuestions } from "./LoopQuestions"; +import type { LoopSettings, LoopView, LoopWriteBody } from "./loopClient"; +import { httpLoopClient } from "./loopClient"; +import type { LoopTone } from "./loopPresentation"; +import { + canRenderLoopConsole, + countWaiting, + describeCheckInRow, + describeEmptyState, + describeLoopState, + describeRefusal, + formatAge, + formatClock, + hasLoop, + hasQuestionSections, + summariseBounds, +} from "./loopPresentation"; +import { useLoopPolling } from "./useLoopPolling"; + +const TONE_DOT: Readonly> = { + muted: "bg-muted-foreground/50", + active: "bg-sky-500", + attention: "bg-primary", + held: "bg-amber-500", + done: "bg-emerald-500", + // Zinc. Never emerald — an exhausted run must not read as a finished one. + spent: "bg-zinc-400", +}; + +const TONE_TEXT: Readonly> = { + muted: "text-muted-foreground", + active: "text-foreground", + attention: "text-foreground", + held: "text-amber-600 dark:text-amber-400", + done: "text-emerald-600 dark:text-emerald-400", + spent: "text-zinc-500 dark:text-zinc-400", +}; + +const DEGRADED_COPY: Readonly> = { + gate_off: + "The agent's own scheduler reported itself off, so T3 is the only thing waking this thread.", + wake_lost: + "A wake the agent scheduled never landed — T3 covered it. That is the gap this loop exists for.", +}; + +function Fact({ label, value }: { readonly label: string; readonly value: string }) { + return ( +
+
{label}
+
{value}
+
+ ); +} + +export interface LoopConsoleProps { + readonly threadRef: { readonly environmentId: string; readonly threadId: string }; +} + +export function LoopConsole({ threadRef }: LoopConsoleProps) { + const threadId = threadRef.threadId; + const [expanded, setExpanded] = useState(false); + const [anchorElement, setAnchorElement] = useState(null); + const anchor = useComposerAnchor(anchorElement); + const [refusal, setRefusal] = useState<{ code: string; message: string } | null>(null); + const [busyBlockerId, setBusyBlockerId] = useState(null); + const [writing, setWriting] = useState(false); + + const browserAccessKnown = usePrimarySettingsAvailable(); + const browserAccessEnabled = usePrimarySettings((settings) => settings.enableAgentBrowserAccess); + + // Every fork route is called against the primary environment, so on a thread that belongs + // to another one the console would answer for a thread id that server has never heard of. + const primaryEnvironmentId = usePrimaryEnvironmentId(); + const consoleTargetsThisThread = canRenderLoopConsole({ + primaryEnvironmentId, + threadEnvironmentId: threadRef.environmentId, + }); + + const loadView = useCallback(() => httpLoopClient.read(threadId), [threadId]); + const loop = useLoopPolling(threadId, loadView); + const loadSettings = useCallback(() => httpLoopClient.readSettings(), []); + const settings = useLoopPolling("global", loadSettings); + + useEffect(() => { + setExpanded(false); + setRefusal(null); + }, [threadId]); + + const applyResult = useCallback( + (result: Awaited>) => { + if (result === null) return; + if (result.ok) { + setRefusal(null); + loop.set(result.value); + return; + } + setRefusal(describeRefusal(result.code)); + }, + [loop], + ); + + const handleWrite = useCallback( + (body: Omit) => { + setWriting(true); + void httpLoopClient.write({ threadId, ...body }).then( + (result) => { + setWriting(false); + applyResult(result); + }, + () => setWriting(false), + ); + }, + [applyResult, threadId], + ); + + const handleAnswer = useCallback( + (blockerId: string, answer: string) => { + setBusyBlockerId(blockerId); + void httpLoopClient.answer({ threadId, blockerId, answer }).then( + (result) => { + setBusyBlockerId(null); + if (result === null) return; + if (result.ok) { + setRefusal(null); + // The answer route returns `{ ok: true }` and nothing else, so re-read rather than + // guessing what the record now looks like. + loop.refresh(); + return; + } + setRefusal(describeRefusal(result.code)); + }, + () => setBusyBlockerId(null), + ); + }, + [loop, threadId], + ); + + const view = loop.value; + // One clock read per load, not one per second. Every age in the panel is relative to it. + const nowMs = loop.lastLoadedAtMs ?? 0; + const state = useMemo( + () => (view === null ? null : describeLoopState(view.derived, nowMs)), + [nowMs, view], + ); + const waiting = view === null ? 0 : countWaiting(view); + + if (view === null || state === null || !consoleTargetsThisThread) { + return null; + } + + const empty = describeEmptyState(view, nowMs); + const hasQuestions = hasQuestionSections(view, { browserAccessKnown, browserAccessEnabled }); + + return ( +
+
+ + + + + +
+
+

{view.record.goal ?? "Loop"}

+ {/* An honest staleness label, not a spinner. */} + + {loop.lastLoadedAtMs === null + ? "" + : `Updated ${formatClock(loop.lastLoadedAtMs)}`} + +
+

{state.detail}

+ + {refusal === null ? null : ( +
+

{refusal.message}

+ {/* The server's own code, carried verbatim: it is what makes an unrecognised + refusal reportable rather than a blank failure. */} +

+ {refusal.code} +

+
+ )} + + {hasQuestions ? ( + + ) : ( +
+

{empty.headline}

+ {empty.lines.map((line) => ( +

+ {line} +

+ ))} +
+ )} + + {hasLoop(view) ? ( + <> +
+ + 0 + ? formatClock(view.derived.deadlineAtMs, nowMs) + : "not set" + } + /> + + +
+ {view.record.degraded === null ? null : ( +

+ {DEGRADED_COPY[view.record.degraded]} +

+ )} + + ) : null} + + {view.record.checkIns.length === 0 ? null : ( +
+
+

Check-ins

+ + what the loop did each time it woke + +
+
    + {view.record.checkIns.toReversed().map((row) => ( +
  1. + + #{row.n} + + {describeCheckInRow(row)} + + {formatAge(row.firedAtMs, nowMs)} + +
  2. + ))} +
+
+ )} + + + + {view.derived.globalEnabled ? null : ( +

+ Loops are switched off in Settings → Loops. Arming still works; nothing will fire + until the master switch is on. +

+ )} +
+
+
+
+
+ ); +} diff --git a/apps/web/src/coil/loop/LoopQuestions.tsx b/apps/web/src/coil/loop/LoopQuestions.tsx new file mode 100644 index 000000000000..d71acb18e343 --- /dev/null +++ b/apps/web/src/coil/loop/LoopQuestions.tsx @@ -0,0 +1,335 @@ +/** + * The two question sections of the loop console: what stopped the loop, and what it worked + * around. + * + * ## Why the blocking section does not mount the native card + * + * A native `AskUserQuestion` is already rendered, live and answerable, by + * `ComposerPendingUserInputPanel` inside the composer — roughly fifty pixels below where this + * overlay sits. Mounting a second instance here would render the identical card twice on one + * screen, register a **second** document-level `1`–`9` keydown handler (so every digit key would + * answer twice), and require ChatView-local draft state that the overlay could only reach by + * widening `ChatView.tsx`, an existing seam row at churn 114. + * + * So this section **names** what is blocking and sends you to the control that already exists, + * rather than duplicating it. The answer path is upstream's, unchanged, with one instance of it on + * the page — which is also why there is no native half of `POST /api/coil/loop/answer`. + * + * The question text comes from the fork's own `userInputs` ledger rather than from the shell, + * because that ledger is the only place a **voided** question is visible at all: upstream settles + * pending inputs as empty answers during session teardown, after which `hasPendingUserInput` reads + * false and a question nobody ever saw is indistinguishable from an answered one. + * + * @module coil/loop/LoopQuestions + */ + +import { AlertTriangleIcon, ArrowDownIcon } from "lucide-react"; +import { useState } from "react"; + +import { Button } from "~/components/ui/button"; +import { Input } from "~/components/ui/input"; +import { cn } from "~/lib/utils"; + +import type { LoopBlocker, LoopUserInput, LoopView } from "./loopClient"; +import { + describeBlockingHint, + formatAge, + hasLoop, + partitionBlockers, + partitionUserInputs, + resolveDeferredChannelNotice, +} from "./loopPresentation"; + +function SectionHeading({ title, why }: { readonly title: string; readonly why: string }) { + return ( +
+

{title}

+ {why} +
+ ); +} + +/** + * One boxed card. + * + * `bg-card` plus a border and a text label, deliberately with **no coloured leading rail**: the + * rail in the prototype read as a status bar on a list rather than as a card, and it put four + * hues on one small panel. + */ +function QuestionCard({ + label, + labelClassName, + at, + nowMs, + children, +}: { + readonly label: string; + readonly labelClassName?: string; + readonly at: number; + readonly nowMs: number; + readonly children: React.ReactNode; +}) { + return ( +
+
+ + {label} + + + {formatAge(at, nowMs)} + +
+ {children} +
+ ); +} + +function BlockingSection({ view, nowMs }: { readonly view: LoopView; readonly nowMs: number }) { + const { open } = partitionUserInputs(view.record.userInputs); + const hint = describeBlockingHint(view.derived); + if (open.length === 0 && hint === null) return null; + + return ( +
+ + {open.map((entry) => ( + +

{entry.question}

+
+ ))} + {hint === null ? null : ( +

+ {/* Pointing at the live control rather than cloning it — see the module note. The + arrow only makes sense when the thing to do IS below; a snooze is undone + elsewhere, so the copy says so and the arrow goes. */} + {view.derived.reason === "snoozed" ? null : ( +

+ )} +
+ ); +} + +function BlockerAnswerForm({ + blocker, + onAnswer, + busy, +}: { + readonly blocker: LoopBlocker; + readonly onAnswer: (answer: string) => void; + readonly busy: boolean; +}) { + const [text, setText] = useState(""); + return ( +
+ {blocker.options.map((option) => ( + + ))} +
{ + event.preventDefault(); + if (text.trim() === "") return; + onAnswer(text.trim()); + setText(""); + }} + > + setText(event.target.value)} + placeholder={blocker.options.length === 0 ? "Answer" : "…or in your own words"} + size="sm" + value={text} + /> + +
+
+ ); +} + +/** + * The deferred channel. + * + * Three states, kept visibly distinct because they are different facts: **open** (nobody has + * answered), **banked** (you answered, but the agent is idle and will not hear it until the next + * check-in prompt), and **delivered** (the agent has been told). + */ +function DeferredSection({ + view, + nowMs, + browserAccessKnown, + browserAccessEnabled, + busyBlockerId, + onAnswer, +}: { + readonly view: LoopView; + readonly nowMs: number; + readonly browserAccessKnown: boolean; + readonly browserAccessEnabled: boolean; + readonly busyBlockerId: string | null; + readonly onAnswer: (blockerId: string, answer: string) => void; +}) { + const { open, banked, delivered } = partitionBlockers(view.record.blockers); + // `loopExists` is what keeps this off every thread in the app: with no loop here nothing + // would have used the channel, so the warning is noise rather than a fact about this thread. + const notice = resolveDeferredChannelNotice({ + browserAccessKnown, + browserAccessEnabled, + loopExists: hasLoop(view), + }); + if (notice === null && open.length === 0 && banked.length === 0 && delivered.length === 0) { + return null; + } + + return ( +
+ + {notice === null ? null : ( +
+
+ )} + {open.map((blocker) => ( + +

{blocker.question}

+ {blocker.context === null ? null : ( +

{blocker.context}

+ )} + onAnswer(blocker.id, answer)} + /> +
+ ))} + {[...banked, ...delivered].map((blocker) => ( + +

+ {blocker.question} +

+

{blocker.answer}

+ {blocker.deliveredToAgent ? null : ( +

+ Banked. The next check-in restates it to the agent. +

+ )} +
+ ))} +
+ ); +} + +/** + * Questions the runtime raised that nobody ever answered because the session was torn down. + * + * Rendered separately from "answered" on purpose: a voided question is a question the human never + * saw, and folding it into the answered pile hides the only evidence it existed. + */ +function VoidedSection({ + voided, + nowMs, +}: { + readonly voided: ReadonlyArray; + readonly nowMs: number; +}) { + if (voided.length === 0) return null; + return ( +
+ + {voided.map((entry) => ( + +

+ {entry.question} +

+

+ Closed by the session ending, not by an answer. +

+
+ ))} +
+ ); +} + +export interface LoopQuestionsProps { + readonly view: LoopView; + readonly nowMs: number; + readonly browserAccessKnown: boolean; + readonly browserAccessEnabled: boolean; + readonly busyBlockerId: string | null; + readonly onAnswer: (blockerId: string, answer: string) => void; +} + +export function LoopQuestions(props: LoopQuestionsProps) { + const { voided } = partitionUserInputs(props.view.record.userInputs); + return ( + <> + + + + + ); +} diff --git a/apps/web/src/coil/loop/loopClient.ts b/apps/web/src/coil/loop/loopClient.ts new file mode 100644 index 000000000000..a6815d709077 --- /dev/null +++ b/apps/web/src/coil/loop/loopClient.ts @@ -0,0 +1,469 @@ +/** + * Transport + wire parsing for the loop console and the Loops settings panel. + * + * Mirrors `coil/autoResumeClient.ts` exactly, for the reasons stated there: the routes are raw + * (`/api/coil/loop*`), so they are called with `resolvePrimaryEnvironmentHttpUrl` over + * `primaryEnvironmentHttpLayer` — the only place in the web app that knows how to authenticate the + * primary environment, and therefore the only call shape that survives remote, relay and tunnel. + * **Never hardcode an origin here.** + * + * Auth is ambient: the environment HTTP layer attaches the credential and the server states the + * 403. There is deliberately **no client-side scope check** in this file or its callers. + * + * The parsers are hand-rolled rather than schema-decoded because the schemas live in + * `apps/server/src/coil/loop/state.ts`, which the web app cannot import. Every field the server can + * omit or malform therefore falls back to the same value the server's own decoding default uses, + * so a partial response degrades to "off" rather than to a crash. `http.test.ts` case 85 is the + * other half of this contract: it asserts every response decodes against its schema, so a drift + * shows up on the server side rather than as a silently empty console. + * + * @module coil/loop/loopClient + */ + +import * as Effect from "effect/Effect"; +import * as ManagedRuntime from "effect/ManagedRuntime"; +import { HttpClient, HttpClientRequest } from "effect/unstable/http"; + +import { primaryEnvironmentHttpLayer } from "~/environments/primary/httpLayer"; +import { resolvePrimaryEnvironmentHttpUrl } from "~/environments/primary/target"; + +export const LOOP_PATH = "/api/coil/loop"; +export const LOOPS_PATH = "/api/coil/loops"; +export const LOOP_SETTINGS_PATH = "/api/coil/loop/settings"; +export const LOOP_ANSWER_PATH = "/api/coil/loop/answer"; + +/** The hard ceiling the arm route enforces. Mirrored here only to word the form's help text. */ +export const LOOP_MAX_CHECK_INS = 20; + +export type LoopState = + | "off" + | "watching" + | "self_pacing" + | "standing_down" + | "held" + | "blocked" + | "stopped"; + +export type LoopStopReason = "done" | "spent" | "stalled" | "handed-back"; + +export interface LoopBlockerOption { + readonly label: string; + readonly description: string; +} + +export interface LoopBlocker { + readonly id: string; + readonly raisedAtMs: number; + readonly question: string; + readonly options: ReadonlyArray; + readonly context: string | null; + readonly answeredAtMs: number | null; + readonly answer: string | null; + /** True once the answer has been restated to the agent in a check-in prompt. */ + readonly deliveredToAgent: boolean; +} + +/** + * A blocking question the runtime raised, as the fork recorded it. + * + * This is the only record that survives upstream settling a pending input as an empty answer + * during session teardown, which is what makes `voided` visible at all. + */ +export interface LoopUserInput { + readonly requestId: string; + readonly raisedAtMs: number; + readonly dialogKind: string | null; + readonly question: string; + readonly resolution: "answered" | "voided" | null; + readonly resolvedAtMs: number | null; +} + +export interface LoopCheckInRow { + readonly n: number; + readonly firedAtMs: number; + readonly createdAtIso: string; + readonly activityCursor: string; + readonly outcome: "productive" | "unproductive" | "unknown"; +} + +export interface LoopStopRecord { + readonly reason: LoopStopReason; + readonly atMs: number; + readonly detail: string; +} + +export interface LoopRecord { + readonly armed: boolean; + readonly armedAtMs: number; + readonly goal: string | null; + readonly maxCheckIns: number; + readonly checkInsUsed: number; + readonly deadlineAtMs: number; + readonly idleMs: number; + readonly busyIdleMs: number; + readonly degraded: "gate_off" | "wake_lost" | null; + readonly userInputs: ReadonlyArray; + readonly checkIns: ReadonlyArray; + readonly strikes: number; + readonly rateLimitedUntilMs: number; + readonly stopped: LoopStopRecord | null; + readonly overridePrompt: string | null; + readonly blockers: ReadonlyArray; +} + +export interface LoopDerived { + readonly state: LoopState; + readonly reason: string | null; + readonly stoppedReason: LoopStopReason | null; + readonly checkInsUsed: number; + readonly maxCheckIns: number; + readonly deadlineAtMs: number; + readonly msUntilDeadline: number; + readonly rateLimitedUntilMs: number; + readonly nextWakeAtMs: number | null; + readonly snoozedUntilMs: number | null; + readonly threadKnown: boolean; + readonly globalEnabled: boolean; + readonly armedCount: number; + readonly maxArmedThreads: number; +} + +export interface LoopView { + readonly threadId: string; + readonly record: LoopRecord; + readonly derived: LoopDerived; + /** The *unanswered* blockers — what is actionable now. */ + readonly blockers: ReadonlyArray; + readonly ledger: ReadonlyArray; +} + +export interface LoopSettings { + readonly enabled: boolean; + readonly maxArmedThreads: number; + readonly defaultMaxCheckIns: number; + readonly defaultRunMs: number; + readonly defaultIdleMs: number; + readonly defaultBusyIdleMs: number; + readonly armedCount: number; +} + +export interface LoopWriteBody { + readonly threadId: string; + /** `clear` forgets a finished run — the reverse of arming, refused while one is armed. */ + readonly action: "arm" | "rearm" | "edit" | "disarm" | "clear"; + readonly maxCheckIns?: number; + readonly deadlineAtMs?: number; + readonly goal?: string | null; + readonly idleMs?: number; + readonly busyIdleMs?: number; + readonly overridePrompt?: string | null; +} + +export interface LoopAnswerBody { + readonly threadId: string; + readonly blockerId: string; + readonly answer: string; +} + +/** + * A refusal the console must word, kept as the server's own code. + * + * The route gives every 400 a distinct code precisely so "you must pick an end time" and "that end + * time has already passed" are different sentences. Collapsing them into a generic failure here + * would throw that away, so the code travels to the UI unchanged. + */ +export interface LoopWriteRefused { + readonly ok: false; + readonly code: string; + readonly status: number; +} + +export type LoopWriteResult
= { readonly ok: true; readonly value: A } | LoopWriteRefused; + +export interface LoopClient { + readonly read: (threadId: string) => Promise; + readonly write: (body: LoopWriteBody) => Promise | null>; + readonly answer: (body: LoopAnswerBody) => Promise | null>; + readonly readSettings: () => Promise; + readonly writeSettings: ( + patch: Partial>, + ) => Promise | null>; + readonly listLoops: () => Promise | null>; +} + +// --- parsing ---------------------------------------------------------------- + +function isJsonObject(value: unknown): value is Record { + return typeof value === "object" && value !== null && !Array.isArray(value); +} + +const num = (value: unknown, fallback: number): number => + typeof value === "number" && Number.isFinite(value) ? value : fallback; + +const nullableNum = (value: unknown): number | null => + typeof value === "number" && Number.isFinite(value) ? value : null; + +const str = (value: unknown, fallback = ""): string => + typeof value === "string" ? value : fallback; + +const nullableStr = (value: unknown): string | null => + typeof value === "string" && value !== "" ? value : null; + +const bool = (value: unknown, fallback: boolean): boolean => + typeof value === "boolean" ? value : fallback; + +const array = (value: unknown): ReadonlyArray => (Array.isArray(value) ? value : []); + +const literal = (value: unknown, allowed: ReadonlyArray, fallback: T): T => + typeof value === "string" && (allowed as ReadonlyArray).includes(value) + ? (value as T) + : fallback; + +const STOP_REASONS = ["done", "spent", "stalled", "handed-back"] as const; +const LOOP_STATES = [ + "off", + "watching", + "self_pacing", + "standing_down", + "held", + "blocked", + "stopped", +] as const; + +function parseBlocker(value: unknown): LoopBlocker | null { + if (!isJsonObject(value)) return null; + const id = str(value.id); + if (id === "") return null; + return { + id, + raisedAtMs: num(value.raisedAtMs, 0), + question: str(value.question), + options: array(value.options).flatMap((option) => + isJsonObject(option) + ? [{ label: str(option.label), description: str(option.description) }] + : [], + ), + context: nullableStr(value.context), + answeredAtMs: nullableNum(value.answeredAtMs), + answer: nullableStr(value.answer), + deliveredToAgent: bool(value.deliveredToAgent, false), + }; +} + +function parseUserInput(value: unknown): LoopUserInput | null { + if (!isJsonObject(value)) return null; + const requestId = str(value.requestId); + if (requestId === "") return null; + const resolution = value.resolution; + return { + requestId, + raisedAtMs: num(value.raisedAtMs, 0), + dialogKind: nullableStr(value.dialogKind), + question: str(value.question), + resolution: + resolution === "answered" || resolution === "voided" ? (resolution as "answered") : null, + resolvedAtMs: nullableNum(value.resolvedAtMs), + }; +} + +function parseCheckInRow(value: unknown): LoopCheckInRow | null { + if (!isJsonObject(value)) return null; + return { + n: num(value.n, 0), + firedAtMs: num(value.firedAtMs, 0), + createdAtIso: str(value.createdAtIso), + activityCursor: str(value.activityCursor), + outcome: literal(value.outcome, ["productive", "unproductive", "unknown"] as const, "unknown"), + }; +} + +function parseStop(value: unknown): LoopStopRecord | null { + if (!isJsonObject(value)) return null; + return { + reason: literal(value.reason, STOP_REASONS, "spent"), + atMs: num(value.atMs, 0), + detail: str(value.detail), + }; +} + +/** Every fallback here is the server's own fail-closed decoding default. */ +export function parseLoopRecord(value: unknown): LoopRecord { + const raw = isJsonObject(value) ? value : {}; + const degraded = raw.degraded; + return { + armed: bool(raw.armed, false), + armedAtMs: num(raw.armedAtMs, 0), + goal: nullableStr(raw.goal), + maxCheckIns: num(raw.maxCheckIns, 0), + checkInsUsed: num(raw.checkInsUsed, 0), + deadlineAtMs: num(raw.deadlineAtMs, 0), + idleMs: num(raw.idleMs, 15 * 60_000), + busyIdleMs: num(raw.busyIdleMs, 45 * 60_000), + degraded: degraded === "gate_off" || degraded === "wake_lost" ? degraded : null, + userInputs: array(raw.userInputs).flatMap((entry) => { + const parsed = parseUserInput(entry); + return parsed === null ? [] : [parsed]; + }), + checkIns: array(raw.checkIns).flatMap((entry) => { + const parsed = parseCheckInRow(entry); + return parsed === null ? [] : [parsed]; + }), + strikes: num(raw.strikes, 0), + rateLimitedUntilMs: num(raw.rateLimitedUntilMs, 0), + stopped: parseStop(raw.stopped), + overridePrompt: nullableStr(raw.overridePrompt), + blockers: array(raw.blockers).flatMap((entry) => { + const parsed = parseBlocker(entry); + return parsed === null ? [] : [parsed]; + }), + }; +} + +export function parseLoopView(value: unknown): LoopView | null { + if (!isJsonObject(value)) return null; + const threadId = str(value.threadId); + if (threadId === "") return null; + const record = parseLoopRecord(value.record); + const rawDerived = isJsonObject(value.derived) ? value.derived : {}; + const derived: LoopDerived = { + state: literal(rawDerived.state, LOOP_STATES, record.stopped === null ? "off" : "stopped"), + reason: nullableStr(rawDerived.reason), + stoppedReason: + typeof rawDerived.stoppedReason === "string" + ? literal(rawDerived.stoppedReason, STOP_REASONS, "spent") + : null, + checkInsUsed: num(rawDerived.checkInsUsed, record.checkInsUsed), + maxCheckIns: num(rawDerived.maxCheckIns, record.maxCheckIns), + deadlineAtMs: num(rawDerived.deadlineAtMs, record.deadlineAtMs), + msUntilDeadline: num(rawDerived.msUntilDeadline, 0), + rateLimitedUntilMs: num(rawDerived.rateLimitedUntilMs, record.rateLimitedUntilMs), + nextWakeAtMs: nullableNum(rawDerived.nextWakeAtMs), + snoozedUntilMs: nullableNum(rawDerived.snoozedUntilMs), + threadKnown: bool(rawDerived.threadKnown, false), + globalEnabled: bool(rawDerived.globalEnabled, false), + armedCount: num(rawDerived.armedCount, 0), + maxArmedThreads: num(rawDerived.maxArmedThreads, 0), + }; + return { + threadId, + record, + derived, + blockers: array(value.blockers).flatMap((entry) => { + const parsed = parseBlocker(entry); + return parsed === null ? [] : [parsed]; + }), + ledger: array(value.ledger).flatMap((entry) => { + const parsed = parseCheckInRow(entry); + return parsed === null ? [] : [parsed]; + }), + }; +} + +export function parseLoopSettings(value: unknown): LoopSettings | null { + if (!isJsonObject(value)) return null; + if (typeof value.enabled !== "boolean") return null; + return { + enabled: value.enabled, + maxArmedThreads: num(value.maxArmedThreads, 3), + defaultMaxCheckIns: num(value.defaultMaxCheckIns, 6), + defaultRunMs: num(value.defaultRunMs, 8 * 3_600_000), + defaultIdleMs: num(value.defaultIdleMs, 15 * 60_000), + defaultBusyIdleMs: num(value.defaultBusyIdleMs, 45 * 60_000), + armedCount: num(value.armedCount, 0), + }; +} + +/** The error code out of a 400 body, or a stand-in so a refusal is never silently blank. */ +function parseRefusal(status: number, body: unknown): LoopWriteRefused { + const code = isJsonObject(body) ? str(body.error, "") : ""; + return { ok: false, status, code: code === "" ? `http_${status}` : code }; +} + +// --- transport -------------------------------------------------------------- + +const loopRuntime = ManagedRuntime.make(primaryEnvironmentHttpLayer); + +/** + * Runs a request, swallowing **every** failure to `null`. + * + * The console is layered over the thread view, so a 401, an undeployed route or an offline client + * must make it disappear rather than degrade chat. `null` means "we do not know"; it is never + * rendered as "there is no loop". + */ +async function run( + effect: Effect.Effect, +): Promise { + try { + return await loopRuntime.runPromise(effect); + } catch { + return null; + } +} + +const postJson = (url: string, body: unknown, parse: (value: unknown) => A | null) => + Effect.gen(function* () { + const response = yield* HttpClient.execute( + HttpClientRequest.bodyJsonUnsafe(HttpClientRequest.post(url), body), + ); + if (response.status !== 200) { + return parseRefusal(response.status, yield* Effect.orElseSucceed(response.json, () => null)); + } + const value = parse(yield* response.json); + return value === null ? null : ({ ok: true, value } as const); + }); + +export const httpLoopClient: LoopClient = { + read: (threadId) => + run( + Effect.gen(function* () { + const response = yield* HttpClient.get( + resolvePrimaryEnvironmentHttpUrl(LOOP_PATH, { threadId }), + ); + if (response.status !== 200) return null; + return parseLoopView(yield* response.json); + }), + ), + + write: (body) => run(postJson(resolvePrimaryEnvironmentHttpUrl(LOOP_PATH), body, parseLoopView)), + + answer: (body) => + run( + postJson(resolvePrimaryEnvironmentHttpUrl(LOOP_ANSWER_PATH), body, (value) => + isJsonObject(value) && value.ok === true ? null : null, + ).pipe( + // The route's success body is `{ ok: true }` and carries nothing the console needs, so a + // 200 resolves to `{ ok: true, value: null }` and the caller re-reads the view. + Effect.map((result) => + result === null || result.ok === false ? result : { ok: true as const, value: null }, + ), + ), + ), + + readSettings: () => + run( + Effect.gen(function* () { + const response = yield* HttpClient.get( + resolvePrimaryEnvironmentHttpUrl(LOOP_SETTINGS_PATH), + ); + if (response.status !== 200) return null; + return parseLoopSettings(yield* response.json); + }), + ), + + writeSettings: (patch) => + run(postJson(resolvePrimaryEnvironmentHttpUrl(LOOP_SETTINGS_PATH), patch, parseLoopSettings)), + + listLoops: () => + run( + Effect.gen(function* () { + const response = yield* HttpClient.get(resolvePrimaryEnvironmentHttpUrl(LOOPS_PATH)); + if (response.status !== 200) return null; + const body = yield* response.json; + if (!isJsonObject(body)) return null; + return array(body.loops).flatMap((entry) => { + const parsed = parseLoopView(entry); + return parsed === null ? [] : [parsed]; + }); + }), + ), +}; diff --git a/apps/web/src/coil/loop/loopPresentation.test.ts b/apps/web/src/coil/loop/loopPresentation.test.ts new file mode 100644 index 000000000000..3364c8c160b6 --- /dev/null +++ b/apps/web/src/coil/loop/loopPresentation.test.ts @@ -0,0 +1,541 @@ +import { describe, expect, it } from "vite-plus/test"; + +import type { LoopBlocker, LoopDerived, LoopRecord, LoopUserInput, LoopView } from "./loopClient"; +import { + canRenderLoopConsole, + countWaiting, + describeBlockingHint, + describeCheckInRow, + describeEmptyState, + describeLoopState, + describeRefusal, + describeStop, + formatAge, + formatClock, + formatDuration, + fromDateTimeLocalValue, + hasQuestionSections, + partitionBlockers, + partitionUserInputs, + resolveDeferredChannelNotice, + seedArmDraft, + summariseBounds, + toDateTimeLocalValue, +} from "./loopPresentation"; + +const NOW_MS = Date.UTC(2026, 8, 2, 9, 4, 0); +const DEADLINE_MS = Date.UTC(2026, 8, 2, 7, 0, 0); + +const record = (overrides: Partial = {}): LoopRecord => ({ + armed: true, + armedAtMs: NOW_MS - 10 * 3_600_000, + goal: null, + maxCheckIns: 6, + checkInsUsed: 2, + deadlineAtMs: DEADLINE_MS, + idleMs: 15 * 60_000, + busyIdleMs: 45 * 60_000, + degraded: null, + userInputs: [], + checkIns: [], + strikes: 0, + rateLimitedUntilMs: 0, + stopped: null, + overridePrompt: null, + blockers: [], + ...overrides, +}); + +const derived = (overrides: Partial = {}): LoopDerived => ({ + state: "watching", + reason: null, + stoppedReason: null, + checkInsUsed: 2, + maxCheckIns: 6, + deadlineAtMs: DEADLINE_MS, + msUntilDeadline: 0, + rateLimitedUntilMs: 0, + nextWakeAtMs: null, + snoozedUntilMs: null, + threadKnown: true, + globalEnabled: true, + armedCount: 1, + maxArmedThreads: 3, + ...overrides, +}); + +const view = (overrides: Partial = {}): LoopView => ({ + threadId: "thread-a", + record: record(), + derived: derived(), + blockers: [], + ledger: [], + ...overrides, +}); + +const blocker = (overrides: Partial = {}): LoopBlocker => ({ + id: "b1", + raisedAtMs: NOW_MS - 3_600_000, + question: "Migrate in place or backfill?", + options: [], + context: null, + answeredAtMs: null, + answer: null, + deliveredToAgent: false, + ...overrides, +}); + +const userInput = (overrides: Partial = {}): LoopUserInput => ({ + requestId: "r1", + raisedAtMs: NOW_MS - 3_600_000, + dialogKind: null, + question: "Which table?", + resolution: null, + resolvedAtMs: null, + ...overrides, +}); + +describe("describeStop", () => { + // The failure this whole feature exists to stop: a run that hit a bound reading as a success. + it("never gives spent the done tone", () => { + expect(describeStop("done").tone).toBe("done"); + expect(describeStop("spent").tone).toBe("spent"); + expect(describeStop("spent").tone).not.toBe("done"); + expect(describeStop("stalled").tone).not.toBe("done"); + expect(describeStop("handed-back").tone).not.toBe("done"); + }); + + it("says out loud that a spent run did not finish", () => { + expect(describeStop("spent").detail).toContain("did not finish"); + }); + + it("states that handing back did not reset the budget", () => { + expect(describeStop("handed-back").detail).toContain("not reset"); + }); +}); + +describe("describeLoopState", () => { + it("reads a terminal state through its stop reason, not through the live guards", () => { + const copy = describeLoopState(derived({ state: "stopped", stoppedReason: "spent" })); + expect(copy.label).toBe("Out of rope"); + expect(copy.tone).toBe("spent"); + }); + + it("defaults a terminal with no reason to spent rather than to done", () => { + expect(describeLoopState(derived({ state: "stopped", stoppedReason: null })).tone).toBe( + "spent", + ); + }); + + it("says standing down is not a disarm", () => { + const copy = describeLoopState(derived({ state: "standing_down", reason: "disabled" })); + expect(copy.tone).toBe("muted"); + expect(copy.detail).toContain("nothing was disarmed"); + }); + + it("distinguishes held from stalled — no check-in was spent", () => { + const copy = describeLoopState( + derived({ state: "held", reason: "rate_limited", rateLimitedUntilMs: DEADLINE_MS }), + ); + expect(copy.tone).toBe("held"); + expect(copy.detail).toContain("No check-in was spent"); + }); + + it("names the agent's own wake when self-pacing", () => { + const copy = describeLoopState(derived({ state: "self_pacing", nextWakeAtMs: DEADLINE_MS })); + expect(copy.detail).toContain("stands by and spends nothing"); + }); + + it("falls back to a generic sentence for a reason it has no copy for", () => { + const copy = describeLoopState(derived({ state: "standing_down", reason: "check_in_floor" })); + expect(copy.detail).toContain("budget and deadline are intact"); + }); +}); + +describe("partitionBlockers", () => { + // Answered and "the agent knows" are different facts: the answer banks until the next check-in. + it("splits open, banked and delivered", () => { + const open = blocker({ id: "open" }); + const banked = blocker({ id: "banked", answeredAtMs: NOW_MS, answer: "yes" }); + const delivered = blocker({ + id: "delivered", + answeredAtMs: NOW_MS, + answer: "yes", + deliveredToAgent: true, + }); + const partition = partitionBlockers([open, banked, delivered]); + expect(partition.open.map((entry) => entry.id)).toEqual(["open"]); + expect(partition.banked.map((entry) => entry.id)).toEqual(["banked"]); + expect(partition.delivered.map((entry) => entry.id)).toEqual(["delivered"]); + }); +}); + +describe("partitionUserInputs", () => { + // A voided question is one nobody ever saw. Upstream leaves no other trace of it. + it("keeps voided distinct from answered", () => { + const partition = partitionUserInputs([ + userInput({ requestId: "open" }), + userInput({ requestId: "answered", resolution: "answered", resolvedAtMs: NOW_MS }), + userInput({ requestId: "voided", resolution: "voided", resolvedAtMs: NOW_MS }), + ]); + expect(partition.open.map((entry) => entry.requestId)).toEqual(["open"]); + expect(partition.answered.map((entry) => entry.requestId)).toEqual(["answered"]); + expect(partition.voided.map((entry) => entry.requestId)).toEqual(["voided"]); + }); +}); + +describe("resolveDeferredChannelNotice", () => { + it("names the missing channel when agent browser access is off", () => { + const notice = resolveDeferredChannelNotice({ + browserAccessKnown: true, + browserAccessEnabled: false, + }); + expect(notice).not.toBeNull(); + expect(notice?.detail).toContain("Settings → Integrations"); + // The whole point: an empty list must not read as "it had nothing to ask". + expect(notice?.detail).toContain("does not mean"); + }); + + it("says nothing when the channel is available", () => { + expect( + resolveDeferredChannelNotice({ browserAccessKnown: true, browserAccessEnabled: true }), + ).toBeNull(); + }); + + it("says nothing rather than guessing when the setting cannot be read", () => { + expect( + resolveDeferredChannelNotice({ browserAccessKnown: false, browserAccessEnabled: false }), + ).toBeNull(); + }); +}); + +describe("describeRefusal", () => { + it("keeps the server's code alongside the sentence", () => { + const refusal = describeRefusal("deadline_required"); + expect(refusal.code).toBe("deadline_required"); + expect(refusal.message).toContain("end time"); + }); + + it("words thread_snoozed as the unsnooze-first rule rather than a generic failure", () => { + expect(describeRefusal("thread_snoozed").message).toContain("Unsnooze"); + }); + + it("still carries an unknown code so a refusal is never blank", () => { + const refusal = describeRefusal("some_future_code"); + expect(refusal.code).toBe("some_future_code"); + expect(refusal.message).not.toBe(""); + }); +}); + +describe("describeEmptyState", () => { + // The acceptance case: spent, and the model never called raise_blocker. + it("still says what happened for a spent run with no blockers", () => { + const empty = describeEmptyState( + view({ + record: record({ + armed: false, + checkInsUsed: 6, + stopped: { reason: "spent", atMs: DEADLINE_MS, detail: "deadline reached" }, + checkIns: [ + { + n: 6, + firedAtMs: DEADLINE_MS - 3_600_000, + createdAtIso: "2026-09-02T06:00:00.000Z", + activityCursor: "c6", + outcome: "productive", + }, + ], + }), + derived: derived({ state: "stopped", stoppedReason: "spent", checkInsUsed: 6 }), + }), + NOW_MS, + ); + expect(empty.headline).toBe("Stopped. Nothing is waiting on you."); + expect(empty.lines.join(" ")).toContain("did not finish"); + expect(empty.lines.join(" ")).toContain("deadline reached"); + expect(empty.lines.join(" ")).toContain("Used 6 of 6 check-ins"); + expect(empty.lines.some((line) => line.includes("Last activity"))).toBe(true); + }); + + it("says so when a run ended without ever checking in", () => { + const empty = describeEmptyState( + view({ + record: record({ armed: false, stopped: { reason: "spent", atMs: 1, detail: "" } }), + derived: derived({ state: "stopped", stoppedReason: "spent" }), + }), + NOW_MS, + ); + expect(empty.lines.some((line) => line.includes("never checked in"))).toBe(true); + }); + + it("does not call a finished run stopped", () => { + const empty = describeEmptyState( + view({ + record: record({ armed: false, stopped: { reason: "done", atMs: 1, detail: "" } }), + derived: derived({ state: "stopped", stoppedReason: "done" }), + }), + NOW_MS, + ); + expect(empty.headline).toBe("Finished. Nothing is waiting on you."); + }); + + it("offers to arm when there is no loop at all", () => { + const empty = describeEmptyState( + view({ record: record({ armed: false }), derived: derived({ state: "off" }) }), + NOW_MS, + ); + expect(empty.headline).toBe("No loop on this thread."); + }); +}); + +describe("countWaiting", () => { + it("counts open blockers and open native questions, and nothing else", () => { + expect( + countWaiting( + view({ + record: record({ + blockers: [ + blocker({ id: "a" }), + blocker({ id: "b", answeredAtMs: NOW_MS, answer: "x" }), + ], + userInputs: [ + userInput({ requestId: "open" }), + userInput({ requestId: "voided", resolution: "voided", resolvedAtMs: NOW_MS }), + ], + }), + }), + ), + ).toBe(2); + }); +}); + +describe("summariseBounds", () => { + it("drops the deadline rather than printing the epoch when there is none", () => { + expect(summariseBounds(derived({ deadlineAtMs: 0 }))).toBe("2 of 6 check-ins"); + }); +}); + +describe("formatDuration", () => { + it("rounds down, so a bound never reads longer than it is", () => { + expect(formatDuration(59_999)).toBe("0m"); + expect(formatDuration(90 * 60_000)).toBe("1h 30m"); + expect(formatDuration(8 * 3_600_000)).toBe("8h"); + expect(formatDuration(51 * 3_600_000)).toBe("2d 3h"); + }); +}); + +describe("formatAge", () => { + it("never reports a negative age against a skewed clock", () => { + expect(formatAge(NOW_MS + 60_000, NOW_MS)).toBe("just now"); + }); +}); + +describe("describeCheckInRow", () => { + // Derived facts only. A model-authored summary of its own night is exactly what we do not show. + it("reports the judged outcome, not a narrative", () => { + expect( + describeCheckInRow({ + n: 1, + firedAtMs: NOW_MS, + createdAtIso: "", + activityCursor: "", + outcome: "unproductive", + }), + ).toContain("nothing moved"); + }); +}); + +describe("seedArmDraft", () => { + it("seeds the deadline from defaultRunMs, which is a form seed and never a server fallback", () => { + const draft = seedArmDraft({ + settings: { + enabled: true, + maxArmedThreads: 3, + defaultMaxCheckIns: 4, + defaultRunMs: 2 * 3_600_000, + defaultIdleMs: 15 * 60_000, + defaultBusyIdleMs: 45 * 60_000, + armedCount: 0, + }, + record: record({ maxCheckIns: 0, deadlineAtMs: 0, goal: null }), + nowMs: NOW_MS, + }); + expect(draft.maxCheckIns).toBe(4); + expect(draft.deadlineAtMs).toBe(NOW_MS + 2 * 3_600_000); + }); + + it("keeps the run's own bounds when they are still ahead", () => { + const draft = seedArmDraft({ + settings: null, + record: record({ maxCheckIns: 9, deadlineAtMs: NOW_MS + 60_000, goal: "ship it" }), + nowMs: NOW_MS, + }); + expect(draft).toEqual({ goal: "ship it", maxCheckIns: 9, deadlineAtMs: NOW_MS + 60_000 }); + }); + + it("re-seeds a deadline that has already passed", () => { + const draft = seedArmDraft({ + settings: null, + record: record({ deadlineAtMs: NOW_MS - 1 }), + nowMs: NOW_MS, + }); + expect(draft.deadlineAtMs).toBe(NOW_MS + 8 * 3_600_000); + }); +}); + +describe("datetime-local round trip", () => { + it("survives a round trip in the reader's own timezone", () => { + const at = new Date(2026, 8, 2, 23, 30, 0, 0).getTime(); + expect(fromDateTimeLocalValue(toDateTimeLocalValue(at))).toBe(at); + }); + + it("reads an empty or half-typed field as no deadline at all", () => { + expect(fromDateTimeLocalValue("")).toBeNull(); + expect(fromDateTimeLocalValue("2026-09-")).toBeNull(); + }); +}); + +describe("formatClock and the date", () => { + const SEVEN_AM_TODAY = Date.UTC(2026, 8, 2, 7, 0, 0); + const SEVEN_AM_TOMORROW = Date.UTC(2026, 8, 3, 7, 0, 0); + + it("prints the time alone for today", () => { + expect(formatClock(SEVEN_AM_TODAY, NOW_MS)).toBe(formatClock(SEVEN_AM_TODAY)); + }); + + it("carries the date when the instant is not today", () => { + // Loops run overnight: a run armed at 23:00 ends tomorrow, and a bare `07:00` on it reads + // as eight hours in the PAST rather than eight hours away. + const withDate = formatClock(SEVEN_AM_TOMORROW, NOW_MS); + expect(withDate).not.toBe(formatClock(SEVEN_AM_TOMORROW)); + expect(withDate).toContain(formatClock(SEVEN_AM_TOMORROW)); + }); + + it("prints the time alone when there is no clock to compare against", () => { + // `lastLoadedAtMs` is null before the first load lands; guessing "not today" there would + // put a date on every timestamp in the panel for one render. + expect(formatClock(SEVEN_AM_TOMORROW, 0)).toBe(formatClock(SEVEN_AM_TOMORROW)); + }); + + it("carries into the bounds summary a reader actually looks at", () => { + expect(summariseBounds(derived({ deadlineAtMs: SEVEN_AM_TOMORROW }), NOW_MS)).toContain( + formatClock(SEVEN_AM_TOMORROW, NOW_MS), + ); + }); +}); + +describe("describeLoopState — held is two different facts", () => { + it("words a snooze as a snooze, not as a usage limit", () => { + const copy = describeLoopState( + derived({ state: "held", reason: "snoozed", snoozedUntilMs: DEADLINE_MS }), + NOW_MS, + ); + expect(copy.label).toBe("Snoozed"); + expect(copy.detail).not.toContain("Usage limit"); + expect(copy.detail).toContain("picks up where it left off"); + }); +}); + +describe("describeBlockingHint", () => { + it("points at the composer for a question that is answerable there", () => { + expect(describeBlockingHint(derived({ state: "blocked", reason: "pending_input" }))).toContain( + "composer below", + ); + }); + + it("tells a snoozed thread to unsnooze rather than to answer something", () => { + // There is no prompt in the composer to answer: the human snoozed the thread themselves, + // and sending them looking for one is a dead end. + const hint = describeBlockingHint(derived({ state: "held", reason: "snoozed" })); + expect(hint).toContain("snoozed"); + expect(hint).not.toContain("composer"); + }); + + it("says nothing about a thread that is simply running", () => { + expect(describeBlockingHint(derived())).toBeNull(); + }); +}); + +describe("hasQuestionSections", () => { + const channel = { browserAccessKnown: true, browserAccessEnabled: true }; + + it("is false when every recorded question was already answered", () => { + // The failure: `userInputs` was non-empty so the console suppressed the empty-state card, + // but every section rendered its own empty branch — so a finished run explained nothing. + const answered = view({ + record: record({ + armed: false, + stopped: { reason: "spent", atMs: NOW_MS, detail: "budget" }, + userInputs: [userInput({ resolution: "answered", resolvedAtMs: NOW_MS })], + }), + derived: derived({ state: "stopped", stoppedReason: "spent" }), + }); + expect(hasQuestionSections(answered, channel)).toBe(false); + expect(describeEmptyState(answered, NOW_MS).headline).toContain("Stopped"); + }); + + it("is true while a question is open, or was voided", () => { + expect( + hasQuestionSections(view({ record: record({ userInputs: [userInput()] }) }), channel), + ).toBe(true); + expect( + hasQuestionSections( + view({ record: record({ userInputs: [userInput({ resolution: "voided" })] }) }), + channel, + ), + ).toBe(true); + }); + + it("is true for an answered blocker, which the deferred section still shows", () => { + const banked = blocker({ answeredAtMs: NOW_MS, answer: "in place" }); + expect(hasQuestionSections(view({ record: record({ blockers: [banked] }) }), channel)).toBe( + true, + ); + }); + + it("raises the missing-channel warning only on a thread that has a loop", () => { + const off = { browserAccessKnown: true, browserAccessEnabled: false }; + // No loop here and none ever: nothing would have used the channel, so warning about it is + // noise on every thread in the app rather than a fact about this one. + const noLoop = view({ + record: record({ armed: false }), + derived: derived({ state: "off" }), + }); + expect(hasQuestionSections(noLoop, off)).toBe(false); + expect(hasQuestionSections(view(), off)).toBe(true); + const ended = view({ + record: record({ armed: false, stopped: { reason: "done", atMs: NOW_MS, detail: "" } }), + derived: derived({ state: "stopped", stoppedReason: "done" }), + }); + expect(hasQuestionSections(ended, off)).toBe(true); + }); +}); + +describe("canRenderLoopConsole", () => { + it("renders on a thread in the primary environment", () => { + expect( + canRenderLoopConsole({ primaryEnvironmentId: "env-1", threadEnvironmentId: "env-1" }), + ).toBe(true); + }); + + it("renders nothing rather than answering for another environment's thread", () => { + // Every fork route is called against the primary environment, so this would report "no + // loop on this thread" for a loop that may well be armed — and arming would 404. + expect( + canRenderLoopConsole({ primaryEnvironmentId: "env-1", threadEnvironmentId: "env-2" }), + ).toBe(false); + }); + + it("does not treat an unresolved primary as a mismatch", () => { + expect(canRenderLoopConsole({ primaryEnvironmentId: null, threadEnvironmentId: "env-2" })).toBe( + true, + ); + }); +}); + +describe("describeRefusal — the armed preconditions", () => { + it("words both 409s rather than falling back to the generic sentence", () => { + expect(describeRefusal("not_armed").message).toContain("not running"); + expect(describeRefusal("armed").message).toContain("Disarm it first"); + }); +}); diff --git a/apps/web/src/coil/loop/loopPresentation.ts b/apps/web/src/coil/loop/loopPresentation.ts new file mode 100644 index 000000000000..fbb886f9f10b --- /dev/null +++ b/apps/web/src/coil/loop/loopPresentation.ts @@ -0,0 +1,525 @@ +/** + * Pure presentation helpers for the loop console and the Loops settings panel. + * + * Deliberately free of React and of `Date.now()` — every function takes the values it needs, so + * the copy shown for each state is pinned by tests rather than by screenshotting the app. The two + * rules this module exists to enforce are both here rather than in JSX, for exactly that reason: + * + * **`spent` is never rendered as success.** "It finished" and "it ran out of rope" are different + * outcomes, and conflating them is the failure this feature is built to stop. `done` is the only + * emerald tone in the file. + * + * **A missing question channel is never rendered as "no questions".** The MCP credential that + * backs `raise_blocker` is minted only when agent browser access is on, so with it off the console + * would otherwise show an empty blocker list — which reads as "nothing to ask about" and is the + * precise failure the deferred channel exists to prevent. + * + * @module coil/loop/loopPresentation + */ + +import type { + LoopBlocker, + LoopCheckInRow, + LoopDerived, + LoopSettings, + LoopStopReason, + LoopUserInput, + LoopView, +} from "./loopClient"; + +/** + * Colour families the console uses, resolved to theme tokens at the call site. + * + * `spent` is its own tone rather than a variant of `done` so that no styling change can ever make + * an exhausted run look finished. + */ +export type LoopTone = "muted" | "active" | "attention" | "held" | "done" | "spent"; + +export interface LoopStateCopy { + readonly label: string; + readonly tone: LoopTone; + /** One sentence saying what that state actually means for the human reading it. */ + readonly detail: string; +} + +const clockFormatter = new Intl.DateTimeFormat(undefined, { hour: "numeric", minute: "2-digit" }); +const dateClockFormatter = new Intl.DateTimeFormat(undefined, { + month: "short", + day: "numeric", + hour: "numeric", + minute: "2-digit", +}); + +/** + * `07:00`, or `Sep 4, 07:00` when that is not today. + * + * Absolute rather than a countdown: a countdown means a timer repainting forever. The date + * is not decoration — loops run overnight and their deadlines are routinely tomorrow, so a + * bare `07:00` on a run armed at 23:00 reads as eight hours ago rather than eight hours + * away. `nowMs` is what "today" is measured against; omit it and you get the time alone, + * which is right wherever the surrounding copy already says which day it means. + */ +export function formatClock(atMs: number, nowMs?: number): string { + const at = new Date(atMs); + if (nowMs === undefined || !Number.isFinite(nowMs) || nowMs === 0) { + return clockFormatter.format(at); + } + const now = new Date(nowMs); + const sameDay = + at.getFullYear() === now.getFullYear() && + at.getMonth() === now.getMonth() && + at.getDate() === now.getDate(); + return sameDay ? clockFormatter.format(at) : dateClockFormatter.format(at); +} + +/** `45m`, `8h`, `8h 30m`, `2d 3h`. Rounded down, because a bound that reads long is a lie. */ +export function formatDuration(ms: number): string { + const totalMinutes = Math.max(0, Math.floor(ms / 60_000)); + if (totalMinutes < 60) return `${totalMinutes}m`; + const hours = Math.floor(totalMinutes / 60); + const minutes = totalMinutes % 60; + if (hours < 24) return minutes === 0 ? `${hours}h` : `${hours}h ${minutes}m`; + const days = Math.floor(hours / 24); + const remainingHours = hours % 24; + return remainingHours === 0 ? `${days}d` : `${days}d ${remainingHours}h`; +} + +/** `just now`, `12m ago`, `3h 05m ago`. Never a negative age, even against a skewed clock. */ +export function formatAge(atMs: number, nowMs: number): string { + const elapsed = nowMs - atMs; + if (!Number.isFinite(elapsed) || elapsed < 60_000) return "just now"; + return `${formatDuration(elapsed)} ago`; +} + +const STOP_COPY: Readonly> = { + done: { + label: "Done", + tone: "done", + detail: "The agent said it had finished, so the loop stopped on its own signal.", + }, + // Zinc, never emerald. The agent never signalled done — the run hit a bound. + spent: { + label: "Out of rope", + tone: "spent", + detail: "The budget or the deadline ran out. It did not finish — it was stopped.", + }, + stalled: { + label: "Stalled", + tone: "spent", + detail: "Two check-ins in a row moved nothing, so the loop stopped spending on it.", + }, + "handed-back": { + label: "Handed back", + tone: "muted", + detail: "You took over, so supervision stopped. The budget was not reset.", + }, +}; + +export function describeStop(reason: LoopStopReason): LoopStateCopy { + return STOP_COPY[reason]; +} + +const STAND_DOWN_DETAIL: Readonly> = { + disabled: "Loops are switched off in Settings. Nothing fires, and nothing was disarmed.", + snoozed: "The thread is snoozed. Unsnooze it and the loop picks up where it left off.", + pending_input: "Something is waiting on you in the composer. Answer it and the loop resumes.", + rate_limited: "A usage limit is in force. Auto-resume owns the wake; no check-in was spent.", +}; + +/** + * The state the console leads with. + * + * Reads the server's `derived` view rather than re-deriving anything: the route already resolved + * the terminal-first ordering, and a second opinion computed in the browser would drift. + */ +export function describeLoopState(derived: LoopDerived, nowMs?: number): LoopStateCopy { + switch (derived.state) { + case "stopped": + return describeStop(derived.stoppedReason ?? "spent"); + case "off": + return { + label: "Not armed", + tone: "muted", + detail: "Nothing is supervising this thread.", + }; + case "standing_down": + return { + label: "Standing down", + tone: "muted", + detail: + STAND_DOWN_DETAIL[derived.reason ?? ""] ?? + "A guard is holding the loop back. Its budget and deadline are intact.", + }; + // `held` carries two facts — a usage limit and a snooze — and they are not the same + // sentence. The server reports both here because both are bounded holds with an expiry; + // wording them identically would tell someone their thread was rate limited when they + // had snoozed it themselves. + case "held": + return derived.reason === "snoozed" + ? { + label: "Snoozed", + tone: "held", + detail: + derived.snoozedUntilMs === null + ? STAND_DOWN_DETAIL.snoozed! + : `Snoozed until ${formatClock(derived.snoozedUntilMs, nowMs)}. The loop picks up where it left off.`, + } + : { + label: "Held", + tone: "held", + detail: + derived.rateLimitedUntilMs > 0 + ? `Usage limit until ${formatClock(derived.rateLimitedUntilMs, nowMs)}. No check-in was spent.` + : STAND_DOWN_DETAIL.rate_limited!, + }; + case "blocked": + return { + label: "Waiting on you", + tone: "attention", + detail: + STAND_DOWN_DETAIL[derived.reason ?? ""] ?? + "The loop cannot go further without a decision from you.", + }; + case "self_pacing": + return { + label: "Self-pacing", + tone: "active", + detail: + derived.nextWakeAtMs === null + ? "The agent is scheduling its own wake-ups. T3 is standing by." + : `The agent scheduled its own wake for ${formatClock(derived.nextWakeAtMs, nowMs)}. T3 stands by and spends nothing.`, + }; + case "watching": + return { + label: "Watching", + tone: "active", + detail: "Running. The loop checks in only if the thread goes quiet.", + }; + } +} + +/** + * `2 of 6 check-ins · ends 07:00`, the one line the collapsed pill has room for. + * + * The deadline carries its date when it is not today, because that is the common case for + * this feature: a run armed at 23:00 ends tomorrow morning, and `ends 07:00` alone reads as + * a deadline that has already passed. + */ +export function summariseBounds(derived: LoopDerived, nowMs?: number): string { + const budget = `${derived.checkInsUsed} of ${derived.maxCheckIns} check-ins`; + if (derived.deadlineAtMs <= 0) return budget; + return `${budget} · ends ${formatClock(derived.deadlineAtMs, nowMs)}`; +} + +export interface BlockerPartition { + /** Unanswered — what is actionable now. */ + readonly open: ReadonlyArray; + /** Answered, but the agent has not been told yet. It banks on the next check-in. */ + readonly banked: ReadonlyArray; + /** Answered and restated to the agent. */ + readonly delivered: ReadonlyArray; +} + +/** + * Splits blockers three ways, because "answered" and "the agent knows" are different facts. + * + * Answering at 09:04 while the thread is idle does nothing until the next check-in prompt carries + * the answer, and a console that showed only "answered" would imply the agent had already acted + * on it. + */ +export function partitionBlockers(blockers: ReadonlyArray): BlockerPartition { + const open: Array = []; + const banked: Array = []; + const delivered: Array = []; + for (const blocker of blockers) { + if (blocker.answeredAtMs === null) open.push(blocker); + else if (blocker.deliveredToAgent) delivered.push(blocker); + else banked.push(blocker); + } + return { open, banked, delivered }; +} + +export interface UserInputPartition { + readonly open: ReadonlyArray; + readonly answered: ReadonlyArray; + /** + * Settled by session teardown as an empty answer — nobody ever saw it. + * + * Upstream leaves no trace of these: `hasPendingUserInput` reads false afterwards, so a voided + * question is otherwise indistinguishable from an answered one. + */ + readonly voided: ReadonlyArray; +} + +export function partitionUserInputs(userInputs: ReadonlyArray): UserInputPartition { + const open: Array = []; + const answered: Array = []; + const voided: Array = []; + for (const entry of userInputs) { + if (entry.resolution === null) open.push(entry); + else if (entry.resolution === "voided") voided.push(entry); + else answered.push(entry); + } + return { open, answered, voided }; +} + +export interface DeferredChannelNotice { + readonly title: string; + readonly detail: string; +} + +/** + * The named degraded state for the deferred-question channel. + * + * `raise_blocker` reaches the agent over the per-thread MCP credential, which + * `prepareMcpSession` mints only while **Settings → Integrations → Agent browser access** is on. + * With it off the agent has no way to raise a question at all, so an empty list here is not + * evidence of "no questions" — it is evidence of no channel. Returns `null` only when the channel + * is genuinely available, or when the client cannot see the setting at all (there is no primary + * environment to read it from, so claiming either way would be a guess). + */ +export function resolveDeferredChannelNotice(input: { + readonly browserAccessKnown: boolean; + readonly browserAccessEnabled: boolean; + /** + * Whether this thread has a loop at all (armed, or one that ended). With no loop there is + * nothing that would have used the channel, so the warning is noise on every thread in the + * app rather than a fact about this one. + */ + readonly loopExists?: boolean; +}): DeferredChannelNotice | null { + if (input.loopExists === false) return null; + if (!input.browserAccessKnown || input.browserAccessEnabled) return null; + return { + title: "Deferred questions unavailable", + detail: + "Agent browser access is off in Settings → Integrations, so the agent cannot raise a question without stopping. An empty list here does not mean it had nothing to ask.", + }; +} + +const REFUSAL_COPY: Readonly> = { + deadline_required: "Pick an end time. A loop with no deadline is not a bound.", + deadline_in_past: "That end time has already passed.", + budget_required: "Set how many check-ins this run may spend.", + budget_too_large: "The most a single run may spend is 20 check-ins.", + budget_too_small: "A run needs at least one check-in.", + ceiling_reached: "Too many loops are already armed. Disarm one, or raise the limit in Settings.", + thread_snoozed: "This thread is snoozed. Unsnooze it first — arming would cancel the snooze.", + unknown_thread: "That thread is not in this environment any more.", + projection_unavailable: "The thread index is unavailable right now. Try again in a moment.", + invalid_body: "The server did not understand that request.", + out_of_range: "One of those values is outside the range the server accepts.", + not_found: "That question is no longer open.", + not_armed: "This loop is not running any more. Refresh to see where it ended up.", + armed: "This loop is still running. Disarm it first.", +}; + +/** + * A refusal, worded for a human, with the server's own code alongside it. + * + * The code is carried verbatim rather than swallowed: the route gives every 400 a distinct code so + * the console can word them differently, and showing it keeps a refusal the console has no copy + * for from rendering as a blank failure. + */ +export function describeRefusal(code: string): { readonly code: string; readonly message: string } { + return { + code, + message: REFUSAL_COPY[code] ?? "The server refused that change.", + }; +} + +/** One ledger row, from derived facts only — never a model-authored summary of its own night. */ +export function describeCheckInRow(row: LoopCheckInRow): string { + const outcome = + row.outcome === "productive" + ? "the thread moved" + : row.outcome === "unproductive" + ? "nothing moved" + : "outcome not yet judged"; + return `Checked in at ${formatClock(row.firedAtMs)} — ${outcome}`; +} + +export interface LoopEmptyState { + readonly headline: string; + readonly lines: ReadonlyArray; +} + +/** + * What the console says when there is nothing to answer. + * + * This is the acceptance case: a run that ended `spent` with a model that never called + * `raise_blocker` still has to say something useful — what happened, what it consumed, and when it + * last moved — rather than rendering an empty list. + */ +export function describeEmptyState(view: LoopView, nowMs: number): LoopEmptyState { + const { derived, record } = view; + const lines: Array = []; + if (derived.state === "stopped") { + const stop = describeStop(derived.stoppedReason ?? "spent"); + lines.push(stop.detail); + if (record.stopped !== null && record.stopped.detail !== "") { + lines.push(record.stopped.detail); + } + lines.push( + `Used ${derived.checkInsUsed} of ${derived.maxCheckIns} check-ins${ + derived.deadlineAtMs > 0 + ? ` against a ${formatClock(derived.deadlineAtMs, nowMs)} deadline` + : "" + }.`, + ); + const lastRow = record.checkIns.at(-1); + lines.push( + lastRow === undefined + ? "It never checked in — the run ended before the thread went quiet." + : `Last activity ${formatAge(lastRow.firedAtMs, nowMs)}.`, + ); + return { + headline: + derived.stoppedReason === "done" + ? "Finished. Nothing is waiting on you." + : "Stopped. Nothing is waiting on you.", + lines, + }; + } + + if (derived.state === "off") { + return { + headline: "No loop on this thread.", + lines: ["Arm one to keep it working while you are away."], + }; + } + + const state = describeLoopState(derived, nowMs); + lines.push(state.detail); + lines.push(summariseBounds(derived, nowMs)); + return { headline: "Nothing is waiting on you.", lines }; +} + +export interface ArmDraft { + readonly goal: string; + readonly maxCheckIns: number; + readonly deadlineAtMs: number; +} + +/** + * Seeds the arm form. + * + * `defaultRunMs` seeds the *form*, and only the form. It is never a fallback deadline on the + * server: a deadline the human did not choose is not a bound they agreed to, which is why the + * route 400s rather than defaulting one. + */ +export function seedArmDraft(input: { + readonly settings: LoopSettings | null; + readonly record: LoopView["record"]; + readonly nowMs: number; +}): ArmDraft { + const settings = input.settings; + const runMs = settings?.defaultRunMs ?? 8 * 3_600_000; + return { + goal: input.record.goal ?? "", + maxCheckIns: + input.record.maxCheckIns > 0 ? input.record.maxCheckIns : (settings?.defaultMaxCheckIns ?? 6), + deadlineAtMs: + input.record.deadlineAtMs > input.nowMs ? input.record.deadlineAtMs : input.nowMs + runMs, + }; +} + +/** + * `2026-09-02T23:00` for ``, in the reader's own timezone. + * + * The server stores an absolute instant; the browser is the only party that knows what "23:00 + * tonight" means to the person typing it. + */ +export function toDateTimeLocalValue(atMs: number): string { + const at = new Date(atMs); + const pad = (value: number) => String(value).padStart(2, "0"); + return `${at.getFullYear()}-${pad(at.getMonth() + 1)}-${pad(at.getDate())}T${pad(at.getHours())}:${pad(at.getMinutes())}`; +} + +/** + * The inverse, or `null` when the field is empty or half-typed. + * + * The shape is checked before `Date` sees it: `new Date("2026-09-")` parses happily as the first + * of September, so leaning on `Number.isFinite` alone would turn a half-typed field into a real + * deadline the human never chose. + */ +const DATE_TIME_LOCAL = /^\d{4}-\d{2}-\d{2}T\d{2}:\d{2}(:\d{2})?$/; + +export function fromDateTimeLocalValue(value: string): number | null { + if (!DATE_TIME_LOCAL.test(value)) return null; + const parsed = new Date(value).getTime(); + return Number.isFinite(parsed) ? parsed : null; +} + +/** A loop this thread has now, or had: armed, or carrying a terminal state. */ +export function hasLoop(view: LoopView): boolean { + return view.record.armed || view.record.stopped !== null; +} + +/** + * Whether `LoopQuestions` would render anything at all. + * + * The console shows either the question sections or the empty-state card, and this is the + * predicate that decides. It has to agree with what those sections actually render, or the + * panel shows neither: a run whose only recorded questions were already **answered** put + * every section into its own empty branch, and the empty-state card — the one thing that + * explains a finished run — was suppressed because `userInputs` was non-empty. + */ +export function hasQuestionSections( + view: LoopView, + channel: { + readonly browserAccessKnown: boolean; + readonly browserAccessEnabled: boolean; + }, +): boolean { + const inputs = partitionUserInputs(view.record.userInputs); + return ( + inputs.open.length > 0 || + inputs.voided.length > 0 || + view.record.blockers.length > 0 || + describeBlockingHint(view.derived) !== null || + resolveDeferredChannelNotice({ ...channel, loopExists: hasLoop(view) }) !== null + ); +} + +/** + * What to say under the blocking section, or `null` when there is nothing to point at. + * + * A snooze is not a question. Telling someone to "answer it in the composer below" when they + * snoozed the thread themselves sends them looking for a prompt that does not exist — the + * way out is to unsnooze, and it is a different sentence. + */ +export function describeBlockingHint(derived: LoopDerived): string | null { + if (derived.reason === "snoozed") { + return "This thread is snoozed. Unsnooze it and the loop picks up where it left off."; + } + if (derived.state !== "blocked") return null; + return "Answer it in the composer below. The loop resumes on its own once you do."; +} + +/** + * Whether the console can speak for this thread. + * + * Every fork route is called against the **primary** environment — the only one the web app + * knows how to authenticate — exactly as the auto-resume overlay's client is. On a thread + * belonging to another environment the reads would therefore answer for a thread id the + * primary server has never heard of: "no loop on this thread" for a loop that may well be + * armed, and an arm that 404s. Rendering nothing is the honest version of that, and + * `docs/user/loops.md` states the limitation. `null` means the primary is not resolved yet, + * which is not evidence of a mismatch. + */ +export function canRenderLoopConsole(input: { + readonly primaryEnvironmentId: string | null; + readonly threadEnvironmentId: string; +}): boolean { + return input.primaryEnvironmentId === null + ? true + : input.primaryEnvironmentId === input.threadEnvironmentId; +} + +/** How many things are actually waiting on a human, for the pill's count. */ +export function countWaiting(view: LoopView): number { + const blockers = partitionBlockers(view.record.blockers).open.length; + const nativeOpen = partitionUserInputs(view.record.userInputs).open.length; + return blockers + nativeOpen; +} diff --git a/apps/web/src/coil/loop/useLoopPolling.ts b/apps/web/src/coil/loop/useLoopPolling.ts new file mode 100644 index 000000000000..b72380171695 --- /dev/null +++ b/apps/web/src/coil/loop/useLoopPolling.ts @@ -0,0 +1,98 @@ +/** + * The fetch cadence every loop surface shares: poll every 30s, and again on window focus. + * + * Same shape as the auto-resume overlay's, and for the same reasons — a fork route is not on the + * websocket, so there is nothing to subscribe to, and 30s is slow enough to be invisible on the + * wire while a focus refresh covers the case that actually matters (you come back to the machine + * at 9am and want today's answer, not last night's). + * + * **No spinner.** A loop that has not moved for eight hours is not "loading", and a spinner over + * it would be a lying one. Callers render `lastLoadedAtMs` instead, so the console says when it + * last heard from the server rather than pretending to be busy. + * + * @module coil/loop/useLoopPolling + */ + +import { useCallback, useEffect, useRef, useState } from "react"; + +export const LOOP_POLL_INTERVAL_MS = 30_000; + +export interface PolledResource { + /** `null` means "we do not know" — never "there is nothing". */ + readonly value: A | null; + /** When the last successful load landed, for an honest "Updated …" label. */ + readonly lastLoadedAtMs: number | null; + /** + * The last attempt did not produce a value. + * + * `value === null` alone cannot tell "the first read has not landed yet" from "the read + * failed", and a surface that renders both as "Loading…" claims to be busy forever on a + * 403 or an offline server — with every control disabled and no way to try again. + */ + readonly failed: boolean; + readonly refresh: () => void; + /** Apply a value the caller already has in hand (e.g. the body of a write response). */ + readonly set: (value: A) => void; +} + +/** + * Polls `load`, discarding any response that lands after `key` changed. + * + * `key` is the identity of the thing being loaded (a threadId, or a constant for a global + * resource). Changing it clears the value first, so a slow in-flight read for the previous thread + * can never populate the new one. + */ +export function useLoopPolling(key: string, load: () => Promise): PolledResource { + const [value, setValue] = useState(null); + const [lastLoadedAtMs, setLastLoadedAtMs] = useState(null); + const [failed, setFailed] = useState(false); + const loadRef = useRef(load); + loadRef.current = load; + // Bumped on every key change; a late response carrying a stale token is dropped. + const tokenRef = useRef(0); + + const refresh = useCallback(() => { + const token = tokenRef.current; + void loadRef.current().then( + (next) => { + if (token !== tokenRef.current) return; + if (next === null) { + // A previously loaded value is kept: a failed poll makes the panel stale, not empty. + setFailed(true); + return; + } + setFailed(false); + setValue(next); + setLastLoadedAtMs(Date.now()); + }, + () => { + // `load` resolves to null on failure; a rejection here is still non-fatal. + if (token === tokenRef.current) setFailed(true); + }, + ); + }, []); + + useEffect(() => { + tokenRef.current += 1; + setValue(null); + setLastLoadedAtMs(null); + setFailed(false); + refresh(); + + const intervalId = window.setInterval(refresh, LOOP_POLL_INTERVAL_MS); + const handleFocus = () => refresh(); + window.addEventListener("focus", handleFocus); + return () => { + window.clearInterval(intervalId); + window.removeEventListener("focus", handleFocus); + }; + }, [key, refresh]); + + const set = useCallback((next: A) => { + setValue(next); + setLastLoadedAtMs(Date.now()); + setFailed(false); + }, []); + + return { value, lastLoadedAtMs, failed, refresh, set }; +} diff --git a/apps/web/src/components/coil/LoopsSettings.tsx b/apps/web/src/components/coil/LoopsSettings.tsx new file mode 100644 index 000000000000..4186f13c457a --- /dev/null +++ b/apps/web/src/components/coil/LoopsSettings.tsx @@ -0,0 +1,338 @@ +/** + * Settings → Loops. Fork-owned, and deliberately not part of upstream's settings machinery. + * + * Loop state lives in `coil-loop.json` and is read and written through `/api/coil/loop/settings`, + * so this panel touches neither `SettingsPanels.tsx` (churn 43) nor + * `packages/contracts/src/settings.ts` (churn 38, persisted). The whole section costs the fork two + * additive lines in `settingsSearch.ts` and two in `SettingsSidebarNav.tsx`, and nothing else. + * + * ## The master switch is a guard, not a lifecycle + * + * This is the one thing the copy here must get right. The supervisor always runs. Switching loops + * off means nothing **fires**: every armed loop reports `standing_down` with reason `disabled`, + * and **nothing is disarmed, nothing is stopped, no budget is spent or reset**. Switching it back + * on resumes the same loops with the same budgets and the same deadlines. Wording it as an + * on/off for loops *themselves* would teach people that flipping it cancels their overnight runs, + * which is exactly what it does not do. + * + * The same rule covers the ceiling: lowering "Loops at once" below the number currently armed is + * accepted, and the excess stand down at the next tick rather than being disarmed. + * + * @module coil/LoopsSettings + */ + +import { Link } from "@tanstack/react-router"; +import { useCallback } from "react"; + +import { Button } from "~/components/ui/button"; +import { + SettingsPageContainer, + SettingsRow, + SettingsSection, +} from "~/components/settings/settingsLayout"; +import { searchableSetting } from "~/components/settings/settingsSearch"; +import { NumberField, NumberFieldGroup, NumberFieldInput } from "~/components/ui/number-field"; +import { Switch } from "~/components/ui/switch"; +import { LOOP_MAX_CHECK_INS, httpLoopClient } from "~/coil/loop/loopClient"; +import type { LoopSettings, LoopView } from "~/coil/loop/loopClient"; +import { describeLoopState, formatClock, summariseBounds } from "~/coil/loop/loopPresentation"; +import { useLoopPolling } from "~/coil/loop/useLoopPolling"; +import { usePrimaryEnvironmentId } from "~/state/environments"; + +const MINUTES = 60_000; +const HOURS = 3_600_000; + +const NO_GROUPING: Intl.NumberFormatOptions = { useGrouping: false }; + +function NumberSetting({ + label, + value, + min, + max, + disabled, + suffix, + onCommit, +}: { + readonly label: string; + readonly value: number; + readonly min: number; + readonly max: number; + readonly disabled: boolean; + readonly suffix?: string; + readonly onCommit: (value: number) => void; +}) { + return ( +
+ { + if (next === null || !Number.isFinite(next)) return; + onCommit(next); + }} + size="sm" + value={value} + > + + + + + {suffix === undefined ? null : ( + {suffix} + )} +
+ ); +} + +/** + * "We could not read that" — with a way to try again. + * + * The panel used to render every non-200 as "Loading…", which is a lie that never resolves: + * a 403, an offline server or a route that is not deployed left the roster claiming to be + * busy forever with every control disabled and nothing to click. + */ +function LoadFailed({ what, onRetry }: { readonly what: string; readonly onRetry: () => void }) { + return ( +
+

Could not load {what}.

+ +
+ ); +} + +/** + * "Did any of my runs give up overnight?" answered from one page. + * + * Arming is deliberately *not* here: it is a decision made at the moment you walk away from a + * thread, not one made in Settings. + */ +function ArmedRoster({ + loops, + failed, + nowMs, + onRetry, +}: { + readonly loops: ReadonlyArray | null; + readonly failed: boolean; + /** The load's own clock read, so a deadline that is not today carries its date. */ + readonly nowMs: number; + readonly onRetry: () => void; +}) { + const primaryEnvironmentId = usePrimaryEnvironmentId(); + + if (loops === null) { + return failed ? ( + + ) : ( +

Loading…

+ ); + } + if (loops.length === 0) { + return ( +

+ No loops are armed. Arm one from a thread, using the loop control above its composer. +

+ ); + } + + return ( +
    + {loops.map((loop) => { + const state = describeLoopState(loop.derived, nowMs); + return ( +
  • + + {primaryEnvironmentId === null ? ( + + {loop.record.goal ?? loop.threadId} + + ) : ( + + {loop.record.goal ?? loop.threadId} + + )} + {state.label} + + + {summariseBounds(loop.derived, nowMs)} + +
  • + ); + })} +
+ ); +} + +export function LoopsSettingsPanel() { + const loadSettings = useCallback(() => httpLoopClient.readSettings(), []); + const settings = useLoopPolling("global", loadSettings); + const loadLoops = useCallback(() => httpLoopClient.listLoops(), []); + const loops = useLoopPolling>("loops", loadLoops); + + const value = settings.value; + const patch = useCallback( + (next: Partial>) => { + void httpLoopClient.writeSettings(next).then((result) => { + if (result === null || !result.ok) { + // The server refused or is unreachable; re-read rather than leaving an optimistic lie + // on screen. + settings.refresh(); + return; + } + settings.set(result.value); + loops.refresh(); + }); + }, + [loops, settings], + ); + + // The controls stay disabled while the current values are unknown, deliberately: every one + // of them writes a value that spends money unattended, and writing one over a state we + // could not read is worse than not offering it. The retry above them is the enabled + // control — the failure this fixes was having none at all. + const disabled = value === null; + + return ( + + + {settings.value === null && settings.failed ? ( + + ) : null} + patch({ enabled: Boolean(checked) })} + /> + } + description="A loop keeps one thread working while you are away and collects what it needs from you in one place. This switch is a guard, not a lifecycle: with it off nothing fires, and armed loops stand down keeping their budget and deadline. Nothing is disarmed and nothing is stopped." + status={ + value === null + ? undefined + : `${value.armedCount} of ${value.maxArmedThreads} loops armed right now.` + } + /> + patch({ maxArmedThreads: next })} + value={value?.maxArmedThreads ?? 3} + /> + } + /> + + + + patch({ defaultIdleMs: next * MINUTES })} + suffix="min" + value={Math.round((value?.defaultIdleMs ?? 15 * MINUTES) / MINUTES)} + /> + } + /> + patch({ defaultBusyIdleMs: next * MINUTES })} + suffix="min" + value={Math.round((value?.defaultBusyIdleMs ?? 45 * MINUTES) / MINUTES)} + /> + } + /> + patch({ defaultMaxCheckIns: next })} + value={value?.defaultMaxCheckIns ?? 6} + /> + } + /> + patch({ defaultRunMs: next * HOURS })} + suffix="hours" + value={Math.round((value?.defaultRunMs ?? 8 * HOURS) / HOURS)} + /> + } + /> + + + + {value.armedCount} of {value.maxArmedThreads} used + + ) + } + > + + {loops.lastLoadedAtMs === null ? null : ( +

+ Updated {formatClock(loops.lastLoadedAtMs)} +

+ )} +
+
+ ); +} diff --git a/apps/web/src/components/settings/SettingsSidebarNav.tsx b/apps/web/src/components/settings/SettingsSidebarNav.tsx index daf1724a38d9..66e808b779d0 100644 --- a/apps/web/src/components/settings/SettingsSidebarNav.tsx +++ b/apps/web/src/components/settings/SettingsSidebarNav.tsx @@ -15,6 +15,7 @@ import { KeyboardIcon, Link2Icon, PaletteIcon, + RefreshCwIcon, SearchIcon, Settings2Icon, XIcon, @@ -52,6 +53,7 @@ const SETTINGS_SECTION_ICONS: Readonly< "/settings/keybindings": KeyboardIcon, "/settings/providers": BotIcon, "/settings/integrations": BlocksIcon, + "/settings/loops": RefreshCwIcon, "/settings/source-control": GitBranchIcon, "/settings/connections": Link2Icon, "/settings/archived": ArchiveIcon, diff --git a/apps/web/src/components/settings/settingsSearch.ts b/apps/web/src/components/settings/settingsSearch.ts index a3e5cf4b9a69..e51f646232d1 100644 --- a/apps/web/src/components/settings/settingsSearch.ts +++ b/apps/web/src/components/settings/settingsSearch.ts @@ -7,6 +7,7 @@ export type SettingsPath = | "/settings/keybindings" | "/settings/providers" | "/settings/integrations" + | "/settings/loops" | "/settings/source-control" | "/settings/connections" | "/settings/archived"; @@ -52,6 +53,7 @@ export const SETTINGS_SECTION_LABELS: Readonly> = { "/settings/keybindings": "Keybindings", "/settings/providers": "Providers", "/settings/integrations": "Integrations", + "/settings/loops": "Loops", "/settings/source-control": "Source Control", "/settings/connections": "Connections", "/settings/archived": "Archive", @@ -431,6 +433,18 @@ export const SETTINGS_SEARCH_ITEMS = [ to: "/settings/connections", searchTerms: ["add pair backend host code ssh config agent tunnel saved t3 connect"], }, + { + id: "loops-enabled", + title: "Let threads run as loops", + to: "/settings/loops", + searchTerms: ["loop overnight unattended supervise check-in autonomous background"], + }, + { + id: "loop-defaults", + title: "Defaults for a new loop", + to: "/settings/loops", + searchTerms: ["loop budget check-ins deadline idle busy how many at once ceiling"], + }, { id: "archive", title: "Archived threads", diff --git a/apps/web/src/routeTree.gen.ts b/apps/web/src/routeTree.gen.ts index f7c47ace6840..cb69c5cb030c 100644 --- a/apps/web/src/routeTree.gen.ts +++ b/apps/web/src/routeTree.gen.ts @@ -17,6 +17,7 @@ import { Route as ChatRouteImport } from './routes/_chat' import { Route as ChatIndexRouteImport } from './routes/_chat.index' import { Route as SettingsSourceControlRouteImport } from './routes/settings.source-control' import { Route as SettingsProvidersRouteImport } from './routes/settings.providers' +import { Route as SettingsLoopsRouteImport } from './routes/settings.loops' import { Route as SettingsKeybindingsRouteImport } from './routes/settings.keybindings' import { Route as SettingsIntegrationsRouteImport } from './routes/settings.integrations' import { Route as SettingsGeneralRouteImport } from './routes/settings.general' @@ -69,6 +70,11 @@ const SettingsProvidersRoute = SettingsProvidersRouteImport.update({ path: '/providers', getParentRoute: () => SettingsRoute, } as any) +const SettingsLoopsRoute = SettingsLoopsRouteImport.update({ + id: '/loops', + path: '/loops', + getParentRoute: () => SettingsRoute, +} as any) const SettingsKeybindingsRoute = SettingsKeybindingsRouteImport.update({ id: '/keybindings', path: '/keybindings', @@ -147,6 +153,7 @@ export interface FileRoutesByFullPath { '/settings/general': typeof SettingsGeneralRoute '/settings/integrations': typeof SettingsIntegrationsRoute '/settings/keybindings': typeof SettingsKeybindingsRoute + '/settings/loops': typeof SettingsLoopsRoute '/settings/providers': typeof SettingsProvidersRoute '/settings/source-control': typeof SettingsSourceControlRoute '/$environmentId/$threadId': typeof ChatEnvironmentIdThreadIdRoute @@ -167,6 +174,7 @@ export interface FileRoutesByTo { '/settings/general': typeof SettingsGeneralRoute '/settings/integrations': typeof SettingsIntegrationsRoute '/settings/keybindings': typeof SettingsKeybindingsRoute + '/settings/loops': typeof SettingsLoopsRoute '/settings/providers': typeof SettingsProvidersRoute '/settings/source-control': typeof SettingsSourceControlRoute '/': typeof ChatIndexRoute @@ -190,6 +198,7 @@ export interface FileRoutesById { '/settings/general': typeof SettingsGeneralRoute '/settings/integrations': typeof SettingsIntegrationsRoute '/settings/keybindings': typeof SettingsKeybindingsRoute + '/settings/loops': typeof SettingsLoopsRoute '/settings/providers': typeof SettingsProvidersRoute '/settings/source-control': typeof SettingsSourceControlRoute '/_chat/': typeof ChatIndexRoute @@ -214,6 +223,7 @@ export interface FileRouteTypes { | '/settings/general' | '/settings/integrations' | '/settings/keybindings' + | '/settings/loops' | '/settings/providers' | '/settings/source-control' | '/$environmentId/$threadId' @@ -234,6 +244,7 @@ export interface FileRouteTypes { | '/settings/general' | '/settings/integrations' | '/settings/keybindings' + | '/settings/loops' | '/settings/providers' | '/settings/source-control' | '/' @@ -256,6 +267,7 @@ export interface FileRouteTypes { | '/settings/general' | '/settings/integrations' | '/settings/keybindings' + | '/settings/loops' | '/settings/providers' | '/settings/source-control' | '/_chat/' @@ -331,6 +343,13 @@ declare module '@tanstack/react-router' { preLoaderRoute: typeof SettingsProvidersRouteImport parentRoute: typeof SettingsRoute } + '/settings/loops': { + id: '/settings/loops' + path: '/loops' + fullPath: '/settings/loops' + preLoaderRoute: typeof SettingsLoopsRouteImport + parentRoute: typeof SettingsRoute + } '/settings/keybindings': { id: '/settings/keybindings' path: '/keybindings' @@ -442,6 +461,7 @@ interface SettingsRouteChildren { SettingsGeneralRoute: typeof SettingsGeneralRoute SettingsIntegrationsRoute: typeof SettingsIntegrationsRoute SettingsKeybindingsRoute: typeof SettingsKeybindingsRoute + SettingsLoopsRoute: typeof SettingsLoopsRoute SettingsProvidersRoute: typeof SettingsProvidersRoute SettingsSourceControlRoute: typeof SettingsSourceControlRoute } @@ -454,6 +474,7 @@ const SettingsRouteChildren: SettingsRouteChildren = { SettingsGeneralRoute: SettingsGeneralRoute, SettingsIntegrationsRoute: SettingsIntegrationsRoute, SettingsKeybindingsRoute: SettingsKeybindingsRoute, + SettingsLoopsRoute: SettingsLoopsRoute, SettingsProvidersRoute: SettingsProvidersRoute, SettingsSourceControlRoute: SettingsSourceControlRoute, } diff --git a/apps/web/src/routes/_chat.$environmentId.$threadId.tsx b/apps/web/src/routes/_chat.$environmentId.$threadId.tsx index 23cd38dcac66..cac4192ebf66 100644 --- a/apps/web/src/routes/_chat.$environmentId.$threadId.tsx +++ b/apps/web/src/routes/_chat.$environmentId.$threadId.tsx @@ -15,7 +15,7 @@ import { } from "../state/entities"; import { useEnvironmentQuery } from "../state/query"; import { environmentShell } from "../state/shell"; -import { AutoResumeOverlay } from "../coil/AutoResumeOverlay"; +import { ThreadCoilOverlay } from "../coil/ThreadCoilOverlay"; function ChatThreadRouteView() { const navigate = useNavigate(); @@ -89,7 +89,7 @@ function ChatThreadRouteView() { routeKind="server" threadSyncPhase={threadSyncPhase} /> - + ) : null} diff --git a/apps/web/src/routes/settings.loops.tsx b/apps/web/src/routes/settings.loops.tsx new file mode 100644 index 000000000000..b196b9ef12a1 --- /dev/null +++ b/apps/web/src/routes/settings.loops.tsx @@ -0,0 +1,11 @@ +import { createFileRoute } from "@tanstack/react-router"; + +import { LoopsSettingsPanel } from "../components/coil/LoopsSettings"; + +function SettingsLoopsRoute() { + return ; +} + +export const Route = createFileRoute("/settings/loops")({ + component: SettingsLoopsRoute, +}); diff --git a/docs/coil/SEAMS.md b/docs/coil/SEAMS.md index 411580e21cf3..301bc562b33a 100644 --- a/docs/coil/SEAMS.md +++ b/docs/coil/SEAMS.md @@ -2,10 +2,44 @@ **The authoritative list of every upstream-owned file this fork edits.** -Measured, not asserted: **53 upstream-owned files, +2609 / -981 lines**, against merge-base +Measured, not asserted: **58 upstream-owned files, +2654 / -982 lines**, against merge-base `941acb4f9` (the 2026-09-02 sync). Everything else the fork adds lives in new files upstream has never seen and cannot conflict. +> **2026-09-02 — the loops feature, phases 1 and 3–5: five rows, +45/-1.** +> `+2609/-981` (53 files) → `+2654/-982` (58 files). Four of the five are `+N/-0`; the fifth spends +> exactly one deletion, and it is worth knowing which and why. +> +> - **`ClaudeAdapter.ts` +3/-0 (row 44, phase 1)** was added to the table but the header totals were +> not updated in the same commit, against the self-reference rule below. Corrected here. +> - **`settingsSearch.ts` +14/-0 (row 45)** and **`SettingsSidebarNav.tsx` +2/-0 (row 46)** are the +> Settings → Loops section. Both are type-forced by the same `Readonly>`, +> so they land together or neither compiles. +> - **`routeTree.gen.ts` +21/-0 (row 47)** is generated, not written. It is listed because the +> regeneration recipe emits it, and it is resolved the way `pnpm-lock.yaml` is: regenerate, never +> merge. +> - **`McpHttpServer.ts` +5/-1 (row 48, phase 5)** is the only row that displaces an upstream line, +> and the whole of the fork's `-982`. The file's terminal `export const layer =` was a single +> `PreviewToolkitRegistrationLive.pipe(...)`; registering a second toolkit needs a +> `Layer.mergeAll(...)` around it, so that one line is rewritten rather than added beside. There +> is no additive form of it — a second `export const layer` would shadow the first — so this is a +> displacement the invariant cannot avoid, not one it caught. **Re-read it at every sync:** an +> upstream change to that expression conflicts here, and taking either side wholesale silently +> drops one of the two toolkits. +> +> **The console cost zero rows, which was the point of the design.** `_chat.$environmentId.$threadId.tsx` +> is still **+10/−6** — phase 3 swapped one JSX element and one import for ``, a +> fork-owned aggregator, so every future per-thread fork surface is now free. `SettingsPanels.tsx` +> (churn 43) and `packages/contracts/src/settings.ts` (churn 38, persisted) were deliberately not +> touched: loop settings live in `coil-loop.json` behind fork-owned routes. +> +> **Two upstream files were deliberately left alone and are worth recording as refusals.** +> `docs/README.md` would take one bullet to link `docs/user/loops.md`, and +> `docs/internals/glossary.md` four terms of loop vocabulary. Neither has a fork edit today, so +> either would open a **new row for prose** against the tripwire below — the vocabulary is in +> `docs/coil/loops-v2/PLAN.md` instead. The user page is therefore currently unlinked from the docs +> index; that is a maintainer call, not an oversight. + > **Re-baselined 2026-09-02** against `941acb4f9`, after a **182-commit** upstream range > (issue #128, which escalated on a `pnpm-lock.yaml` conflict). Both halves were regenerated > against the same merge-base. The footprint moved from **+2613/-1044 to +2609/-981**, and the @@ -463,61 +497,66 @@ upstream never touches is cheap, and a two-line edit to a file upstream rewrites Sorted by risk, worst first. -| Upstream file | fork Δ | churn | risk | Why the fork touches it | -| ----------------------------------------------------------------------- | --------- | ----- | --------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -| `pnpm-lock.yaml` | +459/-802 | 66 | **83226** | Web Push adds `web-push` + `@types/web-push`; update delivery adds the `coil-update-relay` workspace entry (+6, `effect` + `@cloudflare/workers-types` only — `wrangler` is run via `pnpm dlx` precisely to keep it out of here, it would have cost ~500); the electron pin is retired (upstream's 43.4.1 supersedes it); the 2026-08-08 security sweep re-floats astro / postcss / svgo / js-yaml / undici@7 and applies the `overrides:` block below; the 2026-08-11 advisory pass (#58) re-floats sharp 0.35.3, `@modelcontextprotocol/sdk` 1.30.0 and `@hono/node-server` 2.1.0 — **all three inside ranges their parents already declare, so none of them buys an `overrides:` entry**. Net **-420 lines** — astro 7.2.0 drops its old remark/rehype/hast pipeline. Unavoidable and always conflicts; **regenerate rather than merge, seeding from the fork's pre-sync lock and never upstream's**, which is why this row's risk number overstates it — see the note under the header | -| `apps/web/src/components/ChatView.tsx` | +230/-1 | 114 | **26334** | Thread outbox: `handleQueueComposerSubmission`, queue-mode state, `onSend` early-return, ``, `sendLabel`, steer-vs-queue predicate (keyed on the thread's routing binding since #40 A4), per-dispatch-point breadcrumb (which moved below upstream's three new attachment bail-outs, since it must only fire for a submit that really dispatched). The queue path counts **both** attachment classes since upstream #8236 — counting images alone left the Queue button enabled and inert on a files-only composer, and dropped the file on clear | -| `apps/web/src/components/settings/SettingsPanels.tsx` | +58/-0 | 43 | **2494** | Needs-input notifications: import, 3 restore-reducer entries, permission state, a 45-line ``. **Not registered in upstream's new settings-search catalog** — see the note below the table | -| `packages/client-runtime/src/connection/supervisor.test.ts` | +369/-0 | 5 | **1845** | Issue #21: 356-line appended `describe` + harness plumbing | -| `apps/mobile/src/features/threads/ThreadComposer.tsx` | +36/-9 | 28 | **1260** | Mobile Return-key send/queue, plus the line-break toolbar button. Upstream #5625 rewrote this file (-124/+57), replacing `ControlPillMenu` / `buildModelMenuActions` / the provider-option menus with a single `ThreadSettingsSheet` trigger. Resolution keeps upstream's one trigger and re-attaches the fork's line-break button beside it; the fork's model-menu plumbing is gone because the thing it plugged into is gone | -| `packages/contracts/src/ipc.ts` | +44/-0 | 28 | **1232** | `DesktopNotificationRequest` / `Activation` + two optional `DesktopBridge` members; **coil update delivery** adds one type import, one `export type` re-export and a third optional member (`coilUpdate`). Its interfaces live in fork-owned `src/coil/updateDelivery.ts` | -| `apps/server/src/serverRuntimeStartup.test.ts` | +173/-1 | 6 | **1044** | Crash-recovery reconciler coverage | -| `apps/web/src/components/chat/ChatComposer.tsx` | +16/-2 | 57 | **1026** | Threads `sendLabel` / `canQueue` through the composer | -| `packages/client-runtime/src/connection/supervisor.ts` | +195/-66 | 3 | **783** | Issue #21: in-place rewrite of the reconnect/backoff state machine; now also owns the shared `runLivenessProbe` helper upstream's probe path uses. **2026-08-08: that helper had silently dropped upstream's #5561 behaviour.** Upstream marks `wakeProbeFailed` when a _wake_ probe fails, and reads it to reconnect immediately instead of sleeping the first backoff rung; the fork's helper is shared with the heartbeat path and never set the flag, so the Ref was written by nobody and read as always-false. `runLivenessProbe` now takes an `isWakeProbe` argument and only the wake call site passes it, matching where upstream sets it | -| `pnpm-workspace.yaml` | +23/-0 | 25 | **575** | **Row 37, added 2026-08-08.** 13 major-scoped entries appended to upstream's existing `overrides:` block: brace-expansion ×3 lines, builder-util-runtime, fast-uri, form-data, hono, ip-address, nanoid@3, path-to-regexp, shell-quote, tar, undici@6. These are the transitive advisories Dependabot cannot auto-fix — it only ever bumps a `package.json`. Together with the re-resolution pass they took the fork from **107 open alerts to 6**. Additive and contiguous inside a block upstream already owns, so it conflicts as one hunk. This is the row that carries the sweep across a sync — the lockfile is regenerated from it. Drop entries as upstream's tree floats past them | -| `apps/web/src/components/chat/ComposerPrimaryActions.tsx` | +74/-5 | 7 | **553** | Queue button, and the running-turn footer that pairs Stop with either Queue or Send. **2026-08-17: the fork's hoist of upstream's send button was deleted, because upstream #4781 hoisted it to a `const` itself.** That retires the stale-hoist hazard recorded here since 2026-08-08 — a copy cannot conflict, so it reverted upstream restyles silently. What is left is upstream's own button plus one aria-label branch, and the fork now extends upstream's running/idle dispatch instead of displacing it (-55 → -5). Stop still carries the fork's _emphasis_ axis (quiet outline beside Queue) on top of upstream's _size_ axis | -| `docs/user/providers-claude.md` | +86/-33 | 4 | **476** | Fixes the broken multi-account recipe. Upstream renamed this from `docs/providers/claude.md` in #4807 — **the one row worth upstreaming**, which would remove it | -| `scripts/build-desktop-artifact.ts` | +32/-1 | 14 | **462** | **Two env hooks, one displaced line, no new deletions.** Issue #70: `DESKTOP_APP_ID` reads `process.env.T3X_DESKTOP_APP_ID` before falling back to upstream's `com.t3tools.t3code`, so the fork's app owns its own TCC permission rows instead of sharing them with upstream's nightly. Issue #53: `DESKTOP_FILE_EXCLUSIONS` appends `process.env.T3X_DESKTOP_FILE_EXCLUSIONS` (comma-separated globs), taking the packaged asar from 189.66 MiB / 14,765 files to 99.02 MiB / 3,429 and the `.zip` users download by 20.1 MB. Most of the added lines are the comments explaining both. Env hooks rather than changed literals on purpose. For #53 the reason is not row count — `build-desktop-artifact.test.ts` is already a row (see #71 above) — but that the fork's list is **67 globs and grows**: an inline list would make every future size fix an edit to an upstream TEST assertion, resolved by hand at every sync. Through the environment, an unset environment packages precisely what upstream packages and upstream's `deepStrictEqual` keeps passing untouched. Guarded from the fork side by `scripts/coil/mac-signature.test.ts` and `scripts/coil/desktop-bundle-size.test.ts` (both hooks exist, the release workflow sets them, the exclusions never name a package the main process loads) and by two artifact checks: `verify-mac-signature.ts` and `verify-desktop-bundle.mjs` (the shipped app still resolves every import its own bundles make) | -| `apps/desktop/src/preload.ts` | +29/-0 | 14 | **406** | `showNotification` + `onNotificationActivated` on the exposed bridge, plus the `coilUpdate` bridge object (get / subscribe / restart / dismiss) | -| `packages/contracts/src/settings.ts` | +7/-2 | 38 | **342** | `notifyOnNeedsInput` (**persisted schema**) + Claude `homePath` placeholder/description | -| `apps/web/src/branding.test.ts` | +44/-11 | 3 | **165** | #71: app-name fixtures. The injected-branding case deliberately keeps `T3 Code` — it asserts injection WINS over the module constant, so matching the fixture to the constant would make it pass either way. Grew again for the sidebar wordmark: two cases pinning `APP_WORDMARK_SUFFIX`, one on the module constant and one on injected branding | -| `apps/mobile/modules/t3-composer-editor/ios/T3ComposerEditorView.swift` | +37/-0 | 4 | **148** | Shift+Return newline vs. bare Return submit | -| `scripts/build-desktop-artifact.test.ts` | +9/-2 | 13 | **143** | #71: asserts `resolveDesktopProductName` returns the fork's name. Upstream's `T3 Code (Nightly)` literal stays — that branch needs a `-nightly..` version, which this fork never builds | -| `apps/desktop/src/backend/DesktopBackendConfiguration.test.ts` | +41/-0 | 3 | **123** | Heap-headroom assertions | -| `apps/server/src/sourceControl/SourceControlRepositoryService.ts` | +101/-19 | 1 | **120** | **Row 42, added 2026-08-12 (#98).** Derives the provider from the remote URL instead of reporting `unknown`, runs the clone with `allowNonZeroExit` so git's stderr can be classified before it is thrown away, and passes the non-interactive env so a credential prompt fails in seconds rather than hanging to the 120 s timeout. All 19 deletions are the defect itself — see the note under the header. The logic lives in the fork-added `cloneDiagnostics.ts`, which costs no row | -| `apps/web/src/routes/__root.tsx` | +9/-0 | 12 | **108** | Mounts ``, ``, ``, `` | -| `README.md` | +15/-0 | 7 | **105** | **Row 39, added 2026-08-11 (#72).** A callout at the top of Installation saying this repo is a fork whose builds live at coil.curlycloud.dev, that the winget/brew/AUR commands below install upstream's app instead, and the honest platform matrix (macOS arm64, Windows x64, no Linux). Purely inserted — upstream's own text is untouched, so this stays a +N/-0 row | -| `apps/mobile/…/T3ComposerEditorView.kt` | +51/-0 | 2 | **102** | Android bare-Enter intercept | -| `docs/user/install.md` | +17/-0 | 6 | **102** | **Row 40, added 2026-08-11 (#72).** Same callout, for readers who reach the inherited install guide rather than the README. Also states the Gatekeeper _damaged_ wording, since this is the page someone lands on after searching for it. Inserted above upstream's first line | -| `apps/desktop/src/ipc/channels.ts` | +9/-0 | 11 | **99** | Two notification channel constants + four `coil:update-*` constants. Deliberately **not** reusing upstream's `desktop:update-*` channels | -| `apps/server/src/serverRuntimeStartup.ts` | +44/-2 | 2 | **92** | `reconcile.interrupted-turns` startup phase, inside upstream's `startup` effect — plus upstream's own `provider-sessions.reconcile` hoisted above it (see the 2026-08-27 note) | -| `apps/desktop/src/backend/DesktopBackendConfiguration.ts` | +29/-0 | 3 | **87** | Backend heap headroom (`NODE_OPTIONS`) | -| `AGENTS.md` | +6/-0 | 14 | **84** | `## Agent skills` pointer block for the mattpocock engineering skills. Three one-line links into `docs/coil/agents/`; no config lives here. Placed between `## How it works` and `## Where code lives` — stable anchors, deliberately not appended at EOF where upstream adds tips (the issue #29 add/add pattern) | -| `apps/server/src/server.ts` | +3/-0 | 27 | **81** | The intended mount point: one import, one `Layer.provideMerge`, one route entry | -| `apps/desktop/src/ipc/DesktopIpcHandlers.ts` | +11/-0 | 7 | **77** | Registers the `showNotification` handler + three `coilUpdate` handlers (`getCoilUpdateState`, `restartIntoUpdate`, `dismissCoilUpdate`) | -| `apps/web/src/components/sidebar/SidebarChrome.tsx` | +3/-1 | 19 | **76** | The sidebar corner, the one string the running app names itself with. Renders `APP_WORDMARK_SUFFIX` instead of the literal `Code`, so the name resolves from the desktop bundle's injected branding rather than from a second copy. **Not hoisted into a fork component** — see the note below the table | -| `docs/user/source-control.md` | +8/-0 | 9 | **72** | **Row 43, added 2026-08-12 (#98).** States that cloning uses the Git credentials on the machine running T3 Code, not the provider API tokens above it — the confusion the Bitbucket report started from — plus a "Clone failed" troubleshooting entry. Purely inserted | -| `apps/desktop/src/main.ts` | +10/-0 | 6 | **60** | `ElectronNotification` layer + the `T3xUpdateDelivery` layer | -| `apps/desktop/src/app/DesktopEnvironment.ts` | +9/-1 | 5 | **50** | #71: `APP_BASE_NAME`, the source of truth for the visible name — `displayName` derives from it and reaches `app.setName()`, the About panel, window titles, the Linux `.desktop` entry and the whole web UI via `getAppBranding()`. The other 8 lines are a comment on why `legacyUserDataDirName` two lines below must NOT follow it | -| `apps/web/src/connection/platform.ts` | +7/-1 | 6 | **48** | Lazy `import()` of outbox cleanup to dodge a module-init cycle | -| `apps/desktop/src/app/DesktopAppIdentity.test.ts` | +5/-2 | 6 | **42** | #71: asserts the new name for `setName`/About while KEEPING the old one in the legacy-userData assertions. The pair looks like a typo and is not — one is computed, one names a directory already on disk | -| `apps/web/index.html` | +5/-1 | 6 | **36** | PWA manifest + meta tags. #71 replaces the boot ``, the file's only deletion | -| `apps/web/src/routes/_chat.$environmentId.$threadId.tsx` | +10/-6 | 2 | **32** | Mounts `<AutoResumeOverlay>` as a sibling of `<ChatView>` inside upstream's render-state conditional | -| `apps/desktop/package.json` | +1/-1 | 13 | **26** | `productName` renamed for #71. **The electron pin is gone**: upstream #8626 moved to 43.4.1, two majors past the fork's 41.10.3, so the advisory the pin closed is closed by upstream's own version | -| `apps/server/package.json` | +2/-0 | 13 | **26** | `web-push` dependency | -| `apps/web/public/manifest.webmanifest` | +21/-0 | 1 | **21** | **New row 2026-08-17, and not new work.** Upstream now ships its own manifest, so a previously fork-owned file became an add/add seam. Upstream's has no `name`/`short_name`/`description` and no maskable icon — an installable PWA needs all four, which is what Web Push (#23) rides on — so the fork's fields are unioned onto upstream's. The fork's duplicate `apple-touch-icon` entry was dropped in favour of upstream's identical one | -| `apps/desktop/src/settings/DesktopClientSettings.test.ts` | +1/-0 | 19 | **19** | `notifyOnNeedsInput` in a fixture | -| `apps/mobile/src/components/AppSymbol.tsx` | +2/-0 | 9 | **18** | `return:` icon entry | -| `apps/server/src/sourceControl/SourceControlProviderDiscovery.ts` | +12/-5 | 1 | **17** | Issue #4. **2026-08-17: the timeout half was ceded to upstream**, which fixed it independently in #6223 and better — a per-spec `probeTimeoutMs` with `az` at 20s, against the fork's global 15s constant, now deleted. The remaining seam is only the spawn-error classification: solely `VcsProcessSpawnError` means "missing", so a slow-but-present CLI stays "available" and the auth probe still runs. Upstream still does not do this | -| `apps/mobile/src/native/T3ComposerEditor.native.tsx` | +3/-0 | 5 | **15** | Plumbs `onComposerSubmit`, now inside upstream's `<TextInputWrapper>` paste shell | -| `apps/mobile/src/native/T3ComposerEditor.types.ts` | +5/-1 | 2 | **12** | Reworded `onSubmit` doc comment | -| `apps/web/src/components/chat/ComposerPrimaryActions.test.tsx` | +5/-1 | 2 | **12** | **New row 2026-08-17.** One assertion. Upstream #4781 added a test expecting the running-turn send button to carry `aria-label="Send message"`; the fork lengthens exactly that label to "Send message to the running turn" (#35), because on a steer-capable driver the submit folds into the work in progress and nothing else on screen says so. Upstream's actual subject — that a submit button renders beside Stop, at the larger size — is still pinned by the surrounding assertions, which are untouched | -| `apps/desktop/scripts/electron-launcher.mjs` | +1/-1 | 2 | **4** | #71: the dev-mode display name. Dev-only — a packaged build never loads this file | -| `apps/desktop/src/app/DesktopLinuxUrlHandler.test.ts` | +1/-1 | 1 | **2** | #71: app-name fixture | -| `apps/mobile/…/T3ComposerEditorModule.kt` | +1/-0 | 1 | **1** | Event-name list entry | -| `apps/server/src/sourceControl/SourceControlRepositoryService.test.ts` | +194/-1 | 0 | **0** | **Row 41, added 2026-08-12 (#98).** Coverage for each mapped clone failure — auth, not-found, timeout, unwritable destination — plus one asserting a credential-bearing URL is redacted out of the surfaced message, and one asserting the clone spawn is handed the non-interactive env. The single deletion is the assertion that a `git@github.com:` clone reports provider `unknown`; it encoded the bug | -| `apps/web/src/branding.ts` | +15/-1 | 0 | **0** | #71: the browser-only fallback, used when no desktop branding is injected. **+14 additive** for `APP_WORDMARK_SUFFIX`, which the sidebar consumes so the app's name is resolved once rather than written out a second time in an upstream component | -| `packages/shared/src/composerTrigger.test.ts` | +31/-1 | 0 | **0** | `replaceTextRange` newline coverage | +| Upstream file | fork Δ | churn | risk | Why the fork touches it | +| ----------------------------------------------------------------------- | --------- | ----- | --------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| `pnpm-lock.yaml` | +459/-802 | 66 | **83226** | Web Push adds `web-push` + `@types/web-push`; update delivery adds the `coil-update-relay` workspace entry (+6, `effect` + `@cloudflare/workers-types` only — `wrangler` is run via `pnpm dlx` precisely to keep it out of here, it would have cost ~500); the electron pin is retired (upstream's 43.4.1 supersedes it); the 2026-08-08 security sweep re-floats astro / postcss / svgo / js-yaml / undici@7 and applies the `overrides:` block below; the 2026-08-11 advisory pass (#58) re-floats sharp 0.35.3, `@modelcontextprotocol/sdk` 1.30.0 and `@hono/node-server` 2.1.0 — **all three inside ranges their parents already declare, so none of them buys an `overrides:` entry**. Net **-420 lines** — astro 7.2.0 drops its old remark/rehype/hast pipeline. Unavoidable and always conflicts; **regenerate rather than merge, seeding from the fork's pre-sync lock and never upstream's**, which is why this row's risk number overstates it — see the note under the header | +| `apps/web/src/components/ChatView.tsx` | +230/-1 | 114 | **26334** | Thread outbox: `handleQueueComposerSubmission`, queue-mode state, `onSend` early-return, `<ThreadOutboxQueueList>`, `sendLabel`, steer-vs-queue predicate (keyed on the thread's routing binding since #40 A4), per-dispatch-point breadcrumb (which moved below upstream's three new attachment bail-outs, since it must only fire for a submit that really dispatched). The queue path counts **both** attachment classes since upstream #8236 — counting images alone left the Queue button enabled and inert on a files-only composer, and dropped the file on clear | +| `apps/web/src/components/settings/SettingsPanels.tsx` | +58/-0 | 43 | **2494** | Needs-input notifications: import, 3 restore-reducer entries, permission state, a 45-line `<SettingsRow>`. **Not registered in upstream's new settings-search catalog** — see the note below the table | +| `packages/client-runtime/src/connection/supervisor.test.ts` | +369/-0 | 5 | **1845** | Issue #21: 356-line appended `describe` + harness plumbing | +| `apps/mobile/src/features/threads/ThreadComposer.tsx` | +36/-9 | 28 | **1260** | Mobile Return-key send/queue, plus the line-break toolbar button. Upstream #5625 rewrote this file (-124/+57), replacing `ControlPillMenu` / `buildModelMenuActions` / the provider-option menus with a single `ThreadSettingsSheet` trigger. Resolution keeps upstream's one trigger and re-attaches the fork's line-break button beside it; the fork's model-menu plumbing is gone because the thing it plugged into is gone | +| `packages/contracts/src/ipc.ts` | +44/-0 | 28 | **1232** | `DesktopNotificationRequest` / `Activation` + two optional `DesktopBridge` members; **coil update delivery** adds one type import, one `export type` re-export and a third optional member (`coilUpdate`). Its interfaces live in fork-owned `src/coil/updateDelivery.ts` | +| `apps/server/src/serverRuntimeStartup.test.ts` | +173/-1 | 6 | **1044** | Crash-recovery reconciler coverage | +| `apps/web/src/components/chat/ChatComposer.tsx` | +16/-2 | 57 | **1026** | Threads `sendLabel` / `canQueue` through the composer | +| `packages/client-runtime/src/connection/supervisor.ts` | +195/-66 | 3 | **783** | Issue #21: in-place rewrite of the reconnect/backoff state machine; now also owns the shared `runLivenessProbe` helper upstream's probe path uses. **2026-08-08: that helper had silently dropped upstream's #5561 behaviour.** Upstream marks `wakeProbeFailed` when a _wake_ probe fails, and reads it to reconnect immediately instead of sleeping the first backoff rung; the fork's helper is shared with the heartbeat path and never set the flag, so the Ref was written by nobody and read as always-false. `runLivenessProbe` now takes an `isWakeProbe` argument and only the wake call site passes it, matching where upstream sets it | +| `pnpm-workspace.yaml` | +23/-0 | 25 | **575** | **Row 37, added 2026-08-08.** 13 major-scoped entries appended to upstream's existing `overrides:` block: brace-expansion ×3 lines, builder-util-runtime, fast-uri, form-data, hono, ip-address, nanoid@3, path-to-regexp, shell-quote, tar, undici@6. These are the transitive advisories Dependabot cannot auto-fix — it only ever bumps a `package.json`. Together with the re-resolution pass they took the fork from **107 open alerts to 6**. Additive and contiguous inside a block upstream already owns, so it conflicts as one hunk. This is the row that carries the sweep across a sync — the lockfile is regenerated from it. Drop entries as upstream's tree floats past them | +| `apps/web/src/components/chat/ComposerPrimaryActions.tsx` | +74/-5 | 7 | **553** | Queue button, and the running-turn footer that pairs Stop with either Queue or Send. **2026-08-17: the fork's hoist of upstream's send button was deleted, because upstream #4781 hoisted it to a `const` itself.** That retires the stale-hoist hazard recorded here since 2026-08-08 — a copy cannot conflict, so it reverted upstream restyles silently. What is left is upstream's own button plus one aria-label branch, and the fork now extends upstream's running/idle dispatch instead of displacing it (-55 → -5). Stop still carries the fork's _emphasis_ axis (quiet outline beside Queue) on top of upstream's _size_ axis | +| `docs/user/providers-claude.md` | +86/-33 | 4 | **476** | Fixes the broken multi-account recipe. Upstream renamed this from `docs/providers/claude.md` in #4807 — **the one row worth upstreaming**, which would remove it | +| `scripts/build-desktop-artifact.ts` | +32/-1 | 14 | **462** | **Two env hooks, one displaced line, no new deletions.** Issue #70: `DESKTOP_APP_ID` reads `process.env.T3X_DESKTOP_APP_ID` before falling back to upstream's `com.t3tools.t3code`, so the fork's app owns its own TCC permission rows instead of sharing them with upstream's nightly. Issue #53: `DESKTOP_FILE_EXCLUSIONS` appends `process.env.T3X_DESKTOP_FILE_EXCLUSIONS` (comma-separated globs), taking the packaged asar from 189.66 MiB / 14,765 files to 99.02 MiB / 3,429 and the `.zip` users download by 20.1 MB. Most of the added lines are the comments explaining both. Env hooks rather than changed literals on purpose. For #53 the reason is not row count — `build-desktop-artifact.test.ts` is already a row (see #71 above) — but that the fork's list is **67 globs and grows**: an inline list would make every future size fix an edit to an upstream TEST assertion, resolved by hand at every sync. Through the environment, an unset environment packages precisely what upstream packages and upstream's `deepStrictEqual` keeps passing untouched. Guarded from the fork side by `scripts/coil/mac-signature.test.ts` and `scripts/coil/desktop-bundle-size.test.ts` (both hooks exist, the release workflow sets them, the exclusions never name a package the main process loads) and by two artifact checks: `verify-mac-signature.ts` and `verify-desktop-bundle.mjs` (the shipped app still resolves every import its own bundles make) | +| `apps/desktop/src/preload.ts` | +29/-0 | 14 | **406** | `showNotification` + `onNotificationActivated` on the exposed bridge, plus the `coilUpdate` bridge object (get / subscribe / restart / dismiss) | +| `packages/contracts/src/settings.ts` | +7/-2 | 38 | **342** | `notifyOnNeedsInput` (**persisted schema**) + Claude `homePath` placeholder/description | +| `apps/web/src/components/settings/settingsSearch.ts` | +14/-0 | 24 | **336** | **Row 45, added 2026-09-02 (loops phase 4).** The `SettingsPath` union member for `/settings/loops`, its `SETTINGS_SECTION_LABELS` entry, and two `SETTINGS_SEARCH_ITEMS` objects whose ids the fork-owned panel spreads through `searchableSetting`, so the catalog and the rendered rows cannot drift apart. **Type-forced, not stylistic:** `SETTINGS_SECTION_LABELS` and `SETTINGS_SECTION_ICONS` are both `Readonly<Record<SettingsPath, …>>` and `SETTINGS_NAV_ITEMS` derives from the label record's keys, so a `SettingsPath` member without both entries is a type error. It buys the **command palette for free** — `CommandPalette.tsx` builds its settings results from `searchSettings` over this same catalog — which is why the loops feature adds no palette or keybinding row. The most expensive row the feature takes, and it buys discoverability rather than function. **Zero-row fallback if it is ever refused:** a fork route reached only from the console, the way `/settings/diagnostics` is reached — a real settings route that is deliberately **not** a `SettingsPath` member. Fully additive | +| `apps/web/src/routeTree.gen.ts` | +21/-0 | 9 | **189** | **Row 47, added 2026-09-02 (loops phase 4).** Generated by the TanStack Router plugin from `apps/web/src/routes/`, and listed only because the regeneration recipe at the bottom of this file emits it. Same instruction as `pnpm-lock.yaml`: **regenerate, never merge** — every line is the `/settings/loops` entry that the fork-owned `routes/settings.loops.tsx` produces, and deleting that file and re-running the generator removes the whole row. The risk number overstates it: a conflict here is resolved by running the generator, never by reading the diff | +| `apps/web/src/branding.test.ts` | +44/-11 | 3 | **165** | #71: app-name fixtures. The injected-branding case deliberately keeps `T3 Code` — it asserts injection WINS over the module constant, so matching the fixture to the constant would make it pass either way. Grew again for the sidebar wordmark: two cases pinning `APP_WORDMARK_SUFFIX`, one on the module constant and one on injected branding | +| `apps/mobile/modules/t3-composer-editor/ios/T3ComposerEditorView.swift` | +37/-0 | 4 | **148** | Shift+Return newline vs. bare Return submit | +| `scripts/build-desktop-artifact.test.ts` | +9/-2 | 13 | **143** | #71: asserts `resolveDesktopProductName` returns the fork's name. Upstream's `T3 Code (Nightly)` literal stays — that branch needs a `-nightly.<d>.<d>` version, which this fork never builds | +| `apps/desktop/src/backend/DesktopBackendConfiguration.test.ts` | +41/-0 | 3 | **123** | Heap-headroom assertions | +| `apps/server/src/sourceControl/SourceControlRepositoryService.ts` | +101/-19 | 1 | **120** | **Row 42, added 2026-08-12 (#98).** Derives the provider from the remote URL instead of reporting `unknown`, runs the clone with `allowNonZeroExit` so git's stderr can be classified before it is thrown away, and passes the non-interactive env so a credential prompt fails in seconds rather than hanging to the 120 s timeout. All 19 deletions are the defect itself — see the note under the header. The logic lives in the fork-added `cloneDiagnostics.ts`, which costs no row | +| `apps/web/src/routes/__root.tsx` | +9/-0 | 12 | **108** | Mounts `<NotificationCoordinator>`, `<ThreadOutboxDrain>`, `<PushSubscriptionManager>`, `<T3xUpdateToast>` | +| `README.md` | +15/-0 | 7 | **105** | **Row 39, added 2026-08-11 (#72).** A callout at the top of Installation saying this repo is a fork whose builds live at coil.curlycloud.dev, that the winget/brew/AUR commands below install upstream's app instead, and the honest platform matrix (macOS arm64, Windows x64, no Linux). Purely inserted — upstream's own text is untouched, so this stays a +N/-0 row | +| `apps/mobile/…/T3ComposerEditorView.kt` | +51/-0 | 2 | **102** | Android bare-Enter intercept | +| `docs/user/install.md` | +17/-0 | 6 | **102** | **Row 40, added 2026-08-11 (#72).** Same callout, for readers who reach the inherited install guide rather than the README. Also states the Gatekeeper _damaged_ wording, since this is the page someone lands on after searching for it. Inserted above upstream's first line | +| `apps/desktop/src/ipc/channels.ts` | +9/-0 | 11 | **99** | Two notification channel constants + four `coil:update-*` constants. Deliberately **not** reusing upstream's `desktop:update-*` channels | +| `apps/server/src/serverRuntimeStartup.ts` | +44/-2 | 2 | **92** | `reconcile.interrupted-turns` startup phase, inside upstream's `startup` effect — plus upstream's own `provider-sessions.reconcile` hoisted above it (see the 2026-08-27 note) | +| `apps/desktop/src/backend/DesktopBackendConfiguration.ts` | +29/-0 | 3 | **87** | Backend heap headroom (`NODE_OPTIONS`) | +| `AGENTS.md` | +6/-0 | 14 | **84** | `## Agent skills` pointer block for the mattpocock engineering skills. Three one-line links into `docs/coil/agents/`; no config lives here. Placed between `## How it works` and `## Where code lives` — stable anchors, deliberately not appended at EOF where upstream adds tips (the issue #29 add/add pattern) | +| `apps/server/src/server.ts` | +3/-0 | 27 | **81** | The intended mount point: one import, one `Layer.provideMerge`, one route entry | +| `apps/desktop/src/ipc/DesktopIpcHandlers.ts` | +11/-0 | 7 | **77** | Registers the `showNotification` handler + three `coilUpdate` handlers (`getCoilUpdateState`, `restartIntoUpdate`, `dismissCoilUpdate`) | +| `apps/web/src/components/sidebar/SidebarChrome.tsx` | +3/-1 | 19 | **76** | The sidebar corner, the one string the running app names itself with. Renders `APP_WORDMARK_SUFFIX` instead of the literal `Code`, so the name resolves from the desktop bundle's injected branding rather than from a second copy. **Not hoisted into a fork component** — see the note below the table | +| `docs/user/source-control.md` | +8/-0 | 9 | **72** | **Row 43, added 2026-08-12 (#98).** States that cloning uses the Git credentials on the machine running T3 Code, not the provider API tokens above it — the confusion the Bitbucket report started from — plus a "Clone failed" troubleshooting entry. Purely inserted | +| `apps/server/src/provider/Layers/ClaudeAdapter.ts` | +3/-0 | 23 | **69** | **Row 44, added 2026-09-02 (loops phase 1).** The only upstream edit the loops feature takes. One import, one `const loopHooks = yield* loopHooksFor(threadId)` inside `startSession`, and `...(loopHooks ? { hooks: loopHooks } : {})` beside the `mcpServers` spread in `queryOptions`. It buys the fork a durable copy of the agent's own scheduled wakes (`session_crons` on the `Stop` / `SubagentStop` hooks), which is the one thing T3 can supply and the binary cannot: `cron_durable` is false, so the provider's cron table is in-process and a wake lost to a restart otherwise leaves no trace. `loopHooksFor` reads `LoopStore` through `Effect.serviceOption`, so the adapter's layer requirements do not widen and no second row appears in `server.ts` or upstream's adapter tests. That keeps the row at one line but does **not** make it work: the adapter's fiber runs in upstream's layer graph, where `CoilLayerLive` has already discharged `LoopStore` with `Layer.provide`, so in production the service is genuinely absent and the option is always `None`. The supervisor therefore publishes its store in a fork-owned process-level holder (`coil/loop/hooksRegistry.ts`) for the life of its scope, and `loopHooksFor` falls back to it — context first, so every test that provides the store stays honest. A server built without the loop layer finds neither and gets `undefined`, and no `hooks` key is written. Fully additive, read-only, and it fails to a **type error** rather than silent drift. The neighbourhood is live: upstream #8144 added `onUserDialog` / `supportedDialogKinds` to this same object, so re-read the spread's surroundings every sync | +| `apps/desktop/src/main.ts` | +10/-0 | 6 | **60** | `ElectronNotification` layer + the `T3xUpdateDelivery` layer | +| `apps/desktop/src/app/DesktopEnvironment.ts` | +9/-1 | 5 | **50** | #71: `APP_BASE_NAME`, the source of truth for the visible name — `displayName` derives from it and reaches `app.setName()`, the About panel, window titles, the Linux `.desktop` entry and the whole web UI via `getAppBranding()`. The other 8 lines are a comment on why `legacyUserDataDirName` two lines below must NOT follow it | +| `apps/web/src/connection/platform.ts` | +7/-1 | 6 | **48** | Lazy `import()` of outbox cleanup to dodge a module-init cycle | +| `apps/desktop/src/app/DesktopAppIdentity.test.ts` | +5/-2 | 6 | **42** | #71: asserts the new name for `setName`/About while KEEPING the old one in the legacy-userData assertions. The pair looks like a typo and is not — one is computed, one names a directory already on disk | +| `apps/web/src/components/settings/SettingsSidebarNav.tsx` | +2/-0 | 19 | **38** | **Row 46, added 2026-09-02 (loops phase 4).** One icon import (`RefreshCwIcon`) and its `SETTINGS_SECTION_ICONS` entry for `/settings/loops`. Type-forced by the same `Readonly<Record<SettingsPath, …>>` that forces row 45, so the two land together or neither compiles. The smallest possible shape of a settings section, and fully additive | +| `apps/web/index.html` | +5/-1 | 6 | **36** | PWA manifest + meta tags. #71 replaces the boot `<title>`, the file's only deletion | +| `apps/web/src/routes/_chat.$environmentId.$threadId.tsx` | +10/-6 | 2 | **32** | Mounts `<ThreadCoilOverlay>` as a sibling of `<ChatView>` inside upstream's render-state conditional. **2026-09-02 (loops phase 3):** the element and its import were swapped one-for-one from `<AutoResumeOverlay>`, so the +10/-6 is unchanged and adding the loop console cost this row nothing. `coil/ThreadCoilOverlay.tsx` is fork-owned and renders both overlays, so every future per-thread fork surface is free here | +| `apps/desktop/package.json` | +1/-1 | 13 | **26** | `productName` renamed for #71. **The electron pin is gone**: upstream #8626 moved to 43.4.1, two majors past the fork's 41.10.3, so the advisory the pin closed is closed by upstream's own version | +| `apps/server/package.json` | +2/-0 | 13 | **26** | `web-push` dependency | +| `apps/web/public/manifest.webmanifest` | +21/-0 | 1 | **21** | **New row 2026-08-17, and not new work.** Upstream now ships its own manifest, so a previously fork-owned file became an add/add seam. Upstream's has no `name`/`short_name`/`description` and no maskable icon — an installable PWA needs all four, which is what Web Push (#23) rides on — so the fork's fields are unioned onto upstream's. The fork's duplicate `apple-touch-icon` entry was dropped in favour of upstream's identical one | +| `apps/desktop/src/settings/DesktopClientSettings.test.ts` | +1/-0 | 19 | **19** | `notifyOnNeedsInput` in a fixture | +| `apps/mobile/src/components/AppSymbol.tsx` | +2/-0 | 9 | **18** | `return:` icon entry | +| `apps/server/src/sourceControl/SourceControlProviderDiscovery.ts` | +12/-5 | 1 | **17** | Issue #4. **2026-08-17: the timeout half was ceded to upstream**, which fixed it independently in #6223 and better — a per-spec `probeTimeoutMs` with `az` at 20s, against the fork's global 15s constant, now deleted. The remaining seam is only the spawn-error classification: solely `VcsProcessSpawnError` means "missing", so a slow-but-present CLI stays "available" and the auth probe still runs. Upstream still does not do this | +| `apps/mobile/src/native/T3ComposerEditor.native.tsx` | +3/-0 | 5 | **15** | Plumbs `onComposerSubmit`, now inside upstream's `<TextInputWrapper>` paste shell | +| `apps/mobile/src/native/T3ComposerEditor.types.ts` | +5/-1 | 2 | **12** | Reworded `onSubmit` doc comment | +| `apps/web/src/components/chat/ComposerPrimaryActions.test.tsx` | +5/-1 | 2 | **12** | **New row 2026-08-17.** One assertion. Upstream #4781 added a test expecting the running-turn send button to carry `aria-label="Send message"`; the fork lengthens exactly that label to "Send message to the running turn" (#35), because on a steer-capable driver the submit folds into the work in progress and nothing else on screen says so. Upstream's actual subject — that a submit button renders beside Stop, at the larger size — is still pinned by the surrounding assertions, which are untouched | +| `apps/server/src/mcp/McpHttpServer.ts` | +5/-1 | 2 | **12** | **Row 48, added 2026-09-02 (loops phase 5).** The fork's MCP toolkit (`raise_blocker`, `loop_status`, `loop_done`) has to be registered on the same `McpServer` the HTTP transport builds, so the terminal `export const layer` becomes `Layer.mergeAll(PreviewToolkitRegistrationLive, LoopToolkitRegistrationLive).pipe(...)` plus one import. Deliberately **not** a gated `"loop"` `McpCapability`: that would cost four rows and force a runtime `Schema.Literal("preview")` widening in `packages/contracts/src/previewAutomation.ts` (D12), and it would buy nothing — nothing at registration or dispatch consults `capabilities`. The real gate is `store.global.enabled` plus the armed record, checked inside the handlers. `LoopToolkitRegistrationLive` provides its own `LoopStore` (the shared `coil/loop/layer.ts` value), so `makeRoutesLayer`'s signature is unchanged and the fork does not leak a requirement into upstream's types. **Inherited coupling:** the MCP credential is only minted when `enableAgentBrowserAccess` is on, so switching agent browser access off removes all three tools — the console must name that degraded state rather than render an empty blocker list | +| `apps/desktop/scripts/electron-launcher.mjs` | +1/-1 | 2 | **4** | #71: the dev-mode display name. Dev-only — a packaged build never loads this file | +| `apps/desktop/src/app/DesktopLinuxUrlHandler.test.ts` | +1/-1 | 1 | **2** | #71: app-name fixture | +| `apps/mobile/…/T3ComposerEditorModule.kt` | +1/-0 | 1 | **1** | Event-name list entry | +| `apps/server/src/sourceControl/SourceControlRepositoryService.test.ts` | +194/-1 | 0 | **0** | **Row 41, added 2026-08-12 (#98).** Coverage for each mapped clone failure — auth, not-found, timeout, unwritable destination — plus one asserting a credential-bearing URL is redacted out of the surfaced message, and one asserting the clone spawn is handed the non-interactive env. The single deletion is the assertion that a `git@github.com:` clone reports provider `unknown`; it encoded the bug | +| `apps/web/src/branding.ts` | +15/-1 | 0 | **0** | #71: the browser-only fallback, used when no desktop branding is injected. **+14 additive** for `APP_WORDMARK_SUFFIX`, which the sidebar consumes so the app's name is resolved once rather than written out a second time in an upstream component | +| `packages/shared/src/composerTrigger.test.ts` | +31/-1 | 0 | **0** | `replaceTextRange` newline coverage | **Per surface:** `apps/web` 8 · `apps/mobile` 7 · `apps/desktop` 7 · `apps/server` 5 · `packages/**` 5 · `docs/` 1 · repo root 2. @@ -544,15 +583,15 @@ Sorted by risk, worst first. Upstream helpers the fork **replicates** rather than imports, to avoid a code seam. These never conflict during a sync, so nothing warns you when the original changes and the mirror drifts. -| Fork mirror | Mirrors upstream | Risk if upstream changes | -| --------------------------------------------------------------------------------------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -| `apps/server/src/coil/autoResume/guards.ts` (`hasOpenBlockingRequest`) | `decider.ts` (private, unexported) | Could miss a new blocking-request activity kind and auto-resume into a prompt. **Re-checked 2026-08-08: byte-identical.** Upstream's two commits to `decider.ts` this range were both thread pinning (#5312, #5581) and added no activity kind; the four `requested`/`resolved` kinds and the stale-failure escape hatch match exactly. **2026-09-02: `threadIsGone` in this file no longer reads `settledOverride`.** #8600 gave the server a one-minute sweep that dispatches `thread.auto-settle` through `thread.settle`'s own decider case, emitting an identical `thread.settled` with no provenance — so settledness stopped meaning "the user is done here" and was cancelling week-long arms on day three. Watch `ThreadSettlementReactor.ts` / `ThreadSettlementPolicy.ts`: if upstream ever adds a provenance marker, the "user settled it" cancel can come back. | -| `apps/server/src/coil/http/auth.ts` (`authenticateWithScope`) | `http.ts` (`authenticateRawRouteWithScope`, private, unexported) | Every fork raw route could authenticate more weakly than the routes beside it. **Re-checked 2026-09-02 (Phase 0): still accurate.** Previously re-checked 2026-08-08. Upstream's only commit to `apps/server/src/http.ts` this range was the Effect beta.103 upgrade (#5331), which deleted the gzip helpers and left `authenticateRawRouteWithScope` untouched. Compared line by line, the mirror matches upstream's behaviour with three intentional deltas: (1) it is exported, so fork routes import it instead of pasting it; (2) it returns the authenticated session, which upstream's discards, because the Web Push subscribe route records which device registered; (3) its scope parameter is typed `AuthEnvironmentScope` (all 8 literals) where upstream narrows to the two orchestration scopes — inherited from the webPush copy, and both callers pass orchestration scopes, so it is a wider type over identical use. **One mirror as of Phase 0:** it used to be pasted separately into `coil/autoResume/http.ts` and `coil/webPush/http.ts`, and the two had drifted (only auto-resume forwarded `dpopFailureReason`); both now call this module. | -| `apps/server/src/orchestration/Layers/CrashRecoveryReconciler.ts` (`getSnapshot()`) | `ProjectionSnapshotQuery.getSnapshot()` vs. the lighter `getCommandReadModel()` | **Live risk, found at the 2026-08-02 sync, re-confirmed unchanged 2026-08-08 — still not fixed.** Upstream moved its own orchestration-snapshot route off `getSnapshot()` onto `getCommandReadModel()` in this range, commenting that hydrating every message and activity payload "has OOM-killed servers". The fork's boot reconciler still calls `getSnapshot()`, and it runs on the startup path **before commands are accepted**, so an OOM there is a hard boot failure rather than one slow request. Both return `OrchestrationReadModel` and the reconciler only reads `thread.session` / `thread.latestTurn` metadata, so the swap looks like a drop-in — but it was deliberately left out of the sync commit and needs its own PR with coverage. | -| `apps/web/src/outbox/**` (thread outbox) | `apps/mobile/src/state/thread-outbox-*.ts` (upstream-authored, still maintained) | The web outbox is a hand port of upstream's mobile one, function for function. Two divergences are deliberate: the web queue drops image attachments (mobile persists them as base64 data URLs, which localStorage cannot hold) and orders on an explicit `sortKey` for user reordering where mobile sorts on `createdAt` alone. A third divergence closed at the 2026-08-14 sync: upstream #6543 made mobile steer active turns by default too (its outbox delivery gate is now connectivity-only — `thread-outbox-model.ts` dropped the `!threadBusy` term), so both platforms steer. Upstream reworking its mobile outbox produces no conflict here. | -| `apps/web/src/coil/AutoResumeOverlay.tsx` (`COMPOSER_OVERLAY_SELECTOR`, `chat-composer-horizontal-inset`) | `apps/web/src/components/ChatView.tsx` — the `[data-chat-composer-overlay="true"]` element it measures for `composerOverlayHeight`, and the `.chat-composer-horizontal-inset` class in `index.css` that the composer wrapper uses | **Read-only presentation dependencies, not code edits.** The auto-resume capsule is anchored bottom-right, immediately above the docked composer and flush with its right edge. The overlay mounts as a sibling of `<ChatView>` in the route file and cannot receive the composer's geometry as a prop without widening that seam, so it measures the data attribute with a `ResizeObserver` instead. Horizontal alignment additionally reuses the composer's own inset class, because that inset is `0.75rem` at base, `1.25rem` from `40rem` up, and carries `env(safe-area-inset-right)` — any hard-coded value overhangs on wide viewports. If the attribute disappears the capsule falls back to a fixed 76px offset; if the class is renamed the capsule's right edge drifts from the composer's. Both degrade visually, neither breaks. Re-check at each sync. **2026-08-10 (#67): the measurement now converts coordinate spaces rather than assuming they match.** The composer's `offsetParent` is the chat column; the capsule's is `SidebarInset`. Those boxes coincide only while no panel is open, so the original "measure against the composer's parent, apply against ours" shortcut stranded the capsule by 352px horizontally when the inline right panel opened and by the drawer's full height when the terminal drawer opened. `autoResumeAnchor.ts` now derives `bottom`/`left`/`width` from the composer's rect expressed in the capsule's own offsetParent coordinates, and all three boxes are observed. **This adds no new upstream dependency** — same selector, same class — but it does mean the capsule now depends on the chat column remaining the composer's nearest positioned ancestor. | -| `apps/web/src/outbox/composerSteering.logic.ts` (steer allowlist) | Each adapter's mid-turn `sendTurn` behaviour (ClaudeAdapter.ts:3729, CursorAdapter.ts:916, GrokAdapter.ts:921, OpenCodeAdapter.ts:1417) | **No capability flag exists** — `ProviderAdapterCapabilities` has no `supportsSteering`, so which drivers fold a mid-turn send into the running turn is asserted by a hand-maintained allowlist. If upstream changes an adapter to open a new turn instead, nothing fails here; the fork would keep sending mid-turn into a provider that no longer steers, and a refusal is invisible because `sendTurn` is forked in `ProviderCommandReactor`. Re-check those four `sendTurn` implementations at every sync. **2026-08-08: the allowlist is now backed by upstream's own tests.** `ClaudeAdapter.test.ts`, `CursorAdapter.test.ts` and `OpenCodeAdapter.test.ts` each carry an upstream-authored `"steers a running turn instead of opening a new one on mid-turn sendTurn"` case, and all four adapter test files are byte-identical to upstream here — so three of the four allowlisted drivers would break upstream's suite, not just the fork's, if steering regressed. `grok` remains asserted by reading alone. **Re-checked 2026-08-17: all four still steer, and Codex still does not.** Upstream touched every adapter this range, but the two commits that touched a `sendTurn` (`e9e46972f`, `afca73d36`) changed session-scoped permissions and notification-consumer lifetime, not turn routing. The three upstream steering tests are still present and green, and `GrokAdapter.ts` still reuses `ctx.activeTurnId` when `promptsInFlight > 0`. `grep -rn 'turn/steer' apps/server` is still empty, so the deliberate Codex exclusion still holds. | -| `apps/mobile/src/features/threads/composerSendLabel.ts` (`resolveComposerSendLabel`) | `ThreadComposer.tsx` — the inline `sendLabel` ternary the fork lifted out | **The predicted bite landed at the 2026-08-14 sync — and was caught.** Upstream #6543 removed `activeThreadBusy` from the ternary (steer-by-default); because the prop itself was deleted, the conflict surfaced rather than merging silently, and the helper was re-mirrored to the new two-condition expression (`connectionState !== "connected" \|\| queueCount > 0`), its `activeThreadBusy` parameter and test case removed. Earlier history: upstream #5625 rewrote the file end to end and left the conditions untouched. The hazard shape stands — a future edit to the conditions alone (no prop change) still merges cleanly into a call site that ignores it. Diff `git show <upstream-commit> -- ThreadComposer.tsx` for `sendLabel` at every sync **Re-checked 2026-08-17: the mirror still matches.** Upstream's only commit to mobile `ThreadComposer.tsx` this range was `d23b181da` (built-in themes), which changes colours and the appearance provider and does not touch the send-label ternary or reintroduce `activeThreadBusy`. The fork's two-condition expression still follows upstream's. | +| Fork mirror | Mirrors upstream | Risk if upstream changes | +| ----------------------------------------------------------------------------------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| `apps/server/src/coil/autoResume/guards.ts` (`hasOpenBlockingRequest`) | `decider.ts` (private, unexported) | Could miss a new blocking-request activity kind and auto-resume into a prompt. **Re-checked 2026-08-08: byte-identical.** Upstream's two commits to `decider.ts` this range were both thread pinning (#5312, #5581) and added no activity kind; the four `requested`/`resolved` kinds and the stale-failure escape hatch match exactly. **2026-09-02: `threadIsGone` in this file no longer reads `settledOverride`.** #8600 gave the server a one-minute sweep that dispatches `thread.auto-settle` through `thread.settle`'s own decider case, emitting an identical `thread.settled` with no provenance — so settledness stopped meaning "the user is done here" and was cancelling week-long arms on day three. Watch `ThreadSettlementReactor.ts` / `ThreadSettlementPolicy.ts`: if upstream ever adds a provenance marker, the "user settled it" cancel can come back. | +| `apps/server/src/coil/http/auth.ts` (`authenticateWithScope`) | `http.ts` (`authenticateRawRouteWithScope`, private, unexported) | Every fork raw route could authenticate more weakly than the routes beside it. **Re-checked 2026-09-02 (Phase 0): still accurate.** Previously re-checked 2026-08-08. Upstream's only commit to `apps/server/src/http.ts` this range was the Effect beta.103 upgrade (#5331), which deleted the gzip helpers and left `authenticateRawRouteWithScope` untouched. Compared line by line, the mirror matches upstream's behaviour with three intentional deltas: (1) it is exported, so fork routes import it instead of pasting it; (2) it returns the authenticated session, which upstream's discards, because the Web Push subscribe route records which device registered; (3) its scope parameter is typed `AuthEnvironmentScope` (all 8 literals) where upstream narrows to the two orchestration scopes — inherited from the webPush copy, and both callers pass orchestration scopes, so it is a wider type over identical use. **One mirror as of Phase 0:** it used to be pasted separately into `coil/autoResume/http.ts` and `coil/webPush/http.ts`, and the two had drifted (only auto-resume forwarded `dpopFailureReason`); both now call this module. | +| `apps/server/src/orchestration/Layers/CrashRecoveryReconciler.ts` (`getSnapshot()`) | `ProjectionSnapshotQuery.getSnapshot()` vs. the lighter `getCommandReadModel()` | **Live risk, found at the 2026-08-02 sync, re-confirmed unchanged 2026-08-08 — still not fixed.** Upstream moved its own orchestration-snapshot route off `getSnapshot()` onto `getCommandReadModel()` in this range, commenting that hydrating every message and activity payload "has OOM-killed servers". The fork's boot reconciler still calls `getSnapshot()`, and it runs on the startup path **before commands are accepted**, so an OOM there is a hard boot failure rather than one slow request. Both return `OrchestrationReadModel` and the reconciler only reads `thread.session` / `thread.latestTurn` metadata, so the swap looks like a drop-in — but it was deliberately left out of the sync commit and needs its own PR with coverage. | +| `apps/web/src/outbox/**` (thread outbox) | `apps/mobile/src/state/thread-outbox-*.ts` (upstream-authored, still maintained) | The web outbox is a hand port of upstream's mobile one, function for function. Two divergences are deliberate: the web queue drops image attachments (mobile persists them as base64 data URLs, which localStorage cannot hold) and orders on an explicit `sortKey` for user reordering where mobile sorts on `createdAt` alone. A third divergence closed at the 2026-08-14 sync: upstream #6543 made mobile steer active turns by default too (its outbox delivery gate is now connectivity-only — `thread-outbox-model.ts` dropped the `!threadBusy` term), so both platforms steer. Upstream reworking its mobile outbox produces no conflict here. | +| `apps/web/src/coil/composerAnchor.ts` (`COMPOSER_OVERLAY_SELECTOR`, `chat-composer-horizontal-inset`) | `apps/web/src/components/ChatView.tsx` — the `[data-chat-composer-overlay="true"]` element it measures for `composerOverlayHeight`, and the `.chat-composer-horizontal-inset` class in `index.css` that the composer wrapper uses | **Read-only presentation dependencies, not code edits.** The auto-resume capsule is anchored bottom-right, immediately above the docked composer and flush with its right edge. The overlay mounts as a sibling of `<ChatView>` in the route file and cannot receive the composer's geometry as a prop without widening that seam, so it measures the data attribute with a `ResizeObserver` instead. Horizontal alignment additionally reuses the composer's own inset class, because that inset is `0.75rem` at base, `1.25rem` from `40rem` up, and carries `env(safe-area-inset-right)` — any hard-coded value overhangs on wide viewports. If the attribute disappears the capsule falls back to a fixed 76px offset; if the class is renamed the capsule's right edge drifts from the composer's. Both degrade visually, neither breaks. Re-check at each sync. **2026-08-10 (#67): the measurement now converts coordinate spaces rather than assuming they match.** The composer's `offsetParent` is the chat column; the capsule's is `SidebarInset`. Those boxes coincide only while no panel is open, so the original "measure against the composer's parent, apply against ours" shortcut stranded the capsule by 352px horizontally when the inline right panel opened and by the drawer's full height when the terminal drawer opened. `autoResumeAnchor.ts` now derives `bottom`/`left`/`width` from the composer's rect expressed in the capsule's own offsetParent coordinates, and all three boxes are observed. **This adds no new upstream dependency** — same selector, same class — but it does mean the capsule now depends on the chat column remaining the composer's nearest positioned ancestor. **2026-09-02: the hook moved out of `AutoResumeOverlay.tsx` into `coil/composerAnchor.ts`**, because the loop console is a second overlay with the same placement problem and a second copy of this measurement would be exactly the parallel path this document exists to catch. Both overlays now read one set of numbers. They keep separate positioned wrappers on purpose: the maths resolves against `anchorElement.offsetParent`, so a shared positioned box between them and `SidebarInset` would silently change the third rect. Still one upstream dependency, now with one reader. | +| `apps/web/src/outbox/composerSteering.logic.ts` (steer allowlist) | Each adapter's mid-turn `sendTurn` behaviour (ClaudeAdapter.ts:3729, CursorAdapter.ts:916, GrokAdapter.ts:921, OpenCodeAdapter.ts:1417) | **No capability flag exists** — `ProviderAdapterCapabilities` has no `supportsSteering`, so which drivers fold a mid-turn send into the running turn is asserted by a hand-maintained allowlist. If upstream changes an adapter to open a new turn instead, nothing fails here; the fork would keep sending mid-turn into a provider that no longer steers, and a refusal is invisible because `sendTurn` is forked in `ProviderCommandReactor`. Re-check those four `sendTurn` implementations at every sync. **2026-08-08: the allowlist is now backed by upstream's own tests.** `ClaudeAdapter.test.ts`, `CursorAdapter.test.ts` and `OpenCodeAdapter.test.ts` each carry an upstream-authored `"steers a running turn instead of opening a new one on mid-turn sendTurn"` case, and all four adapter test files are byte-identical to upstream here — so three of the four allowlisted drivers would break upstream's suite, not just the fork's, if steering regressed. `grok` remains asserted by reading alone. **Re-checked 2026-08-17: all four still steer, and Codex still does not.** Upstream touched every adapter this range, but the two commits that touched a `sendTurn` (`e9e46972f`, `afca73d36`) changed session-scoped permissions and notification-consumer lifetime, not turn routing. The three upstream steering tests are still present and green, and `GrokAdapter.ts` still reuses `ctx.activeTurnId` when `promptsInFlight > 0`. `grep -rn 'turn/steer' apps/server` is still empty, so the deliberate Codex exclusion still holds. | +| `apps/mobile/src/features/threads/composerSendLabel.ts` (`resolveComposerSendLabel`) | `ThreadComposer.tsx` — the inline `sendLabel` ternary the fork lifted out | **The predicted bite landed at the 2026-08-14 sync — and was caught.** Upstream #6543 removed `activeThreadBusy` from the ternary (steer-by-default); because the prop itself was deleted, the conflict surfaced rather than merging silently, and the helper was re-mirrored to the new two-condition expression (`connectionState !== "connected" \|\| queueCount > 0`), its `activeThreadBusy` parameter and test case removed. Earlier history: upstream #5625 rewrote the file end to end and left the conditions untouched. The hazard shape stands — a future edit to the conditions alone (no prop change) still merges cleanly into a call site that ignores it. Diff `git show <upstream-commit> -- ThreadComposer.tsx` for `sendLabel` at every sync **Re-checked 2026-08-17: the mirror still matches.** Upstream's only commit to mobile `ThreadComposer.tsx` this range was `d23b181da` (built-in themes), which changes colours and the appearance provider and does not touch the send-label ternary or reintroduce `activeThreadBusy`. The fork's two-condition expression still follows upstream's. | ### Parallel paths (fork controls that must honour upstream's guards) diff --git a/docs/coil/loop/DESIGN.md b/docs/coil/loop/DESIGN.md index 69484e1a8ca4..93ddc6d5295f 100644 --- a/docs/coil/loop/DESIGN.md +++ b/docs/coil/loop/DESIGN.md @@ -1,20 +1,34 @@ # Loop Watch — design (radroid/t3code#38) -> **Superseded by [`docs/coil/loops-v2/`](../loops-v2/PLAN.md) (2026-08-17).** The re-checks the -> second paragraph below asks for are covered in -> [`loops-v2/UPSTREAM-DELTA.md`](../loops-v2/UPSTREAM-DELTA.md): #5219 as `ThreadBackgroundLiveness` -> in its §2, and #3638 confirmed still not on `upstream/main` in its §1. Three of this design's -> premises did not survive them. +> **Superseded by [`docs/coil/loops-v2/`](../loops-v2/PLAN.md) (2026-08-17).** > > **Status: designed, not built.** No `apps/server/src/coil/loop/` exists — every path this > document writes in the present tense is a proposal. It is archived here because the research -> underneath it is reusable, not because the feature shipped. +> underneath it is reusable, not because the feature shipped. The reactor detail is the part that +> survived best and is what `loops-v2/BACKEND.md` builds on. > -> Two things have moved since 2026-08-02 and must be re-checked before anyone builds from it. -> Upstream #5219 supersedes the subagent-tracking gap that motivated the issue, and upstream -> PR #3638 ships `schedule_task` / `delegate_task` MCP tools. The codebase line numbers cited -> throughout were measured against the 2026-08-02 merge-base and have since drifted through -> several upstream syncs — re-verify each one rather than trusting it. +> **Corrected 2026-09-02 (issue #125 §A6).** An earlier version of this banner named upstream +> **#5219** and PR **#3638** as the premises that "did not survive". That is backwards: both +> re-checks came back **confirming** — #5219 landed and is exactly the `ThreadBackgroundLiveness` +> the successor design composes with +> ([`UPSTREAM-DELTA.md`](../loops-v2/UPSTREAM-DELTA.md) §2), and #3638 was re-confirmed as **still +> not on `upstream/main`** (§1), which is why a fork-local scheduler is not a parallel path. +> +> The three premises that actually moved are in +> [`loops-v2/FINDINGS.md`](../loops-v2/FINDINGS.md) §A: +> +> - **§A1** — upstream shipped **thread pinning** two days after this design froze, so the sidebar +> affordance this document argues at length against building already exists. +> - **§A2** — `Sidebar.tsx` / `Sidebar.logic.ts` are **not** seam rows, which cuts both ways: a +> pinned thread is free, a bespoke row type is the most expensive kind of row this fork can take. +> - **§A3** — the settings mount this design targets (`BetaSettingsPanel.tsx`) **no longer exists**, +> so its "+2 lines, risk 6" costing is void. +> +> A fourth has moved since: upstream **inverted pin-vs-settle** (`f70eeeeb0`), so a pin no longer +> outranks settlement — see `UPSTREAM-DELTA.md` §9.1. +> +> The codebase line numbers cited throughout were measured against the 2026-08-02 merge-base and +> have since drifted through many upstream syncs — re-verify each one rather than trusting it. _Chosen 2026-08-02 from a 4-design / 12-judgement panel. Full alternatives in [OPTIONS.md](OPTIONS.md); codebase evidence in [RESEARCH.md](RESEARCH.md)._ @@ -26,7 +40,7 @@ _Chosen 2026-08-02 from a 4-design / 12-judgement panel. Full alternatives in [O | --------------------------- | ---- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | | quiescence (Deadman) | 7.67 | WINNER. The only design with zero fatal flaws across all three lenses. Its trigger — `now - projection_threads.updated_at` — is the one signal every one of the twelve judgements independently named as the best idea worth stealing. I re-verified it: ProjectionPipeline.ts:794-808 groups `thread.activity-appended` with `thread.message-sent` and rewrites the row with `updatedAt: event.occurredAt`, and | | budget (RUNWAY) | 6.33 | Headline mechanism is dead; its product instinct survives. I confirmed the SRE's kill shot: `stopSessionInternal` calls `completeTurn(context, "interrupted", "Session stopped.")` at ClaudeAdapter.ts:3057, emitting a real-turnId `turn.completed`, so arming on any non-synthetic turn.completed restarts exactly the work the user just hit Stop on. I also confirmed the second hole at ClaudeAdapter.ts:19 | -| observability (Night Watch) | 6.33 | Right diagnosis, two mechanisms that do not exist. I read the user's own template at /Users/rajdholakia/.claude/skills/auto-loop-bootstrap/assets/templates/.loop/state.json: it is `{stage, iter, pr_mode, pr_size_policy, base_branch, backlog_source}` — there is no `status` key and no `remaining`/`backlog` array, and the companion CLAUDE.md:50 says outright `The loop NEVER halts on a semantic event` | +| observability (Night Watch) | 6.33 | Right diagnosis, two mechanisms that do not exist. I read the user's own template at `~/.claude/skills/auto-loop-bootstrap/assets/templates/.loop/state.json`: it is `{stage, iter, pr_mode, pr_size_policy, base_branch, backlog_source}` — there is no `status` key and no `remaining`/`backlog` array, and the companion CLAUDE.md:50 says outright `The loop NEVER halts on a semantic event` | | contract (BATON) | 6 | Loses on three independently-verified defects, all in the load-bearing mechanism. (1) The contract is read from the wrong directory on most threads: `resolveThreadWorkspaceCwd` (checkpointing/Utils.ts:21-27) returns `worktreePath` FIRST and only falls back to the project root, so on any worktree-backed thread the agent writes into the worktree while the supervisor stats the project root — three fa | # Loop Watch — `apps/server/src/coil/loop/` diff --git a/docs/coil/loops-v2/BACKEND.md b/docs/coil/loops-v2/BACKEND.md index 2c2f098beb01..e42f10ba01b2 100644 --- a/docs/coil/loops-v2/BACKEND.md +++ b/docs/coil/loops-v2/BACKEND.md @@ -12,15 +12,19 @@ scheduler rather than replacing it**. Fork-owned, provider-agnostic. Line references were measured against merge-base `196c8ea0d` (2026-08-14 sync), not the 2026-08-02 base the archived design used; UPSTREAM-DELTA §5 lists the ones that have since moved. Churn and -file-size figures are re-measured against the current merge-base **`cebac353d`**. - -> **Corrected 2026-08-19 by review.** An earlier revision named `a4cc1367b` as the current -> merge-base and said the 2026-08-18 daily sync force-rewrote the fork onto the _same_ base. The -> base **moved**: two upstream commits (`3723722f7`, `cebac353d`) sit under the replayed fork stack, -> and `cebac353d` is an upstream ancestor, so `cebac353d` is the merge-base. The distinction is not -> cosmetic — the package's headline ledger figure is only true against `cebac353d`; measured against -> `a4cc1367b` two of upstream's own edits get counted as the fork's. `docs/coil/SEAMS.md` still -> carries the old base; that defect predates this package and is a follow-up, not part of it. +file-size figures are re-measured against the current merge-base **`941acb4f9`** (the 2026-09-02 +sync), which is also the base `docs/coil/SEAMS.md` now carries — **53 upstream-owned files, ++2609 / −981**. + +> **Re-baselined 2026-09-02 (issue #125 §D).** Two earlier bases are named in the history of this +> document and both are now superseded. The 2026-08-18 sync moved the base from `a4cc1367b` to +> `cebac353d`; the 2026-09-02 sync (182 upstream commits, issue #128) moved it again to +> `941acb4f9`. **Every churn and risk figure below is re-measured against `941acb4f9`**, and several +> moved a lot — `ClaudeAdapter.ts` 16 → **23**, `settingsSearch.ts` 14 → **24**, +> `_chat.$environmentId.$threadId.tsx` 4 → **2**. Churn is measured over the 60 days _before_ the +> merge-base, so a busier upstream window moves every number without any fork change; compare risk +> within this table, not against a previous revision. The SEAMS.md header no longer lags — that +> follow-up landed as `93c1e67b8`. **Marker discipline.** `[A]` is an assumption this design has not verified. `[V]` is verified by command against this tree. **`[V - external]`** is verified by command too, but against a shipped @@ -50,8 +54,9 @@ fires is the measure of a correct implementation, not a sign it is doing nothing Total new upstream surface for phases 1–4: **3 new seam rows, ~+16 lines** — the `hooks` spread into `ClaudeAdapter`'s existing `queryOptions` object (+1), `settingsSearch.ts` (**~+13**) and `SettingsSidebarNav.tsx` (+2) — plus one existing seam row rewritten in place at delta zero, and -~6 lines in an existing fork-owned file, which is not upstream surface at all. The budget is -PLAN §6; §11 below prices each row. +~6 lines in an existing fork-owned file, which is not upstream surface at all. **Phase 5 adds one +more row, measured 2026-09-02 and no longer an estimate: `McpHttpServer.ts` `+2/−1`, churn 2, +risk 6** (§11). The budget is PLAN §6; §11 below prices each row. Everything else is new files upstream has never seen. --- @@ -205,23 +210,48 @@ decision table testable without a server, a clock, or a provider. Every case in ## 3. Data model -One file, `coil-loop.json`, in `ServerConfig.stateDir`. +One file, `coil-loop.json`, in `ServerConfig.stateDir`. Its top level is +`{ version: 1, global: LoopGlobalSettings, threads: Record<threadId, LoopRecord> }` — the `global` +key is the master toggle's home, added by review (issue #125 §B7) because the toggle was the +headline kill switch with no data model, no route and no test. ```ts +LoopGlobalSettings = { + enabled: boolean // default FALSE. The master toggle (guard 2). + maxArmedThreads: number // default 3, enforced in the route AND re-checked per tick + defaultMaxCheckIns: number // default 6, <= 20 + defaultRunMs: number // default 8h — seeds the arm form, never a fallback deadline + defaultIdleMs: number // default 15 min + defaultBusyIdleMs: number // default 45 min +} + LoopRecord = { armed: boolean // default FALSE. Nothing is supervised implicitly. armedAtMs: number goal: string | null // what the user said they wanted, for the console header - // budget — mandatory, no unlimited mode - maxCheckIns: number // <= 20, enforced in the route with a 400 + // budget — mandatory, no unlimited mode, no null state + maxCheckIns: number // 1..20, enforced in the route with a 400 checkInsUsed: number - deadlineAtMs: number | null + deadlineAtMs: number // MANDATORY at arm time. Not nullable. See the note below. // thresholds, per-thread overridable idleMs: number // default 15 min busyIdleMs: number // default 45 min + // what the agent scheduled for itself, from the Stop / SubagentStop hook (§4) + crons: { + recordedAtMs: number + entries: Array<{ + id: string + schedule: string // a 5-field cron expression, NOT a timestamp + recurring: boolean + prompt: string // truncated to 1000 chars by the binary — a label, not the prompt + nextFireAtMs: number | null // computed fork-side; null when the expression did not parse + }> + } | null // null = never observed; { entries: [] } = observed and empty + degraded: null | "gate_off" | "wake_lost" + // liveness bookkeeping lastCheckIn: { firedAtMs: number; createdAtIso: string } | null checkIns: Array<{ // the iteration ledger, bounded by maxCheckIns (so <= 20) @@ -234,6 +264,9 @@ LoopRecord = { strikes: number rateLimitedUntilMs: number + // pin bookkeeping — see §7's "Arming pins, and what that costs" + pinnedByLoop: boolean // true only when the arm route created the pin + // terminal state — sticky, only a human re-arm clears it stopped: null | { reason: "done" | "spent" | "stalled" | "handed-back" @@ -245,6 +278,22 @@ LoopRecord = { } ``` +**`deadlineAtMs` is mandatory and is not nullable — decided by review (issue #125 §A1).** Three +files disagreed about this: an earlier revision of §4 flagged "reject arming without a deadline" as +an unresolved change to this contract, PLAN stated the deadline bound with no caveat, and guard 10b +carried a `deadlineAtMs == null` branch that meant _no deference at all_ — so a deadline-less loop +would fire on top of a healthy self-pacing thread, the exact case §0 declares impossible. The +resolution is the one that makes the wrong behaviour unrepresentable: **a null deadline is not a +state.** The route returns `400 deadline_required` (never a clamp — D9), the field is `number`, and +guard 10b's null branch is gone. + +Its **decoding default is `0`**, deliberately, and that is the only reason a record can ever be seen +without a real deadline: a file written by a build that predates the field, or one hand-edited to +drop it, decodes to epoch — which is always `<= now`, so guard 4b stops the loop as `spent` on the +first evaluation. Fail-closed. `maxCheckIns` takes the same treatment for the same reason (default +`0` ⇒ immediately spent). A decoding default that meant "unbounded" would turn a corrupted write +into an unbounded overnight spend, which is the single worst outcome this feature can produce. + The `checkIns` array is what makes the console's iteration ledger reconstructable at all: without `firedAtMs` and the activity cursor recorded _at nudge time_, the history cannot be rebuilt later. Ledger rows therefore render **derived facts** — turns, activities, files moved between two cursors @@ -255,7 +304,9 @@ absent here rather than stubbed. **Every field is `Schema.withDecodingDefaultKey`.** This is not style. A missing _required_ key fails the whole-file decode, and the boot path turns a decode failure into `EMPTY_STATE` — which would silently disarm every loop on the machine. `autoResume/state.ts` carries this warning in a -comment and it is the highest-severity footgun in the module. +comment and it is the highest-severity footgun in the module. **Choose each default so the +fail-closed reading is the one you get**: `armed: false`, `deadlineAtMs: 0`, `maxCheckIns: 0`, +`crons: null`, `pinnedByLoop: false`, `global.enabled: false`. Blockers live in a sibling map keyed by thread: @@ -286,13 +337,16 @@ threshold = busyTurn ? config.busyIdleMs : config.idleMs selfPacedWakeMs = record.crons.nextFireAtMs // from the Stop hook, persisted graceMs = wakeGrace(record.crons) // derived per entry, see "the grace" below deferrable = selfPacedWakeMs != null - && record.deadlineAtMs != null // no deadline, no cap to defer to && selfPacedWakeMs <= record.deadlineAtMs // the loop's own clock is the cap fire when idleMs >= threshold && !(deferrable && now < selfPacedWakeMs + graceMs) ``` +`deadlineAtMs` is a `number`, never null (§3), so there is no third branch here and no +"deadline-less" case to reason about. The grace boundary is **inclusive**: at exactly +`selfPacedWakeMs + graceMs` the wake counts as lost and T3 fires. + `busyTurn` is `shell.session?.status ∈ {running, starting} || shell.latestTurn?.state === "running"`. **`shell.backgroundLiveness` is deliberately not in `busyTurn`**, even though it is the exact, @@ -329,17 +383,33 @@ stands supervision down for up to 24 hours, and a one-shot pinned to a future da indefinitely, all while the run is nominally armed. The rule is now: **T3 defers only while the recorded next fire is at or before the loop's wall-clock deadline.** Past the deadline the run is over on T3's clock either way, so the deadline is the natural cap and no new config knob is needed; -beyond it T3 paces on its own clock and the ordinary staleness trigger applies. Two consequences -worth arguing with: whether the deadline is the right cap or an explicit `maxDeferMs` would be -honest — that belongs beside PLAN's Q1 — and that `deadlineAtMs` is nullable today (§3), so a loop -armed with no deadline has no cap to defer to. The proposal is to **reject arming without a -deadline** in the route rather than to defer forever; that is a change to §3's contract and is -flagged, not assumed. +beyond it T3 paces on its own clock and the ordinary staleness trigger applies. + +**Both loose ends are now closed (issue #125 §A1).** The deadline is the cap; no `maxDeferMs` knob +is added, because a second knob would have to be explained in terms of the first and every value +other than "the deadline" describes a run that is nominally armed but knowingly unsupervised. +And `deadlineAtMs` is **not nullable** (§3) — the route rejects an arm without one with +`400 deadline_required`, so "a loop with no cap to defer to" is not a reachable state rather than a +branch to handle. Deference is therefore exactly one sentence: _T3 stands down while a recorded +wake's `nextFireAtMs` is at or before the deadline and is not yet overdue by its grace. Past the +deadline there is nothing left to defer to._ `record.crons` is written by the `Stop` / `SubagentStop` hook callbacks (see §2's `crons.ts`): read `input.session_crons`, compute `nextFireAtMs` fork-side from `schedule` (one-shot = single fire time encoded in the fields; server-local tz) `[A — the parse is ours]`, persist per thread. +**The parse is fork-owned, and it must not bring a dependency.** `git grep -i cron -- '*package.json'` +and a `cron` search over `pnpm-lock.yaml` both return **zero** — there is no cron parser anywhere in +this repo `[V]`. Nor should one be added: a general parser handles a grammar the producer never +emits. The producer's grammar is documented and narrow — `CronCreateInput.cron` is +_"Standard 5-field cron expression in local time: `M H DoM Mon DoW`"_, with `*/5` steps and `1-5` +ranges in its own examples, **no seconds field and no `@daily`-style macros** `[V - external]`. So +`crons.ts` ships a ~120-line parser for exactly that grammar: five space-separated fields, each +`*`, `N`, `A-B`, `A-B/S`, `*/S` or a comma-list of those, evaluated in the **server's local +timezone** (the tool says "local time", and the binary and the server run in the same process +group). Anything it cannot parse yields `nextFireAtMs: null` for that entry, which means **no +deference from that entry** — an unparseable schedule must never stand supervision down. + **The delivery is no longer an assumption.** This was the package's single largest `[A]` — a design whose strongest trigger rode on a hook payload nobody had confirmed. Read out of the shipped binary (2.1.236) `[V - external]`: the payload `{ background_tasks, session_crons }` is spread @@ -380,9 +450,27 @@ off, and `wake_lost` when a recorded wake did not land. Both surface on the cons `gate_off` needs a source, because `session_crons` carries no gate status: a **`PostToolUse` hook on `ScheduleWakeup`**, declared in the _same_ fork-built hooks object as the `Stop` callbacks, reading the tool's own response. That is zero extra seam — the object is fork-side, so the upstream spread -does not grow — and like every other callback here it only writes to the fork store `[A — the -response shape is not verified]`. If it turns out not to be readable, the state is dropped rather -than guessed at. +does not grow — and like every other callback here it only writes to the fork store. + +**The plumbing is verified; only the response body is still an assumption.** `HookCallbackMatcher` +carries an optional `matcher` string, which for `PreToolUse` / `PostToolUse` selects by tool name, +and `PostToolUseHookInput` is `BaseHookInput & { hook_event_name: "PostToolUse"; tool_name: string; +tool_input: unknown; tool_response: unknown; tool_use_id: string; duration_ms?: number }` +`[V - external]`. So `{ PostToolUse: [{ matcher: "ScheduleWakeup", hooks: [cb] }] }` reliably +delivers the tool's own response to a fork callback. What is **not** verified is the response body: +issue #42 read `ScheduleWakeupTool.call` out of the binary as `if(!zTH())return OsH("gate_off"),…`, +so the marker string is expected to be `gate_off`, but nothing has observed it on the wire +`[A — the response body]`. Spec it as a **substring probe, not a parse**: JSON-stringify +`tool_response`, and record `degraded: "gate_off"` only when the result contains `gate_off` +case-insensitively. Anything else leaves `degraded` untouched. A probe that finds nothing is the +same as no probe, which is the correct degradation — the state is dropped rather than guessed at. + +**Where it renders.** `gate_off` and `wake_lost` are the two values of `record.degraded` and both +render in the console's loop-state section, in the same slot as the rate-limit hold (§8) and with +the same rule: they must read differently from "stalled". `gate_off` reads +_"Self-pacing unavailable — the agent's scheduler is switched off upstream. T3 is pacing this run."_ +None of the prototypes draw this row; that is a gap in the mocks, declared here rather than +hidden (issue #125 §B10). **Why `updatedAt` and nothing else.** `thread.activity-appended` is grouped with `thread.message-sent` in the projection pipeline and rewrites the row with @@ -512,30 +600,108 @@ own deadline — and because `CronCreate` is unbounded and `recurring` entries l (§1.1), it can keep walking for days. The bound would be advisory exactly where the spend is unattended. +**One semantic for the master toggle, settled by review (issue #125 §A3).** Guards 1 and 2 were +described in three incompatible ways across the package — "no fiber", "stand down at next tick", and +"existing loops stop at their next check-in". Only one of them can be true at a time, and a tick +requires a fiber. The rule is: + +- **The supervisor fiber always runs**, exactly as `autoResume`'s does. One tick loop, forked once at + layer construction. +- **`COIL_LOOP_ENABLED=0` is the only thing that stops a fiber existing.** It is read at layer + construction and never again — an env kill switch for an operator, not a product control. +- **The master toggle is a guard evaluated every tick**, and again immediately pre-dispatch so it is + never one-tick-stale. Toggle off ⇒ nothing fires, every armed loop reports `standing_down` with + reason `disabled`, **nothing is disarmed and nothing is stopped**. Toggle back on and the same + loops resume with their budgets intact. + +That is the semantic a kill switch has to have for a feature that spends money unattended: switching +it off must be reversible without asking the user to re-arm anything, and it must not quietly +manufacture terminal states nobody chose. + **Guards, in order.** Non-consuming skips are marked ○ — they keep budget and surface a reason. -| # | Guard | On fail | -| --- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------ | -| 1 | `config.enabled` (env kill switch, checked at layer construction) | no fiber forks | -| 2 | `store.global.enabled` — re-read **every tick and again pre-dispatch** | ○ the settings toggle is a true kill switch, not one-tick-stale | -| 3 | `record.armed === true` | ○ | -| 4 | shell is `Some`, not archived, is a supported thread | **disarm** | -| 5 | `settledOverride !== "settled"` | ○ | -| 6 | `snoozedUntil == null \|\| <= now` | ○ the only way to honour a snooze | -| 7 | `settledOverride === "active"` ⇒ nudge **then** repair pin | — | -| 8 | `!hasPendingApprovals && !hasPendingUserInput && !hasActionableProposedPlan` | ○ | -| 9 | `autoResumeStore.getThread(threadId).pending === null` | ○ | -| 10 | `now >= record.rateLimitedUntilMs` | ○ | -| 10b | **no deferrable wake still pending** — `nextFireAtMs == null \|\| record.deadlineAtMs == null \|\| nextFireAtMs > record.deadlineAtMs \|\| now >= nextFireAtMs + wakeGrace(entry)` | ○ the deference rule — while a wake is pending _inside the loop's deadline_ the agent is pacing itself, so T3 stands by. A wake past the deadline is not deferred to: `CronCreate` is unbounded (§1.1, §4). Console: _"Self-pacing · next wake 02:35"_ | -| 11 | `now - lastCheckIn.firedAtMs >= config.idleMs` | ○ structural floor: a tight loop stays impossible even if `updatedAt` fails to bump | -| 12 | idle threshold met, on a freshly re-read shell | ○ | -| 13 | budget, deadline, strikes, sentinel | **stop** | -| 14 | armed threads `< maxArmedThreads` — enforced in the route **and** re-checked in the tick | ○ not bypassable by hand-editing the state file | +| # | Guard | On fail | +| --- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| 1 | `config.enabled` (env kill switch, `COIL_LOOP_ENABLED`, checked at layer construction) | no fiber forks — the **only** condition under which no fiber exists | +| 2 | `store.global.enabled` — re-read **every tick and again pre-dispatch** | ○ reason `disabled`; the loop reports `standing_down` and stays armed. Nothing is disarmed, nothing is stopped | +| 3 | `record.armed === true` | ○ | +| 4 | shell is `Some`, not archived, is a supported thread | **disarm** | +| 4b | **stop conditions, swept before any ○ guard**, in this order: done signal (sentinel or `loop_done`), `now >= deadlineAtMs`, `checkInsUsed >= maxCheckIns`, two strikes, takeover | **stop** — see "Why the stop sweep moved" and "Why done outranks the bounds" below | +| 5 | _retired._ `settledOverride !== "settled"` — see "Why guard 5 is gone" below | — | +| 6 | `snoozedUntil == null \|\| <= now` | ○ the only way to honour a snooze. Phase `held`, the same word `status.ts` and the route's `derived` use — a snooze is a bounded hold with an expiry, not a question waiting on an answer | +| 7 | `settledOverride === "active"` ⇒ nudge **then** repair pin | — | +| 8 | `!hasPendingApprovals && !hasPendingUserInput && !hasActionableProposedPlan` | ○ | +| 9 | `autoResumeStore.getThread(threadId).pending === null` | ○ | +| 10 | `now >= record.rateLimitedUntilMs` | ○ | +| 10b | **no deferrable wake still pending** — `nextFireAtMs == null \|\| nextFireAtMs > record.deadlineAtMs \|\| now >= nextFireAtMs + wakeGrace(entry)` | ○ the deference rule — while a wake is pending _inside the loop's deadline_ the agent is pacing itself, so T3 stands by. A wake past the deadline is not deferred to: `CronCreate` is unbounded (§1.1, §4). The `now >=` is **inclusive**. Console: _"Self-pacing · next wake 02:35"_ | +| 11 | `now - lastCheckIn.firedAtMs >= config.idleMs` | ○ structural floor: a tight loop stays impossible even if `updatedAt` fails to bump | +| 12 | idle threshold met, on a freshly re-read shell | ○ | +| 13 | _moved to 4b._ Kept as a number so cited cases keep meaning | — | +| 14 | **other** armed threads `< maxArmedThreads` — enforced in the route **and** re-checked in the tick | ○ not bypassable by hand-editing the state file. The loop under evaluation does not count against the ceiling: counting it made every loop stand down at exactly `maxArmedThreads` armed, i.e. the ceiling refusing the population it was sized for | Guard 8's third clause is the one every design in the original panel missed. `Sidebar.logic.ts` treats plan-ready as **not** pending-input, so a thread parked on an unapproved plan otherwise passes every other blocking guard and gets pushed past the human's yes. +**Why the stop sweep moved from 13 to 4b (added by review, 2026-09-02).** Guard 13 sat _after_ the +idle guards, so a stop condition was only ever evaluated on a tick that had already decided the +thread was idle enough to nudge. Every ○ guard above it is therefore a way to walk past a deadline: +a thread that never goes idle passes 12's check never, so 13 never ran, and a self-paced run +strolled through its own deadline indefinitely — which is precisely the outcome the `stopSession` +paragraph above says must not happen. The same hole swallowed the sentinel: an agent that wrote +`.coil/loop-done` while still working was not recorded as `done` until it happened to go quiet. +Stop conditions are facts about the **run**, not about the thread's current activity, so they are +swept immediately after the shell resolves and before anything can skip. `13` is retained as a +retired number rather than reused, the same discipline TESTS.md uses for its case numbering. + +**Why done outranks the bounds (added by review, 2026-09-02).** Within 4b the done signal is read +_first_. `checkInsUsed` reaches `maxCheckIns` on the final fire, so a run that succeeds on its last +check-in is over on the budget before the tick that would read its `.coil/loop-done` or its +`loop_done` call — and with the bounds swept first, every such run was recorded as `spent`, "it ran +out of rope", for a run that had actually finished. Worse, `stopsSession` fires for `spent` and +deliberately not for `done`, so T3 also ended the session of an agent that had just declared itself +finished and might still have had background work in it. A stale signal is still no signal: freshness +is measured against `armedAtMs`, so the bounds still win when the only done-file is a leftover. + +**Why guard 5 is gone (added by review, 2026-09-02).** `settledOverride === "settled"` was a ○ skip, +read as "the human is done here". It has not meant that since upstream **#8600** moved settlement +server-side: `ThreadSettlementReactor` sweeps every minute and dispatches `thread.auto-settle`, +which shares `thread.settle`'s decider case and emits the same `thread.settled` event **with no +provenance marker**. Auto-settle is on by default, so a settled flag is now overwhelmingly likely to +be a timer rather than a person, and there is no discriminator to recover the difference from. The +fork has already paid for this once: `autoResume/guards.ts` removed settledness from `threadIsGone` +for exactly this reason, and its comment records that an armed week-long resume was reliably +destroyed on day 3 by a timer. Keeping guard 5 would have reproduced the bug in a worse shape — a ○ +skip never stops the loop, so an auto-settled loop would sit armed and silently do nothing until its +deadline, then report `spent`. Settling is also not a one-way door: sending a turn to a settled +thread clears the override on upstream's own path. The user's opt-out is **disarm**, and the console +says so. Guard 7 (repair the `active` pin) is unaffected and stays. + +Snooze is **not** in the same position and guard 6 stands: there is no auto-snooze command — +`InternalOrchestrationCommand` contains `thread.auto-settle` and no snooze counterpart `[V]` — so +`snoozedUntil` still carries a human's intent. + +**Arming pins, and what that costs.** `thread.pin` is not a decoration. Its decider case emits +**companion `thread.unsettled` and `thread.unsnoozed` events**, with a comment saying so: _"Pinning +is a promotion: it clears the parked states rather than silently outranking them"_ `[V]`. Three +consequences the design has to state rather than discover: + +1. **Arming a snoozed thread would silently cancel the snooze.** The route therefore returns + `400 thread_snoozed` on an arm while `snoozedUntil > now`. Unsnooze first; that is a decision the + human makes, not one a supervisor makes for them. +2. **Arming a settled thread unsettles it, and that is correct** — the human asked for the thread to + run, which is the definition of a promotion. +3. **Disarm unpins only what the loop pinned.** `pinnedByLoop` is recorded at arm time (`true` only + when `pinnedAt` was null before the arm), and disarm dispatches `thread.unpin` only when it is + `true`. Without it, disarming a loop on a thread the user had pinned themselves would remove + their pin, and nothing would record that it had ever been theirs. + +There is **no actor check** on any of this: orchestration commands carry no actor field, `dispatch` +takes only an optional descriptive `origin`, and the decider's `thread.pin` case guards on archival +alone `[V]`. A fork reactor may dispatch `thread.pin` exactly as it dispatches `thread.turn.start`. +The corollary is that nothing upstream will stop the fork from doing the wrong thing here, which is +why the three rules above are rules rather than notes. + --- ## 8. Not fighting auto-resume, and surviving limits @@ -600,6 +766,31 @@ Two design consequences: - It is a second, independent argument for `raise_blocker`: a deferred blocker is durable fork-side state and cannot be discarded by a session stop. +### 9.1c And there is now a _second_ blocking dialog, which the loop can trigger itself + +Upstream **#8144** (`c7222ca4d`, 2026-08-25) added `onUserDialog` with +`supportedDialogKinds: ["resume_return"]` to the same `queryOptions` object phase 1 edits `[V]` — +present in the fork's tree today. It routes into the same blocking `Deferred` as `AskUserQuestion`, +and it fires **on session resume**. So the loop's own nudge, landing on a thread whose session was +torn down, can manufacture a pending user-input that guard 8 then treats as a hard skip. + +The failure is not that the loop pushes past a human — guard 8 correctly refuses — it is that the +loop can **cause** the thing that parks it, silently, and then sit at zero spend until its deadline +reports `spent`. Three rules, and no new guard: + +1. **Guard 8 does not change.** A pending input is a pending input whatever produced it; nudging + past one is worse than waiting. +2. **The fork's `user-input.requested` record (§9.1b) carries the dialog kind**, so the console can + say _"waiting on a session-resume confirmation since 01:04"_ rather than showing an unexplained + idle loop. A `resume_return` park that the loop caused must be legible as exactly that. +3. **It resolves through the deadline, and that is acceptable.** A loop parked on a dialog spends + nothing and ends `spent`, distinctly coloured and distinctly worded, with the reason on the + console. That is the correct outcome for "a human is needed and was not there". + +Whether this is common enough to want a `resume_return` auto-answer is a question for the first +dogfooding run, not for the design. It is deliberately **not** answered here: auto-answering a +dialog the human never saw is precisely the class of move that made the console worth building. + ### 9.2 The fix — a second, non-blocking channel A fork-owned MCP tool, `raise_blocker`, that **records and returns immediately**: @@ -656,32 +847,56 @@ GET /api/coil/loop?threadId=… -> record + derived state + blockers + led POST /api/coil/loop -> arm / disarm / edit bounds / re-arm after terminal POST /api/coil/loop/answer -> answer a blocker or a native pending input GET /api/coil/loops -> all loops (the workspace view) +GET /api/coil/loop/settings -> the global block: enabled + defaults + armed count +POST /api/coil/loop/settings -> write the master toggle and the defaults ``` +**The settings pair is new, added by review (issue #125 §B7), and it moves phase.** `store.global.enabled` +was the headline kill switch with no data model, no route and no test — and the sequencing +consequence was worse than the omission: phase 2 shipped "default off behind the master toggle" +while only phase 4 could flip it, so phases 2 and 3 were **unswitchable as ordered**. §3 now carries +`LoopGlobalSettings`, and **the settings routes ship in phase 2 with the reactor**. Phase 4 adds the +_UI_ over a route that already works, which is also how it should have been sequenced anyway: a +control surface built on a live endpoint is testable, one built on a stub is not. + Operate scope, not read — these mutate scheduling. Upstream's private scope-auth path currently exists as **two independent fork mirrors** — `authenticateWithOperateScope` in `autoResume/http.ts` (scope hardcoded) and `authenticateWithScope(scope)` in `webPush/http.ts` (parameterised) — and only one is in the ledger; this feature moves the general form to a shared `coil/http/auth.ts` rather than becoming a third. +**Correction to phase 0's brief: the webPush form is _not_ strictly more general.** It is +parameterised on scope and returns the session, which `autoResume`'s is not — but it calls +`failEnvironmentAuthInvalid` with **one** argument, where `autoResume`'s passes +`EnvironmentAuth.serverAuthDpopFailureReason(error)` as the second. That second argument is the +whole point of `3bdf109e2`, which landed on 2026-09-02 precisely because the stale one-argument call +compiled and drifted silently: a relay client whose DPoP proof failed got a precise reason from +every environment endpoint except this one. Promoting webPush's body verbatim would **re-introduce +the bug it just fixed, in a shared helper, for all three callers**. The promoted +`authenticateWithScope(scope)` must be the union of both: parameterised on scope, returning the +session, **and** passing the DPoP failure reason. + --- ## 11. Seam cost -| File | Delta | Note | -| --------------------------------------------------------- | -------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -| `apps/server/src/coil/index.ts` | +~6 | **fork-owned, churn 0** — store, reactor, routes | -| `apps/web/src/routes/_chat.$environmentId.$threadId.tsx` | **±0** | existing row (+10/−6, churn 4) rewritten in place — **delta 0** (the row already carries risk 64): `<AutoResumeOverlay>` → `<ThreadCoilOverlay>`, which then hosts both. Genuinely delta-zero, and every future per-thread fork surface is free forever. | -| `apps/server/src/server.ts` | **0** | already has its 3-line row | -| `packages/contracts` | **0** | activity `kind` is an open `TrimmedNonEmptyString`, payload is `Schema.Unknown` | -| `Sidebar.tsx` / `Sidebar.logic.ts` | **0** | a loop is a **pinned thread** — `thread.pin` already exists | -| `apps/server/src/provider/Layers/ClaudeAdapter.ts` | **+1** | **new row.** A single spread into the existing `queryOptions` object: `...(loop ? { hooks: loop.claudeHooks(threadId) } : {})`. `options.hooks` is set nowhere in this repo today, so this is the first subscription to a 30-event surface — though the neighbourhood is no longer empty: #4466 sets `settings: { disableAllHooks: true }` on the capability probe, so upstream has begun touching hook-adjacent config. | -| `apps/web/src/components/settings/settingsSearch.ts` | **~+13** | **new row** (phase 4) — the `SettingsPath` union, a `SETTINGS_SECTION_LABELS` entry, and 2 `SETTINGS_SEARCH_ITEMS`. Measured, not estimated: applying that recipe to the real file and diffing gives **+13**, because each search item is a 5–6 line object literal (37 items span 202 lines); upstream's own +26 on this file decomposes the same way, as 1 union + 1 label + 4 items. Three items is ~+19. Append-ordered arrays, additive. | -| `apps/web/src/components/settings/SettingsSidebarNav.tsx` | **+2** | **new row** (phase 4) — the icon import and its `SETTINGS_SECTION_ICONS` entry. | +| File | Delta | Note | +| --------------------------------------------------------- | -------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| `apps/server/src/coil/index.ts` | +~6 | **fork-owned, churn 0** — store, reactor, routes | +| `apps/web/src/routes/_chat.$environmentId.$threadId.tsx` | **±0** | existing row (+10/−6, churn **2**, risk **32**) rewritten in place — **delta 0**: `<AutoResumeOverlay threadRef={threadRef} />` → `<ThreadCoilOverlay threadRef={threadRef} />`, one JSX element swapped for one JSX element inside the fragment the fork already added. The aggregator then hosts both. Genuinely delta-zero, and every future per-thread fork surface is free forever. Re-measured 2026-09-02: churn fell 4 → 2, so the row is **cheaper** than the docs said, not dearer. | +| `apps/server/src/server.ts` | **0** | already has its 3-line row | +| `packages/contracts` | **0** | activity `kind` is an open `TrimmedNonEmptyString`, payload is `Schema.Unknown` | +| `Sidebar.tsx` / `Sidebar.logic.ts` | **0** | a loop is a **pinned thread** — `thread.pin` already exists | +| `apps/server/src/provider/Layers/ClaudeAdapter.ts` | **+1** | **new row.** A single spread into the existing `queryOptions` object: `...(loop ? { hooks: loop.claudeHooks(threadId) } : {})`. `options.hooks` is set nowhere in this repo today, so this is the first subscription to a 30-event surface — though the neighbourhood is no longer empty: #4466 sets `settings: { disableAllHooks: true }` on the capability probe, so upstream has begun touching hook-adjacent config. | +| `apps/web/src/components/settings/settingsSearch.ts` | **~+13** | **new row** (phase 4) — the `SettingsPath` union, a `SETTINGS_SECTION_LABELS` entry, and 2 `SETTINGS_SEARCH_ITEMS`. Measured, not estimated: applying that recipe to the real file and diffing gives **+13**, because each search item is a 5–6 line object literal (37 items span 202 lines); upstream's own +26 on this file decomposes the same way, as 1 union + 1 label + 4 items. Three items is ~+19. Append-ordered arrays, additive. | +| `apps/web/src/components/settings/SettingsSidebarNav.tsx` | **+2** | **new row** (phase 4) — the icon import and its `SETTINGS_SECTION_ICONS` entry. | + +| `apps/server/src/mcp/McpHttpServer.ts` | **+2/−1** | **new row** (phase 5) — one import of the fork's `LoopToolkitRegistrationLive`, and the terminal `export const layer = PreviewToolkitRegistrationLive.pipe(…)` wrapped in a `Layer.mergeAll`. Churn **2**, risk **6**. Measured, not estimated — see "Phase 5, measured" below. | **The `ClaudeAdapter` row is new since the first draft** and it is the cost of the deference rule. -It is worth arguing rather than waving through: the file is churn-16 and ~4.6k lines (4,644 at the -merge-base), which is the most expensive kind of row this fork can take. Three things keep it cheap: +It is worth arguing rather than waving through: the file is churn-**23** and ~4.8k lines (4,820 at +merge-base `941acb4f9`), which is the most expensive kind of row this fork can take. Three things +keep it cheap: 1. **It is one line, and it is additive.** A spread into an object literal alongside the existing `mcpServers` spread. Every line of logic lives fork-side in `coil/loop/crons.ts`. @@ -715,6 +930,63 @@ sleep should be findable by name. `SETTINGS_SEARCH_ITEMS`, so the fork's own rows become searchable the moment the items are added — which is part of what the `settingsSearch.ts` row buys. +**Both settings rows are mandatory, not a style choice** `[V]`. `SETTINGS_SECTION_LABELS` in +`settingsSearch.ts` and `SETTINGS_SECTION_ICONS` in `SettingsSidebarNav.tsx` are both +`Readonly<Record<SettingsPath, …>>`, and `SETTINGS_NAV_ITEMS` is _derived_ by mapping the label +record's keys. So adding `"/settings/loops"` to the `SettingsPath` union without also adding a label +**and** an icon is a type error. There is no cheaper additive shape. + +Their price has risen since the estimate and should be re-read before phase 4 starts: measured at +merge-base `941acb4f9`, `settingsSearch.ts` is churn **24** (was 14) and `SettingsSidebarNav.tsx` +churn **19**. At ~+13 and +2 lines that is risk ~312 and ~38 — an order of magnitude above phase 1's +adapter row, and phase 4's own upstream neighbourhood keeps moving (`a19f01fc1`, 2026-09-02, adds +another `settingsSearch.ts` entry). D10 stands; the number is simply larger than it was. + +**The zero-row fallback, recorded because it now exists in the tree.** `/settings/diagnostics` is a +real, navigable settings route that is **not** a `SettingsPath` member — `SETTINGS_BREADCRUMB_LABELS` +patches its label in by hand `[V]`. So a fork-owned `settings.loops.tsx` reached from the console +rather than from the sidebar costs **zero upstream rows**, at the price of not being findable by +name in the nav or in settings search. That is exactly the trade D10 rejected ("a feature that can +spend money while you sleep should be findable by name") and the decision does not change — but if +the two rows are ever refused at review, this is the fallback, and it is upstream's own pattern +rather than an invention. + +**Phase 5, measured (issue #125 asked; PLAN had it as `0–3 rows [A]`).** Three paths, all measured +against `941acb4f9`: + +| Path | Upstream edits | Rows | Risk | +| ------------------------------------------ | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ----- | ---- | +| **A** — a gated `"loop"` capability | `McpInvocationContext.ts` +1/−1 (churn 1) · `McpSessionRegistry.ts` +1/−1 (churn 1) · **`packages/contracts/src/previewAutomation.ts`** +1/−1 (churn 3) · `McpHttpServer.ts` +2/−1 (churn 2) | **4** | 16 | +| **B** — ungated toolkit **(the decision)** | `McpHttpServer.ts` +2/−1 (churn 2) | **1** | 6 | +| **C** — register from `coil/index.ts` | none | **0** | 0 | + +**Path B is the decision.** Path A is rejected on two counts and neither is the row count: it forces +an edit to `packages/contracts` (`PreviewAutomationUnavailableError`'s `capability` is a runtime +`Schema.Literal("preview")`, so widening the TypeScript union alone does not typecheck), which +breaks **D12**; and it makes a tool named `raise_blocker` fail with an error class named +`PreviewAutomationUnavailableError`. The capability gate buys nothing here in exchange: nothing at +registration or dispatch consults `capabilities` — `requireMcpCapability` is one voluntary line +inside one handler helper, and the credential is **already per-thread and already all-or-nothing** +`[V]`. The loop toolkit's real gate is `store.global.enabled` plus the per-thread armed record, both +of which it must check anyway. + +**Try Path C first, and fall back to B without ceremony.** `McpServer.toolkit(t)` is +`Layer.effectDiscard(registerToolkit(t)).pipe(Layer.provide(McpServer.layer))` and `layerHttp` +provides the _same_ layer value, so Effect's MemoMap should hand both the one `McpServer` instance — +the identical memoisation property `coil/index.ts` already depends on and `autoResume/sharing.test.ts` +already pins. The unresolved blocker is typed, not behavioural: a declared `McpInvocationContext` +dependency leaks into the layer's `R`, which today is discharged only inside the non-exported +`McpTransportLive`. Upstream's own `registerPreviewSnapshot` shows the way out — +`Effect.withFiber` + `Context.getUnsafe(fiber.context, McpInvocationContext)` — which keeps `R` +empty. **This has not been typechecked.** Attempt it, give it one focused session, and take Path B +the moment it fights back; the difference is one row at risk 6. + +**A coupling phase 5 inherits and cannot fix.** The MCP credential is minted only when +`enableAgentBrowserAccess` is true (`ProviderService.prepareMcpSession` revokes it otherwise) `[V]`. +So a user who turns **Settings → Integrations → Agent browser access** off loses `raise_blocker`, +`loop_status` and `loop_done` along with it, with no loop-shaped explanation. The console must +render that as a named degraded state, not as an empty blocker list. + --- ## 12. Restart safety @@ -728,6 +1000,36 @@ This is the property **no event-edge design has**, because both `providerService `ThreadBackgroundLiveness` explicitly does not provide — its own module doc says _"no persistence, no migration. After a server restart the registry is empty."_ +**Upstream has started work in this area and it does not take the premise away — `5b7d72aad` +(#9167, 2026-09-02), arriving on the next sync.** It continues _active threads_ across a server +restart: before a self-update it stamps the running `activeTurnId` into +`provider_session_runtime.runtime_payload_json.continueAfterServerUpdate`, and on the next cold +start it re-establishes the binding and sends a fresh turn (`"Continue where you left off."`, or +promptless where the adapter supports it). Four reasons D1's durability argument survives intact, +each from the commit `[V]`: + +1. **It marks only threads with `session.status === "running" && activeTurnId !== null`.** A thread + waiting on a scheduled wake is neither. **A pending wake is never marked, so it is never + continued** — which is the exact gap this feature exists to cover. +2. **It is opt-in and ships off**: `continueThreadsAfterServerUpdate` decodes to `false`. +3. **It only writes the marker on the intentional self-update path.** A crash, an OOM, a `kill`, a + machine reboot — the cases a supervisor is for — write nothing and behave exactly as before, + settling the thread with _"Provider session did not survive a server restart."_ +4. **It does not continue the interrupted turn.** `activeTurnId` is nulled and a new turn is + started, so a partially-finished turn is lost either way. + +What it does introduce is a **double-fire window**, and it is worth naming precisely because it is +the kind of thing that is invisible until 3am. Reconciliation dispatches `session.status: "starting", +activeTurnId: null` synchronously at startup, but the actual `sendTurn` is parked behind server +activation — so there is a real interval in which a continued thread looks idle, with no error, and +nothing takes a lease. **This design already covers it, by two independent mechanisms**: +`busyTurn` counts `status === "starting"` (§4), so the fuse is `busyIdleMs`; and +`processStartedAtMs` floors the idle clock at process start, so no armed thread can fire for at +least `idleMs` after boot regardless. Neither was added for this, which is the reassuring part. +TESTS case 130b pins it. If a stronger guard is ever wanted, the honest signal is the marker itself — +non-null means upstream has claimed the thread — but reading `provider_session_runtime` from a fork +reactor would be a new coupling to an upstream table, and the two mechanisms above cost nothing. + --- ## 13. What this honestly does not solve diff --git a/docs/coil/loops-v2/FINDINGS.md b/docs/coil/loops-v2/FINDINGS.md index ca75bd6e1183..536454a6d1ca 100644 --- a/docs/coil/loops-v2/FINDINGS.md +++ b/docs/coil/loops-v2/FINDINGS.md @@ -34,13 +34,25 @@ da6e1a967 2026-08-04 feat(sidebar-v2): thread pinning for sidebar v2 (#5312) > A pin overrides the settled/snoozed lifecycle: while `pinnedAt` is set the thread renders > in the pinned block and never classifies into a shelf. -That comment overstates its own rule, and this file repeated the overstatement. What a pin -actually overrides is **auto-settle**. **Snooze still outranks a pin**, which the sidebar's -partition loop says outright (`Sidebar.tsx`, the `supportsSnooze` branch): "Snooze outranks -everything, including a pin: 'hide until Tuesday' temporarily suspends 'keep on top'. The pin -survives underneath". Read every "a pin overrides the lifecycle" below as _overrides -auto-settle_ — a snoozed Loop leaves the pinned block until it wakes, and returns to its exact -slot. +That comment overstated its own rule, and this file repeated the overstatement. It has since been +**rewritten upstream, and in the other direction**. + +> **Corrected 2026-09-02 (issue #125 §D).** `f70eeeeb0` (#7969, 2026-08-23) inverted pin-vs-settle. +> The contract comment now reads _"Settled and snoozed threads remain in their respective shelves +> even when pinned"_, and the sidebar partition is a single `if / else if` chain — **snoozed, then +> settled, then pinned, then active** `[V]` — mirrored verbatim in +> `apps/mobile/src/features/threads/threadListV2.ts`. So the earlier claim here, that a pin +> overrides auto-settle, is **false**: a settled Loop leaves the pinned block, not just a snoozed +> one. What a pin still buys is a pin marker in whichever shelf the thread lands in, plus +> user-arranged order within the pinned block. +> +> **This does not sink D3, and the reason matters.** D3's load-bearing claim was never "pinning +> keeps a Loop visible forever" — it was **"Direction A costs zero rows in `Sidebar.tsx`"**, and +> that is unchanged. What did change is more useful than what was lost: `thread.pin`'s decider case +> emits companion `thread.unsettled` and `thread.unsnoozed` events `[V]` — _"Pinning is a promotion: +> it clears the parked states rather than silently outranking them"_ — so arming a Loop actively +> un-parks it rather than relying on an ordering rule. The cost of that promotion (arming a snoozed +> thread would cancel the snooze; disarming could remove a user's own pin) is priced in BACKEND §7. Also real today: `thread.pinned` / `thread.unpinned` **events** (`:1070-1071`; the commands that produce them are `thread.pin` / `thread.unpin`, `:739,749`), a fractional @@ -271,6 +283,23 @@ console's primary content is a queue of human-answerable items the loop has accu questions, decisions, blockers — each answerable in place, without reading back through the night's output. +> **The plan diverges from the first half of this, deliberately (issue #125 §A2).** PLAN §3 decides +> that **the transcript stays the default view and the console is an overlay on the same route**. +> This paragraph and prototype P7's Shape A both read as though the opposite had been decided, and +> the reversal was never declared — that is the contradiction #125 caught, and this note is the +> declaration. +> +> The reason is #38's, restated: an overlay has _"nothing to toggle back from"_. A second default +> view means a sticky per-thread toggle that has to survive reload, agree across two windows, agree +> on mobile, and be discoverable when it is wrong — and landing on a different view means owning the +> thread route's render decision, which is `ChatView.tsx` territory (an existing seam row at churn 83) rather than the delta-zero overlay row. +> +> **The content half of this requirement is met in full.** Everything below — the three sources, the +> degradation property, answering without steering into a live turn — is unchanged. What is given up +> is one interaction: the console is a click away rather than zero, so a human opening the thread +> still sees the night's tail first. If dogfooding shows that tail is what sends people back to +> typing "are you still working on it?", Shape A is the fallback and P7 prices it. + Design consequences to work through in the prototypes: - **Where do the items come from?** Three candidate sources, in descending order of how diff --git a/docs/coil/loops-v2/PLAN.md b/docs/coil/loops-v2/PLAN.md index eac72216d59c..aa427dde8fb2 100644 --- a/docs/coil/loops-v2/PLAN.md +++ b/docs/coil/loops-v2/PLAN.md @@ -1,22 +1,26 @@ # Loops — implementation plan -**Status:** proposed — nothing built. -**Baseline:** upstream merge-base `cebac353d`, with the fork **zero commits behind upstream at -verification (2026-08-17)**. The merge-base is the anchor rather than a fork `main` SHA, because -every sync rewrites `main`. Verification originally quoted `a4cc1367b`; the 2026-08-18 daily sync -**moved the merge-base** to `cebac353d`, and the seam numbers below are only correct against -`cebac353d` — measured against `a4cc1367b` the same recipe mis-attributes two of upstream's own -edits to the fork. Every claim below was verified against that tree — see -[UPSTREAM-DELTA.md](UPSTREAM-DELTA.md) §7. - -| Companion doc | What it holds | -| -------------------------------------- | --------------------------------------------------- | -| [report.html](report.html) | the design report, 8 clickable prototypes embedded | -| [report.src.html](report.src.html) | the source the report is built from — edit this one | -| [BACKEND.md](BACKEND.md) | full backend design + rejected architectures | -| [TESTS.md](TESTS.md) | 162 test cases | -| [FINDINGS.md](FINDINGS.md) | raw research notes | -| [UPSTREAM-DELTA.md](UPSTREAM-DELTA.md) | the 2026-08-17 re-verification | +**Status:** phases 0–2 are built on `coil/loops-backend` — the durable record, the Claude hooks, +the pure decision table, the sentinel, the HTTP surface and the supervisor. Phases 3–5 (the console, +the settings section, the deferred question channel) remain proposed. +**Reviewed and re-baselined 2026-09-02** against the findings in issue #125; §12 lists what changed. +**Baseline:** upstream merge-base **`941acb4f9`** (the 2026-09-02 sync). The merge-base is the +anchor rather than a fork `main` SHA, because every sync rewrote `main` until the 2026-09-02 switch to merge-based syncs (PR #132); anchoring on the merge-base still holds, because that is what the seam ledger measures against. The package has now moved +base three times — `a4cc1367b` → `cebac353d` → `941acb4f9` — and **every churn and risk figure below +is re-measured against `941acb4f9`**. Several moved materially: `ClaudeAdapter.ts` 16 → **23**, +`settingsSearch.ts` 14 → **24**, `_chat.$environmentId.$threadId.tsx` 4 → **2**. Churn is counted +over the 60 days _before_ the merge-base, so a busier upstream window moves every number without any +fork change. Structural claims were re-verified against that tree — see +[UPSTREAM-DELTA.md](UPSTREAM-DELTA.md) §7 and §9. + +| Companion doc | What it holds | +| -------------------------------------- | ---------------------------------------------------- | +| [report.html](report.html) | the design report, 8 clickable prototypes embedded | +| [report.src.html](report.src.html) | the source the report is built from — edit this one | +| [BACKEND.md](BACKEND.md) | full backend design + rejected architectures | +| [TESTS.md](TESTS.md) | 183 test cases | +| [FINDINGS.md](FINDINGS.md) | raw research notes | +| [UPSTREAM-DELTA.md](UPSTREAM-DELTA.md) | the 2026-08-17 re-verification, plus §9 (2026-09-02) | `report.html` is generated. Edit `report.src.html` (or a prototype under `prototypes/`), then run both steps — `node docs/coil/loops-v2/build-report.mjs` and @@ -72,7 +76,7 @@ Three properties define it: **T3 Coil is a fork of `pingdotgg/t3code`** that rebases onto upstream continuously — 116 upstream commits landed in the three days before this plan was written, and the sync carrying them landed on `main` the same day `[V]`. The fork maintains a **seam ledger** (`docs/coil/SEAMS.md`) listing every -upstream-owned file it edits: currently **53 files, +2590/−1042 lines** `[V]`. Each edited file is a +upstream-owned file it edits: currently **53 files, +2609/−981 lines** `[V]`. Each edited file is a permanent, recurring merge cost. This produces three rules that shape everything below and would look strange otherwise: @@ -98,20 +102,41 @@ has already survived review. Each with the reasoning, so a reviewer can attack the reasoning rather than guess at it. -| # | Decision | Why | Confidence | -| --- | --------------------------------------------------------------------------------------- | ----------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ------------------------------- | -| D1 | **Durable T3-native reactor, backstopping Claude's scheduler** | Claude's scheduler works ≤30min but is in-process and Claude-only `[V]`. Only `ScheduleWakeup` is clamped, to [60, 3600]s; `CronCreate` takes an unbounded cron expression and both write `session_crons` `[V - external]` | High — evidence in BACKEND §1.1 | -| D2 | **Trigger on `updatedAt` staleness + recorded `session_crons`, never `session.status`** | `ProviderSessionReaper` skips any binding whose thread has `session.activeTurnId != null` `[V]`, so a turn whose completion never arrives pins `status = running` with nothing automated to clear it — gating on it deadlocks the exact threads this is for | High | -| D3 | **A loop is a pinned thread** (Direction A) | Upstream shipped pinning 2026-08-04; `pinnedAt` overrides auto-settle — snooze still outranks a pin `[V]`; costs zero sidebar edits `[V]` | High | -| D4 | **The Loops workspace (Direction C) is a later phase (Phase 6), not phase 1** | User agreed. Lives in fork-owned routes, so cost is low, but it is only worth it once several loops exist | High — user-confirmed | -| D5 | **Never Direction B** (a bespoke Loops section in the sidebar) | Would open a row in `Sidebar.tsx` (3911 lines, 7 commits in 3 days `[V]`) and `Sidebar.logic.ts`; also has no mobile equivalent | High | -| D6 | **Two question channels: blocking (native) + deferred (`raise_blocker`)** | `AskUserQuestion` blocks on a `Deferred` `[V]`; a loop that waits loses the night, and one that nudges past a pending decision is worse | High | -| D7 | **Console reads three sources, two needing no model cooperation** | Degradation test: a model that never calls `raise_blocker` must still produce a useful console | High | -| D8 | **Budget is check-ins + wall-clock, not dollars** | The adapter stamps `total_cost_usd` onto `turn.completed.totalCostUsd` and nothing downstream aggregates it; its per-turn vs session-accumulated semantics are unmeasured `[A]`, so summing per turn could inflate quadratically | Medium — revisit if metered | -| D9 | **Mandatory budget, no unlimited option; route returns 400 rather than clamping** | A silent clamp hides a mistake in a feature that spends money unattended | Medium | -| D10 | **Own settings section** (`/settings/loops`) | Now priced at 2 small additive seam rows, with an upstream precedent that landed the same day (2026-08-17) `[V]` | High | -| D11 | **Fork-owned durable JSON, not a DB migration** | The migration registry is upstream-owned; `autoResume` set this precedent and it has held | High | -| D12 | **Zero `packages/contracts` edits** | Activity `kind` is an open string with an `Unknown` payload, so breadcrumbs are free; and upstream has pre-announced this file as its own automations landing zone | High | +| # | Decision | Why | Confidence | +| --- | --------------------------------------------------------------------------------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ------------------------------- | +| D1 | **Durable T3-native reactor, backstopping Claude's scheduler** | Claude's scheduler works ≤30min but is in-process and Claude-only `[V]`. Only `ScheduleWakeup` is clamped, to [60, 3600]s; `CronCreate` takes an unbounded cron expression and both write `session_crons` `[V - external]` | High — evidence in BACKEND §1.1 | +| D2 | **Trigger on `updatedAt` staleness + recorded `session_crons`, never `session.status`** | `ProviderSessionReaper` skips any binding whose thread has `session.activeTurnId != null` `[V]`, so a turn whose completion never arrives pins `status = running` with nothing automated to clear it — gating on it deadlocks the exact threads this is for. **"Never" is scoped, not absolute**: `status ∈ {running, starting}` is read as `busyTurn`, which selects `busyIdleMs` (45 min) over `idleMs` (15 min). It **lengthens the fuse and never vetoes a fire** — that distinction is the long-tool-call case, and a reader who takes "never `session.status`" literally drops it (BACKEND §4) | High | +| D3 | **A loop is a pinned thread** (Direction A) | Upstream shipped pinning 2026-08-04 and it still costs **zero** sidebar edits `[V]`. **What a pin buys has shrunk twice and the decision survives both** — see the note below the table | Medium-High — see the note | +| D4 | **The Loops workspace (Direction C) is a later phase (Phase 6), not phase 1** | User agreed. Lives in fork-owned routes, so cost is low, but it is only worth it once several loops exist | High — user-confirmed | +| D5 | **Never Direction B** (a bespoke Loops section in the sidebar) | Would open a row in `Sidebar.tsx` (3911 lines, 7 commits in 3 days `[V]`) and `Sidebar.logic.ts`; also has no mobile equivalent | High | +| D6 | **Two question channels: blocking (native) + deferred (`raise_blocker`)** | `AskUserQuestion` blocks on a `Deferred` `[V]`; a loop that waits loses the night, and one that nudges past a pending decision is worse | High | +| D7 | **Console reads three sources, two needing no model cooperation** | Degradation test: a model that never calls `raise_blocker` must still produce a useful console | High | +| D8 | **Budget is check-ins + wall-clock, not dollars** | The adapter stamps `total_cost_usd` onto `turn.completed.totalCostUsd` and nothing downstream aggregates it; its per-turn vs session-accumulated semantics are unmeasured `[A]`, so summing per turn could inflate quadratically | Medium — revisit if metered | +| D9 | **Mandatory budget, no unlimited option; route returns 400 rather than clamping** | A silent clamp hides a mistake in a feature that spends money unattended | Medium | +| D10 | **Own settings section** (`/settings/loops`) | Now priced at 2 small additive seam rows, with an upstream precedent that landed the same day (2026-08-17) `[V]` | High | +| D11 | **Fork-owned durable JSON, not a DB migration** | The migration registry is upstream-owned; `autoResume` set this precedent and it has held | High | +| D12 | **Zero `packages/contracts` edits** | Activity `kind` is an open string with an `Unknown` payload, so breadcrumbs are free; and upstream has pre-announced this file as its own automations landing zone | High | + +**Correction on D3, added by the #125 review.** Two facts moved under it and neither kills it. + +_First, upstream inverted pin-vs-settle._ `f70eeeeb0` (#7969, 2026-08-23) rewrote the contract +comment to _"Settled and snoozed threads remain in their respective shelves even when pinned"_, and +the sidebar's partition is now a single `if/else if` chain — snoozed, then settled, then pinned, +then active `[V]`, mirrored verbatim in `apps/mobile/src/features/threads/threadListV2.ts`. So the +old claim that "`pinnedAt` overrides auto-settle" is **false**: a settled loop leaves the pinned +block. What survives is that a pinned thread still shows its pin marker in whichever shelf it lands +in, and that the whole arrangement costs zero fork lines. + +_Second, and more usefully, `thread.pin` turns out to be a promotion rather than a decoration._ Its +decider case emits companion `thread.unsettled` and `thread.unsnoozed` events, with a comment +saying so `[V]`. That mostly makes the inversion moot for a loop — arming pins, and pinning +unsettles — but it also means arming a **snoozed** thread would silently cancel the snooze, and +disarming would remove a pin the user set themselves. BACKEND §7 now carries the three rules that +follow (`400 thread_snoozed` on arm, unsettle-on-arm is correct, `pinnedByLoop` gates the unpin). + +_The decision stands_ because its actual load-bearing claim was never "pinning keeps a loop +visible forever" — it was **"Direction A costs zero rows in `Sidebar.tsx`"**, and that is unchanged. +Confidence drops from High to Medium-High because the visibility a pin buys is now conditional. **Retraction on D2, added by review.** The justification previously read "a background subagent's message auto-opens a synthetic turn that pins `status = running` and nothing closes it", marked @@ -133,12 +158,14 @@ The `updatedAt` half of the trigger is independently verified and unchanged. - A per-thread loop: arm, bound, supervise, stop, re-arm. - Deference to the agent's own scheduler, via recorded `session_crons` — **bounded by the loop's own - wall-clock deadline** (see Q1). + wall-clock deadline**, which is mandatory at arm time (Q1, resolved). - The console: blocking items, deferred blockers, loop state, iteration ledger. - `raise_blocker` / `loop_status` / `loop_done` as a fork MCP toolkit. - Settings: master toggle, defaults, armed roster. - Web + desktop (same app). Mobile read-only surfacing — **deferred until after Phase 3** and priced when it is built; this plan takes no mobile seam row. +- The master toggle's **data model and routes** (`LoopGlobalSettings`, `GET`/`POST +/api/coil/loop/settings`) ship in **Phase 2**, with the reactor. Phase 4 adds the UI over them. ### Out (explicitly, with reasons in report §12) @@ -151,6 +178,36 @@ The `updatedAt` half of the trigger is independently verified and unchanged. - **Loops answering their own low-stakes questions** — destroys the console's completeness, which is the only reason to trust it. - **Maintainer bots (#44)** — the same reactor with a different work source; sequenced after. +- **Push notifications and mobile surfacing.** Designed in the report (§6) and named as a + requirement in prototype P8, but they have **no phase, no seam line, no acceptance criteria and no + tests**, and this revision does not invent them (issue #125 §B8, §B9). Two specific things move + from "designed" to "not built": the three push reasons, and the budget on the mobile thread-list + row. The latter is not free the way P8 implies — chrome on the mobile row is a **mobile seam row**, + in a file the fork does not touch today. Price both when mobile is built. + +### Vocabulary + +Kept here, and **not** in `docs/internals/glossary.md`. That file is upstream-owned and the fork +does not touch it today, so adding four terms would open a **new seam row** for prose — against the +tripwire in `docs/coil/SEAMS.md`, and for a feature whose whole budget is four rows. Move it to +`docs/coil/CONTEXT.md` when that file is created (§12.6); move it upstream only if the feature is +ever upstreamed. + +- **Loop** — a bounded, durable supervision record on one thread: a goal, a check-in budget, a + wall-clock deadline, and the ledger of what happened. Armed by a human, never implicitly. +- **Check-in** — one nudge the loop sends a quiet thread, spending one unit of the budget. Firing + is the exception; standing down is the normal tick. +- **Stand down** — a tick that did nothing and spent nothing, with a reason. Distinct from a stop: + the loop stays armed and keeps its bounds. +- **Blocker** — a question the agent raised through `raise_blocker` _without_ stopping, answered + asynchronously and restated to the agent on the next check-in. Distinct from a native + `AskUserQuestion`, which blocks the turn. +- **Deference** — standing down because the agent has a wake of its own pending inside the run's + deadline. The loop covers that wake only if it never lands. +- **Spent** — a run that ended on its budget or its deadline. **Never rendered as success**: the + agent never signalled done. +- **Iteration ledger** — the per-check-in rows, built from observed facts (cursors, timestamps, + outcomes) and never from a model-authored summary of its own run. ### Divergences from #42 and #38 (deliberate, and open to challenge) @@ -192,7 +249,9 @@ The `updatedAt` half of the trigger is independently verified and unchanged. console covers the parts that need no model cooperation: the per-check-in rollup (the iteration ledger), loop state and bounds, blocking items, deferred blockers. It deliberately does **not** take over the thread route — the transcript stays the default view and the console is an overlay - on the same route, so there is nothing to toggle back from. And it deliberately carries no + on the same route, so there is nothing to toggle back from. **This is a declared divergence from a + requirement the user stated, not an oversight** — see the note closing this section. And it + deliberately carries no model-authored summary of the run: the ledger renders derived facts, never the model's account of its own night, which is the thing that was untrustworthy to begin with. The "right now" block is the one part worth reconsidering, and it depends on the same in-memory roster that guard #15 does. @@ -201,6 +260,29 @@ The `updatedAt` half of the trigger is independently verified and unchanged. the agent inside the loop, and nothing else. Cross-thread reads are a Phase 6 concern (the cross-loop inbox); until then the answer to "how is it going" is the console, opened by a human. +**Declared divergence — the console does not become the default view (issue #125 §A2).** The user's +own words were _"at any time I open the chat there should be a page where I have a questionnaire +ready to be answered"_ (FINDINGS §F1), and prototype P7's Shape A — the recommended shape — draws +exactly that: opening a loop lands on the console, transcript one click away. **This plan decides +the opposite, and the reversal is deliberate.** Three reasons, in order of weight: + +1. **A second route to toggle back from is a one-way door with a hinge on it.** The fork's own rule + is that if you add a way in you add the way out and the way to see it. A sticky per-thread view + toggle is a small piece of state with a large surface: it has to survive reload, agree across + two windows, agree on mobile, and be discoverable when it is wrong. The overlay has none of that + — it is additive chrome over a view that already works everywhere. +2. **The seam is genuinely zero, and Shape A's is not.** The overlay rewrites a row the fork already + owns at delta zero (§6). Landing on a different default view means owning the thread route's + render decision, which is `ChatView.tsx` territory — an existing row at churn 83. +3. **The requirement is about content, not placement.** What was asked for is a standing answer to + "what do you need from me?". An overlay that opens on top of the transcript, on the same route, + with the same content, answers it. Nothing in the ask requires the transcript to go away. + +**What is given up, stated plainly:** the console is one interaction away rather than zero, so a +human who opens the thread still sees the night's tail first. If dogfooding shows that tail is what +sends people back to typing "are you still working on it?", Shape A is the fallback and P7 prices +it. FINDINGS §F1 and P7's Shape A now both carry this note; they previously read as the decision. + --- ## 4. Architecture @@ -247,13 +329,26 @@ one focused agent-assisted session per unit. - Promote the HTTP scope-auth helper into a shared `apps/server/src/coil/http/auth.ts`. Today there are **two independent implementations of the same mirror of upstream's private auth path** `[V]`: - `autoResume/http.ts:45` has `authenticateWithOperateScope` (scope hardcoded) and - `webPush/http.ts:49` has `authenticateWithScope(scope)` (parameterised). The webPush form is - strictly more general — promote it, re-point `autoResume`, and let the loop routes be the third - caller rather than the third paste. + `autoResume/http.ts` has `authenticateWithOperateScope` (scope hardcoded) and `webPush/http.ts` + has `authenticateWithScope(scope)` (parameterised). Let the loop routes be the third caller rather + than the third paste. + **The webPush form is not simply the better one — promote the union of both.** It is + parameterised and returns the session, which `autoResume`'s is not; but it calls + `failEnvironmentAuthInvalid` with one argument where `autoResume`'s passes + `EnvironmentAuth.serverAuthDpopFailureReason(error)` as the second. That second argument is + `3bdf109e2` (2026-09-02), which exists because the stale one-argument call compiled and drifted + silently. Promoting webPush's body verbatim would re-introduce that bug in a shared helper, for + all three callers. The promoted helper is: parameterised on scope, returns the session, **and** + passes the DPoP failure reason. - Widen `isClaudeThread` from `(thread: OrchestrationThread)` to - `Pick<OrchestrationThread, "session">` so an `OrchestrationThreadShell` is assignable `[A — the -archived design verified this; re-check]`. + `Pick<OrchestrationThread, "session">` so an `OrchestrationThreadShell` is assignable `[V]`. The + two structs declare the field identically — `session: Schema.NullOr(OrchestrationSession)` in + both, so both resolve to `OrchestrationSession | null`. Reproduce with + `git grep -n "session: Schema.NullOr(OrchestrationSession)" packages/contracts/src/orchestration.ts` + (two hits, one per struct). `Pick<…, "session">` is the minimal shape; nothing else in the + predicate is read. Note the shells are **not** interchangeable in general — + `OrchestrationThreadShell` has no `deletedAt`, so guard 4 on a shell tests `archivedAt` plus the + shell simply being absent. **Files:** `coil/http/auth.ts` (new), `coil/autoResume/http.ts`, `coil/webPush/http.ts`, `coil/autoResume/guards.ts` — all fork-owned. @@ -277,8 +372,18 @@ reads its state. `prompt` is **truncated to 1000 characters by the binary** `[V - external]`, so nothing here may assume it holds the agent's full prompt), compute `nextFireAtMs` fork-side from `schedule` (one-shot = single fire time encoded in the fields; server-local tz) `[A — the parse is ours]`, - persist per thread. -- `coil/loop/config.ts` — env-overridable defaults. + persist per thread. **The parse brings no dependency**: there is no cron parser anywhere in this + repo (`git grep -i cron -- '*package.json'` and a `cron` search over `pnpm-lock.yaml` are both + empty `[V]`), and none is added. `CronCreateInput.cron` is documented as _"Standard 5-field cron + expression in local time"_ with `*/5` steps and `1-5` ranges — no seconds field, no macros + `[V - external]` — so `crons.ts` parses exactly that grammar in ~120 lines. An entry that does not + parse yields `nextFireAtMs: null`, which means **no deference from that entry**. +- `coil/loop/config.ts` — env-overridable defaults (`COIL_LOOP_*`). +- A `PostToolUse` hook matched on `ScheduleWakeup`, in the **same** fork-built hooks object, to + source the `gate_off` degraded state. Zero extra seam. The plumbing is verified — + `HookCallbackMatcher.matcher` selects by tool name and `PostToolUseHookInput` carries + `tool_name` + `tool_response` `[V - external]` — the response body is not, so it is a substring + probe rather than a parse (BACKEND §4). - **The one upstream edit:** a single spread into `ClaudeAdapter`'s existing `queryOptions` object, beside the `mcpServers` spread: ```ts @@ -293,8 +398,12 @@ reads its state. - Arming is impossible (no arm route yet); nothing dispatches. - On a Claude thread that self-paces, `GET` shows the pending wake with a plausible `nextFireAtMs`. -- On a non-Claude thread, the record exists and `crons` is empty — no errors. -- A hook callback that throws does not break the turn `[A — to be proven by the §4b hook-failure case]`. +- On a non-Claude thread, the record exists and `crons` is `null` — no errors, and no hooks object + is built at all. +- A hook callback that throws does not break the turn `[A — to be proven by the §4b hook-failure +case]`. The callback returns `{ continue: true }` on every path and never rethrows; a `Stop` hook + **can** halt a turn by returning `{ decision: "block" }` or `{ continue: false }` `[V - external]`, + so "observability only" is a property the code has to hold, not one the surface gives for free. - Killing and restarting the server preserves the record. **Why first:** it is the only phase that can be validated purely by observation, and it de-risks the @@ -318,8 +427,17 @@ here and nothing has been wasted.** expression, so an unbounded rule would let a recorded `0 9 * * *` stand T3 down for a day and a one-shot pinned to a future date stand it down indefinitely. Past the deadline there is nothing left to defer _to_, so the deadline is the natural cap and no new knob is needed. See Q1. -- Arm / disarm / re-arm via `POST /api/coil/loop`, with the 400s from D9. -- Arming also dispatches `thread.pin`; disarming unpins `[A — verify pin/unpin from a fork reactor]`. +- Arm / disarm / re-arm via `POST /api/coil/loop`, with the 400s from D9 — including + `400 deadline_required`, because `deadlineAtMs` is mandatory and is not nullable (BACKEND §3). +- **The master toggle's data model and its routes** — `LoopGlobalSettings` in the store, plus + `GET`/`POST /api/coil/loop/settings`. Moved here from phase 4 by the #125 review: shipping + "default off behind the master toggle" in phase 2 while only phase 4 could flip it left phases 2 + and 3 **unswitchable as ordered**. +- Arming also dispatches `thread.pin`; disarming unpins **only when the loop created the pin** + (`pinnedByLoop`). There is no actor check on `thread.pin` — commands carry no actor and the + decider guards on archival alone `[V]` — so the rules are the fork's to keep. `thread.pin` also + emits companion `thread.unsettled` / `thread.unsnoozed` events `[V]`, so the arm route returns + `400 thread_snoozed` rather than silently cancelling a snooze. BACKEND §7. - Rate-limit tap fiber; `rateLimitedUntilMs` persisted. - Terminal states and disarm stop the session when recorded crons are still pending (`providerService.stopSession` — BACKEND §7), so a bound can actually stop a self-paced run. @@ -334,6 +452,10 @@ here and nothing has been wasted.** - A stall: exactly one fire at the threshold, none during background activity. - Budget exhaustion reports `spent`, never `done`. - Human takeover disarms without resetting budget. +- The master toggle is flippable over HTTP in this phase, with no UI — and flipping it off leaves + every loop **armed** and reporting `standing_down`, disarming nothing. +- A deadline that has passed stops the loop **even while the thread is busy** (guard 4b), which is + what stops a self-paced run walking through its own deadline. **Size:** L. This is the bulk of the work. @@ -345,8 +467,14 @@ here and nothing has been wasted.** - `apps/web/src/coil/ThreadCoilOverlay.tsx` (new) — a fork-owned aggregator that mounts `AutoResumeOverlay` and the console; then **rewrite the existing overlay row in place** to mount - it, so the seam delta is zero `[V — the row is +10/−6, churn 4; delta 0 (row already carries risk -64)]`. + it, so the seam delta is zero `[V — re-measured 2026-09-02: the row is +10/−6, churn **2**, risk +**32**; the edit swaps one JSX element for one JSX element inside the fragment the fork already +added, so the delta is unchanged]`. The fork's web client pattern to reuse is + `AutoResumeOverlay.tsx`: `ManagedRuntime.make(primaryEnvironmentHttpLayer)` + + `resolvePrimaryEnvironmentHttpUrl`, 30s poll plus a focus listener, every failure swallowed to + `null` so the overlay disappears rather than degrading chat. Auth is ambient (the environment HTTP + layer attaches the credential); **there is no client-side scope check and none should be added** — + the 403 is the server's to state. - Console UI: blocking / deferred / loop-state sections, iteration ledger, empty state. - `POST /api/coil/loop/answer`, routing native pending-inputs to the existing resolve path and blockers to the fork store. @@ -367,20 +495,37 @@ here and nothing has been wasted.** **Goal:** the on/off switch and the bounds. - `apps/web/src/routes/settings.loops.tsx` (new, fork-owned) + a fork-owned panel component. -- `settingsSearch.ts`: `SettingsPath` union + label + 2 search items — **~+13 lines**, because each - `SETTINGS_SEARCH_ITEMS` entry is a 5–6 line object literal in the file's existing style (37 items - span 202 lines). A third search item takes it to ~+19. For scale, upstream's own `integrations` - addition to this file measured +26. -- `SettingsSidebarNav.tsx`: icon import + record entry — +2. -- **Not** `SettingsPanels.tsx` (churn 36, risk 2088), **not** `contracts/settings.ts` (churn 26, persisted). - -**Seam cost:** **2 new rows**, ~+15 lines total, both additive. +- `settingsSearch.ts`: `SettingsPath` union + `SETTINGS_SECTION_LABELS` entry + 2 + `SETTINGS_SEARCH_ITEMS` — **~+13 lines**, because each search item is a 5–6 line object literal in + the file's existing style. A third search item takes it to ~+19. +- `SettingsSidebarNav.tsx`: icon import + `SETTINGS_SECTION_ICONS` entry — +2. +- **Both rows are mandatory, not stylistic** `[V]`: `SETTINGS_SECTION_LABELS` and + `SETTINGS_SECTION_ICONS` are both `Readonly<Record<SettingsPath, …>>` and `SETTINGS_NAV_ITEMS` is + derived from the label record's keys, so adding a `SettingsPath` member without both entries is a + type error. +- **Not** `SettingsPanels.tsx` (churn **43**, risk ~2900), **not** `contracts/settings.ts` + (churn **38**, persisted). Loop state stays in `coil-loop.json`; the panel reads and writes + `/api/coil/loop/settings`, which phase 2 already shipped. + +**Seam cost:** **2 new rows**, ~+15 lines total, both additive. **Re-measured 2026-09-02 and the +price has risen**: `settingsSearch.ts` is now churn **24** (was 14) and `SettingsSidebarNav.tsx` +churn **19**, so risk is ~312 and ~38. D10 stands, but this is now the most expensive phase in the +plan by risk, and it buys discoverability rather than function. **The zero-row fallback, if the rows +are ever refused:** `/settings/diagnostics` is a real settings route that is **not** a `SettingsPath` +member — its label is patched in by `SETTINGS_BREADCRUMB_LABELS` `[V]` — so a fork route reached +from the console costs zero rows and loses only nav and search discoverability. **Sequencing:** no longer a constraint. This originally had to wait for the sync carrying upstream #7082, or the `settingsSearch.ts` entry would have conflicted with `integrations` on the way in. That sync has landed — `routes/settings.integrations.tsx` is in the tree and `settingsSearch.ts` carries 7 `integrations` references `[V]` — so phase 4 now adds its entry beside a row that is already there, which was the cheap ordering all along. -**Acceptance:** master toggle off ⇒ no fiber, nothing armed, existing loops stand down at next tick. +**Acceptance:** the toggle is a **guard, not a lifecycle** (issue #125 §A3). The supervisor fiber +always runs — one tick loop, like auto-resume — and the toggle is re-read every tick and again +pre-dispatch. Toggle off ⇒ nothing fires, every armed loop reports `standing_down` with reason +`disabled`, **nothing is disarmed and nothing is stopped**; toggle back on and the same loops resume +with budgets intact. The only thing that stops a fiber existing is `COIL_LOOP_ENABLED=0`, read once +at layer construction. An earlier draft of this line said "no fiber" **and** "stand down at next +tick", which cannot both be true — a tick requires a fiber. **Size:** S–M. --- @@ -392,15 +537,26 @@ already there, which was the cheap ordering all along. - `apps/server/src/mcp/toolkits/loop/` — `raise_blocker`, `loop_status`, `loop_done`. - Console renders deferred blockers; answers are delivered on the next check-in prompt. -**Seam cost:** 0–3 rows `[A]` — the toolkit itself is new, but the capability gate may require -edits to `McpInvocationContext.ts` / `McpSessionRegistry.ts` / `McpHttpServer.ts`. **Measure before -committing to this phase** — the archived design estimated three files and that has not been -re-verified. +**Seam cost: 1 new row — `McpHttpServer.ts` `+2/−1`, churn 2, risk 6.** Measured 2026-09-02, no +longer `[A]`. The measurement changed the design: the archived estimate of three files assumed a +gated `"loop"` capability, and that path is now **rejected** — `McpCapability` is backed by a runtime +`Schema.Literal("preview")` in `packages/contracts/src/previewAutomation.ts`, so widening it is a +**contracts edit** and breaks D12, and it would make `raise_blocker` fail with an error class named +`PreviewAutomationUnavailableError`. It buys nothing in exchange: nothing at registration or +dispatch consults `capabilities`, and the MCP credential is already per-thread and already +all-or-nothing `[V]`. The toolkit's real gate is `store.global.enabled` plus the per-thread armed +record. A **zero-row** path exists (register the toolkit from `coil/index.ts`, using upstream's own +`Effect.withFiber` + `Context.getUnsafe` shape to keep the layer's `R` empty); it is worth one +focused attempt and is **not typechecked**, so budget the one row and treat zero as upside. +**Inherited coupling:** the MCP credential is minted only when `enableAgentBrowserAccess` is true +`[V]`, so turning off **Settings → Integrations → Agent browser access** silently removes all three +tools. The console must name that state rather than render an empty blocker list. **Acceptance:** - `raise_blocker` returns immediately (assert elapsed time, not just the value). - Attribution comes from `McpInvocationContext`, never from an argument. - The answer reaches the agent on the next check-in. +- With `enableAgentBrowserAccess` off, the console says so by name. **Size:** M. --- @@ -414,29 +570,40 @@ cross-loop inbox. **Seam cost:** ~1 row wherever the switch mounts. **Size:** M ## 6. Seam budget -The total upstream cost of the whole feature, which is the number that matters for a fork: +The total upstream cost of the whole feature, which is the number that matters for a fork. +All churn and risk re-measured against merge-base `941acb4f9`, 2026-09-02. + +| Phase | File | Delta | Churn | Risk | Kind | +| ----- | ------------------------------------------- | ----- | ----- | ---- | ---------------------------------------- | +| 1 | `provider/Layers/ClaudeAdapter.ts` | +1 | 23 | 23 | **new row** — additive, read-only | +| 3 | `routes/_chat.$environmentId.$threadId.tsx` | ±0 | 2 | 32 | existing row rewritten in place | +| 4 | `settings/settingsSearch.ts` | ~+13 | 24 | ~312 | **new row** — additive | +| 4 | `settings/SettingsSidebarNav.tsx` | +2 | 19 | ~38 | **new row** — additive | +| 5 | `mcp/McpHttpServer.ts` | +2/−1 | 2 | 6 | **new row** — measured, not `[A]` | +| 6 | mode-switch mount | ~+2 | — | — | **new row** — additive, deferred | +| — | `packages/contracts` | 0 | — | — | D12 — and it is why phase 5 takes path B | +| — | `Sidebar.tsx` / `Sidebar.logic.ts` | 0 | — | — | a loop is a pinned thread | +| — | `server.ts` | 0 | — | — | already has its row | + +**Total: 3 new rows for phases 1–4** (~16 lines), **4 for phases 1–5** (~18 lines), all additive, +plus one deferred. For comparison, the ledger carries 53 rows at +2609/−981. -| Phase | File | Delta | Kind | -| ----- | ------------------------------------------- | ----- | --------------------------------- | -| 1 | `provider/Layers/ClaudeAdapter.ts` | +1 | **new row** — additive, read-only | -| 3 | `routes/_chat.$environmentId.$threadId.tsx` | ±0 | existing row rewritten in place | -| 4 | `settings/settingsSearch.ts` | ~+13 | **new row** — additive | -| 4 | `settings/SettingsSidebarNav.tsx` | +2 | **new row** — additive | -| 6 | mode-switch mount | ~+2 | **new row** — additive, deferred | -| — | `packages/contracts` | 0 | — | -| — | `Sidebar.tsx` / `Sidebar.logic.ts` | 0 | a loop is a pinned thread | -| — | `server.ts` | 0 | already has its row | +Two things the re-measurement changed, both worth stating because they invert the intuition the +earlier drafts built: -**Total: 3 new rows for phases 1–4** (~16 lines), all additive, plus one deferred. For comparison, -the ledger currently carries 53 rows. The row count is what recurs at every sync; the line count is -a one-time write, and review corrected it upward from ~7 after measuring `settingsSearch.ts` -against the real file. +- **Phase 5 got much cheaper.** It was `0–3 rows [A]`, priced as the scary one. Measured, it is one + row at risk **6** — the three `mcp/` files are churn 1–2, the quietest neighbourhood in this plan. +- **Phase 4 got much dearer.** `settingsSearch.ts` went churn 14 → 24, so the two settings rows now + carry ~350 of the plan's ~411 total risk. The most expensive thing in this feature is the + navigation entry, not the reactor, not the adapter hook, and not the agent-facing tools. + +The row count is what recurs at every sync; the line count is a one-time write. --- ## 7. Test strategy -162 cases in TESTS.md, plus the three coverage gates in its §11. Structure: +183 cases in TESTS.md, plus the three coverage gates in its §11. Structure: - **Pure and fast** — `decide.ts` and `guards.ts` are pure so the entire decision table tests without a server or clock. Target **100% branch coverage** on both. @@ -452,15 +619,24 @@ The four to write first: 3. The empty console. _(the degradation property)_ 4. `spent` is never reported as `done`. +Two more that the #125 review added to the front of the list, because each pins a hole that was +open in the docs rather than a behaviour that might regress: + +5. A deadline that has passed stops the loop **while the thread is busy** (guard 4b). Without it, + every ○ guard above the old guard 13 was a way to walk past a deadline. +6. An **auto**-settle does not stand the loop down (guard 5 retired). The fork has already lost an + armed auto-resume to this exact server sweep. + --- ## 8. Rollout - **Default off** at every level: env kill switch, master setting, per-thread arm. - **Dogfood** on this repo's own overnight runs before anything else. -- **Kill path:** `COIL_LOOP_ENABLED=0` stops the fiber forking at layer construction. The master - toggle is re-read every tick _and_ immediately pre-dispatch, so it is a true kill switch rather - than one-tick-stale. +- **Kill path:** `COIL_LOOP_ENABLED=0` stops the fiber forking at layer construction — the only + condition under which no fiber exists. The master toggle is a **guard**: re-read every tick _and_ + immediately pre-dispatch, so it is a true kill switch rather than one-tick-stale, and it stands + loops down without disarming or stopping any of them. - **Blast radius if wrong:** bounded by the mandatory budget (≤20 check-ins/loop) and the armed-loop ceiling (default 3). Worst case is `3 × 20 = 60` unwanted turns, and the reserve-before-dispatch discipline means a broken provider burns budget rather than tight-looping. @@ -469,16 +645,17 @@ The four to write first: ## 9. Risk register -| Risk | Likelihood | Impact | Mitigation | -| --------------------------------------------------------------------------------------------------------------------- | ------------------------------- | ---------------------------- | -------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | -| `session_crons` arrives as a cron _expression_, not a fire time, so the `nextFireAtMs` parse is ours and can be wrong | Medium `[A]` | Deference misfires | Narrowed by review: **delivery of the field is settled** `[V - external]`, only the parse is still `[A]`. **Phase 1 is designed to find this out cheaply** — it logs the parse beside the raw `schedule`. Fallback is pure staleness with a longer threshold — worse, but zero upstream cost | -| Upstream ships its own automations feature | Medium | Duplicated work | Zero contracts edits means the fork becomes a _caller_, not a migration. Re-check each sync | -| The `ClaudeAdapter` row conflicts on a sync | Low-Medium | Recurring cost | One additive line beside an existing spread; fails to a type error, not silent drift | -| A loop pushes past a human decision | Low | Trust | Three separate guards (approvals, pending input, plan-ready), each non-consuming | -| Token burn from a runaway loop | Low | Cost | Mandatory budget, deadline, armed ceiling, reserve-before-dispatch, strike detector | -| Model never calls `raise_blocker` | **High** | Console thinner | By design: two of three console sources need no model cooperation. This is the acceptance test | -| `spent` misread as success | Medium | The original problem returns | Distinct colour, distinct word, distinct push copy; asserted in tests | -| `settingsSearch.ts` conflicts | Medium `[V — 3 commits/3 days]` | Small recurring | Append-ordered array, same add/add shape as #29. The #7082 ordering is already satisfied; the residual risk is ordinary and priced | +| Risk | Likelihood | Impact | Mitigation | +| --------------------------------------------------------------------------------------------------------------------- | ------------------------- | ---------------------------- | --------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| `session_crons` arrives as a cron _expression_, not a fire time, so the `nextFireAtMs` parse is ours and can be wrong | Medium `[A]` | Deference misfires | Narrowed by review: **delivery of the field is settled** `[V - external]`, only the parse is still `[A]`. **Phase 1 is designed to find this out cheaply** — it logs the parse beside the raw `schedule`. Fallback is pure staleness with a longer threshold — worse, but zero upstream cost | +| Upstream ships its own automations feature | Medium | Duplicated work | Zero contracts edits means the fork becomes a _caller_, not a migration. Re-check each sync | +| The `ClaudeAdapter` row conflicts on a sync | Low-Medium | Recurring cost | One additive line beside an existing spread; fails to a type error, not silent drift | +| A loop pushes past a human decision | Low | Trust | Three separate guards (approvals, pending input, plan-ready), each non-consuming | +| Token burn from a runaway loop | Low | Cost | Mandatory budget, deadline, armed ceiling, reserve-before-dispatch, strike detector | +| Model never calls `raise_blocker` | **High** | Console thinner | By design: two of three console sources need no model cooperation. This is the acceptance test | +| `spent` misread as success | Medium | The original problem returns | Distinct colour, distinct word; asserted in tests (cases 17, 137). **Push copy is designed in the report and not built** — it has no phase and no tests, so it is listed under §3 Out rather than counted as mitigation (issue #125 §B8) | +| `settingsSearch.ts` conflicts | **High** `[V — churn 24]` | Small recurring | Append-ordered array, same add/add shape as #29. The #7082 ordering is already satisfied. Re-measured 2026-09-02 the churn is 24, not 14, and `a19f01fc1` added another entry the same day — so expect this row to conflict most syncs. §6 records the zero-row fallback if it stops being worth it | +| An **auto**-settle or **auto**-anything reads as human intent | **High** `[V]` | A loop silently retires | Guard 5 retired (BACKEND §7). The general rule the fork keeps re-learning: upstream automates a user-only signal, and every guard reading _intent_ from it inverts. Audit each guard against "could a server timer write this?" before adding it | --- @@ -486,7 +663,17 @@ The four to write first: The four places an outside opinion is most valuable. -**Q1 — Is the deference rule right, or too clever?** +**Q1 — Is the deference rule right, or too clever? — RESOLVED 2026-09-02.** +The rule stands, and both of its loose ends are closed. **The deadline is the cap**; no `maxDeferMs` +knob is added, because a second knob would have to be explained in terms of the first and every +value other than "the deadline" describes a run that is nominally armed but knowingly unsupervised. +And **`deadlineAtMs` is mandatory at arm time and is not nullable** — the route returns +`400 deadline_required` and never clamps (D9). So "a loop with no cap to defer to" is not a +reachable state rather than a branch anyone has to handle, and the whole rule is one sentence: _T3 +stands down while a recorded wake's `nextFireAtMs` is at or before the deadline and is not yet +overdue by its grace._ The grace boundary is **inclusive**. The original text follows, for the +reasoning. + T3 stands down while a recorded wake is still pending — any legal delay, not merely one inside the threshold window — **up to the loop's own wall-clock deadline**, and covers the wake once it is overdue by a grace derived from the recorded entry: `max(90s, min(10% of the period, 15min))` for a @@ -499,8 +686,7 @@ would fire the design's strongest trigger against a perfectly healthy thread. **The deadline cap was added by review**, and it replaces a retracted argument that the exposure was bounded without a second rule because `ScheduleWakeup` clamps a delay to an hour. It is not: `CronCreate` writes the same `session_crons` field with an unbounded cron expression, so a recorded -`0 9 * * *` would stand T3 down for a day. Whether the deadline is the right cap, or whether this -wants an explicit `maxDeferMs`, is open along with the rest of Q1. +`0 9 * * *` would stand T3 down for a day. The alternative to all of it is that T3 always paces on its own clock and simply tolerates occasional double-firing. Deference is more correct and more complex. _Is the complexity worth it, @@ -511,7 +697,11 @@ Prototype P7 shape C makes it a cross-thread "needs you" surface. It is barely m useful with loops switched off, and it might be the more valuable feature. _Is the loop the right container for this at all?_ -**Q3 — Is `raise_blocker` worth the MCP toolkit, or should it be a file?** +**Q3 — Is `raise_blocker` worth the MCP toolkit, or should it be a file? — the price is now +measured.** The toolkit costs **one** seam row at risk 6, not the three the archived design +estimated (§6, BACKEND §11). That does not settle the question — a file is still cheaper and still +worse — but it removes the seam argument from the file's side of it. The original follows. + A blocker could be a line the agent appends to `.coil/blockers.jsonl`, read by the same read-only supervisor that stats the done-file. That is provider-agnostic with **zero** MCP work, at the cost of no structured options and no immediate confirmation to the agent. _Cheaper and worse, or cheaper @@ -528,14 +718,15 @@ design?_ Stated so a reviewer can aim at evidence rather than opinion. -| If this turned out to be true | Then | -| ----------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------ | -| Our parse of `schedule` disagrees with what actually fires | Deference degrades to pure staleness with a longer threshold; the `ClaudeAdapter` row still earns its keep on restart coverage alone | -| `cron_durable` flips to **true** upstream | Claude's crons survive restarts; the durability argument weakens sharply and this feature shrinks to bounds + console | -| Upstream ships a supervision/automations feature | Re-cut against it. The console and the question channel probably survive; the reactor probably does not | -| `AskUserQuestion` becomes non-blocking | D6 collapses — one channel suffices, and `raise_blocker` is unnecessary | -| The user's real loops are almost never blocked on questions | The console is over-built; ship the reactor and the status pill only | -| Pinning is removed or reworked upstream | D3 collapses back to the Direction A/B/C comparison, and C becomes the answer | +| If this turned out to be true | Then | +| ----------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| Our parse of `schedule` disagrees with what actually fires | Deference degrades to pure staleness with a longer threshold; the `ClaudeAdapter` row still earns its keep on restart coverage alone | +| `cron_durable` flips to **true** upstream | Claude's crons survive restarts; the durability argument weakens sharply and this feature shrinks to bounds + console | +| Upstream ships a supervision/automations feature | Re-cut against it. The console and the question channel probably survive; the reactor probably does not | +| `AskUserQuestion` becomes non-blocking | D6 collapses — one channel suffices, and `raise_blocker` is unnecessary | +| The user's real loops are almost never blocked on questions | The console is over-built; ship the reactor and the status pill only | +| Pinning is removed or reworked upstream | D3 collapses back to the Direction A/B/C comparison, and C becomes the answer. **Partly realised already**: `f70eeeeb0` reworked pin-vs-settle so a settled loop leaves the pinned block. D3 survived because its load-bearing claim was the zero row count, not the visibility — see the D3 note in §2 | +| Upstream continues threads across restarts by default | D1's durability argument narrows. **Checked**: `5b7d72aad` (#9167) does this, but only for threads with a live `activeTurnId`, only on the intentional self-update path, and only behind a setting that ships **off** — a pending wake is never marked. BACKEND §12 | **One falsifier has been struck.** The table's top entry used to read "`session_crons` is empty/absent in practice on real Claude sessions". It is **refuted**: the binary spreads @@ -548,8 +739,10 @@ wrong — takes its place, and Phase 1 still answers it. ## 12. Immediate next actions -1. **Review this plan** (that is what it is for). -2. Decide Q1–Q4. +1. ~~Review this plan~~ — done, twice: four reviews before #120 merged, then issue **#125**, whose + findings this revision resolves. What changed is listed below. +2. Decide Q1–Q4. **Q1 is resolved** (§10). Q2, Q3 and Q4 remain open; Q3's seam argument is now + measured rather than estimated. 3. Open a consolidated issue; close #42 pointing at it; note the residue of #38 is now covered. 4. Start **Phase 0** — it is pure debt paydown, zero seam, and safe to do before any decision lands. 5. Start **Phase 1** — it answers the biggest remaining `[A]` in the plan, the `schedule` parse, at @@ -557,3 +750,33 @@ wrong — takes its place, and Phase 1 still answers it. 6. When the design is accepted, register its vocabulary (loop, check-in, blocker, deference, held / spent / stalled, iteration ledger) in `docs/coil/CONTEXT.md` — created lazily then, per `docs/coil/agents/domain.md`, not before. + +### What the #125 review changed + +Grouped by whether it changes behaviour, cost, or only the words. + +**Behaviour — four fixes, two of which were latent bugs rather than contradictions:** + +- `deadlineAtMs` is **mandatory and non-nullable**; the route 400s. The null branch in guard 10b — + which meant _no deference at all_, so a deadline-less loop fired on top of a healthy self-pacing + thread — is gone (§A1). +- **Stop conditions moved from guard 13 to guard 4b**, ahead of every non-consuming skip. They were + evaluated only on a tick that had already decided the thread was idle, so a busy self-paced run + walked past its own deadline and a `done` sentinel went unnoticed until the thread went quiet. +- **Guard 5 (`settledOverride !== "settled"`) is retired.** Upstream #8600 made settlement a + server-side sweep with no provenance marker, so the flag no longer carries human intent — the same + correction `autoResume/guards.ts` already made after a timer destroyed an armed week-long resume. +- **`thread.pin` is a promotion, not a decoration** — it emits companion `thread.unsettled` / + `thread.unsnoozed`. Arming a snoozed thread now 400s; disarm unpins only what the loop pinned. + +**Cost — re-measured against the 2026-09-02 merge-base `941acb4f9`:** phase 5 fell from `0–3 rows +[A]` to **1 row at risk 6**; phase 4 rose to ~350 risk and is now the most expensive part of the +feature; the phase 3 row is cheaper than recorded. `store.global.enabled` gained a data model and +routes and **moved from phase 4 to phase 2**, because phases 2 and 3 were otherwise unswitchable. + +**Words — contradictions closed:** the master toggle has one semantic (§A3, a guard, never "no +fiber"); the grace boundary is inclusive everywhere (§A4); D2's "never `session.status`" is scoped +so a PLAN-only reader keeps the long-tool-call case (§A5); the console-precedence reversal is +declared with its reason (§A2); push, mobile row chrome and the run digest move from "designed" to +**Out** (§B8, §B9); `gate_off` gets a verified source and a named render slot (§B10); the archived +design's superseded-by banner names the premises that actually moved (§A6). diff --git a/docs/coil/loops-v2/TESTS.md b/docs/coil/loops-v2/TESTS.md index 6a6bbad17259..9c1c00595ded 100644 --- a/docs/coil/loops-v2/TESTS.md +++ b/docs/coil/loops-v2/TESTS.md @@ -6,8 +6,18 @@ Written against the conventions already in the tree, not invented: `decide.ts`, `guards.ts`, `config.ts`, `sentinel.ts` are pure precisely so most of this list runs in microseconds. - **Reactor / time** — `@effect/vitest` with `TestClock`, stub `OrchestrationEngineService`, - `ProjectionSnapshotQuery` and `ProviderService` layers. Pattern is - `coil/autoResume/Reactor.test.ts` verbatim. + `ProjectionSnapshotQuery` and `ProviderService` layers, from + `coil/autoResume/Reactor.test.ts` — except for how the scenarios **wait**. That pattern spins + the schedulers a fixed number of turns and looks again, which is a bet on how fast the machine + is: the store persists through real filesystem I/O, so on a loaded two-core runner a turn buys + less progress and a budget that is generous on a laptop runs out. It cost this branch three CI + failures naming three different sets of tests, and because the harness kept advancing the + simulated clock while it guessed, they arrived as assertions about the _product_ — a wake + covered twice, a check-in five simulated minutes late — rather than as anything that looked + like a timing bug. The reactor therefore announces: `coil/loop/receipts.ts` publishes a + receipt at every milestone, `tick.completed` last of all, and every wait in `reactorHarness.ts` + is an await on one. `advancePolls(n)` is exactly `n` whole ticks on any machine. The service is + optional and no production layer provides it, so the reactor's emitter is a no-op there. - **Store** — real `FileSystem` against a temp dir (`NodeServices`), same as `autoResume/state.test.ts`. - **Routes** — `http.test.ts` shape from `autoResume/http.test.ts`. @@ -16,9 +26,11 @@ Written against the conventions already in the tree, not invented: Counts below are cases, not files. **★** marks a case that encodes a bug this design exists to prevent — if you cut scope, do not cut these. -**The total is 162.** Cases inserted into an existing sequence carry a letter suffix rather than -renumbering everything after them (`11b`–`11k`, `70b`–`70i`, `118b`–`118f`, `136b`–`136c`), so a -case number cited elsewhere keeps meaning the same case. +**The total is 183.** Cases inserted into an existing sequence carry a letter suffix rather than +renumbering everything after them (`11b`–`11l`, `45b`–`45d`, `70b`–`70k`, `86b`–`86e`, +`118b`–`118h`, `136b`–`136c`), so a case number cited elsewhere keeps meaning the same case. The +2026-09-02 review (issue #125) added 21 and rewrote four; every one of them is marked +**`[#125]`** so the delta is auditable. --- @@ -57,8 +69,11 @@ decide whether the two schedulers cooperate or fight. inside the run. 11d. `nextFireAtMs` in the past **with** `updatedAt` movement after it → the wake landed; clear it and treat the thread as normally active. Detected immediately, without waiting out `graceMs`. -11e. `nextFireAtMs` overdue by more than `graceMs` **without** `updatedAt` movement → `fire`, -reason `wake_lost`. ★ Inside `graceMs` T3 still stands down. `graceMs` is **derived from the +11e. `nextFireAtMs` overdue by `graceMs` **or more**, without `updatedAt` movement → `fire`, +reason `wake_lost`. ★ **The boundary is inclusive** — guard 10b is `now >= nextFireAtMs + graceMs`, +the same way case 2's threshold is inclusive, and case 11k says so at the other end. `[#125]` an +earlier draft of this case said "more than", which disagreed with both. Strictly inside `graceMs` +T3 still stands down. `graceMs` is **derived from the recorded entry**, not a constant: `max(90s, min(0.10 × period, 15min))` when `recurring: true`, 90s for a one-shot. This is the strongest trigger in the design: an unmet commitment, not an inference — so the grace has to be wide enough that jitter alone never trips it (11k). @@ -75,7 +90,8 @@ that the exposure was bounded without a second rule because the binary clamps a `ScheduleWakeup` takes `delaySeconds`, clamped; `CronCreate` takes a 5-field expression with no hour bound. Unbounded deference therefore stands T3 down for up to 24h on the first, and indefinitely on the second. The deadline is the cap because past it there is nothing left to -defer to; whether that is the right cap, or an explicit `maxDeferMs` is, is open beside Q1. +defer to. `[#125]` **Resolved**: the deadline is the cap and no `maxDeferMs` knob is added +(PLAN Q1). 11k. A `recurring: true` 30-minute wake that lands **2 minutes late** → **no** `wake_lost` and no fire. ★ From the binary's own scheduler text: "recurring tasks fire up to 10% of their period late (max 15 min); one-shot tasks landing on :00 or :30 fire up to 90 s early." The 90s is the @@ -86,15 +102,32 @@ derivation too: a 30-minute period tolerates 3 min, a 20-minute period 2 min, an period 15 min and not more (the cap binds). The floor is inclusive the same way case 2 is: at exactly `graceMs` the wake counts as lost. +11l. `[#125]` A record whose `deadlineAtMs` is the fail-closed decoding default `0` **never +defers** and never fires: guard 4b stops it as `spent` before 10b is reached. ★ Assert the stop +reason, and assert that no `nextFireAtMs` — however near — can produce a `skip` from such a +record. This is the case the old `deadlineAtMs == null` branch got backwards: it read a missing +deadline as "defer to nothing", i.e. fire freely on a healthy self-pacing thread. + ### 1.2 Budget and deadline 12. `checkInsUsed < maxCheckIns` → allowed. 13. `checkInsUsed === maxCheckIns` → `stop("spent")`. 14. `now >= deadlineAtMs` → `stop("spent")` **even when budget remains**. ★ -15. Deadline null → never stops on time. +15. `[#125]` **Rewritten.** `deadlineAtMs` is a `number` and is never null — a null deadline is + not a state (BACKEND §3). What replaces the old "deadline null → never stops on time" case is + its inverse: a record decoded with the fail-closed default `0` → `stop("spent")` on the first + evaluation. ★ A deadline that did not survive a write must mean "over", never "unbounded"; + the opposite default turns one corrupted byte into an unbounded overnight spend. 16. Deadline in the past at arm time is rejected by the route, not silently accepted (see §5). 17. `spent` is returned as `spent`, never as `done`. ★ (assert the literal, not truthiness) +15b. `[#125]` The deadline stops the loop **while the thread is busy** — `busyTurn` true, idle +below `busyIdleMs`, `now >= deadlineAtMs` → `stop("spent")`. ★ Guard 4b is swept before every +○ guard for exactly this reason: with the old ordering a thread that never went idle never +reached the stop check, and a self-paced run strolled through its own deadline indefinitely. +15c. `[#125]` The sentinel is honoured while the thread is busy, for the same reason → +`stop("done")`, not "noticed once it goes quiet". ★ + ### 1.3 Strikes 18. Movement ≥ `productiveMs` after a check-in resets strikes to 0. @@ -129,7 +162,16 @@ Each guard gets: passes-when-satisfied, blocks-when-not, and **the right kind of 31. `armed === false` → skip. 32. Shell `None` → **disarm** (thread deleted). 33. `archivedAt !== null` → disarm. -34. `settledOverride === "settled"` → skip, budget intact. ★ +34. `[#125]` **Inverted.** `settledOverride === "settled"` → the loop **proceeds**; settledness + never blocks a check-in. ★ Guard 5 is retired. Upstream #8600 moved settlement server-side — + `ThreadSettlementReactor` sweeps every minute and dispatches `thread.auto-settle`, which shares + `thread.settle`'s decider case and emits the same event **with no provenance marker** — so the + flag no longer distinguishes "a human is done here" from "a timer fired". `autoResume/guards.ts` + already made this correction after a timer destroyed an armed week-long resume on day 3. + 34b. `[#125]` The failure the retirement prevents, as a scenario: arm a loop, let a simulated + auto-settle write `settledOverride: "settled"`, and assert the loop still checks in. ★ With the + old guard it would have sat armed and silently done nothing until its deadline, then reported + `spent` — a ○ skip never stops, so the failure is invisible rather than loud. 35. `snoozedUntil` in the future → skip, budget intact. ★ 36. `snoozedUntil` in the past → passes. 37. `hasPendingApprovals` → skip. ★ @@ -141,6 +183,15 @@ Each guard gets: passes-when-satisfied, blocks-when-not, and **the right kind of 43. `now - lastCheckIn.firedAtMs < idleMs` → skip, **even if the idle threshold appears met**. ★ (structural anti-tight-loop floor; must hold even when `updatedAt` never bumps) 44. `armedCount >= maxArmedThreads` → skip. + 45b. `[#125]` **Guard 4b is swept before every ○ guard.** A record that is both past its deadline + _and_ rate-limited reports `stop("spent")`, not `skip("rate-limited")`. ★ Assert the returned + decision, because "the loop is held" and "the loop is over" are different words on the console + and the wrong one hides a finished run behind a hold. + 45c. `[#125]` Guard 4b runs after guard 4, not before it: a deleted thread **disarms**, it does not + report `spent`. Order matters in both directions. + 45d. `[#125]` Guard 2 (the master toggle) still precedes 4b: with the toggle off, a loop past its + deadline reports `standing_down` / `disabled` and is **not** stopped. ★ The toggle stands loops + down; it never manufactures terminal states nobody chose. 45. **Guard order is asserted explicitly**: a record that trips several guards reports the _first_ one, because that string is what the console renders. ★ 46. A skip never increments `checkInsUsed` — asserted across every ○ guard in one table-driven case. ★ @@ -174,6 +225,11 @@ Each guard gets: passes-when-satisfied, blocks-when-not, and **the right kind of One case per field — this is the highest-severity footgun in the module, because a whole-file decode failure becomes `EMPTY_STATE` and silently disarms every loop. 62. A corrupt/truncated file → `EMPTY_STATE` and an error log, never a throw at boot. + 61b. `[#125]` Every decoding default is asserted to be the **fail-closed** value, not merely + present: `armed: false`, `deadlineAtMs: 0`, `maxCheckIns: 0`, `crons: null`, + `pinnedByLoop: false`, `global.enabled: false`. ★ Case 61 proves a missing field still decodes; + this proves it decodes to the reading that spends nothing. A default meaning "unbounded" would + turn one truncated write into an unbounded overnight spend. 63. Unknown extra keys are tolerated (forward compatibility with a newer build). 64. Concurrent mutations from two fibers serialize through the `SynchronizedRef` with no lost update. ★ @@ -209,6 +265,14 @@ today's. A fork observability bug must not be able to break a turn. 70g. `SubagentStop` is handled identically to `Stop`. 70h. The record survives a store round-trip (it is the only durable copy of the wake). +70j. `[#125]` A `PostToolUse` callback matched on `ScheduleWakeup` whose `tool_response` +stringifies to something containing `gate_off` writes `degraded: "gate_off"`. ★ The plumbing is +verified — `HookCallbackMatcher.matcher` selects by tool name, and `PostToolUseHookInput` carries +`tool_name` and `tool_response` `[V - external]` — but the response **body** is not, so this is a +substring probe, not a parse. +70k. `[#125]` A `tool_response` that does **not** contain the marker leaves `degraded` untouched — +it is never inferred, never guessed, and a successful call never clears an unrelated degraded +state by accident. ★ A probe that finds nothing must behave exactly like no probe. 70i. The entry's `prompt` arrives **truncated to 1000 characters by the binary**. ★ Assert a longer prompt round-trips as the truncation, and that no fork path treats it as the agent's full prompt — the trigger reads `schedule` and `recurring`; `prompt` is console display text and @@ -227,8 +291,20 @@ receive. non-bypassable; a silent clamp hides a mistake) 76. `POST` arm with `maxCheckIns < 1` → 400. 77. `POST` arm with a deadline in the past → 400. ★ + 77b. `[#125]` `POST` arm with **no** deadline → `400 deadline_required`. ★ Not a clamp and not a + default: a null deadline is not a state (D9, BACKEND §3). Assert the error code, because the + console distinguishes it from the past-deadline 400. 78. `POST` arm when already at `maxArmedThreads` → 400, and **the tick re-checks it too**, so a hand-edited state file cannot exceed the ceiling. ★ + 78b. `[#125]` `POST` arm on a **snoozed** thread → `400 thread_snoozed`, and **no** `thread.pin` is + dispatched. ★ `thread.pin`'s decider case emits companion `thread.unsettled` / + `thread.unsnoozed` events `[V]`, so arming would otherwise silently cancel a snooze the human + set. Assert the absence of the dispatch, not just the status code. + 78c. `[#125]` `POST` arm on a **settled** thread succeeds and pins; the resulting unsettle is + correct, because arming is the human asking for the thread to run. + 78d. `[#125]` `pinnedByLoop` gates the unpin: arming a thread that was **already pinned** records + `pinnedByLoop: false`, and disarming it dispatches **no** `thread.unpin`. ★ Otherwise disarming + a loop removes a pin the user set themselves, and nothing records that it was ever theirs. 79. `POST` disarm on a running loop → disarmed, terminal reason `handed-back`. 80. `POST` re-arm after `spent` → clears terminal, fresh budget. 81. `POST answer` on a blocker → recorded, `deliveredToAgent` false. @@ -238,6 +314,17 @@ receive. 84. Malformed JSON body → 400, no state mutation. 85. Every response shape decodes against its schema (guards against drift with the client). 86. `GET /api/coil/loops` returns every armed loop across projects, ordered deterministically. + 86b. `[#125]` `GET /api/coil/loop/settings` on a fresh install returns the global block with + `enabled: false` and every default populated from config. The route exists in **phase 2**, with + the reactor — shipping "default off behind the master toggle" while only phase 4 could flip it + left phases 2 and 3 unswitchable. + 86c. `[#125]` `POST /api/coil/loop/settings` writes the master toggle durably, and the next tick + observes it. ★ The toggle is re-read every tick _and_ pre-dispatch, so assert both. + 86d. `[#125]` Toggling off leaves every armed loop **armed**: `armed` stays true, `stopped` stays + null, `checkInsUsed` is unchanged, and each reports `standing_down` / `disabled`. ★ Toggling + back on resumes the same loops with the same budgets. The toggle is a guard, not a lifecycle. + 86e. `[#125]` `POST` settings with `maxArmedThreads` below the current armed count is accepted and + the excess loops stand down at the next tick rather than being disarmed — same rule as 86d. --- @@ -295,6 +382,10 @@ receive. 116. `loop_done` writes the terminal state and is equivalent to the sentinel file. 117. `loop_done` from a thread with no loop is a no-op, not a crash. 118. All three tools are unavailable when the global toggle is off. ★ + 118g. `[#125]` With `enableAgentBrowserAccess` off, the MCP credential is never minted `[V]`, so all + three tools vanish. Assert the console renders a **named** degraded state, not an empty blocker + list. ★ A missing question channel that looks like "no questions" is the exact failure mode + `raise_blocker` exists to prevent. ### 7b. Voided questions (upstream #5127) @@ -306,6 +397,14 @@ The agent gets `{}` from upstream; the console must not report that as a human d 118e. The console renders a voided question as still needing attention, with the reason. ★ 118f. A voided question does **not** count as a blocking guard hit on the next tick — the block is gone, so the loop may proceed, but the console still shows it. ★ +118h. `[#125]` **The loop can manufacture its own blocker.** Upstream #8144 added `onUserDialog` +with `supportedDialogKinds: ["resume_return"]` to the same `queryOptions` object phase 1 edits +`[V]`, routing into the same blocking `Deferred` as `AskUserQuestion` and firing on **session +resume** — so a check-in landing on a torn-down session can park the loop on a dialog the loop +itself caused. Assert three things: guard 8 still skips (nudging past a pending input is worse); +the fork's `user-input.requested` record carries the **dialog kind**, so the console can say +"waiting on a session-resume confirmation since 01:04" rather than showing an unexplained idle +loop; and the loop spends nothing while parked and ends `spent`, not `stalled`. ★ --- @@ -338,6 +437,15 @@ Heavier tests; a handful, each replaying a real failure. deadline, strikes and `armedAtMs` all survive and the loop continues from check-in 3. ★ 130. **Reboot storm.** Three armed threads, all long-idle, server restarts. Assert none fire on the first tick and they do not all fire simultaneously afterwards. ★ + 130b. `[#125]` **Upstream's restart continuation.** `5b7d72aad` (#9167) re-establishes a binding + after a self-update and dispatches `session.status: "starting", activeTurnId: null` + synchronously at startup, while the actual `sendTurn` waits on server activation — a real + window in which a continued thread looks idle with no error. Assert the loop does not fire in + it and spends nothing. ★ Two independent mechanisms already cover it and **neither was added + for this**, which is the point of the case: `busyTurn` counts `"starting"`, so the fuse is + `busyIdleMs`; and `processStartedAtMs` floors the idle clock at process start. Also assert the + premise the feature rests on: a thread waiting on a scheduled wake has no `activeTurnId`, so + upstream never marks it for continuation and the durability gap is untouched. 131. **Loop vs auto-resume.** A rate limit arrives while a loop is armed with auto-resume off. Assert the loop does not nudge into the live limit and spends no budget. ★ 132. **Loop vs auto-resume, the other direction.** A pending auto-resume exists; assert the loop @@ -390,3 +498,8 @@ Heavier tests; a handful, each replaying a real failure. without that property fails the suite. - A test that fails if a new field is added to `LoopRecord` without a decoding default. ★ (schema-reflective; this is the failure that silently disarms every loop on the machine) +- `[#125]` **A guard-provenance review, once per guard, written down rather than tested.** Before + any guard is added, answer: _could a server timer write this value?_ Guard 5 died because + `settledOverride` changed answer from "no" to "yes" when upstream #8600 landed, and nothing in the + suite could notice — the old test passed, asserting the wrong behaviour. No test catches a + predicate whose _meaning_ moved, so this is a checklist item, not a case. diff --git a/docs/coil/loops-v2/UPSTREAM-DELTA.md b/docs/coil/loops-v2/UPSTREAM-DELTA.md index cde744091e86..55408499ea09 100644 --- a/docs/coil/loops-v2/UPSTREAM-DELTA.md +++ b/docs/coil/loops-v2/UPSTREAM-DELTA.md @@ -13,7 +13,7 @@ origin/main df027ec08 116 behind upstream — the sync has not lan > **Superseded the same day — the sync landed.** While this was being written, the daily sync > force-landed onto `main`, putting the fork on merge-base **`a4cc1367b`** — exactly the tree > everything below was measured against — **0 commits behind upstream**. The merge-base is cited -> throughout rather than a fork `main` SHA, because every sync rewrites `main`. Every +> throughout rather than a fork `main` SHA, because every sync rewrote `main` until the 2026-09-02 switch to merge-based syncs (PR #132); anchoring on the merge-base still holds, because that is what the seam ledger measures against. Every > finding therefore describes the fork's _current_ `main`, not a future one, and every check below > was re-run against that tree and still passes. Two consequences, both good: the sequencing > question in §6 is moot, and the seam ledger re-baselined to **53 files, +2590 / −1042**. @@ -271,8 +271,9 @@ because it is the tree the work would actually be built on. No claim in this document changed on the way across. `origin/main` was **`f6355f06f`** when this table was run. The next daily sync force-rewrote it to -**`94c6328ef`** on 2026-08-18. Fork `main` SHAs are rewritten by every sync — which is exactly why -the merge-base is the anchor this package cites. +**`94c6328ef`** on 2026-08-18. Fork `main` SHAs were rewritten by every sync until the 2026-09-02 +switch to merge-based syncs (PR #132) — which is why the merge-base is the anchor this package +cites, and it still is: the seam ledger measures against it either way. **Retracted: that sync did not hold the merge-base.** This paragraph said `94c6328ef` sat on the _same_ base `a4cc1367b`, with two extra upstream commits riding along. The base **moved**, and the @@ -328,3 +329,104 @@ Undocumented anywhere in this package until now: the binary truncates each entry the agent's full prompt text is wrong — it receives a prefix. Harmless for the `id`, `schedule` and `recurring` fields the deference rule reads; not harmless for anything that wants to match on prompt content, hash it, or round-trip it back into a turn. + +--- + +## 9. Second re-verification — 2026-09-02 (issue #125 §D) + +Two syncs have landed since §7. The merge-base moved `cebac353d` → **`941acb4f9`** (the 2026-09-02 +sync, 182 upstream commits, issue #128), and the seam ledger re-baselined to **53 files, ++2609 / −981** — the same 53 files, so no row was added or retired by the sync itself. + +### 9.1 Both facts #125 flagged as stale are confirmed, and both are now in the fork's tree + +| Change | Effect on this package | +| ---------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | ------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------------- | +| **`f70eeeeb0` (#7969, 2026-08-23) inverted pin-vs-settle.** The contract now reads _"Settled and snoozed threads remain in their respective shelves even when pinned"_, and the sidebar partition is a single chain: **snoozed → settled → pinned → active** `[V]`, mirrored in `apps/mobile/src/features/threads/threadListV2.ts` | **D3 survives, with lower confidence.** Loop-as-pinned-thread no longer keeps a _settled_ loop visible. But D3's load-bearing claim was the **zero row count** in `Sidebar.tsx`, not the visibility, and that is untouched. FINDINGS §A1 and PLAN D3 corrected | +| **`c7222ca4d` (#8144, 2026-08-25) added a second blocking dialog.** `onUserDialog` with `supportedDialogKinds: ["resume_return"]` sits in the _same_ `queryOptions` object phase 1 edits, routes into the same blocking `Deferred` as `AskUserQuestion`, and fires on **session resume** `[V]` | **A real new failure mode**: the loop's own nudge, landing on a torn-down session, can manufacture a pending user-input that guard 8 then treats as a hard skip — the loop causes the thing that parks it. No new guard; the fork's `user-input.requested` record carries the dialog kind so the console can name it, and the deadline ends it. BACKEND §9.1c, TESTS 118h | + +### 9.2 Churn has moved a lot, and it moved the plan's conclusions + +Re-measured with the SEAMS.md recipe against `941acb4f9`. #125 was right that the figures only +reproduced against the older base; they now reproduce against the stated one. + +| File | Churn then | Churn now | Consequence | +| ------------------------------------------- | ---------- | --------- | -------------------------------------------------------------------- | +| `provider/Layers/ClaudeAdapter.ts` | 16 | **23** | Phase 1's row is dearer but still one additive line — risk 23 | +| `settings/settingsSearch.ts` | 14 | **24** | Phase 4 is now **the most expensive phase in the plan** at ~312 risk | +| `settings/SettingsSidebarNav.tsx` | — | **19** | ~38 risk; both settings rows are type-forced, not optional | +| `routes/_chat.$environmentId.$threadId.tsx` | 4 | **2** | Phase 3's row is _cheaper_ than recorded — risk 32, delta still zero | +| `mcp/McpHttpServer.ts` | 6 | **2** | Phase 5 is one row at risk **6** | +| `mcp/McpSessionRegistry.ts` | 7 | **1** | not taken — see §9.3 | +| `mcp/McpInvocationContext.ts` | 3 | **1** | not taken — see §9.3 | +| `settings/SettingsPanels.tsx` | 36 | **43** | still avoided; ~2900 risk if it were not | +| `contracts/settings.ts` | 26 | **38** | still avoided; D12 | + +**The inversion is the finding.** The plan was written expecting phase 5 (agent-facing MCP tools) +to be the expensive, scary one and phase 4 (a settings entry) to be routine. Measured, it is the +other way round: the `mcp/` neighbourhood is the quietest in the plan, and the **navigation entry +carries ~350 of the feature's ~411 total risk**. + +### 9.3 Phase 5's seam cost, measured rather than estimated + +PLAN carried `0–3 rows [A]`; the archived design guessed three files. Three paths exist: + +- **Path A — a gated `"loop"` capability: 4 rows, risk 16. Rejected.** `McpCapability` is backed by + a runtime `Schema.Literal("preview")` in `packages/contracts/src/previewAutomation.ts`, so + widening the union is a **contracts edit** (breaking D12), and it would make `raise_blocker` fail + with an error class named `PreviewAutomationUnavailableError`. It buys nothing: nothing at + registration or dispatch consults `capabilities` — `requireMcpCapability` is one voluntary line + inside one handler helper — and the MCP credential is already per-thread and already + all-or-nothing `[V]`. +- **Path B — an ungated toolkit: 1 row (`McpHttpServer.ts` `+2/−1`, churn 2, risk 6). The decision.** +- **Path C — register from `coil/index.ts`: 0 rows. Worth one attempt, not typechecked.** + `McpServer.toolkit(t)` and `layerHttp` provide the same `McpServer` layer value, so Effect's + MemoMap should hand both the one instance — the property `coil/index.ts` already relies on. The + blocker is typed, not behavioural: a declared `McpInvocationContext` dependency leaks into the + layer's `R`. Upstream's own `registerPreviewSnapshot` shows the way out (`Effect.withFiber` + + `Context.getUnsafe`). Take Path B the moment it fights back; the difference is one row at risk 6. + +**A coupling phase 5 inherits and cannot fix:** the MCP credential is minted only when +`enableAgentBrowserAccess` is true `[V]`, so turning **Settings → Integrations → Agent browser +access** off silently removes `raise_blocker`, `loop_status` and `loop_done`. TESTS 118g. + +### 9.4 `5b7d72aad` (#9167) — upstream continues active threads across a restart + +Arriving on the next sync, and the closest upstream has come to this feature's territory. It does +**not** take D1's premise away, for four reasons, each from the commit `[V]`: + +1. It marks only threads with `session.status === "running" && activeTurnId !== null`. **A thread + waiting on a scheduled wake is neither, so a pending wake is never marked and never continued** — + which is exactly the gap this feature covers. +2. It is opt-in and ships **off** (`continueThreadsAfterServerUpdate` decodes to `false`). +3. It writes its marker only on the **intentional self-update** path. A crash, an OOM, a `kill` or a + machine reboot — the cases a supervisor is for — behave exactly as before. +4. It does not continue the interrupted turn; `activeTurnId` is nulled and a fresh turn is started. + +It does introduce a **double-fire window**: reconciliation dispatches `status: "starting", +activeTurnId: null` synchronously at startup while the `sendTurn` waits on server activation, so a +continued thread looks idle for a real interval with no error and no lease. This design already +covers it twice over — `busyTurn` counts `"starting"` (so the fuse is `busyIdleMs`) and +`processStartedAtMs` floors the idle clock at process start — and **neither mechanism was added for +this**. BACKEND §12; TESTS 130b. + +### 9.5 Everything else re-checked, and still true + +| Claim | Result at `941acb4f9` | +| ----------------------------------------------------------------------- | --------------------------------------------------------------------------------------------------------- | +| `session_crons` / `ScheduleWakeup` / `CronCreate` in server + contracts | **0 files** | +| `options.hooks` set in `ClaudeAdapter` | **0** — the beachhead is still unclaimed | +| `mcp/toolkits/` contents | **`preview` only** | +| the `mcpServers` spread anchor in `queryOptions` | **present**, and `onUserDialog` now sits beside it | +| `pinnedAt` / `thread.pin` / `thread.unpin` in contracts | **present** | +| an actor check on `thread.pin` | **none** — commands carry no actor; the decider guards on archival alone | +| a cron parser dependency anywhere in the repo | **none** — `git grep -i cron -- '*package.json'` and a `cron` search over `pnpm-lock.yaml` are both empty | +| two independent scope-auth mirrors (phase 0's debt) | **still two** — and they are no longer equivalent, see below | + +**One correction to phase 0's brief.** The webPush helper is _not_ strictly more general than the +autoResume one. It is parameterised on scope and returns the session, which autoResume's is not — +but it calls `failEnvironmentAuthInvalid` with **one** argument where autoResume's passes +`EnvironmentAuth.serverAuthDpopFailureReason(error)` as the second. That argument is `3bdf109e2` +(2026-09-02), which landed precisely because the stale one-argument call compiled and drifted +silently. Promoting webPush's body verbatim would re-introduce that bug in a shared helper, for all +three callers. Promote the **union**: parameterised, session-returning, and DPoP-reporting. diff --git a/docs/coil/loops-v2/build-report.mjs b/docs/coil/loops-v2/build-report.mjs index d10ebfca5dc3..c78c5eea9d55 100755 --- a/docs/coil/loops-v2/build-report.mjs +++ b/docs/coil/loops-v2/build-report.mjs @@ -74,7 +74,32 @@ const out = src.replace( }, ); -if (embedded === 0) throw new Error("no EMBED markers matched — the marker syntax has drifted"); +/** + * Assert the COUNT, not merely "at least one". + * + * The old guard only fired when *zero* markers matched, so a single drifted marker — the hint + * regex terminates on `>` and splits on `|`, both of which are easy to type into a caption — + * silently emitted a report missing that frame. Invisible in the browser and invisible in a + * 15k-line diff. Reproduced with 7 of 8 inlined and no error (issue #125 §C11). + * + * Two directions, because they catch different mistakes: a marker that stopped matching, and a + * prototype nobody referenced. + */ +const prototypeFiles = NodeFS.readdirSync(protoDir) + .filter((name) => name.endsWith(".html") && !name.startsWith("_")) + .sort(); + +if (embedded !== prototypeFiles.length) { + throw new Error( + `inlined ${embedded} prototypes but prototypes/ holds ${prototypeFiles.length} ` + + `(${prototypeFiles.join(", ")}) — a marker has drifted, or a prototype was added without one`, + ); +} + +const unreferenced = prototypeFiles.filter((name) => !src.includes(`<!--EMBED:${name}|`)); +if (unreferenced.length > 0) { + throw new Error(`prototypes never referenced by report.src.html: ${unreferenced.join(", ")}`); +} const doctype = "<!doctype html>"; if (!out.startsWith(doctype)) throw new Error("report.src.html no longer starts with a doctype"); diff --git a/docs/coil/loops-v2/prototypes/p4-settings.html b/docs/coil/loops-v2/prototypes/p4-settings.html index 9b4bdc1b6dac..94d5dfa4026d 100644 --- a/docs/coil/loops-v2/prototypes/p4-settings.html +++ b/docs/coil/loops-v2/prototypes/p4-settings.html @@ -283,8 +283,9 @@ <h1 class="set-h1">Loops</h1> <div> <div class="gate-t">Let threads run as loops</div> <div class="gate-d" id="gateDesc"> - When this is off, no thread can be armed, nothing is scheduled, and the loop - supervisor never starts. Existing loops stop at their next check-in. + When this is off, nothing fires and no thread can be armed. Loops that are + already armed stand down — they keep their budget and their deadline, and resume + when you switch this back on. Nothing is disarmed and nothing is stopped. </div> </div> <button @@ -627,8 +628,8 @@ <h2 class="set-section-h">What the loop says when it checks in</h2> gate.classList.toggle("off", !on); body.classList.toggle("disabled", !on); gateDesc.textContent = on - ? "When this is off, no thread can be armed, nothing is scheduled, and the loop supervisor never starts. Existing loops stop at their next check-in." - : "Loops are off. Nothing is armed, nothing is scheduled, and the supervisor is not running. Three armed loops were stood down."; + ? "When this is off, nothing fires and no thread can be armed. Loops that are already armed stand down — they keep their budget and their deadline, and resume when you switch this back on. Nothing is disarmed and nothing is stopped." + : "Loops are off. Three armed loops are standing down. They keep their budgets and deadlines and will resume where they left off when you switch this back on."; }); document.querySelectorAll(".switch:not(#master):not([disabled])").forEach((s) => s.addEventListener("click", () => { diff --git a/docs/coil/loops-v2/prototypes/p7-console-shapes.html b/docs/coil/loops-v2/prototypes/p7-console-shapes.html index d7d9e1cbb010..4f1283cf9509 100644 --- a/docs/coil/loops-v2/prototypes/p7-console-shapes.html +++ b/docs/coil/loops-v2/prototypes/p7-console-shapes.html @@ -293,6 +293,19 @@ <h1>Three places the console could live</h1> reading of "at any time I open the chat there should be a page where I have a questionnaire ready". </p> + <p class="shape-d"> + <b>Not the decision.</b> The plan takes the console as an + <b>overlay on the thread route</b> + instead, so the transcript stays the default view — a declared divergence from the stated + requirement, not an oversight (PLAN §3). The reason is #38's: an overlay has + <i>nothing to toggle back from</i>. A second default view needs a sticky per-thread toggle + that has to survive reload, agree across two windows, agree on mobile, and be discoverable + when it is wrong; and choosing the view means owning the thread route's render decision, + which is <code>ChatView.tsx</code> territory (an existing seam row at churn 83) rather + than the delta-zero overlay row. What is given up is one interaction. Shape A stays priced + here because it is the fallback if the night's tail turns out to be what sends people back + to typing "are you still working on it?". + </p> <div class="mock"> <div class="sb"> diff --git a/docs/coil/loops-v2/prototypes/p8-mobile.html b/docs/coil/loops-v2/prototypes/p8-mobile.html index 3447df123966..1f9eb56ec39c 100644 --- a/docs/coil/loops-v2/prototypes/p8-mobile.html +++ b/docs/coil/loops-v2/prototypes/p8-mobile.html @@ -657,6 +657,11 @@ <h2 class="sec">1 · The list, and the console</h2> (it carries the word, not just the zinc dot). That is why the tick row exists at all: it started as a mobile requirement and turned out to be the clearest thing on the desktop card too. <br /><br /> + <b>None of this is in phases 0–5, and the row chrome is not free.</b> Mobile surfacing + and push are <b>designed here and not built</b> — no phase, no seam line, no acceptance + criteria, no tests (PLAN §3 Out). And a budget on the mobile thread-list row is a + <b>mobile seam row</b>, in a file the fork does not touch today; the report priced it at + zero, which was wrong. Price it when mobile is built, not before. <br /><br /> Mobile also settles decision 1 in the report. Direction B — a bespoke "Loops" section in the sidebar — has no mobile equivalent, because mobile has no persistent sidebar to put it in. A loop-as-pinned-thread and a Loops tab both work on all three surfaces. That is not decisive diff --git a/docs/coil/loops-v2/report.html b/docs/coil/loops-v2/report.html index 9d1562b20546..b339f9e9e2a4 100644 --- a/docs/coil/loops-v2/report.html +++ b/docs/coil/loops-v2/report.html @@ -4228,6 +4228,19 @@ <h3 class="sub">Where the console renders — you said you were open to other wa reading of "at any time I open the chat there should be a page where I have a questionnaire ready". </p> + <p class="shape-d"> + <b>Not the decision.</b> The plan takes the console as an + <b>overlay on the thread route</b> + instead, so the transcript stays the default view — a declared divergence from the stated + requirement, not an oversight (PLAN &sect;3). The reason is #38's: an overlay has + <i>nothing to toggle back from</i>. A second default view needs a sticky per-thread toggle + that has to survive reload, agree across two windows, agree on mobile, and be discoverable + when it is wrong; and choosing the view means owning the thread route's render decision, + which is <code>ChatView.tsx</code> territory (an existing seam row at churn 83) rather + than the delta-zero overlay row. What is given up is one interaction. Shape A stays priced + here because it is the fallback if the night's tail turns out to be what sends people back + to typing "are you still working on it?". + </p> <div class="mock"> <div class="sb"> @@ -9614,6 +9627,11 @@ <h2 class="sec">Loops on the phone</h2> (it carries the word, not just the zinc dot). That is why the tick row exists at all: it started as a mobile requirement and turned out to be the clearest thing on the desktop card too. <br /><br /> + <b>None of this is in phases 0&ndash;5, and the row chrome is not free.</b> Mobile surfacing + and push are <b>designed here and not built</b> — no phase, no seam line, no acceptance + criteria, no tests (PLAN &sect;3 Out). And a budget on the mobile thread-list row is a + <b>mobile seam row</b>, in a file the fork does not touch today; the report priced it at + zero, which was wrong. Price it when mobile is built, not before. <br /><br /> Mobile also settles decision 1 in the report. Direction B — a bespoke "Loops" section in the sidebar — has no mobile equivalent, because mobile has no persistent sidebar to put it in. A loop-as-pinned-thread and a Loops tab both work on all three surfaces. That is not decisive @@ -10954,8 +10972,9 @@ <h2 class="sec">Settings — the on/off switch, and the ones that matter more</h <div> <div class="gate-t">Let threads run as loops</div> <div class="gate-d" id="gateDesc"> - When this is off, no thread can be armed, nothing is scheduled, and the loop - supervisor never starts. Existing loops stop at their next check-in. + When this is off, nothing fires and no thread can be armed. Loops that are + already armed stand down — they keep their budget and their deadline, and resume + when you switch this back on. Nothing is disarmed and nothing is stopped. </div> </div> <button @@ -11298,8 +11317,8 @@ <h2 class="sec">Settings — the on/off switch, and the ones that matter more</h gate.classList.toggle("off", !on); body.classList.toggle("disabled", !on); gateDesc.textContent = on - ? "When this is off, no thread can be armed, nothing is scheduled, and the loop supervisor never starts. Existing loops stop at their next check-in." - : "Loops are off. Nothing is armed, nothing is scheduled, and the supervisor is not running. Three armed loops were stood down."; + ? "When this is off, nothing fires and no thread can be armed. Loops that are already armed stand down — they keep their budget and their deadline, and resume when you switch this back on. Nothing is disarmed and nothing is stopped." + : "Loops are off. Three armed loops are standing down. They keep their budgets and deadlines and will resume where they left off when you switch this back on."; }); document.querySelectorAll(".switch:not(#master):not([disabled])").forEach((s) => s.addEventListener("click", () => { diff --git a/docs/user/loops.md b/docs/user/loops.md new file mode 100644 index 000000000000..fc1dbecdf83f --- /dev/null +++ b/docs/user/loops.md @@ -0,0 +1,126 @@ +# Loops + +A loop keeps one thread working while you are away, and collects everything it needs from you in +one place so that opening the thread in the morning answers a single question: **what do you need +from me?** + +T3 Code does not replace the agent's own pacing. Claude schedules its own wake-ups and is good at +it. What it cannot do is survive a restart — a wake it scheduled lives inside the running session, +so if the app restarts or the session stops, that wake disappears with no record it ever existed. +A loop is T3 Code's written-down copy: it stands by while the agent is pacing itself, covers a +wake that never lands, and enforces the bounds you set. + +## Arm a loop + +Open the thread you are about to walk away from. Above the composer there is a loop control; open +it and set: + +- **What it is working on.** One line. It is restated to the agent at every check-in, because a + long run compacts and a contract taught once is gone a few hours later. +- **Check-ins.** How many times the loop may restart the thread if it goes quiet. Between 1 and 20. + There is no unlimited option, deliberately. +- **Stop by.** A wall-clock time, in your timezone. Every run has one; a loop with no deadline is + refused rather than given a default. + +Arming pins the thread so it stays at the top of your sidebar. Disarming removes the pin again — +unless the thread was already pinned when you armed it, in which case your pin is left alone. + +A snoozed thread cannot be armed. Arming would cancel the snooze as a side effect, and that is your +decision to make, not the loop's. Unsnooze it first. + +## What the loop actually does + +Between check-ins the loop watches and spends nothing. It only nudges the thread when **all** of +these are true: + +- the thread has been completely silent for longer than the check-in threshold (a longer threshold + applies while a turn still looks busy, so a forty-minute test suite is not mistaken for a stall); +- there is no wake the agent scheduled for itself still pending inside the run's deadline; +- nothing is waiting on you — an approval, a question, or a plan you have not accepted; +- there is no usage limit in force. + +Each of those is a reason to **stand down**, not to stop. Standing down costs no check-in, and the +run keeps its budget and its deadline. + +On a healthy, self-pacing thread a loop should almost never fire. That is what a correct one looks +like. + +## The console + +The loop control above the composer expands into the console. It has five parts, and which ones +appear depends on what happened overnight. + +**Stopped on these.** Questions and approvals the agent could not get past. These are answered in +the composer, where the question already is — the console names them and points you at it. The +loop resumes on its own once you answer. + +**Answer when you can.** Questions the agent raised _without_ stopping, so it could keep working +around them. Answer them here, at whatever time suits you. An answer given while the thread is idle +is **banked**: the console says so, and the next check-in restates it to the agent. Answered and +"the agent has been told" are shown as different things, because they are. + +**Never answered.** Questions that were open when the session ended and were closed by the session +ending rather than by an answer. Nothing else in the app records that these existed. + +**State and bounds.** Which state the loop is in, how much of the budget is gone, when it stops, and +the next wake the agent scheduled for itself. + +**Check-ins.** One line per time the loop woke the thread, with what moved. These are observed +facts, never the agent's own account of its night. + +The console is an overlay: the transcript stays what you land on, and the console is one click +above it. + +## Deferred questions need agent browser access + +The "answer when you can" channel reaches the agent over the same per-thread connection that backs +the preview browser tools. If **Settings → Integrations → Agent browser access** is off, the agent +has no way to raise a question without stopping and waiting for you. + +The console says so by name when that happens. An empty list there does not mean the agent had +nothing to ask. + +## How a run ends + +- **Finished.** The agent said it was done — either by writing a `.coil/loop-done` file in the + project (one line saying why is enough) or by telling the loop directly. The file is the primary + contract: it works from a plain terminal. +- **Out of rope.** The budget or the deadline ran out. This is shown differently from "finished", + on purpose. The agent never said it was done; it was stopped. +- **Stalled.** Two check-ins in a row moved nothing, so the loop stopped spending. +- **Handed back.** You sent a message, or disarmed it. Taking over is not a budget reset: the run + keeps the check-ins it already spent. + +All four are sticky. A stopped loop stays stopped until you deliberately give it another run, which +clears the budget and starts fresh. **Clear** dismisses a finished run instead, and the panel goes +back to offering you a fresh loop. + +When a run ends while the agent still has wakes of its own pending, T3 Code ends the session too. +Those wakes live inside the session, and a bound that cannot stop the agent is not a bound. It is a +blunt instrument, and it will end any other background work in that session with it. + +## Settings → Loops + +**Let threads run as loops** is the master switch, and it is a **guard, not a lifecycle**. With it +off, nothing fires: every armed loop reports that it is standing down, and keeps its budget, its +deadline and its arm. Nothing is disarmed and nothing is stopped. Switch it back on and the same +loops carry on where they were. + +The same rule covers **Loops at once**. Lowering it below the number currently armed is allowed; the +excess stand down rather than being disarmed. + +The rest of the page is defaults — the thresholds and the budget a newly armed loop starts from, and +how far ahead its deadline is set. They seed the form when you arm a loop; changing them never +alters a run already in flight. + +**Armed right now** lists every loop across every project, with its state and what it has spent, so +"did any of my runs give up overnight?" is one page rather than three threads. Arming is not on this +page: it is a decision you make as you walk away from a thread. + +## One environment at a time + +Loops belong to the environment that runs the thread, and the app reads them from the environment +it is connected to as its primary one. On a thread hosted by a _different_ environment — a second +machine you have added, for instance — the loop control does not appear, rather than showing you +another machine's answer for that thread. Open that environment as your primary one to arm, watch +or disarm its loops. Auto-resume has the same reach today.