From 798eb68afe36c40d18ec1553e4d42ac29ed9d8e7 Mon Sep 17 00:00:00 2001 From: song <22676124+songoow@users.noreply.github.com> Date: Tue, 29 Sep 2026 06:23:40 -0400 Subject: [PATCH 01/22] feat(collaboration): emit a wake intent on the first accepted transition The transition to accepted is the one durable moment a requester can be continued without polling. transitionDelegationObservation now returns a wake_intent beside the accepted status: schema, an intent id derived from the requester, operation, request and accepted artifact digests, and the requester identity. accepted -> accepted stays an idempotent readback and emits nothing; rejection wakes nobody. The intent grants no Turn. Co-Authored-By: Claude Fable 5.1 Signed-off-by: song <22676124+songoow@users.noreply.github.com> --- .../control_plane/collaboration/delegation.ts | 25 ++++++++++++++ tests/control_plane_ts/delegation.test.ts | 34 +++++++++++++++++-- 2 files changed, 57 insertions(+), 2 deletions(-) diff --git a/loopx/control_plane/collaboration/delegation.ts b/loopx/control_plane/collaboration/delegation.ts index f661435365..a9d3becb56 100644 --- a/loopx/control_plane/collaboration/delegation.ts +++ b/loopx/control_plane/collaboration/delegation.ts @@ -338,9 +338,34 @@ export function transitionDelegationObservation(params: JsonObject): JsonObject if (to === "accepted") requireThat(params.canonical_done === true && params.acceptance_ready === true && params.artifacts_current === true, "accepted return requires current canonical completion and artifacts"); + if (to === "accepted" && from !== "accepted") return {status: to, wake_intent: delegationWakeIntent(params)}; return {status: to}; } +/** The first transition to ``accepted`` is the one durable moment a requester + * can be continued without polling. The intent names the requester and the + * exact accepted result; it grants no Turn and is not a second settlement. */ +function delegationWakeIntent(params: JsonObject): JsonObject { + const requester = requireJsonObject(params.requester, "wake requester"); + requireThat([requester.goal_id, requester.agent_id, requester.operation_id, requester.request_id].every(text), + "wake intent requires the requester and result identity"); + requireThat(requester.goal_ref === null || typeof requester.goal_ref === "object", "invalid requester goal reference"); + requireThat(Array.isArray(requester.artifacts) && requester.artifacts.length > 0, "wake intent requires accepted artifacts"); + const digests = requester.artifacts.map(value => { + const artifact = requireJsonObject(value, "accepted artifact"); + requireThat(text(artifact.ref) && typeof artifact.sha256 === "string" + && /^[a-f0-9]{64}$/.test(artifact.sha256), "invalid accepted artifact reference"); + return {ref: artifact.ref, sha256: artifact.sha256}; + }); + return { + schema: "loopx_delegation_wake_intent_v0", + intent_id: canonicalAuthoritySha256([requester.goal_id, requester.agent_id, requester.operation_id, + requester.request_id, digests]), + requester: {goal_id: requester.goal_id, agent_id: requester.agent_id, goal_ref: requester.goal_ref ?? null}, + operation_id: requester.operation_id, request_id: requester.request_id, + }; +} + /** Repair only a false terminal observation after the exact Turn validated. * * This does not retry model work. The host boundary must prove that the diff --git a/tests/control_plane_ts/delegation.test.ts b/tests/control_plane_ts/delegation.test.ts index 7ae9fe4d41..86b49acd2c 100644 --- a/tests/control_plane_ts/delegation.test.ts +++ b/tests/control_plane_ts/delegation.test.ts @@ -94,12 +94,42 @@ test("malformed operator binding fails before launch", () => { } }); +const acceptedRequester = {goal_id: "research", agent_id: "coordinator", goal_ref: null, + operation_id: "analysis-1", request_id: "req-1", artifacts: [{ref: "output.json", sha256: "a".repeat(64)}]}; +const acceptedFacts = {canonical_done: true, acceptance_ready: true, artifacts_current: true, requester: acceptedRequester}; + test("message receipt and model return do not imply accepted work", () => { assert.throws(() => transitionDelegationObservation({from: "prepared", to: "accepted"}), /transition/); assert.throws(() => transitionDelegationObservation({from: "turn_returned", to: "accepted"}), /canonical/); assert.throws(() => transitionDelegationObservation({from: "rejected", to: "running"}), /transition/); - assert.deepEqual(transitionDelegationObservation({from: "turn_returned", to: "accepted", - canonical_done: true, acceptance_ready: true, artifacts_current: true}), {status: "accepted"}); + const accepted = transitionDelegationObservation({from: "turn_returned", to: "accepted", ...acceptedFacts}); + assert.equal(accepted.status, "accepted"); +}); + +test("only the first transition to accepted leaves a wake intent for the exact requester and result", () => { + const accepted = transitionDelegationObservation({from: "turn_returned", to: "accepted", ...acceptedFacts}); + const intent = accepted.wake_intent as Record; + assert.equal(intent.schema, "loopx_delegation_wake_intent_v0"); + assert.match(String(intent.intent_id), /^[a-f0-9]{64}$/); + assert.deepEqual(intent.requester, {goal_id: "research", agent_id: "coordinator", goal_ref: null}); + assert.equal(intent.operation_id, "analysis-1"); + assert.equal(intent.request_id, "req-1"); + // Same requester and result: same intent, so a replayed transition cannot mint a second wake. + assert.deepEqual(transitionDelegationObservation({from: "turn_returned", to: "accepted", ...acceptedFacts}), accepted); + const changed = transitionDelegationObservation({from: "turn_returned", to: "accepted", ...acceptedFacts, + requester: {...acceptedRequester, artifacts: [{ref: "output.json", sha256: "b".repeat(64)}]}}); + assert.notEqual((changed.wake_intent as Record).intent_id, intent.intent_id); + // accepted -> accepted is an idempotent readback, never a new wake. + assert.deepEqual(transitionDelegationObservation({from: "accepted", to: "accepted", ...acceptedFacts}), {status: "accepted"}); + // A transition to accepted without the requester identity, or with a forged digest, has no wake to record. + assert.throws(() => transitionDelegationObservation({from: "turn_returned", to: "accepted", + canonical_done: true, acceptance_ready: true, artifacts_current: true}), /wake requester/); + assert.throws(() => transitionDelegationObservation({from: "turn_returned", to: "accepted", ...acceptedFacts, + requester: {...acceptedRequester, artifacts: [{ref: "output.json", sha256: "short"}]}}), /artifact/); + assert.throws(() => transitionDelegationObservation({from: "turn_returned", to: "accepted", ...acceptedFacts, + requester: {...acceptedRequester, artifacts: []}}), /artifacts/); + // Rejection is terminal and wakes nobody. + assert.deepEqual(transitionDelegationObservation({from: "turn_returned", to: "rejected"}), {status: "rejected"}); }); test("a false rejection can reopen only for exact validated settlement recovery", () => { From 7beb4a11ed06729621682267beef9e1d7f6cac12 Mon Sep 17 00:00:00 2001 From: song <22676124+songoow@users.noreply.github.com> Date: Tue, 29 Sep 2026 06:24:59 -0400 Subject: [PATCH 02/22] feat(collaboration): persist the wake intent and retire it when seen in-Turn The accepted observation now stores row["wake"] = {intent, state: pending} in the same atomic write that records accepted, so a result and its wake receipt cannot diverge. record_wake settles a pending intent under the same lock adopt_result uses and never touches a non-accepted row; the Chat LoopX tool marks the intent observed_in_turn whenever the lead sees an accepted result inside its own Turn, so no redundant wake follows. read_delegation exposes the receipt as a fact distinct from the result. Co-Authored-By: Claude Fable 5.1 Signed-off-by: song <22676124+songoow@users.noreply.github.com> --- loopx/chat_loopx_mode.py | 3 ++ loopx/collaboration_mcp.py | 58 ++++++++++++++++++++++++++++++++++++-- 2 files changed, 59 insertions(+), 2 deletions(-) diff --git a/loopx/chat_loopx_mode.py b/loopx/chat_loopx_mode.py index c3d44bde00..0a2ef42ca9 100644 --- a/loopx/chat_loopx_mode.py +++ b/loopx/chat_loopx_mode.py @@ -688,6 +688,9 @@ def dispatch(name, arguments): else: raise ValueError("unsupported collaboration action") if "status" in result: + if result["status"] == "accepted": + # Seen inside this Turn: the pending wake would be redundant. + result["wake"] = service.wake_observed_in_turn(operation_id) or result.get("wake") summary = { key: result.get(key) for key in ("operation_id", "agent_id", "todo_id", "status") diff --git a/loopx/collaboration_mcp.py b/loopx/collaboration_mcp.py index c3f2b9304b..bd320ec82d 100644 --- a/loopx/collaboration_mcp.py +++ b/loopx/collaboration_mcp.py @@ -300,6 +300,31 @@ def consume_peer_result(request_id: str) -> dict: ) +def execution_row_path(root: Path, goal_id: str, agent_id: str, operation_id: str) -> Path: + """The requester-scoped durable operation record; readable without a service.""" + return _root(root) / "executions" / _hash([goal_id, agent_id]) / (_hash(operation_id) + ".json") + + +def record_wake(path: Path, decide) -> dict | None: + """Settle a pending wake receipt under the same lock adopt_result uses. + + ``decide`` receives the pending intent and returns the replacement receipt, + or None to leave it unchanged. Only an accepted result with a pending + intent is decidable; any other terminal state wakes nobody. + """ + with exclusive_file_lock(path): + row = _read(path) + wake = row.get("wake") + if not isinstance(wake, dict) or wake.get("state") != "pending" or row.get("status") != "accepted": + return None + updated = decide(wake) + if updated is None: + return None + row["wake"] = updated + _write(path, row) + return updated + + class Delegations: """Host IO for bound peer work; typed grants and observations stay in TS. @@ -350,7 +375,7 @@ def directory(self) -> dict: for row in bindings]} def path(self, operation_id: str) -> Path: - return _root(self.root) / "executions" / _hash([self.goal_id, self.agent_id]) / (_hash(operation_id) + ".json") + return execution_row_path(self.root, self.goal_id, self.agent_id, operation_id) def operations(self, *, limit: int = 20, cursor: str | None = None) -> dict: from .control_plane.collaboration.delegation_inventory import read_delegation_inventory @@ -620,6 +645,10 @@ def adopt_result(self, operation_id: str, consumer_operation_id: str) -> dict: def read(self, operation_id: str) -> dict: result = self._read_current(operation_id) result.update(delegation_results.result_relationships(self, operation_id)) + wake = _read(self.path(operation_id)).get("wake") + if isinstance(wake, dict): + # Distinct from the result itself: whether the requester was continued. + result["wake"] = wake return result def _read_current(self, operation_id: str) -> dict: @@ -654,8 +683,32 @@ def _observe(self, path: Path, row: dict, status: str, **facts) -> None: "from": row["status"], "to": status, **facts, }) row.update(status=decision["status"]) + if isinstance(decision.get("wake_intent"), dict): + row["wake"] = {**decision["wake_intent"], "state": "pending"} _write(path, row) + def _wake_requester(self, row: dict) -> dict: + """Requester and exact result identity for the typed wake intent.""" + return { + "goal_id": self.goal_id, + "agent_id": self.agent_id, + "goal_ref": self._caller_goal_ref(), + "operation_id": row["identity"]["operation_id"], + "request_id": row["identity"]["request_id"], + "artifacts": [{k: v for k, v in item.items() if k != "text"} for item in row["artifacts"]], + } + + def wake_observed_in_turn(self, operation_id: str) -> dict | None: + """The requester read this accepted result inside its own Turn; no wake follows.""" + try: + return record_wake(self.path(require_operation_id(operation_id)), lambda wake: { + **wake, "state": "observed_in_turn", "observed_at": time.time(), + }) + except LockAcquireTimeoutError: + # The worker or another decision still holds the record; the pump + # re-reads the current state and the observation remains readable. + return None + def _cli(self, binding: dict, *args: str, timeout: int = 60) -> dict: completed = subprocess.run([*_python_module_command("loopx.cli"), "--registry", str(self.registry), @@ -1092,7 +1145,8 @@ def _execute(self, path: Path, row: dict, binding: dict) -> None: registry=self.registry, caller_goal_ref=self._caller_goal_ref(), ) - self._observe(path, row, "accepted", canonical_done=True, acceptance_ready=True, artifacts_current=True) + self._observe(path, row, "accepted", canonical_done=True, acceptance_ready=True, + artifacts_current=True, requester=self._wake_requester(row)) except (ValueError, KeyError, subprocess.TimeoutExpired, EffectRuntimeRemoteError) as exc: # Retain uncertain execution for explicit same-operation recovery. # No fresh Turn is ever created because its client timed out. From bcca92bd99f73f66f5ff1910e2f307fd1248002e Mon Sep 17 00:00:00 2001 From: song <22676124+songoow@users.noreply.github.com> Date: Tue, 29 Sep 2026 06:27:14 -0400 Subject: [PATCH 03/22] feat(chat): admit a host wake of the LoopX lead through the existing owner planChatMode gains a host-only "wake" operation that reuses the owner resume facts (Goal active, no running Turn, registered coordinator, valid bindings, resumable native Goal, allowance above consumed tokens) plus mode enabled and not paused, and returns a typed outcome instead of an error: admitted, pending (lead_turn_active, lead_paused, allowance_exhausted) or refused (goal_stopped, no_wake_owner, lead_unbound, binding_revoked, native_goal_complete, native_goal_absent). ChatLoopXMode.wake takes the session lock before the record lock, asks the rule, and submits one idempotent "/goal resume" Turn keyed by the intent id, so a crash between submit and receipt recovers through turn_for_client with created=false. apply rejects the host-only operation. The wake Turn tells the owner why it started and tells the lead which operation to read; it never unpauses the lead or starts a native Goal. Co-Authored-By: Claude Fable 5.1 Signed-off-by: song <22676124+songoow@users.noreply.github.com> --- loopx/chat_loopx_mode.py | 123 ++++++++++++++++++ loopx/chat_runtime.py | 13 +- .../control_plane/collaboration/chat_mode.ts | 33 ++++- 3 files changed, 160 insertions(+), 9 deletions(-) diff --git a/loopx/chat_loopx_mode.py b/loopx/chat_loopx_mode.py index 0a2ef42ca9..a924a987c6 100644 --- a/loopx/chat_loopx_mode.py +++ b/loopx/chat_loopx_mode.py @@ -71,6 +71,23 @@ "Pause/block when tools, authorization or evidence are insufficient." ) +WAKE_OPERATIONS = {"configure", "start", "resume", "pause", "exit", "message"} + + +def execution_guidance(turn: dict | None) -> str: + """GUIDANCE plus the exact reason a host-initiated Turn started.""" + request = (turn or {}).get("loopx_request") or {} + wake = request.get("wake") if request.get("operation") == "wake" else None + if not isinstance(wake, dict) or not wake.get("operation_id"): + return GUIDANCE + return ( + GUIDANCE + + "\nThis Turn started because delegated operation " + + str(wake["operation_id"]) + + " returned an accepted result. Read it with action=read before continuing;" + " member acceptance is not Goal completion." + ) + class ChatLoopXMode: def __init__(self, controller): @@ -288,6 +305,8 @@ def apply(self, session_id, body, *, work_dir, objective): "delivery_mode", }: raise ValueError("unknown LoopX mode input") + if body.get("operation") not in WAKE_OPERATIONS: + raise ValueError("unknown LoopX mode operation") with self._lock(session_id): if body.get("operation") == "message": return self.message(session_id, body) @@ -427,6 +446,110 @@ def apply(self, session_id, body, *, work_dir, objective): raise return {**self.snapshot(session_id), "turn_id": turn["turn_id"]} + def wake(self, session_id, path, *, work_dir, objective): + """Continue the lead once a delegated result is accepted; the owner does not poll. + + ``path`` is the accepted operation record. The session lock is taken + before the record lock, the same order the in-Turn tool uses, so an + in-Turn observation and a host wake never race. The receipt written + beside the result is a fact distinct from the result itself. + """ + from .collaboration_mcp import record_wake + + with self._lock(session_id): + return record_wake( + path, + lambda intent: self._wake_decision( + session_id, intent, work_dir=work_dir, objective=objective + ), + ) + + def _wake_decision(self, session_id, intent, *, work_dir, objective): + intent_id = intent.get("intent_id") + if not isinstance(intent_id, str) or len(intent_id) < 32: + raise ValueError("invalid wake intent") + client_turn_id = "wake-" + intent_id[:32] + now = time.time() + + def settled(state, reason): + if state == "pending" and intent.get("reason") == reason: + return None # unchanged: no churn on the record + return { + **intent, + "state": state, + "reason": reason, + **({"refused_at": now} if state == "refused" else {"checked_at": now}), + } + + try: + session = self._session(session_id) + except ValueError: + return settled("refused", "no_wake_owner") + # A crash after submit but before the receipt write recovers here: + # the exact client turn already exists, so it is recorded once. + existing = self.store.turn_for_client(session_id, client_turn_id) + if existing: + return self._woken(intent, session_id, existing["turn_id"], created=False, now=now) + mode = session.get("loopx_mode") or {} + settings = mode.get("settings") or {} + try: + goal = self._goal(session) + except ValueError: + goal = None + binding_valid = False + if goal is not None and settings.get("agent_id"): + try: + self._execution(session, settings) + binding_valid = True + except (ValueError, KeyError, OSError): + binding_valid = False + decision = effect_runtime_result( + "collaboration.chat_mode", + { + "session": session, + "origin": "host", + "operation": "wake", + "settings": settings, + "native": session.get("native_goal") or {}, + "registered_agents": registered_agent_ids_for_goal(goal) if goal else [], + "goal_active": goal is not None + and goal.get("status") not in {"stopped", "archived"}, + "execution_binding_valid": binding_valid, + }, + ) + if decision["state"] != "admitted": + return settled(decision["state"], decision["reason"]) + turn, created = self.controller.submit_turn( + session_id=session_id, + client_turn_id=client_turn_id, + message=f"/goal resume --tokens {settings['token_budget']}", + attachments=[], + work_dir=work_dir, + objective=objective, + loopx_execution=True, + loopx_request={ + "operation": "wake", + "settings": settings, + "wake": { + key: intent.get(key) + for key in ("intent_id", "operation_id", "request_id") + }, + }, + ) + return self._woken(intent, session_id, turn["turn_id"], created=created, now=now) + + @staticmethod + def _woken(intent, session_id, turn_id, *, created, now): + receipt = {key: value for key, value in intent.items() if key not in {"reason", "checked_at"}} + return { + **receipt, + "state": "woken", + "session_id": session_id, + "turn_id": turn_id, + "created": created, + "woken_at": now, + } + def recover(self, session_id, adapter): driver = CodexGoalDriver(adapter.session) driver.pause() diff --git a/loopx/chat_runtime.py b/loopx/chat_runtime.py index 7b32d1e5bd..ed36290f44 100644 --- a/loopx/chat_runtime.py +++ b/loopx/chat_runtime.py @@ -951,11 +951,10 @@ def submit_turn( message=message, attachments=attachments, display_message=( - ( - "开启 LoopX 模式,持续推进当前 Goal。" - if (loopx_request or {}).get("operation") == "start" - else "恢复 LoopX 模式。" - ) + { + "start": "开启 LoopX 模式,持续推进当前 Goal。", + "wake": "成员结果已验收,继续推进 LoopX 模式。", + }.get(str((loopx_request or {}).get("operation")), "恢复 LoopX 模式。") if loopx_execution else None ), @@ -1463,9 +1462,9 @@ def event_sink(kind: str, payload: dict[str, Any]) -> None: return execution_context = None if loopx_execution: - from .chat_loopx_mode import GUIDANCE + from .chat_loopx_mode import execution_guidance execution_lock = self.loopx_mode.prepare(session_id, turn_id, adapter, adapter.session.read_tool_handler, event_sink) - execution_context = GUIDANCE + "\nFresh scoped evidence:\n" + json.dumps(context, ensure_ascii=False) + execution_context = execution_guidance(self.store.load_turn(session_id, turn_id)) + "\nFresh scoped evidence:\n" + json.dumps(context, ensure_ascii=False) response = adapter.goal_driver.run(native_command, event_sink, execution_context=execution_context) finally: if execution_lock is not None: diff --git a/loopx/control_plane/collaboration/chat_mode.ts b/loopx/control_plane/collaboration/chat_mode.ts index 78c31564e0..1b5f0f8221 100644 --- a/loopx/control_plane/collaboration/chat_mode.ts +++ b/loopx/control_plane/collaboration/chat_mode.ts @@ -12,12 +12,14 @@ function requireThat(ok: unknown, message: string): asserts ok { export function planChatMode(input: JsonObject): JsonObject { const session = requireJsonObject(input.session, "conversation session"); const operation = input.operation; - requireThat(["configure", "start", "resume", "pause", "exit", "message"].includes(String(operation)), "unsupported conversation operation"); + requireThat(["configure", "start", "resume", "pause", "exit", "message", "wake"].includes(String(operation)), "unsupported conversation operation"); requireThat(resolveConversationScope(session).kind === "owner_goal" - && input.origin === "web" && session.session_mode !== "attached_host" + && (input.origin === "web" || (operation === "wake" && input.origin === "host")) + && session.session_mode !== "attached_host" && session.agent_id === "codex", "LoopX mode requires a local managed Codex Goal conversation"); const settings = requireJsonObject(input.settings, "conversation settings"); const native = requireJsonObject(input.native ?? {}, "native Goal observation"); + if (operation === "wake") return planDelegationWake(input, session, settings, native); if (operation === "message") { const mode = requireJsonObject(session.loopx_mode ?? {}, "mode"); const turn = requireJsonObject(input.turn ?? {}, "active execution turn"); @@ -48,3 +50,30 @@ export function planChatMode(input: JsonObject): JsonObject { } return {operation, enabled: operation !== "configure", settings}; } + +const RESUMABLE_NATIVE = ["paused", "blocked", "usageLimited", "budgetLimited"]; + +/** A delegated result was accepted; decide only whether the lead may continue now. + * + * The same facts that admit an owner resume admit a host wake, plus mode + * enabled and not paused. A refusal is terminal for that intent; pending + * keeps it for a later tick. A wake never unpauses the lead, never starts a + * native Goal and never raises the conversation allowance. */ +function planDelegationWake(input: JsonObject, session: JsonObject, settings: JsonObject, native: JsonObject): JsonObject { + const mode = requireJsonObject(session.loopx_mode ?? {}, "mode"); + const outcome = (state: "pending" | "refused", reason: string) => ({operation: "wake", state, reason}); + if (input.goal_active !== true) return outcome("refused", "goal_stopped"); + if (mode.enabled !== true) return outcome("refused", "no_wake_owner"); + if (!(typeof settings.agent_id === "string" && Array.isArray(input.registered_agents) + && input.registered_agents.includes(settings.agent_id))) return outcome("refused", "lead_unbound"); + if (input.execution_binding_valid !== true) return outcome("refused", "binding_revoked"); + const status = String(native.status ?? "absent"); + if (status === "complete") return outcome("refused", "native_goal_complete"); + if (status === "absent") return outcome("refused", "native_goal_absent"); + if (mode.paused === true) return outcome("pending", "lead_paused"); + if (session.active_turn_id || !RESUMABLE_NATIVE.includes(status)) return outcome("pending", "lead_turn_active"); + if (!(Number.isSafeInteger(settings.token_budget) && Number(settings.token_budget) > 0 + && Number(settings.token_budget) <= 2147483647 + && Number(settings.token_budget) > Number(native.tokensUsed ?? 0))) return outcome("pending", "allowance_exhausted"); + return {operation: "wake", state: "admitted", reason: null, settings}; +} From ddcfe9f7eeb9376d8244b94bfa2fad039a80f4a6 Mon Sep 17 00:00:00 2001 From: song <22676124+songoow@users.noreply.github.com> Date: Tue, 29 Sep 2026 06:29:19 -0400 Subject: [PATCH 04/22] feat(chat): pump pending delegation wakes beside the return service DelegationWakeService scans conversations that configured a coordinator, maps their observed operations (loopx_deliveries) to the requester-scoped records, and asks ChatLoopXMode.wake to settle each pending intent under the same record lock adopt_result uses. It prefers a conversation that is still enabled over one that exited, changes a pending receipt only when its reason changes, and skips records held by a worker until the next tick. Requesters outside a Chat LoopX conversation are never visited. Closing the Chat server stops the pump; intents left pending are the rollback state. Co-Authored-By: Claude Fable 5.1 Signed-off-by: song <22676124+songoow@users.noreply.github.com> --- loopx/chat_delegation_wake.py | 133 ++++++++++++++++++++++++++++++++++ loopx/chat_server.py | 17 +++++ 2 files changed, 150 insertions(+) create mode 100644 loopx/chat_delegation_wake.py diff --git a/loopx/chat_delegation_wake.py b/loopx/chat_delegation_wake.py new file mode 100644 index 0000000000..f719641c55 --- /dev/null +++ b/loopx/chat_delegation_wake.py @@ -0,0 +1,133 @@ +"""Continue a Chat LoopX lead once a delegated result is accepted. + +The accepted transition leaves a pending wake intent beside the result. This +pump maps the intent to the conversation that observed the operation (its +``loopx_deliveries``) and asks the existing Chat LoopX owner for one bounded +Turn. It never unpauses a lead, never starts a native Goal and never settles +canonical work. Refusals are terminal facts; pending intents wait for a later +tick. Stopping the pump leaves intents pending, so disabling it is a complete +rollback. Requesters outside a Chat LoopX conversation are never visited: their +intent stays pending and the scheduler deadline recheck remains their fallback. +""" + +from __future__ import annotations + +import logging +import threading +from pathlib import Path +from typing import Any, Callable + +from .collaboration_mcp import execution_row_path +from .control_plane.collaboration.inbox import _read +from .file_lock import LockAcquireTimeoutError + +WAKE_INTERVAL_SECONDS = 3.0 + + +def default_goal_context(controller, session: dict[str, Any]) -> dict[str, Any]: + """Project and objective for a host-initiated Turn when no richer reader exists.""" + from .agent_registry import load_goal_from_registry + + goal = load_goal_from_registry(controller.registry_path, session["goal_id"]) or {} + return { + "project": Path(str(goal.get("repo") or ".")).expanduser().resolve(), + "objective": str(goal.get("domain") or session["goal_id"]), + } + + +def _pending_owner(sessions: list[dict[str, Any]]) -> dict[str, Any]: + """Prefer a conversation that can still be continued over one that exited.""" + return next( + (row for row in sessions if (row.get("loopx_mode") or {}).get("enabled") is True), + sessions[0], + ) + + +def pump_delegation_wakes( + controller, + *, + goal_context: Callable[[dict[str, Any]], dict[str, Any]] | None = None, + cancelled: Callable[[], bool] = lambda: False, + limit: int = 20, +) -> list[dict[str, Any]]: + """One restart-safe tick; returns the receipts it changed.""" + store = controller.store + context_for = goal_context or (lambda session: default_goal_context(controller, session)) + owners: dict[Path, list[dict[str, Any]]] = {} + for path in sorted(store.sessions_root.glob("*/session.json")): + if cancelled(): + return [] + session = store.load_session(path.parent.name) + if not session: + continue + settings = (session.get("loopx_mode") or {}).get("settings") or {} + agent_id = settings.get("agent_id") + if not isinstance(agent_id, str) or not agent_id: + continue + for delivery in session.get("loopx_deliveries") or []: + operation_id = delivery.get("operation_id") + if not isinstance(operation_id, str) or not operation_id: + continue + record = execution_row_path( + store.root.parent, str(session["goal_id"]), agent_id, operation_id + ) + owners.setdefault(record, []).append(session) + changed: list[dict[str, Any]] = [] + for record, sessions in owners.items(): + if cancelled() or len(changed) >= limit: + break + try: + wake = _read(record).get("wake") if record.exists() else None + except (OSError, ValueError): + continue + if not isinstance(wake, dict) or wake.get("state") != "pending": + continue # unlocked pre-check; record_wake re-reads under the lock + session = _pending_owner(sessions) + context = context_for(session) + try: + receipt = controller.loopx_mode.wake( + session["session_id"], + record, + work_dir=context["project"], + objective=context["objective"], + ) + except LockAcquireTimeoutError: + continue # a worker or decision holds the record; retry next tick + if receipt is not None: + changed.append({"operation_id": wake.get("operation_id"), **receipt}) + return changed + + +class DelegationWakeService: + """Cheap local pump hosted by the existing Chat server, beside the return service.""" + + def __init__(self, controller, *, goal_context=None, interval=WAKE_INTERVAL_SECONDS): + self.controller = controller + self.goal_context = goal_context + self.interval = interval + self.stop = threading.Event() + self.thread = threading.Thread( + target=self.run, daemon=True, name="loopx-delegation-wakes" + ) + + def start(self): + self.thread.start() + return self + + def run(self): + while not self.stop.is_set(): + try: + pump_delegation_wakes( + self.controller, + goal_context=self.goal_context, + cancelled=self.stop.is_set, + ) + except (OSError, ValueError, KeyError, TypeError, RuntimeError): + logging.getLogger(__name__).warning( + "Delegation wake receipts unavailable; retrying" + ) + self.stop.wait(self.interval) + + def close(self): + self.stop.set() + self.thread.join(timeout=3) diff --git a/loopx/chat_server.py b/loopx/chat_server.py index d2818075fa..68974a6e8c 100644 --- a/loopx/chat_server.py +++ b/loopx/chat_server.py @@ -422,6 +422,8 @@ def server_close(self) -> None: self.lark_app_setup_manager.close() if hasattr(self, "manager_return_service"): self.manager_return_service.close() + if hasattr(self, "delegation_wake_service"): + self.delegation_wake_service.close() if hasattr(self, "lark_goal_topic_runtime"): self.lark_goal_topic_runtime.close() if hasattr(self, "runtime_controller"): @@ -1568,6 +1570,21 @@ def serve_chat( server.lark_goal_topic_runtime.start() from .extensions.lark.manager_returns import start_return_service server.manager_return_service = start_return_service(server, runtime_root) + from .chat_delegation_wake import DelegationWakeService + + def _wake_goal_context(session): + registry = load_registry(server.registry_path) + goal = next( + (item for item in registry_goals(registry) if str(item.get("id") or "") == str(session["goal_id"])), + None, + ) + if goal is None: + raise ValueError("goal_id was not found in the active LoopX registry") + return _goal_public_context(registry, goal) + + server.delegation_wake_service = DelegationWakeService( + server.runtime_controller, goal_context=_wake_goal_context + ).start() url = f"http://{host}:{port}{DEFAULT_CHAT_PATH}" print(f"Serving LoopX Chat at {url}", flush=True) print("Agent boundary: local adapters, read-only sandbox, approval policy never", flush=True) From 5748a37d578fe7d468cb6e20ffa1091ee6d965c9 Mon Sep 17 00:00:00 2001 From: song <22676124+songoow@users.noreply.github.com> Date: Tue, 29 Sep 2026 07:56:38 -0400 Subject: [PATCH 05/22] fix(collaboration): keep a wake pending while the lead Turn is activating The wake rule checked the native status before the running-Turn fence, so an owner start that was still activating (Turn active, native "absent" or the previous run's "complete") refused the wake terminally. The fence now comes first; native complete/absent are refused only when no Turn runs. The rule test covers every admitted, pending and refused outcome, and that the host origin admits only wake. The intent uses schema_version like every other collaboration contract and requires the requester goal reference to be an object or null. Co-Authored-By: Claude Opus 5.5 (1M context) Signed-off-by: song <22676124+songoow@users.noreply.github.com> --- .../control_plane/collaboration/chat_mode.ts | 9 ++-- .../control_plane/collaboration/delegation.ts | 6 +-- tests/control_plane_ts/chat_mode.test.ts | 41 +++++++++++++++++++ tests/control_plane_ts/delegation.test.ts | 4 +- 4 files changed, 53 insertions(+), 7 deletions(-) diff --git a/loopx/control_plane/collaboration/chat_mode.ts b/loopx/control_plane/collaboration/chat_mode.ts index 1b5f0f8221..83dfb632f0 100644 --- a/loopx/control_plane/collaboration/chat_mode.ts +++ b/loopx/control_plane/collaboration/chat_mode.ts @@ -57,8 +57,10 @@ const RESUMABLE_NATIVE = ["paused", "blocked", "usageLimited", "budgetLimited"]; * * The same facts that admit an owner resume admit a host wake, plus mode * enabled and not paused. A refusal is terminal for that intent; pending - * keeps it for a later tick. A wake never unpauses the lead, never starts a - * native Goal and never raises the conversation allowance. */ + * keeps it for a later tick. A running Turn is checked before the native + * status: while a start is still activating, "absent" or a previous run's + * "complete" is not yet a stable fact. A wake never unpauses the lead, never + * starts a native Goal and never raises the conversation allowance. */ function planDelegationWake(input: JsonObject, session: JsonObject, settings: JsonObject, native: JsonObject): JsonObject { const mode = requireJsonObject(session.loopx_mode ?? {}, "mode"); const outcome = (state: "pending" | "refused", reason: string) => ({operation: "wake", state, reason}); @@ -67,11 +69,12 @@ function planDelegationWake(input: JsonObject, session: JsonObject, settings: Js if (!(typeof settings.agent_id === "string" && Array.isArray(input.registered_agents) && input.registered_agents.includes(settings.agent_id))) return outcome("refused", "lead_unbound"); if (input.execution_binding_valid !== true) return outcome("refused", "binding_revoked"); + if (session.active_turn_id) return outcome("pending", "lead_turn_active"); const status = String(native.status ?? "absent"); if (status === "complete") return outcome("refused", "native_goal_complete"); if (status === "absent") return outcome("refused", "native_goal_absent"); if (mode.paused === true) return outcome("pending", "lead_paused"); - if (session.active_turn_id || !RESUMABLE_NATIVE.includes(status)) return outcome("pending", "lead_turn_active"); + if (!RESUMABLE_NATIVE.includes(status)) return outcome("pending", "lead_turn_active"); if (!(Number.isSafeInteger(settings.token_budget) && Number(settings.token_budget) > 0 && Number(settings.token_budget) <= 2147483647 && Number(settings.token_budget) > Number(native.tokensUsed ?? 0))) return outcome("pending", "allowance_exhausted"); diff --git a/loopx/control_plane/collaboration/delegation.ts b/loopx/control_plane/collaboration/delegation.ts index a9d3becb56..4afd6879eb 100644 --- a/loopx/control_plane/collaboration/delegation.ts +++ b/loopx/control_plane/collaboration/delegation.ts @@ -349,7 +349,7 @@ function delegationWakeIntent(params: JsonObject): JsonObject { const requester = requireJsonObject(params.requester, "wake requester"); requireThat([requester.goal_id, requester.agent_id, requester.operation_id, requester.request_id].every(text), "wake intent requires the requester and result identity"); - requireThat(requester.goal_ref === null || typeof requester.goal_ref === "object", "invalid requester goal reference"); + const goalRef = requester.goal_ref == null ? null : requireJsonObject(requester.goal_ref, "requester goal reference"); requireThat(Array.isArray(requester.artifacts) && requester.artifacts.length > 0, "wake intent requires accepted artifacts"); const digests = requester.artifacts.map(value => { const artifact = requireJsonObject(value, "accepted artifact"); @@ -358,10 +358,10 @@ function delegationWakeIntent(params: JsonObject): JsonObject { return {ref: artifact.ref, sha256: artifact.sha256}; }); return { - schema: "loopx_delegation_wake_intent_v0", + schema_version: "loopx_delegation_wake_intent_v0", intent_id: canonicalAuthoritySha256([requester.goal_id, requester.agent_id, requester.operation_id, requester.request_id, digests]), - requester: {goal_id: requester.goal_id, agent_id: requester.agent_id, goal_ref: requester.goal_ref ?? null}, + requester: {goal_id: requester.goal_id, agent_id: requester.agent_id, goal_ref: goalRef}, operation_id: requester.operation_id, request_id: requester.request_id, }; } diff --git a/tests/control_plane_ts/chat_mode.test.ts b/tests/control_plane_ts/chat_mode.test.ts index 18c255f805..179d050b4a 100644 --- a/tests/control_plane_ts/chat_mode.test.ts +++ b/tests/control_plane_ts/chat_mode.test.ts @@ -45,3 +45,44 @@ test("all three delivery modes require the exact active execution turn", () => { assert.throws(() => planChatMode({...message, delivery_mode: "queue", turn: {turn_id: "other", loopx_execution: true}})); assert.throws(() => planChatMode({...message, delivery_mode: "queue", session: {...active, loopx_mode: {enabled: true, paused: true}}})); }); + +test("a host wake reuses the resume facts and returns a typed outcome, never an owner change", () => { + const enabled = {...session, loopx_mode: {enabled: true, paused: false}}; + const wake: JsonObject = {...input, operation: "wake", origin: "host", session: enabled, + native: {status: "paused", tokensUsed: 400}}; + const admitted = planChatMode(wake); + assert.equal(admitted.state, "admitted"); + assert.equal(admitted.reason, null); + assert.deepEqual(admitted.settings, input.settings); + // Terminal refusals: the intent is settled and nothing waits. + const refused: [JsonObject, string][] = [ + [{goal_active: false}, "goal_stopped"], + [{session: {...enabled, loopx_mode: {enabled: false}}}, "no_wake_owner"], + [{registered_agents: ["other"]}, "lead_unbound"], + [{execution_binding_valid: false}, "binding_revoked"], + [{native: {status: "complete", tokensUsed: 400}}, "native_goal_complete"], + [{native: {status: "absent"}}, "native_goal_absent"], + ]; + for (const [changes, reason] of refused) { + assert.deepEqual(planChatMode({...wake, ...changes}), {operation: "wake", state: "refused", reason}, reason); + } + // Pending: the intent waits for a later tick; a wake never unpauses the lead. + const pending: [JsonObject, string][] = [ + [{session: {...enabled, loopx_mode: {enabled: true, paused: true}}}, "lead_paused"], + [{session: {...enabled, active_turn_id: "running"}}, "lead_turn_active"], + [{native: {status: "active", tokensUsed: 400}}, "lead_turn_active"], + [{native: {status: "paused", tokensUsed: 1000}}, "allowance_exhausted"], + [{settings: {agent_id: "lead", token_budget: 0}}, "allowance_exhausted"], + // An owner start that is still activating: native facts are not yet stable. + [{session: {...enabled, active_turn_id: "start"}, native: {status: "absent"}}, "lead_turn_active"], + [{session: {...enabled, active_turn_id: "start"}, native: {status: "complete", tokensUsed: 400}}, "lead_turn_active"], + [{session: {...enabled, active_turn_id: "pausing", loopx_mode: {enabled: true, paused: true}}}, "lead_turn_active"], + ]; + for (const [changes, reason] of pending) { + assert.deepEqual(planChatMode({...wake, ...changes}), {operation: "wake", state: "pending", reason}, reason); + } + // The host origin is only for wake; owner operations keep the web origin. + assert.throws(() => planChatMode({...input, origin: "host"}), /local managed/); + assert.throws(() => planChatMode({...wake, origin: "external"}), /local managed/); + assert.throws(() => planChatMode({...wake, session: {...enabled, channel_id: "manager"}})); +}); diff --git a/tests/control_plane_ts/delegation.test.ts b/tests/control_plane_ts/delegation.test.ts index 86b49acd2c..4861506b9b 100644 --- a/tests/control_plane_ts/delegation.test.ts +++ b/tests/control_plane_ts/delegation.test.ts @@ -109,7 +109,7 @@ test("message receipt and model return do not imply accepted work", () => { test("only the first transition to accepted leaves a wake intent for the exact requester and result", () => { const accepted = transitionDelegationObservation({from: "turn_returned", to: "accepted", ...acceptedFacts}); const intent = accepted.wake_intent as Record; - assert.equal(intent.schema, "loopx_delegation_wake_intent_v0"); + assert.equal(intent.schema_version, "loopx_delegation_wake_intent_v0"); assert.match(String(intent.intent_id), /^[a-f0-9]{64}$/); assert.deepEqual(intent.requester, {goal_id: "research", agent_id: "coordinator", goal_ref: null}); assert.equal(intent.operation_id, "analysis-1"); @@ -128,6 +128,8 @@ test("only the first transition to accepted leaves a wake intent for the exact r requester: {...acceptedRequester, artifacts: [{ref: "output.json", sha256: "short"}]}}), /artifact/); assert.throws(() => transitionDelegationObservation({from: "turn_returned", to: "accepted", ...acceptedFacts, requester: {...acceptedRequester, artifacts: []}}), /artifacts/); + assert.throws(() => transitionDelegationObservation({from: "turn_returned", to: "accepted", ...acceptedFacts, + requester: {...acceptedRequester, goal_ref: ["not", "a", "reference"]}}), /goal reference/); // Rejection is terminal and wakes nobody. assert.deepEqual(transitionDelegationObservation({from: "turn_returned", to: "rejected"}), {status: "rejected"}); }); From 6776157ea61ac51d6e0cfcefb555d026d3256443 Mon Sep 17 00:00:00 2001 From: song <22676124+songoow@users.noreply.github.com> Date: Tue, 29 Sep 2026 08:13:56 -0400 Subject: [PATCH 06/22] fix(chat): resolve wake owners by requester and refuse unowned intents The pump visited only operations listed in a conversation's loopx_deliveries, so requesters outside Chat LoopX mode stayed pending forever and a lead with more than twenty operations lost wakes to the bounded list. It now scans this runtime's operation records and resolves the owner from the intent's requester identity, preferring a conversation that can still continue and then the one that observed the operation; a requester with no such conversation is refused with no_wake_owner. An intent must name the requester that owns its storage address, and one failing record no longer aborts the tick. Receipts share one constructor that keeps only the typed intent and the current state's facts. An existing wake-* Turn is recorded only when it carries this intent; otherwise the wake is refused as wake_identity_conflict rather than claiming another Turn. The Goal context is read only after admission, so a removed Goal refuses instead of raising. Adopting a result also retires the consumer's wake. Co-Authored-By: Claude Opus 5.5 (1M context) Signed-off-by: song <22676124+songoow@users.noreply.github.com> --- loopx/chat_delegation_wake.py | 137 ++++++++++++++++++++-------------- loopx/chat_loopx_mode.py | 61 ++++++++------- loopx/collaboration_mcp.py | 16 +++- 3 files changed, 130 insertions(+), 84 deletions(-) diff --git a/loopx/chat_delegation_wake.py b/loopx/chat_delegation_wake.py index f719641c55..7deff880c6 100644 --- a/loopx/chat_delegation_wake.py +++ b/loopx/chat_delegation_wake.py @@ -1,46 +1,84 @@ """Continue a Chat LoopX lead once a delegated result is accepted. -The accepted transition leaves a pending wake intent beside the result. This -pump maps the intent to the conversation that observed the operation (its -``loopx_deliveries``) and asks the existing Chat LoopX owner for one bounded -Turn. It never unpauses a lead, never starts a native Goal and never settles -canonical work. Refusals are terminal facts; pending intents wait for a later -tick. Stopping the pump leaves intents pending, so disabling it is a complete -rollback. Requesters outside a Chat LoopX conversation are never visited: their -intent stays pending and the scheduler deadline recheck remains their fallback. +The accepted transition leaves a pending wake intent beside the result. Each +tick scans this runtime's operation records for pending intents and resolves +the owner from the intent's requester: a Chat LoopX conversation configured +with that coordinator identity, preferring one that can still continue and, +among those, the one that observed the operation. The existing Chat LoopX +owner decides and records the receipt. A requester without such a +conversation is refused with ``no_wake_owner``; the scheduler deadline recheck +remains its continuation. The pump never unpauses a lead, never starts a +native Goal and never settles canonical work. Stopping it leaves intents +pending, which is its rollback. """ from __future__ import annotations import logging import threading +import time from pathlib import Path from typing import Any, Callable -from .collaboration_mcp import execution_row_path -from .control_plane.collaboration.inbox import _read +from .chat_runtime import ChatTurnAcceptanceUnavailableError +from .collaboration_mcp import execution_row_path, record_wake, wake_receipt +from .control_plane.collaboration.inbox import _read, _root from .file_lock import LockAcquireTimeoutError WAKE_INTERVAL_SECONDS = 3.0 +_LOG = logging.getLogger(__name__) def default_goal_context(controller, session: dict[str, Any]) -> dict[str, Any]: """Project and objective for a host-initiated Turn when no richer reader exists.""" from .agent_registry import load_goal_from_registry - goal = load_goal_from_registry(controller.registry_path, session["goal_id"]) or {} + goal = load_goal_from_registry(controller.registry_path, session["goal_id"]) + if not goal: + raise ValueError("Goal unavailable") return { "project": Path(str(goal.get("repo") or ".")).expanduser().resolve(), "objective": str(goal.get("domain") or session["goal_id"]), } -def _pending_owner(sessions: list[dict[str, Any]]) -> dict[str, Any]: - """Prefer a conversation that can still be continued over one that exited.""" - return next( - (row for row in sessions if (row.get("loopx_mode") or {}).get("enabled") is True), - sessions[0], - ) +def _owners(store) -> dict[tuple[str, str], list[dict[str, Any]]]: + """Chat conversations by their configured coordinator (requester) identity.""" + owners: dict[tuple[str, str], list[dict[str, Any]]] = {} + for path in sorted(store.sessions_root.glob("*/session.json")): + session = store.load_session(path.parent.name) + settings = ((session or {}).get("loopx_mode") or {}).get("settings") or {} + agent_id = settings.get("agent_id") + if session and isinstance(agent_id, str) and agent_id: + owners.setdefault((str(session.get("goal_id") or ""), agent_id), []).append(session) + return owners + + +def _select_owner(sessions: list[dict[str, Any]], operation_id: str) -> dict[str, Any]: + def rank(session): + usable = session.get("status") != "closed" and ( + (session.get("loopx_mode") or {}).get("enabled") is True + ) + observed = any( + row.get("operation_id") == operation_id + for row in session.get("loopx_deliveries") or [] + ) + return (usable, observed, str(session.get("updated_at") or "")) + + return max(sessions, key=rank) + + +def _pending_identity(record: Path) -> tuple[str, str, str] | None: + """Requester and operation of a pending intent; an unlocked pre-check only.""" + row = _read(record) + wake = row.get("wake") + if row.get("status") != "accepted" or not isinstance(wake, dict) or wake.get("state") != "pending": + return None + requester = wake.get("requester") or {} + identity = (requester.get("goal_id"), requester.get("agent_id"), wake.get("operation_id")) + if not all(isinstance(value, str) and value for value in identity): + return None + return identity # type: ignore[return-value] def pump_delegation_wakes( @@ -52,49 +90,40 @@ def pump_delegation_wakes( ) -> list[dict[str, Any]]: """One restart-safe tick; returns the receipts it changed.""" store = controller.store + root = store.root.parent context_for = goal_context or (lambda session: default_goal_context(controller, session)) - owners: dict[Path, list[dict[str, Any]]] = {} - for path in sorted(store.sessions_root.glob("*/session.json")): - if cancelled(): - return [] - session = store.load_session(path.parent.name) - if not session: - continue - settings = (session.get("loopx_mode") or {}).get("settings") or {} - agent_id = settings.get("agent_id") - if not isinstance(agent_id, str) or not agent_id: - continue - for delivery in session.get("loopx_deliveries") or []: - operation_id = delivery.get("operation_id") - if not isinstance(operation_id, str) or not operation_id: - continue - record = execution_row_path( - store.root.parent, str(session["goal_id"]), agent_id, operation_id - ) - owners.setdefault(record, []).append(session) + owners = _owners(store) changed: list[dict[str, Any]] = [] - for record, sessions in owners.items(): + for record in sorted((_root(root) / "executions").glob("*/*.json")): if cancelled() or len(changed) >= limit: break try: - wake = _read(record).get("wake") if record.exists() else None - except (OSError, ValueError): - continue - if not isinstance(wake, dict) or wake.get("state") != "pending": - continue # unlocked pre-check; record_wake re-reads under the lock - session = _pending_owner(sessions) - context = context_for(session) - try: - receipt = controller.loopx_mode.wake( - session["session_id"], - record, - work_dir=context["project"], - objective=context["objective"], - ) + identity = _pending_identity(record) + if identity is None: + continue + goal_id, agent_id, operation_id = identity + if execution_row_path(root, goal_id, agent_id, operation_id) != record: + continue # an intent must name the requester that owns its storage address + sessions = owners.get((goal_id, agent_id)) + if not sessions: + receipt = record_wake(record, lambda wake: wake_receipt( + wake, "refused", reason="no_wake_owner", refused_at=time.time())) + else: + session = _select_owner(sessions, operation_id) + receipt = controller.loopx_mode.wake( + session["session_id"], + record, + goal_context=lambda session=session: context_for(session), + ) except LockAcquireTimeoutError: continue # a worker or decision holds the record; retry next tick + except (OSError, ValueError, KeyError, TypeError, RuntimeError, + ChatTurnAcceptanceUnavailableError) as exc: + # Isolate one record; the others still progress this tick. + _LOG.warning("Delegation wake unavailable for one operation (%s); retrying", type(exc).__name__) + continue if receipt is not None: - changed.append({"operation_id": wake.get("operation_id"), **receipt}) + changed.append(receipt) return changed @@ -123,9 +152,7 @@ def run(self): cancelled=self.stop.is_set, ) except (OSError, ValueError, KeyError, TypeError, RuntimeError): - logging.getLogger(__name__).warning( - "Delegation wake receipts unavailable; retrying" - ) + _LOG.warning("Delegation wake receipts unavailable; retrying") self.stop.wait(self.interval) def close(self): diff --git a/loopx/chat_loopx_mode.py b/loopx/chat_loopx_mode.py index a924a987c6..c9d22c59f6 100644 --- a/loopx/chat_loopx_mode.py +++ b/loopx/chat_loopx_mode.py @@ -71,7 +71,8 @@ "Pause/block when tools, authorization or evidence are insufficient." ) -WAKE_OPERATIONS = {"configure", "start", "resume", "pause", "exit", "message"} +# Owner operations; a host wake is admitted only through ``wake``. +OWNER_OPERATIONS = {"configure", "start", "resume", "pause", "exit", "message"} def execution_guidance(turn: dict | None) -> str: @@ -305,7 +306,7 @@ def apply(self, session_id, body, *, work_dir, objective): "delivery_mode", }: raise ValueError("unknown LoopX mode input") - if body.get("operation") not in WAKE_OPERATIONS: + if body.get("operation") not in OWNER_OPERATIONS: raise ValueError("unknown LoopX mode operation") with self._lock(session_id): if body.get("operation") == "message": @@ -446,13 +447,15 @@ def apply(self, session_id, body, *, work_dir, objective): raise return {**self.snapshot(session_id), "turn_id": turn["turn_id"]} - def wake(self, session_id, path, *, work_dir, objective): + def wake(self, session_id, path, *, goal_context): """Continue the lead once a delegated result is accepted; the owner does not poll. - ``path`` is the accepted operation record. The session lock is taken - before the record lock, the same order the in-Turn tool uses, so an - in-Turn observation and a host wake never race. The receipt written - beside the result is a fact distinct from the result itself. + ``path`` is the accepted operation record; ``goal_context`` returns the + Turn's project and objective and is read only after admission. The + session lock is taken before the record lock, the same order the + in-Turn tool uses, so an in-Turn observation and a host wake never + race. The receipt written beside the result is a fact distinct from + the result itself. """ from .collaboration_mcp import record_wake @@ -460,11 +463,13 @@ def wake(self, session_id, path, *, work_dir, objective): return record_wake( path, lambda intent: self._wake_decision( - session_id, intent, work_dir=work_dir, objective=objective + session_id, intent, goal_context=goal_context ), ) - def _wake_decision(self, session_id, intent, *, work_dir, objective): + def _wake_decision(self, session_id, intent, *, goal_context): + from .collaboration_mcp import wake_receipt + intent_id = intent.get("intent_id") if not isinstance(intent_id, str) or len(intent_id) < 32: raise ValueError("invalid wake intent") @@ -474,12 +479,8 @@ def _wake_decision(self, session_id, intent, *, work_dir, objective): def settled(state, reason): if state == "pending" and intent.get("reason") == reason: return None # unchanged: no churn on the record - return { - **intent, - "state": state, - "reason": reason, - **({"refused_at": now} if state == "refused" else {"checked_at": now}), - } + at = "refused_at" if state == "refused" else "checked_at" + return wake_receipt(intent, state, reason=reason, **{at: now}) try: session = self._session(session_id) @@ -489,6 +490,14 @@ def settled(state, reason): # the exact client turn already exists, so it is recorded once. existing = self.store.turn_for_client(session_id, client_turn_id) if existing: + request = existing.get("loopx_request") or {} + if ( + not existing.get("loopx_execution") + or request.get("operation") != "wake" + or (request.get("wake") or {}).get("intent_id") != intent_id + ): + # Another Turn owns this client id; claiming it would be a false receipt. + return settled("refused", "wake_identity_conflict") return self._woken(intent, session_id, existing["turn_id"], created=False, now=now) mode = session.get("loopx_mode") or {} settings = mode.get("settings") or {} @@ -519,13 +528,14 @@ def settled(state, reason): ) if decision["state"] != "admitted": return settled(decision["state"], decision["reason"]) + context = goal_context() turn, created = self.controller.submit_turn( session_id=session_id, client_turn_id=client_turn_id, message=f"/goal resume --tokens {settings['token_budget']}", attachments=[], - work_dir=work_dir, - objective=objective, + work_dir=context["project"], + objective=str(context.get("objective") or context.get("title") or session["goal_id"]), loopx_execution=True, loopx_request={ "operation": "wake", @@ -540,15 +550,12 @@ def settled(state, reason): @staticmethod def _woken(intent, session_id, turn_id, *, created, now): - receipt = {key: value for key, value in intent.items() if key not in {"reason", "checked_at"}} - return { - **receipt, - "state": "woken", - "session_id": session_id, - "turn_id": turn_id, - "created": created, - "woken_at": now, - } + from .collaboration_mcp import wake_receipt + + return wake_receipt( + intent, "woken", session_id=session_id, turn_id=turn_id, + created=created, woken_at=now, + ) def recover(self, session_id, adapter): driver = CodexGoalDriver(adapter.session) @@ -785,6 +792,8 @@ def dispatch(name, arguments): if set(arguments) != {"action", "operation_id", "consumer_operation_id"}: raise ValueError("adopt requires source and consumer operation ids only") result = service.adopt_result(operation_id, arguments["consumer_operation_id"]) + # Adoption requires both results accepted: neither needs a wake now. + service.wake_observed_in_turn(arguments["consumer_operation_id"]) elif action == "start": result = service.start( arguments.get("binding_id", ""), diff --git a/loopx/collaboration_mcp.py b/loopx/collaboration_mcp.py index bd320ec82d..415d93bf7d 100644 --- a/loopx/collaboration_mcp.py +++ b/loopx/collaboration_mcp.py @@ -305,6 +305,14 @@ def execution_row_path(root: Path, goal_id: str, agent_id: str, operation_id: st return _root(root) / "executions" / _hash([goal_id, agent_id]) / (_hash(operation_id) + ".json") +_WAKE_INTENT_KEYS = ("schema_version", "intent_id", "requester", "operation_id", "request_id") + + +def wake_receipt(intent: dict, state: str, **facts) -> dict: + """One receipt shape: the typed intent plus only the current state's facts.""" + return {**{key: intent[key] for key in _WAKE_INTENT_KEYS if key in intent}, "state": state, **facts} + + def record_wake(path: Path, decide) -> dict | None: """Settle a pending wake receipt under the same lock adopt_result uses. @@ -700,10 +708,12 @@ def _wake_requester(self, row: dict) -> dict: def wake_observed_in_turn(self, operation_id: str) -> dict | None: """The requester read this accepted result inside its own Turn; no wake follows.""" + path = self.path(require_operation_id(operation_id)) + if not path.exists(): + return None try: - return record_wake(self.path(require_operation_id(operation_id)), lambda wake: { - **wake, "state": "observed_in_turn", "observed_at": time.time(), - }) + return record_wake(path, lambda wake: wake_receipt( + wake, "observed_in_turn", observed_at=time.time())) except LockAcquireTimeoutError: # The worker or another decision still holds the record; the pump # re-reads the current state and the observation remains readable. From 1806dc93b2b89100ad0cd87717b2a9989324557d Mon Sep 17 00:00:00 2001 From: song <22676124+songoow@users.noreply.github.com> Date: Tue, 29 Sep 2026 09:00:51 -0400 Subject: [PATCH 07/22] fix(collaboration): admit a delegation wake only from the host origin The typed rule accepted origin "web" for a wake, so the host-only guarantee rested on the Python apply guard alone. Each origin now maps to exactly one operation family: the host may only wake, and the owner's web origin may never request one. Co-Authored-By: Claude Opus 5.5 (1M context) Signed-off-by: song <22676124+songoow@users.noreply.github.com> --- loopx/control_plane/collaboration/chat_mode.ts | 2 +- tests/control_plane_ts/chat_mode.test.ts | 3 ++- 2 files changed, 3 insertions(+), 2 deletions(-) diff --git a/loopx/control_plane/collaboration/chat_mode.ts b/loopx/control_plane/collaboration/chat_mode.ts index 83dfb632f0..1349c25008 100644 --- a/loopx/control_plane/collaboration/chat_mode.ts +++ b/loopx/control_plane/collaboration/chat_mode.ts @@ -14,7 +14,7 @@ export function planChatMode(input: JsonObject): JsonObject { const operation = input.operation; requireThat(["configure", "start", "resume", "pause", "exit", "message", "wake"].includes(String(operation)), "unsupported conversation operation"); requireThat(resolveConversationScope(session).kind === "owner_goal" - && (input.origin === "web" || (operation === "wake" && input.origin === "host")) + && input.origin === (operation === "wake" ? "host" : "web") && session.session_mode !== "attached_host" && session.agent_id === "codex", "LoopX mode requires a local managed Codex Goal conversation"); const settings = requireJsonObject(input.settings, "conversation settings"); diff --git a/tests/control_plane_ts/chat_mode.test.ts b/tests/control_plane_ts/chat_mode.test.ts index 179d050b4a..8529edd74b 100644 --- a/tests/control_plane_ts/chat_mode.test.ts +++ b/tests/control_plane_ts/chat_mode.test.ts @@ -81,8 +81,9 @@ test("a host wake reuses the resume facts and returns a typed outcome, never an for (const [changes, reason] of pending) { assert.deepEqual(planChatMode({...wake, ...changes}), {operation: "wake", state: "pending", reason}, reason); } - // The host origin is only for wake; owner operations keep the web origin. + // The host origin is only for wake, and wake only for the host: an owner cannot request one. assert.throws(() => planChatMode({...input, origin: "host"}), /local managed/); + assert.throws(() => planChatMode({...wake, origin: "web"}), /local managed/); assert.throws(() => planChatMode({...wake, origin: "external"}), /local managed/); assert.throws(() => planChatMode({...wake, session: {...enabled, channel_id: "manager"}})); }); From e7607697cb7597cf8290401251e51ce0943058c9 Mon Sep 17 00:00:00 2001 From: song <22676124+songoow@users.noreply.github.com> Date: Tue, 29 Sep 2026 09:57:18 -0400 Subject: [PATCH 08/22] refactor(chat): host the delegation wake pump in the Chat LoopX owner The pump was a new top-level module, which raised loopx/ from 147 to 148 top-level modules and failed the module budget. It only drives LoopXMode.wake, so it now lives beside that owner in chat_loopx_mode. The chat_runtime and collaboration_mcp dependencies are imported inside the pump because chat_runtime imports this module. No behavior change. Co-Authored-By: Claude Opus 5.5 (1M context) Signed-off-by: song <22676124+songoow@users.noreply.github.com> --- loopx/chat_delegation_wake.py | 160 ---------------------------------- loopx/chat_loopx_mode.py | 157 ++++++++++++++++++++++++++++++++- loopx/chat_server.py | 2 +- 3 files changed, 157 insertions(+), 162 deletions(-) delete mode 100644 loopx/chat_delegation_wake.py diff --git a/loopx/chat_delegation_wake.py b/loopx/chat_delegation_wake.py deleted file mode 100644 index 7deff880c6..0000000000 --- a/loopx/chat_delegation_wake.py +++ /dev/null @@ -1,160 +0,0 @@ -"""Continue a Chat LoopX lead once a delegated result is accepted. - -The accepted transition leaves a pending wake intent beside the result. Each -tick scans this runtime's operation records for pending intents and resolves -the owner from the intent's requester: a Chat LoopX conversation configured -with that coordinator identity, preferring one that can still continue and, -among those, the one that observed the operation. The existing Chat LoopX -owner decides and records the receipt. A requester without such a -conversation is refused with ``no_wake_owner``; the scheduler deadline recheck -remains its continuation. The pump never unpauses a lead, never starts a -native Goal and never settles canonical work. Stopping it leaves intents -pending, which is its rollback. -""" - -from __future__ import annotations - -import logging -import threading -import time -from pathlib import Path -from typing import Any, Callable - -from .chat_runtime import ChatTurnAcceptanceUnavailableError -from .collaboration_mcp import execution_row_path, record_wake, wake_receipt -from .control_plane.collaboration.inbox import _read, _root -from .file_lock import LockAcquireTimeoutError - -WAKE_INTERVAL_SECONDS = 3.0 -_LOG = logging.getLogger(__name__) - - -def default_goal_context(controller, session: dict[str, Any]) -> dict[str, Any]: - """Project and objective for a host-initiated Turn when no richer reader exists.""" - from .agent_registry import load_goal_from_registry - - goal = load_goal_from_registry(controller.registry_path, session["goal_id"]) - if not goal: - raise ValueError("Goal unavailable") - return { - "project": Path(str(goal.get("repo") or ".")).expanduser().resolve(), - "objective": str(goal.get("domain") or session["goal_id"]), - } - - -def _owners(store) -> dict[tuple[str, str], list[dict[str, Any]]]: - """Chat conversations by their configured coordinator (requester) identity.""" - owners: dict[tuple[str, str], list[dict[str, Any]]] = {} - for path in sorted(store.sessions_root.glob("*/session.json")): - session = store.load_session(path.parent.name) - settings = ((session or {}).get("loopx_mode") or {}).get("settings") or {} - agent_id = settings.get("agent_id") - if session and isinstance(agent_id, str) and agent_id: - owners.setdefault((str(session.get("goal_id") or ""), agent_id), []).append(session) - return owners - - -def _select_owner(sessions: list[dict[str, Any]], operation_id: str) -> dict[str, Any]: - def rank(session): - usable = session.get("status") != "closed" and ( - (session.get("loopx_mode") or {}).get("enabled") is True - ) - observed = any( - row.get("operation_id") == operation_id - for row in session.get("loopx_deliveries") or [] - ) - return (usable, observed, str(session.get("updated_at") or "")) - - return max(sessions, key=rank) - - -def _pending_identity(record: Path) -> tuple[str, str, str] | None: - """Requester and operation of a pending intent; an unlocked pre-check only.""" - row = _read(record) - wake = row.get("wake") - if row.get("status") != "accepted" or not isinstance(wake, dict) or wake.get("state") != "pending": - return None - requester = wake.get("requester") or {} - identity = (requester.get("goal_id"), requester.get("agent_id"), wake.get("operation_id")) - if not all(isinstance(value, str) and value for value in identity): - return None - return identity # type: ignore[return-value] - - -def pump_delegation_wakes( - controller, - *, - goal_context: Callable[[dict[str, Any]], dict[str, Any]] | None = None, - cancelled: Callable[[], bool] = lambda: False, - limit: int = 20, -) -> list[dict[str, Any]]: - """One restart-safe tick; returns the receipts it changed.""" - store = controller.store - root = store.root.parent - context_for = goal_context or (lambda session: default_goal_context(controller, session)) - owners = _owners(store) - changed: list[dict[str, Any]] = [] - for record in sorted((_root(root) / "executions").glob("*/*.json")): - if cancelled() or len(changed) >= limit: - break - try: - identity = _pending_identity(record) - if identity is None: - continue - goal_id, agent_id, operation_id = identity - if execution_row_path(root, goal_id, agent_id, operation_id) != record: - continue # an intent must name the requester that owns its storage address - sessions = owners.get((goal_id, agent_id)) - if not sessions: - receipt = record_wake(record, lambda wake: wake_receipt( - wake, "refused", reason="no_wake_owner", refused_at=time.time())) - else: - session = _select_owner(sessions, operation_id) - receipt = controller.loopx_mode.wake( - session["session_id"], - record, - goal_context=lambda session=session: context_for(session), - ) - except LockAcquireTimeoutError: - continue # a worker or decision holds the record; retry next tick - except (OSError, ValueError, KeyError, TypeError, RuntimeError, - ChatTurnAcceptanceUnavailableError) as exc: - # Isolate one record; the others still progress this tick. - _LOG.warning("Delegation wake unavailable for one operation (%s); retrying", type(exc).__name__) - continue - if receipt is not None: - changed.append(receipt) - return changed - - -class DelegationWakeService: - """Cheap local pump hosted by the existing Chat server, beside the return service.""" - - def __init__(self, controller, *, goal_context=None, interval=WAKE_INTERVAL_SECONDS): - self.controller = controller - self.goal_context = goal_context - self.interval = interval - self.stop = threading.Event() - self.thread = threading.Thread( - target=self.run, daemon=True, name="loopx-delegation-wakes" - ) - - def start(self): - self.thread.start() - return self - - def run(self): - while not self.stop.is_set(): - try: - pump_delegation_wakes( - self.controller, - goal_context=self.goal_context, - cancelled=self.stop.is_set, - ) - except (OSError, ValueError, KeyError, TypeError, RuntimeError): - _LOG.warning("Delegation wake receipts unavailable; retrying") - self.stop.wait(self.interval) - - def close(self): - self.stop.set() - self.thread.join(timeout=3) diff --git a/loopx/chat_loopx_mode.py b/loopx/chat_loopx_mode.py index c9d22c59f6..a9f0897e00 100644 --- a/loopx/chat_loopx_mode.py +++ b/loopx/chat_loopx_mode.py @@ -7,15 +7,18 @@ from __future__ import annotations import hashlib +import logging import json from pathlib import Path import threading import time +from typing import Any, Callable from .agent_registry import load_goal_from_registry, registered_agent_ids_for_goal from .chat_codex_goal import CodexGoalDriver, validate_goal_chat from .control_plane.effect_runtime import effect_runtime_result -from .file_lock import exclusive_file_lock, LockAcquisitionPolicy +from .control_plane.collaboration.inbox import _read, _root +from .file_lock import exclusive_file_lock, LockAcquisitionPolicy, LockAcquireTimeoutError from .orchestration import ( compact_orchestration_policy, normalize_subagent_execution_config, @@ -874,3 +877,155 @@ def handle_loopx_request(handler, session_id: str, *, apply: bool = False) -> No handler._send_error(str(exc), status=409, error_code="loopx_mode_unavailable") except Exception: handler._send_error("LoopX mode could not access its configured executor or bindings. Check the local configuration and reconnect the conversation.", status=409, error_code="loopx_mode_unavailable") + +# Delegation wake pump. +# +# Continue a Chat LoopX lead once a delegated result is accepted. +# +# The accepted transition leaves a pending wake intent beside the result. Each +# tick scans this runtime's operation records for pending intents and resolves +# the owner from the intent's requester: a Chat LoopX conversation configured +# with that coordinator identity, preferring one that can still continue and, +# among those, the one that observed the operation. The existing Chat LoopX +# owner decides and records the receipt. A requester without such a +# conversation is refused with ``no_wake_owner``; the scheduler deadline recheck +# remains its continuation. The pump never unpauses a lead, never starts a +# native Goal and never settles canonical work. Stopping it leaves intents +# pending, which is its rollback. + +WAKE_INTERVAL_SECONDS = 3.0 +_LOG = logging.getLogger(__name__) + + +def default_goal_context(controller, session: dict[str, Any]) -> dict[str, Any]: + """Project and objective for a host-initiated Turn when no richer reader exists.""" + goal = load_goal_from_registry(controller.registry_path, session["goal_id"]) + if not goal: + raise ValueError("Goal unavailable") + return { + "project": Path(str(goal.get("repo") or ".")).expanduser().resolve(), + "objective": str(goal.get("domain") or session["goal_id"]), + } + + +def _owners(store) -> dict[tuple[str, str], list[dict[str, Any]]]: + """Chat conversations by their configured coordinator (requester) identity.""" + owners: dict[tuple[str, str], list[dict[str, Any]]] = {} + for path in sorted(store.sessions_root.glob("*/session.json")): + session = store.load_session(path.parent.name) + settings = ((session or {}).get("loopx_mode") or {}).get("settings") or {} + agent_id = settings.get("agent_id") + if session and isinstance(agent_id, str) and agent_id: + owners.setdefault((str(session.get("goal_id") or ""), agent_id), []).append(session) + return owners + + +def _select_owner(sessions: list[dict[str, Any]], operation_id: str) -> dict[str, Any]: + def rank(session): + usable = session.get("status") != "closed" and ( + (session.get("loopx_mode") or {}).get("enabled") is True + ) + observed = any( + row.get("operation_id") == operation_id + for row in session.get("loopx_deliveries") or [] + ) + return (usable, observed, str(session.get("updated_at") or "")) + + return max(sessions, key=rank) + + +def _pending_identity(record: Path) -> tuple[str, str, str] | None: + """Requester and operation of a pending intent; an unlocked pre-check only.""" + row = _read(record) + wake = row.get("wake") + if row.get("status") != "accepted" or not isinstance(wake, dict) or wake.get("state") != "pending": + return None + requester = wake.get("requester") or {} + identity = (requester.get("goal_id"), requester.get("agent_id"), wake.get("operation_id")) + if not all(isinstance(value, str) and value for value in identity): + return None + return identity # type: ignore[return-value] + + +def pump_delegation_wakes( + controller, + *, + goal_context: Callable[[dict[str, Any]], dict[str, Any]] | None = None, + cancelled: Callable[[], bool] = lambda: False, + limit: int = 20, +) -> list[dict[str, Any]]: + """One restart-safe tick; returns the receipts it changed.""" + # Imported here: chat_runtime imports this module, and collaboration_mcp + # is already a lazy dependency of LoopXMode. + from .chat_runtime import ChatTurnAcceptanceUnavailableError + from .collaboration_mcp import execution_row_path, record_wake, wake_receipt + + store = controller.store + root = store.root.parent + context_for = goal_context or (lambda session: default_goal_context(controller, session)) + owners = _owners(store) + changed: list[dict[str, Any]] = [] + for record in sorted((_root(root) / "executions").glob("*/*.json")): + if cancelled() or len(changed) >= limit: + break + try: + identity = _pending_identity(record) + if identity is None: + continue + goal_id, agent_id, operation_id = identity + if execution_row_path(root, goal_id, agent_id, operation_id) != record: + continue # an intent must name the requester that owns its storage address + sessions = owners.get((goal_id, agent_id)) + if not sessions: + receipt = record_wake(record, lambda wake: wake_receipt( + wake, "refused", reason="no_wake_owner", refused_at=time.time())) + else: + session = _select_owner(sessions, operation_id) + receipt = controller.loopx_mode.wake( + session["session_id"], + record, + goal_context=lambda session=session: context_for(session), + ) + except LockAcquireTimeoutError: + continue # a worker or decision holds the record; retry next tick + except (OSError, ValueError, KeyError, TypeError, RuntimeError, + ChatTurnAcceptanceUnavailableError) as exc: + # Isolate one record; the others still progress this tick. + _LOG.warning("Delegation wake unavailable for one operation (%s); retrying", type(exc).__name__) + continue + if receipt is not None: + changed.append(receipt) + return changed + + +class DelegationWakeService: + """Cheap local pump hosted by the existing Chat server, beside the return service.""" + + def __init__(self, controller, *, goal_context=None, interval=WAKE_INTERVAL_SECONDS): + self.controller = controller + self.goal_context = goal_context + self.interval = interval + self.stop = threading.Event() + self.thread = threading.Thread( + target=self.run, daemon=True, name="loopx-delegation-wakes" + ) + + def start(self): + self.thread.start() + return self + + def run(self): + while not self.stop.is_set(): + try: + pump_delegation_wakes( + self.controller, + goal_context=self.goal_context, + cancelled=self.stop.is_set, + ) + except (OSError, ValueError, KeyError, TypeError, RuntimeError): + _LOG.warning("Delegation wake receipts unavailable; retrying") + self.stop.wait(self.interval) + + def close(self): + self.stop.set() + self.thread.join(timeout=3) diff --git a/loopx/chat_server.py b/loopx/chat_server.py index 68974a6e8c..1dd147e099 100644 --- a/loopx/chat_server.py +++ b/loopx/chat_server.py @@ -1570,7 +1570,7 @@ def serve_chat( server.lark_goal_topic_runtime.start() from .extensions.lark.manager_returns import start_return_service server.manager_return_service = start_return_service(server, runtime_root) - from .chat_delegation_wake import DelegationWakeService + from .chat_loopx_mode import DelegationWakeService def _wake_goal_context(session): registry = load_registry(server.registry_path) From a187a272e508b42482e76451723643596f157c97 Mon Sep 17 00:00:00 2001 From: song <22676124+songoow@users.noreply.github.com> Date: Tue, 29 Sep 2026 09:57:18 -0400 Subject: [PATCH 09/22] test(chat): cover the host wake of a Chat LoopX lead Expectations follow the wake contract: an accepted result with a pending intent continues its requester's lead exactly once; a second tick, a crash after submit, a client id owned by another Turn, an active or paused lead, a stopped Goal, a requester without a lead conversation, an intent stored under another requester and a non-accepted result each leave the matching receipt without starting a Turn. Mutating both accepted-status guards or the existing-Turn recovery fails these tests. Co-Authored-By: Claude Opus 5.5 (1M context) Signed-off-by: song <22676124+songoow@users.noreply.github.com> --- tests/test_chat_delegation_wake.py | 198 +++++++++++++++++++++++++++++ 1 file changed, 198 insertions(+) create mode 100644 tests/test_chat_delegation_wake.py diff --git a/tests/test_chat_delegation_wake.py b/tests/test_chat_delegation_wake.py new file mode 100644 index 0000000000..b6c50ec6cc --- /dev/null +++ b/tests/test_chat_delegation_wake.py @@ -0,0 +1,198 @@ +"""Host wake of a Chat LoopX lead after a delegated result is accepted. + +Expectations follow the wake contract, not the implementation: an accepted +result with a pending intent continues its requester's lead at most once, the +existing Chat LoopX owner decides admission, and every other state leaves a +distinct receipt without starting a Turn. +""" + +import json + +import pytest + +from loopx.chat_loopx_mode import pump_delegation_wakes +from loopx.collaboration_mcp import execution_row_path +from test_chat_loopx_mode import apply, mode # noqa: F401 +from test_chat_project_coordination import project # noqa: F401 + +INTENT_ID = "a" * 64 +OPERATION_ID = "review-1" + + +def _runtime_root(service): + return service.store.root.parent + + +def _write_record(service, *, status="accepted", agent_id="coordinator", + stored_as=None, operation_id=OPERATION_ID): + root = _runtime_root(service) + owner_goal, owner_agent = stored_as or ("research", agent_id) + path = execution_row_path(root, owner_goal, owner_agent, operation_id) + path.parent.mkdir(parents=True, exist_ok=True) + path.write_text(json.dumps({ + "operation_id": operation_id, + "status": status, + "wake": { + "schema_version": "loopx_delegation_wake_intent_v0", + "intent_id": INTENT_ID, + "requester": {"goal_id": "research", "agent_id": agent_id, "goal_ref": None}, + "operation_id": operation_id, + "request_id": "request-1", + "state": "pending", + }, + })) + return path + + +def _wake(path): + return json.loads(path.read_text())["wake"] + + +def _idle_resumable_lead(mode): # noqa: F811 + """Enable the mode, then leave the lead idle on a resumable native Goal.""" + service, sid, _, settings, calls = mode + apply(mode, "start", settings=settings) + service.store.update_session( + sid, active_turn_id=None, native_goal={"status": "paused", "tokensUsed": 0} + ) + calls.clear() + return service, sid, calls + + +def _pump(service, repo): + return pump_delegation_wakes( + service.controller, + goal_context=lambda session: {"project": repo, "objective": "Continue"}, + ) + + +def test_accepted_result_wakes_the_lead_exactly_once(mode): # noqa: F811 + service, sid, calls = _idle_resumable_lead(mode) + repo = mode[2] + path = _write_record(service) + + changed = _pump(service, repo) + + assert len(calls) == 1 + assert calls[0]["client_turn_id"] == "wake-" + INTENT_ID[:32] + assert calls[0]["loopx_request"]["operation"] == "wake" + assert calls[0]["loopx_request"]["wake"]["intent_id"] == INTENT_ID + receipt = _wake(path) + assert receipt["state"] == "woken" and receipt["created"] is True + assert receipt["session_id"] == sid and changed == [receipt] + + # A second tick finds no pending intent: no second Turn, receipt unchanged. + assert _pump(service, repo) == [] + assert len(calls) == 1 and _wake(path) == receipt + + +def test_crash_after_submit_records_the_existing_turn_once(mode): # noqa: F811 + service, sid, calls = _idle_resumable_lead(mode) + repo = mode[2] + turn, _ = service.store.create_turn( + sid, client_turn_id="wake-" + INTENT_ID[:32], message="/goal resume" + ) + service.store.update_turn( + sid, turn["turn_id"], loopx_execution=True, + loopx_request={"operation": "wake", "wake": {"intent_id": INTENT_ID}}, + ) + service.store.update_session(sid, active_turn_id=None) + path = _write_record(service) + + _pump(service, repo) + + receipt = _wake(path) + assert receipt["state"] == "woken" and receipt["created"] is False + assert receipt["turn_id"] == turn["turn_id"] and calls == [] + + +def test_client_id_owned_by_another_turn_is_not_claimed(mode): # noqa: F811 + service, sid, calls = _idle_resumable_lead(mode) + repo = mode[2] + service.store.create_turn( + sid, client_turn_id="wake-" + INTENT_ID[:32], message="unrelated" + ) + service.store.update_session(sid, active_turn_id=None) + path = _write_record(service) + + _pump(service, repo) + + assert _wake(path)["state"] == "refused" + assert _wake(path)["reason"] == "wake_identity_conflict" and calls == [] + + +def test_active_lead_turn_keeps_the_intent_pending_without_churn(mode): # noqa: F811 + service, sid, _, settings, calls = mode + repo = mode[2] + apply(mode, "start", settings=settings) + calls.clear() + assert service.store.load_session(sid)["active_turn_id"] + path = _write_record(service) + + _pump(service, repo) + first = _wake(path) + assert first["state"] == "pending" and first["reason"] == "lead_turn_active" + before = path.read_bytes() + _pump(service, repo) + assert path.read_bytes() == before and calls == [] + + +def test_paused_lead_stays_pending_and_is_not_unpaused(mode): # noqa: F811 + service, sid, calls = _idle_resumable_lead(mode) + repo = mode[2] + session = service.store.load_session(sid) + service.store.update_session( + sid, loopx_mode={**session["loopx_mode"], "paused": True} + ) + path = _write_record(service) + + _pump(service, repo) + + assert _wake(path)["state"] == "pending" and _wake(path)["reason"] == "lead_paused" + assert service.store.load_session(sid)["loopx_mode"]["paused"] is True + assert calls == [] + + +def test_stopped_goal_refuses_the_wake(mode): # noqa: F811 + service, _, calls = _idle_resumable_lead(mode) + repo = mode[2] + registry = service.controller.registry_path + payload = json.loads(registry.read_text()) + next(goal for goal in payload["goals"] if goal["id"] == "research")["status"] = "stopped" + registry.write_text(json.dumps(payload)) + path = _write_record(service) + + _pump(service, repo) + + assert _wake(path)["state"] == "refused" and _wake(path)["reason"] == "goal_stopped" + assert calls == [] + + +def test_requester_without_a_lead_conversation_is_refused(mode): # noqa: F811 + service, _, calls = _idle_resumable_lead(mode) + repo = mode[2] + path = _write_record(service, agent_id="someone-else") + + _pump(service, repo) + + assert _wake(path)["state"] == "refused" and _wake(path)["reason"] == "no_wake_owner" + assert calls == [] + + +def test_intent_stored_under_another_requester_is_not_decided(mode): # noqa: F811 + service, _, calls = _idle_resumable_lead(mode) + repo = mode[2] + path = _write_record(service, stored_as=("research", "reviewer")) + + assert _pump(service, repo) == [] + assert _wake(path)["state"] == "pending" and calls == [] + + +@pytest.mark.parametrize("status", ["rejected", "running", "stopped"]) +def test_only_an_accepted_result_can_wake(mode, status): # noqa: F811 + service, _, calls = _idle_resumable_lead(mode) + repo = mode[2] + path = _write_record(service, status=status) + + assert _pump(service, repo) == [] + assert _wake(path)["state"] == "pending" and calls == [] From 13bb8a5c17a4d73e09dd9fa7350b8d8e17a2db41 Mon Sep 17 00:00:00 2001 From: song <22676124+songoow@users.noreply.github.com> Date: Tue, 29 Sep 2026 09:57:18 -0400 Subject: [PATCH 10/22] docs(delegation): describe the one-time lead wake and its receipts Goal Chat continuation now lists the wake receipt states and reasons and the limits (no unpause, no new native Goal, no higher allowance, requires the Chat service). The shell entrypoint note points to it. English and Chinese sections stay aligned. Co-Authored-By: Claude Opus 5.5 (1M context) Signed-off-by: song <22676124+songoow@users.noreply.github.com> --- docs/reference/goal-chat-continuation.md | 16 ++++++++++++++++ docs/reference/local-delegation.md | 5 +++-- 2 files changed, 19 insertions(+), 2 deletions(-) diff --git a/docs/reference/goal-chat-continuation.md b/docs/reference/goal-chat-continuation.md index 00569ce34a..482ae63a52 100644 --- a/docs/reference/goal-chat-continuation.md +++ b/docs/reference/goal-chat-continuation.md @@ -98,6 +98,15 @@ messages; images use the ordinary conversation after pausing. - Browser refresh reconnects to the existing local run. After service loss, reconnect restores and pauses the original native thread before resuming. The Chat hard timeout remains in force; this is not an unattended daemon. +- When a delegated result is first accepted, the Chat service continues the + lead once through the same admission as `resume`. The operation record keeps + a separate wake receipt: `woken` with the Turn id, `pending` with + `lead_turn_active`, `lead_paused` or `allowance_exhausted`, or `refused` with + `goal_stopped`, `lead_unbound`, `binding_revoked`, `native_goal_complete`, + `native_goal_absent`, `wake_identity_conflict` or `no_wake_owner`. A wake never + unpauses the lead, never starts a native Goal and never raises the allowance. A lead that already read + the result in its own Turn is marked `observed_in_turn` and is not woken again. + It requires the Chat service to be running; stopping it leaves intents pending. - To roll back, pause/close the Chat service before installing an older build. Disabling mode or deleting a binding does not cancel already admitted children; use their own execution/recovery controls and retain their evidence. @@ -140,6 +149,13 @@ For a disposable mixed-team setup, use the 服务重启先恢复并暂停原线程;浏览器刷新不会另起运行。运行仍有硬超时,尚非 无人值守 daemon。配置文件在未结束的 Goal 中保持摘要绑定,改变后需先协调处理。 +成员结果首次被接受时,Chat 服务按与 `resume` 相同的准入规则继续协调员一次。 +操作记录里另存唤醒回执:`woken` 附 Turn id;`pending` 附 `lead_turn_active`、 +`lead_paused` 或 `allowance_exhausted`;`refused` 附 `goal_stopped`、`lead_unbound`、 +`binding_revoked`、`native_goal_complete`、`native_goal_absent`、`wake_identity_conflict` 或 `no_wake_owner`。 +唤醒不会解除暂停、不会新开原生 Goal、也不会提高额度。协调员已在自己回合内读到结果 +时记为 `observed_in_turn`,不再唤醒。它依赖 Chat 服务在运行;停掉服务时意图保持待处理。 + 协调员保留只读沙箱,成员权限来自各自执行绑定,不继承管家的扩大权限。 成员通过验收与协调员报告、整个 Goal 验收分别显示;本模式不直接完成报告 Todo 或整个 Goal。额度是含历史用量的总量,正在执行的请求可能超额,成员另行计量。 diff --git a/docs/reference/local-delegation.md b/docs/reference/local-delegation.md index 26015e161e..271f96d697 100644 --- a/docs/reference/local-delegation.md +++ b/docs/reference/local-delegation.md @@ -223,7 +223,8 @@ than maintaining separate rules. This entrypoint does not create Agents, grant bindings or wake an idle Codex conversation. The existing host/LoopX continuation policy owns the next lead -turn. The conversation remains persistent independently of whether autonomous +turn; for a Goal Chat LoopX lead that is the Chat service's one-time wake after +acceptance, described in [Goal Chat continuation](goal-chat-continuation.md). The conversation remains persistent independently of whether autonomous LoopX mode is enabled. Dashboard, CLI/managed Turn and Lark keep their existing conversation and runtime owners; they may consume the shared bounded route projection described below, but they do not get another grant or scheduler. @@ -400,7 +401,7 @@ of the new tool description, not injected into those older threads' shared promp 仍显式调用 `resume --execute`。分页回读会重新核验 accepted,单条失效显示 `unavailable`,不能当成失败重派或静默隐藏。`has_more` 表示还有下一页, `page_readback_complete` 只表示本页是否均成功读取;二者都不代表整个团队已完成。 -此入口不创建 Agent、不扩大授权,也不唤醒闲置的 Codex 对话。 +此入口不创建 Agent、不扩大授权,也不唤醒闲置的 Codex 对话。Goal Chat LoopX 模式的协调员由 Chat 服务在结果被接受后唤醒一次,见 [Goal Chat 续跑](goal-chat-continuation.md)。 ### Check a binding before new work From 9983c274b0af140aa3808660afe389b2cd5b82e6 Mon Sep 17 00:00:00 2001 From: song <22676124+songoow@users.noreply.github.com> Date: Tue, 29 Sep 2026 10:02:28 -0400 Subject: [PATCH 11/22] chore(census): register the wake pump's Goal context registry read The Chat server reads the registry to build the Goal context for a host-initiated wake Turn, so that read is a new codec_api site and the server's other sites moved. Regenerated with scripts/generate_project_registry_io_manifest.py; no unclassified sites. Co-Authored-By: Claude Opus 5.5 (1M context) Signed-off-by: song <22676124+songoow@users.noreply.github.com> --- .../semantics/project_registry_io_manifest_v1.json | 14 +++++++++++--- 1 file changed, 11 insertions(+), 3 deletions(-) diff --git a/loopx/semantics/project_registry_io_manifest_v1.json b/loopx/semantics/project_registry_io_manifest_v1.json index 1d92d29240..16164b1ce2 100644 --- a/loopx/semantics/project_registry_io_manifest_v1.json +++ b/loopx/semantics/project_registry_io_manifest_v1.json @@ -431,7 +431,7 @@ }, { "site": "loopx/chat_server.py::.ChatRequestHandler._goal_channel_extension_ready::codec_read:load_registry#1", - "line": 964, + "line": 966, "column": 24, "kind": "codec_read", "api": "load_registry", @@ -439,7 +439,7 @@ }, { "site": "loopx/chat_server.py::.ChatRequestHandler._registry_and_goal::codec_read:load_registry#1", - "line": 511, + "line": 513, "column": 20, "kind": "codec_read", "api": "load_registry", @@ -447,12 +447,20 @@ }, { "site": "loopx/chat_server.py::.serve_chat::codec_read:load_registry#1", - "line": 1486, + "line": 1488, "column": 16, "kind": "codec_read", "api": "load_registry", "classification": "codec_api" }, + { + "site": "loopx/chat_server.py::.serve_chat._wake_goal_context::codec_read:load_registry#1", + "line": 1576, + "column": 20, + "kind": "codec_read", + "api": "load_registry", + "classification": "codec_api" + }, { "site": "loopx/chat_status_api.py::.ChatStatusRequestMixin._status::codec_read:load_registry#1", "line": 112, From 3f99898d65d2d0c880f56797d1e46f35efab87e2 Mon Sep 17 00:00:00 2001 From: song <22676124+songoow@users.noreply.github.com> Date: Tue, 29 Sep 2026 12:31:51 -0400 Subject: [PATCH 12/22] chore(census): follow registry reads moved on main Regenerated with scripts/generate_project_registry_io_manifest.py after rebasing onto main. Site ids and classifications are unchanged. Co-Authored-By: Claude Opus 5.5 (1M context) Signed-off-by: song <22676124+songoow@users.noreply.github.com> --- loopx/semantics/project_registry_io_manifest_v1.json | 8 ++++---- 1 file changed, 4 insertions(+), 4 deletions(-) diff --git a/loopx/semantics/project_registry_io_manifest_v1.json b/loopx/semantics/project_registry_io_manifest_v1.json index 16164b1ce2..8d52be146e 100644 --- a/loopx/semantics/project_registry_io_manifest_v1.json +++ b/loopx/semantics/project_registry_io_manifest_v1.json @@ -1807,7 +1807,7 @@ }, { "site": "loopx/history.py::.collect_history::codec_read:load_registry#1", - "line": 341, + "line": 342, "column": 20, "kind": "codec_read", "api": "load_registry", @@ -1815,7 +1815,7 @@ }, { "site": "loopx/history.py::.inspect_index_duplicates::codec_read:load_registry#1", - "line": 597, + "line": 598, "column": 16, "kind": "codec_read", "api": "load_registry", @@ -1823,7 +1823,7 @@ }, { "site": "loopx/history.py::.rebuild_index_artifact_collisions::codec_read:load_registry#1", - "line": 811, + "line": 812, "column": 16, "kind": "codec_read", "api": "load_registry", @@ -1831,7 +1831,7 @@ }, { "site": "loopx/history.py::.repair_index_duplicates::codec_read:load_registry#1", - "line": 701, + "line": 702, "column": 16, "kind": "codec_read", "api": "load_registry", From 43fcdb8343504bae2bc25e01dd941f058a03b19c Mon Sep 17 00:00:00 2001 From: song Date: Wed, 30 Sep 2026 12:57:31 +0800 Subject: [PATCH 13/22] fix(chat): recover a queued wake Turn through native dispatch A persisted queued Turn is not a completed wake dispatch. The wake decision now carries the Turn already accepted under the intent's client id to the typed planner: only a Turn that started is recorded as woken, a still-queued Turn is replayed through the same native acceptance and dispatch owner, and a Turn that ended before it started is refused instead of being reported as a wake. Pending intents therefore recover from an adapter failure between acceptance and dispatch. Signed-off-by: song --- loopx/chat_loopx_mode.py | 126 ++++++++++++++++++--------------------- 1 file changed, 59 insertions(+), 67 deletions(-) diff --git a/loopx/chat_loopx_mode.py b/loopx/chat_loopx_mode.py index a9f0897e00..901ab410f2 100644 --- a/loopx/chat_loopx_mode.py +++ b/loopx/chat_loopx_mode.py @@ -485,23 +485,23 @@ def settled(state, reason): at = "refused_at" if state == "refused" else "checked_at" return wake_receipt(intent, state, reason=reason, **{at: now}) - try: - session = self._session(session_id) - except ValueError: + session = self.store.load_session(session_id) + if not session: return settled("refused", "no_wake_owner") - # A crash after submit but before the receipt write recovers here: - # the exact client turn already exists, so it is recorded once. + # The Turn this intent's client id already owns in the pinned + # conversation: a crash or lost receipt after acceptance recovers it. existing = self.store.turn_for_client(session_id, client_turn_id) + wake_turn = None if existing: request = existing.get("loopx_request") or {} - if ( - not existing.get("loopx_execution") - or request.get("operation") != "wake" - or (request.get("wake") or {}).get("intent_id") != intent_id - ): - # Another Turn owns this client id; claiming it would be a false receipt. - return settled("refused", "wake_identity_conflict") - return self._woken(intent, session_id, existing["turn_id"], created=False, now=now) + wake_turn = { + "turn_id": existing.get("turn_id"), + "status": existing.get("status"), + "started": existing.get("started_at") is not None, + "loopx_execution": existing.get("loopx_execution") is True, + "operation": request.get("operation"), + "intent_id": (request.get("wake") or {}).get("intent_id"), + } mode = session.get("loopx_mode") or {} settings = mode.get("settings") or {} try: @@ -521,6 +521,8 @@ def settled(state, reason): "session": session, "origin": "host", "operation": "wake", + "intent": intent, + "wake_turn": wake_turn, "settings": settings, "native": session.get("native_goal") or {}, "registered_agents": registered_agent_ids_for_goal(goal) if goal else [], @@ -529,25 +531,40 @@ def settled(state, reason): "execution_binding_valid": binding_valid, }, ) + if decision["state"] == "woken": + # The exact Turn already started; dispatch is not repeated. The + # session itself is an ordinary Goal Chat conversation (not an + # attached host, not another Goal), the same rule ``start`` uses. + validate_goal_chat(session, []) + return self._woken(intent, session_id, existing["turn_id"], created=False, now=now) if decision["state"] != "admitted": return settled(decision["state"], decision["reason"]) + validate_goal_chat(session, []) context = goal_context() + if decision["dispatch"] == "replay": + # Not yet started: the native acceptance owner repairs or replays + # the original request and dispatches that same Turn. + message = existing["message"] + loopx_request = existing["loopx_request"] + else: + message = f"/goal resume --tokens {settings['token_budget']}" + loopx_request = { + "operation": "wake", + "settings": settings, + "wake": { + key: intent.get(key) + for key in ("intent_id", "operation_id", "request_id") + }, + } turn, created = self.controller.submit_turn( session_id=session_id, client_turn_id=client_turn_id, - message=f"/goal resume --tokens {settings['token_budget']}", + message=message, attachments=[], work_dir=context["project"], objective=str(context.get("objective") or context.get("title") or session["goal_id"]), loopx_execution=True, - loopx_request={ - "operation": "wake", - "settings": settings, - "wake": { - key: intent.get(key) - for key in ("intent_id", "operation_id", "request_id") - }, - }, + loopx_request=loopx_request, ) return self._woken(intent, session_id, turn["turn_id"], created=created, now=now) @@ -802,6 +819,7 @@ def dispatch(name, arguments): arguments.get("binding_id", ""), operation_id, arguments.get("brief", {}), + conversation={"session_id": session_id, "turn_id": turn_id}, ) elif action in {"read", "wait", "resume"}: result = ( @@ -883,15 +901,15 @@ def handle_loopx_request(handler, session_id: str, *, apply: bool = False) -> No # Continue a Chat LoopX lead once a delegated result is accepted. # # The accepted transition leaves a pending wake intent beside the result. Each -# tick scans this runtime's operation records for pending intents and resolves -# the owner from the intent's requester: a Chat LoopX conversation configured -# with that coordinator identity, preferring one that can still continue and, -# among those, the one that observed the operation. The existing Chat LoopX -# owner decides and records the receipt. A requester without such a -# conversation is refused with ``no_wake_owner``; the scheduler deadline recheck -# remains its continuation. The pump never unpauses a lead, never starts a -# native Goal and never settles canonical work. Stopping it leaves intents -# pending, which is its rollback. +# tick scans this runtime's operation records for pending intents. An intent +# names the conversation whose Turn started the operation; only that +# conversation is woken, and the existing Chat LoopX owner decides and records +# the receipt. Another conversation of the same Goal and coordinator is never +# selected instead: an intent without a conversation, or whose conversation is +# gone, closed or out of the mode, is refused with ``no_wake_owner`` and the +# scheduler deadline recheck remains its continuation. The pump never +# unpauses a lead, never starts a native Goal and never settles canonical +# work. Stopping it leaves intents pending, which is its rollback. WAKE_INTERVAL_SECONDS = 3.0 _LOG = logging.getLogger(__name__) @@ -908,34 +926,8 @@ def default_goal_context(controller, session: dict[str, Any]) -> dict[str, Any]: } -def _owners(store) -> dict[tuple[str, str], list[dict[str, Any]]]: - """Chat conversations by their configured coordinator (requester) identity.""" - owners: dict[tuple[str, str], list[dict[str, Any]]] = {} - for path in sorted(store.sessions_root.glob("*/session.json")): - session = store.load_session(path.parent.name) - settings = ((session or {}).get("loopx_mode") or {}).get("settings") or {} - agent_id = settings.get("agent_id") - if session and isinstance(agent_id, str) and agent_id: - owners.setdefault((str(session.get("goal_id") or ""), agent_id), []).append(session) - return owners - - -def _select_owner(sessions: list[dict[str, Any]], operation_id: str) -> dict[str, Any]: - def rank(session): - usable = session.get("status") != "closed" and ( - (session.get("loopx_mode") or {}).get("enabled") is True - ) - observed = any( - row.get("operation_id") == operation_id - for row in session.get("loopx_deliveries") or [] - ) - return (usable, observed, str(session.get("updated_at") or "")) - - return max(sessions, key=rank) - - -def _pending_identity(record: Path) -> tuple[str, str, str] | None: - """Requester and operation of a pending intent; an unlocked pre-check only.""" +def _pending_identity(record: Path) -> tuple[str, str, str, str | None] | None: + """Requester, operation and pinned conversation of a pending intent; an unlocked pre-check only.""" row = _read(record) wake = row.get("wake") if row.get("status") != "accepted" or not isinstance(wake, dict) or wake.get("state") != "pending": @@ -944,7 +936,8 @@ def _pending_identity(record: Path) -> tuple[str, str, str] | None: identity = (requester.get("goal_id"), requester.get("agent_id"), wake.get("operation_id")) if not all(isinstance(value, str) and value for value in identity): return None - return identity # type: ignore[return-value] + session_id = (wake.get("conversation") or {}).get("session_id") + return (*identity, session_id if isinstance(session_id, str) and session_id else None) # type: ignore[return-value] def pump_delegation_wakes( @@ -963,7 +956,6 @@ def pump_delegation_wakes( store = controller.store root = store.root.parent context_for = goal_context or (lambda session: default_goal_context(controller, session)) - owners = _owners(store) changed: list[dict[str, Any]] = [] for record in sorted((_root(root) / "executions").glob("*/*.json")): if cancelled() or len(changed) >= limit: @@ -972,19 +964,19 @@ def pump_delegation_wakes( identity = _pending_identity(record) if identity is None: continue - goal_id, agent_id, operation_id = identity + goal_id, agent_id, operation_id, session_id = identity if execution_row_path(root, goal_id, agent_id, operation_id) != record: continue # an intent must name the requester that owns its storage address - sessions = owners.get((goal_id, agent_id)) - if not sessions: + if session_id is None: receipt = record_wake(record, lambda wake: wake_receipt( wake, "refused", reason="no_wake_owner", refused_at=time.time())) else: - session = _select_owner(sessions, operation_id) receipt = controller.loopx_mode.wake( - session["session_id"], + session_id, record, - goal_context=lambda session=session: context_for(session), + goal_context=lambda session_id=session_id: context_for( + store.load_session(session_id) + ), ) except LockAcquireTimeoutError: continue # a worker or decision holds the record; retry next tick From 5e74c7beb70cb94f30e795ab053e711cab84c3ae Mon Sep 17 00:00:00 2001 From: song Date: Wed, 30 Sep 2026 12:57:50 +0800 Subject: [PATCH 14/22] feat(chat): decide a wake against its pinned conversation and Turn planDelegationWake now refuses an intent whose conversation is not this conversation, or whose coordinator is not the configured one, and it reads the wake Turn's own facts: a started Turn is dispatch evidence, a queued one is replayed under the same admission (including its own active Turn id), and a Turn that ended before it started refuses the wake. The planner returns woken/admitted/refused/pending with the exact dispatch the host must perform, so the host can no longer infer dispatch from a Turn's existence. Signed-off-by: song --- .../control_plane/collaboration/chat_mode.ts | 35 +++++++++++++++++-- 1 file changed, 32 insertions(+), 3 deletions(-) diff --git a/loopx/control_plane/collaboration/chat_mode.ts b/loopx/control_plane/collaboration/chat_mode.ts index 1349c25008..635e982d64 100644 --- a/loopx/control_plane/collaboration/chat_mode.ts +++ b/loopx/control_plane/collaboration/chat_mode.ts @@ -54,6 +54,17 @@ export function planChatMode(input: JsonObject): JsonObject { const RESUMABLE_NATIVE = ["paused", "blocked", "usageLimited", "budgetLimited"]; /** A delegated result was accepted; decide only whether the lead may continue now. + * + * The intent is pinned to the conversation whose Turn started the operation; + * no other conversation of the same Goal and coordinator can consume it, and + * a closed or exited origin is refused rather than handed over implicitly. + * + * ``wake_turn`` is the Turn already accepted under this intent's client id in + * that conversation, if any. Only a started Turn is dispatch evidence: it is + * recorded as ``woken`` without another dispatch. A still-queued Turn is not; + * it is replayed through the native acceptance owner under the same admission + * as a new wake, so pause and revocation still hold. A Turn that ended before + * it started is refused, since its client id cannot admit another Turn. * * The same facts that admit an owner resume admit a host wake, plus mode * enabled and not paused. A refusal is terminal for that intent; pending @@ -63,13 +74,31 @@ const RESUMABLE_NATIVE = ["paused", "blocked", "usageLimited", "budgetLimited"]; * starts a native Goal and never raises the conversation allowance. */ function planDelegationWake(input: JsonObject, session: JsonObject, settings: JsonObject, native: JsonObject): JsonObject { const mode = requireJsonObject(session.loopx_mode ?? {}, "mode"); + const intent = requireJsonObject(input.intent, "wake intent"); + const requester = requireJsonObject(intent.requester, "wake requester"); + const conversation = requireJsonObject(intent.conversation, "wake conversation"); + requireThat(typeof intent.intent_id === "string" && /^[a-f0-9]{64}$/.test(intent.intent_id), "invalid wake intent"); const outcome = (state: "pending" | "refused", reason: string) => ({operation: "wake", state, reason}); + if (conversation.session_id !== session.session_id || requester.goal_id !== session.goal_id) { + return outcome("refused", "wake_identity_conflict"); + } + const turn = input.wake_turn == null ? null : requireJsonObject(input.wake_turn, "wake Turn"); + if (turn !== null) { + // Another request owns this client id; claiming it would be a false receipt. + if (turn.loopx_execution !== true || turn.operation !== "wake" || turn.intent_id !== intent.intent_id) { + return outcome("refused", "wake_identity_conflict"); + } + if (turn.started === true) return {operation: "wake", state: "woken", reason: null, dispatch: "recorded"}; + if (turn.status !== "queued") return outcome("refused", "wake_turn_not_started"); + } if (input.goal_active !== true) return outcome("refused", "goal_stopped"); - if (mode.enabled !== true) return outcome("refused", "no_wake_owner"); + if (session.status === "closed" || mode.enabled !== true) return outcome("refused", "no_wake_owner"); + if (settings.agent_id !== requester.agent_id) return outcome("refused", "wake_identity_conflict"); if (!(typeof settings.agent_id === "string" && Array.isArray(input.registered_agents) && input.registered_agents.includes(settings.agent_id))) return outcome("refused", "lead_unbound"); if (input.execution_binding_valid !== true) return outcome("refused", "binding_revoked"); - if (session.active_turn_id) return outcome("pending", "lead_turn_active"); + if (session.active_turn_id && session.active_turn_id !== turn?.turn_id) return outcome("pending", "lead_turn_active"); + // The wake's own queued Turn has not run, so the native status is still the lead's. const status = String(native.status ?? "absent"); if (status === "complete") return outcome("refused", "native_goal_complete"); if (status === "absent") return outcome("refused", "native_goal_absent"); @@ -78,5 +107,5 @@ function planDelegationWake(input: JsonObject, session: JsonObject, settings: Js if (!(Number.isSafeInteger(settings.token_budget) && Number(settings.token_budget) > 0 && Number(settings.token_budget) <= 2147483647 && Number(settings.token_budget) > Number(native.tokensUsed ?? 0))) return outcome("pending", "allowance_exhausted"); - return {operation: "wake", state: "admitted", reason: null, settings}; + return {operation: "wake", state: "admitted", reason: null, dispatch: turn === null ? "create" : "replay", settings}; } From 03efe2150653588fd216760c8d4ab714f6399992 Mon Sep 17 00:00:00 2001 From: song Date: Wed, 30 Sep 2026 12:57:50 +0800 Subject: [PATCH 15/22] feat(collaboration): pin the starting conversation beside an operation The trusted Chat host records the session and Turn that started a delegated operation, once, on first creation; a replay under the same operation id never rebinds it, and the model supplies no routing. The accepted-result wake intent carries that conversation, so a wake returns to the conversation that owes it and to no other. Signed-off-by: song --- loopx/collaboration_mcp.py | 17 ++++++++++++++--- 1 file changed, 14 insertions(+), 3 deletions(-) diff --git a/loopx/collaboration_mcp.py b/loopx/collaboration_mcp.py index 415d93bf7d..883552e6a4 100644 --- a/loopx/collaboration_mcp.py +++ b/loopx/collaboration_mcp.py @@ -305,7 +305,7 @@ def execution_row_path(root: Path, goal_id: str, agent_id: str, operation_id: st return _root(root) / "executions" / _hash([goal_id, agent_id]) / (_hash(operation_id) + ".json") -_WAKE_INTENT_KEYS = ("schema_version", "intent_id", "requester", "operation_id", "request_id") +_WAKE_INTENT_KEYS = ("schema_version", "intent_id", "requester", "conversation", "operation_id", "request_id") def wake_receipt(intent: dict, state: str, **facts) -> dict: @@ -496,7 +496,14 @@ def _inspect(self, binding_id: str) -> dict[str, object]: }) def start(self, binding_id: str, operation_id: str, brief: dict, - parent_request_id: str | None = None) -> dict: + parent_request_id: str | None = None, *, conversation: dict | None = None) -> dict: + """Start or replay one bound operation. + + ``conversation`` is supplied only by the trusted Chat host, never by + the model: the session and Turn that started the operation. It is + kept on first creation and never replaced, so a later wake returns to + that conversation and no other. + """ binding = self.binding(binding_id, require_active=True) require_operation_id(operation_id) brief = normalize_request({"goal_id": self.goal_id, "agent_id": binding["agent_id"], "brief": brief})["brief"] @@ -515,7 +522,10 @@ def start(self, binding_id: str, operation_id: str, brief: dict, if _read(path).get("identity") != identity: raise ValueError("delegation operation identity conflict") else: - _write(path, {"identity": identity, "status": "prepared", "created_at": time.time()}) + origin = ({"session_id": str(conversation["session_id"]), "turn_id": str(conversation["turn_id"])} + if conversation else None) + _write(path, {"identity": identity, "status": "prepared", "created_at": time.time(), + **({"conversation": origin} if origin else {})}) self._spawn(operation_id) return self.read(operation_id) @@ -704,6 +714,7 @@ def _wake_requester(self, row: dict) -> dict: "operation_id": row["identity"]["operation_id"], "request_id": row["identity"]["request_id"], "artifacts": [{k: v for k, v in item.items() if k != "text"} for item in row["artifacts"]], + "conversation": row.get("conversation"), } def wake_observed_in_turn(self, operation_id: str) -> dict | None: From be361c1609be4fc29451a014136554aca77ebad9 Mon Sep 17 00:00:00 2001 From: song Date: Wed, 30 Sep 2026 12:58:19 +0800 Subject: [PATCH 16/22] feat(delegation): bind the wake intent to the starting conversation The originating conversation is part of the intent identity, so the same accepted result cannot be woken into a different conversation of the same requester. Signed-off-by: song --- loopx/control_plane/collaboration/delegation.ts | 15 ++++++++++++--- 1 file changed, 12 insertions(+), 3 deletions(-) diff --git a/loopx/control_plane/collaboration/delegation.ts b/loopx/control_plane/collaboration/delegation.ts index 4afd6879eb..7b29f678e0 100644 --- a/loopx/control_plane/collaboration/delegation.ts +++ b/loopx/control_plane/collaboration/delegation.ts @@ -343,13 +343,21 @@ export function transitionDelegationObservation(params: JsonObject): JsonObject } /** The first transition to ``accepted`` is the one durable moment a requester - * can be continued without polling. The intent names the requester and the - * exact accepted result; it grants no Turn and is not a second settlement. */ + * can be continued without polling. The intent names the requester, the + * conversation whose Turn started the operation (null when it was not started + * from one) and the exact accepted result; it grants no Turn and is not a + * second settlement. The conversation is part of the intent identity, so the + * wake cannot be consumed by another conversation of the same requester. */ function delegationWakeIntent(params: JsonObject): JsonObject { const requester = requireJsonObject(params.requester, "wake requester"); requireThat([requester.goal_id, requester.agent_id, requester.operation_id, requester.request_id].every(text), "wake intent requires the requester and result identity"); const goalRef = requester.goal_ref == null ? null : requireJsonObject(requester.goal_ref, "requester goal reference"); + const origin = requester.conversation == null ? null + : requireJsonObject(requester.conversation, "requester conversation"); + requireThat(origin === null || (text(origin.session_id) && text(origin.turn_id)), + "requester conversation requires its session and Turn"); + const conversation = origin === null ? null : {session_id: origin.session_id, turn_id: origin.turn_id}; requireThat(Array.isArray(requester.artifacts) && requester.artifacts.length > 0, "wake intent requires accepted artifacts"); const digests = requester.artifacts.map(value => { const artifact = requireJsonObject(value, "accepted artifact"); @@ -360,8 +368,9 @@ function delegationWakeIntent(params: JsonObject): JsonObject { return { schema_version: "loopx_delegation_wake_intent_v0", intent_id: canonicalAuthoritySha256([requester.goal_id, requester.agent_id, requester.operation_id, - requester.request_id, digests]), + requester.request_id, digests, conversation]), requester: {goal_id: requester.goal_id, agent_id: requester.agent_id, goal_ref: goalRef}, + conversation, operation_id: requester.operation_id, request_id: requester.request_id, }; } From 67eb11951c062847ca7874a62c462c33a0ac97a0 Mon Sep 17 00:00:00 2001 From: song Date: Wed, 30 Sep 2026 12:58:19 +0800 Subject: [PATCH 17/22] test(delegation): cover wake dispatch recovery and conversation pinning Real ChatRuntimeController.submit_turn, TS acceptance and durable store with faults injected only at the adapter and worker transport: adapter failure after acceptance is re-dispatched as the same Turn, a prepared capsule is repaired, a lost receipt replays only the original Turn, a Turn cancelled before it started is not woken, a queued Turn keeps the pause boundary, and neither exiting, closing nor replacing the origin conversation moves the wake to another session. Signed-off-by: song --- tests/control_plane_ts/chat_mode.test.ts | 55 +++- tests/control_plane_ts/delegation.test.ts | 12 + tests/test_chat_delegation_wake.py | 310 ++++++++++++++++++++-- tests/test_local_delegation.py | 25 ++ 4 files changed, 382 insertions(+), 20 deletions(-) diff --git a/tests/control_plane_ts/chat_mode.test.ts b/tests/control_plane_ts/chat_mode.test.ts index 8529edd74b..c1c8bae765 100644 --- a/tests/control_plane_ts/chat_mode.test.ts +++ b/tests/control_plane_ts/chat_mode.test.ts @@ -46,18 +46,28 @@ test("all three delivery modes require the exact active execution turn", () => { assert.throws(() => planChatMode({...message, delivery_mode: "queue", session: {...active, loopx_mode: {enabled: true, paused: true}}})); }); +const enabled = {...session, session_id: "origin", loopx_mode: {enabled: true, paused: false}}; +const intent = {intent_id: "a".repeat(64), requester: {goal_id: "research", agent_id: "lead"}, + conversation: {session_id: "origin", turn_id: "lead-turn"}}; +const wake: JsonObject = {...input, operation: "wake", origin: "host", session: enabled, intent, + native: {status: "paused", tokensUsed: 400}}; + test("a host wake reuses the resume facts and returns a typed outcome, never an owner change", () => { - const enabled = {...session, loopx_mode: {enabled: true, paused: false}}; - const wake: JsonObject = {...input, operation: "wake", origin: "host", session: enabled, - native: {status: "paused", tokensUsed: 400}}; const admitted = planChatMode(wake); assert.equal(admitted.state, "admitted"); assert.equal(admitted.reason, null); + assert.equal(admitted.dispatch, "create"); assert.deepEqual(admitted.settings, input.settings); // Terminal refusals: the intent is settled and nothing waits. const refused: [JsonObject, string][] = [ [{goal_active: false}, "goal_stopped"], [{session: {...enabled, loopx_mode: {enabled: false}}}, "no_wake_owner"], + [{session: {...enabled, status: "closed"}}, "no_wake_owner"], + // Pinned: another conversation or coordinator of the requester is not the origin. + [{session: {...enabled, session_id: "other"}}, "wake_identity_conflict"], + [{session: {...enabled, session_id: "origin"}, intent: {...intent, requester: {goal_id: "other", agent_id: "lead"}}}, + "wake_identity_conflict"], + [{settings: {agent_id: "other", token_budget: 1000}, registered_agents: ["other"]}, "wake_identity_conflict"], [{registered_agents: ["other"]}, "lead_unbound"], [{execution_binding_valid: false}, "binding_revoked"], [{native: {status: "complete", tokensUsed: 400}}, "native_goal_complete"], @@ -87,3 +97,42 @@ test("a host wake reuses the resume facts and returns a typed outcome, never an assert.throws(() => planChatMode({...wake, origin: "external"}), /local managed/); assert.throws(() => planChatMode({...wake, session: {...enabled, channel_id: "manager"}})); }); + +test("only a started wake Turn is dispatch evidence; a queued one is replayed under the same admission", () => { + const own = {turn_id: "wake-turn", loopx_execution: true, operation: "wake", intent_id: intent.intent_id}; + const queued = {...own, status: "queued", started: false}; + // Queued, not started: replay the same Turn; its own active id does not block it. + const replay = planChatMode({...wake, wake_turn: queued, session: {...enabled, active_turn_id: "wake-turn"}}); + assert.equal(replay.state, "admitted"); + assert.equal(replay.dispatch, "replay"); + // The replay keeps every boundary of a new wake. + const held: [JsonObject, string, string][] = [ + [{session: {...enabled, active_turn_id: "wake-turn", loopx_mode: {enabled: true, paused: true}}}, "pending", "lead_paused"], + [{session: {...enabled, active_turn_id: "other"}}, "pending", "lead_turn_active"], + [{session: {...enabled, loopx_mode: {enabled: false}}}, "refused", "no_wake_owner"], + [{execution_binding_valid: false}, "refused", "binding_revoked"], + [{goal_active: false}, "refused", "goal_stopped"], + ]; + for (const [changes, state, reason] of held) { + assert.deepEqual(planChatMode({...wake, wake_turn: queued, ...changes}), {operation: "wake", state, reason}, reason); + } + // Started (running or terminal): recorded without another dispatch, even after the mode changed. + for (const status of ["starting", "running", "completed", "failed"]) { + assert.deepEqual(planChatMode({...wake, wake_turn: {...own, status, started: true}, + session: {...enabled, loopx_mode: {enabled: false}}}), + {operation: "wake", state: "woken", reason: null, dispatch: "recorded"}, status); + } + // Ended before it started: its client id cannot admit another Turn. + for (const status of ["interrupted", "interrupting", "failed", "completing"]) { + assert.deepEqual(planChatMode({...wake, wake_turn: {...own, status, started: false}}), + {operation: "wake", state: "refused", reason: "wake_turn_not_started"}, status); + } + // A client id owned by another request is never claimed. + for (const other of [{loopx_execution: false}, {operation: "resume"}, {intent_id: "b".repeat(64)}]) { + assert.deepEqual(planChatMode({...wake, wake_turn: {...queued, ...other, started: true}}), + {operation: "wake", state: "refused", reason: "wake_identity_conflict"}); + } + // An intent without its origin conversation cannot be decided by any conversation. + assert.throws(() => planChatMode({...wake, intent: {...intent, conversation: null}}), /conversation/); + assert.throws(() => planChatMode({...wake, intent: {...intent, intent_id: "short"}}), /intent/); +}); diff --git a/tests/control_plane_ts/delegation.test.ts b/tests/control_plane_ts/delegation.test.ts index 4861506b9b..e6c343c9c3 100644 --- a/tests/control_plane_ts/delegation.test.ts +++ b/tests/control_plane_ts/delegation.test.ts @@ -114,6 +114,18 @@ test("only the first transition to accepted leaves a wake intent for the exact r assert.deepEqual(intent.requester, {goal_id: "research", agent_id: "coordinator", goal_ref: null}); assert.equal(intent.operation_id, "analysis-1"); assert.equal(intent.request_id, "req-1"); + assert.equal(intent.conversation, null); + // The originating conversation is pinned into the intent identity. + const pinned = transitionDelegationObservation({from: "turn_returned", to: "accepted", ...acceptedFacts, + requester: {...acceptedRequester, conversation: {session_id: "s-1", turn_id: "t-1", extra: "dropped"}}}); + const pinnedIntent = pinned.wake_intent as Record; + assert.deepEqual(pinnedIntent.conversation, {session_id: "s-1", turn_id: "t-1"}); + assert.notEqual(pinnedIntent.intent_id, intent.intent_id); + const elsewhere = transitionDelegationObservation({from: "turn_returned", to: "accepted", ...acceptedFacts, + requester: {...acceptedRequester, conversation: {session_id: "s-2", turn_id: "t-1"}}}); + assert.notEqual((elsewhere.wake_intent as Record).intent_id, pinnedIntent.intent_id); + assert.throws(() => transitionDelegationObservation({from: "turn_returned", to: "accepted", ...acceptedFacts, + requester: {...acceptedRequester, conversation: {session_id: "s-1"}}}), /conversation/); // Same requester and result: same intent, so a replayed transition cannot mint a second wake. assert.deepEqual(transitionDelegationObservation({from: "turn_returned", to: "accepted", ...acceptedFacts}), accepted); const changed = transitionDelegationObservation({from: "turn_returned", to: "accepted", ...acceptedFacts, diff --git a/tests/test_chat_delegation_wake.py b/tests/test_chat_delegation_wake.py index b6c50ec6cc..0bc55c8ccb 100644 --- a/tests/test_chat_delegation_wake.py +++ b/tests/test_chat_delegation_wake.py @@ -10,7 +10,9 @@ import pytest +import loopx.collaboration_mcp as collaboration_mcp from loopx.chat_loopx_mode import pump_delegation_wakes +from loopx.chat_runtime import ChatRuntimeController from loopx.collaboration_mcp import execution_row_path from test_chat_loopx_mode import apply, mode # noqa: F401 from test_chat_project_coordination import project # noqa: F401 @@ -24,7 +26,7 @@ def _runtime_root(service): def _write_record(service, *, status="accepted", agent_id="coordinator", - stored_as=None, operation_id=OPERATION_ID): + stored_as=None, operation_id=OPERATION_ID, session_id=None): root = _runtime_root(service) owner_goal, owner_agent = stored_as or ("research", agent_id) path = execution_row_path(root, owner_goal, owner_agent, operation_id) @@ -38,6 +40,9 @@ def _write_record(service, *, status="accepted", agent_id="coordinator", "requester": {"goal_id": "research", "agent_id": agent_id, "goal_ref": None}, "operation_id": operation_id, "request_id": "request-1", + # The conversation whose Turn started the operation; the only wake target. + **({"conversation": {"session_id": session_id, "turn_id": "lead-turn"}} + if session_id else {}), "state": "pending", }, })) @@ -69,7 +74,7 @@ def _pump(service, repo): def test_accepted_result_wakes_the_lead_exactly_once(mode): # noqa: F811 service, sid, calls = _idle_resumable_lead(mode) repo = mode[2] - path = _write_record(service) + path = _write_record(service, session_id=sid) changed = _pump(service, repo) @@ -86,18 +91,20 @@ def test_accepted_result_wakes_the_lead_exactly_once(mode): # noqa: F811 assert len(calls) == 1 and _wake(path) == receipt -def test_crash_after_submit_records_the_existing_turn_once(mode): # noqa: F811 +def test_lost_receipt_after_the_turn_started_records_it_once(mode): # noqa: F811 service, sid, calls = _idle_resumable_lead(mode) repo = mode[2] turn, _ = service.store.create_turn( sid, client_turn_id="wake-" + INTENT_ID[:32], message="/goal resume" ) + # A started Turn is dispatch evidence; a merely persisted one is not. service.store.update_turn( - sid, turn["turn_id"], loopx_execution=True, + sid, turn["turn_id"], loopx_execution=True, status="completed", + started_at="2026-09-30T00:00:00Z", completed_at="2026-09-30T00:01:00Z", loopx_request={"operation": "wake", "wake": {"intent_id": INTENT_ID}}, ) service.store.update_session(sid, active_turn_id=None) - path = _write_record(service) + path = _write_record(service, session_id=sid) _pump(service, repo) @@ -113,7 +120,7 @@ def test_client_id_owned_by_another_turn_is_not_claimed(mode): # noqa: F811 sid, client_turn_id="wake-" + INTENT_ID[:32], message="unrelated" ) service.store.update_session(sid, active_turn_id=None) - path = _write_record(service) + path = _write_record(service, session_id=sid) _pump(service, repo) @@ -127,7 +134,7 @@ def test_active_lead_turn_keeps_the_intent_pending_without_churn(mode): # noqa: apply(mode, "start", settings=settings) calls.clear() assert service.store.load_session(sid)["active_turn_id"] - path = _write_record(service) + path = _write_record(service, session_id=sid) _pump(service, repo) first = _wake(path) @@ -144,7 +151,7 @@ def test_paused_lead_stays_pending_and_is_not_unpaused(mode): # noqa: F811 service.store.update_session( sid, loopx_mode={**session["loopx_mode"], "paused": True} ) - path = _write_record(service) + path = _write_record(service, session_id=sid) _pump(service, repo) @@ -154,13 +161,13 @@ def test_paused_lead_stays_pending_and_is_not_unpaused(mode): # noqa: F811 def test_stopped_goal_refuses_the_wake(mode): # noqa: F811 - service, _, calls = _idle_resumable_lead(mode) + service, sid, calls = _idle_resumable_lead(mode) repo = mode[2] registry = service.controller.registry_path payload = json.loads(registry.read_text()) next(goal for goal in payload["goals"] if goal["id"] == "research")["status"] = "stopped" registry.write_text(json.dumps(payload)) - path = _write_record(service) + path = _write_record(service, session_id=sid) _pump(service, repo) @@ -168,10 +175,12 @@ def test_stopped_goal_refuses_the_wake(mode): # noqa: F811 assert calls == [] -def test_requester_without_a_lead_conversation_is_refused(mode): # noqa: F811 - service, _, calls = _idle_resumable_lead(mode) +def test_intent_without_an_originating_conversation_wakes_nobody(mode): # noqa: F811 + """An operation started outside a Chat conversation has no wake target; + a same-identity conversation is never selected in its place.""" + service, sid, calls = _idle_resumable_lead(mode) repo = mode[2] - path = _write_record(service, agent_id="someone-else") + path = _write_record(service) _pump(service, repo) @@ -179,10 +188,21 @@ def test_requester_without_a_lead_conversation_is_refused(mode): # noqa: F811 assert calls == [] +def test_pinned_conversation_rebound_to_another_requester_is_refused(mode): # noqa: F811 + service, sid, calls = _idle_resumable_lead(mode) + repo = mode[2] + path = _write_record(service, agent_id="someone-else", session_id=sid) + + _pump(service, repo) + + assert _wake(path)["state"] == "refused" + assert _wake(path)["reason"] == "wake_identity_conflict" and calls == [] + + def test_intent_stored_under_another_requester_is_not_decided(mode): # noqa: F811 - service, _, calls = _idle_resumable_lead(mode) + service, sid, calls = _idle_resumable_lead(mode) repo = mode[2] - path = _write_record(service, stored_as=("research", "reviewer")) + path = _write_record(service, stored_as=("research", "reviewer"), session_id=sid) assert _pump(service, repo) == [] assert _wake(path)["state"] == "pending" and calls == [] @@ -190,9 +210,265 @@ def test_intent_stored_under_another_requester_is_not_decided(mode): # noqa: F8 @pytest.mark.parametrize("status", ["rejected", "running", "stopped"]) def test_only_an_accepted_result_can_wake(mode, status): # noqa: F811 - service, _, calls = _idle_resumable_lead(mode) + service, sid, calls = _idle_resumable_lead(mode) repo = mode[2] - path = _write_record(service, status=status) + path = _write_record(service, status=status, session_id=sid) assert _pump(service, repo) == [] assert _wake(path)["state"] == "pending" and calls == [] + + +# Native dispatch recovery and conversation pinning. +# +# These run the real ChatRuntimeController.submit_turn, TS turn acceptance and +# the durable Chat store. Only the adapter and worker transport are replaced, +# so no model runs; "dispatch" below is a request to start the worker. + +class Transport: + """Adapter/worker boundary with injectable faults.""" + + def __init__(self, service, *, adapter_failures=0, start_turns=False): + self.service = service + self.adapter_failures = adapter_failures + self.start_turns = start_turns + self.adapter_attempts = [] + self.dispatches = [] + + def ensure_adapter(self, session, **kwargs): + self.adapter_attempts.append(session["session_id"]) + if len(self.adapter_attempts) <= self.adapter_failures: + raise RuntimeError("adapter initialization failed") + return object() + + def start_worker(self, *, session_id, turn_id, **kwargs): + self.dispatches.append((session_id, turn_id)) + if self.start_turns: # the worker's first durable step + self.service.store.update_turn( + session_id, turn_id, expected_statuses={"queued"}, + status="starting", started_at="2026-09-30T00:00:00Z", + ) + return True + + +def _native(mode, monkeypatch, **faults): # noqa: F811 + service, sid, repo, settings, _ = mode + apply(mode, "start", settings=settings) + store = service.store + start_turn = store.load_session(sid)["active_turn_id"] + store.update_turn(sid, start_turn, status="completed", started_at="2026-09-30T00:00:00Z", + completed_at="2026-09-30T00:00:01Z") + store.update_session(sid, active_turn_id=None, status="ready", loopx_tools=True, + native_goal={"status": "paused", "tokensUsed": 0}) + controller = service.controller + transport = Transport(service, **faults) + monkeypatch.setattr(controller, "submit_turn", + ChatRuntimeController.submit_turn.__get__(controller)) + monkeypatch.setattr(controller, "_ensure_adapter_locked", transport.ensure_adapter) + monkeypatch.setattr(controller, "_start_accepted_turn_worker", transport.start_worker) + return service, sid, repo, transport + + +def _wake_turns(service, sid): + turns = service.store.root / "sessions" / sid / "turns" + return [ + row for row in (json.loads(p.read_text()) for p in turns.glob("*.json") + if not p.name.endswith(".events.json")) + if str(row.get("client_turn_id", "")).startswith("wake-") + ] + + +def _second_lead(service, sid, settings): + """Another enabled conversation for the same Goal and coordinator identity.""" + store = service.store + other = store.create_session( + goal_id="research", agent_id="codex", channel_id="goal.research", + upstream_thread_id="fixture-2", upstream_mode="chat", adapter_kind="codex_app_server", + )["session_id"] + mode_state = store.load_session(sid)["loopx_mode"] + store.update_session(other, loopx_mode={**mode_state, "enabled": True, "paused": False}, + loopx_tools=True, status="ready", + native_goal={"status": "paused", "tokensUsed": 0}) + return other + + +def _fail_next_wake_receipt(monkeypatch): + real = collaboration_mcp._write + failed = [] + + def write(path, row): + if not failed and (row.get("wake") or {}).get("state") == "woken": + failed.append(path) + raise OSError("receipt write lost") + return real(path, row) + + monkeypatch.setattr(collaboration_mcp, "_write", write) + return failed + + +def test_adapter_failure_after_acceptance_is_redispatched_not_recorded(mode, monkeypatch): # noqa: F811 + service, sid, repo, transport = _native(mode, monkeypatch, adapter_failures=1) + path = _write_record(service, session_id=sid) + + assert _pump(service, repo) == [] + [queued] = _wake_turns(service, sid) + assert queued["status"] == "queued" and queued["started_at"] is None + assert _wake(path)["state"] == "pending" and transport.dispatches == [] + + _pump(service, repo) + + # The native acceptance owner re-dispatches the same Turn; no second Turn. + assert len(transport.adapter_attempts) == 2 + assert transport.dispatches == [(sid, queued["turn_id"])] + assert [row["turn_id"] for row in _wake_turns(service, sid)] == [queued["turn_id"]] + receipt = _wake(path) + assert receipt["state"] == "woken" and receipt["turn_id"] == queued["turn_id"] + assert receipt["created"] is False + + +def test_failure_after_the_prepared_capsule_is_repaired_and_dispatched(mode, monkeypatch): # noqa: F811 + service, sid, repo, transport = _native(mode, monkeypatch) + path = _write_record(service, session_id=sid) + real = service.store.append_message + failed = [] + + def append_message(*args, **kwargs): + if not failed: + failed.append(True) + raise OSError("transcript unavailable") + return real(*args, **kwargs) + + monkeypatch.setattr(service.store, "append_message", append_message) + assert _pump(service, repo) == [] + [prepared] = _wake_turns(service, sid) + assert "_acceptance" in prepared and transport.dispatches == [] + assert _wake(path)["state"] == "pending" + + _pump(service, repo) + + [settled] = _wake_turns(service, sid) + assert settled["turn_id"] == prepared["turn_id"] and "_acceptance" not in settled + assert transport.dispatches == [(sid, prepared["turn_id"])] + assert _wake(path)["state"] == "woken" + + +@pytest.mark.parametrize("started", [False, True]) +def test_lost_receipt_replays_only_the_original_turn(mode, monkeypatch, started): # noqa: F811 + service, sid, repo, transport = _native(mode, monkeypatch, start_turns=started) + path = _write_record(service, session_id=sid) + lost = _fail_next_wake_receipt(monkeypatch) + + assert _pump(service, repo) == [] and lost + assert _wake(path)["state"] == "pending" and len(transport.dispatches) == 1 + + _pump(service, repo) + + [turn] = _wake_turns(service, sid) + receipt = _wake(path) + assert receipt["state"] == "woken" and receipt["turn_id"] == turn["turn_id"] + # A started Turn is recorded without another dispatch; a still-queued one + # is handed to the native owner again for the same Turn. + expected = 1 if started else 2 + assert transport.dispatches == [(sid, turn["turn_id"])] * expected + + +def test_wake_turn_cancelled_before_it_started_is_not_woken(mode, monkeypatch): # noqa: F811 + service, sid, repo, transport = _native(mode, monkeypatch, adapter_failures=1) + path = _write_record(service, session_id=sid) + _pump(service, repo) + [queued] = _wake_turns(service, sid) + service.store.update_turn(sid, queued["turn_id"], status="interrupted", + completed_at="2026-09-30T00:00:02Z") + service.store.update_session(sid, active_turn_id=None, status="ready") + + _pump(service, repo) + + receipt = _wake(path) + assert receipt["state"] == "refused" and receipt["reason"] == "wake_turn_not_started" + assert transport.dispatches == [] + + +def test_queued_wake_turn_keeps_the_pause_boundary(mode, monkeypatch): # noqa: F811 + service, sid, repo, transport = _native(mode, monkeypatch, adapter_failures=1) + path = _write_record(service, session_id=sid) + _pump(service, repo) + session = service.store.load_session(sid) + service.store.update_session(sid, loopx_mode={**session["loopx_mode"], "paused": True}) + + _pump(service, repo) + + assert _wake(path)["state"] == "pending" and _wake(path)["reason"] == "lead_paused" + assert transport.dispatches == [] and len(transport.adapter_attempts) == 1 + + +def test_wake_never_moves_to_another_conversation_after_exit(mode, monkeypatch): # noqa: F811 + service, sid, repo, transport = _native(mode, monkeypatch) + other = _second_lead(service, sid, mode[3]) + session = service.store.load_session(sid) + service.store.update_session(sid, loopx_mode={**session["loopx_mode"], "enabled": False}) + path = _write_record(service, session_id=sid) + + _pump(service, repo) + + assert _wake(path)["state"] == "refused" and _wake(path)["reason"] == "no_wake_owner" + assert transport.dispatches == [] and _wake_turns(service, other) == [] + + +def test_closed_origin_conversation_is_refused_without_a_substitute(mode, monkeypatch): # noqa: F811 + service, sid, repo, transport = _native(mode, monkeypatch) + other = _second_lead(service, sid, mode[3]) + service.store.update_session(sid, status="closed") + path = _write_record(service, session_id=sid) + + _pump(service, repo) + + assert _wake(path)["state"] == "refused" and _wake(path)["reason"] == "no_wake_owner" + assert transport.dispatches == [] and _wake_turns(service, other) == [] + + +def test_lost_receipt_then_new_conversation_cannot_dispatch_twice(mode, monkeypatch): # noqa: F811 + service, sid, repo, transport = _native(mode, monkeypatch, start_turns=True) + path = _write_record(service, session_id=sid) + _fail_next_wake_receipt(monkeypatch) + _pump(service, repo) + assert len(transport.dispatches) == 1 and _wake(path)["state"] == "pending" + [turn] = _wake_turns(service, sid) + # The original conversation leaves the mode; a new one takes the same identity. + service.store.update_turn(sid, turn["turn_id"], status="completed", + completed_at="2026-09-30T00:00:03Z") + session = service.store.load_session(sid) + service.store.update_session(sid, active_turn_id=None, status="ready", + loopx_mode={**session["loopx_mode"], "enabled": False}) + other = _second_lead(service, sid, mode[3]) + + _pump(service, repo) + + receipt = _wake(path) + assert receipt["state"] == "woken" and receipt["session_id"] == sid + assert receipt["turn_id"] == turn["turn_id"] and receipt["created"] is False + assert len(transport.dispatches) == 1 and _wake_turns(service, other) == [] + +def test_the_in_turn_tool_pins_the_conversation_it_runs_in(mode, monkeypatch): # noqa: F811 + """The model supplies the brief, never the wake routing: the host fixes it.""" + from loopx.collaboration_mcp import Delegations + + pinned = [] + + def start(self, binding_id, operation_id, brief, parent_request_id=None, *, conversation=None): + pinned.append(conversation) + return {"operation_id": operation_id, "status": "prepared"} + + monkeypatch.setattr(Delegations, "start", start) + service, sid, _, settings, _ = mode + apply(mode, "start", settings=settings) + adapter = type("A", (), {})() + adapter.session = type("S", (), {})() + adapter.goal_driver = type("D", (), { + "turn_start_handler": None, "stopped": __import__("threading").Event()})() + turn_id = service.store.load_session(sid)["active_turn_id"] + service.prepare(sid, turn_id, adapter, lambda name, arguments: {"ok": True}, lambda *a, **k: None) + + adapter.session.read_tool_handler("loopx_collaboration", { + "action": "start", "binding_id": "review", "operation_id": "op-1", + "brief": {"schema_version": "collaboration_brief_v0"}}) + + assert pinned == [{"session_id": sid, "turn_id": turn_id}] diff --git a/tests/test_local_delegation.py b/tests/test_local_delegation.py index d29b13fd7e..4e8189835b 100644 --- a/tests/test_local_delegation.py +++ b/tests/test_local_delegation.py @@ -234,6 +234,31 @@ def timeout(*_args, **_kwargs): assert runner.read("analysis-timeout")["error"] == "TimeoutExpired" +def test_the_starting_conversation_is_pinned_beside_the_operation(service, monkeypatch): + """The wake can only return to the conversation whose Turn started the work. + + A second start under the same operation id (another conversation, or a + requester recovering its context) replays the original request; it never + rebinds the pin, which is part of the operation's own identity. + """ + root, runner = service + monkeypatch.setattr(runner, "_spawn", lambda _operation_id: None) + start = {"session_id": "chat-session-1", "turn_id": "turn-1"} + runner.start("analysis", "analysis-1", brief(), conversation=start) + assert _read(runner.path("analysis-1"))["conversation"] == start + + elsewhere = {"session_id": "chat-session-2", "turn_id": "turn-9"} + runner.start("analysis", "analysis-1", brief(), conversation=elsewhere) + assert _read(runner.path("analysis-1"))["conversation"] == start + # Starting without a conversation neither adds nor clears one. + runner.start("analysis", "analysis-1", brief()) + assert _read(runner.path("analysis-1"))["conversation"] == start + # An operation started outside a conversation carries no wake target. + monkeypatch.setattr(runner, "_spawn", lambda _operation_id: None) + runner.start("analysis", "analysis-2", brief()) + assert "conversation" not in _read(runner.path("analysis-2")) + + def test_model_success_without_receiver_adoption_cannot_complete(service): root, runner = service (root / "skip-adoption").touch() From 3573ade2d7993e2c72b8b41741c3804ea52c6886 Mon Sep 17 00:00:00 2001 From: song Date: Wed, 30 Sep 2026 12:58:19 +0800 Subject: [PATCH 18/22] docs(delegation): describe pinned wake dispatch and recovery Signed-off-by: song --- docs/reference/goal-chat-continuation.md | 39 ++++++++++++++++-------- docs/reference/local-delegation.md | 7 +++-- 2 files changed, 31 insertions(+), 15 deletions(-) diff --git a/docs/reference/goal-chat-continuation.md b/docs/reference/goal-chat-continuation.md index 482ae63a52..f4c9e6be9a 100644 --- a/docs/reference/goal-chat-continuation.md +++ b/docs/reference/goal-chat-continuation.md @@ -99,14 +99,22 @@ messages; images use the ordinary conversation after pausing. reconnect restores and pauses the original native thread before resuming. The Chat hard timeout remains in force; this is not an unattended daemon. - When a delegated result is first accepted, the Chat service continues the - lead once through the same admission as `resume`. The operation record keeps - a separate wake receipt: `woken` with the Turn id, `pending` with - `lead_turn_active`, `lead_paused` or `allowance_exhausted`, or `refused` with - `goal_stopped`, `lead_unbound`, `binding_revoked`, `native_goal_complete`, - `native_goal_absent`, `wake_identity_conflict` or `no_wake_owner`. A wake never - unpauses the lead, never starts a native Goal and never raises the allowance. A lead that already read - the result in its own Turn is marked `observed_in_turn` and is not woken again. - It requires the Chat service to be running; stopping it leaves intents pending. + lead once through the same admission as `resume`. The wake belongs to the + conversation whose Turn started the operation: another conversation of the + same Goal and coordinator is never selected instead, so exiting, closing or + reconfiguring that conversation refuses the wake rather than moving it. The + operation record keeps a separate wake receipt: `woken` with the Turn id, + `pending` with `lead_turn_active`, `lead_paused` or `allowance_exhausted`, or + `refused` with `goal_stopped`, `lead_unbound`, `binding_revoked`, + `native_goal_complete`, `native_goal_absent`, `wake_identity_conflict`, + `wake_turn_not_started` or `no_wake_owner`. Its Turn is accepted once under a + client id derived from the intent, and only a Turn that started is recorded as + dispatched: a still-queued Turn from an interrupted dispatch is handed back to + the same native acceptance and dispatch owner, so the lead is continued once + without creating a second Turn. A wake never unpauses the lead, never starts a + native Goal and never raises the allowance. A lead that already read the result + in its own Turn is marked `observed_in_turn` and is not woken again. It + requires the Chat service to be running; stopping it leaves intents pending. - To roll back, pause/close the Chat service before installing an older build. Disabling mode or deleting a binding does not cancel already admitted children; use their own execution/recovery controls and retain their evidence. @@ -150,11 +158,16 @@ For a disposable mixed-team setup, use the 无人值守 daemon。配置文件在未结束的 Goal 中保持摘要绑定,改变后需先协调处理。 成员结果首次被接受时,Chat 服务按与 `resume` 相同的准入规则继续协调员一次。 -操作记录里另存唤醒回执:`woken` 附 Turn id;`pending` 附 `lead_turn_active`、 -`lead_paused` 或 `allowance_exhausted`;`refused` 附 `goal_stopped`、`lead_unbound`、 -`binding_revoked`、`native_goal_complete`、`native_goal_absent`、`wake_identity_conflict` 或 `no_wake_owner`。 -唤醒不会解除暂停、不会新开原生 Goal、也不会提高额度。协调员已在自己回合内读到结果 -时记为 `observed_in_turn`,不再唤醒。它依赖 Chat 服务在运行;停掉服务时意图保持待处理。 +唤醒绑定在「启动该操作的回合所属会话」上:同一 Goal 与身份下的其他会话不会被选为 +替代,所以该会话退出、关闭或改配置时只做拒绝,而不是把唤醒移交给别的会话。操作记录 +里另存唤醒回执:`woken` 附 Turn id;`pending` 附 `lead_turn_active`、`lead_paused` +或 `allowance_exhausted`;`refused` 附 `goal_stopped`、`lead_unbound`、`binding_revoked`、 +`native_goal_complete`、`native_goal_absent`、`wake_identity_conflict`、`wake_turn_not_started` +或 `no_wake_owner`。其 Turn 由意图派生的 client id 只接受一次,且只有真正启动过的 +Turn 才算已派发:中断派发后仍处于 queued 的 Turn 会交回同一套原生接受与派发 owner 继续, +因此协调员只被续跑一次,也不会新建第二个 Turn。唤醒不会解除暂停、不会新开原生 Goal、 +也不会提高额度。协调员已在自己回合内读到结果时记为 `observed_in_turn`,不再唤醒。 +它依赖 Chat 服务在运行;停掉服务时意图保持待处理。 协调员保留只读沙箱,成员权限来自各自执行绑定,不继承管家的扩大权限。 成员通过验收与协调员报告、整个 Goal 验收分别显示;本模式不直接完成报告 Todo diff --git a/docs/reference/local-delegation.md b/docs/reference/local-delegation.md index 271f96d697..97c329e271 100644 --- a/docs/reference/local-delegation.md +++ b/docs/reference/local-delegation.md @@ -224,7 +224,10 @@ than maintaining separate rules. This entrypoint does not create Agents, grant bindings or wake an idle Codex conversation. The existing host/LoopX continuation policy owns the next lead turn; for a Goal Chat LoopX lead that is the Chat service's one-time wake after -acceptance, described in [Goal Chat continuation](goal-chat-continuation.md). The conversation remains persistent independently of whether autonomous +acceptance, described in [Goal Chat continuation](goal-chat-continuation.md). +That wake returns to the conversation whose Turn started the operation, which +the trusted Chat host records beside the operation; the model supplies no +routing. The conversation remains persistent independently of whether autonomous LoopX mode is enabled. Dashboard, CLI/managed Turn and Lark keep their existing conversation and runtime owners; they may consume the shared bounded route projection described below, but they do not get another grant or scheduler. @@ -401,7 +404,7 @@ of the new tool description, not injected into those older threads' shared promp 仍显式调用 `resume --execute`。分页回读会重新核验 accepted,单条失效显示 `unavailable`,不能当成失败重派或静默隐藏。`has_more` 表示还有下一页, `page_readback_complete` 只表示本页是否均成功读取;二者都不代表整个团队已完成。 -此入口不创建 Agent、不扩大授权,也不唤醒闲置的 Codex 对话。Goal Chat LoopX 模式的协调员由 Chat 服务在结果被接受后唤醒一次,见 [Goal Chat 续跑](goal-chat-continuation.md)。 +此入口不创建 Agent、不扩大授权,也不唤醒闲置的 Codex 对话。Goal Chat LoopX 模式的协调员由 Chat 服务在结果被接受后唤醒一次,且只回到「启动该操作的回合所属会话」——该绑定由受信任的 Chat 宿主写在操作记录旁,模型不提供路由;见 [Goal Chat 续跑](goal-chat-continuation.md)。 ### Check a binding before new work From 97b9298aa70d1a86a326fc537689bec115f7f3f8 Mon Sep 17 00:00:00 2001 From: song Date: Wed, 30 Sep 2026 06:05:17 -0400 Subject: [PATCH 19/22] fix(collaboration): keep a wake intent recoverable until its start fact lands `submit_turn` returning proves admission and an asynchronous dispatch, not that the Turn's start fact is durable. The wake path wrote a terminal `woken` receipt right after it returned, so a first worker-start write failure left the Turn queued with the intent already settled: storage recovery and a freshly built controller both stopped rediscovering it. Two changes, in their owning modules: - `_wake_decision` now records `pending` / `wake_dispatch_pending` after admission and lets the next tick read the durable `started_at` back, so the receipt stays a statement about a start that actually persisted. The typed owner distinguishes `started_at` from a status guess, keeps a Turn that is still activating pending, and refuses only a Turn that ended without starting, reusing `isTerminalTurnStatus` rather than restating the terminal set. - `_run_turn` released its single-flight guard only when `worker.start()` itself raised. A failure writing `queued -> starting` escaped the worker body and left the key in `turn_done_events` forever, so every later dispatch of that Turn was a silent no-op. The start write and the Turn body now run under one `finally` that hands the guard back. Coverage pins the failure window itself: a one-shot `OSError` on the first `queued -> starting` write, then the same Turn replaying through native dispatch to one start, a rebuilt controller over the persisted store rediscovering the intent, and terminal de-duplication after that. Characterization of the pass path is unchanged. Signed-off-by: song --- loopx/chat_loopx_mode.py | 19 +- loopx/chat_runtime.py | 57 +++-- .../control_plane/collaboration/chat_mode.ts | 16 +- .../turn_driver/chat_turn_acceptance.ts | 8 + tests/control_plane_ts/chat_mode.test.ts | 19 +- tests/test_chat_delegation_wake.py | 236 ++++++++++++++++-- 6 files changed, 300 insertions(+), 55 deletions(-) diff --git a/loopx/chat_loopx_mode.py b/loopx/chat_loopx_mode.py index 901ab410f2..a3ba5f5581 100644 --- a/loopx/chat_loopx_mode.py +++ b/loopx/chat_loopx_mode.py @@ -479,11 +479,15 @@ def _wake_decision(self, session_id, intent, *, goal_context): client_turn_id = "wake-" + intent_id[:32] now = time.time() - def settled(state, reason): - if state == "pending" and intent.get("reason") == reason: + def settled(state, reason, **facts): + if ( + state == "pending" + and intent.get("reason") == reason + and all(intent.get(key) == value for key, value in facts.items()) + ): return None # unchanged: no churn on the record at = "refused_at" if state == "refused" else "checked_at" - return wake_receipt(intent, state, reason=reason, **{at: now}) + return wake_receipt(intent, state, reason=reason, **facts, **{at: now}) session = self.store.load_session(session_id) if not session: @@ -497,7 +501,7 @@ def settled(state, reason): wake_turn = { "turn_id": existing.get("turn_id"), "status": existing.get("status"), - "started": existing.get("started_at") is not None, + "started_at": existing.get("started_at"), "loopx_execution": existing.get("loopx_execution") is True, "operation": request.get("operation"), "intent_id": (request.get("wake") or {}).get("intent_id"), @@ -556,7 +560,7 @@ def settled(state, reason): for key in ("intent_id", "operation_id", "request_id") }, } - turn, created = self.controller.submit_turn( + turn, _created = self.controller.submit_turn( session_id=session_id, client_turn_id=client_turn_id, message=message, @@ -566,7 +570,10 @@ def settled(state, reason): loopx_execution=True, loopx_request=loopx_request, ) - return self._woken(intent, session_id, turn["turn_id"], created=created, now=now) + # Accepted is not started. ``submit_turn`` returning proves admission + # and an asynchronous dispatch, not that the start fact is durable, so + # the intent stays pending until a later tick reads it back. + return settled("pending", "wake_dispatch_pending", turn_id=turn["turn_id"]) @staticmethod def _woken(intent, session_id, turn_id, *, created, now): diff --git a/loopx/chat_runtime.py b/loopx/chat_runtime.py index ed36290f44..ede6fe95c9 100644 --- a/loopx/chat_runtime.py +++ b/loopx/chat_runtime.py @@ -1373,21 +1373,51 @@ def _run_turn( if done_event is None: done_event = threading.Event() self.turn_done_events[key] = done_event - started = utc_now() - started_turn = self.store.update_turn( - session_id, - turn_id, - expected_statuses={"queued"}, - status="starting", - started_at=started, - ) - if started_turn is None: + # Everything runs inside the terminal release. The start fact is the + # first durable write and is not guaranteed to land: if it raises, the + # Turn was never dispatched, so the single-flight guard must be given + # back here rather than left holding the key, which would make every + # later dispatch of this same Turn a silent no-op. + try: + started_turn = self.store.update_turn( + session_id, + turn_id, + expected_statuses={"queued"}, + status="starting", + started_at=utc_now(), + ) + if started_turn is None: + with self.lock: + self.cancelled_turns.discard(key) + return + self._run_started_turn( + session_id=session_id, + turn_id=turn_id, + message=message, + attachments=attachments, + adapter=adapter, + loopx_execution=loopx_execution, + done_event=done_event, + ) + finally: + done_event.set() with self.lock: - self.cancelled_turns.discard((session_id, turn_id)) if self.turn_done_events.get(key) is done_event: self.turn_done_events.pop(key, None) - done_event.set() - return + + def _run_started_turn( + self, + *, + session_id: str, + turn_id: str, + message: str, + attachments: list[dict[str, Any]], + adapter: ChatRuntimeAdapter, + loopx_execution: bool, + done_event: threading.Event, + ) -> None: + """The body of a Turn whose `queued -> starting` fact is already durable.""" + key = (session_id, turn_id) event_buffer = _TurnEventBuffer( store=self.store, session_id=session_id, @@ -1573,9 +1603,6 @@ def event_sink(kind: str, payload: dict[str, Any]) -> None: event_buffer.close() with self.lock: self.turn_event_buffers.pop(key, None) - if self.turn_done_events.get(key) is done_event: - self.turn_done_events.pop(key, None) - done_event.set() def _fail_turn( self, diff --git a/loopx/control_plane/collaboration/chat_mode.ts b/loopx/control_plane/collaboration/chat_mode.ts index 635e982d64..b988259c18 100644 --- a/loopx/control_plane/collaboration/chat_mode.ts +++ b/loopx/control_plane/collaboration/chat_mode.ts @@ -4,6 +4,7 @@ import type {JsonObject} from "../effect_program.ts"; import {EffectRuntimeRequestError} from "../effect_runtime_errors.ts"; import {requireJsonObject} from "../runtime_decode.ts"; import {resolveConversationScope} from "./conversation_scope.ts"; +import {isTerminalTurnStatus} from "../turn_driver/chat_turn_acceptance.ts"; function requireThat(ok: unknown, message: string): asserts ok { if (!ok) throw new EffectRuntimeRequestError(message); @@ -63,8 +64,10 @@ const RESUMABLE_NATIVE = ["paused", "blocked", "usageLimited", "budgetLimited"]; * that conversation, if any. Only a started Turn is dispatch evidence: it is * recorded as ``woken`` without another dispatch. A still-queued Turn is not; * it is replayed through the native acceptance owner under the same admission - * as a new wake, so pause and revocation still hold. A Turn that ended before - * it started is refused, since its client id cannot admit another Turn. + * as a new wake, so pause and revocation still hold. ``started_at`` is the + * durable start fact, so a Turn still activating stays pending rather than + * being read as a start. A Turn that ended before it started is refused, + * since its client id cannot admit another Turn. * * The same facts that admit an owner resume admit a host wake, plus mode * enabled and not paused. A refusal is terminal for that intent; pending @@ -88,8 +91,13 @@ function planDelegationWake(input: JsonObject, session: JsonObject, settings: Js if (turn.loopx_execution !== true || turn.operation !== "wake" || turn.intent_id !== intent.intent_id) { return outcome("refused", "wake_identity_conflict"); } - if (turn.started === true) return {operation: "wake", state: "woken", reason: null, dispatch: "recorded"}; - if (turn.status !== "queued") return outcome("refused", "wake_turn_not_started"); + if (typeof turn.started_at === "string" && turn.started_at) { + return {operation: "wake", state: "woken", reason: null, dispatch: "recorded"}; + } + // Ended without ever starting: its client id cannot admit another Turn. + if (isTerminalTurnStatus(turn.status)) return outcome("refused", "wake_turn_ended_unstarted"); + // Accepted but not yet started, including a start still activating. + if (turn.status !== "queued") return outcome("pending", "wake_dispatch_pending"); } if (input.goal_active !== true) return outcome("refused", "goal_stopped"); if (session.status === "closed" || mode.enabled !== true) return outcome("refused", "no_wake_owner"); diff --git a/loopx/control_plane/turn_driver/chat_turn_acceptance.ts b/loopx/control_plane/turn_driver/chat_turn_acceptance.ts index 3ea714fee0..6a384ad3b4 100644 --- a/loopx/control_plane/turn_driver/chat_turn_acceptance.ts +++ b/loopx/control_plane/turn_driver/chat_turn_acceptance.ts @@ -35,6 +35,14 @@ const TERMINAL_TURN_STATUSES = new Set([ "timed_out", "failed", ]); + +/** A terminal Turn can never be dispatched again under its client identity. + * + * Callers deciding whether a persisted Turn is still recoverable share this + * owner instead of restating the terminal set beside it. */ +export function isTerminalTurnStatus(value: unknown): boolean { + return TERMINAL_TURN_STATUSES.has(value as TurnStatus); +} const OPAQUE_ID = /^[A-Za-z0-9._-]{1,160}$/; const SHA256 = ENVELOPED_SHA256_PATTERN; const EMPTY_OBJECT_SHA256 = sha256("{}"); diff --git a/tests/control_plane_ts/chat_mode.test.ts b/tests/control_plane_ts/chat_mode.test.ts index c1c8bae765..3332fb9ad0 100644 --- a/tests/control_plane_ts/chat_mode.test.ts +++ b/tests/control_plane_ts/chat_mode.test.ts @@ -100,7 +100,7 @@ test("a host wake reuses the resume facts and returns a typed outcome, never an test("only a started wake Turn is dispatch evidence; a queued one is replayed under the same admission", () => { const own = {turn_id: "wake-turn", loopx_execution: true, operation: "wake", intent_id: intent.intent_id}; - const queued = {...own, status: "queued", started: false}; + const queued = {...own, status: "queued", started_at: null}; // Queued, not started: replay the same Turn; its own active id does not block it. const replay = planChatMode({...wake, wake_turn: queued, session: {...enabled, active_turn_id: "wake-turn"}}); assert.equal(replay.state, "admitted"); @@ -116,20 +116,25 @@ test("only a started wake Turn is dispatch evidence; a queued one is replayed un for (const [changes, state, reason] of held) { assert.deepEqual(planChatMode({...wake, wake_turn: queued, ...changes}), {operation: "wake", state, reason}, reason); } - // Started (running or terminal): recorded without another dispatch, even after the mode changed. + // The durable start fact, in any later status and even after the mode changed. for (const status of ["starting", "running", "completed", "failed"]) { - assert.deepEqual(planChatMode({...wake, wake_turn: {...own, status, started: true}, + assert.deepEqual(planChatMode({...wake, wake_turn: {...own, status, started_at: "2026-09-30T00:00:00Z"}, session: {...enabled, loopx_mode: {enabled: false}}}), {operation: "wake", state: "woken", reason: null, dispatch: "recorded"}, status); } // Ended before it started: its client id cannot admit another Turn. - for (const status of ["interrupted", "interrupting", "failed", "completing"]) { - assert.deepEqual(planChatMode({...wake, wake_turn: {...own, status, started: false}}), - {operation: "wake", state: "refused", reason: "wake_turn_not_started"}, status); + for (const status of ["interrupted", "failed", "timed_out", "completed"]) { + assert.deepEqual(planChatMode({...wake, wake_turn: {...own, status, started_at: null}}), + {operation: "wake", state: "refused", reason: "wake_turn_ended_unstarted"}, status); + } + // Accepted but not yet started: still pending, so the next tick re-reads the fact. + for (const status of ["interrupting", "completing"]) { + assert.deepEqual(planChatMode({...wake, wake_turn: {...own, status, started_at: null}}), + {operation: "wake", state: "pending", reason: "wake_dispatch_pending"}, status); } // A client id owned by another request is never claimed. for (const other of [{loopx_execution: false}, {operation: "resume"}, {intent_id: "b".repeat(64)}]) { - assert.deepEqual(planChatMode({...wake, wake_turn: {...queued, ...other, started: true}}), + assert.deepEqual(planChatMode({...wake, wake_turn: {...queued, ...other, started_at: "2026-09-30T00:00:00Z"}}), {operation: "wake", state: "refused", reason: "wake_identity_conflict"}); } // An intent without its origin conversation cannot be decided by any conversation. diff --git a/tests/test_chat_delegation_wake.py b/tests/test_chat_delegation_wake.py index 0bc55c8ccb..5c441a4178 100644 --- a/tests/test_chat_delegation_wake.py +++ b/tests/test_chat_delegation_wake.py @@ -64,13 +64,31 @@ def _idle_resumable_lead(mode): # noqa: F811 return service, sid, calls -def _pump(service, repo): +def _pump_with(controller, repo): return pump_delegation_wakes( - service.controller, + controller, goal_context=lambda session: {"project": repo, "objective": "Continue"}, ) +def _pump(service, repo): + return _pump_with(service.controller, repo) + + +def _start_worker_turn(service, sid): + """The worker's first durable step, for the stubbed ``submit_turn`` fixture. + + The shared fixture's ``submit_turn`` records acceptance only, exactly as the + real one returns before the worker writes ``started_at``. + """ + turn = service.store.turn_for_client(sid, "wake-" + INTENT_ID[:32]) + service.store.update_turn( + sid, turn["turn_id"], expected_statuses={"queued"}, + status="starting", started_at="2026-09-30T00:00:00Z", + ) + return turn + + def test_accepted_result_wakes_the_lead_exactly_once(mode): # noqa: F811 service, sid, calls = _idle_resumable_lead(mode) repo = mode[2] @@ -82,13 +100,24 @@ def test_accepted_result_wakes_the_lead_exactly_once(mode): # noqa: F811 assert calls[0]["client_turn_id"] == "wake-" + INTENT_ID[:32] assert calls[0]["loopx_request"]["operation"] == "wake" assert calls[0]["loopx_request"]["wake"]["intent_id"] == INTENT_ID + # Dispatch is accepted; the receipt stays pending until the start fact lands. receipt = _wake(path) - assert receipt["state"] == "woken" and receipt["created"] is True - assert receipt["session_id"] == sid and changed == [receipt] + assert receipt["state"] == "pending" and changed == [receipt] + assert _pump(service, repo) == [] and _wake(path) == receipt + + # Once the worker's start lands, the next tick records it without a Turn. + _start_worker_turn(service, sid) + changed = _pump(service, repo) + woken = _wake(path) + assert woken["state"] == "woken" and woken["created"] is False + assert woken["session_id"] == sid and changed == [woken] # A second tick finds no pending intent: no second Turn, receipt unchanged. assert _pump(service, repo) == [] - assert len(calls) == 1 and _wake(path) == receipt + assert _wake(path) == woken + # Every attempt used the one stable client identity, so one Turn resulted. + assert {call["client_turn_id"] for call in calls} == {"wake-" + INTENT_ID[:32]} + assert len(_wake_turns(service, sid)) == 1 def test_lost_receipt_after_the_turn_started_records_it_once(mode): # noqa: F811 @@ -225,12 +254,20 @@ def test_only_an_accepted_result_can_wake(mode, status): # noqa: F811 # so no model runs; "dispatch" below is a request to start the worker. class Transport: - """Adapter/worker boundary with injectable faults.""" + """Adapter/worker boundary with injectable faults. - def __init__(self, service, *, adapter_failures=0, start_turns=False): + ``start_worker`` drives the real ``_run_turn`` body when the wake Turn can + start, so the worker's durable start fact, its single-flight release and + the native acceptance recovery path are the shipped ones; only the adapter + and the thread are replaced. No model runs. + """ + + def __init__(self, service, *, adapter_failures=0, start_write_failures=0, + real_worker=False): self.service = service self.adapter_failures = adapter_failures - self.start_turns = start_turns + self.start_write_failures = start_write_failures + self.real_worker = real_worker self.adapter_attempts = [] self.dispatches = [] @@ -238,15 +275,70 @@ def ensure_adapter(self, session, **kwargs): self.adapter_attempts.append(session["session_id"]) if len(self.adapter_attempts) <= self.adapter_failures: raise RuntimeError("adapter initialization failed") - return object() - - def start_worker(self, *, session_id, turn_id, **kwargs): + return StubAdapter(self.service, session) + + def arm(self, monkeypatch, controller): + """Bind this transport to ``controller``, including the start-fact fault.""" + store = self.service.store + real_update_turn = store.update_turn + + def update_turn(*args, **changes): + # Exactly the worker's first durable step, on its first attempt. + if self.start_write_failures and changes.get("status") == "starting" \ + and changes.get("started_at"): + self.start_write_failures -= 1 + raise OSError("chat turn store unavailable") + return real_update_turn(*args, **changes) + + monkeypatch.setattr(store, "update_turn", update_turn) + monkeypatch.setattr(controller, "_ensure_adapter_locked", self.ensure_adapter) + monkeypatch.setattr(controller, "_start_accepted_turn_worker", self.start_worker) + return self + + def start_worker(self, *, session_id, turn_id, message="", attachments=None, + adapter=None, loopx_execution=False, **kwargs): self.dispatches.append((session_id, turn_id)) - if self.start_turns: # the worker's first durable step + if not self.real_worker: # the worker's first durable step only self.service.store.update_turn( session_id, turn_id, expected_statuses={"queued"}, status="starting", started_at="2026-09-30T00:00:00Z", ) + return True + if (self.service.store.load_turn(session_id, turn_id) or {}).get("status") != "queued": + return False + # The shipped worker body, over a stub adapter: the failure is after the + # durable start write, which is the fact under test. + self.service.controller._run_turn( + session_id=session_id, turn_id=turn_id, message=message, + attachments=attachments or [], adapter=adapter, loopx_execution=loopx_execution, + ) + return True + + +class StubAdapter: + """Enough adapter for the real worker to reach a bounded, non-model failure.""" + + upstream_thread_id = "stub-upstream" + + def __init__(self, service, session): + self.service = service + self.session = session + self.goal_driver = None + self.team_plan_context = None + + def capabilities(self): + return {} + + def start_turn(self, message, event_sink): # pragma: no cover - never reached + raise AssertionError("the stub adapter must not start a native turn") + + def interrupt_turn(self, turn_id=None): + return None + + def close_session(self): + return None + + def healthcheck(self): return True @@ -260,11 +352,9 @@ def _native(mode, monkeypatch, **faults): # noqa: F811 store.update_session(sid, active_turn_id=None, status="ready", loopx_tools=True, native_goal={"status": "paused", "tokensUsed": 0}) controller = service.controller - transport = Transport(service, **faults) + transport = Transport(service, **faults).arm(monkeypatch, controller) monkeypatch.setattr(controller, "submit_turn", ChatRuntimeController.submit_turn.__get__(controller)) - monkeypatch.setattr(controller, "_ensure_adapter_locked", transport.ensure_adapter) - monkeypatch.setattr(controller, "_start_accepted_turn_worker", transport.start_worker) return service, sid, repo, transport @@ -292,11 +382,12 @@ def _second_lead(service, sid, settings): def _fail_next_wake_receipt(monkeypatch): + """Lose the next wake receipt write, whatever state it settles to.""" real = collaboration_mcp._write failed = [] def write(path, row): - if not failed and (row.get("wake") or {}).get("state") == "woken": + if not failed and (row.get("wake") or {}).get("state") in {"woken", "pending"}: failed.append(path) raise OSError("receipt write lost") return real(path, row) @@ -320,9 +411,14 @@ def test_adapter_failure_after_acceptance_is_redispatched_not_recorded(mode, mon assert len(transport.adapter_attempts) == 2 assert transport.dispatches == [(sid, queued["turn_id"])] assert [row["turn_id"] for row in _wake_turns(service, sid)] == [queued["turn_id"]] + assert _wake(path)["state"] == "pending" + + # Only after the start fact is durable does the next tick record the wake. + _pump(service, repo) + receipt = _wake(path) assert receipt["state"] == "woken" and receipt["turn_id"] == queued["turn_id"] - assert receipt["created"] is False + assert receipt["created"] is False and transport.dispatches == [(sid, queued["turn_id"])] def test_failure_after_the_prepared_capsule_is_repaired_and_dispatched(mode, monkeypatch): # noqa: F811 @@ -348,27 +444,46 @@ def append_message(*args, **kwargs): [settled] = _wake_turns(service, sid) assert settled["turn_id"] == prepared["turn_id"] and "_acceptance" not in settled assert transport.dispatches == [(sid, prepared["turn_id"])] + assert _wake(path)["state"] == "pending" + + _pump(service, repo) + assert _wake(path)["state"] == "woken" + assert transport.dispatches == [(sid, prepared["turn_id"])] @pytest.mark.parametrize("started", [False, True]) def test_lost_receipt_replays_only_the_original_turn(mode, monkeypatch, started): # noqa: F811 - service, sid, repo, transport = _native(mode, monkeypatch, start_turns=started) + # ``started=False`` fails the start write itself, before any receipt exists; + # the native owner replays that same Turn. ``started=True`` loses only the + # receipt, after the start fact landed. Both must start it exactly once. + service, sid, repo, transport = _native( + mode, monkeypatch, start_write_failures=0 if started else 1) path = _write_record(service, session_id=sid) lost = _fail_next_wake_receipt(monkeypatch) - assert _pump(service, repo) == [] and lost + assert _pump(service, repo) == [] assert _wake(path)["state"] == "pending" and len(transport.dispatches) == 1 + # The receipt is the write that was lost, when the start fact survived. + assert bool(lost) is started + [queued] = _wake_turns(service, sid) + assert (queued["started_at"] is not None) is started + # Replay through native dispatch: the still-queued case dispatches once + # more for that same Turn; the started case does not dispatch again. + _pump(service, repo) + # A started Turn is recorded without another dispatch. _pump(service, repo) [turn] = _wake_turns(service, sid) + assert turn["turn_id"] == queued["turn_id"] and turn["started_at"] receipt = _wake(path) assert receipt["state"] == "woken" and receipt["turn_id"] == turn["turn_id"] - # A started Turn is recorded without another dispatch; a still-queued one - # is handed to the native owner again for the same Turn. expected = 1 if started else 2 assert transport.dispatches == [(sid, turn["turn_id"])] * expected + # Both paths start the Turn exactly once and never mint a second one. + assert _pump(service, repo) == [] and _wake(path) == receipt + assert len(_wake_turns(service, sid)) == 1 def test_wake_turn_cancelled_before_it_started_is_not_woken(mode, monkeypatch): # noqa: F811 @@ -383,7 +498,7 @@ def test_wake_turn_cancelled_before_it_started_is_not_woken(mode, monkeypatch): _pump(service, repo) receipt = _wake(path) - assert receipt["state"] == "refused" and receipt["reason"] == "wake_turn_not_started" + assert receipt["state"] == "refused" and receipt["reason"] == "wake_turn_ended_unstarted" assert transport.dispatches == [] @@ -425,8 +540,83 @@ def test_closed_origin_conversation_is_refused_without_a_substitute(mode, monkey assert transport.dispatches == [] and _wake_turns(service, other) == [] +def _rebuilt(mode, monkeypatch, **faults): # noqa: F811 + """A fresh real controller over the same persisted store, as after a restart.""" + service = mode[0] + controller = ChatRuntimeController( + store=service.store, + codex_bin="codex", + registry_path=service.controller.registry_path, + ) + transport = Transport(service, **faults).arm(monkeypatch, controller) + monkeypatch.setattr(controller, "submit_turn", + ChatRuntimeController.submit_turn.__get__(controller)) + return controller, transport + + +def test_start_fact_write_failure_is_not_a_wake(mode, monkeypatch): # noqa: F811 + """An accepted Turn whose start never persisted is pending, never woken. + + ``submit_turn`` returning proves admission and a queued Turn; only the + worker's durable ``started_at`` proves dispatch. The first write of that + fact fails here, so the intent must stay recoverable. + """ + service, sid, repo, transport = _native( + mode, monkeypatch, real_worker=True, start_write_failures=1) + path = _write_record(service, session_id=sid) + + # The worker's start write fails: no dispatch fact, so no terminal receipt. + assert _pump(service, repo) == [] + [queued] = _wake_turns(service, sid) + assert queued["status"] == "queued" and queued["started_at"] is None + assert _wake(path)["state"] == "pending" and "woken_at" not in _wake(path) + # The failed start released the worker single-flight guard for this Turn. + assert (sid, queued["turn_id"]) not in service.controller.turn_done_events + + # The next tick replays that same Turn through native dispatch; it starts + # once, and the wake is still not terminal on the same tick's return. + _pump(service, repo) + [started] = _wake_turns(service, sid) + assert started["turn_id"] == queued["turn_id"] and started["started_at"] + assert _wake(path)["state"] == "pending" + + # A later tick reads the durable start fact back: one receipt, no new Turn. + changed = _pump(service, repo) + receipt = _wake(path) + assert receipt["state"] == "woken" and receipt["turn_id"] == started["turn_id"] + assert receipt["created"] is False and changed == [receipt] + assert [row["turn_id"] for row in _wake_turns(service, sid)] == [started["turn_id"]] + # Settled: another tick is a no-op. + assert _pump(service, repo) == [] and _wake(path) == receipt + + +def test_start_fact_write_failure_recovers_after_a_rebuilt_controller(mode, monkeypatch): # noqa: F811 + """A restart rediscovers the same intent and starts the same Turn once.""" + service, sid, repo, transport = _native( + mode, monkeypatch, real_worker=True, start_write_failures=1) + path = _write_record(service, session_id=sid) + + assert _pump(service, repo) == [] + [queued] = _wake_turns(service, sid) + assert queued["status"] == "queued" and _wake(path)["state"] == "pending" + + # A fresh controller over the same persisted store, as after a restart. + controller, rebuilt = _rebuilt(mode, monkeypatch, real_worker=True) + _pump_with(controller, repo) + + [started] = _wake_turns(service, sid) + assert started["turn_id"] == queued["turn_id"] and started["started_at"] + assert rebuilt.dispatches == [(sid, queued["turn_id"])] + + changed = _pump_with(controller, repo) + receipt = _wake(path) + assert receipt["state"] == "woken" and receipt["session_id"] == sid + assert receipt["turn_id"] == started["turn_id"] and changed == [receipt] + assert _pump_with(controller, repo) == [] + + def test_lost_receipt_then_new_conversation_cannot_dispatch_twice(mode, monkeypatch): # noqa: F811 - service, sid, repo, transport = _native(mode, monkeypatch, start_turns=True) + service, sid, repo, transport = _native(mode, monkeypatch) path = _write_record(service, session_id=sid) _fail_next_wake_receipt(monkeypatch) _pump(service, repo) From 4b5b3d44ba7b402ff7bd84245a19ccce82d1e20e Mon Sep 17 00:00:00 2001 From: song Date: Wed, 30 Sep 2026 10:06:56 -0400 Subject: [PATCH 20/22] fix(collaboration): prove a wake dispatch with the provider start fact MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit The planner read `started_at` as dispatch evidence, but the worker stamps that field before it builds the turn context, prepares LoopX mode and hands the message to the adapter. A Turn that failed in that window — or whose process exited before the provider call — was therefore recorded as `woken` although the provider never accepted it. A terminal receipt is never rescanned, so the intent was lost and the lead was never continued. Dispatch is now proven by `upstream_turn_id`, which the runtime writes when the provider reports `turn.started` and checkpoints immediately. A Turn carried past its start write but not dispatched stays `wake_dispatch_pending`; one that ended without dispatch is `refused` / `wake_turn_ended_unstarted`. The Python owner passes the provider fact through and no longer records a terminal `woken` of its own after `submit_turn` returns, so the typed decision is the only writer of that state. The public reference now lists the reasons the planner actually produces — `wake_dispatch_pending` added, `wake_turn_not_started` replaced by `wake_turn_ended_unstarted` — in both languages, and two contract tests pin the vocabulary to the planner so a rename cannot drift from the docs again. `chat_runtime.py` had grown past its reviewed `any_count` ceiling. The growth was the attachment payload annotation on the extracted start body, so the payload is named once as `AttachmentPayload` and the four call sites share it; the module is now below the ceiling rather than above it, and the extracted body no longer takes a `done_event` its caller already owns. Coverage: a worker start with no provider dispatch stays pending across a tick and a rebuilt controller, and only the provider's own fact wakes it once; a Turn that failed before dispatch is refused and mints no second Turn. Signed-off-by: song --- docs/reference/goal-chat-continuation.md | 31 ++-- loopx/chat_loopx_mode.py | 3 + loopx/chat_runtime.py | 23 ++- .../control_plane/collaboration/chat_mode.ts | 29 ++- tests/control_plane_ts/chat_mode.test.ts | 22 ++- tests/test_chat_delegation_wake.py | 174 ++++++++++++++++-- 6 files changed, 233 insertions(+), 49 deletions(-) diff --git a/docs/reference/goal-chat-continuation.md b/docs/reference/goal-chat-continuation.md index f4c9e6be9a..0c3b775d1a 100644 --- a/docs/reference/goal-chat-continuation.md +++ b/docs/reference/goal-chat-continuation.md @@ -104,14 +104,16 @@ messages; images use the ordinary conversation after pausing. same Goal and coordinator is never selected instead, so exiting, closing or reconfiguring that conversation refuses the wake rather than moving it. The operation record keeps a separate wake receipt: `woken` with the Turn id, - `pending` with `lead_turn_active`, `lead_paused` or `allowance_exhausted`, or - `refused` with `goal_stopped`, `lead_unbound`, `binding_revoked`, - `native_goal_complete`, `native_goal_absent`, `wake_identity_conflict`, - `wake_turn_not_started` or `no_wake_owner`. Its Turn is accepted once under a - client id derived from the intent, and only a Turn that started is recorded as - dispatched: a still-queued Turn from an interrupted dispatch is handed back to - the same native acceptance and dispatch owner, so the lead is continued once - without creating a second Turn. A wake never unpauses the lead, never starts a + `pending` with `lead_turn_active`, `lead_paused`, `allowance_exhausted` or + `wake_dispatch_pending`, or `refused` with `goal_stopped`, `lead_unbound`, + `binding_revoked`, `native_goal_complete`, `native_goal_absent`, + `wake_identity_conflict`, `wake_turn_ended_unstarted` or `no_wake_owner`. Its + Turn is accepted once under a client id derived from the intent, and only a + Turn the provider actually started — recorded as `upstream_turn_id` — is + recorded as dispatched: a Turn accepted but not yet dispatched stays + `wake_dispatch_pending`, and a still-queued Turn from an interrupted dispatch + is handed back to the same native acceptance and dispatch owner, so the lead is + continued once without creating a second Turn. A wake never unpauses the lead, never starts a native Goal and never raises the allowance. A lead that already read the result in its own Turn is marked `observed_in_turn` and is not woken again. It requires the Chat service to be running; stopping it leaves intents pending. @@ -160,12 +162,13 @@ For a disposable mixed-team setup, use the 成员结果首次被接受时,Chat 服务按与 `resume` 相同的准入规则继续协调员一次。 唤醒绑定在「启动该操作的回合所属会话」上:同一 Goal 与身份下的其他会话不会被选为 替代,所以该会话退出、关闭或改配置时只做拒绝,而不是把唤醒移交给别的会话。操作记录 -里另存唤醒回执:`woken` 附 Turn id;`pending` 附 `lead_turn_active`、`lead_paused` -或 `allowance_exhausted`;`refused` 附 `goal_stopped`、`lead_unbound`、`binding_revoked`、 -`native_goal_complete`、`native_goal_absent`、`wake_identity_conflict`、`wake_turn_not_started` -或 `no_wake_owner`。其 Turn 由意图派生的 client id 只接受一次,且只有真正启动过的 -Turn 才算已派发:中断派发后仍处于 queued 的 Turn 会交回同一套原生接受与派发 owner 继续, -因此协调员只被续跑一次,也不会新建第二个 Turn。唤醒不会解除暂停、不会新开原生 Goal、 +里另存唤醒回执:`woken` 附 Turn id;`pending` 附 `lead_turn_active`、`lead_paused`、 +`allowance_exhausted` 或 `wake_dispatch_pending`;`refused` 附 `goal_stopped`、`lead_unbound`、 +`binding_revoked`、`native_goal_complete`、`native_goal_absent`、`wake_identity_conflict`、 +`wake_turn_ended_unstarted` 或 `no_wake_owner`。其 Turn 由意图派生的 client id 只接受一次, +且只有 provider 真正开始过的 Turn(记为 `upstream_turn_id`)才算已派发:已接受但尚未派发的 +Turn 保持 `wake_dispatch_pending`;中断派发后仍处于 queued 的 Turn 会交回同一套原生接受与 +派发 owner 继续,因此协调员只被续跑一次,也不会新建第二个 Turn。唤醒不会解除暂停、不会新开原生 Goal、 也不会提高额度。协调员已在自己回合内读到结果时记为 `observed_in_turn`,不再唤醒。 它依赖 Chat 服务在运行;停掉服务时意图保持待处理。 diff --git a/loopx/chat_loopx_mode.py b/loopx/chat_loopx_mode.py index a3ba5f5581..bd966d6760 100644 --- a/loopx/chat_loopx_mode.py +++ b/loopx/chat_loopx_mode.py @@ -501,6 +501,9 @@ def settled(state, reason, **facts): wake_turn = { "turn_id": existing.get("turn_id"), "status": existing.get("status"), + # The provider-start fact. `started_at` is stamped before the + # provider is reached, so it cannot prove a dispatch. + "upstream_turn_id": existing.get("upstream_turn_id"), "started_at": existing.get("started_at"), "loopx_execution": existing.get("loopx_execution") is True, "operation": request.get("operation"), diff --git a/loopx/chat_runtime.py b/loopx/chat_runtime.py index 08363ac572..8e6f102701 100644 --- a/loopx/chat_runtime.py +++ b/loopx/chat_runtime.py @@ -75,6 +75,11 @@ "live_steering_turn_not_started", }) +# The opaque payload an adapter forwards to its provider. Named here so the +# worker signature that carries it between methods does not open another +# module-level `Any`. +AttachmentPayload = dict[str, Any] + class ChatTurnAcceptanceUnavailableError(Exception): """The durable acceptance attempt can be retried with the same request.""" @@ -159,7 +164,7 @@ def start_turn_with_attachments( self, message: str, event_sink: EventSink, - attachments: list[dict[str, Any]], + attachments: list[AttachmentPayload], ) -> dict[str, Any]: return self.session.send(message, attachments=attachments, on_event=event_sink) @@ -939,7 +944,7 @@ def submit_turn( session_id: str, client_turn_id: str, message: str, - attachments: list[dict[str, Any]] | None = None, + attachments: list[AttachmentPayload] | None = None, work_dir: Path, objective: str, loopx_execution: bool = False, @@ -1024,7 +1029,7 @@ def _start_accepted_turn_worker( session_id: str, turn_id: str, message: str, - attachments: list[dict[str, Any]], + attachments: list[AttachmentPayload], adapter: ChatRuntimeAdapter, loopx_execution: bool, ) -> bool: @@ -1385,7 +1390,7 @@ def _run_turn( session_id: str, turn_id: str, message: str, - attachments: list[dict[str, Any]], + attachments: list[AttachmentPayload], adapter: ChatRuntimeAdapter, done_event: threading.Event | None = None, loopx_execution: bool = False, @@ -1421,7 +1426,6 @@ def _run_turn( attachments=attachments, adapter=adapter, loopx_execution=loopx_execution, - done_event=done_event, ) finally: done_event.set() @@ -1435,12 +1439,15 @@ def _run_started_turn( session_id: str, turn_id: str, message: str, - attachments: list[dict[str, Any]], + attachments: list[AttachmentPayload], adapter: ChatRuntimeAdapter, loopx_execution: bool, - done_event: threading.Event, ) -> None: - """The body of a Turn whose `queued -> starting` fact is already durable.""" + """The body of a Turn whose `queued -> starting` fact is already durable. + + `_run_turn` owns the single-flight release, so nothing here needs the + done event. + """ key = (session_id, turn_id) event_buffer = _TurnEventBuffer( store=self.store, diff --git a/loopx/control_plane/collaboration/chat_mode.ts b/loopx/control_plane/collaboration/chat_mode.ts index b988259c18..826037256b 100644 --- a/loopx/control_plane/collaboration/chat_mode.ts +++ b/loopx/control_plane/collaboration/chat_mode.ts @@ -61,13 +61,15 @@ const RESUMABLE_NATIVE = ["paused", "blocked", "usageLimited", "budgetLimited"]; * a closed or exited origin is refused rather than handed over implicitly. * * ``wake_turn`` is the Turn already accepted under this intent's client id in - * that conversation, if any. Only a started Turn is dispatch evidence: it is - * recorded as ``woken`` without another dispatch. A still-queued Turn is not; - * it is replayed through the native acceptance owner under the same admission - * as a new wake, so pause and revocation still hold. ``started_at`` is the - * durable start fact, so a Turn still activating stays pending rather than - * being read as a start. A Turn that ended before it started is refused, - * since its client id cannot admit another Turn. + * that conversation, if any. Only a dispatched Turn is dispatch evidence: it + * is recorded as ``woken`` without another dispatch. A still-queued Turn is + * not; it is replayed through the native acceptance owner under the same + * admission as a new wake, so pause and revocation still hold. Dispatch is + * proven by ``upstream_turn_id``, which the runtime writes when the provider + * reports ``turn.started`` — not by ``started_at``, which the worker stamps + * before it builds the context and reaches the provider, so a Turn that is + * merely activating stays pending. A Turn that ended without dispatch is + * refused, since its client id cannot admit another Turn. * * The same facts that admit an owner resume admit a host wake, plus mode * enabled and not paused. A refusal is terminal for that intent; pending @@ -91,12 +93,19 @@ function planDelegationWake(input: JsonObject, session: JsonObject, settings: Js if (turn.loopx_execution !== true || turn.operation !== "wake" || turn.intent_id !== intent.intent_id) { return outcome("refused", "wake_identity_conflict"); } - if (typeof turn.started_at === "string" && turn.started_at) { + // `upstream_turn_id` is written when the provider reports `turn.started`, + // so it is evidence that a Turn was actually dispatched. `started_at` is + // earlier than that: the worker stamps it before it builds the turn + // context, prepares LoopX mode and hands the message to the adapter, so + // recording `woken` on it would claim a dispatch that may never happen and + // lose the intent, since a terminal receipt is never rescanned. + if (typeof turn.upstream_turn_id === "string" && turn.upstream_turn_id) { return {operation: "wake", state: "woken", reason: null, dispatch: "recorded"}; } - // Ended without ever starting: its client id cannot admit another Turn. + // Ended without ever being dispatched: its client id cannot admit another + // Turn, but the owner must be told rather than left waiting. if (isTerminalTurnStatus(turn.status)) return outcome("refused", "wake_turn_ended_unstarted"); - // Accepted but not yet started, including a start still activating. + // Accepted but not yet dispatched, including a start still activating. if (turn.status !== "queued") return outcome("pending", "wake_dispatch_pending"); } if (input.goal_active !== true) return outcome("refused", "goal_stopped"); diff --git a/tests/control_plane_ts/chat_mode.test.ts b/tests/control_plane_ts/chat_mode.test.ts index 3332fb9ad0..5892700680 100644 --- a/tests/control_plane_ts/chat_mode.test.ts +++ b/tests/control_plane_ts/chat_mode.test.ts @@ -98,7 +98,7 @@ test("a host wake reuses the resume facts and returns a typed outcome, never an assert.throws(() => planChatMode({...wake, session: {...enabled, channel_id: "manager"}})); }); -test("only a started wake Turn is dispatch evidence; a queued one is replayed under the same admission", () => { +test("only a provider-dispatched wake Turn is dispatch evidence; a queued or merely starting one is replayed", () => { const own = {turn_id: "wake-turn", loopx_execution: true, operation: "wake", intent_id: intent.intent_id}; const queued = {...own, status: "queued", started_at: null}; // Queued, not started: replay the same Turn; its own active id does not block it. @@ -116,25 +116,35 @@ test("only a started wake Turn is dispatch evidence; a queued one is replayed un for (const [changes, state, reason] of held) { assert.deepEqual(planChatMode({...wake, wake_turn: queued, ...changes}), {operation: "wake", state, reason}, reason); } - // The durable start fact, in any later status and even after the mode changed. + // The provider-start fact, in any later status and even after the mode changed. for (const status of ["starting", "running", "completed", "failed"]) { - assert.deepEqual(planChatMode({...wake, wake_turn: {...own, status, started_at: "2026-09-30T00:00:00Z"}, + assert.deepEqual(planChatMode({...wake, wake_turn: {...own, status, upstream_turn_id: "upstream-1"}, session: {...enabled, loopx_mode: {enabled: false}}}), {operation: "wake", state: "woken", reason: null, dispatch: "recorded"}, status); } - // Ended before it started: its client id cannot admit another Turn. + // `started_at` lands before the provider is reached, so it is not dispatch + // evidence: a Turn carrying it but no `upstream_turn_id` stays pending rather + // than claiming a wake the provider never accepted. + for (const status of ["starting", "failed", "completed"]) { + assert.deepEqual(planChatMode({...wake, wake_turn: {...own, status, started_at: "2026-09-30T00:00:00Z"}, + session: {...enabled, loopx_mode: {enabled: false}}}), + status === "starting" + ? {operation: "wake", state: "pending", reason: "wake_dispatch_pending"} + : {operation: "wake", state: "refused", reason: "wake_turn_ended_unstarted"}, status); + } + // Ended without a provider dispatch: its client id cannot admit another Turn. for (const status of ["interrupted", "failed", "timed_out", "completed"]) { assert.deepEqual(planChatMode({...wake, wake_turn: {...own, status, started_at: null}}), {operation: "wake", state: "refused", reason: "wake_turn_ended_unstarted"}, status); } - // Accepted but not yet started: still pending, so the next tick re-reads the fact. + // Accepted but not yet dispatched: still pending, so the next tick re-reads the fact. for (const status of ["interrupting", "completing"]) { assert.deepEqual(planChatMode({...wake, wake_turn: {...own, status, started_at: null}}), {operation: "wake", state: "pending", reason: "wake_dispatch_pending"}, status); } // A client id owned by another request is never claimed. for (const other of [{loopx_execution: false}, {operation: "resume"}, {intent_id: "b".repeat(64)}]) { - assert.deepEqual(planChatMode({...wake, wake_turn: {...queued, ...other, started_at: "2026-09-30T00:00:00Z"}}), + assert.deepEqual(planChatMode({...wake, wake_turn: {...queued, ...other, upstream_turn_id: "upstream-1"}}), {operation: "wake", state: "refused", reason: "wake_identity_conflict"}); } // An intent without its origin conversation cannot be decided by any conversation. diff --git a/tests/test_chat_delegation_wake.py b/tests/test_chat_delegation_wake.py index 5c441a4178..d841993665 100644 --- a/tests/test_chat_delegation_wake.py +++ b/tests/test_chat_delegation_wake.py @@ -7,6 +7,7 @@ """ import json +from pathlib import Path import pytest @@ -79,7 +80,9 @@ def _start_worker_turn(service, sid): """The worker's first durable step, for the stubbed ``submit_turn`` fixture. The shared fixture's ``submit_turn`` records acceptance only, exactly as the - real one returns before the worker writes ``started_at``. + real one returns before the worker writes ``started_at``. This is *not* + dispatch evidence: the real worker stamps it before it builds the turn + context and reaches the provider. """ turn = service.store.turn_for_client(sid, "wake-" + INTENT_ID[:32]) service.store.update_turn( @@ -89,6 +92,21 @@ def _start_worker_turn(service, sid): return turn +def _dispatch_worker_turn(service, sid): + """The provider-start fact, as ``turn.started`` records it. + + The runtime writes ``status=running`` and ``upstream_turn_id`` when the + provider reports the Turn actually started, and checkpoints immediately. + Only this proves a dispatch. + """ + turn = _start_worker_turn(service, sid) + service.store.update_turn( + sid, turn["turn_id"], expected_statuses={"starting"}, + status="running", upstream_turn_id="upstream-1", + ) + return turn + + def test_accepted_result_wakes_the_lead_exactly_once(mode): # noqa: F811 service, sid, calls = _idle_resumable_lead(mode) repo = mode[2] @@ -105,8 +123,14 @@ def test_accepted_result_wakes_the_lead_exactly_once(mode): # noqa: F811 assert receipt["state"] == "pending" and changed == [receipt] assert _pump(service, repo) == [] and _wake(path) == receipt - # Once the worker's start lands, the next tick records it without a Turn. + # The worker's first durable step is not a dispatch: the receipt stays + # pending, because the provider has not been reached yet. _start_worker_turn(service, sid) + assert _pump(service, repo) == [] and _wake(path) == receipt + + # Once the provider reports the start, the next tick records it without a + # second Turn. + _dispatch_worker_turn(service, sid) changed = _pump(service, repo) woken = _wake(path) assert woken["state"] == "woken" and woken["created"] is False @@ -126,10 +150,11 @@ def test_lost_receipt_after_the_turn_started_records_it_once(mode): # noqa: F81 turn, _ = service.store.create_turn( sid, client_turn_id="wake-" + INTENT_ID[:32], message="/goal resume" ) - # A started Turn is dispatch evidence; a merely persisted one is not. + # A dispatched Turn is dispatch evidence; a merely persisted one is not. service.store.update_turn( sid, turn["turn_id"], loopx_execution=True, status="completed", started_at="2026-09-30T00:00:00Z", completed_at="2026-09-30T00:01:00Z", + upstream_turn_id="upstream-1", loopx_request={"operation": "wake", "wake": {"intent_id": INTENT_ID}}, ) service.store.update_session(sid, active_turn_id=None) @@ -263,11 +288,14 @@ class Transport: """ def __init__(self, service, *, adapter_failures=0, start_write_failures=0, - real_worker=False): + real_worker=False, dispatch_provider=True): self.service = service self.adapter_failures = adapter_failures self.start_write_failures = start_write_failures self.real_worker = real_worker + # Whether the stubbed provider reports ``turn.started``. ``False`` models + # a Turn whose worker started but never reached a provider. + self.dispatch_provider = dispatch_provider self.adapter_attempts = [] self.dispatches = [] @@ -298,11 +326,17 @@ def update_turn(*args, **changes): def start_worker(self, *, session_id, turn_id, message="", attachments=None, adapter=None, loopx_execution=False, **kwargs): self.dispatches.append((session_id, turn_id)) - if not self.real_worker: # the worker's first durable step only + if not self.real_worker: # the worker's start, then the provider's self.service.store.update_turn( session_id, turn_id, expected_statuses={"queued"}, status="starting", started_at="2026-09-30T00:00:00Z", ) + if self.dispatch_provider: + # ``turn.started``: the runtime's provider-start fact. + self.service.store.update_turn( + session_id, turn_id, expected_statuses={"starting"}, + status="running", upstream_turn_id="upstream-1", + ) return True if (self.service.store.load_turn(session_id, turn_id) or {}).get("status") != "queued": return False @@ -486,6 +520,83 @@ def test_lost_receipt_replays_only_the_original_turn(mode, monkeypatch, started) assert len(_wake_turns(service, sid)) == 1 +def test_a_worker_start_without_a_provider_dispatch_is_not_a_wake(mode, monkeypatch): # noqa: F811 + """The pre-dispatch window: `started_at` lands, the provider is never reached. + + The worker stamps `started_at` before it builds the turn context, prepares + LoopX mode and hands the message to the adapter. Reading that as dispatch + evidence would record `woken` for a Turn the provider never accepted — and + because a terminal receipt is never rescanned, the intent would be lost. + """ + service, sid, repo, transport = _native( + mode, monkeypatch, dispatch_provider=False) + path = _write_record(service, session_id=sid) + + # Admission is pending; the worker stamps its pre-dispatch start fact and + # the provider is never reached, so no upstream identity is recorded. + _pump(service, repo) + [queued] = _wake_turns(service, sid) + assert queued["status"] == "starting" and queued["started_at"] + assert queued["upstream_turn_id"] is None + # No provider accepted this Turn, so the intent stays open. + receipt = _wake(path) + assert receipt["state"] == "pending" and receipt["reason"] == "wake_dispatch_pending" + assert "woken_at" not in receipt + + # A restart rediscovers it and still does not claim a wake. + controller, rebuilt = _rebuilt(mode, monkeypatch, dispatch_provider=False) + assert _pump_with(controller, repo) == [] + assert _wake(path)["state"] == "pending" + assert rebuilt.dispatches == [] + + # Only the provider's own start fact turns it into a wake, exactly once. + service.store.update_turn( + sid, queued["turn_id"], status="running", upstream_turn_id="upstream-1", + ) + changed = _pump_with(controller, repo) + receipt = _wake(path) + assert receipt["state"] == "woken" and receipt["turn_id"] == queued["turn_id"] + assert receipt["created"] is False and changed == [receipt] + assert len(_wake_turns(service, sid)) == 1 + assert _pump_with(controller, repo) == [] and _wake(path) == receipt + + +def test_a_pre_provider_failure_does_not_claim_a_wake(mode, monkeypatch): # noqa: F811 + """A Turn that fails after its start write but before dispatch is refused. + + The real error path marks the Turn `failed` with no `turn.started` event, so + no provider accepted it. Its client id cannot admit another Turn, and the + intent must end in an actionable refusal rather than claiming a wake. + """ + service, sid, repo, transport = _native(mode, monkeypatch, dispatch_provider=False) + path = _write_record(service, session_id=sid) + + _pump(service, repo) + [turn] = _wake_turns(service, sid) + # The worker wrote its start fact; the provider call then failed. + service.store.update_turn( + sid, turn["turn_id"], status="starting", started_at="2026-09-30T00:00:00Z", + ) + service.store.update_turn( + sid, turn["turn_id"], status="failed", error_code="adapter_unavailable", + expected_statuses={"starting"}, + ) + [failed] = _wake_turns(service, sid) + assert failed["started_at"] and failed["upstream_turn_id"] is None + + # The next tick re-reads the Turn and refuses: no provider accepted it, so + # its client id cannot admit another one. This is the receipt the previous + # head wrongly wrote as `woken`. + changed = _pump(service, repo) + receipt = _wake(path) + assert receipt["state"] == "refused", receipt + assert receipt["reason"] == "wake_turn_ended_unstarted" + assert "woken_at" not in receipt and changed == [receipt] + # Refused is terminal for this intent and no second Turn was minted. + assert _pump(service, repo) == [] and _wake(path) == receipt + assert len(_wake_turns(service, sid)) == 1 + + def test_wake_turn_cancelled_before_it_started_is_not_woken(mode, monkeypatch): # noqa: F811 service, sid, repo, transport = _native(mode, monkeypatch, adapter_failures=1) path = _write_record(service, session_id=sid) @@ -562,7 +673,7 @@ def test_start_fact_write_failure_is_not_a_wake(mode, monkeypatch): # noqa: F81 fact fails here, so the intent must stay recoverable. """ service, sid, repo, transport = _native( - mode, monkeypatch, real_worker=True, start_write_failures=1) + mode, monkeypatch, start_write_failures=1, dispatch_provider=False) path = _write_record(service, session_id=sid) # The worker's start write fails: no dispatch fact, so no terminal receipt. @@ -573,14 +684,18 @@ def test_start_fact_write_failure_is_not_a_wake(mode, monkeypatch): # noqa: F81 # The failed start released the worker single-flight guard for this Turn. assert (sid, queued["turn_id"]) not in service.controller.turn_done_events - # The next tick replays that same Turn through native dispatch; it starts - # once, and the wake is still not terminal on the same tick's return. + # The next tick replays that same Turn through native dispatch; the worker + # starts it, but the provider's own start fact has not landed yet. _pump(service, repo) [started] = _wake_turns(service, sid) assert started["turn_id"] == queued["turn_id"] and started["started_at"] + assert started["upstream_turn_id"] is None assert _wake(path)["state"] == "pending" - # A later tick reads the durable start fact back: one receipt, no new Turn. + # Only the provider's start fact makes it a wake: one receipt, no new Turn. + service.store.update_turn( + sid, started["turn_id"], status="running", upstream_turn_id="upstream-1", + ) changed = _pump(service, repo) receipt = _wake(path) assert receipt["state"] == "woken" and receipt["turn_id"] == started["turn_id"] @@ -593,7 +708,7 @@ def test_start_fact_write_failure_is_not_a_wake(mode, monkeypatch): # noqa: F81 def test_start_fact_write_failure_recovers_after_a_rebuilt_controller(mode, monkeypatch): # noqa: F811 """A restart rediscovers the same intent and starts the same Turn once.""" service, sid, repo, transport = _native( - mode, monkeypatch, real_worker=True, start_write_failures=1) + mode, monkeypatch, start_write_failures=1, dispatch_provider=False) path = _write_record(service, session_id=sid) assert _pump(service, repo) == [] @@ -601,13 +716,17 @@ def test_start_fact_write_failure_recovers_after_a_rebuilt_controller(mode, monk assert queued["status"] == "queued" and _wake(path)["state"] == "pending" # A fresh controller over the same persisted store, as after a restart. - controller, rebuilt = _rebuilt(mode, monkeypatch, real_worker=True) + controller, rebuilt = _rebuilt(mode, monkeypatch, dispatch_provider=False) _pump_with(controller, repo) [started] = _wake_turns(service, sid) assert started["turn_id"] == queued["turn_id"] and started["started_at"] assert rebuilt.dispatches == [(sid, queued["turn_id"])] + # The provider's own start fact is what turns it into a wake, exactly once. + service.store.update_turn( + sid, started["turn_id"], status="running", upstream_turn_id="upstream-1", + ) changed = _pump_with(controller, repo) receipt = _wake(path) assert receipt["state"] == "woken" and receipt["session_id"] == sid @@ -662,3 +781,36 @@ def start(self, binding_id, operation_id, brief, parent_request_id=None, *, conv "brief": {"schema_version": "collaboration_brief_v0"}}) assert pinned == [{"session_id": sid, "turn_id": turn_id}] + + +# The public reference enumerates the wake receipt vocabulary, and operators read +# those names out of the operation record. A rename in the planner that the docs +# do not follow leaves a machine state nobody can look up, so the two are pinned +# together here instead of by review attention. +WAKE_REASONS = { + "pending": {"lead_turn_active", "lead_paused", "allowance_exhausted", "wake_dispatch_pending"}, + "refused": { + "goal_stopped", "lead_unbound", "binding_revoked", "native_goal_complete", + "native_goal_absent", "wake_identity_conflict", "wake_turn_ended_unstarted", + "no_wake_owner", + }, +} + + +def test_the_documented_wake_vocabulary_matches_the_planner(): + """Every documented reason is produced by the typed owner, and none is stale.""" + source = (Path(__file__).resolve().parents[1] / "loopx/control_plane/collaboration/chat_mode.ts").read_text() + for state, reasons in WAKE_REASONS.items(): + for reason in sorted(reasons): + assert f'outcome("{state}", "{reason}")' in source, f"{reason} is documented but never produced" + + +def test_the_public_reference_lists_every_wake_reason(markdown_path=None): + """The reference names every reason the planner can write, in both languages.""" + reference = (Path(__file__).resolve().parents[1] / "docs/reference/goal-chat-continuation.md").read_text() + for state, reasons in WAKE_REASONS.items(): + for reason in sorted(reasons): + assert reference.count(reason) >= 2, ( + f"{reason} must be listed in the English and Chinese wake-receipt lists" + ) + assert "wake_turn_not_started" not in reference, "the retired reason name is still documented" From 6e5f11117a96fe29e4fa452d07269aadfccc4eea Mon Sep 17 00:00:00 2001 From: song Date: Wed, 30 Sep 2026 14:31:51 -0400 Subject: [PATCH 21/22] fix(collaboration): produce a wake intent only for a conversation-origin result MIME-Version: 1.0 Content-Type: text/plain; charset=UTF-8 Content-Transfer-Encoding: 8bit An ordinary CLI/MCP delegation was gaining wake state it had no use for. The typed accepted transition produced an intent for every first acceptance and let it carry a null conversation, so a plain `wait`/`read` grew a `wake` field, the record gained a persisted pending intent, and the runtime pump rewrote it to refused/no_wake_owner — on both File and SQLite authority, for a caller who never created a conversation or enabled this capability. Producing an intent and then refusing it is not opt-in isolation: the shared persistent projection and the public readback changed for everyone. The requester's conversation is the capability's own precondition, so it is now also its producer's: an accepted result leaves a wake intent only when the operation was started from a conversation, which only the trusted Chat host ever supplies. A conversationless acceptance keeps exactly the transition it had — no intent, no wake state, and no change to what a plain read returns — while the enabled path keeps its single delivery, pause, allowance, binding and restart-recovery behaviour. `test_an_ordinary_delegation_gains_no_wake_state` pins the isolation on both authorities: acceptance succeeds, no `wake` and no `conversation` appear on the record. Mutation-checked by restoring the unconditional producer, which fails it. The TS contract states both directions — a conversationless acceptance has no intent, a conversation-origin one is pinned to that conversation — and the forged-digest assertions now travel on the path that actually validates them. Signed-off-by: song --- .../control_plane/collaboration/delegation.ts | 19 +++++++- tests/control_plane_ts/delegation.test.ts | 43 +++++++++++-------- tests/test_local_delegation.py | 17 ++++++++ 3 files changed, 61 insertions(+), 18 deletions(-) diff --git a/loopx/control_plane/collaboration/delegation.ts b/loopx/control_plane/collaboration/delegation.ts index 7b29f678e0..bbcf0396dc 100644 --- a/loopx/control_plane/collaboration/delegation.ts +++ b/loopx/control_plane/collaboration/delegation.ts @@ -338,10 +338,27 @@ export function transitionDelegationObservation(params: JsonObject): JsonObject if (to === "accepted") requireThat(params.canonical_done === true && params.acceptance_ready === true && params.artifacts_current === true, "accepted return requires current canonical completion and artifacts"); - if (to === "accepted" && from !== "accepted") return {status: to, wake_intent: delegationWakeIntent(params)}; + if (to === "accepted" && from !== "accepted" && wakesItsConversation(params)) { + return {status: to, wake_intent: delegationWakeIntent(params)}; + } return {status: to}; } +/** Whether an accepted result may produce a wake intent at all. + * + * Only an operation started from a conversation can be continued there. An + * ordinary CLI/MCP delegation has no conversation, so it keeps the transition it + * always had: no intent, no wake state, and no change to what a plain + * `wait`/`read` returns. Producing an intent and then refusing it in the pump + * would still widen a shared persistent projection for every caller who never + * enabled this capability. + */ +function wakesItsConversation(params: JsonObject): boolean { + if (params.requester == null) return false; + const requester = requireJsonObject(params.requester, "wake requester"); + return requester.conversation != null; +} + /** The first transition to ``accepted`` is the one durable moment a requester * can be continued without polling. The intent names the requester, the * conversation whose Turn started the operation (null when it was not started diff --git a/tests/control_plane_ts/delegation.test.ts b/tests/control_plane_ts/delegation.test.ts index e6c343c9c3..5c6f75828d 100644 --- a/tests/control_plane_ts/delegation.test.ts +++ b/tests/control_plane_ts/delegation.test.ts @@ -106,42 +106,51 @@ test("message receipt and model return do not imply accepted work", () => { assert.equal(accepted.status, "accepted"); }); -test("only the first transition to accepted leaves a wake intent for the exact requester and result", () => { - const accepted = transitionDelegationObservation({from: "turn_returned", to: "accepted", ...acceptedFacts}); +test("only a conversation-origin accepted result leaves a wake intent for the exact requester", () => { + // An ordinary CLI/MCP delegation has no conversation. It must keep the + // transition it always had: no intent, no wake state, no widened readback. + const conversationless = transitionDelegationObservation({from: "turn_returned", to: "accepted", ...acceptedFacts}); + assert.deepEqual(conversationless, {status: "accepted"}); + assert.equal(conversationless.wake_intent, undefined); + + // With an originating conversation the intent is produced and pinned to it. + const accepted = transitionDelegationObservation({from: "turn_returned", to: "accepted", ...acceptedFacts, + requester: {...acceptedRequester, conversation: {session_id: "s-1", turn_id: "t-1", extra: "dropped"}}}); const intent = accepted.wake_intent as Record; assert.equal(intent.schema_version, "loopx_delegation_wake_intent_v0"); assert.match(String(intent.intent_id), /^[a-f0-9]{64}$/); assert.deepEqual(intent.requester, {goal_id: "research", agent_id: "coordinator", goal_ref: null}); assert.equal(intent.operation_id, "analysis-1"); assert.equal(intent.request_id, "req-1"); - assert.equal(intent.conversation, null); - // The originating conversation is pinned into the intent identity. - const pinned = transitionDelegationObservation({from: "turn_returned", to: "accepted", ...acceptedFacts, - requester: {...acceptedRequester, conversation: {session_id: "s-1", turn_id: "t-1", extra: "dropped"}}}); - const pinnedIntent = pinned.wake_intent as Record; - assert.deepEqual(pinnedIntent.conversation, {session_id: "s-1", turn_id: "t-1"}); - assert.notEqual(pinnedIntent.intent_id, intent.intent_id); + assert.deepEqual(intent.conversation, {session_id: "s-1", turn_id: "t-1"}); + // Another conversation of the same requester yields a different identity. + const pinned = accepted; + const pinnedIntent = intent; const elsewhere = transitionDelegationObservation({from: "turn_returned", to: "accepted", ...acceptedFacts, requester: {...acceptedRequester, conversation: {session_id: "s-2", turn_id: "t-1"}}}); assert.notEqual((elsewhere.wake_intent as Record).intent_id, pinnedIntent.intent_id); assert.throws(() => transitionDelegationObservation({from: "turn_returned", to: "accepted", ...acceptedFacts, requester: {...acceptedRequester, conversation: {session_id: "s-1"}}}), /conversation/); // Same requester and result: same intent, so a replayed transition cannot mint a second wake. - assert.deepEqual(transitionDelegationObservation({from: "turn_returned", to: "accepted", ...acceptedFacts}), accepted); + assert.deepEqual(transitionDelegationObservation({from: "turn_returned", to: "accepted", ...acceptedFacts, + requester: {...acceptedRequester, conversation: {session_id: "s-1", turn_id: "t-1"}}}), accepted); const changed = transitionDelegationObservation({from: "turn_returned", to: "accepted", ...acceptedFacts, - requester: {...acceptedRequester, artifacts: [{ref: "output.json", sha256: "b".repeat(64)}]}}); + requester: {...acceptedRequester, conversation: {session_id: "s-1", turn_id: "t-1"}, + artifacts: [{ref: "output.json", sha256: "b".repeat(64)}]}}); assert.notEqual((changed.wake_intent as Record).intent_id, intent.intent_id); // accepted -> accepted is an idempotent readback, never a new wake. assert.deepEqual(transitionDelegationObservation({from: "accepted", to: "accepted", ...acceptedFacts}), {status: "accepted"}); - // A transition to accepted without the requester identity, or with a forged digest, has no wake to record. - assert.throws(() => transitionDelegationObservation({from: "turn_returned", to: "accepted", - canonical_done: true, acceptance_ready: true, artifacts_current: true}), /wake requester/); + // A transition to accepted without any requester identity is not a wake, and a + // conversation-origin one with a forged digest is rejected. + assert.deepEqual(transitionDelegationObservation({from: "turn_returned", to: "accepted", + canonical_done: true, acceptance_ready: true, artifacts_current: true}), {status: "accepted"}); + const conversation = {session_id: "s-1", turn_id: "t-1"}; assert.throws(() => transitionDelegationObservation({from: "turn_returned", to: "accepted", ...acceptedFacts, - requester: {...acceptedRequester, artifacts: [{ref: "output.json", sha256: "short"}]}}), /artifact/); + requester: {...acceptedRequester, conversation, artifacts: [{ref: "output.json", sha256: "short"}]}}), /artifact/); assert.throws(() => transitionDelegationObservation({from: "turn_returned", to: "accepted", ...acceptedFacts, - requester: {...acceptedRequester, artifacts: []}}), /artifacts/); + requester: {...acceptedRequester, conversation, artifacts: []}}), /artifacts/); assert.throws(() => transitionDelegationObservation({from: "turn_returned", to: "accepted", ...acceptedFacts, - requester: {...acceptedRequester, goal_ref: ["not", "a", "reference"]}}), /goal reference/); + requester: {...acceptedRequester, conversation, goal_ref: ["not", "a", "reference"]}}), /goal reference/); // Rejection is terminal and wakes nobody. assert.deepEqual(transitionDelegationObservation({from: "turn_returned", to: "rejected"}), {status: "rejected"}); }); diff --git a/tests/test_local_delegation.py b/tests/test_local_delegation.py index 4e8189835b..8ba1f88aed 100644 --- a/tests/test_local_delegation.py +++ b/tests/test_local_delegation.py @@ -290,3 +290,20 @@ def read_on_publish(path, row, status, **facts): assert len(terminal_reads) == 1 assert not demo.canonical_tasks(root)["todo_analyst-initial"]["done"] assert returns(runner.root, runner.goal_id, "lead")["items"] == [] + + +def test_an_ordinary_delegation_gains_no_wake_state(service): + """A delegation started outside a conversation is never a wake candidate. + + The wake capability is opt-in and belongs to a Chat conversation. An + ordinary CLI/MCP delegation must keep the acceptance shape it always had: no + intent, no persisted wake, and no change to what a plain read returns. + """ + root, runner = service + runner.start("analysis", "analysis-1", brief()) + acceptance = wait(runner) + assert acceptance["status"] == "accepted" + recorded = _read(runner.path("analysis-1")) + assert "wake" not in recorded, "an ordinary delegation must not carry wake state" + # Nor does it gain a wake target it could be routed to later. + assert "conversation" not in recorded From 5ef0c0855029e81b851fae6e58e7a87aa18b06b1 Mon Sep 17 00:00:00 2001 From: song Date: Thu, 1 Oct 2026 05:40:51 -0400 Subject: [PATCH 22/22] fix(delegation): refresh registry anchors and reuse digest authority Signed-off-by: song --- loopx/control_plane/collaboration/chat_mode.ts | 3 ++- loopx/control_plane/collaboration/delegation.ts | 2 +- loopx/semantics/project_registry_io_manifest_v1.json | 6 +++--- tests/control_plane_ts/content_digest_single_owner.test.ts | 1 + 4 files changed, 7 insertions(+), 5 deletions(-) diff --git a/loopx/control_plane/collaboration/chat_mode.ts b/loopx/control_plane/collaboration/chat_mode.ts index 826037256b..03cce4cf7d 100644 --- a/loopx/control_plane/collaboration/chat_mode.ts +++ b/loopx/control_plane/collaboration/chat_mode.ts @@ -5,6 +5,7 @@ import {EffectRuntimeRequestError} from "../effect_runtime_errors.ts"; import {requireJsonObject} from "../runtime_decode.ts"; import {resolveConversationScope} from "./conversation_scope.ts"; import {isTerminalTurnStatus} from "../turn_driver/chat_turn_acceptance.ts"; +import {BARE_SHA256_PATTERN} from "../content_digest.ts"; function requireThat(ok: unknown, message: string): asserts ok { if (!ok) throw new EffectRuntimeRequestError(message); @@ -82,7 +83,7 @@ function planDelegationWake(input: JsonObject, session: JsonObject, settings: Js const intent = requireJsonObject(input.intent, "wake intent"); const requester = requireJsonObject(intent.requester, "wake requester"); const conversation = requireJsonObject(intent.conversation, "wake conversation"); - requireThat(typeof intent.intent_id === "string" && /^[a-f0-9]{64}$/.test(intent.intent_id), "invalid wake intent"); + requireThat(typeof intent.intent_id === "string" && BARE_SHA256_PATTERN.test(intent.intent_id), "invalid wake intent"); const outcome = (state: "pending" | "refused", reason: string) => ({operation: "wake", state, reason}); if (conversation.session_id !== session.session_id || requester.goal_id !== session.goal_id) { return outcome("refused", "wake_identity_conflict"); diff --git a/loopx/control_plane/collaboration/delegation.ts b/loopx/control_plane/collaboration/delegation.ts index 600fca3f57..e24d7e1a59 100644 --- a/loopx/control_plane/collaboration/delegation.ts +++ b/loopx/control_plane/collaboration/delegation.ts @@ -380,7 +380,7 @@ function delegationWakeIntent(params: JsonObject): JsonObject { const digests = requester.artifacts.map(value => { const artifact = requireJsonObject(value, "accepted artifact"); requireThat(text(artifact.ref) && typeof artifact.sha256 === "string" - && /^[a-f0-9]{64}$/.test(artifact.sha256), "invalid accepted artifact reference"); + && BARE_SHA256_PATTERN.test(artifact.sha256), "invalid accepted artifact reference"); return {ref: artifact.ref, sha256: artifact.sha256}; }); return { diff --git a/loopx/semantics/project_registry_io_manifest_v1.json b/loopx/semantics/project_registry_io_manifest_v1.json index dacae11375..92762387e3 100644 --- a/loopx/semantics/project_registry_io_manifest_v1.json +++ b/loopx/semantics/project_registry_io_manifest_v1.json @@ -439,7 +439,7 @@ }, { "site": "loopx/chat_server.py::.ChatRequestHandler._goal_channel_extension_ready::codec_read:load_registry#1", - "line": 966, + "line": 968, "column": 24, "kind": "codec_read", "api": "load_registry", @@ -455,7 +455,7 @@ }, { "site": "loopx/chat_server.py::.serve_chat::codec_read:load_registry#1", - "line": 1488, + "line": 1490, "column": 16, "kind": "codec_read", "api": "load_registry", @@ -463,7 +463,7 @@ }, { "site": "loopx/chat_server.py::.serve_chat._wake_goal_context::codec_read:load_registry#1", - "line": 1576, + "line": 1578, "column": 20, "kind": "codec_read", "api": "load_registry", diff --git a/tests/control_plane_ts/content_digest_single_owner.test.ts b/tests/control_plane_ts/content_digest_single_owner.test.ts index 26841f961f..6a6dc4428a 100644 --- a/tests/control_plane_ts/content_digest_single_owner.test.ts +++ b/tests/control_plane_ts/content_digest_single_owner.test.ts @@ -87,6 +87,7 @@ const DECLARED_UNFOLDABLE: Record = { const CANONICAL_CONSUMERS = [ "control_plane/agents/supervisor_event_append.ts", "control_plane/capabilities/external_evidence.ts", + "control_plane/collaboration/chat_mode.ts", "control_plane/collaboration/delegation.ts", "control_plane/collaboration/goal_instance_lifecycle.ts", "control_plane/collaboration/return_delivery.ts",