From 45abc4cdf327dd92ed2da333f335973c676b8b2a Mon Sep 17 00:00:00 2001 From: huangruiteng <14976749+huangruiteng@users.noreply.github.com> Date: Wed, 30 Sep 2026 03:57:42 +0800 Subject: [PATCH 01/13] feat(operations): consume confirmed authority in the original Agent Signed-off-by: huangruiteng <14976749+huangruiteng@users.noreply.github.com> --- .../capabilities/manager_context/__init__.py | 6 +- loopx/chat_action_normalization.py | 70 ++- loopx/chat_action_store.py | 128 +++++- loopx/cli_commands/goal_channel.py | 21 + loopx/cli_commands/goal_channel_operation.py | 107 ++++- loopx/control_plane/collaboration/inbox.py | 18 + .../collaboration/operation_handoff.py | 178 ++++++++ .../control_plane/effect_runtime_handlers.ts | 4 + .../presentation/action_review_plan.ts | 20 +- .../work_items/operation_agent_handoff.ts | 143 ++++++ .../extensions/lark/goal_channel_operation.py | 104 ++++- .../action_review_plan.test.ts | 29 ++ .../operation_agent_handoff.test.ts | 91 ++++ .../test_lark_goal_channel_operation.py | 249 +++++++++- tests/test_chat_operation_actions.py | 432 ++++++++++++++++++ 15 files changed, 1564 insertions(+), 36 deletions(-) create mode 100644 loopx/control_plane/collaboration/operation_handoff.py create mode 100644 loopx/control_plane/work_items/operation_agent_handoff.ts create mode 100644 tests/control_plane_ts/operation_agent_handoff.test.ts diff --git a/loopx/capabilities/manager_context/__init__.py b/loopx/capabilities/manager_context/__init__.py index 0b46faab4d..098b01938a 100644 --- a/loopx/capabilities/manager_context/__init__.py +++ b/loopx/capabilities/manager_context/__init__.py @@ -279,7 +279,11 @@ def produce(): agent_id, scope=goal_scope, ) - count = len(inbox["items"]) + len(inbox.get("peer_returns", {}).get("items", [])) + count = ( + len(inbox["items"]) + + len(inbox.get("peer_returns", {}).get("items", [])) + + len(inbox.get("operation_handoffs", [])) + ) status, error = ("observed" if count else "empty"), None except (OSError, ValueError): count, status, error = 0, "unavailable", "manager_context_unreadable" diff --git a/loopx/chat_action_normalization.py b/loopx/chat_action_normalization.py index 94f42e8282..2d93396c7d 100644 --- a/loopx/chat_action_normalization.py +++ b/loopx/chat_action_normalization.py @@ -147,22 +147,67 @@ def _normalize( normalized_projection["fields"] = normalized_fields raw_executor = values.get("executor") - if not isinstance(raw_executor, Mapping) or set(raw_executor) != { + origin_goal_ref = None + if ( + isinstance(raw_executor, Mapping) + and raw_executor.get("kind") == "agent_session" + ): + from .control_plane.collaboration.goal_instance_scope import ( + collaboration_goal_scope, + ) + from .control_plane.effect_runtime import ( + EffectRuntimeRejected, + effect_runtime_result, + ) + from .control_plane.goals.activation import goal_is_stopped + from .thread_agent_binding import resolve_registry_thread_agent_binding + + try: + executor = dict( + effect_runtime_result( + "operation.agent_executor.normalize", + {"executor": dict(raw_executor)}, + ) + ) + except EffectRuntimeRejected as exc: + raise ValueError(str(exc)) from exc + binding = resolve_registry_thread_agent_binding( + registry_path=self.registry_path, + host_surface=str(executor["host_surface"]), + thread_id=str(executor["thread_id"]), + ) + if binding.get("status") != "bound" or ( + binding.get("goal_id"), + binding.get("agent_id"), + ) != (goal_id, agent_id): + raise ValueError( + "agent operation requires the original registered session" + ) + if normalized_projection["simulated"] is not False: + raise ValueError( + "agent execution handoff cannot masquerade as simulation" + ) + with collaboration_goal_scope( + self.registry_path, + goal_id=goal_id, + agents=(agent_id,), + require_active=True, + ) as scope: + if goal_is_stopped(scope.goal): + raise ValueError("agent operation Goal is stopped") + origin_goal_ref = scope.current_goal_ref + elif not isinstance(raw_executor, Mapping) or set(raw_executor) != { "extension_id", "protocol", "permission", "revision", }: raise ValueError("operation executor binding is invalid") - executor = { - field: _opaque(raw_executor.get(field), field=f"executor.{field}") - for field in ( - "extension_id", - "protocol", - "permission", - "revision", - ) - } + else: + executor = { + field: _opaque(raw_executor.get(field), field=f"executor.{field}") + for field in ("extension_id", "protocol", "permission", "revision") + } expires_at = parse_timestamp( _text(values.get("expires_at"), field="expires_at", limit=80) ) @@ -211,6 +256,11 @@ def _normalize( "expires_at": utc_isoformat(expires_at), "authorized_principals": principals, "executor": executor, + **( + {"origin_goal_ref": origin_goal_ref} + if origin_goal_ref is not None + else {} + ), } if action_kind == "todo.create": values = self._allowed_parameters( diff --git a/loopx/chat_action_store.py b/loopx/chat_action_store.py index 93d98901ec..5fcc922547 100644 --- a/loopx/chat_action_store.py +++ b/loopx/chat_action_store.py @@ -952,6 +952,8 @@ def observe_operation_outcome( proposal_id: str, *, outcome: Mapping[str, Any], + agent_actor: Mapping[str, Any] | None = None, + agent_binding_current: bool = False, ) -> dict[str, Any]: """Persist the domain result without making it a retryable submission.""" @@ -976,6 +978,28 @@ def observe_operation_outcome( ) if not isinstance(operation, dict): raise KeyError("typed operation was not found") + parameters = proposal.get("normalized_parameters") or {} + if (parameters.get("executor") or {}).get("kind") == "agent_session": + plan = self._agent_operation_plan( + proposal, + action="report", + actor=agent_actor, + binding_current=agent_binding_current, + outcome=safe_outcome, + ) + if plan.get("write_reconciliation") is True: + existing = operation.get("reconciliation") + if existing is not None: + if existing != safe_outcome: + raise ActionConflictError( + "operation reconciliation is already immutable" + ) + return proposal + operation["reconciliation"] = safe_outcome + proposal["receipt"] = safe_outcome + proposal["updated_at"] = _utc_now() + self._write(payload) + return proposal if operation.get("lifecycle_state") == "outcome_observed": if operation.get("outcome") != safe_outcome: raise ActionConflictError("operation outcome is already immutable") @@ -997,6 +1021,78 @@ def observe_operation_outcome( self._write(payload) return proposal + @staticmethod + def _agent_operation_plan( + proposal: Mapping[str, Any], **request: Any + ) -> dict[str, Any]: + """One typed decision, using the existing canonical JSON byte format.""" + from .control_plane.effect_runtime import ( + EffectRuntimeRemoteError, + effect_runtime_result, + ) + + parameters = proposal.get("normalized_parameters") or {} + digests = { + "payload_digest": _canonical_digest(parameters.get("payload") or {}), + "projection_digest": _canonical_digest(parameters.get("projection") or {}), + "confirmation_digest": _canonical_digest( + { + "action_kind": "operation.execute", + "normalized_parameters": parameters, + "request_digest": proposal.get("request_digest"), + } + ), + "outcome_digest": _canonical_digest(proposal["operation"]["outcome"]) + if isinstance((proposal.get("operation") or {}).get("outcome"), dict) + else None, + } + try: + return dict( + effect_runtime_result( + "operation.agent_handoff.plan", + { + "proposal": dict(proposal), + "digests": digests, + "now": _utc_now(), + **request, + }, + ) + ) + except EffectRuntimeRemoteError as exc: + raise ActionConflictError(str(exc)) from exc + + def consume_agent_operation( + self, + proposal_id: str, + *, + actor: Mapping[str, Any], + binding_current: bool, + consumption_id: str, + ) -> dict[str, Any]: + """Commit before external execution. Even an exact replay grants no retry.""" + token = _opaque_id(proposal_id, field="proposal_id") + with exclusive_file_lock( + self.path, agent_id="loopx-chat", operation="consume_agent_operation" + ): + payload = self._read() + proposal = payload["proposals"].get(token) + if not isinstance(proposal, dict): + raise KeyError("typed operation was not found") + plan = self._agent_operation_plan( + proposal, + action="consume", + actor=dict(actor), + binding_current=binding_current, + consumption_id=consumption_id, + ) + handoff = plan.pop("write_handoff", None) + if handoff is not None: + proposal["operation"]["agent_handoff"] = handoff + proposal["updated_at"] = _utc_now() + self._write(payload) + plan["consumption_id"] = handoff["consumption_id"] + return plan + def record_operation_result_delivery( self, proposal_id: str, @@ -1019,7 +1115,7 @@ def record_operation_result_delivery( "transport", "delivered_at", } - if set(safe_delivery) != required: + if set(safe_delivery) not in (required, required | {"outcome_stage"}): raise ValueError( "operation result delivery has unsupported or missing fields" ) @@ -1046,6 +1142,22 @@ def record_operation_result_delivery( raise ActionConflictError( "operation outcome is unavailable for result delivery" ) + is_agent = ( + (proposal.get("normalized_parameters") or {}).get("executor") or {} + ).get("kind") == "agent_session" + expected_stage = ( + "reconciled" + if operation.get("reconciliation") is not None + else "initial" + ) + if is_agent and safe_delivery.get("outcome_stage") != expected_stage: + raise ActionConflictError( + "result delivery is not for the current operation outcome" + ) + if not is_agent and "outcome_stage" in safe_delivery: + raise ValueError( + "result outcome stage is only supported for Agent handoff" + ) source_delivery = operation.get("delivery") if not isinstance(source_delivery, dict) or any( safe_delivery.get(field) != source_delivery.get(field) @@ -1057,10 +1169,16 @@ def record_operation_result_delivery( existing = operation.get("result_delivery") if isinstance(existing, dict): if existing != safe_delivery: - raise ActionConflictError( - "operation result delivery is already immutable" - ) - return proposal + if not ( + is_agent + and existing.get("outcome_stage") == "initial" + and expected_stage == "reconciled" + ): + raise ActionConflictError( + "operation result delivery is already immutable" + ) + else: + return proposal operation["result_delivery"] = safe_delivery proposal["updated_at"] = _utc_now() self._write(payload) diff --git a/loopx/cli_commands/goal_channel.py b/loopx/cli_commands/goal_channel.py index 0644d58d6e..cc3357eea2 100644 --- a/loopx/cli_commands/goal_channel.py +++ b/loopx/cli_commands/goal_channel.py @@ -458,6 +458,27 @@ def handle_goal_channel_command( ) print_payload(payload, output_format(args), render_goal_channel_markdown) return 0 if payload.get("ok") else 1 + if command in {"inspect-operation", "consume-operation", "report-operation"}: + # Original-Agent continuations do not call Lark. They must remain + # readable/settleable even if that transport extension is unavailable. + assert goal_id is not None + _, source_path, source_binding, source_root = _source_context( + registry=registry, + registry_path=registry_path, + goal_id=goal_id, + ) + payload = run_goal_channel_operation( + args, + context=GoalChannelOperationContext( + invoked_runtime_root=runtime_root, + source_registry_path=source_path, + source_runtime_root=source_root, + binding_path=source_binding, + ), + ) + assert payload is not None + print_payload(payload, output_format(args), render_goal_channel_markdown) + return 0 if payload.get("ok") else 1 if command == "configure" and bool(args.auto_notify_human_gates): assert goal_id is not None _, source_registry_path, binding_path, _ = _source_context( diff --git a/loopx/cli_commands/goal_channel_operation.py b/loopx/cli_commands/goal_channel_operation.py index c1070551eb..4e672cb3d3 100644 --- a/loopx/cli_commands/goal_channel_operation.py +++ b/loopx/cli_commands/goal_channel_operation.py @@ -11,8 +11,9 @@ from pathlib import Path from typing import Any, ClassVar, Self -from ..chat_action_store import ChatActionStore +from ..chat_action_store import ActionConflictError, ChatActionStore from ..chat_actions import ChatActionService +from ..control_plane.collaboration.operation_handoff import agent_operation_action from ..extensions.lark.goal_channel import ( default_goal_channel_target_path, deliver_goal_channel_operation_card, @@ -41,6 +42,9 @@ class GoalChannelOperationContext: class _GoalChannelOperationCommand(str, Enum): PREPARE = "prepare-operation" DELIVER = "deliver-operation" + INSPECT = "inspect-operation" + CONSUME = "consume-operation" + REPORT = "report-operation" @classmethod def parse(cls, value: object) -> Self | None: @@ -75,7 +79,21 @@ class _DeliverOperation: ) -_OperationRequest = _PrepareOperation | _DeliverOperation +@dataclass(frozen=True, slots=True) +class _AgentOperation: + command: _GoalChannelOperationCommand + goal_id: str + agent_id: str + proposal_id: str + host_surface: str + thread_id: str + execute: bool + consumption_id: str | None = None + outcome_path: Path | None = None + target_path_override: Path | None = None + + +_OperationRequest = _PrepareOperation | _DeliverOperation | _AgentOperation def register_goal_channel_operation_commands( @@ -110,6 +128,34 @@ def register_goal_channel_operation_commands( deliver.add_argument("--proposal-id", required=True) deliver.add_argument("--execute", action="store_true") + for name, help_text in ( + ( + "inspect-operation", + "Read the original confirmed operation; never grants execution.", + ), + ( + "consume-operation", + "Atomically consume the original Agent's authorization once. Requires --execute.", + ), + ( + "report-operation", + "Record original external outcome evidence. Does not execute the operation.", + ), + ): + parser = subparsers.add_parser(name, help=help_text) + add_subcommand_format(parser) + add_common_args(parser) + parser.add_argument("--agent-id", required=True) + parser.add_argument("--proposal-id", required=True) + parser.add_argument("--host-surface", required=True) + parser.add_argument("--thread-id", required=True) + if name != "inspect-operation": + parser.add_argument("--execute", action="store_true") + if name == "consume-operation": + parser.add_argument("--consumption-id", required=True) + if name == "report-operation": + parser.add_argument("--outcome-json", required=True) + def _parse_operation_request(args: argparse.Namespace) -> _OperationRequest | None: command = _GoalChannelOperationCommand.parse( @@ -121,6 +167,23 @@ def _parse_operation_request(args: argparse.Namespace) -> _OperationRequest | No target_path_override = ( Path(str(target_path_arg)).expanduser() if target_path_arg else None ) + if command in { + _GoalChannelOperationCommand.INSPECT, + _GoalChannelOperationCommand.CONSUME, + _GoalChannelOperationCommand.REPORT, + }: + outcome_path = getattr(args, "outcome_json", None) + return _AgentOperation( + command=command, + goal_id=str(args.goal_id), + agent_id=str(args.agent_id), + proposal_id=str(args.proposal_id), + host_surface=str(args.host_surface), + thread_id=str(args.thread_id), + execute=bool(getattr(args, "execute", False)), + consumption_id=getattr(args, "consumption_id", None), + outcome_path=Path(str(outcome_path)).expanduser() if outcome_path else None, + ) if command is _GoalChannelOperationCommand.PREPARE: return _PrepareOperation( goal_id=str(args.goal_id), @@ -187,6 +250,40 @@ def run_goal_channel_operation( if request is None: return None try: + if isinstance(request, _AgentOperation): + # These commands use the original registry/store, not Lark target + # settings or a new approval source. No external executor is called. + action = request.command.value.removesuffix("-operation") + if not request.execute: + action = "inspect" + outcome = None + if action == "report" and request.outcome_path is not None: + if request.outcome_path.stat().st_size > 65536: + raise ValueError("operation outcome exceeds its bounded envelope") + outcome = json.loads(request.outcome_path.read_text(encoding="utf-8")) + if not isinstance(outcome, dict): + raise ValueError("operation outcome must be an object") + result = agent_operation_action( + context.source_runtime_root, + context.source_registry_path, + proposal_id=request.proposal_id, + actor={ + "goal_id": request.goal_id, + "agent_id": request.agent_id, + "host_surface": request.host_surface, + "thread_id": request.thread_id, + }, + action=action, + consumption_id=request.consumption_id, + outcome=outcome, + ) + return { + "ok": True, + "goal_id": request.goal_id, + "execute": request.execute, + "operation": request.command.value.replace("-", "_"), + **result, + } target_path = _operation_target_path(request, context) binding = ( binding_for_goal( @@ -252,6 +349,12 @@ def run_goal_channel_operation( {"external_write_outcome": "unknown"} if outcome is None else None ), ) + except ActionConflictError as exc: + return _operation_error_packet( + request=request, + blocker="operation_handoff_conflict", + summary=str(exc), + ) except ValueError: return _operation_error_packet( request=request, diff --git a/loopx/control_plane/collaboration/inbox.py b/loopx/control_plane/collaboration/inbox.py index 2b07008c18..176b31e8c0 100644 --- a/loopx/control_plane/collaboration/inbox.py +++ b/loopx/control_plane/collaboration/inbox.py @@ -209,8 +209,26 @@ def append_batch(): from .peers import returns peer_returns = returns(runtime_root, goal_id, agent_id, scope=scope) + from .operation_handoff import pending_operation_handoffs + + operation_handoffs = pending_operation_handoffs( + runtime_root, + goal_id, + agent_id, + registry_path=scope.registry_path if scope is not None else None, + scope=scope, + ) return { "ok": True, + **( + { + "operation_handoffs": operation_handoffs["items"], + "operation_handoff_pending_count": operation_handoffs["pending_count"], + "operation_handoff_overflow": operation_handoffs["overflow"], + } + if operation_handoffs["items"] + else {} + ), **({"peer_returns": peer_returns} if peer_returns["items"] else {}), "items": items[:20], "has_more": len(items) > 20, diff --git a/loopx/control_plane/collaboration/operation_handoff.py b/loopx/control_plane/collaboration/operation_handoff.py new file mode 100644 index 0000000000..63be399d2b --- /dev/null +++ b/loopx/control_plane/collaboration/operation_handoff.py @@ -0,0 +1,178 @@ +"""IO for confirmed operation continuations; no second approval/inbox store. + +The original registry owns the session route. The typed action store owns human +confirmation and consumption. An inbox/wakeup message is only a locator. +""" + +from __future__ import annotations + +from collections.abc import Mapping +from pathlib import Path +from typing import Any + +from ...chat_action_store import ActionConflictError, ChatActionStore +from ...thread_agent_binding import resolve_registry_thread_agent_binding +from ..goals.activation import goal_is_stopped +from .goal_instance_scope import ( + CollaborationGoalScope, + collaboration_goal_scope, + decide_collaboration_lifecycle, +) + + +def _store(runtime_root: Path) -> ChatActionStore: + root = runtime_root / "chat" / "actions" + if not (root / "actions.json").is_file(): + raise ValueError("canonical operation store is unavailable") + return ChatActionStore(root) + + +def _binding(registry_path: Path, parameters: Mapping[str, Any]) -> bool: + executor = parameters["executor"] + observed = resolve_registry_thread_agent_binding( + registry_path=registry_path, + host_surface=executor["host_surface"], + thread_id=executor["thread_id"], + ) + return observed.get("status") == "bound" and ( + observed.get("goal_id"), + observed.get("agent_id"), + ) == (parameters["goal_id"], parameters["agent_id"]) + + +def pending_operation_handoffs( + runtime_root: Path, + goal_id: str, + agent_id: str, + *, + registry_path: Path | None = None, + scope: CollaborationGoalScope | None = None, +) -> dict[str, Any]: + """Project canonical tickets into the existing Inbox; do not copy authority.""" + if not (runtime_root / "chat" / "actions" / "actions.json").is_file(): + return {"items": [], "pending_count": 0, "overflow": None} + store = _store(runtime_root) + result = [] + for proposal in store.list(goal_id=goal_id): + parameters = proposal.get("normalized_parameters") or {} + executor = parameters.get("executor") or {} + if ( + proposal.get("action_kind") != "operation.execute" + or parameters.get("agent_id") != agent_id + or executor.get("kind") != "agent_session" + ): + continue + if ( + scope is not None + and decide_collaboration_lifecycle( + scope, + operation="inbox_observe", + record={"goal_ref": parameters.get("origin_goal_ref")}, + ).get("kind") + == "omit" + ): + continue + plan = store._agent_operation_plan(proposal, action="project") + if plan["status"] not in { + "authorized_pending", + "consumed_outcome_pending", + "submission_unknown", + }: + continue + current = registry_path is None or _binding(registry_path, parameters) + if not current and not plan["needs_reconciliation"]: + continue + result.append( + { + **plan, + "binding_current": current, + "summary": proposal["summary"], + "instruction": "Read the original canonical operation and consume it once before any external effect. " + "Only the first successful consumption permits execution; consumed/unknown results require " + "original venue reconciliation, never another submission. Inbox delivery is not trade authority.", + "next_action": "goal-channel consume-operation" + if plan["status"] == "authorized_pending" + else "Reconcile the original external result; do not submit again.", + } + ) + from ..effect_runtime import effect_runtime_result + + return dict( + effect_runtime_result("operation.agent_handoff.inbox", {"items": result}) + ) + + +def agent_operation_action( + runtime_root: Path, + registry_path: Path, + *, + proposal_id: str, + actor: Mapping[str, Any], + action: str, + consumption_id: str | None = None, + outcome: Mapping[str, Any] | None = None, +) -> dict[str, Any]: + store = _store(runtime_root) + proposal = store.load(proposal_id) + if proposal is None: + raise ValueError("canonical operation was not found") + parameters = proposal.get("normalized_parameters") or {} + if (parameters.get("goal_id"), parameters.get("agent_id")) != ( + actor.get("goal_id"), + actor.get("agent_id"), + ): + raise ActionConflictError("operation is not bound to this Goal and Agent") + with collaboration_goal_scope( + registry_path, + goal_id=actor["goal_id"], + agents=(actor["agent_id"],), + caller_goal_ref=parameters.get("origin_goal_ref"), + require_active=action == "consume", + ) as scope: + if action == "consume": + decide_collaboration_lifecycle(scope, operation="request_create") + if goal_is_stopped(scope.goal): + raise ActionConflictError("confirmed operation Goal is stopped") + else: + # Historical result publication cannot re-grant execution. + ref = {"goal_ref": parameters.get("origin_goal_ref")} + decide_collaboration_lifecycle( + scope, + operation="result_publish" if action == "report" else "history_inspect", + record=ref, + route=ref, + ) + current = _binding(registry_path, parameters) + if action == "inspect": + plan = store._agent_operation_plan(proposal, action="project") + return { + **plan, + "binding_current": current, + "parameters": parameters, + "confirmation": proposal["operation"].get("confirmation"), + "consumption": proposal["operation"].get("agent_handoff"), + "outcome": proposal["operation"].get("outcome"), + "reconciliation": proposal["operation"].get("reconciliation"), + } + if action == "consume": + return store.consume_agent_operation( + proposal_id, + actor=actor, + binding_current=current, + consumption_id=str(consumption_id or ""), + ) + if action == "report": + updated = store.observe_operation_outcome( + proposal_id, + outcome=outcome or {}, + agent_actor=actor, + agent_binding_current=current, + ) + plan = store._agent_operation_plan(updated, action="project") + return { + "ok": True, + **plan, + "outcome": updated["operation"].get("reconciliation") + or updated["operation"]["outcome"], + } + raise ValueError("unsupported agent operation action") diff --git a/loopx/control_plane/effect_runtime_handlers.ts b/loopx/control_plane/effect_runtime_handlers.ts index 7ab9b47db8..939671b2de 100644 --- a/loopx/control_plane/effect_runtime_handlers.ts +++ b/loopx/control_plane/effect_runtime_handlers.ts @@ -1,4 +1,5 @@ import {manageNewGoalStorage} from "./coordination/local_authority_defaults.ts"; +import {normalizeAgentOperationExecutor, planAgentOperationHandoff, projectAgentOperationInbox} from "./work_items/operation_agent_handoff.ts"; import {projectDecisionNotice} from "./presentation/decision_notice.ts"; import {normalizeResearchObservation, validateResearchAttribution, projectResearchFrontier} from "./capabilities/explore_research.ts"; import {projectTodoSummary} from "./todos/summary_projection.ts"; @@ -630,6 +631,9 @@ export function createEffectRuntimeHandlers( ["presentation.decision_notice.project", projectDecisionNotice], ["presentation.action_review_plan.compile", (params) => compileActionReviewPlan(params.proposal)], + ["operation.agent_executor.normalize", normalizeAgentOperationExecutor], + ["operation.agent_handoff.plan", planAgentOperationHandoff], + ["operation.agent_handoff.inbox", projectAgentOperationInbox], ["scheduler.monitor_successor.plan", planMonitorSuccessor], ["scheduler.monitor_target.select", selectMonitorTodoRequest], ["capabilities.issue_fix.monitor_reconciliation.plan", planIssueFixMonitorReconciliation], diff --git a/loopx/control_plane/presentation/action_review_plan.ts b/loopx/control_plane/presentation/action_review_plan.ts index d6e0c5b046..d5a2614be3 100644 --- a/loopx/control_plane/presentation/action_review_plan.ts +++ b/loopx/control_plane/presentation/action_review_plan.ts @@ -41,12 +41,13 @@ export type OperationReviewFrame = OperationReviewFrameBase & ( kind: "pending"; attentionKind: "progress"; interactionMode: "inform"; + executionState?: "authorized_pending" | "consumed_outcome_pending"; } | { kind: "result"; attentionKind: "progress"; interactionMode: "inform"; - resultKind: "rejected" | "simulation_completed" | "completed"; + resultKind: "rejected" | "simulation_completed" | "completed" | "unknown" | "not_executed"; resultDeliveryVerified: boolean; summary: string; } @@ -321,9 +322,12 @@ export function compileOperationReviewFrame(proposalValue: unknown): OperationRe kind: "pending", attentionKind: "progress", interactionMode: "inform", + ...(objectValue(parameters.executor)?.kind === "agent_session" + ? {executionState: objectValue(operation.agent_handoff) ? "consumed_outcome_pending" as const : "authorized_pending" as const} + : {}), }; } - const outcome = objectValue(operation.outcome); + const outcome = objectValue(operation.reconciliation) ?? objectValue(operation.outcome); if (!outcome) return undefined; const rejected = outcome.outcome === "rejected_by_operator"; const simulated = outcome.simulation === true || base.simulated; @@ -332,8 +336,11 @@ export function compileOperationReviewFrame(proposalValue: unknown): OperationRe kind: "result", attentionKind: "progress", interactionMode: "inform", - resultKind: rejected ? "rejected" : simulated ? "simulation_completed" : "completed", - resultDeliveryVerified: objectValue(operation.result_delivery) !== null, + resultKind: outcome.outcome === "submission_unknown" ? "unknown" : outcome.outcome === "not_executed" ? "not_executed" + : rejected ? "rejected" : simulated ? "simulation_completed" : "completed", + resultDeliveryVerified: objectValue(operation.result_delivery) !== null + && (objectValue(parameters.executor)?.kind !== "agent_session" + || objectValue(operation.result_delivery)?.outcome_stage === (operation.reconciliation ? "reconciled" : "initial")), summary: textValue(outcome.summary) ?? "", }; } @@ -369,9 +376,12 @@ export function compileActionReviewPlan(proposalValue: unknown): ActionReviewPla if ((lifecycle && proposal.gate != null) || proposal.status === "gated") return held("gated", "authority_gate"); if ((lifecycle && proposal.stale != null) || proposal.status === "stale") return held("refresh", "stale_proposal"); if (proposal.status === "applied") { + if (operationFrame?.kind === "result" && operationFrame.resultKind === "unknown") { + return held("repair", "readback_unverified"); + } const receipt = objectValue(proposal.receipt); return receipt?.projection_verified === true - && (proposal.action_kind !== "operation.execute" || objectValue(objectValue(proposal.operation)?.result_delivery) !== null) + && (proposal.action_kind !== "operation.execute" || (operationFrame?.kind === "result" && operationFrame.resultDeliveryVerified)) ? held("completed", "readback_verified") : held("repair", "readback_unverified"); } diff --git a/loopx/control_plane/work_items/operation_agent_handoff.ts b/loopx/control_plane/work_items/operation_agent_handoff.ts new file mode 100644 index 0000000000..f78de71a2f --- /dev/null +++ b/loopx/control_plane/work_items/operation_agent_handoff.ts @@ -0,0 +1,143 @@ +/** Agent execution is a continuation of the original typed operation, not a + * second approval store. Python supplies locked storage and registry facts; + * this owner decides admission, one-shot consumption and result binding. */ +import type {JsonObject} from "../effect_program.ts"; +import {EffectRuntimeConflictError, EffectRuntimeRequestError} from "../effect_runtime_errors.ts"; +import {requireJsonObject, requireNonEmptyString} from "../runtime_decode.ts"; + +export const AGENT_OPERATION_REVISION = "agent-session-handoff-v0"; +const ID = /^[A-Za-z0-9._:-]{1,200}$/; + +function id(value: unknown, field: string): string { + const result = requireNonEmptyString(value, field); + if (!ID.test(result)) throw new EffectRuntimeRequestError(`${field} must be a compact opaque id`); + return result; +} +function requireThat(value: unknown, message: string): asserts value { + if (!value) throw new EffectRuntimeConflictError(message, "operation_handoff_conflict"); +} +function timestamp(value: unknown): number { + const text = requireNonEmptyString(value, "operation timestamp"); + const parsed = Date.parse(text); + if (!/(Z|[+-]\d\d:\d\d)$/.test(text) || !Number.isFinite(parsed)) { + throw new EffectRuntimeRequestError("operation timestamp requires a timezone"); + } + return parsed; +} + +export function normalizeAgentOperationExecutor(input: JsonObject): JsonObject { + const executor = requireJsonObject(input.executor, "agent executor"); + const keys = ["kind", "host_surface", "thread_id", "revision"]; + if (Object.keys(executor).length !== keys.length || keys.some(key => !(key in executor)) + || executor.kind !== "agent_session" || executor.revision !== AGENT_OPERATION_REVISION) { + throw new EffectRuntimeRequestError("agent operation executor binding is invalid"); + } + return {kind: "agent_session", host_surface: id(executor.host_surface, "host_surface"), + thread_id: id(executor.thread_id, "thread_id"), revision: AGENT_OPERATION_REVISION}; +} + +export function planAgentOperationHandoff(input: JsonObject): JsonObject { + const proposal = requireJsonObject(input.proposal, "proposal"); + const parameters = requireJsonObject(proposal.normalized_parameters, "parameters"); + const operation = requireJsonObject(proposal.operation, "operation"); + const executor = normalizeAgentOperationExecutor({executor: parameters.executor}); + const action = input.action; + const digests = requireJsonObject(input.digests, "locked operation digests"); + requireThat(proposal.action_kind === "operation.execute" && operation.operation_id === proposal.proposal_id, + "agent handoff requires the original typed operation"); + requireThat(digests.payload_digest === parameters.payload_digest && parameters.payload_digest === operation.payload_digest + && digests.projection_digest === parameters.projection_digest && parameters.projection_digest === operation.projection_digest + && digests.confirmation_digest === operation.confirmation_digest + && executor.revision === operation.executor_revision + && parameters.destination_account_ref === operation.destination_account_ref + && parameters.expires_at === operation.expires_at + && JSON.stringify(parameters.authorized_principals) === JSON.stringify(operation.authorized_principals), + "immutable operation binding drifted"); + const confirmation = operation.confirmation == null ? null + : requireJsonObject(operation.confirmation, "operation confirmation"); + const claim = operation.claim == null ? null : requireJsonObject(operation.claim, "operation claim"); + const now = timestamp(input.now); + const expires = timestamp(operation.expires_at); + const route = {goal_id: parameters.goal_id, agent_id: parameters.agent_id, + host_surface: executor.host_surface, thread_id: executor.thread_id}; + const base: JsonObject = {schema_version: "loopx_operation_agent_handoff_v0", operation_id: operation.operation_id, + payload_digest: operation.payload_digest, confirmation_digest: operation.confirmation_digest, + claim_id: claim?.claim_id ?? null, executor_revision: executor.revision, expires_at: operation.expires_at, + route, authorization_source: "canonical_typed_operation", execution_allowed: false, + host_delivery: "not_attempted", external_write_performed: false}; + const handoff = operation.agent_handoff == null ? null + : requireJsonObject(operation.agent_handoff, "agent handoff"); + const observed = operation.reconciliation ?? operation.outcome; + const unknownResult = observed != null + && requireJsonObject(observed, "observed result").outcome === "submission_unknown"; + if (action === "project") { + return {...base, outcome_digest: digests.outcome_digest ?? null, + status: unknownResult ? "submission_unknown" + : operation.lifecycle_state === "outcome_observed" ? "outcome_observed" + : handoff ? "consumed_outcome_pending" : now >= expires ? "expired" + : operation.lifecycle_state === "claimed" ? "authorized_pending" : "awaiting_confirmation", + needs_reconciliation: unknownResult || (!!handoff && operation.lifecycle_state !== "outcome_observed")}; + } + requireThat(confirmation?.decision === "confirm" + && confirmation.confirmation_digest === operation.confirmation_digest && claim, + "agent execution requires authenticated confirmation"); + const actor = requireJsonObject(input.actor, "execution actor"); + requireThat(Object.entries(route).every(([key, value]) => actor[key] === value), + "execution actor is not the original bound session"); + if (action === "consume") { + requireThat(input.binding_current === true, "original session binding is no longer current"); + // Even a same-id retry returns no execute permission. A lost response after + // this commit is ambiguous, never permission to submit a second order. + if (handoff || operation.lifecycle_state === "outcome_observed") { + return {...base, status: "already_consumed", needs_reconciliation: unknownResult || operation.lifecycle_state !== "outcome_observed"}; + } + requireThat(operation.lifecycle_state === "claimed" && proposal.status === "applying", "operation is not claimed"); + requireThat(now < expires, "confirmed operation expired before execution consumption"); + return {...base, status: "consumed_outcome_pending", execution_allowed: true, + write_handoff: {...base, status: "consumed_outcome_pending", consumed_at: input.now, + consumption_id: id(input.consumption_id, "consumption_id")}}; + } + if (action === "report") { + requireThat(handoff, "operation authorization has not been consumed"); + const outcome = requireJsonObject(input.outcome, "operation outcome"); + for (const key of ["operation_id", "payload_digest", "confirmation_digest", "claim_id", "executor_revision"]) { + requireThat(outcome[key] === base[key], "operation outcome does not match the consumed authorization"); + } + requireThat(outcome.consumption_id === handoff.consumption_id, "operation consumption identity drifted"); + requireThat(outcome.schema_version === "loopx_operation_outcome_v0" && outcome.projection_verified === true + && outcome.simulation === false && typeof outcome.external_write_performed === "boolean", + "agent result must separately disclose real external effect status"); + requireThat(["executed", "not_executed", "submission_unknown"].includes(String(outcome.outcome)), "agent outcome is unsupported"); + requireThat(Array.isArray(outcome.evidence_refs) && outcome.evidence_refs.length > 0 + && outcome.evidence_refs.length <= 20 && outcome.evidence_refs.every(ref => typeof ref === "string" && ref.length > 0 && ref.length <= 512), + "agent outcome requires bounded original evidence references"); + requireThat(outcome.outcome !== "executed" || outcome.external_write_performed === true, + "execution completion requires a disclosed external effect"); + requireThat(outcome.outcome !== "not_executed" || outcome.external_write_performed === false, + "a no-execution result must not hide an external effect"); + requireThat(outcome.outcome !== "submission_unknown" || outcome.external_write_performed === true, + "an ambiguous submission must conservatively disclose a possible external effect"); + const original = operation.outcome == null ? null : requireJsonObject(operation.outcome, "original outcome"); + if (original?.outcome === "submission_unknown" && outcome.outcome !== "submission_unknown") { + requireThat(outcome.reconciles_outcome_digest === digests.outcome_digest, + "reconciliation must reference the exact original unknown result"); + return {...base, status: "outcome_observed", outcome, write_reconciliation: true}; + } + return {...base, status: outcome.outcome === "submission_unknown" ? "submission_unknown" : "outcome_observed", outcome}; + } + throw new EffectRuntimeRequestError("unsupported agent operation action"); +} + +/** Bounded attention does not discard original recovery obligations. The + * overflow locator can be inspected directly in the canonical action store. */ +export function projectAgentOperationInbox(input: JsonObject): JsonObject { + if (!Array.isArray(input.items)) throw new EffectRuntimeRequestError("handoff items must be an array"); + const items = input.items.map(value => requireJsonObject(value, "handoff item")); + const rank = (item: JsonObject) => item.needs_reconciliation === true ? 0 : 1; + items.sort((a, b) => rank(a) - rank(b) + || String(a.operation_id).localeCompare(String(b.operation_id), "en")); + return {items: items.slice(0, 20), pending_count: items.length, + overflow: items.length > 20 ? {reason: "attention_page_capacity", count: items.length - 20, + next_operation_id: items[20].operation_id, + instruction: "Inspect the next original operation by id; do not treat this page as the entire inbox."} : null}; +} diff --git a/loopx/extensions/lark/goal_channel_operation.py b/loopx/extensions/lark/goal_channel_operation.py index 505132b794..62b0637878 100644 --- a/loopx/extensions/lark/goal_channel_operation.py +++ b/loopx/extensions/lark/goal_channel_operation.py @@ -111,6 +111,26 @@ def _operation_review_frame(proposal: Mapping[str, Any]) -> dict[str, Any]: return dict(frame) +def _result_delivery_stage(proposal: Mapping[str, Any]) -> dict[str, str]: + parameters, operation = _proposal_operation(proposal) + if (parameters.get("executor") or {}).get("kind") != "agent_session": + return {} + return { + "outcome_stage": "reconciled" + if operation.get("reconciliation") is not None + else "initial" + } + + +def _result_delivery_current(proposal: Mapping[str, Any]) -> bool: + _parameters, operation = _proposal_operation(proposal) + delivery = operation.get("result_delivery") + return isinstance(delivery, Mapping) and all( + delivery.get(key) == value + for key, value in _result_delivery_stage(proposal).items() + ) + + def build_goal_channel_operation_card( proposal: Mapping[str, Any], ) -> dict[str, Any]: @@ -324,16 +344,39 @@ def build_goal_channel_operation_result_card( proposal: Mapping[str, Any], ) -> dict[str, Any]: frame = _operation_review_frame(proposal) - if frame.get("kind") != "result": + if frame.get("kind") not in {"result", "pending"}: raise ActionConflictError("operation outcome is not available") projection = frame.get("content") if not isinstance(projection, Mapping): raise ValueError("operation result projection is unavailable") result_kind = frame.get("resultKind") + pending = frame.get("kind") == "pending" + unknown = result_kind == "unknown" + not_executed = result_kind == "not_executed" rejected = result_kind == "rejected" simulated = result_kind == "simulation_completed" - template = "red" if rejected else "green" - result_label = "已拒绝" if rejected else "模拟完成" if simulated else "已完成" + template = ( + "orange" + if pending or unknown or not_executed + else "red" + if rejected + else "green" + ) + result_label = ( + "已确认,等待原 Agent 执行" + if pending + else "结果未知,须核对原操作,不可重复提交" + if unknown + else "已结束,未执行" + if not_executed + else "已拒绝" + if rejected + else "模拟完成" + if simulated + else "已完成" + ) + if pending and frame.get("executionState") == "consumed_outcome_pending": + result_label = "执行授权已消费,等待真实结果" summary = str(frame.get("summary") or result_label) return { "schema": "2.0", @@ -355,7 +398,11 @@ def build_goal_channel_operation_result_card( { "tag": "text_tag", "text": {"tag": "plain_text", "content": result_label}, - "color": "red" if rejected else "green", + "color": "orange" + if pending or unknown or not_executed + else "red" + if rejected + else "green", } ], }, @@ -626,6 +673,12 @@ def _resolve_operation_executor_binding( parameters: Mapping[str, Any], *, runtime_root: Path ) -> dict[str, Any]: executor = parameters.get("executor") + if isinstance(executor, Mapping) and executor.get("kind") == "agent_session": + return dict( + effect_runtime_result( + "operation.agent_executor.normalize", {"executor": dict(executor)} + ) + ) if not isinstance(executor, Mapping): raise ValueError("operation executor binding is unavailable") return resolve_extension_binding( @@ -782,7 +835,7 @@ def recover_goal_channel_operation_results( delivery = operation.get("delivery") if ( operation.get("lifecycle_state") != "outcome_observed" - or isinstance(operation.get("result_delivery"), Mapping) + or _result_delivery_current(proposal) or not isinstance(delivery, Mapping) or delivery.get("provider") != "lark" or delivery.get("app_id") != profile_app_id @@ -801,7 +854,7 @@ def recover_goal_channel_operation_results( failed += 1 continue _current_parameters, current_operation = _proposal_operation(current) - if isinstance(current_operation.get("result_delivery"), Mapping): + if _result_delivery_current(current): continue current_delivery = current_operation.get("delivery") if not isinstance(current_delivery, Mapping): @@ -830,6 +883,7 @@ def recover_goal_channel_operation_results( "card_digest": _digest(result_card), "transport": "message_patch", "delivered_at": datetime.now(timezone.utc).isoformat(), + **_result_delivery_stage(current), }, ) delivered += 1 @@ -1002,7 +1056,15 @@ def handle_goal_channel_operation_callback( if current is None: raise ValueError("claimed operation disappeared before dispatch") _parameters, current_operation = _proposal_operation(current) - if current_operation.get("lifecycle_state") == "claimed": + if ( + current_operation.get("lifecycle_state") == "claimed" + and (_parameters.get("executor") or {}).get("kind") == "agent_session" + ): + # Confirmation exposes an exact canonical continuation in the + # existing Inbox. No simulator, host resume or financial effect is + # run in the callback process, and no outcome is manufactured. + store._agent_operation_plan(current, action="project") + elif current_operation.get("lifecycle_state") == "claimed": outcome = dict( executor(current) if executor is not None @@ -1023,7 +1085,7 @@ def handle_goal_channel_operation_callback( raise ValueError("operation disappeared before result delivery") decided = current result_card = build_goal_channel_operation_result_card(decided) - if isinstance(decided["operation"].get("result_delivery"), Mapping): + if _result_delivery_current(decided): update = {"external_write_performed": False, "readback_verified": True} else: update = _update_callback_card( @@ -1036,7 +1098,10 @@ def handle_goal_channel_operation_callback( chat_id=str(delivery["chat_id"]), app_id=str(delivery["app_id"]), ) - if update["readback_verified"] is True: + if ( + update["readback_verified"] is True + and decided["operation"]["lifecycle_state"] == "outcome_observed" + ): decided = store.record_operation_result_delivery( action["operation_id"], delivery={ @@ -1047,16 +1112,22 @@ def handle_goal_channel_operation_callback( "card_digest": _digest(result_card), "transport": "callback_update", "delivered_at": datetime.now(timezone.utc).isoformat(), + **_result_delivery_stage(decided), }, ) update_verified = update["readback_verified"] + observed_outcome = ( + decided["operation"].get("reconciliation") + or decided["operation"].get("outcome") + or {} + ) return { "ok": update_verified, "schema_version": OPERATION_CALLBACK_RECEIPT_SCHEMA_VERSION, "operation_id": action["operation_id"], "decision": action["decision"], "lifecycle_state": decided["operation"]["lifecycle_state"], - "outcome": decided["operation"]["outcome"]["outcome"], + "outcome": observed_outcome.get("outcome"), "claim_id": ( decided["operation"]["claim"]["claim_id"] if isinstance(decided["operation"].get("claim"), Mapping) @@ -1064,9 +1135,18 @@ def handle_goal_channel_operation_callback( ), "callback_ack_is_execution_receipt": False, "card_update_verified": update_verified, - "status": "outcome_observed" if update_verified else "result_delivery_pending", + "status": ( + "authorization_pending" + if decided["operation"]["lifecycle_state"] == "claimed" + else "submission_unknown" + if observed_outcome.get("outcome") == "submission_unknown" + else "outcome_observed" + ) + if update_verified + else "result_delivery_pending", + "needs_reconciliation": observed_outcome.get("outcome") == "submission_unknown", "domain_external_write_performed": bool( - decided["operation"]["outcome"].get("external_write_performed") is True + observed_outcome.get("external_write_performed") is True ), "external_write_performed": update["external_write_performed"], } diff --git a/tests/control_plane_ts/action_review_plan.test.ts b/tests/control_plane_ts/action_review_plan.test.ts index 4f692934be..25379d08da 100644 --- a/tests/control_plane_ts/action_review_plan.test.ts +++ b/tests/control_plane_ts/action_review_plan.test.ts @@ -119,6 +119,35 @@ test("operation frame rejects identity drift and malformed projection fields", ( assert.equal(compileOperationReviewFrame(malformed), undefined); }); +test("original-Agent pending, unknown and reconciled results share truthful surface semantics", () => { + const proposal: Record = operationProposal("claimed"); + proposal.status = "applying"; + proposal.normalized_parameters.executor = {kind: "agent_session"}; + proposal.normalized_parameters.projection.simulated = false; + let frame = compileOperationReviewFrame(proposal); + assert.equal(frame?.kind === "pending" && frame.executionState, "authorized_pending"); + proposal.operation.agent_handoff = {consumption_id: "attempt-1"}; + frame = compileOperationReviewFrame(proposal); + assert.equal(frame?.kind === "pending" && frame.executionState, "consumed_outcome_pending"); + proposal.status = "applied"; + proposal.receipt = {projection_verified: true}; + proposal.operation.lifecycle_state = "outcome_observed"; + proposal.operation.outcome = {outcome: "submission_unknown", simulation: false, summary: "Reconciliation required."}; + proposal.operation.result_delivery = {receipt_id: "delivery-1", outcome_stage: "initial"}; + const unknown = compileActionReviewPlan(proposal); + assert.equal(unknown.interaction, "repair"); + assert.equal(unknown.canApply, false); + assert.equal(unknown.operationFrame?.kind === "result" && unknown.operationFrame.resultKind, "unknown"); + proposal.operation.reconciliation = {outcome: "not_executed", simulation: false, summary: "No external effect verified."}; + assert.equal(compileActionReviewPlan(proposal).interaction, "repair", "Old unknown-result card is not final delivery"); + proposal.operation.result_delivery.outcome_stage = "reconciled"; + const reconciled = compileActionReviewPlan(proposal); + assert.equal(reconciled.interaction, "completed"); + assert.equal(reconciled.canApply, false); + assert.equal(reconciled.operationFrame?.kind === "result" && reconciled.operationFrame.resultKind, "not_executed"); + assert.equal(proposal.operation.outcome.outcome, "submission_unknown"); +}); + test("generic action review keeps state precedence and stale classification", () => { const proposal = { proposal_id: "preview-1", diff --git a/tests/control_plane_ts/operation_agent_handoff.test.ts b/tests/control_plane_ts/operation_agent_handoff.test.ts new file mode 100644 index 0000000000..063688e9b7 --- /dev/null +++ b/tests/control_plane_ts/operation_agent_handoff.test.ts @@ -0,0 +1,91 @@ +import assert from "node:assert/strict"; +import test from "node:test"; +import type {JsonObject} from "../../loopx/control_plane/effect_program.ts"; +import {AGENT_OPERATION_REVISION, normalizeAgentOperationExecutor, planAgentOperationHandoff, + projectAgentOperationInbox} from "../../loopx/control_plane/work_items/operation_agent_handoff.ts"; + +function input(): JsonObject { + const executor = {kind: "agent_session", host_surface: "codex-app", thread_id: "original-thread", + revision: AGENT_OPERATION_REVISION}; + const parameters = {goal_id: "test-goal", agent_id: "test-agent", executor, + payload_digest: "payload", projection_digest: "projection", destination_account_ref: "opaque-account", + expires_at: "2030-01-01T01:00:00Z", authorized_principals: ["opaque-principal"]}; + const operation = {...parameters, operation_id: "operation-1", executor_revision: executor.revision, + confirmation_digest: "confirmation", lifecycle_state: "claimed", + confirmation: {decision: "confirm", confirmation_digest: "confirmation"}, claim: {claim_id: "claim-1"}}; + return {action: "consume", now: "2030-01-01T00:00:00Z", binding_current: true, consumption_id: "attempt-1", + actor: {goal_id: "test-goal", agent_id: "test-agent", host_surface: "codex-app", thread_id: "original-thread"}, + digests: {payload_digest: "payload", projection_digest: "projection", confirmation_digest: "confirmation", + outcome_digest: "unknown-result-digest"}, + proposal: {action_kind: "operation.execute", proposal_id: "operation-1", status: "applying", + normalized_parameters: parameters, operation}}; +} +const operation = (value: JsonObject) => (value.proposal as JsonObject).operation as JsonObject; + +test("only the first consumed canonical confirmation grants the original session execution", () => { + const value = input(); + const plan = planAgentOperationHandoff(value); + assert.equal(plan.execution_allowed, true); + assert.equal(plan.host_delivery, "not_attempted"); + operation(value).agent_handoff = plan.write_handoff; + for (const consumption_id of ["attempt-1", "new-attempt"]) { + const replay = planAgentOperationHandoff({...value, consumption_id}); + assert.equal(replay.execution_allowed, false); + assert.equal(replay.needs_reconciliation, true); + } +}); + +test("each immutable term, confirmation, route and current binding fails closed independently", () => { + const mutate: Array<(value: JsonObject) => void> = [ + value => {operation(value).confirmation = null;}, + value => {operation(value).claim = null;}, + value => {operation(value).authorized_principals = ["other-principal"];}, + value => {operation(value).expires_at = "2030-01-01T02:00:00Z";}, + value => {operation(value).destination_account_ref = "other-account";}, + value => {(value.actor as JsonObject).thread_id = "replacement-thread";}, + value => {value.binding_current = false;}, + value => {value.now = "2030-01-01T01:00:00Z";}, + ]; + for (const change of mutate) { + const value = input(); change(value); + assert.throws(() => planAgentOperationHandoff(value)); + } + assert.throws(() => normalizeAgentOperationExecutor({executor: { + ...((input().proposal as JsonObject).normalized_parameters as JsonObject).executor as JsonObject, + resume_prompt: "Injected instructions are not an executor binding", + }})); +}); + +test("unknown submission remains a reconciliation obligation after expiry with no new execution", () => { + const value = input(); + operation(value).agent_handoff = planAgentOperationHandoff(value).write_handoff; + operation(value).lifecycle_state = "outcome_observed"; + operation(value).outcome = {outcome: "submission_unknown"}; + const projected = planAgentOperationHandoff({...value, action: "project", now: "2031-01-01T00:00:00Z"}); + assert.equal(projected.status, "submission_unknown"); + assert.equal(projected.needs_reconciliation, true); + assert.equal(projected.execution_allowed, false); + const outcome: JsonObject = {schema_version: "loopx_operation_outcome_v0", operation_id: "operation-1", + payload_digest: "payload", confirmation_digest: "confirmation", claim_id: "claim-1", + executor_revision: AGENT_OPERATION_REVISION, consumption_id: "attempt-1", outcome: "not_executed", + projection_verified: true, simulation: false, external_write_performed: false, + evidence_refs: ["receipt:fixture-original"]}; + assert.throws(() => planAgentOperationHandoff({...value, action: "report", outcome})); + outcome.reconciles_outcome_digest = "unknown-result-digest"; + const final = planAgentOperationHandoff({...value, action: "report", now: "2031-01-01T00:00:00Z", + binding_current: false, outcome}); + assert.equal(final.execution_allowed, false); + assert.equal(final.write_reconciliation, true); +}); + +test("bounded inbox retains recovery first and declares overflow instead of silently discarding it", () => { + const items = Array.from({length: 22}, (_, index) => ({operation_id: `operation-${String(index).padStart(2, "0")}`, + needs_reconciliation: index === 21, execution_allowed: false})); + const page = projectAgentOperationInbox({items}); + assert.equal((page.items as JsonObject[]).length, 20); + assert.equal((page.items as JsonObject[])[0].operation_id, "operation-21"); + assert.equal(page.pending_count, 22); + assert.equal((page.overflow as JsonObject).count, 2); + assert.equal((page.overflow as JsonObject).next_operation_id, "operation-19"); + assert.equal(items[0].operation_id, "operation-00"); +}); diff --git a/tests/extensions/test_lark_goal_channel_operation.py b/tests/extensions/test_lark_goal_channel_operation.py index c4c179eede..388ffe088c 100644 --- a/tests/extensions/test_lark_goal_channel_operation.py +++ b/tests/extensions/test_lark_goal_channel_operation.py @@ -10,10 +10,18 @@ from typing import Any import pytest +from argparse import Namespace +import sys from loopx.chat_action_store import ActionConflictError, ChatActionStore from loopx.chat_actions import ChatActionService -from loopx.cli_commands.goal_channel_operation import _prepare_goal_channel_operation +from loopx.cli_commands.goal_channel_operation import ( + GoalChannelOperationContext, + _prepare_goal_channel_operation, + run_goal_channel_operation, +) +from loopx.control_plane.collaboration.operation_handoff import agent_operation_action +from loopx.control_plane.collaboration.inbox import pending from loopx.extensions.lark.goal_channel_contracts import ( GOAL_CHANNEL_BINDING_SCHEMA_VERSION, write_goal_channel_binding, @@ -44,6 +52,245 @@ TENANT_KEY = "tenant_operation_fixture" +def _prepare_agent_handoff(store: ChatActionStore, registry: Path) -> dict[str, Any]: + baseline = _prepare(store, registry) + data = json.loads(registry.read_text()) + data["goals"][0]["coordination"]["thread_agent_bindings"] = [ + { + "agent_id": AGENT_ID, + "host_surface": "codex-app", + "thread_id": "thread-operation-fixture", + } + ] + registry.write_text(json.dumps(data)) + parameters = dict(baseline["normalized_parameters"]) + parameters.pop("projection_digest") + parameters["executor"] = { + "kind": "agent_session", + "host_surface": "codex-app", + "thread_id": "thread-operation-fixture", + "revision": "agent-session-handoff-v0", + } + parameters["operation_kind"] = "fixture.submit" + parameters["projection"] = { + **parameters["projection"], + "simulated": False, + "title": "Synthetic Agent execution handoff", + "warning": "Engineering fixture; never sent to a live provider.", + } + parameters["destination_account_ref"] = "account:synthetic-fixture" + return ChatActionService(store=store, registry_path=registry).preview( + { + "action_kind": "operation.execute", + "summary": "Synthetic original-Agent handoff", + "idempotency_key": "agent-operation-fixture-v1", + "context": {"kind": "goal", "goal_id": GOAL_ID}, + "normalized_parameters": parameters, + } + ) + + +def test_authenticated_callback_hands_off_without_calling_any_executor_and_reconciles_original_result( + tmp_path: Path, +) -> None: + store, registry, runtime, binding, target = _fixture(tmp_path) + proposal = _prepare_agent_handoff(store, registry) + cards: dict[str, dict[str, Any]] = {} + runner = _runner([], cards) + deliver_goal_channel_operation_card( + proposal_id=proposal["proposal_id"], + action_store_root=store.root, + runtime_root=runtime, + binding_path=binding, + target_path=target, + execute=True, + runner=runner, + ) + delivered = store.load(proposal["proposal_id"]) + card = cards[delivered["operation"]["delivery"]["message_id"]] + event = _event(delivered, card) + + def no_executor(_proposal): + pytest.fail("a human callback must not run an Agent or simulation executor") + + kwargs = dict( + runtime_root=runtime, + action_store_root=store.root, + profile_app_id=APP_ID, + cli_bin="lark-cli", + profile="operation-bot", + runner=runner, + executor=no_executor, + ) + for invalid in [ + {"operator_id": "ou_wrong_operator"}, + {"message_id": "om_wrong_card"}, + ]: + with pytest.raises(ActionConflictError): + handle_goal_channel_operation_callback({**event, **invalid}, **kwargs) + first = handle_goal_channel_operation_callback(event, **kwargs) + replay = handle_goal_channel_operation_callback(event, **kwargs) + assert first["status"] == replay["status"] == "authorization_pending" + assert first["outcome"] is None and not first["domain_external_write_performed"] + assert first["callback_ack_is_execution_receipt"] is False + claimed = store.load(proposal["proposal_id"]) + assert claimed["operation"]["result_delivery"] is None + assert "等待原 Agent" in normalized_card_text(next(iter(cards.values()))) + actor = { + "goal_id": GOAL_ID, + "agent_id": AGENT_ID, + "host_surface": "codex-app", + "thread_id": "thread-operation-fixture", + } + context = GoalChannelOperationContext(runtime, registry, runtime, binding) + args = Namespace( + goal_channel_command="consume-operation", + goal_id=GOAL_ID, + agent_id=AGENT_ID, + proposal_id=proposal["proposal_id"], + host_surface="codex-app", + thread_id="thread-operation-fixture", + consumption_id="attempt-1", + execute=False, + ) + dry = run_goal_channel_operation(args, context=context) + assert dry["execution_allowed"] is False and not store.load( + proposal["proposal_id"] + )["operation"].get("agent_handoff") + args.execute = True + consumed = run_goal_channel_operation(args, context=context) + assert consumed["execution_allowed"] is True + assert ( + run_goal_channel_operation(args, context=context)["execution_allowed"] is False + ) + operation = claimed["operation"] + unknown = { + "schema_version": "loopx_operation_outcome_v0", + "operation_id": proposal["proposal_id"], + "payload_digest": operation["payload_digest"], + "confirmation_digest": operation["confirmation_digest"], + "claim_id": operation["claim"]["claim_id"], + "executor_revision": operation["executor_revision"], + "consumption_id": "attempt-1", + "projection_verified": True, + "simulation": False, + "outcome": "submission_unknown", + "external_write_performed": True, + "summary": "Synthetic submission result is unknown; do not resubmit.", + "evidence_refs": ["receipt:unknown-fixture"], + } + outcome_path = tmp_path / "outcome.json" + outcome_path.write_text(json.dumps(unknown)) + report_args = Namespace( + **{ + **vars(args), + "goal_channel_command": "report-operation", + "outcome_json": str(outcome_path), + } + ) + reported = run_goal_channel_operation(report_args, context=context) + assert ( + reported["status"] == "submission_unknown" and reported["needs_reconciliation"] + ) + assert pending(runtime, GOAL_ID, AGENT_ID)["operation_handoffs"][0][ + "needs_reconciliation" + ] + recovered = recover_goal_channel_operation_results( + action_store_root=store.root, + profile_app_id=APP_ID, + allowed_chat_ids={CHAT_ID}, + cli_bin="lark-cli", + profile="operation-bot", + runner=runner, + ) + assert recovered["delivered"] == 1 + assert "不可重复提交" in normalized_card_text(next(iter(cards.values()))) + final = { + **unknown, + "outcome": "not_executed", + "external_write_performed": False, + "summary": "Synthetic original venue evidence proves no submission.", + "reconciles_outcome_digest": _digest(unknown), + "evidence_refs": ["receipt:reconciled-fixture"], + } + outcome_path.write_text(json.dumps(final)) + assert run_goal_channel_operation(report_args, context=context)["outcome"] == final + updated = store.load(proposal["proposal_id"]) + frame = goal_channel_operation._operation_review_frame(updated) + assert frame["resultDeliveryVerified"] is False + assert frame["resultKind"] == "not_executed" + recovered = recover_goal_channel_operation_results( + action_store_root=store.root, + profile_app_id=APP_ID, + allowed_chat_ids={CHAT_ID}, + cli_bin="lark-cli", + profile="operation-bot", + runner=runner, + ) + assert recovered["delivered"] == 1 + assert "已结束,未执行" in normalized_card_text(next(iter(cards.values()))) + assert store.load(proposal["proposal_id"])["operation"]["outcome"] == unknown + assert ( + store.load(proposal["proposal_id"])["operation"]["result_delivery"][ + "outcome_stage" + ] + == "reconciled" + ) + assert ( + run_goal_channel_operation(args, context=context)["execution_allowed"] is False + ) + assert "operation_handoffs" not in pending(runtime, GOAL_ID, AGENT_ID) + assert ( + agent_operation_action( + runtime, + registry, + proposal_id=proposal["proposal_id"], + actor=actor, + action="inspect", + )["execution_allowed"] + is False + ) + + +def test_real_cli_inspects_canonical_handoff_in_source_runtime(tmp_path: Path) -> None: + store, registry, runtime, _binding, _target = _fixture(tmp_path) + proposal = _prepare_agent_handoff(store, registry) + result = subprocess.run( + [ + sys.executable, + "-m", + "loopx.cli", + "--format", + "json", + "--registry", + str(registry), + "--runtime-root", + str(runtime), + "goal-channel", + "inspect-operation", + "--goal-id", + GOAL_ID, + "--agent-id", + AGENT_ID, + "--proposal-id", + proposal["proposal_id"], + "--host-surface", + "codex-app", + "--thread-id", + "thread-operation-fixture", + ], + text=True, + capture_output=True, + check=True, + timeout=30, + ) + packet = json.loads(result.stdout) + assert ( + packet["status"] == "awaiting_confirmation" + and packet["execution_allowed"] is False + ) + + def _digest(value: object) -> str: return hashlib.sha256( json.dumps( diff --git a/tests/test_chat_operation_actions.py b/tests/test_chat_operation_actions.py index e428dcc196..36314b8f9d 100644 --- a/tests/test_chat_operation_actions.py +++ b/tests/test_chat_operation_actions.py @@ -3,16 +3,448 @@ from datetime import datetime, timedelta, timezone import hashlib import json +from concurrent.futures import ThreadPoolExecutor from pathlib import Path import pytest from loopx.chat_action_store import ActionConflictError, ChatActionStore from loopx.chat_actions import ChatActionService, ProtectedActionGate +from loopx.control_plane.collaboration.operation_handoff import agent_operation_action +from loopx.control_plane.collaboration.inbox import pending +from loopx.capabilities.manager_context import turn_start_hook GOAL_ID = "goal-operation-fixture" OPERATOR_ID = "ou_authorized_fixture" +EXECUTION_ACTOR = { + "goal_id": GOAL_ID, + "agent_id": "finance-fixture-agent", + "host_surface": "codex-app", + "thread_id": "thread-fixture-original", +} + + +def _agent_request(service: ChatActionService) -> dict[str, object]: + registry = json.loads(service.registry_path.read_text()) + registry["goals"][0]["coordination"]["thread_agent_bindings"] = [ + {key: value for key, value in EXECUTION_ACTOR.items() if key != "goal_id"} + ] + service.registry_path.write_text(json.dumps(registry)) + request = _request() + parameters = request["normalized_parameters"] + parameters["executor"] = { + "kind": "agent_session", + "revision": "agent-session-handoff-v0", + "host_surface": EXECUTION_ACTOR["host_surface"], + "thread_id": EXECUTION_ACTOR["thread_id"], + } + parameters["operation_kind"] = "finance.order.execute" + parameters["projection"]["simulated"] = False + parameters["projection"]["warning"] = ( + "Synthetic engineering terms; no live account or venue is used." + ) + parameters["destination_account_ref"] = "account:synthetic-fixture" + return request + + +def _claim_agent_operation(service: ChatActionService, store: ChatActionStore) -> dict: + proposal = service.preview(_agent_request(service)) + delivered = store.record_operation_delivery( + proposal["proposal_id"], delivery=_delivery(proposal) + ) + return store.decide_operation( + proposal["proposal_id"], + decision="confirm", + confirmation=_confirmation(delivered), + ) + + +def _agent_result( + proposal: dict, consumption_id: str, *, result: str = "executed" +) -> dict: + operation = proposal["operation"] + return { + "schema_version": "loopx_operation_outcome_v0", + "operation_id": proposal["proposal_id"], + "payload_digest": operation["payload_digest"], + "confirmation_digest": operation["confirmation_digest"], + "claim_id": operation["claim"]["claim_id"], + "executor_revision": operation["executor_revision"], + "consumption_id": consumption_id, + "outcome": result, + "projection_verified": True, + "simulation": False, + "external_write_performed": result != "not_executed", + "evidence_refs": ["receipt:synthetic-fixture-1"], + "summary": "Synthetic recorded execution evidence.", + "observed_at": datetime.now(timezone.utc).isoformat(), + } + + +@pytest.mark.parametrize( + "field,value", + [ + ("revision", "future-revision"), + ("host_surface", "unregistered-host"), + ("thread_id", "invalid thread"), + ("extra_authority", True), + ], +) +def test_agent_executor_invalid_binding_is_a_bounded_validation_error( + tmp_path: Path, field: str, value: object +) -> None: + service, store = _service(tmp_path) + request = _agent_request(service) + request["normalized_parameters"]["executor"][field] = value + before = store.path.read_bytes() if store.path.exists() else None + with pytest.raises(ValueError): + service.preview(request) + assert (store.path.read_bytes() if store.path.exists() else None) == before + + +def test_agent_handoff_requires_confirmation_and_original_session( + tmp_path: Path, +) -> None: + service, store = _service(tmp_path) + proposal = service.preview(_agent_request(service)) + runtime = store.root.parent.parent + args = dict(proposal_id=proposal["proposal_id"], actor=EXECUTION_ACTOR) + preview = agent_operation_action( + runtime, service.registry_path, action="inspect", **args + ) + assert ( + preview["status"] == "awaiting_confirmation" + and preview["execution_allowed"] is False + ) + with pytest.raises(ActionConflictError, match="authenticated confirmation"): + agent_operation_action( + runtime, + service.registry_path, + action="consume", + consumption_id="attempt-1", + **args, + ) + delivered = store.record_operation_delivery( + proposal["proposal_id"], delivery=_delivery(proposal) + ) + claimed = store.decide_operation( + proposal["proposal_id"], + decision="confirm", + confirmation=_confirmation(delivered), + ) + with pytest.raises(ActionConflictError, match="original bound session"): + agent_operation_action( + runtime, + service.registry_path, + action="consume", + consumption_id="attempt-1", + **{**args, "actor": {**EXECUTION_ACTOR, "thread_id": "thread-other"}}, + ) + with pytest.raises(ActionConflictError, match="not been consumed"): + store.observe_operation_outcome( + proposal["proposal_id"], + outcome=_agent_result(claimed, "attempt-1"), + agent_actor=EXECUTION_ACTOR, + agent_binding_current=True, + ) + assert not store.load(proposal["proposal_id"])["operation"].get("agent_handoff") + + +def test_agent_handoff_one_shot_consumption_survives_concurrent_retry_and_restart( + tmp_path: Path, +) -> None: + service, store = _service(tmp_path) + proposal = _claim_agent_operation(service, store) + runtime = store.root.parent.parent + inbox = pending(runtime, GOAL_ID, EXECUTION_ACTOR["agent_id"]) + assert len(inbox["operation_handoffs"]) == 1 + assert inbox["operation_handoffs"][0]["host_delivery"] == "not_attempted" + assert inbox["operation_handoffs"][0]["execution_allowed"] is False + assert "operation_handoffs" not in pending(runtime, GOAL_ID, "different-agent") + + def consume(index: int) -> dict: + return agent_operation_action( + runtime, + service.registry_path, + proposal_id=proposal["proposal_id"], + actor=EXECUTION_ACTOR, + action="consume", + consumption_id=f"attempt-{index}", + ) + + with ThreadPoolExecutor(max_workers=4) as workers: + receipts = list(workers.map(consume, range(4))) + first = [r for r in receipts if r["execution_allowed"]] + assert len(first) == 1 + assert all( + r["status"] == "already_consumed" + for r in receipts + if not r["execution_allowed"] + ) + restarted = ChatActionStore(store.root) + handoff = restarted.load(proposal["proposal_id"])["operation"]["agent_handoff"] + assert consume(99)["execution_allowed"] is False + assert ( + restarted.load(proposal["proposal_id"])["operation"]["agent_handoff"] == handoff + ) + assert pending(runtime, GOAL_ID, EXECUTION_ACTOR["agent_id"])["operation_handoffs"][ + 0 + ]["needs_reconciliation"] + result = _agent_result(proposal, first[0]["consumption_id"]) + observed = agent_operation_action( + runtime, + service.registry_path, + proposal_id=proposal["proposal_id"], + actor=EXECUTION_ACTOR, + action="report", + outcome=result, + ) + assert observed["execution_allowed"] is False and observed["outcome"] == result + assert "operation_handoffs" not in pending( + runtime, GOAL_ID, EXECUTION_ACTOR["agent_id"] + ) + assert consume(99)["execution_allowed"] is False + + +@pytest.mark.parametrize("change", ["expires", "rebound", "stopped", "payload"]) +def test_agent_handoff_fails_closed_on_expiry_binding_activation_or_terms_drift( + tmp_path: Path, change: str +) -> None: + service, store = _service(tmp_path) + proposal = _claim_agent_operation(service, store) + if change in {"rebound", "stopped"}: + registry = json.loads(service.registry_path.read_text()) + if change == "rebound": + registry["goals"][0]["coordination"]["thread_agent_bindings"][0][ + "thread_id" + ] = "thread-replacement" + else: + registry["goals"][0]["activation_state"] = "stopped" + service.registry_path.write_text(json.dumps(registry)) + else: + data = json.loads(store.path.read_text()) + stored = data["proposals"][proposal["proposal_id"]] + if change == "expires": + stored["operation"]["expires_at"] = "2000-01-01T00:00:00Z" + else: + stored["normalized_parameters"]["payload"]["quantity"] = "2.00" + store.path.write_text(json.dumps(data)) + before = store.path.read_bytes() + with pytest.raises(ActionConflictError): + agent_operation_action( + store.root.parent.parent, + service.registry_path, + proposal_id=proposal["proposal_id"], + actor=EXECUTION_ACTOR, + action="consume", + consumption_id="attempt-1", + ) + assert store.path.read_bytes() == before + + +@pytest.mark.parametrize( + "field,value", + [ + ("consumption_id", "different-attempt"), + ("payload_digest", "0" * 64), + ("confirmation_digest", "0" * 64), + ("claim_id", "different-claim"), + ("executor_revision", "future-revision"), + ("simulation", True), + ("evidence_refs", []), + ("external_write_performed", False), + ], +) +def test_agent_result_is_bound_to_one_consumed_operation( + tmp_path: Path, field: str, value: object +) -> None: + service, store = _service(tmp_path) + proposal = _claim_agent_operation(service, store) + runtime = store.root.parent.parent + args = dict(proposal_id=proposal["proposal_id"], actor=EXECUTION_ACTOR) + consumed = agent_operation_action( + runtime, + service.registry_path, + action="consume", + consumption_id="attempt-1", + **args, + ) + outcome = {**_agent_result(proposal, consumed["consumption_id"]), field: value} + before = store.path.read_bytes() + with pytest.raises(ActionConflictError): + agent_operation_action( + runtime, service.registry_path, action="report", outcome=outcome, **args + ) + assert store.path.read_bytes() == before + + +def test_unknown_result_never_grants_resubmission_and_late_evidence_does_not_require_active_goal( + tmp_path: Path, +) -> None: + service, store = _service(tmp_path) + proposal = _claim_agent_operation(service, store) + runtime = store.root.parent.parent + args = dict(proposal_id=proposal["proposal_id"], actor=EXECUTION_ACTOR) + agent_operation_action( + runtime, + service.registry_path, + action="consume", + consumption_id="attempt-1", + **args, + ) + registry = json.loads(service.registry_path.read_text()) + registry["goals"][0]["activation_state"] = "stopped" + registry["goals"][0]["coordination"]["thread_agent_bindings"][0]["thread_id"] = ( + "thread-replacement" + ) + service.registry_path.write_text(json.dumps(registry)) + outcome = _agent_result(proposal, "attempt-1", result="submission_unknown") + reported = agent_operation_action( + runtime, service.registry_path, action="report", outcome=outcome, **args + ) + assert reported["outcome"] == outcome + readback = agent_operation_action( + runtime, service.registry_path, action="inspect", **args + ) + assert readback["needs_reconciliation"] and not readback["execution_allowed"] + assert not readback["binding_current"] + from loopx.control_plane.collaboration.goal_instance_scope import ( + collaboration_goal_scope, + ) + + with collaboration_goal_scope( + service.registry_path, + goal_id=GOAL_ID, + agents=(EXECUTION_ACTOR["agent_id"],), + require_active=False, + ) as scope: + inbox = pending(runtime, GOAL_ID, EXECUTION_ACTOR["agent_id"], scope=scope) + assert inbox["operation_handoffs"][0]["needs_reconciliation"] + assert inbox["operation_handoffs"][0]["binding_current"] is False + + +def test_unknown_result_stays_in_original_inbox_past_expiry_until_bound_reconciliation( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + service, store = _service(tmp_path) + proposal = _claim_agent_operation(service, store) + runtime = store.root.parent.parent + args = dict(proposal_id=proposal["proposal_id"], actor=EXECUTION_ACTOR) + agent_operation_action( + runtime, + service.registry_path, + action="consume", + consumption_id="attempt-1", + **args, + ) + unknown = _agent_result(proposal, "attempt-1", result="submission_unknown") + report = agent_operation_action( + runtime, service.registry_path, action="report", outcome=unknown, **args + ) + assert report["status"] == "submission_unknown" and report["needs_reconciliation"] + assert agent_operation_action( + runtime, + service.registry_path, + action="consume", + consumption_id="attempt-1", + **args, + )["needs_reconciliation"] + for _ in range(2): + assert ( + pending(runtime, GOAL_ID, EXECUTION_ACTOR["agent_id"])[ + "operation_handoffs" + ][0]["status"] + == "submission_unknown" + ) + after_expiry = (datetime.now(timezone.utc) + timedelta(hours=2)).isoformat() + monkeypatch.setattr("loopx.chat_action_store._utc_now", lambda: after_expiry) + projected = pending(runtime, GOAL_ID, EXECUTION_ACTOR["agent_id"])[ + "operation_handoffs" + ][0] + assert projected["needs_reconciliation"] and not projected["execution_allowed"] + hook = turn_start_hook( + runtime, service.registry_path, GOAL_ID, EXECUTION_ACTOR["agent_id"] + ).producer() + assert hook["agent_read_required"] and hook["observation_count"] == 1 + final = _agent_result(proposal, "attempt-1", result="not_executed") + with pytest.raises(ActionConflictError, match="exact original unknown result"): + agent_operation_action( + runtime, service.registry_path, action="report", outcome=final, **args + ) + final["reconciles_outcome_digest"] = _digest(unknown) + settled = agent_operation_action( + runtime, service.registry_path, action="report", outcome=final, **args + ) + assert settled["outcome"] == final and not settled["needs_reconciliation"] + assert settled["execution_allowed"] is False + updated = store.load(proposal["proposal_id"]) + assert updated["operation"]["outcome"] == unknown + assert updated["operation"]["reconciliation"] == final + assert "operation_handoffs" not in pending( + runtime, GOAL_ID, EXECUTION_ACTOR["agent_id"] + ) + assert not agent_operation_action( + runtime, + service.registry_path, + action="consume", + consumption_id="attempt-2", + **args, + )["execution_allowed"] + + +def test_lifecycle_only_source_profile_cannot_acquire_new_operation_authority( + tmp_path: Path, +) -> None: + service, store = _service(tmp_path) + registry = json.loads(service.registry_path.read_text()) + registry.update( + profile_id="source_session_v1", + session_bindings=[], + session_receipts=[], + lifetime_receipts=[], + ) + first_instance = "ginst_aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa" + registry["goals"][0]["goal_instance_id"] = first_instance + service.registry_path.write_text(json.dumps(registry)) + request = _agent_request(service) + before = store.path.read_bytes() if store.path.exists() else None + with pytest.raises(ValueError, match="lifecycle-only profile"): + service.preview(request) + assert (store.path.read_bytes() if store.path.exists() else None) == before + + +def test_inbox_uses_shared_recovery_priority_and_explicit_overflow( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + from copy import deepcopy + + service, store = _service(tmp_path) + original = _claim_agent_operation(service, store) + proposals = [] + for index in range(22): + projected = deepcopy(original) + projected["proposal_id"] = projected["operation"]["operation_id"] = ( + f"operation-{index:02}" + ) + if index == 21: + projected["operation"].update( + lifecycle_state="outcome_observed", + agent_handoff={"consumption_id": "attempt-1"}, + outcome={"outcome": "submission_unknown"}, + ) + proposals.append(projected) + monkeypatch.setattr(ChatActionStore, "list", lambda *args, **kwargs: proposals) + inbox = pending(store.root.parent.parent, GOAL_ID, EXECUTION_ACTOR["agent_id"]) + assert len(inbox["operation_handoffs"]) == 20 + assert inbox["operation_handoffs"][0]["operation_id"] == "operation-21" + assert inbox["operation_handoff_pending_count"] == 22 + assert inbox["operation_handoff_overflow"] == { + "reason": "attention_page_capacity", + "count": 2, + "next_operation_id": "operation-19", + "instruction": "Inspect the next original operation by id; do not treat this page as the entire inbox.", + } def _digest(value: object) -> str: From bdc6b3396dd21270137d06b693d137635aae1773 Mon Sep 17 00:00:00 2001 From: huangruiteng <14976749+huangruiteng@users.noreply.github.com> Date: Wed, 30 Sep 2026 03:58:01 +0800 Subject: [PATCH 02/13] feat(presentation): show confirmed handoff and unknown-result boundaries Signed-off-by: huangruiteng <14976749+huangruiteng@users.noreply.github.com> --- .../smoke/action-review-plan-smoke.ts | 17 +++ .../personal-workspace/context-drawer.tsx | 12 +- .../src/features/personal-workspace/i18n.tsx | 18 +++ .../personal-workspace-contract.test.mjs | 5 +- .../personal-workspace-page.tsx | 23 +++- .../personal-workspace/personal-workspace.css | 2 + .../human-confirmed-domain-operations-v0.md | 107 +++++++++++++++++- ...an-confirmed-domain-operations-v0.zh-CN.md | 81 ++++++++++++- 8 files changed, 256 insertions(+), 9 deletions(-) diff --git a/apps/presentation/dashboard/smoke/action-review-plan-smoke.ts b/apps/presentation/dashboard/smoke/action-review-plan-smoke.ts index f9137aed4f..ef6c532e8d 100644 --- a/apps/presentation/dashboard/smoke/action-review-plan-smoke.ts +++ b/apps/presentation/dashboard/smoke/action-review-plan-smoke.ts @@ -121,3 +121,20 @@ check(operationPlan.interaction === "gated", "Operation execution keeps its auth check(operationPlan.operationFrame?.kind === "confirmation", "Dashboard consumes the shared confirmation frame"); check(operationPlan.operationFrame?.interactionMode === "confirm_reject", "The shared frame preserves confirm/reject interaction"); check(operationPlan.operationFrame?.content.fields[0]?.value === "Limit · GTC", "The shared frame preserves bounded projection fields"); + +const agentPending = typedActionProposalSchema.parse({...operationProposal, status: "applying", + normalized_parameters: {...operationProposal.normalized_parameters, executor: {kind: "agent_session"}}, + operation: {...operationProposal.operation, lifecycle_state: "claimed", agent_handoff: {consumption_id: "attempt-1"}}}); +const agentPendingFrame = compileActionReviewPlan(agentPending).operationFrame; +check(agentPendingFrame?.kind === "pending" && agentPendingFrame.executionState === "consumed_outcome_pending", + "Transport retains the original consumption; it does not imply an external result"); +const unknownAgentResult = typedActionProposalSchema.parse({...agentPending, status: "applied", + receipt: {projection_verified: true}, operation: {...agentPending.operation, lifecycle_state: "outcome_observed", + outcome: {outcome: "submission_unknown", simulation: false}, result_delivery: {outcome_stage: "initial"}}}); +check(compileActionReviewPlan(unknownAgentResult).interaction === "repair", "Delivered unknown submission is not completion"); +const reconciledAgentResult = typedActionProposalSchema.parse({...unknownAgentResult, + operation: {...unknownAgentResult.operation, reconciliation: {outcome: "not_executed", simulation: false}}}); +check(compileActionReviewPlan(reconciledAgentResult).interaction === "repair", "Old card delivery cannot certify a new reconciliation"); +check(compileActionReviewPlan({...reconciledAgentResult, operation: {...reconciledAgentResult.operation, + result_delivery: {outcome_stage: "reconciled"}}}).interaction === "completed", "Only reconciled card readback completes presentation"); +console.log("PASS: original-Agent handoff and append-only reconciliation survive the frontend transport"); diff --git a/apps/presentation/dashboard/src/features/personal-workspace/context-drawer.tsx b/apps/presentation/dashboard/src/features/personal-workspace/context-drawer.tsx index ddafdcf533..aa4cebab83 100644 --- a/apps/presentation/dashboard/src/features/personal-workspace/context-drawer.tsx +++ b/apps/presentation/dashboard/src/features/personal-workspace/context-drawer.tsx @@ -283,11 +283,16 @@ export function ContextDrawer({ agents, attentionHistory = [], onSelectAttention }; }, [closeDrawer, selection.kind]); + const operationUnknown = selection.kind === "proposal" + && selection.item.reviewPlan?.operationFrame?.kind === "result" + && selection.item.reviewPlan.operationFrame.resultKind === "unknown"; const title = selection.kind === "attention" ? t("drawer.titleAttention") : selection.kind === "todo" ? t("drawer.taskDetails") : selection.kind === "run" ? t("drawer.runDetails") : selection.kind === "output" ? t("drawer.titleOutput") : selection.kind === "proposal" && selection.item.actionKind === "team.plan" && selection.item.status === "applied" ? t("proposal.teamPlan.resultTitle") + : selection.kind === "proposal" && selection.item.actionKind === "operation.execute" + && selection.item.reviewPlan?.operationFrame?.kind !== "confirmation" ? t("drawer.operationReadOnly") : selection.kind === "proposal" ? t(selection.item.reviewPlan?.retryOriginal ? "drawer.recoverEditResult" : selection.item.status === "applied" ? "drawer.titleProposalApplied" : "drawer.titleProposalConfirm") : selection.kind === "schedule" ? (selection.item.scheduleKind === "heartbeat" ? "Heartbeat" : t("drawer.titleSchedule")) : t("drawer.goalDetails"); @@ -300,6 +305,7 @@ export function ContextDrawer({ agents, attentionHistory = [], onSelectAttention : selection.kind === "goal" ? selection.item.title : selection.kind === "schedule" ? t("drawer.goalAutoRun") : selection.kind === "proposal" && selection.item.status === "applied" && selection.item.actionKind === "team.plan" ? selection.item.goalId ?? t("drawer.currentGoal") + : selection.kind === "proposal" && selection.item.actionKind === "operation.execute" ? t("drawer.operationCanonicalStatus") : selection.item.goalId ? t("drawer.goalChanges") : t("drawer.managerChanges"); const selectedGoalRun = selection.kind === "goal" ? runs.find((run) => run.goalId === selection.item.goalId && Boolean(run.sessionId)) @@ -1109,7 +1115,9 @@ export function ContextDrawer({ agents, attentionHistory = [], onSelectAttention {selection.item.actionKind} · {selection.item.status}

{selection.item.title}

{selection.item.impact ?

{selection.item.impact}

: null} - {selection.item.reviewPlan && !selection.item.reviewPlan.retryOriginal && selection.item.actionKind !== "team.plan" ?

{selection.item.actionKind === "operation.execute" && selection.item.status === "gated" + {selection.item.reviewPlan && !selection.item.reviewPlan.retryOriginal && selection.item.actionKind !== "team.plan" ?

{operationUnknown + ? t("actionReview.operation_reconcile_original") + : selection.item.actionKind === "operation.execute" && selection.item.status === "gated" ? t("actionReview.operation_group_confirmation") : selection.item.actionKind === "operation.execute" && selection.item.reviewPlan.reason === "readback_unverified" ? t("actionReview.operation_result_delivery_pending") @@ -1120,7 +1128,7 @@ export function ContextDrawer({ agents, attentionHistory = [], onSelectAttention {selection.item.status === "applied" && selection.item.actionKind !== "team.plan" ?

{selection.item.actionKind === "operation.execute" ? selection.item.primaryLabel : t("drawer.proposalApplied")}

: null} {selection.item.status === "applied" && selection.item.actionKind !== "operation.execute" && selection.item.goalId ? : null} {selection.item.status === "stale" ?

{t("drawer.proposalStale")}

: null} - {selection.item.status === "error" && !selection.item.reviewPlan?.retryOriginal ?
{selection.item.reviewPlan?.reason === "readback_unverified" ? t("actionReview.readback_unverified") : t("drawer.proposalApplyFailed")}{selection.item.errorMessage ? {selection.item.errorMessage} : null}{t(selection.item.actionKind === "team.plan" ? "proposal.teamPlan.retryHint" : "drawer.proposalApplyFailedHint")}
: null} + {selection.item.status === "error" && !selection.item.reviewPlan?.retryOriginal ?
{operationUnknown ? t("proposal.operationState.submission_unknown") : selection.item.reviewPlan?.reason === "readback_unverified" ? t("actionReview.readback_unverified") : t("drawer.proposalApplyFailed")}{selection.item.errorMessage ? {selection.item.errorMessage} : null}{t(operationUnknown ? "actionReview.operation_reconcile_original" : selection.item.actionKind === "team.plan" ? "proposal.teamPlan.retryHint" : "drawer.proposalApplyFailedHint")}
: null} {selection.item.status === "rejected" ?

{t("drawer.proposalRejected")}

: null} {selection.item.status === "deferred" ?

{t("drawer.proposalDeferred")}

: null} {selection.item.status === "gated" ?
{selection.item.actionKind === "operation.execute" ? selection.item.primaryLabel : selection.item.workspaceCandidates?.length ? selection.item.title : t("drawer.gateRequiresHost")}{selection.item.actionKind === "operation.execute" || selection.item.workspaceCandidates?.length ? selection.item.impact : t("drawer.gateRequiresHostDescription")}{selection.item.gate?.nextAction ? {selection.item.gate.nextAction} : null}
: null} diff --git a/apps/presentation/dashboard/src/features/personal-workspace/i18n.tsx b/apps/presentation/dashboard/src/features/personal-workspace/i18n.tsx index aef9e1058e..1f7fa30342 100644 --- a/apps/presentation/dashboard/src/features/personal-workspace/i18n.tsx +++ b/apps/presentation/dashboard/src/features/personal-workspace/i18n.tsx @@ -376,6 +376,8 @@ const en = { "drawer.titleOutput": "Output details", "drawer.titleProposalApplied": "Execution result", "drawer.titleProposalConfirm": "Confirm execution", + "drawer.operationReadOnly": "Original operation status", + "drawer.operationCanonicalStatus": "Read-only canonical operation; no new execution", "drawer.titleSchedule": "Scheduled check", "drawer.unconfigured": "Not configured", "drawer.workspaceCandidates": "Available workspaces", @@ -674,6 +676,9 @@ const en = { "proposal.field.objective": "Objective", "proposal.field.operation": "Operation", "proposal.field.operationState": "Operation state", + "proposal.operationState.authorized_pending": "Confirmed; waiting for the original Agent", + "proposal.operationState.consumed_outcome_pending": "Authorization consumed; waiting for the real result", + "proposal.operationState.submission_unknown": "Result unknown; reconcile the original operation, do not resubmit", "proposal.field.resultDelivery": "Result delivery", "proposal.field.confirmationBoundary": "Confirmation boundary", "proposal.field.expiresAt": "Expires at", @@ -698,6 +703,7 @@ const en = { "actionReview.apply_pending": "Execution is in progress. Wait for its result before retrying.", "actionReview.readback_verified": "The action completed and its resulting state was verified.", "actionReview.readback_unverified": "The action returned without verified readback. Completion is not confirmed; recheck the state.", + "actionReview.operation_reconcile_original": "Keep the original request. The original Agent must reconcile venue evidence; confirmation or expiry cannot grant another submission.", "actionReview.operation_group_confirmation": "This exact request can only be confirmed on its original card in the bound Feishu group. The Dashboard does not expose a local execution control.", "actionReview.operation_result_delivery_pending": "The operation outcome was recorded, but the original group result card has not passed readback verification yet.", "drawer.recoverEditResult": "Recover operation result", @@ -713,6 +719,9 @@ const en = { "proposal.impact.lifecycleStop": "After confirmation, automatic progress stops and the Goal moves to the collapsed Stopped list. History, Todos, and evidence remain available for resuming.", "proposal.impact.protected": "This action must be completed through the protected LoopX write service.", "proposal.impact.operation": "The exact terms are read-only here. Confirm or reject the same immutable request in the bound Feishu group; confirmation consumes one canonical claim.", + "proposal.impact.operationAuthorized": "Human confirmation is recorded. The original Agent must read and consume this exact authorization once; no external result is recorded yet.", + "proposal.impact.operationConsumed": "The authorization has been consumed. Wait for original external evidence; a retry or lost response must not grant another submission.", + "proposal.impact.operationUnknown": "Submission may have had an external effect. Reconcile the original operation using its evidence; do not resubmit or treat card delivery as execution completion.", "proposal.primary.apply": "Confirm and apply", "proposal.primary.goalCreate": "Create Goal and start first run", "proposal.primary.lifecycleDelete": "Delete Goal", @@ -1555,6 +1564,8 @@ const zhCN: Record = { "drawer.titleOutput": "产出详情", "drawer.titleProposalApplied": "执行结果", "drawer.titleProposalConfirm": "确认执行", + "drawer.operationReadOnly": "原操作状态", + "drawer.operationCanonicalStatus": "只读规范操作,不产生新执行", "drawer.titleSchedule": "定时检查", "drawer.unconfigured": "未配置", "drawer.workspaceCandidates": "可选择的工作区", @@ -1853,6 +1864,9 @@ const zhCN: Record = { "proposal.field.objective": "目标", "proposal.field.operation": "操作", "proposal.field.operationState": "操作状态", + "proposal.operationState.authorized_pending": "已确认,等待原 Agent 接手", + "proposal.operationState.consumed_outcome_pending": "授权已消费,等待真实结果", + "proposal.operationState.submission_unknown": "结果未知;核对原操作,不可重复提交", "proposal.field.resultDelivery": "结果回传", "proposal.field.confirmationBoundary": "确认边界", "proposal.field.expiresAt": "过期时间", @@ -1877,6 +1891,7 @@ const zhCN: Record = { "actionReview.apply_pending": "正在执行,请等待读回结果后再重试。", "actionReview.readback_verified": "操作已完成,结果状态已通过读回验证。", "actionReview.readback_unverified": "操作返回但未通过读回验证。尚不能确认完成,请重新检查状态。", + "actionReview.operation_reconcile_original": "保留原请求;原 Agent 须按平台原始证据对账。确认或到期均不会授予再次提交许可。", "actionReview.operation_group_confirmation": "这份精确请求只能在已绑定飞书群的原始卡片确认;Dashboard 不提供本地执行入口。", "actionReview.operation_result_delivery_pending": "操作结果已经记录,但原群结果卡尚未通过回读核验。", "drawer.recoverEditResult": "恢复操作结果", @@ -1892,6 +1907,9 @@ const zhCN: Record = { "proposal.impact.lifecycleStop": "确认后会停止自动推进,并将 Goal 移入折叠的「已停止」列表;历史、Todo 和证据都会保留,可随时恢复。", "proposal.impact.protected": "该操作需要通过受保护的 LoopX 写入服务完成。", "proposal.impact.operation": "这里仅展示同一份不可变条款。请在已绑定的飞书群确认或拒绝;确认只会消费一个规范 claim。", + "proposal.impact.operationAuthorized": "用户确认已记录。原 Agent 须读取并一次消费这份精确授权;目前尚无外部执行结果。", + "proposal.impact.operationConsumed": "授权已消费。等待原始外部证据;重试或响应丢失均不得重新授予提交许可。", + "proposal.impact.operationUnknown": "提交可能已产生外部副作用。须以原始证据核对原操作,不可重提,也不能把卡片投递当作执行完成。", "proposal.primary.apply": "确认并应用", "proposal.primary.goalCreate": "创建 Goal 并开始首轮", "proposal.primary.lifecycleDelete": "删除 Goal", diff --git a/apps/presentation/dashboard/src/features/personal-workspace/personal-workspace-contract.test.mjs b/apps/presentation/dashboard/src/features/personal-workspace/personal-workspace-contract.test.mjs index 9307c196ba..05dc90e905 100644 --- a/apps/presentation/dashboard/src/features/personal-workspace/personal-workspace-contract.test.mjs +++ b/apps/presentation/dashboard/src/features/personal-workspace/personal-workspace-contract.test.mjs @@ -109,10 +109,13 @@ for (const state of ["delivered", "verification_required", "explicit_unverified" assert.match(returnDelivery, new RegExp(state), `Return delivery renders ${state}`); } assert.doesNotMatch(returnDelivery, /message_ref|provider_receipt|intent_digest/, "Provider-private locator facts never enter the return status badge"); -assert.match(actionReview, /proposal\.action_kind !== "operation\.execute" \|\| objectValue\(objectValue\(proposal\.operation\)\?\.result_delivery\) !== null/, "An operation is not complete in the Dashboard until result delivery is verified"); +assert.match(actionReview, /proposal\.action_kind !== "operation\.execute" \|\| \(operationFrame\?\.kind === "result" && operationFrame\.resultDeliveryVerified\)/, "An operation is not complete in the Dashboard until the current result delivery is verified"); assert.match(page, /reviewPlan\.operationFrame/, "Dashboard operation details consume the shared TS review frame"); assert.match(page, /operation\.execute" && proposal\.status === "applied"/, "Dashboard restores terminal operation receipts from the canonical action store"); assert.match(page, /proposal\.action_kind !== "operation\.execute"[\s\S]*reviewPlan\.interaction !== "completed"/, "Pending operation result-card readback remains visible instead of becoming a generic apply error"); +assert.match(page, /operationFrame\?\.kind === "result"[\s\S]*operationFrame\.resultKind === "unknown"/, "Unknown operations survive workspace restoration and generic error-card filtering"); +assert.match(styles, /\.personal-proposal-row\[data-action-kind="operation\.execute"\]\s*\{\s*grid-template-columns:\s*36px minmax\(0, 1fr\);/, "Operation safety labels cannot take an unbounded third column from the request terms"); +assert.match(styles, /\.personal-proposal-row\[data-action-kind="operation\.execute"\] > b\s*\{\s*grid-column:\s*2;\s*overflow-wrap:\s*anywhere;/, "Long operation status remains fully visible on its own wrapping row"); assert.match(drawer, /selection\.item\.actionKind !== "operation\.execute"/, "Dashboard hides generic local controls for authenticated group operations"); assert.match(dashboard, /response\.protected_action/, "Agent semantic protected intent is projected only after the Chat response"); assert.match(dashboard, /normalizedMessage\.includes\(normalizedTarget\)/, "A model-invented protected target cannot reach typed preview"); diff --git a/apps/presentation/dashboard/src/features/personal-workspace/personal-workspace-page.tsx b/apps/presentation/dashboard/src/features/personal-workspace/personal-workspace-page.tsx index 7df0bfb510..8de52d1ed5 100644 --- a/apps/presentation/dashboard/src/features/personal-workspace/personal-workspace-page.tsx +++ b/apps/presentation/dashboard/src/features/personal-workspace/personal-workspace-page.tsx @@ -554,7 +554,11 @@ function operationProposalFields( { key: "operation_state", label: t("proposal.field.operationState"), - value: frame?.lifecycleState ?? proposal.status, + value: frame?.kind === "pending" && frame.executionState + ? t(`proposal.operationState.${frame.executionState}`) + : frame?.kind === "result" && frame.resultKind === "unknown" + ? t("proposal.operationState.submission_unknown") + : frame?.lifecycleState ?? proposal.status, }, ...(frame?.kind === "result" ? [{ key: "result_delivery", @@ -651,7 +655,11 @@ function workspaceProposal(proposal: TypedActionProposal, t: WorkspaceTranslate) : proposalFields(proposal.normalized_parameters, t), goalId: typeof proposal.normalized_parameters.goal_id === "string" ? proposal.normalized_parameters.goal_id : undefined, impact: reviewPlan.retryOriginal ? t(`actionReview.${reviewPlan.reason}`) : proposal.action_kind === "operation.execute" - ? t("proposal.impact.operation") + ? operationFrame?.kind === "pending" && operationFrame.executionState + ? t(operationFrame.executionState === "consumed_outcome_pending" + ? "proposal.impact.operationConsumed" : "proposal.impact.operationAuthorized") + : operationFrame?.kind === "result" && operationFrame.resultKind === "unknown" + ? t("proposal.impact.operationUnknown") : t("proposal.impact.operation") : proposal.action_kind === "team.plan" ? proposal.status === "applied" ? t("proposal.teamPlan.assignedHint") : t("proposal.impact.teamPlan") : proposal.action_kind === "goal.create" @@ -681,7 +689,11 @@ function workspaceProposal(proposal: TypedActionProposal, t: WorkspaceTranslate) } : undefined, workspaceCandidates, primaryLabel: reviewPlan.retryOriginal ? t("drawer.retryOriginal") : proposal.action_kind === "operation.execute" - ? operationFrame?.kind === "result" + ? operationFrame?.kind === "pending" && operationFrame.executionState + ? t(`proposal.operationState.${operationFrame.executionState}`) + : operationFrame?.kind === "result" && operationFrame.resultKind === "unknown" + ? t("proposal.operationState.submission_unknown") + : operationFrame?.kind === "result" ? operationFrame.resultDeliveryVerified ? t("proposal.primary.operationResultVerified") : t("proposal.primary.operationResultPending") @@ -697,7 +709,8 @@ function workspaceProposal(proposal: TypedActionProposal, t: WorkspaceTranslate) : proposal.action_kind === "todo.create" && proposal.normalized_parameters.start_execution === true ? t("proposal.primary.todoStart") : t("proposal.primary.apply"), - status: reviewPlan.retryOriginal ? "error" : proposal.status === "applied" + status: reviewPlan.retryOriginal || (operationFrame?.kind === "result" && operationFrame.resultKind === "unknown") + ? "error" : proposal.status === "applied" && proposal.action_kind !== "operation.execute" && reviewPlan.interaction !== "completed" ? "error" @@ -942,6 +955,8 @@ export function PersonalWorkspacePage({ .filter((item) => item.kind !== "proposal" || !["stale", "error"].includes(item.proposal.status) || item.proposal.reviewPlan?.retryOriginal === true + || (item.proposal.reviewPlan?.operationFrame?.kind === "result" + && item.proposal.reviewPlan.operationFrame.resultKind === "unknown") || sessionProposalIds.includes(item.proposal.previewId)); return projected.filter((item) => { if (!selectedGoalId) return true; diff --git a/apps/presentation/dashboard/src/features/personal-workspace/personal-workspace.css b/apps/presentation/dashboard/src/features/personal-workspace/personal-workspace.css index dc72547fac..64c896e4f9 100644 --- a/apps/presentation/dashboard/src/features/personal-workspace/personal-workspace.css +++ b/apps/presentation/dashboard/src/features/personal-workspace/personal-workspace.css @@ -756,6 +756,8 @@ button.personal-execution-chip:focus-visible { outline: 2px solid #0070f3; outli .personal-proposal-row strong { font-size: 13px; } .personal-proposal-row p { margin: 0; color: #686f7c; font-size: 11px; } .personal-proposal-row > b { color: #315fc8; font-size: 11px; } +.personal-proposal-row[data-action-kind="operation.execute"] { grid-template-columns: 36px minmax(0, 1fr); } +.personal-proposal-row[data-action-kind="operation.execute"] > b { grid-column: 2; overflow-wrap: anywhere; } .personal-proposal-row.is-applied { border-color: #b5ddcc; background: #f3fbf7; } .personal-proposal-row.is-error, .personal-proposal-row.is-stale { border-color: #efc3c3; background: #fff7f7; } .personal-proposal-row.is-gated { border-color: #ead39c; background: #fffaf0; } diff --git a/docs/architecture/rfcs/human-confirmed-domain-operations-v0.md b/docs/architecture/rfcs/human-confirmed-domain-operations-v0.md index 18d8a28942..1b1a45cb5c 100644 --- a/docs/architecture/rfcs/human-confirmed-domain-operations-v0.md +++ b/docs/architecture/rfcs/human-confirmed-domain-operations-v0.md @@ -5,7 +5,7 @@ - **Delivery maturity:** Proposal - **Authors / owners:** LoopX maintainers and optional domain-provider maintainers - **Created:** 2026-09-12 -- **Last normative revision:** 2026-09-12 +- **Last normative revision:** 2026-09-30 - **Implementation baseline:** `72e557586` - **Related contracts:** [Extensions](../../reference/extensions.md), [Effect interpreter](agent-loop-effect-interpreter-v0.md) @@ -17,6 +17,8 @@ the contract; a difference in their requirements or boundaries is a defect. Sections 1–11 define the proposed contract, not shipped commands. Section 12 records unresolved implementation choices. This document changes no runtime, default permission, configuration or user entry point. +Section 13 describes the original-Agent continuation implementation slice; +its deployment and live acceptance remain separate from local validation. ## 1. Decision summary @@ -289,3 +291,106 @@ extract the actual shared seam, not a speculative adapter framework. 4. **Deployment qualification:** verify the actual Lark app callback and authenticated web-owner mechanism. A healthy event process alone is not evidence that either user path works. Required before M1 acceptance. + +## 13. Original-Agent continuation slice + +For an existing, user-authorized Agent that already owns a domain's browser or +adapter workflow, do not require a new API credential path just to return an +exact human confirmation to that Agent. This is an alternative execution seam, +not a relaxation of the financial preflight, account-wide constraints or +original-source evidence requirements above. Core does not interpret a price, +choose a venue, resume a browser, sign or submit an order. + +### Owners and entry points + +- The original `operation.execute` proposal in `chat/actions/actions.json` + remains the only confirmation, claim, consumption and outcome store. +- The explicit executor shape is `{kind: "agent_session", host_surface, + thread_id, revision: "agent-session-handoff-v0"}`. Preparation checks the + original registry's exact Goal/registered-Agent/session binding and rejects + simulation masquerading as real execution. The lifecycle-only + `source_session_v1` registry currently rejects business-operation preparation; + this slice does not bypass that owner or enable a replacement instance. +- `operation_agent_handoff.ts` owns admission, one-shot consumption, + reconciliation binding and bounded Inbox attention. Python supplies locked + canonical storage, original-registry facts and existing lifecycle guards; + it is not another decision owner. +- Existing Lark prepare/deliver and authenticated callback handling are reused. + A successful confirmation leaves the original proposal claimed, with no + external result. Callback replay, simulator and card-delivery recovery must + never invoke this Agent's browser or adapter. +- The existing manager Inbox projects locators directly from canonical + operations, without a copied approval record. Turn-start hooks include them + in `agent_read_required`. Consumed/unknown obligations sort before unconsumed + tickets; a 20-item page reports total count, typed overflow reason and next + operation ID rather than silently dropping work. Inspect that ID directly. +- Dashboard details and the original Lark card use the shared operation frame: + confirmed/waiting for the original Agent; consumed/waiting for real evidence; + unknown/reconcile without resubmitting; and a separately verified result. + The Dashboard remains read-only for human operation confirmation. There is + no new configuration owner: the original request chooses the executor and + the existing channel/binding owner remains authoritative. + +### Original-runtime CLI + +Use the original registry and runtime, not a copied session or another home's +records. These local continuation commands do not require Lark to be installed +or reachable. Preparing/delivering new cards remains subject to its normal +extension and authenticated-ingress checks. + +```sh +loopx --registry REGISTRY --runtime-root RUNTIME goal-channel inspect-operation \ + --goal-id GOAL --agent-id AGENT --proposal-id OPERATION \ + --host-surface HOST --thread-id ORIGINAL_THREAD +loopx --registry REGISTRY --runtime-root RUNTIME goal-channel consume-operation \ + --goal-id GOAL --agent-id AGENT --proposal-id OPERATION \ + --host-surface HOST --thread-id ORIGINAL_THREAD \ + --consumption-id STABLE_ATTEMPT --execute +loopx --registry REGISTRY --runtime-root RUNTIME goal-channel report-operation \ + --goal-id GOAL --agent-id AGENT --proposal-id OPERATION \ + --host-surface HOST --thread-id ORIGINAL_THREAD --outcome-json OUTCOME --execute +``` + +`inspect-operation` and a command without `--execute` never consume authority. +Only the first successful atomic consumption returns `execution_allowed: true`. +It verifies authenticated confirmation, immutable terms, the current original +session, active Goal and expiry, then persists consumption before any browser +effect. Every retry, including the same attempt after a lost response or +restart, returns no execution permission. This intentionally does not promise +exactly-once venue execution: ambiguity requires original-venue reconciliation. + +The original Agent reports `loopx_operation_outcome_v0` with the exact operation, +payload and confirmation digests, claim, executor revision, consumption ID, +`projection_verified: true`, `simulation: false`, bounded original evidence +references and separate `external_write_performed`. Outcomes are `executed`, +`not_executed` or `submission_unknown`. Unknown conservatively reports a possible +external effect; a transport success cannot certify a trade or protection order. +References must be safe opaque receipt identifiers, not credentials or private +absolute paths. Domain evidence and private trading journals retain detail. + +### Recovery, delivery and remaining acceptance + +Consumed or unknown operations remain recovery obligations after expiry and +session rebinding; those changes cannot grant a new execution. Evidence-only +reporting may continue for a stopped/historical Goal through existing lifecycle +guards. Existing exact-instance lifecycle guards remain in place, not an +implicit migration into lifecycle-only registries or another home. Missing +original Goal/Agent registration or an unsupported registry profile is an +explicit error, not permission to transplant the operation. + +An unknown original outcome is immutable. A definitive report appends +`operation.reconciliation` and binds `reconciles_outcome_digest` to the exact +original unknown result. Only that evidence closes the recovery obligation. +The original result card is updated by existing delivery recovery; an earlier +unknown-result delivery cannot certify the reconciled result. Its readback must +match the current `initial` or `reconciled` stage. None of these paths resubmits. + +The slice does **not** implement immediate host wakeup. Inbox visibility reports +`host_delivery: "not_attempted"`; existing Turn/heartbeat reads are not a host +delivery receipt. Host wakeup must later reuse the original host transport and +publish truthful attempt/readback evidence, without starting a parallel resumed +session. Local synthetic callback, CLI, concurrency, expiry, reconciliation and +packaged-UI checks establish protocol behavior only. Installation, a genuine +group click, original-Agent receipt consumption, real venue/protection evidence +and original-card readback are still required before claiming a live minimum +loop. No synthetic engineering card is sent to a live group for acceptance. diff --git a/docs/architecture/rfcs/human-confirmed-domain-operations-v0.zh-CN.md b/docs/architecture/rfcs/human-confirmed-domain-operations-v0.zh-CN.md index 93bc78f3fa..a5dfc42d5c 100644 --- a/docs/architecture/rfcs/human-confirmed-domain-operations-v0.zh-CN.md +++ b/docs/architecture/rfcs/human-confirmed-domain-operations-v0.zh-CN.md @@ -5,7 +5,7 @@ - **交付成熟度:** 提案 - **作者 / 负责人:** LoopX 维护者与可选垂域 provider 维护者 - **创建日期:** 2026-09-12 -- **最近规范修订:** 2026-09-12 +- **最近规范修订:** 2026-09-30 - **实现基线:** `72e557586` - **相关契约:** [扩展](../../reference/extensions.md)、 [Effect interpreter](agent-loop-effect-interpreter-v0.md) @@ -15,6 +15,7 @@ 第 1–11 节定义拟议契约,并非已发布的命令。第 12 节记录尚未解决的实现选择。 本文不修改运行时、默认权限、配置或用户入口。 +第 13 节记录原 Agent 续接的实现切片;部署与真实验收独立于本地验证。 ## 1. 决策摘要 @@ -232,3 +233,81 @@ M1 同时涵盖 UI 与后端,不要拆成“后端 PR 已完成”而遗忘前 不把签名加入现有行情采集器。由 adapter 维护者在 M2 决定。 4. **部署资格:** 核实真实飞书应用回调和 Web owner 身份认证机制。 事件进程健康本身不能证明任一用户路径可用。M1 验收前必须完成。 + +## 13. 原 Agent 续接切片 + +已有用户授权的 Agent 若已掌握垂域浏览器或 adapter 工作流,不必仅为将精确确认 +回传给它而新增 API 凭据路径。这是另一种执行接缝,不放松上文的金融提交前检查、 +账户级约束或原始证据要求。Core 不解释价格、不选平台、不恢复浏览器、不签名、不下单。 + +### 权威与用户入口 + +- `chat/actions/actions.json` 中原始 `operation.execute` 提案仍是确认、claim、 + 消费及结果的唯一存储。 +- 显式执行器形状为 `{kind: "agent_session", host_surface, thread_id, + revision: "agent-session-handoff-v0"}`。准备时核对原 registry 中精确的 + Goal/已注册 Agent/session 绑定,拒绝以模拟冒充真实执行。生命周期专用的 + `source_session_v1` registry 目前拒绝业务操作准备;本切片不绕过该 owner,也不为 + 替换的 instance 开启业务权限。 +- `operation_agent_handoff.ts` 管理准入、一次消费、对账绑定及有界 Inbox 注意力。 + Python 提供锁定的规范存储、原 registry 事实及现有生命周期保护,不另建决策源。 +- 复用现有 Lark prepare/deliver 与经认证的回调。确认成功后原提案保持 claimed, + 没有外部结果。回调重放、模拟器、卡片投递恢复均不得调用该 Agent 的浏览器或 adapter。 +- 原 manager Inbox 直接从规范操作投影定位信息,不复制审批记录。Turn-start hook + 将它计入 `agent_read_required`。已消费/未知结果义务先于未消费票据展示;每页 20 条 + 之外明确给出总数量、类型化 overflow 原因及下一条操作 ID,不静默丢弃工作;可按该 + ID 直接检查原操作。 +- Dashboard 详情与原 Lark 卡使用同一操作 frame:已确认待原 Agent、已消费待真实 + 证据、未知须对账不得重提,以及独立核验的结果。Dashboard 对用户操作确认保持只读。 + 不新增配置权威:执行器由原请求选择,原渠道/绑定 owner 仍是权威。 + +### 原运行时 CLI + +使用原 registry 和 runtime,不复制 session,也不借用另一个 home 的记录。 +以下本地续接命令不要求 Lark 已安装或可达;新卡片的准备/投递仍须通过原扩展与 +经认证入口的检查。 + +```sh +loopx --registry REGISTRY --runtime-root RUNTIME goal-channel inspect-operation \ + --goal-id GOAL --agent-id AGENT --proposal-id OPERATION \ + --host-surface HOST --thread-id ORIGINAL_THREAD +loopx --registry REGISTRY --runtime-root RUNTIME goal-channel consume-operation \ + --goal-id GOAL --agent-id AGENT --proposal-id OPERATION \ + --host-surface HOST --thread-id ORIGINAL_THREAD \ + --consumption-id STABLE_ATTEMPT --execute +loopx --registry REGISTRY --runtime-root RUNTIME goal-channel report-operation \ + --goal-id GOAL --agent-id AGENT --proposal-id OPERATION \ + --host-surface HOST --thread-id ORIGINAL_THREAD --outcome-json OUTCOME --execute +``` + +`inspect-operation` 和没有 `--execute` 的命令均不消费授权。只有首次成功的原子消费 +返回 `execution_allowed: true`。它检查经认证的确认、不可变条款、原 session 当前绑定、 +有效 Goal 与到期时间,并在任何浏览器操作之前持久记录消费。所有重试,包括响应丢失 +或重启后使用同一 attempt,均不再获得执行许可。这不承诺平台恰好执行一次;存在歧义 +必须按原平台证据对账。 + +原 Agent 回写 `loopx_operation_outcome_v0`,绑定精确 operation、载荷与确认摘要、 +claim、执行器 revision、consumption ID,要求 `projection_verified: true`、 +`simulation: false`、有界原始证据引用,并单独声明 `external_write_performed`。 +结果为 `executed`、`not_executed` 或 `submission_unknown`。未知保守声明可能存在外部 +副作用;传输成功不能证明交易或保护单。引用应为安全的不透明回执标识,不能含凭据或 +私有绝对路径;垂域证据及私有交易日记保留详细材料。 + +### 恢复、投递与剩余验收 + +已消费或未知操作在到期或 session 重绑后仍是恢复义务;这些变化不能授予新执行。 +仅回写证据可通过既有生命周期保护继续用于已停止/历史 Goal。现有精确 instance +生命周期保护保持不变,不隐式迁入生命周期专用 registry 或另一个 home。 +原 Goal/Agent 注册缺失或 registry profile 不受支持均明确报错,不允许移植操作权限。 + +未知的原始 outcome 不可修改。确定性回写追加 `operation.reconciliation`,并通过 +`reconciles_outcome_digest` 绑定原未知结果的精确摘要;只有该证据才关闭恢复义务。 +现有投递恢复更新原结果卡,旧未知结果的投递不能证明新的对账结果;读回必须匹配当前 +`initial` 或 `reconciled` 阶段。上述路径均不会重新提交。 + +本切片**没有实现主机即时唤醒**。Inbox 可见性如实返回 +`host_delivery: "not_attempted"`;既有 Turn/heartbeat 读取不等于主机投递回执。 +后续主机唤醒应复用原宿主传输,给出真实尝试/读回证据,不启动平行的 resumed session。 +本地合成回调、CLI、并发、到期、对账及打包 UI 检查只证明协议行为。宣称真实最小闭环 +之前,仍须安装、真实群确认、原 Agent 消费回执、真实平台/保护证据及原卡片读回。 +不得向真实群发送合成工程卡来冒充验收。 From 0b028548292ef67cf6269842709df2cd7240627d Mon Sep 17 00:00:00 2001 From: huangruiteng <14976749+huangruiteng@users.noreply.github.com> Date: Wed, 30 Sep 2026 04:30:16 +0800 Subject: [PATCH 03/13] fix(operations): fence local session consumption and enumerate recovery pages Signed-off-by: huangruiteng <14976749+huangruiteng@users.noreply.github.com> --- .../src/features/personal-workspace/i18n.tsx | 2 +- .../human-confirmed-domain-operations-v0.md | 22 +- ...an-confirmed-domain-operations-v0.zh-CN.md | 17 +- loopx/cli_commands/_host_thread.py | 15 +- loopx/cli_commands/goal_channel_operation.py | 32 ++- loopx/cli_commands/manager_inbox.py | 6 +- .../collaboration/goal_instance_scope.py | 23 +- loopx/control_plane/collaboration/inbox.py | 6 +- .../collaboration/operation_handoff.py | 41 +++- loopx/control_plane/collaboration/peers.py | 2 + .../control_plane/effect_runtime_handlers.ts | 3 +- .../work_items/operation_agent_handoff.ts | 44 +++- .../operation_agent_handoff.test.ts | 21 +- .../test_lark_goal_channel_operation.py | 40 ++++ tests/test_chat_operation_actions.py | 207 ++++++++++++++++-- 15 files changed, 418 insertions(+), 63 deletions(-) diff --git a/apps/presentation/dashboard/src/features/personal-workspace/i18n.tsx b/apps/presentation/dashboard/src/features/personal-workspace/i18n.tsx index 1f7fa30342..5551ce0471 100644 --- a/apps/presentation/dashboard/src/features/personal-workspace/i18n.tsx +++ b/apps/presentation/dashboard/src/features/personal-workspace/i18n.tsx @@ -703,7 +703,7 @@ const en = { "actionReview.apply_pending": "Execution is in progress. Wait for its result before retrying.", "actionReview.readback_verified": "The action completed and its resulting state was verified.", "actionReview.readback_unverified": "The action returned without verified readback. Completion is not confirmed; recheck the state.", - "actionReview.operation_reconcile_original": "Keep the original request. The original Agent must reconcile venue evidence; confirmation or expiry cannot grant another submission.", + "actionReview.operation_reconcile_original": "Keep the original request. The original Agent must reconcile original external evidence; confirmation or expiry cannot grant another submission.", "actionReview.operation_group_confirmation": "This exact request can only be confirmed on its original card in the bound Feishu group. The Dashboard does not expose a local execution control.", "actionReview.operation_result_delivery_pending": "The operation outcome was recorded, but the original group result card has not passed readback verification yet.", "drawer.recoverEditResult": "Recover operation result", diff --git a/docs/architecture/rfcs/human-confirmed-domain-operations-v0.md b/docs/architecture/rfcs/human-confirmed-domain-operations-v0.md index 1b1a45cb5c..e1c662afa6 100644 --- a/docs/architecture/rfcs/human-confirmed-domain-operations-v0.md +++ b/docs/architecture/rfcs/human-confirmed-domain-operations-v0.md @@ -323,7 +323,11 @@ choose a venue, resume a browser, sign or submit an order. operations, without a copied approval record. Turn-start hooks include them in `agent_read_required`. Consumed/unknown obligations sort before unconsumed tickets; a 20-item page reports total count, typed overflow reason and next - operation ID rather than silently dropping work. Inspect that ID directly. + operation ID plus an independent `operation_handoff_next_cursor`. Continue + with `manager-inbox read --operation-cursor CURSOR` even when the first 20 + unknown outcomes remain unresolved. The cursor is bound to the original + runtime/Goal/Agent scope, not the ordinary request cursor. Restart without it + for new/changed work; finishing a page sequence does not resolve obligations. - Dashboard details and the original Lark card use the shared operation frame: confirmed/waiting for the original Agent; consumed/waiting for real evidence; unknown/reconcile without resubmitting; and a separately verified result. @@ -352,12 +356,28 @@ loopx --registry REGISTRY --runtime-root RUNTIME goal-channel report-operation \ ``` `inspect-operation` and a command without `--execute` never consume authority. +All three CLI commands require the existing host-exported ambient thread +(for example, `CODEX_THREAD_ID`) to match the original route; CLI identifiers +are selectors, not caller authentication. Missing, foreign or unsupported host +context fails closed before inspecting private parameters or writing a receipt. +The returned `caller_context_source: trusted_local_host_environment` names a +trusted-local-OS-user fence, **not cryptographic session isolation**. A hostile +process that can forge the environment or rewrite the same user's canonical +files is outside this slice. Do not advertise exclusive execution across +untrusted processes; that requires an independently qualified authenticated +host transport. Internal storage adapters accept host-validated actor facts, +not unauthenticated network requests. Only the first successful atomic consumption returns `execution_allowed: true`. It verifies authenticated confirmation, immutable terms, the current original session, active Goal and expiry, then persists consumption before any browser effect. Every retry, including the same attempt after a lost response or restart, returns no execution permission. This intentionally does not promise exactly-once venue execution: ambiguity requires original-venue reconciliation. +Binding/registration/activation read and consumption commit hold the existing +registry-writer lock, in Goal-lifetime → registry → action-store order. +Revocation that commits first prevents consumption; revocation waiting behind +a consumption cannot retroactively revoke the already committed receipt. The +lock is released before external work and never claims to fence that work. The original Agent reports `loopx_operation_outcome_v0` with the exact operation, payload and confirmation digests, claim, executor revision, consumption ID, diff --git a/docs/architecture/rfcs/human-confirmed-domain-operations-v0.zh-CN.md b/docs/architecture/rfcs/human-confirmed-domain-operations-v0.zh-CN.md index a5dfc42d5c..5a5f775d7d 100644 --- a/docs/architecture/rfcs/human-confirmed-domain-operations-v0.zh-CN.md +++ b/docs/architecture/rfcs/human-confirmed-domain-operations-v0.zh-CN.md @@ -256,7 +256,10 @@ M1 同时涵盖 UI 与后端,不要拆成“后端 PR 已完成”而遗忘前 - 原 manager Inbox 直接从规范操作投影定位信息,不复制审批记录。Turn-start hook 将它计入 `agent_read_required`。已消费/未知结果义务先于未消费票据展示;每页 20 条 之外明确给出总数量、类型化 overflow 原因及下一条操作 ID,不静默丢弃工作;可按该 - ID 直接检查原操作。 + ID 直接检查原操作,并用独立的 `operation_handoff_next_cursor` 通过 + `manager-inbox read --operation-cursor CURSOR` 逐页找回其余操作,即使前 20 条未知 + 结果长期未解决。游标绑定原 runtime/Goal/Agent 范围,与普通请求游标独立;新增或 + 改变的工作应无游标重读,遍历结束不代表义务已解决。 - Dashboard 详情与原 Lark 卡使用同一操作 frame:已确认待原 Agent、已消费待真实 证据、未知须对账不得重提,以及独立核验的结果。Dashboard 对用户操作确认保持只读。 不新增配置权威:执行器由原请求选择,原渠道/绑定 owner 仍是权威。 @@ -280,11 +283,21 @@ loopx --registry REGISTRY --runtime-root RUNTIME goal-channel report-operation \ --host-surface HOST --thread-id ORIGINAL_THREAD --outcome-json OUTCOME --execute ``` -`inspect-operation` 和没有 `--execute` 的命令均不消费授权。只有首次成功的原子消费 +`inspect-operation` 和没有 `--execute` 的命令均不消费授权。三个 CLI 命令均要求 +既有宿主导出的环境线程(例如 `CODEX_THREAD_ID`)匹配原路由;命令行标识只是选择器, +不是调用者认证。环境缺失、不匹配或不支持的宿主在读取私有参数或写回之前即拒绝。 +返回的 `caller_context_source: trusted_local_host_environment` 明确表示可信本地 +OS 用户边界,**不等于密码学 session 隔离**。能伪造环境或改写同用户规范文件的恶意 +进程不在本切片保证内;不得宣称在不可信进程间独占执行,这需要另行验收经认证的宿主 +传输。内部存储 adapter 只接收宿主已验证的 actor 事实,不接受未认证的网络请求。 +只有首次成功的原子消费 返回 `execution_allowed: true`。它检查经认证的确认、不可变条款、原 session 当前绑定、 有效 Goal 与到期时间,并在任何浏览器操作之前持久记录消费。所有重试,包括响应丢失 或重启后使用同一 attempt,均不再获得执行许可。这不承诺平台恰好执行一次;存在歧义 必须按原平台证据对账。 +绑定/注册/启用状态读取和消费提交共享既有 registry writer 锁,顺序为 Goal 生命周期 +→ registry → action store。先提交的撤销阻止消费;排在消费之后的撤销不能追溯抹除 +已提交的回执。锁在外部执行前释放,不宣称约束其后的外部操作。 原 Agent 回写 `loopx_operation_outcome_v0`,绑定精确 operation、载荷与确认摘要、 claim、执行器 revision、consumption ID,要求 `projection_verified: true`、 diff --git a/loopx/cli_commands/_host_thread.py b/loopx/cli_commands/_host_thread.py index 2a8435e29c..dde98cef63 100644 --- a/loopx/cli_commands/_host_thread.py +++ b/loopx/cli_commands/_host_thread.py @@ -20,11 +20,18 @@ } +def ambient_host_thread_id(host_surface: str) -> str | None: + """Read the host-provided local context, never a caller's route override. + + This is a trusted-local-runtime fence, not authentication against another + process that can change this OS user's environment or canonical files. + """ + variable = HOST_THREAD_ID_ENV.get(host_surface) + return (os.environ.get(variable) or None) if variable else None + + def current_host_thread_id(args: argparse.Namespace) -> str | None: explicit = getattr(args, "thread_id", None) if explicit: return str(explicit) - variable = HOST_THREAD_ID_ENV.get(str(getattr(args, "host_surface", None) or "")) - if not variable: - return None - return os.environ.get(variable) or None + return ambient_host_thread_id(str(getattr(args, "host_surface", None) or "")) diff --git a/loopx/cli_commands/goal_channel_operation.py b/loopx/cli_commands/goal_channel_operation.py index 4e672cb3d3..7edc091b9c 100644 --- a/loopx/cli_commands/goal_channel_operation.py +++ b/loopx/cli_commands/goal_channel_operation.py @@ -14,6 +14,12 @@ from ..chat_action_store import ActionConflictError, ChatActionStore from ..chat_actions import ChatActionService from ..control_plane.collaboration.operation_handoff import agent_operation_action +from ..control_plane.effect_runtime import ( + EffectRuntimeConflict, + EffectRuntimeRejected, + effect_runtime_result, +) +from ._host_thread import ambient_host_thread_id from ..extensions.lark.goal_channel import ( default_goal_channel_target_path, deliver_goal_channel_operation_card, @@ -253,6 +259,24 @@ def run_goal_channel_operation( if isinstance(request, _AgentOperation): # These commands use the original registry/store, not Lark target # settings or a new approval source. No external executor is called. + try: + actor = effect_runtime_result( + "operation.agent_handoff.actor", + { + "requested": { + "goal_id": request.goal_id, + "agent_id": request.agent_id, + "host_surface": request.host_surface, + "thread_id": request.thread_id, + }, + "ambient": { + "host_surface": request.host_surface, + "thread_id": ambient_host_thread_id(request.host_surface), + }, + }, + ) + except (EffectRuntimeConflict, EffectRuntimeRejected) as exc: + raise ActionConflictError(str(exc)) from exc action = request.command.value.removesuffix("-operation") if not request.execute: action = "inspect" @@ -267,12 +291,7 @@ def run_goal_channel_operation( context.source_runtime_root, context.source_registry_path, proposal_id=request.proposal_id, - actor={ - "goal_id": request.goal_id, - "agent_id": request.agent_id, - "host_surface": request.host_surface, - "thread_id": request.thread_id, - }, + actor=actor, action=action, consumption_id=request.consumption_id, outcome=outcome, @@ -282,6 +301,7 @@ def run_goal_channel_operation( "goal_id": request.goal_id, "execute": request.execute, "operation": request.command.value.replace("-", "_"), + "caller_context_source": "trusted_local_host_environment", **result, } target_path = _operation_target_path(request, context) diff --git a/loopx/cli_commands/manager_inbox.py b/loopx/cli_commands/manager_inbox.py index 75a73249cb..2f05f74d3e 100644 --- a/loopx/cli_commands/manager_inbox.py +++ b/loopx/cli_commands/manager_inbox.py @@ -65,6 +65,7 @@ def register_manager_inbox(subparsers, add_format): parser.add_argument("--offset", type=int, default=0) parser.add_argument("--limit", type=int, default=8) parser.add_argument("--cursor", help="For read: continue with the previous page's next_cursor.") + parser.add_argument("--operation-cursor", help="For read: continue the independent original-operation page.") parser.add_argument("--decision", choices=("adopt", "defer", "reject", "no_change")) parser.add_argument("--reason") @@ -72,8 +73,11 @@ def register_manager_inbox(subparsers, add_format): def handle_manager_inbox(args, registry_path, runtime_root): try: cursor = getattr(args, "cursor", None) + operation_cursor = getattr(args, "operation_cursor", None) if cursor is not None and args.manager_inbox_action != "read": raise ValueError("--cursor is only supported for read") + if operation_cursor is not None and args.manager_inbox_action != "read": + raise ValueError("--operation-cursor is only supported for read") if args.manager_inbox_action == "configure-ssh-read-scope": from ..capabilities.manager_context.ssh_evidence import configure result = configure(runtime_root, channel=args.channel_id or "", host=args.ssh_host, @@ -161,7 +165,7 @@ def handle_manager_inbox(args, registry_path, runtime_root): elif args.manager_inbox_action == "read": from ..control_plane.collaboration.peers import read_inbox result = read_inbox(runtime_root, registry_path, args.goal_id, args.agent_id, - workspace=Path.cwd(), cursor=cursor) + workspace=Path.cwd(), cursor=cursor, operation_cursor=operation_cursor) result["followthrough"] = ( "After reading and deciding, associate Core work with manager-inbox link. Then use manager-inbox report --phase conclusion --reply-text to return this request's concrete result, replan decision, or explicit blocker/defer reason to its original audience automatically. Use optional --phase decision only for meaningful interim news during longer work. Adoption/linking alone is not a completed exchange. Do not wait for the owner to ask again. Write audience-ready text, not private deliberation." ) diff --git a/loopx/control_plane/collaboration/goal_instance_scope.py b/loopx/control_plane/collaboration/goal_instance_scope.py index 66581e212e..ec7246174f 100644 --- a/loopx/control_plane/collaboration/goal_instance_scope.py +++ b/loopx/control_plane/collaboration/goal_instance_scope.py @@ -1,7 +1,7 @@ from __future__ import annotations from collections.abc import Iterator -from contextlib import contextmanager +from contextlib import ExitStack, contextmanager from dataclasses import dataclass from pathlib import Path from typing import Any @@ -97,14 +97,27 @@ def collaboration_goal_scope( agents: tuple[str, ...], caller_goal_ref: dict[str, str] | None = None, require_active: bool = False, + lock_registry: bool = False, ) -> Iterator[CollaborationGoalScope]: """Hold the alias lifetime guard while one collaboration operation commits.""" registry_path = Path(registry_path).expanduser().resolve() - with exclusive_cross_runtime_file_lock( - guard_path(registry_path, goal_id), - operation="collaboration_goal_lifetime", - ): + with ExitStack() as guards: + guards.enter_context( + exclusive_cross_runtime_file_lock( + guard_path(registry_path, goal_id), + operation="collaboration_goal_lifetime", + ) + ) + if lock_registry: + # Same lock as thread binding transactions, held through the + # downstream commit. Order: Goal lifetime -> registry -> work store. + guards.enter_context( + exclusive_cross_runtime_file_lock( + registry_path, + operation="collaboration_registry_snapshot", + ) + ) registry = load_project_registry(registry_path) goal = _registered_goal( registry, diff --git a/loopx/control_plane/collaboration/inbox.py b/loopx/control_plane/collaboration/inbox.py index 176b31e8c0..b0fe839d0a 100644 --- a/loopx/control_plane/collaboration/inbox.py +++ b/loopx/control_plane/collaboration/inbox.py @@ -123,6 +123,7 @@ def pending( agent_id: str, *, cursor: str | None = None, + operation_cursor: str | None = None, scope: CollaborationGoalScope | None = None, ) -> dict: cursor_scope = _hash( @@ -217,6 +218,8 @@ def append_batch(): agent_id, registry_path=scope.registry_path if scope is not None else None, scope=scope, + cursor=operation_cursor, + cursor_scope=cursor_scope, ) return { "ok": True, @@ -225,8 +228,9 @@ def append_batch(): "operation_handoffs": operation_handoffs["items"], "operation_handoff_pending_count": operation_handoffs["pending_count"], "operation_handoff_overflow": operation_handoffs["overflow"], + "operation_handoff_next_cursor": operation_handoffs["next_cursor"], } - if operation_handoffs["items"] + if operation_handoffs["items"] or operation_cursor is not None else {} ), **({"peer_returns": peer_returns} if peer_returns["items"] else {}), diff --git a/loopx/control_plane/collaboration/operation_handoff.py b/loopx/control_plane/collaboration/operation_handoff.py index 63be399d2b..12a8d946be 100644 --- a/loopx/control_plane/collaboration/operation_handoff.py +++ b/loopx/control_plane/collaboration/operation_handoff.py @@ -47,13 +47,19 @@ def pending_operation_handoffs( *, registry_path: Path | None = None, scope: CollaborationGoalScope | None = None, + cursor: str | None = None, + cursor_scope: str, ) -> dict[str, Any]: """Project canonical tickets into the existing Inbox; do not copy authority.""" - if not (runtime_root / "chat" / "actions" / "actions.json").is_file(): - return {"items": [], "pending_count": 0, "overflow": None} - store = _store(runtime_root) + store = ( + _store(runtime_root) + if (runtime_root / "chat" / "actions" / "actions.json").is_file() + else None + ) + if store is None and cursor is None: + return {"items": [], "pending_count": 0, "overflow": None, "next_cursor": None} result = [] - for proposal in store.list(goal_id=goal_id): + for proposal in store.list(goal_id=goal_id) if store is not None else []: parameters = proposal.get("normalized_parameters") or {} executor = parameters.get("executor") or {} if ( @@ -89,17 +95,29 @@ def pending_operation_handoffs( "summary": proposal["summary"], "instruction": "Read the original canonical operation and consume it once before any external effect. " "Only the first successful consumption permits execution; consumed/unknown results require " - "original venue reconciliation, never another submission. Inbox delivery is not trade authority.", + "original external-system reconciliation, never another submission. Inbox delivery is not execution authority.", "next_action": "goal-channel consume-operation" if plan["status"] == "authorized_pending" else "Reconcile the original external result; do not submit again.", } ) - from ..effect_runtime import effect_runtime_result + if not result and cursor is None: + return {"items": [], "pending_count": 0, "overflow": None, "next_cursor": None} + from ..effect_runtime import EffectRuntimeRejected, effect_runtime_result - return dict( - effect_runtime_result("operation.agent_handoff.inbox", {"items": result}) - ) + try: + return dict( + effect_runtime_result( + "operation.agent_handoff.inbox", + { + "items": result, + "cursor": cursor, + "cursor_scope": cursor_scope, + }, + ) + ) + except EffectRuntimeRejected as exc: + raise ValueError(str(exc)) from exc def agent_operation_action( @@ -128,6 +146,7 @@ def agent_operation_action( agents=(actor["agent_id"],), caller_goal_ref=parameters.get("origin_goal_ref"), require_active=action == "consume", + lock_registry=action == "consume", ) as scope: if action == "consume": decide_collaboration_lifecycle(scope, operation="request_create") @@ -144,7 +163,9 @@ def agent_operation_action( ) current = _binding(registry_path, parameters) if action == "inspect": - plan = store._agent_operation_plan(proposal, action="project") + plan = store._agent_operation_plan( + proposal, action="inspect", actor=dict(actor) + ) return { **plan, "binding_current": current, diff --git a/loopx/control_plane/collaboration/peers.py b/loopx/control_plane/collaboration/peers.py index 4516c3cb80..f2e83fd375 100644 --- a/loopx/control_plane/collaboration/peers.py +++ b/loopx/control_plane/collaboration/peers.py @@ -560,6 +560,7 @@ def read_inbox( *, workspace=None, cursor=None, + operation_cursor=None, caller_goal_ref=None, ): from .inbox import pending @@ -576,6 +577,7 @@ def read_inbox( goal_id, agent_id, cursor=cursor, + operation_cursor=operation_cursor, scope=goal_scope, ) peer_returns = returns( diff --git a/loopx/control_plane/effect_runtime_handlers.ts b/loopx/control_plane/effect_runtime_handlers.ts index 939671b2de..69830c25a5 100644 --- a/loopx/control_plane/effect_runtime_handlers.ts +++ b/loopx/control_plane/effect_runtime_handlers.ts @@ -1,5 +1,5 @@ import {manageNewGoalStorage} from "./coordination/local_authority_defaults.ts"; -import {normalizeAgentOperationExecutor, planAgentOperationHandoff, projectAgentOperationInbox} from "./work_items/operation_agent_handoff.ts"; +import {deriveAgentOperationActor, normalizeAgentOperationExecutor, planAgentOperationHandoff, projectAgentOperationInbox} from "./work_items/operation_agent_handoff.ts"; import {projectDecisionNotice} from "./presentation/decision_notice.ts"; import {normalizeResearchObservation, validateResearchAttribution, projectResearchFrontier} from "./capabilities/explore_research.ts"; import {projectTodoSummary} from "./todos/summary_projection.ts"; @@ -632,6 +632,7 @@ export function createEffectRuntimeHandlers( ["presentation.action_review_plan.compile", (params) => compileActionReviewPlan(params.proposal)], ["operation.agent_executor.normalize", normalizeAgentOperationExecutor], + ["operation.agent_handoff.actor", deriveAgentOperationActor], ["operation.agent_handoff.plan", planAgentOperationHandoff], ["operation.agent_handoff.inbox", projectAgentOperationInbox], ["scheduler.monitor_successor.plan", planMonitorSuccessor], diff --git a/loopx/control_plane/work_items/operation_agent_handoff.ts b/loopx/control_plane/work_items/operation_agent_handoff.ts index f78de71a2f..dc7761685c 100644 --- a/loopx/control_plane/work_items/operation_agent_handoff.ts +++ b/loopx/control_plane/work_items/operation_agent_handoff.ts @@ -36,6 +36,20 @@ export function normalizeAgentOperationExecutor(input: JsonObject): JsonObject { thread_id: id(executor.thread_id, "thread_id"), revision: AGENT_OPERATION_REVISION}; } +/** CLI selectors are not caller identity. The transport supplies the existing + * host's ambient thread in the trusted local OS-user boundary; hostile local + * processes/file writers require a separate authenticated host transport. */ +export function deriveAgentOperationActor(input: JsonObject): JsonObject { + const requested = requireJsonObject(input.requested, "requested actor"); + const ambient = requireJsonObject(input.ambient, "ambient host context"); + requireThat(typeof ambient.thread_id === "string" && ambient.thread_id.length > 0, + "original host session context is unavailable; route flags are not caller identity"); + requireThat(requested.host_surface === ambient.host_surface && requested.thread_id === ambient.thread_id, + "requested route is not the current host session"); + return {goal_id: id(requested.goal_id, "goal_id"), agent_id: id(requested.agent_id, "agent_id"), + host_surface: id(ambient.host_surface, "host_surface"), thread_id: id(ambient.thread_id, "thread_id")}; +} + export function planAgentOperationHandoff(input: JsonObject): JsonObject { const proposal = requireJsonObject(input.proposal, "proposal"); const parameters = requireJsonObject(proposal.normalized_parameters, "parameters"); @@ -70,7 +84,12 @@ export function planAgentOperationHandoff(input: JsonObject): JsonObject { const observed = operation.reconciliation ?? operation.outcome; const unknownResult = observed != null && requireJsonObject(observed, "observed result").outcome === "submission_unknown"; - if (action === "project") { + if (action === "project" || action === "inspect") { + if (action === "inspect") { + const actor = requireJsonObject(input.actor, "inspection actor"); + requireThat(Object.entries(route).every(([key, value]) => actor[key] === value), + "inspection actor is not the original bound session"); + } return {...base, outcome_digest: digests.outcome_digest ?? null, status: unknownResult ? "submission_unknown" : operation.lifecycle_state === "outcome_observed" ? "outcome_observed" @@ -136,8 +155,23 @@ export function projectAgentOperationInbox(input: JsonObject): JsonObject { const rank = (item: JsonObject) => item.needs_reconciliation === true ? 0 : 1; items.sort((a, b) => rank(a) - rank(b) || String(a.operation_id).localeCompare(String(b.operation_id), "en")); - return {items: items.slice(0, 20), pending_count: items.length, - overflow: items.length > 20 ? {reason: "attention_page_capacity", count: items.length - 20, - next_operation_id: items[20].operation_id, - instruction: "Inspect the next original operation by id; do not treat this page as the entire inbox."} : null}; + const scope = requireNonEmptyString(input.cursor_scope, "operation cursor scope"); + if (!/^[a-f0-9]{64}$/.test(scope)) throw new EffectRuntimeRequestError("operation cursor scope is invalid"); + let remaining = items; + if (input.cursor != null) { + const cursor = requireNonEmptyString(input.cursor, "operation cursor"); + const match = /^op1:([a-f0-9]{64}):([01]):([A-Za-z0-9._:-]{1,200})$/.exec(cursor); + if (!match || match[1] !== scope) throw new EffectRuntimeRequestError("operation cursor scope mismatch"); + const afterRank = Number(match[2]); + remaining = items.filter(item => rank(item) > afterRank + || (rank(item) === afterRank && String(item.operation_id).localeCompare(match[3], "en") > 0)); + } + const page = remaining.slice(0, 20); + const last = page.at(-1); + const next = remaining.length > 20 && last + ? `op1:${scope}:${rank(last)}:${id(last.operation_id, "operation_id")}` : null; + return {items: page, pending_count: items.length, next_cursor: next, + overflow: remaining.length > 20 ? {reason: "attention_page_capacity", count: remaining.length - 20, + next_operation_id: remaining[20].operation_id, next_cursor: next, + instruction: "Continue with manager-inbox read --operation-cursor. Restart without that cursor to discover new or changed work; a page boundary is not completion."} : null}; } diff --git a/tests/control_plane_ts/operation_agent_handoff.test.ts b/tests/control_plane_ts/operation_agent_handoff.test.ts index 063688e9b7..f28ca67846 100644 --- a/tests/control_plane_ts/operation_agent_handoff.test.ts +++ b/tests/control_plane_ts/operation_agent_handoff.test.ts @@ -1,7 +1,7 @@ import assert from "node:assert/strict"; import test from "node:test"; import type {JsonObject} from "../../loopx/control_plane/effect_program.ts"; -import {AGENT_OPERATION_REVISION, normalizeAgentOperationExecutor, planAgentOperationHandoff, +import {AGENT_OPERATION_REVISION, deriveAgentOperationActor, normalizeAgentOperationExecutor, planAgentOperationHandoff, projectAgentOperationInbox} from "../../loopx/control_plane/work_items/operation_agent_handoff.ts"; function input(): JsonObject { @@ -22,6 +22,17 @@ function input(): JsonObject { } const operation = (value: JsonObject) => (value.proposal as JsonObject).operation as JsonObject; +test("caller selectors cannot replace missing or foreign host context", () => { + const requested = input().actor as JsonObject; + for (const thread_id of [null, "", "another-thread"]) { + assert.throws(() => deriveAgentOperationActor({requested, ambient: {host_surface: "codex-app", thread_id}})); + } + assert.throws(() => deriveAgentOperationActor({requested, + ambient: {host_surface: "unsupported-host", thread_id: "original-thread"}})); + assert.deepEqual(deriveAgentOperationActor({requested, + ambient: {host_surface: "codex-app", thread_id: "original-thread"}}), requested); +}); + test("only the first consumed canonical confirmation grants the original session execution", () => { const value = input(); const plan = planAgentOperationHandoff(value); @@ -81,11 +92,17 @@ test("unknown submission remains a reconciliation obligation after expiry with n test("bounded inbox retains recovery first and declares overflow instead of silently discarding it", () => { const items = Array.from({length: 22}, (_, index) => ({operation_id: `operation-${String(index).padStart(2, "0")}`, needs_reconciliation: index === 21, execution_allowed: false})); - const page = projectAgentOperationInbox({items}); + const cursor_scope = "a".repeat(64); + const page = projectAgentOperationInbox({items, cursor_scope}); assert.equal((page.items as JsonObject[]).length, 20); assert.equal((page.items as JsonObject[])[0].operation_id, "operation-21"); assert.equal(page.pending_count, 22); assert.equal((page.overflow as JsonObject).count, 2); assert.equal((page.overflow as JsonObject).next_operation_id, "operation-19"); assert.equal(items[0].operation_id, "operation-00"); + const rest = projectAgentOperationInbox({items, cursor_scope, cursor: page.next_cursor}); + assert.deepEqual((rest.items as JsonObject[]).map(item => item.operation_id), ["operation-19", "operation-20"]); + assert.equal(rest.next_cursor, null); + assert.equal(rest.pending_count, 22); + assert.throws(() => projectAgentOperationInbox({items, cursor_scope: "b".repeat(64), cursor: page.next_cursor})); }); diff --git a/tests/extensions/test_lark_goal_channel_operation.py b/tests/extensions/test_lark_goal_channel_operation.py index 388ffe088c..0c195354f1 100644 --- a/tests/extensions/test_lark_goal_channel_operation.py +++ b/tests/extensions/test_lark_goal_channel_operation.py @@ -4,6 +4,7 @@ from collections.abc import Mapping import hashlib import json +import os from pathlib import Path import subprocess import threading @@ -92,7 +93,9 @@ def _prepare_agent_handoff(store: ChatActionStore, registry: Path) -> dict[str, def test_authenticated_callback_hands_off_without_calling_any_executor_and_reconciles_original_result( tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, ) -> None: + monkeypatch.setenv("CODEX_THREAD_ID", "thread-operation-fixture") store, registry, runtime, binding, target = _fixture(tmp_path) proposal = _prepare_agent_handoff(store, registry) cards: dict[str, dict[str, Any]] = {} @@ -158,8 +161,17 @@ def no_executor(_proposal): proposal["proposal_id"] )["operation"].get("agent_handoff") args.execute = True + before = store.path.read_bytes() + for ambient_thread in ["", "thread-unrelated-fixture"]: + monkeypatch.setenv("CODEX_THREAD_ID", ambient_thread) + rejected = run_goal_channel_operation(args, context=context) + assert rejected["ok"] is False + assert rejected["blocker"] == "operation_handoff_conflict" + assert store.path.read_bytes() == before + monkeypatch.setenv("CODEX_THREAD_ID", "thread-operation-fixture") consumed = run_goal_channel_operation(args, context=context) assert consumed["execution_allowed"] is True + assert consumed["caller_context_source"] == "trusted_local_host_environment" assert ( run_goal_channel_operation(args, context=context)["execution_allowed"] is False ) @@ -283,12 +295,40 @@ def test_real_cli_inspects_canonical_handoff_in_source_runtime(tmp_path: Path) - capture_output=True, check=True, timeout=30, + env={**os.environ, "CODEX_THREAD_ID": "thread-operation-fixture"}, ) packet = json.loads(result.stdout) assert ( packet["status"] == "awaiting_confirmation" and packet["execution_allowed"] is False ) + before = store.path.read_bytes() + wrong_thread = subprocess.run( + [*result.args[:-1], "thread-other"], + text=True, + capture_output=True, + check=False, + timeout=30, + env={**os.environ, "CODEX_THREAD_ID": "thread-operation-fixture"}, + ) + assert wrong_thread.returncode == 1 + assert json.loads(wrong_thread.stdout)["blocker"] == "operation_handoff_conflict" + assert store.path.read_bytes() == before + for ambient_thread in ["", "thread-unrelated-fixture"]: + foreign_process = subprocess.run( + result.args, + text=True, + capture_output=True, + check=False, + timeout=30, + env={**os.environ, "CODEX_THREAD_ID": ambient_thread}, + ) + assert foreign_process.returncode == 1 + assert ( + json.loads(foreign_process.stdout)["blocker"] + == "operation_handoff_conflict" + ) + assert store.path.read_bytes() == before def _digest(value: object) -> str: diff --git a/tests/test_chat_operation_actions.py b/tests/test_chat_operation_actions.py index 36314b8f9d..813a432bab 100644 --- a/tests/test_chat_operation_actions.py +++ b/tests/test_chat_operation_actions.py @@ -48,8 +48,16 @@ def _agent_request(service: ChatActionService) -> dict[str, object]: return request -def _claim_agent_operation(service: ChatActionService, store: ChatActionStore) -> dict: - proposal = service.preview(_agent_request(service)) +def _claim_agent_operation( + service: ChatActionService, + store: ChatActionStore, + *, + idempotency_key: str | None = None, +) -> dict: + request = _agent_request(service) + if idempotency_key is not None: + request["idempotency_key"] = idempotency_key + proposal = service.preview(request) delivered = store.record_operation_delivery( proposal["proposal_id"], delivery=_delivery(proposal) ) @@ -117,6 +125,13 @@ def test_agent_handoff_requires_confirmation_and_original_session( preview["status"] == "awaiting_confirmation" and preview["execution_allowed"] is False ) + with pytest.raises(ActionConflictError, match="original bound session"): + agent_operation_action( + runtime, + service.registry_path, + action="inspect", + **{**args, "actor": {**EXECUTION_ACTOR, "thread_id": "thread-other"}}, + ) with pytest.raises(ActionConflictError, match="authenticated confirmation"): agent_operation_action( runtime, @@ -243,6 +258,95 @@ def test_agent_handoff_fails_closed_on_expiry_binding_activation_or_terms_drift( assert store.path.read_bytes() == before +def test_binding_revocation_and_consumption_share_the_registry_commit_boundary( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + from contextlib import contextmanager + from threading import Event + from loopx.control_plane.projects import registry_codec + from loopx.control_plane.collaboration import operation_handoff + from loopx.thread_agent_binding import unbind_thread_agent_in_registry + + service, store = _service(tmp_path) + proposal = _claim_agent_operation(service, store) + attempted, acquired = Event(), Event() + original_transaction = registry_codec._registry_transaction + original_binding = operation_handoff._binding + original_write = ChatActionStore._write + commits = [] + + @contextmanager + def observed_transaction(*args, **kwargs): + attempted.set() + with original_transaction(*args, **kwargs) as transaction: + acquired.set() + yield transaction + + def revoke(): + result = unbind_thread_agent_in_registry( + registry_path=service.registry_path, + goal_id=GOAL_ID, + host_surface=EXECUTION_ACTOR["host_surface"], + thread_id=EXECUTION_ACTOR["thread_id"], + agent_id=EXECUTION_ACTOR["agent_id"], + execute=True, + ) + assert result["written"] is True + commits.append("revocation") + + def record_consumption(self, payload): + original_write(self, payload) + commits.append("consumption") + + monkeypatch.setattr(registry_codec, "_registry_transaction", observed_transaction) + monkeypatch.setattr(ChatActionStore, "_write", record_consumption) + with ThreadPoolExecutor(max_workers=1) as executor: + futures = [] + + def interleaved_binding(*args): + current = original_binding(*args) + futures.append(executor.submit(revoke)) + assert attempted.wait(5) + assert not acquired.wait(0.2), ( + "revocation cannot commit between binding read and consumption" + ) + return current + + monkeypatch.setattr(operation_handoff, "_binding", interleaved_binding) + result = agent_operation_action( + store.root.parent.parent, + service.registry_path, + proposal_id=proposal["proposal_id"], + actor=EXECUTION_ACTOR, + action="consume", + consumption_id="attempt-1", + ) + assert result["execution_allowed"] is True + futures[0].result(timeout=5) + assert commits == ["consumption", "revocation"] + monkeypatch.setattr(operation_handoff, "_binding", original_binding) + before = store.path.read_bytes() + with pytest.raises(ActionConflictError, match="binding is no longer current"): + agent_operation_action( + store.root.parent.parent, + service.registry_path, + proposal_id=proposal["proposal_id"], + actor=EXECUTION_ACTOR, + action="consume", + consumption_id="attempt-2", + ) + assert store.path.read_bytes() == before + report = agent_operation_action( + store.root.parent.parent, + service.registry_path, + proposal_id=proposal["proposal_id"], + actor=EXECUTION_ACTOR, + action="report", + outcome=_agent_result(proposal, "attempt-1", result="not_executed"), + ) + assert report["execution_allowed"] is False + + @pytest.mark.parametrize( "field,value", [ @@ -415,36 +519,91 @@ def test_lifecycle_only_source_profile_cannot_acquire_new_operation_authority( def test_inbox_uses_shared_recovery_priority_and_explicit_overflow( - tmp_path: Path, monkeypatch: pytest.MonkeyPatch + tmp_path: Path, ) -> None: - from copy import deepcopy + import subprocess + import sys service, store = _service(tmp_path) - original = _claim_agent_operation(service, store) - proposals = [] + operation_ids = set() for index in range(22): - projected = deepcopy(original) - projected["proposal_id"] = projected["operation"]["operation_id"] = ( - f"operation-{index:02}" + proposal = _claim_agent_operation( + service, store, idempotency_key=f"overflow-{index}" + ) + operation_ids.add(proposal["proposal_id"]) + args = dict(proposal_id=proposal["proposal_id"], actor=EXECUTION_ACTOR) + agent_operation_action( + store.root.parent.parent, + service.registry_path, + action="consume", + consumption_id="attempt-1", + **args, + ) + agent_operation_action( + store.root.parent.parent, + service.registry_path, + action="report", + outcome=_agent_result(proposal, "attempt-1", result="submission_unknown"), + **args, ) - if index == 21: - projected["operation"].update( - lifecycle_state="outcome_observed", - agent_handoff={"consumption_id": "attempt-1"}, - outcome={"outcome": "submission_unknown"}, - ) - proposals.append(projected) - monkeypatch.setattr(ChatActionStore, "list", lambda *args, **kwargs: proposals) inbox = pending(store.root.parent.parent, GOAL_ID, EXECUTION_ACTOR["agent_id"]) assert len(inbox["operation_handoffs"]) == 20 - assert inbox["operation_handoffs"][0]["operation_id"] == "operation-21" assert inbox["operation_handoff_pending_count"] == 22 - assert inbox["operation_handoff_overflow"] == { - "reason": "attention_page_capacity", - "count": 2, - "next_operation_id": "operation-19", - "instruction": "Inspect the next original operation by id; do not treat this page as the entire inbox.", - } + assert inbox["operation_handoff_overflow"]["reason"] == "attention_page_capacity" + assert inbox["operation_handoff_overflow"]["count"] == 2 + cursor = inbox["operation_handoff_next_cursor"] + assert cursor == inbox["operation_handoff_overflow"]["next_cursor"] + before = store.path.read_bytes() + process = subprocess.run( + [ + sys.executable, + "-m", + "loopx.cli", + "--format", + "json", + "--registry", + str(service.registry_path), + "--runtime-root", + str(store.root.parent.parent), + "manager-inbox", + "read", + "--goal-id", + GOAL_ID, + "--agent-id", + EXECUTION_ACTOR["agent_id"], + "--operation-cursor", + cursor, + ], + text=True, + capture_output=True, + check=True, + timeout=30, + ) + rest = json.loads(process.stdout) + assert len(rest["operation_handoffs"]) == 2 + assert rest["operation_handoff_next_cursor"] is None + assert rest["operation_handoff_pending_count"] == 22 + assert { + item["operation_id"] + for item in [*inbox["operation_handoffs"], *rest["operation_handoffs"]] + } == operation_ids + assert all( + item["needs_reconciliation"] + for item in [*inbox["operation_handoffs"], *rest["operation_handoffs"]] + ) + assert store.path.read_bytes() == before + assert ( + len( + pending(store.root.parent.parent, GOAL_ID, EXECUTION_ACTOR["agent_id"])[ + "operation_handoffs" + ] + ) + == 20 + ) + with pytest.raises(ValueError, match="cursor scope mismatch"): + pending( + store.root.parent.parent, GOAL_ID, "other-agent", operation_cursor=cursor + ) def _digest(value: object) -> str: From af34f59fc6ff7a54bdbea923793640e8ae32c491 Mon Sep 17 00:00:00 2001 From: huangruiteng <14976749+huangruiteng@users.noreply.github.com> Date: Wed, 30 Sep 2026 05:50:27 +0800 Subject: [PATCH 04/13] fix(operations): recover historical results after session rebinding Signed-off-by: huangruiteng <14976749+huangruiteng@users.noreply.github.com> --- loopx/chat_action_store.py | 7 + .../collaboration/operation_handoff.py | 24 +- .../work_items/operation_agent_handoff.ts | 37 ++- .../project_registry_io_manifest_v1.json | 8 +- .../operation_agent_handoff.test.ts | 37 +++ .../test_lark_goal_channel_operation.py | 201 ++++++++++++++++ tests/test_chat_operation_actions.py | 220 ++++++++++++++++++ 7 files changed, 517 insertions(+), 17 deletions(-) diff --git a/loopx/chat_action_store.py b/loopx/chat_action_store.py index 5fcc922547..38cd7bca41 100644 --- a/loopx/chat_action_store.py +++ b/loopx/chat_action_store.py @@ -954,6 +954,7 @@ def observe_operation_outcome( outcome: Mapping[str, Any], agent_actor: Mapping[str, Any] | None = None, agent_binding_current: bool = False, + agent_actor_binding_current: bool = False, ) -> dict[str, Any]: """Persist the domain result without making it a retryable submission.""" @@ -979,14 +980,17 @@ def observe_operation_outcome( if not isinstance(operation, dict): raise KeyError("typed operation was not found") parameters = proposal.get("normalized_parameters") or {} + report_provenance = None if (parameters.get("executor") or {}).get("kind") == "agent_session": plan = self._agent_operation_plan( proposal, action="report", actor=agent_actor, binding_current=agent_binding_current, + actor_binding_current=agent_actor_binding_current, outcome=safe_outcome, ) + report_provenance = plan["report_provenance"] if plan.get("write_reconciliation") is True: existing = operation.get("reconciliation") if existing is not None: @@ -996,6 +1000,7 @@ def observe_operation_outcome( ) return proposal operation["reconciliation"] = safe_outcome + operation["reconciliation_report"] = report_provenance proposal["receipt"] = safe_outcome proposal["updated_at"] = _utc_now() self._write(payload) @@ -1014,6 +1019,8 @@ def observe_operation_outcome( now = _utc_now() operation["lifecycle_state"] = "outcome_observed" operation["outcome"] = safe_outcome + if report_provenance is not None: + operation["outcome_report"] = report_provenance proposal["status"] = "applied" proposal["receipt"] = safe_outcome proposal["applied_at"] = now diff --git a/loopx/control_plane/collaboration/operation_handoff.py b/loopx/control_plane/collaboration/operation_handoff.py index 12a8d946be..269ba3b18e 100644 --- a/loopx/control_plane/collaboration/operation_handoff.py +++ b/loopx/control_plane/collaboration/operation_handoff.py @@ -146,7 +146,9 @@ def agent_operation_action( agents=(actor["agent_id"],), caller_goal_ref=parameters.get("origin_goal_ref"), require_active=action == "consume", - lock_registry=action == "consume", + # Recovery-owner binding must stay valid through evidence commit too. + # Use the same Goal -> registry -> action-store ordering as consumption. + lock_registry=True, ) as scope: if action == "consume": decide_collaboration_lifecycle(scope, operation="request_create") @@ -162,9 +164,18 @@ def agent_operation_action( route=ref, ) current = _binding(registry_path, parameters) + actor_current = ( + _binding(registry_path, {**parameters, "executor": actor}) + if action in {"inspect", "report"} + else False + ) if action == "inspect": plan = store._agent_operation_plan( - proposal, action="inspect", actor=dict(actor) + proposal, + action="inspect", + actor=dict(actor), + binding_current=current, + actor_binding_current=actor_current, ) return { **plan, @@ -174,6 +185,10 @@ def agent_operation_action( "consumption": proposal["operation"].get("agent_handoff"), "outcome": proposal["operation"].get("outcome"), "reconciliation": proposal["operation"].get("reconciliation"), + "outcome_report": proposal["operation"].get("outcome_report"), + "reconciliation_report": proposal["operation"].get( + "reconciliation_report" + ), } if action == "consume": return store.consume_agent_operation( @@ -188,6 +203,7 @@ def agent_operation_action( outcome=outcome or {}, agent_actor=actor, agent_binding_current=current, + agent_actor_binding_current=actor_current, ) plan = store._agent_operation_plan(updated, action="project") return { @@ -195,5 +211,9 @@ def agent_operation_action( **plan, "outcome": updated["operation"].get("reconciliation") or updated["operation"]["outcome"], + "outcome_report": updated["operation"].get("outcome_report"), + "reconciliation_report": updated["operation"].get( + "reconciliation_report" + ), } raise ValueError("unsupported agent operation action") diff --git a/loopx/control_plane/work_items/operation_agent_handoff.ts b/loopx/control_plane/work_items/operation_agent_handoff.ts index dc7761685c..1c5f0af704 100644 --- a/loopx/control_plane/work_items/operation_agent_handoff.ts +++ b/loopx/control_plane/work_items/operation_agent_handoff.ts @@ -50,6 +50,21 @@ export function deriveAgentOperationActor(input: JsonObject): JsonObject { host_surface: id(ambient.host_surface, "host_surface"), thread_id: id(ambient.thread_id, "thread_id")}; } +/** A registry-authorized replacement may recover evidence, never inherit the + * immutable executor or consume an unspent authorization. Caller identity + * still requires a trusted host transport; a registry binding is not one. */ +function historicalAccess(input: JsonObject, route: JsonObject, consumed: boolean): JsonObject { + const actor = requireJsonObject(input.actor, "historical evidence actor"); + const original = Object.entries(route).every(([key, value]) => actor[key] === value); + requireThat(original || (consumed && input.binding_current === false && input.actor_binding_current === true + && actor.goal_id === route.goal_id && actor.agent_id === route.agent_id), + "historical evidence actor is not the original bound session or its current recovery owner"); + const owner = Object.fromEntries(Object.keys(route).map(key => [key, id(actor[key], `actor.${key}`)])); + return {mode: original ? "original_session" : "replacement_reconciliation", owner, + original_route: route, permission: "historical_evidence_only", execution_allowed: false, + authority_source: original ? "original_operation_route" : "current_registry_binding"}; +} + export function planAgentOperationHandoff(input: JsonObject): JsonObject { const proposal = requireJsonObject(input.proposal, "proposal"); const parameters = requireJsonObject(proposal.normalized_parameters, "parameters"); @@ -85,12 +100,8 @@ export function planAgentOperationHandoff(input: JsonObject): JsonObject { const unknownResult = observed != null && requireJsonObject(observed, "observed result").outcome === "submission_unknown"; if (action === "project" || action === "inspect") { - if (action === "inspect") { - const actor = requireJsonObject(input.actor, "inspection actor"); - requireThat(Object.entries(route).every(([key, value]) => actor[key] === value), - "inspection actor is not the original bound session"); - } - return {...base, outcome_digest: digests.outcome_digest ?? null, + const access = action === "inspect" ? historicalAccess(input, route, !!handoff) : null; + return {...base, ...(access ? {access} : {}), outcome_digest: digests.outcome_digest ?? null, status: unknownResult ? "submission_unknown" : operation.lifecycle_state === "outcome_observed" ? "outcome_observed" : handoff ? "consumed_outcome_pending" : now >= expires ? "expired" @@ -100,10 +111,10 @@ export function planAgentOperationHandoff(input: JsonObject): JsonObject { requireThat(confirmation?.decision === "confirm" && confirmation.confirmation_digest === operation.confirmation_digest && claim, "agent execution requires authenticated confirmation"); - const actor = requireJsonObject(input.actor, "execution actor"); - requireThat(Object.entries(route).every(([key, value]) => actor[key] === value), - "execution actor is not the original bound session"); if (action === "consume") { + const actor = requireJsonObject(input.actor, "execution actor"); + requireThat(Object.entries(route).every(([key, value]) => actor[key] === value), + "execution actor is not the original bound session"); requireThat(input.binding_current === true, "original session binding is no longer current"); // Even a same-id retry returns no execute permission. A lost response after // this commit is ambiguous, never permission to submit a second order. @@ -118,6 +129,9 @@ export function planAgentOperationHandoff(input: JsonObject): JsonObject { } if (action === "report") { requireThat(handoff, "operation authorization has not been consumed"); + const access = historicalAccess(input, route, true); + const report_provenance = {schema_version: "loopx_operation_report_provenance_v0", ...access, + operation_id: operation.operation_id, consumption_id: handoff.consumption_id, recorded_at: input.now}; const outcome = requireJsonObject(input.outcome, "operation outcome"); for (const key of ["operation_id", "payload_digest", "confirmation_digest", "claim_id", "executor_revision"]) { requireThat(outcome[key] === base[key], "operation outcome does not match the consumed authorization"); @@ -140,9 +154,10 @@ export function planAgentOperationHandoff(input: JsonObject): JsonObject { if (original?.outcome === "submission_unknown" && outcome.outcome !== "submission_unknown") { requireThat(outcome.reconciles_outcome_digest === digests.outcome_digest, "reconciliation must reference the exact original unknown result"); - return {...base, status: "outcome_observed", outcome, write_reconciliation: true}; + return {...base, access, report_provenance, status: "outcome_observed", outcome, write_reconciliation: true}; } - return {...base, status: outcome.outcome === "submission_unknown" ? "submission_unknown" : "outcome_observed", outcome}; + return {...base, access, report_provenance, + status: outcome.outcome === "submission_unknown" ? "submission_unknown" : "outcome_observed", outcome}; } throw new EffectRuntimeRequestError("unsupported agent operation action"); } diff --git a/loopx/semantics/project_registry_io_manifest_v1.json b/loopx/semantics/project_registry_io_manifest_v1.json index 5ff87ea248..bf1f48417b 100644 --- a/loopx/semantics/project_registry_io_manifest_v1.json +++ b/loopx/semantics/project_registry_io_manifest_v1.json @@ -223,7 +223,7 @@ }, { "site": "loopx/capabilities/manager_context/__init__.py::.configure_delivery_target::codec_read:load_project_registry#1", - "line": 406, + "line": 410, "column": 20, "kind": "codec_read", "api": "load_project_registry", @@ -231,7 +231,7 @@ }, { "site": "loopx/capabilities/manager_context/__init__.py::.configure_evidence_scope::codec_read:load_project_registry#1", - "line": 364, + "line": 368, "column": 16, "kind": "codec_read", "api": "load_project_registry", @@ -719,7 +719,7 @@ }, { "site": "loopx/cli_commands/manager_inbox.py::.handle_manager_inbox::codec_read:load_project_registry#1", - "line": 105, + "line": 109, "column": 20, "kind": "codec_read", "api": "load_project_registry", @@ -959,7 +959,7 @@ }, { "site": "loopx/control_plane/collaboration/goal_instance_scope.py::.collaboration_goal_scope::codec_read:load_project_registry#1", - "line": 108, + "line": 121, "column": 20, "kind": "codec_read", "api": "load_project_registry", diff --git a/tests/control_plane_ts/operation_agent_handoff.test.ts b/tests/control_plane_ts/operation_agent_handoff.test.ts index f28ca67846..1cbcd78279 100644 --- a/tests/control_plane_ts/operation_agent_handoff.test.ts +++ b/tests/control_plane_ts/operation_agent_handoff.test.ts @@ -89,6 +89,43 @@ test("unknown submission remains a reconciliation obligation after expiry with n assert.equal(final.write_reconciliation, true); }); +test("a currently bound replacement owns historical reconciliation, never the original consumption", () => { + const value = input(); + operation(value).agent_handoff = planAgentOperationHandoff(value).write_handoff; + operation(value).lifecycle_state = "outcome_observed"; + operation(value).outcome = {outcome: "submission_unknown"}; + const replacement = {...value.actor as JsonObject, thread_id: "replacement-thread"}; + const recovery = {...value, actor: replacement, binding_current: false, actor_binding_current: true, + now: "2031-01-01T00:00:00Z"}; + const inspected = planAgentOperationHandoff({...recovery, action: "inspect"}); + assert.deepEqual(inspected.route, value.actor); + assert.deepEqual((inspected.access as JsonObject).owner, replacement); + assert.equal((inspected.access as JsonObject).permission, "historical_evidence_only"); + assert.equal(inspected.execution_allowed, false); + assert.throws(() => planAgentOperationHandoff({...recovery, action: "consume", consumption_id: "attempt-2"})); + const outcome: JsonObject = {schema_version: "loopx_operation_outcome_v0", operation_id: "operation-1", + payload_digest: "payload", confirmation_digest: "confirmation", claim_id: "claim-1", + executor_revision: AGENT_OPERATION_REVISION, consumption_id: "attempt-1", outcome: "not_executed", + projection_verified: true, simulation: false, external_write_performed: false, + evidence_refs: ["receipt:original-system-reconciliation"], reconciles_outcome_digest: "unknown-result-digest"}; + const report = planAgentOperationHandoff({...recovery, action: "report", outcome}); + assert.equal(report.execution_allowed, false); + assert.equal(report.write_reconciliation, true); + assert.deepEqual((report.report_provenance as JsonObject).owner, replacement); + assert.deepEqual((report.report_provenance as JsonObject).original_route, value.actor); + for (const rejected of [ + {...recovery, actor_binding_current: false}, + {...recovery, binding_current: true}, + {...recovery, actor: {...replacement, agent_id: "other-agent"}}, + {...recovery, actor: {...replacement, goal_id: "other-goal"}}, + ]) { + assert.throws(() => planAgentOperationHandoff({...rejected, action: "inspect"})); + assert.throws(() => planAgentOperationHandoff({...rejected, action: "report", outcome})); + } + operation(value).agent_handoff = null; + assert.throws(() => planAgentOperationHandoff({...recovery, action: "inspect"})); +}); + test("bounded inbox retains recovery first and declares overflow instead of silently discarding it", () => { const items = Array.from({length: 22}, (_, index) => ({operation_id: `operation-${String(index).padStart(2, "0")}`, needs_reconciliation: index === 21, execution_allowed: false})); diff --git a/tests/extensions/test_lark_goal_channel_operation.py b/tests/extensions/test_lark_goal_channel_operation.py index 0c195354f1..307a6d5edf 100644 --- a/tests/extensions/test_lark_goal_channel_operation.py +++ b/tests/extensions/test_lark_goal_channel_operation.py @@ -339,6 +339,207 @@ def _digest(value: object) -> str: ).hexdigest() +def test_real_cli_replacement_recovers_unknown_and_updates_original_card_without_consuming( + tmp_path: Path, +) -> None: + from loopx.thread_agent_binding import ( + bind_thread_agent_in_registry, + unbind_thread_agent_in_registry, + ) + + store, registry, runtime, binding, target = _fixture(tmp_path) + proposal = _prepare_agent_handoff(store, registry) + cards: dict[str, dict[str, Any]] = {} + runner = _runner([], cards) + deliver_goal_channel_operation_card( + proposal_id=proposal["proposal_id"], + action_store_root=store.root, + runtime_root=runtime, + binding_path=binding, + target_path=target, + execute=True, + runner=runner, + ) + delivered = store.load(proposal["proposal_id"]) + message_id = delivered["operation"]["delivery"]["message_id"] + handle_goal_channel_operation_callback( + _event(delivered, cards[message_id]), + runtime_root=runtime, + action_store_root=store.root, + profile_app_id=APP_ID, + cli_bin="lark-cli", + profile="operation-bot", + runner=runner, + executor=lambda _: pytest.fail("recovery fixture must not execute externally"), + ) + original = { + "goal_id": GOAL_ID, + "agent_id": AGENT_ID, + "host_surface": "codex-app", + "thread_id": "thread-operation-fixture", + } + replacement = {**original, "thread_id": "thread-replacement-fixture"} + + def cli(*arguments: str, thread: str, expected_exit: int = 0) -> dict[str, Any]: + # This exercises the real CLI's existing ambient-context fence, not a + # claim of hostile-process authentication (tracked independently). + result = subprocess.run( + [ + sys.executable, + "-m", + "loopx.cli", + "--format", + "json", + "--registry", + str(registry), + "--runtime-root", + str(runtime), + *arguments, + ], + text=True, + capture_output=True, + check=False, + timeout=30, + env={**os.environ, "CODEX_THREAD_ID": thread}, + ) + assert result.returncode == expected_exit, result.stdout + result.stderr + return json.loads(result.stdout) + + def operation( + command: str, actor: dict[str, str], *extra: str, expected_exit: int = 0 + ): + return cli( + "goal-channel", + command, + "--goal-id", + GOAL_ID, + "--agent-id", + AGENT_ID, + "--proposal-id", + proposal["proposal_id"], + "--host-surface", + actor["host_surface"], + "--thread-id", + actor["thread_id"], + *extra, + thread=actor["thread_id"], + expected_exit=expected_exit, + ) + + consumed = operation( + "consume-operation", original, "--consumption-id", "attempt-1", "--execute" + ) + assert consumed["execution_allowed"] + unknown = { + "schema_version": "loopx_operation_outcome_v0", + **{ + key: consumed[key] + for key in ( + "operation_id", + "payload_digest", + "confirmation_digest", + "claim_id", + "executor_revision", + ) + }, + "consumption_id": "attempt-1", + "projection_verified": True, + "simulation": False, + "outcome": "submission_unknown", + "external_write_performed": True, + "summary": "Synthetic unknown submission requires original-system evidence.", + "evidence_refs": ["receipt:unknown-fixture"], + } + outcome_path = tmp_path / "outcome.json" + outcome_path.write_text(json.dumps(unknown)) + operation( + "report-operation", original, "--outcome-json", str(outcome_path), "--execute" + ) + unbind_thread_agent_in_registry(registry_path=registry, **original, execute=True) + bind_thread_agent_in_registry(registry_path=registry, **replacement, execute=True) + inbox = cli( + "manager-inbox", + "read", + "--goal-id", + GOAL_ID, + "--agent-id", + AGENT_ID, + thread=replacement["thread_id"], + ) + assert inbox["operation_handoffs"][0]["binding_current"] is False + inspected = operation("inspect-operation", replacement) + assert inspected["outcome"] == unknown and inspected["route"] == original + assert inspected["access"]["owner"] == replacement + assert inspected["access"]["permission"] == "historical_evidence_only" + before = store.path.read_bytes() + rejected = operation( + "consume-operation", + replacement, + "--consumption-id", + "attempt-2", + "--execute", + expected_exit=1, + ) + assert rejected["blocker"] == "operation_handoff_conflict" + assert store.path.read_bytes() == before + final = { + **unknown, + "outcome": "not_executed", + "external_write_performed": False, + "summary": "Synthetic original-system evidence proves no submission.", + "evidence_refs": ["receipt:reconciled-fixture"], + } + outcome_path.write_text(json.dumps(final)) + operation( + "report-operation", + replacement, + "--outcome-json", + str(outcome_path), + "--execute", + expected_exit=1, + ) + assert store.path.read_bytes() == before + final["reconciles_outcome_digest"] = _digest(unknown) + outcome_path.write_text(json.dumps(final)) + dry = operation( + "report-operation", replacement, "--outcome-json", str(outcome_path) + ) + assert not dry["execution_allowed"] and store.path.read_bytes() == before + reported = operation( + "report-operation", + replacement, + "--outcome-json", + str(outcome_path), + "--execute", + ) + assert not reported["execution_allowed"] and not reported["needs_reconciliation"] + assert reported["reconciliation_report"]["owner"] == replacement + inspected = operation("inspect-operation", replacement) + assert inspected["reconciliation"] == final and inspected["outcome"] == unknown + assert inspected["consumption"]["route"] == original + assert inspected["reconciliation_report"]["original_route"] == original + frame = goal_channel_operation._operation_review_frame( + store.load(proposal["proposal_id"]) + ) + assert frame["resultKind"] == "not_executed" and not frame["resultDeliveryVerified"] + recovered = recover_goal_channel_operation_results( + action_store_root=store.root, + profile_app_id=APP_ID, + allowed_chat_ids={CHAT_ID}, + cli_bin="lark-cli", + profile="operation-bot", + runner=runner, + ) + assert recovered["delivered"] == 1 and set(cards) == {message_id} + assert "已结束,未执行" in normalized_card_text(cards[message_id]) + updated = store.load(proposal["proposal_id"]) + assert updated["operation"]["outcome"] == unknown + assert updated["operation"]["result_delivery"]["outcome_stage"] == "reconciled" + assert goal_channel_operation._operation_review_frame(updated)[ + "resultDeliveryVerified" + ] + + def _fixture( tmp_path: Path, ) -> tuple[ChatActionStore, Path, Path, Path, Path]: diff --git a/tests/test_chat_operation_actions.py b/tests/test_chat_operation_actions.py index 813a432bab..425c3a98e3 100644 --- a/tests/test_chat_operation_actions.py +++ b/tests/test_chat_operation_actions.py @@ -518,6 +518,226 @@ def test_lifecycle_only_source_profile_cannot_acquire_new_operation_authority( assert (store.path.read_bytes() if store.path.exists() else None) == before +@pytest.mark.parametrize("historical", [False, True]) +def test_replacement_session_reconciles_under_its_current_binding_without_reconsumption( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, + historical: bool, +) -> None: + from loopx.thread_agent_binding import ( + bind_thread_agent_in_registry, + unbind_thread_agent_in_registry, + ) + + service, store = _service(tmp_path) + proposal = _claim_agent_operation(service, store) + runtime = store.root.parent.parent + args = dict(proposal_id=proposal["proposal_id"], actor=EXECUTION_ACTOR) + agent_operation_action( + runtime, + service.registry_path, + action="consume", + consumption_id="attempt-1", + **args, + ) + unknown = _agent_result(proposal, "attempt-1", result="submission_unknown") + agent_operation_action( + runtime, service.registry_path, action="report", outcome=unknown, **args + ) + replacement = {**EXECUTION_ACTOR, "thread_id": "thread-fixture-replacement"} + final = _agent_result(proposal, "attempt-1", result="not_executed") + final["reconciles_outcome_digest"] = _digest(unknown) + recovery_args = {**args, "actor": replacement} + before = store.path.read_bytes() + with pytest.raises(ActionConflictError): + agent_operation_action( + runtime, service.registry_path, action="inspect", **recovery_args + ) + bind_thread_agent_in_registry( + registry_path=service.registry_path, **replacement, execute=True + ) + with pytest.raises(ActionConflictError): + agent_operation_action( + runtime, + service.registry_path, + action="report", + outcome=final, + **recovery_args, + ) + assert store.path.read_bytes() == before + unbind_thread_agent_in_registry( + registry_path=service.registry_path, **EXECUTION_ACTOR, execute=True + ) + if historical: + registry = json.loads(service.registry_path.read_text()) + registry["goals"][0]["activation_state"] = "stopped" + service.registry_path.write_text(json.dumps(registry)) + after_expiry = (datetime.now(timezone.utc) + timedelta(hours=2)).isoformat() + monkeypatch.setattr("loopx.chat_action_store._utc_now", lambda: after_expiry) + inspected = agent_operation_action( + runtime, service.registry_path, action="inspect", **recovery_args + ) + assert inspected["route"] == EXECUTION_ACTOR + assert inspected["access"]["owner"] == replacement + assert inspected["access"]["permission"] == "historical_evidence_only" + assert inspected["outcome"] == unknown and not inspected["binding_current"] + for attempt in ("attempt-1", "attempt-2"): + reason = "Goal is stopped" if historical else "original bound session" + with pytest.raises(ActionConflictError, match=reason): + agent_operation_action( + runtime, + service.registry_path, + action="consume", + consumption_id=attempt, + **recovery_args, + ) + settled = agent_operation_action( + runtime, service.registry_path, action="report", outcome=final, **recovery_args + ) + assert not settled["execution_allowed"] and not settled["needs_reconciliation"] + updated = store.load(proposal["proposal_id"]) + assert updated["operation"]["outcome"] == unknown + assert updated["operation"]["reconciliation"] == final + provenance = updated["operation"]["reconciliation_report"] + assert provenance["owner"] == replacement + assert provenance["original_route"] == EXECUTION_ACTOR + assert provenance["authority_source"] == "current_registry_binding" + assert provenance["permission"] == "historical_evidence_only" + stable = store.path.read_bytes() + agent_operation_action( + runtime, service.registry_path, action="report", outcome=final, **recovery_args + ) + assert store.path.read_bytes() == stable + + with pytest.raises(ActionConflictError, match="already immutable"): + agent_operation_action( + runtime, + service.registry_path, + action="report", + outcome={ + **final, + "summary": "A conflicting historical result is not a retry.", + }, + **recovery_args, + ) + assert store.path.read_bytes() == stable + + unbind_thread_agent_in_registry( + registry_path=service.registry_path, **replacement, execute=True + ) + with pytest.raises(ActionConflictError): + agent_operation_action( + runtime, + service.registry_path, + action="report", + outcome=final, + **recovery_args, + ) + assert store.path.read_bytes() == stable + + +@pytest.mark.parametrize("stage", ["initial", "reconciled"]) +def test_recovery_binding_revocation_cannot_split_report_validation_from_commit( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch, stage: str +) -> None: + from contextlib import contextmanager + from threading import Event + from loopx.control_plane.projects import registry_codec + from loopx.control_plane.collaboration import operation_handoff + from loopx.thread_agent_binding import ( + bind_thread_agent_in_registry, + unbind_thread_agent_in_registry, + ) + + service, store = _service(tmp_path) + proposal = _claim_agent_operation(service, store) + runtime = store.root.parent.parent + args = dict(proposal_id=proposal["proposal_id"], actor=EXECUTION_ACTOR) + agent_operation_action( + runtime, + service.registry_path, + action="consume", + consumption_id="attempt-1", + **args, + ) + unknown = _agent_result(proposal, "attempt-1", result="submission_unknown") + if stage == "reconciled": + agent_operation_action( + runtime, service.registry_path, action="report", outcome=unknown, **args + ) + replacement = {**EXECUTION_ACTOR, "thread_id": "thread-fixture-replacement"} + unbind_thread_agent_in_registry( + registry_path=service.registry_path, **EXECUTION_ACTOR, execute=True + ) + bind_thread_agent_in_registry( + registry_path=service.registry_path, **replacement, execute=True + ) + final = _agent_result(proposal, "attempt-1", result="not_executed") + if stage == "reconciled": + final["reconciles_outcome_digest"] = _digest(unknown) + attempted, acquired = Event(), Event() + original_transaction = registry_codec._registry_transaction + original_binding = operation_handoff._binding + original_write = ChatActionStore._write + commits = [] + + @contextmanager + def observed_transaction(*args, **kwargs): + attempted.set() + with original_transaction(*args, **kwargs) as transaction: + acquired.set() + yield transaction + + def revoke(): + result = unbind_thread_agent_in_registry( + registry_path=service.registry_path, **replacement, execute=True + ) + assert result["written"] + commits.append("revocation") + + def record_report(self, payload): + original_write(self, payload) + commits.append("report") + + monkeypatch.setattr(registry_codec, "_registry_transaction", observed_transaction) + monkeypatch.setattr(ChatActionStore, "_write", record_report) + with ThreadPoolExecutor(max_workers=1) as executor: + futures = [] + + def interleaved_binding(registry_path, parameters): + current = original_binding(registry_path, parameters) + if parameters["executor"]["thread_id"] == replacement["thread_id"]: + futures.append(executor.submit(revoke)) + assert attempted.wait(5) + assert not acquired.wait(0.2), ( + "revocation must wait for historical report commit" + ) + return current + + monkeypatch.setattr(operation_handoff, "_binding", interleaved_binding) + result = agent_operation_action( + runtime, + service.registry_path, + action="report", + outcome=final, + **{**args, "actor": replacement}, + ) + assert not result["execution_allowed"] + futures[0].result(timeout=5) + assert commits == ["report", "revocation"] + monkeypatch.setattr(operation_handoff, "_binding", original_binding) + before = store.path.read_bytes() + with pytest.raises(ActionConflictError): + agent_operation_action( + runtime, + service.registry_path, + action="report", + outcome=final, + **{**args, "actor": replacement}, + ) + assert store.path.read_bytes() == before + + def test_inbox_uses_shared_recovery_priority_and_explicit_overflow( tmp_path: Path, ) -> None: From 64c1f2bc54747d24667e09e71f41b2b97a4252dc Mon Sep 17 00:00:00 2001 From: huangruiteng <14976749+huangruiteng@users.noreply.github.com> Date: Wed, 30 Sep 2026 05:50:57 +0800 Subject: [PATCH 05/13] docs(operations): distinguish historical recovery from host authentication Signed-off-by: huangruiteng <14976749+huangruiteng@users.noreply.github.com> --- .../human-confirmed-domain-operations-v0.md | 43 +++++++++++++++---- ...an-confirmed-domain-operations-v0.zh-CN.md | 27 +++++++++--- 2 files changed, 57 insertions(+), 13 deletions(-) diff --git a/docs/architecture/rfcs/human-confirmed-domain-operations-v0.md b/docs/architecture/rfcs/human-confirmed-domain-operations-v0.md index e1c662afa6..4f56933c88 100644 --- a/docs/architecture/rfcs/human-confirmed-domain-operations-v0.md +++ b/docs/architecture/rfcs/human-confirmed-domain-operations-v0.md @@ -357,16 +357,19 @@ loopx --registry REGISTRY --runtime-root RUNTIME goal-channel report-operation \ `inspect-operation` and a command without `--execute` never consume authority. All three CLI commands require the existing host-exported ambient thread -(for example, `CODEX_THREAD_ID`) to match the original route; CLI identifiers +(for example, `CODEX_THREAD_ID`) to match the caller's route; CLI identifiers are selectors, not caller authentication. Missing, foreign or unsupported host context fails closed before inspecting private parameters or writing a receipt. -The returned `caller_context_source: trusted_local_host_environment` names a -trusted-local-OS-user fence, **not cryptographic session isolation**. A hostile -process that can forge the environment or rewrite the same user's canonical -files is outside this slice. Do not advertise exclusive execution across -untrusted processes; that requires an independently qualified authenticated -host transport. Internal storage adapters accept host-validated actor facts, -not unauthenticated network requests. +The returned `caller_context_source: trusted_local_host_environment` names an +ambient-context check, **not authenticated session isolation**. Exact-head review +reproduced a same-user process forging this environment and consuming the +original session's authorization. Original-session-exclusive execution is an +unresolved acceptance blocker: this path must not be installed or declared a +live minimum loop until an independently qualified trusted-host transport +proves caller identity. A current registry binding authorizes a route, but does +not authenticate its caller. Internal storage adapters are local IO seams, not +unauthenticated network endpoints; historical recovery below does not close +this separate authentication gap. Only the first successful atomic consumption returns `execution_allowed: true`. It verifies authenticated confirmation, immutable terms, the current original session, active Goal and expiry, then persists consumption before any browser @@ -398,6 +401,30 @@ implicit migration into lifecycle-only registries or another home. Missing original Goal/Agent registration or an unsupported registry profile is an explicit error, not permission to transplant the operation. +After the original route is withdrawn, a **currently registered and bound +replacement session of the same Goal and Agent** may use its own caller route +with `inspect-operation` and `report-operation`. It may inspect only an already +consumed operation and report original-system historical evidence. It cannot +consume an unspent ticket, change the original executor or obtain a second +execution permission. Recovery is not admitted while the original binding is +still current, for an unbound replacement or for a different Goal/Agent. The +CLI returns an explicit `access.owner`, `original_route`, +`permission: "historical_evidence_only"` and binding authority; it never asks the +replacement to impersonate the old thread. The original executor route remains +immutable. The same registry lock is held from recovery-binding validation +through result commit; revocation that commits first rejects the report. + +Result evidence remains unchanged. The canonical operation separately appends +`outcome_report` or `reconciliation_report` provenance with the actual reporter, +original route, evidence-only permission, authority source, consumption ID and +recorded time. Identical result retries preserve the first committed provenance; +they do not relabel its author or grant execution. CLI inspection exposes both +the original unknown outcome and its immutable reconciliation/provenance. The +existing shared Dashboard frame and original Lark-card recovery consume the +same canonical result; neither gets a separate recovery approval store or an +execution control. These synthetic CLI/result-card checks are historical +recovery acceptance, not trusted-host authentication or live-group acceptance. + An unknown original outcome is immutable. A definitive report appends `operation.reconciliation` and binds `reconciles_outcome_digest` to the exact original unknown result. Only that evidence closes the recovery obligation. diff --git a/docs/architecture/rfcs/human-confirmed-domain-operations-v0.zh-CN.md b/docs/architecture/rfcs/human-confirmed-domain-operations-v0.zh-CN.md index 5a5f775d7d..2f2e8fee7c 100644 --- a/docs/architecture/rfcs/human-confirmed-domain-operations-v0.zh-CN.md +++ b/docs/architecture/rfcs/human-confirmed-domain-operations-v0.zh-CN.md @@ -284,12 +284,14 @@ loopx --registry REGISTRY --runtime-root RUNTIME goal-channel report-operation \ ``` `inspect-operation` 和没有 `--execute` 的命令均不消费授权。三个 CLI 命令均要求 -既有宿主导出的环境线程(例如 `CODEX_THREAD_ID`)匹配原路由;命令行标识只是选择器, +既有宿主导出的环境线程(例如 `CODEX_THREAD_ID`)匹配调用会话的路由;命令行标识只是选择器, 不是调用者认证。环境缺失、不匹配或不支持的宿主在读取私有参数或写回之前即拒绝。 -返回的 `caller_context_source: trusted_local_host_environment` 明确表示可信本地 -OS 用户边界,**不等于密码学 session 隔离**。能伪造环境或改写同用户规范文件的恶意 -进程不在本切片保证内;不得宣称在不可信进程间独占执行,这需要另行验收经认证的宿主 -传输。内部存储 adapter 只接收宿主已验证的 actor 事实,不接受未认证的网络请求。 +返回的 `caller_context_source: trusted_local_host_environment` 只表示环境上下文检查, +**不等于经认证的 session 隔离**。精确头复审已复现:同用户进程可自行设置该环境, +消费原会话授权。“原会话专属执行”仍是未解决的验收阻塞;在独立验收受信宿主传输能 +证明调用者身份之前,不得安装此路径或宣称真实最小闭环。当前 registry 绑定只授权 +路由,不认证调用者。内部存储 adapter 是本地 IO 接缝,不是未认证的网络端点;下述 +历史对账恢复并不关闭这个独立的认证缺口。 只有首次成功的原子消费 返回 `execution_allowed: true`。它检查经认证的确认、不可变条款、原 session 当前绑定、 有效 Goal 与到期时间,并在任何浏览器操作之前持久记录消费。所有重试,包括响应丢失 @@ -313,6 +315,21 @@ claim、执行器 revision、consumption ID,要求 `projection_verified: true` 生命周期保护保持不变,不隐式迁入生命周期专用 registry 或另一个 home。 原 Goal/Agent 注册缺失或 registry profile 不受支持均明确报错,不允许移植操作权限。 +原路由撤销后,**同 Goal、同 Agent、当前已注册并绑定的接手会话**可以用自己的调用 +路由执行 `inspect-operation` 与 `report-operation`。它只能检查已消费操作,回写 +原系统的历史证据,不能消费未使用的票据、改写原执行器或取得第二次执行许可。原绑定 +仍有效、接手会话未绑定、Goal/Agent 不同均拒绝恢复准入。CLI 明确返回 `access.owner`、 +`original_route`、`permission: "historical_evidence_only"` 与绑定权威,不要求接手者 +冒充旧线程。原执行路由保持不可变;从接手绑定核对到结果提交共用 registry 锁,先提交 +的撤销拒绝回写。 + +结果证据本身保持原样。规范操作单独追加 `outcome_report` 或 `reconciliation_report` +来源,记录实际报告者、原路由、仅历史证据权限、权威来源、consumption ID 与记录时间。 +相同结果重试保留首次已提交来源,不重标作者或授予执行。CLI 检查同时返回原未知 +outcome 与不可变对账/来源;既有 Dashboard 共享 frame 与原 Lark 卡恢复消费同一 +规范结果,不另建恢复审批存储或执行控件。这些合成 CLI/原卡读回只验收历史对账恢复, +不代替受信宿主认证或真实群验收。 + 未知的原始 outcome 不可修改。确定性回写追加 `operation.reconciliation`,并通过 `reconciles_outcome_digest` 绑定原未知结果的精确摘要;只有该证据才关闭恢复义务。 现有投递恢复更新原结果卡,旧未知结果的投递不能证明新的对账结果;读回必须匹配当前 From 45c3d7b633921c2c0f609cb4882d0aa8837ea10e Mon Sep 17 00:00:00 2001 From: huangruiteng <14976749+huangruiteng@users.noreply.github.com> Date: Wed, 30 Sep 2026 07:50:27 +0800 Subject: [PATCH 06/13] fix(operations): reject unauthenticated public session handoff Signed-off-by: huangruiteng <14976749+huangruiteng@users.noreply.github.com> --- .../smoke/action-review-plan-smoke.ts | 4 + .../src/features/personal-workspace/i18n.tsx | 8 +- loopx/cli_commands/goal_channel_operation.py | 51 +-- .../collaboration/operation_handoff.py | 13 +- .../presentation/action_review_plan.ts | 4 +- .../work_items/operation_agent_handoff.ts | 25 +- .../extensions/lark/goal_channel_operation.py | 2 +- .../action_review_plan.test.ts | 2 +- .../operation_agent_handoff.test.ts | 14 +- .../test_lark_goal_channel_operation.py | 336 ++++++++++-------- 10 files changed, 252 insertions(+), 207 deletions(-) diff --git a/apps/presentation/dashboard/smoke/action-review-plan-smoke.ts b/apps/presentation/dashboard/smoke/action-review-plan-smoke.ts index ef6c532e8d..f85566db5a 100644 --- a/apps/presentation/dashboard/smoke/action-review-plan-smoke.ts +++ b/apps/presentation/dashboard/smoke/action-review-plan-smoke.ts @@ -128,6 +128,10 @@ const agentPending = typedActionProposalSchema.parse({...operationProposal, stat const agentPendingFrame = compileActionReviewPlan(agentPending).operationFrame; check(agentPendingFrame?.kind === "pending" && agentPendingFrame.executionState === "consumed_outcome_pending", "Transport retains the original consumption; it does not imply an external result"); +const unauthenticatedFrame = compileActionReviewPlan({...agentPending, + operation: {...agentPending.operation, agent_handoff: null}}).operationFrame; +check(unauthenticatedFrame?.kind === "pending" && unauthenticatedFrame.executionState === "host_authentication_required", + "Human confirmation alone cannot qualify original-host authentication"); const unknownAgentResult = typedActionProposalSchema.parse({...agentPending, status: "applied", receipt: {projection_verified: true}, operation: {...agentPending.operation, lifecycle_state: "outcome_observed", outcome: {outcome: "submission_unknown", simulation: false}, result_delivery: {outcome_stage: "initial"}}}); diff --git a/apps/presentation/dashboard/src/features/personal-workspace/i18n.tsx b/apps/presentation/dashboard/src/features/personal-workspace/i18n.tsx index 5551ce0471..6b96e2f74b 100644 --- a/apps/presentation/dashboard/src/features/personal-workspace/i18n.tsx +++ b/apps/presentation/dashboard/src/features/personal-workspace/i18n.tsx @@ -676,7 +676,7 @@ const en = { "proposal.field.objective": "Objective", "proposal.field.operation": "Operation", "proposal.field.operationState": "Operation state", - "proposal.operationState.authorized_pending": "Confirmed; waiting for the original Agent", + "proposal.operationState.host_authentication_required": "Confirmed; original-host authentication unavailable", "proposal.operationState.consumed_outcome_pending": "Authorization consumed; waiting for the real result", "proposal.operationState.submission_unknown": "Result unknown; reconcile the original operation, do not resubmit", "proposal.field.resultDelivery": "Result delivery", @@ -719,7 +719,7 @@ const en = { "proposal.impact.lifecycleStop": "After confirmation, automatic progress stops and the Goal moves to the collapsed Stopped list. History, Todos, and evidence remain available for resuming.", "proposal.impact.protected": "This action must be completed through the protected LoopX write service.", "proposal.impact.operation": "The exact terms are read-only here. Confirm or reject the same immutable request in the bound Feishu group; confirmation consumes one canonical claim.", - "proposal.impact.operationAuthorized": "Human confirmation is recorded. The original Agent must read and consume this exact authorization once; no external result is recorded yet.", + "proposal.impact.operationAuthorized": "Human confirmation is recorded, but the original host's authenticated tool transport is not connected. Thread flags or environment ids cannot authorize execution. No external result is recorded yet.", "proposal.impact.operationConsumed": "The authorization has been consumed. Wait for original external evidence; a retry or lost response must not grant another submission.", "proposal.impact.operationUnknown": "Submission may have had an external effect. Reconcile the original operation using its evidence; do not resubmit or treat card delivery as execution completion.", "proposal.primary.apply": "Confirm and apply", @@ -1864,7 +1864,7 @@ const zhCN: Record = { "proposal.field.objective": "目标", "proposal.field.operation": "操作", "proposal.field.operationState": "操作状态", - "proposal.operationState.authorized_pending": "已确认,等待原 Agent 接手", + "proposal.operationState.host_authentication_required": "已确认,原宿主身份认证尚未接通", "proposal.operationState.consumed_outcome_pending": "授权已消费,等待真实结果", "proposal.operationState.submission_unknown": "结果未知;核对原操作,不可重复提交", "proposal.field.resultDelivery": "结果回传", @@ -1907,7 +1907,7 @@ const zhCN: Record = { "proposal.impact.lifecycleStop": "确认后会停止自动推进,并将 Goal 移入折叠的「已停止」列表;历史、Todo 和证据都会保留,可随时恢复。", "proposal.impact.protected": "该操作需要通过受保护的 LoopX 写入服务完成。", "proposal.impact.operation": "这里仅展示同一份不可变条款。请在已绑定的飞书群确认或拒绝;确认只会消费一个规范 claim。", - "proposal.impact.operationAuthorized": "用户确认已记录。原 Agent 须读取并一次消费这份精确授权;目前尚无外部执行结果。", + "proposal.impact.operationAuthorized": "用户确认已记录,但原宿主的认证工具通道尚未接通。线程参数或环境变量不能授权执行;目前尚无外部执行结果。", "proposal.impact.operationConsumed": "授权已消费。等待原始外部证据;重试或响应丢失均不得重新授予提交许可。", "proposal.impact.operationUnknown": "提交可能已产生外部副作用。须以原始证据核对原操作,不可重提,也不能把卡片投递当作执行完成。", "proposal.primary.apply": "确认并应用", diff --git a/loopx/cli_commands/goal_channel_operation.py b/loopx/cli_commands/goal_channel_operation.py index 7edc091b9c..92eb137c1e 100644 --- a/loopx/cli_commands/goal_channel_operation.py +++ b/loopx/cli_commands/goal_channel_operation.py @@ -13,13 +13,11 @@ from ..chat_action_store import ActionConflictError, ChatActionStore from ..chat_actions import ChatActionService -from ..control_plane.collaboration.operation_handoff import agent_operation_action from ..control_plane.effect_runtime import ( EffectRuntimeConflict, EffectRuntimeRejected, effect_runtime_result, ) -from ._host_thread import ambient_host_thread_id from ..extensions.lark.goal_channel import ( default_goal_channel_target_path, deliver_goal_channel_operation_card, @@ -257,10 +255,12 @@ def run_goal_channel_operation( return None try: if isinstance(request, _AgentOperation): - # These commands use the original registry/store, not Lark target - # settings or a new approval source. No external executor is called. + # No independent host identity producer is connected here. Ask the + # TS owner for the explicit gate before reading private parameters, + # outcome files or canonical operation state. Never promote an + # environment id, --thread-id or a local "verified" flag to identity. try: - actor = effect_runtime_result( + effect_runtime_result( "operation.agent_handoff.actor", { "requested": { @@ -269,41 +269,18 @@ def run_goal_channel_operation( "host_surface": request.host_surface, "thread_id": request.thread_id, }, - "ambient": { - "host_surface": request.host_surface, - "thread_id": ambient_host_thread_id(request.host_surface), - }, }, ) except (EffectRuntimeConflict, EffectRuntimeRejected) as exc: - raise ActionConflictError(str(exc)) from exc - action = request.command.value.removesuffix("-operation") - if not request.execute: - action = "inspect" - outcome = None - if action == "report" and request.outcome_path is not None: - if request.outcome_path.stat().st_size > 65536: - raise ValueError("operation outcome exceeds its bounded envelope") - outcome = json.loads(request.outcome_path.read_text(encoding="utf-8")) - if not isinstance(outcome, dict): - raise ValueError("operation outcome must be an object") - result = agent_operation_action( - context.source_runtime_root, - context.source_registry_path, - proposal_id=request.proposal_id, - actor=actor, - action=action, - consumption_id=request.consumption_id, - outcome=outcome, - ) - return { - "ok": True, - "goal_id": request.goal_id, - "execute": request.execute, - "operation": request.command.value.replace("-", "_"), - "caller_context_source": "trusted_local_host_environment", - **result, - } + return _operation_error_packet( + request=request, + blocker=exc.diagnostic_code, + summary=str(exc), + details={"execution_allowed": False}, + ) + # An unexpected success from a mismatched/older runtime still may + # not bypass this adapter's missing authenticated transport. + raise ActionConflictError("no authenticated host transport is connected") target_path = _operation_target_path(request, context) binding = ( binding_for_goal( diff --git a/loopx/control_plane/collaboration/operation_handoff.py b/loopx/control_plane/collaboration/operation_handoff.py index 269ba3b18e..b2e0f37695 100644 --- a/loopx/control_plane/collaboration/operation_handoff.py +++ b/loopx/control_plane/collaboration/operation_handoff.py @@ -93,10 +93,12 @@ def pending_operation_handoffs( **plan, "binding_current": current, "summary": proposal["summary"], - "instruction": "Read the original canonical operation and consume it once before any external effect. " + "instruction": "The original host must authenticate through its session-bound tool transport before reading " + "private operation terms or consuming authority. No qualified producer is connected to the CLI; " + "environment thread ids are not identity proof. " "Only the first successful consumption permits execution; consumed/unknown results require " "original external-system reconciliation, never another submission. Inbox delivery is not execution authority.", - "next_action": "goal-channel consume-operation" + "next_action": "Integrate the original host's authenticated session-bound tool transport; do not retry via environment identity." if plan["status"] == "authorized_pending" else "Reconcile the original external result; do not submit again.", } @@ -130,6 +132,13 @@ def agent_operation_action( consumption_id: str | None = None, outcome: Mapping[str, Any] | None = None, ) -> dict[str, Any]: + """Internal locked IO seam, not a public caller-authentication endpoint. + + `actor` must be supplied by a qualified host transport, never by CLI flags, + environment variables or an arbitrary model tool argument. Fixture calls + validate storage semantics only; the public CLI remains blocked until a + real producer/verifier pair is integrated. + """ store = _store(runtime_root) proposal = store.load(proposal_id) if proposal is None: diff --git a/loopx/control_plane/presentation/action_review_plan.ts b/loopx/control_plane/presentation/action_review_plan.ts index d5a2614be3..320e979085 100644 --- a/loopx/control_plane/presentation/action_review_plan.ts +++ b/loopx/control_plane/presentation/action_review_plan.ts @@ -41,7 +41,7 @@ export type OperationReviewFrame = OperationReviewFrameBase & ( kind: "pending"; attentionKind: "progress"; interactionMode: "inform"; - executionState?: "authorized_pending" | "consumed_outcome_pending"; + executionState?: "host_authentication_required" | "consumed_outcome_pending"; } | { kind: "result"; @@ -323,7 +323,7 @@ export function compileOperationReviewFrame(proposalValue: unknown): OperationRe attentionKind: "progress", interactionMode: "inform", ...(objectValue(parameters.executor)?.kind === "agent_session" - ? {executionState: objectValue(operation.agent_handoff) ? "consumed_outcome_pending" as const : "authorized_pending" as const} + ? {executionState: objectValue(operation.agent_handoff) ? "consumed_outcome_pending" as const : "host_authentication_required" as const} : {}), }; } diff --git a/loopx/control_plane/work_items/operation_agent_handoff.ts b/loopx/control_plane/work_items/operation_agent_handoff.ts index 1c5f0af704..9fd1dfafec 100644 --- a/loopx/control_plane/work_items/operation_agent_handoff.ts +++ b/loopx/control_plane/work_items/operation_agent_handoff.ts @@ -36,18 +36,16 @@ export function normalizeAgentOperationExecutor(input: JsonObject): JsonObject { thread_id: id(executor.thread_id, "thread_id"), revision: AGENT_OPERATION_REVISION}; } -/** CLI selectors are not caller identity. The transport supplies the existing - * host's ambient thread in the trusted local OS-user boundary; hostile local - * processes/file writers require a separate authenticated host transport. */ -export function deriveAgentOperationActor(input: JsonObject): JsonObject { - const requested = requireJsonObject(input.requested, "requested actor"); - const ambient = requireJsonObject(input.ambient, "ambient host context"); - requireThat(typeof ambient.thread_id === "string" && ambient.thread_id.length > 0, - "original host session context is unavailable; route flags are not caller identity"); - requireThat(requested.host_surface === ambient.host_surface && requested.thread_id === ambient.thread_id, - "requested route is not the current host session"); - return {goal_id: id(requested.goal_id, "goal_id"), agent_id: id(requested.agent_id, "agent_id"), - host_surface: id(ambient.host_surface, "host_surface"), thread_id: id(ambient.thread_id, "thread_id")}; +/** No qualified host producer is connected to the public CLI. Ambient thread + * ids, route flags and caller-supplied "verified" fields cannot authenticate + * a session. Keep the old RPC fail-closed until a real transport-owned issuer + * and its owner-pinned verifier are qualified together; do not mint a local + * bearer credential from the same untrusted environment. */ +export function deriveAgentOperationActor(_input: JsonObject): never { + throw new EffectRuntimeRequestError( + "Original-host authentication is unavailable. Integrate the original host's session-bound tool transport; environment ids and route flags are not identity proof.", + "operation_host_authentication_unavailable", + ); } /** A registry-authorized replacement may recover evidence, never inherit the @@ -93,7 +91,8 @@ export function planAgentOperationHandoff(input: JsonObject): JsonObject { payload_digest: operation.payload_digest, confirmation_digest: operation.confirmation_digest, claim_id: claim?.claim_id ?? null, executor_revision: executor.revision, expires_at: operation.expires_at, route, authorization_source: "canonical_typed_operation", execution_allowed: false, - host_delivery: "not_attempted", external_write_performed: false}; + host_delivery: "not_attempted", external_write_performed: false, + host_authentication_required: true}; const handoff = operation.agent_handoff == null ? null : requireJsonObject(operation.agent_handoff, "agent handoff"); const observed = operation.reconciliation ?? operation.outcome; diff --git a/loopx/extensions/lark/goal_channel_operation.py b/loopx/extensions/lark/goal_channel_operation.py index 62b0637878..e89fb5db73 100644 --- a/loopx/extensions/lark/goal_channel_operation.py +++ b/loopx/extensions/lark/goal_channel_operation.py @@ -363,7 +363,7 @@ def build_goal_channel_operation_result_card( else "green" ) result_label = ( - "已确认,等待原 Agent 执行" + "已确认,原宿主身份认证尚未接通" if pending else "结果未知,须核对原操作,不可重复提交" if unknown diff --git a/tests/control_plane_ts/action_review_plan.test.ts b/tests/control_plane_ts/action_review_plan.test.ts index 25379d08da..b73bb4b183 100644 --- a/tests/control_plane_ts/action_review_plan.test.ts +++ b/tests/control_plane_ts/action_review_plan.test.ts @@ -125,7 +125,7 @@ test("original-Agent pending, unknown and reconciled results share truthful surf proposal.normalized_parameters.executor = {kind: "agent_session"}; proposal.normalized_parameters.projection.simulated = false; let frame = compileOperationReviewFrame(proposal); - assert.equal(frame?.kind === "pending" && frame.executionState, "authorized_pending"); + assert.equal(frame?.kind === "pending" && frame.executionState, "host_authentication_required"); proposal.operation.agent_handoff = {consumption_id: "attempt-1"}; frame = compileOperationReviewFrame(proposal); assert.equal(frame?.kind === "pending" && frame.executionState, "consumed_outcome_pending"); diff --git a/tests/control_plane_ts/operation_agent_handoff.test.ts b/tests/control_plane_ts/operation_agent_handoff.test.ts index 1cbcd78279..1688e9a0f2 100644 --- a/tests/control_plane_ts/operation_agent_handoff.test.ts +++ b/tests/control_plane_ts/operation_agent_handoff.test.ts @@ -22,15 +22,19 @@ function input(): JsonObject { } const operation = (value: JsonObject) => (value.proposal as JsonObject).operation as JsonObject; -test("caller selectors cannot replace missing or foreign host context", () => { +test("public caller identity stays blocked, including exact same-user environment forgery", () => { const requested = input().actor as JsonObject; - for (const thread_id of [null, "", "another-thread"]) { - assert.throws(() => deriveAgentOperationActor({requested, ambient: {host_surface: "codex-app", thread_id}})); + for (const thread_id of [null, "", "another-thread", "original-thread"]) { + assert.throws(() => deriveAgentOperationActor({requested, ambient: {host_surface: "codex-app", thread_id}}), + {code: "operation_host_authentication_unavailable", kind: "request_rejected"}); } assert.throws(() => deriveAgentOperationActor({requested, ambient: {host_surface: "unsupported-host", thread_id: "original-thread"}})); - assert.deepEqual(deriveAgentOperationActor({requested, - ambient: {host_surface: "codex-app", thread_id: "original-thread"}}), requested); + for (const claimedProof of [{verified: true}, {signature: "self-signed", issuer: "local-cli"}, + {caller_context_source: "authenticated_host_transport"}]) { + assert.throws(() => deriveAgentOperationActor({requested, ambient: requested, proof: claimedProof}), + {code: "operation_host_authentication_unavailable"}); + } }); test("only the first consumed canonical confirmation grants the original session execution", () => { diff --git a/tests/extensions/test_lark_goal_channel_operation.py b/tests/extensions/test_lark_goal_channel_operation.py index 307a6d5edf..9433999552 100644 --- a/tests/extensions/test_lark_goal_channel_operation.py +++ b/tests/extensions/test_lark_goal_channel_operation.py @@ -138,7 +138,7 @@ def no_executor(_proposal): assert first["callback_ack_is_execution_receipt"] is False claimed = store.load(proposal["proposal_id"]) assert claimed["operation"]["result_delivery"] is None - assert "等待原 Agent" in normalized_card_text(next(iter(cards.values()))) + assert "原宿主身份认证尚未接通" in normalized_card_text(next(iter(cards.values()))) actor = { "goal_id": GOAL_ID, "agent_id": AGENT_ID, @@ -157,23 +157,38 @@ def no_executor(_proposal): execute=False, ) dry = run_goal_channel_operation(args, context=context) - assert dry["execution_allowed"] is False and not store.load( - proposal["proposal_id"] - )["operation"].get("agent_handoff") + assert dry["ok"] is False and not store.load(proposal["proposal_id"])[ + "operation" + ].get("agent_handoff") args.execute = True before = store.path.read_bytes() - for ambient_thread in ["", "thread-unrelated-fixture"]: + for ambient_thread in ["", "thread-unrelated-fixture", "thread-operation-fixture"]: monkeypatch.setenv("CODEX_THREAD_ID", ambient_thread) rejected = run_goal_channel_operation(args, context=context) assert rejected["ok"] is False - assert rejected["blocker"] == "operation_handoff_conflict" + assert rejected["blocker"] == "operation_host_authentication_unavailable" assert store.path.read_bytes() == before - monkeypatch.setenv("CODEX_THREAD_ID", "thread-operation-fixture") - consumed = run_goal_channel_operation(args, context=context) + # The internal IO fixture qualifies one-shot/reconciliation semantics, + # not a production host issuer. The public CLI above must remain blocked. + consumed = agent_operation_action( + runtime, + registry, + proposal_id=proposal["proposal_id"], + actor=actor, + action="consume", + consumption_id="attempt-1", + ) assert consumed["execution_allowed"] is True - assert consumed["caller_context_source"] == "trusted_local_host_environment" assert ( - run_goal_channel_operation(args, context=context)["execution_allowed"] is False + agent_operation_action( + runtime, + registry, + proposal_id=proposal["proposal_id"], + actor=actor, + action="consume", + consumption_id="attempt-1", + )["execution_allowed"] + is False ) operation = claimed["operation"] unknown = { @@ -191,16 +206,14 @@ def no_executor(_proposal): "summary": "Synthetic submission result is unknown; do not resubmit.", "evidence_refs": ["receipt:unknown-fixture"], } - outcome_path = tmp_path / "outcome.json" - outcome_path.write_text(json.dumps(unknown)) - report_args = Namespace( - **{ - **vars(args), - "goal_channel_command": "report-operation", - "outcome_json": str(outcome_path), - } + reported = agent_operation_action( + runtime, + registry, + proposal_id=proposal["proposal_id"], + actor=actor, + action="report", + outcome=unknown, ) - reported = run_goal_channel_operation(report_args, context=context) assert ( reported["status"] == "submission_unknown" and reported["needs_reconciliation"] ) @@ -225,8 +238,17 @@ def no_executor(_proposal): "reconciles_outcome_digest": _digest(unknown), "evidence_refs": ["receipt:reconciled-fixture"], } - outcome_path.write_text(json.dumps(final)) - assert run_goal_channel_operation(report_args, context=context)["outcome"] == final + assert ( + agent_operation_action( + runtime, + registry, + proposal_id=proposal["proposal_id"], + actor=actor, + action="report", + outcome=final, + )["outcome"] + == final + ) updated = store.load(proposal["proposal_id"]) frame = goal_channel_operation._operation_review_frame(updated) assert frame["resultDeliveryVerified"] is False @@ -249,7 +271,15 @@ def no_executor(_proposal): == "reconciled" ) assert ( - run_goal_channel_operation(args, context=context)["execution_allowed"] is False + agent_operation_action( + runtime, + registry, + proposal_id=proposal["proposal_id"], + actor=actor, + action="consume", + consumption_id="attempt-1", + )["execution_allowed"] + is False ) assert "operation_handoffs" not in pending(runtime, GOAL_ID, AGENT_ID) assert ( @@ -264,71 +294,133 @@ def no_executor(_proposal): ) -def test_real_cli_inspects_canonical_handoff_in_source_runtime(tmp_path: Path) -> None: - store, registry, runtime, _binding, _target = _fixture(tmp_path) +def test_real_cli_rejects_same_user_environment_forgery_before_private_reads_or_writes( + tmp_path: Path, +) -> None: + store, registry, runtime, binding, target = _fixture(tmp_path) proposal = _prepare_agent_handoff(store, registry) - result = subprocess.run( - [ - sys.executable, - "-m", - "loopx.cli", - "--format", - "json", - "--registry", - str(registry), - "--runtime-root", - str(runtime), - "goal-channel", - "inspect-operation", - "--goal-id", - GOAL_ID, - "--agent-id", - AGENT_ID, - "--proposal-id", - proposal["proposal_id"], - "--host-surface", - "codex-app", - "--thread-id", - "thread-operation-fixture", - ], - text=True, - capture_output=True, - check=True, - timeout=30, - env={**os.environ, "CODEX_THREAD_ID": "thread-operation-fixture"}, + cards: dict[str, dict[str, Any]] = {} + runner = _runner([], cards) + deliver_goal_channel_operation_card( + proposal_id=proposal["proposal_id"], + action_store_root=store.root, + runtime_root=runtime, + binding_path=binding, + target_path=target, + execute=True, + runner=runner, + ) + delivered = store.load(proposal["proposal_id"]) + handle_goal_channel_operation_callback( + _event(delivered, cards[delivered["operation"]["delivery"]["message_id"]]), + runtime_root=runtime, + action_store_root=store.root, + profile_app_id=APP_ID, + cli_bin="lark-cli", + profile="operation-bot", + runner=runner, + executor=lambda _: pytest.fail("no external execution"), ) - packet = json.loads(result.stdout) assert ( - packet["status"] == "awaiting_confirmation" - and packet["execution_allowed"] is False + store.load(proposal["proposal_id"])["operation"]["lifecycle_state"] == "claimed" ) before = store.path.read_bytes() - wrong_thread = subprocess.run( - [*result.args[:-1], "thread-other"], - text=True, - capture_output=True, - check=False, - timeout=30, - env={**os.environ, "CODEX_THREAD_ID": "thread-operation-fixture"}, - ) - assert wrong_thread.returncode == 1 - assert json.loads(wrong_thread.stdout)["blocker"] == "operation_handoff_conflict" + command_args = { + "consume-operation": ["--consumption-id", "attempt-forged", "--execute"], + "inspect-operation": [], + # Missing outcome is intentional: authentication must precede its read. + "report-operation": [ + "--outcome-json", + str(tmp_path / "absent-outcome.json"), + "--execute", + ], + } + for command, extra in command_args.items(): + for ambient_thread in [ + "thread-operation-fixture", + "thread-unrelated-fixture", + "", + ]: + result = subprocess.run( + [ + sys.executable, + "-m", + "loopx.cli", + "--format", + "json", + "--registry", + str(registry), + "--runtime-root", + str(runtime), + "goal-channel", + command, + "--goal-id", + GOAL_ID, + "--agent-id", + AGENT_ID, + "--proposal-id", + proposal["proposal_id"], + "--host-surface", + "codex-app", + "--thread-id", + "thread-operation-fixture", + *extra, + ], + text=True, + capture_output=True, + check=False, + timeout=30, + env={**os.environ, "CODEX_THREAD_ID": ambient_thread}, + ) + assert result.returncode == 1, result.stdout + result.stderr + packet = json.loads(result.stdout) + assert packet["blocker"] == "operation_host_authentication_unavailable" + assert packet["details"]["execution_allowed"] is False + assert packet["external_write_performed"] is False + assert ( + "environment ids and route flags are not identity proof" + in packet["public_summary"] + ) + assert "normalized_parameters" not in packet and "outcome" not in packet + assert store.path.read_bytes() == before + + +def test_cli_host_gate_survives_unexpected_success_from_an_older_effect_runtime( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, +) -> None: + from loopx.cli_commands import goal_channel_operation as command_module + + store, registry, runtime, binding, _target = _fixture(tmp_path) + proposal = _prepare_agent_handoff(store, registry) + before = store.path.read_bytes() + monkeypatch.setattr( + command_module, + "effect_runtime_result", + lambda *_: { + "goal_id": GOAL_ID, + "agent_id": AGENT_ID, + "host_surface": "codex-app", + "thread_id": "thread-operation-fixture", + }, + ) + args = Namespace( + goal_channel_command="consume-operation", + goal_id=GOAL_ID, + agent_id=AGENT_ID, + proposal_id=proposal["proposal_id"], + host_surface="codex-app", + thread_id="thread-operation-fixture", + consumption_id="attempt-legacy", + execute=True, + ) + packet = run_goal_channel_operation( + args, + context=GoalChannelOperationContext(runtime, registry, runtime, binding), + ) + assert packet["ok"] is False and packet["external_write_performed"] is False + assert "no authenticated host transport" in packet["public_summary"] assert store.path.read_bytes() == before - for ambient_thread in ["", "thread-unrelated-fixture"]: - foreign_process = subprocess.run( - result.args, - text=True, - capture_output=True, - check=False, - timeout=30, - env={**os.environ, "CODEX_THREAD_ID": ambient_thread}, - ) - assert foreign_process.returncode == 1 - assert ( - json.loads(foreign_process.stdout)["blocker"] - == "operation_handoff_conflict" - ) - assert store.path.read_bytes() == before def _digest(value: object) -> str: @@ -339,7 +431,7 @@ def _digest(value: object) -> str: ).hexdigest() -def test_real_cli_replacement_recovers_unknown_and_updates_original_card_without_consuming( +def test_internal_replacement_io_fixture_recovers_unknown_without_qualifying_host_identity( tmp_path: Path, ) -> None: from loopx.thread_agent_binding import ( @@ -381,8 +473,8 @@ def test_real_cli_replacement_recovers_unknown_and_updates_original_card_without replacement = {**original, "thread_id": "thread-replacement-fixture"} def cli(*arguments: str, thread: str, expected_exit: int = 0) -> dict[str, Any]: - # This exercises the real CLI's existing ambient-context fence, not a - # claim of hostile-process authentication (tracked independently). + # Only the locator-only Inbox read remains public here. Positive + # operation calls below exercise internal IO, not host authentication. result = subprocess.run( [ sys.executable, @@ -405,30 +497,17 @@ def cli(*arguments: str, thread: str, expected_exit: int = 0) -> dict[str, Any]: assert result.returncode == expected_exit, result.stdout + result.stderr return json.loads(result.stdout) - def operation( - command: str, actor: dict[str, str], *extra: str, expected_exit: int = 0 - ): - return cli( - "goal-channel", - command, - "--goal-id", - GOAL_ID, - "--agent-id", - AGENT_ID, - "--proposal-id", - proposal["proposal_id"], - "--host-surface", - actor["host_surface"], - "--thread-id", - actor["thread_id"], - *extra, - thread=actor["thread_id"], - expected_exit=expected_exit, + def operation(actor: dict[str, str], action: str, **request: Any): + return agent_operation_action( + runtime, + registry, + proposal_id=proposal["proposal_id"], + actor=actor, + action=action, + **request, ) - consumed = operation( - "consume-operation", original, "--consumption-id", "attempt-1", "--execute" - ) + consumed = operation(original, "consume", consumption_id="attempt-1") assert consumed["execution_allowed"] unknown = { "schema_version": "loopx_operation_outcome_v0", @@ -450,11 +529,7 @@ def operation( "summary": "Synthetic unknown submission requires original-system evidence.", "evidence_refs": ["receipt:unknown-fixture"], } - outcome_path = tmp_path / "outcome.json" - outcome_path.write_text(json.dumps(unknown)) - operation( - "report-operation", original, "--outcome-json", str(outcome_path), "--execute" - ) + operation(original, "report", outcome=unknown) unbind_thread_agent_in_registry(registry_path=registry, **original, execute=True) bind_thread_agent_in_registry(registry_path=registry, **replacement, execute=True) inbox = cli( @@ -467,20 +542,13 @@ def operation( thread=replacement["thread_id"], ) assert inbox["operation_handoffs"][0]["binding_current"] is False - inspected = operation("inspect-operation", replacement) + inspected = operation(replacement, "inspect") assert inspected["outcome"] == unknown and inspected["route"] == original assert inspected["access"]["owner"] == replacement assert inspected["access"]["permission"] == "historical_evidence_only" before = store.path.read_bytes() - rejected = operation( - "consume-operation", - replacement, - "--consumption-id", - "attempt-2", - "--execute", - expected_exit=1, - ) - assert rejected["blocker"] == "operation_handoff_conflict" + with pytest.raises(ActionConflictError): + operation(replacement, "consume", consumption_id="attempt-2") assert store.path.read_bytes() == before final = { **unknown, @@ -489,32 +557,16 @@ def operation( "summary": "Synthetic original-system evidence proves no submission.", "evidence_refs": ["receipt:reconciled-fixture"], } - outcome_path.write_text(json.dumps(final)) - operation( - "report-operation", - replacement, - "--outcome-json", - str(outcome_path), - "--execute", - expected_exit=1, - ) + with pytest.raises(ActionConflictError): + operation(replacement, "report", outcome=final) assert store.path.read_bytes() == before final["reconciles_outcome_digest"] = _digest(unknown) - outcome_path.write_text(json.dumps(final)) - dry = operation( - "report-operation", replacement, "--outcome-json", str(outcome_path) - ) + dry = operation(replacement, "inspect") assert not dry["execution_allowed"] and store.path.read_bytes() == before - reported = operation( - "report-operation", - replacement, - "--outcome-json", - str(outcome_path), - "--execute", - ) + reported = operation(replacement, "report", outcome=final) assert not reported["execution_allowed"] and not reported["needs_reconciliation"] assert reported["reconciliation_report"]["owner"] == replacement - inspected = operation("inspect-operation", replacement) + inspected = operation(replacement, "inspect") assert inspected["reconciliation"] == final and inspected["outcome"] == unknown assert inspected["consumption"]["route"] == original assert inspected["reconciliation_report"]["original_route"] == original From 132b107eac29362cd4d6d59d52a330c6a309fa14 Mon Sep 17 00:00:00 2001 From: huangruiteng <14976749+huangruiteng@users.noreply.github.com> Date: Wed, 30 Sep 2026 07:51:21 +0800 Subject: [PATCH 07/13] docs(operations): specify the required original-host authentication bridge Signed-off-by: huangruiteng <14976749+huangruiteng@users.noreply.github.com> --- .../human-confirmed-domain-operations-v0.md | 112 ++++++++++++++---- ...an-confirmed-domain-operations-v0.zh-CN.md | 78 +++++++++--- 2 files changed, 148 insertions(+), 42 deletions(-) diff --git a/docs/architecture/rfcs/human-confirmed-domain-operations-v0.md b/docs/architecture/rfcs/human-confirmed-domain-operations-v0.md index 4f56933c88..5e0107933d 100644 --- a/docs/architecture/rfcs/human-confirmed-domain-operations-v0.md +++ b/docs/architecture/rfcs/human-confirmed-domain-operations-v0.md @@ -329,7 +329,7 @@ choose a venue, resume a browser, sign or submit an order. runtime/Goal/Agent scope, not the ordinary request cursor. Restart without it for new/changed work; finishing a page sequence does not resolve obligations. - Dashboard details and the original Lark card use the shared operation frame: - confirmed/waiting for the original Agent; consumed/waiting for real evidence; + confirmed/original-host authentication unavailable; consumed/waiting for real evidence; unknown/reconcile without resubmitting; and a separately verified result. The Dashboard remains read-only for human operation confirmation. There is no new configuration owner: the original request chooses the executor and @@ -338,9 +338,11 @@ choose a venue, resume a browser, sign or submit an order. ### Original-runtime CLI Use the original registry and runtime, not a copied session or another home's -records. These local continuation commands do not require Lark to be installed -or reachable. Preparing/delivering new cards remains subject to its normal -extension and authenticated-ingress checks. +records. The following selectors are reserved for the continuation interface; +**all three public CLI commands currently fail closed** because no qualified +host identity producer is connected. They do not require Lark to report that +gate. Preparing/delivering new cards retains its normal extension and +authenticated-ingress checks, but a confirmation cannot remove this host gate. ```sh loopx --registry REGISTRY --runtime-root RUNTIME goal-channel inspect-operation \ @@ -356,20 +358,80 @@ loopx --registry REGISTRY --runtime-root RUNTIME goal-channel report-operation \ ``` `inspect-operation` and a command without `--execute` never consume authority. -All three CLI commands require the existing host-exported ambient thread -(for example, `CODEX_THREAD_ID`) to match the caller's route; CLI identifiers -are selectors, not caller authentication. Missing, foreign or unsupported host -context fails closed before inspecting private parameters or writing a receipt. -The returned `caller_context_source: trusted_local_host_environment` names an -ambient-context check, **not authenticated session isolation**. Exact-head review -reproduced a same-user process forging this environment and consuming the -original session's authorization. Original-session-exclusive execution is an -unresolved acceptance blocker: this path must not be installed or declared a -live minimum loop until an independently qualified trusted-host transport -proves caller identity. A current registry binding authorizes a route, but does -not authenticate its caller. Internal storage adapters are local IO seams, not -unauthenticated network endpoints; historical recovery below does not close -this separate authentication gap. +The former ambient-thread check (for example, `CODEX_THREAD_ID`) was forgeable +by another same-user process. It is removed, not upgraded to authentication: +even an exact environment/route match returns +`operation_host_authentication_unavailable` before reading private operation +terms, outcome files or writing receipts. Unexpected actor success from an +older runtime also cannot bypass the CLI adapter's absent transport. There is +no `--verified`, self-signing command or environment-token fallback. + +The Inbox still exposes bounded locators and preserves consumed/unknown +obligations. It explicitly requests host integration instead of recommending +another blocked CLI consumption. The shared Dashboard/Lark frame says human +confirmation is recorded but original-host authentication is unavailable. +This containment removes the public environment-forgery path; **it does not +deliver an authenticated positive execution path**. Original-session-exclusive +execution therefore remains an acceptance blocker. Internal locked storage +adapters and their synthetic fixtures validate protocol semantics only, not a +host producer or a live minimum loop. A registry binding authorizes a route, +but does not authenticate its caller. + +### Required host-adapter integration + +The chosen boundary is a **transport-owned, non-exporting operation tool**. +The host handles `loopx_operation` (`inspect`, `consume`, `report`) on the +original session's authenticated tool connection and returns the receipt to +that same connection. This is a required companion contract, not an installed +tool, accepted proof field or a new approval store. + +1. The original configuration owner enrolls and revokes the host issuer against + the existing session binding. A request cannot select its own trust key or + enroll a replacement issuer. Issuer rotation does not change the immutable + operation executor or inherit unused approvals. +2. The host derives session/Turn identity from its native tool-dispatch + metadata, not tool arguments, environment variables, an MCP subprocess's + self-report or an agent-readable key file. It does not expose a general + signer or a reusable bearer token to the model/CLI. Private signing material + remains within a separately trusted host service; same-user environment + spoofing must not reach that service's identity or signing authority. +3. Across a process boundary, the issuer signs the canonical invocation with + Ed25519. The invocation binds issuer/key revision, original GoalRef and + registered Agent, host/session/Turn, original operation and its payload/ + confirmation digests, action and action-argument digest, audience/runtime, + authenticated connection, bounded issue/expiry times and a unique request + ID. Core verifies the owner-pinned issuer, signature, exact scope, current + binding and freshness in TypeScript before the existing locked IO seam. + The authenticated response stays on the original host connection; forwarding + a signed payload to a public CLI must not reveal an execution permission. +4. Authentication proves origin only. Original human confirmation, immutable + terms, active Goal, expiry and one-shot atomic consumption remain separate + gates. Request replay and unknown submission never authorize a second + external effect. A recovery host needs its own authenticated connection; + it may report evidence only under the existing replacement-recovery rules. + +**Concrete dependency:** attached Codex Desktop sessions need a session-bound +operation tool in the Desktop's native tool server, alongside its existing +app-owned tools, plus owner-controlled issuer enrollment. That native server +is not implemented in this LoopX checkout or by its CLI/MCP adapters. The host +maintainer must deliver the producer; this PR cannot substitute environment +identity or silently resume the session in a different process. +`CodexChatAgentSession._check_server_gate` already validates thread/Turn +metadata on LoopX-owned app-server tool calls, but those are different owned +sessions, not proof for an attached Desktop thread. An owned-host integration +must be qualified for its own route and cannot stand in for Desktop acceptance. + +Integrate the real producer and owner-pinned verifier as one follow-up slice; +do not ship an unused signing API or fixture-generated credentials. Qualify +forged environment/route/proof fields, foreign sessions, wrong audience, expiry, +tampering, replay, issuer/binding revocation and original-connection receipt +return. Retain the current public-CLI rejection regression when enabling the +host tool. A real original-session invocation must pass through the same +boundary before the slice can be installed. No new session, copied trajectory, +synthetic group click or real financial side effect is part of engineering QA. + +### One-shot and evidence semantics behind the host gate + Only the first successful atomic consumption returns `execution_allowed: true`. It verifies authenticated confirmation, immutable terms, the current original session, active Goal and expiry, then persists consumption before any browser @@ -402,13 +464,14 @@ original Goal/Agent registration or an unsupported registry profile is an explicit error, not permission to transplant the operation. After the original route is withdrawn, a **currently registered and bound -replacement session of the same Goal and Agent** may use its own caller route -with `inspect-operation` and `report-operation`. It may inspect only an already +replacement session of the same Goal and Agent** may use its own authenticated +host route for inspection and reporting once that transport is qualified. The +internal IO seam may inspect only an already consumed operation and report original-system historical evidence. It cannot consume an unspent ticket, change the original executor or obtain a second execution permission. Recovery is not admitted while the original binding is still current, for an unbound replacement or for a different Goal/Agent. The -CLI returns an explicit `access.owner`, `original_route`, +protocol returns an explicit `access.owner`, `original_route`, `permission: "historical_evidence_only"` and binding authority; it never asks the replacement to impersonate the old thread. The original executor route remains immutable. The same registry lock is held from recovery-binding validation @@ -418,12 +481,13 @@ Result evidence remains unchanged. The canonical operation separately appends `outcome_report` or `reconciliation_report` provenance with the actual reporter, original route, evidence-only permission, authority source, consumption ID and recorded time. Identical result retries preserve the first committed provenance; -they do not relabel its author or grant execution. CLI inspection exposes both +they do not relabel its author or grant execution. Internal inspection exposes both the original unknown outcome and its immutable reconciliation/provenance. The existing shared Dashboard frame and original Lark-card recovery consume the same canonical result; neither gets a separate recovery approval store or an -execution control. These synthetic CLI/result-card checks are historical -recovery acceptance, not trusted-host authentication or live-group acceptance. +execution control. The prior positive CLI fixtures are now internal-IO/result-card +checks because the public host gate is closed. They establish historical +recovery semantics, not trusted-host authentication or live-group acceptance. An unknown original outcome is immutable. A definitive report appends `operation.reconciliation` and binds `reconciles_outcome_digest` to the exact diff --git a/docs/architecture/rfcs/human-confirmed-domain-operations-v0.zh-CN.md b/docs/architecture/rfcs/human-confirmed-domain-operations-v0.zh-CN.md index 2f2e8fee7c..cfd8732e68 100644 --- a/docs/architecture/rfcs/human-confirmed-domain-operations-v0.zh-CN.md +++ b/docs/architecture/rfcs/human-confirmed-domain-operations-v0.zh-CN.md @@ -260,15 +260,16 @@ M1 同时涵盖 UI 与后端,不要拆成“后端 PR 已完成”而遗忘前 `manager-inbox read --operation-cursor CURSOR` 逐页找回其余操作,即使前 20 条未知 结果长期未解决。游标绑定原 runtime/Goal/Agent 范围,与普通请求游标独立;新增或 改变的工作应无游标重读,遍历结束不代表义务已解决。 -- Dashboard 详情与原 Lark 卡使用同一操作 frame:已确认待原 Agent、已消费待真实 +- Dashboard 详情与原 Lark 卡使用同一操作 frame:已确认但原宿主认证尚未接通、已消费待真实 证据、未知须对账不得重提,以及独立核验的结果。Dashboard 对用户操作确认保持只读。 不新增配置权威:执行器由原请求选择,原渠道/绑定 owner 仍是权威。 ### 原运行时 CLI 使用原 registry 和 runtime,不复制 session,也不借用另一个 home 的记录。 -以下本地续接命令不要求 Lark 已安装或可达;新卡片的准备/投递仍须通过原扩展与 -经认证入口的检查。 +以下选择器保留为续接接口;由于尚未接通合格的宿主身份签发端,**三个公开 CLI 命令 +目前均拒绝执行**。报告该门禁不要求 Lark 已安装或可达;新卡片准备/投递仍须通过原 +扩展与经认证入口的检查,但用户确认不能消除此宿主门禁。 ```sh loopx --registry REGISTRY --runtime-root RUNTIME goal-channel inspect-operation \ @@ -283,15 +284,56 @@ loopx --registry REGISTRY --runtime-root RUNTIME goal-channel report-operation \ --host-surface HOST --thread-id ORIGINAL_THREAD --outcome-json OUTCOME --execute ``` -`inspect-operation` 和没有 `--execute` 的命令均不消费授权。三个 CLI 命令均要求 -既有宿主导出的环境线程(例如 `CODEX_THREAD_ID`)匹配调用会话的路由;命令行标识只是选择器, -不是调用者认证。环境缺失、不匹配或不支持的宿主在读取私有参数或写回之前即拒绝。 -返回的 `caller_context_source: trusted_local_host_environment` 只表示环境上下文检查, -**不等于经认证的 session 隔离**。精确头复审已复现:同用户进程可自行设置该环境, -消费原会话授权。“原会话专属执行”仍是未解决的验收阻塞;在独立验收受信宿主传输能 -证明调用者身份之前,不得安装此路径或宣称真实最小闭环。当前 registry 绑定只授权 -路由,不认证调用者。内部存储 adapter 是本地 IO 接缝,不是未认证的网络端点;下述 -历史对账恢复并不关闭这个独立的认证缺口。 +`inspect-operation` 和没有 `--execute` 的命令均不消费授权。旧环境线程检查 +(例如 `CODEX_THREAD_ID`)可被另一同用户进程伪造,现已移除,而非升级成认证: +即使环境/路由完全匹配,仍在读取私有操作条款、结果文件或写回前返回 +`operation_host_authentication_unavailable`。旧运行时意外返回 actor 成功也不能 +绕过 CLI adapter 尚未接通的认证传输;没有 `--verified`、自行签发命令或环境令牌兜底。 + +Inbox 继续展示有界定位信息,保留已消费/未知结果义务;它明确要求宿主接入,不再建议 +重试被阻断的 CLI 消费。Dashboard/Lark 共享 frame 明确说明用户确认已记录,但原 +宿主身份认证尚未接通。这项收紧消除了公开环境伪造路径,**并未交付经认证的正向执行 +路径**。“原会话专属执行”因此仍是验收阻塞。内部锁定存储 adapter 与合成夹具只验证 +协议语义,不证明宿主签发端或真实最小闭环。registry 绑定授权路由,不认证调用者。 + +### 必需的宿主 adapter 接入 + +选定边界是**由传输拥有、不可导出的操作工具**。宿主在原 session 的认证工具连接上 +处理 `loopx_operation`(`inspect`、`consume`、`report`),并把回执返回同一连接。 +这是必需的配套契约,不是已安装工具、已接受的 proof 字段或新的审批存储。 + +1. 原配置 owner 在既有 session binding 上登记和撤销宿主签发者。请求不能自行选择 + 信任公钥或登记替代签发者;密钥轮换不改变不可变执行器,也不继承未消费授权。 +2. 宿主从原生工具分发元数据取得 session/Turn 身份,不使用工具参数、环境变量、 + MCP 子进程自报或 Agent 可读的密钥文件。不向模型/CLI 暴露通用签名器或可复用 + bearer token;私钥留在独立受信的宿主服务中,同用户环境伪造不得到达其身份或 + 签名权威。 +3. 跨进程时,签发者用 Ed25519 签署规范 invocation,绑定签发者/密钥 revision、 + 原 GoalRef/注册 Agent、host/session/Turn、原 operation 及载荷/确认摘要、 + action/参数摘要、audience/runtime、认证连接、有界签发/到期时间及唯一 request ID。 + Core 在 TypeScript 中核验 owner 固定的签发者、签名、精确范围、当前绑定与时效, + 再调用既有锁定 IO 接缝。认证响应仅返回原宿主连接;向公开 CLI 转发签名载荷不能 + 获得执行许可。 +4. 认证只证明来源。原用户确认、不可变条款、有效 Goal、到期与原子一次消费仍是 + 独立门禁;重放或未知提交不允许第二次外部操作。恢复宿主也须使用自己的认证连接, + 仅按既有接手恢复规则回写证据。 + +**具体依赖:**外接 Codex Desktop session 需要 Desktop 原生工具服务在既有 app +自有工具旁提供 session-bound 操作工具,并接入 owner 控制的签发者登记。该原生 +服务不在这个 LoopX checkout 内,现有 CLI/MCP adapter 也未实现它。宿主维护者 +须交付签发端;本 PR 不能用环境身份替代,也不能在另一进程悄悄恢复原 session。 +`CodexChatAgentSession._check_server_gate` 已核对 LoopX 自建 app-server 工具 +调用的 thread/Turn 元数据,但那是另一类自有 session,不证明外接 Desktop 线程。 +自有宿主接入须独立验收自己的路由,不能替代 Desktop 验收。 + +真实签发端与 owner 固定的验签端应在同一后续切片接通,不交付无人使用的签名 API +或夹具生成的凭据。验收须覆盖伪造环境/路由/proof 字段、外来 session、错误 audience、 +到期、篡改、重放、签发者/绑定撤销及原连接回执返回;启用宿主工具时保留当前公开 CLI +拒绝回归。在同一边界通过真实原会话调用前不得安装本切片。工程 QA 不创建新 session、 +复制轨迹、制造群确认或触发真实金融副作用。 + +### 宿主门禁后的单次消费与证据语义 + 只有首次成功的原子消费 返回 `execution_allowed: true`。它检查经认证的确认、不可变条款、原 session 当前绑定、 有效 Goal 与到期时间,并在任何浏览器操作之前持久记录消费。所有重试,包括响应丢失 @@ -315,20 +357,20 @@ claim、执行器 revision、consumption ID,要求 `projection_verified: true` 生命周期保护保持不变,不隐式迁入生命周期专用 registry 或另一个 home。 原 Goal/Agent 注册缺失或 registry profile 不受支持均明确报错,不允许移植操作权限。 -原路由撤销后,**同 Goal、同 Agent、当前已注册并绑定的接手会话**可以用自己的调用 -路由执行 `inspect-operation` 与 `report-operation`。它只能检查已消费操作,回写 +原路由撤销后,**同 Goal、同 Agent、当前已注册并绑定的接手会话**在认证传输验收后 +可以用自己的宿主路由检查与回写。内部 IO 接缝只能检查已消费操作,回写 原系统的历史证据,不能消费未使用的票据、改写原执行器或取得第二次执行许可。原绑定 -仍有效、接手会话未绑定、Goal/Agent 不同均拒绝恢复准入。CLI 明确返回 `access.owner`、 +仍有效、接手会话未绑定、Goal/Agent 不同均拒绝恢复准入。协议明确返回 `access.owner`、 `original_route`、`permission: "historical_evidence_only"` 与绑定权威,不要求接手者 冒充旧线程。原执行路由保持不可变;从接手绑定核对到结果提交共用 registry 锁,先提交 的撤销拒绝回写。 结果证据本身保持原样。规范操作单独追加 `outcome_report` 或 `reconciliation_report` 来源,记录实际报告者、原路由、仅历史证据权限、权威来源、consumption ID 与记录时间。 -相同结果重试保留首次已提交来源,不重标作者或授予执行。CLI 检查同时返回原未知 +相同结果重试保留首次已提交来源,不重标作者或授予执行。内部检查同时返回原未知 outcome 与不可变对账/来源;既有 Dashboard 共享 frame 与原 Lark 卡恢复消费同一 -规范结果,不另建恢复审批存储或执行控件。这些合成 CLI/原卡读回只验收历史对账恢复, -不代替受信宿主认证或真实群验收。 +规范结果,不另建恢复审批存储或执行控件。因公开宿主门禁关闭,旧正向 CLI 夹具已改为 +内部 IO/原卡读回检查;它们只证明历史对账语义,不代替受信宿主认证或真实群验收。 未知的原始 outcome 不可修改。确定性回写追加 `operation.reconciliation`,并通过 `reconciles_outcome_digest` 绑定原未知结果的精确摘要;只有该证据才关闭恢复义务。 From 69ea8ce2790fbd99d6d531689fc066bbe58d1e26 Mon Sep 17 00:00:00 2001 From: huangruiteng <14976749+huangruiteng@users.noreply.github.com> Date: Wed, 30 Sep 2026 13:20:03 +0800 Subject: [PATCH 08/13] feat(operations): bind confirmed execution to an owned managed Turn Signed-off-by: huangruiteng <14976749+huangruiteng@users.noreply.github.com> --- loopx/chat_action_normalization.py | 74 +++- loopx/chat_action_store.py | 4 +- loopx/chat_agent.py | 74 +++- loopx/cli_commands/turn.py | 10 + loopx/cli_commands/turn_registration.py | 13 +- loopx/cli_commands/turn_selection.py | 2 + .../control_plane/collaboration/delegation.ts | 1 + .../collaboration/delegation_context.py | 2 +- .../collaboration/operation_handoff.py | 228 +++++++++--- .../control_plane/effect_runtime_handlers.ts | 4 +- loopx/control_plane/turn_driver/codex_cli.py | 39 +- .../turn_driver/codex_operation_host.py | 349 ++++++++++++++++++ .../control_plane/turn_driver/host_binding.py | 41 +- .../work_items/operation_agent_handoff.ts | 69 +++- .../extensions/lark/goal_channel_operation.py | 19 +- .../operation_agent_handoff.test.ts | 62 +++- .../test_lark_goal_channel_operation.py | 46 ++- tests/test_chat_operation_actions.py | 312 +++++++++++++++- tests/test_codex_operation_host.py | 283 ++++++++++++++ tests/test_delegation_preflight.py | 14 +- tests/test_turn_managed_executor_binding.py | 28 ++ 21 files changed, 1568 insertions(+), 106 deletions(-) create mode 100644 loopx/control_plane/turn_driver/codex_operation_host.py create mode 100644 tests/test_codex_operation_host.py diff --git a/loopx/chat_action_normalization.py b/loopx/chat_action_normalization.py index 2d93396c7d..bc66278ec7 100644 --- a/loopx/chat_action_normalization.py +++ b/loopx/chat_action_normalization.py @@ -148,10 +148,10 @@ def _normalize( raw_executor = values.get("executor") origin_goal_ref = None - if ( - isinstance(raw_executor, Mapping) - and raw_executor.get("kind") == "agent_session" - ): + if isinstance(raw_executor, Mapping) and raw_executor.get("kind") in { + "agent_session", + "managed_turn", + }: from .control_plane.collaboration.goal_instance_scope import ( collaboration_goal_scope, ) @@ -171,15 +171,23 @@ def _normalize( ) except EffectRuntimeRejected as exc: raise ValueError(str(exc)) from exc - binding = resolve_registry_thread_agent_binding( - registry_path=self.registry_path, - host_surface=str(executor["host_surface"]), - thread_id=str(executor["thread_id"]), + binding = ( + resolve_registry_thread_agent_binding( + registry_path=self.registry_path, + host_surface=str(executor.get("host_surface", "")), + thread_id=str(executor.get("thread_id", "")), + ) + if executor["kind"] == "agent_session" + else None ) - if binding.get("status") != "bound" or ( - binding.get("goal_id"), - binding.get("agent_id"), - ) != (goal_id, agent_id): + if executor["kind"] == "agent_session" and ( + binding.get("status") != "bound" + or ( + binding.get("goal_id"), + binding.get("agent_id"), + ) + != (goal_id, agent_id) + ): raise ValueError( "agent operation requires the original registered session" ) @@ -196,6 +204,23 @@ def _normalize( if goal_is_stopped(scope.goal): raise ValueError("agent operation Goal is stopped") origin_goal_ref = scope.current_goal_ref + if executor["kind"] == "managed_turn": + from .control_plane.collaboration.operation_handoff import ( + managed_operation_binding_current, + ) + + if not managed_operation_binding_current( + self.store.root.parent.parent, + { + "goal_id": goal_id, + "agent_id": agent_id, + "executor": executor, + "origin_goal_ref": origin_goal_ref, + }, + ): + raise ValueError( + "managed operation requires its current Turn session and profile" + ) elif not isinstance(raw_executor, Mapping) or set(raw_executor) != { "extension_id", "protocol", @@ -208,6 +233,26 @@ def _normalize( field: _opaque(raw_executor.get(field), field=f"executor.{field}") for field in ("extension_id", "protocol", "permission", "revision") } + source_routes = [ + route + for route in (goal.get("coordination") or {}).get( + "thread_agent_bindings", [] + ) + if isinstance(route, Mapping) and route.get("agent_id") == agent_id + and all(isinstance(route.get(key), str) and route[key] + for key in ("host_surface", "thread_id")) + ] + source_route = ( + { + "goal_id": goal_id, + **{ + key: source_routes[0][key] + for key in ("agent_id", "host_surface", "thread_id") + }, + } + if len(source_routes) == 1 + else None + ) expires_at = parse_timestamp( _text(values.get("expires_at"), field="expires_at", limit=80) ) @@ -256,6 +301,11 @@ def _normalize( "expires_at": utc_isoformat(expires_at), "authorized_principals": principals, "executor": executor, + **( + {"source_route": source_route} + if executor.get("kind") == "managed_turn" + else {} + ), **( {"origin_goal_ref": origin_goal_ref} if origin_goal_ref is not None diff --git a/loopx/chat_action_store.py b/loopx/chat_action_store.py index 38cd7bca41..d42b3f1bf6 100644 --- a/loopx/chat_action_store.py +++ b/loopx/chat_action_store.py @@ -981,7 +981,7 @@ def observe_operation_outcome( raise KeyError("typed operation was not found") parameters = proposal.get("normalized_parameters") or {} report_provenance = None - if (parameters.get("executor") or {}).get("kind") == "agent_session": + if (parameters.get("executor") or {}).get("kind") in {"agent_session", "managed_turn"}: plan = self._agent_operation_plan( proposal, action="report", @@ -1151,7 +1151,7 @@ def record_operation_result_delivery( ) is_agent = ( (proposal.get("normalized_parameters") or {}).get("executor") or {} - ).get("kind") == "agent_session" + ).get("kind") in {"agent_session", "managed_turn"} expected_stage = ( "reconciled" if operation.get("reconciliation") is not None diff --git a/loopx/chat_agent.py b/loopx/chat_agent.py index 12f4932e82..e66ed6924d 100644 --- a/loopx/chat_agent.py +++ b/loopx/chat_agent.py @@ -423,6 +423,7 @@ class CodexChatAgentSession: work_dir: Path context_summary: str = "" execution_mode: bool = False + process_tree_owned: bool = False runtime_profile: str = "restricted" sandbox: str = "read-only" model: str | None = None @@ -436,6 +437,9 @@ class CodexChatAgentSession: read_tool_handler: Callable[[str, Any], dict[str, Any]] | None = field( default=None, repr=False ) + bound_tool_handler: Callable[[str, Any, dict[str, Any]], dict[str, Any]] | None = ( + field(default=None, repr=False) + ) _pending_events: "queue.Queue[dict[str, Any]]" = field( default_factory=queue.Queue, repr=False ) @@ -465,12 +469,14 @@ def start( hard_timeout_sec: float = 900.0, resume_thread_id: str | None = None, execution_mode: bool = False, + isolate_process_tree: bool = False, runtime_profile: str = "restricted", sandbox: str | None = None, codex_home: Path | None = None, model: str | None = None, reasoning_effort: str | None = None, dynamic_tools: list[dict[str, Any]] | None = None, + host_config: dict[str, Any] | None = None, _compatibility_catalog_path: Path | None = None, ) -> "CodexChatAgentSession": resolved = shutil.which(codex_bin) @@ -535,6 +541,7 @@ def start( text=True, encoding="utf-8", bufsize=1, + start_new_session=isolate_process_tree and os.name == "posix", ) except OSError as exc: raise CodexChatAgentError( @@ -561,6 +568,7 @@ def start( idle_timeout_sec=idle_timeout_sec, hard_timeout_sec=hard_timeout_sec, execution_mode=execution_mode, + process_tree_owned=isolate_process_tree, runtime_profile=runtime_profile, sandbox=selected_sandbox, model=model, @@ -592,8 +600,17 @@ def start( "cwd": str(root), **({"model": model} if model else {}), **( - {"config": {"model_reasoning_effort": reasoning_effort}} - if reasoning_effort + { + "config": { + **(host_config or {}), + **( + {"model_reasoning_effort": reasoning_effort} + if reasoning_effort + else {} + ), + } + } + if reasoning_effort or host_config else {} ), "sandbox": selected_sandbox, @@ -649,12 +666,14 @@ def start( hard_timeout_sec=hard_timeout_sec, resume_thread_id=resume_thread_id, execution_mode=execution_mode, + isolate_process_tree=isolate_process_tree, runtime_profile=runtime_profile, sandbox=selected_sandbox, codex_home=runtime_home, model=model, reasoning_effort=reasoning_effort, dynamic_tools=dynamic_tools, + host_config=host_config, _compatibility_catalog_path=catalog_path, ) except Exception: @@ -731,7 +750,10 @@ def _check_server_gate(self, message: dict[str, Any]) -> bool: if ( message.get("id") is not None and message.get("method") == "item/tool/call" - and self.read_tool_handler + and ( + self.read_tool_handler + or (self.execution_mode and self.bound_tool_handler) + ) ): params = message.get("params") or {} valid = ( @@ -743,8 +765,20 @@ def _check_server_gate(self, message: dict[str, Any]) -> bool: ) try: result = ( - self.read_tool_handler( - params.get("tool", ""), params.get("arguments") + ( + self.bound_tool_handler( + params.get("tool", ""), + params.get("arguments"), + { + "thread_id": params["threadId"], + "host_turn_id": params["turnId"], + "call_id": params.get("callId"), + }, + ) + if self.execution_mode and self.bound_tool_handler + else self.read_tool_handler( + params.get("tool", ""), params.get("arguments") + ) ) if valid else { @@ -805,7 +839,9 @@ def _request( except queue.Empty: remaining = deadline - time.monotonic() if remaining <= 0: - raise self._runtime_error("Codex app-server timed out.") + raise self._timeout_error( + "response_timeout", "Codex app-server timed out." + ) try: raw = self.messages.get(timeout=min(0.1, remaining)) except queue.Empty: @@ -902,17 +938,22 @@ def send( *, attachments: list[dict[str, Any]] | None = None, on_event: Callable[[str, dict[str, Any]], None] | None = None, + output_schema: dict[str, Any] | None = None, ) -> dict[str, Any]: text = " ".join(str(user_message or "").split()) if not text: raise ValueError("user message is required") + if output_schema is not None and not self.execution_mode: + raise ValueError("structured Turn output requires an execution session") with self._request_id_lock: request_id = self.next_request_id self.next_request_id += 1 turn_input: list[dict[str, Any]] = [ { "type": "text", - "text": _turn_prompt( + "text": text + if output_schema is not None + else _turn_prompt( text, context_summary=self.context_summary, execution_mode=self.execution_mode, @@ -933,6 +974,9 @@ def send( **({"model": self.model} if self.model else {}), **({"effort": self.reasoning_effort} if self.reasoning_effort else {}), "approvalPolicy": "never", + **( + {"outputSchema": output_schema} if output_schema is not None else {} + ), }, request_id=request_id, ) @@ -1068,6 +1112,17 @@ def send( visible_delta_count += 1 on_event("answer.delta", {"text": visible_tail}) raw_response = "".join(parts) + if output_schema is not None: + try: + result = json.loads(raw_response) + except (ValueError, TypeError) as exc: + raise self._runtime_error( + "Codex Turn did not return structured output." + ) from exc + if not isinstance(result, dict): + raise self._runtime_error("Codex Turn output is not an object.") + self.current_turn_id = "" + return result response = parse_agent_response( raw_response, protected_paths=[self.work_dir], @@ -1095,6 +1150,11 @@ def _timeout_error(self, error_code: str, summary: str) -> CodexChatTimeoutError ) def close(self) -> None: + if self.process_tree_owned: + from .extensions.process_runtime import terminate_process_tree + + terminate_process_tree(self.process, grace_seconds=0.1) + return if self.process.poll() is not None: return self.process.terminate() diff --git a/loopx/cli_commands/turn.py b/loopx/cli_commands/turn.py index 568ad9130f..0cd7c3e24d 100644 --- a/loopx/cli_commands/turn.py +++ b/loopx/cli_commands/turn.py @@ -129,6 +129,8 @@ def handle_turn_command( planned_goal_ref=goal_ref, ) strict_goal_admission = goal_admission if goal_admission.enabled else None + if getattr(args, "codex_operation_tools", False) and args.host != "codex-cli": + raise ValueError("--codex-operation-tools requires the codex-cli host") # Planning and dry-run execution inspect existing admitted intents. # Only an executing wake may sync inboxes or reserve a calendar window. turn_start_hook_dispatch = {} @@ -994,6 +996,14 @@ def run_built_in_host( } if strict_goal_admission is not None: options["goal_admission"] = strict_goal_admission + if getattr(args, "codex_operation_tools", False): + from ..control_plane.turn_driver.codex_operation_host import ( + run_codex_operation_host, + ) + + return run_codex_operation_host( + request, registry_path=registry_path, **options + ) return run_codex_cli_host(request, **options) host_runner = run_built_in_host diff --git a/loopx/cli_commands/turn_registration.py b/loopx/cli_commands/turn_registration.py index d916f2fc85..67ee763808 100644 --- a/loopx/cli_commands/turn_registration.py +++ b/loopx/cli_commands/turn_registration.py @@ -219,6 +219,11 @@ def register_turn_commands( help="Codex CLI executable used by the built-in codex-cli host.", ) run_once.add_argument("--codex-model") + run_once.add_argument( + "--codex-operation-tools", + action="store_true", + help="Opt in to the owned app-server operation transport for this admitted codex-cli Turn. Reuses the original Todo/session; does not authenticate an attached Desktop or grant domain effects.", + ) run_once.add_argument( "--codex-reasoning-effort", choices=list(REASONING_EFFORTS), @@ -232,9 +237,11 @@ def register_turn_commands( "--codex-sandbox", choices=["read-only", "workspace-write", "danger-full-access"], default="read-only", - help=("Codex CLI sandbox (default: read-only). danger-full-access explicitly " - "disables the inner sandbox; callers must provide their own isolation. " - "The setting is passed explicitly for both new and resumed sessions."), + help=( + "Codex CLI sandbox (default: read-only). danger-full-access explicitly " + "disables the inner sandbox; callers must provide their own isolation. " + "The setting is passed explicitly for both new and resumed sessions." + ), ) run_once.add_argument( "--codex-mcp-server-json", diff --git a/loopx/cli_commands/turn_selection.py b/loopx/cli_commands/turn_selection.py index ab48bbe9b0..8253444c38 100644 --- a/loopx/cli_commands/turn_selection.py +++ b/loopx/cli_commands/turn_selection.py @@ -125,4 +125,6 @@ def managed_executor_cli_binding( if args.host == "dsh" else None ), + codex_operation_tools=bool(getattr(args, "codex_operation_tools", False)), + codex_sandbox=getattr(args, "codex_sandbox", "read-only"), ) diff --git a/loopx/control_plane/collaboration/delegation.ts b/loopx/control_plane/collaboration/delegation.ts index f661435365..a09f5afcd2 100644 --- a/loopx/control_plane/collaboration/delegation.ts +++ b/loopx/control_plane/collaboration/delegation.ts @@ -269,6 +269,7 @@ export function delegationPreflight(params: JsonObject): JsonObject { promotion_from_surface_allowed: false, executor: {host: executor.executor, available: executor.available, reason: executor.unavailable_reason, profile: executor.execution_profile, + ...(executor.operation_transport ? {operation_transport: executor.operation_transport} : {}), ...delegationRuntimeFacts(executor)}, effects, note: "Point-in-time preflight, not an execution permit or evidence of running work. " diff --git a/loopx/control_plane/collaboration/delegation_context.py b/loopx/control_plane/collaboration/delegation_context.py index 1886379374..881ac31c9f 100644 --- a/loopx/control_plane/collaboration/delegation_context.py +++ b/loopx/control_plane/collaboration/delegation_context.py @@ -67,7 +67,7 @@ def _route(binding: dict[str, Any]) -> dict[str, Any]: row["reason_code"] = reason # Transport the existing host owner's public observations unchanged. The # Python adapter neither reprobes nor derives another admission decision. - for key in ("runtime_probe", "unavailable_remediation"): + for key in ("runtime_probe", "unavailable_remediation", "operation_transport"): if key in executor: row[key] = executor[key] return row diff --git a/loopx/control_plane/collaboration/operation_handoff.py b/loopx/control_plane/collaboration/operation_handoff.py index b2e0f37695..70c162dd5a 100644 --- a/loopx/control_plane/collaboration/operation_handoff.py +++ b/loopx/control_plane/collaboration/operation_handoff.py @@ -8,6 +8,7 @@ from collections.abc import Mapping from pathlib import Path +from contextlib import ExitStack from typing import Any from ...chat_action_store import ActionConflictError, ChatActionStore @@ -27,8 +28,37 @@ def _store(runtime_root: Path) -> ChatActionStore: return ChatActionStore(root) -def _binding(registry_path: Path, parameters: Mapping[str, Any]) -> bool: +def managed_operation_binding_current( + runtime_root: Path, parameters: Mapping[str, Any] +) -> bool: + """Read the original Turn session owner; the shared TS owner judges it.""" + from ..effect_runtime import effect_runtime_result + from ..turn_driver.codex_cli import load_codex_cli_session + + executor = parameters["executor"] + lineage = { + "goal_id": parameters["goal_id"], + "agent_id": parameters["agent_id"], + "todo_id": executor["todo_id"], + } + return ( + effect_runtime_result( + "operation.managed_binding.current", + { + "parameters": dict(parameters), + "session": load_codex_cli_session(runtime_root, lineage=lineage), + }, + )["current"] + is True + ) + + +def _binding( + registry_path: Path, parameters: Mapping[str, Any], runtime_root: Path +) -> bool: executor = parameters["executor"] + if executor.get("kind") == "managed_turn": + return managed_operation_binding_current(runtime_root, parameters) observed = resolve_registry_thread_agent_binding( registry_path=registry_path, host_surface=executor["host_surface"], @@ -65,7 +95,7 @@ def pending_operation_handoffs( if ( proposal.get("action_kind") != "operation.execute" or parameters.get("agent_id") != agent_id - or executor.get("kind") != "agent_session" + or executor.get("kind") not in {"agent_session", "managed_turn"} ): continue if ( @@ -85,7 +115,9 @@ def pending_operation_handoffs( "submission_unknown", }: continue - current = registry_path is None or _binding(registry_path, parameters) + current = registry_path is None or _binding( + registry_path, parameters, runtime_root + ) if not current and not plan["needs_reconciliation"]: continue result.append( @@ -93,12 +125,20 @@ def pending_operation_handoffs( **plan, "binding_current": current, "summary": proposal["summary"], - "instruction": "The original host must authenticate through its session-bound tool transport before reading " - "private operation terms or consuming authority. No qualified producer is connected to the CLI; " - "environment thread ids are not identity proof. " - "Only the first successful consumption permits execution; consumed/unknown results require " + "instruction": ( + "Use the bound managed Turn's loopx_operation tool; source conversation flags grant no authority. " + if executor.get("kind") == "managed_turn" + else "The original host must authenticate through its session-bound tool transport before reading " + "private operation terms or consuming authority. No qualified producer is connected to the CLI; " + "environment thread ids are not identity proof. " + ) + + "Only the first successful consumption permits execution; consumed/unknown results require " "original external-system reconciliation, never another submission. Inbox delivery is not execution authority.", - "next_action": "Integrate the original host's authenticated session-bound tool transport; do not retry via environment identity." + "next_action": ( + "Resume the original managed binding through delegation/Turn; inspect and consume through loopx_operation." + if executor.get("kind") == "managed_turn" + else "Integrate the original host's authenticated session-bound tool transport; do not retry via environment identity." + ) if plan["status"] == "authorized_pending" else "Reconcile the original external result; do not submit again.", } @@ -172,57 +212,129 @@ def agent_operation_action( record=ref, route=ref, ) - current = _binding(registry_path, parameters) - actor_current = ( - _binding(registry_path, {**parameters, "executor": actor}) - if action in {"inspect", "report"} - else False - ) - if action == "inspect": - plan = store._agent_operation_plan( - proposal, - action="inspect", - actor=dict(actor), - binding_current=current, - actor_binding_current=actor_current, + # Keep the existing session owner stable through the action-store commit. + # Session writes and operation commits share Goal -> registry -> session + # -> action-store order. No lock is held across a domain effect. + from ...file_lock import exclusive_file_lock + from ..turn_driver.codex_cli import _session_path + + executor = parameters["executor"] + session_paths = set() + if executor.get("kind") == "managed_turn": + session_paths.add( + _session_path( + runtime_root, + { + "goal_id": parameters["goal_id"], + "agent_id": parameters["agent_id"], + "todo_id": executor["todo_id"], + }, + ) ) - return { - **plan, - "binding_current": current, - "parameters": parameters, - "confirmation": proposal["operation"].get("confirmation"), - "consumption": proposal["operation"].get("agent_handoff"), - "outcome": proposal["operation"].get("outcome"), - "reconciliation": proposal["operation"].get("reconciliation"), - "outcome_report": proposal["operation"].get("outcome_report"), - "reconciliation_report": proposal["operation"].get( - "reconciliation_report" - ), - } - if action == "consume": - return store.consume_agent_operation( - proposal_id, - actor=actor, - binding_current=current, - consumption_id=str(consumption_id or ""), + if actor.get("host_surface") == "loopx-managed-codex": + session_paths.add( + _session_path( + runtime_root, + { + "goal_id": actor["goal_id"], + "agent_id": actor["agent_id"], + "todo_id": actor["todo_id"], + }, + ) ) - if action == "report": - updated = store.observe_operation_outcome( + with ExitStack() as locks: + # A recovery Turn may have a different Todo. Hold both owners in + # deterministic order, so replacement revocation cannot race report. + for path in sorted(session_paths): + locks.enter_context(exclusive_file_lock(path)) + return _commit_agent_operation( + store, + runtime_root, + registry_path, + parameters, + proposal, proposal_id, - outcome=outcome or {}, - agent_actor=actor, - agent_binding_current=current, - agent_actor_binding_current=actor_current, + actor, + action, + consumption_id, + outcome, ) - plan = store._agent_operation_plan(updated, action="project") - return { - "ok": True, - **plan, - "outcome": updated["operation"].get("reconciliation") - or updated["operation"]["outcome"], - "outcome_report": updated["operation"].get("outcome_report"), - "reconciliation_report": updated["operation"].get( - "reconciliation_report" - ), - } - raise ValueError("unsupported agent operation action") + + +def _commit_agent_operation( + store: ChatActionStore, + runtime_root: Path, + registry_path: Path, + parameters: Mapping[str, Any], + proposal: dict, + proposal_id: str, + actor: Mapping[str, Any], + action: str, + consumption_id: str | None, + outcome: Mapping[str, Any] | None, +) -> dict[str, Any]: + current = _binding(registry_path, parameters, runtime_root) + actor_executor = ( + { + "kind": "managed_turn", + "session_id": actor.get("thread_id"), + "todo_id": actor.get("todo_id"), + "profile_digest": actor.get("profile_digest"), + "revision": "managed-turn-handoff-v0", + "model": actor.get("model"), + "reasoning_effort": actor.get("reasoning_effort"), + } + if actor.get("host_surface") == "loopx-managed-codex" + else actor + ) + actor_current = ( + _binding( + registry_path, {**parameters, "executor": actor_executor}, runtime_root + ) + if action in {"inspect", "report"} + else False + ) + if action == "inspect": + plan = store._agent_operation_plan( + proposal, + action="inspect", + actor=dict(actor), + binding_current=current, + actor_binding_current=actor_current, + ) + return { + **plan, + "binding_current": current, + "parameters": parameters, + "confirmation": proposal["operation"].get("confirmation"), + "consumption": proposal["operation"].get("agent_handoff"), + "outcome": proposal["operation"].get("outcome"), + "reconciliation": proposal["operation"].get("reconciliation"), + "outcome_report": proposal["operation"].get("outcome_report"), + "reconciliation_report": proposal["operation"].get("reconciliation_report"), + } + if action == "consume": + return store.consume_agent_operation( + proposal_id, + actor=actor, + binding_current=current, + consumption_id=str(consumption_id or ""), + ) + if action == "report": + updated = store.observe_operation_outcome( + proposal_id, + outcome=outcome or {}, + agent_actor=actor, + agent_binding_current=current, + agent_actor_binding_current=actor_current, + ) + plan = store._agent_operation_plan(updated, action="project") + return { + "ok": True, + **plan, + "outcome": updated["operation"].get("reconciliation") + or updated["operation"]["outcome"], + "outcome_report": updated["operation"].get("outcome_report"), + "reconciliation_report": updated["operation"].get("reconciliation_report"), + } + raise ValueError("unsupported agent operation action") diff --git a/loopx/control_plane/effect_runtime_handlers.ts b/loopx/control_plane/effect_runtime_handlers.ts index 69830c25a5..cf04abcb75 100644 --- a/loopx/control_plane/effect_runtime_handlers.ts +++ b/loopx/control_plane/effect_runtime_handlers.ts @@ -1,5 +1,5 @@ import {manageNewGoalStorage} from "./coordination/local_authority_defaults.ts"; -import {deriveAgentOperationActor, normalizeAgentOperationExecutor, planAgentOperationHandoff, projectAgentOperationInbox} from "./work_items/operation_agent_handoff.ts"; +import {deriveAgentOperationActor, managedOperationBindingCurrent, normalizeAgentOperationExecutor, planAgentOperationHandoff, projectAgentOperationInbox, projectManagedOperationTransport} from "./work_items/operation_agent_handoff.ts"; import {projectDecisionNotice} from "./presentation/decision_notice.ts"; import {normalizeResearchObservation, validateResearchAttribution, projectResearchFrontier} from "./capabilities/explore_research.ts"; import {projectTodoSummary} from "./todos/summary_projection.ts"; @@ -632,6 +632,8 @@ export function createEffectRuntimeHandlers( ["presentation.action_review_plan.compile", (params) => compileActionReviewPlan(params.proposal)], ["operation.agent_executor.normalize", normalizeAgentOperationExecutor], + ["operation.managed_binding.current", managedOperationBindingCurrent], + ["operation.managed_transport.project", projectManagedOperationTransport], ["operation.agent_handoff.actor", deriveAgentOperationActor], ["operation.agent_handoff.plan", planAgentOperationHandoff], ["operation.agent_handoff.inbox", projectAgentOperationInbox], diff --git a/loopx/control_plane/turn_driver/codex_cli.py b/loopx/control_plane/turn_driver/codex_cli.py index 6070e4b51f..cdd0610ff9 100644 --- a/loopx/control_plane/turn_driver/codex_cli.py +++ b/loopx/control_plane/turn_driver/codex_cli.py @@ -13,6 +13,7 @@ from typing import Any from ...runtime import validate_goal_id_path_segment +from ...file_lock import exclusive_file_lock from ..goals.first_party_host_admission import FirstPartyHostGoalAdmission from .subagent_execution_topology import ( child_execution_receipts_json_schema, @@ -305,6 +306,31 @@ def _store_codex_cli_session( lineage: Mapping[str, str], session_id: str, goal_ref: Mapping[str, Any] | None = None, + operation_profile_digest: str | None = None, + operation_model: str | None = None, + operation_reasoning_effort: str | None = None, +) -> None: + with exclusive_file_lock(_session_path(runtime_root, lineage)): + _write_codex_cli_session( + runtime_root, + lineage=lineage, + session_id=session_id, + goal_ref=goal_ref, + operation_profile_digest=operation_profile_digest, + operation_model=operation_model, + operation_reasoning_effort=operation_reasoning_effort, + ) + + +def _write_codex_cli_session( + runtime_root: Path, + *, + lineage: Mapping[str, str], + session_id: str, + goal_ref: Mapping[str, Any] | None = None, + operation_profile_digest: str | None = None, + operation_model: str | None = None, + operation_reasoning_effort: str | None = None, ) -> None: normalized_session_id = _valid_session_id(session_id) if not normalized_session_id: @@ -329,6 +355,11 @@ def _store_codex_cli_session( } if goal_ref is not None: payload["goal_ref"] = dict(goal_ref) + if operation_profile_digest is not None: + payload["operation_transport"] = "app-server-operation-tools-v0" + payload["operation_profile_digest"] = operation_profile_digest + payload["operation_model"] = operation_model + payload["operation_reasoning_effort"] = operation_reasoning_effort json.dump( payload, handle, @@ -350,7 +381,9 @@ def _discard_codex_cli_session( *, lineage: Mapping[str, str], ) -> None: - _session_path(runtime_root, lineage).unlink(missing_ok=True) + path = _session_path(runtime_root, lineage) + with exclusive_file_lock(path): + path.unlink(missing_ok=True) def _has_subagent_topology(request: Mapping[str, Any] | None) -> bool: @@ -864,6 +897,10 @@ def run_codex_cli_host( if planned_action not in {"resume", "start_new"}: raise ValueError("Codex CLI host request has no executable session action") session_id = str(binding.get("session_id")) if binding else None + if binding and binding.get("operation_transport"): + raise ValueError( + "operation-equipped session requires its original managed transport; select a fresh iteration explicitly to change it" + ) goal_ref = request.get("goal_ref") exact_goal_ref = dict(goal_ref) if isinstance(goal_ref, Mapping) else None diff --git a/loopx/control_plane/turn_driver/codex_operation_host.py b/loopx/control_plane/turn_driver/codex_operation_host.py new file mode 100644 index 0000000000..38e15ee544 --- /dev/null +++ b/loopx/control_plane/turn_driver/codex_operation_host.py @@ -0,0 +1,349 @@ +"""Owned app-server transport for an opt-in, already-admitted Codex Turn. + +No new approval, scheduler or execution store. Python owns the subprocess and +native tool IO; TS owns executor binding, consumption and outcome semantics. +An attached Desktop conversation is never resumed or impersonated here. +""" + +from __future__ import annotations + +import hashlib +import json +import os +import shutil +from collections.abc import Mapping +from pathlib import Path +from typing import Any + +from ...chat_agent import CodexChatAgentError, CodexChatAgentSession +from ...chat_action_store import ChatActionStore +from ...chat_actions import ChatActionService +from ..collaboration.operation_handoff import ( + agent_operation_action, + pending_operation_handoffs, +) +from ..goals.first_party_host_admission import FirstPartyHostGoalAdmission +from .codex_cli import ( + _lineage, + _prompt, + _read_codex_cli_session_document, + _codex_session_goal_ref, + _store_codex_cli_session, + load_codex_cli_session, + normalize_codex_stdio_mcp_server, + codex_cli_result_schema, +) +from .executor import LOOPX_TURN_HOST_REQUEST_SCHEMA_VERSION +from .host_failure import BuiltInHostError + +TRANSPORT = "app-server-operation-tools-v0" +REVISION = "managed-turn-handoff-v0" +OPERATION_TOOL = { + "name": "loopx_operation", + "description": "Read/prepare a scoped operation, consume exact human approval once, or report original evidence. Identity comes from this native connection, never arguments. A receipt is not permission to retry an effect.", + "inputSchema": { + "type": "object", + "properties": { + "action": { + "type": "string", + "enum": [ + "context", + "pending", + "prepare", + "inspect", + "consume", + "report", + ], + }, + "proposal_id": {"type": "string"}, + "consumption_id": {"type": "string"}, + "request": {"type": "object"}, + "outcome": {"type": "object"}, + "cursor": {"type": "string"}, + }, + "required": ["action"], + "additionalProperties": False, + }, +} + + +def operation_tool_handler( + *, + runtime_root: Path, + registry_path: Path, + lineage: Mapping[str, str], + session_id: str, + profile_digest: str, + model: str, + reasoning_effort: str, + goal_admission: FirstPartyHostGoalAdmission | None = None, +): + """Private closure installed only on a process owned by the Turn driver. + + The caller must already have checked native thread/Turn metadata. This is + not an MCP endpoint or a public actor/proof deserializer. + """ + executor = { + "kind": "managed_turn", + "todo_id": lineage["todo_id"], + "session_id": session_id, + "profile_digest": profile_digest, + "revision": REVISION, + "model": model, + "reasoning_effort": reasoning_effort, + } + + def handle(tool: str, arguments: Any, native: dict[str, Any]) -> dict[str, Any]: + if tool != "loopx_operation" or not isinstance(arguments, dict): + return {"ok": False, "error": "unsupported_operation_tool"} + if native.get("thread_id") != session_id or not native.get("host_turn_id"): + return {"ok": False, "error": "tool_turn_mismatch"} + allowed = { + "context": {"action"}, + "pending": {"action"}, + "prepare": {"action", "request"}, + "inspect": {"action", "proposal_id"}, + "consume": {"action", "proposal_id", "consumption_id"}, + "report": {"action", "proposal_id", "outcome"}, + } + action = arguments.get("action") + keys = set(arguments) - ({"cursor"} if action == "pending" else set()) + if ( + not isinstance(action, str) + or action not in allowed + or keys != allowed[action] + ): + return {"ok": False, "error": "invalid_operation_arguments"} + try: + if goal_admission is not None: + goal_admission.require_current() + if action == "context": + return { + "ok": True, + **lineage, + "executor": executor, + "authority": "approval_and_first_consumption_required", + "external_write_performed": False, + } + if action == "pending": + return { + "ok": True, + **pending_operation_handoffs( + runtime_root, + lineage["goal_id"], + lineage["agent_id"], + registry_path=registry_path, + cursor=arguments.get("cursor"), + cursor_scope="managed-operation:" + session_id, + ), + } + if action == "prepare": + request = dict(arguments["request"]) + terms = dict(request.get("normalized_parameters") or {}) + for key, expected in { + "goal_id": lineage["goal_id"], + "agent_id": lineage["agent_id"], + "executor": executor, + }.items(): + if key in terms and terms[key] != expected: + raise ValueError( + "operation request cannot select another execution subject" + ) + terms[key] = expected + if request.get("action_kind") != "operation.execute": + raise ValueError("operation tool prepares only typed operations") + request["normalized_parameters"] = terms + service = ChatActionService( + store=ChatActionStore(runtime_root / "chat" / "actions"), + registry_path=registry_path, + ) + return { + "ok": True, + "proposal": service.preview(request), + "execution_allowed": False, + } + actor = { + **lineage, + "host_surface": "loopx-managed-codex", + "thread_id": session_id, + "profile_digest": profile_digest, + "model": model, + "reasoning_effort": reasoning_effort, + "host_turn_id": native["host_turn_id"], + } + return { + "ok": True, + **agent_operation_action( + runtime_root, + registry_path, + proposal_id=arguments["proposal_id"], + actor=actor, + action=action, + consumption_id=arguments.get("consumption_id"), + outcome=arguments.get("outcome"), + ), + } + except (ValueError, KeyError, TypeError, RuntimeError): + # Private payloads, paths and adapter error text never enter the model tool error. + return { + "ok": False, + "error": "operation_admission_rejected", + "execution_allowed": False, + } + + return handle + + +def run_codex_operation_host( + request: Mapping[str, Any], + *, + runtime_root: Path, + registry_path: Path, + project: Path, + codex_bin: str = "codex", + sandbox: str = "read-only", + model: str | None = None, + reasoning_effort: str | None = None, + mcp_server: Mapping[str, Any] | None = None, + timeout_seconds: float = 115, + goal_admission: FirstPartyHostGoalAdmission | None = None, +) -> dict[str, Any]: + if request.get("schema_version") != LOOPX_TURN_HOST_REQUEST_SCHEMA_VERSION: + raise ValueError("unsupported LoopX Turn host request schema") + if sandbox not in {"read-only", "workspace-write"}: + raise ValueError("operation tools require a restricted Codex execution sandbox") + if not model or not reasoning_effort: + raise ValueError( + "managed operation tools require an explicit model and reasoning effort" + ) + lineage = _lineage(request) + if not all(lineage.values()): + raise ValueError("managed operation tools require a Todo-bound Turn") + mcp_server = normalize_codex_stdio_mcp_server(mcp_server) + resolved_bin = shutil.which(codex_bin) + if not resolved_bin: + raise ValueError("Codex executable is unavailable") + with Path(resolved_bin).open("rb") as stream: + binary_digest = hashlib.file_digest(stream, "sha256").hexdigest() + profile = { + "transport": TRANSPORT, + "model": model, + "reasoning_effort": reasoning_effort, + "sandbox": sandbox, + "workspace": str(project.resolve()), + "codex_home": str( + Path(os.environ.get("CODEX_HOME") or "~/.codex").expanduser().resolve() + ), + "codex_binary_digest": binary_digest, + "mcp_server": mcp_server, + } + profile_digest = hashlib.sha256( + json.dumps(profile, sort_keys=True, separators=(",", ":")).encode() + ).hexdigest() + binding = load_codex_cli_session(runtime_root, lineage=lineage) + if goal_admission is not None: + selected = goal_admission.select_state( + read_state=lambda: _read_codex_cli_session_document( + runtime_root, lineage=lineage + ), + goal_ref_of=lambda value: _codex_session_goal_ref(value, lineage=lineage), + ) + binding = dict(selected) if selected is not None else None + session_plan = request.get("session") or {} + if (session_plan.get("context_policy") or {}).get("mode") == "fresh": + binding = None + action = session_plan.get("action") + if ( + action not in {"start_new", "resume"} + or (action == "resume" and not binding) + or (action == "start_new" and binding) + ): + raise ValueError("managed Codex session changed after Turn planning") + if binding and ( + binding.get("operation_transport") != TRANSPORT + or binding.get("operation_profile_digest") != profile_digest + ): + raise ValueError( + "managed operation profile changed; explicitly select a fresh iteration and obtain fresh approval" + ) + host_config = ( + { + "mcp_servers": { + mcp_server["name"]: { + "command": mcp_server["command"][0], + "args": mcp_server["command"][1:], + "enabled": True, + "required": True, + "default_tools_approval_mode": "approve", + "startup_timeout_sec": 30, + "tool_timeout_sec": 60, + } + } + } + if mcp_server + else None + ) + session = None + try: + session = CodexChatAgentSession.start( + codex_bin=codex_bin, + work_dir=project, + goal_id=lineage["goal_id"], + objective="One admitted LoopX Turn", + execution_mode=True, + isolate_process_tree=True, + sandbox=sandbox, + model=model, + reasoning_effort=reasoning_effort, + resume_thread_id=binding["session_id"] if binding else None, + dynamic_tools=[OPERATION_TOOL], + host_config=host_config, + response_timeout_sec=min(timeout_seconds, 30), + hard_timeout_sec=timeout_seconds, + idle_timeout_sec=min(timeout_seconds, 180), + ) + + def store_binding(): + _store_codex_cli_session( + runtime_root, + lineage=lineage, + session_id=session.thread_id, + goal_ref=request.get("goal_ref"), + operation_profile_digest=profile_digest, + operation_model=model, + operation_reasoning_effort=reasoning_effort, + ) + + if goal_admission is None: + store_binding() + else: + goal_admission.accept_result(store_binding) + session.bound_tool_handler = operation_tool_handler( + runtime_root=runtime_root, + registry_path=registry_path, + lineage=lineage, + session_id=session.thread_id, + profile_digest=profile_digest, + model=model, + reasoning_effort=reasoning_effort, + goal_admission=goal_admission, + ) + return session.send( + _prompt(request) + + "\nUse loopx_operation for pending/prepare/inspect/consume/report. " + "Source conversations are not executor identity. Execute only after the first receipt says execution_allowed=true. " + "Already consumed/unknown effects require evidence reconciliation, never retry. Never treat final-answer prose as an outcome receipt.", + output_schema=codex_cli_result_schema(request), + ) + except CodexChatAgentError as exc: + raise BuiltInHostError( + "codex_operation_host_" + exc.error_code, + failure_kind="executor_timeout" + if "timeout" in exc.error_code + else "unknown", + recovery_kind="resume_session" if session else None, + ) from exc + finally: + if session is not None: + session.bound_tool_handler = None + session.close() diff --git a/loopx/control_plane/turn_driver/host_binding.py b/loopx/control_plane/turn_driver/host_binding.py index 43ac3ce965..6a0be83f0a 100644 --- a/loopx/control_plane/turn_driver/host_binding.py +++ b/loopx/control_plane/turn_driver/host_binding.py @@ -177,6 +177,31 @@ def dsh_output_token_budget(max_tokens: int | None = None) -> dict[str, Any]: } +def _with_operation_transport( + binding: dict[str, Any], + *, + enabled: bool, + host: str, + sandbox: str, + model: str | None, + reasoning_effort: str | None, +) -> dict[str, Any]: + if enabled: + from ..effect_runtime import effect_runtime_result + + observed = effect_runtime_result( + "operation.managed_transport.project", + {"host": host, "sandbox": sandbox, + "model": model, "reasoning_effort": reasoning_effort}, + ) + binding["operation_transport"] = observed["transport"] + if observed["reason"]: + binding["available"] = False + binding["unavailable_reason"] = observed["reason"] + binding["unavailable_remediation"] = [REMEDY_CORRECT_EXECUTION_PROFILE] + return binding + + def managed_executor_binding( host: str, *, @@ -187,6 +212,8 @@ def managed_executor_binding( model: str | None = None, reasoning_effort: str | None = None, max_tokens: int | None = None, + codex_operation_tools: bool = False, + codex_sandbox: str = "read-only", ) -> dict[str, Any]: """Project the executor one planned Turn would run on. @@ -228,7 +255,7 @@ def managed_executor_binding( unavailable_reason = INVALID_OUTPUT_TOKEN_LIMIT else: unavailable_reason = profile_reason - return { + binding = { "schema_version": MANAGED_EXECUTOR_BINDING_SCHEMA_VERSION, "executor": host, "executor_kind": EXECUTOR_KIND_MANAGED, @@ -252,12 +279,16 @@ def managed_executor_binding( "available": runtime_available, }, } + return _with_operation_transport( + binding, enabled=codex_operation_tools, host=host, + sandbox=codex_sandbox, model=model, reasoning_effort=reasoning_effort, + ) individual_profile: str | None = None if host == INDIVIDUAL_TURN_HOST and (model or reasoning_effort): individual_profile = ( f"{model or 'host-default'}@{reasoning_effort or 'host-default'}" ) - return { + binding = { "schema_version": MANAGED_EXECUTOR_BINDING_SCHEMA_VERSION, "executor": host, "executor_kind": ( @@ -276,6 +307,10 @@ def managed_executor_binding( # branches on its presence; only a managed executor probes a runtime. "runtime_probe": None, } + return _with_operation_transport( + binding, enabled=codex_operation_tools, host=host, + sandbox=codex_sandbox, model=model, reasoning_effort=reasoning_effort, + ) def turn_host_arg_option(host_args: Sequence[str], name: str) -> str | None: @@ -338,6 +373,8 @@ def managed_executor_binding_from_host_args( if host == INDIVIDUAL_TURN_HOST else None ), + codex_operation_tools="--codex-operation-tools" in host_args, + codex_sandbox=turn_host_arg_option(host_args, "--codex-sandbox") or "read-only", ) diff --git a/loopx/control_plane/work_items/operation_agent_handoff.ts b/loopx/control_plane/work_items/operation_agent_handoff.ts index 9fd1dfafec..287a6ec12e 100644 --- a/loopx/control_plane/work_items/operation_agent_handoff.ts +++ b/loopx/control_plane/work_items/operation_agent_handoff.ts @@ -6,7 +6,9 @@ import {EffectRuntimeConflictError, EffectRuntimeRequestError} from "../effect_r import {requireJsonObject, requireNonEmptyString} from "../runtime_decode.ts"; export const AGENT_OPERATION_REVISION = "agent-session-handoff-v0"; +export const MANAGED_OPERATION_REVISION = "managed-turn-handoff-v0"; const ID = /^[A-Za-z0-9._:-]{1,200}$/; +const SHA256 = /^[a-f0-9]{64}$/; function id(value: unknown, field: string): string { const result = requireNonEmptyString(value, field); @@ -25,8 +27,35 @@ function timestamp(value: unknown): number { return parsed; } +/** Readback of an operator-selected transport. This is not a session binding, + * a runtime qualification or an execution permit. Python supplies argv facts. */ +export function projectManagedOperationTransport(input: JsonObject): JsonObject { + const reason = input.host !== "codex-cli" ? "operation_transport_host_unsupported" + : !["read-only", "workspace-write"].includes(String(input.sandbox)) ? "operation_transport_sandbox_unsupported" + : typeof input.model !== "string" || !ID.test(input.model) + || !["minimal", "low", "medium", "high", "xhigh"].includes(String(input.reasoning_effort)) + ? "operation_transport_profile_required" : null; + return {reason, transport: {schema_version: "loopx_operation_transport_v0", + kind: "owned_app_server", revision: "app-server-operation-tools-v0", + configuration_valid: reason === null, runtime_qualified: false, + identity_source: "native_thread_turn_metadata", human_confirmation_required: true, + first_consumption_required: true, source_conversation_is_executor: false}}; +} + export function normalizeAgentOperationExecutor(input: JsonObject): JsonObject { const executor = requireJsonObject(input.executor, "agent executor"); + if (executor.kind === "managed_turn") { + const keys = ["kind", "todo_id", "session_id", "profile_digest", "model", "reasoning_effort", "revision"]; + if (Object.keys(executor).length !== keys.length || keys.some(key => !(key in executor)) + || executor.revision !== MANAGED_OPERATION_REVISION + || typeof executor.profile_digest !== "string" || !SHA256.test(executor.profile_digest)) { + throw new EffectRuntimeRequestError("managed operation executor binding is invalid"); + } + return {kind: "managed_turn", todo_id: id(executor.todo_id, "todo_id"), + session_id: id(executor.session_id, "session_id"), profile_digest: executor.profile_digest, + model: id(executor.model, "model"), reasoning_effort: id(executor.reasoning_effort, "reasoning_effort"), + revision: MANAGED_OPERATION_REVISION}; + } const keys = ["kind", "host_surface", "thread_id", "revision"]; if (Object.keys(executor).length !== keys.length || keys.some(key => !(key in executor)) || executor.kind !== "agent_session" || executor.revision !== AGENT_OPERATION_REVISION) { @@ -36,6 +65,30 @@ export function normalizeAgentOperationExecutor(input: JsonObject): JsonObject { thread_id: id(executor.thread_id, "thread_id"), revision: AGENT_OPERATION_REVISION}; } +/** Filesystem/session observations are supplied by the existing Turn session + * owner. A registered source conversation is context/return routing, not the + * identity of a managed executor. No transport proof is accepted by this RPC. */ +export function managedOperationBindingCurrent(input: JsonObject): JsonObject { + const parameters = requireJsonObject(input.parameters, "operation parameters"); + const executor = normalizeAgentOperationExecutor({executor: parameters.executor}); + const session = input.session == null ? null : requireJsonObject(input.session, "Turn session"); + const expectedRef = parameters.origin_goal_ref == null ? null + : requireJsonObject(parameters.origin_goal_ref, "origin Goal ref"); + const observedRef = session?.goal_ref == null ? null : requireJsonObject(session.goal_ref, "session Goal ref"); + const sameGoalRef = expectedRef === null + ? observedRef === null || (observedRef.goal_id === parameters.goal_id && observedRef.goal_instance_id == null) + : observedRef !== null && expectedRef.goal_id === observedRef.goal_id + && expectedRef.goal_instance_id === observedRef.goal_instance_id; + return {current: executor.kind === "managed_turn" && session !== null + && session.schema_version === "loopx_codex_cli_session_v1" + && session.goal_id === parameters.goal_id && session.agent_id === parameters.agent_id + && session.todo_id === executor.todo_id && session.session_id === executor.session_id + && session.operation_transport === "app-server-operation-tools-v0" + && session.operation_profile_digest === executor.profile_digest + && session.operation_model === executor.model && session.operation_reasoning_effort === executor.reasoning_effort + && sameGoalRef}; +} + /** No qualified host producer is connected to the public CLI. Ambient thread * ids, route flags and caller-supplied "verified" fields cannot authenticate * a session. Keep the old RPC fail-closed until a real transport-owned issuer @@ -60,7 +113,8 @@ function historicalAccess(input: JsonObject, route: JsonObject, consumed: boolea const owner = Object.fromEntries(Object.keys(route).map(key => [key, id(actor[key], `actor.${key}`)])); return {mode: original ? "original_session" : "replacement_reconciliation", owner, original_route: route, permission: "historical_evidence_only", execution_allowed: false, - authority_source: original ? "original_operation_route" : "current_registry_binding"}; + authority_source: original ? "original_operation_route" + : actor.host_surface === "loopx-managed-codex" ? "current_turn_session_binding" : "current_registry_binding"}; } export function planAgentOperationHandoff(input: JsonObject): JsonObject { @@ -85,14 +139,18 @@ export function planAgentOperationHandoff(input: JsonObject): JsonObject { const claim = operation.claim == null ? null : requireJsonObject(operation.claim, "operation claim"); const now = timestamp(input.now); const expires = timestamp(operation.expires_at); - const route = {goal_id: parameters.goal_id, agent_id: parameters.agent_id, - host_surface: executor.host_surface, thread_id: executor.thread_id}; + const managed = executor.kind === "managed_turn"; + const route: JsonObject = {goal_id: parameters.goal_id, agent_id: parameters.agent_id, + ...(managed ? {host_surface: "loopx-managed-codex", thread_id: executor.session_id, + todo_id: executor.todo_id, profile_digest: executor.profile_digest} + : {host_surface: executor.host_surface, thread_id: executor.thread_id})}; const base: JsonObject = {schema_version: "loopx_operation_agent_handoff_v0", operation_id: operation.operation_id, payload_digest: operation.payload_digest, confirmation_digest: operation.confirmation_digest, claim_id: claim?.claim_id ?? null, executor_revision: executor.revision, expires_at: operation.expires_at, route, authorization_source: "canonical_typed_operation", execution_allowed: false, host_delivery: "not_attempted", external_write_performed: false, - host_authentication_required: true}; + executor_kind: executor.kind, source_route: parameters.source_route ?? null, + host_authentication_required: !managed}; const handoff = operation.agent_handoff == null ? null : requireJsonObject(operation.agent_handoff, "agent handoff"); const observed = operation.reconciliation ?? operation.outcome; @@ -115,6 +173,8 @@ export function planAgentOperationHandoff(input: JsonObject): JsonObject { requireThat(Object.entries(route).every(([key, value]) => actor[key] === value), "execution actor is not the original bound session"); requireThat(input.binding_current === true, "original session binding is no longer current"); + if (managed) requireThat(typeof actor.host_turn_id === "string" && ID.test(actor.host_turn_id), + "managed operation requires its transport-owned active Turn"); // Even a same-id retry returns no execute permission. A lost response after // this commit is ambiguous, never permission to submit a second order. if (handoff || operation.lifecycle_state === "outcome_observed") { @@ -124,6 +184,7 @@ export function planAgentOperationHandoff(input: JsonObject): JsonObject { requireThat(now < expires, "confirmed operation expired before execution consumption"); return {...base, status: "consumed_outcome_pending", execution_allowed: true, write_handoff: {...base, status: "consumed_outcome_pending", consumed_at: input.now, + ...(managed ? {host_turn_id: actor.host_turn_id} : {}), consumption_id: id(input.consumption_id, "consumption_id")}}; } if (action === "report") { diff --git a/loopx/extensions/lark/goal_channel_operation.py b/loopx/extensions/lark/goal_channel_operation.py index e89fb5db73..704e2ed4bb 100644 --- a/loopx/extensions/lark/goal_channel_operation.py +++ b/loopx/extensions/lark/goal_channel_operation.py @@ -113,7 +113,10 @@ def _operation_review_frame(proposal: Mapping[str, Any]) -> dict[str, Any]: def _result_delivery_stage(proposal: Mapping[str, Any]) -> dict[str, str]: parameters, operation = _proposal_operation(proposal) - if (parameters.get("executor") or {}).get("kind") != "agent_session": + if (parameters.get("executor") or {}).get("kind") not in { + "agent_session", + "managed_turn", + }: return {} return { "outcome_stage": "reconciled" @@ -377,6 +380,8 @@ def build_goal_channel_operation_result_card( ) if pending and frame.get("executionState") == "consumed_outcome_pending": result_label = "执行授权已消费,等待真实结果" + elif pending and frame.get("executionState") == "managed_turn_pending": + result_label = "已确认,等待绑定的受管回合;尚未执行" summary = str(frame.get("summary") or result_label) return { "schema": "2.0", @@ -673,7 +678,10 @@ def _resolve_operation_executor_binding( parameters: Mapping[str, Any], *, runtime_root: Path ) -> dict[str, Any]: executor = parameters.get("executor") - if isinstance(executor, Mapping) and executor.get("kind") == "agent_session": + if isinstance(executor, Mapping) and executor.get("kind") in { + "agent_session", + "managed_turn", + }: return dict( effect_runtime_result( "operation.agent_executor.normalize", {"executor": dict(executor)} @@ -1056,10 +1064,9 @@ def handle_goal_channel_operation_callback( if current is None: raise ValueError("claimed operation disappeared before dispatch") _parameters, current_operation = _proposal_operation(current) - if ( - current_operation.get("lifecycle_state") == "claimed" - and (_parameters.get("executor") or {}).get("kind") == "agent_session" - ): + if current_operation.get("lifecycle_state") == "claimed" and ( + _parameters.get("executor") or {} + ).get("kind") in {"agent_session", "managed_turn"}: # Confirmation exposes an exact canonical continuation in the # existing Inbox. No simulator, host resume or financial effect is # run in the callback process, and no outcome is manufactured. diff --git a/tests/control_plane_ts/operation_agent_handoff.test.ts b/tests/control_plane_ts/operation_agent_handoff.test.ts index 1688e9a0f2..f904f4f180 100644 --- a/tests/control_plane_ts/operation_agent_handoff.test.ts +++ b/tests/control_plane_ts/operation_agent_handoff.test.ts @@ -1,8 +1,8 @@ import assert from "node:assert/strict"; import test from "node:test"; import type {JsonObject} from "../../loopx/control_plane/effect_program.ts"; -import {AGENT_OPERATION_REVISION, deriveAgentOperationActor, normalizeAgentOperationExecutor, planAgentOperationHandoff, - projectAgentOperationInbox} from "../../loopx/control_plane/work_items/operation_agent_handoff.ts"; +import {AGENT_OPERATION_REVISION, MANAGED_OPERATION_REVISION, managedOperationBindingCurrent, deriveAgentOperationActor, normalizeAgentOperationExecutor, planAgentOperationHandoff, + projectAgentOperationInbox, projectManagedOperationTransport} from "../../loopx/control_plane/work_items/operation_agent_handoff.ts"; function input(): JsonObject { const executor = {kind: "agent_session", host_surface: "codex-app", thread_id: "original-thread", @@ -22,6 +22,64 @@ function input(): JsonObject { } const operation = (value: JsonObject) => (value.proposal as JsonObject).operation as JsonObject; +test("operator transport preflight projects pinned configuration without claiming authentication or runtime qualification", () => { + const input = {host: "codex-cli", sandbox: "read-only", model: "test-model", reasoning_effort: "xhigh"}; + const projected = projectManagedOperationTransport(input); + assert.equal(projected.reason, null); + assert.equal((projected.transport as JsonObject).configuration_valid, true); + assert.equal((projected.transport as JsonObject).runtime_qualified, false); + assert.equal((projected.transport as JsonObject).source_conversation_is_executor, false); + for (const invalid of [{host: "dsh"}, {sandbox: "danger-full-access"}, {model: null}, {reasoning_effort: "unknown"}]) { + const refused = projectManagedOperationTransport({...input, ...invalid}); + assert.equal((refused.transport as JsonObject).configuration_valid, false); + assert.notEqual(refused.reason, null); + } +}); + +function managedInput(): JsonObject { + const value = input(); + const parameters = (value.proposal as JsonObject).normalized_parameters as JsonObject; + parameters.executor = {kind: "managed_turn", todo_id: "todo-worker", session_id: "owned-thread", + profile_digest: "a".repeat(64), model: "test-model", reasoning_effort: "xhigh", revision: MANAGED_OPERATION_REVISION}; + parameters.source_route = {...value.actor as JsonObject}; + operation(value).executor_revision = MANAGED_OPERATION_REVISION; + value.actor = {goal_id: "test-goal", agent_id: "test-agent", host_surface: "loopx-managed-codex", + thread_id: "owned-thread", todo_id: "todo-worker", profile_digest: "a".repeat(64), host_turn_id: "native-turn"}; + return value; +} + +test("source context is not managed execution identity and old approvals never migrate", () => { + const value = managedInput(); + const plan = planAgentOperationHandoff(value); + assert.equal(plan.execution_allowed, true); + assert.equal(plan.host_authentication_required, false); + assert.equal((plan.source_route as JsonObject).host_surface, "codex-app"); + assert.equal((plan.route as JsonObject).host_surface, "loopx-managed-codex"); + assert.equal((plan.write_handoff as JsonObject).host_turn_id, "native-turn"); + for (const change of [ + {thread_id: "other-thread"}, {profile_digest: "b".repeat(64)}, {todo_id: "todo-other"}, + {host_surface: "codex-app"}, {host_turn_id: null}, + ]) assert.throws(() => planAgentOperationHandoff({...value, actor: {...value.actor as JsonObject, ...change}})); + assert.throws(() => planAgentOperationHandoff({...input(), actor: value.actor})); + operation(value).agent_handoff = plan.write_handoff; + assert.equal(planAgentOperationHandoff(value).execution_allowed, false); +}); + +test("managed binding is the existing exact Goal/Todo/session/profile owner, never a source route", () => { + const parameters = (managedInput().proposal as JsonObject).normalized_parameters as JsonObject; + parameters.origin_goal_ref = {goal_id: "test-goal", goal_instance_id: "instance-1"}; + const session = {schema_version: "loopx_codex_cli_session_v1", goal_id: "test-goal", agent_id: "test-agent", + todo_id: "todo-worker", session_id: "owned-thread", operation_profile_digest: "a".repeat(64), + operation_transport: "app-server-operation-tools-v0", + operation_model: "test-model", operation_reasoning_effort: "xhigh", + goal_ref: {goal_instance_id: "instance-1", goal_id: "test-goal"}}; + assert.equal(managedOperationBindingCurrent({parameters, session}).current, true); + for (const changed of [{session_id: "replacement"}, {todo_id: "other"}, {agent_id: "other"}, + {operation_profile_digest: "b".repeat(64)}, {operation_transport: null}, {goal_ref: {goal_id: "test-goal", goal_instance_id: "instance-2"}}, + {schema_version: "future"}]) assert.equal(managedOperationBindingCurrent({parameters, session: {...session, ...changed}}).current, false); + assert.equal(managedOperationBindingCurrent({parameters, session: null}).current, false); +}); + test("public caller identity stays blocked, including exact same-user environment forgery", () => { const requested = input().actor as JsonObject; for (const thread_id of [null, "", "another-thread", "original-thread"]) { diff --git a/tests/extensions/test_lark_goal_channel_operation.py b/tests/extensions/test_lark_goal_channel_operation.py index 9433999552..69e1300714 100644 --- a/tests/extensions/test_lark_goal_channel_operation.py +++ b/tests/extensions/test_lark_goal_channel_operation.py @@ -53,7 +53,9 @@ TENANT_KEY = "tenant_operation_fixture" -def _prepare_agent_handoff(store: ChatActionStore, registry: Path) -> dict[str, Any]: +def _prepare_agent_handoff( + store: ChatActionStore, registry: Path, *, managed: bool = False +) -> dict[str, Any]: baseline = _prepare(store, registry) data = json.loads(registry.read_text()) data["goals"][0]["coordination"]["thread_agent_bindings"] = [ @@ -73,6 +75,30 @@ def _prepare_agent_handoff(store: ChatActionStore, registry: Path) -> dict[str, "revision": "agent-session-handoff-v0", } parameters["operation_kind"] = "fixture.submit" + if managed: + from loopx.control_plane.turn_driver.codex_cli import _store_codex_cli_session + + _store_codex_cli_session( + store.root.parent.parent, + lineage={ + "goal_id": GOAL_ID, + "agent_id": AGENT_ID, + "todo_id": "todo-managed", + }, + session_id="owned-managed-thread", + operation_profile_digest="c" * 64, + operation_model="test-model", + operation_reasoning_effort="xhigh", + ) + parameters["executor"] = { + "kind": "managed_turn", + "todo_id": "todo-managed", + "session_id": "owned-managed-thread", + "profile_digest": "c" * 64, + "model": "test-model", + "reasoning_effort": "xhigh", + "revision": "managed-turn-handoff-v0", + } parameters["projection"] = { **parameters["projection"], "simulated": False, @@ -91,13 +117,15 @@ def _prepare_agent_handoff(store: ChatActionStore, registry: Path) -> dict[str, ) +@pytest.mark.parametrize("managed", [False, True]) def test_authenticated_callback_hands_off_without_calling_any_executor_and_reconciles_original_result( tmp_path: Path, monkeypatch: pytest.MonkeyPatch, + managed: bool, ) -> None: monkeypatch.setenv("CODEX_THREAD_ID", "thread-operation-fixture") store, registry, runtime, binding, target = _fixture(tmp_path) - proposal = _prepare_agent_handoff(store, registry) + proposal = _prepare_agent_handoff(store, registry, managed=managed) cards: dict[str, dict[str, Any]] = {} runner = _runner([], cards) deliver_goal_channel_operation_card( @@ -138,7 +166,9 @@ def no_executor(_proposal): assert first["callback_ack_is_execution_receipt"] is False claimed = store.load(proposal["proposal_id"]) assert claimed["operation"]["result_delivery"] is None - assert "原宿主身份认证尚未接通" in normalized_card_text(next(iter(cards.values()))) + assert ( + "等待绑定的受管回合" if managed else "原宿主身份认证尚未接通" + ) in normalized_card_text(next(iter(cards.values()))) actor = { "goal_id": GOAL_ID, "agent_id": AGENT_ID, @@ -146,6 +176,16 @@ def no_executor(_proposal): "thread_id": "thread-operation-fixture", } context = GoalChannelOperationContext(runtime, registry, runtime, binding) + if managed: + actor.update( + host_surface="loopx-managed-codex", + thread_id="owned-managed-thread", + todo_id="todo-managed", + profile_digest="c" * 64, + host_turn_id="native-turn-1", + model="test-model", + reasoning_effort="xhigh", + ) args = Namespace( goal_channel_command="consume-operation", goal_id=GOAL_ID, diff --git a/tests/test_chat_operation_actions.py b/tests/test_chat_operation_actions.py index 425c3a98e3..00d03850b1 100644 --- a/tests/test_chat_operation_actions.py +++ b/tests/test_chat_operation_actions.py @@ -68,6 +68,314 @@ def _claim_agent_operation( ) +def _managed_handler( + service: ChatActionService, + store: ChatActionStore, + *, + session_id="owned-managed-thread", + profile_digest="c" * 64, + todo_id="todo-managed", + model="test-model", + reasoning_effort="xhigh", +): + from loopx.control_plane.turn_driver.codex_cli import _store_codex_cli_session + from loopx.control_plane.turn_driver.codex_operation_host import ( + operation_tool_handler, + ) + + lineage = { + "goal_id": GOAL_ID, + "agent_id": "finance-fixture-agent", + "todo_id": todo_id, + } + runtime = store.root.parent.parent + _store_codex_cli_session( + runtime, + lineage=lineage, + session_id=session_id, + operation_profile_digest=profile_digest, + operation_model=model, + operation_reasoning_effort=reasoning_effort, + ) + return operation_tool_handler( + runtime_root=runtime, + registry_path=service.registry_path, + lineage=lineage, + session_id=session_id, + profile_digest=profile_digest, + model=model, + reasoning_effort=reasoning_effort, + ) + + +def test_owned_managed_tool_uses_canonical_approval_once_without_desktop_binding( + tmp_path: Path, +) -> None: + service, store = _service(tmp_path) + handler = _managed_handler(service, store) + native = {"thread_id": "owned-managed-thread", "host_turn_id": "native-turn-1"} + request = _request() + request["normalized_parameters"].pop("executor") + request["normalized_parameters"]["projection"]["simulated"] = False + prepared = handler( + "loopx_operation", {"action": "prepare", "request": request}, native + ) + assert prepared["ok"] is True and prepared["execution_allowed"] is False + proposal = prepared["proposal"] + executor = proposal["normalized_parameters"]["executor"] + assert executor["kind"] == "managed_turn" + assert executor["session_id"] == native["thread_id"] + args = { + "action": "consume", + "proposal_id": proposal["proposal_id"], + "consumption_id": "managed-attempt", + } + assert handler("loopx_operation", args, native)["ok"] is False # no human approval + delivered = store.record_operation_delivery( + proposal["proposal_id"], delivery=_delivery(proposal) + ) + claimed = store.decide_operation( + proposal["proposal_id"], + decision="confirm", + confirmation=_confirmation(delivered), + ) + consumed = handler("loopx_operation", args, native) + assert consumed["ok"] is True and consumed["execution_allowed"] is True + assert handler("loopx_operation", args, native)["execution_allowed"] is False + persisted = store.load(proposal["proposal_id"]) + assert persisted["operation"]["agent_handoff"]["host_turn_id"] == "native-turn-1" + outcome = _agent_result(claimed, "managed-attempt", result="submission_unknown") + reported = handler( + "loopx_operation", + { + "action": "report", + "proposal_id": proposal["proposal_id"], + "outcome": outcome, + }, + native, + ) + assert reported["ok"] is True and reported["needs_reconciliation"] is True + assert handler("loopx_operation", args, native)["execution_allowed"] is False + final = { + **outcome, + "outcome": "not_executed", + "external_write_performed": False, + "reconciles_outcome_digest": reported["outcome_digest"], + } + recovered = handler( + "loopx_operation", + {"action": "report", "proposal_id": proposal["proposal_id"], "outcome": final}, + native, + ) + assert recovered["ok"] is True and recovered["needs_reconciliation"] is False + + +def test_managed_replacement_has_evidence_only_access_and_never_inherits_execution( + tmp_path: Path, + monkeypatch: pytest.MonkeyPatch, +) -> None: + from contextlib import contextmanager + from threading import Event, current_thread + from loopx.control_plane.collaboration import operation_handoff + from loopx.control_plane.turn_driver import codex_cli + from loopx.control_plane.turn_driver.codex_cli import _discard_codex_cli_session + + service, store = _service(tmp_path) + original = _managed_handler(service, store) + native = {"thread_id": "owned-managed-thread", "host_turn_id": "native-turn-1"} + request = _request() + request["normalized_parameters"].pop("executor") + request["normalized_parameters"]["projection"]["simulated"] = False + proposal = original( + "loopx_operation", {"action": "prepare", "request": request}, native + )["proposal"] + delivered = store.record_operation_delivery( + proposal["proposal_id"], delivery=_delivery(proposal) + ) + confirmed = store.decide_operation( + proposal["proposal_id"], + decision="confirm", + confirmation=_confirmation(delivered), + ) + args = { + "action": "consume", + "proposal_id": proposal["proposal_id"], + "consumption_id": "managed-attempt", + } + assert original("loopx_operation", args, native)["execution_allowed"] is True + unknown = _agent_result(confirmed, "managed-attempt", result="submission_unknown") + reported = original( + "loopx_operation", + { + "action": "report", + "proposal_id": proposal["proposal_id"], + "outcome": unknown, + }, + native, + ) + replacement = _managed_handler( + service, + store, + session_id="replacement-thread", + profile_digest="d" * 64, + todo_id="todo-recovery", + model="replacement-model", + reasoning_effort="high", + ) + replacement_native = { + "thread_id": "replacement-thread", + "host_turn_id": "native-recovery-turn", + } + inspect = {"action": "inspect", "proposal_id": proposal["proposal_id"]} + assert replacement("loopx_operation", inspect, replacement_native)["ok"] is False + _discard_codex_cli_session( + store.root.parent.parent, + lineage={ + "goal_id": GOAL_ID, + "agent_id": "finance-fixture-agent", + "todo_id": "todo-managed", + }, + ) + observed = replacement("loopx_operation", inspect, replacement_native) + assert observed["ok"] is True and observed["execution_allowed"] is False + assert observed["access"]["permission"] == "historical_evidence_only" + assert observed["access"]["owner"]["todo_id"] == "todo-recovery" + assert observed["access"]["authority_source"] == "current_turn_session_binding" + assert replacement("loopx_operation", args, replacement_native)["ok"] is False + final = { + **unknown, + "outcome": "not_executed", + "external_write_performed": False, + "reconciles_outcome_digest": reported["outcome_digest"], + } + attempted, acquired = Event(), Event() + original_lock = codex_cli.exclusive_file_lock + original_binding = operation_handoff._binding + original_write = ChatActionStore._write + commits = [] + + @contextmanager + def observed_lock(path, *args, **kwargs): + revoking = current_thread().name.startswith("managed-revoker") + if revoking: + attempted.set() + with original_lock(path, *args, **kwargs) as proof: + if revoking: + acquired.set() + yield proof + + def revoke(): + _discard_codex_cli_session( + store.root.parent.parent, + lineage={"goal_id": GOAL_ID, "agent_id": "finance-fixture-agent", + "todo_id": "todo-recovery"}, + ) + commits.append("revocation") + + def record_report(self, payload): + original_write(self, payload) + commits.append("report") + + monkeypatch.setattr(codex_cli, "exclusive_file_lock", observed_lock) + monkeypatch.setattr(ChatActionStore, "_write", record_report) + with ThreadPoolExecutor(max_workers=1, thread_name_prefix="managed-revoker") as executor: + futures = [] + + def interleaved_binding(registry_path, parameters, runtime_root): + current = original_binding(registry_path, parameters, runtime_root) + if parameters["executor"].get("todo_id") == "todo-recovery": + futures.append(executor.submit(revoke)) + assert attempted.wait(5) + assert not acquired.wait(0.2), "replacement revocation must wait for report commit" + return current + + monkeypatch.setattr(operation_handoff, "_binding", interleaved_binding) + assert replacement( + "loopx_operation", + {"action": "report", "proposal_id": proposal["proposal_id"], "outcome": final}, + replacement_native, + )["ok"] is True + futures[0].result(timeout=5) + assert commits == ["report", "revocation"] + monkeypatch.setattr(operation_handoff, "_binding", original_binding) + before = store.path.read_bytes() + assert replacement("loopx_operation", inspect, replacement_native)["ok"] is False + assert store.path.read_bytes() == before + stored = store.load(proposal["proposal_id"]) + assert stored["operation"]["outcome"] == unknown + assert stored["operation"]["reconciliation"] == final + assert ( + stored["operation"]["reconciliation_report"]["owner"]["thread_id"] + == "replacement-thread" + ) + assert ( + stored["normalized_parameters"]["executor"]["session_id"] + == "owned-managed-thread" + ) + + +def test_managed_tool_rejects_actor_injection_native_mismatch_and_revoked_profile( + tmp_path: Path, +) -> None: + service, store = _service(tmp_path) + handler = _managed_handler(service, store) + native = {"thread_id": "owned-managed-thread", "host_turn_id": "native-turn-1"} + assert ( + handler( + "loopx_operation", {"action": "context", "actor": EXECUTION_ACTOR}, native + )["ok"] + is False + ) + assert ( + handler( + "loopx_operation", {"action": "context"}, {**native, "thread_id": "forged"} + )["ok"] + is False + ) + request = _request() + request["normalized_parameters"].pop("executor") + request["normalized_parameters"]["projection"]["simulated"] = False + prepared = handler( + "loopx_operation", {"action": "prepare", "request": request}, native + ) + proposal = prepared["proposal"] + delivered = store.record_operation_delivery( + proposal["proposal_id"], delivery=_delivery(proposal) + ) + store.decide_operation( + proposal["proposal_id"], + decision="confirm", + confirmation=_confirmation(delivered), + ) + from loopx.control_plane.turn_driver.codex_cli import _store_codex_cli_session + + _store_codex_cli_session( + store.root.parent.parent, + lineage={ + "goal_id": GOAL_ID, + "agent_id": "finance-fixture-agent", + "todo_id": "todo-managed", + }, + session_id="replacement-managed-thread", + operation_profile_digest="d" * 64, + operation_model="test-model", + operation_reasoning_effort="xhigh", + ) + assert ( + handler( + "loopx_operation", + { + "action": "consume", + "proposal_id": proposal["proposal_id"], + "consumption_id": "managed-attempt", + }, + native, + )["ok"] + is False + ) + assert store.load(proposal["proposal_id"])["operation"].get("agent_handoff") is None + + def _agent_result( proposal: dict, consumption_id: str, *, result: str = "executed" ) -> dict: @@ -704,8 +1012,8 @@ def record_report(self, payload): with ThreadPoolExecutor(max_workers=1) as executor: futures = [] - def interleaved_binding(registry_path, parameters): - current = original_binding(registry_path, parameters) + def interleaved_binding(registry_path, parameters, runtime_root): + current = original_binding(registry_path, parameters, runtime_root) if parameters["executor"]["thread_id"] == replacement["thread_id"]: futures.append(executor.submit(revoke)) assert attempted.wait(5) diff --git a/tests/test_codex_operation_host.py b/tests/test_codex_operation_host.py new file mode 100644 index 0000000000..0903ca0a2c --- /dev/null +++ b/tests/test_codex_operation_host.py @@ -0,0 +1,283 @@ +"""Non-financial managed transport qualification; no live user/account effects.""" + +from __future__ import annotations + +import os +import time +import contextlib +import io +import json +import sys +from pathlib import Path + +import pytest + +from loopx.control_plane.turn_driver.codex_cli import ( + _lineage, + load_codex_cli_session, + run_codex_cli_host, +) +from loopx.control_plane.turn_driver.codex_operation_host import ( + run_codex_operation_host, +) +from loopx.control_plane.turn_driver.host_failure import BuiltInHostError +from loopx.control_plane.turn_driver.executor import validate_loopx_turn_host_result +from test_loopx_turn_codex_cli import _request +from test_loopx_turn_driver import _write_live_fixture + + +FAKE_SERVER = """#!/usr/bin/env python3 +import json, sys +thread = "owned-app-server-thread" +turn = "native-app-server-turn" +key = None +def emit(value): + print(json.dumps(value), flush=True) +for line in sys.stdin: + row = json.loads(line) + method = row.get("method") + if method == "initialize": + emit({"id": row["id"], "result": {}}) + elif method in {"thread/start", "thread/resume"}: + assert row["params"]["sandbox"] == "read-only" + if method == "thread/start": + assert row["params"]["dynamicTools"][0]["name"] == "loopx_operation" + emit({"id": row["id"], "result": {"thread": {"id": thread}, "model": "test-model", "reasoningEffort": "xhigh"}}) + elif method == "turn/start": + assert row["params"]["outputSchema"]["type"] == "object" + properties = row["params"]["outputSchema"]["properties"] + import os, pathlib, subprocess, time + marker = os.environ.get("FAKE_OPERATION_CHILD_MARKER") + if marker: + child = subprocess.Popen([sys.executable, "-c", + "import pathlib,signal,time;signal.signal(signal.SIGTERM,signal.SIG_IGN);" + "p=pathlib.Path(" + repr(marker) + ");n=0\\n" + "while True:\\n p.write_text(str(n));n+=1;time.sleep(.01)"]) + pathlib.Path(marker + ".pid").write_text(str(child.pid)) + while not pathlib.Path(marker).exists(): time.sleep(.01) + if os.environ.get("FAKE_OPERATION_HANG") == "1": + while True: time.sleep(.1) + text = row["params"]["input"][0]["text"] + import re + key = re.search(r'"turn_key":"([^"]+)"', text).group(1) + emit({"id": row["id"], "result": {"turn": {"id": turn}}}) + emit({"id": 50, "method": "item/tool/call", "params": {"threadId": "forged-thread", "turnId": turn, "tool": "loopx_operation", "arguments": {"action": "context"}}}) + elif row.get("id") == 50: + assert row["result"]["success"] is False + emit({"id": 51, "method": "item/tool/call", "params": {"threadId": thread, "turnId": turn, "tool": "loopx_operation", "arguments": {"action": "context"}}}) + elif row.get("id") == 51: + result = json.loads(row["result"]["contentItems"][0]["text"]) + assert result["executor"]["session_id"] == thread + assert result["authority"] == "approval_and_first_consumption_required" + assert row["result"]["success"] is True + answer = {name: "" for name, schema in properties.items() if schema["type"] == "string"} + answer.update({"schema_version": "loopx_turn_result_v0", "turn_key": key, "result_kind": "wait", "completed_phases": ["host_execute", "typed_result"], "classification": "awaiting_operation_confirmation", "recommended_action": "Wait for actual human approval", "summary": "Owned native context retrieved; no approval or effect"}) + emit({"method": "item/agentMessage/delta", "params": {"threadId": thread, "turnId": turn, "delta": json.dumps(answer)}}) + emit({"method": "turn/completed", "params": {"threadId": thread, "turn": {"id": turn, "status": "completed"}}}) +""" + + +def test_owned_app_server_process_authenticates_native_metadata_and_resumes_same_profile( + tmp_path: Path, +) -> None: + executable = tmp_path / "fake-codex-operation" + executable.write_text(FAKE_SERVER) + executable.chmod(0o700) + options = { + "runtime_root": tmp_path / "runtime", + "registry_path": tmp_path / "registry.json", + "project": tmp_path, + "codex_bin": str(executable), + "model": "test-model", + "reasoning_effort": "xhigh", + "timeout_seconds": 5, + } + first = run_codex_operation_host(_request(), **options) + assert first["result_kind"] == "wait" + validation = validate_loopx_turn_host_result( + {"transaction": {"turn_key": _request()["turn_key"]}}, first + ) + assert validation["ok"], validation["errors"] + binding = load_codex_cli_session( + options["runtime_root"], lineage=_lineage(_request()) + ) + assert binding["operation_transport"] == "app-server-operation-tools-v0" + second = run_codex_operation_host( + _request(session_action="resume", turn_key="sha256:" + "b" * 64), **options + ) + assert second["turn_key"] == "sha256:" + "b" * 64 + assert ( + load_codex_cli_session(options["runtime_root"], lineage=_lineage(_request())) + == binding + ) + with pytest.raises(ValueError, match="profile changed"): + run_codex_operation_host( + _request(session_action="resume"), **{**options, "reasoning_effort": "high"} + ) + with pytest.raises(ValueError, match="original managed transport"): + run_codex_cli_host( + _request(session_action="resume"), + **{key: value for key, value in options.items() if key != "registry_path"}, + ) + + +def test_admitted_turn_cli_launches_owned_transport_without_plain_cli_fallback( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + from loopx.cli import main as cli_main + + project, runtime, registry = _write_live_fixture(tmp_path) + executable = tmp_path / "fake-codex-operation" + executable.write_text(FAKE_SERVER) + executable.chmod(0o700) + + def forbidden_plain_cli(*args, **kwargs): + pytest.fail("Operation opt-in must not downgrade to plain Codex exec") + + monkeypatch.setattr("loopx.cli_commands.turn.run_codex_cli_host", forbidden_plain_cli) + arguments = [ + "--registry", str(registry), "--runtime-root", str(runtime), "--format", "json", + "turn", "run-once", "--goal-id", "loopx-turn-fixture", "--agent-id", "codex-fixture", + "--host", "codex-cli", "--project", str(project), "--scan-root", str(project), + "--no-global-sync", "--codex-operation-tools", "--codex-bin", str(executable), + "--codex-model", "test-model", "--codex-reasoning-effort", "xhigh", + "--codex-sandbox", "read-only", "--timeout-seconds", "5", + "--validation-command-json", json.dumps([sys.executable, "-c", "import json,sys; json.load(sys.stdin)"]), + ] + lineage = {"goal_id": "loopx-turn-fixture", "agent_id": "codex-fixture", "todo_id": "todo_fixture0001"} + output = io.StringIO() + with contextlib.redirect_stdout(output): + dry_exit = cli_main(arguments) + dry = json.loads(output.getvalue()) + assert dry_exit == 0, dry + assert load_codex_cli_session(runtime, lineage=lineage) is None + output = io.StringIO() + with contextlib.redirect_stdout(output): + exit_code = cli_main([*arguments, "--execute"]) + payload = json.loads(output.getvalue()) + assert exit_code == 0, json.dumps(payload, indent=2) + assert payload["host"] == {"executable": "built-in", "kind": "codex-cli"} + binding = load_codex_cli_session(runtime, lineage=lineage) + assert binding["operation_transport"] == "app-server-operation-tools-v0" + assert binding["session_id"] == "owned-app-server-thread" + assert payload["effects"]["quota_spent"] is False + + +@pytest.mark.parametrize( + "change", + [{"model": None}, {"reasoning_effort": None}, {"sandbox": "danger-full-access"}], +) +def test_operation_host_refuses_unpinned_or_widened_profile_before_launch( + tmp_path: Path, change: dict +) -> None: + with pytest.raises(ValueError): + run_codex_operation_host( + _request(), + runtime_root=tmp_path / "runtime", + registry_path=tmp_path / "registry.json", + project=tmp_path, + **{"model": "test-model", "reasoning_effort": "xhigh", **change}, + ) + assert not (tmp_path / "runtime").exists() + + +@pytest.mark.skipif(os.name != "posix", reason="POSIX process-group qualification") +@pytest.mark.parametrize("hang", [False, True]) +def test_owned_operation_host_reaps_descendants_on_success_and_timeout( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch, hang: bool +) -> None: + executable = tmp_path / "fake-codex-operation" + executable.write_text(FAKE_SERVER) + executable.chmod(0o700) + marker = tmp_path / "owned-child" + monkeypatch.setenv("FAKE_OPERATION_CHILD_MARKER", str(marker)) + if hang: + monkeypatch.setenv("FAKE_OPERATION_HANG", "1") + options = dict( + runtime_root=tmp_path / "runtime", + registry_path=tmp_path / "registry.json", + project=tmp_path, + codex_bin=str(executable), + model="test-model", + reasoning_effort="xhigh", + timeout_seconds=1 if hang else 5, + ) + if hang: + with pytest.raises(BuiltInHostError, match="timeout"): + run_codex_operation_host(_request(), **options) + else: + assert run_codex_operation_host(_request(), **options)["result_kind"] == "wait" + assert marker.exists() + modified = marker.stat().st_mtime_ns + time.sleep(0.1) + assert marker.stat().st_mtime_ns == modified + + +@pytest.mark.skipif( + not os.environ.get("LOOPX_QUALIFY_CODEX_OPERATION_HOST"), + reason="explicit live-host release qualification only", +) +def test_live_owned_app_server_native_tool_metadata( + tmp_path: Path, monkeypatch: pytest.MonkeyPatch +) -> None: + from loopx.control_plane.turn_driver import codex_operation_host + + original = codex_operation_host.operation_tool_handler + calls = [] + + def observe(**options): + handler = original(**options) + + def record(tool, arguments, native): + result = handler(tool, arguments, native) + calls.append((arguments.get("action"), native, result.get("ok"))) + return result + + return record + + monkeypatch.setattr(codex_operation_host, "operation_tool_handler", observe) + request = _request() + request["turn_envelope"]["action"]["selected_todo"]["text"] = ( + "Non-financial transport qualification only. Call loopx_operation context once. " + "Do not use shell, create proposals, send messages or perform any external effect. " + "Return result_kind wait with the exact supplied Turn key, using only the existing result schema. " + "Report the context retrieval in summary; leave non-applicable material-work fields empty." + ) + options = dict( + runtime_root=tmp_path / "runtime", + registry_path=tmp_path / "registry.json", + project=tmp_path, + model="gpt-6-sol", + reasoning_effort="xhigh", + timeout_seconds=120, + ) + result = run_codex_operation_host(request, **options) + validation = validate_loopx_turn_host_result( + {"transaction": {"turn_key": request["turn_key"]}}, result + ) + assert validation["ok"], validation["errors"] + assert result["turn_key"] == request["turn_key"] + assert result["result_kind"] == "wait" + assert calls and all( + action == "context" and ok and native["host_turn_id"] + for action, native, ok in calls + ) + first_thread = calls[0][1]["thread_id"] + first_turn = calls[0][1]["host_turn_id"] + calls.clear() + request["session"]["action"] = "resume" + request["turn_key"] = "sha256:" + "b" * 64 + continued = run_codex_operation_host(request, **options) + validation = validate_loopx_turn_host_result( + {"transaction": {"turn_key": request["turn_key"]}}, continued + ) + assert validation["ok"], validation["errors"] + assert ( + continued["result_kind"] == "wait" + and continued["turn_key"] == request["turn_key"] + ) + assert calls and all( + action == "context" and ok and native["thread_id"] == first_thread + and native["host_turn_id"] != first_turn + for action, native, ok in calls + ) diff --git a/tests/test_delegation_preflight.py b/tests/test_delegation_preflight.py index 80dbabaddb..b89f0fa5f9 100644 --- a/tests/test_delegation_preflight.py +++ b/tests/test_delegation_preflight.py @@ -583,7 +583,8 @@ def test_selected_dsh_profile_is_not_replaced_by_the_default(service): assert not (root / "host-started").exists() -def test_selected_codex_managed_agent_profile_is_projected_exactly(service): +@pytest.mark.parametrize("operation_tools", [False, True]) +def test_selected_codex_managed_agent_profile_is_projected_exactly(service, operation_tools): root, runner = service config = json.loads(runner.config.read_text()) config["bindings"][0]["host_args"] = [ @@ -594,12 +595,14 @@ def test_selected_codex_managed_agent_profile_is_projected_exactly(service): "--codex-reasoning-effort", "xhigh", ] + if operation_tools: + config["bindings"][0]["host_args"].append("--codex-operation-tools") runner.config.write_text(json.dumps(config)) status, result = cli(runner, "inspect", "--binding-id", "analysis") assert status == 0, result - assert result["executor"] == { + expected = { "host": "codex-cli", "available": None, "reason": None, @@ -607,12 +610,19 @@ def test_selected_codex_managed_agent_profile_is_projected_exactly(service): "runtime_probe": None, "unavailable_remediation": [], } + if operation_tools: + from loopx.control_plane.turn_driver.host_binding import managed_executor_binding_from_host_args + expected["operation_transport"] = managed_executor_binding_from_host_args( + config["bindings"][0]["host_args"] + )["operation_transport"] + assert result["executor"] == expected assert result["state"] == "runtime_unverified" assert not any(result["effects"].values()) assert not (root / "host-started").exists() binding = runner.binding("analysis", require_active=True) execution = runner._execution_arguments(binding, "native-tool-inspection") + assert ("--codex-operation-tools" in execution) is operation_tools encoded = execution[execution.index("--codex-mcp-server-json") + 1] native = json.loads(encoded) assert native["schema_version"] == "codex_stdio_mcp_server_v0" diff --git a/tests/test_turn_managed_executor_binding.py b/tests/test_turn_managed_executor_binding.py index 7ab7261dc6..b735c17d27 100644 --- a/tests/test_turn_managed_executor_binding.py +++ b/tests/test_turn_managed_executor_binding.py @@ -108,6 +108,34 @@ def test_trusted_codex_binding_projects_its_independent_agent_profile(): assert "unrelated-managed-credential" not in json.dumps(binding) +def test_owned_operation_transport_is_opt_in_and_read_back_by_the_existing_host_owner(): + options = ["--host", "codex-cli", "--codex-operation-tools"] + blocked = managed_executor_binding_from_host_args(options, environ={}) + assert blocked["available"] is False + assert blocked["unavailable_reason"] == "operation_transport_profile_required" + pinned = managed_executor_binding_from_host_args( + [*options, "--codex-model", "test-model", "--codex-reasoning-effort", "xhigh"], + environ={}, + ) + assert pinned["available"] is None # argv cannot qualify a real host + assert pinned["execution_profile"] == "test-model@xhigh" + assert pinned["operation_transport"]["identity_source"] == "native_thread_turn_metadata" + assert pinned["operation_transport"]["runtime_qualified"] is False + assert "operation_transport" not in managed_executor_binding("codex-cli") + + +@pytest.mark.parametrize("host", ["dsh", "claude-code", "generic-cli"]) +def test_operation_transport_never_disappears_or_falls_back_on_another_host(host): + observed = managed_executor_binding( + host, environ={}, dsh_runner_configured=True, + codex_operation_tools=True, model="test-model", reasoning_effort="xhigh", + ) + assert observed["available"] is False + assert observed["unavailable_reason"] == "operation_transport_host_unsupported" + assert observed["operation_transport"]["configuration_valid"] is False + assert observed["operation_transport"]["runtime_qualified"] is False + + def test_managed_executor_reports_the_operator_credential_and_endpoint(): binding = managed_executor_binding( "dsh", From 650023c585ce1935a578a6461d77bbe5e10e00c4 Mon Sep 17 00:00:00 2001 From: huangruiteng <14976749+huangruiteng@users.noreply.github.com> Date: Wed, 30 Sep 2026 13:21:05 +0800 Subject: [PATCH 09/13] feat(presentation): distinguish managed approval from actual execution Signed-off-by: huangruiteng <14976749+huangruiteng@users.noreply.github.com> --- .../smoke/action-review-plan-smoke.ts | 11 +++ .../delegation-preflight-browser-smoke.mjs | 25 ++++- .../smoke/delegation-preflight-smoke.tsx | 17 ++++ .../src/data/delegation-preflight.ts | 4 +- .../delegation-preflight-status.tsx | 9 ++ .../src/features/personal-workspace/i18n.tsx | 8 ++ .../personal-workspace-page.tsx | 3 +- examples/personal-workspace-browser-smoke.mjs | 2 + .../confirmed-operation-fixtures.py | 98 +++++++++++++++++++ .../confirmed-operations.mjs | 54 ++++++++++ .../presentation/action_review_plan.ts | 25 ++++- .../action_review_plan.test.ts | 20 ++++ 12 files changed, 268 insertions(+), 8 deletions(-) create mode 100644 examples/personal-workspace-browser/confirmed-operation-fixtures.py create mode 100644 examples/personal-workspace-browser/confirmed-operations.mjs diff --git a/apps/presentation/dashboard/smoke/action-review-plan-smoke.ts b/apps/presentation/dashboard/smoke/action-review-plan-smoke.ts index f85566db5a..fecf558d32 100644 --- a/apps/presentation/dashboard/smoke/action-review-plan-smoke.ts +++ b/apps/presentation/dashboard/smoke/action-review-plan-smoke.ts @@ -135,6 +135,17 @@ check(unauthenticatedFrame?.kind === "pending" && unauthenticatedFrame.execution const unknownAgentResult = typedActionProposalSchema.parse({...agentPending, status: "applied", receipt: {projection_verified: true}, operation: {...agentPending.operation, lifecycle_state: "outcome_observed", outcome: {outcome: "submission_unknown", simulation: false}, result_delivery: {outcome_stage: "initial"}}}); +const managedPending = typedActionProposalSchema.parse({...agentPending, + normalized_parameters: {...agentPending.normalized_parameters, agent_id: "managed-worker", executor: { + kind: "managed_turn", todo_id: "todo-managed", session_id: "owned-thread", profile_digest: "a".repeat(64), + model: "test-model", reasoning_effort: "xhigh", revision: "managed-turn-handoff-v0"}}, + operation: {...agentPending.operation, agent_handoff: null}}); +const managedFrame = compileActionReviewPlan(managedPending).operationFrame; +check(managedFrame?.kind === "pending" && managedFrame.executionState === "managed_turn_pending", + "Managed confirmation waits for its exact admitted executor rather than Desktop authentication"); +check(managedFrame?.content.fields.some(field => field.value.includes("test-model@xhigh")) === true, + "The shared managed profile survives the frontend schema transport"); +check(compileActionReviewPlan(managedPending).canApply === false, "Managed approval exposes no local execute control"); check(compileActionReviewPlan(unknownAgentResult).interaction === "repair", "Delivered unknown submission is not completion"); const reconciledAgentResult = typedActionProposalSchema.parse({...unknownAgentResult, operation: {...unknownAgentResult.operation, reconciliation: {outcome: "not_executed", simulation: false}}}); diff --git a/apps/presentation/dashboard/smoke/delegation-preflight-browser-smoke.mjs b/apps/presentation/dashboard/smoke/delegation-preflight-browser-smoke.mjs index e13b91c9e8..f5e040583b 100644 --- a/apps/presentation/dashboard/smoke/delegation-preflight-browser-smoke.mjs +++ b/apps/presentation/dashboard/smoke/delegation-preflight-browser-smoke.mjs @@ -3,6 +3,7 @@ import assert from "node:assert/strict"; import {mkdir} from "node:fs/promises"; import {resolve} from "node:path"; import {delegationPreflight} from "../../../../loopx/control_plane/collaboration/delegation.ts"; +import {projectManagedOperationTransport} from "../../../../loopx/control_plane/work_items/operation_agent_handoff.ts"; import {launchBrowser, loadPlaywright, waitForHttp} from "../../../../examples/dashboard-browser-smoke-support.mjs"; import {outputDir, packaged, port, startServer} from "../../../../examples/personal-workspace-browser/fixture.mjs"; import {openWorkspacePage} from "../../../../examples/personal-workspace-browser/scenario-context.mjs"; @@ -35,7 +36,16 @@ try { acceptance_declaration_mismatch: "completion_validation_declaration_mismatch", acceptance_unknown: "/private/validator PRIVATE_VALUE", }; - const check = state.startsWith("acceptance_") + const operation = projectManagedOperationTransport({host: "codex-cli", sandbox: "read-only", + model: state === "operation_invalid" ? null : "test-model", reasoning_effort: "xhigh"}); + const check = state.startsWith("operation_") + ? delegationPreflight({binding, acceptance: {todo_id: binding.todo_id, state: "ready"}, + validation_files_current: true, preview: {dry_run: true, status: "preview", + effects: {host_invoked: false, state_written: false, quota_spent: false, scheduler_acknowledged: false}, + route: {kind: "ready_for_host", would_invoke_host: true, selected_todo_id: binding.todo_id}, + managed_executor: {executor: "codex-cli", available: operation.reason ? false : null, + unavailable_reason: operation.reason, execution_profile: "test-model@xhigh", operation_transport: operation.transport}}}) + : state.startsWith("acceptance_") ? delegationPreflight({binding, acceptance: {todo_id: binding.todo_id, state: state === "acceptance_files" ? "ready" : "unbound", reason: acceptanceReasons[state]}, validation_files_current: false, @@ -113,6 +123,19 @@ try { } } } + for (const valid of [true, false]) { + state = valid ? "operation_valid" : "operation_invalid"; + await team.getByRole("button", {name: zh ? "检查整个团队" : "Check whole team", exact: true}).click(); + const observations = team.locator(".goal-team-bindings > li > p[role=status]"); + const expected = valid ? (zh ? "运行未核验" : "runtime unqualified") + : (zh ? "操作传输配置未获准" : "Operation transport configuration not admitted"); + await observations.filter({hasText: expected}).nth(2).waitFor(); + for (const [size, viewport] of [["desktop", {width: 1512, height: 982}], ["mobile", {width: 390, height: 844}]]) { + await page.setViewportSize(viewport); + assert(await dialog.evaluate(el => el.scrollWidth <= el.clientWidth), "Operation transport readback must not overflow"); + if (valid) await page.screenshot({path: resolve(outputDir, `operation-preflight-${zh ? "zh" : "en"}-${size}.png`), animations: "disabled"}); + } + } const member = team.locator(".goal-team-bindings > li").filter({hasText: "local-analyst"}); await member.locator("summary").click(); state = "available"; diff --git a/apps/presentation/dashboard/smoke/delegation-preflight-smoke.tsx b/apps/presentation/dashboard/smoke/delegation-preflight-smoke.tsx index 8a5c7d9f6b..3c2956b633 100644 --- a/apps/presentation/dashboard/smoke/delegation-preflight-smoke.tsx +++ b/apps/presentation/dashboard/smoke/delegation-preflight-smoke.tsx @@ -80,3 +80,20 @@ for (const zh of [true, false]) { } console.log("delegation authority/workspace/validation preflight smoke passed"); + +for (const valid of [true, false]) { + const check = {...unavailable, state: "runtime_unverified", authority_ready: true, + authority_reason: null, authority_state: "promoted", authority_next_action: "none", + executor: {host: "codex-cli", available: valid, reason: valid ? null : "operation_transport_profile_required", + profile: "test-model@xhigh", operation_transport: {schema_version: "loopx_operation_transport_v0", + configuration_valid: valid, runtime_qualified: false}}} as DelegationPreflight; + for (const zh of [false, true]) { + const html = renderToStaticMarkup(); + if (!html.includes(valid ? (zh ? "运行未核验" : "runtime unqualified") + : (zh ? "操作传输配置未获准" : "Operation transport configuration not admitted"))) { + throw new Error("Shared transport projection lost truthful configuration readback"); + } + if (/Runtime qualified|运行已核验/.test(html)) throw new Error("Preflight invented transport qualification"); + } +} +console.log("managed operation transport remains configuration-only in delegation preflight"); diff --git a/apps/presentation/dashboard/src/data/delegation-preflight.ts b/apps/presentation/dashboard/src/data/delegation-preflight.ts index 67d2bcfd0a..c6064a3cb7 100644 --- a/apps/presentation/dashboard/src/data/delegation-preflight.ts +++ b/apps/presentation/dashboard/src/data/delegation-preflight.ts @@ -12,6 +12,8 @@ export type DelegationPreflight = { authority_state: "promotion_required" | "unavailable" | "promoted" | "uninspected"; authority_next_action: "preview_reviewed_goal_authority_promotion" | "repair_canonical_authority" | "none"; promotion_from_surface_allowed: false; - executor: {host: string; available: boolean | null; reason: string | null; profile: string | null} | null; + executor: {host: string; available: boolean | null; reason: string | null; profile: string | null; + operation_transport?: {schema_version: "loopx_operation_transport_v0"; + configuration_valid: boolean; runtime_qualified: false}} | null; effects: {host_invoked: boolean; state_written: boolean; quota_spent: boolean; scheduler_acknowledged: boolean}; }; diff --git a/apps/presentation/dashboard/src/features/personal-workspace/delegation-preflight-status.tsx b/apps/presentation/dashboard/src/features/personal-workspace/delegation-preflight-status.tsx index 22f4fd5a95..f002f62fa4 100644 --- a/apps/presentation/dashboard/src/features/personal-workspace/delegation-preflight-status.tsx +++ b/apps/presentation/dashboard/src/features/personal-workspace/delegation-preflight-status.tsx @@ -48,6 +48,14 @@ export function DelegationPreflightStatus({check, zh}: {check: DelegationPreflig ? acceptanceActions[check.acceptance_next_action] : null; const executorDetail = check.executor ? [check.executor.host, check.executor.reason].filter(Boolean).join(" · ") : check.authority_reason; + const operationTransport = check.executor?.operation_transport; + const operationDetail = operationTransport?.schema_version === "loopx_operation_transport_v0" + ? operationTransport.configuration_valid === true && operationTransport.runtime_qualified === false + ? (zh ? "自有原生工具传输已配置,运行未核验;执行仍需精确批准与首次消费" + : "Owned native tool transport configured, runtime unqualified; exact approval and first consumption still required") + : (zh ? "操作传输配置未获准;请原配置责任人核对宿主、固定模型/思考深度和 sandbox" + : "Operation transport configuration not admitted; ask its original owner to review host, pinned model/effort and sandbox") + : null; const detail = check.state === "workspace_unavailable" ? (workspaceDetail ? (zh ? workspaceDetail.zh : workspaceDetail.en) : null) : [acceptanceDetail ? (zh ? acceptanceDetail.zh : acceptanceDetail.en) : null, executorDetail].filter(Boolean).join(" · "); @@ -66,6 +74,7 @@ export function DelegationPreflightStatus({check, zh}: {check: DelegationPreflig return

{zh ? labels[check.state].zh : labels[check.state].en} {detail ? ` · ${detail}` : ""} + {operationDetail ? ` · ${operationDetail}` : ""} {nextAction ? ` · ${nextAction}` : ""} {` · ${disclaimer}`}

; diff --git a/apps/presentation/dashboard/src/features/personal-workspace/i18n.tsx b/apps/presentation/dashboard/src/features/personal-workspace/i18n.tsx index 6b96e2f74b..963b917045 100644 --- a/apps/presentation/dashboard/src/features/personal-workspace/i18n.tsx +++ b/apps/presentation/dashboard/src/features/personal-workspace/i18n.tsx @@ -677,6 +677,7 @@ const en = { "proposal.field.operation": "Operation", "proposal.field.operationState": "Operation state", "proposal.operationState.host_authentication_required": "Confirmed; original-host authentication unavailable", + "proposal.operationState.managed_turn_pending": "Confirmed; waiting for the bound managed Turn", "proposal.operationState.consumed_outcome_pending": "Authorization consumed; waiting for the real result", "proposal.operationState.submission_unknown": "Result unknown; reconcile the original operation, do not resubmit", "proposal.field.resultDelivery": "Result delivery", @@ -701,6 +702,8 @@ const en = { "actionReview.authority_gate": "A gate prevents execution. Resolve its requirements, then recheck the proposal.", "actionReview.stale_proposal": "The source state changed. Generate a new preview; the previous decision no longer applies.", "actionReview.apply_pending": "Execution is in progress. Wait for its result before retrying.", + "actionReview.operation_authorization_pending": "Confirmed, not executed. The bound executor must consume this exact approval before acting.", + "actionReview.operation_outcome_pending": "Approval consumed; no real outcome is established yet. Reconcile the original result, never submit again.", "actionReview.readback_verified": "The action completed and its resulting state was verified.", "actionReview.readback_unverified": "The action returned without verified readback. Completion is not confirmed; recheck the state.", "actionReview.operation_reconcile_original": "Keep the original request. The original Agent must reconcile original external evidence; confirmation or expiry cannot grant another submission.", @@ -720,6 +723,7 @@ const en = { "proposal.impact.protected": "This action must be completed through the protected LoopX write service.", "proposal.impact.operation": "The exact terms are read-only here. Confirm or reject the same immutable request in the bound Feishu group; confirmation consumes one canonical claim.", "proposal.impact.operationAuthorized": "Human confirmation is recorded, but the original host's authenticated tool transport is not connected. Thread flags or environment ids cannot authorize execution. No external result is recorded yet.", + "proposal.impact.operationManagedPending": "Confirmation is bound to the selected managed session and execution profile, not the source conversation. The admitted Turn must consume it once through its owned tool connection. No execution or external result is established yet.", "proposal.impact.operationConsumed": "The authorization has been consumed. Wait for original external evidence; a retry or lost response must not grant another submission.", "proposal.impact.operationUnknown": "Submission may have had an external effect. Reconcile the original operation using its evidence; do not resubmit or treat card delivery as execution completion.", "proposal.primary.apply": "Confirm and apply", @@ -1865,6 +1869,7 @@ const zhCN: Record = { "proposal.field.operation": "操作", "proposal.field.operationState": "操作状态", "proposal.operationState.host_authentication_required": "已确认,原宿主身份认证尚未接通", + "proposal.operationState.managed_turn_pending": "已确认,等待绑定的受管回合", "proposal.operationState.consumed_outcome_pending": "授权已消费,等待真实结果", "proposal.operationState.submission_unknown": "结果未知;核对原操作,不可重复提交", "proposal.field.resultDelivery": "结果回传", @@ -1889,6 +1894,8 @@ const zhCN: Record = { "actionReview.authority_gate": "Gate 阻止执行。请满足其要求后重新检查提案。", "actionReview.stale_proposal": "来源状态已变化。请重新生成预览,原决定不再适用。", "actionReview.apply_pending": "正在执行,请等待读回结果后再重试。", + "actionReview.operation_authorization_pending": "已确认,尚未执行;绑定执行者须先消费本次精确批准,才能行动。", + "actionReview.operation_outcome_pending": "批准已消费,尚无真实结果;核对原操作,不可再次提交。", "actionReview.readback_verified": "操作已完成,结果状态已通过读回验证。", "actionReview.readback_unverified": "操作返回但未通过读回验证。尚不能确认完成,请重新检查状态。", "actionReview.operation_reconcile_original": "保留原请求;原 Agent 须按平台原始证据对账。确认或到期均不会授予再次提交许可。", @@ -1908,6 +1915,7 @@ const zhCN: Record = { "proposal.impact.protected": "该操作需要通过受保护的 LoopX 写入服务完成。", "proposal.impact.operation": "这里仅展示同一份不可变条款。请在已绑定的飞书群确认或拒绝;确认只会消费一个规范 claim。", "proposal.impact.operationAuthorized": "用户确认已记录,但原宿主的认证工具通道尚未接通。线程参数或环境变量不能授权执行;目前尚无外部执行结果。", + "proposal.impact.operationManagedPending": "批准绑定选定的受管会话和执行配置,不绑定来源对话。通过准入的回合须在自有工具通道上消费一次;目前不代表已执行,也没有外部结果。", "proposal.impact.operationConsumed": "授权已消费。等待原始外部证据;重试或响应丢失均不得重新授予提交许可。", "proposal.impact.operationUnknown": "提交可能已产生外部副作用。须以原始证据核对原操作,不可重提,也不能把卡片投递当作执行完成。", "proposal.primary.apply": "确认并应用", diff --git a/apps/presentation/dashboard/src/features/personal-workspace/personal-workspace-page.tsx b/apps/presentation/dashboard/src/features/personal-workspace/personal-workspace-page.tsx index 8de52d1ed5..ee8c967983 100644 --- a/apps/presentation/dashboard/src/features/personal-workspace/personal-workspace-page.tsx +++ b/apps/presentation/dashboard/src/features/personal-workspace/personal-workspace-page.tsx @@ -657,7 +657,8 @@ function workspaceProposal(proposal: TypedActionProposal, t: WorkspaceTranslate) impact: reviewPlan.retryOriginal ? t(`actionReview.${reviewPlan.reason}`) : proposal.action_kind === "operation.execute" ? operationFrame?.kind === "pending" && operationFrame.executionState ? t(operationFrame.executionState === "consumed_outcome_pending" - ? "proposal.impact.operationConsumed" : "proposal.impact.operationAuthorized") + ? "proposal.impact.operationConsumed" : operationFrame.executionState === "managed_turn_pending" + ? "proposal.impact.operationManagedPending" : "proposal.impact.operationAuthorized") : operationFrame?.kind === "result" && operationFrame.resultKind === "unknown" ? t("proposal.impact.operationUnknown") : t("proposal.impact.operation") : proposal.action_kind === "team.plan" diff --git a/examples/personal-workspace-browser-smoke.mjs b/examples/personal-workspace-browser-smoke.mjs index af4eeddb2f..300c442a12 100644 --- a/examples/personal-workspace-browser-smoke.mjs +++ b/examples/personal-workspace-browser-smoke.mjs @@ -33,6 +33,7 @@ import { progressiveLoadingScenario } from "./personal-workspace-browser/progres import { stewardJourneyScenario } from "./personal-workspace-browser/steward-journey.mjs"; import { teamPlanScenario } from "./personal-workspace-browser/team-plan.mjs"; import { typedActionsScenario } from "./personal-workspace-browser/typed-actions.mjs"; +import { confirmedOperationsScenario } from "./personal-workspace-browser/confirmed-operations.mjs"; import { stewardModelSettingsScenario } from "./personal-workspace-browser/steward-model-settings.mjs"; import { workspaceLocaleScenario } from "./personal-workspace-browser/workspace-locale.mjs"; import { answerPresentationScenario } from "./personal-workspace-browser/answer-presentation.mjs"; @@ -50,6 +51,7 @@ import { executionServiceOfflineScenario } from "./personal-workspace-browser/ex import { conversationStartupScenario } from "./personal-workspace-browser/conversation-startup.mjs"; const scenarioCatalog = [conversationStartupScenario,goalDraftScenario, capabilityScopeScenario, stewardGroupTriggerScenario, conversationInputScenario, goalActivityScenario, conversationActivityScenario, navigationSortingScenario, automationCadenceScenario, chatRecoveryScenario, conversationReturnContinuityScenario, answerPresentationScenario, loopxModeScenario, teamEvidenceScenario, managedGoalResultsScenario, typedActionsScenario, teamPlanScenario, stewardJourneyScenario, executionChipScenario, stewardModelSettingsScenario, progressiveLoadingScenario, workspaceLocaleScenario, newestDraftScenario, larkCliMissingScenario, executionServiceOfflineScenario]; +scenarioCatalog.push(confirmedOperationsScenario); const requestedScenario = process.env.LOOPX_PERSONAL_WORKSPACE_SCENARIO; const scenarios = requestedScenario ? scenarioCatalog.filter((scenario) => scenario.id === requestedScenario) diff --git a/examples/personal-workspace-browser/confirmed-operation-fixtures.py b/examples/personal-workspace-browser/confirmed-operation-fixtures.py new file mode 100644 index 0000000000..58fe521c6f --- /dev/null +++ b/examples/personal-workspace-browser/confirmed-operation-fixtures.py @@ -0,0 +1,98 @@ +"""Isolated canonical-store fixtures; never contact a real channel or executor.""" + +from __future__ import annotations + +import json +from pathlib import Path +import sys +from tempfile import TemporaryDirectory + +sys.path.insert(0, str(Path(__file__).resolve().parents[2] / "tests")) +import test_chat_operation_actions as fixtures # noqa: E402 + + +def main() -> None: + fixtures.GOAL_ID = "product-release" + with TemporaryDirectory(prefix="loopx-operation-ui-") as root: + service, store = fixtures._service(Path(root)) + handler = fixtures._managed_handler(service, store) + native = { + "thread_id": "owned-managed-thread", + "host_turn_id": "fixture-native-turn", + } + request = fixtures._request() + terms = request["normalized_parameters"] + terms.pop("executor") + terms["projection"].update( + title="Synthetic managed operation", + subtitle="Isolated backend fixture; no real account", + warning="Engineering fixture only. No real channel, financial effect or automatic wakeup.", + simulated=False, + ) + proposal = handler( + "loopx_operation", {"action": "prepare", "request": request}, native + )["proposal"] + delivered = store.record_operation_delivery( + proposal["proposal_id"], delivery=fixtures._delivery(proposal) + ) + confirmed = store.decide_operation( + proposal["proposal_id"], + decision="confirm", + confirmation=fixtures._confirmation(delivered), + ) + consumed = handler( + "loopx_operation", + { + "action": "consume", + "proposal_id": proposal["proposal_id"], + "consumption_id": "fixture-attempt", + }, + native, + ) + assert consumed["execution_allowed"] is True + waiting = store.load(proposal["proposal_id"]) + unknown = fixtures._agent_result( + confirmed, "fixture-attempt", result="submission_unknown" + ) + reported = handler( + "loopx_operation", + { + "action": "report", + "proposal_id": proposal["proposal_id"], + "outcome": unknown, + }, + native, + ) + assert reported["ok"] is True + ambiguous = store.load(proposal["proposal_id"]) + final = { + **unknown, + "outcome": "not_executed", + "external_write_performed": False, + "reconciles_outcome_digest": reported["outcome_digest"], + } + recovered = handler( + "loopx_operation", + { + "action": "report", + "proposal_id": proposal["proposal_id"], + "outcome": final, + }, + native, + ) + assert recovered["ok"] is True + reconciled = store.load(proposal["proposal_id"]) + print( + json.dumps( + { + "confirmed": confirmed, + "waiting": waiting, + "unknown": ambiguous, + "reconciled": reconciled, + } + ) + ) + + +if __name__ == "__main__": + main() diff --git a/examples/personal-workspace-browser/confirmed-operations.mjs b/examples/personal-workspace-browser/confirmed-operations.mjs new file mode 100644 index 0000000000..7bed7259f4 --- /dev/null +++ b/examples/personal-workspace-browser/confirmed-operations.mjs @@ -0,0 +1,54 @@ +import assert from "node:assert/strict"; +import { execFileSync } from "node:child_process"; +import { resolve } from "node:path"; +import { resolveTestPython } from "../../scripts/test-python.mjs"; +import { outputDir, repoRoot } from "./fixture.mjs"; +import { openWorkspacePage } from "./scenario-context.mjs"; + +export const confirmedOperationsScenario = { + id: "confirmed-operations", + async run({ browser, url }) { + const fixtures = JSON.parse(execFileSync(resolveTestPython(), [ + resolve(repoRoot, "examples/personal-workspace-browser/confirmed-operation-fixtures.py"), + ], { cwd: repoRoot, env: { ...process.env, PYTHONPATH: repoRoot }, encoding: "utf8" })); + const states = { + confirmed: { en: "Confirmed; waiting for the bound managed Turn", "zh-CN": "已确认,等待绑定的受管回合" }, + waiting: { en: "Authorization consumed; waiting for the real result", "zh-CN": "授权已消费,等待真实结果" }, + unknown: { en: "Result unknown; reconcile the original operation, do not resubmit", "zh-CN": "结果未知;核对原操作,不可重复提交" }, + reconciled: { en: "Result card delivery pending", "zh-CN": "等待回传并核验原群卡片" }, + }; + for (const locale of ["en", "zh-CN"]) for (const width of [1512, 390]) { + for (const [state, proposal] of Object.entries(fixtures)) { + const ui = await openWorkspacePage(browser, url, { + viewport: { width, height: 982 }, + apiOptions: { initialActionProposals: [proposal] }, + beforeGoto: async (_api, page) => page.addInitScript(value => localStorage.setItem("loopx-pw-locale", value), locale), + }); + try { + const { page } = ui; + if (width < 640) await page.locator(".personal-mobile-menu").click(); + await page.locator(".personal-goal-link", { hasText: "Product Release" }).click(); + await page.getByRole("navigation", { name: locale === "en" ? "Goal view" : "Goal 视图" }).getByRole("button", { name: /^(Chat|对话)$/ }).click(); + await page.locator(".personal-proposal-row").filter({ hasText: proposal.normalized_parameters.projection.title }).click(); + const drawer = page.locator('.personal-context-drawer[data-context-kind="proposal"]'); + await drawer.getByText(states[state][locale], { exact: true }).first().waitFor({ state: "visible" }); + assert.ok((await drawer.innerText()).includes("test-model@xhigh")); + assert.ok((await drawer.innerText()).includes("todo-managed")); + if (state === "confirmed" || state === "waiting") { + assert.ok(!/Execution is in progress|正在执行/u.test(await drawer.innerText()), "Approval and consumption are not proof of execution"); + } + assert.equal(await drawer.getByRole("button", { name: /确认并应用|拒绝|重新生成|Confirm and apply|Reject|Regenerate/ }).count(), 0); + assert.equal(await page.evaluate(() => document.documentElement.scrollWidth > window.innerWidth + 1), false); + assert.equal(ui.errors.length, 0, ui.errors.join(" | ")); + assert.equal(ui.api.durableWriteCount, 0, "Inspection must never execute an operation"); + if (state === "confirmed") await page.screenshot({ + path: resolve(outputDir, `managed-operation-${locale}-${width}.png`), animations: "disabled", + }); + } finally { + await ui.close(); + } + } + } + return { coverageEntries: [], note: "Canonical backend fixtures survived packaged EN/ZH desktop/mobile readback; inspection remained read-only" }; + }, +}; diff --git a/loopx/control_plane/presentation/action_review_plan.ts b/loopx/control_plane/presentation/action_review_plan.ts index 320e979085..2aec54d11b 100644 --- a/loopx/control_plane/presentation/action_review_plan.ts +++ b/loopx/control_plane/presentation/action_review_plan.ts @@ -10,6 +10,7 @@ export type ActionReviewReason = | "incomplete_proposal" | "authority_gate" | "stale_proposal" | "apply_pending" | "readback_verified" | "readback_unverified" | "apply_failed" | "inactive_proposal" + | "operation_authorization_pending" | "operation_outcome_pending" | "canonical_update_retry" | "canonical_update_projection_pending"; export type OperationReviewContent = { @@ -41,7 +42,7 @@ export type OperationReviewFrame = OperationReviewFrameBase & ( kind: "pending"; attentionKind: "progress"; interactionMode: "inform"; - executionState?: "host_authentication_required" | "consumed_outcome_pending"; + executionState?: "host_authentication_required" | "managed_turn_pending" | "consumed_outcome_pending"; } | { kind: "result"; @@ -276,6 +277,13 @@ function operationContent(parameters: JsonRecord): OperationReviewContent | null if (!label || !value) return null; fields.push({ label, value }); } + const executor = objectValue(parameters.executor); + if (executor?.kind === "managed_turn") { + fields.push({label: "Executor / 执行者", value: `Managed Turn / 受管回合 · ${compactValue(executor.model, 80)}@${compactValue(executor.reasoning_effort, 20)}`}); + fields.push({label: "Scope / 范围", value: `${compactValue(parameters.agent_id, 80)} · ${compactValue(executor.todo_id, 80)}`}); + const source = objectValue(parameters.source_route); + if (source) fields.push({label: "Source context / 来源上下文", value: `${compactValue(source.host_surface, 60)} · ${compactValue(source.agent_id, 80)}`}); + } return { title, subtitle, focus, fields, warning }; } @@ -322,8 +330,9 @@ export function compileOperationReviewFrame(proposalValue: unknown): OperationRe kind: "pending", attentionKind: "progress", interactionMode: "inform", - ...(objectValue(parameters.executor)?.kind === "agent_session" - ? {executionState: objectValue(operation.agent_handoff) ? "consumed_outcome_pending" as const : "host_authentication_required" as const} + ...(["agent_session", "managed_turn"].includes(String(objectValue(parameters.executor)?.kind)) + ? {executionState: objectValue(operation.agent_handoff) ? "consumed_outcome_pending" as const + : objectValue(parameters.executor)?.kind === "managed_turn" ? "managed_turn_pending" as const : "host_authentication_required" as const} : {}), }; } @@ -339,7 +348,7 @@ export function compileOperationReviewFrame(proposalValue: unknown): OperationRe resultKind: outcome.outcome === "submission_unknown" ? "unknown" : outcome.outcome === "not_executed" ? "not_executed" : rejected ? "rejected" : simulated ? "simulation_completed" : "completed", resultDeliveryVerified: objectValue(operation.result_delivery) !== null - && (objectValue(parameters.executor)?.kind !== "agent_session" + && (!["agent_session", "managed_turn"].includes(String(objectValue(parameters.executor)?.kind)) || objectValue(operation.result_delivery)?.outcome_stage === (operation.reconciliation ? "reconciled" : "initial")), summary: textValue(outcome.summary) ?? "", }; @@ -401,7 +410,13 @@ export function compileActionReviewPlan(proposalValue: unknown): ActionReviewPla reason: failure?.error_code === "canonical_update_projection_pending" ? "canonical_update_projection_pending" : "canonical_update_retry"}), retryOriginal: true}; } - if (proposal.status === "applying") return held("pending", "apply_pending"); + if (proposal.status === "applying") { + if (operationFrame?.kind === "pending" && operationFrame.executionState) { + return held("pending", operationFrame.executionState === "consumed_outcome_pending" + ? "operation_outcome_pending" : "operation_authorization_pending"); + } + return held("pending", "apply_pending"); + } if (proposal.status === "failed" || proposal.error != null) return held("repair", "apply_failed"); if (proposal.status !== "preview_ready" && proposal.status !== "deferred") return held("inactive", "inactive_proposal"); const reviewed = (reason: ActionReviewReason, canApply = true): ActionReviewPlan => diff --git a/tests/control_plane_ts/action_review_plan.test.ts b/tests/control_plane_ts/action_review_plan.test.ts index b73bb4b183..cc0653b650 100644 --- a/tests/control_plane_ts/action_review_plan.test.ts +++ b/tests/control_plane_ts/action_review_plan.test.ts @@ -148,6 +148,26 @@ test("original-Agent pending, unknown and reconciled results share truthful surf assert.equal(proposal.operation.outcome.outcome, "submission_unknown"); }); +test("managed executor and source context use the same frame without turning approval into execution", () => { + const proposal: Record = operationProposal("claimed"); + proposal.status = "applying"; + proposal.normalized_parameters.agent_id = "worker"; + proposal.normalized_parameters.executor = {kind: "managed_turn", todo_id: "todo-worker", model: "test-model", reasoning_effort: "xhigh"}; + proposal.normalized_parameters.source_route = {host_surface: "codex-app", agent_id: "source-agent", thread_id: "private-source-thread"}; + let frame = compileOperationReviewFrame(proposal); + assert.equal(frame?.kind === "pending" && frame.executionState, "managed_turn_pending"); + assert.equal(frame?.content.fields.at(-3)?.value, "Managed Turn / 受管回合 · test-model@xhigh"); + assert.equal(frame?.content.fields.at(-2)?.value, "worker · todo-worker"); + assert.equal(frame?.content.fields.at(-1)?.value, "codex-app · source-agent"); + assert.equal(JSON.stringify(frame).includes("private-source-thread"), false); + assert.equal(compileActionReviewPlan(proposal).canApply, false); + assert.equal(compileActionReviewPlan(proposal).reason, "operation_authorization_pending"); + proposal.operation.agent_handoff = {consumption_id: "managed-attempt"}; + frame = compileOperationReviewFrame(proposal); + assert.equal(frame?.kind === "pending" && frame.executionState, "consumed_outcome_pending"); + assert.equal(compileActionReviewPlan(proposal).reason, "operation_outcome_pending"); +}); + test("generic action review keeps state precedence and stale classification", () => { const proposal = { proposal_id: "preview-1", From 25d4354197b809146480877527b02a8e41977fd0 Mon Sep 17 00:00:00 2001 From: huangruiteng <14976749+huangruiteng@users.noreply.github.com> Date: Wed, 30 Sep 2026 13:21:49 +0800 Subject: [PATCH 10/13] docs(operations): separate source context from admitted execution Signed-off-by: huangruiteng <14976749+huangruiteng@users.noreply.github.com> --- .../human-confirmed-domain-operations-v0.md | 368 ++++++++---------- ...an-confirmed-domain-operations-v0.zh-CN.md | 266 ++++++------- .../rfcs/loopx-overall-roadmap-v0.md | 2 +- .../rfcs/loopx-overall-roadmap-v0.zh-CN.md | 2 +- 4 files changed, 278 insertions(+), 360 deletions(-) diff --git a/docs/architecture/rfcs/human-confirmed-domain-operations-v0.md b/docs/architecture/rfcs/human-confirmed-domain-operations-v0.md index 5e0107933d..f8c791a9db 100644 --- a/docs/architecture/rfcs/human-confirmed-domain-operations-v0.md +++ b/docs/architecture/rfcs/human-confirmed-domain-operations-v0.md @@ -17,7 +17,7 @@ the contract; a difference in their requirements or boundaries is a defect. Sections 1–11 define the proposed contract, not shipped commands. Section 12 records unresolved implementation choices. This document changes no runtime, default permission, configuration or user entry point. -Section 13 describes the original-Agent continuation implementation slice; +Section 13 describes the source-context/admitted-executor continuation slice; its deployment and live acceptance remain separate from local validation. ## 1. Decision summary @@ -292,216 +292,164 @@ extract the actual shared seam, not a speculative adapter framework. authenticated web-owner mechanism. A healthy event process alone is not evidence that either user path works. Required before M1 acceptance. -## 13. Original-Agent continuation slice - -For an existing, user-authorized Agent that already owns a domain's browser or -adapter workflow, do not require a new API credential path just to return an -exact human confirmation to that Agent. This is an alternative execution seam, -not a relaxation of the financial preflight, account-wide constraints or -original-source evidence requirements above. Core does not interpret a price, -choose a venue, resume a browser, sign or submit an order. - -### Owners and entry points - -- The original `operation.execute` proposal in `chat/actions/actions.json` - remains the only confirmation, claim, consumption and outcome store. -- The explicit executor shape is `{kind: "agent_session", host_surface, - thread_id, revision: "agent-session-handoff-v0"}`. Preparation checks the - original registry's exact Goal/registered-Agent/session binding and rejects - simulation masquerading as real execution. The lifecycle-only - `source_session_v1` registry currently rejects business-operation preparation; - this slice does not bypass that owner or enable a replacement instance. -- `operation_agent_handoff.ts` owns admission, one-shot consumption, - reconciliation binding and bounded Inbox attention. Python supplies locked - canonical storage, original-registry facts and existing lifecycle guards; - it is not another decision owner. -- Existing Lark prepare/deliver and authenticated callback handling are reused. - A successful confirmation leaves the original proposal claimed, with no - external result. Callback replay, simulator and card-delivery recovery must - never invoke this Agent's browser or adapter. -- The existing manager Inbox projects locators directly from canonical - operations, without a copied approval record. Turn-start hooks include them - in `agent_read_required`. Consumed/unknown obligations sort before unconsumed - tickets; a 20-item page reports total count, typed overflow reason and next - operation ID plus an independent `operation_handoff_next_cursor`. Continue - with `manager-inbox read --operation-cursor CURSOR` even when the first 20 - unknown outcomes remain unresolved. The cursor is bound to the original - runtime/Goal/Agent scope, not the ordinary request cursor. Restart without it - for new/changed work; finishing a page sequence does not resolve obligations. -- Dashboard details and the original Lark card use the shared operation frame: - confirmed/original-host authentication unavailable; consumed/waiting for real evidence; - unknown/reconcile without resubmitting; and a separately verified result. - The Dashboard remains read-only for human operation confirmation. There is - no new configuration owner: the original request chooses the executor and - the existing channel/binding owner remains authoritative. - -### Original-runtime CLI - -Use the original registry and runtime, not a copied session or another home's -records. The following selectors are reserved for the continuation interface; -**all three public CLI commands currently fail closed** because no qualified -host identity producer is connected. They do not require Lark to report that -gate. Preparing/delivering new cards retains its normal extension and -authenticated-ingress checks, but a confirmation cannot remove this host gate. - -```sh -loopx --registry REGISTRY --runtime-root RUNTIME goal-channel inspect-operation \ - --goal-id GOAL --agent-id AGENT --proposal-id OPERATION \ - --host-surface HOST --thread-id ORIGINAL_THREAD -loopx --registry REGISTRY --runtime-root RUNTIME goal-channel consume-operation \ - --goal-id GOAL --agent-id AGENT --proposal-id OPERATION \ - --host-surface HOST --thread-id ORIGINAL_THREAD \ - --consumption-id STABLE_ATTEMPT --execute -loopx --registry REGISTRY --runtime-root RUNTIME goal-channel report-operation \ - --goal-id GOAL --agent-id AGENT --proposal-id OPERATION \ - --host-surface HOST --thread-id ORIGINAL_THREAD --outcome-json OUTCOME --execute +## 13. Human-confirmed Agent execution: source and executor are separate + +An original conversation supplies context and a result-return audience. It need +not be the process that executes an operation. The long-term contract is +**human approval → admitted execution → one-shot consumption → original-system +evidence → verified return**, not “a Desktop thread ID is an execution token”. +An existing domain Agent may reuse its authorized workflow, or an explicitly +configured delegation may use a LoopX-owned managed Turn. Neither route grants +new account access, trading permission or broad write authority. + +### Single owners and immutable execution subjects + +- The original `operation.execute` in `chat/actions/actions.json` remains the + sole confirmation, claim, consumption, outcome and reconciliation store. +- The existing attached subject stays + `{kind: "agent_session", host_surface, thread_id, revision: "agent-session-handoff-v0"}`. + Its public CLI authentication gate remains closed. +- The new opt-in subject is + `{kind: "managed_turn", todo_id, session_id, profile_digest, model, reasoning_effort, revision: "managed-turn-handoff-v0"}`. + It refers to the existing Codex Turn session owner, not a second runtime + directory. Preparation verifies the registered Goal/Agent, exact Todo/session, + current Goal instance when applicable, transport and pinned profile. +- `source_route` is an optional projection of the existing registered + conversation binding. It is immutable context/return information, not caller + identity, an execution grant or proof that a message was delivered. +- TypeScript owns executor normalization, binding judgments, transport + configuration readback, one-shot admission, recovery and presentation. + Python is the native subprocess, session/storage-lock and Lark IO adapter. + No parallel Python approval or domain-neutral policy owner is introduced. +- An old attached approval is never converted to a managed approval. Changing + session, Todo, model, effort, sandbox, workspace, home, executable or + invocation-scoped MCP configuration requires an explicit fresh iteration and + fresh approval. No trajectories, SQLite rows or credentials are copied. + +Lifecycle-only `source_session_v1` registries still reject business-operation +preparation. This slice does not enable a replacement Goal instance, change +provider authority, or transplant an old operation into another runtime. + +### Existing delegation and Turn entry + +The original operator-owned delegation configuration remains the launch grant. +A Codex binding can opt in through its existing `host_args`: + +```text +--host codex-cli --codex-operation-tools +--codex-model MODEL --codex-reasoning-effort EFFORT +--codex-sandbox read-only ``` -`inspect-operation` and a command without `--execute` never consume authority. -The former ambient-thread check (for example, `CODEX_THREAD_ID`) was forgeable -by another same-user process. It is removed, not upgraded to authentication: -even an exact environment/route match returns -`operation_host_authentication_unavailable` before reading private operation -terms, outcome files or writing receipts. Unexpected actor success from an -older runtime also cannot bypass the CLI adapter's absent transport. There is -no `--verified`, self-signing command or environment-token fallback. - -The Inbox still exposes bounded locators and preserves consumed/unknown -obligations. It explicitly requests host integration instead of recommending -another blocked CLI consumption. The shared Dashboard/Lark frame says human -confirmation is recorded but original-host authentication is unavailable. -This containment removes the public environment-forgery path; **it does not -deliver an authenticated positive execution path**. Original-session-exclusive -execution therefore remains an acceptance blocker. Internal locked storage -adapters and their synthetic fixtures validate protocol semantics only, not a -host producer or a live minimum loop. A registry binding authorizes a route, -but does not authenticate its caller. - -### Required host-adapter integration - -The chosen boundary is a **transport-owned, non-exporting operation tool**. -The host handles `loopx_operation` (`inspect`, `consume`, `report`) on the -original session's authenticated tool connection and returns the receipt to -that same connection. This is a required companion contract, not an installed -tool, accepted proof field or a new approval store. - -1. The original configuration owner enrolls and revokes the host issuer against - the existing session binding. A request cannot select its own trust key or - enroll a replacement issuer. Issuer rotation does not change the immutable - operation executor or inherit unused approvals. -2. The host derives session/Turn identity from its native tool-dispatch - metadata, not tool arguments, environment variables, an MCP subprocess's - self-report or an agent-readable key file. It does not expose a general - signer or a reusable bearer token to the model/CLI. Private signing material - remains within a separately trusted host service; same-user environment - spoofing must not reach that service's identity or signing authority. -3. Across a process boundary, the issuer signs the canonical invocation with - Ed25519. The invocation binds issuer/key revision, original GoalRef and - registered Agent, host/session/Turn, original operation and its payload/ - confirmation digests, action and action-argument digest, audience/runtime, - authenticated connection, bounded issue/expiry times and a unique request - ID. Core verifies the owner-pinned issuer, signature, exact scope, current - binding and freshness in TypeScript before the existing locked IO seam. - The authenticated response stays on the original host connection; forwarding - a signed payload to a public CLI must not reveal an execution permission. -4. Authentication proves origin only. Original human confirmation, immutable - terms, active Goal, expiry and one-shot atomic consumption remain separate - gates. Request replay and unknown submission never authorize a second - external effect. A recovery host needs its own authenticated connection; - it may report evidence only under the existing replacement-recovery rules. - -**Concrete dependency:** attached Codex Desktop sessions need a session-bound -operation tool in the Desktop's native tool server, alongside its existing -app-owned tools, plus owner-controlled issuer enrollment. That native server -is not implemented in this LoopX checkout or by its CLI/MCP adapters. The host -maintainer must deliver the producer; this PR cannot substitute environment -identity or silently resume the session in a different process. -`CodexChatAgentSession._check_server_gate` already validates thread/Turn -metadata on LoopX-owned app-server tool calls, but those are different owned -sessions, not proof for an attached Desktop thread. An owned-host integration -must be qualified for its own route and cannot stand in for Desktop acceptance. - -Integrate the real producer and owner-pinned verifier as one follow-up slice; -do not ship an unused signing API or fixture-generated credentials. Qualify -forged environment/route/proof fields, foreign sessions, wrong audience, expiry, -tampering, replay, issuer/binding revocation and original-connection receipt -return. Retain the current public-CLI rejection regression when enabling the -host tool. A real original-session invocation must pass through the same -boundary before the slice can be installed. No new session, copied trajectory, -synthetic group click or real financial side effect is part of engineering QA. - -### One-shot and evidence semantics behind the host gate +The same options are available on `turn run-once`. Existing delegation +inspect/preflight, start, admission, lease, session continuation, result +validation and acceptance are reused. The profile is read back by the existing +host owner; no new frontend configuration store or hidden default is added. +Preflight reports an unpinned/unsupported configuration as unavailable. A valid +argv is still runtime-unverified, not proof that a host or an operation ran. +Use `workspace-write` only when the existing work grant requires it; +`danger-full-access` is not accepted by this transport. + +The admitted host launches the existing `CodexChatAgentSession` app-server +adapter, persists its opaque thread under the existing Goal/Agent/Todo session +owner, and installs a non-exporting `loopx_operation` dynamic tool. The owned +stdio connection checks native thread and active Turn metadata before dispatch. +The tool arguments cannot supply an actor, a verification flag, trust key, +signature or bearer token. Receipts return on that same native connection. +A registered source Desktop thread is neither resumed nor impersonated. + +`context` exposes the exact managed subject without an execution permit; +`prepare` previews immutable terms in the canonical action store; +`pending` reads a bounded Inbox; `inspect`, `consume` and `report` use the +same locked operation seam. The invocation-scoped collaboration MCP server is +preserved for ordinary delegation work; it is not an operation-identity issuer. +The host result uses the existing typed Turn result contract and validation. +Final-answer prose does not count as an operation outcome. + +This first transport is Codex-specific IO, not a new Codex-specific approval +model. Other managed hosts can implement the same contract only after their +native identity producer, effective-profile readback, revocation and response +route are qualified. Attached Desktop support remains an independent optional +adapter; it is **not a prerequisite for the managed route**. A future remote +boundary may require owner-enrolled authenticated invocation verification, but +an unused signer or locally minted credential is not a prerequisite for an +owned in-process dispatch seam. + +### Confirmation, one-shot consumption and recovery + +Existing authenticated Lark callbacks record the exact human confirmation and +claim the proposal; they never launch a browser or domain adapter. Replay, +simulation and card-delivery recovery do not execute a domain effect. +The Dashboard remains read-only for human operation confirmation. Only the first successful atomic consumption returns `execution_allowed: true`. -It verifies authenticated confirmation, immutable terms, the current original -session, active Goal and expiry, then persists consumption before any browser -effect. Every retry, including the same attempt after a lost response or -restart, returns no execution permission. This intentionally does not promise -exactly-once venue execution: ambiguity requires original-venue reconciliation. -Binding/registration/activation read and consumption commit hold the existing -registry-writer lock, in Goal-lifetime → registry → action-store order. -Revocation that commits first prevents consumption; revocation waiting behind -a consumption cannot retroactively revoke the already committed receipt. The -lock is released before external work and never claims to fence that work. - -The original Agent reports `loopx_operation_outcome_v0` with the exact operation, -payload and confirmation digests, claim, executor revision, consumption ID, -`projection_verified: true`, `simulation: false`, bounded original evidence -references and separate `external_write_performed`. Outcomes are `executed`, -`not_executed` or `submission_unknown`. Unknown conservatively reports a possible -external effect; a transport success cannot certify a trade or protection order. -References must be safe opaque receipt identifiers, not credentials or private -absolute paths. Domain evidence and private trading journals retain detail. - -### Recovery, delivery and remaining acceptance - -Consumed or unknown operations remain recovery obligations after expiry and -session rebinding; those changes cannot grant a new execution. Evidence-only -reporting may continue for a stopped/historical Goal through existing lifecycle -guards. Existing exact-instance lifecycle guards remain in place, not an -implicit migration into lifecycle-only registries or another home. Missing -original Goal/Agent registration or an unsupported registry profile is an -explicit error, not permission to transplant the operation. - -After the original route is withdrawn, a **currently registered and bound -replacement session of the same Goal and Agent** may use its own authenticated -host route for inspection and reporting once that transport is qualified. The -internal IO seam may inspect only an already -consumed operation and report original-system historical evidence. It cannot -consume an unspent ticket, change the original executor or obtain a second -execution permission. Recovery is not admitted while the original binding is -still current, for an unbound replacement or for a different Goal/Agent. The -protocol returns an explicit `access.owner`, `original_route`, -`permission: "historical_evidence_only"` and binding authority; it never asks the -replacement to impersonate the old thread. The original executor route remains -immutable. The same registry lock is held from recovery-binding validation -through result commit; revocation that commits first rejects the report. - -Result evidence remains unchanged. The canonical operation separately appends -`outcome_report` or `reconciliation_report` provenance with the actual reporter, -original route, evidence-only permission, authority source, consumption ID and -recorded time. Identical result retries preserve the first committed provenance; -they do not relabel its author or grant execution. Internal inspection exposes both -the original unknown outcome and its immutable reconciliation/provenance. The -existing shared Dashboard frame and original Lark-card recovery consume the -same canonical result; neither gets a separate recovery approval store or an -execution control. The prior positive CLI fixtures are now internal-IO/result-card -checks because the public host gate is closed. They establish historical -recovery semantics, not trusted-host authentication or live-group acceptance. - -An unknown original outcome is immutable. A definitive report appends -`operation.reconciliation` and binds `reconciles_outcome_digest` to the exact -original unknown result. Only that evidence closes the recovery obligation. -The original result card is updated by existing delivery recovery; an earlier -unknown-result delivery cannot certify the reconciled result. Its readback must -match the current `initial` or `reconciled` stage. None of these paths resubmits. - -The slice does **not** implement immediate host wakeup. Inbox visibility reports -`host_delivery: "not_attempted"`; existing Turn/heartbeat reads are not a host -delivery receipt. Host wakeup must later reuse the original host transport and -publish truthful attempt/readback evidence, without starting a parallel resumed -session. Local synthetic callback, CLI, concurrency, expiry, reconciliation and -packaged-UI checks establish protocol behavior only. Installation, a genuine -group click, original-Agent receipt consumption, real venue/protection evidence -and original-card readback are still required before claiming a live minimum -loop. No synthetic engineering card is sent to a live group for acceptance. +It checks authenticated confirmation, immutable terms, current execution +binding, active Goal and expiry, then persists consumption before the Agent's +domain work. Every retry, including the same attempt after response loss or a +restart, returns no execution permission. This fences **authorization +consumption**, not every possible tool call by a trusted Agent, and does not +promise exactly-once venue execution. + +The lock order is Goal lifetime → registry → existing Turn session (managed +route only) → action store. Session replacement/discard uses the same lock as +consumption. Revocation committed first prevents consumption; a later +revocation cannot erase a committed receipt. No lock is held across domain work. + +The bound host reports `loopx_operation_outcome_v0`: exact operation/payload/ +confirmation digests, claim, executor revision, consumption ID, verified +projection, `simulation: false`, bounded original-system evidence references +and separate `external_write_performed`. Outcomes are `executed`, +`not_executed` or `submission_unknown`; unknown conservatively discloses a +possible external effect. A native transport success is not venue evidence. +Core does not interpret prices, venues, fees, positions or protection orders. + +Consumed/unknown obligations survive expiry and session withdrawal. A current +same-Goal/Agent replacement may inspect/report historical evidence only after +the original binding is withdrawn. It cannot consume an unused ticket or +rewrite its executor. The original outcome is immutable; reconciliation binds +`reconciles_outcome_digest` and appends actual reporter/route provenance. +Readback retains both original and reconciled evidence. A stopped/historical +Goal uses the existing evidence-only lifecycle guards, not a new execution grant. + +### Shared Inbox, frontend, Lark and truthful return + +The existing Inbox projects canonical locators directly, with recovery first, +20-item pages, total/typed overflow and an independent operation cursor. +`loopx_operation pending` accepts its bound cursor; the CLI projection uses +`manager-inbox read --operation-cursor CURSOR`. New/changed work restarts +without a cursor. Reading or exhausting a page does not resolve obligations. + +The shared TypeScript operation frame shows executor, pinned model/effort, +Goal/Agent/Todo scope, optional source context, and distinct states: confirmed +but attached authentication unavailable; confirmed and awaiting the bound +managed Turn; consumed awaiting evidence; unknown requiring reconciliation; +and independently verified result delivery. CLI, Dashboard and the original +Lark card consume the same frame. Result readback must match the current +`initial` or `reconciled` stage; an older unknown-result delivery cannot certify +a reconciled result. Existing delivery recovery updates the original card, +never resubmits the operation. Delegation result acceptance and requester +adoption remain separate receipts, not an automatic new chat protocol. + +### Qualification and remaining delivery + +This slice qualifies owned-process native tool dispatch (including bounded +real Codex context calls on a new Turn and a same-thread resumed Turn, both +accepted by the existing typed result validator), canonical approval/consumption/result fixtures, +profile/session revocation, original-route isolation, delegation preflight, +and packaged presentation. The live context probe creates a new managed GPT +session in its owning home; it performs no proposals, group clicks or financial +effects. Synthetic approval fixtures are not genuine user approval. + +Public `goal-channel inspect-operation/consume-operation/report-operation` +remain fail-closed before private reads or writes, even with matching +`CODEX_THREAD_ID`, route flags, self-signed proof or an older runtime's actor +success. No proof-import shortcut is exposed. + +Immediate confirmation-triggered host wakeup is not implemented: +`host_delivery: "not_attempted"` remains truthful. Continue through the +existing admitted Turn/delegation route; later durable wakeup must reuse its +original scheduling/session owner, not start a parallel resumed executor. +Before claiming the investment minimum loop, still prove installation, +genuine human approval, bound native consumption, domain preflight and +original-system evidence, accepted result and original-card/audience readback. +The core PR requires owner review and is not self-installed before merge. diff --git a/docs/architecture/rfcs/human-confirmed-domain-operations-v0.zh-CN.md b/docs/architecture/rfcs/human-confirmed-domain-operations-v0.zh-CN.md index cfd8732e68..5dabf38c26 100644 --- a/docs/architecture/rfcs/human-confirmed-domain-operations-v0.zh-CN.md +++ b/docs/architecture/rfcs/human-confirmed-domain-operations-v0.zh-CN.md @@ -15,7 +15,7 @@ 第 1–11 节定义拟议契约,并非已发布的命令。第 12 节记录尚未解决的实现选择。 本文不修改运行时、默认权限、配置或用户入口。 -第 13 节记录原 Agent 续接的实现切片;部署与真实验收独立于本地验证。 +第 13 节记录来源上下文/已准入执行者续接的实现切片;部署与真实验收独立于本地验证。 ## 1. 决策摘要 @@ -234,152 +234,122 @@ M1 同时涵盖 UI 与后端,不要拆成“后端 PR 已完成”而遗忘前 4. **部署资格:** 核实真实飞书应用回调和 Web owner 身份认证机制。 事件进程健康本身不能证明任一用户路径可用。M1 验收前必须完成。 -## 13. 原 Agent 续接切片 - -已有用户授权的 Agent 若已掌握垂域浏览器或 adapter 工作流,不必仅为将精确确认 -回传给它而新增 API 凭据路径。这是另一种执行接缝,不放松上文的金融提交前检查、 -账户级约束或原始证据要求。Core 不解释价格、不选平台、不恢复浏览器、不签名、不下单。 - -### 权威与用户入口 - -- `chat/actions/actions.json` 中原始 `operation.execute` 提案仍是确认、claim、 - 消费及结果的唯一存储。 -- 显式执行器形状为 `{kind: "agent_session", host_surface, thread_id, - revision: "agent-session-handoff-v0"}`。准备时核对原 registry 中精确的 - Goal/已注册 Agent/session 绑定,拒绝以模拟冒充真实执行。生命周期专用的 - `source_session_v1` registry 目前拒绝业务操作准备;本切片不绕过该 owner,也不为 - 替换的 instance 开启业务权限。 -- `operation_agent_handoff.ts` 管理准入、一次消费、对账绑定及有界 Inbox 注意力。 - Python 提供锁定的规范存储、原 registry 事实及现有生命周期保护,不另建决策源。 -- 复用现有 Lark prepare/deliver 与经认证的回调。确认成功后原提案保持 claimed, - 没有外部结果。回调重放、模拟器、卡片投递恢复均不得调用该 Agent 的浏览器或 adapter。 -- 原 manager Inbox 直接从规范操作投影定位信息,不复制审批记录。Turn-start hook - 将它计入 `agent_read_required`。已消费/未知结果义务先于未消费票据展示;每页 20 条 - 之外明确给出总数量、类型化 overflow 原因及下一条操作 ID,不静默丢弃工作;可按该 - ID 直接检查原操作,并用独立的 `operation_handoff_next_cursor` 通过 - `manager-inbox read --operation-cursor CURSOR` 逐页找回其余操作,即使前 20 条未知 - 结果长期未解决。游标绑定原 runtime/Goal/Agent 范围,与普通请求游标独立;新增或 - 改变的工作应无游标重读,遍历结束不代表义务已解决。 -- Dashboard 详情与原 Lark 卡使用同一操作 frame:已确认但原宿主认证尚未接通、已消费待真实 - 证据、未知须对账不得重提,以及独立核验的结果。Dashboard 对用户操作确认保持只读。 - 不新增配置权威:执行器由原请求选择,原渠道/绑定 owner 仍是权威。 - -### 原运行时 CLI - -使用原 registry 和 runtime,不复制 session,也不借用另一个 home 的记录。 -以下选择器保留为续接接口;由于尚未接通合格的宿主身份签发端,**三个公开 CLI 命令 -目前均拒绝执行**。报告该门禁不要求 Lark 已安装或可达;新卡片准备/投递仍须通过原 -扩展与经认证入口的检查,但用户确认不能消除此宿主门禁。 - -```sh -loopx --registry REGISTRY --runtime-root RUNTIME goal-channel inspect-operation \ - --goal-id GOAL --agent-id AGENT --proposal-id OPERATION \ - --host-surface HOST --thread-id ORIGINAL_THREAD -loopx --registry REGISTRY --runtime-root RUNTIME goal-channel consume-operation \ - --goal-id GOAL --agent-id AGENT --proposal-id OPERATION \ - --host-surface HOST --thread-id ORIGINAL_THREAD \ - --consumption-id STABLE_ATTEMPT --execute -loopx --registry REGISTRY --runtime-root RUNTIME goal-channel report-operation \ - --goal-id GOAL --agent-id AGENT --proposal-id OPERATION \ - --host-surface HOST --thread-id ORIGINAL_THREAD --outcome-json OUTCOME --execute +## 13. 用户确认后的 Agent 执行:来源与执行者分离 + +原对话提供上下文及结果返回的受众,不必同时成为执行进程。长期契约是 +**用户批准 → 受管执行准入 → 单次消费 → 原系统证据 → 核验回传**,不是“Desktop +线程 ID 就是执行令牌”。已有垂域 Agent 可复用其已授权工作流,显式配置的 delegation +也可由 LoopX 自有受管 Turn 执行;两条路径都不增加账户访问、交易或宽泛写权限。 + +### 单一权威与不可变执行主体 + +- `chat/actions/actions.json` 中原始 `operation.execute` 仍是确认、claim、 + 消费、结果和对账的唯一存储。 +- 原外接主体保持 + `{kind: "agent_session", host_surface, thread_id, revision: "agent-session-handoff-v0"}`; + 其公开 CLI 认证门禁仍关闭。 +- 新增显式 opt-in 主体为 + `{kind: "managed_turn", todo_id, session_id, profile_digest, model, reasoning_effort, revision: "managed-turn-handoff-v0"}`。 + 它指向既有 Codex Turn session owner,不另建运行时目录。准备时核对已注册 + Goal/Agent、精确 Todo/session、适用的 Goal instance、传输与固定 profile。 +- `source_route` 仅从既有登记对话绑定投影,可为空;它是不可变的上下文/返回信息, + 不是调用者身份、执行许可或消息已送达证明。 +- TypeScript 管理执行器归一化、绑定判断、传输配置读回、单次准入、恢复和展示; + Python 只承载原生进程、session/存储锁及 Lark IO。不新增平行 Python 审批或通用决策源。 +- 旧外接批准不转换为受管批准。session、Todo、模型、思考深度、sandbox、workspace、 + home、可执行文件或 invocation-scoped MCP 配置改变时,必须显式开启 fresh + iteration 并重新批准。不复制轨迹、SQLite 行或凭据。 + +生命周期专用 `source_session_v1` registry 仍拒绝业务操作准备;本切片不启用替换 +Goal instance,不改变 provider 权威,也不把旧操作移植到另一运行时。 + +### 既有 delegation 与 Turn 入口 + +启动许可仍归原 operator-owned delegation 配置。Codex binding 通过既有 +`host_args` 显式启用: + +```text +--host codex-cli --codex-operation-tools +--codex-model MODEL --codex-reasoning-effort EFFORT +--codex-sandbox read-only ``` -`inspect-operation` 和没有 `--execute` 的命令均不消费授权。旧环境线程检查 -(例如 `CODEX_THREAD_ID`)可被另一同用户进程伪造,现已移除,而非升级成认证: -即使环境/路由完全匹配,仍在读取私有操作条款、结果文件或写回前返回 -`operation_host_authentication_unavailable`。旧运行时意外返回 actor 成功也不能 -绕过 CLI adapter 尚未接通的认证传输;没有 `--verified`、自行签发命令或环境令牌兜底。 - -Inbox 继续展示有界定位信息,保留已消费/未知结果义务;它明确要求宿主接入,不再建议 -重试被阻断的 CLI 消费。Dashboard/Lark 共享 frame 明确说明用户确认已记录,但原 -宿主身份认证尚未接通。这项收紧消除了公开环境伪造路径,**并未交付经认证的正向执行 -路径**。“原会话专属执行”因此仍是验收阻塞。内部锁定存储 adapter 与合成夹具只验证 -协议语义,不证明宿主签发端或真实最小闭环。registry 绑定授权路由,不认证调用者。 - -### 必需的宿主 adapter 接入 - -选定边界是**由传输拥有、不可导出的操作工具**。宿主在原 session 的认证工具连接上 -处理 `loopx_operation`(`inspect`、`consume`、`report`),并把回执返回同一连接。 -这是必需的配套契约,不是已安装工具、已接受的 proof 字段或新的审批存储。 - -1. 原配置 owner 在既有 session binding 上登记和撤销宿主签发者。请求不能自行选择 - 信任公钥或登记替代签发者;密钥轮换不改变不可变执行器,也不继承未消费授权。 -2. 宿主从原生工具分发元数据取得 session/Turn 身份,不使用工具参数、环境变量、 - MCP 子进程自报或 Agent 可读的密钥文件。不向模型/CLI 暴露通用签名器或可复用 - bearer token;私钥留在独立受信的宿主服务中,同用户环境伪造不得到达其身份或 - 签名权威。 -3. 跨进程时,签发者用 Ed25519 签署规范 invocation,绑定签发者/密钥 revision、 - 原 GoalRef/注册 Agent、host/session/Turn、原 operation 及载荷/确认摘要、 - action/参数摘要、audience/runtime、认证连接、有界签发/到期时间及唯一 request ID。 - Core 在 TypeScript 中核验 owner 固定的签发者、签名、精确范围、当前绑定与时效, - 再调用既有锁定 IO 接缝。认证响应仅返回原宿主连接;向公开 CLI 转发签名载荷不能 - 获得执行许可。 -4. 认证只证明来源。原用户确认、不可变条款、有效 Goal、到期与原子一次消费仍是 - 独立门禁;重放或未知提交不允许第二次外部操作。恢复宿主也须使用自己的认证连接, - 仅按既有接手恢复规则回写证据。 - -**具体依赖:**外接 Codex Desktop session 需要 Desktop 原生工具服务在既有 app -自有工具旁提供 session-bound 操作工具,并接入 owner 控制的签发者登记。该原生 -服务不在这个 LoopX checkout 内,现有 CLI/MCP adapter 也未实现它。宿主维护者 -须交付签发端;本 PR 不能用环境身份替代,也不能在另一进程悄悄恢复原 session。 -`CodexChatAgentSession._check_server_gate` 已核对 LoopX 自建 app-server 工具 -调用的 thread/Turn 元数据,但那是另一类自有 session,不证明外接 Desktop 线程。 -自有宿主接入须独立验收自己的路由,不能替代 Desktop 验收。 - -真实签发端与 owner 固定的验签端应在同一后续切片接通,不交付无人使用的签名 API -或夹具生成的凭据。验收须覆盖伪造环境/路由/proof 字段、外来 session、错误 audience、 -到期、篡改、重放、签发者/绑定撤销及原连接回执返回;启用宿主工具时保留当前公开 CLI -拒绝回归。在同一边界通过真实原会话调用前不得安装本切片。工程 QA 不创建新 session、 -复制轨迹、制造群确认或触发真实金融副作用。 - -### 宿主门禁后的单次消费与证据语义 - -只有首次成功的原子消费 -返回 `execution_allowed: true`。它检查经认证的确认、不可变条款、原 session 当前绑定、 -有效 Goal 与到期时间,并在任何浏览器操作之前持久记录消费。所有重试,包括响应丢失 -或重启后使用同一 attempt,均不再获得执行许可。这不承诺平台恰好执行一次;存在歧义 -必须按原平台证据对账。 -绑定/注册/启用状态读取和消费提交共享既有 registry writer 锁,顺序为 Goal 生命周期 -→ registry → action store。先提交的撤销阻止消费;排在消费之后的撤销不能追溯抹除 -已提交的回执。锁在外部执行前释放,不宣称约束其后的外部操作。 - -原 Agent 回写 `loopx_operation_outcome_v0`,绑定精确 operation、载荷与确认摘要、 -claim、执行器 revision、consumption ID,要求 `projection_verified: true`、 -`simulation: false`、有界原始证据引用,并单独声明 `external_write_performed`。 -结果为 `executed`、`not_executed` 或 `submission_unknown`。未知保守声明可能存在外部 -副作用;传输成功不能证明交易或保护单。引用应为安全的不透明回执标识,不能含凭据或 -私有绝对路径;垂域证据及私有交易日记保留详细材料。 - -### 恢复、投递与剩余验收 - -已消费或未知操作在到期或 session 重绑后仍是恢复义务;这些变化不能授予新执行。 -仅回写证据可通过既有生命周期保护继续用于已停止/历史 Goal。现有精确 instance -生命周期保护保持不变,不隐式迁入生命周期专用 registry 或另一个 home。 -原 Goal/Agent 注册缺失或 registry profile 不受支持均明确报错,不允许移植操作权限。 - -原路由撤销后,**同 Goal、同 Agent、当前已注册并绑定的接手会话**在认证传输验收后 -可以用自己的宿主路由检查与回写。内部 IO 接缝只能检查已消费操作,回写 -原系统的历史证据,不能消费未使用的票据、改写原执行器或取得第二次执行许可。原绑定 -仍有效、接手会话未绑定、Goal/Agent 不同均拒绝恢复准入。协议明确返回 `access.owner`、 -`original_route`、`permission: "historical_evidence_only"` 与绑定权威,不要求接手者 -冒充旧线程。原执行路由保持不可变;从接手绑定核对到结果提交共用 registry 锁,先提交 -的撤销拒绝回写。 - -结果证据本身保持原样。规范操作单独追加 `outcome_report` 或 `reconciliation_report` -来源,记录实际报告者、原路由、仅历史证据权限、权威来源、consumption ID 与记录时间。 -相同结果重试保留首次已提交来源,不重标作者或授予执行。内部检查同时返回原未知 -outcome 与不可变对账/来源;既有 Dashboard 共享 frame 与原 Lark 卡恢复消费同一 -规范结果,不另建恢复审批存储或执行控件。因公开宿主门禁关闭,旧正向 CLI 夹具已改为 -内部 IO/原卡读回检查;它们只证明历史对账语义,不代替受信宿主认证或真实群验收。 - -未知的原始 outcome 不可修改。确定性回写追加 `operation.reconciliation`,并通过 -`reconciles_outcome_digest` 绑定原未知结果的精确摘要;只有该证据才关闭恢复义务。 -现有投递恢复更新原结果卡,旧未知结果的投递不能证明新的对账结果;读回必须匹配当前 -`initial` 或 `reconciled` 阶段。上述路径均不会重新提交。 - -本切片**没有实现主机即时唤醒**。Inbox 可见性如实返回 -`host_delivery: "not_attempted"`;既有 Turn/heartbeat 读取不等于主机投递回执。 -后续主机唤醒应复用原宿主传输,给出真实尝试/读回证据,不启动平行的 resumed session。 -本地合成回调、CLI、并发、到期、对账及打包 UI 检查只证明协议行为。宣称真实最小闭环 -之前,仍须安装、真实群确认、原 Agent 消费回执、真实平台/保护证据及原卡片读回。 -不得向真实群发送合成工程卡来冒充验收。 +相同选项可用于 `turn run-once`。复用既有 delegation inspect/preflight、start、 +准入、租约、session 续接、结果验证和验收。原 host owner 读回 profile; +不新增前端配置存储或隐藏默认。不固定或不受支持的配置在预检中标为 unavailable; +有效 argv 仍是 runtime-unverified,不能证明宿主或操作已运行。仅在现有工作授权 +确需时使用 `workspace-write`;此传输不接受 `danger-full-access`。 + +已准入宿主复用 `CodexChatAgentSession` app-server adapter,将不透明线程保存于 +既有 Goal/Agent/Todo session owner,安装不可导出的 `loopx_operation` dynamic tool。 +自有 stdio 连接在分发前核对原生 thread 与活跃 Turn 元数据;工具参数不能传 actor、 +verified、信任密钥、签名或 bearer token。回执返回同一原生连接,不恢复或冒充 +登记的来源 Desktop 线程。 + +`context` 读回精确受管主体但不给执行许可;`prepare` 在规范 action store 预览 +不可变条款;`pending` 读取有界 Inbox;`inspect/consume/report` 复用锁定操作接缝。 +保留 invocation-scoped collaboration MCP 的普通委派能力,它不是操作身份签发端。 +宿主结果沿用 typed Turn result 及验证;最终答复文字不能冒充操作结果回执。 + +首个传输是 Codex 专用 IO,不是 Codex 专用审批模型。其他受管宿主只有在原生身份 +来源、实际 profile 读回、撤销和响应路由通过验收后才能实现同一契约。外接 Desktop +仍是独立可选 adapter,**不是受管路径的前置依赖**。未来远端边界可能需要 owner +登记的认证 invocation 核验,但不应让无人消费的签名器或本地伪造凭据成为自有进程 +分发的前置条件。 + +### 确认、单次消费与恢复 + +既有经认证 Lark 回调记录精确用户确认并 claim 原提案,不启动浏览器或垂域 adapter。 +重放、模拟及卡片投递恢复都不产生垂域效果;Dashboard 对用户操作确认仍只读。 + +只有首次成功原子消费返回 `execution_allowed: true`,核对经认证的确认、不可变 +条款、当前执行绑定、有效 Goal 和到期时间,在 Agent 垂域执行前持久记录消费。 +所有重试,包括丢失响应或重启后的同一 attempt,都不再授予执行。 +它约束的是**授权消费**,不宣称能禁止可信 Agent 的所有工具调用,也不承诺平台恰好执行一次。 + +锁序是 Goal 生命周期 → registry → 既有 Turn session(仅受管路径)→ action store。 +session 替换/丢弃与消费共用锁。先提交的撤销阻止消费,之后的撤销不能抹去已提交回执; +外部执行期间不持锁。 + +绑定宿主回写 `loopx_operation_outcome_v0`:精确 operation/载荷/确认摘要、claim、 +executor revision、consumption ID、核验投影、`simulation: false`、 +有界原系统证据引用及单独的 `external_write_performed`。结果为 `executed`、 +`not_executed` 或 `submission_unknown`;未知保守披露可能的外部副作用。 +原生传输成功不等于平台证据;Core 不解释价格、平台、费用、持仓或保护单。 + +已消费/未知义务在到期或原绑定撤销后仍保留。同 Goal/Agent 当前接手者只有在原绑定 +撤销后才能检查/回写历史证据,不能消费未用票据或重写执行器。原 outcome 不可变, +对账以 `reconciles_outcome_digest` 绑定并追加实际报告者/路由来源,读回保留原始与 +对账证据。已停止/历史 Goal 沿用证据专用生命周期保护,不因此获得新的执行许可。 + +### 共享 Inbox、前端、Lark 与真实回传 + +既有 Inbox 直接投影规范定位信息:恢复优先、每页 20 条、总数/类型化 overflow +及独立 operation cursor。`loopx_operation pending` 接受绑定游标;CLI 投影用 +`manager-inbox read --operation-cursor CURSOR`。新增/改变工作应无游标重读, +读完一页或遍历结束不代表义务已解决。 + +共享 TS 操作 frame 展示执行者、固定模型/思考深度、Goal/Agent/Todo 范围、可选来源 +上下文,并区分:已确认但外接认证不可用、已确认待绑定受管回合、已消费待证据、未知须 +对账,以及独立核验的结果投递。CLI、Dashboard 和原 Lark 卡消费同一 frame。 +结果读回必须匹配当前 `initial/reconciled` 阶段;旧未知结果的投递不能证明新对账结果。 +现有投递恢复更新原卡,不重提操作。delegation 结果验收及 requester 采用仍是独立 +回执,不新增自动 chat 回传协议。 + +### 资格化与剩余交付 + +本切片验证自有进程原生工具分发(含新 Turn 与同线程恢复 Turn 的有界真实 Codex +context 调用,结果均由既有 typed result validator 接受)、规范 +批准/消费/结果夹具、profile/session 撤销、原路由隔离、delegation 预检和打包展示。 +真实 context 探针只在当前拥有的 home 创建新的受管 GPT session,不建提案、 +不造群确认、不执行金融副作用;合成批准夹具不是真实用户批准。 + +公开 `goal-channel inspect-operation/consume-operation/report-operation` 仍在 +私有读写前拒绝,即使 `CODEX_THREAD_ID`、路由、自签 proof 完全匹配或旧运行时 +意外返回 actor 成功,也没有 proof-import 捷径。 + +本切片不实现确认后的即时宿主唤醒,`host_delivery: "not_attempted"` 保持真实。 +沿既有已准入 Turn/delegation 续接;后续持久唤醒复用其调度/session owner, +不启动平行 resumed 执行者。宣称投研最小闭环前,仍须证明安装、真实用户批准、 +绑定原生消费、垂域提交前检查与原系统证据、结果验收及原卡/受众读回。 +Core PR 仍须 owner review,不在合并前自行安装。 diff --git a/docs/architecture/rfcs/loopx-overall-roadmap-v0.md b/docs/architecture/rfcs/loopx-overall-roadmap-v0.md index fd8238b38f..9857b803c0 100644 --- a/docs/architecture/rfcs/loopx-overall-roadmap-v0.md +++ b/docs/architecture/rfcs/loopx-overall-roadmap-v0.md @@ -326,7 +326,7 @@ subsystem was not performed. Section 8 records the focused audit. | [Per-Goal Usage, Token, and Cost Surfacing v0](goal-usage-token-cost-v0.md) | S7/S5 | Accepted; Codex aggregate/cost display slice exists | P0 observation→P1 provider coverage: unknown is not zero, deduplicate accounting, price source/freshness; usage grants no budget | | [Intelligent Review and Dynamic Presentation Surfaces v0](intelligent-review-presentation-surfaces-v0.md) | S5 | Accepted; action/attention verticals and local delivery-chain/acceptance review implemented | P1: cross-channel disclosure and governed amendment/settlement review; local visibility does not qualify G2 | | [Human Attention Wishlist v0](human-attention-wishlist-v0.md) | S5/S11 | Accepted; Held | P3: reopen only on repeated second real need; sidecar cannot alter gates/quota/scheduling | -| [Human-confirmed domain operations (v0)](human-confirmed-domain-operations-v0.md) | S8/S9 | Accepted; proposal only | P2: simulated immutable confirmation→effect→reconciliation→return; finance provider separate, no broader coordination grant | +| [Human-confirmed domain operations (v0)](human-confirmed-domain-operations-v0.md) | S8/S9 + R2/R3 | Accepted; canonical operation seam exists, managed native transport under owner review | Qualify source-context vs admitted-executor separation, exact human approval→one-shot consumption→domain evidence→original-route return; owned Turn/delegation need not await Desktop authentication. Live approval/effect/wakeup remain unqualified; finance provider stays separate | | [Provider-side authorization at effect acceptance (v0)](provider-effect-acceptance-v0.md) | S8/S9, supporting S2/S4 | Accepted design only; no runtime integration or qualified provider | M1: controlled provider and deterministic revoke/crash/replay conformance; strict production wiring remains gated by exact Goal lifetime, receipt retention and independent provider qualification | | [Research Exploration Control Plane v0](research-exploration-control-plane-v0.md) | S11/S3 | Accepted; partial M2 composition/successor | P1: independently verify observation/write-time gate/closure basis; defer inferred triggers and model selection | | [Hierarchical Agent Stride Control v0](hierarchical-agent-stride-control-v0.md) | S11/S7 | Accepted; M1 read-only observation | P2: matched shadow stride experiment with costs/events; no direct production cadence change | diff --git a/docs/architecture/rfcs/loopx-overall-roadmap-v0.zh-CN.md b/docs/architecture/rfcs/loopx-overall-roadmap-v0.zh-CN.md index 960510fb01..3ecd7dcb78 100644 --- a/docs/architecture/rfcs/loopx-overall-roadmap-v0.zh-CN.md +++ b/docs/architecture/rfcs/loopx-overall-roadmap-v0.zh-CN.md @@ -263,7 +263,7 @@ Muse 设计页在浏览器超时,其文章通过网页检索读取。本次调 | [Per-Goal Usage, Token, and Cost Surfacing v0](goal-usage-token-cost-v0.md) | S7/S5 | 已接受;Codex aggregate/cost 展示已有切片 | P0 观测→P1 多 provider:未知不作零、重复扣费去重、价格来源/时效;usage 不自动授权预算 | | [Intelligent Review and Dynamic Presentation Surfaces v0](intelligent-review-presentation-surfaces-v0.zh-CN.md) | S5 | 已接受;action/attention 纵切及本地交付链/验收复盘已实现 | P1:跨渠道披露和受治理的修订/结算复盘;本地可见性不代表 G2 通过 | | [Human Attention Wishlist v0](human-attention-wishlist-v0.zh-CN.md) | S5/S11 | 已接受;Held | P3:第二个重复真实需求出现才重开;sidecar 不改变 gate/quota/调度 | -| [Human-confirmed domain operations (v0)](human-confirmed-domain-operations-v0.zh-CN.md) | S8/S9 | 已接受;proposal only | P2:模拟 adapter 的一次不可变确认→effect→对账→原路回报;金融 provider 独立包,不扩普通协调权限 | +| [Human-confirmed domain operations (v0)](human-confirmed-domain-operations-v0.zh-CN.md) | S8/S9 + R2/R3 | 已接受;规范操作接缝存在,受管原生传输待 owner review | 验收来源上下文与已准入执行者分离、精确用户批准→单次消费→垂域证据→原路返回;自有 Turn/delegation 不等待 Desktop 认证。真实批准/效果/唤醒仍未资格化;金融 provider 保持独立 | | [Provider 在效果接受点执行授权(v0)](provider-effect-acceptance-v0.zh-CN.md) | S8/S9,S2/S4 支撑 | 已接受设计;尚未接入 runtime,也未准入 provider | M1:controlled provider 与 deterministic revoke/crash/replay conformance;strict production 接线仍需精确 Goal 生命周期、receipt retention 与独立 provider 资格 | | [Research Exploration Control Plane v0](research-exploration-control-plane-v0.zh-CN.md) | S11/S3 | 已接受;M2 composition/successor 局部实现 | P1:observation/write-time gate/closure basis 独立验证;自选模型和推断触发继续 defer | | [Hierarchical Agent Stride Control v0](hierarchical-agent-stride-control-v0.zh-CN.md) | S11/S7 | 已接受;M1 只读观测 | P2:matched shadow stride 实验,定义代价与事件;不直接改变生产节奏 | From 876a507b54d9516871a4e78c4aae2f9bb9862085 Mon Sep 17 00:00:00 2001 From: huangruiteng <14976749+huangruiteng@users.noreply.github.com> Date: Wed, 30 Sep 2026 13:36:21 +0800 Subject: [PATCH 11/13] fix(operations): scope native pending discovery and cursor identity Signed-off-by: huangruiteng <14976749+huangruiteng@users.noreply.github.com> --- .../turn_driver/codex_operation_host.py | 36 +++++++---- tests/test_chat_operation_actions.py | 64 +++++++++++++++++++ 2 files changed, 89 insertions(+), 11 deletions(-) diff --git a/loopx/control_plane/turn_driver/codex_operation_host.py b/loopx/control_plane/turn_driver/codex_operation_host.py index 38e15ee544..783c10f436 100644 --- a/loopx/control_plane/turn_driver/codex_operation_host.py +++ b/loopx/control_plane/turn_driver/codex_operation_host.py @@ -18,6 +18,7 @@ from ...chat_agent import CodexChatAgentError, CodexChatAgentSession from ...chat_action_store import ChatActionStore from ...chat_actions import ChatActionService +from ..collaboration.goal_instance_scope import collaboration_goal_scope from ..collaboration.operation_handoff import ( agent_operation_action, pending_operation_handoffs, @@ -126,17 +127,30 @@ def handle(tool: str, arguments: Any, native: dict[str, Any]) -> dict[str, Any]: "external_write_performed": False, } if action == "pending": - return { - "ok": True, - **pending_operation_handoffs( - runtime_root, - lineage["goal_id"], - lineage["agent_id"], - registry_path=registry_path, - cursor=arguments.get("cursor"), - cursor_scope="managed-operation:" + session_id, - ), - } + with collaboration_goal_scope( + registry_path, + goal_id=lineage["goal_id"], + agents=(lineage["agent_id"],), + ) as scope: + cursor_scope = hashlib.sha256( + json.dumps( + ["managed-operation-v0", str(runtime_root.resolve()), + scope.target(lineage["agent_id"]), executor], + sort_keys=True, separators=(",", ":"), + ).encode() + ).hexdigest() + return { + "ok": True, + **pending_operation_handoffs( + runtime_root, + lineage["goal_id"], + lineage["agent_id"], + registry_path=registry_path, + scope=scope, + cursor=arguments.get("cursor"), + cursor_scope=cursor_scope, + ), + } if action == "prepare": request = dict(arguments["request"]) terms = dict(request.get("normalized_parameters") or {}) diff --git a/tests/test_chat_operation_actions.py b/tests/test_chat_operation_actions.py index 00d03850b1..c1e3e16b0d 100644 --- a/tests/test_chat_operation_actions.py +++ b/tests/test_chat_operation_actions.py @@ -170,6 +170,70 @@ def test_owned_managed_tool_uses_canonical_approval_once_without_desktop_binding assert recovered["ok"] is True and recovered["needs_reconciliation"] is False +def test_managed_pending_reuses_registered_agent_and_goal_instance_scope( + tmp_path: Path, +) -> None: + service, store = _service(tmp_path) + handler = _managed_handler(service, store) + native = {"thread_id": "owned-managed-thread", "host_turn_id": "native-turn-1"} + request = _request() + request["normalized_parameters"].pop("executor") + request["normalized_parameters"]["projection"]["simulated"] = False + proposal = handler( + "loopx_operation", {"action": "prepare", "request": request}, native + )["proposal"] + delivered = store.record_operation_delivery( + proposal["proposal_id"], delivery=_delivery(proposal) + ) + store.decide_operation( + proposal["proposal_id"], decision="confirm", confirmation=_confirmation(delivered) + ) + consumed = handler( + "loopx_operation", + { + "action": "consume", + "proposal_id": proposal["proposal_id"], + "consumption_id": "managed-attempt", + }, + native, + ) + assert consumed["execution_allowed"] is True + args = {"action": "pending"} + assert ( + handler("loopx_operation", args, native)["items"][0]["operation_id"] + == proposal["proposal_id"] + ) + + registry = json.loads(service.registry_path.read_text()) + registry["goals"][0]["activation_state"] = "stopped" + service.registry_path.write_text(json.dumps(registry)) + # Stopping execution does not erase the original evidence-reconciliation locator. + historical = handler("loopx_operation", args, native) + assert historical["ok"] is True and historical["items"][0]["needs_reconciliation"] + assert historical["items"][0]["execution_allowed"] is False + + registry.update( + profile_id="source_session_v1", + session_bindings=[], + session_receipts=[], + lifetime_receipts=[], + ) + registry["goals"][0]["goal_instance_id"] = "ginst_aaaaaaaaaaaaaaaaaaaaaaaaaaaaaaaa" + service.registry_path.write_text(json.dumps(registry)) + # A current exact-instance Inbox must not expose legacy, unbound locators. + scoped = handler("loopx_operation", args, native) + assert scoped["ok"] is True and scoped["items"] == [] + + registry["goals"][0]["coordination"]["registered_agents"] = [] + service.registry_path.write_text(json.dumps(registry)) + rejected = handler("loopx_operation", args, native) + assert rejected == { + "ok": False, + "error": "operation_admission_rejected", + "execution_allowed": False, + } + + def test_managed_replacement_has_evidence_only_access_and_never_inherits_execution( tmp_path: Path, monkeypatch: pytest.MonkeyPatch, From 2d912cfe2a9225c19a3febf514594f37943e1a42 Mon Sep 17 00:00:00 2001 From: huangruiteng <14976749+huangruiteng@users.noreply.github.com> Date: Wed, 30 Sep 2026 15:28:09 +0800 Subject: [PATCH 12/13] fix(operation): keep packaged fixtures runtime-only and reuse digest owner Signed-off-by: huangruiteng <14976749+huangruiteng@users.noreply.github.com> --- examples/operation_action_fixtures.py | 209 ++++++++++++++ .../confirmed-operation-fixtures.py | 18 +- .../confirmed-operations.mjs | 8 +- .../work_items/operation_agent_handoff.ts | 6 +- .../content_digest_single_owner.test.ts | 1 + tests/test_chat_operation_actions.py | 270 +++++------------- 6 files changed, 305 insertions(+), 207 deletions(-) create mode 100644 examples/operation_action_fixtures.py diff --git a/examples/operation_action_fixtures.py b/examples/operation_action_fixtures.py new file mode 100644 index 0000000000..02197a71da --- /dev/null +++ b/examples/operation_action_fixtures.py @@ -0,0 +1,209 @@ +"""Synthetic canonical-operation fixtures shared by examples and unit tests. + +Only stdlib and installed LoopX runtime imports belong here. No test framework, +real channel, account, credential or executor is required by the browser smoke. +Policy remains in the existing TypeScript owners; these are storage/wire inputs. +""" + +from __future__ import annotations + +from datetime import datetime, timedelta, timezone +import hashlib +import json +from pathlib import Path + +from loopx.chat_action_store import ChatActionStore +from loopx.chat_actions import ChatActionService + +GOAL_ID = "goal-operation-fixture" +OPERATOR_ID = "ou_authorized_fixture" + + +def digest(value: object) -> str: + encoded = json.dumps( + value, ensure_ascii=False, sort_keys=True, separators=(",", ":") + ).encode() + return hashlib.sha256(encoded).hexdigest() + + +def service( + tmp_path: Path, *, goal_id: str = GOAL_ID +) -> tuple[ChatActionService, ChatActionStore]: + project = tmp_path / "project" + project.mkdir() + (project / "ACTIVE_GOAL_STATE.md").write_text( + f"---\ngoal_id: {goal_id}\n---\n\n## User Todo\n\n## Agent Todo\n", + encoding="utf-8", + ) + registry = project / ".loopx" / "registry.json" + registry.parent.mkdir() + registry.write_text( + json.dumps( + { + "goals": [ + { + "id": goal_id, + "repo": str(project), + "state_file": "ACTIVE_GOAL_STATE.md", + "coordination": { + "registered_agents": ["finance-fixture-agent"] + }, + } + ] + } + ), + encoding="utf-8", + ) + store = ChatActionStore(tmp_path / "runtime" / "chat" / "actions") + return ChatActionService(store=store, registry_path=registry), store + + +def managed_handler( + service: ChatActionService, + store: ChatActionStore, + *, + goal_id=GOAL_ID, + session_id="owned-managed-thread", + profile_digest="c" * 64, + todo_id="todo-managed", + model="test-model", + reasoning_effort="xhigh", +): + from loopx.control_plane.turn_driver.codex_cli import _store_codex_cli_session + from loopx.control_plane.turn_driver.codex_operation_host import ( + operation_tool_handler, + ) + + lineage = { + "goal_id": goal_id, + "agent_id": "finance-fixture-agent", + "todo_id": todo_id, + } + runtime = store.root.parent.parent + _store_codex_cli_session( + runtime, + lineage=lineage, + session_id=session_id, + operation_profile_digest=profile_digest, + operation_model=model, + operation_reasoning_effort=reasoning_effort, + ) + return operation_tool_handler( + runtime_root=runtime, + registry_path=service.registry_path, + lineage=lineage, + session_id=session_id, + profile_digest=profile_digest, + model=model, + reasoning_effort=reasoning_effort, + ) + + +def request( + *, payload: dict[str, object] | None = None, goal_id: str = GOAL_ID +) -> dict[str, object]: + operation_payload = payload or { + "schema_version": "finance_order_intent_v0", + "side": "buy", + "asset": "SYNTH", + "quantity": "1.00", + "order_type": "limit", + "limit_price": "10.00", + "time_in_force": "GTC", + "reduce_only": False, + } + return { + "action_kind": "operation.execute", + "summary": "Confirm one simulated finance order", + "idempotency_key": "operation-fixture-v1", + "context": {"kind": "goal", "goal_id": goal_id}, + "normalized_parameters": { + "schema_version": "loopx_operation_request_v0", + "goal_id": goal_id, + "agent_id": "finance-fixture-agent", + "domain": "finance", + "operation_kind": "finance.order.simulate", + "operation_schema": "finance_order_intent_v0", + "payload_ref": "finance-order:synthetic-1", + "payload": operation_payload, + "payload_digest": digest(operation_payload), + "projection": { + "schema_version": "loopx_operation_projection_v0", + "title": "Simulated trade request", + "subtitle": "Synthetic fixture · no venue call", + "focus": "BUY 1.00 SYNTH @ 10.00", + "fields": [ + {"label": "Order type", "value": "Limit · GTC"}, + {"label": "Maximum notional", "value": "10.00 TEST"}, + ], + "warning": "Simulation only. This cannot submit, sign, or transfer.", + "simulated": True, + }, + "destination_account_ref": "account:simulation", + "expires_at": (datetime.now(timezone.utc) + timedelta(hours=1)).isoformat(), + "authorized_principals": [f"lark:{OPERATOR_ID}"], + "executor": { + "extension_id": "loopx-finance-execution", + "protocol": "finance_operation_executor_v0", + "permission": "finance.operation.simulate", + "revision": "simulator-v0", + }, + }, + } + + +def delivery(proposal: dict[str, object]) -> dict[str, str]: + assert isinstance(proposal["operation"], dict) + return { + "provider": "lark", + "message_id": "om_operation_fixture", + "chat_id": "oc_operation_fixture", + "app_id": "cli_operation_fixture", + "binding_digest": "a" * 64, + "card_digest": "b" * 64, + "delivered_at": datetime.now(timezone.utc).isoformat(), + } + + +def confirmation( + proposal: dict[str, object], *, event_id: str = "evt-operation-1" +) -> dict[str, str]: + operation = proposal["operation"] + assert isinstance(operation, dict) + delivered = operation["delivery"] + assert isinstance(delivered, dict) + return { + "provider": "lark", + "event_id": event_id, + "principal": f"lark:{OPERATOR_ID}", + "message_id": str(delivered["message_id"]), + "chat_id": str(delivered["chat_id"]), + "app_id": str(delivered["app_id"]), + "surface_kind": "group_message_card", + "interaction_kind": "button_callback", + "confirmation_digest": str(operation["confirmation_digest"]), + "card_digest": str(delivered["card_digest"]), + "confirmed_at": datetime.now(timezone.utc).isoformat(), + } + + +def agent_result( + proposal: dict, consumption_id: str, *, result: str = "executed" +) -> dict: + operation = proposal["operation"] + return { + "schema_version": "loopx_operation_outcome_v0", + "operation_id": proposal["proposal_id"], + "payload_digest": operation["payload_digest"], + "confirmation_digest": operation["confirmation_digest"], + "claim_id": operation["claim"]["claim_id"], + "executor_revision": operation["executor_revision"], + "consumption_id": consumption_id, + "outcome": result, + "projection_verified": True, + "simulation": False, + "external_write_performed": result != "not_executed", + "evidence_refs": ["receipt:synthetic-fixture-1"], + "summary": "Synthetic recorded execution evidence.", + "observed_at": datetime.now(timezone.utc).isoformat(), + } diff --git a/examples/personal-workspace-browser/confirmed-operation-fixtures.py b/examples/personal-workspace-browser/confirmed-operation-fixtures.py index 58fe521c6f..7cf94021dd 100644 --- a/examples/personal-workspace-browser/confirmed-operation-fixtures.py +++ b/examples/personal-workspace-browser/confirmed-operation-fixtures.py @@ -7,20 +7,20 @@ import sys from tempfile import TemporaryDirectory -sys.path.insert(0, str(Path(__file__).resolve().parents[2] / "tests")) -import test_chat_operation_actions as fixtures # noqa: E402 +sys.path.insert(0, str(Path(__file__).resolve().parents[1])) +import operation_action_fixtures as fixtures # noqa: E402 def main() -> None: - fixtures.GOAL_ID = "product-release" + goal_id = "product-release" with TemporaryDirectory(prefix="loopx-operation-ui-") as root: - service, store = fixtures._service(Path(root)) - handler = fixtures._managed_handler(service, store) + service, store = fixtures.service(Path(root), goal_id=goal_id) + handler = fixtures.managed_handler(service, store, goal_id=goal_id) native = { "thread_id": "owned-managed-thread", "host_turn_id": "fixture-native-turn", } - request = fixtures._request() + request = fixtures.request(goal_id=goal_id) terms = request["normalized_parameters"] terms.pop("executor") terms["projection"].update( @@ -33,12 +33,12 @@ def main() -> None: "loopx_operation", {"action": "prepare", "request": request}, native )["proposal"] delivered = store.record_operation_delivery( - proposal["proposal_id"], delivery=fixtures._delivery(proposal) + proposal["proposal_id"], delivery=fixtures.delivery(proposal) ) confirmed = store.decide_operation( proposal["proposal_id"], decision="confirm", - confirmation=fixtures._confirmation(delivered), + confirmation=fixtures.confirmation(delivered), ) consumed = handler( "loopx_operation", @@ -51,7 +51,7 @@ def main() -> None: ) assert consumed["execution_allowed"] is True waiting = store.load(proposal["proposal_id"]) - unknown = fixtures._agent_result( + unknown = fixtures.agent_result( confirmed, "fixture-attempt", result="submission_unknown" ) reported = handler( diff --git a/examples/personal-workspace-browser/confirmed-operations.mjs b/examples/personal-workspace-browser/confirmed-operations.mjs index 7bed7259f4..2b0f25143c 100644 --- a/examples/personal-workspace-browser/confirmed-operations.mjs +++ b/examples/personal-workspace-browser/confirmed-operations.mjs @@ -8,9 +8,15 @@ import { openWorkspacePage } from "./scenario-context.mjs"; export const confirmedOperationsScenario = { id: "confirmed-operations", async run({ browser, url }) { + const fixtureEnv = { ...process.env }; + // An explicitly selected release interpreter must exercise its installed + // LoopX, not silently shadow the wheel with this source checkout. + if (!["LOOPX_TEST_PYTHON", "LOOPX_PYTHON_BIN", "LOOPX_PYTHON"].some(key => process.env[key])) { + fixtureEnv.PYTHONPATH = repoRoot; + } const fixtures = JSON.parse(execFileSync(resolveTestPython(), [ resolve(repoRoot, "examples/personal-workspace-browser/confirmed-operation-fixtures.py"), - ], { cwd: repoRoot, env: { ...process.env, PYTHONPATH: repoRoot }, encoding: "utf8" })); + ], { cwd: repoRoot, env: fixtureEnv, encoding: "utf8" })); const states = { confirmed: { en: "Confirmed; waiting for the bound managed Turn", "zh-CN": "已确认,等待绑定的受管回合" }, waiting: { en: "Authorization consumed; waiting for the real result", "zh-CN": "授权已消费,等待真实结果" }, diff --git a/loopx/control_plane/work_items/operation_agent_handoff.ts b/loopx/control_plane/work_items/operation_agent_handoff.ts index 287a6ec12e..5814501af5 100644 --- a/loopx/control_plane/work_items/operation_agent_handoff.ts +++ b/loopx/control_plane/work_items/operation_agent_handoff.ts @@ -2,13 +2,13 @@ * second approval store. Python supplies locked storage and registry facts; * this owner decides admission, one-shot consumption and result binding. */ import type {JsonObject} from "../effect_program.ts"; +import {BARE_SHA256_PATTERN} from "../content_digest.ts"; import {EffectRuntimeConflictError, EffectRuntimeRequestError} from "../effect_runtime_errors.ts"; import {requireJsonObject, requireNonEmptyString} from "../runtime_decode.ts"; export const AGENT_OPERATION_REVISION = "agent-session-handoff-v0"; export const MANAGED_OPERATION_REVISION = "managed-turn-handoff-v0"; const ID = /^[A-Za-z0-9._:-]{1,200}$/; -const SHA256 = /^[a-f0-9]{64}$/; function id(value: unknown, field: string): string { const result = requireNonEmptyString(value, field); @@ -48,7 +48,7 @@ export function normalizeAgentOperationExecutor(input: JsonObject): JsonObject { const keys = ["kind", "todo_id", "session_id", "profile_digest", "model", "reasoning_effort", "revision"]; if (Object.keys(executor).length !== keys.length || keys.some(key => !(key in executor)) || executor.revision !== MANAGED_OPERATION_REVISION - || typeof executor.profile_digest !== "string" || !SHA256.test(executor.profile_digest)) { + || typeof executor.profile_digest !== "string" || !BARE_SHA256_PATTERN.test(executor.profile_digest)) { throw new EffectRuntimeRequestError("managed operation executor binding is invalid"); } return {kind: "managed_turn", todo_id: id(executor.todo_id, "todo_id"), @@ -231,7 +231,7 @@ export function projectAgentOperationInbox(input: JsonObject): JsonObject { items.sort((a, b) => rank(a) - rank(b) || String(a.operation_id).localeCompare(String(b.operation_id), "en")); const scope = requireNonEmptyString(input.cursor_scope, "operation cursor scope"); - if (!/^[a-f0-9]{64}$/.test(scope)) throw new EffectRuntimeRequestError("operation cursor scope is invalid"); + if (!BARE_SHA256_PATTERN.test(scope)) throw new EffectRuntimeRequestError("operation cursor scope is invalid"); let remaining = items; if (input.cursor != null) { const cursor = requireNonEmptyString(input.cursor, "operation cursor"); diff --git a/tests/control_plane_ts/content_digest_single_owner.test.ts b/tests/control_plane_ts/content_digest_single_owner.test.ts index a8ea3796ae..47b3c0d4c0 100644 --- a/tests/control_plane_ts/content_digest_single_owner.test.ts +++ b/tests/control_plane_ts/content_digest_single_owner.test.ts @@ -120,6 +120,7 @@ const CANONICAL_CONSUMERS = [ "control_plane/todos/completion_transaction.ts", "control_plane/todos/completion_validation_revision.ts", "control_plane/turn_driver/chat_turn_acceptance.ts", + "control_plane/work_items/operation_agent_handoff.ts", "control_plane/work_items/pending_capability_intent.ts", "control_plane/work_items/replan_history_snapshot.ts", "control_plane/work_items/task_lease_acquire.ts", diff --git a/tests/test_chat_operation_actions.py b/tests/test_chat_operation_actions.py index c1e3e16b0d..3e9e0f06c3 100644 --- a/tests/test_chat_operation_actions.py +++ b/tests/test_chat_operation_actions.py @@ -1,8 +1,10 @@ from __future__ import annotations from datetime import datetime, timedelta, timezone -import hashlib import json +import os +import subprocess +import sys from concurrent.futures import ThreadPoolExecutor from pathlib import Path @@ -13,10 +15,18 @@ from loopx.control_plane.collaboration.operation_handoff import agent_operation_action from loopx.control_plane.collaboration.inbox import pending from loopx.capabilities.manager_context import turn_start_hook +from examples.operation_action_fixtures import ( + GOAL_ID, + agent_result as _agent_result, + confirmation as _confirmation, + delivery as _delivery, + digest as _digest, + managed_handler as _managed_handler, + request as _request, + service as _service, +) -GOAL_ID = "goal-operation-fixture" -OPERATOR_ID = "ou_authorized_fixture" EXECUTION_ACTOR = { "goal_id": GOAL_ID, "agent_id": "finance-fixture-agent", @@ -68,43 +78,45 @@ def _claim_agent_operation( ) -def _managed_handler( - service: ChatActionService, - store: ChatActionStore, - *, - session_id="owned-managed-thread", - profile_digest="c" * 64, - todo_id="todo-managed", - model="test-model", - reasoning_effort="xhigh", -): - from loopx.control_plane.turn_driver.codex_cli import _store_codex_cli_session - from loopx.control_plane.turn_driver.codex_operation_host import ( - operation_tool_handler, +def test_browser_operation_fixture_needs_no_test_framework(tmp_path: Path) -> None: + repo = Path(__file__).resolve().parents[1] + script = ( + repo / "examples/personal-workspace-browser/confirmed-operation-fixtures.py" + ) + isolated = tmp_path / "effect-server" + isolated.mkdir() + # -S excludes site packages (including pytest); LoopX itself has no Python + # runtime dependencies. The release rehearsal separately installs the wheel. + result = subprocess.run( + [sys.executable, "-S", str(script)], + cwd=tmp_path, + env={ + **os.environ, + "PYTHONPATH": str(repo), + "TMPDIR": str(isolated), + "TEMP": str(isolated), + "TMP": str(isolated), + }, + capture_output=True, + text=True, + timeout=30, + check=True, ) - - lineage = { - "goal_id": GOAL_ID, - "agent_id": "finance-fixture-agent", - "todo_id": todo_id, - } - runtime = store.root.parent.parent - _store_codex_cli_session( - runtime, - lineage=lineage, - session_id=session_id, - operation_profile_digest=profile_digest, - operation_model=model, - operation_reasoning_effort=reasoning_effort, + fixtures = json.loads(result.stdout) + assert list(fixtures) == ["confirmed", "waiting", "unknown", "reconciled"] + for proposal in fixtures.values(): + assert proposal["normalized_parameters"]["goal_id"] == "product-release" + assert proposal["normalized_parameters"]["executor"]["kind"] == "managed_turn" + assert ( + fixtures["unknown"]["operation"]["outcome"]["outcome"] == "submission_unknown" ) - return operation_tool_handler( - runtime_root=runtime, - registry_path=service.registry_path, - lineage=lineage, - session_id=session_id, - profile_digest=profile_digest, - model=model, - reasoning_effort=reasoning_effort, + assert ( + fixtures["reconciled"]["operation"]["outcome"] + == fixtures["unknown"]["operation"]["outcome"] + ) + assert ( + fixtures["reconciled"]["operation"]["reconciliation"]["outcome"] + == "not_executed" ) @@ -186,7 +198,9 @@ def test_managed_pending_reuses_registered_agent_and_goal_instance_scope( proposal["proposal_id"], delivery=_delivery(proposal) ) store.decide_operation( - proposal["proposal_id"], decision="confirm", confirmation=_confirmation(delivered) + proposal["proposal_id"], + decision="confirm", + confirmation=_confirmation(delivered), ) consumed = handler( "loopx_operation", @@ -331,8 +345,11 @@ def observed_lock(path, *args, **kwargs): def revoke(): _discard_codex_cli_session( store.root.parent.parent, - lineage={"goal_id": GOAL_ID, "agent_id": "finance-fixture-agent", - "todo_id": "todo-recovery"}, + lineage={ + "goal_id": GOAL_ID, + "agent_id": "finance-fixture-agent", + "todo_id": "todo-recovery", + }, ) commits.append("revocation") @@ -342,7 +359,9 @@ def record_report(self, payload): monkeypatch.setattr(codex_cli, "exclusive_file_lock", observed_lock) monkeypatch.setattr(ChatActionStore, "_write", record_report) - with ThreadPoolExecutor(max_workers=1, thread_name_prefix="managed-revoker") as executor: + with ThreadPoolExecutor( + max_workers=1, thread_name_prefix="managed-revoker" + ) as executor: futures = [] def interleaved_binding(registry_path, parameters, runtime_root): @@ -350,15 +369,24 @@ def interleaved_binding(registry_path, parameters, runtime_root): if parameters["executor"].get("todo_id") == "todo-recovery": futures.append(executor.submit(revoke)) assert attempted.wait(5) - assert not acquired.wait(0.2), "replacement revocation must wait for report commit" + assert not acquired.wait(0.2), ( + "replacement revocation must wait for report commit" + ) return current monkeypatch.setattr(operation_handoff, "_binding", interleaved_binding) - assert replacement( - "loopx_operation", - {"action": "report", "proposal_id": proposal["proposal_id"], "outcome": final}, - replacement_native, - )["ok"] is True + assert ( + replacement( + "loopx_operation", + { + "action": "report", + "proposal_id": proposal["proposal_id"], + "outcome": final, + }, + replacement_native, + )["ok"] + is True + ) futures[0].result(timeout=5) assert commits == ["report", "revocation"] monkeypatch.setattr(operation_handoff, "_binding", original_binding) @@ -440,28 +468,6 @@ def test_managed_tool_rejects_actor_injection_native_mismatch_and_revoked_profil assert store.load(proposal["proposal_id"])["operation"].get("agent_handoff") is None -def _agent_result( - proposal: dict, consumption_id: str, *, result: str = "executed" -) -> dict: - operation = proposal["operation"] - return { - "schema_version": "loopx_operation_outcome_v0", - "operation_id": proposal["proposal_id"], - "payload_digest": operation["payload_digest"], - "confirmation_digest": operation["confirmation_digest"], - "claim_id": operation["claim"]["claim_id"], - "executor_revision": operation["executor_revision"], - "consumption_id": consumption_id, - "outcome": result, - "projection_verified": True, - "simulation": False, - "external_write_performed": result != "not_executed", - "evidence_refs": ["receipt:synthetic-fixture-1"], - "summary": "Synthetic recorded execution evidence.", - "observed_at": datetime.now(timezone.utc).isoformat(), - } - - @pytest.mark.parametrize( "field,value", [ @@ -1198,130 +1204,6 @@ def test_inbox_uses_shared_recovery_priority_and_explicit_overflow( ) -def _digest(value: object) -> str: - encoded = json.dumps( - value, ensure_ascii=False, sort_keys=True, separators=(",", ":") - ).encode() - return hashlib.sha256(encoded).hexdigest() - - -def _service(tmp_path: Path) -> tuple[ChatActionService, ChatActionStore]: - project = tmp_path / "project" - project.mkdir() - (project / "ACTIVE_GOAL_STATE.md").write_text( - f"---\ngoal_id: {GOAL_ID}\n---\n\n## User Todo\n\n## Agent Todo\n", - encoding="utf-8", - ) - registry = project / ".loopx" / "registry.json" - registry.parent.mkdir() - registry.write_text( - json.dumps( - { - "goals": [ - { - "id": GOAL_ID, - "repo": str(project), - "state_file": "ACTIVE_GOAL_STATE.md", - "coordination": { - "registered_agents": ["finance-fixture-agent"] - }, - } - ] - } - ), - encoding="utf-8", - ) - store = ChatActionStore(tmp_path / "runtime" / "chat" / "actions") - return ChatActionService(store=store, registry_path=registry), store - - -def _request(*, payload: dict[str, object] | None = None) -> dict[str, object]: - operation_payload = payload or { - "schema_version": "finance_order_intent_v0", - "side": "buy", - "asset": "SYNTH", - "quantity": "1.00", - "order_type": "limit", - "limit_price": "10.00", - "time_in_force": "GTC", - "reduce_only": False, - } - return { - "action_kind": "operation.execute", - "summary": "Confirm one simulated finance order", - "idempotency_key": "operation-fixture-v1", - "context": {"kind": "goal", "goal_id": GOAL_ID}, - "normalized_parameters": { - "schema_version": "loopx_operation_request_v0", - "goal_id": GOAL_ID, - "agent_id": "finance-fixture-agent", - "domain": "finance", - "operation_kind": "finance.order.simulate", - "operation_schema": "finance_order_intent_v0", - "payload_ref": "finance-order:synthetic-1", - "payload": operation_payload, - "payload_digest": _digest(operation_payload), - "projection": { - "schema_version": "loopx_operation_projection_v0", - "title": "Simulated trade request", - "subtitle": "Synthetic fixture · no venue call", - "focus": "BUY 1.00 SYNTH @ 10.00", - "fields": [ - {"label": "Order type", "value": "Limit · GTC"}, - {"label": "Maximum notional", "value": "10.00 TEST"}, - ], - "warning": "Simulation only. This cannot submit, sign, or transfer.", - "simulated": True, - }, - "destination_account_ref": "account:simulation", - "expires_at": (datetime.now(timezone.utc) + timedelta(hours=1)).isoformat(), - "authorized_principals": [f"lark:{OPERATOR_ID}"], - "executor": { - "extension_id": "loopx-finance-execution", - "protocol": "finance_operation_executor_v0", - "permission": "finance.operation.simulate", - "revision": "simulator-v0", - }, - }, - } - - -def _delivery(proposal: dict[str, object]) -> dict[str, str]: - operation = proposal["operation"] - assert isinstance(operation, dict) - return { - "provider": "lark", - "message_id": "om_operation_fixture", - "chat_id": "oc_operation_fixture", - "app_id": "cli_operation_fixture", - "binding_digest": "a" * 64, - "card_digest": "b" * 64, - "delivered_at": datetime.now(timezone.utc).isoformat(), - } - - -def _confirmation( - proposal: dict[str, object], *, event_id: str = "evt-operation-1" -) -> dict[str, str]: - operation = proposal["operation"] - assert isinstance(operation, dict) - delivery = operation["delivery"] - assert isinstance(delivery, dict) - return { - "provider": "lark", - "event_id": event_id, - "principal": f"lark:{OPERATOR_ID}", - "message_id": str(delivery["message_id"]), - "chat_id": str(delivery["chat_id"]), - "app_id": str(delivery["app_id"]), - "surface_kind": "group_message_card", - "interaction_kind": "button_callback", - "confirmation_digest": str(operation["confirmation_digest"]), - "card_digest": str(delivery["card_digest"]), - "confirmed_at": datetime.now(timezone.utc).isoformat(), - } - - def test_operation_preview_arms_one_canonical_gate_and_local_apply_cannot_claim( tmp_path: Path, ) -> None: From 674d6abf1b2148030bc34f0bdf72cab785163128 Mon Sep 17 00:00:00 2001 From: huangruiteng Date: Wed, 30 Sep 2026 17:22:14 +0800 Subject: [PATCH 13/13] test(ui): validate interactive execution chip against its role Signed-off-by: huangruiteng --- .../personal-workspace-browser/execution-chip.mjs | 12 +++++++++--- 1 file changed, 9 insertions(+), 3 deletions(-) diff --git a/examples/personal-workspace-browser/execution-chip.mjs b/examples/personal-workspace-browser/execution-chip.mjs index 9bc3863534..08e8b48547 100644 --- a/examples/personal-workspace-browser/execution-chip.mjs +++ b/examples/personal-workspace-browser/execution-chip.mjs @@ -170,14 +170,20 @@ async function waitForPickerLabel(page, settled, timeoutMs = 15_000) { throw new Error(`Chat runtime picker never settled: ${resolution}`); } -async function assertHairlineRow(page, maxHeight = 26) { +async function assertHairlineRow(page, maxHeight) { const headerBox = await page.locator(".personal-channel-header").boundingBox(); - const chipBox = await page.locator(".personal-execution-chip").boundingBox(); + const chip = page.locator(".personal-execution-chip"); + const chipBox = await chip.boundingBox(); + // The model editor turns the read-only label into a desktop button with a + // 28px target. Keep the original 26px label budget and the explicit mobile + // budget; do not derive the limit from the observed height or its CSS. + const desktopBudget = await chip.evaluate((element) => element.tagName === "BUTTON" ? 28 : 26); + const heightBudget = maxHeight ?? desktopBudget; if (!headerBox || !chipBox) throw new Error("Execution chip has no layout box"); if (chipBox.y < headerBox.y || chipBox.y + chipBox.height > headerBox.y + headerBox.height) { throw new Error("Execution chip escaped the channel header row"); } - if (chipBox.height > maxHeight) { + if (chipBox.height > heightBudget) { throw new Error(`Execution chip is not a compact hairline row: ${chipBox.height}px tall`); } }