From 691d60d9b44ef178635af5db8e0fa785f9a1ea85 Mon Sep 17 00:00:00 2001 From: mika <211269698+mikamikasuki@users.noreply.github.com> Date: Thu, 8 Oct 2026 07:20:37 -0700 Subject: [PATCH 1/2] test(explore): align replan qualification with inline context Signed-off-by: mika <211269698+mikamikasuki@users.noreply.github.com> --- loopx/capabilities/explore/README.md | 17 ++-- .../replan_semantic_action_behavior.py | 18 +++- .../test_replan_semantic_action_behavior.py | 93 +++++++++++-------- 3 files changed, 80 insertions(+), 48 deletions(-) diff --git a/loopx/capabilities/explore/README.md b/loopx/capabilities/explore/README.md index 7ee395e5a6..dae96268fc 100644 --- a/loopx/capabilities/explore/README.md +++ b/loopx/capabilities/explore/README.md @@ -546,13 +546,16 @@ planning remains enabled fails with an actionable mode command. Do not combine `--explore-mode` with the legacy enable flags in one request. Both enabled modes register an `explore.turn_context` turn-start hook. Its -`required_reads` entry is a **before-work read obligation** in the normal packet. -The command returns at most three recent nodes/findings and, in planning mode, -three suggested Todo branches, plus structured commands for detail or evidence -recording. The read folds existing history but bounds the returned context; -it does not claim to reduce history IO. The agent chooses evidence-backed work; -planner suggestions do not require branching on every turn or recording empty -ceremonial nodes. Use the detail command when the short view is insufficient. +bounded result is projected inline under +`interaction_contract.agent_channel.work_context.sources` before work. The +source retains a command for replay, but the inline context is not a separate +`required_reads` obligation. It contains at most three recent nodes/findings +and, in planning mode, three suggested Todo branches, plus structured commands +for detail or evidence recording. The read folds existing history but bounds +the returned context; it does not claim to reduce history IO. The agent chooses +evidence-backed work; planner suggestions do not require branching on every +turn or recording empty ceremonial nodes. Use the detail command when the +short view is insufficient. Explicit writeback results default to three full scoped details, not a hard visibility limit. `graph.result_page` reports total/remaining counts and an diff --git a/loopx/control_plane/testing/replan_semantic_action_behavior.py b/loopx/control_plane/testing/replan_semantic_action_behavior.py index f332e7d811..6189d43a44 100644 --- a/loopx/control_plane/testing/replan_semantic_action_behavior.py +++ b/loopx/control_plane/testing/replan_semantic_action_behavior.py @@ -536,7 +536,23 @@ def _is_replan_successor_create(command: str) -> bool: def _required_explore_read_command(packet: Mapping[str, Any]) -> str | None: """Accept only this fixture's current, scoped turn-start read contract.""" - reads = packet.get("required_reads") or [] + interaction = packet.get("interaction_contract") + agent_channel = ( + interaction.get("agent_channel") + if isinstance(interaction, Mapping) + else None + ) + raw_reads = ( + agent_channel.get("required_reads") + if isinstance(agent_channel, Mapping) + else None + ) + if not isinstance(raw_reads, list): + raw_reads = packet.get("required_reads") + reads = [ + read for read in raw_reads or [] + if isinstance(read, Mapping) and read.get("kind") == "explore_turn_context" + ] if not reads: return None if not isinstance(reads, list) or len(reads) != 1: diff --git a/tests/control_plane/test_replan_semantic_action_behavior.py b/tests/control_plane/test_replan_semantic_action_behavior.py index cb1e72044d..bc168bf7dd 100644 --- a/tests/control_plane/test_replan_semantic_action_behavior.py +++ b/tests/control_plane/test_replan_semantic_action_behavior.py @@ -89,13 +89,38 @@ def _exhausted_action( ) -def _explore_context_action( +def _inline_explore_context_action( request: Mapping[str, object], ) -> ScriptedExecToolAction: - read = _latest_quota_packet(request)["required_reads"][0] - assert read["kind"] == "explore_turn_context" - assert read["ordering"] == "before_work" - return ScriptedExecToolAction(command=read["command"]) + packet = _latest_quota_packet(request) + interaction = packet["interaction_contract"] + assert isinstance(interaction, Mapping) + channel = interaction["agent_channel"] + assert isinstance(channel, Mapping) + work_context = channel["work_context"] + assert isinstance(work_context, Mapping) + sources = work_context["sources"] + assert isinstance(sources, list) + explore_sources = [ + source for source in sources + if isinstance(source, Mapping) + and source.get("kind") == "explore_turn_context" + ] + assert len(explore_sources) == 1 + source = explore_sources[0] + content = source.get("content") + assert isinstance(content, Mapping) + assert content.get("goal_id") == "replan-semantic-action-fixture" + assert content.get("agent_id") == "codex-replan-semantic-action" + boundary = content.get("boundary") + assert isinstance(boundary, Mapping) + assert boundary.get("read_only") is True + assert isinstance(source.get("command"), str) and source["command"] + assert not any( + isinstance(read, Mapping) and read.get("kind") == "explore_turn_context" + for read in channel.get("required_reads", []) + ) + return ScriptedExecToolAction(command="cat replan-frontier.json") def _composition_successor_action( @@ -435,8 +460,7 @@ def test_real_tool_loop_selects_composition_gap_and_creates_bound_successor( transport = ScriptedDoubaoExecTransport( [ ScriptedExecToolAction(command=fixture.quota_guard_command), - _explore_context_action, - ScriptedExecToolAction(command="cat replan-frontier.json"), + _inline_explore_context_action, ScriptedExecToolAction(command="cat fixture/permission-config.json"), _composition_successor_action, ] @@ -453,7 +477,6 @@ def test_real_tool_loop_selects_composition_gap_and_creates_bound_successor( assert receipt["qualification_passed"] is True, receipt assert receipt["observed_tool_sequence"] == [ "quota_should_run", - "explore_turn_context", "workspace_read", "workspace_read", "replan_successor_create", @@ -479,38 +502,28 @@ def test_real_tool_loop_selects_composition_gap_and_creates_bound_successor( ] == selected_gap["experiment_node_ref"] -@pytest.mark.parametrize("attempt", ["omit", "repeat", "wrong_scope"]) -def test_composition_read_obligation_cannot_be_skipped_replayed_or_retargeted( - tmp_path: Path, attempt: str, -) -> None: - fixture = _build_fixture(tmp_path / "oracle", composition_frontier=True) - actions = [ScriptedExecToolAction(command=fixture.quota_guard_command)] - if attempt == "omit": - actions.append(ScriptedExecToolAction(command="cat replan-frontier.json")) - elif attempt == "repeat": - actions.extend([_explore_context_action, _explore_context_action]) - else: - def wrong_scope(request: Mapping[str, object]) -> ScriptedExecToolAction: - action = _explore_context_action(request) - return ScriptedExecToolAction( - command=action.command.replace( - "--goal-id replan-semantic-action-fixture", "--goal-id another-goal" - ) - ) - actions.append(wrong_scope) - receipt = DoubaoReplanSemanticActionBehaviorActor( - api_key="test-only-placeholder", transport=ScriptedDoubaoExecTransport(actions), - ).qualify( - qualification_id=f"composition-read-{attempt}", - fixture_root=tmp_path / "actor", composition_frontier=True, - ) - assert receipt["qualification_passed"] is False - assert receipt["failure_code"] == ( - "repeated_explore_turn_context" - if attempt == "repeat" else "required_explore_turn_context_missing" - ) - assert receipt["semantic_action_accepted"] is False - assert "replan_successor_create" not in receipt["observed_tool_sequence"] +def test_required_explore_read_uses_nested_interaction_contract() -> None: + command = ( + "loopx --format json explore turn-context --goal-id " + "replan-semantic-action-fixture --agent-id codex-replan-semantic-action" + ) + packet = { + "interaction_contract": { + "agent_channel": { + "required_reads": [ + {"kind": "goal_state", "command": "cat goal-state.md"}, + { + "kind": "explore_turn_context", + "source": "turn_start_capability_hook", + "ordering": "before_work", + "command": command, + }, + ], + }, + }, + } + + assert replan_behavior._required_explore_read_command(packet) == command def test_action_outside_observed_frontier_is_rejected(tmp_path: Path) -> None: From 86f3a792e46108f43ac7bcbfe0cece06d1afd74c Mon Sep 17 00:00:00 2001 From: mika <211269698+mikamikasuki@users.noreply.github.com> Date: Thu, 8 Oct 2026 09:47:36 -0700 Subject: [PATCH 2/2] test(control-plane): use valid replan trigger in recovery fixture Signed-off-by: mika <211269698+mikamikasuki@users.noreply.github.com> --- tests/control_plane/test_selection_replan_reentry.py | 7 ++++--- 1 file changed, 4 insertions(+), 3 deletions(-) diff --git a/tests/control_plane/test_selection_replan_reentry.py b/tests/control_plane/test_selection_replan_reentry.py index 9267636168..00f400d12d 100644 --- a/tests/control_plane/test_selection_replan_reentry.py +++ b/tests/control_plane/test_selection_replan_reentry.py @@ -6,7 +6,6 @@ from test_quota_settlement_cli import ( AGENT_ID, GOAL_ID, SELECTED_REPLAN_TODO_ID, TODO_ID, - AUTONOMOUS_REPLAN_PERIODIC_RUN_THRESHOLD, _append_surface_only_runs, _configure_autonomous_replan_fixture, _configure_selected_todo_replan_fixture, _configure_selectable_alternative, _heartbeat_receipt_count, _projected_cli_args, _run_cli, _run_generated_cli, _spend_run_count, _write_fixture, @@ -21,7 +20,6 @@ def test_replan_selection_recovers_inline_or_refuses_ineligible_choice_and_settl _configure_selectable_alternative(project) if initial_replan: _configure_selected_todo_replan_fixture(project, registry) - _append_surface_only_runs(runtime, count=AUTONOMOUS_REPLAN_PERIODIC_RUN_THRESHOLD) turn = "turn-selection-preempted" guard = ("quota", "should-run", "--codex-app", "--goal-id", GOAL_ID, "--agent-id", AGENT_ID, "--turn-instance-id", turn, "--scan-path", str(project)) @@ -31,7 +29,10 @@ def test_replan_selection_recovers_inline_or_refuses_ineligible_choice_and_settl assert first["interaction_contract"]["cli_channel"]["selection_required"] assert "settlement_identity" not in first["heartbeat_receipt"] if initial_replan: - assert "periodic_review_due" in { + # This test exercises selection recovery from an existing replan. + # Periodic review has its own settlement-qualified coverage; these + # setup runs are not settled Turns and must not trigger its cadence. + assert "long_todo_chain" in { trigger["kind"] for trigger in first["autonomous_replan_obligation"]["triggers"] } if binding == "todo":