Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
Show all changes
22 commits
Select commit Hold shift + click to select a range
095c998
fix(dual-franka): fresh-observations review slice
codex Oct 10, 2026
8e12a26
fix(dual-franka): dashboard-operator review slice
codex Oct 10, 2026
d519892
fix(dual-franka): terminal-lifecycle review slice
codex Oct 10, 2026
12ab2af
Merge branch 'nieyi/split-05-terminal-lifecycle' into nieyi/split-06-…
codex Oct 10, 2026
ec9cfa0
test(interaction): focus attended regression coverage on complete att…
codex Oct 10, 2026
e1068f8
test(dashboard): narrow operator coverage to complete web task flows
codex Oct 10, 2026
ae54792
Merge branch 'nieyi/split-05-terminal-lifecycle' into nieyi/split-06-…
codex Oct 10, 2026
93c4af4
test(interaction): keep only primary attended execution paths
codex Oct 10, 2026
85b73b8
test(dashboard): keep one complete operator-controlled task flow
codex Oct 10, 2026
0515c3b
test(observation): exercise public acquisition and segmentation paths
codex Oct 10, 2026
a8d4409
Merge branch 'nieyi/split-04-fresh-observations' into nieyi/split-05-…
codex Oct 10, 2026
af5b7ab
test(interaction): fold feedback coverage into the existing input flow
codex Oct 10, 2026
9cf4369
Merge branch 'nieyi/split-05-terminal-lifecycle' into nieyi/split-06-…
codex Oct 10, 2026
c431779
refactor(interaction): reuse lifecycle guards and narrow operator rou…
codex Oct 10, 2026
02d34d4
refactor(interaction): reuse planner sessions and operator confirmations
codex Oct 10, 2026
1b067e3
refactor(cli): share operator outcome finalization
codex Oct 10, 2026
aec6c47
fix(exploration): serialize observations and retain stable outcome ev…
codex Oct 11, 2026
2b96676
test(interaction): reuse attended workflow fixtures
codex Oct 11, 2026
ec6f04f
test(interaction): keep prompt cleanup independently mergeable
codex Oct 11, 2026
bbd4d14
fix(interaction): preserve unaccepted input across session cancellation
codex Oct 11, 2026
6294826
fix(codex): resubmit input after a completed-turn steer rejection
codex Oct 11, 2026
68635bd
test: narrow offline regressions to essential workflows
codex Oct 11, 2026
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
12 changes: 12 additions & 0 deletions robots/dual_franka/perception.py
Original file line number Diff line number Diff line change
Expand Up @@ -843,12 +843,24 @@ def _mask_to_camera_world(
result.update({"point_xyz": None, "world_error": "empty mask"})
return result

# Keep the rejected mask visible too; a missing valid 3-D point must not
# hide where segmentation actually landed in the image.
result["centroid_pixel"] = [int(np.median(rows)), int(np.median(cols))]
result["mask_bbox_rc"] = [
int(rows.min()),
int(cols.min()),
int(rows.max()),
int(cols.max()),
]

depths = depth[rows, cols].astype(np.float64)
valid_depth = np.isfinite(depths) & (depths > 0.0)
rows = rows[valid_depth]
cols = cols[valid_depth]
depths = depths[valid_depth]
result["valid_depth_pixels"] = int(depths.size)
if depths.size:
result["raw_depth_median_m"] = round(float(np.median(depths)), 5)
if depths.size < min_valid:
result.update(
{
Expand Down
2 changes: 2 additions & 0 deletions robots/dual_franka/prompt_bundle.py
Original file line number Diff line number Diff line change
Expand Up @@ -53,4 +53,6 @@ def user_prompt(variables: Mapping[str, object] | None = None) -> PromptNode:
"Use {{memory_inbox}} for reviewable exploration notes and "
"{{output_dir}}/attempts/ for failed-attempt archives."
)
if (variables or {}).get("dashboard"):
node["OPERATOR INPUT"] = user_parts.DASHBOARD_OPERATOR
return node
22 changes: 16 additions & 6 deletions robots/dual_franka/prompts/explore.py
Original file line number Diff line number Diff line change
Expand Up @@ -29,24 +29,34 @@
state take precedence over using the remaining exploration budget."""

RULES = (
"Acknowledge explicit operator feedback and explain the resulting plan change. "
"An operator may request practicing a later task stage while preserving earlier "
"progress. Request confirmation of that partial starting scene, not full restoration. "
"Do not repeat completed stages merely to satisfy the preset order. Still obtain "
"fresh reset confirmation before motion, inspect the scene and re-localize. "
"Record partial-start practice as such, not as an end-to-end successful run. "
"If feedback is ambiguous, ask a specific question instead of ignoring it.",
"Before the first motion in every session, request_scene_reset(reason, "
"expected_scene_state). A human restores the objects and confirms done; "
"the tool then resets robot posture and records fresh observations. A new "
"toolkit or planner session does not restore the tabletop.",
"After reset, re-read the new state and re-localize. Never reuse a pixel, "
"point or TCP target from an earlier attempt. Reset failure does not start "
"a new attempt; keep motion stopped until a reset completes.",
"Read describe_dual_franka_setup for registered tools and cameras. Use move_delta, "
"rotate_delta, gripper tools, recover_joint_posture, back_project, optional "
"segment, and the named vla_right_grasp/vla_handoff/vla_left_place skills. "
"Do not assume LIBERO primitives or the older vla_grasp tool exist.",
"Read describe_dual_franka_setup for registered tools and cameras. Use only "
"the tools available for the current task; do not assume a fixed skill set.",
"Ask request_operator_verdict after apparent success or failure. The human "
"answers success, failure, continue or abort. Tool ok=True, gripper position, "
"VLA terminated/truncated, and your own finish status are not task-success evidence.",
"Any subsequent physical action invalidates the verdict. continue clears "
"the previous verdict. Obtain another judgment before finish.",
"the previous verdict. A returned continue is already authorization to "
"resume this attempt, not a request to wait for another continue. Read its "
"notes and fresh observation and continue without resetting. Do not ask "
"again about the same answered question unless new evidence introduces a "
"concrete blocker. Obtain another judgment before finish.",
"For a failed attempt, archive evidence and update wip before requesting "
"a new scene reset. Change a named lever: order, staging, target or VLA prompt/chunk budget. "
"a new scene reset. Change one meaningful lever, such as perception inputs "
"or use of available tools, within the current task requirements. "
"Use in-place recovery only when safe. Never force a restart after operator abort.",
)

Expand Down
5 changes: 5 additions & 0 deletions robots/dual_franka/prompts/user.py
Original file line number Diff line number Diff line change
Expand Up @@ -26,3 +26,8 @@
together with both arms' TCP/gripper/joint-health state, and execute the task
conservatively with the exposed bounded tools. Auxiliary camera views are
artifact views for targeted follow-up inspection only."""

DASHBOARD_OPERATOR = """Before the first motion call request_scene_reset and wait
for the operator's Dashboard confirmation. Final success/failure requires
request_operator_verdict or an explicit operator verdict. There is no terminal
input; use the Dashboard message input for operator replies."""
7 changes: 5 additions & 2 deletions robots/dual_franka/robot_spec.py
Original file line number Diff line number Diff line change
Expand Up @@ -56,7 +56,7 @@
"name": "task_id",
"kind": "integer",
"minimum": 0,
"suggestions": tuple(sorted(DUAL_FRANKA_TASKS)),
"suggestions": tuple(str(key) for key in sorted(DUAL_FRANKA_TASKS)),
},
),
"display": "Dual Franka task {task_id}",
Expand Down Expand Up @@ -424,7 +424,10 @@ def _init_runtime(
connectors = {
"env": lambda rpc: {
"env": DualFrankaEnvClient(
rpc, reset_on_connect=not getattr(args, "explore", False)
rpc,
reset_on_connect=not (
getattr(args, "explore", False) or getattr(args, "dashboard", False)
),
),
"task_description": get_dual_franka_task(args.task_id).instruction,
"vla_instruction": get_dual_franka_task(args.task_id).vla_instruction,
Expand Down
130 changes: 83 additions & 47 deletions robots/dual_franka/toolkit.py
Original file line number Diff line number Diff line change
Expand Up @@ -83,11 +83,12 @@ def __init__(
if attempts_per_session < 0:
raise ValueError("attempts_per_session must be nonnegative")
self._mode = mode
self._attended = mode == "exploration" or operator_input is not None
self._attempt = 0 # first attempt begins only after scene confirmation
self._budget = attempts_per_session
self._budget = attempts_per_session if mode == "exploration" else 0
self._attempts_per_session = attempts_per_session
self._session_attempt = 1
self._scene_ready = mode != "exploration"
self._scene_ready = not self._attended
self._operator_input = operator_input
self._operator_verdict = None
self._operator_notes = ""
Expand All @@ -97,6 +98,7 @@ def __init__(
self._events = []
self._direct_verdict_event = threading.Event()
self._direct_verdict: str | None = None
self._direct_verdict_notes = ""
super().__init__(
runtime_kwargs=runtime_kwargs,
dashboard_events=dashboard_events,
Expand All @@ -109,19 +111,20 @@ def direct_verdict_requested(self) -> bool:
"""Whether operator control has sealed the active session."""
return self._direct_verdict_event.is_set()

def request_direct_verdict(self, verdict: str) -> bool:
def request_direct_verdict(self, verdict: str, notes: str = "") -> bool:
"""Seal this attempt immediately; cancel in-flight work at its next boundary."""
if verdict not in {"success", "failure", "abort"}:
raise ValueError("verdict must be success, failure or abort")
with self._scheduler.condition:
if self._direct_verdict_event.is_set():
return self._direct_verdict == verdict
if self._mode != "exploration" or (
if self._scheduler.closed:
return False
if not self._attended or (
verdict == "success"
and (not self._scene_ready or self._operator_aborted)
):
return False
self._direct_verdict = verdict
self._direct_verdict_notes = notes
self._direct_verdict_event.set()
self._scheduler.cancel(close=True)
return True
Expand Down Expand Up @@ -160,21 +163,20 @@ def finalize_direct_verdict(self) -> dict[str, Any]:
"operator_finished": True,
"operator_aborted": True,
"verdict_source": "interactive_command",
"operator_notes": self._direct_verdict_notes,
}
self._event("verdict", verdict="abort", source="interactive_command")
self._event("finish", **result)
return result
self.get_env_state(
command={"action": "observe_for_verdict"}, result={}, elapsed_s=0.0
)
self._validate_observation()
self._observe_current("observe_for_verdict")
record = self.state.latest_record()
self._publish_step(record)
self._scene_ready = True
self._operator_verdict = self._direct_verdict
self._operator_notes = (
f"Operator entered /{self._direct_verdict} in the interactive terminal."
)
if self._direct_verdict_notes:
self._operator_notes += " " + self._direct_verdict_notes
self._verdict_step = record.step_idx
self._event(
"verdict",
Expand All @@ -185,9 +187,10 @@ def finalize_direct_verdict(self) -> dict[str, Any]:
result = {
"status": self._direct_verdict,
"operator_verdict": self._direct_verdict,
"operator_finished": True,
"operator_finished": False,
"verdict_source": "interactive_command",
"operator_aborted": False,
"operator_notes": self._direct_verdict_notes,
}
self._event("finish", **result)
return result
Expand Down Expand Up @@ -224,15 +227,38 @@ def _guard_motion(self, inner, **kwargs) -> ToolResult:
self._clear_verdict()
return inner(**kwargs)

def _is_readonly_call(self, tool: Tool, kwargs: dict[str, Any]) -> bool:
if tool.name == "view_env_state":
return kwargs["step"] >= 0
return tool.readonly and tool.name != "request_operator_verdict"

def _view_env_state(self, step: int = -1) -> ToolResult:
if step >= 0:
return dual_franka_tools.view_env_state(step, state=self._state)
return self._observe_current("observe_current")

def _observe_current(self, action: str) -> ToolResult:
output = self.get_env_state(
command={"action": action}, result={}, elapsed_s=0.0
)
self._validate_observation()
self._publish_step(self.state.latest_record())
return output

def _current_perception(self, inner, **kwargs) -> ToolResult:
step = kwargs.get("step")
if not self._scene_ready or (
step is not None and step != -1 and step < self._attempt_start_step
):
if not self._scene_ready:
return ToolResult(
data={
"error": "localization refused; use fresh observations after confirmed scene reset"
}
)
if step is not None and step != -1 and step < self.state.latest_step:
return ToolResult(
data={
"error": (
"localization refused; use fresh observations after confirmed scene reset"
f"localization refused: step {step} is stale; use current step "
f"{self.state.latest_step} and reselect the target on its image"
)
}
)
Expand Down Expand Up @@ -306,10 +332,11 @@ def _request_scene_reset(
kind="reset",
)
self._event("reset_response", response=response)
if response is None or response.strip().lower() != "done":
self._operator_aborted = (
response is None or response.strip().lower() == "abort"
)
response_parts = (response or "").strip().split(maxsplit=1)
confirmation = response_parts[0].lower() if response_parts else ""
notes = response_parts[1] if len(response_parts) > 1 else ""
if confirmation != "done":
self._operator_aborted = response is None or confirmation == "abort"
return ToolResult(
data={
"error": "scene reset not confirmed",
Expand All @@ -332,6 +359,7 @@ def _request_scene_reset(
"ok": True,
"robot_reset": result,
"scene_reset_confirmed": True,
"operator_notes": notes,
"notice": "Scene restored by operator; robot posture reset. Re-localize from the new images.",
}
)
Expand Down Expand Up @@ -359,12 +387,8 @@ def _request_operator_verdict(
data={"error": "verdict refused; no active confirmed attempt"}
)
# Save the evidence being judged using the existing camera/state logger.
self.get_env_state(
command={"action": "observe_for_verdict"}, result={}, elapsed_s=0.0
)
self._validate_observation()
self._observe_current("observe_for_verdict")
record = self.state.latest_record()
self._publish_step(record)
response = self._ask_operator(
f"{question}\nAttempt {self._attempt}, observation step {record.step_idx}. "
"Reply success, failure, continue, or abort; optional notes may follow.",
Expand All @@ -381,7 +405,19 @@ def _request_operator_verdict(
self._operator_verdict = verdict
self._operator_notes = notes
self._verdict_step = record.step_idx
elif verdict != "continue":
elif verdict == "continue":
observation = self._view_env_state()
observation.data.update(
ok=True,
status=verdict,
operator_notes=notes,
attempt=self._attempt,
evidence_step=record.step_idx,
observation_step=self.state.latest_step,
operator_aborted=False,
)
return observation
else:
return ToolResult(
data={"error": "invalid operator verdict", "evidence": event}
)
Expand Down Expand Up @@ -422,8 +458,8 @@ def _guarded_finish(self, inner, **kwargs) -> ToolResult:
self._event("finish", **result.to_dict())
return result

def get_env_state(self, *, command, result, elapsed_s):
if self._mode != "exploration":
def get_env_state(self, *, command, result, elapsed_s) -> ToolResult:
if not self._attended:
return super().get_env_state(
command=command, result=result, elapsed_s=elapsed_s
)
Expand All @@ -446,8 +482,11 @@ def get_env_state(self, *, command, result, elapsed_s):
"operator_verdict": self._operator_verdict,
"attempt_start_step": self._attempt_start_step,
}
self.state.save("exploration.json", status)
output.data["exploration"] = status
lifecycle = (
"exploration" if self._mode == "exploration" else "operator_lifecycle"
)
self.state.save(f"{lifecycle}.json", status)
output.data[lifecycle] = status
if result.get("error"):
output.data["error"] = result["error"]
return output
Expand Down Expand Up @@ -482,7 +521,7 @@ def _validate_observation(self) -> None:
def solved(self) -> bool:
if self.flash_options is not None:
return self._flash_solved
if self._mode != "exploration":
if not self._attended:
return self._operator_verdict == "success"
return (
self._scene_ready
Expand Down Expand Up @@ -558,6 +597,7 @@ def _register_tools(self) -> None:
}
state_handlers.update(
{
"view_env_state": self._view_env_state,
"view_camera_meta": partial(
franka_tools.view_camera_meta, state=self._state
),
Expand All @@ -568,16 +608,16 @@ def _register_tools(self) -> None:
),
"request_scene_reset": self._request_scene_reset,
"request_operator_verdict": self._request_operator_verdict
if self._mode == "exploration"
if self._attended
else self._evaluation_verdict,
}
)
for definition in self.declared_tools():
name = definition.name
if name in _EXPLORATION_ONLY_TOOLS and self._mode != "exploration":
if name in _EXPLORATION_ONLY_TOOLS and not self._attended:
continue
handler = state_handlers.get(name) or getattr(self._primitives, name)
if self._mode == "exploration":
if self._attended:
if name in _MOTION_TOOLS:
handler = partial(self._guard_motion, handler)
elif name in {"back_project", "segment"}:
Expand All @@ -586,11 +626,7 @@ def _register_tools(self) -> None:
handler = partial(self._describe_exploration_setup, handler)
self.add_tool(definition.with_handler(handler))
finish = self._tools["finish"]
guard = (
self._guarded_finish
if self._mode == "exploration"
else self._evaluation_finish
)
guard = self._guarded_finish if self._attended else self._evaluation_finish
self.add_tool(finish.with_handler(partial(guard, finish)), replace=True)

def _read_operator_line(self, prompt: str) -> str | None:
Expand Down Expand Up @@ -631,15 +667,15 @@ def _evaluation_verdict(
}
)
if verdict == "continue":
return ToolResult(
data={
"ok": True,
"status": "continue",
"operator_notes": notes,
"attempt": self._attempt,
"notice": "Operator requested more action; do not finish yet.",
}
observation = self._view_env_state()
observation.data.update(
ok=True,
status="continue",
operator_notes=notes,
attempt=self._attempt,
notice="Operator requested more action; do not finish yet.",
)
return observation
self._operator_verdict = verdict
self._operator_notes = notes
return ToolResult(
Expand Down
2 changes: 1 addition & 1 deletion robots/dual_franka/tools.py
Original file line number Diff line number Diff line change
Expand Up @@ -580,7 +580,7 @@ def view_env_state(
*,
state: EnvState,
) -> ToolResult:
"""Read a dual-Franka state snapshot. Configured inline camera views are returned directly; other available views are returned as artifact paths; use read_image to inspect these artifacts."""
"""Capture a fresh dual-Franka observation when step=-1 (default); a nonnegative step reads that historical snapshot without acquisition. Use the returned step for segmentation and back-projection; do not reuse pixels or masks from an older step after refreshing. Configured inline camera views are returned directly; other available views are returned as artifact paths for targeted read_image inspection."""
images: list[bytes] = []
record = state.get(step)
output = record.to_blob()
Expand Down
Loading
Loading