diff --git a/demos/moveit_pick_place/README.md b/demos/moveit_pick_place/README.md index 11fe34e..eeb7cd6 100644 --- a/demos/moveit_pick_place/README.md +++ b/demos/moveit_pick_place/README.md @@ -73,7 +73,27 @@ Use the interactive arm controller to send joint trajectories: ``` The script sends goals directly to the `panda_arm_controller/follow_joint_trajectory` action. -It works both from outside (via `docker exec`) and from inside the container. +It works both from outside (via `docker exec`, no TTY required) and from inside the container. +Where there is no `docker` CLI (inside the container), it uses the local `ros2`. Where there +is one, it uses a local `ros2` only if `ros2 action list` shows the arm action; it asks twice, +because a first listing without a running `ros2` daemon can miss it. Otherwise it runs the +goal through `docker exec` in the demo container. + +`pick_place_loop.py` keeps sending its own goals to the same controller, so a manual move +can be preempted mid-motion by the demo's own workload. `move-arm.sh` reports the goal's +real final status: if the controller aborts it (`error_string: Current goal preempted by +new incoming action`), the script prints `Failed: (status: ABORTED)` and exits +non-zero instead of claiming success. `./move-arm.sh demo` runs all three steps regardless +of earlier failures and reports each one; the command exits non-zero if any step failed. + +Each `ros2 action send_goal` run is limited to 30 seconds. The controller can fail to +deliver the goal response to a freshly started CLI (the container log shows `Failed to send +goal response`); it then never runs that goal. When the CLI printed `Sending goal:` and no +goal response arrived, the script sends the goal again, up to three times. A goal that was +accepted is never sent twice: if its result does not arrive in time, the script prints +`Failed: (status: UNKNOWN)`. A goal that was never sent (no demo container, no action +server within 30 seconds, a `ros2` error) is not sent again: the script prints the command's +output and `Failed: (goal not sent: ...)`, and exits non-zero. ### 4. Viewing Logs @@ -209,6 +229,10 @@ curl http://localhost:8080/api/v1/apps/move-group/operations | jq curl http://localhost:8080/api/v1/faults | jq ``` +While the fault manager is not available (for example right after startup), `GET /faults` +answers `503`. `./check-faults.sh` then prints `Could not read faults (HTTP 503)` and exits +non-zero; it reports "No active faults" only after a successful read of an empty list. + ### Clear All Faults ```bash @@ -308,8 +332,8 @@ The gateway supports condition-based triggers that fire when specific events occ ### How It Works -1. `setup-triggers.sh` creates a trigger via `POST /api/v1/apps/manipulation_monitor/triggers`: - - **Resource:** `/api/v1/apps/manipulation_monitor/faults` (watches fault collection) +1. `setup-triggers.sh` creates a trigger via `POST /api/v1/apps/manipulation-monitor/triggers`: + - **Resource:** `/api/v1/apps/manipulation-monitor/faults` (watches fault collection) - **Condition:** `OnChange` (fires on any new or updated fault) - **Multishot:** `true` (fires repeatedly, not just once) - **Lifetime:** 3600 seconds (auto-expires after 1 hour) @@ -320,23 +344,23 @@ The gateway supports condition-based triggers that fire when specific events occ ```bash # Create a trigger -curl -X POST http://localhost:8080/api/v1/apps/manipulation_monitor/triggers \ +curl -X POST http://localhost:8080/api/v1/apps/manipulation-monitor/triggers \ -H "Content-Type: application/json" \ -d '{ - "resource": "/api/v1/apps/manipulation_monitor/faults", + "resource": "/api/v1/apps/manipulation-monitor/faults", "trigger_condition": {"condition_type": "OnChange"}, "multishot": true, "lifetime": 3600 }' | jq # List triggers -curl http://localhost:8080/api/v1/apps/manipulation_monitor/triggers | jq +curl http://localhost:8080/api/v1/apps/manipulation-monitor/triggers | jq # Watch events (replace TRIGGER_ID) -curl -N http://localhost:8080/api/v1/apps/manipulation_monitor/triggers/TRIGGER_ID/events +curl -N http://localhost:8080/api/v1/apps/manipulation-monitor/triggers/TRIGGER_ID/events # Delete a trigger -curl -X DELETE http://localhost:8080/api/v1/apps/manipulation_monitor/triggers/TRIGGER_ID +curl -X DELETE http://localhost:8080/api/v1/apps/manipulation-monitor/triggers/TRIGGER_ID ``` ## Fault Injection Scenarios @@ -445,7 +469,7 @@ Container scripts are stored under `/var/lib/ros2_medkit/scripts/moveit-planning | Docker build fails | Apt package missing | Check if MoveIt 2 Jazzy packages are available | | "MoveGroup not available" | Slow startup | Wait 60-90 seconds after container starts | | Controller not loading | Missing config | Verify `moveit_controllers.yaml` is correct | -| Joint states empty | Controllers not loaded | Check `ros2 control list_controllers` inside container | +| Joint states empty (`check-entities.sh` prints "Joint state data not available") | Controllers not loaded | Check `ros2 control list_controllers` inside container | | `ros2` CLI hangs in `docker exec` | DDS discovery across container boundaries | Use gateway REST API instead of `ros2` CLI for parameter/service operations | ## Comparison with Other Demos diff --git a/demos/moveit_pick_place/check-entities.sh b/demos/moveit_pick_place/check-entities.sh index 3a32ef0..70412a6 100755 --- a/demos/moveit_pick_place/check-entities.sh +++ b/demos/moveit_pick_place/check-entities.sh @@ -40,24 +40,31 @@ echo_step "1. Areas (Functional Groupings)" curl -s "${API_BASE}/areas" | jq '.items[] | {id: .id, name: .name, description: .description}' echo_step "2. Components (Hardware/Logical Units)" -curl -s "${API_BASE}/components" | jq '.items[] | {id: .id, name: .name, type: .type, area: .area}' +curl -s "${API_BASE}/components" | jq '.items[] | {id: .id, name: .name, description: .description}' echo_step "3. Apps (ROS 2 Nodes)" -curl -s "${API_BASE}/apps" | jq '.items[] | {id: .id, name: .name, category: .category, component: .is_located_on}' +curl -s "${API_BASE}/apps" | jq '.items[] | {id: .id, name: .name, description: .description, component: .["x-medkit"].component_id}' echo_step "4. Functions (High-level Capabilities)" -curl -s "${API_BASE}/functions" | jq '.items[] | {id: .id, name: .name, category: .category, hosted_by: .hosted_by}' +curl -s "${API_BASE}/functions" | jq '.items[] | {id: .id, name: .name, description: .description}' echo_step "5. Sample Data (Joint States)" echo "Getting latest joint states from Panda arm..." -curl -s "${API_BASE}/apps/joint-state-broadcaster/data/joint_states" 2>/dev/null | jq '{ - joint_names: .data.name, - positions: .data.position, - velocities: .data.velocity -}' || echo " (Joint state data not available — robot may still be starting)" +# An error body or a reply without data (no /joint_states publisher) has no +# joint names, and jq still exits 0 on it. +JOINT_STATES=$(curl -s "${API_BASE}/apps/joint-state-broadcaster/data/joint_states" 2>/dev/null) || JOINT_STATES="" +if jq -e '.data.name | arrays | length > 0' <<< "${JOINT_STATES}" > /dev/null 2>&1; then + jq '{ + joint_names: .data.name, + positions: .data.position, + velocities: .data.velocity + }' <<< "${JOINT_STATES}" +else + echo " (Joint state data not available - the robot may still be starting, or joint_state_broadcaster is not running)" +fi echo_step "6. Faults" -curl -s "${API_BASE}/faults" | jq '.items[] | {code: .code, severity: .severity, reporter: .reporter_id}' +curl -s "${API_BASE}/faults" | jq '.items[] | {code: .fault_code, severity: .severity_label, status: .status, sources: .reporting_sources}' echo "" echo -e "${GREEN}✓ Entity hierarchy exploration complete!${NC}" diff --git a/demos/moveit_pick_place/check-faults.sh b/demos/moveit_pick_place/check-faults.sh index 51ac30b..72273c4 100755 --- a/demos/moveit_pick_place/check-faults.sh +++ b/demos/moveit_pick_place/check-faults.sh @@ -25,12 +25,19 @@ fi echo "✓ Gateway is healthy" echo "" -# Get all faults +# Get all faults. A failed read is not "no faults": the gateway answers 503 +# while the fault manager is unavailable (for example during startup). echo "📋 Active Faults:" -FAULTS=$(curl -s "${API_BASE}/faults") +RESPONSE=$(curl -s -m 30 -w '\n%{http_code}' "${API_BASE}/faults") || true +HTTP_CODE="${RESPONSE##*$'\n'}" +FAULTS="${RESPONSE%$'\n'*}" -# Check if there are any faults -FAULT_COUNT=$(echo "$FAULTS" | jq '.items | length') +if [ "$HTTP_CODE" != "200" ] || ! FAULT_COUNT=$(echo "$FAULTS" | jq -e '.items | arrays | length' 2>/dev/null); then + DETAIL=$(echo "$FAULTS" | jq -r '[.message, .parameters.details] | map(select(. != null)) | join(": ")' 2>/dev/null) + echo "❌ Could not read faults (HTTP ${HTTP_CODE:-000})${DETAIL:+: ${DETAIL}}" + echo " The fault list is unknown, not empty. Retry in a few seconds." + exit 1 +fi if [ "$FAULT_COUNT" = "0" ]; then echo " No active faults — system is healthy! ✅" diff --git a/demos/moveit_pick_place/move-arm.sh b/demos/moveit_pick_place/move-arm.sh index 91af056..7d5593a 100755 --- a/demos/moveit_pick_place/move-arm.sh +++ b/demos/moveit_pick_place/move-arm.sh @@ -12,13 +12,19 @@ set -eu -CONTAINER="${CONTAINER_NAME:-$(docker ps --format '{{.Names}}' | grep -E '^moveit_medkit_demo(_nvidia)?(_local)?$' | head -n1)}" ACTION="/panda_arm_controller/follow_joint_trajectory" JOINT_NAMES='["panda_joint1","panda_joint2","panda_joint3","panda_joint4","panda_joint5","panda_joint6","panda_joint7"]' # Duration in seconds for trajectory execution DURATION_SEC=3 +# Time limit for one `ros2 action send_goal` run, and how many runs per goal. +# The controller can fail to deliver the goal response to a new CLI +# ("Failed to send goal response"). It then never runs the goal and the CLI +# waits forever, so a goal with no response is sent again. +SEND_TIMEOUT_SEC=30 +SEND_ATTEMPTS=3 + # --- Preset joint positions (radians) --- # Ready: default MoveIt pose (from SRDF) READY="[0.0, -0.785, 0.0, -2.356, 0.0, 1.571, 0.785]" @@ -45,6 +51,51 @@ RIGHT="[1.5, -0.785, 0.0, -2.356, 0.0, 1.571, 0.785]" WAVE="[0.0, -1.0, 0.0, -0.5, 0.0, 2.5, 0.785]" +have() { + command -v "$1" &> /dev/null +} + +# True only if a LOCAL ros2 can actually reach the target action server. +# `ros2 node list` exits 0 even on an empty graph (wrong ROS_DOMAIN_ID, no +# multicast route), so a host with ROS 2 sourced but not connected to the +# demo looks identical to being inside the container. Checking that the +# action itself is listed avoids that false positive. A cold listing (no +# ros2 daemon yet) can miss a running server, so a miss is listed once more. +can_reach_action_locally() { + have ros2 || return 1 + ros2 action list 2> /dev/null | grep -qFx "${ACTION}" \ + || ros2 action list 2> /dev/null | grep -qFx "${ACTION}" +} + +# Picks how goals are sent: USE_LOCAL=true for the local ros2, else +# `docker exec` into CONTAINER. Without a docker CLI the script runs inside +# the container or on a ROS host, so the local ros2 is the only way. +# On failure sets NOT_SENT_REASON and returns 1. +USE_LOCAL="" +CONTAINER="" +NOT_SENT_REASON="" +choose_transport() { + [[ -z "${USE_LOCAL}" ]] || return 0 + if ! have docker; then + if ! have ros2; then + NOT_SENT_REASON="needs the docker CLI or a sourced ROS 2 (ros2 CLI), found neither" + return 1 + fi + USE_LOCAL=true + return 0 + fi + if can_reach_action_locally; then + USE_LOCAL=true + return 0 + fi + CONTAINER="${CONTAINER_NAME:-$(docker ps --format '{{.Names}}' | grep -E '^moveit_medkit_demo(_nvidia)?(_local)?$' | head -n1)}" + if [[ -z "${CONTAINER}" ]]; then + NOT_SENT_REASON="no running moveit_medkit_demo container, start it with ./run-demo.sh or set CONTAINER_NAME" + return 1 + fi + USE_LOCAL=false +} + send_trajectory() { local positions="$1" local label="$2" @@ -64,27 +115,76 @@ send_trajectory() { } }" - # Check if we're inside the container or outside - if command -v ros2 &> /dev/null && ros2 node list &> /dev/null 2>&1; then - # Inside the container (or ROS 2 env is set up) - ros2 action send_goal "${ACTION}" \ - control_msgs/action/FollowJointTrajectory \ - "${goal_msg}" \ - --feedback - else - # Outside — exec into container - docker exec -it "${CONTAINER}" bash -c " - source /opt/ros/jazzy/setup.bash && \ - source /root/demo_ws/install/setup.bash && \ - ros2 action send_goal ${ACTION} \ - control_msgs/action/FollowJointTrajectory \ - \"${goal_msg}\" \ - --feedback - " + if ! choose_transport; then + echo "Failed: ${label} (goal not sent: ${NOT_SENT_REASON})" >&2 + return 1 fi + # `ros2 action send_goal` always exits 0, whatever the goal's outcome - + # the real result is in its own printed "Goal finished with status:" + # line, so capture output and parse that instead of the exit code. + # PYTHONUNBUFFERED keeps the lines printed before a timeout kills the CLI. + local attempt output rc + for ((attempt = 1; attempt <= SEND_ATTEMPTS; attempt++)); do + rc=0 + if [[ "${USE_LOCAL}" == true ]]; then + output=$(PYTHONUNBUFFERED=1 timeout "${SEND_TIMEOUT_SEC}" \ + ros2 action send_goal "${ACTION}" \ + control_msgs/action/FollowJointTrajectory \ + "${goal_msg}" \ + --feedback 2>&1) || rc=$? + else + # Outside the container: exec into it. No -it: this must also + # work without a TTY (CI, a pipe), and the command needs no stdin. + # timeout runs in the container: killing `docker exec` would + # leave the CLI running there. + output=$(docker exec "${CONTAINER}" bash -c " + source /opt/ros/jazzy/setup.bash && \ + source /root/demo_ws/install/setup.bash && \ + PYTHONUNBUFFERED=1 timeout ${SEND_TIMEOUT_SEC} \ + ros2 action send_goal ${ACTION} \ + control_msgs/action/FollowJointTrajectory \ + \"${goal_msg}\" \ + --feedback + " 2>&1) || rc=$? + fi + printf '%s\n' "${output}" + # An accepted or rejected goal has its answer. + if grep -qE '^(Goal accepted with ID|Goal was rejected)' <<< "${output}"; then + break + fi + # Never sent (no container, no action server, a CLI error): sending + # again changes nothing. rc 124 is the timeout. + if ! grep -q '^Sending goal:' <<< "${output}"; then + if ((rc == 124)) && grep -q '^Waiting for an action server' <<< "${output}"; then + echo "Failed: ${label} (goal not sent: no action server ${ACTION} within ${SEND_TIMEOUT_SEC} s)" >&2 + elif ((rc == 124)); then + echo "Failed: ${label} (goal not sent: timed out after ${SEND_TIMEOUT_SEC} s)" >&2 + else + echo "Failed: ${label} (goal not sent: exit status ${rc})" >&2 + fi + return 1 + fi + # Sent with no response: the controller never runs it, so send again. + if ((attempt < SEND_ATTEMPTS)); then + if ((rc == 124)); then + echo "No goal response within ${SEND_TIMEOUT_SEC} s, sending the goal again" + else + echo "No goal response (exit status ${rc}), sending the goal again" + fi + fi + done + + local status + status=$(printf '%s\n' "${output}" | grep -F 'Goal finished with status:' | tail -n1 | sed -E 's/.*status: *//') + echo "" - echo "✅ Done: ${label}" + if [[ "${status}" == "SUCCEEDED" ]]; then + echo "✅ Done: ${label}" + return 0 + fi + echo "Failed: ${label} (status: ${status:-UNKNOWN})" >&2 + return 1 } show_menu() { @@ -110,13 +210,19 @@ show_menu() { run_demo_cycle() { echo "🔄 Running pick → place → home cycle..." echo "" - send_trajectory "${PICK}" "pick" + local failed=0 + send_trajectory "${PICK}" "pick" || failed=1 sleep 2 - send_trajectory "${PLACE}" "place" + send_trajectory "${PLACE}" "place" || failed=1 sleep 2 - send_trajectory "${READY}" "ready (home)" + send_trajectory "${READY}" "ready (home)" || failed=1 echo "" - echo "🔄 Cycle complete!" + if [[ "${failed}" -eq 0 ]]; then + echo "🔄 Cycle complete!" + else + echo "🔄 Cycle complete with failures" >&2 + fi + return "${failed}" } handle_choice() { @@ -138,15 +244,20 @@ handle_choice() { # --- Main --- -# If argument provided, run directly +# If argument provided, run directly. Reflect the goal's real result in the +# exit code instead of always exiting 0. if [[ $# -gt 0 ]]; then - handle_choice "$1" - exit 0 + if handle_choice "$1"; then + exit 0 + else + exit 1 + fi fi -# Interactive mode +# Interactive mode. A failed goal reports failure and the menu continues - +# it must not kill the session (set -e would, without this guard). while true; do show_menu read -rp "Choose position (1-8, d, q): " choice - handle_choice "${choice}" + handle_choice "${choice}" || true done diff --git a/demos/multi_ecu_aggregation/Dockerfile b/demos/multi_ecu_aggregation/Dockerfile index 4f1363f..a446e4e 100644 --- a/demos/multi_ecu_aggregation/Dockerfile +++ b/demos/multi_ecu_aggregation/Dockerfile @@ -49,10 +49,6 @@ COPY src/ ${COLCON_WS}/src/multi_ecu_demo/src/ COPY config/ ${COLCON_WS}/src/multi_ecu_demo/config/ COPY launch/ ${COLCON_WS}/src/multi_ecu_demo/launch/ -# Copy container scripts -COPY container_scripts/ /var/lib/ros2_medkit/scripts/ -RUN find /var/lib/ros2_medkit/scripts -name "*.bash" -exec chmod +x {} \; - # Build all packages (skip test dependencies that aren't in ros-base) WORKDIR ${COLCON_WS} RUN bash -c "source /opt/ros/jazzy/setup.bash && \ @@ -61,6 +57,11 @@ RUN bash -c "source /opt/ros/jazzy/setup.bash && \ --skip-keys='ament_cmake_clang_format ament_cmake_clang_tidy test_msgs example_interfaces sqlite3' && \ MAKEFLAGS='-j 2' colcon build --executor sequential --symlink-install --cmake-args -DBUILD_TESTING=OFF" +# Copy container scripts. Not a build input: kept after colcon build so a +# script-only change does not invalidate the compile cache. +COPY container_scripts/ /var/lib/ros2_medkit/scripts/ +RUN find /var/lib/ros2_medkit/scripts -name "*.bash" -exec chmod +x {} \; + # Setup environment RUN echo "source /opt/ros/jazzy/setup.bash" >> ~/.bashrc && \ echo "source ${COLCON_WS}/install/setup.bash" >> ~/.bashrc diff --git a/demos/multi_ecu_aggregation/README.md b/demos/multi_ecu_aggregation/README.md index 6e052a1..4cc0803 100644 --- a/demos/multi_ecu_aggregation/README.md +++ b/demos/multi_ecu_aggregation/README.md @@ -337,6 +337,31 @@ curl -X POST http://localhost:8080/api/v1/components/actuation-ecu/scripts/injec -d '{"execution_type":"now"}' | jq ``` +### How the Scripts Change Parameters + +Each script sets node parameters through its own ECU's gateway, with the +configuration API. The same call works from the host through the perception +gateway: + +```bash +curl -X PUT http://localhost:8080/api/v1/apps/path-planner/configurations/planning_delay_ms \ + -H "Content-Type: application/json" -d '{"value": 5000}' +``` + +A script whose write is refused exits non-zero, and the execution's error +message names each failed write, for example +`FAIL: gripper-controller/inject_jam (HTTP 409)`. `restore-normal` clears the +ECU's faults only after all its writes succeeded. It clears twice, 2 s apart, +and the second clear decides the result. If that clear fails, the script exits +non-zero and the error names the status it got, for example +`FAIL: clear faults (HTTP 503)` while the ECU's fault manager does not answer. +The ECU may then still hold faults. + +`path-planner` runs its planning timer in its own callback group, so its +parameters stay writable while `inject-planning-delay` is active, and a new +`planning_delay_ms` also shortens the cycle already in progress. +`restore-normal` right after `inject-planning-delay` takes a few seconds. + ### Available Scripts per ECU | ECU | Script | Description | diff --git a/demos/multi_ecu_aggregation/container_scripts/actuation-ecu/inject-gripper-jam/script.bash b/demos/multi_ecu_aggregation/container_scripts/actuation-ecu/inject-gripper-jam/script.bash index 6913093..7f05905 100644 --- a/demos/multi_ecu_aggregation/container_scripts/actuation-ecu/inject-gripper-jam/script.bash +++ b/demos/multi_ecu_aggregation/container_scripts/actuation-ecu/inject-gripper-jam/script.bash @@ -2,14 +2,32 @@ # Inject gripper jam - gripper controller stuck set -eu -# ROS setup.bash dereferences AMENT_TRACE_SETUP_FILES; relax nounset around it. -set +u -# shellcheck source=/dev/null -source /opt/ros/jazzy/setup.bash -# shellcheck source=/dev/null -source /root/demo_ws/install/setup.bash -set -u - -ros2 param set /actuation/gripper_controller inject_jam true +GATEWAY_URL="${GATEWAY_URL:-http://localhost:8080}" +API_BASE="${GATEWAY_URL}/api/v1" + +ERRORS=0 + +# Sets one parameter through this ECU's gateway. A refused write is named on +# stderr, which the Scripts API returns as the error message. +put_config() { + local app="$1" param="$2" value="$3" code + code=$(curl -s -m 30 -o /dev/null -w '%{http_code}' -X PUT \ + "${API_BASE}/apps/${app}/configurations/${param}" \ + -H "Content-Type: application/json" -d "{\"value\": ${value}}") || true + case "$code" in + 2??) echo "${app}: ${param}=${value}" ;; + *) + echo "FAIL: ${app}/${param} (HTTP ${code})" >&2 + ERRORS=$((ERRORS + 1)) + ;; + esac +} + +put_config gripper-controller inject_jam true + +if [ "$ERRORS" -gt 0 ]; then + echo "${ERRORS} parameter write(s) failed" >&2 + exit 1 +fi echo '{"status": "injected", "parameter": "inject_jam", "value": true}' diff --git a/demos/multi_ecu_aggregation/container_scripts/actuation-ecu/restore-normal/script.bash b/demos/multi_ecu_aggregation/container_scripts/actuation-ecu/restore-normal/script.bash index a7d6e72..02005f6 100644 --- a/demos/multi_ecu_aggregation/container_scripts/actuation-ecu/restore-normal/script.bash +++ b/demos/multi_ecu_aggregation/container_scripts/actuation-ecu/restore-normal/script.bash @@ -1,41 +1,62 @@ #!/bin/bash -# Reset all actuation node parameters to defaults +# Reset all actuation node parameters to defaults and clear faults set -eu -# ROS setup.bash dereferences AMENT_TRACE_SETUP_FILES; relax nounset around it. -set +u -# shellcheck source=/dev/null -source /opt/ros/jazzy/setup.bash -# shellcheck source=/dev/null -source /root/demo_ws/install/setup.bash -set -u +GATEWAY_URL="${GATEWAY_URL:-http://localhost:8080}" +API_BASE="${GATEWAY_URL}/api/v1" ERRORS=0 +# Sets one parameter through this ECU's gateway. A refused write is named on +# stderr, which the Scripts API returns as the error message. +put_config() { + local app="$1" param="$2" value="$3" code + code=$(curl -s -m 30 -o /dev/null -w '%{http_code}' -X PUT \ + "${API_BASE}/apps/${app}/configurations/${param}" \ + -H "Content-Type: application/json" -d "{\"value\": ${value}}") || true + case "$code" in + 2??) echo "${app}: ${param}=${value}" ;; + *) + echo "FAIL: ${app}/${param} (HTTP ${code})" >&2 + ERRORS=$((ERRORS + 1)) + ;; + esac +} + # Motor controller -ros2 param set /actuation/motor_controller torque_noise 0.01 || ERRORS=$((ERRORS + 1)) -ros2 param set /actuation/motor_controller failure_probability 0.0 || ERRORS=$((ERRORS + 1)) +put_config motor-controller torque_noise 0.01 +put_config motor-controller failure_probability 0.0 # Joint driver -ros2 param set /actuation/joint_driver inject_overheat false || ERRORS=$((ERRORS + 1)) -ros2 param set /actuation/joint_driver drift_rate 0.0 || ERRORS=$((ERRORS + 1)) -ros2 param set /actuation/joint_driver failure_probability 0.0 || ERRORS=$((ERRORS + 1)) +put_config joint-driver inject_overheat false +put_config joint-driver drift_rate 0.0 +put_config joint-driver failure_probability 0.0 # Gripper controller -ros2 param set /actuation/gripper_controller inject_jam false || ERRORS=$((ERRORS + 1)) -ros2 param set /actuation/gripper_controller failure_probability 0.0 || ERRORS=$((ERRORS + 1)) +put_config gripper-controller inject_jam false +put_config gripper-controller failure_probability 0.0 -if [ $ERRORS -gt 0 ]; then - echo "{\"status\": \"partial\", \"errors\": $ERRORS}" +if [ "$ERRORS" -gt 0 ]; then + echo "${ERRORS} parameter write(s) failed" >&2 exit 1 fi -# Clear faults -GATEWAY_URL="${GATEWAY_URL:-http://localhost:8080}" -API_BASE="${GATEWAY_URL}/api/v1" +# Clears the faults of this ECU's fault manager. Prints the HTTP status. +clear_faults() { + curl -s -m 30 -o /dev/null -w '%{http_code}' -X DELETE "${API_BASE}/faults" || true +} + +# The second clear decides the result. echo "Clearing faults..." -curl -sf -X DELETE "${API_BASE}/faults" > /dev/null 2>&1 || true +clear_faults > /dev/null sleep 2 -curl -sf -X DELETE "${API_BASE}/faults" > /dev/null 2>&1 || true +code=$(clear_faults) +case "$code" in + 2??) ;; + *) + echo "FAIL: clear faults (HTTP ${code})" >&2 + exit 1 + ;; +esac echo '{"status": "restored", "ecu": "actuation"}' diff --git a/demos/multi_ecu_aggregation/container_scripts/perception-ecu/inject-sensor-failure/script.bash b/demos/multi_ecu_aggregation/container_scripts/perception-ecu/inject-sensor-failure/script.bash index 4d784ce..8e4edc6 100644 --- a/demos/multi_ecu_aggregation/container_scripts/perception-ecu/inject-sensor-failure/script.bash +++ b/demos/multi_ecu_aggregation/container_scripts/perception-ecu/inject-sensor-failure/script.bash @@ -2,14 +2,32 @@ # Inject LiDAR sensor failure - high failure probability set -eu -# ROS setup.bash dereferences AMENT_TRACE_SETUP_FILES; relax nounset around it. -set +u -# shellcheck source=/dev/null -source /opt/ros/jazzy/setup.bash -# shellcheck source=/dev/null -source /root/demo_ws/install/setup.bash -set -u - -ros2 param set /perception/lidar_driver failure_probability 0.8 +GATEWAY_URL="${GATEWAY_URL:-http://localhost:8080}" +API_BASE="${GATEWAY_URL}/api/v1" + +ERRORS=0 + +# Sets one parameter through this ECU's gateway. A refused write is named on +# stderr, which the Scripts API returns as the error message. +put_config() { + local app="$1" param="$2" value="$3" code + code=$(curl -s -m 30 -o /dev/null -w '%{http_code}' -X PUT \ + "${API_BASE}/apps/${app}/configurations/${param}" \ + -H "Content-Type: application/json" -d "{\"value\": ${value}}") || true + case "$code" in + 2??) echo "${app}: ${param}=${value}" ;; + *) + echo "FAIL: ${app}/${param} (HTTP ${code})" >&2 + ERRORS=$((ERRORS + 1)) + ;; + esac +} + +put_config lidar-driver failure_probability 0.8 + +if [ "$ERRORS" -gt 0 ]; then + echo "${ERRORS} parameter write(s) failed" >&2 + exit 1 +fi echo '{"status": "injected", "parameter": "failure_probability", "value": 0.8}' diff --git a/demos/multi_ecu_aggregation/container_scripts/perception-ecu/restore-normal/script.bash b/demos/multi_ecu_aggregation/container_scripts/perception-ecu/restore-normal/script.bash index 23d7e73..f9c5a5b 100644 --- a/demos/multi_ecu_aggregation/container_scripts/perception-ecu/restore-normal/script.bash +++ b/demos/multi_ecu_aggregation/container_scripts/perception-ecu/restore-normal/script.bash @@ -1,49 +1,70 @@ #!/bin/bash -# Reset all perception node parameters to defaults +# Reset all perception node parameters to defaults and clear faults set -eu -# ROS setup.bash dereferences AMENT_TRACE_SETUP_FILES; relax nounset around it. -set +u -# shellcheck source=/dev/null -source /opt/ros/jazzy/setup.bash -# shellcheck source=/dev/null -source /root/demo_ws/install/setup.bash -set -u +GATEWAY_URL="${GATEWAY_URL:-http://localhost:8080}" +API_BASE="${GATEWAY_URL}/api/v1" ERRORS=0 +# Sets one parameter through this ECU's gateway. A refused write is named on +# stderr, which the Scripts API returns as the error message. +put_config() { + local app="$1" param="$2" value="$3" code + code=$(curl -s -m 30 -o /dev/null -w '%{http_code}' -X PUT \ + "${API_BASE}/apps/${app}/configurations/${param}" \ + -H "Content-Type: application/json" -d "{\"value\": ${value}}") || true + case "$code" in + 2??) echo "${app}: ${param}=${value}" ;; + *) + echo "FAIL: ${app}/${param} (HTTP ${code})" >&2 + ERRORS=$((ERRORS + 1)) + ;; + esac +} + # LiDAR driver -ros2 param set /perception/lidar_driver failure_probability 0.0 || ERRORS=$((ERRORS + 1)) -ros2 param set /perception/lidar_driver inject_nan false || ERRORS=$((ERRORS + 1)) -ros2 param set /perception/lidar_driver noise_stddev 0.01 || ERRORS=$((ERRORS + 1)) -ros2 param set /perception/lidar_driver drift_rate 0.0 || ERRORS=$((ERRORS + 1)) +put_config lidar-driver failure_probability 0.0 +put_config lidar-driver inject_nan false +put_config lidar-driver noise_stddev 0.01 +put_config lidar-driver drift_rate 0.0 # Camera driver -ros2 param set /perception/camera_driver failure_probability 0.0 || ERRORS=$((ERRORS + 1)) -ros2 param set /perception/camera_driver noise_level 0.0 || ERRORS=$((ERRORS + 1)) -ros2 param set /perception/camera_driver inject_black_frames false || ERRORS=$((ERRORS + 1)) +put_config camera-driver failure_probability 0.0 +put_config camera-driver noise_level 0.0 +put_config camera-driver inject_black_frames false # Point cloud filter -ros2 param set /perception/point_cloud_filter failure_probability 0.0 || ERRORS=$((ERRORS + 1)) -ros2 param set /perception/point_cloud_filter drop_rate 0.0 || ERRORS=$((ERRORS + 1)) -ros2 param set /perception/point_cloud_filter delay_ms 0 || ERRORS=$((ERRORS + 1)) +put_config point-cloud-filter failure_probability 0.0 +put_config point-cloud-filter drop_rate 0.0 +put_config point-cloud-filter delay_ms 0 # Object detector -ros2 param set /perception/object_detector failure_probability 0.0 || ERRORS=$((ERRORS + 1)) -ros2 param set /perception/object_detector false_positive_rate 0.0 || ERRORS=$((ERRORS + 1)) -ros2 param set /perception/object_detector miss_rate 0.0 || ERRORS=$((ERRORS + 1)) +put_config object-detector failure_probability 0.0 +put_config object-detector false_positive_rate 0.0 +put_config object-detector miss_rate 0.0 -if [ $ERRORS -gt 0 ]; then - echo "{\"status\": \"partial\", \"errors\": $ERRORS}" +if [ "$ERRORS" -gt 0 ]; then + echo "${ERRORS} parameter write(s) failed" >&2 exit 1 fi -# Clear faults -GATEWAY_URL="${GATEWAY_URL:-http://localhost:8080}" -API_BASE="${GATEWAY_URL}/api/v1" +# Clears the faults of this ECU's fault manager. Prints the HTTP status. +clear_faults() { + curl -s -m 30 -o /dev/null -w '%{http_code}' -X DELETE "${API_BASE}/faults" || true +} + +# The second clear decides the result. echo "Clearing faults..." -curl -sf -X DELETE "${API_BASE}/faults" > /dev/null 2>&1 || true +clear_faults > /dev/null sleep 2 -curl -sf -X DELETE "${API_BASE}/faults" > /dev/null 2>&1 || true +code=$(clear_faults) +case "$code" in + 2??) ;; + *) + echo "FAIL: clear faults (HTTP ${code})" >&2 + exit 1 + ;; +esac echo '{"status": "restored", "ecu": "perception"}' diff --git a/demos/multi_ecu_aggregation/container_scripts/planning-ecu/inject-planning-delay/script.bash b/demos/multi_ecu_aggregation/container_scripts/planning-ecu/inject-planning-delay/script.bash index d186826..cbc93d7 100644 --- a/demos/multi_ecu_aggregation/container_scripts/planning-ecu/inject-planning-delay/script.bash +++ b/demos/multi_ecu_aggregation/container_scripts/planning-ecu/inject-planning-delay/script.bash @@ -2,14 +2,32 @@ # Inject path planning delay - 5000ms processing time set -eu -# ROS setup.bash dereferences AMENT_TRACE_SETUP_FILES; relax nounset around it. -set +u -# shellcheck source=/dev/null -source /opt/ros/jazzy/setup.bash -# shellcheck source=/dev/null -source /root/demo_ws/install/setup.bash -set -u - -ros2 param set /planning/path_planner planning_delay_ms 5000 +GATEWAY_URL="${GATEWAY_URL:-http://localhost:8080}" +API_BASE="${GATEWAY_URL}/api/v1" + +ERRORS=0 + +# Sets one parameter through this ECU's gateway. A refused write is named on +# stderr, which the Scripts API returns as the error message. +put_config() { + local app="$1" param="$2" value="$3" code + code=$(curl -s -m 30 -o /dev/null -w '%{http_code}' -X PUT \ + "${API_BASE}/apps/${app}/configurations/${param}" \ + -H "Content-Type: application/json" -d "{\"value\": ${value}}") || true + case "$code" in + 2??) echo "${app}: ${param}=${value}" ;; + *) + echo "FAIL: ${app}/${param} (HTTP ${code})" >&2 + ERRORS=$((ERRORS + 1)) + ;; + esac +} + +put_config path-planner planning_delay_ms 5000 + +if [ "$ERRORS" -gt 0 ]; then + echo "${ERRORS} parameter write(s) failed" >&2 + exit 1 +fi echo '{"status": "injected", "parameter": "planning_delay_ms", "value": 5000}' diff --git a/demos/multi_ecu_aggregation/container_scripts/planning-ecu/restore-normal/script.bash b/demos/multi_ecu_aggregation/container_scripts/planning-ecu/restore-normal/script.bash index 4ff9ced..e7cf83d 100644 --- a/demos/multi_ecu_aggregation/container_scripts/planning-ecu/restore-normal/script.bash +++ b/demos/multi_ecu_aggregation/container_scripts/planning-ecu/restore-normal/script.bash @@ -1,40 +1,61 @@ #!/bin/bash -# Reset all planning node parameters to defaults +# Reset all planning node parameters to defaults and clear faults set -eu -# ROS setup.bash dereferences AMENT_TRACE_SETUP_FILES; relax nounset around it. -set +u -# shellcheck source=/dev/null -source /opt/ros/jazzy/setup.bash -# shellcheck source=/dev/null -source /root/demo_ws/install/setup.bash -set -u +GATEWAY_URL="${GATEWAY_URL:-http://localhost:8080}" +API_BASE="${GATEWAY_URL}/api/v1" ERRORS=0 +# Sets one parameter through this ECU's gateway. A refused write is named on +# stderr, which the Scripts API returns as the error message. +put_config() { + local app="$1" param="$2" value="$3" code + code=$(curl -s -m 30 -o /dev/null -w '%{http_code}' -X PUT \ + "${API_BASE}/apps/${app}/configurations/${param}" \ + -H "Content-Type: application/json" -d "{\"value\": ${value}}") || true + case "$code" in + 2??) echo "${app}: ${param}=${value}" ;; + *) + echo "FAIL: ${app}/${param} (HTTP ${code})" >&2 + ERRORS=$((ERRORS + 1)) + ;; + esac +} + # Path planner -ros2 param set /planning/path_planner planning_delay_ms 0 || ERRORS=$((ERRORS + 1)) -ros2 param set /planning/path_planner failure_probability 0.0 || ERRORS=$((ERRORS + 1)) +put_config path-planner planning_delay_ms 0 +put_config path-planner failure_probability 0.0 # Behavior planner -ros2 param set /planning/behavior_planner inject_wrong_direction false || ERRORS=$((ERRORS + 1)) -ros2 param set /planning/behavior_planner failure_probability 0.0 || ERRORS=$((ERRORS + 1)) +put_config behavior-planner inject_wrong_direction false +put_config behavior-planner failure_probability 0.0 # Task scheduler -ros2 param set /planning/task_scheduler inject_stuck false || ERRORS=$((ERRORS + 1)) -ros2 param set /planning/task_scheduler failure_probability 0.0 || ERRORS=$((ERRORS + 1)) +put_config task-scheduler inject_stuck false +put_config task-scheduler failure_probability 0.0 -if [ $ERRORS -gt 0 ]; then - echo "{\"status\": \"partial\", \"errors\": $ERRORS}" +if [ "$ERRORS" -gt 0 ]; then + echo "${ERRORS} parameter write(s) failed" >&2 exit 1 fi -# Clear faults -GATEWAY_URL="${GATEWAY_URL:-http://localhost:8080}" -API_BASE="${GATEWAY_URL}/api/v1" +# Clears the faults of this ECU's fault manager. Prints the HTTP status. +clear_faults() { + curl -s -m 30 -o /dev/null -w '%{http_code}' -X DELETE "${API_BASE}/faults" || true +} + +# The second clear decides the result. echo "Clearing faults..." -curl -sf -X DELETE "${API_BASE}/faults" > /dev/null 2>&1 || true +clear_faults > /dev/null sleep 2 -curl -sf -X DELETE "${API_BASE}/faults" > /dev/null 2>&1 || true +code=$(clear_faults) +case "$code" in + 2??) ;; + *) + echo "FAIL: clear faults (HTTP ${code})" >&2 + exit 1 + ;; +esac echo '{"status": "restored", "ecu": "planning"}' diff --git a/demos/multi_ecu_aggregation/src/planning/path_planner.cpp b/demos/multi_ecu_aggregation/src/planning/path_planner.cpp index d07c724..7d54558 100644 --- a/demos/multi_ecu_aggregation/src/planning/path_planner.cpp +++ b/demos/multi_ecu_aggregation/src/planning/path_planner.cpp @@ -8,13 +8,19 @@ /// 10-waypoint path in the "map" frame. Offsets the path when detections are /// present to simulate obstacle avoidance. Supports artificial planning delay /// and failure probability for fault injection. +/// +/// The planning timer runs in its own callback group on a multi-threaded +/// executor, so parameter services answer while a planning cycle waits out +/// the injected delay. +#include #include #include +#include #include +#include #include #include -#include #include #include "diagnostic_msgs/msg/diagnostic_array.hpp" @@ -53,6 +59,8 @@ class PathPlanner : public rclcpp::Node "/perception/detections", 10, std::bind(&PathPlanner::on_detections, this, std::placeholders::_1)); + planning_group_ = this->create_callback_group(rclcpp::CallbackGroupType::MutuallyExclusive); + // Create timer (with rate validation) double rate = planning_rate_; if (rate <= 0.0) { @@ -66,7 +74,7 @@ class PathPlanner : public rclcpp::Node auto period = std::chrono::duration(1.0 / rate); timer_ = this->create_wall_timer( std::chrono::duration_cast(period), - std::bind(&PathPlanner::plan_path, this)); + std::bind(&PathPlanner::plan_path, this), planning_group_); // Register parameter callback param_callback_handle_ = this->add_on_set_parameters_callback( @@ -91,12 +99,17 @@ class PathPlanner : public rclcpp::Node for (const auto & param : parameters) { if (param.get_name() == "planning_delay_ms") { - planning_delay_ms_ = param.as_int(); - RCLCPP_INFO(this->get_logger(), "Planning delay changed to %ld ms", planning_delay_ms_); + { + std::lock_guard lock(delay_mutex_); + planning_delay_ms_ = param.as_int(); + } + delay_changed_.notify_all(); + RCLCPP_INFO( + this->get_logger(), "Planning delay changed to %ld ms", planning_delay_ms_.load()); } else if (param.get_name() == "failure_probability") { failure_probability_ = param.as_double(); RCLCPP_INFO( - this->get_logger(), "Failure probability changed to %.2f", failure_probability_); + this->get_logger(), "Failure probability changed to %.2f", failure_probability_.load()); } else if (param.get_name() == "planning_rate") { double rate = param.as_double(); if (rate <= 0.0) { @@ -108,7 +121,7 @@ class PathPlanner : public rclcpp::Node auto period = std::chrono::duration(1.0 / rate); timer_ = this->create_wall_timer( std::chrono::duration_cast(period), - std::bind(&PathPlanner::plan_path, this)); + std::bind(&PathPlanner::plan_path, this), planning_group_); RCLCPP_INFO(this->get_logger(), "Planning rate changed to %.1f Hz", planning_rate_); } } @@ -125,11 +138,18 @@ class PathPlanner : public rclcpp::Node { plan_count_++; - // Intentional blocking sleep to simulate slow computation pipeline. - // This blocks the single-threaded executor, preventing parameter changes - // and other callbacks from being processed during the delay. - if (planning_delay_ms_ > 0) { - std::this_thread::sleep_for(std::chrono::milliseconds(planning_delay_ms_)); + // Simulated slow computation: the cycle waits planning_delay_ms before it + // publishes. A new planning_delay_ms applies to the cycle in progress. + { + const auto start = std::chrono::steady_clock::now(); + std::unique_lock lock(delay_mutex_); + while (true) { + const auto deadline = start + std::chrono::milliseconds(planning_delay_ms_.load()); + if (std::chrono::steady_clock::now() >= deadline) { + break; + } + delay_changed_.wait_until(lock, deadline); + } } // Check for failure injection @@ -175,11 +195,11 @@ class PathPlanner : public rclcpp::Node if (planning_delay_ms_ > 100) { publish_diagnostics( "SLOW_PLANNING", - "Planning delay: " + std::to_string(planning_delay_ms_) + " ms"); + "Planning delay: " + std::to_string(planning_delay_ms_.load()) + " ms"); } else if (last_detection_count_ > 5) { publish_diagnostics( "HIGH_OBSTACLE_COUNT", - "Avoiding " + std::to_string(last_detection_count_) + " detections"); + "Avoiding " + std::to_string(last_detection_count_.load()) + " detections"); } else { publish_diagnostics("OK", "Operating normally"); } @@ -212,11 +232,11 @@ class PathPlanner : public rclcpp::Node diag.values.push_back(kv); kv.key = "detection_count"; - kv.value = std::to_string(last_detection_count_); + kv.value = std::to_string(last_detection_count_.load()); diag.values.push_back(kv); kv.key = "planning_delay_ms"; - kv.value = std::to_string(planning_delay_ms_); + kv.value = std::to_string(planning_delay_ms_.load()); diag.values.push_back(kv); diag_array.status.push_back(diag); @@ -230,7 +250,8 @@ class PathPlanner : public rclcpp::Node // Subscription rclcpp::Subscription::SharedPtr detection_sub_; - // Timer + // Timer, in its own callback group so it never blocks parameter services + rclcpp::CallbackGroup::SharedPtr planning_group_; rclcpp::TimerBase::SharedPtr timer_; // Parameter callback @@ -240,13 +261,17 @@ class PathPlanner : public rclcpp::Node std::mt19937 rng_; std::uniform_real_distribution uniform_dist_; - // Parameters - int64_t planning_delay_ms_; - double failure_probability_; + // Parameters. The planning cycle reads them on another executor thread. + std::atomic planning_delay_ms_{0}; + std::atomic failure_probability_{0.0}; double planning_rate_; + // Wakes a waiting planning cycle when planning_delay_ms changes + std::mutex delay_mutex_; + std::condition_variable delay_changed_; + // State - int last_detection_count_{0}; + std::atomic last_detection_count_{0}; uint64_t plan_count_{0}; }; @@ -255,7 +280,11 @@ class PathPlanner : public rclcpp::Node int main(int argc, char ** argv) { rclcpp::init(argc, argv); - rclcpp::spin(std::make_shared()); + auto node = std::make_shared(); + // One thread for the planning group, one for everything else. + rclcpp::executors::MultiThreadedExecutor executor(rclcpp::ExecutorOptions(), 2); + executor.add_node(node); + executor.spin(); rclcpp::shutdown(); return 0; } diff --git a/demos/ota_nav2_sensor_fix/run-demo.sh b/demos/ota_nav2_sensor_fix/run-demo.sh index 7781801..dc18069 100755 --- a/demos/ota_nav2_sensor_fix/run-demo.sh +++ b/demos/ota_nav2_sensor_fix/run-demo.sh @@ -1,11 +1,11 @@ #!/bin/bash # OTA over SOVD - nav2 sensor-fix demo runner. # Brings up the gateway (with the dev-grade ota_update_plugin baked in) and -# the FastAPI artifact server. The gateway image bundles a full TurtleBot3 + -# Nav2 + headless Gazebo stack and runs foxglove_bridge on :8765, so the -# demo is self-contained: broken_lidar publishes /scan with a phantom -# obstacle that nav2 + a Foxglove 3D panel both react to. The OTA flow -# swaps broken_lidar -> fixed_lidar and the phantom disappears. +# the FastAPI artifact server. The gateway image bundles a full Robotnik +# RB-Theron AMR + Nav2 + headless Gazebo stack and runs foxglove_bridge on +# :8765, so the demo is self-contained: broken_lidar publishes /scan with a +# phantom obstacle that nav2 + a Foxglove 3D panel both react to. The OTA +# flow swaps broken_lidar -> fixed_lidar and the phantom disappears. set -eu @@ -203,6 +203,6 @@ echo " open http://localhost:5173 -> Connect -> ${GATEWAY_URL}" echo "" echo " Foxglove (recommended for the 3D narrative):" echo " Open connection -> Foxglove WebSocket -> ws://localhost:${OTA_FOXGLOVE_BRIDGE_PORT:-8765}" -echo " Add a 3D panel: TurtleBot3 in the world, /scan cone shows the phantom" +echo " Add a 3D panel: RB-Theron in the world, /scan cone shows the phantom" echo " Install ros2_medkit_foxglove_extension (npm run local-install) for the" echo " 'ros2_medkit Updates' panel; set baseUrl to ${GATEWAY_URL}/api/v1" diff --git a/demos/sensor_diagnostics/README.md b/demos/sensor_diagnostics/README.md index d89a033..4c994dd 100644 --- a/demos/sensor_diagnostics/README.md +++ b/demos/sensor_diagnostics/README.md @@ -206,8 +206,8 @@ The gateway supports condition-based triggers that fire when specific events occ ### How It Works -1. `setup-triggers.sh` creates a trigger via `POST /api/v1/apps/diagnostic_bridge/triggers`: - - **Resource:** `/api/v1/apps/diagnostic_bridge/faults` (watches fault collection) +1. `setup-triggers.sh` creates a trigger via `POST /api/v1/apps/diagnostic-bridge/triggers`: + - **Resource:** `/api/v1/apps/diagnostic-bridge/faults` (watches fault collection) - **Condition:** `OnChange` (fires on any new or updated fault) - **Multishot:** `true` (fires repeatedly, not just once) - **Lifetime:** 3600 seconds (auto-expires after 1 hour) @@ -218,23 +218,23 @@ The gateway supports condition-based triggers that fire when specific events occ ```bash # Create a trigger -curl -X POST http://localhost:8080/api/v1/apps/diagnostic_bridge/triggers \ +curl -X POST http://localhost:8080/api/v1/apps/diagnostic-bridge/triggers \ -H "Content-Type: application/json" \ -d '{ - "resource": "/api/v1/apps/diagnostic_bridge/faults", + "resource": "/api/v1/apps/diagnostic-bridge/faults", "trigger_condition": {"condition_type": "OnChange"}, "multishot": true, "lifetime": 3600 }' | jq # List triggers -curl http://localhost:8080/api/v1/apps/diagnostic_bridge/triggers | jq +curl http://localhost:8080/api/v1/apps/diagnostic-bridge/triggers | jq # Watch events (replace TRIGGER_ID) -curl -N http://localhost:8080/api/v1/apps/diagnostic_bridge/triggers/TRIGGER_ID/events +curl -N http://localhost:8080/api/v1/apps/diagnostic-bridge/triggers/TRIGGER_ID/events # Delete a trigger -curl -X DELETE http://localhost:8080/api/v1/apps/diagnostic_bridge/triggers/TRIGGER_ID +curl -X DELETE http://localhost:8080/api/v1/apps/diagnostic-bridge/triggers/TRIGGER_ID ``` ## API Examples @@ -242,14 +242,14 @@ curl -X DELETE http://localhost:8080/api/v1/apps/diagnostic_bridge/triggers/TRIG ### Read Sensor Data ```bash -# Get LiDAR scan -curl http://localhost:8080/api/v1/apps/lidar-sim/data/scan | jq '.ranges[:5]' +# Get LiDAR scan (topic id is /sensors/scan, percent-encoded in the URL) +curl http://localhost:8080/api/v1/apps/lidar-sim/data/sensors%2Fscan | jq '.data.ranges[:5]' -# Get IMU data -curl http://localhost:8080/api/v1/apps/imu-sim/data/imu | jq '.linear_acceleration' +# Get IMU data (topic id is /sensors/imu) +curl http://localhost:8080/api/v1/apps/imu-sim/data/sensors%2Fimu | jq '.data.linear_acceleration' -# Get GPS fix -curl http://localhost:8080/api/v1/apps/gps-sim/data/fix | jq '{lat: .latitude, lon: .longitude}' +# Get GPS fix (topic id is /sensors/fix) +curl http://localhost:8080/api/v1/apps/gps-sim/data/sensors%2Ffix | jq '{lat: .data.latitude, lon: .data.longitude}' ``` ### View Configurations @@ -304,7 +304,7 @@ curl http://localhost:8080/api/v1/faults | jq |--------|-------------| | `run-demo.sh` | Start Docker services (daemon mode) | | `stop-demo.sh` | Stop Docker services | -| `check-demo.sh` | Interactive API demonstration and exploration | +| `check-demo.sh` | Interactive API demonstration and exploration; waits for the sensor data after the demo starts (see below), exits 1 when the fault list cannot be read | | `run-diagnostics.sh` | Check health of all sensors | | `inject-fault-scenario.sh` | Composite fault injection across all sensors | | `inject-noise.sh` | Inject high noise fault | @@ -317,6 +317,16 @@ curl http://localhost:8080/api/v1/faults | jq > **Note:** All diagnostic scripts (`inject-*.sh`, `restore-normal.sh`, `run-diagnostics.sh`, `inject-fault-scenario.sh`) are also available via the [Scripts API](#scripts-api) - callable as REST endpoints without requiring the host-side scripts. +`check-demo.sh` waits up to 30 s for the gateway to link the sensor nodes, and up to 5 s for a first +message from each linked sensor. A sensor past its 5 s is still read while the wait goes on for others. +Set `DATA_WAIT_SEC=` to change the 30 s. A sensor that sends nothing, such as the IMU after +`inject-failure.sh`, is named in the output with the time it was waited for. Its data section reads it +again and says it has no data, and the fault sections still run. + +The fault sections show the first listed fault on the App whose node reported it. This includes faults +the anomaly detector reports as `/processing/anomaly_detector/`. For those faults the rosbag list +of `apps/anomaly-detector` can be empty. + ## Sensor Parameters ### LiDAR (`/sensors/lidar_sim`) diff --git a/demos/sensor_diagnostics/check-demo.sh b/demos/sensor_diagnostics/check-demo.sh index 48cec64..2cdf20b 100755 --- a/demos/sensor_diagnostics/check-demo.sh +++ b/demos/sensor_diagnostics/check-demo.sh @@ -45,6 +45,102 @@ if ! curl -sf "${API_BASE}/health" > /dev/null 2>&1; then fi echo_success "Gateway is healthy!" +# /health answers before the gateway links the sensor nodes, and a node's +# topic reads come back empty until then. A linked sensor that publishes +# nothing (inject-failure.sh) never gets a message, so it holds the wait for +# SAMPLE_WAIT_SEC at most; it is still read while the wait goes on for others. +# The whole wait ends within DATA_WAIT_SEC plus one request. +DATA_WAIT_SEC="${DATA_WAIT_SEC:-30}" +SAMPLE_WAIT_SEC=5 +REQUEST_TIMEOUT_SEC=3 +SENSOR_APPS=(lidar-sim imu-sim gps-sim) +SENSOR_TOPICS=(sensors/scan sensors/imu sensors/fix) + +case "$DATA_WAIT_SEC" in + '' | *[!0-9]*) + echo_error "DATA_WAIT_SEC must be a whole number of seconds, got '${DATA_WAIT_SEC}'." + exit 1 + ;; +esac +# Base 10: bash reads a number with a leading zero as octal. +DATA_WAIT_SEC=$((10#$DATA_WAIT_SEC)) + +# True when the gateway has linked APP to its node: its data list is not empty. +sensor_linked() { + curl -sf -m "$REQUEST_TIMEOUT_SEC" "${API_BASE}/apps/$1/data" \ + | jq -e '.items | length > 0' > /dev/null 2>&1 +} + +# True when the gateway has a message on APP's TOPIC. +sensor_has_data() { + curl -sf -m "$REQUEST_TIMEOUT_SEC" "${API_BASE}/apps/$1/data/${2//\//%2F}" \ + | jq -e '.data | type == "object" and length > 0' > /dev/null 2>&1 +} + +wait_start=$(date +%s) +deadline=$((wait_start + DATA_WAIT_SEC)) +# Per sensor: when the link was seen, when it was last read, whether a +# message was read, and which waiting line was printed. +linked_at=() +polled_at=() +has_data=() +announced=() +while :; do + pending=false + for i in "${!SENSOR_APPS[@]}"; do + [ -n "${has_data[$i]:-}" ] && continue + app=${SENSOR_APPS[$i]} + [ "$(date +%s)" -lt "$deadline" ] || break 2 + if [ -z "${linked_at[$i]:-}" ]; then + sensor_linked "$app" && linked_at[i]=$(date +%s) + polled_at[i]=$(date +%s) + if [ -z "${linked_at[$i]:-}" ]; then + if [ -z "${announced[$i]:-}" ]; then + echo "Waiting for the gateway to link ${app} (max ${DATA_WAIT_SEC}s)..." + announced[i]="link" + fi + pending=true + continue + fi + [ "$(date +%s)" -lt "$deadline" ] || break 2 + fi + if sensor_has_data "$app" "${SENSOR_TOPICS[$i]}"; then + has_data[i]=1 + continue + fi + polled_at[i]=$(date +%s) + # Past its first-message window a sensor no longer holds the wait. + [ "$(date +%s)" -lt "$((linked_at[i] + SAMPLE_WAIT_SEC))" ] || continue + if [ "${announced[$i]:-}" != sample ]; then + echo "Waiting for a first message from ${app} on /${SENSOR_TOPICS[$i]} (max ${SAMPLE_WAIT_SEC}s)..." + announced[i]="sample" + fi + pending=true + done + "$pending" || break + [ "$(date +%s)" -lt "$deadline" ] || break + sleep 1 +done + +# The time reported is how long each sensor was read for. The data sections +# below read every sensor again. +missing=false +for i in "${!SENSOR_APPS[@]}"; do + [ -n "${has_data[$i]:-}" ] && continue + missing=true + app=${SENSOR_APPS[$i]} + if [ -z "${polled_at[$i]:-}" ]; then + echo " ${app} was not waited for." + elif [ -n "${linked_at[$i]:-}" ]; then + echo " No message from ${app} on /${SENSOR_TOPICS[$i]} in the $((polled_at[i] - wait_start))s it was waited for; the sensor may have failed." + else + echo " The gateway did not link ${app} in the $((polled_at[i] - wait_start))s it was waited for." + fi +done +if "$missing"; then + echo " Sections 5-7 read each sensor again." +fi + echo_step "1. Checking Gateway Health" curl -s "${API_BASE}/health" | jq '.' @@ -52,70 +148,100 @@ echo_step "2. Listing All Areas (Namespaces)" curl -s "${API_BASE}/areas" | jq '.items[] | {id: .id, name: .name, description: .description}' echo_step "3. Listing All Components" -curl -s "${API_BASE}/components" | jq '.items[] | {id: .id, name: .name, area: .area}' +curl -s "${API_BASE}/components" | jq '.items[] | {id: .id, name: .name, description: .description}' echo_step "4. Listing All Apps (ROS 2 Nodes)" -curl -s "${API_BASE}/apps" | jq '.items[] | {id: .id, name: .name, namespace: .namespace}' +curl -s "${API_BASE}/apps" | jq '.items[] | {id: .id, name: .name, component: .["x-medkit"].component_id}' + +# Prints FILTER applied to the latest message on APP's TOPIC, or says that +# the gateway has none. +# Usage: show_sensor_data LABEL APP TOPIC FILTER +show_sensor_data() { + local body + body=$(curl -s -m 10 "${API_BASE}/apps/$2/data/${3//\//%2F}") + if echo "$body" | jq -e '.data | type == "object" and length > 0' > /dev/null 2>&1; then + echo "$body" | jq "$4" + else + echo " No $1 data: the gateway has no message from $2 on /$3." + fi +} echo_step "5. Reading LiDAR Data" echo "Getting latest scan from LiDAR simulator..." -curl -s "${API_BASE}/apps/lidar-sim/data/scan" | jq '{ - angle_min: .angle_min, - angle_max: .angle_max, - range_min: .range_min, - range_max: .range_max, - sample_ranges: .ranges[:5] +show_sensor_data "LiDAR" lidar-sim sensors/scan '{ + angle_min: .data.angle_min, + angle_max: .data.angle_max, + range_min: .data.range_min, + range_max: .data.range_max, + sample_ranges: .data.ranges[:5] }' echo_step "6. Reading IMU Data" echo "Getting latest IMU reading..." -curl -s "${API_BASE}/apps/imu-sim/data/imu" | jq '{ - linear_acceleration: .linear_acceleration, - angular_velocity: .angular_velocity +show_sensor_data "IMU" imu-sim sensors/imu '{ + linear_acceleration: .data.linear_acceleration, + angular_velocity: .data.angular_velocity }' echo_step "7. Reading GPS Fix" echo "Getting current GPS position..." -curl -s "${API_BASE}/apps/gps-sim/data/fix" | jq '{ - latitude: .latitude, - longitude: .longitude, - altitude: .altitude, - status: .status +show_sensor_data "GPS" gps-sim sensors/fix '{ + latitude: .data.latitude, + longitude: .data.longitude, + altitude: .data.altitude, + status: .data.status }' echo_step "8. Listing LiDAR Configurations" echo "These parameters can be modified at runtime to inject faults..." -curl -s "${API_BASE}/apps/lidar-sim/configurations" | jq '.items[] | {name: .name, value: .value, type: .type}' +# The list endpoint carries id/name/type only; the value and the ROS type are +# on each parameter's own detail endpoint. +LIDAR_CONFIG_IDS=$(curl -s "${API_BASE}/apps/lidar-sim/configurations" | jq -r '.items[]?.id' 2>/dev/null) +if [ -z "$LIDAR_CONFIG_IDS" ]; then + echo " No LiDAR configurations available." +fi +while IFS= read -r cfg_id; do + [ -n "$cfg_id" ] || continue + curl -s "${API_BASE}/apps/lidar-sim/configurations/${cfg_id}" \ + | jq '{name: .id, value: .data, type: .["x-medkit"].parameter.type}' +done <<< "$LIDAR_CONFIG_IDS" echo_step "9. Checking Current Faults" -FAULTS_JSON=$(curl -s "${API_BASE}/faults") +# A failed read leaves the fault list unknown, not empty. +FAULTS_RESPONSE=$(curl -s -w "\n%{http_code}" "${API_BASE}/faults") +FAULTS_CODE=$(tail -n 1 <<< "$FAULTS_RESPONSE") +FAULTS_JSON=$(sed '$d' <<< "$FAULTS_RESPONSE") +if [ "$FAULTS_CODE" != "200" ] || ! echo "$FAULTS_JSON" | jq -e '.items | type == "array"' > /dev/null 2>&1; then + echo_error "Could not read faults (HTTP ${FAULTS_CODE}): $(echo "$FAULTS_JSON" | jq -r '.message // empty' 2>/dev/null)" + echo " The fault list is unknown. Check that the fault manager is running, then retry." + exit 1 +fi echo "$FAULTS_JSON" | jq '.' # If there are faults, demonstrate snapshot / bulk-data endpoints FAULT_COUNT=$(echo "$FAULTS_JSON" | jq '.items | length') if [ "$FAULT_COUNT" -gt 0 ]; then - # Find the first fault that has both a non-null entity_id and code - FIRST_FAULT_ENTRY=$(echo "$FAULTS_JSON" | jq -r '.items[] | select(.entity_id != null and .code != null) | "\(.entity_type) \(.entity_id) \(.code)"' | head -n 1) + # The fault collection carries fault_code and reporting_sources (ROS node + # paths), not an entity id. The owning App is the one whose ROS node is the + # first reporting source or a path above it, whole segments only: the + # anomaly detector reports as /processing/anomaly_detector/. + FIRST_FAULT=$(echo "$FAULTS_JSON" | jq -r '.items[0].fault_code') + REPORTING_SOURCE=$(echo "$FAULTS_JSON" | jq -r '.items[0].reporting_sources[0] // empty') + FIRST_ENTITY=$(curl -s "${API_BASE}/apps" | jq -r --arg src "$REPORTING_SOURCE" ' + [.items[] | .["x-medkit"].ros2.node as $node + | select(($node | type) == "string" + and ($src == $node or ($src | startswith($node + "/")))) + | {id, depth: ($node | length)}] + | max_by(.depth) | .id // empty' 2>/dev/null) - if [ -z "$FIRST_FAULT_ENTRY" ]; then + if [ -z "$FIRST_ENTITY" ]; then echo "" - echo " Faults exist but none provide both 'entity_id' and 'code'." + echo " Could not map fault ${FIRST_FAULT} to a reporting App (source: ${REPORTING_SOURCE:-none})." echo " Skipping snapshot and bulk-data demonstration." else - FIRST_ENTITY_TYPE=$(echo "$FIRST_FAULT_ENTRY" | awk '{print $1}') - FIRST_ENTITY=$(echo "$FIRST_FAULT_ENTRY" | awk '{print $2}') - FIRST_FAULT=$(echo "$FIRST_FAULT_ENTRY" | awk '{print $3}') - # Map entity_type to plural resource path (e.g., "app" -> "apps") - case "$FIRST_ENTITY_TYPE" in - app|apps) ENTITY_PATH="apps" ;; - component|components) ENTITY_PATH="components" ;; - area|areas) ENTITY_PATH="areas" ;; - *) ENTITY_PATH="apps" ;; - esac - echo_step "10. Fault Detail with Environment Data (Snapshots)" - echo "Fetching fault ${FIRST_FAULT} on ${ENTITY_PATH}/${FIRST_ENTITY}..." - curl -s "${API_BASE}/${ENTITY_PATH}/${FIRST_ENTITY}/faults/${FIRST_FAULT}" | jq '{ + echo "Fetching fault ${FIRST_FAULT} on apps/${FIRST_ENTITY}..." + curl -s "${API_BASE}/apps/${FIRST_ENTITY}/faults/${FIRST_FAULT}" | jq '{ code: .item.code, status: .item.status, environment_data: { @@ -126,17 +252,22 @@ if [ "$FAULT_COUNT" -gt 0 ]; then echo_step "11. Bulk-Data Categories (Rosbag Recordings)" echo "Checking available bulk-data categories..." - curl -s "${API_BASE}/${ENTITY_PATH}/${FIRST_ENTITY}/bulk-data" | jq '.' + curl -s "${API_BASE}/apps/${FIRST_ENTITY}/bulk-data" | jq '.' echo_step "12. Bulk-Data Descriptors (Rosbag Files)" echo "Listing available rosbag recordings..." - curl -s "${API_BASE}/${ENTITY_PATH}/${FIRST_ENTITY}/bulk-data/rosbags" | jq '.items[] | { - id: .id, - name: .name, - size: .size, - mimetype: .mimetype, - "x-medkit": ."x-medkit" - }' + ROSBAGS_JSON=$(curl -s "${API_BASE}/apps/${FIRST_ENTITY}/bulk-data/rosbags") + if echo "$ROSBAGS_JSON" | jq -e '.items | length > 0' > /dev/null 2>&1; then + echo "$ROSBAGS_JSON" | jq '.items[] | { + id: .id, + name: .name, + size: .size, + mimetype: .mimetype, + "x-medkit": ."x-medkit" + }' + else + echo " No rosbag recordings listed for apps/${FIRST_ENTITY}." + fi fi else echo "" @@ -155,9 +286,9 @@ echo " ./inject-drift.sh # Inject sensor drift" echo " ./restore-normal.sh # Restore normal operation" echo "" echo "📸 After injecting a fault, check snapshots and rosbags:" -echo " curl ${API_BASE}/faults | jq # List faults" -echo " curl ${API_BASE}/components/lidar-unit/faults/ | jq # Fault detail + snapshots" -echo " curl ${API_BASE}/components/lidar-unit/bulk-data/rosbags | jq # List rosbag recordings" +echo " curl ${API_BASE}/faults | jq # List faults" +echo " curl ${API_BASE}/apps/diagnostic-bridge/faults/ | jq # Fault detail + snapshots" +echo " curl ${API_BASE}/apps/diagnostic-bridge/bulk-data/rosbags | jq # List rosbag recordings" echo "" echo "🌐 Web UI: http://localhost:3000" echo "🌐 REST API: http://localhost:8080/api/v1/" diff --git a/demos/turtlebot3_integration/Dockerfile b/demos/turtlebot3_integration/Dockerfile index f92fd61..4c99dd9 100644 --- a/demos/turtlebot3_integration/Dockerfile +++ b/demos/turtlebot3_integration/Dockerfile @@ -72,10 +72,6 @@ COPY config/ ${COLCON_WS}/src/turtlebot3_medkit_demo/config/ COPY launch/ ${COLCON_WS}/src/turtlebot3_medkit_demo/launch/ COPY scripts/ ${COLCON_WS}/src/turtlebot3_medkit_demo/scripts/ -# TODO(#49): Move to manifest-defined scripts once ros2_medkit#303 lands -COPY container_scripts/ /var/lib/ros2_medkit/scripts/ -RUN find /var/lib/ros2_medkit/scripts -name "*.bash" -exec chmod +x {} \; - # Build ros2_medkit and demo package WORKDIR ${COLCON_WS} RUN bash -c "source /opt/ros/jazzy/setup.bash && \ @@ -83,6 +79,11 @@ RUN bash -c "source /opt/ros/jazzy/setup.bash && \ rosdep install --from-paths src --ignore-src -r -y && \ MAKEFLAGS='-j 2' colcon build --executor sequential --symlink-install --cmake-args -DBUILD_TESTING=OFF" +# Scripts are not build inputs, so a script change does not rebuild the workspace. +# TODO(#49): Move to manifest-defined scripts once ros2_medkit#303 lands +COPY container_scripts/ /var/lib/ros2_medkit/scripts/ +RUN find /var/lib/ros2_medkit/scripts -name "*.bash" -exec chmod +x {} \; + # Setup environment RUN echo "source /opt/ros/jazzy/setup.bash" >> ~/.bashrc && \ echo "source ${COLCON_WS}/install/setup.bash" >> ~/.bashrc && \ diff --git a/demos/turtlebot3_integration/README.md b/demos/turtlebot3_integration/README.md index a2c0566..1d385dd 100644 --- a/demos/turtlebot3_integration/README.md +++ b/demos/turtlebot3_integration/README.md @@ -179,10 +179,10 @@ curl http://localhost:8080/api/v1/health curl http://localhost:8080/api/v1/areas | jq '.items[] | {id, name}' # List all components (hardware/logical units) -curl http://localhost:8080/api/v1/components | jq '.items[] | {id, name, area}' +curl http://localhost:8080/api/v1/components | jq '.items[] | {id, name, description}' # List all apps (ROS 2 nodes) -curl http://localhost:8080/api/v1/apps | jq '.items[] | {id, name, namespace}' +curl http://localhost:8080/api/v1/apps | jq '.items[] | {id, name, component: .["x-medkit"].component_id, online: .["x-medkit"].is_online}' # Get specific app details curl http://localhost:8080/api/v1/apps/amcl | jq @@ -193,19 +193,19 @@ curl http://localhost:8080/api/v1/apps/amcl | jq ```bash # Get LiDAR scan data curl http://localhost:8080/api/v1/apps/turtlebot3-node/data/scan | jq '{ - angle_min: .angle_min, - angle_max: .angle_max, - sample_ranges: .ranges[:5] + angle_min: .data.angle_min, + angle_max: .data.angle_max, + sample_ranges: .data.ranges[:5] }' # Get odometry data curl http://localhost:8080/api/v1/apps/turtlebot3-node/data/odom | jq '{ - position: .pose.pose.position, - orientation: .pose.pose.orientation + position: .data.pose.pose.position, + orientation: .data.pose.pose.orientation }' # List all data topics for an app -curl http://localhost:8080/api/v1/apps/turtlebot3-node/data | jq +curl http://localhost:8080/api/v1/apps/anomaly-detector/data | jq ``` ### Fault Management @@ -218,10 +218,10 @@ curl http://localhost:8080/api/v1/faults | jq curl http://localhost:8080/api/v1/areas/robot/faults | jq # Get fault details with environment data (includes snapshots) -curl http://localhost:8080/api/v1/faults/NAVIGATION_GOAL_ABORTED | jq +curl http://localhost:8080/api/v1/apps/anomaly-detector/faults/NAVIGATION_GOAL_ABORTED | jq # Clear a specific fault -curl -X DELETE http://localhost:8080/api/v1/apps/diagnostic-bridge/faults/TURTLEBOT3_NODE +curl -X DELETE http://localhost:8080/api/v1/apps/anomaly-detector/faults/NAVIGATION_GOAL_ABORTED ``` ### Rosbag Snapshots (Bulk Data) @@ -400,11 +400,13 @@ GATEWAY_URL=http://192.168.1.10:8080 ./inject-nav-failure.sh | `reset-navigation` | Cancel goals and reset AMCL | | `inject-localization-failure` | Inject AMCL localization failure | | `inject-nav-failure` | Inject navigation failure (unreachable goal) | -| `restore-normal` | Reset parameters and clear faults | +| `restore-normal` | Cancel goals, reset parameters, re-localize AMCL and clear faults | ## Triggers (Condition-Based Alerts) -The gateway supports condition-based triggers that fire when specific events occur, delivering notifications via Server-Sent Events (SSE). This demo creates a fault-monitoring trigger that alerts on any new or updated faults reported by the anomaly detector (including navigation failures). +The gateway supports condition-based triggers that fire when specific events occur, delivering notifications via Server-Sent Events (SSE). This demo creates a fault-monitoring trigger on the anomaly detector. It fires for the localization fault that `./inject-localization-failure.sh` causes (`LOCALIZATION_UNCERTAINTY`). + +Navigation goal faults (`NAVIGATION_GOAL_ABORTED`, `NAVIGATION_GOAL_CANCELED`), such as the one `./inject-nav-failure.sh` causes, do not fire this trigger. They still appear in `GET /api/v1/faults` and in `./check-faults.sh`. ### Setup @@ -419,40 +421,40 @@ The gateway supports condition-based triggers that fire when specific events occ ./watch-triggers.sh # Terminal 2: Inject a fault - the trigger fires in Terminal 3! -./inject-nav-failure.sh +./inject-localization-failure.sh ``` ### How It Works -1. `setup-triggers.sh` creates a trigger via `POST /api/v1/apps/anomaly_detector/triggers`: - - **Resource:** `/api/v1/apps/anomaly_detector/faults` (watches fault collection) - - **Condition:** `OnChange` (fires on any new or updated fault) +1. `setup-triggers.sh` creates a trigger via `POST /api/v1/apps/anomaly-detector/triggers`: + - **Resource:** `/api/v1/apps/anomaly-detector/faults` (watches fault collection) + - **Condition:** `OnChange` (fires when the localization fault is reported or updated; navigation goal faults do not fire it) - **Multishot:** `true` (fires repeatedly, not just once) - **Lifetime:** 3600 seconds (auto-expires after 1 hour) 2. `watch-triggers.sh` connects to the SSE event stream at the trigger's `event_source` URL -3. When a fault is injected and detected by the gateway, the trigger fires and an SSE event is delivered +3. When `./inject-localization-failure.sh` makes the anomaly detector report the localization fault, the trigger fires and an SSE event is delivered ### Manual API Usage ```bash # Create a trigger -curl -X POST http://localhost:8080/api/v1/apps/anomaly_detector/triggers \ +curl -X POST http://localhost:8080/api/v1/apps/anomaly-detector/triggers \ -H "Content-Type: application/json" \ -d '{ - "resource": "/api/v1/apps/anomaly_detector/faults", + "resource": "/api/v1/apps/anomaly-detector/faults", "trigger_condition": {"condition_type": "OnChange"}, "multishot": true, "lifetime": 3600 }' | jq # List triggers -curl http://localhost:8080/api/v1/apps/anomaly_detector/triggers | jq +curl http://localhost:8080/api/v1/apps/anomaly-detector/triggers | jq # Watch events (replace TRIGGER_ID) -curl -N http://localhost:8080/api/v1/apps/anomaly_detector/triggers/TRIGGER_ID/events +curl -N http://localhost:8080/api/v1/apps/anomaly-detector/triggers/TRIGGER_ID/events # Delete a trigger -curl -X DELETE http://localhost:8080/api/v1/apps/anomaly_detector/triggers/TRIGGER_ID +curl -X DELETE http://localhost:8080/api/v1/apps/anomaly-detector/triggers/TRIGGER_ID ``` ## Fault Injection Scenarios @@ -466,7 +468,7 @@ Faults are detected by `anomaly_detector` and reported directly to FaultManager. |--------|-----------|-------------|-----------------| | `inject-nav-failure.sh` | Navigation | Send goal to unreachable location | `NAVIGATION_GOAL_ABORTED` | | `inject-localization-failure.sh` | Localization | Reset AMCL with high uncertainty | `LOCALIZATION_UNCERTAINTY` | -| `restore-normal.sh` | Recovery | Restore defaults and clear faults | - | +| `restore-normal.sh` | Recovery | Restore defaults, re-localize AMCL and clear faults | - | ### Fault Injection Examples @@ -493,10 +495,17 @@ curl http://localhost:8080/api/v1/faults | jq #### Restore Normal Operation ```bash -# Clear all faults and restore default parameters +# Cancel goals, restore default parameters, re-localize AMCL and clear all faults ./restore-normal.sh ``` +`restore-normal.sh` also recovers from `inject-localization-failure.sh`: it reads the +robot's pose from the running Gazebo simulation and sets it on AMCL through +`POST /api/v1/apps/amcl/operations/set_initial_pose/executions`, so the robot need not be +at its spawn point. When the gateway refuses one of its writes, for example with 409 while +another client holds a lock on the app, the script still runs its other steps, then fails +and names the refused write. + ### Fault Monitoring via API ```bash @@ -591,13 +600,13 @@ demos/turtlebot3_integration/ | `run-demo.sh` | Start the full demo (Docker) | | `stop-demo.sh` | Stop containers and cleanup | | `send-nav-goal.sh [x] [y] [yaw]` | Send navigation goal via SOVD API | -| `check-entities.sh` | Explore SOVD entity hierarchy | -| `check-faults.sh` | View active faults from gateway | +| `check-entities.sh` | Explore SOVD entity hierarchy; prints a hint instead of scan values when the LiDAR has no data | +| `check-faults.sh` | View active faults from gateway; exits 1 when the fault list cannot be read | | `nav-health-check.sh` | Check Nav2 stack health | | `reset-navigation.sh` | Cancel goals and reset AMCL | | `inject-nav-failure.sh` | Inject navigation failure (unreachable goal) | | `inject-localization-failure.sh` | Inject localization failure (AMCL reset) | -| `restore-normal.sh` | Restore normal operation and clear faults | +| `restore-normal.sh` | Restore normal operation, re-localize AMCL and clear faults | | `setup-triggers.sh` | Create OnChange fault trigger | | `watch-triggers.sh` | Watch trigger events via SSE stream | diff --git a/demos/turtlebot3_integration/check-entities.sh b/demos/turtlebot3_integration/check-entities.sh index 5b04968..84ee44f 100755 --- a/demos/turtlebot3_integration/check-entities.sh +++ b/demos/turtlebot3_integration/check-entities.sh @@ -40,26 +40,32 @@ echo_step "1. Areas (Namespace Groupings)" curl -s "${API_BASE}/areas" | jq '.items[] | {id: .id, name: .name, description: .description}' echo_step "2. Components (Hardware/Logical Units)" -curl -s "${API_BASE}/components" | jq '.items[] | {id: .id, name: .name, type: .type, area: .area}' +curl -s "${API_BASE}/components" | jq '.items[] | {id: .id, name: .name, type: .type, description: .description}' echo_step "3. Apps (ROS 2 Nodes)" -curl -s "${API_BASE}/apps" | jq '.items[] | {id: .id, name: .name, category: .category, component: .is_located_on}' +curl -s "${API_BASE}/apps" | jq '.items[] | {id: .id, name: .name, component: .["x-medkit"].component_id}' echo_step "4. Functions (High-level Capabilities)" -curl -s "${API_BASE}/functions" | jq '.items[] | {id: .id, name: .name, category: .category, hosted_by: .hosted_by}' +curl -s "${API_BASE}/functions" | jq '.items[] | {id: .id, name: .name, description: .description}' echo_step "5. Sample Data (LiDAR Scan)" echo "Getting latest LiDAR scan from TurtleBot3..." -curl -s "${API_BASE}/apps/turtlebot3-node/data/scan" 2>/dev/null | jq '{ - angle_min: .angle_min, - angle_max: .angle_max, - range_min: .range_min, - range_max: .range_max, - sample_ranges: .ranges[:5] -}' || echo " (LiDAR data not available - Gazebo may still be starting)" +# A failed read or a scan topic without a message leaves .data empty. +SCAN_JSON=$(curl -s "${API_BASE}/apps/turtlebot3-node/data/scan" 2>/dev/null) +if echo "$SCAN_JSON" | jq -e '.data | type == "object" and length > 0' > /dev/null 2>&1; then + echo "$SCAN_JSON" | jq '{ + angle_min: .data.angle_min, + angle_max: .data.angle_max, + range_min: .data.range_min, + range_max: .data.range_max, + sample_ranges: .data.ranges[:5] + }' +else + echo " (LiDAR data not available - Gazebo may still be starting)" +fi echo_step "6. Faults" -curl -s "${API_BASE}/faults" | jq '.items[] | {code: .code, severity: .severity, reporter: .reporter_id}' +curl -s "${API_BASE}/faults" | jq '.items[] | {code: .fault_code, severity: .severity_label, sources: .reporting_sources}' echo "" echo -e "${GREEN}✓ Entity hierarchy exploration complete!${NC}" diff --git a/demos/turtlebot3_integration/check-faults.sh b/demos/turtlebot3_integration/check-faults.sh index 944ed6d..1990d36 100755 --- a/demos/turtlebot3_integration/check-faults.sh +++ b/demos/turtlebot3_integration/check-faults.sh @@ -1,6 +1,7 @@ #!/bin/bash # Check current faults from ros2_medkit gateway -# Faults are collected from Nav2/TurtleBot3 via diagnostic_bridge +# Faults are collected from Nav2/TurtleBot3 via anomaly_detector (direct) and +# diagnostic_bridge (legacy /diagnostics path) GATEWAY_URL="${GATEWAY_URL:-http://localhost:8080}" API_BASE="${GATEWAY_URL}/api/v1" @@ -25,9 +26,17 @@ fi echo "✓ Gateway is healthy" echo "" -# Get all faults +# Get all faults. A failed read leaves the fault list unknown, not empty. +FAULTS_RESPONSE=$(curl -s -w "\n%{http_code}" "${API_BASE}/faults") +FAULTS_CODE=$(tail -n 1 <<< "$FAULTS_RESPONSE") +FAULTS=$(sed '$d' <<< "$FAULTS_RESPONSE") +if [ "$FAULTS_CODE" != "200" ] || ! echo "$FAULTS" | jq -e '.items | type == "array"' > /dev/null 2>&1; then + echo "❌ Could not read faults from ${GATEWAY_URL} (HTTP ${FAULTS_CODE}): $(echo "$FAULTS" | jq -r '.message // empty' 2>/dev/null)" + echo " The fault list is unknown. Check that the fault manager is running, then retry." + exit 1 +fi + echo "📋 Active Faults:" -FAULTS=$(curl -s "${API_BASE}/faults") # Check if there are any faults FAULT_COUNT=$(echo "$FAULTS" | jq '.items | length') @@ -36,18 +45,30 @@ if [ "$FAULT_COUNT" = "0" ]; then echo " No active faults - system is healthy!" else echo "$FAULTS" | jq '.items[] | { - code: .code, - severity: .severity, - reporter: .reporter_id, - message: .message, - timestamp: .timestamp + code: .fault_code, + severity: .severity_label, + status: .status, + description: .description, + sources: .reporting_sources, + occurrences: .occurrence_count, + first_occurred: .first_occurred, + last_occurred: .last_occurred }' fi echo "" echo "📊 Fault Summary:" echo " Total active faults: $FAULT_COUNT" + +# Show fault counts by severity if any exist +if [ "$FAULT_COUNT" != "0" ]; then + echo "" + echo " By severity:" + echo "$FAULTS" | jq -r '.items | group_by(.severity_label) | .[] | " \(.[0].severity_label): \(length)"' +fi + echo "" echo "Commands:" echo " Clear all faults: curl -X DELETE ${API_BASE}/faults" -echo " Check specific area: curl ${API_BASE}/areas/robot/faults | jq" +echo " Check area faults: curl ${API_BASE}/areas/navigation/faults | jq" +echo " Check component faults: curl ${API_BASE}/components/nav2-stack/faults | jq" diff --git a/demos/turtlebot3_integration/container_scripts/nav2-stack/inject-localization-failure/script.bash b/demos/turtlebot3_integration/container_scripts/nav2-stack/inject-localization-failure/script.bash index 64a3f57..27b5041 100755 --- a/demos/turtlebot3_integration/container_scripts/nav2-stack/inject-localization-failure/script.bash +++ b/demos/turtlebot3_integration/container_scripts/nav2-stack/inject-localization-failure/script.bash @@ -15,8 +15,8 @@ echo "Waiting for particles to scatter..." sleep 2 echo "Sending navigation goal with high localization uncertainty..." -# Drop -f: action server typically rejects the goal under scattered particles, -# which is the demo's intended failure mode - treat HTTP 400 as expected. +# Drop -f so an error body is still printed. Nav2 accepts the goal and drives from +# the scattered pose estimate. RESPONSE=$(curl -s -X POST "${API_BASE}/apps/bt-navigator/operations/navigate_to_pose/executions" \ -H "Content-Type: application/json" \ -d '{ diff --git a/demos/turtlebot3_integration/container_scripts/nav2-stack/restore-normal/metadata.json b/demos/turtlebot3_integration/container_scripts/nav2-stack/restore-normal/metadata.json index c7bcd64..099f93d 100644 --- a/demos/turtlebot3_integration/container_scripts/nav2-stack/restore-normal/metadata.json +++ b/demos/turtlebot3_integration/container_scripts/nav2-stack/restore-normal/metadata.json @@ -1,5 +1,5 @@ { "name": "Restore Normal Operation", - "description": "Cancel active navigation goals, restore velocity parameters to defaults, and clear all faults", + "description": "Cancel active navigation goals, restore velocity parameters to defaults, re-localize AMCL at the robot's pose in the simulation, and clear all faults", "format": "bash" } diff --git a/demos/turtlebot3_integration/container_scripts/nav2-stack/restore-normal/script.bash b/demos/turtlebot3_integration/container_scripts/nav2-stack/restore-normal/script.bash index d55c7af..46e8aa4 100755 --- a/demos/turtlebot3_integration/container_scripts/nav2-stack/restore-normal/script.bash +++ b/demos/turtlebot3_integration/container_scripts/nav2-stack/restore-normal/script.bash @@ -1,10 +1,33 @@ #!/bin/bash -# Restore normal operation: cancel goals, restore velocity params, clear all faults +# Restore normal operation: cancel goals, restore velocity params, put AMCL back on +# the robot's pose in the simulation, clear all faults set -eu GATEWAY_URL="${GATEWAY_URL:-http://localhost:8080}" API_BASE="${GATEWAY_URL}/api/v1" +# A failed step is named on stderr, which the Scripts API returns as the error +# message. The remaining steps still run; the script fails at the end. +FAILURES=0 +report_failure() { + echo "FAIL: $1" >&2 + FAILURES=$((FAILURES + 1)) +} + +# Sets one parameter and reports a refused or failed write. +# Usage: put_config APP PARAM JSON_VALUE +put_config() { + local app="$1" param="$2" value="$3" response code + response=$(curl -s -m 30 -w "\n%{http_code}" -X PUT \ + "${API_BASE}/apps/${app}/configurations/${param}" \ + -H "Content-Type: application/json" -d "{\"value\": ${value}}" 2>/dev/null) || response=$'\nnone' + code=$(tail -n 1 <<< "${response}") + case "${code}" in + 2??) echo " ${app}/${param} = ${value}" ;; + *) report_failure "${app}/${param} (HTTP ${code}: $(sed '$d' <<< "${response}" | jq -r '.message // empty' 2>/dev/null))" ;; + esac +} + echo "Canceling active navigation goals..." EXECUTIONS=$(curl -s "${API_BASE}/apps/bt-navigator/operations/navigate_to_pose/executions" 2>/dev/null || echo '{"items":[]}') if echo "${EXECUTIONS}" | jq -e '.items[]' > /dev/null 2>&1; then @@ -20,19 +43,63 @@ fi echo "" echo "Restoring velocity parameters to defaults..." -curl -s -X PUT "${API_BASE}/apps/velocity-smoother/configurations/max_velocity" \ - -H "Content-Type: application/json" \ - -d '{"value": [0.26, 0.0, 1.0]}' > /dev/null 2>&1 || true +put_config velocity-smoother max_velocity "[0.26, 0.0, 1.0]" +put_config controller-server FollowPath.max_vel_x 0.26 + +# The map frame of this demo is the Gazebo world frame, so the robot's world pose +# is the pose AMCL must hold. It is read from the running simulation after the +# robot has stopped, never assumed. +echo "" +echo "Re-localizing AMCL at the robot's pose in the simulation..." +sleep 1 +MODEL="${TURTLEBOT3_MODEL:-burger}" +WORLD=$(timeout 10 gz topic -l 2>/dev/null | sed -n 's|^/world/\([^/]*\)/dynamic_pose/info$|\1|p' | head -n 1) || WORLD="" +POSE_REQUEST="" +if [ -n "${WORLD}" ]; then + # A single read has returned the model twice, so only the first match is used. + POSE_REQUEST=$(timeout 10 gz topic -e -n 1 -t "/world/${WORLD}/dynamic_pose/info" --json-output 2>/dev/null \ + | jq -c -n --arg m "${MODEL}" ' + first(inputs | .pose[] | select(.name == $m)) + | (.orientation | [(.w // 0), (.x // 0), (.y // 0), (.z // 0)] as [$w, $x, $y, $z] + | atan2(2 * ($w * $z + $x * $y); 1 - 2 * ($y * $y + $z * $z))) as $yaw + | {parameters: {pose: {header: {frame_id: "map"}, + pose: {pose: {position: {x: (.position.x // 0), y: (.position.y // 0), z: 0.0}, + orientation: {x: 0.0, y: 0.0, z: ($yaw / 2 | sin), w: ($yaw / 2 | cos)}}, + covariance: [range(36) | if . == 0 or . == 7 or . == 35 then 0.01 else 0.0 end]}}}}' \ + 2>/dev/null) || POSE_REQUEST="" +fi -curl -s -X PUT "${API_BASE}/apps/controller-server/configurations/FollowPath.max_vel_x" \ - -H "Content-Type: application/json" \ - -d '{"value": 0.26}' > /dev/null 2>&1 || true +if [ -z "${POSE_REQUEST}" ]; then + report_failure "amcl/set_initial_pose (no pose for '${MODEL}' in the simulation)" +else + echo " Robot pose: $(echo "${POSE_REQUEST}" | jq -c '.parameters.pose.pose.pose.position | {x, y}')" + HTTP_CODE=$(curl -s -o /dev/null -w "%{http_code}" -X POST \ + "${API_BASE}/apps/amcl/operations/set_initial_pose/executions" \ + -H "Content-Type: application/json" -d "${POSE_REQUEST}" 2>/dev/null) || HTTP_CODE="none" + if [ "${HTTP_CODE}" != "200" ]; then + report_failure "amcl/set_initial_pose (HTTP ${HTTP_CODE})" + fi +fi +# Faults are cleared after the re-localization, so no pose from the scattered +# particle cloud is reported after the clear. +echo "" echo "Clearing all faults..." -curl -s -X DELETE "${API_BASE}/faults" > /dev/null || true +CLEAR_CODE=$(curl -s -o /dev/null -w "%{http_code}" -X DELETE "${API_BASE}/faults" 2>/dev/null) || CLEAR_CODE="none" +case "${CLEAR_CODE}" in + 2??) ;; + *) report_failure "faults clear (HTTP ${CLEAR_CODE})" ;; +esac + +FAULT_COUNT=$(curl -sf "${API_BASE}/faults" | jq '.items | length' 2>/dev/null || echo "?") +if [ "${FAILURES}" -gt 0 ]; then + echo "" + echo "Normal operation is not restored: ${FAILURES} step(s) failed. Active faults: ${FAULT_COUNT}" + echo "${FAILURES} restore step(s) failed" >&2 + exit 1 +fi echo "" echo "Normal operation restored." -FAULT_COUNT=$(curl -sf "${API_BASE}/faults" | jq '.items | length' 2>/dev/null || echo "?") echo "Active faults: ${FAULT_COUNT}" exit 0 diff --git a/demos/turtlebot3_integration/setup-triggers.sh b/demos/turtlebot3_integration/setup-triggers.sh index b9390f6..f10974a 100755 --- a/demos/turtlebot3_integration/setup-triggers.sh +++ b/demos/turtlebot3_integration/setup-triggers.sh @@ -1,10 +1,10 @@ #!/bin/bash # Create fault-monitoring trigger for turtlebot3 integration demo -# Alerts on any fault change reported via the diagnostic bridge - the -# anomaly-detector app has no faults of its own, faults arrive from -# /diagnostics through the bridge. +# Alerts on any fault change reported by the anomaly detector - navigation +# and localization faults are reported directly, not via the diagnostic +# bridge. export ENTITY_TYPE="apps" -export ENTITY_ID="diagnostic-bridge" -export INJECT_HINT="./inject-nav-failure.sh" +export ENTITY_ID="anomaly-detector" +export INJECT_HINT="./inject-localization-failure.sh" # shellcheck disable=SC1091 source "$(cd "$(dirname "$0")" && pwd)/../../lib/setup-trigger.sh" diff --git a/demos/turtlebot3_integration/watch-triggers.sh b/demos/turtlebot3_integration/watch-triggers.sh index d6d9b6c..aa0746f 100755 --- a/demos/turtlebot3_integration/watch-triggers.sh +++ b/demos/turtlebot3_integration/watch-triggers.sh @@ -2,6 +2,6 @@ # Watch trigger events for turtlebot3 integration demo # Connects to SSE stream and prints fault events in real time export ENTITY_TYPE="apps" -export ENTITY_ID="diagnostic-bridge" +export ENTITY_ID="anomaly-detector" # shellcheck disable=SC1091 source "$(cd "$(dirname "$0")" && pwd)/../../lib/watch-trigger.sh" "$@" diff --git a/tests/smoke_test.sh b/tests/smoke_test.sh index cb03779..bb55b20 100755 --- a/tests/smoke_test.sh +++ b/tests/smoke_test.sh @@ -1,6 +1,8 @@ #!/bin/bash # Smoke tests for sensor_diagnostics demo -# Runs from the host against the containerized gateway on localhost:8080 +# Runs from the host against the containerized gateway on localhost:8080. Needs +# docker access to the demo container (DEMO_CONTAINER) to pause its fault +# manager and to restart it. # # Usage: ./tests/smoke_test.sh [GATEWAY_URL] # Default GATEWAY_URL: http://localhost:8080 @@ -12,12 +14,204 @@ SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" # shellcheck source=tests/smoke_lib.sh source "${SCRIPT_DIR}/smoke_lib.sh" -trap print_summary EXIT +DEMO_CONTAINER="${DEMO_CONTAINER:-sensor_diagnostics_demo_ci}" +# Anchored so it does not match the bash -c wrapper that runs pgrep. +FAULT_MANAGER_PATTERN='^/root/demo_ws/install/ros2_medkit_fault_manager/lib/ros2_medkit_fault_manager/fault_manager_node ' + +# SIGSTOP on the fault manager makes the gateway's ListFaults call time out, +# so GET /faults answers 503 as it does before the fault manager is up. +stop_fault_manager() { + docker exec "${DEMO_CONTAINER}" bash -c " + pid=\$(pgrep -f '${FAULT_MANAGER_PATTERN}') || exit 1 + kill -STOP \"\${pid}\" + " > /dev/null 2>&1 +} + +resume_fault_manager() { + docker exec "${DEMO_CONTAINER}" bash -c " + pid=\$(pgrep -f '${FAULT_MANAGER_PATTERN}') || exit 0 + kill -CONT \"\${pid}\" + " > /dev/null 2>&1 || true +} + +SENSOR_BIN_DIR=/root/demo_ws/install/sensor_diagnostics_demo/lib/sensor_diagnostics_demo + +# Sends SIGNAL to the demo's NODE executable; fails when it is not running. +# SIGSTOP keeps a sensor node in the ROS graph but stops its messages. +# Usage: signal_sensor NODE SIGNAL +signal_sensor() { + docker exec "${DEMO_CONTAINER}" bash -c " + pid=\$(pgrep -f '^${SENSOR_BIN_DIR}/$1 ') || exit 1 + kill -$2 \"\${pid}\" + " > /dev/null 2>&1 +} + +# Takes lidar_sim off the ROS graph. Its command line, which carries the +# launch's parameter files, is kept for restore_lidar. +remove_lidar() { + docker exec "${DEMO_CONTAINER}" bash -c " + pid=\$(pgrep -f '^${SENSOR_BIN_DIR}/lidar_sim_node ') || exit 1 + tr '\\0' '\\n' < /proc/\${pid}/cmdline > /tmp/lidar_sim.cmdline + kill -TERM \"\${pid}\" + for _ in \$(seq 1 50); do kill -0 \"\${pid}\" 2> /dev/null || exit 0; sleep 0.1; done + exit 1 + " > /dev/null 2>&1 +} + +# Starts lidar_sim from the kept command line unless it is running. +restore_lidar() { + docker exec "${DEMO_CONTAINER}" bash -c " + ! pgrep -f '^${SENSOR_BIN_DIR}/lidar_sim_node ' && test -s /tmp/lidar_sim.cmdline + " > /dev/null 2>&1 || return 0 + # shellcheck disable=SC2016 # expanded by the container's bash + docker exec -d "${DEMO_CONTAINER}" bash -c 'source /opt/ros/jazzy/setup.bash \ + && source /root/demo_ws/install/setup.bash \ + && mapfile -t args < /tmp/lidar_sim.cmdline && exec "${args[@]}"' > /dev/null 2>&1 || true +} + +# print_summary reads the script's exit status from $?, so hand it the status +# saved on entry. set +e: under errexit `(exit rc)` would end the trap before +# print_summary runs. +cleanup_on_exit() { + local rc=$? + set +e + resume_fault_manager + signal_sensor lidar_sim_node CONT + signal_sensor imu_sim_node CONT + restore_lidar + (exit "${rc}") + print_summary +} +trap cleanup_on_exit EXIT + +CHECK_DEMO_SCRIPT="${SCRIPT_DIR}/../demos/sensor_diagnostics/check-demo.sh" + +# Runs check-demo.sh and prints its output without ANSI colours. +run_check_demo() { + GATEWAY_URL="$GATEWAY_URL" bash "$CHECK_DEMO_SCRIPT" 2>&1 | sed 's/\x1b\[[0-9;]*m//g' +} + +# Runs check-demo.sh with DATA_WAIT_SEC=$1 (empty: the script's default) and +# prints each output line without ANSI colours, prefixed with the +# milliseconds since the run started. +run_check_demo_timed() { + local start=${EPOCHREALTIME//[!0-9]/} line + env ${1:+"DATA_WAIT_SEC=$1"} GATEWAY_URL="$GATEWAY_URL" timeout 180 bash "$CHECK_DEMO_SCRIPT" < /dev/null 2>&1 \ + | while IFS= read -r line; do + printf '%d %s\n' $(( (${EPOCHREALTIME//[!0-9]/} - start) / 1000 )) "$line" + done | sed 's/\x1b\[[0-9;]*m//g' +} + +# Prints the milliseconds check-demo.sh spent between the health check and +# section 1, from a run_check_demo_timed output. Without section 1, the time +# of the last line. +readiness_wait_ms() { + awk '/^[0-9]+ .*Gateway is healthy/ { s = $1 } + /^[0-9]+ === 1\. / { e = $1; exit } + { last = $1 } + END { if (e == "") e = last; print e - s }' <<< "$1" +} + +# Drops the time prefix of a run_check_demo_timed output. +untimed() { + awk '{ sub(/^[0-9]+ /, ""); print }' <<< "$1" +} + +# Prints what check-demo.sh printed between the health check and section 1. +# One awk reads the output: an early exit behind a pipe would SIGPIPE the +# writer, and pipefail would fail the caller. +readiness_wait_output() { + awk '{ sub(/^[0-9]+ /, "") } /Gateway is healthy/ { on = 1; next } /^=== 1\. / { exit } on' <<< "$1" +} + +# Prints the text lines of section N of a check-demo.sh output. +check_demo_section_text() { + awk -v hdr="^=== $2\\\\. " ' + $0 ~ hdr { on = 1; next } + on && /^=== [0-9]+\. / { exit } + on' <<< "$1" +} + +# Prints the JSON values printed in section N of a check-demo.sh output as +# one array: [] when the section is missing, nothing when it is not JSON. +# Usage: check_demo_section OUTPUT N +check_demo_section() { + check_demo_section_text "$1" "$2" | sed -n '/^[[{]/,$p' | jq -s '.' 2>/dev/null || true +} + +# Prints the value of a configuration, or nothing when it cannot be read. +# Usage: config_value APP PARAM +config_value() { + curl -s -m 10 "${API_BASE}/apps/$1/configurations/$2" | jq -c '.data' 2>/dev/null || true +} + +# True when section N (5 LiDAR, 6 IMU, 7 GPS, 8 configurations) of a +# check-demo.sh output carries values of the right type. +# Usage: section_values_printed OUTPUT N +section_values_printed() { + local filter + case "$2" in + 5) filter='length == 1 and ([.[0] | .angle_min, .angle_max, .range_min, .range_max, .sample_ranges[]] + | length == 9 and all(type == "number"))' ;; + 6) filter='length == 1 and ([.[0] | .linear_acceleration[], .angular_velocity[]] + | length == 6 and all(type == "number"))' ;; + 7) filter='length == 1 and ([.[0] | .latitude, .longitude, .altitude] | all(type == "number"))' ;; + 8) filter='length > 0' ;; + esac + check_demo_section "$1" "$2" | jq -e "$filter" > /dev/null 2>&1 +} + +# Reports a FAILED fault through the fault manager's report_fault service. +# Usage: report_fault CODE SOURCE_ID +report_fault() { + curl -s -m 20 -X POST "${API_BASE}/apps/medkit-fault-manager/operations/report_fault/executions" \ + -H "Content-Type: application/json" \ + -d "$(jq -nc --arg code "$1" --arg src "$2" '{parameters: {fault_code: $code, event_type: 0, + severity: 2, description: "smoke test", source_id: $src}}')" \ + | jq -e '.parameters.accepted == true' > /dev/null 2>&1 +} # --- Wait for gateway startup --- +# /health answers about a second before the gateway links the sensor nodes, and +# wait_for_gateway polls only every 2 s. The tighter poll lets the first +# check-demo.sh run below start inside that window on a demo that has just started. +for _ in $(seq 1 450); do + curl -sf -m 2 "${API_BASE}/health" > /dev/null 2>&1 && break + sleep 0.2 +done wait_for_gateway 90 +section "Check-Demo Before Linking" + +# Until the sensor nodes are linked, their data reads come back empty. +# check-demo.sh must wait for the links and then print real values. +EARLY_LIDAR_ITEMS=$(curl -s -m 5 "${API_BASE}/apps/lidar-sim/data" | jq '.items | length' 2>/dev/null) || true +EARLY_RC=0 +EARLY_TIMED=$(run_check_demo_timed "") || EARLY_RC=$? +EARLY_PLAIN=$(untimed "$EARLY_TIMED") +echo " LiDAR data items when check-demo.sh started: ${EARLY_LIDAR_ITEMS:-unreadable}; exit code ${EARLY_RC}" +echo " readiness wait $(readiness_wait_ms "$EARLY_TIMED") ms:" \ + "$(readiness_wait_output "$EARLY_TIMED" | grep -vE '^ *$' | tr '\n' ';' | head -c 300)" + +if grep -q ': null' <<< "$EARLY_PLAIN"; then + fail "check-demo.sh started right after /health prints no null fields" \ + "$(grep -B1 ': null' <<< "$EARLY_PLAIN" | head -10)" +else + pass "check-demo.sh started right after /health prints no null fields" +fi + +EARLY_MISSING="" +for n in 5 6 7 8; do + section_values_printed "$EARLY_PLAIN" "$n" || EARLY_MISSING="${EARLY_MISSING} ${n}" +done +if [ "$EARLY_RC" -eq 0 ] && [ -z "$EARLY_MISSING" ]; then + pass "check-demo.sh started right after /health exits 0 and prints values in sections 5-8" +else + fail "check-demo.sh started right after /health exits 0 and prints values in sections 5-8" \ + "exit code ${EARLY_RC}; sections without values:${EARLY_MISSING:- none}; last lines: $(grep -vE '^ *$' <<< "$EARLY_PLAIN" | tail -n 3 | tr '\n' ';')" +fi + # Wait for entity discovery + runtime linking (nodes need to be linked to manifest apps) # In hybrid mode, manifest entities appear instantly but data/configurations require # the runtime refresh cycle to link ROS 2 nodes to manifest apps. @@ -96,6 +290,93 @@ section "Logs" assert_non_empty_items "/apps/medkit-gateway/logs" +section "Check-Demo Sensor Values" + +# Runs before the fault injection below, while the noise is at its default: at +# the injected 0.5 m, 8 sigma spans the whole LiDAR range. + +# Checks section N of the first run against a direct read of ENDPOINT, and +# that the second run printed a different sample. FILTER sees the printed +# values as input, the direct read's .data as $g and SIGMA as $sigma. +# Usage: assert_section_live N DESCRIPTION ENDPOINT SIGMA FILTER +assert_section_live() { + local n="$1" description="$2" endpoint="$3" sigma="$4" filter="$5" + local printed printed_again direct + printed=$(check_demo_section "$SENSOR_RUN_1" "$n") + printed_again=$(check_demo_section "$SENSOR_RUN_2" "$n") + if ! api_get "$endpoint"; then + fail "$description" "direct read of ${endpoint} failed" + return + fi + direct=$(jq -c '.data' <<< "$RESPONSE" 2>/dev/null) || true + if ! jq -e --argjson g "${direct:-null}" --argjson sigma "${sigma:-null}" "$filter" \ + <<< "$printed" > /dev/null 2>&1; then + fail "$description" "printed $(jq -c '.' <<< "$printed" 2>/dev/null || echo "no JSON"); sigma ${sigma}; direct read ${direct:0:300}" + elif [ "$printed" = "$printed_again" ]; then + fail "$description" "two runs printed the same values: $(jq -c '.' <<< "$printed")" + else + pass "$description" + fi +} + +SENSOR_RUN_1=$(run_check_demo) || true +# Every sensor publishes at 1 Hz or faster, so a run a second later reads new samples. +sleep 1 +SENSOR_RUN_2=$(run_check_demo) || true + +if grep -q ': null' <<< "$SENSOR_RUN_1"; then + fail "check-demo.sh prints no null fields" "$(grep -B1 ': null' <<< "$SENSOR_RUN_1" | head -10)" +else + pass "check-demo.sh prints no null fields" +fi + +# Tolerances are 8 sigma of the live noise configuration: two independent +# samples differ by more than that with a probability below 1e-7. The LiDAR +# check also requires 8 sigma below a tenth of the scan's range span, or it +# could not tell a real range from an invented one. +LIDAR_FILTER=$(cat <<'JQ' +8 * $sigma < ($g.range_max - $g.range_min) / 10 +and length == 1 and (.[0] as $p + | $p.angle_min == $g.angle_min and $p.angle_max == $g.angle_max + and $p.range_min == $g.range_min and $p.range_max == $g.range_max + and ($p.sample_ranges | type == "array" and length == 5) + and ([range(5)] | all(. as $i | $p.sample_ranges[$i] as $r + | ($r | type) == "number" + and $r >= $g.range_min and $r <= $g.range_max + and (($r - $g.ranges[$i]) | fabs) <= 8 * $sigma))) +JQ +) +IMU_FILTER=$(cat <<'JQ' +length == 1 and (.[0] as $p + | [["linear_acceleration", $sigma.accel], ["angular_velocity", $sigma.gyro]] + | all(.[0] as $k | .[1] as $s | ["x", "y", "z"] + | all(. as $a | ($p[$k][$a] | type) == "number" + and (($p[$k][$a] - $g[$k][$a]) | fabs) <= 8 * $s))) +JQ +) +# 111 km per degree of latitude; a degree of longitude shrinks with cos(latitude). +GPS_FILTER=$(cat <<'JQ' +length == 1 and (.[0] as $p + | ([$p.latitude, $p.longitude, $p.altitude] | all(type == "number")) + and (($p.latitude - $g.latitude) | fabs) <= 8 * $sigma.pos / 111000 + and (($p.longitude - $g.longitude) | fabs) + <= 8 * $sigma.pos / (111000 * (($g.latitude * 3.141592653589793 / 180) | cos)) + and (($p.altitude - $g.altitude) | fabs) <= 8 * $sigma.alt + and $p.status == $g.status) +JQ +) + +assert_section_live 5 "check-demo.sh section 5 shows live LiDAR values matching a direct read" \ + "/apps/lidar-sim/data/sensors%2Fscan" "$(config_value lidar-sim noise_stddev)" "$LIDAR_FILTER" +assert_section_live 6 "check-demo.sh section 6 shows live IMU values matching a direct read" \ + "/apps/imu-sim/data/sensors%2Fimu" \ + "{\"accel\": $(config_value imu-sim accel_noise_stddev), \"gyro\": $(config_value imu-sim gyro_noise_stddev)}" \ + "$IMU_FILTER" +assert_section_live 7 "check-demo.sh section 7 shows live GPS values matching a direct read" \ + "/apps/gps-sim/data/sensors%2Ffix" \ + "{\"pos\": $(config_value gps-sim position_noise_stddev), \"alt\": $(config_value gps-sim altitude_noise_stddev)}" \ + "$GPS_FILTER" + section "Fault Injection" # Inject noise fault via configuration API @@ -133,6 +414,75 @@ else fail "GET fault detail returns 200" "unexpected status code" fi +section "Check-Demo Script" + +# The rosbag recording finalizes duration_after_sec after confirmation, so +# poll for it rather than racing check-demo.sh against the write. +echo " Waiting for rosbag recording to finish (max 15s)..." +if poll_until "/apps/diagnostic-bridge/bulk-data/rosbags" '.items | length > 0' 15; then + pass "rosbag recording available before running check-demo.sh" +else + fail "rosbag recording available before running check-demo.sh" "no rosbag after 15s" +fi + +# Run check-demo.sh again while the LIDAR_SIM fault above is active: section 8 +# must show the injected noise, and sections 10-12 need an active fault. +CHECK_DEMO_PLAIN=$(run_check_demo) || true + +if grep -q ': null' <<< "$CHECK_DEMO_PLAIN"; then + fail "check-demo.sh prints no null fields with a fault active" \ + "$(grep -B1 ': null' <<< "$CHECK_DEMO_PLAIN" | head -10)" +else + pass "check-demo.sh prints no null fields with a fault active" +fi + +# Section 8: exactly the parameters the list endpoint lists, each with the +# value and ROS type its own detail endpoint returns. +CONFIG_EXPECTED="" +if api_get "/apps/lidar-sim/configurations"; then + CONFIG_EXPECTED="[]" + while IFS= read -r cfg_id; do + if ! api_get "/apps/lidar-sim/configurations/${cfg_id//\//%2F}"; then + CONFIG_EXPECTED="" + break + fi + CONFIG_EXPECTED=$(jq -c --argjson d "$RESPONSE" \ + '. + [{name: $d.id, value: $d.data, type: $d["x-medkit"].parameter.type}]' <<< "$CONFIG_EXPECTED") + done < <(jq -r '.items[].id' <<< "$RESPONSE") +fi +CONFIG_PRINTED=$(check_demo_section "$CHECK_DEMO_PLAIN" 8) +if [ -z "$CONFIG_EXPECTED" ]; then + fail "check-demo.sh section 8 lists every LiDAR configuration with its value and type" \ + "direct read of /apps/lidar-sim/configurations failed" +elif [ "$(jq -cS 'sort_by(.name)' <<< "$CONFIG_PRINTED" 2>/dev/null)" = "$(jq -cS 'sort_by(.name)' <<< "$CONFIG_EXPECTED")" ]; then + pass "check-demo.sh section 8 lists every LiDAR configuration with its value and type" +else + fail "check-demo.sh section 8 lists every LiDAR configuration with its value and type" \ + "$(jq -nc --argjson p "${CONFIG_PRINTED:-[]}" --argjson e "$CONFIG_EXPECTED" \ + '{missing: ($e - $p), unexpected: ($p - $e)}' 2>/dev/null || echo "section 8 is not JSON")" +fi + +if grep -q "10\. Fault Detail with Environment Data" <<< "$CHECK_DEMO_PLAIN"; then + pass "check-demo.sh runs the fault detail section for the active fault" +else + fail "check-demo.sh runs the fault detail section for the active fault" \ + "section 10 did not run: the owning App was not resolved" +fi + +if grep -q '"snapshot_count": 0' <<< "$CHECK_DEMO_PLAIN"; then + fail "check-demo.sh fault detail shows real snapshot data" "snapshot_count is 0" +elif grep -q '"snapshot_count":' <<< "$CHECK_DEMO_PLAIN"; then + pass "check-demo.sh fault detail shows real snapshot data" +else + fail "check-demo.sh fault detail shows real snapshot data" "snapshot_count field missing" +fi + +if grep -q '"id": "fault_LIDAR_SIM' <<< "$CHECK_DEMO_PLAIN"; then + pass "check-demo.sh bulk-data section lists a real rosbag recording" +else + fail "check-demo.sh bulk-data section lists a real rosbag recording" "no fault_LIDAR_SIM rosbag id found" +fi + # Cleanup: restore config + delete fault echo " Cleaning up: restoring config and clearing fault..." curl -s -X PUT "${API_BASE}/apps/lidar-sim/configurations/noise_stddev" \ @@ -190,6 +540,338 @@ if [ "$beacon_found" = false ]; then echo -e " ${BLUE}SKIP${NC} beacon not active (BEACON_MODE=none or plugin not loaded)" fi +section "check-demo.sh: a failed fault read is not reported as no faults" + +FAILED_READ_LOG=$(mktemp) +if stop_fault_manager; then + if api_get "/faults" 503; then + pass "setup: GET /faults answers 503 while the fault manager is stopped" + else + fail "setup: GET /faults answers 503 while the fault manager is stopped" "got: $(head -c 200 <<< "${RESPONSE}")" + fi + if GATEWAY_URL="$GATEWAY_URL" timeout 120 bash "$CHECK_DEMO_SCRIPT" < /dev/null > "$FAILED_READ_LOG" 2>&1; then + FAILED_READ_RC=0 + else + FAILED_READ_RC=$? + fi + resume_fault_manager + if [ "$FAILED_READ_RC" -ne 0 ] && grep -q 'Could not read faults .*(HTTP 503)' "$FAILED_READ_LOG" \ + && ! grep -q 'No active faults' "$FAILED_READ_LOG"; then + pass "check-demo.sh exits non-zero on HTTP 503 and does not claim there are no faults" + else + fail "check-demo.sh exits non-zero on HTTP 503 and does not claim there are no faults" \ + "rc=${FAILED_READ_RC}; output: $(sed 's/\x1b\[[0-9;]*m//g' "$FAILED_READ_LOG" | grep -vE '^ *$' | tail -n 6 | tr '\n' ';')" + fi + if poll_until "/faults" '.items | arrays' 30; then + pass "setup: GET /faults answers again after the fault manager resumes" + else + fail "setup: GET /faults answers again after the fault manager resumes" "still failing after 30 s" + fi +else + fail "setup: fault manager stopped" "no fault_manager_node process in ${DEMO_CONTAINER}" +fi +rm -f "$FAILED_READ_LOG" + +section "check-demo.sh: a fault reported from a sub-path of an App's node" + +# The anomaly detector reports as /processing/anomaly_detector/, a +# sub-path of its node. Sections 10-12 use the first listed fault, so the +# other faults are cleared before each report. +SUBPATH_CODE="SMOKE_SUBPATH_SOURCE" +SIBLING_CODE="SMOKE_SIBLING_SOURCE" +curl -s -m 20 -X DELETE "${API_BASE}/faults" > /dev/null || true +if report_fault "$SUBPATH_CODE" "/processing/anomaly_detector/imu_sim" \ + && poll_until "/faults" ".items[0].fault_code == \"${SUBPATH_CODE}\"" 15; then + pass "setup: ${SUBPATH_CODE} from /processing/anomaly_detector/imu_sim is the first listed fault" + SUBPATH_PLAIN=$(run_check_demo) || true + SUBPATH_SHOWN=$(check_demo_section "$SUBPATH_PLAIN" 10 | jq -r '.[0].code // empty' 2>/dev/null) || true + if [ "$SUBPATH_SHOWN" = "$SUBPATH_CODE" ] \ + && grep -q "Fetching fault ${SUBPATH_CODE} on apps/anomaly-detector\.\.\." <<< "$SUBPATH_PLAIN"; then + pass "check-demo.sh section 10 shows a sub-path fault on the App that owns the node" + else + fail "check-demo.sh section 10 shows a sub-path fault on the App that owns the node" \ + "section 10 code: ${SUBPATH_SHOWN:-none}; $(grep -m1 -E 'Could not map|Fetching fault' <<< "$SUBPATH_PLAIN")" + fi + if grep -q "=== 11\. " <<< "$SUBPATH_PLAIN" && grep -q "=== 12\. " <<< "$SUBPATH_PLAIN"; then + pass "check-demo.sh runs sections 11 and 12 for a sub-path fault" + else + fail "check-demo.sh runs sections 11 and 12 for a sub-path fault" "section 11 or 12 missing" + fi + if grep -q ': null' <<< "$SUBPATH_PLAIN"; then + fail "check-demo.sh prints no null fields for a sub-path fault" \ + "$(grep -B1 ': null' <<< "$SUBPATH_PLAIN" | head -10)" + else + pass "check-demo.sh prints no null fields for a sub-path fault" + fi + # The gateway lists a node's rosbags only for faults reported under the + # exact node path, so this section may have no recording to list. + SUBPATH_ROSBAGS_TEXT=$(check_demo_section_text "$SUBPATH_PLAIN" 12) + if check_demo_section "$SUBPATH_PLAIN" 12 | jq -e 'length > 0' > /dev/null 2>&1 \ + || grep -q "No rosbag recordings" <<< "$SUBPATH_ROSBAGS_TEXT"; then + pass "check-demo.sh section 12 lists rosbags or says there are none" + else + fail "check-demo.sh section 12 lists rosbags or says there are none" \ + "section 12: $(tr '\n' ';' <<< "$SUBPATH_ROSBAGS_TEXT" | head -c 300)" + fi +else + fail "setup: ${SUBPATH_CODE} from /processing/anomaly_detector/imu_sim is the first listed fault" \ + "first listed: $(curl -s -m 10 "${API_BASE}/faults" | jq -c '.items[0] | {fault_code, reporting_sources}' 2>/dev/null)" +fi + +# A source that only shares a name prefix with a node is not under it. +curl -s -m 20 -X DELETE "${API_BASE}/faults" > /dev/null || true +if report_fault "$SIBLING_CODE" "/processing/anomaly_detector_extra" \ + && poll_until "/faults" ".items[0].fault_code == \"${SIBLING_CODE}\"" 15; then + pass "setup: ${SIBLING_CODE} from /processing/anomaly_detector_extra is the first listed fault" + SIBLING_PLAIN=$(run_check_demo) || true + if grep -q "Could not map fault ${SIBLING_CODE} to a reporting App" <<< "$SIBLING_PLAIN" \ + && ! grep -q "=== 10\. " <<< "$SIBLING_PLAIN"; then + pass "check-demo.sh maps no App to a source that only shares a name prefix with its node" + else + fail "check-demo.sh maps no App to a source that only shares a name prefix with its node" \ + "$(grep -m1 -E 'Could not map|Fetching fault' <<< "$SIBLING_PLAIN")" + fi +else + fail "setup: ${SIBLING_CODE} from /processing/anomaly_detector_extra is the first listed fault" \ + "first listed: $(curl -s -m 10 "${API_BASE}/faults" | jq -c '.items[0] | {fault_code, reporting_sources}' 2>/dev/null)" +fi +curl -s -m 20 -X DELETE "${API_BASE}/faults" > /dev/null || true + +section "check-demo.sh waits for a sensor without a first message" + +# The first run above races the linking and often starts after it. Here the +# wait is needed on every run: on a fresh gateway the LiDAR is held before +# anything reads it and resumed 2 s into the run, well inside both the +# 30 s link wait and the 5 s first-message wait. +if docker restart "${DEMO_CONTAINER}" > /dev/null 2>&1; then + pass "setup: ${DEMO_CONTAINER} restarted" +else + fail "setup: ${DEMO_CONTAINER} restarted" "docker restart failed" +fi +wait_for_gateway 90 +if signal_sensor lidar_sim_node STOP; then + pass "setup: lidar_sim held right after /health" + ( sleep 2; signal_sensor lidar_sim_node CONT || true ) & + HOLD_PID=$! + HELD_RC=0 + HELD_TIMED=$(run_check_demo_timed "") || HELD_RC=$? + wait "$HOLD_PID" 2>/dev/null || true + signal_sensor lidar_sim_node CONT || true + HELD_PLAIN=$(untimed "$HELD_TIMED") + HELD_WAIT_TEXT=$(readiness_wait_output "$HELD_TIMED") + echo " readiness wait $(readiness_wait_ms "$HELD_TIMED") ms:" \ + "$(grep -vE '^ *$' <<< "$HELD_WAIT_TEXT" | tr '\n' ';' | head -c 300)" + HELD_MISSING="" + for n in 5 6 7 8; do + section_values_printed "$HELD_PLAIN" "$n" || HELD_MISSING="${HELD_MISSING} ${n}" + done + if [ "$HELD_RC" -eq 0 ] && [ -z "$HELD_MISSING" ] && ! grep -q ': null' <<< "$HELD_PLAIN"; then + pass "check-demo.sh with the LiDAR held exits 0 and prints values in sections 5-8" + else + fail "check-demo.sh with the LiDAR held exits 0 and prints values in sections 5-8" \ + "exit code ${HELD_RC}; sections without values:${HELD_MISSING:- none}; last lines: $(grep -vE '^ *$' <<< "$HELD_PLAIN" | tail -n 3 | tr '\n' ';')" + fi + if grep -q "^Waiting for .*lidar-sim" <<< "$HELD_WAIT_TEXT"; then + pass "check-demo.sh says it waits for the held LiDAR" + else + fail "check-demo.sh says it waits for the held LiDAR" \ + "wait printed: $(grep -vE '^ *$' <<< "$HELD_WAIT_TEXT" | tr '\n' ';' | head -c 300)" + fi +else + fail "setup: lidar_sim held right after /health" "no lidar_sim_node process in ${DEMO_CONTAINER}" +fi + +section "check-demo.sh with a late link and a sensor that resumes while it waits" + +# lidar_sim leaves the ROS graph and comes back 9 s into the run, past the 5 s +# first-message window, so only the link wait brings its values. The IMU is +# linked but held until 7 s, past its own window while the wait goes on for +# the LiDAR: the wait must still notice its messages and must not report it +# silent. +if docker restart "${DEMO_CONTAINER}" > /dev/null 2>&1; then + pass "setup: ${DEMO_CONTAINER} restarted" +else + fail "setup: ${DEMO_CONTAINER} restarted" "docker restart failed" +fi +wait_for_gateway 90 +wait_for_runtime_linking "/apps/imu-sim/data" 60 +if signal_sensor imu_sim_node STOP && remove_lidar; then + pass "setup: imu_sim held and lidar_sim off the ROS graph" + if poll_until "/apps/lidar-sim/data" '.items | length == 0' 30; then + pass "setup: the gateway has unlinked lidar-sim before the run" + else + fail "setup: the gateway has unlinked lidar-sim before the run" "lidar-sim data still listed after 30 s" + fi + ( sleep 7; signal_sensor imu_sim_node CONT || true ) & + IMU_TIMER_PID=$! + ( sleep 9; restore_lidar ) & + LIDAR_TIMER_PID=$! + LATE_RC=0 + LATE_TIMED=$(run_check_demo_timed "") || LATE_RC=$? + wait "$IMU_TIMER_PID" "$LIDAR_TIMER_PID" 2>/dev/null || true + signal_sensor imu_sim_node CONT || true + restore_lidar + LATE_PLAIN=$(untimed "$LATE_TIMED") + LATE_WAIT_TEXT=$(readiness_wait_output "$LATE_TIMED") + LATE_WAIT_MS=$(readiness_wait_ms "$LATE_TIMED") + echo " readiness wait ${LATE_WAIT_MS} ms:" \ + "$(grep -vE '^ *$' <<< "$LATE_WAIT_TEXT" | tr '\n' ';' | head -c 400)" + + LATE_MISSING="" + for n in 5 6 7 8; do + section_values_printed "$LATE_PLAIN" "$n" || LATE_MISSING="${LATE_MISSING} ${n}" + done + if [ "$LATE_RC" -eq 0 ] && [ -z "$LATE_MISSING" ] && ! grep -q ': null' <<< "$LATE_PLAIN" \ + && grep -q "^Waiting for the gateway to link lidar-sim" <<< "$LATE_WAIT_TEXT"; then + pass "check-demo.sh waits for a link that comes after the 5 s window, exits 0 and prints values in sections 5-8" + else + fail "check-demo.sh waits for a link that comes after the 5 s window, exits 0 and prints values in sections 5-8" \ + "exit code ${LATE_RC}; sections without values:${LATE_MISSING:- none}; waited ${LATE_WAIT_MS} ms" + fi + + if grep -q "^Waiting for a first message from imu-sim" <<< "$LATE_WAIT_TEXT" && [ "$LATE_WAIT_MS" -ge 8000 ]; then + pass "setup: the wait held the IMU past its 5 s window and went on past 8 s" + else + fail "setup: the wait held the IMU past its 5 s window and went on past 8 s" \ + "waited ${LATE_WAIT_MS} ms" + fi + if section_values_printed "$LATE_PLAIN" 6 && ! grep -q "No message from imu-sim" <<< "$LATE_WAIT_TEXT"; then + pass "check-demo.sh does not report the IMU silent after it sent data during the wait" + else + fail "check-demo.sh does not report the IMU silent after it sent data during the wait" \ + "wait printed: $(grep -vE '^ *$' <<< "$LATE_WAIT_TEXT" | grep -v '^Waiting' | tr '\n' ';' | head -c 300); section 6: $(check_demo_section "$LATE_PLAIN" 6 | jq -c . 2>/dev/null | head -c 120)" + fi + curl -s -m 20 -X DELETE "${API_BASE}/faults" > /dev/null || true +else + fail "setup: imu_sim held and lidar_sim off the ROS graph" "sensor nodes not found in ${DEMO_CONTAINER}" + signal_sensor imu_sim_node CONT || true + restore_lidar +fi + +section "check-demo.sh with a failed IMU on a fresh gateway" + +# The gateway keeps the last message of a topic it has read. A sensor that +# stops before its first read has no message at all, so the demo restarts +# and the IMU fails before anything reads it. +if docker restart "${DEMO_CONTAINER}" > /dev/null 2>&1; then + pass "setup: ${DEMO_CONTAINER} restarted" +else + fail "setup: ${DEMO_CONTAINER} restarted" "docker restart failed" +fi +wait_for_gateway 90 +wait_for_runtime_linking "/apps/imu-sim/data" 60 +assert_script_execution "compute-unit" "inject-failure" 30 +if poll_until "/faults" '.items | length > 0' 30; then + pass "setup: a fault is listed after the IMU failure" +else + fail "setup: a fault is listed after the IMU failure" "no fault after 30 s" +fi +if api_get "/health" && jq -e '."x-medkit-data-provider".pool_size == 0' <<< "$RESPONSE" > /dev/null 2>&1; then + pass "setup: the gateway has read no topic before check-demo.sh runs" +else + fail "setup: the gateway has read no topic before check-demo.sh runs" \ + "$(jq -c '."x-medkit-data-provider"' <<< "$RESPONSE" 2>/dev/null)" +fi + +# check-demo.sh bounds each request of its readiness wait to this many seconds. +CHECK_DEMO_REQUEST_TIMEOUT_SEC=3 + +FAILED_IMU_RC=0 +FAILED_IMU_TIMED=$(run_check_demo_timed "") || FAILED_IMU_RC=$? +FAILED_IMU_PLAIN=$(untimed "$FAILED_IMU_TIMED") +FAILED_IMU_WAIT_MS=$(readiness_wait_ms "$FAILED_IMU_TIMED") +echo " readiness wait ${FAILED_IMU_WAIT_MS} ms, exit code ${FAILED_IMU_RC}" + +if [ "$FAILED_IMU_RC" -eq 0 ]; then + pass "check-demo.sh exits 0 with the IMU failed" +else + fail "check-demo.sh exits 0 with the IMU failed" \ + "exit code ${FAILED_IMU_RC}: $(grep -vE '^ *$' <<< "$FAILED_IMU_PLAIN" | tail -n 3 | tr '\n' ';')" +fi +if grep -q ': null' <<< "$FAILED_IMU_PLAIN"; then + fail "check-demo.sh prints no null fields with the IMU failed" \ + "$(grep -B1 ': null' <<< "$FAILED_IMU_PLAIN" | head -10)" +else + pass "check-demo.sh prints no null fields with the IMU failed" +fi +FAILED_IMU_SECTION_6=$(check_demo_section_text "$FAILED_IMU_PLAIN" 6) +if grep -q "=== 6\. " <<< "$FAILED_IMU_PLAIN" \ + && check_demo_section "$FAILED_IMU_PLAIN" 6 | jq -e 'length == 0' > /dev/null 2>&1 \ + && grep -qi "no IMU data" <<< "$FAILED_IMU_SECTION_6"; then + pass "check-demo.sh section 6 says the IMU has no data" +else + fail "check-demo.sh section 6 says the IMU has no data" \ + "section 6: $(tr '\n' ';' <<< "$FAILED_IMU_SECTION_6" | head -c 300)" +fi +if section_values_printed "$FAILED_IMU_PLAIN" 5 && section_values_printed "$FAILED_IMU_PLAIN" 7; then + pass "check-demo.sh prints LiDAR and GPS values with the IMU failed" +else + fail "check-demo.sh prints LiDAR and GPS values with the IMU failed" \ + "section 5: $(check_demo_section "$FAILED_IMU_PLAIN" 5 | jq -c . 2>/dev/null | head -c 150); section 7: $(check_demo_section "$FAILED_IMU_PLAIN" 7 | jq -c . 2>/dev/null | head -c 150)" +fi +FAILED_IMU_MISSING="" +for n in 9 10 11 12; do + grep -q "=== ${n}\. " <<< "$FAILED_IMU_PLAIN" || FAILED_IMU_MISSING="${FAILED_IMU_MISSING} ${n}" +done +if [ -z "$FAILED_IMU_MISSING" ]; then + pass "check-demo.sh runs sections 9-12 with the IMU failed" +else + fail "check-demo.sh runs sections 9-12 with the IMU failed" "missing sections:${FAILED_IMU_MISSING}" +fi + +# The wait names only the sensor it waited for; LiDAR and GPS had data. +FAILED_IMU_WAIT_TEXT=$(readiness_wait_output "$FAILED_IMU_TIMED") +if grep -q "imu-sim" <<< "$FAILED_IMU_WAIT_TEXT" && ! grep -qE "lidar-sim|gps-sim" <<< "$FAILED_IMU_WAIT_TEXT"; then + pass "check-demo.sh's wait names the sensor it waited for" +else + fail "check-demo.sh's wait names the sensor it waited for" \ + "wait printed: $(grep -vE '^ *$' <<< "$FAILED_IMU_WAIT_TEXT" | tr '\n' ';' | head -c 300)" +fi + +# The IMU line reports how long the wait went on for it, which here is the +# whole wait. +FAILED_IMU_REPORTED_S=$(awk '/imu-sim/ && !/^Waiting/ && match($0, /[0-9]+s/) { + print substr($0, RSTART, RLENGTH - 1); exit }' <<< "$FAILED_IMU_WAIT_TEXT") +if [ -n "$FAILED_IMU_REPORTED_S" ] \ + && [ "$(( FAILED_IMU_REPORTED_S * 1000 - FAILED_IMU_WAIT_MS ))" -le 2000 ] \ + && [ "$(( FAILED_IMU_WAIT_MS - FAILED_IMU_REPORTED_S * 1000 ))" -le 2000 ]; then + pass "check-demo.sh reports the time it waited for the IMU" +else + fail "check-demo.sh reports the time it waited for the IMU" \ + "reported ${FAILED_IMU_REPORTED_S:-nothing} s; waited ${FAILED_IMU_WAIT_MS} ms" +fi + +# The wait is bounded by wall-clock time: DATA_WAIT_SEC plus one request. +# A cold IMU read blocks for the gateway's sample timeout, so a wait that +# counts passes instead of seconds overruns here. +if [ "$FAILED_IMU_WAIT_MS" -le $(( (30 + CHECK_DEMO_REQUEST_TIMEOUT_SEC) * 1000 )) ]; then + pass "check-demo.sh's readiness wait ends within the default 30 s plus one request" +else + fail "check-demo.sh's readiness wait ends within the default 30 s plus one request" \ + "waited ${FAILED_IMU_WAIT_MS} ms" +fi +# 08 is a whole number of seconds with a leading zero, not an octal number. +# Here the IMU's 5 s first-message window ends the wait before 8 s, so the +# wait must last at least 3 s: a value read as 0 or refused ends it at once. +for wait_sec in 0 2 08; do + WAIT_RC=0 + WAIT_TIMED=$(run_check_demo_timed "$wait_sec") || WAIT_RC=$? + WAIT_MS=$(readiness_wait_ms "$WAIT_TIMED") + WAIT_MIN_MS=0 + [ "$wait_sec" = 08 ] && WAIT_MIN_MS=3000 + echo " DATA_WAIT_SEC=${wait_sec}: readiness wait ${WAIT_MS} ms, exit code ${WAIT_RC}" + if [ "$WAIT_RC" -eq 0 ] && [ "$WAIT_MS" -ge "$WAIT_MIN_MS" ] \ + && [ "$WAIT_MS" -le $(( (10#$wait_sec + CHECK_DEMO_REQUEST_TIMEOUT_SEC) * 1000 )) ] \ + && grep -q "=== 12\. " <<< "$WAIT_TIMED"; then + pass "check-demo.sh with DATA_WAIT_SEC=${wait_sec} waits at most ${wait_sec} s plus one request and runs to section 12" + else + fail "check-demo.sh with DATA_WAIT_SEC=${wait_sec} waits at most ${wait_sec} s plus one request and runs to section 12" \ + "exit code ${WAIT_RC}, waited ${WAIT_MS} ms: $(grep -vE '^[0-9]+ *$' <<< "$WAIT_TIMED" | tail -n 2 | tr '\n' ';')" + fi +done + +assert_script_execution "compute-unit" "restore-normal" 30 + # --- Summary --- # print_summary runs via EXIT trap; exit code reflects test results diff --git a/tests/smoke_test_debounce.sh b/tests/smoke_test_debounce.sh index 39505bc..0087d6a 100755 --- a/tests/smoke_test_debounce.sh +++ b/tests/smoke_test_debounce.sh @@ -24,11 +24,12 @@ # burst and why healing_threshold is 0. # # Faults are injected through the gateway's own SOVD operation endpoint for -# /fault_manager/report_fault rather than by driving Gazebo. Navigation-driven -# injection is too timing-dependent for CI (see the header of -# smoke_test_turtlebot3.sh), and the contract under test is how the fault -# manager resolves thresholds per source, which a report exercises exactly as -# the detector does. +# /fault_manager/report_fault rather than by driving Gazebo. The thresholds are +# counts of FAILED and PASSED events per source, and only a direct report sends +# an exact number of them; a navigation goal makes the detector report as often +# as its state changes. The contract under test is how the fault manager +# resolves thresholds per source, which a report exercises exactly as the +# detector does. GATEWAY_URL="${1:-http://localhost:8080}" API_BASE="${GATEWAY_URL}/api/v1" diff --git a/tests/smoke_test_moveit.sh b/tests/smoke_test_moveit.sh index 30533e4..67b7208 100755 --- a/tests/smoke_test_moveit.sh +++ b/tests/smoke_test_moveit.sh @@ -18,7 +18,408 @@ SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" # shellcheck source=tests/smoke_lib.sh source "${SCRIPT_DIR}/smoke_lib.sh" -trap print_summary EXIT +DEMO_DIR="$(cd "${SCRIPT_DIR}/../demos/moveit_pick_place" && pwd)" +DEMO_CONTAINER="${MOVEIT_DEMO_CONTAINER:-moveit_medkit_demo_ci}" +ACTION="/panda_arm_controller/follow_joint_trajectory" +PICK_PLACE_PATTERN='^python3 /root/demo_ws/install/moveit_medkit_demo/lib/moveit_medkit_demo/pick_place_loop\.py' +FAULT_MANAGER_PATTERN='^/root/demo_ws/install/ros2_medkit_fault_manager/lib/ros2_medkit_fault_manager/fault_manager_node ' + +# container_python : run the Python script on stdin inside the demo +# container, with the ROS 2 environment sourced. The process is killed after +# 240 s, which is longer than every wait inside CONTROLLER_PROBE_PY together. +container_python() { + docker exec -i "${DEMO_CONTAINER}" bash -c ' + set +u + source /opt/ros/jazzy/setup.bash + source /root/demo_ws/install/setup.bash + exec timeout 240 python3 - "$@" + ' container_python "$@" +} + +# Arm controller probe, run with container_python. Every mode first waits up +# to 30 s for the action server, then up to 60 s until neither the arm +# controller nor MoveGroup has an active goal for 1 s. Exits non-zero when a +# wait runs out. +# idle : only these waits. +# fault : send two goals back to back. The second preempts the +# first, which manipulation_monitor reports as a fault. +# preempt : print ARMED, then preempt the next goal the arm +# controller starts, and print that goal's final status. +CONTROLLER_PROBE_PY=$(cat <<'PY' +import sys +import time + +import rclpy +from action_msgs.msg import GoalStatus, GoalStatusArray +from builtin_interfaces.msg import Duration +from control_msgs.action import FollowJointTrajectory +from rclpy.action import ActionClient +from rclpy.node import Node +from rclpy.qos import qos_profile_action_status_default +from trajectory_msgs.msg import JointTrajectoryPoint + +mode, arm_action = sys.argv[1], sys.argv[2] +ACTIVE = (GoalStatus.STATUS_ACCEPTED, GoalStatus.STATUS_EXECUTING, GoalStatus.STATUS_CANCELING) +TERMINAL = {GoalStatus.STATUS_SUCCEEDED: "SUCCEEDED", GoalStatus.STATUS_CANCELED: "CANCELED", + GoalStatus.STATUS_ABORTED: "ABORTED"} +READY = [0.0, -0.785, 0.0, -2.356, 0.0, 1.571, 0.785] + +rclpy.init() +node = Node("smoke_controller_probe") +goals = {} +subscriptions = [] +for action in (arm_action, "/move_action"): + goals[action] = [] + subscriptions.append(node.create_subscription( + GoalStatusArray, action + "/_action/status", + lambda msg, action=action: goals.__setitem__(action, list(msg.status_list)), + qos_profile_action_status_default)) +client = ActionClient(node, FollowJointTrajectory, arm_action) + + +def spin_until(condition, timeout): + deadline = time.monotonic() + timeout + while time.monotonic() < deadline: + rclpy.spin_once(node, timeout_sec=0.05) + if condition(): + return True + return False + + +def wait_future(future, timeout, what): + if not spin_until(future.done, timeout): + sys.exit(f"no {what} within {timeout} s") + return future.result() + + +def send_ready_goal(sec): + goal = FollowJointTrajectory.Goal() + goal.trajectory.joint_names = [f"panda_joint{i}" for i in range(1, 8)] + goal.trajectory.points = [JointTrajectoryPoint(positions=READY, time_from_start=Duration(sec=sec))] + handle = wait_future(client.send_goal_async(goal), 10, "goal response") + if not handle.accepted: + sys.exit("arm controller rejected the goal") + return handle + + +def final_status(handle): + return TERMINAL.get(wait_future(handle.get_result_async(), 10, "goal result").status, "UNKNOWN") + + +quiet_since = None + + +def idle(): + global quiet_since + busy = any(s.status in ACTIVE for status_list in goals.values() for s in status_list) + matched = all(sub.get_publisher_count() > 0 for sub in subscriptions) + if busy or not matched: + quiet_since = None + return False + quiet_since = quiet_since or time.monotonic() + return time.monotonic() - quiet_since >= 1.0 + + +if not client.wait_for_server(timeout_sec=30): + sys.exit("arm controller action server not available within 30 s") +if not spin_until(idle, 60): + sys.exit("arm controller or MoveGroup still busy after 60 s") + +if mode == "fault": + first = send_ready_goal(3) + second = send_ready_goal(1) + first_status, second_status = final_status(first), final_status(second) + print(f"first goal {first_status}, second goal {second_status}", flush=True) + if first_status == "SUCCEEDED": + sys.exit("the second goal did not preempt the first") + +if mode == "preempt": + seen = {bytes(s.goal_info.goal_id.uuid) for s in goals[arm_action]} + print("ARMED", flush=True) + target = [] + + def new_goal_executing(): + target[:] = [bytes(s.goal_info.goal_id.uuid) for s in goals[arm_action] + if s.status == GoalStatus.STATUS_EXECUTING + and bytes(s.goal_info.goal_id.uuid) not in seen][:1] + return bool(target) + + if not spin_until(new_goal_executing, 90): + sys.exit("no new arm goal started within 90 s") + competing = send_ready_goal(1) + target_status = [] + + def target_finished(): + target_status[:] = [TERMINAL[s.status] for s in goals[arm_action] + if bytes(s.goal_info.goal_id.uuid) == target[0] and s.status in TERMINAL] + return bool(target_status) + + if not spin_until(target_finished, 10): + sys.exit("the goal to preempt did not finish within 10 s") + print(f"PREEMPTED goal {target_status[0]}, competing goal {final_status(competing)}", flush=True) + if target_status[0] == "SUCCEEDED": + sys.exit("the competing goal did not preempt the running goal") + +node.destroy_node() +rclpy.shutdown() +PY +) + +# pick_place_loop sends MoveGroup goals to the arm controller that move-arm.sh +# drives. SIGSTOP pauses it without changing the demo. A MoveGroup goal it +# already sent still runs to its end, so wait until the arm is idle. +stop_pick_place_loop() { + docker exec "${DEMO_CONTAINER}" bash -c " + pid=\$(pgrep -f '${PICK_PLACE_PATTERN}') || exit 0 + kill -STOP \"\${pid}\" + " > /dev/null 2>&1 || true + if ! container_python idle "${ACTION}" <<< "${CONTROLLER_PROBE_PY}" \ + > /tmp/moveit_smoke_idle.log 2>&1; then + fail "setup: arm idle after pausing pick_place_loop" "$(tail -n 3 /tmp/moveit_smoke_idle.log)" + fi +} + +resume_pick_place_loop() { + docker exec "${DEMO_CONTAINER}" bash -c " + pid=\$(pgrep -f '${PICK_PLACE_PATTERN}') || exit 0 + kill -CONT \"\${pid}\" + " > /dev/null 2>&1 || true +} + +# SIGSTOP on the fault manager makes the gateway's ListFaults call time out, +# so GET /faults answers 503 as it does before the fault manager is up. +stop_fault_manager() { + docker exec "${DEMO_CONTAINER}" bash -c " + pid=\$(pgrep -f '${FAULT_MANAGER_PATTERN}') || exit 1 + kill -STOP \"\${pid}\" + " > /dev/null 2>&1 +} + +resume_fault_manager() { + docker exec "${DEMO_CONTAINER}" bash -c " + pid=\$(pgrep -f '${FAULT_MANAGER_PATTERN}') || exit 0 + kill -CONT \"\${pid}\" + " > /dev/null 2>&1 || true +} + +# start_preemptor : start the preempt probe in the background and return +# once it listens. Sets PREEMPTOR_PID. +start_preemptor() { + local log="$1" waited=0 + # Empty the log first: an ARMED line left by an earlier run must not count. + : > "${log}" + container_python preempt "${ACTION}" <<< "${CONTROLLER_PROBE_PY}" >> "${log}" 2>&1 & + PREEMPTOR_PID=$! + until grep -q '^ARMED' "${log}"; do + if ! kill -0 "${PREEMPTOR_PID}" 2> /dev/null || [ "${waited}" -ge 900 ]; then + return 1 + fi + sleep 0.1 + waited=$((waited + 1)) + done +} + +# finish_preemptor : wait for the preempt probe. It must have preempted a +# running goal. Sets PREEMPTED_GOAL_STATUS to that goal's final status. +finish_preemptor() { + local log="$1" rc=0 + wait "${PREEMPTOR_PID}" || rc=$? + PREEMPTED_GOAL_STATUS=$(sed -nE 's/^PREEMPTED goal ([A-Z]+),.*/\1/p' "${log}") + if [ "${rc}" -eq 0 ] && [ -n "${PREEMPTED_GOAL_STATUS}" ] && [ "${PREEMPTED_GOAL_STATUS}" != "SUCCEEDED" ]; then + pass "setup: the competing goal preempted a running goal (${PREEMPTED_GOAL_STATUS})" + else + fail "setup: the competing goal preempted a running goal" \ + "probe rc=${rc}: $(tail -n 2 "${log}" | tr '\n' ';')" + fi +} + +# Upper bound for one move-arm.sh run. The script itself allows 3 x 30 s per +# goal, and the demo cycle runs three goals. +MOVE_ARM_LIMIT_SEC=400 + +# run_move_arm : move-arm.sh without a TTY. Sets MOVE_ARM_RC. +run_move_arm() { + local log="$1" + shift + if CONTAINER_NAME="${DEMO_CONTAINER}" timeout "${MOVE_ARM_LIMIT_SEC}" \ + "${DEMO_DIR}/move-arm.sh" "$@" < /dev/null > "${log}" 2>&1; then + MOVE_ARM_RC=0 + else + MOVE_ARM_RC=$? + fi + if [ "${MOVE_ARM_RC}" -eq 124 ]; then + fail "move-arm.sh $* finishes within ${MOVE_ARM_LIMIT_SEC} s" "killed by timeout" + fi +} + +# Log of every run_move_arm_inside run, for checks across all of them. +INSIDE_ALL_LOG=/tmp/moveit_smoke_inside_all.log +: > "${INSIDE_ALL_LOG}" + +# Sources the ROS environment, as a shell in the container does. +ROS_ENV='source /opt/ros/jazzy/setup.bash && source /root/demo_ws/install/setup.bash' + +# run_move_arm_inside : run the copy of move-arm.sh at +# /tmp/move-arm.sh in the demo container. Bash code runs first; the +# ROS environment is there only if sources it. move-arm.sh runs only +# if succeeds, else MOVE_ARM_RC is 255 and the log is empty. Output +# goes to a file in the container: docker exec can drop the end of a ros2 +# CLI's stdout. Sets MOVE_ARM_RC. +run_move_arm_inside() { + local log="$1" setup="$2" + shift 2 + docker exec "${DEMO_CONTAINER}" bash -c " + rm -f /tmp/moveit_smoke_inside.log /tmp/moveit_smoke_inside.rc + { ${setup}; } || exit 0 + timeout ${MOVE_ARM_LIMIT_SEC} bash /tmp/move-arm.sh $* < /dev/null > /tmp/moveit_smoke_inside.log 2>&1 + echo \$? > /tmp/moveit_smoke_inside.rc + " > /dev/null 2>&1 || true + docker exec "${DEMO_CONTAINER}" cat /tmp/moveit_smoke_inside.log > "${log}" 2> /dev/null || : > "${log}" + MOVE_ARM_RC=$(docker exec "${DEMO_CONTAINER}" cat /tmp/moveit_smoke_inside.rc 2> /dev/null) || MOVE_ARM_RC=255 + cat "${log}" >> "${INSIDE_ALL_LOG}" + if [ "${MOVE_ARM_RC}" = 124 ]; then + fail "move-arm.sh $* in the container finishes within ${MOVE_ARM_LIMIT_SEC} s" "killed by timeout" + fi +} + +# write_fake_ros2 : a ros2 CLI at /ros2 that counts its calls in +# /lists and /calls. `action list` shows the arm action from call +# FAKE_LIST_MISSES + 1 on (default 0). `action send_goal` hands the goal to +# the real CLI in CONTAINER_NAME, except the first goal when FAKE_LOST_GOAL is +# set: it prints "Sending goal:" and then hangs ("hang"), exits 1 ("exit"), +# or prints "Goal accepted with ID" and hangs without a result ("accepted"). +write_fake_ros2() { + cat > "$1/ros2" <<'FAKE' +#!/bin/sh +count() { + n=$(($(cat "${FAKE_ROS2_DIR}/$1" 2> /dev/null || echo 0) + 1)) + echo "${n}" > "${FAKE_ROS2_DIR}/$1" +} +if [ "$1 $2" = "action list" ]; then + count lists + if [ "${n}" -gt "${FAKE_LIST_MISSES:-0}" ]; then + echo /panda_arm_controller/follow_joint_trajectory + fi + exit 0 +fi +[ "$1 $2" = "action send_goal" ] || exit 1 +shift 2 +count calls +if [ "${n}" -eq 1 ] && [ -n "${FAKE_LOST_GOAL:-}" ]; then + echo "Waiting for an action server to become available..." + echo "Sending goal:" + [ "${FAKE_LOST_GOAL}" != exit ] || exit 1 + if [ "${FAKE_LOST_GOAL}" = accepted ]; then + echo "Goal accepted with ID: 0123456789abcdef0123456789abcdef" + fi + echo "$$" > "${FAKE_ROS2_DIR}/hung.pid" + exec sleep 3600 +fi +exec docker exec -i "${CONTAINER_NAME}" bash -s -- "$@" <<'REMOTE' +set +u +source /opt/ros/jazzy/setup.bash +source /root/demo_ws/install/setup.bash +exec timeout 60 ros2 action send_goal "$@" +REMOTE +FAKE + chmod +x "$1/ros2" +} + +# check_not_sent : a goal that was never sent. The +# run exits non-zero on its own, prints a line matching (the +# failure's own output) exactly once, one failure line, and no resend line. +check_not_sent() { + local name="$1" log="$2" rc="$3" pattern="$4" matches failures + matches=$(grep -cE -- "${pattern}" "${log}" || true) + failures=$(grep -c '^Failed: ' "${log}" || true) + if [ "${rc}" -ne 0 ] && [ "${rc}" -ne 124 ] && [ "${matches}" -eq 1 ] && [ "${failures}" -eq 1 ] \ + && ! grep -q 'sending the goal again' "${log}"; then + pass "move-arm.sh ${name}: reported once, exit ${rc}, not sent again" + else + fail "move-arm.sh ${name}: reported once, non-zero exit, not sent again" \ + "rc=${rc}; lines matching '${pattern}': ${matches}; failure lines: ${failures}; tail: $(tail -n 4 "${log}" | tr '\n' ';')" + fi +} + +# ros2_control : `ros2 control ` in the demo container. Output +# goes to /tmp/moveit_smoke_ros2_control.log there. +ros2_control() { + docker exec "${DEMO_CONTAINER}" bash -c " + ${ROS_ENV} && timeout 60 ros2 control \"\$@\" > /tmp/moveit_smoke_ros2_control.log 2>&1 + " ros2_control "$@" +} + +# True while a test has joint_state_broadcaster inactive or unloaded. +JSB_DOWN=false + +reload_joint_state_broadcaster() { + { ros2_control load_controller --set-state active joint_state_broadcaster \ + || ros2_control set_controller_state joint_state_broadcaster active; } && JSB_DOWN=false +} + +# real_status : the final status the action client printed itself, read +# without move-arm.sh's own verdict. Empty when no goal finished. +real_status() { + { grep -F 'Goal finished with status:' "$1" || true; } | tail -n 1 | sed -E 's/.*status: *//' +} + +# reports_status