From 78a7d550374bbc45aa7bb55a354b4849819f2011 Mon Sep 17 00:00:00 2001 From: Bartosz Burda Date: Mon, 28 Sep 2026 15:39:31 +0200 Subject: [PATCH 01/53] fix(demos/sensor-diagnostics): show real values in check-demo.sh Data resources are keyed by full topic id (/sensors/scan, not scan) and the reading is nested under .data, so sections 5-7 printed null for every field. Section 8 read .value from the configurations list, which never carries a value; read each parameter's own detail endpoint instead. The fault collection has no entity_id/code fields, so the snapshot/bulk-data walkthrough always fell back to "skip" - resolve the owning App from the fault's reporting_sources instead. Also drop the Components/Apps columns that read fields the API never returns (area, namespace) for ones that do (description, component id), and fix the diagnostic_bridge id typo in the README (live id is hyphenated). Covered by a new smoke_test.sh section that runs check-demo.sh live against a fault and asserts no null fields. --- demos/sensor_diagnostics/README.md | 26 ++++----- demos/sensor_diagnostics/check-demo.sh | 79 +++++++++++++------------- tests/smoke_test.sh | 47 +++++++++++++++ 3 files changed, 99 insertions(+), 53 deletions(-) diff --git a/demos/sensor_diagnostics/README.md b/demos/sensor_diagnostics/README.md index d89a033..040d052 100644 --- a/demos/sensor_diagnostics/README.md +++ b/demos/sensor_diagnostics/README.md @@ -206,8 +206,8 @@ The gateway supports condition-based triggers that fire when specific events occ ### How It Works -1. `setup-triggers.sh` creates a trigger via `POST /api/v1/apps/diagnostic_bridge/triggers`: - - **Resource:** `/api/v1/apps/diagnostic_bridge/faults` (watches fault collection) +1. `setup-triggers.sh` creates a trigger via `POST /api/v1/apps/diagnostic-bridge/triggers`: + - **Resource:** `/api/v1/apps/diagnostic-bridge/faults` (watches fault collection) - **Condition:** `OnChange` (fires on any new or updated fault) - **Multishot:** `true` (fires repeatedly, not just once) - **Lifetime:** 3600 seconds (auto-expires after 1 hour) @@ -218,23 +218,23 @@ The gateway supports condition-based triggers that fire when specific events occ ```bash # Create a trigger -curl -X POST http://localhost:8080/api/v1/apps/diagnostic_bridge/triggers \ +curl -X POST http://localhost:8080/api/v1/apps/diagnostic-bridge/triggers \ -H "Content-Type: application/json" \ -d '{ - "resource": "/api/v1/apps/diagnostic_bridge/faults", + "resource": "/api/v1/apps/diagnostic-bridge/faults", "trigger_condition": {"condition_type": "OnChange"}, "multishot": true, "lifetime": 3600 }' | jq # List triggers -curl http://localhost:8080/api/v1/apps/diagnostic_bridge/triggers | jq +curl http://localhost:8080/api/v1/apps/diagnostic-bridge/triggers | jq # Watch events (replace TRIGGER_ID) -curl -N http://localhost:8080/api/v1/apps/diagnostic_bridge/triggers/TRIGGER_ID/events +curl -N http://localhost:8080/api/v1/apps/diagnostic-bridge/triggers/TRIGGER_ID/events # Delete a trigger -curl -X DELETE http://localhost:8080/api/v1/apps/diagnostic_bridge/triggers/TRIGGER_ID +curl -X DELETE http://localhost:8080/api/v1/apps/diagnostic-bridge/triggers/TRIGGER_ID ``` ## API Examples @@ -242,14 +242,14 @@ curl -X DELETE http://localhost:8080/api/v1/apps/diagnostic_bridge/triggers/TRIG ### Read Sensor Data ```bash -# Get LiDAR scan -curl http://localhost:8080/api/v1/apps/lidar-sim/data/scan | jq '.ranges[:5]' +# Get LiDAR scan (topic id is /sensors/scan, percent-encoded in the URL) +curl http://localhost:8080/api/v1/apps/lidar-sim/data/sensors%2Fscan | jq '.data.ranges[:5]' -# Get IMU data -curl http://localhost:8080/api/v1/apps/imu-sim/data/imu | jq '.linear_acceleration' +# Get IMU data (topic id is /sensors/imu) +curl http://localhost:8080/api/v1/apps/imu-sim/data/sensors%2Fimu | jq '.data.linear_acceleration' -# Get GPS fix -curl http://localhost:8080/api/v1/apps/gps-sim/data/fix | jq '{lat: .latitude, lon: .longitude}' +# Get GPS fix (topic id is /sensors/fix) +curl http://localhost:8080/api/v1/apps/gps-sim/data/sensors%2Ffix | jq '{lat: .data.latitude, lon: .data.longitude}' ``` ### View Configurations diff --git a/demos/sensor_diagnostics/check-demo.sh b/demos/sensor_diagnostics/check-demo.sh index 48cec64..ed34d88 100755 --- a/demos/sensor_diagnostics/check-demo.sh +++ b/demos/sensor_diagnostics/check-demo.sh @@ -52,40 +52,45 @@ echo_step "2. Listing All Areas (Namespaces)" curl -s "${API_BASE}/areas" | jq '.items[] | {id: .id, name: .name, description: .description}' echo_step "3. Listing All Components" -curl -s "${API_BASE}/components" | jq '.items[] | {id: .id, name: .name, area: .area}' +curl -s "${API_BASE}/components" | jq '.items[] | {id: .id, name: .name, description: .description}' echo_step "4. Listing All Apps (ROS 2 Nodes)" -curl -s "${API_BASE}/apps" | jq '.items[] | {id: .id, name: .name, namespace: .namespace}' +curl -s "${API_BASE}/apps" | jq '.items[] | {id: .id, name: .name, component: .["x-medkit"].component_id}' echo_step "5. Reading LiDAR Data" echo "Getting latest scan from LiDAR simulator..." -curl -s "${API_BASE}/apps/lidar-sim/data/scan" | jq '{ - angle_min: .angle_min, - angle_max: .angle_max, - range_min: .range_min, - range_max: .range_max, - sample_ranges: .ranges[:5] +curl -s "${API_BASE}/apps/lidar-sim/data/sensors%2Fscan" | jq '{ + angle_min: .data.angle_min, + angle_max: .data.angle_max, + range_min: .data.range_min, + range_max: .data.range_max, + sample_ranges: .data.ranges[:5] }' echo_step "6. Reading IMU Data" echo "Getting latest IMU reading..." -curl -s "${API_BASE}/apps/imu-sim/data/imu" | jq '{ - linear_acceleration: .linear_acceleration, - angular_velocity: .angular_velocity +curl -s "${API_BASE}/apps/imu-sim/data/sensors%2Fimu" | jq '{ + linear_acceleration: .data.linear_acceleration, + angular_velocity: .data.angular_velocity }' echo_step "7. Reading GPS Fix" echo "Getting current GPS position..." -curl -s "${API_BASE}/apps/gps-sim/data/fix" | jq '{ - latitude: .latitude, - longitude: .longitude, - altitude: .altitude, - status: .status +curl -s "${API_BASE}/apps/gps-sim/data/sensors%2Ffix" | jq '{ + latitude: .data.latitude, + longitude: .data.longitude, + altitude: .data.altitude, + status: .data.status }' echo_step "8. Listing LiDAR Configurations" echo "These parameters can be modified at runtime to inject faults..." -curl -s "${API_BASE}/apps/lidar-sim/configurations" | jq '.items[] | {name: .name, value: .value, type: .type}' +# The list endpoint carries id/name/type only; the value is on each parameter's +# own detail endpoint. +LIDAR_CONFIG_IDS=$(curl -s "${API_BASE}/apps/lidar-sim/configurations" | jq -r '.items[].id') +while IFS= read -r cfg_id; do + curl -s "${API_BASE}/apps/lidar-sim/configurations/${cfg_id}" | jq '{name: .id, value: .data, type: "parameter"}' +done <<< "$LIDAR_CONFIG_IDS" echo_step "9. Checking Current Faults" FAULTS_JSON=$(curl -s "${API_BASE}/faults") @@ -94,28 +99,22 @@ echo "$FAULTS_JSON" | jq '.' # If there are faults, demonstrate snapshot / bulk-data endpoints FAULT_COUNT=$(echo "$FAULTS_JSON" | jq '.items | length') if [ "$FAULT_COUNT" -gt 0 ]; then - # Find the first fault that has both a non-null entity_id and code - FIRST_FAULT_ENTRY=$(echo "$FAULTS_JSON" | jq -r '.items[] | select(.entity_id != null and .code != null) | "\(.entity_type) \(.entity_id) \(.code)"' | head -n 1) - - if [ -z "$FIRST_FAULT_ENTRY" ]; then + # The fault collection carries fault_code and reporting_sources (ROS node + # paths), not an entity id. Resolve the owning App by matching the first + # reporting source against each App's ROS node. + FIRST_FAULT=$(echo "$FAULTS_JSON" | jq -r '.items[0].fault_code') + REPORTING_SOURCE=$(echo "$FAULTS_JSON" | jq -r '.items[0].reporting_sources[0] // empty') + FIRST_ENTITY=$(curl -s "${API_BASE}/apps" | jq -r --arg node "$REPORTING_SOURCE" \ + '.items[] | select(.["x-medkit"].ros2.node == $node) | .id' | head -n 1) + + if [ -z "$FIRST_ENTITY" ]; then echo "" - echo " Faults exist but none provide both 'entity_id' and 'code'." + echo " Could not map fault ${FIRST_FAULT} to a reporting App (source: ${REPORTING_SOURCE:-none})." echo " Skipping snapshot and bulk-data demonstration." else - FIRST_ENTITY_TYPE=$(echo "$FIRST_FAULT_ENTRY" | awk '{print $1}') - FIRST_ENTITY=$(echo "$FIRST_FAULT_ENTRY" | awk '{print $2}') - FIRST_FAULT=$(echo "$FIRST_FAULT_ENTRY" | awk '{print $3}') - # Map entity_type to plural resource path (e.g., "app" -> "apps") - case "$FIRST_ENTITY_TYPE" in - app|apps) ENTITY_PATH="apps" ;; - component|components) ENTITY_PATH="components" ;; - area|areas) ENTITY_PATH="areas" ;; - *) ENTITY_PATH="apps" ;; - esac - echo_step "10. Fault Detail with Environment Data (Snapshots)" - echo "Fetching fault ${FIRST_FAULT} on ${ENTITY_PATH}/${FIRST_ENTITY}..." - curl -s "${API_BASE}/${ENTITY_PATH}/${FIRST_ENTITY}/faults/${FIRST_FAULT}" | jq '{ + echo "Fetching fault ${FIRST_FAULT} on apps/${FIRST_ENTITY}..." + curl -s "${API_BASE}/apps/${FIRST_ENTITY}/faults/${FIRST_FAULT}" | jq '{ code: .item.code, status: .item.status, environment_data: { @@ -126,11 +125,11 @@ if [ "$FAULT_COUNT" -gt 0 ]; then echo_step "11. Bulk-Data Categories (Rosbag Recordings)" echo "Checking available bulk-data categories..." - curl -s "${API_BASE}/${ENTITY_PATH}/${FIRST_ENTITY}/bulk-data" | jq '.' + curl -s "${API_BASE}/apps/${FIRST_ENTITY}/bulk-data" | jq '.' echo_step "12. Bulk-Data Descriptors (Rosbag Files)" echo "Listing available rosbag recordings..." - curl -s "${API_BASE}/${ENTITY_PATH}/${FIRST_ENTITY}/bulk-data/rosbags" | jq '.items[] | { + curl -s "${API_BASE}/apps/${FIRST_ENTITY}/bulk-data/rosbags" | jq '.items[] | { id: .id, name: .name, size: .size, @@ -155,9 +154,9 @@ echo " ./inject-drift.sh # Inject sensor drift" echo " ./restore-normal.sh # Restore normal operation" echo "" echo "📸 After injecting a fault, check snapshots and rosbags:" -echo " curl ${API_BASE}/faults | jq # List faults" -echo " curl ${API_BASE}/components/lidar-unit/faults/ | jq # Fault detail + snapshots" -echo " curl ${API_BASE}/components/lidar-unit/bulk-data/rosbags | jq # List rosbag recordings" +echo " curl ${API_BASE}/faults | jq # List faults" +echo " curl ${API_BASE}/apps/diagnostic-bridge/faults/ | jq # Fault detail + snapshots" +echo " curl ${API_BASE}/apps/diagnostic-bridge/bulk-data/rosbags | jq # List rosbag recordings" echo "" echo "🌐 Web UI: http://localhost:3000" echo "🌐 REST API: http://localhost:8080/api/v1/" diff --git a/tests/smoke_test.sh b/tests/smoke_test.sh index cb03779..5847805 100755 --- a/tests/smoke_test.sh +++ b/tests/smoke_test.sh @@ -133,6 +133,53 @@ else fail "GET fault detail returns 200" "unexpected status code" fi +section "Check-Demo Script" + +# The rosbag recording finalizes duration_after_sec after confirmation, so +# poll for it rather than racing check-demo.sh against the write. +echo " Waiting for rosbag recording to finish (max 15s)..." +if poll_until "/apps/diagnostic-bridge/bulk-data/rosbags" '.items | length > 0' 15; then + pass "rosbag recording available before running check-demo.sh" +else + fail "rosbag recording available before running check-demo.sh" "no rosbag after 15s" +fi + +# check-demo.sh is the interactive tour a user runs by hand. Run it live +# against the gateway while the LIDAR_SIM fault above is still active, and +# check that every field it labels carries a real value, not a stale/wrong +# resource path resolving to null. +CHECK_DEMO_SCRIPT="${SCRIPT_DIR}/../demos/sensor_diagnostics/check-demo.sh" +CHECK_DEMO_OUTPUT=$(GATEWAY_URL="$GATEWAY_URL" bash "$CHECK_DEMO_SCRIPT" 2>&1) || true +# shellcheck disable=SC2001 +CHECK_DEMO_PLAIN=$(sed 's/\x1b\[[0-9;]*m//g' <<< "$CHECK_DEMO_OUTPUT") + +if grep -q ': null' <<< "$CHECK_DEMO_PLAIN"; then + fail "check-demo.sh prints no null fields" "$(grep -B1 ': null' <<< "$CHECK_DEMO_PLAIN" | head -10)" +else + pass "check-demo.sh prints no null fields" +fi + +if grep -q "10\. Fault Detail with Environment Data" <<< "$CHECK_DEMO_PLAIN"; then + pass "check-demo.sh runs the fault detail section for the active fault" +else + fail "check-demo.sh runs the fault detail section for the active fault" \ + "section 10 did not run: the owning App was not resolved" +fi + +if grep -q '"snapshot_count": 0' <<< "$CHECK_DEMO_PLAIN"; then + fail "check-demo.sh fault detail shows real snapshot data" "snapshot_count is 0" +elif grep -q '"snapshot_count":' <<< "$CHECK_DEMO_PLAIN"; then + pass "check-demo.sh fault detail shows real snapshot data" +else + fail "check-demo.sh fault detail shows real snapshot data" "snapshot_count field missing" +fi + +if grep -q '"id": "fault_LIDAR_SIM' <<< "$CHECK_DEMO_PLAIN"; then + pass "check-demo.sh bulk-data section lists a real rosbag recording" +else + fail "check-demo.sh bulk-data section lists a real rosbag recording" "no fault_LIDAR_SIM rosbag id found" +fi + # Cleanup: restore config + delete fault echo " Cleaning up: restoring config and clearing fault..." curl -s -X PUT "${API_BASE}/apps/lidar-sim/configurations/noise_stddev" \ From 337c2ff216328b1e27720b04271958a2fd59732a Mon Sep 17 00:00:00 2001 From: Bartosz Burda Date: Mon, 28 Sep 2026 15:39:45 +0200 Subject: [PATCH 02/53] fix(demos/turtlebot3): show real values and fix the fault trigger check-entities.sh and check-faults.sh read fields the SOVD entity and fault responses never carry (area, category, is_located_on, hosted_by, code, reporter_id, message, timestamp), so every labeled field printed null. Read the real fields instead (fault_code, severity_label, reporting_sources, x-medkit.component_id, ...), matching the shape moveit_pick_place/check-faults.sh already uses. setup-triggers.sh and watch-triggers.sh both watched apps/diagnostic-bridge, which reports nothing for this demo - faults arrive from apps/anomaly-detector. Fixed both to the live entity id. The nav-failure inject script never brings a fault to CONFIRMED state under the gateway's own OnChange notification path (verified live: inject-localization-failure fires reliably, inject-nav-failure never does across repeated attempts), so the hinted inject script and README example now point at inject-localization-failure.sh. Covered by a new smoke_test_turtlebot3.sh section that runs both scripts against a live fault and asserts no null fields, and a new smoke_test_navigation.sh section that drives setup-triggers.sh / watch-triggers.sh / inject-localization-failure.sh and asserts the SSE stream delivers an event. --- demos/turtlebot3_integration/README.md | 18 ++--- .../turtlebot3_integration/check-entities.sh | 18 ++--- demos/turtlebot3_integration/check-faults.sh | 27 ++++++-- .../turtlebot3_integration/setup-triggers.sh | 10 +-- .../turtlebot3_integration/watch-triggers.sh | 2 +- tests/smoke_test_navigation.sh | 45 +++++++++++++ tests/smoke_test_turtlebot3.sh | 67 +++++++++++++++++++ 7 files changed, 156 insertions(+), 31 deletions(-) diff --git a/demos/turtlebot3_integration/README.md b/demos/turtlebot3_integration/README.md index a2c0566..ae1e255 100644 --- a/demos/turtlebot3_integration/README.md +++ b/demos/turtlebot3_integration/README.md @@ -404,7 +404,7 @@ GATEWAY_URL=http://192.168.1.10:8080 ./inject-nav-failure.sh ## Triggers (Condition-Based Alerts) -The gateway supports condition-based triggers that fire when specific events occur, delivering notifications via Server-Sent Events (SSE). This demo creates a fault-monitoring trigger that alerts on any new or updated faults reported by the anomaly detector (including navigation failures). +The gateway supports condition-based triggers that fire when specific events occur, delivering notifications via Server-Sent Events (SSE). This demo creates a fault-monitoring trigger that alerts on any new or updated faults reported by the anomaly detector, such as localization uncertainty. ### Setup @@ -419,13 +419,13 @@ The gateway supports condition-based triggers that fire when specific events occ ./watch-triggers.sh # Terminal 2: Inject a fault - the trigger fires in Terminal 3! -./inject-nav-failure.sh +./inject-localization-failure.sh ``` ### How It Works -1. `setup-triggers.sh` creates a trigger via `POST /api/v1/apps/anomaly_detector/triggers`: - - **Resource:** `/api/v1/apps/anomaly_detector/faults` (watches fault collection) +1. `setup-triggers.sh` creates a trigger via `POST /api/v1/apps/anomaly-detector/triggers`: + - **Resource:** `/api/v1/apps/anomaly-detector/faults` (watches fault collection) - **Condition:** `OnChange` (fires on any new or updated fault) - **Multishot:** `true` (fires repeatedly, not just once) - **Lifetime:** 3600 seconds (auto-expires after 1 hour) @@ -436,23 +436,23 @@ The gateway supports condition-based triggers that fire when specific events occ ```bash # Create a trigger -curl -X POST http://localhost:8080/api/v1/apps/anomaly_detector/triggers \ +curl -X POST http://localhost:8080/api/v1/apps/anomaly-detector/triggers \ -H "Content-Type: application/json" \ -d '{ - "resource": "/api/v1/apps/anomaly_detector/faults", + "resource": "/api/v1/apps/anomaly-detector/faults", "trigger_condition": {"condition_type": "OnChange"}, "multishot": true, "lifetime": 3600 }' | jq # List triggers -curl http://localhost:8080/api/v1/apps/anomaly_detector/triggers | jq +curl http://localhost:8080/api/v1/apps/anomaly-detector/triggers | jq # Watch events (replace TRIGGER_ID) -curl -N http://localhost:8080/api/v1/apps/anomaly_detector/triggers/TRIGGER_ID/events +curl -N http://localhost:8080/api/v1/apps/anomaly-detector/triggers/TRIGGER_ID/events # Delete a trigger -curl -X DELETE http://localhost:8080/api/v1/apps/anomaly_detector/triggers/TRIGGER_ID +curl -X DELETE http://localhost:8080/api/v1/apps/anomaly-detector/triggers/TRIGGER_ID ``` ## Fault Injection Scenarios diff --git a/demos/turtlebot3_integration/check-entities.sh b/demos/turtlebot3_integration/check-entities.sh index 5b04968..259fcc5 100755 --- a/demos/turtlebot3_integration/check-entities.sh +++ b/demos/turtlebot3_integration/check-entities.sh @@ -40,26 +40,26 @@ echo_step "1. Areas (Namespace Groupings)" curl -s "${API_BASE}/areas" | jq '.items[] | {id: .id, name: .name, description: .description}' echo_step "2. Components (Hardware/Logical Units)" -curl -s "${API_BASE}/components" | jq '.items[] | {id: .id, name: .name, type: .type, area: .area}' +curl -s "${API_BASE}/components" | jq '.items[] | {id: .id, name: .name, type: .type, description: .description}' echo_step "3. Apps (ROS 2 Nodes)" -curl -s "${API_BASE}/apps" | jq '.items[] | {id: .id, name: .name, category: .category, component: .is_located_on}' +curl -s "${API_BASE}/apps" | jq '.items[] | {id: .id, name: .name, component: .["x-medkit"].component_id}' echo_step "4. Functions (High-level Capabilities)" -curl -s "${API_BASE}/functions" | jq '.items[] | {id: .id, name: .name, category: .category, hosted_by: .hosted_by}' +curl -s "${API_BASE}/functions" | jq '.items[] | {id: .id, name: .name, description: .description}' echo_step "5. Sample Data (LiDAR Scan)" echo "Getting latest LiDAR scan from TurtleBot3..." curl -s "${API_BASE}/apps/turtlebot3-node/data/scan" 2>/dev/null | jq '{ - angle_min: .angle_min, - angle_max: .angle_max, - range_min: .range_min, - range_max: .range_max, - sample_ranges: .ranges[:5] + angle_min: .data.angle_min, + angle_max: .data.angle_max, + range_min: .data.range_min, + range_max: .data.range_max, + sample_ranges: .data.ranges[:5] }' || echo " (LiDAR data not available - Gazebo may still be starting)" echo_step "6. Faults" -curl -s "${API_BASE}/faults" | jq '.items[] | {code: .code, severity: .severity, reporter: .reporter_id}' +curl -s "${API_BASE}/faults" | jq '.items[] | {code: .fault_code, severity: .severity_label, sources: .reporting_sources}' echo "" echo -e "${GREEN}✓ Entity hierarchy exploration complete!${NC}" diff --git a/demos/turtlebot3_integration/check-faults.sh b/demos/turtlebot3_integration/check-faults.sh index 944ed6d..989b49f 100755 --- a/demos/turtlebot3_integration/check-faults.sh +++ b/demos/turtlebot3_integration/check-faults.sh @@ -1,6 +1,7 @@ #!/bin/bash # Check current faults from ros2_medkit gateway -# Faults are collected from Nav2/TurtleBot3 via diagnostic_bridge +# Faults are collected from Nav2/TurtleBot3 via anomaly_detector (direct) and +# diagnostic_bridge (legacy /diagnostics path) GATEWAY_URL="${GATEWAY_URL:-http://localhost:8080}" API_BASE="${GATEWAY_URL}/api/v1" @@ -36,18 +37,30 @@ if [ "$FAULT_COUNT" = "0" ]; then echo " No active faults - system is healthy!" else echo "$FAULTS" | jq '.items[] | { - code: .code, - severity: .severity, - reporter: .reporter_id, - message: .message, - timestamp: .timestamp + code: .fault_code, + severity: .severity_label, + status: .status, + description: .description, + sources: .reporting_sources, + occurrences: .occurrence_count, + first_occurred: .first_occurred, + last_occurred: .last_occurred }' fi echo "" echo "📊 Fault Summary:" echo " Total active faults: $FAULT_COUNT" + +# Show fault counts by severity if any exist +if [ "$FAULT_COUNT" != "0" ]; then + echo "" + echo " By severity:" + echo "$FAULTS" | jq -r '.items | group_by(.severity_label) | .[] | " \(.[0].severity_label): \(length)"' +fi + echo "" echo "Commands:" echo " Clear all faults: curl -X DELETE ${API_BASE}/faults" -echo " Check specific area: curl ${API_BASE}/areas/robot/faults | jq" +echo " Check area faults: curl ${API_BASE}/areas/navigation/faults | jq" +echo " Check component faults: curl ${API_BASE}/components/nav2-stack/faults | jq" diff --git a/demos/turtlebot3_integration/setup-triggers.sh b/demos/turtlebot3_integration/setup-triggers.sh index b9390f6..f10974a 100755 --- a/demos/turtlebot3_integration/setup-triggers.sh +++ b/demos/turtlebot3_integration/setup-triggers.sh @@ -1,10 +1,10 @@ #!/bin/bash # Create fault-monitoring trigger for turtlebot3 integration demo -# Alerts on any fault change reported via the diagnostic bridge - the -# anomaly-detector app has no faults of its own, faults arrive from -# /diagnostics through the bridge. +# Alerts on any fault change reported by the anomaly detector - navigation +# and localization faults are reported directly, not via the diagnostic +# bridge. export ENTITY_TYPE="apps" -export ENTITY_ID="diagnostic-bridge" -export INJECT_HINT="./inject-nav-failure.sh" +export ENTITY_ID="anomaly-detector" +export INJECT_HINT="./inject-localization-failure.sh" # shellcheck disable=SC1091 source "$(cd "$(dirname "$0")" && pwd)/../../lib/setup-trigger.sh" diff --git a/demos/turtlebot3_integration/watch-triggers.sh b/demos/turtlebot3_integration/watch-triggers.sh index d6d9b6c..aa0746f 100755 --- a/demos/turtlebot3_integration/watch-triggers.sh +++ b/demos/turtlebot3_integration/watch-triggers.sh @@ -2,6 +2,6 @@ # Watch trigger events for turtlebot3 integration demo # Connects to SSE stream and prints fault events in real time export ENTITY_TYPE="apps" -export ENTITY_ID="diagnostic-bridge" +export ENTITY_ID="anomaly-detector" # shellcheck disable=SC1091 source "$(cd "$(dirname "$0")" && pwd)/../../lib/watch-trigger.sh" "$@" diff --git a/tests/smoke_test_navigation.sh b/tests/smoke_test_navigation.sh index 18b3d90..ec7fcf2 100755 --- a/tests/smoke_test_navigation.sh +++ b/tests/smoke_test_navigation.sh @@ -274,6 +274,51 @@ section "Localization held while driving" sleep "$FAULT_SETTLE" assert_localization_certain "after the drive" +section "Trigger delivers fault events" + +# Drive setup-triggers.sh / watch-triggers.sh / inject-localization-failure.sh +# exactly as a user would: the trigger watches apps/${DETECTOR_APP}, which is +# what reports both navigation and localization faults directly. +# inject-nav-failure's goal is rejected by the planner before a fault confirms, +# which never reaches the confirmed state the trigger fires on; +# inject-localization-failure reliably confirms LOCALIZATION_UNCERTAINTY. +TB3_DIR="${SCRIPT_DIR}/../demos/turtlebot3_integration" + +SETUP_OUTPUT=$(cd "$TB3_DIR" && GATEWAY_URL="$GATEWAY_URL" bash ./setup-triggers.sh 2>&1) || true +TRIGGER_ID=$(sed -n 's/^ ID:[[:space:]]*//p' <<< "$SETUP_OUTPUT" | head -1) + +if [ -n "$TRIGGER_ID" ]; then + pass "setup-triggers.sh creates a trigger on apps/${DETECTOR_APP}" +else + fail "setup-triggers.sh creates a trigger on apps/${DETECTOR_APP}" "$(tail -5 <<< "$SETUP_OUTPUT")" +fi + +if [ -n "$TRIGGER_ID" ]; then + WATCH_LOG=$(mktemp) + (cd "$TB3_DIR" && GATEWAY_URL="$GATEWAY_URL" timeout 20 bash ./watch-triggers.sh "$TRIGGER_ID") \ + > "$WATCH_LOG" 2>&1 & + WATCH_PID=$! + sleep 2 + + (cd "$TB3_DIR" && GATEWAY_URL="$GATEWAY_URL" bash ./inject-localization-failure.sh) > /dev/null 2>&1 || true + + wait "$WATCH_PID" 2>/dev/null || true + + if grep -q "Event received" "$WATCH_LOG"; then + pass "watch-triggers.sh receives at least one event after inject-localization-failure.sh" + else + fail "watch-triggers.sh receives at least one event after inject-localization-failure.sh" \ + "$(tail -5 "$WATCH_LOG")" + fi + rm -f "$WATCH_LOG" + + curl -s -o /dev/null -X DELETE "${API_BASE}/apps/${DETECTOR_APP}/triggers/${TRIGGER_ID}" || true +fi + +# Cleanup: clear the injected fault and any incidental localization latch so +# a re-run of this script on the same container starts clean. +curl -s -X DELETE "${API_BASE}/faults" > /dev/null || true + # --- Summary --- # print_summary runs via EXIT trap; exit code reflects test results diff --git a/tests/smoke_test_turtlebot3.sh b/tests/smoke_test_turtlebot3.sh index 54a1fcd..075e5ab 100755 --- a/tests/smoke_test_turtlebot3.sh +++ b/tests/smoke_test_turtlebot3.sh @@ -94,6 +94,73 @@ section "Logs" assert_non_empty_items "/apps/medkit-gateway/logs" +section "Check-Entities and Check-Faults Scripts" + +TB3_DIR="${SCRIPT_DIR}/../demos/turtlebot3_integration" + +# Inject a real fault via the Scripts API so check-entities.sh (section 6) +# and check-faults.sh exercise the fault-carrying fields, not just the +# empty case. +echo " Injecting navigation failure via Scripts API..." +INJECT_RESPONSE=$(curl -s -m 30 -X POST "${API_BASE}/components/nav2-stack/scripts/inject-nav-failure/executions" \ + -H "Content-Type: application/json" -d '{"execution_type": "now"}') || true +INJECT_EXEC_ID=$(echo "$INJECT_RESPONSE" | jq -r '.id // empty') +if [ -n "$INJECT_EXEC_ID" ]; then + elapsed=0 + while [ $elapsed -lt 30 ]; do + st=$(curl -s "${API_BASE}/components/nav2-stack/scripts/inject-nav-failure/executions/${INJECT_EXEC_ID}" | jq -r '.status') + if [ "$st" = "completed" ] || [ "$st" = "failed" ]; then + break + fi + sleep 1 + elapsed=$((elapsed + 1)) + done +fi + +echo " Waiting for NAVIGATION_GOAL_ABORTED fault to appear (max 15s)..." +if poll_until "/faults" '.items[] | select(.fault_code == "NAVIGATION_GOAL_ABORTED")' 15; then + pass "NAVIGATION_GOAL_ABORTED fault appeared in /faults" +else + fail "NAVIGATION_GOAL_ABORTED fault appeared in /faults" "fault not found after 15s" +fi + +CHECK_ENTITIES_OUTPUT=$(cd "$TB3_DIR" && GATEWAY_URL="$GATEWAY_URL" bash ./check-entities.sh 2>&1) || true +# shellcheck disable=SC2001 +CHECK_ENTITIES_PLAIN=$(sed 's/\x1b\[[0-9;]*m//g' <<< "$CHECK_ENTITIES_OUTPUT") + +if grep -q ': null' <<< "$CHECK_ENTITIES_PLAIN"; then + fail "check-entities.sh prints no null fields" "$(grep -B1 ': null' <<< "$CHECK_ENTITIES_PLAIN" | head -10)" +else + pass "check-entities.sh prints no null fields" +fi + +if grep -q "NAVIGATION_GOAL_ABORTED" <<< "$CHECK_ENTITIES_PLAIN"; then + pass "check-entities.sh faults section shows the active fault code" +else + fail "check-entities.sh faults section shows the active fault code" "NAVIGATION_GOAL_ABORTED not in output" +fi + +CHECK_FAULTS_OUTPUT=$(cd "$TB3_DIR" && GATEWAY_URL="$GATEWAY_URL" bash ./check-faults.sh 2>&1) || true +# shellcheck disable=SC2001 +CHECK_FAULTS_PLAIN=$(sed 's/\x1b\[[0-9;]*m//g' <<< "$CHECK_FAULTS_OUTPUT") + +if grep -q ': null' <<< "$CHECK_FAULTS_PLAIN"; then + fail "check-faults.sh prints no null fields" "$(grep -B1 ': null' <<< "$CHECK_FAULTS_PLAIN" | head -10)" +else + pass "check-faults.sh prints no null fields" +fi + +if grep -q "NAVIGATION_GOAL_ABORTED" <<< "$CHECK_FAULTS_PLAIN"; then + pass "check-faults.sh shows the active fault code" +else + fail "check-faults.sh shows the active fault code" "NAVIGATION_GOAL_ABORTED not in output" +fi + +# Cleanup: clear all faults so smoke_test_navigation.sh (run next on this +# stack) does not inherit a latched fault confirmation. +echo " Cleaning up: clearing faults..." +curl -s -X DELETE "${API_BASE}/faults" > /dev/null || true + section "Triggers" assert_triggers_crud "apps" "diagnostic-bridge" "/api/v1/apps/diagnostic-bridge/faults" From a207b6e38ad839432f80aabd37ead4f42d7c5116 Mon Sep 17 00:00:00 2001 From: Bartosz Burda Date: Mon, 28 Sep 2026 15:39:52 +0200 Subject: [PATCH 03/53] fix(demos/ota-nav2-sensor-fix): name the robot the image actually builds run-demo.sh's text called the simulated robot a TurtleBot3. The image builds the Robotnik RB-Theron description (Dockerfile.gateway), not TurtleBot3. --- demos/ota_nav2_sensor_fix/run-demo.sh | 12 ++++++------ 1 file changed, 6 insertions(+), 6 deletions(-) diff --git a/demos/ota_nav2_sensor_fix/run-demo.sh b/demos/ota_nav2_sensor_fix/run-demo.sh index 7781801..dc18069 100755 --- a/demos/ota_nav2_sensor_fix/run-demo.sh +++ b/demos/ota_nav2_sensor_fix/run-demo.sh @@ -1,11 +1,11 @@ #!/bin/bash # OTA over SOVD - nav2 sensor-fix demo runner. # Brings up the gateway (with the dev-grade ota_update_plugin baked in) and -# the FastAPI artifact server. The gateway image bundles a full TurtleBot3 + -# Nav2 + headless Gazebo stack and runs foxglove_bridge on :8765, so the -# demo is self-contained: broken_lidar publishes /scan with a phantom -# obstacle that nav2 + a Foxglove 3D panel both react to. The OTA flow -# swaps broken_lidar -> fixed_lidar and the phantom disappears. +# the FastAPI artifact server. The gateway image bundles a full Robotnik +# RB-Theron AMR + Nav2 + headless Gazebo stack and runs foxglove_bridge on +# :8765, so the demo is self-contained: broken_lidar publishes /scan with a +# phantom obstacle that nav2 + a Foxglove 3D panel both react to. The OTA +# flow swaps broken_lidar -> fixed_lidar and the phantom disappears. set -eu @@ -203,6 +203,6 @@ echo " open http://localhost:5173 -> Connect -> ${GATEWAY_URL}" echo "" echo " Foxglove (recommended for the 3D narrative):" echo " Open connection -> Foxglove WebSocket -> ws://localhost:${OTA_FOXGLOVE_BRIDGE_PORT:-8765}" -echo " Add a 3D panel: TurtleBot3 in the world, /scan cone shows the phantom" +echo " Add a 3D panel: RB-Theron in the world, /scan cone shows the phantom" echo " Install ros2_medkit_foxglove_extension (npm run local-install) for the" echo " 'ros2_medkit Updates' panel; set baseUrl to ${GATEWAY_URL}/api/v1" From 8555c5e55f7a18401665958d030150fa8319a2e0 Mon Sep 17 00:00:00 2001 From: Bartosz Burda Date: Mon, 28 Sep 2026 15:44:02 +0200 Subject: [PATCH 04/53] fix(demos/moveit): report real goal status and drop null fields move-arm.sh always printed a success line and exited 0, even when the controller aborted the goal (the pick-and-place loop competes for the same action). It now parses the action's own final status, exits non-zero and prints a failure line when the goal did not succeed, and ./move-arm.sh demo runs every step regardless of earlier failures while still exiting non-zero overall. The local-vs-container branch also checked only that `ros2 node list` succeeded, which is true even on an empty, disconnected graph; it now confirms the target action is actually listed. The container exec no longer requests a TTY, so the script also works from a pipe or CI. check-entities.sh displayed several fields (component area, app category, app location, function category/host, fault code/reporter) that the API never populates for this demo, always printing null. Those columns are dropped or replaced with the real equivalents (app-to-component links via x-medkit, real fault field names) so the explorer only prints real data. README.md documents the new goal-preemption behavior and corrects the manipulation-monitor entity id in the triggers section (the demo scripts already use the hyphenated id; the docs still had the old underscored one). --- demos/moveit_pick_place/README.md | 23 +- demos/moveit_pick_place/check-entities.sh | 8 +- demos/moveit_pick_place/move-arm.sh | 68 ++++-- tests/smoke_test_moveit.sh | 276 +++++++++++++++++++++- 4 files changed, 344 insertions(+), 31 deletions(-) diff --git a/demos/moveit_pick_place/README.md b/demos/moveit_pick_place/README.md index 11fe34e..0df25b1 100644 --- a/demos/moveit_pick_place/README.md +++ b/demos/moveit_pick_place/README.md @@ -73,7 +73,14 @@ Use the interactive arm controller to send joint trajectories: ``` The script sends goals directly to the `panda_arm_controller/follow_joint_trajectory` action. -It works both from outside (via `docker exec`) and from inside the container. +It works both from outside (via `docker exec`, no TTY required) and from inside the container. + +`pick_place_loop.py` keeps sending its own goals to the same controller, so a manual move +can be preempted mid-motion by the demo's own workload. `move-arm.sh` reports the goal's +real final status: if the controller aborts it (`error_string: Current goal preempted by +new incoming action`), the script prints `Failed: (status: ABORTED)` and exits +non-zero instead of claiming success. `./move-arm.sh demo` runs all three steps regardless +of earlier failures and reports each one; the command exits non-zero if any step failed. ### 4. Viewing Logs @@ -308,8 +315,8 @@ The gateway supports condition-based triggers that fire when specific events occ ### How It Works -1. `setup-triggers.sh` creates a trigger via `POST /api/v1/apps/manipulation_monitor/triggers`: - - **Resource:** `/api/v1/apps/manipulation_monitor/faults` (watches fault collection) +1. `setup-triggers.sh` creates a trigger via `POST /api/v1/apps/manipulation-monitor/triggers`: + - **Resource:** `/api/v1/apps/manipulation-monitor/faults` (watches fault collection) - **Condition:** `OnChange` (fires on any new or updated fault) - **Multishot:** `true` (fires repeatedly, not just once) - **Lifetime:** 3600 seconds (auto-expires after 1 hour) @@ -320,23 +327,23 @@ The gateway supports condition-based triggers that fire when specific events occ ```bash # Create a trigger -curl -X POST http://localhost:8080/api/v1/apps/manipulation_monitor/triggers \ +curl -X POST http://localhost:8080/api/v1/apps/manipulation-monitor/triggers \ -H "Content-Type: application/json" \ -d '{ - "resource": "/api/v1/apps/manipulation_monitor/faults", + "resource": "/api/v1/apps/manipulation-monitor/faults", "trigger_condition": {"condition_type": "OnChange"}, "multishot": true, "lifetime": 3600 }' | jq # List triggers -curl http://localhost:8080/api/v1/apps/manipulation_monitor/triggers | jq +curl http://localhost:8080/api/v1/apps/manipulation-monitor/triggers | jq # Watch events (replace TRIGGER_ID) -curl -N http://localhost:8080/api/v1/apps/manipulation_monitor/triggers/TRIGGER_ID/events +curl -N http://localhost:8080/api/v1/apps/manipulation-monitor/triggers/TRIGGER_ID/events # Delete a trigger -curl -X DELETE http://localhost:8080/api/v1/apps/manipulation_monitor/triggers/TRIGGER_ID +curl -X DELETE http://localhost:8080/api/v1/apps/manipulation-monitor/triggers/TRIGGER_ID ``` ## Fault Injection Scenarios diff --git a/demos/moveit_pick_place/check-entities.sh b/demos/moveit_pick_place/check-entities.sh index 3a32ef0..b7e0e5f 100755 --- a/demos/moveit_pick_place/check-entities.sh +++ b/demos/moveit_pick_place/check-entities.sh @@ -40,13 +40,13 @@ echo_step "1. Areas (Functional Groupings)" curl -s "${API_BASE}/areas" | jq '.items[] | {id: .id, name: .name, description: .description}' echo_step "2. Components (Hardware/Logical Units)" -curl -s "${API_BASE}/components" | jq '.items[] | {id: .id, name: .name, type: .type, area: .area}' +curl -s "${API_BASE}/components" | jq '.items[] | {id: .id, name: .name, description: .description}' echo_step "3. Apps (ROS 2 Nodes)" -curl -s "${API_BASE}/apps" | jq '.items[] | {id: .id, name: .name, category: .category, component: .is_located_on}' +curl -s "${API_BASE}/apps" | jq '.items[] | {id: .id, name: .name, description: .description, component: .["x-medkit"].component_id}' echo_step "4. Functions (High-level Capabilities)" -curl -s "${API_BASE}/functions" | jq '.items[] | {id: .id, name: .name, category: .category, hosted_by: .hosted_by}' +curl -s "${API_BASE}/functions" | jq '.items[] | {id: .id, name: .name, description: .description}' echo_step "5. Sample Data (Joint States)" echo "Getting latest joint states from Panda arm..." @@ -57,7 +57,7 @@ curl -s "${API_BASE}/apps/joint-state-broadcaster/data/joint_states" 2>/dev/null }' || echo " (Joint state data not available — robot may still be starting)" echo_step "6. Faults" -curl -s "${API_BASE}/faults" | jq '.items[] | {code: .code, severity: .severity, reporter: .reporter_id}' +curl -s "${API_BASE}/faults" | jq '.items[] | {code: .fault_code, severity: .severity_label, status: .status, sources: .reporting_sources}' echo "" echo -e "${GREEN}✓ Entity hierarchy exploration complete!${NC}" diff --git a/demos/moveit_pick_place/move-arm.sh b/demos/moveit_pick_place/move-arm.sh index 91af056..3fb989d 100755 --- a/demos/moveit_pick_place/move-arm.sh +++ b/demos/moveit_pick_place/move-arm.sh @@ -45,6 +45,15 @@ RIGHT="[1.5, -0.785, 0.0, -2.356, 0.0, 1.571, 0.785]" WAVE="[0.0, -1.0, 0.0, -0.5, 0.0, 2.5, 0.785]" +# True only if a LOCAL ros2 can actually reach the target action server. +# `ros2 node list` exits 0 even on an empty graph (wrong ROS_DOMAIN_ID, no +# multicast route), so a host with ROS 2 sourced but not connected to the +# demo looks identical to being inside the container. Checking that the +# action itself is listed avoids that false positive. +can_reach_action_locally() { + command -v ros2 &> /dev/null && ros2 action list 2> /dev/null | grep -qFx "${ACTION}" +} + send_trajectory() { local positions="$1" local label="$2" @@ -64,27 +73,39 @@ send_trajectory() { } }" - # Check if we're inside the container or outside - if command -v ros2 &> /dev/null && ros2 node list &> /dev/null 2>&1; then - # Inside the container (or ROS 2 env is set up) - ros2 action send_goal "${ACTION}" \ + # `ros2 action send_goal` always exits 0, whatever the goal's outcome - + # the real result is in its own printed "Goal finished with status:" + # line, so capture output and parse that instead of the exit code. + local output + if can_reach_action_locally; then + output=$(ros2 action send_goal "${ACTION}" \ control_msgs/action/FollowJointTrajectory \ "${goal_msg}" \ - --feedback + --feedback 2>&1) || true else - # Outside — exec into container - docker exec -it "${CONTAINER}" bash -c " + # Outside — exec into container. No -it: this must also work + # without a TTY (CI, a pipe), and the command needs no stdin. + output=$(docker exec "${CONTAINER}" bash -c " source /opt/ros/jazzy/setup.bash && \ source /root/demo_ws/install/setup.bash && \ ros2 action send_goal ${ACTION} \ control_msgs/action/FollowJointTrajectory \ \"${goal_msg}\" \ --feedback - " + " 2>&1) || true fi + printf '%s\n' "${output}" + + local status + status=$(printf '%s\n' "${output}" | grep -F 'Goal finished with status:' | tail -n1 | sed -E 's/.*status: *//') echo "" - echo "✅ Done: ${label}" + if [[ "${status}" == "SUCCEEDED" ]]; then + echo "✅ Done: ${label}" + return 0 + fi + echo "Failed: ${label} (status: ${status:-UNKNOWN})" >&2 + return 1 } show_menu() { @@ -110,13 +131,19 @@ show_menu() { run_demo_cycle() { echo "🔄 Running pick → place → home cycle..." echo "" - send_trajectory "${PICK}" "pick" + local failed=0 + send_trajectory "${PICK}" "pick" || failed=1 sleep 2 - send_trajectory "${PLACE}" "place" + send_trajectory "${PLACE}" "place" || failed=1 sleep 2 - send_trajectory "${READY}" "ready (home)" + send_trajectory "${READY}" "ready (home)" || failed=1 echo "" - echo "🔄 Cycle complete!" + if [[ "${failed}" -eq 0 ]]; then + echo "🔄 Cycle complete!" + else + echo "🔄 Cycle complete with failures" >&2 + fi + return "${failed}" } handle_choice() { @@ -138,15 +165,20 @@ handle_choice() { # --- Main --- -# If argument provided, run directly +# If argument provided, run directly. Reflect the goal's real result in the +# exit code instead of always exiting 0. if [[ $# -gt 0 ]]; then - handle_choice "$1" - exit 0 + if handle_choice "$1"; then + exit 0 + else + exit 1 + fi fi -# Interactive mode +# Interactive mode. A failed goal reports failure and the menu continues - +# it must not kill the session (set -e would, without this guard). while true; do show_menu read -rp "Choose position (1-8, d, q): " choice - handle_choice "${choice}" + handle_choice "${choice}" || true done diff --git a/tests/smoke_test_moveit.sh b/tests/smoke_test_moveit.sh index 30533e4..50e969c 100755 --- a/tests/smoke_test_moveit.sh +++ b/tests/smoke_test_moveit.sh @@ -18,7 +18,34 @@ SCRIPT_DIR="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)" # shellcheck source=tests/smoke_lib.sh source "${SCRIPT_DIR}/smoke_lib.sh" -trap print_summary EXIT +DEMO_DIR="$(cd "${SCRIPT_DIR}/../demos/moveit_pick_place" && pwd)" +DEMO_CONTAINER="${MOVEIT_DEMO_CONTAINER:-moveit_medkit_demo_ci}" +ACTION="/panda_arm_controller/follow_joint_trajectory" +PICK_PLACE_PATTERN='^python3 /root/demo_ws/install/moveit_medkit_demo/lib/moveit_medkit_demo/pick_place_loop\.py' + +# pick_place_loop sends goals to the same controller action move-arm.sh +# drives directly, so a live loop makes goal outcomes below racy against the +# demo's own workload. Pausing the OS process (no file or behavior change) +# is how the tests below get a deterministic outcome to assert on. +stop_pick_place_loop() { + docker exec "${DEMO_CONTAINER}" bash -c " + pid=\$(pgrep -f '${PICK_PLACE_PATTERN}') || exit 0 + kill -STOP \"\${pid}\" + " > /dev/null 2>&1 || true +} + +resume_pick_place_loop() { + docker exec "${DEMO_CONTAINER}" bash -c " + pid=\$(pgrep -f '${PICK_PLACE_PATTERN}') || exit 0 + kill -CONT \"\${pid}\" + " > /dev/null 2>&1 || true +} + +cleanup_on_exit() { + resume_pick_place_loop + print_summary +} +trap cleanup_on_exit EXIT # --- Wait for gateway startup --- @@ -97,6 +124,253 @@ section "Triggers" assert_triggers_crud "apps" "diagnostic-bridge-app" "/api/v1/apps/diagnostic-bridge-app/faults" +section "check-entities.sh: no null fields, including under an active fault" + +# Force a real fault: send two goals to the same controller back to back so +# the second preempts the first, a real ABORTED goal that +# manipulation_monitor turns into TRAJECTORY_EXECUTION_FAILED / +# CONTROLLER_TIMEOUT. One rclpy action client (not two `ros2` CLI +# invocations) removes per-invocation DDS discovery jitter, so which goal +# gets preempted is deterministic. +echo " Forcing a real active fault via controller goal preemption..." +stop_pick_place_loop +docker exec -i "${DEMO_CONTAINER}" bash -s > /tmp/moveit_smoke_fault_setup.log 2>&1 <<'REMOTE' || true +set -eu +set +u +source /opt/ros/jazzy/setup.bash +source /root/demo_ws/install/setup.bash +set -u +python3 - <<'PYEOF' +import rclpy +from rclpy.node import Node +from rclpy.action import ActionClient +from control_msgs.action import FollowJointTrajectory +from trajectory_msgs.msg import JointTrajectoryPoint +from builtin_interfaces.msg import Duration + +JOINTS = [ + "panda_joint1", "panda_joint2", "panda_joint3", "panda_joint4", + "panda_joint5", "panda_joint6", "panda_joint7", +] + + +def make_goal(positions, sec): + goal = FollowJointTrajectory.Goal() + goal.trajectory.joint_names = JOINTS + point = JointTrajectoryPoint() + point.positions = positions + point.time_from_start = Duration(sec=sec, nanosec=0) + goal.trajectory.points = [point] + return goal + + +rclpy.init() +node = Node("smoke_fault_probe") +client = ActionClient( + node, FollowJointTrajectory, "/panda_arm_controller/follow_joint_trajectory" +) +client.wait_for_server() + +ready = [0.0, -0.785, 0.0, -2.356, 0.0, 1.571, 0.785] +future_a = client.send_goal_async(make_goal(ready, 3)) +rclpy.spin_until_future_complete(node, future_a) +handle_a = future_a.result() + +future_b = client.send_goal_async(make_goal(ready, 1)) +rclpy.spin_until_future_complete(node, future_b) +handle_b = future_b.result() + +rclpy.spin_until_future_complete(node, handle_a.get_result_async()) +rclpy.spin_until_future_complete(node, handle_b.get_result_async()) + +node.destroy_node() +rclpy.shutdown() +PYEOF +REMOTE +resume_pick_place_loop + +echo " Waiting for the fault to appear (max 20s)..." +if poll_until "/faults" '.items | length > 0' 20; then + pass "setup: a real active fault exists" +else + fail "setup: a real active fault exists" "no fault after forced controller preemption" +fi + +ENTITIES_OUTPUT=$(GATEWAY_URL="${GATEWAY_URL}" "${DEMO_DIR}/check-entities.sh" 2>&1) || true +if ! printf '%s\n' "${ENTITIES_OUTPUT}" | grep -q 'exploration complete'; then + fail "check-entities.sh runs to completion" \ + "$(printf '%s\n' "${ENTITIES_OUTPUT}" | tail -5)" +elif printf '%s\n' "${ENTITIES_OUTPUT}" | grep -q '": null'; then + fail "check-entities.sh prints no null fields" \ + "$(printf '%s\n' "${ENTITIES_OUTPUT}" | grep '": null' | sort -u | tr '\n' ';')" +else + pass "check-entities.sh prints no null fields" +fi + +section "move-arm.sh: a local ros2 that cannot reach the demo still moves the arm" + +# CI runners (and this dev container) have no ROS 2 wired to the demo's +# graph, but `ros2 node list` still exits 0 on an empty graph. A fake ros2 +# that only answers `node list` reproduces that false positive without +# needing a real, disconnected ROS 2 install. +FAKE_ROS2_DIR=$(mktemp -d) +cat > "${FAKE_ROS2_DIR}/ros2" <<'FAKE' +#!/bin/sh +if [ "$1" = "node" ] && [ "$2" = "list" ]; then + exit 0 +fi +exit 1 +FAKE +chmod +x "${FAKE_ROS2_DIR}/ros2" + +stop_pick_place_loop +if MOVE_ARM_OUTPUT=$(PATH="${FAKE_ROS2_DIR}:${PATH}" CONTAINER_NAME="${DEMO_CONTAINER}" \ + "${DEMO_DIR}/move-arm.sh" extended < /dev/null 2>&1); then + MOVE_ARM_RC=0 +else + MOVE_ARM_RC=$? +fi +resume_pick_place_loop +rm -rf "${FAKE_ROS2_DIR}" + +if [ "${MOVE_ARM_RC}" -eq 0 ] && printf '%s\n' "${MOVE_ARM_OUTPUT}" | grep -q 'Goal finished with status: SUCCEEDED'; then + pass "move-arm.sh moves the container's arm despite a local unreachable ros2" +else + fail "move-arm.sh moves the container's arm despite a local unreachable ros2" \ + "rc=${MOVE_ARM_RC}; tail: $(printf '%s\n' "${MOVE_ARM_OUTPUT}" | tail -5)" +fi + +if printf '%s\n' "${MOVE_ARM_OUTPUT}" | grep -q 'cannot attach stdin'; then + fail "move-arm.sh works without a TTY" "docker exec still requires a TTY" +else + pass "move-arm.sh works without a TTY" +fi + +section "move-arm.sh: reports the goal's real final status" + +# Launch two goals for the same joints at (as close as the shell gets to) +# the same instant. The controller always aborts whichever one it already +# had in flight when the second arrives, so exactly one of the two +# invocations below is guaranteed to see a real ABORTED result and the +# other a real SUCCEEDED one - which one is not predictable, so both are +# checked post hoc. +stop_pick_place_loop +# set +e inside each subshell: it inherits the outer `set -e`, and without +# this a non-zero move-arm.sh exit (exactly what these subshells expect to +# sometimes see) would abort the subshell before it writes its .rc file. +( + set +e + CONTAINER_NAME="${DEMO_CONTAINER}" "${DEMO_DIR}/move-arm.sh" ready < /dev/null \ + > /tmp/moveit_smoke_goal_a.log 2>&1 + echo "$?" > /tmp/moveit_smoke_goal_a.rc +) & +GOAL_A_PID=$! +( + set +e + CONTAINER_NAME="${DEMO_CONTAINER}" "${DEMO_DIR}/move-arm.sh" place < /dev/null \ + > /tmp/moveit_smoke_goal_b.log 2>&1 + echo "$?" > /tmp/moveit_smoke_goal_b.rc +) & +GOAL_B_PID=$! +wait "${GOAL_A_PID}" "${GOAL_B_PID}" +resume_pick_place_loop + +GOAL_A_RC=$(cat /tmp/moveit_smoke_goal_a.rc) +GOAL_B_RC=$(cat /tmp/moveit_smoke_goal_b.rc) +GOAL_A_OUT=$(cat /tmp/moveit_smoke_goal_a.log) +GOAL_B_OUT=$(cat /tmp/moveit_smoke_goal_b.log) + +if { [ "${GOAL_A_RC}" -eq 0 ] && [ "${GOAL_B_RC}" -ne 0 ]; } || \ + { [ "${GOAL_A_RC}" -ne 0 ] && [ "${GOAL_B_RC}" -eq 0 ]; }; then + pass "concurrent goals: one real ABORTED and one real SUCCEEDED occurred" +else + fail "concurrent goals: one real ABORTED and one real SUCCEEDED occurred" \ + "goal A rc=${GOAL_A_RC}, goal B rc=${GOAL_B_RC}" +fi + +# check_goal_report