From dfb8c64845004d07c2964ee7dc14fa9a3d5f8e71 Mon Sep 17 00:00:00 2001 From: Siyao Meng <50227127+smengcl@users.noreply.github.com> Date: Wed, 12 Aug 2026 22:27:31 -0700 Subject: [PATCH 1/4] HDDS-11128. Increase timeout for decommission-triggered replica-count waits in TestReconAndAdminContainerCLI testNodesInDecommissionOrMaintenance intermittently times out at OzoneTestHelper.waitForReplicaCount while waiting for a decommission- triggered replica copy (3 -> 4 for the first node, 4 -> 5 for the second) to be reflected in SCM. The shared waitForReplicaCount helper used a fixed 30s budget for all 14 callers; HDDS-10582 only reduced its poll interval (1000ms -> 200ms) and kept the 30s total, so under a loaded CI runner the replica copy is not always observed in time and the test flakes. Add a 4-arg waitForReplicaCount overload that accepts a timeout, leaving the existing 3-arg method delegating with the same 30s default (no behavior change for the other callers). The decommission/maintenance replica-copy waits in TestReconAndAdminContainerCLI now use a 120s budget, matching the larger timeouts adopted elsewhere for the same class of replication waits. Co-Authored-By: Claude Opus 4.8 (1M context) --- .../ozone/recon/TestReconAndAdminContainerCLI.java | 10 ++++++++-- .../apache/hadoop/ozone/container/OzoneTestHelper.java | 8 ++++++-- 2 files changed, 14 insertions(+), 4 deletions(-) diff --git a/hadoop-ozone/integration-test-recon/src/test/java/org/apache/hadoop/ozone/recon/TestReconAndAdminContainerCLI.java b/hadoop-ozone/integration-test-recon/src/test/java/org/apache/hadoop/ozone/recon/TestReconAndAdminContainerCLI.java index 89ce3172c917..125d26c9ee93 100644 --- a/hadoop-ozone/integration-test-recon/src/test/java/org/apache/hadoop/ozone/recon/TestReconAndAdminContainerCLI.java +++ b/hadoop-ozone/integration-test-recon/src/test/java/org/apache/hadoop/ozone/recon/TestReconAndAdminContainerCLI.java @@ -114,6 +114,12 @@ class TestReconAndAdminContainerCLI { * are still drifting past each other (HDDS-15223). */ private static final int RM_RECON_COMPARE_STABLE_POLLS = 2; + /** + * Max wait for a decommission/maintenance triggered replica copy to be reflected in SCM. The + * default {@link OzoneTestHelper#waitForReplicaCount} budget (30s) is not always enough on a + * loaded CI runner, which is the recurring flake tracked in HDDS-11128 (see also HDDS-10582). + */ + private static final int REPLICA_COPY_WAIT_MS = 120_000; private static final OzoneConfiguration CONF = new OzoneConfiguration(); private static ScmClient scmClient; @@ -272,7 +278,7 @@ void testNodesInDecommissionOrMaintenance( // a new replica-copy is made to another node. // For maintenance, there is no replica-copy in this case. if (!isMaintenance) { - OzoneTestHelper.waitForReplicaCount(containerIdR3, 4, cluster); + OzoneTestHelper.waitForReplicaCount(containerIdR3, 4, cluster, REPLICA_COPY_WAIT_MS); } compareRMReportToReconResponse(underReplicatedState); @@ -299,7 +305,7 @@ void testNodesInDecommissionOrMaintenance( // There will be a replica copy for both maintenance and decommission. // maintenance 3 -> 4, decommission 4 -> 5. int expectedReplicaNum = isMaintenance ? 4 : 5; - OzoneTestHelper.waitForReplicaCount(containerIdR3, expectedReplicaNum, cluster); + OzoneTestHelper.waitForReplicaCount(containerIdR3, expectedReplicaNum, cluster, REPLICA_COPY_WAIT_MS); compareRMReportToReconResponse(underReplicatedState); compareRMReportToReconResponse(overReplicatedState); diff --git a/hadoop-ozone/integration-test/src/test/java/org/apache/hadoop/ozone/container/OzoneTestHelper.java b/hadoop-ozone/integration-test/src/test/java/org/apache/hadoop/ozone/container/OzoneTestHelper.java index 9558be282560..94b0bc9c39cc 100644 --- a/hadoop-ozone/integration-test/src/test/java/org/apache/hadoop/ozone/container/OzoneTestHelper.java +++ b/hadoop-ozone/integration-test/src/test/java/org/apache/hadoop/ozone/container/OzoneTestHelper.java @@ -461,8 +461,12 @@ public static int countReplicas(long containerID, MiniOzoneCluster cluster) { public static void waitForReplicaCount(long containerID, int count, MiniOzoneCluster cluster) throws TimeoutException, InterruptedException { - GenericTestUtils.waitFor(() -> countReplicas(containerID, cluster) == count, - 200, 30000); + waitForReplicaCount(containerID, count, cluster, 30000); + } + + public static void waitForReplicaCount(long containerID, int count, MiniOzoneCluster cluster, int timeoutMillis) + throws TimeoutException, InterruptedException { + GenericTestUtils.waitFor(() -> countReplicas(containerID, cluster) == count, 200, timeoutMillis); } /** From 4c72d98659c12aa9f583d49d74e21452799e43b2 Mon Sep 17 00:00:00 2001 From: Siyao Meng <50227127+smengcl@users.noreply.github.com> Date: Wed, 12 Aug 2026 22:58:24 -0700 Subject: [PATCH 2/4] HDDS-11128. Reduce replica-count wait to 60s and disambiguate waitForReplicaCount Javadoc link The happy path returns as soon as the copy lands, so the timeout only matters on a genuine failure; 120s was more than needed and, because this @Flaky test is rerun on failure in the flaky split, an oversized budget multiplies wasted CI time on a real break. 60s (double the previous 30s) gives comfortable headroom for the tail latency while keeping the failing path bounded. Also qualify the OzoneTestHelper#waitForReplicaCount(long, int, MiniOzoneCluster) Javadoc link, which became ambiguous once the timeout overload was added (flagged by review, avoids doclint resolution warnings). Co-Authored-By: Claude Opus 4.8 (1M context) --- .../hadoop/ozone/recon/TestReconAndAdminContainerCLI.java | 8 +++++--- 1 file changed, 5 insertions(+), 3 deletions(-) diff --git a/hadoop-ozone/integration-test-recon/src/test/java/org/apache/hadoop/ozone/recon/TestReconAndAdminContainerCLI.java b/hadoop-ozone/integration-test-recon/src/test/java/org/apache/hadoop/ozone/recon/TestReconAndAdminContainerCLI.java index 125d26c9ee93..3daeea4b9fcd 100644 --- a/hadoop-ozone/integration-test-recon/src/test/java/org/apache/hadoop/ozone/recon/TestReconAndAdminContainerCLI.java +++ b/hadoop-ozone/integration-test-recon/src/test/java/org/apache/hadoop/ozone/recon/TestReconAndAdminContainerCLI.java @@ -116,10 +116,12 @@ class TestReconAndAdminContainerCLI { private static final int RM_RECON_COMPARE_STABLE_POLLS = 2; /** * Max wait for a decommission/maintenance triggered replica copy to be reflected in SCM. The - * default {@link OzoneTestHelper#waitForReplicaCount} budget (30s) is not always enough on a - * loaded CI runner, which is the recurring flake tracked in HDDS-11128 (see also HDDS-10582). + * default {@link OzoneTestHelper#waitForReplicaCount(long, int, MiniOzoneCluster)} budget (30s) + * is occasionally not enough on a loaded CI runner (the recurring flake tracked in HDDS-11128, + * see also HDDS-10582), so give these waits double the headroom. The happy path returns as soon + * as the copy lands. */ - private static final int REPLICA_COPY_WAIT_MS = 120_000; + private static final int REPLICA_COPY_WAIT_MS = 60_000; private static final OzoneConfiguration CONF = new OzoneConfiguration(); private static ScmClient scmClient; From fec80c4ef46ac9182ac36c35658b51a51138a1a3 Mon Sep 17 00:00:00 2001 From: Siyao Meng <50227127+smengcl@users.noreply.github.com> Date: Thu, 13 Aug 2026 00:11:59 -0700 Subject: [PATCH 3/4] HDDS-11128. Wait for replication to settle instead of a transient replica count testNodesInDecommissionOrMaintenance waited on a bare replica-count equality (countReplicas(...) == N) after each decommission/maintenance step. During decommission SCM is actively adding (and re-evaluating) replicas, so the count passes through the expected value transiently; a fixed-value poll can sample the wrong instant and time out. HDDS-10582 only shrank the poll interval (1000ms -> 200ms) to narrow that window, so the flake recurred (HDDS-11128). Add OzoneTestHelper.waitForStableReplicaCount, which returns only once replication has quiesced (ReplicationManager has no pending add/delete ops for the container) AND the count equals N, so the assertion is on a settled state rather than a transient one. It uses the same 30s budget as waitForReplicaCount; the elevated timeout is no longer needed because the call runs right after the DECOMMISSIONED/IN_MAINTENANCE gate, which already requires the new replica to exist, so the settle returns almost immediately. The existing waitForReplicaCount is left untouched for its other callers. The test remains @Flaky("HDDS-11128") because no wait can prove non-flakiness. Co-Authored-By: Claude Opus 4.8 (1M context) --- .../recon/TestReconAndAdminContainerCLI.java | 12 ++---------- .../ozone/container/OzoneTestHelper.java | 19 ++++++++++++++++--- 2 files changed, 18 insertions(+), 13 deletions(-) diff --git a/hadoop-ozone/integration-test-recon/src/test/java/org/apache/hadoop/ozone/recon/TestReconAndAdminContainerCLI.java b/hadoop-ozone/integration-test-recon/src/test/java/org/apache/hadoop/ozone/recon/TestReconAndAdminContainerCLI.java index 3daeea4b9fcd..911f5c277603 100644 --- a/hadoop-ozone/integration-test-recon/src/test/java/org/apache/hadoop/ozone/recon/TestReconAndAdminContainerCLI.java +++ b/hadoop-ozone/integration-test-recon/src/test/java/org/apache/hadoop/ozone/recon/TestReconAndAdminContainerCLI.java @@ -114,14 +114,6 @@ class TestReconAndAdminContainerCLI { * are still drifting past each other (HDDS-15223). */ private static final int RM_RECON_COMPARE_STABLE_POLLS = 2; - /** - * Max wait for a decommission/maintenance triggered replica copy to be reflected in SCM. The - * default {@link OzoneTestHelper#waitForReplicaCount(long, int, MiniOzoneCluster)} budget (30s) - * is occasionally not enough on a loaded CI runner (the recurring flake tracked in HDDS-11128, - * see also HDDS-10582), so give these waits double the headroom. The happy path returns as soon - * as the copy lands. - */ - private static final int REPLICA_COPY_WAIT_MS = 60_000; private static final OzoneConfiguration CONF = new OzoneConfiguration(); private static ScmClient scmClient; @@ -280,7 +272,7 @@ void testNodesInDecommissionOrMaintenance( // a new replica-copy is made to another node. // For maintenance, there is no replica-copy in this case. if (!isMaintenance) { - OzoneTestHelper.waitForReplicaCount(containerIdR3, 4, cluster, REPLICA_COPY_WAIT_MS); + OzoneTestHelper.waitForStableReplicaCount(containerIdR3, 4, cluster); } compareRMReportToReconResponse(underReplicatedState); @@ -307,7 +299,7 @@ void testNodesInDecommissionOrMaintenance( // There will be a replica copy for both maintenance and decommission. // maintenance 3 -> 4, decommission 4 -> 5. int expectedReplicaNum = isMaintenance ? 4 : 5; - OzoneTestHelper.waitForReplicaCount(containerIdR3, expectedReplicaNum, cluster, REPLICA_COPY_WAIT_MS); + OzoneTestHelper.waitForStableReplicaCount(containerIdR3, expectedReplicaNum, cluster); compareRMReportToReconResponse(underReplicatedState); compareRMReportToReconResponse(overReplicatedState); diff --git a/hadoop-ozone/integration-test/src/test/java/org/apache/hadoop/ozone/container/OzoneTestHelper.java b/hadoop-ozone/integration-test/src/test/java/org/apache/hadoop/ozone/container/OzoneTestHelper.java index 94b0bc9c39cc..122c9d19bd0d 100644 --- a/hadoop-ozone/integration-test/src/test/java/org/apache/hadoop/ozone/container/OzoneTestHelper.java +++ b/hadoop-ozone/integration-test/src/test/java/org/apache/hadoop/ozone/container/OzoneTestHelper.java @@ -51,6 +51,7 @@ import org.apache.hadoop.hdds.scm.container.ContainerManager; import org.apache.hadoop.hdds.scm.container.ContainerNotFoundException; import org.apache.hadoop.hdds.scm.container.ContainerReplica; +import org.apache.hadoop.hdds.scm.container.replication.ReplicationManager; import org.apache.hadoop.hdds.scm.events.SCMEvents; import org.apache.hadoop.hdds.scm.pipeline.Pipeline; import org.apache.hadoop.hdds.scm.pipeline.PipelineNotFoundException; @@ -461,12 +462,24 @@ public static int countReplicas(long containerID, MiniOzoneCluster cluster) { public static void waitForReplicaCount(long containerID, int count, MiniOzoneCluster cluster) throws TimeoutException, InterruptedException { - waitForReplicaCount(containerID, count, cluster, 30000); + GenericTestUtils.waitFor(() -> countReplicas(containerID, cluster) == count, + 200, 30000); } - public static void waitForReplicaCount(long containerID, int count, MiniOzoneCluster cluster, int timeoutMillis) + /** + * Like {@link #waitForReplicaCount(long, int, MiniOzoneCluster)}, but only returns once + * replication has quiesced: no pending add or delete ops remain for the container, so the count + * has settled at {@code count} instead of being observed transiently while SCM is still adding or + * removing replicas. This removes the race where the count passes through {@code count} between + * polls under a loaded runner (HDDS-10582, HDDS-11128). + */ + public static void waitForStableReplicaCount(long containerID, int count, MiniOzoneCluster cluster) throws TimeoutException, InterruptedException { - GenericTestUtils.waitFor(() -> countReplicas(containerID, cluster) == count, 200, timeoutMillis); + ReplicationManager replicationManager = cluster.getStorageContainerManager().getReplicationManager(); + ContainerID cid = ContainerID.valueOf(containerID); + GenericTestUtils.waitFor(() -> + replicationManager.getPendingReplicationOps(cid).isEmpty() && countReplicas(containerID, cluster) == count, + 200, 30000); } /** From c2f62866c6d14577c34bbc25d8f209b51b776773 Mon Sep 17 00:00:00 2001 From: Siyao Meng <50227127+smengcl@users.noreply.github.com> Date: Thu, 20 Aug 2026 16:02:36 -0700 Subject: [PATCH 4/4] HDDS-11128. Isolate maintenance and decommission test invocations The maintenance invocation can return with an excess replica before SCM finishes over replication cleanup. The following decommission invocation can select that replica, then cleanup deletes it and decommission completes without creating another copy. The test then waits for a replica count that SCM does not need to reach. Wait for a healthy three replica baseline before and after each invocation, and select nodes from the current SCM replica set instead of the original pipeline. Also require current healthy replication state in the stable replica count helper so an empty pending operation list alone does not establish quiescence. --- .../recon/TestReconAndAdminContainerCLI.java | 9 ++++--- .../ozone/container/OzoneTestHelper.java | 27 +++++++++++++------ 2 files changed, 25 insertions(+), 11 deletions(-) diff --git a/hadoop-ozone/integration-test-recon/src/test/java/org/apache/hadoop/ozone/recon/TestReconAndAdminContainerCLI.java b/hadoop-ozone/integration-test-recon/src/test/java/org/apache/hadoop/ozone/recon/TestReconAndAdminContainerCLI.java index 911f5c277603..714703a5da6f 100644 --- a/hadoop-ozone/integration-test-recon/src/test/java/org/apache/hadoop/ozone/recon/TestReconAndAdminContainerCLI.java +++ b/hadoop-ozone/integration-test-recon/src/test/java/org/apache/hadoop/ozone/recon/TestReconAndAdminContainerCLI.java @@ -60,6 +60,7 @@ import org.apache.hadoop.hdds.scm.container.ContainerHealthState; import org.apache.hadoop.hdds.scm.container.ContainerID; import org.apache.hadoop.hdds.scm.container.ContainerManager; +import org.apache.hadoop.hdds.scm.container.ContainerReplica; import org.apache.hadoop.hdds.scm.container.ReplicationManagerReport; import org.apache.hadoop.hdds.scm.container.replication.ReplicationManager; import org.apache.hadoop.hdds.scm.node.NodeManager; @@ -235,11 +236,11 @@ void testMissingContainer() throws Exception { void testNodesInDecommissionOrMaintenance( NodeOperationalState initialState, NodeOperationalState finalState, boolean isMaintenance) throws Exception { - Pipeline pipeline = - scmClient.getContainerWithPipeline(containerIdR3).getPipeline(); + OzoneTestHelper.waitForStableReplicaCount(containerIdR3, 3, cluster); List details = - pipeline.getNodes().stream() + scmContainerManager.getContainerReplicas(ContainerID.valueOf(containerIdR3)).stream() + .map(ContainerReplica::getDatanodeDetails) .filter(d -> d.getPersistedOpState().equals(IN_SERVICE)) .collect(Collectors.toList()); @@ -316,6 +317,8 @@ void testNodesInDecommissionOrMaintenance( NodeTestUtil.waitForDnToReachPersistedOpState(nodeToGoOffline1, IN_SERVICE); NodeTestUtil.waitForDnToReachPersistedOpState(nodeToGoOffline2, IN_SERVICE); + OzoneTestHelper.waitForStableReplicaCount(containerIdR3, 3, cluster); + compareRMReportToReconResponse(underReplicatedState); compareRMReportToReconResponse(overReplicatedState); } diff --git a/hadoop-ozone/integration-test/src/test/java/org/apache/hadoop/ozone/container/OzoneTestHelper.java b/hadoop-ozone/integration-test/src/test/java/org/apache/hadoop/ozone/container/OzoneTestHelper.java index 122c9d19bd0d..198d02106a3d 100644 --- a/hadoop-ozone/integration-test/src/test/java/org/apache/hadoop/ozone/container/OzoneTestHelper.java +++ b/hadoop-ozone/integration-test/src/test/java/org/apache/hadoop/ozone/container/OzoneTestHelper.java @@ -51,6 +51,7 @@ import org.apache.hadoop.hdds.scm.container.ContainerManager; import org.apache.hadoop.hdds.scm.container.ContainerNotFoundException; import org.apache.hadoop.hdds.scm.container.ContainerReplica; +import org.apache.hadoop.hdds.scm.container.replication.ContainerHealthResult; import org.apache.hadoop.hdds.scm.container.replication.ReplicationManager; import org.apache.hadoop.hdds.scm.events.SCMEvents; import org.apache.hadoop.hdds.scm.pipeline.Pipeline; @@ -468,18 +469,28 @@ public static void waitForReplicaCount(long containerID, int count, /** * Like {@link #waitForReplicaCount(long, int, MiniOzoneCluster)}, but only returns once - * replication has quiesced: no pending add or delete ops remain for the container, so the count - * has settled at {@code count} instead of being observed transiently while SCM is still adding or - * removing replicas. This removes the race where the count passes through {@code count} between - * polls under a loaded runner (HDDS-10582, HDDS-11128). + * replication has quiesced: the container is healthy with {@code count} replicas and no pending + * add or delete ops. Checking the current replication health avoids treating an empty pending-op + * list as settled before ReplicationManager has evaluated the latest replica or node state. */ public static void waitForStableReplicaCount(long containerID, int count, MiniOzoneCluster cluster) throws TimeoutException, InterruptedException { - ReplicationManager replicationManager = cluster.getStorageContainerManager().getReplicationManager(); + ContainerManager containerManager = cluster.getStorageContainerManager().getContainerManager(); + ReplicationManager replicationManager = cluster.getStorageContainerManager() + .getReplicationManager(); ContainerID cid = ContainerID.valueOf(containerID); - GenericTestUtils.waitFor(() -> - replicationManager.getPendingReplicationOps(cid).isEmpty() && countReplicas(containerID, cluster) == count, - 200, 30000); + GenericTestUtils.waitFor(() -> { + try { + ContainerInfo container = containerManager.getContainer(cid); + Set replicas = containerManager.getContainerReplicas(cid); + return replicas.size() == count + && replicationManager.getPendingReplicationOps(cid).isEmpty() + && replicationManager.getContainerReplicationHealth(container, replicas).getHealthState() + == ContainerHealthResult.HealthState.HEALTHY; + } catch (ContainerNotFoundException e) { + return false; + } + }, 200, 30000); } /**