]> git-server-git.apps.pok.os.sepia.ceph.com Git - ceph.git/commitdiff
nvmeofgw: stretched cluster fix relocate 69490/head
authorLeonid Chernin <leonidc@il.ibm.com>
Sun, 7 Jun 2026 05:26:38 +0000 (08:26 +0300)
committerLeonid Chernin <leonidc@il.ibm.com>
Tue, 30 Jun 2026 07:05:28 +0000 (10:05 +0300)
fixes: https://tracker.ceph.com/issues/77803

Signed-off-by: Leonid Chernin <leonidc@il.ibm.com>
src/mon/NVMeofGwMap.cc
src/mon/NVMeofGwMap.h

index b72dc3e898b5c0c4ca4b10fd7fa7340e199a65c1..494983b5a4e17eeafb1b066e1a8c4bd66b8fc0bf 100644 (file)
@@ -818,7 +818,9 @@ void NVMeofGwMap::handle_abandoned_ana_groups(bool& propose)
 
 void NVMeofGwMap::check_relocate_ana_groups(const NvmeGroupKey& group_key,
          bool &propose) {
-  /* if location in disaster_locations found in recovering state state - find all gws in location.
+  /* loop for all locations:
+   * if location in normal state or "disaster_locations"
+   * found in recovering state state - find all gws in location.
    * add ana-grp of not Available gws to the list.
    * if ana-grp is already active on some gw in location skip it
    * for ana-grp in list make relocation.
@@ -830,48 +832,57 @@ void NVMeofGwMap::check_relocate_ana_groups(const NvmeGroupKey& group_key,
        return ;
   }
   std::list<NvmeAnaGrpId>  reloc_list;
+  std::unordered_set<NvmeLocation> locations_set;
   auto& gws_states = created_gws[group_key];
-  FailbackLocation location;
-  if (get_location_in_disaster_cleanup(group_key, location)) {
-    uint32_t num_gw_in_location = 0;
-    uint32_t num_active_ana_in_location = 0;
-    for (auto& gw_state : gws_states) { // loop for GWs inside group-key
-      NvmeGwMonState& state = gw_state.second;
-      if (state.location == location) {
-        num_gw_in_location ++;
-        if (state.availability != gw_availability_t::GW_AVAILABLE) {
-          reloc_list.push_back(state.ana_grp_id);
-        } else { // in parallel check condition to complete failback-in-process
-          for (auto& state_it: state.sm_state) {
-            if (state_it.second == gw_states_per_group_t::GW_ACTIVE_STATE) {
-              num_active_ana_in_location ++;
+  for (auto& gw_state : gws_states) {// build locations set
+    locations_set.insert(gw_state.second.location);
+  }
+  for (NvmeLocation location : locations_set) {
+    bool cleanup_in_process;
+    reloc_list.clear();
+    bool disaster =  is_location_in_disaster(group_key, location, cleanup_in_process);
+    if ((disaster && cleanup_in_process) || (!disaster)) {
+      uint32_t num_gw_in_location = 0;
+      uint32_t num_active_ana_in_location = 0;
+      for (auto& gw_state : gws_states) { // loop for GWs inside group-key
+        NvmeGwMonState& state = gw_state.second;
+        if (state.location == location) {
+          num_gw_in_location++;
+          if (state.availability != gw_availability_t::GW_AVAILABLE) {
+            reloc_list.push_back(state.ana_grp_id);
+          } else { // in parallel check condition to complete failback-in-process
+            for (auto& state_it: state.sm_state) {
+              if (state_it.second == gw_states_per_group_t::GW_ACTIVE_STATE) {
+                num_active_ana_in_location ++;
+              }
             }
           }
         }
       }
-    }
-    if (num_gw_in_location == num_active_ana_in_location) {// All ana groups of disaster location are in Active
-      disaster_map_remove_location(group_key, location);
-      dout(4) <<  "the location entry is erased "<< location
-          << " from disaster-locations num_ana_groups in location "
-          << num_gw_in_location
-          << " from the failbacks-in-progress of group " << group_key <<dendl;
-      propose = true;
-      return;
-    }
+      if (num_gw_in_location == num_active_ana_in_location) {// All ana groups of disaster location are in Active
+        disaster_map_remove_location(group_key, location);
+        dout(4) <<  "the location entry is erased "<< location
+            << " from disaster-locations num_ana_groups in location "
+            << num_gw_in_location
+            << " from the failbacks-in-progress of group " << group_key <<dendl;
+        propose = true;
+        return;
+      }
     // for all ana groups in the list do relocate
-    for (auto& anagrp : reloc_list) {
-      for (auto& gw_state : gws_states) { // loop for GWs inside group-key
-        NvmeGwMonState& state = gw_state.second;
-        if (state.sm_state[anagrp] == gw_states_per_group_t::GW_ACTIVE_STATE) {
-          if (state.location == location) { // already relocated to the location
-            dout(10) << "ana " << anagrp << " already in " << location << dendl;
-            break;
-          } else { // try to relocate
-              dout(10) << "ana " << anagrp
-                  << " to relocate to " << location << dendl;
-              relocate_ana_grp(gw_state.first, group_key, anagrp,
-                    location, propose);
+      for (auto& anagrp : reloc_list) {
+        for (auto& gw_state : gws_states) { // loop for GWs inside group-key
+          NvmeGwMonState& state = gw_state.second;
+          if (state.sm_state[anagrp] == gw_states_per_group_t::GW_ACTIVE_STATE) {
+            if (state.location == location) { // already relocated to the location
+              dout(10) << "ana " << anagrp << " already in " << location << dendl;
+              break;
+            } else { // try to relocate
+                dout(10) << "ana " << anagrp
+                    << " relocate to " << location << dendl;
+                relocate_ana_grp(gw_state.first, group_key, anagrp,
+                      location, propose);
+                return; // allow just 1 relocation during a tick()
+            }
           }
         }
       }
@@ -926,14 +937,8 @@ int NVMeofGwMap::relocate_ana_grp(const NvmeGwId &src_gw_id,
       << location << "min load " << min_num_ana_groups_in_gw << dendl;
 
   if (min_num_ana_groups_in_gw <  MAX_NUM_ANA_GROUPS_FOR_RELOCATE) {
-    dout(4) << "relocate starts " << grpid << " location " << location << dendl;
-    gws_states[src_gw_id].sm_state[grpid] =
-       gw_states_per_group_t::GW_WAIT_FAILBACK_PREPARED;
-    // Add timestamp of start Failback preparation
-    start_timer(src_gw_id, group_key, grpid, 3);
-    gws_states[min_gw_id].sm_state[grpid] =
-       gw_states_per_group_t::GW_OWNER_WAIT_FAILBACK_PREPARED;
-    propose = true;
+    dout(4) << "relocate starts ANA group " << grpid << " location " << location << dendl;
+    fsm_handle_failback_and_relocation(min_gw_id, src_gw_id, group_key, grpid, propose);
   }
   return 0;
 }
@@ -1031,14 +1036,8 @@ void NVMeofGwMap::find_failback_gw(
         << gw_id << dendl;
         return;
       }
-      st.sm_state[gw_state.ana_grp_id] =
-       gw_states_per_group_t::GW_WAIT_FAILBACK_PREPARED;
-
-      // Add timestamp of start Failback preparation
-      start_timer(failback_gw_id, group_key, gw_state.ana_grp_id, 3);
-      gw_state.sm_state[gw_state.ana_grp_id] =
-       gw_states_per_group_t::GW_OWNER_WAIT_FAILBACK_PREPARED;
-      propose = true;
+      fsm_handle_failback_and_relocation(gw_id, failback_gw_id, group_key,
+                    gw_state.ana_grp_id, propose);
       break;
     }
   }
@@ -1362,6 +1361,21 @@ void NVMeofGwMap::fsm_handle_gw_delete(
   }
 }
 
+void NVMeofGwMap::fsm_handle_failback_and_relocation(
+  const NvmeGwId &owner_gw_id, const NvmeGwId &failover_gw_id,
+  const NvmeGroupKey& group_key,
+  NvmeAnaGrpId grpid,  bool &map_modified)
+{
+  auto& gws_states = created_gws[group_key];
+  gws_states[failover_gw_id].sm_state[grpid] =
+    gw_states_per_group_t::GW_WAIT_FAILBACK_PREPARED;
+  start_timer(failover_gw_id, group_key, grpid, 3);
+
+  gws_states[owner_gw_id].sm_state[grpid] =
+    gw_states_per_group_t::GW_OWNER_WAIT_FAILBACK_PREPARED;
+  map_modified = true;
+}
+
 void NVMeofGwMap::fsm_handle_to_expired(
   const NvmeGwId &gw_id, const NvmeGroupKey& group_key,
   NvmeAnaGrpId grpid,  bool &map_modified)
index e5fa568d722b8a61274418e61318169922881c93..080c86ca0cb8ecaefe6af4679e3070191b0d1408 100644 (file)
@@ -165,6 +165,9 @@ private:
   void fsm_handle_to_expired(
     const NvmeGwId &gw_id, const NvmeGroupKey& group_key,
     NvmeAnaGrpId grpid,  bool &map_modified);
+  void fsm_handle_failback_and_relocation(const NvmeGwId &owner_gw_id,
+     const NvmeGwId &failover_gw_id, const NvmeGroupKey& group_key,
+     NvmeAnaGrpId grpid, bool &map_modified);
   void find_failover_candidate(
     const NvmeGwId &gw_id, const NvmeGroupKey& group_key,
     NvmeAnaGrpId grpid, bool &propose_pending);