Merge pull request #69490 from leonidc/fix_relocation_stretched

nvmeofgw: stretched cluster fix relocate
This commit is contained in:
leonidc 2026-07-19 09:40:38 +03:00 committed by GitHub
commit f72cb692d6
No known key found for this signature in database
GPG Key ID: B5690EEEBB952194
2 changed files with 70 additions and 53 deletions

View File

@ -818,7 +818,9 @@ void NVMeofGwMap::handle_abandoned_ana_groups(bool& propose)
void NVMeofGwMap::check_relocate_ana_groups(const NvmeGroupKey& group_key,
bool &propose) {
/* if location in disaster_locations found in recovering state state - find all gws in location.
/* loop for all locations:
* if location in normal state or "disaster_locations"
* found in recovering state state - find all gws in location.
* add ana-grp of not Available gws to the list.
* if ana-grp is already active on some gw in location skip it
* for ana-grp in list make relocation.
@ -830,48 +832,57 @@ void NVMeofGwMap::check_relocate_ana_groups(const NvmeGroupKey& group_key,
return ;
}
std::list<NvmeAnaGrpId> reloc_list;
std::unordered_set<NvmeLocation> locations_set;
auto& gws_states = created_gws[group_key];
FailbackLocation location;
if (get_location_in_disaster_cleanup(group_key, location)) {
uint32_t num_gw_in_location = 0;
uint32_t num_active_ana_in_location = 0;
for (auto& gw_state : gws_states) { // loop for GWs inside group-key
NvmeGwMonState& state = gw_state.second;
if (state.location == location) {
num_gw_in_location ++;
if (state.availability != gw_availability_t::GW_AVAILABLE) {
reloc_list.push_back(state.ana_grp_id);
} else { // in parallel check condition to complete failback-in-process
for (auto& state_it: state.sm_state) {
if (state_it.second == gw_states_per_group_t::GW_ACTIVE_STATE) {
num_active_ana_in_location ++;
for (auto& gw_state : gws_states) {// build locations set
locations_set.insert(gw_state.second.location);
}
for (NvmeLocation location : locations_set) {
bool cleanup_in_process;
reloc_list.clear();
bool disaster = is_location_in_disaster(group_key, location, cleanup_in_process);
if ((disaster && cleanup_in_process) || (!disaster)) {
uint32_t num_gw_in_location = 0;
uint32_t num_active_ana_in_location = 0;
for (auto& gw_state : gws_states) { // loop for GWs inside group-key
NvmeGwMonState& state = gw_state.second;
if (state.location == location) {
num_gw_in_location++;
if (state.availability != gw_availability_t::GW_AVAILABLE) {
reloc_list.push_back(state.ana_grp_id);
} else { // in parallel check condition to complete failback-in-process
for (auto& state_it: state.sm_state) {
if (state_it.second == gw_states_per_group_t::GW_ACTIVE_STATE) {
num_active_ana_in_location ++;
}
}
}
}
}
}
if (num_gw_in_location == num_active_ana_in_location) {// All ana groups of disaster location are in Active
disaster_map_remove_location(group_key, location);
dout(4) << "the location entry is erased "<< location
<< " from disaster-locations num_ana_groups in location "
<< num_gw_in_location
<< " from the failbacks-in-progress of group " << group_key <<dendl;
propose = true;
return;
}
if (num_gw_in_location == num_active_ana_in_location) {// All ana groups of disaster location are in Active
disaster_map_remove_location(group_key, location);
dout(4) << "the location entry is erased "<< location
<< " from disaster-locations num_ana_groups in location "
<< num_gw_in_location
<< " from the failbacks-in-progress of group " << group_key <<dendl;
propose = true;
return;
}
// for all ana groups in the list do relocate
for (auto& anagrp : reloc_list) {
for (auto& gw_state : gws_states) { // loop for GWs inside group-key
NvmeGwMonState& state = gw_state.second;
if (state.sm_state[anagrp] == gw_states_per_group_t::GW_ACTIVE_STATE) {
if (state.location == location) { // already relocated to the location
dout(10) << "ana " << anagrp << " already in " << location << dendl;
break;
} else { // try to relocate
dout(10) << "ana " << anagrp
<< " to relocate to " << location << dendl;
relocate_ana_grp(gw_state.first, group_key, anagrp,
location, propose);
for (auto& anagrp : reloc_list) {
for (auto& gw_state : gws_states) { // loop for GWs inside group-key
NvmeGwMonState& state = gw_state.second;
if (state.sm_state[anagrp] == gw_states_per_group_t::GW_ACTIVE_STATE) {
if (state.location == location) { // already relocated to the location
dout(10) << "ana " << anagrp << " already in " << location << dendl;
break;
} else { // try to relocate
dout(10) << "ana " << anagrp
<< " relocate to " << location << dendl;
relocate_ana_grp(gw_state.first, group_key, anagrp,
location, propose);
return; // allow just 1 relocation during a tick()
}
}
}
}
@ -926,14 +937,8 @@ int NVMeofGwMap::relocate_ana_grp(const NvmeGwId &src_gw_id,
<< location << "min load " << min_num_ana_groups_in_gw << dendl;
if (min_num_ana_groups_in_gw < MAX_NUM_ANA_GROUPS_FOR_RELOCATE) {
dout(4) << "relocate starts " << grpid << " location " << location << dendl;
gws_states[src_gw_id].sm_state[grpid] =
gw_states_per_group_t::GW_WAIT_FAILBACK_PREPARED;
// Add timestamp of start Failback preparation
start_timer(src_gw_id, group_key, grpid, 3);
gws_states[min_gw_id].sm_state[grpid] =
gw_states_per_group_t::GW_OWNER_WAIT_FAILBACK_PREPARED;
propose = true;
dout(4) << "relocate starts ANA group " << grpid << " location " << location << dendl;
fsm_handle_failback_and_relocation(min_gw_id, src_gw_id, group_key, grpid, propose);
}
return 0;
}
@ -1031,14 +1036,8 @@ void NVMeofGwMap::find_failback_gw(
<< gw_id << dendl;
return;
}
st.sm_state[gw_state.ana_grp_id] =
gw_states_per_group_t::GW_WAIT_FAILBACK_PREPARED;
// Add timestamp of start Failback preparation
start_timer(failback_gw_id, group_key, gw_state.ana_grp_id, 3);
gw_state.sm_state[gw_state.ana_grp_id] =
gw_states_per_group_t::GW_OWNER_WAIT_FAILBACK_PREPARED;
propose = true;
fsm_handle_failback_and_relocation(gw_id, failback_gw_id, group_key,
gw_state.ana_grp_id, propose);
break;
}
}
@ -1362,6 +1361,21 @@ void NVMeofGwMap::fsm_handle_gw_delete(
}
}
void NVMeofGwMap::fsm_handle_failback_and_relocation(
const NvmeGwId &owner_gw_id, const NvmeGwId &failover_gw_id,
const NvmeGroupKey& group_key,
NvmeAnaGrpId grpid, bool &map_modified)
{
auto& gws_states = created_gws[group_key];
gws_states[failover_gw_id].sm_state[grpid] =
gw_states_per_group_t::GW_WAIT_FAILBACK_PREPARED;
start_timer(failover_gw_id, group_key, grpid, 3);
gws_states[owner_gw_id].sm_state[grpid] =
gw_states_per_group_t::GW_OWNER_WAIT_FAILBACK_PREPARED;
map_modified = true;
}
void NVMeofGwMap::fsm_handle_to_expired(
const NvmeGwId &gw_id, const NvmeGroupKey& group_key,
NvmeAnaGrpId grpid, bool &map_modified)

View File

@ -165,6 +165,9 @@ private:
void fsm_handle_to_expired(
const NvmeGwId &gw_id, const NvmeGroupKey& group_key,
NvmeAnaGrpId grpid, bool &map_modified);
void fsm_handle_failback_and_relocation(const NvmeGwId &owner_gw_id,
const NvmeGwId &failover_gw_id, const NvmeGroupKey& group_key,
NvmeAnaGrpId grpid, bool &map_modified);
void find_failover_candidate(
const NvmeGwId &gw_id, const NvmeGroupKey& group_key,
NvmeAnaGrpId grpid, bool &propose_pending);