|
|
190130 |
From a30a5fdef2e252eba9f44a3c671de8f3aa4f17d7 Mon Sep 17 00:00:00 2001
|
|
|
190130 |
From: Vishal Pandey <vpandey@redhat.com>
|
|
|
190130 |
Date: Tue, 19 Nov 2019 11:39:22 +0530
|
|
|
190130 |
Subject: [PATCH 392/449] glusterd: Brick process fails to come up with
|
|
|
190130 |
brickmux on
|
|
|
190130 |
|
|
|
190130 |
Issue:
|
|
|
190130 |
1- In a cluster of 3 Nodes N1, N2, N3. Create 3 volumes vol1,
|
|
|
190130 |
vol2, vol3 with 3 bricks (one from each node)
|
|
|
190130 |
2- Set cluster.brick-multiplex on
|
|
|
190130 |
3- Start all 3 volumes
|
|
|
190130 |
4- Check if all bricks on a node are running on same port
|
|
|
190130 |
5- Kill N1
|
|
|
190130 |
6- Set performance.readdir-ahead for volumes vol1, vol2, vol3
|
|
|
190130 |
7- Bring N1 up and check volume status
|
|
|
190130 |
8- All bricks processes not running on N1.
|
|
|
190130 |
|
|
|
190130 |
Root Cause -
|
|
|
190130 |
Since, There is a diff in volfile versions in N1 as compared
|
|
|
190130 |
to N2 and N3 therefore glusterd_import_friend_volume() is called.
|
|
|
190130 |
glusterd_import_friend_volume() copies the new_volinfo and deletes
|
|
|
190130 |
old_volinfo and then calls glusterd_start_bricks().
|
|
|
190130 |
glusterd_start_bricks() looks for the volfiles and sends an rpc
|
|
|
190130 |
request to glusterfs_handle_attach(). Now, since the volinfo
|
|
|
190130 |
has been deleted by glusterd_delete_stale_volume()
|
|
|
190130 |
from priv->volumes list before glusterd_start_bricks() and
|
|
|
190130 |
glusterd_create_volfiles_and_notify_services() and
|
|
|
190130 |
glusterd_list_add_order is called after glusterd_start_bricks(),
|
|
|
190130 |
therefore the attach RPC req gets an empty volfile path
|
|
|
190130 |
and that causes the brick to crash.
|
|
|
190130 |
|
|
|
190130 |
Fix- Call glusterd_list_add_order() and
|
|
|
190130 |
glusterd_create_volfiles_and_notify_services before
|
|
|
190130 |
glusterd_start_bricks() cal is made in glusterd_import_friend_volume
|
|
|
190130 |
|
|
|
190130 |
> upstream patch link: https://review.gluster.org/#/c/glusterfs/+/23724/
|
|
|
190130 |
> Change-Id: Idfe0e8710f7eb77ca3ddfa1cabeb45b2987f41aa
|
|
|
190130 |
> Fixes: bz#1773856
|
|
|
190130 |
> Signed-off-by: Mohammed Rafi KC <rkavunga@redhat.com>
|
|
|
190130 |
|
|
|
190130 |
BUG: 1683602
|
|
|
190130 |
Change-Id: Idfe0e8710f7eb77ca3ddfa1cabeb45b2987f41aa
|
|
|
190130 |
Signed-off-by: Sanju Rakonde <srakonde@redhat.com>
|
|
|
190130 |
Reviewed-on: https://code.engineering.redhat.com/gerrit/202255
|
|
|
190130 |
Tested-by: RHGS Build Bot <nigelb@redhat.com>
|
|
|
190130 |
Reviewed-by: Mohit Agrawal <moagrawa@redhat.com>
|
|
|
190130 |
Reviewed-by: Sunil Kumar Heggodu Gopala Acharya <sheggodu@redhat.com>
|
|
|
190130 |
---
|
|
|
190130 |
.../glusterd/brick-mux-validation-in-cluster.t | 61 +++++++++++++++++++++-
|
|
|
190130 |
xlators/mgmt/glusterd/src/glusterd-utils.c | 28 +++++-----
|
|
|
190130 |
2 files changed, 75 insertions(+), 14 deletions(-)
|
|
|
190130 |
|
|
|
190130 |
diff --git a/tests/bugs/glusterd/brick-mux-validation-in-cluster.t b/tests/bugs/glusterd/brick-mux-validation-in-cluster.t
|
|
|
190130 |
index 4e57038..f088dbb 100644
|
|
|
190130 |
--- a/tests/bugs/glusterd/brick-mux-validation-in-cluster.t
|
|
|
190130 |
+++ b/tests/bugs/glusterd/brick-mux-validation-in-cluster.t
|
|
|
190130 |
@@ -7,6 +7,20 @@ function count_brick_processes {
|
|
|
190130 |
pgrep glusterfsd | wc -l
|
|
|
190130 |
}
|
|
|
190130 |
|
|
|
190130 |
+function count_brick_pids {
|
|
|
190130 |
+ $CLI_1 --xml volume status all | sed -n '/.*<pid>\([^<]*\).*/s//\1/p' \
|
|
|
190130 |
+ | grep -v "N/A" | sort | uniq | wc -l
|
|
|
190130 |
+}
|
|
|
190130 |
+
|
|
|
190130 |
+function count_N/A_brick_pids {
|
|
|
190130 |
+ $CLI_1 --xml volume status all | sed -n '/.*<pid>\([^<]*\).*/s//\1/p' \
|
|
|
190130 |
+ | grep -- '\-1' | sort | uniq | wc -l
|
|
|
190130 |
+}
|
|
|
190130 |
+
|
|
|
190130 |
+function check_peers {
|
|
|
190130 |
+ $CLI_2 peer status | grep 'Peer in Cluster (Connected)' | wc -l
|
|
|
190130 |
+}
|
|
|
190130 |
+
|
|
|
190130 |
cleanup;
|
|
|
190130 |
|
|
|
190130 |
TEST launch_cluster 3
|
|
|
190130 |
@@ -48,4 +62,49 @@ TEST $CLI_1 volume stop $V1
|
|
|
190130 |
|
|
|
190130 |
EXPECT 3 count_brick_processes
|
|
|
190130 |
|
|
|
190130 |
-cleanup
|
|
|
190130 |
+TEST $CLI_1 volume stop $META_VOL
|
|
|
190130 |
+
|
|
|
190130 |
+TEST $CLI_1 volume delete $META_VOL
|
|
|
190130 |
+TEST $CLI_1 volume delete $V0
|
|
|
190130 |
+TEST $CLI_1 volume delete $V1
|
|
|
190130 |
+
|
|
|
190130 |
+#bug-1773856 - Brick process fails to come up with brickmux on
|
|
|
190130 |
+
|
|
|
190130 |
+TEST $CLI_1 volume create $V0 $H1:$B1/${V0}1 $H2:$B2/${V0}1 $H3:$B3/${V0}1 force
|
|
|
190130 |
+TEST $CLI_1 volume start $V0
|
|
|
190130 |
+
|
|
|
190130 |
+
|
|
|
190130 |
+EXPECT 3 count_brick_processes
|
|
|
190130 |
+
|
|
|
190130 |
+#create and start a new volume
|
|
|
190130 |
+TEST $CLI_1 volume create $V1 $H1:$B1/${V1}2 $H2:$B2/${V1}2 $H3:$B3/${V1}2 force
|
|
|
190130 |
+TEST $CLI_1 volume start $V1
|
|
|
190130 |
+
|
|
|
190130 |
+EXPECT 3 count_brick_processes
|
|
|
190130 |
+
|
|
|
190130 |
+V2=patchy2
|
|
|
190130 |
+TEST $CLI_1 volume create $V2 $H1:$B1/${V2}3 $H2:$B2/${V2}3 $H3:$B3/${V2}3 force
|
|
|
190130 |
+TEST $CLI_1 volume start $V2
|
|
|
190130 |
+
|
|
|
190130 |
+EXPECT 3 count_brick_processes
|
|
|
190130 |
+
|
|
|
190130 |
+EXPECT_WITHIN $PROCESS_UP_TIMEOUT 3 count_brick_pids
|
|
|
190130 |
+
|
|
|
190130 |
+TEST kill_node 1
|
|
|
190130 |
+
|
|
|
190130 |
+sleep 10
|
|
|
190130 |
+
|
|
|
190130 |
+EXPECT_WITHIN $PROBE_TIMEOUT 1 check_peers;
|
|
|
190130 |
+
|
|
|
190130 |
+$CLI_2 volume set $V0 performance.readdir-ahead on
|
|
|
190130 |
+$CLI_2 volume set $V1 performance.readdir-ahead on
|
|
|
190130 |
+
|
|
|
190130 |
+TEST $glusterd_1;
|
|
|
190130 |
+
|
|
|
190130 |
+sleep 10
|
|
|
190130 |
+
|
|
|
190130 |
+EXPECT 4 count_brick_processes
|
|
|
190130 |
+EXPECT_WITHIN $PROCESS_UP_TIMEOUT 4 count_brick_pids
|
|
|
190130 |
+EXPECT_WITHIN $PROCESS_UP_TIMEOUT 0 count_N/A_brick_pids
|
|
|
190130 |
+
|
|
|
190130 |
+cleanup;
|
|
|
190130 |
diff --git a/xlators/mgmt/glusterd/src/glusterd-utils.c b/xlators/mgmt/glusterd/src/glusterd-utils.c
|
|
|
190130 |
index 6654741..1b78812 100644
|
|
|
190130 |
--- a/xlators/mgmt/glusterd/src/glusterd-utils.c
|
|
|
190130 |
+++ b/xlators/mgmt/glusterd/src/glusterd-utils.c
|
|
|
190130 |
@@ -4988,16 +4988,6 @@ glusterd_import_friend_volume(dict_t *peer_data, int count)
|
|
|
190130 |
glusterd_volinfo_unref(old_volinfo);
|
|
|
190130 |
}
|
|
|
190130 |
|
|
|
190130 |
- if (glusterd_is_volume_started(new_volinfo)) {
|
|
|
190130 |
- (void)glusterd_start_bricks(new_volinfo);
|
|
|
190130 |
- if (glusterd_is_snapd_enabled(new_volinfo)) {
|
|
|
190130 |
- svc = &(new_volinfo->snapd.svc);
|
|
|
190130 |
- if (svc->manager(svc, new_volinfo, PROC_START_NO_WAIT)) {
|
|
|
190130 |
- gf_event(EVENT_SVC_MANAGER_FAILED, "svc_name=%s", svc->name);
|
|
|
190130 |
- }
|
|
|
190130 |
- }
|
|
|
190130 |
- }
|
|
|
190130 |
-
|
|
|
190130 |
ret = glusterd_store_volinfo(new_volinfo, GLUSTERD_VOLINFO_VER_AC_NONE);
|
|
|
190130 |
if (ret) {
|
|
|
190130 |
gf_msg(this->name, GF_LOG_ERROR, 0, GD_MSG_VOLINFO_STORE_FAIL,
|
|
|
190130 |
@@ -5007,19 +4997,31 @@ glusterd_import_friend_volume(dict_t *peer_data, int count)
|
|
|
190130 |
goto out;
|
|
|
190130 |
}
|
|
|
190130 |
|
|
|
190130 |
- ret = glusterd_create_volfiles_and_notify_services(new_volinfo);
|
|
|
190130 |
+ ret = glusterd_create_volfiles(new_volinfo);
|
|
|
190130 |
if (ret)
|
|
|
190130 |
goto out;
|
|
|
190130 |
|
|
|
190130 |
+ glusterd_list_add_order(&new_volinfo->vol_list, &priv->volumes,
|
|
|
190130 |
+ glusterd_compare_volume_name);
|
|
|
190130 |
+
|
|
|
190130 |
+ if (glusterd_is_volume_started(new_volinfo)) {
|
|
|
190130 |
+ (void)glusterd_start_bricks(new_volinfo);
|
|
|
190130 |
+ if (glusterd_is_snapd_enabled(new_volinfo)) {
|
|
|
190130 |
+ svc = &(new_volinfo->snapd.svc);
|
|
|
190130 |
+ if (svc->manager(svc, new_volinfo, PROC_START_NO_WAIT)) {
|
|
|
190130 |
+ gf_event(EVENT_SVC_MANAGER_FAILED, "svc_name=%s", svc->name);
|
|
|
190130 |
+ }
|
|
|
190130 |
+ }
|
|
|
190130 |
+ }
|
|
|
190130 |
+
|
|
|
190130 |
ret = glusterd_import_quota_conf(peer_data, count, new_volinfo, "volume");
|
|
|
190130 |
if (ret) {
|
|
|
190130 |
gf_event(EVENT_IMPORT_QUOTA_CONF_FAILED, "volume=%s",
|
|
|
190130 |
new_volinfo->volname);
|
|
|
190130 |
goto out;
|
|
|
190130 |
}
|
|
|
190130 |
- glusterd_list_add_order(&new_volinfo->vol_list, &priv->volumes,
|
|
|
190130 |
- glusterd_compare_volume_name);
|
|
|
190130 |
|
|
|
190130 |
+ ret = glusterd_fetchspec_notify(this);
|
|
|
190130 |
out:
|
|
|
190130 |
gf_msg_debug("glusterd", 0, "Returning with ret: %d", ret);
|
|
|
190130 |
return ret;
|
|
|
190130 |
--
|
|
|
190130 |
1.8.3.1
|
|
|
190130 |
|