a3470f
From 4bf98e63a481aea6143e8f404aa4650f7a80e317 Mon Sep 17 00:00:00 2001
a3470f
From: Atin Mukherjee <amukherj@redhat.com>
a3470f
Date: Wed, 3 Jan 2018 14:29:51 +0530
a3470f
Subject: [PATCH 120/128] glusterd: connect to an existing brick process when
a3470f
 qourum status is NOT_APPLICABLE_QUORUM
a3470f
a3470f
First of all, this patch reverts commit 635c1c3 as the same is causing a
a3470f
regression with bricks not coming up on time when a node is rebooted.
a3470f
This patch tries to fix the problem in a different way by just trying to
a3470f
connect to an existing running brick when quorum status is not
a3470f
applicable.
a3470f
a3470f
> upstream patch : https://review.gluster.org/#/c/19134/
a3470f
a3470f
Change-Id: I0efb5901832824b1c15dcac529bffac85173e097
a3470f
BUG: 1509102
a3470f
Signed-off-by: Atin Mukherjee <amukherj@redhat.com>
a3470f
Reviewed-on: https://code.engineering.redhat.com/gerrit/126996
a3470f
Tested-by: RHGS Build Bot <nigelb@redhat.com>
a3470f
---
a3470f
 xlators/mgmt/glusterd/src/glusterd-brick-ops.c     |  2 +-
a3470f
 xlators/mgmt/glusterd/src/glusterd-handshake.c     |  2 +-
a3470f
 xlators/mgmt/glusterd/src/glusterd-op-sm.c         |  1 +
a3470f
 xlators/mgmt/glusterd/src/glusterd-replace-brick.c |  3 ++-
a3470f
 xlators/mgmt/glusterd/src/glusterd-server-quorum.c | 27 ++++++++++++++++++----
a3470f
 xlators/mgmt/glusterd/src/glusterd-utils.c         | 13 +++++++----
a3470f
 xlators/mgmt/glusterd/src/glusterd-utils.h         |  3 ++-
a3470f
 xlators/mgmt/glusterd/src/glusterd-volume-ops.c    |  3 ++-
a3470f
 8 files changed, 40 insertions(+), 14 deletions(-)
a3470f
a3470f
diff --git a/xlators/mgmt/glusterd/src/glusterd-brick-ops.c b/xlators/mgmt/glusterd/src/glusterd-brick-ops.c
a3470f
index e88fa3f..416412e 100644
a3470f
--- a/xlators/mgmt/glusterd/src/glusterd-brick-ops.c
a3470f
+++ b/xlators/mgmt/glusterd/src/glusterd-brick-ops.c
a3470f
@@ -1554,7 +1554,7 @@ glusterd_op_perform_add_bricks (glusterd_volinfo_t *volinfo, int32_t count,
a3470f
                         }
a3470f
                 }
a3470f
                 ret = glusterd_brick_start (volinfo, brickinfo,
a3470f
-                                            _gf_true);
a3470f
+                                            _gf_true, _gf_false);
a3470f
                 if (ret)
a3470f
                         goto out;
a3470f
                 i++;
a3470f
diff --git a/xlators/mgmt/glusterd/src/glusterd-handshake.c b/xlators/mgmt/glusterd/src/glusterd-handshake.c
a3470f
index 35aeca3..3d1dfb2 100644
a3470f
--- a/xlators/mgmt/glusterd/src/glusterd-handshake.c
a3470f
+++ b/xlators/mgmt/glusterd/src/glusterd-handshake.c
a3470f
@@ -658,7 +658,7 @@ glusterd_create_missed_snap (glusterd_missed_snap_info *missed_snapinfo,
a3470f
         }
a3470f
 
a3470f
         brickinfo->snap_status = 0;
a3470f
-        ret = glusterd_brick_start (snap_vol, brickinfo, _gf_false);
a3470f
+        ret = glusterd_brick_start (snap_vol, brickinfo, _gf_false, _gf_false);
a3470f
         if (ret) {
a3470f
                 gf_msg (this->name, GF_LOG_WARNING, 0,
a3470f
                         GD_MSG_BRICK_DISCONNECTED, "starting the "
a3470f
diff --git a/xlators/mgmt/glusterd/src/glusterd-op-sm.c b/xlators/mgmt/glusterd/src/glusterd-op-sm.c
a3470f
index 86f18f0..b1a6e06 100644
a3470f
--- a/xlators/mgmt/glusterd/src/glusterd-op-sm.c
a3470f
+++ b/xlators/mgmt/glusterd/src/glusterd-op-sm.c
a3470f
@@ -2437,6 +2437,7 @@ glusterd_start_bricks (glusterd_volinfo_t *volinfo)
a3470f
                         pthread_mutex_lock (&brickinfo->restart_mutex);
a3470f
                         {
a3470f
                                 ret = glusterd_brick_start (volinfo, brickinfo,
a3470f
+                                                            _gf_false,
a3470f
                                                             _gf_false);
a3470f
                         }
a3470f
                         pthread_mutex_unlock (&brickinfo->restart_mutex);
a3470f
diff --git a/xlators/mgmt/glusterd/src/glusterd-replace-brick.c b/xlators/mgmt/glusterd/src/glusterd-replace-brick.c
a3470f
index b11adf1..a037323 100644
a3470f
--- a/xlators/mgmt/glusterd/src/glusterd-replace-brick.c
a3470f
+++ b/xlators/mgmt/glusterd/src/glusterd-replace-brick.c
a3470f
@@ -429,7 +429,8 @@ glusterd_op_perform_replace_brick (glusterd_volinfo_t  *volinfo,
a3470f
                 goto out;
a3470f
 
a3470f
         if (GLUSTERD_STATUS_STARTED == volinfo->status) {
a3470f
-                ret = glusterd_brick_start (volinfo, new_brickinfo, _gf_false);
a3470f
+                ret = glusterd_brick_start (volinfo, new_brickinfo, _gf_false,
a3470f
+                                            _gf_false);
a3470f
                 if (ret)
a3470f
                         goto out;
a3470f
         }
a3470f
diff --git a/xlators/mgmt/glusterd/src/glusterd-server-quorum.c b/xlators/mgmt/glusterd/src/glusterd-server-quorum.c
a3470f
index 995a568..b01bfaa 100644
a3470f
--- a/xlators/mgmt/glusterd/src/glusterd-server-quorum.c
a3470f
+++ b/xlators/mgmt/glusterd/src/glusterd-server-quorum.c
a3470f
@@ -314,6 +314,7 @@ glusterd_do_volume_quorum_action (xlator_t *this, glusterd_volinfo_t *volinfo,
a3470f
         glusterd_brickinfo_t *brickinfo     = NULL;
a3470f
         gd_quorum_status_t   quorum_status  = NOT_APPLICABLE_QUORUM;
a3470f
         gf_boolean_t         follows_quorum = _gf_false;
a3470f
+        gf_boolean_t         quorum_status_unchanged = _gf_false;
a3470f
 
a3470f
         if (volinfo->status != GLUSTERD_STATUS_STARTED) {
a3470f
                 volinfo->quorum_status = NOT_APPLICABLE_QUORUM;
a3470f
@@ -341,9 +342,10 @@ glusterd_do_volume_quorum_action (xlator_t *this, glusterd_volinfo_t *volinfo,
a3470f
          * the bricks that are down are brought up again. In this process it
a3470f
          * also brings up the brick that is purposefully taken down.
a3470f
          */
a3470f
-        if (quorum_status != NOT_APPLICABLE_QUORUM &&
a3470f
-            volinfo->quorum_status == quorum_status)
a3470f
+        if (volinfo->quorum_status == quorum_status) {
a3470f
+                quorum_status_unchanged = _gf_true;
a3470f
                 goto out;
a3470f
+        }
a3470f
 
a3470f
         if (quorum_status == MEETS_QUORUM) {
a3470f
                 gf_msg (this->name, GF_LOG_CRITICAL, 0,
a3470f
@@ -368,9 +370,10 @@ glusterd_do_volume_quorum_action (xlator_t *this, glusterd_volinfo_t *volinfo,
a3470f
                         if (!brickinfo->start_triggered) {
a3470f
                                 pthread_mutex_lock (&brickinfo->restart_mutex);
a3470f
                                 {
a3470f
-                                        glusterd_brick_start (volinfo,
a3470f
-                                                              brickinfo,
a3470f
-                                                              _gf_false);
a3470f
+                                        ret = glusterd_brick_start (volinfo,
a3470f
+                                                                    brickinfo,
a3470f
+                                                                    _gf_false,
a3470f
+                                                                    _gf_false);
a3470f
                                 }
a3470f
                                 pthread_mutex_unlock (&brickinfo->restart_mutex);
a3470f
                         }
a3470f
@@ -392,6 +395,20 @@ glusterd_do_volume_quorum_action (xlator_t *this, glusterd_volinfo_t *volinfo,
a3470f
                 }
a3470f
         }
a3470f
 out:
a3470f
+        if (quorum_status_unchanged) {
a3470f
+                list_for_each_entry (brickinfo, &volinfo->bricks, brick_list) {
a3470f
+                        if (!glusterd_is_local_brick (this, volinfo, brickinfo))
a3470f
+                                continue;
a3470f
+                        ret = glusterd_brick_start (volinfo, brickinfo,
a3470f
+                                                    _gf_false, _gf_true);
a3470f
+                        if (ret) {
a3470f
+                                gf_msg (this->name, GF_LOG_ERROR, 0,
a3470f
+                                        GD_MSG_BRICK_DISCONNECTED, "Failed to "
a3470f
+                                        "connect to %s:%s", brickinfo->hostname,
a3470f
+                                        brickinfo->path);
a3470f
+                        }
a3470f
+                }
a3470f
+        }
a3470f
         return;
a3470f
 }
a3470f
 
a3470f
diff --git a/xlators/mgmt/glusterd/src/glusterd-utils.c b/xlators/mgmt/glusterd/src/glusterd-utils.c
a3470f
index 1b2cc43..f1b365f 100644
a3470f
--- a/xlators/mgmt/glusterd/src/glusterd-utils.c
a3470f
+++ b/xlators/mgmt/glusterd/src/glusterd-utils.c
a3470f
@@ -5796,7 +5796,8 @@ glusterd_get_sock_from_brick_pid (int pid, char *sockpath, size_t len)
a3470f
 int
a3470f
 glusterd_brick_start (glusterd_volinfo_t *volinfo,
a3470f
                       glusterd_brickinfo_t *brickinfo,
a3470f
-                      gf_boolean_t wait)
a3470f
+                      gf_boolean_t wait,
a3470f
+                      gf_boolean_t only_connect)
a3470f
 {
a3470f
         int                     ret   = -1;
a3470f
         xlator_t                *this = NULL;
a3470f
@@ -5847,7 +5848,9 @@ glusterd_brick_start (glusterd_volinfo_t *volinfo,
a3470f
                 ret = 0;
a3470f
                 goto out;
a3470f
         }
a3470f
-        brickinfo->start_triggered = _gf_true;
a3470f
+        if (!only_connect)
a3470f
+                brickinfo->start_triggered = _gf_true;
a3470f
+
a3470f
         GLUSTERD_GET_BRICK_PIDFILE (pidfile, volinfo, brickinfo, conf);
a3470f
         if (gf_is_service_running (pidfile, &pid)) {
a3470f
                 if (brickinfo->status != GF_BRICK_STARTING &&
a3470f
@@ -5905,6 +5908,8 @@ glusterd_brick_start (glusterd_volinfo_t *volinfo,
a3470f
                 }
a3470f
                 return 0;
a3470f
         }
a3470f
+        if (only_connect)
a3470f
+                return 0;
a3470f
 
a3470f
 run:
a3470f
         ret = _mk_rundir_p (volinfo);
a3470f
@@ -6032,7 +6037,7 @@ glusterd_restart_bricks (glusterd_conf_t *conf)
a3470f
                                         {
a3470f
                                                 glusterd_brick_start
a3470f
                                                          (volinfo, brickinfo,
a3470f
-                                                          _gf_false);
a3470f
+                                                          _gf_false, _gf_false);
a3470f
                                         }
a3470f
                                         pthread_mutex_unlock
a3470f
                                                 (&brickinfo->restart_mutex);
a3470f
@@ -6081,7 +6086,7 @@ glusterd_restart_bricks (glusterd_conf_t *conf)
a3470f
                                         {
a3470f
                                                 glusterd_brick_start
a3470f
                                                          (volinfo, brickinfo,
a3470f
-                                                          _gf_false);
a3470f
+                                                          _gf_false, _gf_false);
a3470f
                                         }
a3470f
                                         pthread_mutex_unlock
a3470f
                                                 (&brickinfo->restart_mutex);
a3470f
diff --git a/xlators/mgmt/glusterd/src/glusterd-utils.h b/xlators/mgmt/glusterd/src/glusterd-utils.h
a3470f
index abaec4b..9194da0 100644
a3470f
--- a/xlators/mgmt/glusterd/src/glusterd-utils.h
a3470f
+++ b/xlators/mgmt/glusterd/src/glusterd-utils.h
a3470f
@@ -277,7 +277,8 @@ glusterd_all_volume_cond_check (glusterd_condition_func func, int status,
a3470f
 int
a3470f
 glusterd_brick_start (glusterd_volinfo_t *volinfo,
a3470f
                       glusterd_brickinfo_t *brickinfo,
a3470f
-                      gf_boolean_t wait);
a3470f
+                      gf_boolean_t wait,
a3470f
+                      gf_boolean_t only_connect);
a3470f
 int
a3470f
 glusterd_brick_stop (glusterd_volinfo_t *volinfo,
a3470f
                      glusterd_brickinfo_t *brickinfo,
a3470f
diff --git a/xlators/mgmt/glusterd/src/glusterd-volume-ops.c b/xlators/mgmt/glusterd/src/glusterd-volume-ops.c
a3470f
index de97e6a..414f9ba 100644
a3470f
--- a/xlators/mgmt/glusterd/src/glusterd-volume-ops.c
a3470f
+++ b/xlators/mgmt/glusterd/src/glusterd-volume-ops.c
a3470f
@@ -2564,7 +2564,8 @@ glusterd_start_volume (glusterd_volinfo_t *volinfo, int flags,
a3470f
                 if (flags & GF_CLI_FLAG_OP_FORCE) {
a3470f
                         brickinfo->start_triggered = _gf_false;
a3470f
                 }
a3470f
-                ret = glusterd_brick_start (volinfo, brickinfo, wait);
a3470f
+                ret = glusterd_brick_start (volinfo, brickinfo, wait,
a3470f
+                                            _gf_false);
a3470f
                 /* If 'force' try to start all bricks regardless of success or
a3470f
                  * failure
a3470f
                  */
a3470f
-- 
a3470f
1.8.3.1
a3470f