burnin: do not use offline nodes

[ganeti-local] / lib / cmdlib.py
diff --git a/lib/cmdlib.py b/lib/cmdlib.py

index f0521aa..b7b2fa8 100644 (file)
--- a/lib/cmdlib.py
+++ b/lib/cmdlib.py
@@ -415,6 +415,32 @@ def _CheckOutputFields(static, dynamic, selected):
                                 % ",".join(delta))
  
  
+def _CheckBooleanOpField(op, name):
+  """Validates boolean opcode parameters.
+
+  This will ensure that an opcode parameter is either a boolean value,
+  or None (but that it always exists).
+
+  """
+  val = getattr(op, name, None)
+  if not (val is None or isinstance(val, bool)):
+    raise errors.OpPrereqError("Invalid boolean parameter '%s' (%s)" %
+                               (name, str(val)))
+  setattr(op, name, val)
+
+
+def _CheckNodeOnline(lu, node):
+  """Ensure that a given node is online.
+
+  @param lu: the LU on behalf of which we make the check
+  @param node: the node to check
+  @raise errors.OpPrereqError: if the nodes is offline
+
+  """
+  if lu.cfg.GetNodeInfo(node).offline:
+    raise errors.OpPrereqError("Can't use offline node %s" % node)
+
+
  def _BuildInstanceHookEnv(name, primary_node, secondary_nodes, os_type, status,
                            memory, vcpus, nics):
    """Builds instance related env variables for hooks
@@ -500,6 +526,22 @@ def _BuildInstanceHookEnvByObject(lu, instance, override=None):
    return _BuildInstanceHookEnv(**args)
  
  
+def _AdjustCandidatePool(lu):
+  """Adjust the candidate pool after node operations.
+
+  """
+  mod_list = lu.cfg.MaintainCandidatePool()
+  if mod_list:
+    lu.LogInfo("Promoted nodes to master candidate role: %s",
+               ", ".join(node.name for node in mod_list))
+    for name in mod_list:
+      lu.context.ReaddNode(name)
+  mc_now, mc_max = lu.cfg.GetMasterCandidateStats()
+  if mc_now > mc_max:
+    lu.LogInfo("Note: more nodes are candidates (%d) than desired (%d)" %
+               (mc_now, mc_max))
+
+
  def _CheckInstanceBridgesExist(lu, instance):
    """Check that the brigdes needed by an instance exist.
  
@@ -685,7 +727,7 @@ class LUVerifyCluster(LogicalUnit):
      return bad
  
    def _VerifyInstance(self, instance, instanceconfig, node_vol_is,
-                      node_instance, feedback_fn):
+                      node_instance, feedback_fn, n_offline):
      """Verify an instance.
  
      This function checks to see if the required block devices are
@@ -700,6 +742,9 @@ class LUVerifyCluster(LogicalUnit):
      instanceconfig.MapLVsByNode(node_vol_should)
  
      for node in node_vol_should:
+      if node in n_offline:
+        # ignore missing volumes on offline nodes
+        continue
        for volume in node_vol_should[node]:
          if node not in node_vol_is or volume not in node_vol_is[node]:
            feedback_fn("  - ERROR: volume %s missing on node %s" %
@@ -707,8 +752,9 @@ class LUVerifyCluster(LogicalUnit):
            bad = True
  
      if not instanceconfig.status == 'down':
-      if (node_current not in node_instance or
-          not instance in node_instance[node_current]):
+      if ((node_current not in node_instance or
+          not instance in node_instance[node_current]) and
+          node_current not in n_offline):
          feedback_fn("  - ERROR: instance %s not running on node %s" %
                          (instance, node_current))
          bad = True
@@ -823,6 +869,7 @@ class LUVerifyCluster(LogicalUnit):
      instancelist = utils.NiceSort(self.cfg.GetInstanceList())
      i_non_redundant = [] # Non redundant instances
      i_non_a_balanced = [] # Non auto-balanced instances
+    n_offline = [] # List of offline nodes
      node_volume = {}
      node_instance = {}
      node_info = {}
@@ -834,6 +881,7 @@ class LUVerifyCluster(LogicalUnit):
  
      file_names = ssconf.SimpleStore().GetFileList()
      file_names.append(constants.SSL_CERT_FILE)
+    file_names.append(constants.RAPI_CERT_FILE)
      file_names.extend(master_files)
  
      local_checksums = utils.FingerprintFiles(file_names)
@@ -841,10 +889,12 @@ class LUVerifyCluster(LogicalUnit):
      feedback_fn("* Gathering data (%d nodes)" % len(nodelist))
      node_verify_param = {
        constants.NV_FILELIST: file_names,
-      constants.NV_NODELIST: nodelist,
+      constants.NV_NODELIST: [node.name for node in nodeinfo
+                              if not node.offline],
        constants.NV_HYPERVISOR: hypervisors,
        constants.NV_NODENETTEST: [(node.name, node.primary_ip,
-                                  node.secondary_ip) for node in nodeinfo],
+                                  node.secondary_ip) for node in nodeinfo
+                                 if not node.offline],
        constants.NV_LVLIST: vg_name,
        constants.NV_INSTANCELIST: hypervisors,
        constants.NV_VGLIST: None,
@@ -860,6 +910,11 @@ class LUVerifyCluster(LogicalUnit):
        node = node_i.name
        nresult = all_nvinfo[node].data
  
+      if node_i.offline:
+        feedback_fn("* Skipping offline node %s" % (node,))
+        n_offline.append(node)
+        continue
+
        if node == master_node:
          ntype = "master"
        elif node_i.master_candidate:
@@ -932,8 +987,9 @@ class LUVerifyCluster(LogicalUnit):
        feedback_fn("* Verifying instance %s" % instance)
        inst_config = self.cfg.GetInstanceInfo(instance)
        result =  self._VerifyInstance(instance, inst_config, node_volume,
-                                     node_instance, feedback_fn)
+                                     node_instance, feedback_fn, n_offline)
        bad = bad or result
+      inst_nodes_offline = []
  
        inst_config.MapLVsByNode(node_vol_should)
  
@@ -942,11 +998,14 @@ class LUVerifyCluster(LogicalUnit):
        pnode = inst_config.primary_node
        if pnode in node_info:
          node_info[pnode]['pinst'].append(instance)
-      else:
+      elif pnode not in n_offline:
          feedback_fn("  - ERROR: instance %s, connection to primary node"
                      " %s failed" % (instance, pnode))
          bad = True
  
+      if pnode in n_offline:
+        inst_nodes_offline.append(pnode)
+
        # If the instance is non-redundant we cannot survive losing its primary
        # node, so we are not N+1 compliant. On the other hand we have no disk
        # templates with more than one secondary so that situation is not well
@@ -967,9 +1026,18 @@ class LUVerifyCluster(LogicalUnit):
            if pnode not in node_info[snode]['sinst-by-pnode']:
              node_info[snode]['sinst-by-pnode'][pnode] = []
            node_info[snode]['sinst-by-pnode'][pnode].append(instance)
-        else:
+        elif snode not in n_offline:
            feedback_fn("  - ERROR: instance %s, connection to secondary node"
                        " %s failed" % (instance, snode))
+          bad = True
+        if snode in n_offline:
+          inst_nodes_offline.append(snode)
+
+      if inst_nodes_offline:
+        # warn that the instance lives on offline nodes, and set bad=True
+        feedback_fn("  - ERROR: instance lives on offline node(s) %s" %
+                    ", ".join(inst_nodes_offline))
+        bad = True
  
      feedback_fn("* Verifying orphan volumes")
      result = self._VerifyOrphanVolumes(node_vol_should, node_volume,
@@ -995,6 +1063,9 @@ class LUVerifyCluster(LogicalUnit):
        feedback_fn("  - NOTICE: %d non-auto-balanced instance(s) found."
                    % len(i_non_a_balanced))
  
+    if n_offline:
+      feedback_fn("  - NOTICE: %d offline node(s) found." % len(n_offline))
+
      return not bad
  
    def HooksCallBack(self, phase, hooks_results, feedback_fn, lu_result):
@@ -1026,6 +1097,9 @@ class LUVerifyCluster(LogicalUnit):
            show_node_header = True
            res = hooks_results[node_name]
            if res.failed or res.data is False or not isinstance(res.data, list):
+            if res.offline:
+              # no need to warn or set fail return value
+              continue
              feedback_fn("    Communication failure in hooks execution")
              lu_result = 1
              continue
@@ -1099,8 +1173,9 @@ class LUVerifyDisks(NoHooksLU):
        # node_volume
        lvs = node_lvs[node]
        if lvs.failed:
-        self.LogWarning("Connection to node %s failed: %s" %
-                        (node, lvs.data))
+        if not lvs.offline:
+          self.LogWarning("Connection to node %s failed: %s" %
+                          (node, lvs.data))
          continue
        lvs = lvs.data
        if isinstance(lvs, basestring):
@@ -1198,7 +1273,8 @@ class LURenameCluster(LogicalUnit):
                                           constants.SSH_KNOWN_HOSTS_FILE)
        for to_node, to_result in result.iteritems():
          if to_result.failed or not to_result.data:
-          logging.error("Copy of file %s to node %s failed", fname, to_node)
+          logging.error("Copy of file %s to node %s failed",
+                        constants.SSH_KNOWN_HOSTS_FILE, to_node)
  
      finally:
        result = self.rpc.call_node_start_master(master, False)
@@ -1358,26 +1434,7 @@ class LUSetClusterParams(LogicalUnit):
      # we want to update nodes after the cluster so that if any errors
      # happen, we have recorded and saved the cluster info
      if self.op.candidate_pool_size is not None:
-      node_info = self.cfg.GetAllNodesInfo().values()
-      num_candidates = len([node for node in node_info
-                            if node.master_candidate])
-      num_nodes = len(node_info)
-      if num_candidates < self.op.candidate_pool_size:
-        random.shuffle(node_info)
-        for node in node_info:
-          if num_candidates >= self.op.candidate_pool_size:
-            break
-          if node.master_candidate:
-            continue
-          node.master_candidate = True
-          self.LogInfo("Promoting node %s to master candidate", node.name)
-          self.cfg.Update(node)
-          self.context.ReaddNode(node)
-          num_candidates += 1
-      elif num_candidates > self.op.candidate_pool_size:
-        self.LogInfo("Note: more nodes are candidates (%d) than the new value"
-                     " of candidate_pool_size (%d)" %
-                     (num_candidates, self.op.candidate_pool_size))
+      _AdjustCandidatePool(self)
  
  
  def _WaitForSync(lu, instance, oneshot=False, unlock=False):
@@ -1530,10 +1587,12 @@ class LUDiagnoseOS(NoHooksLU):
  
      """
      node_list = self.acquired_locks[locking.LEVEL_NODE]
-    node_data = self.rpc.call_os_diagnose(node_list)
+    valid_nodes = [node for node in self.cfg.GetOnlineNodeList()
+                   if node in node_list]
+    node_data = self.rpc.call_os_diagnose(valid_nodes)
      if node_data == False:
        raise errors.OpExecError("Can't gather the list of OSes")
-    pol = self._DiagnoseByOS(node_list, node_data)
+    pol = self._DiagnoseByOS(valid_nodes, node_data)
      output = []
      for os_name, os_data in pol.iteritems():
        row = []
@@ -1623,22 +1682,7 @@ class LURemoveNode(LogicalUnit):
      self.rpc.call_node_leave_cluster(node.name)
  
      # Promote nodes to master candidate as needed
-    cp_size = self.cfg.GetClusterInfo().candidate_pool_size
-    node_info = self.cfg.GetAllNodesInfo().values()
-    num_candidates = len([n for n in node_info
-                          if n.master_candidate])
-    num_nodes = len(node_info)
-    random.shuffle(node_info)
-    for node in node_info:
-      if num_candidates >= cp_size or num_candidates >= num_nodes:
-        break
-      if node.master_candidate:
-        continue
-      node.master_candidate = True
-      self.LogInfo("Promoting node %s to master candidate", node.name)
-      self.cfg.Update(node)
-      self.context.ReaddNode(node)
-      num_candidates += 1
+    _AdjustCandidatePool(self)
  
  
  class LUQueryNodes(NoHooksLU):
@@ -1972,10 +2016,8 @@ class LUAddNode(LogicalUnit):
                                     " based ping to noded port")
  
      cp_size = self.cfg.GetClusterInfo().candidate_pool_size
-    node_info = self.cfg.GetAllNodesInfo().values()
-    num_candidates = len([n for n in node_info
-                          if n.master_candidate])
-    master_candidate = num_candidates < cp_size
+    mc_now, _ = self.cfg.GetMasterCandidateStats()
+    master_candidate = mc_now < cp_size
  
      self.new_node = objects.Node(name=node,
                                   primary_ip=primary_ip,
@@ -2099,9 +2141,13 @@ class LUSetNodeParams(LogicalUnit):
      if node_name is None:
        raise errors.OpPrereqError("Invalid node name '%s'" % self.op.node_name)
      self.op.node_name = node_name
-    if not hasattr(self.op, 'master_candidate'):
+    _CheckBooleanOpField(self.op, 'master_candidate')
+    _CheckBooleanOpField(self.op, 'offline')
+    if self.op.master_candidate is None and self.op.offline is None:
        raise errors.OpPrereqError("Please pass at least one modification")
-    self.op.master_candidate = bool(self.op.master_candidate)
+    if self.op.offline == True and self.op.master_candidate == True:
+      raise errors.OpPrereqError("Can't set the node into offline and"
+                                 " master_candidate at the same time")
  
    def ExpandNames(self):
      self.needed_locks = {locking.LEVEL_NODE: self.op.node_name}
@@ -2115,6 +2161,7 @@ class LUSetNodeParams(LogicalUnit):
      env = {
        "OP_TARGET": self.op.node_name,
        "MASTER_CANDIDATE": str(self.op.master_candidate),
+      "OFFLINE": str(self.op.offline),
        }
      nl = [self.cfg.GetMasterNode(),
            self.op.node_name]
@@ -2126,34 +2173,46 @@ class LUSetNodeParams(LogicalUnit):
      This only checks the instance list against the existing names.
  
      """
-    force = self.force = self.op.force
+    node = self.node = self.cfg.GetNodeInfo(self.op.node_name)
  
-    if self.op.master_candidate == False:
+    if ((self.op.master_candidate == False or self.op.offline == True)
+        and node.master_candidate):
+      # we will demote the node from master_candidate
        if self.op.node_name == self.cfg.GetMasterNode():
          raise errors.OpPrereqError("The master node has to be a"
-                                   " master candidate")
+                                   " master candidate and online")
        cp_size = self.cfg.GetClusterInfo().candidate_pool_size
-      node_info = self.cfg.GetAllNodesInfo().values()
-      num_candidates = len([node for node in node_info
-                            if node.master_candidate])
+      num_candidates, _ = self.cfg.GetMasterCandidateStats()
        if num_candidates <= cp_size:
          msg = ("Not enough master candidates (desired"
                 " %d, new value will be %d)" % (cp_size, num_candidates-1))
-        if force:
+        if self.op.force:
            self.LogWarning(msg)
          else:
            raise errors.OpPrereqError(msg)
  
+    if (self.op.master_candidate == True and node.offline and
+        not self.op.offline == False):
+      raise errors.OpPrereqError("Can't set an offline node to"
+                                 " master_candidate")
+
      return
  
    def Exec(self, feedback_fn):
      """Modifies a node.
  
      """
-    node = self.cfg.GetNodeInfo(self.op.node_name)
+    node = self.node
  
      result = []
  
+    if self.op.offline is not None:
+      node.offline = self.op.offline
+      result.append(("offline", str(self.op.offline)))
+      if self.op.offline == True and node.master_candidate:
+        node.master_candidate = False
+        result.append(("master_candidate", "auto-demotion due to offline"))
+
      if self.op.master_candidate is not None:
        node.master_candidate = self.op.master_candidate
        result.append(("master_candidate", str(self.op.master_candidate)))
@@ -2279,6 +2338,7 @@ class LUActivateInstanceDisks(NoHooksLU):
      self.instance = self.cfg.GetInstanceInfo(self.op.instance_name)
      assert self.instance is not None, \
        "Cannot retrieve locked instance %s" % self.op.instance_name
+    _CheckNodeOnline(self, self.instance.primary_node)
  
    def Exec(self, feedback_fn):
      """Activate the disks.
@@ -2346,7 +2406,7 @@ def _AssembleInstanceDisks(lu, instance, ignore_secondaries=False):
                             " (is_primary=True, pass=2)",
                             inst_disk.iv_name, node)
          disks_ok = False
-    device_info.append((instance.primary_node, inst_disk.iv_name, result))
+    device_info.append((instance.primary_node, inst_disk.iv_name, result.data))
  
    # leave the disks configured for the primary node
    # this is a workaround that would be fixed better by
@@ -2449,7 +2509,7 @@ def _ShutdownInstanceDisks(lu, instance, ignore_primary=False):
    return result
  
  
-def _CheckNodeFreeMemory(lu, node, reason, requested, hypervisor):
+def _CheckNodeFreeMemory(lu, node, reason, requested, hypervisor_name):
    """Checks if a node has enough free memory.
  
    This function check if a given node has the needed amount of free
@@ -2465,13 +2525,13 @@ def _CheckNodeFreeMemory(lu, node, reason, requested, hypervisor):
    @param reason: string to use in the error message
    @type requested: C{int}
    @param requested: the amount of memory in MiB to check for
-  @type hypervisor: C{str}
-  @param hypervisor: the hypervisor to ask for memory stats
+  @type hypervisor_name: C{str}
+  @param hypervisor_name: the hypervisor to ask for memory stats
    @raise errors.OpPrereqError: if the node doesn't have enough memory, or
        we cannot check the node
  
    """
-  nodeinfo = lu.rpc.call_node_info([node], lu.cfg.GetVGName(), hypervisor)
+  nodeinfo = lu.rpc.call_node_info([node], lu.cfg.GetVGName(), hypervisor_name)
    nodeinfo[node].Raise()
    free_mem = nodeinfo[node].data.get('memory_free')
    if not isinstance(free_mem, int):
@@ -2519,6 +2579,8 @@ class LUStartupInstance(LogicalUnit):
      assert self.instance is not None, \
        "Cannot retrieve locked instance %s" % self.op.instance_name
  
+    _CheckNodeOnline(self, instance.primary_node)
+
      bep = self.cfg.GetClusterInfo().FillBE(instance)
      # check bridges existance
      _CheckInstanceBridgesExist(self, instance)
@@ -2590,6 +2652,8 @@ class LURebootInstance(LogicalUnit):
      assert self.instance is not None, \
        "Cannot retrieve locked instance %s" % self.op.instance_name
  
+    _CheckNodeOnline(self, instance.primary_node)
+
      # check bridges existance
      _CheckInstanceBridgesExist(self, instance)
  
@@ -2655,6 +2719,7 @@ class LUShutdownInstance(LogicalUnit):
      self.instance = self.cfg.GetInstanceInfo(self.op.instance_name)
      assert self.instance is not None, \
        "Cannot retrieve locked instance %s" % self.op.instance_name
+    _CheckNodeOnline(self, self.instance.primary_node)
  
    def Exec(self, feedback_fn):
      """Shutdown the instance.
@@ -2702,6 +2767,7 @@ class LUReinstallInstance(LogicalUnit):
      instance = self.cfg.GetInstanceInfo(self.op.instance_name)
      assert instance is not None, \
        "Cannot retrieve locked instance %s" % self.op.instance_name
+    _CheckNodeOnline(self, instance.primary_node)
  
      if instance.disk_template == constants.DT_DISKLESS:
        raise errors.OpPrereqError("Instance '%s' has no disks" %
@@ -2788,6 +2854,8 @@ class LURenameInstance(LogicalUnit):
      if instance is None:
        raise errors.OpPrereqError("Instance '%s' not known" %
                                   self.op.instance_name)
+    _CheckNodeOnline(self, instance.primary_node)
+
      if instance.status != "down":
        raise errors.OpPrereqError("Instance '%s' is marked to be up" %
                                   self.op.instance_name)
@@ -3214,6 +3282,7 @@ class LUFailoverInstance(LogicalUnit):
                                     "a mirrored disk template")
  
      target_node = secondary_nodes[0]
+    _CheckNodeOnline(self, target_node)
      # check memory requirements on the secondary node
      _CheckNodeFreeMemory(self, target_node, "failing over instance %s" %
                           instance.name, bep[constants.BE_MEMORY],
@@ -3856,6 +3925,7 @@ class LUCreateInstance(LogicalUnit):
            raise errors.OpPrereqError("No export found for relative path %s" %
                                        src_path)
  
+      _CheckNodeOnline(self, src_node)
        result = self.rpc.call_export_info(src_node, src_path)
        result.Raise()
        if not result.data:
@@ -3922,6 +3992,10 @@ class LUCreateInstance(LogicalUnit):
      self.pnode = pnode = self.cfg.GetNodeInfo(self.op.pnode)
      assert self.pnode is not None, \
        "Cannot retrieve locked node %s" % self.op.pnode
+    if pnode.offline:
+      raise errors.OpPrereqError("Cannot use offline primary node '%s'" %
+                                 pnode.name)
+
      self.secondaries = []
  
      # mirror node verification
@@ -3933,6 +4007,7 @@ class LUCreateInstance(LogicalUnit):
          raise errors.OpPrereqError("The secondary node cannot be"
                                     " the primary node.")
        self.secondaries.append(self.op.snode)
+      _CheckNodeOnline(self, self.op.snode)
  
      nodenames = [pnode.name] + self.secondaries
  
@@ -4148,6 +4223,7 @@ class LUConnectConsole(NoHooksLU):
      self.instance = self.cfg.GetInstanceInfo(self.op.instance_name)
      assert self.instance is not None, \
        "Cannot retrieve locked instance %s" % self.op.instance_name
+    _CheckNodeOnline(self, self.instance.primary_node)
  
    def Exec(self, feedback_fn):
      """Connect to the console of an instance
@@ -4181,17 +4257,32 @@ class LUReplaceDisks(LogicalUnit):
    _OP_REQP = ["instance_name", "mode", "disks"]
    REQ_BGL = False
  
-  def ExpandNames(self):
-    self._ExpandAndLockInstance()
-
+  def CheckArguments(self):
      if not hasattr(self.op, "remote_node"):
        self.op.remote_node = None
-
-    ia_name = getattr(self.op, "iallocator", None)
-    if ia_name is not None:
-      if self.op.remote_node is not None:
+    if not hasattr(self.op, "iallocator"):
+      self.op.iallocator = None
+
+    # check for valid parameter combination
+    cnt = [self.op.remote_node, self.op.iallocator].count(None)
+    if self.op.mode == constants.REPLACE_DISK_CHG:
+      if cnt == 2:
+        raise errors.OpPrereqError("When changing the secondary either an"
+                                   " iallocator script must be used or the"
+                                   " new node given")
+      elif cnt == 0:
          raise errors.OpPrereqError("Give either the iallocator or the new"
                                     " secondary, not both")
+    else: # not replacing the secondary
+      if cnt != 2:
+        raise errors.OpPrereqError("The iallocator and new node options can"
+                                   " be used only when changing the"
+                                   " secondary node")
+
+  def ExpandNames(self):
+    self._ExpandAndLockInstance()
+
+    if self.op.iallocator is not None:
        self.needed_locks[locking.LEVEL_NODE] = locking.ALL_SET
      elif self.op.remote_node is not None:
        remote_node = self.cfg.ExpandNodeName(self.op.remote_node)
@@ -4266,9 +4357,9 @@ class LUReplaceDisks(LogicalUnit):
        "Cannot retrieve locked instance %s" % self.op.instance_name
      self.instance = instance
  
-    if instance.disk_template not in constants.DTS_NET_MIRROR:
-      raise errors.OpPrereqError("Instance's disk layout is not"
-                                 " network mirrored.")
+    if instance.disk_template != constants.DT_DRBD8:
+      raise errors.OpPrereqError("Can only run replace disks for DRBD8-based"
+                                 " instances")
  
      if len(instance.secondary_nodes) != 1:
        raise errors.OpPrereqError("The instance has a strange layout,"
@@ -4277,8 +4368,7 @@ class LUReplaceDisks(LogicalUnit):
  
      self.sec_node = instance.secondary_nodes[0]
  
-    ia_name = getattr(self.op, "iallocator", None)
-    if ia_name is not None:
+    if self.op.iallocator is not None:
        self._RunAllocator()
  
      remote_node = self.op.remote_node
@@ -4292,35 +4382,24 @@ class LUReplaceDisks(LogicalUnit):
        raise errors.OpPrereqError("The specified node is the primary node of"
                                   " the instance.")
      elif remote_node == self.sec_node:
-      if self.op.mode == constants.REPLACE_DISK_SEC:
-        # this is for DRBD8, where we can't execute the same mode of
-        # replacement as for drbd7 (no different port allocated)
-        raise errors.OpPrereqError("Same secondary given, cannot execute"
-                                   " replacement")
-    if instance.disk_template == constants.DT_DRBD8:
-      if (self.op.mode == constants.REPLACE_DISK_ALL and
-          remote_node is not None):
-        # switch to replace secondary mode
-        self.op.mode = constants.REPLACE_DISK_SEC
-
-      if self.op.mode == constants.REPLACE_DISK_ALL:
-        raise errors.OpPrereqError("Template 'drbd' only allows primary or"
-                                   " secondary disk replacement, not"
-                                   " both at once")
-      elif self.op.mode == constants.REPLACE_DISK_PRI:
-        if remote_node is not None:
-          raise errors.OpPrereqError("Template 'drbd' does not allow changing"
-                                     " the secondary while doing a primary"
-                                     " node disk replacement")
-        self.tgt_node = instance.primary_node
-        self.oth_node = instance.secondary_nodes[0]
-      elif self.op.mode == constants.REPLACE_DISK_SEC:
-        self.new_node = remote_node # this can be None, in which case
-                                    # we don't change the secondary
-        self.tgt_node = instance.secondary_nodes[0]
-        self.oth_node = instance.primary_node
-      else:
-        raise errors.ProgrammerError("Unhandled disk replace mode")
+      raise errors.OpPrereqError("The specified node is already the"
+                                 " secondary node of the instance.")
+
+    if self.op.mode == constants.REPLACE_DISK_PRI:
+      n1 = self.tgt_node = instance.primary_node
+      n2 = self.oth_node = self.sec_node
+    elif self.op.mode == constants.REPLACE_DISK_SEC:
+      n1 = self.tgt_node = self.sec_node
+      n2 = self.oth_node = instance.primary_node
+    elif self.op.mode == constants.REPLACE_DISK_CHG:
+      n1 = self.new_node = remote_node
+      n2 = self.oth_node = instance.primary_node
+      self.tgt_node = self.sec_node
+    else:
+      raise errors.ProgrammerError("Unhandled disk replace mode")
+
+    _CheckNodeOnline(self, n1)
+    _CheckNodeOnline(self, n2)
  
      if not self.op.disks:
        self.op.disks = range(len(instance.disks))
@@ -4475,7 +4554,7 @@ class LUReplaceDisks(LogicalUnit):
  
        # now that the new lvs have the old name, we can add them to the device
        info("adding new mirror component on %s" % tgt_node)
-      result =self.rpc.call_blockdev_addchildren(tgt_node, dev, new_lvs)
+      result = self.rpc.call_blockdev_addchildren(tgt_node, dev, new_lvs)
        if result.failed or not result.data:
          for new_lv in new_lvs:
            result = self.rpc.call_blockdev_remove(tgt_node, new_lv)
@@ -4536,7 +4615,6 @@ class LUReplaceDisks(LogicalUnit):
      warning, info = (self.proc.LogWarning, self.proc.LogInfo)
      instance = self.instance
      iv_names = {}
-    vgname = self.cfg.GetVGName()
      # start of work
      cfg = self.cfg
      old_node = self.tgt_node
@@ -4578,7 +4656,6 @@ class LUReplaceDisks(LogicalUnit):
      # Step: create new storage
      self.proc.LogStep(3, steps_total, "allocate new storage")
      for idx, dev in enumerate(instance.disks):
-      size = dev.size
        info("adding new local storage on %s for disk/%d" %
             (new_node, idx))
        # since we *always* want to create this LV, we use the
@@ -4666,7 +4743,6 @@ class LUReplaceDisks(LogicalUnit):
  
      # and now perform the drbd attach
      info("attaching primary drbds to new secondary (standalone => connected)")
-    failures = []
      for idx, dev in enumerate(instance.disks):
        info("attaching primary drbd for disk/%d to new secondary node" % idx)
        # since the attach is smart, it's enough to 'find' the device,
@@ -4715,13 +4791,10 @@ class LUReplaceDisks(LogicalUnit):
      if instance.status == "down":
        _StartInstanceDisks(self, instance, True)
  
-    if instance.disk_template == constants.DT_DRBD8:
-      if self.op.remote_node is None:
-        fn = self._ExecD8DiskOnly
-      else:
-        fn = self._ExecD8Secondary
+    if self.op.mode == constants.REPLACE_DISK_CHG:
+      fn = self._ExecD8Secondary
      else:
-      raise errors.ProgrammerError("Unhandled disk replacement case")
+      fn = self._ExecD8DiskOnly
  
      ret = fn(feedback_fn)
  
@@ -4776,6 +4849,10 @@ class LUGrowDisk(LogicalUnit):
      instance = self.cfg.GetInstanceInfo(self.op.instance_name)
      assert instance is not None, \
        "Cannot retrieve locked instance %s" % self.op.instance_name
+    _CheckNodeOnline(self, instance.primary_node)
+    for node in instance.secondary_nodes:
+      _CheckNodeOnline(self, node)
+
  
      self.instance = instance
  
@@ -5445,6 +5522,7 @@ class LUExportInstance(LogicalUnit):
      self.instance = self.cfg.GetInstanceInfo(instance_name)
      assert self.instance is not None, \
            "Cannot retrieve locked instance %s" % self.op.instance_name
+    _CheckNodeOnline(self, self.instance.primary_node)
  
      self.dst_node = self.cfg.GetNodeInfo(
        self.cfg.ExpandNodeName(self.op.target_node))
@@ -5452,6 +5530,7 @@ class LUExportInstance(LogicalUnit):
      if self.dst_node is None:
        # This is wrong node name, not a non-locked node
        raise errors.OpPrereqError("Wrong node name %s" % self.op.target_node)
+    _CheckNodeOnline(self, self.op.target_node)
  
      # instance disk type verification
      for disk in self.instance.disks:
@@ -5832,6 +5911,7 @@ class IAllocator(object):
      self.name = name
      self.mem_size = self.disks = self.disk_template = None
      self.os = self.tags = self.nics = self.vcpus = None
+    self.hypervisor = None
      self.relocate_from = None
      # computed fields
      self.required_nodes = None
@@ -5879,12 +5959,12 @@ class IAllocator(object):
      node_list = cfg.GetNodeList()
  
      if self.mode == constants.IALLOCATOR_MODE_ALLOC:
-      hypervisor = self.hypervisor
+      hypervisor_name = self.hypervisor
      elif self.mode == constants.IALLOCATOR_MODE_RELOC:
-      hypervisor = cfg.GetInstanceInfo(self.name).hypervisor
+      hypervisor_name = cfg.GetInstanceInfo(self.name).hypervisor
  
      node_data = self.lu.rpc.call_node_info(node_list, cfg.GetVGName(),
-                                           hypervisor)
+                                           hypervisor_name)
      node_iinfo = self.lu.rpc.call_all_instances_info(node_list,
                         cluster_info.enabled_hypervisors)
      for nname in node_list: