Add default instance shutdown timeout constant

[ganeti-local] / lib / cmdlib.py
diff --git a/lib/cmdlib.py b/lib/cmdlib.py

index 089cd36..0023cac 100644 (file)
--- a/lib/cmdlib.py
+++ b/lib/cmdlib.py
@@ -670,22 +670,33 @@ def _BuildInstanceHookEnvByObject(lu, instance, override=None):
    return _BuildInstanceHookEnv(**args)
  
  
    return _BuildInstanceHookEnv(**args)
  
  
-def _AdjustCandidatePool(lu):
+def _AdjustCandidatePool(lu, exceptions):
    """Adjust the candidate pool after node operations.
  
    """
    """Adjust the candidate pool after node operations.
  
    """
-  mod_list = lu.cfg.MaintainCandidatePool()
+  mod_list = lu.cfg.MaintainCandidatePool(exceptions)
    if mod_list:
      lu.LogInfo("Promoted nodes to master candidate role: %s",
                 ", ".join(node.name for node in mod_list))
      for name in mod_list:
        lu.context.ReaddNode(name)
    if mod_list:
      lu.LogInfo("Promoted nodes to master candidate role: %s",
                 ", ".join(node.name for node in mod_list))
      for name in mod_list:
        lu.context.ReaddNode(name)
-  mc_now, mc_max = lu.cfg.GetMasterCandidateStats()
+  mc_now, mc_max, _ = lu.cfg.GetMasterCandidateStats(exceptions)
    if mc_now > mc_max:
      lu.LogInfo("Note: more nodes are candidates (%d) than desired (%d)" %
                 (mc_now, mc_max))
  
  
    if mc_now > mc_max:
      lu.LogInfo("Note: more nodes are candidates (%d) than desired (%d)" %
                 (mc_now, mc_max))
  
  
+def _DecideSelfPromotion(lu, exceptions=None):
+  """Decide whether I should promote myself as a master candidate.
+
+  """
+  cp_size = lu.cfg.GetClusterInfo().candidate_pool_size
+  mc_now, mc_should, _ = lu.cfg.GetMasterCandidateStats(exceptions)
+  # the new node will increase mc_max with one, so:
+  mc_should = min(mc_should + 1, cp_size)
+  return mc_now < mc_should
+
+
  def _CheckNicsBridgesExist(lu, target_nics, target_node,
                                 profile=constants.PP_DEFAULT):
    """Check that the brigdes needed by a list of nics exist.
  def _CheckNicsBridgesExist(lu, target_nics, target_node,
                                 profile=constants.PP_DEFAULT):
    """Check that the brigdes needed by a list of nics exist.
@@ -711,6 +722,26 @@ def _CheckInstanceBridgesExist(lu, instance, node=None):
    _CheckNicsBridgesExist(lu, instance.nics, node)
  
  
    _CheckNicsBridgesExist(lu, instance.nics, node)
  
  
+def _CheckOSVariant(os, name):
+  """Check whether an OS name conforms to the os variants specification.
+
+  @type os: L{objects.OS}
+  @param os: OS object to check
+  @type name: string
+  @param name: OS name passed by the user, to check for validity
+
+  """
+  if not os.supported_variants:
+    return
+  try:
+    variant = name.split("+", 1)[1]
+  except IndexError:
+    raise errors.OpPrereqError("OS name must include a variant")
+
+  if variant not in os.supported_variants:
+    raise errors.OpPrereqError("Unsupported OS variant")
+
+
  def _GetNodeInstancesInner(cfg, fn):
    return [i for i in cfg.GetAllInstancesInfo().values() if fn(i)]
  
  def _GetNodeInstancesInner(cfg, fn):
    return [i for i in cfg.GetAllInstancesInfo().values() if fn(i)]
  
@@ -797,12 +828,21 @@ class LUPostInitCluster(LogicalUnit):
      return True
  
  
      return True
  
  
-class LUDestroyCluster(NoHooksLU):
+class LUDestroyCluster(LogicalUnit):
    """Logical unit for destroying the cluster.
  
    """
    """Logical unit for destroying the cluster.
  
    """
+  HPATH = "cluster-destroy"
+  HTYPE = constants.HTYPE_CLUSTER
    _OP_REQP = []
  
    _OP_REQP = []
  
+  def BuildHooksEnv(self):
+    """Build hooks env.
+
+    """
+    env = {"OP_TARGET": self.cfg.GetClusterName()}
+    return env, [], []
+
    def CheckPrereq(self):
      """Check prerequisites.
  
    def CheckPrereq(self):
      """Check prerequisites.
  
@@ -827,6 +867,14 @@ class LUDestroyCluster(NoHooksLU):
  
      """
      master = self.cfg.GetMasterNode()
  
      """
      master = self.cfg.GetMasterNode()
+
+    # Run post hooks on master node before it's removed
+    hm = self.proc.hmclass(self.rpc.call_hooks_runner, self)
+    try:
+      hm.RunPhase(constants.HOOKS_PHASE_POST, [master])
+    except:
+      self.LogWarning("Errors occurred running hooks on %s" % master)
+
      result = self.rpc.call_node_stop_master(master, False)
      result.Raise("Could not disable the master role")
      priv_key, pub_key, _ = ssh.GetUserFiles(constants.GANETI_RUNAS)
      result = self.rpc.call_node_stop_master(master, False)
      result.Raise("Could not disable the master role")
      priv_key, pub_key, _ = ssh.GetUserFiles(constants.GANETI_RUNAS)
@@ -841,9 +889,37 @@ class LUVerifyCluster(LogicalUnit):
    """
    HPATH = "cluster-verify"
    HTYPE = constants.HTYPE_CLUSTER
    """
    HPATH = "cluster-verify"
    HTYPE = constants.HTYPE_CLUSTER
-  _OP_REQP = ["skip_checks"]
+  _OP_REQP = ["skip_checks", "verbose", "error_codes", "debug_simulate_errors"]
    REQ_BGL = False
  
    REQ_BGL = False
  
+  TCLUSTER = "cluster"
+  TNODE = "node"
+  TINSTANCE = "instance"
+
+  ECLUSTERCFG = (TCLUSTER, "ECLUSTERCFG")
+  EINSTANCEBADNODE = (TINSTANCE, "EINSTANCEBADNODE")
+  EINSTANCEDOWN = (TINSTANCE, "EINSTANCEDOWN")
+  EINSTANCELAYOUT = (TINSTANCE, "EINSTANCELAYOUT")
+  EINSTANCEMISSINGDISK = (TINSTANCE, "EINSTANCEMISSINGDISK")
+  EINSTANCEMISSINGDISK = (TINSTANCE, "EINSTANCEMISSINGDISK")
+  EINSTANCEWRONGNODE = (TINSTANCE, "EINSTANCEWRONGNODE")
+  ENODEDRBD = (TNODE, "ENODEDRBD")
+  ENODEFILECHECK = (TNODE, "ENODEFILECHECK")
+  ENODEHOOKS = (TNODE, "ENODEHOOKS")
+  ENODEHV = (TNODE, "ENODEHV")
+  ENODELVM = (TNODE, "ENODELVM")
+  ENODEN1 = (TNODE, "ENODEN1")
+  ENODENET = (TNODE, "ENODENET")
+  ENODEORPHANINSTANCE = (TNODE, "ENODEORPHANINSTANCE")
+  ENODEORPHANLV = (TNODE, "ENODEORPHANLV")
+  ENODERPC = (TNODE, "ENODERPC")
+  ENODESSH = (TNODE, "ENODESSH")
+  ENODEVERSION = (TNODE, "ENODEVERSION")
+
+  ETYPE_FIELD = "code"
+  ETYPE_ERROR = "ERROR"
+  ETYPE_WARNING = "WARNING"
+
    def ExpandNames(self):
      self.needed_locks = {
        locking.LEVEL_NODE: locking.ALL_SET,
    def ExpandNames(self):
      self.needed_locks = {
        locking.LEVEL_NODE: locking.ALL_SET,
@@ -851,9 +927,45 @@ class LUVerifyCluster(LogicalUnit):
      }
      self.share_locks = dict.fromkeys(locking.LEVELS, 1)
  
      }
      self.share_locks = dict.fromkeys(locking.LEVELS, 1)
  
+  def _Error(self, ecode, item, msg, *args, **kwargs):
+    """Format an error message.
+
+    Based on the opcode's error_codes parameter, either format a
+    parseable error code, or a simpler error string.
+
+    This must be called only from Exec and functions called from Exec.
+
+    """
+    ltype = kwargs.get(self.ETYPE_FIELD, self.ETYPE_ERROR)
+    itype, etxt = ecode
+    # first complete the msg
+    if args:
+      msg = msg % args
+    # then format the whole message
+    if self.op.error_codes:
+      msg = "%s:%s:%s:%s:%s" % (ltype, etxt, itype, item, msg)
+    else:
+      if item:
+        item = " " + item
+      else:
+        item = ""
+      msg = "%s: %s%s: %s" % (ltype, itype, item, msg)
+    # and finally report it via the feedback_fn
+    self._feedback_fn("  - %s" % msg)
+
+  def _ErrorIf(self, cond, *args, **kwargs):
+    """Log an error message if the passed condition is True.
+
+    """
+    cond = bool(cond) or self.op.debug_simulate_errors
+    if cond:
+      self._Error(*args, **kwargs)
+    # do not mark the operation as failed for WARN cases only
+    if kwargs.get(self.ETYPE_FIELD, self.ETYPE_ERROR) == self.ETYPE_ERROR:
+      self.bad = self.bad or cond
+
    def _VerifyNode(self, nodeinfo, file_list, local_cksum,
    def _VerifyNode(self, nodeinfo, file_list, local_cksum,
-                  node_result, feedback_fn, master_files,
-                  drbd_map, vg_name):
+                  node_result, master_files, drbd_map, vg_name):
      """Run multiple tests against a node.
  
      Test list:
      """Run multiple tests against a node.
  
      Test list:
@@ -868,7 +980,6 @@ class LUVerifyCluster(LogicalUnit):
      @param file_list: required list of files
      @param local_cksum: dictionary of local files and their checksums
      @param node_result: the results from the node
      @param file_list: required list of files
      @param local_cksum: dictionary of local files and their checksums
      @param node_result: the results from the node
-    @param feedback_fn: function used to accumulate results
      @param master_files: list of files that only masters should have
      @param drbd_map: the useddrbd minors for this node, in
          form of minor: (instance, must_exist) which correspond to instances
      @param master_files: list of files that only masters should have
      @param drbd_map: the useddrbd minors for this node, in
          form of minor: (instance, must_exist) which correspond to instances
@@ -877,138 +988,136 @@ class LUVerifyCluster(LogicalUnit):
  
      """
      node = nodeinfo.name
  
      """
      node = nodeinfo.name
+    _ErrorIf = self._ErrorIf
  
      # main result, node_result should be a non-empty dict
  
      # main result, node_result should be a non-empty dict
-    if not node_result or not isinstance(node_result, dict):
-      feedback_fn("  - ERROR: unable to verify node %s." % (node,))
-      return True
+    test = not node_result or not isinstance(node_result, dict)
+    _ErrorIf(test, self.ENODERPC, node,
+                  "unable to verify node: no data returned")
+    if test:
+      return
  
      # compares ganeti version
      local_version = constants.PROTOCOL_VERSION
      remote_version = node_result.get('version', None)
  
      # compares ganeti version
      local_version = constants.PROTOCOL_VERSION
      remote_version = node_result.get('version', None)
-    if not (remote_version and isinstance(remote_version, (list, tuple)) and
-            len(remote_version) == 2):
-      feedback_fn("  - ERROR: connection to %s failed" % (node))
-      return True
+    test = not (remote_version and
+                isinstance(remote_version, (list, tuple)) and
+                len(remote_version) == 2)
+    _ErrorIf(test, self.ENODERPC, node,
+             "connection to node returned invalid data")
+    if test:
+      return
  
  
-    if local_version != remote_version[0]:
-      feedback_fn("  - ERROR: incompatible protocol versions: master %s,"
-                  " node %s %s" % (local_version, node, remote_version[0]))
-      return True
+    test = local_version != remote_version[0]
+    _ErrorIf(test, self.ENODEVERSION, node,
+             "incompatible protocol versions: master %s,"
+             " node %s", local_version, remote_version[0])
+    if test:
+      return
  
      # node seems compatible, we can actually try to look into its results
  
  
      # node seems compatible, we can actually try to look into its results
  
-    bad = False
-
      # full package version
      # full package version
-    if constants.RELEASE_VERSION != remote_version[1]:
-      feedback_fn("  - WARNING: software version mismatch: master %s,"
-                  " node %s %s" %
-                  (constants.RELEASE_VERSION, node, remote_version[1]))
+    self._ErrorIf(constants.RELEASE_VERSION != remote_version[1],
+                  self.ENODEVERSION, node,
+                  "software version mismatch: master %s, node %s",
+                  constants.RELEASE_VERSION, remote_version[1],
+                  code=self.ETYPE_WARNING)
  
      # checks vg existence and size > 20G
      if vg_name is not None:
        vglist = node_result.get(constants.NV_VGLIST, None)
  
      # checks vg existence and size > 20G
      if vg_name is not None:
        vglist = node_result.get(constants.NV_VGLIST, None)
-      if not vglist:
-        feedback_fn("  - ERROR: unable to check volume groups on node %s." %
-                        (node,))
-        bad = True
-      else:
+      test = not vglist
+      _ErrorIf(test, self.ENODELVM, node, "unable to check volume groups")
+      if not test:
          vgstatus = utils.CheckVolumeGroupSize(vglist, vg_name,
                                                constants.MIN_VG_SIZE)
          vgstatus = utils.CheckVolumeGroupSize(vglist, vg_name,
                                                constants.MIN_VG_SIZE)
-        if vgstatus:
-          feedback_fn("  - ERROR: %s on node %s" % (vgstatus, node))
-          bad = True
+        _ErrorIf(vgstatus, self.ENODELVM, node, vgstatus)
  
      # checks config file checksum
  
      remote_cksum = node_result.get(constants.NV_FILELIST, None)
  
      # checks config file checksum
  
      remote_cksum = node_result.get(constants.NV_FILELIST, None)
-    if not isinstance(remote_cksum, dict):
-      bad = True
-      feedback_fn("  - ERROR: node hasn't returned file checksum data")
-    else:
+    test = not isinstance(remote_cksum, dict)
+    _ErrorIf(test, self.ENODEFILECHECK, node,
+             "node hasn't returned file checksum data")
+    if not test:
        for file_name in file_list:
          node_is_mc = nodeinfo.master_candidate
        for file_name in file_list:
          node_is_mc = nodeinfo.master_candidate
-        must_have_file = file_name not in master_files
-        if file_name not in remote_cksum:
-          if node_is_mc or must_have_file:
-            bad = True
-            feedback_fn("  - ERROR: file '%s' missing" % file_name)
-        elif remote_cksum[file_name] != local_cksum[file_name]:
-          if node_is_mc or must_have_file:
-            bad = True
-            feedback_fn("  - ERROR: file '%s' has wrong checksum" % file_name)
-          else:
-            # not candidate and this is not a must-have file
-            bad = True
-            feedback_fn("  - ERROR: file '%s' should not exist on non master"
-                        " candidates (and the file is outdated)" % file_name)
-        else:
-          # all good, except non-master/non-must have combination
-          if not node_is_mc and not must_have_file:
-            feedback_fn("  - ERROR: file '%s' should not exist on non master"
-                        " candidates" % file_name)
+        must_have = (file_name not in master_files) or node_is_mc
+        # missing
+        test1 = file_name not in remote_cksum
+        # invalid checksum
+        test2 = not test1 and remote_cksum[file_name] != local_cksum[file_name]
+        # existing and good
+        test3 = not test1 and remote_cksum[file_name] == local_cksum[file_name]
+        _ErrorIf(test1 and must_have, self.ENODEFILECHECK, node,
+                 "file '%s' missing", file_name)
+        _ErrorIf(test2 and must_have, self.ENODEFILECHECK, node,
+                 "file '%s' has wrong checksum", file_name)
+        # not candidate and this is not a must-have file
+        _ErrorIf(test2 and not must_have, self.ENODEFILECHECK, node,
+                 "file '%s' should not exist on non master"
+                 " candidates (and the file is outdated)", file_name)
+        # all good, except non-master/non-must have combination
+        _ErrorIf(test3 and not must_have, self.ENODEFILECHECK, node,
+                 "file '%s' should not exist"
+                 " on non master candidates", file_name)
  
      # checks ssh to any
  
  
      # checks ssh to any
  
-    if constants.NV_NODELIST not in node_result:
-      bad = True
-      feedback_fn("  - ERROR: node hasn't returned node ssh connectivity data")
-    else:
+    test = constants.NV_NODELIST not in node_result
+    _ErrorIf(test, self.ENODESSH, node,
+             "node hasn't returned node ssh connectivity data")
+    if not test:
        if node_result[constants.NV_NODELIST]:
        if node_result[constants.NV_NODELIST]:
-        bad = True
-        for node in node_result[constants.NV_NODELIST]:
-          feedback_fn("  - ERROR: ssh communication with node '%s': %s" %
-                          (node, node_result[constants.NV_NODELIST][node]))
-
-    if constants.NV_NODENETTEST not in node_result:
-      bad = True
-      feedback_fn("  - ERROR: node hasn't returned node tcp connectivity data")
-    else:
+        for a_node, a_msg in node_result[constants.NV_NODELIST].items():
+          _ErrorIf(True, self.ENODESSH, node,
+                   "ssh communication with node '%s': %s", a_node, a_msg)
+
+    test = constants.NV_NODENETTEST not in node_result
+    _ErrorIf(test, self.ENODENET, node,
+             "node hasn't returned node tcp connectivity data")
+    if not test:
        if node_result[constants.NV_NODENETTEST]:
        if node_result[constants.NV_NODENETTEST]:
-        bad = True
          nlist = utils.NiceSort(node_result[constants.NV_NODENETTEST].keys())
          nlist = utils.NiceSort(node_result[constants.NV_NODENETTEST].keys())
-        for node in nlist:
-          feedback_fn("  - ERROR: tcp communication with node '%s': %s" %
-                          (node, node_result[constants.NV_NODENETTEST][node]))
+        for anode in nlist:
+          _ErrorIf(True, self.ENODENET, node,
+                   "tcp communication with node '%s': %s",
+                   anode, node_result[constants.NV_NODENETTEST][anode])
  
      hyp_result = node_result.get(constants.NV_HYPERVISOR, None)
      if isinstance(hyp_result, dict):
        for hv_name, hv_result in hyp_result.iteritems():
  
      hyp_result = node_result.get(constants.NV_HYPERVISOR, None)
      if isinstance(hyp_result, dict):
        for hv_name, hv_result in hyp_result.iteritems():
-        if hv_result is not None:
-          feedback_fn("  - ERROR: hypervisor %s verify failure: '%s'" %
-                      (hv_name, hv_result))
+        test = hv_result is not None
+        _ErrorIf(test, self.ENODEHV, node,
+                 "hypervisor %s verify failure: '%s'", hv_name, hv_result)
  
      # check used drbd list
      if vg_name is not None:
        used_minors = node_result.get(constants.NV_DRBDLIST, [])
  
      # check used drbd list
      if vg_name is not None:
        used_minors = node_result.get(constants.NV_DRBDLIST, [])
-      if not isinstance(used_minors, (tuple, list)):
-        feedback_fn("  - ERROR: cannot parse drbd status file: %s" %
-                    str(used_minors))
-      else:
+      test = not isinstance(used_minors, (tuple, list))
+      _ErrorIf(test, self.ENODEDRBD, node,
+               "cannot parse drbd status file: %s", str(used_minors))
+      if not test:
          for minor, (iname, must_exist) in drbd_map.items():
          for minor, (iname, must_exist) in drbd_map.items():
-          if minor not in used_minors and must_exist:
-            feedback_fn("  - ERROR: drbd minor %d of instance %s is"
-                        " not active" % (minor, iname))
-            bad = True
+          test = minor not in used_minors and must_exist
+          _ErrorIf(test, self.ENODEDRBD, node,
+                   "drbd minor %d of instance %s is not active",
+                   minor, iname)
          for minor in used_minors:
          for minor in used_minors:
-          if minor not in drbd_map:
-            feedback_fn("  - ERROR: unallocated drbd minor %d is in use" %
-                        minor)
-            bad = True
-
-    return bad
+          test = minor not in drbd_map
+          _ErrorIf(test, self.ENODEDRBD, node,
+                   "unallocated drbd minor %d is in use", minor)
  
    def _VerifyInstance(self, instance, instanceconfig, node_vol_is,
  
    def _VerifyInstance(self, instance, instanceconfig, node_vol_is,
-                      node_instance, feedback_fn, n_offline):
+                      node_instance, n_offline):
      """Verify an instance.
  
      This function checks to see if the required block devices are
      available on the instance's node.
  
      """
      """Verify an instance.
  
      This function checks to see if the required block devices are
      available on the instance's node.
  
      """
-    bad = False
-
+    _ErrorIf = self._ErrorIf
      node_current = instanceconfig.primary_node
  
      node_vol_should = {}
      node_current = instanceconfig.primary_node
  
      node_vol_should = {}
@@ -1019,69 +1128,57 @@ class LUVerifyCluster(LogicalUnit):
          # ignore missing volumes on offline nodes
          continue
        for volume in node_vol_should[node]:
          # ignore missing volumes on offline nodes
          continue
        for volume in node_vol_should[node]:
-        if node not in node_vol_is or volume not in node_vol_is[node]:
-          feedback_fn("  - ERROR: volume %s missing on node %s" %
-                          (volume, node))
-          bad = True
+        test = node not in node_vol_is or volume not in node_vol_is[node]
+        _ErrorIf(test, self.EINSTANCEMISSINGDISK, instance,
+                 "volume %s missing on node %s", volume, node)
  
      if instanceconfig.admin_up:
  
      if instanceconfig.admin_up:
-      if ((node_current not in node_instance or
-          not instance in node_instance[node_current]) and
-          node_current not in n_offline):
-        feedback_fn("  - ERROR: instance %s not running on node %s" %
-                        (instance, node_current))
-        bad = True
+      test = ((node_current not in node_instance or
+               not instance in node_instance[node_current]) and
+              node_current not in n_offline)
+      _ErrorIf(test, self.EINSTANCEDOWN, instance,
+               "instance not running on its primary node %s",
+               node_current)
  
      for node in node_instance:
        if (not node == node_current):
  
      for node in node_instance:
        if (not node == node_current):
-        if instance in node_instance[node]:
-          feedback_fn("  - ERROR: instance %s should not run on node %s" %
-                          (instance, node))
-          bad = True
-
-    return bad
+        test = instance in node_instance[node]
+        _ErrorIf(test, self.EINSTANCEWRONGNODE, instance,
+                 "instance should not run on node %s", node)
  
  
-  def _VerifyOrphanVolumes(self, node_vol_should, node_vol_is, feedback_fn):
+  def _VerifyOrphanVolumes(self, node_vol_should, node_vol_is):
      """Verify if there are any unknown volumes in the cluster.
  
      The .os, .swap and backup volumes are ignored. All other volumes are
      reported as unknown.
  
      """
      """Verify if there are any unknown volumes in the cluster.
  
      The .os, .swap and backup volumes are ignored. All other volumes are
      reported as unknown.
  
      """
-    bad = False
-
      for node in node_vol_is:
        for volume in node_vol_is[node]:
      for node in node_vol_is:
        for volume in node_vol_is[node]:
-        if node not in node_vol_should or volume not in node_vol_should[node]:
-          feedback_fn("  - ERROR: volume %s on node %s should not exist" %
-                      (volume, node))
-          bad = True
-    return bad
+        test = (node not in node_vol_should or
+                volume not in node_vol_should[node])
+        self._ErrorIf(test, self.ENODEORPHANLV, node,
+                      "volume %s is unknown", volume)
  
  
-  def _VerifyOrphanInstances(self, instancelist, node_instance, feedback_fn):
+  def _VerifyOrphanInstances(self, instancelist, node_instance):
      """Verify the list of running instances.
  
      This checks what instances are running but unknown to the cluster.
  
      """
      """Verify the list of running instances.
  
      This checks what instances are running but unknown to the cluster.
  
      """
-    bad = False
      for node in node_instance:
      for node in node_instance:
-      for runninginstance in node_instance[node]:
-        if runninginstance not in instancelist:
-          feedback_fn("  - ERROR: instance %s on node %s should not exist" %
-                          (runninginstance, node))
-          bad = True
-    return bad
-
-  def _VerifyNPlusOneMemory(self, node_info, instance_cfg, feedback_fn):
+      for o_inst in node_instance[node]:
+        test = o_inst not in instancelist
+        self._ErrorIf(test, self.ENODEORPHANINSTANCE, node,
+                      "instance %s on node %s should not exist", o_inst, node)
+
+  def _VerifyNPlusOneMemory(self, node_info, instance_cfg):
      """Verify N+1 Memory Resilience.
  
      Check that if one single node dies we can still start all the instances it
      was primary for.
  
      """
      """Verify N+1 Memory Resilience.
  
      Check that if one single node dies we can still start all the instances it
      was primary for.
  
      """
-    bad = False
-
      for node, nodeinfo in node_info.iteritems():
        # This code checks that every node which is now listed as secondary has
        # enough memory to host all instances it is supposed to should a single
      for node, nodeinfo in node_info.iteritems():
        # This code checks that every node which is now listed as secondary has
        # enough memory to host all instances it is supposed to should a single
@@ -1097,11 +1194,10 @@ class LUVerifyCluster(LogicalUnit):
            bep = self.cfg.GetClusterInfo().FillBE(instance_cfg[instance])
            if bep[constants.BE_AUTO_BALANCE]:
              needed_mem += bep[constants.BE_MEMORY]
            bep = self.cfg.GetClusterInfo().FillBE(instance_cfg[instance])
            if bep[constants.BE_AUTO_BALANCE]:
              needed_mem += bep[constants.BE_MEMORY]
-        if nodeinfo['mfree'] < needed_mem:
-          feedback_fn("  - ERROR: not enough memory on node %s to accommodate"
-                      " failovers should node %s fail" % (node, prinode))
-          bad = True
-    return bad
+        test = nodeinfo['mfree'] < needed_mem
+        self._ErrorIf(test, self.ENODEN1, node,
+                      "not enough memory on to accommodate"
+                      " failovers should peer node %s fail", prinode)
  
    def CheckPrereq(self):
      """Check prerequisites.
  
    def CheckPrereq(self):
      """Check prerequisites.
@@ -1134,10 +1230,13 @@ class LUVerifyCluster(LogicalUnit):
      """Verify integrity of cluster, performing various test on nodes.
  
      """
      """Verify integrity of cluster, performing various test on nodes.
  
      """
-    bad = False
+    self.bad = False
+    _ErrorIf = self._ErrorIf
+    verbose = self.op.verbose
+    self._feedback_fn = feedback_fn
      feedback_fn("* Verifying global settings")
      for msg in self.cfg.VerifyConfig():
      feedback_fn("* Verifying global settings")
      for msg in self.cfg.VerifyConfig():
-      feedback_fn("  - ERROR: %s" % msg)
+      _ErrorIf(True, self.ECLUSTERCFG, None, msg)
  
      vg_name = self.cfg.GetVGName()
      hypervisors = self.cfg.GetClusterInfo().enabled_hypervisors
  
      vg_name = self.cfg.GetVGName()
      hypervisors = self.cfg.GetClusterInfo().enabled_hypervisors
@@ -1190,11 +1289,13 @@ class LUVerifyCluster(LogicalUnit):
      master_node = self.cfg.GetMasterNode()
      all_drbd_map = self.cfg.ComputeDRBDMap()
  
      master_node = self.cfg.GetMasterNode()
      all_drbd_map = self.cfg.ComputeDRBDMap()
  
+    feedback_fn("* Verifying node status")
      for node_i in nodeinfo:
        node = node_i.name
  
        if node_i.offline:
      for node_i in nodeinfo:
        node = node_i.name
  
        if node_i.offline:
-        feedback_fn("* Skipping offline node %s" % (node,))
+        if verbose:
+          feedback_fn("* Skipping offline node %s" % (node,))
          n_offline.append(node)
          continue
  
          n_offline.append(node)
          continue
  
@@ -1207,62 +1308,59 @@ class LUVerifyCluster(LogicalUnit):
          n_drained.append(node)
        else:
          ntype = "regular"
          n_drained.append(node)
        else:
          ntype = "regular"
-      feedback_fn("* Verifying node %s (%s)" % (node, ntype))
+      if verbose:
+        feedback_fn("* Verifying node %s (%s)" % (node, ntype))
  
        msg = all_nvinfo[node].fail_msg
  
        msg = all_nvinfo[node].fail_msg
+      _ErrorIf(msg, self.ENODERPC, node, "while contacting node: %s", msg)
        if msg:
        if msg:
-        feedback_fn("  - ERROR: while contacting node %s: %s" % (node, msg))
-        bad = True
          continue
  
        nresult = all_nvinfo[node].payload
        node_drbd = {}
        for minor, instance in all_drbd_map[node].items():
          continue
  
        nresult = all_nvinfo[node].payload
        node_drbd = {}
        for minor, instance in all_drbd_map[node].items():
-        if instance not in instanceinfo:
-          feedback_fn("  - ERROR: ghost instance '%s' in temporary DRBD map" %
-                      instance)
+        test = instance not in instanceinfo
+        _ErrorIf(test, self.ECLUSTERCFG, None,
+                 "ghost instance '%s' in temporary DRBD map", instance)
            # ghost instance should not be running, but otherwise we
            # don't give double warnings (both ghost instance and
            # unallocated minor in use)
            # ghost instance should not be running, but otherwise we
            # don't give double warnings (both ghost instance and
            # unallocated minor in use)
+        if test:
            node_drbd[minor] = (instance, False)
          else:
            instance = instanceinfo[instance]
            node_drbd[minor] = (instance.name, instance.admin_up)
            node_drbd[minor] = (instance, False)
          else:
            instance = instanceinfo[instance]
            node_drbd[minor] = (instance.name, instance.admin_up)
-      result = self._VerifyNode(node_i, file_names, local_checksums,
-                                nresult, feedback_fn, master_files,
-                                node_drbd, vg_name)
-      bad = bad or result
+      self._VerifyNode(node_i, file_names, local_checksums,
+                       nresult, master_files, node_drbd, vg_name)
  
        lvdata = nresult.get(constants.NV_LVLIST, "Missing LV data")
        if vg_name is None:
          node_volume[node] = {}
        elif isinstance(lvdata, basestring):
  
        lvdata = nresult.get(constants.NV_LVLIST, "Missing LV data")
        if vg_name is None:
          node_volume[node] = {}
        elif isinstance(lvdata, basestring):
-        feedback_fn("  - ERROR: LVM problem on node %s: %s" %
-                    (node, utils.SafeEncode(lvdata)))
-        bad = True
+        _ErrorIf(True, self.ENODELVM, node, "LVM problem on node: %s",
+                 utils.SafeEncode(lvdata))
          node_volume[node] = {}
        elif not isinstance(lvdata, dict):
          node_volume[node] = {}
        elif not isinstance(lvdata, dict):
-        feedback_fn("  - ERROR: connection to %s failed (lvlist)" % (node,))
-        bad = True
+        _ErrorIf(True, self.ENODELVM, node, "rpc call to node failed (lvlist)")
          continue
        else:
          node_volume[node] = lvdata
  
        # node_instance
        idata = nresult.get(constants.NV_INSTANCELIST, None)
          continue
        else:
          node_volume[node] = lvdata
  
        # node_instance
        idata = nresult.get(constants.NV_INSTANCELIST, None)
-      if not isinstance(idata, list):
-        feedback_fn("  - ERROR: connection to %s failed (instancelist)" %
-                    (node,))
-        bad = True
+      test = not isinstance(idata, list)
+      _ErrorIf(test, self.ENODEHV, node,
+               "rpc call to node failed (instancelist)")
+      if test:
          continue
  
        node_instance[node] = idata
  
        # node_info
        nodeinfo = nresult.get(constants.NV_HVINFO, None)
          continue
  
        node_instance[node] = idata
  
        # node_info
        nodeinfo = nresult.get(constants.NV_HVINFO, None)
-      if not isinstance(nodeinfo, dict):
-        feedback_fn("  - ERROR: connection to %s failed (hvinfo)" % (node,))
-        bad = True
+      test = not isinstance(nodeinfo, dict)
+      _ErrorIf(test, self.ENODEHV, node, "rpc call to node failed (hvinfo)")
+      if test:
          continue
  
        try:
          continue
  
        try:
@@ -1280,28 +1378,28 @@ class LUVerifyCluster(LogicalUnit):
          }
          # FIXME: devise a free space model for file based instances as well
          if vg_name is not None:
          }
          # FIXME: devise a free space model for file based instances as well
          if vg_name is not None:
-          if (constants.NV_VGLIST not in nresult or
-              vg_name not in nresult[constants.NV_VGLIST]):
-            feedback_fn("  - ERROR: node %s didn't return data for the"
-                        " volume group '%s' - it is either missing or broken" %
-                        (node, vg_name))
-            bad = True
+          test = (constants.NV_VGLIST not in nresult or
+                  vg_name not in nresult[constants.NV_VGLIST])
+          _ErrorIf(test, self.ENODELVM, node,
+                   "node didn't return data for the volume group '%s'"
+                   " - it is either missing or broken", vg_name)
+          if test:
              continue
            node_info[node]["dfree"] = int(nresult[constants.NV_VGLIST][vg_name])
        except (ValueError, KeyError):
              continue
            node_info[node]["dfree"] = int(nresult[constants.NV_VGLIST][vg_name])
        except (ValueError, KeyError):
-        feedback_fn("  - ERROR: invalid nodeinfo value returned"
-                    " from node %s" % (node,))
-        bad = True
+        _ErrorIf(True, self.ENODERPC, node,
+                 "node returned invalid nodeinfo, check lvm/hypervisor")
          continue
  
      node_vol_should = {}
  
          continue
  
      node_vol_should = {}
  
+    feedback_fn("* Verifying instance status")
      for instance in instancelist:
      for instance in instancelist:
-      feedback_fn("* Verifying instance %s" % instance)
+      if verbose:
+        feedback_fn("* Verifying instance %s" % instance)
        inst_config = instanceinfo[instance]
        inst_config = instanceinfo[instance]
-      result =  self._VerifyInstance(instance, inst_config, node_volume,
-                                     node_instance, feedback_fn, n_offline)
-      bad = bad or result
+      self._VerifyInstance(instance, inst_config, node_volume,
+                           node_instance, n_offline)
        inst_nodes_offline = []
  
        inst_config.MapLVsByNode(node_vol_should)
        inst_nodes_offline = []
  
        inst_config.MapLVsByNode(node_vol_should)
@@ -1309,12 +1407,11 @@ class LUVerifyCluster(LogicalUnit):
        instance_cfg[instance] = inst_config
  
        pnode = inst_config.primary_node
        instance_cfg[instance] = inst_config
  
        pnode = inst_config.primary_node
+      _ErrorIf(pnode not in node_info and pnode not in n_offline,
+               self.ENODERPC, pnode, "instance %s, connection to"
+               " primary node failed", instance)
        if pnode in node_info:
          node_info[pnode]['pinst'].append(instance)
        if pnode in node_info:
          node_info[pnode]['pinst'].append(instance)
-      elif pnode not in n_offline:
-        feedback_fn("  - ERROR: instance %s, connection to primary node"
-                    " %s failed" % (instance, pnode))
-        bad = True
  
        if pnode in n_offline:
          inst_nodes_offline.append(pnode)
  
        if pnode in n_offline:
          inst_nodes_offline.append(pnode)
@@ -1326,46 +1423,42 @@ class LUVerifyCluster(LogicalUnit):
        # FIXME: does not support file-backed instances
        if len(inst_config.secondary_nodes) == 0:
          i_non_redundant.append(instance)
        # FIXME: does not support file-backed instances
        if len(inst_config.secondary_nodes) == 0:
          i_non_redundant.append(instance)
-      elif len(inst_config.secondary_nodes) > 1:
-        feedback_fn("  - WARNING: multiple secondaries for instance %s"
-                    % instance)
+      _ErrorIf(len(inst_config.secondary_nodes) > 1,
+               self.EINSTANCELAYOUT, instance,
+               "instance has multiple secondary nodes", code="WARNING")
  
        if not cluster.FillBE(inst_config)[constants.BE_AUTO_BALANCE]:
          i_non_a_balanced.append(instance)
  
        for snode in inst_config.secondary_nodes:
  
        if not cluster.FillBE(inst_config)[constants.BE_AUTO_BALANCE]:
          i_non_a_balanced.append(instance)
  
        for snode in inst_config.secondary_nodes:
+        _ErrorIf(snode not in node_info and snode not in n_offline,
+                 self.ENODERPC, snode,
+                 "instance %s, connection to secondary node"
+                 "failed", instance)
+
          if snode in node_info:
            node_info[snode]['sinst'].append(instance)
            if pnode not in node_info[snode]['sinst-by-pnode']:
              node_info[snode]['sinst-by-pnode'][pnode] = []
            node_info[snode]['sinst-by-pnode'][pnode].append(instance)
          if snode in node_info:
            node_info[snode]['sinst'].append(instance)
            if pnode not in node_info[snode]['sinst-by-pnode']:
              node_info[snode]['sinst-by-pnode'][pnode] = []
            node_info[snode]['sinst-by-pnode'][pnode].append(instance)
-        elif snode not in n_offline:
-          feedback_fn("  - ERROR: instance %s, connection to secondary node"
-                      " %s failed" % (instance, snode))
-          bad = True
+
          if snode in n_offline:
            inst_nodes_offline.append(snode)
  
          if snode in n_offline:
            inst_nodes_offline.append(snode)
  
-      if inst_nodes_offline:
-        # warn that the instance lives on offline nodes, and set bad=True
-        feedback_fn("  - ERROR: instance lives on offline node(s) %s" %
-                    ", ".join(inst_nodes_offline))
-        bad = True
+      # warn that the instance lives on offline nodes
+      _ErrorIf(inst_nodes_offline, self.EINSTANCEBADNODE, instance,
+               "instance lives on offline node(s) %s",
+               ", ".join(inst_nodes_offline))
  
      feedback_fn("* Verifying orphan volumes")
  
      feedback_fn("* Verifying orphan volumes")
-    result = self._VerifyOrphanVolumes(node_vol_should, node_volume,
-                                       feedback_fn)
-    bad = bad or result
+    self._VerifyOrphanVolumes(node_vol_should, node_volume)
  
      feedback_fn("* Verifying remaining instances")
  
      feedback_fn("* Verifying remaining instances")
-    result = self._VerifyOrphanInstances(instancelist, node_instance,
-                                         feedback_fn)
-    bad = bad or result
+    self._VerifyOrphanInstances(instancelist, node_instance)
  
      if constants.VERIFY_NPLUSONE_MEM not in self.skip_set:
        feedback_fn("* Verifying N+1 Memory redundancy")
  
      if constants.VERIFY_NPLUSONE_MEM not in self.skip_set:
        feedback_fn("* Verifying N+1 Memory redundancy")
-      result = self._VerifyNPlusOneMemory(node_info, instance_cfg, feedback_fn)
-      bad = bad or result
+      self._VerifyNPlusOneMemory(node_info, instance_cfg)
  
      feedback_fn("* Other Notes")
      if i_non_redundant:
  
      feedback_fn("* Other Notes")
      if i_non_redundant:
@@ -1382,7 +1475,7 @@ class LUVerifyCluster(LogicalUnit):
      if n_drained:
        feedback_fn("  - NOTICE: %d drained node(s) found." % len(n_drained))
  
      if n_drained:
        feedback_fn("  - NOTICE: %d drained node(s) found." % len(n_drained))
  
-    return not bad
+    return not self.bad
  
    def HooksCallBack(self, phase, hooks_results, feedback_fn, lu_result):
      """Analyze the post-hooks' result
  
    def HooksCallBack(self, phase, hooks_results, feedback_fn, lu_result):
      """Analyze the post-hooks' result
@@ -1405,33 +1498,28 @@ class LUVerifyCluster(LogicalUnit):
        # Used to change hooks' output to proper indentation
        indent_re = re.compile('^', re.M)
        feedback_fn("* Hooks Results")
        # Used to change hooks' output to proper indentation
        indent_re = re.compile('^', re.M)
        feedback_fn("* Hooks Results")
-      if not hooks_results:
-        feedback_fn("  - ERROR: general communication failure")
-        lu_result = 1
-      else:
-        for node_name in hooks_results:
-          show_node_header = True
-          res = hooks_results[node_name]
-          msg = res.fail_msg
-          if msg:
-            if res.offline:
-              # no need to warn or set fail return value
-              continue
-            feedback_fn("    Communication failure in hooks execution: %s" %
-                        msg)
+      assert hooks_results, "invalid result from hooks"
+
+      for node_name in hooks_results:
+        show_node_header = True
+        res = hooks_results[node_name]
+        msg = res.fail_msg
+        test = msg and not res.offline
+        self._ErrorIf(test, self.ENODEHOOKS, node_name,
+                      "Communication failure in hooks execution: %s", msg)
+        if test:
+          # override manually lu_result here as _ErrorIf only
+          # overrides self.bad
+          lu_result = 1
+          continue
+        for script, hkr, output in res.payload:
+          test = hkr == constants.HKR_FAIL
+          self._ErrorIf(test, self.ENODEHOOKS, node_name,
+                        "Script %s failed, output:", script)
+          if test:
+            output = indent_re.sub('      ', output)
+            feedback_fn("%s" % output)
              lu_result = 1
              lu_result = 1
-            continue
-          for script, hkr, output in res.payload:
-            if hkr == constants.HKR_FAIL:
-              # The node header is only shown once, if there are
-              # failing hooks on that node
-              if show_node_header:
-                feedback_fn("  Node %s:" % node_name)
-                show_node_header = False
-              feedback_fn("    ERROR: Script %s failed, output:" % script)
-              output = indent_re.sub('      ', output)
-              feedback_fn("%s" % output)
-              lu_result = 1
  
        return lu_result
  
  
        return lu_result
  
@@ -1527,7 +1615,6 @@ class LURepairDiskSizes(NoHooksLU):
    REQ_BGL = False
  
    def ExpandNames(self):
    REQ_BGL = False
  
    def ExpandNames(self):
-
      if not isinstance(self.op.instances, list):
        raise errors.OpPrereqError("Invalid argument type 'instances'")
  
      if not isinstance(self.op.instances, list):
        raise errors.OpPrereqError("Invalid argument type 'instances'")
  
@@ -1538,7 +1625,6 @@ class LURepairDiskSizes(NoHooksLU):
          if full_name is None:
            raise errors.OpPrereqError("Instance '%s' not known" % name)
          self.wanted_names.append(full_name)
          if full_name is None:
            raise errors.OpPrereqError("Instance '%s' not known" % name)
          self.wanted_names.append(full_name)
-      self.needed_locks[locking.LEVEL_INSTANCE] = self.wanted_names
        self.needed_locks = {
          locking.LEVEL_NODE: [],
          locking.LEVEL_INSTANCE: self.wanted_names,
        self.needed_locks = {
          locking.LEVEL_NODE: [],
          locking.LEVEL_INSTANCE: self.wanted_names,
@@ -1568,6 +1654,29 @@ class LURepairDiskSizes(NoHooksLU):
      self.wanted_instances = [self.cfg.GetInstanceInfo(name) for name
                               in self.wanted_names]
  
      self.wanted_instances = [self.cfg.GetInstanceInfo(name) for name
                               in self.wanted_names]
  
+  def _EnsureChildSizes(self, disk):
+    """Ensure children of the disk have the needed disk size.
+
+    This is valid mainly for DRBD8 and fixes an issue where the
+    children have smaller disk size.
+
+    @param disk: an L{ganeti.objects.Disk} object
+
+    """
+    if disk.dev_type == constants.LD_DRBD8:
+      assert disk.children, "Empty children for DRBD8?"
+      fchild = disk.children[0]
+      mismatch = fchild.size < disk.size
+      if mismatch:
+        self.LogInfo("Child disk has size %d, parent %d, fixing",
+                     fchild.size, disk.size)
+        fchild.size = disk.size
+
+      # and we recurse on this child only, not on the metadev
+      return self._EnsureChildSizes(fchild) or mismatch
+    else:
+      return False
+
    def Exec(self, feedback_fn):
      """Verify the size of cluster disks.
  
    def Exec(self, feedback_fn):
      """Verify the size of cluster disks.
  
@@ -1584,8 +1693,11 @@ class LURepairDiskSizes(NoHooksLU):
  
      changed = []
      for node, dskl in per_node_disks.items():
  
      changed = []
      for node, dskl in per_node_disks.items():
-      result = self.rpc.call_blockdev_getsizes(node, [v[2] for v in dskl])
-      if result.failed:
+      newl = [v[2].Copy() for v in dskl]
+      for dsk in newl:
+        self.cfg.SetDiskID(dsk, node)
+      result = self.rpc.call_blockdev_getsizes(node, newl)
+      if result.fail_msg:
          self.LogWarning("Failure in blockdev_getsizes call to node"
                          " %s, ignoring", node)
          continue
          self.LogWarning("Failure in blockdev_getsizes call to node"
                          " %s, ignoring", node)
          continue
@@ -1610,6 +1722,9 @@ class LURepairDiskSizes(NoHooksLU):
            disk.size = size
            self.cfg.Update(instance)
            changed.append((instance.name, idx, size))
            disk.size = size
            self.cfg.Update(instance)
            changed.append((instance.name, idx, size))
+        if self._EnsureChildSizes(disk):
+          self.cfg.Update(instance)
+          changed.append((instance.name, idx, disk.size))
      return changed
  
  
      return changed
  
  
@@ -1820,7 +1935,7 @@ class LUSetClusterParams(LogicalUnit):
        invalid_hvs = set(self.hv_list) - constants.HYPER_TYPES
        if invalid_hvs:
          raise errors.OpPrereqError("Enabled hypervisors contains invalid"
        invalid_hvs = set(self.hv_list) - constants.HYPER_TYPES
        if invalid_hvs:
          raise errors.OpPrereqError("Enabled hypervisors contains invalid"
-                                   " entries: %s" % invalid_hvs)
+                                   " entries: %s" % " ,".join(invalid_hvs))
      else:
        self.hv_list = cluster.enabled_hypervisors
  
      else:
        self.hv_list = cluster.enabled_hypervisors
  
@@ -1861,7 +1976,7 @@ class LUSetClusterParams(LogicalUnit):
      if self.op.candidate_pool_size is not None:
        self.cluster.candidate_pool_size = self.op.candidate_pool_size
        # we need to update the pool size here, otherwise the save will fail
      if self.op.candidate_pool_size is not None:
        self.cluster.candidate_pool_size = self.op.candidate_pool_size
        # we need to update the pool size here, otherwise the save will fail
-      _AdjustCandidatePool(self)
+      _AdjustCandidatePool(self, [])
  
      self.cfg.Update(self.cluster)
  
  
      self.cfg.Update(self.cluster)
  
@@ -1986,7 +2101,8 @@ def _WaitForSync(lu, instance, oneshot=False, unlock=False):
          else:
            rem_time = "no time estimate"
          lu.proc.LogInfo("- device %s: %5.2f%% done, %s" %
          else:
            rem_time = "no time estimate"
          lu.proc.LogInfo("- device %s: %5.2f%% done, %s" %
-                        (instance.disks[i].iv_name, mstat.sync_percent, rem_time))
+                        (instance.disks[i].iv_name, mstat.sync_percent,
+                         rem_time))
  
      # if we're done but degraded, let's do a few small retries, to
      # make sure we see a stable and not transient situation; therefore
  
      # if we're done but degraded, let's do a few small retries, to
      # make sure we see a stable and not transient situation; therefore
@@ -2048,7 +2164,9 @@ class LUDiagnoseOS(NoHooksLU):
    _OP_REQP = ["output_fields", "names"]
    REQ_BGL = False
    _FIELDS_STATIC = utils.FieldSet()
    _OP_REQP = ["output_fields", "names"]
    REQ_BGL = False
    _FIELDS_STATIC = utils.FieldSet()
-  _FIELDS_DYNAMIC = utils.FieldSet("name", "valid", "node_status")
+  _FIELDS_DYNAMIC = utils.FieldSet("name", "valid", "node_status", "variants")
+  # Fields that need calculation of global os validity
+  _FIELDS_NEEDVALID = frozenset(["valid", "variants"])
  
    def ExpandNames(self):
      if self.op.names:
  
    def ExpandNames(self):
      if self.op.names:
@@ -2096,14 +2214,14 @@ class LUDiagnoseOS(NoHooksLU):
      for node_name, nr in rlist.items():
        if nr.fail_msg or not nr.payload:
          continue
      for node_name, nr in rlist.items():
        if nr.fail_msg or not nr.payload:
          continue
-      for name, path, status, diagnose in nr.payload:
+      for name, path, status, diagnose, variants in nr.payload:
          if name not in all_os:
            # build a list of nodes for this os containing empty lists
            # for each node in node_list
            all_os[name] = {}
            for nname in good_nodes:
              all_os[name][nname] = []
          if name not in all_os:
            # build a list of nodes for this os containing empty lists
            # for each node in node_list
            all_os[name] = {}
            for nname in good_nodes:
              all_os[name][nname] = []
-        all_os[name][node_name].append((path, status, diagnose))
+        all_os[name][node_name].append((path, status, diagnose, variants))
      return all_os
  
    def Exec(self, feedback_fn):
      return all_os
  
    def Exec(self, feedback_fn):
@@ -2114,18 +2232,38 @@ class LUDiagnoseOS(NoHooksLU):
      node_data = self.rpc.call_os_diagnose(valid_nodes)
      pol = self._DiagnoseByOS(valid_nodes, node_data)
      output = []
      node_data = self.rpc.call_os_diagnose(valid_nodes)
      pol = self._DiagnoseByOS(valid_nodes, node_data)
      output = []
+    calc_valid = self._FIELDS_NEEDVALID.intersection(self.op.output_fields)
+    calc_variants = "variants" in self.op.output_fields
+
      for os_name, os_data in pol.items():
        row = []
      for os_name, os_data in pol.items():
        row = []
+      if calc_valid:
+        valid = True
+        variants = None
+        for osl in os_data.values():
+          valid = valid and osl and osl[0][1]
+          if not valid:
+            variants = None
+            break
+          if calc_variants:
+            node_variants = osl[0][3]
+            if variants is None:
+              variants = node_variants
+            else:
+              variants = [v for v in variants if v in node_variants]
+
        for field in self.op.output_fields:
          if field == "name":
            val = os_name
          elif field == "valid":
        for field in self.op.output_fields:
          if field == "name":
            val = os_name
          elif field == "valid":
-          val = utils.all([osl and osl[0][1] for osl in os_data.values()])
+          val = valid
          elif field == "node_status":
            # this is just a copy of the dict
            val = {}
            for node_name, nos_list in os_data.items():
              val[node_name] = nos_list
          elif field == "node_status":
            # this is just a copy of the dict
            val = {}
            for node_name, nos_list in os_data.items():
              val[node_name] = nos_list
+        elif field == "variants":
+          val =  variants
          else:
            raise errors.ParameterError(field)
          row.append(val)
          else:
            raise errors.ParameterError(field)
          row.append(val)
@@ -2154,7 +2292,8 @@ class LURemoveNode(LogicalUnit):
        "NODE_NAME": self.op.node_name,
        }
      all_nodes = self.cfg.GetNodeList()
        "NODE_NAME": self.op.node_name,
        }
      all_nodes = self.cfg.GetNodeList()
-    all_nodes.remove(self.op.node_name)
+    if self.op.node_name in all_nodes:
+      all_nodes.remove(self.op.node_name)
      return env, all_nodes, all_nodes
  
    def CheckPrereq(self):
      return env, all_nodes, all_nodes
  
    def CheckPrereq(self):
@@ -2195,17 +2334,23 @@ class LURemoveNode(LogicalUnit):
      logging.info("Stopping the node daemon and removing configs from node %s",
                   node.name)
  
      logging.info("Stopping the node daemon and removing configs from node %s",
                   node.name)
  
+    # Promote nodes to master candidate as needed
+    _AdjustCandidatePool(self, exceptions=[node.name])
      self.context.RemoveNode(node.name)
  
      self.context.RemoveNode(node.name)
  
+    # Run post hooks on the node before it's removed
+    hm = self.proc.hmclass(self.rpc.call_hooks_runner, self)
+    try:
+      h_results = hm.RunPhase(constants.HOOKS_PHASE_POST, [node.name])
+    except:
+      self.LogWarning("Errors occurred running hooks on %s" % node.name)
+
      result = self.rpc.call_node_leave_cluster(node.name)
      msg = result.fail_msg
      if msg:
        self.LogWarning("Errors encountered on the remote node while leaving"
                        " the cluster: %s", msg)
  
      result = self.rpc.call_node_leave_cluster(node.name)
      msg = result.fail_msg
      if msg:
        self.LogWarning("Errors encountered on the remote node while leaving"
                        " the cluster: %s", msg)
  
-    # Promote nodes to master candidate as needed
-    _AdjustCandidatePool(self)
-
  
  class LUQueryNodes(NoHooksLU):
    """Logical unit for querying nodes.
  
  class LUQueryNodes(NoHooksLU):
    """Logical unit for querying nodes.
@@ -2213,6 +2358,10 @@ class LUQueryNodes(NoHooksLU):
    """
    _OP_REQP = ["output_fields", "names", "use_locking"]
    REQ_BGL = False
    """
    _OP_REQP = ["output_fields", "names", "use_locking"]
    REQ_BGL = False
+
+  _SIMPLE_FIELDS = ["name", "serial_no", "ctime", "mtime", "uuid",
+                    "master_candidate", "offline", "drained"]
+
    _FIELDS_DYNAMIC = utils.FieldSet(
      "dtotal", "dfree",
      "mtotal", "mnode", "mfree",
    _FIELDS_DYNAMIC = utils.FieldSet(
      "dtotal", "dfree",
      "mtotal", "mnode", "mfree",
@@ -2220,16 +2369,12 @@ class LUQueryNodes(NoHooksLU):
      "ctotal", "cnodes", "csockets",
      )
  
      "ctotal", "cnodes", "csockets",
      )
  
-  _FIELDS_STATIC = utils.FieldSet(
-    "name", "pinst_cnt", "sinst_cnt",
+  _FIELDS_STATIC = utils.FieldSet(*[
+    "pinst_cnt", "sinst_cnt",
      "pinst_list", "sinst_list",
      "pip", "sip", "tags",
      "pinst_list", "sinst_list",
      "pip", "sip", "tags",
-    "serial_no", "ctime", "mtime",
-    "master_candidate",
      "master",
      "master",
-    "offline",
-    "drained",
-    "role",
+    "role"] + _SIMPLE_FIELDS
      )
  
    def ExpandNames(self):
      )
  
    def ExpandNames(self):
@@ -2330,8 +2475,8 @@ class LUQueryNodes(NoHooksLU):
      for node in nodelist:
        node_output = []
        for field in self.op.output_fields:
      for node in nodelist:
        node_output = []
        for field in self.op.output_fields:
-        if field == "name":
-          val = node.name
+        if field in self._SIMPLE_FIELDS:
+          val = getattr(node, field)
          elif field == "pinst_list":
            val = list(node_to_primary[node.name])
          elif field == "sinst_list":
          elif field == "pinst_list":
            val = list(node_to_primary[node.name])
          elif field == "sinst_list":
@@ -2346,20 +2491,8 @@ class LUQueryNodes(NoHooksLU):
            val = node.secondary_ip
          elif field == "tags":
            val = list(node.GetTags())
            val = node.secondary_ip
          elif field == "tags":
            val = list(node.GetTags())
-        elif field == "serial_no":
-          val = node.serial_no
-        elif field == "ctime":
-          val = node.ctime
-        elif field == "mtime":
-          val = node.mtime
-        elif field == "master_candidate":
-          val = node.master_candidate
          elif field == "master":
            val = node.name == master_node
          elif field == "master":
            val = node.name == master_node
-        elif field == "offline":
-          val = node.offline
-        elif field == "drained":
-          val = node.drained
          elif self._FIELDS_DYNAMIC.Matches(field):
            val = live_data[node.name].get(field, None)
          elif field == "role":
          elif self._FIELDS_DYNAMIC.Matches(field):
            val = live_data[node.name].get(field, None)
          elif field == "role":
@@ -2712,15 +2845,12 @@ class LUAddNode(LogicalUnit):
          raise errors.OpPrereqError("Node secondary ip not reachable by TCP"
                                     " based ping to noded port")
  
          raise errors.OpPrereqError("Node secondary ip not reachable by TCP"
                                     " based ping to noded port")
  
-    cp_size = self.cfg.GetClusterInfo().candidate_pool_size
      if self.op.readd:
        exceptions = [node]
      else:
        exceptions = []
      if self.op.readd:
        exceptions = [node]
      else:
        exceptions = []
-    mc_now, mc_max = self.cfg.GetMasterCandidateStats(exceptions)
-    # the new node will increase mc_max with one, so:
-    mc_max = min(mc_max + 1, cp_size)
-    self.master_candidate = mc_now < mc_max
+
+    self.master_candidate = _DecideSelfPromotion(self, exceptions=exceptions)
  
      if self.op.readd:
        self.new_node = self.cfg.GetNodeInfo(node)
  
      if self.op.readd:
        self.new_node = self.cfg.GetNodeInfo(node)
@@ -2773,11 +2903,7 @@ class LUAddNode(LogicalUnit):
                  priv_key, pub_key]
  
      for i in keyfiles:
                  priv_key, pub_key]
  
      for i in keyfiles:
-      f = open(i, 'r')
-      try:
-        keyarray.append(f.read())
-      finally:
-        f.close()
+      keyarray.append(utils.ReadFile(i))
  
      result = self.rpc.call_node_add(node, keyarray[0], keyarray[1],
                                      keyarray[2],
  
      result = self.rpc.call_node_add(node, keyarray[0], keyarray[1],
                                      keyarray[2],
@@ -2800,7 +2926,7 @@ class LUAddNode(LogicalUnit):
  
      node_verify_list = [self.cfg.GetMasterNode()]
      node_verify_param = {
  
      node_verify_list = [self.cfg.GetMasterNode()]
      node_verify_param = {
-      'nodelist': [node],
+      constants.NV_NODELIST: [node],
        # TODO: do a node-net-test as well?
      }
  
        # TODO: do a node-net-test as well?
      }
  
@@ -2808,10 +2934,11 @@ class LUAddNode(LogicalUnit):
                                         self.cfg.GetClusterName())
      for verifier in node_verify_list:
        result[verifier].Raise("Cannot communicate with node %s" % verifier)
                                         self.cfg.GetClusterName())
      for verifier in node_verify_list:
        result[verifier].Raise("Cannot communicate with node %s" % verifier)
-      nl_payload = result[verifier].payload['nodelist']
+      nl_payload = result[verifier].payload[constants.NV_NODELIST]
        if nl_payload:
          for failed in nl_payload:
        if nl_payload:
          for failed in nl_payload:
-          feedback_fn("ssh/hostname verification failed %s -> %s" %
+          feedback_fn("ssh/hostname verification failed"
+                      " (checking from %s): %s" %
                        (verifier, nl_payload[failed]))
          raise errors.OpExecError("ssh/hostname verification failed.")
  
                        (verifier, nl_payload[failed]))
          raise errors.OpExecError("ssh/hostname verification failed.")
  
@@ -2823,7 +2950,7 @@ class LUAddNode(LogicalUnit):
        # and make sure the new node will not have old files around
        if not new_node.master_candidate:
          result = self.rpc.call_node_demote_from_mc(new_node.name)
        # and make sure the new node will not have old files around
        if not new_node.master_candidate:
          result = self.rpc.call_node_demote_from_mc(new_node.name)
-        msg = result.RemoteFailMsg()
+        msg = result.fail_msg
          if msg:
            self.LogWarning("Node failed to demote itself from master"
                            " candidate status: %s" % msg)
          if msg:
            self.LogWarning("Node failed to demote itself from master"
                            " candidate status: %s" % msg)
@@ -2883,18 +3010,30 @@ class LUSetNodeParams(LogicalUnit):
      """
      node = self.node = self.cfg.GetNodeInfo(self.op.node_name)
  
      """
      node = self.node = self.cfg.GetNodeInfo(self.op.node_name)
  
-    if ((self.op.master_candidate == False or self.op.offline == True or
-         self.op.drained == True) and node.master_candidate):
-      # we will demote the node from master_candidate
+    if (self.op.master_candidate is not None or
+        self.op.drained is not None or
+        self.op.offline is not None):
+      # we can't change the master's node flags
        if self.op.node_name == self.cfg.GetMasterNode():
        if self.op.node_name == self.cfg.GetMasterNode():
-        raise errors.OpPrereqError("The master node has to be a"
-                                   " master candidate, online and not drained")
+        raise errors.OpPrereqError("The master role can be changed"
+                                   " only via masterfailover")
+
+    # Boolean value that tells us whether we're offlining or draining the node
+    offline_or_drain = self.op.offline == True or self.op.drained == True
+    deoffline_or_drain = self.op.offline == False or self.op.drained == False
+
+    if (node.master_candidate and
+        (self.op.master_candidate == False or offline_or_drain)):
        cp_size = self.cfg.GetClusterInfo().candidate_pool_size
        cp_size = self.cfg.GetClusterInfo().candidate_pool_size
-      num_candidates, _ = self.cfg.GetMasterCandidateStats()
-      if num_candidates <= cp_size:
+      mc_now, mc_should, mc_max = self.cfg.GetMasterCandidateStats()
+      if mc_now <= cp_size:
          msg = ("Not enough master candidates (desired"
          msg = ("Not enough master candidates (desired"
-               " %d, new value will be %d)" % (cp_size, num_candidates-1))
-        if self.op.force:
+               " %d, new value will be %d)" % (cp_size, mc_now-1))
+        # Only allow forcing the operation if it's an offline/drain operation,
+        # and we could not possibly promote more nodes.
+        # FIXME: this can still lead to issues if in any way another node which
+        # could be promoted appears in the meantime.
+        if self.op.force and offline_or_drain and mc_should == mc_max:
            self.LogWarning(msg)
          else:
            raise errors.OpPrereqError(msg)
            self.LogWarning(msg)
          else:
            raise errors.OpPrereqError(msg)
@@ -2905,6 +3044,13 @@ class LUSetNodeParams(LogicalUnit):
        raise errors.OpPrereqError("Node '%s' is offline or drained, can't set"
                                   " to master_candidate" % node.name)
  
        raise errors.OpPrereqError("Node '%s' is offline or drained, can't set"
                                   " to master_candidate" % node.name)
  
+    # If we're being deofflined/drained, we'll MC ourself if needed
+    if (deoffline_or_drain and not offline_or_drain and not
+        self.op.master_candidate == True):
+      self.op.master_candidate = _DecideSelfPromotion(self)
+      if self.op.master_candidate:
+        self.LogInfo("Autopromoting node to master candidate")
+
      return
  
    def Exec(self, feedback_fn):
      return
  
    def Exec(self, feedback_fn):
@@ -2947,7 +3093,7 @@ class LUSetNodeParams(LogicalUnit):
            changed_mc = True
            result.append(("master_candidate", "auto-demotion due to drain"))
            rrc = self.rpc.call_node_demote_from_mc(node.name)
            changed_mc = True
            result.append(("master_candidate", "auto-demotion due to drain"))
            rrc = self.rpc.call_node_demote_from_mc(node.name)
-          msg = rrc.RemoteFailMsg()
+          msg = rrc.fail_msg
            if msg:
              self.LogWarning("Node failed to demote itself: %s" % msg)
          if node.offline:
            if msg:
              self.LogWarning("Node failed to demote itself: %s" % msg)
          if node.offline:
@@ -3048,6 +3194,8 @@ class LUQueryClusterInfo(NoHooksLU):
        "file_storage_dir": cluster.file_storage_dir,
        "ctime": cluster.ctime,
        "mtime": cluster.mtime,
        "file_storage_dir": cluster.file_storage_dir,
        "ctime": cluster.ctime,
        "mtime": cluster.mtime,
+      "uuid": cluster.uuid,
+      "tags": list(cluster.GetTags()),
        }
  
      return result
        }
  
      return result
@@ -3060,7 +3208,8 @@ class LUQueryConfigValues(NoHooksLU):
    _OP_REQP = []
    REQ_BGL = False
    _FIELDS_DYNAMIC = utils.FieldSet()
    _OP_REQP = []
    REQ_BGL = False
    _FIELDS_DYNAMIC = utils.FieldSet()
-  _FIELDS_STATIC = utils.FieldSet("cluster_name", "master_node", "drain_flag")
+  _FIELDS_STATIC = utils.FieldSet("cluster_name", "master_node", "drain_flag",
+                                  "watcher_pause")
  
    def ExpandNames(self):
      self.needed_locks = {}
  
    def ExpandNames(self):
      self.needed_locks = {}
@@ -3087,6 +3236,8 @@ class LUQueryConfigValues(NoHooksLU):
          entry = self.cfg.GetMasterNode()
        elif field == "drain_flag":
          entry = os.path.exists(constants.JOB_QUEUE_DRAIN_FILE)
          entry = self.cfg.GetMasterNode()
        elif field == "drain_flag":
          entry = os.path.exists(constants.JOB_QUEUE_DRAIN_FILE)
+      elif field == "watcher_pause":
+        return utils.ReadWatcherPauseFile(constants.WATCHER_PAUSEFILE)
        else:
          raise errors.ParameterError(field)
        values.append(entry)
        else:
          raise errors.ParameterError(field)
        values.append(entry)
@@ -3617,6 +3768,7 @@ class LUReinstallInstance(LogicalUnit):
                                    instance.primary_node))
  
      self.op.os_type = getattr(self.op, "os_type", None)
                                    instance.primary_node))
  
      self.op.os_type = getattr(self.op, "os_type", None)
+    self.op.force_variant = getattr(self.op, "force_variant", False)
      if self.op.os_type is not None:
        # OS verification
        pnode = self.cfg.GetNodeInfo(
      if self.op.os_type is not None:
        # OS verification
        pnode = self.cfg.GetNodeInfo(
@@ -3627,6 +3779,8 @@ class LUReinstallInstance(LogicalUnit):
        result = self.rpc.call_os_get(pnode.name, self.op.os_type)
        result.Raise("OS '%s' not in supported OS list for primary node %s" %
                     (self.op.os_type, pnode.name), prereq=True)
        result = self.rpc.call_os_get(pnode.name, self.op.os_type)
        result.Raise("OS '%s' not in supported OS list for primary node %s" %
                     (self.op.os_type, pnode.name), prereq=True)
+      if not self.op.force_variant:
+        _CheckOSVariant(result.payload, self.op.os_type)
  
      self.instance = instance
  
  
      self.instance = instance
  
@@ -3913,6 +4067,8 @@ class LUQueryInstances(NoHooksLU):
    """
    _OP_REQP = ["output_fields", "names", "use_locking"]
    REQ_BGL = False
    """
    _OP_REQP = ["output_fields", "names", "use_locking"]
    REQ_BGL = False
+  _SIMPLE_FIELDS = ["name", "os", "network_port", "hypervisor",
+                    "serial_no", "ctime", "mtime", "uuid"]
    _FIELDS_STATIC = utils.FieldSet(*["name", "os", "pnode", "snodes",
                                      "admin_state",
                                      "disk_template", "ip", "mac", "bridge",
    _FIELDS_STATIC = utils.FieldSet(*["name", "os", "pnode", "snodes",
                                      "admin_state",
                                      "disk_template", "ip", "mac", "bridge",
@@ -3925,9 +4081,8 @@ class LUQueryInstances(NoHooksLU):
                                      r"(nic)\.(bridge)/([0-9]+)",
                                      r"(nic)\.(macs|ips|modes|links|bridges)",
                                      r"(disk|nic)\.(count)",
                                      r"(nic)\.(bridge)/([0-9]+)",
                                      r"(nic)\.(macs|ips|modes|links|bridges)",
                                      r"(disk|nic)\.(count)",
-                                    "serial_no", "hypervisor", "hvparams",
-                                    "ctime", "mtime",
-                                    ] +
+                                    "hvparams",
+                                    ] + _SIMPLE_FIELDS +
                                    ["hv/%s" % name
                                     for name in constants.HVS_PARAMETERS] +
                                    ["be/%s" % name
                                    ["hv/%s" % name
                                     for name in constants.HVS_PARAMETERS] +
                                    ["be/%s" % name
@@ -4007,7 +4162,7 @@ class LUQueryInstances(NoHooksLU):
          if result.offline:
            # offline nodes will be in both lists
            off_nodes.append(name)
          if result.offline:
            # offline nodes will be in both lists
            off_nodes.append(name)
-        if result.failed or result.fail_msg:
+        if result.fail_msg:
            bad_nodes.append(name)
          else:
            if result.payload:
            bad_nodes.append(name)
          else:
            if result.payload:
@@ -4030,10 +4185,8 @@ class LUQueryInstances(NoHooksLU):
                                   nic.nicparams) for nic in instance.nics]
        for field in self.op.output_fields:
          st_match = self._FIELDS_STATIC.Matches(field)
                                   nic.nicparams) for nic in instance.nics]
        for field in self.op.output_fields:
          st_match = self._FIELDS_STATIC.Matches(field)
-        if field == "name":
-          val = instance.name
-        elif field == "os":
-          val = instance.os
+        if field in self._SIMPLE_FIELDS:
+          val = getattr(instance, field)
          elif field == "pnode":
            val = instance.primary_node
          elif field == "snodes":
          elif field == "pnode":
            val = instance.primary_node
          elif field == "snodes":
@@ -4110,16 +4263,6 @@ class LUQueryInstances(NoHooksLU):
            val = _ComputeDiskSize(instance.disk_template, disk_sizes)
          elif field == "tags":
            val = list(instance.GetTags())
            val = _ComputeDiskSize(instance.disk_template, disk_sizes)
          elif field == "tags":
            val = list(instance.GetTags())
-        elif field == "serial_no":
-          val = instance.serial_no
-        elif field == "ctime":
-          val = instance.ctime
-        elif field == "mtime":
-          val = instance.mtime
-        elif field == "network_port":
-          val = instance.network_port
-        elif field == "hypervisor":
-          val = instance.hypervisor
          elif field == "hvparams":
            val = i_hv
          elif (field.startswith(HVPREFIX) and
          elif field == "hvparams":
            val = i_hv
          elif (field.startswith(HVPREFIX) and
@@ -4369,6 +4512,181 @@ class LUMigrateInstance(LogicalUnit):
      return env, nl, nl
  
  
      return env, nl, nl
  
  
+class LUMoveInstance(LogicalUnit):
+  """Move an instance by data-copying.
+
+  """
+  HPATH = "instance-move"
+  HTYPE = constants.HTYPE_INSTANCE
+  _OP_REQP = ["instance_name", "target_node"]
+  REQ_BGL = False
+
+  def ExpandNames(self):
+    self._ExpandAndLockInstance()
+    target_node = self.cfg.ExpandNodeName(self.op.target_node)
+    if target_node is None:
+      raise errors.OpPrereqError("Node '%s' not known" %
+                                  self.op.target_node)
+    self.op.target_node = target_node
+    self.needed_locks[locking.LEVEL_NODE] = [target_node]
+    self.recalculate_locks[locking.LEVEL_NODE] = constants.LOCKS_APPEND
+
+  def DeclareLocks(self, level):
+    if level == locking.LEVEL_NODE:
+      self._LockInstancesNodes(primary_only=True)
+
+  def BuildHooksEnv(self):
+    """Build hooks env.
+
+    This runs on master, primary and secondary nodes of the instance.
+
+    """
+    env = {
+      "TARGET_NODE": self.op.target_node,
+      }
+    env.update(_BuildInstanceHookEnvByObject(self, self.instance))
+    nl = [self.cfg.GetMasterNode()] + [self.instance.primary_node,
+                                       self.op.target_node]
+    return env, nl, nl
+
+  def CheckPrereq(self):
+    """Check prerequisites.
+
+    This checks that the instance is in the cluster.
+
+    """
+    self.instance = instance = self.cfg.GetInstanceInfo(self.op.instance_name)
+    assert self.instance is not None, \
+      "Cannot retrieve locked instance %s" % self.op.instance_name
+
+    node = self.cfg.GetNodeInfo(self.op.target_node)
+    assert node is not None, \
+      "Cannot retrieve locked node %s" % self.op.target_node
+
+    self.target_node = target_node = node.name
+
+    if target_node == instance.primary_node:
+      raise errors.OpPrereqError("Instance %s is already on the node %s" %
+                                 (instance.name, target_node))
+
+    bep = self.cfg.GetClusterInfo().FillBE(instance)
+
+    for idx, dsk in enumerate(instance.disks):
+      if dsk.dev_type not in (constants.LD_LV, constants.LD_FILE):
+        raise errors.OpPrereqError("Instance disk %d has a complex layout,"
+                                   " cannot copy")
+
+    _CheckNodeOnline(self, target_node)
+    _CheckNodeNotDrained(self, target_node)
+
+    if instance.admin_up:
+      # check memory requirements on the secondary node
+      _CheckNodeFreeMemory(self, target_node, "failing over instance %s" %
+                           instance.name, bep[constants.BE_MEMORY],
+                           instance.hypervisor)
+    else:
+      self.LogInfo("Not checking memory on the secondary node as"
+                   " instance will not be started")
+
+    # check bridge existance
+    _CheckInstanceBridgesExist(self, instance, node=target_node)
+
+  def Exec(self, feedback_fn):
+    """Move an instance.
+
+    The move is done by shutting it down on its present node, copying
+    the data over (slow) and starting it on the new node.
+
+    """
+    instance = self.instance
+
+    source_node = instance.primary_node
+    target_node = self.target_node
+
+    self.LogInfo("Shutting down instance %s on source node %s",
+                 instance.name, source_node)
+
+    result = self.rpc.call_instance_shutdown(source_node, instance)
+    msg = result.fail_msg
+    if msg:
+      if self.op.ignore_consistency:
+        self.proc.LogWarning("Could not shutdown instance %s on node %s."
+                             " Proceeding anyway. Please make sure node"
+                             " %s is down. Error details: %s",
+                             instance.name, source_node, source_node, msg)
+      else:
+        raise errors.OpExecError("Could not shutdown instance %s on"
+                                 " node %s: %s" %
+                                 (instance.name, source_node, msg))
+
+    # create the target disks
+    try:
+      _CreateDisks(self, instance, target_node=target_node)
+    except errors.OpExecError:
+      self.LogWarning("Device creation failed, reverting...")
+      try:
+        _RemoveDisks(self, instance, target_node=target_node)
+      finally:
+        self.cfg.ReleaseDRBDMinors(instance.name)
+        raise
+
+    cluster_name = self.cfg.GetClusterInfo().cluster_name
+
+    errs = []
+    # activate, get path, copy the data over
+    for idx, disk in enumerate(instance.disks):
+      self.LogInfo("Copying data for disk %d", idx)
+      result = self.rpc.call_blockdev_assemble(target_node, disk,
+                                               instance.name, True)
+      if result.fail_msg:
+        self.LogWarning("Can't assemble newly created disk %d: %s",
+                        idx, result.fail_msg)
+        errs.append(result.fail_msg)
+        break
+      dev_path = result.payload
+      result = self.rpc.call_blockdev_export(source_node, disk,
+                                             target_node, dev_path,
+                                             cluster_name)
+      if result.fail_msg:
+        self.LogWarning("Can't copy data over for disk %d: %s",
+                        idx, result.fail_msg)
+        errs.append(result.fail_msg)
+        break
+
+    if errs:
+      self.LogWarning("Some disks failed to copy, aborting")
+      try:
+        _RemoveDisks(self, instance, target_node=target_node)
+      finally:
+        self.cfg.ReleaseDRBDMinors(instance.name)
+        raise errors.OpExecError("Errors during disk copy: %s" %
+                                 (",".join(errs),))
+
+    instance.primary_node = target_node
+    self.cfg.Update(instance)
+
+    self.LogInfo("Removing the disks on the original node")
+    _RemoveDisks(self, instance, target_node=source_node)
+
+    # Only start the instance if it's marked as up
+    if instance.admin_up:
+      self.LogInfo("Starting instance %s on node %s",
+                   instance.name, target_node)
+
+      disks_ok, _ = _AssembleInstanceDisks(self, instance,
+                                           ignore_secondaries=True)
+      if not disks_ok:
+        _ShutdownInstanceDisks(self, instance)
+        raise errors.OpExecError("Can't activate the instance's disks")
+
+      result = self.rpc.call_instance_start(target_node, instance, None, None)
+      msg = result.fail_msg
+      if msg:
+        _ShutdownInstanceDisks(self, instance)
+        raise errors.OpExecError("Could not start instance %s on node %s: %s" %
+                                 (instance.name, target_node, msg))
+
+
  class LUMigrateNode(LogicalUnit):
    """Migrate all instances from a node.
  
  class LUMigrateNode(LogicalUnit):
    """Migrate all instances from a node.
  
@@ -4932,7 +5250,7 @@ def _GetInstanceInfoText(instance):
    return "originstname+%s" % instance.name
  
  
    return "originstname+%s" % instance.name
  
  
-def _CreateDisks(lu, instance, to_skip=None):
+def _CreateDisks(lu, instance, to_skip=None, target_node=None):
    """Create all disks for an instance.
  
    This abstracts away some work from AddInstance.
    """Create all disks for an instance.
  
    This abstracts away some work from AddInstance.
@@ -4943,19 +5261,26 @@ def _CreateDisks(lu, instance, to_skip=None):
    @param instance: the instance whose disks we should create
    @type to_skip: list
    @param to_skip: list of indices to skip
    @param instance: the instance whose disks we should create
    @type to_skip: list
    @param to_skip: list of indices to skip
+  @type target_node: string
+  @param target_node: if passed, overrides the target node for creation
    @rtype: boolean
    @return: the success of the creation
  
    """
    info = _GetInstanceInfoText(instance)
    @rtype: boolean
    @return: the success of the creation
  
    """
    info = _GetInstanceInfoText(instance)
-  pnode = instance.primary_node
+  if target_node is None:
+    pnode = instance.primary_node
+    all_nodes = instance.all_nodes
+  else:
+    pnode = target_node
+    all_nodes = [pnode]
  
    if instance.disk_template == constants.DT_FILE:
      file_storage_dir = os.path.dirname(instance.disks[0].logical_id[1])
      result = lu.rpc.call_file_storage_dir_create(pnode, file_storage_dir)
  
      result.Raise("Failed to create directory '%s' on"
  
    if instance.disk_template == constants.DT_FILE:
      file_storage_dir = os.path.dirname(instance.disks[0].logical_id[1])
      result = lu.rpc.call_file_storage_dir_create(pnode, file_storage_dir)
  
      result.Raise("Failed to create directory '%s' on"
-                 " node %s: %s" % (file_storage_dir, pnode))
+                 " node %s" % (file_storage_dir, pnode))
  
    # Note: this needs to be kept in sync with adding of disks in
    # LUSetInstanceParams
  
    # Note: this needs to be kept in sync with adding of disks in
    # LUSetInstanceParams
@@ -4965,12 +5290,12 @@ def _CreateDisks(lu, instance, to_skip=None):
      logging.info("Creating volume %s for instance %s",
                   device.iv_name, instance.name)
      #HARDCODE
      logging.info("Creating volume %s for instance %s",
                   device.iv_name, instance.name)
      #HARDCODE
-    for node in instance.all_nodes:
+    for node in all_nodes:
        f_create = node == pnode
        _CreateBlockDev(lu, node, instance, device, f_create, info, f_create)
  
  
        f_create = node == pnode
        _CreateBlockDev(lu, node, instance, device, f_create, info, f_create)
  
  
-def _RemoveDisks(lu, instance):
+def _RemoveDisks(lu, instance, target_node=None):
    """Remove all disks for an instance.
  
    This abstracts away some work from `AddInstance()` and
    """Remove all disks for an instance.
  
    This abstracts away some work from `AddInstance()` and
@@ -4982,6 +5307,8 @@ def _RemoveDisks(lu, instance):
    @param lu: the logical unit on whose behalf we execute
    @type instance: L{objects.Instance}
    @param instance: the instance whose disks we should remove
    @param lu: the logical unit on whose behalf we execute
    @type instance: L{objects.Instance}
    @param instance: the instance whose disks we should remove
+  @type target_node: string
+  @param target_node: used to override the node on which to remove the disks
    @rtype: boolean
    @return: the success of the removal
  
    @rtype: boolean
    @return: the success of the removal
  
@@ -4990,7 +5317,11 @@ def _RemoveDisks(lu, instance):
  
    all_result = True
    for device in instance.disks:
  
    all_result = True
    for device in instance.disks:
-    for node, disk in device.ComputeNodeTree(instance.primary_node):
+    if target_node:
+      edata = [(target_node, device)]
+    else:
+      edata = device.ComputeNodeTree(instance.primary_node)
+    for node, disk in edata:
        lu.cfg.SetDiskID(disk, node)
        msg = lu.rpc.call_blockdev_remove(node, disk).fail_msg
        if msg:
        lu.cfg.SetDiskID(disk, node)
        msg = lu.rpc.call_blockdev_remove(node, disk).fail_msg
        if msg:
@@ -5000,12 +5331,14 @@ def _RemoveDisks(lu, instance):
  
    if instance.disk_template == constants.DT_FILE:
      file_storage_dir = os.path.dirname(instance.disks[0].logical_id[1])
  
    if instance.disk_template == constants.DT_FILE:
      file_storage_dir = os.path.dirname(instance.disks[0].logical_id[1])
-    result = lu.rpc.call_file_storage_dir_remove(instance.primary_node,
-                                                 file_storage_dir)
-    msg = result.fail_msg
-    if msg:
+    if target_node:
+      tgt = target_node
+    else:
+      tgt = instance.primary_node
+    result = lu.rpc.call_file_storage_dir_remove(tgt, file_storage_dir)
+    if result.fail_msg:
        lu.LogWarning("Could not remove directory '%s' on node %s: %s",
        lu.LogWarning("Could not remove directory '%s' on node %s: %s",
-                    file_storage_dir, instance.primary_node, msg)
+                    file_storage_dir, instance.primary_node, result.fail_msg)
        all_result = False
  
    return all_result
        all_result = False
  
    return all_result
@@ -5177,6 +5510,12 @@ class LUCreateInstance(LogicalUnit):
          if not utils.IsValidMac(mac.lower()):
            raise errors.OpPrereqError("Invalid MAC address specified: %s" %
                                       mac)
          if not utils.IsValidMac(mac.lower()):
            raise errors.OpPrereqError("Invalid MAC address specified: %s" %
                                       mac)
+        else:
+          # or validate/reserve the current one
+          if self.cfg.IsMacInUse(mac):
+            raise errors.OpPrereqError("MAC address %s already in use"
+                                       " in cluster" % mac)
+
        # bridge verification
        bridge = nic.get("bridge", None)
        link = nic.get("link", None)
        # bridge verification
        bridge = nic.get("bridge", None)
        link = nic.get("link", None)
@@ -5264,9 +5603,15 @@ class LUCreateInstance(LogicalUnit):
            self.op.src_path = src_path = \
              os.path.join(constants.EXPORT_DIR, src_path)
  
            self.op.src_path = src_path = \
              os.path.join(constants.EXPORT_DIR, src_path)
  
+      # On import force_variant must be True, because if we forced it at
+      # initial install, our only chance when importing it back is that it
+      # works again!
+      self.op.force_variant = True
+
      else: # INSTANCE_CREATE
        if getattr(self.op, "os_type", None) is None:
          raise errors.OpPrereqError("No guest OS specified")
      else: # INSTANCE_CREATE
        if getattr(self.op, "os_type", None) is None:
          raise errors.OpPrereqError("No guest OS specified")
+      self.op.force_variant = getattr(self.op, "force_variant", False)
  
    def _RunAllocator(self):
      """Run the allocator based on input opcode.
  
    def _RunAllocator(self):
      """Run the allocator based on input opcode.
@@ -5496,6 +5841,8 @@ class LUCreateInstance(LogicalUnit):
      result = self.rpc.call_os_get(pnode.name, self.op.os_type)
      result.Raise("OS '%s' not in supported os list for primary node %s" %
                   (self.op.os_type, pnode.name), prereq=True)
      result = self.rpc.call_os_get(pnode.name, self.op.os_type)
      result.Raise("OS '%s' not in supported os list for primary node %s" %
                   (self.op.os_type, pnode.name), prereq=True)
+    if not self.op.force_variant:
+      _CheckOSVariant(result.payload, self.op.os_type)
  
      _CheckNicsBridgesExist(self, self.nics, self.pnode.name)
  
  
      _CheckNicsBridgesExist(self, self.nics, self.pnode.name)
  
@@ -6243,7 +6590,8 @@ class TLReplaceDisks(Tasklet):
      for dev, old_lvs, new_lvs in iv_names.itervalues():
        self.lu.LogInfo("Detaching %s drbd from local storage" % dev.iv_name)
  
      for dev, old_lvs, new_lvs in iv_names.itervalues():
        self.lu.LogInfo("Detaching %s drbd from local storage" % dev.iv_name)
  
-      result = self.rpc.call_blockdev_removechildren(self.target_node, dev, old_lvs)
+      result = self.rpc.call_blockdev_removechildren(self.target_node, dev,
+                                                     old_lvs)
        result.Raise("Can't detach drbd from local storage on node"
                     " %s for device %s" % (self.target_node, dev.iv_name))
        #dev.children = []
        result.Raise("Can't detach drbd from local storage on node"
                     " %s for device %s" % (self.target_node, dev.iv_name))
        #dev.children = []
@@ -6269,14 +6617,16 @@ class TLReplaceDisks(Tasklet):
            rename_old_to_new.append((to_ren, ren_fn(to_ren, temp_suffix)))
  
        self.lu.LogInfo("Renaming the old LVs on the target node")
            rename_old_to_new.append((to_ren, ren_fn(to_ren, temp_suffix)))
  
        self.lu.LogInfo("Renaming the old LVs on the target node")
-      result = self.rpc.call_blockdev_rename(self.target_node, rename_old_to_new)
+      result = self.rpc.call_blockdev_rename(self.target_node,
+                                             rename_old_to_new)
        result.Raise("Can't rename old LVs on node %s" % self.target_node)
  
        # Now we rename the new LVs to the old LVs
        self.lu.LogInfo("Renaming the new LVs on the target node")
        rename_new_to_old = [(new, old.physical_id)
                             for old, new in zip(old_lvs, new_lvs)]
        result.Raise("Can't rename old LVs on node %s" % self.target_node)
  
        # Now we rename the new LVs to the old LVs
        self.lu.LogInfo("Renaming the new LVs on the target node")
        rename_new_to_old = [(new, old.physical_id)
                             for old, new in zip(old_lvs, new_lvs)]
-      result = self.rpc.call_blockdev_rename(self.target_node, rename_new_to_old)
+      result = self.rpc.call_blockdev_rename(self.target_node,
+                                             rename_new_to_old)
        result.Raise("Can't rename new LVs on node %s" % self.target_node)
  
        for old, new in zip(old_lvs, new_lvs):
        result.Raise("Can't rename new LVs on node %s" % self.target_node)
  
        for old, new in zip(old_lvs, new_lvs):
@@ -6289,11 +6639,13 @@ class TLReplaceDisks(Tasklet):
  
        # Now that the new lvs have the old name, we can add them to the device
        self.lu.LogInfo("Adding new mirror component on %s" % self.target_node)
  
        # Now that the new lvs have the old name, we can add them to the device
        self.lu.LogInfo("Adding new mirror component on %s" % self.target_node)
-      result = self.rpc.call_blockdev_addchildren(self.target_node, dev, new_lvs)
+      result = self.rpc.call_blockdev_addchildren(self.target_node, dev,
+                                                  new_lvs)
        msg = result.fail_msg
        if msg:
          for new_lv in new_lvs:
        msg = result.fail_msg
        if msg:
          for new_lv in new_lvs:
-          msg2 = self.rpc.call_blockdev_remove(self.target_node, new_lv).fail_msg
+          msg2 = self.rpc.call_blockdev_remove(self.target_node,
+                                               new_lv).fail_msg
            if msg2:
              self.lu.LogWarning("Can't rollback device %s: %s", dev, msg2,
                                 hint=("cleanup manually the unused logical"
            if msg2:
              self.lu.LogWarning("Can't rollback device %s: %s", dev, msg2,
                                 hint=("cleanup manually the unused logical"
@@ -6361,13 +6713,15 @@ class TLReplaceDisks(Tasklet):
      # after this, we must manually remove the drbd minors on both the
      # error and the success paths
      self.lu.LogStep(4, steps_total, "Changing drbd configuration")
      # after this, we must manually remove the drbd minors on both the
      # error and the success paths
      self.lu.LogStep(4, steps_total, "Changing drbd configuration")
-    minors = self.cfg.AllocateDRBDMinor([self.new_node for dev in self.instance.disks],
+    minors = self.cfg.AllocateDRBDMinor([self.new_node
+                                         for dev in self.instance.disks],
                                          self.instance.name)
      logging.debug("Allocated minors %r" % (minors,))
  
      iv_names = {}
      for idx, (dev, new_minor) in enumerate(zip(self.instance.disks, minors)):
                                          self.instance.name)
      logging.debug("Allocated minors %r" % (minors,))
  
      iv_names = {}
      for idx, (dev, new_minor) in enumerate(zip(self.instance.disks, minors)):
-      self.lu.LogInfo("activating a new drbd on %s for disk/%d" % (self.new_node, idx))
+      self.lu.LogInfo("activating a new drbd on %s for disk/%d" %
+                      (self.new_node, idx))
        # create new devices on new_node; note that we create two IDs:
        # one without port, so the drbd will be activated without
        # networking information on the new node at this stage, and one
        # create new devices on new_node; note that we create two IDs:
        # one without port, so the drbd will be activated without
        # networking information on the new node at this stage, and one
@@ -6378,8 +6732,10 @@ class TLReplaceDisks(Tasklet):
        else:
          p_minor = o_minor2
  
        else:
          p_minor = o_minor2
  
-      new_alone_id = (self.instance.primary_node, self.new_node, None, p_minor, new_minor, o_secret)
-      new_net_id = (self.instance.primary_node, self.new_node, o_port, p_minor, new_minor, o_secret)
+      new_alone_id = (self.instance.primary_node, self.new_node, None,
+                      p_minor, new_minor, o_secret)
+      new_net_id = (self.instance.primary_node, self.new_node, o_port,
+                    p_minor, new_minor, o_secret)
  
        iv_names[idx] = (dev, dev.children, new_net_id)
        logging.debug("Allocated new_minor: %s, new_logical_id: %s", new_minor,
  
        iv_names[idx] = (dev, dev.children, new_net_id)
        logging.debug("Allocated new_minor: %s, new_logical_id: %s", new_minor,
@@ -6407,8 +6763,10 @@ class TLReplaceDisks(Tasklet):
                                   " soon as possible"))
  
      self.lu.LogInfo("Detaching primary drbds from the network (=> standalone)")
                                   " soon as possible"))
  
      self.lu.LogInfo("Detaching primary drbds from the network (=> standalone)")
-    result = self.rpc.call_drbd_disconnect_net([self.instance.primary_node], self.node_secondary_ip,
-                                               self.instance.disks)[self.instance.primary_node]
+    result = self.rpc.call_drbd_disconnect_net([self.instance.primary_node],
+                                               self.node_secondary_ip,
+                                               self.instance.disks)\
+                                              [self.instance.primary_node]
  
      msg = result.fail_msg
      if msg:
  
      msg = result.fail_msg
      if msg:
@@ -6429,13 +6787,17 @@ class TLReplaceDisks(Tasklet):
      # and now perform the drbd attach
      self.lu.LogInfo("Attaching primary drbds to new secondary"
                      " (standalone => connected)")
      # and now perform the drbd attach
      self.lu.LogInfo("Attaching primary drbds to new secondary"
                      " (standalone => connected)")
-    result = self.rpc.call_drbd_attach_net([self.instance.primary_node, self.new_node], self.node_secondary_ip,
-                                           self.instance.disks, self.instance.name,
+    result = self.rpc.call_drbd_attach_net([self.instance.primary_node,
+                                            self.new_node],
+                                           self.node_secondary_ip,
+                                           self.instance.disks,
+                                           self.instance.name,
                                             False)
      for to_node, to_result in result.items():
        msg = to_result.fail_msg
        if msg:
                                             False)
      for to_node, to_result in result.items():
        msg = to_result.fail_msg
        if msg:
-        self.lu.LogWarning("Can't attach drbd disks on node %s: %s", to_node, msg,
+        self.lu.LogWarning("Can't attach drbd disks on node %s: %s",
+                           to_node, msg,
                             hint=("please do a gnt-instance info to see the"
                                   " status of disks"))
  
                             hint=("please do a gnt-instance info to see the"
                                   " status of disks"))
  
@@ -6476,7 +6838,7 @@ class LURepairNodeStorage(NoHooksLU):
      if _FindFaultyInstanceDisks(self.cfg, self.rpc, instance,
                                  node_name, True):
        raise errors.OpPrereqError("Instance '%s' has faulty disks on"
      if _FindFaultyInstanceDisks(self.cfg, self.rpc, instance,
                                  node_name, True):
        raise errors.OpPrereqError("Instance '%s' has faulty disks on"
-                                 " node '%s'" % (inst.name, node_name))
+                                 " node '%s'" % (instance.name, node_name))
  
    def CheckPrereq(self):
      """Check prerequisites.
  
    def CheckPrereq(self):
      """Check prerequisites.
@@ -6749,6 +7111,7 @@ class LUQueryInstanceData(NoHooksLU):
          "serial_no": instance.serial_no,
          "mtime": instance.mtime,
          "ctime": instance.ctime,
          "serial_no": instance.serial_no,
          "mtime": instance.mtime,
          "ctime": instance.ctime,
+        "uuid": instance.uuid,
          }
  
        result[instance.name] = idict
          }
  
        result[instance.name] = idict
@@ -7381,8 +7744,10 @@ class LUExportInstance(LogicalUnit):
      instance = self.instance
      dst_node = self.dst_node
      src_node = instance.primary_node
      instance = self.instance
      dst_node = self.dst_node
      src_node = instance.primary_node
+
      if self.op.shutdown:
        # shutdown the instance, but not the disks
      if self.op.shutdown:
        # shutdown the instance, but not the disks
+      feedback_fn("Shutting down instance %s" % instance.name)
        result = self.rpc.call_instance_shutdown(src_node, instance)
        result.Raise("Could not shutdown instance %s on"
                     " node %s" % (instance.name, src_node))
        result = self.rpc.call_instance_shutdown(src_node, instance)
        result.Raise("Could not shutdown instance %s on"
                     " node %s" % (instance.name, src_node))
@@ -7400,6 +7765,9 @@ class LUExportInstance(LogicalUnit):
      dresults = []
      try:
        for idx, disk in enumerate(instance.disks):
      dresults = []
      try:
        for idx, disk in enumerate(instance.disks):
+        feedback_fn("Creating a snapshot of disk/%s on node %s" %
+                    (idx, src_node))
+
          # result.payload will be a snapshot of an lvm leaf of the one we passed
          result = self.rpc.call_blockdev_snapshot(src_node, disk)
          msg = result.fail_msg
          # result.payload will be a snapshot of an lvm leaf of the one we passed
          result = self.rpc.call_blockdev_snapshot(src_node, disk)
          msg = result.fail_msg
@@ -7416,6 +7784,7 @@ class LUExportInstance(LogicalUnit):
  
      finally:
        if self.op.shutdown and instance.admin_up:
  
      finally:
        if self.op.shutdown and instance.admin_up:
+        feedback_fn("Starting instance %s" % instance.name)
          result = self.rpc.call_instance_start(src_node, instance, None, None)
          msg = result.fail_msg
          if msg:
          result = self.rpc.call_instance_start(src_node, instance, None, None)
          msg = result.fail_msg
          if msg:
@@ -7426,6 +7795,8 @@ class LUExportInstance(LogicalUnit):
  
      cluster_name = self.cfg.GetClusterName()
      for idx, dev in enumerate(snap_disks):
  
      cluster_name = self.cfg.GetClusterName()
      for idx, dev in enumerate(snap_disks):
+      feedback_fn("Exporting snapshot %s from %s to %s" %
+                  (idx, src_node, dst_node.name))
        if dev:
          result = self.rpc.call_snapshot_export(src_node, dev, dst_node.name,
                                                 instance, cluster_name, idx)
        if dev:
          result = self.rpc.call_snapshot_export(src_node, dev, dst_node.name,
                                                 instance, cluster_name, idx)
@@ -7443,6 +7814,7 @@ class LUExportInstance(LogicalUnit):
        else:
          dresults.append(False)
  
        else:
          dresults.append(False)
  
+    feedback_fn("Finalizing export on %s" % dst_node.name)
      result = self.rpc.call_finalize_export(dst_node.name, instance, snap_disks)
      fin_resu = True
      msg = result.fail_msg
      result = self.rpc.call_finalize_export(dst_node.name, instance, snap_disks)
      fin_resu = True
      msg = result.fail_msg
@@ -7459,6 +7831,7 @@ class LUExportInstance(LogicalUnit):
      # substitutes an empty list with the full cluster node list.
      iname = instance.name
      if nodelist:
      # substitutes an empty list with the full cluster node list.
      iname = instance.name
      if nodelist:
+      feedback_fn("Removing old exports for instance %s" % iname)
        exportlist = self.rpc.call_export_list(nodelist)
        for node in exportlist:
          if exportlist[node].fail_msg:
        exportlist = self.rpc.call_export_list(nodelist)
        for node in exportlist:
          if exportlist[node].fail_msg: