Treat empty oob_program param as default

[ganeti-local] / lib / cmdlib.py
diff --git a/lib/cmdlib.py b/lib/cmdlib.py

index 51e94ff..1cc765a 100644 (file)
--- a/lib/cmdlib.py
+++ b/lib/cmdlib.py
@@ -1397,6 +1397,13 @@ class LUClusterVerify(LogicalUnit):
          _ErrorIf(test, self.ENODEHV, node,
                   "hypervisor %s verify failure: '%s'", hv_name, hv_result)
  
+    hvp_result = nresult.get(constants.NV_HVPARAMS, None)
+    if ninfo.vm_capable and isinstance(hvp_result, list):
+      for item, hv_name, hv_result in hvp_result:
+        _ErrorIf(True, self.ENODEHV, node,
+                 "hypervisor %s parameter verify failure (source %s): %s",
+                 hv_name, item, hv_result)
+
      test = nresult.get(constants.NV_NODESETUP,
                             ["Missing NODESETUP results"])
      _ErrorIf(test, self.ENODESETUP, node, "node setup error: %s",
@@ -1547,7 +1554,7 @@ class LUClusterVerify(LogicalUnit):
                 node_current)
  
      for node, n_img in node_image.items():
-      if (not node == node_current):
+      if node != node_current:
          test = instance in n_img.instances
          _ErrorIf(test, self.EINSTANCEWRONGNODE, instance,
                   "instance should not run on node %s", node)
@@ -1557,7 +1564,11 @@ class LUClusterVerify(LogicalUnit):
                  for idx, (success, status) in enumerate(disks)]
  
      for nname, success, bdev_status, idx in diskdata:
-      _ErrorIf(instanceconfig.admin_up and not success,
+      # the 'ghost node' construction in Exec() ensures that we have a
+      # node here
+      snode = node_image[nname]
+      bad_snode = snode.ghost or snode.offline
+      _ErrorIf(instanceconfig.admin_up and not success and not bad_snode,
                 self.EINSTANCEFAULTYDISK, instance,
                 "couldn't retrieve status for disk/%s on %s: %s",
                 idx, nname, bdev_status)
@@ -1615,6 +1626,12 @@ class LUClusterVerify(LogicalUnit):
        # WARNING: we currently take into account down instances as well
        # as up ones, considering that even if they're down someone
        # might want to start them even in the event of a node failure.
+      if n_img.offline:
+        # we're skipping offline nodes from the N+1 warning, since
+        # most likely we don't have good memory infromation from them;
+        # we already list instances living on such nodes, and that's
+        # enough warning
+        continue
        for prinode, instances in n_img.sbp.items():
          needed_mem = 0
          for instance in instances:
@@ -2029,6 +2046,21 @@ class LUClusterVerify(LogicalUnit):
  
      return instdisk
  
+  def _VerifyHVP(self, hvp_data):
+    """Verifies locally the syntax of the hypervisor parameters.
+
+    """
+    for item, hv_name, hv_params in hvp_data:
+      msg = ("hypervisor %s parameters syntax check (source %s): %%s" %
+             (item, hv_name))
+      try:
+        hv_class = hypervisor.GetHypervisor(hv_name)
+        utils.ForceDictType(hv_params, constants.HVS_PARAMETER_TYPES)
+        hv_class.CheckParameterSyntax(hv_params)
+      except errors.GenericError, err:
+        self._ErrorIf(True, self.ECLUSTERCFG, None, msg % str(err))
+
+
    def BuildHooksEnv(self):
      """Build hooks env.
  
@@ -2094,12 +2126,32 @@ class LUClusterVerify(LogicalUnit):
  
      local_checksums = utils.FingerprintFiles(file_names)
  
+    # Compute the set of hypervisor parameters
+    hvp_data = []
+    for hv_name in hypervisors:
+      hvp_data.append(("cluster", hv_name, cluster.GetHVDefaults(hv_name)))
+    for os_name, os_hvp in cluster.os_hvp.items():
+      for hv_name, hv_params in os_hvp.items():
+        if not hv_params:
+          continue
+        full_params = cluster.GetHVDefaults(hv_name, os_name=os_name)
+        hvp_data.append(("os %s" % os_name, hv_name, full_params))
+    # TODO: collapse identical parameter values in a single one
+    for instance in instanceinfo.values():
+      if not instance.hvparams:
+        continue
+      hvp_data.append(("instance %s" % instance.name, instance.hypervisor,
+                       cluster.FillHV(instance)))
+    # and verify them locally
+    self._VerifyHVP(hvp_data)
+
      feedback_fn("* Gathering data (%d nodes)" % len(nodelist))
      node_verify_param = {
        constants.NV_FILELIST: file_names,
        constants.NV_NODELIST: [node.name for node in nodeinfo
                                if not node.offline],
        constants.NV_HYPERVISOR: hypervisors,
+      constants.NV_HVPARAMS: hvp_data,
        constants.NV_NODENETTEST: [(node.name, node.primary_ip,
                                    node.secondary_ip) for node in nodeinfo
                                   if not node.offline],
@@ -2248,8 +2300,8 @@ class LUClusterVerify(LogicalUnit):
                 self.ENODERPC, pnode, "instance %s, connection to"
                 " primary node failed", instance)
  
-      if pnode_img.offline:
-        inst_nodes_offline.append(pnode)
+      _ErrorIf(pnode_img.offline, self.EINSTANCEBADNODE, instance,
+               "instance lives on offline node %s", inst_config.primary_node)
  
        # If the instance is non-redundant we cannot survive losing its primary
        # node, so we are not N+1 compliant. On the other hand we have no disk
@@ -2298,7 +2350,7 @@ class LUClusterVerify(LogicalUnit):
  
        # warn that the instance lives on offline nodes
        _ErrorIf(inst_nodes_offline, self.EINSTANCEBADNODE, instance,
-               "instance lives on offline node(s) %s",
+               "instance has offline secondary node(s) %s",
                 utils.CommaJoin(inst_nodes_offline))
        # ... or ghost/non-vm_capable nodes
        for node in inst_config.all_nodes:
@@ -2406,14 +2458,12 @@ class LUClusterVerifyDisks(NoHooksLU):
      result = res_nodes, res_instances, res_missing = {}, [], {}
  
      nodes = utils.NiceSort(self.cfg.GetVmCapableNodeList())
-    instances = [self.cfg.GetInstanceInfo(name)
-                 for name in self.cfg.GetInstanceList()]
+    instances = self.cfg.GetAllInstancesInfo().values()
  
      nv_dict = {}
      for inst in instances:
        inst_lvs = {}
-      if (not inst.admin_up or
-          inst.disk_template not in constants.DTS_NET_MIRROR):
+      if not inst.admin_up:
          continue
        inst.MapLVsByNode(inst_lvs)
        # transform { iname: {node: [vol,],},} to {(node, vol): iname}
@@ -2424,14 +2474,8 @@ class LUClusterVerifyDisks(NoHooksLU):
      if not nv_dict:
        return result
  
-    vg_names = self.rpc.call_vg_list(nodes)
-    for node in nodes:
-      vg_names[node].Raise("Cannot get list of VGs")
-
-    for node in nodes:
-      # node_volume
-      node_res = self.rpc.call_lv_list([node],
-                                       vg_names[node].payload.keys())[node]
+    node_lvs = self.rpc.call_lv_list(nodes, [])
+    for node, node_res in node_lvs.items():
        if node_res.offline:
          continue
        msg = node_res.fail_msg
@@ -2540,16 +2584,18 @@ class LUClusterRepairDiskSizes(NoHooksLU):
        newl = [v[2].Copy() for v in dskl]
        for dsk in newl:
          self.cfg.SetDiskID(dsk, node)
-      result = self.rpc.call_blockdev_getsizes(node, newl)
+      result = self.rpc.call_blockdev_getsize(node, newl)
        if result.fail_msg:
-        self.LogWarning("Failure in blockdev_getsizes call to node"
+        self.LogWarning("Failure in blockdev_getsize call to node"
                          " %s, ignoring", node)
          continue
-      if len(result.data) != len(dskl):
+      if len(result.payload) != len(dskl):
+        logging.warning("Invalid result from node %s: len(dksl)=%d,"
+                        " result.payload=%s", node, len(dskl), result.payload)
          self.LogWarning("Invalid result from node %s, ignoring node results",
                          node)
          continue
-      for ((instance, idx, disk), size) in zip(dskl, result.data):
+      for ((instance, idx, disk), size) in zip(dskl, result.payload):
          if size is None:
            self.LogWarning("Disk %d of instance %s did not return size"
                            " information, ignoring", idx, instance.name)
@@ -2755,6 +2801,12 @@ class LUClusterSetParams(LogicalUnit):
        utils.ForceDictType(self.op.ndparams, constants.NDS_PARAMETER_TYPES)
        self.new_ndparams = cluster.SimpleFillND(self.op.ndparams)
  
+      # TODO: we need a more general way to handle resetting
+      # cluster-level parameters to default values
+      if self.new_ndparams["oob_program"] == "":
+        self.new_ndparams["oob_program"] = \
+            constants.NDC_DEFAULTS[constants.ND_OOB_PROGRAM]
+
      if self.op.nicparams:
        utils.ForceDictType(self.op.nicparams, constants.NICS_PARAMETER_TYPES)
        self.new_nicparams = cluster.SimpleFillNIC(self.op.nicparams)
@@ -3395,7 +3447,9 @@ class LUOsDiagnose(NoHooksLU):
      """Compute the list of OSes.
  
      """
-    valid_nodes = [node for node in self.cfg.GetOnlineNodeList()]
+    valid_nodes = [node.name
+                   for node in self.cfg.GetAllNodesInfo().values()
+                   if not node.offline and node.vm_capable]
      node_data = self.rpc.call_os_diagnose(valid_nodes)
      pol = self._DiagnoseByOS(node_data)
      output = []
@@ -3584,7 +3638,10 @@ class _NodeQuery(_QueryBase):
  
      # Gather data as requested
      if query.NQ_LIVE in self.requested_data:
-      node_data = lu.rpc.call_node_info(nodenames, lu.cfg.GetVGName(),
+      # filter out non-vm_capable nodes
+      toquery_nodes = [name for name in nodenames if all_info[name].vm_capable]
+
+      node_data = lu.rpc.call_node_info(toquery_nodes, lu.cfg.GetVGName(),
                                          lu.cfg.GetHypervisorType())
        live_data = dict((name, nresult.payload)
                         for (name, nresult) in node_data.items()
@@ -3832,18 +3889,21 @@ class _InstanceQuery(_QueryBase):
      """Computes the list of instances and their attributes.
  
      """
+    cluster = lu.cfg.GetClusterInfo()
      all_info = lu.cfg.GetAllInstancesInfo()
  
      instance_names = self._GetNames(lu, all_info.keys(), locking.LEVEL_INSTANCE)
  
      instance_list = [all_info[name] for name in instance_names]
-    nodes = frozenset([inst.primary_node for inst in instance_list])
+    nodes = frozenset(itertools.chain(*(inst.all_nodes
+                                        for inst in instance_list)))
      hv_list = list(set([inst.hypervisor for inst in instance_list]))
      bad_nodes = []
      offline_nodes = []
+    wrongnode_inst = set()
  
      # Gather data as requested
-    if query.IQ_LIVE in self.requested_data:
+    if self.requested_data & set([query.IQ_LIVE, query.IQ_CONSOLE]):
        live_data = {}
        node_data = lu.rpc.call_all_instances_info(nodes, hv_list)
        for name in nodes:
@@ -3855,7 +3915,17 @@ class _InstanceQuery(_QueryBase):
          if result.fail_msg:
            bad_nodes.append(name)
          elif result.payload:
-          live_data.update(result.payload)
+          for inst in result.payload:
+            if inst in all_info:
+              if all_info[inst].primary_node == name:
+                live_data.update(result.payload)
+              else:
+                wrongnode_inst.add(inst)
+            else:
+              # orphan instance; we don't list it here as we don't
+              # handle this case yet in the output of instance listing
+              logging.warning("Orphan instance '%s' found on node %s",
+                              inst, name)
          # else no instance is alive
      else:
        live_data = {}
@@ -3869,9 +3939,21 @@ class _InstanceQuery(_QueryBase):
      else:
        disk_usage = None
  
+    if query.IQ_CONSOLE in self.requested_data:
+      consinfo = {}
+      for inst in instance_list:
+        if inst.name in live_data:
+          # Instance is running
+          consinfo[inst.name] = _GetInstanceConsole(cluster, inst)
+        else:
+          consinfo[inst.name] = None
+      assert set(consinfo.keys()) == set(instance_names)
+    else:
+      consinfo = None
+
      return query.InstanceQueryData(instance_list, lu.cfg.GetClusterInfo(),
                                     disk_usage, offline_nodes, bad_nodes,
-                                   live_data)
+                                   live_data, wrongnode_inst, consinfo)
  
  
  class LUQuery(NoHooksLU):
@@ -4345,15 +4427,15 @@ class LUNodeSetParams(LogicalUnit):
                                     errors.ECODE_STATE)
  
      if node.master_candidate and self.might_demote and not self.lock_all:
-      assert not self.op.auto_promote, "auto-promote set but lock_all not"
+      assert not self.op.auto_promote, "auto_promote set but lock_all not"
        # check if after removing the current node, we're missing master
        # candidates
        (mc_remaining, mc_should, _) = \
            self.cfg.GetMasterCandidateStats(exceptions=[node.name])
        if mc_remaining < mc_should:
          raise errors.OpPrereqError("Not enough master candidates, please"
-                                   " pass auto_promote to allow promotion",
-                                   errors.ECODE_STATE)
+                                   " pass auto promote option to allow"
+                                   " promotion", errors.ECODE_STATE)
  
      self.old_flags = old_flags = (node.master_candidate,
                                    node.drained, node.offline)
@@ -4728,13 +4810,13 @@ def _AssembleInstanceDisks(lu, instance, disks=None, ignore_secondaries=False,
    # SyncSource, etc.)
  
    # 1st pass, assemble on all nodes in secondary mode
-  for inst_disk in disks:
+  for idx, inst_disk in enumerate(disks):
      for node, node_disk in inst_disk.ComputeNodeTree(instance.primary_node):
        if ignore_size:
          node_disk = node_disk.Copy()
          node_disk.UnsetSize()
        lu.cfg.SetDiskID(node_disk, node)
-      result = lu.rpc.call_blockdev_assemble(node, node_disk, iname, False)
+      result = lu.rpc.call_blockdev_assemble(node, node_disk, iname, False, idx)
        msg = result.fail_msg
        if msg:
          lu.proc.LogWarning("Could not prepare block device %s on node %s"
@@ -4746,7 +4828,7 @@ def _AssembleInstanceDisks(lu, instance, disks=None, ignore_secondaries=False,
    # FIXME: race condition on drbd migration to primary
  
    # 2nd pass, do only the primary node
-  for inst_disk in disks:
+  for idx, inst_disk in enumerate(disks):
      dev_path = None
  
      for node, node_disk in inst_disk.ComputeNodeTree(instance.primary_node):
@@ -4756,7 +4838,7 @@ def _AssembleInstanceDisks(lu, instance, disks=None, ignore_secondaries=False,
          node_disk = node_disk.Copy()
          node_disk.UnsetSize()
        lu.cfg.SetDiskID(node_disk, node)
-      result = lu.rpc.call_blockdev_assemble(node, node_disk, iname, True)
+      result = lu.rpc.call_blockdev_assemble(node, node_disk, iname, True, idx)
        msg = result.fail_msg
        if msg:
          lu.proc.LogWarning("Could not prepare block device %s on node %s"
@@ -4822,7 +4904,10 @@ class LUInstanceDeactivateDisks(NoHooksLU):
  
      """
      instance = self.instance
-    _SafeShutdownInstanceDisks(self, instance)
+    if self.op.force:
+      _ShutdownInstanceDisks(self, instance)
+    else:
+      _SafeShutdownInstanceDisks(self, instance)
  
  
  def _SafeShutdownInstanceDisks(lu, instance, disks=None):
@@ -5891,7 +5976,7 @@ class LUInstanceMove(LogicalUnit):
      for idx, disk in enumerate(instance.disks):
        self.LogInfo("Copying data for disk %d", idx)
        result = self.rpc.call_blockdev_assemble(target_node, disk,
-                                               instance.name, True)
+                                               instance.name, True, idx)
        if result.fail_msg:
          self.LogWarning("Can't assemble newly created disk %d: %s",
                          idx, result.fail_msg)
@@ -6556,6 +6641,10 @@ def _WipeDisks(lu, instance):
  
    """
    node = instance.primary_node
+
+  for device in instance.disks:
+    lu.cfg.SetDiskID(device, node)
+
    logging.info("Pause sync of instance %s disks", instance.name)
    result = lu.rpc.call_blockdev_pause_resume_sync(node, instance.disks, True)
  
@@ -6567,7 +6656,8 @@ def _WipeDisks(lu, instance):
    try:
      for idx, device in enumerate(instance.disks):
        lu.LogInfo("* Wiping disk %d", idx)
-      logging.info("Wiping disk %d for instance %s", idx, instance.name)
+      logging.info("Wiping disk %d for instance %s, node %s",
+                   idx, instance.name, node)
  
        # The wipe size is MIN_WIPE_CHUNK_PERCENT % of the instance disk but
        # MAX_WIPE_CHUNK at max
@@ -6748,6 +6838,21 @@ def _ComputeDiskSize(disk_template, disks):
    return req_size_dict[disk_template]
  
  
+def _FilterVmNodes(lu, nodenames):
+  """Filters out non-vm_capable nodes from a list.
+
+  @type lu: L{LogicalUnit}
+  @param lu: the logical unit for which we check
+  @type nodenames: list
+  @param nodenames: the list of nodes on which we should check
+  @rtype: list
+  @return: the list of vm-capable nodes
+
+  """
+  vm_nodes = frozenset(lu.cfg.GetNonVmCapableNodeList())
+  return [name for name in nodenames if name not in vm_nodes]
+
+
  def _CheckHVParams(lu, nodenames, hvname, hvparams):
    """Hypervisor parameter validation.
  
@@ -6765,6 +6870,7 @@ def _CheckHVParams(lu, nodenames, hvname, hvparams):
    @raise errors.OpPrereqError: if the parameters are not valid
  
    """
+  nodenames = _FilterVmNodes(lu, nodenames)
    hvinfo = lu.rpc.call_hypervisor_validate_params(nodenames,
                                                    hvname,
                                                    hvparams)
@@ -6792,6 +6898,7 @@ def _CheckOSParams(lu, required, nodenames, osname, osparams):
    @raise errors.OpPrereqError: if the parameters are not valid
  
    """
+  nodenames = _FilterVmNodes(lu, nodenames)
    result = lu.rpc.call_os_validate(required, nodenames, osname,
                                     [constants.OS_VALIDATE_PARAMETERS],
                                     osparams)
@@ -7759,18 +7866,28 @@ class LUInstanceConsole(NoHooksLU):
  
      logging.debug("Connecting to console of %s on %s", instance.name, node)
  
-    hyper = hypervisor.GetHypervisor(instance.hypervisor)
-    cluster = self.cfg.GetClusterInfo()
-    # beparams and hvparams are passed separately, to avoid editing the
-    # instance and then saving the defaults in the instance itself.
-    hvparams = cluster.FillHV(instance)
-    beparams = cluster.FillBE(instance)
-    console = hyper.GetInstanceConsole(instance, hvparams, beparams)
+    return _GetInstanceConsole(self.cfg.GetClusterInfo(), instance)
  
-    assert console.instance == instance.name
-    assert console.Validate()
  
-    return console.ToDict()
+def _GetInstanceConsole(cluster, instance):
+  """Returns console information for an instance.
+
+  @type cluster: L{objects.Cluster}
+  @type instance: L{objects.Instance}
+  @rtype: dict
+
+  """
+  hyper = hypervisor.GetHypervisor(instance.hypervisor)
+  # beparams and hvparams are passed separately, to avoid editing the
+  # instance and then saving the defaults in the instance itself.
+  hvparams = cluster.FillHV(instance)
+  beparams = cluster.FillBE(instance)
+  console = hyper.GetInstanceConsole(instance, hvparams, beparams)
+
+  assert console.instance == instance.name
+  assert console.Validate()
+
+  return console.ToDict()
  
  
  class LUInstanceReplaceDisks(LogicalUnit):
@@ -10314,9 +10431,9 @@ class LUGroupRemove(LogicalUnit):
  
      # Verify the cluster would not be left group-less.
      if len(self.cfg.GetNodeGroupList()) == 1:
-      raise errors.OpPrereqError("Group '%s' is the last group in the cluster,"
-                                 " which cannot be left without at least one"
-                                 " group" % self.op.group_name,
+      raise errors.OpPrereqError("Group '%s' is the only group,"
+                                 " cannot be removed" %
+                                 self.op.group_name,
                                   errors.ECODE_STATE)
  
    def BuildHooksEnv(self):
@@ -10957,8 +11074,7 @@ class IAllocator(object):
            "i_pri_up_memory": i_p_up_mem,
            }
          pnr_dyn.update(node_results[nname])
-
-      node_results[nname] = pnr_dyn
+        node_results[nname] = pnr_dyn
  
      return node_results