cmdlib: Fix broken QueryInstanceData for plain instances

[ganeti-local] / lib / cmdlib.py
diff --git a/lib/cmdlib.py b/lib/cmdlib.py

index 899a3e1..089cd36 100644 (file)
--- a/lib/cmdlib.py
+++ b/lib/cmdlib.py
@@ -25,9 +25,7 @@
  
  import os
  import os.path
  
  import os
  import os.path
-import sha
  import time
  import time
-import tempfile
  import re
  import platform
  import logging
  import re
  import platform
  import logging
@@ -40,8 +38,8 @@ from ganeti import hypervisor
  from ganeti import locking
  from ganeti import constants
  from ganeti import objects
  from ganeti import locking
  from ganeti import constants
  from ganeti import objects
-from ganeti import opcodes
  from ganeti import serializer
  from ganeti import serializer
+from ganeti import ssconf
  
  
  class LogicalUnit(object):
  
  
  class LogicalUnit(object):
@@ -49,27 +47,28 @@ class LogicalUnit(object):
  
    Subclasses must follow these rules:
      - implement ExpandNames
  
    Subclasses must follow these rules:
      - implement ExpandNames
-    - implement CheckPrereq
-    - implement Exec
+    - implement CheckPrereq (except when tasklets are used)
+    - implement Exec (except when tasklets are used)
      - implement BuildHooksEnv
      - redefine HPATH and HTYPE
      - optionally redefine their run requirements:
      - implement BuildHooksEnv
      - redefine HPATH and HTYPE
      - optionally redefine their run requirements:
-        REQ_MASTER: the LU needs to run on the master node
          REQ_BGL: the LU needs to hold the Big Ganeti Lock exclusively
  
    Note that all commands require root permissions.
  
          REQ_BGL: the LU needs to hold the Big Ganeti Lock exclusively
  
    Note that all commands require root permissions.
  
+  @ivar dry_run_result: the value (if any) that will be returned to the caller
+      in dry-run mode (signalled by opcode dry_run parameter)
+
    """
    HPATH = None
    HTYPE = None
    _OP_REQP = []
    """
    HPATH = None
    HTYPE = None
    _OP_REQP = []
-  REQ_MASTER = True
    REQ_BGL = True
  
    def __init__(self, processor, op, context, rpc):
      """Constructor for LogicalUnit.
  
    REQ_BGL = True
  
    def __init__(self, processor, op, context, rpc):
      """Constructor for LogicalUnit.
  
-    This needs to be overriden in derived classes in order to check op
+    This needs to be overridden in derived classes in order to check op
      validity.
  
      """
      validity.
  
      """
@@ -81,7 +80,7 @@ class LogicalUnit(object):
      # Dicts used to declare locking needs to mcpu
      self.needed_locks = None
      self.acquired_locks = {}
      # Dicts used to declare locking needs to mcpu
      self.needed_locks = None
      self.acquired_locks = {}
-    self.share_locks = dict(((i, 0) for i in locking.LEVELS))
+    self.share_locks = dict.fromkeys(locking.LEVELS, 0)
      self.add_locks = {}
      self.remove_locks = {}
      # Used to force good behavior when calling helper functions
      self.add_locks = {}
      self.remove_locks = {}
      # Used to force good behavior when calling helper functions
@@ -90,6 +89,12 @@ class LogicalUnit(object):
      # logging
      self.LogWarning = processor.LogWarning
      self.LogInfo = processor.LogInfo
      # logging
      self.LogWarning = processor.LogWarning
      self.LogInfo = processor.LogInfo
+    self.LogStep = processor.LogStep
+    # support for dry-run
+    self.dry_run_result = None
+
+    # Tasklets
+    self.tasklets = None
  
      for attr_name in self._OP_REQP:
        attr_val = getattr(op, attr_name, None)
  
      for attr_name in self._OP_REQP:
        attr_val = getattr(op, attr_name, None)
@@ -97,14 +102,7 @@ class LogicalUnit(object):
          raise errors.OpPrereqError("Required parameter '%s' missing" %
                                     attr_name)
  
          raise errors.OpPrereqError("Required parameter '%s' missing" %
                                     attr_name)
  
-    if not self.cfg.IsCluster():
-      raise errors.OpPrereqError("Cluster not initialized yet,"
-                                 " use 'gnt-cluster init' first.")
-    if self.REQ_MASTER:
-      master = self.cfg.GetMasterNode()
-      if master != utils.HostInfo().name:
-        raise errors.OpPrereqError("Commands must be run on the master"
-                                   " node %s" % master)
+    self.CheckArguments()
  
    def __GetSSH(self):
      """Returns the SshRunner object
  
    def __GetSSH(self):
      """Returns the SshRunner object
@@ -116,6 +114,24 @@ class LogicalUnit(object):
  
    ssh = property(fget=__GetSSH)
  
  
    ssh = property(fget=__GetSSH)
  
+  def CheckArguments(self):
+    """Check syntactic validity for the opcode arguments.
+
+    This method is for doing a simple syntactic check and ensure
+    validity of opcode parameters, without any cluster-related
+    checks. While the same can be accomplished in ExpandNames and/or
+    CheckPrereq, doing these separate is better because:
+
+      - ExpandNames is left as as purely a lock-related function
+      - CheckPrereq is run after we have acquired locks (and possible
+        waited for them)
+
+    The function is allowed to change the self.op attribute so that
+    later methods can no longer worry about missing parameters.
+
+    """
+    pass
+
    def ExpandNames(self):
      """Expand names for this LU.
  
    def ExpandNames(self):
      """Expand names for this LU.
  
@@ -137,6 +153,10 @@ class LogicalUnit(object):
      level you can modify self.share_locks, setting a true value (usually 1) for
      that level. By default locks are not shared.
  
      level you can modify self.share_locks, setting a true value (usually 1) for
      that level. By default locks are not shared.
  
+    This function can also define a list of tasklets, which then will be
+    executed in order instead of the usual LU-level CheckPrereq and Exec
+    functions, if those are not defined by the LU.
+
      Examples::
  
        # Acquire all nodes and one instance
      Examples::
  
        # Acquire all nodes and one instance
@@ -193,7 +213,13 @@ class LogicalUnit(object):
      their canonical form if it hasn't been done by ExpandNames before.
  
      """
      their canonical form if it hasn't been done by ExpandNames before.
  
      """
-    raise NotImplementedError
+    if self.tasklets is not None:
+      for (idx, tl) in enumerate(self.tasklets):
+        logging.debug("Checking prerequisites for tasklet %s/%s",
+                      idx + 1, len(self.tasklets))
+        tl.CheckPrereq()
+    else:
+      raise NotImplementedError
  
    def Exec(self, feedback_fn):
      """Execute the LU.
  
    def Exec(self, feedback_fn):
      """Execute the LU.
@@ -203,7 +229,12 @@ class LogicalUnit(object):
      code, or expected.
  
      """
      code, or expected.
  
      """
-    raise NotImplementedError
+    if self.tasklets is not None:
+      for (idx, tl) in enumerate(self.tasklets):
+        logging.debug("Executing tasklet %s/%s", idx + 1, len(self.tasklets))
+        tl.Exec(feedback_fn)
+    else:
+      raise NotImplementedError
  
    def BuildHooksEnv(self):
      """Build hooks environment for this LU.
  
    def BuildHooksEnv(self):
      """Build hooks environment for this LU.
@@ -327,6 +358,52 @@ class NoHooksLU(LogicalUnit):
    HTYPE = None
  
  
    HTYPE = None
  
  
+class Tasklet:
+  """Tasklet base class.
+
+  Tasklets are subcomponents for LUs. LUs can consist entirely of tasklets or
+  they can mix legacy code with tasklets. Locking needs to be done in the LU,
+  tasklets know nothing about locks.
+
+  Subclasses must follow these rules:
+    - Implement CheckPrereq
+    - Implement Exec
+
+  """
+  def __init__(self, lu):
+    self.lu = lu
+
+    # Shortcuts
+    self.cfg = lu.cfg
+    self.rpc = lu.rpc
+
+  def CheckPrereq(self):
+    """Check prerequisites for this tasklets.
+
+    This method should check whether the prerequisites for the execution of
+    this tasklet are fulfilled. It can do internode communication, but it
+    should be idempotent - no cluster or system changes are allowed.
+
+    The method should raise errors.OpPrereqError in case something is not
+    fulfilled. Its return value is ignored.
+
+    This method should also update all parameters to their canonical form if it
+    hasn't been done before.
+
+    """
+    raise NotImplementedError
+
+  def Exec(self, feedback_fn):
+    """Execute the tasklet.
+
+    This method should implement the actual work. It should raise
+    errors.OpExecError for failures that are somewhat dealt with in code, or
+    expected.
+
+    """
+    raise NotImplementedError
+
+
  def _GetWantedNodes(lu, nodes):
    """Returns list of checked and expanded node names.
  
  def _GetWantedNodes(lu, nodes):
    """Returns list of checked and expanded node names.
  
@@ -382,8 +459,8 @@ def _GetWantedInstances(lu, instances):
        wanted.append(instance)
  
    else:
        wanted.append(instance)
  
    else:
-    wanted = lu.cfg.GetInstanceList()
-  return utils.NiceSort(wanted)
+    wanted = utils.NiceSort(lu.cfg.GetInstanceList())
+  return wanted
  
  
  def _CheckOutputFields(static, dynamic, selected):
  
  
  def _CheckOutputFields(static, dynamic, selected):
@@ -405,8 +482,47 @@ def _CheckOutputFields(static, dynamic, selected):
                                 % ",".join(delta))
  
  
                                 % ",".join(delta))
  
  
+def _CheckBooleanOpField(op, name):
+  """Validates boolean opcode parameters.
+
+  This will ensure that an opcode parameter is either a boolean value,
+  or None (but that it always exists).
+
+  """
+  val = getattr(op, name, None)
+  if not (val is None or isinstance(val, bool)):
+    raise errors.OpPrereqError("Invalid boolean parameter '%s' (%s)" %
+                               (name, str(val)))
+  setattr(op, name, val)
+
+
+def _CheckNodeOnline(lu, node):
+  """Ensure that a given node is online.
+
+  @param lu: the LU on behalf of which we make the check
+  @param node: the node to check
+  @raise errors.OpPrereqError: if the node is offline
+
+  """
+  if lu.cfg.GetNodeInfo(node).offline:
+    raise errors.OpPrereqError("Can't use offline node %s" % node)
+
+
+def _CheckNodeNotDrained(lu, node):
+  """Ensure that a given node is not drained.
+
+  @param lu: the LU on behalf of which we make the check
+  @param node: the node to check
+  @raise errors.OpPrereqError: if the node is drained
+
+  """
+  if lu.cfg.GetNodeInfo(node).drained:
+    raise errors.OpPrereqError("Can't use drained node %s" % node)
+
+
  def _BuildInstanceHookEnv(name, primary_node, secondary_nodes, os_type, status,
  def _BuildInstanceHookEnv(name, primary_node, secondary_nodes, os_type, status,
-                          memory, vcpus, nics):
+                          memory, vcpus, nics, disk_template, disks,
+                          bep, hvp, hypervisor_name):
    """Builds instance related env variables for hooks
  
    This builds the hook environment from individual variables.
    """Builds instance related env variables for hooks
  
    This builds the hook environment from individual variables.
@@ -419,46 +535,103 @@ def _BuildInstanceHookEnv(name, primary_node, secondary_nodes, os_type, status,
    @param secondary_nodes: list of secondary nodes as strings
    @type os_type: string
    @param os_type: the name of the instance's OS
    @param secondary_nodes: list of secondary nodes as strings
    @type os_type: string
    @param os_type: the name of the instance's OS
-  @type status: string
-  @param status: the desired status of the instances
+  @type status: boolean
+  @param status: the should_run status of the instance
    @type memory: string
    @param memory: the memory size of the instance
    @type vcpus: string
    @param vcpus: the count of VCPUs the instance has
    @type nics: list
    @type memory: string
    @param memory: the memory size of the instance
    @type vcpus: string
    @param vcpus: the count of VCPUs the instance has
    @type nics: list
-  @param nics: list of tuples (ip, bridge, mac) representing
-      the NICs the instance  has
+  @param nics: list of tuples (ip, mac, mode, link) representing
+      the NICs the instance has
+  @type disk_template: string
+  @param disk_template: the disk template of the instance
+  @type disks: list
+  @param disks: the list of (size, mode) pairs
+  @type bep: dict
+  @param bep: the backend parameters for the instance
+  @type hvp: dict
+  @param hvp: the hypervisor parameters for the instance
+  @type hypervisor_name: string
+  @param hypervisor_name: the hypervisor for the instance
    @rtype: dict
    @return: the hook environment for this instance
  
    """
    @rtype: dict
    @return: the hook environment for this instance
  
    """
+  if status:
+    str_status = "up"
+  else:
+    str_status = "down"
    env = {
      "OP_TARGET": name,
      "INSTANCE_NAME": name,
      "INSTANCE_PRIMARY": primary_node,
      "INSTANCE_SECONDARIES": " ".join(secondary_nodes),
      "INSTANCE_OS_TYPE": os_type,
    env = {
      "OP_TARGET": name,
      "INSTANCE_NAME": name,
      "INSTANCE_PRIMARY": primary_node,
      "INSTANCE_SECONDARIES": " ".join(secondary_nodes),
      "INSTANCE_OS_TYPE": os_type,
-    "INSTANCE_STATUS": status,
+    "INSTANCE_STATUS": str_status,
      "INSTANCE_MEMORY": memory,
      "INSTANCE_VCPUS": vcpus,
      "INSTANCE_MEMORY": memory,
      "INSTANCE_VCPUS": vcpus,
+    "INSTANCE_DISK_TEMPLATE": disk_template,
+    "INSTANCE_HYPERVISOR": hypervisor_name,
    }
  
    if nics:
      nic_count = len(nics)
    }
  
    if nics:
      nic_count = len(nics)
-    for idx, (ip, bridge, mac) in enumerate(nics):
+    for idx, (ip, mac, mode, link) in enumerate(nics):
        if ip is None:
          ip = ""
        env["INSTANCE_NIC%d_IP" % idx] = ip
        if ip is None:
          ip = ""
        env["INSTANCE_NIC%d_IP" % idx] = ip
-      env["INSTANCE_NIC%d_BRIDGE" % idx] = bridge
-      env["INSTANCE_NIC%d_HWADDR" % idx] = mac
+      env["INSTANCE_NIC%d_MAC" % idx] = mac
+      env["INSTANCE_NIC%d_MODE" % idx] = mode
+      env["INSTANCE_NIC%d_LINK" % idx] = link
+      if mode == constants.NIC_MODE_BRIDGED:
+        env["INSTANCE_NIC%d_BRIDGE" % idx] = link
    else:
      nic_count = 0
  
    env["INSTANCE_NIC_COUNT"] = nic_count
  
    else:
      nic_count = 0
  
    env["INSTANCE_NIC_COUNT"] = nic_count
  
+  if disks:
+    disk_count = len(disks)
+    for idx, (size, mode) in enumerate(disks):
+      env["INSTANCE_DISK%d_SIZE" % idx] = size
+      env["INSTANCE_DISK%d_MODE" % idx] = mode
+  else:
+    disk_count = 0
+
+  env["INSTANCE_DISK_COUNT"] = disk_count
+
+  for source, kind in [(bep, "BE"), (hvp, "HV")]:
+    for key, value in source.items():
+      env["INSTANCE_%s_%s" % (kind, key)] = value
+
    return env
  
  
    return env
  
  
+def _NICListToTuple(lu, nics):
+  """Build a list of nic information tuples.
+
+  This list is suitable to be passed to _BuildInstanceHookEnv or as a return
+  value in LUQueryInstanceData.
+
+  @type lu:  L{LogicalUnit}
+  @param lu: the logical unit on whose behalf we execute
+  @type nics: list of L{objects.NIC}
+  @param nics: list of nics to convert to hooks tuples
+
+  """
+  hooks_nics = []
+  c_nicparams = lu.cfg.GetClusterInfo().nicparams[constants.PP_DEFAULT]
+  for nic in nics:
+    ip = nic.ip
+    mac = nic.mac
+    filled_params = objects.FillDict(c_nicparams, nic.nicparams)
+    mode = filled_params[constants.NIC_MODE]
+    link = filled_params[constants.NIC_LINK]
+    hooks_nics.append((ip, mac, mode, link))
+  return hooks_nics
+
+
  def _BuildInstanceHookEnvByObject(lu, instance, override=None):
    """Builds instance related env variables for hooks from an object.
  
  def _BuildInstanceHookEnvByObject(lu, instance, override=None):
    """Builds instance related env variables for hooks from an object.
  
@@ -474,32 +647,154 @@ def _BuildInstanceHookEnvByObject(lu, instance, override=None):
    @return: the hook environment dictionary
  
    """
    @return: the hook environment dictionary
  
    """
-  bep = lu.cfg.GetClusterInfo().FillBE(instance)
+  cluster = lu.cfg.GetClusterInfo()
+  bep = cluster.FillBE(instance)
+  hvp = cluster.FillHV(instance)
    args = {
      'name': instance.name,
      'primary_node': instance.primary_node,
      'secondary_nodes': instance.secondary_nodes,
      'os_type': instance.os,
    args = {
      'name': instance.name,
      'primary_node': instance.primary_node,
      'secondary_nodes': instance.secondary_nodes,
      'os_type': instance.os,
-    'status': instance.os,
+    'status': instance.admin_up,
      'memory': bep[constants.BE_MEMORY],
      'vcpus': bep[constants.BE_VCPUS],
      'memory': bep[constants.BE_MEMORY],
      'vcpus': bep[constants.BE_VCPUS],
-    'nics': [(nic.ip, nic.bridge, nic.mac) for nic in instance.nics],
+    'nics': _NICListToTuple(lu, instance.nics),
+    'disk_template': instance.disk_template,
+    'disks': [(disk.size, disk.mode) for disk in instance.disks],
+    'bep': bep,
+    'hvp': hvp,
+    'hypervisor_name': instance.hypervisor,
    }
    if override:
      args.update(override)
    return _BuildInstanceHookEnv(**args)
  
  
    }
    if override:
      args.update(override)
    return _BuildInstanceHookEnv(**args)
  
  
-def _CheckInstanceBridgesExist(lu, instance):
+def _AdjustCandidatePool(lu):
+  """Adjust the candidate pool after node operations.
+
+  """
+  mod_list = lu.cfg.MaintainCandidatePool()
+  if mod_list:
+    lu.LogInfo("Promoted nodes to master candidate role: %s",
+               ", ".join(node.name for node in mod_list))
+    for name in mod_list:
+      lu.context.ReaddNode(name)
+  mc_now, mc_max = lu.cfg.GetMasterCandidateStats()
+  if mc_now > mc_max:
+    lu.LogInfo("Note: more nodes are candidates (%d) than desired (%d)" %
+               (mc_now, mc_max))
+
+
+def _CheckNicsBridgesExist(lu, target_nics, target_node,
+                               profile=constants.PP_DEFAULT):
+  """Check that the brigdes needed by a list of nics exist.
+
+  """
+  c_nicparams = lu.cfg.GetClusterInfo().nicparams[profile]
+  paramslist = [objects.FillDict(c_nicparams, nic.nicparams)
+                for nic in target_nics]
+  brlist = [params[constants.NIC_LINK] for params in paramslist
+            if params[constants.NIC_MODE] == constants.NIC_MODE_BRIDGED]
+  if brlist:
+    result = lu.rpc.call_bridges_exist(target_node, brlist)
+    result.Raise("Error checking bridges on destination node '%s'" %
+                 target_node, prereq=True)
+
+
+def _CheckInstanceBridgesExist(lu, instance, node=None):
    """Check that the brigdes needed by an instance exist.
  
    """
    """Check that the brigdes needed by an instance exist.
  
    """
-  # check bridges existance
-  brlist = [nic.bridge for nic in instance.nics]
-  if not lu.rpc.call_bridges_exist(instance.primary_node, brlist):
-    raise errors.OpPrereqError("one or more target bridges %s does not"
-                               " exist on destination node '%s'" %
-                               (brlist, instance.primary_node))
+  if node is None:
+    node = instance.primary_node
+  _CheckNicsBridgesExist(lu, instance.nics, node)
+
+
+def _GetNodeInstancesInner(cfg, fn):
+  return [i for i in cfg.GetAllInstancesInfo().values() if fn(i)]
+
+
+def _GetNodeInstances(cfg, node_name):
+  """Returns a list of all primary and secondary instances on a node.
+
+  """
+
+  return _GetNodeInstancesInner(cfg, lambda inst: node_name in inst.all_nodes)
+
+
+def _GetNodePrimaryInstances(cfg, node_name):
+  """Returns primary instances on a node.
+
+  """
+  return _GetNodeInstancesInner(cfg,
+                                lambda inst: node_name == inst.primary_node)
+
+
+def _GetNodeSecondaryInstances(cfg, node_name):
+  """Returns secondary instances on a node.
+
+  """
+  return _GetNodeInstancesInner(cfg,
+                                lambda inst: node_name in inst.secondary_nodes)
+
+
+def _GetStorageTypeArgs(cfg, storage_type):
+  """Returns the arguments for a storage type.
+
+  """
+  # Special case for file storage
+  if storage_type == constants.ST_FILE:
+    # storage.FileStorage wants a list of storage directories
+    return [[cfg.GetFileStorageDir()]]
+
+  return []
+
+
+def _FindFaultyInstanceDisks(cfg, rpc, instance, node_name, prereq):
+  faulty = []
+
+  for dev in instance.disks:
+    cfg.SetDiskID(dev, node_name)
+
+  result = rpc.call_blockdev_getmirrorstatus(node_name, instance.disks)
+  result.Raise("Failed to get disk status from node %s" % node_name,
+               prereq=prereq)
+
+  for idx, bdev_status in enumerate(result.payload):
+    if bdev_status and bdev_status.ldisk_status == constants.LDS_FAULTY:
+      faulty.append(idx)
+
+  return faulty
+
+
+class LUPostInitCluster(LogicalUnit):
+  """Logical unit for running hooks after cluster initialization.
+
+  """
+  HPATH = "cluster-init"
+  HTYPE = constants.HTYPE_CLUSTER
+  _OP_REQP = []
+
+  def BuildHooksEnv(self):
+    """Build hooks env.
+
+    """
+    env = {"OP_TARGET": self.cfg.GetClusterName()}
+    mn = self.cfg.GetMasterNode()
+    return env, [], [mn]
+
+  def CheckPrereq(self):
+    """No prerequisites to check.
+
+    """
+    return True
+
+  def Exec(self, feedback_fn):
+    """Nothing to do.
+
+    """
+    return True
  
  
  class LUDestroyCluster(NoHooksLU):
  
  
  class LUDestroyCluster(NoHooksLU):
@@ -513,7 +808,7 @@ class LUDestroyCluster(NoHooksLU):
  
      This checks whether the cluster is empty.
  
  
      This checks whether the cluster is empty.
  
-    Any errors are signalled by raising errors.OpPrereqError.
+    Any errors are signaled by raising errors.OpPrereqError.
  
      """
      master = self.cfg.GetMasterNode()
  
      """
      master = self.cfg.GetMasterNode()
@@ -532,8 +827,8 @@ class LUDestroyCluster(NoHooksLU):
  
      """
      master = self.cfg.GetMasterNode()
  
      """
      master = self.cfg.GetMasterNode()
-    if not self.rpc.call_node_stop_master(master, False):
-      raise errors.OpExecError("Could not disable the master role")
+    result = self.rpc.call_node_stop_master(master, False)
+    result.Raise("Could not disable the master role")
      priv_key, pub_key, _ = ssh.GetUserFiles(constants.GANETI_RUNAS)
      utils.CreateBackup(priv_key)
      utils.CreateBackup(pub_key)
      priv_key, pub_key, _ = ssh.GetUserFiles(constants.GANETI_RUNAS)
      utils.CreateBackup(priv_key)
      utils.CreateBackup(pub_key)
@@ -554,105 +849,158 @@ class LUVerifyCluster(LogicalUnit):
        locking.LEVEL_NODE: locking.ALL_SET,
        locking.LEVEL_INSTANCE: locking.ALL_SET,
      }
        locking.LEVEL_NODE: locking.ALL_SET,
        locking.LEVEL_INSTANCE: locking.ALL_SET,
      }
-    self.share_locks = dict(((i, 1) for i in locking.LEVELS))
+    self.share_locks = dict.fromkeys(locking.LEVELS, 1)
  
  
-  def _VerifyNode(self, node, file_list, local_cksum, vglist, node_result,
-                  remote_version, feedback_fn):
+  def _VerifyNode(self, nodeinfo, file_list, local_cksum,
+                  node_result, feedback_fn, master_files,
+                  drbd_map, vg_name):
      """Run multiple tests against a node.
  
      """Run multiple tests against a node.
  
-    Test list::
+    Test list:
  
        - compares ganeti version
  
        - compares ganeti version
-      - checks vg existance and size > 20G
+      - checks vg existence and size > 20G
        - checks config file checksum
        - checks ssh to other nodes
  
        - checks config file checksum
        - checks ssh to other nodes
  
-    @type node: string
-    @param node: the name of the node to check
+    @type nodeinfo: L{objects.Node}
+    @param nodeinfo: the node to check
      @param file_list: required list of files
      @param local_cksum: dictionary of local files and their checksums
      @param file_list: required list of files
      @param local_cksum: dictionary of local files and their checksums
-    @type vglist: dict
-    @param vglist: dictionary of volume group names and their size
      @param node_result: the results from the node
      @param node_result: the results from the node
-    @param remote_version: the RPC version from the remote node
      @param feedback_fn: function used to accumulate results
      @param feedback_fn: function used to accumulate results
+    @param master_files: list of files that only masters should have
+    @param drbd_map: the useddrbd minors for this node, in
+        form of minor: (instance, must_exist) which correspond to instances
+        and their running status
+    @param vg_name: Ganeti Volume Group (result of self.cfg.GetVGName())
  
      """
  
      """
+    node = nodeinfo.name
+
+    # main result, node_result should be a non-empty dict
+    if not node_result or not isinstance(node_result, dict):
+      feedback_fn("  - ERROR: unable to verify node %s." % (node,))
+      return True
+
      # compares ganeti version
      local_version = constants.PROTOCOL_VERSION
      # compares ganeti version
      local_version = constants.PROTOCOL_VERSION
-    if not remote_version:
+    remote_version = node_result.get('version', None)
+    if not (remote_version and isinstance(remote_version, (list, tuple)) and
+            len(remote_version) == 2):
        feedback_fn("  - ERROR: connection to %s failed" % (node))
        return True
  
        feedback_fn("  - ERROR: connection to %s failed" % (node))
        return True
  
-    if local_version != remote_version:
-      feedback_fn("  - ERROR: sw version mismatch: master %s, node(%s) %s" %
-                      (local_version, node, remote_version))
+    if local_version != remote_version[0]:
+      feedback_fn("  - ERROR: incompatible protocol versions: master %s,"
+                  " node %s %s" % (local_version, node, remote_version[0]))
        return True
  
        return True
  
-    # checks vg existance and size > 20G
+    # node seems compatible, we can actually try to look into its results
  
      bad = False
  
      bad = False
-    if not vglist:
-      feedback_fn("  - ERROR: unable to check volume groups on node %s." %
-                      (node,))
-      bad = True
-    else:
-      vgstatus = utils.CheckVolumeGroupSize(vglist, self.cfg.GetVGName(),
-                                            constants.MIN_VG_SIZE)
-      if vgstatus:
-        feedback_fn("  - ERROR: %s on node %s" % (vgstatus, node))
-        bad = True
  
  
-    if not node_result:
-      feedback_fn("  - ERROR: unable to verify node %s." % (node,))
-      return True
+    # full package version
+    if constants.RELEASE_VERSION != remote_version[1]:
+      feedback_fn("  - WARNING: software version mismatch: master %s,"
+                  " node %s %s" %
+                  (constants.RELEASE_VERSION, node, remote_version[1]))
+
+    # checks vg existence and size > 20G
+    if vg_name is not None:
+      vglist = node_result.get(constants.NV_VGLIST, None)
+      if not vglist:
+        feedback_fn("  - ERROR: unable to check volume groups on node %s." %
+                        (node,))
+        bad = True
+      else:
+        vgstatus = utils.CheckVolumeGroupSize(vglist, vg_name,
+                                              constants.MIN_VG_SIZE)
+        if vgstatus:
+          feedback_fn("  - ERROR: %s on node %s" % (vgstatus, node))
+          bad = True
  
      # checks config file checksum
  
      # checks config file checksum
-    # checks ssh to any
  
  
-    if 'filelist' not in node_result:
+    remote_cksum = node_result.get(constants.NV_FILELIST, None)
+    if not isinstance(remote_cksum, dict):
        bad = True
        feedback_fn("  - ERROR: node hasn't returned file checksum data")
      else:
        bad = True
        feedback_fn("  - ERROR: node hasn't returned file checksum data")
      else:
-      remote_cksum = node_result['filelist']
        for file_name in file_list:
        for file_name in file_list:
+        node_is_mc = nodeinfo.master_candidate
+        must_have_file = file_name not in master_files
          if file_name not in remote_cksum:
          if file_name not in remote_cksum:
-          bad = True
-          feedback_fn("  - ERROR: file '%s' missing" % file_name)
+          if node_is_mc or must_have_file:
+            bad = True
+            feedback_fn("  - ERROR: file '%s' missing" % file_name)
          elif remote_cksum[file_name] != local_cksum[file_name]:
          elif remote_cksum[file_name] != local_cksum[file_name]:
-          bad = True
-          feedback_fn("  - ERROR: file '%s' has wrong checksum" % file_name)
+          if node_is_mc or must_have_file:
+            bad = True
+            feedback_fn("  - ERROR: file '%s' has wrong checksum" % file_name)
+          else:
+            # not candidate and this is not a must-have file
+            bad = True
+            feedback_fn("  - ERROR: file '%s' should not exist on non master"
+                        " candidates (and the file is outdated)" % file_name)
+        else:
+          # all good, except non-master/non-must have combination
+          if not node_is_mc and not must_have_file:
+            feedback_fn("  - ERROR: file '%s' should not exist on non master"
+                        " candidates" % file_name)
  
  
-    if 'nodelist' not in node_result:
+    # checks ssh to any
+
+    if constants.NV_NODELIST not in node_result:
        bad = True
        feedback_fn("  - ERROR: node hasn't returned node ssh connectivity data")
      else:
        bad = True
        feedback_fn("  - ERROR: node hasn't returned node ssh connectivity data")
      else:
-      if node_result['nodelist']:
+      if node_result[constants.NV_NODELIST]:
          bad = True
          bad = True
-        for node in node_result['nodelist']:
+        for node in node_result[constants.NV_NODELIST]:
            feedback_fn("  - ERROR: ssh communication with node '%s': %s" %
            feedback_fn("  - ERROR: ssh communication with node '%s': %s" %
-                          (node, node_result['nodelist'][node]))
-    if 'node-net-test' not in node_result:
+                          (node, node_result[constants.NV_NODELIST][node]))
+
+    if constants.NV_NODENETTEST not in node_result:
        bad = True
        feedback_fn("  - ERROR: node hasn't returned node tcp connectivity data")
      else:
        bad = True
        feedback_fn("  - ERROR: node hasn't returned node tcp connectivity data")
      else:
-      if node_result['node-net-test']:
+      if node_result[constants.NV_NODENETTEST]:
          bad = True
          bad = True
-        nlist = utils.NiceSort(node_result['node-net-test'].keys())
+        nlist = utils.NiceSort(node_result[constants.NV_NODENETTEST].keys())
          for node in nlist:
            feedback_fn("  - ERROR: tcp communication with node '%s': %s" %
          for node in nlist:
            feedback_fn("  - ERROR: tcp communication with node '%s': %s" %
-                          (node, node_result['node-net-test'][node]))
+                          (node, node_result[constants.NV_NODENETTEST][node]))
  
  
-    hyp_result = node_result.get('hypervisor', None)
+    hyp_result = node_result.get(constants.NV_HYPERVISOR, None)
      if isinstance(hyp_result, dict):
        for hv_name, hv_result in hyp_result.iteritems():
          if hv_result is not None:
            feedback_fn("  - ERROR: hypervisor %s verify failure: '%s'" %
                        (hv_name, hv_result))
      if isinstance(hyp_result, dict):
        for hv_name, hv_result in hyp_result.iteritems():
          if hv_result is not None:
            feedback_fn("  - ERROR: hypervisor %s verify failure: '%s'" %
                        (hv_name, hv_result))
+
+    # check used drbd list
+    if vg_name is not None:
+      used_minors = node_result.get(constants.NV_DRBDLIST, [])
+      if not isinstance(used_minors, (tuple, list)):
+        feedback_fn("  - ERROR: cannot parse drbd status file: %s" %
+                    str(used_minors))
+      else:
+        for minor, (iname, must_exist) in drbd_map.items():
+          if minor not in used_minors and must_exist:
+            feedback_fn("  - ERROR: drbd minor %d of instance %s is"
+                        " not active" % (minor, iname))
+            bad = True
+        for minor in used_minors:
+          if minor not in drbd_map:
+            feedback_fn("  - ERROR: unallocated drbd minor %d is in use" %
+                        minor)
+            bad = True
+
      return bad
  
    def _VerifyInstance(self, instance, instanceconfig, node_vol_is,
      return bad
  
    def _VerifyInstance(self, instance, instanceconfig, node_vol_is,
-                      node_instance, feedback_fn):
+                      node_instance, feedback_fn, n_offline):
      """Verify an instance.
  
      This function checks to see if the required block devices are
      """Verify an instance.
  
      This function checks to see if the required block devices are
@@ -667,15 +1015,19 @@ class LUVerifyCluster(LogicalUnit):
      instanceconfig.MapLVsByNode(node_vol_should)
  
      for node in node_vol_should:
      instanceconfig.MapLVsByNode(node_vol_should)
  
      for node in node_vol_should:
+      if node in n_offline:
+        # ignore missing volumes on offline nodes
+        continue
        for volume in node_vol_should[node]:
          if node not in node_vol_is or volume not in node_vol_is[node]:
            feedback_fn("  - ERROR: volume %s missing on node %s" %
                            (volume, node))
            bad = True
  
        for volume in node_vol_should[node]:
          if node not in node_vol_is or volume not in node_vol_is[node]:
            feedback_fn("  - ERROR: volume %s missing on node %s" %
                            (volume, node))
            bad = True
  
-    if not instanceconfig.status == 'down':
-      if (node_current not in node_instance or
-          not instance in node_instance[node_current]):
+    if instanceconfig.admin_up:
+      if ((node_current not in node_instance or
+          not instance in node_instance[node_current]) and
+          node_current not in n_offline):
          feedback_fn("  - ERROR: instance %s not running on node %s" %
                          (instance, node_current))
          bad = True
          feedback_fn("  - ERROR: instance %s not running on node %s" %
                          (instance, node_current))
          bad = True
@@ -746,7 +1098,7 @@ class LUVerifyCluster(LogicalUnit):
            if bep[constants.BE_AUTO_BALANCE]:
              needed_mem += bep[constants.BE_MEMORY]
          if nodeinfo['mfree'] < needed_mem:
            if bep[constants.BE_AUTO_BALANCE]:
              needed_mem += bep[constants.BE_MEMORY]
          if nodeinfo['mfree'] < needed_mem:
-          feedback_fn("  - ERROR: not enough memory on node %s to accomodate"
+          feedback_fn("  - ERROR: not enough memory on node %s to accommodate"
                        " failovers should node %s fail" % (node, prinode))
            bad = True
      return bad
                        " failovers should node %s fail" % (node, prinode))
            bad = True
      return bad
@@ -765,13 +1117,17 @@ class LUVerifyCluster(LogicalUnit):
    def BuildHooksEnv(self):
      """Build hooks env.
  
    def BuildHooksEnv(self):
      """Build hooks env.
  
-    Cluster-Verify hooks just rone in the post phase and their failure makes
+    Cluster-Verify hooks just ran in the post phase and their failure makes
      the output be logged in the verify output and the verification to fail.
  
      """
      all_nodes = self.cfg.GetNodeList()
      the output be logged in the verify output and the verification to fail.
  
      """
      all_nodes = self.cfg.GetNodeList()
-    # TODO: populate the environment with useful information for verify hooks
-    env = {}
+    env = {
+      "CLUSTER_TAGS": " ".join(self.cfg.GetClusterInfo().GetTags())
+      }
+    for node in self.cfg.GetAllNodesInfo().values():
+      env["NODE_TAGS_%s" % node.name] = " ".join(node.GetTags())
+
      return env, [], all_nodes
  
    def Exec(self, feedback_fn):
      return env, [], all_nodes
  
    def Exec(self, feedback_fn):
@@ -788,8 +1144,12 @@ class LUVerifyCluster(LogicalUnit):
      nodelist = utils.NiceSort(self.cfg.GetNodeList())
      nodeinfo = [self.cfg.GetNodeInfo(nname) for nname in nodelist]
      instancelist = utils.NiceSort(self.cfg.GetInstanceList())
      nodelist = utils.NiceSort(self.cfg.GetNodeList())
      nodeinfo = [self.cfg.GetNodeInfo(nname) for nname in nodelist]
      instancelist = utils.NiceSort(self.cfg.GetInstanceList())
+    instanceinfo = dict((iname, self.cfg.GetInstanceInfo(iname))
+                        for iname in instancelist)
      i_non_redundant = [] # Non redundant instances
      i_non_a_balanced = [] # Non auto-balanced instances
      i_non_redundant = [] # Non redundant instances
      i_non_a_balanced = [] # Non auto-balanced instances
+    n_offline = [] # List of offline nodes
+    n_drained = [] # List of nodes being drained
      node_volume = {}
      node_instance = {}
      node_info = {}
      node_volume = {}
      node_instance = {}
      node_info = {}
@@ -797,71 +1157,117 @@ class LUVerifyCluster(LogicalUnit):
  
      # FIXME: verify OS list
      # do local checksums
  
      # FIXME: verify OS list
      # do local checksums
-    file_names = []
+    master_files = [constants.CLUSTER_CONF_FILE]
+
+    file_names = ssconf.SimpleStore().GetFileList()
      file_names.append(constants.SSL_CERT_FILE)
      file_names.append(constants.SSL_CERT_FILE)
-    file_names.append(constants.CLUSTER_CONF_FILE)
+    file_names.append(constants.RAPI_CERT_FILE)
+    file_names.extend(master_files)
+
      local_checksums = utils.FingerprintFiles(file_names)
  
      feedback_fn("* Gathering data (%d nodes)" % len(nodelist))
      local_checksums = utils.FingerprintFiles(file_names)
  
      feedback_fn("* Gathering data (%d nodes)" % len(nodelist))
-    all_volumeinfo = self.rpc.call_volume_list(nodelist, vg_name)
-    all_instanceinfo = self.rpc.call_instance_list(nodelist, hypervisors)
-    all_vglist = self.rpc.call_vg_list(nodelist)
      node_verify_param = {
      node_verify_param = {
-      'filelist': file_names,
-      'nodelist': nodelist,
-      'hypervisor': hypervisors,
-      'node-net-test': [(node.name, node.primary_ip, node.secondary_ip)
-                        for node in nodeinfo]
+      constants.NV_FILELIST: file_names,
+      constants.NV_NODELIST: [node.name for node in nodeinfo
+                              if not node.offline],
+      constants.NV_HYPERVISOR: hypervisors,
+      constants.NV_NODENETTEST: [(node.name, node.primary_ip,
+                                  node.secondary_ip) for node in nodeinfo
+                                 if not node.offline],
+      constants.NV_INSTANCELIST: hypervisors,
+      constants.NV_VERSION: None,
+      constants.NV_HVINFO: self.cfg.GetHypervisorType(),
        }
        }
+    if vg_name is not None:
+      node_verify_param[constants.NV_VGLIST] = None
+      node_verify_param[constants.NV_LVLIST] = vg_name
+      node_verify_param[constants.NV_DRBDLIST] = None
      all_nvinfo = self.rpc.call_node_verify(nodelist, node_verify_param,
                                             self.cfg.GetClusterName())
      all_nvinfo = self.rpc.call_node_verify(nodelist, node_verify_param,
                                             self.cfg.GetClusterName())
-    all_rversion = self.rpc.call_version(nodelist)
-    all_ninfo = self.rpc.call_node_info(nodelist, self.cfg.GetVGName(),
-                                        self.cfg.GetHypervisorType())
  
      cluster = self.cfg.GetClusterInfo()
  
      cluster = self.cfg.GetClusterInfo()
-    for node in nodelist:
-      feedback_fn("* Verifying node %s" % node)
-      result = self._VerifyNode(node, file_names, local_checksums,
-                                all_vglist[node], all_nvinfo[node],
-                                all_rversion[node], feedback_fn)
-      bad = bad or result
+    master_node = self.cfg.GetMasterNode()
+    all_drbd_map = self.cfg.ComputeDRBDMap()
  
  
-      # node_volume
-      volumeinfo = all_volumeinfo[node]
+    for node_i in nodeinfo:
+      node = node_i.name
+
+      if node_i.offline:
+        feedback_fn("* Skipping offline node %s" % (node,))
+        n_offline.append(node)
+        continue
+
+      if node == master_node:
+        ntype = "master"
+      elif node_i.master_candidate:
+        ntype = "master candidate"
+      elif node_i.drained:
+        ntype = "drained"
+        n_drained.append(node)
+      else:
+        ntype = "regular"
+      feedback_fn("* Verifying node %s (%s)" % (node, ntype))
+
+      msg = all_nvinfo[node].fail_msg
+      if msg:
+        feedback_fn("  - ERROR: while contacting node %s: %s" % (node, msg))
+        bad = True
+        continue
+
+      nresult = all_nvinfo[node].payload
+      node_drbd = {}
+      for minor, instance in all_drbd_map[node].items():
+        if instance not in instanceinfo:
+          feedback_fn("  - ERROR: ghost instance '%s' in temporary DRBD map" %
+                      instance)
+          # ghost instance should not be running, but otherwise we
+          # don't give double warnings (both ghost instance and
+          # unallocated minor in use)
+          node_drbd[minor] = (instance, False)
+        else:
+          instance = instanceinfo[instance]
+          node_drbd[minor] = (instance.name, instance.admin_up)
+      result = self._VerifyNode(node_i, file_names, local_checksums,
+                                nresult, feedback_fn, master_files,
+                                node_drbd, vg_name)
+      bad = bad or result
  
  
-      if isinstance(volumeinfo, basestring):
+      lvdata = nresult.get(constants.NV_LVLIST, "Missing LV data")
+      if vg_name is None:
+        node_volume[node] = {}
+      elif isinstance(lvdata, basestring):
          feedback_fn("  - ERROR: LVM problem on node %s: %s" %
          feedback_fn("  - ERROR: LVM problem on node %s: %s" %
-                    (node, volumeinfo[-400:].encode('string_escape')))
+                    (node, utils.SafeEncode(lvdata)))
          bad = True
          node_volume[node] = {}
          bad = True
          node_volume[node] = {}
-      elif not isinstance(volumeinfo, dict):
-        feedback_fn("  - ERROR: connection to %s failed" % (node,))
+      elif not isinstance(lvdata, dict):
+        feedback_fn("  - ERROR: connection to %s failed (lvlist)" % (node,))
          bad = True
          continue
        else:
          bad = True
          continue
        else:
-        node_volume[node] = volumeinfo
+        node_volume[node] = lvdata
  
        # node_instance
  
        # node_instance
-      nodeinstance = all_instanceinfo[node]
-      if type(nodeinstance) != list:
-        feedback_fn("  - ERROR: connection to %s failed" % (node,))
+      idata = nresult.get(constants.NV_INSTANCELIST, None)
+      if not isinstance(idata, list):
+        feedback_fn("  - ERROR: connection to %s failed (instancelist)" %
+                    (node,))
          bad = True
          continue
  
          bad = True
          continue
  
-      node_instance[node] = nodeinstance
+      node_instance[node] = idata
  
        # node_info
  
        # node_info
-      nodeinfo = all_ninfo[node]
+      nodeinfo = nresult.get(constants.NV_HVINFO, None)
        if not isinstance(nodeinfo, dict):
        if not isinstance(nodeinfo, dict):
-        feedback_fn("  - ERROR: connection to %s failed" % (node,))
+        feedback_fn("  - ERROR: connection to %s failed (hvinfo)" % (node,))
          bad = True
          continue
  
        try:
          node_info[node] = {
            "mfree": int(nodeinfo['memory_free']),
          bad = True
          continue
  
        try:
          node_info[node] = {
            "mfree": int(nodeinfo['memory_free']),
-          "dfree": int(nodeinfo['vg_free']),
            "pinst": [],
            "sinst": [],
            # dictionary holding all instances this node is secondary for,
            "pinst": [],
            "sinst": [],
            # dictionary holding all instances this node is secondary for,
@@ -872,8 +1278,19 @@ class LUVerifyCluster(LogicalUnit):
            # secondary.
            "sinst-by-pnode": {},
          }
            # secondary.
            "sinst-by-pnode": {},
          }
-      except ValueError:
-        feedback_fn("  - ERROR: invalid value returned from node %s" % (node,))
+        # FIXME: devise a free space model for file based instances as well
+        if vg_name is not None:
+          if (constants.NV_VGLIST not in nresult or
+              vg_name not in nresult[constants.NV_VGLIST]):
+            feedback_fn("  - ERROR: node %s didn't return data for the"
+                        " volume group '%s' - it is either missing or broken" %
+                        (node, vg_name))
+            bad = True
+            continue
+          node_info[node]["dfree"] = int(nresult[constants.NV_VGLIST][vg_name])
+      except (ValueError, KeyError):
+        feedback_fn("  - ERROR: invalid nodeinfo value returned"
+                    " from node %s" % (node,))
          bad = True
          continue
  
          bad = True
          continue
  
@@ -881,10 +1298,11 @@ class LUVerifyCluster(LogicalUnit):
  
      for instance in instancelist:
        feedback_fn("* Verifying instance %s" % instance)
  
      for instance in instancelist:
        feedback_fn("* Verifying instance %s" % instance)
-      inst_config = self.cfg.GetInstanceInfo(instance)
+      inst_config = instanceinfo[instance]
        result =  self._VerifyInstance(instance, inst_config, node_volume,
        result =  self._VerifyInstance(instance, inst_config, node_volume,
-                                     node_instance, feedback_fn)
+                                     node_instance, feedback_fn, n_offline)
        bad = bad or result
        bad = bad or result
+      inst_nodes_offline = []
  
        inst_config.MapLVsByNode(node_vol_should)
  
  
        inst_config.MapLVsByNode(node_vol_should)
  
@@ -893,11 +1311,14 @@ class LUVerifyCluster(LogicalUnit):
        pnode = inst_config.primary_node
        if pnode in node_info:
          node_info[pnode]['pinst'].append(instance)
        pnode = inst_config.primary_node
        if pnode in node_info:
          node_info[pnode]['pinst'].append(instance)
-      else:
+      elif pnode not in n_offline:
          feedback_fn("  - ERROR: instance %s, connection to primary node"
                      " %s failed" % (instance, pnode))
          bad = True
  
          feedback_fn("  - ERROR: instance %s, connection to primary node"
                      " %s failed" % (instance, pnode))
          bad = True
  
+      if pnode in n_offline:
+        inst_nodes_offline.append(pnode)
+
        # If the instance is non-redundant we cannot survive losing its primary
        # node, so we are not N+1 compliant. On the other hand we have no disk
        # templates with more than one secondary so that situation is not well
        # If the instance is non-redundant we cannot survive losing its primary
        # node, so we are not N+1 compliant. On the other hand we have no disk
        # templates with more than one secondary so that situation is not well
@@ -918,9 +1339,18 @@ class LUVerifyCluster(LogicalUnit):
            if pnode not in node_info[snode]['sinst-by-pnode']:
              node_info[snode]['sinst-by-pnode'][pnode] = []
            node_info[snode]['sinst-by-pnode'][pnode].append(instance)
            if pnode not in node_info[snode]['sinst-by-pnode']:
              node_info[snode]['sinst-by-pnode'][pnode] = []
            node_info[snode]['sinst-by-pnode'][pnode].append(instance)
-        else:
+        elif snode not in n_offline:
            feedback_fn("  - ERROR: instance %s, connection to secondary node"
                        " %s failed" % (instance, snode))
            feedback_fn("  - ERROR: instance %s, connection to secondary node"
                        " %s failed" % (instance, snode))
+          bad = True
+        if snode in n_offline:
+          inst_nodes_offline.append(snode)
+
+      if inst_nodes_offline:
+        # warn that the instance lives on offline nodes, and set bad=True
+        feedback_fn("  - ERROR: instance lives on offline node(s) %s" %
+                    ", ".join(inst_nodes_offline))
+        bad = True
  
      feedback_fn("* Verifying orphan volumes")
      result = self._VerifyOrphanVolumes(node_vol_should, node_volume,
  
      feedback_fn("* Verifying orphan volumes")
      result = self._VerifyOrphanVolumes(node_vol_should, node_volume,
@@ -946,10 +1376,16 @@ class LUVerifyCluster(LogicalUnit):
        feedback_fn("  - NOTICE: %d non-auto-balanced instance(s) found."
                    % len(i_non_a_balanced))
  
        feedback_fn("  - NOTICE: %d non-auto-balanced instance(s) found."
                    % len(i_non_a_balanced))
  
+    if n_offline:
+      feedback_fn("  - NOTICE: %d offline node(s) found." % len(n_offline))
+
+    if n_drained:
+      feedback_fn("  - NOTICE: %d drained node(s) found." % len(n_drained))
+
      return not bad
  
    def HooksCallBack(self, phase, hooks_results, feedback_fn, lu_result):
      return not bad
  
    def HooksCallBack(self, phase, hooks_results, feedback_fn, lu_result):
-    """Analize the post-hooks' result
+    """Analyze the post-hooks' result
  
      This method analyses the hook result, handles it, and sends some
      nicely-formatted feedback back to the user.
  
      This method analyses the hook result, handles it, and sends some
      nicely-formatted feedback back to the user.
@@ -976,11 +1412,16 @@ class LUVerifyCluster(LogicalUnit):
          for node_name in hooks_results:
            show_node_header = True
            res = hooks_results[node_name]
          for node_name in hooks_results:
            show_node_header = True
            res = hooks_results[node_name]
-          if res is False or not isinstance(res, list):
-            feedback_fn("    Communication failure")
+          msg = res.fail_msg
+          if msg:
+            if res.offline:
+              # no need to warn or set fail return value
+              continue
+            feedback_fn("    Communication failure in hooks execution: %s" %
+                        msg)
              lu_result = 1
              continue
              lu_result = 1
              continue
-          for script, hkr, output in res:
+          for script, hkr, output in res.payload:
              if hkr == constants.HKR_FAIL:
                # The node header is only shown once, if there are
                # failing hooks on that node
              if hkr == constants.HKR_FAIL:
                # The node header is only shown once, if there are
                # failing hooks on that node
@@ -1007,7 +1448,7 @@ class LUVerifyDisks(NoHooksLU):
        locking.LEVEL_NODE: locking.ALL_SET,
        locking.LEVEL_INSTANCE: locking.ALL_SET,
      }
        locking.LEVEL_NODE: locking.ALL_SET,
        locking.LEVEL_INSTANCE: locking.ALL_SET,
      }
-    self.share_locks = dict(((i, 1) for i in locking.LEVELS))
+    self.share_locks = dict.fromkeys(locking.LEVELS, 1)
  
    def CheckPrereq(self):
      """Check prerequisites.
  
    def CheckPrereq(self):
      """Check prerequisites.
@@ -1020,8 +1461,13 @@ class LUVerifyDisks(NoHooksLU):
    def Exec(self, feedback_fn):
      """Verify integrity of cluster disks.
  
    def Exec(self, feedback_fn):
      """Verify integrity of cluster disks.
  
+    @rtype: tuple of three items
+    @return: a tuple of (dict of node-to-node_error, list of instances
+        which need activate-disks, dict of instance: (node, volume) for
+        missing volumes
+
      """
      """
-    result = res_nodes, res_nlvm, res_instances, res_missing = [], {}, [], {}
+    result = res_nodes, res_instances, res_missing = {}, [], {}
  
      vg_name = self.cfg.GetVGName()
      nodes = utils.NiceSort(self.cfg.GetNodeList())
  
      vg_name = self.cfg.GetVGName()
      nodes = utils.NiceSort(self.cfg.GetNodeList())
@@ -1031,7 +1477,7 @@ class LUVerifyDisks(NoHooksLU):
      nv_dict = {}
      for inst in instances:
        inst_lvs = {}
      nv_dict = {}
      for inst in instances:
        inst_lvs = {}
-      if (inst.status != "up" or
+      if (not inst.admin_up or
            inst.disk_template not in constants.DTS_NET_MIRROR):
          continue
        inst.MapLVsByNode(inst_lvs)
            inst.disk_template not in constants.DTS_NET_MIRROR):
          continue
        inst.MapLVsByNode(inst_lvs)
@@ -1043,23 +1489,21 @@ class LUVerifyDisks(NoHooksLU):
      if not nv_dict:
        return result
  
      if not nv_dict:
        return result
  
-    node_lvs = self.rpc.call_volume_list(nodes, vg_name)
+    node_lvs = self.rpc.call_lv_list(nodes, vg_name)
  
  
-    to_act = set()
      for node in nodes:
        # node_volume
      for node in nodes:
        # node_volume
-      lvs = node_lvs[node]
-
-      if isinstance(lvs, basestring):
-        logging.warning("Error enumerating LVs on node %s: %s", node, lvs)
-        res_nlvm[node] = lvs
-      elif not isinstance(lvs, dict):
-        logging.warning("Connection to node %s failed or invalid data"
-                        " returned", node)
-        res_nodes.append(node)
+      node_res = node_lvs[node]
+      if node_res.offline:
+        continue
+      msg = node_res.fail_msg
+      if msg:
+        logging.warning("Error enumerating LVs on node %s: %s", node, msg)
+        res_nodes[node] = msg
          continue
  
          continue
  
-      for lv_name, (_, lv_inactive, lv_online) in lvs.iteritems():
+      lvs = node_res.payload
+      for lv_name, (_, lv_inactive, lv_online) in lvs.items():
          inst = nv_dict.pop((node, lv_name), None)
          if (not lv_online and inst is not None
              and inst.name not in res_instances):
          inst = nv_dict.pop((node, lv_name), None)
          if (not lv_online and inst is not None
              and inst.name not in res_instances):
@@ -1075,6 +1519,100 @@ class LUVerifyDisks(NoHooksLU):
      return result
  
  
      return result
  
  
+class LURepairDiskSizes(NoHooksLU):
+  """Verifies the cluster disks sizes.
+
+  """
+  _OP_REQP = ["instances"]
+  REQ_BGL = False
+
+  def ExpandNames(self):
+
+    if not isinstance(self.op.instances, list):
+      raise errors.OpPrereqError("Invalid argument type 'instances'")
+
+    if self.op.instances:
+      self.wanted_names = []
+      for name in self.op.instances:
+        full_name = self.cfg.ExpandInstanceName(name)
+        if full_name is None:
+          raise errors.OpPrereqError("Instance '%s' not known" % name)
+        self.wanted_names.append(full_name)
+      self.needed_locks[locking.LEVEL_INSTANCE] = self.wanted_names
+      self.needed_locks = {
+        locking.LEVEL_NODE: [],
+        locking.LEVEL_INSTANCE: self.wanted_names,
+        }
+      self.recalculate_locks[locking.LEVEL_NODE] = constants.LOCKS_REPLACE
+    else:
+      self.wanted_names = None
+      self.needed_locks = {
+        locking.LEVEL_NODE: locking.ALL_SET,
+        locking.LEVEL_INSTANCE: locking.ALL_SET,
+        }
+    self.share_locks = dict(((i, 1) for i in locking.LEVELS))
+
+  def DeclareLocks(self, level):
+    if level == locking.LEVEL_NODE and self.wanted_names is not None:
+      self._LockInstancesNodes(primary_only=True)
+
+  def CheckPrereq(self):
+    """Check prerequisites.
+
+    This only checks the optional instance list against the existing names.
+
+    """
+    if self.wanted_names is None:
+      self.wanted_names = self.acquired_locks[locking.LEVEL_INSTANCE]
+
+    self.wanted_instances = [self.cfg.GetInstanceInfo(name) for name
+                             in self.wanted_names]
+
+  def Exec(self, feedback_fn):
+    """Verify the size of cluster disks.
+
+    """
+    # TODO: check child disks too
+    # TODO: check differences in size between primary/secondary nodes
+    per_node_disks = {}
+    for instance in self.wanted_instances:
+      pnode = instance.primary_node
+      if pnode not in per_node_disks:
+        per_node_disks[pnode] = []
+      for idx, disk in enumerate(instance.disks):
+        per_node_disks[pnode].append((instance, idx, disk))
+
+    changed = []
+    for node, dskl in per_node_disks.items():
+      result = self.rpc.call_blockdev_getsizes(node, [v[2] for v in dskl])
+      if result.failed:
+        self.LogWarning("Failure in blockdev_getsizes call to node"
+                        " %s, ignoring", node)
+        continue
+      if len(result.data) != len(dskl):
+        self.LogWarning("Invalid result from node %s, ignoring node results",
+                        node)
+        continue
+      for ((instance, idx, disk), size) in zip(dskl, result.data):
+        if size is None:
+          self.LogWarning("Disk %d of instance %s did not return size"
+                          " information, ignoring", idx, instance.name)
+          continue
+        if not isinstance(size, (int, long)):
+          self.LogWarning("Disk %d of instance %s did not return valid"
+                          " size information, ignoring", idx, instance.name)
+          continue
+        size = size >> 20
+        if size != disk.size:
+          self.LogInfo("Disk %d of instance %s has mismatched size,"
+                       " correcting: recorded %d, actual %d", idx,
+                       instance.name, disk.size, size)
+          disk.size = size
+          self.cfg.Update(instance)
+          changed.append((instance.name, idx, size))
+    return changed
+
+
  class LURenameCluster(LogicalUnit):
    """Rename the cluster.
  
  class LURenameCluster(LogicalUnit):
    """Rename the cluster.
  
@@ -1124,33 +1662,37 @@ class LURenameCluster(LogicalUnit):
  
      # shutdown the master IP
      master = self.cfg.GetMasterNode()
  
      # shutdown the master IP
      master = self.cfg.GetMasterNode()
-    if not self.rpc.call_node_stop_master(master, False):
-      raise errors.OpExecError("Could not disable the master role")
+    result = self.rpc.call_node_stop_master(master, False)
+    result.Raise("Could not disable the master role")
  
      try:
  
      try:
-      # modify the sstore
-      # TODO: sstore
-      ss.SetKey(ss.SS_MASTER_IP, ip)
-      ss.SetKey(ss.SS_CLUSTER_NAME, clustername)
-
-      # Distribute updated ss config to all nodes
-      myself = self.cfg.GetNodeInfo(master)
-      dist_nodes = self.cfg.GetNodeList()
-      if myself.name in dist_nodes:
-        dist_nodes.remove(myself.name)
-
-      logging.debug("Copying updated ssconf data to all nodes")
-      for keyname in [ss.SS_CLUSTER_NAME, ss.SS_MASTER_IP]:
-        fname = ss.KeyToFilename(keyname)
-        result = self.rpc.call_upload_file(dist_nodes, fname)
-        for to_node in dist_nodes:
-          if not result[to_node]:
-            self.LogWarning("Copy of file %s to node %s failed",
-                            fname, to_node)
+      cluster = self.cfg.GetClusterInfo()
+      cluster.cluster_name = clustername
+      cluster.master_ip = ip
+      self.cfg.Update(cluster)
+
+      # update the known hosts file
+      ssh.WriteKnownHostsFile(self.cfg, constants.SSH_KNOWN_HOSTS_FILE)
+      node_list = self.cfg.GetNodeList()
+      try:
+        node_list.remove(master)
+      except ValueError:
+        pass
+      result = self.rpc.call_upload_file(node_list,
+                                         constants.SSH_KNOWN_HOSTS_FILE)
+      for to_node, to_result in result.iteritems():
+        msg = to_result.fail_msg
+        if msg:
+          msg = ("Copy of file %s to node %s failed: %s" %
+                 (constants.SSH_KNOWN_HOSTS_FILE, to_node, msg))
+          self.proc.LogWarning(msg)
+
      finally:
      finally:
-      if not self.rpc.call_node_start_master(master, False):
+      result = self.rpc.call_node_start_master(master, False, False)
+      msg = result.fail_msg
+      if msg:
          self.LogWarning("Could not re-enable the master role on"
          self.LogWarning("Could not re-enable the master role on"
-                        " the master, please restart manually.")
+                        " the master, please restart manually: %s", msg)
  
  
  def _RecursiveCheckIfLVMBased(disk):
  
  
  def _RecursiveCheckIfLVMBased(disk):
@@ -1158,7 +1700,7 @@ def _RecursiveCheckIfLVMBased(disk):
  
    @type disk: L{objects.Disk}
    @param disk: the disk to check
  
    @type disk: L{objects.Disk}
    @param disk: the disk to check
-  @rtype: booleean
+  @rtype: boolean
    @return: boolean indicating whether a LD_LV dev_type was found or not
  
    """
    @return: boolean indicating whether a LD_LV dev_type was found or not
  
    """
@@ -1178,6 +1720,21 @@ class LUSetClusterParams(LogicalUnit):
    _OP_REQP = []
    REQ_BGL = False
  
    _OP_REQP = []
    REQ_BGL = False
  
+  def CheckArguments(self):
+    """Check parameters
+
+    """
+    if not hasattr(self.op, "candidate_pool_size"):
+      self.op.candidate_pool_size = None
+    if self.op.candidate_pool_size is not None:
+      try:
+        self.op.candidate_pool_size = int(self.op.candidate_pool_size)
+      except (ValueError, TypeError), err:
+        raise errors.OpPrereqError("Invalid candidate_pool_size value: %s" %
+                                   str(err))
+      if self.op.candidate_pool_size < 1:
+        raise errors.OpPrereqError("At least one master candidate needed")
+
    def ExpandNames(self):
      # FIXME: in the future maybe other cluster params won't require checking on
      # all nodes to be modified.
    def ExpandNames(self):
      # FIXME: in the future maybe other cluster params won't require checking on
      # all nodes to be modified.
@@ -1204,8 +1761,6 @@ class LUSetClusterParams(LogicalUnit):
      if the given volume group is valid.
  
      """
      if the given volume group is valid.
  
      """
-    # FIXME: This only works because there is only one parameter that can be
-    # changed or removed.
      if self.op.vg_name is not None and not self.op.vg_name:
        instances = self.cfg.GetAllInstancesInfo().values()
        for inst in instances:
      if self.op.vg_name is not None and not self.op.vg_name:
        instances = self.cfg.GetAllInstancesInfo().values()
        for inst in instances:
@@ -1220,21 +1775,34 @@ class LUSetClusterParams(LogicalUnit):
      if self.op.vg_name:
        vglist = self.rpc.call_vg_list(node_list)
        for node in node_list:
      if self.op.vg_name:
        vglist = self.rpc.call_vg_list(node_list)
        for node in node_list:
-        vgstatus = utils.CheckVolumeGroupSize(vglist[node], self.op.vg_name,
+        msg = vglist[node].fail_msg
+        if msg:
+          # ignoring down node
+          self.LogWarning("Error while gathering data on node %s"
+                          " (ignoring node): %s", node, msg)
+          continue
+        vgstatus = utils.CheckVolumeGroupSize(vglist[node].payload,
+                                              self.op.vg_name,
                                                constants.MIN_VG_SIZE)
          if vgstatus:
            raise errors.OpPrereqError("Error on node '%s': %s" %
                                       (node, vgstatus))
  
      self.cluster = cluster = self.cfg.GetClusterInfo()
                                                constants.MIN_VG_SIZE)
          if vgstatus:
            raise errors.OpPrereqError("Error on node '%s': %s" %
                                       (node, vgstatus))
  
      self.cluster = cluster = self.cfg.GetClusterInfo()
-    # beparams changes do not need validation (we can't validate?),
-    # but we still process here
+    # validate params changes
      if self.op.beparams:
      if self.op.beparams:
-      self.new_beparams = cluster.FillDict(
-        cluster.beparams[constants.BEGR_DEFAULT], self.op.beparams)
+      utils.ForceDictType(self.op.beparams, constants.BES_PARAMETER_TYPES)
+      self.new_beparams = objects.FillDict(
+        cluster.beparams[constants.PP_DEFAULT], self.op.beparams)
+
+    if self.op.nicparams:
+      utils.ForceDictType(self.op.nicparams, constants.NICS_PARAMETER_TYPES)
+      self.new_nicparams = objects.FillDict(
+        cluster.nicparams[constants.PP_DEFAULT], self.op.nicparams)
+      objects.NIC.CheckParameterSyntax(self.new_nicparams)
  
      # hypervisor list/parameters
  
      # hypervisor list/parameters
-    self.new_hvparams = cluster.FillDict(cluster.hvparams, {})
+    self.new_hvparams = objects.FillDict(cluster.hvparams, {})
      if self.op.hvparams:
        if not isinstance(self.op.hvparams, dict):
          raise errors.OpPrereqError("Invalid 'hvparams' parameter on input")
      if self.op.hvparams:
        if not isinstance(self.op.hvparams, dict):
          raise errors.OpPrereqError("Invalid 'hvparams' parameter on input")
@@ -1246,6 +1814,13 @@ class LUSetClusterParams(LogicalUnit):
  
      if self.op.enabled_hypervisors is not None:
        self.hv_list = self.op.enabled_hypervisors
  
      if self.op.enabled_hypervisors is not None:
        self.hv_list = self.op.enabled_hypervisors
+      if not self.hv_list:
+        raise errors.OpPrereqError("Enabled hypervisors list must contain at"
+                                   " least one member")
+      invalid_hvs = set(self.hv_list) - constants.HYPER_TYPES
+      if invalid_hvs:
+        raise errors.OpPrereqError("Enabled hypervisors contains invalid"
+                                   " entries: %s" % invalid_hvs)
      else:
        self.hv_list = cluster.enabled_hypervisors
  
      else:
        self.hv_list = cluster.enabled_hypervisors
  
@@ -1257,6 +1832,7 @@ class LUSetClusterParams(LogicalUnit):
               hv_name in self.op.enabled_hypervisors)):
            # either this is a new hypervisor, or its parameters have changed
            hv_class = hypervisor.GetHypervisor(hv_name)
               hv_name in self.op.enabled_hypervisors)):
            # either this is a new hypervisor, or its parameters have changed
            hv_class = hypervisor.GetHypervisor(hv_name)
+          utils.ForceDictType(hv_params, constants.HVS_PARAMETER_TYPES)
            hv_class.CheckParameterSyntax(hv_params)
            _CheckHVParams(self, node_list, hv_name, hv_params)
  
            hv_class.CheckParameterSyntax(hv_params)
            _CheckHVParams(self, node_list, hv_name, hv_params)
  
@@ -1265,8 +1841,11 @@ class LUSetClusterParams(LogicalUnit):
  
      """
      if self.op.vg_name is not None:
  
      """
      if self.op.vg_name is not None:
-      if self.op.vg_name != self.cfg.GetVGName():
-        self.cfg.SetVGName(self.op.vg_name)
+      new_volume = self.op.vg_name
+      if not new_volume:
+        new_volume = None
+      if new_volume != self.cfg.GetVGName():
+        self.cfg.SetVGName(new_volume)
        else:
          feedback_fn("Cluster LVM configuration already in desired"
                      " state, not changing")
        else:
          feedback_fn("Cluster LVM configuration already in desired"
                      " state, not changing")
@@ -1275,58 +1854,149 @@ class LUSetClusterParams(LogicalUnit):
      if self.op.enabled_hypervisors is not None:
        self.cluster.enabled_hypervisors = self.op.enabled_hypervisors
      if self.op.beparams:
      if self.op.enabled_hypervisors is not None:
        self.cluster.enabled_hypervisors = self.op.enabled_hypervisors
      if self.op.beparams:
-      self.cluster.beparams[constants.BEGR_DEFAULT] = self.new_beparams
+      self.cluster.beparams[constants.PP_DEFAULT] = self.new_beparams
+    if self.op.nicparams:
+      self.cluster.nicparams[constants.PP_DEFAULT] = self.new_nicparams
+
+    if self.op.candidate_pool_size is not None:
+      self.cluster.candidate_pool_size = self.op.candidate_pool_size
+      # we need to update the pool size here, otherwise the save will fail
+      _AdjustCandidatePool(self)
+
      self.cfg.Update(self.cluster)
  
  
      self.cfg.Update(self.cluster)
  
  
-def _WaitForSync(lu, instance, oneshot=False, unlock=False):
-  """Sleep and poll for an instance's disk to sync.
+def _RedistributeAncillaryFiles(lu, additional_nodes=None):
+  """Distribute additional files which are part of the cluster configuration.
  
  
-  """
-  if not instance.disks:
-    return True
+  ConfigWriter takes care of distributing the config and ssconf files, but
+  there are more files which should be distributed to all nodes. This function
+  makes sure those are copied.
  
  
-  if not oneshot:
-    lu.proc.LogInfo("Waiting for instance %s to sync disks." % instance.name)
+  @param lu: calling logical unit
+  @param additional_nodes: list of nodes not in the config to distribute to
  
  
-  node = instance.primary_node
+  """
+  # 1. Gather target nodes
+  myself = lu.cfg.GetNodeInfo(lu.cfg.GetMasterNode())
+  dist_nodes = lu.cfg.GetNodeList()
+  if additional_nodes is not None:
+    dist_nodes.extend(additional_nodes)
+  if myself.name in dist_nodes:
+    dist_nodes.remove(myself.name)
+  # 2. Gather files to distribute
+  dist_files = set([constants.ETC_HOSTS,
+                    constants.SSH_KNOWN_HOSTS_FILE,
+                    constants.RAPI_CERT_FILE,
+                    constants.RAPI_USERS_FILE,
+                    constants.HMAC_CLUSTER_KEY,
+                   ])
+
+  enabled_hypervisors = lu.cfg.GetClusterInfo().enabled_hypervisors
+  for hv_name in enabled_hypervisors:
+    hv_class = hypervisor.GetHypervisor(hv_name)
+    dist_files.update(hv_class.GetAncillaryFiles())
+
+  # 3. Perform the files upload
+  for fname in dist_files:
+    if os.path.exists(fname):
+      result = lu.rpc.call_upload_file(dist_nodes, fname)
+      for to_node, to_result in result.items():
+        msg = to_result.fail_msg
+        if msg:
+          msg = ("Copy of file %s to node %s failed: %s" %
+                 (fname, to_node, msg))
+          lu.proc.LogWarning(msg)
+
+
+class LURedistributeConfig(NoHooksLU):
+  """Force the redistribution of cluster configuration.
+
+  This is a very simple LU.
+
+  """
+  _OP_REQP = []
+  REQ_BGL = False
+
+  def ExpandNames(self):
+    self.needed_locks = {
+      locking.LEVEL_NODE: locking.ALL_SET,
+    }
+    self.share_locks[locking.LEVEL_NODE] = 1
+
+  def CheckPrereq(self):
+    """Check prerequisites.
+
+    """
+
+  def Exec(self, feedback_fn):
+    """Redistribute the configuration.
+
+    """
+    self.cfg.Update(self.cfg.GetClusterInfo())
+    _RedistributeAncillaryFiles(self)
+
+
+def _WaitForSync(lu, instance, oneshot=False, unlock=False):
+  """Sleep and poll for an instance's disk to sync.
+
+  """
+  if not instance.disks:
+    return True
+
+  if not oneshot:
+    lu.proc.LogInfo("Waiting for instance %s to sync disks." % instance.name)
+
+  node = instance.primary_node
  
    for dev in instance.disks:
      lu.cfg.SetDiskID(dev, node)
  
    retries = 0
  
    for dev in instance.disks:
      lu.cfg.SetDiskID(dev, node)
  
    retries = 0
+  degr_retries = 10 # in seconds, as we sleep 1 second each time
    while True:
      max_time = 0
      done = True
      cumul_degraded = False
      rstats = lu.rpc.call_blockdev_getmirrorstatus(node, instance.disks)
    while True:
      max_time = 0
      done = True
      cumul_degraded = False
      rstats = lu.rpc.call_blockdev_getmirrorstatus(node, instance.disks)
-    if not rstats:
-      lu.LogWarning("Can't get any data from node %s", node)
+    msg = rstats.fail_msg
+    if msg:
+      lu.LogWarning("Can't get any data from node %s: %s", node, msg)
        retries += 1
        if retries >= 10:
          raise errors.RemoteError("Can't contact node %s for mirror data,"
                                   " aborting." % node)
        time.sleep(6)
        continue
        retries += 1
        if retries >= 10:
          raise errors.RemoteError("Can't contact node %s for mirror data,"
                                   " aborting." % node)
        time.sleep(6)
        continue
+    rstats = rstats.payload
      retries = 0
      retries = 0
-    for i in range(len(rstats)):
-      mstat = rstats[i]
+    for i, mstat in enumerate(rstats):
        if mstat is None:
          lu.LogWarning("Can't compute data for node %s/%s",
                             node, instance.disks[i].iv_name)
          continue
        if mstat is None:
          lu.LogWarning("Can't compute data for node %s/%s",
                             node, instance.disks[i].iv_name)
          continue
-      # we ignore the ldisk parameter
-      perc_done, est_time, is_degraded, _ = mstat
-      cumul_degraded = cumul_degraded or (is_degraded and perc_done is None)
-      if perc_done is not None:
+
+      cumul_degraded = (cumul_degraded or
+                        (mstat.is_degraded and mstat.sync_percent is None))
+      if mstat.sync_percent is not None:
          done = False
          done = False
-        if est_time is not None:
-          rem_time = "%d estimated seconds remaining" % est_time
-          max_time = est_time
+        if mstat.estimated_time is not None:
+          rem_time = "%d estimated seconds remaining" % mstat.estimated_time
+          max_time = mstat.estimated_time
          else:
            rem_time = "no time estimate"
          lu.proc.LogInfo("- device %s: %5.2f%% done, %s" %
          else:
            rem_time = "no time estimate"
          lu.proc.LogInfo("- device %s: %5.2f%% done, %s" %
-                        (instance.disks[i].iv_name, perc_done, rem_time))
+                        (instance.disks[i].iv_name, mstat.sync_percent, rem_time))
+
+    # if we're done but degraded, let's do a few small retries, to
+    # make sure we see a stable and not transient situation; therefore
+    # we force restart of the loop
+    if (done or oneshot) and cumul_degraded and degr_retries > 0:
+      logging.info("Degraded disks found, %d retries left", degr_retries)
+      degr_retries -= 1
+      time.sleep(1)
+      continue
+
      if done or oneshot:
        break
  
      if done or oneshot:
        break
  
@@ -1346,19 +2016,24 @@ def _CheckDiskConsistency(lu, dev, node, on_primary, ldisk=False):
  
    """
    lu.cfg.SetDiskID(dev, node)
  
    """
    lu.cfg.SetDiskID(dev, node)
-  if ldisk:
-    idx = 6
-  else:
-    idx = 5
  
    result = True
  
    result = True
+
    if on_primary or dev.AssembleOnSecondary():
      rstats = lu.rpc.call_blockdev_find(node, dev)
    if on_primary or dev.AssembleOnSecondary():
      rstats = lu.rpc.call_blockdev_find(node, dev)
-    if not rstats:
-      logging.warning("Node %s: disk degraded, not found or node down", node)
+    msg = rstats.fail_msg
+    if msg:
+      lu.LogWarning("Can't find disk on node %s: %s", node, msg)
+      result = False
+    elif not rstats.payload:
+      lu.LogWarning("Can't find disk on node %s", node)
        result = False
      else:
        result = False
      else:
-      result = result and (not rstats[idx])
+      if ldisk:
+        result = result and rstats.payload.ldisk_status == constants.LDS_OKAY
+      else:
+        result = result and not rstats.payload.is_degraded
+
    if dev.children:
      for child in dev.children:
        result = result and _CheckDiskConsistency(lu, child, node, on_primary)
    if dev.children:
      for child in dev.children:
        result = result and _CheckDiskConsistency(lu, child, node, on_primary)
@@ -1384,9 +2059,11 @@ class LUDiagnoseOS(NoHooksLU):
                         selected=self.op.output_fields)
  
      # Lock all nodes, in shared mode
                         selected=self.op.output_fields)
  
      # Lock all nodes, in shared mode
+    # Temporary removal of locks, should be reverted later
+    # TODO: reintroduce locks when they are lighter-weight
      self.needed_locks = {}
      self.needed_locks = {}
-    self.share_locks[locking.LEVEL_NODE] = 1
-    self.needed_locks[locking.LEVEL_NODE] = locking.ALL_SET
+    #self.share_locks[locking.LEVEL_NODE] = 1
+    #self.needed_locks[locking.LEVEL_NODE] = locking.ALL_SET
  
    def CheckPrereq(self):
      """Check prerequisites.
  
    def CheckPrereq(self):
      """Check prerequisites.
@@ -1401,49 +2078,54 @@ class LUDiagnoseOS(NoHooksLU):
      @param rlist: a map with node names as keys and OS objects as values
  
      @rtype: dict
      @param rlist: a map with node names as keys and OS objects as values
  
      @rtype: dict
-    @returns: a dictionary with osnames as keys and as value another map, with
-        nodes as keys and list of OS objects as values, eg::
+    @return: a dictionary with osnames as keys and as value another map, with
+        nodes as keys and tuples of (path, status, diagnose) as values, eg::
  
  
-          {"debian-etch": {"node1": [<object>,...],
-                           "node2": [<object>,]}
+          {"debian-etch": {"node1": [(/usr/lib/..., True, ""),
+                                     (/srv/..., False, "invalid api")],
+                           "node2": [(/srv/..., True, "")]}
            }
  
      """
      all_os = {}
            }
  
      """
      all_os = {}
-    for node_name, nr in rlist.iteritems():
-      if not nr:
+    # we build here the list of nodes that didn't fail the RPC (at RPC
+    # level), so that nodes with a non-responding node daemon don't
+    # make all OSes invalid
+    good_nodes = [node_name for node_name in rlist
+                  if not rlist[node_name].fail_msg]
+    for node_name, nr in rlist.items():
+      if nr.fail_msg or not nr.payload:
          continue
          continue
-      for os_obj in nr:
-        if os_obj.name not in all_os:
+      for name, path, status, diagnose in nr.payload:
+        if name not in all_os:
            # build a list of nodes for this os containing empty lists
            # for each node in node_list
            # build a list of nodes for this os containing empty lists
            # for each node in node_list
-          all_os[os_obj.name] = {}
-          for nname in node_list:
-            all_os[os_obj.name][nname] = []
-        all_os[os_obj.name][node_name].append(os_obj)
+          all_os[name] = {}
+          for nname in good_nodes:
+            all_os[name][nname] = []
+        all_os[name][node_name].append((path, status, diagnose))
      return all_os
  
    def Exec(self, feedback_fn):
      """Compute the list of OSes.
  
      """
      return all_os
  
    def Exec(self, feedback_fn):
      """Compute the list of OSes.
  
      """
-    node_list = self.acquired_locks[locking.LEVEL_NODE]
-    node_data = self.rpc.call_os_diagnose(node_list)
-    if node_data == False:
-      raise errors.OpExecError("Can't gather the list of OSes")
-    pol = self._DiagnoseByOS(node_list, node_data)
+    valid_nodes = [node for node in self.cfg.GetOnlineNodeList()]
+    node_data = self.rpc.call_os_diagnose(valid_nodes)
+    pol = self._DiagnoseByOS(valid_nodes, node_data)
      output = []
      output = []
-    for os_name, os_data in pol.iteritems():
+    for os_name, os_data in pol.items():
        row = []
        for field in self.op.output_fields:
          if field == "name":
            val = os_name
          elif field == "valid":
        row = []
        for field in self.op.output_fields:
          if field == "name":
            val = os_name
          elif field == "valid":
-          val = utils.all([osl and osl[0] for osl in os_data.values()])
+          val = utils.all([osl and osl[0][1] for osl in os_data.values()])
          elif field == "node_status":
          elif field == "node_status":
+          # this is just a copy of the dict
            val = {}
            val = {}
-          for node_name, nos_list in os_data.iteritems():
-            val[node_name] = [(v.status, v.path) for v in nos_list]
+          for node_name, nos_list in os_data.items():
+            val[node_name] = nos_list
          else:
            raise errors.ParameterError(field)
          row.append(val)
          else:
            raise errors.ParameterError(field)
          row.append(val)
@@ -1483,7 +2165,7 @@ class LURemoveNode(LogicalUnit):
       - it does not have primary or secondary instances
       - it's not the master
  
       - it does not have primary or secondary instances
       - it's not the master
  
-    Any errors are signalled by raising errors.OpPrereqError.
+    Any errors are signaled by raising errors.OpPrereqError.
  
      """
      node = self.cfg.GetNodeInfo(self.cfg.ExpandNodeName(self.op.node_name))
  
      """
      node = self.cfg.GetNodeInfo(self.cfg.ExpandNodeName(self.op.node_name))
@@ -1499,11 +2181,8 @@ class LURemoveNode(LogicalUnit):
  
      for instance_name in instance_list:
        instance = self.cfg.GetInstanceInfo(instance_name)
  
      for instance_name in instance_list:
        instance = self.cfg.GetInstanceInfo(instance_name)
-      if node.name == instance.primary_node:
-        raise errors.OpPrereqError("Instance %s still running on the node,"
-                                   " please remove first." % instance_name)
-      if node.name in instance.secondary_nodes:
-        raise errors.OpPrereqError("Instance %s has node as a secondary,"
+      if node.name in instance.all_nodes:
+        raise errors.OpPrereqError("Instance %s is still running on the node,"
                                     " please remove first." % instance_name)
      self.op.node_name = node.name
      self.node = node
                                     " please remove first." % instance_name)
      self.op.node_name = node.name
      self.node = node
@@ -1518,27 +2197,39 @@ class LURemoveNode(LogicalUnit):
  
      self.context.RemoveNode(node.name)
  
  
      self.context.RemoveNode(node.name)
  
-    self.rpc.call_node_leave_cluster(node.name)
+    result = self.rpc.call_node_leave_cluster(node.name)
+    msg = result.fail_msg
+    if msg:
+      self.LogWarning("Errors encountered on the remote node while leaving"
+                      " the cluster: %s", msg)
+
+    # Promote nodes to master candidate as needed
+    _AdjustCandidatePool(self)
  
  
  class LUQueryNodes(NoHooksLU):
    """Logical unit for querying nodes.
  
    """
  
  
  class LUQueryNodes(NoHooksLU):
    """Logical unit for querying nodes.
  
    """
-  _OP_REQP = ["output_fields", "names"]
+  _OP_REQP = ["output_fields", "names", "use_locking"]
    REQ_BGL = False
    _FIELDS_DYNAMIC = utils.FieldSet(
      "dtotal", "dfree",
      "mtotal", "mnode", "mfree",
      "bootid",
    REQ_BGL = False
    _FIELDS_DYNAMIC = utils.FieldSet(
      "dtotal", "dfree",
      "mtotal", "mnode", "mfree",
      "bootid",
-    "ctotal",
+    "ctotal", "cnodes", "csockets",
      )
  
    _FIELDS_STATIC = utils.FieldSet(
      "name", "pinst_cnt", "sinst_cnt",
      "pinst_list", "sinst_list",
      "pip", "sip", "tags",
      )
  
    _FIELDS_STATIC = utils.FieldSet(
      "name", "pinst_cnt", "sinst_cnt",
      "pinst_list", "sinst_list",
      "pip", "sip", "tags",
-    "serial_no",
+    "serial_no", "ctime", "mtime",
+    "master_candidate",
+    "master",
+    "offline",
+    "drained",
+    "role",
      )
  
    def ExpandNames(self):
      )
  
    def ExpandNames(self):
@@ -1554,7 +2245,8 @@ class LUQueryNodes(NoHooksLU):
      else:
        self.wanted = locking.ALL_SET
  
      else:
        self.wanted = locking.ALL_SET
  
-    self.do_locking = self._FIELDS_STATIC.NonMatching(self.op.output_fields)
+    self.do_node_query = self._FIELDS_STATIC.NonMatching(self.op.output_fields)
+    self.do_locking = self.do_node_query and self.op.use_locking
      if self.do_locking:
        # if we don't request only static fields, we need to lock the nodes
        self.needed_locks[locking.LEVEL_NODE] = self.wanted
      if self.do_locking:
        # if we don't request only static fields, we need to lock the nodes
        self.needed_locks[locking.LEVEL_NODE] = self.wanted
@@ -1589,21 +2281,25 @@ class LUQueryNodes(NoHooksLU):
  
      # begin data gathering
  
  
      # begin data gathering
  
-    if self.do_locking:
+    if self.do_node_query:
        live_data = {}
        node_data = self.rpc.call_node_info(nodenames, self.cfg.GetVGName(),
                                            self.cfg.GetHypervisorType())
        for name in nodenames:
        live_data = {}
        node_data = self.rpc.call_node_info(nodenames, self.cfg.GetVGName(),
                                            self.cfg.GetHypervisorType())
        for name in nodenames:
-        nodeinfo = node_data.get(name, None)
-        if nodeinfo:
+        nodeinfo = node_data[name]
+        if not nodeinfo.fail_msg and nodeinfo.payload:
+          nodeinfo = nodeinfo.payload
+          fn = utils.TryConvert
            live_data[name] = {
            live_data[name] = {
-            "mtotal": utils.TryConvert(int, nodeinfo['memory_total']),
-            "mnode": utils.TryConvert(int, nodeinfo['memory_dom0']),
-            "mfree": utils.TryConvert(int, nodeinfo['memory_free']),
-            "dtotal": utils.TryConvert(int, nodeinfo['vg_size']),
-            "dfree": utils.TryConvert(int, nodeinfo['vg_free']),
-            "ctotal": utils.TryConvert(int, nodeinfo['cpu_total']),
-            "bootid": nodeinfo['bootid'],
+            "mtotal": fn(int, nodeinfo.get('memory_total', None)),
+            "mnode": fn(int, nodeinfo.get('memory_dom0', None)),
+            "mfree": fn(int, nodeinfo.get('memory_free', None)),
+            "dtotal": fn(int, nodeinfo.get('vg_size', None)),
+            "dfree": fn(int, nodeinfo.get('vg_free', None)),
+            "ctotal": fn(int, nodeinfo.get('cpu_total', None)),
+            "bootid": nodeinfo.get('bootid', None),
+            "cnodes": fn(int, nodeinfo.get('cpu_nodes', None)),
+            "csockets": fn(int, nodeinfo.get('cpu_sockets', None)),
              }
          else:
            live_data[name] = {}
              }
          else:
            live_data[name] = {}
@@ -1626,6 +2322,8 @@ class LUQueryNodes(NoHooksLU):
            if secnode in node_to_secondary:
              node_to_secondary[secnode].add(inst.name)
  
            if secnode in node_to_secondary:
              node_to_secondary[secnode].add(inst.name)
  
+    master_node = self.cfg.GetMasterNode()
+
      # end data gathering
  
      output = []
      # end data gathering
  
      output = []
@@ -1650,8 +2348,31 @@ class LUQueryNodes(NoHooksLU):
            val = list(node.GetTags())
          elif field == "serial_no":
            val = node.serial_no
            val = list(node.GetTags())
          elif field == "serial_no":
            val = node.serial_no
+        elif field == "ctime":
+          val = node.ctime
+        elif field == "mtime":
+          val = node.mtime
+        elif field == "master_candidate":
+          val = node.master_candidate
+        elif field == "master":
+          val = node.name == master_node
+        elif field == "offline":
+          val = node.offline
+        elif field == "drained":
+          val = node.drained
          elif self._FIELDS_DYNAMIC.Matches(field):
            val = live_data[node.name].get(field, None)
          elif self._FIELDS_DYNAMIC.Matches(field):
            val = live_data[node.name].get(field, None)
+        elif field == "role":
+          if node.name == master_node:
+            val = "M"
+          elif node.master_candidate:
+            val = "C"
+          elif node.drained:
+            val = "D"
+          elif node.offline:
+            val = "O"
+          else:
+            val = "R"
          else:
            raise errors.ParameterError(field)
          node_output.append(val)
          else:
            raise errors.ParameterError(field)
          node_output.append(val)
@@ -1704,10 +2425,15 @@ class LUQueryNodeVolumes(NoHooksLU):
  
      output = []
      for node in nodenames:
  
      output = []
      for node in nodenames:
-      if node not in volumes or not volumes[node]:
+      nresult = volumes[node]
+      if nresult.offline:
+        continue
+      msg = nresult.fail_msg
+      if msg:
+        self.LogWarning("Can't compute volume data on node %s: %s", node, msg)
          continue
  
          continue
  
-      node_vols = volumes[node][:]
+      node_vols = nresult.payload[:]
        node_vols.sort(key=lambda vol: vol['dev'])
  
        for vol in node_vols:
        node_vols.sort(key=lambda vol: vol['dev'])
  
        for vol in node_vols:
@@ -1741,6 +2467,154 @@ class LUQueryNodeVolumes(NoHooksLU):
      return output
  
  
      return output
  
  
+class LUQueryNodeStorage(NoHooksLU):
+  """Logical unit for getting information on storage units on node(s).
+
+  """
+  _OP_REQP = ["nodes", "storage_type", "output_fields"]
+  REQ_BGL = False
+  _FIELDS_STATIC = utils.FieldSet("node")
+
+  def ExpandNames(self):
+    storage_type = self.op.storage_type
+
+    if storage_type not in constants.VALID_STORAGE_FIELDS:
+      raise errors.OpPrereqError("Unknown storage type: %s" % storage_type)
+
+    dynamic_fields = constants.VALID_STORAGE_FIELDS[storage_type]
+
+    _CheckOutputFields(static=self._FIELDS_STATIC,
+                       dynamic=utils.FieldSet(*dynamic_fields),
+                       selected=self.op.output_fields)
+
+    self.needed_locks = {}
+    self.share_locks[locking.LEVEL_NODE] = 1
+
+    if self.op.nodes:
+      self.needed_locks[locking.LEVEL_NODE] = \
+        _GetWantedNodes(self, self.op.nodes)
+    else:
+      self.needed_locks[locking.LEVEL_NODE] = locking.ALL_SET
+
+  def CheckPrereq(self):
+    """Check prerequisites.
+
+    This checks that the fields required are valid output fields.
+
+    """
+    self.op.name = getattr(self.op, "name", None)
+
+    self.nodes = self.acquired_locks[locking.LEVEL_NODE]
+
+  def Exec(self, feedback_fn):
+    """Computes the list of nodes and their attributes.
+
+    """
+    # Always get name to sort by
+    if constants.SF_NAME in self.op.output_fields:
+      fields = self.op.output_fields[:]
+    else:
+      fields = [constants.SF_NAME] + self.op.output_fields
+
+    # Never ask for node as it's only known to the LU
+    while "node" in fields:
+      fields.remove("node")
+
+    field_idx = dict([(name, idx) for (idx, name) in enumerate(fields)])
+    name_idx = field_idx[constants.SF_NAME]
+
+    st_args = _GetStorageTypeArgs(self.cfg, self.op.storage_type)
+    data = self.rpc.call_storage_list(self.nodes,
+                                      self.op.storage_type, st_args,
+                                      self.op.name, fields)
+
+    result = []
+
+    for node in utils.NiceSort(self.nodes):
+      nresult = data[node]
+      if nresult.offline:
+        continue
+
+      msg = nresult.fail_msg
+      if msg:
+        self.LogWarning("Can't get storage data from node %s: %s", node, msg)
+        continue
+
+      rows = dict([(row[name_idx], row) for row in nresult.payload])
+
+      for name in utils.NiceSort(rows.keys()):
+        row = rows[name]
+
+        out = []
+
+        for field in self.op.output_fields:
+          if field == "node":
+            val = node
+          elif field in field_idx:
+            val = row[field_idx[field]]
+          else:
+            raise errors.ParameterError(field)
+
+          out.append(val)
+
+        result.append(out)
+
+    return result
+
+
+class LUModifyNodeStorage(NoHooksLU):
+  """Logical unit for modifying a storage volume on a node.
+
+  """
+  _OP_REQP = ["node_name", "storage_type", "name", "changes"]
+  REQ_BGL = False
+
+  def CheckArguments(self):
+    node_name = self.cfg.ExpandNodeName(self.op.node_name)
+    if node_name is None:
+      raise errors.OpPrereqError("Invalid node name '%s'" % self.op.node_name)
+
+    self.op.node_name = node_name
+
+    storage_type = self.op.storage_type
+    if storage_type not in constants.VALID_STORAGE_FIELDS:
+      raise errors.OpPrereqError("Unknown storage type: %s" % storage_type)
+
+  def ExpandNames(self):
+    self.needed_locks = {
+      locking.LEVEL_NODE: self.op.node_name,
+      }
+
+  def CheckPrereq(self):
+    """Check prerequisites.
+
+    """
+    storage_type = self.op.storage_type
+
+    try:
+      modifiable = constants.MODIFIABLE_STORAGE_FIELDS[storage_type]
+    except KeyError:
+      raise errors.OpPrereqError("Storage units of type '%s' can not be"
+                                 " modified" % storage_type)
+
+    diff = set(self.op.changes.keys()) - modifiable
+    if diff:
+      raise errors.OpPrereqError("The following fields can not be modified for"
+                                 " storage units of type '%s': %r" %
+                                 (storage_type, list(diff)))
+
+  def Exec(self, feedback_fn):
+    """Computes the list of nodes and their attributes.
+
+    """
+    st_args = _GetStorageTypeArgs(self.cfg, self.op.storage_type)
+    result = self.rpc.call_storage_modify(self.op.node_name,
+                                          self.op.storage_type, st_args,
+                                          self.op.name, self.op.changes)
+    result.Raise("Failed to modify storage unit '%s' on %s" %
+                 (self.op.name, self.op.node_name))
+
+
  class LUAddNode(LogicalUnit):
    """Logical unit for adding node to the cluster.
  
  class LUAddNode(LogicalUnit):
    """Logical unit for adding node to the cluster.
  
@@ -1773,7 +2647,7 @@ class LUAddNode(LogicalUnit):
       - it is resolvable
       - its parameters (single/dual homed) matches the cluster
  
       - it is resolvable
       - its parameters (single/dual homed) matches the cluster
  
-    Any errors are signalled by raising errors.OpPrereqError.
+    Any errors are signaled by raising errors.OpPrereqError.
  
      """
      node_name = self.op.node_name
  
      """
      node_name = self.op.node_name
@@ -1827,7 +2701,7 @@ class LUAddNode(LogicalUnit):
          raise errors.OpPrereqError("The master has a private ip but the"
                                     " new node doesn't have one")
  
          raise errors.OpPrereqError("The master has a private ip but the"
                                     " new node doesn't have one")
  
-    # checks reachablity
+    # checks reachability
      if not utils.TcpPing(primary_ip, constants.DEFAULT_NODED_PORT):
        raise errors.OpPrereqError("Node not reachable by ping")
  
      if not utils.TcpPing(primary_ip, constants.DEFAULT_NODED_PORT):
        raise errors.OpPrereqError("Node not reachable by ping")
  
@@ -1838,9 +2712,25 @@ class LUAddNode(LogicalUnit):
          raise errors.OpPrereqError("Node secondary ip not reachable by TCP"
                                     " based ping to noded port")
  
          raise errors.OpPrereqError("Node secondary ip not reachable by TCP"
                                     " based ping to noded port")
  
-    self.new_node = objects.Node(name=node,
-                                 primary_ip=primary_ip,
-                                 secondary_ip=secondary_ip)
+    cp_size = self.cfg.GetClusterInfo().candidate_pool_size
+    if self.op.readd:
+      exceptions = [node]
+    else:
+      exceptions = []
+    mc_now, mc_max = self.cfg.GetMasterCandidateStats(exceptions)
+    # the new node will increase mc_max with one, so:
+    mc_max = min(mc_max + 1, cp_size)
+    self.master_candidate = mc_now < mc_max
+
+    if self.op.readd:
+      self.new_node = self.cfg.GetNodeInfo(node)
+      assert self.new_node is not None, "Can't retrieve locked node %s" % node
+    else:
+      self.new_node = objects.Node(name=node,
+                                   primary_ip=primary_ip,
+                                   secondary_ip=secondary_ip,
+                                   master_candidate=self.master_candidate,
+                                   offline=False, drained=False)
  
    def Exec(self, feedback_fn):
      """Adds the new node to the cluster.
  
    def Exec(self, feedback_fn):
      """Adds the new node to the cluster.
@@ -1849,18 +2739,30 @@ class LUAddNode(LogicalUnit):
      new_node = self.new_node
      node = new_node.name
  
      new_node = self.new_node
      node = new_node.name
  
+    # for re-adds, reset the offline/drained/master-candidate flags;
+    # we need to reset here, otherwise offline would prevent RPC calls
+    # later in the procedure; this also means that if the re-add
+    # fails, we are left with a non-offlined, broken node
+    if self.op.readd:
+      new_node.drained = new_node.offline = False
+      self.LogInfo("Readding a node, the offline/drained flags were reset")
+      # if we demote the node, we do cleanup later in the procedure
+      new_node.master_candidate = self.master_candidate
+
+    # notify the user about any possible mc promotion
+    if new_node.master_candidate:
+      self.LogInfo("Node will be a master candidate")
+
      # check connectivity
      result = self.rpc.call_version([node])[node]
      # check connectivity
      result = self.rpc.call_version([node])[node]
-    if result:
-      if constants.PROTOCOL_VERSION == result:
-        logging.info("Communication to node %s fine, sw version %s match",
-                     node, result)
-      else:
-        raise errors.OpExecError("Version mismatch master version %s,"
-                                 " node version %s" %
-                                 (constants.PROTOCOL_VERSION, result))
+    result.Raise("Can't get version information from node %s" % node)
+    if constants.PROTOCOL_VERSION == result.payload:
+      logging.info("Communication to node %s fine, sw version %s match",
+                   node, result.payload)
      else:
      else:
-      raise errors.OpExecError("Cannot get version from the new node")
+      raise errors.OpExecError("Version mismatch master version %s,"
+                               " node version %s" %
+                               (constants.PROTOCOL_VERSION, result.payload))
  
      # setup ssh on node
      logging.info("Copy ssh key to node %s", node)
  
      # setup ssh on node
      logging.info("Copy ssh key to node %s", node)
@@ -1880,16 +2782,18 @@ class LUAddNode(LogicalUnit):
      result = self.rpc.call_node_add(node, keyarray[0], keyarray[1],
                                      keyarray[2],
                                      keyarray[3], keyarray[4], keyarray[5])
      result = self.rpc.call_node_add(node, keyarray[0], keyarray[1],
                                      keyarray[2],
                                      keyarray[3], keyarray[4], keyarray[5])
-
-    if not result:
-      raise errors.OpExecError("Cannot transfer ssh keys to the new node")
+    result.Raise("Cannot transfer ssh keys to the new node")
  
      # Add node to our /etc/hosts, and add key to known_hosts
  
      # Add node to our /etc/hosts, and add key to known_hosts
-    utils.AddHostToEtcHosts(new_node.name)
+    if self.cfg.GetClusterInfo().modify_etc_hosts:
+      utils.AddHostToEtcHosts(new_node.name)
  
      if new_node.secondary_ip != new_node.primary_ip:
  
      if new_node.secondary_ip != new_node.primary_ip:
-      if not self.rpc.call_node_has_ip_address(new_node.name,
-                                               new_node.secondary_ip):
+      result = self.rpc.call_node_has_ip_address(new_node.name,
+                                                 new_node.secondary_ip)
+      result.Raise("Failure checking secondary ip on node %s" % new_node.name,
+                   prereq=True)
+      if not result.payload:
          raise errors.OpExecError("Node claims it doesn't have the secondary ip"
                                   " you gave (%s). Please fix and re-run this"
                                   " command." % new_node.secondary_ip)
          raise errors.OpExecError("Node claims it doesn't have the secondary ip"
                                   " you gave (%s). Please fix and re-run this"
                                   " command." % new_node.secondary_ip)
@@ -1903,51 +2807,210 @@ class LUAddNode(LogicalUnit):
      result = self.rpc.call_node_verify(node_verify_list, node_verify_param,
                                         self.cfg.GetClusterName())
      for verifier in node_verify_list:
      result = self.rpc.call_node_verify(node_verify_list, node_verify_param,
                                         self.cfg.GetClusterName())
      for verifier in node_verify_list:
-      if not result[verifier]:
-        raise errors.OpExecError("Cannot communicate with %s's node daemon"
-                                 " for remote verification" % verifier)
-      if result[verifier]['nodelist']:
-        for failed in result[verifier]['nodelist']:
+      result[verifier].Raise("Cannot communicate with node %s" % verifier)
+      nl_payload = result[verifier].payload['nodelist']
+      if nl_payload:
+        for failed in nl_payload:
            feedback_fn("ssh/hostname verification failed %s -> %s" %
            feedback_fn("ssh/hostname verification failed %s -> %s" %
-                      (verifier, result[verifier]['nodelist'][failed]))
+                      (verifier, nl_payload[failed]))
          raise errors.OpExecError("ssh/hostname verification failed.")
  
          raise errors.OpExecError("ssh/hostname verification failed.")
  
-    # Distribute updated /etc/hosts and known_hosts to all nodes,
-    # including the node just added
-    myself = self.cfg.GetNodeInfo(self.cfg.GetMasterNode())
-    dist_nodes = self.cfg.GetNodeList()
-    if not self.op.readd:
-      dist_nodes.append(node)
-    if myself.name in dist_nodes:
-      dist_nodes.remove(myself.name)
-
-    logging.debug("Copying hosts and known_hosts to all nodes")
-    for fname in (constants.ETC_HOSTS, constants.SSH_KNOWN_HOSTS_FILE):
-      result = self.rpc.call_upload_file(dist_nodes, fname)
-      for to_node in dist_nodes:
-        if not result[to_node]:
-          logging.error("Copy of file %s to node %s failed", fname, to_node)
-
-    to_copy = []
-    if constants.HT_XEN_HVM in self.cfg.GetClusterInfo().enabled_hypervisors:
-      to_copy.append(constants.VNC_PASSWORD_FILE)
-    for fname in to_copy:
-      result = self.rpc.call_upload_file([node], fname)
-      if not result[node]:
-        logging.error("Could not copy file %s to node %s", fname, node)
-
      if self.op.readd:
      if self.op.readd:
+      _RedistributeAncillaryFiles(self)
        self.context.ReaddNode(new_node)
        self.context.ReaddNode(new_node)
+      # make sure we redistribute the config
+      self.cfg.Update(new_node)
+      # and make sure the new node will not have old files around
+      if not new_node.master_candidate:
+        result = self.rpc.call_node_demote_from_mc(new_node.name)
+        msg = result.RemoteFailMsg()
+        if msg:
+          self.LogWarning("Node failed to demote itself from master"
+                          " candidate status: %s" % msg)
      else:
      else:
+      _RedistributeAncillaryFiles(self, additional_nodes=[node])
        self.context.AddNode(new_node)
  
  
        self.context.AddNode(new_node)
  
  
+class LUSetNodeParams(LogicalUnit):
+  """Modifies the parameters of a node.
+
+  """
+  HPATH = "node-modify"
+  HTYPE = constants.HTYPE_NODE
+  _OP_REQP = ["node_name"]
+  REQ_BGL = False
+
+  def CheckArguments(self):
+    node_name = self.cfg.ExpandNodeName(self.op.node_name)
+    if node_name is None:
+      raise errors.OpPrereqError("Invalid node name '%s'" % self.op.node_name)
+    self.op.node_name = node_name
+    _CheckBooleanOpField(self.op, 'master_candidate')
+    _CheckBooleanOpField(self.op, 'offline')
+    _CheckBooleanOpField(self.op, 'drained')
+    all_mods = [self.op.offline, self.op.master_candidate, self.op.drained]
+    if all_mods.count(None) == 3:
+      raise errors.OpPrereqError("Please pass at least one modification")
+    if all_mods.count(True) > 1:
+      raise errors.OpPrereqError("Can't set the node into more than one"
+                                 " state at the same time")
+
+  def ExpandNames(self):
+    self.needed_locks = {locking.LEVEL_NODE: self.op.node_name}
+
+  def BuildHooksEnv(self):
+    """Build hooks env.
+
+    This runs on the master node.
+
+    """
+    env = {
+      "OP_TARGET": self.op.node_name,
+      "MASTER_CANDIDATE": str(self.op.master_candidate),
+      "OFFLINE": str(self.op.offline),
+      "DRAINED": str(self.op.drained),
+      }
+    nl = [self.cfg.GetMasterNode(),
+          self.op.node_name]
+    return env, nl, nl
+
+  def CheckPrereq(self):
+    """Check prerequisites.
+
+    This only checks the instance list against the existing names.
+
+    """
+    node = self.node = self.cfg.GetNodeInfo(self.op.node_name)
+
+    if ((self.op.master_candidate == False or self.op.offline == True or
+         self.op.drained == True) and node.master_candidate):
+      # we will demote the node from master_candidate
+      if self.op.node_name == self.cfg.GetMasterNode():
+        raise errors.OpPrereqError("The master node has to be a"
+                                   " master candidate, online and not drained")
+      cp_size = self.cfg.GetClusterInfo().candidate_pool_size
+      num_candidates, _ = self.cfg.GetMasterCandidateStats()
+      if num_candidates <= cp_size:
+        msg = ("Not enough master candidates (desired"
+               " %d, new value will be %d)" % (cp_size, num_candidates-1))
+        if self.op.force:
+          self.LogWarning(msg)
+        else:
+          raise errors.OpPrereqError(msg)
+
+    if (self.op.master_candidate == True and
+        ((node.offline and not self.op.offline == False) or
+         (node.drained and not self.op.drained == False))):
+      raise errors.OpPrereqError("Node '%s' is offline or drained, can't set"
+                                 " to master_candidate" % node.name)
+
+    return
+
+  def Exec(self, feedback_fn):
+    """Modifies a node.
+
+    """
+    node = self.node
+
+    result = []
+    changed_mc = False
+
+    if self.op.offline is not None:
+      node.offline = self.op.offline
+      result.append(("offline", str(self.op.offline)))
+      if self.op.offline == True:
+        if node.master_candidate:
+          node.master_candidate = False
+          changed_mc = True
+          result.append(("master_candidate", "auto-demotion due to offline"))
+        if node.drained:
+          node.drained = False
+          result.append(("drained", "clear drained status due to offline"))
+
+    if self.op.master_candidate is not None:
+      node.master_candidate = self.op.master_candidate
+      changed_mc = True
+      result.append(("master_candidate", str(self.op.master_candidate)))
+      if self.op.master_candidate == False:
+        rrc = self.rpc.call_node_demote_from_mc(node.name)
+        msg = rrc.fail_msg
+        if msg:
+          self.LogWarning("Node failed to demote itself: %s" % msg)
+
+    if self.op.drained is not None:
+      node.drained = self.op.drained
+      result.append(("drained", str(self.op.drained)))
+      if self.op.drained == True:
+        if node.master_candidate:
+          node.master_candidate = False
+          changed_mc = True
+          result.append(("master_candidate", "auto-demotion due to drain"))
+          rrc = self.rpc.call_node_demote_from_mc(node.name)
+          msg = rrc.RemoteFailMsg()
+          if msg:
+            self.LogWarning("Node failed to demote itself: %s" % msg)
+        if node.offline:
+          node.offline = False
+          result.append(("offline", "clear offline status due to drain"))
+
+    # this will trigger configuration file update, if needed
+    self.cfg.Update(node)
+    # this will trigger job queue propagation or cleanup
+    if changed_mc:
+      self.context.ReaddNode(node)
+
+    return result
+
+
+class LUPowercycleNode(NoHooksLU):
+  """Powercycles a node.
+
+  """
+  _OP_REQP = ["node_name", "force"]
+  REQ_BGL = False
+
+  def CheckArguments(self):
+    node_name = self.cfg.ExpandNodeName(self.op.node_name)
+    if node_name is None:
+      raise errors.OpPrereqError("Invalid node name '%s'" % self.op.node_name)
+    self.op.node_name = node_name
+    if node_name == self.cfg.GetMasterNode() and not self.op.force:
+      raise errors.OpPrereqError("The node is the master and the force"
+                                 " parameter was not set")
+
+  def ExpandNames(self):
+    """Locking for PowercycleNode.
+
+    This is a last-resort option and shouldn't block on other
+    jobs. Therefore, we grab no locks.
+
+    """
+    self.needed_locks = {}
+
+  def CheckPrereq(self):
+    """Check prerequisites.
+
+    This LU has no prereqs.
+
+    """
+    pass
+
+  def Exec(self, feedback_fn):
+    """Reboots a node.
+
+    """
+    result = self.rpc.call_node_powercycle(self.op.node_name,
+                                           self.cfg.GetHypervisorType())
+    result.Raise("Failed to schedule the reboot")
+    return result.payload
+
+
  class LUQueryClusterInfo(NoHooksLU):
    """Query cluster configuration.
  
    """
    _OP_REQP = []
  class LUQueryClusterInfo(NoHooksLU):
    """Query cluster configuration.
  
    """
    _OP_REQP = []
-  REQ_MASTER = False
    REQ_BGL = False
  
    def ExpandNames(self):
    REQ_BGL = False
  
    def ExpandNames(self):
@@ -1968,15 +3031,23 @@ class LUQueryClusterInfo(NoHooksLU):
        "software_version": constants.RELEASE_VERSION,
        "protocol_version": constants.PROTOCOL_VERSION,
        "config_version": constants.CONFIG_VERSION,
        "software_version": constants.RELEASE_VERSION,
        "protocol_version": constants.PROTOCOL_VERSION,
        "config_version": constants.CONFIG_VERSION,
-      "os_api_version": constants.OS_API_VERSION,
+      "os_api_version": max(constants.OS_API_VERSIONS),
        "export_version": constants.EXPORT_VERSION,
        "architecture": (platform.architecture()[0], platform.machine()),
        "name": cluster.cluster_name,
        "master": cluster.master_node,
        "export_version": constants.EXPORT_VERSION,
        "architecture": (platform.architecture()[0], platform.machine()),
        "name": cluster.cluster_name,
        "master": cluster.master_node,
-      "default_hypervisor": cluster.default_hypervisor,
+      "default_hypervisor": cluster.enabled_hypervisors[0],
        "enabled_hypervisors": cluster.enabled_hypervisors,
        "enabled_hypervisors": cluster.enabled_hypervisors,
-      "hvparams": cluster.hvparams,
+      "hvparams": dict([(hypervisor_name, cluster.hvparams[hypervisor_name])
+                        for hypervisor_name in cluster.enabled_hypervisors]),
        "beparams": cluster.beparams,
        "beparams": cluster.beparams,
+      "nicparams": cluster.nicparams,
+      "candidate_pool_size": cluster.candidate_pool_size,
+      "master_netdev": cluster.master_netdev,
+      "volume_group_name": cluster.volume_group_name,
+      "file_storage_dir": cluster.file_storage_dir,
+      "ctime": cluster.ctime,
+      "mtime": cluster.mtime,
        }
  
      return result
        }
  
      return result
@@ -2047,19 +3118,25 @@ class LUActivateInstanceDisks(NoHooksLU):
      self.instance = self.cfg.GetInstanceInfo(self.op.instance_name)
      assert self.instance is not None, \
        "Cannot retrieve locked instance %s" % self.op.instance_name
      self.instance = self.cfg.GetInstanceInfo(self.op.instance_name)
      assert self.instance is not None, \
        "Cannot retrieve locked instance %s" % self.op.instance_name
+    _CheckNodeOnline(self, self.instance.primary_node)
+    if not hasattr(self.op, "ignore_size"):
+      self.op.ignore_size = False
  
    def Exec(self, feedback_fn):
      """Activate the disks.
  
      """
  
    def Exec(self, feedback_fn):
      """Activate the disks.
  
      """
-    disks_ok, disks_info = _AssembleInstanceDisks(self, self.instance)
+    disks_ok, disks_info = \
+              _AssembleInstanceDisks(self, self.instance,
+                                     ignore_size=self.op.ignore_size)
      if not disks_ok:
        raise errors.OpExecError("Cannot activate block devices")
  
      return disks_info
  
  
      if not disks_ok:
        raise errors.OpExecError("Cannot activate block devices")
  
      return disks_info
  
  
-def _AssembleInstanceDisks(lu, instance, ignore_secondaries=False):
+def _AssembleInstanceDisks(lu, instance, ignore_secondaries=False,
+                           ignore_size=False):
    """Prepare the block devices for an instance.
  
    This sets up the block devices on all nodes.
    """Prepare the block devices for an instance.
  
    This sets up the block devices on all nodes.
@@ -2071,6 +3148,10 @@ def _AssembleInstanceDisks(lu, instance, ignore_secondaries=False):
    @type ignore_secondaries: boolean
    @param ignore_secondaries: if true, errors on secondary nodes
        won't result in an error return from the function
    @type ignore_secondaries: boolean
    @param ignore_secondaries: if true, errors on secondary nodes
        won't result in an error return from the function
+  @type ignore_size: boolean
+  @param ignore_size: if true, the current known size of the disk
+      will not be used during the disk activation, useful for cases
+      when the size is wrong
    @return: False if the operation failed, otherwise a list of
        (host, instance_visible_name, node_visible_name)
        with the mapping from node devices to instance devices
    @return: False if the operation failed, otherwise a list of
        (host, instance_visible_name, node_visible_name)
        with the mapping from node devices to instance devices
@@ -2091,12 +3172,16 @@ def _AssembleInstanceDisks(lu, instance, ignore_secondaries=False):
    # 1st pass, assemble on all nodes in secondary mode
    for inst_disk in instance.disks:
      for node, node_disk in inst_disk.ComputeNodeTree(instance.primary_node):
    # 1st pass, assemble on all nodes in secondary mode
    for inst_disk in instance.disks:
      for node, node_disk in inst_disk.ComputeNodeTree(instance.primary_node):
+      if ignore_size:
+        node_disk = node_disk.Copy()
+        node_disk.UnsetSize()
        lu.cfg.SetDiskID(node_disk, node)
        result = lu.rpc.call_blockdev_assemble(node, node_disk, iname, False)
        lu.cfg.SetDiskID(node_disk, node)
        result = lu.rpc.call_blockdev_assemble(node, node_disk, iname, False)
-      if not result:
+      msg = result.fail_msg
+      if msg:
          lu.proc.LogWarning("Could not prepare block device %s on node %s"
          lu.proc.LogWarning("Could not prepare block device %s on node %s"
-                           " (is_primary=False, pass=1)",
-                           inst_disk.iv_name, node)
+                           " (is_primary=False, pass=1): %s",
+                           inst_disk.iv_name, node, msg)
          if not ignore_secondaries:
            disks_ok = False
  
          if not ignore_secondaries:
            disks_ok = False
  
@@ -2107,14 +3192,19 @@ def _AssembleInstanceDisks(lu, instance, ignore_secondaries=False):
      for node, node_disk in inst_disk.ComputeNodeTree(instance.primary_node):
        if node != instance.primary_node:
          continue
      for node, node_disk in inst_disk.ComputeNodeTree(instance.primary_node):
        if node != instance.primary_node:
          continue
+      if ignore_size:
+        node_disk = node_disk.Copy()
+        node_disk.UnsetSize()
        lu.cfg.SetDiskID(node_disk, node)
        result = lu.rpc.call_blockdev_assemble(node, node_disk, iname, True)
        lu.cfg.SetDiskID(node_disk, node)
        result = lu.rpc.call_blockdev_assemble(node, node_disk, iname, True)
-      if not result:
+      msg = result.fail_msg
+      if msg:
          lu.proc.LogWarning("Could not prepare block device %s on node %s"
          lu.proc.LogWarning("Could not prepare block device %s on node %s"
-                           " (is_primary=True, pass=2)",
-                           inst_disk.iv_name, node)
+                           " (is_primary=True, pass=2): %s",
+                           inst_disk.iv_name, node, msg)
          disks_ok = False
          disks_ok = False
-    device_info.append((instance.primary_node, inst_disk.iv_name, result))
+    device_info.append((instance.primary_node, inst_disk.iv_name,
+                        result.payload))
  
    # leave the disks configured for the primary node
    # this is a workaround that would be fixed better by
  
    # leave the disks configured for the primary node
    # this is a workaround that would be fixed better by
@@ -2129,7 +3219,7 @@ def _StartInstanceDisks(lu, instance, force):
    """Start the disks of an instance.
  
    """
    """Start the disks of an instance.
  
    """
-  disks_ok, dummy = _AssembleInstanceDisks(lu, instance,
+  disks_ok, _ = _AssembleInstanceDisks(lu, instance,
                                             ignore_secondaries=force)
    if not disks_ok:
      _ShutdownInstanceDisks(lu, instance)
                                             ignore_secondaries=force)
    if not disks_ok:
      _ShutdownInstanceDisks(lu, instance)
@@ -2181,14 +3271,11 @@ def _SafeShutdownInstanceDisks(lu, instance):
    _ShutdownInstanceDisks.
  
    """
    _ShutdownInstanceDisks.
  
    """
-  ins_l = lu.rpc.call_instance_list([instance.primary_node],
-                                      [instance.hypervisor])
-  ins_l = ins_l[instance.primary_node]
-  if not type(ins_l) is list:
-    raise errors.OpExecError("Can't contact node '%s'" %
-                             instance.primary_node)
-
-  if instance.name in ins_l:
+  pnode = instance.primary_node
+  ins_l = lu.rpc.call_instance_list([pnode], [instance.hypervisor])[pnode]
+  ins_l.Raise("Can't contact node %s" % pnode)
+
+  if instance.name in ins_l.payload:
      raise errors.OpExecError("Instance is running, can't shutdown"
                               " block devices.")
  
      raise errors.OpExecError("Instance is running, can't shutdown"
                               " block devices.")
  
@@ -2204,19 +3291,21 @@ def _ShutdownInstanceDisks(lu, instance, ignore_primary=False):
    ignored.
  
    """
    ignored.
  
    """
-  result = True
+  all_result = True
    for disk in instance.disks:
      for node, top_disk in disk.ComputeNodeTree(instance.primary_node):
        lu.cfg.SetDiskID(top_disk, node)
    for disk in instance.disks:
      for node, top_disk in disk.ComputeNodeTree(instance.primary_node):
        lu.cfg.SetDiskID(top_disk, node)
-      if not lu.rpc.call_blockdev_shutdown(node, top_disk):
-        logging.error("Could not shutdown block device %s on node %s",
-                      disk.iv_name, node)
+      result = lu.rpc.call_blockdev_shutdown(node, top_disk)
+      msg = result.fail_msg
+      if msg:
+        lu.LogWarning("Could not shutdown block device %s on node %s: %s",
+                      disk.iv_name, node, msg)
          if not ignore_primary or node != instance.primary_node:
          if not ignore_primary or node != instance.primary_node:
-          result = False
-  return result
+          all_result = False
+  return all_result
  
  
  
  
-def _CheckNodeFreeMemory(lu, node, reason, requested, hypervisor):
+def _CheckNodeFreeMemory(lu, node, reason, requested, hypervisor_name):
    """Checks if a node has enough free memory.
  
    This function check if a given node has the needed amount of free
    """Checks if a node has enough free memory.
  
    This function check if a given node has the needed amount of free
@@ -2232,25 +3321,22 @@ def _CheckNodeFreeMemory(lu, node, reason, requested, hypervisor):
    @param reason: string to use in the error message
    @type requested: C{int}
    @param requested: the amount of memory in MiB to check for
    @param reason: string to use in the error message
    @type requested: C{int}
    @param requested: the amount of memory in MiB to check for
-  @type hypervisor: C{str}
-  @param hypervisor: the hypervisor to ask for memory stats
+  @type hypervisor_name: C{str}
+  @param hypervisor_name: the hypervisor to ask for memory stats
    @raise errors.OpPrereqError: if the node doesn't have enough memory, or
        we cannot check the node
  
    """
    @raise errors.OpPrereqError: if the node doesn't have enough memory, or
        we cannot check the node
  
    """
-  nodeinfo = lu.rpc.call_node_info([node], lu.cfg.GetVGName(), hypervisor)
-  if not nodeinfo or not isinstance(nodeinfo, dict):
-    raise errors.OpPrereqError("Could not contact node %s for resource"
-                             " information" % (node,))
-
-  free_mem = nodeinfo[node].get('memory_free')
+  nodeinfo = lu.rpc.call_node_info([node], lu.cfg.GetVGName(), hypervisor_name)
+  nodeinfo[node].Raise("Can't get data from node %s" % node, prereq=True)
+  free_mem = nodeinfo[node].payload.get('memory_free', None)
    if not isinstance(free_mem, int):
      raise errors.OpPrereqError("Can't compute free memory on node %s, result"
    if not isinstance(free_mem, int):
      raise errors.OpPrereqError("Can't compute free memory on node %s, result"
-                             " was '%s'" % (node, free_mem))
+                               " was '%s'" % (node, free_mem))
    if requested > free_mem:
      raise errors.OpPrereqError("Not enough memory on node %s for %s:"
    if requested > free_mem:
      raise errors.OpPrereqError("Not enough memory on node %s for %s:"
-                             " needed %s MiB, available %s MiB" %
-                             (node, reason, requested, free_mem))
+                               " needed %s MiB, available %s MiB" %
+                               (node, reason, requested, free_mem))
  
  
  class LUStartupInstance(LogicalUnit):
  
  
  class LUStartupInstance(LogicalUnit):
@@ -2275,8 +3361,7 @@ class LUStartupInstance(LogicalUnit):
        "FORCE": self.op.force,
        }
      env.update(_BuildInstanceHookEnvByObject(self, self.instance))
        "FORCE": self.op.force,
        }
      env.update(_BuildInstanceHookEnvByObject(self, self.instance))
-    nl = ([self.cfg.GetMasterNode(), self.instance.primary_node] +
-          list(self.instance.secondary_nodes))
+    nl = [self.cfg.GetMasterNode()] + list(self.instance.all_nodes)
      return env, nl, nl
  
    def CheckPrereq(self):
      return env, nl, nl
  
    def CheckPrereq(self):
@@ -2289,13 +3374,49 @@ class LUStartupInstance(LogicalUnit):
      assert self.instance is not None, \
        "Cannot retrieve locked instance %s" % self.op.instance_name
  
      assert self.instance is not None, \
        "Cannot retrieve locked instance %s" % self.op.instance_name
  
+    # extra beparams
+    self.beparams = getattr(self.op, "beparams", {})
+    if self.beparams:
+      if not isinstance(self.beparams, dict):
+        raise errors.OpPrereqError("Invalid beparams passed: %s, expected"
+                                   " dict" % (type(self.beparams), ))
+      # fill the beparams dict
+      utils.ForceDictType(self.beparams, constants.BES_PARAMETER_TYPES)
+      self.op.beparams = self.beparams
+
+    # extra hvparams
+    self.hvparams = getattr(self.op, "hvparams", {})
+    if self.hvparams:
+      if not isinstance(self.hvparams, dict):
+        raise errors.OpPrereqError("Invalid hvparams passed: %s, expected"
+                                   " dict" % (type(self.hvparams), ))
+
+      # check hypervisor parameter syntax (locally)
+      cluster = self.cfg.GetClusterInfo()
+      utils.ForceDictType(self.hvparams, constants.HVS_PARAMETER_TYPES)
+      filled_hvp = objects.FillDict(cluster.hvparams[instance.hypervisor],
+                                    instance.hvparams)
+      filled_hvp.update(self.hvparams)
+      hv_type = hypervisor.GetHypervisor(instance.hypervisor)
+      hv_type.CheckParameterSyntax(filled_hvp)
+      _CheckHVParams(self, instance.all_nodes, instance.hypervisor, filled_hvp)
+      self.op.hvparams = self.hvparams
+
+    _CheckNodeOnline(self, instance.primary_node)
+
      bep = self.cfg.GetClusterInfo().FillBE(instance)
      bep = self.cfg.GetClusterInfo().FillBE(instance)
-    # check bridges existance
+    # check bridges existence
      _CheckInstanceBridgesExist(self, instance)
  
      _CheckInstanceBridgesExist(self, instance)
  
-    _CheckNodeFreeMemory(self, instance.primary_node,
-                         "starting instance %s" % instance.name,
-                         bep[constants.BE_MEMORY], instance.hypervisor)
+    remote_info = self.rpc.call_instance_info(instance.primary_node,
+                                              instance.name,
+                                              instance.hypervisor)
+    remote_info.Raise("Error checking node %s" % instance.primary_node,
+                      prereq=True)
+    if not remote_info.payload: # not running already
+      _CheckNodeFreeMemory(self, instance.primary_node,
+                           "starting instance %s" % instance.name,
+                           bep[constants.BE_MEMORY], instance.hypervisor)
  
    def Exec(self, feedback_fn):
      """Start the instance.
  
    def Exec(self, feedback_fn):
      """Start the instance.
@@ -2303,7 +3424,6 @@ class LUStartupInstance(LogicalUnit):
      """
      instance = self.instance
      force = self.op.force
      """
      instance = self.instance
      force = self.op.force
-    extra_args = getattr(self.op, "extra_args", "")
  
      self.cfg.MarkInstanceUp(instance.name)
  
  
      self.cfg.MarkInstanceUp(instance.name)
  
@@ -2311,9 +3431,12 @@ class LUStartupInstance(LogicalUnit):
  
      _StartInstanceDisks(self, instance, force)
  
  
      _StartInstanceDisks(self, instance, force)
  
-    if not self.rpc.call_instance_start(node_current, instance, extra_args):
+    result = self.rpc.call_instance_start(node_current, instance,
+                                          self.hvparams, self.beparams)
+    msg = result.fail_msg
+    if msg:
        _ShutdownInstanceDisks(self, instance)
        _ShutdownInstanceDisks(self, instance)
-      raise errors.OpExecError("Could not start instance")
+      raise errors.OpExecError("Could not start instance: %s" % msg)
  
  
  class LURebootInstance(LogicalUnit):
  
  
  class LURebootInstance(LogicalUnit):
@@ -2343,10 +3466,10 @@ class LURebootInstance(LogicalUnit):
      """
      env = {
        "IGNORE_SECONDARIES": self.op.ignore_secondaries,
      """
      env = {
        "IGNORE_SECONDARIES": self.op.ignore_secondaries,
+      "REBOOT_TYPE": self.op.reboot_type,
        }
      env.update(_BuildInstanceHookEnvByObject(self, self.instance))
        }
      env.update(_BuildInstanceHookEnvByObject(self, self.instance))
-    nl = ([self.cfg.GetMasterNode(), self.instance.primary_node] +
-          list(self.instance.secondary_nodes))
+    nl = [self.cfg.GetMasterNode()] + list(self.instance.all_nodes)
      return env, nl, nl
  
    def CheckPrereq(self):
      return env, nl, nl
  
    def CheckPrereq(self):
@@ -2359,7 +3482,9 @@ class LURebootInstance(LogicalUnit):
      assert self.instance is not None, \
        "Cannot retrieve locked instance %s" % self.op.instance_name
  
      assert self.instance is not None, \
        "Cannot retrieve locked instance %s" % self.op.instance_name
  
-    # check bridges existance
+    _CheckNodeOnline(self, instance.primary_node)
+
+    # check bridges existence
      _CheckInstanceBridgesExist(self, instance)
  
    def Exec(self, feedback_fn):
      _CheckInstanceBridgesExist(self, instance)
  
    def Exec(self, feedback_fn):
@@ -2369,23 +3494,27 @@ class LURebootInstance(LogicalUnit):
      instance = self.instance
      ignore_secondaries = self.op.ignore_secondaries
      reboot_type = self.op.reboot_type
      instance = self.instance
      ignore_secondaries = self.op.ignore_secondaries
      reboot_type = self.op.reboot_type
-    extra_args = getattr(self.op, "extra_args", "")
  
      node_current = instance.primary_node
  
      if reboot_type in [constants.INSTANCE_REBOOT_SOFT,
                         constants.INSTANCE_REBOOT_HARD]:
  
      node_current = instance.primary_node
  
      if reboot_type in [constants.INSTANCE_REBOOT_SOFT,
                         constants.INSTANCE_REBOOT_HARD]:
-      if not self.rpc.call_instance_reboot(node_current, instance,
-                                           reboot_type, extra_args):
-        raise errors.OpExecError("Could not reboot instance")
+      for disk in instance.disks:
+        self.cfg.SetDiskID(disk, node_current)
+      result = self.rpc.call_instance_reboot(node_current, instance,
+                                             reboot_type)
+      result.Raise("Could not reboot instance")
      else:
      else:
-      if not self.rpc.call_instance_shutdown(node_current, instance):
-        raise errors.OpExecError("could not shutdown instance for full reboot")
+      result = self.rpc.call_instance_shutdown(node_current, instance)
+      result.Raise("Could not shutdown instance for full reboot")
        _ShutdownInstanceDisks(self, instance)
        _StartInstanceDisks(self, instance, ignore_secondaries)
        _ShutdownInstanceDisks(self, instance)
        _StartInstanceDisks(self, instance, ignore_secondaries)
-      if not self.rpc.call_instance_start(node_current, instance, extra_args):
+      result = self.rpc.call_instance_start(node_current, instance, None, None)
+      msg = result.fail_msg
+      if msg:
          _ShutdownInstanceDisks(self, instance)
          _ShutdownInstanceDisks(self, instance)
-        raise errors.OpExecError("Could not start instance for full reboot")
+        raise errors.OpExecError("Could not start instance for"
+                                 " full reboot: %s" % msg)
  
      self.cfg.MarkInstanceUp(instance.name)
  
  
      self.cfg.MarkInstanceUp(instance.name)
  
@@ -2409,8 +3538,7 @@ class LUShutdownInstance(LogicalUnit):
  
      """
      env = _BuildInstanceHookEnvByObject(self, self.instance)
  
      """
      env = _BuildInstanceHookEnvByObject(self, self.instance)
-    nl = ([self.cfg.GetMasterNode(), self.instance.primary_node] +
-          list(self.instance.secondary_nodes))
+    nl = [self.cfg.GetMasterNode()] + list(self.instance.all_nodes)
      return env, nl, nl
  
    def CheckPrereq(self):
      return env, nl, nl
  
    def CheckPrereq(self):
@@ -2422,6 +3550,7 @@ class LUShutdownInstance(LogicalUnit):
      self.instance = self.cfg.GetInstanceInfo(self.op.instance_name)
      assert self.instance is not None, \
        "Cannot retrieve locked instance %s" % self.op.instance_name
      self.instance = self.cfg.GetInstanceInfo(self.op.instance_name)
      assert self.instance is not None, \
        "Cannot retrieve locked instance %s" % self.op.instance_name
+    _CheckNodeOnline(self, self.instance.primary_node)
  
    def Exec(self, feedback_fn):
      """Shutdown the instance.
  
    def Exec(self, feedback_fn):
      """Shutdown the instance.
@@ -2430,8 +3559,10 @@ class LUShutdownInstance(LogicalUnit):
      instance = self.instance
      node_current = instance.primary_node
      self.cfg.MarkInstanceDown(instance.name)
      instance = self.instance
      node_current = instance.primary_node
      self.cfg.MarkInstanceDown(instance.name)
-    if not self.rpc.call_instance_shutdown(node_current, instance):
-      self.proc.LogWarning("Could not shutdown instance")
+    result = self.rpc.call_instance_shutdown(node_current, instance)
+    msg = result.fail_msg
+    if msg:
+      self.proc.LogWarning("Could not shutdown instance: %s" % msg)
  
      _ShutdownInstanceDisks(self, instance)
  
  
      _ShutdownInstanceDisks(self, instance)
  
@@ -2455,8 +3586,7 @@ class LUReinstallInstance(LogicalUnit):
  
      """
      env = _BuildInstanceHookEnvByObject(self, self.instance)
  
      """
      env = _BuildInstanceHookEnvByObject(self, self.instance)
-    nl = ([self.cfg.GetMasterNode(), self.instance.primary_node] +
-          list(self.instance.secondary_nodes))
+    nl = [self.cfg.GetMasterNode()] + list(self.instance.all_nodes)
      return env, nl, nl
  
    def CheckPrereq(self):
      return env, nl, nl
  
    def CheckPrereq(self):
@@ -2468,17 +3598,20 @@ class LUReinstallInstance(LogicalUnit):
      instance = self.cfg.GetInstanceInfo(self.op.instance_name)
      assert instance is not None, \
        "Cannot retrieve locked instance %s" % self.op.instance_name
      instance = self.cfg.GetInstanceInfo(self.op.instance_name)
      assert instance is not None, \
        "Cannot retrieve locked instance %s" % self.op.instance_name
+    _CheckNodeOnline(self, instance.primary_node)
  
      if instance.disk_template == constants.DT_DISKLESS:
        raise errors.OpPrereqError("Instance '%s' has no disks" %
                                   self.op.instance_name)
  
      if instance.disk_template == constants.DT_DISKLESS:
        raise errors.OpPrereqError("Instance '%s' has no disks" %
                                   self.op.instance_name)
-    if instance.status != "down":
+    if instance.admin_up:
        raise errors.OpPrereqError("Instance '%s' is marked to be up" %
                                   self.op.instance_name)
      remote_info = self.rpc.call_instance_info(instance.primary_node,
                                                instance.name,
                                                instance.hypervisor)
        raise errors.OpPrereqError("Instance '%s' is marked to be up" %
                                   self.op.instance_name)
      remote_info = self.rpc.call_instance_info(instance.primary_node,
                                                instance.name,
                                                instance.hypervisor)
-    if remote_info:
+    remote_info.Raise("Error checking node %s" % instance.primary_node,
+                      prereq=True)
+    if remote_info.payload:
        raise errors.OpPrereqError("Instance '%s' is running on the node %s" %
                                   (self.op.instance_name,
                                    instance.primary_node))
        raise errors.OpPrereqError("Instance '%s' is running on the node %s" %
                                   (self.op.instance_name,
                                    instance.primary_node))
@@ -2491,10 +3624,9 @@ class LUReinstallInstance(LogicalUnit):
        if pnode is None:
          raise errors.OpPrereqError("Primary node '%s' is unknown" %
                                     self.op.pnode)
        if pnode is None:
          raise errors.OpPrereqError("Primary node '%s' is unknown" %
                                     self.op.pnode)
-      os_obj = self.rpc.call_os_get(pnode.name, self.op.os_type)
-      if not os_obj:
-        raise errors.OpPrereqError("OS '%s' not in supported OS list for"
-                                   " primary node"  % self.op.os_type)
+      result = self.rpc.call_os_get(pnode.name, self.op.os_type)
+      result.Raise("OS '%s' not in supported OS list for primary node %s" %
+                   (self.op.os_type, pnode.name), prereq=True)
  
      self.instance = instance
  
  
      self.instance = instance
  
@@ -2512,21 +3644,36 @@ class LUReinstallInstance(LogicalUnit):
      _StartInstanceDisks(self, inst, None)
      try:
        feedback_fn("Running the instance OS create scripts...")
      _StartInstanceDisks(self, inst, None)
      try:
        feedback_fn("Running the instance OS create scripts...")
-      if not self.rpc.call_instance_os_add(inst.primary_node, inst):
-        raise errors.OpExecError("Could not install OS for instance %s"
-                                 " on node %s" %
-                                 (inst.name, inst.primary_node))
+      result = self.rpc.call_instance_os_add(inst.primary_node, inst, True)
+      result.Raise("Could not install OS for instance %s on node %s" %
+                   (inst.name, inst.primary_node))
      finally:
        _ShutdownInstanceDisks(self, inst)
  
  
      finally:
        _ShutdownInstanceDisks(self, inst)
  
  
-class LURenameInstance(LogicalUnit):
-  """Rename an instance.
+class LURecreateInstanceDisks(LogicalUnit):
+  """Recreate an instance's missing disks.
  
    """
  
    """
-  HPATH = "instance-rename"
+  HPATH = "instance-recreate-disks"
    HTYPE = constants.HTYPE_INSTANCE
    HTYPE = constants.HTYPE_INSTANCE
-  _OP_REQP = ["instance_name", "new_name"]
+  _OP_REQP = ["instance_name", "disks"]
+  REQ_BGL = False
+
+  def CheckArguments(self):
+    """Check the arguments.
+
+    """
+    if not isinstance(self.op.disks, list):
+      raise errors.OpPrereqError("Invalid disks parameter")
+    for item in self.op.disks:
+      if (not isinstance(item, int) or
+          item < 0):
+        raise errors.OpPrereqError("Invalid disk specification '%s'" %
+                                   str(item))
+
+  def ExpandNames(self):
+    self._ExpandAndLockInstance()
  
    def BuildHooksEnv(self):
      """Build hooks env.
  
    def BuildHooksEnv(self):
      """Build hooks env.
@@ -2535,9 +3682,7 @@ class LURenameInstance(LogicalUnit):
  
      """
      env = _BuildInstanceHookEnvByObject(self, self.instance)
  
      """
      env = _BuildInstanceHookEnvByObject(self, self.instance)
-    env["INSTANCE_NEW_NAME"] = self.op.new_name
-    nl = ([self.cfg.GetMasterNode(), self.instance.primary_node] +
-          list(self.instance.secondary_nodes))
+    nl = [self.cfg.GetMasterNode()] + list(self.instance.all_nodes)
      return env, nl, nl
  
    def CheckPrereq(self):
      return env, nl, nl
  
    def CheckPrereq(self):
@@ -2546,34 +3691,106 @@ class LURenameInstance(LogicalUnit):
      This checks that the instance is in the cluster and is not running.
  
      """
      This checks that the instance is in the cluster and is not running.
  
      """
-    instance = self.cfg.GetInstanceInfo(
-      self.cfg.ExpandInstanceName(self.op.instance_name))
-    if instance is None:
-      raise errors.OpPrereqError("Instance '%s' not known" %
+    instance = self.cfg.GetInstanceInfo(self.op.instance_name)
+    assert instance is not None, \
+      "Cannot retrieve locked instance %s" % self.op.instance_name
+    _CheckNodeOnline(self, instance.primary_node)
+
+    if instance.disk_template == constants.DT_DISKLESS:
+      raise errors.OpPrereqError("Instance '%s' has no disks" %
                                   self.op.instance_name)
                                   self.op.instance_name)
-    if instance.status != "down":
+    if instance.admin_up:
        raise errors.OpPrereqError("Instance '%s' is marked to be up" %
                                   self.op.instance_name)
      remote_info = self.rpc.call_instance_info(instance.primary_node,
                                                instance.name,
                                                instance.hypervisor)
        raise errors.OpPrereqError("Instance '%s' is marked to be up" %
                                   self.op.instance_name)
      remote_info = self.rpc.call_instance_info(instance.primary_node,
                                                instance.name,
                                                instance.hypervisor)
-    if remote_info:
+    remote_info.Raise("Error checking node %s" % instance.primary_node,
+                      prereq=True)
+    if remote_info.payload:
        raise errors.OpPrereqError("Instance '%s' is running on the node %s" %
                                   (self.op.instance_name,
                                    instance.primary_node))
        raise errors.OpPrereqError("Instance '%s' is running on the node %s" %
                                   (self.op.instance_name,
                                    instance.primary_node))
-    self.instance = instance
  
  
-    # new name verification
-    name_info = utils.HostInfo(self.op.new_name)
+    if not self.op.disks:
+      self.op.disks = range(len(instance.disks))
+    else:
+      for idx in self.op.disks:
+        if idx >= len(instance.disks):
+          raise errors.OpPrereqError("Invalid disk index passed '%s'" % idx)
  
  
-    self.op.new_name = new_name = name_info.name
-    instance_list = self.cfg.GetInstanceList()
-    if new_name in instance_list:
-      raise errors.OpPrereqError("Instance '%s' is already in the cluster" %
-                                 new_name)
+    self.instance = instance
  
  
-    if not getattr(self.op, "ignore_ip", False):
-      if utils.TcpPing(name_info.ip, constants.DEFAULT_NODED_PORT):
+  def Exec(self, feedback_fn):
+    """Recreate the disks.
+
+    """
+    to_skip = []
+    for idx, disk in enumerate(self.instance.disks):
+      if idx not in self.op.disks: # disk idx has not been passed in
+        to_skip.append(idx)
+        continue
+
+    _CreateDisks(self, self.instance, to_skip=to_skip)
+
+
+class LURenameInstance(LogicalUnit):
+  """Rename an instance.
+
+  """
+  HPATH = "instance-rename"
+  HTYPE = constants.HTYPE_INSTANCE
+  _OP_REQP = ["instance_name", "new_name"]
+
+  def BuildHooksEnv(self):
+    """Build hooks env.
+
+    This runs on master, primary and secondary nodes of the instance.
+
+    """
+    env = _BuildInstanceHookEnvByObject(self, self.instance)
+    env["INSTANCE_NEW_NAME"] = self.op.new_name
+    nl = [self.cfg.GetMasterNode()] + list(self.instance.all_nodes)
+    return env, nl, nl
+
+  def CheckPrereq(self):
+    """Check prerequisites.
+
+    This checks that the instance is in the cluster and is not running.
+
+    """
+    instance = self.cfg.GetInstanceInfo(
+      self.cfg.ExpandInstanceName(self.op.instance_name))
+    if instance is None:
+      raise errors.OpPrereqError("Instance '%s' not known" %
+                                 self.op.instance_name)
+    _CheckNodeOnline(self, instance.primary_node)
+
+    if instance.admin_up:
+      raise errors.OpPrereqError("Instance '%s' is marked to be up" %
+                                 self.op.instance_name)
+    remote_info = self.rpc.call_instance_info(instance.primary_node,
+                                              instance.name,
+                                              instance.hypervisor)
+    remote_info.Raise("Error checking node %s" % instance.primary_node,
+                      prereq=True)
+    if remote_info.payload:
+      raise errors.OpPrereqError("Instance '%s' is running on the node %s" %
+                                 (self.op.instance_name,
+                                  instance.primary_node))
+    self.instance = instance
+
+    # new name verification
+    name_info = utils.HostInfo(self.op.new_name)
+
+    self.op.new_name = new_name = name_info.name
+    instance_list = self.cfg.GetInstanceList()
+    if new_name in instance_list:
+      raise errors.OpPrereqError("Instance '%s' is already in the cluster" %
+                                 new_name)
+
+    if not getattr(self.op, "ignore_ip", False):
+      if utils.TcpPing(name_info.ip, constants.DEFAULT_NODED_PORT):
          raise errors.OpPrereqError("IP %s of instance %s already in use" %
                                     (name_info.ip, new_name))
  
          raise errors.OpPrereqError("IP %s of instance %s already in use" %
                                     (name_info.ip, new_name))
  
@@ -2601,27 +3818,20 @@ class LURenameInstance(LogicalUnit):
        result = self.rpc.call_file_storage_dir_rename(inst.primary_node,
                                                       old_file_storage_dir,
                                                       new_file_storage_dir)
        result = self.rpc.call_file_storage_dir_rename(inst.primary_node,
                                                       old_file_storage_dir,
                                                       new_file_storage_dir)
-
-      if not result:
-        raise errors.OpExecError("Could not connect to node '%s' to rename"
-                                 " directory '%s' to '%s' (but the instance"
-                                 " has been renamed in Ganeti)" % (
-                                 inst.primary_node, old_file_storage_dir,
-                                 new_file_storage_dir))
-
-      if not result[0]:
-        raise errors.OpExecError("Could not rename directory '%s' to '%s'"
-                                 " (but the instance has been renamed in"
-                                 " Ganeti)" % (old_file_storage_dir,
-                                               new_file_storage_dir))
+      result.Raise("Could not rename on node %s directory '%s' to '%s'"
+                   " (but the instance has been renamed in Ganeti)" %
+                   (inst.primary_node, old_file_storage_dir,
+                    new_file_storage_dir))
  
      _StartInstanceDisks(self, inst, None)
      try:
  
      _StartInstanceDisks(self, inst, None)
      try:
-      if not self.rpc.call_instance_run_rename(inst.primary_node, inst,
-                                               old_name):
+      result = self.rpc.call_instance_run_rename(inst.primary_node, inst,
+                                                 old_name)
+      msg = result.fail_msg
+      if msg:
          msg = ("Could not run OS rename script for instance %s on node %s"
          msg = ("Could not run OS rename script for instance %s on node %s"
-               " (but the instance has been renamed in Ganeti)" %
-               (inst.name, inst.primary_node))
+               " (but the instance has been renamed in Ganeti): %s" %
+               (inst.name, inst.primary_node, msg))
          self.proc.LogWarning(msg)
      finally:
        _ShutdownInstanceDisks(self, inst)
          self.proc.LogWarning(msg)
      finally:
        _ShutdownInstanceDisks(self, inst)
@@ -2673,12 +3883,15 @@ class LURemoveInstance(LogicalUnit):
      logging.info("Shutting down instance %s on node %s",
                   instance.name, instance.primary_node)
  
      logging.info("Shutting down instance %s on node %s",
                   instance.name, instance.primary_node)
  
-    if not self.rpc.call_instance_shutdown(instance.primary_node, instance):
+    result = self.rpc.call_instance_shutdown(instance.primary_node, instance)
+    msg = result.fail_msg
+    if msg:
        if self.op.ignore_failures:
        if self.op.ignore_failures:
-        feedback_fn("Warning: can't shutdown instance")
+        feedback_fn("Warning: can't shutdown instance: %s" % msg)
        else:
        else:
-        raise errors.OpExecError("Could not shutdown instance %s on node %s" %
-                                 (instance.name, instance.primary_node))
+        raise errors.OpExecError("Could not shutdown instance %s on"
+                                 " node %s: %s" %
+                                 (instance.name, instance.primary_node, msg))
  
      logging.info("Removing block devices for instance %s", instance.name)
  
  
      logging.info("Removing block devices for instance %s", instance.name)
  
@@ -2698,19 +3911,23 @@ class LUQueryInstances(NoHooksLU):
    """Logical unit for querying instances.
  
    """
    """Logical unit for querying instances.
  
    """
-  _OP_REQP = ["output_fields", "names"]
+  _OP_REQP = ["output_fields", "names", "use_locking"]
    REQ_BGL = False
    _FIELDS_STATIC = utils.FieldSet(*["name", "os", "pnode", "snodes",
    REQ_BGL = False
    _FIELDS_STATIC = utils.FieldSet(*["name", "os", "pnode", "snodes",
-                                    "admin_state", "admin_ram",
+                                    "admin_state",
                                      "disk_template", "ip", "mac", "bridge",
                                      "disk_template", "ip", "mac", "bridge",
+                                    "nic_mode", "nic_link",
                                      "sda_size", "sdb_size", "vcpus", "tags",
                                      "network_port", "beparams",
                                      "sda_size", "sdb_size", "vcpus", "tags",
                                      "network_port", "beparams",
-                                    "(disk).(size)/([0-9]+)",
-                                    "(disk).(sizes)",
-                                    "(nic).(mac|ip|bridge)/([0-9]+)",
-                                    "(nic).(macs|ips|bridges)",
-                                    "(disk|nic).(count)",
-                                    "serial_no", "hypervisor", "hvparams",] +
+                                    r"(disk)\.(size)/([0-9]+)",
+                                    r"(disk)\.(sizes)", "disk_usage",
+                                    r"(nic)\.(mac|ip|mode|link)/([0-9]+)",
+                                    r"(nic)\.(bridge)/([0-9]+)",
+                                    r"(nic)\.(macs|ips|modes|links|bridges)",
+                                    r"(disk|nic)\.(count)",
+                                    "serial_no", "hypervisor", "hvparams",
+                                    "ctime", "mtime",
+                                    ] +
                                    ["hv/%s" % name
                                     for name in constants.HVS_PARAMETERS] +
                                    ["be/%s" % name
                                    ["hv/%s" % name
                                     for name in constants.HVS_PARAMETERS] +
                                    ["be/%s" % name
@@ -2732,7 +3949,8 @@ class LUQueryInstances(NoHooksLU):
      else:
        self.wanted = locking.ALL_SET
  
      else:
        self.wanted = locking.ALL_SET
  
-    self.do_locking = self._FIELDS_STATIC.NonMatching(self.op.output_fields)
+    self.do_node_query = self._FIELDS_STATIC.NonMatching(self.op.output_fields)
+    self.do_locking = self.do_node_query and self.op.use_locking
      if self.do_locking:
        self.needed_locks[locking.LEVEL_INSTANCE] = self.wanted
        self.needed_locks[locking.LEVEL_NODE] = []
      if self.do_locking:
        self.needed_locks[locking.LEVEL_INSTANCE] = self.wanted
        self.needed_locks[locking.LEVEL_NODE] = []
@@ -2753,19 +3971,25 @@ class LUQueryInstances(NoHooksLU):
  
      """
      all_info = self.cfg.GetAllInstancesInfo()
  
      """
      all_info = self.cfg.GetAllInstancesInfo()
-    if self.do_locking:
-      instance_names = self.acquired_locks[locking.LEVEL_INSTANCE]
-    elif self.wanted != locking.ALL_SET:
-      instance_names = self.wanted
-      missing = set(instance_names).difference(all_info.keys())
-      if missing:
-        raise errors.OpExecError(
-          "Some instances were removed before retrieving their data: %s"
-          % missing)
+    if self.wanted == locking.ALL_SET:
+      # caller didn't specify instance names, so ordering is not important
+      if self.do_locking:
+        instance_names = self.acquired_locks[locking.LEVEL_INSTANCE]
+      else:
+        instance_names = all_info.keys()
+      instance_names = utils.NiceSort(instance_names)
      else:
      else:
-      instance_names = all_info.keys()
+      # caller did specify names, so we must keep the ordering
+      if self.do_locking:
+        tgt_set = self.acquired_locks[locking.LEVEL_INSTANCE]
+      else:
+        tgt_set = all_info.keys()
+      missing = set(self.wanted).difference(tgt_set)
+      if missing:
+        raise errors.OpExecError("Some instances were removed before"
+                                 " retrieving their data: %s" % missing)
+      instance_names = self.wanted
  
  
-    instance_names = utils.NiceSort(instance_names)
      instance_list = [all_info[iname] for iname in instance_names]
  
      # begin data gathering
      instance_list = [all_info[iname] for iname in instance_names]
  
      # begin data gathering
@@ -2774,16 +3998,21 @@ class LUQueryInstances(NoHooksLU):
      hv_list = list(set([inst.hypervisor for inst in instance_list]))
  
      bad_nodes = []
      hv_list = list(set([inst.hypervisor for inst in instance_list]))
  
      bad_nodes = []
-    if self.do_locking:
+    off_nodes = []
+    if self.do_node_query:
        live_data = {}
        node_data = self.rpc.call_all_instances_info(nodes, hv_list)
        for name in nodes:
          result = node_data[name]
        live_data = {}
        node_data = self.rpc.call_all_instances_info(nodes, hv_list)
        for name in nodes:
          result = node_data[name]
-        if result:
-          live_data.update(result)
-        elif result == False:
+        if result.offline:
+          # offline nodes will be in both lists
+          off_nodes.append(name)
+        if result.failed or result.fail_msg:
            bad_nodes.append(name)
            bad_nodes.append(name)
-        # else no instance is alive
+        else:
+          if result.payload:
+            live_data.update(result.payload)
+          # else no instance is alive
      else:
        live_data = dict([(name, {}) for name in instance_names])
  
      else:
        live_data = dict([(name, {}) for name in instance_names])
  
@@ -2792,10 +4021,13 @@ class LUQueryInstances(NoHooksLU):
      HVPREFIX = "hv/"
      BEPREFIX = "be/"
      output = []
      HVPREFIX = "hv/"
      BEPREFIX = "be/"
      output = []
+    cluster = self.cfg.GetClusterInfo()
      for instance in instance_list:
        iout = []
      for instance in instance_list:
        iout = []
-      i_hv = self.cfg.GetClusterInfo().FillHV(instance)
-      i_be = self.cfg.GetClusterInfo().FillBE(instance)
+      i_hv = cluster.FillHV(instance)
+      i_be = cluster.FillBE(instance)
+      i_nicp = [objects.FillDict(cluster.nicparams[constants.PP_DEFAULT],
+                                 nic.nicparams) for nic in instance.nics]
        for field in self.op.output_fields:
          st_match = self._FIELDS_STATIC.Matches(field)
          if field == "name":
        for field in self.op.output_fields:
          st_match = self._FIELDS_STATIC.Matches(field)
          if field == "name":
@@ -2807,24 +4039,26 @@ class LUQueryInstances(NoHooksLU):
          elif field == "snodes":
            val = list(instance.secondary_nodes)
          elif field == "admin_state":
          elif field == "snodes":
            val = list(instance.secondary_nodes)
          elif field == "admin_state":
-          val = (instance.status != "down")
+          val = instance.admin_up
          elif field == "oper_state":
            if instance.primary_node in bad_nodes:
              val = None
            else:
              val = bool(live_data.get(instance.name))
          elif field == "status":
          elif field == "oper_state":
            if instance.primary_node in bad_nodes:
              val = None
            else:
              val = bool(live_data.get(instance.name))
          elif field == "status":
-          if instance.primary_node in bad_nodes:
+          if instance.primary_node in off_nodes:
+            val = "ERROR_nodeoffline"
+          elif instance.primary_node in bad_nodes:
              val = "ERROR_nodedown"
            else:
              running = bool(live_data.get(instance.name))
              if running:
              val = "ERROR_nodedown"
            else:
              running = bool(live_data.get(instance.name))
              if running:
-              if instance.status != "down":
+              if instance.admin_up:
                  val = "running"
                else:
                  val = "ERROR_up"
              else:
                  val = "running"
                else:
                  val = "ERROR_up"
              else:
-              if instance.status != "down":
+              if instance.admin_up:
                  val = "ERROR_down"
                else:
                  val = "ADMIN_down"
                  val = "ERROR_down"
                else:
                  val = "ADMIN_down"
@@ -2835,24 +4069,53 @@ class LUQueryInstances(NoHooksLU):
              val = live_data[instance.name].get("memory", "?")
            else:
              val = "-"
              val = live_data[instance.name].get("memory", "?")
            else:
              val = "-"
+        elif field == "vcpus":
+          val = i_be[constants.BE_VCPUS]
          elif field == "disk_template":
            val = instance.disk_template
          elif field == "ip":
          elif field == "disk_template":
            val = instance.disk_template
          elif field == "ip":
-          val = instance.nics[0].ip
+          if instance.nics:
+            val = instance.nics[0].ip
+          else:
+            val = None
+        elif field == "nic_mode":
+          if instance.nics:
+            val = i_nicp[0][constants.NIC_MODE]
+          else:
+            val = None
+        elif field == "nic_link":
+          if instance.nics:
+            val = i_nicp[0][constants.NIC_LINK]
+          else:
+            val = None
          elif field == "bridge":
          elif field == "bridge":
-          val = instance.nics[0].bridge
+          if (instance.nics and
+              i_nicp[0][constants.NIC_MODE] == constants.NIC_MODE_BRIDGED):
+            val = i_nicp[0][constants.NIC_LINK]
+          else:
+            val = None
          elif field == "mac":
          elif field == "mac":
-          val = instance.nics[0].mac
+          if instance.nics:
+            val = instance.nics[0].mac
+          else:
+            val = None
          elif field == "sda_size" or field == "sdb_size":
            idx = ord(field[2]) - ord('a')
            try:
              val = instance.FindDisk(idx).size
            except errors.OpPrereqError:
              val = None
          elif field == "sda_size" or field == "sdb_size":
            idx = ord(field[2]) - ord('a')
            try:
              val = instance.FindDisk(idx).size
            except errors.OpPrereqError:
              val = None
+        elif field == "disk_usage": # total disk usage per node
+          disk_sizes = [{'size': disk.size} for disk in instance.disks]
+          val = _ComputeDiskSize(instance.disk_template, disk_sizes)
          elif field == "tags":
            val = list(instance.GetTags())
          elif field == "serial_no":
            val = instance.serial_no
          elif field == "tags":
            val = list(instance.GetTags())
          elif field == "serial_no":
            val = instance.serial_no
+        elif field == "ctime":
+          val = instance.ctime
+        elif field == "mtime":
+          val = instance.mtime
          elif field == "network_port":
            val = instance.network_port
          elif field == "hypervisor":
          elif field == "network_port":
            val = instance.network_port
          elif field == "hypervisor":
@@ -2889,8 +4152,17 @@ class LUQueryInstances(NoHooksLU):
                val = [nic.mac for nic in instance.nics]
              elif st_groups[1] == "ips":
                val = [nic.ip for nic in instance.nics]
                val = [nic.mac for nic in instance.nics]
              elif st_groups[1] == "ips":
                val = [nic.ip for nic in instance.nics]
+            elif st_groups[1] == "modes":
+              val = [nicp[constants.NIC_MODE] for nicp in i_nicp]
+            elif st_groups[1] == "links":
+              val = [nicp[constants.NIC_LINK] for nicp in i_nicp]
              elif st_groups[1] == "bridges":
              elif st_groups[1] == "bridges":
-              val = [nic.bridge for nic in instance.nics]
+              val = []
+              for nicp in i_nicp:
+                if nicp[constants.NIC_MODE] == constants.NIC_MODE_BRIDGED:
+                  val.append(nicp[constants.NIC_LINK])
+                else:
+                  val.append(None)
              else:
                # index-based item
                nic_idx = int(st_groups[2])
              else:
                # index-based item
                nic_idx = int(st_groups[2])
@@ -2901,14 +4173,23 @@ class LUQueryInstances(NoHooksLU):
                    val = instance.nics[nic_idx].mac
                  elif st_groups[1] == "ip":
                    val = instance.nics[nic_idx].ip
                    val = instance.nics[nic_idx].mac
                  elif st_groups[1] == "ip":
                    val = instance.nics[nic_idx].ip
+                elif st_groups[1] == "mode":
+                  val = i_nicp[nic_idx][constants.NIC_MODE]
+                elif st_groups[1] == "link":
+                  val = i_nicp[nic_idx][constants.NIC_LINK]
                  elif st_groups[1] == "bridge":
                  elif st_groups[1] == "bridge":
-                  val = instance.nics[nic_idx].bridge
+                  nic_mode = i_nicp[nic_idx][constants.NIC_MODE]
+                  if nic_mode == constants.NIC_MODE_BRIDGED:
+                    val = i_nicp[nic_idx][constants.NIC_LINK]
+                  else:
+                    val = None
                  else:
                    assert False, "Unhandled NIC parameter"
            else:
                  else:
                    assert False, "Unhandled NIC parameter"
            else:
-            assert False, "Unhandled variable parameter"
+            assert False, ("Declared but unhandled variable parameter '%s'" %
+                           field)
          else:
          else:
-          raise errors.ParameterError(field)
+          assert False, "Declared but unhandled parameter '%s'" % field
          iout.append(val)
        output.append(iout)
  
          iout.append(val)
        output.append(iout)
  
@@ -2967,17 +4248,19 @@ class LUFailoverInstance(LogicalUnit):
                                     "a mirrored disk template")
  
      target_node = secondary_nodes[0]
                                     "a mirrored disk template")
  
      target_node = secondary_nodes[0]
-    # check memory requirements on the secondary node
-    _CheckNodeFreeMemory(self, target_node, "failing over instance %s" %
-                         instance.name, bep[constants.BE_MEMORY],
-                         instance.hypervisor)
+    _CheckNodeOnline(self, target_node)
+    _CheckNodeNotDrained(self, target_node)
+    if instance.admin_up:
+      # check memory requirements on the secondary node
+      _CheckNodeFreeMemory(self, target_node, "failing over instance %s" %
+                           instance.name, bep[constants.BE_MEMORY],
+                           instance.hypervisor)
+    else:
+      self.LogInfo("Not checking memory on the secondary node as"
+                   " instance will not be started")
  
      # check bridge existance
  
      # check bridge existance
-    brlist = [nic.bridge for nic in instance.nics]
-    if not self.rpc.call_bridges_exist(target_node, brlist):
-      raise errors.OpPrereqError("One or more target bridges %s does not"
-                                 " exist on destination node '%s'" %
-                                 (brlist, target_node))
+    _CheckInstanceBridgesExist(self, instance, node=target_node)
  
    def Exec(self, feedback_fn):
      """Failover an instance.
  
    def Exec(self, feedback_fn):
      """Failover an instance.
@@ -2995,7 +4278,7 @@ class LUFailoverInstance(LogicalUnit):
      for dev in instance.disks:
        # for drbd, these are drbd over lvm
        if not _CheckDiskConsistency(self, dev, target_node, False):
      for dev in instance.disks:
        # for drbd, these are drbd over lvm
        if not _CheckDiskConsistency(self, dev, target_node, False):
-        if instance.status == "up" and not self.op.ignore_consistency:
+        if instance.admin_up and not self.op.ignore_consistency:
            raise errors.OpExecError("Disk %s is degraded on target node,"
                                     " aborting failover." % dev.iv_name)
  
            raise errors.OpExecError("Disk %s is degraded on target node,"
                                     " aborting failover." % dev.iv_name)
  
@@ -3003,15 +4286,18 @@ class LUFailoverInstance(LogicalUnit):
      logging.info("Shutting down instance %s on node %s",
                   instance.name, source_node)
  
      logging.info("Shutting down instance %s on node %s",
                   instance.name, source_node)
  
-    if not self.rpc.call_instance_shutdown(source_node, instance):
+    result = self.rpc.call_instance_shutdown(source_node, instance)
+    msg = result.fail_msg
+    if msg:
        if self.op.ignore_consistency:
          self.proc.LogWarning("Could not shutdown instance %s on node %s."
        if self.op.ignore_consistency:
          self.proc.LogWarning("Could not shutdown instance %s on node %s."
-                             " Proceeding"
-                             " anyway. Please make sure node %s is down",
-                             instance.name, source_node, source_node)
+                             " Proceeding anyway. Please make sure node"
+                             " %s is down. Error details: %s",
+                             instance.name, source_node, source_node, msg)
        else:
        else:
-        raise errors.OpExecError("Could not shutdown instance %s on node %s" %
-                                 (instance.name, source_node))
+        raise errors.OpExecError("Could not shutdown instance %s on"
+                                 " node %s: %s" %
+                                 (instance.name, source_node, msg))
  
      feedback_fn("* deactivating the instance's disks on source node")
      if not _ShutdownInstanceDisks(self, instance, ignore_primary=True):
  
      feedback_fn("* deactivating the instance's disks on source node")
      if not _ShutdownInstanceDisks(self, instance, ignore_primary=True):
@@ -3022,72 +4308,521 @@ class LUFailoverInstance(LogicalUnit):
      self.cfg.Update(instance)
  
      # Only start the instance if it's marked as up
      self.cfg.Update(instance)
  
      # Only start the instance if it's marked as up
-    if instance.status == "up":
+    if instance.admin_up:
        feedback_fn("* activating the instance's disks on target node")
        logging.info("Starting instance %s on node %s",
                     instance.name, target_node)
  
        feedback_fn("* activating the instance's disks on target node")
        logging.info("Starting instance %s on node %s",
                     instance.name, target_node)
  
-      disks_ok, dummy = _AssembleInstanceDisks(self, instance,
+      disks_ok, _ = _AssembleInstanceDisks(self, instance,
                                                 ignore_secondaries=True)
        if not disks_ok:
          _ShutdownInstanceDisks(self, instance)
          raise errors.OpExecError("Can't activate the instance's disks")
  
        feedback_fn("* starting the instance on the target node")
                                                 ignore_secondaries=True)
        if not disks_ok:
          _ShutdownInstanceDisks(self, instance)
          raise errors.OpExecError("Can't activate the instance's disks")
  
        feedback_fn("* starting the instance on the target node")
-      if not self.rpc.call_instance_start(target_node, instance, None):
+      result = self.rpc.call_instance_start(target_node, instance, None, None)
+      msg = result.fail_msg
+      if msg:
          _ShutdownInstanceDisks(self, instance)
          _ShutdownInstanceDisks(self, instance)
-        raise errors.OpExecError("Could not start instance %s on node %s." %
-                                 (instance.name, target_node))
+        raise errors.OpExecError("Could not start instance %s on node %s: %s" %
+                                 (instance.name, target_node, msg))
  
  
  
  
-def _CreateBlockDevOnPrimary(lu, node, instance, device, info):
-  """Create a tree of block devices on the primary node.
+class LUMigrateInstance(LogicalUnit):
+  """Migrate an instance.
  
  
-  This always creates all devices.
+  This is migration without shutting down, compared to the failover,
+  which is done with shutdown.
  
    """
  
    """
-  if device.children:
-    for child in device.children:
-      if not _CreateBlockDevOnPrimary(lu, node, instance, child, info):
-        return False
+  HPATH = "instance-migrate"
+  HTYPE = constants.HTYPE_INSTANCE
+  _OP_REQP = ["instance_name", "live", "cleanup"]
  
  
-  lu.cfg.SetDiskID(device, node)
-  new_id = lu.rpc.call_blockdev_create(node, device, device.size,
-                                       instance.name, True, info)
-  if not new_id:
-    return False
-  if device.physical_id is None:
-    device.physical_id = new_id
-  return True
+  REQ_BGL = False
+
+  def ExpandNames(self):
+    self._ExpandAndLockInstance()
+
+    self.needed_locks[locking.LEVEL_NODE] = []
+    self.recalculate_locks[locking.LEVEL_NODE] = constants.LOCKS_REPLACE
+
+    self._migrater = TLMigrateInstance(self, self.op.instance_name,
+                                       self.op.live, self.op.cleanup)
+    self.tasklets = [self._migrater]
+
+  def DeclareLocks(self, level):
+    if level == locking.LEVEL_NODE:
+      self._LockInstancesNodes()
+
+  def BuildHooksEnv(self):
+    """Build hooks env.
+
+    This runs on master, primary and secondary nodes of the instance.
+
+    """
+    instance = self._migrater.instance
+    env = _BuildInstanceHookEnvByObject(self, instance)
+    env["MIGRATE_LIVE"] = self.op.live
+    env["MIGRATE_CLEANUP"] = self.op.cleanup
+    nl = [self.cfg.GetMasterNode()] + list(instance.secondary_nodes)
+    return env, nl, nl
+
+
+class LUMigrateNode(LogicalUnit):
+  """Migrate all instances from a node.
+
+  """
+  HPATH = "node-migrate"
+  HTYPE = constants.HTYPE_NODE
+  _OP_REQP = ["node_name", "live"]
+  REQ_BGL = False
+
+  def ExpandNames(self):
+    self.op.node_name = self.cfg.ExpandNodeName(self.op.node_name)
+    if self.op.node_name is None:
+      raise errors.OpPrereqError("Node '%s' not known" % self.op.node_name)
+
+    self.needed_locks = {
+      locking.LEVEL_NODE: [self.op.node_name],
+      }
+
+    self.recalculate_locks[locking.LEVEL_NODE] = constants.LOCKS_APPEND
+
+    # Create tasklets for migrating instances for all instances on this node
+    names = []
+    tasklets = []
+
+    for inst in _GetNodePrimaryInstances(self.cfg, self.op.node_name):
+      logging.debug("Migrating instance %s", inst.name)
+      names.append(inst.name)
+
+      tasklets.append(TLMigrateInstance(self, inst.name, self.op.live, False))
+
+    self.tasklets = tasklets
+
+    # Declare instance locks
+    self.needed_locks[locking.LEVEL_INSTANCE] = names
+
+  def DeclareLocks(self, level):
+    if level == locking.LEVEL_NODE:
+      self._LockInstancesNodes()
+
+  def BuildHooksEnv(self):
+    """Build hooks env.
+
+    This runs on the master, the primary and all the secondaries.
+
+    """
+    env = {
+      "NODE_NAME": self.op.node_name,
+      }
+
+    nl = [self.cfg.GetMasterNode()]
+
+    return (env, nl, nl)
+
+
+class TLMigrateInstance(Tasklet):
+  def __init__(self, lu, instance_name, live, cleanup):
+    """Initializes this class.
+
+    """
+    Tasklet.__init__(self, lu)
+
+    # Parameters
+    self.instance_name = instance_name
+    self.live = live
+    self.cleanup = cleanup
+
+  def CheckPrereq(self):
+    """Check prerequisites.
+
+    This checks that the instance is in the cluster.
+
+    """
+    instance = self.cfg.GetInstanceInfo(
+      self.cfg.ExpandInstanceName(self.instance_name))
+    if instance is None:
+      raise errors.OpPrereqError("Instance '%s' not known" %
+                                 self.instance_name)
+
+    if instance.disk_template != constants.DT_DRBD8:
+      raise errors.OpPrereqError("Instance's disk layout is not"
+                                 " drbd8, cannot migrate.")
+
+    secondary_nodes = instance.secondary_nodes
+    if not secondary_nodes:
+      raise errors.ConfigurationError("No secondary node but using"
+                                      " drbd8 disk template")
+
+    i_be = self.cfg.GetClusterInfo().FillBE(instance)
+
+    target_node = secondary_nodes[0]
+    # check memory requirements on the secondary node
+    _CheckNodeFreeMemory(self, target_node, "migrating instance %s" %
+                         instance.name, i_be[constants.BE_MEMORY],
+                         instance.hypervisor)
+
+    # check bridge existance
+    _CheckInstanceBridgesExist(self, instance, node=target_node)
+
+    if not self.cleanup:
+      _CheckNodeNotDrained(self, target_node)
+      result = self.rpc.call_instance_migratable(instance.primary_node,
+                                                 instance)
+      result.Raise("Can't migrate, please use failover", prereq=True)
+
+    self.instance = instance
+
+  def _WaitUntilSync(self):
+    """Poll with custom rpc for disk sync.
+
+    This uses our own step-based rpc call.
+
+    """
+    self.feedback_fn("* wait until resync is done")
+    all_done = False
+    while not all_done:
+      all_done = True
+      result = self.rpc.call_drbd_wait_sync(self.all_nodes,
+                                            self.nodes_ip,
+                                            self.instance.disks)
+      min_percent = 100
+      for node, nres in result.items():
+        nres.Raise("Cannot resync disks on node %s" % node)
+        node_done, node_percent = nres.payload
+        all_done = all_done and node_done
+        if node_percent is not None:
+          min_percent = min(min_percent, node_percent)
+      if not all_done:
+        if min_percent < 100:
+          self.feedback_fn("   - progress: %.1f%%" % min_percent)
+        time.sleep(2)
+
+  def _EnsureSecondary(self, node):
+    """Demote a node to secondary.
+
+    """
+    self.feedback_fn("* switching node %s to secondary mode" % node)
+
+    for dev in self.instance.disks:
+      self.cfg.SetDiskID(dev, node)
+
+    result = self.rpc.call_blockdev_close(node, self.instance.name,
+                                          self.instance.disks)
+    result.Raise("Cannot change disk to secondary on node %s" % node)
+
+  def _GoStandalone(self):
+    """Disconnect from the network.
+
+    """
+    self.feedback_fn("* changing into standalone mode")
+    result = self.rpc.call_drbd_disconnect_net(self.all_nodes, self.nodes_ip,
+                                               self.instance.disks)
+    for node, nres in result.items():
+      nres.Raise("Cannot disconnect disks node %s" % node)
+
+  def _GoReconnect(self, multimaster):
+    """Reconnect to the network.
+
+    """
+    if multimaster:
+      msg = "dual-master"
+    else:
+      msg = "single-master"
+    self.feedback_fn("* changing disks into %s mode" % msg)
+    result = self.rpc.call_drbd_attach_net(self.all_nodes, self.nodes_ip,
+                                           self.instance.disks,
+                                           self.instance.name, multimaster)
+    for node, nres in result.items():
+      nres.Raise("Cannot change disks config on node %s" % node)
+
+  def _ExecCleanup(self):
+    """Try to cleanup after a failed migration.
+
+    The cleanup is done by:
+      - check that the instance is running only on one node
+        (and update the config if needed)
+      - change disks on its secondary node to secondary
+      - wait until disks are fully synchronized
+      - disconnect from the network
+      - change disks into single-master mode
+      - wait again until disks are fully synchronized
+
+    """
+    instance = self.instance
+    target_node = self.target_node
+    source_node = self.source_node
+
+    # check running on only one node
+    self.feedback_fn("* checking where the instance actually runs"
+                     " (if this hangs, the hypervisor might be in"
+                     " a bad state)")
+    ins_l = self.rpc.call_instance_list(self.all_nodes, [instance.hypervisor])
+    for node, result in ins_l.items():
+      result.Raise("Can't contact node %s" % node)
+
+    runningon_source = instance.name in ins_l[source_node].payload
+    runningon_target = instance.name in ins_l[target_node].payload
+
+    if runningon_source and runningon_target:
+      raise errors.OpExecError("Instance seems to be running on two nodes,"
+                               " or the hypervisor is confused. You will have"
+                               " to ensure manually that it runs only on one"
+                               " and restart this operation.")
+
+    if not (runningon_source or runningon_target):
+      raise errors.OpExecError("Instance does not seem to be running at all."
+                               " In this case, it's safer to repair by"
+                               " running 'gnt-instance stop' to ensure disk"
+                               " shutdown, and then restarting it.")
+
+    if runningon_target:
+      # the migration has actually succeeded, we need to update the config
+      self.feedback_fn("* instance running on secondary node (%s),"
+                       " updating config" % target_node)
+      instance.primary_node = target_node
+      self.cfg.Update(instance)
+      demoted_node = source_node
+    else:
+      self.feedback_fn("* instance confirmed to be running on its"
+                       " primary node (%s)" % source_node)
+      demoted_node = target_node
+
+    self._EnsureSecondary(demoted_node)
+    try:
+      self._WaitUntilSync()
+    except errors.OpExecError:
+      # we ignore here errors, since if the device is standalone, it
+      # won't be able to sync
+      pass
+    self._GoStandalone()
+    self._GoReconnect(False)
+    self._WaitUntilSync()
+
+    self.feedback_fn("* done")
+
+  def _RevertDiskStatus(self):
+    """Try to revert the disk status after a failed migration.
+
+    """
+    target_node = self.target_node
+    try:
+      self._EnsureSecondary(target_node)
+      self._GoStandalone()
+      self._GoReconnect(False)
+      self._WaitUntilSync()
+    except errors.OpExecError, err:
+      self.lu.LogWarning("Migration failed and I can't reconnect the"
+                         " drives: error '%s'\n"
+                         "Please look and recover the instance status" %
+                         str(err))
+
+  def _AbortMigration(self):
+    """Call the hypervisor code to abort a started migration.
+
+    """
+    instance = self.instance
+    target_node = self.target_node
+    migration_info = self.migration_info
+
+    abort_result = self.rpc.call_finalize_migration(target_node,
+                                                    instance,
+                                                    migration_info,
+                                                    False)
+    abort_msg = abort_result.fail_msg
+    if abort_msg:
+      logging.error("Aborting migration failed on target node %s: %s" %
+                    (target_node, abort_msg))
+      # Don't raise an exception here, as we stil have to try to revert the
+      # disk status, even if this step failed.
+
+  def _ExecMigration(self):
+    """Migrate an instance.
+
+    The migrate is done by:
+      - change the disks into dual-master mode
+      - wait until disks are fully synchronized again
+      - migrate the instance
+      - change disks on the new secondary node (the old primary) to secondary
+      - wait until disks are fully synchronized
+      - change disks into single-master mode
+
+    """
+    instance = self.instance
+    target_node = self.target_node
+    source_node = self.source_node
+
+    self.feedback_fn("* checking disk consistency between source and target")
+    for dev in instance.disks:
+      if not _CheckDiskConsistency(self, dev, target_node, False):
+        raise errors.OpExecError("Disk %s is degraded or not fully"
+                                 " synchronized on target node,"
+                                 " aborting migrate." % dev.iv_name)
+
+    # First get the migration information from the remote node
+    result = self.rpc.call_migration_info(source_node, instance)
+    msg = result.fail_msg
+    if msg:
+      log_err = ("Failed fetching source migration information from %s: %s" %
+                 (source_node, msg))
+      logging.error(log_err)
+      raise errors.OpExecError(log_err)
+
+    self.migration_info = migration_info = result.payload
+
+    # Then switch the disks to master/master mode
+    self._EnsureSecondary(target_node)
+    self._GoStandalone()
+    self._GoReconnect(True)
+    self._WaitUntilSync()
+
+    self.feedback_fn("* preparing %s to accept the instance" % target_node)
+    result = self.rpc.call_accept_instance(target_node,
+                                           instance,
+                                           migration_info,
+                                           self.nodes_ip[target_node])
+
+    msg = result.fail_msg
+    if msg:
+      logging.error("Instance pre-migration failed, trying to revert"
+                    " disk status: %s", msg)
+      self._AbortMigration()
+      self._RevertDiskStatus()
+      raise errors.OpExecError("Could not pre-migrate instance %s: %s" %
+                               (instance.name, msg))
+
+    self.feedback_fn("* migrating instance to %s" % target_node)
+    time.sleep(10)
+    result = self.rpc.call_instance_migrate(source_node, instance,
+                                            self.nodes_ip[target_node],
+                                            self.live)
+    msg = result.fail_msg
+    if msg:
+      logging.error("Instance migration failed, trying to revert"
+                    " disk status: %s", msg)
+      self._AbortMigration()
+      self._RevertDiskStatus()
+      raise errors.OpExecError("Could not migrate instance %s: %s" %
+                               (instance.name, msg))
+    time.sleep(10)
+
+    instance.primary_node = target_node
+    # distribute new instance config to the other nodes
+    self.cfg.Update(instance)
+
+    result = self.rpc.call_finalize_migration(target_node,
+                                              instance,
+                                              migration_info,
+                                              True)
+    msg = result.fail_msg
+    if msg:
+      logging.error("Instance migration succeeded, but finalization failed:"
+                    " %s" % msg)
+      raise errors.OpExecError("Could not finalize instance migration: %s" %
+                               msg)
+
+    self._EnsureSecondary(source_node)
+    self._WaitUntilSync()
+    self._GoStandalone()
+    self._GoReconnect(False)
+    self._WaitUntilSync()
+
+    self.feedback_fn("* done")
  
  
+  def Exec(self, feedback_fn):
+    """Perform the migration.
+
+    """
+    feedback_fn("Migrating instance %s" % self.instance.name)
+
+    self.feedback_fn = feedback_fn
  
  
-def _CreateBlockDevOnSecondary(lu, node, instance, device, force, info):
-  """Create a tree of block devices on a secondary node.
+    self.source_node = self.instance.primary_node
+    self.target_node = self.instance.secondary_nodes[0]
+    self.all_nodes = [self.source_node, self.target_node]
+    self.nodes_ip = {
+      self.source_node: self.cfg.GetNodeInfo(self.source_node).secondary_ip,
+      self.target_node: self.cfg.GetNodeInfo(self.target_node).secondary_ip,
+      }
+
+    if self.cleanup:
+      return self._ExecCleanup()
+    else:
+      return self._ExecMigration()
+
+
+def _CreateBlockDev(lu, node, instance, device, force_create,
+                    info, force_open):
+  """Create a tree of block devices on a given node.
  
    If this device type has to be created on secondaries, create it and
    all its children.
  
    If not, just recurse to children keeping the same 'force' value.
  
  
    If this device type has to be created on secondaries, create it and
    all its children.
  
    If not, just recurse to children keeping the same 'force' value.
  
+  @param lu: the lu on whose behalf we execute
+  @param node: the node on which to create the device
+  @type instance: L{objects.Instance}
+  @param instance: the instance which owns the device
+  @type device: L{objects.Disk}
+  @param device: the device to create
+  @type force_create: boolean
+  @param force_create: whether to force creation of this device; this
+      will be change to True whenever we find a device which has
+      CreateOnSecondary() attribute
+  @param info: the extra 'metadata' we should attach to the device
+      (this will be represented as a LVM tag)
+  @type force_open: boolean
+  @param force_open: this parameter will be passes to the
+      L{backend.BlockdevCreate} function where it specifies
+      whether we run on primary or not, and it affects both
+      the child assembly and the device own Open() execution
+
    """
    if device.CreateOnSecondary():
    """
    if device.CreateOnSecondary():
-    force = True
+    force_create = True
+
    if device.children:
      for child in device.children:
    if device.children:
      for child in device.children:
-      if not _CreateBlockDevOnSecondary(lu, node, instance,
-                                        child, force, info):
-        return False
+      _CreateBlockDev(lu, node, instance, child, force_create,
+                      info, force_open)
  
  
-  if not force:
-    return True
+  if not force_create:
+    return
+
+  _CreateSingleBlockDev(lu, node, instance, device, info, force_open)
+
+
+def _CreateSingleBlockDev(lu, node, instance, device, info, force_open):
+  """Create a single block device on a given node.
+
+  This will not recurse over children of the device, so they must be
+  created in advance.
+
+  @param lu: the lu on whose behalf we execute
+  @param node: the node on which to create the device
+  @type instance: L{objects.Instance}
+  @param instance: the instance which owns the device
+  @type device: L{objects.Disk}
+  @param device: the device to create
+  @param info: the extra 'metadata' we should attach to the device
+      (this will be represented as a LVM tag)
+  @type force_open: boolean
+  @param force_open: this parameter will be passes to the
+      L{backend.BlockdevCreate} function where it specifies
+      whether we run on primary or not, and it affects both
+      the child assembly and the device own Open() execution
+
+  """
    lu.cfg.SetDiskID(device, node)
    lu.cfg.SetDiskID(device, node)
-  new_id = lu.rpc.call_blockdev_create(node, device, device.size,
-                                       instance.name, False, info)
-  if not new_id:
-    return False
+  result = lu.rpc.call_blockdev_create(node, device, device.size,
+                                       instance.name, force_open, info)
+  result.Raise("Can't create block device %s on"
+               " node %s for instance %s" % (device, node, instance.name))
    if device.physical_id is None:
    if device.physical_id is None:
-    device.physical_id = new_id
-  return True
+    device.physical_id = result.payload
  
  
  def _GenerateUniqueNames(lu, exts):
  
  
  def _GenerateUniqueNames(lu, exts):
@@ -3127,7 +4862,8 @@ def _GenerateDRBD8Branch(lu, primary, secondary, size, names, iv_name,
  def _GenerateDiskTemplate(lu, template_name,
                            instance_name, primary_node,
                            secondary_nodes, disk_info,
  def _GenerateDiskTemplate(lu, template_name,
                            instance_name, primary_node,
                            secondary_nodes, disk_info,
-                          file_storage_dir, file_driver):
+                          file_storage_dir, file_driver,
+                          base_index):
    """Generate the entire disk layout for a given template type.
  
    """
    """Generate the entire disk layout for a given template type.
  
    """
@@ -3142,12 +4878,14 @@ def _GenerateDiskTemplate(lu, template_name,
      if len(secondary_nodes) != 0:
        raise errors.ProgrammerError("Wrong template configuration")
  
      if len(secondary_nodes) != 0:
        raise errors.ProgrammerError("Wrong template configuration")
  
-    names = _GenerateUniqueNames(lu, [".disk%d" % i
+    names = _GenerateUniqueNames(lu, [".disk%d" % (base_index + i)
                                        for i in range(disk_count)])
      for idx, disk in enumerate(disk_info):
                                        for i in range(disk_count)])
      for idx, disk in enumerate(disk_info):
+      disk_index = idx + base_index
        disk_dev = objects.Disk(dev_type=constants.LD_LV, size=disk["size"],
                                logical_id=(vgname, names[idx]),
        disk_dev = objects.Disk(dev_type=constants.LD_LV, size=disk["size"],
                                logical_id=(vgname, names[idx]),
-                              iv_name = "disk/%d" % idx)
+                              iv_name="disk/%d" % disk_index,
+                              mode=disk["mode"])
        disks.append(disk_dev)
    elif template_name == constants.DT_DRBD8:
      if len(secondary_nodes) != 1:
        disks.append(disk_dev)
    elif template_name == constants.DT_DRBD8:
      if len(secondary_nodes) != 1:
@@ -3156,28 +4894,31 @@ def _GenerateDiskTemplate(lu, template_name,
      minors = lu.cfg.AllocateDRBDMinor(
        [primary_node, remote_node] * len(disk_info), instance_name)
  
      minors = lu.cfg.AllocateDRBDMinor(
        [primary_node, remote_node] * len(disk_info), instance_name)
  
-    names = _GenerateUniqueNames(lu,
-                                 [".disk%d_%s" % (i, s)
-                                  for i in range(disk_count)
-                                  for s in ("data", "meta")
-                                  ])
+    names = []
+    for lv_prefix in _GenerateUniqueNames(lu, [".disk%d" % (base_index + i)
+                                               for i in range(disk_count)]):
+      names.append(lv_prefix + "_data")
+      names.append(lv_prefix + "_meta")
      for idx, disk in enumerate(disk_info):
      for idx, disk in enumerate(disk_info):
+      disk_index = idx + base_index
        disk_dev = _GenerateDRBD8Branch(lu, primary_node, remote_node,
                                        disk["size"], names[idx*2:idx*2+2],
        disk_dev = _GenerateDRBD8Branch(lu, primary_node, remote_node,
                                        disk["size"], names[idx*2:idx*2+2],
-                                      "disk/%d" % idx,
+                                      "disk/%d" % disk_index,
                                        minors[idx*2], minors[idx*2+1])
                                        minors[idx*2], minors[idx*2+1])
+      disk_dev.mode = disk["mode"]
        disks.append(disk_dev)
    elif template_name == constants.DT_FILE:
      if len(secondary_nodes) != 0:
        raise errors.ProgrammerError("Wrong template configuration")
  
      for idx, disk in enumerate(disk_info):
        disks.append(disk_dev)
    elif template_name == constants.DT_FILE:
      if len(secondary_nodes) != 0:
        raise errors.ProgrammerError("Wrong template configuration")
  
      for idx, disk in enumerate(disk_info):
-
+      disk_index = idx + base_index
        disk_dev = objects.Disk(dev_type=constants.LD_FILE, size=disk["size"],
        disk_dev = objects.Disk(dev_type=constants.LD_FILE, size=disk["size"],
-                              iv_name="disk/%d" % idx,
+                              iv_name="disk/%d" % disk_index,
                                logical_id=(file_driver,
                                            "%s/disk%d" % (file_storage_dir,
                                logical_id=(file_driver,
                                            "%s/disk%d" % (file_storage_dir,
-                                                         idx)))
+                                                         disk_index)),
+                              mode=disk["mode"])
        disks.append(disk_dev)
    else:
      raise errors.ProgrammerError("Invalid disk template '%s'" % template_name)
        disks.append(disk_dev)
    else:
      raise errors.ProgrammerError("Invalid disk template '%s'" % template_name)
@@ -3191,7 +4932,7 @@ def _GetInstanceInfoText(instance):
    return "originstname+%s" % instance.name
  
  
    return "originstname+%s" % instance.name
  
  
-def _CreateDisks(lu, instance):
+def _CreateDisks(lu, instance, to_skip=None):
    """Create all disks for an instance.
  
    This abstracts away some work from AddInstance.
    """Create all disks for an instance.
  
    This abstracts away some work from AddInstance.
@@ -3200,42 +4941,33 @@ def _CreateDisks(lu, instance):
    @param lu: the logical unit on whose behalf we execute
    @type instance: L{objects.Instance}
    @param instance: the instance whose disks we should create
    @param lu: the logical unit on whose behalf we execute
    @type instance: L{objects.Instance}
    @param instance: the instance whose disks we should create
+  @type to_skip: list
+  @param to_skip: list of indices to skip
    @rtype: boolean
    @return: the success of the creation
  
    """
    info = _GetInstanceInfoText(instance)
    @rtype: boolean
    @return: the success of the creation
  
    """
    info = _GetInstanceInfoText(instance)
+  pnode = instance.primary_node
  
    if instance.disk_template == constants.DT_FILE:
      file_storage_dir = os.path.dirname(instance.disks[0].logical_id[1])
  
    if instance.disk_template == constants.DT_FILE:
      file_storage_dir = os.path.dirname(instance.disks[0].logical_id[1])
-    result = lu.rpc.call_file_storage_dir_create(instance.primary_node,
-                                                 file_storage_dir)
-
-    if not result:
-      logging.error("Could not connect to node '%s'", instance.primary_node)
-      return False
+    result = lu.rpc.call_file_storage_dir_create(pnode, file_storage_dir)
  
  
-    if not result[0]:
-      logging.error("Failed to create directory '%s'", file_storage_dir)
-      return False
+    result.Raise("Failed to create directory '%s' on"
+                 " node %s: %s" % (file_storage_dir, pnode))
  
  
-  for device in instance.disks:
+  # Note: this needs to be kept in sync with adding of disks in
+  # LUSetInstanceParams
+  for idx, device in enumerate(instance.disks):
+    if to_skip and idx in to_skip:
+      continue
      logging.info("Creating volume %s for instance %s",
                   device.iv_name, instance.name)
      #HARDCODE
      logging.info("Creating volume %s for instance %s",
                   device.iv_name, instance.name)
      #HARDCODE
-    for secondary_node in instance.secondary_nodes:
-      if not _CreateBlockDevOnSecondary(lu, secondary_node, instance,
-                                        device, False, info):
-        logging.error("Failed to create volume %s (%s) on secondary node %s!",
-                      device.iv_name, device, secondary_node)
-        return False
-    #HARDCODE
-    if not _CreateBlockDevOnPrimary(lu, instance.primary_node,
-                                    instance, device, info):
-      logging.error("Failed to create volume %s on primary!", device.iv_name)
-      return False
-
-  return True
+    for node in instance.all_nodes:
+      f_create = node == pnode
+      _CreateBlockDev(lu, node, instance, device, f_create, info, f_create)
  
  
  def _RemoveDisks(lu, instance):
  
  
  def _RemoveDisks(lu, instance):
@@ -3256,23 +4988,27 @@ def _RemoveDisks(lu, instance):
    """
    logging.info("Removing block devices for instance %s", instance.name)
  
    """
    logging.info("Removing block devices for instance %s", instance.name)
  
-  result = True
+  all_result = True
    for device in instance.disks:
      for node, disk in device.ComputeNodeTree(instance.primary_node):
        lu.cfg.SetDiskID(disk, node)
    for device in instance.disks:
      for node, disk in device.ComputeNodeTree(instance.primary_node):
        lu.cfg.SetDiskID(disk, node)
-      if not lu.rpc.call_blockdev_remove(node, disk):
-        lu.proc.LogWarning("Could not remove block device %s on node %s,"
-                           " continuing anyway", device.iv_name, node)
-        result = False
+      msg = lu.rpc.call_blockdev_remove(node, disk).fail_msg
+      if msg:
+        lu.LogWarning("Could not remove block device %s on node %s,"
+                      " continuing anyway: %s", device.iv_name, node, msg)
+        all_result = False
  
    if instance.disk_template == constants.DT_FILE:
      file_storage_dir = os.path.dirname(instance.disks[0].logical_id[1])
  
    if instance.disk_template == constants.DT_FILE:
      file_storage_dir = os.path.dirname(instance.disks[0].logical_id[1])
-    if not lu.rpc.call_file_storage_dir_remove(instance.primary_node,
-                                               file_storage_dir):
-      logging.error("Could not remove directory '%s'", file_storage_dir)
-      result = False
+    result = lu.rpc.call_file_storage_dir_remove(instance.primary_node,
+                                                 file_storage_dir)
+    msg = result.fail_msg
+    if msg:
+      lu.LogWarning("Could not remove directory '%s' on node %s: %s",
+                    file_storage_dir, instance.primary_node, msg)
+      all_result = False
  
  
-  return result
+  return all_result
  
  
  def _ComputeDiskSize(disk_template, disks):
  
  
  def _ComputeDiskSize(disk_template, disks):
@@ -3316,13 +5052,10 @@ def _CheckHVParams(lu, nodenames, hvname, hvparams):
                                                    hvname,
                                                    hvparams)
    for node in nodenames:
                                                    hvname,
                                                    hvparams)
    for node in nodenames:
-    info = hvinfo.get(node, None)
-    if not info or not isinstance(info, (tuple, list)):
-      raise errors.OpPrereqError("Cannot get current information"
-                                 " from node '%s' (%s)" % (node, info))
-    if not info[0]:
-      raise errors.OpPrereqError("Hypervisor parameter validation failed:"
-                                 " %s" % info[1])
+    info = hvinfo[node]
+    if info.offline:
+      continue
+    info.Raise("Hypervisor parameter validation failed on node %s" % node)
  
  
  class LUCreateInstance(LogicalUnit):
  
  
  class LUCreateInstance(LogicalUnit):
@@ -3382,14 +5115,16 @@ class LUCreateInstance(LogicalUnit):
                                    ",".join(enabled_hvs)))
  
      # check hypervisor parameter syntax (locally)
                                    ",".join(enabled_hvs)))
  
      # check hypervisor parameter syntax (locally)
-
-    filled_hvp = cluster.FillDict(cluster.hvparams[self.op.hypervisor],
+    utils.ForceDictType(self.op.hvparams, constants.HVS_PARAMETER_TYPES)
+    filled_hvp = objects.FillDict(cluster.hvparams[self.op.hypervisor],
                                    self.op.hvparams)
      hv_type = hypervisor.GetHypervisor(self.op.hypervisor)
      hv_type.CheckParameterSyntax(filled_hvp)
                                    self.op.hvparams)
      hv_type = hypervisor.GetHypervisor(self.op.hypervisor)
      hv_type.CheckParameterSyntax(filled_hvp)
+    self.hv_full = filled_hvp
  
      # fill and remember the beparams dict
  
      # fill and remember the beparams dict
-    self.be_full = cluster.FillDict(cluster.beparams[constants.BEGR_DEFAULT],
+    utils.ForceDictType(self.op.beparams, constants.BES_PARAMETER_TYPES)
+    self.be_full = objects.FillDict(cluster.beparams[constants.PP_DEFAULT],
                                      self.op.beparams)
  
      #### instance parameters check
                                      self.op.beparams)
  
      #### instance parameters check
@@ -3408,10 +5143,21 @@ class LUCreateInstance(LogicalUnit):
  
      # NIC buildup
      self.nics = []
  
      # NIC buildup
      self.nics = []
-    for nic in self.op.nics:
+    for idx, nic in enumerate(self.op.nics):
+      nic_mode_req = nic.get("mode", None)
+      nic_mode = nic_mode_req
+      if nic_mode is None:
+        nic_mode = cluster.nicparams[constants.PP_DEFAULT][constants.NIC_MODE]
+
+      # in routed mode, for the first nic, the default ip is 'auto'
+      if nic_mode == constants.NIC_MODE_ROUTED and idx == 0:
+        default_ip_mode = constants.VALUE_AUTO
+      else:
+        default_ip_mode = constants.VALUE_NONE
+
        # ip validity checks
        # ip validity checks
-      ip = nic.get("ip", None)
-      if ip is None or ip.lower() == "none":
+      ip = nic.get("ip", default_ip_mode)
+      if ip is None or ip.lower() == constants.VALUE_NONE:
          nic_ip = None
        elif ip.lower() == constants.VALUE_AUTO:
          nic_ip = hostname1.ip
          nic_ip = None
        elif ip.lower() == constants.VALUE_AUTO:
          nic_ip = hostname1.ip
@@ -3421,6 +5167,10 @@ class LUCreateInstance(LogicalUnit):
                                       " like a valid IP" % ip)
          nic_ip = ip
  
                                       " like a valid IP" % ip)
          nic_ip = ip
  
+      # TODO: check the ip for uniqueness !!
+      if nic_mode == constants.NIC_MODE_ROUTED and not nic_ip:
+        raise errors.OpPrereqError("Routed nic mode requires an ip address")
+
        # MAC address verification
        mac = nic.get("mac", constants.VALUE_AUTO)
        if mac not in (constants.VALUE_AUTO, constants.VALUE_GENERATE):
        # MAC address verification
        mac = nic.get("mac", constants.VALUE_AUTO)
        if mac not in (constants.VALUE_AUTO, constants.VALUE_GENERATE):
@@ -3428,8 +5178,26 @@ class LUCreateInstance(LogicalUnit):
            raise errors.OpPrereqError("Invalid MAC address specified: %s" %
                                       mac)
        # bridge verification
            raise errors.OpPrereqError("Invalid MAC address specified: %s" %
                                       mac)
        # bridge verification
-      bridge = nic.get("bridge", self.cfg.GetDefBridge())
-      self.nics.append(objects.NIC(mac=mac, ip=nic_ip, bridge=bridge))
+      bridge = nic.get("bridge", None)
+      link = nic.get("link", None)
+      if bridge and link:
+        raise errors.OpPrereqError("Cannot pass 'bridge' and 'link'"
+                                   " at the same time")
+      elif bridge and nic_mode == constants.NIC_MODE_ROUTED:
+        raise errors.OpPrereqError("Cannot pass 'bridge' on a routed nic")
+      elif bridge:
+        link = bridge
+
+      nicparams = {}
+      if nic_mode_req:
+        nicparams[constants.NIC_MODE] = nic_mode_req
+      if link:
+        nicparams[constants.NIC_LINK] = link
+
+      check_params = objects.FillDict(cluster.nicparams[constants.PP_DEFAULT],
+                                      nicparams)
+      objects.NIC.CheckParameterSyntax(check_params)
+      self.nics.append(objects.NIC(mac=mac, ip=nic_ip, nicparams=nicparams))
  
      # disk checks/pre-build
      self.disks = []
  
      # disk checks/pre-build
      self.disks = []
@@ -3479,16 +5247,22 @@ class LUCreateInstance(LogicalUnit):
        src_node = getattr(self.op, "src_node", None)
        src_path = getattr(self.op, "src_path", None)
  
        src_node = getattr(self.op, "src_node", None)
        src_path = getattr(self.op, "src_path", None)
  
-      if src_node is None or src_path is None:
-        raise errors.OpPrereqError("Importing an instance requires source"
-                                   " node and path options")
-
-      if not os.path.isabs(src_path):
-        raise errors.OpPrereqError("The source path must be absolute")
+      if src_path is None:
+        self.op.src_path = src_path = self.op.instance_name
  
  
-      self.op.src_node = src_node = self._ExpandNode(src_node)
-      if self.needed_locks[locking.LEVEL_NODE] is not locking.ALL_SET:
-        self.needed_locks[locking.LEVEL_NODE].append(src_node)
+      if src_node is None:
+        self.needed_locks[locking.LEVEL_NODE] = locking.ALL_SET
+        self.op.src_node = None
+        if os.path.isabs(src_path):
+          raise errors.OpPrereqError("Importing an instance from an absolute"
+                                     " path requires a source node option.")
+      else:
+        self.op.src_node = src_node = self._ExpandNode(src_node)
+        if self.needed_locks[locking.LEVEL_NODE] is not locking.ALL_SET:
+          self.needed_locks[locking.LEVEL_NODE].append(src_node)
+        if not os.path.isabs(src_path):
+          self.op.src_path = src_path = \
+            os.path.join(constants.EXPORT_DIR, src_path)
  
      else: # INSTANCE_CREATE
        if getattr(self.op, "os_type", None) is None:
  
      else: # INSTANCE_CREATE
        if getattr(self.op, "os_type", None) is None:
@@ -3499,7 +5273,7 @@ class LUCreateInstance(LogicalUnit):
  
      """
      nics = [n.ToDict() for n in self.nics]
  
      """
      nics = [n.ToDict() for n in self.nics]
-    ial = IAllocator(self,
+    ial = IAllocator(self.cfg, self.rpc,
                       mode=constants.IALLOCATOR_MODE_ALLOC,
                       name=self.op.instance_name,
                       disk_template=self.op.disk_template,
                       mode=constants.IALLOCATOR_MODE_ALLOC,
                       name=self.op.instance_name,
                       disk_template=self.op.disk_template,
@@ -3509,6 +5283,7 @@ class LUCreateInstance(LogicalUnit):
                       mem_size=self.be_full[constants.BE_MEMORY],
                       disks=self.disks,
                       nics=nics,
                       mem_size=self.be_full[constants.BE_MEMORY],
                       disks=self.disks,
                       nics=nics,
+                     hypervisor=self.op.hypervisor,
                       )
  
      ial.Run(self.op.iallocator)
                       )
  
      ial.Run(self.op.iallocator)
@@ -3536,23 +5311,27 @@ class LUCreateInstance(LogicalUnit):
  
      """
      env = {
  
      """
      env = {
-      "INSTANCE_DISK_TEMPLATE": self.op.disk_template,
-      "INSTANCE_DISK_SIZE": ",".join(str(d["size"]) for d in self.disks),
-      "INSTANCE_ADD_MODE": self.op.mode,
+      "ADD_MODE": self.op.mode,
        }
      if self.op.mode == constants.INSTANCE_IMPORT:
        }
      if self.op.mode == constants.INSTANCE_IMPORT:
-      env["INSTANCE_SRC_NODE"] = self.op.src_node
-      env["INSTANCE_SRC_PATH"] = self.op.src_path
-      env["INSTANCE_SRC_IMAGES"] = self.src_images
+      env["SRC_NODE"] = self.op.src_node
+      env["SRC_PATH"] = self.op.src_path
+      env["SRC_IMAGES"] = self.src_images
  
  
-    env.update(_BuildInstanceHookEnv(name=self.op.instance_name,
+    env.update(_BuildInstanceHookEnv(
+      name=self.op.instance_name,
        primary_node=self.op.pnode,
        secondary_nodes=self.secondaries,
        primary_node=self.op.pnode,
        secondary_nodes=self.secondaries,
-      status=self.instance_status,
+      status=self.op.start,
        os_type=self.op.os_type,
        memory=self.be_full[constants.BE_MEMORY],
        vcpus=self.be_full[constants.BE_VCPUS],
        os_type=self.op.os_type,
        memory=self.be_full[constants.BE_MEMORY],
        vcpus=self.be_full[constants.BE_VCPUS],
-      nics=[(n.ip, n.bridge, n.mac) for n in self.nics],
+      nics=_NICListToTuple(self, self.nics),
+      disk_template=self.op.disk_template,
+      disks=[(d["size"], d["mode"]) for d in self.disks],
+      bep=self.be_full,
+      hvp=self.hv_full,
+      hypervisor_name=self.op.hypervisor,
      ))
  
      nl = ([self.cfg.GetMasterNode(), self.op.pnode] +
      ))
  
      nl = ([self.cfg.GetMasterNode(), self.op.pnode] +
@@ -3569,16 +5348,32 @@ class LUCreateInstance(LogicalUnit):
        raise errors.OpPrereqError("Cluster does not support lvm-based"
                                   " instances")
  
        raise errors.OpPrereqError("Cluster does not support lvm-based"
                                   " instances")
  
-
      if self.op.mode == constants.INSTANCE_IMPORT:
        src_node = self.op.src_node
        src_path = self.op.src_path
  
      if self.op.mode == constants.INSTANCE_IMPORT:
        src_node = self.op.src_node
        src_path = self.op.src_path
  
-      export_info = self.rpc.call_export_info(src_node, src_path)
-
-      if not export_info:
-        raise errors.OpPrereqError("No export found in dir %s" % src_path)
-
+      if src_node is None:
+        locked_nodes = self.acquired_locks[locking.LEVEL_NODE]
+        exp_list = self.rpc.call_export_list(locked_nodes)
+        found = False
+        for node in exp_list:
+          if exp_list[node].fail_msg:
+            continue
+          if src_path in exp_list[node].payload:
+            found = True
+            self.op.src_node = src_node = node
+            self.op.src_path = src_path = os.path.join(constants.EXPORT_DIR,
+                                                       src_path)
+            break
+        if not found:
+          raise errors.OpPrereqError("No export found for relative path %s" %
+                                      src_path)
+
+      _CheckNodeOnline(self, src_node)
+      result = self.rpc.call_export_info(src_node, src_path)
+      result.Raise("No export or invalid export found in dir %s" % src_path)
+
+      export_info = objects.SerializableConfigParser.Loads(str(result.payload))
        if not export_info.has_section(constants.INISECT_EXP):
          raise errors.ProgrammerError("Corrupted export config")
  
        if not export_info.has_section(constants.INISECT_EXP):
          raise errors.ProgrammerError("Corrupted export config")
  
@@ -3593,7 +5388,7 @@ class LUCreateInstance(LogicalUnit):
        if instance_disks < export_disks:
          raise errors.OpPrereqError("Not enough disks to import."
                                     " (instance: %d, export: %d)" %
        if instance_disks < export_disks:
          raise errors.OpPrereqError("Not enough disks to import."
                                     " (instance: %d, export: %d)" %
-                                   (2, export_disks))
+                                   (instance_disks, export_disks))
  
        self.op.os_type = export_info.get(constants.INISECT_EXP, 'os')
        disk_images = []
  
        self.op.os_type = export_info.get(constants.INISECT_EXP, 'os')
        disk_images = []
@@ -3609,16 +5404,17 @@ class LUCreateInstance(LogicalUnit):
  
        self.src_images = disk_images
  
  
        self.src_images = disk_images
  
-      if self.op.mac == constants.VALUE_AUTO:
-        old_name = export_info.get(constants.INISECT_INS, 'name')
-        if self.op.instance_name == old_name:
-          # FIXME: adjust every nic, when we'll be able to create instances
-          # with more than one
-          if int(export_info.get(constants.INISECT_INS, 'nic_count')) >= 1:
-            self.op.mac = export_info.get(constants.INISECT_INS, 'nic_0_mac')
+      old_name = export_info.get(constants.INISECT_INS, 'name')
+      # FIXME: int() here could throw a ValueError on broken exports
+      exp_nic_count = int(export_info.get(constants.INISECT_INS, 'nic_count'))
+      if self.op.instance_name == old_name:
+        for idx, nic in enumerate(self.nics):
+          if nic.mac == constants.VALUE_AUTO and exp_nic_count >= idx:
+            nic_mac_ini = 'nic%d_mac' % idx
+            nic.mac = export_info.get(constants.INISECT_INS, nic_mac_ini)
  
  
+    # ENDIF: self.op.mode == constants.INSTANCE_IMPORT
      # ip ping checks (we use the same ip that was resolved in ExpandNames)
      # ip ping checks (we use the same ip that was resolved in ExpandNames)
-
      if self.op.start and not self.op.ip_check:
        raise errors.OpPrereqError("Cannot ignore IP address conflicts when"
                                   " adding an instance in start mode")
      if self.op.start and not self.op.ip_check:
        raise errors.OpPrereqError("Cannot ignore IP address conflicts when"
                                   " adding an instance in start mode")
@@ -3628,6 +5424,18 @@ class LUCreateInstance(LogicalUnit):
          raise errors.OpPrereqError("IP %s of instance %s already in use" %
                                     (self.check_ip, self.op.instance_name))
  
          raise errors.OpPrereqError("IP %s of instance %s already in use" %
                                     (self.check_ip, self.op.instance_name))
  
+    #### mac address generation
+    # By generating here the mac address both the allocator and the hooks get
+    # the real final mac address rather than the 'auto' or 'generate' value.
+    # There is a race condition between the generation and the instance object
+    # creation, which means that we know the mac is valid now, but we're not
+    # sure it will be when we actually add the instance. If things go bad
+    # adding the instance will abort because of a duplicate mac, and the
+    # creation job will fail.
+    for nic in self.nics:
+      if nic.mac in (constants.VALUE_AUTO, constants.VALUE_GENERATE):
+        nic.mac = self.cfg.GenerateMAC()
+
      #### allocator run
  
      if self.op.iallocator is not None:
      #### allocator run
  
      if self.op.iallocator is not None:
@@ -3639,6 +5447,13 @@ class LUCreateInstance(LogicalUnit):
      self.pnode = pnode = self.cfg.GetNodeInfo(self.op.pnode)
      assert self.pnode is not None, \
        "Cannot retrieve locked node %s" % self.op.pnode
      self.pnode = pnode = self.cfg.GetNodeInfo(self.op.pnode)
      assert self.pnode is not None, \
        "Cannot retrieve locked node %s" % self.op.pnode
+    if pnode.offline:
+      raise errors.OpPrereqError("Cannot use offline primary node '%s'" %
+                                 pnode.name)
+    if pnode.drained:
+      raise errors.OpPrereqError("Cannot use drained primary node '%s'" %
+                                 pnode.name)
+
      self.secondaries = []
  
      # mirror node verification
      self.secondaries = []
  
      # mirror node verification
@@ -3649,6 +5464,8 @@ class LUCreateInstance(LogicalUnit):
        if self.op.snode == pnode.name:
          raise errors.OpPrereqError("The secondary node cannot be"
                                     " the primary node.")
        if self.op.snode == pnode.name:
          raise errors.OpPrereqError("The secondary node cannot be"
                                     " the primary node.")
+      _CheckNodeOnline(self, self.op.snode)
+      _CheckNodeNotDrained(self, self.op.snode)
        self.secondaries.append(self.op.snode)
  
      nodenames = [pnode.name] + self.secondaries
        self.secondaries.append(self.op.snode)
  
      nodenames = [pnode.name] + self.secondaries
@@ -3661,34 +5478,26 @@ class LUCreateInstance(LogicalUnit):
        nodeinfo = self.rpc.call_node_info(nodenames, self.cfg.GetVGName(),
                                           self.op.hypervisor)
        for node in nodenames:
        nodeinfo = self.rpc.call_node_info(nodenames, self.cfg.GetVGName(),
                                           self.op.hypervisor)
        for node in nodenames:
-        info = nodeinfo.get(node, None)
-        if not info:
-          raise errors.OpPrereqError("Cannot get current information"
-                                     " from node '%s'" % node)
+        info = nodeinfo[node]
+        info.Raise("Cannot get current information from node %s" % node)
+        info = info.payload
          vg_free = info.get('vg_free', None)
          if not isinstance(vg_free, int):
            raise errors.OpPrereqError("Can't compute free disk space on"
                                       " node %s" % node)
          vg_free = info.get('vg_free', None)
          if not isinstance(vg_free, int):
            raise errors.OpPrereqError("Can't compute free disk space on"
                                       " node %s" % node)
-        if req_size > info['vg_free']:
+        if req_size > vg_free:
            raise errors.OpPrereqError("Not enough disk space on target node %s."
                                       " %d MB available, %d MB required" %
            raise errors.OpPrereqError("Not enough disk space on target node %s."
                                       " %d MB available, %d MB required" %
-                                     (node, info['vg_free'], req_size))
+                                     (node, vg_free, req_size))
  
      _CheckHVParams(self, nodenames, self.op.hypervisor, self.op.hvparams)
  
      # os verification
  
      _CheckHVParams(self, nodenames, self.op.hypervisor, self.op.hvparams)
  
      # os verification
-    os_obj = self.rpc.call_os_get(pnode.name, self.op.os_type)
-    if not os_obj:
-      raise errors.OpPrereqError("OS '%s' not in supported os list for"
-                                 " primary node"  % self.op.os_type)
-
-    # bridge check on primary node
-    bridges = [n.bridge for n in self.nics]
-    if not self.rpc.call_bridges_exist(self.pnode.name, bridges):
-      raise errors.OpPrereqError("one of the target bridges '%s' does not"
-                                 " exist on"
-                                 " destination node '%s'" %
-                                 (",".join(bridges), pnode.name))
+    result = self.rpc.call_os_get(pnode.name, self.op.os_type)
+    result.Raise("OS '%s' not in supported os list for primary node %s" %
+                 (self.op.os_type, pnode.name), prereq=True)
+
+    _CheckNicsBridgesExist(self, self.nics, self.pnode.name)
  
      # memory check on primary node
      if self.op.start:
  
      # memory check on primary node
      if self.op.start:
@@ -3697,10 +5506,7 @@ class LUCreateInstance(LogicalUnit):
                             self.be_full[constants.BE_MEMORY],
                             self.op.hypervisor)
  
                             self.be_full[constants.BE_MEMORY],
                             self.op.hypervisor)
  
-    if self.op.start:
-      self.instance_status = 'up'
-    else:
-      self.instance_status = 'down'
+    self.dry_run_result = list(nodenames)
  
    def Exec(self, feedback_fn):
      """Create and add the instance to the cluster.
  
    def Exec(self, feedback_fn):
      """Create and add the instance to the cluster.
@@ -3709,10 +5515,6 @@ class LUCreateInstance(LogicalUnit):
      instance = self.op.instance_name
      pnode_name = self.pnode.name
  
      instance = self.op.instance_name
      pnode_name = self.pnode.name
  
-    for nic in self.nics:
-      if nic.mac in (constants.VALUE_AUTO, constants.VALUE_GENERATE):
-        nic.mac = self.cfg.GenerateMAC()
-
      ht_kind = self.op.hypervisor
      if ht_kind in constants.HTS_REQ_PORT:
        network_port = self.cfg.AllocatePort()
      ht_kind = self.op.hypervisor
      if ht_kind in constants.HTS_REQ_PORT:
        network_port = self.cfg.AllocatePort()
@@ -3740,13 +5542,14 @@ class LUCreateInstance(LogicalUnit):
                                    self.secondaries,
                                    self.disks,
                                    file_storage_dir,
                                    self.secondaries,
                                    self.disks,
                                    file_storage_dir,
-                                  self.op.file_driver)
+                                  self.op.file_driver,
+                                  0)
  
      iobj = objects.Instance(name=instance, os=self.op.os_type,
                              primary_node=pnode_name,
                              nics=self.nics, disks=disks,
                              disk_template=self.op.disk_template,
  
      iobj = objects.Instance(name=instance, os=self.op.os_type,
                              primary_node=pnode_name,
                              nics=self.nics, disks=disks,
                              disk_template=self.op.disk_template,
-                            status=self.instance_status,
+                            admin_up=False,
                              network_port=network_port,
                              beparams=self.op.beparams,
                              hvparams=self.op.hvparams,
                              network_port=network_port,
                              beparams=self.op.beparams,
                              hvparams=self.op.hvparams,
@@ -3754,10 +5557,15 @@ class LUCreateInstance(LogicalUnit):
                              )
  
      feedback_fn("* creating instance disks...")
                              )
  
      feedback_fn("* creating instance disks...")
-    if not _CreateDisks(self, iobj):
-      _RemoveDisks(self, iobj)
-      self.cfg.ReleaseDRBDMinors(instance)
-      raise errors.OpExecError("Device creation failed, reverting...")
+    try:
+      _CreateDisks(self, iobj)
+    except errors.OpExecError:
+      self.LogWarning("Device creation failed, reverting...")
+      try:
+        _RemoveDisks(self, iobj)
+      finally:
+        self.cfg.ReleaseDRBDMinors(instance)
+        raise
  
      feedback_fn("adding instance %s to cluster config" % instance)
  
  
      feedback_fn("adding instance %s to cluster config" % instance)
  
@@ -3765,8 +5573,16 @@ class LUCreateInstance(LogicalUnit):
      # Declare that we don't want to remove the instance lock anymore, as we've
      # added the instance to the config
      del self.remove_locks[locking.LEVEL_INSTANCE]
      # Declare that we don't want to remove the instance lock anymore, as we've
      # added the instance to the config
      del self.remove_locks[locking.LEVEL_INSTANCE]
-    # Remove the temp. assignements for the instance's drbds
-    self.cfg.ReleaseDRBDMinors(instance)
+    # Unlock all the nodes
+    if self.op.mode == constants.INSTANCE_IMPORT:
+      nodes_keep = [self.op.src_node]
+      nodes_release = [node for node in self.acquired_locks[locking.LEVEL_NODE]
+                       if node != self.op.src_node]
+      self.context.glm.release(locking.LEVEL_NODE, nodes_release)
+      self.acquired_locks[locking.LEVEL_NODE] = nodes_keep
+    else:
+      self.context.glm.release(locking.LEVEL_NODE)
+      del self.acquired_locks[locking.LEVEL_NODE]
  
      if self.op.wait_for_sync:
        disk_abort = not _WaitForSync(self, iobj)
  
      if self.op.wait_for_sync:
        disk_abort = not _WaitForSync(self, iobj)
@@ -3792,10 +5608,9 @@ class LUCreateInstance(LogicalUnit):
      if iobj.disk_template != constants.DT_DISKLESS:
        if self.op.mode == constants.INSTANCE_CREATE:
          feedback_fn("* running the instance OS create scripts...")
      if iobj.disk_template != constants.DT_DISKLESS:
        if self.op.mode == constants.INSTANCE_CREATE:
          feedback_fn("* running the instance OS create scripts...")
-        if not self.rpc.call_instance_os_add(pnode_name, iobj):
-          raise errors.OpExecError("could not add os for instance %s"
-                                   " on node %s" %
-                                   (instance, pnode_name))
+        result = self.rpc.call_instance_os_add(pnode_name, iobj, False)
+        result.Raise("Could not add os for instance %s"
+                     " on node %s" % (instance, pnode_name))
  
        elif self.op.mode == constants.INSTANCE_IMPORT:
          feedback_fn("* running the instance OS import scripts...")
  
        elif self.op.mode == constants.INSTANCE_IMPORT:
          feedback_fn("* running the instance OS import scripts...")
@@ -3805,21 +5620,24 @@ class LUCreateInstance(LogicalUnit):
          import_result = self.rpc.call_instance_os_import(pnode_name, iobj,
                                                           src_node, src_images,
                                                           cluster_name)
          import_result = self.rpc.call_instance_os_import(pnode_name, iobj,
                                                           src_node, src_images,
                                                           cluster_name)
-        for idx, result in enumerate(import_result):
-          if not result:
-            self.LogWarning("Could not image %s for on instance %s, disk %d,"
-                            " on node %s" % (src_images[idx], instance, idx,
-                                             pnode_name))
+        msg = import_result.fail_msg
+        if msg:
+          self.LogWarning("Error while importing the disk images for instance"
+                          " %s on node %s: %s" % (instance, pnode_name, msg))
        else:
          # also checked in the prereq part
          raise errors.ProgrammerError("Unknown OS initialization mode '%s'"
                                       % self.op.mode)
  
      if self.op.start:
        else:
          # also checked in the prereq part
          raise errors.ProgrammerError("Unknown OS initialization mode '%s'"
                                       % self.op.mode)
  
      if self.op.start:
+      iobj.admin_up = True
+      self.cfg.Update(iobj)
        logging.info("Starting instance %s on node %s", instance, pnode_name)
        feedback_fn("* starting instance...")
        logging.info("Starting instance %s on node %s", instance, pnode_name)
        feedback_fn("* starting instance...")
-      if not self.rpc.call_instance_start(pnode_name, iobj, None):
-        raise errors.OpExecError("Could not start instance")
+      result = self.rpc.call_instance_start(pnode_name, iobj, None, None)
+      result.Raise("Could not start instance")
+
+    return list(iobj.all_nodes)
  
  
  class LUConnectConsole(NoHooksLU):
  
  
  class LUConnectConsole(NoHooksLU):
@@ -3845,6 +5663,7 @@ class LUConnectConsole(NoHooksLU):
      self.instance = self.cfg.GetInstanceInfo(self.op.instance_name)
      assert self.instance is not None, \
        "Cannot retrieve locked instance %s" % self.op.instance_name
      self.instance = self.cfg.GetInstanceInfo(self.op.instance_name)
      assert self.instance is not None, \
        "Cannot retrieve locked instance %s" % self.op.instance_name
+    _CheckNodeOnline(self, self.instance.primary_node)
  
    def Exec(self, feedback_fn):
      """Connect to the console of an instance
  
    def Exec(self, feedback_fn):
      """Connect to the console of an instance
@@ -3855,16 +5674,20 @@ class LUConnectConsole(NoHooksLU):
  
      node_insts = self.rpc.call_instance_list([node],
                                               [instance.hypervisor])[node]
  
      node_insts = self.rpc.call_instance_list([node],
                                               [instance.hypervisor])[node]
-    if node_insts is False:
-      raise errors.OpExecError("Can't connect to node %s." % node)
+    node_insts.Raise("Can't get node information from %s" % node)
  
  
-    if instance.name not in node_insts:
+    if instance.name not in node_insts.payload:
        raise errors.OpExecError("Instance %s is not running." % instance.name)
  
      logging.debug("Connecting to console of %s on %s", instance.name, node)
  
      hyper = hypervisor.GetHypervisor(instance.hypervisor)
        raise errors.OpExecError("Instance %s is not running." % instance.name)
  
      logging.debug("Connecting to console of %s on %s", instance.name, node)
  
      hyper = hypervisor.GetHypervisor(instance.hypervisor)
-    console_cmd = hyper.GetShellCommandForConsole(instance)
+    cluster = self.cfg.GetClusterInfo()
+    # beparams and hvparams are passed separately, to avoid editing the
+    # instance and then saving the defaults in the instance itself.
+    hvparams = cluster.FillHV(instance)
+    beparams = cluster.FillBE(instance)
+    console_cmd = hyper.GetShellCommandForConsole(instance, hvparams, beparams)
  
      # build ssh cmdline
      return self.ssh.BuildCmd(node, "root", console_cmd, batch=True, tty=True)
  
      # build ssh cmdline
      return self.ssh.BuildCmd(node, "root", console_cmd, batch=True, tty=True)
@@ -3879,30 +5702,46 @@ class LUReplaceDisks(LogicalUnit):
    _OP_REQP = ["instance_name", "mode", "disks"]
    REQ_BGL = False
  
    _OP_REQP = ["instance_name", "mode", "disks"]
    REQ_BGL = False
  
-  def ExpandNames(self):
-    self._ExpandAndLockInstance()
-
+  def CheckArguments(self):
      if not hasattr(self.op, "remote_node"):
        self.op.remote_node = None
      if not hasattr(self.op, "remote_node"):
        self.op.remote_node = None
+    if not hasattr(self.op, "iallocator"):
+      self.op.iallocator = None
  
  
-    ia_name = getattr(self.op, "iallocator", None)
-    if ia_name is not None:
-      if self.op.remote_node is not None:
-        raise errors.OpPrereqError("Give either the iallocator or the new"
-                                   " secondary, not both")
+    TLReplaceDisks.CheckArguments(self.op.mode, self.op.remote_node,
+                                  self.op.iallocator)
+
+  def ExpandNames(self):
+    self._ExpandAndLockInstance()
+
+    if self.op.iallocator is not None:
        self.needed_locks[locking.LEVEL_NODE] = locking.ALL_SET
        self.needed_locks[locking.LEVEL_NODE] = locking.ALL_SET
+
      elif self.op.remote_node is not None:
        remote_node = self.cfg.ExpandNodeName(self.op.remote_node)
        if remote_node is None:
          raise errors.OpPrereqError("Node '%s' not known" %
                                     self.op.remote_node)
      elif self.op.remote_node is not None:
        remote_node = self.cfg.ExpandNodeName(self.op.remote_node)
        if remote_node is None:
          raise errors.OpPrereqError("Node '%s' not known" %
                                     self.op.remote_node)
+
        self.op.remote_node = remote_node
        self.op.remote_node = remote_node
+
+      # Warning: do not remove the locking of the new secondary here
+      # unless DRBD8.AddChildren is changed to work in parallel;
+      # currently it doesn't since parallel invocations of
+      # FindUnusedMinor will conflict
        self.needed_locks[locking.LEVEL_NODE] = [remote_node]
        self.recalculate_locks[locking.LEVEL_NODE] = constants.LOCKS_APPEND
        self.needed_locks[locking.LEVEL_NODE] = [remote_node]
        self.recalculate_locks[locking.LEVEL_NODE] = constants.LOCKS_APPEND
+
      else:
        self.needed_locks[locking.LEVEL_NODE] = []
        self.recalculate_locks[locking.LEVEL_NODE] = constants.LOCKS_REPLACE
  
      else:
        self.needed_locks[locking.LEVEL_NODE] = []
        self.recalculate_locks[locking.LEVEL_NODE] = constants.LOCKS_REPLACE
  
+    self.replacer = TLReplaceDisks(self, self.op.instance_name, self.op.mode,
+                                   self.op.iallocator, self.op.remote_node,
+                                   self.op.disks)
+
+    self.tasklets = [self.replacer]
+
    def DeclareLocks(self, level):
      # If we're not already locking all nodes in the set we have to declare the
      # instance's primary/secondary nodes.
    def DeclareLocks(self, level):
      # If we're not already locking all nodes in the set we have to declare the
      # instance's primary/secondary nodes.
@@ -3910,124 +5749,458 @@ class LUReplaceDisks(LogicalUnit):
          self.needed_locks[locking.LEVEL_NODE] is not locking.ALL_SET):
        self._LockInstancesNodes()
  
          self.needed_locks[locking.LEVEL_NODE] is not locking.ALL_SET):
        self._LockInstancesNodes()
  
-  def _RunAllocator(self):
-    """Compute a new secondary node using an IAllocator.
-
-    """
-    ial = IAllocator(self,
-                     mode=constants.IALLOCATOR_MODE_RELOC,
-                     name=self.op.instance_name,
-                     relocate_from=[self.sec_node])
-
-    ial.Run(self.op.iallocator)
-
-    if not ial.success:
-      raise errors.OpPrereqError("Can't compute nodes using"
-                                 " iallocator '%s': %s" % (self.op.iallocator,
-                                                           ial.info))
-    if len(ial.nodes) != ial.required_nodes:
-      raise errors.OpPrereqError("iallocator '%s' returned invalid number"
-                                 " of nodes (%s), required %s" %
-                                 (len(ial.nodes), ial.required_nodes))
-    self.op.remote_node = ial.nodes[0]
-    self.LogInfo("Selected new secondary for the instance: %s",
-                 self.op.remote_node)
-
    def BuildHooksEnv(self):
      """Build hooks env.
  
      This runs on the master, the primary and all the secondaries.
  
      """
    def BuildHooksEnv(self):
      """Build hooks env.
  
      This runs on the master, the primary and all the secondaries.
  
      """
+    instance = self.replacer.instance
      env = {
        "MODE": self.op.mode,
        "NEW_SECONDARY": self.op.remote_node,
      env = {
        "MODE": self.op.mode,
        "NEW_SECONDARY": self.op.remote_node,
-      "OLD_SECONDARY": self.instance.secondary_nodes[0],
+      "OLD_SECONDARY": instance.secondary_nodes[0],
        }
        }
-    env.update(_BuildInstanceHookEnvByObject(self, self.instance))
+    env.update(_BuildInstanceHookEnvByObject(self, instance))
      nl = [
        self.cfg.GetMasterNode(),
      nl = [
        self.cfg.GetMasterNode(),
-      self.instance.primary_node,
+      instance.primary_node,
        ]
      if self.op.remote_node is not None:
        nl.append(self.op.remote_node)
      return env, nl, nl
  
        ]
      if self.op.remote_node is not None:
        nl.append(self.op.remote_node)
      return env, nl, nl
  
+
+class LUEvacuateNode(LogicalUnit):
+  """Relocate the secondary instances from a node.
+
+  """
+  HPATH = "node-evacuate"
+  HTYPE = constants.HTYPE_NODE
+  _OP_REQP = ["node_name"]
+  REQ_BGL = False
+
+  def CheckArguments(self):
+    if not hasattr(self.op, "remote_node"):
+      self.op.remote_node = None
+    if not hasattr(self.op, "iallocator"):
+      self.op.iallocator = None
+
+    TLReplaceDisks.CheckArguments(constants.REPLACE_DISK_CHG,
+                                  self.op.remote_node,
+                                  self.op.iallocator)
+
+  def ExpandNames(self):
+    self.op.node_name = self.cfg.ExpandNodeName(self.op.node_name)
+    if self.op.node_name is None:
+      raise errors.OpPrereqError("Node '%s' not known" % self.op.node_name)
+
+    self.needed_locks = {}
+
+    # Declare node locks
+    if self.op.iallocator is not None:
+      self.needed_locks[locking.LEVEL_NODE] = locking.ALL_SET
+
+    elif self.op.remote_node is not None:
+      remote_node = self.cfg.ExpandNodeName(self.op.remote_node)
+      if remote_node is None:
+        raise errors.OpPrereqError("Node '%s' not known" %
+                                   self.op.remote_node)
+
+      self.op.remote_node = remote_node
+
+      # Warning: do not remove the locking of the new secondary here
+      # unless DRBD8.AddChildren is changed to work in parallel;
+      # currently it doesn't since parallel invocations of
+      # FindUnusedMinor will conflict
+      self.needed_locks[locking.LEVEL_NODE] = [remote_node]
+      self.recalculate_locks[locking.LEVEL_NODE] = constants.LOCKS_APPEND
+
+    else:
+      raise errors.OpPrereqError("Invalid parameters")
+
+    # Create tasklets for replacing disks for all secondary instances on this
+    # node
+    names = []
+    tasklets = []
+
+    for inst in _GetNodeSecondaryInstances(self.cfg, self.op.node_name):
+      logging.debug("Replacing disks for instance %s", inst.name)
+      names.append(inst.name)
+
+      replacer = TLReplaceDisks(self, inst.name, constants.REPLACE_DISK_CHG,
+                                self.op.iallocator, self.op.remote_node, [])
+      tasklets.append(replacer)
+
+    self.tasklets = tasklets
+    self.instance_names = names
+
+    # Declare instance locks
+    self.needed_locks[locking.LEVEL_INSTANCE] = self.instance_names
+
+  def DeclareLocks(self, level):
+    # If we're not already locking all nodes in the set we have to declare the
+    # instance's primary/secondary nodes.
+    if (level == locking.LEVEL_NODE and
+        self.needed_locks[locking.LEVEL_NODE] is not locking.ALL_SET):
+      self._LockInstancesNodes()
+
+  def BuildHooksEnv(self):
+    """Build hooks env.
+
+    This runs on the master, the primary and all the secondaries.
+
+    """
+    env = {
+      "NODE_NAME": self.op.node_name,
+      }
+
+    nl = [self.cfg.GetMasterNode()]
+
+    if self.op.remote_node is not None:
+      env["NEW_SECONDARY"] = self.op.remote_node
+      nl.append(self.op.remote_node)
+
+    return (env, nl, nl)
+
+
+class TLReplaceDisks(Tasklet):
+  """Replaces disks for an instance.
+
+  Note: Locking is not within the scope of this class.
+
+  """
+  def __init__(self, lu, instance_name, mode, iallocator_name, remote_node,
+               disks):
+    """Initializes this class.
+
+    """
+    Tasklet.__init__(self, lu)
+
+    # Parameters
+    self.instance_name = instance_name
+    self.mode = mode
+    self.iallocator_name = iallocator_name
+    self.remote_node = remote_node
+    self.disks = disks
+
+    # Runtime data
+    self.instance = None
+    self.new_node = None
+    self.target_node = None
+    self.other_node = None
+    self.remote_node_info = None
+    self.node_secondary_ip = None
+
+  @staticmethod
+  def CheckArguments(mode, remote_node, iallocator):
+    """Helper function for users of this class.
+
+    """
+    # check for valid parameter combination
+    if mode == constants.REPLACE_DISK_CHG:
+      if remote_node is None and iallocator is None:
+        raise errors.OpPrereqError("When changing the secondary either an"
+                                   " iallocator script must be used or the"
+                                   " new node given")
+
+      if remote_node is not None and iallocator is not None:
+        raise errors.OpPrereqError("Give either the iallocator or the new"
+                                   " secondary, not both")
+
+    elif remote_node is not None or iallocator is not None:
+      # Not replacing the secondary
+      raise errors.OpPrereqError("The iallocator and new node options can"
+                                 " only be used when changing the"
+                                 " secondary node")
+
+  @staticmethod
+  def _RunAllocator(lu, iallocator_name, instance_name, relocate_from):
+    """Compute a new secondary node using an IAllocator.
+
+    """
+    ial = IAllocator(lu.cfg, lu.rpc,
+                     mode=constants.IALLOCATOR_MODE_RELOC,
+                     name=instance_name,
+                     relocate_from=relocate_from)
+
+    ial.Run(iallocator_name)
+
+    if not ial.success:
+      raise errors.OpPrereqError("Can't compute nodes using iallocator '%s':"
+                                 " %s" % (iallocator_name, ial.info))
+
+    if len(ial.nodes) != ial.required_nodes:
+      raise errors.OpPrereqError("iallocator '%s' returned invalid number"
+                                 " of nodes (%s), required %s" %
+                                 (len(ial.nodes), ial.required_nodes))
+
+    remote_node_name = ial.nodes[0]
+
+    lu.LogInfo("Selected new secondary for instance '%s': %s",
+               instance_name, remote_node_name)
+
+    return remote_node_name
+
+  def _FindFaultyDisks(self, node_name):
+    return _FindFaultyInstanceDisks(self.cfg, self.rpc, self.instance,
+                                    node_name, True)
+
    def CheckPrereq(self):
      """Check prerequisites.
  
      This checks that the instance is in the cluster.
  
      """
    def CheckPrereq(self):
      """Check prerequisites.
  
      This checks that the instance is in the cluster.
  
      """
-    instance = self.cfg.GetInstanceInfo(self.op.instance_name)
-    assert instance is not None, \
-      "Cannot retrieve locked instance %s" % self.op.instance_name
-    self.instance = instance
+    self.instance = self.cfg.GetInstanceInfo(self.instance_name)
+    assert self.instance is not None, \
+      "Cannot retrieve locked instance %s" % self.instance_name
  
  
-    if instance.disk_template not in constants.DTS_NET_MIRROR:
-      raise errors.OpPrereqError("Instance's disk layout is not"
-                                 " network mirrored.")
+    if self.instance.disk_template != constants.DT_DRBD8:
+      raise errors.OpPrereqError("Can only run replace disks for DRBD8-based"
+                                 " instances")
  
  
-    if len(instance.secondary_nodes) != 1:
+    if len(self.instance.secondary_nodes) != 1:
        raise errors.OpPrereqError("The instance has a strange layout,"
                                   " expected one secondary but found %d" %
        raise errors.OpPrereqError("The instance has a strange layout,"
                                   " expected one secondary but found %d" %
-                                 len(instance.secondary_nodes))
+                                 len(self.instance.secondary_nodes))
  
  
-    self.sec_node = instance.secondary_nodes[0]
+    secondary_node = self.instance.secondary_nodes[0]
  
  
-    ia_name = getattr(self.op, "iallocator", None)
-    if ia_name is not None:
-      self._RunAllocator()
+    if self.iallocator_name is None:
+      remote_node = self.remote_node
+    else:
+      remote_node = self._RunAllocator(self.lu, self.iallocator_name,
+                                       self.instance.name, secondary_node)
  
  
-    remote_node = self.op.remote_node
      if remote_node is not None:
        self.remote_node_info = self.cfg.GetNodeInfo(remote_node)
        assert self.remote_node_info is not None, \
          "Cannot retrieve locked node %s" % remote_node
      else:
        self.remote_node_info = None
      if remote_node is not None:
        self.remote_node_info = self.cfg.GetNodeInfo(remote_node)
        assert self.remote_node_info is not None, \
          "Cannot retrieve locked node %s" % remote_node
      else:
        self.remote_node_info = None
-    if remote_node == instance.primary_node:
+
+    if remote_node == self.instance.primary_node:
        raise errors.OpPrereqError("The specified node is the primary node of"
                                   " the instance.")
        raise errors.OpPrereqError("The specified node is the primary node of"
                                   " the instance.")
-    elif remote_node == self.sec_node:
-      if self.op.mode == constants.REPLACE_DISK_SEC:
-        # this is for DRBD8, where we can't execute the same mode of
-        # replacement as for drbd7 (no different port allocated)
-        raise errors.OpPrereqError("Same secondary given, cannot execute"
-                                   " replacement")
-    if instance.disk_template == constants.DT_DRBD8:
-      if (self.op.mode == constants.REPLACE_DISK_ALL and
-          remote_node is not None):
-        # switch to replace secondary mode
-        self.op.mode = constants.REPLACE_DISK_SEC
-
-      if self.op.mode == constants.REPLACE_DISK_ALL:
-        raise errors.OpPrereqError("Template 'drbd' only allows primary or"
-                                   " secondary disk replacement, not"
-                                   " both at once")
-      elif self.op.mode == constants.REPLACE_DISK_PRI:
-        if remote_node is not None:
-          raise errors.OpPrereqError("Template 'drbd' does not allow changing"
-                                     " the secondary while doing a primary"
-                                     " node disk replacement")
-        self.tgt_node = instance.primary_node
-        self.oth_node = instance.secondary_nodes[0]
-      elif self.op.mode == constants.REPLACE_DISK_SEC:
-        self.new_node = remote_node # this can be None, in which case
-                                    # we don't change the secondary
-        self.tgt_node = instance.secondary_nodes[0]
-        self.oth_node = instance.primary_node
+
+    if remote_node == secondary_node:
+      raise errors.OpPrereqError("The specified node is already the"
+                                 " secondary node of the instance.")
+
+    if self.disks and self.mode in (constants.REPLACE_DISK_AUTO,
+                                    constants.REPLACE_DISK_CHG):
+      raise errors.OpPrereqError("Cannot specify disks to be replaced")
+
+    if self.mode == constants.REPLACE_DISK_AUTO:
+      faulty_primary = self._FindFaultyDisks(self.instance.primary_node)
+      faulty_secondary = self._FindFaultyDisks(secondary_node)
+
+      if faulty_primary and faulty_secondary:
+        raise errors.OpPrereqError("Instance %s has faulty disks on more than"
+                                   " one node and can not be repaired"
+                                   " automatically" % self.instance_name)
+
+      if faulty_primary:
+        self.disks = faulty_primary
+        self.target_node = self.instance.primary_node
+        self.other_node = secondary_node
+        check_nodes = [self.target_node, self.other_node]
+      elif faulty_secondary:
+        self.disks = faulty_secondary
+        self.target_node = secondary_node
+        self.other_node = self.instance.primary_node
+        check_nodes = [self.target_node, self.other_node]
        else:
        else:
-        raise errors.ProgrammerError("Unhandled disk replace mode")
+        self.disks = []
+        check_nodes = []
  
  
-    if not self.op.disks:
-      self.op.disks = range(len(instance.disks))
+    else:
+      # Non-automatic modes
+      if self.mode == constants.REPLACE_DISK_PRI:
+        self.target_node = self.instance.primary_node
+        self.other_node = secondary_node
+        check_nodes = [self.target_node, self.other_node]
+
+      elif self.mode == constants.REPLACE_DISK_SEC:
+        self.target_node = secondary_node
+        self.other_node = self.instance.primary_node
+        check_nodes = [self.target_node, self.other_node]
+
+      elif self.mode == constants.REPLACE_DISK_CHG:
+        self.new_node = remote_node
+        self.other_node = self.instance.primary_node
+        self.target_node = secondary_node
+        check_nodes = [self.new_node, self.other_node]
+
+        _CheckNodeNotDrained(self.lu, remote_node)
+
+      else:
+        raise errors.ProgrammerError("Unhandled disk replace mode (%s)" %
+                                     self.mode)
+
+      # If not specified all disks should be replaced
+      if not self.disks:
+        self.disks = range(len(self.instance.disks))
+
+    for node in check_nodes:
+      _CheckNodeOnline(self.lu, node)
+
+    # Check whether disks are valid
+    for disk_idx in self.disks:
+      self.instance.FindDisk(disk_idx)
+
+    # Get secondary node IP addresses
+    node_2nd_ip = {}
+
+    for node_name in [self.target_node, self.other_node, self.new_node]:
+      if node_name is not None:
+        node_2nd_ip[node_name] = self.cfg.GetNodeInfo(node_name).secondary_ip
+
+    self.node_secondary_ip = node_2nd_ip
+
+  def Exec(self, feedback_fn):
+    """Execute disk replacement.
+
+    This dispatches the disk replacement to the appropriate handler.
+
+    """
+    if not self.disks:
+      feedback_fn("No disks need replacement")
+      return
+
+    feedback_fn("Replacing disk(s) %s for %s" %
+                (", ".join([str(i) for i in self.disks]), self.instance.name))
+
+    activate_disks = (not self.instance.admin_up)
+
+    # Activate the instance disks if we're replacing them on a down instance
+    if activate_disks:
+      _StartInstanceDisks(self.lu, self.instance, True)
+
+    try:
+      # Should we replace the secondary node?
+      if self.new_node is not None:
+        return self._ExecDrbd8Secondary()
+      else:
+        return self._ExecDrbd8DiskOnly()
+
+    finally:
+      # Deactivate the instance disks if we're replacing them on a down instance
+      if activate_disks:
+        _SafeShutdownInstanceDisks(self.lu, self.instance)
+
+  def _CheckVolumeGroup(self, nodes):
+    self.lu.LogInfo("Checking volume groups")
+
+    vgname = self.cfg.GetVGName()
+
+    # Make sure volume group exists on all involved nodes
+    results = self.rpc.call_vg_list(nodes)
+    if not results:
+      raise errors.OpExecError("Can't list volume groups on the nodes")
+
+    for node in nodes:
+      res = results[node]
+      res.Raise("Error checking node %s" % node)
+      if vgname not in res.payload:
+        raise errors.OpExecError("Volume group '%s' not found on node %s" %
+                                 (vgname, node))
+
+  def _CheckDisksExistence(self, nodes):
+    # Check disk existence
+    for idx, dev in enumerate(self.instance.disks):
+      if idx not in self.disks:
+        continue
+
+      for node in nodes:
+        self.lu.LogInfo("Checking disk/%d on %s" % (idx, node))
+        self.cfg.SetDiskID(dev, node)
+
+        result = self.rpc.call_blockdev_find(node, dev)
+
+        msg = result.fail_msg
+        if msg or not result.payload:
+          if not msg:
+            msg = "disk not found"
+          raise errors.OpExecError("Can't find disk/%d on node %s: %s" %
+                                   (idx, node, msg))
+
+  def _CheckDisksConsistency(self, node_name, on_primary, ldisk):
+    for idx, dev in enumerate(self.instance.disks):
+      if idx not in self.disks:
+        continue
+
+      self.lu.LogInfo("Checking disk/%d consistency on node %s" %
+                      (idx, node_name))
+
+      if not _CheckDiskConsistency(self.lu, dev, node_name, on_primary,
+                                   ldisk=ldisk):
+        raise errors.OpExecError("Node %s has degraded storage, unsafe to"
+                                 " replace disks for instance %s" %
+                                 (node_name, self.instance.name))
+
+  def _CreateNewStorage(self, node_name):
+    vgname = self.cfg.GetVGName()
+    iv_names = {}
  
  
-    for disk_idx in self.op.disks:
-      instance.FindDisk(disk_idx)
+    for idx, dev in enumerate(self.instance.disks):
+      if idx not in self.disks:
+        continue
+
+      self.lu.LogInfo("Adding storage on %s for disk/%d" % (node_name, idx))
+
+      self.cfg.SetDiskID(dev, node_name)
+
+      lv_names = [".disk%d_%s" % (idx, suffix) for suffix in ["data", "meta"]]
+      names = _GenerateUniqueNames(self.lu, lv_names)
+
+      lv_data = objects.Disk(dev_type=constants.LD_LV, size=dev.size,
+                             logical_id=(vgname, names[0]))
+      lv_meta = objects.Disk(dev_type=constants.LD_LV, size=128,
+                             logical_id=(vgname, names[1]))
+
+      new_lvs = [lv_data, lv_meta]
+      old_lvs = dev.children
+      iv_names[dev.iv_name] = (dev, old_lvs, new_lvs)
+
+      # we pass force_create=True to force the LVM creation
+      for new_lv in new_lvs:
+        _CreateBlockDev(self.lu, node_name, self.instance, new_lv, True,
+                        _GetInstanceInfoText(self.instance), False)
+
+    return iv_names
+
+  def _CheckDevices(self, node_name, iv_names):
+    for name, (dev, old_lvs, new_lvs) in iv_names.iteritems():
+      self.cfg.SetDiskID(dev, node_name)
+
+      result = self.rpc.call_blockdev_find(node_name, dev)
+
+      msg = result.fail_msg
+      if msg or not result.payload:
+        if not msg:
+          msg = "disk not found"
+        raise errors.OpExecError("Can't find DRBD device %s: %s" %
+                                 (name, msg))
+
+      if result.payload.is_degraded:
+        raise errors.OpExecError("DRBD device %s is degraded!" % name)
+
+  def _RemoveOldStorage(self, node_name, iv_names):
+    for name, (dev, old_lvs, _) in iv_names.iteritems():
+      self.lu.LogInfo("Remove logical volumes for %s" % name)
+
+      for lv in old_lvs:
+        self.cfg.SetDiskID(lv, node_name)
  
  
-  def _ExecD8DiskOnly(self, feedback_fn):
-    """Replace a disk on the primary or secondary for dbrd8.
+        msg = self.rpc.call_blockdev_remove(node_name, lv).fail_msg
+        if msg:
+          self.lu.LogWarning("Can't remove old LV: %s" % msg,
+                             hint="remove unused LVs manually")
+
+  def _ExecDrbd8DiskOnly(self):
+    """Replace a disk on the primary or secondary for DRBD 8.
  
      The algorithm for replace is quite complicated:
  
  
      The algorithm for replace is quite complicated:
  
@@ -4049,85 +6222,30 @@ class LUReplaceDisks(LogicalUnit):
  
      """
      steps_total = 6
  
      """
      steps_total = 6
-    warning, info = (self.proc.LogWarning, self.proc.LogInfo)
-    instance = self.instance
-    iv_names = {}
-    vgname = self.cfg.GetVGName()
-    # start of work
-    cfg = self.cfg
-    tgt_node = self.tgt_node
-    oth_node = self.oth_node
  
      # Step: check device activation
  
      # Step: check device activation
-    self.proc.LogStep(1, steps_total, "check device existence")
-    info("checking volume groups")
-    my_vg = cfg.GetVGName()
-    results = self.rpc.call_vg_list([oth_node, tgt_node])
-    if not results:
-      raise errors.OpExecError("Can't list volume groups on the nodes")
-    for node in oth_node, tgt_node:
-      res = results.get(node, False)
-      if not res or my_vg not in res:
-        raise errors.OpExecError("Volume group '%s' not found on %s" %
-                                 (my_vg, node))
-    for idx, dev in enumerate(instance.disks):
-      if idx not in self.op.disks:
-        continue
-      for node in tgt_node, oth_node:
-        info("checking disk/%d on %s" % (idx, node))
-        cfg.SetDiskID(dev, node)
-        if not self.rpc.call_blockdev_find(node, dev):
-          raise errors.OpExecError("Can't find disk/%d on node %s" %
-                                   (idx, node))
+    self.lu.LogStep(1, steps_total, "Check device existence")
+    self._CheckDisksExistence([self.other_node, self.target_node])
+    self._CheckVolumeGroup([self.target_node, self.other_node])
  
      # Step: check other node consistency
  
      # Step: check other node consistency
-    self.proc.LogStep(2, steps_total, "check peer consistency")
-    for idx, dev in enumerate(instance.disks):
-      if idx not in self.op.disks:
-        continue
-      info("checking disk/%d consistency on %s" % (idx, oth_node))
-      if not _CheckDiskConsistency(self, dev, oth_node,
-                                   oth_node==instance.primary_node):
-        raise errors.OpExecError("Peer node (%s) has degraded storage, unsafe"
-                                 " to replace disks on this node (%s)" %
-                                 (oth_node, tgt_node))
+    self.lu.LogStep(2, steps_total, "Check peer consistency")
+    self._CheckDisksConsistency(self.other_node,
+                                self.other_node == self.instance.primary_node,
+                                False)
  
      # Step: create new storage
  
      # Step: create new storage
-    self.proc.LogStep(3, steps_total, "allocate new storage")
-    for idx, dev in enumerate(instance.disks):
-      if idx not in self.op.disks:
-        continue
-      size = dev.size
-      cfg.SetDiskID(dev, tgt_node)
-      lv_names = [".disk%d_%s" % (idx, suf)
-                  for suf in ["data", "meta"]]
-      names = _GenerateUniqueNames(self, lv_names)
-      lv_data = objects.Disk(dev_type=constants.LD_LV, size=size,
-                             logical_id=(vgname, names[0]))
-      lv_meta = objects.Disk(dev_type=constants.LD_LV, size=128,
-                             logical_id=(vgname, names[1]))
-      new_lvs = [lv_data, lv_meta]
-      old_lvs = dev.children
-      iv_names[dev.iv_name] = (dev, old_lvs, new_lvs)
-      info("creating new local storage on %s for %s" %
-           (tgt_node, dev.iv_name))
-      # since we *always* want to create this LV, we use the
-      # _Create...OnPrimary (which forces the creation), even if we
-      # are talking about the secondary node
-      for new_lv in new_lvs:
-        if not _CreateBlockDevOnPrimary(self, tgt_node, instance, new_lv,
-                                        _GetInstanceInfoText(instance)):
-          raise errors.OpExecError("Failed to create new LV named '%s' on"
-                                   " node '%s'" %
-                                   (new_lv.logical_id[1], tgt_node))
+    self.lu.LogStep(3, steps_total, "Allocate new storage")
+    iv_names = self._CreateNewStorage(self.target_node)
  
      # Step: for each lv, detach+rename*2+attach
  
      # Step: for each lv, detach+rename*2+attach
-    self.proc.LogStep(4, steps_total, "change drbd configuration")
+    self.lu.LogStep(4, steps_total, "Changing drbd configuration")
      for dev, old_lvs, new_lvs in iv_names.itervalues():
      for dev, old_lvs, new_lvs in iv_names.itervalues():
-      info("detaching %s drbd from local storage" % dev.iv_name)
-      if not self.rpc.call_blockdev_removechildren(tgt_node, dev, old_lvs):
-        raise errors.OpExecError("Can't detach drbd from local storage on node"
-                                 " %s for device %s" % (tgt_node, dev.iv_name))
+      self.lu.LogInfo("Detaching %s drbd from local storage" % dev.iv_name)
+
+      result = self.rpc.call_blockdev_removechildren(self.target_node, dev, old_lvs)
+      result.Raise("Can't detach drbd from local storage on node"
+                   " %s for device %s" % (self.target_node, dev.iv_name))
        #dev.children = []
        #cfg.Update(instance)
  
        #dev.children = []
        #cfg.Update(instance)
  
@@ -4141,69 +6259,66 @@ class LUReplaceDisks(LogicalUnit):
        temp_suffix = int(time.time())
        ren_fn = lambda d, suff: (d.physical_id[0],
                                  d.physical_id[1] + "_replaced-%s" % suff)
        temp_suffix = int(time.time())
        ren_fn = lambda d, suff: (d.physical_id[0],
                                  d.physical_id[1] + "_replaced-%s" % suff)
-      # build the rename list based on what LVs exist on the node
-      rlist = []
+
+      # Build the rename list based on what LVs exist on the node
+      rename_old_to_new = []
        for to_ren in old_lvs:
        for to_ren in old_lvs:
-        find_res = self.rpc.call_blockdev_find(tgt_node, to_ren)
-        if find_res is not None: # device exists
-          rlist.append((to_ren, ren_fn(to_ren, temp_suffix)))
-
-      info("renaming the old LVs on the target node")
-      if not self.rpc.call_blockdev_rename(tgt_node, rlist):
-        raise errors.OpExecError("Can't rename old LVs on node %s" % tgt_node)
-      # now we rename the new LVs to the old LVs
-      info("renaming the new LVs on the target node")
-      rlist = [(new, old.physical_id) for old, new in zip(old_lvs, new_lvs)]
-      if not self.rpc.call_blockdev_rename(tgt_node, rlist):
-        raise errors.OpExecError("Can't rename new LVs on node %s" % tgt_node)
+        result = self.rpc.call_blockdev_find(self.target_node, to_ren)
+        if not result.fail_msg and result.payload:
+          # device exists
+          rename_old_to_new.append((to_ren, ren_fn(to_ren, temp_suffix)))
+
+      self.lu.LogInfo("Renaming the old LVs on the target node")
+      result = self.rpc.call_blockdev_rename(self.target_node, rename_old_to_new)
+      result.Raise("Can't rename old LVs on node %s" % self.target_node)
+
+      # Now we rename the new LVs to the old LVs
+      self.lu.LogInfo("Renaming the new LVs on the target node")
+      rename_new_to_old = [(new, old.physical_id)
+                           for old, new in zip(old_lvs, new_lvs)]
+      result = self.rpc.call_blockdev_rename(self.target_node, rename_new_to_old)
+      result.Raise("Can't rename new LVs on node %s" % self.target_node)
  
        for old, new in zip(old_lvs, new_lvs):
          new.logical_id = old.logical_id
  
        for old, new in zip(old_lvs, new_lvs):
          new.logical_id = old.logical_id
-        cfg.SetDiskID(new, tgt_node)
+        self.cfg.SetDiskID(new, self.target_node)
  
        for disk in old_lvs:
          disk.logical_id = ren_fn(disk, temp_suffix)
  
        for disk in old_lvs:
          disk.logical_id = ren_fn(disk, temp_suffix)
-        cfg.SetDiskID(disk, tgt_node)
+        self.cfg.SetDiskID(disk, self.target_node)
  
  
-      # now that the new lvs have the old name, we can add them to the device
-      info("adding new mirror component on %s" % tgt_node)
-      if not self.rpc.call_blockdev_addchildren(tgt_node, dev, new_lvs):
+      # Now that the new lvs have the old name, we can add them to the device
+      self.lu.LogInfo("Adding new mirror component on %s" % self.target_node)
+      result = self.rpc.call_blockdev_addchildren(self.target_node, dev, new_lvs)
+      msg = result.fail_msg
+      if msg:
          for new_lv in new_lvs:
          for new_lv in new_lvs:
-          if not self.rpc.call_blockdev_remove(tgt_node, new_lv):
-            warning("Can't rollback device %s", hint="manually cleanup unused"
-                    " logical volumes")
-        raise errors.OpExecError("Can't add local storage to drbd")
+          msg2 = self.rpc.call_blockdev_remove(self.target_node, new_lv).fail_msg
+          if msg2:
+            self.lu.LogWarning("Can't rollback device %s: %s", dev, msg2,
+                               hint=("cleanup manually the unused logical"
+                                     "volumes"))
+        raise errors.OpExecError("Can't add local storage to drbd: %s" % msg)
  
        dev.children = new_lvs
  
        dev.children = new_lvs
-      cfg.Update(instance)
  
  
-    # Step: wait for sync
+      self.cfg.Update(self.instance)
  
  
-    # this can fail as the old devices are degraded and _WaitForSync
-    # does a combined result over all disks, so we don't check its
-    # return value
-    self.proc.LogStep(5, steps_total, "sync devices")
-    _WaitForSync(self, instance, unlock=True)
+    # Wait for sync
+    # This can fail as the old devices are degraded and _WaitForSync
+    # does a combined result over all disks, so we don't check its return value
+    self.lu.LogStep(5, steps_total, "Sync devices")
+    _WaitForSync(self.lu, self.instance, unlock=True)
  
  
-    # so check manually all the devices
-    for name, (dev, old_lvs, new_lvs) in iv_names.iteritems():
-      cfg.SetDiskID(dev, instance.primary_node)
-      is_degr = self.rpc.call_blockdev_find(instance.primary_node, dev)[5]
-      if is_degr:
-        raise errors.OpExecError("DRBD device %s is degraded!" % name)
+    # Check all devices manually
+    self._CheckDevices(self.instance.primary_node, iv_names)
  
      # Step: remove old storage
  
      # Step: remove old storage
-    self.proc.LogStep(6, steps_total, "removing old storage")
-    for name, (dev, old_lvs, new_lvs) in iv_names.iteritems():
-      info("remove logical volumes for %s" % name)
-      for lv in old_lvs:
-        cfg.SetDiskID(lv, tgt_node)
-        if not self.rpc.call_blockdev_remove(tgt_node, lv):
-          warning("Can't remove old LV", hint="manually remove unused LVs")
-          continue
+    self.lu.LogStep(6, steps_total, "Removing old storage")
+    self._RemoveOldStorage(self.target_node, iv_names)
  
  
-  def _ExecD8Secondary(self, feedback_fn):
-    """Replace the secondary node for drbd8.
+  def _ExecDrbd8Secondary(self):
+    """Replace the secondary node for DRBD 8.
  
      The algorithm for replace is quite complicated:
        - for all disks of the instance:
  
      The algorithm for replace is quite complicated:
        - for all disks of the instance:
@@ -4222,198 +6337,176 @@ class LUReplaceDisks(LogicalUnit):
  
      """
      steps_total = 6
  
      """
      steps_total = 6
-    warning, info = (self.proc.LogWarning, self.proc.LogInfo)
-    instance = self.instance
-    iv_names = {}
-    vgname = self.cfg.GetVGName()
-    # start of work
-    cfg = self.cfg
-    old_node = self.tgt_node
-    new_node = self.new_node
-    pri_node = instance.primary_node
  
      # Step: check device activation
  
      # Step: check device activation
-    self.proc.LogStep(1, steps_total, "check device existence")
-    info("checking volume groups")
-    my_vg = cfg.GetVGName()
-    results = self.rpc.call_vg_list([pri_node, new_node])
-    if not results:
-      raise errors.OpExecError("Can't list volume groups on the nodes")
-    for node in pri_node, new_node:
-      res = results.get(node, False)
-      if not res or my_vg not in res:
-        raise errors.OpExecError("Volume group '%s' not found on %s" %
-                                 (my_vg, node))
-    for idx, dev in enumerate(instance.disks):
-      if idx not in self.op.disks:
-        continue
-      info("checking disk/%d on %s" % (idx, pri_node))
-      cfg.SetDiskID(dev, pri_node)
-      if not self.rpc.call_blockdev_find(pri_node, dev):
-        raise errors.OpExecError("Can't find disk/%d on node %s" %
-                                 (idx, pri_node))
+    self.lu.LogStep(1, steps_total, "Check device existence")
+    self._CheckDisksExistence([self.instance.primary_node])
+    self._CheckVolumeGroup([self.instance.primary_node])
  
      # Step: check other node consistency
  
      # Step: check other node consistency
-    self.proc.LogStep(2, steps_total, "check peer consistency")
-    for idx, dev in enumerate(instance.disks):
-      if idx not in self.op.disks:
-        continue
-      info("checking disk/%d consistency on %s" % (idx, pri_node))
-      if not _CheckDiskConsistency(self, dev, pri_node, True, ldisk=True):
-        raise errors.OpExecError("Primary node (%s) has degraded storage,"
-                                 " unsafe to replace the secondary" %
-                                 pri_node)
+    self.lu.LogStep(2, steps_total, "Check peer consistency")
+    self._CheckDisksConsistency(self.instance.primary_node, True, True)
  
      # Step: create new storage
  
      # Step: create new storage
-    self.proc.LogStep(3, steps_total, "allocate new storage")
-    for idx, dev in enumerate(instance.disks):
-      size = dev.size
-      info("adding new local storage on %s for disk/%d" %
-           (new_node, idx))
-      # since we *always* want to create this LV, we use the
-      # _Create...OnPrimary (which forces the creation), even if we
-      # are talking about the secondary node
+    self.lu.LogStep(3, steps_total, "Allocate new storage")
+    for idx, dev in enumerate(self.instance.disks):
+      self.lu.LogInfo("Adding new local storage on %s for disk/%d" %
+                      (self.new_node, idx))
+      # we pass force_create=True to force LVM creation
        for new_lv in dev.children:
        for new_lv in dev.children:
-        if not _CreateBlockDevOnPrimary(self, new_node, instance, new_lv,
-                                        _GetInstanceInfoText(instance)):
-          raise errors.OpExecError("Failed to create new LV named '%s' on"
-                                   " node '%s'" %
-                                   (new_lv.logical_id[1], new_node))
+        _CreateBlockDev(self.lu, self.new_node, self.instance, new_lv, True,
+                        _GetInstanceInfoText(self.instance), False)
  
      # Step 4: dbrd minors and drbd setups changes
      # after this, we must manually remove the drbd minors on both the
      # error and the success paths
  
      # Step 4: dbrd minors and drbd setups changes
      # after this, we must manually remove the drbd minors on both the
      # error and the success paths
-    minors = cfg.AllocateDRBDMinor([new_node for dev in instance.disks],
-                                   instance.name)
-    logging.debug("Allocated minors %s" % (minors,))
-    self.proc.LogStep(4, steps_total, "changing drbd configuration")
-    for idx, (dev, new_minor) in enumerate(zip(instance.disks, minors)):
-      size = dev.size
-      info("activating a new drbd on %s for disk/%d" % (new_node, idx))
-      # create new devices on new_node
-      if pri_node == dev.logical_id[0]:
-        new_logical_id = (pri_node, new_node,
-                          dev.logical_id[2], dev.logical_id[3], new_minor,
-                          dev.logical_id[5])
+    self.lu.LogStep(4, steps_total, "Changing drbd configuration")
+    minors = self.cfg.AllocateDRBDMinor([self.new_node for dev in self.instance.disks],
+                                        self.instance.name)
+    logging.debug("Allocated minors %r" % (minors,))
+
+    iv_names = {}
+    for idx, (dev, new_minor) in enumerate(zip(self.instance.disks, minors)):
+      self.lu.LogInfo("activating a new drbd on %s for disk/%d" % (self.new_node, idx))
+      # create new devices on new_node; note that we create two IDs:
+      # one without port, so the drbd will be activated without
+      # networking information on the new node at this stage, and one
+      # with network, for the latter activation in step 4
+      (o_node1, o_node2, o_port, o_minor1, o_minor2, o_secret) = dev.logical_id
+      if self.instance.primary_node == o_node1:
+        p_minor = o_minor1
        else:
        else:
-        new_logical_id = (new_node, pri_node,
-                          dev.logical_id[2], new_minor, dev.logical_id[4],
-                          dev.logical_id[5])
-      iv_names[idx] = (dev, dev.children, new_logical_id)
+        p_minor = o_minor2
+
+      new_alone_id = (self.instance.primary_node, self.new_node, None, p_minor, new_minor, o_secret)
+      new_net_id = (self.instance.primary_node, self.new_node, o_port, p_minor, new_minor, o_secret)
+
+      iv_names[idx] = (dev, dev.children, new_net_id)
        logging.debug("Allocated new_minor: %s, new_logical_id: %s", new_minor,
        logging.debug("Allocated new_minor: %s, new_logical_id: %s", new_minor,
-                    new_logical_id)
+                    new_net_id)
        new_drbd = objects.Disk(dev_type=constants.LD_DRBD8,
        new_drbd = objects.Disk(dev_type=constants.LD_DRBD8,
-                              logical_id=new_logical_id,
-                              children=dev.children)
-      if not _CreateBlockDevOnSecondary(self, new_node, instance,
-                                        new_drbd, False,
-                                        _GetInstanceInfoText(instance)):
-        self.cfg.ReleaseDRBDMinors(instance.name)
-        raise errors.OpExecError("Failed to create new DRBD on"
-                                 " node '%s'" % new_node)
-
-    for idx, dev in enumerate(instance.disks):
-      # we have new devices, shutdown the drbd on the old secondary
-      info("shutting down drbd for disk/%d on old node" % idx)
-      cfg.SetDiskID(dev, old_node)
-      if not self.rpc.call_blockdev_shutdown(old_node, dev):
-        warning("Failed to shutdown drbd for disk/%d on old node" % idx,
-                hint="Please cleanup this device manually as soon as possible")
-
-    info("detaching primary drbds from the network (=> standalone)")
-    done = 0
-    for idx, dev in enumerate(instance.disks):
-      cfg.SetDiskID(dev, pri_node)
-      # set the network part of the physical (unique in bdev terms) id
-      # to None, meaning detach from network
-      dev.physical_id = (None, None, None, None) + dev.physical_id[4:]
-      # and 'find' the device, which will 'fix' it to match the
-      # standalone state
-      if self.rpc.call_blockdev_find(pri_node, dev):
-        done += 1
-      else:
-        warning("Failed to detach drbd disk/%d from network, unusual case" %
-                idx)
-
-    if not done:
-      # no detaches succeeded (very unlikely)
-      self.cfg.ReleaseDRBDMinors(instance.name)
-      raise errors.OpExecError("Can't detach at least one DRBD from old node")
+                              logical_id=new_alone_id,
+                              children=dev.children,
+                              size=dev.size)
+      try:
+        _CreateSingleBlockDev(self.lu, self.new_node, self.instance, new_drbd,
+                              _GetInstanceInfoText(self.instance), False)
+      except errors.GenericError:
+        self.cfg.ReleaseDRBDMinors(self.instance.name)
+        raise
+
+    # We have new devices, shutdown the drbd on the old secondary
+    for idx, dev in enumerate(self.instance.disks):
+      self.lu.LogInfo("Shutting down drbd for disk/%d on old node" % idx)
+      self.cfg.SetDiskID(dev, self.target_node)
+      msg = self.rpc.call_blockdev_shutdown(self.target_node, dev).fail_msg
+      if msg:
+        self.lu.LogWarning("Failed to shutdown drbd for disk/%d on old"
+                           "node: %s" % (idx, msg),
+                           hint=("Please cleanup this device manually as"
+                                 " soon as possible"))
+
+    self.lu.LogInfo("Detaching primary drbds from the network (=> standalone)")
+    result = self.rpc.call_drbd_disconnect_net([self.instance.primary_node], self.node_secondary_ip,
+                                               self.instance.disks)[self.instance.primary_node]
+
+    msg = result.fail_msg
+    if msg:
+      # detaches didn't succeed (unlikely)
+      self.cfg.ReleaseDRBDMinors(self.instance.name)
+      raise errors.OpExecError("Can't detach the disks from the network on"
+                               " old node: %s" % (msg,))
  
      # if we managed to detach at least one, we update all the disks of
      # the instance to point to the new secondary
  
      # if we managed to detach at least one, we update all the disks of
      # the instance to point to the new secondary
-    info("updating instance configuration")
+    self.lu.LogInfo("Updating instance configuration")
      for dev, _, new_logical_id in iv_names.itervalues():
        dev.logical_id = new_logical_id
      for dev, _, new_logical_id in iv_names.itervalues():
        dev.logical_id = new_logical_id
-      cfg.SetDiskID(dev, pri_node)
-    cfg.Update(instance)
-    # we can remove now the temp minors as now the new values are
-    # written to the config file (and therefore stable)
-    self.cfg.ReleaseDRBDMinors(instance.name)
+      self.cfg.SetDiskID(dev, self.instance.primary_node)
+
+    self.cfg.Update(self.instance)
  
      # and now perform the drbd attach
  
      # and now perform the drbd attach
-    info("attaching primary drbds to new secondary (standalone => connected)")
-    failures = []
-    for idx, dev in enumerate(instance.disks):
-      info("attaching primary drbd for disk/%d to new secondary node" % idx)
-      # since the attach is smart, it's enough to 'find' the device,
-      # it will automatically activate the network, if the physical_id
-      # is correct
-      cfg.SetDiskID(dev, pri_node)
-      logging.debug("Disk to attach: %s", dev)
-      if not self.rpc.call_blockdev_find(pri_node, dev):
-        warning("can't attach drbd disk/%d to new secondary!" % idx,
-                "please do a gnt-instance info to see the status of disks")
-
-    # this can fail as the old devices are degraded and _WaitForSync
-    # does a combined result over all disks, so we don't check its
-    # return value
-    self.proc.LogStep(5, steps_total, "sync devices")
-    _WaitForSync(self, instance, unlock=True)
-
-    # so check manually all the devices
-    for idx, (dev, old_lvs, _) in iv_names.iteritems():
-      cfg.SetDiskID(dev, pri_node)
-      is_degr = self.rpc.call_blockdev_find(pri_node, dev)[5]
-      if is_degr:
-        raise errors.OpExecError("DRBD device disk/%d is degraded!" % idx)
-
-    self.proc.LogStep(6, steps_total, "removing old storage")
-    for idx, (dev, old_lvs, _) in iv_names.iteritems():
-      info("remove logical volumes for disk/%d" % idx)
-      for lv in old_lvs:
-        cfg.SetDiskID(lv, old_node)
-        if not self.rpc.call_blockdev_remove(old_node, lv):
-          warning("Can't remove LV on old secondary",
-                  hint="Cleanup stale volumes by hand")
+    self.lu.LogInfo("Attaching primary drbds to new secondary"
+                    " (standalone => connected)")
+    result = self.rpc.call_drbd_attach_net([self.instance.primary_node, self.new_node], self.node_secondary_ip,
+                                           self.instance.disks, self.instance.name,
+                                           False)
+    for to_node, to_result in result.items():
+      msg = to_result.fail_msg
+      if msg:
+        self.lu.LogWarning("Can't attach drbd disks on node %s: %s", to_node, msg,
+                           hint=("please do a gnt-instance info to see the"
+                                 " status of disks"))
+
+    # Wait for sync
+    # This can fail as the old devices are degraded and _WaitForSync
+    # does a combined result over all disks, so we don't check its return value
+    self.lu.LogStep(5, steps_total, "Sync devices")
+    _WaitForSync(self.lu, self.instance, unlock=True)
+
+    # Check all devices manually
+    self._CheckDevices(self.instance.primary_node, iv_names)
  
  
-  def Exec(self, feedback_fn):
-    """Execute disk replacement.
+    # Step: remove old storage
+    self.lu.LogStep(6, steps_total, "Removing old storage")
+    self._RemoveOldStorage(self.target_node, iv_names)
  
  
-    This dispatches the disk replacement to the appropriate handler.
  
  
-    """
-    instance = self.instance
+class LURepairNodeStorage(NoHooksLU):
+  """Repairs the volume group on a node.
  
  
-    # Activate the instance disks if we're replacing them on a down instance
-    if instance.status == "down":
-      _StartInstanceDisks(self, instance, True)
+  """
+  _OP_REQP = ["node_name"]
+  REQ_BGL = False
  
  
-    if instance.disk_template == constants.DT_DRBD8:
-      if self.op.remote_node is None:
-        fn = self._ExecD8DiskOnly
-      else:
-        fn = self._ExecD8Secondary
-    else:
-      raise errors.ProgrammerError("Unhandled disk replacement case")
+  def CheckArguments(self):
+    node_name = self.cfg.ExpandNodeName(self.op.node_name)
+    if node_name is None:
+      raise errors.OpPrereqError("Invalid node name '%s'" % self.op.node_name)
+
+    self.op.node_name = node_name
+
+  def ExpandNames(self):
+    self.needed_locks = {
+      locking.LEVEL_NODE: [self.op.node_name],
+      }
+
+  def _CheckFaultyDisks(self, instance, node_name):
+    if _FindFaultyInstanceDisks(self.cfg, self.rpc, instance,
+                                node_name, True):
+      raise errors.OpPrereqError("Instance '%s' has faulty disks on"
+                                 " node '%s'" % (inst.name, node_name))
+
+  def CheckPrereq(self):
+    """Check prerequisites.
+
+    """
+    storage_type = self.op.storage_type
+
+    if (constants.SO_FIX_CONSISTENCY not in
+        constants.VALID_STORAGE_OPERATIONS.get(storage_type, [])):
+      raise errors.OpPrereqError("Storage units of type '%s' can not be"
+                                 " repaired" % storage_type)
  
  
-    ret = fn(feedback_fn)
+    # Check whether any instance on this node has faulty disks
+    for inst in _GetNodeInstances(self.cfg, self.op.node_name):
+      check_nodes = set(inst.all_nodes)
+      check_nodes.discard(self.op.node_name)
+      for inst_node_name in check_nodes:
+        self._CheckFaultyDisks(inst, inst_node_name)
  
  
-    # Deactivate the instance disks if we're replacing them on a down instance
-    if instance.status == "down":
-      _SafeShutdownInstanceDisks(self, instance)
+  def Exec(self, feedback_fn):
+    feedback_fn("Repairing storage unit '%s' on %s ..." %
+                (self.op.name, self.op.node_name))
  
  
-    return ret
+    st_args = _GetStorageTypeArgs(self.cfg, self.op.storage_type)
+    result = self.rpc.call_storage_execute(self.op.node_name,
+                                           self.op.storage_type, st_args,
+                                           self.op.name,
+                                           constants.SO_FIX_CONSISTENCY)
+    result.Raise("Failed to repair storage unit '%s' on %s" %
+                 (self.op.name, self.op.node_name))
  
  
  class LUGrowDisk(LogicalUnit):
  
  
  class LUGrowDisk(LogicalUnit):
@@ -4460,6 +6553,10 @@ class LUGrowDisk(LogicalUnit):
      instance = self.cfg.GetInstanceInfo(self.op.instance_name)
      assert instance is not None, \
        "Cannot retrieve locked instance %s" % self.op.instance_name
      instance = self.cfg.GetInstanceInfo(self.op.instance_name)
      assert instance is not None, \
        "Cannot retrieve locked instance %s" % self.op.instance_name
+    nodenames = list(instance.all_nodes)
+    for node in nodenames:
+      _CheckNodeOnline(self, node)
+
  
      self.instance = instance
  
  
      self.instance = instance
  
@@ -4469,22 +6566,19 @@ class LUGrowDisk(LogicalUnit):
  
      self.disk = instance.FindDisk(self.op.disk)
  
  
      self.disk = instance.FindDisk(self.op.disk)
  
-    nodenames = [instance.primary_node] + list(instance.secondary_nodes)
      nodeinfo = self.rpc.call_node_info(nodenames, self.cfg.GetVGName(),
                                         instance.hypervisor)
      for node in nodenames:
      nodeinfo = self.rpc.call_node_info(nodenames, self.cfg.GetVGName(),
                                         instance.hypervisor)
      for node in nodenames:
-      info = nodeinfo.get(node, None)
-      if not info:
-        raise errors.OpPrereqError("Cannot get current information"
-                                   " from node '%s'" % node)
-      vg_free = info.get('vg_free', None)
+      info = nodeinfo[node]
+      info.Raise("Cannot get current information from node %s" % node)
+      vg_free = info.payload.get('vg_free', None)
        if not isinstance(vg_free, int):
          raise errors.OpPrereqError("Can't compute free disk space on"
                                     " node %s" % node)
        if not isinstance(vg_free, int):
          raise errors.OpPrereqError("Can't compute free disk space on"
                                     " node %s" % node)
-      if self.op.amount > info['vg_free']:
+      if self.op.amount > vg_free:
          raise errors.OpPrereqError("Not enough disk space on target node %s:"
                                     " %d MiB available, %d MiB required" %
          raise errors.OpPrereqError("Not enough disk space on target node %s:"
                                     " %d MiB available, %d MiB required" %
-                                   (node, info['vg_free'], self.op.amount))
+                                   (node, vg_free, self.op.amount))
  
    def Exec(self, feedback_fn):
      """Execute disk grow.
  
    def Exec(self, feedback_fn):
      """Execute disk grow.
@@ -4492,15 +6586,10 @@ class LUGrowDisk(LogicalUnit):
      """
      instance = self.instance
      disk = self.disk
      """
      instance = self.instance
      disk = self.disk
-    for node in (instance.secondary_nodes + (instance.primary_node,)):
+    for node in instance.all_nodes:
        self.cfg.SetDiskID(disk, node)
        result = self.rpc.call_blockdev_grow(node, disk, self.op.amount)
        self.cfg.SetDiskID(disk, node)
        result = self.rpc.call_blockdev_grow(node, disk, self.op.amount)
-      if (not result or not isinstance(result, (list, tuple)) or
-          len(result) != 2):
-        raise errors.OpExecError("grow request failed to node %s" % node)
-      elif not result[0]:
-        raise errors.OpExecError("grow request failed to node %s: %s" %
-                                 (node, result[1]))
+      result.Raise("Grow request failed to node %s" % node)
      disk.RecordGrow(self.op.amount)
      self.cfg.Update(instance)
      if self.op.wait_for_sync:
      disk.RecordGrow(self.op.amount)
      self.cfg.Update(instance)
      if self.op.wait_for_sync:
@@ -4519,7 +6608,7 @@ class LUQueryInstanceData(NoHooksLU):
  
    def ExpandNames(self):
      self.needed_locks = {}
  
    def ExpandNames(self):
      self.needed_locks = {}
-    self.share_locks = dict(((i, 1) for i in locking.LEVELS))
+    self.share_locks = dict.fromkeys(locking.LEVELS, 1)
  
      if not isinstance(self.op.instances, list):
        raise errors.OpPrereqError("Invalid argument type 'instances'")
  
      if not isinstance(self.op.instances, list):
        raise errors.OpPrereqError("Invalid argument type 'instances'")
@@ -4529,8 +6618,7 @@ class LUQueryInstanceData(NoHooksLU):
        for name in self.op.instances:
          full_name = self.cfg.ExpandInstanceName(name)
          if full_name is None:
        for name in self.op.instances:
          full_name = self.cfg.ExpandInstanceName(name)
          if full_name is None:
-          raise errors.OpPrereqError("Instance '%s' not known" %
-                                     self.op.instance_name)
+          raise errors.OpPrereqError("Instance '%s' not known" % name)
          self.wanted_names.append(full_name)
        self.needed_locks[locking.LEVEL_INSTANCE] = self.wanted_names
      else:
          self.wanted_names.append(full_name)
        self.needed_locks[locking.LEVEL_INSTANCE] = self.wanted_names
      else:
@@ -4557,17 +6645,33 @@ class LUQueryInstanceData(NoHooksLU):
                               in self.wanted_names]
      return
  
                               in self.wanted_names]
      return
  
+  def _ComputeBlockdevStatus(self, node, instance_name, dev):
+    """Returns the status of a block device
+
+    """
+    if self.op.static or not node:
+      return None
+
+    self.cfg.SetDiskID(dev, node)
+
+    result = self.rpc.call_blockdev_find(node, dev)
+    if result.offline:
+      return None
+
+    result.Raise("Can't compute disk status for %s" % instance_name)
+
+    status = result.payload
+    if status is None:
+      return None
+
+    return (status.dev_path, status.major, status.minor,
+            status.sync_percent, status.estimated_time,
+            status.is_degraded, status.ldisk_status)
+
    def _ComputeDiskStatus(self, instance, snode, dev):
      """Compute block device status.
  
      """
    def _ComputeDiskStatus(self, instance, snode, dev):
      """Compute block device status.
  
      """
-    static = self.op.static
-    if not static:
-      self.cfg.SetDiskID(dev, instance.primary_node)
-      dev_pstatus = self.rpc.call_blockdev_find(instance.primary_node, dev)
-    else:
-      dev_pstatus = None
-
      if dev.dev_type in constants.LDS_DRBD:
        # we change the snode then (otherwise we use the one passed in)
        if dev.logical_id[0] == instance.primary_node:
      if dev.dev_type in constants.LDS_DRBD:
        # we change the snode then (otherwise we use the one passed in)
        if dev.logical_id[0] == instance.primary_node:
@@ -4575,11 +6679,9 @@ class LUQueryInstanceData(NoHooksLU):
        else:
          snode = dev.logical_id[0]
  
        else:
          snode = dev.logical_id[0]
  
-    if snode and not static:
-      self.cfg.SetDiskID(dev, snode)
-      dev_sstatus = self.rpc.call_blockdev_find(snode, dev)
-    else:
-      dev_sstatus = None
+    dev_pstatus = self._ComputeBlockdevStatus(instance.primary_node,
+                                              instance.name, dev)
+    dev_sstatus = self._ComputeBlockdevStatus(snode, instance.name, dev)
  
      if dev.children:
        dev_children = [self._ComputeDiskStatus(instance, snode, child)
  
      if dev.children:
        dev_children = [self._ComputeDiskStatus(instance, snode, child)
@@ -4595,6 +6697,8 @@ class LUQueryInstanceData(NoHooksLU):
        "pstatus": dev_pstatus,
        "sstatus": dev_sstatus,
        "children": dev_children,
        "pstatus": dev_pstatus,
        "sstatus": dev_sstatus,
        "children": dev_children,
+      "mode": dev.mode,
+      "size": dev.size,
        }
  
      return data
        }
  
      return data
@@ -4610,16 +6714,18 @@ class LUQueryInstanceData(NoHooksLU):
          remote_info = self.rpc.call_instance_info(instance.primary_node,
                                                    instance.name,
                                                    instance.hypervisor)
          remote_info = self.rpc.call_instance_info(instance.primary_node,
                                                    instance.name,
                                                    instance.hypervisor)
+        remote_info.Raise("Error checking node %s" % instance.primary_node)
+        remote_info = remote_info.payload
          if remote_info and "state" in remote_info:
            remote_state = "up"
          else:
            remote_state = "down"
        else:
          remote_state = None
          if remote_info and "state" in remote_info:
            remote_state = "up"
          else:
            remote_state = "down"
        else:
          remote_state = None
-      if instance.status == "down":
-        config_state = "down"
-      else:
+      if instance.admin_up:
          config_state = "up"
          config_state = "up"
+      else:
+        config_state = "down"
  
        disks = [self._ComputeDiskStatus(instance, None, device)
                 for device in instance.disks]
  
        disks = [self._ComputeDiskStatus(instance, None, device)
                 for device in instance.disks]
@@ -4631,7 +6737,8 @@ class LUQueryInstanceData(NoHooksLU):
          "pnode": instance.primary_node,
          "snodes": instance.secondary_nodes,
          "os": instance.os,
          "pnode": instance.primary_node,
          "snodes": instance.secondary_nodes,
          "os": instance.os,
-        "nics": [(nic.mac, nic.ip, nic.bridge) for nic in instance.nics],
+        # this happens to be the same format used for hooks
+        "nics": _NICListToTuple(self, instance.nics),
          "disks": disks,
          "hypervisor": instance.hypervisor,
          "network_port": instance.network_port,
          "disks": disks,
          "hypervisor": instance.hypervisor,
          "network_port": instance.network_port,
@@ -4639,6 +6746,9 @@ class LUQueryInstanceData(NoHooksLU):
          "hv_actual": cluster.FillHV(instance),
          "be_instance": instance.beparams,
          "be_actual": cluster.FillBE(instance),
          "hv_actual": cluster.FillHV(instance),
          "be_instance": instance.beparams,
          "be_actual": cluster.FillBE(instance),
+        "serial_no": instance.serial_no,
+        "mtime": instance.mtime,
+        "ctime": instance.ctime,
          }
  
        result[instance.name] = idict
          }
  
        result[instance.name] = idict
@@ -4652,15 +6762,118 @@ class LUSetInstanceParams(LogicalUnit):
    """
    HPATH = "instance-modify"
    HTYPE = constants.HTYPE_INSTANCE
    """
    HPATH = "instance-modify"
    HTYPE = constants.HTYPE_INSTANCE
-  _OP_REQP = ["instance_name", "hvparams"]
+  _OP_REQP = ["instance_name"]
    REQ_BGL = False
  
    REQ_BGL = False
  
+  def CheckArguments(self):
+    if not hasattr(self.op, 'nics'):
+      self.op.nics = []
+    if not hasattr(self.op, 'disks'):
+      self.op.disks = []
+    if not hasattr(self.op, 'beparams'):
+      self.op.beparams = {}
+    if not hasattr(self.op, 'hvparams'):
+      self.op.hvparams = {}
+    self.op.force = getattr(self.op, "force", False)
+    if not (self.op.nics or self.op.disks or
+            self.op.hvparams or self.op.beparams):
+      raise errors.OpPrereqError("No changes submitted")
+
+    # Disk validation
+    disk_addremove = 0
+    for disk_op, disk_dict in self.op.disks:
+      if disk_op == constants.DDM_REMOVE:
+        disk_addremove += 1
+        continue
+      elif disk_op == constants.DDM_ADD:
+        disk_addremove += 1
+      else:
+        if not isinstance(disk_op, int):
+          raise errors.OpPrereqError("Invalid disk index")
+        if not isinstance(disk_dict, dict):
+          msg = "Invalid disk value: expected dict, got '%s'" % disk_dict
+          raise errors.OpPrereqError(msg)
+
+      if disk_op == constants.DDM_ADD:
+        mode = disk_dict.setdefault('mode', constants.DISK_RDWR)
+        if mode not in constants.DISK_ACCESS_SET:
+          raise errors.OpPrereqError("Invalid disk access mode '%s'" % mode)
+        size = disk_dict.get('size', None)
+        if size is None:
+          raise errors.OpPrereqError("Required disk parameter size missing")
+        try:
+          size = int(size)
+        except ValueError, err:
+          raise errors.OpPrereqError("Invalid disk size parameter: %s" %
+                                     str(err))
+        disk_dict['size'] = size
+      else:
+        # modification of disk
+        if 'size' in disk_dict:
+          raise errors.OpPrereqError("Disk size change not possible, use"
+                                     " grow-disk")
+
+    if disk_addremove > 1:
+      raise errors.OpPrereqError("Only one disk add or remove operation"
+                                 " supported at a time")
+
+    # NIC validation
+    nic_addremove = 0
+    for nic_op, nic_dict in self.op.nics:
+      if nic_op == constants.DDM_REMOVE:
+        nic_addremove += 1
+        continue
+      elif nic_op == constants.DDM_ADD:
+        nic_addremove += 1
+      else:
+        if not isinstance(nic_op, int):
+          raise errors.OpPrereqError("Invalid nic index")
+        if not isinstance(nic_dict, dict):
+          msg = "Invalid nic value: expected dict, got '%s'" % nic_dict
+          raise errors.OpPrereqError(msg)
+
+      # nic_dict should be a dict
+      nic_ip = nic_dict.get('ip', None)
+      if nic_ip is not None:
+        if nic_ip.lower() == constants.VALUE_NONE:
+          nic_dict['ip'] = None
+        else:
+          if not utils.IsValidIP(nic_ip):
+            raise errors.OpPrereqError("Invalid IP address '%s'" % nic_ip)
+
+      nic_bridge = nic_dict.get('bridge', None)
+      nic_link = nic_dict.get('link', None)
+      if nic_bridge and nic_link:
+        raise errors.OpPrereqError("Cannot pass 'bridge' and 'link'"
+                                   " at the same time")
+      elif nic_bridge and nic_bridge.lower() == constants.VALUE_NONE:
+        nic_dict['bridge'] = None
+      elif nic_link and nic_link.lower() == constants.VALUE_NONE:
+        nic_dict['link'] = None
+
+      if nic_op == constants.DDM_ADD:
+        nic_mac = nic_dict.get('mac', None)
+        if nic_mac is None:
+          nic_dict['mac'] = constants.VALUE_AUTO
+
+      if 'mac' in nic_dict:
+        nic_mac = nic_dict['mac']
+        if nic_mac not in (constants.VALUE_AUTO, constants.VALUE_GENERATE):
+          if not utils.IsValidMac(nic_mac):
+            raise errors.OpPrereqError("Invalid MAC address %s" % nic_mac)
+        if nic_op != constants.DDM_ADD and nic_mac == constants.VALUE_AUTO:
+          raise errors.OpPrereqError("'auto' is not a valid MAC address when"
+                                     " modifying an existing nic")
+
+    if nic_addremove > 1:
+      raise errors.OpPrereqError("Only one NIC add or remove operation"
+                                 " supported at a time")
+
    def ExpandNames(self):
      self._ExpandAndLockInstance()
      self.needed_locks[locking.LEVEL_NODE] = []
      self.recalculate_locks[locking.LEVEL_NODE] = constants.LOCKS_REPLACE
  
    def ExpandNames(self):
      self._ExpandAndLockInstance()
      self.needed_locks[locking.LEVEL_NODE] = []
      self.recalculate_locks[locking.LEVEL_NODE] = constants.LOCKS_REPLACE
  
-
    def DeclareLocks(self, level):
      if level == locking.LEVEL_NODE:
        self._LockInstancesNodes()
    def DeclareLocks(self, level):
      if level == locking.LEVEL_NODE:
        self._LockInstancesNodes()
@@ -4676,93 +6889,101 @@ class LUSetInstanceParams(LogicalUnit):
        args['memory'] = self.be_new[constants.BE_MEMORY]
      if constants.BE_VCPUS in self.be_new:
        args['vcpus'] = self.be_new[constants.BE_VCPUS]
        args['memory'] = self.be_new[constants.BE_MEMORY]
      if constants.BE_VCPUS in self.be_new:
        args['vcpus'] = self.be_new[constants.BE_VCPUS]
-    if self.do_ip or self.do_bridge or self.mac:
-      if self.do_ip:
-        ip = self.ip
-      else:
-        ip = self.instance.nics[0].ip
-      if self.bridge:
-        bridge = self.bridge
-      else:
-        bridge = self.instance.nics[0].bridge
-      if self.mac:
-        mac = self.mac
-      else:
-        mac = self.instance.nics[0].mac
-      args['nics'] = [(ip, bridge, mac)]
+    # TODO: export disk changes. Note: _BuildInstanceHookEnv* don't export disk
+    # information at all.
+    if self.op.nics:
+      args['nics'] = []
+      nic_override = dict(self.op.nics)
+      c_nicparams = self.cluster.nicparams[constants.PP_DEFAULT]
+      for idx, nic in enumerate(self.instance.nics):
+        if idx in nic_override:
+          this_nic_override = nic_override[idx]
+        else:
+          this_nic_override = {}
+        if 'ip' in this_nic_override:
+          ip = this_nic_override['ip']
+        else:
+          ip = nic.ip
+        if 'mac' in this_nic_override:
+          mac = this_nic_override['mac']
+        else:
+          mac = nic.mac
+        if idx in self.nic_pnew:
+          nicparams = self.nic_pnew[idx]
+        else:
+          nicparams = objects.FillDict(c_nicparams, nic.nicparams)
+        mode = nicparams[constants.NIC_MODE]
+        link = nicparams[constants.NIC_LINK]
+        args['nics'].append((ip, mac, mode, link))
+      if constants.DDM_ADD in nic_override:
+        ip = nic_override[constants.DDM_ADD].get('ip', None)
+        mac = nic_override[constants.DDM_ADD]['mac']
+        nicparams = self.nic_pnew[constants.DDM_ADD]
+        mode = nicparams[constants.NIC_MODE]
+        link = nicparams[constants.NIC_LINK]
+        args['nics'].append((ip, mac, mode, link))
+      elif constants.DDM_REMOVE in nic_override:
+        del args['nics'][-1]
+
      env = _BuildInstanceHookEnvByObject(self, self.instance, override=args)
      env = _BuildInstanceHookEnvByObject(self, self.instance, override=args)
-    nl = [self.cfg.GetMasterNode(),
-          self.instance.primary_node] + list(self.instance.secondary_nodes)
+    nl = [self.cfg.GetMasterNode()] + list(self.instance.all_nodes)
      return env, nl, nl
  
      return env, nl, nl
  
+  def _GetUpdatedParams(self, old_params, update_dict,
+                        default_values, parameter_types):
+    """Return the new params dict for the given params.
+
+    @type old_params: dict
+    @param old_params: old parameters
+    @type update_dict: dict
+    @param update_dict: dict containing new parameter values,
+                        or constants.VALUE_DEFAULT to reset the
+                        parameter to its default value
+    @type default_values: dict
+    @param default_values: default values for the filled parameters
+    @type parameter_types: dict
+    @param parameter_types: dict mapping target dict keys to types
+                            in constants.ENFORCEABLE_TYPES
+    @rtype: (dict, dict)
+    @return: (new_parameters, filled_parameters)
+
+    """
+    params_copy = copy.deepcopy(old_params)
+    for key, val in update_dict.iteritems():
+      if val == constants.VALUE_DEFAULT:
+        try:
+          del params_copy[key]
+        except KeyError:
+          pass
+      else:
+        params_copy[key] = val
+    utils.ForceDictType(params_copy, parameter_types)
+    params_filled = objects.FillDict(default_values, params_copy)
+    return (params_copy, params_filled)
+
    def CheckPrereq(self):
      """Check prerequisites.
  
      This only checks the instance list against the existing names.
  
      """
    def CheckPrereq(self):
      """Check prerequisites.
  
      This only checks the instance list against the existing names.
  
      """
-    # FIXME: all the parameters could be checked before, in ExpandNames, or in
-    # a separate CheckArguments function, if we implement one, so the operation
-    # can be aborted without waiting for any lock, should it have an error...
-    self.ip = getattr(self.op, "ip", None)
-    self.mac = getattr(self.op, "mac", None)
-    self.bridge = getattr(self.op, "bridge", None)
-    self.kernel_path = getattr(self.op, "kernel_path", None)
-    self.initrd_path = getattr(self.op, "initrd_path", None)
-    self.force = getattr(self.op, "force", None)
-    all_parms = [self.ip, self.bridge, self.mac]
-    if (all_parms.count(None) == len(all_parms) and
-        not self.op.hvparams and
-        not self.op.beparams):
-      raise errors.OpPrereqError("No changes submitted")
-    for item in (constants.BE_MEMORY, constants.BE_VCPUS):
-      val = self.op.beparams.get(item, None)
-      if val is not None:
-        try:
-          val = int(val)
-        except ValueError, err:
-          raise errors.OpPrereqError("Invalid %s size: %s" % (item, str(err)))
-        self.op.beparams[item] = val
-    if self.ip is not None:
-      self.do_ip = True
-      if self.ip.lower() == "none":
-        self.ip = None
-      else:
-        if not utils.IsValidIP(self.ip):
-          raise errors.OpPrereqError("Invalid IP address '%s'." % self.ip)
-    else:
-      self.do_ip = False
-    self.do_bridge = (self.bridge is not None)
-    if self.mac is not None:
-      if self.cfg.IsMacInUse(self.mac):
-        raise errors.OpPrereqError('MAC address %s already in use in cluster' %
-                                   self.mac)
-      if not utils.IsValidMac(self.mac):
-        raise errors.OpPrereqError('Invalid MAC address %s' % self.mac)
+    self.force = self.op.force
  
      # checking the new params on the primary/secondary nodes
  
      instance = self.instance = self.cfg.GetInstanceInfo(self.op.instance_name)
  
      # checking the new params on the primary/secondary nodes
  
      instance = self.instance = self.cfg.GetInstanceInfo(self.op.instance_name)
+    cluster = self.cluster = self.cfg.GetClusterInfo()
      assert self.instance is not None, \
        "Cannot retrieve locked instance %s" % self.op.instance_name
      assert self.instance is not None, \
        "Cannot retrieve locked instance %s" % self.op.instance_name
-    pnode = self.instance.primary_node
-    nodelist = [pnode]
-    nodelist.extend(instance.secondary_nodes)
+    pnode = instance.primary_node
+    nodelist = list(instance.all_nodes)
  
      # hvparams processing
      if self.op.hvparams:
  
      # hvparams processing
      if self.op.hvparams:
-      i_hvdict = copy.deepcopy(instance.hvparams)
-      for key, val in self.op.hvparams.iteritems():
-        if val is None:
-          try:
-            del i_hvdict[key]
-          except KeyError:
-            pass
-        else:
-          i_hvdict[key] = val
-      cluster = self.cfg.GetClusterInfo()
-      hv_new = cluster.FillDict(cluster.hvparams[instance.hypervisor],
-                                i_hvdict)
+      i_hvdict, hv_new = self._GetUpdatedParams(
+                             instance.hvparams, self.op.hvparams,
+                             cluster.hvparams[instance.hypervisor],
+                             constants.HVS_PARAMETER_TYPES)
        # local check
        hypervisor.GetHypervisor(
          instance.hypervisor).CheckParameterSyntax(hv_new)
        # local check
        hypervisor.GetHypervisor(
          instance.hypervisor).CheckParameterSyntax(hv_new)
@@ -4774,22 +6995,14 @@ class LUSetInstanceParams(LogicalUnit):
  
      # beparams processing
      if self.op.beparams:
  
      # beparams processing
      if self.op.beparams:
-      i_bedict = copy.deepcopy(instance.beparams)
-      for key, val in self.op.beparams.iteritems():
-        if val is None:
-          try:
-            del i_bedict[key]
-          except KeyError:
-            pass
-        else:
-          i_bedict[key] = val
-      cluster = self.cfg.GetClusterInfo()
-      be_new = cluster.FillDict(cluster.beparams[constants.BEGR_DEFAULT],
-                                i_bedict)
+      i_bedict, be_new = self._GetUpdatedParams(
+                             instance.beparams, self.op.beparams,
+                             cluster.beparams[constants.PP_DEFAULT],
+                             constants.BES_PARAMETER_TYPES)
        self.be_new = be_new # the new actual values
        self.be_inst = i_bedict # the new dict (without defaults)
      else:
        self.be_new = be_new # the new actual values
        self.be_inst = i_bedict # the new dict (without defaults)
      else:
-      self.hv_new = self.hv_inst = {}
+      self.be_new = self.be_inst = {}
  
      self.warn = []
  
  
      self.warn = []
  
@@ -4802,39 +7015,151 @@ class LUSetInstanceParams(LogicalUnit):
                                                    instance.hypervisor)
        nodeinfo = self.rpc.call_node_info(mem_check_list, self.cfg.GetVGName(),
                                           instance.hypervisor)
                                                    instance.hypervisor)
        nodeinfo = self.rpc.call_node_info(mem_check_list, self.cfg.GetVGName(),
                                           instance.hypervisor)
-
-      if pnode not in nodeinfo or not isinstance(nodeinfo[pnode], dict):
+      pninfo = nodeinfo[pnode]
+      msg = pninfo.fail_msg
+      if msg:
          # Assume the primary node is unreachable and go ahead
          # Assume the primary node is unreachable and go ahead
-        self.warn.append("Can't get info from primary node %s" % pnode)
+        self.warn.append("Can't get info from primary node %s: %s" %
+                         (pnode,  msg))
+      elif not isinstance(pninfo.payload.get('memory_free', None), int):
+        self.warn.append("Node data from primary node %s doesn't contain"
+                         " free memory information" % pnode)
+      elif instance_info.fail_msg:
+        self.warn.append("Can't get instance runtime information: %s" %
+                        instance_info.fail_msg)
        else:
        else:
-        if instance_info:
-          current_mem = instance_info['memory']
+        if instance_info.payload:
+          current_mem = int(instance_info.payload['memory'])
          else:
            # Assume instance not running
            # (there is a slight race condition here, but it's not very probable,
            # and we have no other way to check)
            current_mem = 0
          miss_mem = (be_new[constants.BE_MEMORY] - current_mem -
          else:
            # Assume instance not running
            # (there is a slight race condition here, but it's not very probable,
            # and we have no other way to check)
            current_mem = 0
          miss_mem = (be_new[constants.BE_MEMORY] - current_mem -
-                    nodeinfo[pnode]['memory_free'])
+                    pninfo.payload['memory_free'])
          if miss_mem > 0:
            raise errors.OpPrereqError("This change will prevent the instance"
                                       " from starting, due to %d MB of memory"
                                       " missing on its primary node" % miss_mem)
  
        if be_new[constants.BE_AUTO_BALANCE]:
          if miss_mem > 0:
            raise errors.OpPrereqError("This change will prevent the instance"
                                       " from starting, due to %d MB of memory"
                                       " missing on its primary node" % miss_mem)
  
        if be_new[constants.BE_AUTO_BALANCE]:
-        for node in instance.secondary_nodes:
-          if node not in nodeinfo or not isinstance(nodeinfo[node], dict):
-            self.warn.append("Can't get info from secondary node %s" % node)
-          elif be_new[constants.BE_MEMORY] > nodeinfo[node]['memory_free']:
+        for node, nres in nodeinfo.items():
+          if node not in instance.secondary_nodes:
+            continue
+          msg = nres.fail_msg
+          if msg:
+            self.warn.append("Can't get info from secondary node %s: %s" %
+                             (node, msg))
+          elif not isinstance(nres.payload.get('memory_free', None), int):
+            self.warn.append("Secondary node %s didn't return free"
+                             " memory information" % node)
+          elif be_new[constants.BE_MEMORY] > nres.payload['memory_free']:
              self.warn.append("Not enough memory to failover instance to"
                               " secondary node %s" % node)
  
              self.warn.append("Not enough memory to failover instance to"
                               " secondary node %s" % node)
  
+    # NIC processing
+    self.nic_pnew = {}
+    self.nic_pinst = {}
+    for nic_op, nic_dict in self.op.nics:
+      if nic_op == constants.DDM_REMOVE:
+        if not instance.nics:
+          raise errors.OpPrereqError("Instance has no NICs, cannot remove")
+        continue
+      if nic_op != constants.DDM_ADD:
+        # an existing nic
+        if nic_op < 0 or nic_op >= len(instance.nics):
+          raise errors.OpPrereqError("Invalid NIC index %s, valid values"
+                                     " are 0 to %d" %
+                                     (nic_op, len(instance.nics)))
+        old_nic_params = instance.nics[nic_op].nicparams
+        old_nic_ip = instance.nics[nic_op].ip
+      else:
+        old_nic_params = {}
+        old_nic_ip = None
+
+      update_params_dict = dict([(key, nic_dict[key])
+                                 for key in constants.NICS_PARAMETERS
+                                 if key in nic_dict])
+
+      if 'bridge' in nic_dict:
+        update_params_dict[constants.NIC_LINK] = nic_dict['bridge']
+
+      new_nic_params, new_filled_nic_params = \
+          self._GetUpdatedParams(old_nic_params, update_params_dict,
+                                 cluster.nicparams[constants.PP_DEFAULT],
+                                 constants.NICS_PARAMETER_TYPES)
+      objects.NIC.CheckParameterSyntax(new_filled_nic_params)
+      self.nic_pinst[nic_op] = new_nic_params
+      self.nic_pnew[nic_op] = new_filled_nic_params
+      new_nic_mode = new_filled_nic_params[constants.NIC_MODE]
+
+      if new_nic_mode == constants.NIC_MODE_BRIDGED:
+        nic_bridge = new_filled_nic_params[constants.NIC_LINK]
+        msg = self.rpc.call_bridges_exist(pnode, [nic_bridge]).fail_msg
+        if msg:
+          msg = "Error checking bridges on node %s: %s" % (pnode, msg)
+          if self.force:
+            self.warn.append(msg)
+          else:
+            raise errors.OpPrereqError(msg)
+      if new_nic_mode == constants.NIC_MODE_ROUTED:
+        if 'ip' in nic_dict:
+          nic_ip = nic_dict['ip']
+        else:
+          nic_ip = old_nic_ip
+        if nic_ip is None:
+          raise errors.OpPrereqError('Cannot set the nic ip to None'
+                                     ' on a routed nic')
+      if 'mac' in nic_dict:
+        nic_mac = nic_dict['mac']
+        if nic_mac is None:
+          raise errors.OpPrereqError('Cannot set the nic mac to None')
+        elif nic_mac in (constants.VALUE_AUTO, constants.VALUE_GENERATE):
+          # otherwise generate the mac
+          nic_dict['mac'] = self.cfg.GenerateMAC()
+        else:
+          # or validate/reserve the current one
+          if self.cfg.IsMacInUse(nic_mac):
+            raise errors.OpPrereqError("MAC address %s already in use"
+                                       " in cluster" % nic_mac)
+
+    # DISK processing
+    if self.op.disks and instance.disk_template == constants.DT_DISKLESS:
+      raise errors.OpPrereqError("Disk operations not supported for"
+                                 " diskless instances")
+    for disk_op, disk_dict in self.op.disks:
+      if disk_op == constants.DDM_REMOVE:
+        if len(instance.disks) == 1:
+          raise errors.OpPrereqError("Cannot remove the last disk of"
+                                     " an instance")
+        ins_l = self.rpc.call_instance_list([pnode], [instance.hypervisor])
+        ins_l = ins_l[pnode]
+        msg = ins_l.fail_msg
+        if msg:
+          raise errors.OpPrereqError("Can't contact node %s: %s" %
+                                     (pnode, msg))
+        if instance.name in ins_l.payload:
+          raise errors.OpPrereqError("Instance is running, can't remove"
+                                     " disks.")
+
+      if (disk_op == constants.DDM_ADD and
+          len(instance.nics) >= constants.MAX_DISKS):
+        raise errors.OpPrereqError("Instance has too many disks (%d), cannot"
+                                   " add more" % constants.MAX_DISKS)
+      if disk_op not in (constants.DDM_ADD, constants.DDM_REMOVE):
+        # an existing disk
+        if disk_op < 0 or disk_op >= len(instance.disks):
+          raise errors.OpPrereqError("Invalid disk index %s, valid values"
+                                     " are 0 to %d" %
+                                     (disk_op, len(instance.disks)))
+
      return
  
    def Exec(self, feedback_fn):
      """Modifies an instance.
  
      All parameters take effect only at the next restart of the instance.
      return
  
    def Exec(self, feedback_fn):
      """Modifies an instance.
  
      All parameters take effect only at the next restart of the instance.
+
      """
      # Process here the warnings from CheckPrereq, as we don't have a
      # feedback_fn there.
      """
      # Process here the warnings from CheckPrereq, as we don't have a
      # feedback_fn there.
@@ -4843,19 +7168,93 @@ class LUSetInstanceParams(LogicalUnit):
  
      result = []
      instance = self.instance
  
      result = []
      instance = self.instance
-    if self.do_ip:
-      instance.nics[0].ip = self.ip
-      result.append(("ip", self.ip))
-    if self.bridge:
-      instance.nics[0].bridge = self.bridge
-      result.append(("bridge", self.bridge))
-    if self.mac:
-      instance.nics[0].mac = self.mac
-      result.append(("mac", self.mac))
+    cluster = self.cluster
+    # disk changes
+    for disk_op, disk_dict in self.op.disks:
+      if disk_op == constants.DDM_REMOVE:
+        # remove the last disk
+        device = instance.disks.pop()
+        device_idx = len(instance.disks)
+        for node, disk in device.ComputeNodeTree(instance.primary_node):
+          self.cfg.SetDiskID(disk, node)
+          msg = self.rpc.call_blockdev_remove(node, disk).fail_msg
+          if msg:
+            self.LogWarning("Could not remove disk/%d on node %s: %s,"
+                            " continuing anyway", device_idx, node, msg)
+        result.append(("disk/%d" % device_idx, "remove"))
+      elif disk_op == constants.DDM_ADD:
+        # add a new disk
+        if instance.disk_template == constants.DT_FILE:
+          file_driver, file_path = instance.disks[0].logical_id
+          file_path = os.path.dirname(file_path)
+        else:
+          file_driver = file_path = None
+        disk_idx_base = len(instance.disks)
+        new_disk = _GenerateDiskTemplate(self,
+                                         instance.disk_template,
+                                         instance.name, instance.primary_node,
+                                         instance.secondary_nodes,
+                                         [disk_dict],
+                                         file_path,
+                                         file_driver,
+                                         disk_idx_base)[0]
+        instance.disks.append(new_disk)
+        info = _GetInstanceInfoText(instance)
+
+        logging.info("Creating volume %s for instance %s",
+                     new_disk.iv_name, instance.name)
+        # Note: this needs to be kept in sync with _CreateDisks
+        #HARDCODE
+        for node in instance.all_nodes:
+          f_create = node == instance.primary_node
+          try:
+            _CreateBlockDev(self, node, instance, new_disk,
+                            f_create, info, f_create)
+          except errors.OpExecError, err:
+            self.LogWarning("Failed to create volume %s (%s) on"
+                            " node %s: %s",
+                            new_disk.iv_name, new_disk, node, err)
+        result.append(("disk/%d" % disk_idx_base, "add:size=%s,mode=%s" %
+                       (new_disk.size, new_disk.mode)))
+      else:
+        # change a given disk
+        instance.disks[disk_op].mode = disk_dict['mode']
+        result.append(("disk.mode/%d" % disk_op, disk_dict['mode']))
+    # NIC changes
+    for nic_op, nic_dict in self.op.nics:
+      if nic_op == constants.DDM_REMOVE:
+        # remove the last nic
+        del instance.nics[-1]
+        result.append(("nic.%d" % len(instance.nics), "remove"))
+      elif nic_op == constants.DDM_ADD:
+        # mac and bridge should be set, by now
+        mac = nic_dict['mac']
+        ip = nic_dict.get('ip', None)
+        nicparams = self.nic_pinst[constants.DDM_ADD]
+        new_nic = objects.NIC(mac=mac, ip=ip, nicparams=nicparams)
+        instance.nics.append(new_nic)
+        result.append(("nic.%d" % (len(instance.nics) - 1),
+                       "add:mac=%s,ip=%s,mode=%s,link=%s" %
+                       (new_nic.mac, new_nic.ip,
+                        self.nic_pnew[constants.DDM_ADD][constants.NIC_MODE],
+                        self.nic_pnew[constants.DDM_ADD][constants.NIC_LINK]
+                       )))
+      else:
+        for key in 'mac', 'ip':
+          if key in nic_dict:
+            setattr(instance.nics[nic_op], key, nic_dict[key])
+        if nic_op in self.nic_pnew:
+          instance.nics[nic_op].nicparams = self.nic_pnew[nic_op]
+        for key, val in nic_dict.iteritems():
+          result.append(("nic.%s/%d" % (key, nic_op), val))
+
+    # hvparams changes
      if self.op.hvparams:
      if self.op.hvparams:
-      instance.hvparams = self.hv_new
+      instance.hvparams = self.hv_inst
        for key, val in self.op.hvparams.iteritems():
          result.append(("hv/%s" % key, val))
        for key, val in self.op.hvparams.iteritems():
          result.append(("hv/%s" % key, val))
+
+    # beparams changes
      if self.op.beparams:
        instance.beparams = self.be_inst
        for key, val in self.op.beparams.iteritems():
      if self.op.beparams:
        instance.beparams = self.be_inst
        for key, val in self.op.beparams.iteritems():
@@ -4897,7 +7296,15 @@ class LUQueryExports(NoHooksLU):
          that node.
  
      """
          that node.
  
      """
-    return self.rpc.call_export_list(self.nodes)
+    rpcresult = self.rpc.call_export_list(self.nodes)
+    result = {}
+    for node in rpcresult:
+      if rpcresult[node].fail_msg:
+        result[node] = False
+      else:
+        result[node] = rpcresult[node].payload
+
+    return result
  
  
  class LUExportInstance(LogicalUnit):
  
  
  class LUExportInstance(LogicalUnit):
@@ -4918,7 +7325,7 @@ class LUExportInstance(LogicalUnit):
      # remove it from its current node. In the future we could fix this by:
      #  - making a tasklet to search (share-lock all), then create the new one,
      #    then one to remove, after
      # remove it from its current node. In the future we could fix this by:
      #  - making a tasklet to search (share-lock all), then create the new one,
      #    then one to remove, after
-    #  - removing the removal operation altoghether
+    #  - removing the removal operation altogether
      self.needed_locks[locking.LEVEL_NODE] = locking.ALL_SET
  
    def DeclareLocks(self, level):
      self.needed_locks[locking.LEVEL_NODE] = locking.ALL_SET
  
    def DeclareLocks(self, level):
@@ -4950,12 +7357,16 @@ class LUExportInstance(LogicalUnit):
      self.instance = self.cfg.GetInstanceInfo(instance_name)
      assert self.instance is not None, \
            "Cannot retrieve locked instance %s" % self.op.instance_name
      self.instance = self.cfg.GetInstanceInfo(instance_name)
      assert self.instance is not None, \
            "Cannot retrieve locked instance %s" % self.op.instance_name
+    _CheckNodeOnline(self, self.instance.primary_node)
  
      self.dst_node = self.cfg.GetNodeInfo(
        self.cfg.ExpandNodeName(self.op.target_node))
  
  
      self.dst_node = self.cfg.GetNodeInfo(
        self.cfg.ExpandNodeName(self.op.target_node))
  
-    assert self.dst_node is not None, \
-          "Cannot retrieve locked node %s" % self.op.target_node
+    if self.dst_node is None:
+      # This is wrong node name, not a non-locked node
+      raise errors.OpPrereqError("Wrong node name %s" % self.op.target_node)
+    _CheckNodeOnline(self, self.dst_node.name)
+    _CheckNodeNotDrained(self, self.dst_node.name)
  
      # instance disk type verification
      for disk in self.instance.disks:
  
      # instance disk type verification
      for disk in self.instance.disks:
@@ -4972,53 +7383,73 @@ class LUExportInstance(LogicalUnit):
      src_node = instance.primary_node
      if self.op.shutdown:
        # shutdown the instance, but not the disks
      src_node = instance.primary_node
      if self.op.shutdown:
        # shutdown the instance, but not the disks
-      if not self.rpc.call_instance_shutdown(src_node, instance):
-        raise errors.OpExecError("Could not shutdown instance %s on node %s" %
-                                 (instance.name, src_node))
+      result = self.rpc.call_instance_shutdown(src_node, instance)
+      result.Raise("Could not shutdown instance %s on"
+                   " node %s" % (instance.name, src_node))
  
      vgname = self.cfg.GetVGName()
  
      snap_disks = []
  
  
      vgname = self.cfg.GetVGName()
  
      snap_disks = []
  
-    try:
-      for disk in instance.disks:
-        # new_dev_name will be a snapshot of an lvm leaf of the one we passed
-        new_dev_name = self.rpc.call_blockdev_snapshot(src_node, disk)
+    # set the disks ID correctly since call_instance_start needs the
+    # correct drbd minor to create the symlinks
+    for disk in instance.disks:
+      self.cfg.SetDiskID(disk, src_node)
  
  
-        if not new_dev_name:
-          self.LogWarning("Could not snapshot block device %s on node %s",
-                          disk.logical_id[1], src_node)
+    # per-disk results
+    dresults = []
+    try:
+      for idx, disk in enumerate(instance.disks):
+        # result.payload will be a snapshot of an lvm leaf of the one we passed
+        result = self.rpc.call_blockdev_snapshot(src_node, disk)
+        msg = result.fail_msg
+        if msg:
+          self.LogWarning("Could not snapshot disk/%s on node %s: %s",
+                          idx, src_node, msg)
            snap_disks.append(False)
          else:
            snap_disks.append(False)
          else:
+          disk_id = (vgname, result.payload)
            new_dev = objects.Disk(dev_type=constants.LD_LV, size=disk.size,
            new_dev = objects.Disk(dev_type=constants.LD_LV, size=disk.size,
-                                 logical_id=(vgname, new_dev_name),
-                                 physical_id=(vgname, new_dev_name),
+                                 logical_id=disk_id, physical_id=disk_id,
                                   iv_name=disk.iv_name)
            snap_disks.append(new_dev)
  
      finally:
                                   iv_name=disk.iv_name)
            snap_disks.append(new_dev)
  
      finally:
-      if self.op.shutdown and instance.status == "up":
-        if not self.rpc.call_instance_start(src_node, instance, None):
+      if self.op.shutdown and instance.admin_up:
+        result = self.rpc.call_instance_start(src_node, instance, None, None)
+        msg = result.fail_msg
+        if msg:
            _ShutdownInstanceDisks(self, instance)
            _ShutdownInstanceDisks(self, instance)
-          raise errors.OpExecError("Could not start instance")
+          raise errors.OpExecError("Could not start instance: %s" % msg)
  
      # TODO: check for size
  
      cluster_name = self.cfg.GetClusterName()
      for idx, dev in enumerate(snap_disks):
        if dev:
  
      # TODO: check for size
  
      cluster_name = self.cfg.GetClusterName()
      for idx, dev in enumerate(snap_disks):
        if dev:
-        if not self.rpc.call_snapshot_export(src_node, dev, dst_node.name,
-                                             instance, cluster_name, idx):
-          self.LogWarning("Could not export block device %s from node %s to"
-                          " node %s", dev.logical_id[1], src_node,
-                          dst_node.name)
-        if not self.rpc.call_blockdev_remove(src_node, dev):
-          self.LogWarning("Could not remove snapshot block device %s from node"
-                          " %s", dev.logical_id[1], src_node)
-
-    if not self.rpc.call_finalize_export(dst_node.name, instance, snap_disks):
-      self.LogWarning("Could not finalize export for instance %s on node %s",
-                      instance.name, dst_node.name)
+        result = self.rpc.call_snapshot_export(src_node, dev, dst_node.name,
+                                               instance, cluster_name, idx)
+        msg = result.fail_msg
+        if msg:
+          self.LogWarning("Could not export disk/%s from node %s to"
+                          " node %s: %s", idx, src_node, dst_node.name, msg)
+          dresults.append(False)
+        else:
+          dresults.append(True)
+        msg = self.rpc.call_blockdev_remove(src_node, dev).fail_msg
+        if msg:
+          self.LogWarning("Could not remove snapshot for disk/%d from node"
+                          " %s: %s", idx, src_node, msg)
+      else:
+        dresults.append(False)
+
+    result = self.rpc.call_finalize_export(dst_node.name, instance, snap_disks)
+    fin_resu = True
+    msg = result.fail_msg
+    if msg:
+      self.LogWarning("Could not finalize export for instance %s"
+                      " on node %s: %s", instance.name, dst_node.name, msg)
+      fin_resu = False
  
      nodelist = self.cfg.GetNodeList()
      nodelist.remove(dst_node.name)
  
      nodelist = self.cfg.GetNodeList()
      nodelist.remove(dst_node.name)
@@ -5026,13 +7457,18 @@ class LUExportInstance(LogicalUnit):
      # on one-node clusters nodelist will be empty after the removal
      # if we proceed the backup would be removed because OpQueryExports
      # substitutes an empty list with the full cluster node list.
      # on one-node clusters nodelist will be empty after the removal
      # if we proceed the backup would be removed because OpQueryExports
      # substitutes an empty list with the full cluster node list.
+    iname = instance.name
      if nodelist:
        exportlist = self.rpc.call_export_list(nodelist)
        for node in exportlist:
      if nodelist:
        exportlist = self.rpc.call_export_list(nodelist)
        for node in exportlist:
-        if instance.name in exportlist[node]:
-          if not self.rpc.call_export_remove(node, instance.name):
+        if exportlist[node].fail_msg:
+          continue
+        if iname in exportlist[node].payload:
+          msg = self.rpc.call_export_remove(node, iname).fail_msg
+          if msg:
              self.LogWarning("Could not remove older export for instance %s"
              self.LogWarning("Could not remove older export for instance %s"
-                            " on node %s", instance.name, node)
+                            " on node %s: %s", iname, node, msg)
+    return fin_resu, dresults
  
  
  class LURemoveExport(NoHooksLU):
  
  
  class LURemoveExport(NoHooksLU):
@@ -5066,15 +7502,21 @@ class LURemoveExport(NoHooksLU):
        fqdn_warn = True
        instance_name = self.op.instance_name
  
        fqdn_warn = True
        instance_name = self.op.instance_name
  
-    exportlist = self.rpc.call_export_list(self.acquired_locks[
-      locking.LEVEL_NODE])
+    locked_nodes = self.acquired_locks[locking.LEVEL_NODE]
+    exportlist = self.rpc.call_export_list(locked_nodes)
      found = False
      for node in exportlist:
      found = False
      for node in exportlist:
-      if instance_name in exportlist[node]:
+      msg = exportlist[node].fail_msg
+      if msg:
+        self.LogWarning("Failed to query node %s (continuing): %s", node, msg)
+        continue
+      if instance_name in exportlist[node].payload:
          found = True
          found = True
-        if not self.rpc.call_export_remove(node, instance_name):
+        result = self.rpc.call_export_remove(node, instance_name)
+        msg = result.fail_msg
+        if msg:
            logging.error("Could not remove export for instance %s"
            logging.error("Could not remove export for instance %s"
-                        " on node %s", instance_name, node)
+                        " on node %s: %s", instance_name, node, msg)
  
      if fqdn_warn and not found:
        feedback_fn("Export not found. If trying to remove an export belonging"
  
      if fqdn_warn and not found:
        feedback_fn("Export not found. If trying to remove an export belonging"
@@ -5286,12 +7728,8 @@ class LUTestDelay(NoHooksLU):
          raise errors.OpExecError("Error during master delay test")
      if self.op.on_nodes:
        result = self.rpc.call_test_delay(self.op.on_nodes, self.op.duration)
          raise errors.OpExecError("Error during master delay test")
      if self.op.on_nodes:
        result = self.rpc.call_test_delay(self.op.on_nodes, self.op.duration)
-      if not result:
-        raise errors.OpExecError("Complete failure from rpc call")
        for node, node_result in result.items():
        for node, node_result in result.items():
-        if not node_result:
-          raise errors.OpExecError("Failure during rpc call to node %s,"
-                                   " result: %s" % (node, node_result))
+        node_result.Raise("Failure during rpc call to node %s" % node)
  
  
  class IAllocator(object):
  
  
  class IAllocator(object):
@@ -5309,14 +7747,15 @@ class IAllocator(object):
    """
    _ALLO_KEYS = [
      "mem_size", "disks", "disk_template",
    """
    _ALLO_KEYS = [
      "mem_size", "disks", "disk_template",
-    "os", "tags", "nics", "vcpus",
+    "os", "tags", "nics", "vcpus", "hypervisor",
      ]
    _RELO_KEYS = [
      "relocate_from",
      ]
  
      ]
    _RELO_KEYS = [
      "relocate_from",
      ]
  
-  def __init__(self, lu, mode, name, **kwargs):
-    self.lu = lu
+  def __init__(self, cfg, rpc, mode, name, **kwargs):
+    self.cfg = cfg
+    self.rpc = rpc
      # init buffer variables
      self.in_text = self.out_text = self.in_data = self.out_data = None
      # init all input fields so that pylint is happy
      # init buffer variables
      self.in_text = self.out_text = self.in_data = self.out_data = None
      # init all input fields so that pylint is happy
@@ -5324,6 +7763,7 @@ class IAllocator(object):
      self.name = name
      self.mem_size = self.disks = self.disk_template = None
      self.os = self.tags = self.nics = self.vcpus = None
      self.name = name
      self.mem_size = self.disks = self.disk_template = None
      self.os = self.tags = self.nics = self.vcpus = None
+    self.hypervisor = None
      self.relocate_from = None
      # computed fields
      self.required_nodes = None
      self.relocate_from = None
      # computed fields
      self.required_nodes = None
@@ -5353,87 +7793,121 @@ class IAllocator(object):
      This is the data that is independent of the actual operation.
  
      """
      This is the data that is independent of the actual operation.
  
      """
-    cfg = self.lu.cfg
+    cfg = self.cfg
      cluster_info = cfg.GetClusterInfo()
      # cluster data
      data = {
      cluster_info = cfg.GetClusterInfo()
      # cluster data
      data = {
-      "version": 1,
+      "version": constants.IALLOCATOR_VERSION,
        "cluster_name": cfg.GetClusterName(),
        "cluster_tags": list(cluster_info.GetTags()),
        "cluster_name": cfg.GetClusterName(),
        "cluster_tags": list(cluster_info.GetTags()),
-      "enable_hypervisors": list(cluster_info.enabled_hypervisors),
+      "enabled_hypervisors": list(cluster_info.enabled_hypervisors),
        # we don't have job IDs
        }
        # we don't have job IDs
        }
-
-    i_list = []
-    cluster = self.cfg.GetClusterInfo()
-    for iname in cfg.GetInstanceList():
-      i_obj = cfg.GetInstanceInfo(iname)
-      i_list.append((i_obj, cluster.FillBE(i_obj)))
+    iinfo = cfg.GetAllInstancesInfo().values()
+    i_list = [(inst, cluster_info.FillBE(inst)) for inst in iinfo]
  
      # node data
      node_results = {}
      node_list = cfg.GetNodeList()
  
      # node data
      node_results = {}
      node_list = cfg.GetNodeList()
-    # FIXME: here we have only one hypervisor information, but
-    # instance can belong to different hypervisors
-    node_data = self.lu.rpc.call_node_info(node_list, cfg.GetVGName(),
-                                           cfg.GetHypervisorType())
-    for nname in node_list:
+
+    if self.mode == constants.IALLOCATOR_MODE_ALLOC:
+      hypervisor_name = self.hypervisor
+    elif self.mode == constants.IALLOCATOR_MODE_RELOC:
+      hypervisor_name = cfg.GetInstanceInfo(self.name).hypervisor
+
+    node_data = self.rpc.call_node_info(node_list, cfg.GetVGName(),
+                                        hypervisor_name)
+    node_iinfo = \
+      self.rpc.call_all_instances_info(node_list,
+                                       cluster_info.enabled_hypervisors)
+    for nname, nresult in node_data.items():
+      # first fill in static (config-based) values
        ninfo = cfg.GetNodeInfo(nname)
        ninfo = cfg.GetNodeInfo(nname)
-      if nname not in node_data or not isinstance(node_data[nname], dict):
-        raise errors.OpExecError("Can't get data for node %s" % nname)
-      remote_info = node_data[nname]
-      for attr in ['memory_total', 'memory_free', 'memory_dom0',
-                   'vg_size', 'vg_free', 'cpu_total']:
-        if attr not in remote_info:
-          raise errors.OpExecError("Node '%s' didn't return attribute '%s'" %
-                                   (nname, attr))
-        try:
-          remote_info[attr] = int(remote_info[attr])
-        except ValueError, err:
-          raise errors.OpExecError("Node '%s' returned invalid value for '%s':"
-                                   " %s" % (nname, attr, str(err)))
-      # compute memory used by primary instances
-      i_p_mem = i_p_up_mem = 0
-      for iinfo, beinfo in i_list:
-        if iinfo.primary_node == nname:
-          i_p_mem += beinfo[constants.BE_MEMORY]
-          if iinfo.status == "up":
-            i_p_up_mem += beinfo[constants.BE_MEMORY]
-
-      # compute memory used by instances
        pnr = {
          "tags": list(ninfo.GetTags()),
        pnr = {
          "tags": list(ninfo.GetTags()),
-        "total_memory": remote_info['memory_total'],
-        "reserved_memory": remote_info['memory_dom0'],
-        "free_memory": remote_info['memory_free'],
-        "i_pri_memory": i_p_mem,
-        "i_pri_up_memory": i_p_up_mem,
-        "total_disk": remote_info['vg_size'],
-        "free_disk": remote_info['vg_free'],
          "primary_ip": ninfo.primary_ip,
          "secondary_ip": ninfo.secondary_ip,
          "primary_ip": ninfo.primary_ip,
          "secondary_ip": ninfo.secondary_ip,
-        "total_cpus": remote_info['cpu_total'],
+        "offline": ninfo.offline,
+        "drained": ninfo.drained,
+        "master_candidate": ninfo.master_candidate,
          }
          }
+
+      if not (ninfo.offline or ninfo.drained):
+        nresult.Raise("Can't get data for node %s" % nname)
+        node_iinfo[nname].Raise("Can't get node instance info from node %s" %
+                                nname)
+        remote_info = nresult.payload
+
+        for attr in ['memory_total', 'memory_free', 'memory_dom0',
+                     'vg_size', 'vg_free', 'cpu_total']:
+          if attr not in remote_info:
+            raise errors.OpExecError("Node '%s' didn't return attribute"
+                                     " '%s'" % (nname, attr))
+          if not isinstance(remote_info[attr], int):
+            raise errors.OpExecError("Node '%s' returned invalid value"
+                                     " for '%s': %s" %
+                                     (nname, attr, remote_info[attr]))
+        # compute memory used by primary instances
+        i_p_mem = i_p_up_mem = 0
+        for iinfo, beinfo in i_list:
+          if iinfo.primary_node == nname:
+            i_p_mem += beinfo[constants.BE_MEMORY]
+            if iinfo.name not in node_iinfo[nname].payload:
+              i_used_mem = 0
+            else:
+              i_used_mem = int(node_iinfo[nname].payload[iinfo.name]['memory'])
+            i_mem_diff = beinfo[constants.BE_MEMORY] - i_used_mem
+            remote_info['memory_free'] -= max(0, i_mem_diff)
+
+            if iinfo.admin_up:
+              i_p_up_mem += beinfo[constants.BE_MEMORY]
+
+        # compute memory used by instances
+        pnr_dyn = {
+          "total_memory": remote_info['memory_total'],
+          "reserved_memory": remote_info['memory_dom0'],
+          "free_memory": remote_info['memory_free'],
+          "total_disk": remote_info['vg_size'],
+          "free_disk": remote_info['vg_free'],
+          "total_cpus": remote_info['cpu_total'],
+          "i_pri_memory": i_p_mem,
+          "i_pri_up_memory": i_p_up_mem,
+          }
+        pnr.update(pnr_dyn)
+
        node_results[nname] = pnr
      data["nodes"] = node_results
  
      # instance data
      instance_data = {}
      for iinfo, beinfo in i_list:
        node_results[nname] = pnr
      data["nodes"] = node_results
  
      # instance data
      instance_data = {}
      for iinfo, beinfo in i_list:
-      nic_data = [{"mac": n.mac, "ip": n.ip, "bridge": n.bridge}
-                  for n in iinfo.nics]
+      nic_data = []
+      for nic in iinfo.nics:
+        filled_params = objects.FillDict(
+            cluster_info.nicparams[constants.PP_DEFAULT],
+            nic.nicparams)
+        nic_dict = {"mac": nic.mac,
+                    "ip": nic.ip,
+                    "mode": filled_params[constants.NIC_MODE],
+                    "link": filled_params[constants.NIC_LINK],
+                   }
+        if filled_params[constants.NIC_MODE] == constants.NIC_MODE_BRIDGED:
+          nic_dict["bridge"] = filled_params[constants.NIC_LINK]
+        nic_data.append(nic_dict)
        pir = {
          "tags": list(iinfo.GetTags()),
        pir = {
          "tags": list(iinfo.GetTags()),
-        "should_run": iinfo.status == "up",
+        "admin_up": iinfo.admin_up,
          "vcpus": beinfo[constants.BE_VCPUS],
          "memory": beinfo[constants.BE_MEMORY],
          "os": iinfo.os,
          "nodes": [iinfo.primary_node] + list(iinfo.secondary_nodes),
          "nics": nic_data,
          "vcpus": beinfo[constants.BE_VCPUS],
          "memory": beinfo[constants.BE_MEMORY],
          "os": iinfo.os,
          "nodes": [iinfo.primary_node] + list(iinfo.secondary_nodes),
          "nics": nic_data,
-        "disks": [{"size": dsk.size, "mode": "w"} for dsk in iinfo.disks],
+        "disks": [{"size": dsk.size, "mode": dsk.mode} for dsk in iinfo.disks],
          "disk_template": iinfo.disk_template,
          "hypervisor": iinfo.hypervisor,
          }
          "disk_template": iinfo.disk_template,
          "hypervisor": iinfo.hypervisor,
          }
+      pir["disk_space_total"] = _ComputeDiskSize(iinfo.disk_template,
+                                                 pir["disks"])
        instance_data[iinfo.name] = pir
  
      data["instances"] = instance_data
        instance_data[iinfo.name] = pir
  
      data["instances"] = instance_data
@@ -5451,8 +7925,6 @@ class IAllocator(object):
  
      """
      data = self.in_data
  
      """
      data = self.in_data
-    if len(self.disks) != 2:
-      raise errors.OpExecError("Only two-disk configurations supported")
  
      disk_space = _ComputeDiskSize(self.disk_template, self.disks)
  
  
      disk_space = _ComputeDiskSize(self.disk_template, self.disks)
  
@@ -5485,7 +7957,7 @@ class IAllocator(object):
      done.
  
      """
      done.
  
      """
-    instance = self.lu.cfg.GetInstanceInfo(self.name)
+    instance = self.cfg.GetInstanceInfo(self.name)
      if instance is None:
        raise errors.ProgrammerError("Unknown instance '%s' passed to"
                                     " IAllocator" % self.name)
      if instance is None:
        raise errors.ProgrammerError("Unknown instance '%s' passed to"
                                     " IAllocator" % self.name)
@@ -5527,22 +7999,12 @@ class IAllocator(object):
  
      """
      if call_fn is None:
  
      """
      if call_fn is None:
-      call_fn = self.lu.rpc.call_iallocator_runner
-    data = self.in_text
-
-    result = call_fn(self.lu.cfg.GetMasterNode(), name, self.in_text)
-
-    if not isinstance(result, (list, tuple)) or len(result) != 4:
-      raise errors.OpExecError("Invalid result from master iallocator runner")
+      call_fn = self.rpc.call_iallocator_runner
  
  
-    rcode, stdout, stderr, fail = result
+    result = call_fn(self.cfg.GetMasterNode(), name, self.in_text)
+    result.Raise("Failure while running the iallocator script")
  
  
-    if rcode == constants.IARUN_NOTFOUND:
-      raise errors.OpExecError("Can't find allocator '%s'" % name)
-    elif rcode == constants.IARUN_FAILURE:
-      raise errors.OpExecError("Instance allocator call failed: %s,"
-                               " output: %s" % (fail, stdout+stderr))
-    self.out_text = stdout
+    self.out_text = result.payload
      if validate:
        self._ValidateResult()
  
      if validate:
        self._ValidateResult()
  
@@ -5608,8 +8070,6 @@ class LUTestAllocator(NoHooksLU):
                                       " 'nics' parameter")
        if not isinstance(self.op.disks, list):
          raise errors.OpPrereqError("Invalid parameter 'disks'")
                                       " 'nics' parameter")
        if not isinstance(self.op.disks, list):
          raise errors.OpPrereqError("Invalid parameter 'disks'")
-      if len(self.op.disks) != 2:
-        raise errors.OpPrereqError("Only two-disk configurations supported")
        for row in self.op.disks:
          if (not isinstance(row, dict) or
              "size" not in row or
        for row in self.op.disks:
          if (not isinstance(row, dict) or
              "size" not in row or
@@ -5618,6 +8078,8 @@ class LUTestAllocator(NoHooksLU):
              row["mode"] not in ['r', 'w']):
            raise errors.OpPrereqError("Invalid contents of the"
                                       " 'disks' parameter")
              row["mode"] not in ['r', 'w']):
            raise errors.OpPrereqError("Invalid contents of the"
                                       " 'disks' parameter")
+      if not hasattr(self.op, "hypervisor") or self.op.hypervisor is None:
+        self.op.hypervisor = self.cfg.GetHypervisorType()
      elif self.op.mode == constants.IALLOCATOR_MODE_RELOC:
        if not hasattr(self.op, "name"):
          raise errors.OpPrereqError("Missing attribute 'name' on opcode input")
      elif self.op.mode == constants.IALLOCATOR_MODE_RELOC:
        if not hasattr(self.op, "name"):
          raise errors.OpPrereqError("Missing attribute 'name' on opcode input")
@@ -5643,7 +8105,7 @@ class LUTestAllocator(NoHooksLU):
  
      """
      if self.op.mode == constants.IALLOCATOR_MODE_ALLOC:
  
      """
      if self.op.mode == constants.IALLOCATOR_MODE_ALLOC:
-      ial = IAllocator(self,
+      ial = IAllocator(self.cfg, self.rpc,
                         mode=self.op.mode,
                         name=self.op.name,
                         mem_size=self.op.mem_size,
                         mode=self.op.mode,
                         name=self.op.name,
                         mem_size=self.op.mem_size,
@@ -5653,9 +8115,10 @@ class LUTestAllocator(NoHooksLU):
                         tags=self.op.tags,
                         nics=self.op.nics,
                         vcpus=self.op.vcpus,
                         tags=self.op.tags,
                         nics=self.op.nics,
                         vcpus=self.op.vcpus,
+                       hypervisor=self.op.hypervisor,
                         )
      else:
                         )
      else:
-      ial = IAllocator(self,
+      ial = IAllocator(self.cfg, self.rpc,
                         mode=self.op.mode,
                         name=self.op.name,
                         relocate_from=list(self.relocate_from),
                         mode=self.op.mode,
                         name=self.op.name,
                         relocate_from=list(self.relocate_from),