ganeti-watcher 11.9 KB
Newer Older
Iustin Pop's avatar
Iustin Pop committed
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
#!/usr/bin/python
#

# Copyright (C) 2006, 2007 Google Inc.
#
# This program is free software; you can redistribute it and/or modify
# it under the terms of the GNU General Public License as published by
# the Free Software Foundation; either version 2 of the License, or
# (at your option) any later version.
#
# This program is distributed in the hope that it will be useful, but
# WITHOUT ANY WARRANTY; without even the implied warranty of
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE.  See the GNU
# General Public License for more details.
#
# You should have received a copy of the GNU General Public License
# along with this program; if not, write to the Free Software
# Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA
# 02110-1301, USA.


"""Tool to restart erronously downed virtual machines.

This program and set of classes implement a watchdog to restart
virtual machines in a Ganeti cluster that have crashed or been killed
by a node reboot.  Run from cron or similar.

28
"""
Iustin Pop's avatar
Iustin Pop committed
29
30
31

import os
import sys
32
import re
Iustin Pop's avatar
Iustin Pop committed
33
34
35
import time
import fcntl
import errno
36
import simplejson
Iustin Pop's avatar
Iustin Pop committed
37
38
39
40
from optparse import OptionParser

from ganeti import utils
from ganeti import constants
41
from ganeti import ssconf
42
from ganeti import errors
Iustin Pop's avatar
Iustin Pop committed
43
44


45
46
47
48
49
50
51
MAXTRIES = 5
BAD_STATES = ['stopped']
HELPLESS_STATES = ['(node down)']
NOTICE = 'NOTICE'
ERROR = 'ERROR'


Iustin Pop's avatar
Iustin Pop committed
52
53
class Error(Exception):
  """Generic custom error class."""
54
55
56
57


class NotMasterError(Error):
  """Exception raised when this host is not the master."""
Iustin Pop's avatar
Iustin Pop committed
58
59
60
61
62
63
64
65


def Indent(s, prefix='| '):
  """Indent a piece of text with a given prefix before each line.

  Args:
    s: The string to indent
    prefix: The string to prepend each line.
66

Iustin Pop's avatar
Iustin Pop committed
67
68
69
70
71
72
73
74
75
76
77
  """
  return "%s%s\n" % (prefix, ('\n' + prefix).join(s.splitlines()))


def DoCmd(cmd):
  """Run a shell command.

  Args:
    cmd: the command to run.

  Raises CommandError with verbose commentary on error.
78

Iustin Pop's avatar
Iustin Pop committed
79
80
81
82
83
84
85
86
87
88
89
90
91
  """
  res = utils.RunCmd(cmd)

  if res.failed:
    raise Error("Command %s failed:\n%s\nstdout:\n%sstderr:\n%s" %
                (repr(cmd),
                 Indent(res.fail_reason),
                 Indent(res.stdout),
                 Indent(res.stderr)))

  return res


92
class WatcherState(object):
Iustin Pop's avatar
Iustin Pop committed
93
94
95
96
  """Interface to a state file recording restart attempts.

  """
  def __init__(self):
97
98
99
100
101
    """Open, lock, read and parse the file.

    Raises StandardError on lock contention.

    """
Iustin Pop's avatar
Iustin Pop committed
102
103
104
105
106
107
108
109
110
111
    # The two-step dance below is necessary to allow both opening existing
    # file read/write and creating if not existing.  Vanilla open will truncate
    # an existing file -or- allow creating if not existing.
    f = os.open(constants.WATCHER_STATEFILE, os.O_RDWR | os.O_CREAT)
    f = os.fdopen(f, 'w+')

    try:
      fcntl.flock(f.fileno(), fcntl.LOCK_EX|fcntl.LOCK_NB)
    except IOError, x:
      if x.errno == errno.EAGAIN:
112
        raise StandardError("State file already locked")
Iustin Pop's avatar
Iustin Pop committed
113
114
115
116
      raise

    self.statefile = f

117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
    try:
      self.data = simplejson.load(self.statefile)
    except Exception, msg:
      # Ignore errors while loading the file and treat it as empty
      self.data = {}
      sys.stderr.write("Empty or invalid state file. "
          "Using defaults. Error message: %s\n" % msg)

    if "instance" not in self.data:
      self.data["instance"] = {}
    if "node" not in self.data:
      self.data["node"] = {}

  def __del__(self):
    """Called on destruction.

    """
    if self.statefile:
      self._Close()

  def _Close(self):
    """Unlock configuration file and close it.

    """
    assert self.statefile

    fcntl.flock(self.statefile.fileno(), fcntl.LOCK_UN)

    self.statefile.close()
    self.statefile = None

  def GetNodeBootID(self, name):
    """Returns the last boot ID of a node or None.
Iustin Pop's avatar
Iustin Pop committed
150

151
152
153
154
155
156
157
158
159
160
161
162
    """
    ndata = self.data["node"]

    if name in ndata and "bootid" in ndata[name]:
      return ndata[name]["bootid"]
    return None

  def SetNodeBootID(self, name, bootid):
    """Sets the boot ID of a node.

    """
    assert bootid
Iustin Pop's avatar
Iustin Pop committed
163

164
    ndata = self.data["node"]
Iustin Pop's avatar
Iustin Pop committed
165

166
167
168
169
170
171
    if name not in ndata:
      ndata[name] = {}

    ndata[name]["bootid"] = bootid

  def NumberOfRestartAttempts(self, instance):
Iustin Pop's avatar
Iustin Pop committed
172
173
174
175
    """Returns number of previous restart attempts.

    Args:
      instance - the instance to look up.
176

Iustin Pop's avatar
Iustin Pop committed
177
    """
178
    idata = self.data["instance"]
Iustin Pop's avatar
Iustin Pop committed
179

180
181
    if instance.name in idata:
      return idata[instance.name]["restart_count"]
Iustin Pop's avatar
Iustin Pop committed
182
183
184

    return 0

185
  def RecordRestartAttempt(self, instance):
Iustin Pop's avatar
Iustin Pop committed
186
187
188
189
    """Record a restart attempt.

    Args:
      instance - the instance being restarted
190

Iustin Pop's avatar
Iustin Pop committed
191
    """
192
    idata = self.data["instance"]
Iustin Pop's avatar
Iustin Pop committed
193

194
195
196
197
    if instance.name not in idata:
      inst = idata[instance.name] = {}
    else:
      inst = idata[instance.name]
Iustin Pop's avatar
Iustin Pop committed
198

199
200
    inst["restart_when"] = time.time()
    inst["restart_count"] = idata.get("restart_count", 0) + 1
Iustin Pop's avatar
Iustin Pop committed
201

202
  def RemoveInstance(self, instance):
203
    """Update state to reflect that a machine is running, i.e. remove record.
Iustin Pop's avatar
Iustin Pop committed
204
205
206
207

    Args:
      instance - the instance to remove from books

208
209
    This method removes the record for a named instance.

Iustin Pop's avatar
Iustin Pop committed
210
    """
211
    idata = self.data["instance"]
Iustin Pop's avatar
Iustin Pop committed
212

213
214
    if instance.name in idata:
      del idata[instance.name]
Iustin Pop's avatar
Iustin Pop committed
215
216

  def Save(self):
217
    """Save state to file, then unlock and close it.
218

Iustin Pop's avatar
Iustin Pop committed
219
220
221
222
223
224
    """
    assert self.statefile

    self.statefile.seek(0)
    self.statefile.truncate()

225
    simplejson.dump(self.data, self.statefile)
Iustin Pop's avatar
Iustin Pop committed
226

227
    self._Close()
Iustin Pop's avatar
Iustin Pop committed
228
229
230
231
232
233
234


class Instance(object):
  """Abstraction for a Virtual Machine instance.

  Methods:
    Restart(): issue a command to restart the represented machine.
235

Iustin Pop's avatar
Iustin Pop committed
236
  """
237
  def __init__(self, name, state, autostart):
Iustin Pop's avatar
Iustin Pop committed
238
239
    self.name = name
    self.state = state
240
    self.autostart = autostart
Iustin Pop's avatar
Iustin Pop committed
241
242

  def Restart(self):
243
244
245
    """Encapsulates the start of an instance.

    """
Iustin Pop's avatar
Iustin Pop committed
246
247
    DoCmd(['gnt-instance', 'startup', '--lock-retries=15', self.name])

248
249
250
251
252
253
  def ActivateDisks(self):
    """Encapsulates the activation of all disks of an instance.

    """
    DoCmd(['gnt-instance', 'activate-disks', '--lock-retries=15', self.name])

Iustin Pop's avatar
Iustin Pop committed
254

255
256
def _RunListCmd(cmd):
  """Runs a command and parses its output into lists.
257

Iustin Pop's avatar
Iustin Pop committed
258
  """
259
260
  for line in DoCmd(cmd).stdout.splitlines():
    yield line.split(':')
Iustin Pop's avatar
Iustin Pop committed
261
262


263
264
265
266
267
268
269
270
def GetInstanceList(with_secondaries=None):
  """Get a list of instances on this cluster.

  """
  cmd = ['gnt-instance', 'list', '--lock-retries=15', '--no-headers',
         '--separator=:']

  fields = 'name,oper_state,admin_state'
Iustin Pop's avatar
Iustin Pop committed
271

272
273
  if with_secondaries is not None:
    fields += ',snodes'
Iustin Pop's avatar
Iustin Pop committed
274

275
276
277
278
279
280
281
282
283
  cmd.append('-o')
  cmd.append(fields)

  instances = []
  for fields in _RunListCmd(cmd):
    if with_secondaries is not None:
      (name, status, autostart, snodes) = fields

      if snodes == "-":
Iustin Pop's avatar
Iustin Pop committed
284
        continue
285
286
287
288
289

      for node in with_secondaries:
        if node in snodes.split(','):
          break
      else:
Iustin Pop's avatar
Iustin Pop committed
290
291
        continue

292
293
294
295
    else:
      (name, status, autostart) = fields

    instances.append(Instance(name, status, autostart != "no"))
Iustin Pop's avatar
Iustin Pop committed
296

297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
  return instances


def GetNodeBootIDs():
  """Get a dict mapping nodes to boot IDs.

  """
  cmd = ['gnt-node', 'list', '--lock-retries=15', '--no-headers',
         '--separator=:', '-o', 'name,bootid']

  ids = {}
  for fields in _RunListCmd(cmd):
    (name, bootid) = fields
    ids[name] = bootid

  return ids
Iustin Pop's avatar
Iustin Pop committed
313
314
315
316


class Message(object):
  """Encapsulation of a notice or error message.
317

Iustin Pop's avatar
Iustin Pop committed
318
319
320
321
322
323
324
325
326
327
  """
  def __init__(self, level, msg):
    self.level = level
    self.msg = msg
    self.when = time.time()

  def __str__(self):
    return self.level + ' ' + time.ctime(self.when) + '\n' + Indent(self.msg)


328
class Watcher(object):
Iustin Pop's avatar
Iustin Pop committed
329
330
331
332
333
  """Encapsulate the logic for restarting erronously halted virtual machines.

  The calling program should periodically instantiate me and call Run().
  This will traverse the list of instances, and make up to MAXTRIES attempts
  to restart machines that are down.
334

Iustin Pop's avatar
Iustin Pop committed
335
336
  """
  def __init__(self):
337
338
    sstore = ssconf.SimpleStore()
    master = sstore.GetMasterNode()
339
    if master != utils.HostInfo().name:
340
      raise NotMasterError("This is not the master node")
341
342
    self.instances = GetInstanceList()
    self.bootids = GetNodeBootIDs()
Iustin Pop's avatar
Iustin Pop committed
343
344
345
    self.messages = []

  def Run(self):
346
347
348
349
350
351
352
    notepad = WatcherState()
    self.CheckInstances(notepad)
    self.CheckDisks(notepad)
    notepad.Save()

  def CheckDisks(self, notepad):
    """Check all nodes for restarted ones.
353

Iustin Pop's avatar
Iustin Pop committed
354
    """
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
    check_nodes = []
    for name, id in self.bootids.iteritems():
      old = notepad.GetNodeBootID(name)
      if old != id:
        # Node's boot ID has changed, proably through a reboot.
        check_nodes.append(name)

    if check_nodes:
      # Activate disks for all instances with any of the checked nodes as a
      # secondary node.
      for instance in GetInstanceList(with_secondaries=check_nodes):
        try:
          self.messages.append(Message(NOTICE,
                                       "Activating disks for %s." %
                                       instance.name))
          instance.ActivateDisks()
        except Error, x:
          self.messages.append(Message(ERROR, str(x)))

      # Keep changed boot IDs
      for name in check_nodes:
        notepad.SetNodeBootID(name, self.bootids[name])
Iustin Pop's avatar
Iustin Pop committed
377

378
379
380
381
  def CheckInstances(self, notepad):
    """Make a pass over the list of instances, restarting downed ones.

    """
Iustin Pop's avatar
Iustin Pop committed
382
    for instance in self.instances:
383
384
385
386
      # Don't care about manually stopped instances
      if not instance.autostart:
        continue

Iustin Pop's avatar
Iustin Pop committed
387
      if instance.state in BAD_STATES:
388
        n = notepad.NumberOfRestartAttempts(instance)
Iustin Pop's avatar
Iustin Pop committed
389
390
391
392
393
394
395

        if n > MAXTRIES:
          # stay quiet.
          continue
        elif n < MAXTRIES:
          last = " (Attempt #%d)" % (n + 1)
        else:
396
          notepad.RecordRestartAttempt(instance)
Iustin Pop's avatar
Iustin Pop committed
397
398
399
400
401
402
403
404
405
406
407
408
          self.messages.append(Message(ERROR, "Could not restart %s for %d"
                                       " times, giving up..." %
                                       (instance.name, MAXTRIES)))
          continue
        try:
          self.messages.append(Message(NOTICE,
                                       "Restarting %s%s." %
                                       (instance.name, last)))
          instance.Restart()
        except Error, x:
          self.messages.append(Message(ERROR, str(x)))

409
        notepad.RecordRestartAttempt(instance)
Iustin Pop's avatar
Iustin Pop committed
410
      elif instance.state in HELPLESS_STATES:
411
412
        if notepad.NumberOfRestartAttempts(instance):
          notepad.RemoveInstance(instance)
Iustin Pop's avatar
Iustin Pop committed
413
      else:
414
415
        if notepad.NumberOfRestartAttempts(instance):
          notepad.RemoveInstance(instance)
Iustin Pop's avatar
Iustin Pop committed
416
417
418
419
420
          msg = Message(NOTICE,
                        "Restart of %s succeeded." % instance.name)
          self.messages.append(msg)

  def WriteReport(self, logfile):
421
    """Log all messages to file.
Iustin Pop's avatar
Iustin Pop committed
422
423
424

    Args:
      logfile: file object open for writing (the log file)
425

Iustin Pop's avatar
Iustin Pop committed
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
    """
    for msg in self.messages:
      print >> logfile, str(msg)


def ParseOptions():
  """Parse the command line options.

  Returns:
    (options, args) as from OptionParser.parse_args()

  """
  parser = OptionParser(description="Ganeti cluster watcher",
                        usage="%prog [-d]",
                        version="%%prog (ganeti) %s" %
                        constants.RELEASE_VERSION)

  parser.add_option("-d", "--debug", dest="debug",
                    help="Don't redirect messages to the log file",
                    default=False, action="store_true")
  options, args = parser.parse_args()
  return options, args


def main():
  """Main function.

  """
  options, args = ParseOptions()

  if not options.debug:
457
    sys.stderr = sys.stdout = open(constants.LOG_WATCHER, 'a')
Iustin Pop's avatar
Iustin Pop committed
458
459

  try:
460
461
462
463
464
    try:
      watcher = Watcher()
    except errors.ConfigurationError:
      # Just exit if there's no configuration
      sys.exit(constants.EXIT_SUCCESS)
465
466
    watcher.Run()
    watcher.WriteReport(sys.stdout)
467
468
469
470
  except NotMasterError:
    if options.debug:
      sys.stderr.write("Not master, exiting.\n")
    sys.exit(constants.EXIT_NOTMASTER)
471
472
473
  except errors.ResolverError, err:
    sys.stderr.write("Cannot resolve hostname '%s', exiting.\n" % err.args[0])
    sys.exit(constants.EXIT_NODESETUP_ERROR)
Iustin Pop's avatar
Iustin Pop committed
474
475
476
  except Error, err:
    print err

477

Iustin Pop's avatar
Iustin Pop committed
478
479
if __name__ == '__main__':
  main()