ganeti-watcher 11.9 KB
Newer Older
Iustin Pop's avatar
Iustin Pop committed
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
#!/usr/bin/python
#

# Copyright (C) 2006, 2007 Google Inc.
#
# This program is free software; you can redistribute it and/or modify
# it under the terms of the GNU General Public License as published by
# the Free Software Foundation; either version 2 of the License, or
# (at your option) any later version.
#
# This program is distributed in the hope that it will be useful, but
# WITHOUT ANY WARRANTY; without even the implied warranty of
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE.  See the GNU
# General Public License for more details.
#
# You should have received a copy of the GNU General Public License
# along with this program; if not, write to the Free Software
# Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA
# 02110-1301, USA.


"""Tool to restart erronously downed virtual machines.

This program and set of classes implement a watchdog to restart
virtual machines in a Ganeti cluster that have crashed or been killed
by a node reboot.  Run from cron or similar.

28
"""
Iustin Pop's avatar
Iustin Pop committed
29
30
31

import os
import sys
32
import re
Iustin Pop's avatar
Iustin Pop committed
33
34
35
import time
import fcntl
import errno
36
import simplejson
Iustin Pop's avatar
Iustin Pop committed
37
38
39
40
from optparse import OptionParser

from ganeti import utils
from ganeti import constants
41
from ganeti import ssconf
42
from ganeti import errors
Iustin Pop's avatar
Iustin Pop committed
43
44


45
46
47
48
49
MAXTRIES = 5
BAD_STATES = ['stopped']
HELPLESS_STATES = ['(node down)']
NOTICE = 'NOTICE'
ERROR = 'ERROR'
50
51
52
KEY_RESTART_COUNT = "restart_count"
KEY_RESTART_WHEN = "restart_when"
KEY_BOOT_ID = "bootid"
53
54


Iustin Pop's avatar
Iustin Pop committed
55
56
class Error(Exception):
  """Generic custom error class."""
57
58
59
60


class NotMasterError(Error):
  """Exception raised when this host is not the master."""
Iustin Pop's avatar
Iustin Pop committed
61
62
63
64
65
66
67
68


def Indent(s, prefix='| '):
  """Indent a piece of text with a given prefix before each line.

  Args:
    s: The string to indent
    prefix: The string to prepend each line.
69

Iustin Pop's avatar
Iustin Pop committed
70
71
72
73
74
75
76
77
78
79
80
  """
  return "%s%s\n" % (prefix, ('\n' + prefix).join(s.splitlines()))


def DoCmd(cmd):
  """Run a shell command.

  Args:
    cmd: the command to run.

  Raises CommandError with verbose commentary on error.
81

Iustin Pop's avatar
Iustin Pop committed
82
83
84
85
86
87
88
89
90
91
92
93
94
  """
  res = utils.RunCmd(cmd)

  if res.failed:
    raise Error("Command %s failed:\n%s\nstdout:\n%sstderr:\n%s" %
                (repr(cmd),
                 Indent(res.fail_reason),
                 Indent(res.stdout),
                 Indent(res.stderr)))

  return res


95
class WatcherState(object):
Iustin Pop's avatar
Iustin Pop committed
96
97
98
99
  """Interface to a state file recording restart attempts.

  """
  def __init__(self):
100
101
102
103
104
    """Open, lock, read and parse the file.

    Raises StandardError on lock contention.

    """
Iustin Pop's avatar
Iustin Pop committed
105
106
107
108
109
110
111
112
113
114
    # The two-step dance below is necessary to allow both opening existing
    # file read/write and creating if not existing.  Vanilla open will truncate
    # an existing file -or- allow creating if not existing.
    f = os.open(constants.WATCHER_STATEFILE, os.O_RDWR | os.O_CREAT)
    f = os.fdopen(f, 'w+')

    try:
      fcntl.flock(f.fileno(), fcntl.LOCK_EX|fcntl.LOCK_NB)
    except IOError, x:
      if x.errno == errno.EAGAIN:
115
        raise StandardError("State file already locked")
Iustin Pop's avatar
Iustin Pop committed
116
117
118
119
      raise

    self.statefile = f

120
121
122
123
124
    try:
      self.data = simplejson.load(self.statefile)
    except Exception, msg:
      # Ignore errors while loading the file and treat it as empty
      self.data = {}
125
126
      sys.stderr.write("Empty or invalid state file."
                       " Using defaults. Error message: %s\n" % msg)
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152

    if "instance" not in self.data:
      self.data["instance"] = {}
    if "node" not in self.data:
      self.data["node"] = {}

  def __del__(self):
    """Called on destruction.

    """
    if self.statefile:
      self._Close()

  def _Close(self):
    """Unlock configuration file and close it.

    """
    assert self.statefile

    fcntl.flock(self.statefile.fileno(), fcntl.LOCK_UN)

    self.statefile.close()
    self.statefile = None

  def GetNodeBootID(self, name):
    """Returns the last boot ID of a node or None.
Iustin Pop's avatar
Iustin Pop committed
153

154
155
156
    """
    ndata = self.data["node"]

157
158
    if name in ndata and KEY_BOOT_ID in ndata[name]:
      return ndata[name][KEY_BOOT_ID]
159
160
161
162
163
164
165
    return None

  def SetNodeBootID(self, name, bootid):
    """Sets the boot ID of a node.

    """
    assert bootid
Iustin Pop's avatar
Iustin Pop committed
166

167
    ndata = self.data["node"]
Iustin Pop's avatar
Iustin Pop committed
168

169
170
171
    if name not in ndata:
      ndata[name] = {}

172
    ndata[name][KEY_BOOT_ID] = bootid
173
174

  def NumberOfRestartAttempts(self, instance):
Iustin Pop's avatar
Iustin Pop committed
175
176
177
178
    """Returns number of previous restart attempts.

    Args:
      instance - the instance to look up.
179

Iustin Pop's avatar
Iustin Pop committed
180
    """
181
    idata = self.data["instance"]
Iustin Pop's avatar
Iustin Pop committed
182

183
    if instance.name in idata:
184
      return idata[instance.name][KEY_RESTART_COUNT]
Iustin Pop's avatar
Iustin Pop committed
185
186
187

    return 0

188
  def RecordRestartAttempt(self, instance):
Iustin Pop's avatar
Iustin Pop committed
189
190
191
192
    """Record a restart attempt.

    Args:
      instance - the instance being restarted
193

Iustin Pop's avatar
Iustin Pop committed
194
    """
195
    idata = self.data["instance"]
Iustin Pop's avatar
Iustin Pop committed
196

197
198
199
200
    if instance.name not in idata:
      inst = idata[instance.name] = {}
    else:
      inst = idata[instance.name]
Iustin Pop's avatar
Iustin Pop committed
201

202
203
    inst[KEY_RESTART_WHEN] = time.time()
    inst[KEY_RESTART_COUNT] = inst.get(KEY_RESTART_COUNT, 0) + 1
Iustin Pop's avatar
Iustin Pop committed
204

205
  def RemoveInstance(self, instance):
206
    """Update state to reflect that a machine is running, i.e. remove record.
Iustin Pop's avatar
Iustin Pop committed
207
208
209
210

    Args:
      instance - the instance to remove from books

211
212
    This method removes the record for a named instance.

Iustin Pop's avatar
Iustin Pop committed
213
    """
214
    idata = self.data["instance"]
Iustin Pop's avatar
Iustin Pop committed
215

216
217
    if instance.name in idata:
      del idata[instance.name]
Iustin Pop's avatar
Iustin Pop committed
218
219

  def Save(self):
220
    """Save state to file, then unlock and close it.
221

Iustin Pop's avatar
Iustin Pop committed
222
223
224
225
226
227
    """
    assert self.statefile

    self.statefile.seek(0)
    self.statefile.truncate()

228
    simplejson.dump(self.data, self.statefile)
Iustin Pop's avatar
Iustin Pop committed
229

230
    self._Close()
Iustin Pop's avatar
Iustin Pop committed
231
232
233
234
235
236
237


class Instance(object):
  """Abstraction for a Virtual Machine instance.

  Methods:
    Restart(): issue a command to restart the represented machine.
238

Iustin Pop's avatar
Iustin Pop committed
239
  """
240
  def __init__(self, name, state, autostart):
Iustin Pop's avatar
Iustin Pop committed
241
242
    self.name = name
    self.state = state
243
    self.autostart = autostart
Iustin Pop's avatar
Iustin Pop committed
244
245

  def Restart(self):
246
247
248
    """Encapsulates the start of an instance.

    """
Iustin Pop's avatar
Iustin Pop committed
249
250
    DoCmd(['gnt-instance', 'startup', '--lock-retries=15', self.name])

251
252
253
254
255
256
  def ActivateDisks(self):
    """Encapsulates the activation of all disks of an instance.

    """
    DoCmd(['gnt-instance', 'activate-disks', '--lock-retries=15', self.name])

Iustin Pop's avatar
Iustin Pop committed
257

258
259
def _RunListCmd(cmd):
  """Runs a command and parses its output into lists.
260

Iustin Pop's avatar
Iustin Pop committed
261
  """
262
263
  for line in DoCmd(cmd).stdout.splitlines():
    yield line.split(':')
Iustin Pop's avatar
Iustin Pop committed
264
265


266
267
268
269
270
271
272
273
def GetInstanceList(with_secondaries=None):
  """Get a list of instances on this cluster.

  """
  cmd = ['gnt-instance', 'list', '--lock-retries=15', '--no-headers',
         '--separator=:']

  fields = 'name,oper_state,admin_state'
Iustin Pop's avatar
Iustin Pop committed
274

275
276
  if with_secondaries is not None:
    fields += ',snodes'
Iustin Pop's avatar
Iustin Pop committed
277

278
279
280
281
282
283
284
285
286
  cmd.append('-o')
  cmd.append(fields)

  instances = []
  for fields in _RunListCmd(cmd):
    if with_secondaries is not None:
      (name, status, autostart, snodes) = fields

      if snodes == "-":
Iustin Pop's avatar
Iustin Pop committed
287
        continue
288
289
290
291
292

      for node in with_secondaries:
        if node in snodes.split(','):
          break
      else:
Iustin Pop's avatar
Iustin Pop committed
293
294
        continue

295
296
297
298
    else:
      (name, status, autostart) = fields

    instances.append(Instance(name, status, autostart != "no"))
Iustin Pop's avatar
Iustin Pop committed
299

300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
  return instances


def GetNodeBootIDs():
  """Get a dict mapping nodes to boot IDs.

  """
  cmd = ['gnt-node', 'list', '--lock-retries=15', '--no-headers',
         '--separator=:', '-o', 'name,bootid']

  ids = {}
  for fields in _RunListCmd(cmd):
    (name, bootid) = fields
    ids[name] = bootid

  return ids
Iustin Pop's avatar
Iustin Pop committed
316
317
318
319


class Message(object):
  """Encapsulation of a notice or error message.
320

Iustin Pop's avatar
Iustin Pop committed
321
322
323
324
325
326
327
328
329
330
  """
  def __init__(self, level, msg):
    self.level = level
    self.msg = msg
    self.when = time.time()

  def __str__(self):
    return self.level + ' ' + time.ctime(self.when) + '\n' + Indent(self.msg)


331
class Watcher(object):
Iustin Pop's avatar
Iustin Pop committed
332
333
334
335
336
  """Encapsulate the logic for restarting erronously halted virtual machines.

  The calling program should periodically instantiate me and call Run().
  This will traverse the list of instances, and make up to MAXTRIES attempts
  to restart machines that are down.
337

Iustin Pop's avatar
Iustin Pop committed
338
339
  """
  def __init__(self):
340
341
    sstore = ssconf.SimpleStore()
    master = sstore.GetMasterNode()
342
    if master != utils.HostInfo().name:
343
      raise NotMasterError("This is not the master node")
344
345
    self.instances = GetInstanceList()
    self.bootids = GetNodeBootIDs()
Iustin Pop's avatar
Iustin Pop committed
346
347
348
    self.messages = []

  def Run(self):
349
350
351
352
353
354
355
    notepad = WatcherState()
    self.CheckInstances(notepad)
    self.CheckDisks(notepad)
    notepad.Save()

  def CheckDisks(self, notepad):
    """Check all nodes for restarted ones.
356

Iustin Pop's avatar
Iustin Pop committed
357
    """
358
359
360
361
362
363
364
365
366
367
368
369
    check_nodes = []
    for name, id in self.bootids.iteritems():
      old = notepad.GetNodeBootID(name)
      if old != id:
        # Node's boot ID has changed, proably through a reboot.
        check_nodes.append(name)

    if check_nodes:
      # Activate disks for all instances with any of the checked nodes as a
      # secondary node.
      for instance in GetInstanceList(with_secondaries=check_nodes):
        try:
370
371
          self.messages.append(Message(NOTICE, ("Activating disks for %s." %
                                                instance.name)))
372
373
374
375
376
377
378
          instance.ActivateDisks()
        except Error, x:
          self.messages.append(Message(ERROR, str(x)))

      # Keep changed boot IDs
      for name in check_nodes:
        notepad.SetNodeBootID(name, self.bootids[name])
Iustin Pop's avatar
Iustin Pop committed
379

380
381
382
383
  def CheckInstances(self, notepad):
    """Make a pass over the list of instances, restarting downed ones.

    """
Iustin Pop's avatar
Iustin Pop committed
384
    for instance in self.instances:
385
386
387
388
      # Don't care about manually stopped instances
      if not instance.autostart:
        continue

Iustin Pop's avatar
Iustin Pop committed
389
      if instance.state in BAD_STATES:
390
        n = notepad.NumberOfRestartAttempts(instance)
Iustin Pop's avatar
Iustin Pop committed
391
392
393
394
395
396
397

        if n > MAXTRIES:
          # stay quiet.
          continue
        elif n < MAXTRIES:
          last = " (Attempt #%d)" % (n + 1)
        else:
398
          notepad.RecordRestartAttempt(instance)
Iustin Pop's avatar
Iustin Pop committed
399
400
401
402
403
          self.messages.append(Message(ERROR, "Could not restart %s for %d"
                                       " times, giving up..." %
                                       (instance.name, MAXTRIES)))
          continue
        try:
404
405
          self.messages.append(Message(NOTICE, ("Restarting %s%s." %
                                                (instance.name, last))))
Iustin Pop's avatar
Iustin Pop committed
406
407
408
409
          instance.Restart()
        except Error, x:
          self.messages.append(Message(ERROR, str(x)))

410
        notepad.RecordRestartAttempt(instance)
Iustin Pop's avatar
Iustin Pop committed
411
      elif instance.state in HELPLESS_STATES:
412
413
        if notepad.NumberOfRestartAttempts(instance):
          notepad.RemoveInstance(instance)
Iustin Pop's avatar
Iustin Pop committed
414
      else:
415
416
        if notepad.NumberOfRestartAttempts(instance):
          notepad.RemoveInstance(instance)
417
          msg = Message(NOTICE, "Restart of %s succeeded." % instance.name)
Iustin Pop's avatar
Iustin Pop committed
418
419
420
          self.messages.append(msg)

  def WriteReport(self, logfile):
421
    """Log all messages to file.
Iustin Pop's avatar
Iustin Pop committed
422
423
424

    Args:
      logfile: file object open for writing (the log file)
425

Iustin Pop's avatar
Iustin Pop committed
426
427
428
429
430
431
432
433
434
435
436
437
438
439
440
441
442
443
444
445
446
447
448
449
450
451
452
453
454
455
456
    """
    for msg in self.messages:
      print >> logfile, str(msg)


def ParseOptions():
  """Parse the command line options.

  Returns:
    (options, args) as from OptionParser.parse_args()

  """
  parser = OptionParser(description="Ganeti cluster watcher",
                        usage="%prog [-d]",
                        version="%%prog (ganeti) %s" %
                        constants.RELEASE_VERSION)

  parser.add_option("-d", "--debug", dest="debug",
                    help="Don't redirect messages to the log file",
                    default=False, action="store_true")
  options, args = parser.parse_args()
  return options, args


def main():
  """Main function.

  """
  options, args = ParseOptions()

  if not options.debug:
457
    sys.stderr = sys.stdout = open(constants.LOG_WATCHER, 'a')
Iustin Pop's avatar
Iustin Pop committed
458
459

  try:
460
461
462
463
464
    try:
      watcher = Watcher()
    except errors.ConfigurationError:
      # Just exit if there's no configuration
      sys.exit(constants.EXIT_SUCCESS)
465
466
    watcher.Run()
    watcher.WriteReport(sys.stdout)
467
468
469
470
  except NotMasterError:
    if options.debug:
      sys.stderr.write("Not master, exiting.\n")
    sys.exit(constants.EXIT_NOTMASTER)
471
472
473
  except errors.ResolverError, err:
    sys.stderr.write("Cannot resolve hostname '%s', exiting.\n" % err.args[0])
    sys.exit(constants.EXIT_NODESETUP_ERROR)
Iustin Pop's avatar
Iustin Pop committed
474
475
476
  except Error, err:
    print err

477

Iustin Pop's avatar
Iustin Pop committed
478
479
if __name__ == '__main__':
  main()