ganeti-watcher 9.08 KB
Newer Older
Iustin Pop's avatar
Iustin Pop committed
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
#!/usr/bin/python
#

# Copyright (C) 2006, 2007 Google Inc.
#
# This program is free software; you can redistribute it and/or modify
# it under the terms of the GNU General Public License as published by
# the Free Software Foundation; either version 2 of the License, or
# (at your option) any later version.
#
# This program is distributed in the hope that it will be useful, but
# WITHOUT ANY WARRANTY; without even the implied warranty of
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE.  See the GNU
# General Public License for more details.
#
# You should have received a copy of the GNU General Public License
# along with this program; if not, write to the Free Software
# Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA
# 02110-1301, USA.


"""Tool to restart erronously downed virtual machines.

This program and set of classes implement a watchdog to restart
virtual machines in a Ganeti cluster that have crashed or been killed
by a node reboot.  Run from cron or similar.
"""


LOGFILE = '/var/log/ganeti/watcher.log'
MAXTRIES = 5
BAD_STATES = ['stopped']
HELPLESS_STATES = ['(node down)']
NOTICE = 'NOTICE'
ERROR = 'ERROR'

import os
import sys
import time
import fcntl
import errno
42
import socket
Iustin Pop's avatar
Iustin Pop committed
43
44
45
46
47
from optparse import OptionParser


from ganeti import utils
from ganeti import constants
48
from ganeti import ssconf
Iustin Pop's avatar
Iustin Pop committed
49
50
51
52


class Error(Exception):
  """Generic custom error class."""
53
54
55
56


class NotMasterError(Error):
  """Exception raised when this host is not the master."""
Iustin Pop's avatar
Iustin Pop committed
57
58
59
60
61
62
63
64


def Indent(s, prefix='| '):
  """Indent a piece of text with a given prefix before each line.

  Args:
    s: The string to indent
    prefix: The string to prepend each line.
65

Iustin Pop's avatar
Iustin Pop committed
66
67
68
69
70
71
72
73
74
75
76
  """
  return "%s%s\n" % (prefix, ('\n' + prefix).join(s.splitlines()))


def DoCmd(cmd):
  """Run a shell command.

  Args:
    cmd: the command to run.

  Raises CommandError with verbose commentary on error.
77

Iustin Pop's avatar
Iustin Pop committed
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
  """
  res = utils.RunCmd(cmd)

  if res.failed:
    raise Error("Command %s failed:\n%s\nstdout:\n%sstderr:\n%s" %
                (repr(cmd),
                 Indent(res.fail_reason),
                 Indent(res.stdout),
                 Indent(res.stderr)))

  return res


class RestarterState(object):
  """Interface to a state file recording restart attempts.

  Methods:
    Open(): open, lock, read and parse the file.
            Raises StandardError on lock contention.

    NumberOfAttempts(name): returns the number of times in succession
                            a restart has been attempted of the named instance.

    RecordAttempt(name, when): records one restart attempt of name at
                               time in when.

    Remove(name): remove record given by name, if exists.

    Save(name): saves all records to file, releases lock and closes file.
107

Iustin Pop's avatar
Iustin Pop committed
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
  """
  def __init__(self):
    # The two-step dance below is necessary to allow both opening existing
    # file read/write and creating if not existing.  Vanilla open will truncate
    # an existing file -or- allow creating if not existing.
    f = os.open(constants.WATCHER_STATEFILE, os.O_RDWR | os.O_CREAT)
    f = os.fdopen(f, 'w+')

    try:
      fcntl.flock(f.fileno(), fcntl.LOCK_EX|fcntl.LOCK_NB)
    except IOError, x:
      if x.errno == errno.EAGAIN:
        raise StandardError('State file already locked')
      raise

    self.statefile = f
    self.inst_map = {}

    for line in f:
      name, when, count = line.rstrip().split(':')

      when = int(when)
      count = int(count)

      self.inst_map[name] = (when, count)

  def NumberOfAttempts(self, instance):
    """Returns number of previous restart attempts.

    Args:
      instance - the instance to look up.
139

Iustin Pop's avatar
Iustin Pop committed
140
141
142
143
144
145
146
147
148
149
150
151
152
    """
    assert self.statefile

    if instance.name in self.inst_map:
      return self.inst_map[instance.name][1]

    return 0

  def RecordAttempt(self, instance):
    """Record a restart attempt.

    Args:
      instance - the instance being restarted
153

Iustin Pop's avatar
Iustin Pop committed
154
155
156
157
158
159
160
161
    """
    assert self.statefile

    when = time.time()

    self.inst_map[instance.name] = (when, 1 + self.NumberOfAttempts(instance))

  def Remove(self, instance):
162
    """Update state to reflect that a machine is running, i.e. remove record.
Iustin Pop's avatar
Iustin Pop committed
163
164
165
166

    Args:
      instance - the instance to remove from books

167
168
    This method removes the record for a named instance.

Iustin Pop's avatar
Iustin Pop committed
169
170
171
172
173
174
175
176
    """
    assert self.statefile

    if instance.name in self.inst_map:
      del self.inst_map[instance.name]

  def Save(self):
    """Save records to file, then unlock and close file.
177

Iustin Pop's avatar
Iustin Pop committed
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
    """
    assert self.statefile

    self.statefile.seek(0)
    self.statefile.truncate()

    for name in self.inst_map:
      print >> self.statefile, "%s:%d:%d" % ((name,) + self.inst_map[name])

    fcntl.flock(self.statefile.fileno(), fcntl.LOCK_UN)

    self.statefile.close()
    self.statefile = None


class Instance(object):
  """Abstraction for a Virtual Machine instance.

  Methods:
    Restart(): issue a command to restart the represented machine.
  """
  def __init__(self, name, state):
    self.name = name
    self.state = state

  def Restart(self):
    DoCmd(['gnt-instance', 'startup', '--lock-retries=15', self.name])


class InstanceList(object):
  """The set of Virtual Machine instances on a cluster.
209

Iustin Pop's avatar
Iustin Pop committed
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
  """
  cmd = ['gnt-instance', 'list', '--lock-retries=15',
         '-o', 'name,admin_state,oper_state', '--no-headers', '--separator=:']

  def __init__(self):
    res = DoCmd(self.cmd)

    lines = res.stdout.splitlines()

    self.instances = []
    for line in lines:
      fields = [fld.strip() for fld in line.split(':')]

      if len(fields) != 3:
        continue
      if fields[1] == "no": #no autostart, we don't care about this instance
        continue
      name, status = fields[0], fields[2]

      self.instances.append(Instance(name, status))

  def __iter__(self):
    return self.instances.__iter__()


class Message(object):
  """Encapsulation of a notice or error message.
237

Iustin Pop's avatar
Iustin Pop committed
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
  """
  def __init__(self, level, msg):
    self.level = level
    self.msg = msg
    self.when = time.time()

  def __str__(self):
    return self.level + ' ' + time.ctime(self.when) + '\n' + Indent(self.msg)


class Restarter(object):
  """Encapsulate the logic for restarting erronously halted virtual machines.

  The calling program should periodically instantiate me and call Run().
  This will traverse the list of instances, and make up to MAXTRIES attempts
  to restart machines that are down.
254

Iustin Pop's avatar
Iustin Pop committed
255
256
  """
  def __init__(self):
257
258
259
260
    sstore = ssconf.SimpleStore()
    master = sstore.GetMasterNode()
    if master != socket.gethostname():
      raise NotMasterError, ("This is not the master node")
Iustin Pop's avatar
Iustin Pop committed
261
262
263
264
265
    self.instances = InstanceList()
    self.messages = []

  def Run(self):
    """Make a pass over the list of instances, restarting downed ones.
266

Iustin Pop's avatar
Iustin Pop committed
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
    """
    notepad = RestarterState()

    for instance in self.instances:
      if instance.state in BAD_STATES:
        n = notepad.NumberOfAttempts(instance)

        if n > MAXTRIES:
          # stay quiet.
          continue
        elif n < MAXTRIES:
          last = " (Attempt #%d)" % (n + 1)
        else:
          notepad.RecordAttempt(instance)
          self.messages.append(Message(ERROR, "Could not restart %s for %d"
                                       " times, giving up..." %
                                       (instance.name, MAXTRIES)))
          continue
        try:
          self.messages.append(Message(NOTICE,
                                       "Restarting %s%s." %
                                       (instance.name, last)))
          instance.Restart()
        except Error, x:
          self.messages.append(Message(ERROR, str(x)))

        notepad.RecordAttempt(instance)
      elif instance.state in HELPLESS_STATES:
        if notepad.NumberOfAttempts(instance):
          notepad.Remove(instance)
      else:
        if notepad.NumberOfAttempts(instance):
          notepad.Remove(instance)
          msg = Message(NOTICE,
                        "Restart of %s succeeded." % instance.name)
          self.messages.append(msg)

    notepad.Save()

  def WriteReport(self, logfile):
307
    """Log all messages to file.
Iustin Pop's avatar
Iustin Pop committed
308
309
310

    Args:
      logfile: file object open for writing (the log file)
311

Iustin Pop's avatar
Iustin Pop committed
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
    """
    for msg in self.messages:
      print >> logfile, str(msg)


def ParseOptions():
  """Parse the command line options.

  Returns:
    (options, args) as from OptionParser.parse_args()

  """
  parser = OptionParser(description="Ganeti cluster watcher",
                        usage="%prog [-d]",
                        version="%%prog (ganeti) %s" %
                        constants.RELEASE_VERSION)

  parser.add_option("-d", "--debug", dest="debug",
                    help="Don't redirect messages to the log file",
                    default=False, action="store_true")
  options, args = parser.parse_args()
  return options, args


def main():
  """Main function.

  """
  options, args = ParseOptions()

  if not options.debug:
    sys.stderr = sys.stdout = open(LOGFILE, 'a')

  try:
    restarter = Restarter()
    restarter.Run()
    restarter.WriteReport(sys.stdout)
349
350
351
352
  except NotMasterError:
    if options.debug:
      sys.stderr.write("Not master, exiting.\n")
    sys.exit(constants.EXIT_NOTMASTER)
Iustin Pop's avatar
Iustin Pop committed
353
354
355
356
357
  except Error, err:
    print err

if __name__ == '__main__':
  main()