ganeti-watcher 9.39 KB
Newer Older
Iustin Pop's avatar
Iustin Pop committed
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
#!/usr/bin/python
#

# Copyright (C) 2006, 2007 Google Inc.
#
# This program is free software; you can redistribute it and/or modify
# it under the terms of the GNU General Public License as published by
# the Free Software Foundation; either version 2 of the License, or
# (at your option) any later version.
#
# This program is distributed in the hope that it will be useful, but
# WITHOUT ANY WARRANTY; without even the implied warranty of
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE.  See the GNU
# General Public License for more details.
#
# You should have received a copy of the GNU General Public License
# along with this program; if not, write to the Free Software
# Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA
# 02110-1301, USA.


"""Tool to restart erronously downed virtual machines.

This program and set of classes implement a watchdog to restart
virtual machines in a Ganeti cluster that have crashed or been killed
by a node reboot.  Run from cron or similar.
"""


LOGFILE = '/var/log/ganeti/watcher.log'
MAXTRIES = 5
BAD_STATES = ['stopped']
HELPLESS_STATES = ['(node down)']
NOTICE = 'NOTICE'
ERROR = 'ERROR'

import os
import sys
import time
import fcntl
import errno
from optparse import OptionParser


from ganeti import utils
from ganeti import constants
47
from ganeti import ssconf
48
from ganeti import errors
Iustin Pop's avatar
Iustin Pop committed
49
50
51
52


class Error(Exception):
  """Generic custom error class."""
53
54
55
56


class NotMasterError(Error):
  """Exception raised when this host is not the master."""
Iustin Pop's avatar
Iustin Pop committed
57
58
59
60
61
62
63
64


def Indent(s, prefix='| '):
  """Indent a piece of text with a given prefix before each line.

  Args:
    s: The string to indent
    prefix: The string to prepend each line.
65

Iustin Pop's avatar
Iustin Pop committed
66
67
68
69
70
71
72
73
74
75
76
  """
  return "%s%s\n" % (prefix, ('\n' + prefix).join(s.splitlines()))


def DoCmd(cmd):
  """Run a shell command.

  Args:
    cmd: the command to run.

  Raises CommandError with verbose commentary on error.
77

Iustin Pop's avatar
Iustin Pop committed
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
  """
  res = utils.RunCmd(cmd)

  if res.failed:
    raise Error("Command %s failed:\n%s\nstdout:\n%sstderr:\n%s" %
                (repr(cmd),
                 Indent(res.fail_reason),
                 Indent(res.stdout),
                 Indent(res.stderr)))

  return res


class RestarterState(object):
  """Interface to a state file recording restart attempts.

  Methods:
    Open(): open, lock, read and parse the file.
            Raises StandardError on lock contention.

    NumberOfAttempts(name): returns the number of times in succession
                            a restart has been attempted of the named instance.

    RecordAttempt(name, when): records one restart attempt of name at
                               time in when.

    Remove(name): remove record given by name, if exists.

    Save(name): saves all records to file, releases lock and closes file.
107

Iustin Pop's avatar
Iustin Pop committed
108
109
110
111
112
113
114
115
116
117
118
119
  """
  def __init__(self):
    # The two-step dance below is necessary to allow both opening existing
    # file read/write and creating if not existing.  Vanilla open will truncate
    # an existing file -or- allow creating if not existing.
    f = os.open(constants.WATCHER_STATEFILE, os.O_RDWR | os.O_CREAT)
    f = os.fdopen(f, 'w+')

    try:
      fcntl.flock(f.fileno(), fcntl.LOCK_EX|fcntl.LOCK_NB)
    except IOError, x:
      if x.errno == errno.EAGAIN:
120
        raise StandardError("State file already locked")
Iustin Pop's avatar
Iustin Pop committed
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
      raise

    self.statefile = f
    self.inst_map = {}

    for line in f:
      name, when, count = line.rstrip().split(':')

      when = int(when)
      count = int(count)

      self.inst_map[name] = (when, count)

  def NumberOfAttempts(self, instance):
    """Returns number of previous restart attempts.

    Args:
      instance - the instance to look up.
139

Iustin Pop's avatar
Iustin Pop committed
140
141
142
143
144
145
146
147
148
149
150
151
152
    """
    assert self.statefile

    if instance.name in self.inst_map:
      return self.inst_map[instance.name][1]

    return 0

  def RecordAttempt(self, instance):
    """Record a restart attempt.

    Args:
      instance - the instance being restarted
153

Iustin Pop's avatar
Iustin Pop committed
154
155
156
157
158
159
160
161
    """
    assert self.statefile

    when = time.time()

    self.inst_map[instance.name] = (when, 1 + self.NumberOfAttempts(instance))

  def Remove(self, instance):
162
    """Update state to reflect that a machine is running, i.e. remove record.
Iustin Pop's avatar
Iustin Pop committed
163
164
165
166

    Args:
      instance - the instance to remove from books

167
168
    This method removes the record for a named instance.

Iustin Pop's avatar
Iustin Pop committed
169
170
171
172
173
174
175
176
    """
    assert self.statefile

    if instance.name in self.inst_map:
      del self.inst_map[instance.name]

  def Save(self):
    """Save records to file, then unlock and close file.
177

Iustin Pop's avatar
Iustin Pop committed
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
    """
    assert self.statefile

    self.statefile.seek(0)
    self.statefile.truncate()

    for name in self.inst_map:
      print >> self.statefile, "%s:%d:%d" % ((name,) + self.inst_map[name])

    fcntl.flock(self.statefile.fileno(), fcntl.LOCK_UN)

    self.statefile.close()
    self.statefile = None


class Instance(object):
  """Abstraction for a Virtual Machine instance.

  Methods:
    Restart(): issue a command to restart the represented machine.
198

Iustin Pop's avatar
Iustin Pop committed
199
200
201
202
203
204
  """
  def __init__(self, name, state):
    self.name = name
    self.state = state

  def Restart(self):
205
206
207
208
209
210
    """Encapsulates the start of an instance.

    This is currently done using the command line interface and not
    the Ganeti modules.

    """
Iustin Pop's avatar
Iustin Pop committed
211
212
213
214
215
    DoCmd(['gnt-instance', 'startup', '--lock-retries=15', self.name])


class InstanceList(object):
  """The set of Virtual Machine instances on a cluster.
216

Iustin Pop's avatar
Iustin Pop committed
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
  """
  cmd = ['gnt-instance', 'list', '--lock-retries=15',
         '-o', 'name,admin_state,oper_state', '--no-headers', '--separator=:']

  def __init__(self):
    res = DoCmd(self.cmd)

    lines = res.stdout.splitlines()

    self.instances = []
    for line in lines:
      fields = [fld.strip() for fld in line.split(':')]

      if len(fields) != 3:
        continue
      if fields[1] == "no": #no autostart, we don't care about this instance
        continue
      name, status = fields[0], fields[2]

      self.instances.append(Instance(name, status))

  def __iter__(self):
    return self.instances.__iter__()


class Message(object):
  """Encapsulation of a notice or error message.
244

Iustin Pop's avatar
Iustin Pop committed
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
  """
  def __init__(self, level, msg):
    self.level = level
    self.msg = msg
    self.when = time.time()

  def __str__(self):
    return self.level + ' ' + time.ctime(self.when) + '\n' + Indent(self.msg)


class Restarter(object):
  """Encapsulate the logic for restarting erronously halted virtual machines.

  The calling program should periodically instantiate me and call Run().
  This will traverse the list of instances, and make up to MAXTRIES attempts
  to restart machines that are down.
261

Iustin Pop's avatar
Iustin Pop committed
262
263
  """
  def __init__(self):
264
265
    sstore = ssconf.SimpleStore()
    master = sstore.GetMasterNode()
266
    if master != utils.HostInfo().name:
267
      raise NotMasterError("This is not the master node")
Iustin Pop's avatar
Iustin Pop committed
268
269
270
271
272
    self.instances = InstanceList()
    self.messages = []

  def Run(self):
    """Make a pass over the list of instances, restarting downed ones.
273

Iustin Pop's avatar
Iustin Pop committed
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
    """
    notepad = RestarterState()

    for instance in self.instances:
      if instance.state in BAD_STATES:
        n = notepad.NumberOfAttempts(instance)

        if n > MAXTRIES:
          # stay quiet.
          continue
        elif n < MAXTRIES:
          last = " (Attempt #%d)" % (n + 1)
        else:
          notepad.RecordAttempt(instance)
          self.messages.append(Message(ERROR, "Could not restart %s for %d"
                                       " times, giving up..." %
                                       (instance.name, MAXTRIES)))
          continue
        try:
          self.messages.append(Message(NOTICE,
                                       "Restarting %s%s." %
                                       (instance.name, last)))
          instance.Restart()
        except Error, x:
          self.messages.append(Message(ERROR, str(x)))

        notepad.RecordAttempt(instance)
      elif instance.state in HELPLESS_STATES:
        if notepad.NumberOfAttempts(instance):
          notepad.Remove(instance)
      else:
        if notepad.NumberOfAttempts(instance):
          notepad.Remove(instance)
          msg = Message(NOTICE,
                        "Restart of %s succeeded." % instance.name)
          self.messages.append(msg)

    notepad.Save()

  def WriteReport(self, logfile):
314
    """Log all messages to file.
Iustin Pop's avatar
Iustin Pop committed
315
316
317

    Args:
      logfile: file object open for writing (the log file)
318

Iustin Pop's avatar
Iustin Pop committed
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
    """
    for msg in self.messages:
      print >> logfile, str(msg)


def ParseOptions():
  """Parse the command line options.

  Returns:
    (options, args) as from OptionParser.parse_args()

  """
  parser = OptionParser(description="Ganeti cluster watcher",
                        usage="%prog [-d]",
                        version="%%prog (ganeti) %s" %
                        constants.RELEASE_VERSION)

  parser.add_option("-d", "--debug", dest="debug",
                    help="Don't redirect messages to the log file",
                    default=False, action="store_true")
  options, args = parser.parse_args()
  return options, args


def main():
  """Main function.

  """
  options, args = ParseOptions()

  if not options.debug:
    sys.stderr = sys.stdout = open(LOGFILE, 'a')

  try:
    restarter = Restarter()
    restarter.Run()
    restarter.WriteReport(sys.stdout)
356
357
358
359
  except NotMasterError:
    if options.debug:
      sys.stderr.write("Not master, exiting.\n")
    sys.exit(constants.EXIT_NOTMASTER)
360
361
362
  except errors.ResolverError, err:
    sys.stderr.write("Cannot resolve hostname '%s', exiting.\n" % err.args[0])
    sys.exit(constants.EXIT_NODESETUP_ERROR)
Iustin Pop's avatar
Iustin Pop committed
363
364
365
366
367
  except Error, err:
    print err

if __name__ == '__main__':
  main()