added two new sequences to rins a node if not installed.

[monitor.git] / bootman.py
diff --git a/bootman.py b/bootman.py

index b9a161f..c26335f 100755 (executable)
--- a/bootman.py
+++ b/bootman.py
@@ -7,7 +7,7 @@ api = plc.getAuthAPI()
  
  import sys
  import os
-import policy
+import const
  
  from getsshkeys import SSHKnownHosts
  
@@ -24,7 +24,7 @@ from unified_model import *
  from emailTxt import mailtxt
  from nodeconfig import network_config_to_str
  import traceback
-import monitorconfig
+import config
  
  import signal
  class Sopen(subprocess.Popen):
@@ -34,9 +34,12 @@ class Sopen(subprocess.Popen):
  #from Rpyc import SocketConnection, Async
  from Rpyc import SocketConnection, Async
  from Rpyc.Utils import *
+fb = None
  
  def get_fbnode(node):
-       fb = database.dbLoad("findbad")
+       global fb
+       if fb is None:
+               fb = database.dbLoad("findbad")
         fbnode = fb['nodes'][node]['values']
         return fbnode
  
@@ -204,7 +207,7 @@ class PlanetLabSession:
                 args['port'] = self.port
                 args['user'] = 'root'
                 args['hostname'] = self.node
-               args['monitordir'] = monitorconfig.MONITOR_SCRIPT_ROOT
+               args['monitordir'] = config.MONITOR_SCRIPT_ROOT
                 ssh_port = 22
  
                 if self.nosetup:
@@ -321,7 +324,8 @@ def reboot(hostname, config=None, forced_action=None):
                                                         mailtxt.newbootcd_one[1] % args, True, db='bootcd_persistmessages')
  
                 loginbase = plc.siteId(hostname)
-               m.send([policy.PIEMAIL % loginbase, policy.TECHEMAIL % loginbase])
+               emails = plc.getTechEmails(loginbase)
+               m.send(emails) 
  
                 print "\tDisabling %s due to out-of-date BOOTCD" % hostname
                 api.UpdateNode(hostname, {'boot_state' : 'disable'})
@@ -334,6 +338,8 @@ def reboot(hostname, config=None, forced_action=None):
         try:
                 k = SSHKnownHosts(); k.update(node); k.write(); del k
         except:
+               from nodecommon import email_exception
+               email_exception()
                 print traceback.print_exc()
                 return False
  
@@ -343,8 +349,11 @@ def reboot(hostname, config=None, forced_action=None):
                 else:
                         session = PlanetLabSession(node, config.nosetup, config.verbose)
         except Exception, e:
-               print "ERROR setting up session for %s" % hostname
+               msg = "ERROR setting up session for %s" % hostname
+               print msg
                 print traceback.print_exc()
+               from nodecommon import email_exception
+               email_exception(msg)
                 print e
                 return False
  
@@ -358,8 +367,9 @@ def reboot(hostname, config=None, forced_action=None):
                         conn = session.get_connection(config)
                 except:
                         print traceback.print_exc()
+                       from nodecommon import email_exception
+                       email_exception()
                         return False
-                       
  
         if forced_action == "reboot":
                 conn.restart_node('rins')
@@ -400,25 +410,34 @@ def reboot(hostname, config=None, forced_action=None):
                         ('ccisserror' , 'cciss: cmd \w+ has CHECK CONDITION  byte \w+ = \w+'),
  
                         ('buffererror', 'Buffer I/O error on device dm-\d, logical block \d+'),
+
+                       ('hdaseekerror', 'hda: dma_intr: status=0x\d+ { DriveReady SeekComplete Error }'),
+                       ('hdacorrecterror', 'hda: dma_intr: error=0x\d+ { UncorrectableError }, LBAsect=\d+, sector=\d+'),
+
                         ('atareadyerror'   , 'ata\d+: status=0x\d+ { DriveReady SeekComplete Error }'),
                         ('atacorrecterror' , 'ata\d+: error=0x\d+ { UncorrectableError }'),
+
                         ('sdXerror'   , 'sd\w: Current: sense key: Medium Error'),
                         ('ext3error'   , 'EXT3-fs error (device dm-\d+): ext3_find_entry: reading directory #\d+ offset \d+'),
+
                         ('floppytimeout','floppy0: floppy timeout called'),
                         ('floppyerror',  'end_request: I/O error, dev fd\w+, sector \d+'),
  
+                       # hda: dma_intr: status=0x51 { DriveReady SeekComplete Error }
+                       # hda: dma_intr: error=0x40 { UncorrectableError }, LBAsect=23331263, sector=23331263
+
                         # floppy0: floppy timeout called
                         # end_request: I/O error, dev fd0, sector 0
  
-                       #Buffer I/O error on device dm-2, logical block 8888896
-                       #ata1: status=0x51 { DriveReady SeekComplete Error }
-                       #ata1: error=0x40 { UncorrectableError }
-                       #SCSI error : <0 0 0 0> return code = 0x8000002
-                       #sda: Current: sense key: Medium Error
+                       # Buffer I/O error on device dm-2, logical block 8888896
+                       # ata1: status=0x51 { DriveReady SeekComplete Error }
+                       # ata1: error=0x40 { UncorrectableError }
+                       # SCSI error : <0 0 0 0> return code = 0x8000002
+                       # sda: Current: sense key: Medium Error
                         #       Additional sense: Unrecovered read error - auto reallocate failed
  
-                       #SCSI error : <0 2 0 0> return code = 0x40001
-                       #end_request: I/O error, dev sda, sector 572489600
+                       # SCSI error : <0 2 0 0> return code = 0x40001
+                       # end_request: I/O error, dev sda, sector 572489600
                 ]
                 id = index_to_id(steps, child.expect( steps_to_list(steps) + [ pexpect.EOF ]))
                 sequence.append(id)
@@ -444,7 +463,8 @@ def reboot(hostname, config=None, forced_action=None):
                                                                                  mailtxt.baddisk[1] % args, True, db='hardware_persistmessages')
  
                         loginbase = plc.siteId(hostname)
-                       m.send([policy.PIEMAIL % loginbase, policy.TECHEMAIL % loginbase])
+                       emails = plc.getTechEmails(loginbase)
+                       m.send(emails) 
                         conn.set_nodestate('disable')
                         return False
  
@@ -503,6 +523,7 @@ def reboot(hostname, config=None, forced_action=None):
                         ('hardwarerequirefail' , 'Hardware requirements not met'),
                         ('mkfsfail'         , 'while running: Running mkfs.ext2 -q  -m 0 -j /dev/planetlab/vservers failed'),
                         ('nofilereference', "No such file or directory: '/tmp/mnt/sysimg//vservers/.vref/planetlab-f8-i386/etc/hosts'"),
+                       ('kernelcopyfail', "cp: cannot stat `/tmp/mnt/sysimg/boot/kernel-boot': No such file or directory"),
                         ('chrootfail'   , 'Running chroot /tmp/mnt/sysimg'),
                         ('modulefail'   , 'Unable to get list of system modules'),
                         ('writeerror'   , 'write error: No space left on device'),
@@ -530,11 +551,11 @@ def reboot(hostname, config=None, forced_action=None):
         #  By using the sequence identifier, we guarantee that there will be no
         #  frequent loops.  I'm guessing there is a better way to track loops,
         #  though.
-       if not config.force and pflags.getRecentFlag(s):
-               pflags.setRecentFlag(s)
-               pflags.save() 
-               print "... flag is set or it has already run recently. Skipping %s" % node
-               return True
+       #if not config.force and pflags.getRecentFlag(s):
+       #       pflags.setRecentFlag(s)
+       #       pflags.save() 
+       #       print "... flag is set or it has already run recently. Skipping %s" % node
+       #       return True
  
         sequences = {}
  
@@ -572,7 +593,14 @@ def reboot(hostname, config=None, forced_action=None):
                         "bminit-cfg-auth-getplc-update-installinit-validate-rebuildinitrd-netcfg-update3-implementerror-nofilereference-update-debug-done",
                         "bminit-cfg-auth-getplc-update-hardware-installinit-installdisk-exception-mkfsfail-update-debug-done",
                         "bminit-cfg-auth-getplc-installinit-validate-rebuildinitrd-exception-chrootfail-update-debug-done",
+                       "bminit-cfg-auth-getplc-update-hardware-installinit-installdisk-installbootfs-installcfg-installstop-update-installinit-validate-rebuildinitrd-netcfg-disk-update4-update3-update3-kernelcopyfail-exception-update-debug-done",
+                       "bminit-cfg-auth-getplc-hardware-installinit-installdisk-installbootfs-installcfg-installstop-update-installinit-validate-rebuildinitrd-netcfg-disk-update4-update3-update3-kernelcopyfail-exception-update-debug-done",
                         "bminit-cfg-auth-getplc-installinit-validate-exception-noinstall-update-debug-done",
+                       # actual solution appears to involve removing the bad files, and
+                       # continually trying to boot the node.
+                       "bminit-cfg-auth-getplc-update-installinit-validate-rebuildinitrd-netcfg-disk-update4-update3-update3-implementerror-update-debug-done",
+                       "bminit-cfg-auth-getplc-installinit-validate-exception-bmexceptmount-exception-noinstall-update-debug-done",
+                       "bminit-cfg-auth-getplc-update-installinit-validate-exception-bmexceptmount-exception-noinstall-update-debug-done",
                         ]:
                 sequences.update({n : "restart_bootmanager_rins"})
  
@@ -601,16 +629,20 @@ def reboot(hostname, config=None, forced_action=None):
                          "bminit-cfg-auth-implementerror-bootcheckfail-update-implementerror-bootupdatefail-done",
                          "bminit-cfg-auth-getplc-update-installinit-validate-rebuildinitrd-netcfg-update3-implementerror-nospace-update-debug-done",
                          "bminit-cfg-auth-getplc-hardware-installinit-installdisk-installbootfs-exception-downloadfail-update-debug-done",
+                        "bminit-cfg-auth-getplc-update-installinit-validate-implementerror-update-debug-done",
                          ]:
                 sequences.update({n: "restart_node_boot"})
  
         # update_node_config_email
         for n in ["bminit-cfg-exception-nocfg-update-bootupdatefail-nonode-debug-done",
-                       "bminit-cfg-exception-update-bootupdatefail-nonode-debug-done",
+                         "bminit-cfg-exception-update-bootupdatefail-nonode-debug-done",
+                         "bminit-cfg-auth-bootcheckfail-nonode-exception-update-bootupdatefail-nonode-debug-done",
                         ]:
                 sequences.update({n : "update_node_config_email"})
  
-       for n in [ "bminit-cfg-exception-nodehostname-update-debug-done", ]:
+       for n in [ "bminit-cfg-exception-nodehostname-update-debug-done", 
+                          "bminit-cfg-update-exception-nodehostname-update-debug-done", 
+                       ]:
                 sequences.update({n : "nodenetwork_email"})
  
         # update_bootcd_email
@@ -634,7 +666,11 @@ def reboot(hostname, config=None, forced_action=None):
         sequences.update({"bminit-cfg-auth-getplc-update-hardware-exception-hardwarerequirefail-update-debug-done" : "broken_hardware_email"})
  
         # bad_dns_email
-       sequences.update({"bminit-cfg-update-implementerror-bootupdatefail-dnserror-update-implementerror-bootupdatefail-dnserror-done" : "bad_dns_email"})
+       for n in [ 
+        "bminit-cfg-update-implementerror-bootupdatefail-dnserror-update-implementerror-bootupdatefail-dnserror-done",
+               "bminit-cfg-auth-implementerror-bootcheckfail-dnserror-update-implementerror-bootupdatefail-dnserror-done",
+               ]:
+               sequences.update( { n : "bad_dns_email"})
  
         flag_set = True
  
@@ -650,7 +686,7 @@ def reboot(hostname, config=None, forced_action=None):
                 m = PersistMessage(hostname, mailtxt.unknownsequence[0] % args,
                                                                          mailtxt.unknownsequence[1] % args, False, db='unknown_persistmessages')
                 m.reset()
-               m.send(['monitor-list@lists.planet-lab.org'])
+               m.send([config.cc_email]) 
  
                 conn.restart_bootmanager('boot')
  
@@ -688,7 +724,7 @@ def reboot(hostname, config=None, forced_action=None):
                         m = PersistMessage(hostname, "Suspicous error from BootManager on %s" % args,
                                                                                  mailtxt.unknownsequence[1] % args, False, db='suspect_persistmessages')
                         m.reset()
-                       m.send(['monitor-list@lists.planet-lab.org'])
+                       m.send([config.cc_email]) 
  
                         conn.restart_bootmanager('boot')
  
@@ -699,7 +735,8 @@ def reboot(hostname, config=None, forced_action=None):
                         m = PersistMessage(hostname,  mailtxt.plnode_cfg[0] % args,  mailtxt.plnode_cfg[1] % args, 
                                                                 True, db='nodeid_persistmessages')
                         loginbase = plc.siteId(hostname)
-                       m.send([policy.PIEMAIL % loginbase, policy.TECHEMAIL % loginbase])
+                       emails = plc.getTechEmails(loginbase)
+                       m.send(emails) 
                         conn.dump_plconf_file()
                         conn.set_nodestate('disable')
  
@@ -708,10 +745,11 @@ def reboot(hostname, config=None, forced_action=None):
                         args = {}
                         args['hostname'] = hostname
                         args['bmlog'] = conn.get_bootmanager_log().read()
-                       m = PersistMessage(hostname,  mailtxt.plnode_network[0] % args,  mailtxt.plnode_cfg[1] % args, 
+                       m = PersistMessage(hostname,  mailtxt.plnode_cfg[0] % args,  mailtxt.plnode_cfg[1] % args, 
                                                                 True, db='nodenet_persistmessages')
                         loginbase = plc.siteId(hostname)
-                       m.send([policy.PIEMAIL % loginbase, policy.TECHEMAIL % loginbase])
+                       emails = plc.getTechEmails(loginbase)
+                       m.send(emails) 
                         conn.dump_plconf_file()
                         conn.set_nodestate('disable')
  
@@ -726,7 +764,8 @@ def reboot(hostname, config=None, forced_action=None):
                                                                 mailtxt.newalphacd_one[1] % args, True, db='bootcd_persistmessages')
  
                         loginbase = plc.siteId(hostname)
-                       m.send([policy.PIEMAIL % loginbase, policy.TECHEMAIL % loginbase])
+                       emails = plc.getTechEmails(loginbase)
+                       m.send(emails) 
  
                         print "\tDisabling %s due to out-of-date BOOTCD" % hostname
                         conn.set_nodestate('disable')
@@ -744,7 +783,8 @@ def reboot(hostname, config=None, forced_action=None):
                                                                                  mailtxt.baddisk[1] % args, True, db='hardware_persistmessages')
  
                         loginbase = plc.siteId(hostname)
-                       m.send([policy.PIEMAIL % loginbase, policy.TECHEMAIL % loginbase])
+                       emails = plc.getTechEmails(loginbase)
+                       m.send(emails) 
                         conn.set_nodestate('disable')
  
                 elif sequences[s] == "update_hardware_email":
@@ -756,7 +796,8 @@ def reboot(hostname, config=None, forced_action=None):
                                                                                  mailtxt.minimalhardware[1] % args, True, db='minhardware_persistmessages')
  
                         loginbase = plc.siteId(hostname)
-                       m.send([policy.PIEMAIL % loginbase, policy.TECHEMAIL % loginbase])
+                       emails = plc.getTechEmails(loginbase)
+                       m.send(emails) 
                         conn.set_nodestate('disable')
  
                 elif sequences[s] == "bad_dns_email":
@@ -766,6 +807,8 @@ def reboot(hostname, config=None, forced_action=None):
                                 node = api.GetNodes(hostname)[0]
                                 net = api.GetNodeNetworks(node['nodenetwork_ids'])[0]
                         except:
+                               from nodecommon import email_exception
+                               email_exception()
                                 print traceback.print_exc()
                                 # TODO: api error. skip email, b/c all info is not available,
                                 # flag_set will not be recorded.
@@ -779,7 +822,8 @@ def reboot(hostname, config=None, forced_action=None):
                                                                                  mailtxt.baddns[1] % args, True, db='baddns_persistmessages')
  
                         loginbase = plc.siteId(hostname)
-                       m.send([policy.PIEMAIL % loginbase, policy.TECHEMAIL % loginbase])
+                       emails = plc.getTechEmails(loginbase)
+                       m.send(emails) 
                         conn.set_nodestate('disable')
  
         if flag_set: