#!/usr/whatap/infra/lib/bin/python2.7


import errno, os, time, sys, signal, Queue, threading

# Ensure WHATAP_HOME reflects this launcher's actual install location
# (<install>/conf) when not explicitly set, so the pid file, alive socket and
# all downstream config-path resolution follow a relocated install instead of
# the hardcoded /usr/whatap/infra/conf.
if not os.environ.get("WHATAP_HOME"):
    os.environ["WHATAP_HOME"] = os.path.join(
        os.path.dirname(os.path.abspath(__file__)), "conf")

from whatap.agent.conf import configure as config
import whatap.util.logging_util as logging_util

AliveSockAddr       = os.path.join(os.environ["WHATAP_HOME"], "whatap_infrad.sock")
KEEP_ALIVE_PROTOCOL = 0x5B9B
AGENT_VERSION = "1.4.16"

childPid = 0
# Number of times the watchdog has (re)started the worker child; a rising count
# is a restart storm (data-gap investigation).
childRestartCount = 0

# SERVER-2095: minimum delay between two child starts, and the escalation used
# when children keep dying young. The old backoff (`if childHealthy:
# time.sleep(30)` at the end of watchChildProcess) was unreachable - the loop
# only exits with childHealthy False - so restarts ran at fork speed.
RESTART_MIN_INTERVAL = 10
RESTART_BACKOFF_MAX = 60
SHORT_LIVED_CHILD_SECONDS = 60
lastChildStartTime = 0
shortLivedChildStreak = 0

# SERVER-2107: the watchdog judges the child's death from the process itself
# (Popen.poll every HEALTH_POLL_SECONDS), not from the health socket's EOF. A
# socket can lie - a spawned command holding an inherited copy delays the EOF
# (SERVER-2095), and a child killed before its first frame leaves an EOF that
# names no generation, so its death went unnoticed until the 180s timeout. The
# health pipe now only carries what the child says about itself: PROBLEM
# frames, and their absence (HEALTH_TIMEOUT_SECONDS) for a hung child.
HEALTH_POLL_SECONDS = 10
HEALTH_TIMEOUT_SECONDS = 180
# SERVER-2107: a health connection that says nothing for this long is dropped
# (the child sends a frame every 60s), and each connection has its own reader
# thread, so a silent connection - e.g. a dead child's socket held open by a
# spawned command - can no longer occupy the only reader while the current
# child's connection waits in the backlog.
HEALTH_CONN_IDLE_SECONDS = 120
HEALTH_LISTEN_BACKLOG = 8

# SERVER-2106: on SIGTERM the master used to kill its child and exit 5s later,
# but in between the watchdog saw the child die and forked a replacement, which
# was orphaned (ppid=1) the moment the master exited - and ran forever,
# reporting under the same oid. shuttingDown stops any further fork; childLock
# makes "check the flag, fork, publish childPid" atomic against the signal
# handler, so the pid the handler kills is the last child that will ever exist.
# RLock: a second signal may re-enter the handler while it holds the lock.
shuttingDown = False
childLock = threading.RLock()

# SERVER-2106: the master passes its pid to the child. The child used to watch
# for a change of os.getppid() relative to the value it saw when its watcher
# thread started - a child forked just before the master exited started out
# with ppid=1, saw no change and never exited.
MASTER_PID_ENV = "WHATAP_MASTER_PID"

# SERVER-2106: the child is started with this exact path, and the orphan sweep
# matches it as a ps token, so both must use the same absolute form (resolved
# once, before anything could change the working directory).
EXEC_PATH = os.path.abspath(sys.argv[0])


def setCloseOnExec(sock):
    """Mark a health-pipe socket close-on-exec.

    SERVER-2095: python 2.7 defaults subprocess.Popen to close_fds=False, so
    every command the agent spawns (vmstat, lspv, df, entstat, ...) inherits a
    copy of this socket. A leaked copy keeps the health connection open after
    the owning process is killed, so the master never sees EOF for the dead
    child, stays blocked on that connection while the new child's connect waits
    in the backlog, and finally reads the late EOF as "the current child died" -
    killing an innocent child. That cascade is self-sustaining (restart storm).
    Verified on AIX 7.1 / python 2.7.18: with the leak the EOF after SIGKILL
    arrived 8.0s late (the spawned command's remaining lifetime), without it 0.0s.
    """
    try:
        import fcntl
        flags = fcntl.fcntl(sock.fileno(), fcntl.F_GETFD)
        fcntl.fcntl(sock.fileno(), fcntl.F_SETFD, flags | fcntl.FD_CLOEXEC)
    except Exception, e:
        logging_util.error("whatap_infrad failed to set FD_CLOEXEC on health socket: {}".format(e))

def setPid():
    whatap_home = os.environ.get("WHATAP_HOME")
    if whatap_home:
        filepath = os.path.join(whatap_home,'whatap.pid')
        f = open(filepath,'w')
        f.write(str(os.getpid()))
        f.close()

import platform
import re

def is_fork():
    system = platform.system().lower()
    if system == "sunos" or system == "hp-ux":
        return True

    # Attempt to read the release information from /etc/redhat-release
    try:
        with open("/etc/redhat-release", "r") as file:
            release_info = file.read()
            
            # Check if the release information mentions CentOS and version 5
            if "CentOS" in release_info and re.search(r"release 5(\.\d+)?", release_info):
                return True
    except IOError:
        # Handle error (e.g., file not found) if /etc/redhat-release does not exist
        pass

    # Fallback check using platform module
    dist_name, version, id = platform.linux_distribution(full_distribution_name=0)
    return dist_name.lower() in ["centos", "redhat"] and version.startswith("5")


def setLogger(logfileName):
    conf = config.GetConfig()
    logPath = os.path.join(conf.logPath, logfileName)
    logSize = conf.logSize * 1024 * 1024 #MB
    logBackup = conf.logBackup
    logging_util.initLogger(logPath, isDebug = conf.Debug, logMaxBytes = logSize, logBackupCount = logBackup) 



class _ControlTimeout(Exception):
    """Startup housekeeping ran out of its time budget."""


def _run_bounded(seconds, func):
    """Run func() under a wall-clock bound. Never raises, never blocks.

    Python 2.7 has no timeout anywhere in subprocess, so a shell-out that
    hangs (inittab contention, a stuck filesystem holding /usr/sbin) would
    hang whoever called it - here that is the agent's own startup, before
    setPid(), with SRC believing the agent came up. SIGALRM is the only
    bound available in 2.7 without threads. A timeout is treated exactly
    like any other failure: give up and move on.

    If the alarm cannot be armed (not the main thread, no SIGALRM) func is
    not called at all - best-effort self-healing is worth less than the
    guarantee that startup cannot block.
    """
    def _on_alarm(signum, frame):
        raise _ControlTimeout("timed out after %ss" % seconds)

    prev_remaining = 0
    try:
        prev_handler = signal.signal(signal.SIGALRM, _on_alarm)
    except Exception:
        return
    try:
        prev_remaining = signal.alarm(seconds)
        try:
            func()
        except Exception:
            # _ControlTimeout included: a hang is just another failure.
            pass
    finally:
        try:
            signal.alarm(0)
            signal.signal(signal.SIGALRM, prev_handler)
            # Someone else's alarm was pending; put back what is left of it.
            if prev_remaining:
                signal.alarm(prev_remaining)
        except Exception:
            pass


def _control_startup_housekeeping():
    """Clean up what a control operation may have left behind.

    Nothing in here may prevent the agent from starting: every step is
    guarded on its own AND the whole body is guarded again, because the
    setup around the steps can fail too (os.path.abspath() calls getcwd(),
    which raises when the directory the init script ran from has been
    deleted - an upgrade does exactly that). A late SIGALRM from
    _run_bounded landing between the steps is caught by the outer guard as
    well.
    """
    try:
        install_root = os.path.dirname(os.path.abspath(sys.argv[0]))

        # (1) A hard-link alias survives a killed executor. A leftover link
        #     pins the old inode when a patch replaces the agent binary.
        try:
            import whatap.agent.control.agentstop.alias as control_alias
            _run_bounded(5, lambda: control_alias.remove_stale(install_root))
        except Exception:
            pass

        # (2) AIX: if the agent came up with inittab respawn still off, turn
        #     it back on. Old patch scripts only do stopsrc -> startsrc and
        #     never touch inittab, so that path leaves respawn off and the
        #     agent does not come back after a reboot. A running agent
        #     implies respawn should be on. While STOPped there is no agent
        #     to run this, so a STOP is not undone by it. It stays here, in
        #     front of everything else, rather than moving after startup:
        #     the executor disables respawn and re-checks it once the stop
        #     is confirmed (service_aix.after_stopped), and that ordering
        #     only holds while the agent's own restore happens at the very
        #     start of its startup. Bounded instead of moved.
        try:
            import platform as _platform
            if _platform.system().lower() == "aix":
                from whatap.agent.control.agentstop import service_aix
                _run_bounded(20, service_aix.Adapter().enable_respawn)
        except Exception:
            pass
    except Exception:
        pass

def daemonize(uid, isfork=is_fork()):
    _control_startup_housekeeping()
    if isfork:
        if not hasattr ( os, 'fork'):
            return
        if os.fork():
            os._exit(0)
        os.setsid()
        os.setuid( uid )
        signal.signal(signal.SIGHUP, signal.SIG_IGN)

        if os.fork():
            os._exit(0)
        sys.stdin = open("/dev/null", "r")

    setPid()
    reapOrphanedChildren()

    #deamon
    sys.stdout = open("/dev/null", "w")
    sys.stderr = open("/dev/null", "w")

    import whatap.util.thread_util as thread_util
    logging_util.isDebug = config.GetConfig().Debug
    from datetime import datetime, timedelta
    if config.GetConfig().HouseKeepEnabled:
        childHealthChannel = Queue.Queue()
        thread_util.async(listenChildAlive, childHealthChannel)
        time.sleep(1)
        def watchChildProcess():
            global childPid
            global childRestartCount
            global lastChildStartTime
            global shortLivedChildStreak

            # SERVER-2095 (backoff): hold off before starting the next child so a
            # repeating failure cannot run at fork speed. The delay grows while
            # children keep dying young and resets as soon as one survives.
            if lastChildStartTime:
                backoff = RESTART_MIN_INTERVAL
                if shortLivedChildStreak > 1:
                    backoff = min(RESTART_MIN_INTERVAL * shortLivedChildStreak,
                                  RESTART_BACKOFF_MAX)
                # Clamped to backoff: time.time() is wall clock, so an NTP step
                # backwards would otherwise park the watchdog (and leave the agent
                # with no worker) for as long as the clock moved.
                waitFor = min(backoff, backoff - (time.time() - lastChildStartTime))
                if waitFor > 0:
                    logging_util.info(
                        "whatap_infrad watchdog waiting {:.0f}s before restarting the child "
                        "(short-lived streak={})".format(waitFor, shortLivedChildStreak))
                    time.sleep(waitFor)

            logging_util.debug("whatap.daemonize step -2 start forking child ")
            childProcess = startChildUnlessShuttingDown()
            if childProcess is None:
                return
            lastChildStartTime = time.time()

            childRestartCount += 1
            # Watchdog restart trace: every (re)start of the worker child is a
            # data-gap boundary, so log the new pid + count (rising count = restart
            # storm) explicitly (data-gap investigation).
            logging_util.info("whatap_infrad watchdog started child pid={} start#={}".format(childPid, childRestartCount))
            childHealthy = True
            childStarted = datetime.now()
            lastHealthTime = time.time()

            while childHealthy :
                logging_util.debug("whatap.daemonize step -3.1 wait health ")
                try:
                    report = childHealthChannel.get(timeout = HEALTH_POLL_SECONDS)
                    logging_util.debug("whatap.daemonize step -4 received child health: ", report)
                except Queue.Empty, e:
                    report = None
                verdict, detail, lastHealthTime = judgeChild(
                    childProcess, report, lastHealthTime, time.time())
                uptime = (datetime.now() - childStarted).total_seconds()
                if verdict == "CHILD_EXITED":
                    logging_util.error(
                        "whatap_infrad watchdog restarting child pid={}: reason=CHILD_EXITED rc={} uptime={:.0f}s".format(
                            childProcess.pid, detail, uptime))
                    childHealthy = False
                elif verdict is not None:
                    if verdict == "BAD_HEALTH":
                        logging_util.error(
                            "whatap_infrad watchdog restarting child pid={}: reason=BAD_HEALTH check={} uptime={:.0f}s".format(
                                childProcess.pid, detail, uptime))
                    else:
                        logging_util.error(
                            "whatap_infrad watchdog restarting child pid={}: reason=HEALTH_TIMEOUT "
                            "(no health report for {}s; child hung or keepAlive stalled) uptime={:.0f}s".format(
                                childProcess.pid, HEALTH_TIMEOUT_SECONDS, uptime))
                    try:
                        os.kill(childProcess.pid, signal.SIGKILL)
                    except OSError, e:
                        logging_util.error("whatap_infrad failed to kill child pid={}: {}".format(childProcess.pid, e))
                    childProcess.wait()
                    childHealthy = False
                    logging_util.debug("whatap.daemonize step -8 kill complete ")

                now = datetime.now()
                cooltimeThreshold = timedelta(hours = config.GetConfig().HouseKeepCooltime)
                if config.GetConfig().HouseKeepCooltime < 1:
                    cooltimeThreshold += timedelta(minutes = 1)

                # SERVER-1850: schedule the restart on hour AND minute so many
                # agents can be staggered within an hour instead of all firing on
                # the same hour boundary. Health reports (and therefore this
                # check) arrive only about once a minute, so an exact "minute =="
                # match is unreliable. Instead treat the scheduled time as the
                # start of a window that runs to a few minutes past the end of the
                # configured hour: any tick landing in the window triggers the
                # restart, so a ~60s poll that drifts across an hour (or midnight)
                # boundary (e.g. 23:58 -> 00:00) is still caught. The window is
                # computed in minute-of-day units to avoid datetime edge cases and
                # to wrap cleanly past midnight. The daily cadence is still
                # enforced by cooltimeThreshold (a second restart cannot happen
                # until the child has run at least that long). Default
                # HouseKeepMinute=0 yields a full-hour window, matching the legacy
                # hour-only behavior.
                houseKeepHour = config.GetConfig().HouseKeepHour
                houseKeepMinute = config.GetConfig().HouseKeepMinute
                cooltimePassed = now - childStarted > cooltimeThreshold

                # An out-of-range hour/minute simply disables the scheduled
                # restart (as the legacy "now.hour == HouseKeepHour" test did for
                # a bad hour) rather than crashing the watchdog loop.
                if 0 <= houseKeepHour <= 23 and 0 <= houseKeepMinute <= 59:
                    nowMinuteOfDay = now.hour * 60 + now.minute
                    windowStart = houseKeepHour * 60 + houseKeepMinute
                    windowEnd = (houseKeepHour + 1) * 60 + 5
                    if windowEnd <= 24 * 60:
                        houseKeepDue = windowStart <= nowMinuteOfDay < windowEnd
                    else:
                        # Window runs past midnight (HouseKeepHour == 23): match
                        # either the tail of today or the wrapped head of the day.
                        houseKeepDue = (nowMinuteOfDay >= windowStart
                                        or nowMinuteOfDay < windowEnd - 24 * 60)
                else:
                    houseKeepDue = False

                logging_util.debug("whatap.daemonize step -9  ",
                    childHealthy, cooltimePassed, now.hour, now.minute,
                    houseKeepHour, houseKeepMinute,
                    "kill switch:",
                    childHealthy and cooltimePassed and houseKeepDue)
                if childHealthy and cooltimePassed and houseKeepDue:
                    uptime = (now - childStarted).total_seconds()
                    logging_util.info(
                        "whatap_infrad watchdog restarting child pid={}: reason=HOUSEKEEP "
                        "(scheduled daily restart at {:02d}:{:02d}) uptime={:.0f}s".format(
                            childProcess.pid, houseKeepHour, houseKeepMinute, uptime))
                    os.kill(childProcess.pid, signal.SIGKILL)
                    childProcess.wait()
                    childHealthy = False

            # SERVER-2095: feed the backoff. A child that did not even reach
            # SHORT_LIVED_CHILD_SECONDS is a failing restart, so the next start is
            # delayed further; one that ran longer clears the streak. (This
            # replaces `if childHealthy: time.sleep(30)`, which could never run
            # because the loop above only exits with childHealthy False.)
            if (datetime.now() - childStarted).total_seconds() < SHORT_LIVED_CHILD_SECONDS:
                shortLivedChildStreak += 1
            else:
                shortLivedChildStreak = 0
        def _loop():
            while not shuttingDown:
                watchChildProcess()
            logging_util.info("whatap_infrad watchdog stopped: master is shutting down")
        thread_util.async(_loop)
    else:
        thread_util.async(boot.boot)

def judgeChild(childProcess, report, lastHealthTime, now):
    """One watchdog step: (verdict, detail, lastHealthTime).

    SERVER-2107: verdict is None while the child is fine, "CHILD_EXITED"
    (detail = return code) when the process is gone, "BAD_HEALTH" (detail =
    the failing check) when the child reported PROBLEM, or "HEALTH_TIMEOUT"
    when it has not reported for HEALTH_TIMEOUT_SECONDS. report is a queued
    (pid, ok, reason) frame or None.
    """
    rc = childProcess.poll()
    if rc is not None:
        return "CHILD_EXITED", rc, lastHealthTime
    if report is not None:
        reportedPid, ok, reason = report
        # SERVER-2095: a report names the child it describes. The reader
        # thread can only decide which generation it is reading when it reads;
        # by the time it queues the report the watchdog may already have
        # started the next child, so the generation is checked here too.
        if str(reportedPid) != str(childProcess.pid):
            logging_util.info(
                "whatap_infrad watchdog ignoring health report about child pid={} "
                "(current child pid={})".format(reportedPid, childProcess.pid))
        else:
            lastHealthTime = now
            if not ok:
                return "BAD_HEALTH", reason or "unknown", lastHealthTime
    # Wall clock stepped backwards (NTP): restart the count instead of letting
    # a negative age hide a hung child for as long as the clock moved.
    if now < lastHealthTime:
        lastHealthTime = now
    if now - lastHealthTime >= HEALTH_TIMEOUT_SECONDS:
        return "HEALTH_TIMEOUT", None, lastHealthTime
    return None, None, lastHealthTime

def startChildUnlessShuttingDown():
    """Fork the next child and publish its pid, unless the master is stopping.

    SERVER-2106: runs under childLock so the signal handler either sees this
    child's pid (and kills it) or has already set shuttingDown (and no child is
    started). Returns the Popen object, or None when shutting down.
    """
    global childPid
    with childLock:
        if shuttingDown:
            logging_util.info("whatap_infrad watchdog not starting a child: master is shutting down")
            return None
        childProcess = forkChild()
        childPid = childProcess.pid
        return childProcess

def findOrphanedChildren(psOutput, execPath, myPid):
    """Pids of this install's worker children that have lost their master.

    SERVER-2106: parses `ps -ef`. A worker runs as "<python> <execPath>
    foreground"; one with ppid 1 has no master, and no master will ever adopt
    it. The path must be an exact token (not a substring) so another install
    root, or the whatap_infrad.bak-<ts> files the patch script leaves next to
    the binary, never match - same rule as the HP-UX service script
    (SERVER-1901). A command line ps truncated before "foreground" is simply
    not matched.
    """
    pids = []
    for line in psOutput.splitlines():
        fields = line.split()
        if len(fields) < 4 or fields[2] != "1":
            continue
        try:
            pid = int(fields[1])
        except ValueError:
            continue
        if pid == myPid:
            continue
        for i in range(3, len(fields) - 1):
            if fields[i] == execPath and fields[i + 1] == "foreground":
                pids.append(pid)
                break
    return pids

def reapOrphanedChildren():
    """Kill worker children orphaned by an earlier master (SERVER-2106).

    The fix in the signal handler only helps once it is the code being stopped:
    a patch stops the OLD master, which can still orphan a child running the
    old code - and that child never exits by itself. The new master clears
    them before it starts its own child.
    """
    import whatap.util.process_util as process_util
    try:
        output = process_util.executeCommandArgsWithTimeout(["ps", "-ef"], 10)
        for pid in findOrphanedChildren(output, EXEC_PATH, os.getpid()):
            logging_util.error(
                "whatap_infrad killing orphaned child pid={} left by a previous master".format(pid))
            try:
                os.kill(pid, signal.SIGKILL)
            except OSError, e:
                logging_util.error("whatap_infrad failed to kill orphaned child pid={}: {}".format(pid, e))
    except Exception, e:
        logging_util.error("whatap_infrad orphaned child scan failed: {}".format(e))

def listenChildAlive(badHealthChannel ):
    import socket
    import whatap.util.logging_util as logging_util
    import whatap.util.thread_util as thread_util
    while True:
        logging_util.debug("listenChildAlive step -1")
        if os.path.exists(AliveSockAddr):
            os.unlink(AliveSockAddr)
        alive_sock_path= os.path.split(AliveSockAddr)[0]
        if not os.path.exists(alive_sock_path):
            os.makedirs(alive_sock_path)
        logging_util.debug("listenChildAlive step -2")
        l = socket.socket(socket.AF_UNIX, socket.SOCK_STREAM)
        setCloseOnExec(l)
        l.bind(AliveSockAddr)
        l.listen(HEALTH_LISTEN_BACKLOG)

        while True:
            conn = None
            try:
                conn, _ = l.accept()
                setCloseOnExec(conn)
                conn.settimeout(HEALTH_CONN_IDLE_SECONDS)
            except Exception, e:
                logging_util.debugStack(e)
                if conn:
                    conn.close()
                time.sleep(3)
                continue

            logging_util.debug("listenChildAlive step -6")
            # SERVER-2107: one reader per connection (see HEALTH_CONN_IDLE_SECONDS)
            thread_util.async(keepAlive, conn, badHealthChannel)
            logging_util.debug("listenChildAlive step -7")

def keepAlive(conn, healthChannel ):
    import whatap.util.logging_util as logging_util
    try:
        _keepAlive(conn, healthChannel)
    except Exception, e:

        logging_util.debugStack(e)
        return
    finally:
        if conn :
            conn.close()

def isCurrentChildConnection(connChildPid, cause):
    """Whether a broken health connection says anything about the current child.

    SERVER-2095: a connection can outlive its child (a spawned command holding an
    inherited copy of the socket, a child killed before its first frame), so its
    end-of-stream can arrive long after the watchdog has already started the next
    child. Charging that to the current child kills a healthy process and starts
    the next round of the same mistake. A connection that is not the current
    child's is simply dropped; if the current child really stops reporting, the
    180s health timeout still covers it.
    """
    if connChildPid is not None and str(connChildPid) == str(childPid):
        return True
    logging_util.info(
        "whatap_infrad ignoring health connection failure from {} child pid={} "
        "(current child pid={}): {}".format(
            "an unidentified" if connChildPid is None else "a previous",
            connChildPid or "unknown", childPid, cause))
    return False


def _keepAlive(conn, healthChannel):
    from whatap.io.data_inputx import DataInputX
    import whatap.util.logging_util as logging_util
    from whatap.value.map_value import MapValue

    reader = DataInputX(conn)
    # SERVER-2095: the generation this connection belongs to, taken from the pid
    # the child stamps on every health frame. It stays None until the first frame
    # identifies the owner. A connection only speaks for the child that owns it,
    # so its end-of-stream must never be charged to a different generation.
    connChildPid = None
    while True:
        logging_util.debug("whatap.keepAlive step -1 ")

        # Protocol/parse errors mean a live child is writing a corrupt stream
        # and restart it with a named reason. A read error (EOF, or the idle
        # timeout) is only the end of this connection: SERVER-2107 - whether
        # the child is dead is decided from the process (judgeChild), and a
        # live child that stopped reporting is caught by HEALTH_TIMEOUT.
        try:
            protocol = reader.readShort()
        except Exception, e:
            if not isCurrentChildConnection(connChildPid, e):
                return
            logging_util.info(
                "whatap_infrad health connection of child pid={} ended: {} "
                "(liveness is judged from the process)".format(connChildPid, e))
            return
        if protocol != KEEP_ALIVE_PROTOCOL:
            if not isCurrentChildConnection(connChildPid, "protocol mismatch"):
                return
            logging_util.error("whatap_infrad health pipe protocol mismatch: got={} expected=0x{:x} (corrupt/misframed health stream)".format(protocol, KEEP_ALIVE_PROTOCOL))
            healthChannel.put((connChildPid, False, "health_pipe_protocol_mismatch(got={})".format(protocol)))
            return

        logging_util.debug("whatap.keepAlive step -2 ")
        try:
            msgbytes = reader.readIntBytes(2048)
            datainputx = DataInputX(msgbytes)
            msgmap = MapValue()
            msgmap.read(datainputx)
        except Exception, e:
            if not isCurrentChildConnection(connChildPid, e):
                return
            logging_util.error("whatap_infrad health message parse failed (malformed health frame): {}".format(e))
            healthChannel.put((connChildPid, False, "health_pipe_parse_error({})".format(e)))
            return

        # SERVER-2095: bind this connection to the child that opened it. Frames
        # from a generation the watchdog has already replaced are not evidence
        # about the current child, so the connection is dropped instead.
        framePid = msgmap.getText("Pid")
        if connChildPid is None:
            connChildPid = framePid
        if str(connChildPid) != str(childPid):
            logging_util.info(
                "whatap_infrad discarding health stream from a previous child "
                "pid={} (current child pid={})".format(connChildPid or "unknown", childPid))
            return

        logging_util.debug("whatap.keepAlive step -3 ", msgmap.getText("Health"))
        if msgmap.getText("Health") != "OK":
            healthChannel.put((connChildPid, False, msgmap.getText("Reason") or "unknown"))
            return

        healthChannel.put((connChildPid, True, ""))
        logging_util.debug("whatap.keepAlive step -4 ")

def forkChild():
    import whatap.util.logging_util as logging_util
    import subprocess
    import os,sys

    logging_util.debug("whatap.forkChild step -1 ")
    execPath = EXEC_PATH
    logging_util.debug("whatap.forkChild step -1.1 ", execPath)
    env = dict(os.environ)
    env[MASTER_PID_ENV] = str(os.getpid())
    p = subprocess.Popen((execPath, "foreground"), env=env)
    logging_util.debug("whatap.forkChild step -3 ")

    return p


def keepAliveSender():
    import whatap.util.logging_util as logging_util
    from whatap.io.data_outputx import DataOutputX
    from whatap.value.map_value import MapValue
    import whatap.agent.boot.boot as boot
    import socket

    while True:
        logging_util.debug("whatap.keepAliveSender step -1 ")

        serverconn = None
        while not serverconn:
            logging_util.debug("whatap.keepAliveSender step -1.1 ")
            s = socket.socket(socket.AF_UNIX, socket.SOCK_STREAM)
            # Mark it before connecting: other threads spawn collection commands
            # continuously, and any Popen between socket() and this call would
            # inherit the descriptor (SERVER-2095).
            setCloseOnExec(s)
            try:
                s.connect(AliveSockAddr)
                s.settimeout(30)
            except socket.error, msg:
                logging_util.debug("whatap.keepAliveSender step -2 ", msg)
                time.sleep(3)
                continue

            serverconn = s
        try:
            logging_util.debug("whatap.keepAliveSender step -3 ")
            while True:
                msgmap = MapValue()
                # SERVER-2095: stamp every frame with this child's pid so the
                # master can tell which generation a health connection belongs to.
                msgmap.putString("Pid", str(os.getpid()))
                check, reason = boot.checkHealth()
                logging_util.debug("whatap.keepAliveSender step -4 boot.checkHealth(): ", check, reason)

                if check:
                    msgmap.putString("Health", "OK")
                else:
                    # Carry the failing check to the master so its restart log
                    # names the cause (data-gap investigation).
                    logging_util.error("HealthCheck Fail reason={}".format(reason))
                    msgmap.putString("Health", "PROBLEM")
                    msgmap.putString("Reason", reason)

                dout = DataOutputX()
                dout.writeShort(KEEP_ALIVE_PROTOCOL)
                doutx = DataOutputX()
                msgmap.write(doutx)
                dout.writeIntBytes(doutx.toByteArray())
                try:
                    serverconn.sendall(dout.toByteArray())
                except Exception, e:
                    logging_util.debugStack(e)
                    break
                logging_util.debug("whatap.keepAliveSender step -5 ")
                time.sleep(60 )
        finally:
            if serverconn:
                serverconn.close()
        time.sleep(10 )


def popMasterPid():
    """The master's pid handed down by forkChild, or None (older master).

    Removed from the environment so the commands this child spawns do not
    carry it.
    """
    value = os.environ.pop(MASTER_PID_ENV, None)
    try:
        return int(value) if value else None
    except ValueError:
        return None

def exitOnStdinClose(masterPid=None):
    import whatap.util.logging_util as logging_util
    logging_util.debug("whatap.exitOnStdinClose scanning stdin step -1 ")
    # SERVER-2106: compare against the master's real pid when known; the ppid
    # sampled here may already be 1 if the master exited during our startup.
    ppid = masterPid if masterPid is not None else os.getppid()
    while True:
        if ppid != os.getppid():
            # Abnormal-termination trace: the master (parent) went away, so this
            # foreground child exits and will be re-forked by a new master. This
            # is a data-gap boundary (data-gap investigation).
            logging_util.error(
                "whatap_infrad child exiting: parent changed origPpid={} nowPpid={} pid={}".format(
                    ppid, os.getppid(), os.getpid()))
            os._exit(0)

        time.sleep(0.05)

def printusage():
    print "Usage"
    print "%s " % ( sys.argv[0] )

_SIGNAL_NAMES = {
    signal.SIGINT: "SIGINT", signal.SIGTERM: "SIGTERM",
    signal.SIGSEGV: "SIGSEGV", signal.SIGABRT: "SIGABRT",
    signal.SIGQUIT: "SIGQUIT",
}

def acquireWithin(lock, timeout):
    # py2.7 locks have no acquire timeout.
    deadline = time.time() + timeout
    while not lock.acquire(False):
        if time.time() >= deadline:
            return False
        time.sleep(0.05)
    return True

def waitChildGone(pid, timeout):
    # Reap here rather than sleep a fixed time; ECHILD means the watchdog's
    # Popen.wait() already reaped it. Popen.wait treats ECHILD as "exited"
    # (checked in the bundled 2.7.9 and 2.7.18), so reaping from both threads
    # is safe.
    deadline = time.time() + timeout
    while time.time() < deadline:
        try:
            if os.waitpid(pid, os.WNOHANG)[0] == pid:
                return
        except OSError:
            return
        time.sleep(0.1)

def signal_handler(sig, frame):
    global shuttingDown
    import whatap.util.logging_util as logging_util
    import whatap.agent.boot.boot as boot
    signame = _SIGNAL_NAMES.get(sig, str(sig))
    # SERVER-2106: stop the watchdog from forking first, then read childPid
    # under childLock - that waits out a fork already in progress, so the pid
    # read is the last child ever started and killing it leaves no orphan.
    # Bounded: a fork stuck in the kernel must not block shutdown. If it ever
    # completes, that child finds its master gone and exits (MASTER_PID_ENV).
    shuttingDown = True
    locked = acquireWithin(childLock, 5)
    try:
        pid = childPid
    finally:
        if locked:
            childLock.release()
    if not locked:
        logging_util.error("whatap_infrad a child fork did not finish within 5s; exiting anyway")
    # Abnormal-termination trace: record which signal is bringing the agent down
    # and whether we are killing the child too (data-gap investigation).
    logging_util.error(
        "whatap_infrad terminating on signal {} (pid={}, childPid={})".format(
            signame, os.getpid(), pid))
    if pid != 0:
        logging_util.error("whatap_infrad killing child pid={} before exit".format(pid))
        try:
            os.kill(pid, signal.SIGKILL)
        except OSError, e:
            logging_util.error("whatap_infrad failed to kill child pid={}: {}".format(pid, e))
        waitChildGone(pid, 5)
    sys.exit(0)

def main():
    import argparse
    parser = argparse.ArgumentParser(description='Whatap Server Health Monitoring')
    parser.add_argument('cmd', metavar='cmd', type=str,
                        help='command', nargs='?')
    parser.add_argument('--user', metavar='user', type=str,
                        help='user to run custom script')
    parser.add_argument('--request-id', metavar='request_id', type=int,
                        default=0, help='control request id')

    args = parser.parse_args()

    if args.cmd == 'version':
        print(AGENT_VERSION)
        return

    import whatap.agent.boot.boot as boot
    import whatap.util.logging_util as logging_util
    import whatap.agent.counter.counter_manager as counter_manager
    counter_manager.AGENT_VERSION = AGENT_VERSION

    signal.signal(signal.SIGINT, signal_handler)
    signal.signal(signal.SIGTERM, signal_handler)
    signal.signal(signal.SIGSEGV, signal_handler)
    signal.signal(signal.SIGABRT, signal_handler)
    # Old Solaris init scripts send SIGQUIT. With no handler the watchdog dies
    # on the default action and never kills its child, leaving an orphan that
    # keeps sending data while the service is believed stopped.
    signal.signal(signal.SIGQUIT, signal_handler)

    import whatap.agent.conf.configure as conf
    conf.entrypoint_path = os.path.split(os.path.abspath(__file__))[0]
    conf.GetConfig()

    if args.cmd:
        import whatap.util.thread_util as thread_util
        import whatap.util.logging_util as logging_util
        if args.cmd == 'boot-test':
            setLogger("whatap_infra_boottest.log")
            ret = boot.boot_test()
            if ret:
                sys.exit(0)
            else:
                sys.exit(1)
        elif args.cmd == 'debug':
            boot.boot()
        elif args.cmd == 'foreground':
            setLogger("whatap_infra_foreground.log")
            # SERVER-2106: a child forked while its master was exiting must not
            # boot at all - it would open a second session under the same oid.
            masterPid = popMasterPid()
            if masterPid is not None and os.getppid() != masterPid:
                logging_util.error(
                    "whatap_infrad child exiting at startup: master pid={} is gone (ppid={}, pid={})".format(
                        masterPid, os.getppid(), os.getpid()))
                os._exit(0)
            thread_util.async(boot.boot)
            thread_util.async(exitOnStdinClose, masterPid)
            thread_util.async(keepAliveSender)
        elif args.cmd == 'init-script':
            if args.user:
                conf.updateScripts(args.user)
                import whatap.agentless.checkmain as checkmain
                conf.GetConfig().load()
                checkmain.testRun()
            else:
                print("--user is required")
            return
        elif args.cmd in ('control-stop', 'control-restart'):
            # The control executor. It runs as its own short-lived process.
            # Both commands need the same preparation - default signal
            # handlers, the alias re-exec, the control log - so they share
            # one branch; a second copy would drift from this one and the
            # drift would only show up on a customer box.
            try:
                setLogger("whatap_infra_control.log")
            except Exception:
                pass

            # The handlers registered above belong to the long-running agent:
            # they log, kill the child and then exit(0). For the executor
            # exit(0) means "the agent was stopped", so dying on a signal must
            # never look like that. Back to the default action - the caller
            # then sees a signal death, and the status file keeps the truth.
            for _sig in (signal.SIGINT, signal.SIGTERM, signal.SIGQUIT,
                         signal.SIGSEGV, signal.SIGABRT):
                signal.signal(_sig, signal.SIG_DFL)

            def _control_warn(message):
                logging_util.error(message)
                try:
                    sys.stderr.write("whatap_infrad: %s\n" % message)
                except Exception:
                    pass

            # This process is about to kill the agent, so it must not look
            # like the agent. The HP-UX patch script and install.sh sweep and
            # kill every process whose command line holds the exact token
            # "<install>/whatap_infrad"; re-invoking the agent binary puts
            # that token on this command line, so an old patch script run by
            # hand would kill the executor in the middle of a stop. Re-exec
            # under a hard-link alias instead.
            import whatap.agent.control.agentstop.alias as control_alias
            agent_path = os.path.abspath(sys.argv[0])
            install_root = os.path.dirname(agent_path)
            if not control_alias.is_alias(sys.argv[0]):
                alias_path = control_alias.prepare(agent_path)
                if alias_path:
                    # Retry once on ENOENT. Another executor for this same
                    # install root removes the alias in its `finally`
                    # (run.py: alias.remove_stale), so the link this process
                    # just made can be unlinked in the window between
                    # prepare() and execv(). Falling back on that would leave
                    # this process running under the agent's own name - the
                    # exact state the alias exists to prevent, since a
                    # name-based patch-script sweep would then kill it in the
                    # middle of a stop. Remaking the link is cheap and the
                    # window is narrow, so one retry closes it.
                    for _attempt in (0, 1):
                        try:
                            # The process image is replaced here; this call
                            # does not return. Afterwards sys.argv[0] is the
                            # alias, so is_alias() is true and this branch is
                            # not re-entered (no re-exec loop).
                            os.execv(alias_path, [alias_path] + sys.argv[1:])
                        except Exception as e:
                            alias_error = e
                            if _attempt or getattr(e, "errno", None) != errno.ENOENT:
                                break
                            alias_path = control_alias.prepare(agent_path)
                            if not alias_path:
                                break
                    _control_warn(
                        "%s could not re-exec as the alias, continuing "
                        "under the agent name: %s" % (args.cmd, alias_error))
                else:
                    _control_warn(
                        "%s could not create the alias link, continuing "
                        "under the agent name (an old patch script sweep "
                        "may kill this process)" % args.cmd)

            import whatap.agent.control.agentstop.run as control_run
            conf_dir = os.environ["WHATAP_HOME"]
            if args.cmd == 'control-restart':
                res = control_run.run_restart(
                    install_root, conf_dir, args.request_id, logging_util.info)
            else:
                res = control_run.run_stop(
                    install_root, conf_dir, args.request_id, logging_util.info)
            sys.exit(0 if res["code"] in ("", "OK") else 1)
        else:
            # Without this branch an unrecognised subcommand matches nothing,
            # skips daemonize and falls into the `while Enabled: sleep(1)`
            # loop below - it hangs forever and says nothing (measured on
            # Solaris: still alive after 8 seconds, silent). A new service
            # script talking to an old binary hit exactly that, and a silent
            # hang during a patch stalls the upgrade. Old binaries cannot be
            # fixed, but from this version on the trap is gone.
            sys.stderr.write("unknown command: %s\n" % args.cmd)
            sys.exit(2)
    else:
        setLogger("whatap_infra.log")
        daemonize( os.getuid() )

    logging_util.debug("whatap_infrad.main ", conf.entrypoint_path)

    while conf.GetConfig().Enabled:
        time.sleep(1)

    # Abnormal-termination trace: the agent was disabled remotely, so this
    # process exits (the child then re-forks only if re-enabled).
    logging_util.error("whatap_infrad exiting: agent disabled from service.whatap.io (pid={})".format(os.getpid()))
    print( 'agent disabled from service.whatap.io  exiting')

if __name__ == '__main__':
    try:
        main()
    except KeyboardInterrupt, e:
        logging_util.error("whatap_infrad exiting on KeyboardInterrupt (pid={})".format(os.getpid()))
        sys.exit(0)
    except Exception, e:
        # Abnormal-termination trace: an uncaught exception is crashing the
        # process; record it before dying (data-gap investigation).
        try:
            logging_util.error("whatap_infrad crashing on uncaught exception (pid={}): {}".format(os.getpid(), e))
            logging_util.errorStack(e)
        except Exception:
            pass
        raise
