mirror of
https://github.com/ZoneMinder/zoneminder.git
synced 2026-10-02 23:45:08 -04:00
The capture daemons and this watchdog share the database. If the DB (or the host/network) drops out, every zmc stops updating its shared-memory heartbeat at the same instant. When the DB connection recovers, the next check pass saw the whole fleet as simultaneously stale and restarted -- and, for cameras with control capability, rebooted -- every monitor, even though none of them individually failed. Track wall-clock time between check passes. If we just came out of a DB reconnect, or far more time elapsed than a normal check interval (host stall/clock step), skip restart+reboot actions for one pass so capture daemons can refresh their heartbeats. A genuinely dead camera is still stale next pass and gets handled normally -- reboot behaviour is preserved, just no longer fired on a false fleet-wide positive. Co-Authored-By: Claude Sonnet 5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_01HoNiNxwgyaHV29CbiueCUf
246 lines
9.2 KiB
Plaintext
246 lines
9.2 KiB
Plaintext
#!@PERL_EXECUTABLE@ -wT
|
|
#
|
|
# ==========================================================================
|
|
#
|
|
# ZoneMinder WatchDog Script, $Date$, $Revision$
|
|
# Copyright (C) 2001-2008 Philip Coombes
|
|
#
|
|
# This program is free software; you can redistribute it and/or
|
|
# modify it under the terms of the GNU General Public License
|
|
# as published by the Free Software Foundation; either version 2
|
|
# of the License, or (at your option) any later version.
|
|
#
|
|
# This program is distributed in the hope that it will be useful,
|
|
# but WITHOUT ANY WARRANTY; without even the implied warranty of
|
|
# MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the
|
|
# GNU General Public License for more details.
|
|
#
|
|
# You should have received a copy of the GNU General Public License
|
|
# along with this program; if not, write to the Free Software
|
|
# Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA.
|
|
#
|
|
# ==========================================================================
|
|
|
|
=head1 NAME
|
|
|
|
zmwatch.pl - ZoneMinder WatchDog Script
|
|
|
|
=head1 SYNOPSIS
|
|
|
|
zmwatch.pl
|
|
|
|
=head1 DESCRIPTION
|
|
|
|
This does some basic setup for ZoneMinder to run and then periodically
|
|
checks the fps output of the active daemons to check they haven't
|
|
locked up. If they have then they are killed and restarted
|
|
|
|
=cut
|
|
use strict;
|
|
use bytes;
|
|
|
|
# ==========================================================================
|
|
#
|
|
# These are the elements you can edit to suit your installation
|
|
#
|
|
# ==========================================================================
|
|
|
|
use constant START_DELAY => 30; # To give everything else time to start
|
|
|
|
# ==========================================================================
|
|
#
|
|
# Don't change anything below here
|
|
#
|
|
# ==========================================================================
|
|
|
|
@EXTRA_PERL_LIB@
|
|
use ZoneMinder;
|
|
use ZoneMinder::Monitor;
|
|
use POSIX;
|
|
use DBI;
|
|
use autouse 'Data::Dumper'=>qw(Dumper);
|
|
|
|
$| = 1;
|
|
|
|
$ENV{PATH} = '/bin:/usr/bin:/usr/local/bin';
|
|
$ENV{SHELL} = '/bin/sh' if exists $ENV{SHELL};
|
|
delete @ENV{qw(IFS CDPATH ENV BASH_ENV)};
|
|
|
|
my $log = logInit();
|
|
logSetSignal();
|
|
my $zm_terminate = 0;
|
|
sub TermHandler {
|
|
Debug('Received TERM, exiting');
|
|
$zm_terminate = 1;
|
|
}
|
|
$SIG{TERM} = \&TermHandler;
|
|
$SIG{INT} = \&TermHandler;
|
|
|
|
my $dbh = zmDbConnect();
|
|
|
|
# We don't need to keep objects cached
|
|
$ZoneMinder::Object::no_cache = 1;
|
|
|
|
Debug('Watchdog starting, pausing for '.START_DELAY.' seconds');
|
|
sleep(START_DELAY);
|
|
|
|
# Wall-clock of the last completed check pass, to detect systemic stalls (see below).
|
|
my $last_pass_time = time();
|
|
|
|
while (!$zm_terminate) {
|
|
my $db_was_down = 0;
|
|
while (!($dbh and $dbh->ping()) and !$zm_terminate) {
|
|
$db_was_down = 1;
|
|
if (!($dbh = zmDbConnect())) {
|
|
sleep($Config{ZM_WATCH_CHECK_INTERVAL});
|
|
}
|
|
}
|
|
|
|
# --- Systemic-stall guard -------------------------------------------------
|
|
# The capture daemons and this watchdog share the database. If the DB (or the
|
|
# host/network) drops out, EVERY zmc stops updating its shared-memory
|
|
# heartbeat at the same instant. When the DB returns, this loop would see the
|
|
# whole fleet as simultaneously "stale" and restart -- and reboot -- every
|
|
# camera, even though nothing is wrong with any of them. That is the cause of
|
|
# the fleet-wide restart storms (each one coincided with a DB outage).
|
|
#
|
|
# Distinguish "this camera died" from "the thing we both depend on died": if
|
|
# we just came out of a DB reconnect, or far more wall-clock than a normal
|
|
# cycle has elapsed (host stall / clock step), skip restart+reboot actions for
|
|
# ONE pass so the daemons can refresh their heartbeats. A genuinely dead
|
|
# camera is still stale next pass (~check interval later) and gets handled
|
|
# then -- reboot behaviour is preserved, just no longer fired on a false
|
|
# fleet-wide positive.
|
|
my $pass_now = time();
|
|
my $since_last_pass = $pass_now - $last_pass_time;
|
|
$last_pass_time = $pass_now;
|
|
my $stall_grace = $db_was_down
|
|
|| ($since_last_pass > 2 * $Config{ZM_WATCH_CHECK_INTERVAL});
|
|
if ($stall_grace) {
|
|
Warning("zmwatch: systemic stall detected (db_was_down=$db_was_down, "
|
|
."${since_last_pass}s since last check) - skipping restart/reboot actions "
|
|
."this pass so capture daemons can refresh their heartbeats");
|
|
}
|
|
|
|
foreach my $monitor (ZoneMinder::Monitor->find(Deleted=>0, $Config{ZM_SERVER_ID} ? (ServerId=>$Config{ZM_SERVER_ID}) : ())) {
|
|
next if $stall_grace; # systemic DB/host stall this pass: don't restart/reboot on false staleness
|
|
next if $monitor->{Capturing} eq 'None';
|
|
next if $monitor->{Type} eq 'WebSite';
|
|
|
|
my $now = time();
|
|
my $restart = 0;
|
|
if (!zmMemVerify($monitor)) {
|
|
Info("Restarting capture daemon for $monitor->{Id} $monitor->{Name}, shared data not valid");
|
|
$monitor->control('restart');
|
|
next;
|
|
}
|
|
|
|
my $heartbeat_time = zmMemRead($monitor, 'shared_data:heartbeat_time');
|
|
my $heartbeat_elapsed = $now-$heartbeat_time;
|
|
if ($heartbeat_elapsed > $Config{ZM_WATCH_MAX_DELAY}) {
|
|
Info("Restarting capture daemon for $monitor->{Id} $monitor->{Name}, $now - heartbeat time $heartbeat_time $heartbeat_elapsed > $Config{ZM_WATCH_MAX_DELAY}");
|
|
$monitor->control('restart');
|
|
next;
|
|
} else {
|
|
Debug("Monitor $monitor->{Id} $monitor->{Name}, heartbeat time $now - $heartbeat_time $heartbeat_elapsed < $Config{ZM_WATCH_MAX_DELAY}");
|
|
}
|
|
|
|
next if $monitor->{Capturing} eq 'Ondemand';
|
|
next if $monitor->{Decoding} eq 'None' or $monitor->{Decoding} eq 'Ondemand';
|
|
|
|
# Check we have got an image recently
|
|
my $capture_time = zmGetLastWriteTime($monitor);
|
|
if (!defined($capture_time)) {
|
|
# Can't read from shared data
|
|
Warning('LastWriteTime is not defined.');
|
|
next;
|
|
}
|
|
Debug("Monitor $$monitor{Id} LastWriteTime is $capture_time.");
|
|
if (!$capture_time) {
|
|
# We can't get the last capture time so can't be sure it's died, it might just be starting up.
|
|
my $startup_time = zmGetStartupTime($monitor);
|
|
my $startup_elapsed = $now - $startup_time;
|
|
if ($startup_elapsed > $Config{ZM_WATCH_MAX_DELAY}) {
|
|
Debug("Monitor $monitor->{Id} $monitor->{Name}, startup time $now - $startup_time $startup_elapsed <? $Config{ZM_WATCH_MAX_DELAY}");
|
|
if ($monitor->ControlId()) {
|
|
my $control = $monitor->Control();
|
|
# Only try to reboot the camera if it actually answers. Otherwise
|
|
# open() blocks until it times out on a camera that is down (the
|
|
# common reason there is no image since startup). ping() resolves the
|
|
# host from the monitor itself, so no need to dig the ip out here.
|
|
if ($control and $control->CanReboot()) {
|
|
if (!$control->ping()) {
|
|
Debug("Not rebooting $monitor->{Id} $monitor->{Name}: camera is not reachable");
|
|
} elsif ($control->open()) {
|
|
$control->reboot();
|
|
}
|
|
}
|
|
}
|
|
$log->logPrint(ZoneMinder::Logger::WARNING+$monitor->ImportanceNumber(),
|
|
"Restarting capture daemon for $monitor->{Id} $$monitor{Name}, no image since startup. ".
|
|
"Startup time was $startup_time - now $now > $Config{ZM_WATCH_MAX_DELAY}"
|
|
);
|
|
$monitor->control('restart');
|
|
}
|
|
next;
|
|
}
|
|
|
|
my $max_image_delay = (
|
|
$monitor->{MaxFPS}
|
|
&&($monitor->{MaxFPS}>0)
|
|
&&($monitor->{MaxFPS}<1)
|
|
) ? (3/$monitor->{MaxFPS})
|
|
: $Config{ZM_WATCH_MAX_DELAY};
|
|
|
|
my $image_delay = $now - $capture_time;
|
|
Debug("Monitor $monitor->{Id} last captured $image_delay seconds ago, max is $max_image_delay");
|
|
if ($image_delay > $max_image_delay) {
|
|
$log->logPrint(ZoneMinder::Logger::WARNING+$monitor->ImportanceNumber(),
|
|
'Restarting capture daemon for '.$monitor->{Name}.
|
|
", time since last capture $image_delay seconds ($now-$capture_time)");
|
|
$monitor->control('restart');
|
|
next;
|
|
}
|
|
|
|
if ($monitor->{Analysing} ne 'None') {
|
|
# Now check analysis thread
|
|
# Check we have got an image recently
|
|
my $image_time = zmGetLastReadTime($monitor);
|
|
if (!defined($image_time)) {
|
|
# Can't read from shared data
|
|
Error("Error reading shared data for $$monitor{Id} $$monitor{Name}");
|
|
$monitor->control('restart');
|
|
next;
|
|
} elsif (!$image_time) {
|
|
Debug("Last analyse time for $$monitor{Id} $$monitor{Name} was zero.");
|
|
} else {
|
|
my $max_image_delay = ( $monitor->{MaxFPS}
|
|
&&($monitor->{MaxFPS}>0)
|
|
&&($monitor->{MaxFPS}<1)
|
|
) ? (3/$monitor->{MaxFPS})
|
|
: $Config{ZM_WATCH_MAX_DELAY}
|
|
;
|
|
my $image_delay = $now-$image_time;
|
|
Debug("Monitor $monitor->{Id} last analysed $image_delay seconds ago, max is $max_image_delay");
|
|
if ($image_delay > $max_image_delay) {
|
|
$log->logPrint(ZoneMinder::Logger::WARNING+$monitor->ImportanceNumber(),
|
|
"daemon for $$monitor{Id} $$monitor{Name} needs restarting,"
|
|
." time since last analysis $image_delay seconds ($now-$image_time)");
|
|
$monitor->control('restart');
|
|
next;
|
|
}
|
|
}
|
|
} # end if check analysis daemon
|
|
} # end foreach monitor
|
|
|
|
Debug("Sleeping $Config{ZM_WATCH_CHECK_INTERVAL}");
|
|
sleep($Config{ZM_WATCH_CHECK_INTERVAL});
|
|
Debug("Done Sleeping $Config{ZM_WATCH_CHECK_INTERVAL}");
|
|
} # end while (!$zm_terminate)
|
|
|
|
Debug('Watchdog exiting');
|
|
exit();
|
|
|
|
1;
|
|
__END__
|