Files
zoneminder/src/zm_audio_detector.cpp
T
Isaac ConnorandClaude Opus 5 428c048bac feat: measure the audio level on demand, with a meter in the editor
Reverts the previous commit's always-on measurement. Decoding audio for every
monitor that has it spends CPU on a number almost nothing reads, which is the
wrong trade even though it did solve the chicken and egg of picking a
threshold without ever seeing a level.

Measure when something is actually going to use the reading instead:

 - AudioDetection is on, as before, so nothing changes for a monitor that
   scores on audio; or
 - somebody asked. SharedData gains audio_level_until, a wall clock second
   the capture thread keeps measuring up to. The monitor editor's new level
   meter pushes it forward while it is on screen and the measurement lapses a
   few seconds after the page is left, so nothing has to send a stop and a
   crashed browser cannot leave a monitor decoding forever.

When the reading stops being wanted the decoder is released and the published
level and peak are cleared, so a stale number is not left looking current and
an old peak does not land on the next frame row written.

audio_level_until is carved out of analysis_pad rather than appended, so
SharedData stays 888 bytes and no existing offset moves; the static_asserts,
Memory.pm and Monitor.php are updated together and all three now agree the
field is at +880.

The meter itself is on the audio settings, shown whether or not
AudioDetection is checked, because the level is what you need in order to
choose a threshold. It draws the threshold currently in the input as a mark on
the bar so a reading can be judged against it before saving, and a monitor
whose zmc is not running reads "no reading" rather than a confident 0, which
would be indistinguishable from silence.

Frames.AudioLevel is therefore 0 again on monitors that do not score on audio.
That is what the graph already treats as "no audio data", so it draws no line
rather than a flat one.

Co-Authored-By: Claude Opus 5 (1M context) <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_01JpiSWBmtQkR5bcgpHWY4ME
2026-09-20 18:17:44 -05:00

219 lines
7.2 KiB
C++

/*
* This file is part of the ZoneMinder Project. See AUTHORS file for Copyright information
*
* This program is free software; you can redistribute it and/or modify it
* under the terms of the GNU General Public License as published by the
* Free Software Foundation; either version 2 of the License, or (at your
* option) any later version.
*
* This program is distributed in the hope that it will be useful, but WITHOUT
* ANY WARRANTY; without even the implied warranty of MERCHANTABILITY or
* FITNESS FOR A PARTICULAR PURPOSE. See the GNU General Public License for
* more details.
*
* You should have received a copy of the GNU General Public License along
* with this program. If not, see <http://www.gnu.org/licenses/>.
*/
#include "zm_audio_detector.h"
#include "zm_logger.h"
#include <algorithm>
#include <cmath>
namespace {
int FrameChannels(const AVFrame *frame) {
#if LIBAVUTIL_VERSION_CHECK(57, 28, 100, 28, 0)
return frame->ch_layout.nb_channels;
#else
return frame->channels;
#endif
}
} // namespace
AudioDetector::~AudioDetector() {
Close();
}
bool AudioDetector::Open(const AVCodecParameters *codecpar) {
// Already established that this codec cannot be decoded. Say so without
// logging again: Monitor::Capture retries on every audio packet.
if (codecpar and (codecpar->codec_id == failed_codec_)) return false;
Close();
if (!codecpar) return false;
const AVCodec *codec = avcodec_find_decoder(codecpar->codec_id);
if (!codec) {
Warning("Audio detection: no decoder for codec %d, level reporting disabled", codecpar->codec_id);
failed_codec_ = codecpar->codec_id;
return false;
}
codec_context_ = avcodec_alloc_context3(codec);
if (!codec_context_) {
Error("Audio detection: could not allocate a decoder context");
failed_codec_ = codecpar->codec_id;
return false;
}
if (avcodec_parameters_to_context(codec_context_, codecpar) < 0) {
Error("Audio detection: could not copy stream parameters");
Close();
failed_codec_ = codecpar->codec_id;
return false;
}
if (avcodec_open2(codec_context_, codec, nullptr) < 0) {
Error("Audio detection: could not open the %s decoder", codec->name);
Close();
failed_codec_ = codecpar->codec_id;
return false;
}
failed_codec_ = AV_CODEC_ID_NONE;
Debug(1, "Audio detection: opened %s decoder", codec->name);
return true;
}
void AudioDetector::Close() {
if (codec_context_) avcodec_free_context(&codec_context_);
codec_context_ = nullptr;
level_.store(0, std::memory_order_relaxed);
}
double AudioDetector::RmsS16(const int16_t *samples, size_t count) {
if (!samples or !count) return 0.0;
double sum = 0.0;
for (size_t i = 0; i < count; i++) {
// 32768 rather than 32767: -32768 is a legal sample and would otherwise
// push the ratio just over 1.0.
const double v = static_cast<double>(samples[i]) / 32768.0;
sum += v * v;
}
return std::sqrt(sum / static_cast<double>(count));
}
double AudioDetector::RmsFloat(const float *samples, size_t count) {
if (!samples or !count) return 0.0;
double sum = 0.0;
for (size_t i = 0; i < count; i++) {
const double v = std::max(-1.0, std::min(1.0, static_cast<double>(samples[i])));
sum += v * v;
}
return std::sqrt(sum / static_cast<double>(count));
}
int AudioDetector::LevelFromRms(double rms) {
if (!(rms > 0.0)) return 0; // also catches NaN
const double db = 20.0 * std::log10(std::min(rms, 1.0));
if (db <= AUDIO_FLOOR_DB) return 0;
const double level = 100.0 * (1.0 - db / AUDIO_FLOOR_DB);
return static_cast<int>(std::lround(std::max(0.0, std::min(100.0, level))));
}
bool AudioDetector::IsAlarm(int level, int threshold) {
// Threshold 0 is "off". Without this a monitor that had detection enabled
// but never had a threshold set would alarm on every packet, silence
// included, because a level of 0 is >= a threshold of 0.
if (threshold <= 0) return false;
return level >= threshold;
}
bool AudioDetector::LevelWanted(bool audio_detection, uint32_t request_until, uint32_t now) {
// A monitor that scores on audio needs the level on every packet anyway.
if (audio_detection) return true;
// Nobody has asked. Distinguished from an expired request only for clarity;
// the comparison below would reject 0 anyway for any plausible clock.
if (!request_until) return false;
return now <= request_until;
}
double AudioDetector::RmsFromFrame(const AVFrame *frame) const {
const int channels = FrameChannels(frame);
const int samples = frame->nb_samples;
if (channels <= 0 or samples <= 0) return 0.0;
const AVSampleFormat format = static_cast<AVSampleFormat>(frame->format);
const bool planar = av_sample_fmt_is_planar(format) != 0;
// Planar frames keep one plane per channel; interleaved keeps everything in
// plane 0, so one "plane" of channels * samples values.
const int planes = planar ? channels : 1;
const size_t per_plane = static_cast<size_t>(samples) * (planar ? 1 : channels);
// Averaging the per-plane mean squares gives the same answer as one pass
// over every sample, and keeps interleaved and planar on the same scale.
double sum_of_squares = 0.0;
for (int p = 0; p < planes; p++) {
if (!frame->extended_data[p]) continue;
double rms = 0.0;
switch (format) {
case AV_SAMPLE_FMT_S16:
case AV_SAMPLE_FMT_S16P:
rms = RmsS16(reinterpret_cast<const int16_t *>(frame->extended_data[p]), per_plane);
break;
case AV_SAMPLE_FMT_FLT:
case AV_SAMPLE_FMT_FLTP:
rms = RmsFloat(reinterpret_cast<const float *>(frame->extended_data[p]), per_plane);
break;
default:
// Every codec ZM sees over RTSP decodes to s16 or float. Anything else
// is reported once rather than silently scoring 0 forever.
Debug(1, "Audio detection: unhandled sample format %s",
av_get_sample_fmt_name(format));
return 0.0;
}
sum_of_squares += rms * rms;
}
return std::sqrt(sum_of_squares / static_cast<double>(planes));
}
int AudioDetector::Process(const AVPacket *packet) {
if (!codec_context_ or !packet) return Level();
if (avcodec_send_packet(codec_context_, packet) < 0) {
Debug(2, "Audio detection: decoder rejected a packet");
return Level();
}
AVFrame *frame = av_frame_alloc();
if (!frame) return Level();
int level = -1;
while (avcodec_receive_frame(codec_context_, frame) == 0) {
// A packet can yield several frames; the loudest wins, so a short sound at
// the head of the packet is not averaged away by the quiet that follows.
level = std::max(level, LevelFromRms(RmsFromFrame(frame)));
av_frame_unref(frame);
}
av_frame_free(&frame);
// No frames means the decoder is still priming, which is normal. Holding the
// previous level is better than reporting a spurious 0.
if (level < 0) return Level();
level_.store(level, std::memory_order_relaxed);
RaisePeak(level);
return level;
}
void AudioDetector::RaisePeak(int level) {
int current = peak_.load(std::memory_order_relaxed);
while (level > current &&
!peak_.compare_exchange_weak(current, level,
std::memory_order_relaxed,
std::memory_order_relaxed)) {
// compare_exchange_weak refreshed current; loop unless it now wins.
}
}