Skip to content
Open
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
12 changes: 12 additions & 0 deletions CMakeLists.txt
Original file line number Diff line number Diff line change
Expand Up @@ -753,6 +753,7 @@ set(CORE_SOURCES
src/core/SpecbleachFilter.cpp
src/core/CwDecoder.cpp
src/core/DeepCwEngine.cpp # neural CW decoder backend (ONNX; inert without HAVE_ONNX)
src/core/DeepCwCommitter.cpp # DeepCW sliding window + time-anchored commit
src/core/CwCallsignSpotter.cpp
src/core/CallsignInfo.cpp
src/core/QrzClient.cpp
Expand Down Expand Up @@ -2791,3 +2792,14 @@ else()
)
endif()


# DeepCW offline replay harness (RFC #4817 regression take): not built by default.
if(ORT_FOUND)
add_executable(deepcw_replay EXCLUDE_FROM_ALL
tools/deepcw_replay.cpp src/core/DeepCwEngine.cpp src/core/DeepCwCommitter.cpp
src/core/Resampler.cpp)
target_compile_definitions(deepcw_replay PRIVATE HAVE_ONNX)
target_include_directories(deepcw_replay PRIVATE src
${CMAKE_SOURCE_DIR}/third_party/r8brain ${ORT_INCLUDE_DIRS})
target_link_libraries(deepcw_replay PRIVATE Qt6::Widgets ${ORT_LIBRARIES})
endif()
92 changes: 28 additions & 64 deletions src/core/CwDecoder.cpp
Original file line number Diff line number Diff line change
@@ -1,5 +1,6 @@
#include "CwDecoder.h"
#include "LogManager.h"
#include "DeepCwCommitter.h"
#include "DeepCwEngine.h"
#include "Resampler.h"
#include "ggmorse/ggmorse.h"
Expand Down Expand Up @@ -268,37 +269,31 @@ void CwDecoder::decodeLoop()
qCDebug(lcDsp) << "CwDecoder: decode loop exiting, total frames:" << feedCount;
}

// DeepCW (neural) worker loop. The model is a whole-window CTC decoder trained
// on 5-20 s clips, so we accumulate a rolling audio segment and re-decode it as
// it grows, emitting only the newly-decoded suffix (the decode of a longer clip
// is normally a prefix-extension of the shorter one). Near the model's 20 s cap
// we finalize the segment and start fresh so inference stays in-distribution.
//
// feedAudio() has downmixed the RX audio to mono float32 @24 kHz into m_ringBuf;
// here we drain it, resample to the model's 3200 Hz with an anti-aliased r8brain
// SRC (a 7.5x decimation — a naive drop/linear resample would fold energy into
// the 400-1200 Hz analysis band), and grow a rolling 3200 Hz segment that we
// re-decode as it lengthens. The SRC stays continuous across segment resets
// (the audio stream is continuous even though the analysis window restarts).
// Prototype heuristics — a later revision can add overlap-merge and silence
// segmentation.
// DeepCW (neural) worker loop. feedAudio() has downmixed the RX audio to mono
// float32 @24 kHz into m_ringBuf; here we drain it, resample to the model's
// 3200 Hz with an anti-aliased r8brain SRC (a 7.5x decimation — a naive
// drop/linear resample would fold energy into the 400-1200 Hz analysis band),
// and hand it to DeepCwCommitter: a sliding window re-decoded every 2 s whose
// characters are shown only once they are holdSec behind the live edge, so the
// model's full-context reading reaches the panel instead of its first guess at
// the ragged end of a short window, and no hard reset cuts words at a seam.
// holdSec defaults to 5 s; AETHER_DEEPCW_HOLD_S overrides it (local bench knob).
void CwDecoder::decodeLoopDeep()
{
constexpr int kRate = DeepCwEngine::kModelSampleRate; // 3200 Hz (post-resample)
const size_t kMinDecode = kRate * 5; // model floor: 5 s
const size_t kHopSamples = kRate * 2; // re-decode every ~2 s of new audio
const size_t kMaxSamples = kRate * 15; // finalize before the 20 s cap (headroom)
constexpr int kRate = DeepCwEngine::kModelSampleRate; // 3200 Hz (post-resample)

// Anti-aliased 24k -> 3200 Hz SRC (r8brain via the in-tree wrapper). Owned by
// and used only on this worker thread, so its non-thread-safety is moot.
Resampler resampler(24000.0, static_cast<double>(kRate));

std::vector<float> seg; // accumulated audio at 3200 Hz
seg.reserve(kMaxSamples + kRate);
std::string emitted; // text already emitted for the current segment
size_t lastDecodeSamples = 0;
double holdSec = 5.0;
bool holdOk = false;
const double envHold = qEnvironmentVariable("AETHER_DEEPCW_HOLD_S").toDouble(&holdOk);
if (holdOk && envHold >= 1.0 && envHold <= 14.0) holdSec = envHold;
DeepCwCommitter committer(holdSec);

qCDebug(lcDsp) << "CwDecoder: DeepCW loop running, modelLoaded:" << m_deepLoaded.load();
qCDebug(lcDsp) << "CwDecoder: DeepCW loop running, modelLoaded:" << m_deepLoaded.load()
<< "hold" << committer.holdSec() << "s window" << committer.windowSec() << "s";

while (m_running) {
// Drain the handoff ring (mono float32 @24k) and resample to 3200 Hz.
Expand All @@ -312,53 +307,22 @@ void CwDecoder::decodeLoopDeep()
m_ringBuf.clear();
}
}
if (!in24k.empty()) {
if (!in24k.empty() && m_deepLoaded && m_deepcw) {
const QByteArray out = resampler.process(in24k.data(), static_cast<int>(in24k.size()));
const auto* r = reinterpret_cast<const float*>(out.constData());
const int m = out.size() / static_cast<int>(sizeof(float));
seg.insert(seg.end(), r, r + m);
}

const bool haveMin = seg.size() >= kMinDecode;
const bool grewEnough = seg.size() >= lastDecodeSamples + kHopSamples;

if (m_deepLoaded && m_deepcw && haveMin && grewEnough) {
float conf = 1.0f;
float pitchHz = 0.0f;
const std::string text = m_deepcw->decode(seg, kRate, &conf, &pitchHz);
lastDecodeSamples = seg.size();
// Map mean CTC confidence to the panel's cost convention (lower =
// better) so the Sensitivity slider filters shaky neural decodes.
const float cost = 1.0f - conf;
const auto m = static_cast<std::size_t>(out.size() / static_cast<int>(sizeof(float)));
const DeepCwCommitter::Result res = committer.push(r, m, *m_deepcw);

// Publish the dominant-tone pitch (no speed estimate for a CTC model)
// so zero-beat and the pitch readout work in neural mode too.
if (pitchHz > 0.0f) {
m_pitch = pitchHz;
emit statsUpdated(pitchHz, 0.0f);
if (res.decoded && res.pitchHz > 0.0f) {
m_pitch = res.pitchHz;
emit statsUpdated(res.pitchHz, 0.0f);
}

// Emit the suffix beyond what we've shown when the new decode extends
// the old as a prefix; on a divergent revision, silently adopt the new
// baseline (a rare correction may drop/duplicate a few chars — accepted
// for the prototype).
if (text.size() >= emitted.size()
&& text.compare(0, emitted.size(), emitted) == 0) {
const std::string delta = text.substr(emitted.size());
if (!delta.empty()) {
emit textDecoded(QString::fromStdString(delta), cost);
emitted = text;
}
} else {
emitted = text;
}
}

// Finalize near the model's max window and start a fresh segment.
if (seg.size() >= kMaxSamples) {
seg.clear();
emitted.clear();
lastDecodeSamples = 0;
// Map mean CTC confidence to the panel's cost convention (lower =
// better) so the Sensitivity slider filters shaky neural decodes.
if (!res.text.empty())
emit textDecoded(QString::fromStdString(res.text), 1.0f - res.meanConf);
}

QThread::msleep(200);
Expand Down
103 changes: 103 additions & 0 deletions src/core/DeepCwCommitter.cpp
Original file line number Diff line number Diff line change
@@ -0,0 +1,103 @@
#include "DeepCwCommitter.h"

#include <algorithm>

namespace AetherSDR {

namespace {
constexpr int kRate = DeepCwEngine::kModelSampleRate;
constexpr double kFrameSec = static_cast<double>(DeepCwEngine::kHopLength) / kRate;
constexpr int kSnapBlankFrames = 3;
} // namespace

DeepCwCommitter::DeepCwCommitter(double holdSec, double leftContextSec, double hopSec)
: m_hold(holdSec)
, m_left(leftContextSec)
, m_hop(hopSec)
// Stay inside the model's trained 5-20 s window range.
, m_window(std::min(DeepCwEngine::kMaxWindowSec,
std::max(15.0, holdSec + leftContextSec + hopSec)))
, m_keep(std::min(m_window - hopSec, holdSec + leftContextSec))
{
}

void DeepCwCommitter::reset()
{
m_buf.clear();
m_bufStart = 0;
m_lastDecodeEnd = 0;
m_tCommit = -1.0;
m_tLastChar = -1.0;
m_lastOut = 0;
}

DeepCwCommitter::Result DeepCwCommitter::push(const float* audio3200, std::size_t n,
const DeepCwEngine& eng)
{
Result r;
if (n > 0) m_buf.insert(m_buf.end(), audio3200, audio3200 + n);
const std::size_t absEnd = m_bufStart + m_buf.size();
if (m_buf.size() >= static_cast<std::size_t>(kRate * DeepCwEngine::kMinWindowSec)
&& absEnd >= m_lastDecodeEnd + static_cast<std::size_t>(m_hop * kRate)) {
m_lastDecodeEnd = absEnd;
r = decodeAndCommit(eng, false);
}
if (m_buf.size() >= static_cast<std::size_t>(m_window * kRate)) {
const std::size_t drop = m_buf.size() - static_cast<std::size_t>(m_keep * kRate);
m_buf.erase(m_buf.begin(), m_buf.begin() + static_cast<std::ptrdiff_t>(drop));
m_bufStart += drop;
}
return r;
}

DeepCwCommitter::Result DeepCwCommitter::flush(const DeepCwEngine& eng)
{
return decodeAndCommit(eng, true);
}

DeepCwCommitter::Result DeepCwCommitter::decodeAndCommit(const DeepCwEngine& eng, bool flushAll)
{
Result r;
int frames = 0;
const std::vector<float> lp = eng.inferLogProbs(m_buf, &frames, &r.pitchHz);
if (frames == 0) return r;
r.decoded = true;

std::vector<uint8_t> blank;
const std::vector<DeepCwEngine::Emission> em = eng.greedyEmissions(lp.data(), frames, &blank);

const double t0 = static_cast<double>(m_bufStart) / kRate;
const double nowT = static_cast<double>(m_bufStart + m_buf.size()) / kRate;
double cutoff = 1e18;
if (!flushAll) {
// Snap back to the latest frame <= (now - hold) that ends a blank run.
int f = std::min(frames - 1, static_cast<int>((nowT - m_hold - t0) / kFrameSec));
while (f >= kSnapBlankFrames
&& !(blank[f] && blank[f - 1] && blank[f - 2])) --f;
cutoff = t0 + f * kFrameSec;
}

double confSum = 0.0;
int confN = 0;
for (const auto& e : em) {
const double t = t0 + e.frame * kFrameSec;
// Word spaces are emitted inside the silent gap the cutoff snaps into,
// so a space may land just behind tCommit on the next decode: accept it
// anywhere after the last committed letter.
const double from = (e.ch == ' ') ? m_tLastChar : m_tCommit;
if (t <= from || t > cutoff) continue;
if (e.ch == ' ' && (m_lastOut == 0 || m_lastOut == ' ')) continue;
r.text.push_back(e.ch);
m_lastOut = e.ch;
if (e.ch != ' ') {
m_tLastChar = t;
confSum += e.conf;
++confN;
}
}
if (confN > 0) r.meanConf = static_cast<float>(confSum / confN);
if (cutoff > m_tCommit) m_tCommit = cutoff;
return r;
}

} // namespace AetherSDR
59 changes: 59 additions & 0 deletions src/core/DeepCwCommitter.h
Original file line number Diff line number Diff line change
@@ -0,0 +1,59 @@
#pragma once

// DeepCW streaming front: sliding analysis window + time-anchored commit.
//
// The model is a whole-window CTC decoder, and its reading of a character
// keeps improving while more audio arrives after it. So each hop the whole
// window is re-decoded, but a character is committed (shown) only once it was
// emitted more than holdSec behind the live edge; everything newer stays
// provisional and is re-read on the next hop. The commit point is snapped back
// into a run of blank frames so no character straddles it. When the window
// reaches its maximum, the oldest audio is dropped but holdSec + leftContextSec
// is kept, so there is no hard reset and no character or word is cut at a seam.
//
// Qt-free (std + DeepCwEngine) so the offline replay tool runs exactly this code.

#include "DeepCwEngine.h"

#include <cstddef>
#include <string>
#include <vector>

namespace AetherSDR {

class DeepCwCommitter {
public:
explicit DeepCwCommitter(double holdSec = 5.0, double leftContextSec = 3.0,
double hopSec = 2.0);

struct Result {
bool decoded{false}; // a model decode ran on this push
std::string text; // newly committed characters (may be empty)
float meanConf{1.0f}; // mean posterior of the committed characters
float pitchHz{0.0f}; // dominant tone of the decoded window
};

// Append model-rate (3200 Hz) mono audio; decodes when a hop has elapsed.
Result push(const float* audio3200, std::size_t n, const DeepCwEngine& eng);

// Commit everything still provisional (offline end-of-file use).
Result flush(const DeepCwEngine& eng);

void reset();

double holdSec() const { return m_hold; }
double windowSec() const { return m_window; }

private:
Result decodeAndCommit(const DeepCwEngine& eng, bool flushAll);

double m_hold, m_left, m_hop, m_window, m_keep;
std::vector<float> m_buf;
std::size_t m_bufStart{0}; // absolute sample index of m_buf[0]
std::size_t m_lastDecodeEnd{0};
double m_tCommit{-1.0}; // committed up to this time (s)
double m_tLastChar{-1.0}; // emission time of the last committed letter
char m_lastOut{0};
};

} // namespace AetherSDR
78 changes: 78 additions & 0 deletions src/core/DeepCwEngine.cpp
Original file line number Diff line number Diff line change
Expand Up @@ -189,6 +189,84 @@ std::string DeepCwEngine::ctcDecode(const float* logProbs, int frames, float* av
return text;
}

std::vector<float> DeepCwEngine::inferLogProbs(const std::vector<float>& audio3200, int* frames,
float* pitchHz) const
{
*frames = 0;
if (pitchHz) *pitchHz = 0.0f;
#ifdef HAVE_ONNX
if (!m_loaded || !m_session) { return {}; }
int specFrames = 0;
const std::vector<float> spec = spectrogram(audio3200, &specFrames);
if (specFrames == 0 || spec.empty()) { return {}; }
if (pitchHz) { // same peak-energy-bin estimate as decode()
const double binHz = static_cast<double>(kModelSampleRate) / kFftLength;
const int startBin = static_cast<int>(std::ceil(kMinFreqHz / binHz));
int peak = 0; double peakE = -1.0;
for (int k = 0; k < kFreqBins; ++k) {
double e = 0.0;
for (int f = 0; f < specFrames; ++f) e += spec[static_cast<size_t>(f) * kFreqBins + k];
if (e > peakE) { peakE = e; peak = k; }
}
*pitchHz = static_cast<float>((startBin + peak) * binHz);
}
try {
const std::array<int64_t, 4> shape{1, 1, specFrames, kFreqBins};
Ort::Value input = Ort::Value::CreateTensor<float>(
m_memInfo, const_cast<float*>(spec.data()), spec.size(),
shape.data(), shape.size());
const char* inputNames[] = {"spectrogram"};
const char* outputNames[] = {"log_probs"};
auto outputs = m_session->Run(Ort::RunOptions{nullptr},
inputNames, &input, 1, outputNames, 1);
const float* lp = outputs[0].GetTensorData<float>();
const auto outShape = outputs[0].GetTensorTypeAndShapeInfo().GetShape();
const int outFrames = outShape.size() >= 2
? static_cast<int>(outShape[outShape.size() - 2]) : specFrames;
*frames = outFrames;
return std::vector<float>(lp, lp + static_cast<size_t>(outFrames) * kNumClasses);
} catch (const Ort::Exception& ex) {
std::fprintf(stderr, "DeepCwEngine: inference error: %s\n", ex.what());
return {};
}
#else
(void)audio3200;
return {};
#endif
}

std::vector<DeepCwEngine::Emission> DeepCwEngine::greedyEmissions(
const float* logProbs, int frames, std::vector<uint8_t>* blankFrame) const
{
// Same greedy rule as ctcDecode(): argmax per frame, blank resets, a new
// character on each change of non-blank argmax. Adds the frame index and
// the softmax posterior of the chosen class.
std::vector<Emission> out;
if (blankFrame) blankFrame->assign(static_cast<size_t>(std::max(frames, 0)), 0);
int previous = -1;
for (int t = 0; t < frames; ++t) {
const float* row = logProbs + static_cast<size_t>(t) * kNumClasses;
int best = 0;
float bestVal = row[0];
for (int c = 1; c < kNumClasses; ++c) {
if (row[c] > bestVal) { bestVal = row[c]; best = c; }
}
if (best == kBlankIndex) {
if (blankFrame) (*blankFrame)[static_cast<size_t>(t)] = 1;
previous = -1;
continue;
}
if (best != previous) {
double sumExp = 0.0;
for (int c = 0; c < kNumClasses; ++c)
sumExp += std::exp(static_cast<double>(row[c]) - bestVal);
out.push_back({kChars[best], t, static_cast<float>(1.0 / sumExp)});
}
previous = best;
}
return out;
}

std::string DeepCwEngine::decode(const std::vector<float>& audio, int sampleRateHz,
float* avgConfidence, float* pitchHz) const
{
Expand Down
Loading