Round 4 of Task 4 review. Four defects in FrameDecoder's rule 3: - The undeclared-rate (hrt) path positioned each burst at an ABSOLUTE hrt/ticksPerSecond(). hrt counts from the producer's boot, so it is ~1e11 ticks by the time a scope attaches, and the rate is refitted every packet with a few parts in 1e4 of wobble. The product is tens of milliseconds of jitter in BOTH directions -- not merely imprecise, non-monotonic. Integrate short tick deltas into accProdSec instead and let ClockOffset latch the epoch that leaves behind. - The lead bleed used a fixed 0.9 factor, which converges only while the declared rate is within ~10%. Squeeze proportionally to the excess instead (floored at kMinBleedFactor), settling it in a single burst. - A single-sample flush fell through to the plain-scalar rule, dating it from arrival and leaving lastCounter stale so the next real burst reinstated a hole that never existed. Accumulate mode flushes on a timer, so a short cycle legitimately yields one sample; keep it on the chain. - kMaxCounterGap was inert: an absurd gap yields an absurd prediction that the arrival backstop already rejects, and no input can distinguish the two rules. Removed rather than left implying a behaviour it did not have. FrameDecoder.h now states the deliberate divergence from StreamHub -- which converts hrt with the LOCAL MARTe timer frequency, valid only because it runs on the producer's host -- and why a remote scope's drift is irreducible. Three new tests, each sabotage-proven non-vacuous: producer restart, short flushes staying on the chain, and a 20000-packet undeclared run after a day of producer uptime that asserts SPACING as well as ordering (the monotonic guard alone restores order while leaving positions wrong). Co-Authored-By: Claude Opus 4.6 <noreply@anthropic.com>
324 lines
16 KiB
C++
324 lines
16 KiB
C++
#include "FrameDecoder.h"
|
|
|
|
#include <cmath>
|
|
|
|
namespace udpscope {
|
|
|
|
/** Fallback cycle period before the first inter-packet gap is known. */
|
|
static constexpr double kDefaultDt = 1.0e-3;
|
|
|
|
/**
|
|
* How far a chained burst prediction may sit from where arrival time says it
|
|
* should be before the chain is abandoned and time is re-anchored on arrival.
|
|
*
|
|
* This is a backstop, not the primary mechanism: the packet counter normally
|
|
* accounts for lost datagrams exactly, so the prediction and arrival agree.
|
|
* It catches what the counter cannot describe — a producer restart (the
|
|
* counter returns to zero), a counter that never advances, and a declared
|
|
* sampling rate that does not match the producer's real one. A kernel draining
|
|
* a backlog of queued datagrams can legitimately put the prediction a couple of
|
|
* hundred milliseconds from arrival, so the threshold sits well clear of that.
|
|
* Same value and same reasoning as ClockOffset::kRecalibThresholdS.
|
|
*/
|
|
static constexpr double kBurstResyncThresholdS = 0.5;
|
|
|
|
/**
|
|
* Narrowest a burst may be drawn, as a fraction of its nominal width, while a
|
|
* leading timeline is being pulled back. Only a floor: the squeeze is normally
|
|
* proportional to the excess and removes it in a single burst. See the sole use
|
|
* site.
|
|
*/
|
|
static constexpr double kMinBleedFactor = 0.05;
|
|
|
|
void FrameDecoder::setSignals(const std::vector<SignalMeta>& signals) {
|
|
signals_ = signals;
|
|
state_.assign(signals_.size(), SigState{});
|
|
hrtFit_.reset();
|
|
}
|
|
|
|
void FrameDecoder::reset() {
|
|
state_.assign(signals_.size(), SigState{});
|
|
hrtFit_.reset();
|
|
}
|
|
|
|
void FrameDecoder::beginFrame(const FrameView& f) {
|
|
if (f.hrt != 0u) { hrtFit_.add(f.hrt, f.recvTime); }
|
|
}
|
|
|
|
bool FrameDecoder::packetBurst(uint32_t idx, uint32_t nElems, double wallNow,
|
|
std::vector<double>& tsOut) {
|
|
SigState& st = state_[idx];
|
|
if (!st.lastPacketValid || wallNow <= st.lastPacketWall) {
|
|
/* No previous arrival to span from, or time went backwards. Remember
|
|
* this one and drop the samples rather than store them at made-up
|
|
* spacing. */
|
|
st.lastPacketWall = wallNow;
|
|
st.lastPacketValid = true;
|
|
return false;
|
|
}
|
|
|
|
const double dt = (wallNow - st.lastPacketWall) / static_cast<double>(nElems);
|
|
tsOut.resize(nElems);
|
|
for (uint32_t e = 0; e < nElems; e++) {
|
|
tsOut[e] = st.lastPacketWall + static_cast<double>(e + 1u) * dt;
|
|
}
|
|
st.lastPacketWall = wallNow;
|
|
return true;
|
|
}
|
|
|
|
bool FrameDecoder::timestamps(const FrameView& f, uint32_t idx,
|
|
std::vector<double>& tsOut) {
|
|
tsOut.clear();
|
|
if (idx >= signals_.size() || idx >= f.numSignals ||
|
|
f.counts == nullptr || f.values == nullptr) {
|
|
return false;
|
|
}
|
|
|
|
const SignalMeta& d = signals_[idx];
|
|
const uint32_t nElems = f.counts[idx];
|
|
if (nElems == 0u) { return false; }
|
|
|
|
const double wallNow = f.recvTime;
|
|
SigState& st = state_[idx];
|
|
|
|
/* A repeated counter is a duplicated datagram — the same update arriving
|
|
* twice because the host joined the multicast group on two interfaces, say.
|
|
* The C client only de-duplicates fragments, so an unfragmented update
|
|
* reaches us intact both times; emitting it again would double the values
|
|
* and advance the timeline by a burst that never existed. Counter zero is
|
|
* excluded because a producer that never sets one leaves it there. */
|
|
if (st.lastEmittedValid && f.counter != 0u && f.counter == st.lastCounter) {
|
|
return false;
|
|
}
|
|
|
|
/* hasTimeSignal() bounds the index against the FRAME's signal count, but
|
|
* the time signal's type code is read from our own table, whose size is
|
|
* independent — a frame carrying more signals than the installed table
|
|
* (briefly possible after a CONFIG change) would otherwise read past it. */
|
|
const bool hasTimeSig = d.hasTimeSignal(f.numSignals) &&
|
|
d.timeSignalIdx < signals_.size();
|
|
const uint32_t tIdx = hasTimeSig ? d.timeSignalIdx : 0u;
|
|
const double tScale = hasTimeSig
|
|
? TimeSignalScale(signals_[tIdx].typeCode)
|
|
: 1.0e-6;
|
|
|
|
/* Rule 1: one stamp per element, straight from the time signal. */
|
|
if (d.timeMode == kTimeFullArray && hasTimeSig &&
|
|
f.counts[tIdx] >= nElems && f.values[tIdx] != nullptr) {
|
|
const double* tv = f.values[tIdx];
|
|
const double t0 = tv[0] * tScale;
|
|
(void) st.offset.map(t0, wallNow);
|
|
const double base = st.offset.offset();
|
|
tsOut.resize(nElems);
|
|
for (uint32_t e = 0; e < nElems; e++) {
|
|
tsOut[e] = base + tv[e] * tScale;
|
|
}
|
|
return true;
|
|
}
|
|
|
|
/* Rule 2: anchor from the time signal, spread by the sampling rate. */
|
|
if ((d.timeMode == kTimeFirstSample || d.timeMode == kTimeLastSample) &&
|
|
hasTimeSig && f.counts[tIdx] >= 1u && f.values[tIdx] != nullptr) {
|
|
const double anchor = st.offset.map(f.values[tIdx][0] * tScale, wallNow);
|
|
const double dt = (d.samplingRate > 0.0) ? (1.0 / d.samplingRate) : 0.0;
|
|
tsOut.resize(nElems);
|
|
for (uint32_t e = 0; e < nElems; e++) {
|
|
tsOut[e] = (d.timeMode == kTimeFirstSample)
|
|
? (anchor + static_cast<double>(e) * dt)
|
|
: (anchor - static_cast<double>(nElems - 1u - e) * dt);
|
|
}
|
|
return true;
|
|
}
|
|
|
|
/* Rule 3: accumulated scalar, based on declared sampling rate or hrt.
|
|
*
|
|
* When samplingRate is declared the inter-element step is exact and we
|
|
* anchor from the end of the previous burst rather than from arrival time
|
|
* or hrt. This makes the output immune to arrival jitter: even when the
|
|
* kernel delivers two packets microseconds apart each burst starts exactly
|
|
* one sample period after the previous burst ended.
|
|
*
|
|
* When samplingRate is absent we must derive dt from the hrt gap, which
|
|
* requires the HrtRateFit to be ready. Until then we fall back to
|
|
* packetBurst (arrival-time spanning), which is accurate during the normal
|
|
* pre-burst delivery phase that precedes the fit becoming ready.
|
|
*
|
|
* A signal that has already produced a burst stays on this rule even when a
|
|
* later packet carries a single sample — Accumulate mode flushes on a timer,
|
|
* so a short cycle legitimately yields one. Dropping such a packet to rule 5
|
|
* would date it from arrival while its neighbours are chained, and would
|
|
* leave lastCounter behind so the next real burst read the skip as a lost
|
|
* datagram and reinstated a hole that never existed. A signal that has never
|
|
* burst is a genuine scalar and is left to rule 5. */
|
|
if (d.numElements() == 1u && (nElems > 1u || st.lastEmittedValid)) {
|
|
const double dt = (d.samplingRate > 0.0)
|
|
? (1.0 / d.samplingRate)
|
|
: 0.0;
|
|
|
|
if (d.samplingRate > 0.0) {
|
|
/* Where arrival time says this burst begins: its last element was
|
|
* acquired just before the packet landed. */
|
|
const double arrivalAnchor =
|
|
wallNow - static_cast<double>(nElems - 1u) * dt;
|
|
|
|
/* Chaining onto the end of the previous burst is immune to arrival
|
|
* jitter — a kernel draining several queued datagrams microseconds
|
|
* apart still yields contiguous timestamps. What a bare chain gets
|
|
* wrong is loss: it closes the hole a dropped datagram left, and
|
|
* every later sample is then dated early for the rest of the run.
|
|
*
|
|
* The wire says exactly how much is missing. counter increments
|
|
* once per update, so a gap of g means g-1 lost packets, each
|
|
* carrying (as far as we can tell) as many samples as the last one
|
|
* we saw. Reinstating that duration keeps the chain honest without
|
|
* consulting arrival time at all. */
|
|
double base = arrivalAnchor;
|
|
double step = dt;
|
|
if (st.lastEmittedValid) {
|
|
/* Unsigned subtraction wraps, so this stays right across the
|
|
* counter's own 2^32 rollover.
|
|
*
|
|
* A producer restart or a reordered datagram makes the wrapped
|
|
* gap enormous, and this deliberately does NOT special-case
|
|
* that: an absurd gap yields an absurd prediction, which the
|
|
* arrival backstop below then rejects on its own. Clamping the
|
|
* gap first would only decide the same question earlier, by a
|
|
* second rule that no stream can distinguish from this one. */
|
|
const uint32_t gap = f.counter - st.lastCounter;
|
|
const double lost = (gap > 1u)
|
|
? static_cast<double>(gap - 1u) *
|
|
static_cast<double>(st.prevAccCount)
|
|
: 0.0;
|
|
const double predicted = st.lastEmittedEnd + dt * (1.0 + lost);
|
|
|
|
/* Backstop for what the counter cannot express: a producer
|
|
* restart, a counter stuck at zero, or a declared rate that is
|
|
* simply wrong. Beyond this the chain is not recoverable and
|
|
* arrival time is the better of two bad answers. */
|
|
if (std::fabs(predicted - arrivalAnchor) <= kBurstResyncThresholdS) {
|
|
base = predicted;
|
|
}
|
|
|
|
if (base <= st.lastEmittedEnd) {
|
|
/* Re-anchoring here would step backwards, and the ring, the
|
|
* trigger and the exporter all require a signal's stamps to
|
|
* increase. Rejecting the correction outright is not an
|
|
* option either: `predicted` is never less than
|
|
* lastEmittedEnd + dt, so rejection would make the backstop
|
|
* one-directional and let a timeline that runs FAST — two
|
|
* hosts' crystals differ by tens of ppm, so this is certain
|
|
* on a long session, not hypothetical — drift ahead of the
|
|
* wall clock without bound.
|
|
*
|
|
* So compress instead of stepping back: start immediately
|
|
* after the previous burst and spread this one out to
|
|
* arrival. A single packet is drawn narrower than its true
|
|
* width, and in exchange the timeline is back in step. */
|
|
if (wallNow > st.lastEmittedEnd) {
|
|
step = (wallNow - st.lastEmittedEnd) /
|
|
static_cast<double>(nElems);
|
|
base = st.lastEmittedEnd + step;
|
|
} else {
|
|
/* The timeline has run PAST arrival: our last burst is
|
|
* dated later than the moment this packet landed, so
|
|
* there is no room to spread into and no burst can end
|
|
* on arrival without starting before it. Squeeze this
|
|
* one by exactly the excess instead. That lands its end
|
|
* one nominal burst ahead of arrival — the closest a
|
|
* forward-only timeline can legally get — and the excess
|
|
* settles at (nominal width - true burst period), a
|
|
* couple of hundred microseconds for the ppm-scale
|
|
* crystal mismatch that causes this.
|
|
*
|
|
* The floor keeps the step positive when the excess is
|
|
* larger than a whole burst (a declared rate that is
|
|
* wrong by a factor, not by ppm). It only slows the
|
|
* recovery: each burst then advances by almost nothing
|
|
* while arrival keeps advancing, so the excess still
|
|
* falls to zero, just over several packets. */
|
|
const double nominal = static_cast<double>(nElems) * dt;
|
|
const double excess = st.lastEmittedEnd - wallNow;
|
|
double factor = 1.0 - excess / nominal;
|
|
if (factor < kMinBleedFactor) { factor = kMinBleedFactor; }
|
|
step = dt * factor;
|
|
base = st.lastEmittedEnd + step;
|
|
}
|
|
}
|
|
}
|
|
tsOut.resize(nElems);
|
|
for (uint32_t e = 0; e < nElems; e++) {
|
|
tsOut[e] = base + static_cast<double>(e) * step;
|
|
}
|
|
st.lastEmittedEnd = tsOut[nElems - 1u];
|
|
st.lastCounter = f.counter;
|
|
st.prevAccCount = nElems;
|
|
st.lastEmittedValid = true;
|
|
return true;
|
|
}
|
|
|
|
/* No declared rate: need hrt-derived dt. */
|
|
if (!hrtFit_.ready() || f.hrt == 0u) {
|
|
return packetBurst(idx, nElems, wallNow, tsOut);
|
|
}
|
|
const double rate = hrtFit_.ticksPerSecond();
|
|
|
|
/* Integrate short tick DELTAS. Never convert an absolute tick count, and
|
|
* never subtract two such conversions.
|
|
*
|
|
* hrt counts from the producer's boot, so it is already ~1e11 ticks when
|
|
* the scope attaches, while the fit is re-estimated on every packet and
|
|
* wobbles by a few parts in 1e4. Any absolute hrt/rate therefore carries
|
|
* that relative wobble multiplied by the whole elapsed epoch — tens of
|
|
* milliseconds, moving in either direction from one packet to the next.
|
|
* As a burst's position that is not merely imprecise, it is
|
|
* NON-MONOTONIC: on a 2 h stream with ordinary scheduling jitter a few
|
|
* percent of samples land before their own predecessor.
|
|
*
|
|
* A delta spans one packet, so its share of the wobble is microseconds,
|
|
* and summing deltas keeps it there. ClockOffset then latches the
|
|
* arbitrary epoch that leaves behind, exactly as it would have latched
|
|
* the producer's boot epoch. */
|
|
double elapsed = 0.0;
|
|
if (st.lastAccValid && f.hrt > st.lastAccHrt) {
|
|
elapsed = static_cast<double>(f.hrt - st.lastAccHrt) / rate;
|
|
}
|
|
st.accProdSec += elapsed;
|
|
double base = st.offset.map(st.accProdSec, wallNow);
|
|
|
|
/* The flushes carry contiguous RT cycles, so the gap divided by the
|
|
* previous packet's sample count is exactly one cycle period. */
|
|
const double hrtDt = (elapsed > 0.0 && st.prevAccCount > 0u)
|
|
? (elapsed / static_cast<double>(st.prevAccCount))
|
|
: kDefaultDt;
|
|
|
|
/* ClockOffset recalibrates once true drift passes its threshold, and a
|
|
* recalibration can land behind where this signal already is.
|
|
* Downstream requires increasing stamps, so step forward minimally. */
|
|
if (st.lastEmittedValid && base <= st.lastEmittedEnd) {
|
|
base = st.lastEmittedEnd + hrtDt;
|
|
}
|
|
|
|
tsOut.resize(nElems);
|
|
for (uint32_t e = 0; e < nElems; e++) {
|
|
tsOut[e] = base + static_cast<double>(e) * hrtDt;
|
|
}
|
|
st.lastAccHrt = f.hrt;
|
|
st.lastAccValid = true;
|
|
st.prevAccCount = nElems;
|
|
st.lastEmittedEnd = tsOut[nElems - 1u];
|
|
st.lastEmittedValid = true;
|
|
return true;
|
|
}
|
|
|
|
/* Rule 4: PACKET burst with no time reference at all. */
|
|
if (nElems > 1u) {
|
|
return packetBurst(idx, nElems, wallNow, tsOut);
|
|
}
|
|
|
|
/* Rule 5: plain scalar. */
|
|
tsOut.assign(1, wallNow);
|
|
return true;
|
|
}
|
|
|
|
} /* namespace udpscope */
|