Commit 77506207d2 for ffmpeg
commit 77506207d22820db688c0bff409d127b1f7eb22b
Author: Lynne <dev@lynne.ee>
Date: Wed Sep 23 20:29:15 2026 +0900
avcodec/aaccoder_nmr: bounded, escalating reservoir-cap enforcement and a bucket-full spend floor
Three fixes to the CBR corridor's handling of the reservoir.
Bound the coarse-pass cap re-solve. When the coarse pass's trellis bits
exceeded the reservoir cap, its re-solve bisected lambda in [lam, 1e4]
on the coarse candidate grid, whose steppy bit curve can demand a jump
of orders of magnitude - a one-frame excursion that potato-codes the
burst recovery frame: on bass-heavy content the frame after an isolated
transient came out 20-40dB coarser below 500Hz, an audible digital
scratch (ENC_TEST_SAMPLE_05 at 2.1s and 2.8s, with every coding tool
off; err 0.0038 -> 0.0004). Coarsen at most 3x past the corridor there.
The fine pass then works from the bounded lambda, and the reservoir cap
is still enforced on the fine grid after it, where the fit only moves
lambda as far as the frame needs (1.03-1.2x on the test clips; larger
moves happen only at a stream head whose corridor bootstrapped near
lossless, where the cap is what keeps the head at the requested rate).
Corpus neutral at 128k, -2% at 64k, transient panel unchanged.
Escalate it under sustained overage. The debt ledger clips at -rc_bmax:
once saturated, further overage was silently forgiven, and sustained
transient content rode hot forever (castanets 143 kbps on a 128k CBR ask
- its opening roll spent 215 kbps for a full second; tightening the
bound made it worse, since less fit means more forgiven overage). A
one-frame hard fit fixes the rate but re-creates the potato-frame click
the bound exists to prevent (You look good 64k click +53%). The
enforcement now escalates with consecutive saturated frames, counted
once per frame across element groups (any frame without a saturated
overage ends the run): each one widens the allowed coarsening by another
3x step, so a chronic roll converges within a few frames while isolated
bursts never see more than one escalation. Castanets 142.9 -> 134.1
kbps, You-look click at baseline, the E5 recovery-frame scratch stays
fixed, corpus unchanged at both rates.
Bucket-full spend floor. Bits saved beyond the reservoir's remaining
headroom are simply lost (the bucket is capped), so with a full bucket
the plan must never sit below nominal minus what can still be banked.
Without this, a quiet intro poisons the corridor centre high and the
loud entrance rate-limits through it for over a second - listening
feedback flagged a muffled first second on a fade-in sample while our
rate crawled from 47 to 128 kbps over 0.9s where reference encoders
enter flat. Applied in the corridor and in the pre-bootstrap phase, so a
fade-in bootstraps lam_rc at a spending operating point; finer-only,
true silence still codes nothing. Fade-in head now enters at full rate
from the first content frame; corpus standings and the E5 artifact
fixes unchanged.
The id3v2 re-encode refs follow the default NMR output.
diff --git a/libavcodec/aaccoder_nmr.h b/libavcodec/aaccoder_nmr.h
index bb5c4538df..255675a4f3 100644
--- a/libavcodec/aaccoder_nmr.h
+++ b/libavcodec/aaccoder_nmr.h
@@ -85,6 +85,9 @@
/* Corridor: bisect within [lam_rc/NMR_RC_CORR, lam_rc*NMR_RC_CORR] so quality stays
* smooth while per-frame demand is tracked; 1.5 cuts lambda jitter ~25%. */
#define NMR_RC_CORR 1.5f
+/* Reservoir-cap re-solve coarsens at most this far past the corridor per
+ * escalation step; the residual overage rides as reservoir debt. */
+#define NMR_RC_CAPK 3.0f
/* Reservoir half-window (bits/ch); swept 512/1536/3072, 1536 optimal. */
#define NMR_CBR_BUF 1536
@@ -707,7 +710,36 @@ static void nmr_solve_group(AVCodecContext *avctx, AACEncContext *s,
/* legality cap only; no spend-floor (rc_off spends the bank) */
rc_cap = FFMIN(hardcap, (s->nmr->rc_fill + rc_rate_frame + rc_bmax) * chans / s->channels);
if (tot > rc_cap) {
- lam = nmr_solve_slots(s, sl, nsl, cstep, rc_cap, lam, 1e4f, NMR_CITERS);
+ /* reservoir-empty: coarsen, but never past a bounded excursion
+ * of the corridor. The coarse grid's bit curve is steppy: an
+ * unbounded fit here can jump lambda by orders of magnitude and
+ * potato-frame the burst recovery frame (audible LF scratch).
+ * The fine pass works from the bounded lambda, and the reservoir
+ * cap is enforced once more on the fine grid after it, where the
+ * fit only moves lambda as far as the frame really needs. */
+ lam = nmr_solve_slots(s, sl, nsl, cstep, rc_cap, lam, lam * NMR_RC_CAPK, NMR_CITERS);
+ tot = 0;
+ for (int k = 0; k < nsl; k++)
+ tot += nmr_slot_bits(sl[k], s->nmr->nb[sl[k]->si], cstep);
+ if (tot > rc_cap && s->nmr->rc_fill <= -(rc_bmax * 9 / 10)) {
+ /* the debt ledger clips at -rc_bmax: with no capacity
+ * left, further overage would be silently forgiven and
+ * sustained transient content rides hot forever (a
+ * castanets roll hit 215 kbps for its first second on a
+ * 128k ask). A one-frame hard fit trades that for a
+ * potato frame (audible click); instead the bound
+ * ESCALATES with consecutive saturated frames - chronic
+ * rolls converge within a few frames, isolated bursts
+ * never see more than one escalation step. */
+ float ek = NMR_RC_CAPK * (1 + FFMIN(s->nmr->rc_satrun, 8));
+ lam = nmr_solve_slots(s, sl, nsl, cstep, rc_cap, lam, lam * ek, NMR_CITERS);
+ s->nmr->rc_sat_frame = 1;
+ tot = 0;
+ for (int k = 0; k < nsl; k++)
+ tot += nmr_slot_bits(sl[k], s->nmr->nb[sl[k]->si], cstep);
+ }
+ if (tot > hardcap) /* decoder-buffer legality is absolute */
+ lam = nmr_solve_slots(s, sl, nsl, cstep, hardcap, lam, 1e4f, NMR_CITERS);
}
} else {
/* per-frame bisection, warm-started off the previous frame's lambda;
@@ -764,7 +796,8 @@ static void nmr_solve_group(AVCodecContext *avctx, AACEncContext *s,
lam_dem = lam; /* demand-solved lambda, pre bucket clamp: what content wants */
if (rc_global) {
- /* legality clamp, then the quality slew limiter */
+ /* reservoir cap on the fine grid (see the coarse-pass bound above),
+ * then the quality slew limiter */
int hardcap = av_clip((int)(5800.f * FFMIN(1.f, lambda / 120.f)), 256, 5800) * chans;
int tot = 0, rc_cap;
for (int k = 0; k < nsl; k++)
@@ -773,6 +806,21 @@ static void nmr_solve_group(AVCodecContext *avctx, AACEncContext *s,
if (tot > rc_cap) {
lam = nmr_solve_slots(s, sl, nsl, NMR_STEP, rc_cap, lam, 1e4f, NMR_RC_ITERS);
}
+ { /* bucket-full spend floor: bits saved beyond the reservoir's
+ * remaining headroom are simply lost, so with a full bucket the
+ * plan must not sit below nominal minus what can still be
+ * banked. Without this a quiet intro poisons the corridor high
+ * and the loud entrance rate-limits through it - a muffled
+ * first second at the exact moment the listener tunes in. */
+ int headroom = rc_bmax - av_clip(s->nmr->rc_fill, -rc_bmax, rc_bmax);
+ int fbits = (rc_rate_frame - headroom) * chans / s->channels;
+ if (tot < fbits) {
+ lam = nmr_solve_slots(s, sl, nsl, NMR_STEP, fbits, lam / 64.0f, lam, NMR_RC_ITERS);
+ tot = 0;
+ for (int k = 0; k < nsl; k++)
+ tot += nmr_slot_bits(sl[k], s->nmr->nb[sl[k]->si], NMR_STEP);
+ }
+ }
if (s->nmr->lam_slew > 0.0f) {
float kup, kdn;
/* hold lambda near-constant within short runs; bits follow content */
@@ -791,6 +839,17 @@ static void nmr_solve_group(AVCodecContext *avctx, AACEncContext *s,
}
}
s->nmr->lam_slew = lam;
+ } else if (rc_eligible) {
+ /* corridor not yet bootstrapped: the same bucket-full floor, so a
+ * fade-in bootstraps lam_rc at a spending operating point instead
+ * of memorizing the intro's starvation lambda */
+ int headroom = rc_bmax - av_clip(s->nmr->rc_fill, -rc_bmax, rc_bmax);
+ int fbits = (rc_rate_frame - headroom) * chans / s->channels;
+ int tot = 0;
+ for (int k = 0; k < nsl; k++)
+ tot += nmr_slot_bits(sl[k], s->nmr->nb[sl[k]->si], NMR_STEP);
+ if (tot < fbits)
+ lam = nmr_solve_slots(s, sl, nsl, NMR_STEP, fbits, lam / 64.0f, lam, NMR_RC_ITERS);
}
for (int k = 0; k < nsl; k++)
@@ -984,6 +1043,10 @@ static void search_for_quantizers_nmr(AVCodecContext *avctx,
-rc_bmax, rc_bmax);
n->rc_frame_num = avctx->frame_num;
n->pending = 0; /* a deferred first channel never crosses a frame */
+ /* consecutive saturated frames, once per frame across all element
+ * groups: any frame without a saturated overage ends the run */
+ n->rc_satrun = n->rc_sat_frame ? n->rc_satrun + 1 : 0;
+ n->rc_sat_frame = 0;
/* latch the RC mode per frame: a mid-frame bootstrap must not flip
* the CPE defer logic between channels */
n->rc_gl = rc_eligible && n->lam_rc > 0.0f;
diff --git a/libavcodec/aacenc.h b/libavcodec/aacenc.h
index dfd5c23a2b..08bc0fb961 100644
--- a/libavcodec/aacenc.h
+++ b/libavcodec/aacenc.h
@@ -234,6 +234,8 @@ typedef struct AACNMRCurves {
int64_t rc_frame_num; ///< frame the reservoir was last advanced for
float lam_rc; ///< global-lambda rate control: operating lambda, 0 until bootstrapped
int rc_fill; ///< virtual bit reservoir fill, + = bits saved vs nominal
+ int rc_satrun; ///< consecutive frames with saturated reservoir debt (cap escalation)
+ int rc_sat_frame; ///< the current frame hit a saturated overage
int frames_since_short; ///< long-block frames since the last short run (the "gap"): large = isolated transient
int prev_was_short; ///< previous frame was a short block (for run-start detection)
float run_burst; ///< transient bit-burst factor, set at run start and held across the short run