FFmpeg
Loading...
Searching...
No Matches
aaccoder_nmr.h
Go to the documentation of this file.
1/*
2 * AAC encoder NMR (noise-to-mask ratio) scalefactor coder
3 * Copyright (c) 2026 Lynne <dev@lynne.ee>
4 *
5 * This file is part of FFmpeg.
6 *
7 * FFmpeg is free software; you can redistribute it and/or
8 * modify it under the terms of the GNU Lesser General Public
9 * License as published by the Free Software Foundation; either
10 * version 2.1 of the License, or (at your option) any later version.
11 *
12 * FFmpeg is distributed in the hope that it will be useful,
13 * but WITHOUT ANY WARRANTY; without even the implied warranty of
14 * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU
15 * Lesser General Public License for more details.
16 *
17 * You should have received a copy of the GNU Lesser General Public
18 * License along with FFmpeg; if not, write to the Free Software
19 * Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA
20 */
21
22/**
23 * AAC encoder NMR scalefactor coder.
24 *
25 * Optimizes the same noise-to-mask objective as the two-loop coder, but with an
26 * optimal Viterbi search over scalefactors instead of a heuristic loop. For each
27 * coded band the per-scalefactor distortion/bits curve is precomputed, then a
28 * trellis over the (window-group, band) coding sequence minimizes
29 * sum_g = dist_g(sf_g)/threshold_g +
30 * lambda * (spectral_bits_g(sf_g) + scalefactor_differential_bits)
31 * with |sf_g - sf_{g-1}| <= SCALE_MAX_DIFF as a constraint, and lambda
32 * binary-searched so the coded size meets the per-frame bit budget
33 *
34 * Perceptual noise substitution (PNS) is integrated into the same objective: once
35 * the trellis settles on its operating lambda, each noise-like band (flagged by
36 * mark_pns) is offered a terminal "code as noise" candidate whose cost is
37 * nmr_pns + lambda*NMR_PNS_BITS. Because NMR_PNS_BITS is far below a band's spectral bit
38 * count, this candidate only wins when lambda is large, i.e. when the encoder is
39 * struggling to hold the bitrate. The bits freed by the chosen PNS bands are
40 * then re-spent by a second trellis pass over the remaining bands.
41 */
42
43#ifndef AVCODEC_AACCODER_NMR_H
44#define AVCODEC_AACCODER_NMR_H
45
46#include <float.h>
47#include <string.h>
49#include "mathops.h"
50#include "avcodec.h"
51#include "put_bits.h"
52#include "aac.h"
53#include "aacenc.h"
54#include "aactab.h"
55#include "aacenctab.h"
56
57/* differential scalefactor coding cost, clamped to the legal delta range */
58#define NMR_SFBITS(d) ff_aac_scalefactor_bits[av_clip((d) + SCALE_DIFF_ZERO, 0, 2*SCALE_MAX_DIFF)]
59
60#define NMR_ITERS 14 /* lambda binary-search iters */
61#define NMR_IFINE 9 /* fine-pass lambda iters */
62#define NMR_CITERS 7 /* coarse-pass lambda iters */
63#define NMR_CWARM 5 /* coarse-pass iters when warm-started off the previous frame's
64 * lambda: the bracket spans 10 octaves instead of ~43, so fewer
65 * bisection steps reach the same resolution */
66#define NMR_COARSE 8 /* two-pass coarse->fine grid step, cuts the Viterbi ncand^2 with no
67 * quality loss, 0 disables it (single full-resolution pass) */
68#define NMR_STEP 1 /* fine-pass scalefactor candidate granularity */
69
70#define NMR_PNS_BITS 9 /* approx cost in bits of signalling PNS */
72/* Spectral-hole fill: noise-like bands the trellis left mostly empty are filled with
73 * energy-matched noise (PNS); an audible hole sounds worse than matched noise. */
74#define NMR_PNS_HOLE_FRAC 0.5f
75#define NMR_PNS_HOLE_SPREAD 0.5f
76/* Hole fill applies at any operating lambda above this (comfortable rates
77 * included): sparse-line renditions of noise are audible at every rate. */
78#define NMR_PNS_HOLE_LAM 20.0f
80/* RC servo gain: scale the corridor centre by exp2(-K*fill/R) each frame to hold
81 * the long-run mean rate; without it a bad centre drifts for dozens of frames. */
82#define NMR_RC_K_CBR 0.5f
83
84#define NMR_RC_ITERS 8 /* lambda bisection iters when clamping an over-cap frame */
85/* Corridor: bisect within [lam_rc/NMR_RC_CORR, lam_rc*NMR_RC_CORR] so quality stays
86 * smooth while per-frame demand is tracked; 1.5 cuts lambda jitter ~25%. */
87#define NMR_RC_CORR 1.5f
88/* Reservoir-cap re-solve coarsens at most this far past the corridor per
89 * escalation step; the residual overage rides as reservoir debt. */
90#define NMR_RC_CAPK 3.0f
91
92/* Quality-target calibration anchor: the nd set-point of -q:a 1, which lands
93 * near 128-136 kbps stereo on the tuning corpora with the q ladder's own
94 * bandwidth. The ABR servo seeds off the same constant (referenced to 81.5
95 * kbps/ch, where it measured before the bandwidth retune; the boot corrects
96 * the per-content seed error), so they are one constant. It moved
97 * -2.1 -> -0.75 with the quality-target mask floor (PSY_THRFL_QUALITY), which
98 * lowers thr_real and so lifts the whole statistic.
99 * The reachable-range clamps did NOT move
100 * with it: their floors are set by the stat's -4/band sub-mask clip, which the
101 * mask floor does not touch. Targets below them are asymptotically unreachable
102 * - the bisect rails at the finest lambda, the frame exceeds the decoder buffer
103 * and the outer re-encode loop never converges (the 384k stall). Raising them
104 * with the anchor railed the ABR servo instead: it wanted a finer target to
105 * meet its rate ask on easy content and could not ask for one. */
106#define NMR_VBR_ANCHOR (-0.75f)
107#define NMR_VBR_TMIN (-3.8f)
108/* ABR set-point range. The coarse end is a quality guard: set-points past it
109 * buy little rate for a lot of quality (2.5 -> 6.0 at a 48k ask: -5% rate,
110 * +32% Zimtohrli), so asks below ~56 kbps stereo land above the ask - CBR
111 * is the mode for those rates. At the fine end, easy content can fall short
112 * of very high asks (no padding). */
113#define NMR_ABR_TMIN (-3.0f)
114#define NMR_ABR_TMAX 2.5f
116/* ABR servo: integrator gain on the log rate error, rate EMA weight after the
117 * boot, set-point step size and minimum frames between steps */
118#define NMR_ABR_K 0.003f
119#define NMR_ABR_EMA 0.0023f
120#define NMR_ABR_STEP 0.15f
121#define NMR_ABR_HOLD 120
122/* ABR boot: open-loop correction gain over the nominal loop gain, the rate
123 * error that re-arms it, the re-arm budget and the settle time before one */
124#define NMR_ABR_BOOT_GAIN 1.2f
125#define NMR_ABR_TOL 0.02f
126#define NMR_ABR_BOOTS 6
127#define NMR_ABR_SETTLE 100
129/* VBR shorts: the measured stat inflation counts fully once this share of
130 * the frame's energy sits above 6 kHz */
131#define NMR_INFL_HF 0.10f
132
133/* Reservoir half-window (bits/ch); swept 512/1536/3072, 1536 optimal. */
134#define NMR_CBR_BUF 1536
135/* Slew limit on the FINAL operating lambda per frame; bits deviate instead,
136 * the reservoir absorbs. See memory: aac-castanets-transient-rc. */
137#define NMR_SLEW 1.6f
138#define NMR_SLEW_RUN 1.15f /* within short runs */
139#define NMR_RC_CITERS 3 /* corridor coarse-pass iters */
141/* Transition premask: an attack cannot mask backwards; clamp a START frame's
142 * thresholds toward the previous long frame's. */
143#define NMR_TRANS_PM 2.0f
145/* Zero-decision hysteresis: previously-coded bands need this margin below
146 * threshold to zero (marginal bands flicker audibly otherwise). */
147#define NMR_ZERO_STICKY 0.5f
149/* Short-window groups keep a borderline window coded down to this fraction
150 * of its zero threshold rather than muting it inside a coded group. */
151#define NMR_GROUP_KEEP 0.25f
153/* HF precision taper: thresholds rise by NMR_HF_TAPER dB per kHz^2 above the
154 * knee (Apple's measured HF precision rolloff). */
155#define NMR_HF_TAPER 0.08f
156#define NMR_HF_TAPER_KNEE 8000.0f
158/* Transient bit-burst: an isolated onset (preceded by >= NMR_BURST_GAP long frames)
159 * is coded NMR_BURST_GAIN x finer, held uniform across the run, repaid from steady stretches. */
160#define NMR_BURST_GAP 10
161#define NMR_BURST_GAIN 8.0f
162/* Dense-beat boost: short runs with gap < NMR_BURST_GAP get a budget factor
163 * ramping with the gap (starvation-scaled at the use site). */
164#define NMR_SHORT_BOOST 2.0f
165#define NMR_RC_FITERS 4 /* corridor fine-pass iters */
166#define NMR_RC_TRACK 0.1f /* per-frame pull of the corridor centre toward the realized lambda */
167
168/* PNS noise-distortion gate: only bands coded well above the masking floor become noise. */
169#define NMR_PNS_NDGATE 4.0f
171/* Energy/threshold cap for PNS: loud bands (energy >> mask) yield clipping random peaks;
172 * only near-masked bands are safe substitution targets. */
173#define NMR_PNS_MAX_ET 8.0f
175/* Operating-lambda floor for PNS: below it the encoder is not struggling, so
176 * substituting real texture for 9 signalling bits is net-negative. */
177#define NMR_PNS_LAM 100.0f
179/* PNS decision hysteresis: enter and leave both cost a margin. */
180#define NMR_PNS_ENTER 0.7f
181#define NMR_PNS_STAY 1.4f
182/* PNS debounce: enter after NMR_PNS_ON consecutive wants, leave after
183 * NMR_PNS_OFF (chronically marginal bands never qualify). */
184#define NMR_PNS_ON 8
185#define NMR_PNS_OFF 4
186
187/**
188 * Viterbi over the coding sequence act[0..nact-1] (indices into the per-band
189 * curves nd/nb), with lambda binary-searched so the coded size ~ destbits.
190 * Fills chosen[band] for every band referenced by act. Returns the operating
191 * lambda. node cost = dist/threshold + lambda*spectral_bits;
192 * edge cost = lambda*sf_differential_bits; |delta sf| <= SCALE_MAX_DIFF hard.
193 */
194static float nmr_solve(AACEncContext *s,
195 const float (*nd)[NMR_NCAND], const int (*nb)[NMR_NCAND],
196 const int *blo, const int *bnc, int step,
197 const int *act, int nact, int destbits, int *chosen,
198 float lo_l, float hi_l, int iters)
199{
200 float dp[NMR_NCAND], dpp[NMR_NCAND], node[NMR_NCAND];
201 float lamsf[2*SCALE_MAX_DIFF + 1]; /* lam*sfdiff bit cost, per lambda */
202 uint8_t bp[128][NMR_NCAND];
203 float lam = 1.0f;
204
205 if (nact <= 0)
206 return lam;
207
208 for (int it = 0; it < iters; it++) {
209 lam = sqrtf(lo_l * hi_l);
210 for (int i = 0; i <= 2*SCALE_MAX_DIFF; i++)
211 lamsf[i] = lam * ff_aac_scalefactor_bits[i]; /* edge cost for this lambda */
212
213 int b0 = act[0];
214 for (int o = 0; o < bnc[b0]; o++)
215 dp[o] = nd[b0][o] + lam * nb[b0][o]; /* anchor band node cost */
216
217 for (int k = 1; k < nact; k++) {
218 int b = act[k], pb = act[k-1];
219 memcpy(dpp, dp, sizeof(dp));
220 for (int o = 0; o < bnc[b]; o++)
221 node[o] = nd[b][o] + lam * nb[b][o];
222 /* dp[o] = node[o] + min_op(dpp[op] + edge cost) */
223 s->aacdsp.nmr_trellis_step(dp, bp[k], dpp, node, lamsf,
224 bnc[b], bnc[pb], blo[b] - blo[pb], step,
226 }
227
228 /* backtrack */
229 int beo = 0, b = act[nact-1];
230 float bec = FLT_MAX;
231 for (int o = 0; o < bnc[b]; o++)
232 if (dp[o] < bec) { bec = dp[o]; beo = o; }
233 chosen[b] = beo;
234 for (int k = nact-1; k > 0; k--)
235 chosen[act[k-1]] = bp[k][chosen[act[k]]];
236
237 /* calc cost */
238 int total = 0;
239 for (int k = 0; k < nact; k++)
240 total += nb[act[k]][chosen[act[k]]];
241 for (int k = 1; k < nact; k++)
242 total += NMR_SFBITS((blo[act[k]]+chosen[act[k]]*step) - (blo[act[k-1]]+chosen[act[k-1]]*step));
243
244 if (it == iters - 1)
245 break;
246
247 /* check if we went over budget, go coarser if we did */
248 if (total > destbits)
249 lo_l = lam;
250 else
251 hi_l = lam;
252 }
253 return lam;
254}
256/* Build one coded band's (dist/threshold, bits) cost curve, candidates sf = lo + o*step
257 * for o in [0,maxn), stopping when the band would drop (cb <= 0). Returns the bit count. */
258static int nmr_band_curve(AACEncContext *s, SingleChannelElement *sce, int w, int g,
259 int start, int lo, int step, int maxn, float invthr,
260 float maxval, float *nd_row, int *nb_row)
261{
262 int ncand = 0;
263 for (int o = 0; o < maxn && lo + o*step <= SCALE_MAX_POS; o++) {
264 int sf = lo + o*step, btot = 0, cb = find_min_book(maxval, sf);
265 float dist = 0.0f;
266 if (cb <= 0)
267 break;
268 for (int w2 = 0; w2 < sce->ics.group_len[w]; w2++) {
269 int bb;
270 dist += quantize_band_cost_cached(s, w + w2, g, sce->coeffs + start + w2*128,
271 s->scoefs + start + w2*128, sce->ics.swb_sizes[g],
272 sf, cb, 1.0f, INFINITY, &bb, NULL, 0);
273 btot += bb;
274 }
275 nd_row[ncand] = (dist - btot) * invthr;
276 nb_row[ncand] = btot;
277 ncand++;
278 }
279 return ncand;
280}
282/* Zero a channel with nothing codeable; stale band_types would resurrect
283 * bands with chain-illegal scalefactors. */
285{
286 for (int i = 0; i < 128; i++) {
287 if (sce->band_type[i] == INTENSITY_BT || sce->band_type[i] == INTENSITY_BT2)
288 continue;
289 sce->zeroes[i] = 1;
290 sce->band_type[i] = 0;
291 }
292}
293
294/* Per-channel setup into slot t: short-block threshold shaping, the
295 * allocation law, zero decisions, and the PASS 1 coarse candidate curves.
296 * Returns the coded-band count; 0 = nothing codeable (caller bails). */
299{
300 float (*nd)[NMR_NCAND] = s->nmr->nd[t->si];
301 int (*nb)[NMR_NCAND] = s->nmr->nb[t->si];
302 const int cstep = NMR_COARSE > 0 ? NMR_COARSE : NMR_STEP;
303 int allz = 0, cutoff = 1024, nbnd = 0;
304
305 uint8_t *zprev = s->nmr->zero_prev[s->cur_channel & 15];
306 if (s->nmr->zero_nw[s->cur_channel & 15] != sce->ics.num_windows) {
307 memset(zprev, 1, 128);
308 s->nmr->zero_nw[s->cur_channel & 15] = sce->ics.num_windows;
309 }
310
311 t->sce = sce;
312 t->cur_ch = s->cur_channel;
314 t->nbnd = t->nact = 0;
315
316 /* band cutoff index for this frame's window size; the bandwidth is fixed
317 * at init and shared with the psy model */
318 cutoff = s->bandwidth * 2 * (1024 / sce->ics.num_windows) / avctx->sample_rate;
319
320 /* Short-block shaping: temporal premask + per-window threshold flatten. */
322 const float pm_p1 = 0.1f, pm_p2 = 2.0f, pm_p3 = 4.0f;
323 for (int g = 0; g < sce->ics.num_swb; g++) {
324 float t1 = FLT_MAX, t2 = FLT_MAX; /* original thr of w-1, w-2 */
325 for (int w = 0; w < sce->ics.num_windows; w++) {
326 FFPsyBand *b = &s->psy.ch[s->cur_channel].psy_bands[w*16+g];
327 float th = b->threshold;
328 float c = FFMIN(th, FFMIN(t1*pm_p2, t2*pm_p3));
329 b->threshold = FFMAX(c, th*pm_p1);
330 t2 = t1; t1 = th;
331 }
332 }
333 {
334 for (int w = 0; w < sce->ics.num_windows; w++) {
335 float sum = 0.0f, esum = 0.0f; int n = 0;
336 for (int g = 0; g < sce->ics.num_swb; g++) {
337 FFPsyBand *b = &s->psy.ch[s->cur_channel].psy_bands[w*16+g];
338 if (b->energy > b->threshold && b->threshold > 0.0f) { sum += b->threshold; esum += b->energy; n++; }
339 }
340 if (n > 0) {
341 /* keep each window codeable: cap the mean 12dB under the
342 * window's mean audible energy */
343 float mean = FFMIN(sum / n, (esum / n) * expf(-12.0f * (float)M_LN10 / 10.0f));
344 for (int g = 0; g < sce->ics.num_swb; g++) {
345 FFPsyBand *b = &s->psy.ch[s->cur_channel].psy_bands[w*16+g];
346 if (b->energy > b->threshold && b->threshold > 0.0f)
347 b->threshold = FFMIN(mean, b->threshold * 1e9f);
348 }
349 }
350 }
351 }
352 }
353
354 /* Allocation law; short frames blend to softer energy exponents under
355 * pressure (roll anti-starvation, see memory). */
356 /* The Apple-RE constants (0.443/0.111) describe THEIR bit distribution,
357 * not this trellis's optimum: more psy-threshold influence is worth -38%
358 * Zim / +0.06 ViSQOL at 128k stereo and -15% at 64k, flat plateau beyond. */
359 float a_ae = 0.50f, a_at = 0.18f;
360 if (sce->ics.num_windows == 8 && s->nmr) {
361 /* blend to mask-weighted exponents under rate pressure */
362 a_ae += (0.35f - a_ae) * s->nmr->press;
363 a_at += (0.3f - a_at) * s->nmr->press;
364 }
365 for (int w = 0; w < sce->ics.num_windows; w += sce->ics.group_len[w]) {
366 int start = 0;
367 for (int g = 0; g < sce->ics.num_swb; start += sce->ics.swb_sizes[g++]) {
368 float uplim = 0.0f, ener = 0.0f, spread = 2.0f;
369 int nz = 0;
370 if (sce->band_type[w*16+g] == INTENSITY_BT ||
371 sce->band_type[w*16+g] == INTENSITY_BT2) {
372 /* pre-decided intensity band (right channel): keep its
373 * signalling, it is not trellis-coded */
374 for (int w2 = 0; w2 < sce->ics.group_len[w]; w2++)
375 sce->zeroes[(w+w2)*16+g] = 0;
376 continue;
377 }
378 float zthr_mul = zprev[w*16+g] ? 1.0f : NMR_ZERO_STICKY;
379 /* M/S side bands: zero-reluctance scaled by side/mid ratio (a tiny
380 * side IS the image; zeroing it flickers). */
381 if ((t->cur_ch & 1) && s->nmr && s->nmr->pair &&
382 s->nmr->smode_band[(t->cur_ch >> 1) & 7][w*16+g] == 1) {
383 const FFPsyBand *mb = &s->psy.ch[s->cur_channel - 1].psy_bands[w*16+g];
384 float ratio = 0.0f;
385 float eside = 0.0f;
386 for (int w2 = 0; w2 < sce->ics.group_len[w]; w2++) {
387 const FFPsyBand *bb = &s->psy.ch[s->cur_channel].psy_bands[(w+w2)*16+g];
388 eside += bb->energy;
389 }
390 ratio = eside / FFMAX(mb->energy * sce->ics.group_len[w], 1e-9f);
391 zthr_mul *= 0.25f + 0.75f * av_clipf(ratio / 0.3f, 0.0f, 1.0f);
392 }
393 /* HF precision taper (bitstream RE): at 128k Apple codes
394 * 7.5-12.6k ~6dB coarser and 12.6k+ ~28dB coarser than we do,
395 * funding their LF/mid SNR edge; a quadratic thr rise above the
396 * knee reproduces that rolloff */
397 float hftdb;
398 {
399 float bfreq = start * (avctx->sample_rate * (sce->ics.num_windows == 8 ? 4.0f : 0.5f)) / 1024.0f;
400 float k = FFMAX(0.0f, bfreq - NMR_HF_TAPER_KNEE) / 1000.0f;
401 hftdb = NMR_HF_TAPER * k * k;
402 }
403 t->hftx[w*16+g] = 0;
404 for (int w2 = 0; w2 < sce->ics.group_len[w]; w2++) {
405 FFPsyBand *band = &s->psy.ch[s->cur_channel].psy_bands[(w+w2)*16+g];
406 ener += band->energy;
407 spread = FFMIN(spread, band->spread);
408 if (start >= cutoff || band->energy <= band->threshold * zthr_mul ||
409 band->threshold == 0.0f) {
410 sce->zeroes[(w+w2)*16+g] = 1;
411 continue;
412 }
413 uplim += band->threshold;
414 nz = 1;
415 }
416 if (nz && sce->ics.group_len[w] > 1) {
417 /* a coded group must not mute individual borderline windows:
418 * they share the group's scalefactor (near-free to keep) and
419 * a 3-6ms in-band mute right before an attack reads as a
420 * gated crunch on every beat */
421 for (int w2 = 0; w2 < sce->ics.group_len[w]; w2++) {
422 FFPsyBand *band = &s->psy.ch[s->cur_channel].psy_bands[(w+w2)*16+g];
423 if (sce->zeroes[(w+w2)*16+g] && start < cutoff &&
424 band->threshold > 0.0f &&
425 band->energy > band->threshold * zthr_mul * NMR_GROUP_KEEP) {
426 sce->zeroes[(w+w2)*16+g] = 0;
427 uplim += band->threshold;
428 }
429 }
430 }
431 zprev[w*16+g] = !nz;
432 sce->zeroes[w*16+g] = !nz;
433 t->thr_real[w*16+g] = uplim; /* real mask, before the allocation law (PNS gate) */
434 if (nz && ener > 0.0f && uplim > 0.0f) { /* allocation law */
435 uplim = expf(a_ae * logf(ener) + a_at * logf(uplim));
436 if (hftdb != 0.0f) {
437 uplim *= powf(10.0f, hftdb * 0.1f);
438 /* a band pushed well past its real mask reads as huge
439 * achieved-nd; keep the deliberate deficit out of the
440 * psy-reliability stat or press ramps into noise-class
441 * RC on clean tonal content */
442 if (hftdb > 2.0f)
443 t->hftx[w*16+g] = 1;
444 }
445 }
446 t->thr[w*16+g] = uplim;
447 t->pener[w*16+g] = ener;
448 t->pspread[w*16+g] = spread;
449 allz |= nz;
450 }
451 }
452 if (!allz)
453 return 0;
454
455 /* transition premask (see NMR_TRANS_PM) */
456 if (sce->ics.num_windows == 1) {
457 int ci = t->cur_ch & 15;
458 if (sce->ics.window_sequence[0] == LONG_START_SEQUENCE &&
459 s->nmr->thr_prev_ok[ci]) {
460 for (int g = 0; g < sce->ics.num_swb && g < 64; g++)
461 if (t->thr[g] > 0.0f && s->nmr->thr_prev[ci][g] > 0.0f)
462 t->thr[g] = FFMIN(t->thr[g], s->nmr->thr_prev[ci][g] * NMR_TRANS_PM);
463 }
464 for (int g = 0; g < sce->ics.num_swb && g < 64; g++)
465 s->nmr->thr_prev[ci][g] = t->thr[g];
466 s->nmr->thr_prev_ok[ci] = 1;
467 } else {
468 s->nmr->thr_prev_ok[t->cur_ch & 15] = 0;
469 }
470
471 s->aacdsp.abs_pow34(s->scoefs, sce->coeffs, 1024);
473
474 /* TNS synthesis gain per band: the decoder re-amplifies residual-domain
475 * quantization noise by the whitening gain (shorts only). */
476 for (int i = 0; i < 128; i++)
477 t->tnsg[i] = 1.0f;
478 if (sce->ics.num_windows == 8 && sce->tns.present) {
479 const int mmm2 = FFMIN(sce->ics.tns_max_bands, sce->ics.max_sfb ? sce->ics.max_sfb : sce->ics.num_swb);
480 for (int w = 0; w < 8; w++) {
481 int bottom2 = sce->ics.num_swb;
482 for (int filt = 0; filt < sce->tns.n_filt[w]; filt++) {
483 int top2 = bottom2;
484 bottom2 = FFMAX(0, top2 - sce->tns.length[w][filt]);
485 if (!sce->tns.order[w][filt])
486 continue;
487 for (int g = FFMIN(bottom2, mmm2); g < FFMIN(top2, mmm2); g++) {
488 int s0 = sce->ics.swb_offset[g] + w*128;
489 int s1 = sce->ics.swb_offset[g+1] + w*128;
490 float eres = 0.0f;
491 const FFPsyBand *pb = &s->psy.ch[s->cur_channel].psy_bands[w*16+g];
492 for (int k = s0; k < s1; k++)
493 eres += sce->coeffs[k]*sce->coeffs[k];
494 t->tnsg[w*16+g] = av_clipf(pb->energy / FFMAX(eres, 1e-12f), 1.0f, 64.0f);
495 }
496 }
497 }
498 }
499
500 /* finest codeable scalefactor and max value per band */
501 for (int w = 0; w < sce->ics.num_windows; w += sce->ics.group_len[w]) {
502 int start = w*128;
503 for (int g = 0; g < sce->ics.num_swb; g++) {
504 t->maxvals[w*16+g] = find_max_val(sce->ics.group_len[w], sce->ics.swb_sizes[g], s->scoefs + start);
505 t->minsf[w*16+g] = t->maxvals[w*16+g] > 0 ? coef2minsf(t->maxvals[w*16+g]) : 0;
506 start += sce->ics.swb_sizes[g];
507 }
508 }
509
510 /* PASS 1: coarse candidate curves per coded band
511 * (the lambda search runs on this cheap grid, PASS 2 refines the winner) */
512 {
513 for (int w = 0; w < sce->ics.num_windows; w += sce->ics.group_len[w]) {
514 int start = w*128;
515 for (int g = 0; g < sce->ics.num_swb; g++) {
516 if (!sce->zeroes[w*16+g] && t->maxvals[w*16+g] > 0 && nbnd < 128) {
517 int lo = av_clip(t->minsf[w*16+g], 0, SCALE_MAX_POS);
518 float invthr = 1.0f / FFMAX(t->thr[w*16+g], 1e-9f);
519 int ncand = nmr_band_curve(s, sce, w, g, start, lo, cstep, NMR_NCAND,
520 invthr, t->maxvals[w*16+g], nd[nbnd], nb[nbnd]);
521 if (t->tnsg[w*16+g] > 1.0f)
522 for (int o = 0; o < ncand; o++)
523 nd[nbnd][o] *= t->tnsg[w*16+g];
524 if (ncand == 0) {
525 /* nothing codeable: drop the group band incl. subwindow
526 * flags (group flag is re-derived by ANDing) */
527 for (int w2 = 0; w2 < sce->ics.group_len[w]; w2++)
528 sce->zeroes[(w+w2)*16+g] = 1;
529 } else {
530 t->bidx[nbnd] = w*16+g;
531 t->bw[nbnd] = w;
532 t->bg[nbnd] = g;
533 t->bst[nbnd] = start;
534 t->blo[nbnd] = lo;
535 t->bnc[nbnd] = ncand;
536 nbnd++;
537 }
538 }
539 start += sce->ics.swb_sizes[g];
540 }
541 }
542 }
543 t->nbnd = nbnd;
544 for (int b = 0; b < nbnd; b++) {
545 t->act[b] = b;
546 t->is_pns[b] = 0;
547 }
548 t->nact = nbnd;
549 return nbnd;
551
552/* total bits of a slot's current chosen[] on grid `step`, incl. sf deltas */
553static int nmr_slot_bits(const NMRSlot *t, const int (*nb)[NMR_NCAND], int step)
554{
555 int tot = 0;
556 for (int k = 0; k < t->nact; k++)
557 tot += nb[t->act[k]][t->chosen[t->act[k]]];
558 for (int k = 1; k < t->nact; k++)
559 tot += NMR_SFBITS((t->blo[t->act[k]]+t->chosen[t->act[k]]*step) -
560 (t->blo[t->act[k-1]]+t->chosen[t->act[k-1]]*step));
561 return tot;
563
564/* Run every slot's trellis at one fixed lambda; returns the pooled bits. */
565static int nmr_eval_slots(AACEncContext *s, NMRSlot *const *sl, int nsl, int step, float lam)
566{
567 int total = 0;
568 for (int k = 0; k < nsl; k++) {
569 NMRSlot *t = sl[k];
570 if (!t->nact)
571 continue;
572 nmr_solve(s, s->nmr->nd[t->si], s->nmr->nb[t->si], t->blo, t->bnc, step,
573 t->act, t->nact, 0, t->chosen, lam, lam, 1);
574 total += nmr_slot_bits(t, s->nmr->nb[t->si], step);
575 }
576 return total;
577}
578
579/* Bisect ONE shared lambda across the slots so the POOLED bits meet destbits.
580 * This is the CPE budget pool: bits flow to whichever channel of the pair has
581 * demand at the common operating point, instead of an equal per-channel split. */
582static float nmr_solve_slots(AACEncContext *s, NMRSlot *const *sl, int nsl, int step,
583 int destbits, float lo_l, float hi_l, int iters)
584{
585 float lam = 1.0f;
586 for (int it = 0; it < iters; it++) {
587 lam = sqrtf(lo_l * hi_l);
588 int total = nmr_eval_slots(s, sl, nsl, step, lam);
589 if (it == iters - 1)
590 break;
591 /* over budget -> go coarser */
592 if (total > destbits)
593 lo_l = lam;
594 else
595 hi_l = lam;
596 }
597 return lam;
598}
599
600/* Mean log2 achieved dist/real-mask over the active bands of all slots at
601 * their current chosen[] - average dB-distance from the mask (Brandenburg
602 * mean-NMR). Log compression keeps unavoidably-loud bands from swamping the
603 * statistic. NAN when nothing is coded. */
604static float nmr_nd_stat(AACEncContext *s, NMRSlot *const *sl, int nsl)
605{
606 float ndsum = 0.0f;
607 int n = 0;
608 for (int k = 0; k < nsl; k++) {
609 NMRSlot *t = sl[k];
610 const float (*ndk)[NMR_NCAND] = (const float (*)[NMR_NCAND])s->nmr->nd[t->si];
611 for (int b_ = 0; b_ < t->nact; b_++) {
612 int b = t->act[b_], bi = t->bidx[b];
613 if (t->thr_real[bi] > 0.0f && t->thr[bi] > 0.0f) {
614 /* floor the sub-mask credit: a flat sf-chain over-codes
615 * quiet bands and their unbounded negative log2 would let a
616 * COARSER lambda read as a finer statistic */
617 ndsum += FFMAX(log2f(FFMAX(ndk[b][t->chosen[b]] * t->thr[bi] /
618 t->thr_real[bi], 1e-6f)), -4.0f);
619 n++;
620 }
621 }
622 }
623 return n ? ndsum / n : NAN;
624}
625
626/* Bisect ONE shared lambda across the slots so the achieved noise-to-mask
627 * statistic meets the target: quality-domain VBR. Monotone: coarser lambda ->
628 * more distortion. */
629static float nmr_solve_slots_nd(AACEncContext *s, NMRSlot *const *sl, int nsl,
630 int step, float target, float lo_l, float hi_l,
631 int iters)
632{
633 float lam = 1.0f;
634 for (int it = 0; it < iters; it++) {
635 float st;
636 lam = sqrtf(lo_l * hi_l);
637 nmr_eval_slots(s, sl, nsl, step, lam);
638 st = nmr_nd_stat(s, sl, nsl);
639 if (isnan(st) || it == iters - 1)
640 break;
641 if (st > target)
642 hi_l = lam;
643 else
644 lo_l = lam;
645 }
646 return lam;
647}
649/* Write a solved slot back into its channel: band types, scalefactors, and the
650 * SCALE_MAX_DIFF legality fixups. Verbatim from the pre-pool single-channel tail. */
652{
653 SingleChannelElement *sce = t->sce;
654 const int (*nb)[NMR_NCAND] = (const int (*)[NMR_NCAND])s->nmr->nb[t->si];
655
656 for (int b = 0; b < t->nbnd; b++) {
657 int bi = t->bidx[b];
658 if (t->is_pns[b]) {
659 sce->band_type[bi] = NOISE_BT;
660 sce->zeroes[bi] = 0;
661 sce->pns_ener[bi] = t->pener[bi] * FFMIN(1.0f, t->pspread[bi]*t->pspread[bi]);
662 } else {
663 sce->sf_idx[bi] = av_clip(t->blo[b] + t->chosen[b]*NMR_STEP, 0, SCALE_MAX_POS);
664 }
665 }
666
667
668 /* record the bits this solve accounted for; the encoder compares them
669 * against the channel's real output to keep the budget honest. Only the
670 * surviving active bands are coded: bands shed for legality are not. */
671 s->nmr->counted[t->cur_ch] = nmr_slot_bits(t, nb, NMR_STEP);
672
673 /* SCALE_MAX_DIFF condition:
674 * re-clamp, codebook fixup, drop uncodeable, set global gain
675 * NOISE_BT bands keep their own scalefactor chain via set_special_band_scalefactors) */
676 {
677 uint8_t nextband[128];
678 int prev = -1;
679 ff_init_nextband_map(sce, nextband);
680 for (int w = 0; w < sce->ics.num_windows; w += sce->ics.group_len[w]) {
681 for (int g = 0; g < sce->ics.num_swb; g++) {
682 if (sce->band_type[w*16+g] == NOISE_BT ||
683 sce->band_type[w*16+g] == INTENSITY_BT ||
684 sce->band_type[w*16+g] == INTENSITY_BT2)
685 continue;
686 if (sce->zeroes[w*16+g]) {
687 sce->band_type[w*16+g] = 0;
688 continue;
689 }
690
691 if (prev != -1)
692 sce->sf_idx[w*16+g] = av_clip(sce->sf_idx[w*16+g], prev - SCALE_MAX_DIFF, prev + SCALE_MAX_DIFF);
693 sce->band_type[w*16+g] = find_min_book(t->maxvals[w*16+g], sce->sf_idx[w*16+g]);
694 if (sce->band_type[w*16+g] <= 0) {
695 if (!ff_sfdelta_can_remove_band(sce, nextband, prev, w*16+g)) {
696 sce->band_type[w*16+g] = 1;
697 } else {
698 /* drop subwindow flags too, see the PASS 1 drop above */
699 for (int w2 = 0; w2 < sce->ics.group_len[w]; w2++)
700 sce->zeroes[(w+w2)*16+g] = 1;
701 sce->band_type[w*16+g] = 0;
702 continue;
703 }
704 }
705 if (prev == -1)
706 sce->sf_idx[0] = sce->sf_idx[w*16+g]; /* global gain */
707 prev = sce->sf_idx[w*16+g];
708 }
709 }
710
711 /* every band must carry a chain-legal scalefactor (re-clamp, codebook
712 * fixup, global gain) */
713 if (prev != -1) {
714 int last = sce->sf_idx[0];
715 for (int w = 0; w < sce->ics.num_windows; w += sce->ics.group_len[w]) {
716 for (int g = 0; g < sce->ics.num_swb; g++) {
717 if (!sce->zeroes[w*16+g] && sce->band_type[w*16+g] != NOISE_BT &&
718 sce->band_type[w*16+g] < RESERVED_BT)
719 last = sce->sf_idx[w*16+g];
720 else if (sce->band_type[w*16+g] < RESERVED_BT && (w*16+g) > 0)
721 sce->sf_idx[w*16+g] = last;
722 }
723 }
724 }
725 }
726}
728/* Solve one element group (a solo channel, or a CPE pair pooled under one
729 * shared lambda and one pooled budget), then PNS and commit. */
730static void nmr_solve_group(AVCodecContext *avctx, AACEncContext *s,
731 const float lambda, NMRSlot *const *sl, int nsl,
732 int chans, int rc_eligible, int rc_global,
733 int rc_rate_frame, int rc_bmax)
734{
735 const int cstep = NMR_COARSE > 0 ? NMR_COARSE : NMR_STEP;
736 int bch = ((avctx->flags & AV_CODEC_FLAG_QSCALE) ? 2.0f : avctx->ch_layout.nb_channels);
737 int destbits = avctx->bit_rate * 1024.0 / avctx->sample_rate / bch * (lambda / 120.f) * chans;
738 int is8_any = 0;
739 float lam;
740 float rc_off = 1.0f, lam_dem = 0.0f;
741 /* the outer loop's starting lambda: legality caps scale with lambda
742 * relative to it, so they only shrink on a hard-overflow retry */
743 const float lam_ref = (avctx->flags & AV_CODEC_FLAG_QSCALE) && avctx->global_quality > 0 ?
744 avctx->global_quality : 120.0f;
745
746 for (int k = 0; k < nsl; k++)
747 is8_any |= sl[k]->is8;
748
749 /* decoupled solo solves keep per-channel slew state: sharing one
750 * lam_slew clamps each channel against the OTHER's operating lambda */
751 float *vslew_st = &s->nmr->lam_slew;
752 if (nsl == 1 && s->channels > 1) {
753 vslew_st = &s->nmr->lam_slew_ch[sl[0]->cur_ch & 15];
754 if (*vslew_st <= 0.0f)
755 *vslew_st = s->nmr->lam_slew;
756 }
757
758 if (s->psy.bitres.alloc >= 0)
759 destbits = s->psy.bitres.alloc *
760 (lambda / (avctx->global_quality ? avctx->global_quality : 120)) * chans;
761 if (rc_global && s->psy.bitres.alloc >= 0) {
762 /* CBR target: nominal + repayment, bounded +-30%/frame */
763 double rr = avctx->bit_rate * 1024.0 / avctx->sample_rate;
764 destbits = (rr + av_clipd(s->nmr->rc_fill / 2.0, -0.3 * rr, 0.3 * rr)) * chans / s->channels;
765 } else if (rc_eligible && s->psy.bitres.alloc >= 0) {
766 /* pre-bootstrap CBR frames: target nominal (psy bitres is cold) */
767 destbits = (avctx->bit_rate * 1024.0 / avctx->sample_rate) * chans / s->channels;
768 }
769 destbits = FFMIN(destbits, 5800 * chans);
770 /* honest budget: subtract the measured non-trellis overhead (section data, ICS,
771 * sf/PNS signalling), which is rate-dependent hence adaptive. */
772 if (s->nmr->side_inited)
773 destbits = av_clip(destbits - (int)(s->nmr->side_ema * chans / s->channels), 64, 5800 * chans);
774
775 /* Held transient burst, bank-aware: spend banked bits, never borrow deep
776 * (payback troughs starve the next transient). */
777 if (s->nmr->run_burst > 1.0f) {
778 int extra = destbits * (s->nmr->run_burst - 1.0f);
779 int avail = FFMAX(0, (int)((s->nmr->rc_fill + rc_bmax / 2) * (int64_t)chans / s->channels));
780 destbits = av_clip(destbits + FFMIN(extra, avail), 64, 6800 * chans);
781 }
782
783 int vbr = (avctx->flags & AV_CODEC_FLAG_QSCALE) && s->nmr;
784 int abr = !vbr && s->options.rc == 1 && avctx->bit_rate > 0 && s->nmr;
785 int vbr_subst = 1;
786 float vbr_t = 0.0f;
787 if (abr) {
788 /* ABR: same quality-target solve, set-point owned by the rate servo */
789 vbr_t = s->nmr->abr_t;
790 vbr = 1;
791 } else if (vbr) {
792 /* nd-target VBR: constant achieved noise-to-mask, bits float. */
793 /* target in log2(dist/real-mask), anchored so -q:a 1 lands near
794 * 128 kbps stereo on the tuning corpus; higher q is finer, each q
795 * doubling ~ -1.2 */
796 vbr_t = FFMAX(NMR_VBR_ANCHOR - 1.2f * log2f(avctx->global_quality > 0 ?
797 avctx->global_quality / (float)FF_QP2LAMBDA : 1.0f),
799 }
800 if (vbr && is8_any) {
801 /* short frames: the grouped-band statistic is inflated (same reason
802 * nd_ema excludes shorts), so they bisect against an offset target
803 * rather than the long-frame one - and never write solver state
804 * (transient-dense content would otherwise starve the long-frame
805 * warm start of updates). Warm off the surrounding operating point. */
806 float lam0 = *vslew_st > 0.0f ? FFMIN(*vslew_st, 1e4f) : 0.0f;
807 float *infl = &s->nmr->vbr_infl[sl[0]->cur_ch & 15];
808 if (lam0 > 0.0f) {
809 /* measured stat offset, decided at shorts-run entry: evaluate
810 * the grouped stat at the long-anchor lambda - continuity says
811 * true quality is on target there, so any excess IS this
812 * content's inflation. Broadband beats (velvet) measure +2..+4,
813 * tonal musical decays measure none; the old fixed +3 picked
814 * one class and audibly damaged the other. Re-measured every
815 * short frame: run-entry-only attribution tars a whole tonal
816 * decay run with its mini-attack entry frame. */
817 float st0, ehf = 0.0f, etot = 0.0f, hfw;
818 nmr_eval_slots(s, sl, nsl, cstep, lam0);
819 st0 = nmr_nd_stat(s, sl, nsl);
820 /* the stat reads inflated on BOTH classes; audibility does not.
821 * Coarse shorts hide under HF-dominant transients (hats, claps)
822 * and stick out on LF/mid tonal material - weight the offset by
823 * the frame's HF energy share. */
824 for (int k = 0; k < nsl; k++)
825 for (int b_ = 0; b_ < sl[k]->nact; b_++) {
826 int b = sl[k]->act[b_], bi = sl[k]->bidx[b];
827 float en = sl[k]->pener[bi];
828 int bin = sl[k]->bst[b] - sl[k]->bw[b]*128;
829 etot += en;
830 if (bin * (avctx->sample_rate * 4.0f) / 1024.0f > 6000.0f)
831 ehf += en;
832 }
833 hfw = etot > 0.0f ? av_clipf(ehf / (etot * NMR_INFL_HF), 0.0f, 1.0f) : 1.0f;
834 *infl = isnan(st0) ? 0.0f : av_clipf(st0 - vbr_t, 0.0f, 4.0f) * hfw;
835 }
836 if (lam0 > 0.0f)
837 lam = nmr_solve_slots_nd(s, sl, nsl, cstep, vbr_t + *infl,
838 lam0/32.0f, FFMIN(lam0*32.0f, 1e4f), NMR_CWARM + 2);
839 else
840 lam = nmr_solve_slots_nd(s, sl, nsl, cstep, vbr_t, 1e-9f, 1e4f, NMR_ITERS);
841 if (lam0 > 0.0f) {
842 /* the offset-bisect assumes isolated transients between long
843 * anchors; on continuous-shorts content the per-frame grouped
844 * stat is bisect noise and lambda careens across decades -
845 * audible as whooshing. Hold shorts to the same near-constant
846 * run slew CBR uses (dives toward finer stay free: transient
847 * bits float by design). */
848 float kup = s->nmr->prev_was_short ? NMR_SLEW_RUN : NMR_SLEW;
849 if (lam > lam0 * kup || lam < lam0 / 4.0f) {
850 lam = av_clipf(lam, lam0 / 4.0f, lam0 * kup);
851 nmr_eval_slots(s, sl, nsl, cstep, lam);
852 }
853 }
854 vbr_subst = 0;
855 } else if (vbr) {
856 /* warm-started off the previous frame's lambda */
857 float lam0 = s->nmr->lam[sl[0]->cur_ch];
858 lam = 1.0f;
859 if (lam0 > 0.0f) {
860 lam0 = FFMIN(lam0, 1e4f);
861 lam = nmr_solve_slots_nd(s, sl, nsl, cstep, vbr_t,
862 lam0/32.0f, FFMIN(lam0*32.0f, 1e4f), NMR_CWARM + 2);
863 if (lam < lam0/16.0f || lam > lam0*16.0f)
864 lam0 = 0.0f;
865 }
866 if (lam0 <= 0.0f)
867 lam = nmr_solve_slots_nd(s, sl, nsl, cstep, vbr_t, 1e-9f, 1e4f, NMR_ITERS);
868 } else if (rc_global) {
869 /* corridor bisect around the servoed centre; pressure = stateless
870 * rc_off multiplier (folding it into lam_rc winds up) */
871 float R = avctx->bit_rate * 1024.0 / avctx->sample_rate;
872 float cen;
873 int tot, hardcap, rc_cap;
874 float lo;
875 rc_off = exp2f(-NMR_RC_K_CBR * s->nmr->rc_fill / R);
876 cen = s->nmr->lam_rc * rc_off;
877 lo = cen / NMR_RC_CORR;
878 /* transient burst: widen the lower bound so the boosted destbits can
879 * actually pour into the onset frame */
880 if (is8_any && s->nmr->run_burst > 1.0f)
881 lo /= s->nmr->run_burst;
882 lam = nmr_solve_slots(s, sl, nsl, cstep, destbits,
883 lo, cen * NMR_RC_CORR, NMR_RC_CITERS);
884
885 tot = 0;
886 for (int k = 0; k < nsl; k++)
887 tot += nmr_slot_bits(sl[k], s->nmr->nb[sl[k]->si], cstep);
888 hardcap = av_clip((int)(5800.f * FFMIN(1.f, lambda / 120.f)), 256, 5800) * chans;
889 /* legality cap only; no spend-floor (rc_off spends the bank) */
890 rc_cap = FFMIN(hardcap, (s->nmr->rc_fill + rc_rate_frame + rc_bmax) * chans / s->channels);
891 if (tot > rc_cap) {
892 /* reservoir-empty: coarsen, but never past a bounded excursion
893 * of the corridor. The coarse grid's bit curve is steppy: an
894 * unbounded fit here can jump lambda by orders of magnitude and
895 * potato-frame the burst recovery frame (audible LF scratch).
896 * The fine pass works from the bounded lambda, and the reservoir
897 * cap is enforced once more on the fine grid after it, where the
898 * fit only moves lambda as far as the frame really needs. */
899 lam = nmr_solve_slots(s, sl, nsl, cstep, rc_cap, lam, lam * NMR_RC_CAPK, NMR_CITERS);
900 tot = 0;
901 for (int k = 0; k < nsl; k++)
902 tot += nmr_slot_bits(sl[k], s->nmr->nb[sl[k]->si], cstep);
903 if (tot > rc_cap && s->nmr->rc_fill <= -(rc_bmax * 9 / 10)) {
904 /* the debt ledger clips at -rc_bmax: with no capacity
905 * left, further overage would be silently forgiven and
906 * sustained transient content rides hot forever (a
907 * castanets roll hit 215 kbps for its first second on a
908 * 128k ask). A one-frame hard fit trades that for a
909 * potato frame (audible click); instead the bound
910 * ESCALATES with consecutive saturated frames - chronic
911 * rolls converge within a few frames, isolated bursts
912 * never see more than one escalation step. */
913 float ek = NMR_RC_CAPK * (1 + FFMIN(s->nmr->rc_satrun, 8));
914 lam = nmr_solve_slots(s, sl, nsl, cstep, rc_cap, lam, lam * ek, NMR_CITERS);
915 s->nmr->rc_sat_frame = 1;
916 tot = 0;
917 for (int k = 0; k < nsl; k++)
918 tot += nmr_slot_bits(sl[k], s->nmr->nb[sl[k]->si], cstep);
919 }
920 if (tot > hardcap) /* decoder-buffer legality is absolute */
921 lam = nmr_solve_slots(s, sl, nsl, cstep, hardcap, lam, 1e4f, NMR_CITERS);
922 }
923 } else {
924 /* per-frame bisection, warm-started off the previous frame's lambda;
925 * a result at the bracket edge means redo the full search */
926 float lam0 = s->nmr->lam[sl[0]->cur_ch];
927 lam = 1.0f;
928 if (NMR_COARSE > 0 && lam0 > 0.0f) {
929 lam = nmr_solve_slots(s, sl, nsl, cstep, destbits, lam0/32.0f, lam0*32.0f, NMR_CWARM);
930 if (lam < lam0/16.0f || lam > lam0*16.0f)
931 lam0 = 0.0f;
932 }
933 if (lam0 <= 0.0f)
934 lam = nmr_solve_slots(s, sl, nsl, cstep, destbits,
935 1e-9f, 1e4f, NMR_COARSE > 0 ? NMR_CITERS : NMR_ITERS);
936 }
937
938 /* PASS 2:
939 * refine each band at full granularity (NMR_STEP) in a +/-cstep window
940 * around the coarse pick, then re-solve. Recovers single-pass quality while the
941 * lambda search stayed cheap on the coarse grid. */
942 if (NMR_COARSE > 0) {
943 /* nmr_speed, 0 = slowest/best, higher = faster; see the option docs. */
944 int win = NMR_COARSE - av_clip(s->options.nmr_speed, 0, 4);
945 for (int k = 0; k < nsl; k++) {
946 NMRSlot *t = sl[k];
947 float (*ndk)[NMR_NCAND] = s->nmr->nd[t->si];
948 int (*nbk)[NMR_NCAND] = s->nmr->nb[t->si];
949 if (!t->nact)
950 continue;
951 /* the pow34 spectrum and the quantize cache are per-channel state */
952 s->aacdsp.abs_pow34(s->scoefs, t->sce->coeffs, 1024);
954 for (int b = 0; b < t->nbnd; b++) {
955 int center = t->blo[b] + t->chosen[b]*cstep;
956 int flo = av_clip(center - win, av_clip(t->minsf[t->bidx[b]], 0, SCALE_MAX_POS), SCALE_MAX_POS);
957 int maxn = FFMIN(NMR_NCAND, 2*win/NMR_STEP + 1);
958 float invthr = 1.0f / FFMAX(t->thr[t->bidx[b]], 1e-9f);
959 int ncand = nmr_band_curve(s, t->sce, t->bw[b], t->bg[b], t->bst[b], flo, NMR_STEP, maxn,
960 invthr, t->maxvals[t->bidx[b]], ndk[b], nbk[b]);
961 if (t->tnsg[t->bidx[b]] > 1.0f)
962 for (int o = 0; o < ncand; o++)
963 ndk[b][o] *= t->tnsg[t->bidx[b]];
964 t->blo[b] = flo;
965 t->bnc[b] = FFMAX(1, ncand);
966 }
967 }
968 /* fine pass: narrow corridor around the coarse solve */
969 if (vbr && !vbr_subst) {
970 float infl2 = is8_any ? s->nmr->vbr_infl[sl[0]->cur_ch & 15] : 0.0f;
971 lam = nmr_solve_slots_nd(s, sl, nsl, NMR_STEP, vbr_t + infl2,
972 lam/32.0f, FFMIN(lam*32.0f, 1e4f), NMR_IFINE);
973 } else if (vbr)
974 /* wide re-bisect: the fine grid sits systematically finer than
975 * the coarse grid at equal lambda, well past a 1-octave bracket */
976 lam = nmr_solve_slots_nd(s, sl, nsl, NMR_STEP, vbr_t,
977 lam/32.0f, FFMIN(lam*32.0f, 1e4f), NMR_IFINE);
978 else if (rc_global)
979 lam = nmr_solve_slots(s, sl, nsl, NMR_STEP, destbits, lam/2.0f, lam*2.0f, NMR_RC_FITERS);
980 else
981 lam = nmr_solve_slots(s, sl, nsl, NMR_STEP, destbits, lam/16.0f, lam*16.0f, NMR_IFINE);
982 }
983
984 lam_dem = lam; /* demand-solved lambda, pre bucket clamp: what content wants */
985
986 if (vbr) {
987 /* quality slew: lambda moves smoothly, bits follow content (the same
988 * anti-flutter rule the CBR path uses; unconstrained per-frame lambda
989 * reads as framerate-rate quality modulation). Low-content frames
990 * (silence, stop-gaps) rail lambda high legitimately - they must
991 * neither be clamped nor write back solver state, or every gap
992 * poisons the slew/warm-start and the solve re-traps at the rail. */
993 int subst = 0;
994 float st_fin = nmr_nd_stat(s, sl, nsl);
995 for (int k = 0; k < nsl; k++)
996 subst += sl[k]->nact;
997 /* a frame counts as an operating point only if it bisected the
998 * statistic (not a short frame holding lambda) and the solve
999 * actually REACHED the target: cheap frames (room tone, gaps) sit
1000 * far below it at any lambda and rail high while costing nothing */
1001 /* two-sided: a frame railed COARSE (st far above target, e.g. the
1002 * first content frames of a quiet channel bisecting from nothing)
1003 * is no more an operating point than a railed-fine cheap frame -
1004 * writing its rail lambda builds a slew prison the channel then
1005 * climbs out of one rung per frame, muffling the stream head */
1006 subst = vbr_subst && subst >= 8 && !isnan(st_fin) &&
1007 st_fin > vbr_t - 0.5f && st_fin < vbr_t + 0.5f;
1008 if (subst) {
1009 if (*vslew_st > 0.0f) {
1010 /* only bisected LONG frames land here (shorts run free); on
1011 * steady-tonal content per-frame lambda jitter reads as
1012 * framerate-rate noise modulation (see NMR_RC_CORR) */
1013 if (lam > *vslew_st * NMR_SLEW || lam < *vslew_st / NMR_SLEW) {
1014 lam = av_clipf(lam, *vslew_st / NMR_SLEW, *vslew_st * NMR_SLEW);
1015 nmr_eval_slots(s, sl, nsl, NMR_STEP, lam);
1016 }
1017 }
1018 *vslew_st = s->nmr->lam_slew = lam;
1019 } else if (is8_any && *vslew_st > 0.0f) {
1020 /* short frames: PASS 2 re-bisects the inflated grouped stat on
1021 * a fresh +-32x bracket, undoing any earlier bound - the clamp
1022 * must sit here, after the last solve (love.flac whoosh: lambda
1023 * railed at exactly lam*32 through an all-shorts passage) */
1024 float kup = s->nmr->prev_was_short ? NMR_SLEW_RUN : NMR_SLEW;
1025 float lam0 = FFMIN(*vslew_st, 1e4f);
1026 if (lam > lam0 * kup || lam < lam0 / 4.0f) {
1027 lam = av_clipf(lam, lam0 / 4.0f, lam0 * kup);
1028 nmr_eval_slots(s, sl, nsl, NMR_STEP, lam);
1029 }
1030 }
1031 vbr_subst = subst;
1032 }
1033 if (vbr) {
1034 /* legality only: a frame may never exceed the bit reservoir bound.
1035 * Side bits ride on top of the trellis count; a negative side
1036 * estimate must never enlarge the cap. The cap is the format limit
1037 * at the reference lambda (the -q:a quality itself for VBR) and
1038 * only shrinks on an outer hard-overflow retry, so the retry
1039 * converges. */
1040 int hardcap = av_clip((int)(5800.f * FFMIN(1.f, lambda / lam_ref)), 256, 5800) * chans;
1041 int tot = 0;
1042 if (s->nmr->side_inited)
1043 hardcap = FFMAX(hardcap - (int)(FFMAX(s->nmr->side_ema, 0.0f) * chans / s->channels), 256);
1044 for (int k = 0; k < nsl; k++)
1045 tot += nmr_slot_bits(sl[k], s->nmr->nb[sl[k]->si], NMR_STEP);
1046 if (abr && s->nmr->abr_ema < rc_rate_frame) {
1047 /* bits-fill: under the long-run target, top an easy frame up
1048 * toward the ask by moving lambda FINER only - the quality
1049 * target is a floor, never traded away. Spends the rate the
1050 * user asked for on sub-mask margin the nd statistic cannot
1051 * see (it clips at -4/band); without this, stat-transparent
1052 * content caps the rate below any ask. Active from frame one
1053 * (the immature EMA reads as a deficit, so the stream head is
1054 * held near the ask) and fades continuously as the EMA
1055 * converges - a boot-gated fill flips regimes at ~2s, an
1056 * audible quality step.
1057 * The fill target follows the psy PE demand shape: per-frame
1058 * demand carries perceptual information the nd model does not
1059 * (the constant-lambda lesson) - and routing it through the
1060 * fill keeps the quality floor intact. */
1061 float dshape = 1.0f;
1062 int fill;
1063 if (s->psy.bitres.alloc > 0) {
1064 float *aema = &s->nmr->abr_alloc_ema;
1065 if (*aema <= 0.0f) *aema = s->psy.bitres.alloc;
1066 else *aema += 0.01f * (s->psy.bitres.alloc - *aema);
1067 dshape = av_clipf(s->psy.bitres.alloc / *aema, 0.6f, 1.7f);
1068 }
1069 fill = (int)((rc_rate_frame + (rc_rate_frame - (int)s->nmr->abr_ema)) * dshape) *
1070 chans / s->channels;
1071 fill = FFMIN3(fill, 3 * rc_rate_frame * chans / s->channels / 2, hardcap);
1072 if (tot < fill) {
1073 /* the fill dive is the easy-frame operating point: slew it
1074 * like any other lambda move and let it carry the
1075 * continuity state, or per-frame depth variation reads as
1076 * frame-rate HF wobble */
1077 float flo = lam / 64.0f;
1078 if (*vslew_st > 0.0f)
1079 flo = FFMIN(FFMAX(flo, *vslew_st / NMR_SLEW), lam);
1080 lam = nmr_solve_slots(s, sl, nsl, NMR_STEP, fill, flo, lam, NMR_RC_ITERS);
1081 if (!is8_any)
1082 *vslew_st = s->nmr->lam_slew = lam;
1083 tot = 0;
1084 for (int k = 0; k < nsl; k++)
1085 tot += nmr_slot_bits(sl[k], s->nmr->nb[sl[k]->si], NMR_STEP);
1086 }
1087 }
1088 if (tot > hardcap) {
1089 lam = nmr_solve_slots(s, sl, nsl, NMR_STEP, hardcap, lam, 1e4f, NMR_RC_ITERS);
1090 tot = 0;
1091 for (int k = 0; k < nsl; k++)
1092 tot += nmr_slot_bits(sl[k], s->nmr->nb[sl[k]->si], NMR_STEP);
1093 /* dense short frames at high rates can exceed the decoder
1094 * buffer even at the coarsest candidate grid; an illegal frame
1095 * traps the outer re-encode loop forever. Shed the highest
1096 * bands until the frame fits. act[] runs window group by window
1097 * group, so pick the highest band across all groups; the first
1098 * band anchors the scalefactor chain and stays. */
1099 while (tot > hardcap) {
1100 int dropped = 0;
1101 for (int k = 0; k < nsl; k++) {
1102 NMRSlot *t = sl[k];
1103 SingleChannelElement *sce = t->sce;
1104 int hi = 1, b, w0, g;
1105 if (t->nact <= 1)
1106 continue;
1107 for (int b_ = 2; b_ < t->nact; b_++)
1108 if (t->bg[t->act[b_]] >= t->bg[t->act[hi]])
1109 hi = b_;
1110 b = t->act[hi];
1111 w0 = t->bw[b]; g = t->bg[b];
1112 for (int w2 = 0; w2 < sce->ics.group_len[w0]; w2++)
1113 sce->zeroes[(w0+w2)*16+g] = 1;
1114 memmove(&t->act[hi], &t->act[hi + 1], (t->nact - hi - 1) * sizeof(t->act[0]));
1115 t->nact--;
1116 dropped = 1;
1117 }
1118 if (!dropped)
1119 break;
1120 tot = nmr_eval_slots(s, sl, nsl, NMR_STEP, lam);
1121 }
1122 }
1123 }
1124
1125 if (rc_global) {
1126 /* reservoir cap on the fine grid (see the coarse-pass bound above),
1127 * then the quality slew limiter */
1128 int hardcap = av_clip((int)(5800.f * FFMIN(1.f, lambda / 120.f)), 256, 5800) * chans;
1129 int tot = 0, rc_cap;
1130 for (int k = 0; k < nsl; k++)
1131 tot += nmr_slot_bits(sl[k], s->nmr->nb[sl[k]->si], NMR_STEP);
1132 rc_cap = FFMIN(hardcap, (s->nmr->rc_fill + rc_rate_frame + rc_bmax) * chans / s->channels);
1133 if (tot > rc_cap) {
1134 lam = nmr_solve_slots(s, sl, nsl, NMR_STEP, rc_cap, lam, 1e4f, NMR_RC_ITERS);
1135 }
1136 { /* bucket-full spend floor: bits saved beyond the reservoir's
1137 * remaining headroom are simply lost, so with a full bucket the
1138 * plan must not sit below nominal minus what can still be
1139 * banked. Without this a quiet intro poisons the corridor high
1140 * and the loud entrance rate-limits through it - a muffled
1141 * first second at the exact moment the listener tunes in. */
1142 int headroom = rc_bmax - av_clip(s->nmr->rc_fill, -rc_bmax, rc_bmax);
1143 int fbits = (rc_rate_frame - headroom) * chans / s->channels;
1144 if (tot < fbits) {
1145 lam = nmr_solve_slots(s, sl, nsl, NMR_STEP, fbits, lam / 64.0f, lam, NMR_RC_ITERS);
1146 tot = 0;
1147 for (int k = 0; k < nsl; k++)
1148 tot += nmr_slot_bits(sl[k], s->nmr->nb[sl[k]->si], NMR_STEP);
1149 }
1150 }
1151 if (s->nmr->lam_slew > 0.0f) {
1152 float kup, kdn;
1153 /* hold lambda near-constant within short runs; bits follow content */
1154 kup = (is8_any && s->nmr->prev_was_short) ? NMR_SLEW_RUN : NMR_SLEW;
1155 /* a deliberate onset burst may dive as far as its widened corridor
1156 * allows; the RECOVERY back up is what must stay gradual */
1157 kdn = (is8_any && s->nmr->run_burst > 1.0f) ? NMR_SLEW * s->nmr->run_burst :
1158 (is8_any && s->nmr->prev_was_short) ? NMR_SLEW_RUN : NMR_SLEW;
1159 if (lam > s->nmr->lam_slew * kup || lam < s->nmr->lam_slew / kdn) {
1160 lam = av_clipf(lam, s->nmr->lam_slew / kdn, s->nmr->lam_slew * kup);
1161 tot = nmr_eval_slots(s, sl, nsl, NMR_STEP, lam);
1162 /* never at the price of an illegal reservoir excursion */
1163 if (tot > rc_cap) {
1164 lam = nmr_solve_slots(s, sl, nsl, NMR_STEP, rc_cap, lam, 1e4f, NMR_RC_ITERS);
1165 }
1166 }
1167 }
1168 s->nmr->lam_slew = lam;
1169 } else if (rc_eligible && !vbr) {
1170 /* corridor not yet bootstrapped: the same bucket-full floor, so a
1171 * fade-in bootstraps lam_rc at a spending operating point instead
1172 * of memorizing the intro's starvation lambda */
1173 int headroom = rc_bmax - av_clip(s->nmr->rc_fill, -rc_bmax, rc_bmax);
1174 int fbits = (rc_rate_frame - headroom) * chans / s->channels;
1175 int tot = 0;
1176 for (int k = 0; k < nsl; k++)
1177 tot += nmr_slot_bits(sl[k], s->nmr->nb[sl[k]->si], NMR_STEP);
1178 if (tot < fbits)
1179 lam = nmr_solve_slots(s, sl, nsl, NMR_STEP, fbits, lam / 64.0f, lam, NMR_RC_ITERS);
1180 }
1181
1182 if (!vbr || vbr_subst)
1183 for (int k = 0; k < nsl; k++)
1184 s->nmr->lam[sl[k]->cur_ch] = lam; /* warm start for the next frame */
1185 { /* nd: mean achieved dist/real-mask (dimensionless starvation +
1186 * noise-class signal) */
1187 float ndsum = 0.0f; int ndn = 0;
1188 for (int k = 0; k < nsl; k++) {
1189 NMRSlot *t = sl[k];
1190 float (*ndk)[NMR_NCAND] = s->nmr->nd[t->si];
1191 for (int b_ = 0; b_ < t->nact; b_++) {
1192 int b = t->act[b_], bi = t->bidx[b];
1193 if (t->thr_real[bi] > 0.0f && t->thr[bi] > 0.0f && !t->hftx[bi]) {
1194 ndsum += ndk[b][t->chosen[b]] * t->thr[bi] / t->thr_real[bi];
1195 ndn++;
1196 }
1197 }
1198 }
1199 /* long frames only (short groups inflate the ratio) */
1200 if (ndn >= 8 && !is8_any) {
1201 float nd = ndsum / ndn;
1202 s->nmr->nd_ema = s->nmr->nd_ema > 0.0f ?
1203 0.95f * s->nmr->nd_ema + 0.05f * nd : nd;
1204 }
1205 }
1206 { /* track short vs long operating lambda (dense-beat boost scaling) */
1207 float *ema = is8_any ? &s->nmr->lam_short_ema : &s->nmr->lam_long_ema;
1208 *ema = *ema > 0.0f ? 0.9f * *ema + 0.1f * lam : lam;
1209 /* sustained-strain floor: snaps down at any comfortable moment,
1210 * recovers only slowly, so bursty content cannot bank pressure
1211 * credit between its lambda valleys. */
1212 s->nmr->lam_floor = s->nmr->lam_floor > 0.0f ?
1213 fminf(s->nmr->lam_floor * 1.02f, lam) : lam;
1214 }
1215 { /* shared rate-pressure ramp: lambda vs nd-scaled anchors */
1216 float scale, ramp;
1217 scale = 1.0f + av_clipf(s->nmr->nd_ema / 50.0f, 0.0f, 8.0f);
1218 ramp = s->nmr->lam_long_ema > 0.0f ?
1219 av_clipf((s->nmr->lam_long_ema - 120.0f * scale) /
1220 (350.0f * scale - 120.0f * scale), 0.0f, 1.0f) : 0.0f;
1221 /* transparency veto: lambda*nd below ~74 = comfortable */
1222 if (s->nmr->nd_ema > 0.0f)
1223 ramp *= av_clipf((s->nmr->lam_long_ema * s->nmr->nd_ema - 60.0f) /
1224 (120.0f - 60.0f), 0.0f, 1.0f);
1225 s->nmr->press = ramp;
1226 }
1227 if (rc_global) {
1228 /* track the centre toward the CONTENT lambda (demand-solved, pressure
1229 * divided out); clamped lambda is rate noise, not content */
1230 float c = s->nmr->lam_rc * powf(lam_dem / rc_off / s->nmr->lam_rc, NMR_RC_TRACK);
1231 s->nmr->lam_rc = av_clipf(c, 1e-6f, 1e4f);
1232 } else if (rc_eligible) {
1233 /* bootstrap the servo off the first substantive frame (silent lead-ins
1234 * have degenerate budgets) */
1235 int nbnd_max = 0;
1236 for (int k = 0; k < nsl; k++)
1237 nbnd_max = FFMAX(nbnd_max, sl[k]->nbnd);
1238 if (nbnd_max >= 8) {
1239 s->nmr->lam_rc = av_clipf(lam, 1e-4f, 1e4f);
1240 s->nmr->lam_slew = s->nmr->lam_rc;
1241 }
1242 }
1243
1244 { /* PNS, per channel at the group's operating lambda */
1245 const float pns_lam = NMR_PNS_LAM;
1246 int pns_total = 0;
1247 for (int k = 0; k < nsl; k++) {
1248 NMRSlot *t = sl[k];
1249 const float (*ndk)[NMR_NCAND] = (const float (*)[NMR_NCAND])s->nmr->nd[t->si];
1250 const int (*nbk)[NMR_NCAND] = (const int (*)[NMR_NCAND])s->nmr->nb[t->si];
1251 int pns_count = 0;
1252 /* band 0 (lowest freq) is kept as the global-gain / sf-chain anchor */
1253 for (int b = 1; b < t->nbnd; b++) {
1254 int bi = t->bidx[b];
1255 float spread = t->pspread[bi];
1256 float nmr_pns, cost_keep, cost_pns, frac;
1257 /* zeroes[] is only set on a coded band when it was shed for
1258 * legality; such a band must not come back as noise */
1259 if (!t->sce->can_pns[bi] || t->sce->zeroes[bi])
1260 continue;
1261
1262 int was = s->nmr->pns_prev[t->cur_ch & 15][bi];
1263 float bias = was ? NMR_PNS_STAY : NMR_PNS_ENTER;
1264 int want = 0, force_exit = 0;
1265
1266 /* (can_pns was already checked above; gates below fill `want`) */
1267 if (t->pener[bi] > NMR_PNS_MAX_ET * t->thr_real[bi]) {
1268 force_exit = 1; /* loud-band guard */
1269 } else if (lam > NMR_PNS_HOLE_LAM &&
1270 (frac = ndk[b][t->chosen[b]] * t->thr[bi] /
1271 FFMAX(t->pener[bi], 1e-9f)) > NMR_PNS_HOLE_FRAC * (was ? 0.7f : 1.0f) &&
1272 spread > NMR_PNS_HOLE_SPREAD) {
1273 /* Spectral-hole fill: a noise-like band whose chosen
1274 * rendition leaves most of its energy uncoded turns to
1275 * sparse lines - structure damage the NMR objective
1276 * cannot see - at ANY comfortable lambda */
1277 want = 1;
1278 } else if (lam > pns_lam) {
1279 if (ndk[b][t->chosen[b]] * t->thr[bi] >
1280 NMR_PNS_NDGATE * t->thr_real[bi] * (was ? 0.5f : 1.0f)) {
1281 /* replace only a band coded audibly badly; cost of
1282 * energy-matched noise = its non-noise-like fraction */
1283 nmr_pns = FFMAX(0.0f, t->pener[bi] * (1.0f - spread*spread))
1284 / FFMAX(t->thr[bi], 1e-9f);
1285 cost_keep = ndk[b][t->chosen[b]] + lam * nbk[b][t->chosen[b]];
1286 cost_pns = nmr_pns + lam * NMR_PNS_BITS;
1287 want = cost_pns < cost_keep * bias;
1288 }
1289 }
1290 { /* debounce; near-mask deletion candidates skip entry
1291 * (noise beats the ~silent rendition they'd get) */
1292 uint8_t *ron = &s->nmr->pns_run_on [t->cur_ch & 15][bi];
1293 uint8_t *roff = &s->nmr->pns_run_off[t->cur_ch & 15][bi];
1294 int near = t->pener[bi] < 2.0f * t->thr_real[bi];
1295 if (want) { if (*ron < 255) (*ron)++; *roff = 0; }
1296 else { if (*roff < 255) (*roff)++; *ron = 0; }
1297 if (force_exit)
1298 want = was && *roff < 2; /* tolerate 1-frame loudness blips */
1299 else if (!was)
1300 want = *ron >= (near ? 2 : NMR_PNS_ON);
1301 else if (near)
1302 want = 1; /* physics-hysteresis: noise until audible */
1303 else
1304 want = !(*roff >= NMR_PNS_OFF);
1305 }
1306 if (want) {
1307 t->is_pns[b] = 1;
1308 pns_count++;
1309 }
1310 }
1311 if (pns_count) {
1312 /* filter the active list rather than rebuilding it from
1313 * nbnd: bands shed for legality must stay out */
1314 int n = 0;
1315 for (int b_ = 0; b_ < t->nact; b_++)
1316 if (!t->is_pns[t->act[b_]])
1317 t->act[n++] = t->act[b_];
1318 t->nact = n;
1319 }
1320 pns_total += pns_count;
1321 }
1322 if (pns_total) {
1323 /* re-solve over the survivors: at fixed lambda the allocation is
1324 * the same except for the repaired sf-delta chain; in bisection
1325 * mode re-spend the freed budget */
1326 if (rc_global || vbr)
1327 nmr_eval_slots(s, sl, nsl, NMR_STEP, lam);
1328 else
1329 nmr_solve_slots(s, sl, nsl, NMR_STEP, destbits - pns_total * NMR_PNS_BITS,
1330 1e-9f, 1e4f, NMR_ITERS);
1331 }
1332 }
1333
1334 for (int k = 0; k < nsl; k++) {
1335 NMRSlot *t = sl[k];
1336 uint8_t *pp = s->nmr->pns_prev[t->cur_ch & 15];
1337 uint8_t now[128] = {0};
1338 for (int b = 0; b < t->nbnd; b++)
1339 if (t->is_pns[b])
1340 now[t->bidx[b]] = 1;
1341 memcpy(pp, now, 128);
1342 }
1343 for (int k = 0; k < nsl; k++)
1344 nmr_commit_channel(s, sl[k]);
1346}
1347
1348static void search_for_quantizers_nmr(AVCodecContext *avctx,
1351 const float lambda)
1352{
1353 AACNMRCurves *n = s->nmr;
1354 /* Global-lambda RC: one solve per frame at a servoed centre lambda; the reservoir
1355 * holds the long-run mean rate. Bypassed for VBR (-q:a) and the bootstrap frame. */
1356 int rc_eligible = !(avctx->flags & AV_CODEC_FLAG_QSCALE) && avctx->bit_rate > 0 &&
1357 avctx->bit_rate_tolerance != 0 && s->options.rc == 0;
1358 /* Signed reservoir; soft steering (bounded repay + rc_off), hard cap =
1359 * legality only. */
1360 int rc_rate_frame = avctx->bit_rate * 1024.0 / avctx->sample_rate;
1361 int rc_bmax = FFMIN(FFMAX(6144 * s->channels - rc_rate_frame, 256), NMR_CBR_BUF * s->channels);
1362
1363 int rc_global, defer;
1364 NMRSlot *t;
1365
1366 s->nmr->counted[s->cur_channel] = 0;
1367
1368 if (s->options.rc == 1 && !(avctx->flags & AV_CODEC_FLAG_QSCALE) &&
1369 avctx->bit_rate > 0 && avctx->frame_num != n->abr_frame_num) {
1370 /* ABR servo: integrate the log rate error, but apply it to the nd
1371 * set-point in RARE, DISCRETE steps. The set-point must be
1372 * quasi-static: any drift on content timescales moves lambda, and
1373 * with it band/PNS/scalefactor state - measured 2x worse than a
1374 * fixed target at equal mean rate. Locally this mode IS fixed-target
1375 * VBR; rate honesty converges on the minutes scale. */
1376 if (n->abr_frame_num == 0 && n->abr_ema <= 0.0f) {
1377 /* seed the set-point from the VBR calibration anchor (see
1378 * NMR_VBR_ANCHOR for the 81.5 kbps/ch reference); the servo
1379 * trims the rest */
1380 n->abr_t = av_clipf(NMR_VBR_ANCHOR - 2.5f * log2f(avctx->bit_rate /
1381 (81500.0f * s->channels)), NMR_ABR_TMIN, NMR_ABR_TMAX);
1382 n->abr_ema = rc_rate_frame;
1383 } else if (s->last_frame_pb_count > 0) {
1384 /* bootstrap: fast rate measurement for ~2s, then one open-loop
1385 * jump over the measured loop gain (~ -0.24 log2 rate per target
1386 * unit) corrects the seed's per-content error; the quasi-static
1387 * stepper handles drift from there */
1388 /* seed on-target: an EMA warming up from zero reads as a fake
1389 * deficit and the fill overspends the whole stream head */
1390 if (n->abr_ema <= 0.0f)
1391 n->abr_ema = rc_rate_frame;
1392 n->abr_ema += (n->abr_booted ? NMR_ABR_EMA : 0.02f) *
1393 (s->last_frame_pb_count - n->abr_ema);
1394 n->abr_hold++;
1395 if (!n->abr_booted) {
1397 /* boot only off a REPRESENTATIVE window: an all-transient
1398 * head (castanets roll) reads over-target and would coarsen
1399 * the very content that needs bits. Time out eventually. */
1400 if ((n->abr_hold >= 86 && n->abr_longs >= n->abr_hold / 2) ||
1401 n->abr_hold >= 400) {
1402 /* glide the seed correction in, never step it: a
1403 * one-frame noise-floor jump at a fixed stream time is
1404 * audible against revealing content */
1405 n->abr_glide = NMR_ABR_BOOT_GAIN * log2f(n->abr_ema / rc_rate_frame) / 0.24f;
1406 n->abr_booted = 1;
1407 n->abr_boots++;
1408 n->abr_hold = 0;
1409 }
1410 } else {
1411 /* Re-arm the open-loop correction while the rate is still
1412 * off. One boot leaves a residual whenever the real loop gain
1413 * differs from the nominal 0.24, and the estimator is biased
1414 * toward the ask by construction (the EMA is seeded ON target
1415 * so a cold start cannot read as a deficit), so it understates
1416 * the error it is correcting. The quasi-static stepper cannot
1417 * drain that inside a track - measured, male_speech ended a
1418 * 774-frame file 23% short and still moving. Re-measuring and
1419 * firing again converges regardless of the gain estimate, and
1420 * keeps the set-point quasi-static: corrections stay rare,
1421 * glided, and bounded in number. */
1422 if (n->abr_boots < NMR_ABR_BOOTS && n->abr_glide == 0.0f &&
1423 n->abr_hold >= NMR_ABR_SETTLE &&
1424 fabsf(log2f(n->abr_ema / rc_rate_frame)) > NMR_ABR_TOL) {
1425 /* Re-seed the estimator ON target, exactly as the cold
1426 * start does - never to zero. The value is read again
1427 * before anything re-seeds it: by the integrator below
1428 * (log2f(0) is -inf, which pinned the accumulator at its
1429 * clip) and by the bits-fill later in this same frame
1430 * (a zero EMA reads as a 100% deficit and authorises a
1431 * one-frame overspend). Drop the residual integral with
1432 * it: it describes the regime being left, and the boot
1433 * is about to re-measure that same error open-loop. */
1434 n->abr_booted = 0;
1435 n->abr_ema = rc_rate_frame;
1436 n->abr_acc = 0.0f;
1437 n->abr_hold = 0;
1438 n->abr_longs = 0;
1439 } else {
1440 n->abr_acc = av_clipf(n->abr_acc + NMR_ABR_K * log2f(n->abr_ema / rc_rate_frame),
1441 -4.0f * NMR_ABR_STEP, 4.0f * NMR_ABR_STEP);
1442 /* step in calm stretches (a set-point move during a transient
1443 * section coarsens exactly what needs the bits) - but dense
1444 * content must not deadlock the servo: after 3x the hold the
1445 * step fires regardless. Large errors step at double size;
1446 * the remainder carries over instead of being discarded. */
1447 if (fabsf(n->abr_acc) >= NMR_ABR_STEP && n->abr_hold >= NMR_ABR_HOLD &&
1448 (n->frames_since_short >= 12 || n->abr_hold >= 3 * NMR_ABR_HOLD)) {
1449 float ms = NMR_ABR_STEP * (fabsf(n->abr_acc) > 3.0f * NMR_ABR_STEP ? 2.0f : 1.0f);
1450 float step = av_clipf(n->abr_acc, -ms, ms);
1451 n->abr_glide += step;
1452 n->abr_acc -= step;
1453 n->abr_hold = 0;
1454 }
1455 }
1456 }
1457 if (n->abr_glide != 0.0f) {
1458 /* drain pending set-point corrections smoothly (~0.5
1459 * log2-units/s): the decision stays quasi-static and
1460 * calm-gated, only the application glides */
1461 float d = av_clipf(n->abr_glide, -0.012f, 0.012f);
1463 n->abr_glide -= d;
1464 if (fabsf(n->abr_glide) < 1e-4f)
1465 n->abr_glide = 0.0f;
1466 }
1467 }
1468 n->abr_frame_num = avctx->frame_num;
1469 }
1470 if (rc_eligible && !n->rc_fill_seeded) {
1471 /* the decoder bit reservoir starts FULL: seed it so the head may frontload */
1472 n->rc_fill = rc_bmax;
1473 n->rc_fill_seeded = 1;
1474 }
1475 if (rc_eligible && avctx->frame_num != n->rc_frame_num) {
1476 if (n->rc_frame_num > 0 && n->lam_rc > 0.0f)
1477 n->rc_fill = av_clip(n->rc_fill + rc_rate_frame - s->last_frame_pb_count,
1478 -rc_bmax, rc_bmax);
1479 n->rc_frame_num = avctx->frame_num;
1480 n->pending = 0; /* a deferred first channel never crosses a frame */
1481 /* consecutive saturated frames, once per frame across all element
1482 * groups: any frame without a saturated overage ends the run */
1483 n->rc_satrun = n->rc_sat_frame ? n->rc_satrun + 1 : 0;
1484 n->rc_sat_frame = 0;
1485 /* latch the RC mode per frame: a mid-frame bootstrap must not flip
1486 * the CPE defer logic between channels */
1487 n->rc_gl = rc_eligible && n->lam_rc > 0.0f;
1488 }
1489 if (avctx->frame_num != n->win_frame_num) {
1490 /* Window history, once per frame in every rate-control mode (the
1491 * quality-target slew and the ABR servo read it too).
1492 * Transient burst run state: set at run start and held across the run so
1493 * coding stays uniform; repaid from the reservoir's steady stretches. */
1494 int is_short = sce->ics.window_sequence[0] == EIGHT_SHORT_SEQUENCE;
1495 n->win_frame_num = avctx->frame_num;
1496 if (is_short) {
1497 if (!n->prev_was_short) { /* run start */
1500 } else {
1501 /* dense-beat boost, scaled by measured short-frame starvation */
1502 float imb = 0.0f;
1503 if (n->lam_long_ema > 0.0f && n->lam_short_ema > 0.0f)
1504 imb = av_clipf(n->lam_short_ema / n->lam_long_ema - 1.0f,
1505 0.0f, 1.0f);
1506 n->run_burst = 1.0f + (NMR_SHORT_BOOST - 1.0f) * imb *
1508 }
1509 }
1510 n->frames_since_short = 0;
1511 } else {
1512 /* the frame closing a run (the STOP) absorbs the corridor recoil
1513 * of the boosted shorts; give it half the run's factor so the
1514 * repayment spreads into the steady stretch instead */
1515 n->run_burst = n->prev_was_short ? sqrtf(n->run_burst) : 1.0f;
1516 n->frames_since_short++;
1517 }
1518 n->prev_was_short = is_short;
1519 }
1520 rc_global = rc_eligible && n->rc_gl;
1521
1522 /* CPE budget pool: under global-lambda RC, defer the pair's first channel
1523 * and solve both against one pooled budget when the second one arrives. */
1524 defer = n->pair && rc_global;
1525
1526 t = &n->slot[(defer && n->pending) ? 1 : 0];
1527 t->si = (defer && n->pending) ? 1 : 0;
1528
1529 if (!nmr_setup_channel(avctx, s, sce, t)) {
1530 nmr_bail_channel(sce);
1531 t->nbnd = t->nact = 0;
1532 }
1533
1534 if (defer && !n->pending) {
1535 n->pending = 1; /* wait for the partner channel */
1536 return;
1537 }
1538
1539 {
1540 NMRSlot *sl[2];
1541 int nsl = 0, chans = 1;
1542 if (defer) {
1543 n->pending = 0;
1544 chans = 2;
1545 if (n->slot[0].nact)
1546 sl[nsl++] = &n->slot[0];
1547 if (n->slot[1].nact)
1548 sl[nsl++] = &n->slot[1];
1549 } else if (t->nact) {
1550 sl[nsl++] = t;
1551 }
1552 if (!nsl)
1553 return; /* nothing codeable in the group */
1554 nmr_solve_group(avctx, s, lambda, sl, nsl, chans,
1555 rc_eligible, rc_global, rc_rate_frame, rc_bmax);
1556 }
1557}
1558
1559#endif /* AVCODEC_AACCODER_NMR_H */
AAC definitions and structures.
#define SCALE_MAX_DIFF
maximum scalefactor difference allowed by standard
Definition aac.h:94
@ EIGHT_SHORT_SEQUENCE
Definition aac.h:66
@ LONG_START_SEQUENCE
Definition aac.h:65
@ INTENSITY_BT
Scalefactor data are intensity stereo positions (in phase).
Definition aac.h:77
@ INTENSITY_BT2
Scalefactor data are intensity stereo positions (out of phase).
Definition aac.h:76
@ RESERVED_BT
Band types following are encoded differently from others.
Definition aac.h:74
@ NOISE_BT
Spectral data are scaled white noise not coded in the bitstream.
Definition aac.h:75
#define SCALE_MAX_POS
scalefactor index maximum value
Definition aac.h:93
#define NMR_VBR_TMIN
static void search_for_quantizers_nmr(AVCodecContext *avctx, AACEncContext *s, SingleChannelElement *sce, const float lambda)
#define NMR_GROUP_KEEP
#define NMR_RC_CITERS
#define NMR_CBR_BUF
#define NMR_ITERS
#define NMR_SLEW
static float nmr_solve_slots(AACEncContext *s, NMRSlot *const *sl, int nsl, int step, int destbits, float lo_l, float hi_l, int iters)
static int nmr_setup_channel(AVCodecContext *avctx, AACEncContext *s, SingleChannelElement *sce, NMRSlot *t)
static void nmr_solve_group(AVCodecContext *avctx, AACEncContext *s, const float lambda, NMRSlot *const *sl, int nsl, int chans, int rc_eligible, int rc_global, int rc_rate_frame, int rc_bmax)
#define NMR_PNS_HOLE_SPREAD
#define NMR_RC_TRACK
#define NMR_PNS_HOLE_LAM
#define NMR_PNS_STAY
#define NMR_PNS_HOLE_FRAC
#define NMR_VBR_ANCHOR
#define NMR_PNS_ON
#define NMR_RC_ITERS
#define NMR_PNS_BITS
#define NMR_COARSE
static float nmr_solve_slots_nd(AACEncContext *s, NMRSlot *const *sl, int nsl, int step, float target, float lo_l, float hi_l, int iters)
#define NMR_ABR_EMA
#define NMR_PNS_ENTER
#define NMR_TRANS_PM
#define NMR_ABR_BOOTS
static int nmr_slot_bits(const NMRSlot *t, const int(*nb)[NMR_NCAND], int step)
#define NMR_ABR_TMAX
#define NMR_BURST_GAIN
#define NMR_SHORT_BOOST
#define NMR_HF_TAPER
#define NMR_ABR_TOL
#define NMR_INFL_HF
#define NMR_RC_CAPK
static int nmr_eval_slots(AACEncContext *s, NMRSlot *const *sl, int nsl, int step, float lam)
#define NMR_SLEW_RUN
static int nmr_band_curve(AACEncContext *s, SingleChannelElement *sce, int w, int g, int start, int lo, int step, int maxn, float invthr, float maxval, float *nd_row, int *nb_row)
#define NMR_SFBITS(d)
AAC encoder NMR scalefactor coder.
static float nmr_solve(AACEncContext *s, const float(*nd)[NMR_NCAND], const int(*nb)[NMR_NCAND], const int *blo, const int *bnc, int step, const int *act, int nact, int destbits, int *chosen, float lo_l, float hi_l, int iters)
Viterbi over the coding sequence act[0..nact-1] (indices into the per-band curves nd/nb),...
#define NMR_CWARM
#define NMR_ABR_SETTLE
#define NMR_RC_K_CBR
#define NMR_ABR_HOLD
#define NMR_STEP
#define NMR_PNS_OFF
#define NMR_IFINE
#define NMR_ABR_BOOT_GAIN
static float nmr_nd_stat(AACEncContext *s, NMRSlot *const *sl, int nsl)
#define NMR_CITERS
#define NMR_RC_FITERS
#define NMR_ABR_STEP
#define NMR_ABR_K
#define NMR_HF_TAPER_KNEE
#define NMR_BURST_GAP
#define NMR_PNS_LAM
#define NMR_ZERO_STICKY
#define NMR_PNS_MAX_ET
static void nmr_commit_channel(AACEncContext *s, NMRSlot *t)
static void nmr_bail_channel(SingleChannelElement *sce)
#define NMR_ABR_TMIN
#define NMR_PNS_NDGATE
#define NMR_RC_CORR
void ff_quantize_band_cost_cache_init(struct AACEncContext *s)
Definition aacenc.c:557
#define NMR_NCAND
per-band scalefactor candidates above the finest codeable sf (NMR coder)
Definition aacenc.h:173
static float quantize_band_cost_cached(struct AACEncContext *s, int w, int g, const float *in, const float *scaled, int size, int scale_idx, int cb, const float lambda, const float uplim, int *bits, float *energy, int rtz)
static void ff_init_nextband_map(const SingleChannelElement *sce, uint8_t *nextband)
static int find_min_book(float maxval, int sf)
static float find_max_val(int group_len, int swb_size, const float *scaled)
static uint8_t coef2minsf(float coef)
Return the minimum scalefactor where the quantized coef does not clip.
static int ff_sfdelta_can_remove_band(const SingleChannelElement *sce, const uint8_t *nextband, int prev_sf, int band)
AAC encoder data.
const uint8_t ff_aac_scalefactor_bits[121]
Definition aactab.c:200
AAC data declarations.
static const int8_t filt[NUMTAPS *2]
Definition af_earwax.c:40
static float win(SuperEqualizerContext *s, float n, int N)
#define log2f(x)
Definition math.h:27
Libavcodec external API header.
#define i(width, name, range_min, range_max)
Definition cbs_h264.c:63
#define mb(name)
Definition cbs_lcevc.c:95
#define s(width, name)
Definition cbs_vp9.c:198
#define av_clipd
Definition common.h:148
#define av_clip
Definition common.h:100
#define av_clipf
Definition common.h:145
#define NULL
Definition coverity.c:32
long long int64_t
Definition coverity.c:34
static __device__ float sqrtf(float a)
static __device__ float fabsf(float a)
float fminf(float, float)
#define AV_CODEC_FLAG_QSCALE
Use fixed qscale.
Definition avcodec.h:213
#define FF_QP2LAMBDA
factor to convert from H.263 QP to lambda
Definition avutil.h:226
#define R
Definition huffyuv.h:44
#define b
Definition input.c:43
static void scale(int *out, const int *in, const int w, const int h, const int shift)
Definition intra.c:278
uint8_t w
Definition llvidencdsp.c:39
#define FFMIN(a, b)
Definition macros.h:49
#define FFMAX(a, b)
Definition macros.h:47
#define FFMIN3(a, b, c)
Definition macros.h:50
float exp2f(float x)
Definition math.c:49
#define M_LN10
Definition mathematics.h:49
#define NAN
#define INFINITY
static int headroom(int *la)
Definition nellymoser.c:106
bitstream writer API
AAC encoder context.
Definition aacenc.h:279
int frames_since_short
long-block frames since the last short run (the "gap"): large = isolated transient
Definition aacenc.h:240
int64_t win_frame_num
frame the window history was last advanced for
Definition aacenc.h:242
NMRSlot slot[2]
pair slots (solo solves use slot 0)
Definition aacenc.h:210
int rc_sat_frame
the current frame hit a saturated overage
Definition aacenc.h:239
int rc_satrun
consecutive frames with saturated reservoir debt (cap escalation)
Definition aacenc.h:238
int abr_longs
long frames seen during bootstrap
Definition aacenc.h:262
float run_burst
transient bit-burst factor, set at run start and held across the short run
Definition aacenc.h:243
int64_t abr_frame_num
once-per-frame servo guard
Definition aacenc.h:263
float abr_ema
EMA of real frame bits.
Definition aacenc.h:256
int rc_fill
virtual bit reservoir fill, + = bits saved vs nominal
Definition aacenc.h:237
float abr_t
current nd target (log2 dist/mask)
Definition aacenc.h:254
int rc_gl
rc_global latched at frame start: the corridor bootstrap must not flip the CPE defer logic between ch...
Definition aacenc.h:212
float lam_short_ema
smoothed operating lambda of short frames
Definition aacenc.h:249
int64_t rc_frame_num
frame the reservoir was last advanced for
Definition aacenc.h:235
float lam_long_ema
smoothed operating lambda of long frames
Definition aacenc.h:250
int pending
slot 0 holds a deferred first channel
Definition aacenc.h:214
int abr_boots
open-loop corrections fired so far
Definition aacenc.h:260
float abr_glide
pending set-point correction, drained per-frame (no discrete quality steps)
Definition aacenc.h:255
int abr_booted
seed correction applied (re-armed while the rate is still off)
Definition aacenc.h:259
int abr_hold
frames since the target last stepped
Definition aacenc.h:258
int prev_was_short
previous frame was a short block (for run-start detection)
Definition aacenc.h:241
float lam_rc
global-lambda rate control: operating lambda, 0 until bootstrapped
Definition aacenc.h:236
int rc_fill_seeded
reservoir seeded full at stream start (decoder buffer starts full)
Definition aacenc.h:213
float abr_acc
accumulated set-point correction
Definition aacenc.h:257
int pair
current element is a CPE: pool the pair budget
Definition aacenc.h:211
int nb_channels
Number of channels in this layout.
main external API structure.
Definition avcodec.h:443
AVChannelLayout ch_layout
Audio channel layout.
Definition avcodec.h:1055
int global_quality
Global quality for codecs which cannot change it per frame.
Definition avcodec.h:1235
int64_t frame_num
Frame counter, set by libavcodec.
Definition avcodec.h:1888
int bit_rate_tolerance
number of bits the bitstream is allowed to diverge from the reference.
Definition avcodec.h:1227
int64_t bit_rate
the average bitrate
Definition avcodec.h:493
int sample_rate
samples per second
Definition avcodec.h:1040
int flags
AV_CODEC_FLAG_*.
Definition avcodec.h:500
single band psychoacoustic information
Definition psymodel.h:50
float spread
Definition psymodel.h:54
float threshold
Definition psymodel.h:53
float energy
Definition psymodel.h:52
uint8_t max_sfb
number of scalefactor bands per group
Definition aacdec.h:170
int num_swb
number of scalefactor window bands
Definition aacdec.h:178
uint8_t group_len[8]
Definition aacdec.h:175
const uint8_t * swb_sizes
table of scalefactor band sizes for a particular window
Definition aacenc.h:87
enum WindowSequence window_sequence[2]
Definition aacdec.h:171
const uint16_t * swb_offset
table of offsets to the lowest spectral coefficient of a scalefactor band, sfb, for a particular wind...
Definition aacdec.h:177
NMR coder per-band candidate cost curves (~96 KiB) and rate-control carry-over.
Definition aacenc.h:183
int chosen[128]
Definition aacenc.h:193
float pspread[128]
band tonality spread (1 = noise)
Definition aacenc.h:202
int nact
Definition aacenc.h:195
int minsf[128]
Definition aacenc.h:196
float tnsg[128]
TNS synthesis gain per band for THIS solve (1 = uncovered), M/S-aware (pair max)
Definition aacenc.h:200
int bst[128]
window group, swb, coef start
Definition aacenc.h:190
struct SingleChannelElement * sce
Definition aacenc.h:184
int bnc[128]
number of candidates
Definition aacenc.h:192
uint8_t hftx[128]
band strongly HF-tapered: excluded from the nd stat (deficit is by design)
Definition aacenc.h:204
float thr_real[128]
real masking threshold (PNS gates)
Definition aacenc.h:199
uint8_t is_pns[128]
band coded as noise
Definition aacenc.h:203
int si
curve-bank index (nd/nb slot)
Definition aacenc.h:185
int bw[128]
Definition aacenc.h:190
float pener[128]
band energy (PNS noise target)
Definition aacenc.h:201
int bg[128]
Definition aacenc.h:190
float thr[128]
allocation-law effective threshold
Definition aacenc.h:198
int bidx[128]
sce band index (w*16+g)
Definition aacenc.h:189
int nbnd
coded-band count, 0 = nothing codeable
Definition aacenc.h:187
int blo[128]
finest candidate scalefactor
Definition aacenc.h:191
int act[128]
active (non-PNS) band coding order
Definition aacenc.h:194
float maxvals[128]
Definition aacenc.h:197
int is8
EIGHT_SHORT frame.
Definition aacenc.h:188
int cur_ch
encoder channel index (psy/cache context)
Definition aacenc.h:186
Single Channel Element - used for both SCE and LFE elements.
Definition aacdec.h:217
uint8_t zeroes[128]
band is not coded
Definition aacenc.h:118
float coeffs[1024]
coefficients for IMDCT, maybe processed
Definition aacenc.h:123
uint8_t can_pns[128]
band is allowed to PNS (informative)
Definition aacenc.h:119
TemporalNoiseShaping tns
Definition aacdec.h:220
float pns_ener[128]
Noise energy values.
Definition aacenc.h:121
enum BandType band_type[128]
band types
Definition aacdec.h:221
IndividualChannelStream ics
Definition aacdec.h:218
int sf_idx[128]
scalefactor indices
Definition aacenc.h:117
int length[8][4]
Definition aacdec.h:194
int order[8][4]
Definition aacdec.h:196
#define imb
const char * g
Definition vf_curves.c:128
static double cb(void *priv, double x, double y)
Definition vf_geq.c:247
static float mean(const float *input, int size)
Definition vf_nnedi.c:861
static double b0(void *priv, double x, double y)
Definition vf_xfade.c:2040
static int bias(int x, int c)
Definition vqcdec.c:115
static double c[64]