FFmpeg
Loading...
Searching...
No Matches
aaccoder_nmr.h
Go to the documentation of this file.
1/*
2 * AAC encoder NMR (noise-to-mask ratio) scalefactor coder
3 * Copyright (c) 2026 Lynne <dev@lynne.ee>
4 *
5 * This file is part of FFmpeg.
6 *
7 * FFmpeg is free software; you can redistribute it and/or
8 * modify it under the terms of the GNU Lesser General Public
9 * License as published by the Free Software Foundation; either
10 * version 2.1 of the License, or (at your option) any later version.
11 *
12 * FFmpeg is distributed in the hope that it will be useful,
13 * but WITHOUT ANY WARRANTY; without even the implied warranty of
14 * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU
15 * Lesser General Public License for more details.
16 *
17 * You should have received a copy of the GNU Lesser General Public
18 * License along with FFmpeg; if not, write to the Free Software
19 * Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA
20 */
21
22/**
23 * AAC encoder NMR scalefactor coder.
24 *
25 * Optimizes the same noise-to-mask objective as the two-loop coder, but with an
26 * optimal Viterbi search over scalefactors instead of a heuristic loop. For each
27 * coded band the per-scalefactor distortion/bits curve is precomputed, then a
28 * trellis over the (window-group, band) coding sequence minimizes
29 * sum_g = dist_g(sf_g)/threshold_g +
30 * lambda * (spectral_bits_g(sf_g) + scalefactor_differential_bits)
31 * with |sf_g - sf_{g-1}| <= SCALE_MAX_DIFF as a constraint, and lambda
32 * binary-searched so the coded size meets the per-frame bit budget
33 *
34 * Perceptual noise substitution (PNS) is integrated into the same objective: once
35 * the trellis settles on its operating lambda, each noise-like band (flagged by
36 * mark_pns) is offered a terminal "code as noise" candidate whose cost is
37 * nmr_pns + lambda*NMR_PNS_BITS. Because NMR_PNS_BITS is far below a band's spectral bit
38 * count, this candidate only wins when lambda is large, i.e. when the encoder is
39 * struggling to hold the bitrate. The bits freed by the chosen PNS bands are
40 * then re-spent by a second trellis pass over the remaining bands.
41 */
42
43#ifndef AVCODEC_AACCODER_NMR_H
44#define AVCODEC_AACCODER_NMR_H
45
46#include <float.h>
47#include <string.h>
49#include "mathops.h"
50#include "avcodec.h"
51#include "put_bits.h"
52#include "aac.h"
53#include "aacenc.h"
54#include "aactab.h"
55#include "aacenctab.h"
56
57/* differential scalefactor coding cost, clamped to the legal delta range */
58#define NMR_SFBITS(d) ff_aac_scalefactor_bits[av_clip((d) + SCALE_DIFF_ZERO, 0, 2*SCALE_MAX_DIFF)]
59
60#define NMR_ITERS 14 /* lambda binary-search iters */
61#define NMR_IFINE 9 /* fine-pass lambda iters */
62#define NMR_CITERS 7 /* coarse-pass lambda iters */
63#define NMR_CWARM 5 /* coarse-pass iters when warm-started off the previous frame's
64 * lambda: the bracket spans 10 octaves instead of ~43, so fewer
65 * bisection steps reach the same resolution */
66#define NMR_COARSE 8 /* two-pass coarse->fine grid step, cuts the Viterbi ncand^2 with no
67 * quality loss, 0 disables it (single full-resolution pass) */
68#define NMR_STEP 1 /* fine-pass scalefactor candidate granularity */
69
70#define NMR_PNS_BITS 9 /* approx cost in bits of signalling PNS */
72/* Spectral-hole fill: noise-like bands the trellis left mostly empty are filled with
73 * energy-matched noise (PNS); an audible hole sounds worse than matched noise. */
74#define NMR_PNS_HOLE_FRAC 0.5f
75#define NMR_PNS_HOLE_SPREAD 0.5f
77/* RC servo gain: scale the corridor centre by exp2(-K*fill/R) each frame to hold
78 * the long-run mean rate; without it a bad centre drifts for dozens of frames. */
79#define NMR_RC_K_CBR 0.5f
80
81#define NMR_RC_ITERS 8 /* lambda bisection iters when clamping an over-cap frame */
82/* Corridor: bisect within [lam_rc/NMR_RC_CORR, lam_rc*NMR_RC_CORR] so quality stays
83 * smooth while per-frame demand is tracked; 1.5 cuts lambda jitter ~25%. */
84#define NMR_RC_CORR 1.5f
85
86/* Reservoir half-window (bits/ch); swept 512/1536/3072, 1536 optimal. */
87#define NMR_CBR_BUF 1536
88/* Slew limit on the FINAL operating lambda per frame; bits deviate instead,
89 * the reservoir absorbs. See memory: aac-castanets-transient-rc. */
90#define NMR_SLEW 1.6f
91#define NMR_SLEW_RUN 1.15f /* within short runs */
92#define NMR_RC_CITERS 3 /* corridor coarse-pass iters */
94/* Transition premask: an attack cannot mask backwards; clamp a START frame's
95 * thresholds toward the previous long frame's. */
96#define NMR_TRANS_PM 2.0f
98/* Zero-decision hysteresis: previously-coded bands need this margin below
99 * threshold to zero (marginal bands flicker audibly otherwise). */
100#define NMR_ZERO_STICKY 0.5f
102/* Transient bit-burst: an isolated onset (preceded by >= NMR_BURST_GAP long frames)
103 * is coded NMR_BURST_GAIN x finer, held uniform across the run, repaid from steady stretches. */
104#define NMR_BURST_GAP 10
105#define NMR_BURST_GAIN 8.0f
106/* Dense-beat boost: short runs with gap < NMR_BURST_GAP get a budget factor
107 * ramping with the gap (starvation-scaled at the use site). */
108#define NMR_SHORT_BOOST 2.0f
109#define NMR_RC_FITERS 4 /* corridor fine-pass iters */
110#define NMR_RC_TRACK 0.1f /* per-frame pull of the corridor centre toward the realized lambda */
111
112/* PNS noise-distortion gate: only bands coded well above the masking floor become noise. */
113#define NMR_PNS_NDGATE 4.0f
115/* Energy/threshold cap for PNS: loud bands (energy >> mask) yield clipping random peaks;
116 * only near-masked bands are safe substitution targets. */
117#define NMR_PNS_MAX_ET 8.0f
119/* Operating-lambda floor for PNS: below it the encoder is not struggling, so
120 * substituting real texture for 9 signalling bits is net-negative. */
121#define NMR_PNS_LAM 100.0f
123/* PNS decision hysteresis: enter and leave both cost a margin. */
124#define NMR_PNS_ENTER 0.7f
125#define NMR_PNS_STAY 1.4f
126/* PNS debounce: enter after NMR_PNS_ON consecutive wants, leave after
127 * NMR_PNS_OFF (chronically marginal bands never qualify). */
128#define NMR_PNS_ON 8
129#define NMR_PNS_OFF 4
130
131/**
132 * Viterbi over the coding sequence act[0..nact-1] (indices into the per-band
133 * curves nd/nb), with lambda binary-searched so the coded size ~ destbits.
134 * Fills chosen[band] for every band referenced by act. Returns the operating
135 * lambda. node cost = dist/threshold + lambda*spectral_bits;
136 * edge cost = lambda*sf_differential_bits; |delta sf| <= SCALE_MAX_DIFF hard.
137 */
138static float nmr_solve(AACEncContext *s,
139 const float (*nd)[NMR_NCAND], const int (*nb)[NMR_NCAND],
140 const int *blo, const int *bnc, int step,
141 const int *act, int nact, int destbits, int *chosen,
142 float lo_l, float hi_l, int iters)
143{
144 float dp[NMR_NCAND], dpp[NMR_NCAND], node[NMR_NCAND];
145 float lamsf[2*SCALE_MAX_DIFF + 1]; /* lam*sfdiff bit cost, per lambda */
146 uint8_t bp[128][NMR_NCAND];
147 float lam = 1.0f;
148
149 if (nact <= 0)
150 return lam;
151
152 for (int it = 0; it < iters; it++) {
153 lam = sqrtf(lo_l * hi_l);
154 for (int i = 0; i <= 2*SCALE_MAX_DIFF; i++)
155 lamsf[i] = lam * ff_aac_scalefactor_bits[i]; /* edge cost for this lambda */
156
157 int b0 = act[0];
158 for (int o = 0; o < bnc[b0]; o++)
159 dp[o] = nd[b0][o] + lam * nb[b0][o]; /* anchor band node cost */
160
161 for (int k = 1; k < nact; k++) {
162 int b = act[k], pb = act[k-1];
163 memcpy(dpp, dp, sizeof(dp));
164 for (int o = 0; o < bnc[b]; o++)
165 node[o] = nd[b][o] + lam * nb[b][o];
166 /* dp[o] = node[o] + min_op(dpp[op] + edge cost) */
167 s->aacdsp.nmr_trellis_step(dp, bp[k], dpp, node, lamsf,
168 bnc[b], bnc[pb], blo[b] - blo[pb], step,
170 }
171
172 /* backtrack */
173 int beo = 0, b = act[nact-1];
174 float bec = FLT_MAX;
175 for (int o = 0; o < bnc[b]; o++)
176 if (dp[o] < bec) { bec = dp[o]; beo = o; }
177 chosen[b] = beo;
178 for (int k = nact-1; k > 0; k--)
179 chosen[act[k-1]] = bp[k][chosen[act[k]]];
180
181 /* calc cost */
182 int total = 0;
183 for (int k = 0; k < nact; k++)
184 total += nb[act[k]][chosen[act[k]]];
185 for (int k = 1; k < nact; k++)
186 total += NMR_SFBITS((blo[act[k]]+chosen[act[k]]*step) - (blo[act[k-1]]+chosen[act[k-1]]*step));
187
188 if (it == iters - 1)
189 break;
190
191 /* check if we went over budget, go coarser if we did */
192 if (total > destbits)
193 lo_l = lam;
194 else
195 hi_l = lam;
196 }
197 return lam;
198}
200/* Build one coded band's (dist/threshold, bits) cost curve, candidates sf = lo + o*step
201 * for o in [0,maxn), stopping when the band would drop (cb <= 0). Returns the bit count. */
202static int nmr_band_curve(AACEncContext *s, SingleChannelElement *sce, int w, int g,
203 int start, int lo, int step, int maxn, float invthr,
204 float maxval, float *nd_row, int *nb_row)
205{
206 int ncand = 0;
207 for (int o = 0; o < maxn && lo + o*step <= SCALE_MAX_POS; o++) {
208 int sf = lo + o*step, btot = 0, cb = find_min_book(maxval, sf);
209 float dist = 0.0f;
210 if (cb <= 0)
211 break;
212 for (int w2 = 0; w2 < sce->ics.group_len[w]; w2++) {
213 int bb;
214 dist += quantize_band_cost_cached(s, w + w2, g, sce->coeffs + start + w2*128,
215 s->scoefs + start + w2*128, sce->ics.swb_sizes[g],
216 sf, cb, 1.0f, INFINITY, &bb, NULL, 0);
217 btot += bb;
218 }
219 nd_row[ncand] = (dist - btot) * invthr;
220 nb_row[ncand] = btot;
221 ncand++;
222 }
223 return ncand;
224}
226/* Zero a channel with nothing codeable; stale band_types would resurrect
227 * bands with chain-illegal scalefactors. */
229{
230 for (int i = 0; i < 128; i++) {
231 if (sce->band_type[i] == INTENSITY_BT || sce->band_type[i] == INTENSITY_BT2)
232 continue;
233 sce->zeroes[i] = 1;
234 sce->band_type[i] = 0;
235 }
236}
237
238/* Per-channel setup into slot t: short-block threshold shaping, the
239 * allocation law, zero decisions, and the PASS 1 coarse candidate curves.
240 * Returns the coded-band count; 0 = nothing codeable (caller bails). */
243{
244 float (*nd)[NMR_NCAND] = s->nmr->nd[t->si];
245 int (*nb)[NMR_NCAND] = s->nmr->nb[t->si];
246 const int cstep = NMR_COARSE > 0 ? NMR_COARSE : NMR_STEP;
247 int allz = 0, cutoff = 1024, nbnd = 0;
248
249 uint8_t *zprev = s->nmr->zero_prev[s->cur_channel & 15];
250 if (s->nmr->zero_nw[s->cur_channel & 15] != sce->ics.num_windows) {
251 memset(zprev, 1, 128);
252 s->nmr->zero_nw[s->cur_channel & 15] = sce->ics.num_windows;
253 }
254
255 t->sce = sce;
256 t->cur_ch = s->cur_channel;
258 t->nbnd = t->nact = 0;
259
260 /* band cutoff index for this frame's window size; the bandwidth is fixed
261 * at init and shared with the psy model */
262 cutoff = s->bandwidth * 2 * (1024 / sce->ics.num_windows) / avctx->sample_rate;
263
264 /* Short-block shaping: temporal premask + per-window threshold flatten. */
266 const float pm_p1 = 0.1f, pm_p2 = 2.0f, pm_p3 = 4.0f;
267 for (int g = 0; g < sce->ics.num_swb; g++) {
268 float t1 = FLT_MAX, t2 = FLT_MAX; /* original thr of w-1, w-2 */
269 for (int w = 0; w < sce->ics.num_windows; w++) {
270 FFPsyBand *b = &s->psy.ch[s->cur_channel].psy_bands[w*16+g];
271 float th = b->threshold;
272 float c = FFMIN(th, FFMIN(t1*pm_p2, t2*pm_p3));
273 b->threshold = FFMAX(c, th*pm_p1);
274 t2 = t1; t1 = th;
275 }
276 }
277 {
278 for (int w = 0; w < sce->ics.num_windows; w++) {
279 float sum = 0.0f, esum = 0.0f; int n = 0;
280 for (int g = 0; g < sce->ics.num_swb; g++) {
281 FFPsyBand *b = &s->psy.ch[s->cur_channel].psy_bands[w*16+g];
282 if (b->energy > b->threshold && b->threshold > 0.0f) { sum += b->threshold; esum += b->energy; n++; }
283 }
284 if (n > 0) {
285 /* keep each window codeable: cap the mean 12dB under the
286 * window's mean audible energy */
287 float mean = FFMIN(sum / n, (esum / n) * expf(-12.0f * (float)M_LN10 / 10.0f));
288 for (int g = 0; g < sce->ics.num_swb; g++) {
289 FFPsyBand *b = &s->psy.ch[s->cur_channel].psy_bands[w*16+g];
290 if (b->energy > b->threshold && b->threshold > 0.0f)
291 b->threshold = FFMIN(mean, b->threshold * 1e9f);
292 }
293 }
294 }
295 }
296 }
297
298 /* Allocation law; short frames blend to softer energy exponents under
299 * pressure (roll anti-starvation, see memory). */
300 float a_ae = 0.443f, a_at = 0.111f;
301 if (sce->ics.num_windows == 8 && s->nmr) {
302 /* blend to mask-weighted exponents under rate pressure */
303 a_ae += (0.35f - a_ae) * s->nmr->press;
304 a_at += (0.3f - a_at) * s->nmr->press;
305 }
306 for (int w = 0; w < sce->ics.num_windows; w += sce->ics.group_len[w]) {
307 int start = 0;
308 for (int g = 0; g < sce->ics.num_swb; start += sce->ics.swb_sizes[g++]) {
309 float uplim = 0.0f, ener = 0.0f, spread = 2.0f;
310 int nz = 0;
311 if (sce->band_type[w*16+g] == INTENSITY_BT ||
312 sce->band_type[w*16+g] == INTENSITY_BT2) {
313 /* pre-decided intensity band (right channel): keep its
314 * signalling, it is not trellis-coded */
315 for (int w2 = 0; w2 < sce->ics.group_len[w]; w2++)
316 sce->zeroes[(w+w2)*16+g] = 0;
317 continue;
318 }
319 float zthr_mul = zprev[w*16+g] ? 1.0f : NMR_ZERO_STICKY;
320 /* M/S side bands: zero-reluctance scaled by side/mid ratio (a tiny
321 * side IS the image; zeroing it flickers). */
322 if ((t->cur_ch & 1) && s->nmr && s->nmr->pair &&
323 s->nmr->smode_band[(t->cur_ch >> 1) & 7][w*16+g] == 1) {
324 const FFPsyBand *mb = &s->psy.ch[s->cur_channel - 1].psy_bands[w*16+g];
325 float ratio = 0.0f;
326 float eside = 0.0f;
327 for (int w2 = 0; w2 < sce->ics.group_len[w]; w2++) {
328 const FFPsyBand *bb = &s->psy.ch[s->cur_channel].psy_bands[(w+w2)*16+g];
329 eside += bb->energy;
330 }
331 ratio = eside / FFMAX(mb->energy * sce->ics.group_len[w], 1e-9f);
332 zthr_mul *= 0.25f + 0.75f * av_clipf(ratio / 0.3f, 0.0f, 1.0f);
333 }
334 for (int w2 = 0; w2 < sce->ics.group_len[w]; w2++) {
335 FFPsyBand *band = &s->psy.ch[s->cur_channel].psy_bands[(w+w2)*16+g];
336 ener += band->energy;
337 spread = FFMIN(spread, band->spread);
338 if (start >= cutoff || band->energy <= band->threshold * zthr_mul ||
339 band->threshold == 0.0f) {
340 sce->zeroes[(w+w2)*16+g] = 1;
341 continue;
342 }
343 uplim += band->threshold;
344 nz = 1;
345 }
346 zprev[w*16+g] = !nz;
347 sce->zeroes[w*16+g] = !nz;
348 t->thr_real[w*16+g] = uplim; /* real mask, before the allocation law (PNS gate) */
349 if (nz && ener > 0.0f && uplim > 0.0f) /* allocation law */
350 uplim = expf(a_ae * logf(ener) + a_at * logf(uplim));
351 t->thr[w*16+g] = uplim;
352 t->pener[w*16+g] = ener;
353 t->pspread[w*16+g] = spread;
354 allz |= nz;
355 }
356 }
357 if (!allz)
358 return 0;
359
360 /* transition premask (see NMR_TRANS_PM) */
361 if (sce->ics.num_windows == 1) {
362 int ci = t->cur_ch & 15;
363 if (sce->ics.window_sequence[0] == LONG_START_SEQUENCE &&
364 s->nmr->thr_prev_ok[ci]) {
365 for (int g = 0; g < sce->ics.num_swb && g < 64; g++)
366 if (t->thr[g] > 0.0f && s->nmr->thr_prev[ci][g] > 0.0f)
367 t->thr[g] = FFMIN(t->thr[g], s->nmr->thr_prev[ci][g] * NMR_TRANS_PM);
368 }
369 for (int g = 0; g < sce->ics.num_swb && g < 64; g++)
370 s->nmr->thr_prev[ci][g] = t->thr[g];
371 s->nmr->thr_prev_ok[ci] = 1;
372 } else {
373 s->nmr->thr_prev_ok[t->cur_ch & 15] = 0;
374 }
375
376 s->aacdsp.abs_pow34(s->scoefs, sce->coeffs, 1024);
378
379 /* TNS synthesis gain per band: the decoder re-amplifies residual-domain
380 * quantization noise by the whitening gain (shorts only). */
381 for (int i = 0; i < 128; i++)
382 t->tnsg[i] = 1.0f;
383 if (sce->ics.num_windows == 8 && sce->tns.present) {
384 const int mmm2 = FFMIN(sce->ics.tns_max_bands, sce->ics.max_sfb ? sce->ics.max_sfb : sce->ics.num_swb);
385 for (int w = 0; w < 8; w++) {
386 int bottom2 = sce->ics.num_swb;
387 for (int filt = 0; filt < sce->tns.n_filt[w]; filt++) {
388 int top2 = bottom2;
389 bottom2 = FFMAX(0, top2 - sce->tns.length[w][filt]);
390 if (!sce->tns.order[w][filt])
391 continue;
392 for (int g = FFMIN(bottom2, mmm2); g < FFMIN(top2, mmm2); g++) {
393 int s0 = sce->ics.swb_offset[g] + w*128;
394 int s1 = sce->ics.swb_offset[g+1] + w*128;
395 float eres = 0.0f;
396 const FFPsyBand *pb = &s->psy.ch[s->cur_channel].psy_bands[w*16+g];
397 for (int k = s0; k < s1; k++)
398 eres += sce->coeffs[k]*sce->coeffs[k];
399 t->tnsg[w*16+g] = av_clipf(pb->energy / FFMAX(eres, 1e-12f), 1.0f, 64.0f);
400 }
401 }
402 }
403 }
404
405 /* finest codeable scalefactor and max value per band */
406 for (int w = 0; w < sce->ics.num_windows; w += sce->ics.group_len[w]) {
407 int start = w*128;
408 for (int g = 0; g < sce->ics.num_swb; g++) {
409 t->maxvals[w*16+g] = find_max_val(sce->ics.group_len[w], sce->ics.swb_sizes[g], s->scoefs + start);
410 t->minsf[w*16+g] = t->maxvals[w*16+g] > 0 ? coef2minsf(t->maxvals[w*16+g]) : 0;
411 start += sce->ics.swb_sizes[g];
412 }
413 }
414
415 /* PASS 1: coarse candidate curves per coded band
416 * (the lambda search runs on this cheap grid, PASS 2 refines the winner) */
417 {
418 for (int w = 0; w < sce->ics.num_windows; w += sce->ics.group_len[w]) {
419 int start = w*128;
420 for (int g = 0; g < sce->ics.num_swb; g++) {
421 if (!sce->zeroes[w*16+g] && t->maxvals[w*16+g] > 0 && nbnd < 128) {
422 int lo = av_clip(t->minsf[w*16+g], 0, SCALE_MAX_POS);
423 float invthr = 1.0f / FFMAX(t->thr[w*16+g], 1e-9f);
424 int ncand = nmr_band_curve(s, sce, w, g, start, lo, cstep, NMR_NCAND,
425 invthr, t->maxvals[w*16+g], nd[nbnd], nb[nbnd]);
426 if (t->tnsg[w*16+g] > 1.0f)
427 for (int o = 0; o < ncand; o++)
428 nd[nbnd][o] *= t->tnsg[w*16+g];
429 if (ncand == 0) {
430 /* nothing codeable: drop the group band incl. subwindow
431 * flags (group flag is re-derived by ANDing) */
432 for (int w2 = 0; w2 < sce->ics.group_len[w]; w2++)
433 sce->zeroes[(w+w2)*16+g] = 1;
434 } else {
435 t->bidx[nbnd] = w*16+g;
436 t->bw[nbnd] = w;
437 t->bg[nbnd] = g;
438 t->bst[nbnd] = start;
439 t->blo[nbnd] = lo;
440 t->bnc[nbnd] = ncand;
441 nbnd++;
442 }
443 }
444 start += sce->ics.swb_sizes[g];
445 }
446 }
447 }
448 t->nbnd = nbnd;
449 for (int b = 0; b < nbnd; b++) {
450 t->act[b] = b;
451 t->is_pns[b] = 0;
452 }
453 t->nact = nbnd;
454 return nbnd;
456
457/* total bits of a slot's current chosen[] on grid `step`, incl. sf deltas */
458static int nmr_slot_bits(const NMRSlot *t, const int (*nb)[NMR_NCAND], int step)
459{
460 int tot = 0;
461 for (int k = 0; k < t->nact; k++)
462 tot += nb[t->act[k]][t->chosen[t->act[k]]];
463 for (int k = 1; k < t->nact; k++)
464 tot += NMR_SFBITS((t->blo[t->act[k]]+t->chosen[t->act[k]]*step) -
465 (t->blo[t->act[k-1]]+t->chosen[t->act[k-1]]*step));
466 return tot;
468
469/* Run every slot's trellis at one fixed lambda; returns the pooled bits. */
470static int nmr_eval_slots(AACEncContext *s, NMRSlot *const *sl, int nsl, int step, float lam)
471{
472 int total = 0;
473 for (int k = 0; k < nsl; k++) {
474 NMRSlot *t = sl[k];
475 if (!t->nact)
476 continue;
477 nmr_solve(s, s->nmr->nd[t->si], s->nmr->nb[t->si], t->blo, t->bnc, step,
478 t->act, t->nact, 0, t->chosen, lam, lam, 1);
479 total += nmr_slot_bits(t, s->nmr->nb[t->si], step);
480 }
481 return total;
482}
483
484/* Bisect ONE shared lambda across the slots so the POOLED bits meet destbits.
485 * This is the CPE budget pool: bits flow to whichever channel of the pair has
486 * demand at the common operating point, instead of an equal per-channel split. */
487static float nmr_solve_slots(AACEncContext *s, NMRSlot *const *sl, int nsl, int step,
488 int destbits, float lo_l, float hi_l, int iters)
489{
490 float lam = 1.0f;
491 for (int it = 0; it < iters; it++) {
492 lam = sqrtf(lo_l * hi_l);
493 int total = nmr_eval_slots(s, sl, nsl, step, lam);
494 if (it == iters - 1)
495 break;
496 /* over budget -> go coarser */
497 if (total > destbits)
498 lo_l = lam;
499 else
500 hi_l = lam;
501 }
502 return lam;
503}
505/* Write a solved slot back into its channel: band types, scalefactors, and the
506 * SCALE_MAX_DIFF legality fixups. Verbatim from the pre-pool single-channel tail. */
508{
509 SingleChannelElement *sce = t->sce;
510 const int (*nb)[NMR_NCAND] = (const int (*)[NMR_NCAND])s->nmr->nb[t->si];
511
512 for (int b = 0; b < t->nbnd; b++) {
513 int bi = t->bidx[b];
514 if (t->is_pns[b]) {
515 sce->band_type[bi] = NOISE_BT;
516 sce->zeroes[bi] = 0;
517 sce->pns_ener[bi] = t->pener[bi] * FFMIN(1.0f, t->pspread[bi]*t->pspread[bi]);
518 } else {
519 sce->sf_idx[bi] = av_clip(t->blo[b] + t->chosen[b]*NMR_STEP, 0, SCALE_MAX_POS);
520 }
521 }
522
523
524 { /* record the bits this solve accounted for; the encoder compares them
525 * against the channel's real output to keep the budget honest */
526 int tot = 0, prevb = -1;
527 for (int b = 0; b < t->nbnd; b++) {
528 if (t->is_pns[b])
529 continue;
530 tot += nb[b][t->chosen[b]];
531 if (prevb >= 0)
532 tot += NMR_SFBITS((t->blo[b]+t->chosen[b]*NMR_STEP) - (t->blo[prevb]+t->chosen[prevb]*NMR_STEP));
533 prevb = b;
534 }
535 s->nmr->counted[t->cur_ch] = tot;
536 }
537
538 /* SCALE_MAX_DIFF condition:
539 * re-clamp, codebook fixup, drop uncodeable, set global gain
540 * NOISE_BT bands keep their own scalefactor chain via set_special_band_scalefactors) */
541 {
542 uint8_t nextband[128];
543 int prev = -1;
544 ff_init_nextband_map(sce, nextband);
545 for (int w = 0; w < sce->ics.num_windows; w += sce->ics.group_len[w]) {
546 for (int g = 0; g < sce->ics.num_swb; g++) {
547 if (sce->band_type[w*16+g] == NOISE_BT ||
548 sce->band_type[w*16+g] == INTENSITY_BT ||
549 sce->band_type[w*16+g] == INTENSITY_BT2)
550 continue;
551 if (sce->zeroes[w*16+g]) {
552 sce->band_type[w*16+g] = 0;
553 continue;
554 }
555
556 if (prev != -1)
557 sce->sf_idx[w*16+g] = av_clip(sce->sf_idx[w*16+g], prev - SCALE_MAX_DIFF, prev + SCALE_MAX_DIFF);
558 sce->band_type[w*16+g] = find_min_book(t->maxvals[w*16+g], sce->sf_idx[w*16+g]);
559 if (sce->band_type[w*16+g] <= 0) {
560 if (!ff_sfdelta_can_remove_band(sce, nextband, prev, w*16+g)) {
561 sce->band_type[w*16+g] = 1;
562 } else {
563 /* drop subwindow flags too, see the PASS 1 drop above */
564 for (int w2 = 0; w2 < sce->ics.group_len[w]; w2++)
565 sce->zeroes[(w+w2)*16+g] = 1;
566 sce->band_type[w*16+g] = 0;
567 continue;
568 }
569 }
570 if (prev == -1)
571 sce->sf_idx[0] = sce->sf_idx[w*16+g]; /* global gain */
572 prev = sce->sf_idx[w*16+g];
573 }
574 }
575
576 /* every band must carry a chain-legal scalefactor (re-clamp, codebook
577 * fixup, global gain) */
578 if (prev != -1) {
579 int last = sce->sf_idx[0];
580 for (int w = 0; w < sce->ics.num_windows; w += sce->ics.group_len[w]) {
581 for (int g = 0; g < sce->ics.num_swb; g++) {
582 if (!sce->zeroes[w*16+g] && sce->band_type[w*16+g] != NOISE_BT &&
583 sce->band_type[w*16+g] < RESERVED_BT)
584 last = sce->sf_idx[w*16+g];
585 else if (sce->band_type[w*16+g] < RESERVED_BT && (w*16+g) > 0)
586 sce->sf_idx[w*16+g] = last;
587 }
588 }
589 }
590 }
591}
593/* Solve one element group (a solo channel, or a CPE pair pooled under one
594 * shared lambda and one pooled budget), then PNS and commit. */
595static void nmr_solve_group(AVCodecContext *avctx, AACEncContext *s,
596 const float lambda, NMRSlot *const *sl, int nsl,
597 int chans, int rc_eligible, int rc_global,
598 int rc_rate_frame, int rc_bmax)
599{
600 const int cstep = NMR_COARSE > 0 ? NMR_COARSE : NMR_STEP;
601 int bch = ((avctx->flags & AV_CODEC_FLAG_QSCALE) ? 2.0f : avctx->ch_layout.nb_channels);
602 int destbits = avctx->bit_rate * 1024.0 / avctx->sample_rate / bch * (lambda / 120.f) * chans;
603 int is8_any = 0;
604 float lam;
605 float rc_off = 1.0f, lam_dem = 0.0f;
606
607 for (int k = 0; k < nsl; k++)
608 is8_any |= sl[k]->is8;
609
610 if (s->psy.bitres.alloc >= 0)
611 destbits = s->psy.bitres.alloc *
612 (lambda / (avctx->global_quality ? avctx->global_quality : 120)) * chans;
613 if (rc_global && s->psy.bitres.alloc >= 0) {
614 /* CBR target: nominal + repayment, bounded +-30%/frame */
615 double rr = avctx->bit_rate * 1024.0 / avctx->sample_rate;
616 destbits = (rr + av_clipd(s->nmr->rc_fill / 2.0, -0.3 * rr, 0.3 * rr)) * chans / s->channels;
617 } else if (rc_eligible && s->psy.bitres.alloc >= 0) {
618 /* pre-bootstrap CBR frames: target nominal (psy bitres is cold) */
619 destbits = (avctx->bit_rate * 1024.0 / avctx->sample_rate) * chans / s->channels;
620 }
621 destbits = FFMIN(destbits, 5800 * chans);
622 /* honest budget: subtract the measured non-trellis overhead (section data, ICS,
623 * sf/PNS signalling), which is rate-dependent hence adaptive. */
624 if (s->nmr->side_inited)
625 destbits = av_clip(destbits - (int)(s->nmr->side_ema * chans / s->channels), 64, 5800 * chans);
626
627 /* Held transient burst, bank-aware: spend banked bits, never borrow deep
628 * (payback troughs starve the next transient). */
629 if (s->nmr->run_burst > 1.0f) {
630 int extra = destbits * (s->nmr->run_burst - 1.0f);
631 int avail = FFMAX(0, (int)((s->nmr->rc_fill + rc_bmax / 2) * (int64_t)chans / s->channels));
632 destbits = av_clip(destbits + FFMIN(extra, avail), 64, 6800 * chans);
633 }
634
635 if (rc_global) {
636 /* corridor bisect around the servoed centre; pressure = stateless
637 * rc_off multiplier (folding it into lam_rc winds up) */
638 float R = avctx->bit_rate * 1024.0 / avctx->sample_rate;
639 float cen;
640 int tot, hardcap, rc_cap;
641 float lo;
642 rc_off = exp2f(-NMR_RC_K_CBR * s->nmr->rc_fill / R);
643 cen = s->nmr->lam_rc * rc_off;
644 lo = cen / NMR_RC_CORR;
645 /* transient burst: widen the lower bound so the boosted destbits can
646 * actually pour into the onset frame */
647 if (is8_any && s->nmr->run_burst > 1.0f)
648 lo /= s->nmr->run_burst;
649 lam = nmr_solve_slots(s, sl, nsl, cstep, destbits,
650 lo, cen * NMR_RC_CORR, NMR_RC_CITERS);
651
652 tot = 0;
653 for (int k = 0; k < nsl; k++)
654 tot += nmr_slot_bits(sl[k], s->nmr->nb[sl[k]->si], cstep);
655 hardcap = av_clip((int)(5800.f * FFMIN(1.f, lambda / 120.f)), 256, 5800) * chans;
656 /* legality cap only; no spend-floor (rc_off spends the bank) */
657 rc_cap = FFMIN(hardcap, (s->nmr->rc_fill + rc_rate_frame + rc_bmax) * chans / s->channels);
658 if (tot > rc_cap) {
659 lam = nmr_solve_slots(s, sl, nsl, cstep, rc_cap, lam, 1e4f, NMR_CITERS);
660 }
661 } else {
662 /* per-frame bisection, warm-started off the previous frame's lambda;
663 * a result at the bracket edge means redo the full search */
664 float lam0 = s->nmr->lam[sl[0]->cur_ch];
665 lam = 1.0f;
666 if (NMR_COARSE > 0 && lam0 > 0.0f) {
667 lam = nmr_solve_slots(s, sl, nsl, cstep, destbits, lam0/32.0f, lam0*32.0f, NMR_CWARM);
668 if (lam < lam0/16.0f || lam > lam0*16.0f)
669 lam0 = 0.0f;
670 }
671 if (lam0 <= 0.0f)
672 lam = nmr_solve_slots(s, sl, nsl, cstep, destbits,
673 1e-9f, 1e4f, NMR_COARSE > 0 ? NMR_CITERS : NMR_ITERS);
674 }
675
676 /* PASS 2:
677 * refine each band at full granularity (NMR_STEP) in a +/-cstep window
678 * around the coarse pick, then re-solve. Recovers single-pass quality while the
679 * lambda search stayed cheap on the coarse grid. */
680 if (NMR_COARSE > 0) {
681 /* nmr_speed, 0 = slowest/best, higher = faster; see the option docs. */
682 int win = NMR_COARSE - av_clip(s->options.nmr_speed, 0, 4);
683 for (int k = 0; k < nsl; k++) {
684 NMRSlot *t = sl[k];
685 float (*ndk)[NMR_NCAND] = s->nmr->nd[t->si];
686 int (*nbk)[NMR_NCAND] = s->nmr->nb[t->si];
687 if (!t->nact)
688 continue;
689 /* the pow34 spectrum and the quantize cache are per-channel state */
690 s->aacdsp.abs_pow34(s->scoefs, t->sce->coeffs, 1024);
692 for (int b = 0; b < t->nbnd; b++) {
693 int center = t->blo[b] + t->chosen[b]*cstep;
694 int flo = av_clip(center - win, av_clip(t->minsf[t->bidx[b]], 0, SCALE_MAX_POS), SCALE_MAX_POS);
695 int maxn = FFMIN(NMR_NCAND, 2*win/NMR_STEP + 1);
696 float invthr = 1.0f / FFMAX(t->thr[t->bidx[b]], 1e-9f);
697 int ncand = nmr_band_curve(s, t->sce, t->bw[b], t->bg[b], t->bst[b], flo, NMR_STEP, maxn,
698 invthr, t->maxvals[t->bidx[b]], ndk[b], nbk[b]);
699 if (t->tnsg[t->bidx[b]] > 1.0f)
700 for (int o = 0; o < ncand; o++)
701 ndk[b][o] *= t->tnsg[t->bidx[b]];
702 t->blo[b] = flo;
703 t->bnc[b] = FFMAX(1, ncand);
704 }
705 }
706 /* fine pass: narrow corridor around the coarse solve */
707 if (rc_global)
708 lam = nmr_solve_slots(s, sl, nsl, NMR_STEP, destbits, lam/2.0f, lam*2.0f, NMR_RC_FITERS);
709 else
710 lam = nmr_solve_slots(s, sl, nsl, NMR_STEP, destbits, lam/16.0f, lam*16.0f, NMR_IFINE);
711 }
712
713 lam_dem = lam; /* demand-solved lambda, pre bucket clamp: what content wants */
714
715 if (rc_global) {
716 /* legality clamp, then the quality slew limiter */
717 int hardcap = av_clip((int)(5800.f * FFMIN(1.f, lambda / 120.f)), 256, 5800) * chans;
718 int tot = 0, rc_cap;
719 for (int k = 0; k < nsl; k++)
720 tot += nmr_slot_bits(sl[k], s->nmr->nb[sl[k]->si], NMR_STEP);
721 rc_cap = FFMIN(hardcap, (s->nmr->rc_fill + rc_rate_frame + rc_bmax) * chans / s->channels);
722 if (tot > rc_cap) {
723 lam = nmr_solve_slots(s, sl, nsl, NMR_STEP, rc_cap, lam, 1e4f, NMR_RC_ITERS);
724 }
725 if (s->nmr->lam_slew > 0.0f) {
726 float kup, kdn;
727 /* hold lambda near-constant within short runs; bits follow content */
728 kup = (is8_any && s->nmr->prev_was_short) ? NMR_SLEW_RUN : NMR_SLEW;
729 /* a deliberate onset burst may dive as far as its widened corridor
730 * allows; the RECOVERY back up is what must stay gradual */
731 kdn = (is8_any && s->nmr->run_burst > 1.0f) ? NMR_SLEW * s->nmr->run_burst :
732 (is8_any && s->nmr->prev_was_short) ? NMR_SLEW_RUN : NMR_SLEW;
733 if (lam > s->nmr->lam_slew * kup || lam < s->nmr->lam_slew / kdn) {
734 lam = av_clipf(lam, s->nmr->lam_slew / kdn, s->nmr->lam_slew * kup);
735 tot = nmr_eval_slots(s, sl, nsl, NMR_STEP, lam);
736 /* never at the price of an illegal reservoir excursion */
737 if (tot > rc_cap) {
738 lam = nmr_solve_slots(s, sl, nsl, NMR_STEP, rc_cap, lam, 1e4f, NMR_RC_ITERS);
739 }
740 }
741 }
742 s->nmr->lam_slew = lam;
743 }
744
745 for (int k = 0; k < nsl; k++)
746 s->nmr->lam[sl[k]->cur_ch] = lam; /* warm start for the next frame */
747 { /* nd: mean achieved dist/real-mask (dimensionless starvation +
748 * noise-class signal) */
749 float ndsum = 0.0f; int ndn = 0;
750 for (int k = 0; k < nsl; k++) {
751 NMRSlot *t = sl[k];
752 float (*ndk)[NMR_NCAND] = s->nmr->nd[t->si];
753 for (int b_ = 0; b_ < t->nact; b_++) {
754 int b = t->act[b_], bi = t->bidx[b];
755 if (t->thr_real[bi] > 0.0f && t->thr[bi] > 0.0f) {
756 ndsum += ndk[b][t->chosen[b]] * t->thr[bi] / t->thr_real[bi];
757 ndn++;
758 }
759 }
760 }
761 /* long frames only (short groups inflate the ratio) */
762 if (ndn >= 8 && !is8_any) {
763 float nd = ndsum / ndn;
764 s->nmr->nd_ema = s->nmr->nd_ema > 0.0f ?
765 0.95f * s->nmr->nd_ema + 0.05f * nd : nd;
766 }
767 }
768 { /* track short vs long operating lambda (dense-beat boost scaling) */
769 float *ema = is8_any ? &s->nmr->lam_short_ema : &s->nmr->lam_long_ema;
770 *ema = *ema > 0.0f ? 0.9f * *ema + 0.1f * lam : lam;
771 /* sustained-strain floor: snaps down at any comfortable moment,
772 * recovers only slowly, so bursty content cannot bank pressure
773 * credit between its lambda valleys. */
774 s->nmr->lam_floor = s->nmr->lam_floor > 0.0f ?
775 fminf(s->nmr->lam_floor * 1.02f, lam) : lam;
776 }
777 { /* shared rate-pressure ramp: lambda vs nd-scaled anchors */
778 float scale, ramp;
779 scale = 1.0f + av_clipf(s->nmr->nd_ema / 50.0f, 0.0f, 8.0f);
780 ramp = s->nmr->lam_long_ema > 0.0f ?
781 av_clipf((s->nmr->lam_long_ema - 120.0f * scale) /
782 (350.0f * scale - 120.0f * scale), 0.0f, 1.0f) : 0.0f;
783 /* transparency veto: lambda*nd below ~74 = comfortable */
784 if (s->nmr->nd_ema > 0.0f)
785 ramp *= av_clipf((s->nmr->lam_long_ema * s->nmr->nd_ema - 60.0f) /
786 (120.0f - 60.0f), 0.0f, 1.0f);
787 s->nmr->press = ramp;
788 }
789 if (rc_global) {
790 /* track the centre toward the CONTENT lambda (demand-solved, pressure
791 * divided out); clamped lambda is rate noise, not content */
792 float c = s->nmr->lam_rc * powf(lam_dem / rc_off / s->nmr->lam_rc, NMR_RC_TRACK);
793 s->nmr->lam_rc = av_clipf(c, 1e-6f, 1e4f);
794 } else if (rc_eligible) {
795 /* bootstrap the servo off the first substantive frame (silent lead-ins
796 * have degenerate budgets) */
797 int nbnd_max = 0;
798 for (int k = 0; k < nsl; k++)
799 nbnd_max = FFMAX(nbnd_max, sl[k]->nbnd);
800 if (nbnd_max >= 8) {
801 s->nmr->lam_rc = av_clipf(lam, 1e-4f, 1e4f);
802 s->nmr->lam_slew = s->nmr->lam_rc;
803 }
804 }
805
806 { /* PNS, per channel at the group's operating lambda */
807 const float pns_lam = NMR_PNS_LAM;
808 int pns_total = 0;
809 for (int k = 0; k < nsl; k++) {
810 NMRSlot *t = sl[k];
811 const float (*ndk)[NMR_NCAND] = (const float (*)[NMR_NCAND])s->nmr->nd[t->si];
812 const int (*nbk)[NMR_NCAND] = (const int (*)[NMR_NCAND])s->nmr->nb[t->si];
813 int pns_count = 0;
814 /* band 0 (lowest freq) is kept as the global-gain / sf-chain anchor */
815 for (int b = 1; b < t->nbnd; b++) {
816 int bi = t->bidx[b];
817 float spread = t->pspread[bi];
818 float nmr_pns, cost_keep, cost_pns, frac;
819 if (!t->sce->can_pns[bi])
820 continue;
821
822 int was = s->nmr->pns_prev[t->cur_ch & 15][bi];
823 float bias = was ? NMR_PNS_STAY : NMR_PNS_ENTER;
824 int want = 0, force_exit = 0;
825
826 /* (can_pns was already checked above; gates below fill `want`) */
827 if (t->pener[bi] > NMR_PNS_MAX_ET * t->thr_real[bi]) {
828 force_exit = 1; /* loud-band guard */
829 } else if (lam > pns_lam) {
830 /* Spectral-hole fill: a noise-like band left mostly empty */
831 frac = ndk[b][t->chosen[b]] * t->thr[bi] / FFMAX(t->pener[bi], 1e-9f);
832 if (spread > NMR_PNS_HOLE_SPREAD &&
833 frac > NMR_PNS_HOLE_FRAC * (was ? 0.7f : 1.0f)) {
834 want = 1;
835 } else if (ndk[b][t->chosen[b]] * t->thr[bi] >
836 NMR_PNS_NDGATE * t->thr_real[bi] * (was ? 0.5f : 1.0f)) {
837 /* replace only a band coded audibly badly; cost of
838 * energy-matched noise = its non-noise-like fraction */
839 nmr_pns = FFMAX(0.0f, t->pener[bi] * (1.0f - spread*spread))
840 / FFMAX(t->thr[bi], 1e-9f);
841 cost_keep = ndk[b][t->chosen[b]] + lam * nbk[b][t->chosen[b]];
842 cost_pns = nmr_pns + lam * NMR_PNS_BITS;
843 want = cost_pns < cost_keep * bias;
844 }
845 }
846 { /* debounce; near-mask deletion candidates skip entry
847 * (noise beats the ~silent rendition they'd get) */
848 uint8_t *ron = &s->nmr->pns_run_on [t->cur_ch & 15][bi];
849 uint8_t *roff = &s->nmr->pns_run_off[t->cur_ch & 15][bi];
850 int near = t->pener[bi] < 2.0f * t->thr_real[bi];
851 if (want) { if (*ron < 255) (*ron)++; *roff = 0; }
852 else { if (*roff < 255) (*roff)++; *ron = 0; }
853 if (force_exit)
854 want = 0;
855 else if (!was)
856 want = near ? want : *ron >= NMR_PNS_ON;
857 else if (near)
858 want = 1; /* physics-hysteresis: noise until audible */
859 else
860 want = !(*roff >= NMR_PNS_OFF);
861 }
862 if (want) {
863 t->is_pns[b] = 1;
864 pns_count++;
865 }
866 }
867 if (pns_count) {
868 t->nact = 0;
869 for (int b = 0; b < t->nbnd; b++)
870 if (!t->is_pns[b])
871 t->act[t->nact++] = b;
872 }
873 pns_total += pns_count;
874 }
875 if (pns_total) {
876 /* re-solve over the survivors: at fixed lambda the allocation is
877 * the same except for the repaired sf-delta chain; in bisection
878 * mode re-spend the freed budget */
879 if (rc_global)
880 nmr_eval_slots(s, sl, nsl, NMR_STEP, lam);
881 else
882 nmr_solve_slots(s, sl, nsl, NMR_STEP, destbits - pns_total * NMR_PNS_BITS,
883 1e-9f, 1e4f, NMR_ITERS);
884 }
885 }
886
887 for (int k = 0; k < nsl; k++) {
888 NMRSlot *t = sl[k];
889 uint8_t *pp = s->nmr->pns_prev[t->cur_ch & 15];
890 uint8_t now[128] = {0};
891 for (int b = 0; b < t->nbnd; b++)
892 if (t->is_pns[b])
893 now[t->bidx[b]] = 1;
894 memcpy(pp, now, 128);
895 }
896 for (int k = 0; k < nsl; k++)
897 nmr_commit_channel(s, sl[k]);
899}
900
904 const float lambda)
905{
906 AACNMRCurves *n = s->nmr;
907 /* Global-lambda RC: one solve per frame at a servoed centre lambda; the reservoir
908 * holds the long-run mean rate. Bypassed for VBR (-q:a) and the bootstrap frame. */
909 int rc_eligible = !(avctx->flags & AV_CODEC_FLAG_QSCALE) && avctx->bit_rate > 0 &&
910 avctx->bit_rate_tolerance != 0;
911 /* Signed reservoir; soft steering (bounded repay + rc_off), hard cap =
912 * legality only. */
913 int rc_rate_frame = avctx->bit_rate * 1024.0 / avctx->sample_rate;
914 int rc_bmax = FFMIN(FFMAX(6144 * s->channels - rc_rate_frame, 256), NMR_CBR_BUF * s->channels);
915
916 int rc_global, defer;
917 NMRSlot *t;
918
919 s->nmr->counted[s->cur_channel] = 0;
920
921 if (rc_eligible && !n->rc_fill_seeded) {
922 /* the decoder bit reservoir starts FULL: seed it so the head may frontload */
923 n->rc_fill = rc_bmax;
924 n->rc_fill_seeded = 1;
925 }
926 if (rc_eligible && avctx->frame_num != n->rc_frame_num) {
927 if (n->rc_frame_num > 0 && n->lam_rc > 0.0f)
928 n->rc_fill = av_clip(n->rc_fill + rc_rate_frame - s->last_frame_pb_count,
929 -rc_bmax, rc_bmax);
930 n->rc_frame_num = avctx->frame_num;
931 n->pending = 0; /* a deferred first channel never crosses a frame */
932 /* latch the RC mode per frame: a mid-frame bootstrap must not flip
933 * the CPE defer logic between channels */
934 n->rc_gl = rc_eligible && n->lam_rc > 0.0f;
935
936 /* Transient burst run state: set at run start and held across the run so
937 * coding stays uniform; repaid from the reservoir's steady stretches. */
938 int is_short = sce->ics.window_sequence[0] == EIGHT_SHORT_SEQUENCE;
939 if (is_short) {
940 if (!n->prev_was_short) { /* run start */
943 } else {
944 /* dense-beat boost, scaled by measured short-frame starvation */
945 float imb = 0.0f;
946 if (n->lam_long_ema > 0.0f && n->lam_short_ema > 0.0f)
947 imb = av_clipf(n->lam_short_ema / n->lam_long_ema - 1.0f,
948 0.0f, 1.0f);
949 n->run_burst = 1.0f + (NMR_SHORT_BOOST - 1.0f) * imb *
951 }
952 }
953 n->frames_since_short = 0;
954 } else {
955 /* the frame closing a run (the STOP) absorbs the corridor recoil
956 * of the boosted shorts; give it half the run's factor so the
957 * repayment spreads into the steady stretch instead */
958 n->run_burst = n->prev_was_short ? sqrtf(n->run_burst) : 1.0f;
960 }
961 n->prev_was_short = is_short;
962 }
963 rc_global = rc_eligible && n->rc_gl;
964
965 /* CPE budget pool: under global-lambda RC, defer the pair's first channel
966 * and solve both against one pooled budget when the second one arrives. */
967 defer = n->pair && rc_global;
968
969 t = &n->slot[(defer && n->pending) ? 1 : 0];
970 t->si = (defer && n->pending) ? 1 : 0;
971
972 if (!nmr_setup_channel(avctx, s, sce, t)) {
973 nmr_bail_channel(sce);
974 t->nbnd = t->nact = 0;
975 }
976
977 if (defer && !n->pending) {
978 n->pending = 1; /* wait for the partner channel */
979 return;
980 }
981
982 {
983 NMRSlot *sl[2];
984 int nsl = 0, chans = 1;
985 if (defer) {
986 n->pending = 0;
987 chans = 2;
988 if (n->slot[0].nact)
989 sl[nsl++] = &n->slot[0];
990 if (n->slot[1].nact)
991 sl[nsl++] = &n->slot[1];
992 } else if (t->nact) {
993 sl[nsl++] = t;
994 }
995 if (!nsl)
996 return; /* nothing codeable in the group */
997 nmr_solve_group(avctx, s, lambda, sl, nsl, chans,
998 rc_eligible, rc_global, rc_rate_frame, rc_bmax);
999 }
1000}
1001
1002#endif /* AVCODEC_AACCODER_NMR_H */
AAC definitions and structures.
#define SCALE_MAX_DIFF
maximum scalefactor difference allowed by standard
Definition aac.h:94
@ EIGHT_SHORT_SEQUENCE
Definition aac.h:66
@ LONG_START_SEQUENCE
Definition aac.h:65
@ INTENSITY_BT
Scalefactor data are intensity stereo positions (in phase).
Definition aac.h:77
@ INTENSITY_BT2
Scalefactor data are intensity stereo positions (out of phase).
Definition aac.h:76
@ RESERVED_BT
Band types following are encoded differently from others.
Definition aac.h:74
@ NOISE_BT
Spectral data are scaled white noise not coded in the bitstream.
Definition aac.h:75
#define SCALE_MAX_POS
scalefactor index maximum value
Definition aac.h:93
static void search_for_quantizers_nmr(AVCodecContext *avctx, AACEncContext *s, SingleChannelElement *sce, const float lambda)
#define NMR_RC_CITERS
#define NMR_CBR_BUF
#define NMR_ITERS
#define NMR_SLEW
static float nmr_solve_slots(AACEncContext *s, NMRSlot *const *sl, int nsl, int step, int destbits, float lo_l, float hi_l, int iters)
static int nmr_setup_channel(AVCodecContext *avctx, AACEncContext *s, SingleChannelElement *sce, NMRSlot *t)
static void nmr_solve_group(AVCodecContext *avctx, AACEncContext *s, const float lambda, NMRSlot *const *sl, int nsl, int chans, int rc_eligible, int rc_global, int rc_rate_frame, int rc_bmax)
#define NMR_PNS_HOLE_SPREAD
#define NMR_RC_TRACK
#define NMR_PNS_STAY
#define NMR_PNS_HOLE_FRAC
#define NMR_PNS_ON
#define NMR_RC_ITERS
#define NMR_PNS_BITS
#define NMR_COARSE
#define NMR_PNS_ENTER
#define NMR_TRANS_PM
static int nmr_slot_bits(const NMRSlot *t, const int(*nb)[NMR_NCAND], int step)
#define NMR_BURST_GAIN
#define NMR_SHORT_BOOST
static int nmr_eval_slots(AACEncContext *s, NMRSlot *const *sl, int nsl, int step, float lam)
#define NMR_SLEW_RUN
static int nmr_band_curve(AACEncContext *s, SingleChannelElement *sce, int w, int g, int start, int lo, int step, int maxn, float invthr, float maxval, float *nd_row, int *nb_row)
#define NMR_SFBITS(d)
AAC encoder NMR scalefactor coder.
static float nmr_solve(AACEncContext *s, const float(*nd)[NMR_NCAND], const int(*nb)[NMR_NCAND], const int *blo, const int *bnc, int step, const int *act, int nact, int destbits, int *chosen, float lo_l, float hi_l, int iters)
Viterbi over the coding sequence act[0..nact-1] (indices into the per-band curves nd/nb),...
#define NMR_CWARM
#define NMR_RC_K_CBR
#define NMR_STEP
#define NMR_PNS_OFF
#define NMR_IFINE
#define NMR_CITERS
#define NMR_RC_FITERS
#define NMR_BURST_GAP
#define NMR_PNS_LAM
#define NMR_ZERO_STICKY
#define NMR_PNS_MAX_ET
static void nmr_commit_channel(AACEncContext *s, NMRSlot *t)
static void nmr_bail_channel(SingleChannelElement *sce)
#define NMR_PNS_NDGATE
#define NMR_RC_CORR
void ff_quantize_band_cost_cache_init(struct AACEncContext *s)
Definition aacenc.c:373
#define NMR_NCAND
per-band scalefactor candidates above the finest codeable sf (NMR coder)
Definition aacenc.h:171
static float quantize_band_cost_cached(struct AACEncContext *s, int w, int g, const float *in, const float *scaled, int size, int scale_idx, int cb, const float lambda, const float uplim, int *bits, float *energy, int rtz)
static void ff_init_nextband_map(const SingleChannelElement *sce, uint8_t *nextband)
static int find_min_book(float maxval, int sf)
static float find_max_val(int group_len, int swb_size, const float *scaled)
static uint8_t coef2minsf(float coef)
Return the minimum scalefactor where the quantized coef does not clip.
static int ff_sfdelta_can_remove_band(const SingleChannelElement *sce, const uint8_t *nextband, int prev_sf, int band)
AAC encoder data.
const uint8_t ff_aac_scalefactor_bits[121]
Definition aactab.c:200
AAC data declarations.
static const int8_t filt[NUMTAPS *2]
Definition af_earwax.c:40
static float win(SuperEqualizerContext *s, float n, int N)
Libavcodec external API header.
#define i(width, name, range_min, range_max)
Definition cbs_h264.c:63
#define mb(name)
Definition cbs_lcevc.c:95
#define s(width, name)
Definition cbs_vp9.c:198
#define av_clipd
Definition common.h:148
#define av_clip
Definition common.h:100
#define av_clipf
Definition common.h:145
#define NULL
Definition coverity.c:32
long long int64_t
Definition coverity.c:34
static __device__ float sqrtf(float a)
float fminf(float, float)
#define AV_CODEC_FLAG_QSCALE
Use fixed qscale.
Definition avcodec.h:213
#define R
Definition huffyuv.h:44
#define b
Definition input.c:43
static void scale(int *out, const int *in, const int w, const int h, const int shift)
Definition intra.c:278
#define expf(x)
Definition libm.h:285
#define exp2f(x)
Definition libm.h:295
#define powf(x, y)
Definition libm.h:52
uint8_t w
Definition llvidencdsp.c:39
#define FFMIN(a, b)
Definition macros.h:49
#define FFMAX(a, b)
Definition macros.h:47
#define M_LN10
Definition mathematics.h:49
#define INFINITY
bitstream writer API
AAC encoder context.
Definition aacenc.h:258
int frames_since_short
long-block frames since the last short run (the "gap"): large = isolated transient
Definition aacenc.h:235
NMRSlot slot[2]
pair slots (solo solves use slot 0)
Definition aacenc.h:207
float run_burst
transient bit-burst factor, set at run start and held across the short run
Definition aacenc.h:237
int rc_fill
virtual bit reservoir fill, + = bits saved vs nominal
Definition aacenc.h:234
int rc_gl
rc_global latched at frame start: the corridor bootstrap must not flip the CPE defer logic between ch...
Definition aacenc.h:209
float lam_short_ema
smoothed operating lambda of short frames
Definition aacenc.h:241
int64_t rc_frame_num
frame the reservoir was last advanced for
Definition aacenc.h:232
float lam_long_ema
smoothed operating lambda of long frames
Definition aacenc.h:242
int pending
slot 0 holds a deferred first channel
Definition aacenc.h:211
int prev_was_short
previous frame was a short block (for run-start detection)
Definition aacenc.h:236
float lam_rc
global-lambda rate control: operating lambda, 0 until bootstrapped
Definition aacenc.h:233
int rc_fill_seeded
reservoir seeded full at stream start (decoder buffer starts full)
Definition aacenc.h:210
int pair
current element is a CPE: pool the pair budget
Definition aacenc.h:208
int nb_channels
Number of channels in this layout.
main external API structure.
Definition avcodec.h:443
AVChannelLayout ch_layout
Audio channel layout.
Definition avcodec.h:1055
int global_quality
Global quality for codecs which cannot change it per frame.
Definition avcodec.h:1235
int64_t frame_num
Frame counter, set by libavcodec.
Definition avcodec.h:1883
int bit_rate_tolerance
number of bits the bitstream is allowed to diverge from the reference.
Definition avcodec.h:1227
int64_t bit_rate
the average bitrate
Definition avcodec.h:493
int sample_rate
samples per second
Definition avcodec.h:1040
int flags
AV_CODEC_FLAG_*.
Definition avcodec.h:500
single band psychoacoustic information
Definition psymodel.h:50
float spread
Definition psymodel.h:54
float threshold
Definition psymodel.h:53
float energy
Definition psymodel.h:52
uint8_t max_sfb
number of scalefactor bands per group
Definition aacdec.h:170
int num_swb
number of scalefactor window bands
Definition aacdec.h:178
uint8_t group_len[8]
Definition aacdec.h:175
const uint8_t * swb_sizes
table of scalefactor band sizes for a particular window
Definition aacenc.h:85
enum WindowSequence window_sequence[2]
Definition aacdec.h:171
const uint16_t * swb_offset
table of offsets to the lowest spectral coefficient of a scalefactor band, sfb, for a particular wind...
Definition aacdec.h:177
NMR coder per-band candidate cost curves (~96 KiB) and rate-control carry-over.
Definition aacenc.h:181
int chosen[128]
Definition aacenc.h:191
float pspread[128]
band tonality spread (1 = noise)
Definition aacenc.h:200
int nact
Definition aacenc.h:193
int minsf[128]
Definition aacenc.h:194
float tnsg[128]
TNS synthesis gain per band for THIS solve (1 = uncovered), M/S-aware (pair max)
Definition aacenc.h:198
int bst[128]
window group, swb, coef start
Definition aacenc.h:188
struct SingleChannelElement * sce
Definition aacenc.h:182
int bnc[128]
number of candidates
Definition aacenc.h:190
float thr_real[128]
real masking threshold (PNS gates)
Definition aacenc.h:197
uint8_t is_pns[128]
band coded as noise
Definition aacenc.h:201
int si
curve-bank index (nd/nb slot)
Definition aacenc.h:183
int bw[128]
Definition aacenc.h:188
float pener[128]
band energy (PNS noise target)
Definition aacenc.h:199
int bg[128]
Definition aacenc.h:188
float thr[128]
allocation-law effective threshold
Definition aacenc.h:196
int bidx[128]
sce band index (w*16+g)
Definition aacenc.h:187
int nbnd
coded-band count, 0 = nothing codeable
Definition aacenc.h:185
int blo[128]
finest candidate scalefactor
Definition aacenc.h:189
int act[128]
active (non-PNS) band coding order
Definition aacenc.h:192
float maxvals[128]
Definition aacenc.h:195
int is8
EIGHT_SHORT frame.
Definition aacenc.h:186
int cur_ch
encoder channel index (psy/cache context)
Definition aacenc.h:184
Single Channel Element - used for both SCE and LFE elements.
Definition aacdec.h:217
uint8_t zeroes[128]
band is not coded
Definition aacenc.h:116
float coeffs[1024]
coefficients for IMDCT, maybe processed
Definition aacenc.h:121
uint8_t can_pns[128]
band is allowed to PNS (informative)
Definition aacenc.h:117
TemporalNoiseShaping tns
Definition aacdec.h:220
float pns_ener[128]
Noise energy values.
Definition aacenc.h:119
enum BandType band_type[128]
band types
Definition aacdec.h:221
IndividualChannelStream ics
Definition aacdec.h:218
int sf_idx[128]
scalefactor indices
Definition aacenc.h:115
int length[8][4]
Definition aacdec.h:194
int order[8][4]
Definition aacdec.h:196
#define imb
const char * g
Definition vf_curves.c:128
static double cb(void *priv, double x, double y)
Definition vf_geq.c:247
static float mean(const float *input, int size)
Definition vf_nnedi.c:861
static double b0(void *priv, double x, double y)
Definition vf_xfade.c:2033
static int bias(int x, int c)
Definition vqcdec.c:115
static double c[64]