FFmpeg
Loading...
Searching...
No Matches
aacenc.c
Go to the documentation of this file.
1/*
2 * AAC encoder
3 * Copyright (C) 2008 Konstantin Shishkov
4 *
5 * This file is part of FFmpeg.
6 *
7 * FFmpeg is free software; you can redistribute it and/or
8 * modify it under the terms of the GNU Lesser General Public
9 * License as published by the Free Software Foundation; either
10 * version 2.1 of the License, or (at your option) any later version.
11 *
12 * FFmpeg is distributed in the hope that it will be useful,
13 * but WITHOUT ANY WARRANTY; without even the implied warranty of
14 * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU
15 * Lesser General Public License for more details.
16 *
17 * You should have received a copy of the GNU Lesser General Public
18 * License along with FFmpeg; if not, write to the Free Software
19 * Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA
20 */
21
22/**
23 * @file
24 * AAC encoder
25 */
26
27/***********************************
28 * TODOs:
29 * add sane pulse detection
30 ***********************************/
31#include <float.h>
32#include <math.h>
33
35#include "libavutil/crc.h"
36#include "libavutil/float_dsp.h"
37#include "libavutil/mem.h"
38#include "libavutil/opt.h"
39#include "avcodec.h"
40#include "codec_internal.h"
41#include "encode.h"
42#include "put_bits.h"
43#include "mpeg4audio.h"
44#include "sinewin.h"
45#include "profiles.h"
46#include "version.h"
47
48#include "aac.h"
49#include "aactab.h"
50#include "aacenc.h"
51#include "aacenctab.h"
52#include "aacenc_utils.h"
53
54#include "psymodel.h"
55
56/**
57 * List of PCE (Program Configuration Element) for the channel layouts listed
58 * in channel_layout.h
59 *
60 * For those wishing in the future to add other layouts:
61 *
62 * - num_ele: number of elements in each group of front, side, back, lfe channels
63 * (an element is of type SCE (single channel), CPE (channel pair) for
64 * the first 3 groups; and is LFE for LFE group).
65 *
66 * - pairing: 0 for an SCE element or 1 for a CPE; does not apply to LFE group
67 *
68 * - index: there are three independent indices for SCE, CPE and LFE;
69 * they are incremented irrespective of the group to which the element belongs;
70 * they are not reset when going from one group to another
71 *
72 * Example: for 7.0 channel layout,
73 * .pairing = { { 1, 0 }, { 1 }, { 1 }, }, (3 CPE and 1 SCE in front group)
74 * .index = { { 0, 0 }, { 1 }, { 2 }, },
75 * (index is 0 for the single SCE but goes from 0 to 2 for the CPEs)
76 *
77 * The index order impacts the channel ordering. But is otherwise arbitrary
78 * (the sequence could have been 2, 0, 1 instead of 0, 1, 2).
79 *
80 * Spec allows for discontinuous indices, e.g. if one has a total of two SCE,
81 * SCE.0 SCE.15 is OK per spec; BUT it won't be decoded by our AAC decoder
82 * which at this time requires that indices fully cover some range starting
83 * from 0 (SCE.1 SCE.0 is OK but not SCE.0 SCE.15).
84 *
85 * - height: 0 for a base layer element, 1 for a top layer element, 2 for a bottom
86 * layer element.
87 *
88 * - config_map: total number of elements and their types. Beware, the way the
89 * types are ordered impacts the final channel ordering.
90 *
91 * - reorder_map: reorders the channels.
92 *
93 */
94static const AACPCEInfo aac_pce_configs[] = {
95 {
96 .layout = AV_CHANNEL_LAYOUT_MONO,
97 .num_ele = { 1, 0, 0, 0 },
98 .pairing = { { 0 }, },
99 .index = { { 0 }, },
100 .config_map = { 1, TYPE_SCE, },
101 .reorder_map = { 0 },
102 },
103 {
104 .layout = AV_CHANNEL_LAYOUT_STEREO,
105 .num_ele = { 1, 0, 0, 0 },
106 .pairing = { { 1 }, },
107 .index = { { 0 }, },
108 .config_map = { 1, TYPE_CPE, },
109 .reorder_map = { 0, 1 },
110 },
111 {
113 .num_ele = { 1, 0, 0, 1 },
114 .pairing = { { 1 }, },
115 .index = { { 0 },{ 0 },{ 0 },{ 0 } },
116 .config_map = { 2, TYPE_CPE, TYPE_LFE },
117 .reorder_map = { 0, 1, 2 },
118 },
119 {
120 .layout = AV_CHANNEL_LAYOUT_2_1,
121 .num_ele = { 1, 0, 1, 0 },
122 .pairing = { { 1 },{ 0 },{ 0 } },
123 .index = { { 0 },{ 0 },{ 0 }, },
124 .config_map = { 2, TYPE_CPE, TYPE_SCE },
125 .reorder_map = { 0, 1, 2 },
126 },
127 {
129 .num_ele = { 2, 0, 0, 0 },
130 .pairing = { { 0, 1 }, },
131 .index = { { 0, 0 }, },
132 .config_map = { 2, TYPE_SCE, TYPE_CPE },
133 .reorder_map = { 2, 0, 1 },
134 },
135 {
137 .num_ele = { 2, 0, 0, 1 },
138 .pairing = { { 0, 1 }, },
139 .index = { { 0, 0 }, { 0 }, { 0 }, { 0 }, },
140 .config_map = { 3, TYPE_SCE, TYPE_CPE, TYPE_LFE },
141 .reorder_map = { 2, 0, 1, 3 },
142 },
143 {
145 .num_ele = { 2, 0, 1, 0 },
146 .pairing = { { 0, 1 }, { 0 }, { 0 }, },
147 .index = { { 0, 0 }, { 0 }, { 1 } },
148 .config_map = { 3, TYPE_SCE, TYPE_CPE, TYPE_SCE },
149 .reorder_map = { 2, 0, 1, 3 },
150 },
151 {
153 .num_ele = { 2, 0, 1, 1 },
154 .pairing = { { 0, 1 }, { 0 }, { 0 }, },
155 .index = { { 0, 0 }, { 0 }, { 1 }, { 0 } },
156 .config_map = { 4, TYPE_SCE, TYPE_CPE, TYPE_SCE, TYPE_LFE },
157 .reorder_map = { 2, 0, 1, 4, 3 },
158 },
159 {
160 .layout = AV_CHANNEL_LAYOUT_2_2,
161 .num_ele = { 1, 0, 1, 0 },
162 .pairing = { { 1 }, { 0 }, { 1 }, },
163 .index = { { 0 }, { 0 }, { 1 } },
164 .config_map = { 2, TYPE_CPE, TYPE_CPE },
165 .reorder_map = { 0, 1, 2, 3 },
166 },
167 {
168 .layout = AV_CHANNEL_LAYOUT_QUAD,
169 .num_ele = { 1, 0, 1, 0 },
170 .pairing = { { 1 }, { 0 }, { 1 }, },
171 .index = { { 0 }, { 0 }, { 1 } },
172 .config_map = { 2, TYPE_CPE, TYPE_CPE },
173 .reorder_map = { 0, 1, 2, 3 },
174 },
175 {
177 .num_ele = { 2, 0, 1, 0 },
178 .pairing = { { 0, 1 }, { 0 }, { 1 } },
179 .index = { { 0, 0 }, { 0 }, { 1 } },
180 .config_map = { 3, TYPE_SCE, TYPE_CPE, TYPE_CPE },
181 .reorder_map = { 2, 0, 1, 3, 4 },
182 },
183 {
185 .num_ele = { 2, 0, 1, 1 },
186 .pairing = { { 0, 1 }, { 0 }, { 1 }, },
187 .index = { { 0, 0 }, { 0 }, { 1 }, { 0 } },
188 .config_map = { 4, TYPE_SCE, TYPE_CPE, TYPE_CPE, TYPE_LFE },
189 .reorder_map = { 2, 0, 1, 4, 5, 3 },
190 },
191 {
193 .num_ele = { 2, 0, 1, 0 },
194 .pairing = { { 0, 1 }, { 0 }, { 1 } },
195 .index = { { 0, 0 }, { 0 }, { 1 } },
196 .config_map = { 3, TYPE_SCE, TYPE_CPE, TYPE_CPE },
197 .reorder_map = { 2, 0, 1, 3, 4 },
198 },
199 {
201 .num_ele = { 2, 0, 1, 1 },
202 .pairing = { { 0, 1 }, { 0 }, { 1 }, },
203 .index = { { 0, 0 }, { 0 }, { 1 }, { 0 } },
204 .config_map = { 4, TYPE_SCE, TYPE_CPE, TYPE_CPE, TYPE_LFE },
205 .reorder_map = { 2, 0, 1, 4, 5, 3 },
206 },
207 {
209 .num_ele = { 2, 0, 2, 0 },
210 .pairing = { { 0, 1 }, { 0 }, { 1, 0 } },
211 .index = { { 0, 0 }, { 0 }, { 1, 1 } },
212 .config_map = { 4, TYPE_SCE, TYPE_CPE, TYPE_CPE, TYPE_SCE },
213 .reorder_map = { 2, 0, 1, 4, 5, 3 },
214 },
215 {
217 .num_ele = { 2, 0, 1, 0 },
218 .pairing = { { 1, 1 }, { 0 }, { 1 } },
219 .index = { { 0, 1 }, { 0 }, { 2 }, },
220 .config_map = { 3, TYPE_CPE, TYPE_CPE, TYPE_CPE, },
221 .reorder_map = { 2, 3, 0, 1, 4, 5 },
222 },
223 {
225 .num_ele = { 2, 0, 2, 0 },
226 .pairing = { { 0, 1 }, { 0 }, { 1, 0 } },
227 .index = { { 0, 0 }, { 0 }, { 1, 1 } },
228 .config_map = { 4, TYPE_SCE, TYPE_CPE, TYPE_CPE, TYPE_SCE },
229 .reorder_map = { 2, 0, 1, 3, 4, 5 },
230 },
231 {
233 .num_ele = { 2, 0, 2, 1 },
234 .pairing = { { 0, 1 }, { 0 }, { 1, 0 }, },
235 .index = { { 0, 0 }, { 0 }, { 1, 1 }, { 0 } },
236 .config_map = { 5, TYPE_SCE, TYPE_CPE, TYPE_CPE, TYPE_SCE, TYPE_LFE },
237 .reorder_map = { 2, 0, 1, 5, 6, 4, 3 },
238 },
239 {
241 .num_ele = { 2, 0, 2, 1 },
242 .pairing = { { 0, 1 },{ 0 },{ 1, 0 }, },
243 .index = { { 0, 0 },{ 0 },{ 1, 1 },{ 0 } },
244 .config_map = { 5, TYPE_SCE, TYPE_CPE, TYPE_CPE, TYPE_SCE, TYPE_LFE },
245 .reorder_map = { 2, 0, 1, 4, 5, 6, 3 },
246 },
247 {
249 .num_ele = { 2, 0, 1, 1 },
250 .pairing = { { 1, 1 }, { 0 }, { 1 }, },
251 .index = { { 0, 1 }, { 0 }, { 2 }, { 0 }, },
252 .config_map = { 4, TYPE_CPE, TYPE_CPE, TYPE_CPE, TYPE_LFE, },
253 .reorder_map = { 3, 4, 0, 1, 5, 6, 2 },
254 },
255 {
257 .num_ele = { 2, 0, 2, 0 },
258 .pairing = { { 0, 1 }, { 0 }, { 1, 1 }, },
259 .index = { { 0, 0 }, { 0 }, { 2, 1 }, },
260 .config_map = { 4, TYPE_SCE, TYPE_CPE, TYPE_CPE, TYPE_CPE },
261 .reorder_map = { 2, 0, 1, 3, 4, 5, 6 },
262 },
263 {
265 .num_ele = { 3, 0, 1, 0 },
266 .pairing = { { 0, 1, 1 }, { 0 }, { 1 }, },
267 .index = { { 0, 0, 1 }, { 0 }, { 2 }, },
268 .config_map = { 4, TYPE_SCE, TYPE_CPE, TYPE_CPE, TYPE_CPE },
269 .reorder_map = { 2, 3, 4, 0, 1, 5, 6 },
270 },
271 {
273 .num_ele = { 2, 0, 2, 1 },
274 .pairing = { { 0, 1 }, { 0 }, { 1, 1 }, },
275 .index = { { 0, 0 }, { 0 }, { 2, 1 }, { 0 } },
276 .config_map = { 5, TYPE_SCE, TYPE_CPE, TYPE_CPE, TYPE_CPE, TYPE_LFE },
277 .reorder_map = { 2, 0, 1, 4, 5, 6, 7, 3 },
278 },
279 {
281 .num_ele = { 3, 0, 1, 1 },
282 .pairing = { { 0, 1, 1 }, { 0 }, { 1 }, },
283 .index = { { 0, 0, 1 }, { 0 }, { 2 }, { 0 }, },
284 .config_map = { 5, TYPE_SCE, TYPE_CPE, TYPE_CPE, TYPE_CPE, TYPE_LFE },
285 .reorder_map = { 2, 4, 5, 0, 1, 6, 7, 3 },
286 },
287 {
289 .num_ele = { 3, 0, 1, 1 },
290 .pairing = { { 0, 1, 1 }, { 0 }, { 1 } },
291 .index = { { 0, 0, 1 }, { 0 }, { 2 }, { 0 } },
292 .config_map = { 5, TYPE_SCE, TYPE_CPE, TYPE_CPE, TYPE_CPE, TYPE_LFE },
293 .reorder_map = { 2, 6, 7, 0, 1, 4, 5, 3 },
294 },
295 {
297 .num_ele = { 2, 0, 3, 0 },
298 .pairing = { { 0, 1 }, { 0 }, { 1, 1, 0 }, },
299 .index = { { 0, 0 }, { 0 }, { 1, 2, 1 }, },
300 .config_map = { 5, TYPE_SCE, TYPE_CPE, TYPE_CPE, TYPE_CPE, TYPE_SCE },
301 .reorder_map = { 2, 0, 1, 6, 7, 3, 4, 5 },
302 },
303 {
305 .num_ele = { 3, 0, 1, 1 },
306 .pairing = { { 0, 1, 1 }, { 0 }, { 1 }, },
307 .index = { { 0, 0, 2 }, { 0 }, { 1 }, { 0 }, },
308 .height = { { 0, 0, 1 }, { 0 }, { 0 } },
309 .config_map = { 5, TYPE_SCE, TYPE_CPE, TYPE_CPE, TYPE_LFE, TYPE_CPE },
310 .reorder_map = { 2, 0, 1, 4, 5, 3, 6, 7 },
311 },
312 {
313 // ITU-R BS.2051-3 Sound System C
315 .num_ele = { 3, 0, 1, 1 },
316 .pairing = { { 0, 1, 1 }, { 0 }, { 1 }, },
317 .index = { { 0, 0, 2 }, { 0 }, { 1 }, { 0 }, },
318 .height = { { 0, 0, 1 }, { 0 }, { 0 } },
319 .config_map = { 5, TYPE_SCE, TYPE_CPE, TYPE_CPE, TYPE_LFE, TYPE_CPE },
320 .reorder_map = { 2, 0, 1, 4, 5, 3, 6, 7 },
321 },
322 {
324 .num_ele = { 3, 0, 2, 1 },
325 .pairing = { { 0, 1, 1 }, { 0 }, { 1, 1 }, },
326 .index = { { 0, 0, 2 }, { 0 }, { 1, 3 }, { 0 }, },
327 .height = { { 0, 0, 1 }, { 0 }, { 0, 1 } },
328 .config_map = { 6, TYPE_SCE, TYPE_CPE, TYPE_CPE, TYPE_LFE, TYPE_CPE, TYPE_CPE },
329 .reorder_map = { 2, 0, 1, 4, 5, 3, 6, 7, 8, 9 },
330 },
331 // ITU-R BS.2051-3 Sound System D
332 {
333 .layout = {
334 .nb_channels = 10,
337 },
338 .num_ele = { 3, 0, 2, 1 },
339 .pairing = { { 0, 1, 1 }, { 0 }, { 1, 1 }, },
340 .index = { { 0, 0, 2 }, { 0 }, { 1, 3 }, { 0 }, },
341 .height = { { 0, 0, 1 }, { 0 }, { 0, 1 } },
342 .config_map = { 6, TYPE_SCE, TYPE_CPE, TYPE_CPE, TYPE_LFE, TYPE_CPE, TYPE_CPE },
343 .reorder_map = { 2, 0, 1, 4, 5, 3, 6, 7, 8, 9 },
344 },
345 {
346 // ITU-R BS.2051-3 Sound System E
347 .layout = {
348 .nb_channels = 11,
351 },
352 .num_ele = { 4, 0, 2, 1 },
353 .pairing = { { 0, 1, 1, 0 }, { 0 }, { 1, 1 }, },
354 .index = { { 0, 0, 2, 1 }, { 0 }, { 1, 3 }, { 0 }, },
355 .height = { { 0, 0, 1, 2 }, { 0 }, { 0, 1 } },
356 .config_map = { 7, TYPE_SCE, TYPE_CPE, TYPE_CPE, TYPE_LFE, TYPE_CPE, TYPE_CPE, TYPE_SCE },
357 .reorder_map = { 2, 0, 1, 4, 5, 3, 6, 7, 8, 9, 10 },
358 },
359 {
361 .num_ele = { 4, 1, 2, 1 },
362 .pairing = { { 0, 1, 0, 1 }, { 0 }, { 1, 1 }, },
363 .index = { { 0, 0, 1, 2 }, { 2 }, { 1, 3 }, { 0 }, },
364 .height = { { 0, 0, 1, 1 }, { 1 }, { 0, 1 } },
366 .reorder_map = { 2, 0, 1, 4, 5, 3, 8, 7, 9, 6, 10, 11 },
367 },
368 {
369 .layout = {
370 .nb_channels = 12,
374 },
375 .num_ele = { 4, 1, 2, 1 },
376 .pairing = { { 0, 1, 0, 1 }, { 0 }, { 1, 1 }, },
377 .index = { { 0, 0, 1, 2 }, { 2 }, { 1, 3 }, { 0 }, },
378 .height = { { 0, 0, 1, 1 }, { 1 }, { 0, 1 } },
380 .reorder_map = { 2, 0, 1, 4, 5, 3, 8, 7, 9, 6, 10, 11 },
381 },
382 {
384 .num_ele = { 3, 0, 2, 1 },
385 .pairing = { { 0, 1, 1 }, { 0 }, { 1, 1 }, },
386 .index = { { 0, 0, 3 }, { 0 }, { 2, 1 }, { 0 } },
387 .height = { { 0, 0, 1 }, { 0 }, { 0, 0 } },
388 .config_map = { 6, TYPE_SCE, TYPE_CPE, TYPE_CPE, TYPE_CPE, TYPE_LFE, TYPE_CPE },
389 .reorder_map = { 2, 0, 1, 4, 5, 6, 7, 3, 8, 9 },
390 },
391 {
392 // ITU-R BS.2051-3 Sound System F
394 .num_ele = { 3, 0, 3, 2 },
395 .pairing = { { 0, 1, 1 }, { 0 }, { 1, 1, 0 }, },
396 .index = { { 0, 0, 3 }, { 0 }, { 2, 1, 1 }, { 0, 1 } },
397 .height = { { 0, 0, 1 }, { 0 }, { 0, 0, 1 } },
399 .reorder_map = { 2, 0, 1, 4, 5, 6, 7, 3, 11, 8, 9, 10 },
400 },
401 {
402 // ITU-R BS.2051-3 Sound System J
404 .num_ele = { 3, 0, 3, 1 },
405 .pairing = { { 0, 1, 1 }, { 0 }, { 1, 1, 1 }, },
406 .index = { { 0, 0, 3 }, { 0 }, { 2, 1, 4 }, { 0 } },
407 .height = { { 0, 0, 1 }, { 0 }, { 0, 0, 1 } },
408 .config_map = { 7, TYPE_SCE, TYPE_CPE, TYPE_CPE, TYPE_CPE, TYPE_LFE, TYPE_CPE, TYPE_CPE },
409 .reorder_map = { 2, 0, 1, 4, 5, 6, 7, 3, 8, 9, 10, 11 },
410 },
411 {
413 .num_ele = { 4, 1, 3, 1 },
414 .pairing = { { 0, 1, 0, 1 }, { 0 }, { 1, 1, 1 }, },
415 .index = { { 0, 0, 1, 3 }, { 2 }, { 2, 1, 4 }, { 0 } },
416 .height = { { 0, 0, 1, 1 }, { 1 }, { 0, 0, 1 } },
418 .reorder_map = { 2, 0, 1, 4, 5, 6, 7, 3, 10, 9, 11, 8, 12, 13 },
419 },
420 {
421 // ITU-R BS.2051-3 Sound System G
423 .num_ele = { 4, 0, 3, 1 },
424 .pairing = { { 0, 1, 1, 1 }, { 0 }, { 1, 1, 1 }, },
425 .index = { { 0, 0, 1, 4 }, { 0 }, { 2, 3, 5 }, { 0 } },
426 .height = { { 0, 0, 0, 1 }, { 0 }, { 0, 0, 1 } },
428 .reorder_map = { 2, 6, 7, 0, 1, 8, 9, 4, 5, 3, 10, 11, 12, 13 },
429 },
430 {
432 .num_ele = { 4, 1, 3, 1 },
433 .pairing = { { 0, 1, 1, 1 }, { 1 }, { 1, 1, 1 }, },
434 .index = { { 0, 0, 1, 4 }, { 5 }, { 2, 3, 6 }, { 0 } },
435 .height = { { 0, 0, 0, 1 }, { 1 }, { 0, 0, 1 } },
437 .reorder_map = { 2, 6, 7, 0, 1, 8, 9, 4, 5, 3, 10, 11, 14, 15, 12, 13 },
438 },
439 {
441 .num_ele = { 1, 0, 1, 0 },
442 .pairing = { { 1 }, { 0 }, { 1 }, },
443 .index = { { 0 }, { 0 }, { 1 } },
444 .config_map = { 2, TYPE_CPE, TYPE_CPE },
445 .reorder_map = { 0, 1, 2, 3 },
446 },
447 {
448 .layout = { .order = AV_CHANNEL_ORDER_AMBISONIC, .nb_channels = 9 },
449 .num_ele = { 3, 0, 2, 0 },
450 .pairing = { { 0, 1, 1 }, { 0 }, { 1, 1 }, },
451 .index = { { 0, 0, 1 }, { 0 }, { 2, 3 }, },
452 .config_map = { 5, TYPE_SCE, TYPE_CPE, TYPE_CPE, TYPE_CPE, TYPE_CPE },
453 .reorder_map = { 2, 5, 6, 0, 1, 7, 8, 3, 4 },
454 },
455 {
456 .layout = { .order = AV_CHANNEL_ORDER_AMBISONIC, .nb_channels = 16 },
457 .num_ele = { 4, 0, 4, 0 },
458 .pairing = { { 1, 1, 1, 1 }, { 0 }, { 1, 1, 1, 1 }, },
459 .index = { { 0, 1, 2, 3 }, { 0 }, { 4, 5, 6, 7 }, },
461 .reorder_map = { 0, 1, 2, 3, 4, 5, 6, 7, 8, 9, 10, 11, 12, 13, 14, 15 },
462 },
463};
464
465static void put_pce(PutBitContext *pb, AVCodecContext *avctx)
466{
467 int i, j;
468 AACEncContext *s = avctx->priv_data;
469 AACPCEInfo *pce = &s->pce;
470 const int bitexact = avctx->flags & AV_CODEC_FLAG_BITEXACT;
471 const char *aux_data = bitexact ? "Lavc" : LIBAVCODEC_IDENT;
472
473 put_bits(pb, 4, 0);
474
475 put_bits(pb, 2, avctx->profile);
476 put_bits(pb, 4, s->samplerate_index);
477
478 put_bits(pb, 4, pce->num_ele[0]); /* Front */
479 put_bits(pb, 4, pce->num_ele[1]); /* Side */
480 put_bits(pb, 4, pce->num_ele[2]); /* Back */
481 put_bits(pb, 2, pce->num_ele[3]); /* LFE */
482 put_bits(pb, 3, 0); /* Assoc data */
483 put_bits(pb, 4, 0); /* CCs */
484
485 put_bits(pb, 1, 0); /* Stereo mixdown */
486 put_bits(pb, 1, 0); /* Mono mixdown */
487 put_bits(pb, 1, 0); /* Something else */
488
489 for (i = 0; i < 4; i++) {
490 for (j = 0; j < pce->num_ele[i]; j++) {
491 if (i < 3)
492 put_bits(pb, 1, pce->pairing[i][j]);
493 put_bits(pb, 4, pce->index[i][j]);
494 }
495 }
496
497 align_put_bits(pb);
498 if (s->needs_height_ext) {
499 const AVCRC *crc_ctx = av_crc_get_table(AV_CRC_8_ATM);
500 PutBitContext height_pb;
501 uint8_t buf[16];
502 int bits = 8 + pce->num_ele[0] * 2 + pce->num_ele[1] * 2 + pce->num_ele[2] * 2;
503 int bytes = (bits + 7) / 8;
504
505 init_put_bits(&height_pb, buf, bytes);
506 put_bits(&height_pb, 8, 0xAC);
507 for (i = 0; i < 3; i++)
508 for (j = 0; j < pce->num_ele[i]; j++)
509 put_bits(&height_pb, 2, pce->height[i][j]);
510 flush_put_bits(&height_pb);
511
512 put_bits(pb, 8, bytes + 1);
513 ff_copy_bits(pb, buf, bits);
514 align_put_bits(pb);
515 put_bits(pb, 8, av_crc(crc_ctx, 0xFF, buf, bytes));
516 } else {
517 put_bits(pb, 8, strlen(aux_data));
518 ff_put_string(pb, aux_data, 0);
519 }
520}
521
522/**
523 * Make AAC audio config object.
524 * @see 1.6.2.1 "Syntax - AudioSpecificConfig"
525 */
526static int put_audio_specific_config(AVCodecContext *avctx, int chcfg)
527{
528 PutBitContext pb;
529 AACEncContext *s = avctx->priv_data;
530 const int max_size = 32;
531
532 avctx->extradata = av_mallocz(max_size);
533 if (!avctx->extradata)
534 return AVERROR(ENOMEM);
535
536 init_put_bits(&pb, avctx->extradata, max_size);
537 put_bits(&pb, 5, s->profile+1); //profile
538 put_bits(&pb, 4, s->samplerate_index); //sample rate index
539 put_bits(&pb, 4, chcfg);
540 //GASpecificConfig
541 put_bits(&pb, 1, 0); //frame length - 1024 samples
542 put_bits(&pb, 1, 0); //does not depend on core coder
543 put_bits(&pb, 1, 0); //is not extension
544 if (s->needs_pce)
545 put_pce(&pb, avctx);
546
547 //Explicitly Mark SBR absent
548 put_bits(&pb, 11, 0x2b7); //sync extension
549 put_bits(&pb, 5, AOT_SBR);
550 put_bits(&pb, 1, 0);
551 flush_put_bits(&pb);
552 avctx->extradata_size = put_bytes_output(&pb);
553
554 return 0;
555}
556
558{
559 ++s->quantize_band_cost_cache_generation;
560 if (s->quantize_band_cost_cache_generation == 0) {
561 memset(s->quantize_band_cost_cache, 0, sizeof(s->quantize_band_cost_cache));
562 s->quantize_band_cost_cache_generation = 1;
563 }
564}
565
566#define WINDOW_FUNC(type) \
567static void apply_ ##type ##_window(AVFloatDSPContext *fdsp, \
568 SingleChannelElement *sce, \
569 const float *audio)
570
571WINDOW_FUNC(only_long)
572{
573 const float *lwindow = sce->ics.use_kb_window[0] ? ff_aac_kbd_long_1024 : ff_sine_1024;
574 const float *pwindow = sce->ics.use_kb_window[1] ? ff_aac_kbd_long_1024 : ff_sine_1024;
575 float *out = sce->ret_buf;
576
577 fdsp->vector_fmul (out, audio, lwindow, 1024);
578 fdsp->vector_fmul_reverse(out + 1024, audio + 1024, pwindow, 1024);
579}
580
581WINDOW_FUNC(long_start)
582{
583 const float *lwindow = sce->ics.use_kb_window[1] ? ff_aac_kbd_long_1024 : ff_sine_1024;
584 const float *swindow = sce->ics.use_kb_window[0] ? ff_aac_kbd_short_128 : ff_sine_128;
585 float *out = sce->ret_buf;
586
587 fdsp->vector_fmul(out, audio, lwindow, 1024);
588 memcpy(out + 1024, audio + 1024, sizeof(out[0]) * 448);
589 fdsp->vector_fmul_reverse(out + 1024 + 448, audio + 1024 + 448, swindow, 128);
590 memset(out + 1024 + 576, 0, sizeof(out[0]) * 448);
591}
592
593WINDOW_FUNC(long_stop)
594{
595 const float *lwindow = sce->ics.use_kb_window[0] ? ff_aac_kbd_long_1024 : ff_sine_1024;
596 const float *swindow = sce->ics.use_kb_window[1] ? ff_aac_kbd_short_128 : ff_sine_128;
597 float *out = sce->ret_buf;
598
599 memset(out, 0, sizeof(out[0]) * 448);
600 fdsp->vector_fmul(out + 448, audio + 448, swindow, 128);
601 memcpy(out + 576, audio + 576, sizeof(out[0]) * 448);
602 fdsp->vector_fmul_reverse(out + 1024, audio + 1024, lwindow, 1024);
603}
604
605WINDOW_FUNC(eight_short)
606{
607 const float *swindow = sce->ics.use_kb_window[0] ? ff_aac_kbd_short_128 : ff_sine_128;
608 const float *pwindow = sce->ics.use_kb_window[1] ? ff_aac_kbd_short_128 : ff_sine_128;
609 const float *in = audio + 448;
610 float *out = sce->ret_buf;
611 int w;
612
613 for (w = 0; w < 8; w++) {
614 fdsp->vector_fmul (out, in, w ? pwindow : swindow, 128);
615 out += 128;
616 in += 128;
617 fdsp->vector_fmul_reverse(out, in, swindow, 128);
618 out += 128;
619 }
620}
621
622static void (*const apply_window[4])(AVFloatDSPContext *fdsp,
624 const float *audio) = {
625 [ONLY_LONG_SEQUENCE] = apply_only_long_window,
626 [LONG_START_SEQUENCE] = apply_long_start_window,
627 [EIGHT_SHORT_SEQUENCE] = apply_eight_short_window,
628 [LONG_STOP_SEQUENCE] = apply_long_stop_window
629};
630
632 float *audio)
633{
634 int i;
635 float *output = sce->ret_buf;
636
637 apply_window[sce->ics.window_sequence[0]](s->fdsp, sce, audio);
638
640 s->mdct1024_fn(s->mdct1024, sce->coeffs, output, sizeof(float));
641 else
642 for (i = 0; i < 1024; i += 128)
643 s->mdct128_fn(s->mdct128, &sce->coeffs[i], output + i*2, sizeof(float));
644 memcpy(audio, audio + 1024, sizeof(audio[0]) * 1024);
645 memcpy(sce->pcoeffs, sce->coeffs, sizeof(sce->pcoeffs));
646}
647
648/**
649 * Encode ics_info element.
650 * @see Table 4.6 (syntax of ics_info)
651 */
653{
654 int w;
655
656 put_bits(&s->pb, 1, 0); // ics_reserved bit
657 put_bits(&s->pb, 2, info->window_sequence[0]);
658 put_bits(&s->pb, 1, info->use_kb_window[0]);
659 if (info->window_sequence[0] != EIGHT_SHORT_SEQUENCE) {
660 put_bits(&s->pb, 6, info->max_sfb);
661 put_bits(&s->pb, 1, 0); /* No predictor present */
662 } else {
663 put_bits(&s->pb, 4, info->max_sfb);
664 for (w = 1; w < 8; w++)
665 put_bits(&s->pb, 1, !info->group_len[w]);
666 }
667}
668
669/**
670 * Encode MS data.
671 * @see 4.6.8.1 "Joint Coding - M/S Stereo"
672 */
674{
675 int i, w;
676
677 put_bits(pb, 2, cpe->ms_mode);
678 if (cpe->ms_mode == 1)
679 for (w = 0; w < cpe->ch[0].ics.num_windows; w += cpe->ch[0].ics.group_len[w])
680 for (i = 0; i < cpe->ch[0].ics.max_sfb; i++)
681 put_bits(pb, 1, cpe->ms_mask[w*16 + i]);
682}
683
684/**
685 * Produce integer coefficients from scalefactors provided by the model.
686 */
687static void adjust_frame_information(ChannelElement *cpe, int chans)
688{
689 int i, w, w2, g, ch;
690 int maxsfb, cmaxsfb;
691
692 for (ch = 0; ch < chans; ch++) {
693 IndividualChannelStream *ics = &cpe->ch[ch].ics;
694 maxsfb = 0;
695 cpe->ch[ch].pulse.num_pulse = 0;
696 for (w = 0; w < ics->num_windows; w += ics->group_len[w]) {
697 for (cmaxsfb = ics->num_swb; cmaxsfb > 0 && cpe->ch[ch].zeroes[w*16+cmaxsfb-1]; cmaxsfb--)
698 ;
699 maxsfb = FFMAX(maxsfb, cmaxsfb);
700 }
701 ics->max_sfb = maxsfb;
702
703 //adjust zero bands for window groups
704 for (w = 0; w < ics->num_windows; w += ics->group_len[w]) {
705 for (g = 0; g < ics->max_sfb; g++) {
706 i = 1;
707 for (w2 = w; w2 < w + ics->group_len[w]; w2++) {
708 if (!cpe->ch[ch].zeroes[w2*16 + g]) {
709 i = 0;
710 break;
711 }
712 }
713 cpe->ch[ch].zeroes[w*16 + g] = i;
714 }
715 }
716 }
717
718 if (chans > 1 && cpe->common_window) {
719 IndividualChannelStream *ics0 = &cpe->ch[0].ics;
720 IndividualChannelStream *ics1 = &cpe->ch[1].ics;
721 int msc = 0;
722 ics0->max_sfb = FFMAX(ics0->max_sfb, ics1->max_sfb);
723 ics1->max_sfb = ics0->max_sfb;
724 for (w = 0; w < ics0->num_windows*16; w += 16)
725 for (i = 0; i < ics0->max_sfb; i++)
726 if (cpe->ms_mask[w+i])
727 msc++;
728 if (msc == 0 || ics0->max_sfb == 0)
729 cpe->ms_mode = 0;
730 else
731 cpe->ms_mode = msc < ics0->max_sfb * ics0->num_windows ? 1 : 2;
732 }
733}
734
736{
737 int w, w2, g, i;
738 IndividualChannelStream *ics = &cpe->ch[0].ics;
739 if (!cpe->common_window)
740 return;
741 for (w = 0; w < ics->num_windows; w += ics->group_len[w]) {
742 for (w2 = 0; w2 < ics->group_len[w]; w2++) {
743 int start = (w+w2) * 128;
744 for (g = 0; g < ics->num_swb; g++) {
745 int p = -1 + 2 * (cpe->ch[1].band_type[w*16+g] - 14);
746 float scale = cpe->ch[0].is_ener[w*16+g];
747 if (!cpe->is_mask[w*16 + g]) {
748 start += ics->swb_sizes[g];
749 continue;
750 }
751 if (cpe->ms_mask[w*16 + g])
752 p *= -1;
753 for (i = 0; i < ics->swb_sizes[g]; i++) {
754 float sum = (cpe->ch[0].coeffs[start+i] + p*cpe->ch[1].coeffs[start+i])*scale;
755 cpe->ch[0].coeffs[start+i] = sum;
756 cpe->ch[1].coeffs[start+i] = 0.0f;
757 }
758 start += ics->swb_sizes[g];
759 }
760 }
761 }
762}
763
764/* I/S acceptance level for the image-error EMA at full rate pressure */
765#define NMR_IS_IMG_GATE 8000.0f
766
767/* Frequency in Hz for the lower limit of intensity stereo */
768#define NMR_IS_LOW_LIMIT 6100
769
770/* M/S adoption: es < 0.5*em, content-driven and rate-free */
771#define NMR_MS_EQUIV 0.5f
772#define NMR_MS_MASK 0.0f
773
774/* Pair decouple threshold on the joint-tool candidacy fraction EMA: pairs
775 * whose joint tools are mostly dead (diffuse decorrelated content) window
776 * per-channel and skip M/S; recouple above 1.3x. */
777#define NMR_DECORR_LO 0.20f
778
779/* Stereo-decision hysteresis: leaving a joint mode costs a margin. */
780#define NMR_STICKY 2.0f
781
782/* Decision statistics are EMA-smoothed across frames. */
783#define NMR_SDEC_EMA 0.75f
784
785/* PNS-stereo gate: substitute only clearly-decorrelated (wide) bands. */
786#define NMR_PNS_STEREO_DECORR 0.6f
787
788/* M/S balance gate: no M/S on bands panned harder than this energy ratio */
789#define NMR_MS_BALANCE 0.25f
790
791/* Perceptual I/S: highly correlated long-window bands above this frequency
792 * take I/S at this image-error budget regardless of rate pressure */
793#define NMR_IS_PERC_FREQ 8000.0f
794#define NMR_IS_PERC_CORR 0.85f
795#define NMR_IS_PERC_GATE 50.0f
796
797/* Recode one band's window group as mid+side in place. */
799 int w, int g, int start, int len, int gl)
800{
801 SingleChannelElement *sce0 = &cpe->ch[0];
802 SingleChannelElement *sce1 = &cpe->ch[1];
803 cpe->ms_mask[w*16+g] = 1;
804 for (int w2 = 0; w2 < gl; w2++) {
805 FFPsyBand *b0 = &s->psy.ch[s->cur_channel+0].psy_bands[(w+w2)*16+g];
806 FFPsyBand *b1 = &s->psy.ch[s->cur_channel+1].psy_bands[(w+w2)*16+g];
807 float *L = sce0->coeffs + start + (w+w2)*128;
808 float *R = sce1->coeffs + start + (w+w2)*128;
809 float em = 0.0f, es = 0.0f;
810 for (int i = 0; i < len; i++) {
811 float m = (L[i] + R[i]) * 0.5f;
812 R[i] = m - R[i]; L[i] = m;
813 em += L[i]*L[i]; es += R[i]*R[i];
814 }
815 b0->threshold = FFMIN(b0->threshold, b1->threshold) * 0.5f;
816 b1->threshold = b0->threshold;
817 b0->energy = em; b1->energy = es;
818 }
819}
820
821/* I/S perceptual test: reconstruction image error vs the pair's masks. */
823 int w, int g, int start, int len, int gl,
824 float ener0, float ener1, float dot,
825 float minthr0, float minthr1, float *ratio_out,
826 float *scale_out, float *sr_out, int *p_out)
827{
828 int p = dot >= 0.0f ? 1 : -1;
829 float ener01 = ener0 + ener1 + 2*p*dot; /* energy of L + p*R */
830 *ratio_out = FLT_MAX;
831 if (ener01 <= FLT_MIN)
832 return 0;
833 float scale = sqrtf(ener0 / ener01); /* carrier = (L + p*R)*scale */
834 float sr_ = sqrtf(ener1 / ener0); /* decoder: R = p*sr_*carrier */
835 float img0 = 0.0f, img1 = 0.0f;
836 for (int w2 = 0; w2 < gl; w2++) {
837 const float *L = cpe->ch[0].coeffs + start + (w+w2)*128;
838 const float *R = cpe->ch[1].coeffs + start + (w+w2)*128;
839 for (int i = 0; i < len; i++) {
840 float c = (L[i] + p*R[i]) * scale;
841 float dl = L[i] - c, dr = R[i] - p*sr_*c;
842 img0 += dl*dl; img1 += dr*dr;
843 }
844 }
845 *ratio_out = FFMAX(img0 / FFMAX(minthr0 * gl, FLT_MIN),
846 img1 / FFMAX(minthr1 * gl, FLT_MIN));
847 *scale_out = scale; *sr_out = sr_; *p_out = p;
848 return 1;
849}
850
851/* Recode one band's window group as intensity stereo in place: replace L with the
852 * carrier, zero R, signal the phase via the side channel's band type, and fold the
853 * pair's masking into the surviving (carrier) channel. */
855 int w, int g, int start, int len, int gl,
856 float scale, float sr_, int p,
857 float ener0, float ener1)
858{
859 cpe->is_mask[w*16+g] = 1;
860 cpe->ch[0].is_ener[w*16+g] = scale;
861 cpe->ch[1].is_ener[w*16+g] = ener0 / ener1;
862 cpe->ch[1].band_type[w*16+g] = p > 0 ? INTENSITY_BT : INTENSITY_BT2;
863 for (int w2 = 0; w2 < gl; w2++) {
864 FFPsyBand *b0 = &s->psy.ch[s->cur_channel+0].psy_bands[(w+w2)*16+g];
865 FFPsyBand *b1 = &s->psy.ch[s->cur_channel+1].psy_bands[(w+w2)*16+g];
866 float *L = cpe->ch[0].coeffs + start + (w+w2)*128;
867 float *R = cpe->ch[1].coeffs + start + (w+w2)*128;
868 float ec = 0.0f;
869 for (int i = 0; i < len; i++) {
870 L[i] = (L[i] + p*R[i]) * scale;
871 R[i] = 0.0f;
872 ec += L[i]*L[i];
873 }
874 b0->threshold = FFMIN(b0->threshold, b1->threshold / FFMAX(sr_*sr_, 1e-9f));
875 b0->energy = ec; b1->energy = 0.0f;
876 }
877}
878
879/*
880 * Per-band stereo-mode decision (L/R vs M/S vs intensity) for the NMR coder,
881 * made before quantization from the psychoacoustic model alone, so the
882 * quantizer search allocates natively on the spectra that are actually coded.
883 */
885{
886 SingleChannelElement *sce0 = &cpe->ch[0];
887 SingleChannelElement *sce1 = &cpe->ch[1];
888 IndividualChannelStream *ics = &sce0->ics;
889 const AVCodecContext *avctx = s->psy.avctx;
890 const float freq_mult = avctx->sample_rate / (1024.0f / ics->num_windows) / 2.0f;
891 int is_count = 0;
892
893 if (s->nmr) {
894 int pi = (s->cur_channel >> 1) & 7;
895 pi = pi * 2 + (ics->num_windows == 8); /* per-grid state bank */
896 if (!s->nmr->sinit[pi]) {
897 /* one-time init; per-grid banks persist across window switches
898 * (wiping them churned stereo modes audibly) */
899 memset(s->nmr->smode[pi], 0, sizeof(s->nmr->smode[pi]));
900 for (int b = 0; b < 128; b++) {
901 s->nmr->sema_em[pi][b] = 0.0f;
902 s->nmr->sema_img[pi][b] = -1.0f;
903 }
904 s->nmr->sinit[pi] = 1;
905 }
906 }
907
908 /* Per-band stereo decision (L/R vs M/S vs I/S), made pre-quantization from
909 * the psy model so the trellis allocates on the coded spectra. */
910
911 /* I/S engages under SUSTAINED strain only: rate pressure gated by the
912 * lambda floor (pressure spikes at a comfortable operating point must
913 * not admit it). Unengaged candidates fall back to M/S. */
914 float is_ramp = s->nmr ? s->nmr->press *
915 av_clipf((s->nmr->lam_floor - 40.0f) / (120.0f - 40.0f), 0.0f, 1.0f) : 0.0f;
916 /* perceptual I/S (below) makes candidacy pressure-independent */
917 const int allow_is = s->options.intensity_stereo;
918
919 const int pidx = (s->cur_channel >> 1) & 15;
920 const int decoupled = s->psy.pair_decoupled[pidx];
921 int njoint = 0, nbands = 0; /* joint-tool candidacy census, decouple feed */
922
923 for (int w = 0; w < ics->num_windows; w += ics->group_len[w]) {
924 int start = 0;
925 for (int g = 0; g < ics->num_swb; start += ics->swb_sizes[g++]) {
926 int len = ics->swb_sizes[g], gl = ics->group_len[w];
927 float ener0 = 0.0f, ener1 = 0.0f, dot = 0.0f, es_tot = 0.0f, em_tot = 0.0f;
928 float minthr0 = FLT_MAX, minthr1 = FLT_MAX;
929
930 cpe->is_mask[w*16+g] = 0;
931 cpe->ms_mask[w*16+g] = 0;
932
933 for (int w2 = 0; w2 < gl; w2++) {
934 FFPsyBand *b0 = &s->psy.ch[s->cur_channel+0].psy_bands[(w+w2)*16+g];
935 FFPsyBand *b1 = &s->psy.ch[s->cur_channel+1].psy_bands[(w+w2)*16+g];
936 const float *L = sce0->coeffs + start + (w+w2)*128;
937 const float *R = sce1->coeffs + start + (w+w2)*128;
938 float el = 0.0f, er = 0.0f, em = 0.0f, es = 0.0f, d = 0.0f;
939 for (int i = 0; i < len; i++) {
940 float m = (L[i] + R[i]) * 0.5f;
941 float sv = m - R[i];
942 el += L[i]*L[i]; er += R[i]*R[i];
943 em += m*m; es += sv*sv; d += L[i]*R[i];
944 }
945 ener0 += el; ener1 += er; dot += d; es_tot += es; em_tot += em;
946 minthr0 = FFMIN(minthr0, b0->threshold);
947 minthr1 = FFMIN(minthr1, b1->threshold);
948 }
949 float thr_g = FFMIN(minthr0, minthr1) * gl; /* group masking budget */
950
951 /* PNS-stereo reservation: keep clearly-wide noise bands for PNS. */
952 const int sidx = w*16+g;
953 {
954 float es_w = es_tot, em_w = em_tot;
955 if (s->nmr) {
956 int pi_ = ((s->cur_channel >> 1) & 7) * 2 + (cpe->ch[0].ics.num_windows == 8);
957 float pe = s->nmr->sema_es[pi_][sidx];
958 float pm = s->nmr->sema_em[pi_][sidx];
959 if (pm > 0.0f) {
960 es_w = NMR_SDEC_EMA * pe + (1.0f - NMR_SDEC_EMA) * es_tot;
961 em_w = NMR_SDEC_EMA * pm + (1.0f - NMR_SDEC_EMA) * em_tot;
962 }
963 }
964 if (cpe->ch[0].can_pns[w*16+g] && cpe->ch[1].can_pns[w*16+g] &&
965 es_w > NMR_PNS_STEREO_DECORR * em_w)
966 continue;
967 }
968 cpe->ch[0].can_pns[w*16+g] = cpe->ch[1].can_pns[w*16+g] = 0;
969
970 int pi = ((s->cur_channel >> 1) & 7) * 2 + (cpe->ch[0].ics.num_windows == 8);
971 uint8_t *pmode = s->nmr ? s->nmr->smode[pi] : NULL;
972 int prev = pmode ? pmode[sidx] : 0;
973 float eqgate = NMR_MS_EQUIV * (prev == 1 ? 1.5f : 1.0f); /* stay-until es>0.75em */
974 /* I/S = lossy economy: image-error budget scales with pressure */
975 float imgate = NMR_IS_IMG_GATE * is_ramp * (prev == 2 ? NMR_STICKY : 1.0f);
976 /* Perceptual I/S (metric-blind by design - Zimtohrli penalizes
977 * even sub-mask image error, ears above ~8k do not hear
978 * interaural fine structure): engage on genuinely intensity-
979 * panned HF - high inter-channel correlation, long windows,
980 * sticky - regardless of rate pressure. Freed bits are judged
981 * by the sub-8k spectrum; the image itself is judged by ears. */
982 if (cpe->ch[0].ics.num_windows != 8 && ener0 > FLT_MIN && ener1 > FLT_MIN) {
983 float corr = dot / sqrtf(ener0 * ener1);
984 if (start * freq_mult > NMR_IS_PERC_FREQ &&
985 fabsf(corr) > NMR_IS_PERC_CORR * (prev == 2 ? 0.9f : 1.0f))
986 imgate = FFMAX(imgate, NMR_IS_PERC_GATE * (prev == 2 ? NMR_STICKY : 1.0f));
987 }
988 float es_d = es_tot, em_d = em_tot;
989 if (s->nmr) {
990 float *ees = &s->nmr->sema_es[pi][sidx];
991 float *eem = &s->nmr->sema_em[pi][sidx];
992 if (*eem <= 0.0f) { *ees = es_tot; *eem = em_tot; }
993 else {
994 *ees = NMR_SDEC_EMA * *ees + (1.0f - NMR_SDEC_EMA) * es_tot;
995 *eem = NMR_SDEC_EMA * *eem + (1.0f - NMR_SDEC_EMA) * em_tot;
996 }
997 es_d = *ees; em_d = *eem;
998 }
999 /* Balance gate: M/S has no coding gain on a hard-panned band
1000 * (|S| ~ |M|), and M/S quantization noise decorrelates across
1001 * the unfold, smearing the panned source into the far channel. */
1002 int bal_ok = FFMIN(ener0, ener1) > NMR_MS_BALANCE * FFMAX(ener0, ener1);
1003 int ms_would = s->options.mid_side &&
1004 (s->options.mid_side == 1 ||
1005 ((es_d < eqgate * em_d ||
1006 es_tot < NMR_MS_MASK * thr_g) && bal_ok));
1007 int ms_ok = ms_would && !decoupled;
1008 float scale, sr_, imgratio; int p;
1009 /* I/S competes with M/S above the frequency limit (candidacy must
1010 * not be gated on !ms_ok - that leaves only unrenderable bands) */
1011 int is_cand = start * freq_mult > NMR_IS_LOW_LIMIT &&
1012 ener0 > FLT_MIN && ener1 > FLT_MIN &&
1013 nmr_is_image_masked(s, cpe, w, g, start, len, gl,
1014 ener0, ener1, dot, minthr0, minthr1,
1015 &imgratio, &scale, &sr_, &p);
1016 int is_ok = is_cand;
1017 if (s->nmr && start * freq_mult > NMR_IS_LOW_LIMIT) {
1018 /* smoothed image-error; updated only while candidate (fail-value
1019 * feeding jammed it permanently high) */
1020 float *eim = &s->nmr->sema_img[pi][sidx];
1021 if (is_cand) {
1022 /* seed from first measurement; freeze when not candidate */
1023 if (*eim < 0.0f) *eim = imgratio;
1024 else *eim = NMR_SDEC_EMA * *eim + (1.0f - NMR_SDEC_EMA) * FFMIN(imgratio, 100.0f * NMR_IS_IMG_GATE);
1025 }
1026 is_ok = is_cand && *eim >= 0.0f && *eim < imgate;
1027 }
1028
1029 njoint += ms_would || is_ok; nbands++;
1030 if (pmode) {
1031 int m_ = (is_ok && allow_is) ? 2 : ms_ok ? 1 :
1032 (is_ok && s->options.mid_side) ? 1 : 0;
1033 pmode[sidx] = m_;
1034 s->nmr->smode_band[(s->cur_channel >> 1) & 7][w*16+g] = m_;
1035 }
1036 if (is_ok && allow_is) {
1037 nmr_apply_is_band(s, cpe, w, g, start, len, gl,
1038 scale, sr_, p, ener0, ener1);
1039 is_count++;
1040 } else if (ms_ok || (is_ok && s->options.mid_side)) {
1041 nmr_apply_ms_band(s, cpe, w, g, start, len, gl);
1042 }
1043 /* else: keep full L/R stereo */
1044 }
1045 }
1046 cpe->is_mode = !!is_count;
1047
1048 if (nbands > 0) {
1049 /* Pair joint-tool value, read next frame by the psy pair-synced window
1050 * decision and the M/S candidacy above. Measured as CANDIDACY (not
1051 * adoption) so decoupling cannot starve its own signal and self-lock. */
1052 float r = (float)njoint / nbands;
1053 float *pj = &s->psy.pair_joint[pidx];
1054 *pj = *pj > 0.0f ? 0.95f * *pj + 0.05f * r : r;
1055 s->psy.pair_decoupled[pidx] = *pj <
1056 (s->psy.pair_decoupled[pidx] ? 1.3f * NMR_DECORR_LO : NMR_DECORR_LO);
1057 }
1058}
1059
1061{
1062 int w, w2, g, i;
1063 IndividualChannelStream *ics = &cpe->ch[0].ics;
1064 if (!cpe->common_window)
1065 return;
1066 for (w = 0; w < ics->num_windows; w += ics->group_len[w]) {
1067 for (w2 = 0; w2 < ics->group_len[w]; w2++) {
1068 int start = (w+w2) * 128;
1069 for (g = 0; g < ics->num_swb; g++) {
1070 /* ms_mask can be used for other purposes in PNS and I/S,
1071 * so must not apply M/S if any band uses either, even if
1072 * ms_mask is set.
1073 */
1074 if (!cpe->ms_mask[w*16 + g] || cpe->is_mask[w*16 + g]
1075 || cpe->ch[0].band_type[w*16 + g] >= NOISE_BT
1076 || cpe->ch[1].band_type[w*16 + g] >= NOISE_BT) {
1077 start += ics->swb_sizes[g];
1078 continue;
1079 }
1080 for (i = 0; i < ics->swb_sizes[g]; i++) {
1081 float L = (cpe->ch[0].coeffs[start+i] + cpe->ch[1].coeffs[start+i]) * 0.5f;
1082 float R = L - cpe->ch[1].coeffs[start+i];
1083 cpe->ch[0].coeffs[start+i] = L;
1084 cpe->ch[1].coeffs[start+i] = R;
1085 }
1086 start += ics->swb_sizes[g];
1087 }
1088 }
1089 }
1090}
1091
1092/**
1093 * Encode scalefactor band coding type.
1094 */
1096{
1097 int w;
1098
1099 if (s->coder->set_special_band_scalefactors)
1100 s->coder->set_special_band_scalefactors(s, sce);
1101
1102 for (w = 0; w < sce->ics.num_windows; w += sce->ics.group_len[w])
1103 {
1104 /* the sectioning trellis must trade section bits against
1105 * spectral bits at the coder's REAL operating lambda; the
1106 * NMR outer-loop lambda is a static 120 */
1107 float slam = s->lambda;
1108 if (s->options.coder == AAC_CODER_NMR && s->nmr &&
1109 s->nmr->lam_slew > 0.0f)
1110 slam = s->nmr->lam_slew;
1111 s->coder->encode_window_bands_info(s, sce, w, sce->ics.group_len[w], slam);
1112 }
1113}
1114
1115/**
1116 * Encode scalefactors.
1117 */
1120{
1121 int diff, off_sf = sce->sf_idx[0], off_pns = sce->sf_idx[0] - NOISE_OFFSET;
1122 int off_is = 0, noise_flag = 1;
1123 int i, w;
1124
1125 for (w = 0; w < sce->ics.num_windows; w += sce->ics.group_len[w]) {
1126 for (i = 0; i < sce->ics.max_sfb; i++) {
1127 if (!sce->zeroes[w*16 + i]) {
1128 if (sce->band_type[w*16 + i] == NOISE_BT) {
1129 diff = sce->sf_idx[w*16 + i] - off_pns;
1130 off_pns = sce->sf_idx[w*16 + i];
1131 if (noise_flag-- > 0) {
1133 continue;
1134 }
1135 } else if (sce->band_type[w*16 + i] == INTENSITY_BT ||
1136 sce->band_type[w*16 + i] == INTENSITY_BT2) {
1137 diff = sce->sf_idx[w*16 + i] - off_is;
1138 off_is = sce->sf_idx[w*16 + i];
1139 } else {
1140 diff = sce->sf_idx[w*16 + i] - off_sf;
1141 off_sf = sce->sf_idx[w*16 + i];
1142 }
1144 av_assert0(diff >= 0 && diff <= 120);
1146 }
1147 }
1148 }
1149}
1150
1151/**
1152 * Encode pulse data.
1153 */
1154static void encode_pulses(AACEncContext *s, Pulse *pulse)
1155{
1156 int i;
1157
1158 put_bits(&s->pb, 1, !!pulse->num_pulse);
1159 if (!pulse->num_pulse)
1160 return;
1161
1162 put_bits(&s->pb, 2, pulse->num_pulse - 1);
1163 put_bits(&s->pb, 6, pulse->start);
1164 for (i = 0; i < pulse->num_pulse; i++) {
1165 put_bits(&s->pb, 5, pulse->pos[i]);
1166 put_bits(&s->pb, 4, pulse->amp[i]);
1167 }
1168}
1169
1170/**
1171 * Encode spectral coefficients processed by psychoacoustic model.
1172 */
1174{
1175 int start, i, w, w2;
1176
1177 for (w = 0; w < sce->ics.num_windows; w += sce->ics.group_len[w]) {
1178 start = 0;
1179 for (i = 0; i < sce->ics.max_sfb; i++) {
1180 if (sce->zeroes[w*16 + i]) {
1181 start += sce->ics.swb_sizes[i];
1182 continue;
1183 }
1184 for (w2 = w; w2 < w + sce->ics.group_len[w]; w2++) {
1185 s->coder->quantize_and_encode_band(s, &s->pb,
1186 &sce->coeffs[start + w2*128],
1187 NULL, sce->ics.swb_sizes[i],
1188 sce->sf_idx[w*16 + i],
1189 sce->band_type[w*16 + i],
1190 s->lambda,
1191 sce->ics.window_clipping[w]);
1192 }
1193 start += sce->ics.swb_sizes[i];
1194 }
1195 }
1196}
1197
1198/**
1199 * Downscale spectral coefficients for near-clipping windows to avoid artifacts
1200 */
1202{
1203 int start, i, j, w;
1204
1205 if (sce->ics.clip_avoidance_factor < 1.0f) {
1206 for (w = 0; w < sce->ics.num_windows; w++) {
1207 start = 0;
1208 for (i = 0; i < sce->ics.max_sfb; i++) {
1209 float *swb_coeffs = &sce->coeffs[start + w*128];
1210 for (j = 0; j < sce->ics.swb_sizes[i]; j++)
1211 swb_coeffs[j] *= sce->ics.clip_avoidance_factor;
1212 start += sce->ics.swb_sizes[i];
1213 }
1214 }
1215 }
1216}
1217
1218/**
1219 * Encode one channel of audio data.
1220 */
1223 int common_window)
1224{
1225 put_bits(&s->pb, 8, sce->sf_idx[0]);
1226 if (!common_window)
1227 put_ics_info(s, &sce->ics);
1228 encode_band_info(s, sce);
1229 encode_scale_factors(avctx, s, sce);
1230 encode_pulses(s, &sce->pulse);
1231 put_bits(&s->pb, 1, !!sce->tns.present);
1232 if (s->coder->encode_tns_info)
1233 s->coder->encode_tns_info(s, sce);
1234 put_bits(&s->pb, 1, 0); //ssr
1236 return 0;
1237}
1238
1239/**
1240 * Write some auxiliary information about the created AAC file.
1241 */
1242static void put_bitstream_info(AACEncContext *s, const char *name)
1243{
1244 int i, namelen, padbits;
1245
1246 namelen = strlen(name) + 2;
1247 put_bits(&s->pb, 3, TYPE_FIL);
1248 put_bits(&s->pb, 4, FFMIN(namelen, 15));
1249 if (namelen >= 15)
1250 put_bits(&s->pb, 8, namelen - 14);
1251 put_bits(&s->pb, 4, 0); //extension type - filler
1252 padbits = -put_bits_count(&s->pb) & 7;
1253 align_put_bits(&s->pb);
1254 for (i = 0; i < namelen - 2; i++)
1255 put_bits(&s->pb, 8, name[i]);
1256 put_bits(&s->pb, 12 - padbits, 0);
1257}
1258
1259/*
1260 * Copy input samples.
1261 * Channels are reordered from libavcodec's default order to AAC order.
1262 */
1264{
1265 int ch;
1266 int end = 2048 + (frame ? frame->nb_samples : 0);
1267 const uint8_t *channel_map = s->reorder_map;
1268
1269 /* copy and remap input samples */
1270 for (ch = 0; ch < s->channels; ch++) {
1271 /* copy last 1024 samples of previous frame to the start of the current frame */
1272 memcpy(&s->planar_samples[ch][1024], &s->planar_samples[ch][2048], 1024 * sizeof(s->planar_samples[0][0]));
1273
1274 /* copy new samples and zero any remaining samples */
1275 if (frame) {
1276 memcpy(&s->planar_samples[ch][2048],
1277 frame->extended_data[channel_map[ch]],
1278 frame->nb_samples * sizeof(s->planar_samples[0][0]));
1279 }
1280 memset(&s->planar_samples[ch][end], 0,
1281 (3072 - end) * sizeof(s->planar_samples[0][0]));
1282 }
1283}
1284
1285static int aac_encode_frame(AVCodecContext *avctx, AVPacket *avpkt,
1286 const AVFrame *frame, int *got_packet_ptr)
1287{
1288 AACEncContext *s = avctx->priv_data;
1289 float **samples = s->planar_samples, *samples2, *la, *overlap;
1290 ChannelElement *cpe;
1293 int i, its, ch, w, chans, tag, start_ch, ret, frame_bits;
1294 int target_bits, rate_bits, too_many_bits, too_few_bits;
1295 int ms_mode = 0, is_mode = 0, tns_mode = 0, pred_mode = 0;
1296 int chan_el_counter[4];
1298
1299 /* add current frame to queue */
1300 if (frame) {
1301 if ((ret = ff_af_queue_add(&s->afq, frame)) < 0)
1302 return ret;
1303 } else {
1304 if (!s->afq.remaining_samples || (!s->afq.frame_alloc && !s->afq.frame_count))
1305 return 0;
1306 }
1307
1309
1310 if (!avctx->frame_num)
1311 return 0;
1312
1313 start_ch = 0;
1314 for (i = 0; i < s->chan_map[0]; i++) {
1315 FFPsyWindowInfo* wi = windows + start_ch;
1316 tag = s->chan_map[i+1];
1317 chans = tag == TYPE_CPE ? 2 : 1;
1318 cpe = &s->cpe[i];
1319 {
1320 int wi_paired = 0;
1321 /* Synced pair windows: decide both channels of a CPE together so
1322 * their block switching never diverges (see psy window_pair). */
1323 if (chans == 2 && tag != TYPE_LFE && s->psy.model->window_pair && frame) {
1324 const float *ov0 = &samples[start_ch][0], *ov1 = &samples[start_ch + 1][0];
1325 s->psy.model->window_pair(&s->psy,
1326 ov0 + 1024, ov0 + 1024 + 448 + 64,
1327 ov1 + 1024, ov1 + 1024 + 448 + 64,
1328 start_ch, start_ch + 1,
1329 cpe->ch[0].ics.window_sequence[0],
1330 cpe->ch[1].ics.window_sequence[0],
1331 wi);
1332 wi_paired = 1;
1333 }
1334 for (ch = 0; ch < chans; ch++) {
1335 int k;
1336 float clip_avoidance_factor;
1337 sce = &cpe->ch[ch];
1338 ics = &sce->ics;
1339 s->cur_channel = start_ch + ch;
1340 overlap = &samples[s->cur_channel][0];
1341 samples2 = overlap + 1024;
1342 la = samples2 + (448+64);
1343 if (!frame)
1344 la = NULL;
1345 if (tag == TYPE_LFE) {
1346 wi[ch].window_type[0] = wi[ch].window_type[1] = ONLY_LONG_SEQUENCE;
1347 wi[ch].window_shape = 0;
1348 wi[ch].num_windows = 1;
1349 wi[ch].grouping[0] = 1;
1350 wi[ch].clipping[0] = 0;
1351
1352 /* Only the lowest 12 coefficients are used in a LFE channel.
1353 * The expression below results in only the bottom 8 coefficients
1354 * being used for 11.025kHz to 16kHz sample rates.
1355 */
1356 ics->num_swb = s->samplerate_index >= 8 ? 1 : 3;
1357 } else if (!wi_paired) {
1358 wi[ch] = s->psy.model->window(&s->psy, samples2, la, s->cur_channel,
1359 ics->window_sequence[0]);
1360 }
1361 ics->window_sequence[1] = ics->window_sequence[0];
1362 ics->window_sequence[0] = wi[ch].window_type[0];
1363 ics->use_kb_window[1] = ics->use_kb_window[0];
1364 ics->use_kb_window[0] = wi[ch].window_shape;
1365 ics->num_windows = wi[ch].num_windows;
1366 ics->swb_sizes = s->psy.bands [ics->num_windows == 8];
1367 ics->num_swb = tag == TYPE_LFE ? ics->num_swb : s->psy.num_bands[ics->num_windows == 8];
1368 ics->max_sfb = FFMIN(ics->max_sfb, ics->num_swb);
1369 ics->swb_offset = wi[ch].window_type[0] == EIGHT_SHORT_SEQUENCE ?
1370 ff_swb_offset_128 [s->samplerate_index]:
1371 ff_swb_offset_1024[s->samplerate_index];
1372 ics->tns_max_bands = wi[ch].window_type[0] == EIGHT_SHORT_SEQUENCE ?
1373 ff_tns_max_bands_128 [s->samplerate_index]:
1374 ff_tns_max_bands_1024[s->samplerate_index];
1375
1376 for (w = 0; w < ics->num_windows; w++)
1377 ics->group_len[w] = wi[ch].grouping[w];
1378
1379 /* Calculate input sample maximums and evaluate clipping risk */
1380 clip_avoidance_factor = 0.0f;
1381 for (w = 0; w < ics->num_windows; w++) {
1382 const float *wbuf = overlap + w * 128;
1383 const int wlen = 2048 / ics->num_windows;
1384 float max = 0;
1385 int j;
1386 /* mdct input is 2 * output */
1387 for (j = 0; j < wlen; j++)
1388 max = FFMAX(max, fabsf(wbuf[j]));
1389 wi[ch].clipping[w] = max;
1390 }
1391 /* Pre-attenuating hot frames costs 0.45 dB of level accuracy on
1392 * loud masters; float decoders don't clip, so the NMR coder
1393 * keeps levels exact. */
1394 if (s->options.coder == AAC_CODER_NMR)
1395 for (w = 0; w < ics->num_windows; w++)
1396 wi[ch].clipping[w] = 0;
1397 for (w = 0; w < ics->num_windows; w++) {
1398 if (wi[ch].clipping[w] > CLIP_AVOIDANCE_FACTOR) {
1399 ics->window_clipping[w] = 1;
1400 clip_avoidance_factor = FFMAX(clip_avoidance_factor, wi[ch].clipping[w]);
1401 } else {
1402 ics->window_clipping[w] = 0;
1403 }
1404 }
1405 if (clip_avoidance_factor > CLIP_AVOIDANCE_FACTOR) {
1406 ics->clip_avoidance_factor = CLIP_AVOIDANCE_FACTOR / clip_avoidance_factor;
1407 } else {
1408 ics->clip_avoidance_factor = 1.0f;
1409 }
1410
1411 apply_window_and_mdct(s, sce, overlap);
1412
1413 for (k = 0; k < 1024; k++) {
1414 if (!(fabs(cpe->ch[ch].coeffs[k]) < 1E16)) { // Ensure headroom for energy calculation
1415 av_log(avctx, AV_LOG_ERROR, "Input contains (near) NaN/+-Inf\n");
1416 return AVERROR(EINVAL);
1417 }
1418 }
1419 avoid_clipping(s, sce);
1420 }
1421 }
1422 start_ch += chans;
1423 }
1424 if ((ret = ff_alloc_packet(avctx, avpkt, 8192 * s->channels)) < 0)
1425 return ret;
1426 frame_bits = its = 0;
1427 do {
1428 init_put_bits(&s->pb, avpkt->data, avpkt->size);
1429
1430 if ((avctx->frame_num & 0xFF)==1 && !(avctx->flags & AV_CODEC_FLAG_BITEXACT))
1432 start_ch = 0;
1433 target_bits = 0;
1434 memset(chan_el_counter, 0, sizeof(chan_el_counter));
1435 for (i = 0; i < s->chan_map[0]; i++) {
1436 FFPsyWindowInfo* wi = windows + start_ch;
1437 const float *coeffs[2];
1438 tag = s->chan_map[i+1];
1439 chans = tag == TYPE_CPE ? 2 : 1;
1440 cpe = &s->cpe[i];
1441 cpe->common_window = 0;
1442 memset(cpe->is_mask, 0, sizeof(cpe->is_mask));
1443 memset(cpe->ms_mask, 0, sizeof(cpe->ms_mask));
1444 put_bits(&s->pb, 3, tag);
1445 put_bits(&s->pb, 4, chan_el_counter[tag]++);
1446 for (ch = 0; ch < chans; ch++) {
1447 sce = &cpe->ch[ch];
1448 coeffs[ch] = sce->coeffs;
1449 memset(&sce->tns, 0, sizeof(TemporalNoiseShaping));
1450 for (w = 0; w < 128; w++)
1451 if (sce->band_type[w] > RESERVED_BT)
1452 sce->band_type[w] = 0;
1453 }
1454 s->psy.bitres.alloc = -1;
1455 s->psy.bitres.bits = s->last_frame_pb_count / s->channels;
1456 s->psy.model->analyze(&s->psy, start_ch, coeffs, wi);
1457 if (s->psy.bitres.alloc > 0) {
1458 /* Lambda unused here on purpose, we need to take psy's unscaled allocation */
1459 target_bits += s->psy.bitres.alloc
1460 * (s->lambda / (avctx->global_quality ? avctx->global_quality : 120));
1461 s->psy.bitres.alloc /= chans;
1462 }
1463 s->cur_type = tag;
1464 if (chans > 1
1465 && wi[0].window_type[0] == wi[1].window_type[0]
1466 && wi[0].window_shape == wi[1].window_shape) {
1467
1468 cpe->common_window = 1;
1469 for (w = 0; w < wi[0].num_windows; w++) {
1470 if (wi[0].grouping[w] != wi[1].grouping[w]) {
1471 cpe->common_window = 0;
1472 break;
1473 }
1474 }
1475 }
1476
1477 const int use_tns = s->options.tns && s->coder->search_for_tns &&
1478 s->coder->apply_tns_filt;
1479
1480 /* The NMR coder rate-controls itself and never re-quantizes, so TNS must run
1481 * before the quantizer */
1482 const int tns_first = s->options.coder == AAC_CODER_NMR;
1483 if (tns_first && use_tns) {
1484 for (ch = 0; ch < chans; ch++) {
1485 sce = &cpe->ch[ch];
1486 s->cur_channel = start_ch + ch;
1487 /* mono: mark_pns before TNS so the region cap sees PNS bands. Stereo
1488 * PNS is marked in its own block (below) after the stereo decision. */
1489 if (chans == 1 && s->options.pns && s->coder->mark_pns)
1490 s->coder->mark_pns(s, avctx, sce);
1491 s->coder->search_for_tns(s, sce);
1492 s->coder->apply_tns_filt(s, sce);
1493 if (sce->tns.present)
1494 tns_mode = 1;
1495 }
1496 }
1497
1498 /* NMR stereo PNS (imaging-safe). Mark each channel's noise-like bands on the
1499 * original L/R psy, then keep PNS only where BOTH channels are noise-like. */
1500 if (chans == 2 && cpe->common_window && tns_first &&
1501 s->options.pns && s->coder->mark_pns) {
1502 s->cur_channel = start_ch; s->coder->mark_pns(s, avctx, &cpe->ch[0]);
1503 s->cur_channel = start_ch + 1; s->coder->mark_pns(s, avctx, &cpe->ch[1]);
1504 for (int b = 0; b < 128; b++)
1505 if (!cpe->ch[0].can_pns[b] || !cpe->ch[1].can_pns[b])
1506 cpe->ch[0].can_pns[b] = cpe->ch[1].can_pns[b] = 0;
1507 }
1508
1509 /* The NMR coder decides I/S and M/S BEFORE quantization, from the psy model,
1510 * and the trellis then allocates natively on the coeffs actually coded. */
1511 if (chans == 2 && cpe->common_window && s->options.coder == AAC_CODER_NMR &&
1512 (s->options.mid_side || s->options.intensity_stereo)) {
1513 s->cur_channel = start_ch;
1514 nmr_decide_stereo(s, cpe);
1515 }
1516 /* NMR pools the CPE bit budget: both channels of a pair are solved
1517 * jointly under one shared lambda (see aaccoder_nmr.h). */
1518 if (s->options.coder == AAC_CODER_NMR && s->nmr)
1519 s->nmr->pair = (chans == 2);
1520 for (ch = 0; ch < chans; ch++) {
1521 s->cur_channel = start_ch + ch;
1522 /* NMR PNS is mono-only */
1523 if (s->options.pns && s->coder->mark_pns && !tns_first)
1524 s->coder->mark_pns(s, avctx, &cpe->ch[ch]);
1525 s->coder->search_for_quantizers(avctx, s, &cpe->ch[ch], s->lambda);
1526 }
1527 for (ch = 0; ch < chans; ch++) { /* TNS (non-NMR) and PNS */
1528 sce = &cpe->ch[ch];
1529 s->cur_channel = start_ch + ch;
1530 if (!tns_first && use_tns) {
1531 s->coder->search_for_tns(s, sce);
1532 s->coder->apply_tns_filt(s, sce);
1533 if (sce->tns.present)
1534 tns_mode = 1;
1535 }
1536 if (s->options.pns && s->coder->search_for_pns)
1537 s->coder->search_for_pns(s, avctx, sce);
1538 }
1539 s->cur_channel = start_ch;
1540 if (s->options.intensity_stereo) { /* Intensity Stereo */
1541 if (s->options.coder != AAC_CODER_NMR) { /* NMR: decided pre-search */
1542 if (s->coder->search_for_is)
1543 s->coder->search_for_is(s, avctx, cpe);
1545 }
1546 if (cpe->is_mode) is_mode = 1;
1547 }
1548 if (s->options.mid_side && s->options.coder != AAC_CODER_NMR) { /* Mid/Side stereo */
1549 if (s->options.mid_side == -1 && s->coder->search_for_ms)
1550 s->coder->search_for_ms(s, cpe);
1551 else if (cpe->common_window)
1552 memset(cpe->ms_mask, 1, sizeof(cpe->ms_mask));
1554 }
1555 adjust_frame_information(cpe, chans);
1556 if (chans == 2) {
1557 put_bits(&s->pb, 1, cpe->common_window);
1558 if (cpe->common_window) {
1559 put_ics_info(s, &cpe->ch[0].ics);
1560 encode_ms_info(&s->pb, cpe);
1561 if (cpe->ms_mode) ms_mode = 1;
1562 }
1563 }
1564 for (ch = 0; ch < chans; ch++) {
1565 s->cur_channel = start_ch + ch;
1566 encode_individual_channel(avctx, s, &cpe->ch[ch], cpe->common_window);
1567 }
1568 start_ch += chans;
1569 }
1570
1571 if (avctx->flags & AV_CODEC_FLAG_QSCALE) {
1572 /* When using a constant Q-scale, don't mess with lambda, unless
1573 * the frame does not fit the decoder buffer: retry coarser (the
1574 * coders' legality caps shrink with lambda) */
1575 frame_bits = put_bits_count(&s->pb);
1576 if (frame_bits < 6144 * s->channels - 3 || its >= 16)
1577 break;
1578 s->lambda *= FFMIN(0.9f, (6144.0f * s->channels - 3) / frame_bits);
1579 for (i = 0; i < s->chan_map[0]; i++) {
1580 chans = s->chan_map[i + 1] == TYPE_CPE ? 2 : 1;
1581 for (ch = 0; ch < chans; ch++)
1582 memcpy(s->cpe[i].ch[ch].coeffs, s->cpe[i].ch[ch].pcoeffs,
1583 sizeof(s->cpe[i].ch[ch].coeffs));
1584 }
1585 its++;
1586 continue;
1587 }
1588
1589 frame_bits = put_bits_count(&s->pb);
1590
1591 /* The NMR coder rate-controls itself (global-lambda reservoir servo):
1592 * per-frame bits intentionally float around the nominal rate, so skip
1593 * the lambda rate loop and only intervene on a hard overflow. */
1594 if (s->options.coder == AAC_CODER_NMR && avctx->bit_rate_tolerance != 0 &&
1595 frame_bits < 6144 * s->channels - 3)
1596 break;
1597
1598 /* rate control stuff
1599 * allow between the nominal bitrate, and what psy's bit reservoir says to target
1600 * but drift towards the nominal bitrate always
1601 */
1602 rate_bits = avctx->bit_rate * 1024 / avctx->sample_rate;
1603 rate_bits = FFMIN(rate_bits, 6144 * s->channels - 3);
1604 too_many_bits = FFMAX(target_bits, rate_bits);
1605 too_many_bits = FFMIN(too_many_bits, 6144 * s->channels - 3);
1606 too_few_bits = FFMIN(FFMAX(rate_bits - rate_bits/4, target_bits), too_many_bits);
1607
1608 /* When strict bit-rate control is demanded */
1609 if (avctx->bit_rate_tolerance == 0) {
1610 if (rate_bits < frame_bits) {
1611 float ratio = ((float)rate_bits) / frame_bits;
1612 s->lambda *= FFMIN(0.9f, ratio);
1613 continue;
1614 }
1615 /* reset lambda when solution is found */
1616 s->lambda = avctx->global_quality > 0 ? avctx->global_quality : 120;
1617 break;
1618 }
1619
1620 /* When using ABR, be strict (but only for increasing) */
1621 too_few_bits = too_few_bits - too_few_bits/8;
1622 too_many_bits = too_many_bits + too_many_bits/2;
1623
1624 if ( its == 0 /* for steady-state Q-scale tracking */
1625 || (its < 5 && (frame_bits < too_few_bits || frame_bits > too_many_bits))
1626 || frame_bits >= 6144 * s->channels - 3 )
1627 {
1628 float ratio = ((float)rate_bits) / frame_bits;
1629
1630 if (frame_bits >= too_few_bits && frame_bits <= too_many_bits) {
1631 /*
1632 * This path is for steady-state Q-scale tracking
1633 * When frame bits fall within the stable range, we still need to adjust
1634 * lambda to maintain it like so in a stable fashion (large jumps in lambda
1635 * create artifacts and should be avoided), but slowly
1636 */
1637 ratio = sqrtf(sqrtf(ratio));
1638 ratio = av_clipf(ratio, 0.9f, 1.1f);
1639 } else {
1640 /* Not so fast though */
1641 ratio = sqrtf(ratio);
1642 }
1643 s->lambda = av_clipf(s->lambda * ratio, FLT_EPSILON, 65536.f);
1644
1645 /* Keep iterating if we must reduce and lambda is in the sky */
1646 if (ratio > 0.9f && ratio < 1.1f) {
1647 break;
1648 } else {
1649 if (is_mode || ms_mode || tns_mode || pred_mode) {
1650 for (i = 0; i < s->chan_map[0]; i++) {
1651 // Must restore coeffs
1652 chans = tag == TYPE_CPE ? 2 : 1;
1653 cpe = &s->cpe[i];
1654 for (ch = 0; ch < chans; ch++)
1655 memcpy(cpe->ch[ch].coeffs, cpe->ch[ch].pcoeffs, sizeof(cpe->ch[ch].coeffs));
1656 }
1657 }
1658 its++;
1659 }
1660 } else {
1661 break;
1662 }
1663 } while (1);
1664 if (avctx->flags & AV_CODEC_FLAG_QSCALE)
1665 s->lambda = avctx->global_quality > 0 ? avctx->global_quality : 120;
1666
1667 /* tool-usage stats over the final per-band decisions of this frame */
1668 for (i = 0; i < s->chan_map[0]; i++) {
1669 int etag = s->chan_map[i + 1], echans = etag == TYPE_CPE ? 2 : 1;
1670 ChannelElement *ce = &s->cpe[i];
1671 IndividualChannelStream *ics = &ce->ch[0].ics;
1672 for (ch = 0; ch < echans; ch++) { /* per-channel frame stats */
1673 int is_short = ce->ch[ch].ics.window_sequence[0] == EIGHT_SHORT_SEQUENCE;
1674 s->stat_chans++;
1675 if (is_short)
1676 s->stat_short++;
1677 if (ce->ch[ch].tns.present) {
1678 if (is_short) s->stat_tns_short++;
1679 else s->stat_tns_long++;
1680 }
1681 }
1682 for (w = 0; w < ics->num_windows; w += ics->group_len[w]) {
1683 for (int g = 0; g < ics->num_swb; g++) {
1684 int idx = w*16 + g, coded = 0;
1685 for (ch = 0; ch < echans; ch++) {
1686 SingleChannelElement *sce = &ce->ch[ch];
1687 if (sce->zeroes[idx] && sce->band_type[idx] == 0)
1688 continue;
1689 s->stat_ch_bands++;
1690 if (sce->band_type[idx] == NOISE_BT)
1691 s->stat_pns++;
1692 coded = 1;
1693 }
1694 if (etag == TYPE_CPE && coded) {
1695 s->stat_cpe_bands++;
1696 if (ce->ms_mask[idx]) s->stat_ms++;
1697 if (ce->is_mask[idx]) s->stat_is++;
1698 }
1699 }
1700 }
1701 }
1702
1703 put_bits(&s->pb, 3, TYPE_END);
1704 flush_put_bits(&s->pb);
1705
1706 s->last_frame_pb_count = put_bits_count(&s->pb);
1707
1708 /* NMR rate accounting: how many bits the frame really took beyond what the
1709 * trellis counted; feeds the next frame's budget correction */
1710 if (s->nmr) {
1711 int counted = 0;
1712 for (i = 0; i < s->channels; i++)
1713 counted += s->nmr->counted[i];
1714 if (counted > 0) {
1715 float side = (float)s->last_frame_pb_count - counted;
1716 if (s->nmr->side_inited) {
1717 s->nmr->side_ema += 0.125f * (side - s->nmr->side_ema);
1718 } else {
1719 s->nmr->side_ema = side;
1720 s->nmr->side_inited = 1;
1721 }
1722 }
1723 }
1724 avpkt->size = put_bytes_output(&s->pb);
1725
1726 /* NMR reports its real operating lambda: the corridor centre in CBR,
1727 * the quality-mode slew state in VBR/ABR - the lambda of the last long
1728 * operating-point solve, which short frames and legality re-solves do
1729 * not update (the outer-loop s->lambda is never touched for this coder
1730 * and would pin Qavg at its 120 init) */
1731 s->lambda_sum += (s->nmr && s->nmr->lam_slew > 0.0f &&
1732 ((avctx->flags & AV_CODEC_FLAG_QSCALE) || s->options.rc == 1)) ?
1733 s->nmr->lam_slew :
1734 (s->nmr && s->nmr->lam_rc > 0.0f) ? s->nmr->lam_rc : s->lambda;
1735 s->lambda_count++;
1736
1737 ret = ff_af_queue_remove(&s->afq, avctx->frame_size, avpkt);
1738 if (ret < 0)
1739 return ret;
1740
1741 avpkt->flags |= AV_PKT_FLAG_KEY;
1742
1743 *got_packet_ptr = 1;
1744 return 0;
1745}
1746
1748{
1749 AACEncContext *s = avctx->priv_data;
1750
1751 av_log(avctx, AV_LOG_INFO,
1752 "Qavg: %.3f Tr: %.1f%% TNS(L): %.1f%% TNS(S): %.1f%% M/S: %.1f%% I/S: %.1f%% PNS: %.1f%%\n",
1753 s->lambda_count ? s->lambda_sum / s->lambda_count : NAN,
1754 s->stat_chans ? 100.0 * s->stat_short / s->stat_chans : 0.0,
1755 s->stat_chans - s->stat_short ? 100.0 * s->stat_tns_long / (s->stat_chans - s->stat_short) : 0.0,
1756 s->stat_short ? 100.0 * s->stat_tns_short / s->stat_short : 0.0,
1757 s->stat_cpe_bands ? 100.0 * s->stat_ms / s->stat_cpe_bands : 0.0,
1758 s->stat_cpe_bands ? 100.0 * s->stat_is / s->stat_cpe_bands : 0.0,
1759 s->stat_ch_bands ? 100.0 * s->stat_pns / s->stat_ch_bands : 0.0);
1760
1761 av_tx_uninit(&s->mdct1024);
1762 av_tx_uninit(&s->mdct128);
1763 ff_psy_end(&s->psy);
1764 ff_lpc_end(&s->lpc);
1765 av_freep(&s->buffer.samples);
1766 av_freep(&s->cpe);
1767 av_freep(&s->fdsp);
1768 av_freep(&s->nmr);
1769 ff_af_queue_close(&s->afq);
1770 return 0;
1771}
1772
1774{
1775 int ret = 0;
1776 float scale = 32768.0f;
1777
1779 if (!s->fdsp)
1780 return AVERROR(ENOMEM);
1781
1782 if ((ret = av_tx_init(&s->mdct1024, &s->mdct1024_fn, AV_TX_FLOAT_MDCT, 0,
1783 1024, &scale, 0)) < 0)
1784 return ret;
1785 if ((ret = av_tx_init(&s->mdct128, &s->mdct128_fn, AV_TX_FLOAT_MDCT, 0,
1786 128, &scale, 0)) < 0)
1787 return ret;
1788
1789 return 0;
1790}
1791
1793{
1794 int ch;
1795 if (!FF_ALLOCZ_TYPED_ARRAY(s->buffer.samples, s->channels * 3 * 1024) ||
1796 !FF_ALLOCZ_TYPED_ARRAY(s->cpe, s->chan_map[0]))
1797 return AVERROR(ENOMEM);
1798
1799 for(ch = 0; ch < s->channels; ch++)
1800 s->planar_samples[ch] = s->buffer.samples + 3 * 1024 * ch;
1801
1802 if (s->options.coder == AAC_CODER_NMR) {
1803 s->nmr = av_mallocz(sizeof(*s->nmr));
1804 if (!s->nmr)
1805 return AVERROR(ENOMEM);
1806 }
1807
1808 return 0;
1809}
1810
1812{
1813 for (int i = 0; i < avctx->ch_layout.nb_channels; i++) {
1816 return 1;
1817 // Layouts with TOP_SIDE channels also include the above.
1818 }
1819
1820 return 0;
1821}
1822
1824{
1825 AACEncContext *s = avctx->priv_data;
1826 int i, ret = 0;
1827 int chcfg;
1828 const uint8_t *sizes[2];
1829 uint8_t grouping[AAC_MAX_CHANNELS];
1830 int lengths[2];
1831
1832 /* Constants */
1833 s->last_frame_pb_count = 0;
1834 avctx->frame_size = 1024;
1835 avctx->initial_padding = 1024;
1836 s->lambda = avctx->global_quality > 0 ? avctx->global_quality : 120;
1837
1838 /* Channel map and unspecified bitrate guessing */
1839 s->channels = avctx->ch_layout.nb_channels;
1840
1841 s->needs_pce = 1;
1842 for (chcfg = 1; chcfg < FF_ARRAY_ELEMS(aac_normal_chan_layouts); chcfg++) {
1844 s->needs_pce = s->options.pce;
1845 break;
1846 }
1847 }
1848
1849 if (!s->needs_pce && chcfg == 7 /* 7.1(wide) */ && !s->options.allow_71wide) {
1850 /**
1851 * FFmpeg used to produce out-of-spec AAC files that mistagged 7.1
1852 * as 7.1(wide), and this wark-around is still enabled by default in
1853 * aacdec.c, so avoid producing such files in the rare case that the
1854 * user correctly passed 7.1(wide) channel layout content.
1855 */
1856 av_log(avctx, AV_LOG_INFO, "Forcing the use of PCE to encode 7.1(wide) "
1857 "channel layout to avoid ambiguity. Set -aac_allow_71wide 1 to "
1858 "override this behavior and force the use of spec-compliant "
1859 "channel configuration ID.\n");
1860 s->needs_pce = 1;
1861 }
1862
1863 if (s->needs_pce) {
1864 char buf[64];
1865 for (i = 0; i < FF_ARRAY_ELEMS(aac_pce_configs); i++)
1866 if (!av_channel_layout_compare(&avctx->ch_layout, &aac_pce_configs[i].layout))
1867 break;
1868 av_channel_layout_describe(&avctx->ch_layout, buf, sizeof(buf));
1870 av_log(avctx, AV_LOG_ERROR, "Unsupported channel layout \"%s\"\n", buf);
1871 return AVERROR(EINVAL);
1872 }
1873 av_log(avctx, AV_LOG_INFO, "Using a PCE to encode channel layout \"%s\"\n", buf);
1874 s->pce = aac_pce_configs[i];
1875 s->reorder_map = s->pce.reorder_map;
1876 s->chan_map = s->pce.config_map;
1877 s->needs_height_ext = check_height_ext(avctx, s);
1878 chcfg = 0;
1879 } else {
1880 s->reorder_map = aac_chan_maps[chcfg - 1];
1881 s->chan_map = aac_chan_configs[chcfg - 1];
1882 }
1883
1884 if (!avctx->bit_rate) {
1885 for (i = 1; i <= s->chan_map[0]; i++) {
1886 avctx->bit_rate += s->chan_map[i] == TYPE_CPE ? 128000 : /* Pair */
1887 s->chan_map[i] == TYPE_LFE ? 16000 : /* LFE */
1888 69000 ; /* SCE */
1889 }
1890 }
1891
1892 /* Samplerate */
1893 for (int i = 0;; i++) {
1894 av_assert1(i < 13);
1895 if (avctx->sample_rate == ff_mpeg4audio_sample_rates[i]) {
1896 s->samplerate_index = i;
1897 break;
1898 }
1899 }
1900
1901 /* Bitrate limiting */
1902 WARN_IF(1024.0 * avctx->bit_rate / avctx->sample_rate > 6144 * s->channels,
1903 "Too many bits %f > %d per frame requested, clamping to max\n",
1904 1024.0 * avctx->bit_rate / avctx->sample_rate,
1905 6144 * s->channels);
1906 avctx->bit_rate = (int64_t)FFMIN(6144 * s->channels / 1024.0 * avctx->sample_rate,
1907 avctx->bit_rate);
1908
1909 /* Profile and option setting */
1911 avctx->profile;
1912 for (i = 0; i < FF_ARRAY_ELEMS(aacenc_profiles); i++)
1913 if (avctx->profile == aacenc_profiles[i])
1914 break;
1915 ERROR_IF(i == FF_ARRAY_ELEMS(aacenc_profiles), "Profile not supported!\n");
1916 if (avctx->profile == AV_PROFILE_MPEG2_AAC_LOW) {
1917 avctx->profile = AV_PROFILE_AAC_LOW;
1918 WARN_IF(s->options.pns,
1919 "PNS unavailable in the \"mpeg2_aac_low\" profile, turning off\n");
1920 s->options.pns = 0;
1921 }
1922 s->profile = avctx->profile;
1923
1924 /* Coder limitations */
1925 s->coder = &ff_aac_coders[s->options.coder];
1926
1927 /* M/S introduces horrible artifacts with multichannel files, this is temporary */
1928 if (s->channels > 3)
1929 s->options.mid_side = 0;
1930
1931 /* Coding bandwidth, fixed at init time */
1932 if (avctx->cutoff > 0) {
1933 s->bandwidth = avctx->cutoff;
1934 } else {
1935 int frame_br;
1936 if (avctx->flags & AV_CODEC_FLAG_QSCALE) {
1937 if (s->options.coder == AAC_CODER_NMR) {
1938 /* nd-target VBR: expected per-channel rate from the quality
1939 * ladder (measured: q=1 ~ 64.5 kbps/ch, x1.27 per doubling) */
1940 float q = avctx->global_quality > 0 ?
1941 avctx->global_quality / (float)FF_QP2LAMBDA : 1.0f;
1942 frame_br = 66000 * powf(q, 0.29f);
1943 } else {
1944 frame_br = avctx->bit_rate / 2.0f * (s->lambda / 120.f) * 1.5f;
1945 }
1946 } else {
1947 frame_br = avctx->bit_rate / avctx->ch_layout.nb_channels;
1948 }
1949
1950 if (s->options.coder == AAC_CODER_NMR && frame_br >= 24000) {
1951 /* Ear-tuned, not metric-tuned: Zim rewards HF presence and cannot
1952 * hear HF graininess, so metric sweeps push this table wide. At
1953 * these rates coarse HF reads as beat-synchronous crunch (velvet);
1954 * FDK sits at 14k and Apple at ~16k for 64 kbps/ch. */
1955 static const int rates[] = { 24000, 32000, 48000, 64000, 96000, 192000 };
1956 static const int bws[] = { 14000, 14000, 15000, 16000, 19500, 22000 };
1957 int bw_i = 0;
1958 for (; bw_i < FF_ARRAY_ELEMS(rates) - 2 && frame_br > rates[bw_i + 1]; bw_i++);
1959 s->bandwidth = bws[bw_i] + (int)((int64_t)(bws[bw_i + 1] - bws[bw_i]) *
1960 (frame_br - rates[bw_i]) / (rates[bw_i + 1] - rates[bw_i]));
1961 s->bandwidth = FFMIN3(s->bandwidth, 22000, avctx->sample_rate / 2);
1962 } else {
1963 if (s->options.pns || s->options.intensity_stereo)
1964 frame_br *= 1.15f;
1965 s->bandwidth = FFMAX(3000, AAC_CUTOFF_FROM_BITRATE(frame_br, 1,
1966 avctx->sample_rate));
1967 }
1968
1969 s->bandwidth = FFMIN(FFMAX(s->bandwidth, 8000), avctx->sample_rate / 2);
1970 }
1971
1972 if (!(avctx->flags & AV_CODEC_FLAG_QSCALE) && avctx->bit_rate > 0) {
1973 int bpc = avctx->bit_rate / avctx->ch_layout.nb_channels;
1974 if (bpc <= 32000 && avctx->sample_rate > 32000)
1975 av_log(avctx, AV_LOG_INFO,
1976 "%d kb/s per channel at %d Hz: consider resampling the "
1977 "input to 32000 Hz or lower for better quality.\n",
1978 bpc / 1000, avctx->sample_rate);
1979 }
1980
1981 // Initialize static tables
1983
1984 if ((ret = dsp_init(avctx, s)) < 0)
1985 return ret;
1986
1987 if ((ret = alloc_buffers(avctx, s)) < 0)
1988 return ret;
1989
1990 if ((ret = put_audio_specific_config(avctx, chcfg)))
1991 return ret;
1992
1993 sizes[0] = ff_aac_swb_size_1024[s->samplerate_index];
1994 sizes[1] = ff_aac_swb_size_128[s->samplerate_index];
1995 lengths[0] = ff_aac_num_swb_1024[s->samplerate_index];
1996 lengths[1] = ff_aac_num_swb_128[s->samplerate_index];
1997 for (i = 0; i < s->chan_map[0]; i++)
1998 grouping[i] = s->chan_map[i + 1] == TYPE_CPE;
1999 s->psy.unbounded_pe = ((avctx->flags & AV_CODEC_FLAG_QSCALE) || s->options.rc == 1) &&
2000 s->options.coder == AAC_CODER_NMR;
2001 if ((ret = ff_psy_init(&s->psy, avctx, 2, sizes, lengths,
2002 s->chan_map[0], grouping, s->bandwidth)) < 0)
2003 return ret;
2004
2006 s->random_state = 0x1f2e3d4c;
2007
2008 ff_aacenc_dsp_init(&s->aacdsp);
2009
2010 ff_af_queue_init(avctx, &s->afq);
2011
2012 return 0;
2013}
2014
2015#define AACENC_FLAGS AV_OPT_FLAG_ENCODING_PARAM | AV_OPT_FLAG_AUDIO_PARAM
2016static const AVOption aacenc_options[] = {
2017 {"aac_coder", "Coding algorithm", offsetof(AACEncContext, options.coder), AV_OPT_TYPE_INT, {.i64 = AAC_CODER_NMR}, 0, AAC_CODER_NB-1, AACENC_FLAGS, .unit = "coder"},
2018 {"twoloop", "Two loop searching method", 0, AV_OPT_TYPE_CONST, {.i64 = AAC_CODER_TWOLOOP}, INT_MIN, INT_MAX, AACENC_FLAGS, .unit = "coder"},
2019 {"fast", "Fast search", 0, AV_OPT_TYPE_CONST, {.i64 = AAC_CODER_FAST}, INT_MIN, INT_MAX, AACENC_FLAGS, .unit = "coder"},
2020 {"nmr", "Noise-to-mask ratio scalefactor trellis", 0, AV_OPT_TYPE_CONST, {.i64 = AAC_CODER_NMR}, INT_MIN, INT_MAX, AACENC_FLAGS, .unit = "coder"},
2021 {"aac_ms", "Force M/S stereo coding", offsetof(AACEncContext, options.mid_side), AV_OPT_TYPE_BOOL, {.i64 = -1}, -1, 1, AACENC_FLAGS},
2022 {"aac_is", "Intensity stereo coding", offsetof(AACEncContext, options.intensity_stereo), AV_OPT_TYPE_BOOL, {.i64 = 1}, -1, 1, AACENC_FLAGS},
2023 {"aac_pns", "Perceptual noise substitution", offsetof(AACEncContext, options.pns), AV_OPT_TYPE_BOOL, {.i64 = 1}, -1, 1, AACENC_FLAGS},
2024 {"aac_tns", "Temporal noise shaping", offsetof(AACEncContext, options.tns), AV_OPT_TYPE_BOOL, {.i64 = 1}, -1, 1, AACENC_FLAGS},
2025 {"aac_pce", "Forces the use of PCEs", offsetof(AACEncContext, options.pce), AV_OPT_TYPE_BOOL, {.i64 = 0}, -1, 1, AACENC_FLAGS},
2026 {"aac_rc", "Rate-control mode (NMR coder)", offsetof(AACEncContext, options.rc), AV_OPT_TYPE_INT, {.i64 = 0}, 0, 1, AACENC_FLAGS, .unit = "aac_rc"},
2027 {"cbr", "Constant bitrate (corridor + leaky bucket)", 0, AV_OPT_TYPE_CONST, {.i64 = 0}, INT_MIN, INT_MAX, AACENC_FLAGS, .unit = "aac_rc"},
2028 {"abr", "Average bitrate (constant-quality target, slow rate servo)", 0, AV_OPT_TYPE_CONST, {.i64 = 1}, INT_MIN, INT_MAX, AACENC_FLAGS, .unit = "aac_rc"},
2029 {"aac_nmr_speed", "NMR coder speed level: 0 = slowest/best, higher trades quality for speed", offsetof(AACEncContext, options.nmr_speed), AV_OPT_TYPE_INT, {.i64 = 0}, 0, 4, AACENC_FLAGS},
2030 {"aac_allow_71wide", "Allow non-PCE use of 7.1(wide) channel layout", offsetof(AACEncContext, options.allow_71wide), AV_OPT_TYPE_BOOL, {.i64 = 0}, 0, 1, AACENC_FLAGS},
2032 {NULL}
2033};
2034
2035static const AVClass aacenc_class = {
2036 .class_name = "AAC encoder",
2037 .item_name = av_default_item_name,
2038 .option = aacenc_options,
2039 .version = LIBAVUTIL_VERSION_INT,
2040};
2041
2043 { "b", "0" },
2044 { NULL }
2045};
2046
2048 .p.name = "aac",
2049 CODEC_LONG_NAME("AAC (Advanced Audio Coding)"),
2050 .p.type = AVMEDIA_TYPE_AUDIO,
2051 .p.id = AV_CODEC_ID_AAC,
2052 .p.capabilities = AV_CODEC_CAP_DR1 | AV_CODEC_CAP_DELAY |
2054 .priv_data_size = sizeof(AACEncContext),
2057 .close = aac_encode_end,
2058 .defaults = aac_encode_defaults,
2060 .caps_internal = FF_CODEC_CAP_INIT_CLEANUP,
2062 .p.priv_class = &aacenc_class,
2063};
AAC definitions and structures.
@ EIGHT_SHORT_SEQUENCE
Definition aac.h:66
@ LONG_STOP_SEQUENCE
Definition aac.h:67
@ ONLY_LONG_SEQUENCE
Definition aac.h:64
@ LONG_START_SEQUENCE
Definition aac.h:65
#define NOISE_PRE
preamble for NOISE_BT, put in bitstream with the first noise band
Definition aac.h:99
@ INTENSITY_BT
Scalefactor data are intensity stereo positions (in phase).
Definition aac.h:77
@ INTENSITY_BT2
Scalefactor data are intensity stereo positions (out of phase).
Definition aac.h:76
@ RESERVED_BT
Band types following are encoded differently from others.
Definition aac.h:74
@ NOISE_BT
Spectral data are scaled white noise not coded in the bitstream.
Definition aac.h:75
@ TYPE_CPE
Definition aac.h:45
@ TYPE_SCE
Definition aac.h:44
@ TYPE_FIL
Definition aac.h:50
@ TYPE_LFE
Definition aac.h:47
@ TYPE_END
Definition aac.h:51
#define TNS_MAX_ORDER
Definition aac.h:36
#define NOISE_PRE_BITS
length of preamble
Definition aac.h:100
#define SCALE_DIFF_ZERO
codebook index corresponding to zero scalefactor indices difference
Definition aac.h:95
#define NOISE_OFFSET
subtracted from global gain, used as offset for the preamble
Definition aac.h:101
const AACCoefficientsEncoder ff_aac_coders[AAC_CODER_NB]
Definition aaccoder.c:827
static void put_bitstream_info(AACEncContext *s, const char *name)
Write some auxiliary information about the created AAC file.
Definition aacenc.c:1242
static void avoid_clipping(AACEncContext *s, SingleChannelElement *sce)
Downscale spectral coefficients for near-clipping windows to avoid artifacts.
Definition aacenc.c:1201
static void apply_mid_side_stereo(ChannelElement *cpe)
Definition aacenc.c:1060
static void adjust_frame_information(ChannelElement *cpe, int chans)
Produce integer coefficients from scalefactors provided by the model.
Definition aacenc.c:687
static int encode_individual_channel(AVCodecContext *avctx, AACEncContext *s, SingleChannelElement *sce, int common_window)
Encode one channel of audio data.
Definition aacenc.c:1221
#define NMR_SDEC_EMA
Definition aacenc.c:783
static av_cold int dsp_init(AVCodecContext *avctx, AACEncContext *s)
Definition aacenc.c:1773
static const AVOption aacenc_options[]
Definition aacenc.c:2016
static int put_audio_specific_config(AVCodecContext *avctx, int chcfg)
Make AAC audio config object.
Definition aacenc.c:526
const FFCodec ff_aac_encoder
Definition aacenc.c:2047
static void(*const apply_window[4])(AVFloatDSPContext *fdsp, SingleChannelElement *sce, const float *audio)
Definition aacenc.c:622
static void nmr_apply_ms_band(AACEncContext *s, ChannelElement *cpe, int w, int g, int start, int len, int gl)
Definition aacenc.c:798
#define NMR_MS_BALANCE
Definition aacenc.c:789
static void encode_pulses(AACEncContext *s, Pulse *pulse)
Encode pulse data.
Definition aacenc.c:1154
#define WINDOW_FUNC(type)
Definition aacenc.c:566
static const AVClass aacenc_class
Definition aacenc.c:2035
static av_cold int aac_encode_end(AVCodecContext *avctx)
Definition aacenc.c:1747
static void encode_band_info(AACEncContext *s, SingleChannelElement *sce)
Encode scalefactor band coding type.
Definition aacenc.c:1095
static int nmr_is_image_masked(AACEncContext *s, ChannelElement *cpe, int w, int g, int start, int len, int gl, float ener0, float ener1, float dot, float minthr0, float minthr1, float *ratio_out, float *scale_out, float *sr_out, int *p_out)
Definition aacenc.c:822
#define NMR_MS_MASK
Definition aacenc.c:772
static const AACPCEInfo aac_pce_configs[]
List of PCE (Program Configuration Element) for the channel layouts listed in channel_layout....
Definition aacenc.c:94
static void nmr_decide_stereo(AACEncContext *s, ChannelElement *cpe)
Definition aacenc.c:884
static void apply_window_and_mdct(AACEncContext *s, SingleChannelElement *sce, float *audio)
Definition aacenc.c:631
#define NMR_IS_IMG_GATE
Definition aacenc.c:765
static av_cold int aac_encode_init(AVCodecContext *avctx)
Definition aacenc.c:1823
static void apply_intensity_stereo(ChannelElement *cpe)
Definition aacenc.c:735
static void put_pce(PutBitContext *pb, AVCodecContext *avctx)
Definition aacenc.c:465
#define NMR_IS_LOW_LIMIT
Definition aacenc.c:768
static void encode_ms_info(PutBitContext *pb, ChannelElement *cpe)
Encode MS data.
Definition aacenc.c:673
static void put_ics_info(AACEncContext *s, IndividualChannelStream *info)
Encode ics_info element.
Definition aacenc.c:652
#define AACENC_FLAGS
Definition aacenc.c:2015
static const FFCodecDefault aac_encode_defaults[]
Definition aacenc.c:2042
static void encode_spectral_coeffs(AACEncContext *s, SingleChannelElement *sce)
Encode spectral coefficients processed by psychoacoustic model.
Definition aacenc.c:1173
#define NMR_IS_PERC_FREQ
Definition aacenc.c:793
#define NMR_IS_PERC_CORR
Definition aacenc.c:794
#define NMR_STICKY
Definition aacenc.c:780
static av_cold int check_height_ext(AVCodecContext *avctx, AACEncContext *s)
Definition aacenc.c:1811
static void nmr_apply_is_band(AACEncContext *s, ChannelElement *cpe, int w, int g, int start, int len, int gl, float scale, float sr_, int p, float ener0, float ener1)
Definition aacenc.c:854
static av_cold int alloc_buffers(AVCodecContext *avctx, AACEncContext *s)
Definition aacenc.c:1792
#define NMR_MS_EQUIV
Definition aacenc.c:771
static int aac_encode_frame(AVCodecContext *avctx, AVPacket *avpkt, const AVFrame *frame, int *got_packet_ptr)
Definition aacenc.c:1285
static void copy_input_samples(AACEncContext *s, const AVFrame *frame)
Definition aacenc.c:1263
#define NMR_DECORR_LO
Definition aacenc.c:777
static void encode_scale_factors(AVCodecContext *avctx, AACEncContext *s, SingleChannelElement *sce)
Encode scalefactors.
Definition aacenc.c:1118
#define NMR_PNS_STEREO_DECORR
Definition aacenc.c:786
void ff_quantize_band_cost_cache_init(struct AACEncContext *s)
Definition aacenc.c:557
#define NMR_IS_PERC_GATE
Definition aacenc.c:795
@ AAC_CODER_FAST
Definition aacenc.h:46
@ AAC_CODER_NMR
Definition aacenc.h:47
@ AAC_CODER_NB
Definition aacenc.h:49
@ AAC_CODER_TWOLOOP
Definition aacenc.h:45
#define CLIP_AVOIDANCE_FACTOR
Definition aacenc.h:42
AAC encoder utilities.
#define WARN_IF(cond,...)
#define ERROR_IF(cond,...)
const uint8_t *const ff_aac_swb_size_1024[]
Definition aacenctab.c:97
const uint8_t *const ff_aac_swb_size_128[]
Definition aacenctab.c:89
AAC encoder data.
#define AAC_MAX_CHANNELS
Definition aacenctab.h:41
static const uint8_t aac_chan_maps[14][AAC_MAX_CHANNELS]
Table to remap channels from libavcodec's default order to AAC order.
Definition aacenctab.h:86
static const AVChannelLayout aac_normal_chan_layouts[15]
Definition aacenctab.h:47
static const uint8_t aac_chan_configs[14][6]
default channel configurations
Definition aacenctab.h:66
static const int aacenc_profiles[]
Definition aacenctab.h:145
const uint32_t ff_aac_scalefactor_code[121]
Definition aactab.c:181
const uint8_t ff_tns_max_bands_1024[]
Definition aactab.c:1974
const uint16_t *const ff_swb_offset_128[]
Definition aactab.c:1940
const uint16_t *const ff_swb_offset_1024[]
Definition aactab.c:1900
const uint8_t ff_aac_scalefactor_bits[121]
Definition aactab.c:200
const uint8_t ff_aac_num_swb_1024[]
Definition aactab.c:149
const uint8_t ff_aac_num_swb_128[]
Definition aactab.c:169
const uint8_t ff_tns_max_bands_128[]
Definition aactab.c:1990
AAC data declarations.
float ff_aac_kbd_long_1024[1024]
void ff_aac_float_common_init(void)
float ff_aac_kbd_short_128[128]
static FILE * out
channels
Definition aptx.h:31
#define L(x)
Definition vpx_arith.h:36
static const uint8_t channel_map[8][8]
av_cold void ff_af_queue_close(AudioFrameQueue *afq)
Close AudioFrameQueue.
av_cold void ff_af_queue_init(AVCodecContext *avctx, AudioFrameQueue *afq)
Initialize AudioFrameQueue.
int ff_af_queue_remove(AudioFrameQueue *afq, int nb_samples, AVPacket *pkt)
Remove frame(s) from the queue.
int ff_af_queue_add(AudioFrameQueue *afq, const AVFrame *f)
Add a frame to the queue.
#define av_assert1(cond)
assert() equivalent, that does not lie in speed critical code.
Definition avassert.h:58
#define av_assert0(cond)
assert() equivalent, that is always enabled.
Definition avassert.h:42
Libavcodec external API header.
void ff_copy_bits(PutBitContext *pb, const uint8_t *src, int length)
Copy the content of src to the bitstream.
Definition bitstream.c:49
void ff_put_string(PutBitContext *pb, const char *string, int terminate_string)
Put the string string in the bitstream.
Definition bitstream.c:39
#define i(width, name, range_min, range_max)
Definition cbs_h264.c:63
#define s(width, name)
Definition cbs_vp9.c:198
Public libavutil channel layout APIs header.
#define CODEC_SAMPLERATES_ARRAY(array)
#define FF_CODEC_ENCODE_CB(func)
#define CODEC_LONG_NAME(str)
#define FF_CODEC_CAP_INIT_CLEANUP
The codec allows calling the close function for deallocation even if the init function returned a fai...
#define CODEC_SAMPLEFMTS(...)
#define av_clipf
Definition common.h:145
#define NULL
Definition coverity.c:32
long long int64_t
Definition coverity.c:34
Public header for CRC hash function implementation.
static __device__ float sqrtf(float a)
static __device__ float fabsf(float a)
static __device__ float fabs(float a)
#define max(a, b)
#define AV_PROFILE_MPEG2_AAC_LOW
Definition defs.h:77
#define AV_PROFILE_UNKNOWN
Definition defs.h:65
#define AV_PROFILE_AAC_LOW
Definition defs.h:69
static AVFrame * frame
int(* init)(AVBSFContext *ctx)
Definition dts2pts.c:608
int ff_alloc_packet(AVCodecContext *avctx, AVPacket *avpkt, int64_t size)
Check AVPacket size and allocate data.
Definition encode.c:62
static const uint8_t bits[8]
Definition fastaudio.c:100
@ AV_OPT_TYPE_CONST
Special option type for declaring named constants.
Definition opt.h:298
@ AV_OPT_TYPE_INT
Underlying C type is int.
Definition opt.h:258
@ AV_OPT_TYPE_BOOL
Underlying C type is int.
Definition opt.h:326
#define AV_CH_LAYOUT_5POINT1POINT2_BACK
#define AV_CH_LAYOUT_5POINT1POINT4
#define AV_CH_BOTTOM_FRONT_CENTER
#define AV_CH_TOP_FRONT_CENTER
#define AV_CH_TOP_BACK_RIGHT
#define AV_CH_TOP_CENTER
#define AV_CH_TOP_BACK_LEFT
#define AV_CODEC_FLAG_BITEXACT
Use only bitexact stuff (except (I)DCT).
Definition avcodec.h:322
#define AV_CODEC_CAP_DELAY
Encoder or decoder requires flushing with NULL input at the end in order to give the complete and cor...
Definition codec.h:79
#define AV_CODEC_CAP_DR1
Codec uses get_buffer() or get_encode_buffer() for allocating buffers and supports custom allocators.
Definition codec.h:49
#define AV_CODEC_FLAG_QSCALE
Use fixed qscale.
Definition avcodec.h:213
#define AV_CODEC_CAP_SMALL_LAST_FRAME
Codec can be fed a final frame with a smaller size.
Definition codec.h:84
@ AV_CODEC_ID_AAC
Definition codec_id.h:458
#define AV_PKT_FLAG_KEY
The packet contains a keyframe.
Definition packet.h:650
#define AV_CHANNEL_LAYOUT_4POINT0
#define AV_CHANNEL_LAYOUT_9POINT1POINT4
#define AV_CHANNEL_LAYOUT_7POINT2POINT3
#define AV_CHANNEL_LAYOUT_5POINT1POINT2
#define AV_CHANNEL_LAYOUT_HEXAGONAL
#define AV_CHANNEL_LAYOUT_4POINT1
#define AV_CHANNEL_LAYOUT_3POINT1
#define AV_CHANNEL_LAYOUT_7POINT0_FRONT
#define AV_CHANNEL_LAYOUT_5POINT1_BACK
#define AV_CHANNEL_LAYOUT_7POINT1_WIDE
#define AV_CHANNEL_LAYOUT_7POINT1POINT2
#define AV_CHANNEL_LAYOUT_6POINT1_FRONT
#define AV_CHANNEL_LAYOUT_5POINT0
#define AV_CHANNEL_LAYOUT_AMBISONIC_FIRST_ORDER
#define AV_CHANNEL_LAYOUT_7POINT1POINT4
#define AV_CHANNEL_LAYOUT_5POINT1POINT6
#define AV_CHANNEL_LAYOUT_STEREO
int av_channel_layout_compare(const AVChannelLayout *chl, const AVChannelLayout *chl1)
Check whether two channel layouts are semantically the same, i.e.
#define AV_CHANNEL_LAYOUT_6POINT0
#define AV_CHANNEL_LAYOUT_2_2
#define AV_CHANNEL_LAYOUT_6POINT1_BACK
#define AV_CHANNEL_LAYOUT_5POINT0_BACK
#define AV_CHANNEL_LAYOUT_5POINT1
#define AV_CHANNEL_LAYOUT_MONO
enum AVChannel av_channel_layout_channel_from_index(const AVChannelLayout *channel_layout, unsigned int idx)
Get the channel with the given index in a channel layout.
#define AV_CHANNEL_LAYOUT_7POINT0
AVChannel
#define AV_CHANNEL_LAYOUT_SURROUND
#define AV_CHANNEL_LAYOUT_5POINT1POINT2_BACK
#define AV_CHANNEL_LAYOUT_2_1
#define AV_CHANNEL_LAYOUT_OCTAGONAL
int av_channel_layout_describe(const AVChannelLayout *channel_layout, char *buf, size_t buf_size)
Get a human-readable string describing the channel layout properties.
#define AV_CHANNEL_LAYOUT_7POINT1
#define AV_CHANNEL_LAYOUT_5POINT1POINT4
#define AV_CHANNEL_LAYOUT_7POINT1POINT6
#define AV_CHANNEL_LAYOUT_9POINT1POINT6
#define AV_CHANNEL_LAYOUT_QUAD
#define AV_CHANNEL_LAYOUT_6POINT0_FRONT
#define AV_CHANNEL_LAYOUT_2POINT1
#define AV_CHANNEL_LAYOUT_7POINT1_WIDE_BACK
#define AV_CHANNEL_LAYOUT_6POINT1
@ AV_CHANNEL_ORDER_NATIVE
The native channel order, i.e.
@ AV_CHANNEL_ORDER_AMBISONIC
The audio is represented as the decomposition of the sound field into spherical harmonics.
@ AV_CHAN_TOP_FRONT_LEFT
@ AV_CHAN_TOP_BACK_RIGHT
const AVCRC * av_crc_get_table(AVCRCId crc_id)
Get an initialized standard CRC table.
Definition crc.c:389
uint32_t AVCRC
Definition crc.h:46
uint32_t av_crc(const AVCRC *ctx, uint32_t crc, const uint8_t *buffer, size_t length)
Calculate the CRC of a block.
Definition crc.c:421
@ AV_CRC_8_ATM
Definition crc.h:49
#define FF_QP2LAMBDA
factor to convert from H.263 QP to lambda
Definition avutil.h:226
#define AVERROR(e)
Definition error.h:45
#define AV_LOG_INFO
Standard information.
Definition log.h:221
#define AV_LOG_ERROR
Something went wrong and cannot losslessly be recovered.
Definition log.h:210
const char * av_default_item_name(void *ptr)
Return the context name.
Definition log.c:241
@ AVMEDIA_TYPE_AUDIO
Definition avutil.h:201
@ AV_SAMPLE_FMT_FLTP
float, planar
Definition samplefmt.h:66
#define LIBAVUTIL_VERSION_INT
Definition version.h:85
#define R
Definition huffyuv.h:44
static const int sizes[][2]
Definition img2dec.c:62
#define r
Definition input.c:42
#define b
Definition input.c:43
static void scale(int *out, const int *in, const int w, const int h, const int shift)
Definition intra.c:278
static void put_bits(Jpeg2000EncoderContext *s, int val, int n)
put n times val bit
Definition j2kenc.c:154
void ff_aacenc_dsp_init(AACEncDSPContext *s)
Definition aacencdsp.c:75
av_cold void ff_lpc_end(LPCContext *s)
Uninitialize LPCContext.
Definition lpc.c:367
av_cold int ff_lpc_init(LPCContext *s, int blocksize, int max_order, enum FFLPCType lpc_type)
Initialize LPCContext.
Definition lpc.c:342
Libavcodec version macros.
#define LIBAVCODEC_IDENT
Definition version.h:43
#define av_cold
Definition attributes.h:117
av_cold AVFloatDSPContext * avpriv_float_dsp_alloc(int bit_exact)
Allocate a float DSP context.
Definition float_dsp.c:135
#define FF_ALLOCZ_TYPED_ARRAY(p, nelem)
Definition internal.h:81
uint8_t w
Definition llvidencdsp.c:39
@ FF_LPC_TYPE_LEVINSON
Levinson-Durbin recursion.
Definition lpc.h:46
#define FFMIN(a, b)
Definition macros.h:49
#define FFMAX(a, b)
Definition macros.h:47
#define FFMIN3(a, b, c)
Definition macros.h:50
#define NAN
Memory handling functions.
uint32_t tag
Definition movenc.c:2128
const int ff_mpeg4audio_sample_rates[16]
@ AOT_SBR
Y Spectral Band Replication.
Definition mpeg4audio.h:78
AVOptions.
#define FF_AAC_PROFILE_OPTS
Definition profiles.h:29
av_cold int ff_psy_init(FFPsyContext *ctx, AVCodecContext *avctx, int num_lens, const uint8_t **bands, const int *num_bands, int num_groups, const uint8_t *group_map, int cutoff)
Initialize psychoacoustic model.
Definition psymodel.c:28
av_cold void ff_psy_end(FFPsyContext *ctx)
Cleanup model context at the end.
Definition psymodel.c:77
#define AAC_CUTOFF_FROM_BITRATE(bit_rate, channels, sample_rate)
Definition psymodel.h:35
bitstream writer API
static void init_put_bits(PutBitContext *s, uint8_t *buffer, int buffer_size)
Initialize the PutBitContext s.
Definition put_bits.h:62
static int put_bits_count(PutBitContext *s)
Definition put_bits.h:90
static void flush_put_bits(PutBitContext *s)
Pad the end of the output stream with zeros.
Definition put_bits.h:153
static int put_bytes_output(const PutBitContext *s)
Definition put_bits.h:99
static void align_put_bits(PutBitContext *s)
Pad the bitstream with zeros up to the next byte boundary.
Definition put_bits.h:445
const char * name
Definition qsvenc.c:142
#define FF_ARRAY_ELEMS(a)
AAC encoder context.
Definition aacenc.h:279
uint8_t num_ele[4]
front, side, back, lfe
Definition aacenc.h:268
uint8_t index[4][8]
front, side, back, lfe
Definition aacenc.h:270
uint8_t pairing[3][8]
front, side, back
Definition aacenc.h:269
uint8_t height[3][8]
front, side, back
Definition aacenc.h:271
int nb_channels
Number of channels in this layout.
Describe the class of an AVClass context structure.
Definition log.h:76
main external API structure.
Definition avcodec.h:443
AVChannelLayout ch_layout
Audio channel layout.
Definition avcodec.h:1055
int global_quality
Global quality for codecs which cannot change it per frame.
Definition avcodec.h:1235
int64_t frame_num
Frame counter, set by libavcodec.
Definition avcodec.h:1888
int bit_rate_tolerance
number of bits the bitstream is allowed to diverge from the reference.
Definition avcodec.h:1227
int64_t bit_rate
the average bitrate
Definition avcodec.h:493
int profile
profile
Definition avcodec.h:1641
int initial_padding
Audio only.
Definition avcodec.h:1114
int sample_rate
samples per second
Definition avcodec.h:1040
int flags
AV_CODEC_FLAG_*.
Definition avcodec.h:500
uint8_t * extradata
Out-of-band global headers that may be used by some codecs.
Definition avcodec.h:526
int extradata_size
Definition avcodec.h:527
int cutoff
Audio cutoff bandwidth (0 means "automatic")
Definition avcodec.h:1082
int frame_size
Number of samples per channel in an audio frame.
Definition avcodec.h:1068
void * priv_data
Definition avcodec.h:470
This structure describes decoded (raw) audio or video data.
Definition frame.h:479
AVOption.
Definition opt.h:428
This structure stores compressed data.
Definition packet.h:580
int flags
A combination of AV_PKT_FLAG values.
Definition packet.h:609
int size
Definition packet.h:604
uint8_t * data
Definition packet.h:603
channel element - generic struct for SCE/CPE/CCE/LFE
Definition aacdec.h:296
uint8_t ms_mask[128]
Set if mid/side stereo is used for each scalefactor window band.
Definition aacdec.h:300
SingleChannelElement ch[2]
Definition aacdec.h:302
uint8_t is_mask[128]
Set if intensity stereo is used.
Definition aacenc.h:137
int ms_mode
Signals mid/side stereo flags coding mode.
Definition aacenc.h:134
int common_window
Set if channels share a common 'IndividualChannelStream' in bitstream.
Definition aacenc.h:133
uint8_t is_mode
Set if any bands have been encoded using intensity stereo.
Definition aacenc.h:135
single band psychoacoustic information
Definition psymodel.h:50
windowing related information
Definition psymodel.h:77
int num_windows
number of windows in a frame
Definition psymodel.h:80
int grouping[8]
window grouping (for e.g. AAC)
Definition psymodel.h:81
float clipping[8]
maximum absolute normalized intensity in the given window for clip avoidance
Definition psymodel.h:82
int window_shape
window shape (sine/KBD/whatever)
Definition psymodel.h:79
int window_type[3]
window type (short/long/transitional, etc.) - current, previous and next
Definition psymodel.h:78
Individual Channel Stream.
Definition aacdec.h:169
uint8_t max_sfb
number of scalefactor bands per group
Definition aacdec.h:170
int num_swb
number of scalefactor window bands
Definition aacdec.h:178
uint8_t group_len[8]
Definition aacdec.h:175
uint8_t use_kb_window[2]
If set, use Kaiser-Bessel window, otherwise use a sine window.
Definition aacdec.h:172
float clip_avoidance_factor
set if any window is near clipping to the necessary atennuation factor to avoid it
Definition aacenc.h:92
const uint8_t * swb_sizes
table of scalefactor band sizes for a particular window
Definition aacenc.h:87
enum WindowSequence window_sequence[2]
Definition aacdec.h:171
const uint16_t * swb_offset
table of offsets to the lowest spectral coefficient of a scalefactor band, sfb, for a particular wind...
Definition aacdec.h:177
uint8_t window_clipping[8]
set if a certain window is near clipping
Definition aacdec.h:185
Definition aac.h:103
int pos[4]
Definition aac.h:106
int start
Definition aac.h:105
int amp[4]
Definition aac.h:107
int num_pulse
Definition aac.h:104
Single Channel Element - used for both SCE and LFE elements.
Definition aacdec.h:217
float pcoeffs[1024]
coefficients for IMDCT, pristine
Definition aacenc.h:122
uint8_t zeroes[128]
band is not coded
Definition aacenc.h:118
float coeffs[1024]
coefficients for IMDCT, maybe processed
Definition aacenc.h:123
float is_ener[128]
Intensity stereo pos.
Definition aacenc.h:120
uint8_t can_pns[128]
band is allowed to PNS (informative)
Definition aacenc.h:119
TemporalNoiseShaping tns
Definition aacdec.h:220
float ret_buf[2048]
PCM output buffer.
Definition aacenc.h:124
enum BandType band_type[128]
band types
Definition aacdec.h:221
IndividualChannelStream ics
Definition aacdec.h:218
int sf_idx[128]
scalefactor indices
Definition aacenc.h:117
Temporal Noise Shaping.
Definition aacdec.h:191
#define av_mallocz(s)
#define av_freep(p)
#define av_log(a,...)
static const int rates[]
Definition swresample.c:100
av_cold void av_tx_uninit(AVTXContext **ctx)
Frees a context and sets *ctx to NULL, does nothing when *ctx == NULL.
Definition tx.c:295
av_cold int av_tx_init(AVTXContext **ctx, av_tx_fn *tx, enum AVTXType type, int inv, int len, const void *scale, uint64_t flags)
Initialize a transform context with the given configuration (i)MDCTs with an odd length are currently...
Definition tx.c:903
@ AV_TX_FLOAT_MDCT
Standard MDCT with a sample data type of float, double or int32_t, respectively.
Definition tx.h:68
const char * g
Definition vf_curves.c:128
static av_always_inline int diff(const struct color_info *a, const struct color_info *b, const int trans_thresh)
static double b1(void *priv, double x, double y)
Definition vf_xfade.c:2041
static double b0(void *priv, double x, double y)
Definition vf_xfade.c:2040
int len
static double c[64]