FFmpeg
aacenc.c
Go to the documentation of this file.
1 /*
2  * AAC encoder
3  * Copyright (C) 2008 Konstantin Shishkov
4  *
5  * This file is part of FFmpeg.
6  *
7  * FFmpeg is free software; you can redistribute it and/or
8  * modify it under the terms of the GNU Lesser General Public
9  * License as published by the Free Software Foundation; either
10  * version 2.1 of the License, or (at your option) any later version.
11  *
12  * FFmpeg is distributed in the hope that it will be useful,
13  * but WITHOUT ANY WARRANTY; without even the implied warranty of
14  * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU
15  * Lesser General Public License for more details.
16  *
17  * You should have received a copy of the GNU Lesser General Public
18  * License along with FFmpeg; if not, write to the Free Software
19  * Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA
20  */
21 
22 /**
23  * @file
24  * AAC encoder
25  */
26 
27 /***********************************
28  * TODOs:
29  * add sane pulse detection
30  ***********************************/
31 #include <float.h>
32 
34 #include "libavutil/libm.h"
35 #include "libavutil/float_dsp.h"
36 #include "libavutil/mem.h"
37 #include "libavutil/opt.h"
38 #include "avcodec.h"
39 #include "codec_internal.h"
40 #include "encode.h"
41 #include "put_bits.h"
42 #include "mpeg4audio.h"
43 #include "sinewin.h"
44 #include "profiles.h"
45 #include "version.h"
46 
47 #include "aac.h"
48 #include "aactab.h"
49 #include "aacenc.h"
50 #include "aacenctab.h"
51 #include "aacenc_utils.h"
52 
53 #include "psymodel.h"
54 
55 /**
56  * List of PCE (Program Configuration Element) for the channel layouts listed
57  * in channel_layout.h
58  *
59  * For those wishing in the future to add other layouts:
60  *
61  * - num_ele: number of elements in each group of front, side, back, lfe channels
62  * (an element is of type SCE (single channel), CPE (channel pair) for
63  * the first 3 groups; and is LFE for LFE group).
64  *
65  * - pairing: 0 for an SCE element or 1 for a CPE; does not apply to LFE group
66  *
67  * - index: there are three independent indices for SCE, CPE and LFE;
68  * they are incremented irrespective of the group to which the element belongs;
69  * they are not reset when going from one group to another
70  *
71  * Example: for 7.0 channel layout,
72  * .pairing = { { 1, 0 }, { 1 }, { 1 }, }, (3 CPE and 1 SCE in front group)
73  * .index = { { 0, 0 }, { 1 }, { 2 }, },
74  * (index is 0 for the single SCE but goes from 0 to 2 for the CPEs)
75  *
76  * The index order impacts the channel ordering. But is otherwise arbitrary
77  * (the sequence could have been 2, 0, 1 instead of 0, 1, 2).
78  *
79  * Spec allows for discontinuous indices, e.g. if one has a total of two SCE,
80  * SCE.0 SCE.15 is OK per spec; BUT it won't be decoded by our AAC decoder
81  * which at this time requires that indices fully cover some range starting
82  * from 0 (SCE.1 SCE.0 is OK but not SCE.0 SCE.15).
83  *
84  * - config_map: total number of elements and their types. Beware, the way the
85  * types are ordered impacts the final channel ordering.
86  *
87  * - reorder_map: reorders the channels.
88  *
89  */
90 static const AACPCEInfo aac_pce_configs[] = {
91  {
93  .num_ele = { 1, 0, 0, 0 },
94  .pairing = { { 0 }, },
95  .index = { { 0 }, },
96  .config_map = { 1, TYPE_SCE, },
97  .reorder_map = { 0 },
98  },
99  {
100  .layout = AV_CHANNEL_LAYOUT_STEREO,
101  .num_ele = { 1, 0, 0, 0 },
102  .pairing = { { 1 }, },
103  .index = { { 0 }, },
104  .config_map = { 1, TYPE_CPE, },
105  .reorder_map = { 0, 1 },
106  },
107  {
108  .layout = AV_CHANNEL_LAYOUT_2POINT1,
109  .num_ele = { 1, 0, 0, 1 },
110  .pairing = { { 1 }, },
111  .index = { { 0 },{ 0 },{ 0 },{ 0 } },
112  .config_map = { 2, TYPE_CPE, TYPE_LFE },
113  .reorder_map = { 0, 1, 2 },
114  },
115  {
116  .layout = AV_CHANNEL_LAYOUT_2_1,
117  .num_ele = { 1, 0, 1, 0 },
118  .pairing = { { 1 },{ 0 },{ 0 } },
119  .index = { { 0 },{ 0 },{ 0 }, },
120  .config_map = { 2, TYPE_CPE, TYPE_SCE },
121  .reorder_map = { 0, 1, 2 },
122  },
123  {
124  .layout = AV_CHANNEL_LAYOUT_SURROUND,
125  .num_ele = { 2, 0, 0, 0 },
126  .pairing = { { 0, 1 }, },
127  .index = { { 0, 0 }, },
128  .config_map = { 2, TYPE_SCE, TYPE_CPE },
129  .reorder_map = { 2, 0, 1 },
130  },
131  {
132  .layout = AV_CHANNEL_LAYOUT_3POINT1,
133  .num_ele = { 2, 0, 0, 1 },
134  .pairing = { { 0, 1 }, },
135  .index = { { 0, 0 }, { 0 }, { 0 }, { 0 }, },
136  .config_map = { 3, TYPE_SCE, TYPE_CPE, TYPE_LFE },
137  .reorder_map = { 2, 0, 1, 3 },
138  },
139  {
140  .layout = AV_CHANNEL_LAYOUT_4POINT0,
141  .num_ele = { 2, 0, 1, 0 },
142  .pairing = { { 0, 1 }, { 0 }, { 0 }, },
143  .index = { { 0, 0 }, { 0 }, { 1 } },
144  .config_map = { 3, TYPE_SCE, TYPE_CPE, TYPE_SCE },
145  .reorder_map = { 2, 0, 1, 3 },
146  },
147  {
148  .layout = AV_CHANNEL_LAYOUT_4POINT1,
149  .num_ele = { 2, 0, 1, 1 },
150  .pairing = { { 0, 1 }, { 0 }, { 0 }, },
151  .index = { { 0, 0 }, { 0 }, { 1 }, { 0 } },
152  .config_map = { 4, TYPE_SCE, TYPE_CPE, TYPE_SCE, TYPE_LFE },
153  .reorder_map = { 2, 0, 1, 4, 3 },
154  },
155  {
156  .layout = AV_CHANNEL_LAYOUT_2_2,
157  .num_ele = { 1, 0, 1, 0 },
158  .pairing = { { 1 }, { 0 }, { 1 }, },
159  .index = { { 0 }, { 0 }, { 1 } },
160  .config_map = { 2, TYPE_CPE, TYPE_CPE },
161  .reorder_map = { 0, 1, 2, 3 },
162  },
163  {
164  .layout = AV_CHANNEL_LAYOUT_QUAD,
165  .num_ele = { 1, 0, 1, 0 },
166  .pairing = { { 1 }, { 0 }, { 1 }, },
167  .index = { { 0 }, { 0 }, { 1 } },
168  .config_map = { 2, TYPE_CPE, TYPE_CPE },
169  .reorder_map = { 0, 1, 2, 3 },
170  },
171  {
172  .layout = AV_CHANNEL_LAYOUT_5POINT0,
173  .num_ele = { 2, 0, 1, 0 },
174  .pairing = { { 0, 1 }, { 0 }, { 1 } },
175  .index = { { 0, 0 }, { 0 }, { 1 } },
176  .config_map = { 3, TYPE_SCE, TYPE_CPE, TYPE_CPE },
177  .reorder_map = { 2, 0, 1, 3, 4 },
178  },
179  {
180  .layout = AV_CHANNEL_LAYOUT_5POINT1,
181  .num_ele = { 2, 0, 1, 1 },
182  .pairing = { { 0, 1 }, { 0 }, { 1 }, },
183  .index = { { 0, 0 }, { 0 }, { 1 }, { 0 } },
184  .config_map = { 4, TYPE_SCE, TYPE_CPE, TYPE_CPE, TYPE_LFE },
185  .reorder_map = { 2, 0, 1, 4, 5, 3 },
186  },
187  {
189  .num_ele = { 2, 0, 1, 0 },
190  .pairing = { { 0, 1 }, { 0 }, { 1 } },
191  .index = { { 0, 0 }, { 0 }, { 1 } },
192  .config_map = { 3, TYPE_SCE, TYPE_CPE, TYPE_CPE },
193  .reorder_map = { 2, 0, 1, 3, 4 },
194  },
195  {
197  .num_ele = { 2, 0, 1, 1 },
198  .pairing = { { 0, 1 }, { 0 }, { 1 }, },
199  .index = { { 0, 0 }, { 0 }, { 1 }, { 0 } },
200  .config_map = { 4, TYPE_SCE, TYPE_CPE, TYPE_CPE, TYPE_LFE },
201  .reorder_map = { 2, 0, 1, 4, 5, 3 },
202  },
203  {
204  .layout = AV_CHANNEL_LAYOUT_6POINT0,
205  .num_ele = { 2, 0, 2, 0 },
206  .pairing = { { 0, 1 }, { 0 }, { 1, 0 } },
207  .index = { { 0, 0 }, { 0 }, { 1, 1 } },
208  .config_map = { 4, TYPE_SCE, TYPE_CPE, TYPE_CPE, TYPE_SCE },
209  .reorder_map = { 2, 0, 1, 4, 5, 3 },
210  },
211  {
213  .num_ele = { 2, 0, 1, 0 },
214  .pairing = { { 1, 1 }, { 0 }, { 1 } },
215  .index = { { 0, 1 }, { 0 }, { 2 }, },
216  .config_map = { 3, TYPE_CPE, TYPE_CPE, TYPE_CPE, },
217  .reorder_map = { 2, 3, 0, 1, 4, 5 },
218  },
219  {
220  .layout = AV_CHANNEL_LAYOUT_HEXAGONAL,
221  .num_ele = { 2, 0, 2, 0 },
222  .pairing = { { 0, 1 }, { 0 }, { 1, 0 } },
223  .index = { { 0, 0 }, { 0 }, { 1, 1 } },
224  .config_map = { 4, TYPE_SCE, TYPE_CPE, TYPE_CPE, TYPE_SCE },
225  .reorder_map = { 2, 0, 1, 3, 4, 5 },
226  },
227  {
228  .layout = AV_CHANNEL_LAYOUT_6POINT1,
229  .num_ele = { 2, 0, 2, 1 },
230  .pairing = { { 0, 1 }, { 0 }, { 1, 0 }, },
231  .index = { { 0, 0 }, { 0 }, { 1, 1 }, { 0 } },
232  .config_map = { 5, TYPE_SCE, TYPE_CPE, TYPE_CPE, TYPE_SCE, TYPE_LFE },
233  .reorder_map = { 2, 0, 1, 5, 6, 4, 3 },
234  },
235  {
237  .num_ele = { 2, 0, 2, 1 },
238  .pairing = { { 0, 1 },{ 0 },{ 1, 0 }, },
239  .index = { { 0, 0 },{ 0 },{ 1, 1 },{ 0 } },
240  .config_map = { 5, TYPE_SCE, TYPE_CPE, TYPE_CPE, TYPE_SCE, TYPE_LFE },
241  .reorder_map = { 2, 0, 1, 4, 5, 6, 3 },
242  },
243  {
245  .num_ele = { 2, 0, 1, 1 },
246  .pairing = { { 1, 1 }, { 0 }, { 1 }, },
247  .index = { { 0, 1 }, { 0 }, { 2 }, { 0 }, },
248  .config_map = { 4, TYPE_CPE, TYPE_CPE, TYPE_CPE, TYPE_LFE, },
249  .reorder_map = { 3, 4, 0, 1, 5, 6, 2 },
250  },
251  {
252  .layout = AV_CHANNEL_LAYOUT_7POINT0,
253  .num_ele = { 2, 0, 2, 0 },
254  .pairing = { { 0, 1 }, { 0 }, { 1, 1 }, },
255  .index = { { 0, 0 }, { 0 }, { 2, 1 }, },
256  .config_map = { 4, TYPE_SCE, TYPE_CPE, TYPE_CPE, TYPE_CPE },
257  .reorder_map = { 2, 0, 1, 3, 4, 5, 6 },
258  },
259  {
261  .num_ele = { 3, 0, 1, 0 },
262  .pairing = { { 0, 1, 1 }, { 0 }, { 1 }, },
263  .index = { { 0, 0, 1 }, { 0 }, { 2 }, },
264  .config_map = { 4, TYPE_SCE, TYPE_CPE, TYPE_CPE, TYPE_CPE },
265  .reorder_map = { 2, 3, 4, 0, 1, 5, 6 },
266  },
267  {
268  .layout = AV_CHANNEL_LAYOUT_7POINT1,
269  .num_ele = { 2, 0, 2, 1 },
270  .pairing = { { 0, 1 }, { 0 }, { 1, 1 }, },
271  .index = { { 0, 0 }, { 0 }, { 2, 1 }, { 0 } },
272  .config_map = { 5, TYPE_SCE, TYPE_CPE, TYPE_CPE, TYPE_CPE, TYPE_LFE },
273  .reorder_map = { 2, 0, 1, 4, 5, 6, 7, 3 },
274  },
275  {
277  .num_ele = { 3, 0, 1, 1 },
278  .pairing = { { 0, 1, 1 }, { 0 }, { 1 }, },
279  .index = { { 0, 0, 1 }, { 0 }, { 2 }, { 0 }, },
280  .config_map = { 5, TYPE_SCE, TYPE_CPE, TYPE_CPE, TYPE_CPE, TYPE_LFE },
281  .reorder_map = { 2, 4, 5, 0, 1, 6, 7, 3 },
282  },
283  {
285  .num_ele = { 3, 0, 1, 1 },
286  .pairing = { { 0, 1, 1 }, { 0 }, { 1 } },
287  .index = { { 0, 0, 1 }, { 0 }, { 2 }, { 0 } },
288  .config_map = { 5, TYPE_SCE, TYPE_CPE, TYPE_CPE, TYPE_CPE, TYPE_LFE },
289  .reorder_map = { 2, 6, 7, 0, 1, 4, 5, 3 },
290  },
291  {
292  .layout = AV_CHANNEL_LAYOUT_OCTAGONAL,
293  .num_ele = { 2, 0, 3, 0 },
294  .pairing = { { 0, 1 }, { 0 }, { 1, 1, 0 }, },
295  .index = { { 0, 0 }, { 0 }, { 1, 2, 1 }, },
296  .config_map = { 5, TYPE_SCE, TYPE_CPE, TYPE_CPE, TYPE_CPE, TYPE_SCE },
297  .reorder_map = { 2, 0, 1, 6, 7, 3, 4, 5 },
298  },
299 };
300 
301 static void put_pce(PutBitContext *pb, AVCodecContext *avctx)
302 {
303  int i, j;
304  AACEncContext *s = avctx->priv_data;
305  AACPCEInfo *pce = &s->pce;
306  const int bitexact = avctx->flags & AV_CODEC_FLAG_BITEXACT;
307  const char *aux_data = bitexact ? "Lavc" : LIBAVCODEC_IDENT;
308 
309  put_bits(pb, 4, 0);
310 
311  put_bits(pb, 2, avctx->profile);
312  put_bits(pb, 4, s->samplerate_index);
313 
314  put_bits(pb, 4, pce->num_ele[0]); /* Front */
315  put_bits(pb, 4, pce->num_ele[1]); /* Side */
316  put_bits(pb, 4, pce->num_ele[2]); /* Back */
317  put_bits(pb, 2, pce->num_ele[3]); /* LFE */
318  put_bits(pb, 3, 0); /* Assoc data */
319  put_bits(pb, 4, 0); /* CCs */
320 
321  put_bits(pb, 1, 0); /* Stereo mixdown */
322  put_bits(pb, 1, 0); /* Mono mixdown */
323  put_bits(pb, 1, 0); /* Something else */
324 
325  for (i = 0; i < 4; i++) {
326  for (j = 0; j < pce->num_ele[i]; j++) {
327  if (i < 3)
328  put_bits(pb, 1, pce->pairing[i][j]);
329  put_bits(pb, 4, pce->index[i][j]);
330  }
331  }
332 
333  align_put_bits(pb);
334  put_bits(pb, 8, strlen(aux_data));
335  ff_put_string(pb, aux_data, 0);
336 }
337 
338 /**
339  * Make AAC audio config object.
340  * @see 1.6.2.1 "Syntax - AudioSpecificConfig"
341  */
342 static int put_audio_specific_config(AVCodecContext *avctx, int chcfg)
343 {
344  PutBitContext pb;
345  AACEncContext *s = avctx->priv_data;
346  const int max_size = 32;
347 
348  avctx->extradata = av_mallocz(max_size);
349  if (!avctx->extradata)
350  return AVERROR(ENOMEM);
351 
352  init_put_bits(&pb, avctx->extradata, max_size);
353  put_bits(&pb, 5, s->profile+1); //profile
354  put_bits(&pb, 4, s->samplerate_index); //sample rate index
355  put_bits(&pb, 4, chcfg);
356  //GASpecificConfig
357  put_bits(&pb, 1, 0); //frame length - 1024 samples
358  put_bits(&pb, 1, 0); //does not depend on core coder
359  put_bits(&pb, 1, 0); //is not extension
360  if (s->needs_pce)
361  put_pce(&pb, avctx);
362 
363  //Explicitly Mark SBR absent
364  put_bits(&pb, 11, 0x2b7); //sync extension
365  put_bits(&pb, 5, AOT_SBR);
366  put_bits(&pb, 1, 0);
367  flush_put_bits(&pb);
368  avctx->extradata_size = put_bytes_output(&pb);
369 
370  return 0;
371 }
372 
374 {
375  ++s->quantize_band_cost_cache_generation;
376  if (s->quantize_band_cost_cache_generation == 0) {
377  memset(s->quantize_band_cost_cache, 0, sizeof(s->quantize_band_cost_cache));
378  s->quantize_band_cost_cache_generation = 1;
379  }
380 }
381 
382 #define WINDOW_FUNC(type) \
383 static void apply_ ##type ##_window(AVFloatDSPContext *fdsp, \
384  SingleChannelElement *sce, \
385  const float *audio)
386 
387 WINDOW_FUNC(only_long)
388 {
389  const float *lwindow = sce->ics.use_kb_window[0] ? ff_aac_kbd_long_1024 : ff_sine_1024;
390  const float *pwindow = sce->ics.use_kb_window[1] ? ff_aac_kbd_long_1024 : ff_sine_1024;
391  float *out = sce->ret_buf;
392 
393  fdsp->vector_fmul (out, audio, lwindow, 1024);
394  fdsp->vector_fmul_reverse(out + 1024, audio + 1024, pwindow, 1024);
395 }
396 
397 WINDOW_FUNC(long_start)
398 {
399  const float *lwindow = sce->ics.use_kb_window[1] ? ff_aac_kbd_long_1024 : ff_sine_1024;
400  const float *swindow = sce->ics.use_kb_window[0] ? ff_aac_kbd_short_128 : ff_sine_128;
401  float *out = sce->ret_buf;
402 
403  fdsp->vector_fmul(out, audio, lwindow, 1024);
404  memcpy(out + 1024, audio + 1024, sizeof(out[0]) * 448);
405  fdsp->vector_fmul_reverse(out + 1024 + 448, audio + 1024 + 448, swindow, 128);
406  memset(out + 1024 + 576, 0, sizeof(out[0]) * 448);
407 }
408 
409 WINDOW_FUNC(long_stop)
410 {
411  const float *lwindow = sce->ics.use_kb_window[0] ? ff_aac_kbd_long_1024 : ff_sine_1024;
412  const float *swindow = sce->ics.use_kb_window[1] ? ff_aac_kbd_short_128 : ff_sine_128;
413  float *out = sce->ret_buf;
414 
415  memset(out, 0, sizeof(out[0]) * 448);
416  fdsp->vector_fmul(out + 448, audio + 448, swindow, 128);
417  memcpy(out + 576, audio + 576, sizeof(out[0]) * 448);
418  fdsp->vector_fmul_reverse(out + 1024, audio + 1024, lwindow, 1024);
419 }
420 
421 WINDOW_FUNC(eight_short)
422 {
423  const float *swindow = sce->ics.use_kb_window[0] ? ff_aac_kbd_short_128 : ff_sine_128;
424  const float *pwindow = sce->ics.use_kb_window[1] ? ff_aac_kbd_short_128 : ff_sine_128;
425  const float *in = audio + 448;
426  float *out = sce->ret_buf;
427  int w;
428 
429  for (w = 0; w < 8; w++) {
430  fdsp->vector_fmul (out, in, w ? pwindow : swindow, 128);
431  out += 128;
432  in += 128;
433  fdsp->vector_fmul_reverse(out, in, swindow, 128);
434  out += 128;
435  }
436 }
437 
438 static void (*const apply_window[4])(AVFloatDSPContext *fdsp,
440  const float *audio) = {
441  [ONLY_LONG_SEQUENCE] = apply_only_long_window,
442  [LONG_START_SEQUENCE] = apply_long_start_window,
443  [EIGHT_SHORT_SEQUENCE] = apply_eight_short_window,
444  [LONG_STOP_SEQUENCE] = apply_long_stop_window
445 };
446 
448  float *audio)
449 {
450  int i;
451  float *output = sce->ret_buf;
452 
453  apply_window[sce->ics.window_sequence[0]](s->fdsp, sce, audio);
454 
456  s->mdct1024_fn(s->mdct1024, sce->coeffs, output, sizeof(float));
457  else
458  for (i = 0; i < 1024; i += 128)
459  s->mdct128_fn(s->mdct128, &sce->coeffs[i], output + i*2, sizeof(float));
460  memcpy(audio, audio + 1024, sizeof(audio[0]) * 1024);
461  memcpy(sce->pcoeffs, sce->coeffs, sizeof(sce->pcoeffs));
462 }
463 
464 /**
465  * Encode ics_info element.
466  * @see Table 4.6 (syntax of ics_info)
467  */
469 {
470  int w;
471 
472  put_bits(&s->pb, 1, 0); // ics_reserved bit
473  put_bits(&s->pb, 2, info->window_sequence[0]);
474  put_bits(&s->pb, 1, info->use_kb_window[0]);
475  if (info->window_sequence[0] != EIGHT_SHORT_SEQUENCE) {
476  put_bits(&s->pb, 6, info->max_sfb);
477  put_bits(&s->pb, 1, 0); /* No predictor present */
478  } else {
479  put_bits(&s->pb, 4, info->max_sfb);
480  for (w = 1; w < 8; w++)
481  put_bits(&s->pb, 1, !info->group_len[w]);
482  }
483 }
484 
485 /**
486  * Encode MS data.
487  * @see 4.6.8.1 "Joint Coding - M/S Stereo"
488  */
490 {
491  int i, w;
492 
493  put_bits(pb, 2, cpe->ms_mode);
494  if (cpe->ms_mode == 1)
495  for (w = 0; w < cpe->ch[0].ics.num_windows; w += cpe->ch[0].ics.group_len[w])
496  for (i = 0; i < cpe->ch[0].ics.max_sfb; i++)
497  put_bits(pb, 1, cpe->ms_mask[w*16 + i]);
498 }
499 
500 /**
501  * Produce integer coefficients from scalefactors provided by the model.
502  */
503 static void adjust_frame_information(ChannelElement *cpe, int chans)
504 {
505  int i, w, w2, g, ch;
506  int maxsfb, cmaxsfb;
507 
508  for (ch = 0; ch < chans; ch++) {
509  IndividualChannelStream *ics = &cpe->ch[ch].ics;
510  maxsfb = 0;
511  cpe->ch[ch].pulse.num_pulse = 0;
512  for (w = 0; w < ics->num_windows; w += ics->group_len[w]) {
513  for (cmaxsfb = ics->num_swb; cmaxsfb > 0 && cpe->ch[ch].zeroes[w*16+cmaxsfb-1]; cmaxsfb--)
514  ;
515  maxsfb = FFMAX(maxsfb, cmaxsfb);
516  }
517  ics->max_sfb = maxsfb;
518 
519  //adjust zero bands for window groups
520  for (w = 0; w < ics->num_windows; w += ics->group_len[w]) {
521  for (g = 0; g < ics->max_sfb; g++) {
522  i = 1;
523  for (w2 = w; w2 < w + ics->group_len[w]; w2++) {
524  if (!cpe->ch[ch].zeroes[w2*16 + g]) {
525  i = 0;
526  break;
527  }
528  }
529  cpe->ch[ch].zeroes[w*16 + g] = i;
530  }
531  }
532  }
533 
534  if (chans > 1 && cpe->common_window) {
535  IndividualChannelStream *ics0 = &cpe->ch[0].ics;
536  IndividualChannelStream *ics1 = &cpe->ch[1].ics;
537  int msc = 0;
538  ics0->max_sfb = FFMAX(ics0->max_sfb, ics1->max_sfb);
539  ics1->max_sfb = ics0->max_sfb;
540  for (w = 0; w < ics0->num_windows*16; w += 16)
541  for (i = 0; i < ics0->max_sfb; i++)
542  if (cpe->ms_mask[w+i])
543  msc++;
544  if (msc == 0 || ics0->max_sfb == 0)
545  cpe->ms_mode = 0;
546  else
547  cpe->ms_mode = msc < ics0->max_sfb * ics0->num_windows ? 1 : 2;
548  }
549 }
550 
552 {
553  int w, w2, g, i;
554  IndividualChannelStream *ics = &cpe->ch[0].ics;
555  if (!cpe->common_window)
556  return;
557  for (w = 0; w < ics->num_windows; w += ics->group_len[w]) {
558  for (w2 = 0; w2 < ics->group_len[w]; w2++) {
559  int start = (w+w2) * 128;
560  for (g = 0; g < ics->num_swb; g++) {
561  int p = -1 + 2 * (cpe->ch[1].band_type[w*16+g] - 14);
562  float scale = cpe->ch[0].is_ener[w*16+g];
563  if (!cpe->is_mask[w*16 + g]) {
564  start += ics->swb_sizes[g];
565  continue;
566  }
567  if (cpe->ms_mask[w*16 + g])
568  p *= -1;
569  for (i = 0; i < ics->swb_sizes[g]; i++) {
570  float sum = (cpe->ch[0].coeffs[start+i] + p*cpe->ch[1].coeffs[start+i])*scale;
571  cpe->ch[0].coeffs[start+i] = sum;
572  cpe->ch[1].coeffs[start+i] = 0.0f;
573  }
574  start += ics->swb_sizes[g];
575  }
576  }
577  }
578 }
579 
580 /* I/S acceptance level for the image-error EMA at full rate pressure */
581 #define NMR_IS_IMG_GATE 8000.0f
582 
583 /* Frequency in Hz for the lower limit of intensity stereo */
584 #define NMR_IS_LOW_LIMIT 6100
585 
586 /* M/S adoption: es < 0.5*em, content-driven and rate-free */
587 #define NMR_MS_EQUIV 0.5f
588 #define NMR_MS_MASK 0.0f
589 
590 /* Pair decouple threshold on the joint-tool candidacy fraction EMA: pairs
591  * whose joint tools are mostly dead (diffuse decorrelated content) window
592  * per-channel and skip M/S; recouple above 1.3x. */
593 #define NMR_DECORR_LO 0.20f
594 
595 /* Stereo-decision hysteresis: leaving a joint mode costs a margin. */
596 #define NMR_STICKY 2.0f
597 
598 /* Decision statistics are EMA-smoothed across frames. */
599 #define NMR_SDEC_EMA 0.75f
600 
601 /* PNS-stereo gate: substitute only clearly-decorrelated (wide) bands. */
602 #define NMR_PNS_STEREO_DECORR 0.6f
603 
604 /* Recode one band's window group as mid+side in place. */
606  int w, int g, int start, int len, int gl)
607 {
608  SingleChannelElement *sce0 = &cpe->ch[0];
609  SingleChannelElement *sce1 = &cpe->ch[1];
610  cpe->ms_mask[w*16+g] = 1;
611  for (int w2 = 0; w2 < gl; w2++) {
612  FFPsyBand *b0 = &s->psy.ch[s->cur_channel+0].psy_bands[(w+w2)*16+g];
613  FFPsyBand *b1 = &s->psy.ch[s->cur_channel+1].psy_bands[(w+w2)*16+g];
614  float *L = sce0->coeffs + start + (w+w2)*128;
615  float *R = sce1->coeffs + start + (w+w2)*128;
616  float em = 0.0f, es = 0.0f;
617  for (int i = 0; i < len; i++) {
618  float m = (L[i] + R[i]) * 0.5f;
619  R[i] = m - R[i]; L[i] = m;
620  em += L[i]*L[i]; es += R[i]*R[i];
621  }
622  b0->threshold = FFMIN(b0->threshold, b1->threshold) * 0.5f;
623  b1->threshold = b0->threshold;
624  b0->energy = em; b1->energy = es;
625  }
626 }
627 
628 /* I/S perceptual test: reconstruction image error vs the pair's masks. */
630  int w, int g, int start, int len, int gl,
631  float ener0, float ener1, float dot,
632  float minthr0, float minthr1, float *ratio_out,
633  float *scale_out, float *sr_out, int *p_out)
634 {
635  int p = dot >= 0.0f ? 1 : -1;
636  float ener01 = ener0 + ener1 + 2*p*dot; /* energy of L + p*R */
637  *ratio_out = FLT_MAX;
638  if (ener01 <= FLT_MIN)
639  return 0;
640  float scale = sqrtf(ener0 / ener01); /* carrier = (L + p*R)*scale */
641  float sr_ = sqrtf(ener1 / ener0); /* decoder: R = p*sr_*carrier */
642  float img0 = 0.0f, img1 = 0.0f;
643  for (int w2 = 0; w2 < gl; w2++) {
644  const float *L = cpe->ch[0].coeffs + start + (w+w2)*128;
645  const float *R = cpe->ch[1].coeffs + start + (w+w2)*128;
646  for (int i = 0; i < len; i++) {
647  float c = (L[i] + p*R[i]) * scale;
648  float dl = L[i] - c, dr = R[i] - p*sr_*c;
649  img0 += dl*dl; img1 += dr*dr;
650  }
651  }
652  *ratio_out = FFMAX(img0 / FFMAX(minthr0 * gl, FLT_MIN),
653  img1 / FFMAX(minthr1 * gl, FLT_MIN));
654  *scale_out = scale; *sr_out = sr_; *p_out = p;
655  return 1;
656 }
657 
658 /* Recode one band's window group as intensity stereo in place: replace L with the
659  * carrier, zero R, signal the phase via the side channel's band type, and fold the
660  * pair's masking into the surviving (carrier) channel. */
662  int w, int g, int start, int len, int gl,
663  float scale, float sr_, int p,
664  float ener0, float ener1)
665 {
666  cpe->is_mask[w*16+g] = 1;
667  cpe->ch[0].is_ener[w*16+g] = scale;
668  cpe->ch[1].is_ener[w*16+g] = ener0 / ener1;
669  cpe->ch[1].band_type[w*16+g] = p > 0 ? INTENSITY_BT : INTENSITY_BT2;
670  for (int w2 = 0; w2 < gl; w2++) {
671  FFPsyBand *b0 = &s->psy.ch[s->cur_channel+0].psy_bands[(w+w2)*16+g];
672  FFPsyBand *b1 = &s->psy.ch[s->cur_channel+1].psy_bands[(w+w2)*16+g];
673  float *L = cpe->ch[0].coeffs + start + (w+w2)*128;
674  float *R = cpe->ch[1].coeffs + start + (w+w2)*128;
675  float ec = 0.0f;
676  for (int i = 0; i < len; i++) {
677  L[i] = (L[i] + p*R[i]) * scale;
678  R[i] = 0.0f;
679  ec += L[i]*L[i];
680  }
681  b0->threshold = FFMIN(b0->threshold, b1->threshold / FFMAX(sr_*sr_, 1e-9f));
682  b0->energy = ec; b1->energy = 0.0f;
683  }
684 }
685 
686 /*
687  * Per-band stereo-mode decision (L/R vs M/S vs intensity) for the NMR coder,
688  * made before quantization from the psychoacoustic model alone, so the
689  * quantizer search allocates natively on the spectra that are actually coded.
690  */
692 {
693  SingleChannelElement *sce0 = &cpe->ch[0];
694  SingleChannelElement *sce1 = &cpe->ch[1];
695  IndividualChannelStream *ics = &sce0->ics;
696  const AVCodecContext *avctx = s->psy.avctx;
697  const float freq_mult = avctx->sample_rate / (1024.0f / ics->num_windows) / 2.0f;
698  int is_count = 0;
699 
700  if (s->nmr) {
701  int pi = (s->cur_channel >> 1) & 7;
702  pi = pi * 2 + (ics->num_windows == 8); /* per-grid state bank */
703  if (!s->nmr->sinit[pi]) {
704  /* one-time init; per-grid banks persist across window switches
705  * (wiping them churned stereo modes audibly) */
706  memset(s->nmr->smode[pi], 0, sizeof(s->nmr->smode[pi]));
707  for (int b = 0; b < 128; b++) {
708  s->nmr->sema_em[pi][b] = 0.0f;
709  s->nmr->sema_img[pi][b] = -1.0f;
710  }
711  s->nmr->sinit[pi] = 1;
712  }
713  }
714 
715  /* Per-band stereo decision (L/R vs M/S vs I/S), made pre-quantization from
716  * the psy model so the trellis allocates on the coded spectra. */
717 
718  /* I/S engages under SUSTAINED strain only: rate pressure gated by the
719  * lambda floor (pressure spikes at a comfortable operating point must
720  * not admit it). Unengaged candidates fall back to M/S. */
721  float is_ramp = s->nmr ? s->nmr->press *
722  av_clipf((s->nmr->lam_floor - 40.0f) / (120.0f - 40.0f), 0.0f, 1.0f) : 0.0f;
723  const int allow_is = s->options.intensity_stereo && is_ramp > 0.0f;
724 
725  const int pidx = (s->cur_channel >> 1) & 15;
726  const int decoupled = s->psy.pair_decoupled[pidx];
727  int njoint = 0, nbands = 0; /* joint-tool candidacy census, decouple feed */
728 
729  for (int w = 0; w < ics->num_windows; w += ics->group_len[w]) {
730  int start = 0;
731  for (int g = 0; g < ics->num_swb; start += ics->swb_sizes[g++]) {
732  int len = ics->swb_sizes[g], gl = ics->group_len[w];
733  float ener0 = 0.0f, ener1 = 0.0f, dot = 0.0f, es_tot = 0.0f, em_tot = 0.0f;
734  float minthr0 = FLT_MAX, minthr1 = FLT_MAX;
735 
736  cpe->is_mask[w*16+g] = 0;
737  cpe->ms_mask[w*16+g] = 0;
738 
739  for (int w2 = 0; w2 < gl; w2++) {
740  FFPsyBand *b0 = &s->psy.ch[s->cur_channel+0].psy_bands[(w+w2)*16+g];
741  FFPsyBand *b1 = &s->psy.ch[s->cur_channel+1].psy_bands[(w+w2)*16+g];
742  const float *L = sce0->coeffs + start + (w+w2)*128;
743  const float *R = sce1->coeffs + start + (w+w2)*128;
744  float el = 0.0f, er = 0.0f, em = 0.0f, es = 0.0f, d = 0.0f;
745  for (int i = 0; i < len; i++) {
746  float m = (L[i] + R[i]) * 0.5f;
747  float sv = m - R[i];
748  el += L[i]*L[i]; er += R[i]*R[i];
749  em += m*m; es += sv*sv; d += L[i]*R[i];
750  }
751  ener0 += el; ener1 += er; dot += d; es_tot += es; em_tot += em;
752  minthr0 = FFMIN(minthr0, b0->threshold);
753  minthr1 = FFMIN(minthr1, b1->threshold);
754  }
755  float thr_g = FFMIN(minthr0, minthr1) * gl; /* group masking budget */
756 
757  /* PNS-stereo reservation: keep clearly-wide noise bands for PNS. */
758  const int sidx = w*16+g;
759  {
760  float es_w = es_tot, em_w = em_tot;
761  if (s->nmr) {
762  int pi_ = ((s->cur_channel >> 1) & 7) * 2 + (cpe->ch[0].ics.num_windows == 8);
763  float pe = s->nmr->sema_es[pi_][sidx];
764  float pm = s->nmr->sema_em[pi_][sidx];
765  if (pm > 0.0f) {
766  es_w = NMR_SDEC_EMA * pe + (1.0f - NMR_SDEC_EMA) * es_tot;
767  em_w = NMR_SDEC_EMA * pm + (1.0f - NMR_SDEC_EMA) * em_tot;
768  }
769  }
770  if (cpe->ch[0].can_pns[w*16+g] && cpe->ch[1].can_pns[w*16+g] &&
771  es_w > NMR_PNS_STEREO_DECORR * em_w)
772  continue;
773  }
774  cpe->ch[0].can_pns[w*16+g] = cpe->ch[1].can_pns[w*16+g] = 0;
775 
776  int pi = ((s->cur_channel >> 1) & 7) * 2 + (cpe->ch[0].ics.num_windows == 8);
777  uint8_t *pmode = s->nmr ? s->nmr->smode[pi] : NULL;
778  int prev = pmode ? pmode[sidx] : 0;
779  float eqgate = NMR_MS_EQUIV * (prev == 1 ? 1.5f : 1.0f); /* stay-until es>0.75em */
780  /* I/S = lossy economy: image-error budget scales with pressure */
781  float imgate = NMR_IS_IMG_GATE * is_ramp * (prev == 2 ? NMR_STICKY : 1.0f);
782  float es_d = es_tot, em_d = em_tot;
783  if (s->nmr) {
784  float *ees = &s->nmr->sema_es[pi][sidx];
785  float *eem = &s->nmr->sema_em[pi][sidx];
786  if (*eem <= 0.0f) { *ees = es_tot; *eem = em_tot; }
787  else {
788  *ees = NMR_SDEC_EMA * *ees + (1.0f - NMR_SDEC_EMA) * es_tot;
789  *eem = NMR_SDEC_EMA * *eem + (1.0f - NMR_SDEC_EMA) * em_tot;
790  }
791  es_d = *ees; em_d = *eem;
792  }
793  int ms_would = s->options.mid_side &&
794  (s->options.mid_side == 1 ||
795  es_d < eqgate * em_d ||
796  es_tot < NMR_MS_MASK * thr_g);
797  int ms_ok = ms_would && !decoupled;
798  float scale, sr_, imgratio; int p;
799  /* I/S competes with M/S above the frequency limit (candidacy must
800  * not be gated on !ms_ok - that leaves only unrenderable bands) */
801  int is_cand = start * freq_mult > NMR_IS_LOW_LIMIT &&
802  ener0 > FLT_MIN && ener1 > FLT_MIN &&
803  nmr_is_image_masked(s, cpe, w, g, start, len, gl,
804  ener0, ener1, dot, minthr0, minthr1,
805  &imgratio, &scale, &sr_, &p);
806  int is_ok = is_cand;
807  if (s->nmr && start * freq_mult > NMR_IS_LOW_LIMIT) {
808  /* smoothed image-error; updated only while candidate (fail-value
809  * feeding jammed it permanently high) */
810  float *eim = &s->nmr->sema_img[pi][sidx];
811  if (is_cand) {
812  /* seed from first measurement; freeze when not candidate */
813  if (*eim < 0.0f) *eim = imgratio;
814  else *eim = NMR_SDEC_EMA * *eim + (1.0f - NMR_SDEC_EMA) * FFMIN(imgratio, 100.0f * NMR_IS_IMG_GATE);
815  }
816  is_ok = is_cand && *eim >= 0.0f && *eim < imgate;
817  }
818 
819  njoint += ms_would || is_ok; nbands++;
820  if (pmode) {
821  int m_ = (is_ok && allow_is) ? 2 : ms_ok ? 1 :
822  (is_ok && s->options.mid_side) ? 1 : 0;
823  pmode[sidx] = m_;
824  s->nmr->smode_band[(s->cur_channel >> 1) & 7][w*16+g] = m_;
825  }
826  if (is_ok && allow_is) {
827  nmr_apply_is_band(s, cpe, w, g, start, len, gl,
828  scale, sr_, p, ener0, ener1);
829  is_count++;
830  } else if (ms_ok || (is_ok && s->options.mid_side)) {
831  nmr_apply_ms_band(s, cpe, w, g, start, len, gl);
832  }
833  /* else: keep full L/R stereo */
834  }
835  }
836  cpe->is_mode = !!is_count;
837 
838  if (nbands > 0) {
839  /* Pair joint-tool value, read next frame by the psy pair-synced window
840  * decision and the M/S candidacy above. Measured as CANDIDACY (not
841  * adoption) so decoupling cannot starve its own signal and self-lock. */
842  float r = (float)njoint / nbands;
843  float *pj = &s->psy.pair_joint[pidx];
844  *pj = *pj > 0.0f ? 0.95f * *pj + 0.05f * r : r;
845  s->psy.pair_decoupled[pidx] = *pj <
846  (s->psy.pair_decoupled[pidx] ? 1.3f * NMR_DECORR_LO : NMR_DECORR_LO);
847  }
848 }
849 
851 {
852  int w, w2, g, i;
853  IndividualChannelStream *ics = &cpe->ch[0].ics;
854  if (!cpe->common_window)
855  return;
856  for (w = 0; w < ics->num_windows; w += ics->group_len[w]) {
857  for (w2 = 0; w2 < ics->group_len[w]; w2++) {
858  int start = (w+w2) * 128;
859  for (g = 0; g < ics->num_swb; g++) {
860  /* ms_mask can be used for other purposes in PNS and I/S,
861  * so must not apply M/S if any band uses either, even if
862  * ms_mask is set.
863  */
864  if (!cpe->ms_mask[w*16 + g] || cpe->is_mask[w*16 + g]
865  || cpe->ch[0].band_type[w*16 + g] >= NOISE_BT
866  || cpe->ch[1].band_type[w*16 + g] >= NOISE_BT) {
867  start += ics->swb_sizes[g];
868  continue;
869  }
870  for (i = 0; i < ics->swb_sizes[g]; i++) {
871  float L = (cpe->ch[0].coeffs[start+i] + cpe->ch[1].coeffs[start+i]) * 0.5f;
872  float R = L - cpe->ch[1].coeffs[start+i];
873  cpe->ch[0].coeffs[start+i] = L;
874  cpe->ch[1].coeffs[start+i] = R;
875  }
876  start += ics->swb_sizes[g];
877  }
878  }
879  }
880 }
881 
882 /**
883  * Encode scalefactor band coding type.
884  */
886 {
887  int w;
888 
889  if (s->coder->set_special_band_scalefactors)
890  s->coder->set_special_band_scalefactors(s, sce);
891 
892  for (w = 0; w < sce->ics.num_windows; w += sce->ics.group_len[w])
893  s->coder->encode_window_bands_info(s, sce, w, sce->ics.group_len[w], s->lambda);
894 }
895 
896 /**
897  * Encode scalefactors.
898  */
901 {
902  int diff, off_sf = sce->sf_idx[0], off_pns = sce->sf_idx[0] - NOISE_OFFSET;
903  int off_is = 0, noise_flag = 1;
904  int i, w;
905 
906  for (w = 0; w < sce->ics.num_windows; w += sce->ics.group_len[w]) {
907  for (i = 0; i < sce->ics.max_sfb; i++) {
908  if (!sce->zeroes[w*16 + i]) {
909  if (sce->band_type[w*16 + i] == NOISE_BT) {
910  diff = sce->sf_idx[w*16 + i] - off_pns;
911  off_pns = sce->sf_idx[w*16 + i];
912  if (noise_flag-- > 0) {
914  continue;
915  }
916  } else if (sce->band_type[w*16 + i] == INTENSITY_BT ||
917  sce->band_type[w*16 + i] == INTENSITY_BT2) {
918  diff = sce->sf_idx[w*16 + i] - off_is;
919  off_is = sce->sf_idx[w*16 + i];
920  } else {
921  diff = sce->sf_idx[w*16 + i] - off_sf;
922  off_sf = sce->sf_idx[w*16 + i];
923  }
925  av_assert0(diff >= 0 && diff <= 120);
927  }
928  }
929  }
930 }
931 
932 /**
933  * Encode pulse data.
934  */
935 static void encode_pulses(AACEncContext *s, Pulse *pulse)
936 {
937  int i;
938 
939  put_bits(&s->pb, 1, !!pulse->num_pulse);
940  if (!pulse->num_pulse)
941  return;
942 
943  put_bits(&s->pb, 2, pulse->num_pulse - 1);
944  put_bits(&s->pb, 6, pulse->start);
945  for (i = 0; i < pulse->num_pulse; i++) {
946  put_bits(&s->pb, 5, pulse->pos[i]);
947  put_bits(&s->pb, 4, pulse->amp[i]);
948  }
949 }
950 
951 /**
952  * Encode spectral coefficients processed by psychoacoustic model.
953  */
955 {
956  int start, i, w, w2;
957 
958  for (w = 0; w < sce->ics.num_windows; w += sce->ics.group_len[w]) {
959  start = 0;
960  for (i = 0; i < sce->ics.max_sfb; i++) {
961  if (sce->zeroes[w*16 + i]) {
962  start += sce->ics.swb_sizes[i];
963  continue;
964  }
965  for (w2 = w; w2 < w + sce->ics.group_len[w]; w2++) {
966  s->coder->quantize_and_encode_band(s, &s->pb,
967  &sce->coeffs[start + w2*128],
968  NULL, sce->ics.swb_sizes[i],
969  sce->sf_idx[w*16 + i],
970  sce->band_type[w*16 + i],
971  s->lambda,
972  sce->ics.window_clipping[w]);
973  }
974  start += sce->ics.swb_sizes[i];
975  }
976  }
977 }
978 
979 /**
980  * Downscale spectral coefficients for near-clipping windows to avoid artifacts
981  */
983 {
984  int start, i, j, w;
985 
986  if (sce->ics.clip_avoidance_factor < 1.0f) {
987  for (w = 0; w < sce->ics.num_windows; w++) {
988  start = 0;
989  for (i = 0; i < sce->ics.max_sfb; i++) {
990  float *swb_coeffs = &sce->coeffs[start + w*128];
991  for (j = 0; j < sce->ics.swb_sizes[i]; j++)
992  swb_coeffs[j] *= sce->ics.clip_avoidance_factor;
993  start += sce->ics.swb_sizes[i];
994  }
995  }
996  }
997 }
998 
999 /**
1000  * Encode one channel of audio data.
1001  */
1003  SingleChannelElement *sce,
1004  int common_window)
1005 {
1006  put_bits(&s->pb, 8, sce->sf_idx[0]);
1007  if (!common_window)
1008  put_ics_info(s, &sce->ics);
1009  encode_band_info(s, sce);
1010  encode_scale_factors(avctx, s, sce);
1011  encode_pulses(s, &sce->pulse);
1012  put_bits(&s->pb, 1, !!sce->tns.present);
1013  if (s->coder->encode_tns_info)
1014  s->coder->encode_tns_info(s, sce);
1015  put_bits(&s->pb, 1, 0); //ssr
1016  encode_spectral_coeffs(s, sce);
1017  return 0;
1018 }
1019 
1020 /**
1021  * Write some auxiliary information about the created AAC file.
1022  */
1023 static void put_bitstream_info(AACEncContext *s, const char *name)
1024 {
1025  int i, namelen, padbits;
1026 
1027  namelen = strlen(name) + 2;
1028  put_bits(&s->pb, 3, TYPE_FIL);
1029  put_bits(&s->pb, 4, FFMIN(namelen, 15));
1030  if (namelen >= 15)
1031  put_bits(&s->pb, 8, namelen - 14);
1032  put_bits(&s->pb, 4, 0); //extension type - filler
1033  padbits = -put_bits_count(&s->pb) & 7;
1034  align_put_bits(&s->pb);
1035  for (i = 0; i < namelen - 2; i++)
1036  put_bits(&s->pb, 8, name[i]);
1037  put_bits(&s->pb, 12 - padbits, 0);
1038 }
1039 
1040 /*
1041  * Copy input samples.
1042  * Channels are reordered from libavcodec's default order to AAC order.
1043  */
1045 {
1046  int ch;
1047  int end = 2048 + (frame ? frame->nb_samples : 0);
1048  const uint8_t *channel_map = s->reorder_map;
1049 
1050  /* copy and remap input samples */
1051  for (ch = 0; ch < s->channels; ch++) {
1052  /* copy last 1024 samples of previous frame to the start of the current frame */
1053  memcpy(&s->planar_samples[ch][1024], &s->planar_samples[ch][2048], 1024 * sizeof(s->planar_samples[0][0]));
1054 
1055  /* copy new samples and zero any remaining samples */
1056  if (frame) {
1057  memcpy(&s->planar_samples[ch][2048],
1058  frame->extended_data[channel_map[ch]],
1059  frame->nb_samples * sizeof(s->planar_samples[0][0]));
1060  }
1061  memset(&s->planar_samples[ch][end], 0,
1062  (3072 - end) * sizeof(s->planar_samples[0][0]));
1063  }
1064 }
1065 
1066 static int aac_encode_frame(AVCodecContext *avctx, AVPacket *avpkt,
1067  const AVFrame *frame, int *got_packet_ptr)
1068 {
1069  AACEncContext *s = avctx->priv_data;
1070  float **samples = s->planar_samples, *samples2, *la, *overlap;
1071  ChannelElement *cpe;
1072  SingleChannelElement *sce;
1074  int i, its, ch, w, chans, tag, start_ch, ret, frame_bits;
1075  int target_bits, rate_bits, too_many_bits, too_few_bits;
1076  int ms_mode = 0, is_mode = 0, tns_mode = 0, pred_mode = 0;
1077  int chan_el_counter[4];
1079 
1080  /* add current frame to queue */
1081  if (frame) {
1082  if ((ret = ff_af_queue_add(&s->afq, frame)) < 0)
1083  return ret;
1084  } else {
1085  if (!s->afq.remaining_samples || (!s->afq.frame_alloc && !s->afq.frame_count))
1086  return 0;
1087  }
1088 
1090 
1091  if (!avctx->frame_num)
1092  return 0;
1093 
1094  start_ch = 0;
1095  for (i = 0; i < s->chan_map[0]; i++) {
1096  FFPsyWindowInfo* wi = windows + start_ch;
1097  tag = s->chan_map[i+1];
1098  chans = tag == TYPE_CPE ? 2 : 1;
1099  cpe = &s->cpe[i];
1100  {
1101  int wi_paired = 0;
1102  /* Synced pair windows: decide both channels of a CPE together so
1103  * their block switching never diverges (see psy window_pair). */
1104  if (chans == 2 && tag != TYPE_LFE && s->psy.model->window_pair && frame) {
1105  const float *ov0 = &samples[start_ch][0], *ov1 = &samples[start_ch + 1][0];
1106  s->psy.model->window_pair(&s->psy,
1107  ov0 + 1024, ov0 + 1024 + 448 + 64,
1108  ov1 + 1024, ov1 + 1024 + 448 + 64,
1109  start_ch, start_ch + 1,
1110  cpe->ch[0].ics.window_sequence[0],
1111  cpe->ch[1].ics.window_sequence[0],
1112  wi);
1113  wi_paired = 1;
1114  }
1115  for (ch = 0; ch < chans; ch++) {
1116  int k;
1117  float clip_avoidance_factor;
1118  sce = &cpe->ch[ch];
1119  ics = &sce->ics;
1120  s->cur_channel = start_ch + ch;
1121  overlap = &samples[s->cur_channel][0];
1122  samples2 = overlap + 1024;
1123  la = samples2 + (448+64);
1124  if (!frame)
1125  la = NULL;
1126  if (tag == TYPE_LFE) {
1127  wi[ch].window_type[0] = wi[ch].window_type[1] = ONLY_LONG_SEQUENCE;
1128  wi[ch].window_shape = 0;
1129  wi[ch].num_windows = 1;
1130  wi[ch].grouping[0] = 1;
1131  wi[ch].clipping[0] = 0;
1132 
1133  /* Only the lowest 12 coefficients are used in a LFE channel.
1134  * The expression below results in only the bottom 8 coefficients
1135  * being used for 11.025kHz to 16kHz sample rates.
1136  */
1137  ics->num_swb = s->samplerate_index >= 8 ? 1 : 3;
1138  } else if (!wi_paired) {
1139  wi[ch] = s->psy.model->window(&s->psy, samples2, la, s->cur_channel,
1140  ics->window_sequence[0]);
1141  }
1142  ics->window_sequence[1] = ics->window_sequence[0];
1143  ics->window_sequence[0] = wi[ch].window_type[0];
1144  ics->use_kb_window[1] = ics->use_kb_window[0];
1145  ics->use_kb_window[0] = wi[ch].window_shape;
1146  ics->num_windows = wi[ch].num_windows;
1147  ics->swb_sizes = s->psy.bands [ics->num_windows == 8];
1148  ics->num_swb = tag == TYPE_LFE ? ics->num_swb : s->psy.num_bands[ics->num_windows == 8];
1149  ics->max_sfb = FFMIN(ics->max_sfb, ics->num_swb);
1150  ics->swb_offset = wi[ch].window_type[0] == EIGHT_SHORT_SEQUENCE ?
1151  ff_swb_offset_128 [s->samplerate_index]:
1152  ff_swb_offset_1024[s->samplerate_index];
1153  ics->tns_max_bands = wi[ch].window_type[0] == EIGHT_SHORT_SEQUENCE ?
1154  ff_tns_max_bands_128 [s->samplerate_index]:
1155  ff_tns_max_bands_1024[s->samplerate_index];
1156 
1157  for (w = 0; w < ics->num_windows; w++)
1158  ics->group_len[w] = wi[ch].grouping[w];
1159 
1160  /* Calculate input sample maximums and evaluate clipping risk */
1161  clip_avoidance_factor = 0.0f;
1162  for (w = 0; w < ics->num_windows; w++) {
1163  const float *wbuf = overlap + w * 128;
1164  const int wlen = 2048 / ics->num_windows;
1165  float max = 0;
1166  int j;
1167  /* mdct input is 2 * output */
1168  for (j = 0; j < wlen; j++)
1169  max = FFMAX(max, fabsf(wbuf[j]));
1170  wi[ch].clipping[w] = max;
1171  }
1172  for (w = 0; w < ics->num_windows; w++) {
1173  if (wi[ch].clipping[w] > CLIP_AVOIDANCE_FACTOR) {
1174  ics->window_clipping[w] = 1;
1175  clip_avoidance_factor = FFMAX(clip_avoidance_factor, wi[ch].clipping[w]);
1176  } else {
1177  ics->window_clipping[w] = 0;
1178  }
1179  }
1180  if (clip_avoidance_factor > CLIP_AVOIDANCE_FACTOR) {
1181  ics->clip_avoidance_factor = CLIP_AVOIDANCE_FACTOR / clip_avoidance_factor;
1182  } else {
1183  ics->clip_avoidance_factor = 1.0f;
1184  }
1185 
1186  apply_window_and_mdct(s, sce, overlap);
1187 
1188  for (k = 0; k < 1024; k++) {
1189  if (!(fabs(cpe->ch[ch].coeffs[k]) < 1E16)) { // Ensure headroom for energy calculation
1190  av_log(avctx, AV_LOG_ERROR, "Input contains (near) NaN/+-Inf\n");
1191  return AVERROR(EINVAL);
1192  }
1193  }
1194  avoid_clipping(s, sce);
1195  }
1196  }
1197  start_ch += chans;
1198  }
1199  if ((ret = ff_alloc_packet(avctx, avpkt, 8192 * s->channels)) < 0)
1200  return ret;
1201  frame_bits = its = 0;
1202  do {
1203  init_put_bits(&s->pb, avpkt->data, avpkt->size);
1204 
1205  if ((avctx->frame_num & 0xFF)==1 && !(avctx->flags & AV_CODEC_FLAG_BITEXACT))
1207  start_ch = 0;
1208  target_bits = 0;
1209  memset(chan_el_counter, 0, sizeof(chan_el_counter));
1210  for (i = 0; i < s->chan_map[0]; i++) {
1211  FFPsyWindowInfo* wi = windows + start_ch;
1212  const float *coeffs[2];
1213  tag = s->chan_map[i+1];
1214  chans = tag == TYPE_CPE ? 2 : 1;
1215  cpe = &s->cpe[i];
1216  cpe->common_window = 0;
1217  memset(cpe->is_mask, 0, sizeof(cpe->is_mask));
1218  memset(cpe->ms_mask, 0, sizeof(cpe->ms_mask));
1219  put_bits(&s->pb, 3, tag);
1220  put_bits(&s->pb, 4, chan_el_counter[tag]++);
1221  for (ch = 0; ch < chans; ch++) {
1222  sce = &cpe->ch[ch];
1223  coeffs[ch] = sce->coeffs;
1224  memset(&sce->tns, 0, sizeof(TemporalNoiseShaping));
1225  for (w = 0; w < 128; w++)
1226  if (sce->band_type[w] > RESERVED_BT)
1227  sce->band_type[w] = 0;
1228  }
1229  s->psy.bitres.alloc = -1;
1230  s->psy.bitres.bits = s->last_frame_pb_count / s->channels;
1231  s->psy.model->analyze(&s->psy, start_ch, coeffs, wi);
1232  if (s->psy.bitres.alloc > 0) {
1233  /* Lambda unused here on purpose, we need to take psy's unscaled allocation */
1234  target_bits += s->psy.bitres.alloc
1235  * (s->lambda / (avctx->global_quality ? avctx->global_quality : 120));
1236  s->psy.bitres.alloc /= chans;
1237  }
1238  s->cur_type = tag;
1239  if (chans > 1
1240  && wi[0].window_type[0] == wi[1].window_type[0]
1241  && wi[0].window_shape == wi[1].window_shape) {
1242 
1243  cpe->common_window = 1;
1244  for (w = 0; w < wi[0].num_windows; w++) {
1245  if (wi[0].grouping[w] != wi[1].grouping[w]) {
1246  cpe->common_window = 0;
1247  break;
1248  }
1249  }
1250  }
1251 
1252  const int use_tns = s->options.tns && s->coder->search_for_tns &&
1253  s->coder->apply_tns_filt;
1254 
1255  /* The NMR coder rate-controls itself and never re-quantizes, so TNS must run
1256  * before the quantizer */
1257  const int tns_first = s->options.coder == AAC_CODER_NMR;
1258  if (tns_first && use_tns) {
1259  for (ch = 0; ch < chans; ch++) {
1260  sce = &cpe->ch[ch];
1261  s->cur_channel = start_ch + ch;
1262  /* mono: mark_pns before TNS so the region cap sees PNS bands. Stereo
1263  * PNS is marked in its own block (below) after the stereo decision. */
1264  if (chans == 1 && s->options.pns && s->coder->mark_pns)
1265  s->coder->mark_pns(s, avctx, sce);
1266  s->coder->search_for_tns(s, sce);
1267  s->coder->apply_tns_filt(s, sce);
1268  if (sce->tns.present)
1269  tns_mode = 1;
1270  }
1271  }
1272 
1273  /* NMR stereo PNS (imaging-safe). Mark each channel's noise-like bands on the
1274  * original L/R psy, then keep PNS only where BOTH channels are noise-like. */
1275  if (chans == 2 && cpe->common_window && tns_first &&
1276  s->options.pns && s->coder->mark_pns) {
1277  s->cur_channel = start_ch; s->coder->mark_pns(s, avctx, &cpe->ch[0]);
1278  s->cur_channel = start_ch + 1; s->coder->mark_pns(s, avctx, &cpe->ch[1]);
1279  for (int b = 0; b < 128; b++)
1280  if (!cpe->ch[0].can_pns[b] || !cpe->ch[1].can_pns[b])
1281  cpe->ch[0].can_pns[b] = cpe->ch[1].can_pns[b] = 0;
1282  }
1283 
1284  /* The NMR coder decides I/S and M/S BEFORE quantization, from the psy model,
1285  * and the trellis then allocates natively on the coeffs actually coded. */
1286  if (chans == 2 && cpe->common_window && s->options.coder == AAC_CODER_NMR &&
1287  (s->options.mid_side || s->options.intensity_stereo)) {
1288  s->cur_channel = start_ch;
1289  nmr_decide_stereo(s, cpe);
1290  }
1291  /* NMR pools the CPE bit budget: both channels of a pair are solved
1292  * jointly under one shared lambda (see aaccoder_nmr.h). */
1293  if (s->options.coder == AAC_CODER_NMR && s->nmr)
1294  s->nmr->pair = (chans == 2);
1295  for (ch = 0; ch < chans; ch++) {
1296  s->cur_channel = start_ch + ch;
1297  /* NMR PNS is mono-only */
1298  if (s->options.pns && s->coder->mark_pns && !tns_first)
1299  s->coder->mark_pns(s, avctx, &cpe->ch[ch]);
1300  s->coder->search_for_quantizers(avctx, s, &cpe->ch[ch], s->lambda);
1301  }
1302  for (ch = 0; ch < chans; ch++) { /* TNS (non-NMR) and PNS */
1303  sce = &cpe->ch[ch];
1304  s->cur_channel = start_ch + ch;
1305  if (!tns_first && use_tns) {
1306  s->coder->search_for_tns(s, sce);
1307  s->coder->apply_tns_filt(s, sce);
1308  if (sce->tns.present)
1309  tns_mode = 1;
1310  }
1311  if (s->options.pns && s->coder->search_for_pns)
1312  s->coder->search_for_pns(s, avctx, sce);
1313  }
1314  s->cur_channel = start_ch;
1315  if (s->options.intensity_stereo) { /* Intensity Stereo */
1316  if (s->options.coder != AAC_CODER_NMR) { /* NMR: decided pre-search */
1317  if (s->coder->search_for_is)
1318  s->coder->search_for_is(s, avctx, cpe);
1320  }
1321  if (cpe->is_mode) is_mode = 1;
1322  }
1323  if (s->options.mid_side && s->options.coder != AAC_CODER_NMR) { /* Mid/Side stereo */
1324  if (s->options.mid_side == -1 && s->coder->search_for_ms)
1325  s->coder->search_for_ms(s, cpe);
1326  else if (cpe->common_window)
1327  memset(cpe->ms_mask, 1, sizeof(cpe->ms_mask));
1328  apply_mid_side_stereo(cpe);
1329  }
1330  adjust_frame_information(cpe, chans);
1331  if (chans == 2) {
1332  put_bits(&s->pb, 1, cpe->common_window);
1333  if (cpe->common_window) {
1334  put_ics_info(s, &cpe->ch[0].ics);
1335  encode_ms_info(&s->pb, cpe);
1336  if (cpe->ms_mode) ms_mode = 1;
1337  }
1338  }
1339  for (ch = 0; ch < chans; ch++) {
1340  s->cur_channel = start_ch + ch;
1341  encode_individual_channel(avctx, s, &cpe->ch[ch], cpe->common_window);
1342  }
1343  start_ch += chans;
1344  }
1345 
1346  if (avctx->flags & AV_CODEC_FLAG_QSCALE) {
1347  /* When using a constant Q-scale, don't mess with lambda */
1348  break;
1349  }
1350 
1351  frame_bits = put_bits_count(&s->pb);
1352 
1353  /* The NMR coder rate-controls itself (global-lambda reservoir servo):
1354  * per-frame bits intentionally float around the nominal rate, so skip
1355  * the lambda rate loop and only intervene on a hard overflow. */
1356  if (s->options.coder == AAC_CODER_NMR && avctx->bit_rate_tolerance != 0 &&
1357  frame_bits < 6144 * s->channels - 3)
1358  break;
1359 
1360  /* rate control stuff
1361  * allow between the nominal bitrate, and what psy's bit reservoir says to target
1362  * but drift towards the nominal bitrate always
1363  */
1364  rate_bits = avctx->bit_rate * 1024 / avctx->sample_rate;
1365  rate_bits = FFMIN(rate_bits, 6144 * s->channels - 3);
1366  too_many_bits = FFMAX(target_bits, rate_bits);
1367  too_many_bits = FFMIN(too_many_bits, 6144 * s->channels - 3);
1368  too_few_bits = FFMIN(FFMAX(rate_bits - rate_bits/4, target_bits), too_many_bits);
1369 
1370  /* When strict bit-rate control is demanded */
1371  if (avctx->bit_rate_tolerance == 0) {
1372  if (rate_bits < frame_bits) {
1373  float ratio = ((float)rate_bits) / frame_bits;
1374  s->lambda *= FFMIN(0.9f, ratio);
1375  continue;
1376  }
1377  /* reset lambda when solution is found */
1378  s->lambda = avctx->global_quality > 0 ? avctx->global_quality : 120;
1379  break;
1380  }
1381 
1382  /* When using ABR, be strict (but only for increasing) */
1383  too_few_bits = too_few_bits - too_few_bits/8;
1384  too_many_bits = too_many_bits + too_many_bits/2;
1385 
1386  if ( its == 0 /* for steady-state Q-scale tracking */
1387  || (its < 5 && (frame_bits < too_few_bits || frame_bits > too_many_bits))
1388  || frame_bits >= 6144 * s->channels - 3 )
1389  {
1390  float ratio = ((float)rate_bits) / frame_bits;
1391 
1392  if (frame_bits >= too_few_bits && frame_bits <= too_many_bits) {
1393  /*
1394  * This path is for steady-state Q-scale tracking
1395  * When frame bits fall within the stable range, we still need to adjust
1396  * lambda to maintain it like so in a stable fashion (large jumps in lambda
1397  * create artifacts and should be avoided), but slowly
1398  */
1399  ratio = sqrtf(sqrtf(ratio));
1400  ratio = av_clipf(ratio, 0.9f, 1.1f);
1401  } else {
1402  /* Not so fast though */
1403  ratio = sqrtf(ratio);
1404  }
1405  s->lambda = av_clipf(s->lambda * ratio, FLT_EPSILON, 65536.f);
1406 
1407  /* Keep iterating if we must reduce and lambda is in the sky */
1408  if (ratio > 0.9f && ratio < 1.1f) {
1409  break;
1410  } else {
1411  if (is_mode || ms_mode || tns_mode || pred_mode) {
1412  for (i = 0; i < s->chan_map[0]; i++) {
1413  // Must restore coeffs
1414  chans = tag == TYPE_CPE ? 2 : 1;
1415  cpe = &s->cpe[i];
1416  for (ch = 0; ch < chans; ch++)
1417  memcpy(cpe->ch[ch].coeffs, cpe->ch[ch].pcoeffs, sizeof(cpe->ch[ch].coeffs));
1418  }
1419  }
1420  its++;
1421  }
1422  } else {
1423  break;
1424  }
1425  } while (1);
1426 
1427  /* tool-usage stats over the final per-band decisions of this frame */
1428  for (i = 0; i < s->chan_map[0]; i++) {
1429  int etag = s->chan_map[i + 1], echans = etag == TYPE_CPE ? 2 : 1;
1430  ChannelElement *ce = &s->cpe[i];
1431  IndividualChannelStream *ics = &ce->ch[0].ics;
1432  for (ch = 0; ch < echans; ch++) { /* per-channel frame stats */
1433  int is_short = ce->ch[ch].ics.window_sequence[0] == EIGHT_SHORT_SEQUENCE;
1434  s->stat_chans++;
1435  if (is_short)
1436  s->stat_short++;
1437  if (ce->ch[ch].tns.present) {
1438  if (is_short) s->stat_tns_short++;
1439  else s->stat_tns_long++;
1440  }
1441  }
1442  for (w = 0; w < ics->num_windows; w += ics->group_len[w]) {
1443  for (int g = 0; g < ics->num_swb; g++) {
1444  int idx = w*16 + g, coded = 0;
1445  for (ch = 0; ch < echans; ch++) {
1446  SingleChannelElement *sce = &ce->ch[ch];
1447  if (sce->zeroes[idx] && sce->band_type[idx] == 0)
1448  continue;
1449  s->stat_ch_bands++;
1450  if (sce->band_type[idx] == NOISE_BT)
1451  s->stat_pns++;
1452  coded = 1;
1453  }
1454  if (etag == TYPE_CPE && coded) {
1455  s->stat_cpe_bands++;
1456  if (ce->ms_mask[idx]) s->stat_ms++;
1457  if (ce->is_mask[idx]) s->stat_is++;
1458  }
1459  }
1460  }
1461  }
1462 
1463  put_bits(&s->pb, 3, TYPE_END);
1464  flush_put_bits(&s->pb);
1465 
1466  s->last_frame_pb_count = put_bits_count(&s->pb);
1467 
1468  /* NMR rate accounting: how many bits the frame really took beyond what the
1469  * trellis counted; feeds the next frame's budget correction */
1470  if (s->nmr) {
1471  int counted = 0;
1472  for (i = 0; i < s->channels; i++)
1473  counted += s->nmr->counted[i];
1474  if (counted > 0) {
1475  float side = (float)s->last_frame_pb_count - counted;
1476  if (s->nmr->side_inited) {
1477  s->nmr->side_ema += 0.125f * (side - s->nmr->side_ema);
1478  } else {
1479  s->nmr->side_ema = side;
1480  s->nmr->side_inited = 1;
1481  }
1482  }
1483  }
1484  avpkt->size = put_bytes_output(&s->pb);
1485 
1486  s->lambda_sum += (s->nmr && s->nmr->lam_rc > 0.0f) ? s->nmr->lam_rc : s->lambda;
1487  s->lambda_count++;
1488 
1489  ret = ff_af_queue_remove(&s->afq, avctx->frame_size, avpkt);
1490  if (ret < 0)
1491  return ret;
1492 
1493  avpkt->flags |= AV_PKT_FLAG_KEY;
1494 
1495  *got_packet_ptr = 1;
1496  return 0;
1497 }
1498 
1500 {
1501  AACEncContext *s = avctx->priv_data;
1502 
1503  av_log(avctx, AV_LOG_INFO,
1504  "Qavg: %.3f Tr: %.1f%% TNS(L): %.1f%% TNS(S): %.1f%% M/S: %.1f%% I/S: %.1f%% PNS: %.1f%%\n",
1505  s->lambda_count ? s->lambda_sum / s->lambda_count : NAN,
1506  s->stat_chans ? 100.0 * s->stat_short / s->stat_chans : 0.0,
1507  s->stat_chans - s->stat_short ? 100.0 * s->stat_tns_long / (s->stat_chans - s->stat_short) : 0.0,
1508  s->stat_short ? 100.0 * s->stat_tns_short / s->stat_short : 0.0,
1509  s->stat_cpe_bands ? 100.0 * s->stat_ms / s->stat_cpe_bands : 0.0,
1510  s->stat_cpe_bands ? 100.0 * s->stat_is / s->stat_cpe_bands : 0.0,
1511  s->stat_ch_bands ? 100.0 * s->stat_pns / s->stat_ch_bands : 0.0);
1512 
1513  av_tx_uninit(&s->mdct1024);
1514  av_tx_uninit(&s->mdct128);
1515  ff_psy_end(&s->psy);
1516  ff_lpc_end(&s->lpc);
1517  av_freep(&s->buffer.samples);
1518  av_freep(&s->cpe);
1519  av_freep(&s->fdsp);
1520  av_freep(&s->nmr);
1521  ff_af_queue_close(&s->afq);
1522  return 0;
1523 }
1524 
1526 {
1527  int ret = 0;
1528  float scale = 32768.0f;
1529 
1531  if (!s->fdsp)
1532  return AVERROR(ENOMEM);
1533 
1534  if ((ret = av_tx_init(&s->mdct1024, &s->mdct1024_fn, AV_TX_FLOAT_MDCT, 0,
1535  1024, &scale, 0)) < 0)
1536  return ret;
1537  if ((ret = av_tx_init(&s->mdct128, &s->mdct128_fn, AV_TX_FLOAT_MDCT, 0,
1538  128, &scale, 0)) < 0)
1539  return ret;
1540 
1541  return 0;
1542 }
1543 
1545 {
1546  int ch;
1547  if (!FF_ALLOCZ_TYPED_ARRAY(s->buffer.samples, s->channels * 3 * 1024) ||
1548  !FF_ALLOCZ_TYPED_ARRAY(s->cpe, s->chan_map[0]))
1549  return AVERROR(ENOMEM);
1550 
1551  for(ch = 0; ch < s->channels; ch++)
1552  s->planar_samples[ch] = s->buffer.samples + 3 * 1024 * ch;
1553 
1554  if (s->options.coder == AAC_CODER_NMR) {
1555  s->nmr = av_mallocz(sizeof(*s->nmr));
1556  if (!s->nmr)
1557  return AVERROR(ENOMEM);
1558  }
1559 
1560  return 0;
1561 }
1562 
1564 {
1565  AACEncContext *s = avctx->priv_data;
1566  int i, ret = 0;
1567  int chcfg;
1568  const uint8_t *sizes[2];
1569  uint8_t grouping[AAC_MAX_CHANNELS];
1570  int lengths[2];
1571 
1572  /* Constants */
1573  s->last_frame_pb_count = 0;
1574  avctx->frame_size = 1024;
1575  avctx->initial_padding = 1024;
1576  s->lambda = avctx->global_quality > 0 ? avctx->global_quality : 120;
1577 
1578  /* Channel map and unspecified bitrate guessing */
1579  s->channels = avctx->ch_layout.nb_channels;
1580 
1581  s->needs_pce = 1;
1582  for (chcfg = 1; chcfg < FF_ARRAY_ELEMS(aac_normal_chan_layouts); chcfg++) {
1584  s->needs_pce = s->options.pce;
1585  break;
1586  }
1587  }
1588 
1589  if (s->needs_pce) {
1590  char buf[64];
1591  for (i = 0; i < FF_ARRAY_ELEMS(aac_pce_configs); i++)
1593  break;
1594  av_channel_layout_describe(&avctx->ch_layout, buf, sizeof(buf));
1595  if (i == FF_ARRAY_ELEMS(aac_pce_configs)) {
1596  av_log(avctx, AV_LOG_ERROR, "Unsupported channel layout \"%s\"\n", buf);
1597  return AVERROR(EINVAL);
1598  }
1599  av_log(avctx, AV_LOG_INFO, "Using a PCE to encode channel layout \"%s\"\n", buf);
1600  s->pce = aac_pce_configs[i];
1601  s->reorder_map = s->pce.reorder_map;
1602  s->chan_map = s->pce.config_map;
1603  chcfg = 0;
1604  } else {
1605  s->reorder_map = aac_chan_maps[chcfg - 1];
1606  s->chan_map = aac_chan_configs[chcfg - 1];
1607  }
1608 
1609  if (!avctx->bit_rate) {
1610  for (i = 1; i <= s->chan_map[0]; i++) {
1611  avctx->bit_rate += s->chan_map[i] == TYPE_CPE ? 128000 : /* Pair */
1612  s->chan_map[i] == TYPE_LFE ? 16000 : /* LFE */
1613  69000 ; /* SCE */
1614  }
1615  }
1616 
1617  /* Samplerate */
1618  for (int i = 0;; i++) {
1619  av_assert1(i < 13);
1620  if (avctx->sample_rate == ff_mpeg4audio_sample_rates[i]) {
1621  s->samplerate_index = i;
1622  break;
1623  }
1624  }
1625 
1626  /* Bitrate limiting */
1627  WARN_IF(1024.0 * avctx->bit_rate / avctx->sample_rate > 6144 * s->channels,
1628  "Too many bits %f > %d per frame requested, clamping to max\n",
1629  1024.0 * avctx->bit_rate / avctx->sample_rate,
1630  6144 * s->channels);
1631  avctx->bit_rate = (int64_t)FFMIN(6144 * s->channels / 1024.0 * avctx->sample_rate,
1632  avctx->bit_rate);
1633 
1634  /* Profile and option setting */
1635  avctx->profile = avctx->profile == AV_PROFILE_UNKNOWN ? AV_PROFILE_AAC_LOW :
1636  avctx->profile;
1637  for (i = 0; i < FF_ARRAY_ELEMS(aacenc_profiles); i++)
1638  if (avctx->profile == aacenc_profiles[i])
1639  break;
1640  ERROR_IF(i == FF_ARRAY_ELEMS(aacenc_profiles), "Profile not supported!\n");
1641  if (avctx->profile == AV_PROFILE_MPEG2_AAC_LOW) {
1642  avctx->profile = AV_PROFILE_AAC_LOW;
1643  WARN_IF(s->options.pns,
1644  "PNS unavailable in the \"mpeg2_aac_low\" profile, turning off\n");
1645  s->options.pns = 0;
1646  }
1647  s->profile = avctx->profile;
1648 
1649  /* Coder limitations */
1650  s->coder = &ff_aac_coders[s->options.coder];
1651 
1652  /* M/S introduces horrible artifacts with multichannel files, this is temporary */
1653  if (s->channels > 3)
1654  s->options.mid_side = 0;
1655 
1656  /* Coding bandwidth, fixed at init time */
1657  if (avctx->cutoff > 0) {
1658  s->bandwidth = avctx->cutoff;
1659  } else {
1660  int frame_br = (avctx->flags & AV_CODEC_FLAG_QSCALE) ?
1661  (avctx->bit_rate / 2.0f * (s->lambda / 120.f) * 1.5f) :
1662  (avctx->bit_rate / avctx->ch_layout.nb_channels);
1663 
1664  if (s->options.coder == AAC_CODER_NMR && frame_br >= 24000) {
1665  static const int rates[] = { 24000, 32000, 48000, 64000, 96000, 192000 };
1666  static const int bws[] = { 14000, 14000, 18500, 20000, 21000, 22000 };
1667  int bw_i = 0;
1668  for (; bw_i < FF_ARRAY_ELEMS(rates) - 2 && frame_br > rates[bw_i + 1]; bw_i++);
1669  s->bandwidth = bws[bw_i] + (int)((int64_t)(bws[bw_i + 1] - bws[bw_i]) *
1670  (frame_br - rates[bw_i]) / (rates[bw_i + 1] - rates[bw_i]));
1671  s->bandwidth = FFMIN3(s->bandwidth, 22000, avctx->sample_rate / 2);
1672  } else {
1673  if (s->options.pns || s->options.intensity_stereo)
1674  frame_br *= 1.15f;
1675  s->bandwidth = FFMAX(3000, AAC_CUTOFF_FROM_BITRATE(frame_br, 1,
1676  avctx->sample_rate));
1677  }
1678 
1679  s->bandwidth = FFMIN(FFMAX(s->bandwidth, 8000), avctx->sample_rate / 2);
1680  }
1681 
1682  if (!(avctx->flags & AV_CODEC_FLAG_QSCALE) && avctx->bit_rate > 0) {
1683  int bpc = avctx->bit_rate / avctx->ch_layout.nb_channels;
1684  if (bpc <= 32000 && avctx->sample_rate > 32000)
1685  av_log(avctx, AV_LOG_INFO,
1686  "%d kb/s per channel at %d Hz: consider resampling the "
1687  "input to 32000 Hz or lower for better quality.\n",
1688  bpc / 1000, avctx->sample_rate);
1689  }
1690 
1691  // Initialize static tables
1693 
1694  if ((ret = dsp_init(avctx, s)) < 0)
1695  return ret;
1696 
1697  if ((ret = alloc_buffers(avctx, s)) < 0)
1698  return ret;
1699 
1700  if ((ret = put_audio_specific_config(avctx, chcfg)))
1701  return ret;
1702 
1703  sizes[0] = ff_aac_swb_size_1024[s->samplerate_index];
1704  sizes[1] = ff_aac_swb_size_128[s->samplerate_index];
1705  lengths[0] = ff_aac_num_swb_1024[s->samplerate_index];
1706  lengths[1] = ff_aac_num_swb_128[s->samplerate_index];
1707  for (i = 0; i < s->chan_map[0]; i++)
1708  grouping[i] = s->chan_map[i + 1] == TYPE_CPE;
1709  if ((ret = ff_psy_init(&s->psy, avctx, 2, sizes, lengths,
1710  s->chan_map[0], grouping, s->bandwidth)) < 0)
1711  return ret;
1713  s->random_state = 0x1f2e3d4c;
1714 
1715  ff_aacenc_dsp_init(&s->aacdsp);
1716 
1717  ff_af_queue_init(avctx, &s->afq);
1718 
1719  return 0;
1720 }
1721 
1722 #define AACENC_FLAGS AV_OPT_FLAG_ENCODING_PARAM | AV_OPT_FLAG_AUDIO_PARAM
1723 static const AVOption aacenc_options[] = {
1724  {"aac_coder", "Coding algorithm", offsetof(AACEncContext, options.coder), AV_OPT_TYPE_INT, {.i64 = AAC_CODER_NMR}, 0, AAC_CODER_NB-1, AACENC_FLAGS, .unit = "coder"},
1725  {"twoloop", "Two loop searching method", 0, AV_OPT_TYPE_CONST, {.i64 = AAC_CODER_TWOLOOP}, INT_MIN, INT_MAX, AACENC_FLAGS, .unit = "coder"},
1726  {"fast", "Fast search", 0, AV_OPT_TYPE_CONST, {.i64 = AAC_CODER_FAST}, INT_MIN, INT_MAX, AACENC_FLAGS, .unit = "coder"},
1727  {"nmr", "Noise-to-mask ratio scalefactor trellis", 0, AV_OPT_TYPE_CONST, {.i64 = AAC_CODER_NMR}, INT_MIN, INT_MAX, AACENC_FLAGS, .unit = "coder"},
1728  {"aac_ms", "Force M/S stereo coding", offsetof(AACEncContext, options.mid_side), AV_OPT_TYPE_BOOL, {.i64 = -1}, -1, 1, AACENC_FLAGS},
1729  {"aac_is", "Intensity stereo coding", offsetof(AACEncContext, options.intensity_stereo), AV_OPT_TYPE_BOOL, {.i64 = 1}, -1, 1, AACENC_FLAGS},
1730  {"aac_pns", "Perceptual noise substitution", offsetof(AACEncContext, options.pns), AV_OPT_TYPE_BOOL, {.i64 = 1}, -1, 1, AACENC_FLAGS},
1731  {"aac_tns", "Temporal noise shaping", offsetof(AACEncContext, options.tns), AV_OPT_TYPE_BOOL, {.i64 = 1}, -1, 1, AACENC_FLAGS},
1732  {"aac_pce", "Forces the use of PCEs", offsetof(AACEncContext, options.pce), AV_OPT_TYPE_BOOL, {.i64 = 0}, -1, 1, AACENC_FLAGS},
1733  {"aac_nmr_speed", "NMR coder speed level: 0 = slowest/best, higher trades quality for speed", offsetof(AACEncContext, options.nmr_speed), AV_OPT_TYPE_INT, {.i64 = 0}, 0, 4, AACENC_FLAGS},
1735  {NULL}
1736 };
1737 
1738 static const AVClass aacenc_class = {
1739  .class_name = "AAC encoder",
1740  .item_name = av_default_item_name,
1741  .option = aacenc_options,
1742  .version = LIBAVUTIL_VERSION_INT,
1743 };
1744 
1746  { "b", "0" },
1747  { NULL }
1748 };
1749 
1751  .p.name = "aac",
1752  CODEC_LONG_NAME("AAC (Advanced Audio Coding)"),
1753  .p.type = AVMEDIA_TYPE_AUDIO,
1754  .p.id = AV_CODEC_ID_AAC,
1755  .p.capabilities = AV_CODEC_CAP_DR1 | AV_CODEC_CAP_DELAY |
1757  .priv_data_size = sizeof(AACEncContext),
1758  .init = aac_encode_init,
1760  .close = aac_encode_end,
1761  .defaults = aac_encode_defaults,
1763  .caps_internal = FF_CODEC_CAP_INIT_CLEANUP,
1765  .p.priv_class = &aacenc_class,
1766 };
FF_ALLOCZ_TYPED_ARRAY
#define FF_ALLOCZ_TYPED_ARRAY(p, nelem)
Definition: internal.h:72
AVCodecContext::frame_size
int frame_size
Number of samples per channel in an audio frame.
Definition: avcodec.h:1068
AV_SAMPLE_FMT_FLTP
@ AV_SAMPLE_FMT_FLTP
float, planar
Definition: samplefmt.h:66
name
it s the only field you need to keep assuming you have a context There is some magic you don t need to care about around this just let it vf default minimum maximum flags name is the option name
Definition: writing_filters.txt:88
ff_tns_max_bands_128
const uint8_t ff_tns_max_bands_128[]
Definition: aactab.c:1990
AV_CHANNEL_LAYOUT_OCTAGONAL
#define AV_CHANNEL_LAYOUT_OCTAGONAL
Definition: channel_layout.h:422
FF_CODEC_CAP_INIT_CLEANUP
#define FF_CODEC_CAP_INIT_CLEANUP
The codec allows calling the close function for deallocation even if the init function returned a fai...
Definition: codec_internal.h:43
aacenc_class
static const AVClass aacenc_class
Definition: aacenc.c:1738
aac_normal_chan_layouts
static const AVChannelLayout aac_normal_chan_layouts[15]
Definition: aacenctab.h:47
NMR_DECORR_LO
#define NMR_DECORR_LO
Definition: aacenc.c:593
r
const char * r
Definition: vf_curves.c:127
AVERROR
Filter the word “frame” indicates either a video frame or a group of audio as stored in an AVFrame structure Format for each input and each output the list of supported formats For video that means pixel format For audio that means channel sample they are references to shared objects When the negotiation mechanism computes the intersection of the formats supported at each end of a all references to both lists are replaced with a reference to the intersection And when a single format is eventually chosen for a link amongst the remaining all references to the list are updated That means that if a filter requires that its input and output have the same format amongst a supported all it has to do is use a reference to the same list of formats query_formats can leave some formats unset and return AVERROR(EAGAIN) to cause the negotiation mechanism toagain later. That can be used by filters with complex requirements to use the format negotiated on one link to set the formats supported on another. Frame references ownership and permissions
opt.h
SingleChannelElement::can_pns
uint8_t can_pns[128]
band is allowed to PNS (informative)
Definition: aacenc.h:117
LIBAVCODEC_IDENT
#define LIBAVCODEC_IDENT
Definition: version.h:43
put_bitstream_info
static void put_bitstream_info(AACEncContext *s, const char *name)
Write some auxiliary information about the created AAC file.
Definition: aacenc.c:1023
ff_aac_kbd_short_128
float ff_aac_kbd_short_128[128]
libm.h
SingleChannelElement::pulse
Pulse pulse
Definition: aacenc.h:112
align_put_bits
static void align_put_bits(PutBitContext *s)
Pad the bitstream with zeros up to the next byte boundary.
Definition: put_bits.h:445
TYPE_FIL
@ TYPE_FIL
Definition: aac.h:50
out
static FILE * out
Definition: movenc.c:55
AV_CHANNEL_LAYOUT_STEREO
#define AV_CHANNEL_LAYOUT_STEREO
Definition: channel_layout.h:395
put_bytes_output
static int put_bytes_output(const PutBitContext *s)
Definition: put_bits.h:99
AVCodecContext::sample_rate
int sample_rate
samples per second
Definition: avcodec.h:1040
AV_CHANNEL_LAYOUT_4POINT1
#define AV_CHANNEL_LAYOUT_4POINT1
Definition: channel_layout.h:401
aacenctab.h
AV_CHANNEL_LAYOUT_HEXAGONAL
#define AV_CHANNEL_LAYOUT_HEXAGONAL
Definition: channel_layout.h:411
copy_input_samples
static void copy_input_samples(AACEncContext *s, const AVFrame *frame)
Definition: aacenc.c:1044
aac_encode_init
static av_cold int aac_encode_init(AVCodecContext *avctx)
Definition: aacenc.c:1563
aacenc_profiles
static const int aacenc_profiles[]
Definition: aacenctab.h:145
Pulse::num_pulse
int num_pulse
Definition: aac.h:104
AV_CODEC_FLAG_QSCALE
#define AV_CODEC_FLAG_QSCALE
Use fixed qscale.
Definition: avcodec.h:213
av_cold
#define av_cold
Definition: attributes.h:119
int64_t
long long int64_t
Definition: coverity.c:34
output
filter_frame For filters that do not use the this method is called when a frame is pushed to the filter s input It can be called at any time except in a reentrant way If the input frame is enough to produce output
Definition: filter_design.txt:226
SingleChannelElement::zeroes
uint8_t zeroes[128]
band is not coded
Definition: aacenc.h:116
init_put_bits
static void init_put_bits(PutBitContext *s, uint8_t *buffer, int buffer_size)
Initialize the PutBitContext s.
Definition: put_bits.h:62
ff_af_queue_init
av_cold void ff_af_queue_init(AVCodecContext *avctx, AudioFrameQueue *afq)
Initialize AudioFrameQueue.
Definition: audio_frame_queue.c:29
ff_lpc_init
av_cold int ff_lpc_init(LPCContext *s, int blocksize, int max_order, enum FFLPCType lpc_type)
Initialize LPCContext.
Definition: lpc.c:342
AV_CHANNEL_LAYOUT_2_2
#define AV_CHANNEL_LAYOUT_2_2
Definition: channel_layout.h:402
NMR_IS_IMG_GATE
#define NMR_IS_IMG_GATE
Definition: aacenc.c:581
AVFrame
This structure describes decoded (raw) audio or video data.
Definition: frame.h:472
put_bits
static void put_bits(Jpeg2000EncoderContext *s, int val, int n)
put n times val bit
Definition: j2kenc.c:154
WARN_IF
#define WARN_IF(cond,...)
Definition: aacenc_utils.h:250
AVPacket::data
uint8_t * data
Definition: packet.h:603
ff_aac_coders
const AACCoefficientsEncoder ff_aac_coders[AAC_CODER_NB]
Definition: aaccoder.c:827
AVOption
AVOption.
Definition: opt.h:428
encode.h
b
#define b
Definition: input.c:43
R
#define R
Definition: huffyuv.h:44
NMR_MS_MASK
#define NMR_MS_MASK
Definition: aacenc.c:588
encode_band_info
static void encode_band_info(AACEncContext *s, SingleChannelElement *sce)
Encode scalefactor band coding type.
Definition: aacenc.c:885
AV_PROFILE_MPEG2_AAC_LOW
#define AV_PROFILE_MPEG2_AAC_LOW
Definition: defs.h:77
TemporalNoiseShaping::present
int present
Definition: aacdec.h:192
FFCodec
Definition: codec_internal.h:127
version.h
FFPsyWindowInfo::window_shape
int window_shape
window shape (sine/KBD/whatever)
Definition: psymodel.h:79
float.h
AAC_CODER_NB
@ AAC_CODER_NB
Definition: aacenc.h:49
max
#define max(a, b)
Definition: cuda_runtime.h:33
FFMAX
#define FFMAX(a, b)
Definition: macros.h:47
AVChannelLayout::nb_channels
int nb_channels
Number of channels in this layout.
Definition: channel_layout.h:329
ChannelElement::ch
SingleChannelElement ch[2]
Definition: aacdec.h:302
AV_PKT_FLAG_KEY
#define AV_PKT_FLAG_KEY
The packet contains a keyframe.
Definition: packet.h:650
ff_swb_offset_128
const uint16_t *const ff_swb_offset_128[]
Definition: aactab.c:1940
av_tx_init
av_cold int av_tx_init(AVTXContext **ctx, av_tx_fn *tx, enum AVTXType type, int inv, int len, const void *scale, uint64_t flags)
Initialize a transform context with the given configuration (i)MDCTs with an odd length are currently...
Definition: tx.c:903
encode_spectral_coeffs
static void encode_spectral_coeffs(AACEncContext *s, SingleChannelElement *sce)
Encode spectral coefficients processed by psychoacoustic model.
Definition: aacenc.c:954
ff_tns_max_bands_1024
const uint8_t ff_tns_max_bands_1024[]
Definition: aactab.c:1974
AAC_CODER_FAST
@ AAC_CODER_FAST
Definition: aacenc.h:46
IndividualChannelStream::num_swb
int num_swb
number of scalefactor window bands
Definition: aacdec.h:178
AV_CHANNEL_LAYOUT_7POINT1_WIDE
#define AV_CHANNEL_LAYOUT_7POINT1_WIDE
Definition: channel_layout.h:418
WINDOW_FUNC
#define WINDOW_FUNC(type)
Definition: aacenc.c:382
SingleChannelElement::coeffs
float coeffs[1024]
coefficients for IMDCT, maybe processed
Definition: aacenc.h:121
avoid_clipping
static void avoid_clipping(AACEncContext *s, SingleChannelElement *sce)
Downscale spectral coefficients for near-clipping windows to avoid artifacts.
Definition: aacenc.c:982
put_audio_specific_config
static int put_audio_specific_config(AVCodecContext *avctx, int chcfg)
Make AAC audio config object.
Definition: aacenc.c:342
FFCodecDefault
Definition: codec_internal.h:97
FFCodec::p
AVCodec p
The public AVCodec.
Definition: codec_internal.h:131
mpeg4audio.h
b1
static double b1(void *priv, double x, double y)
Definition: vf_xfade.c:2034
AVCodecContext::ch_layout
AVChannelLayout ch_layout
Audio channel layout.
Definition: avcodec.h:1055
SingleChannelElement::ret_buf
float ret_buf[2048]
PCM output buffer.
Definition: aacenc.h:122
apply_mid_side_stereo
static void apply_mid_side_stereo(ChannelElement *cpe)
Definition: aacenc.c:850
AV_CHANNEL_LAYOUT_2POINT1
#define AV_CHANNEL_LAYOUT_2POINT1
Definition: channel_layout.h:396
TYPE_CPE
@ TYPE_CPE
Definition: aac.h:45
ChannelElement::ms_mode
int ms_mode
Signals mid/side stereo flags coding mode.
Definition: aacenc.h:132
AVCodecContext::initial_padding
int initial_padding
Audio only.
Definition: avcodec.h:1114
IndividualChannelStream::window_clipping
uint8_t window_clipping[8]
set if a certain window is near clipping
Definition: aacdec.h:185
AVCodecContext::flags
int flags
AV_CODEC_FLAG_*.
Definition: avcodec.h:500
Pulse::amp
int amp[4]
Definition: aac.h:107
Pulse::pos
int pos[4]
Definition: aac.h:106
AVCodecContext::bit_rate_tolerance
int bit_rate_tolerance
number of bits the bitstream is allowed to diverge from the reference.
Definition: avcodec.h:1227
NMR_IS_LOW_LIMIT
#define NMR_IS_LOW_LIMIT
Definition: aacenc.c:584
put_pce
static void put_pce(PutBitContext *pb, AVCodecContext *avctx)
Definition: aacenc.c:301
ff_psy_end
av_cold void ff_psy_end(FFPsyContext *ctx)
Cleanup model context at the end.
Definition: psymodel.c:77
Pulse::start
int start
Definition: aac.h:105
FF_CODEC_ENCODE_CB
#define FF_CODEC_ENCODE_CB(func)
Definition: codec_internal.h:376
fabsf
static __device__ float fabsf(float a)
Definition: cuda_runtime.h:181
ff_af_queue_add
int ff_af_queue_add(AudioFrameQueue *afq, const AVFrame *f)
Add a frame to the queue.
Definition: audio_frame_queue.c:46
AV_CHANNEL_LAYOUT_6POINT1_FRONT
#define AV_CHANNEL_LAYOUT_6POINT1_FRONT
Definition: channel_layout.h:414
SingleChannelElement::ics
IndividualChannelStream ics
Definition: aacdec.h:218
AV_CHANNEL_LAYOUT_SURROUND
#define AV_CHANNEL_LAYOUT_SURROUND
Definition: channel_layout.h:398
FFPsyWindowInfo
windowing related information
Definition: psymodel.h:77
NMR_SDEC_EMA
#define NMR_SDEC_EMA
Definition: aacenc.c:599
NMR_PNS_STEREO_DECORR
#define NMR_PNS_STEREO_DECORR
Definition: aacenc.c:602
adjust_frame_information
static void adjust_frame_information(ChannelElement *cpe, int chans)
Produce integer coefficients from scalefactors provided by the model.
Definition: aacenc.c:503
AV_LOG_ERROR
#define AV_LOG_ERROR
Something went wrong and cannot losslessly be recovered.
Definition: log.h:210
FF_ARRAY_ELEMS
#define FF_ARRAY_ELEMS(a)
Definition: sinewin_tablegen.c:29
AV_PROFILE_UNKNOWN
#define AV_PROFILE_UNKNOWN
Definition: defs.h:65
IndividualChannelStream::clip_avoidance_factor
float clip_avoidance_factor
set if any window is near clipping to the necessary atennuation factor to avoid it
Definition: aacenc.h:90
AACPCEInfo::index
uint8_t index[4][8]
front, side, back, lfe
Definition: aacenc.h:250
nmr_apply_is_band
static void nmr_apply_is_band(AACEncContext *s, ChannelElement *cpe, int w, int g, int start, int len, int gl, float scale, float sr_, int p, float ener0, float ener1)
Definition: aacenc.c:661
av_channel_layout_describe
int av_channel_layout_describe(const AVChannelLayout *channel_layout, char *buf, size_t buf_size)
Get a human-readable string describing the channel layout properties.
Definition: channel_layout.c:654
AV_CHANNEL_LAYOUT_4POINT0
#define AV_CHANNEL_LAYOUT_4POINT0
Definition: channel_layout.h:400
float
float
Definition: af_crystalizer.c:122
AVCodecContext::extradata_size
int extradata_size
Definition: avcodec.h:527
NOISE_BT
@ NOISE_BT
Spectral data are scaled white noise not coded in the bitstream.
Definition: aac.h:75
AV_TX_FLOAT_MDCT
@ AV_TX_FLOAT_MDCT
Standard MDCT with a sample data type of float, double or int32_t, respectively.
Definition: tx.h:68
AV_CHANNEL_LAYOUT_7POINT1
#define AV_CHANNEL_LAYOUT_7POINT1
Definition: channel_layout.h:417
AVCodecContext::global_quality
int global_quality
Global quality for codecs which cannot change it per frame.
Definition: avcodec.h:1235
IndividualChannelStream::swb_sizes
const uint8_t * swb_sizes
table of scalefactor band sizes for a particular window
Definition: aacenc.h:85
g
const char * g
Definition: vf_curves.c:128
AVMEDIA_TYPE_AUDIO
@ AVMEDIA_TYPE_AUDIO
Definition: avutil.h:201
EIGHT_SHORT_SEQUENCE
@ EIGHT_SHORT_SEQUENCE
Definition: aac.h:66
info
MIPS optimizations info
Definition: mips.txt:2
AV_CHANNEL_LAYOUT_5POINT0_BACK
#define AV_CHANNEL_LAYOUT_5POINT0_BACK
Definition: channel_layout.h:406
INTENSITY_BT2
@ INTENSITY_BT2
Scalefactor data are intensity stereo positions (out of phase).
Definition: aac.h:76
av_assert0
#define av_assert0(cond)
assert() equivalent, that is always enabled.
Definition: avassert.h:42
alloc_buffers
static av_cold int alloc_buffers(AVCodecContext *avctx, AACEncContext *s)
Definition: aacenc.c:1544
channels
channels
Definition: aptx.h:31
channel_map
static const uint8_t channel_map[8][8]
Definition: atrac3plusdec.c:52
ff_put_string
void ff_put_string(PutBitContext *pb, const char *string, int terminate_string)
Put the string string in the bitstream.
Definition: bitstream.c:39
IndividualChannelStream
Individual Channel Stream.
Definition: aacdec.h:169
av_mallocz
#define av_mallocz(s)
Definition: tableprint_vlc.h:31
SCALE_DIFF_ZERO
#define SCALE_DIFF_ZERO
codebook index corresponding to zero scalefactor indices difference
Definition: aac.h:95
NAN
#define NAN
Definition: mathematics.h:115
NOISE_PRE
#define NOISE_PRE
preamble for NOISE_BT, put in bitstream with the first noise band
Definition: aac.h:99
PutBitContext
Definition: put_bits.h:50
CODEC_LONG_NAME
#define CODEC_LONG_NAME(str)
Definition: codec_internal.h:349
if
if(ret)
Definition: filter_design.txt:179
ff_af_queue_close
av_cold void ff_af_queue_close(AudioFrameQueue *afq)
Close AudioFrameQueue.
Definition: audio_frame_queue.c:38
AV_CHANNEL_LAYOUT_7POINT1_WIDE_BACK
#define AV_CHANNEL_LAYOUT_7POINT1_WIDE_BACK
Definition: channel_layout.h:419
INTENSITY_BT
@ INTENSITY_BT
Scalefactor data are intensity stereo positions (in phase).
Definition: aac.h:77
FFPsyWindowInfo::window_type
int window_type[3]
window type (short/long/transitional, etc.) - current, previous and next
Definition: psymodel.h:78
AAC_MAX_CHANNELS
#define AAC_MAX_CHANNELS
Definition: aacenctab.h:41
LIBAVUTIL_VERSION_INT
#define LIBAVUTIL_VERSION_INT
Definition: version.h:85
AVClass
Describe the class of an AVClass context structure.
Definition: log.h:76
fabs
static __device__ float fabs(float a)
Definition: cuda_runtime.h:182
ChannelElement::is_mask
uint8_t is_mask[128]
Set if intensity stereo is used.
Definition: aacenc.h:135
NULL
#define NULL
Definition: coverity.c:32
AACPCEInfo::pairing
uint8_t pairing[3][8]
front, side, back
Definition: aacenc.h:249
sizes
static const int sizes[][2]
Definition: img2dec.c:62
encode_pulses
static void encode_pulses(AACEncContext *s, Pulse *pulse)
Encode pulse data.
Definition: aacenc.c:935
SingleChannelElement::is_ener
float is_ener[128]
Intensity stereo pos.
Definition: aacenc.h:118
IndividualChannelStream::use_kb_window
uint8_t use_kb_window[2]
If set, use Kaiser-Bessel window, otherwise use a sine window.
Definition: aacdec.h:172
ff_aac_num_swb_128
const uint8_t ff_aac_num_swb_128[]
Definition: aactab.c:169
nmr_is_image_masked
static int nmr_is_image_masked(AACEncContext *s, ChannelElement *cpe, int w, int g, int start, int len, int gl, float ener0, float ener1, float dot, float minthr0, float minthr1, float *ratio_out, float *scale_out, float *sr_out, int *p_out)
Definition: aacenc.c:629
AVCodecContext::bit_rate
int64_t bit_rate
the average bitrate
Definition: avcodec.h:493
av_default_item_name
const char * av_default_item_name(void *ptr)
Return the context name.
Definition: log.c:242
profiles.h
ff_lpc_end
av_cold void ff_lpc_end(LPCContext *s)
Uninitialize LPCContext.
Definition: lpc.c:367
ChannelElement::ms_mask
uint8_t ms_mask[128]
Set if mid/side stereo is used for each scalefactor window band.
Definition: aacdec.h:300
options
Definition: swscale.c:50
FFPsyBand
single band psychoacoustic information
Definition: psymodel.h:50
aac.h
aactab.h
nmr_decide_stereo
static void nmr_decide_stereo(AACEncContext *s, ChannelElement *cpe)
Definition: aacenc.c:691
sqrtf
static __device__ float sqrtf(float a)
Definition: cuda_runtime.h:184
FFPsyWindowInfo::grouping
int grouping[8]
window grouping (for e.g. AAC)
Definition: psymodel.h:81
av_clipf
av_clipf
Definition: af_crystalizer.c:122
TNS_MAX_ORDER
#define TNS_MAX_ORDER
Definition: aac.h:36
c
Undefined Behavior In the C some operations are like signed integer dereferencing freed accessing outside allocated Undefined Behavior must not occur in a C it is not safe even if the output of undefined operations is unused The unsafety may seem nit picking but Optimizing compilers have in fact optimized code on the assumption that no undefined Behavior occurs Optimizing code based on wrong assumptions can and has in some cases lead to effects beyond the output of computations The signed integer overflow problem in speed critical code Code which is highly optimized and works with signed integers sometimes has the problem that often the output of the computation does not c
Definition: undefined.txt:32
SingleChannelElement::sf_idx
int sf_idx[128]
scalefactor indices
Definition: aacenc.h:115
float_dsp.h
AV_CODEC_ID_AAC
@ AV_CODEC_ID_AAC
Definition: codec_id.h:454
aac_encode_frame
static int aac_encode_frame(AVCodecContext *avctx, AVPacket *avpkt, const AVFrame *frame, int *got_packet_ptr)
Definition: aacenc.c:1066
ff_aac_scalefactor_bits
const uint8_t ff_aac_scalefactor_bits[121]
Definition: aactab.c:200
NMR_MS_EQUIV
#define NMR_MS_EQUIV
Definition: aacenc.c:587
AACPCEInfo
Definition: aacenc.h:246
FFPsyWindowInfo::clipping
float clipping[8]
maximum absolute normalized intensity in the given window for clip avoidance
Definition: psymodel.h:82
IndividualChannelStream::window_sequence
enum WindowSequence window_sequence[2]
Definition: aacdec.h:171
f
f
Definition: af_crystalizer.c:122
init
int(* init)(AVBSFContext *ctx)
Definition: dts2pts.c:608
AV_CODEC_CAP_DR1
#define AV_CODEC_CAP_DR1
Codec uses get_buffer() or get_encode_buffer() for allocating buffers and supports custom allocators.
Definition: codec.h:49
ff_swb_offset_1024
const uint16_t *const ff_swb_offset_1024[]
Definition: aactab.c:1900
AVPacket::size
int size
Definition: packet.h:604
codec_internal.h
ONLY_LONG_SEQUENCE
@ ONLY_LONG_SEQUENCE
Definition: aac.h:64
TYPE_END
@ TYPE_END
Definition: aac.h:51
ff_aac_float_common_init
void ff_aac_float_common_init(void)
i
#define i(width, name, range_min, range_max)
Definition: cbs_h264.c:63
encode_scale_factors
static void encode_scale_factors(AVCodecContext *avctx, AACEncContext *s, SingleChannelElement *sce)
Encode scalefactors.
Definition: aacenc.c:899
for
for(k=2;k<=8;++k)
Definition: h264pred_template.c:424
apply_window_and_mdct
static void apply_window_and_mdct(AACEncContext *s, SingleChannelElement *sce, float *audio)
Definition: aacenc.c:447
AVFloatDSPContext
Definition: float_dsp.h:24
AAC_CODER_TWOLOOP
@ AAC_CODER_TWOLOOP
Definition: aacenc.h:45
aac_chan_configs
static const uint8_t aac_chan_configs[14][6]
default channel configurations
Definition: aacenctab.h:66
AV_CHANNEL_LAYOUT_6POINT0
#define AV_CHANNEL_LAYOUT_6POINT0
Definition: channel_layout.h:408
diff
static av_always_inline int diff(const struct color_info *a, const struct color_info *b, const int trans_thresh)
Definition: vf_paletteuse.c:166
CLIP_AVOIDANCE_FACTOR
#define CLIP_AVOIDANCE_FACTOR
Definition: aacenc.h:42
ChannelElement::common_window
int common_window
Set if channels share a common 'IndividualChannelStream' in bitstream.
Definition: aacenc.h:131
sinewin.h
apply_intensity_stereo
static void apply_intensity_stereo(ChannelElement *cpe)
Definition: aacenc.c:551
AVPacket::flags
int flags
A combination of AV_PKT_FLAG values.
Definition: packet.h:609
ff_af_queue_remove
int ff_af_queue_remove(AudioFrameQueue *afq, int nb_samples, AVPacket *pkt)
Remove frame(s) from the queue.
Definition: audio_frame_queue.c:87
CODEC_SAMPLEFMTS
#define CODEC_SAMPLEFMTS(...)
Definition: codec_internal.h:395
nmr_apply_ms_band
static void nmr_apply_ms_band(AACEncContext *s, ChannelElement *cpe, int w, int g, int start, int len, int gl)
Definition: aacenc.c:605
SingleChannelElement::band_type
enum BandType band_type[128]
band types
Definition: aacdec.h:221
av_tx_uninit
av_cold void av_tx_uninit(AVTXContext **ctx)
Frees a context and sets *ctx to NULL, does nothing when *ctx == NULL.
Definition: tx.c:295
av_channel_layout_compare
int av_channel_layout_compare(const AVChannelLayout *chl, const AVChannelLayout *chl1)
Check whether two channel layouts are semantically the same, i.e.
Definition: channel_layout.c:811
AV_LOG_INFO
#define AV_LOG_INFO
Standard information.
Definition: log.h:221
AAC_CUTOFF_FROM_BITRATE
#define AAC_CUTOFF_FROM_BITRATE(bit_rate, channels, sample_rate)
Definition: psymodel.h:35
AV_CHANNEL_LAYOUT_6POINT1_BACK
#define AV_CHANNEL_LAYOUT_6POINT1_BACK
Definition: channel_layout.h:413
aac_pce_configs
static const AACPCEInfo aac_pce_configs[]
List of PCE (Program Configuration Element) for the channel layouts listed in channel_layout....
Definition: aacenc.c:90
SingleChannelElement
Single Channel Element - used for both SCE and LFE elements.
Definition: aacdec.h:217
put_bits_count
static int put_bits_count(PutBitContext *s)
Definition: put_bits.h:90
IndividualChannelStream::num_windows
int num_windows
Definition: aacdec.h:179
NMR_STICKY
#define NMR_STICKY
Definition: aacenc.c:596
AVCodecContext::extradata
uint8_t * extradata
Out-of-band global headers that may be used by some codecs.
Definition: avcodec.h:526
FFMIN3
#define FFMIN3(a, b, c)
Definition: macros.h:50
aacenc_options
static const AVOption aacenc_options[]
Definition: aacenc.c:1723
AV_CHANNEL_LAYOUT_QUAD
#define AV_CHANNEL_LAYOUT_QUAD
Definition: channel_layout.h:403
SingleChannelElement::pcoeffs
float pcoeffs[1024]
coefficients for IMDCT, pristine
Definition: aacenc.h:120
LONG_STOP_SEQUENCE
@ LONG_STOP_SEQUENCE
Definition: aac.h:67
ChannelElement
channel element - generic struct for SCE/CPE/CCE/LFE
Definition: aacdec.h:296
IndividualChannelStream::swb_offset
const uint16_t * swb_offset
table of offsets to the lowest spectral coefficient of a scalefactor band, sfb, for a particular wind...
Definition: aacdec.h:177
ff_psy_init
av_cold int ff_psy_init(FFPsyContext *ctx, AVCodecContext *avctx, int num_lens, const uint8_t **bands, const int *num_bands, int num_groups, const uint8_t *group_map, int cutoff)
Initialize psychoacoustic model.
Definition: psymodel.c:28
AVCodecContext::cutoff
int cutoff
Audio cutoff bandwidth (0 means "automatic")
Definition: avcodec.h:1082
AV_CHANNEL_LAYOUT_7POINT0_FRONT
#define AV_CHANNEL_LAYOUT_7POINT0_FRONT
Definition: channel_layout.h:416
apply_window
static void(*const apply_window[4])(AVFloatDSPContext *fdsp, SingleChannelElement *sce, const float *audio)
Definition: aacenc.c:438
av_assert1
#define av_assert1(cond)
assert() equivalent, that does not lie in speed critical code.
Definition: avassert.h:58
s
uint8_t s
Definition: llvidencdsp.c:39
NOISE_PRE_BITS
#define NOISE_PRE_BITS
length of preamble
Definition: aac.h:100
AV_CHANNEL_LAYOUT_3POINT1
#define AV_CHANNEL_LAYOUT_3POINT1
Definition: channel_layout.h:399
FFMIN
#define FFMIN(a, b)
Definition: macros.h:49
TYPE_LFE
@ TYPE_LFE
Definition: aac.h:47
ff_aac_kbd_long_1024
float ff_aac_kbd_long_1024[1024]
AACPCEInfo::num_ele
uint8_t num_ele[4]
front, side, back, lfe
Definition: aacenc.h:248
AVCodec::name
const char * name
Name of the codec implementation.
Definition: codec.h:176
TYPE_SCE
@ TYPE_SCE
Definition: aac.h:44
AACENC_FLAGS
#define AACENC_FLAGS
Definition: aacenc.c:1722
len
int len
Definition: vorbis_enc_data.h:426
IndividualChannelStream::tns_max_bands
int tns_max_bands
Definition: aacdec.h:180
AAC_CODER_NMR
@ AAC_CODER_NMR
Definition: aacenc.h:47
avcodec.h
AVCodecContext::frame_num
int64_t frame_num
Frame counter, set by libavcodec.
Definition: avcodec.h:1883
aac_encode_defaults
static const FFCodecDefault aac_encode_defaults[]
Definition: aacenc.c:1745
tag
uint32_t tag
Definition: movenc.c:2073
ret
ret
Definition: filter_design.txt:187
ff_aac_num_swb_1024
const uint8_t ff_aac_num_swb_1024[]
Definition: aactab.c:149
AVClass::class_name
const char * class_name
The name of the class; usually it is the same name as the context structure type to which the AVClass...
Definition: log.h:81
frame
these buffered frames must be flushed immediately if a new input produces new the filter must not call request_frame to get more It must just process the frame or queue it The task of requesting more frames is left to the filter s request_frame method or the application If a filter has several the filter must be ready for frames arriving randomly on any input any filter with several inputs will most likely require some kind of queuing mechanism It is perfectly acceptable to have a limited queue and to drop frames when the inputs are too unbalanced request_frame For filters that do not use the this method is called when a frame is wanted on an output For a it should directly call filter_frame on the corresponding output For a if there are queued frames already one of these frames should be pushed If the filter should request a frame on one of its repeatedly until at least one frame has been pushed Return or at least make progress towards producing a frame
Definition: filter_design.txt:265
ff_aac_encoder
const FFCodec ff_aac_encoder
Definition: aacenc.c:1750
encode_ms_info
static void encode_ms_info(PutBitContext *pb, ChannelElement *cpe)
Encode MS data.
Definition: aacenc.c:489
AV_CHANNEL_LAYOUT_7POINT0
#define AV_CHANNEL_LAYOUT_7POINT0
Definition: channel_layout.h:415
RESERVED_BT
@ RESERVED_BT
Band types following are encoded differently from others.
Definition: aac.h:74
LONG_START_SEQUENCE
@ LONG_START_SEQUENCE
Definition: aac.h:65
SingleChannelElement::tns
TemporalNoiseShaping tns
Definition: aacdec.h:220
AACEncContext
AAC encoder context.
Definition: aacenc.h:258
AV_PROFILE_AAC_LOW
#define AV_PROFILE_AAC_LOW
Definition: defs.h:69
AV_CHANNEL_LAYOUT_2_1
#define AV_CHANNEL_LAYOUT_2_1
Definition: channel_layout.h:397
AVCodecContext
main external API structure.
Definition: avcodec.h:443
channel_layout.h
CODEC_SAMPLERATES_ARRAY
#define CODEC_SAMPLERATES_ARRAY(array)
Definition: codec_internal.h:393
encode_individual_channel
static int encode_individual_channel(AVCodecContext *avctx, AACEncContext *s, SingleChannelElement *sce, int common_window)
Encode one channel of audio data.
Definition: aacenc.c:1002
NOISE_OFFSET
#define NOISE_OFFSET
subtracted from global gain, used as offset for the preamble
Definition: aac.h:101
ERROR_IF
#define ERROR_IF(cond,...)
Definition: aacenc_utils.h:244
rates
static const int rates[]
Definition: swresample.c:101
ff_aac_swb_size_1024
const uint8_t *const ff_aac_swb_size_1024[]
Definition: aacenctab.c:97
AV_OPT_TYPE_INT
@ AV_OPT_TYPE_INT
Underlying C type is int.
Definition: opt.h:258
TemporalNoiseShaping
Temporal Noise Shaping.
Definition: aacdec.h:191
AVCodecContext::profile
int profile
profile
Definition: avcodec.h:1636
AOT_SBR
@ AOT_SBR
Y Spectral Band Replication.
Definition: mpeg4audio.h:78
L
#define L(x)
Definition: vpx_arith.h:36
AV_CHANNEL_LAYOUT_6POINT0_FRONT
#define AV_CHANNEL_LAYOUT_6POINT0_FRONT
Definition: channel_layout.h:409
AV_CODEC_CAP_DELAY
#define AV_CODEC_CAP_DELAY
Encoder or decoder requires flushing with NULL input at the end in order to give the complete and cor...
Definition: codec.h:73
samples
Filter the word “frame” indicates either a video frame or a group of audio samples
Definition: filter_design.txt:8
ChannelElement::is_mode
uint8_t is_mode
Set if any bands have been encoded using intensity stereo.
Definition: aacenc.h:133
Windows::Graphics::DirectX::Direct3D11::p
IDirect3DDxgiInterfaceAccess _COM_Outptr_ void ** p
Definition: vsrc_gfxcapture_winrt.hpp:53
ff_aacenc_dsp_init
void ff_aacenc_dsp_init(AACEncDSPContext *s)
Definition: aacencdsp.c:75
put_ics_info
static void put_ics_info(AACEncContext *s, IndividualChannelStream *info)
Encode ics_info element.
Definition: aacenc.c:468
ff_mpeg4audio_sample_rates
const int ff_mpeg4audio_sample_rates[16]
Definition: mpeg4audio_sample_rates.h:30
ff_aac_swb_size_128
const uint8_t *const ff_aac_swb_size_128[]
Definition: aacenctab.c:89
mem.h
AV_CODEC_FLAG_BITEXACT
#define AV_CODEC_FLAG_BITEXACT
Use only bitexact stuff (except (I)DCT).
Definition: avcodec.h:322
aac_encode_end
static av_cold int aac_encode_end(AVCodecContext *avctx)
Definition: aacenc.c:1499
flush_put_bits
static void flush_put_bits(PutBitContext *s)
Pad the end of the output stream with zeros.
Definition: put_bits.h:153
w
uint8_t w
Definition: llvidencdsp.c:39
AV_CHANNEL_LAYOUT_MONO
#define AV_CHANNEL_LAYOUT_MONO
Definition: channel_layout.h:394
scale
static void scale(int *out, const int *in, const int w, const int h, const int shift)
Definition: intra.c:278
FF_AAC_PROFILE_OPTS
#define FF_AAC_PROFILE_OPTS
Definition: profiles.h:29
AVPacket
This structure stores compressed data.
Definition: packet.h:580
AVCodecContext::priv_data
void * priv_data
Definition: avcodec.h:470
AV_OPT_TYPE_BOOL
@ AV_OPT_TYPE_BOOL
Underlying C type is int.
Definition: opt.h:326
av_freep
#define av_freep(p)
Definition: tableprint_vlc.h:35
avpriv_float_dsp_alloc
av_cold AVFloatDSPContext * avpriv_float_dsp_alloc(int bit_exact)
Allocate a float DSP context.
Definition: float_dsp.c:135
AV_CHANNEL_LAYOUT_5POINT1_BACK
#define AV_CHANNEL_LAYOUT_5POINT1_BACK
Definition: channel_layout.h:407
IndividualChannelStream::max_sfb
uint8_t max_sfb
number of scalefactor bands per group
Definition: aacdec.h:170
Pulse
Definition: aac.h:103
av_log
#define av_log(a,...)
Definition: tableprint_vlc.h:27
AV_CHANNEL_LAYOUT_6POINT1
#define AV_CHANNEL_LAYOUT_6POINT1
Definition: channel_layout.h:412
b0
static double b0(void *priv, double x, double y)
Definition: vf_xfade.c:2033
dsp_init
static av_cold int dsp_init(AVCodecContext *avctx, AACEncContext *s)
Definition: aacenc.c:1525
AV_CHANNEL_LAYOUT_5POINT0
#define AV_CHANNEL_LAYOUT_5POINT0
Definition: channel_layout.h:404
aacenc_utils.h
aac_chan_maps
static const uint8_t aac_chan_maps[14][AAC_MAX_CHANNELS]
Table to remap channels from libavcodec's default order to AAC order.
Definition: aacenctab.h:86
AV_CODEC_CAP_SMALL_LAST_FRAME
#define AV_CODEC_CAP_SMALL_LAST_FRAME
Codec can be fed a final frame with a smaller size.
Definition: codec.h:78
AV_CHANNEL_LAYOUT_5POINT1
#define AV_CHANNEL_LAYOUT_5POINT1
Definition: channel_layout.h:405
put_bits.h
IndividualChannelStream::group_len
uint8_t group_len[8]
Definition: aacdec.h:175
psymodel.h
AV_OPT_TYPE_CONST
@ AV_OPT_TYPE_CONST
Special option type for declaring named constants.
Definition: opt.h:298
ff_alloc_packet
int ff_alloc_packet(AVCodecContext *avctx, AVPacket *avpkt, int64_t size)
Check AVPacket size and allocate data.
Definition: encode.c:61
FF_LPC_TYPE_LEVINSON
@ FF_LPC_TYPE_LEVINSON
Levinson-Durbin recursion.
Definition: lpc.h:46
FFPsyWindowInfo::num_windows
int num_windows
number of windows in a frame
Definition: psymodel.h:80
ff_aac_scalefactor_code
const uint32_t ff_aac_scalefactor_code[121]
Definition: aactab.c:181
ff_quantize_band_cost_cache_init
void ff_quantize_band_cost_cache_init(struct AACEncContext *s)
Definition: aacenc.c:373
AACPCEInfo::layout
AVChannelLayout layout
Definition: aacenc.h:247
aacenc.h