Coverage Report

Created: 2026-07-25 07:52

next uncovered line (L), next uncovered region (R), next uncovered branch (B)
/src/ffmpeg/libavcodec/aacenc.c
Line
Count
Source
1
/*
2
 * AAC encoder
3
 * Copyright (C) 2008 Konstantin Shishkov
4
 *
5
 * This file is part of FFmpeg.
6
 *
7
 * FFmpeg is free software; you can redistribute it and/or
8
 * modify it under the terms of the GNU Lesser General Public
9
 * License as published by the Free Software Foundation; either
10
 * version 2.1 of the License, or (at your option) any later version.
11
 *
12
 * FFmpeg is distributed in the hope that it will be useful,
13
 * but WITHOUT ANY WARRANTY; without even the implied warranty of
14
 * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE.  See the GNU
15
 * Lesser General Public License for more details.
16
 *
17
 * You should have received a copy of the GNU Lesser General Public
18
 * License along with FFmpeg; if not, write to the Free Software
19
 * Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA
20
 */
21
22
/**
23
 * @file
24
 * AAC encoder
25
 */
26
27
/***********************************
28
 *              TODOs:
29
 * add sane pulse detection
30
 ***********************************/
31
#include <float.h>
32
33
#include "libavutil/channel_layout.h"
34
#include "libavutil/libm.h"
35
#include "libavutil/float_dsp.h"
36
#include "libavutil/mem.h"
37
#include "libavutil/opt.h"
38
#include "avcodec.h"
39
#include "codec_internal.h"
40
#include "encode.h"
41
#include "put_bits.h"
42
#include "mpeg4audio.h"
43
#include "sinewin.h"
44
#include "profiles.h"
45
#include "version.h"
46
47
#include "aac.h"
48
#include "aactab.h"
49
#include "aacenc.h"
50
#include "aacenctab.h"
51
#include "aacenc_utils.h"
52
53
#include "psymodel.h"
54
55
/**
56
 * List of PCE (Program Configuration Element) for the channel layouts listed
57
 * in channel_layout.h
58
 *
59
 * For those wishing in the future to add other layouts:
60
 *
61
 * - num_ele: number of elements in each group of front, side, back, lfe channels
62
 *            (an element is of type SCE (single channel), CPE (channel pair) for
63
 *            the first 3 groups; and is LFE for LFE group).
64
 *
65
 * - pairing: 0 for an SCE element or 1 for a CPE; does not apply to LFE group
66
 *
67
 * - index: there are three independent indices for SCE, CPE and LFE;
68
 *     they are incremented irrespective of the group to which the element belongs;
69
 *     they are not reset when going from one group to another
70
 *
71
 *     Example: for 7.0 channel layout,
72
 *        .pairing = { { 1, 0 }, { 1 }, { 1 }, }, (3 CPE and 1 SCE in front group)
73
 *        .index = { { 0, 0 }, { 1 }, { 2 }, },
74
 *               (index is 0 for the single SCE but goes from 0 to 2 for the CPEs)
75
 *
76
 *     The index order impacts the channel ordering. But is otherwise arbitrary
77
 *     (the sequence could have been 2, 0, 1 instead of 0, 1, 2).
78
 *
79
 *     Spec allows for discontinuous indices, e.g. if one has a total of two SCE,
80
 *     SCE.0 SCE.15 is OK per spec; BUT it won't be decoded by our AAC decoder
81
 *     which at this time requires that indices fully cover some range starting
82
 *     from 0 (SCE.1 SCE.0 is OK but not SCE.0 SCE.15).
83
 *
84
 * - config_map: total number of elements and their types. Beware, the way the
85
 *               types are ordered impacts the final channel ordering.
86
 *
87
 * - reorder_map: reorders the channels.
88
 *
89
 */
90
static const AACPCEInfo aac_pce_configs[] = {
91
    {
92
        .layout = AV_CHANNEL_LAYOUT_MONO,
93
        .num_ele = { 1, 0, 0, 0 },
94
        .pairing = { { 0 }, },
95
        .index = { { 0 }, },
96
        .config_map = { 1, TYPE_SCE, },
97
        .reorder_map = { 0 },
98
    },
99
    {
100
        .layout = AV_CHANNEL_LAYOUT_STEREO,
101
        .num_ele = { 1, 0, 0, 0 },
102
        .pairing = { { 1 }, },
103
        .index = { { 0 }, },
104
        .config_map = { 1, TYPE_CPE, },
105
        .reorder_map = { 0, 1 },
106
    },
107
    {
108
        .layout = AV_CHANNEL_LAYOUT_2POINT1,
109
        .num_ele = { 1, 0, 0, 1 },
110
        .pairing = { { 1 }, },
111
        .index = { { 0 },{ 0 },{ 0 },{ 0 } },
112
        .config_map = { 2, TYPE_CPE, TYPE_LFE },
113
        .reorder_map = { 0, 1, 2 },
114
    },
115
    {
116
        .layout = AV_CHANNEL_LAYOUT_2_1,
117
        .num_ele = { 1, 0, 1, 0 },
118
        .pairing = { { 1 },{ 0 },{ 0 } },
119
        .index = { { 0 },{ 0 },{ 0 }, },
120
        .config_map = { 2, TYPE_CPE, TYPE_SCE },
121
        .reorder_map = { 0, 1, 2 },
122
    },
123
    {
124
        .layout = AV_CHANNEL_LAYOUT_SURROUND,
125
        .num_ele = { 2, 0, 0, 0 },
126
        .pairing = { { 0, 1 }, },
127
        .index = { { 0, 0 }, },
128
        .config_map = { 2, TYPE_SCE, TYPE_CPE },
129
        .reorder_map = { 2, 0, 1 },
130
    },
131
    {
132
        .layout = AV_CHANNEL_LAYOUT_3POINT1,
133
        .num_ele = { 2, 0, 0, 1 },
134
        .pairing = { { 0, 1 }, },
135
        .index = { { 0, 0 }, { 0 }, { 0 }, { 0 }, },
136
        .config_map = { 3, TYPE_SCE, TYPE_CPE, TYPE_LFE },
137
        .reorder_map = { 2, 0, 1, 3 },
138
    },
139
    {
140
        .layout = AV_CHANNEL_LAYOUT_4POINT0,
141
        .num_ele = { 2, 0, 1, 0 },
142
        .pairing = { { 0, 1 }, { 0 }, { 0 }, },
143
        .index = { { 0, 0 }, { 0 }, { 1 } },
144
        .config_map = { 3, TYPE_SCE, TYPE_CPE, TYPE_SCE },
145
        .reorder_map = { 2, 0, 1, 3 },
146
    },
147
    {
148
        .layout = AV_CHANNEL_LAYOUT_4POINT1,
149
        .num_ele = { 2, 0, 1, 1 },
150
        .pairing = { { 0, 1 }, { 0 }, { 0 }, },
151
        .index = { { 0, 0 }, { 0 }, { 1 }, { 0 } },
152
        .config_map = { 4, TYPE_SCE, TYPE_CPE, TYPE_SCE, TYPE_LFE },
153
        .reorder_map = { 2, 0, 1, 4, 3 },
154
    },
155
    {
156
        .layout = AV_CHANNEL_LAYOUT_2_2,
157
        .num_ele = { 1, 0, 1, 0 },
158
        .pairing = { { 1 }, { 0 }, { 1 }, },
159
        .index = { { 0 }, { 0 }, { 1 } },
160
        .config_map = { 2, TYPE_CPE, TYPE_CPE },
161
        .reorder_map = { 0, 1, 2, 3 },
162
    },
163
    {
164
        .layout = AV_CHANNEL_LAYOUT_QUAD,
165
        .num_ele = { 1, 0, 1, 0 },
166
        .pairing = { { 1 }, { 0 }, { 1 }, },
167
        .index = { { 0 }, { 0 }, { 1 } },
168
        .config_map = { 2, TYPE_CPE, TYPE_CPE },
169
        .reorder_map = { 0, 1, 2, 3 },
170
    },
171
    {
172
        .layout = AV_CHANNEL_LAYOUT_5POINT0,
173
        .num_ele = { 2, 0, 1, 0 },
174
        .pairing = { { 0, 1 }, { 0 }, { 1 } },
175
        .index = { { 0, 0 }, { 0 }, { 1 } },
176
        .config_map = { 3, TYPE_SCE, TYPE_CPE, TYPE_CPE },
177
        .reorder_map = { 2, 0, 1, 3, 4 },
178
    },
179
    {
180
        .layout = AV_CHANNEL_LAYOUT_5POINT1,
181
        .num_ele = { 2, 0, 1, 1 },
182
        .pairing = { { 0, 1 }, { 0 }, { 1 }, },
183
        .index = { { 0, 0 }, { 0 }, { 1 }, { 0 } },
184
        .config_map = { 4, TYPE_SCE, TYPE_CPE, TYPE_CPE, TYPE_LFE },
185
        .reorder_map = { 2, 0, 1, 4, 5, 3 },
186
    },
187
    {
188
        .layout = AV_CHANNEL_LAYOUT_5POINT0_BACK,
189
        .num_ele = { 2, 0, 1, 0 },
190
        .pairing = { { 0, 1 }, { 0 }, { 1 } },
191
        .index = { { 0, 0 }, { 0 }, { 1 } },
192
        .config_map = { 3, TYPE_SCE, TYPE_CPE, TYPE_CPE },
193
        .reorder_map = { 2, 0, 1, 3, 4 },
194
    },
195
    {
196
        .layout = AV_CHANNEL_LAYOUT_5POINT1_BACK,
197
        .num_ele = { 2, 0, 1, 1 },
198
        .pairing = { { 0, 1 }, { 0 }, { 1 }, },
199
        .index = { { 0, 0 }, { 0 }, { 1 }, { 0 } },
200
        .config_map = { 4, TYPE_SCE, TYPE_CPE, TYPE_CPE, TYPE_LFE },
201
        .reorder_map = { 2, 0, 1, 4, 5, 3 },
202
    },
203
    {
204
        .layout = AV_CHANNEL_LAYOUT_6POINT0,
205
        .num_ele = { 2, 0, 2, 0 },
206
        .pairing = { { 0, 1 }, { 0 }, { 1, 0 } },
207
        .index = { { 0, 0 }, { 0 }, { 1, 1 } },
208
        .config_map = { 4, TYPE_SCE, TYPE_CPE, TYPE_CPE, TYPE_SCE },
209
        .reorder_map = { 2, 0, 1, 4, 5, 3 },
210
    },
211
    {
212
        .layout = AV_CHANNEL_LAYOUT_6POINT0_FRONT,
213
        .num_ele = { 2, 0, 1, 0 },
214
        .pairing = { { 1, 1 }, { 0 }, { 1 } },
215
        .index = { { 0, 1 }, { 0 }, { 2 }, },
216
        .config_map = { 3, TYPE_CPE, TYPE_CPE, TYPE_CPE, },
217
        .reorder_map = { 2, 3, 0, 1, 4, 5 },
218
    },
219
    {
220
        .layout = AV_CHANNEL_LAYOUT_HEXAGONAL,
221
        .num_ele = { 2, 0, 2, 0 },
222
        .pairing = { { 0, 1 }, { 0 }, { 1, 0 } },
223
        .index = { { 0, 0 }, { 0 }, { 1, 1 } },
224
        .config_map = { 4, TYPE_SCE, TYPE_CPE, TYPE_CPE, TYPE_SCE },
225
        .reorder_map = { 2, 0, 1, 3, 4, 5 },
226
    },
227
    {
228
        .layout = AV_CHANNEL_LAYOUT_6POINT1,
229
        .num_ele = { 2, 0, 2, 1 },
230
        .pairing = { { 0, 1 }, { 0 }, { 1, 0 }, },
231
        .index = { { 0, 0 }, { 0 }, { 1, 1 }, { 0 } },
232
        .config_map = { 5, TYPE_SCE, TYPE_CPE, TYPE_CPE, TYPE_SCE, TYPE_LFE },
233
        .reorder_map = { 2, 0, 1, 5, 6, 4, 3 },
234
    },
235
    {
236
        .layout = AV_CHANNEL_LAYOUT_6POINT1_BACK,
237
        .num_ele = { 2, 0, 2, 1 },
238
        .pairing = { { 0, 1 },{ 0 },{ 1, 0 }, },
239
        .index = { { 0, 0 },{ 0 },{ 1, 1 },{ 0 } },
240
        .config_map = { 5, TYPE_SCE, TYPE_CPE, TYPE_CPE, TYPE_SCE, TYPE_LFE },
241
        .reorder_map = { 2, 0, 1, 4, 5, 6, 3 },
242
    },
243
    {
244
        .layout = AV_CHANNEL_LAYOUT_6POINT1_FRONT,
245
        .num_ele = { 2, 0, 1, 1 },
246
        .pairing = { { 1, 1 }, { 0 }, { 1 }, },
247
        .index = { { 0, 1 }, { 0 }, { 2 }, { 0 }, },
248
        .config_map = { 4, TYPE_CPE, TYPE_CPE, TYPE_CPE, TYPE_LFE, },
249
        .reorder_map = { 3, 4, 0, 1, 5, 6, 2 },
250
    },
251
    {
252
        .layout = AV_CHANNEL_LAYOUT_7POINT0,
253
        .num_ele = { 2, 0, 2, 0 },
254
        .pairing = { { 0, 1 }, { 0 }, { 1, 1 }, },
255
        .index = { { 0, 0 }, { 0 }, { 2, 1 }, },
256
        .config_map = { 4, TYPE_SCE, TYPE_CPE, TYPE_CPE, TYPE_CPE },
257
        .reorder_map = { 2, 0, 1, 3, 4, 5, 6 },
258
    },
259
    {
260
        .layout = AV_CHANNEL_LAYOUT_7POINT0_FRONT,
261
        .num_ele = { 3, 0, 1, 0 },
262
        .pairing = { { 0, 1, 1 }, { 0 }, { 1 }, },
263
        .index = { { 0, 0, 1 }, { 0 }, { 2 }, },
264
        .config_map = { 4, TYPE_SCE, TYPE_CPE, TYPE_CPE, TYPE_CPE },
265
        .reorder_map = { 2, 3, 4, 0, 1, 5, 6 },
266
    },
267
    {
268
        .layout = AV_CHANNEL_LAYOUT_7POINT1,
269
        .num_ele = { 2, 0, 2, 1 },
270
        .pairing = { { 0, 1 }, { 0 }, { 1, 1 }, },
271
        .index = { { 0, 0 }, { 0 }, { 2, 1 }, { 0 } },
272
        .config_map = { 5, TYPE_SCE, TYPE_CPE, TYPE_CPE, TYPE_CPE, TYPE_LFE },
273
        .reorder_map = { 2, 0, 1, 4, 5, 6, 7, 3 },
274
    },
275
    {
276
        .layout = AV_CHANNEL_LAYOUT_7POINT1_WIDE,
277
        .num_ele = { 3, 0, 1, 1 },
278
        .pairing = { { 0, 1, 1 }, { 0 }, { 1 }, },
279
        .index = { { 0, 0, 1 }, { 0 }, { 2 }, { 0 }, },
280
        .config_map = { 5, TYPE_SCE, TYPE_CPE, TYPE_CPE, TYPE_CPE, TYPE_LFE },
281
        .reorder_map = { 2, 4, 5, 0, 1, 6, 7, 3 },
282
    },
283
    {
284
        .layout = AV_CHANNEL_LAYOUT_7POINT1_WIDE_BACK,
285
        .num_ele = { 3, 0, 1, 1 },
286
        .pairing = { { 0, 1, 1 }, { 0 }, { 1 } },
287
        .index = { { 0, 0, 1 }, { 0 }, { 2 }, { 0 } },
288
        .config_map = { 5, TYPE_SCE, TYPE_CPE, TYPE_CPE, TYPE_CPE, TYPE_LFE },
289
        .reorder_map = { 2, 6, 7, 0, 1, 4, 5, 3 },
290
    },
291
    {
292
        .layout = AV_CHANNEL_LAYOUT_OCTAGONAL,
293
        .num_ele = { 2, 0, 3, 0 },
294
        .pairing = { { 0, 1 }, { 0 }, { 1, 1, 0 }, },
295
        .index = { { 0, 0 }, { 0 }, { 1, 2, 1 }, },
296
        .config_map = { 5, TYPE_SCE, TYPE_CPE, TYPE_CPE, TYPE_CPE, TYPE_SCE },
297
        .reorder_map = { 2, 0, 1, 6, 7, 3, 4, 5 },
298
    },
299
};
300
301
static void put_pce(PutBitContext *pb, AVCodecContext *avctx)
302
0
{
303
0
    int i, j;
304
0
    AACEncContext *s = avctx->priv_data;
305
0
    AACPCEInfo *pce = &s->pce;
306
0
    const int bitexact = avctx->flags & AV_CODEC_FLAG_BITEXACT;
307
0
    const char *aux_data = bitexact ? "Lavc" : LIBAVCODEC_IDENT;
308
309
0
    put_bits(pb, 4, 0);
310
311
0
    put_bits(pb, 2, avctx->profile);
312
0
    put_bits(pb, 4, s->samplerate_index);
313
314
0
    put_bits(pb, 4, pce->num_ele[0]); /* Front */
315
0
    put_bits(pb, 4, pce->num_ele[1]); /* Side */
316
0
    put_bits(pb, 4, pce->num_ele[2]); /* Back */
317
0
    put_bits(pb, 2, pce->num_ele[3]); /* LFE */
318
0
    put_bits(pb, 3, 0); /* Assoc data */
319
0
    put_bits(pb, 4, 0); /* CCs */
320
321
0
    put_bits(pb, 1, 0); /* Stereo mixdown */
322
0
    put_bits(pb, 1, 0); /* Mono mixdown */
323
0
    put_bits(pb, 1, 0); /* Something else */
324
325
0
    for (i = 0; i < 4; i++) {
326
0
        for (j = 0; j < pce->num_ele[i]; j++) {
327
0
            if (i < 3)
328
0
                put_bits(pb, 1, pce->pairing[i][j]);
329
0
            put_bits(pb, 4, pce->index[i][j]);
330
0
        }
331
0
    }
332
333
0
    align_put_bits(pb);
334
0
    put_bits(pb, 8, strlen(aux_data));
335
0
    ff_put_string(pb, aux_data, 0);
336
0
}
337
338
/**
339
 * Make AAC audio config object.
340
 * @see 1.6.2.1 "Syntax - AudioSpecificConfig"
341
 */
342
static int put_audio_specific_config(AVCodecContext *avctx, int chcfg)
343
0
{
344
0
    PutBitContext pb;
345
0
    AACEncContext *s = avctx->priv_data;
346
0
    const int max_size = 32;
347
348
0
    avctx->extradata = av_mallocz(max_size);
349
0
    if (!avctx->extradata)
350
0
        return AVERROR(ENOMEM);
351
352
0
    init_put_bits(&pb, avctx->extradata, max_size);
353
0
    put_bits(&pb, 5, s->profile+1); //profile
354
0
    put_bits(&pb, 4, s->samplerate_index); //sample rate index
355
0
    put_bits(&pb, 4, chcfg);
356
    //GASpecificConfig
357
0
    put_bits(&pb, 1, 0); //frame length - 1024 samples
358
0
    put_bits(&pb, 1, 0); //does not depend on core coder
359
0
    put_bits(&pb, 1, 0); //is not extension
360
0
    if (s->needs_pce)
361
0
        put_pce(&pb, avctx);
362
363
    //Explicitly Mark SBR absent
364
0
    put_bits(&pb, 11, 0x2b7); //sync extension
365
0
    put_bits(&pb, 5,  AOT_SBR);
366
0
    put_bits(&pb, 1,  0);
367
0
    flush_put_bits(&pb);
368
0
    avctx->extradata_size = put_bytes_output(&pb);
369
370
0
    return 0;
371
0
}
372
373
void ff_quantize_band_cost_cache_init(struct AACEncContext *s)
374
0
{
375
0
    ++s->quantize_band_cost_cache_generation;
376
0
    if (s->quantize_band_cost_cache_generation == 0) {
377
0
        memset(s->quantize_band_cost_cache, 0, sizeof(s->quantize_band_cost_cache));
378
0
        s->quantize_band_cost_cache_generation = 1;
379
0
    }
380
0
}
381
382
#define WINDOW_FUNC(type) \
383
static void apply_ ##type ##_window(AVFloatDSPContext *fdsp, \
384
                                    SingleChannelElement *sce, \
385
                                    const float *audio)
386
387
WINDOW_FUNC(only_long)
388
0
{
389
0
    const float *lwindow = sce->ics.use_kb_window[0] ? ff_aac_kbd_long_1024 : ff_sine_1024;
390
0
    const float *pwindow = sce->ics.use_kb_window[1] ? ff_aac_kbd_long_1024 : ff_sine_1024;
391
0
    float *out = sce->ret_buf;
392
393
0
    fdsp->vector_fmul        (out,        audio,        lwindow, 1024);
394
0
    fdsp->vector_fmul_reverse(out + 1024, audio + 1024, pwindow, 1024);
395
0
}
396
397
WINDOW_FUNC(long_start)
398
0
{
399
0
    const float *lwindow = sce->ics.use_kb_window[1] ? ff_aac_kbd_long_1024 : ff_sine_1024;
400
0
    const float *swindow = sce->ics.use_kb_window[0] ? ff_aac_kbd_short_128 : ff_sine_128;
401
0
    float *out = sce->ret_buf;
402
403
0
    fdsp->vector_fmul(out, audio, lwindow, 1024);
404
0
    memcpy(out + 1024, audio + 1024, sizeof(out[0]) * 448);
405
0
    fdsp->vector_fmul_reverse(out + 1024 + 448, audio + 1024 + 448, swindow, 128);
406
0
    memset(out + 1024 + 576, 0, sizeof(out[0]) * 448);
407
0
}
408
409
WINDOW_FUNC(long_stop)
410
0
{
411
0
    const float *lwindow = sce->ics.use_kb_window[0] ? ff_aac_kbd_long_1024 : ff_sine_1024;
412
0
    const float *swindow = sce->ics.use_kb_window[1] ? ff_aac_kbd_short_128 : ff_sine_128;
413
0
    float *out = sce->ret_buf;
414
415
0
    memset(out, 0, sizeof(out[0]) * 448);
416
0
    fdsp->vector_fmul(out + 448, audio + 448, swindow, 128);
417
0
    memcpy(out + 576, audio + 576, sizeof(out[0]) * 448);
418
0
    fdsp->vector_fmul_reverse(out + 1024, audio + 1024, lwindow, 1024);
419
0
}
420
421
WINDOW_FUNC(eight_short)
422
0
{
423
0
    const float *swindow = sce->ics.use_kb_window[0] ? ff_aac_kbd_short_128 : ff_sine_128;
424
0
    const float *pwindow = sce->ics.use_kb_window[1] ? ff_aac_kbd_short_128 : ff_sine_128;
425
0
    const float *in = audio + 448;
426
0
    float *out = sce->ret_buf;
427
0
    int w;
428
429
0
    for (w = 0; w < 8; w++) {
430
0
        fdsp->vector_fmul        (out, in, w ? pwindow : swindow, 128);
431
0
        out += 128;
432
0
        in  += 128;
433
0
        fdsp->vector_fmul_reverse(out, in, swindow, 128);
434
0
        out += 128;
435
0
    }
436
0
}
437
438
static void (*const apply_window[4])(AVFloatDSPContext *fdsp,
439
                                     SingleChannelElement *sce,
440
                                     const float *audio) = {
441
    [ONLY_LONG_SEQUENCE]   = apply_only_long_window,
442
    [LONG_START_SEQUENCE]  = apply_long_start_window,
443
    [EIGHT_SHORT_SEQUENCE] = apply_eight_short_window,
444
    [LONG_STOP_SEQUENCE]   = apply_long_stop_window
445
};
446
447
static void apply_window_and_mdct(AACEncContext *s, SingleChannelElement *sce,
448
                                  float *audio)
449
0
{
450
0
    int i;
451
0
    float *output = sce->ret_buf;
452
453
0
    apply_window[sce->ics.window_sequence[0]](s->fdsp, sce, audio);
454
455
0
    if (sce->ics.window_sequence[0] != EIGHT_SHORT_SEQUENCE)
456
0
        s->mdct1024_fn(s->mdct1024, sce->coeffs, output, sizeof(float));
457
0
    else
458
0
        for (i = 0; i < 1024; i += 128)
459
0
            s->mdct128_fn(s->mdct128, &sce->coeffs[i], output + i*2, sizeof(float));
460
0
    memcpy(audio, audio + 1024, sizeof(audio[0]) * 1024);
461
0
    memcpy(sce->pcoeffs, sce->coeffs, sizeof(sce->pcoeffs));
462
0
}
463
464
/**
465
 * Encode ics_info element.
466
 * @see Table 4.6 (syntax of ics_info)
467
 */
468
static void put_ics_info(AACEncContext *s, IndividualChannelStream *info)
469
0
{
470
0
    int w;
471
472
0
    put_bits(&s->pb, 1, 0);                // ics_reserved bit
473
0
    put_bits(&s->pb, 2, info->window_sequence[0]);
474
0
    put_bits(&s->pb, 1, info->use_kb_window[0]);
475
0
    if (info->window_sequence[0] != EIGHT_SHORT_SEQUENCE) {
476
0
        put_bits(&s->pb, 6, info->max_sfb);
477
0
        put_bits(&s->pb, 1, 0); /* No predictor present */
478
0
    } else {
479
0
        put_bits(&s->pb, 4, info->max_sfb);
480
0
        for (w = 1; w < 8; w++)
481
0
            put_bits(&s->pb, 1, !info->group_len[w]);
482
0
    }
483
0
}
484
485
/**
486
 * Encode MS data.
487
 * @see 4.6.8.1 "Joint Coding - M/S Stereo"
488
 */
489
static void encode_ms_info(PutBitContext *pb, ChannelElement *cpe)
490
0
{
491
0
    int i, w;
492
493
0
    put_bits(pb, 2, cpe->ms_mode);
494
0
    if (cpe->ms_mode == 1)
495
0
        for (w = 0; w < cpe->ch[0].ics.num_windows; w += cpe->ch[0].ics.group_len[w])
496
0
            for (i = 0; i < cpe->ch[0].ics.max_sfb; i++)
497
0
                put_bits(pb, 1, cpe->ms_mask[w*16 + i]);
498
0
}
499
500
/**
501
 * Produce integer coefficients from scalefactors provided by the model.
502
 */
503
static void adjust_frame_information(ChannelElement *cpe, int chans)
504
0
{
505
0
    int i, w, w2, g, ch;
506
0
    int maxsfb, cmaxsfb;
507
508
0
    for (ch = 0; ch < chans; ch++) {
509
0
        IndividualChannelStream *ics = &cpe->ch[ch].ics;
510
0
        maxsfb = 0;
511
0
        cpe->ch[ch].pulse.num_pulse = 0;
512
0
        for (w = 0; w < ics->num_windows; w += ics->group_len[w]) {
513
0
            for (cmaxsfb = ics->num_swb; cmaxsfb > 0 && cpe->ch[ch].zeroes[w*16+cmaxsfb-1]; cmaxsfb--)
514
0
                ;
515
0
            maxsfb = FFMAX(maxsfb, cmaxsfb);
516
0
        }
517
0
        ics->max_sfb = maxsfb;
518
519
        //adjust zero bands for window groups
520
0
        for (w = 0; w < ics->num_windows; w += ics->group_len[w]) {
521
0
            for (g = 0; g < ics->max_sfb; g++) {
522
0
                i = 1;
523
0
                for (w2 = w; w2 < w + ics->group_len[w]; w2++) {
524
0
                    if (!cpe->ch[ch].zeroes[w2*16 + g]) {
525
0
                        i = 0;
526
0
                        break;
527
0
                    }
528
0
                }
529
0
                cpe->ch[ch].zeroes[w*16 + g] = i;
530
0
            }
531
0
        }
532
0
    }
533
534
0
    if (chans > 1 && cpe->common_window) {
535
0
        IndividualChannelStream *ics0 = &cpe->ch[0].ics;
536
0
        IndividualChannelStream *ics1 = &cpe->ch[1].ics;
537
0
        int msc = 0;
538
0
        ics0->max_sfb = FFMAX(ics0->max_sfb, ics1->max_sfb);
539
0
        ics1->max_sfb = ics0->max_sfb;
540
0
        for (w = 0; w < ics0->num_windows*16; w += 16)
541
0
            for (i = 0; i < ics0->max_sfb; i++)
542
0
                if (cpe->ms_mask[w+i])
543
0
                    msc++;
544
0
        if (msc == 0 || ics0->max_sfb == 0)
545
0
            cpe->ms_mode = 0;
546
0
        else
547
0
            cpe->ms_mode = msc < ics0->max_sfb * ics0->num_windows ? 1 : 2;
548
0
    }
549
0
}
550
551
static void apply_intensity_stereo(ChannelElement *cpe)
552
0
{
553
0
    int w, w2, g, i;
554
0
    IndividualChannelStream *ics = &cpe->ch[0].ics;
555
0
    if (!cpe->common_window)
556
0
        return;
557
0
    for (w = 0; w < ics->num_windows; w += ics->group_len[w]) {
558
0
        for (w2 =  0; w2 < ics->group_len[w]; w2++) {
559
0
            int start = (w+w2) * 128;
560
0
            for (g = 0; g < ics->num_swb; g++) {
561
0
                int p  = -1 + 2 * (cpe->ch[1].band_type[w*16+g] - 14);
562
0
                float scale = cpe->ch[0].is_ener[w*16+g];
563
0
                if (!cpe->is_mask[w*16 + g]) {
564
0
                    start += ics->swb_sizes[g];
565
0
                    continue;
566
0
                }
567
0
                if (cpe->ms_mask[w*16 + g])
568
0
                    p *= -1;
569
0
                for (i = 0; i < ics->swb_sizes[g]; i++) {
570
0
                    float sum = (cpe->ch[0].coeffs[start+i] + p*cpe->ch[1].coeffs[start+i])*scale;
571
0
                    cpe->ch[0].coeffs[start+i] = sum;
572
0
                    cpe->ch[1].coeffs[start+i] = 0.0f;
573
0
                }
574
0
                start += ics->swb_sizes[g];
575
0
            }
576
0
        }
577
0
    }
578
0
}
579
580
/* I/S acceptance level for the image-error EMA at full rate pressure */
581
0
#define NMR_IS_IMG_GATE 8000.0f
582
583
/* Frequency in Hz for the lower limit of intensity stereo */
584
0
#define NMR_IS_LOW_LIMIT 6100
585
586
/* M/S adoption: es < 0.5*em, content-driven and rate-free */
587
0
#define NMR_MS_EQUIV 0.5f
588
0
#define NMR_MS_MASK  0.0f
589
590
/* Pair decouple threshold on the joint-tool candidacy fraction EMA: pairs
591
 * whose joint tools are mostly dead (diffuse decorrelated content) window
592
 * per-channel and skip M/S; recouple above 1.3x. */
593
0
#define NMR_DECORR_LO 0.20f
594
595
/* Stereo-decision hysteresis: leaving a joint mode costs a margin. */
596
0
#define NMR_STICKY 2.0f
597
598
/* Decision statistics are EMA-smoothed across frames. */
599
0
#define NMR_SDEC_EMA 0.75f
600
601
/* PNS-stereo gate: substitute only clearly-decorrelated (wide) bands. */
602
0
#define NMR_PNS_STEREO_DECORR 0.6f
603
604
/* Recode one band's window group as mid+side in place. */
605
static void nmr_apply_ms_band(AACEncContext *s, ChannelElement *cpe,
606
                              int w, int g, int start, int len, int gl)
607
0
{
608
0
    SingleChannelElement *sce0 = &cpe->ch[0];
609
0
    SingleChannelElement *sce1 = &cpe->ch[1];
610
0
    cpe->ms_mask[w*16+g] = 1;
611
0
    for (int w2 = 0; w2 < gl; w2++) {
612
0
        FFPsyBand *b0 = &s->psy.ch[s->cur_channel+0].psy_bands[(w+w2)*16+g];
613
0
        FFPsyBand *b1 = &s->psy.ch[s->cur_channel+1].psy_bands[(w+w2)*16+g];
614
0
        float *L = sce0->coeffs + start + (w+w2)*128;
615
0
        float *R = sce1->coeffs + start + (w+w2)*128;
616
0
        float em = 0.0f, es = 0.0f;
617
0
        for (int i = 0; i < len; i++) {
618
0
            float m = (L[i] + R[i]) * 0.5f;
619
0
            R[i] = m - R[i]; L[i] = m;
620
0
            em += L[i]*L[i]; es += R[i]*R[i];
621
0
        }
622
0
        b0->threshold = FFMIN(b0->threshold, b1->threshold) * 0.5f;
623
0
        b1->threshold = b0->threshold;
624
0
        b0->energy = em; b1->energy = es;
625
0
    }
626
0
}
627
628
/* I/S perceptual test: reconstruction image error vs the pair's masks. */
629
static int nmr_is_image_masked(AACEncContext *s, ChannelElement *cpe,
630
                               int w, int g, int start, int len, int gl,
631
                               float ener0, float ener1, float dot,
632
                               float minthr0, float minthr1, float *ratio_out,
633
                               float *scale_out, float *sr_out, int *p_out)
634
0
{
635
0
    int p = dot >= 0.0f ? 1 : -1;
636
0
    float ener01 = ener0 + ener1 + 2*p*dot;     /* energy of L + p*R */
637
0
    *ratio_out = FLT_MAX;
638
0
    if (ener01 <= FLT_MIN)
639
0
        return 0;
640
0
    float scale = sqrtf(ener0 / ener01);        /* carrier = (L + p*R)*scale */
641
0
    float sr_   = sqrtf(ener1 / ener0);         /* decoder: R = p*sr_*carrier */
642
0
    float img0 = 0.0f, img1 = 0.0f;
643
0
    for (int w2 = 0; w2 < gl; w2++) {
644
0
        const float *L = cpe->ch[0].coeffs + start + (w+w2)*128;
645
0
        const float *R = cpe->ch[1].coeffs + start + (w+w2)*128;
646
0
        for (int i = 0; i < len; i++) {
647
0
            float c  = (L[i] + p*R[i]) * scale;
648
0
            float dl = L[i] - c, dr = R[i] - p*sr_*c;
649
0
            img0 += dl*dl; img1 += dr*dr;
650
0
        }
651
0
    }
652
0
    *ratio_out = FFMAX(img0 / FFMAX(minthr0 * gl, FLT_MIN),
653
0
                       img1 / FFMAX(minthr1 * gl, FLT_MIN));
654
0
    *scale_out = scale; *sr_out = sr_; *p_out = p;
655
0
    return 1;
656
0
}
657
658
/* Recode one band's window group as intensity stereo in place: replace L with the
659
 * carrier, zero R, signal the phase via the side channel's band type, and fold the
660
 * pair's masking into the surviving (carrier) channel. */
661
static void nmr_apply_is_band(AACEncContext *s, ChannelElement *cpe,
662
                              int w, int g, int start, int len, int gl,
663
                              float scale, float sr_, int p,
664
                              float ener0, float ener1)
665
0
{
666
0
    cpe->is_mask[w*16+g] = 1;
667
0
    cpe->ch[0].is_ener[w*16+g] = scale;
668
0
    cpe->ch[1].is_ener[w*16+g] = ener0 / ener1;
669
0
    cpe->ch[1].band_type[w*16+g] = p > 0 ? INTENSITY_BT : INTENSITY_BT2;
670
0
    for (int w2 = 0; w2 < gl; w2++) {
671
0
        FFPsyBand *b0 = &s->psy.ch[s->cur_channel+0].psy_bands[(w+w2)*16+g];
672
0
        FFPsyBand *b1 = &s->psy.ch[s->cur_channel+1].psy_bands[(w+w2)*16+g];
673
0
        float *L = cpe->ch[0].coeffs + start + (w+w2)*128;
674
0
        float *R = cpe->ch[1].coeffs + start + (w+w2)*128;
675
0
        float ec = 0.0f;
676
0
        for (int i = 0; i < len; i++) {
677
0
            L[i] = (L[i] + p*R[i]) * scale;
678
0
            R[i] = 0.0f;
679
0
            ec += L[i]*L[i];
680
0
        }
681
0
        b0->threshold = FFMIN(b0->threshold, b1->threshold / FFMAX(sr_*sr_, 1e-9f));
682
0
        b0->energy = ec; b1->energy = 0.0f;
683
0
    }
684
0
}
685
686
/*
687
 * Per-band stereo-mode decision (L/R vs M/S vs intensity) for the NMR coder,
688
 * made before quantization from the psychoacoustic model alone, so the
689
 * quantizer search allocates natively on the spectra that are actually coded.
690
 */
691
static void nmr_decide_stereo(AACEncContext *s, ChannelElement *cpe)
692
0
{
693
0
    SingleChannelElement *sce0 = &cpe->ch[0];
694
0
    SingleChannelElement *sce1 = &cpe->ch[1];
695
0
    IndividualChannelStream *ics = &sce0->ics;
696
0
    const AVCodecContext *avctx = s->psy.avctx;
697
0
    const float freq_mult = avctx->sample_rate / (1024.0f / ics->num_windows) / 2.0f;
698
0
    int is_count = 0;
699
700
0
    if (s->nmr) {
701
0
        int pi = (s->cur_channel >> 1) & 7;
702
0
        pi = pi * 2 + (ics->num_windows == 8);   /* per-grid state bank */
703
0
        if (!s->nmr->sinit[pi]) {
704
            /* one-time init; per-grid banks persist across window switches
705
             * (wiping them churned stereo modes audibly) */
706
0
            memset(s->nmr->smode[pi], 0, sizeof(s->nmr->smode[pi]));
707
0
            for (int b = 0; b < 128; b++) {
708
0
                s->nmr->sema_em[pi][b]  = 0.0f;
709
0
                s->nmr->sema_img[pi][b] = -1.0f;
710
0
            }
711
0
            s->nmr->sinit[pi] = 1;
712
0
        }
713
0
    }
714
715
    /* Per-band stereo decision (L/R vs M/S vs I/S), made pre-quantization from
716
     * the psy model so the trellis allocates on the coded spectra. */
717
718
    /* I/S engages under SUSTAINED strain only: rate pressure gated by the
719
     * lambda floor (pressure spikes at a comfortable operating point must
720
     * not admit it). Unengaged candidates fall back to M/S. */
721
0
    float is_ramp = s->nmr ? s->nmr->press *
722
0
        av_clipf((s->nmr->lam_floor - 40.0f) / (120.0f - 40.0f), 0.0f, 1.0f) : 0.0f;
723
0
    const int allow_is = s->options.intensity_stereo && is_ramp > 0.0f;
724
725
0
    const int pidx = (s->cur_channel >> 1) & 15;
726
0
    const int decoupled = s->psy.pair_decoupled[pidx];
727
0
    int njoint = 0, nbands = 0;   /* joint-tool candidacy census, decouple feed */
728
729
0
    for (int w = 0; w < ics->num_windows; w += ics->group_len[w]) {
730
0
        int start = 0;
731
0
        for (int g = 0; g < ics->num_swb; start += ics->swb_sizes[g++]) {
732
0
            int len = ics->swb_sizes[g], gl = ics->group_len[w];
733
0
            float ener0 = 0.0f, ener1 = 0.0f, dot = 0.0f, es_tot = 0.0f, em_tot = 0.0f;
734
0
            float minthr0 = FLT_MAX, minthr1 = FLT_MAX;
735
736
0
            cpe->is_mask[w*16+g] = 0;
737
0
            cpe->ms_mask[w*16+g] = 0;
738
739
0
            for (int w2 = 0; w2 < gl; w2++) {
740
0
                FFPsyBand *b0 = &s->psy.ch[s->cur_channel+0].psy_bands[(w+w2)*16+g];
741
0
                FFPsyBand *b1 = &s->psy.ch[s->cur_channel+1].psy_bands[(w+w2)*16+g];
742
0
                const float *L = sce0->coeffs + start + (w+w2)*128;
743
0
                const float *R = sce1->coeffs + start + (w+w2)*128;
744
0
                float el = 0.0f, er = 0.0f, em = 0.0f, es = 0.0f, d = 0.0f;
745
0
                for (int i = 0; i < len; i++) {
746
0
                    float m  = (L[i] + R[i]) * 0.5f;
747
0
                    float sv = m - R[i];
748
0
                    el += L[i]*L[i]; er += R[i]*R[i];
749
0
                    em += m*m; es += sv*sv; d += L[i]*R[i];
750
0
                }
751
0
                ener0 += el; ener1 += er; dot += d; es_tot += es; em_tot += em;
752
0
                minthr0 = FFMIN(minthr0, b0->threshold);
753
0
                minthr1 = FFMIN(minthr1, b1->threshold);
754
0
            }
755
0
            float thr_g = FFMIN(minthr0, minthr1) * gl;   /* group masking budget */
756
757
            /* PNS-stereo reservation: keep clearly-wide noise bands for PNS. */
758
0
            const int sidx = w*16+g;
759
0
            {
760
0
                float es_w = es_tot, em_w = em_tot;
761
0
                if (s->nmr) {
762
0
                    int pi_ = ((s->cur_channel >> 1) & 7) * 2 + (cpe->ch[0].ics.num_windows == 8);
763
0
                    float pe = s->nmr->sema_es[pi_][sidx];
764
0
                    float pm = s->nmr->sema_em[pi_][sidx];
765
0
                    if (pm > 0.0f) {
766
0
                        es_w = NMR_SDEC_EMA * pe + (1.0f - NMR_SDEC_EMA) * es_tot;
767
0
                        em_w = NMR_SDEC_EMA * pm + (1.0f - NMR_SDEC_EMA) * em_tot;
768
0
                    }
769
0
                }
770
0
                if (cpe->ch[0].can_pns[w*16+g] && cpe->ch[1].can_pns[w*16+g] &&
771
0
                    es_w > NMR_PNS_STEREO_DECORR * em_w)
772
0
                    continue;
773
0
            }
774
0
            cpe->ch[0].can_pns[w*16+g] = cpe->ch[1].can_pns[w*16+g] = 0;
775
776
0
            int pi = ((s->cur_channel >> 1) & 7) * 2 + (cpe->ch[0].ics.num_windows == 8);
777
0
            uint8_t *pmode = s->nmr ? s->nmr->smode[pi] : NULL;
778
0
            int prev = pmode ? pmode[sidx] : 0;
779
0
            float eqgate = NMR_MS_EQUIV * (prev == 1 ? 1.5f : 1.0f);   /* stay-until es>0.75em */
780
            /* I/S = lossy economy: image-error budget scales with pressure */
781
0
            float imgate = NMR_IS_IMG_GATE * is_ramp * (prev == 2 ? NMR_STICKY : 1.0f);
782
0
            float es_d = es_tot, em_d = em_tot;
783
0
            if (s->nmr) {
784
0
                float *ees = &s->nmr->sema_es[pi][sidx];
785
0
                float *eem = &s->nmr->sema_em[pi][sidx];
786
0
                if (*eem <= 0.0f) { *ees = es_tot; *eem = em_tot; }
787
0
                else {
788
0
                    *ees = NMR_SDEC_EMA * *ees + (1.0f - NMR_SDEC_EMA) * es_tot;
789
0
                    *eem = NMR_SDEC_EMA * *eem + (1.0f - NMR_SDEC_EMA) * em_tot;
790
0
                }
791
0
                es_d = *ees; em_d = *eem;
792
0
            }
793
0
            int ms_would = s->options.mid_side &&
794
0
                           (s->options.mid_side == 1 ||
795
0
                            es_d < eqgate * em_d ||
796
0
                            es_tot < NMR_MS_MASK  * thr_g);
797
0
            int ms_ok = ms_would && !decoupled;
798
0
            float scale, sr_, imgratio; int p;
799
            /* I/S competes with M/S above the frequency limit (candidacy must
800
             * not be gated on !ms_ok - that leaves only unrenderable bands) */
801
0
            int is_cand = start * freq_mult > NMR_IS_LOW_LIMIT &&
802
0
                          ener0 > FLT_MIN && ener1 > FLT_MIN &&
803
0
                          nmr_is_image_masked(s, cpe, w, g, start, len, gl,
804
0
                                              ener0, ener1, dot, minthr0, minthr1,
805
0
                                              &imgratio, &scale, &sr_, &p);
806
0
            int is_ok = is_cand;
807
0
            if (s->nmr && start * freq_mult > NMR_IS_LOW_LIMIT) {
808
                /* smoothed image-error; updated only while candidate (fail-value
809
                 * feeding jammed it permanently high) */
810
0
                float *eim = &s->nmr->sema_img[pi][sidx];
811
0
                if (is_cand) {
812
                    /* seed from first measurement; freeze when not candidate */
813
0
                    if (*eim < 0.0f) *eim = imgratio;
814
0
                    else *eim = NMR_SDEC_EMA * *eim + (1.0f - NMR_SDEC_EMA) * FFMIN(imgratio, 100.0f * NMR_IS_IMG_GATE);
815
0
                }
816
0
                is_ok = is_cand && *eim >= 0.0f && *eim < imgate;
817
0
            }
818
819
0
            njoint += ms_would || is_ok; nbands++;
820
0
            if (pmode) {
821
0
                int m_ = (is_ok && allow_is) ? 2 : ms_ok ? 1 :
822
0
                         (is_ok && s->options.mid_side) ? 1 : 0;
823
0
                pmode[sidx] = m_;
824
0
                s->nmr->smode_band[(s->cur_channel >> 1) & 7][w*16+g] = m_;
825
0
            }
826
0
            if (is_ok && allow_is) {
827
0
                nmr_apply_is_band(s, cpe, w, g, start, len, gl,
828
0
                                  scale, sr_, p, ener0, ener1);
829
0
                is_count++;
830
0
            } else if (ms_ok || (is_ok && s->options.mid_side)) {
831
0
                nmr_apply_ms_band(s, cpe, w, g, start, len, gl);
832
0
            }
833
            /* else: keep full L/R stereo */
834
0
        }
835
0
    }
836
0
    cpe->is_mode = !!is_count;
837
838
0
    if (nbands > 0) {
839
        /* Pair joint-tool value, read next frame by the psy pair-synced window
840
         * decision and the M/S candidacy above. Measured as CANDIDACY (not
841
         * adoption) so decoupling cannot starve its own signal and self-lock. */
842
0
        float r = (float)njoint / nbands;
843
0
        float *pj = &s->psy.pair_joint[pidx];
844
0
        *pj = *pj > 0.0f ? 0.95f * *pj + 0.05f * r : r;
845
0
        s->psy.pair_decoupled[pidx] = *pj <
846
0
            (s->psy.pair_decoupled[pidx] ? 1.3f * NMR_DECORR_LO : NMR_DECORR_LO);
847
0
    }
848
0
}
849
850
static void apply_mid_side_stereo(ChannelElement *cpe)
851
0
{
852
0
    int w, w2, g, i;
853
0
    IndividualChannelStream *ics = &cpe->ch[0].ics;
854
0
    if (!cpe->common_window)
855
0
        return;
856
0
    for (w = 0; w < ics->num_windows; w += ics->group_len[w]) {
857
0
        for (w2 =  0; w2 < ics->group_len[w]; w2++) {
858
0
            int start = (w+w2) * 128;
859
0
            for (g = 0; g < ics->num_swb; g++) {
860
                /* ms_mask can be used for other purposes in PNS and I/S,
861
                 * so must not apply M/S if any band uses either, even if
862
                 * ms_mask is set.
863
                 */
864
0
                if (!cpe->ms_mask[w*16 + g] || cpe->is_mask[w*16 + g]
865
0
                    || cpe->ch[0].band_type[w*16 + g] >= NOISE_BT
866
0
                    || cpe->ch[1].band_type[w*16 + g] >= NOISE_BT) {
867
0
                    start += ics->swb_sizes[g];
868
0
                    continue;
869
0
                }
870
0
                for (i = 0; i < ics->swb_sizes[g]; i++) {
871
0
                    float L = (cpe->ch[0].coeffs[start+i] + cpe->ch[1].coeffs[start+i]) * 0.5f;
872
0
                    float R = L - cpe->ch[1].coeffs[start+i];
873
0
                    cpe->ch[0].coeffs[start+i] = L;
874
0
                    cpe->ch[1].coeffs[start+i] = R;
875
0
                }
876
0
                start += ics->swb_sizes[g];
877
0
            }
878
0
        }
879
0
    }
880
0
}
881
882
/**
883
 * Encode scalefactor band coding type.
884
 */
885
static void encode_band_info(AACEncContext *s, SingleChannelElement *sce)
886
0
{
887
0
    int w;
888
889
0
    if (s->coder->set_special_band_scalefactors)
890
0
        s->coder->set_special_band_scalefactors(s, sce);
891
892
0
    for (w = 0; w < sce->ics.num_windows; w += sce->ics.group_len[w])
893
0
        s->coder->encode_window_bands_info(s, sce, w, sce->ics.group_len[w], s->lambda);
894
0
}
895
896
/**
897
 * Encode scalefactors.
898
 */
899
static void encode_scale_factors(AVCodecContext *avctx, AACEncContext *s,
900
                                 SingleChannelElement *sce)
901
0
{
902
0
    int diff, off_sf = sce->sf_idx[0], off_pns = sce->sf_idx[0] - NOISE_OFFSET;
903
0
    int off_is = 0, noise_flag = 1;
904
0
    int i, w;
905
906
0
    for (w = 0; w < sce->ics.num_windows; w += sce->ics.group_len[w]) {
907
0
        for (i = 0; i < sce->ics.max_sfb; i++) {
908
0
            if (!sce->zeroes[w*16 + i]) {
909
0
                if (sce->band_type[w*16 + i] == NOISE_BT) {
910
0
                    diff = sce->sf_idx[w*16 + i] - off_pns;
911
0
                    off_pns = sce->sf_idx[w*16 + i];
912
0
                    if (noise_flag-- > 0) {
913
0
                        put_bits(&s->pb, NOISE_PRE_BITS, diff + NOISE_PRE);
914
0
                        continue;
915
0
                    }
916
0
                } else if (sce->band_type[w*16 + i] == INTENSITY_BT  ||
917
0
                           sce->band_type[w*16 + i] == INTENSITY_BT2) {
918
0
                    diff = sce->sf_idx[w*16 + i] - off_is;
919
0
                    off_is = sce->sf_idx[w*16 + i];
920
0
                } else {
921
0
                    diff = sce->sf_idx[w*16 + i] - off_sf;
922
0
                    off_sf = sce->sf_idx[w*16 + i];
923
0
                }
924
0
                diff += SCALE_DIFF_ZERO;
925
0
                av_assert0(diff >= 0 && diff <= 120);
926
0
                put_bits(&s->pb, ff_aac_scalefactor_bits[diff], ff_aac_scalefactor_code[diff]);
927
0
            }
928
0
        }
929
0
    }
930
0
}
931
932
/**
933
 * Encode pulse data.
934
 */
935
static void encode_pulses(AACEncContext *s, Pulse *pulse)
936
0
{
937
0
    int i;
938
939
0
    put_bits(&s->pb, 1, !!pulse->num_pulse);
940
0
    if (!pulse->num_pulse)
941
0
        return;
942
943
0
    put_bits(&s->pb, 2, pulse->num_pulse - 1);
944
0
    put_bits(&s->pb, 6, pulse->start);
945
0
    for (i = 0; i < pulse->num_pulse; i++) {
946
0
        put_bits(&s->pb, 5, pulse->pos[i]);
947
0
        put_bits(&s->pb, 4, pulse->amp[i]);
948
0
    }
949
0
}
950
951
/**
952
 * Encode spectral coefficients processed by psychoacoustic model.
953
 */
954
static void encode_spectral_coeffs(AACEncContext *s, SingleChannelElement *sce)
955
0
{
956
0
    int start, i, w, w2;
957
958
0
    for (w = 0; w < sce->ics.num_windows; w += sce->ics.group_len[w]) {
959
0
        start = 0;
960
0
        for (i = 0; i < sce->ics.max_sfb; i++) {
961
0
            if (sce->zeroes[w*16 + i]) {
962
0
                start += sce->ics.swb_sizes[i];
963
0
                continue;
964
0
            }
965
0
            for (w2 = w; w2 < w + sce->ics.group_len[w]; w2++) {
966
0
                s->coder->quantize_and_encode_band(s, &s->pb,
967
0
                                                   &sce->coeffs[start + w2*128],
968
0
                                                   NULL, sce->ics.swb_sizes[i],
969
0
                                                   sce->sf_idx[w*16 + i],
970
0
                                                   sce->band_type[w*16 + i],
971
0
                                                   s->lambda,
972
0
                                                   sce->ics.window_clipping[w]);
973
0
            }
974
0
            start += sce->ics.swb_sizes[i];
975
0
        }
976
0
    }
977
0
}
978
979
/**
980
 * Downscale spectral coefficients for near-clipping windows to avoid artifacts
981
 */
982
static void avoid_clipping(AACEncContext *s, SingleChannelElement *sce)
983
0
{
984
0
    int start, i, j, w;
985
986
0
    if (sce->ics.clip_avoidance_factor < 1.0f) {
987
0
        for (w = 0; w < sce->ics.num_windows; w++) {
988
0
            start = 0;
989
0
            for (i = 0; i < sce->ics.max_sfb; i++) {
990
0
                float *swb_coeffs = &sce->coeffs[start + w*128];
991
0
                for (j = 0; j < sce->ics.swb_sizes[i]; j++)
992
0
                    swb_coeffs[j] *= sce->ics.clip_avoidance_factor;
993
0
                start += sce->ics.swb_sizes[i];
994
0
            }
995
0
        }
996
0
    }
997
0
}
998
999
/**
1000
 * Encode one channel of audio data.
1001
 */
1002
static int encode_individual_channel(AVCodecContext *avctx, AACEncContext *s,
1003
                                     SingleChannelElement *sce,
1004
                                     int common_window)
1005
0
{
1006
0
    put_bits(&s->pb, 8, sce->sf_idx[0]);
1007
0
    if (!common_window)
1008
0
        put_ics_info(s, &sce->ics);
1009
0
    encode_band_info(s, sce);
1010
0
    encode_scale_factors(avctx, s, sce);
1011
0
    encode_pulses(s, &sce->pulse);
1012
0
    put_bits(&s->pb, 1, !!sce->tns.present);
1013
0
    if (s->coder->encode_tns_info)
1014
0
        s->coder->encode_tns_info(s, sce);
1015
0
    put_bits(&s->pb, 1, 0); //ssr
1016
0
    encode_spectral_coeffs(s, sce);
1017
0
    return 0;
1018
0
}
1019
1020
/**
1021
 * Write some auxiliary information about the created AAC file.
1022
 */
1023
static void put_bitstream_info(AACEncContext *s, const char *name)
1024
0
{
1025
0
    int i, namelen, padbits;
1026
1027
0
    namelen = strlen(name) + 2;
1028
0
    put_bits(&s->pb, 3, TYPE_FIL);
1029
0
    put_bits(&s->pb, 4, FFMIN(namelen, 15));
1030
0
    if (namelen >= 15)
1031
0
        put_bits(&s->pb, 8, namelen - 14);
1032
0
    put_bits(&s->pb, 4, 0); //extension type - filler
1033
0
    padbits = -put_bits_count(&s->pb) & 7;
1034
0
    align_put_bits(&s->pb);
1035
0
    for (i = 0; i < namelen - 2; i++)
1036
0
        put_bits(&s->pb, 8, name[i]);
1037
0
    put_bits(&s->pb, 12 - padbits, 0);
1038
0
}
1039
1040
/*
1041
 * Copy input samples.
1042
 * Channels are reordered from libavcodec's default order to AAC order.
1043
 */
1044
static void copy_input_samples(AACEncContext *s, const AVFrame *frame)
1045
0
{
1046
0
    int ch;
1047
0
    int end = 2048 + (frame ? frame->nb_samples : 0);
1048
0
    const uint8_t *channel_map = s->reorder_map;
1049
1050
    /* copy and remap input samples */
1051
0
    for (ch = 0; ch < s->channels; ch++) {
1052
        /* copy last 1024 samples of previous frame to the start of the current frame */
1053
0
        memcpy(&s->planar_samples[ch][1024], &s->planar_samples[ch][2048], 1024 * sizeof(s->planar_samples[0][0]));
1054
1055
        /* copy new samples and zero any remaining samples */
1056
0
        if (frame) {
1057
0
            memcpy(&s->planar_samples[ch][2048],
1058
0
                   frame->extended_data[channel_map[ch]],
1059
0
                   frame->nb_samples * sizeof(s->planar_samples[0][0]));
1060
0
        }
1061
0
        memset(&s->planar_samples[ch][end], 0,
1062
0
               (3072 - end) * sizeof(s->planar_samples[0][0]));
1063
0
    }
1064
0
}
1065
1066
static int aac_encode_frame(AVCodecContext *avctx, AVPacket *avpkt,
1067
                            const AVFrame *frame, int *got_packet_ptr)
1068
0
{
1069
0
    AACEncContext *s = avctx->priv_data;
1070
0
    float **samples = s->planar_samples, *samples2, *la, *overlap;
1071
0
    ChannelElement *cpe;
1072
0
    SingleChannelElement *sce;
1073
0
    IndividualChannelStream *ics;
1074
0
    int i, its, ch, w, chans, tag, start_ch, ret, frame_bits;
1075
0
    int target_bits, rate_bits, too_many_bits, too_few_bits;
1076
0
    int ms_mode = 0, is_mode = 0, tns_mode = 0, pred_mode = 0;
1077
0
    int chan_el_counter[4];
1078
0
    FFPsyWindowInfo windows[AAC_MAX_CHANNELS];
1079
1080
    /* add current frame to queue */
1081
0
    if (frame) {
1082
0
        if ((ret = ff_af_queue_add(&s->afq, frame)) < 0)
1083
0
            return ret;
1084
0
    } else {
1085
0
        if (!s->afq.remaining_samples || (!s->afq.frame_alloc && !s->afq.frame_count))
1086
0
            return 0;
1087
0
    }
1088
1089
0
    copy_input_samples(s, frame);
1090
1091
0
    if (!avctx->frame_num)
1092
0
        return 0;
1093
1094
0
    start_ch = 0;
1095
0
    for (i = 0; i < s->chan_map[0]; i++) {
1096
0
        FFPsyWindowInfo* wi = windows + start_ch;
1097
0
        tag      = s->chan_map[i+1];
1098
0
        chans    = tag == TYPE_CPE ? 2 : 1;
1099
0
        cpe      = &s->cpe[i];
1100
0
        {
1101
0
            int wi_paired = 0;
1102
            /* Synced pair windows: decide both channels of a CPE together so
1103
             * their block switching never diverges (see psy window_pair). */
1104
0
            if (chans == 2 && tag != TYPE_LFE && s->psy.model->window_pair && frame) {
1105
0
                const float *ov0 = &samples[start_ch][0],     *ov1 = &samples[start_ch + 1][0];
1106
0
                s->psy.model->window_pair(&s->psy,
1107
0
                                          ov0 + 1024, ov0 + 1024 + 448 + 64,
1108
0
                                          ov1 + 1024, ov1 + 1024 + 448 + 64,
1109
0
                                          start_ch, start_ch + 1,
1110
0
                                          cpe->ch[0].ics.window_sequence[0],
1111
0
                                          cpe->ch[1].ics.window_sequence[0],
1112
0
                                          wi);
1113
0
                wi_paired = 1;
1114
0
            }
1115
0
            for (ch = 0; ch < chans; ch++) {
1116
0
            int k;
1117
0
            float clip_avoidance_factor;
1118
0
            sce = &cpe->ch[ch];
1119
0
            ics = &sce->ics;
1120
0
            s->cur_channel = start_ch + ch;
1121
0
            overlap  = &samples[s->cur_channel][0];
1122
0
            samples2 = overlap + 1024;
1123
0
            la       = samples2 + (448+64);
1124
0
            if (!frame)
1125
0
                la = NULL;
1126
0
            if (tag == TYPE_LFE) {
1127
0
                wi[ch].window_type[0] = wi[ch].window_type[1] = ONLY_LONG_SEQUENCE;
1128
0
                wi[ch].window_shape   = 0;
1129
0
                wi[ch].num_windows    = 1;
1130
0
                wi[ch].grouping[0]    = 1;
1131
0
                wi[ch].clipping[0]    = 0;
1132
1133
                /* Only the lowest 12 coefficients are used in a LFE channel.
1134
                 * The expression below results in only the bottom 8 coefficients
1135
                 * being used for 11.025kHz to 16kHz sample rates.
1136
                 */
1137
0
                ics->num_swb = s->samplerate_index >= 8 ? 1 : 3;
1138
0
            } else if (!wi_paired) {
1139
0
                wi[ch] = s->psy.model->window(&s->psy, samples2, la, s->cur_channel,
1140
0
                                              ics->window_sequence[0]);
1141
0
            }
1142
0
            ics->window_sequence[1] = ics->window_sequence[0];
1143
0
            ics->window_sequence[0] = wi[ch].window_type[0];
1144
0
            ics->use_kb_window[1]   = ics->use_kb_window[0];
1145
0
            ics->use_kb_window[0]   = wi[ch].window_shape;
1146
0
            ics->num_windows        = wi[ch].num_windows;
1147
0
            ics->swb_sizes          = s->psy.bands    [ics->num_windows == 8];
1148
0
            ics->num_swb            = tag == TYPE_LFE ? ics->num_swb : s->psy.num_bands[ics->num_windows == 8];
1149
0
            ics->max_sfb            = FFMIN(ics->max_sfb, ics->num_swb);
1150
0
            ics->swb_offset         = wi[ch].window_type[0] == EIGHT_SHORT_SEQUENCE ?
1151
0
                                        ff_swb_offset_128 [s->samplerate_index]:
1152
0
                                        ff_swb_offset_1024[s->samplerate_index];
1153
0
            ics->tns_max_bands      = wi[ch].window_type[0] == EIGHT_SHORT_SEQUENCE ?
1154
0
                                        ff_tns_max_bands_128 [s->samplerate_index]:
1155
0
                                        ff_tns_max_bands_1024[s->samplerate_index];
1156
1157
0
            for (w = 0; w < ics->num_windows; w++)
1158
0
                ics->group_len[w] = wi[ch].grouping[w];
1159
1160
            /* Calculate input sample maximums and evaluate clipping risk */
1161
0
            clip_avoidance_factor = 0.0f;
1162
0
            for (w = 0; w < ics->num_windows; w++) {
1163
0
                const float *wbuf = overlap + w * 128;
1164
0
                const int wlen = 2048 / ics->num_windows;
1165
0
                float max = 0;
1166
0
                int j;
1167
                /* mdct input is 2 * output */
1168
0
                for (j = 0; j < wlen; j++)
1169
0
                    max = FFMAX(max, fabsf(wbuf[j]));
1170
0
                wi[ch].clipping[w] = max;
1171
0
            }
1172
0
            for (w = 0; w < ics->num_windows; w++) {
1173
0
                if (wi[ch].clipping[w] > CLIP_AVOIDANCE_FACTOR) {
1174
0
                    ics->window_clipping[w] = 1;
1175
0
                    clip_avoidance_factor = FFMAX(clip_avoidance_factor, wi[ch].clipping[w]);
1176
0
                } else {
1177
0
                    ics->window_clipping[w] = 0;
1178
0
                }
1179
0
            }
1180
0
            if (clip_avoidance_factor > CLIP_AVOIDANCE_FACTOR) {
1181
0
                ics->clip_avoidance_factor = CLIP_AVOIDANCE_FACTOR / clip_avoidance_factor;
1182
0
            } else {
1183
0
                ics->clip_avoidance_factor = 1.0f;
1184
0
            }
1185
1186
0
            apply_window_and_mdct(s, sce, overlap);
1187
1188
0
            for (k = 0; k < 1024; k++) {
1189
0
                if (!(fabs(cpe->ch[ch].coeffs[k]) < 1E16)) { // Ensure headroom for energy calculation
1190
0
                    av_log(avctx, AV_LOG_ERROR, "Input contains (near) NaN/+-Inf\n");
1191
0
                    return AVERROR(EINVAL);
1192
0
                }
1193
0
            }
1194
0
            avoid_clipping(s, sce);
1195
0
        }
1196
0
        }
1197
0
        start_ch += chans;
1198
0
    }
1199
0
    if ((ret = ff_alloc_packet(avctx, avpkt, 8192 * s->channels)) < 0)
1200
0
        return ret;
1201
0
    frame_bits = its = 0;
1202
0
    do {
1203
0
        init_put_bits(&s->pb, avpkt->data, avpkt->size);
1204
1205
0
        if ((avctx->frame_num & 0xFF)==1 && !(avctx->flags & AV_CODEC_FLAG_BITEXACT))
1206
0
            put_bitstream_info(s, LIBAVCODEC_IDENT);
1207
0
        start_ch = 0;
1208
0
        target_bits = 0;
1209
0
        memset(chan_el_counter, 0, sizeof(chan_el_counter));
1210
0
        for (i = 0; i < s->chan_map[0]; i++) {
1211
0
            FFPsyWindowInfo* wi = windows + start_ch;
1212
0
            const float *coeffs[2];
1213
0
            tag      = s->chan_map[i+1];
1214
0
            chans    = tag == TYPE_CPE ? 2 : 1;
1215
0
            cpe      = &s->cpe[i];
1216
0
            cpe->common_window = 0;
1217
0
            memset(cpe->is_mask, 0, sizeof(cpe->is_mask));
1218
0
            memset(cpe->ms_mask, 0, sizeof(cpe->ms_mask));
1219
0
            put_bits(&s->pb, 3, tag);
1220
0
            put_bits(&s->pb, 4, chan_el_counter[tag]++);
1221
0
            for (ch = 0; ch < chans; ch++) {
1222
0
                sce = &cpe->ch[ch];
1223
0
                coeffs[ch] = sce->coeffs;
1224
0
                memset(&sce->tns, 0, sizeof(TemporalNoiseShaping));
1225
0
                for (w = 0; w < 128; w++)
1226
0
                    if (sce->band_type[w] > RESERVED_BT)
1227
0
                        sce->band_type[w] = 0;
1228
0
            }
1229
0
            s->psy.bitres.alloc = -1;
1230
0
            s->psy.bitres.bits = s->last_frame_pb_count / s->channels;
1231
0
            s->psy.model->analyze(&s->psy, start_ch, coeffs, wi);
1232
0
            if (s->psy.bitres.alloc > 0) {
1233
                /* Lambda unused here on purpose, we need to take psy's unscaled allocation */
1234
0
                target_bits += s->psy.bitres.alloc
1235
0
                    * (s->lambda / (avctx->global_quality ? avctx->global_quality : 120));
1236
0
                s->psy.bitres.alloc /= chans;
1237
0
            }
1238
0
            s->cur_type = tag;
1239
0
            if (chans > 1
1240
0
                && wi[0].window_type[0] == wi[1].window_type[0]
1241
0
                && wi[0].window_shape   == wi[1].window_shape) {
1242
1243
0
                cpe->common_window = 1;
1244
0
                for (w = 0; w < wi[0].num_windows; w++) {
1245
0
                    if (wi[0].grouping[w] != wi[1].grouping[w]) {
1246
0
                        cpe->common_window = 0;
1247
0
                        break;
1248
0
                    }
1249
0
                }
1250
0
            }
1251
1252
0
            const int use_tns = s->options.tns && s->coder->search_for_tns &&
1253
0
                                s->coder->apply_tns_filt;
1254
1255
            /* The NMR coder rate-controls itself and never re-quantizes, so TNS must run
1256
             * before the quantizer */
1257
0
            const int tns_first = s->options.coder == AAC_CODER_NMR;
1258
0
            if (tns_first && use_tns) {
1259
0
                for (ch = 0; ch < chans; ch++) {
1260
0
                    sce = &cpe->ch[ch];
1261
0
                    s->cur_channel = start_ch + ch;
1262
                    /* mono: mark_pns before TNS so the region cap sees PNS bands. Stereo
1263
                     * PNS is marked in its own block (below) after the stereo decision. */
1264
0
                    if (chans == 1 && s->options.pns && s->coder->mark_pns)
1265
0
                        s->coder->mark_pns(s, avctx, sce);
1266
0
                    s->coder->search_for_tns(s, sce);
1267
0
                    s->coder->apply_tns_filt(s, sce);
1268
0
                    if (sce->tns.present)
1269
0
                        tns_mode = 1;
1270
0
                }
1271
0
            }
1272
1273
            /* NMR stereo PNS (imaging-safe). Mark each channel's noise-like bands on the
1274
             * original L/R psy, then keep PNS only where BOTH channels are noise-like. */
1275
0
            if (chans == 2 && cpe->common_window && tns_first &&
1276
0
                s->options.pns && s->coder->mark_pns) {
1277
0
                s->cur_channel = start_ch;     s->coder->mark_pns(s, avctx, &cpe->ch[0]);
1278
0
                s->cur_channel = start_ch + 1; s->coder->mark_pns(s, avctx, &cpe->ch[1]);
1279
0
                for (int b = 0; b < 128; b++)
1280
0
                    if (!cpe->ch[0].can_pns[b] || !cpe->ch[1].can_pns[b])
1281
0
                        cpe->ch[0].can_pns[b] = cpe->ch[1].can_pns[b] = 0;
1282
0
            }
1283
1284
            /* The NMR coder decides I/S and M/S BEFORE quantization, from the psy model,
1285
             * and the trellis then allocates natively on the coeffs actually coded. */
1286
0
            if (chans == 2 && cpe->common_window && s->options.coder == AAC_CODER_NMR &&
1287
0
                (s->options.mid_side || s->options.intensity_stereo)) {
1288
0
                s->cur_channel = start_ch;
1289
0
                nmr_decide_stereo(s, cpe);
1290
0
            }
1291
            /* NMR pools the CPE bit budget: both channels of a pair are solved
1292
             * jointly under one shared lambda (see aaccoder_nmr.h). */
1293
0
            if (s->options.coder == AAC_CODER_NMR && s->nmr)
1294
0
                s->nmr->pair = (chans == 2);
1295
0
            for (ch = 0; ch < chans; ch++) {
1296
0
                s->cur_channel = start_ch + ch;
1297
                /* NMR PNS is mono-only */
1298
0
                if (s->options.pns && s->coder->mark_pns && !tns_first)
1299
0
                    s->coder->mark_pns(s, avctx, &cpe->ch[ch]);
1300
0
                s->coder->search_for_quantizers(avctx, s, &cpe->ch[ch], s->lambda);
1301
0
            }
1302
0
            for (ch = 0; ch < chans; ch++) { /* TNS (non-NMR) and PNS */
1303
0
                sce = &cpe->ch[ch];
1304
0
                s->cur_channel = start_ch + ch;
1305
0
                if (!tns_first && use_tns) {
1306
0
                    s->coder->search_for_tns(s, sce);
1307
0
                    s->coder->apply_tns_filt(s, sce);
1308
0
                    if (sce->tns.present)
1309
0
                        tns_mode = 1;
1310
0
                }
1311
0
                if (s->options.pns && s->coder->search_for_pns)
1312
0
                    s->coder->search_for_pns(s, avctx, sce);
1313
0
            }
1314
0
            s->cur_channel = start_ch;
1315
0
            if (s->options.intensity_stereo) { /* Intensity Stereo */
1316
0
                if (s->options.coder != AAC_CODER_NMR) { /* NMR: decided pre-search */
1317
0
                    if (s->coder->search_for_is)
1318
0
                        s->coder->search_for_is(s, avctx, cpe);
1319
0
                    apply_intensity_stereo(cpe);
1320
0
                }
1321
0
                if (cpe->is_mode) is_mode = 1;
1322
0
            }
1323
0
            if (s->options.mid_side && s->options.coder != AAC_CODER_NMR) { /* Mid/Side stereo */
1324
0
                if (s->options.mid_side == -1 && s->coder->search_for_ms)
1325
0
                    s->coder->search_for_ms(s, cpe);
1326
0
                else if (cpe->common_window)
1327
0
                    memset(cpe->ms_mask, 1, sizeof(cpe->ms_mask));
1328
0
                apply_mid_side_stereo(cpe);
1329
0
            }
1330
0
            adjust_frame_information(cpe, chans);
1331
0
            if (chans == 2) {
1332
0
                put_bits(&s->pb, 1, cpe->common_window);
1333
0
                if (cpe->common_window) {
1334
0
                    put_ics_info(s, &cpe->ch[0].ics);
1335
0
                    encode_ms_info(&s->pb, cpe);
1336
0
                    if (cpe->ms_mode) ms_mode = 1;
1337
0
                }
1338
0
            }
1339
0
            for (ch = 0; ch < chans; ch++) {
1340
0
                s->cur_channel = start_ch + ch;
1341
0
                encode_individual_channel(avctx, s, &cpe->ch[ch], cpe->common_window);
1342
0
            }
1343
0
            start_ch += chans;
1344
0
        }
1345
1346
0
        if (avctx->flags & AV_CODEC_FLAG_QSCALE) {
1347
            /* When using a constant Q-scale, don't mess with lambda */
1348
0
            break;
1349
0
        }
1350
1351
0
        frame_bits = put_bits_count(&s->pb);
1352
1353
        /* The NMR coder rate-controls itself (global-lambda reservoir servo):
1354
         * per-frame bits intentionally float around the nominal rate, so skip
1355
         * the lambda rate loop and only intervene on a hard overflow. */
1356
0
        if (s->options.coder == AAC_CODER_NMR && avctx->bit_rate_tolerance != 0 &&
1357
0
            frame_bits < 6144 * s->channels - 3)
1358
0
            break;
1359
1360
        /* rate control stuff
1361
         * allow between the nominal bitrate, and what psy's bit reservoir says to target
1362
         * but drift towards the nominal bitrate always
1363
         */
1364
0
        rate_bits = avctx->bit_rate * 1024 / avctx->sample_rate;
1365
0
        rate_bits = FFMIN(rate_bits, 6144 * s->channels - 3);
1366
0
        too_many_bits = FFMAX(target_bits, rate_bits);
1367
0
        too_many_bits = FFMIN(too_many_bits, 6144 * s->channels - 3);
1368
0
        too_few_bits = FFMIN(FFMAX(rate_bits - rate_bits/4, target_bits), too_many_bits);
1369
1370
        /* When strict bit-rate control is demanded */
1371
0
        if (avctx->bit_rate_tolerance == 0) {
1372
0
            if (rate_bits < frame_bits) {
1373
0
                float ratio = ((float)rate_bits) / frame_bits;
1374
0
                s->lambda *= FFMIN(0.9f, ratio);
1375
0
                continue;
1376
0
            }
1377
            /* reset lambda when solution is found */
1378
0
            s->lambda = avctx->global_quality > 0 ? avctx->global_quality : 120;
1379
0
            break;
1380
0
        }
1381
1382
        /* When using ABR, be strict (but only for increasing) */
1383
0
        too_few_bits = too_few_bits - too_few_bits/8;
1384
0
        too_many_bits = too_many_bits + too_many_bits/2;
1385
1386
0
        if (   its == 0 /* for steady-state Q-scale tracking */
1387
0
            || (its < 5 && (frame_bits < too_few_bits || frame_bits > too_many_bits))
1388
0
            || frame_bits >= 6144 * s->channels - 3  )
1389
0
        {
1390
0
            float ratio = ((float)rate_bits) / frame_bits;
1391
1392
0
            if (frame_bits >= too_few_bits && frame_bits <= too_many_bits) {
1393
                /*
1394
                 * This path is for steady-state Q-scale tracking
1395
                 * When frame bits fall within the stable range, we still need to adjust
1396
                 * lambda to maintain it like so in a stable fashion (large jumps in lambda
1397
                 * create artifacts and should be avoided), but slowly
1398
                 */
1399
0
                ratio = sqrtf(sqrtf(ratio));
1400
0
                ratio = av_clipf(ratio, 0.9f, 1.1f);
1401
0
            } else {
1402
                /* Not so fast though */
1403
0
                ratio = sqrtf(ratio);
1404
0
            }
1405
0
            s->lambda = av_clipf(s->lambda * ratio, FLT_EPSILON, 65536.f);
1406
1407
            /* Keep iterating if we must reduce and lambda is in the sky */
1408
0
            if (ratio > 0.9f && ratio < 1.1f) {
1409
0
                break;
1410
0
            } else {
1411
0
                if (is_mode || ms_mode || tns_mode || pred_mode) {
1412
0
                    for (i = 0; i < s->chan_map[0]; i++) {
1413
                        // Must restore coeffs
1414
0
                        chans = tag == TYPE_CPE ? 2 : 1;
1415
0
                        cpe = &s->cpe[i];
1416
0
                        for (ch = 0; ch < chans; ch++)
1417
0
                            memcpy(cpe->ch[ch].coeffs, cpe->ch[ch].pcoeffs, sizeof(cpe->ch[ch].coeffs));
1418
0
                    }
1419
0
                }
1420
0
                its++;
1421
0
            }
1422
0
        } else {
1423
0
            break;
1424
0
        }
1425
0
    } while (1);
1426
1427
    /* tool-usage stats over the final per-band decisions of this frame */
1428
0
    for (i = 0; i < s->chan_map[0]; i++) {
1429
0
        int etag = s->chan_map[i + 1], echans = etag == TYPE_CPE ? 2 : 1;
1430
0
        ChannelElement *ce = &s->cpe[i];
1431
0
        IndividualChannelStream *ics = &ce->ch[0].ics;
1432
0
        for (ch = 0; ch < echans; ch++) {   /* per-channel frame stats */
1433
0
            int is_short = ce->ch[ch].ics.window_sequence[0] == EIGHT_SHORT_SEQUENCE;
1434
0
            s->stat_chans++;
1435
0
            if (is_short)
1436
0
                s->stat_short++;
1437
0
            if (ce->ch[ch].tns.present) {
1438
0
                if (is_short) s->stat_tns_short++;
1439
0
                else          s->stat_tns_long++;
1440
0
            }
1441
0
        }
1442
0
        for (w = 0; w < ics->num_windows; w += ics->group_len[w]) {
1443
0
            for (int g = 0; g < ics->num_swb; g++) {
1444
0
                int idx = w*16 + g, coded = 0;
1445
0
                for (ch = 0; ch < echans; ch++) {
1446
0
                    SingleChannelElement *sce = &ce->ch[ch];
1447
0
                    if (sce->zeroes[idx] && sce->band_type[idx] == 0)
1448
0
                        continue;
1449
0
                    s->stat_ch_bands++;
1450
0
                    if (sce->band_type[idx] == NOISE_BT)
1451
0
                        s->stat_pns++;
1452
0
                    coded = 1;
1453
0
                }
1454
0
                if (etag == TYPE_CPE && coded) {
1455
0
                    s->stat_cpe_bands++;
1456
0
                    if (ce->ms_mask[idx]) s->stat_ms++;
1457
0
                    if (ce->is_mask[idx]) s->stat_is++;
1458
0
                }
1459
0
            }
1460
0
        }
1461
0
    }
1462
1463
0
    put_bits(&s->pb, 3, TYPE_END);
1464
0
    flush_put_bits(&s->pb);
1465
1466
0
    s->last_frame_pb_count = put_bits_count(&s->pb);
1467
1468
    /* NMR rate accounting: how many bits the frame really took beyond what the
1469
     * trellis counted; feeds the next frame's budget correction */
1470
0
    if (s->nmr) {
1471
0
        int counted = 0;
1472
0
        for (i = 0; i < s->channels; i++)
1473
0
            counted += s->nmr->counted[i];
1474
0
        if (counted > 0) {
1475
0
            float side = (float)s->last_frame_pb_count - counted;
1476
0
            if (s->nmr->side_inited) {
1477
0
                s->nmr->side_ema += 0.125f * (side - s->nmr->side_ema);
1478
0
            } else {
1479
0
                s->nmr->side_ema    = side;
1480
0
                s->nmr->side_inited = 1;
1481
0
            }
1482
0
        }
1483
0
    }
1484
0
    avpkt->size            = put_bytes_output(&s->pb);
1485
1486
0
    s->lambda_sum += (s->nmr && s->nmr->lam_rc > 0.0f) ? s->nmr->lam_rc : s->lambda;
1487
0
    s->lambda_count++;
1488
1489
0
    ret = ff_af_queue_remove(&s->afq, avctx->frame_size, avpkt);
1490
0
    if (ret < 0)
1491
0
        return ret;
1492
1493
0
    avpkt->flags |= AV_PKT_FLAG_KEY;
1494
1495
0
    *got_packet_ptr = 1;
1496
0
    return 0;
1497
0
}
1498
1499
static av_cold int aac_encode_end(AVCodecContext *avctx)
1500
0
{
1501
0
    AACEncContext *s = avctx->priv_data;
1502
1503
0
    av_log(avctx, AV_LOG_INFO,
1504
0
           "Qavg: %.3f  Tr: %.1f%%  TNS(L): %.1f%%  TNS(S): %.1f%%  M/S: %.1f%%  I/S: %.1f%%  PNS: %.1f%%\n",
1505
0
           s->lambda_count ? s->lambda_sum / s->lambda_count : NAN,
1506
0
           s->stat_chans     ? 100.0 * s->stat_short      / s->stat_chans               : 0.0,
1507
0
           s->stat_chans - s->stat_short ? 100.0 * s->stat_tns_long  / (s->stat_chans - s->stat_short) : 0.0,
1508
0
           s->stat_short     ? 100.0 * s->stat_tns_short  / s->stat_short               : 0.0,
1509
0
           s->stat_cpe_bands ? 100.0 * s->stat_ms         / s->stat_cpe_bands           : 0.0,
1510
0
           s->stat_cpe_bands ? 100.0 * s->stat_is         / s->stat_cpe_bands           : 0.0,
1511
0
           s->stat_ch_bands  ? 100.0 * s->stat_pns        / s->stat_ch_bands            : 0.0);
1512
1513
0
    av_tx_uninit(&s->mdct1024);
1514
0
    av_tx_uninit(&s->mdct128);
1515
0
    ff_psy_end(&s->psy);
1516
0
    ff_lpc_end(&s->lpc);
1517
0
    av_freep(&s->buffer.samples);
1518
0
    av_freep(&s->cpe);
1519
0
    av_freep(&s->fdsp);
1520
0
    av_freep(&s->nmr);
1521
0
    ff_af_queue_close(&s->afq);
1522
0
    return 0;
1523
0
}
1524
1525
static av_cold int dsp_init(AVCodecContext *avctx, AACEncContext *s)
1526
0
{
1527
0
    int ret = 0;
1528
0
    float scale = 32768.0f;
1529
1530
0
    s->fdsp = avpriv_float_dsp_alloc(avctx->flags & AV_CODEC_FLAG_BITEXACT);
1531
0
    if (!s->fdsp)
1532
0
        return AVERROR(ENOMEM);
1533
1534
0
    if ((ret = av_tx_init(&s->mdct1024, &s->mdct1024_fn, AV_TX_FLOAT_MDCT, 0,
1535
0
                          1024, &scale, 0)) < 0)
1536
0
        return ret;
1537
0
    if ((ret = av_tx_init(&s->mdct128, &s->mdct128_fn,   AV_TX_FLOAT_MDCT, 0,
1538
0
                          128, &scale, 0)) < 0)
1539
0
        return ret;
1540
1541
0
    return 0;
1542
0
}
1543
1544
static av_cold int alloc_buffers(AVCodecContext *avctx, AACEncContext *s)
1545
0
{
1546
0
    int ch;
1547
0
    if (!FF_ALLOCZ_TYPED_ARRAY(s->buffer.samples, s->channels * 3 * 1024) ||
1548
0
        !FF_ALLOCZ_TYPED_ARRAY(s->cpe,            s->chan_map[0]))
1549
0
        return AVERROR(ENOMEM);
1550
1551
0
    for(ch = 0; ch < s->channels; ch++)
1552
0
        s->planar_samples[ch] = s->buffer.samples + 3 * 1024 * ch;
1553
1554
0
    if (s->options.coder == AAC_CODER_NMR) {
1555
0
        s->nmr = av_mallocz(sizeof(*s->nmr));
1556
0
        if (!s->nmr)
1557
0
            return AVERROR(ENOMEM);
1558
0
    }
1559
1560
0
    return 0;
1561
0
}
1562
1563
static av_cold int aac_encode_init(AVCodecContext *avctx)
1564
0
{
1565
0
    AACEncContext *s = avctx->priv_data;
1566
0
    int i, ret = 0;
1567
0
    int chcfg;
1568
0
    const uint8_t *sizes[2];
1569
0
    uint8_t grouping[AAC_MAX_CHANNELS];
1570
0
    int lengths[2];
1571
1572
    /* Constants */
1573
0
    s->last_frame_pb_count = 0;
1574
0
    avctx->frame_size = 1024;
1575
0
    avctx->initial_padding = 1024;
1576
0
    s->lambda = avctx->global_quality > 0 ? avctx->global_quality : 120;
1577
1578
    /* Channel map and unspecified bitrate guessing */
1579
0
    s->channels = avctx->ch_layout.nb_channels;
1580
1581
0
    s->needs_pce = 1;
1582
0
    for (chcfg = 1; chcfg < FF_ARRAY_ELEMS(aac_normal_chan_layouts); chcfg++) {
1583
0
        if (!av_channel_layout_compare(&avctx->ch_layout, &aac_normal_chan_layouts[chcfg])) {
1584
0
            s->needs_pce = s->options.pce;
1585
0
            break;
1586
0
        }
1587
0
    }
1588
1589
0
    if (s->needs_pce) {
1590
0
        char buf[64];
1591
0
        for (i = 0; i < FF_ARRAY_ELEMS(aac_pce_configs); i++)
1592
0
            if (!av_channel_layout_compare(&avctx->ch_layout, &aac_pce_configs[i].layout))
1593
0
                break;
1594
0
        av_channel_layout_describe(&avctx->ch_layout, buf, sizeof(buf));
1595
0
        if (i == FF_ARRAY_ELEMS(aac_pce_configs)) {
1596
0
            av_log(avctx, AV_LOG_ERROR, "Unsupported channel layout \"%s\"\n", buf);
1597
0
            return AVERROR(EINVAL);
1598
0
        }
1599
0
        av_log(avctx, AV_LOG_INFO, "Using a PCE to encode channel layout \"%s\"\n", buf);
1600
0
        s->pce = aac_pce_configs[i];
1601
0
        s->reorder_map = s->pce.reorder_map;
1602
0
        s->chan_map = s->pce.config_map;
1603
0
        chcfg = 0;
1604
0
    } else {
1605
0
        s->reorder_map = aac_chan_maps[chcfg - 1];
1606
0
        s->chan_map = aac_chan_configs[chcfg - 1];
1607
0
    }
1608
1609
0
    if (!avctx->bit_rate) {
1610
0
        for (i = 1; i <= s->chan_map[0]; i++) {
1611
0
            avctx->bit_rate += s->chan_map[i] == TYPE_CPE ? 128000 : /* Pair */
1612
0
                               s->chan_map[i] == TYPE_LFE ? 16000  : /* LFE  */
1613
0
                                                            69000  ; /* SCE  */
1614
0
        }
1615
0
    }
1616
1617
    /* Samplerate */
1618
0
    for (int i = 0;; i++) {
1619
0
        av_assert1(i < 13);
1620
0
        if (avctx->sample_rate == ff_mpeg4audio_sample_rates[i]) {
1621
0
            s->samplerate_index = i;
1622
0
            break;
1623
0
        }
1624
0
    }
1625
1626
    /* Bitrate limiting */
1627
0
    WARN_IF(1024.0 * avctx->bit_rate / avctx->sample_rate > 6144 * s->channels,
1628
0
             "Too many bits %f > %d per frame requested, clamping to max\n",
1629
0
             1024.0 * avctx->bit_rate / avctx->sample_rate,
1630
0
             6144 * s->channels);
1631
0
    avctx->bit_rate = (int64_t)FFMIN(6144 * s->channels / 1024.0 * avctx->sample_rate,
1632
0
                                     avctx->bit_rate);
1633
1634
    /* Profile and option setting */
1635
0
    avctx->profile = avctx->profile == AV_PROFILE_UNKNOWN ? AV_PROFILE_AAC_LOW :
1636
0
                     avctx->profile;
1637
0
    for (i = 0; i < FF_ARRAY_ELEMS(aacenc_profiles); i++)
1638
0
        if (avctx->profile == aacenc_profiles[i])
1639
0
            break;
1640
0
    ERROR_IF(i == FF_ARRAY_ELEMS(aacenc_profiles), "Profile not supported!\n");
1641
0
    if (avctx->profile == AV_PROFILE_MPEG2_AAC_LOW) {
1642
0
        avctx->profile = AV_PROFILE_AAC_LOW;
1643
0
        WARN_IF(s->options.pns,
1644
0
                "PNS unavailable in the \"mpeg2_aac_low\" profile, turning off\n");
1645
0
        s->options.pns = 0;
1646
0
    }
1647
0
    s->profile = avctx->profile;
1648
1649
    /* Coder limitations */
1650
0
    s->coder = &ff_aac_coders[s->options.coder];
1651
1652
    /* M/S introduces horrible artifacts with multichannel files, this is temporary */
1653
0
    if (s->channels > 3)
1654
0
        s->options.mid_side = 0;
1655
1656
    /* Coding bandwidth, fixed at init time */
1657
0
    if (avctx->cutoff > 0) {
1658
0
        s->bandwidth = avctx->cutoff;
1659
0
    } else {
1660
0
        int frame_br = (avctx->flags & AV_CODEC_FLAG_QSCALE) ?
1661
0
                       (avctx->bit_rate / 2.0f * (s->lambda / 120.f) * 1.5f) :
1662
0
                       (avctx->bit_rate / avctx->ch_layout.nb_channels);
1663
1664
0
        if (s->options.coder == AAC_CODER_NMR && frame_br >= 24000) {
1665
0
            static const int rates[] = { 24000, 32000, 48000, 64000, 96000, 192000 };
1666
0
            static const int bws[]   = { 14000, 14000, 18500, 20000, 21000, 22000 };
1667
0
            int bw_i = 0;
1668
0
            for (; bw_i < FF_ARRAY_ELEMS(rates) - 2 && frame_br > rates[bw_i + 1]; bw_i++);
1669
0
            s->bandwidth = bws[bw_i] + (int)((int64_t)(bws[bw_i + 1] - bws[bw_i]) *
1670
0
                                             (frame_br - rates[bw_i]) / (rates[bw_i + 1] - rates[bw_i]));
1671
0
            s->bandwidth = FFMIN3(s->bandwidth, 22000, avctx->sample_rate / 2);
1672
0
        } else {
1673
0
            if (s->options.pns || s->options.intensity_stereo)
1674
0
                frame_br *= 1.15f;
1675
0
            s->bandwidth = FFMAX(3000, AAC_CUTOFF_FROM_BITRATE(frame_br, 1,
1676
0
                                                               avctx->sample_rate));
1677
0
        }
1678
1679
0
        s->bandwidth = FFMIN(FFMAX(s->bandwidth, 8000), avctx->sample_rate / 2);
1680
0
    }
1681
1682
0
    if (!(avctx->flags & AV_CODEC_FLAG_QSCALE) && avctx->bit_rate > 0) {
1683
0
        int bpc = avctx->bit_rate / avctx->ch_layout.nb_channels;
1684
0
        if (bpc <= 32000 && avctx->sample_rate > 32000)
1685
0
            av_log(avctx, AV_LOG_INFO,
1686
0
                   "%d kb/s per channel at %d Hz: consider resampling the "
1687
0
                   "input to 32000 Hz or lower for better quality.\n",
1688
0
                   bpc / 1000, avctx->sample_rate);
1689
0
    }
1690
1691
    // Initialize static tables
1692
0
    ff_aac_float_common_init();
1693
1694
0
    if ((ret = dsp_init(avctx, s)) < 0)
1695
0
        return ret;
1696
1697
0
    if ((ret = alloc_buffers(avctx, s)) < 0)
1698
0
        return ret;
1699
1700
0
    if ((ret = put_audio_specific_config(avctx, chcfg)))
1701
0
        return ret;
1702
1703
0
    sizes[0]   = ff_aac_swb_size_1024[s->samplerate_index];
1704
0
    sizes[1]   = ff_aac_swb_size_128[s->samplerate_index];
1705
0
    lengths[0] = ff_aac_num_swb_1024[s->samplerate_index];
1706
0
    lengths[1] = ff_aac_num_swb_128[s->samplerate_index];
1707
0
    for (i = 0; i < s->chan_map[0]; i++)
1708
0
        grouping[i] = s->chan_map[i + 1] == TYPE_CPE;
1709
0
    if ((ret = ff_psy_init(&s->psy, avctx, 2, sizes, lengths,
1710
0
                           s->chan_map[0], grouping, s->bandwidth)) < 0)
1711
0
        return ret;
1712
0
    ff_lpc_init(&s->lpc, 2*avctx->frame_size, TNS_MAX_ORDER, FF_LPC_TYPE_LEVINSON);
1713
0
    s->random_state = 0x1f2e3d4c;
1714
1715
0
    ff_aacenc_dsp_init(&s->aacdsp);
1716
1717
0
    ff_af_queue_init(avctx, &s->afq);
1718
1719
0
    return 0;
1720
0
}
1721
1722
#define AACENC_FLAGS AV_OPT_FLAG_ENCODING_PARAM | AV_OPT_FLAG_AUDIO_PARAM
1723
static const AVOption aacenc_options[] = {
1724
    {"aac_coder", "Coding algorithm", offsetof(AACEncContext, options.coder), AV_OPT_TYPE_INT, {.i64 = AAC_CODER_NMR}, 0, AAC_CODER_NB-1, AACENC_FLAGS, .unit = "coder"},
1725
        {"twoloop",  "Two loop searching method", 0, AV_OPT_TYPE_CONST, {.i64 = AAC_CODER_TWOLOOP}, INT_MIN, INT_MAX, AACENC_FLAGS, .unit = "coder"},
1726
        {"fast",     "Fast search",               0, AV_OPT_TYPE_CONST, {.i64 = AAC_CODER_FAST},    INT_MIN, INT_MAX, AACENC_FLAGS, .unit = "coder"},
1727
        {"nmr",      "Noise-to-mask ratio scalefactor trellis", 0, AV_OPT_TYPE_CONST, {.i64 = AAC_CODER_NMR}, INT_MIN, INT_MAX, AACENC_FLAGS, .unit = "coder"},
1728
    {"aac_ms", "Force M/S stereo coding", offsetof(AACEncContext, options.mid_side), AV_OPT_TYPE_BOOL, {.i64 = -1}, -1, 1, AACENC_FLAGS},
1729
    {"aac_is", "Intensity stereo coding", offsetof(AACEncContext, options.intensity_stereo), AV_OPT_TYPE_BOOL, {.i64 = 1}, -1, 1, AACENC_FLAGS},
1730
    {"aac_pns", "Perceptual noise substitution", offsetof(AACEncContext, options.pns), AV_OPT_TYPE_BOOL, {.i64 = 1}, -1, 1, AACENC_FLAGS},
1731
    {"aac_tns", "Temporal noise shaping", offsetof(AACEncContext, options.tns), AV_OPT_TYPE_BOOL, {.i64 = 1}, -1, 1, AACENC_FLAGS},
1732
    {"aac_pce", "Forces the use of PCEs", offsetof(AACEncContext, options.pce), AV_OPT_TYPE_BOOL, {.i64 = 0}, -1, 1, AACENC_FLAGS},
1733
    {"aac_nmr_speed", "NMR coder speed level: 0 = slowest/best, higher trades quality for speed", offsetof(AACEncContext, options.nmr_speed), AV_OPT_TYPE_INT, {.i64 = 0}, 0, 4, AACENC_FLAGS},
1734
    FF_AAC_PROFILE_OPTS
1735
    {NULL}
1736
};
1737
1738
static const AVClass aacenc_class = {
1739
    .class_name = "AAC encoder",
1740
    .item_name  = av_default_item_name,
1741
    .option     = aacenc_options,
1742
    .version    = LIBAVUTIL_VERSION_INT,
1743
};
1744
1745
static const FFCodecDefault aac_encode_defaults[] = {
1746
    { "b", "0" },
1747
    { NULL }
1748
};
1749
1750
const FFCodec ff_aac_encoder = {
1751
    .p.name         = "aac",
1752
    CODEC_LONG_NAME("AAC (Advanced Audio Coding)"),
1753
    .p.type         = AVMEDIA_TYPE_AUDIO,
1754
    .p.id           = AV_CODEC_ID_AAC,
1755
    .p.capabilities = AV_CODEC_CAP_DR1 | AV_CODEC_CAP_DELAY |
1756
                      AV_CODEC_CAP_SMALL_LAST_FRAME,
1757
    .priv_data_size = sizeof(AACEncContext),
1758
    .init           = aac_encode_init,
1759
    FF_CODEC_ENCODE_CB(aac_encode_frame),
1760
    .close          = aac_encode_end,
1761
    .defaults       = aac_encode_defaults,
1762
    CODEC_SAMPLERATES_ARRAY(ff_mpeg4audio_sample_rates),
1763
    .caps_internal  = FF_CODEC_CAP_INIT_CLEANUP,
1764
    CODEC_SAMPLEFMTS(AV_SAMPLE_FMT_FLTP),
1765
    .p.priv_class   = &aacenc_class,
1766
};