/src/ffmpeg/libavcodec/aaccoder_nmr.h
Line | Count | Source |
1 | | /* |
2 | | * AAC encoder NMR (noise-to-mask ratio) scalefactor coder |
3 | | * Copyright (c) 2026 Lynne <dev@lynne.ee> |
4 | | * |
5 | | * This file is part of FFmpeg. |
6 | | * |
7 | | * FFmpeg is free software; you can redistribute it and/or |
8 | | * modify it under the terms of the GNU Lesser General Public |
9 | | * License as published by the Free Software Foundation; either |
10 | | * version 2.1 of the License, or (at your option) any later version. |
11 | | * |
12 | | * FFmpeg is distributed in the hope that it will be useful, |
13 | | * but WITHOUT ANY WARRANTY; without even the implied warranty of |
14 | | * MERCHANTABILITY or FITNESS FOR A PARTICULAR PURPOSE. See the GNU |
15 | | * Lesser General Public License for more details. |
16 | | * |
17 | | * You should have received a copy of the GNU Lesser General Public |
18 | | * License along with FFmpeg; if not, write to the Free Software |
19 | | * Foundation, Inc., 51 Franklin Street, Fifth Floor, Boston, MA 02110-1301 USA |
20 | | */ |
21 | | |
22 | | /** |
23 | | * AAC encoder NMR scalefactor coder. |
24 | | * |
25 | | * Optimizes the same noise-to-mask objective as the two-loop coder, but with an |
26 | | * optimal Viterbi search over scalefactors instead of a heuristic loop. For each |
27 | | * coded band the per-scalefactor distortion/bits curve is precomputed, then a |
28 | | * trellis over the (window-group, band) coding sequence minimizes |
29 | | * sum_g = dist_g(sf_g)/threshold_g + |
30 | | * lambda * (spectral_bits_g(sf_g) + scalefactor_differential_bits) |
31 | | * with |sf_g - sf_{g-1}| <= SCALE_MAX_DIFF as a constraint, and lambda |
32 | | * binary-searched so the coded size meets the per-frame bit budget |
33 | | * |
34 | | * Perceptual noise substitution (PNS) is integrated into the same objective: once |
35 | | * the trellis settles on its operating lambda, each noise-like band (flagged by |
36 | | * mark_pns) is offered a terminal "code as noise" candidate whose cost is |
37 | | * nmr_pns + lambda*NMR_PNS_BITS. Because NMR_PNS_BITS is far below a band's spectral bit |
38 | | * count, this candidate only wins when lambda is large, i.e. when the encoder is |
39 | | * struggling to hold the bitrate. The bits freed by the chosen PNS bands are |
40 | | * then re-spent by a second trellis pass over the remaining bands. |
41 | | */ |
42 | | |
43 | | #ifndef AVCODEC_AACCODER_NMR_H |
44 | | #define AVCODEC_AACCODER_NMR_H |
45 | | |
46 | | #include <float.h> |
47 | | #include <string.h> |
48 | | #include "libavutil/mathematics.h" |
49 | | #include "mathops.h" |
50 | | #include "avcodec.h" |
51 | | #include "put_bits.h" |
52 | | #include "aac.h" |
53 | | #include "aacenc.h" |
54 | | #include "aactab.h" |
55 | | #include "aacenctab.h" |
56 | | |
57 | | /* differential scalefactor coding cost, clamped to the legal delta range */ |
58 | 0 | #define NMR_SFBITS(d) ff_aac_scalefactor_bits[av_clip((d) + SCALE_DIFF_ZERO, 0, 2*SCALE_MAX_DIFF)] |
59 | | |
60 | 0 | #define NMR_ITERS 14 /* lambda binary-search iters */ |
61 | 0 | #define NMR_IFINE 9 /* fine-pass lambda iters */ |
62 | 0 | #define NMR_CITERS 7 /* coarse-pass lambda iters */ |
63 | 0 | #define NMR_CWARM 5 /* coarse-pass iters when warm-started off the previous frame's |
64 | | * lambda: the bracket spans 10 octaves instead of ~43, so fewer |
65 | | * bisection steps reach the same resolution */ |
66 | 0 | #define NMR_COARSE 8 /* two-pass coarse->fine grid step, cuts the Viterbi ncand^2 with no |
67 | | * quality loss, 0 disables it (single full-resolution pass) */ |
68 | 0 | #define NMR_STEP 1 /* fine-pass scalefactor candidate granularity */ |
69 | | |
70 | 0 | #define NMR_PNS_BITS 9 /* approx cost in bits of signalling PNS */ |
71 | | |
72 | | /* Spectral-hole fill: noise-like bands the trellis left mostly empty are filled with |
73 | | * energy-matched noise (PNS); an audible hole sounds worse than matched noise. */ |
74 | 0 | #define NMR_PNS_HOLE_FRAC 0.5f |
75 | 0 | #define NMR_PNS_HOLE_SPREAD 0.5f |
76 | | |
77 | | /* RC servo gain: scale the corridor centre by exp2(-K*fill/R) each frame to hold |
78 | | * the long-run mean rate; without it a bad centre drifts for dozens of frames. */ |
79 | 0 | #define NMR_RC_K_CBR 0.5f |
80 | | |
81 | 0 | #define NMR_RC_ITERS 8 /* lambda bisection iters when clamping an over-cap frame */ |
82 | | /* Corridor: bisect within [lam_rc/NMR_RC_CORR, lam_rc*NMR_RC_CORR] so quality stays |
83 | | * smooth while per-frame demand is tracked; 1.5 cuts lambda jitter ~25%. */ |
84 | 0 | #define NMR_RC_CORR 1.5f |
85 | | |
86 | | /* Reservoir half-window (bits/ch); swept 512/1536/3072, 1536 optimal. */ |
87 | | #define NMR_CBR_BUF 1536 |
88 | | /* Slew limit on the FINAL operating lambda per frame; bits deviate instead, |
89 | | * the reservoir absorbs. See memory: aac-castanets-transient-rc. */ |
90 | 0 | #define NMR_SLEW 1.6f |
91 | 0 | #define NMR_SLEW_RUN 1.15f /* within short runs */ |
92 | 0 | #define NMR_RC_CITERS 3 /* corridor coarse-pass iters */ |
93 | | |
94 | | /* Transition premask: an attack cannot mask backwards; clamp a START frame's |
95 | | * thresholds toward the previous long frame's. */ |
96 | | #define NMR_TRANS_PM 2.0f |
97 | | |
98 | | /* Zero-decision hysteresis: previously-coded bands need this margin below |
99 | | * threshold to zero (marginal bands flicker audibly otherwise). */ |
100 | 0 | #define NMR_ZERO_STICKY 0.5f |
101 | | |
102 | | /* Transient bit-burst: an isolated onset (preceded by >= NMR_BURST_GAP long frames) |
103 | | * is coded NMR_BURST_GAIN x finer, held uniform across the run, repaid from steady stretches. */ |
104 | 0 | #define NMR_BURST_GAP 10 |
105 | 0 | #define NMR_BURST_GAIN 8.0f |
106 | | /* Dense-beat boost: short runs with gap < NMR_BURST_GAP get a budget factor |
107 | | * ramping with the gap (starvation-scaled at the use site). */ |
108 | 0 | #define NMR_SHORT_BOOST 2.0f |
109 | 0 | #define NMR_RC_FITERS 4 /* corridor fine-pass iters */ |
110 | 0 | #define NMR_RC_TRACK 0.1f /* per-frame pull of the corridor centre toward the realized lambda */ |
111 | | |
112 | | /* PNS noise-distortion gate: only bands coded well above the masking floor become noise. */ |
113 | 0 | #define NMR_PNS_NDGATE 4.0f |
114 | | |
115 | | /* Energy/threshold cap for PNS: loud bands (energy >> mask) yield clipping random peaks; |
116 | | * only near-masked bands are safe substitution targets. */ |
117 | 0 | #define NMR_PNS_MAX_ET 8.0f |
118 | | |
119 | | /* Operating-lambda floor for PNS: below it the encoder is not struggling, so |
120 | | * substituting real texture for 9 signalling bits is net-negative. */ |
121 | 0 | #define NMR_PNS_LAM 100.0f |
122 | | |
123 | | /* PNS decision hysteresis: enter and leave both cost a margin. */ |
124 | 0 | #define NMR_PNS_ENTER 0.7f |
125 | 0 | #define NMR_PNS_STAY 1.4f |
126 | | /* PNS debounce: enter after NMR_PNS_ON consecutive wants, leave after |
127 | | * NMR_PNS_OFF (chronically marginal bands never qualify). */ |
128 | 0 | #define NMR_PNS_ON 8 |
129 | 0 | #define NMR_PNS_OFF 4 |
130 | | |
131 | | /** |
132 | | * Viterbi over the coding sequence act[0..nact-1] (indices into the per-band |
133 | | * curves nd/nb), with lambda binary-searched so the coded size ~ destbits. |
134 | | * Fills chosen[band] for every band referenced by act. Returns the operating |
135 | | * lambda. node cost = dist/threshold + lambda*spectral_bits; |
136 | | * edge cost = lambda*sf_differential_bits; |delta sf| <= SCALE_MAX_DIFF hard. |
137 | | */ |
138 | | static float nmr_solve(AACEncContext *s, |
139 | | const float (*nd)[NMR_NCAND], const int (*nb)[NMR_NCAND], |
140 | | const int *blo, const int *bnc, int step, |
141 | | const int *act, int nact, int destbits, int *chosen, |
142 | | float lo_l, float hi_l, int iters) |
143 | 0 | { |
144 | 0 | float dp[NMR_NCAND], dpp[NMR_NCAND], node[NMR_NCAND]; |
145 | 0 | float lamsf[2*SCALE_MAX_DIFF + 1]; /* lam*sfdiff bit cost, per lambda */ |
146 | 0 | uint8_t bp[128][NMR_NCAND]; |
147 | 0 | float lam = 1.0f; |
148 | |
|
149 | 0 | if (nact <= 0) |
150 | 0 | return lam; |
151 | | |
152 | 0 | for (int it = 0; it < iters; it++) { |
153 | 0 | lam = sqrtf(lo_l * hi_l); |
154 | 0 | for (int i = 0; i <= 2*SCALE_MAX_DIFF; i++) |
155 | 0 | lamsf[i] = lam * ff_aac_scalefactor_bits[i]; /* edge cost for this lambda */ |
156 | |
|
157 | 0 | int b0 = act[0]; |
158 | 0 | for (int o = 0; o < bnc[b0]; o++) |
159 | 0 | dp[o] = nd[b0][o] + lam * nb[b0][o]; /* anchor band node cost */ |
160 | |
|
161 | 0 | for (int k = 1; k < nact; k++) { |
162 | 0 | int b = act[k], pb = act[k-1]; |
163 | 0 | memcpy(dpp, dp, sizeof(dp)); |
164 | 0 | for (int o = 0; o < bnc[b]; o++) |
165 | 0 | node[o] = nd[b][o] + lam * nb[b][o]; |
166 | | /* dp[o] = node[o] + min_op(dpp[op] + edge cost) */ |
167 | 0 | s->aacdsp.nmr_trellis_step(dp, bp[k], dpp, node, lamsf, |
168 | 0 | bnc[b], bnc[pb], blo[b] - blo[pb], step, |
169 | 0 | SCALE_MAX_DIFF); |
170 | 0 | } |
171 | | |
172 | | /* backtrack */ |
173 | 0 | int beo = 0, b = act[nact-1]; |
174 | 0 | float bec = FLT_MAX; |
175 | 0 | for (int o = 0; o < bnc[b]; o++) |
176 | 0 | if (dp[o] < bec) { bec = dp[o]; beo = o; } |
177 | 0 | chosen[b] = beo; |
178 | 0 | for (int k = nact-1; k > 0; k--) |
179 | 0 | chosen[act[k-1]] = bp[k][chosen[act[k]]]; |
180 | | |
181 | | /* calc cost */ |
182 | 0 | int total = 0; |
183 | 0 | for (int k = 0; k < nact; k++) |
184 | 0 | total += nb[act[k]][chosen[act[k]]]; |
185 | 0 | for (int k = 1; k < nact; k++) |
186 | 0 | total += NMR_SFBITS((blo[act[k]]+chosen[act[k]]*step) - (blo[act[k-1]]+chosen[act[k-1]]*step)); |
187 | |
|
188 | 0 | if (it == iters - 1) |
189 | 0 | break; |
190 | | |
191 | | /* check if we went over budget, go coarser if we did */ |
192 | 0 | if (total > destbits) |
193 | 0 | lo_l = lam; |
194 | 0 | else |
195 | 0 | hi_l = lam; |
196 | 0 | } |
197 | 0 | return lam; |
198 | 0 | } |
199 | | |
200 | | /* Build one coded band's (dist/threshold, bits) cost curve, candidates sf = lo + o*step |
201 | | * for o in [0,maxn), stopping when the band would drop (cb <= 0). Returns the bit count. */ |
202 | | static int nmr_band_curve(AACEncContext *s, SingleChannelElement *sce, int w, int g, |
203 | | int start, int lo, int step, int maxn, float invthr, |
204 | | float maxval, float *nd_row, int *nb_row) |
205 | 0 | { |
206 | 0 | int ncand = 0; |
207 | 0 | for (int o = 0; o < maxn && lo + o*step <= SCALE_MAX_POS; o++) { |
208 | 0 | int sf = lo + o*step, btot = 0, cb = find_min_book(maxval, sf); |
209 | 0 | float dist = 0.0f; |
210 | 0 | if (cb <= 0) |
211 | 0 | break; |
212 | 0 | for (int w2 = 0; w2 < sce->ics.group_len[w]; w2++) { |
213 | 0 | int bb; |
214 | 0 | dist += quantize_band_cost_cached(s, w + w2, g, sce->coeffs + start + w2*128, |
215 | 0 | s->scoefs + start + w2*128, sce->ics.swb_sizes[g], |
216 | 0 | sf, cb, 1.0f, INFINITY, &bb, NULL, 0); |
217 | 0 | btot += bb; |
218 | 0 | } |
219 | 0 | nd_row[ncand] = (dist - btot) * invthr; |
220 | 0 | nb_row[ncand] = btot; |
221 | 0 | ncand++; |
222 | 0 | } |
223 | 0 | return ncand; |
224 | 0 | } |
225 | | |
226 | | /* Zero a channel with nothing codeable; stale band_types would resurrect |
227 | | * bands with chain-illegal scalefactors. */ |
228 | | static void nmr_bail_channel(SingleChannelElement *sce) |
229 | 0 | { |
230 | 0 | for (int i = 0; i < 128; i++) { |
231 | 0 | if (sce->band_type[i] == INTENSITY_BT || sce->band_type[i] == INTENSITY_BT2) |
232 | 0 | continue; |
233 | 0 | sce->zeroes[i] = 1; |
234 | 0 | sce->band_type[i] = 0; |
235 | 0 | } |
236 | 0 | } |
237 | | |
238 | | /* Per-channel setup into slot t: short-block threshold shaping, the |
239 | | * allocation law, zero decisions, and the PASS 1 coarse candidate curves. |
240 | | * Returns the coded-band count; 0 = nothing codeable (caller bails). */ |
241 | | static int nmr_setup_channel(AVCodecContext *avctx, AACEncContext *s, |
242 | | SingleChannelElement *sce, NMRSlot *t) |
243 | 0 | { |
244 | 0 | float (*nd)[NMR_NCAND] = s->nmr->nd[t->si]; |
245 | 0 | int (*nb)[NMR_NCAND] = s->nmr->nb[t->si]; |
246 | 0 | const int cstep = NMR_COARSE > 0 ? NMR_COARSE : NMR_STEP; |
247 | 0 | int allz = 0, cutoff = 1024, nbnd = 0; |
248 | |
|
249 | 0 | uint8_t *zprev = s->nmr->zero_prev[s->cur_channel & 15]; |
250 | 0 | if (s->nmr->zero_nw[s->cur_channel & 15] != sce->ics.num_windows) { |
251 | 0 | memset(zprev, 1, 128); |
252 | 0 | s->nmr->zero_nw[s->cur_channel & 15] = sce->ics.num_windows; |
253 | 0 | } |
254 | |
|
255 | 0 | t->sce = sce; |
256 | 0 | t->cur_ch = s->cur_channel; |
257 | 0 | t->is8 = sce->ics.window_sequence[0] == EIGHT_SHORT_SEQUENCE; |
258 | 0 | t->nbnd = t->nact = 0; |
259 | | |
260 | | /* band cutoff index for this frame's window size; the bandwidth is fixed |
261 | | * at init and shared with the psy model */ |
262 | 0 | cutoff = s->bandwidth * 2 * (1024 / sce->ics.num_windows) / avctx->sample_rate; |
263 | | |
264 | | /* Short-block shaping: temporal premask + per-window threshold flatten. */ |
265 | 0 | if (sce->ics.window_sequence[0] == EIGHT_SHORT_SEQUENCE) { |
266 | 0 | const float pm_p1 = 0.1f, pm_p2 = 2.0f, pm_p3 = 4.0f; |
267 | 0 | for (int g = 0; g < sce->ics.num_swb; g++) { |
268 | 0 | float t1 = FLT_MAX, t2 = FLT_MAX; /* original thr of w-1, w-2 */ |
269 | 0 | for (int w = 0; w < sce->ics.num_windows; w++) { |
270 | 0 | FFPsyBand *b = &s->psy.ch[s->cur_channel].psy_bands[w*16+g]; |
271 | 0 | float th = b->threshold; |
272 | 0 | float c = FFMIN(th, FFMIN(t1*pm_p2, t2*pm_p3)); |
273 | 0 | b->threshold = FFMAX(c, th*pm_p1); |
274 | 0 | t2 = t1; t1 = th; |
275 | 0 | } |
276 | 0 | } |
277 | 0 | { |
278 | 0 | for (int w = 0; w < sce->ics.num_windows; w++) { |
279 | 0 | float sum = 0.0f, esum = 0.0f; int n = 0; |
280 | 0 | for (int g = 0; g < sce->ics.num_swb; g++) { |
281 | 0 | FFPsyBand *b = &s->psy.ch[s->cur_channel].psy_bands[w*16+g]; |
282 | 0 | if (b->energy > b->threshold && b->threshold > 0.0f) { sum += b->threshold; esum += b->energy; n++; } |
283 | 0 | } |
284 | 0 | if (n > 0) { |
285 | | /* keep each window codeable: cap the mean 12dB under the |
286 | | * window's mean audible energy */ |
287 | 0 | float mean = FFMIN(sum / n, (esum / n) * expf(-12.0f * (float)M_LN10 / 10.0f)); |
288 | 0 | for (int g = 0; g < sce->ics.num_swb; g++) { |
289 | 0 | FFPsyBand *b = &s->psy.ch[s->cur_channel].psy_bands[w*16+g]; |
290 | 0 | if (b->energy > b->threshold && b->threshold > 0.0f) |
291 | 0 | b->threshold = FFMIN(mean, b->threshold * 1e9f); |
292 | 0 | } |
293 | 0 | } |
294 | 0 | } |
295 | 0 | } |
296 | 0 | } |
297 | | |
298 | | /* Allocation law; short frames blend to softer energy exponents under |
299 | | * pressure (roll anti-starvation, see memory). */ |
300 | 0 | float a_ae = 0.443f, a_at = 0.111f; |
301 | 0 | if (sce->ics.num_windows == 8 && s->nmr) { |
302 | | /* blend to mask-weighted exponents under rate pressure */ |
303 | 0 | a_ae += (0.35f - a_ae) * s->nmr->press; |
304 | 0 | a_at += (0.3f - a_at) * s->nmr->press; |
305 | 0 | } |
306 | 0 | for (int w = 0; w < sce->ics.num_windows; w += sce->ics.group_len[w]) { |
307 | 0 | int start = 0; |
308 | 0 | for (int g = 0; g < sce->ics.num_swb; start += sce->ics.swb_sizes[g++]) { |
309 | 0 | float uplim = 0.0f, ener = 0.0f, spread = 2.0f; |
310 | 0 | int nz = 0; |
311 | 0 | if (sce->band_type[w*16+g] == INTENSITY_BT || |
312 | 0 | sce->band_type[w*16+g] == INTENSITY_BT2) { |
313 | | /* pre-decided intensity band (right channel): keep its |
314 | | * signalling, it is not trellis-coded */ |
315 | 0 | for (int w2 = 0; w2 < sce->ics.group_len[w]; w2++) |
316 | 0 | sce->zeroes[(w+w2)*16+g] = 0; |
317 | 0 | continue; |
318 | 0 | } |
319 | 0 | float zthr_mul = zprev[w*16+g] ? 1.0f : NMR_ZERO_STICKY; |
320 | | /* M/S side bands: zero-reluctance scaled by side/mid ratio (a tiny |
321 | | * side IS the image; zeroing it flickers). */ |
322 | 0 | if ((t->cur_ch & 1) && s->nmr && s->nmr->pair && |
323 | 0 | s->nmr->smode_band[(t->cur_ch >> 1) & 7][w*16+g] == 1) { |
324 | 0 | const FFPsyBand *mb = &s->psy.ch[s->cur_channel - 1].psy_bands[w*16+g]; |
325 | 0 | float ratio = 0.0f; |
326 | 0 | float eside = 0.0f; |
327 | 0 | for (int w2 = 0; w2 < sce->ics.group_len[w]; w2++) { |
328 | 0 | const FFPsyBand *bb = &s->psy.ch[s->cur_channel].psy_bands[(w+w2)*16+g]; |
329 | 0 | eside += bb->energy; |
330 | 0 | } |
331 | 0 | ratio = eside / FFMAX(mb->energy * sce->ics.group_len[w], 1e-9f); |
332 | 0 | zthr_mul *= 0.25f + 0.75f * av_clipf(ratio / 0.3f, 0.0f, 1.0f); |
333 | 0 | } |
334 | 0 | for (int w2 = 0; w2 < sce->ics.group_len[w]; w2++) { |
335 | 0 | FFPsyBand *band = &s->psy.ch[s->cur_channel].psy_bands[(w+w2)*16+g]; |
336 | 0 | ener += band->energy; |
337 | 0 | spread = FFMIN(spread, band->spread); |
338 | 0 | if (start >= cutoff || band->energy <= band->threshold * zthr_mul || |
339 | 0 | band->threshold == 0.0f) { |
340 | 0 | sce->zeroes[(w+w2)*16+g] = 1; |
341 | 0 | continue; |
342 | 0 | } |
343 | 0 | uplim += band->threshold; |
344 | 0 | nz = 1; |
345 | 0 | } |
346 | 0 | zprev[w*16+g] = !nz; |
347 | 0 | sce->zeroes[w*16+g] = !nz; |
348 | 0 | t->thr_real[w*16+g] = uplim; /* real mask, before the allocation law (PNS gate) */ |
349 | 0 | if (nz && ener > 0.0f && uplim > 0.0f) /* allocation law */ |
350 | 0 | uplim = expf(a_ae * logf(ener) + a_at * logf(uplim)); |
351 | 0 | t->thr[w*16+g] = uplim; |
352 | 0 | t->pener[w*16+g] = ener; |
353 | 0 | t->pspread[w*16+g] = spread; |
354 | 0 | allz |= nz; |
355 | 0 | } |
356 | 0 | } |
357 | 0 | if (!allz) |
358 | 0 | return 0; |
359 | | |
360 | | /* transition premask (see NMR_TRANS_PM) */ |
361 | 0 | if (sce->ics.num_windows == 1) { |
362 | 0 | int ci = t->cur_ch & 15; |
363 | 0 | if (sce->ics.window_sequence[0] == LONG_START_SEQUENCE && |
364 | 0 | s->nmr->thr_prev_ok[ci]) { |
365 | 0 | for (int g = 0; g < sce->ics.num_swb && g < 64; g++) |
366 | 0 | if (t->thr[g] > 0.0f && s->nmr->thr_prev[ci][g] > 0.0f) |
367 | 0 | t->thr[g] = FFMIN(t->thr[g], s->nmr->thr_prev[ci][g] * NMR_TRANS_PM); |
368 | 0 | } |
369 | 0 | for (int g = 0; g < sce->ics.num_swb && g < 64; g++) |
370 | 0 | s->nmr->thr_prev[ci][g] = t->thr[g]; |
371 | 0 | s->nmr->thr_prev_ok[ci] = 1; |
372 | 0 | } else { |
373 | 0 | s->nmr->thr_prev_ok[t->cur_ch & 15] = 0; |
374 | 0 | } |
375 | |
|
376 | 0 | s->aacdsp.abs_pow34(s->scoefs, sce->coeffs, 1024); |
377 | 0 | ff_quantize_band_cost_cache_init(s); |
378 | | |
379 | | /* TNS synthesis gain per band: the decoder re-amplifies residual-domain |
380 | | * quantization noise by the whitening gain (shorts only). */ |
381 | 0 | for (int i = 0; i < 128; i++) |
382 | 0 | t->tnsg[i] = 1.0f; |
383 | 0 | if (sce->ics.num_windows == 8 && sce->tns.present) { |
384 | 0 | const int mmm2 = FFMIN(sce->ics.tns_max_bands, sce->ics.max_sfb ? sce->ics.max_sfb : sce->ics.num_swb); |
385 | 0 | for (int w = 0; w < 8; w++) { |
386 | 0 | int bottom2 = sce->ics.num_swb; |
387 | 0 | for (int filt = 0; filt < sce->tns.n_filt[w]; filt++) { |
388 | 0 | int top2 = bottom2; |
389 | 0 | bottom2 = FFMAX(0, top2 - sce->tns.length[w][filt]); |
390 | 0 | if (!sce->tns.order[w][filt]) |
391 | 0 | continue; |
392 | 0 | for (int g = FFMIN(bottom2, mmm2); g < FFMIN(top2, mmm2); g++) { |
393 | 0 | int s0 = sce->ics.swb_offset[g] + w*128; |
394 | 0 | int s1 = sce->ics.swb_offset[g+1] + w*128; |
395 | 0 | float eres = 0.0f; |
396 | 0 | const FFPsyBand *pb = &s->psy.ch[s->cur_channel].psy_bands[w*16+g]; |
397 | 0 | for (int k = s0; k < s1; k++) |
398 | 0 | eres += sce->coeffs[k]*sce->coeffs[k]; |
399 | 0 | t->tnsg[w*16+g] = av_clipf(pb->energy / FFMAX(eres, 1e-12f), 1.0f, 64.0f); |
400 | 0 | } |
401 | 0 | } |
402 | 0 | } |
403 | 0 | } |
404 | | |
405 | | /* finest codeable scalefactor and max value per band */ |
406 | 0 | for (int w = 0; w < sce->ics.num_windows; w += sce->ics.group_len[w]) { |
407 | 0 | int start = w*128; |
408 | 0 | for (int g = 0; g < sce->ics.num_swb; g++) { |
409 | 0 | t->maxvals[w*16+g] = find_max_val(sce->ics.group_len[w], sce->ics.swb_sizes[g], s->scoefs + start); |
410 | 0 | t->minsf[w*16+g] = t->maxvals[w*16+g] > 0 ? coef2minsf(t->maxvals[w*16+g]) : 0; |
411 | 0 | start += sce->ics.swb_sizes[g]; |
412 | 0 | } |
413 | 0 | } |
414 | | |
415 | | /* PASS 1: coarse candidate curves per coded band |
416 | | * (the lambda search runs on this cheap grid, PASS 2 refines the winner) */ |
417 | 0 | { |
418 | 0 | for (int w = 0; w < sce->ics.num_windows; w += sce->ics.group_len[w]) { |
419 | 0 | int start = w*128; |
420 | 0 | for (int g = 0; g < sce->ics.num_swb; g++) { |
421 | 0 | if (!sce->zeroes[w*16+g] && t->maxvals[w*16+g] > 0 && nbnd < 128) { |
422 | 0 | int lo = av_clip(t->minsf[w*16+g], 0, SCALE_MAX_POS); |
423 | 0 | float invthr = 1.0f / FFMAX(t->thr[w*16+g], 1e-9f); |
424 | 0 | int ncand = nmr_band_curve(s, sce, w, g, start, lo, cstep, NMR_NCAND, |
425 | 0 | invthr, t->maxvals[w*16+g], nd[nbnd], nb[nbnd]); |
426 | 0 | if (t->tnsg[w*16+g] > 1.0f) |
427 | 0 | for (int o = 0; o < ncand; o++) |
428 | 0 | nd[nbnd][o] *= t->tnsg[w*16+g]; |
429 | 0 | if (ncand == 0) { |
430 | | /* nothing codeable: drop the group band incl. subwindow |
431 | | * flags (group flag is re-derived by ANDing) */ |
432 | 0 | for (int w2 = 0; w2 < sce->ics.group_len[w]; w2++) |
433 | 0 | sce->zeroes[(w+w2)*16+g] = 1; |
434 | 0 | } else { |
435 | 0 | t->bidx[nbnd] = w*16+g; |
436 | 0 | t->bw[nbnd] = w; |
437 | 0 | t->bg[nbnd] = g; |
438 | 0 | t->bst[nbnd] = start; |
439 | 0 | t->blo[nbnd] = lo; |
440 | 0 | t->bnc[nbnd] = ncand; |
441 | 0 | nbnd++; |
442 | 0 | } |
443 | 0 | } |
444 | 0 | start += sce->ics.swb_sizes[g]; |
445 | 0 | } |
446 | 0 | } |
447 | 0 | } |
448 | 0 | t->nbnd = nbnd; |
449 | 0 | for (int b = 0; b < nbnd; b++) { |
450 | 0 | t->act[b] = b; |
451 | 0 | t->is_pns[b] = 0; |
452 | 0 | } |
453 | 0 | t->nact = nbnd; |
454 | 0 | return nbnd; |
455 | 0 | } |
456 | | |
457 | | /* total bits of a slot's current chosen[] on grid `step`, incl. sf deltas */ |
458 | | static int nmr_slot_bits(const NMRSlot *t, const int (*nb)[NMR_NCAND], int step) |
459 | 0 | { |
460 | 0 | int tot = 0; |
461 | 0 | for (int k = 0; k < t->nact; k++) |
462 | 0 | tot += nb[t->act[k]][t->chosen[t->act[k]]]; |
463 | 0 | for (int k = 1; k < t->nact; k++) |
464 | 0 | tot += NMR_SFBITS((t->blo[t->act[k]]+t->chosen[t->act[k]]*step) - |
465 | 0 | (t->blo[t->act[k-1]]+t->chosen[t->act[k-1]]*step)); |
466 | 0 | return tot; |
467 | 0 | } |
468 | | |
469 | | /* Run every slot's trellis at one fixed lambda; returns the pooled bits. */ |
470 | | static int nmr_eval_slots(AACEncContext *s, NMRSlot *const *sl, int nsl, int step, float lam) |
471 | 0 | { |
472 | 0 | int total = 0; |
473 | 0 | for (int k = 0; k < nsl; k++) { |
474 | 0 | NMRSlot *t = sl[k]; |
475 | 0 | if (!t->nact) |
476 | 0 | continue; |
477 | 0 | nmr_solve(s, s->nmr->nd[t->si], s->nmr->nb[t->si], t->blo, t->bnc, step, |
478 | 0 | t->act, t->nact, 0, t->chosen, lam, lam, 1); |
479 | 0 | total += nmr_slot_bits(t, s->nmr->nb[t->si], step); |
480 | 0 | } |
481 | 0 | return total; |
482 | 0 | } |
483 | | |
484 | | /* Bisect ONE shared lambda across the slots so the POOLED bits meet destbits. |
485 | | * This is the CPE budget pool: bits flow to whichever channel of the pair has |
486 | | * demand at the common operating point, instead of an equal per-channel split. */ |
487 | | static float nmr_solve_slots(AACEncContext *s, NMRSlot *const *sl, int nsl, int step, |
488 | | int destbits, float lo_l, float hi_l, int iters) |
489 | 0 | { |
490 | 0 | float lam = 1.0f; |
491 | 0 | for (int it = 0; it < iters; it++) { |
492 | 0 | lam = sqrtf(lo_l * hi_l); |
493 | 0 | int total = nmr_eval_slots(s, sl, nsl, step, lam); |
494 | 0 | if (it == iters - 1) |
495 | 0 | break; |
496 | | /* over budget -> go coarser */ |
497 | 0 | if (total > destbits) |
498 | 0 | lo_l = lam; |
499 | 0 | else |
500 | 0 | hi_l = lam; |
501 | 0 | } |
502 | 0 | return lam; |
503 | 0 | } |
504 | | |
505 | | /* Write a solved slot back into its channel: band types, scalefactors, and the |
506 | | * SCALE_MAX_DIFF legality fixups. Verbatim from the pre-pool single-channel tail. */ |
507 | | static void nmr_commit_channel(AACEncContext *s, NMRSlot *t) |
508 | 0 | { |
509 | 0 | SingleChannelElement *sce = t->sce; |
510 | 0 | const int (*nb)[NMR_NCAND] = (const int (*)[NMR_NCAND])s->nmr->nb[t->si]; |
511 | |
|
512 | 0 | for (int b = 0; b < t->nbnd; b++) { |
513 | 0 | int bi = t->bidx[b]; |
514 | 0 | if (t->is_pns[b]) { |
515 | 0 | sce->band_type[bi] = NOISE_BT; |
516 | 0 | sce->zeroes[bi] = 0; |
517 | 0 | sce->pns_ener[bi] = t->pener[bi] * FFMIN(1.0f, t->pspread[bi]*t->pspread[bi]); |
518 | 0 | } else { |
519 | 0 | sce->sf_idx[bi] = av_clip(t->blo[b] + t->chosen[b]*NMR_STEP, 0, SCALE_MAX_POS); |
520 | 0 | } |
521 | 0 | } |
522 | | |
523 | |
|
524 | 0 | { /* record the bits this solve accounted for; the encoder compares them |
525 | | * against the channel's real output to keep the budget honest */ |
526 | 0 | int tot = 0, prevb = -1; |
527 | 0 | for (int b = 0; b < t->nbnd; b++) { |
528 | 0 | if (t->is_pns[b]) |
529 | 0 | continue; |
530 | 0 | tot += nb[b][t->chosen[b]]; |
531 | 0 | if (prevb >= 0) |
532 | 0 | tot += NMR_SFBITS((t->blo[b]+t->chosen[b]*NMR_STEP) - (t->blo[prevb]+t->chosen[prevb]*NMR_STEP)); |
533 | 0 | prevb = b; |
534 | 0 | } |
535 | 0 | s->nmr->counted[t->cur_ch] = tot; |
536 | 0 | } |
537 | | |
538 | | /* SCALE_MAX_DIFF condition: |
539 | | * re-clamp, codebook fixup, drop uncodeable, set global gain |
540 | | * NOISE_BT bands keep their own scalefactor chain via set_special_band_scalefactors) */ |
541 | 0 | { |
542 | 0 | uint8_t nextband[128]; |
543 | 0 | int prev = -1; |
544 | 0 | ff_init_nextband_map(sce, nextband); |
545 | 0 | for (int w = 0; w < sce->ics.num_windows; w += sce->ics.group_len[w]) { |
546 | 0 | for (int g = 0; g < sce->ics.num_swb; g++) { |
547 | 0 | if (sce->band_type[w*16+g] == NOISE_BT || |
548 | 0 | sce->band_type[w*16+g] == INTENSITY_BT || |
549 | 0 | sce->band_type[w*16+g] == INTENSITY_BT2) |
550 | 0 | continue; |
551 | 0 | if (sce->zeroes[w*16+g]) { |
552 | 0 | sce->band_type[w*16+g] = 0; |
553 | 0 | continue; |
554 | 0 | } |
555 | | |
556 | 0 | if (prev != -1) |
557 | 0 | sce->sf_idx[w*16+g] = av_clip(sce->sf_idx[w*16+g], prev - SCALE_MAX_DIFF, prev + SCALE_MAX_DIFF); |
558 | 0 | sce->band_type[w*16+g] = find_min_book(t->maxvals[w*16+g], sce->sf_idx[w*16+g]); |
559 | 0 | if (sce->band_type[w*16+g] <= 0) { |
560 | 0 | if (!ff_sfdelta_can_remove_band(sce, nextband, prev, w*16+g)) { |
561 | 0 | sce->band_type[w*16+g] = 1; |
562 | 0 | } else { |
563 | | /* drop subwindow flags too, see the PASS 1 drop above */ |
564 | 0 | for (int w2 = 0; w2 < sce->ics.group_len[w]; w2++) |
565 | 0 | sce->zeroes[(w+w2)*16+g] = 1; |
566 | 0 | sce->band_type[w*16+g] = 0; |
567 | 0 | continue; |
568 | 0 | } |
569 | 0 | } |
570 | 0 | if (prev == -1) |
571 | 0 | sce->sf_idx[0] = sce->sf_idx[w*16+g]; /* global gain */ |
572 | 0 | prev = sce->sf_idx[w*16+g]; |
573 | 0 | } |
574 | 0 | } |
575 | | |
576 | | /* every band must carry a chain-legal scalefactor (re-clamp, codebook |
577 | | * fixup, global gain) */ |
578 | 0 | if (prev != -1) { |
579 | 0 | int last = sce->sf_idx[0]; |
580 | 0 | for (int w = 0; w < sce->ics.num_windows; w += sce->ics.group_len[w]) { |
581 | 0 | for (int g = 0; g < sce->ics.num_swb; g++) { |
582 | 0 | if (!sce->zeroes[w*16+g] && sce->band_type[w*16+g] != NOISE_BT && |
583 | 0 | sce->band_type[w*16+g] < RESERVED_BT) |
584 | 0 | last = sce->sf_idx[w*16+g]; |
585 | 0 | else if (sce->band_type[w*16+g] < RESERVED_BT && (w*16+g) > 0) |
586 | 0 | sce->sf_idx[w*16+g] = last; |
587 | 0 | } |
588 | 0 | } |
589 | 0 | } |
590 | 0 | } |
591 | 0 | } |
592 | | |
593 | | /* Solve one element group (a solo channel, or a CPE pair pooled under one |
594 | | * shared lambda and one pooled budget), then PNS and commit. */ |
595 | | static void nmr_solve_group(AVCodecContext *avctx, AACEncContext *s, |
596 | | const float lambda, NMRSlot *const *sl, int nsl, |
597 | | int chans, int rc_eligible, int rc_global, |
598 | | int rc_rate_frame, int rc_bmax) |
599 | 0 | { |
600 | 0 | const int cstep = NMR_COARSE > 0 ? NMR_COARSE : NMR_STEP; |
601 | 0 | int bch = ((avctx->flags & AV_CODEC_FLAG_QSCALE) ? 2.0f : avctx->ch_layout.nb_channels); |
602 | 0 | int destbits = avctx->bit_rate * 1024.0 / avctx->sample_rate / bch * (lambda / 120.f) * chans; |
603 | 0 | int is8_any = 0; |
604 | 0 | float lam; |
605 | 0 | float rc_off = 1.0f, lam_dem = 0.0f; |
606 | |
|
607 | 0 | for (int k = 0; k < nsl; k++) |
608 | 0 | is8_any |= sl[k]->is8; |
609 | |
|
610 | 0 | if (s->psy.bitres.alloc >= 0) |
611 | 0 | destbits = s->psy.bitres.alloc * |
612 | 0 | (lambda / (avctx->global_quality ? avctx->global_quality : 120)) * chans; |
613 | 0 | if (rc_global && s->psy.bitres.alloc >= 0) { |
614 | | /* CBR target: nominal + repayment, bounded +-30%/frame */ |
615 | 0 | double rr = avctx->bit_rate * 1024.0 / avctx->sample_rate; |
616 | 0 | destbits = (rr + av_clipd(s->nmr->rc_fill / 2.0, -0.3 * rr, 0.3 * rr)) * chans / s->channels; |
617 | 0 | } else if (rc_eligible && s->psy.bitres.alloc >= 0) { |
618 | | /* pre-bootstrap CBR frames: target nominal (psy bitres is cold) */ |
619 | 0 | destbits = (avctx->bit_rate * 1024.0 / avctx->sample_rate) * chans / s->channels; |
620 | 0 | } |
621 | 0 | destbits = FFMIN(destbits, 5800 * chans); |
622 | | /* honest budget: subtract the measured non-trellis overhead (section data, ICS, |
623 | | * sf/PNS signalling), which is rate-dependent hence adaptive. */ |
624 | 0 | if (s->nmr->side_inited) |
625 | 0 | destbits = av_clip(destbits - (int)(s->nmr->side_ema * chans / s->channels), 64, 5800 * chans); |
626 | | |
627 | | /* Held transient burst, bank-aware: spend banked bits, never borrow deep |
628 | | * (payback troughs starve the next transient). */ |
629 | 0 | if (s->nmr->run_burst > 1.0f) { |
630 | 0 | int extra = destbits * (s->nmr->run_burst - 1.0f); |
631 | 0 | int avail = FFMAX(0, (int)((s->nmr->rc_fill + rc_bmax / 2) * (int64_t)chans / s->channels)); |
632 | 0 | destbits = av_clip(destbits + FFMIN(extra, avail), 64, 6800 * chans); |
633 | 0 | } |
634 | |
|
635 | 0 | if (rc_global) { |
636 | | /* corridor bisect around the servoed centre; pressure = stateless |
637 | | * rc_off multiplier (folding it into lam_rc winds up) */ |
638 | 0 | float R = avctx->bit_rate * 1024.0 / avctx->sample_rate; |
639 | 0 | float cen; |
640 | 0 | int tot, hardcap, rc_cap; |
641 | 0 | float lo; |
642 | 0 | rc_off = exp2f(-NMR_RC_K_CBR * s->nmr->rc_fill / R); |
643 | 0 | cen = s->nmr->lam_rc * rc_off; |
644 | 0 | lo = cen / NMR_RC_CORR; |
645 | | /* transient burst: widen the lower bound so the boosted destbits can |
646 | | * actually pour into the onset frame */ |
647 | 0 | if (is8_any && s->nmr->run_burst > 1.0f) |
648 | 0 | lo /= s->nmr->run_burst; |
649 | 0 | lam = nmr_solve_slots(s, sl, nsl, cstep, destbits, |
650 | 0 | lo, cen * NMR_RC_CORR, NMR_RC_CITERS); |
651 | |
|
652 | 0 | tot = 0; |
653 | 0 | for (int k = 0; k < nsl; k++) |
654 | 0 | tot += nmr_slot_bits(sl[k], s->nmr->nb[sl[k]->si], cstep); |
655 | 0 | hardcap = av_clip((int)(5800.f * FFMIN(1.f, lambda / 120.f)), 256, 5800) * chans; |
656 | | /* legality cap only; no spend-floor (rc_off spends the bank) */ |
657 | 0 | rc_cap = FFMIN(hardcap, (s->nmr->rc_fill + rc_rate_frame + rc_bmax) * chans / s->channels); |
658 | 0 | if (tot > rc_cap) { |
659 | 0 | lam = nmr_solve_slots(s, sl, nsl, cstep, rc_cap, lam, 1e4f, NMR_CITERS); |
660 | 0 | } |
661 | 0 | } else { |
662 | | /* per-frame bisection, warm-started off the previous frame's lambda; |
663 | | * a result at the bracket edge means redo the full search */ |
664 | 0 | float lam0 = s->nmr->lam[sl[0]->cur_ch]; |
665 | 0 | lam = 1.0f; |
666 | 0 | if (NMR_COARSE > 0 && lam0 > 0.0f) { |
667 | 0 | lam = nmr_solve_slots(s, sl, nsl, cstep, destbits, lam0/32.0f, lam0*32.0f, NMR_CWARM); |
668 | 0 | if (lam < lam0/16.0f || lam > lam0*16.0f) |
669 | 0 | lam0 = 0.0f; |
670 | 0 | } |
671 | 0 | if (lam0 <= 0.0f) |
672 | 0 | lam = nmr_solve_slots(s, sl, nsl, cstep, destbits, |
673 | 0 | 1e-9f, 1e4f, NMR_COARSE > 0 ? NMR_CITERS : NMR_ITERS); |
674 | 0 | } |
675 | | |
676 | | /* PASS 2: |
677 | | * refine each band at full granularity (NMR_STEP) in a +/-cstep window |
678 | | * around the coarse pick, then re-solve. Recovers single-pass quality while the |
679 | | * lambda search stayed cheap on the coarse grid. */ |
680 | 0 | if (NMR_COARSE > 0) { |
681 | | /* nmr_speed, 0 = slowest/best, higher = faster; see the option docs. */ |
682 | 0 | int win = NMR_COARSE - av_clip(s->options.nmr_speed, 0, 4); |
683 | 0 | for (int k = 0; k < nsl; k++) { |
684 | 0 | NMRSlot *t = sl[k]; |
685 | 0 | float (*ndk)[NMR_NCAND] = s->nmr->nd[t->si]; |
686 | 0 | int (*nbk)[NMR_NCAND] = s->nmr->nb[t->si]; |
687 | 0 | if (!t->nact) |
688 | 0 | continue; |
689 | | /* the pow34 spectrum and the quantize cache are per-channel state */ |
690 | 0 | s->aacdsp.abs_pow34(s->scoefs, t->sce->coeffs, 1024); |
691 | 0 | ff_quantize_band_cost_cache_init(s); |
692 | 0 | for (int b = 0; b < t->nbnd; b++) { |
693 | 0 | int center = t->blo[b] + t->chosen[b]*cstep; |
694 | 0 | int flo = av_clip(center - win, av_clip(t->minsf[t->bidx[b]], 0, SCALE_MAX_POS), SCALE_MAX_POS); |
695 | 0 | int maxn = FFMIN(NMR_NCAND, 2*win/NMR_STEP + 1); |
696 | 0 | float invthr = 1.0f / FFMAX(t->thr[t->bidx[b]], 1e-9f); |
697 | 0 | int ncand = nmr_band_curve(s, t->sce, t->bw[b], t->bg[b], t->bst[b], flo, NMR_STEP, maxn, |
698 | 0 | invthr, t->maxvals[t->bidx[b]], ndk[b], nbk[b]); |
699 | 0 | if (t->tnsg[t->bidx[b]] > 1.0f) |
700 | 0 | for (int o = 0; o < ncand; o++) |
701 | 0 | ndk[b][o] *= t->tnsg[t->bidx[b]]; |
702 | 0 | t->blo[b] = flo; |
703 | 0 | t->bnc[b] = FFMAX(1, ncand); |
704 | 0 | } |
705 | 0 | } |
706 | | /* fine pass: narrow corridor around the coarse solve */ |
707 | 0 | if (rc_global) |
708 | 0 | lam = nmr_solve_slots(s, sl, nsl, NMR_STEP, destbits, lam/2.0f, lam*2.0f, NMR_RC_FITERS); |
709 | 0 | else |
710 | 0 | lam = nmr_solve_slots(s, sl, nsl, NMR_STEP, destbits, lam/16.0f, lam*16.0f, NMR_IFINE); |
711 | 0 | } |
712 | |
|
713 | 0 | lam_dem = lam; /* demand-solved lambda, pre bucket clamp: what content wants */ |
714 | |
|
715 | 0 | if (rc_global) { |
716 | | /* legality clamp, then the quality slew limiter */ |
717 | 0 | int hardcap = av_clip((int)(5800.f * FFMIN(1.f, lambda / 120.f)), 256, 5800) * chans; |
718 | 0 | int tot = 0, rc_cap; |
719 | 0 | for (int k = 0; k < nsl; k++) |
720 | 0 | tot += nmr_slot_bits(sl[k], s->nmr->nb[sl[k]->si], NMR_STEP); |
721 | 0 | rc_cap = FFMIN(hardcap, (s->nmr->rc_fill + rc_rate_frame + rc_bmax) * chans / s->channels); |
722 | 0 | if (tot > rc_cap) { |
723 | 0 | lam = nmr_solve_slots(s, sl, nsl, NMR_STEP, rc_cap, lam, 1e4f, NMR_RC_ITERS); |
724 | 0 | } |
725 | 0 | if (s->nmr->lam_slew > 0.0f) { |
726 | 0 | float kup, kdn; |
727 | | /* hold lambda near-constant within short runs; bits follow content */ |
728 | 0 | kup = (is8_any && s->nmr->prev_was_short) ? NMR_SLEW_RUN : NMR_SLEW; |
729 | | /* a deliberate onset burst may dive as far as its widened corridor |
730 | | * allows; the RECOVERY back up is what must stay gradual */ |
731 | 0 | kdn = (is8_any && s->nmr->run_burst > 1.0f) ? NMR_SLEW * s->nmr->run_burst : |
732 | 0 | (is8_any && s->nmr->prev_was_short) ? NMR_SLEW_RUN : NMR_SLEW; |
733 | 0 | if (lam > s->nmr->lam_slew * kup || lam < s->nmr->lam_slew / kdn) { |
734 | 0 | lam = av_clipf(lam, s->nmr->lam_slew / kdn, s->nmr->lam_slew * kup); |
735 | 0 | tot = nmr_eval_slots(s, sl, nsl, NMR_STEP, lam); |
736 | | /* never at the price of an illegal reservoir excursion */ |
737 | 0 | if (tot > rc_cap) { |
738 | 0 | lam = nmr_solve_slots(s, sl, nsl, NMR_STEP, rc_cap, lam, 1e4f, NMR_RC_ITERS); |
739 | 0 | } |
740 | 0 | } |
741 | 0 | } |
742 | 0 | s->nmr->lam_slew = lam; |
743 | 0 | } |
744 | |
|
745 | 0 | for (int k = 0; k < nsl; k++) |
746 | 0 | s->nmr->lam[sl[k]->cur_ch] = lam; /* warm start for the next frame */ |
747 | 0 | { /* nd: mean achieved dist/real-mask (dimensionless starvation + |
748 | | * noise-class signal) */ |
749 | 0 | float ndsum = 0.0f; int ndn = 0; |
750 | 0 | for (int k = 0; k < nsl; k++) { |
751 | 0 | NMRSlot *t = sl[k]; |
752 | 0 | float (*ndk)[NMR_NCAND] = s->nmr->nd[t->si]; |
753 | 0 | for (int b_ = 0; b_ < t->nact; b_++) { |
754 | 0 | int b = t->act[b_], bi = t->bidx[b]; |
755 | 0 | if (t->thr_real[bi] > 0.0f && t->thr[bi] > 0.0f) { |
756 | 0 | ndsum += ndk[b][t->chosen[b]] * t->thr[bi] / t->thr_real[bi]; |
757 | 0 | ndn++; |
758 | 0 | } |
759 | 0 | } |
760 | 0 | } |
761 | | /* long frames only (short groups inflate the ratio) */ |
762 | 0 | if (ndn >= 8 && !is8_any) { |
763 | 0 | float nd = ndsum / ndn; |
764 | 0 | s->nmr->nd_ema = s->nmr->nd_ema > 0.0f ? |
765 | 0 | 0.95f * s->nmr->nd_ema + 0.05f * nd : nd; |
766 | 0 | } |
767 | 0 | } |
768 | 0 | { /* track short vs long operating lambda (dense-beat boost scaling) */ |
769 | 0 | float *ema = is8_any ? &s->nmr->lam_short_ema : &s->nmr->lam_long_ema; |
770 | 0 | *ema = *ema > 0.0f ? 0.9f * *ema + 0.1f * lam : lam; |
771 | | /* sustained-strain floor: snaps down at any comfortable moment, |
772 | | * recovers only slowly, so bursty content cannot bank pressure |
773 | | * credit between its lambda valleys. */ |
774 | 0 | s->nmr->lam_floor = s->nmr->lam_floor > 0.0f ? |
775 | 0 | fminf(s->nmr->lam_floor * 1.02f, lam) : lam; |
776 | 0 | } |
777 | 0 | { /* shared rate-pressure ramp: lambda vs nd-scaled anchors */ |
778 | 0 | float scale, ramp; |
779 | 0 | scale = 1.0f + av_clipf(s->nmr->nd_ema / 50.0f, 0.0f, 8.0f); |
780 | 0 | ramp = s->nmr->lam_long_ema > 0.0f ? |
781 | 0 | av_clipf((s->nmr->lam_long_ema - 120.0f * scale) / |
782 | 0 | (350.0f * scale - 120.0f * scale), 0.0f, 1.0f) : 0.0f; |
783 | | /* transparency veto: lambda*nd below ~74 = comfortable */ |
784 | 0 | if (s->nmr->nd_ema > 0.0f) |
785 | 0 | ramp *= av_clipf((s->nmr->lam_long_ema * s->nmr->nd_ema - 60.0f) / |
786 | 0 | (120.0f - 60.0f), 0.0f, 1.0f); |
787 | 0 | s->nmr->press = ramp; |
788 | 0 | } |
789 | 0 | if (rc_global) { |
790 | | /* track the centre toward the CONTENT lambda (demand-solved, pressure |
791 | | * divided out); clamped lambda is rate noise, not content */ |
792 | 0 | float c = s->nmr->lam_rc * powf(lam_dem / rc_off / s->nmr->lam_rc, NMR_RC_TRACK); |
793 | 0 | s->nmr->lam_rc = av_clipf(c, 1e-6f, 1e4f); |
794 | 0 | } else if (rc_eligible) { |
795 | | /* bootstrap the servo off the first substantive frame (silent lead-ins |
796 | | * have degenerate budgets) */ |
797 | 0 | int nbnd_max = 0; |
798 | 0 | for (int k = 0; k < nsl; k++) |
799 | 0 | nbnd_max = FFMAX(nbnd_max, sl[k]->nbnd); |
800 | 0 | if (nbnd_max >= 8) { |
801 | 0 | s->nmr->lam_rc = av_clipf(lam, 1e-4f, 1e4f); |
802 | 0 | s->nmr->lam_slew = s->nmr->lam_rc; |
803 | 0 | } |
804 | 0 | } |
805 | |
|
806 | 0 | { /* PNS, per channel at the group's operating lambda */ |
807 | 0 | const float pns_lam = NMR_PNS_LAM; |
808 | 0 | int pns_total = 0; |
809 | 0 | for (int k = 0; k < nsl; k++) { |
810 | 0 | NMRSlot *t = sl[k]; |
811 | 0 | const float (*ndk)[NMR_NCAND] = (const float (*)[NMR_NCAND])s->nmr->nd[t->si]; |
812 | 0 | const int (*nbk)[NMR_NCAND] = (const int (*)[NMR_NCAND])s->nmr->nb[t->si]; |
813 | 0 | int pns_count = 0; |
814 | | /* band 0 (lowest freq) is kept as the global-gain / sf-chain anchor */ |
815 | 0 | for (int b = 1; b < t->nbnd; b++) { |
816 | 0 | int bi = t->bidx[b]; |
817 | 0 | float spread = t->pspread[bi]; |
818 | 0 | float nmr_pns, cost_keep, cost_pns, frac; |
819 | 0 | if (!t->sce->can_pns[bi]) |
820 | 0 | continue; |
821 | | |
822 | 0 | int was = s->nmr->pns_prev[t->cur_ch & 15][bi]; |
823 | 0 | float bias = was ? NMR_PNS_STAY : NMR_PNS_ENTER; |
824 | 0 | int want = 0, force_exit = 0; |
825 | | |
826 | | /* (can_pns was already checked above; gates below fill `want`) */ |
827 | 0 | if (t->pener[bi] > NMR_PNS_MAX_ET * t->thr_real[bi]) { |
828 | 0 | force_exit = 1; /* loud-band guard */ |
829 | 0 | } else if (lam > pns_lam) { |
830 | | /* Spectral-hole fill: a noise-like band left mostly empty */ |
831 | 0 | frac = ndk[b][t->chosen[b]] * t->thr[bi] / FFMAX(t->pener[bi], 1e-9f); |
832 | 0 | if (spread > NMR_PNS_HOLE_SPREAD && |
833 | 0 | frac > NMR_PNS_HOLE_FRAC * (was ? 0.7f : 1.0f)) { |
834 | 0 | want = 1; |
835 | 0 | } else if (ndk[b][t->chosen[b]] * t->thr[bi] > |
836 | 0 | NMR_PNS_NDGATE * t->thr_real[bi] * (was ? 0.5f : 1.0f)) { |
837 | | /* replace only a band coded audibly badly; cost of |
838 | | * energy-matched noise = its non-noise-like fraction */ |
839 | 0 | nmr_pns = FFMAX(0.0f, t->pener[bi] * (1.0f - spread*spread)) |
840 | 0 | / FFMAX(t->thr[bi], 1e-9f); |
841 | 0 | cost_keep = ndk[b][t->chosen[b]] + lam * nbk[b][t->chosen[b]]; |
842 | 0 | cost_pns = nmr_pns + lam * NMR_PNS_BITS; |
843 | 0 | want = cost_pns < cost_keep * bias; |
844 | 0 | } |
845 | 0 | } |
846 | 0 | { /* debounce; near-mask deletion candidates skip entry |
847 | | * (noise beats the ~silent rendition they'd get) */ |
848 | 0 | uint8_t *ron = &s->nmr->pns_run_on [t->cur_ch & 15][bi]; |
849 | 0 | uint8_t *roff = &s->nmr->pns_run_off[t->cur_ch & 15][bi]; |
850 | 0 | int near = t->pener[bi] < 2.0f * t->thr_real[bi]; |
851 | 0 | if (want) { if (*ron < 255) (*ron)++; *roff = 0; } |
852 | 0 | else { if (*roff < 255) (*roff)++; *ron = 0; } |
853 | 0 | if (force_exit) |
854 | 0 | want = 0; |
855 | 0 | else if (!was) |
856 | 0 | want = near ? want : *ron >= NMR_PNS_ON; |
857 | 0 | else if (near) |
858 | 0 | want = 1; /* physics-hysteresis: noise until audible */ |
859 | 0 | else |
860 | 0 | want = !(*roff >= NMR_PNS_OFF); |
861 | 0 | } |
862 | 0 | if (want) { |
863 | 0 | t->is_pns[b] = 1; |
864 | 0 | pns_count++; |
865 | 0 | } |
866 | 0 | } |
867 | 0 | if (pns_count) { |
868 | 0 | t->nact = 0; |
869 | 0 | for (int b = 0; b < t->nbnd; b++) |
870 | 0 | if (!t->is_pns[b]) |
871 | 0 | t->act[t->nact++] = b; |
872 | 0 | } |
873 | 0 | pns_total += pns_count; |
874 | 0 | } |
875 | 0 | if (pns_total) { |
876 | | /* re-solve over the survivors: at fixed lambda the allocation is |
877 | | * the same except for the repaired sf-delta chain; in bisection |
878 | | * mode re-spend the freed budget */ |
879 | 0 | if (rc_global) |
880 | 0 | nmr_eval_slots(s, sl, nsl, NMR_STEP, lam); |
881 | 0 | else |
882 | 0 | nmr_solve_slots(s, sl, nsl, NMR_STEP, destbits - pns_total * NMR_PNS_BITS, |
883 | 0 | 1e-9f, 1e4f, NMR_ITERS); |
884 | 0 | } |
885 | 0 | } |
886 | |
|
887 | 0 | for (int k = 0; k < nsl; k++) { |
888 | 0 | NMRSlot *t = sl[k]; |
889 | 0 | uint8_t *pp = s->nmr->pns_prev[t->cur_ch & 15]; |
890 | 0 | uint8_t now[128] = {0}; |
891 | 0 | for (int b = 0; b < t->nbnd; b++) |
892 | 0 | if (t->is_pns[b]) |
893 | 0 | now[t->bidx[b]] = 1; |
894 | 0 | memcpy(pp, now, 128); |
895 | 0 | } |
896 | 0 | for (int k = 0; k < nsl; k++) |
897 | 0 | nmr_commit_channel(s, sl[k]); |
898 | |
|
899 | 0 | } |
900 | | |
901 | | static void search_for_quantizers_nmr(AVCodecContext *avctx, |
902 | | AACEncContext *s, |
903 | | SingleChannelElement *sce, |
904 | | const float lambda) |
905 | 0 | { |
906 | 0 | AACNMRCurves *n = s->nmr; |
907 | | /* Global-lambda RC: one solve per frame at a servoed centre lambda; the reservoir |
908 | | * holds the long-run mean rate. Bypassed for VBR (-q:a) and the bootstrap frame. */ |
909 | 0 | int rc_eligible = !(avctx->flags & AV_CODEC_FLAG_QSCALE) && avctx->bit_rate > 0 && |
910 | 0 | avctx->bit_rate_tolerance != 0; |
911 | | /* Signed reservoir; soft steering (bounded repay + rc_off), hard cap = |
912 | | * legality only. */ |
913 | 0 | int rc_rate_frame = avctx->bit_rate * 1024.0 / avctx->sample_rate; |
914 | 0 | int rc_bmax = FFMIN(FFMAX(6144 * s->channels - rc_rate_frame, 256), NMR_CBR_BUF * s->channels); |
915 | |
|
916 | 0 | int rc_global, defer; |
917 | 0 | NMRSlot *t; |
918 | |
|
919 | 0 | s->nmr->counted[s->cur_channel] = 0; |
920 | |
|
921 | 0 | if (rc_eligible && !n->rc_fill_seeded) { |
922 | | /* the decoder bit reservoir starts FULL: seed it so the head may frontload */ |
923 | 0 | n->rc_fill = rc_bmax; |
924 | 0 | n->rc_fill_seeded = 1; |
925 | 0 | } |
926 | 0 | if (rc_eligible && avctx->frame_num != n->rc_frame_num) { |
927 | 0 | if (n->rc_frame_num > 0 && n->lam_rc > 0.0f) |
928 | 0 | n->rc_fill = av_clip(n->rc_fill + rc_rate_frame - s->last_frame_pb_count, |
929 | 0 | -rc_bmax, rc_bmax); |
930 | 0 | n->rc_frame_num = avctx->frame_num; |
931 | 0 | n->pending = 0; /* a deferred first channel never crosses a frame */ |
932 | | /* latch the RC mode per frame: a mid-frame bootstrap must not flip |
933 | | * the CPE defer logic between channels */ |
934 | 0 | n->rc_gl = rc_eligible && n->lam_rc > 0.0f; |
935 | | |
936 | | /* Transient burst run state: set at run start and held across the run so |
937 | | * coding stays uniform; repaid from the reservoir's steady stretches. */ |
938 | 0 | int is_short = sce->ics.window_sequence[0] == EIGHT_SHORT_SEQUENCE; |
939 | 0 | if (is_short) { |
940 | 0 | if (!n->prev_was_short) { /* run start */ |
941 | 0 | if (n->frames_since_short >= NMR_BURST_GAP) { |
942 | 0 | n->run_burst = NMR_BURST_GAIN; |
943 | 0 | } else { |
944 | | /* dense-beat boost, scaled by measured short-frame starvation */ |
945 | 0 | float imb = 0.0f; |
946 | 0 | if (n->lam_long_ema > 0.0f && n->lam_short_ema > 0.0f) |
947 | 0 | imb = av_clipf(n->lam_short_ema / n->lam_long_ema - 1.0f, |
948 | 0 | 0.0f, 1.0f); |
949 | 0 | n->run_burst = 1.0f + (NMR_SHORT_BOOST - 1.0f) * imb * |
950 | 0 | n->frames_since_short / (float)NMR_BURST_GAP; |
951 | 0 | } |
952 | 0 | } |
953 | 0 | n->frames_since_short = 0; |
954 | 0 | } else { |
955 | | /* the frame closing a run (the STOP) absorbs the corridor recoil |
956 | | * of the boosted shorts; give it half the run's factor so the |
957 | | * repayment spreads into the steady stretch instead */ |
958 | 0 | n->run_burst = n->prev_was_short ? sqrtf(n->run_burst) : 1.0f; |
959 | 0 | n->frames_since_short++; |
960 | 0 | } |
961 | 0 | n->prev_was_short = is_short; |
962 | 0 | } |
963 | 0 | rc_global = rc_eligible && n->rc_gl; |
964 | | |
965 | | /* CPE budget pool: under global-lambda RC, defer the pair's first channel |
966 | | * and solve both against one pooled budget when the second one arrives. */ |
967 | 0 | defer = n->pair && rc_global; |
968 | |
|
969 | 0 | t = &n->slot[(defer && n->pending) ? 1 : 0]; |
970 | 0 | t->si = (defer && n->pending) ? 1 : 0; |
971 | |
|
972 | 0 | if (!nmr_setup_channel(avctx, s, sce, t)) { |
973 | 0 | nmr_bail_channel(sce); |
974 | 0 | t->nbnd = t->nact = 0; |
975 | 0 | } |
976 | |
|
977 | 0 | if (defer && !n->pending) { |
978 | 0 | n->pending = 1; /* wait for the partner channel */ |
979 | 0 | return; |
980 | 0 | } |
981 | | |
982 | 0 | { |
983 | 0 | NMRSlot *sl[2]; |
984 | 0 | int nsl = 0, chans = 1; |
985 | 0 | if (defer) { |
986 | 0 | n->pending = 0; |
987 | 0 | chans = 2; |
988 | 0 | if (n->slot[0].nact) |
989 | 0 | sl[nsl++] = &n->slot[0]; |
990 | 0 | if (n->slot[1].nact) |
991 | 0 | sl[nsl++] = &n->slot[1]; |
992 | 0 | } else if (t->nact) { |
993 | 0 | sl[nsl++] = t; |
994 | 0 | } |
995 | 0 | if (!nsl) |
996 | 0 | return; /* nothing codeable in the group */ |
997 | 0 | nmr_solve_group(avctx, s, lambda, sl, nsl, chans, |
998 | 0 | rc_eligible, rc_global, rc_rate_frame, rc_bmax); |
999 | 0 | } |
1000 | 0 | } |
1001 | | |
1002 | | #endif /* AVCODEC_AACCODER_NMR_H */ |