Coverage Report

Created: 2026-09-14 07:05

next uncovered line (L), next uncovered region (R), next uncovered branch (B)
/src/libmpeg2/common/impeg2_idct.c
Line
Count
Source
1
/******************************************************************************
2
 *
3
 * Copyright (C) 2015 The Android Open Source Project
4
 *
5
 * Licensed under the Apache License, Version 2.0 (the "License");
6
 * you may not use this file except in compliance with the License.
7
 * You may obtain a copy of the License at:
8
 *
9
 * http://www.apache.org/licenses/LICENSE-2.0
10
 *
11
 * Unless required by applicable law or agreed to in writing, software
12
 * distributed under the License is distributed on an "AS IS" BASIS,
13
 * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
14
 * See the License for the specific language governing permissions and
15
 * limitations under the License.
16
 *
17
 *****************************************************************************
18
 * Originally developed and contributed by Ittiam Systems Pvt. Ltd, Bangalore
19
*/
20
/*****************************************************************************/
21
/*                                                                           */
22
/*  File Name         : impeg2_idct.c                                        */
23
/*                                                                           */
24
/*  Description       : Contains 2d idct and invese quantization functions   */
25
/*                                                                           */
26
/*  List of Functions : impeg2_idct_recon_dc()                               */
27
/*                      impeg2_idct_recon_dc_mismatch()                      */
28
/*                      impeg2_idct_recon()                                  */
29
/*                                                                           */
30
/*  Issues / Problems : None                                                 */
31
/*                                                                           */
32
/*  Revision History  :                                                      */
33
/*                                                                           */
34
/*         DD MM YYYY   Author(s)       Changes                              */
35
/*         10 09 2005   Hairsh M        First Version                        */
36
/*                                                                           */
37
/*****************************************************************************/
38
/*
39
  IEEE - 1180 results for this IDCT
40
  L                           256         256         5           5           300         300         384         384         Thresholds
41
  H                           255         255         5           5           300         300         383         383
42
  sign                        1           -1          1           -1          1           -1          1           -1
43
  Peak Error                  1           1           1           1           1           1           1           1           1
44
  Peak Mean Square Error      0.0191      0.0188      0.0108      0.0111      0.0176      0.0188      0.0165      0.0177      0.06
45
  Overall Mean Square Error   0.01566406  0.01597656  0.0091875   0.00908906  0.01499063  0.01533281  0.01432344  0.01412344  0.02
46
  Peak Mean Error             0.0027      0.0026      0.0028      0.002       0.0017      0.0033      0.0031      0.0025      0.015
47
  Overall Mean Error          0.00002656  -0.00031406 0.00016875  0.00005469  -0.00003125 0.00011406  0.00009219  0.00004219  0.0015
48
  */
49
#include <stdio.h>
50
#include <string.h>
51
52
#include "iv_datatypedef.h"
53
#include "iv.h"
54
#include "impeg2_defs.h"
55
#include "impeg2_platform_macros.h"
56
57
#include "impeg2_macros.h"
58
#include "impeg2_globals.h"
59
#include "impeg2_idct.h"
60
61
62
void impeg2_idct_recon_dc(WORD16 *pi2_src,
63
                            WORD16 *pi2_tmp,
64
                            UWORD8 *pu1_pred,
65
                            UWORD8 *pu1_dst,
66
                            WORD32 i4_src_strd,
67
                            WORD32 i4_pred_strd,
68
                            WORD32 i4_dst_strd,
69
                            WORD32 i4_zero_cols,
70
                            WORD32 i4_zero_rows)
71
796k
{
72
796k
    WORD32 i4_val, i, j;
73
74
796k
    UNUSED(pi2_tmp);
75
796k
    UNUSED(i4_src_strd);
76
796k
    UNUSED(i4_zero_cols);
77
796k
    UNUSED(i4_zero_rows);
78
79
796k
    i4_val = pi2_src[0] * gai2_impeg2_idct_q15[0];
80
796k
    i4_val = ((i4_val + IDCT_STG1_ROUND) >> IDCT_STG1_SHIFT);
81
796k
    i4_val = i4_val * gai2_impeg2_idct_q11[0];
82
796k
    i4_val = ((i4_val + IDCT_STG2_ROUND) >> IDCT_STG2_SHIFT);
83
84
7.02M
    for(i = 0; i < TRANS_SIZE_8; i++)
85
6.22M
    {
86
55.5M
        for(j = 0; j < TRANS_SIZE_8; j++)
87
49.2M
        {
88
49.2M
            pu1_dst[j] = CLIP_U8(i4_val + pu1_pred[j]);
89
49.2M
        }
90
6.22M
        pu1_dst  += i4_dst_strd;
91
6.22M
        pu1_pred += i4_pred_strd;
92
6.22M
    }
93
796k
}
94
void impeg2_idct_recon_dc_mismatch(WORD16 *pi2_src,
95
                            WORD16 *pi2_tmp,
96
                            UWORD8 *pu1_pred,
97
                            UWORD8 *pu1_dst,
98
                            WORD32 i4_src_strd,
99
                            WORD32 i4_pred_strd,
100
                            WORD32 i4_dst_strd,
101
                            WORD32 i4_zero_cols,
102
                            WORD32 i4_zero_rows)
103
104
127k
{
105
127k
    WORD32 i4_val, i, j;
106
127k
    WORD32 i4_count = 0;
107
127k
    WORD32 i4_sum;
108
109
127k
    UNUSED(pi2_tmp);
110
127k
    UNUSED(i4_src_strd);
111
127k
    UNUSED(i4_zero_cols);
112
127k
    UNUSED(i4_zero_rows);
113
114
127k
    i4_val = pi2_src[0] * gai2_impeg2_idct_q15[0];
115
127k
    i4_val = ((i4_val + IDCT_STG1_ROUND) >> IDCT_STG1_SHIFT);
116
117
127k
    i4_val *= gai2_impeg2_idct_q11[0];
118
1.12M
    for(i = 0; i < TRANS_SIZE_8; i++)
119
998k
    {
120
8.93M
        for (j = 0; j < TRANS_SIZE_8; j++)
121
7.93M
        {
122
7.93M
            i4_sum = i4_val;
123
7.93M
            i4_sum += gai2_impeg2_mismatch_stg2_additive[i4_count];
124
7.93M
            i4_sum = ((i4_sum + IDCT_STG2_ROUND) >> IDCT_STG2_SHIFT);
125
7.93M
            i4_sum += pu1_pred[j];
126
7.93M
            pu1_dst[j] = CLIP_U8(i4_sum);
127
7.93M
            i4_count++;
128
7.93M
        }
129
130
998k
        pu1_dst  += i4_dst_strd;
131
998k
        pu1_pred += i4_pred_strd;
132
998k
    }
133
134
127k
}
135
/**
136
 *******************************************************************************
137
 *
138
 * @brief
139
 *  This function performs Inverse transform  and reconstruction for 8x8
140
 * input block
141
 *
142
 * @par Description:
143
 *  Performs inverse transform and adds the prediction  data and clips output
144
 * to 8 bit
145
 *
146
 * @param[in] pi2_src
147
 *  Input 8x8 coefficients
148
 *
149
 * @param[in] pi2_tmp
150
 *  Temporary 8x8 buffer for storing inverse
151
 *
152
 *  transform
153
 *  1st stage output
154
 *
155
 * @param[in] pu1_pred
156
 *  Prediction 8x8 block
157
 *
158
 * @param[out] pu1_dst
159
 *  Output 8x8 block
160
 *
161
 * @param[in] src_strd
162
 *  Input stride
163
 *
164
 * @param[in] pred_strd
165
 *  Prediction stride
166
 *
167
 * @param[in] dst_strd
168
 *  Output Stride
169
 *
170
 * @param[in] shift
171
 *  Output shift
172
 *
173
 * @param[in] zero_cols
174
 *  Zero columns in pi2_src
175
 *
176
 * @returns  Void
177
 *
178
 * @remarks
179
 *  None
180
 *
181
 *******************************************************************************
182
 */
183
184
void impeg2_idct_recon(WORD16 *pi2_src,
185
                        WORD16 *pi2_tmp,
186
                        UWORD8 *pu1_pred,
187
                        UWORD8 *pu1_dst,
188
                        WORD32 i4_src_strd,
189
                        WORD32 i4_pred_strd,
190
                        WORD32 i4_dst_strd,
191
                        WORD32 i4_zero_cols,
192
                        WORD32 i4_zero_rows)
193
4.89M
{
194
4.89M
    WORD32 j, k;
195
4.89M
    WORD32 ai4_e[4], ai4_o[4];
196
4.89M
    WORD32 ai4_ee[2], ai4_eo[2];
197
4.89M
    WORD32 i4_add;
198
4.89M
    WORD32 i4_shift;
199
4.89M
    WORD16 *pi2_tmp_orig;
200
4.89M
    WORD32 i4_trans_size;
201
4.89M
    WORD32 i4_zero_rows_2nd_stage = i4_zero_cols;
202
4.89M
    WORD32 i4_row_limit_2nd_stage;
203
204
4.89M
    i4_trans_size = TRANS_SIZE_8;
205
206
4.89M
    pi2_tmp_orig = pi2_tmp;
207
208
4.89M
    if((i4_zero_cols & 0xF0) == 0xF0)
209
4.49M
        i4_row_limit_2nd_stage = 4;
210
401k
    else
211
401k
        i4_row_limit_2nd_stage = TRANS_SIZE_8;
212
213
214
4.89M
    if((i4_zero_rows & 0xF0) == 0xF0) /* First 4 rows of input are non-zero */
215
4.35M
    {
216
        /************************************************************************************************/
217
        /**********************************START - IT_RECON_8x8******************************************/
218
        /************************************************************************************************/
219
220
        /* Inverse Transform 1st stage */
221
4.35M
        i4_shift = IDCT_STG1_SHIFT;
222
4.35M
        i4_add = 1 << (i4_shift - 1);
223
224
21.8M
        for(j = 0; j < i4_row_limit_2nd_stage; j++)
225
17.4M
        {
226
            /* Checking for Zero Cols */
227
17.4M
            if((i4_zero_cols & 1) == 1)
228
11.4M
            {
229
11.4M
                memset(pi2_tmp, 0, i4_trans_size * sizeof(WORD16));
230
11.4M
            }
231
6.03M
            else
232
6.03M
            {
233
                /* Utilizing symmetry properties to the maximum to minimize the number of multiplications */
234
30.2M
                for(k = 0; k < 4; k++)
235
24.2M
                {
236
24.2M
                    ai4_o[k] = gai2_impeg2_idct_q15[1 * 8 + k] * pi2_src[i4_src_strd]
237
24.2M
                                    + gai2_impeg2_idct_q15[3 * 8 + k]
238
24.2M
                                                    * pi2_src[3 * i4_src_strd];
239
24.2M
                }
240
6.03M
                ai4_eo[0] = gai2_impeg2_idct_q15[2 * 8 + 0] * pi2_src[2 * i4_src_strd];
241
6.03M
                ai4_eo[1] = gai2_impeg2_idct_q15[2 * 8 + 1] * pi2_src[2 * i4_src_strd];
242
6.03M
                ai4_ee[0] = gai2_impeg2_idct_q15[0 * 8 + 0] * pi2_src[0];
243
6.03M
                ai4_ee[1] = gai2_impeg2_idct_q15[0 * 8 + 1] * pi2_src[0];
244
245
                /* Combining e and o terms at each hierarchy levels to calculate the final spatial domain vector */
246
6.03M
                ai4_e[0] = ai4_ee[0] + ai4_eo[0];
247
6.03M
                ai4_e[3] = ai4_ee[0] - ai4_eo[0];
248
6.03M
                ai4_e[1] = ai4_ee[1] + ai4_eo[1];
249
6.03M
                ai4_e[2] = ai4_ee[1] - ai4_eo[1];
250
30.3M
                for(k = 0; k < 4; k++)
251
24.2M
                {
252
24.2M
                    pi2_tmp[k] =
253
24.2M
                                    CLIP_S16(((ai4_e[k] + ai4_o[k] + i4_add) >> i4_shift));
254
24.2M
                    pi2_tmp[k + 4] =
255
24.2M
                                    CLIP_S16(((ai4_e[3 - k] - ai4_o[3 - k] + i4_add) >> i4_shift));
256
24.2M
                }
257
6.03M
            }
258
17.4M
            pi2_src++;
259
17.4M
            pi2_tmp += i4_trans_size;
260
17.4M
            i4_zero_cols = i4_zero_cols >> 1;
261
17.4M
        }
262
263
4.35M
        pi2_tmp = pi2_tmp_orig;
264
265
        /* Inverse Transform 2nd stage */
266
4.35M
        i4_shift = IDCT_STG2_SHIFT;
267
4.35M
        i4_add = 1 << (i4_shift - 1);
268
4.35M
        if((i4_zero_rows_2nd_stage & 0xF0) == 0xF0) /* First 4 rows of output of 1st stage are non-zero */
269
4.33M
        {
270
38.1M
            for(j = 0; j < i4_trans_size; j++)
271
33.7M
            {
272
                /* Utilizing symmetry properties to the maximum to minimize the number of multiplications */
273
168M
                for(k = 0; k < 4; k++)
274
134M
                {
275
134M
                    ai4_o[k] = gai2_impeg2_idct_q11[1 * 8 + k] * pi2_tmp[i4_trans_size]
276
134M
                                    + gai2_impeg2_idct_q11[3 * 8 + k] * pi2_tmp[3 * i4_trans_size];
277
134M
                }
278
33.7M
                ai4_eo[0] = gai2_impeg2_idct_q11[2 * 8 + 0] * pi2_tmp[2 * i4_trans_size];
279
33.7M
                ai4_eo[1] = gai2_impeg2_idct_q11[2 * 8 + 1] * pi2_tmp[2 * i4_trans_size];
280
33.7M
                ai4_ee[0] = gai2_impeg2_idct_q11[0 * 8 + 0] * pi2_tmp[0];
281
33.7M
                ai4_ee[1] = gai2_impeg2_idct_q11[0 * 8 + 1] * pi2_tmp[0];
282
283
                /* Combining e and o terms at each hierarchy levels to calculate the final spatial domain vector */
284
33.7M
                ai4_e[0] = ai4_ee[0] + ai4_eo[0];
285
33.7M
                ai4_e[3] = ai4_ee[0] - ai4_eo[0];
286
33.7M
                ai4_e[1] = ai4_ee[1] + ai4_eo[1];
287
33.7M
                ai4_e[2] = ai4_ee[1] - ai4_eo[1];
288
165M
                for(k = 0; k < 4; k++)
289
131M
                {
290
131M
                    WORD32 itrans_out;
291
131M
                    itrans_out =
292
131M
                                    CLIP_S16(((ai4_e[k] + ai4_o[k] + i4_add) >> i4_shift));
293
131M
                    pu1_dst[k] = CLIP_U8((itrans_out + pu1_pred[k]));
294
131M
                    itrans_out =
295
131M
                                    CLIP_S16(((ai4_e[3 - k] - ai4_o[3 - k] + i4_add) >> i4_shift));
296
131M
                    pu1_dst[k + 4] = CLIP_U8((itrans_out + pu1_pred[k + 4]));
297
131M
                }
298
33.7M
                pi2_tmp++;
299
33.7M
                pu1_pred += i4_pred_strd;
300
33.7M
                pu1_dst += i4_dst_strd;
301
33.7M
            }
302
4.33M
        }
303
18.8k
        else /* All rows of output of 1st stage are non-zero */
304
18.8k
        {
305
203k
            for(j = 0; j < i4_trans_size; j++)
306
184k
            {
307
                /* Utilizing symmetry properties to the maximum to minimize the number of multiplications */
308
920k
                for(k = 0; k < 4; k++)
309
736k
                {
310
736k
                    ai4_o[k] = gai2_impeg2_idct_q11[1 * 8 + k] * pi2_tmp[i4_trans_size]
311
736k
                                    + gai2_impeg2_idct_q11[3 * 8 + k]
312
736k
                                                    * pi2_tmp[3 * i4_trans_size]
313
736k
                                    + gai2_impeg2_idct_q11[5 * 8 + k]
314
736k
                                                    * pi2_tmp[5 * i4_trans_size]
315
736k
                                    + gai2_impeg2_idct_q11[7 * 8 + k]
316
736k
                                                    * pi2_tmp[7 * i4_trans_size];
317
736k
                }
318
319
184k
                ai4_eo[0] = gai2_impeg2_idct_q11[2 * 8 + 0] * pi2_tmp[2 * i4_trans_size]
320
184k
                                + gai2_impeg2_idct_q11[6 * 8 + 0] * pi2_tmp[6 * i4_trans_size];
321
184k
                ai4_eo[1] = gai2_impeg2_idct_q11[2 * 8 + 1] * pi2_tmp[2 * i4_trans_size]
322
184k
                                + gai2_impeg2_idct_q11[6 * 8 + 1] * pi2_tmp[6 * i4_trans_size];
323
184k
                ai4_ee[0] = gai2_impeg2_idct_q11[0 * 8 + 0] * pi2_tmp[0]
324
184k
                                + gai2_impeg2_idct_q11[4 * 8 + 0] * pi2_tmp[4 * i4_trans_size];
325
184k
                ai4_ee[1] = gai2_impeg2_idct_q11[0 * 8 + 1] * pi2_tmp[0]
326
184k
                                + gai2_impeg2_idct_q11[4 * 8 + 1] * pi2_tmp[4 * i4_trans_size];
327
328
                /* Combining e and o terms at each hierarchy levels to calculate the final spatial domain vector */
329
184k
                ai4_e[0] = ai4_ee[0] + ai4_eo[0];
330
184k
                ai4_e[3] = ai4_ee[0] - ai4_eo[0];
331
184k
                ai4_e[1] = ai4_ee[1] + ai4_eo[1];
332
184k
                ai4_e[2] = ai4_ee[1] - ai4_eo[1];
333
920k
                for(k = 0; k < 4; k++)
334
736k
                {
335
736k
                    WORD32 itrans_out;
336
736k
                    itrans_out =
337
736k
                                    CLIP_S16(((ai4_e[k] + ai4_o[k] + i4_add) >> i4_shift));
338
736k
                    pu1_dst[k] = CLIP_U8((itrans_out + pu1_pred[k]));
339
736k
                    itrans_out =
340
736k
                                    CLIP_S16(((ai4_e[3 - k] - ai4_o[3 - k] + i4_add) >> i4_shift));
341
736k
                    pu1_dst[k + 4] = CLIP_U8((itrans_out + pu1_pred[k + 4]));
342
736k
                }
343
184k
                pi2_tmp++;
344
184k
                pu1_pred += i4_pred_strd;
345
184k
                pu1_dst += i4_dst_strd;
346
184k
            }
347
18.8k
        }
348
        /************************************************************************************************/
349
        /************************************END - IT_RECON_8x8******************************************/
350
        /************************************************************************************************/
351
4.35M
    }
352
541k
    else /* All rows of input are non-zero */
353
541k
    {
354
        /************************************************************************************************/
355
        /**********************************START - IT_RECON_8x8******************************************/
356
        /************************************************************************************************/
357
358
        /* Inverse Transform 1st stage */
359
541k
        i4_shift = IDCT_STG1_SHIFT;
360
541k
        i4_add = 1 << (i4_shift - 1);
361
362
4.20M
        for(j = 0; j < i4_row_limit_2nd_stage; j++)
363
3.66M
        {
364
            /* Checking for Zero Cols */
365
3.66M
            if((i4_zero_cols & 1) == 1)
366
1.83M
            {
367
1.83M
                memset(pi2_tmp, 0, i4_trans_size * sizeof(WORD16));
368
1.83M
            }
369
1.82M
            else
370
1.82M
            {
371
                /* Utilizing symmetry properties to the maximum to minimize the number of multiplications */
372
9.15M
                for(k = 0; k < 4; k++)
373
7.32M
                {
374
7.32M
                    ai4_o[k] = gai2_impeg2_idct_q15[1 * 8 + k] * pi2_src[i4_src_strd]
375
7.32M
                                    + gai2_impeg2_idct_q15[3 * 8 + k]
376
7.32M
                                                    * pi2_src[3 * i4_src_strd]
377
7.32M
                                    + gai2_impeg2_idct_q15[5 * 8 + k]
378
7.32M
                                                    * pi2_src[5 * i4_src_strd]
379
7.32M
                                    + gai2_impeg2_idct_q15[7 * 8 + k]
380
7.32M
                                                    * pi2_src[7 * i4_src_strd];
381
7.32M
                }
382
383
1.82M
                ai4_eo[0] = gai2_impeg2_idct_q15[2 * 8 + 0] * pi2_src[2 * i4_src_strd]
384
1.82M
                                + gai2_impeg2_idct_q15[6 * 8 + 0] * pi2_src[6 * i4_src_strd];
385
1.82M
                ai4_eo[1] = gai2_impeg2_idct_q15[2 * 8 + 1] * pi2_src[2 * i4_src_strd]
386
1.82M
                                + gai2_impeg2_idct_q15[6 * 8 + 1] * pi2_src[6 * i4_src_strd];
387
1.82M
                ai4_ee[0] = gai2_impeg2_idct_q15[0 * 8 + 0] * pi2_src[0]
388
1.82M
                                + gai2_impeg2_idct_q15[4 * 8 + 0] * pi2_src[4 * i4_src_strd];
389
1.82M
                ai4_ee[1] = gai2_impeg2_idct_q15[0 * 8 + 1] * pi2_src[0]
390
1.82M
                                + gai2_impeg2_idct_q15[4 * 8 + 1] * pi2_src[4 * i4_src_strd];
391
392
                /* Combining e and o terms at each hierarchy levels to calculate the final spatial domain vector */
393
1.82M
                ai4_e[0] = ai4_ee[0] + ai4_eo[0];
394
1.82M
                ai4_e[3] = ai4_ee[0] - ai4_eo[0];
395
1.82M
                ai4_e[1] = ai4_ee[1] + ai4_eo[1];
396
1.82M
                ai4_e[2] = ai4_ee[1] - ai4_eo[1];
397
9.14M
                for(k = 0; k < 4; k++)
398
7.31M
                {
399
7.31M
                    pi2_tmp[k] =
400
7.31M
                                    CLIP_S16(((ai4_e[k] + ai4_o[k] + i4_add) >> i4_shift));
401
7.31M
                    pi2_tmp[k + 4] =
402
7.31M
                                    CLIP_S16(((ai4_e[3 - k] - ai4_o[3 - k] + i4_add) >> i4_shift));
403
7.31M
                }
404
1.82M
            }
405
3.66M
            pi2_src++;
406
3.66M
            pi2_tmp += i4_trans_size;
407
3.66M
            i4_zero_cols = i4_zero_cols >> 1;
408
3.66M
        }
409
410
541k
        pi2_tmp = pi2_tmp_orig;
411
412
        /* Inverse Transform 2nd stage */
413
541k
        i4_shift = IDCT_STG2_SHIFT;
414
541k
        i4_add = 1 << (i4_shift - 1);
415
541k
        if((i4_zero_rows_2nd_stage & 0xF0) == 0xF0) /* First 4 rows of output of 1st stage are non-zero */
416
163k
        {
417
1.46M
            for(j = 0; j < i4_trans_size; j++)
418
1.30M
            {
419
                /* Utilizing symmetry properties to the maximum to minimize the number of multiplications */
420
6.51M
                for(k = 0; k < 4; k++)
421
5.20M
                {
422
5.20M
                    ai4_o[k] = gai2_impeg2_idct_q11[1 * 8 + k] * pi2_tmp[i4_trans_size]
423
5.20M
                                    + gai2_impeg2_idct_q11[3 * 8 + k] * pi2_tmp[3 * i4_trans_size];
424
5.20M
                }
425
1.30M
                ai4_eo[0] = gai2_impeg2_idct_q11[2 * 8 + 0] * pi2_tmp[2 * i4_trans_size];
426
1.30M
                ai4_eo[1] = gai2_impeg2_idct_q11[2 * 8 + 1] * pi2_tmp[2 * i4_trans_size];
427
1.30M
                ai4_ee[0] = gai2_impeg2_idct_q11[0 * 8 + 0] * pi2_tmp[0];
428
1.30M
                ai4_ee[1] = gai2_impeg2_idct_q11[0 * 8 + 1] * pi2_tmp[0];
429
430
                /* Combining e and o terms at each hierarchy levels to calculate the final spatial domain vector */
431
1.30M
                ai4_e[0] = ai4_ee[0] + ai4_eo[0];
432
1.30M
                ai4_e[3] = ai4_ee[0] - ai4_eo[0];
433
1.30M
                ai4_e[1] = ai4_ee[1] + ai4_eo[1];
434
1.30M
                ai4_e[2] = ai4_ee[1] - ai4_eo[1];
435
6.50M
                for(k = 0; k < 4; k++)
436
5.20M
                {
437
5.20M
                    WORD32 itrans_out;
438
5.20M
                    itrans_out =
439
5.20M
                                    CLIP_S16(((ai4_e[k] + ai4_o[k] + i4_add) >> i4_shift));
440
5.20M
                    pu1_dst[k] = CLIP_U8((itrans_out + pu1_pred[k]));
441
5.20M
                    itrans_out =
442
5.20M
                                    CLIP_S16(((ai4_e[3 - k] - ai4_o[3 - k] + i4_add) >> i4_shift));
443
5.20M
                    pu1_dst[k + 4] = CLIP_U8((itrans_out + pu1_pred[k + 4]));
444
5.20M
                }
445
1.30M
                pi2_tmp++;
446
1.30M
                pu1_pred += i4_pred_strd;
447
1.30M
                pu1_dst += i4_dst_strd;
448
1.30M
            }
449
163k
        }
450
378k
        else /* All rows of output of 1st stage are non-zero */
451
378k
        {
452
3.35M
            for(j = 0; j < i4_trans_size; j++)
453
2.97M
            {
454
                /* Utilizing symmetry properties to the maximum to minimize the number of multiplications */
455
14.7M
                for(k = 0; k < 4; k++)
456
11.8M
                {
457
11.8M
                    ai4_o[k] = gai2_impeg2_idct_q11[1 * 8 + k] * pi2_tmp[i4_trans_size]
458
11.8M
                                    + gai2_impeg2_idct_q11[3 * 8 + k]
459
11.8M
                                                    * pi2_tmp[3 * i4_trans_size]
460
11.8M
                                    + gai2_impeg2_idct_q11[5 * 8 + k]
461
11.8M
                                                    * pi2_tmp[5 * i4_trans_size]
462
11.8M
                                    + gai2_impeg2_idct_q11[7 * 8 + k]
463
11.8M
                                                    * pi2_tmp[7 * i4_trans_size];
464
11.8M
                }
465
466
2.97M
                ai4_eo[0] = gai2_impeg2_idct_q11[2 * 8 + 0] * pi2_tmp[2 * i4_trans_size]
467
2.97M
                                + gai2_impeg2_idct_q11[6 * 8 + 0] * pi2_tmp[6 * i4_trans_size];
468
2.97M
                ai4_eo[1] = gai2_impeg2_idct_q11[2 * 8 + 1] * pi2_tmp[2 * i4_trans_size]
469
2.97M
                                + gai2_impeg2_idct_q11[6 * 8 + 1] * pi2_tmp[6 * i4_trans_size];
470
2.97M
                ai4_ee[0] = gai2_impeg2_idct_q11[0 * 8 + 0] * pi2_tmp[0]
471
2.97M
                                + gai2_impeg2_idct_q11[4 * 8 + 0] * pi2_tmp[4 * i4_trans_size];
472
2.97M
                ai4_ee[1] = gai2_impeg2_idct_q11[0 * 8 + 1] * pi2_tmp[0]
473
2.97M
                                + gai2_impeg2_idct_q11[4 * 8 + 1] * pi2_tmp[4 * i4_trans_size];
474
475
                /* Combining e and o terms at each hierarchy levels to calculate the final spatial domain vector */
476
2.97M
                ai4_e[0] = ai4_ee[0] + ai4_eo[0];
477
2.97M
                ai4_e[3] = ai4_ee[0] - ai4_eo[0];
478
2.97M
                ai4_e[1] = ai4_ee[1] + ai4_eo[1];
479
2.97M
                ai4_e[2] = ai4_ee[1] - ai4_eo[1];
480
14.6M
                for(k = 0; k < 4; k++)
481
11.7M
                {
482
11.7M
                    WORD32 itrans_out;
483
11.7M
                    itrans_out =
484
11.7M
                                    CLIP_S16(((ai4_e[k] + ai4_o[k] + i4_add) >> i4_shift));
485
11.7M
                    pu1_dst[k] = CLIP_U8((itrans_out + pu1_pred[k]));
486
11.7M
                    itrans_out =
487
11.7M
                                    CLIP_S16(((ai4_e[3 - k] - ai4_o[3 - k] + i4_add) >> i4_shift));
488
11.7M
                    pu1_dst[k + 4] = CLIP_U8((itrans_out + pu1_pred[k + 4]));
489
11.7M
                }
490
2.97M
                pi2_tmp++;
491
2.97M
                pu1_pred += i4_pred_strd;
492
2.97M
                pu1_dst += i4_dst_strd;
493
2.97M
            }
494
378k
        }
495
        /************************************************************************************************/
496
        /************************************END - IT_RECON_8x8******************************************/
497
        /************************************************************************************************/
498
541k
    }
499
4.89M
}
500