Coverage Report

Created: 2026-07-30 07:04

next uncovered line (L), next uncovered region (R), next uncovered branch (B)
/src/libmpeg2/common/impeg2_idct.c
Line
Count
Source
1
/******************************************************************************
2
 *
3
 * Copyright (C) 2015 The Android Open Source Project
4
 *
5
 * Licensed under the Apache License, Version 2.0 (the "License");
6
 * you may not use this file except in compliance with the License.
7
 * You may obtain a copy of the License at:
8
 *
9
 * http://www.apache.org/licenses/LICENSE-2.0
10
 *
11
 * Unless required by applicable law or agreed to in writing, software
12
 * distributed under the License is distributed on an "AS IS" BASIS,
13
 * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
14
 * See the License for the specific language governing permissions and
15
 * limitations under the License.
16
 *
17
 *****************************************************************************
18
 * Originally developed and contributed by Ittiam Systems Pvt. Ltd, Bangalore
19
*/
20
/*****************************************************************************/
21
/*                                                                           */
22
/*  File Name         : impeg2_idct.c                                        */
23
/*                                                                           */
24
/*  Description       : Contains 2d idct and invese quantization functions   */
25
/*                                                                           */
26
/*  List of Functions : impeg2_idct_recon_dc()                               */
27
/*                      impeg2_idct_recon_dc_mismatch()                      */
28
/*                      impeg2_idct_recon()                                  */
29
/*                                                                           */
30
/*  Issues / Problems : None                                                 */
31
/*                                                                           */
32
/*  Revision History  :                                                      */
33
/*                                                                           */
34
/*         DD MM YYYY   Author(s)       Changes                              */
35
/*         10 09 2005   Hairsh M        First Version                        */
36
/*                                                                           */
37
/*****************************************************************************/
38
/*
39
  IEEE - 1180 results for this IDCT
40
  L                           256         256         5           5           300         300         384         384         Thresholds
41
  H                           255         255         5           5           300         300         383         383
42
  sign                        1           -1          1           -1          1           -1          1           -1
43
  Peak Error                  1           1           1           1           1           1           1           1           1
44
  Peak Mean Square Error      0.0191      0.0188      0.0108      0.0111      0.0176      0.0188      0.0165      0.0177      0.06
45
  Overall Mean Square Error   0.01566406  0.01597656  0.0091875   0.00908906  0.01499063  0.01533281  0.01432344  0.01412344  0.02
46
  Peak Mean Error             0.0027      0.0026      0.0028      0.002       0.0017      0.0033      0.0031      0.0025      0.015
47
  Overall Mean Error          0.00002656  -0.00031406 0.00016875  0.00005469  -0.00003125 0.00011406  0.00009219  0.00004219  0.0015
48
  */
49
#include <stdio.h>
50
#include <string.h>
51
52
#include "iv_datatypedef.h"
53
#include "iv.h"
54
#include "impeg2_defs.h"
55
#include "impeg2_platform_macros.h"
56
57
#include "impeg2_macros.h"
58
#include "impeg2_globals.h"
59
#include "impeg2_idct.h"
60
61
62
void impeg2_idct_recon_dc(WORD16 *pi2_src,
63
                            WORD16 *pi2_tmp,
64
                            UWORD8 *pu1_pred,
65
                            UWORD8 *pu1_dst,
66
                            WORD32 i4_src_strd,
67
                            WORD32 i4_pred_strd,
68
                            WORD32 i4_dst_strd,
69
                            WORD32 i4_zero_cols,
70
                            WORD32 i4_zero_rows)
71
649k
{
72
649k
    WORD32 i4_val, i, j;
73
74
649k
    UNUSED(pi2_tmp);
75
649k
    UNUSED(i4_src_strd);
76
649k
    UNUSED(i4_zero_cols);
77
649k
    UNUSED(i4_zero_rows);
78
79
649k
    i4_val = pi2_src[0] * gai2_impeg2_idct_q15[0];
80
649k
    i4_val = ((i4_val + IDCT_STG1_ROUND) >> IDCT_STG1_SHIFT);
81
649k
    i4_val = i4_val * gai2_impeg2_idct_q11[0];
82
649k
    i4_val = ((i4_val + IDCT_STG2_ROUND) >> IDCT_STG2_SHIFT);
83
84
5.75M
    for(i = 0; i < TRANS_SIZE_8; i++)
85
5.10M
    {
86
45.6M
        for(j = 0; j < TRANS_SIZE_8; j++)
87
40.5M
        {
88
40.5M
            pu1_dst[j] = CLIP_U8(i4_val + pu1_pred[j]);
89
40.5M
        }
90
5.10M
        pu1_dst  += i4_dst_strd;
91
5.10M
        pu1_pred += i4_pred_strd;
92
5.10M
    }
93
649k
}
94
void impeg2_idct_recon_dc_mismatch(WORD16 *pi2_src,
95
                            WORD16 *pi2_tmp,
96
                            UWORD8 *pu1_pred,
97
                            UWORD8 *pu1_dst,
98
                            WORD32 i4_src_strd,
99
                            WORD32 i4_pred_strd,
100
                            WORD32 i4_dst_strd,
101
                            WORD32 i4_zero_cols,
102
                            WORD32 i4_zero_rows)
103
104
107k
{
105
107k
    WORD32 i4_val, i, j;
106
107k
    WORD32 i4_count = 0;
107
107k
    WORD32 i4_sum;
108
109
107k
    UNUSED(pi2_tmp);
110
107k
    UNUSED(i4_src_strd);
111
107k
    UNUSED(i4_zero_cols);
112
107k
    UNUSED(i4_zero_rows);
113
114
107k
    i4_val = pi2_src[0] * gai2_impeg2_idct_q15[0];
115
107k
    i4_val = ((i4_val + IDCT_STG1_ROUND) >> IDCT_STG1_SHIFT);
116
117
107k
    i4_val *= gai2_impeg2_idct_q11[0];
118
961k
    for(i = 0; i < TRANS_SIZE_8; i++)
119
853k
    {
120
7.64M
        for (j = 0; j < TRANS_SIZE_8; j++)
121
6.79M
        {
122
6.79M
            i4_sum = i4_val;
123
6.79M
            i4_sum += gai2_impeg2_mismatch_stg2_additive[i4_count];
124
6.79M
            i4_sum = ((i4_sum + IDCT_STG2_ROUND) >> IDCT_STG2_SHIFT);
125
6.79M
            i4_sum += pu1_pred[j];
126
6.79M
            pu1_dst[j] = CLIP_U8(i4_sum);
127
6.79M
            i4_count++;
128
6.79M
        }
129
130
853k
        pu1_dst  += i4_dst_strd;
131
853k
        pu1_pred += i4_pred_strd;
132
853k
    }
133
134
107k
}
135
/**
136
 *******************************************************************************
137
 *
138
 * @brief
139
 *  This function performs Inverse transform  and reconstruction for 8x8
140
 * input block
141
 *
142
 * @par Description:
143
 *  Performs inverse transform and adds the prediction  data and clips output
144
 * to 8 bit
145
 *
146
 * @param[in] pi2_src
147
 *  Input 8x8 coefficients
148
 *
149
 * @param[in] pi2_tmp
150
 *  Temporary 8x8 buffer for storing inverse
151
 *
152
 *  transform
153
 *  1st stage output
154
 *
155
 * @param[in] pu1_pred
156
 *  Prediction 8x8 block
157
 *
158
 * @param[out] pu1_dst
159
 *  Output 8x8 block
160
 *
161
 * @param[in] src_strd
162
 *  Input stride
163
 *
164
 * @param[in] pred_strd
165
 *  Prediction stride
166
 *
167
 * @param[in] dst_strd
168
 *  Output Stride
169
 *
170
 * @param[in] shift
171
 *  Output shift
172
 *
173
 * @param[in] zero_cols
174
 *  Zero columns in pi2_src
175
 *
176
 * @returns  Void
177
 *
178
 * @remarks
179
 *  None
180
 *
181
 *******************************************************************************
182
 */
183
184
void impeg2_idct_recon(WORD16 *pi2_src,
185
                        WORD16 *pi2_tmp,
186
                        UWORD8 *pu1_pred,
187
                        UWORD8 *pu1_dst,
188
                        WORD32 i4_src_strd,
189
                        WORD32 i4_pred_strd,
190
                        WORD32 i4_dst_strd,
191
                        WORD32 i4_zero_cols,
192
                        WORD32 i4_zero_rows)
193
5.55M
{
194
5.55M
    WORD32 j, k;
195
5.55M
    WORD32 ai4_e[4], ai4_o[4];
196
5.55M
    WORD32 ai4_ee[2], ai4_eo[2];
197
5.55M
    WORD32 i4_add;
198
5.55M
    WORD32 i4_shift;
199
5.55M
    WORD16 *pi2_tmp_orig;
200
5.55M
    WORD32 i4_trans_size;
201
5.55M
    WORD32 i4_zero_rows_2nd_stage = i4_zero_cols;
202
5.55M
    WORD32 i4_row_limit_2nd_stage;
203
204
5.55M
    i4_trans_size = TRANS_SIZE_8;
205
206
5.55M
    pi2_tmp_orig = pi2_tmp;
207
208
5.55M
    if((i4_zero_cols & 0xF0) == 0xF0)
209
4.91M
        i4_row_limit_2nd_stage = 4;
210
642k
    else
211
642k
        i4_row_limit_2nd_stage = TRANS_SIZE_8;
212
213
214
5.55M
    if((i4_zero_rows & 0xF0) == 0xF0) /* First 4 rows of input are non-zero */
215
4.76M
    {
216
        /************************************************************************************************/
217
        /**********************************START - IT_RECON_8x8******************************************/
218
        /************************************************************************************************/
219
220
        /* Inverse Transform 1st stage */
221
4.76M
        i4_shift = IDCT_STG1_SHIFT;
222
4.76M
        i4_add = 1 << (i4_shift - 1);
223
224
23.9M
        for(j = 0; j < i4_row_limit_2nd_stage; j++)
225
19.1M
        {
226
            /* Checking for Zero Cols */
227
19.1M
            if((i4_zero_cols & 1) == 1)
228
12.5M
            {
229
12.5M
                memset(pi2_tmp, 0, i4_trans_size * sizeof(WORD16));
230
12.5M
            }
231
6.59M
            else
232
6.59M
            {
233
                /* Utilizing symmetry properties to the maximum to minimize the number of multiplications */
234
33.1M
                for(k = 0; k < 4; k++)
235
26.5M
                {
236
26.5M
                    ai4_o[k] = gai2_impeg2_idct_q15[1 * 8 + k] * pi2_src[i4_src_strd]
237
26.5M
                                    + gai2_impeg2_idct_q15[3 * 8 + k]
238
26.5M
                                                    * pi2_src[3 * i4_src_strd];
239
26.5M
                }
240
6.59M
                ai4_eo[0] = gai2_impeg2_idct_q15[2 * 8 + 0] * pi2_src[2 * i4_src_strd];
241
6.59M
                ai4_eo[1] = gai2_impeg2_idct_q15[2 * 8 + 1] * pi2_src[2 * i4_src_strd];
242
6.59M
                ai4_ee[0] = gai2_impeg2_idct_q15[0 * 8 + 0] * pi2_src[0];
243
6.59M
                ai4_ee[1] = gai2_impeg2_idct_q15[0 * 8 + 1] * pi2_src[0];
244
245
                /* Combining e and o terms at each hierarchy levels to calculate the final spatial domain vector */
246
6.59M
                ai4_e[0] = ai4_ee[0] + ai4_eo[0];
247
6.59M
                ai4_e[3] = ai4_ee[0] - ai4_eo[0];
248
6.59M
                ai4_e[1] = ai4_ee[1] + ai4_eo[1];
249
6.59M
                ai4_e[2] = ai4_ee[1] - ai4_eo[1];
250
33.1M
                for(k = 0; k < 4; k++)
251
26.5M
                {
252
26.5M
                    pi2_tmp[k] =
253
26.5M
                                    CLIP_S16(((ai4_e[k] + ai4_o[k] + i4_add) >> i4_shift));
254
26.5M
                    pi2_tmp[k + 4] =
255
26.5M
                                    CLIP_S16(((ai4_e[3 - k] - ai4_o[3 - k] + i4_add) >> i4_shift));
256
26.5M
                }
257
6.59M
            }
258
19.1M
            pi2_src++;
259
19.1M
            pi2_tmp += i4_trans_size;
260
19.1M
            i4_zero_cols = i4_zero_cols >> 1;
261
19.1M
        }
262
263
4.76M
        pi2_tmp = pi2_tmp_orig;
264
265
        /* Inverse Transform 2nd stage */
266
4.76M
        i4_shift = IDCT_STG2_SHIFT;
267
4.76M
        i4_add = 1 << (i4_shift - 1);
268
4.76M
        if((i4_zero_rows_2nd_stage & 0xF0) == 0xF0) /* First 4 rows of output of 1st stage are non-zero */
269
4.74M
        {
270
41.6M
            for(j = 0; j < i4_trans_size; j++)
271
36.9M
            {
272
                /* Utilizing symmetry properties to the maximum to minimize the number of multiplications */
273
183M
                for(k = 0; k < 4; k++)
274
146M
                {
275
146M
                    ai4_o[k] = gai2_impeg2_idct_q11[1 * 8 + k] * pi2_tmp[i4_trans_size]
276
146M
                                    + gai2_impeg2_idct_q11[3 * 8 + k] * pi2_tmp[3 * i4_trans_size];
277
146M
                }
278
36.9M
                ai4_eo[0] = gai2_impeg2_idct_q11[2 * 8 + 0] * pi2_tmp[2 * i4_trans_size];
279
36.9M
                ai4_eo[1] = gai2_impeg2_idct_q11[2 * 8 + 1] * pi2_tmp[2 * i4_trans_size];
280
36.9M
                ai4_ee[0] = gai2_impeg2_idct_q11[0 * 8 + 0] * pi2_tmp[0];
281
36.9M
                ai4_ee[1] = gai2_impeg2_idct_q11[0 * 8 + 1] * pi2_tmp[0];
282
283
                /* Combining e and o terms at each hierarchy levels to calculate the final spatial domain vector */
284
36.9M
                ai4_e[0] = ai4_ee[0] + ai4_eo[0];
285
36.9M
                ai4_e[3] = ai4_ee[0] - ai4_eo[0];
286
36.9M
                ai4_e[1] = ai4_ee[1] + ai4_eo[1];
287
36.9M
                ai4_e[2] = ai4_ee[1] - ai4_eo[1];
288
181M
                for(k = 0; k < 4; k++)
289
144M
                {
290
144M
                    WORD32 itrans_out;
291
144M
                    itrans_out =
292
144M
                                    CLIP_S16(((ai4_e[k] + ai4_o[k] + i4_add) >> i4_shift));
293
144M
                    pu1_dst[k] = CLIP_U8((itrans_out + pu1_pred[k]));
294
144M
                    itrans_out =
295
144M
                                    CLIP_S16(((ai4_e[3 - k] - ai4_o[3 - k] + i4_add) >> i4_shift));
296
144M
                    pu1_dst[k + 4] = CLIP_U8((itrans_out + pu1_pred[k + 4]));
297
144M
                }
298
36.9M
                pi2_tmp++;
299
36.9M
                pu1_pred += i4_pred_strd;
300
36.9M
                pu1_dst += i4_dst_strd;
301
36.9M
            }
302
4.74M
        }
303
20.6k
        else /* All rows of output of 1st stage are non-zero */
304
20.6k
        {
305
250k
            for(j = 0; j < i4_trans_size; j++)
306
229k
            {
307
                /* Utilizing symmetry properties to the maximum to minimize the number of multiplications */
308
1.14M
                for(k = 0; k < 4; k++)
309
918k
                {
310
918k
                    ai4_o[k] = gai2_impeg2_idct_q11[1 * 8 + k] * pi2_tmp[i4_trans_size]
311
918k
                                    + gai2_impeg2_idct_q11[3 * 8 + k]
312
918k
                                                    * pi2_tmp[3 * i4_trans_size]
313
918k
                                    + gai2_impeg2_idct_q11[5 * 8 + k]
314
918k
                                                    * pi2_tmp[5 * i4_trans_size]
315
918k
                                    + gai2_impeg2_idct_q11[7 * 8 + k]
316
918k
                                                    * pi2_tmp[7 * i4_trans_size];
317
918k
                }
318
319
229k
                ai4_eo[0] = gai2_impeg2_idct_q11[2 * 8 + 0] * pi2_tmp[2 * i4_trans_size]
320
229k
                                + gai2_impeg2_idct_q11[6 * 8 + 0] * pi2_tmp[6 * i4_trans_size];
321
229k
                ai4_eo[1] = gai2_impeg2_idct_q11[2 * 8 + 1] * pi2_tmp[2 * i4_trans_size]
322
229k
                                + gai2_impeg2_idct_q11[6 * 8 + 1] * pi2_tmp[6 * i4_trans_size];
323
229k
                ai4_ee[0] = gai2_impeg2_idct_q11[0 * 8 + 0] * pi2_tmp[0]
324
229k
                                + gai2_impeg2_idct_q11[4 * 8 + 0] * pi2_tmp[4 * i4_trans_size];
325
229k
                ai4_ee[1] = gai2_impeg2_idct_q11[0 * 8 + 1] * pi2_tmp[0]
326
229k
                                + gai2_impeg2_idct_q11[4 * 8 + 1] * pi2_tmp[4 * i4_trans_size];
327
328
                /* Combining e and o terms at each hierarchy levels to calculate the final spatial domain vector */
329
229k
                ai4_e[0] = ai4_ee[0] + ai4_eo[0];
330
229k
                ai4_e[3] = ai4_ee[0] - ai4_eo[0];
331
229k
                ai4_e[1] = ai4_ee[1] + ai4_eo[1];
332
229k
                ai4_e[2] = ai4_ee[1] - ai4_eo[1];
333
1.14M
                for(k = 0; k < 4; k++)
334
917k
                {
335
917k
                    WORD32 itrans_out;
336
917k
                    itrans_out =
337
917k
                                    CLIP_S16(((ai4_e[k] + ai4_o[k] + i4_add) >> i4_shift));
338
917k
                    pu1_dst[k] = CLIP_U8((itrans_out + pu1_pred[k]));
339
917k
                    itrans_out =
340
917k
                                    CLIP_S16(((ai4_e[3 - k] - ai4_o[3 - k] + i4_add) >> i4_shift));
341
917k
                    pu1_dst[k + 4] = CLIP_U8((itrans_out + pu1_pred[k + 4]));
342
917k
                }
343
229k
                pi2_tmp++;
344
229k
                pu1_pred += i4_pred_strd;
345
229k
                pu1_dst += i4_dst_strd;
346
229k
            }
347
20.6k
        }
348
        /************************************************************************************************/
349
        /************************************END - IT_RECON_8x8******************************************/
350
        /************************************************************************************************/
351
4.76M
    }
352
793k
    else /* All rows of input are non-zero */
353
793k
    {
354
        /************************************************************************************************/
355
        /**********************************START - IT_RECON_8x8******************************************/
356
        /************************************************************************************************/
357
358
        /* Inverse Transform 1st stage */
359
793k
        i4_shift = IDCT_STG1_SHIFT;
360
793k
        i4_add = 1 << (i4_shift - 1);
361
362
6.40M
        for(j = 0; j < i4_row_limit_2nd_stage; j++)
363
5.61M
        {
364
            /* Checking for Zero Cols */
365
5.61M
            if((i4_zero_cols & 1) == 1)
366
2.98M
            {
367
2.98M
                memset(pi2_tmp, 0, i4_trans_size * sizeof(WORD16));
368
2.98M
            }
369
2.62M
            else
370
2.62M
            {
371
                /* Utilizing symmetry properties to the maximum to minimize the number of multiplications */
372
13.1M
                for(k = 0; k < 4; k++)
373
10.5M
                {
374
10.5M
                    ai4_o[k] = gai2_impeg2_idct_q15[1 * 8 + k] * pi2_src[i4_src_strd]
375
10.5M
                                    + gai2_impeg2_idct_q15[3 * 8 + k]
376
10.5M
                                                    * pi2_src[3 * i4_src_strd]
377
10.5M
                                    + gai2_impeg2_idct_q15[5 * 8 + k]
378
10.5M
                                                    * pi2_src[5 * i4_src_strd]
379
10.5M
                                    + gai2_impeg2_idct_q15[7 * 8 + k]
380
10.5M
                                                    * pi2_src[7 * i4_src_strd];
381
10.5M
                }
382
383
2.62M
                ai4_eo[0] = gai2_impeg2_idct_q15[2 * 8 + 0] * pi2_src[2 * i4_src_strd]
384
2.62M
                                + gai2_impeg2_idct_q15[6 * 8 + 0] * pi2_src[6 * i4_src_strd];
385
2.62M
                ai4_eo[1] = gai2_impeg2_idct_q15[2 * 8 + 1] * pi2_src[2 * i4_src_strd]
386
2.62M
                                + gai2_impeg2_idct_q15[6 * 8 + 1] * pi2_src[6 * i4_src_strd];
387
2.62M
                ai4_ee[0] = gai2_impeg2_idct_q15[0 * 8 + 0] * pi2_src[0]
388
2.62M
                                + gai2_impeg2_idct_q15[4 * 8 + 0] * pi2_src[4 * i4_src_strd];
389
2.62M
                ai4_ee[1] = gai2_impeg2_idct_q15[0 * 8 + 1] * pi2_src[0]
390
2.62M
                                + gai2_impeg2_idct_q15[4 * 8 + 1] * pi2_src[4 * i4_src_strd];
391
392
                /* Combining e and o terms at each hierarchy levels to calculate the final spatial domain vector */
393
2.62M
                ai4_e[0] = ai4_ee[0] + ai4_eo[0];
394
2.62M
                ai4_e[3] = ai4_ee[0] - ai4_eo[0];
395
2.62M
                ai4_e[1] = ai4_ee[1] + ai4_eo[1];
396
2.62M
                ai4_e[2] = ai4_ee[1] - ai4_eo[1];
397
13.1M
                for(k = 0; k < 4; k++)
398
10.4M
                {
399
10.4M
                    pi2_tmp[k] =
400
10.4M
                                    CLIP_S16(((ai4_e[k] + ai4_o[k] + i4_add) >> i4_shift));
401
10.4M
                    pi2_tmp[k + 4] =
402
10.4M
                                    CLIP_S16(((ai4_e[3 - k] - ai4_o[3 - k] + i4_add) >> i4_shift));
403
10.4M
                }
404
2.62M
            }
405
5.61M
            pi2_src++;
406
5.61M
            pi2_tmp += i4_trans_size;
407
5.61M
            i4_zero_cols = i4_zero_cols >> 1;
408
5.61M
        }
409
410
793k
        pi2_tmp = pi2_tmp_orig;
411
412
        /* Inverse Transform 2nd stage */
413
793k
        i4_shift = IDCT_STG2_SHIFT;
414
793k
        i4_add = 1 << (i4_shift - 1);
415
793k
        if((i4_zero_rows_2nd_stage & 0xF0) == 0xF0) /* First 4 rows of output of 1st stage are non-zero */
416
180k
        {
417
1.61M
            for(j = 0; j < i4_trans_size; j++)
418
1.43M
            {
419
                /* Utilizing symmetry properties to the maximum to minimize the number of multiplications */
420
7.19M
                for(k = 0; k < 4; k++)
421
5.75M
                {
422
5.75M
                    ai4_o[k] = gai2_impeg2_idct_q11[1 * 8 + k] * pi2_tmp[i4_trans_size]
423
5.75M
                                    + gai2_impeg2_idct_q11[3 * 8 + k] * pi2_tmp[3 * i4_trans_size];
424
5.75M
                }
425
1.43M
                ai4_eo[0] = gai2_impeg2_idct_q11[2 * 8 + 0] * pi2_tmp[2 * i4_trans_size];
426
1.43M
                ai4_eo[1] = gai2_impeg2_idct_q11[2 * 8 + 1] * pi2_tmp[2 * i4_trans_size];
427
1.43M
                ai4_ee[0] = gai2_impeg2_idct_q11[0 * 8 + 0] * pi2_tmp[0];
428
1.43M
                ai4_ee[1] = gai2_impeg2_idct_q11[0 * 8 + 1] * pi2_tmp[0];
429
430
                /* Combining e and o terms at each hierarchy levels to calculate the final spatial domain vector */
431
1.43M
                ai4_e[0] = ai4_ee[0] + ai4_eo[0];
432
1.43M
                ai4_e[3] = ai4_ee[0] - ai4_eo[0];
433
1.43M
                ai4_e[1] = ai4_ee[1] + ai4_eo[1];
434
1.43M
                ai4_e[2] = ai4_ee[1] - ai4_eo[1];
435
7.19M
                for(k = 0; k < 4; k++)
436
5.75M
                {
437
5.75M
                    WORD32 itrans_out;
438
5.75M
                    itrans_out =
439
5.75M
                                    CLIP_S16(((ai4_e[k] + ai4_o[k] + i4_add) >> i4_shift));
440
5.75M
                    pu1_dst[k] = CLIP_U8((itrans_out + pu1_pred[k]));
441
5.75M
                    itrans_out =
442
5.75M
                                    CLIP_S16(((ai4_e[3 - k] - ai4_o[3 - k] + i4_add) >> i4_shift));
443
5.75M
                    pu1_dst[k + 4] = CLIP_U8((itrans_out + pu1_pred[k + 4]));
444
5.75M
                }
445
1.43M
                pi2_tmp++;
446
1.43M
                pu1_pred += i4_pred_strd;
447
1.43M
                pu1_dst += i4_dst_strd;
448
1.43M
            }
449
180k
        }
450
613k
        else /* All rows of output of 1st stage are non-zero */
451
613k
        {
452
5.45M
            for(j = 0; j < i4_trans_size; j++)
453
4.84M
            {
454
                /* Utilizing symmetry properties to the maximum to minimize the number of multiplications */
455
24.0M
                for(k = 0; k < 4; k++)
456
19.2M
                {
457
19.2M
                    ai4_o[k] = gai2_impeg2_idct_q11[1 * 8 + k] * pi2_tmp[i4_trans_size]
458
19.2M
                                    + gai2_impeg2_idct_q11[3 * 8 + k]
459
19.2M
                                                    * pi2_tmp[3 * i4_trans_size]
460
19.2M
                                    + gai2_impeg2_idct_q11[5 * 8 + k]
461
19.2M
                                                    * pi2_tmp[5 * i4_trans_size]
462
19.2M
                                    + gai2_impeg2_idct_q11[7 * 8 + k]
463
19.2M
                                                    * pi2_tmp[7 * i4_trans_size];
464
19.2M
                }
465
466
4.84M
                ai4_eo[0] = gai2_impeg2_idct_q11[2 * 8 + 0] * pi2_tmp[2 * i4_trans_size]
467
4.84M
                                + gai2_impeg2_idct_q11[6 * 8 + 0] * pi2_tmp[6 * i4_trans_size];
468
4.84M
                ai4_eo[1] = gai2_impeg2_idct_q11[2 * 8 + 1] * pi2_tmp[2 * i4_trans_size]
469
4.84M
                                + gai2_impeg2_idct_q11[6 * 8 + 1] * pi2_tmp[6 * i4_trans_size];
470
4.84M
                ai4_ee[0] = gai2_impeg2_idct_q11[0 * 8 + 0] * pi2_tmp[0]
471
4.84M
                                + gai2_impeg2_idct_q11[4 * 8 + 0] * pi2_tmp[4 * i4_trans_size];
472
4.84M
                ai4_ee[1] = gai2_impeg2_idct_q11[0 * 8 + 1] * pi2_tmp[0]
473
4.84M
                                + gai2_impeg2_idct_q11[4 * 8 + 1] * pi2_tmp[4 * i4_trans_size];
474
475
                /* Combining e and o terms at each hierarchy levels to calculate the final spatial domain vector */
476
4.84M
                ai4_e[0] = ai4_ee[0] + ai4_eo[0];
477
4.84M
                ai4_e[3] = ai4_ee[0] - ai4_eo[0];
478
4.84M
                ai4_e[1] = ai4_ee[1] + ai4_eo[1];
479
4.84M
                ai4_e[2] = ai4_ee[1] - ai4_eo[1];
480
23.9M
                for(k = 0; k < 4; k++)
481
19.1M
                {
482
19.1M
                    WORD32 itrans_out;
483
19.1M
                    itrans_out =
484
19.1M
                                    CLIP_S16(((ai4_e[k] + ai4_o[k] + i4_add) >> i4_shift));
485
19.1M
                    pu1_dst[k] = CLIP_U8((itrans_out + pu1_pred[k]));
486
19.1M
                    itrans_out =
487
19.1M
                                    CLIP_S16(((ai4_e[3 - k] - ai4_o[3 - k] + i4_add) >> i4_shift));
488
19.1M
                    pu1_dst[k + 4] = CLIP_U8((itrans_out + pu1_pred[k + 4]));
489
19.1M
                }
490
4.84M
                pi2_tmp++;
491
4.84M
                pu1_pred += i4_pred_strd;
492
4.84M
                pu1_dst += i4_dst_strd;
493
4.84M
            }
494
613k
        }
495
        /************************************************************************************************/
496
        /************************************END - IT_RECON_8x8******************************************/
497
        /************************************************************************************************/
498
793k
    }
499
5.55M
}
500