Coverage Report

Created: 2026-09-01 07:15

next uncovered line (L), next uncovered region (R), next uncovered branch (B)
/src/libmpeg2/common/impeg2_idct.c
Line
Count
Source
1
/******************************************************************************
2
 *
3
 * Copyright (C) 2015 The Android Open Source Project
4
 *
5
 * Licensed under the Apache License, Version 2.0 (the "License");
6
 * you may not use this file except in compliance with the License.
7
 * You may obtain a copy of the License at:
8
 *
9
 * http://www.apache.org/licenses/LICENSE-2.0
10
 *
11
 * Unless required by applicable law or agreed to in writing, software
12
 * distributed under the License is distributed on an "AS IS" BASIS,
13
 * WITHOUT WARRANTIES OR CONDITIONS OF ANY KIND, either express or implied.
14
 * See the License for the specific language governing permissions and
15
 * limitations under the License.
16
 *
17
 *****************************************************************************
18
 * Originally developed and contributed by Ittiam Systems Pvt. Ltd, Bangalore
19
*/
20
/*****************************************************************************/
21
/*                                                                           */
22
/*  File Name         : impeg2_idct.c                                        */
23
/*                                                                           */
24
/*  Description       : Contains 2d idct and invese quantization functions   */
25
/*                                                                           */
26
/*  List of Functions : impeg2_idct_recon_dc()                               */
27
/*                      impeg2_idct_recon_dc_mismatch()                      */
28
/*                      impeg2_idct_recon()                                  */
29
/*                                                                           */
30
/*  Issues / Problems : None                                                 */
31
/*                                                                           */
32
/*  Revision History  :                                                      */
33
/*                                                                           */
34
/*         DD MM YYYY   Author(s)       Changes                              */
35
/*         10 09 2005   Hairsh M        First Version                        */
36
/*                                                                           */
37
/*****************************************************************************/
38
/*
39
  IEEE - 1180 results for this IDCT
40
  L                           256         256         5           5           300         300         384         384         Thresholds
41
  H                           255         255         5           5           300         300         383         383
42
  sign                        1           -1          1           -1          1           -1          1           -1
43
  Peak Error                  1           1           1           1           1           1           1           1           1
44
  Peak Mean Square Error      0.0191      0.0188      0.0108      0.0111      0.0176      0.0188      0.0165      0.0177      0.06
45
  Overall Mean Square Error   0.01566406  0.01597656  0.0091875   0.00908906  0.01499063  0.01533281  0.01432344  0.01412344  0.02
46
  Peak Mean Error             0.0027      0.0026      0.0028      0.002       0.0017      0.0033      0.0031      0.0025      0.015
47
  Overall Mean Error          0.00002656  -0.00031406 0.00016875  0.00005469  -0.00003125 0.00011406  0.00009219  0.00004219  0.0015
48
  */
49
#include <stdio.h>
50
#include <string.h>
51
52
#include "iv_datatypedef.h"
53
#include "iv.h"
54
#include "impeg2_defs.h"
55
#include "impeg2_platform_macros.h"
56
57
#include "impeg2_macros.h"
58
#include "impeg2_globals.h"
59
#include "impeg2_idct.h"
60
61
62
void impeg2_idct_recon_dc(WORD16 *pi2_src,
63
                            WORD16 *pi2_tmp,
64
                            UWORD8 *pu1_pred,
65
                            UWORD8 *pu1_dst,
66
                            WORD32 i4_src_strd,
67
                            WORD32 i4_pred_strd,
68
                            WORD32 i4_dst_strd,
69
                            WORD32 i4_zero_cols,
70
                            WORD32 i4_zero_rows)
71
745k
{
72
745k
    WORD32 i4_val, i, j;
73
74
745k
    UNUSED(pi2_tmp);
75
745k
    UNUSED(i4_src_strd);
76
745k
    UNUSED(i4_zero_cols);
77
745k
    UNUSED(i4_zero_rows);
78
79
745k
    i4_val = pi2_src[0] * gai2_impeg2_idct_q15[0];
80
745k
    i4_val = ((i4_val + IDCT_STG1_ROUND) >> IDCT_STG1_SHIFT);
81
745k
    i4_val = i4_val * gai2_impeg2_idct_q11[0];
82
745k
    i4_val = ((i4_val + IDCT_STG2_ROUND) >> IDCT_STG2_SHIFT);
83
84
6.60M
    for(i = 0; i < TRANS_SIZE_8; i++)
85
5.86M
    {
86
52.4M
        for(j = 0; j < TRANS_SIZE_8; j++)
87
46.6M
        {
88
46.6M
            pu1_dst[j] = CLIP_U8(i4_val + pu1_pred[j]);
89
46.6M
        }
90
5.86M
        pu1_dst  += i4_dst_strd;
91
5.86M
        pu1_pred += i4_pred_strd;
92
5.86M
    }
93
745k
}
94
void impeg2_idct_recon_dc_mismatch(WORD16 *pi2_src,
95
                            WORD16 *pi2_tmp,
96
                            UWORD8 *pu1_pred,
97
                            UWORD8 *pu1_dst,
98
                            WORD32 i4_src_strd,
99
                            WORD32 i4_pred_strd,
100
                            WORD32 i4_dst_strd,
101
                            WORD32 i4_zero_cols,
102
                            WORD32 i4_zero_rows)
103
104
135k
{
105
135k
    WORD32 i4_val, i, j;
106
135k
    WORD32 i4_count = 0;
107
135k
    WORD32 i4_sum;
108
109
135k
    UNUSED(pi2_tmp);
110
135k
    UNUSED(i4_src_strd);
111
135k
    UNUSED(i4_zero_cols);
112
135k
    UNUSED(i4_zero_rows);
113
114
135k
    i4_val = pi2_src[0] * gai2_impeg2_idct_q15[0];
115
135k
    i4_val = ((i4_val + IDCT_STG1_ROUND) >> IDCT_STG1_SHIFT);
116
117
135k
    i4_val *= gai2_impeg2_idct_q11[0];
118
1.20M
    for(i = 0; i < TRANS_SIZE_8; i++)
119
1.07M
    {
120
9.61M
        for (j = 0; j < TRANS_SIZE_8; j++)
121
8.54M
        {
122
8.54M
            i4_sum = i4_val;
123
8.54M
            i4_sum += gai2_impeg2_mismatch_stg2_additive[i4_count];
124
8.54M
            i4_sum = ((i4_sum + IDCT_STG2_ROUND) >> IDCT_STG2_SHIFT);
125
8.54M
            i4_sum += pu1_pred[j];
126
8.54M
            pu1_dst[j] = CLIP_U8(i4_sum);
127
8.54M
            i4_count++;
128
8.54M
        }
129
130
1.07M
        pu1_dst  += i4_dst_strd;
131
1.07M
        pu1_pred += i4_pred_strd;
132
1.07M
    }
133
134
135k
}
135
/**
136
 *******************************************************************************
137
 *
138
 * @brief
139
 *  This function performs Inverse transform  and reconstruction for 8x8
140
 * input block
141
 *
142
 * @par Description:
143
 *  Performs inverse transform and adds the prediction  data and clips output
144
 * to 8 bit
145
 *
146
 * @param[in] pi2_src
147
 *  Input 8x8 coefficients
148
 *
149
 * @param[in] pi2_tmp
150
 *  Temporary 8x8 buffer for storing inverse
151
 *
152
 *  transform
153
 *  1st stage output
154
 *
155
 * @param[in] pu1_pred
156
 *  Prediction 8x8 block
157
 *
158
 * @param[out] pu1_dst
159
 *  Output 8x8 block
160
 *
161
 * @param[in] src_strd
162
 *  Input stride
163
 *
164
 * @param[in] pred_strd
165
 *  Prediction stride
166
 *
167
 * @param[in] dst_strd
168
 *  Output Stride
169
 *
170
 * @param[in] shift
171
 *  Output shift
172
 *
173
 * @param[in] zero_cols
174
 *  Zero columns in pi2_src
175
 *
176
 * @returns  Void
177
 *
178
 * @remarks
179
 *  None
180
 *
181
 *******************************************************************************
182
 */
183
184
void impeg2_idct_recon(WORD16 *pi2_src,
185
                        WORD16 *pi2_tmp,
186
                        UWORD8 *pu1_pred,
187
                        UWORD8 *pu1_dst,
188
                        WORD32 i4_src_strd,
189
                        WORD32 i4_pred_strd,
190
                        WORD32 i4_dst_strd,
191
                        WORD32 i4_zero_cols,
192
                        WORD32 i4_zero_rows)
193
6.60M
{
194
6.60M
    WORD32 j, k;
195
6.60M
    WORD32 ai4_e[4], ai4_o[4];
196
6.60M
    WORD32 ai4_ee[2], ai4_eo[2];
197
6.60M
    WORD32 i4_add;
198
6.60M
    WORD32 i4_shift;
199
6.60M
    WORD16 *pi2_tmp_orig;
200
6.60M
    WORD32 i4_trans_size;
201
6.60M
    WORD32 i4_zero_rows_2nd_stage = i4_zero_cols;
202
6.60M
    WORD32 i4_row_limit_2nd_stage;
203
204
6.60M
    i4_trans_size = TRANS_SIZE_8;
205
206
6.60M
    pi2_tmp_orig = pi2_tmp;
207
208
6.60M
    if((i4_zero_cols & 0xF0) == 0xF0)
209
6.01M
        i4_row_limit_2nd_stage = 4;
210
585k
    else
211
585k
        i4_row_limit_2nd_stage = TRANS_SIZE_8;
212
213
214
6.60M
    if((i4_zero_rows & 0xF0) == 0xF0) /* First 4 rows of input are non-zero */
215
5.83M
    {
216
        /************************************************************************************************/
217
        /**********************************START - IT_RECON_8x8******************************************/
218
        /************************************************************************************************/
219
220
        /* Inverse Transform 1st stage */
221
5.83M
        i4_shift = IDCT_STG1_SHIFT;
222
5.83M
        i4_add = 1 << (i4_shift - 1);
223
224
29.2M
        for(j = 0; j < i4_row_limit_2nd_stage; j++)
225
23.4M
        {
226
            /* Checking for Zero Cols */
227
23.4M
            if((i4_zero_cols & 1) == 1)
228
15.2M
            {
229
15.2M
                memset(pi2_tmp, 0, i4_trans_size * sizeof(WORD16));
230
15.2M
            }
231
8.14M
            else
232
8.14M
            {
233
                /* Utilizing symmetry properties to the maximum to minimize the number of multiplications */
234
40.9M
                for(k = 0; k < 4; k++)
235
32.7M
                {
236
32.7M
                    ai4_o[k] = gai2_impeg2_idct_q15[1 * 8 + k] * pi2_src[i4_src_strd]
237
32.7M
                                    + gai2_impeg2_idct_q15[3 * 8 + k]
238
32.7M
                                                    * pi2_src[3 * i4_src_strd];
239
32.7M
                }
240
8.14M
                ai4_eo[0] = gai2_impeg2_idct_q15[2 * 8 + 0] * pi2_src[2 * i4_src_strd];
241
8.14M
                ai4_eo[1] = gai2_impeg2_idct_q15[2 * 8 + 1] * pi2_src[2 * i4_src_strd];
242
8.14M
                ai4_ee[0] = gai2_impeg2_idct_q15[0 * 8 + 0] * pi2_src[0];
243
8.14M
                ai4_ee[1] = gai2_impeg2_idct_q15[0 * 8 + 1] * pi2_src[0];
244
245
                /* Combining e and o terms at each hierarchy levels to calculate the final spatial domain vector */
246
8.14M
                ai4_e[0] = ai4_ee[0] + ai4_eo[0];
247
8.14M
                ai4_e[3] = ai4_ee[0] - ai4_eo[0];
248
8.14M
                ai4_e[1] = ai4_ee[1] + ai4_eo[1];
249
8.14M
                ai4_e[2] = ai4_ee[1] - ai4_eo[1];
250
40.9M
                for(k = 0; k < 4; k++)
251
32.7M
                {
252
32.7M
                    pi2_tmp[k] =
253
32.7M
                                    CLIP_S16(((ai4_e[k] + ai4_o[k] + i4_add) >> i4_shift));
254
32.7M
                    pi2_tmp[k + 4] =
255
32.7M
                                    CLIP_S16(((ai4_e[3 - k] - ai4_o[3 - k] + i4_add) >> i4_shift));
256
32.7M
                }
257
8.14M
            }
258
23.4M
            pi2_src++;
259
23.4M
            pi2_tmp += i4_trans_size;
260
23.4M
            i4_zero_cols = i4_zero_cols >> 1;
261
23.4M
        }
262
263
5.83M
        pi2_tmp = pi2_tmp_orig;
264
265
        /* Inverse Transform 2nd stage */
266
5.83M
        i4_shift = IDCT_STG2_SHIFT;
267
5.83M
        i4_add = 1 << (i4_shift - 1);
268
5.83M
        if((i4_zero_rows_2nd_stage & 0xF0) == 0xF0) /* First 4 rows of output of 1st stage are non-zero */
269
5.81M
        {
270
51.0M
            for(j = 0; j < i4_trans_size; j++)
271
45.2M
            {
272
                /* Utilizing symmetry properties to the maximum to minimize the number of multiplications */
273
224M
                for(k = 0; k < 4; k++)
274
179M
                {
275
179M
                    ai4_o[k] = gai2_impeg2_idct_q11[1 * 8 + k] * pi2_tmp[i4_trans_size]
276
179M
                                    + gai2_impeg2_idct_q11[3 * 8 + k] * pi2_tmp[3 * i4_trans_size];
277
179M
                }
278
45.2M
                ai4_eo[0] = gai2_impeg2_idct_q11[2 * 8 + 0] * pi2_tmp[2 * i4_trans_size];
279
45.2M
                ai4_eo[1] = gai2_impeg2_idct_q11[2 * 8 + 1] * pi2_tmp[2 * i4_trans_size];
280
45.2M
                ai4_ee[0] = gai2_impeg2_idct_q11[0 * 8 + 0] * pi2_tmp[0];
281
45.2M
                ai4_ee[1] = gai2_impeg2_idct_q11[0 * 8 + 1] * pi2_tmp[0];
282
283
                /* Combining e and o terms at each hierarchy levels to calculate the final spatial domain vector */
284
45.2M
                ai4_e[0] = ai4_ee[0] + ai4_eo[0];
285
45.2M
                ai4_e[3] = ai4_ee[0] - ai4_eo[0];
286
45.2M
                ai4_e[1] = ai4_ee[1] + ai4_eo[1];
287
45.2M
                ai4_e[2] = ai4_ee[1] - ai4_eo[1];
288
221M
                for(k = 0; k < 4; k++)
289
176M
                {
290
176M
                    WORD32 itrans_out;
291
176M
                    itrans_out =
292
176M
                                    CLIP_S16(((ai4_e[k] + ai4_o[k] + i4_add) >> i4_shift));
293
176M
                    pu1_dst[k] = CLIP_U8((itrans_out + pu1_pred[k]));
294
176M
                    itrans_out =
295
176M
                                    CLIP_S16(((ai4_e[3 - k] - ai4_o[3 - k] + i4_add) >> i4_shift));
296
176M
                    pu1_dst[k + 4] = CLIP_U8((itrans_out + pu1_pred[k + 4]));
297
176M
                }
298
45.2M
                pi2_tmp++;
299
45.2M
                pu1_pred += i4_pred_strd;
300
45.2M
                pu1_dst += i4_dst_strd;
301
45.2M
            }
302
5.81M
        }
303
18.8k
        else /* All rows of output of 1st stage are non-zero */
304
18.8k
        {
305
234k
            for(j = 0; j < i4_trans_size; j++)
306
215k
            {
307
                /* Utilizing symmetry properties to the maximum to minimize the number of multiplications */
308
1.07M
                for(k = 0; k < 4; k++)
309
863k
                {
310
863k
                    ai4_o[k] = gai2_impeg2_idct_q11[1 * 8 + k] * pi2_tmp[i4_trans_size]
311
863k
                                    + gai2_impeg2_idct_q11[3 * 8 + k]
312
863k
                                                    * pi2_tmp[3 * i4_trans_size]
313
863k
                                    + gai2_impeg2_idct_q11[5 * 8 + k]
314
863k
                                                    * pi2_tmp[5 * i4_trans_size]
315
863k
                                    + gai2_impeg2_idct_q11[7 * 8 + k]
316
863k
                                                    * pi2_tmp[7 * i4_trans_size];
317
863k
                }
318
319
215k
                ai4_eo[0] = gai2_impeg2_idct_q11[2 * 8 + 0] * pi2_tmp[2 * i4_trans_size]
320
215k
                                + gai2_impeg2_idct_q11[6 * 8 + 0] * pi2_tmp[6 * i4_trans_size];
321
215k
                ai4_eo[1] = gai2_impeg2_idct_q11[2 * 8 + 1] * pi2_tmp[2 * i4_trans_size]
322
215k
                                + gai2_impeg2_idct_q11[6 * 8 + 1] * pi2_tmp[6 * i4_trans_size];
323
215k
                ai4_ee[0] = gai2_impeg2_idct_q11[0 * 8 + 0] * pi2_tmp[0]
324
215k
                                + gai2_impeg2_idct_q11[4 * 8 + 0] * pi2_tmp[4 * i4_trans_size];
325
215k
                ai4_ee[1] = gai2_impeg2_idct_q11[0 * 8 + 1] * pi2_tmp[0]
326
215k
                                + gai2_impeg2_idct_q11[4 * 8 + 1] * pi2_tmp[4 * i4_trans_size];
327
328
                /* Combining e and o terms at each hierarchy levels to calculate the final spatial domain vector */
329
215k
                ai4_e[0] = ai4_ee[0] + ai4_eo[0];
330
215k
                ai4_e[3] = ai4_ee[0] - ai4_eo[0];
331
215k
                ai4_e[1] = ai4_ee[1] + ai4_eo[1];
332
215k
                ai4_e[2] = ai4_ee[1] - ai4_eo[1];
333
1.07M
                for(k = 0; k < 4; k++)
334
863k
                {
335
863k
                    WORD32 itrans_out;
336
863k
                    itrans_out =
337
863k
                                    CLIP_S16(((ai4_e[k] + ai4_o[k] + i4_add) >> i4_shift));
338
863k
                    pu1_dst[k] = CLIP_U8((itrans_out + pu1_pred[k]));
339
863k
                    itrans_out =
340
863k
                                    CLIP_S16(((ai4_e[3 - k] - ai4_o[3 - k] + i4_add) >> i4_shift));
341
863k
                    pu1_dst[k + 4] = CLIP_U8((itrans_out + pu1_pred[k + 4]));
342
863k
                }
343
215k
                pi2_tmp++;
344
215k
                pu1_pred += i4_pred_strd;
345
215k
                pu1_dst += i4_dst_strd;
346
215k
            }
347
18.8k
        }
348
        /************************************************************************************************/
349
        /************************************END - IT_RECON_8x8******************************************/
350
        /************************************************************************************************/
351
5.83M
    }
352
768k
    else /* All rows of input are non-zero */
353
768k
    {
354
        /************************************************************************************************/
355
        /**********************************START - IT_RECON_8x8******************************************/
356
        /************************************************************************************************/
357
358
        /* Inverse Transform 1st stage */
359
768k
        i4_shift = IDCT_STG1_SHIFT;
360
768k
        i4_add = 1 << (i4_shift - 1);
361
362
6.06M
        for(j = 0; j < i4_row_limit_2nd_stage; j++)
363
5.29M
        {
364
            /* Checking for Zero Cols */
365
5.29M
            if((i4_zero_cols & 1) == 1)
366
2.70M
            {
367
2.70M
                memset(pi2_tmp, 0, i4_trans_size * sizeof(WORD16));
368
2.70M
            }
369
2.58M
            else
370
2.58M
            {
371
                /* Utilizing symmetry properties to the maximum to minimize the number of multiplications */
372
12.9M
                for(k = 0; k < 4; k++)
373
10.3M
                {
374
10.3M
                    ai4_o[k] = gai2_impeg2_idct_q15[1 * 8 + k] * pi2_src[i4_src_strd]
375
10.3M
                                    + gai2_impeg2_idct_q15[3 * 8 + k]
376
10.3M
                                                    * pi2_src[3 * i4_src_strd]
377
10.3M
                                    + gai2_impeg2_idct_q15[5 * 8 + k]
378
10.3M
                                                    * pi2_src[5 * i4_src_strd]
379
10.3M
                                    + gai2_impeg2_idct_q15[7 * 8 + k]
380
10.3M
                                                    * pi2_src[7 * i4_src_strd];
381
10.3M
                }
382
383
2.58M
                ai4_eo[0] = gai2_impeg2_idct_q15[2 * 8 + 0] * pi2_src[2 * i4_src_strd]
384
2.58M
                                + gai2_impeg2_idct_q15[6 * 8 + 0] * pi2_src[6 * i4_src_strd];
385
2.58M
                ai4_eo[1] = gai2_impeg2_idct_q15[2 * 8 + 1] * pi2_src[2 * i4_src_strd]
386
2.58M
                                + gai2_impeg2_idct_q15[6 * 8 + 1] * pi2_src[6 * i4_src_strd];
387
2.58M
                ai4_ee[0] = gai2_impeg2_idct_q15[0 * 8 + 0] * pi2_src[0]
388
2.58M
                                + gai2_impeg2_idct_q15[4 * 8 + 0] * pi2_src[4 * i4_src_strd];
389
2.58M
                ai4_ee[1] = gai2_impeg2_idct_q15[0 * 8 + 1] * pi2_src[0]
390
2.58M
                                + gai2_impeg2_idct_q15[4 * 8 + 1] * pi2_src[4 * i4_src_strd];
391
392
                /* Combining e and o terms at each hierarchy levels to calculate the final spatial domain vector */
393
2.58M
                ai4_e[0] = ai4_ee[0] + ai4_eo[0];
394
2.58M
                ai4_e[3] = ai4_ee[0] - ai4_eo[0];
395
2.58M
                ai4_e[1] = ai4_ee[1] + ai4_eo[1];
396
2.58M
                ai4_e[2] = ai4_ee[1] - ai4_eo[1];
397
12.9M
                for(k = 0; k < 4; k++)
398
10.3M
                {
399
10.3M
                    pi2_tmp[k] =
400
10.3M
                                    CLIP_S16(((ai4_e[k] + ai4_o[k] + i4_add) >> i4_shift));
401
10.3M
                    pi2_tmp[k + 4] =
402
10.3M
                                    CLIP_S16(((ai4_e[3 - k] - ai4_o[3 - k] + i4_add) >> i4_shift));
403
10.3M
                }
404
2.58M
            }
405
5.29M
            pi2_src++;
406
5.29M
            pi2_tmp += i4_trans_size;
407
5.29M
            i4_zero_cols = i4_zero_cols >> 1;
408
5.29M
        }
409
410
768k
        pi2_tmp = pi2_tmp_orig;
411
412
        /* Inverse Transform 2nd stage */
413
768k
        i4_shift = IDCT_STG2_SHIFT;
414
768k
        i4_add = 1 << (i4_shift - 1);
415
768k
        if((i4_zero_rows_2nd_stage & 0xF0) == 0xF0) /* First 4 rows of output of 1st stage are non-zero */
416
209k
        {
417
1.88M
            for(j = 0; j < i4_trans_size; j++)
418
1.67M
            {
419
                /* Utilizing symmetry properties to the maximum to minimize the number of multiplications */
420
8.38M
                for(k = 0; k < 4; k++)
421
6.70M
                {
422
6.70M
                    ai4_o[k] = gai2_impeg2_idct_q11[1 * 8 + k] * pi2_tmp[i4_trans_size]
423
6.70M
                                    + gai2_impeg2_idct_q11[3 * 8 + k] * pi2_tmp[3 * i4_trans_size];
424
6.70M
                }
425
1.67M
                ai4_eo[0] = gai2_impeg2_idct_q11[2 * 8 + 0] * pi2_tmp[2 * i4_trans_size];
426
1.67M
                ai4_eo[1] = gai2_impeg2_idct_q11[2 * 8 + 1] * pi2_tmp[2 * i4_trans_size];
427
1.67M
                ai4_ee[0] = gai2_impeg2_idct_q11[0 * 8 + 0] * pi2_tmp[0];
428
1.67M
                ai4_ee[1] = gai2_impeg2_idct_q11[0 * 8 + 1] * pi2_tmp[0];
429
430
                /* Combining e and o terms at each hierarchy levels to calculate the final spatial domain vector */
431
1.67M
                ai4_e[0] = ai4_ee[0] + ai4_eo[0];
432
1.67M
                ai4_e[3] = ai4_ee[0] - ai4_eo[0];
433
1.67M
                ai4_e[1] = ai4_ee[1] + ai4_eo[1];
434
1.67M
                ai4_e[2] = ai4_ee[1] - ai4_eo[1];
435
8.37M
                for(k = 0; k < 4; k++)
436
6.70M
                {
437
6.70M
                    WORD32 itrans_out;
438
6.70M
                    itrans_out =
439
6.70M
                                    CLIP_S16(((ai4_e[k] + ai4_o[k] + i4_add) >> i4_shift));
440
6.70M
                    pu1_dst[k] = CLIP_U8((itrans_out + pu1_pred[k]));
441
6.70M
                    itrans_out =
442
6.70M
                                    CLIP_S16(((ai4_e[3 - k] - ai4_o[3 - k] + i4_add) >> i4_shift));
443
6.70M
                    pu1_dst[k + 4] = CLIP_U8((itrans_out + pu1_pred[k + 4]));
444
6.70M
                }
445
1.67M
                pi2_tmp++;
446
1.67M
                pu1_pred += i4_pred_strd;
447
1.67M
                pu1_dst += i4_dst_strd;
448
1.67M
            }
449
209k
        }
450
558k
        else /* All rows of output of 1st stage are non-zero */
451
558k
        {
452
4.95M
            for(j = 0; j < i4_trans_size; j++)
453
4.39M
            {
454
                /* Utilizing symmetry properties to the maximum to minimize the number of multiplications */
455
21.8M
                for(k = 0; k < 4; k++)
456
17.4M
                {
457
17.4M
                    ai4_o[k] = gai2_impeg2_idct_q11[1 * 8 + k] * pi2_tmp[i4_trans_size]
458
17.4M
                                    + gai2_impeg2_idct_q11[3 * 8 + k]
459
17.4M
                                                    * pi2_tmp[3 * i4_trans_size]
460
17.4M
                                    + gai2_impeg2_idct_q11[5 * 8 + k]
461
17.4M
                                                    * pi2_tmp[5 * i4_trans_size]
462
17.4M
                                    + gai2_impeg2_idct_q11[7 * 8 + k]
463
17.4M
                                                    * pi2_tmp[7 * i4_trans_size];
464
17.4M
                }
465
466
4.39M
                ai4_eo[0] = gai2_impeg2_idct_q11[2 * 8 + 0] * pi2_tmp[2 * i4_trans_size]
467
4.39M
                                + gai2_impeg2_idct_q11[6 * 8 + 0] * pi2_tmp[6 * i4_trans_size];
468
4.39M
                ai4_eo[1] = gai2_impeg2_idct_q11[2 * 8 + 1] * pi2_tmp[2 * i4_trans_size]
469
4.39M
                                + gai2_impeg2_idct_q11[6 * 8 + 1] * pi2_tmp[6 * i4_trans_size];
470
4.39M
                ai4_ee[0] = gai2_impeg2_idct_q11[0 * 8 + 0] * pi2_tmp[0]
471
4.39M
                                + gai2_impeg2_idct_q11[4 * 8 + 0] * pi2_tmp[4 * i4_trans_size];
472
4.39M
                ai4_ee[1] = gai2_impeg2_idct_q11[0 * 8 + 1] * pi2_tmp[0]
473
4.39M
                                + gai2_impeg2_idct_q11[4 * 8 + 1] * pi2_tmp[4 * i4_trans_size];
474
475
                /* Combining e and o terms at each hierarchy levels to calculate the final spatial domain vector */
476
4.39M
                ai4_e[0] = ai4_ee[0] + ai4_eo[0];
477
4.39M
                ai4_e[3] = ai4_ee[0] - ai4_eo[0];
478
4.39M
                ai4_e[1] = ai4_ee[1] + ai4_eo[1];
479
4.39M
                ai4_e[2] = ai4_ee[1] - ai4_eo[1];
480
21.7M
                for(k = 0; k < 4; k++)
481
17.3M
                {
482
17.3M
                    WORD32 itrans_out;
483
17.3M
                    itrans_out =
484
17.3M
                                    CLIP_S16(((ai4_e[k] + ai4_o[k] + i4_add) >> i4_shift));
485
17.3M
                    pu1_dst[k] = CLIP_U8((itrans_out + pu1_pred[k]));
486
17.3M
                    itrans_out =
487
17.3M
                                    CLIP_S16(((ai4_e[3 - k] - ai4_o[3 - k] + i4_add) >> i4_shift));
488
17.3M
                    pu1_dst[k + 4] = CLIP_U8((itrans_out + pu1_pred[k + 4]));
489
17.3M
                }
490
4.39M
                pi2_tmp++;
491
4.39M
                pu1_pred += i4_pred_strd;
492
4.39M
                pu1_dst += i4_dst_strd;
493
4.39M
            }
494
558k
        }
495
        /************************************************************************************************/
496
        /************************************END - IT_RECON_8x8******************************************/
497
        /************************************************************************************************/
498
768k
    }
499
6.60M
}
500