Coverage Report

Created: 2026-09-02 06:43

next uncovered line (L), next uncovered region (R), next uncovered branch (B)
/work/vvenc/source/Lib/CommonLib/TrQuant.cpp
Line
Count
Source
1
/* -----------------------------------------------------------------------------
2
The copyright in this software is being made available under the Clear BSD
3
License, included below. No patent rights, trademark rights and/or 
4
other Intellectual Property Rights other than the copyrights concerning 
5
the Software are granted under this license.
6
7
The Clear BSD License
8
9
Copyright (c) 2019-2026, Fraunhofer-Gesellschaft zur Förderung der angewandten Forschung e.V. & The VVenC Authors.
10
All rights reserved.
11
12
Redistribution and use in source and binary forms, with or without modification,
13
are permitted (subject to the limitations in the disclaimer below) provided that
14
the following conditions are met:
15
16
     * Redistributions of source code must retain the above copyright notice,
17
     this list of conditions and the following disclaimer.
18
19
     * Redistributions in binary form must reproduce the above copyright
20
     notice, this list of conditions and the following disclaimer in the
21
     documentation and/or other materials provided with the distribution.
22
23
     * Neither the name of the copyright holder nor the names of its
24
     contributors may be used to endorse or promote products derived from this
25
     software without specific prior written permission.
26
27
NO EXPRESS OR IMPLIED LICENSES TO ANY PARTY'S PATENT RIGHTS ARE GRANTED BY
28
THIS LICENSE. THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND
29
CONTRIBUTORS "AS IS" AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT
30
LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A
31
PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR
32
CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL,
33
EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO,
34
PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR
35
BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER
36
IN CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE)
37
ARISING IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE
38
POSSIBILITY OF SUCH DAMAGE.
39
40
41
------------------------------------------------------------------------------------------- */
42
43
44
/** \file     TrQuant.cpp
45
    \brief    transform and quantization class
46
*/
47
48
#include "TrQuant.h"
49
#include "TrQuant_EMT.h"
50
#include "QuantRDOQ.h"
51
#include "DepQuant.h"
52
#include "UnitTools.h"
53
#include "ContextModelling.h"
54
#include "CodingStructure.h"
55
#include "dtrace_buffer.h"
56
#include "TimeProfiler.h"
57
#include "SearchSpaceCounter.h"
58
59
#include <stdlib.h>
60
#include <memory.h>
61
62
//! \ingroup CommonLib
63
//! \{
64
65
namespace vvenc {
66
67
struct coeffGroupRDStats
68
{
69
  int    iNNZbeforePos0;
70
  double d64CodedLevelandDist; // distortion and level cost only
71
  double d64UncodedDist;    // all zero coded block distortion
72
  double d64SigCost;
73
  double d64SigCost_0;
74
};
75
76
FwdTrans *const fastFwdTrans[NUM_TRANS_TYPE][g_numTransformMatrixSizes] =
77
{
78
  { fastForwardDCT2_B2, fastForwardDCT2_B4, fastForwardDCT2_B8, fastForwardDCT2_B16, fastForwardDCT2_B32, fastForwardDCT2_B64 },
79
  { nullptr,            fastForwardDCT8_B4, fastForwardDCT8_B8, fastForwardDCT8_B16, fastForwardDCT8_B32, nullptr },
80
  { nullptr,            fastForwardDST7_B4, fastForwardDST7_B8, fastForwardDST7_B16, fastForwardDST7_B32, nullptr },
81
};
82
83
InvTrans *const fastInvTrans[NUM_TRANS_TYPE][g_numTransformMatrixSizes] =
84
{
85
  { fastInverseDCT2_B2, fastInverseDCT2_B4, fastInverseDCT2_B8, fastInverseDCT2_B16, fastInverseDCT2_B32, fastInverseDCT2_B64 },
86
  { nullptr,            fastInverseDCT8_B4, fastInverseDCT8_B8, fastInverseDCT8_B16, fastInverseDCT8_B32, nullptr },
87
  { nullptr,            fastInverseDST7_B4, fastInverseDST7_B8, fastInverseDST7_B16, fastInverseDST7_B32, nullptr },
88
};
89
90
//! \ingroup CommonLib
91
//! \{
92
93
334M
static inline int64_t square( const int d ) { return d * (int64_t)d; }
94
95
template<int signedMode> std::pair<int64_t,int64_t> fwdTransformCbCr( const PelBuf& resCb, const PelBuf& resCr, PelBuf& resC1, PelBuf& resC2 )
96
1.01M
{
97
1.01M
  const Pel*  cb  = resCb.buf;
98
1.01M
  const Pel*  cr  = resCr.buf;
99
1.01M
  Pel*        c1  = resC1.buf;
100
1.01M
  Pel*        c2  = resC2.buf;
101
1.01M
  int64_t     d1  = 0;
102
1.01M
  int64_t     d2  = 0;
103
13.5M
  for( SizeType y = 0; y < resCb.height; y++, cb += resCb.stride, cr += resCr.stride, c1 += resC1.stride, c2 += resC2.stride )
104
12.5M
  {
105
179M
    for( SizeType x = 0; x < resCb.width; x++ )
106
167M
    {
107
167M
      int cbx = cb[x], crx = cr[x];
108
167M
      if      ( signedMode ==  1 )
109
41.7M
      {
110
41.7M
        c1[x] = Pel( ( 4*cbx + 2*crx ) / 5 );
111
41.7M
        d1   += square( cbx - c1[x] ) + square( crx - (c1[x]>>1) );
112
41.7M
      }
113
125M
      else if ( signedMode == -1 )
114
0
      {
115
0
        c1[x] = Pel( ( 4*cbx - 2*crx ) / 5 );
116
0
        d1   += square( cbx - c1[x] ) + square( crx - (-c1[x]>>1) );
117
0
      }
118
125M
      else if ( signedMode ==  2 )
119
41.7M
      {
120
41.7M
        c1[x] = Pel( ( cbx + crx ) / 2 );
121
41.7M
        d1   += square( cbx - c1[x] ) + square( crx - c1[x] );
122
41.7M
      }
123
83.5M
      else if ( signedMode == -2 )
124
0
      {
125
0
        c1[x] = Pel( ( cbx - crx ) / 2 );
126
0
        d1   += square( cbx - c1[x] ) + square( crx + c1[x] );
127
0
      }
128
83.5M
      else if ( signedMode ==  3 )
129
41.7M
      {
130
41.7M
        c2[x] = Pel( ( 4*crx + 2*cbx ) / 5 );
131
41.7M
        d1   += square( cbx - (c2[x]>>1) ) + square( crx - c2[x] );
132
41.7M
      }
133
41.7M
      else if ( signedMode == -3 )
134
0
      {
135
0
        c2[x] = Pel( ( 4*crx - 2*cbx ) / 5 );
136
0
        d1   += square( cbx - (-c2[x]>>1) ) + square( crx - c2[x] );
137
0
      }
138
41.7M
      else
139
41.7M
      {
140
41.7M
        d1   += square( cbx );
141
41.7M
        d2   += square( crx );
142
41.7M
      }
143
167M
    }
144
12.5M
  }
145
1.01M
  return std::make_pair(d1,d2);
146
1.01M
}
std::__1::pair<long, long> vvenc::fwdTransformCbCr<0>(vvenc::AreaBuf<short> const&, vvenc::AreaBuf<short> const&, vvenc::AreaBuf<short>&, vvenc::AreaBuf<short>&)
Line
Count
Source
96
252k
{
97
252k
  const Pel*  cb  = resCb.buf;
98
252k
  const Pel*  cr  = resCr.buf;
99
252k
  Pel*        c1  = resC1.buf;
100
252k
  Pel*        c2  = resC2.buf;
101
252k
  int64_t     d1  = 0;
102
252k
  int64_t     d2  = 0;
103
3.39M
  for( SizeType y = 0; y < resCb.height; y++, cb += resCb.stride, cr += resCr.stride, c1 += resC1.stride, c2 += resC2.stride )
104
3.14M
  {
105
44.9M
    for( SizeType x = 0; x < resCb.width; x++ )
106
41.7M
    {
107
41.7M
      int cbx = cb[x], crx = cr[x];
108
41.7M
      if      ( signedMode ==  1 )
109
0
      {
110
0
        c1[x] = Pel( ( 4*cbx + 2*crx ) / 5 );
111
0
        d1   += square( cbx - c1[x] ) + square( crx - (c1[x]>>1) );
112
0
      }
113
41.7M
      else if ( signedMode == -1 )
114
0
      {
115
0
        c1[x] = Pel( ( 4*cbx - 2*crx ) / 5 );
116
0
        d1   += square( cbx - c1[x] ) + square( crx - (-c1[x]>>1) );
117
0
      }
118
41.7M
      else if ( signedMode ==  2 )
119
0
      {
120
0
        c1[x] = Pel( ( cbx + crx ) / 2 );
121
0
        d1   += square( cbx - c1[x] ) + square( crx - c1[x] );
122
0
      }
123
41.7M
      else if ( signedMode == -2 )
124
0
      {
125
0
        c1[x] = Pel( ( cbx - crx ) / 2 );
126
0
        d1   += square( cbx - c1[x] ) + square( crx + c1[x] );
127
0
      }
128
41.7M
      else if ( signedMode ==  3 )
129
0
      {
130
0
        c2[x] = Pel( ( 4*crx + 2*cbx ) / 5 );
131
0
        d1   += square( cbx - (c2[x]>>1) ) + square( crx - c2[x] );
132
0
      }
133
41.7M
      else if ( signedMode == -3 )
134
0
      {
135
0
        c2[x] = Pel( ( 4*crx - 2*cbx ) / 5 );
136
0
        d1   += square( cbx - (-c2[x]>>1) ) + square( crx - c2[x] );
137
0
      }
138
41.7M
      else
139
41.7M
      {
140
41.7M
        d1   += square( cbx );
141
41.7M
        d2   += square( crx );
142
41.7M
      }
143
41.7M
    }
144
3.14M
  }
145
252k
  return std::make_pair(d1,d2);
146
252k
}
std::__1::pair<long, long> vvenc::fwdTransformCbCr<1>(vvenc::AreaBuf<short> const&, vvenc::AreaBuf<short> const&, vvenc::AreaBuf<short>&, vvenc::AreaBuf<short>&)
Line
Count
Source
96
252k
{
97
252k
  const Pel*  cb  = resCb.buf;
98
252k
  const Pel*  cr  = resCr.buf;
99
252k
  Pel*        c1  = resC1.buf;
100
252k
  Pel*        c2  = resC2.buf;
101
252k
  int64_t     d1  = 0;
102
252k
  int64_t     d2  = 0;
103
3.39M
  for( SizeType y = 0; y < resCb.height; y++, cb += resCb.stride, cr += resCr.stride, c1 += resC1.stride, c2 += resC2.stride )
104
3.14M
  {
105
44.9M
    for( SizeType x = 0; x < resCb.width; x++ )
106
41.7M
    {
107
41.7M
      int cbx = cb[x], crx = cr[x];
108
41.7M
      if      ( signedMode ==  1 )
109
41.7M
      {
110
41.7M
        c1[x] = Pel( ( 4*cbx + 2*crx ) / 5 );
111
41.7M
        d1   += square( cbx - c1[x] ) + square( crx - (c1[x]>>1) );
112
41.7M
      }
113
0
      else if ( signedMode == -1 )
114
0
      {
115
0
        c1[x] = Pel( ( 4*cbx - 2*crx ) / 5 );
116
0
        d1   += square( cbx - c1[x] ) + square( crx - (-c1[x]>>1) );
117
0
      }
118
0
      else if ( signedMode ==  2 )
119
0
      {
120
0
        c1[x] = Pel( ( cbx + crx ) / 2 );
121
0
        d1   += square( cbx - c1[x] ) + square( crx - c1[x] );
122
0
      }
123
0
      else if ( signedMode == -2 )
124
0
      {
125
0
        c1[x] = Pel( ( cbx - crx ) / 2 );
126
0
        d1   += square( cbx - c1[x] ) + square( crx + c1[x] );
127
0
      }
128
0
      else if ( signedMode ==  3 )
129
0
      {
130
0
        c2[x] = Pel( ( 4*crx + 2*cbx ) / 5 );
131
0
        d1   += square( cbx - (c2[x]>>1) ) + square( crx - c2[x] );
132
0
      }
133
0
      else if ( signedMode == -3 )
134
0
      {
135
0
        c2[x] = Pel( ( 4*crx - 2*cbx ) / 5 );
136
0
        d1   += square( cbx - (-c2[x]>>1) ) + square( crx - c2[x] );
137
0
      }
138
0
      else
139
0
      {
140
0
        d1   += square( cbx );
141
0
        d2   += square( crx );
142
0
      }
143
41.7M
    }
144
3.14M
  }
145
252k
  return std::make_pair(d1,d2);
146
252k
}
Unexecuted instantiation: std::__1::pair<long, long> vvenc::fwdTransformCbCr<-1>(vvenc::AreaBuf<short> const&, vvenc::AreaBuf<short> const&, vvenc::AreaBuf<short>&, vvenc::AreaBuf<short>&)
std::__1::pair<long, long> vvenc::fwdTransformCbCr<2>(vvenc::AreaBuf<short> const&, vvenc::AreaBuf<short> const&, vvenc::AreaBuf<short>&, vvenc::AreaBuf<short>&)
Line
Count
Source
96
252k
{
97
252k
  const Pel*  cb  = resCb.buf;
98
252k
  const Pel*  cr  = resCr.buf;
99
252k
  Pel*        c1  = resC1.buf;
100
252k
  Pel*        c2  = resC2.buf;
101
252k
  int64_t     d1  = 0;
102
252k
  int64_t     d2  = 0;
103
3.39M
  for( SizeType y = 0; y < resCb.height; y++, cb += resCb.stride, cr += resCr.stride, c1 += resC1.stride, c2 += resC2.stride )
104
3.14M
  {
105
44.9M
    for( SizeType x = 0; x < resCb.width; x++ )
106
41.7M
    {
107
41.7M
      int cbx = cb[x], crx = cr[x];
108
41.7M
      if      ( signedMode ==  1 )
109
0
      {
110
0
        c1[x] = Pel( ( 4*cbx + 2*crx ) / 5 );
111
0
        d1   += square( cbx - c1[x] ) + square( crx - (c1[x]>>1) );
112
0
      }
113
41.7M
      else if ( signedMode == -1 )
114
0
      {
115
0
        c1[x] = Pel( ( 4*cbx - 2*crx ) / 5 );
116
0
        d1   += square( cbx - c1[x] ) + square( crx - (-c1[x]>>1) );
117
0
      }
118
41.7M
      else if ( signedMode ==  2 )
119
41.7M
      {
120
41.7M
        c1[x] = Pel( ( cbx + crx ) / 2 );
121
41.7M
        d1   += square( cbx - c1[x] ) + square( crx - c1[x] );
122
41.7M
      }
123
0
      else if ( signedMode == -2 )
124
0
      {
125
0
        c1[x] = Pel( ( cbx - crx ) / 2 );
126
0
        d1   += square( cbx - c1[x] ) + square( crx + c1[x] );
127
0
      }
128
0
      else if ( signedMode ==  3 )
129
0
      {
130
0
        c2[x] = Pel( ( 4*crx + 2*cbx ) / 5 );
131
0
        d1   += square( cbx - (c2[x]>>1) ) + square( crx - c2[x] );
132
0
      }
133
0
      else if ( signedMode == -3 )
134
0
      {
135
0
        c2[x] = Pel( ( 4*crx - 2*cbx ) / 5 );
136
0
        d1   += square( cbx - (-c2[x]>>1) ) + square( crx - c2[x] );
137
0
      }
138
0
      else
139
0
      {
140
0
        d1   += square( cbx );
141
0
        d2   += square( crx );
142
0
      }
143
41.7M
    }
144
3.14M
  }
145
252k
  return std::make_pair(d1,d2);
146
252k
}
Unexecuted instantiation: std::__1::pair<long, long> vvenc::fwdTransformCbCr<-2>(vvenc::AreaBuf<short> const&, vvenc::AreaBuf<short> const&, vvenc::AreaBuf<short>&, vvenc::AreaBuf<short>&)
std::__1::pair<long, long> vvenc::fwdTransformCbCr<3>(vvenc::AreaBuf<short> const&, vvenc::AreaBuf<short> const&, vvenc::AreaBuf<short>&, vvenc::AreaBuf<short>&)
Line
Count
Source
96
252k
{
97
252k
  const Pel*  cb  = resCb.buf;
98
252k
  const Pel*  cr  = resCr.buf;
99
252k
  Pel*        c1  = resC1.buf;
100
252k
  Pel*        c2  = resC2.buf;
101
252k
  int64_t     d1  = 0;
102
252k
  int64_t     d2  = 0;
103
3.39M
  for( SizeType y = 0; y < resCb.height; y++, cb += resCb.stride, cr += resCr.stride, c1 += resC1.stride, c2 += resC2.stride )
104
3.14M
  {
105
44.9M
    for( SizeType x = 0; x < resCb.width; x++ )
106
41.7M
    {
107
41.7M
      int cbx = cb[x], crx = cr[x];
108
41.7M
      if      ( signedMode ==  1 )
109
0
      {
110
0
        c1[x] = Pel( ( 4*cbx + 2*crx ) / 5 );
111
0
        d1   += square( cbx - c1[x] ) + square( crx - (c1[x]>>1) );
112
0
      }
113
41.7M
      else if ( signedMode == -1 )
114
0
      {
115
0
        c1[x] = Pel( ( 4*cbx - 2*crx ) / 5 );
116
0
        d1   += square( cbx - c1[x] ) + square( crx - (-c1[x]>>1) );
117
0
      }
118
41.7M
      else if ( signedMode ==  2 )
119
0
      {
120
0
        c1[x] = Pel( ( cbx + crx ) / 2 );
121
0
        d1   += square( cbx - c1[x] ) + square( crx - c1[x] );
122
0
      }
123
41.7M
      else if ( signedMode == -2 )
124
0
      {
125
0
        c1[x] = Pel( ( cbx - crx ) / 2 );
126
0
        d1   += square( cbx - c1[x] ) + square( crx + c1[x] );
127
0
      }
128
41.7M
      else if ( signedMode ==  3 )
129
41.7M
      {
130
41.7M
        c2[x] = Pel( ( 4*crx + 2*cbx ) / 5 );
131
41.7M
        d1   += square( cbx - (c2[x]>>1) ) + square( crx - c2[x] );
132
41.7M
      }
133
0
      else if ( signedMode == -3 )
134
0
      {
135
0
        c2[x] = Pel( ( 4*crx - 2*cbx ) / 5 );
136
0
        d1   += square( cbx - (-c2[x]>>1) ) + square( crx - c2[x] );
137
0
      }
138
0
      else
139
0
      {
140
0
        d1   += square( cbx );
141
0
        d2   += square( crx );
142
0
      }
143
41.7M
    }
144
3.14M
  }
145
252k
  return std::make_pair(d1,d2);
146
252k
}
Unexecuted instantiation: std::__1::pair<long, long> vvenc::fwdTransformCbCr<-3>(vvenc::AreaBuf<short> const&, vvenc::AreaBuf<short> const&, vvenc::AreaBuf<short>&, vvenc::AreaBuf<short>&)
147
148
template<int signedMode> void invTransformCbCr( PelBuf& resCb, PelBuf& resCr )
149
249k
{
150
249k
  Pel*  cb  = resCb.buf;
151
249k
  Pel*  cr  = resCr.buf;
152
3.35M
  for( SizeType y = 0; y < resCb.height; y++, cb += resCb.stride, cr += resCr.stride )
153
3.10M
  {
154
44.4M
    for( SizeType x = 0; x < resCb.width; x++ )
155
41.3M
    {
156
41.3M
      if      ( signedMode ==  1 )  { cr[x] =  cb[x] >> 1;  }
157
41.3M
      else if ( signedMode == -1 )  { cr[x] = -cb[x] >> 1;  }
158
41.3M
      else if ( signedMode ==  2 )  { cr[x] =  cb[x]; }
159
229k
      else if ( signedMode == -2 )  { cr[x] = -cb[x]; }
160
229k
      else if ( signedMode ==  3 )  { cb[x] =  cr[x] >> 1; }
161
0
      else if ( signedMode == -3 )  { cb[x] = -cr[x] >> 1; }
162
41.3M
    }
163
3.10M
  }
164
249k
}
Unexecuted instantiation: void vvenc::invTransformCbCr<0>(vvenc::AreaBuf<short>&, vvenc::AreaBuf<short>&)
Unexecuted instantiation: void vvenc::invTransformCbCr<1>(vvenc::AreaBuf<short>&, vvenc::AreaBuf<short>&)
Unexecuted instantiation: void vvenc::invTransformCbCr<-1>(vvenc::AreaBuf<short>&, vvenc::AreaBuf<short>&)
void vvenc::invTransformCbCr<2>(vvenc::AreaBuf<short>&, vvenc::AreaBuf<short>&)
Line
Count
Source
149
247k
{
150
247k
  Pel*  cb  = resCb.buf;
151
247k
  Pel*  cr  = resCr.buf;
152
3.33M
  for( SizeType y = 0; y < resCb.height; y++, cb += resCb.stride, cr += resCr.stride )
153
3.08M
  {
154
44.1M
    for( SizeType x = 0; x < resCb.width; x++ )
155
41.0M
    {
156
41.0M
      if      ( signedMode ==  1 )  { cr[x] =  cb[x] >> 1;  }
157
41.0M
      else if ( signedMode == -1 )  { cr[x] = -cb[x] >> 1;  }
158
41.0M
      else if ( signedMode ==  2 )  { cr[x] =  cb[x]; }
159
0
      else if ( signedMode == -2 )  { cr[x] = -cb[x]; }
160
0
      else if ( signedMode ==  3 )  { cb[x] =  cr[x] >> 1; }
161
0
      else if ( signedMode == -3 )  { cb[x] = -cr[x] >> 1; }
162
41.0M
    }
163
3.08M
  }
164
247k
}
Unexecuted instantiation: void vvenc::invTransformCbCr<-2>(vvenc::AreaBuf<short>&, vvenc::AreaBuf<short>&)
void vvenc::invTransformCbCr<3>(vvenc::AreaBuf<short>&, vvenc::AreaBuf<short>&)
Line
Count
Source
149
1.67k
{
150
1.67k
  Pel*  cb  = resCb.buf;
151
1.67k
  Pel*  cr  = resCr.buf;
152
21.9k
  for( SizeType y = 0; y < resCb.height; y++, cb += resCb.stride, cr += resCr.stride )
153
20.2k
  {
154
249k
    for( SizeType x = 0; x < resCb.width; x++ )
155
229k
    {
156
229k
      if      ( signedMode ==  1 )  { cr[x] =  cb[x] >> 1;  }
157
229k
      else if ( signedMode == -1 )  { cr[x] = -cb[x] >> 1;  }
158
229k
      else if ( signedMode ==  2 )  { cr[x] =  cb[x]; }
159
229k
      else if ( signedMode == -2 )  { cr[x] = -cb[x]; }
160
229k
      else if ( signedMode ==  3 )  { cb[x] =  cr[x] >> 1; }
161
0
      else if ( signedMode == -3 )  { cb[x] = -cr[x] >> 1; }
162
229k
    }
163
20.2k
  }
164
1.67k
}
Unexecuted instantiation: void vvenc::invTransformCbCr<-3>(vvenc::AreaBuf<short>&, vvenc::AreaBuf<short>&)
165
166
void xFwdLfnstNxNCore(int *src, int *dst, const uint32_t mode, const uint32_t index, const uint32_t size, int zeroOutSize)
167
1.11M
{
168
1.11M
  const int8_t *trMat  = (size > 4) ? g_lfnstFwd8x8[mode][index][0] : g_lfnstFwd4x4[mode][index][0];
169
1.11M
  const int     trSize = (size > 4) ? 48 : 16;
170
1.11M
  int           coef;
171
1.11M
  int *         out = dst;
172
173
17.2M
  for (int j = 0; j < zeroOutSize; j++)
174
16.1M
  {
175
16.1M
    int *         srcPtr   = src;
176
16.1M
    const int8_t *trMatTmp = trMat;
177
16.1M
    coef                   = 0;
178
612M
    for (int i = 0; i < trSize; i++)
179
596M
    {
180
596M
      coef += *srcPtr++ * *trMatTmp++;
181
596M
    }
182
16.1M
    *out++ = (coef + 64) >> 7;
183
16.1M
    trMat += trSize;
184
16.1M
  }
185
186
1.11M
  ::memset(out, 0, (trSize - zeroOutSize) * sizeof(int));
187
1.11M
}
188
189
190
void xInvLfnstNxNCore(int *src, int *dst, const uint32_t mode, const uint32_t index, const uint32_t size, int zeroOutSize)
191
475k
{
192
475k
  int           maxLog2TrDynamicRange = 15;
193
475k
  const TCoeff  outputMinimum         = -(1 << maxLog2TrDynamicRange);
194
475k
  const TCoeff  outputMaximum         = (1 << maxLog2TrDynamicRange) - 1;
195
475k
  const int8_t *trMat                 = (size > 4) ? g_lfnstInv8x8[mode][index][0] : g_lfnstInv4x4[mode][index][0];
196
475k
  const int     trSize                = (size > 4) ? 48 : 16;
197
475k
  int           resi;
198
475k
  int *         out                   = dst;
199
200
17.7M
  for( int j = 0; j < trSize; j++, trMat += 16 )
201
17.2M
  {
202
17.2M
    resi = 0;
203
17.2M
    const int8_t* trMatTmp = trMat;
204
17.2M
    int*          srcPtr   = src;
205
206
261M
    for( int i = 0; i < zeroOutSize; i++ )
207
244M
    {
208
244M
      resi += *srcPtr++ * *trMatTmp++;
209
244M
    }
210
211
17.2M
    *out++ = Clip3( outputMinimum, outputMaximum, ( int ) ( resi + 64 ) >> 7 );
212
17.2M
  }
213
475k
}
214
215
// ====================================================================================================================
216
// TrQuant class member functions
217
// ====================================================================================================================
218
18.3k
TrQuant::TrQuant() : m_scalingListEnabled(false), m_quant( nullptr )
219
18.3k
{
220
  // allocate temporary buffers
221
18.3k
  m_plTempCoeff = ( TCoeff* ) xMalloc( TCoeff, MAX_TB_SIZEY * MAX_TB_SIZEY );
222
18.3k
  m_tmp         = ( TCoeff* ) xMalloc( TCoeff, MAX_TB_SIZEY * MAX_TB_SIZEY );
223
18.3k
  m_blk         = ( TCoeff* ) xMalloc( TCoeff, MAX_TB_SIZEY * MAX_TB_SIZEY );
224
225
128k
  for( int i = 0; i < NUM_TRAFO_MODES_MTS; i++ )
226
110k
  {
227
110k
    m_mtsCoeffs[i] = ( TCoeff* ) xMalloc( TCoeff, MAX_TB_SIZEY * MAX_TB_SIZEY );
228
110k
  }
229
230
18.3k
  {
231
18.3k
    m_invICT      = m_invICTMem + maxAbsIctMode;
232
18.3k
    m_invICT[ 0]  = invTransformCbCr< 0>;
233
18.3k
    m_invICT[ 1]  = invTransformCbCr< 1>;
234
18.3k
    m_invICT[-1]  = invTransformCbCr<-1>;
235
18.3k
    m_invICT[ 2]  = invTransformCbCr< 2>;
236
18.3k
    m_invICT[-2]  = invTransformCbCr<-2>;
237
18.3k
    m_invICT[ 3]  = invTransformCbCr< 3>;
238
18.3k
    m_invICT[-3]  = invTransformCbCr<-3>;
239
18.3k
    m_fwdICT      = m_fwdICTMem + maxAbsIctMode;
240
18.3k
    m_fwdICT[ 0]  = fwdTransformCbCr< 0>;
241
18.3k
    m_fwdICT[ 1]  = fwdTransformCbCr< 1>;
242
18.3k
    m_fwdICT[-1]  = fwdTransformCbCr<-1>;
243
18.3k
    m_fwdICT[ 2]  = fwdTransformCbCr< 2>;
244
18.3k
    m_fwdICT[-2]  = fwdTransformCbCr<-2>;
245
18.3k
    m_fwdICT[ 3]  = fwdTransformCbCr< 3>;
246
18.3k
    m_fwdICT[-3]  = fwdTransformCbCr<-3>;
247
18.3k
  }
248
249
18.3k
  m_invLfnstNxN = xInvLfnstNxNCore;
250
18.3k
  m_fwdLfnstNxN = xFwdLfnstNxNCore;
251
252
#if defined( TARGET_SIMD_X86 ) && ENABLE_SIMD_TRAFO
253
  initTrQuantX86();
254
#endif
255
18.3k
}
256
257
TrQuant::~TrQuant()
258
18.3k
{
259
18.3k
  if( m_quant )
260
18.3k
  {
261
18.3k
    delete m_quant;
262
18.3k
    m_quant = nullptr;
263
18.3k
  }
264
265
  // delete temporary buffers
266
18.3k
  if( m_plTempCoeff )
267
18.3k
  {
268
18.3k
    xFree( m_plTempCoeff );
269
18.3k
    m_plTempCoeff = nullptr;
270
18.3k
  }
271
272
18.3k
  if( m_blk )
273
18.3k
  {
274
18.3k
    xFree( m_blk );
275
18.3k
    m_blk = nullptr;
276
18.3k
  }
277
278
18.3k
  if( m_tmp )
279
18.3k
  {
280
18.3k
    xFree( m_tmp );
281
18.3k
    m_tmp = nullptr;
282
18.3k
  }
283
284
128k
  for( int i = 0; i < NUM_TRAFO_MODES_MTS; i++ )
285
110k
  {
286
110k
     xFree( m_mtsCoeffs[i] );
287
110k
  }
288
18.3k
}
289
290
void TrQuant::xDeQuant(const TransformUnit& tu,
291
                             CoeffBuf      &dstCoeff,
292
                       const ComponentID   &compID,
293
                       const QpParam       &cQP)
294
780k
{
295
780k
  PROFILER_SCOPE_AND_STAGE( 1, _TPROF, P_DEQUANT );
296
780k
  m_quant->dequant( tu, dstCoeff, compID, cQP );
297
780k
}
298
299
void TrQuant::init( const Quant* otherQuant,
300
                    const int  rdoq,
301
                    const bool bUseRDOQTS,
302
                    const bool scalingListsEnabled,
303
                    const bool bEnc,
304
                    const int  thrVal
305
)
306
18.3k
{
307
18.3k
  m_bEnc = bEnc;
308
309
18.3k
  delete m_quant;
310
18.3k
  m_quant = nullptr;
311
312
18.3k
  m_quant = new(std::nothrow) DepQuant( otherQuant, bEnc, scalingListsEnabled );
313
18.3k
  CHECK( !m_quant, "allocation failed" );
314
18.3k
  m_quant->init( rdoq, bUseRDOQTS, thrVal );
315
18.3k
}
316
317
318
void TrQuant::invTransformNxN( TransformUnit& tu, const ComponentID compID, PelBuf& pResi, const QpParam& cQP )
319
780k
{
320
780k
  const CompArea& area    = tu.blocks[compID];
321
780k
  const uint32_t uiWidth  = area.width;
322
780k
  const uint32_t uiHeight = area.height;
323
324
780k
  CHECK( uiWidth > tu.cs->sps->getMaxTbSize() || uiHeight > tu.cs->sps->getMaxTbSize(), "Maximal allowed transformation size exceeded!" );
325
326
780k
  {
327
780k
    CoeffBuf tempCoeff = CoeffBuf( m_plTempCoeff, area );
328
780k
    xDeQuant( tu, tempCoeff, compID, cQP );
329
330
780k
    DTRACE_COEFF_BUF( D_TCOEFF, tempCoeff, tu, tu.cu->predMode, compID );
331
332
780k
    if (tu.cs->sps->LFNST)
333
780k
    {
334
780k
      xInvLfnst(tu, compID);
335
780k
    }
336
780k
    if (tu.mtsIdx[compID] == MTS_SKIP)
337
45.8k
    {
338
45.8k
      xITransformSkip(tempCoeff, pResi, tu, compID);
339
45.8k
    }
340
734k
    else
341
734k
    {
342
734k
      xIT(tu, compID, tempCoeff, pResi);
343
734k
    }
344
780k
  }
345
346
  //DTRACE_BLOCK_COEFF(tu.getCoeffs(compID), tu, tu.cu->predMode, compID);
347
780k
  DTRACE_PEL_BUF( D_RESIDUALS, pResi, tu, tu.cu->predMode, compID);
348
780k
}
349
350
std::pair<int64_t,int64_t> TrQuant::fwdTransformICT( const TransformUnit& tu, const PelBuf& resCb, const PelBuf& resCr, PelBuf& resC1, PelBuf& resC2, int jointCbCr )
351
1.01M
{
352
1.01M
  CHECK( Size(resCb) != Size(resCr), "resCb and resCr have different sizes" );
353
1.01M
  CHECK( Size(resCb) != Size(resC1), "resCb and resC1 have different sizes" );
354
1.01M
  CHECK( Size(resCb) != Size(resC2), "resCb and resC2 have different sizes" );
355
1.01M
  return (*m_fwdICT[ TU::getICTMode(tu, jointCbCr) ])( resCb, resCr, resC1, resC2 );
356
1.01M
}
357
358
void TrQuant::invTransformICT( const TransformUnit& tu, PelBuf& resCb, PelBuf& resCr )
359
249k
{
360
249k
  CHECK( Size(resCb) != Size(resCr), "resCb and resCr have different sizes" );
361
249k
  (*m_invICT[ TU::getICTMode(tu) ])( resCb, resCr );
362
249k
}
363
364
std::vector<int> TrQuant::selectICTCandidates( const TransformUnit& tu, CompStorage* resCb, CompStorage* resCr )
365
252k
{
366
252k
  CHECK( !resCb[0].valid() || !resCr[0].valid(), "standard components are not valid" );
367
368
252k
  if( !CU::isIntra( *tu.cu ) )
369
0
  {
370
0
    int cbfMask = 3;
371
0
    fwdTransformICT( tu, resCb[0], resCr[0], resCb[cbfMask], resCr[cbfMask], cbfMask );
372
0
    std::vector<int> cbfMasksToTest;
373
0
    cbfMasksToTest.push_back( cbfMask );
374
0
    return cbfMasksToTest;
375
0
  }
376
377
252k
  std::pair<int64_t,int64_t> pairDist[4];
378
1.26M
  for( int cbfMask = 0; cbfMask < 4; cbfMask++ )
379
1.01M
  {
380
1.01M
    pairDist[cbfMask] = fwdTransformICT( tu, resCb[0], resCr[0], resCb[cbfMask], resCr[cbfMask], cbfMask );
381
1.01M
  }
382
383
252k
  std::vector<int> cbfMasksToTest;
384
252k
  int64_t minDist1  = std::min<int64_t>( pairDist[0].first, pairDist[0].second );
385
252k
  int64_t minDist2  = std::numeric_limits<int64_t>::max();
386
252k
  int     cbfMask1  = 0;
387
252k
  int     cbfMask2  = 0;
388
252k
  for( int cbfMask : { 1, 2, 3 } )
389
757k
  {
390
757k
    if( pairDist[cbfMask].first < minDist1 )
391
500k
    {
392
500k
      cbfMask2  = cbfMask1; minDist2  = minDist1;
393
500k
      cbfMask1  = cbfMask;  minDist1  = pairDist[cbfMask1].first;
394
500k
    }
395
257k
    else if( pairDist[cbfMask].first < minDist2 )
396
252k
    {
397
252k
      cbfMask2  = cbfMask;  minDist2  = pairDist[cbfMask2].first;
398
252k
    }
399
757k
  }
400
252k
  if( cbfMask1 )
401
252k
  {
402
252k
    cbfMasksToTest.push_back( cbfMask1 );
403
252k
  }
404
252k
  if( cbfMask2 && ( ( minDist2 < (9*minDist1)/8 ) || ( !cbfMask1 && minDist2 < (3*minDist1)/2 ) ) )
405
0
  {
406
0
    cbfMasksToTest.push_back( cbfMask2 );
407
0
  }
408
409
252k
  return cbfMasksToTest;
410
252k
}
411
412
413
414
// ------------------------------------------------------------------------------------------------
415
// Logical transform
416
// ------------------------------------------------------------------------------------------------
417
void TrQuant::xSetTrTypes( const TransformUnit& tu, const ComponentID compID, const int width, const int height, int &trTypeHor, int &trTypeVer )
418
2.52M
{
419
2.52M
  const bool isISP = CU::isIntra(*tu.cu) && tu.cu->ispMode && isLuma(compID);
420
2.52M
  if (isISP && tu.cu->lfnstIdx)
421
17.8k
  {
422
17.8k
    return;
423
17.8k
  }
424
2.50M
  if (!tu.cs->sps->MTS)
425
0
  {
426
0
    return;
427
0
  }
428
2.50M
  if (CU::isIntra(*tu.cu) && isLuma(compID) && ((tu.cs->sps->getUseImplicitMTS() && tu.cu->lfnstIdx == 0 && tu.cu->mipFlag == 0) || tu.cu->ispMode))
429
78.5k
  {
430
78.5k
    if (width >= 4 && width <= 16)
431
35.8k
      trTypeHor = DST7;
432
78.5k
    if (height >= 4 && height <= 16)
433
33.7k
      trTypeVer = DST7;
434
78.5k
  }
435
2.42M
  else if( tu.cs->sps->MTS && tu.cu->sbtInfo && isLuma(compID)/*isSBT*/ )
436
0
  {
437
0
    const uint8_t sbtIdx = CU::getSbtIdx( tu.cu->sbtInfo );
438
0
    const uint8_t sbtPos = CU::getSbtPos( tu.cu->sbtInfo );
439
440
0
    if( sbtIdx == SBT_VER_HALF || sbtIdx == SBT_VER_QUAD )
441
0
    {
442
0
      assert( tu.lwidth() <= MTS_INTER_MAX_CU_SIZE );
443
0
      if( tu.lheight() > MTS_INTER_MAX_CU_SIZE )
444
0
      {
445
0
        trTypeHor = trTypeVer = DCT2;
446
0
      }
447
0
      else
448
0
      {
449
0
        if( sbtPos == SBT_POS0 )  { trTypeHor = DCT8;  trTypeVer = DST7; }
450
0
        else                      { trTypeHor = DST7;  trTypeVer = DST7; }
451
0
      }
452
0
    }
453
0
    else
454
0
    {
455
0
      assert( tu.lheight() <= MTS_INTER_MAX_CU_SIZE );
456
0
      if( tu.lwidth() > MTS_INTER_MAX_CU_SIZE )
457
0
      {
458
0
        trTypeHor = trTypeVer = DCT2;
459
0
      }
460
0
      else
461
0
      {
462
0
        if( sbtPos == SBT_POS0 )  { trTypeHor = DST7;  trTypeVer = DCT8; }
463
0
        else                      { trTypeHor = DST7;  trTypeVer = DST7; }
464
0
      }
465
0
    }
466
0
  }
467
2.50M
  const bool isExplicitMTS = (CU::isIntra(*tu.cu) ? tu.cs->sps->MTS : tu.cs->sps->MTSInter && CU::isInter(*tu.cu)) && isLuma(compID);
468
2.50M
  if (isExplicitMTS)
469
185k
  {
470
185k
    if (tu.mtsIdx[compID] > MTS_SKIP)
471
0
    {
472
0
      int indHor = (tu.mtsIdx[compID] - MTS_DST7_DST7) & 1;
473
0
      int indVer = (tu.mtsIdx[compID] - MTS_DST7_DST7) >> 1;
474
0
      trTypeHor  = indHor ? DCT8 : DST7;
475
0
      trTypeVer  = indVer ? DCT8 : DST7;
476
0
    }
477
185k
  }
478
2.50M
}
479
480
481
void TrQuant::xT( const TransformUnit& tu, const ComponentID compID, const CPelBuf& resi, CoeffBuf& dstCoeff, const int width, const int height )
482
1.78M
{
483
1.78M
  PROFILER_SCOPE_AND_STAGE( 1, _TPROF, P_TRAFO );
484
485
1.78M
  const unsigned maxLog2TrDynamicRange  = tu.cs->sps->getMaxLog2TrDynamicRange();
486
1.78M
  const unsigned bitDepth               = tu.cs->sps->bitDepths[toChannelType( compID )];
487
1.78M
  const int      TRANSFORM_MATRIX_SHIFT = g_transformMatrixShift[TRANSFORM_FORWARD];
488
1.78M
  const uint32_t transformWidthIndex    = Log2(width ) - 1;  // nLog2WidthMinus1, since transform start from 2-point
489
1.78M
  const uint32_t transformHeightIndex   = Log2(height) - 1;  // nLog2HeightMinus1, since transform start from 2-point
490
491
1.78M
  int trTypeHor = DCT2;
492
1.78M
  int trTypeVer = DCT2;
493
494
1.78M
  xSetTrTypes( tu, compID, width, height, trTypeHor, trTypeVer );
495
496
1.78M
  int  skipWidth  = ( trTypeHor != DCT2 && width  == 32 ) ? 16 : width  > JVET_C0024_ZERO_OUT_TH ? width  - JVET_C0024_ZERO_OUT_TH : 0;
497
1.78M
  int  skipHeight = ( trTypeVer != DCT2 && height == 32 ) ? 16 : height > JVET_C0024_ZERO_OUT_TH ? height - JVET_C0024_ZERO_OUT_TH : 0;
498
499
1.78M
  if( tu.cu->lfnstIdx )
500
1.11M
  {
501
1.11M
    if ((width == 4 && height > 4) || (width > 4 && height == 4))
502
318k
    {
503
318k
      skipWidth  = width - 4;
504
318k
      skipHeight = height - 4;
505
318k
    }
506
794k
    else if ((width >= 8 && height >= 8))
507
725k
    {
508
725k
      skipWidth  = width - 8;
509
725k
      skipHeight = height - 8;
510
725k
    }
511
1.11M
  }
512
513
1.78M
  TCoeff* block = m_blk;
514
1.78M
  TCoeff* tmp   = m_tmp;
515
516
1.78M
  const Pel* resiBuf    = resi.buf;
517
1.78M
  const int  resiStride = resi.stride;
518
519
1.78M
#if ENABLE_SIMD_TRAFO
520
1.78M
  if( width & 3 )
521
0
#endif
522
0
  {
523
0
    for( int y = 0; y < height; y++ )
524
0
    {
525
0
      for( int x = 0; x < width; x++ )
526
0
      {
527
0
        block[( y * width ) + x] = resiBuf[( y * resiStride ) + x];
528
0
      }
529
0
    }
530
0
  }
531
1.78M
#if ENABLE_SIMD_TRAFO
532
1.78M
  else if( width & 7 )
533
344k
  {
534
344k
    g_tCoeffOps.cpyCoeff4( resiBuf, resiStride, block, width, height );
535
344k
  }
536
1.44M
  else
537
1.44M
  {
538
1.44M
    g_tCoeffOps.cpyCoeff8( resiBuf, resiStride, block, width, height );
539
1.44M
  }
540
1.78M
#endif //ENABLE_SIMD_TRAFO
541
542
1.78M
  if (width > 1 && height > 1)
543
1.78M
  {
544
1.78M
    const int shift_1st = ((Log2(width )) + bitDepth + TRANSFORM_MATRIX_SHIFT) - maxLog2TrDynamicRange;
545
1.78M
    const int shift_2nd =  (Log2(height))            + TRANSFORM_MATRIX_SHIFT;
546
1.78M
    CHECK( shift_1st < 0, "Negative shift" );
547
1.78M
    CHECK( shift_2nd < 0, "Negative shift" );
548
1.78M
    fastFwdTrans[trTypeHor][transformWidthIndex](block, tmp, shift_1st, height, 0, skipWidth);
549
1.78M
    fastFwdTrans[trTypeVer][transformHeightIndex](tmp, dstCoeff.buf, shift_2nd, width, skipWidth, skipHeight);
550
1.78M
  }
551
177
  else if (height == 1)   // 1-D horizontal transform
552
260
  {
553
260
    const int shift = ((Log2(width )) + bitDepth + TRANSFORM_MATRIX_SHIFT) - maxLog2TrDynamicRange;
554
260
    CHECK( shift < 0, "Negative shift" );
555
260
    fastFwdTrans[trTypeHor][transformWidthIndex](block, dstCoeff.buf, shift, 1, 0, skipWidth);
556
260
  }
557
18.4E
  else   // if (iWidth == 1) //1-D vertical transform
558
18.4E
  {
559
18.4E
    int shift = ((floorLog2(height)) + bitDepth + TRANSFORM_MATRIX_SHIFT) - maxLog2TrDynamicRange;
560
18.4E
    CHECK(shift < 0, "Negative shift");
561
18.4E
    CHECKD((transformHeightIndex < 0), "There is a problem with the height.");
562
18.4E
    fastFwdTrans[trTypeVer][transformHeightIndex](block, dstCoeff.buf, shift, 1, 0, skipHeight);
563
18.4E
  }
564
1.78M
}
565
566
567
void TrQuant::xIT( const TransformUnit& tu, const ComponentID compID, const CCoeffBuf& pCoeff, PelBuf& pResidual )
568
734k
{
569
734k
  PROFILER_SCOPE_AND_STAGE( 1, _TPROF, P_TRAFO );
570
571
734k
  const int      width                  = pCoeff.width;
572
734k
  const int      height                 = pCoeff.height;
573
734k
  const unsigned maxLog2TrDynamicRange  = tu.cs->sps->getMaxLog2TrDynamicRange();
574
734k
  const unsigned bitDepth               = tu.cs->sps->bitDepths[toChannelType( compID )];
575
734k
  const int      TRANSFORM_MATRIX_SHIFT = g_transformMatrixShift[TRANSFORM_INVERSE];
576
734k
  const TCoeff   clipMinimum            = -( 1 << maxLog2TrDynamicRange );
577
734k
  const TCoeff   clipMaximum            =  ( 1 << maxLog2TrDynamicRange ) - 1;
578
734k
  const uint32_t transformWidthIndex    = Log2(width )- 1;                                // nLog2WidthMinus1, since transform start from 2-point
579
734k
  const uint32_t transformHeightIndex   = Log2(height) - 1;                                // nLog2HeightMinus1, since transform start from 2-point
580
581
582
734k
  int trTypeHor = DCT2;
583
734k
  int trTypeVer = DCT2;
584
585
734k
  xSetTrTypes( tu, compID, width, height, trTypeHor, trTypeVer );
586
587
734k
  int skipWidth  = ( trTypeHor != DCT2 && width  == 32 ) ? 16 : width  > JVET_C0024_ZERO_OUT_TH ? width  - JVET_C0024_ZERO_OUT_TH : 0;
588
734k
  int skipHeight = ( trTypeVer != DCT2 && height == 32 ) ? 16 : height > JVET_C0024_ZERO_OUT_TH ? height - JVET_C0024_ZERO_OUT_TH : 0;
589
590
734k
  if (tu.cs->sps->LFNST && tu.cu->lfnstIdx)
591
475k
  {
592
475k
    if ((width == 4 && height > 4) || (width > 4 && height == 4))
593
137k
    {
594
137k
      skipWidth = width - 4;
595
137k
      skipHeight = height - 4;
596
137k
    }
597
338k
    else if ((width >= 8 && height >= 8))
598
300k
    {
599
300k
      skipWidth = width - 8;
600
300k
      skipHeight = height - 8;
601
300k
    }
602
475k
  }
603
604
734k
  TCoeff *block = m_blk;
605
734k
  TCoeff *tmp   = m_tmp;
606
734k
  if (width > 1 && height > 1)   // 2-D transform
607
734k
  {
608
734k
    const int shift_1st =   TRANSFORM_MATRIX_SHIFT + 1; // 1 has been added to shift_1st at the expense of shift_2nd
609
734k
    const int shift_2nd = ( TRANSFORM_MATRIX_SHIFT + maxLog2TrDynamicRange - 1 ) - bitDepth;
610
734k
    CHECK( shift_1st < 0, "Negative shift" );
611
734k
    CHECK( shift_2nd < 0, "Negative shift" );
612
734k
    fastInvTrans[trTypeVer][transformHeightIndex](pCoeff.buf, tmp, shift_1st, width, skipWidth, skipHeight, clipMinimum, clipMaximum);
613
734k
    fastInvTrans[trTypeHor][transformWidthIndex](tmp, block, shift_2nd, height, 0, skipWidth, clipMinimum, clipMaximum);
614
734k
  }
615
65
  else if (width == 1)   // 1-D vertical transform
616
0
  {
617
0
    int shift = (TRANSFORM_MATRIX_SHIFT + maxLog2TrDynamicRange - 1) - bitDepth;
618
0
    CHECK(shift < 0, "Negative shift");
619
0
    fastInvTrans[trTypeVer][transformHeightIndex](pCoeff.buf, block, shift + 1, 1, 0, skipHeight, clipMinimum, clipMaximum);
620
0
  }
621
65
  else   // if(iHeight == 1) //1-D horizontal transform
622
65
  {
623
65
    const int shift = (TRANSFORM_MATRIX_SHIFT + maxLog2TrDynamicRange - 1) - bitDepth;
624
65
    CHECK(shift < 0, "Negative shift");
625
65
    fastInvTrans[trTypeHor][transformWidthIndex](pCoeff.buf, block, shift + 1, 1, 0, skipWidth, clipMinimum, clipMaximum);
626
65
  }
627
628
734k
#if ENABLE_SIMD_TRAFO
629
734k
  if( width & 3 )
630
0
#endif //ENABLE_SIMD_TRAFO
631
0
  {
632
0
    Pel       *dst    = pResidual.buf;
633
0
    ptrdiff_t  stride = pResidual.stride;
634
635
0
    for( int y = 0; y < height; y++ )
636
0
    {
637
0
      for( int x = 0; x < width; x++ )
638
0
      {
639
0
        dst[x] = ( Pel ) *block++;
640
0
      }
641
642
0
      dst += stride;
643
0
    }
644
0
  }
645
734k
#if ENABLE_SIMD_TRAFO
646
734k
  else if( width & 7 )
647
159k
  {
648
159k
    g_tCoeffOps.cpyResi4( block, pResidual.buf, pResidual.stride, width, height );
649
159k
  }
650
574k
  else
651
574k
  {
652
574k
    g_tCoeffOps.cpyResi8( block, pResidual.buf, pResidual.stride, width, height );
653
574k
  }
654
734k
#endif //ENABLE_SIMD_TRAFO
655
734k
}
656
657
/** Wrapper function between HM interface and core NxN transform skipping
658
 */
659
void TrQuant::xITransformSkip(const CCoeffBuf& pCoeff,
660
  PelBuf& pResidual,
661
  const TransformUnit& tu,
662
  const ComponentID compID)
663
45.8k
{
664
45.8k
  const CompArea& area = tu.blocks[compID];
665
45.8k
  const int width = area.width;
666
45.8k
  const int height = area.height;
667
668
459k
  for (uint32_t y = 0; y < height; y++)
669
414k
  {
670
4.58M
    for (uint32_t x = 0; x < width; x++)
671
4.16M
    {
672
4.16M
      pResidual.at(x, y) = Pel(pCoeff.at(x, y));
673
4.16M
    }
674
414k
  }
675
45.8k
}
676
677
void TrQuant::xQuant(TransformUnit& tu, const ComponentID compID, const CCoeffBuf& pSrc, TCoeff &uiAbsSum, const QpParam& cQP, const Ctx& ctx)
678
1.88M
{
679
1.88M
  PROFILER_SCOPE_AND_STAGE( 1, _TPROF, P_QUANT );
680
1.88M
  m_quant->quant( tu, compID, pSrc, uiAbsSum, cQP, ctx );
681
#if ENABLE_MEASURE_SEARCH_SPACE
682
683
  g_searchSpaceAcc.addQuant( tu, toChannelType( compID ) );
684
#endif
685
1.88M
}
686
687
688
void TrQuant::transformNxN(TransformUnit &tu, const ComponentID compID, const QpParam &cQP, TCoeff &uiAbsSum, const Ctx &ctx, const bool loadTr)
689
1.88M
{
690
1.88M
        CodingStructure &cs = *tu.cs;
691
1.88M
  const CompArea& rect      = tu.blocks[compID];
692
1.88M
  const uint32_t uiWidth        = rect.width;
693
1.88M
  const uint32_t uiHeight       = rect.height;
694
695
1.88M
  const CPelBuf resiBuf     = cs.getResiBuf(rect);
696
697
1.88M
  if( tu.noResidual )
698
0
  {
699
0
    uiAbsSum = 0;
700
0
    TU::setCbfAtDepth( tu, compID, tu.depth, uiAbsSum > 0 );
701
0
    return;
702
0
  }
703
1.88M
  if (tu.cu->bdpcmM[toChannelType(compID)])
704
94.2k
  {
705
94.2k
    tu.mtsIdx[compID] = MTS_SKIP;
706
94.2k
  }
707
708
1.88M
  uiAbsSum = 0;
709
1.88M
  CHECK( cs.sps->getMaxTbSize() < uiWidth, "Unsupported transformation size" );
710
711
1.88M
  CoeffBuf tempCoeff(loadTr ? m_mtsCoeffs[tu.mtsIdx[compID]] : m_plTempCoeff, rect);
712
1.88M
  if (!loadTr)
713
1.86M
  {
714
1.86M
    DTRACE_PEL_BUF( D_RESIDUALS, resiBuf, tu, tu.cu->predMode, compID );
715
1.86M
    if (tu.mtsIdx[compID] == MTS_SKIP)
716
94.2k
    {
717
94.2k
      xTransformSkip(tu, compID, resiBuf, tempCoeff.buf);
718
94.2k
    }
719
1.76M
    else
720
1.76M
    {
721
1.76M
      xT(tu, compID, resiBuf, tempCoeff, uiWidth, uiHeight);
722
1.76M
    }
723
1.86M
  }
724
1.88M
  if (cs.sps->LFNST)
725
1.88M
  {
726
1.88M
    xFwdLfnst(tu, compID, loadTr);
727
1.88M
  }
728
1.88M
  DTRACE_COEFF_BUF( D_TCOEFF, tempCoeff, tu, tu.cu->predMode, compID );
729
730
1.88M
  xQuant( tu, compID, tempCoeff, uiAbsSum, cQP, ctx );
731
732
1.88M
  DTRACE_COEFF_BUF( D_TCOEFF, tu.getCoeffs( compID ), tu, tu.cu->predMode, compID );
733
734
  // set coded block flag (CBF)
735
1.88M
  TU::setCbfAtDepth (tu, compID, tu.depth, uiAbsSum > 0);
736
1.88M
}
737
738
void TrQuant::checktransformsNxN( TransformUnit &tu, std::vector<TrMode> *trModes, const int maxCand, const ComponentID compID)
739
16.7k
{
740
16.7k
  CodingStructure &cs     = *tu.cs;
741
16.7k
  const CompArea& rect    = tu.blocks[compID];
742
16.7k
  const uint32_t   width  = rect.width;
743
16.7k
  const uint32_t   height = rect.height;
744
745
16.7k
  const CPelBuf resiBuf = cs.getResiBuf(rect);
746
747
16.7k
  CHECK(cs.sps->getMaxTbSize() < width, "Unsupported transformation size");
748
16.7k
  int                           pos = 0;
749
16.7k
  std::vector<TrCost>           trCosts;
750
16.7k
  std::vector<TrMode>::iterator it      = trModes->begin();
751
16.7k
  const double                  facBB[] = { 1.2, 1.3, 1.3, 1.4, 1.5 };
752
50.2k
  while (it != trModes->end())
753
33.4k
  {
754
33.4k
    tu.mtsIdx[compID] = it->first;
755
33.4k
    CoeffBuf tempCoeff(m_mtsCoeffs[tu.mtsIdx[compID]], rect);
756
33.4k
    if (tu.noResidual)
757
0
    {
758
0
      int sumAbs = 0;
759
0
      trCosts.push_back(TrCost(sumAbs, pos++));
760
0
      it++;
761
0
      continue;
762
0
    }
763
33.4k
    if (tu.mtsIdx[compID] == MTS_SKIP)
764
16.7k
    {
765
16.7k
      xTransformSkip(tu, compID, resiBuf, tempCoeff.buf);
766
16.7k
    }
767
16.7k
    else
768
16.7k
    {
769
16.7k
      xT(tu, compID, resiBuf, tempCoeff, width, height);
770
16.7k
    }
771
772
33.4k
    int sumAbs = 0;
773
5.96M
    for (int pos = 0; pos < width * height; pos++)
774
5.93M
    {
775
5.93M
      sumAbs += abs(tempCoeff.buf[pos]);
776
5.93M
    }
777
778
33.4k
    double scaleSAD = 1.0;
779
33.4k
    if (tu.mtsIdx[compID] == MTS_SKIP && ((floorLog2(width) + floorLog2(height)) & 1) == 1)
780
7.01k
    {
781
7.01k
      scaleSAD = 1.0 / 1.414213562;   // compensate for not scaling transform skip coefficients by 1/sqrt(2)
782
7.01k
    }
783
33.4k
    if (tu.mtsIdx[compID] == MTS_SKIP)
784
16.7k
    {
785
16.7k
      int trShift = getTransformShift(tu.cu->slice->sps->bitDepths[CH_L], rect.size(), tu.cu->slice->sps->getMaxLog2TrDynamicRange());
786
16.7k
      scaleSAD *= pow(2, trShift);
787
16.7k
    }
788
33.4k
    trCosts.push_back(TrCost(int(std::min<double>(sumAbs * scaleSAD, std::numeric_limits<int>::max())), pos++));
789
33.4k
    it++;
790
33.4k
  }
791
792
16.7k
  int                           numTests = 0;
793
16.7k
  std::vector<TrCost>::iterator itC      = trCosts.begin();
794
16.7k
  const double                  fac      = facBB[std::max(0, floorLog2(std::max(width, height)) - 2)];
795
16.7k
  const double                  thr      = fac * trCosts.begin()->first;
796
16.7k
  const double                  thrTS    = trCosts.begin()->first;
797
50.2k
  while (itC != trCosts.end())
798
33.4k
  {
799
33.4k
    const bool testTr               = itC->first <= (trModes->at(itC->second).first == 1 ? thrTS : thr) && numTests <= maxCand;
800
33.4k
    trModes->at(itC->second).second = testTr;
801
33.4k
    numTests += testTr;
802
33.4k
    itC++;
803
33.4k
  }
804
16.7k
}
805
806
uint32_t TrQuant::xGetLFNSTIntraMode( const Area& tuArea, const uint32_t dirMode )
807
1.58M
{
808
1.58M
  if (dirMode < 2)
809
894k
  {
810
894k
    return dirMode;
811
894k
  }
812
813
693k
  static const int modeShift[] = { 0, 6, 10, 12, 14, 15 };
814
815
693k
  const int width  = int(tuArea.width);
816
693k
  const int height = int(tuArea.height);
817
818
693k
  if (width > height && dirMode < 2 + modeShift[floorLog2(width) - floorLog2(height)])
819
192
  {
820
192
    return dirMode + (VDIA_IDX - 1) + (NUM_EXT_LUMA_MODE >> 1);
821
192
  }
822
693k
  else if (height > width && dirMode > VDIA_IDX - modeShift[floorLog2(height) - floorLog2(width)])
823
66.3k
  {
824
66.3k
    return dirMode - (VDIA_IDX + 1) + (NUM_EXT_LUMA_MODE >> 1) + NUM_LUMA_MODE;
825
66.3k
  }
826
827
627k
  return dirMode;
828
693k
}
829
830
831
bool TrQuant::xGetTransposeFlag(uint32_t intraMode)
832
1.58M
{
833
1.58M
  return ((intraMode >= NUM_LUMA_MODE) && (intraMode >= (NUM_LUMA_MODE + (NUM_EXT_LUMA_MODE >> 1))))
834
1.58M
         || ((intraMode < NUM_LUMA_MODE) && (intraMode > DIA_IDX));
835
1.58M
}
836
837
838
void TrQuant::xInvLfnst(const TransformUnit &tu, const ComponentID compID)
839
780k
{
840
780k
  const CompArea &area     = tu.blocks[compID];
841
780k
  const uint32_t  width    = area.width;
842
780k
  const uint32_t  height   = area.height;
843
780k
  const uint32_t  lfnstIdx = tu.cu->lfnstIdx;
844
780k
  if (lfnstIdx && tu.mtsIdx[compID] != MTS_SKIP && (CU::isSepTree(*tu.cu) ? true : isLuma(compID)))
845
475k
  {
846
475k
    const CodingUnit& cu = *tu.cs->getCU(area.pos(), toChannelType(compID), TREE_D);
847
475k
    const bool         whge3 = width >= 8 && height >= 8;
848
475k
    const ScanElement *scan =
849
475k
      whge3
850
475k
        ? g_coefTopLeftDiagScan8x8[Log2(width)] 
851
475k
        : getScanOrder(SCAN_GROUPED_4x4, Log2(area.width), Log2(area.height));
852
475k
    uint32_t intraMode = CU::getFinalIntraMode(cu, toChannelType(compID));
853
854
475k
    if (CU::isLMCMode( cu.intraDir[toChannelType(compID)]))
855
78.1k
    {
856
78.1k
      intraMode = CU::getCoLocatedIntraLumaMode(cu);
857
78.1k
    }
858
475k
    if (CU::isMIP(cu, toChannelType(compID)))
859
4.62k
    {
860
4.62k
      intraMode = PLANAR_IDX;
861
4.62k
    }
862
475k
    CHECK(intraMode >= NUM_INTRA_MODE - 1, "Invalid intra mode");
863
864
475k
    if (lfnstIdx < 3)
865
475k
    {
866
475k
      if (tu.cu->ispMode && isLuma(compID))
867
6.94k
      {
868
6.94k
        intraMode = xGetLFNSTIntraMode(tu.cu->blocks[compID], intraMode);
869
6.94k
      }
870
468k
      else
871
468k
        intraMode = xGetLFNSTIntraMode(tu.blocks[compID], intraMode);
872
475k
      bool      transposeFlag = xGetTransposeFlag(intraMode);
873
475k
      const int sbSize        = whge3 ? 8 : 4;
874
475k
      bool      tu4x4Flag     = (width == 4 && height == 4);
875
475k
      bool      tu8x8Flag     = (width == 8 && height == 8);
876
475k
      TCoeff *  lfnstTemp;
877
475k
      TCoeff *  coeffTemp;
878
475k
      int       y;
879
475k
      lfnstTemp                  = m_tempInMatrix;   // inverse spectral rearrangement
880
475k
      coeffTemp                  = m_plTempCoeff;
881
475k
      TCoeff *           dst     = lfnstTemp;
882
475k
      const ScanElement *scanPtr = scan;
883
8.08M
      for (y = 0; y < 16; y++)
884
7.60M
      {
885
7.60M
        *dst++ = coeffTemp[scanPtr->idx];
886
7.60M
        scanPtr++;
887
7.60M
      }
888
889
475k
      m_invLfnstNxN( m_tempInMatrix, m_tempOutMatrix, g_lfnstLut[intraMode], lfnstIdx - 1, sbSize, ( tu4x4Flag || tu8x8Flag ) ? 8 : 16 );
890
891
475k
      lfnstTemp = m_tempOutMatrix;   // inverse spectral rearrangement
892
893
475k
      if (transposeFlag)
894
43.9k
      {
895
43.9k
        if (sbSize == 4)
896
8.63k
        {
897
43.1k
          for (y = 0; y < 4; y++)
898
34.5k
          {
899
34.5k
            coeffTemp[0] = lfnstTemp[0];
900
34.5k
            coeffTemp[1] = lfnstTemp[4];
901
34.5k
            coeffTemp[2] = lfnstTemp[8];
902
34.5k
            coeffTemp[3] = lfnstTemp[12];
903
34.5k
            lfnstTemp++;
904
34.5k
            coeffTemp += width;
905
34.5k
          }
906
8.63k
        }
907
35.2k
        else   // ( sbSize == 8 )
908
35.2k
        {
909
317k
          for (y = 0; y < 8; y++)
910
282k
          {
911
282k
            coeffTemp[0] = lfnstTemp[0];
912
282k
            coeffTemp[1] = lfnstTemp[8];
913
282k
            coeffTemp[2] = lfnstTemp[16];
914
282k
            coeffTemp[3] = lfnstTemp[24];
915
282k
            if (y < 4)
916
141k
            {
917
141k
              coeffTemp[4] = lfnstTemp[32];
918
141k
              coeffTemp[5] = lfnstTemp[36];
919
141k
              coeffTemp[6] = lfnstTemp[40];
920
141k
              coeffTemp[7] = lfnstTemp[44];
921
141k
            }
922
282k
            lfnstTemp++;
923
282k
            coeffTemp += width;
924
282k
          }
925
35.2k
        }
926
43.9k
      }
927
431k
      else
928
431k
      {
929
3.21M
        for (y = 0; y < sbSize; y++)
930
2.78M
        {
931
2.78M
          uint32_t uiStride = (y < 4) ? sbSize : 4;
932
2.78M
          ::memcpy(coeffTemp, lfnstTemp, uiStride * sizeof(TCoeff));
933
2.78M
          lfnstTemp += uiStride;
934
2.78M
          coeffTemp += width;
935
2.78M
        }
936
431k
      }
937
475k
    }
938
475k
  }
939
780k
}
940
941
942
void TrQuant::xFwdLfnst(const TransformUnit &tu, const ComponentID compID, const bool loadTr)
943
1.88M
{
944
1.88M
  const CompArea &area     = tu.blocks[compID];
945
1.88M
  const uint32_t  width    = area.width;
946
1.88M
  const uint32_t  height   = area.height;
947
1.88M
  const uint32_t  lfnstIdx = tu.cu->lfnstIdx;
948
1.88M
  if (lfnstIdx && tu.mtsIdx[compID] != MTS_SKIP && (CU::isSepTree(*tu.cu) ? true : isLuma(compID)))
949
1.11M
  {
950
1.11M
    const CodingUnit& cu = *tu.cs->getCU(area.pos(), toChannelType(compID), TREE_D);
951
1.11M
    const bool         whge3 = width >= 8 && height >= 8;
952
1.11M
    const ScanElement *scan =
953
1.11M
      whge3
954
1.11M
        ? g_coefTopLeftDiagScan8x8[Log2(width)] 
955
1.11M
        : getScanOrder(SCAN_GROUPED_4x4, Log2(area.width), Log2(area.height));   
956
1.11M
    uint32_t intraMode = CU::getFinalIntraMode(cu, toChannelType(compID));
957
958
1.11M
    if (CU::isLMCMode(cu.intraDir[toChannelType(compID)]))
959
99.2k
    {
960
99.2k
      intraMode = CU::getCoLocatedIntraLumaMode(cu);
961
99.2k
    }
962
1.11M
    if (CU::isMIP(cu, toChannelType(compID)))
963
7.67k
    {
964
7.67k
      intraMode = PLANAR_IDX;
965
7.67k
    }
966
1.11M
    CHECK(intraMode >= NUM_INTRA_MODE - 1, "Invalid intra mode");
967
968
1.11M
    if (lfnstIdx < 3)
969
1.11M
    {
970
1.11M
      if (tu.cu->ispMode && isLuma(compID))
971
10.9k
      {
972
10.9k
        intraMode = xGetLFNSTIntraMode(tu.cu->blocks[compID], intraMode);
973
10.9k
      }
974
1.10M
      else
975
1.10M
      {
976
1.10M
        intraMode = xGetLFNSTIntraMode(tu.blocks[compID], intraMode);
977
1.10M
      }
978
1.11M
      bool      transposeFlag = xGetTransposeFlag(intraMode);
979
1.11M
      const int sbSize        = whge3 ? 8 : 4;
980
1.11M
      bool      tu4x4Flag     = (width == 4 && height == 4);
981
1.11M
      bool      tu8x8Flag     = (width == 8 && height == 8);
982
1.11M
      TCoeff*   lfnstTemp;
983
1.11M
      TCoeff*   coeffTemp;
984
1.11M
      TCoeff*   tempCoeff     = loadTr ? m_mtsCoeffs[tu.mtsIdx[compID]] : m_plTempCoeff;
985
986
1.11M
      int y;
987
1.11M
      lfnstTemp = m_tempInMatrix;   // forward low frequency non-separable transform
988
1.11M
      coeffTemp = tempCoeff;
989
990
1.11M
      if (transposeFlag)
991
235k
      {
992
235k
        if (sbSize == 4)
993
70.0k
        {
994
350k
          for (y = 0; y < 4; y++)
995
280k
          {
996
280k
            lfnstTemp[0]  = coeffTemp[0];
997
280k
            lfnstTemp[4]  = coeffTemp[1];
998
280k
            lfnstTemp[8]  = coeffTemp[2];
999
280k
            lfnstTemp[12] = coeffTemp[3];
1000
280k
            lfnstTemp++;
1001
280k
            coeffTemp += width;
1002
280k
          }
1003
70.0k
        }
1004
165k
        else   // ( sbSize == 8 )
1005
165k
        {
1006
1.49M
          for (y = 0; y < 8; y++)
1007
1.32M
          {
1008
1.32M
            lfnstTemp[0]  = coeffTemp[0];
1009
1.32M
            lfnstTemp[8]  = coeffTemp[1];
1010
1.32M
            lfnstTemp[16] = coeffTemp[2];
1011
1.32M
            lfnstTemp[24] = coeffTemp[3];
1012
1.32M
            if (y < 4)
1013
662k
            {
1014
662k
              lfnstTemp[32] = coeffTemp[4];
1015
662k
              lfnstTemp[36] = coeffTemp[5];
1016
662k
              lfnstTemp[40] = coeffTemp[6];
1017
662k
              lfnstTemp[44] = coeffTemp[7];
1018
662k
            }
1019
1.32M
            lfnstTemp++;
1020
1.32M
            coeffTemp += width;
1021
1.32M
          }
1022
165k
        }
1023
235k
      }
1024
877k
      else
1025
877k
      {
1026
6.62M
        for (y = 0; y < sbSize; y++)
1027
5.74M
        {
1028
5.74M
          uint32_t uiStride = (y < 4) ? sbSize : 4;
1029
5.74M
          ::memcpy(lfnstTemp, coeffTemp, uiStride * sizeof(TCoeff));
1030
5.74M
          lfnstTemp += uiStride;
1031
5.74M
          coeffTemp += width;
1032
5.74M
        }
1033
877k
      }
1034
1035
1.11M
      m_fwdLfnstNxN( m_tempInMatrix, m_tempOutMatrix, g_lfnstLut[intraMode], lfnstIdx - 1, sbSize, ( tu4x4Flag || tu8x8Flag ) ? 8 : 16 );
1036
1037
1.11M
      lfnstTemp                        = m_tempOutMatrix;   // forward spectral rearrangement
1038
1.11M
      coeffTemp                        = tempCoeff;
1039
1.11M
      const ScanElement *scanPtr       = scan;
1040
1.11M
      int                lfnstCoeffNum = (sbSize == 4) ? sbSize * sbSize : 48;
1041
42.1M
      for (y = 0; y < lfnstCoeffNum; y++)
1042
41.0M
      {
1043
41.0M
        coeffTemp[scanPtr->idx] = *lfnstTemp++;
1044
41.0M
        scanPtr++;
1045
41.0M
      }
1046
1.11M
    }
1047
1.11M
  }
1048
1.88M
}
1049
1050
void TrQuant::xTransformSkip(const TransformUnit& tu, const ComponentID& compID, const CPelBuf& resi, TCoeff* psCoeff)
1051
111k
{
1052
111k
  const CompArea& rect = tu.blocks[compID];
1053
111k
  const uint32_t width = rect.width;
1054
111k
  const uint32_t height = rect.height;
1055
1056
1.21M
  for (uint32_t y = 0, coefficientIndex = 0; y < height; y++)
1057
1.09M
  {
1058
13.1M
    for (uint32_t x = 0; x < width; x++, coefficientIndex++)
1059
12.0M
    {
1060
12.0M
      psCoeff[coefficientIndex] = TCoeff(resi.at(x, y));
1061
12.0M
    }
1062
1.09M
  }
1063
111k
}
1064
} // namespace vvenc
1065
1066
//! \}
1067