Coverage Report

Created: 2026-09-14 06:44

next uncovered line (L), next uncovered region (R), next uncovered branch (B)
/work/vvenc/source/Lib/CommonLib/TrQuant.cpp
Line
Count
Source
1
/* -----------------------------------------------------------------------------
2
The copyright in this software is being made available under the Clear BSD
3
License, included below. No patent rights, trademark rights and/or 
4
other Intellectual Property Rights other than the copyrights concerning 
5
the Software are granted under this license.
6
7
The Clear BSD License
8
9
Copyright (c) 2019-2026, Fraunhofer-Gesellschaft zur Förderung der angewandten Forschung e.V. & The VVenC Authors.
10
All rights reserved.
11
12
Redistribution and use in source and binary forms, with or without modification,
13
are permitted (subject to the limitations in the disclaimer below) provided that
14
the following conditions are met:
15
16
     * Redistributions of source code must retain the above copyright notice,
17
     this list of conditions and the following disclaimer.
18
19
     * Redistributions in binary form must reproduce the above copyright
20
     notice, this list of conditions and the following disclaimer in the
21
     documentation and/or other materials provided with the distribution.
22
23
     * Neither the name of the copyright holder nor the names of its
24
     contributors may be used to endorse or promote products derived from this
25
     software without specific prior written permission.
26
27
NO EXPRESS OR IMPLIED LICENSES TO ANY PARTY'S PATENT RIGHTS ARE GRANTED BY
28
THIS LICENSE. THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND
29
CONTRIBUTORS "AS IS" AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT
30
LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A
31
PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR
32
CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL,
33
EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO,
34
PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR
35
BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER
36
IN CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE)
37
ARISING IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE
38
POSSIBILITY OF SUCH DAMAGE.
39
40
41
------------------------------------------------------------------------------------------- */
42
43
44
/** \file     TrQuant.cpp
45
    \brief    transform and quantization class
46
*/
47
48
#include "TrQuant.h"
49
#include "TrQuant_EMT.h"
50
#include "QuantRDOQ.h"
51
#include "DepQuant.h"
52
#include "UnitTools.h"
53
#include "ContextModelling.h"
54
#include "CodingStructure.h"
55
#include "dtrace_buffer.h"
56
#include "TimeProfiler.h"
57
#include "SearchSpaceCounter.h"
58
59
#include <stdlib.h>
60
#include <memory.h>
61
62
//! \ingroup CommonLib
63
//! \{
64
65
namespace vvenc {
66
67
struct coeffGroupRDStats
68
{
69
  int    iNNZbeforePos0;
70
  double d64CodedLevelandDist; // distortion and level cost only
71
  double d64UncodedDist;    // all zero coded block distortion
72
  double d64SigCost;
73
  double d64SigCost_0;
74
};
75
76
FwdTrans *const fastFwdTrans[NUM_TRANS_TYPE][g_numTransformMatrixSizes] =
77
{
78
  { fastForwardDCT2_B2, fastForwardDCT2_B4, fastForwardDCT2_B8, fastForwardDCT2_B16, fastForwardDCT2_B32, fastForwardDCT2_B64 },
79
  { nullptr,            fastForwardDCT8_B4, fastForwardDCT8_B8, fastForwardDCT8_B16, fastForwardDCT8_B32, nullptr },
80
  { nullptr,            fastForwardDST7_B4, fastForwardDST7_B8, fastForwardDST7_B16, fastForwardDST7_B32, nullptr },
81
};
82
83
InvTrans *const fastInvTrans[NUM_TRANS_TYPE][g_numTransformMatrixSizes] =
84
{
85
  { fastInverseDCT2_B2, fastInverseDCT2_B4, fastInverseDCT2_B8, fastInverseDCT2_B16, fastInverseDCT2_B32, fastInverseDCT2_B64 },
86
  { nullptr,            fastInverseDCT8_B4, fastInverseDCT8_B8, fastInverseDCT8_B16, fastInverseDCT8_B32, nullptr },
87
  { nullptr,            fastInverseDST7_B4, fastInverseDST7_B8, fastInverseDST7_B16, fastInverseDST7_B32, nullptr },
88
};
89
90
//! \ingroup CommonLib
91
//! \{
92
93
353M
static inline int64_t square( const int d ) { return d * (int64_t)d; }
94
95
template<int signedMode> std::pair<int64_t,int64_t> fwdTransformCbCr( const PelBuf& resCb, const PelBuf& resCr, PelBuf& resC1, PelBuf& resC2 )
96
1.05M
{
97
1.05M
  const Pel*  cb  = resCb.buf;
98
1.05M
  const Pel*  cr  = resCr.buf;
99
1.05M
  Pel*        c1  = resC1.buf;
100
1.05M
  Pel*        c2  = resC2.buf;
101
1.05M
  int64_t     d1  = 0;
102
1.05M
  int64_t     d2  = 0;
103
14.2M
  for( SizeType y = 0; y < resCb.height; y++, cb += resCb.stride, cr += resCr.stride, c1 += resC1.stride, c2 += resC2.stride )
104
13.1M
  {
105
189M
    for( SizeType x = 0; x < resCb.width; x++ )
106
176M
    {
107
176M
      int cbx = cb[x], crx = cr[x];
108
176M
      if      ( signedMode ==  1 )
109
44.1M
      {
110
44.1M
        c1[x] = Pel( ( 4*cbx + 2*crx ) / 5 );
111
44.1M
        d1   += square( cbx - c1[x] ) + square( crx - (c1[x]>>1) );
112
44.1M
      }
113
132M
      else if ( signedMode == -1 )
114
0
      {
115
0
        c1[x] = Pel( ( 4*cbx - 2*crx ) / 5 );
116
0
        d1   += square( cbx - c1[x] ) + square( crx - (-c1[x]>>1) );
117
0
      }
118
132M
      else if ( signedMode ==  2 )
119
44.1M
      {
120
44.1M
        c1[x] = Pel( ( cbx + crx ) / 2 );
121
44.1M
        d1   += square( cbx - c1[x] ) + square( crx - c1[x] );
122
44.1M
      }
123
88.3M
      else if ( signedMode == -2 )
124
0
      {
125
0
        c1[x] = Pel( ( cbx - crx ) / 2 );
126
0
        d1   += square( cbx - c1[x] ) + square( crx + c1[x] );
127
0
      }
128
88.3M
      else if ( signedMode ==  3 )
129
44.1M
      {
130
44.1M
        c2[x] = Pel( ( 4*crx + 2*cbx ) / 5 );
131
44.1M
        d1   += square( cbx - (c2[x]>>1) ) + square( crx - c2[x] );
132
44.1M
      }
133
44.1M
      else if ( signedMode == -3 )
134
0
      {
135
0
        c2[x] = Pel( ( 4*crx - 2*cbx ) / 5 );
136
0
        d1   += square( cbx - (-c2[x]>>1) ) + square( crx - c2[x] );
137
0
      }
138
44.1M
      else
139
44.1M
      {
140
44.1M
        d1   += square( cbx );
141
44.1M
        d2   += square( crx );
142
44.1M
      }
143
176M
    }
144
13.1M
  }
145
1.05M
  return std::make_pair(d1,d2);
146
1.05M
}
std::__1::pair<long, long> vvenc::fwdTransformCbCr<0>(vvenc::AreaBuf<short> const&, vvenc::AreaBuf<short> const&, vvenc::AreaBuf<short>&, vvenc::AreaBuf<short>&)
Line
Count
Source
96
263k
{
97
263k
  const Pel*  cb  = resCb.buf;
98
263k
  const Pel*  cr  = resCr.buf;
99
263k
  Pel*        c1  = resC1.buf;
100
263k
  Pel*        c2  = resC2.buf;
101
263k
  int64_t     d1  = 0;
102
263k
  int64_t     d2  = 0;
103
3.55M
  for( SizeType y = 0; y < resCb.height; y++, cb += resCb.stride, cr += resCr.stride, c1 += resC1.stride, c2 += resC2.stride )
104
3.29M
  {
105
47.4M
    for( SizeType x = 0; x < resCb.width; x++ )
106
44.1M
    {
107
44.1M
      int cbx = cb[x], crx = cr[x];
108
44.1M
      if      ( signedMode ==  1 )
109
0
      {
110
0
        c1[x] = Pel( ( 4*cbx + 2*crx ) / 5 );
111
0
        d1   += square( cbx - c1[x] ) + square( crx - (c1[x]>>1) );
112
0
      }
113
44.1M
      else if ( signedMode == -1 )
114
0
      {
115
0
        c1[x] = Pel( ( 4*cbx - 2*crx ) / 5 );
116
0
        d1   += square( cbx - c1[x] ) + square( crx - (-c1[x]>>1) );
117
0
      }
118
44.1M
      else if ( signedMode ==  2 )
119
0
      {
120
0
        c1[x] = Pel( ( cbx + crx ) / 2 );
121
0
        d1   += square( cbx - c1[x] ) + square( crx - c1[x] );
122
0
      }
123
44.1M
      else if ( signedMode == -2 )
124
0
      {
125
0
        c1[x] = Pel( ( cbx - crx ) / 2 );
126
0
        d1   += square( cbx - c1[x] ) + square( crx + c1[x] );
127
0
      }
128
44.1M
      else if ( signedMode ==  3 )
129
0
      {
130
0
        c2[x] = Pel( ( 4*crx + 2*cbx ) / 5 );
131
0
        d1   += square( cbx - (c2[x]>>1) ) + square( crx - c2[x] );
132
0
      }
133
44.1M
      else if ( signedMode == -3 )
134
0
      {
135
0
        c2[x] = Pel( ( 4*crx - 2*cbx ) / 5 );
136
0
        d1   += square( cbx - (-c2[x]>>1) ) + square( crx - c2[x] );
137
0
      }
138
44.1M
      else
139
44.1M
      {
140
44.1M
        d1   += square( cbx );
141
44.1M
        d2   += square( crx );
142
44.1M
      }
143
44.1M
    }
144
3.29M
  }
145
263k
  return std::make_pair(d1,d2);
146
263k
}
std::__1::pair<long, long> vvenc::fwdTransformCbCr<1>(vvenc::AreaBuf<short> const&, vvenc::AreaBuf<short> const&, vvenc::AreaBuf<short>&, vvenc::AreaBuf<short>&)
Line
Count
Source
96
263k
{
97
263k
  const Pel*  cb  = resCb.buf;
98
263k
  const Pel*  cr  = resCr.buf;
99
263k
  Pel*        c1  = resC1.buf;
100
263k
  Pel*        c2  = resC2.buf;
101
263k
  int64_t     d1  = 0;
102
263k
  int64_t     d2  = 0;
103
3.55M
  for( SizeType y = 0; y < resCb.height; y++, cb += resCb.stride, cr += resCr.stride, c1 += resC1.stride, c2 += resC2.stride )
104
3.29M
  {
105
47.4M
    for( SizeType x = 0; x < resCb.width; x++ )
106
44.1M
    {
107
44.1M
      int cbx = cb[x], crx = cr[x];
108
44.1M
      if      ( signedMode ==  1 )
109
44.1M
      {
110
44.1M
        c1[x] = Pel( ( 4*cbx + 2*crx ) / 5 );
111
44.1M
        d1   += square( cbx - c1[x] ) + square( crx - (c1[x]>>1) );
112
44.1M
      }
113
0
      else if ( signedMode == -1 )
114
0
      {
115
0
        c1[x] = Pel( ( 4*cbx - 2*crx ) / 5 );
116
0
        d1   += square( cbx - c1[x] ) + square( crx - (-c1[x]>>1) );
117
0
      }
118
0
      else if ( signedMode ==  2 )
119
0
      {
120
0
        c1[x] = Pel( ( cbx + crx ) / 2 );
121
0
        d1   += square( cbx - c1[x] ) + square( crx - c1[x] );
122
0
      }
123
0
      else if ( signedMode == -2 )
124
0
      {
125
0
        c1[x] = Pel( ( cbx - crx ) / 2 );
126
0
        d1   += square( cbx - c1[x] ) + square( crx + c1[x] );
127
0
      }
128
0
      else if ( signedMode ==  3 )
129
0
      {
130
0
        c2[x] = Pel( ( 4*crx + 2*cbx ) / 5 );
131
0
        d1   += square( cbx - (c2[x]>>1) ) + square( crx - c2[x] );
132
0
      }
133
0
      else if ( signedMode == -3 )
134
0
      {
135
0
        c2[x] = Pel( ( 4*crx - 2*cbx ) / 5 );
136
0
        d1   += square( cbx - (-c2[x]>>1) ) + square( crx - c2[x] );
137
0
      }
138
0
      else
139
0
      {
140
0
        d1   += square( cbx );
141
0
        d2   += square( crx );
142
0
      }
143
44.1M
    }
144
3.29M
  }
145
263k
  return std::make_pair(d1,d2);
146
263k
}
Unexecuted instantiation: std::__1::pair<long, long> vvenc::fwdTransformCbCr<-1>(vvenc::AreaBuf<short> const&, vvenc::AreaBuf<short> const&, vvenc::AreaBuf<short>&, vvenc::AreaBuf<short>&)
std::__1::pair<long, long> vvenc::fwdTransformCbCr<2>(vvenc::AreaBuf<short> const&, vvenc::AreaBuf<short> const&, vvenc::AreaBuf<short>&, vvenc::AreaBuf<short>&)
Line
Count
Source
96
263k
{
97
263k
  const Pel*  cb  = resCb.buf;
98
263k
  const Pel*  cr  = resCr.buf;
99
263k
  Pel*        c1  = resC1.buf;
100
263k
  Pel*        c2  = resC2.buf;
101
263k
  int64_t     d1  = 0;
102
263k
  int64_t     d2  = 0;
103
3.55M
  for( SizeType y = 0; y < resCb.height; y++, cb += resCb.stride, cr += resCr.stride, c1 += resC1.stride, c2 += resC2.stride )
104
3.29M
  {
105
47.4M
    for( SizeType x = 0; x < resCb.width; x++ )
106
44.1M
    {
107
44.1M
      int cbx = cb[x], crx = cr[x];
108
44.1M
      if      ( signedMode ==  1 )
109
0
      {
110
0
        c1[x] = Pel( ( 4*cbx + 2*crx ) / 5 );
111
0
        d1   += square( cbx - c1[x] ) + square( crx - (c1[x]>>1) );
112
0
      }
113
44.1M
      else if ( signedMode == -1 )
114
0
      {
115
0
        c1[x] = Pel( ( 4*cbx - 2*crx ) / 5 );
116
0
        d1   += square( cbx - c1[x] ) + square( crx - (-c1[x]>>1) );
117
0
      }
118
44.1M
      else if ( signedMode ==  2 )
119
44.1M
      {
120
44.1M
        c1[x] = Pel( ( cbx + crx ) / 2 );
121
44.1M
        d1   += square( cbx - c1[x] ) + square( crx - c1[x] );
122
44.1M
      }
123
0
      else if ( signedMode == -2 )
124
0
      {
125
0
        c1[x] = Pel( ( cbx - crx ) / 2 );
126
0
        d1   += square( cbx - c1[x] ) + square( crx + c1[x] );
127
0
      }
128
0
      else if ( signedMode ==  3 )
129
0
      {
130
0
        c2[x] = Pel( ( 4*crx + 2*cbx ) / 5 );
131
0
        d1   += square( cbx - (c2[x]>>1) ) + square( crx - c2[x] );
132
0
      }
133
0
      else if ( signedMode == -3 )
134
0
      {
135
0
        c2[x] = Pel( ( 4*crx - 2*cbx ) / 5 );
136
0
        d1   += square( cbx - (-c2[x]>>1) ) + square( crx - c2[x] );
137
0
      }
138
0
      else
139
0
      {
140
0
        d1   += square( cbx );
141
0
        d2   += square( crx );
142
0
      }
143
44.1M
    }
144
3.29M
  }
145
263k
  return std::make_pair(d1,d2);
146
263k
}
Unexecuted instantiation: std::__1::pair<long, long> vvenc::fwdTransformCbCr<-2>(vvenc::AreaBuf<short> const&, vvenc::AreaBuf<short> const&, vvenc::AreaBuf<short>&, vvenc::AreaBuf<short>&)
std::__1::pair<long, long> vvenc::fwdTransformCbCr<3>(vvenc::AreaBuf<short> const&, vvenc::AreaBuf<short> const&, vvenc::AreaBuf<short>&, vvenc::AreaBuf<short>&)
Line
Count
Source
96
263k
{
97
263k
  const Pel*  cb  = resCb.buf;
98
263k
  const Pel*  cr  = resCr.buf;
99
263k
  Pel*        c1  = resC1.buf;
100
263k
  Pel*        c2  = resC2.buf;
101
263k
  int64_t     d1  = 0;
102
263k
  int64_t     d2  = 0;
103
3.55M
  for( SizeType y = 0; y < resCb.height; y++, cb += resCb.stride, cr += resCr.stride, c1 += resC1.stride, c2 += resC2.stride )
104
3.29M
  {
105
47.4M
    for( SizeType x = 0; x < resCb.width; x++ )
106
44.1M
    {
107
44.1M
      int cbx = cb[x], crx = cr[x];
108
44.1M
      if      ( signedMode ==  1 )
109
0
      {
110
0
        c1[x] = Pel( ( 4*cbx + 2*crx ) / 5 );
111
0
        d1   += square( cbx - c1[x] ) + square( crx - (c1[x]>>1) );
112
0
      }
113
44.1M
      else if ( signedMode == -1 )
114
0
      {
115
0
        c1[x] = Pel( ( 4*cbx - 2*crx ) / 5 );
116
0
        d1   += square( cbx - c1[x] ) + square( crx - (-c1[x]>>1) );
117
0
      }
118
44.1M
      else if ( signedMode ==  2 )
119
0
      {
120
0
        c1[x] = Pel( ( cbx + crx ) / 2 );
121
0
        d1   += square( cbx - c1[x] ) + square( crx - c1[x] );
122
0
      }
123
44.1M
      else if ( signedMode == -2 )
124
0
      {
125
0
        c1[x] = Pel( ( cbx - crx ) / 2 );
126
0
        d1   += square( cbx - c1[x] ) + square( crx + c1[x] );
127
0
      }
128
44.1M
      else if ( signedMode ==  3 )
129
44.1M
      {
130
44.1M
        c2[x] = Pel( ( 4*crx + 2*cbx ) / 5 );
131
44.1M
        d1   += square( cbx - (c2[x]>>1) ) + square( crx - c2[x] );
132
44.1M
      }
133
0
      else if ( signedMode == -3 )
134
0
      {
135
0
        c2[x] = Pel( ( 4*crx - 2*cbx ) / 5 );
136
0
        d1   += square( cbx - (-c2[x]>>1) ) + square( crx - c2[x] );
137
0
      }
138
0
      else
139
0
      {
140
0
        d1   += square( cbx );
141
0
        d2   += square( crx );
142
0
      }
143
44.1M
    }
144
3.29M
  }
145
263k
  return std::make_pair(d1,d2);
146
263k
}
Unexecuted instantiation: std::__1::pair<long, long> vvenc::fwdTransformCbCr<-3>(vvenc::AreaBuf<short> const&, vvenc::AreaBuf<short> const&, vvenc::AreaBuf<short>&, vvenc::AreaBuf<short>&)
147
148
template<int signedMode> void invTransformCbCr( PelBuf& resCb, PelBuf& resCr )
149
260k
{
150
260k
  Pel*  cb  = resCb.buf;
151
260k
  Pel*  cr  = resCr.buf;
152
3.51M
  for( SizeType y = 0; y < resCb.height; y++, cb += resCb.stride, cr += resCr.stride )
153
3.25M
  {
154
47.0M
    for( SizeType x = 0; x < resCb.width; x++ )
155
43.7M
    {
156
43.7M
      if      ( signedMode ==  1 )  { cr[x] =  cb[x] >> 1;  }
157
43.7M
      else if ( signedMode == -1 )  { cr[x] = -cb[x] >> 1;  }
158
43.7M
      else if ( signedMode ==  2 )  { cr[x] =  cb[x]; }
159
213k
      else if ( signedMode == -2 )  { cr[x] = -cb[x]; }
160
213k
      else if ( signedMode ==  3 )  { cb[x] =  cr[x] >> 1; }
161
0
      else if ( signedMode == -3 )  { cb[x] = -cr[x] >> 1; }
162
43.7M
    }
163
3.25M
  }
164
260k
}
Unexecuted instantiation: void vvenc::invTransformCbCr<0>(vvenc::AreaBuf<short>&, vvenc::AreaBuf<short>&)
Unexecuted instantiation: void vvenc::invTransformCbCr<1>(vvenc::AreaBuf<short>&, vvenc::AreaBuf<short>&)
Unexecuted instantiation: void vvenc::invTransformCbCr<-1>(vvenc::AreaBuf<short>&, vvenc::AreaBuf<short>&)
void vvenc::invTransformCbCr<2>(vvenc::AreaBuf<short>&, vvenc::AreaBuf<short>&)
Line
Count
Source
149
258k
{
150
258k
  Pel*  cb  = resCb.buf;
151
258k
  Pel*  cr  = resCr.buf;
152
3.49M
  for( SizeType y = 0; y < resCb.height; y++, cb += resCb.stride, cr += resCr.stride )
153
3.23M
  {
154
46.7M
    for( SizeType x = 0; x < resCb.width; x++ )
155
43.5M
    {
156
43.5M
      if      ( signedMode ==  1 )  { cr[x] =  cb[x] >> 1;  }
157
43.5M
      else if ( signedMode == -1 )  { cr[x] = -cb[x] >> 1;  }
158
43.5M
      else if ( signedMode ==  2 )  { cr[x] =  cb[x]; }
159
0
      else if ( signedMode == -2 )  { cr[x] = -cb[x]; }
160
0
      else if ( signedMode ==  3 )  { cb[x] =  cr[x] >> 1; }
161
0
      else if ( signedMode == -3 )  { cb[x] = -cr[x] >> 1; }
162
43.5M
    }
163
3.23M
  }
164
258k
}
Unexecuted instantiation: void vvenc::invTransformCbCr<-2>(vvenc::AreaBuf<short>&, vvenc::AreaBuf<short>&)
void vvenc::invTransformCbCr<3>(vvenc::AreaBuf<short>&, vvenc::AreaBuf<short>&)
Line
Count
Source
149
1.68k
{
150
1.68k
  Pel*  cb  = resCb.buf;
151
1.68k
  Pel*  cr  = resCr.buf;
152
20.9k
  for( SizeType y = 0; y < resCb.height; y++, cb += resCb.stride, cr += resCr.stride )
153
19.2k
  {
154
232k
    for( SizeType x = 0; x < resCb.width; x++ )
155
213k
    {
156
213k
      if      ( signedMode ==  1 )  { cr[x] =  cb[x] >> 1;  }
157
213k
      else if ( signedMode == -1 )  { cr[x] = -cb[x] >> 1;  }
158
213k
      else if ( signedMode ==  2 )  { cr[x] =  cb[x]; }
159
213k
      else if ( signedMode == -2 )  { cr[x] = -cb[x]; }
160
213k
      else if ( signedMode ==  3 )  { cb[x] =  cr[x] >> 1; }
161
0
      else if ( signedMode == -3 )  { cb[x] = -cr[x] >> 1; }
162
213k
    }
163
19.2k
  }
164
1.68k
}
Unexecuted instantiation: void vvenc::invTransformCbCr<-3>(vvenc::AreaBuf<short>&, vvenc::AreaBuf<short>&)
165
166
void xFwdLfnstNxNCore(int *src, int *dst, const uint32_t mode, const uint32_t index, const uint32_t size, int zeroOutSize)
167
1.16M
{
168
1.16M
  const int8_t *trMat  = (size > 4) ? g_lfnstFwd8x8[mode][index][0] : g_lfnstFwd4x4[mode][index][0];
169
1.16M
  const int     trSize = (size > 4) ? 48 : 16;
170
1.16M
  int           coef;
171
1.16M
  int *         out = dst;
172
173
18.1M
  for (int j = 0; j < zeroOutSize; j++)
174
16.9M
  {
175
16.9M
    int *         srcPtr   = src;
176
16.9M
    const int8_t *trMatTmp = trMat;
177
16.9M
    coef                   = 0;
178
644M
    for (int i = 0; i < trSize; i++)
179
627M
    {
180
627M
      coef += *srcPtr++ * *trMatTmp++;
181
627M
    }
182
16.9M
    *out++ = (coef + 64) >> 7;
183
16.9M
    trMat += trSize;
184
16.9M
  }
185
186
1.16M
  ::memset(out, 0, (trSize - zeroOutSize) * sizeof(int));
187
1.16M
}
188
189
190
void xInvLfnstNxNCore(int *src, int *dst, const uint32_t mode, const uint32_t index, const uint32_t size, int zeroOutSize)
191
496k
{
192
496k
  int           maxLog2TrDynamicRange = 15;
193
496k
  const TCoeff  outputMinimum         = -(1 << maxLog2TrDynamicRange);
194
496k
  const TCoeff  outputMaximum         = (1 << maxLog2TrDynamicRange) - 1;
195
496k
  const int8_t *trMat                 = (size > 4) ? g_lfnstInv8x8[mode][index][0] : g_lfnstInv4x4[mode][index][0];
196
496k
  const int     trSize                = (size > 4) ? 48 : 16;
197
496k
  int           resi;
198
496k
  int *         out                   = dst;
199
200
18.5M
  for( int j = 0; j < trSize; j++, trMat += 16 )
201
18.0M
  {
202
18.0M
    resi = 0;
203
18.0M
    const int8_t* trMatTmp = trMat;
204
18.0M
    int*          srcPtr   = src;
205
206
274M
    for( int i = 0; i < zeroOutSize; i++ )
207
256M
    {
208
256M
      resi += *srcPtr++ * *trMatTmp++;
209
256M
    }
210
211
18.0M
    *out++ = Clip3( outputMinimum, outputMaximum, ( int ) ( resi + 64 ) >> 7 );
212
18.0M
  }
213
496k
}
214
215
// ====================================================================================================================
216
// TrQuant class member functions
217
// ====================================================================================================================
218
19.1k
TrQuant::TrQuant() : m_scalingListEnabled(false), m_quant( nullptr )
219
19.1k
{
220
  // allocate temporary buffers
221
19.1k
  m_plTempCoeff = ( TCoeff* ) xMalloc( TCoeff, MAX_TB_SIZEY * MAX_TB_SIZEY );
222
19.1k
  m_tmp         = ( TCoeff* ) xMalloc( TCoeff, MAX_TB_SIZEY * MAX_TB_SIZEY );
223
19.1k
  m_blk         = ( TCoeff* ) xMalloc( TCoeff, MAX_TB_SIZEY * MAX_TB_SIZEY );
224
225
134k
  for( int i = 0; i < NUM_TRAFO_MODES_MTS; i++ )
226
115k
  {
227
115k
    m_mtsCoeffs[i] = ( TCoeff* ) xMalloc( TCoeff, MAX_TB_SIZEY * MAX_TB_SIZEY );
228
115k
  }
229
230
19.1k
  {
231
19.1k
    m_invICT      = m_invICTMem + maxAbsIctMode;
232
19.1k
    m_invICT[ 0]  = invTransformCbCr< 0>;
233
19.1k
    m_invICT[ 1]  = invTransformCbCr< 1>;
234
19.1k
    m_invICT[-1]  = invTransformCbCr<-1>;
235
19.1k
    m_invICT[ 2]  = invTransformCbCr< 2>;
236
19.1k
    m_invICT[-2]  = invTransformCbCr<-2>;
237
19.1k
    m_invICT[ 3]  = invTransformCbCr< 3>;
238
19.1k
    m_invICT[-3]  = invTransformCbCr<-3>;
239
19.1k
    m_fwdICT      = m_fwdICTMem + maxAbsIctMode;
240
19.1k
    m_fwdICT[ 0]  = fwdTransformCbCr< 0>;
241
19.1k
    m_fwdICT[ 1]  = fwdTransformCbCr< 1>;
242
19.1k
    m_fwdICT[-1]  = fwdTransformCbCr<-1>;
243
19.1k
    m_fwdICT[ 2]  = fwdTransformCbCr< 2>;
244
19.1k
    m_fwdICT[-2]  = fwdTransformCbCr<-2>;
245
19.1k
    m_fwdICT[ 3]  = fwdTransformCbCr< 3>;
246
19.1k
    m_fwdICT[-3]  = fwdTransformCbCr<-3>;
247
19.1k
  }
248
249
19.1k
  m_invLfnstNxN = xInvLfnstNxNCore;
250
19.1k
  m_fwdLfnstNxN = xFwdLfnstNxNCore;
251
252
#if defined( TARGET_SIMD_X86 ) && ENABLE_SIMD_TRAFO
253
  initTrQuantX86();
254
#endif
255
19.1k
}
256
257
TrQuant::~TrQuant()
258
19.1k
{
259
19.1k
  if( m_quant )
260
19.1k
  {
261
19.1k
    delete m_quant;
262
19.1k
    m_quant = nullptr;
263
19.1k
  }
264
265
  // delete temporary buffers
266
19.1k
  if( m_plTempCoeff )
267
19.1k
  {
268
19.1k
    xFree( m_plTempCoeff );
269
19.1k
    m_plTempCoeff = nullptr;
270
19.1k
  }
271
272
19.1k
  if( m_blk )
273
19.1k
  {
274
19.1k
    xFree( m_blk );
275
19.1k
    m_blk = nullptr;
276
19.1k
  }
277
278
19.1k
  if( m_tmp )
279
19.1k
  {
280
19.1k
    xFree( m_tmp );
281
19.1k
    m_tmp = nullptr;
282
19.1k
  }
283
284
134k
  for( int i = 0; i < NUM_TRAFO_MODES_MTS; i++ )
285
115k
  {
286
115k
     xFree( m_mtsCoeffs[i] );
287
115k
  }
288
19.1k
}
289
290
void TrQuant::xDeQuant(const TransformUnit& tu,
291
                             CoeffBuf      &dstCoeff,
292
                       const ComponentID   &compID,
293
                       const QpParam       &cQP)
294
814k
{
295
814k
  PROFILER_SCOPE_AND_STAGE( 1, _TPROF, P_DEQUANT );
296
814k
  m_quant->dequant( tu, dstCoeff, compID, cQP );
297
814k
}
298
299
void TrQuant::init( const Quant* otherQuant,
300
                    const int  rdoq,
301
                    const bool bUseRDOQTS,
302
                    const bool scalingListsEnabled,
303
                    const bool bEnc,
304
                    const int  thrVal
305
)
306
19.1k
{
307
19.1k
  m_bEnc = bEnc;
308
309
19.1k
  delete m_quant;
310
19.1k
  m_quant = nullptr;
311
312
19.1k
  m_quant = new(std::nothrow) DepQuant( otherQuant, bEnc, scalingListsEnabled );
313
19.1k
  CHECK( !m_quant, "allocation failed" );
314
19.1k
  m_quant->init( rdoq, bUseRDOQTS, thrVal );
315
19.1k
}
316
317
318
void TrQuant::invTransformNxN( TransformUnit& tu, const ComponentID compID, PelBuf& pResi, const QpParam& cQP )
319
814k
{
320
814k
  const CompArea& area    = tu.blocks[compID];
321
814k
  const uint32_t uiWidth  = area.width;
322
814k
  const uint32_t uiHeight = area.height;
323
324
814k
  CHECK( uiWidth > tu.cs->sps->getMaxTbSize() || uiHeight > tu.cs->sps->getMaxTbSize(), "Maximal allowed transformation size exceeded!" );
325
326
814k
  {
327
814k
    CoeffBuf tempCoeff = CoeffBuf( m_plTempCoeff, area );
328
814k
    xDeQuant( tu, tempCoeff, compID, cQP );
329
330
814k
    DTRACE_COEFF_BUF( D_TCOEFF, tempCoeff, tu, tu.cu->predMode, compID );
331
332
814k
    if (tu.cs->sps->LFNST)
333
814k
    {
334
814k
      xInvLfnst(tu, compID);
335
814k
    }
336
814k
    if (tu.mtsIdx[compID] == MTS_SKIP)
337
47.6k
    {
338
47.6k
      xITransformSkip(tempCoeff, pResi, tu, compID);
339
47.6k
    }
340
766k
    else
341
766k
    {
342
766k
      xIT(tu, compID, tempCoeff, pResi);
343
766k
    }
344
814k
  }
345
346
  //DTRACE_BLOCK_COEFF(tu.getCoeffs(compID), tu, tu.cu->predMode, compID);
347
814k
  DTRACE_PEL_BUF( D_RESIDUALS, pResi, tu, tu.cu->predMode, compID);
348
814k
}
349
350
std::pair<int64_t,int64_t> TrQuant::fwdTransformICT( const TransformUnit& tu, const PelBuf& resCb, const PelBuf& resCr, PelBuf& resC1, PelBuf& resC2, int jointCbCr )
351
1.05M
{
352
1.05M
  CHECK( Size(resCb) != Size(resCr), "resCb and resCr have different sizes" );
353
1.05M
  CHECK( Size(resCb) != Size(resC1), "resCb and resC1 have different sizes" );
354
1.05M
  CHECK( Size(resCb) != Size(resC2), "resCb and resC2 have different sizes" );
355
1.05M
  return (*m_fwdICT[ TU::getICTMode(tu, jointCbCr) ])( resCb, resCr, resC1, resC2 );
356
1.05M
}
357
358
void TrQuant::invTransformICT( const TransformUnit& tu, PelBuf& resCb, PelBuf& resCr )
359
260k
{
360
260k
  CHECK( Size(resCb) != Size(resCr), "resCb and resCr have different sizes" );
361
260k
  (*m_invICT[ TU::getICTMode(tu) ])( resCb, resCr );
362
260k
}
363
364
std::vector<int> TrQuant::selectICTCandidates( const TransformUnit& tu, CompStorage* resCb, CompStorage* resCr )
365
263k
{
366
263k
  CHECK( !resCb[0].valid() || !resCr[0].valid(), "standard components are not valid" );
367
368
263k
  if( !CU::isIntra( *tu.cu ) )
369
0
  {
370
0
    int cbfMask = 3;
371
0
    fwdTransformICT( tu, resCb[0], resCr[0], resCb[cbfMask], resCr[cbfMask], cbfMask );
372
0
    std::vector<int> cbfMasksToTest;
373
0
    cbfMasksToTest.push_back( cbfMask );
374
0
    return cbfMasksToTest;
375
0
  }
376
377
263k
  std::pair<int64_t,int64_t> pairDist[4];
378
1.31M
  for( int cbfMask = 0; cbfMask < 4; cbfMask++ )
379
1.05M
  {
380
1.05M
    pairDist[cbfMask] = fwdTransformICT( tu, resCb[0], resCr[0], resCb[cbfMask], resCr[cbfMask], cbfMask );
381
1.05M
  }
382
383
263k
  std::vector<int> cbfMasksToTest;
384
263k
  int64_t minDist1  = std::min<int64_t>( pairDist[0].first, pairDist[0].second );
385
263k
  int64_t minDist2  = std::numeric_limits<int64_t>::max();
386
263k
  int     cbfMask1  = 0;
387
263k
  int     cbfMask2  = 0;
388
263k
  for( int cbfMask : { 1, 2, 3 } )
389
790k
  {
390
790k
    if( pairDist[cbfMask].first < minDist1 )
391
522k
    {
392
522k
      cbfMask2  = cbfMask1; minDist2  = minDist1;
393
522k
      cbfMask1  = cbfMask;  minDist1  = pairDist[cbfMask1].first;
394
522k
    }
395
268k
    else if( pairDist[cbfMask].first < minDist2 )
396
263k
    {
397
263k
      cbfMask2  = cbfMask;  minDist2  = pairDist[cbfMask2].first;
398
263k
    }
399
790k
  }
400
263k
  if( cbfMask1 )
401
263k
  {
402
263k
    cbfMasksToTest.push_back( cbfMask1 );
403
263k
  }
404
263k
  if( cbfMask2 && ( ( minDist2 < (9*minDist1)/8 ) || ( !cbfMask1 && minDist2 < (3*minDist1)/2 ) ) )
405
0
  {
406
0
    cbfMasksToTest.push_back( cbfMask2 );
407
0
  }
408
409
263k
  return cbfMasksToTest;
410
263k
}
411
412
413
414
// ------------------------------------------------------------------------------------------------
415
// Logical transform
416
// ------------------------------------------------------------------------------------------------
417
void TrQuant::xSetTrTypes( const TransformUnit& tu, const ComponentID compID, const int width, const int height, int &trTypeHor, int &trTypeVer )
418
2.63M
{
419
2.63M
  const bool isISP = CU::isIntra(*tu.cu) && tu.cu->ispMode && isLuma(compID);
420
2.63M
  if (isISP && tu.cu->lfnstIdx)
421
18.6k
  {
422
18.6k
    return;
423
18.6k
  }
424
2.61M
  if (!tu.cs->sps->MTS)
425
0
  {
426
0
    return;
427
0
  }
428
2.61M
  if (CU::isIntra(*tu.cu) && isLuma(compID) && ((tu.cs->sps->getUseImplicitMTS() && tu.cu->lfnstIdx == 0 && tu.cu->mipFlag == 0) || tu.cu->ispMode))
429
82.2k
  {
430
82.2k
    if (width >= 4 && width <= 16)
431
37.1k
      trTypeHor = DST7;
432
82.2k
    if (height >= 4 && height <= 16)
433
35.2k
      trTypeVer = DST7;
434
82.2k
  }
435
2.53M
  else if( tu.cs->sps->MTS && tu.cu->sbtInfo && isLuma(compID)/*isSBT*/ )
436
0
  {
437
0
    const uint8_t sbtIdx = CU::getSbtIdx( tu.cu->sbtInfo );
438
0
    const uint8_t sbtPos = CU::getSbtPos( tu.cu->sbtInfo );
439
440
0
    if( sbtIdx == SBT_VER_HALF || sbtIdx == SBT_VER_QUAD )
441
0
    {
442
0
      assert( tu.lwidth() <= MTS_INTER_MAX_CU_SIZE );
443
0
      if( tu.lheight() > MTS_INTER_MAX_CU_SIZE )
444
0
      {
445
0
        trTypeHor = trTypeVer = DCT2;
446
0
      }
447
0
      else
448
0
      {
449
0
        if( sbtPos == SBT_POS0 )  { trTypeHor = DCT8;  trTypeVer = DST7; }
450
0
        else                      { trTypeHor = DST7;  trTypeVer = DST7; }
451
0
      }
452
0
    }
453
0
    else
454
0
    {
455
0
      assert( tu.lheight() <= MTS_INTER_MAX_CU_SIZE );
456
0
      if( tu.lwidth() > MTS_INTER_MAX_CU_SIZE )
457
0
      {
458
0
        trTypeHor = trTypeVer = DCT2;
459
0
      }
460
0
      else
461
0
      {
462
0
        if( sbtPos == SBT_POS0 )  { trTypeHor = DST7;  trTypeVer = DCT8; }
463
0
        else                      { trTypeHor = DST7;  trTypeVer = DST7; }
464
0
      }
465
0
    }
466
0
  }
467
2.61M
  const bool isExplicitMTS = (CU::isIntra(*tu.cu) ? tu.cs->sps->MTS : tu.cs->sps->MTSInter && CU::isInter(*tu.cu)) && isLuma(compID);
468
2.61M
  if (isExplicitMTS)
469
195k
  {
470
195k
    if (tu.mtsIdx[compID] > MTS_SKIP)
471
0
    {
472
0
      int indHor = (tu.mtsIdx[compID] - MTS_DST7_DST7) & 1;
473
0
      int indVer = (tu.mtsIdx[compID] - MTS_DST7_DST7) >> 1;
474
0
      trTypeHor  = indHor ? DCT8 : DST7;
475
0
      trTypeVer  = indVer ? DCT8 : DST7;
476
0
    }
477
195k
  }
478
2.61M
}
479
480
481
void TrQuant::xT( const TransformUnit& tu, const ComponentID compID, const CPelBuf& resi, CoeffBuf& dstCoeff, const int width, const int height )
482
1.87M
{
483
1.87M
  PROFILER_SCOPE_AND_STAGE( 1, _TPROF, P_TRAFO );
484
485
1.87M
  const unsigned maxLog2TrDynamicRange  = tu.cs->sps->getMaxLog2TrDynamicRange();
486
1.87M
  const unsigned bitDepth               = tu.cs->sps->bitDepths[toChannelType( compID )];
487
1.87M
  const int      TRANSFORM_MATRIX_SHIFT = g_transformMatrixShift[TRANSFORM_FORWARD];
488
1.87M
  const uint32_t transformWidthIndex    = Log2(width ) - 1;  // nLog2WidthMinus1, since transform start from 2-point
489
1.87M
  const uint32_t transformHeightIndex   = Log2(height) - 1;  // nLog2HeightMinus1, since transform start from 2-point
490
491
1.87M
  int trTypeHor = DCT2;
492
1.87M
  int trTypeVer = DCT2;
493
494
1.87M
  xSetTrTypes( tu, compID, width, height, trTypeHor, trTypeVer );
495
496
1.87M
  int  skipWidth  = ( trTypeHor != DCT2 && width  == 32 ) ? 16 : width  > JVET_C0024_ZERO_OUT_TH ? width  - JVET_C0024_ZERO_OUT_TH : 0;
497
1.87M
  int  skipHeight = ( trTypeVer != DCT2 && height == 32 ) ? 16 : height > JVET_C0024_ZERO_OUT_TH ? height - JVET_C0024_ZERO_OUT_TH : 0;
498
499
1.87M
  if( tu.cu->lfnstIdx )
500
1.16M
  {
501
1.16M
    if ((width == 4 && height > 4) || (width > 4 && height == 4))
502
328k
    {
503
328k
      skipWidth  = width - 4;
504
328k
      skipHeight = height - 4;
505
328k
    }
506
837k
    else if ((width >= 8 && height >= 8))
507
765k
    {
508
765k
      skipWidth  = width - 8;
509
765k
      skipHeight = height - 8;
510
765k
    }
511
1.16M
  }
512
513
1.87M
  TCoeff* block = m_blk;
514
1.87M
  TCoeff* tmp   = m_tmp;
515
516
1.87M
  const Pel* resiBuf    = resi.buf;
517
1.87M
  const int  resiStride = resi.stride;
518
519
1.87M
#if ENABLE_SIMD_TRAFO
520
1.87M
  if( width & 3 )
521
0
#endif
522
0
  {
523
0
    for( int y = 0; y < height; y++ )
524
0
    {
525
0
      for( int x = 0; x < width; x++ )
526
0
      {
527
0
        block[( y * width ) + x] = resiBuf[( y * resiStride ) + x];
528
0
      }
529
0
    }
530
0
  }
531
1.87M
#if ENABLE_SIMD_TRAFO
532
1.87M
  else if( width & 7 )
533
353k
  {
534
353k
    g_tCoeffOps.cpyCoeff4( resiBuf, resiStride, block, width, height );
535
353k
  }
536
1.51M
  else
537
1.51M
  {
538
1.51M
    g_tCoeffOps.cpyCoeff8( resiBuf, resiStride, block, width, height );
539
1.51M
  }
540
1.87M
#endif //ENABLE_SIMD_TRAFO
541
542
1.87M
  if (width > 1 && height > 1)
543
1.87M
  {
544
1.87M
    const int shift_1st = ((Log2(width )) + bitDepth + TRANSFORM_MATRIX_SHIFT) - maxLog2TrDynamicRange;
545
1.87M
    const int shift_2nd =  (Log2(height))            + TRANSFORM_MATRIX_SHIFT;
546
1.87M
    CHECK( shift_1st < 0, "Negative shift" );
547
1.87M
    CHECK( shift_2nd < 0, "Negative shift" );
548
1.87M
    fastFwdTrans[trTypeHor][transformWidthIndex](block, tmp, shift_1st, height, 0, skipWidth);
549
1.87M
    fastFwdTrans[trTypeVer][transformHeightIndex](tmp, dstCoeff.buf, shift_2nd, width, skipWidth, skipHeight);
550
1.87M
  }
551
222
  else if (height == 1)   // 1-D horizontal transform
552
268
  {
553
268
    const int shift = ((Log2(width )) + bitDepth + TRANSFORM_MATRIX_SHIFT) - maxLog2TrDynamicRange;
554
268
    CHECK( shift < 0, "Negative shift" );
555
268
    fastFwdTrans[trTypeHor][transformWidthIndex](block, dstCoeff.buf, shift, 1, 0, skipWidth);
556
268
  }
557
18.4E
  else   // if (iWidth == 1) //1-D vertical transform
558
18.4E
  {
559
18.4E
    int shift = ((floorLog2(height)) + bitDepth + TRANSFORM_MATRIX_SHIFT) - maxLog2TrDynamicRange;
560
18.4E
    CHECK(shift < 0, "Negative shift");
561
18.4E
    CHECKD((transformHeightIndex < 0), "There is a problem with the height.");
562
18.4E
    fastFwdTrans[trTypeVer][transformHeightIndex](block, dstCoeff.buf, shift, 1, 0, skipHeight);
563
18.4E
  }
564
1.87M
}
565
566
567
void TrQuant::xIT( const TransformUnit& tu, const ComponentID compID, const CCoeffBuf& pCoeff, PelBuf& pResidual )
568
766k
{
569
766k
  PROFILER_SCOPE_AND_STAGE( 1, _TPROF, P_TRAFO );
570
571
766k
  const int      width                  = pCoeff.width;
572
766k
  const int      height                 = pCoeff.height;
573
766k
  const unsigned maxLog2TrDynamicRange  = tu.cs->sps->getMaxLog2TrDynamicRange();
574
766k
  const unsigned bitDepth               = tu.cs->sps->bitDepths[toChannelType( compID )];
575
766k
  const int      TRANSFORM_MATRIX_SHIFT = g_transformMatrixShift[TRANSFORM_INVERSE];
576
766k
  const TCoeff   clipMinimum            = -( 1 << maxLog2TrDynamicRange );
577
766k
  const TCoeff   clipMaximum            =  ( 1 << maxLog2TrDynamicRange ) - 1;
578
766k
  const uint32_t transformWidthIndex    = Log2(width )- 1;                                // nLog2WidthMinus1, since transform start from 2-point
579
766k
  const uint32_t transformHeightIndex   = Log2(height) - 1;                                // nLog2HeightMinus1, since transform start from 2-point
580
581
582
766k
  int trTypeHor = DCT2;
583
766k
  int trTypeVer = DCT2;
584
585
766k
  xSetTrTypes( tu, compID, width, height, trTypeHor, trTypeVer );
586
587
766k
  int skipWidth  = ( trTypeHor != DCT2 && width  == 32 ) ? 16 : width  > JVET_C0024_ZERO_OUT_TH ? width  - JVET_C0024_ZERO_OUT_TH : 0;
588
766k
  int skipHeight = ( trTypeVer != DCT2 && height == 32 ) ? 16 : height > JVET_C0024_ZERO_OUT_TH ? height - JVET_C0024_ZERO_OUT_TH : 0;
589
590
766k
  if (tu.cs->sps->LFNST && tu.cu->lfnstIdx)
591
496k
  {
592
496k
    if ((width == 4 && height > 4) || (width > 4 && height == 4))
593
141k
    {
594
141k
      skipWidth = width - 4;
595
141k
      skipHeight = height - 4;
596
141k
    }
597
354k
    else if ((width >= 8 && height >= 8))
598
315k
    {
599
315k
      skipWidth = width - 8;
600
315k
      skipHeight = height - 8;
601
315k
    }
602
496k
  }
603
604
766k
  TCoeff *block = m_blk;
605
766k
  TCoeff *tmp   = m_tmp;
606
766k
  if (width > 1 && height > 1)   // 2-D transform
607
766k
  {
608
766k
    const int shift_1st =   TRANSFORM_MATRIX_SHIFT + 1; // 1 has been added to shift_1st at the expense of shift_2nd
609
766k
    const int shift_2nd = ( TRANSFORM_MATRIX_SHIFT + maxLog2TrDynamicRange - 1 ) - bitDepth;
610
766k
    CHECK( shift_1st < 0, "Negative shift" );
611
766k
    CHECK( shift_2nd < 0, "Negative shift" );
612
766k
    fastInvTrans[trTypeVer][transformHeightIndex](pCoeff.buf, tmp, shift_1st, width, skipWidth, skipHeight, clipMinimum, clipMaximum);
613
766k
    fastInvTrans[trTypeHor][transformWidthIndex](tmp, block, shift_2nd, height, 0, skipWidth, clipMinimum, clipMaximum);
614
766k
  }
615
67
  else if (width == 1)   // 1-D vertical transform
616
0
  {
617
0
    int shift = (TRANSFORM_MATRIX_SHIFT + maxLog2TrDynamicRange - 1) - bitDepth;
618
0
    CHECK(shift < 0, "Negative shift");
619
0
    fastInvTrans[trTypeVer][transformHeightIndex](pCoeff.buf, block, shift + 1, 1, 0, skipHeight, clipMinimum, clipMaximum);
620
0
  }
621
67
  else   // if(iHeight == 1) //1-D horizontal transform
622
67
  {
623
67
    const int shift = (TRANSFORM_MATRIX_SHIFT + maxLog2TrDynamicRange - 1) - bitDepth;
624
67
    CHECK(shift < 0, "Negative shift");
625
67
    fastInvTrans[trTypeHor][transformWidthIndex](pCoeff.buf, block, shift + 1, 1, 0, skipWidth, clipMinimum, clipMaximum);
626
67
  }
627
628
766k
#if ENABLE_SIMD_TRAFO
629
766k
  if( width & 3 )
630
0
#endif //ENABLE_SIMD_TRAFO
631
0
  {
632
0
    Pel       *dst    = pResidual.buf;
633
0
    ptrdiff_t  stride = pResidual.stride;
634
635
0
    for( int y = 0; y < height; y++ )
636
0
    {
637
0
      for( int x = 0; x < width; x++ )
638
0
      {
639
0
        dst[x] = ( Pel ) *block++;
640
0
      }
641
642
0
      dst += stride;
643
0
    }
644
0
  }
645
766k
#if ENABLE_SIMD_TRAFO
646
766k
  else if( width & 7 )
647
165k
  {
648
165k
    g_tCoeffOps.cpyResi4( block, pResidual.buf, pResidual.stride, width, height );
649
165k
  }
650
601k
  else
651
601k
  {
652
601k
    g_tCoeffOps.cpyResi8( block, pResidual.buf, pResidual.stride, width, height );
653
601k
  }
654
766k
#endif //ENABLE_SIMD_TRAFO
655
766k
}
656
657
/** Wrapper function between HM interface and core NxN transform skipping
658
 */
659
void TrQuant::xITransformSkip(const CCoeffBuf& pCoeff,
660
  PelBuf& pResidual,
661
  const TransformUnit& tu,
662
  const ComponentID compID)
663
47.6k
{
664
47.6k
  const CompArea& area = tu.blocks[compID];
665
47.6k
  const int width = area.width;
666
47.6k
  const int height = area.height;
667
668
479k
  for (uint32_t y = 0; y < height; y++)
669
431k
  {
670
4.78M
    for (uint32_t x = 0; x < width; x++)
671
4.35M
    {
672
4.35M
      pResidual.at(x, y) = Pel(pCoeff.at(x, y));
673
4.35M
    }
674
431k
  }
675
47.6k
}
676
677
void TrQuant::xQuant(TransformUnit& tu, const ComponentID compID, const CCoeffBuf& pSrc, TCoeff &uiAbsSum, const QpParam& cQP, const Ctx& ctx)
678
1.97M
{
679
1.97M
  PROFILER_SCOPE_AND_STAGE( 1, _TPROF, P_QUANT );
680
1.97M
  m_quant->quant( tu, compID, pSrc, uiAbsSum, cQP, ctx );
681
#if ENABLE_MEASURE_SEARCH_SPACE
682
683
  g_searchSpaceAcc.addQuant( tu, toChannelType( compID ) );
684
#endif
685
1.97M
}
686
687
688
void TrQuant::transformNxN(TransformUnit &tu, const ComponentID compID, const QpParam &cQP, TCoeff &uiAbsSum, const Ctx &ctx, const bool loadTr)
689
1.97M
{
690
1.97M
        CodingStructure &cs = *tu.cs;
691
1.97M
  const CompArea& rect      = tu.blocks[compID];
692
1.97M
  const uint32_t uiWidth        = rect.width;
693
1.97M
  const uint32_t uiHeight       = rect.height;
694
695
1.97M
  const CPelBuf resiBuf     = cs.getResiBuf(rect);
696
697
1.97M
  if( tu.noResidual )
698
0
  {
699
0
    uiAbsSum = 0;
700
0
    TU::setCbfAtDepth( tu, compID, tu.depth, uiAbsSum > 0 );
701
0
    return;
702
0
  }
703
1.97M
  if (tu.cu->bdpcmM[toChannelType(compID)])
704
97.7k
  {
705
97.7k
    tu.mtsIdx[compID] = MTS_SKIP;
706
97.7k
  }
707
708
1.97M
  uiAbsSum = 0;
709
1.97M
  CHECK( cs.sps->getMaxTbSize() < uiWidth, "Unsupported transformation size" );
710
711
1.97M
  CoeffBuf tempCoeff(loadTr ? m_mtsCoeffs[tu.mtsIdx[compID]] : m_plTempCoeff, rect);
712
1.97M
  if (!loadTr)
713
1.95M
  {
714
1.95M
    DTRACE_PEL_BUF( D_RESIDUALS, resiBuf, tu, tu.cu->predMode, compID );
715
1.95M
    if (tu.mtsIdx[compID] == MTS_SKIP)
716
97.7k
    {
717
97.7k
      xTransformSkip(tu, compID, resiBuf, tempCoeff.buf);
718
97.7k
    }
719
1.85M
    else
720
1.85M
    {
721
1.85M
      xT(tu, compID, resiBuf, tempCoeff, uiWidth, uiHeight);
722
1.85M
    }
723
1.95M
  }
724
1.97M
  if (cs.sps->LFNST)
725
1.97M
  {
726
1.97M
    xFwdLfnst(tu, compID, loadTr);
727
1.97M
  }
728
1.97M
  DTRACE_COEFF_BUF( D_TCOEFF, tempCoeff, tu, tu.cu->predMode, compID );
729
730
1.97M
  xQuant( tu, compID, tempCoeff, uiAbsSum, cQP, ctx );
731
732
1.97M
  DTRACE_COEFF_BUF( D_TCOEFF, tu.getCoeffs( compID ), tu, tu.cu->predMode, compID );
733
734
  // set coded block flag (CBF)
735
1.97M
  TU::setCbfAtDepth (tu, compID, tu.depth, uiAbsSum > 0);
736
1.97M
}
737
738
void TrQuant::checktransformsNxN( TransformUnit &tu, std::vector<TrMode> *trModes, const int maxCand, const ComponentID compID)
739
17.1k
{
740
17.1k
  CodingStructure &cs     = *tu.cs;
741
17.1k
  const CompArea& rect    = tu.blocks[compID];
742
17.1k
  const uint32_t   width  = rect.width;
743
17.1k
  const uint32_t   height = rect.height;
744
745
17.1k
  const CPelBuf resiBuf = cs.getResiBuf(rect);
746
747
17.1k
  CHECK(cs.sps->getMaxTbSize() < width, "Unsupported transformation size");
748
17.1k
  int                           pos = 0;
749
17.1k
  std::vector<TrCost>           trCosts;
750
17.1k
  std::vector<TrMode>::iterator it      = trModes->begin();
751
17.1k
  const double                  facBB[] = { 1.2, 1.3, 1.3, 1.4, 1.5 };
752
51.3k
  while (it != trModes->end())
753
34.2k
  {
754
34.2k
    tu.mtsIdx[compID] = it->first;
755
34.2k
    CoeffBuf tempCoeff(m_mtsCoeffs[tu.mtsIdx[compID]], rect);
756
34.2k
    if (tu.noResidual)
757
0
    {
758
0
      int sumAbs = 0;
759
0
      trCosts.push_back(TrCost(sumAbs, pos++));
760
0
      it++;
761
0
      continue;
762
0
    }
763
34.2k
    if (tu.mtsIdx[compID] == MTS_SKIP)
764
17.1k
    {
765
17.1k
      xTransformSkip(tu, compID, resiBuf, tempCoeff.buf);
766
17.1k
    }
767
17.1k
    else
768
17.1k
    {
769
17.1k
      xT(tu, compID, resiBuf, tempCoeff, width, height);
770
17.1k
    }
771
772
34.2k
    int sumAbs = 0;
773
6.12M
    for (int pos = 0; pos < width * height; pos++)
774
6.08M
    {
775
6.08M
      sumAbs += abs(tempCoeff.buf[pos]);
776
6.08M
    }
777
778
34.2k
    double scaleSAD = 1.0;
779
34.2k
    if (tu.mtsIdx[compID] == MTS_SKIP && ((floorLog2(width) + floorLog2(height)) & 1) == 1)
780
7.13k
    {
781
7.13k
      scaleSAD = 1.0 / 1.414213562;   // compensate for not scaling transform skip coefficients by 1/sqrt(2)
782
7.13k
    }
783
34.2k
    if (tu.mtsIdx[compID] == MTS_SKIP)
784
17.1k
    {
785
17.1k
      int trShift = getTransformShift(tu.cu->slice->sps->bitDepths[CH_L], rect.size(), tu.cu->slice->sps->getMaxLog2TrDynamicRange());
786
17.1k
      scaleSAD *= pow(2, trShift);
787
17.1k
    }
788
34.2k
    trCosts.push_back(TrCost(int(std::min<double>(sumAbs * scaleSAD, std::numeric_limits<int>::max())), pos++));
789
34.2k
    it++;
790
34.2k
  }
791
792
17.1k
  int                           numTests = 0;
793
17.1k
  std::vector<TrCost>::iterator itC      = trCosts.begin();
794
17.1k
  const double                  fac      = facBB[std::max(0, floorLog2(std::max(width, height)) - 2)];
795
17.1k
  const double                  thr      = fac * trCosts.begin()->first;
796
17.1k
  const double                  thrTS    = trCosts.begin()->first;
797
51.3k
  while (itC != trCosts.end())
798
34.2k
  {
799
34.2k
    const bool testTr               = itC->first <= (trModes->at(itC->second).first == 1 ? thrTS : thr) && numTests <= maxCand;
800
34.2k
    trModes->at(itC->second).second = testTr;
801
34.2k
    numTests += testTr;
802
34.2k
    itC++;
803
34.2k
  }
804
17.1k
}
805
806
uint32_t TrQuant::xGetLFNSTIntraMode( const Area& tuArea, const uint32_t dirMode )
807
1.66M
{
808
1.66M
  if (dirMode < 2)
809
936k
  {
810
936k
    return dirMode;
811
936k
  }
812
813
726k
  static const int modeShift[] = { 0, 6, 10, 12, 14, 15 };
814
815
726k
  const int width  = int(tuArea.width);
816
726k
  const int height = int(tuArea.height);
817
818
726k
  if (width > height && dirMode < 2 + modeShift[floorLog2(width) - floorLog2(height)])
819
212
  {
820
212
    return dirMode + (VDIA_IDX - 1) + (NUM_EXT_LUMA_MODE >> 1);
821
212
  }
822
725k
  else if (height > width && dirMode > VDIA_IDX - modeShift[floorLog2(height) - floorLog2(width)])
823
67.9k
  {
824
67.9k
    return dirMode - (VDIA_IDX + 1) + (NUM_EXT_LUMA_MODE >> 1) + NUM_LUMA_MODE;
825
67.9k
  }
826
827
658k
  return dirMode;
828
726k
}
829
830
831
bool TrQuant::xGetTransposeFlag(uint32_t intraMode)
832
1.66M
{
833
1.66M
  return ((intraMode >= NUM_LUMA_MODE) && (intraMode >= (NUM_LUMA_MODE + (NUM_EXT_LUMA_MODE >> 1))))
834
1.66M
         || ((intraMode < NUM_LUMA_MODE) && (intraMode > DIA_IDX));
835
1.66M
}
836
837
838
void TrQuant::xInvLfnst(const TransformUnit &tu, const ComponentID compID)
839
814k
{
840
814k
  const CompArea &area     = tu.blocks[compID];
841
814k
  const uint32_t  width    = area.width;
842
814k
  const uint32_t  height   = area.height;
843
814k
  const uint32_t  lfnstIdx = tu.cu->lfnstIdx;
844
814k
  if (lfnstIdx && tu.mtsIdx[compID] != MTS_SKIP && (CU::isSepTree(*tu.cu) ? true : isLuma(compID)))
845
496k
  {
846
496k
    const CodingUnit& cu = *tu.cs->getCU(area.pos(), toChannelType(compID), TREE_D);
847
496k
    const bool         whge3 = width >= 8 && height >= 8;
848
496k
    const ScanElement *scan =
849
496k
      whge3
850
496k
        ? g_coefTopLeftDiagScan8x8[Log2(width)] 
851
496k
        : getScanOrder(SCAN_GROUPED_4x4, Log2(area.width), Log2(area.height));
852
496k
    uint32_t intraMode = CU::getFinalIntraMode(cu, toChannelType(compID));
853
854
496k
    if (CU::isLMCMode( cu.intraDir[toChannelType(compID)]))
855
81.9k
    {
856
81.9k
      intraMode = CU::getCoLocatedIntraLumaMode(cu);
857
81.9k
    }
858
496k
    if (CU::isMIP(cu, toChannelType(compID)))
859
4.82k
    {
860
4.82k
      intraMode = PLANAR_IDX;
861
4.82k
    }
862
496k
    CHECK(intraMode >= NUM_INTRA_MODE - 1, "Invalid intra mode");
863
864
496k
    if (lfnstIdx < 3)
865
496k
    {
866
496k
      if (tu.cu->ispMode && isLuma(compID))
867
7.21k
      {
868
7.21k
        intraMode = xGetLFNSTIntraMode(tu.cu->blocks[compID], intraMode);
869
7.21k
      }
870
489k
      else
871
489k
        intraMode = xGetLFNSTIntraMode(tu.blocks[compID], intraMode);
872
496k
      bool      transposeFlag = xGetTransposeFlag(intraMode);
873
496k
      const int sbSize        = whge3 ? 8 : 4;
874
496k
      bool      tu4x4Flag     = (width == 4 && height == 4);
875
496k
      bool      tu8x8Flag     = (width == 8 && height == 8);
876
496k
      TCoeff *  lfnstTemp;
877
496k
      TCoeff *  coeffTemp;
878
496k
      int       y;
879
496k
      lfnstTemp                  = m_tempInMatrix;   // inverse spectral rearrangement
880
496k
      coeffTemp                  = m_plTempCoeff;
881
496k
      TCoeff *           dst     = lfnstTemp;
882
496k
      const ScanElement *scanPtr = scan;
883
8.43M
      for (y = 0; y < 16; y++)
884
7.94M
      {
885
7.94M
        *dst++ = coeffTemp[scanPtr->idx];
886
7.94M
        scanPtr++;
887
7.94M
      }
888
889
496k
      m_invLfnstNxN( m_tempInMatrix, m_tempOutMatrix, g_lfnstLut[intraMode], lfnstIdx - 1, sbSize, ( tu4x4Flag || tu8x8Flag ) ? 8 : 16 );
890
891
496k
      lfnstTemp = m_tempOutMatrix;   // inverse spectral rearrangement
892
893
496k
      if (transposeFlag)
894
45.6k
      {
895
45.6k
        if (sbSize == 4)
896
8.81k
        {
897
44.0k
          for (y = 0; y < 4; y++)
898
35.2k
          {
899
35.2k
            coeffTemp[0] = lfnstTemp[0];
900
35.2k
            coeffTemp[1] = lfnstTemp[4];
901
35.2k
            coeffTemp[2] = lfnstTemp[8];
902
35.2k
            coeffTemp[3] = lfnstTemp[12];
903
35.2k
            lfnstTemp++;
904
35.2k
            coeffTemp += width;
905
35.2k
          }
906
8.81k
        }
907
36.8k
        else   // ( sbSize == 8 )
908
36.8k
        {
909
331k
          for (y = 0; y < 8; y++)
910
294k
          {
911
294k
            coeffTemp[0] = lfnstTemp[0];
912
294k
            coeffTemp[1] = lfnstTemp[8];
913
294k
            coeffTemp[2] = lfnstTemp[16];
914
294k
            coeffTemp[3] = lfnstTemp[24];
915
294k
            if (y < 4)
916
147k
            {
917
147k
              coeffTemp[4] = lfnstTemp[32];
918
147k
              coeffTemp[5] = lfnstTemp[36];
919
147k
              coeffTemp[6] = lfnstTemp[40];
920
147k
              coeffTemp[7] = lfnstTemp[44];
921
147k
            }
922
294k
            lfnstTemp++;
923
294k
            coeffTemp += width;
924
294k
          }
925
36.8k
        }
926
45.6k
      }
927
450k
      else
928
450k
      {
929
3.36M
        for (y = 0; y < sbSize; y++)
930
2.91M
        {
931
2.91M
          uint32_t uiStride = (y < 4) ? sbSize : 4;
932
2.91M
          ::memcpy(coeffTemp, lfnstTemp, uiStride * sizeof(TCoeff));
933
2.91M
          lfnstTemp += uiStride;
934
2.91M
          coeffTemp += width;
935
2.91M
        }
936
450k
      }
937
496k
    }
938
496k
  }
939
814k
}
940
941
942
void TrQuant::xFwdLfnst(const TransformUnit &tu, const ComponentID compID, const bool loadTr)
943
1.97M
{
944
1.97M
  const CompArea &area     = tu.blocks[compID];
945
1.97M
  const uint32_t  width    = area.width;
946
1.97M
  const uint32_t  height   = area.height;
947
1.97M
  const uint32_t  lfnstIdx = tu.cu->lfnstIdx;
948
1.97M
  if (lfnstIdx && tu.mtsIdx[compID] != MTS_SKIP && (CU::isSepTree(*tu.cu) ? true : isLuma(compID)))
949
1.16M
  {
950
1.16M
    const CodingUnit& cu = *tu.cs->getCU(area.pos(), toChannelType(compID), TREE_D);
951
1.16M
    const bool         whge3 = width >= 8 && height >= 8;
952
1.16M
    const ScanElement *scan =
953
1.16M
      whge3
954
1.16M
        ? g_coefTopLeftDiagScan8x8[Log2(width)] 
955
1.16M
        : getScanOrder(SCAN_GROUPED_4x4, Log2(area.width), Log2(area.height));   
956
1.16M
    uint32_t intraMode = CU::getFinalIntraMode(cu, toChannelType(compID));
957
958
1.16M
    if (CU::isLMCMode(cu.intraDir[toChannelType(compID)]))
959
103k
    {
960
103k
      intraMode = CU::getCoLocatedIntraLumaMode(cu);
961
103k
    }
962
1.16M
    if (CU::isMIP(cu, toChannelType(compID)))
963
8.09k
    {
964
8.09k
      intraMode = PLANAR_IDX;
965
8.09k
    }
966
1.16M
    CHECK(intraMode >= NUM_INTRA_MODE - 1, "Invalid intra mode");
967
968
1.16M
    if (lfnstIdx < 3)
969
1.16M
    {
970
1.16M
      if (tu.cu->ispMode && isLuma(compID))
971
11.4k
      {
972
11.4k
        intraMode = xGetLFNSTIntraMode(tu.cu->blocks[compID], intraMode);
973
11.4k
      }
974
1.15M
      else
975
1.15M
      {
976
1.15M
        intraMode = xGetLFNSTIntraMode(tu.blocks[compID], intraMode);
977
1.15M
      }
978
1.16M
      bool      transposeFlag = xGetTransposeFlag(intraMode);
979
1.16M
      const int sbSize        = whge3 ? 8 : 4;
980
1.16M
      bool      tu4x4Flag     = (width == 4 && height == 4);
981
1.16M
      bool      tu8x8Flag     = (width == 8 && height == 8);
982
1.16M
      TCoeff*   lfnstTemp;
983
1.16M
      TCoeff*   coeffTemp;
984
1.16M
      TCoeff*   tempCoeff     = loadTr ? m_mtsCoeffs[tu.mtsIdx[compID]] : m_plTempCoeff;
985
986
1.16M
      int y;
987
1.16M
      lfnstTemp = m_tempInMatrix;   // forward low frequency non-separable transform
988
1.16M
      coeffTemp = tempCoeff;
989
990
1.16M
      if (transposeFlag)
991
248k
      {
992
248k
        if (sbSize == 4)
993
72.4k
        {
994
362k
          for (y = 0; y < 4; y++)
995
289k
          {
996
289k
            lfnstTemp[0]  = coeffTemp[0];
997
289k
            lfnstTemp[4]  = coeffTemp[1];
998
289k
            lfnstTemp[8]  = coeffTemp[2];
999
289k
            lfnstTemp[12] = coeffTemp[3];
1000
289k
            lfnstTemp++;
1001
289k
            coeffTemp += width;
1002
289k
          }
1003
72.4k
        }
1004
175k
        else   // ( sbSize == 8 )
1005
175k
        {
1006
1.58M
          for (y = 0; y < 8; y++)
1007
1.40M
          {
1008
1.40M
            lfnstTemp[0]  = coeffTemp[0];
1009
1.40M
            lfnstTemp[8]  = coeffTemp[1];
1010
1.40M
            lfnstTemp[16] = coeffTemp[2];
1011
1.40M
            lfnstTemp[24] = coeffTemp[3];
1012
1.40M
            if (y < 4)
1013
703k
            {
1014
703k
              lfnstTemp[32] = coeffTemp[4];
1015
703k
              lfnstTemp[36] = coeffTemp[5];
1016
703k
              lfnstTemp[40] = coeffTemp[6];
1017
703k
              lfnstTemp[44] = coeffTemp[7];
1018
703k
            }
1019
1.40M
            lfnstTemp++;
1020
1.40M
            coeffTemp += width;
1021
1.40M
          }
1022
175k
        }
1023
248k
      }
1024
917k
      else
1025
917k
      {
1026
6.94M
        for (y = 0; y < sbSize; y++)
1027
6.02M
        {
1028
6.02M
          uint32_t uiStride = (y < 4) ? sbSize : 4;
1029
6.02M
          ::memcpy(lfnstTemp, coeffTemp, uiStride * sizeof(TCoeff));
1030
6.02M
          lfnstTemp += uiStride;
1031
6.02M
          coeffTemp += width;
1032
6.02M
        }
1033
917k
      }
1034
1035
1.16M
      m_fwdLfnstNxN( m_tempInMatrix, m_tempOutMatrix, g_lfnstLut[intraMode], lfnstIdx - 1, sbSize, ( tu4x4Flag || tu8x8Flag ) ? 8 : 16 );
1036
1037
1.16M
      lfnstTemp                        = m_tempOutMatrix;   // forward spectral rearrangement
1038
1.16M
      coeffTemp                        = tempCoeff;
1039
1.16M
      const ScanElement *scanPtr       = scan;
1040
1.16M
      int                lfnstCoeffNum = (sbSize == 4) ? sbSize * sbSize : 48;
1041
44.3M
      for (y = 0; y < lfnstCoeffNum; y++)
1042
43.1M
      {
1043
43.1M
        coeffTemp[scanPtr->idx] = *lfnstTemp++;
1044
43.1M
        scanPtr++;
1045
43.1M
      }
1046
1.16M
    }
1047
1.16M
  }
1048
1.97M
}
1049
1050
void TrQuant::xTransformSkip(const TransformUnit& tu, const ComponentID& compID, const CPelBuf& resi, TCoeff* psCoeff)
1051
114k
{
1052
114k
  const CompArea& rect = tu.blocks[compID];
1053
114k
  const uint32_t width = rect.width;
1054
114k
  const uint32_t height = rect.height;
1055
1056
1.25M
  for (uint32_t y = 0, coefficientIndex = 0; y < height; y++)
1057
1.13M
  {
1058
13.5M
    for (uint32_t x = 0; x < width; x++, coefficientIndex++)
1059
12.4M
    {
1060
12.4M
      psCoeff[coefficientIndex] = TCoeff(resi.at(x, y));
1061
12.4M
    }
1062
1.13M
  }
1063
114k
}
1064
} // namespace vvenc
1065
1066
//! \}
1067