Coverage Report

Created: 2026-09-02 06:43

next uncovered line (L), next uncovered region (R), next uncovered branch (B)
/work/vvenc/source/Lib/EncoderLib/IntraSearch.cpp
Line
Count
Source
1
/* -----------------------------------------------------------------------------
2
The copyright in this software is being made available under the Clear BSD
3
License, included below. No patent rights, trademark rights and/or 
4
other Intellectual Property Rights other than the copyrights concerning 
5
the Software are granted under this license.
6
7
The Clear BSD License
8
9
Copyright (c) 2019-2026, Fraunhofer-Gesellschaft zur Förderung der angewandten Forschung e.V. & The VVenC Authors.
10
All rights reserved.
11
12
Redistribution and use in source and binary forms, with or without modification,
13
are permitted (subject to the limitations in the disclaimer below) provided that
14
the following conditions are met:
15
16
     * Redistributions of source code must retain the above copyright notice,
17
     this list of conditions and the following disclaimer.
18
19
     * Redistributions in binary form must reproduce the above copyright
20
     notice, this list of conditions and the following disclaimer in the
21
     documentation and/or other materials provided with the distribution.
22
23
     * Neither the name of the copyright holder nor the names of its
24
     contributors may be used to endorse or promote products derived from this
25
     software without specific prior written permission.
26
27
NO EXPRESS OR IMPLIED LICENSES TO ANY PARTY'S PATENT RIGHTS ARE GRANTED BY
28
THIS LICENSE. THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND
29
CONTRIBUTORS "AS IS" AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT
30
LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A
31
PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR
32
CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL,
33
EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO,
34
PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR
35
BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER
36
IN CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE)
37
ARISING IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE
38
POSSIBILITY OF SUCH DAMAGE.
39
40
41
------------------------------------------------------------------------------------------- */
42
43
44
/** \file     EncSearch.cpp
45
 *  \brief    encoder intra search class
46
 */
47
48
#include "IntraSearch.h"
49
#include "EncPicture.h"
50
#include "CommonLib/CommonDef.h"
51
#include "CommonLib/Rom.h"
52
#include "CommonLib/Picture.h"
53
#include "CommonLib/UnitTools.h"
54
#include "CommonLib/dtrace_next.h"
55
#include "CommonLib/dtrace_buffer.h"
56
#include <math.h>
57
#include "vvenc/vvencCfg.h"
58
59
//! \ingroup EncoderLib
60
//! \{
61
62
namespace vvenc {
63
64
#define PLTCtx(c) SubCtx( Ctx::Palette, c )
65
66
IntraSearch::IntraSearch()
67
18.3k
  : m_pSaveCS       (nullptr)
68
18.3k
  , m_pcEncCfg      (nullptr)
69
18.3k
  , m_pcTrQuant     (nullptr)
70
18.3k
  , m_pcRdCost      (nullptr)
71
18.3k
  , m_CABACEstimator(nullptr)
72
18.3k
  , m_CtxCache      (nullptr)
73
18.3k
{
74
18.3k
}
75
76
void IntraSearch::init(const VVEncCfg &encCfg, TrQuant *pTrQuant, RdCost *pRdCost, SortedPelUnitBufs<SORTED_BUFS> *pSortedPelUnitBufs, XUCache &unitCache )
77
18.3k
{
78
18.3k
  IntraPrediction::init( encCfg.m_internChromaFormat, encCfg.m_internalBitDepth[ CH_L ] );
79
80
18.3k
  m_pcEncCfg          = &encCfg;
81
18.3k
  m_pcTrQuant         = pTrQuant;
82
18.3k
  m_pcRdCost          = pRdCost;
83
18.3k
  m_SortedPelUnitBufs = pSortedPelUnitBufs;
84
85
18.3k
  const ChromaFormat chrFormat = encCfg.m_internChromaFormat;
86
18.3k
  const int maxCUSize          = encCfg.m_CTUSize;
87
88
18.3k
  Area area = Area( 0, 0, maxCUSize, maxCUSize );
89
90
18.3k
  m_pTempCS = new CodingStructure( unitCache, nullptr );
91
18.3k
  m_pBestCS = new CodingStructure( unitCache, nullptr );
92
93
18.3k
  m_pTempCS->createForSearch( chrFormat, area );
94
18.3k
  m_pBestCS->createForSearch( chrFormat, area );
95
96
18.3k
  const int uiNumSaveLayersToAllocate = 3;
97
18.3k
  m_pSaveCS = new CodingStructure*[uiNumSaveLayersToAllocate];
98
73.4k
  for( int layer = 0; layer < uiNumSaveLayersToAllocate; layer++ )
99
55.0k
  {
100
55.0k
    m_pSaveCS[ layer ] = new CodingStructure( unitCache, nullptr );
101
55.0k
    m_pSaveCS[ layer ]->createForSearch( chrFormat, Area( 0, 0, maxCUSize, maxCUSize ) );
102
55.0k
    m_pSaveCS[ layer ]->initStructData();
103
55.0k
  }
104
105
18.3k
  CompArea chromaArea( COMP_Cb, chrFormat, area, true );
106
110k
  for( int i = 0; i < 5; i++ )
107
91.7k
  {
108
91.7k
    m_orgResiCb[i].create( chromaArea );
109
91.7k
    m_orgResiCr[i].create( chromaArea );
110
91.7k
  }
111
18.3k
}
112
113
void IntraSearch::destroy()
114
18.3k
{
115
18.3k
  if ( m_pSaveCS )
116
18.3k
  {
117
18.3k
    const int uiNumSaveLayersToAllocate = 3;
118
73.4k
    for( int layer = 0; layer < uiNumSaveLayersToAllocate; layer++ )
119
55.0k
    {
120
55.0k
      if ( m_pSaveCS[ layer ] ) { m_pSaveCS[ layer ]->destroy(); delete m_pSaveCS[ layer ]; }
121
55.0k
    }
122
18.3k
    delete[] m_pSaveCS;
123
18.3k
    m_pSaveCS = nullptr;
124
18.3k
  }
125
126
18.3k
  if( m_pTempCS )
127
18.3k
  {
128
18.3k
    m_pTempCS->destroy();
129
18.3k
    delete m_pTempCS; m_pTempCS = nullptr;
130
18.3k
  }
131
132
18.3k
  if( m_pBestCS )
133
18.3k
  {
134
18.3k
    m_pBestCS->destroy();
135
18.3k
    delete m_pBestCS; m_pBestCS = nullptr;
136
18.3k
  }
137
18.3k
}
138
139
IntraSearch::~IntraSearch()
140
18.3k
{
141
18.3k
  destroy();
142
18.3k
}
143
144
void IntraSearch::setCtuEncRsrc( CABACWriter* cabacEstimator, CtxCache *ctxCache )
145
3.59k
{
146
3.59k
  m_CABACEstimator = cabacEstimator;
147
3.59k
  m_CtxCache       = ctxCache;
148
3.59k
}
149
150
//////////////////////////////////////////////////////////////////////////
151
// INTRA PREDICTION
152
//////////////////////////////////////////////////////////////////////////
153
static constexpr double COST_UNKNOWN = -65536.0;
154
155
double IntraSearch::xFindInterCUCost( CodingUnit &cu )
156
23.9k
{
157
23.9k
  if( CU::isConsIntra(cu) && !cu.slice->isIntra() )
158
0
  {
159
    //search corresponding inter CU cost
160
0
    for( int i = 0; i < m_numCuInSCIPU; i++ )
161
0
    {
162
0
      if( cu.lumaPos() == m_cuAreaInSCIPU[i].pos() && cu.lumaSize() == m_cuAreaInSCIPU[i].size() )
163
0
      {
164
0
        return m_cuCostInSCIPU[i];
165
0
      }
166
0
    }
167
0
  }
168
23.9k
  return COST_UNKNOWN;
169
23.9k
}
170
171
void IntraSearch::xEstimateLumaRdModeList(int& numModesForFullRD,
172
  static_vector<ModeInfo, FAST_UDI_MAX_RDMODE_NUM>& RdModeList,
173
  static_vector<ModeInfo, FAST_UDI_MAX_RDMODE_NUM>& HadModeList,
174
  static_vector<double, FAST_UDI_MAX_RDMODE_NUM>& CandCostList,
175
  static_vector<double, FAST_UDI_MAX_RDMODE_NUM>& CandHadList, CodingUnit& cu, bool testMip )
176
23.9k
{
177
23.9k
  PROFILER_SCOPE_AND_STAGE_EXT( 1, _TPROF, P_INTRA_EST_RD_CAND, cu.cs, CH_L );
178
23.9k
  const uint16_t intra_ctx_size = Ctx::IntraLumaMpmFlag.size() + Ctx::IntraLumaPlanarFlag.size() + Ctx::MultiRefLineIdx.size() + Ctx::ISPMode.size() + Ctx::MipFlag.size();
179
23.9k
  const TempCtx  ctxStartIntraCtx(m_CtxCache, SubCtx(CtxSet(Ctx::IntraLumaMpmFlag(), intra_ctx_size), m_CABACEstimator->getCtx()));
180
23.9k
  const double   sqrtLambdaForFirstPass = m_pcRdCost->getMotionLambda() * FRAC_BITS_SCALE;
181
23.9k
  const int numModesAvailable = NUM_LUMA_MODE; // total number of Intra modes
182
183
23.9k
  CHECK(numModesForFullRD >= numModesAvailable, "Too many modes for full RD search");
184
185
23.9k
  const SPS& sps     = *cu.cs->sps;
186
23.9k
  const bool fastMip = sps.MIP && m_pcEncCfg->m_useFastMIP;
187
188
  // this should always be true
189
23.9k
  CHECK( !cu.Y().valid(), "CU is not valid" );
190
191
23.9k
  const CompArea& area = cu.Y();
192
193
23.9k
  const UnitArea localUnitArea(area.chromaFormat, Area(0, 0, area.width, area.height));
194
23.9k
  if( testMip)
195
18.2k
  {
196
18.2k
    numModesForFullRD += fastMip ? numModesForFullRD - std::min( m_pcEncCfg->m_useFastMIP, numModesForFullRD )
197
18.2k
                                 : numModesForFullRD;
198
18.2k
    m_SortedPelUnitBufs->prepare( localUnitArea, numModesForFullRD + 1 );
199
18.2k
  }
200
5.63k
  else
201
5.63k
  {
202
5.63k
    m_SortedPelUnitBufs->prepare( localUnitArea, numModesForFullRD );
203
5.63k
  }
204
205
23.9k
  CPelBuf piOrg   = cu.cs->getOrgBuf(COMP_Y);
206
23.9k
  PelBuf piPred  = m_SortedPelUnitBufs->getTestBuf(COMP_Y);
207
208
23.9k
  DistParam distParam    = m_pcRdCost->setDistParam( piOrg, piPred, sps.bitDepths[ CH_L ], DF_HAD_2SAD); // Use HAD (SATD) cost
209
210
23.9k
  const int numHadCand = (testMip ? 2 : 1) * 3;
211
212
  //*** Derive (regular) candidates using Hadamard
213
23.9k
  cu.mipFlag = false;
214
23.9k
  cu.multiRefIdx = 0;
215
216
  //===== init pattern for luma prediction =====
217
23.9k
  initIntraPatternChType(cu, cu.Y(), true);
218
219
23.9k
  bool satdChecked[NUM_INTRA_MODE] = { false };
220
221
23.9k
  unsigned mpmLst[NUM_MOST_PROBABLE_MODES];
222
23.9k
  CU::getIntraMPMs(cu, mpmLst);
223
224
23.9k
  const int decMsk = ( 1 << m_pcEncCfg->m_IntraEstDecBit ) - 1;
225
226
23.9k
  m_parentCandList.resize( 0 );
227
23.9k
  m_parentCandList.reserve( ( numModesAvailable >> m_pcEncCfg->m_IntraEstDecBit ) + 2 );
228
229
1.62M
  for( unsigned mode = 0; mode < numModesAvailable; mode++ )
230
1.60M
  {
231
    // Skip checking extended Angular modes in the first round of SATD
232
1.60M
    if( mode > DC_IDX && ( mode & decMsk ) )
233
1.17M
    {
234
1.17M
      continue;
235
1.17M
    }
236
237
430k
    m_parentCandList.push_back( ModeInfo( false, false, 0, NOT_INTRA_SUBPARTITIONS, mode ) );
238
430k
  }
239
   
240
95.6k
  for( int decDst = 1 << m_pcEncCfg->m_IntraEstDecBit; decDst > 0; decDst >>= 1 )
241
71.7k
  {
242
645k
    for( unsigned idx = 0; idx < m_parentCandList.size(); idx++ )
243
573k
    {
244
573k
      int modeParent = m_parentCandList[idx].modeId;
245
246
573k
      int off = decDst & decMsk;
247
573k
      int inc = decDst << 1;
248
249
573k
#if 1 // INTRA_AS_IN_VTM
250
573k
      if( off != 0 && ( modeParent <= ( DC_IDX + 1 ) || modeParent >= ( NUM_LUMA_MODE - 1 ) ) )
251
93.2k
      {
252
93.2k
        continue;
253
93.2k
      }
254
255
480k
#endif
256
1.01M
      for( int mode = modeParent - off; mode < modeParent + off + 1; mode += inc )
257
530k
      {
258
530k
        if( satdChecked[mode] || mode < 0 || mode >= NUM_LUMA_MODE )
259
2.27k
        {
260
2.27k
          continue;
261
2.27k
        }
262
263
528k
        cu.intraDir[0] = mode;
264
265
528k
        initPredIntraParams( cu, cu.Y(), sps );
266
528k
        distParam.cur.buf = piPred.buf = m_SortedPelUnitBufs->getTestBuf().Y().buf;
267
528k
        predIntraAng( COMP_Y, piPred, cu );
268
269
        // Use the min between SAD and HAD as the cost criterion
270
        // SAD is scaled by 2 to align with the scaling of HAD
271
528k
        Distortion minSadHad = distParam.distFunc( distParam );
272
273
528k
        uint64_t fracModeBits = xFracModeBitsIntraLuma( cu, mpmLst );
274
275
        //restore ctx
276
528k
        m_CABACEstimator->getCtx() = SubCtx( CtxSet( Ctx::IntraLumaMpmFlag(), intra_ctx_size ), ctxStartIntraCtx );
277
278
528k
        double cost = ( double ) minSadHad + ( double ) fracModeBits * sqrtLambdaForFirstPass;
279
528k
        DTRACE( g_trace_ctx, D_INTRA_COST, "IntraHAD: %u, %llu, %f (%d)\n", minSadHad, fracModeBits, cost, mode );
280
281
528k
        int insertPos = -1;
282
528k
        updateCandList( ModeInfo( false, false, 0, NOT_INTRA_SUBPARTITIONS, mode ), cost, RdModeList, CandCostList, numModesForFullRD, &insertPos );
283
528k
        updateCandList( ModeInfo( false, false, 0, NOT_INTRA_SUBPARTITIONS, mode ), ( double ) minSadHad, HadModeList, CandHadList, numHadCand );
284
528k
        m_SortedPelUnitBufs->insert( insertPos, ( int ) RdModeList.size() );
285
286
528k
        satdChecked[mode] = true;
287
528k
      }
288
480k
    }
289
290
71.7k
    m_parentCandList.resize( RdModeList.size() );
291
71.7k
    std::copy( RdModeList.cbegin(), RdModeList.cend(), m_parentCandList.begin() );
292
71.7k
  }
293
294
23.9k
  const bool isFirstLineOfCtu = (((cu.block(COMP_Y).y)&((cu.cs->sps)->CTUSize - 1)) == 0);
295
23.9k
  if( m_pcEncCfg->m_MRL && ! isFirstLineOfCtu )
296
14.4k
  {
297
14.4k
    cu.multiRefIdx = 1;
298
14.4k
    unsigned  multiRefMPM [NUM_MOST_PROBABLE_MODES];
299
14.4k
    CU::getIntraMPMs(cu, multiRefMPM);
300
301
43.3k
    for (int mRefNum = 1; mRefNum < MRL_NUM_REF_LINES; mRefNum++)
302
28.8k
    {
303
28.8k
      int multiRefIdx = MULTI_REF_LINE_IDX[mRefNum];
304
305
28.8k
      cu.multiRefIdx = multiRefIdx;
306
28.8k
      initIntraPatternChType(cu, cu.Y(), true);
307
308
173k
      for (int x = 1; x < NUM_MOST_PROBABLE_MODES; x++)
309
144k
      {
310
144k
        cu.intraDir[0] = multiRefMPM[x];
311
144k
        initPredIntraParams(cu, cu.Y(), sps);
312
144k
        distParam.cur.buf = piPred.buf = m_SortedPelUnitBufs->getTestBuf().Y().buf;
313
144k
        predIntraAng(COMP_Y, piPred, cu);
314
315
        // Use the min between SAD and SATD as the cost criterion
316
        // SAD is scaled by 2 to align with the scaling of HAD
317
144k
        Distortion minSadHad = distParam.distFunc(distParam);
318
319
        // NB xFracModeBitsIntra will not affect the mode for chroma that may have already been pre-estimated.
320
144k
        uint64_t fracModeBits = xFracModeBitsIntraLuma( cu, mpmLst );
321
322
        //restore ctx
323
144k
        m_CABACEstimator->getCtx() = SubCtx(CtxSet(Ctx::IntraLumaMpmFlag(), intra_ctx_size), ctxStartIntraCtx);
324
325
144k
        double cost = (double) minSadHad + (double) fracModeBits * sqrtLambdaForFirstPass;
326
//        DTRACE(g_trace_ctx, D_INTRA_COST, "IntraMRL: %u, %llu, %f (%d)\n", minSadHad, fracModeBits, cost, cu.intraDir[0]);
327
328
144k
        int insertPos = -1;
329
144k
        updateCandList( ModeInfo( false, false, multiRefIdx, NOT_INTRA_SUBPARTITIONS, cu.intraDir[0] ), cost, RdModeList,  CandCostList, numModesForFullRD, &insertPos );
330
144k
        updateCandList( ModeInfo( false, false, multiRefIdx, NOT_INTRA_SUBPARTITIONS, cu.intraDir[0] ), (double)minSadHad, HadModeList, CandHadList,  numHadCand );
331
144k
        m_SortedPelUnitBufs->insert(insertPos, (int)RdModeList.size());
332
144k
      }
333
28.8k
    }
334
14.4k
    cu.multiRefIdx = 0;
335
14.4k
  }
336
337
23.9k
  if (testMip)
338
18.2k
  {
339
18.2k
    cu.mipFlag = true;
340
18.2k
    cu.multiRefIdx = 0;
341
342
18.2k
    double mipHadCost[MAX_NUM_MIP_MODE] = { MAX_DOUBLE };
343
344
18.2k
    initIntraPatternChType(cu, cu.Y());
345
18.2k
    initIntraMip( cu );
346
347
18.2k
    const int transpOff    = getNumModesMip( cu.Y() );
348
18.2k
    const int numModesFull = (transpOff << 1);
349
239k
    for( uint32_t uiModeFull = 0; uiModeFull < numModesFull; uiModeFull++ )
350
220k
    {
351
220k
      const bool     isTransposed = (uiModeFull >= transpOff ? true : false);
352
220k
      const uint32_t uiMode       = (isTransposed ? uiModeFull - transpOff : uiModeFull);
353
354
220k
      cu.mipTransposedFlag = isTransposed;
355
220k
      cu.intraDir[CH_L] = uiMode;
356
220k
      distParam.cur.buf = piPred.buf = m_SortedPelUnitBufs->getTestBuf().Y().buf;
357
220k
      predIntraMip(piPred, cu);
358
359
      // Use the min between SAD and HAD as the cost criterion
360
      // SAD is scaled by 2 to align with the scaling of HAD
361
220k
      Distortion minSadHad = distParam.distFunc(distParam);
362
363
220k
      uint64_t fracModeBits = xFracModeBitsIntraLuma( cu, mpmLst );
364
365
      //restore ctx
366
220k
      m_CABACEstimator->getCtx() = SubCtx(CtxSet(Ctx::IntraLumaMpmFlag(), intra_ctx_size), ctxStartIntraCtx);
367
368
220k
      double cost = double(minSadHad) + double(fracModeBits) * sqrtLambdaForFirstPass;
369
220k
      mipHadCost[uiModeFull] = cost;
370
220k
      DTRACE(g_trace_ctx, D_INTRA_COST, "IntraMIP: %u, %llu, %f (%d)\n", minSadHad, fracModeBits, cost, uiModeFull);
371
372
220k
      int insertPos = -1;
373
220k
      updateCandList( ModeInfo( true, isTransposed, 0, NOT_INTRA_SUBPARTITIONS, cu.intraDir[0] ), cost, RdModeList,  CandCostList, numModesForFullRD+1, &insertPos );
374
220k
      updateCandList( ModeInfo( true, isTransposed, 0, NOT_INTRA_SUBPARTITIONS, cu.intraDir[0] ), 0.8*(double)minSadHad, HadModeList, CandHadList,  numHadCand );
375
220k
      m_SortedPelUnitBufs->insert(insertPos, (int)RdModeList.size());
376
220k
    }
377
378
18.2k
    const double thresholdHadCost = 1.0 + 1.4 / sqrt((double)(cu.lwidth()*cu.lheight()));
379
18.2k
    xReduceHadCandList(RdModeList, CandCostList, *m_SortedPelUnitBufs, numModesForFullRD, thresholdHadCost, mipHadCost, cu, fastMip);
380
18.2k
  }
381
382
23.9k
  if( m_pcEncCfg->m_bFastUDIUseMPMEnabled )
383
23.9k
  {
384
23.9k
    const int numMPMs = NUM_MOST_PROBABLE_MODES;
385
23.9k
    unsigned  intraMpms[numMPMs];
386
387
23.9k
    cu.multiRefIdx = 0;
388
389
23.9k
    const int numCand = CU::getIntraMPMs( cu, intraMpms );
390
23.9k
    ModeInfo mostProbableMode(false, false, 0, NOT_INTRA_SUBPARTITIONS, 0);
391
392
48.5k
    for( int j = 0; j < numCand; j++ )
393
24.6k
    {
394
24.6k
      bool mostProbableModeIncluded = false;
395
24.6k
      mostProbableMode.modeId = intraMpms[j];
396
397
125k
      for( int i = 0; i < numModesForFullRD; i++ )
398
100k
      {
399
100k
        mostProbableModeIncluded |= ( mostProbableMode == RdModeList[i] );
400
100k
      }
401
24.6k
      if( !mostProbableModeIncluded )
402
180
      {
403
180
        numModesForFullRD++;
404
180
        RdModeList.push_back( mostProbableMode );
405
180
        CandCostList.push_back(0);
406
180
      }
407
24.6k
    }
408
23.9k
  }
409
23.9k
}
410
411
bool IntraSearch::estIntraPredLumaQT(CodingUnit &cu, Partitioner &partitioner, double bestCost)
412
23.9k
{
413
23.9k
  CodingStructure       &cs           = *cu.cs;
414
23.9k
  const int             width         = partitioner.currArea().lwidth();
415
23.9k
  const int             height        = partitioner.currArea().lheight();
416
417
  //===== loop over partitions =====
418
419
23.9k
  const TempCtx ctxStart           ( m_CtxCache, m_CABACEstimator->getCtx() );
420
421
  // variables for saving fast intra modes scan results across multiple LFNST passes
422
23.9k
  double costInterCU = xFindInterCUCost( cu );
423
424
23.9k
  bool validReturn = false;
425
426
  //===== determine set of modes to be tested (using prediction signal only) =====
427
23.9k
  int numModesAvailable = NUM_LUMA_MODE; // total number of Intra modes
428
23.9k
  static_vector<ModeInfo, FAST_UDI_MAX_RDMODE_NUM> RdModeList;
429
23.9k
  static_vector<ModeInfo, FAST_UDI_MAX_RDMODE_NUM> HadModeList;
430
23.9k
  static_vector<double, FAST_UDI_MAX_RDMODE_NUM> CandCostList;
431
23.9k
  static_vector<double, FAST_UDI_MAX_RDMODE_NUM> CandHadList;
432
433
23.9k
  int numModesForFullRD = g_aucIntraModeNumFast_UseMPM_2D[Log2(width) - MIN_CU_LOG2][Log2(height) - MIN_CU_LOG2];
434
23.9k
  if (m_pcEncCfg->m_numIntraModesFullRD > 0)
435
0
    numModesForFullRD=m_pcEncCfg->m_numIntraModesFullRD;
436
437
#if INTRA_FULL_SEARCH
438
  numModesForFullRD = numModesAvailable;
439
#endif
440
23.9k
  const SPS& sps = *cu.cs->sps;
441
23.9k
  const bool mipAllowed = sps.MIP && cu.lwidth() <= sps.getMaxTbSize() && cu.lheight() <= sps.getMaxTbSize() && ((cu.lfnstIdx == 0) || allowLfnstWithMip(cu.lumaSize()));
442
23.9k
  const int SizeThr     = 8 >> std::max( 0, m_pcEncCfg->m_useFastMIP - 1 );
443
23.9k
  const bool testMip    = mipAllowed && ( cu.lwidth() <= ( SizeThr * cu.lheight() ) && cu.lheight() <= ( SizeThr * cu.lwidth() ) ) && ( cu.lwidth() <= MIP_MAX_WIDTH && cu.lheight() <= MIP_MAX_HEIGHT );
444
23.9k
  bool testISP = sps.ISP && CU::canUseISP(width, height, cu.cs->sps->getMaxTbSize());
445
23.9k
  if (testISP)
446
23.9k
  {
447
23.9k
    int numTotalPartsHor = (int)width >> floorLog2(CU::getISPSplitDim(width, height, TU_1D_VERT_SPLIT));
448
23.9k
    int numTotalPartsVer = (int)height >> floorLog2(CU::getISPSplitDim(width, height, TU_1D_HORZ_SPLIT));
449
23.9k
    m_ispTestedModes[0].init(numTotalPartsHor, numTotalPartsVer, 0);
450
    // the total number of subpartitions is modified to take into account the cases where LFNST cannot be combined with
451
    // ISP due to size restrictions
452
23.9k
    numTotalPartsHor = sps.LFNST && CU::canUseLfnstWithISP(cu.Y(), HOR_INTRA_SUBPARTITIONS) ? numTotalPartsHor : 0;
453
23.9k
    numTotalPartsVer = sps.LFNST && CU::canUseLfnstWithISP(cu.Y(), VER_INTRA_SUBPARTITIONS) ? numTotalPartsVer : 0;
454
71.7k
    for (int j = 1; j < NUM_LFNST_NUM_PER_SET; j++)
455
47.8k
    {
456
47.8k
      m_ispTestedModes[j].init(numTotalPartsHor, numTotalPartsVer, 0);
457
47.8k
    }
458
23.9k
    testISP = m_ispTestedModes[0].numTotalParts[0];
459
23.9k
  }
460
0
  else
461
0
  {
462
0
    m_ispTestedModes[0].init(0, 0, 0);
463
0
  }
464
465
23.9k
  xEstimateLumaRdModeList(numModesForFullRD, RdModeList, HadModeList, CandCostList, CandHadList, cu, testMip);
466
467
23.9k
  CHECK( (size_t)numModesForFullRD != RdModeList.size(), "Inconsistent state!" );
468
469
  // after this point, don't use numModesForFullRD
470
23.9k
  if( m_pcEncCfg->m_usePbIntraFast && !cs.slice->isIntra() && RdModeList.size() < numModesAvailable )
471
0
  {
472
0
    double pbintraRatio = m_pcEncCfg->m_usePbIntraFast == 1 && ( cs.area.lwidth() >= 16 && cs.area.lheight() >= 16 ) ? 1.2 : PBINTRA_RATIO;
473
474
0
    int maxSize = -1;
475
0
    ModeInfo bestMipMode;
476
0
    int bestMipIdx = -1;
477
0
    for( int idx = 0; idx < RdModeList.size(); idx++ )
478
0
    {
479
0
      if( RdModeList[idx].mipFlg )
480
0
      {
481
0
        bestMipMode = RdModeList[idx];
482
0
        bestMipIdx = idx;
483
0
        break;
484
0
      }
485
0
    }
486
0
    const int numHadCand = 3;
487
0
    for (int k = numHadCand - 1; k >= 0; k--)
488
0
    {
489
0
      if (CandHadList.size() < (k + 1) || CandHadList[k] > cs.interHad * pbintraRatio) { maxSize = k; }
490
0
    }
491
0
    if (maxSize > 0)
492
0
    {
493
0
      RdModeList.resize(std::min<size_t>(RdModeList.size(), maxSize));
494
0
      if( bestMipIdx >= 0 )
495
0
      {
496
0
        if( RdModeList.size() <= bestMipIdx )
497
0
        {
498
0
          RdModeList.push_back(bestMipMode);
499
0
          m_SortedPelUnitBufs->swap( maxSize, bestMipIdx );
500
0
        }
501
0
      }
502
0
    }
503
0
    if (maxSize == 0)
504
0
    {
505
0
      cs.dist = MAX_DISTORTION;
506
0
      cs.interHad = 0;
507
0
      return false;
508
0
    }
509
0
  }
510
511
  //===== check modes (using r-d costs) =====
512
23.9k
  ModeInfo bestPUMode;
513
514
23.9k
  CodingStructure *csTemp = m_pTempCS;
515
23.9k
  CodingStructure *csBest = m_pBestCS;
516
517
23.9k
  csTemp->slice   = csBest->slice   = cs.slice;
518
23.9k
  csTemp->picture = csBest->picture = cs.picture;
519
23.9k
  csTemp->compactResize( cu );
520
23.9k
  csBest->compactResize( cu );
521
23.9k
  csTemp->initStructData();
522
23.9k
  csBest->initStructData();
523
524
23.9k
  int   bestLfnstIdx  = 0;
525
23.9k
  const bool useBDPCM = cs.picture->useBDPCM;
526
23.9k
  int   NumBDPCMCand  = (useBDPCM && sps.BDPCM && CU::bdpcmAllowed(cu, ComponentID(partitioner.chType))) ? 2 : 0;
527
23.9k
  int   bestbdpcmMode = 0;
528
23.9k
  int   bestISP       = 0;
529
23.9k
  int   bestMrl       = 0;
530
23.9k
  bool  bestMip       = 0;
531
23.9k
  int   EndMode       = (int)RdModeList.size();
532
23.9k
  bool  useISPlfnst   = testISP && sps.LFNST;
533
23.9k
  bool  noLFNST_ts    = false;
534
23.9k
  double bestCostIsp[2] = { MAX_DOUBLE, MAX_DOUBLE };
535
23.9k
  bool disableMTS = false;
536
23.9k
  bool disableLFNST = false;
537
23.9k
  bool disableDCT2test = false;
538
23.9k
  if (m_pcEncCfg->m_FastIntraTools)
539
23.9k
  {
540
23.9k
    int speedIntra = 0;
541
23.9k
    xSpeedUpIntra(bestCost, EndMode, speedIntra, cu);
542
23.9k
    disableMTS = (speedIntra >> 2 ) & 0x1;
543
23.9k
    disableLFNST = (speedIntra >> 1) & 0x1;
544
23.9k
    disableDCT2test = speedIntra>>3;
545
23.9k
    if (disableLFNST)
546
21.3k
    {
547
21.3k
      noLFNST_ts = true;
548
21.3k
      useISPlfnst = false;
549
21.3k
    }
550
23.9k
    if (speedIntra & 0x1)
551
21.3k
    {
552
21.3k
      testISP = false;
553
21.3k
    }
554
23.9k
  }
555
556
128k
  for (int mode_cur = 0; mode_cur < EndMode + NumBDPCMCand; mode_cur++)
557
104k
  {
558
104k
    int mode = mode_cur;
559
104k
    if (mode_cur >= EndMode)
560
6.90k
    {
561
6.90k
      mode = mode_cur - EndMode ? -1 : -2;
562
6.90k
      testISP = false;
563
6.90k
    }
564
    // set CU/PU to luma prediction mode
565
104k
    ModeInfo testMode;
566
104k
    int noISP = 0;
567
104k
    int endISP = testISP ? 2 : 0;
568
104k
    bool noLFNST = false || noLFNST_ts;
569
104k
    if (mode && useISPlfnst)
570
8.47k
    {
571
8.47k
      noLFNST |= (bestCostIsp[0] > (bestCostIsp[1] * 1.4));
572
8.47k
      if (mode > 2)
573
2.29k
      {
574
2.29k
        endISP = 0;
575
2.29k
        testISP = false;
576
2.29k
      }
577
8.47k
    }
578
104k
    if (testISP)
579
5.36k
    {
580
5.36k
      xSpeedUpISP(1, testISP, mode, noISP, endISP, cu, RdModeList, bestPUMode, bestISP, bestLfnstIdx);
581
5.36k
    }
582
104k
    int startISP = 0;
583
104k
    if (disableDCT2test && mode && bestISP)
584
0
    {
585
0
      startISP = endISP ? 1 : 0;
586
0
    }
587
217k
    for (int ispM = startISP; ispM <= endISP; ispM++)
588
112k
    {
589
112k
      if (ispM && (ispM == noISP))
590
54
      {
591
54
        continue;
592
54
      }
593
594
112k
      if (mode < 0)
595
6.90k
      {
596
6.90k
        cu.bdpcmM[CH_L] = -mode;
597
6.90k
        testMode = ModeInfo(false, false, 0, NOT_INTRA_SUBPARTITIONS, cu.bdpcmM[CH_L] == 2 ? VER_IDX : HOR_IDX);
598
6.90k
      }
599
105k
      else
600
105k
      {
601
105k
        testMode = RdModeList[mode];
602
105k
        cu.bdpcmM[CH_L] = 0;
603
105k
      }
604
605
112k
      cu.ispMode = ispM;
606
112k
      cu.mipFlag = testMode.mipFlg;
607
112k
      cu.mipTransposedFlag = testMode.mipTrFlg;
608
112k
      cu.multiRefIdx = testMode.mRefId;
609
112k
      cu.intraDir[CH_L] = testMode.modeId;
610
112k
      if (cu.ispMode && xSpeedUpISP(0, testISP, mode, noISP, endISP, cu, RdModeList, bestPUMode, bestISP, 0) )
611
2.80k
      {
612
2.80k
        continue;
613
2.80k
      }
614
109k
      if (m_pcEncCfg->m_FastIntraTools && (cu.ispMode || sps.LFNST || sps.MTS))
615
109k
      {
616
109k
        m_ispTestedModes[0].intraWasTested = true;
617
109k
      }
618
109k
      CHECK(cu.mipFlag && cu.multiRefIdx, "Error: combination of MIP and MRL not supported");
619
109k
      CHECK(cu.multiRefIdx && (cu.intraDir[0] == PLANAR_IDX), "Error: combination of MRL and Planar mode not supported");
620
109k
      CHECK(cu.ispMode && cu.mipFlag, "Error: combination of ISP and MIP not supported");
621
109k
      CHECK(cu.ispMode && cu.multiRefIdx, "Error: combination of ISP and MRL not supported");
622
623
      // determine residual for partition
624
109k
      cs.initSubStructure(*csTemp, partitioner.chType, cs.area, true);
625
109k
      int doISP = (((cu.ispMode == 0) && noLFNST) || (useISPlfnst && mode && cu.ispMode && (bestLfnstIdx == 0)) || disableLFNST) ? -mode : mode;
626
109k
      xIntraCodingLumaQT(*csTemp, partitioner, m_SortedPelUnitBufs->getBufFromSortedList(mode), bestCost, doISP, disableMTS);
627
628
109k
      DTRACE(g_trace_ctx, D_INTRA_COST, "IntraCost T [x=%d,y=%d,w=%d,h=%d] %f (%d,%d,%d,%d,%d,%d) \n", cu.blocks[0].x,
629
109k
        cu.blocks[0].y, width, height, csTemp->cost, testMode.modeId, testMode.ispMod,
630
109k
        cu.multiRefIdx, cu.mipFlag, cu.lfnstIdx, cu.mtsFlag);
631
632
109k
      if (cu.ispMode && !csTemp->cus[0]->firstTU->cbf[COMP_Y])
633
1.75k
      {
634
1.75k
        csTemp->cost = MAX_DOUBLE;
635
1.75k
        csTemp->costDbOffset = 0;
636
1.75k
      }
637
109k
      if (useISPlfnst)
638
16.0k
      {
639
16.0k
        int n = (cu.ispMode == 0) ? 0 : 1;
640
16.0k
        bestCostIsp[n] = csTemp->cost < bestCostIsp[n] ? csTemp->cost : bestCostIsp[n];
641
16.0k
      }
642
643
      // check r-d cost
644
109k
      if (csTemp->cost < csBest->cost)
645
30.3k
      {
646
30.3k
        validReturn   = true;
647
30.3k
        std::swap(csTemp, csBest);
648
30.3k
        bestPUMode    = testMode;
649
30.3k
        bestLfnstIdx  = csBest->cus[0]->lfnstIdx;
650
30.3k
        bestISP       = csBest->cus[0]->ispMode;
651
30.3k
        bestMip       = csBest->cus[0]->mipFlag;
652
30.3k
        bestMrl       = csBest->cus[0]->multiRefIdx;
653
30.3k
        bestbdpcmMode = cu.bdpcmM[CH_L];
654
30.3k
        m_ispTestedModes[bestLfnstIdx].bestSplitSoFar = ISPType(bestISP);
655
30.3k
        if (csBest->cost < bestCost)
656
30.3k
        {
657
30.3k
          bestCost = csBest->cost;
658
30.3k
        }
659
30.3k
        if ((csBest->getTU(partitioner.chType)->mtsIdx[COMP_Y] == MTS_SKIP) && ( floorLog2(csBest->getTU(partitioner.chType)->blocks[COMP_Y].area()) >= 6 ))
660
4.14k
        {
661
4.14k
          noLFNST_ts = 1;
662
4.14k
        }
663
30.3k
      }
664
665
      // reset context models
666
109k
      m_CABACEstimator->getCtx() = ctxStart;
667
668
109k
      csTemp->releaseIntermediateData();
669
670
109k
      if (m_pcEncCfg->m_fastLocalDualTreeMode && CU::isConsIntra(cu) && !cu.slice->isIntra() && csBest->cost != MAX_DOUBLE && costInterCU != COST_UNKNOWN && mode >= 0)
671
0
      {
672
0
        if( (m_pcEncCfg->m_fastLocalDualTreeMode == 2) || (csBest->cost > costInterCU * 1.5))
673
0
        {
674
          //Note: only try one intra mode, which is especially useful to reduce EncT for LDB case (around 4%)
675
0
          EndMode = 0;
676
0
          break;
677
0
        }
678
0
      }
679
109k
    }
680
104k
  } // Mode loop
681
682
23.9k
  if (m_pcEncCfg->m_FastIntraTools && (sps.ISP|| sps.LFNST || sps.MTS))
683
23.9k
  {
684
23.9k
    int bestMode = csBest->getTU(partitioner.chType)->mtsIdx[COMP_Y] ? 4 : 0;
685
23.9k
    bestMode |= bestLfnstIdx ? 2 : 0;
686
23.9k
    bestMode |= bestISP ? 1 : 0;
687
23.9k
    m_ispTestedModes[0].bestIntraMode = bestMode;
688
23.9k
  }
689
23.9k
  cu.ispMode = bestISP;
690
23.9k
  if( validReturn )
691
23.9k
  {
692
23.9k
    cs.useSubStructure( *csBest, partitioner.chType, TREE_D, cu.singleChan( CH_L ), true );
693
694
    //=== update PU data ====
695
23.9k
    cu.lfnstIdx           = bestLfnstIdx;
696
23.9k
    cu.mipTransposedFlag  = bestPUMode.mipTrFlg;
697
23.9k
    cu.intraDir[CH_L]     = bestPUMode.modeId;
698
23.9k
    cu.bdpcmM[CH_L]       = bestbdpcmMode;
699
23.9k
    cu.mipFlag            = bestMip;
700
23.9k
    cu.multiRefIdx        = bestMrl;
701
23.9k
  }
702
0
  else
703
0
  {
704
0
    THROW("fix this");
705
0
  }
706
707
23.9k
  csBest->releaseIntermediateData();
708
709
23.9k
  return validReturn;
710
23.9k
}
711
712
void IntraSearch::estIntraPredChromaQT( CodingUnit& cu, Partitioner& partitioner, const double maxCostAllowed )
713
54.0k
{
714
54.0k
  PROFILER_SCOPE_AND_STAGE_EXT( 0, _TPROF, P_INTRA_CHROMA, cu.cs, CH_C );
715
54.0k
  const TempCtx ctxStart( m_CtxCache, m_CABACEstimator->getCtx() );
716
54.0k
  CodingStructure &cs   = *cu.cs;
717
54.0k
  bool lumaUsesISP      = !CU::isSepTree(cu) && cu.ispMode;
718
54.0k
  PartSplit ispType     = lumaUsesISP ? CU::getISPType(cu, COMP_Y) : TU_NO_ISP;
719
54.0k
  double bestCostSoFar  = maxCostAllowed;
720
54.0k
  const uint32_t numberValidComponents = getNumberValidComponents( cu.chromaFormat );
721
54.0k
  const bool useBDPCM   = cs.picture->useBDPCM;
722
723
54.0k
  uint32_t   uiBestMode = 0;
724
54.0k
  Distortion uiBestDist = 0;
725
54.0k
  double     dBestCost  = MAX_DOUBLE;
726
727
  //----- init mode list ----
728
54.0k
  {
729
54.0k
    uint32_t  uiMinMode = 0;
730
54.0k
    uint32_t  uiMaxMode = NUM_CHROMA_MODE;
731
732
54.0k
    const int reducedModeNumber = uiMaxMode >> (m_pcEncCfg->m_reduceIntraChromaModesFullRD ? 1 : 2);
733
    //----- check chroma modes -----
734
54.0k
    uint32_t chromaCandModes[ NUM_CHROMA_MODE ];
735
54.0k
    CU::getIntraChromaCandModes( cu, chromaCandModes );
736
737
    // create a temporary CS
738
54.0k
    CodingStructure &saveCS = *m_pSaveCS[0];
739
54.0k
    saveCS.pcv      = cs.pcv;
740
54.0k
    saveCS.picture  = cs.picture;
741
54.0k
    saveCS.area.repositionTo( cs.area );
742
54.0k
    saveCS.clearTUs();
743
744
54.0k
    if( !CU::isSepTree(cu) && cu.ispMode )
745
0
    {
746
0
      saveCS.clearCUs();
747
0
    }
748
749
54.0k
    if( CU::isSepTree(cu) )
750
54.0k
    {
751
54.0k
      if( partitioner.canSplit( TU_MAX_TR_SPLIT, cs ) )
752
0
      {
753
0
        partitioner.splitCurrArea( TU_MAX_TR_SPLIT, cs );
754
755
0
        do
756
0
        {
757
0
          cs.addTU( CS::getArea( cs, partitioner.currArea(), partitioner.chType, partitioner.treeType ), partitioner.chType, &cu ).depth = partitioner.currTrDepth;
758
0
        } while( partitioner.nextPart( cs ) );
759
760
0
        partitioner.exitCurrSplit();
761
0
      }
762
54.0k
      else
763
54.0k
        cs.addTU( CS::getArea( cs, partitioner.currArea(), partitioner.chType, partitioner.treeType ), partitioner.chType, &cu );
764
54.0k
    }
765
766
    // create a store for the TUs
767
54.0k
    std::vector<TransformUnit*> orgTUs;
768
54.0k
    for( const auto &ptu : cs.tus )
769
54.0k
    {
770
      // for split TUs in HEVC, add the TUs without Chroma parts for correct setting of Cbfs
771
54.0k
      if (lumaUsesISP || cu.contains(*ptu, CH_C))
772
54.0k
      {
773
54.0k
        saveCS.addTU( *ptu, partitioner.chType, nullptr );
774
54.0k
        orgTUs.push_back( ptu );
775
54.0k
      }
776
54.0k
    }
777
778
    // SATD pre-selecting.
779
54.0k
    int     satdModeList  [NUM_CHROMA_MODE] = { 0 };
780
54.0k
    int64_t satdSortedCost[NUM_CHROMA_MODE] = { 0 };
781
54.0k
    bool    modeDisable[NUM_INTRA_MODE + 1] = { false }; // use intra mode idx to check whether enable
782
783
54.0k
    CodingStructure& cs = *(cu.cs);
784
54.0k
    CompArea areaCb = cu.Cb();
785
54.0k
    CompArea areaCr = cu.Cr();
786
54.0k
    CPelBuf orgCb  = cs.getOrgBuf (COMP_Cb);
787
54.0k
    PelBuf predCb  = cs.getPredBuf(COMP_Cb);
788
54.0k
    CPelBuf orgCr  = cs.getOrgBuf (COMP_Cr);
789
54.0k
    PelBuf predCr  = cs.getPredBuf(COMP_Cr);
790
791
54.0k
    DistParam distParamSadCb  = m_pcRdCost->setDistParam( orgCb, predCb, cu.cs->sps->bitDepths[ CH_C ], DF_SAD);
792
54.0k
    DistParam distParamSatdCb = m_pcRdCost->setDistParam( orgCb, predCb, cu.cs->sps->bitDepths[ CH_C ], DF_HAD);
793
54.0k
    DistParam distParamSadCr  = m_pcRdCost->setDistParam( orgCr, predCr, cu.cs->sps->bitDepths[ CH_C ], DF_SAD);
794
54.0k
    DistParam distParamSatdCr = m_pcRdCost->setDistParam( orgCr, predCr, cu.cs->sps->bitDepths[ CH_C ], DF_HAD);
795
796
54.0k
    cu.intraDir[1] = MDLM_L_IDX; // temporary assigned, just to indicate this is a MDLM mode. for luma down-sampling operation.
797
798
54.0k
    initIntraPatternChType(cu, cu.Cb());
799
54.0k
    initIntraPatternChType(cu, cu.Cr());
800
54.0k
    loadLMLumaRecPels(cu, cu.Cb());
801
802
486k
    for (int idx = uiMinMode; idx < uiMaxMode; idx++)
803
432k
    {
804
432k
      int mode = chromaCandModes[idx];
805
432k
      satdModeList[idx] = mode;
806
432k
      if (CU::isLMCMode(mode) && ( !CU::isLMCModeEnabled(cu, mode) || cu.slice->lmChromaCheckDisable ) )
807
47.1k
      {
808
47.1k
        continue;
809
47.1k
      }
810
385k
      if ((mode == LM_CHROMA_IDX) || (mode == PLANAR_IDX) || (mode == DM_CHROMA_IDX)) // only pre-check regular modes and MDLM modes, not including DM ,Planar, and LM
811
94.1k
      {
812
94.1k
        continue;
813
94.1k
      }
814
815
291k
      cu.intraDir[1]    = mode; // temporary assigned, for SATD checking.
816
817
291k
      const bool isLMCMode = CU::isLMCMode(mode);
818
291k
      if( isLMCMode )
819
76.7k
      {
820
76.7k
        predIntraChromaLM(COMP_Cb, predCb, cu, areaCb, mode);
821
76.7k
      }
822
214k
      else
823
214k
      {
824
214k
        initPredIntraParams(cu, cu.Cb(), *cs.sps);
825
214k
        predIntraAng(COMP_Cb, predCb, cu);
826
214k
      }
827
291k
      int64_t sadCb = distParamSadCb.distFunc(distParamSadCb) * 2;
828
291k
      int64_t satdCb = distParamSatdCb.distFunc(distParamSatdCb);
829
291k
      int64_t sad = std::min(sadCb, satdCb);
830
831
291k
      if( isLMCMode )
832
76.7k
      {
833
76.7k
        predIntraChromaLM(COMP_Cr, predCr, cu, areaCr, mode);
834
76.7k
      }
835
214k
      else
836
214k
      {
837
214k
        initPredIntraParams(cu, cu.Cr(), *cs.sps);
838
214k
        predIntraAng(COMP_Cr, predCr, cu);
839
214k
      }
840
291k
      int64_t sadCr = distParamSadCr.distFunc(distParamSadCr) * 2;
841
291k
      int64_t satdCr = distParamSatdCr.distFunc(distParamSatdCr);
842
291k
      sad += std::min(sadCr, satdCr);
843
291k
      satdSortedCost[idx] = sad;
844
291k
    }
845
846
    // sort the mode based on the cost from small to large.
847
486k
    for (int i = uiMinMode; i <= uiMaxMode - 1; i++)
848
432k
    {
849
1.94M
      for (int j = i + 1; j <= uiMaxMode - 1; j++)
850
1.51M
      {
851
1.51M
        if (satdSortedCost[j] < satdSortedCost[i])
852
93.6k
        {
853
93.6k
          std::swap( satdModeList[i],   satdModeList[j]);
854
93.6k
          std::swap( satdSortedCost[i], satdSortedCost[j]);
855
93.6k
        }
856
1.51M
      }
857
432k
    }
858
859
270k
    for (int i = 0; i < reducedModeNumber; i++)
860
216k
    {
861
216k
      modeDisable[satdModeList[uiMaxMode - 1 - i]] = true; // disable the last reducedModeNumber modes
862
216k
    }
863
864
54.0k
    int bestLfnstIdx = 0;
865
    // save the dist
866
54.0k
    Distortion baseDist = cs.dist;
867
54.0k
    int32_t bestbdpcmMode = 0;
868
54.0k
    uint32_t numbdpcmModes = ( useBDPCM && CU::bdpcmAllowed(cu, COMP_Cb)
869
36.2k
        && ((partitioner.chType == CH_C) || (cu.ispMode == 0 && cu.lfnstIdx == 0 && cu.firstTU->mtsIdx[COMP_Y] == MTS_SKIP))) ? 2 : 0;
870
558k
    for (int mode_cur = uiMinMode; mode_cur < (int)(uiMaxMode + numbdpcmModes); mode_cur++)
871
504k
    {
872
504k
      int mode = mode_cur;
873
504k
      if (mode_cur >= uiMaxMode)
874
72.4k
      {
875
72.4k
        mode = mode_cur > uiMaxMode ? -1 : -2; //set bdpcm mode
876
72.4k
        if ((mode == -1) && (saveCS.tus[0]->mtsIdx[COMP_Cb] != MTS_SKIP) && (saveCS.tus[0]->mtsIdx[COMP_Cr] != MTS_SKIP))
877
36.2k
        {
878
36.2k
          continue;
879
36.2k
        }
880
72.4k
      }
881
468k
      int chromaIntraMode;
882
468k
      if (mode < 0)
883
36.2k
      {
884
36.2k
        cu.bdpcmM[CH_C] = -mode;
885
36.2k
        chromaIntraMode = cu.bdpcmM[CH_C] == 2 ? chromaCandModes[1] : chromaCandModes[2];
886
36.2k
      }
887
432k
      else
888
432k
      {
889
432k
        cu.bdpcmM[CH_C] = 0;
890
432k
        chromaIntraMode = chromaCandModes[mode];
891
432k
        if (CU::isLMCMode(chromaIntraMode) && ( !CU::isLMCModeEnabled(cu, chromaIntraMode) || cu.slice->lmChromaCheckDisable ) )
892
47.1k
        {
893
47.1k
          continue;
894
47.1k
        }
895
385k
        if (modeDisable[chromaIntraMode] && CU::isLMCModeEnabled(cu, chromaIntraMode)) // when CCLM is disable, then MDLM is disable. not use satd checking
896
153k
        {
897
153k
          continue;
898
153k
        }
899
385k
      }
900
268k
      cs.dist = baseDist;
901
      //----- restore context models -----
902
268k
      m_CABACEstimator->getCtx() = ctxStart;
903
904
      //----- chroma coding -----
905
268k
      cu.intraDir[1] = chromaIntraMode;
906
268k
      m_ispTestedModes[0].IspType = ispType;
907
268k
      m_ispTestedModes[0].subTuCounter = -1;
908
268k
      xIntraChromaCodingQT( cs, partitioner );
909
268k
      if (lumaUsesISP && cs.dist == MAX_UINT)
910
0
      {
911
0
        continue;
912
0
      }
913
914
268k
      if (cs.sps->transformSkip)
915
268k
      {
916
268k
        m_CABACEstimator->getCtx() = ctxStart;
917
268k
      }
918
268k
      m_ispTestedModes[0].IspType = ispType;
919
268k
      m_ispTestedModes[0].subTuCounter = -1;
920
268k
      uint64_t fracBits   = xGetIntraFracBitsQT( cs, partitioner, false );
921
268k
      Distortion uiDist = cs.dist;
922
268k
      double    dCost   = m_pcRdCost->calcRdCost( fracBits, uiDist - baseDist );
923
924
      //----- compare -----
925
268k
      if( dCost < dBestCost )
926
97.1k
      {
927
97.1k
        if (lumaUsesISP && (dCost < bestCostSoFar))
928
0
        {
929
0
          bestCostSoFar = dCost;
930
0
        }
931
291k
        for( uint32_t i = getFirstComponentOfChannel( CH_C ); i < numberValidComponents; i++ )
932
194k
        {
933
194k
          const CompArea& area = cu.blocks[i];
934
194k
          saveCS.getRecoBuf     ( area ).copyFrom( cs.getRecoBuf   ( area ) );
935
194k
          cs.picture->getRecoBuf( area ).copyFrom( cs.getRecoBuf   ( area ) );
936
388k
          for( uint32_t j = 0; j < saveCS.tus.size(); j++ )
937
194k
          {
938
194k
            saveCS.tus[j]->copyComponentFrom( *orgTUs[j], area.compID );
939
194k
          }
940
194k
        }
941
97.1k
        dBestCost    = dCost;
942
97.1k
        uiBestDist   = uiDist;
943
97.1k
        uiBestMode   = chromaIntraMode;
944
97.1k
        bestLfnstIdx = cu.lfnstIdx;
945
97.1k
        bestbdpcmMode = cu.bdpcmM[CH_C];
946
947
97.1k
      }
948
268k
    }
949
54.0k
    cu.lfnstIdx = bestLfnstIdx;
950
54.0k
    cu.bdpcmM[CH_C]= bestbdpcmMode;
951
952
162k
    for( uint32_t i = getFirstComponentOfChannel( CH_C ); i < numberValidComponents; i++ )
953
108k
    {
954
108k
      const CompArea& area = cu.blocks[i];
955
956
108k
      cs.getRecoBuf         ( area ).copyFrom( saveCS.getRecoBuf( area ) );
957
108k
      cs.picture->getRecoBuf( area ).copyFrom( cs.getRecoBuf    ( area ) );
958
959
216k
      for( uint32_t j = 0; j < saveCS.tus.size(); j++ )
960
108k
      {
961
108k
        orgTUs[ j ]->copyComponentFrom( *saveCS.tus[ j ], area.compID );
962
108k
      }
963
108k
    }
964
54.0k
  }
965
54.0k
  cu.intraDir[1] = uiBestMode;
966
54.0k
  cs.dist        = uiBestDist;
967
968
  //----- restore context models -----
969
54.0k
  m_CABACEstimator->getCtx() = ctxStart;
970
54.0k
  if (lumaUsesISP && bestCostSoFar >= maxCostAllowed)
971
0
  {
972
0
    cu.ispMode = 0;
973
0
  }
974
54.0k
}
975
976
void IntraSearch::saveCuAreaCostInSCIPU( Area area, double cost )
977
0
{
978
0
  if( m_numCuInSCIPU < NUM_INTER_CU_INFO_SAVE )
979
0
  {
980
0
    m_cuAreaInSCIPU[m_numCuInSCIPU] = area;
981
0
    m_cuCostInSCIPU[m_numCuInSCIPU] = cost;
982
0
    m_numCuInSCIPU++;
983
0
  }
984
0
}
985
986
void IntraSearch::initCuAreaCostInSCIPU()
987
0
{
988
0
  for( int i = 0; i < NUM_INTER_CU_INFO_SAVE; i++ )
989
0
  {
990
0
    m_cuAreaInSCIPU[i] = Area();
991
0
    m_cuCostInSCIPU[i] = 0;
992
0
  }
993
0
  m_numCuInSCIPU = 0;
994
0
}
995
// -------------------------------------------------------------------------------------------------------------------
996
// Intra search
997
// -------------------------------------------------------------------------------------------------------------------
998
999
void IntraSearch::xEncIntraHeader( CodingStructure &cs, Partitioner &partitioner, const bool luma )
1000
445k
{
1001
445k
  CodingUnit &cu = *cs.getCU( partitioner.chType, partitioner.treeType );
1002
1003
445k
  if (luma)
1004
177k
  {
1005
177k
    bool isFirst = cu.ispMode ? m_ispTestedModes[0].subTuCounter == 0 : partitioner.currArea().lumaPos() == cs.area.lumaPos();
1006
1007
    // CU header
1008
177k
    if( isFirst )
1009
173k
    {
1010
173k
      if ((!cs.slice->isIntra() || cs.slice->sps->IBC || cs.slice->sps->PLT) && cu.Y().valid())
1011
173k
      {
1012
173k
        m_CABACEstimator->pred_mode   ( cu );
1013
173k
      }
1014
173k
      m_CABACEstimator->bdpcm_mode  ( cu, ComponentID(partitioner.chType) );
1015
173k
    }
1016
1017
    // luma prediction mode
1018
177k
    if (isFirst)
1019
173k
    {
1020
173k
      if ( !cu.Y().valid())
1021
0
      {
1022
0
        m_CABACEstimator->pred_mode( cu );
1023
0
      }
1024
173k
      m_CABACEstimator->intra_luma_pred_mode( cu );
1025
173k
    }
1026
177k
  }
1027
268k
  else //  if (chroma)
1028
268k
  {
1029
268k
    bool isFirst = partitioner.currArea().Cb().valid() && partitioner.currArea().chromaPos() == cs.area.chromaPos();
1030
1031
268k
    if( isFirst )
1032
268k
    {
1033
268k
      m_CABACEstimator->bdpcm_mode(cu, ComponentID(CH_C));
1034
268k
      m_CABACEstimator->intra_chroma_pred_mode(  cu );
1035
268k
    }
1036
268k
  }
1037
445k
}
1038
1039
void IntraSearch::xEncSubdivCbfQT( CodingStructure &cs, Partitioner &partitioner, const bool luma )
1040
445k
{
1041
445k
  const UnitArea& currArea = partitioner.currArea();
1042
445k
  int subTuCounter = m_ispTestedModes[0].subTuCounter;
1043
445k
  TransformUnit  &currTU   = *cs.getTU(currArea.blocks[partitioner.chType], partitioner.chType, subTuCounter);
1044
445k
  CodingUnit     &currCU   = *currTU.cu;
1045
445k
  const uint32_t currDepth = partitioner.currTrDepth;
1046
445k
  const bool  subdiv = currTU.depth > currDepth;
1047
445k
  ComponentID compID = partitioner.chType == CH_L ? COMP_Y : COMP_Cb;
1048
1049
445k
  if (!luma)
1050
268k
  {
1051
268k
    const bool chromaCbfISP = currArea.blocks[COMP_Cb].valid() && currCU.ispMode && !subdiv;
1052
268k
    if (!currCU.ispMode || chromaCbfISP)
1053
268k
    {
1054
268k
      const uint32_t numberValidComponents = getNumberValidComponents(currArea.chromaFormat);
1055
268k
      const uint32_t cbfDepth = (chromaCbfISP ? currDepth - 1 : currDepth);
1056
1057
804k
      for (uint32_t ch = COMP_Cb; ch < numberValidComponents; ch++)
1058
536k
      {
1059
536k
        const ComponentID compID = ComponentID(ch);
1060
536k
        if (currDepth == 0 || TU::getCbfAtDepth(currTU, compID, currDepth - 1) || chromaCbfISP)
1061
536k
        {
1062
536k
          const bool prevCbf = (compID == COMP_Cr ? TU::getCbfAtDepth(currTU, COMP_Cb, currDepth) : false);
1063
536k
          m_CABACEstimator->cbf_comp(currCU, TU::getCbfAtDepth(currTU, compID, currDepth), currArea.blocks[compID], cbfDepth, prevCbf);
1064
536k
        }
1065
536k
      }
1066
268k
    }
1067
268k
  }
1068
1069
445k
  if (subdiv)
1070
0
  {
1071
0
    if (partitioner.canSplit(TU_MAX_TR_SPLIT, cs))
1072
0
    {
1073
0
      partitioner.splitCurrArea(TU_MAX_TR_SPLIT, cs);
1074
0
    }
1075
0
    else if (currCU.ispMode && isLuma(compID))
1076
0
    {
1077
0
      partitioner.splitCurrArea(m_ispTestedModes[0].IspType, cs);
1078
0
    }
1079
0
    else
1080
0
      THROW("Cannot perform an implicit split!");
1081
1082
0
    do
1083
0
    {
1084
0
      xEncSubdivCbfQT(cs, partitioner, luma);   //?
1085
0
      subTuCounter += subTuCounter != -1 ? 1 : 0;
1086
0
    } while (partitioner.nextPart(cs));
1087
1088
0
    partitioner.exitCurrSplit();
1089
0
  }
1090
445k
  else
1091
445k
  {
1092
    //===== Cbfs =====
1093
445k
    if (luma)
1094
177k
    {
1095
177k
      bool previousCbf = false;
1096
177k
      bool lastCbfIsInferred = false;
1097
177k
      if (m_ispTestedModes[0].IspType != TU_NO_ISP)
1098
13.8k
      {
1099
13.8k
        bool     rootCbfSoFar = false;
1100
13.8k
        uint32_t nTus = currCU.ispMode == HOR_INTRA_SUBPARTITIONS ? currCU.lheight() >> floorLog2(currTU.lheight())
1101
13.8k
          : currCU.lwidth() >> floorLog2(currTU.lwidth());
1102
13.8k
        if (subTuCounter == nTus - 1)
1103
1.31k
        {
1104
1.31k
          TransformUnit* tuPointer = currCU.firstTU;
1105
5.24k
          for (int tuIdx = 0; tuIdx < nTus - 1; tuIdx++)
1106
3.93k
          {
1107
3.93k
            rootCbfSoFar |= TU::getCbfAtDepth(*tuPointer, COMP_Y, currDepth);
1108
3.93k
            tuPointer = tuPointer->next;
1109
3.93k
          }
1110
1.31k
          if (!rootCbfSoFar)
1111
0
          {
1112
0
            lastCbfIsInferred = true;
1113
0
          }
1114
1.31k
        }
1115
13.8k
        if (!lastCbfIsInferred)
1116
13.8k
        {
1117
13.8k
          previousCbf = TU::getPrevTuCbfAtDepth(currTU, COMP_Y, partitioner.currTrDepth);
1118
13.8k
        }
1119
13.8k
      }
1120
177k
      if (!lastCbfIsInferred)
1121
177k
      {
1122
177k
        m_CABACEstimator->cbf_comp(currCU, TU::getCbfAtDepth(currTU, COMP_Y, currDepth), currTU.Y(), currTU.depth, previousCbf, currCU.ispMode);
1123
177k
      }
1124
177k
    }
1125
445k
  }
1126
445k
}
1127
void IntraSearch::xEncCoeffQT(CodingStructure& cs, Partitioner& partitioner, const ComponentID compID, CUCtx* cuCtx, const int subTuIdx, const PartSplit ispType)
1128
713k
{
1129
713k
  const UnitArea& currArea  = partitioner.currArea();
1130
1131
713k
  int subTuCounter          = m_ispTestedModes[0].subTuCounter;
1132
713k
  TransformUnit& currTU     = *cs.getTU(currArea.blocks[partitioner.chType], partitioner.chType, subTuCounter);
1133
713k
  uint32_t   currDepth      = partitioner.currTrDepth;
1134
713k
  const bool subdiv         = currTU.depth > currDepth;
1135
1136
713k
  if (subdiv)
1137
0
  {
1138
0
    if (partitioner.canSplit(TU_MAX_TR_SPLIT, cs))
1139
0
    {
1140
0
      partitioner.splitCurrArea(TU_MAX_TR_SPLIT, cs);
1141
0
    }
1142
0
    else if (currTU.cu->ispMode)
1143
0
    {
1144
0
      partitioner.splitCurrArea(m_ispTestedModes[0].IspType, cs);
1145
0
    }
1146
0
    else
1147
0
      THROW("Implicit TU split not available!");
1148
1149
0
    do
1150
0
    {
1151
0
      xEncCoeffQT(cs, partitioner, compID, cuCtx, subTuCounter, m_ispTestedModes[0].IspType);
1152
0
      subTuCounter += subTuCounter != -1 ? 1 : 0;
1153
0
    } while( partitioner.nextPart( cs ) );
1154
1155
0
    partitioner.exitCurrSplit();
1156
0
  }
1157
713k
  else
1158
1159
713k
  if( currArea.blocks[compID].valid() )
1160
713k
  {
1161
713k
    if( compID == COMP_Cr )
1162
268k
    {
1163
268k
      const int cbfMask = ( TU::getCbf( currTU, COMP_Cb ) ? 2 : 0 ) + ( TU::getCbf( currTU, COMP_Cr ) ? 1 : 0 );
1164
268k
      m_CABACEstimator->joint_cb_cr( currTU, cbfMask );
1165
268k
    }
1166
713k
    if( TU::getCbf( currTU, compID ) )
1167
217k
    {
1168
217k
      if( isLuma(compID) )
1169
24.1k
      {
1170
24.1k
        m_CABACEstimator->residual_coding( currTU, compID, cuCtx );
1171
24.1k
        m_CABACEstimator->mts_idx( *currTU.cu, cuCtx );
1172
24.1k
      }
1173
193k
      else
1174
193k
        m_CABACEstimator->residual_coding( currTU, compID );
1175
217k
    }
1176
713k
  }
1177
713k
}
1178
1179
uint64_t IntraSearch::xGetIntraFracBitsQT( CodingStructure &cs, Partitioner &partitioner, const bool luma, CUCtx *cuCtx )
1180
445k
{
1181
445k
  m_CABACEstimator->resetBits();
1182
1183
445k
  xEncIntraHeader( cs, partitioner, luma );
1184
445k
  xEncSubdivCbfQT( cs, partitioner, luma );
1185
1186
445k
  if( luma )
1187
177k
  {
1188
177k
    xEncCoeffQT( cs, partitioner, COMP_Y, cuCtx );
1189
1190
177k
    CodingUnit &cu = *cs.cus[0];
1191
177k
    if (cuCtx /*&& CU::isSepTree(cu)*/
1192
112k
      && (!cu.ispMode || (cu.lfnstIdx && m_ispTestedModes[0].subTuCounter == 0)
1193
8.76k
        || (!cu.lfnstIdx
1194
7.44k
          && m_ispTestedModes[0].subTuCounter == m_ispTestedModes[cu.lfnstIdx].numTotalParts[cu.ispMode - 1] - 1)))
1195
104k
    {
1196
104k
      m_CABACEstimator->residual_lfnst_mode( cu, *cuCtx );
1197
104k
    }
1198
177k
  }
1199
268k
  else
1200
268k
  {
1201
268k
    xEncCoeffQT( cs, partitioner, COMP_Cb );
1202
268k
    xEncCoeffQT( cs, partitioner, COMP_Cr );
1203
268k
  }
1204
1205
445k
  uint64_t fracBits = m_CABACEstimator->getEstFracBits();
1206
445k
  return fracBits;
1207
445k
}
1208
1209
uint64_t IntraSearch::xGetIntraFracBitsQTChroma(const TransformUnit& currTU, const ComponentID compID, CUCtx *cuCtx)
1210
1.67M
{
1211
1.67M
  m_CABACEstimator->resetBits();
1212
1213
1.67M
  if ( currTU.jointCbCr )
1214
249k
  {
1215
249k
    const int cbfMask = ( TU::getCbf( currTU, COMP_Cb ) ? 2 : 0 ) + ( TU::getCbf( currTU, COMP_Cr ) ? 1 : 0 );
1216
249k
    m_CABACEstimator->cbf_comp( *currTU.cu, cbfMask>>1, currTU.blocks[ COMP_Cb ], currTU.depth, false );
1217
249k
    m_CABACEstimator->cbf_comp( *currTU.cu, cbfMask &1, currTU.blocks[ COMP_Cr ], currTU.depth, cbfMask>>1 );
1218
249k
    if( cbfMask )
1219
249k
      m_CABACEstimator->joint_cb_cr( currTU, cbfMask );
1220
249k
    if (cbfMask >> 1)
1221
247k
      m_CABACEstimator->residual_coding( currTU, COMP_Cb, cuCtx );
1222
249k
    if (cbfMask & 1)
1223
249k
      m_CABACEstimator->residual_coding( currTU, COMP_Cr, cuCtx );
1224
249k
  }
1225
1.42M
  else
1226
1.42M
  {
1227
1.42M
    if ( compID == COMP_Cb )
1228
712k
      m_CABACEstimator->cbf_comp( *currTU.cu, TU::getCbf( currTU, compID ), currTU.blocks[ compID ], currTU.depth, false );
1229
712k
    else
1230
712k
    {
1231
712k
      const bool cbCbf    = TU::getCbf( currTU, COMP_Cb );
1232
712k
      const bool crCbf    = TU::getCbf( currTU, compID );
1233
712k
      const int  cbfMask  = ( cbCbf ? 2 : 0 ) + ( crCbf ? 1 : 0 );
1234
712k
      m_CABACEstimator->cbf_comp( *currTU.cu, crCbf, currTU.blocks[ compID ], currTU.depth, cbCbf );
1235
712k
      m_CABACEstimator->joint_cb_cr( currTU, cbfMask );
1236
712k
    }
1237
1.42M
  }
1238
1239
1.67M
  if( !currTU.jointCbCr && TU::getCbf( currTU, compID ) )
1240
501k
  {
1241
501k
    m_CABACEstimator->residual_coding( currTU, compID, cuCtx );
1242
501k
  }
1243
1244
1.67M
  uint64_t fracBits = m_CABACEstimator->getEstFracBits();
1245
1.67M
  return fracBits;
1246
1.67M
}
1247
1248
void IntraSearch::xIntraCodingTUBlock(TransformUnit &tu, const ComponentID compID, const bool checkCrossCPrediction, Distortion &ruiDist, uint32_t *numSig, PelUnitBuf *predBuf, const bool loadTr)
1249
1.85M
{
1250
1.85M
  if (!tu.blocks[compID].valid())
1251
0
  {
1252
0
    return;
1253
0
  }
1254
1255
1.85M
  CodingStructure &cs             = *tu.cs;
1256
1.85M
  const CompArea      &area       = tu.blocks[compID];
1257
1.85M
  const SPS           &sps        = *cs.sps;
1258
1259
1.85M
  const ChannelType    chType     = toChannelType(compID);
1260
1.85M
  const int            bitDepth   = sps.bitDepths[chType];
1261
1262
1.85M
  CPelBuf        piOrg            = cs.getOrgBuf    (area);
1263
1.85M
  PelBuf         piPred           = cs.getPredBuf   (area);
1264
1.85M
  PelBuf         piResi           = cs.getResiBuf   (area);
1265
1.85M
  PelBuf         piReco           = cs.getRecoBuf   (area);
1266
1267
1.85M
  const CodingUnit& cu            = *tu.cu;
1268
1269
  //===== init availability pattern =====
1270
1.85M
  CHECK( tu.jointCbCr && compID == COMP_Cr, "wrong combination of compID and jointCbCr" );
1271
1.85M
  bool jointCbCr = tu.jointCbCr && compID == COMP_Cb;
1272
1273
1.85M
  if ( isLuma(compID) )
1274
182k
  {
1275
182k
    bool predRegDiffFromTB = CU::isPredRegDiffFromTB(*tu.cu );
1276
182k
    bool firstTBInPredReg  = false;
1277
182k
    CompArea areaPredReg(COMP_Y, tu.chromaFormat, area);
1278
182k
    if (tu.cu->ispMode )
1279
18.7k
    {
1280
18.7k
      firstTBInPredReg = CU::isFirstTBInPredReg(*tu.cu, area);
1281
18.7k
      if (predRegDiffFromTB)
1282
0
      {
1283
0
        if (firstTBInPredReg)
1284
0
        {
1285
0
          CU::adjustPredArea(areaPredReg);
1286
0
          initIntraPatternChTypeISP(*tu.cu, areaPredReg, piReco);
1287
0
        }
1288
0
      }
1289
18.7k
      else
1290
18.7k
        initIntraPatternChTypeISP(*tu.cu, area, piReco);
1291
18.7k
    }
1292
163k
    else if( !predBuf )
1293
28.0k
    {
1294
28.0k
      initIntraPatternChType(*tu.cu, area);
1295
28.0k
    }
1296
1297
    //===== get prediction signal =====
1298
182k
    if (predRegDiffFromTB)
1299
0
    {
1300
0
      if (firstTBInPredReg)
1301
0
      {
1302
0
        PelBuf piPredReg = cs.getPredBuf(areaPredReg);
1303
0
        predIntraAng(compID, piPredReg, cu);
1304
0
      }
1305
0
    }
1306
182k
    else
1307
182k
    {
1308
182k
      if( predBuf )
1309
135k
      {
1310
135k
        piPred.copyFrom( predBuf->Y() );
1311
135k
      }
1312
46.8k
      else if( CU::isMIP( cu, CH_L ) )
1313
20.9k
      {
1314
20.9k
        initIntraMip( cu );
1315
20.9k
        predIntraMip( piPred, cu );
1316
20.9k
      }
1317
25.8k
      else
1318
25.8k
      {
1319
25.8k
        predIntraAng(compID, piPred, cu);
1320
25.8k
      }
1321
182k
    }
1322
182k
  }
1323
1.85M
  DTRACE( g_trace_ctx, D_PRED, "@(%4d,%4d) [%2dx%2d] IMode=%d\n", tu.lx(), tu.ly(), tu.lwidth(), tu.lheight(), CU::getFinalIntraMode(cu, chType) );
1324
1325
1.85M
  if (isLuma(compID))
1326
182k
  {
1327
    //===== get residual signal =====
1328
182k
    piResi.subtract( piOrg, piPred );
1329
182k
  }
1330
1331
  //===== transform and quantization =====
1332
  //--- init rate estimation arrays for RDOQ ---
1333
  //--- transform and quantization           ---
1334
1.85M
  TCoeff uiAbsSum = 0;
1335
1.85M
  const QpParam cQP(tu, compID);
1336
1337
1.85M
  m_pcTrQuant->selectLambda(compID);
1338
1339
1.85M
  if ( jointCbCr )
1340
252k
  {
1341
    // Lambda is loosened for the joint mode with respect to single modes as the same residual is used for both chroma blocks
1342
252k
    const int    absIct = abs( TU::getICTMode(tu) );
1343
252k
    const double lfact  = ( absIct == 1 || absIct == 3 ? 0.8 : 0.5 );
1344
252k
    m_pcTrQuant->scaleLambda( lfact );
1345
252k
  }
1346
1.85M
  if ( sps.jointCbCr && isChroma(compID) && (tu.cu->cs->slice->sliceQp > 18) )
1347
1.12M
  {
1348
1.12M
    m_pcTrQuant->scaleLambda( 1.3 );
1349
1.12M
  }
1350
1351
1.85M
  if( isLuma(compID) )
1352
182k
  {
1353
182k
    m_pcTrQuant->transformNxN(tu, compID, cQP, uiAbsSum, m_CABACEstimator->getCtx(), loadTr);
1354
1355
182k
    DTRACE( g_trace_ctx, D_TU_ABS_SUM, "%d: comp=%d, abssum=%d\n", DTRACE_GET_COUNTER( g_trace_ctx, D_TU_ABS_SUM ), compID, uiAbsSum );
1356
182k
    if (tu.cu->ispMode && isLuma(compID) && CU::isISPLast(*tu.cu, area, area.compID) && CU::allLumaCBFsAreZero(*tu.cu))
1357
0
    {
1358
      // ISP has to have at least one non-zero CBF
1359
0
      ruiDist = MAX_INT;
1360
0
      return;
1361
0
    }
1362
    //--- inverse transform ---
1363
182k
    if (uiAbsSum > 0)
1364
29.0k
    {
1365
29.0k
      m_pcTrQuant->invTransformNxN(tu, compID, piResi, cQP);
1366
29.0k
    }
1367
153k
    else
1368
153k
    {
1369
153k
      piResi.fill(0);
1370
153k
    }
1371
182k
  }
1372
1.67M
  else // chroma
1373
1.67M
  {
1374
1.67M
    PelBuf          crPred = cs.getPredBuf ( COMP_Cr );
1375
1.67M
    PelBuf          crResi = cs.getResiBuf ( COMP_Cr );
1376
1.67M
    PelBuf          crReco = cs.getRecoBuf ( COMP_Cr );
1377
1378
1.67M
    int         codedCbfMask  = 0;
1379
1.67M
    ComponentID codeCompId    = (tu.jointCbCr ? (tu.jointCbCr >> 1 ? COMP_Cb : COMP_Cr) : compID);
1380
1.67M
    const QpParam qpCbCr(tu, codeCompId);
1381
1382
1.67M
    if( tu.jointCbCr )
1383
252k
    {
1384
252k
      ComponentID otherCompId = ( codeCompId==COMP_Cr ? COMP_Cb : COMP_Cr );
1385
252k
      tu.getCoeffs( otherCompId ).fill(0); // do we need that?
1386
252k
      TU::setCbfAtDepth (tu, otherCompId, tu.depth, false );
1387
252k
    }
1388
1.67M
    PelBuf& codeResi = ( codeCompId == COMP_Cr ? crResi : piResi );
1389
1.67M
    uiAbsSum = 0;
1390
1.67M
    m_pcTrQuant->transformNxN(tu, codeCompId, qpCbCr, uiAbsSum, m_CABACEstimator->getCtx(), loadTr);
1391
1.67M
    DTRACE( g_trace_ctx, D_TU_ABS_SUM, "%d: comp=%d, abssum=%d\n", DTRACE_GET_COUNTER( g_trace_ctx, D_TU_ABS_SUM ), codeCompId, uiAbsSum );
1392
1.67M
    if( uiAbsSum > 0 )
1393
751k
    {
1394
751k
      m_pcTrQuant->invTransformNxN(tu, codeCompId, codeResi, qpCbCr);
1395
751k
      codedCbfMask += ( codeCompId == COMP_Cb ? 2 : 1 );
1396
751k
    }
1397
926k
    else
1398
926k
    {
1399
926k
      codeResi.fill(0);
1400
926k
    }
1401
1402
1.67M
    if( tu.jointCbCr )
1403
252k
    {
1404
252k
      if( tu.jointCbCr == 3 && codedCbfMask == 2 )
1405
247k
      {
1406
247k
        codedCbfMask = 3;
1407
247k
        TU::setCbfAtDepth (tu, COMP_Cr, tu.depth, true );
1408
247k
      }
1409
252k
      if( tu.jointCbCr != codedCbfMask )
1410
3.27k
      {
1411
3.27k
        ruiDist = MAX_DISTORTION;
1412
3.27k
        return;
1413
3.27k
      }
1414
249k
      m_pcTrQuant->invTransformICT( tu, piResi, crResi );
1415
249k
      uiAbsSum = codedCbfMask;
1416
249k
    }
1417
1418
    //===== reconstruction =====
1419
1.67M
    if( jointCbCr )
1420
249k
    {
1421
249k
      crReco.reconstruct(crPred, crResi, cs.slice->clpRngs[ COMP_Cr ]);
1422
249k
    }
1423
1.67M
  }
1424
1.85M
  piReco.reconstruct(piPred, piResi, cs.slice->clpRngs[ compID ]);
1425
  
1426
1427
1428
  //===== update distortion =====
1429
1.85M
  ruiDist += m_pcRdCost->getDistPart( piOrg, piReco, bitDepth, compID, DF_SSE );
1430
1.85M
  if( jointCbCr )
1431
249k
  {
1432
249k
    CPelBuf         crOrg  = cs.getOrgBuf  ( COMP_Cr );
1433
249k
    PelBuf          crReco = cs.getRecoBuf ( COMP_Cr );
1434
249k
    ruiDist += m_pcRdCost->getDistPart( crOrg, crReco, bitDepth, COMP_Cr, DF_SSE );
1435
249k
  }
1436
1.85M
}
1437
1438
void IntraSearch::xIntraCodingLumaQT(CodingStructure& cs, Partitioner& partitioner, PelUnitBuf* predBuf, const double bestCostSoFar, int numMode, bool disableMTS)
1439
109k
{
1440
109k
  PROFILER_SCOPE_AND_STAGE_EXT( 0, _TPROF, P_INTRA_RD_SEARCH_LUMA, &cs, partitioner.chType );
1441
109k
  const UnitArea& currArea  = partitioner.currArea();
1442
109k
  uint32_t        currDepth = partitioner.currTrDepth;
1443
109k
  Distortion singleDistLuma = 0;
1444
109k
  uint32_t   numSig         = 0;
1445
109k
  const SPS &sps            = *cs.sps;
1446
109k
  CodingUnit &cu            = *cs.cus[0];
1447
109k
  bool mtsAllowed = (numMode < 0) || disableMTS ? false : CU::isMTSAllowed(cu, COMP_Y);
1448
109k
  uint64_t singleFracBits   = 0;
1449
109k
  bool   splitCbfLumaSum    = false;
1450
109k
  double bestCostForISP     = bestCostSoFar;
1451
109k
  double dSingleCost        = MAX_DOUBLE;
1452
109k
  int endLfnstIdx           = (partitioner.isSepTree(cs) && partitioner.chType == CH_C && (currArea.lwidth() < 8 || currArea.lheight() < 8))
1453
109k
                           || (currArea.lwidth() > sps.getMaxTbSize() || currArea.lheight() > sps.getMaxTbSize()) || !sps.LFNST || (numMode < 0) ? 0 : 2;
1454
109k
  const bool useTS          = cs.picture->useTS;
1455
109k
  numMode                   = (numMode < 0) ? -numMode : numMode;
1456
1457
109k
  if (cu.mipFlag && !allowLfnstWithMip(cu.lumaSize()))
1458
1.88k
  {
1459
1.88k
    endLfnstIdx = 0;
1460
1.88k
  }
1461
109k
  int bestMTS = 0;
1462
109k
  int EndMTS  = mtsAllowed ? m_pcEncCfg->m_MTSIntraMaxCand : 0;
1463
109k
  if (cu.ispMode && (EndMTS || endLfnstIdx))
1464
5.07k
  {
1465
5.07k
    EndMTS = 0;
1466
5.07k
    if ((m_ispTestedModes[1].numTotalParts[cu.ispMode - 1] == 0)
1467
272
     && (m_ispTestedModes[2].numTotalParts[cu.ispMode - 1] == 0))
1468
272
    {
1469
272
      endLfnstIdx = 0;
1470
272
    }
1471
5.07k
  }
1472
109k
  if (cu.bdpcmM[CH_L])
1473
6.90k
  {
1474
6.90k
    endLfnstIdx = 0;
1475
6.90k
    EndMTS = 0;
1476
6.90k
  }
1477
109k
  bool checkTransformSkip = sps.transformSkip;
1478
1479
109k
  SizeType transformSkipMaxSize = 1 << sps.log2MaxTransformSkipBlockSize;
1480
109k
  bool tsAllowed = useTS  && cu.cs->sps->transformSkip && (!cu.ispMode) && (!cu.bdpcmM[CH_L]) && (!cu.sbtInfo);
1481
109k
  tsAllowed &= cu.blocks[COMP_Y].width <= transformSkipMaxSize && cu.blocks[COMP_Y].height <= transformSkipMaxSize;
1482
109k
  if (tsAllowed)
1483
13.8k
  {
1484
13.8k
    EndMTS += 1;
1485
13.8k
  }
1486
109k
  if (endLfnstIdx || EndMTS)
1487
44.2k
  {
1488
44.2k
    bool       splitCbfLuma  = false;
1489
44.2k
    const PartSplit ispType  = CU::getISPType(cu, COMP_Y);
1490
44.2k
    CUCtx cuCtx;
1491
44.2k
    cuCtx.isDQPCoded         = true;
1492
44.2k
    cuCtx.isChromaQpAdjCoded = true;
1493
44.2k
    cs.cost                  = 0.0;
1494
44.2k
    Distortion       singleDistTmpLuma = 0;
1495
44.2k
    uint64_t         singleTmpFracBits = 0;
1496
44.2k
    double           singleCostTmp     = 0;
1497
44.2k
    const TempCtx    ctxStart          (m_CtxCache, m_CABACEstimator->getCtx());
1498
44.2k
          TempCtx    ctxBest           (m_CtxCache);
1499
44.2k
    CodingStructure &saveCS            = *m_pSaveCS[cu.ispMode?0:1];
1500
44.2k
    TransformUnit *  tmpTU             = nullptr;
1501
44.2k
    int              bestLfnstIdx      = 0;
1502
44.2k
    int              startLfnstIdx     = 0;
1503
    // speedUps LFNST
1504
44.2k
    bool   rapidLFNST                  = false;
1505
44.2k
    bool   rapidDCT                    = false;
1506
44.2k
    double thresholdDCT                = 1;
1507
1508
44.2k
    if (m_pcEncCfg->m_MTS == 2)
1509
0
    {
1510
0
      thresholdDCT += 1.4 / sqrt(cu.lwidth() * cu.lheight());
1511
0
    }
1512
1513
44.2k
    if (m_pcEncCfg->m_LFNST > 1)
1514
0
    {
1515
0
      rapidLFNST = true;
1516
1517
0
      if (m_pcEncCfg->m_LFNST > 2)
1518
0
      {
1519
0
        rapidDCT    = true;
1520
0
        endLfnstIdx = endLfnstIdx ? 1 : 0;
1521
0
      }
1522
0
    }
1523
1524
44.2k
    saveCS.pcv              = cs.pcv;
1525
44.2k
    saveCS.picture          = cs.picture;
1526
44.2k
    saveCS.area.repositionTo( cs.area);
1527
1528
44.2k
    if (cu.ispMode)
1529
4.80k
    {
1530
4.80k
      partitioner.splitCurrArea(ispType, cs);
1531
4.80k
    }
1532
1533
44.2k
    TransformUnit& tu = cs.addTU(CS::getArea(cs, partitioner.currArea(), partitioner.chType, partitioner.treeType), partitioner.chType, cs.cus[0]);
1534
1535
44.2k
    if (cu.ispMode)
1536
4.80k
    {
1537
4.80k
      saveCS.clearTUs();
1538
4.80k
      do
1539
19.2k
      {
1540
19.2k
        saveCS.addTU(
1541
19.2k
          CS::getArea(cs, partitioner.currArea(), partitioner.chType, partitioner.treeType),
1542
19.2k
          partitioner.chType, cs.cus[0]);
1543
19.2k
      } while (partitioner.nextPart(cs));
1544
1545
4.80k
      partitioner.exitCurrSplit();
1546
4.80k
    }
1547
39.4k
    else
1548
39.4k
    {
1549
39.4k
      tmpTU = saveCS.tus.empty() ? &saveCS.addTU( currArea, partitioner.chType, nullptr ) : saveCS.tus.front();
1550
39.4k
      tmpTU->initData();
1551
39.4k
      tmpTU->UnitArea::operator=( currArea );
1552
39.4k
    }
1553
1554
1555
44.2k
    std::vector<TrMode> trModes{ TrMode(0, true) };
1556
44.2k
    if (tsAllowed)
1557
13.8k
    {
1558
13.8k
      trModes.push_back(TrMode(1, true));
1559
13.8k
    }
1560
44.2k
    double dct2Cost           = MAX_DOUBLE;
1561
44.2k
    double trGrpStopThreshold = 1.001;
1562
44.2k
    double trGrpBestCost      = MAX_DOUBLE;
1563
1564
44.2k
    if (mtsAllowed)
1565
0
    {
1566
0
      if (m_pcEncCfg->m_LFNST)
1567
0
      {
1568
0
        uint32_t uiIntraMode = cs.cus[0]->intraDir[partitioner.chType];
1569
0
        int MTScur           = (uiIntraMode < 34) ? MTS_DST7_DCT8 : MTS_DCT8_DST7;
1570
1571
0
        trModes.push_back(TrMode(     2, true));
1572
0
        trModes.push_back(TrMode(MTScur, true));
1573
1574
0
        MTScur = (uiIntraMode < 34) ? MTS_DCT8_DST7 : MTS_DST7_DCT8;
1575
1576
0
        trModes.push_back(TrMode(MTScur,            true));
1577
0
        trModes.push_back(TrMode(MTS_DST7_DST7 + 3, true));
1578
0
      }
1579
0
      else
1580
0
      {
1581
0
        for (int i = 2; i < 6; i++)
1582
0
        {
1583
0
          trModes.push_back(TrMode(i, true));
1584
0
        }
1585
0
      }
1586
0
    }
1587
1588
44.2k
    if ((EndMTS && !m_pcEncCfg->m_LFNST) || (tsAllowed && !mtsAllowed))
1589
13.8k
    {
1590
13.8k
      xPreCheckMTS(tu, &trModes, m_pcEncCfg->m_MTSIntraMaxCand, predBuf);
1591
13.8k
      if (!mtsAllowed && !trModes[1].second)
1592
2.76k
      {
1593
2.76k
        EndMTS = 0;
1594
2.76k
      }
1595
13.8k
    }
1596
1597
44.2k
    bool NStopMTS = true;
1598
1599
88.4k
    for (int modeId = 0; modeId <= EndMTS && NStopMTS; modeId++)
1600
44.2k
    {
1601
44.2k
      if (modeId > 1)
1602
0
      {
1603
0
        trGrpBestCost = MAX_DOUBLE;
1604
0
      }
1605
157k
      for (int lfnstIdx = startLfnstIdx; lfnstIdx <= endLfnstIdx; lfnstIdx++)
1606
112k
      {
1607
112k
        if (lfnstIdx && modeId)
1608
0
        {
1609
0
          continue;
1610
0
        }
1611
112k
        if (mtsAllowed || tsAllowed)
1612
21.7k
        {
1613
21.7k
          if (m_pcEncCfg->m_TS && bestMTS == MTS_SKIP)
1614
0
          {
1615
0
            break;
1616
0
          }
1617
21.7k
          if (!m_pcEncCfg->m_LFNST && !trModes[modeId].second && mtsAllowed)
1618
0
          {
1619
0
            continue;
1620
0
          }
1621
1622
21.7k
          tu.mtsIdx[COMP_Y] = trModes[modeId].first;
1623
21.7k
        }
1624
1625
112k
        if (cu.ispMode && lfnstIdx)
1626
9.61k
        {
1627
9.61k
          if (m_ispTestedModes[lfnstIdx].numTotalParts[cu.ispMode - 1] == 0)
1628
0
          {
1629
0
            if (lfnstIdx == 2)
1630
0
            {
1631
0
              endLfnstIdx = 1;
1632
0
            }
1633
0
            continue;
1634
0
          }
1635
9.61k
        }
1636
1637
112k
        cu.lfnstIdx                          = lfnstIdx;
1638
112k
        cuCtx.lfnstLastScanPos               = false;
1639
112k
        cuCtx.violatesLfnstConstrained[CH_L] = false;
1640
112k
        cuCtx.violatesLfnstConstrained[CH_C] = false;
1641
1642
112k
        if ((lfnstIdx != startLfnstIdx) || (modeId))
1643
68.7k
        {
1644
68.7k
          m_CABACEstimator->getCtx() = ctxStart;
1645
68.7k
        }
1646
1647
112k
        singleDistTmpLuma = 0;
1648
1649
112k
        if (cu.ispMode)
1650
14.4k
        {
1651
14.4k
          splitCbfLuma = false;
1652
1653
14.4k
          partitioner.splitCurrArea(ispType, cs);
1654
1655
14.4k
          singleCostTmp = xTestISP(cs, partitioner, bestCostForISP, ispType, splitCbfLuma, singleTmpFracBits, singleDistTmpLuma, cuCtx);
1656
1657
14.4k
          partitioner.exitCurrSplit();
1658
1659
14.4k
          if (modeId && (singleCostTmp == MAX_DOUBLE))
1660
0
          {
1661
0
            m_ispTestedModes[lfnstIdx].numTotalParts[cu.ispMode - 1] = 0;
1662
0
          }
1663
1664
14.4k
          bool storeCost = (numMode == 1) ? true : false;
1665
1666
14.4k
          if ((m_pcEncCfg->m_ISP >= 2) && (numMode <= 1))
1667
14.4k
          {
1668
14.4k
            storeCost = true;
1669
14.4k
          }
1670
1671
14.4k
          if (storeCost)
1672
14.4k
          {
1673
14.4k
            m_ispTestedModes[0].bestCost[cu.ispMode - 1] = singleCostTmp;
1674
14.4k
          }
1675
14.4k
        }
1676
98.5k
        else
1677
98.5k
        {
1678
98.5k
          bool TrLoad = (EndMTS && !m_pcEncCfg->m_LFNST) || (tsAllowed && !mtsAllowed && (lfnstIdx == 0)) ? true : false;
1679
1680
98.5k
          xIntraCodingTUBlock(tu, COMP_Y, false, singleDistTmpLuma, &numSig, predBuf, TrLoad);
1681
1682
98.5k
          cuCtx.mtsLastScanPos = false;
1683
          //----- determine rate and r-d cost -----
1684
18.4E
        if ((sps.LFNST ? (modeId == EndMTS && modeId != 0 && checkTransformSkip) : (trModes[modeId].first != 0)) && !TU::getCbfAtDepth(tu, COMP_Y, currDepth))
1685
0
        {
1686
0
          singleCostTmp = MAX_DOUBLE;
1687
0
        }
1688
98.5k
        else
1689
98.5k
        {
1690
98.5k
          m_ispTestedModes[0].IspType      = TU_NO_ISP;
1691
98.5k
          m_ispTestedModes[0].subTuCounter = -1;
1692
98.5k
          singleTmpFracBits = xGetIntraFracBitsQT(cs, partitioner, true, &cuCtx);
1693
1694
98.5k
          if (tu.mtsIdx[COMP_Y] > MTS_SKIP)
1695
0
          {
1696
0
            if (!cuCtx.mtsLastScanPos)
1697
0
            {
1698
0
              singleCostTmp = MAX_DOUBLE;
1699
0
            }
1700
0
            else
1701
0
            {
1702
0
              singleCostTmp = m_pcRdCost->calcRdCost(singleTmpFracBits, singleDistTmpLuma);
1703
0
            }
1704
0
          }
1705
98.5k
          else
1706
98.5k
          {
1707
98.5k
            singleCostTmp = m_pcRdCost->calcRdCost(singleTmpFracBits, singleDistTmpLuma);
1708
98.5k
          }
1709
98.5k
        }
1710
1711
98.5k
          if (((EndMTS && (m_pcEncCfg->m_MTS == 2)) || rapidLFNST) && modeId == 0 && lfnstIdx == 0)
1712
0
          {
1713
0
            if (singleCostTmp > bestCostSoFar * thresholdDCT)
1714
0
            {
1715
0
              EndMTS = 0;
1716
1717
0
              if (rapidDCT)
1718
0
              {
1719
0
                endLfnstIdx = 0;   // break the loop but do not cpy best
1720
0
              }
1721
0
            }
1722
0
          }
1723
1724
98.5k
          if (lfnstIdx && !cuCtx.lfnstLastScanPos && !cu.ispMode)
1725
49.5k
          {
1726
49.5k
            bool rootCbfL = false;
1727
1728
198k
            for (uint32_t t = 0; t < getNumberValidTBlocks(*cu.cs->pcv); t++)
1729
148k
            {
1730
148k
              rootCbfL |= tu.cbf[t] != 0;
1731
148k
            }
1732
1733
49.5k
            if (rapidLFNST && !rootCbfL)
1734
0
            {
1735
0
              endLfnstIdx = lfnstIdx; // break the loop
1736
0
            }
1737
49.5k
            bool cbfAtZeroDepth = CU::isSepTree(cu)
1738
49.5k
              ? rootCbfL
1739
49.5k
              : (cs.area.chromaFormat != CHROMA_400 && std::min(cu.firstTU->blocks[1].width, cu.firstTU->blocks[1].height) < 4)
1740
0
                ? TU::getCbfAtDepth(tu, COMP_Y, currDepth)
1741
0
                : rootCbfL;
1742
1743
49.5k
            if (cbfAtZeroDepth)
1744
372
            {
1745
372
              singleCostTmp = MAX_DOUBLE;
1746
372
            }
1747
49.5k
          }
1748
98.5k
        }
1749
1750
112k
        if (singleCostTmp < dSingleCost)
1751
40.7k
        {
1752
40.7k
          trGrpBestCost  = singleCostTmp;
1753
40.7k
          dSingleCost    = singleCostTmp;
1754
40.7k
          singleDistLuma = singleDistTmpLuma;
1755
40.7k
          singleFracBits = singleTmpFracBits;
1756
40.7k
          bestLfnstIdx   = lfnstIdx;
1757
40.7k
          bestMTS        = modeId;
1758
1759
40.7k
          if (dSingleCost < bestCostForISP)
1760
25.7k
          {
1761
25.7k
            bestCostForISP = dSingleCost;
1762
25.7k
          }
1763
1764
40.7k
          splitCbfLumaSum = splitCbfLuma;
1765
1766
40.7k
          if (lfnstIdx == 0 && modeId == 0 && cu.ispMode == 0)
1767
39.4k
          {
1768
39.4k
            dct2Cost = singleCostTmp;
1769
1770
39.4k
            if (!TU::getCbfAtDepth(tu, COMP_Y, currDepth))
1771
33.3k
            {
1772
33.3k
              if (rapidLFNST)
1773
0
              {
1774
0
                 endLfnstIdx = 0;   // break the loop but do not cpy best
1775
0
              }
1776
1777
33.3k
              EndMTS = 0;
1778
33.3k
            }
1779
39.4k
          }
1780
1781
40.7k
          if (bestLfnstIdx != endLfnstIdx || bestMTS != EndMTS)
1782
30.7k
          {
1783
30.7k
            if (cu.ispMode)
1784
1.01k
            {
1785
1.01k
              saveCS.getRecoBuf(currArea.Y()).copyFrom(cs.getRecoBuf(currArea.Y()));
1786
1787
5.06k
              for (uint32_t j = 0; j < cs.tus.size(); j++)
1788
4.05k
              {
1789
4.05k
                saveCS.tus[j]->copyComponentFrom(*cs.tus[j], COMP_Y);
1790
4.05k
              }
1791
1.01k
            }
1792
29.7k
            else
1793
29.7k
            {
1794
29.7k
              saveCS.getPredBuf(tu.Y()).copyFrom(cs.getPredBuf(tu.Y()));
1795
29.7k
              saveCS.getRecoBuf(tu.Y()).copyFrom(cs.getRecoBuf(tu.Y()));
1796
1797
29.7k
              tmpTU->copyComponentFrom(tu, COMP_Y);
1798
29.7k
            }
1799
1800
30.7k
            ctxBest = m_CABACEstimator->getCtx();
1801
30.7k
          }
1802
      
1803
40.7k
        }
1804
72.1k
        else
1805
72.1k
        {
1806
72.1k
          if( rapidLFNST )
1807
0
          {
1808
0
            endLfnstIdx = lfnstIdx; // break the loop
1809
0
          }
1810
72.1k
        }
1811
112k
      }
1812
44.2k
      if (m_pcEncCfg->m_LFNST && m_pcEncCfg->m_MTS == 2 && modeId && modeId != EndMTS)
1813
0
      {
1814
0
        NStopMTS = false;
1815
1816
0
        if (bestMTS || bestLfnstIdx)
1817
0
        {
1818
0
          if ((modeId > 1 && bestMTS == modeId) || modeId == 1)
1819
0
          {
1820
0
            NStopMTS = (dct2Cost / trGrpBestCost) < trGrpStopThreshold;
1821
0
          }
1822
0
        }
1823
0
      }
1824
44.2k
    }
1825
1826
44.2k
    cu.lfnstIdx = bestLfnstIdx;
1827
44.2k
    if (dSingleCost != MAX_DOUBLE)
1828
40.2k
    {
1829
40.2k
      if (bestLfnstIdx != endLfnstIdx || bestMTS != EndMTS)
1830
30.2k
      {
1831
30.2k
        if (cu.ispMode)
1832
713
        {
1833
713
          const UnitArea& currArea = partitioner.currArea();
1834
713
          cs.getRecoBuf(currArea.Y()).copyFrom(saveCS.getRecoBuf(currArea.Y()));
1835
1836
713
          if (saveCS.tus.size() != cs.tus.size())
1837
0
          {
1838
0
            partitioner.splitCurrArea(ispType, cs);
1839
1840
0
            do
1841
0
            {
1842
0
              partitioner.nextPart(cs);
1843
0
              cs.addTU(CS::getArea(cs, partitioner.currArea(), partitioner.chType, partitioner.treeType),
1844
0
                partitioner.chType, cs.cus[0]);
1845
0
            } while (saveCS.tus.size() != cs.tus.size());
1846
1847
0
            partitioner.exitCurrSplit();
1848
0
          }
1849
1850
3.56k
          for (uint32_t j = 0; j < saveCS.tus.size(); j++)
1851
2.85k
          {
1852
2.85k
            cs.tus[j]->copyComponentFrom(*saveCS.tus[j], COMP_Y);
1853
2.85k
          }
1854
713
        }
1855
29.5k
        else
1856
29.5k
        {
1857
29.5k
          cs.getRecoBuf(tu.Y()).copyFrom(saveCS.getRecoBuf(tu.Y()));
1858
1859
29.5k
          tu.copyComponentFrom(*tmpTU, COMP_Y);
1860
29.5k
        }
1861
1862
30.2k
        m_CABACEstimator->getCtx() = ctxBest;
1863
30.2k
      }
1864
1865
      // otherwise this would've happened in useSubStructure
1866
40.2k
      cs.picture->getRecoBuf(currArea.Y()).copyFrom(cs.getRecoBuf(currArea.Y()));
1867
40.2k
    }
1868
44.2k
  }
1869
65.5k
  else
1870
65.5k
  {
1871
65.5k
    if (cu.ispMode)
1872
272
    {
1873
272
      const PartSplit ispType = CU::getISPType(cu, COMP_Y);
1874
272
      partitioner.splitCurrArea(ispType, cs);
1875
1876
272
      CUCtx      cuCtx;
1877
272
      dSingleCost = xTestISP(cs, partitioner, bestCostForISP, ispType, splitCbfLumaSum, singleFracBits, singleDistLuma, cuCtx);
1878
272
      partitioner.exitCurrSplit();
1879
272
      bool storeCost = (numMode == 1) ? true : false;
1880
272
      if ((m_pcEncCfg->m_ISP >= 2) && (numMode <= 1))
1881
272
      {
1882
272
        storeCost = true;
1883
272
      }
1884
272
      if (storeCost)
1885
272
      {
1886
272
        m_ispTestedModes[0].bestCost[cu.ispMode - 1] = dSingleCost;
1887
272
      }
1888
272
    }
1889
65.2k
    else
1890
65.2k
    {
1891
65.2k
      TransformUnit& tu =
1892
65.2k
        cs.addTU(CS::getArea(cs, currArea, partitioner.chType, partitioner.treeType), partitioner.chType, cs.cus[0]);
1893
65.2k
      tu.depth = currDepth;
1894
1895
65.2k
      CHECK(!tu.Y().valid(), "Invalid TU");
1896
65.2k
      xIntraCodingTUBlock(tu, COMP_Y, false, singleDistLuma, &numSig, predBuf);
1897
      //----- determine rate and r-d cost -----
1898
65.2k
      m_ispTestedModes[0].IspType = TU_NO_ISP;
1899
65.2k
      m_ispTestedModes[0].subTuCounter = -1;
1900
65.2k
      singleFracBits = xGetIntraFracBitsQT(cs, partitioner, true);
1901
65.2k
      dSingleCost = m_pcRdCost->calcRdCost(singleFracBits, singleDistLuma);
1902
65.2k
    }
1903
65.5k
  }
1904
1905
109k
  if (cu.ispMode)
1906
5.07k
  { 
1907
5.07k
    for (auto& ptu : cs.tus)
1908
8.21k
    {
1909
8.21k
      if (currArea.Y().contains(ptu->Y()))
1910
8.21k
      {
1911
8.21k
        TU::setCbfAtDepth(*ptu, COMP_Y, currDepth, splitCbfLumaSum ? 1 : 0);
1912
8.21k
      }
1913
8.21k
    }
1914
5.07k
  }
1915
109k
  cs.dist     += singleDistLuma;
1916
109k
  cs.fracBits += singleFracBits;
1917
109k
  cs.cost      = dSingleCost;
1918
1919
109k
  STAT_COUNT_CU_MODES( partitioner.chType == CH_L, g_cuCounters1D[CU_RD_TESTS][0][!cs.slice->isIntra() + cs.slice->depth] );
1920
109k
  STAT_COUNT_CU_MODES( partitioner.chType == CH_L && !cs.slice->isIntra(), g_cuCounters2D[CU_RD_TESTS][Log2( cs.area.lheight() )][Log2( cs.area.lwidth() )] );
1921
109k
}
1922
1923
ChromaCbfs IntraSearch::xIntraChromaCodingQT(CodingStructure& cs, Partitioner& partitioner)
1924
268k
{
1925
268k
  UnitArea    currArea      = partitioner.currArea();
1926
1927
268k
  if( !currArea.Cb().valid() ) 
1928
0
    return ChromaCbfs(false);
1929
1930
268k
  TransformUnit& currTU     = *cs.getTU( currArea.chromaPos(), CH_C );
1931
268k
  const CodingUnit& cu  = *cs.getCU( currArea.chromaPos(), CH_C, TREE_D );
1932
268k
  ChromaCbfs cbfs(false);
1933
268k
  uint32_t   currDepth = partitioner.currTrDepth;
1934
268k
  const bool useTS = cs.picture->useTS;
1935
268k
  if (currDepth == currTU.depth)
1936
268k
  {
1937
268k
    if (!currArea.Cb().valid() || !currArea.Cr().valid())
1938
0
    {
1939
0
      return cbfs;
1940
0
    }
1941
1942
268k
    CodingStructure& saveCS = *m_pSaveCS[1];
1943
268k
    saveCS.pcv = cs.pcv;
1944
268k
    saveCS.picture = cs.picture;
1945
268k
    saveCS.area.repositionTo(cs.area);
1946
1947
268k
    TransformUnit& tmpTU = saveCS.tus.empty() ? saveCS.addTU(currArea, partitioner.chType, nullptr) : *saveCS.tus.front();
1948
268k
    tmpTU.initData();
1949
268k
    tmpTU.UnitArea::operator=(currArea);
1950
268k
    const unsigned      numTBlocks = getNumberValidTBlocks(*cs.pcv);
1951
1952
268k
    CompArea& cbArea = currTU.blocks[COMP_Cb];
1953
268k
    CompArea& crArea = currTU.blocks[COMP_Cr];
1954
268k
    double     bestCostCb = MAX_DOUBLE;
1955
268k
    double     bestCostCr = MAX_DOUBLE;
1956
268k
    Distortion bestDistCb = 0;
1957
268k
    Distortion bestDistCr = 0;
1958
1959
268k
    TempCtx ctxStartTU(m_CtxCache);
1960
268k
    TempCtx ctxStart(m_CtxCache);
1961
268k
    TempCtx ctxBest(m_CtxCache);
1962
1963
268k
    ctxStartTU = m_CABACEstimator->getCtx();
1964
268k
    ctxStart = m_CABACEstimator->getCtx();
1965
268k
    currTU.jointCbCr = 0;
1966
1967
    // Do predictions here to avoid repeating the "default0Save1Load2" stuff
1968
268k
    int  predMode = cu.bdpcmM[CH_C] ? BDPCM_IDX : CU::getFinalIntraMode(cu, CH_C);
1969
1970
268k
    PelBuf piPredCb = cs.getPredBuf(COMP_Cb);
1971
268k
    PelBuf piPredCr = cs.getPredBuf(COMP_Cr);
1972
1973
268k
    initIntraPatternChType(*currTU.cu, cbArea);
1974
268k
    initIntraPatternChType(*currTU.cu, crArea);
1975
1976
268k
    if (CU::isLMCMode(predMode))
1977
19.7k
    {
1978
19.7k
      loadLMLumaRecPels(cu, cbArea);
1979
19.7k
      predIntraChromaLM(COMP_Cb, piPredCb, cu, cbArea, predMode);
1980
19.7k
      predIntraChromaLM(COMP_Cr, piPredCr, cu, crArea, predMode);
1981
19.7k
    }
1982
248k
    else
1983
248k
    {
1984
248k
      predIntraAng(COMP_Cb, piPredCb, cu);
1985
248k
      predIntraAng(COMP_Cr, piPredCr, cu);
1986
248k
    }
1987
1988
    // determination of chroma residuals including reshaping and cross-component prediction
1989
    //----- get chroma residuals -----
1990
268k
    PelBuf resiCb = cs.getResiBuf(COMP_Cb);
1991
268k
    PelBuf resiCr = cs.getResiBuf(COMP_Cr);
1992
268k
    resiCb.subtract(cs.getOrgBuf(COMP_Cb), piPredCb);
1993
268k
    resiCr.subtract(cs.getOrgBuf(COMP_Cr), piPredCr);
1994
1995
    //===== store original residual signals (std and crossCompPred) =====
1996
1.60M
    for( int k = 0; k < 5; k++ )
1997
1.34M
    {
1998
1.34M
      m_orgResiCb[k].compactResize( cbArea );
1999
1.34M
      m_orgResiCr[k].compactResize( crArea );
2000
1.34M
    }
2001
536k
    for (int k = 0; k < 1; k += 4)
2002
268k
    {
2003
268k
      m_orgResiCb[k].copyFrom(resiCb);
2004
268k
      m_orgResiCr[k].copyFrom(resiCr);
2005
268k
    }
2006
2007
268k
    CUCtx cuCtx;
2008
268k
    cuCtx.isDQPCoded = true;
2009
268k
    cuCtx.isChromaQpAdjCoded = true;
2010
268k
    cuCtx.lfnstLastScanPos = false;
2011
2012
268k
    CodingStructure& saveCScur = *m_pSaveCS[2];
2013
2014
268k
    saveCScur.pcv = cs.pcv;
2015
268k
    saveCScur.picture = cs.picture;
2016
268k
    saveCScur.area.repositionTo(cs.area);
2017
2018
268k
    TransformUnit& tmpTUcur = saveCScur.tus.empty() ? saveCScur.addTU(currArea, partitioner.chType, nullptr) : *saveCScur.tus.front();
2019
268k
    tmpTUcur.initData();
2020
268k
    tmpTUcur.UnitArea::operator=(currArea);
2021
2022
268k
    TempCtx ctxBestTUL(m_CtxCache);
2023
2024
268k
    const SPS& sps = *cs.sps;
2025
268k
    double     bestCostCbcur = MAX_DOUBLE;
2026
268k
    double     bestCostCrcur = MAX_DOUBLE;
2027
268k
    Distortion bestDistCbcur = 0;
2028
268k
    Distortion bestDistCrcur = 0;
2029
2030
268k
    int  endLfnstIdx = (partitioner.isSepTree(cs) && partitioner.chType == CH_C && (partitioner.currArea().lwidth() < 8 || partitioner.currArea().lheight() < 8))
2031
256k
      || (partitioner.currArea().lwidth() > sps.getMaxTbSize() || partitioner.currArea().lheight() > sps.getMaxTbSize()) || !sps.LFNST ? 0 : 2;
2032
268k
    int  startLfnstIdx = 0;
2033
268k
    int  bestLfnstIdx = 0;
2034
268k
    bool testLFNST = sps.LFNST;
2035
2036
    // speedUps LFNST
2037
268k
    bool rapidLFNST = false;
2038
268k
    if (m_pcEncCfg->m_LFNST > 1)
2039
0
    {
2040
0
      rapidLFNST = true;
2041
0
      if (m_pcEncCfg->m_LFNST > 2)
2042
0
      {
2043
0
        endLfnstIdx = endLfnstIdx ? 1 : 0;
2044
0
      }
2045
0
    }
2046
268k
    int ts_used = 0;
2047
268k
    bool testTS = false;
2048
268k
    if (partitioner.chType != CH_C)
2049
0
    {
2050
0
      startLfnstIdx = currTU.cu->lfnstIdx;
2051
0
      endLfnstIdx = currTU.cu->lfnstIdx;
2052
0
      bestLfnstIdx = currTU.cu->lfnstIdx;
2053
0
      testLFNST  = false;
2054
0
      rapidLFNST = false;
2055
0
      ts_used = currTU.mtsIdx[COMP_Y];
2056
0
    }
2057
268k
    if (cu.bdpcmM[CH_C])
2058
36.2k
    {
2059
36.2k
      endLfnstIdx = 0;
2060
36.2k
      testLFNST = false;
2061
36.2k
    }
2062
2063
268k
    double dSingleCostAll = MAX_DOUBLE;
2064
268k
    double singleCostTmpAll = 0;
2065
2066
980k
    for (int lfnstIdx = startLfnstIdx; lfnstIdx <= endLfnstIdx; lfnstIdx++)
2067
712k
    {
2068
712k
      if (rapidLFNST && lfnstIdx)
2069
0
      {
2070
0
        if ((lfnstIdx == 2) && (bestLfnstIdx == 0))
2071
0
        {
2072
0
          continue;
2073
0
        }
2074
0
      }
2075
2076
712k
      currTU.cu->lfnstIdx = lfnstIdx;
2077
712k
      if (lfnstIdx)
2078
444k
      {
2079
444k
        m_CABACEstimator->getCtx() = ctxStartTU;
2080
444k
      }
2081
2082
712k
      cuCtx.lfnstLastScanPos = false;
2083
712k
      cuCtx.violatesLfnstConstrained[CH_L] = false;
2084
712k
      cuCtx.violatesLfnstConstrained[CH_C] = false;
2085
2086
2.13M
      for (uint32_t c = COMP_Cb; c < numTBlocks; c++)
2087
1.42M
      {
2088
1.42M
        const ComponentID compID = ComponentID(c);
2089
1.42M
        const CompArea& area = currTU.blocks[compID];
2090
1.42M
        double     dSingleCost = MAX_DOUBLE;
2091
1.42M
        Distortion singleDistCTmp = 0;
2092
1.42M
        double     singleCostTmp = 0;
2093
1.42M
        bool tsAllowed = useTS && TU::isTSAllowed(currTU, compID) && m_pcEncCfg->m_useChromaTS && !currTU.cu->lfnstIdx && !cu.bdpcmM[CH_C];
2094
1.42M
        if ((partitioner.chType == CH_L) && (!ts_used))
2095
0
        {
2096
0
          tsAllowed = false;
2097
0
        }
2098
1.42M
        uint8_t nNumTransformCands = 1 + (tsAllowed ? 1 : 0); // DCT + TS = 2 tests       
2099
1.42M
        std::vector<TrMode> trModes;
2100
1.42M
        if (nNumTransformCands > 1)
2101
0
        {
2102
0
          trModes.push_back(TrMode(0, true));   // DCT2
2103
0
          trModes.push_back(TrMode(1, true));   // TS
2104
0
          testTS = true;
2105
0
        }
2106
1.42M
        bool cbfDCT2 = true;
2107
1.42M
        const bool isLastMode = testLFNST || cs.sps->jointCbCr ||  tsAllowed ? false : true;
2108
1.42M
        int bestModeId = 0;
2109
1.42M
        ctxStart = m_CABACEstimator->getCtx();
2110
2.84M
        for (int modeId = 0; modeId < nNumTransformCands; modeId++)
2111
1.42M
        {
2112
1.42M
          if (lfnstIdx || modeId)
2113
888k
          {
2114
888k
            resiCb.copyFrom(m_orgResiCb[0]);
2115
888k
            resiCr.copyFrom(m_orgResiCr[0]);
2116
888k
          }
2117
1.42M
          if (modeId == 0)
2118
1.42M
          {
2119
1.42M
            if ( tsAllowed)
2120
0
            {
2121
0
              xPreCheckMTS(currTU, &trModes, m_pcEncCfg->m_MTSIntraMaxCand, 0, compID);
2122
0
            }
2123
1.42M
          }
2124
2125
1.42M
          currTU.mtsIdx[compID] = currTU.cu->bdpcmM[CH_C] ? MTS_SKIP : modeId;
2126
2127
1.42M
          if (modeId)
2128
0
          {
2129
0
            if (!cbfDCT2 && trModes[modeId].first == MTS_SKIP)
2130
0
            {
2131
0
              break;
2132
0
            }
2133
0
            m_CABACEstimator->getCtx() = ctxStart;
2134
0
          }
2135
1.42M
          singleDistCTmp = 0;
2136
1.42M
          if (tsAllowed)
2137
0
          {
2138
0
            xIntraCodingTUBlock(currTU, compID, false, singleDistCTmp, 0, 0, true);
2139
0
            if ((modeId == 0) && (!trModes[modeId + 1].second))
2140
0
            {
2141
0
              nNumTransformCands = 1;
2142
0
            }
2143
0
          }
2144
1.42M
          else
2145
1.42M
        {
2146
1.42M
          xIntraCodingTUBlock(currTU, compID, false, singleDistCTmp);
2147
1.42M
        }
2148
1.42M
        if (((currTU.mtsIdx[compID] == MTS_SKIP && !currTU.cu->bdpcmM[CH_C])
2149
0
          && !TU::getCbf(currTU, compID)))   // In order not to code TS flag when cbf is zero, the case for TS with
2150
                                             // cbf being zero is forbidden.
2151
0
        {
2152
0
          singleCostTmp = MAX_DOUBLE;
2153
0
        }
2154
1.42M
        else
2155
1.42M
        {
2156
1.42M
          uint64_t fracBitsTmp = xGetIntraFracBitsQTChroma(currTU, compID, &cuCtx);
2157
1.42M
          singleCostTmp = m_pcRdCost->calcRdCost(fracBitsTmp, singleDistCTmp);
2158
1.42M
        }
2159
2160
1.42M
        if (singleCostTmp < dSingleCost)
2161
1.42M
        {
2162
1.42M
          dSingleCost = singleCostTmp;
2163
2164
1.42M
          if (compID == COMP_Cb)
2165
712k
          {
2166
712k
            bestCostCb = singleCostTmp;
2167
712k
            bestDistCb = singleDistCTmp;
2168
712k
          }
2169
712k
          else
2170
712k
          {
2171
712k
            bestCostCr = singleCostTmp;
2172
712k
            bestDistCr = singleDistCTmp;
2173
712k
          }
2174
1.42M
          bestModeId = modeId;
2175
1.42M
          if (currTU.mtsIdx[compID] == MTS_DCT2_DCT2)
2176
1.35M
          {
2177
1.35M
            cbfDCT2 = TU::getCbfAtDepth(currTU, compID, currDepth);
2178
1.35M
          }
2179
1.42M
          if (!isLastMode)
2180
1.42M
          {
2181
1.42M
            saveCS.getRecoBuf(area).copyFrom(cs.getRecoBuf(area));
2182
1.42M
            tmpTU.copyComponentFrom(currTU, compID);
2183
1.42M
            ctxBest = m_CABACEstimator->getCtx();
2184
1.42M
          }
2185
1.42M
        }
2186
1.42M
        }
2187
1.42M
        if (testTS && ((c == COMP_Cb && bestModeId < (nNumTransformCands - 1)) ))
2188
0
        {
2189
0
          m_CABACEstimator->getCtx() = ctxBest;
2190
2191
0
          currTU.copyComponentFrom(tmpTU, COMP_Cb); // Cbf of Cb is needed to estimate cost for Cr Cbf
2192
0
        }
2193
1.42M
      }
2194
2195
712k
      singleCostTmpAll = bestCostCb + bestCostCr;
2196
2197
712k
      bool rootCbfL = false;
2198
712k
      if (testLFNST)
2199
676k
      {
2200
2.70M
        for (uint32_t t = 0; t < getNumberValidTBlocks(*cs.pcv); t++)
2201
2.02M
        {
2202
2.02M
          rootCbfL |= bool(tmpTU.cbf[t]);
2203
2.02M
        }
2204
676k
        if (rapidLFNST && !rootCbfL)
2205
0
        {
2206
0
          endLfnstIdx = lfnstIdx; // end this
2207
0
        }
2208
676k
      }
2209
2210
712k
      if (testLFNST && lfnstIdx && !cuCtx.lfnstLastScanPos)
2211
292k
      {
2212
292k
        bool cbfAtZeroDepth = CU::isSepTree(*currTU.cu)
2213
292k
          ? rootCbfL : (cs.area.chromaFormat != CHROMA_400
2214
0
            && std::min(tmpTU.blocks[1].width, tmpTU.blocks[1].height) < 4)
2215
0
          ? TU::getCbfAtDepth(currTU, COMP_Y, currTU.depth) : rootCbfL;
2216
292k
        if (cbfAtZeroDepth)
2217
1.48k
        {
2218
1.48k
          singleCostTmpAll = MAX_DOUBLE;
2219
1.48k
        }
2220
292k
      }
2221
712k
      if ((testLFNST || testTS) && (singleCostTmpAll < dSingleCostAll))
2222
231k
      {
2223
231k
        bestLfnstIdx = lfnstIdx;
2224
231k
        if ((lfnstIdx != endLfnstIdx) || testTS)
2225
222k
        {
2226
222k
          dSingleCostAll = singleCostTmpAll;
2227
2228
222k
          bestCostCbcur = bestCostCb;
2229
222k
          bestCostCrcur = bestCostCr;
2230
222k
          bestDistCbcur = bestDistCb;
2231
222k
          bestDistCrcur = bestDistCr;
2232
2233
222k
          saveCScur.getRecoBuf(cbArea).copyFrom(saveCS.getRecoBuf(cbArea));
2234
222k
          saveCScur.getRecoBuf(crArea).copyFrom(saveCS.getRecoBuf(crArea));
2235
2236
222k
          tmpTUcur.copyComponentFrom(tmpTU, COMP_Cb);
2237
222k
          tmpTUcur.copyComponentFrom(tmpTU, COMP_Cr);
2238
222k
        }
2239
231k
        ctxBestTUL = m_CABACEstimator->getCtx();
2240
231k
      }
2241
712k
    }
2242
268k
    if ((testLFNST && (bestLfnstIdx != endLfnstIdx)) || testTS)
2243
222k
    {
2244
222k
      bestCostCb = bestCostCbcur;
2245
222k
      bestCostCr = bestCostCrcur;
2246
222k
      bestDistCb = bestDistCbcur;
2247
222k
      bestDistCr = bestDistCrcur;
2248
222k
      currTU.cu->lfnstIdx = bestLfnstIdx;
2249
222k
      if (!cs.sps->jointCbCr)
2250
0
      {
2251
0
        cs.getRecoBuf(cbArea).copyFrom(saveCScur.getRecoBuf(cbArea));
2252
0
        cs.getRecoBuf(crArea).copyFrom(saveCScur.getRecoBuf(crArea));
2253
2254
0
        currTU.copyComponentFrom(tmpTUcur, COMP_Cb);
2255
0
        currTU.copyComponentFrom(tmpTUcur, COMP_Cr);
2256
2257
0
        m_CABACEstimator->getCtx() = ctxBestTUL;
2258
0
      }
2259
222k
    }
2260
2261
268k
    Distortion bestDistCbCr = bestDistCb + bestDistCr;
2262
2263
268k
    if (cs.sps->jointCbCr)
2264
268k
    {
2265
268k
      if ((testLFNST && (bestLfnstIdx != endLfnstIdx)) || testTS)
2266
222k
      {
2267
222k
        saveCS.getRecoBuf(cbArea).copyFrom(saveCScur.getRecoBuf(cbArea));
2268
222k
        saveCS.getRecoBuf(crArea).copyFrom(saveCScur.getRecoBuf(crArea));
2269
2270
222k
        tmpTU.copyComponentFrom(tmpTUcur, COMP_Cb);
2271
222k
        tmpTU.copyComponentFrom(tmpTUcur, COMP_Cr);
2272
222k
        m_CABACEstimator->getCtx() = ctxBestTUL;
2273
222k
        ctxBest = m_CABACEstimator->getCtx();
2274
222k
      }
2275
      // Test using joint chroma residual coding
2276
268k
      double     bestCostCbCr = bestCostCb + bestCostCr;
2277
268k
      int        bestJointCbCr = 0;
2278
268k
      bool checkDCTOnly = m_pcEncCfg->m_useChromaTS && ((TU::getCbf(tmpTU, COMP_Cb) && tmpTU.mtsIdx[COMP_Cb] == MTS_DCT2_DCT2 && !TU::getCbf(tmpTU, COMP_Cr)) ||
2279
0
        (TU::getCbf(tmpTU, COMP_Cr) && tmpTU.mtsIdx[COMP_Cr] == MTS_DCT2_DCT2 && !TU::getCbf(tmpTU, COMP_Cb)) ||
2280
0
        (TU::getCbf(tmpTU, COMP_Cb) && tmpTU.mtsIdx[COMP_Cb] == MTS_DCT2_DCT2 && TU::getCbf(tmpTU, COMP_Cr) && tmpTU.mtsIdx[COMP_Cr] == MTS_DCT2_DCT2));
2281
268k
      bool checkTSOnly = m_pcEncCfg->m_useChromaTS && ((TU::getCbf(tmpTU, COMP_Cb) && tmpTU.mtsIdx[COMP_Cb] == MTS_SKIP && !TU::getCbf(tmpTU, COMP_Cr)) ||
2282
0
        (TU::getCbf(tmpTU, COMP_Cr) && tmpTU.mtsIdx[COMP_Cr] == MTS_SKIP && !TU::getCbf(tmpTU, COMP_Cb)) ||
2283
0
        (TU::getCbf(tmpTU, COMP_Cb) && tmpTU.mtsIdx[COMP_Cb] == MTS_SKIP && TU::getCbf(tmpTU, COMP_Cr) && tmpTU.mtsIdx[COMP_Cr] == MTS_SKIP));
2284
268k
      bool       lastIsBest = false;
2285
268k
      bool noLFNST1 = false;
2286
268k
      if (rapidLFNST && (startLfnstIdx != endLfnstIdx))
2287
0
      {
2288
0
        if (bestLfnstIdx == 2)
2289
0
        {
2290
0
          noLFNST1 = true;
2291
0
        }
2292
0
        else
2293
0
        {
2294
0
          endLfnstIdx = 1;
2295
0
        }
2296
0
      }
2297
2298
980k
      for (int lfnstIdxj = startLfnstIdx; lfnstIdxj <= endLfnstIdx; lfnstIdxj++)
2299
712k
      {
2300
712k
        if (rapidLFNST && noLFNST1 && (lfnstIdxj == 1))
2301
0
        {
2302
0
          continue;
2303
0
        }
2304
712k
        currTU.cu->lfnstIdx = lfnstIdxj;
2305
712k
        std::vector<int> jointCbfMasksToTest;
2306
712k
        if (TU::getCbf(tmpTU, COMP_Cb) || TU::getCbf(tmpTU, COMP_Cr))
2307
252k
        {
2308
252k
          jointCbfMasksToTest = m_pcTrQuant->selectICTCandidates(currTU, m_orgResiCb, m_orgResiCr);
2309
252k
        }
2310
712k
        for (int cbfMask : jointCbfMasksToTest)
2311
252k
        {
2312
252k
          currTU.jointCbCr = (uint8_t)cbfMask;
2313
252k
          ComponentID codeCompId = ((currTU.jointCbCr >> 1) ? COMP_Cb : COMP_Cr);
2314
252k
          ComponentID otherCompId = ((codeCompId == COMP_Cb) ? COMP_Cr : COMP_Cb);
2315
252k
          bool tsAllowed = useTS && TU::isTSAllowed(currTU, codeCompId) && (m_pcEncCfg->m_useChromaTS) && !currTU.cu->lfnstIdx && !cu.bdpcmM[CH_C];
2316
252k
          if ((partitioner.chType == CH_L)&& tsAllowed && (currTU.mtsIdx[COMP_Y] != MTS_SKIP))
2317
0
          {
2318
0
            tsAllowed = false;
2319
0
          }
2320
252k
          if (!tsAllowed)
2321
252k
          {
2322
252k
            checkTSOnly = false;
2323
252k
          }
2324
252k
          uint8_t     numTransformCands = 1 + (tsAllowed && !(checkDCTOnly || checkTSOnly)? 1 : 0); // DCT + TS = 2 tests
2325
252k
          std::vector<TrMode> trModes;
2326
252k
          if (numTransformCands > 1)
2327
0
          {
2328
0
            trModes.push_back(TrMode(0, true)); // DCT2
2329
0
            trModes.push_back(TrMode(1, true));//TS
2330
0
          }
2331
252k
          else
2332
252k
          {
2333
252k
            currTU.mtsIdx[codeCompId] = checkTSOnly || currTU.cu->bdpcmM[CH_C] ? 1 : 0;
2334
252k
          }
2335
2336
505k
          for (int modeId = 0; modeId < numTransformCands; modeId++)
2337
252k
          {
2338
252k
            Distortion distTmp = 0;
2339
252k
            currTU.mtsIdx[codeCompId] = currTU.cu->bdpcmM[CH_C] ? MTS_SKIP : MTS_DCT2_DCT2;
2340
252k
            if (numTransformCands > 1)
2341
0
            {
2342
0
              currTU.mtsIdx[codeCompId] = currTU.cu->bdpcmM[CH_C] ? MTS_SKIP : trModes[modeId].first;
2343
0
            }
2344
252k
            currTU.mtsIdx[otherCompId] = MTS_DCT2_DCT2;
2345
2346
252k
            m_CABACEstimator->getCtx() = ctxStartTU;
2347
2348
252k
            resiCb.copyFrom(m_orgResiCb[cbfMask]);
2349
252k
            resiCr.copyFrom(m_orgResiCr[cbfMask]);
2350
252k
            if ((modeId == 0) && (numTransformCands > 1))
2351
0
            {
2352
0
              xPreCheckMTS(currTU, &trModes, m_pcEncCfg->m_MTSIntraMaxCand, 0, COMP_Cb);
2353
0
              currTU.mtsIdx[codeCompId] = trModes[modeId].first;
2354
0
              currTU.mtsIdx[(codeCompId == COMP_Cr) ? COMP_Cb : COMP_Cr] = MTS_DCT2_DCT2;
2355
0
            }
2356
252k
            cuCtx.lfnstLastScanPos = false;
2357
252k
            cuCtx.violatesLfnstConstrained[CH_L] = false;
2358
252k
            cuCtx.violatesLfnstConstrained[CH_C] = false;
2359
252k
            if (numTransformCands > 1)
2360
0
            {
2361
0
              xIntraCodingTUBlock(currTU, COMP_Cb, false, distTmp, 0, 0, true);
2362
0
              if ((modeId == 0) && !trModes[modeId + 1].second)
2363
0
              {
2364
0
                numTransformCands = 1;
2365
0
              }
2366
0
            }
2367
252k
            else
2368
252k
            {
2369
252k
              xIntraCodingTUBlock(currTU, COMP_Cb, false, distTmp, 0);
2370
252k
            }
2371
2372
252k
            double costTmp = std::numeric_limits<double>::max();
2373
252k
            if (distTmp < MAX_DISTORTION)
2374
249k
            {
2375
249k
              uint64_t bits = xGetIntraFracBitsQTChroma(currTU, COMP_Cb, &cuCtx);
2376
249k
              costTmp = m_pcRdCost->calcRdCost(bits, distTmp);
2377
249k
            }
2378
3.27k
            else if (!currTU.mtsIdx[codeCompId])
2379
3.27k
            {
2380
3.27k
              numTransformCands = 1;
2381
3.27k
            }
2382
252k
            bool rootCbfL = false;
2383
1.01M
            for (uint32_t t = 0; t < getNumberValidTBlocks(*cs.pcv); t++)
2384
757k
            {
2385
757k
              rootCbfL |= bool(tmpTU.cbf[t]);
2386
757k
            }
2387
252k
            if (rapidLFNST && !rootCbfL)
2388
0
            {
2389
0
              endLfnstIdx = lfnstIdxj;
2390
0
            }
2391
252k
            if (testLFNST && currTU.cu->lfnstIdx && !cuCtx.lfnstLastScanPos)
2392
3.22k
            {
2393
3.22k
              bool cbfAtZeroDepth = CU::isSepTree(*currTU.cu) ? rootCbfL
2394
3.22k
                : (cs.area.chromaFormat != CHROMA_400 && std::min(tmpTU.blocks[1].width, tmpTU.blocks[1].height) < 4)
2395
0
                ? TU::getCbfAtDepth(currTU, COMP_Y, currTU.depth) : rootCbfL;
2396
3.22k
              if (cbfAtZeroDepth)
2397
3.22k
              {
2398
3.22k
                costTmp = MAX_DOUBLE;
2399
3.22k
              }
2400
3.22k
            }
2401
252k
            if (costTmp < bestCostCbCr)
2402
95.2k
            {
2403
95.2k
              bestCostCbCr = costTmp;
2404
95.2k
              bestDistCbCr = distTmp;
2405
95.2k
              bestJointCbCr = currTU.jointCbCr;
2406
2407
              // store data
2408
95.2k
              bestLfnstIdx = lfnstIdxj;
2409
95.2k
              if ((cbfMask != jointCbfMasksToTest.back() || (lfnstIdxj != endLfnstIdx)) || (modeId != (numTransformCands - 1)))
2410
77.4k
              {
2411
77.4k
                saveCS.getRecoBuf(cbArea).copyFrom(cs.getRecoBuf(cbArea));
2412
77.4k
                saveCS.getRecoBuf(crArea).copyFrom(cs.getRecoBuf(crArea));
2413
2414
77.4k
                tmpTU.copyComponentFrom(currTU, COMP_Cb);
2415
77.4k
                tmpTU.copyComponentFrom(currTU, COMP_Cr);
2416
2417
77.4k
                ctxBest = m_CABACEstimator->getCtx();
2418
77.4k
              }
2419
17.7k
              else
2420
17.7k
              {
2421
17.7k
                lastIsBest = true;
2422
17.7k
                cs.cus[0]->lfnstIdx = bestLfnstIdx;
2423
17.7k
              }
2424
95.2k
            }
2425
252k
          }
2426
252k
        }
2427
2428
        // Retrieve the best CU data (unless it was the very last one tested)
2429
712k
      }
2430
268k
      if (!lastIsBest)
2431
250k
      {
2432
250k
        cs.getRecoBuf(cbArea).copyFrom(saveCS.getRecoBuf(cbArea));
2433
250k
        cs.getRecoBuf(crArea).copyFrom(saveCS.getRecoBuf(crArea));
2434
2435
250k
        cs.cus[0]->lfnstIdx = bestLfnstIdx;
2436
250k
        currTU.copyComponentFrom(tmpTU, COMP_Cb);
2437
250k
        currTU.copyComponentFrom(tmpTU, COMP_Cr);
2438
250k
        m_CABACEstimator->getCtx() = ctxBest;
2439
250k
      }
2440
268k
      currTU.jointCbCr = (TU::getCbf(currTU, COMP_Cb) || TU::getCbf(currTU, COMP_Cr)) ? bestJointCbCr : 0;
2441
268k
    } // jointCbCr
2442
2443
268k
    cs.dist += bestDistCbCr;
2444
268k
    cuCtx.violatesLfnstConstrained[CH_L] = false;
2445
268k
    cuCtx.violatesLfnstConstrained[CH_C] = false;
2446
268k
    cuCtx.lfnstLastScanPos = false;
2447
268k
    cuCtx.violatesMtsCoeffConstraint = false;
2448
268k
    cuCtx.mtsLastScanPos = false;
2449
268k
    cbfs.cbf(COMP_Cb) = TU::getCbf(currTU, COMP_Cb);
2450
268k
    cbfs.cbf(COMP_Cr) = TU::getCbf(currTU, COMP_Cr);
2451
268k
  }
2452
1
  else
2453
1
  {
2454
1
    unsigned   numValidTBlocks = getNumberValidTBlocks(*cs.pcv);
2455
1
    ChromaCbfs SplitCbfs(false);
2456
2457
1
    if (partitioner.canSplit(TU_MAX_TR_SPLIT, cs))
2458
0
    {
2459
0
      partitioner.splitCurrArea(TU_MAX_TR_SPLIT, cs);
2460
0
    }
2461
1
    else if (currTU.cu->ispMode)
2462
0
    {
2463
0
      partitioner.splitCurrArea(m_ispTestedModes[0].IspType, cs);
2464
0
    }
2465
1
    else
2466
1
      THROW("Implicit TU split not available");
2467
2468
0
    do
2469
0
    {
2470
0
      ChromaCbfs subCbfs = xIntraChromaCodingQT(cs, partitioner);
2471
2472
0
      for (uint32_t ch = COMP_Cb; ch < numValidTBlocks; ch++)
2473
0
      {
2474
0
        const ComponentID compID = ComponentID(ch);
2475
0
        SplitCbfs.cbf(compID) |= subCbfs.cbf(compID);
2476
0
      }
2477
0
    } while (partitioner.nextPart(cs));
2478
2479
0
    partitioner.exitCurrSplit();
2480
2481
    /*if (lumaUsesISP && cs.dist == MAX_UINT) //ahenkel
2482
    {
2483
      return cbfs;
2484
    }*/
2485
0
    {
2486
0
      cbfs.Cb |= SplitCbfs.Cb;
2487
0
      cbfs.Cr |= SplitCbfs.Cr;
2488
2489
0
      if (1)   //(!lumaUsesISP)
2490
0
      {
2491
0
        for (auto& ptu : cs.tus)
2492
0
        {
2493
0
          if (currArea.Cb().contains(ptu->Cb()) || (!ptu->Cb().valid() && currArea.Y().contains(ptu->Y())))
2494
0
          {
2495
0
            TU::setCbfAtDepth(*ptu, COMP_Cb, currDepth, SplitCbfs.Cb);
2496
0
            TU::setCbfAtDepth(*ptu, COMP_Cr, currDepth, SplitCbfs.Cr);
2497
0
          }
2498
0
        }
2499
0
      }
2500
0
    }
2501
0
  }
2502
268k
  return cbfs;
2503
268k
}
2504
2505
uint64_t IntraSearch::xFracModeBitsIntraLuma(const CodingUnit& cu, const unsigned* mpmLst)
2506
893k
{
2507
893k
  m_CABACEstimator->resetBits();
2508
2509
893k
  if (!cu.ciip)
2510
893k
  {
2511
893k
    m_CABACEstimator->intra_luma_pred_mode(cu, mpmLst);
2512
893k
  }
2513
2514
893k
  return m_CABACEstimator->getEstFracBits();
2515
893k
}
2516
2517
template<typename T, size_t N, int M>
2518
void IntraSearch::xReduceHadCandList(static_vector<T, N>& candModeList, static_vector<double, N>& candCostList, SortedPelUnitBufs<M>& sortedPelBuffer, int& numModesForFullRD, const double thresholdHadCost, const double* mipHadCost, const CodingUnit& cu, const bool fastMip)
2519
18.2k
{
2520
18.2k
  const int maxCandPerType = numModesForFullRD >> 1;
2521
18.2k
  static_vector<ModeInfo, FAST_UDI_MAX_RDMODE_NUM> tempRdModeList;
2522
18.2k
  static_vector<double, FAST_UDI_MAX_RDMODE_NUM> tempCandCostList;
2523
18.2k
  const double minCost = candCostList[0];
2524
18.2k
  bool keepOneMip = candModeList.size() > numModesForFullRD;
2525
18.2k
  const int maxNumConv = 3; 
2526
2527
18.2k
  int numConv = 0;
2528
18.2k
  int numMip = 0;
2529
82.5k
  for (int idx = 0; idx < candModeList.size() - (keepOneMip?0:1); idx++)
2530
64.3k
  {
2531
64.3k
    bool addMode = false;
2532
64.3k
    const ModeInfo& orgMode = candModeList[idx];
2533
2534
64.3k
    if (!orgMode.mipFlg)
2535
46.0k
    {
2536
46.0k
      addMode = (numConv < maxNumConv);
2537
46.0k
      numConv += addMode ? 1:0;
2538
46.0k
    }
2539
18.2k
    else
2540
18.2k
    {
2541
18.2k
      addMode = ( numMip < maxCandPerType || (candCostList[idx] < thresholdHadCost * minCost) || keepOneMip );
2542
18.2k
      keepOneMip = false;
2543
18.2k
      numMip += addMode ? 1:0;
2544
18.2k
    }
2545
64.3k
    if( addMode )
2546
64.3k
    {
2547
64.3k
      tempRdModeList.push_back(orgMode);
2548
64.3k
      tempCandCostList.push_back(candCostList[idx]);
2549
64.3k
    }
2550
64.3k
  }
2551
2552
  // sort Pel Buffer
2553
18.2k
  int i = -1;
2554
18.2k
  for( auto &m: tempRdModeList)
2555
64.3k
  {
2556
64.3k
    if( ! (m == candModeList.at( ++i )) )
2557
0
    {
2558
0
      for( int j = i; j < (int)candModeList.size()-1; )
2559
0
      {
2560
0
        if( m == candModeList.at( ++j ) )
2561
0
        {
2562
0
          sortedPelBuffer.swap( i, j);
2563
0
          break;
2564
0
        }
2565
0
      }
2566
0
    }
2567
64.3k
  }
2568
18.2k
  sortedPelBuffer.reduceTo( (int)tempRdModeList.size() );
2569
2570
18.2k
  if ((cu.lwidth() > 8 && cu.lheight() > 8))
2571
16.3k
  {
2572
    // Sort MIP candidates by Hadamard cost
2573
16.3k
    const int transpOff = getNumModesMip(cu.Y());
2574
16.3k
    static_vector<uint8_t, FAST_UDI_MAX_RDMODE_NUM> sortedMipModes(0);
2575
16.3k
    static_vector<double, FAST_UDI_MAX_RDMODE_NUM> sortedMipCost(0);
2576
16.3k
    for (uint8_t mode : { 0, 1, 2 })
2577
49.1k
    {
2578
49.1k
      uint8_t candMode = mode + uint8_t((mipHadCost[mode + transpOff] < mipHadCost[mode]) ? transpOff : 0);
2579
49.1k
      updateCandList(candMode, mipHadCost[candMode], sortedMipModes, sortedMipCost, 3);
2580
49.1k
    }
2581
2582
    // Append MIP mode to RD mode list
2583
16.3k
    const int modeListSize = int(tempRdModeList.size());
2584
32.7k
    for (int idx = 0; idx < 3; idx++)
2585
32.7k
    {
2586
32.7k
      const bool     isTransposed = (sortedMipModes[idx] >= transpOff ? true : false);
2587
32.7k
      const uint32_t mipIdx       = (isTransposed ? sortedMipModes[idx] - transpOff : sortedMipModes[idx]);
2588
32.7k
      const ModeInfo mipMode( true, isTransposed, 0, NOT_INTRA_SUBPARTITIONS, mipIdx );
2589
32.7k
      bool alreadyIncluded = false;
2590
130k
      for (int modeListIdx = 0; modeListIdx < modeListSize; modeListIdx++)
2591
114k
      {
2592
114k
        if (tempRdModeList[modeListIdx] == mipMode)
2593
16.3k
        {
2594
16.3k
          alreadyIncluded = true;
2595
16.3k
          break;
2596
16.3k
        }
2597
114k
      }
2598
2599
32.7k
      if (!alreadyIncluded)
2600
16.3k
      {
2601
16.3k
        tempRdModeList.push_back(mipMode);
2602
16.3k
        tempCandCostList.push_back(0);
2603
16.3k
        if( fastMip ) break;
2604
16.3k
      }
2605
32.7k
    }
2606
16.3k
  }
2607
2608
18.2k
  candModeList = tempRdModeList;
2609
18.2k
  candCostList = tempCandCostList;
2610
18.2k
  numModesForFullRD = int(candModeList.size());
2611
18.2k
}
2612
2613
void IntraSearch::xPreCheckMTS(TransformUnit &tu, std::vector<TrMode> *trModes, const int maxCand, PelUnitBuf *predBuf, const ComponentID& compID)
2614
13.8k
{
2615
13.8k
  if (compID == COMP_Y)
2616
13.8k
  {
2617
13.8k
    CodingStructure&  cs = *tu.cs;
2618
13.8k
    const CompArea& area = tu.blocks[compID];
2619
13.8k
    const CodingUnit& cu = *cs.getCU(area.pos(), CH_L,TREE_D);
2620
13.8k
    PelBuf piPred = cs.getPredBuf(area);
2621
13.8k
    PelBuf piResi = cs.getResiBuf(area);
2622
2623
13.8k
    initIntraPatternChType(*tu.cu, area);
2624
13.8k
    if (predBuf)
2625
12.3k
    {
2626
12.3k
      piPred.copyFrom(predBuf->Y());
2627
12.3k
    }
2628
1.47k
    else if (CU::isMIP(cu, CH_L))
2629
1.45k
    {
2630
1.45k
      initIntraMip(cu);
2631
1.45k
      predIntraMip(piPred, cu);
2632
1.45k
    }
2633
20
    else
2634
20
    {
2635
20
      predIntraAng(COMP_Y, piPred, cu);
2636
20
    }
2637
2638
    //===== get residual signal =====
2639
13.8k
    CPelBuf piOrg = cs.getOrgBuf(COMP_Y);
2640
13.8k
    piResi.subtract(piOrg, piPred);
2641
13.8k
    m_pcTrQuant->checktransformsNxN(tu, trModes, m_pcEncCfg->m_MTSIntraMaxCand, compID);
2642
13.8k
  }
2643
0
  else
2644
0
  {
2645
0
    ComponentID codeCompId = (tu.jointCbCr ? (tu.jointCbCr >> 1 ? COMP_Cb : COMP_Cr) : compID);
2646
0
    m_pcTrQuant->checktransformsNxN(tu, trModes, m_pcEncCfg->m_MTSIntraMaxCand, codeCompId);
2647
0
  }
2648
13.8k
}
2649
2650
double IntraSearch::xTestISP(CodingStructure& cs, Partitioner& subTuPartitioner, double bestCostForISP, PartSplit ispType, bool& splitcbf, uint64_t& singleFracBits, Distortion& singleDistLuma, CUCtx& cuCtx)
2651
14.6k
{
2652
14.6k
  int  subTuCounter = 0;
2653
14.6k
  bool earlySkipISP = false;
2654
14.6k
  bool splitCbfLuma = false;
2655
14.6k
  CodingUnit& cu = *cs.cus[0];
2656
2657
14.6k
  Distortion singleDistTmpLumaSUM = 0;
2658
14.6k
  uint64_t   singleTmpFracBitsSUM = 0;
2659
14.6k
  double     singleCostTmpSUM = 0;
2660
14.6k
  cuCtx.isDQPCoded = true;
2661
14.6k
  cuCtx.isChromaQpAdjCoded = true;
2662
2663
14.6k
  do
2664
18.7k
  {
2665
18.7k
    Distortion singleDistTmpLuma = 0;
2666
18.7k
    uint64_t   singleTmpFracBits = 0;
2667
18.7k
    double     singleCostTmp = 0;
2668
18.7k
    TransformUnit& tmpTUcur = ((cs.tus.size() < (subTuCounter + 1)))
2669
18.7k
      ? cs.addTU(CS::getArea(cs, subTuPartitioner.currArea(), subTuPartitioner.chType,
2670
3.40k
        subTuPartitioner.treeType),
2671
3.40k
        subTuPartitioner.chType, cs.cus[0])
2672
18.7k
      : *cs.tus[subTuCounter];
2673
18.7k
    tmpTUcur.depth = subTuPartitioner.currTrDepth;
2674
2675
    // Encode TU
2676
18.7k
    xIntraCodingTUBlock(tmpTUcur, COMP_Y, false, singleDistTmpLuma, 0);
2677
18.7k
    cuCtx.mtsLastScanPos = false;
2678
2679
18.7k
    if (singleDistTmpLuma == MAX_INT)   // all zero CBF skip
2680
0
    {
2681
0
      earlySkipISP = true;
2682
0
      singleCostTmpSUM = MAX_DOUBLE;
2683
0
      break;
2684
0
    }
2685
2686
18.7k
    if (m_pcRdCost->calcRdCost(singleTmpFracBitsSUM, singleDistTmpLumaSUM + singleDistTmpLuma) > bestCostForISP)
2687
4.88k
    {
2688
4.88k
      earlySkipISP = true;
2689
4.88k
    }
2690
13.8k
    else
2691
13.8k
    {
2692
13.8k
      m_ispTestedModes[0].IspType = ispType;
2693
13.8k
      m_ispTestedModes[0].subTuCounter = subTuCounter;
2694
13.8k
      singleTmpFracBits = xGetIntraFracBitsQT(cs, subTuPartitioner, true, &cuCtx);
2695
13.8k
    }
2696
18.7k
    singleCostTmp = m_pcRdCost->calcRdCost(singleTmpFracBits, singleDistTmpLuma);
2697
2698
18.7k
    singleCostTmpSUM     += singleCostTmp;
2699
18.7k
    singleDistTmpLumaSUM += singleDistTmpLuma;
2700
18.7k
    singleTmpFracBitsSUM += singleTmpFracBits;
2701
2702
18.7k
    subTuCounter++;
2703
2704
18.7k
    splitCbfLuma |= TU::getCbfAtDepth( *cs.getTU(subTuPartitioner.currArea().lumaPos(), subTuPartitioner.chType, subTuCounter - 1), 
2705
18.7k
                                       COMP_Y, subTuPartitioner.currTrDepth);
2706
18.7k
    int nSubPartitions = m_ispTestedModes[cu.lfnstIdx].numTotalParts[cu.ispMode - 1];
2707
18.7k
    bool doStop = (m_pcEncCfg->m_ISP != 1) || (subTuCounter < nSubPartitions);
2708
18.7k
    if (doStop)
2709
18.7k
    {
2710
18.7k
      if (singleCostTmpSUM > bestCostForISP)
2711
12.2k
      {
2712
12.2k
        earlySkipISP = true;
2713
12.2k
        break;
2714
12.2k
      }
2715
6.49k
      if (subTuCounter < nSubPartitions)
2716
5.18k
      {
2717
5.18k
        double threshold = nSubPartitions == 2 ? 0.95 : subTuCounter == 1 ? 0.83 : 0.91;
2718
5.18k
        if (singleCostTmpSUM > bestCostForISP * threshold)
2719
1.09k
        {
2720
1.09k
          earlySkipISP = true;
2721
1.09k
          break;
2722
1.09k
        }
2723
5.18k
      }
2724
6.49k
    }
2725
18.7k
  } while (subTuPartitioner.nextPart(cs));
2726
14.6k
  singleDistLuma = singleDistTmpLumaSUM;
2727
14.6k
  singleFracBits = singleTmpFracBitsSUM;
2728
2729
14.6k
  splitcbf = splitCbfLuma;
2730
14.6k
  return earlySkipISP ? MAX_DOUBLE : singleCostTmpSUM;
2731
14.6k
}
2732
2733
int IntraSearch::xSpeedUpISP(int speed, bool& testISP, int mode, int& noISP, int& endISP, CodingUnit& cu, static_vector<ModeInfo, FAST_UDI_MAX_RDMODE_NUM>& RdModeList, const ModeInfo& bestPUMode, int bestISP, int bestLfnstIdx)
2734
13.2k
{
2735
13.2k
  if (speed)
2736
5.36k
  {
2737
5.36k
    if (mode >= 1)
2738
2.82k
    {
2739
2.82k
      if (m_ispTestedModes[0].splitIsFinished[1] && m_ispTestedModes[0].splitIsFinished[0])
2740
0
      {
2741
0
        testISP = false;
2742
0
        endISP = 0;
2743
0
      }
2744
2.82k
      else
2745
2.82k
      {
2746
2.82k
        if (m_pcEncCfg->m_ISP >= 2)
2747
2.82k
        {
2748
2.82k
          if (mode == 1) //best Hor||Ver
2749
2.54k
          {
2750
2.54k
            int bestDir = 0;
2751
7.64k
            for (int d = 0; d < 2; d++)
2752
5.09k
            {
2753
5.09k
              int d2 = d ? 0 : 1;
2754
5.09k
              if ((m_ispTestedModes[0].bestCost[d] <= m_ispTestedModes[0].bestCost[d2])
2755
4.82k
                && (m_ispTestedModes[0].bestCost[d] != MAX_DOUBLE))
2756
273
              {
2757
273
                bestDir = d + 1;
2758
273
                m_ispTestedModes[0].splitIsFinished[d2] = true;
2759
273
              }
2760
5.09k
            }
2761
2.54k
            m_ispTestedModes[0].bestModeSoFar = bestDir;
2762
2.54k
            if (m_ispTestedModes[0].bestModeSoFar <= 0)
2763
2.27k
            {
2764
2.27k
              m_ispTestedModes[0].splitIsFinished[1] = true;
2765
2.27k
              m_ispTestedModes[0].splitIsFinished[0] = true;
2766
2.27k
              testISP = false;
2767
2.27k
              endISP = 0;
2768
2.27k
            }
2769
2.54k
          }
2770
2.82k
          if (m_ispTestedModes[0].bestModeSoFar == 2)
2771
78
          {
2772
78
            noISP = 1;
2773
78
          }
2774
2.74k
          else
2775
2.74k
          {
2776
2.74k
            endISP = 1;
2777
2.74k
          }
2778
2.82k
        }
2779
2.82k
      }
2780
2.82k
    }
2781
5.36k
    if (testISP)
2782
3.09k
    {
2783
3.09k
      if (mode == 2)
2784
273
      {
2785
819
        for (int d = 0; d < 2; d++)
2786
546
        {
2787
546
          int d2 = d ? 0 : 1;
2788
546
          if (m_ispTestedModes[0].bestCost[d] == MAX_DOUBLE)
2789
249
          {
2790
249
            m_ispTestedModes[0].splitIsFinished[d] = true;
2791
249
          }
2792
546
          if ((m_ispTestedModes[0].bestCost[d2] < 1.3 * m_ispTestedModes[0].bestCost[d])
2793
297
            && (int(m_ispTestedModes[0].bestSplitSoFar) != (d + 1)))
2794
226
          {
2795
226
            if (d)
2796
187
            {
2797
187
              endISP = 1;
2798
187
            }
2799
39
            else
2800
39
            {
2801
39
              noISP = 1;
2802
39
            }
2803
226
            m_ispTestedModes[0].splitIsFinished[d] = true;
2804
226
          }
2805
546
        }
2806
273
      }
2807
2.82k
      else
2808
2.82k
      {
2809
2.82k
        if (m_ispTestedModes[0].splitIsFinished[0])
2810
39
        {
2811
39
          noISP = 1;
2812
39
        }
2813
2.82k
        if (m_ispTestedModes[0].splitIsFinished[1])
2814
234
        {
2815
234
          endISP = 1;
2816
234
        }
2817
2.82k
      }
2818
3.09k
    }
2819
5.36k
    if ((noISP == 1) && (endISP == 1))
2820
24
    {
2821
24
      endISP = 0;
2822
24
    }
2823
5.36k
  }
2824
7.88k
  else
2825
7.88k
  {
2826
7.88k
    bool stopFound = false;
2827
7.88k
    if (m_pcEncCfg->m_ISP >= 3)
2828
7.88k
    {
2829
7.88k
      if (mode)
2830
2.79k
      {
2831
2.79k
        if ((bestISP == 0) || ((bestPUMode.modeId != RdModeList[mode - 1].modeId)
2832
87
          && (bestPUMode.modeId != RdModeList[mode].modeId)))
2833
1.92k
        {
2834
1.92k
          stopFound = true;
2835
1.92k
        }
2836
2.79k
      }
2837
7.88k
    }
2838
7.88k
    if (cu.mipFlag || cu.multiRefIdx)
2839
163
    {
2840
163
      cu.mipFlag = false;
2841
163
      cu.multiRefIdx = 0;
2842
163
      if (!stopFound)
2843
0
      {
2844
0
        for (int k = 0; k < mode; k++)
2845
0
        {
2846
0
          if (cu.intraDir[CH_L] == RdModeList[k].modeId)
2847
0
          {
2848
0
            stopFound = true;
2849
0
            break;
2850
0
          }
2851
0
        }
2852
0
      }
2853
163
    }
2854
7.88k
    if (stopFound)
2855
1.92k
    {
2856
1.92k
      testISP = false;
2857
1.92k
      endISP = 0;
2858
1.92k
      return 1;
2859
1.92k
    }
2860
5.95k
    if (!stopFound && (m_pcEncCfg->m_ISP >= 2) && (cu.intraDir[CH_L] == DC_IDX))
2861
877
    {
2862
877
      stopFound = true;
2863
877
      endISP = 0;
2864
877
      return 1;
2865
877
    }
2866
5.95k
  }
2867
10.4k
  return 0;
2868
13.2k
}
2869
2870
void IntraSearch::xSpeedUpIntra(double bestcost, int& EndMode, int& speedIntra, CodingUnit& cu)
2871
23.9k
{
2872
23.9k
  int bestIdxbefore = m_ispTestedModes[0].bestIntraMode;
2873
23.9k
  if (m_ispTestedModes[0].isIntra)
2874
0
  {
2875
0
    if (bestIdxbefore == 1)//ISP
2876
0
    {
2877
0
      speedIntra = 14;
2878
0
    }
2879
0
    if (bestIdxbefore == 4)//MTS
2880
0
    {
2881
0
      speedIntra = 3;
2882
0
    }
2883
0
  }
2884
23.9k
  else if (!cu.cs->slice->isIntra())
2885
0
  {
2886
0
    if (bestcost != MAX_DOUBLE)
2887
0
    {
2888
0
      speedIntra = 10;
2889
0
    }
2890
0
  }
2891
23.9k
  if (m_ispTestedModes[0].bestBefore[0] == -1)
2892
21.3k
  {
2893
21.3k
    speedIntra |= 7;
2894
21.3k
    if (m_pcEncCfg->m_FastIntraTools == 2)
2895
0
    {
2896
0
      EndMode = 1;
2897
0
    }
2898
21.3k
  }
2899
23.9k
  if (!cu.cs->slice->isIntra())
2900
0
  {
2901
0
    if ((m_ispTestedModes[0].bestBefore[1] == 1) || (m_ispTestedModes[0].bestBefore[2] == 1))
2902
0
    {
2903
0
      speedIntra |= 2;
2904
0
    }
2905
0
    if ((m_ispTestedModes[0].bestBefore[1] == 4) || (m_ispTestedModes[0].bestBefore[2] == 4))
2906
0
    {
2907
0
      speedIntra |= 3;
2908
0
    }
2909
0
    if ((m_ispTestedModes[0].bestBefore[1] == 2) || (m_ispTestedModes[0].bestBefore[2] == 2))
2910
0
    {
2911
0
      speedIntra |= 1;
2912
0
    }
2913
0
  }
2914
23.9k
}
2915
2916
} // namespace vvenc
2917
2918
//! \}
2919