Coverage Report

Created: 2026-07-30 06:27

next uncovered line (L), next uncovered region (R), next uncovered branch (B)
/work/vvenc/source/Lib/EncoderLib/IntraSearch.cpp
Line
Count
Source
1
/* -----------------------------------------------------------------------------
2
The copyright in this software is being made available under the Clear BSD
3
License, included below. No patent rights, trademark rights and/or 
4
other Intellectual Property Rights other than the copyrights concerning 
5
the Software are granted under this license.
6
7
The Clear BSD License
8
9
Copyright (c) 2019-2026, Fraunhofer-Gesellschaft zur Förderung der angewandten Forschung e.V. & The VVenC Authors.
10
All rights reserved.
11
12
Redistribution and use in source and binary forms, with or without modification,
13
are permitted (subject to the limitations in the disclaimer below) provided that
14
the following conditions are met:
15
16
     * Redistributions of source code must retain the above copyright notice,
17
     this list of conditions and the following disclaimer.
18
19
     * Redistributions in binary form must reproduce the above copyright
20
     notice, this list of conditions and the following disclaimer in the
21
     documentation and/or other materials provided with the distribution.
22
23
     * Neither the name of the copyright holder nor the names of its
24
     contributors may be used to endorse or promote products derived from this
25
     software without specific prior written permission.
26
27
NO EXPRESS OR IMPLIED LICENSES TO ANY PARTY'S PATENT RIGHTS ARE GRANTED BY
28
THIS LICENSE. THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND
29
CONTRIBUTORS "AS IS" AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT
30
LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A
31
PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR
32
CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL,
33
EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO,
34
PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR
35
BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER
36
IN CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE)
37
ARISING IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE
38
POSSIBILITY OF SUCH DAMAGE.
39
40
41
------------------------------------------------------------------------------------------- */
42
43
44
/** \file     EncSearch.cpp
45
 *  \brief    encoder intra search class
46
 */
47
48
#include "IntraSearch.h"
49
#include "EncPicture.h"
50
#include "CommonLib/CommonDef.h"
51
#include "CommonLib/Rom.h"
52
#include "CommonLib/Picture.h"
53
#include "CommonLib/UnitTools.h"
54
#include "CommonLib/dtrace_next.h"
55
#include "CommonLib/dtrace_buffer.h"
56
#include "CommonLib/Reshape.h"
57
#include <math.h>
58
#include "vvenc/vvencCfg.h"
59
60
//! \ingroup EncoderLib
61
//! \{
62
63
namespace vvenc {
64
65
#define PLTCtx(c) SubCtx( Ctx::Palette, c )
66
67
IntraSearch::IntraSearch()
68
19.4k
  : m_pSaveCS       (nullptr)
69
19.4k
  , m_pcEncCfg      (nullptr)
70
19.4k
  , m_pcTrQuant     (nullptr)
71
19.4k
  , m_pcRdCost      (nullptr)
72
19.4k
  , m_CABACEstimator(nullptr)
73
19.4k
  , m_CtxCache      (nullptr)
74
19.4k
{
75
19.4k
}
76
77
void IntraSearch::init(const VVEncCfg &encCfg, TrQuant *pTrQuant, RdCost *pRdCost, SortedPelUnitBufs<SORTED_BUFS> *pSortedPelUnitBufs, XUCache &unitCache )
78
19.4k
{
79
19.4k
  IntraPrediction::init( encCfg.m_internChromaFormat, encCfg.m_internalBitDepth[ CH_L ] );
80
81
19.4k
  m_pcEncCfg          = &encCfg;
82
19.4k
  m_pcTrQuant         = pTrQuant;
83
19.4k
  m_pcRdCost          = pRdCost;
84
19.4k
  m_SortedPelUnitBufs = pSortedPelUnitBufs;
85
86
19.4k
  const ChromaFormat chrFormat = encCfg.m_internChromaFormat;
87
19.4k
  const int maxCUSize          = encCfg.m_CTUSize;
88
89
19.4k
  Area area = Area( 0, 0, maxCUSize, maxCUSize );
90
91
19.4k
  m_pTempCS = new CodingStructure( unitCache, nullptr );
92
19.4k
  m_pBestCS = new CodingStructure( unitCache, nullptr );
93
94
19.4k
  m_pTempCS->createForSearch( chrFormat, area );
95
19.4k
  m_pBestCS->createForSearch( chrFormat, area );
96
97
19.4k
  const int uiNumSaveLayersToAllocate = 3;
98
19.4k
  m_pSaveCS = new CodingStructure*[uiNumSaveLayersToAllocate];
99
77.8k
  for( int layer = 0; layer < uiNumSaveLayersToAllocate; layer++ )
100
58.3k
  {
101
58.3k
    m_pSaveCS[ layer ] = new CodingStructure( unitCache, nullptr );
102
58.3k
    m_pSaveCS[ layer ]->createForSearch( chrFormat, Area( 0, 0, maxCUSize, maxCUSize ) );
103
58.3k
    m_pSaveCS[ layer ]->initStructData();
104
58.3k
  }
105
106
19.4k
  CompArea chromaArea( COMP_Cb, chrFormat, area, true );
107
116k
  for( int i = 0; i < 5; i++ )
108
97.2k
  {
109
97.2k
    m_orgResiCb[i].create( chromaArea );
110
97.2k
    m_orgResiCr[i].create( chromaArea );
111
97.2k
  }
112
19.4k
}
113
114
void IntraSearch::destroy()
115
19.4k
{
116
19.4k
  if ( m_pSaveCS )
117
19.4k
  {
118
19.4k
    const int uiNumSaveLayersToAllocate = 3;
119
77.8k
    for( int layer = 0; layer < uiNumSaveLayersToAllocate; layer++ )
120
58.3k
    {
121
58.3k
      if ( m_pSaveCS[ layer ] ) { m_pSaveCS[ layer ]->destroy(); delete m_pSaveCS[ layer ]; }
122
58.3k
    }
123
19.4k
    delete[] m_pSaveCS;
124
19.4k
    m_pSaveCS = nullptr;
125
19.4k
  }
126
127
19.4k
  if( m_pTempCS )
128
19.4k
  {
129
19.4k
    m_pTempCS->destroy();
130
19.4k
    delete m_pTempCS; m_pTempCS = nullptr;
131
19.4k
  }
132
133
19.4k
  if( m_pBestCS )
134
19.4k
  {
135
19.4k
    m_pBestCS->destroy();
136
19.4k
    delete m_pBestCS; m_pBestCS = nullptr;
137
19.4k
  }
138
19.4k
}
139
140
IntraSearch::~IntraSearch()
141
19.4k
{
142
19.4k
  destroy();
143
19.4k
}
144
145
void IntraSearch::setCtuEncRsrc( CABACWriter* cabacEstimator, CtxCache *ctxCache )
146
3.82k
{
147
3.82k
  m_CABACEstimator = cabacEstimator;
148
3.82k
  m_CtxCache       = ctxCache;
149
3.82k
}
150
151
//////////////////////////////////////////////////////////////////////////
152
// INTRA PREDICTION
153
//////////////////////////////////////////////////////////////////////////
154
static constexpr double COST_UNKNOWN = -65536.0;
155
156
double IntraSearch::xFindInterCUCost( CodingUnit &cu )
157
25.4k
{
158
25.4k
  if( CU::isConsIntra(cu) && !cu.slice->isIntra() )
159
0
  {
160
    //search corresponding inter CU cost
161
0
    for( int i = 0; i < m_numCuInSCIPU; i++ )
162
0
    {
163
0
      if( cu.lumaPos() == m_cuAreaInSCIPU[i].pos() && cu.lumaSize() == m_cuAreaInSCIPU[i].size() )
164
0
      {
165
0
        return m_cuCostInSCIPU[i];
166
0
      }
167
0
    }
168
0
  }
169
25.4k
  return COST_UNKNOWN;
170
25.4k
}
171
172
void IntraSearch::xEstimateLumaRdModeList(int& numModesForFullRD,
173
  static_vector<ModeInfo, FAST_UDI_MAX_RDMODE_NUM>& RdModeList,
174
  static_vector<ModeInfo, FAST_UDI_MAX_RDMODE_NUM>& HadModeList,
175
  static_vector<double, FAST_UDI_MAX_RDMODE_NUM>& CandCostList,
176
  static_vector<double, FAST_UDI_MAX_RDMODE_NUM>& CandHadList, CodingUnit& cu, bool testMip )
177
25.4k
{
178
25.4k
  PROFILER_SCOPE_AND_STAGE_EXT( 1, _TPROF, P_INTRA_EST_RD_CAND, cu.cs, CH_L );
179
25.4k
  const uint16_t intra_ctx_size = Ctx::IntraLumaMpmFlag.size() + Ctx::IntraLumaPlanarFlag.size() + Ctx::MultiRefLineIdx.size() + Ctx::ISPMode.size() + Ctx::MipFlag.size();
180
25.4k
  const TempCtx  ctxStartIntraCtx(m_CtxCache, SubCtx(CtxSet(Ctx::IntraLumaMpmFlag(), intra_ctx_size), m_CABACEstimator->getCtx()));
181
25.4k
  const double   sqrtLambdaForFirstPass = m_pcRdCost->getMotionLambda() * FRAC_BITS_SCALE;
182
25.4k
  const int numModesAvailable = NUM_LUMA_MODE; // total number of Intra modes
183
184
25.4k
  CHECK(numModesForFullRD >= numModesAvailable, "Too many modes for full RD search");
185
186
25.4k
  const SPS& sps     = *cu.cs->sps;
187
25.4k
  const bool fastMip = sps.MIP && m_pcEncCfg->m_useFastMIP;
188
189
  // this should always be true
190
25.4k
  CHECK( !cu.Y().valid(), "CU is not valid" );
191
192
25.4k
  const CompArea& area = cu.Y();
193
194
25.4k
  const UnitArea localUnitArea(area.chromaFormat, Area(0, 0, area.width, area.height));
195
25.4k
  if( testMip)
196
19.3k
  {
197
19.3k
    numModesForFullRD += fastMip ? numModesForFullRD - std::min( m_pcEncCfg->m_useFastMIP, numModesForFullRD )
198
19.3k
                                 : numModesForFullRD;
199
19.3k
    m_SortedPelUnitBufs->prepare( localUnitArea, numModesForFullRD + 1 );
200
19.3k
  }
201
6.09k
  else
202
6.09k
  {
203
6.09k
    m_SortedPelUnitBufs->prepare( localUnitArea, numModesForFullRD );
204
6.09k
  }
205
206
25.4k
  CPelBuf piOrg   = cu.cs->getOrgBuf(COMP_Y);
207
25.4k
  PelBuf piPred  = m_SortedPelUnitBufs->getTestBuf(COMP_Y);
208
209
25.4k
  const ReshapeData& reshapeData = cu.cs->picture->reshapeData;
210
25.4k
  if (cu.cs->picHeader->lmcsEnabled && reshapeData.getCTUFlag())
211
0
  {
212
0
    piOrg = cu.cs->getRspOrgBuf();
213
0
  }
214
25.4k
  DistParam distParam    = m_pcRdCost->setDistParam( piOrg, piPred, sps.bitDepths[ CH_L ], DF_HAD_2SAD); // Use HAD (SATD) cost
215
216
25.4k
  const int numHadCand = (testMip ? 2 : 1) * 3;
217
218
  //*** Derive (regular) candidates using Hadamard
219
25.4k
  cu.mipFlag = false;
220
25.4k
  cu.multiRefIdx = 0;
221
222
  //===== init pattern for luma prediction =====
223
25.4k
  initIntraPatternChType(cu, cu.Y(), true);
224
225
25.4k
  bool satdChecked[NUM_INTRA_MODE] = { false };
226
227
25.4k
  unsigned mpmLst[NUM_MOST_PROBABLE_MODES];
228
25.4k
  CU::getIntraMPMs(cu, mpmLst);
229
230
25.4k
  const int decMsk = ( 1 << m_pcEncCfg->m_IntraEstDecBit ) - 1;
231
232
25.4k
  m_parentCandList.resize( 0 );
233
25.4k
  m_parentCandList.reserve( ( numModesAvailable >> m_pcEncCfg->m_IntraEstDecBit ) + 2 );
234
235
1.73M
  for( unsigned mode = 0; mode < numModesAvailable; mode++ )
236
1.70M
  {
237
    // Skip checking extended Angular modes in the first round of SATD
238
1.70M
    if( mode > DC_IDX && ( mode & decMsk ) )
239
1.24M
    {
240
1.24M
      continue;
241
1.24M
    }
242
243
458k
    m_parentCandList.push_back( ModeInfo( false, false, 0, NOT_INTRA_SUBPARTITIONS, mode ) );
244
458k
  }
245
   
246
101k
  for( int decDst = 1 << m_pcEncCfg->m_IntraEstDecBit; decDst > 0; decDst >>= 1 )
247
76.4k
  {
248
687k
    for( unsigned idx = 0; idx < m_parentCandList.size(); idx++ )
249
611k
    {
250
611k
      int modeParent = m_parentCandList[idx].modeId;
251
252
611k
      int off = decDst & decMsk;
253
611k
      int inc = decDst << 1;
254
255
611k
#if 1 // INTRA_AS_IN_VTM
256
611k
      if( off != 0 && ( modeParent <= ( DC_IDX + 1 ) || modeParent >= ( NUM_LUMA_MODE - 1 ) ) )
257
99.3k
      {
258
99.3k
        continue;
259
99.3k
      }
260
261
512k
#endif
262
1.07M
      for( int mode = modeParent - off; mode < modeParent + off + 1; mode += inc )
263
565k
      {
264
565k
        if( satdChecked[mode] || mode < 0 || mode >= NUM_LUMA_MODE )
265
2.49k
        {
266
2.49k
          continue;
267
2.49k
        }
268
269
563k
        cu.intraDir[0] = mode;
270
271
563k
        initPredIntraParams( cu, cu.Y(), sps );
272
563k
        distParam.cur.buf = piPred.buf = m_SortedPelUnitBufs->getTestBuf().Y().buf;
273
563k
        predIntraAng( COMP_Y, piPred, cu );
274
275
        // Use the min between SAD and HAD as the cost criterion
276
        // SAD is scaled by 2 to align with the scaling of HAD
277
563k
        Distortion minSadHad = distParam.distFunc( distParam );
278
279
563k
        uint64_t fracModeBits = xFracModeBitsIntraLuma( cu, mpmLst );
280
281
        //restore ctx
282
563k
        m_CABACEstimator->getCtx() = SubCtx( CtxSet( Ctx::IntraLumaMpmFlag(), intra_ctx_size ), ctxStartIntraCtx );
283
284
563k
        double cost = ( double ) minSadHad + ( double ) fracModeBits * sqrtLambdaForFirstPass;
285
563k
        DTRACE( g_trace_ctx, D_INTRA_COST, "IntraHAD: %u, %llu, %f (%d)\n", minSadHad, fracModeBits, cost, mode );
286
287
563k
        int insertPos = -1;
288
563k
        updateCandList( ModeInfo( false, false, 0, NOT_INTRA_SUBPARTITIONS, mode ), cost, RdModeList, CandCostList, numModesForFullRD, &insertPos );
289
563k
        updateCandList( ModeInfo( false, false, 0, NOT_INTRA_SUBPARTITIONS, mode ), ( double ) minSadHad, HadModeList, CandHadList, numHadCand );
290
563k
        m_SortedPelUnitBufs->insert( insertPos, ( int ) RdModeList.size() );
291
292
563k
        satdChecked[mode] = true;
293
563k
      }
294
512k
    }
295
296
76.4k
    m_parentCandList.resize( RdModeList.size() );
297
76.4k
    std::copy( RdModeList.cbegin(), RdModeList.cend(), m_parentCandList.begin() );
298
76.4k
  }
299
300
25.4k
  const bool isFirstLineOfCtu = (((cu.block(COMP_Y).y)&((cu.cs->sps)->CTUSize - 1)) == 0);
301
25.4k
  if( m_pcEncCfg->m_MRL && ! isFirstLineOfCtu )
302
15.4k
  {
303
15.4k
    cu.multiRefIdx = 1;
304
15.4k
    unsigned  multiRefMPM [NUM_MOST_PROBABLE_MODES];
305
15.4k
    CU::getIntraMPMs(cu, multiRefMPM);
306
307
46.2k
    for (int mRefNum = 1; mRefNum < MRL_NUM_REF_LINES; mRefNum++)
308
30.8k
    {
309
30.8k
      int multiRefIdx = MULTI_REF_LINE_IDX[mRefNum];
310
311
30.8k
      cu.multiRefIdx = multiRefIdx;
312
30.8k
      initIntraPatternChType(cu, cu.Y(), true);
313
314
184k
      for (int x = 1; x < NUM_MOST_PROBABLE_MODES; x++)
315
154k
      {
316
154k
        cu.intraDir[0] = multiRefMPM[x];
317
154k
        initPredIntraParams(cu, cu.Y(), sps);
318
154k
        distParam.cur.buf = piPred.buf = m_SortedPelUnitBufs->getTestBuf().Y().buf;
319
154k
        predIntraAng(COMP_Y, piPred, cu);
320
321
        // Use the min between SAD and SATD as the cost criterion
322
        // SAD is scaled by 2 to align with the scaling of HAD
323
154k
        Distortion minSadHad = distParam.distFunc(distParam);
324
325
        // NB xFracModeBitsIntra will not affect the mode for chroma that may have already been pre-estimated.
326
154k
        uint64_t fracModeBits = xFracModeBitsIntraLuma( cu, mpmLst );
327
328
        //restore ctx
329
154k
        m_CABACEstimator->getCtx() = SubCtx(CtxSet(Ctx::IntraLumaMpmFlag(), intra_ctx_size), ctxStartIntraCtx);
330
331
154k
        double cost = (double) minSadHad + (double) fracModeBits * sqrtLambdaForFirstPass;
332
//        DTRACE(g_trace_ctx, D_INTRA_COST, "IntraMRL: %u, %llu, %f (%d)\n", minSadHad, fracModeBits, cost, cu.intraDir[0]);
333
334
154k
        int insertPos = -1;
335
154k
        updateCandList( ModeInfo( false, false, multiRefIdx, NOT_INTRA_SUBPARTITIONS, cu.intraDir[0] ), cost, RdModeList,  CandCostList, numModesForFullRD, &insertPos );
336
154k
        updateCandList( ModeInfo( false, false, multiRefIdx, NOT_INTRA_SUBPARTITIONS, cu.intraDir[0] ), (double)minSadHad, HadModeList, CandHadList,  numHadCand );
337
154k
        m_SortedPelUnitBufs->insert(insertPos, (int)RdModeList.size());
338
154k
      }
339
30.8k
    }
340
15.4k
    cu.multiRefIdx = 0;
341
15.4k
  }
342
343
25.4k
  if (testMip)
344
19.3k
  {
345
19.3k
    cu.mipFlag = true;
346
19.3k
    cu.multiRefIdx = 0;
347
348
19.3k
    double mipHadCost[MAX_NUM_MIP_MODE] = { MAX_DOUBLE };
349
350
19.3k
    initIntraPatternChType(cu, cu.Y());
351
19.3k
    initIntraMip( cu );
352
353
19.3k
    const int transpOff    = getNumModesMip( cu.Y() );
354
19.3k
    const int numModesFull = (transpOff << 1);
355
253k
    for( uint32_t uiModeFull = 0; uiModeFull < numModesFull; uiModeFull++ )
356
234k
    {
357
234k
      const bool     isTransposed = (uiModeFull >= transpOff ? true : false);
358
234k
      const uint32_t uiMode       = (isTransposed ? uiModeFull - transpOff : uiModeFull);
359
360
234k
      cu.mipTransposedFlag = isTransposed;
361
234k
      cu.intraDir[CH_L] = uiMode;
362
234k
      distParam.cur.buf = piPred.buf = m_SortedPelUnitBufs->getTestBuf().Y().buf;
363
234k
      predIntraMip(piPred, cu);
364
365
      // Use the min between SAD and HAD as the cost criterion
366
      // SAD is scaled by 2 to align with the scaling of HAD
367
234k
      Distortion minSadHad = distParam.distFunc(distParam);
368
369
234k
      uint64_t fracModeBits = xFracModeBitsIntraLuma( cu, mpmLst );
370
371
      //restore ctx
372
234k
      m_CABACEstimator->getCtx() = SubCtx(CtxSet(Ctx::IntraLumaMpmFlag(), intra_ctx_size), ctxStartIntraCtx);
373
374
234k
      double cost = double(minSadHad) + double(fracModeBits) * sqrtLambdaForFirstPass;
375
234k
      mipHadCost[uiModeFull] = cost;
376
234k
      DTRACE(g_trace_ctx, D_INTRA_COST, "IntraMIP: %u, %llu, %f (%d)\n", minSadHad, fracModeBits, cost, uiModeFull);
377
378
234k
      int insertPos = -1;
379
234k
      updateCandList( ModeInfo( true, isTransposed, 0, NOT_INTRA_SUBPARTITIONS, cu.intraDir[0] ), cost, RdModeList,  CandCostList, numModesForFullRD+1, &insertPos );
380
234k
      updateCandList( ModeInfo( true, isTransposed, 0, NOT_INTRA_SUBPARTITIONS, cu.intraDir[0] ), 0.8*(double)minSadHad, HadModeList, CandHadList,  numHadCand );
381
234k
      m_SortedPelUnitBufs->insert(insertPos, (int)RdModeList.size());
382
234k
    }
383
384
19.3k
    const double thresholdHadCost = 1.0 + 1.4 / sqrt((double)(cu.lwidth()*cu.lheight()));
385
19.3k
    xReduceHadCandList(RdModeList, CandCostList, *m_SortedPelUnitBufs, numModesForFullRD, thresholdHadCost, mipHadCost, cu, fastMip);
386
19.3k
  }
387
388
25.4k
  if( m_pcEncCfg->m_bFastUDIUseMPMEnabled )
389
25.4k
  {
390
25.4k
    const int numMPMs = NUM_MOST_PROBABLE_MODES;
391
25.4k
    unsigned  intraMpms[numMPMs];
392
393
25.4k
    cu.multiRefIdx = 0;
394
395
25.4k
    const int numCand = CU::getIntraMPMs( cu, intraMpms );
396
25.4k
    ModeInfo mostProbableMode(false, false, 0, NOT_INTRA_SUBPARTITIONS, 0);
397
398
51.7k
    for( int j = 0; j < numCand; j++ )
399
26.2k
    {
400
26.2k
      bool mostProbableModeIncluded = false;
401
26.2k
      mostProbableMode.modeId = intraMpms[j];
402
403
133k
      for( int i = 0; i < numModesForFullRD; i++ )
404
107k
      {
405
107k
        mostProbableModeIncluded |= ( mostProbableMode == RdModeList[i] );
406
107k
      }
407
26.2k
      if( !mostProbableModeIncluded )
408
189
      {
409
189
        numModesForFullRD++;
410
189
        RdModeList.push_back( mostProbableMode );
411
189
        CandCostList.push_back(0);
412
189
      }
413
26.2k
    }
414
25.4k
  }
415
25.4k
}
416
417
bool IntraSearch::estIntraPredLumaQT(CodingUnit &cu, Partitioner &partitioner, double bestCost)
418
25.4k
{
419
25.4k
  CodingStructure       &cs           = *cu.cs;
420
25.4k
  const int             width         = partitioner.currArea().lwidth();
421
25.4k
  const int             height        = partitioner.currArea().lheight();
422
423
  //===== loop over partitions =====
424
425
25.4k
  const TempCtx ctxStart           ( m_CtxCache, m_CABACEstimator->getCtx() );
426
427
  // variables for saving fast intra modes scan results across multiple LFNST passes
428
25.4k
  double costInterCU = xFindInterCUCost( cu );
429
430
25.4k
  bool validReturn = false;
431
432
  //===== determine set of modes to be tested (using prediction signal only) =====
433
25.4k
  int numModesAvailable = NUM_LUMA_MODE; // total number of Intra modes
434
25.4k
  static_vector<ModeInfo, FAST_UDI_MAX_RDMODE_NUM> RdModeList;
435
25.4k
  static_vector<ModeInfo, FAST_UDI_MAX_RDMODE_NUM> HadModeList;
436
25.4k
  static_vector<double, FAST_UDI_MAX_RDMODE_NUM> CandCostList;
437
25.4k
  static_vector<double, FAST_UDI_MAX_RDMODE_NUM> CandHadList;
438
439
25.4k
  int numModesForFullRD = g_aucIntraModeNumFast_UseMPM_2D[Log2(width) - MIN_CU_LOG2][Log2(height) - MIN_CU_LOG2];
440
25.4k
  if (m_pcEncCfg->m_numIntraModesFullRD > 0)
441
0
    numModesForFullRD=m_pcEncCfg->m_numIntraModesFullRD;
442
443
#if INTRA_FULL_SEARCH
444
  numModesForFullRD = numModesAvailable;
445
#endif
446
25.4k
  const SPS& sps = *cu.cs->sps;
447
25.4k
  const bool mipAllowed = sps.MIP && cu.lwidth() <= sps.getMaxTbSize() && cu.lheight() <= sps.getMaxTbSize() && ((cu.lfnstIdx == 0) || allowLfnstWithMip(cu.lumaSize()));
448
25.4k
  const int SizeThr     = 8 >> std::max( 0, m_pcEncCfg->m_useFastMIP - 1 );
449
25.4k
  const bool testMip    = mipAllowed && ( cu.lwidth() <= ( SizeThr * cu.lheight() ) && cu.lheight() <= ( SizeThr * cu.lwidth() ) ) && ( cu.lwidth() <= MIP_MAX_WIDTH && cu.lheight() <= MIP_MAX_HEIGHT );
450
25.4k
  bool testISP = sps.ISP && CU::canUseISP(width, height, cu.cs->sps->getMaxTbSize());
451
25.4k
  if (testISP)
452
25.4k
  {
453
25.4k
    int numTotalPartsHor = (int)width >> floorLog2(CU::getISPSplitDim(width, height, TU_1D_VERT_SPLIT));
454
25.4k
    int numTotalPartsVer = (int)height >> floorLog2(CU::getISPSplitDim(width, height, TU_1D_HORZ_SPLIT));
455
25.4k
    m_ispTestedModes[0].init(numTotalPartsHor, numTotalPartsVer, 0);
456
    // the total number of subpartitions is modified to take into account the cases where LFNST cannot be combined with
457
    // ISP due to size restrictions
458
25.4k
    numTotalPartsHor = sps.LFNST && CU::canUseLfnstWithISP(cu.Y(), HOR_INTRA_SUBPARTITIONS) ? numTotalPartsHor : 0;
459
25.4k
    numTotalPartsVer = sps.LFNST && CU::canUseLfnstWithISP(cu.Y(), VER_INTRA_SUBPARTITIONS) ? numTotalPartsVer : 0;
460
76.4k
    for (int j = 1; j < NUM_LFNST_NUM_PER_SET; j++)
461
50.9k
    {
462
50.9k
      m_ispTestedModes[j].init(numTotalPartsHor, numTotalPartsVer, 0);
463
50.9k
    }
464
25.4k
    testISP = m_ispTestedModes[0].numTotalParts[0];
465
25.4k
  }
466
0
  else
467
0
  {
468
0
    m_ispTestedModes[0].init(0, 0, 0);
469
0
  }
470
471
25.4k
  xEstimateLumaRdModeList(numModesForFullRD, RdModeList, HadModeList, CandCostList, CandHadList, cu, testMip);
472
473
25.4k
  CHECK( (size_t)numModesForFullRD != RdModeList.size(), "Inconsistent state!" );
474
475
  // after this point, don't use numModesForFullRD
476
25.4k
  if( m_pcEncCfg->m_usePbIntraFast && !cs.slice->isIntra() && RdModeList.size() < numModesAvailable )
477
0
  {
478
0
    double pbintraRatio = m_pcEncCfg->m_usePbIntraFast == 1 && ( cs.area.lwidth() >= 16 && cs.area.lheight() >= 16 ) ? 1.2 : PBINTRA_RATIO;
479
480
0
    int maxSize = -1;
481
0
    ModeInfo bestMipMode;
482
0
    int bestMipIdx = -1;
483
0
    for( int idx = 0; idx < RdModeList.size(); idx++ )
484
0
    {
485
0
      if( RdModeList[idx].mipFlg )
486
0
      {
487
0
        bestMipMode = RdModeList[idx];
488
0
        bestMipIdx = idx;
489
0
        break;
490
0
      }
491
0
    }
492
0
    const int numHadCand = 3;
493
0
    for (int k = numHadCand - 1; k >= 0; k--)
494
0
    {
495
0
      if (CandHadList.size() < (k + 1) || CandHadList[k] > cs.interHad * pbintraRatio) { maxSize = k; }
496
0
    }
497
0
    if (maxSize > 0)
498
0
    {
499
0
      RdModeList.resize(std::min<size_t>(RdModeList.size(), maxSize));
500
0
      if( bestMipIdx >= 0 )
501
0
      {
502
0
        if( RdModeList.size() <= bestMipIdx )
503
0
        {
504
0
          RdModeList.push_back(bestMipMode);
505
0
          m_SortedPelUnitBufs->swap( maxSize, bestMipIdx );
506
0
        }
507
0
      }
508
0
    }
509
0
    if (maxSize == 0)
510
0
    {
511
0
      cs.dist = MAX_DISTORTION;
512
0
      cs.interHad = 0;
513
0
      return false;
514
0
    }
515
0
  }
516
517
  //===== check modes (using r-d costs) =====
518
25.4k
  ModeInfo bestPUMode;
519
520
25.4k
  CodingStructure *csTemp = m_pTempCS;
521
25.4k
  CodingStructure *csBest = m_pBestCS;
522
523
25.4k
  csTemp->slice   = csBest->slice   = cs.slice;
524
25.4k
  csTemp->picture = csBest->picture = cs.picture;
525
25.4k
  csTemp->compactResize( cu );
526
25.4k
  csBest->compactResize( cu );
527
25.4k
  csTemp->initStructData();
528
25.4k
  csBest->initStructData();
529
530
25.4k
  int   bestLfnstIdx  = 0;
531
25.4k
  const bool useBDPCM = cs.picture->useBDPCM;
532
25.4k
  int   NumBDPCMCand  = (useBDPCM && sps.BDPCM && CU::bdpcmAllowed(cu, ComponentID(partitioner.chType))) ? 2 : 0;
533
25.4k
  int   bestbdpcmMode = 0;
534
25.4k
  int   bestISP       = 0;
535
25.4k
  int   bestMrl       = 0;
536
25.4k
  bool  bestMip       = 0;
537
25.4k
  int   EndMode       = (int)RdModeList.size();
538
25.4k
  bool  useISPlfnst   = testISP && sps.LFNST;
539
25.4k
  bool  noLFNST_ts    = false;
540
25.4k
  double bestCostIsp[2] = { MAX_DOUBLE, MAX_DOUBLE };
541
25.4k
  bool disableMTS = false;
542
25.4k
  bool disableLFNST = false;
543
25.4k
  bool disableDCT2test = false;
544
25.4k
  if (m_pcEncCfg->m_FastIntraTools)
545
25.4k
  {
546
25.4k
    int speedIntra = 0;
547
25.4k
    xSpeedUpIntra(bestCost, EndMode, speedIntra, cu);
548
25.4k
    disableMTS = (speedIntra >> 2 ) & 0x1;
549
25.4k
    disableLFNST = (speedIntra >> 1) & 0x1;
550
25.4k
    disableDCT2test = speedIntra>>3;
551
25.4k
    if (disableLFNST)
552
22.7k
    {
553
22.7k
      noLFNST_ts = true;
554
22.7k
      useISPlfnst = false;
555
22.7k
    }
556
25.4k
    if (speedIntra & 0x1)
557
22.7k
    {
558
22.7k
      testISP = false;
559
22.7k
    }
560
25.4k
  }
561
562
136k
  for (int mode_cur = 0; mode_cur < EndMode + NumBDPCMCand; mode_cur++)
563
111k
  {
564
111k
    int mode = mode_cur;
565
111k
    if (mode_cur >= EndMode)
566
7.45k
    {
567
7.45k
      mode = mode_cur - EndMode ? -1 : -2;
568
7.45k
      testISP = false;
569
7.45k
    }
570
    // set CU/PU to luma prediction mode
571
111k
    ModeInfo testMode;
572
111k
    int noISP = 0;
573
111k
    int endISP = testISP ? 2 : 0;
574
111k
    bool noLFNST = false || noLFNST_ts;
575
111k
    if (mode && useISPlfnst)
576
8.89k
    {
577
8.89k
      noLFNST |= (bestCostIsp[0] > (bestCostIsp[1] * 1.4));
578
8.89k
      if (mode > 2)
579
2.40k
      {
580
2.40k
        endISP = 0;
581
2.40k
        testISP = false;
582
2.40k
      }
583
8.89k
    }
584
111k
    if (testISP)
585
5.65k
    {
586
5.65k
      xSpeedUpISP(1, testISP, mode, noISP, endISP, cu, RdModeList, bestPUMode, bestISP, bestLfnstIdx);
587
5.65k
    }
588
111k
    int startISP = 0;
589
111k
    if (disableDCT2test && mode && bestISP)
590
0
    {
591
0
      startISP = endISP ? 1 : 0;
592
0
    }
593
231k
    for (int ispM = startISP; ispM <= endISP; ispM++)
594
119k
    {
595
119k
      if (ispM && (ispM == noISP))
596
45
      {
597
45
        continue;
598
45
      }
599
600
119k
      if (mode < 0)
601
7.45k
      {
602
7.45k
        cu.bdpcmM[CH_L] = -mode;
603
7.45k
        testMode = ModeInfo(false, false, 0, NOT_INTRA_SUBPARTITIONS, cu.bdpcmM[CH_L] == 2 ? VER_IDX : HOR_IDX);
604
7.45k
      }
605
112k
      else
606
112k
      {
607
112k
        testMode = RdModeList[mode];
608
112k
        cu.bdpcmM[CH_L] = 0;
609
112k
      }
610
611
119k
      cu.ispMode = ispM;
612
119k
      cu.mipFlag = testMode.mipFlg;
613
119k
      cu.mipTransposedFlag = testMode.mipTrFlg;
614
119k
      cu.multiRefIdx = testMode.mRefId;
615
119k
      cu.intraDir[CH_L] = testMode.modeId;
616
119k
      if (cu.ispMode && xSpeedUpISP(0, testISP, mode, noISP, endISP, cu, RdModeList, bestPUMode, bestISP, 0) )
617
2.96k
      {
618
2.96k
        continue;
619
2.96k
      }
620
116k
      if (m_pcEncCfg->m_FastIntraTools && (cu.ispMode || sps.LFNST || sps.MTS))
621
116k
      {
622
116k
        m_ispTestedModes[0].intraWasTested = true;
623
116k
      }
624
116k
      CHECK(cu.mipFlag && cu.multiRefIdx, "Error: combination of MIP and MRL not supported");
625
116k
      CHECK(cu.multiRefIdx && (cu.intraDir[0] == PLANAR_IDX), "Error: combination of MRL and Planar mode not supported");
626
116k
      CHECK(cu.ispMode && cu.mipFlag, "Error: combination of ISP and MIP not supported");
627
116k
      CHECK(cu.ispMode && cu.multiRefIdx, "Error: combination of ISP and MRL not supported");
628
629
      // determine residual for partition
630
116k
      cs.initSubStructure(*csTemp, partitioner.chType, cs.area, true);
631
116k
      int doISP = (((cu.ispMode == 0) && noLFNST) || (useISPlfnst && mode && cu.ispMode && (bestLfnstIdx == 0)) || disableLFNST) ? -mode : mode;
632
116k
      xIntraCodingLumaQT(*csTemp, partitioner, m_SortedPelUnitBufs->getBufFromSortedList(mode), bestCost, doISP, disableMTS);
633
634
116k
      DTRACE(g_trace_ctx, D_INTRA_COST, "IntraCost T [x=%d,y=%d,w=%d,h=%d] %f (%d,%d,%d,%d,%d,%d) \n", cu.blocks[0].x,
635
116k
        cu.blocks[0].y, width, height, csTemp->cost, testMode.modeId, testMode.ispMod,
636
116k
        cu.multiRefIdx, cu.mipFlag, cu.lfnstIdx, cu.mtsFlag);
637
638
116k
      if (cu.ispMode && !csTemp->cus[0]->firstTU->cbf[COMP_Y])
639
1.83k
      {
640
1.83k
        csTemp->cost = MAX_DOUBLE;
641
1.83k
        csTemp->costDbOffset = 0;
642
1.83k
      }
643
116k
      if (useISPlfnst)
644
16.9k
      {
645
16.9k
        int n = (cu.ispMode == 0) ? 0 : 1;
646
16.9k
        bestCostIsp[n] = csTemp->cost < bestCostIsp[n] ? csTemp->cost : bestCostIsp[n];
647
16.9k
      }
648
649
      // check r-d cost
650
116k
      if (csTemp->cost < csBest->cost)
651
32.3k
      {
652
32.3k
        validReturn   = true;
653
32.3k
        std::swap(csTemp, csBest);
654
32.3k
        bestPUMode    = testMode;
655
32.3k
        bestLfnstIdx  = csBest->cus[0]->lfnstIdx;
656
32.3k
        bestISP       = csBest->cus[0]->ispMode;
657
32.3k
        bestMip       = csBest->cus[0]->mipFlag;
658
32.3k
        bestMrl       = csBest->cus[0]->multiRefIdx;
659
32.3k
        bestbdpcmMode = cu.bdpcmM[CH_L];
660
32.3k
        m_ispTestedModes[bestLfnstIdx].bestSplitSoFar = ISPType(bestISP);
661
32.3k
        if (csBest->cost < bestCost)
662
32.3k
        {
663
32.3k
          bestCost = csBest->cost;
664
32.3k
        }
665
32.3k
        if ((csBest->getTU(partitioner.chType)->mtsIdx[COMP_Y] == MTS_SKIP) && ( floorLog2(csBest->getTU(partitioner.chType)->blocks[COMP_Y].area()) >= 6 ))
666
4.45k
        {
667
4.45k
          noLFNST_ts = 1;
668
4.45k
        }
669
32.3k
      }
670
671
      // reset context models
672
116k
      m_CABACEstimator->getCtx() = ctxStart;
673
674
116k
      csTemp->releaseIntermediateData();
675
676
116k
      if (m_pcEncCfg->m_fastLocalDualTreeMode && CU::isConsIntra(cu) && !cu.slice->isIntra() && csBest->cost != MAX_DOUBLE && costInterCU != COST_UNKNOWN && mode >= 0)
677
0
      {
678
0
        if( (m_pcEncCfg->m_fastLocalDualTreeMode == 2) || (csBest->cost > costInterCU * 1.5))
679
0
        {
680
          //Note: only try one intra mode, which is especially useful to reduce EncT for LDB case (around 4%)
681
0
          EndMode = 0;
682
0
          break;
683
0
        }
684
0
      }
685
116k
    }
686
111k
  } // Mode loop
687
688
25.4k
  if (m_pcEncCfg->m_FastIntraTools && (sps.ISP|| sps.LFNST || sps.MTS))
689
25.4k
  {
690
25.4k
    int bestMode = csBest->getTU(partitioner.chType)->mtsIdx[COMP_Y] ? 4 : 0;
691
25.4k
    bestMode |= bestLfnstIdx ? 2 : 0;
692
25.4k
    bestMode |= bestISP ? 1 : 0;
693
25.4k
    m_ispTestedModes[0].bestIntraMode = bestMode;
694
25.4k
  }
695
25.4k
  cu.ispMode = bestISP;
696
25.4k
  if( validReturn )
697
25.4k
  {
698
25.4k
    cs.useSubStructure( *csBest, partitioner.chType, TREE_D, cu.singleChan( CH_L ), true );
699
25.4k
    const ReshapeData& reshapeData = cs.picture->reshapeData;
700
25.4k
    if (cs.picHeader->lmcsEnabled && reshapeData.getCTUFlag())
701
0
    {
702
0
      cs.getRspRecoBuf().copyFrom(csBest->getRspRecoBuf());
703
0
    }
704
705
    //=== update PU data ====
706
25.4k
    cu.lfnstIdx           = bestLfnstIdx;
707
25.4k
    cu.mipTransposedFlag  = bestPUMode.mipTrFlg;
708
25.4k
    cu.intraDir[CH_L]     = bestPUMode.modeId;
709
25.4k
    cu.bdpcmM[CH_L]       = bestbdpcmMode;
710
25.4k
    cu.mipFlag            = bestMip;
711
25.4k
    cu.multiRefIdx        = bestMrl;
712
25.4k
  }
713
0
  else
714
0
  {
715
0
    THROW("fix this");
716
0
  }
717
718
25.4k
  csBest->releaseIntermediateData();
719
720
25.4k
  return validReturn;
721
25.4k
}
722
723
void IntraSearch::estIntraPredChromaQT( CodingUnit& cu, Partitioner& partitioner, const double maxCostAllowed )
724
57.4k
{
725
57.4k
  PROFILER_SCOPE_AND_STAGE_EXT( 0, _TPROF, P_INTRA_CHROMA, cu.cs, CH_C );
726
57.4k
  const TempCtx ctxStart( m_CtxCache, m_CABACEstimator->getCtx() );
727
57.4k
  CodingStructure &cs   = *cu.cs;
728
57.4k
  bool lumaUsesISP      = !CU::isSepTree(cu) && cu.ispMode;
729
57.4k
  PartSplit ispType     = lumaUsesISP ? CU::getISPType(cu, COMP_Y) : TU_NO_ISP;
730
57.4k
  double bestCostSoFar  = maxCostAllowed;
731
57.4k
  const uint32_t numberValidComponents = getNumberValidComponents( cu.chromaFormat );
732
57.4k
  const bool useBDPCM   = cs.picture->useBDPCM;
733
734
57.4k
  uint32_t   uiBestMode = 0;
735
57.4k
  Distortion uiBestDist = 0;
736
57.4k
  double     dBestCost  = MAX_DOUBLE;
737
738
  //----- init mode list ----
739
57.4k
  {
740
57.4k
    uint32_t  uiMinMode = 0;
741
57.4k
    uint32_t  uiMaxMode = NUM_CHROMA_MODE;
742
743
57.4k
    const int reducedModeNumber = uiMaxMode >> (m_pcEncCfg->m_reduceIntraChromaModesFullRD ? 1 : 2);
744
    //----- check chroma modes -----
745
57.4k
    uint32_t chromaCandModes[ NUM_CHROMA_MODE ];
746
57.4k
    CU::getIntraChromaCandModes( cu, chromaCandModes );
747
748
    // create a temporary CS
749
57.4k
    CodingStructure &saveCS = *m_pSaveCS[0];
750
57.4k
    saveCS.pcv      = cs.pcv;
751
57.4k
    saveCS.picture  = cs.picture;
752
57.4k
    saveCS.area.repositionTo( cs.area );
753
57.4k
    saveCS.clearTUs();
754
755
57.4k
    if( !CU::isSepTree(cu) && cu.ispMode )
756
0
    {
757
0
      saveCS.clearCUs();
758
0
    }
759
760
57.4k
    if( CU::isSepTree(cu) )
761
57.4k
    {
762
57.4k
      if( partitioner.canSplit( TU_MAX_TR_SPLIT, cs ) )
763
0
      {
764
0
        partitioner.splitCurrArea( TU_MAX_TR_SPLIT, cs );
765
766
0
        do
767
0
        {
768
0
          cs.addTU( CS::getArea( cs, partitioner.currArea(), partitioner.chType, partitioner.treeType ), partitioner.chType, &cu ).depth = partitioner.currTrDepth;
769
0
        } while( partitioner.nextPart( cs ) );
770
771
0
        partitioner.exitCurrSplit();
772
0
      }
773
57.4k
      else
774
57.4k
        cs.addTU( CS::getArea( cs, partitioner.currArea(), partitioner.chType, partitioner.treeType ), partitioner.chType, &cu );
775
57.4k
    }
776
777
    // create a store for the TUs
778
57.4k
    std::vector<TransformUnit*> orgTUs;
779
57.4k
    for( const auto &ptu : cs.tus )
780
57.4k
    {
781
      // for split TUs in HEVC, add the TUs without Chroma parts for correct setting of Cbfs
782
57.4k
      if (lumaUsesISP || cu.contains(*ptu, CH_C))
783
57.4k
      {
784
57.4k
        saveCS.addTU( *ptu, partitioner.chType, nullptr );
785
57.4k
        orgTUs.push_back( ptu );
786
57.4k
      }
787
57.4k
    }
788
789
    // SATD pre-selecting.
790
57.4k
    int     satdModeList  [NUM_CHROMA_MODE] = { 0 };
791
57.4k
    int64_t satdSortedCost[NUM_CHROMA_MODE] = { 0 };
792
57.4k
    bool    modeDisable[NUM_INTRA_MODE + 1] = { false }; // use intra mode idx to check whether enable
793
794
57.4k
    CodingStructure& cs = *(cu.cs);
795
57.4k
    CompArea areaCb = cu.Cb();
796
57.4k
    CompArea areaCr = cu.Cr();
797
57.4k
    CPelBuf orgCb  = cs.getOrgBuf (COMP_Cb);
798
57.4k
    PelBuf predCb  = cs.getPredBuf(COMP_Cb);
799
57.4k
    CPelBuf orgCr  = cs.getOrgBuf (COMP_Cr);
800
57.4k
    PelBuf predCr  = cs.getPredBuf(COMP_Cr);
801
802
57.4k
    DistParam distParamSadCb  = m_pcRdCost->setDistParam( orgCb, predCb, cu.cs->sps->bitDepths[ CH_C ], DF_SAD);
803
57.4k
    DistParam distParamSatdCb = m_pcRdCost->setDistParam( orgCb, predCb, cu.cs->sps->bitDepths[ CH_C ], DF_HAD);
804
57.4k
    DistParam distParamSadCr  = m_pcRdCost->setDistParam( orgCr, predCr, cu.cs->sps->bitDepths[ CH_C ], DF_SAD);
805
57.4k
    DistParam distParamSatdCr = m_pcRdCost->setDistParam( orgCr, predCr, cu.cs->sps->bitDepths[ CH_C ], DF_HAD);
806
807
57.4k
    cu.intraDir[1] = MDLM_L_IDX; // temporary assigned, just to indicate this is a MDLM mode. for luma down-sampling operation.
808
809
57.4k
    initIntraPatternChType(cu, cu.Cb());
810
57.4k
    initIntraPatternChType(cu, cu.Cr());
811
57.4k
    loadLMLumaRecPels(cu, cu.Cb());
812
813
517k
    for (int idx = uiMinMode; idx < uiMaxMode; idx++)
814
459k
    {
815
459k
      int mode = chromaCandModes[idx];
816
459k
      satdModeList[idx] = mode;
817
459k
      if (CU::isLMCMode(mode) && ( !CU::isLMCModeEnabled(cu, mode) || cu.slice->lmChromaCheckDisable ) )
818
49.7k
      {
819
49.7k
        continue;
820
49.7k
      }
821
410k
      if ((mode == LM_CHROMA_IDX) || (mode == PLANAR_IDX) || (mode == DM_CHROMA_IDX)) // only pre-check regular modes and MDLM modes, not including DM ,Planar, and LM
822
100k
      {
823
100k
        continue;
824
100k
      }
825
826
309k
      cu.intraDir[1]    = mode; // temporary assigned, for SATD checking.
827
828
309k
      const bool isLMCMode = CU::isLMCMode(mode);
829
309k
      if( isLMCMode )
830
81.8k
      {
831
81.8k
        predIntraChromaLM(COMP_Cb, predCb, cu, areaCb, mode);
832
81.8k
      }
833
228k
      else
834
228k
      {
835
228k
        initPredIntraParams(cu, cu.Cb(), *cs.sps);
836
228k
        predIntraAng(COMP_Cb, predCb, cu);
837
228k
      }
838
309k
      int64_t sadCb = distParamSadCb.distFunc(distParamSadCb) * 2;
839
309k
      int64_t satdCb = distParamSatdCb.distFunc(distParamSatdCb);
840
309k
      int64_t sad = std::min(sadCb, satdCb);
841
842
309k
      if( isLMCMode )
843
81.8k
      {
844
81.8k
        predIntraChromaLM(COMP_Cr, predCr, cu, areaCr, mode);
845
81.8k
      }
846
228k
      else
847
228k
      {
848
228k
        initPredIntraParams(cu, cu.Cr(), *cs.sps);
849
228k
        predIntraAng(COMP_Cr, predCr, cu);
850
228k
      }
851
309k
      int64_t sadCr = distParamSadCr.distFunc(distParamSadCr) * 2;
852
309k
      int64_t satdCr = distParamSatdCr.distFunc(distParamSatdCr);
853
309k
      sad += std::min(sadCr, satdCr);
854
309k
      satdSortedCost[idx] = sad;
855
309k
    }
856
857
    // sort the mode based on the cost from small to large.
858
517k
    for (int i = uiMinMode; i <= uiMaxMode - 1; i++)
859
459k
    {
860
2.06M
      for (int j = i + 1; j <= uiMaxMode - 1; j++)
861
1.60M
      {
862
1.60M
        if (satdSortedCost[j] < satdSortedCost[i])
863
99.5k
        {
864
99.5k
          std::swap( satdModeList[i],   satdModeList[j]);
865
99.5k
          std::swap( satdSortedCost[i], satdSortedCost[j]);
866
99.5k
        }
867
1.60M
      }
868
459k
    }
869
870
287k
    for (int i = 0; i < reducedModeNumber; i++)
871
229k
    {
872
229k
      modeDisable[satdModeList[uiMaxMode - 1 - i]] = true; // disable the last reducedModeNumber modes
873
229k
    }
874
875
57.4k
    int bestLfnstIdx = 0;
876
    // save the dist
877
57.4k
    Distortion baseDist = cs.dist;
878
57.4k
    int32_t bestbdpcmMode = 0;
879
57.4k
    uint32_t numbdpcmModes = ( useBDPCM && CU::bdpcmAllowed(cu, COMP_Cb)
880
38.4k
        && ((partitioner.chType == CH_C) || (cu.ispMode == 0 && cu.lfnstIdx == 0 && cu.firstTU->mtsIdx[COMP_Y] == MTS_SKIP))) ? 2 : 0;
881
594k
    for (int mode_cur = uiMinMode; mode_cur < (int)(uiMaxMode + numbdpcmModes); mode_cur++)
882
536k
    {
883
536k
      int mode = mode_cur;
884
536k
      if (mode_cur >= uiMaxMode)
885
76.8k
      {
886
76.8k
        mode = mode_cur > uiMaxMode ? -1 : -2; //set bdpcm mode
887
76.8k
        if ((mode == -1) && (saveCS.tus[0]->mtsIdx[COMP_Cb] != MTS_SKIP) && (saveCS.tus[0]->mtsIdx[COMP_Cr] != MTS_SKIP))
888
38.4k
        {
889
38.4k
          continue;
890
38.4k
        }
891
76.8k
      }
892
498k
      int chromaIntraMode;
893
498k
      if (mode < 0)
894
38.4k
      {
895
38.4k
        cu.bdpcmM[CH_C] = -mode;
896
38.4k
        chromaIntraMode = cu.bdpcmM[CH_C] == 2 ? chromaCandModes[1] : chromaCandModes[2];
897
38.4k
      }
898
459k
      else
899
459k
      {
900
459k
        cu.bdpcmM[CH_C] = 0;
901
459k
        chromaIntraMode = chromaCandModes[mode];
902
459k
        if (CU::isLMCMode(chromaIntraMode) && ( !CU::isLMCModeEnabled(cu, chromaIntraMode) || cu.slice->lmChromaCheckDisable ) )
903
49.7k
        {
904
49.7k
          continue;
905
49.7k
        }
906
410k
        if (modeDisable[chromaIntraMode] && CU::isLMCModeEnabled(cu, chromaIntraMode)) // when CCLM is disable, then MDLM is disable. not use satd checking
907
163k
        {
908
163k
          continue;
909
163k
        }
910
410k
      }
911
284k
      cs.dist = baseDist;
912
      //----- restore context models -----
913
284k
      m_CABACEstimator->getCtx() = ctxStart;
914
915
      //----- chroma coding -----
916
284k
      cu.intraDir[1] = chromaIntraMode;
917
284k
      m_ispTestedModes[0].IspType = ispType;
918
284k
      m_ispTestedModes[0].subTuCounter = -1;
919
284k
      xIntraChromaCodingQT( cs, partitioner );
920
284k
      if (lumaUsesISP && cs.dist == MAX_UINT)
921
0
      {
922
0
        continue;
923
0
      }
924
925
284k
      if (cs.sps->transformSkip)
926
284k
      {
927
284k
        m_CABACEstimator->getCtx() = ctxStart;
928
284k
      }
929
284k
      m_ispTestedModes[0].IspType = ispType;
930
284k
      m_ispTestedModes[0].subTuCounter = -1;
931
284k
      uint64_t fracBits   = xGetIntraFracBitsQT( cs, partitioner, false );
932
284k
      Distortion uiDist = cs.dist;
933
284k
      double    dCost   = m_pcRdCost->calcRdCost( fracBits, uiDist - baseDist );
934
935
      //----- compare -----
936
284k
      if( dCost < dBestCost )
937
103k
      {
938
103k
        if (lumaUsesISP && (dCost < bestCostSoFar))
939
0
        {
940
0
          bestCostSoFar = dCost;
941
0
        }
942
309k
        for( uint32_t i = getFirstComponentOfChannel( CH_C ); i < numberValidComponents; i++ )
943
206k
        {
944
206k
          const CompArea& area = cu.blocks[i];
945
206k
          saveCS.getRecoBuf     ( area ).copyFrom( cs.getRecoBuf   ( area ) );
946
206k
          cs.picture->getRecoBuf( area ).copyFrom( cs.getRecoBuf   ( area ) );
947
412k
          for( uint32_t j = 0; j < saveCS.tus.size(); j++ )
948
206k
          {
949
206k
            saveCS.tus[j]->copyComponentFrom( *orgTUs[j], area.compID );
950
206k
          }
951
206k
        }
952
103k
        dBestCost    = dCost;
953
103k
        uiBestDist   = uiDist;
954
103k
        uiBestMode   = chromaIntraMode;
955
103k
        bestLfnstIdx = cu.lfnstIdx;
956
103k
        bestbdpcmMode = cu.bdpcmM[CH_C];
957
958
103k
      }
959
284k
    }
960
57.4k
    cu.lfnstIdx = bestLfnstIdx;
961
57.4k
    cu.bdpcmM[CH_C]= bestbdpcmMode;
962
963
172k
    for( uint32_t i = getFirstComponentOfChannel( CH_C ); i < numberValidComponents; i++ )
964
114k
    {
965
114k
      const CompArea& area = cu.blocks[i];
966
967
114k
      cs.getRecoBuf         ( area ).copyFrom( saveCS.getRecoBuf( area ) );
968
114k
      cs.picture->getRecoBuf( area ).copyFrom( cs.getRecoBuf    ( area ) );
969
970
229k
      for( uint32_t j = 0; j < saveCS.tus.size(); j++ )
971
114k
      {
972
114k
        orgTUs[ j ]->copyComponentFrom( *saveCS.tus[ j ], area.compID );
973
114k
      }
974
114k
    }
975
57.4k
  }
976
57.4k
  cu.intraDir[1] = uiBestMode;
977
57.4k
  cs.dist        = uiBestDist;
978
979
  //----- restore context models -----
980
57.4k
  m_CABACEstimator->getCtx() = ctxStart;
981
57.4k
  if (lumaUsesISP && bestCostSoFar >= maxCostAllowed)
982
0
  {
983
0
    cu.ispMode = 0;
984
0
  }
985
57.4k
}
986
987
void IntraSearch::saveCuAreaCostInSCIPU( Area area, double cost )
988
0
{
989
0
  if( m_numCuInSCIPU < NUM_INTER_CU_INFO_SAVE )
990
0
  {
991
0
    m_cuAreaInSCIPU[m_numCuInSCIPU] = area;
992
0
    m_cuCostInSCIPU[m_numCuInSCIPU] = cost;
993
0
    m_numCuInSCIPU++;
994
0
  }
995
0
}
996
997
void IntraSearch::initCuAreaCostInSCIPU()
998
0
{
999
0
  for( int i = 0; i < NUM_INTER_CU_INFO_SAVE; i++ )
1000
0
  {
1001
0
    m_cuAreaInSCIPU[i] = Area();
1002
0
    m_cuCostInSCIPU[i] = 0;
1003
0
  }
1004
0
  m_numCuInSCIPU = 0;
1005
0
}
1006
// -------------------------------------------------------------------------------------------------------------------
1007
// Intra search
1008
// -------------------------------------------------------------------------------------------------------------------
1009
1010
void IntraSearch::xEncIntraHeader( CodingStructure &cs, Partitioner &partitioner, const bool luma )
1011
473k
{
1012
473k
  CodingUnit &cu = *cs.getCU( partitioner.chType, partitioner.treeType );
1013
1014
473k
  if (luma)
1015
188k
  {
1016
188k
    bool isFirst = cu.ispMode ? m_ispTestedModes[0].subTuCounter == 0 : partitioner.currArea().lumaPos() == cs.area.lumaPos();
1017
1018
    // CU header
1019
188k
    if( isFirst )
1020
184k
    {
1021
184k
      if ((!cs.slice->isIntra() || cs.slice->sps->IBC || cs.slice->sps->PLT) && cu.Y().valid())
1022
184k
      {
1023
184k
        m_CABACEstimator->pred_mode   ( cu );
1024
184k
      }
1025
184k
      m_CABACEstimator->bdpcm_mode  ( cu, ComponentID(partitioner.chType) );
1026
184k
    }
1027
1028
    // luma prediction mode
1029
188k
    if (isFirst)
1030
184k
    {
1031
184k
      if ( !cu.Y().valid())
1032
0
      {
1033
0
        m_CABACEstimator->pred_mode( cu );
1034
0
      }
1035
184k
      m_CABACEstimator->intra_luma_pred_mode( cu );
1036
184k
    }
1037
188k
  }
1038
284k
  else //  if (chroma)
1039
284k
  {
1040
284k
    bool isFirst = partitioner.currArea().Cb().valid() && partitioner.currArea().chromaPos() == cs.area.chromaPos();
1041
1042
284k
    if( isFirst )
1043
284k
    {
1044
284k
      m_CABACEstimator->bdpcm_mode(cu, ComponentID(CH_C));
1045
284k
      m_CABACEstimator->intra_chroma_pred_mode(  cu );
1046
284k
    }
1047
284k
  }
1048
473k
}
1049
1050
void IntraSearch::xEncSubdivCbfQT( CodingStructure &cs, Partitioner &partitioner, const bool luma )
1051
473k
{
1052
473k
  const UnitArea& currArea = partitioner.currArea();
1053
473k
  int subTuCounter = m_ispTestedModes[0].subTuCounter;
1054
473k
  TransformUnit  &currTU   = *cs.getTU(currArea.blocks[partitioner.chType], partitioner.chType, subTuCounter);
1055
473k
  CodingUnit     &currCU   = *currTU.cu;
1056
473k
  const uint32_t currDepth = partitioner.currTrDepth;
1057
473k
  const bool  subdiv = currTU.depth > currDepth;
1058
473k
  ComponentID compID = partitioner.chType == CH_L ? COMP_Y : COMP_Cb;
1059
1060
473k
  if (!luma)
1061
284k
  {
1062
284k
    const bool chromaCbfISP = currArea.blocks[COMP_Cb].valid() && currCU.ispMode && !subdiv;
1063
284k
    if (!currCU.ispMode || chromaCbfISP)
1064
284k
    {
1065
284k
      const uint32_t numberValidComponents = getNumberValidComponents(currArea.chromaFormat);
1066
284k
      const uint32_t cbfDepth = (chromaCbfISP ? currDepth - 1 : currDepth);
1067
1068
854k
      for (uint32_t ch = COMP_Cb; ch < numberValidComponents; ch++)
1069
569k
      {
1070
569k
        const ComponentID compID = ComponentID(ch);
1071
569k
        if (currDepth == 0 || TU::getCbfAtDepth(currTU, compID, currDepth - 1) || chromaCbfISP)
1072
569k
        {
1073
569k
          const bool prevCbf = (compID == COMP_Cr ? TU::getCbfAtDepth(currTU, COMP_Cb, currDepth) : false);
1074
569k
          m_CABACEstimator->cbf_comp(currCU, TU::getCbfAtDepth(currTU, compID, currDepth), currArea.blocks[compID], cbfDepth, prevCbf);
1075
569k
        }
1076
569k
      }
1077
284k
    }
1078
284k
  }
1079
1080
473k
  if (subdiv)
1081
0
  {
1082
0
    if (partitioner.canSplit(TU_MAX_TR_SPLIT, cs))
1083
0
    {
1084
0
      partitioner.splitCurrArea(TU_MAX_TR_SPLIT, cs);
1085
0
    }
1086
0
    else if (currCU.ispMode && isLuma(compID))
1087
0
    {
1088
0
      partitioner.splitCurrArea(m_ispTestedModes[0].IspType, cs);
1089
0
    }
1090
0
    else
1091
0
      THROW("Cannot perform an implicit split!");
1092
1093
0
    do
1094
0
    {
1095
0
      xEncSubdivCbfQT(cs, partitioner, luma);   //?
1096
0
      subTuCounter += subTuCounter != -1 ? 1 : 0;
1097
0
    } while (partitioner.nextPart(cs));
1098
1099
0
    partitioner.exitCurrSplit();
1100
0
  }
1101
473k
  else
1102
473k
  {
1103
    //===== Cbfs =====
1104
473k
    if (luma)
1105
188k
    {
1106
188k
      bool previousCbf = false;
1107
188k
      bool lastCbfIsInferred = false;
1108
188k
      if (m_ispTestedModes[0].IspType != TU_NO_ISP)
1109
14.4k
      {
1110
14.4k
        bool     rootCbfSoFar = false;
1111
14.4k
        uint32_t nTus = currCU.ispMode == HOR_INTRA_SUBPARTITIONS ? currCU.lheight() >> floorLog2(currTU.lheight())
1112
14.4k
          : currCU.lwidth() >> floorLog2(currTU.lwidth());
1113
14.4k
        if (subTuCounter == nTus - 1)
1114
1.35k
        {
1115
1.35k
          TransformUnit* tuPointer = currCU.firstTU;
1116
5.41k
          for (int tuIdx = 0; tuIdx < nTus - 1; tuIdx++)
1117
4.05k
          {
1118
4.05k
            rootCbfSoFar |= TU::getCbfAtDepth(*tuPointer, COMP_Y, currDepth);
1119
4.05k
            tuPointer = tuPointer->next;
1120
4.05k
          }
1121
1.35k
          if (!rootCbfSoFar)
1122
0
          {
1123
0
            lastCbfIsInferred = true;
1124
0
          }
1125
1.35k
        }
1126
14.4k
        if (!lastCbfIsInferred)
1127
14.4k
        {
1128
14.4k
          previousCbf = TU::getPrevTuCbfAtDepth(currTU, COMP_Y, partitioner.currTrDepth);
1129
14.4k
        }
1130
14.4k
      }
1131
188k
      if (!lastCbfIsInferred)
1132
188k
      {
1133
188k
        m_CABACEstimator->cbf_comp(currCU, TU::getCbfAtDepth(currTU, COMP_Y, currDepth), currTU.Y(), currTU.depth, previousCbf, currCU.ispMode);
1134
188k
      }
1135
188k
    }
1136
473k
  }
1137
473k
}
1138
void IntraSearch::xEncCoeffQT(CodingStructure& cs, Partitioner& partitioner, const ComponentID compID, CUCtx* cuCtx, const int subTuIdx, const PartSplit ispType)
1139
758k
{
1140
758k
  const UnitArea& currArea  = partitioner.currArea();
1141
1142
758k
  int subTuCounter          = m_ispTestedModes[0].subTuCounter;
1143
758k
  TransformUnit& currTU     = *cs.getTU(currArea.blocks[partitioner.chType], partitioner.chType, subTuCounter);
1144
758k
  uint32_t   currDepth      = partitioner.currTrDepth;
1145
758k
  const bool subdiv         = currTU.depth > currDepth;
1146
1147
758k
  if (subdiv)
1148
0
  {
1149
0
    if (partitioner.canSplit(TU_MAX_TR_SPLIT, cs))
1150
0
    {
1151
0
      partitioner.splitCurrArea(TU_MAX_TR_SPLIT, cs);
1152
0
    }
1153
0
    else if (currTU.cu->ispMode)
1154
0
    {
1155
0
      partitioner.splitCurrArea(m_ispTestedModes[0].IspType, cs);
1156
0
    }
1157
0
    else
1158
0
      THROW("Implicit TU split not available!");
1159
1160
0
    do
1161
0
    {
1162
0
      xEncCoeffQT(cs, partitioner, compID, cuCtx, subTuCounter, m_ispTestedModes[0].IspType);
1163
0
      subTuCounter += subTuCounter != -1 ? 1 : 0;
1164
0
    } while( partitioner.nextPart( cs ) );
1165
1166
0
    partitioner.exitCurrSplit();
1167
0
  }
1168
758k
  else
1169
1170
758k
  if( currArea.blocks[compID].valid() )
1171
758k
  {
1172
758k
    if( compID == COMP_Cr )
1173
284k
    {
1174
284k
      const int cbfMask = ( TU::getCbf( currTU, COMP_Cb ) ? 2 : 0 ) + ( TU::getCbf( currTU, COMP_Cr ) ? 1 : 0 );
1175
284k
      m_CABACEstimator->joint_cb_cr( currTU, cbfMask );
1176
284k
    }
1177
758k
    if( TU::getCbf( currTU, compID ) )
1178
230k
    {
1179
230k
      if( isLuma(compID) )
1180
25.3k
      {
1181
25.3k
        m_CABACEstimator->residual_coding( currTU, compID, cuCtx );
1182
25.3k
        m_CABACEstimator->mts_idx( *currTU.cu, cuCtx );
1183
25.3k
      }
1184
204k
      else
1185
204k
        m_CABACEstimator->residual_coding( currTU, compID );
1186
230k
    }
1187
758k
  }
1188
758k
}
1189
1190
uint64_t IntraSearch::xGetIntraFracBitsQT( CodingStructure &cs, Partitioner &partitioner, const bool luma, CUCtx *cuCtx )
1191
473k
{
1192
473k
  m_CABACEstimator->resetBits();
1193
1194
473k
  xEncIntraHeader( cs, partitioner, luma );
1195
473k
  xEncSubdivCbfQT( cs, partitioner, luma );
1196
1197
473k
  if( luma )
1198
188k
  {
1199
188k
    xEncCoeffQT( cs, partitioner, COMP_Y, cuCtx );
1200
1201
188k
    CodingUnit &cu = *cs.cus[0];
1202
188k
    if (cuCtx /*&& CU::isSepTree(cu)*/
1203
119k
      && (!cu.ispMode || (cu.lfnstIdx && m_ispTestedModes[0].subTuCounter == 0)
1204
9.13k
        || (!cu.lfnstIdx
1205
7.76k
          && m_ispTestedModes[0].subTuCounter == m_ispTestedModes[cu.lfnstIdx].numTotalParts[cu.ispMode - 1] - 1)))
1206
111k
    {
1207
111k
      m_CABACEstimator->residual_lfnst_mode( cu, *cuCtx );
1208
111k
    }
1209
188k
  }
1210
284k
  else
1211
284k
  {
1212
284k
    xEncCoeffQT( cs, partitioner, COMP_Cb );
1213
284k
    xEncCoeffQT( cs, partitioner, COMP_Cr );
1214
284k
  }
1215
1216
473k
  uint64_t fracBits = m_CABACEstimator->getEstFracBits();
1217
473k
  return fracBits;
1218
473k
}
1219
1220
uint64_t IntraSearch::xGetIntraFracBitsQTChroma(const TransformUnit& currTU, const ComponentID compID, CUCtx *cuCtx)
1221
1.77M
{
1222
1.77M
  m_CABACEstimator->resetBits();
1223
1224
1.77M
  if ( currTU.jointCbCr )
1225
264k
  {
1226
264k
    const int cbfMask = ( TU::getCbf( currTU, COMP_Cb ) ? 2 : 0 ) + ( TU::getCbf( currTU, COMP_Cr ) ? 1 : 0 );
1227
264k
    m_CABACEstimator->cbf_comp( *currTU.cu, cbfMask>>1, currTU.blocks[ COMP_Cb ], currTU.depth, false );
1228
264k
    m_CABACEstimator->cbf_comp( *currTU.cu, cbfMask &1, currTU.blocks[ COMP_Cr ], currTU.depth, cbfMask>>1 );
1229
264k
    if( cbfMask )
1230
264k
      m_CABACEstimator->joint_cb_cr( currTU, cbfMask );
1231
264k
    if (cbfMask >> 1)
1232
262k
      m_CABACEstimator->residual_coding( currTU, COMP_Cb, cuCtx );
1233
264k
    if (cbfMask & 1)
1234
264k
      m_CABACEstimator->residual_coding( currTU, COMP_Cr, cuCtx );
1235
264k
  }
1236
1.51M
  else
1237
1.51M
  {
1238
1.51M
    if ( compID == COMP_Cb )
1239
757k
      m_CABACEstimator->cbf_comp( *currTU.cu, TU::getCbf( currTU, compID ), currTU.blocks[ compID ], currTU.depth, false );
1240
757k
    else
1241
757k
    {
1242
757k
      const bool cbCbf    = TU::getCbf( currTU, COMP_Cb );
1243
757k
      const bool crCbf    = TU::getCbf( currTU, compID );
1244
757k
      const int  cbfMask  = ( cbCbf ? 2 : 0 ) + ( crCbf ? 1 : 0 );
1245
757k
      m_CABACEstimator->cbf_comp( *currTU.cu, crCbf, currTU.blocks[ compID ], currTU.depth, cbCbf );
1246
757k
      m_CABACEstimator->joint_cb_cr( currTU, cbfMask );
1247
757k
    }
1248
1.51M
  }
1249
1250
1.77M
  if( !currTU.jointCbCr && TU::getCbf( currTU, compID ) )
1251
531k
  {
1252
531k
    m_CABACEstimator->residual_coding( currTU, compID, cuCtx );
1253
531k
  }
1254
1255
1.77M
  uint64_t fracBits = m_CABACEstimator->getEstFracBits();
1256
1.77M
  return fracBits;
1257
1.77M
}
1258
1259
void IntraSearch::xIntraCodingTUBlock(TransformUnit &tu, const ComponentID compID, const bool checkCrossCPrediction, Distortion &ruiDist, uint32_t *numSig, PelUnitBuf *predBuf, const bool loadTr)
1260
1.97M
{
1261
1.97M
  if (!tu.blocks[compID].valid())
1262
0
  {
1263
0
    return;
1264
0
  }
1265
1266
1.97M
  CodingStructure &cs             = *tu.cs;
1267
1.97M
  const CompArea      &area       = tu.blocks[compID];
1268
1.97M
  const SPS           &sps        = *cs.sps;
1269
1.97M
  const ReshapeData&  reshapeData = cs.picture->reshapeData;
1270
1271
1.97M
  const ChannelType    chType     = toChannelType(compID);
1272
1.97M
  const int            bitDepth   = sps.bitDepths[chType];
1273
1274
1.97M
  CPelBuf        piOrg            = cs.getOrgBuf    (area);
1275
1.97M
  PelBuf         piPred           = cs.getPredBuf   (area);
1276
1.97M
  PelBuf         piResi           = cs.getResiBuf   (area);
1277
1.97M
  PelBuf         piReco           = cs.getRecoBuf   (area);
1278
1279
1.97M
  const CodingUnit& cu            = *tu.cu;
1280
1281
  //===== init availability pattern =====
1282
1.97M
  CHECK( tu.jointCbCr && compID == COMP_Cr, "wrong combination of compID and jointCbCr" );
1283
1.97M
  bool jointCbCr = tu.jointCbCr && compID == COMP_Cb;
1284
1285
1.97M
  if ( isLuma(compID) )
1286
193k
  {
1287
193k
    bool predRegDiffFromTB = CU::isPredRegDiffFromTB(*tu.cu );
1288
193k
    bool firstTBInPredReg  = false;
1289
193k
    CompArea areaPredReg(COMP_Y, tu.chromaFormat, area);
1290
193k
    if (tu.cu->ispMode )
1291
19.6k
    {
1292
19.6k
      firstTBInPredReg = CU::isFirstTBInPredReg(*tu.cu, area);
1293
19.6k
      if (predRegDiffFromTB)
1294
0
      {
1295
0
        if (firstTBInPredReg)
1296
0
        {
1297
0
          CU::adjustPredArea(areaPredReg);
1298
0
          initIntraPatternChTypeISP(*tu.cu, areaPredReg, piReco);
1299
0
        }
1300
0
      }
1301
19.6k
      else
1302
19.6k
        initIntraPatternChTypeISP(*tu.cu, area, piReco);
1303
19.6k
    }
1304
174k
    else if( !predBuf )
1305
29.7k
    {
1306
29.7k
      initIntraPatternChType(*tu.cu, area);
1307
29.7k
    }
1308
1309
    //===== get prediction signal =====
1310
193k
    if (predRegDiffFromTB)
1311
0
    {
1312
0
      if (firstTBInPredReg)
1313
0
      {
1314
0
        PelBuf piPredReg = cs.getPredBuf(areaPredReg);
1315
0
        predIntraAng(compID, piPredReg, cu);
1316
0
      }
1317
0
    }
1318
193k
    else
1319
193k
    {
1320
193k
      if( predBuf )
1321
144k
      {
1322
144k
        piPred.copyFrom( predBuf->Y() );
1323
144k
      }
1324
49.4k
      else if( CU::isMIP( cu, CH_L ) )
1325
22.0k
      {
1326
22.0k
        initIntraMip( cu );
1327
22.0k
        predIntraMip( piPred, cu );
1328
22.0k
      }
1329
27.3k
      else
1330
27.3k
      {
1331
27.3k
        predIntraAng(compID, piPred, cu);
1332
27.3k
      }
1333
193k
    }
1334
193k
  }
1335
1.97M
  DTRACE( g_trace_ctx, D_PRED, "@(%4d,%4d) [%2dx%2d] IMode=%d\n", tu.lx(), tu.ly(), tu.lwidth(), tu.lheight(), CU::getFinalIntraMode(cu, chType) );
1336
1.97M
  const Slice &slice = *cs.slice;
1337
1.97M
  bool flag = cs.picHeader->lmcsEnabled && (slice.isIntra() || (!slice.isIntra() && reshapeData.getCTUFlag()));
1338
1339
1.97M
  if (isLuma(compID))
1340
193k
  {
1341
    //===== get residual signal =====
1342
193k
    if (cs.picHeader->lmcsEnabled && reshapeData.getCTUFlag() )
1343
0
    {
1344
0
      piResi.subtract(cs.getRspOrgBuf(area), piPred);
1345
0
    }
1346
193k
    else
1347
193k
    {
1348
193k
      piResi.subtract( piOrg, piPred );
1349
193k
    }
1350
193k
  }
1351
1352
  //===== transform and quantization =====
1353
  //--- init rate estimation arrays for RDOQ ---
1354
  //--- transform and quantization           ---
1355
1.97M
  TCoeff uiAbsSum = 0;
1356
1.97M
  const QpParam cQP(tu, compID);
1357
1358
1.97M
  m_pcTrQuant->selectLambda(compID);
1359
1360
1.97M
  flag =flag && (tu.blocks[compID].width*tu.blocks[compID].height > 4);
1361
1.97M
  if (flag && isChroma(compID) && cs.picHeader->lmcsChromaResidualScale )
1362
0
  {
1363
0
    int cResScaleInv = tu.chromaAdj;
1364
0
    double cRescale = (double)(1 << CSCALE_FP_PREC) / (double)cResScaleInv;
1365
0
    m_pcTrQuant->scaleLambda( 1.0/(cRescale*cRescale) );
1366
0
  }
1367
1368
1.97M
  if ( jointCbCr )
1369
267k
  {
1370
    // Lambda is loosened for the joint mode with respect to single modes as the same residual is used for both chroma blocks
1371
267k
    const int    absIct = abs( TU::getICTMode(tu) );
1372
267k
    const double lfact  = ( absIct == 1 || absIct == 3 ? 0.8 : 0.5 );
1373
267k
    m_pcTrQuant->scaleLambda( lfact );
1374
267k
  }
1375
1.97M
  if ( sps.jointCbCr && isChroma(compID) && (tu.cu->cs->slice->sliceQp > 18) )
1376
1.20M
  {
1377
1.20M
    m_pcTrQuant->scaleLambda( 1.3 );
1378
1.20M
  }
1379
1380
1.97M
  if( isLuma(compID) )
1381
193k
  {
1382
193k
    m_pcTrQuant->transformNxN(tu, compID, cQP, uiAbsSum, m_CABACEstimator->getCtx(), loadTr);
1383
1384
193k
    DTRACE( g_trace_ctx, D_TU_ABS_SUM, "%d: comp=%d, abssum=%d\n", DTRACE_GET_COUNTER( g_trace_ctx, D_TU_ABS_SUM ), compID, uiAbsSum );
1385
193k
    if (tu.cu->ispMode && isLuma(compID) && CU::isISPLast(*tu.cu, area, area.compID) && CU::allLumaCBFsAreZero(*tu.cu))
1386
0
    {
1387
      // ISP has to have at least one non-zero CBF
1388
0
      ruiDist = MAX_INT;
1389
0
      return;
1390
0
    }
1391
    //--- inverse transform ---
1392
193k
    if (uiAbsSum > 0)
1393
30.5k
    {
1394
30.5k
      m_pcTrQuant->invTransformNxN(tu, compID, piResi, cQP);
1395
30.5k
    }
1396
163k
    else
1397
163k
    {
1398
163k
      piResi.fill(0);
1399
163k
    }
1400
193k
  }
1401
1.78M
  else // chroma
1402
1.78M
  {
1403
1.78M
    PelBuf          crPred = cs.getPredBuf ( COMP_Cr );
1404
1.78M
    PelBuf          crResi = cs.getResiBuf ( COMP_Cr );
1405
1.78M
    PelBuf          crReco = cs.getRecoBuf ( COMP_Cr );
1406
1407
1.78M
    int         codedCbfMask  = 0;
1408
1.78M
    ComponentID codeCompId    = (tu.jointCbCr ? (tu.jointCbCr >> 1 ? COMP_Cb : COMP_Cr) : compID);
1409
1.78M
    const QpParam qpCbCr(tu, codeCompId);
1410
1411
1.78M
    if( tu.jointCbCr )
1412
267k
    {
1413
267k
      ComponentID otherCompId = ( codeCompId==COMP_Cr ? COMP_Cb : COMP_Cr );
1414
267k
      tu.getCoeffs( otherCompId ).fill(0); // do we need that?
1415
267k
      TU::setCbfAtDepth (tu, otherCompId, tu.depth, false );
1416
267k
    }
1417
1.78M
    PelBuf& codeResi = ( codeCompId == COMP_Cr ? crResi : piResi );
1418
1.78M
    uiAbsSum = 0;
1419
1.78M
    m_pcTrQuant->transformNxN(tu, codeCompId, qpCbCr, uiAbsSum, m_CABACEstimator->getCtx(), loadTr);
1420
1.78M
    DTRACE( g_trace_ctx, D_TU_ABS_SUM, "%d: comp=%d, abssum=%d\n", DTRACE_GET_COUNTER( g_trace_ctx, D_TU_ABS_SUM ), codeCompId, uiAbsSum );
1421
1.78M
    if( uiAbsSum > 0 )
1422
796k
    {
1423
796k
      m_pcTrQuant->invTransformNxN(tu, codeCompId, codeResi, qpCbCr);
1424
796k
      codedCbfMask += ( codeCompId == COMP_Cb ? 2 : 1 );
1425
796k
    }
1426
986k
    else
1427
986k
    {
1428
986k
      codeResi.fill(0);
1429
986k
    }
1430
1431
1.78M
    if( tu.jointCbCr )
1432
267k
    {
1433
267k
      if( tu.jointCbCr == 3 && codedCbfMask == 2 )
1434
262k
      {
1435
262k
        codedCbfMask = 3;
1436
262k
        TU::setCbfAtDepth (tu, COMP_Cr, tu.depth, true );
1437
262k
      }
1438
267k
      if( tu.jointCbCr != codedCbfMask )
1439
3.51k
      {
1440
3.51k
        ruiDist = MAX_DISTORTION;
1441
3.51k
        return;
1442
3.51k
      }
1443
264k
      m_pcTrQuant->invTransformICT( tu, piResi, crResi );
1444
264k
      uiAbsSum = codedCbfMask;
1445
264k
    }
1446
1447
    //===== reconstruction =====
1448
1.77M
    if ( flag && uiAbsSum > 0 && cs.picHeader->lmcsChromaResidualScale )
1449
0
    {
1450
0
      piResi.scaleSignal(tu.chromaAdj, 0, slice.clpRngs[compID]);
1451
1452
0
      if( jointCbCr )
1453
0
      {
1454
0
        crResi.scaleSignal(tu.chromaAdj, 0, slice.clpRngs[COMP_Cr]);
1455
0
      }
1456
0
    }
1457
1458
1.77M
    if( jointCbCr )
1459
264k
    {
1460
264k
      crReco.reconstruct(crPred, crResi, cs.slice->clpRngs[ COMP_Cr ]);
1461
264k
    }
1462
1.77M
  }
1463
1.97M
  piReco.reconstruct(piPred, piResi, cs.slice->clpRngs[ compID ]);
1464
  
1465
1466
1467
  //===== update distortion =====
1468
1.97M
  const bool reshapeIntraCMD = m_pcEncCfg->m_reshapeSignalType == RESHAPE_SIGNAL_PQ;
1469
1.97M
  if(((cs.picHeader->lmcsEnabled && (reshapeData.getCTUFlag() || (isChroma(compID) && reshapeIntraCMD))) || m_pcEncCfg->m_lumaLevelToDeltaQPEnabled ) )
1470
0
  {
1471
0
    const CPelBuf orgLuma = cs.getOrgBuf( cs.area.blocks[COMP_Y] );
1472
0
    if( compID == COMP_Y && !m_pcEncCfg->m_lumaLevelToDeltaQPEnabled )
1473
0
    {
1474
0
      PelBuf tmpRecLuma = cs.getRspRecoBuf(area);
1475
0
      tmpRecLuma.rspSignal( piReco, reshapeData.getInvLUT());
1476
0
      ruiDist += m_pcRdCost->getDistPart(piOrg, tmpRecLuma, sps.bitDepths[toChannelType(compID)], compID, DF_SSE_WTD, &orgLuma);
1477
0
    }
1478
0
    else
1479
0
    {
1480
0
      ruiDist += m_pcRdCost->getDistPart( piOrg, piReco, bitDepth, compID, DF_SSE_WTD, &orgLuma );
1481
0
      if( jointCbCr )
1482
0
      {
1483
0
        CPelBuf         crOrg  = cs.getOrgBuf  ( COMP_Cr );
1484
0
        PelBuf          crReco = cs.getRecoBuf ( COMP_Cr );
1485
0
        ruiDist += m_pcRdCost->getDistPart( crOrg, crReco, bitDepth, COMP_Cr, DF_SSE_WTD, &orgLuma );
1486
0
      }
1487
0
    }
1488
0
  }
1489
1.97M
  else
1490
1.97M
  {
1491
1.97M
    ruiDist += m_pcRdCost->getDistPart( piOrg, piReco, bitDepth, compID, DF_SSE );
1492
1.97M
    if( jointCbCr )
1493
264k
    {
1494
264k
      CPelBuf         crOrg  = cs.getOrgBuf  ( COMP_Cr );
1495
264k
      PelBuf          crReco = cs.getRecoBuf ( COMP_Cr );
1496
264k
      ruiDist += m_pcRdCost->getDistPart( crOrg, crReco, bitDepth, COMP_Cr, DF_SSE );
1497
264k
    }
1498
1.97M
  }
1499
1.97M
}
1500
1501
void IntraSearch::xIntraCodingLumaQT(CodingStructure& cs, Partitioner& partitioner, PelUnitBuf* predBuf, const double bestCostSoFar, int numMode, bool disableMTS)
1502
116k
{
1503
116k
  PROFILER_SCOPE_AND_STAGE_EXT( 0, _TPROF, P_INTRA_RD_SEARCH_LUMA, &cs, partitioner.chType );
1504
116k
  const UnitArea& currArea  = partitioner.currArea();
1505
116k
  uint32_t        currDepth = partitioner.currTrDepth;
1506
116k
  Distortion singleDistLuma = 0;
1507
116k
  uint32_t   numSig         = 0;
1508
116k
  const SPS &sps            = *cs.sps;
1509
116k
  CodingUnit &cu            = *cs.cus[0];
1510
116k
  bool mtsAllowed = (numMode < 0) || disableMTS ? false : CU::isMTSAllowed(cu, COMP_Y);
1511
116k
  uint64_t singleFracBits   = 0;
1512
116k
  bool   splitCbfLumaSum    = false;
1513
116k
  double bestCostForISP     = bestCostSoFar;
1514
116k
  double dSingleCost        = MAX_DOUBLE;
1515
116k
  int endLfnstIdx           = (partitioner.isSepTree(cs) && partitioner.chType == CH_C && (currArea.lwidth() < 8 || currArea.lheight() < 8))
1516
116k
                           || (currArea.lwidth() > sps.getMaxTbSize() || currArea.lheight() > sps.getMaxTbSize()) || !sps.LFNST || (numMode < 0) ? 0 : 2;
1517
116k
  const bool useTS          = cs.picture->useTS;
1518
116k
  numMode                   = (numMode < 0) ? -numMode : numMode;
1519
1520
116k
  if (cu.mipFlag && !allowLfnstWithMip(cu.lumaSize()))
1521
2.07k
  {
1522
2.07k
    endLfnstIdx = 0;
1523
2.07k
  }
1524
116k
  int bestMTS = 0;
1525
116k
  int EndMTS  = mtsAllowed ? m_pcEncCfg->m_MTSIntraMaxCand : 0;
1526
116k
  if (cu.ispMode && (EndMTS || endLfnstIdx))
1527
5.34k
  {
1528
5.34k
    EndMTS = 0;
1529
5.34k
    if ((m_ispTestedModes[1].numTotalParts[cu.ispMode - 1] == 0)
1530
295
     && (m_ispTestedModes[2].numTotalParts[cu.ispMode - 1] == 0))
1531
295
    {
1532
295
      endLfnstIdx = 0;
1533
295
    }
1534
5.34k
  }
1535
116k
  if (cu.bdpcmM[CH_L])
1536
7.45k
  {
1537
7.45k
    endLfnstIdx = 0;
1538
7.45k
    EndMTS = 0;
1539
7.45k
  }
1540
116k
  bool checkTransformSkip = sps.transformSkip;
1541
1542
116k
  SizeType transformSkipMaxSize = 1 << sps.log2MaxTransformSkipBlockSize;
1543
116k
  bool tsAllowed = useTS  && cu.cs->sps->transformSkip && (!cu.ispMode) && (!cu.bdpcmM[CH_L]) && (!cu.sbtInfo);
1544
116k
  tsAllowed &= cu.blocks[COMP_Y].width <= transformSkipMaxSize && cu.blocks[COMP_Y].height <= transformSkipMaxSize;
1545
116k
  if (tsAllowed)
1546
14.9k
  {
1547
14.9k
    EndMTS += 1;
1548
14.9k
  }
1549
116k
  if (endLfnstIdx || EndMTS)
1550
47.1k
  {
1551
47.1k
    bool       splitCbfLuma  = false;
1552
47.1k
    const PartSplit ispType  = CU::getISPType(cu, COMP_Y);
1553
47.1k
    CUCtx cuCtx;
1554
47.1k
    cuCtx.isDQPCoded         = true;
1555
47.1k
    cuCtx.isChromaQpAdjCoded = true;
1556
47.1k
    cs.cost                  = 0.0;
1557
47.1k
    Distortion       singleDistTmpLuma = 0;
1558
47.1k
    uint64_t         singleTmpFracBits = 0;
1559
47.1k
    double           singleCostTmp     = 0;
1560
47.1k
    const TempCtx    ctxStart          (m_CtxCache, m_CABACEstimator->getCtx());
1561
47.1k
          TempCtx    ctxBest           (m_CtxCache);
1562
47.1k
    CodingStructure &saveCS            = *m_pSaveCS[cu.ispMode?0:1];
1563
47.1k
    TransformUnit *  tmpTU             = nullptr;
1564
47.1k
    int              bestLfnstIdx      = 0;
1565
47.1k
    int              startLfnstIdx     = 0;
1566
    // speedUps LFNST
1567
47.1k
    bool   rapidLFNST                  = false;
1568
47.1k
    bool   rapidDCT                    = false;
1569
47.1k
    double thresholdDCT                = 1;
1570
1571
47.1k
    if (m_pcEncCfg->m_MTS == 2)
1572
0
    {
1573
0
      thresholdDCT += 1.4 / sqrt(cu.lwidth() * cu.lheight());
1574
0
    }
1575
1576
47.1k
    if (m_pcEncCfg->m_LFNST > 1)
1577
0
    {
1578
0
      rapidLFNST = true;
1579
1580
0
      if (m_pcEncCfg->m_LFNST > 2)
1581
0
      {
1582
0
        rapidDCT    = true;
1583
0
        endLfnstIdx = endLfnstIdx ? 1 : 0;
1584
0
      }
1585
0
    }
1586
1587
47.1k
    saveCS.pcv              = cs.pcv;
1588
47.1k
    saveCS.picture          = cs.picture;
1589
47.1k
    saveCS.area.repositionTo( cs.area);
1590
1591
47.1k
    if (cu.ispMode)
1592
5.05k
    {
1593
5.05k
      partitioner.splitCurrArea(ispType, cs);
1594
5.05k
    }
1595
1596
47.1k
    TransformUnit& tu = cs.addTU(CS::getArea(cs, partitioner.currArea(), partitioner.chType, partitioner.treeType), partitioner.chType, cs.cus[0]);
1597
1598
47.1k
    if (cu.ispMode)
1599
5.05k
    {
1600
5.05k
      saveCS.clearTUs();
1601
5.05k
      do
1602
20.2k
      {
1603
20.2k
        saveCS.addTU(
1604
20.2k
          CS::getArea(cs, partitioner.currArea(), partitioner.chType, partitioner.treeType),
1605
20.2k
          partitioner.chType, cs.cus[0]);
1606
20.2k
      } while (partitioner.nextPart(cs));
1607
1608
5.05k
      partitioner.exitCurrSplit();
1609
5.05k
    }
1610
42.0k
    else
1611
42.0k
    {
1612
42.0k
      tmpTU = saveCS.tus.empty() ? &saveCS.addTU( currArea, partitioner.chType, nullptr ) : saveCS.tus.front();
1613
42.0k
      tmpTU->initData();
1614
42.0k
      tmpTU->UnitArea::operator=( currArea );
1615
42.0k
    }
1616
1617
1618
47.1k
    std::vector<TrMode> trModes{ TrMode(0, true) };
1619
47.1k
    if (tsAllowed)
1620
14.9k
    {
1621
14.9k
      trModes.push_back(TrMode(1, true));
1622
14.9k
    }
1623
47.1k
    double dct2Cost           = MAX_DOUBLE;
1624
47.1k
    double trGrpStopThreshold = 1.001;
1625
47.1k
    double trGrpBestCost      = MAX_DOUBLE;
1626
1627
47.1k
    if (mtsAllowed)
1628
0
    {
1629
0
      if (m_pcEncCfg->m_LFNST)
1630
0
      {
1631
0
        uint32_t uiIntraMode = cs.cus[0]->intraDir[partitioner.chType];
1632
0
        int MTScur           = (uiIntraMode < 34) ? MTS_DST7_DCT8 : MTS_DCT8_DST7;
1633
1634
0
        trModes.push_back(TrMode(     2, true));
1635
0
        trModes.push_back(TrMode(MTScur, true));
1636
1637
0
        MTScur = (uiIntraMode < 34) ? MTS_DCT8_DST7 : MTS_DST7_DCT8;
1638
1639
0
        trModes.push_back(TrMode(MTScur,            true));
1640
0
        trModes.push_back(TrMode(MTS_DST7_DST7 + 3, true));
1641
0
      }
1642
0
      else
1643
0
      {
1644
0
        for (int i = 2; i < 6; i++)
1645
0
        {
1646
0
          trModes.push_back(TrMode(i, true));
1647
0
        }
1648
0
      }
1649
0
    }
1650
1651
47.1k
    if ((EndMTS && !m_pcEncCfg->m_LFNST) || (tsAllowed && !mtsAllowed))
1652
14.9k
    {
1653
14.9k
      xPreCheckMTS(tu, &trModes, m_pcEncCfg->m_MTSIntraMaxCand, predBuf);
1654
14.9k
      if (!mtsAllowed && !trModes[1].second)
1655
2.84k
      {
1656
2.84k
        EndMTS = 0;
1657
2.84k
      }
1658
14.9k
    }
1659
1660
47.1k
    bool NStopMTS = true;
1661
1662
94.2k
    for (int modeId = 0; modeId <= EndMTS && NStopMTS; modeId++)
1663
47.1k
    {
1664
47.1k
      if (modeId > 1)
1665
0
      {
1666
0
        trGrpBestCost = MAX_DOUBLE;
1667
0
      }
1668
167k
      for (int lfnstIdx = startLfnstIdx; lfnstIdx <= endLfnstIdx; lfnstIdx++)
1669
120k
      {
1670
120k
        if (lfnstIdx && modeId)
1671
0
        {
1672
0
          continue;
1673
0
        }
1674
120k
        if (mtsAllowed || tsAllowed)
1675
23.4k
        {
1676
23.4k
          if (m_pcEncCfg->m_TS && bestMTS == MTS_SKIP)
1677
0
          {
1678
0
            break;
1679
0
          }
1680
23.4k
          if (!m_pcEncCfg->m_LFNST && !trModes[modeId].second && mtsAllowed)
1681
0
          {
1682
0
            continue;
1683
0
          }
1684
1685
23.4k
          tu.mtsIdx[COMP_Y] = trModes[modeId].first;
1686
23.4k
        }
1687
1688
120k
        if (cu.ispMode && lfnstIdx)
1689
10.1k
        {
1690
10.1k
          if (m_ispTestedModes[lfnstIdx].numTotalParts[cu.ispMode - 1] == 0)
1691
0
          {
1692
0
            if (lfnstIdx == 2)
1693
0
            {
1694
0
              endLfnstIdx = 1;
1695
0
            }
1696
0
            continue;
1697
0
          }
1698
10.1k
        }
1699
1700
120k
        cu.lfnstIdx                          = lfnstIdx;
1701
120k
        cuCtx.lfnstLastScanPos               = false;
1702
120k
        cuCtx.violatesLfnstConstrained[CH_L] = false;
1703
120k
        cuCtx.violatesLfnstConstrained[CH_C] = false;
1704
1705
120k
        if ((lfnstIdx != startLfnstIdx) || (modeId))
1706
72.9k
        {
1707
72.9k
          m_CABACEstimator->getCtx() = ctxStart;
1708
72.9k
        }
1709
1710
120k
        singleDistTmpLuma = 0;
1711
1712
120k
        if (cu.ispMode)
1713
15.1k
        {
1714
15.1k
          splitCbfLuma = false;
1715
1716
15.1k
          partitioner.splitCurrArea(ispType, cs);
1717
1718
15.1k
          singleCostTmp = xTestISP(cs, partitioner, bestCostForISP, ispType, splitCbfLuma, singleTmpFracBits, singleDistTmpLuma, cuCtx);
1719
1720
15.1k
          partitioner.exitCurrSplit();
1721
1722
15.1k
          if (modeId && (singleCostTmp == MAX_DOUBLE))
1723
0
          {
1724
0
            m_ispTestedModes[lfnstIdx].numTotalParts[cu.ispMode - 1] = 0;
1725
0
          }
1726
1727
15.1k
          bool storeCost = (numMode == 1) ? true : false;
1728
1729
15.1k
          if ((m_pcEncCfg->m_ISP >= 2) && (numMode <= 1))
1730
15.1k
          {
1731
15.1k
            storeCost = true;
1732
15.1k
          }
1733
1734
15.1k
          if (storeCost)
1735
15.1k
          {
1736
15.1k
            m_ispTestedModes[0].bestCost[cu.ispMode - 1] = singleCostTmp;
1737
15.1k
          }
1738
15.1k
        }
1739
104k
        else
1740
104k
        {
1741
104k
          bool TrLoad = (EndMTS && !m_pcEncCfg->m_LFNST) || (tsAllowed && !mtsAllowed && (lfnstIdx == 0)) ? true : false;
1742
1743
104k
          xIntraCodingTUBlock(tu, COMP_Y, false, singleDistTmpLuma, &numSig, predBuf, TrLoad);
1744
1745
104k
          cuCtx.mtsLastScanPos = false;
1746
          //----- determine rate and r-d cost -----
1747
104k
        if ((sps.LFNST ? (modeId == EndMTS && modeId != 0 && checkTransformSkip) : (trModes[modeId].first != 0)) && !TU::getCbfAtDepth(tu, COMP_Y, currDepth))
1748
0
        {
1749
0
          singleCostTmp = MAX_DOUBLE;
1750
0
        }
1751
104k
        else
1752
104k
        {
1753
104k
          m_ispTestedModes[0].IspType      = TU_NO_ISP;
1754
104k
          m_ispTestedModes[0].subTuCounter = -1;
1755
104k
          singleTmpFracBits = xGetIntraFracBitsQT(cs, partitioner, true, &cuCtx);
1756
1757
104k
          if (tu.mtsIdx[COMP_Y] > MTS_SKIP)
1758
0
          {
1759
0
            if (!cuCtx.mtsLastScanPos)
1760
0
            {
1761
0
              singleCostTmp = MAX_DOUBLE;
1762
0
            }
1763
0
            else
1764
0
            {
1765
0
              singleCostTmp = m_pcRdCost->calcRdCost(singleTmpFracBits, singleDistTmpLuma);
1766
0
            }
1767
0
          }
1768
104k
          else
1769
104k
          {
1770
104k
            singleCostTmp = m_pcRdCost->calcRdCost(singleTmpFracBits, singleDistTmpLuma);
1771
104k
          }
1772
104k
        }
1773
1774
104k
          if (((EndMTS && (m_pcEncCfg->m_MTS == 2)) || rapidLFNST) && modeId == 0 && lfnstIdx == 0)
1775
0
          {
1776
0
            if (singleCostTmp > bestCostSoFar * thresholdDCT)
1777
0
            {
1778
0
              EndMTS = 0;
1779
1780
0
              if (rapidDCT)
1781
0
              {
1782
0
                endLfnstIdx = 0;   // break the loop but do not cpy best
1783
0
              }
1784
0
            }
1785
0
          }
1786
1787
104k
          if (lfnstIdx && !cuCtx.lfnstLastScanPos && !cu.ispMode)
1788
52.7k
          {
1789
52.7k
            bool rootCbfL = false;
1790
1791
210k
            for (uint32_t t = 0; t < getNumberValidTBlocks(*cu.cs->pcv); t++)
1792
158k
            {
1793
158k
              rootCbfL |= tu.cbf[t] != 0;
1794
158k
            }
1795
1796
52.7k
            if (rapidLFNST && !rootCbfL)
1797
0
            {
1798
0
              endLfnstIdx = lfnstIdx; // break the loop
1799
0
            }
1800
52.7k
            bool cbfAtZeroDepth = CU::isSepTree(cu)
1801
52.7k
              ? rootCbfL
1802
52.7k
              : (cs.area.chromaFormat != CHROMA_400 && std::min(cu.firstTU->blocks[1].width, cu.firstTU->blocks[1].height) < 4)
1803
0
                ? TU::getCbfAtDepth(tu, COMP_Y, currDepth)
1804
0
                : rootCbfL;
1805
1806
52.7k
            if (cbfAtZeroDepth)
1807
375
            {
1808
375
              singleCostTmp = MAX_DOUBLE;
1809
375
            }
1810
52.7k
          }
1811
104k
        }
1812
1813
120k
        if (singleCostTmp < dSingleCost)
1814
43.4k
        {
1815
43.4k
          trGrpBestCost  = singleCostTmp;
1816
43.4k
          dSingleCost    = singleCostTmp;
1817
43.4k
          singleDistLuma = singleDistTmpLuma;
1818
43.4k
          singleFracBits = singleTmpFracBits;
1819
43.4k
          bestLfnstIdx   = lfnstIdx;
1820
43.4k
          bestMTS        = modeId;
1821
1822
43.4k
          if (dSingleCost < bestCostForISP)
1823
27.3k
          {
1824
27.3k
            bestCostForISP = dSingleCost;
1825
27.3k
          }
1826
1827
43.4k
          splitCbfLumaSum = splitCbfLuma;
1828
1829
43.4k
          if (lfnstIdx == 0 && modeId == 0 && cu.ispMode == 0)
1830
42.0k
          {
1831
42.0k
            dct2Cost = singleCostTmp;
1832
1833
42.0k
            if (!TU::getCbfAtDepth(tu, COMP_Y, currDepth))
1834
35.7k
            {
1835
35.7k
              if (rapidLFNST)
1836
0
              {
1837
0
                 endLfnstIdx = 0;   // break the loop but do not cpy best
1838
0
              }
1839
1840
35.7k
              EndMTS = 0;
1841
35.7k
            }
1842
42.0k
          }
1843
1844
43.4k
          if (bestLfnstIdx != endLfnstIdx || bestMTS != EndMTS)
1845
32.6k
          {
1846
32.6k
            if (cu.ispMode)
1847
1.04k
            {
1848
1.04k
              saveCS.getRecoBuf(currArea.Y()).copyFrom(cs.getRecoBuf(currArea.Y()));
1849
1850
5.22k
              for (uint32_t j = 0; j < cs.tus.size(); j++)
1851
4.17k
              {
1852
4.17k
                saveCS.tus[j]->copyComponentFrom(*cs.tus[j], COMP_Y);
1853
4.17k
              }
1854
1.04k
            }
1855
31.5k
            else
1856
31.5k
            {
1857
31.5k
              saveCS.getPredBuf(tu.Y()).copyFrom(cs.getPredBuf(tu.Y()));
1858
31.5k
              saveCS.getRecoBuf(tu.Y()).copyFrom(cs.getRecoBuf(tu.Y()));
1859
1860
31.5k
              tmpTU->copyComponentFrom(tu, COMP_Y);
1861
31.5k
            }
1862
1863
32.6k
            ctxBest = m_CABACEstimator->getCtx();
1864
32.6k
          }
1865
      
1866
43.4k
        }
1867
76.6k
        else
1868
76.6k
        {
1869
76.6k
          if( rapidLFNST )
1870
0
          {
1871
0
            endLfnstIdx = lfnstIdx; // break the loop
1872
0
          }
1873
76.6k
        }
1874
120k
      }
1875
47.1k
      if (m_pcEncCfg->m_LFNST && m_pcEncCfg->m_MTS == 2 && modeId && modeId != EndMTS)
1876
0
      {
1877
0
        NStopMTS = false;
1878
1879
0
        if (bestMTS || bestLfnstIdx)
1880
0
        {
1881
0
          if ((modeId > 1 && bestMTS == modeId) || modeId == 1)
1882
0
          {
1883
0
            NStopMTS = (dct2Cost / trGrpBestCost) < trGrpStopThreshold;
1884
0
          }
1885
0
        }
1886
0
      }
1887
47.1k
    }
1888
1889
47.1k
    cu.lfnstIdx = bestLfnstIdx;
1890
47.1k
    if (dSingleCost != MAX_DOUBLE)
1891
42.9k
    {
1892
42.9k
      if (bestLfnstIdx != endLfnstIdx || bestMTS != EndMTS)
1893
32.1k
      {
1894
32.1k
        if (cu.ispMode)
1895
734
        {
1896
734
          const UnitArea& currArea = partitioner.currArea();
1897
734
          cs.getRecoBuf(currArea.Y()).copyFrom(saveCS.getRecoBuf(currArea.Y()));
1898
1899
734
          if (saveCS.tus.size() != cs.tus.size())
1900
0
          {
1901
0
            partitioner.splitCurrArea(ispType, cs);
1902
1903
0
            do
1904
0
            {
1905
0
              partitioner.nextPart(cs);
1906
0
              cs.addTU(CS::getArea(cs, partitioner.currArea(), partitioner.chType, partitioner.treeType),
1907
0
                partitioner.chType, cs.cus[0]);
1908
0
            } while (saveCS.tus.size() != cs.tus.size());
1909
1910
0
            partitioner.exitCurrSplit();
1911
0
          }
1912
1913
3.67k
          for (uint32_t j = 0; j < saveCS.tus.size(); j++)
1914
2.93k
          {
1915
2.93k
            cs.tus[j]->copyComponentFrom(*saveCS.tus[j], COMP_Y);
1916
2.93k
          }
1917
734
        }
1918
31.4k
        else
1919
31.4k
        {
1920
31.4k
          cs.getRecoBuf(tu.Y()).copyFrom(saveCS.getRecoBuf(tu.Y()));
1921
1922
31.4k
          tu.copyComponentFrom(*tmpTU, COMP_Y);
1923
31.4k
        }
1924
1925
32.1k
        m_CABACEstimator->getCtx() = ctxBest;
1926
32.1k
      }
1927
1928
      // otherwise this would've happened in useSubStructure
1929
42.9k
      cs.picture->getRecoBuf(currArea.Y()).copyFrom(cs.getRecoBuf(currArea.Y()));
1930
42.9k
    }
1931
47.1k
  }
1932
69.6k
  else
1933
69.6k
  {
1934
69.6k
    if (cu.ispMode)
1935
295
    {
1936
295
      const PartSplit ispType = CU::getISPType(cu, COMP_Y);
1937
295
      partitioner.splitCurrArea(ispType, cs);
1938
1939
295
      CUCtx      cuCtx;
1940
295
      dSingleCost = xTestISP(cs, partitioner, bestCostForISP, ispType, splitCbfLumaSum, singleFracBits, singleDistLuma, cuCtx);
1941
295
      partitioner.exitCurrSplit();
1942
295
      bool storeCost = (numMode == 1) ? true : false;
1943
295
      if ((m_pcEncCfg->m_ISP >= 2) && (numMode <= 1))
1944
295
      {
1945
295
        storeCost = true;
1946
295
      }
1947
295
      if (storeCost)
1948
295
      {
1949
295
        m_ispTestedModes[0].bestCost[cu.ispMode - 1] = dSingleCost;
1950
295
      }
1951
295
    }
1952
69.3k
    else
1953
69.3k
    {
1954
69.3k
      TransformUnit& tu =
1955
69.3k
        cs.addTU(CS::getArea(cs, currArea, partitioner.chType, partitioner.treeType), partitioner.chType, cs.cus[0]);
1956
69.3k
      tu.depth = currDepth;
1957
1958
69.3k
      CHECK(!tu.Y().valid(), "Invalid TU");
1959
69.3k
      xIntraCodingTUBlock(tu, COMP_Y, false, singleDistLuma, &numSig, predBuf);
1960
      //----- determine rate and r-d cost -----
1961
69.3k
      m_ispTestedModes[0].IspType = TU_NO_ISP;
1962
69.3k
      m_ispTestedModes[0].subTuCounter = -1;
1963
69.3k
      singleFracBits = xGetIntraFracBitsQT(cs, partitioner, true);
1964
69.3k
      dSingleCost = m_pcRdCost->calcRdCost(singleFracBits, singleDistLuma);
1965
69.3k
    }
1966
69.6k
  }
1967
1968
116k
  if (cu.ispMode)
1969
5.34k
  { 
1970
5.34k
    for (auto& ptu : cs.tus)
1971
8.58k
    {
1972
8.58k
      if (currArea.Y().contains(ptu->Y()))
1973
8.58k
      {
1974
8.58k
        TU::setCbfAtDepth(*ptu, COMP_Y, currDepth, splitCbfLumaSum ? 1 : 0);
1975
8.58k
      }
1976
8.58k
    }
1977
5.34k
  }
1978
116k
  cs.dist     += singleDistLuma;
1979
116k
  cs.fracBits += singleFracBits;
1980
116k
  cs.cost      = dSingleCost;
1981
1982
116k
  STAT_COUNT_CU_MODES( partitioner.chType == CH_L, g_cuCounters1D[CU_RD_TESTS][0][!cs.slice->isIntra() + cs.slice->depth] );
1983
116k
  STAT_COUNT_CU_MODES( partitioner.chType == CH_L && !cs.slice->isIntra(), g_cuCounters2D[CU_RD_TESTS][Log2( cs.area.lheight() )][Log2( cs.area.lwidth() )] );
1984
116k
}
1985
1986
ChromaCbfs IntraSearch::xIntraChromaCodingQT(CodingStructure& cs, Partitioner& partitioner)
1987
284k
{
1988
284k
  UnitArea    currArea      = partitioner.currArea();
1989
1990
284k
  if( !currArea.Cb().valid() ) 
1991
0
    return ChromaCbfs(false);
1992
1993
284k
  TransformUnit& currTU     = *cs.getTU( currArea.chromaPos(), CH_C );
1994
284k
  const CodingUnit& cu  = *cs.getCU( currArea.chromaPos(), CH_C, TREE_D );
1995
284k
  ChromaCbfs cbfs(false);
1996
284k
  uint32_t   currDepth = partitioner.currTrDepth;
1997
284k
  const bool useTS = cs.picture->useTS;
1998
284k
  if (currDepth == currTU.depth)
1999
284k
  {
2000
284k
    if (!currArea.Cb().valid() || !currArea.Cr().valid())
2001
0
    {
2002
0
      return cbfs;
2003
0
    }
2004
2005
284k
    CodingStructure& saveCS = *m_pSaveCS[1];
2006
284k
    saveCS.pcv = cs.pcv;
2007
284k
    saveCS.picture = cs.picture;
2008
284k
    saveCS.area.repositionTo(cs.area);
2009
2010
284k
    TransformUnit& tmpTU = saveCS.tus.empty() ? saveCS.addTU(currArea, partitioner.chType, nullptr) : *saveCS.tus.front();
2011
284k
    tmpTU.initData();
2012
284k
    tmpTU.UnitArea::operator=(currArea);
2013
284k
    const unsigned      numTBlocks = getNumberValidTBlocks(*cs.pcv);
2014
2015
284k
    CompArea& cbArea = currTU.blocks[COMP_Cb];
2016
284k
    CompArea& crArea = currTU.blocks[COMP_Cr];
2017
284k
    double     bestCostCb = MAX_DOUBLE;
2018
284k
    double     bestCostCr = MAX_DOUBLE;
2019
284k
    Distortion bestDistCb = 0;
2020
284k
    Distortion bestDistCr = 0;
2021
2022
284k
    TempCtx ctxStartTU(m_CtxCache);
2023
284k
    TempCtx ctxStart(m_CtxCache);
2024
284k
    TempCtx ctxBest(m_CtxCache);
2025
2026
284k
    ctxStartTU = m_CABACEstimator->getCtx();
2027
284k
    ctxStart = m_CABACEstimator->getCtx();
2028
284k
    currTU.jointCbCr = 0;
2029
2030
    // Do predictions here to avoid repeating the "default0Save1Load2" stuff
2031
284k
    int  predMode = cu.bdpcmM[CH_C] ? BDPCM_IDX : CU::getFinalIntraMode(cu, CH_C);
2032
2033
284k
    PelBuf piPredCb = cs.getPredBuf(COMP_Cb);
2034
284k
    PelBuf piPredCr = cs.getPredBuf(COMP_Cr);
2035
2036
284k
    initIntraPatternChType(*currTU.cu, cbArea);
2037
284k
    initIntraPatternChType(*currTU.cu, crArea);
2038
2039
284k
    if (CU::isLMCMode(predMode))
2040
20.9k
    {
2041
20.9k
      loadLMLumaRecPels(cu, cbArea);
2042
20.9k
      predIntraChromaLM(COMP_Cb, piPredCb, cu, cbArea, predMode);
2043
20.9k
      predIntraChromaLM(COMP_Cr, piPredCr, cu, crArea, predMode);
2044
20.9k
    }
2045
263k
    else
2046
263k
    {
2047
263k
      predIntraAng(COMP_Cb, piPredCb, cu);
2048
263k
      predIntraAng(COMP_Cr, piPredCr, cu);
2049
263k
    }
2050
2051
    // determination of chroma residuals including reshaping and cross-component prediction
2052
    //----- get chroma residuals -----
2053
284k
    PelBuf resiCb = cs.getResiBuf(COMP_Cb);
2054
284k
    PelBuf resiCr = cs.getResiBuf(COMP_Cr);
2055
284k
    resiCb.subtract(cs.getOrgBuf(COMP_Cb), piPredCb);
2056
284k
    resiCr.subtract(cs.getOrgBuf(COMP_Cr), piPredCr);
2057
2058
    //----- get reshape parameter ----
2059
284k
    ReshapeData& reshapeData = cs.picture->reshapeData;
2060
284k
    bool doReshaping = (cs.picHeader->lmcsEnabled && cs.picHeader->lmcsChromaResidualScale && (cs.slice->isIntra() || reshapeData.getCTUFlag()) && (cbArea.width * cbArea.height > 4));
2061
284k
    if (doReshaping)
2062
0
    {
2063
0
      const Area area = currTU.Y().valid() ? currTU.Y() : Area(recalcPosition(currTU.chromaFormat, currTU.chType, CH_L, currTU.blocks[currTU.chType].pos()), recalcSize(currTU.chromaFormat, currTU.chType, CH_L, currTU.blocks[currTU.chType].size()));
2064
0
      const CompArea& areaY = CompArea(COMP_Y, currTU.chromaFormat, area);
2065
0
      currTU.chromaAdj = reshapeData.calculateChromaAdjVpduNei(currTU, areaY, currTU.cu->treeType);
2066
0
    }
2067
2068
    //===== store original residual signals (std and crossCompPred) =====
2069
1.70M
    for( int k = 0; k < 5; k++ )
2070
1.42M
    {
2071
1.42M
      m_orgResiCb[k].compactResize( cbArea );
2072
1.42M
      m_orgResiCr[k].compactResize( crArea );
2073
1.42M
    }
2074
569k
    for (int k = 0; k < 1; k += 4)
2075
284k
    {
2076
284k
      m_orgResiCb[k].copyFrom(resiCb);
2077
284k
      m_orgResiCr[k].copyFrom(resiCr);
2078
2079
284k
      if (doReshaping)
2080
0
      {
2081
0
        int cResScaleInv = currTU.chromaAdj;
2082
0
        m_orgResiCb[k].scaleSignal(cResScaleInv, 1, cs.slice->clpRngs[COMP_Cb]);
2083
0
        m_orgResiCr[k].scaleSignal(cResScaleInv, 1, cs.slice->clpRngs[COMP_Cr]);
2084
0
      }
2085
284k
    }
2086
2087
284k
    CUCtx cuCtx;
2088
284k
    cuCtx.isDQPCoded = true;
2089
284k
    cuCtx.isChromaQpAdjCoded = true;
2090
284k
    cuCtx.lfnstLastScanPos = false;
2091
2092
284k
    CodingStructure& saveCScur = *m_pSaveCS[2];
2093
2094
284k
    saveCScur.pcv = cs.pcv;
2095
284k
    saveCScur.picture = cs.picture;
2096
284k
    saveCScur.area.repositionTo(cs.area);
2097
2098
284k
    TransformUnit& tmpTUcur = saveCScur.tus.empty() ? saveCScur.addTU(currArea, partitioner.chType, nullptr) : *saveCScur.tus.front();
2099
284k
    tmpTUcur.initData();
2100
284k
    tmpTUcur.UnitArea::operator=(currArea);
2101
2102
284k
    TempCtx ctxBestTUL(m_CtxCache);
2103
2104
284k
    const SPS& sps = *cs.sps;
2105
284k
    double     bestCostCbcur = MAX_DOUBLE;
2106
284k
    double     bestCostCrcur = MAX_DOUBLE;
2107
284k
    Distortion bestDistCbcur = 0;
2108
284k
    Distortion bestDistCrcur = 0;
2109
2110
284k
    int  endLfnstIdx = (partitioner.isSepTree(cs) && partitioner.chType == CH_C && (partitioner.currArea().lwidth() < 8 || partitioner.currArea().lheight() < 8))
2111
272k
      || (partitioner.currArea().lwidth() > sps.getMaxTbSize() || partitioner.currArea().lheight() > sps.getMaxTbSize()) || !sps.LFNST ? 0 : 2;
2112
284k
    int  startLfnstIdx = 0;
2113
284k
    int  bestLfnstIdx = 0;
2114
284k
    bool testLFNST = sps.LFNST;
2115
2116
    // speedUps LFNST
2117
284k
    bool rapidLFNST = false;
2118
284k
    if (m_pcEncCfg->m_LFNST > 1)
2119
0
    {
2120
0
      rapidLFNST = true;
2121
0
      if (m_pcEncCfg->m_LFNST > 2)
2122
0
      {
2123
0
        endLfnstIdx = endLfnstIdx ? 1 : 0;
2124
0
      }
2125
0
    }
2126
284k
    int ts_used = 0;
2127
284k
    bool testTS = false;
2128
284k
    if (partitioner.chType != CH_C)
2129
0
    {
2130
0
      startLfnstIdx = currTU.cu->lfnstIdx;
2131
0
      endLfnstIdx = currTU.cu->lfnstIdx;
2132
0
      bestLfnstIdx = currTU.cu->lfnstIdx;
2133
0
      testLFNST  = false;
2134
0
      rapidLFNST = false;
2135
0
      ts_used = currTU.mtsIdx[COMP_Y];
2136
0
    }
2137
284k
    if (cu.bdpcmM[CH_C])
2138
38.4k
    {
2139
38.4k
      endLfnstIdx = 0;
2140
38.4k
      testLFNST = false;
2141
38.4k
    }
2142
2143
284k
    double dSingleCostAll = MAX_DOUBLE;
2144
284k
    double singleCostTmpAll = 0;
2145
2146
1.04M
    for (int lfnstIdx = startLfnstIdx; lfnstIdx <= endLfnstIdx; lfnstIdx++)
2147
757k
    {
2148
757k
      if (rapidLFNST && lfnstIdx)
2149
0
      {
2150
0
        if ((lfnstIdx == 2) && (bestLfnstIdx == 0))
2151
0
        {
2152
0
          continue;
2153
0
        }
2154
0
      }
2155
2156
757k
      currTU.cu->lfnstIdx = lfnstIdx;
2157
757k
      if (lfnstIdx)
2158
472k
      {
2159
472k
        m_CABACEstimator->getCtx() = ctxStartTU;
2160
472k
      }
2161
2162
757k
      cuCtx.lfnstLastScanPos = false;
2163
757k
      cuCtx.violatesLfnstConstrained[CH_L] = false;
2164
757k
      cuCtx.violatesLfnstConstrained[CH_C] = false;
2165
2166
2.27M
      for (uint32_t c = COMP_Cb; c < numTBlocks; c++)
2167
1.51M
      {
2168
1.51M
        const ComponentID compID = ComponentID(c);
2169
1.51M
        const CompArea& area = currTU.blocks[compID];
2170
1.51M
        double     dSingleCost = MAX_DOUBLE;
2171
1.51M
        Distortion singleDistCTmp = 0;
2172
1.51M
        double     singleCostTmp = 0;
2173
1.51M
        bool tsAllowed = useTS && TU::isTSAllowed(currTU, compID) && m_pcEncCfg->m_useChromaTS && !currTU.cu->lfnstIdx && !cu.bdpcmM[CH_C];
2174
1.51M
        if ((partitioner.chType == CH_L) && (!ts_used))
2175
0
        {
2176
0
          tsAllowed = false;
2177
0
        }
2178
1.51M
        uint8_t nNumTransformCands = 1 + (tsAllowed ? 1 : 0); // DCT + TS = 2 tests       
2179
1.51M
        std::vector<TrMode> trModes;
2180
1.51M
        if (nNumTransformCands > 1)
2181
0
        {
2182
0
          trModes.push_back(TrMode(0, true));   // DCT2
2183
0
          trModes.push_back(TrMode(1, true));   // TS
2184
0
          testTS = true;
2185
0
        }
2186
1.51M
        bool cbfDCT2 = true;
2187
1.51M
        const bool isLastMode = testLFNST || cs.sps->jointCbCr ||  tsAllowed ? false : true;
2188
1.51M
        int bestModeId = 0;
2189
1.51M
        ctxStart = m_CABACEstimator->getCtx();
2190
3.02M
        for (int modeId = 0; modeId < nNumTransformCands; modeId++)
2191
1.51M
        {
2192
1.51M
          if (doReshaping || lfnstIdx || modeId)
2193
944k
          {
2194
944k
            resiCb.copyFrom(m_orgResiCb[0]);
2195
944k
            resiCr.copyFrom(m_orgResiCr[0]);
2196
944k
          }
2197
1.51M
          if (modeId == 0)
2198
1.51M
          {
2199
1.51M
            if ( tsAllowed)
2200
0
            {
2201
0
              xPreCheckMTS(currTU, &trModes, m_pcEncCfg->m_MTSIntraMaxCand, 0, compID);
2202
0
            }
2203
1.51M
          }
2204
2205
1.51M
          currTU.mtsIdx[compID] = currTU.cu->bdpcmM[CH_C] ? MTS_SKIP : modeId;
2206
2207
1.51M
          if (modeId)
2208
0
          {
2209
0
            if (!cbfDCT2 && trModes[modeId].first == MTS_SKIP)
2210
0
            {
2211
0
              break;
2212
0
            }
2213
0
            m_CABACEstimator->getCtx() = ctxStart;
2214
0
          }
2215
1.51M
          singleDistCTmp = 0;
2216
1.51M
          if (tsAllowed)
2217
0
          {
2218
0
            xIntraCodingTUBlock(currTU, compID, false, singleDistCTmp, 0, 0, true);
2219
0
            if ((modeId == 0) && (!trModes[modeId + 1].second))
2220
0
            {
2221
0
              nNumTransformCands = 1;
2222
0
            }
2223
0
          }
2224
1.51M
          else
2225
1.51M
        {
2226
1.51M
          xIntraCodingTUBlock(currTU, compID, false, singleDistCTmp);
2227
1.51M
        }
2228
1.51M
        if (((currTU.mtsIdx[compID] == MTS_SKIP && !currTU.cu->bdpcmM[CH_C])
2229
0
          && !TU::getCbf(currTU, compID)))   // In order not to code TS flag when cbf is zero, the case for TS with
2230
                                             // cbf being zero is forbidden.
2231
0
        {
2232
0
          singleCostTmp = MAX_DOUBLE;
2233
0
        }
2234
1.51M
        else
2235
1.51M
        {
2236
1.51M
          uint64_t fracBitsTmp = xGetIntraFracBitsQTChroma(currTU, compID, &cuCtx);
2237
1.51M
          singleCostTmp = m_pcRdCost->calcRdCost(fracBitsTmp, singleDistCTmp);
2238
1.51M
        }
2239
2240
1.51M
        if (singleCostTmp < dSingleCost)
2241
1.51M
        {
2242
1.51M
          dSingleCost = singleCostTmp;
2243
2244
1.51M
          if (compID == COMP_Cb)
2245
757k
          {
2246
757k
            bestCostCb = singleCostTmp;
2247
757k
            bestDistCb = singleDistCTmp;
2248
757k
          }
2249
757k
          else
2250
757k
          {
2251
757k
            bestCostCr = singleCostTmp;
2252
757k
            bestDistCr = singleDistCTmp;
2253
757k
          }
2254
1.51M
          bestModeId = modeId;
2255
1.51M
          if (currTU.mtsIdx[compID] == MTS_DCT2_DCT2)
2256
1.43M
          {
2257
1.43M
            cbfDCT2 = TU::getCbfAtDepth(currTU, compID, currDepth);
2258
1.43M
          }
2259
1.51M
          if (!isLastMode)
2260
1.51M
          {
2261
1.51M
            saveCS.getRecoBuf(area).copyFrom(cs.getRecoBuf(area));
2262
1.51M
            tmpTU.copyComponentFrom(currTU, compID);
2263
1.51M
            ctxBest = m_CABACEstimator->getCtx();
2264
1.51M
          }
2265
1.51M
        }
2266
1.51M
        }
2267
1.51M
        if (testTS && ((c == COMP_Cb && bestModeId < (nNumTransformCands - 1)) ))
2268
0
        {
2269
0
          m_CABACEstimator->getCtx() = ctxBest;
2270
2271
0
          currTU.copyComponentFrom(tmpTU, COMP_Cb); // Cbf of Cb is needed to estimate cost for Cr Cbf
2272
0
        }
2273
1.51M
      }
2274
2275
757k
      singleCostTmpAll = bestCostCb + bestCostCr;
2276
2277
757k
      bool rootCbfL = false;
2278
757k
      if (testLFNST)
2279
718k
      {
2280
2.87M
        for (uint32_t t = 0; t < getNumberValidTBlocks(*cs.pcv); t++)
2281
2.15M
        {
2282
2.15M
          rootCbfL |= bool(tmpTU.cbf[t]);
2283
2.15M
        }
2284
718k
        if (rapidLFNST && !rootCbfL)
2285
0
        {
2286
0
          endLfnstIdx = lfnstIdx; // end this
2287
0
        }
2288
718k
      }
2289
2290
757k
      if (testLFNST && lfnstIdx && !cuCtx.lfnstLastScanPos)
2291
311k
      {
2292
311k
        bool cbfAtZeroDepth = CU::isSepTree(*currTU.cu)
2293
311k
          ? rootCbfL : (cs.area.chromaFormat != CHROMA_400
2294
0
            && std::min(tmpTU.blocks[1].width, tmpTU.blocks[1].height) < 4)
2295
1
          ? TU::getCbfAtDepth(currTU, COMP_Y, currTU.depth) : rootCbfL;
2296
311k
        if (cbfAtZeroDepth)
2297
1.56k
        {
2298
1.56k
          singleCostTmpAll = MAX_DOUBLE;
2299
1.56k
        }
2300
311k
      }
2301
757k
      if ((testLFNST || testTS) && (singleCostTmpAll < dSingleCostAll))
2302
246k
      {
2303
246k
        bestLfnstIdx = lfnstIdx;
2304
246k
        if ((lfnstIdx != endLfnstIdx) || testTS)
2305
236k
        {
2306
236k
          dSingleCostAll = singleCostTmpAll;
2307
2308
236k
          bestCostCbcur = bestCostCb;
2309
236k
          bestCostCrcur = bestCostCr;
2310
236k
          bestDistCbcur = bestDistCb;
2311
236k
          bestDistCrcur = bestDistCr;
2312
2313
236k
          saveCScur.getRecoBuf(cbArea).copyFrom(saveCS.getRecoBuf(cbArea));
2314
236k
          saveCScur.getRecoBuf(crArea).copyFrom(saveCS.getRecoBuf(crArea));
2315
2316
236k
          tmpTUcur.copyComponentFrom(tmpTU, COMP_Cb);
2317
236k
          tmpTUcur.copyComponentFrom(tmpTU, COMP_Cr);
2318
236k
        }
2319
246k
        ctxBestTUL = m_CABACEstimator->getCtx();
2320
246k
      }
2321
757k
    }
2322
284k
    if ((testLFNST && (bestLfnstIdx != endLfnstIdx)) || testTS)
2323
236k
    {
2324
236k
      bestCostCb = bestCostCbcur;
2325
236k
      bestCostCr = bestCostCrcur;
2326
236k
      bestDistCb = bestDistCbcur;
2327
236k
      bestDistCr = bestDistCrcur;
2328
236k
      currTU.cu->lfnstIdx = bestLfnstIdx;
2329
236k
      if (!cs.sps->jointCbCr)
2330
0
      {
2331
0
        cs.getRecoBuf(cbArea).copyFrom(saveCScur.getRecoBuf(cbArea));
2332
0
        cs.getRecoBuf(crArea).copyFrom(saveCScur.getRecoBuf(crArea));
2333
2334
0
        currTU.copyComponentFrom(tmpTUcur, COMP_Cb);
2335
0
        currTU.copyComponentFrom(tmpTUcur, COMP_Cr);
2336
2337
0
        m_CABACEstimator->getCtx() = ctxBestTUL;
2338
0
      }
2339
236k
    }
2340
2341
284k
    Distortion bestDistCbCr = bestDistCb + bestDistCr;
2342
2343
284k
    if (cs.sps->jointCbCr)
2344
284k
    {
2345
284k
      if ((testLFNST && (bestLfnstIdx != endLfnstIdx)) || testTS)
2346
236k
      {
2347
236k
        saveCS.getRecoBuf(cbArea).copyFrom(saveCScur.getRecoBuf(cbArea));
2348
236k
        saveCS.getRecoBuf(crArea).copyFrom(saveCScur.getRecoBuf(crArea));
2349
2350
236k
        tmpTU.copyComponentFrom(tmpTUcur, COMP_Cb);
2351
236k
        tmpTU.copyComponentFrom(tmpTUcur, COMP_Cr);
2352
236k
        m_CABACEstimator->getCtx() = ctxBestTUL;
2353
236k
        ctxBest = m_CABACEstimator->getCtx();
2354
236k
      }
2355
      // Test using joint chroma residual coding
2356
284k
      double     bestCostCbCr = bestCostCb + bestCostCr;
2357
284k
      int        bestJointCbCr = 0;
2358
284k
      bool checkDCTOnly = m_pcEncCfg->m_useChromaTS && ((TU::getCbf(tmpTU, COMP_Cb) && tmpTU.mtsIdx[COMP_Cb] == MTS_DCT2_DCT2 && !TU::getCbf(tmpTU, COMP_Cr)) ||
2359
0
        (TU::getCbf(tmpTU, COMP_Cr) && tmpTU.mtsIdx[COMP_Cr] == MTS_DCT2_DCT2 && !TU::getCbf(tmpTU, COMP_Cb)) ||
2360
0
        (TU::getCbf(tmpTU, COMP_Cb) && tmpTU.mtsIdx[COMP_Cb] == MTS_DCT2_DCT2 && TU::getCbf(tmpTU, COMP_Cr) && tmpTU.mtsIdx[COMP_Cr] == MTS_DCT2_DCT2));
2361
284k
      bool checkTSOnly = m_pcEncCfg->m_useChromaTS && ((TU::getCbf(tmpTU, COMP_Cb) && tmpTU.mtsIdx[COMP_Cb] == MTS_SKIP && !TU::getCbf(tmpTU, COMP_Cr)) ||
2362
0
        (TU::getCbf(tmpTU, COMP_Cr) && tmpTU.mtsIdx[COMP_Cr] == MTS_SKIP && !TU::getCbf(tmpTU, COMP_Cb)) ||
2363
0
        (TU::getCbf(tmpTU, COMP_Cb) && tmpTU.mtsIdx[COMP_Cb] == MTS_SKIP && TU::getCbf(tmpTU, COMP_Cr) && tmpTU.mtsIdx[COMP_Cr] == MTS_SKIP));
2364
284k
      bool       lastIsBest = false;
2365
284k
      bool noLFNST1 = false;
2366
284k
      if (rapidLFNST && (startLfnstIdx != endLfnstIdx))
2367
0
      {
2368
0
        if (bestLfnstIdx == 2)
2369
0
        {
2370
0
          noLFNST1 = true;
2371
0
        }
2372
0
        else
2373
0
        {
2374
0
          endLfnstIdx = 1;
2375
0
        }
2376
0
      }
2377
2378
1.04M
      for (int lfnstIdxj = startLfnstIdx; lfnstIdxj <= endLfnstIdx; lfnstIdxj++)
2379
757k
      {
2380
757k
        if (rapidLFNST && noLFNST1 && (lfnstIdxj == 1))
2381
0
        {
2382
0
          continue;
2383
0
        }
2384
757k
        currTU.cu->lfnstIdx = lfnstIdxj;
2385
757k
        std::vector<int> jointCbfMasksToTest;
2386
757k
        if (TU::getCbf(tmpTU, COMP_Cb) || TU::getCbf(tmpTU, COMP_Cr))
2387
267k
        {
2388
267k
          jointCbfMasksToTest = m_pcTrQuant->selectICTCandidates(currTU, m_orgResiCb, m_orgResiCr);
2389
267k
        }
2390
757k
        for (int cbfMask : jointCbfMasksToTest)
2391
267k
        {
2392
267k
          currTU.jointCbCr = (uint8_t)cbfMask;
2393
267k
          ComponentID codeCompId = ((currTU.jointCbCr >> 1) ? COMP_Cb : COMP_Cr);
2394
267k
          ComponentID otherCompId = ((codeCompId == COMP_Cb) ? COMP_Cr : COMP_Cb);
2395
267k
          bool tsAllowed = useTS && TU::isTSAllowed(currTU, codeCompId) && (m_pcEncCfg->m_useChromaTS) && !currTU.cu->lfnstIdx && !cu.bdpcmM[CH_C];
2396
267k
          if ((partitioner.chType == CH_L)&& tsAllowed && (currTU.mtsIdx[COMP_Y] != MTS_SKIP))
2397
0
          {
2398
0
            tsAllowed = false;
2399
0
          }
2400
267k
          if (!tsAllowed)
2401
267k
          {
2402
267k
            checkTSOnly = false;
2403
267k
          }
2404
267k
          uint8_t     numTransformCands = 1 + (tsAllowed && !(checkDCTOnly || checkTSOnly)? 1 : 0); // DCT + TS = 2 tests
2405
267k
          std::vector<TrMode> trModes;
2406
267k
          if (numTransformCands > 1)
2407
0
          {
2408
0
            trModes.push_back(TrMode(0, true)); // DCT2
2409
0
            trModes.push_back(TrMode(1, true));//TS
2410
0
          }
2411
267k
          else
2412
267k
          {
2413
267k
            currTU.mtsIdx[codeCompId] = checkTSOnly || currTU.cu->bdpcmM[CH_C] ? 1 : 0;
2414
267k
          }
2415
2416
535k
          for (int modeId = 0; modeId < numTransformCands; modeId++)
2417
267k
          {
2418
267k
            Distortion distTmp = 0;
2419
267k
            currTU.mtsIdx[codeCompId] = currTU.cu->bdpcmM[CH_C] ? MTS_SKIP : MTS_DCT2_DCT2;
2420
267k
            if (numTransformCands > 1)
2421
0
            {
2422
0
              currTU.mtsIdx[codeCompId] = currTU.cu->bdpcmM[CH_C] ? MTS_SKIP : trModes[modeId].first;
2423
0
            }
2424
267k
            currTU.mtsIdx[otherCompId] = MTS_DCT2_DCT2;
2425
2426
267k
            m_CABACEstimator->getCtx() = ctxStartTU;
2427
2428
267k
            resiCb.copyFrom(m_orgResiCb[cbfMask]);
2429
267k
            resiCr.copyFrom(m_orgResiCr[cbfMask]);
2430
267k
            if ((modeId == 0) && (numTransformCands > 1))
2431
0
            {
2432
0
              xPreCheckMTS(currTU, &trModes, m_pcEncCfg->m_MTSIntraMaxCand, 0, COMP_Cb);
2433
0
              currTU.mtsIdx[codeCompId] = trModes[modeId].first;
2434
0
              currTU.mtsIdx[(codeCompId == COMP_Cr) ? COMP_Cb : COMP_Cr] = MTS_DCT2_DCT2;
2435
0
            }
2436
267k
            cuCtx.lfnstLastScanPos = false;
2437
267k
            cuCtx.violatesLfnstConstrained[CH_L] = false;
2438
267k
            cuCtx.violatesLfnstConstrained[CH_C] = false;
2439
267k
            if (numTransformCands > 1)
2440
0
            {
2441
0
              xIntraCodingTUBlock(currTU, COMP_Cb, false, distTmp, 0, 0, true);
2442
0
              if ((modeId == 0) && !trModes[modeId + 1].second)
2443
0
              {
2444
0
                numTransformCands = 1;
2445
0
              }
2446
0
            }
2447
267k
            else
2448
267k
            {
2449
267k
              xIntraCodingTUBlock(currTU, COMP_Cb, false, distTmp, 0);
2450
267k
            }
2451
2452
267k
            double costTmp = std::numeric_limits<double>::max();
2453
267k
            if (distTmp < MAX_DISTORTION)
2454
264k
            {
2455
264k
              uint64_t bits = xGetIntraFracBitsQTChroma(currTU, COMP_Cb, &cuCtx);
2456
264k
              costTmp = m_pcRdCost->calcRdCost(bits, distTmp);
2457
264k
            }
2458
3.51k
            else if (!currTU.mtsIdx[codeCompId])
2459
3.51k
            {
2460
3.51k
              numTransformCands = 1;
2461
3.51k
            }
2462
267k
            bool rootCbfL = false;
2463
1.07M
            for (uint32_t t = 0; t < getNumberValidTBlocks(*cs.pcv); t++)
2464
803k
            {
2465
803k
              rootCbfL |= bool(tmpTU.cbf[t]);
2466
803k
            }
2467
267k
            if (rapidLFNST && !rootCbfL)
2468
0
            {
2469
0
              endLfnstIdx = lfnstIdxj;
2470
0
            }
2471
267k
            if (testLFNST && currTU.cu->lfnstIdx && !cuCtx.lfnstLastScanPos)
2472
3.47k
            {
2473
3.47k
              bool cbfAtZeroDepth = CU::isSepTree(*currTU.cu) ? rootCbfL
2474
3.47k
                : (cs.area.chromaFormat != CHROMA_400 && std::min(tmpTU.blocks[1].width, tmpTU.blocks[1].height) < 4)
2475
0
                ? TU::getCbfAtDepth(currTU, COMP_Y, currTU.depth) : rootCbfL;
2476
3.47k
              if (cbfAtZeroDepth)
2477
3.47k
              {
2478
3.47k
                costTmp = MAX_DOUBLE;
2479
3.47k
              }
2480
3.47k
            }
2481
267k
            if (costTmp < bestCostCbCr)
2482
100k
            {
2483
100k
              bestCostCbCr = costTmp;
2484
100k
              bestDistCbCr = distTmp;
2485
100k
              bestJointCbCr = currTU.jointCbCr;
2486
2487
              // store data
2488
100k
              bestLfnstIdx = lfnstIdxj;
2489
100k
              if ((cbfMask != jointCbfMasksToTest.back() || (lfnstIdxj != endLfnstIdx)) || (modeId != (numTransformCands - 1)))
2490
82.1k
              {
2491
82.1k
                saveCS.getRecoBuf(cbArea).copyFrom(cs.getRecoBuf(cbArea));
2492
82.1k
                saveCS.getRecoBuf(crArea).copyFrom(cs.getRecoBuf(crArea));
2493
2494
82.1k
                tmpTU.copyComponentFrom(currTU, COMP_Cb);
2495
82.1k
                tmpTU.copyComponentFrom(currTU, COMP_Cr);
2496
2497
82.1k
                ctxBest = m_CABACEstimator->getCtx();
2498
82.1k
              }
2499
18.6k
              else
2500
18.6k
              {
2501
18.6k
                lastIsBest = true;
2502
18.6k
                cs.cus[0]->lfnstIdx = bestLfnstIdx;
2503
18.6k
              }
2504
100k
            }
2505
267k
          }
2506
267k
        }
2507
2508
        // Retrieve the best CU data (unless it was the very last one tested)
2509
757k
      }
2510
284k
      if (!lastIsBest)
2511
266k
      {
2512
266k
        cs.getRecoBuf(cbArea).copyFrom(saveCS.getRecoBuf(cbArea));
2513
266k
        cs.getRecoBuf(crArea).copyFrom(saveCS.getRecoBuf(crArea));
2514
2515
266k
        cs.cus[0]->lfnstIdx = bestLfnstIdx;
2516
266k
        currTU.copyComponentFrom(tmpTU, COMP_Cb);
2517
266k
        currTU.copyComponentFrom(tmpTU, COMP_Cr);
2518
266k
        m_CABACEstimator->getCtx() = ctxBest;
2519
266k
      }
2520
284k
      currTU.jointCbCr = (TU::getCbf(currTU, COMP_Cb) || TU::getCbf(currTU, COMP_Cr)) ? bestJointCbCr : 0;
2521
284k
    } // jointCbCr
2522
2523
284k
    cs.dist += bestDistCbCr;
2524
284k
    cuCtx.violatesLfnstConstrained[CH_L] = false;
2525
284k
    cuCtx.violatesLfnstConstrained[CH_C] = false;
2526
284k
    cuCtx.lfnstLastScanPos = false;
2527
284k
    cuCtx.violatesMtsCoeffConstraint = false;
2528
284k
    cuCtx.mtsLastScanPos = false;
2529
284k
    cbfs.cbf(COMP_Cb) = TU::getCbf(currTU, COMP_Cb);
2530
284k
    cbfs.cbf(COMP_Cr) = TU::getCbf(currTU, COMP_Cr);
2531
284k
  }
2532
2
  else
2533
2
  {
2534
2
    unsigned   numValidTBlocks = getNumberValidTBlocks(*cs.pcv);
2535
2
    ChromaCbfs SplitCbfs(false);
2536
2537
2
    if (partitioner.canSplit(TU_MAX_TR_SPLIT, cs))
2538
0
    {
2539
0
      partitioner.splitCurrArea(TU_MAX_TR_SPLIT, cs);
2540
0
    }
2541
2
    else if (currTU.cu->ispMode)
2542
0
    {
2543
0
      partitioner.splitCurrArea(m_ispTestedModes[0].IspType, cs);
2544
0
    }
2545
2
    else
2546
2
      THROW("Implicit TU split not available");
2547
2548
0
    do
2549
0
    {
2550
0
      ChromaCbfs subCbfs = xIntraChromaCodingQT(cs, partitioner);
2551
2552
0
      for (uint32_t ch = COMP_Cb; ch < numValidTBlocks; ch++)
2553
0
      {
2554
0
        const ComponentID compID = ComponentID(ch);
2555
0
        SplitCbfs.cbf(compID) |= subCbfs.cbf(compID);
2556
0
      }
2557
0
    } while (partitioner.nextPart(cs));
2558
2559
0
    partitioner.exitCurrSplit();
2560
2561
    /*if (lumaUsesISP && cs.dist == MAX_UINT) //ahenkel
2562
    {
2563
      return cbfs;
2564
    }*/
2565
0
    {
2566
0
      cbfs.Cb |= SplitCbfs.Cb;
2567
0
      cbfs.Cr |= SplitCbfs.Cr;
2568
2569
0
      if (1)   //(!lumaUsesISP)
2570
0
      {
2571
0
        for (auto& ptu : cs.tus)
2572
0
        {
2573
0
          if (currArea.Cb().contains(ptu->Cb()) || (!ptu->Cb().valid() && currArea.Y().contains(ptu->Y())))
2574
0
          {
2575
0
            TU::setCbfAtDepth(*ptu, COMP_Cb, currDepth, SplitCbfs.Cb);
2576
0
            TU::setCbfAtDepth(*ptu, COMP_Cr, currDepth, SplitCbfs.Cr);
2577
0
          }
2578
0
        }
2579
0
      }
2580
0
    }
2581
0
  }
2582
284k
  return cbfs;
2583
284k
}
2584
2585
uint64_t IntraSearch::xFracModeBitsIntraLuma(const CodingUnit& cu, const unsigned* mpmLst)
2586
951k
{
2587
951k
  m_CABACEstimator->resetBits();
2588
2589
951k
  if (!cu.ciip)
2590
951k
  {
2591
951k
    m_CABACEstimator->intra_luma_pred_mode(cu, mpmLst);
2592
951k
  }
2593
2594
951k
  return m_CABACEstimator->getEstFracBits();
2595
951k
}
2596
2597
template<typename T, size_t N, int M>
2598
void IntraSearch::xReduceHadCandList(static_vector<T, N>& candModeList, static_vector<double, N>& candCostList, SortedPelUnitBufs<M>& sortedPelBuffer, int& numModesForFullRD, const double thresholdHadCost, const double* mipHadCost, const CodingUnit& cu, const bool fastMip)
2599
19.3k
{
2600
19.3k
  const int maxCandPerType = numModesForFullRD >> 1;
2601
19.3k
  static_vector<ModeInfo, FAST_UDI_MAX_RDMODE_NUM> tempRdModeList;
2602
19.3k
  static_vector<double, FAST_UDI_MAX_RDMODE_NUM> tempCandCostList;
2603
19.3k
  const double minCost = candCostList[0];
2604
19.3k
  bool keepOneMip = candModeList.size() > numModesForFullRD;
2605
19.3k
  const int maxNumConv = 3; 
2606
2607
19.3k
  int numConv = 0;
2608
19.3k
  int numMip = 0;
2609
87.5k
  for (int idx = 0; idx < candModeList.size() - (keepOneMip?0:1); idx++)
2610
68.2k
  {
2611
68.2k
    bool addMode = false;
2612
68.2k
    const ModeInfo& orgMode = candModeList[idx];
2613
2614
68.2k
    if (!orgMode.mipFlg)
2615
48.8k
    {
2616
48.8k
      addMode = (numConv < maxNumConv);
2617
48.8k
      numConv += addMode ? 1:0;
2618
48.8k
    }
2619
19.3k
    else
2620
19.3k
    {
2621
19.3k
      addMode = ( numMip < maxCandPerType || (candCostList[idx] < thresholdHadCost * minCost) || keepOneMip );
2622
19.3k
      keepOneMip = false;
2623
19.3k
      numMip += addMode ? 1:0;
2624
19.3k
    }
2625
68.2k
    if( addMode )
2626
68.2k
    {
2627
68.2k
      tempRdModeList.push_back(orgMode);
2628
68.2k
      tempCandCostList.push_back(candCostList[idx]);
2629
68.2k
    }
2630
68.2k
  }
2631
2632
  // sort Pel Buffer
2633
19.3k
  int i = -1;
2634
19.3k
  for( auto &m: tempRdModeList)
2635
68.2k
  {
2636
68.2k
    if( ! (m == candModeList.at( ++i )) )
2637
0
    {
2638
0
      for( int j = i; j < (int)candModeList.size()-1; )
2639
0
      {
2640
0
        if( m == candModeList.at( ++j ) )
2641
0
        {
2642
0
          sortedPelBuffer.swap( i, j);
2643
0
          break;
2644
0
        }
2645
0
      }
2646
0
    }
2647
68.2k
  }
2648
19.3k
  sortedPelBuffer.reduceTo( (int)tempRdModeList.size() );
2649
2650
19.3k
  if ((cu.lwidth() > 8 && cu.lheight() > 8))
2651
17.3k
  {
2652
    // Sort MIP candidates by Hadamard cost
2653
17.3k
    const int transpOff = getNumModesMip(cu.Y());
2654
17.3k
    static_vector<uint8_t, FAST_UDI_MAX_RDMODE_NUM> sortedMipModes(0);
2655
17.3k
    static_vector<double, FAST_UDI_MAX_RDMODE_NUM> sortedMipCost(0);
2656
17.3k
    for (uint8_t mode : { 0, 1, 2 })
2657
51.9k
    {
2658
51.9k
      uint8_t candMode = mode + uint8_t((mipHadCost[mode + transpOff] < mipHadCost[mode]) ? transpOff : 0);
2659
51.9k
      updateCandList(candMode, mipHadCost[candMode], sortedMipModes, sortedMipCost, 3);
2660
51.9k
    }
2661
2662
    // Append MIP mode to RD mode list
2663
17.3k
    const int modeListSize = int(tempRdModeList.size());
2664
34.6k
    for (int idx = 0; idx < 3; idx++)
2665
34.6k
    {
2666
34.6k
      const bool     isTransposed = (sortedMipModes[idx] >= transpOff ? true : false);
2667
34.6k
      const uint32_t mipIdx       = (isTransposed ? sortedMipModes[idx] - transpOff : sortedMipModes[idx]);
2668
34.6k
      const ModeInfo mipMode( true, isTransposed, 0, NOT_INTRA_SUBPARTITIONS, mipIdx );
2669
34.6k
      bool alreadyIncluded = false;
2670
138k
      for (int modeListIdx = 0; modeListIdx < modeListSize; modeListIdx++)
2671
120k
      {
2672
120k
        if (tempRdModeList[modeListIdx] == mipMode)
2673
17.3k
        {
2674
17.3k
          alreadyIncluded = true;
2675
17.3k
          break;
2676
17.3k
        }
2677
120k
      }
2678
2679
34.6k
      if (!alreadyIncluded)
2680
17.3k
      {
2681
17.3k
        tempRdModeList.push_back(mipMode);
2682
17.3k
        tempCandCostList.push_back(0);
2683
17.3k
        if( fastMip ) break;
2684
17.3k
      }
2685
34.6k
    }
2686
17.3k
  }
2687
2688
19.3k
  candModeList = tempRdModeList;
2689
19.3k
  candCostList = tempCandCostList;
2690
19.3k
  numModesForFullRD = int(candModeList.size());
2691
19.3k
}
2692
2693
void IntraSearch::xPreCheckMTS(TransformUnit &tu, std::vector<TrMode> *trModes, const int maxCand, PelUnitBuf *predBuf, const ComponentID& compID)
2694
14.9k
{
2695
14.9k
  if (compID == COMP_Y)
2696
14.9k
  {
2697
14.9k
    CodingStructure&  cs = *tu.cs;
2698
14.9k
    const CompArea& area = tu.blocks[compID];
2699
14.9k
    const ReshapeData& reshapeData = cs.picture->reshapeData;
2700
14.9k
    const CodingUnit& cu = *cs.getCU(area.pos(), CH_L,TREE_D);
2701
14.9k
    PelBuf piPred = cs.getPredBuf(area);
2702
14.9k
    PelBuf piResi = cs.getResiBuf(area);
2703
2704
14.9k
    initIntraPatternChType(*tu.cu, area);
2705
14.9k
    if (predBuf)
2706
13.3k
    {
2707
13.3k
      piPred.copyFrom(predBuf->Y());
2708
13.3k
    }
2709
1.56k
    else if (CU::isMIP(cu, CH_L))
2710
1.54k
    {
2711
1.54k
      initIntraMip(cu);
2712
1.54k
      predIntraMip(piPred, cu);
2713
1.54k
    }
2714
24
    else
2715
24
    {
2716
24
      predIntraAng(COMP_Y, piPred, cu);
2717
24
    }
2718
2719
    //===== get residual signal =====
2720
14.9k
    if (cs.picHeader->lmcsEnabled && reshapeData.getCTUFlag())
2721
0
    {
2722
0
      piResi.subtract(cs.getRspOrgBuf(), piPred);
2723
0
    }
2724
14.9k
    else
2725
14.9k
    {
2726
14.9k
      CPelBuf piOrg = cs.getOrgBuf(COMP_Y);
2727
14.9k
      piResi.subtract(piOrg, piPred);
2728
14.9k
    }
2729
14.9k
    m_pcTrQuant->checktransformsNxN(tu, trModes, m_pcEncCfg->m_MTSIntraMaxCand, compID);
2730
14.9k
  }
2731
0
  else
2732
0
  {
2733
0
    ComponentID codeCompId = (tu.jointCbCr ? (tu.jointCbCr >> 1 ? COMP_Cb : COMP_Cr) : compID);
2734
0
    m_pcTrQuant->checktransformsNxN(tu, trModes, m_pcEncCfg->m_MTSIntraMaxCand, codeCompId);
2735
0
  }
2736
14.9k
}
2737
2738
double IntraSearch::xTestISP(CodingStructure& cs, Partitioner& subTuPartitioner, double bestCostForISP, PartSplit ispType, bool& splitcbf, uint64_t& singleFracBits, Distortion& singleDistLuma, CUCtx& cuCtx)
2739
15.4k
{
2740
15.4k
  int  subTuCounter = 0;
2741
15.4k
  bool earlySkipISP = false;
2742
15.4k
  bool splitCbfLuma = false;
2743
15.4k
  CodingUnit& cu = *cs.cus[0];
2744
2745
15.4k
  Distortion singleDistTmpLumaSUM = 0;
2746
15.4k
  uint64_t   singleTmpFracBitsSUM = 0;
2747
15.4k
  double     singleCostTmpSUM = 0;
2748
15.4k
  cuCtx.isDQPCoded = true;
2749
15.4k
  cuCtx.isChromaQpAdjCoded = true;
2750
2751
15.4k
  do
2752
19.6k
  {
2753
19.6k
    Distortion singleDistTmpLuma = 0;
2754
19.6k
    uint64_t   singleTmpFracBits = 0;
2755
19.6k
    double     singleCostTmp = 0;
2756
19.6k
    TransformUnit& tmpTUcur = ((cs.tus.size() < (subTuCounter + 1)))
2757
19.6k
      ? cs.addTU(CS::getArea(cs, subTuPartitioner.currArea(), subTuPartitioner.chType,
2758
3.53k
        subTuPartitioner.treeType),
2759
3.53k
        subTuPartitioner.chType, cs.cus[0])
2760
19.6k
      : *cs.tus[subTuCounter];
2761
19.6k
    tmpTUcur.depth = subTuPartitioner.currTrDepth;
2762
2763
    // Encode TU
2764
19.6k
    xIntraCodingTUBlock(tmpTUcur, COMP_Y, false, singleDistTmpLuma, 0);
2765
19.6k
    cuCtx.mtsLastScanPos = false;
2766
2767
19.6k
    if (singleDistTmpLuma == MAX_INT)   // all zero CBF skip
2768
0
    {
2769
0
      earlySkipISP = true;
2770
0
      singleCostTmpSUM = MAX_DOUBLE;
2771
0
      break;
2772
0
    }
2773
2774
19.6k
    if (m_pcRdCost->calcRdCost(singleTmpFracBitsSUM, singleDistTmpLumaSUM + singleDistTmpLuma) > bestCostForISP)
2775
5.19k
    {
2776
5.19k
      earlySkipISP = true;
2777
5.19k
    }
2778
14.4k
    else
2779
14.4k
    {
2780
14.4k
      m_ispTestedModes[0].IspType = ispType;
2781
14.4k
      m_ispTestedModes[0].subTuCounter = subTuCounter;
2782
14.4k
      singleTmpFracBits = xGetIntraFracBitsQT(cs, subTuPartitioner, true, &cuCtx);
2783
14.4k
    }
2784
19.6k
    singleCostTmp = m_pcRdCost->calcRdCost(singleTmpFracBits, singleDistTmpLuma);
2785
2786
19.6k
    singleCostTmpSUM     += singleCostTmp;
2787
19.6k
    singleDistTmpLumaSUM += singleDistTmpLuma;
2788
19.6k
    singleTmpFracBitsSUM += singleTmpFracBits;
2789
2790
19.6k
    subTuCounter++;
2791
2792
19.6k
    splitCbfLuma |= TU::getCbfAtDepth( *cs.getTU(subTuPartitioner.currArea().lumaPos(), subTuPartitioner.chType, subTuCounter - 1), 
2793
19.6k
                                       COMP_Y, subTuPartitioner.currTrDepth);
2794
19.6k
    int nSubPartitions = m_ispTestedModes[cu.lfnstIdx].numTotalParts[cu.ispMode - 1];
2795
19.6k
    bool doStop = (m_pcEncCfg->m_ISP != 1) || (subTuCounter < nSubPartitions);
2796
19.6k
    if (doStop)
2797
19.6k
    {
2798
19.6k
      if (singleCostTmpSUM > bestCostForISP)
2799
12.9k
      {
2800
12.9k
        earlySkipISP = true;
2801
12.9k
        break;
2802
12.9k
      }
2803
6.69k
      if (subTuCounter < nSubPartitions)
2804
5.34k
      {
2805
5.34k
        double threshold = nSubPartitions == 2 ? 0.95 : subTuCounter == 1 ? 0.83 : 0.91;
2806
5.34k
        if (singleCostTmpSUM > bestCostForISP * threshold)
2807
1.11k
        {
2808
1.11k
          earlySkipISP = true;
2809
1.11k
          break;
2810
1.11k
        }
2811
5.34k
      }
2812
6.69k
    }
2813
19.6k
  } while (subTuPartitioner.nextPart(cs));
2814
15.4k
  singleDistLuma = singleDistTmpLumaSUM;
2815
15.4k
  singleFracBits = singleTmpFracBitsSUM;
2816
2817
15.4k
  splitcbf = splitCbfLuma;
2818
15.4k
  return earlySkipISP ? MAX_DOUBLE : singleCostTmpSUM;
2819
15.4k
}
2820
2821
int IntraSearch::xSpeedUpISP(int speed, bool& testISP, int mode, int& noISP, int& endISP, CodingUnit& cu, static_vector<ModeInfo, FAST_UDI_MAX_RDMODE_NUM>& RdModeList, const ModeInfo& bestPUMode, int bestISP, int bestLfnstIdx)
2822
13.9k
{
2823
13.9k
  if (speed)
2824
5.65k
  {
2825
5.65k
    if (mode >= 1)
2826
2.97k
    {
2827
2.97k
      if (m_ispTestedModes[0].splitIsFinished[1] && m_ispTestedModes[0].splitIsFinished[0])
2828
0
      {
2829
0
        testISP = false;
2830
0
        endISP = 0;
2831
0
      }
2832
2.97k
      else
2833
2.97k
      {
2834
2.97k
        if (m_pcEncCfg->m_ISP >= 2)
2835
2.97k
        {
2836
2.97k
          if (mode == 1) //best Hor||Ver
2837
2.68k
          {
2838
2.68k
            int bestDir = 0;
2839
8.04k
            for (int d = 0; d < 2; d++)
2840
5.36k
            {
2841
5.36k
              int d2 = d ? 0 : 1;
2842
5.36k
              if ((m_ispTestedModes[0].bestCost[d] <= m_ispTestedModes[0].bestCost[d2])
2843
5.07k
                && (m_ispTestedModes[0].bestCost[d] != MAX_DOUBLE))
2844
290
              {
2845
290
                bestDir = d + 1;
2846
290
                m_ispTestedModes[0].splitIsFinished[d2] = true;
2847
290
              }
2848
5.36k
            }
2849
2.68k
            m_ispTestedModes[0].bestModeSoFar = bestDir;
2850
2.68k
            if (m_ispTestedModes[0].bestModeSoFar <= 0)
2851
2.39k
            {
2852
2.39k
              m_ispTestedModes[0].splitIsFinished[1] = true;
2853
2.39k
              m_ispTestedModes[0].splitIsFinished[0] = true;
2854
2.39k
              testISP = false;
2855
2.39k
              endISP = 0;
2856
2.39k
            }
2857
2.68k
          }
2858
2.97k
          if (m_ispTestedModes[0].bestModeSoFar == 2)
2859
64
          {
2860
64
            noISP = 1;
2861
64
          }
2862
2.90k
          else
2863
2.90k
          {
2864
2.90k
            endISP = 1;
2865
2.90k
          }
2866
2.97k
        }
2867
2.97k
      }
2868
2.97k
    }
2869
5.65k
    if (testISP)
2870
3.26k
    {
2871
3.26k
      if (mode == 2)
2872
290
      {
2873
870
        for (int d = 0; d < 2; d++)
2874
580
        {
2875
580
          int d2 = d ? 0 : 1;
2876
580
          if (m_ispTestedModes[0].bestCost[d] == MAX_DOUBLE)
2877
271
          {
2878
271
            m_ispTestedModes[0].splitIsFinished[d] = true;
2879
271
          }
2880
580
          if ((m_ispTestedModes[0].bestCost[d2] < 1.3 * m_ispTestedModes[0].bestCost[d])
2881
309
            && (int(m_ispTestedModes[0].bestSplitSoFar) != (d + 1)))
2882
235
          {
2883
235
            if (d)
2884
203
            {
2885
203
              endISP = 1;
2886
203
            }
2887
32
            else
2888
32
            {
2889
32
              noISP = 1;
2890
32
            }
2891
235
            m_ispTestedModes[0].splitIsFinished[d] = true;
2892
235
          }
2893
580
        }
2894
290
      }
2895
2.97k
      else
2896
2.97k
      {
2897
2.97k
        if (m_ispTestedModes[0].splitIsFinished[0])
2898
32
        {
2899
32
          noISP = 1;
2900
32
        }
2901
2.97k
        if (m_ispTestedModes[0].splitIsFinished[1])
2902
258
        {
2903
258
          endISP = 1;
2904
258
        }
2905
2.97k
      }
2906
3.26k
    }
2907
5.65k
    if ((noISP == 1) && (endISP == 1))
2908
19
    {
2909
19
      endISP = 0;
2910
19
    }
2911
5.65k
  }
2912
8.30k
  else
2913
8.30k
  {
2914
8.30k
    bool stopFound = false;
2915
8.30k
    if (m_pcEncCfg->m_ISP >= 3)
2916
8.30k
    {
2917
8.30k
      if (mode)
2918
2.95k
      {
2919
2.95k
        if ((bestISP == 0) || ((bestPUMode.modeId != RdModeList[mode - 1].modeId)
2920
93
          && (bestPUMode.modeId != RdModeList[mode].modeId)))
2921
2.04k
        {
2922
2.04k
          stopFound = true;
2923
2.04k
        }
2924
2.95k
      }
2925
8.30k
    }
2926
8.30k
    if (cu.mipFlag || cu.multiRefIdx)
2927
180
    {
2928
180
      cu.mipFlag = false;
2929
180
      cu.multiRefIdx = 0;
2930
180
      if (!stopFound)
2931
0
      {
2932
0
        for (int k = 0; k < mode; k++)
2933
0
        {
2934
0
          if (cu.intraDir[CH_L] == RdModeList[k].modeId)
2935
0
          {
2936
0
            stopFound = true;
2937
0
            break;
2938
0
          }
2939
0
        }
2940
0
      }
2941
180
    }
2942
8.30k
    if (stopFound)
2943
2.04k
    {
2944
2.04k
      testISP = false;
2945
2.04k
      endISP = 0;
2946
2.04k
      return 1;
2947
2.04k
    }
2948
6.26k
    if (!stopFound && (m_pcEncCfg->m_ISP >= 2) && (cu.intraDir[CH_L] == DC_IDX))
2949
913
    {
2950
913
      stopFound = true;
2951
913
      endISP = 0;
2952
913
      return 1;
2953
913
    }
2954
6.26k
  }
2955
11.0k
  return 0;
2956
13.9k
}
2957
2958
void IntraSearch::xSpeedUpIntra(double bestcost, int& EndMode, int& speedIntra, CodingUnit& cu)
2959
25.4k
{
2960
25.4k
  int bestIdxbefore = m_ispTestedModes[0].bestIntraMode;
2961
25.4k
  if (m_ispTestedModes[0].isIntra)
2962
0
  {
2963
0
    if (bestIdxbefore == 1)//ISP
2964
0
    {
2965
0
      speedIntra = 14;
2966
0
    }
2967
0
    if (bestIdxbefore == 4)//MTS
2968
0
    {
2969
0
      speedIntra = 3;
2970
0
    }
2971
0
  }
2972
25.4k
  else if (!cu.cs->slice->isIntra())
2973
0
  {
2974
0
    if (bestcost != MAX_DOUBLE)
2975
0
    {
2976
0
      speedIntra = 10;
2977
0
    }
2978
0
  }
2979
25.4k
  if (m_ispTestedModes[0].bestBefore[0] == -1)
2980
22.7k
  {
2981
22.7k
    speedIntra |= 7;
2982
22.7k
    if (m_pcEncCfg->m_FastIntraTools == 2)
2983
0
    {
2984
0
      EndMode = 1;
2985
0
    }
2986
22.7k
  }
2987
25.4k
  if (!cu.cs->slice->isIntra())
2988
0
  {
2989
0
    if ((m_ispTestedModes[0].bestBefore[1] == 1) || (m_ispTestedModes[0].bestBefore[2] == 1))
2990
0
    {
2991
0
      speedIntra |= 2;
2992
0
    }
2993
0
    if ((m_ispTestedModes[0].bestBefore[1] == 4) || (m_ispTestedModes[0].bestBefore[2] == 4))
2994
0
    {
2995
0
      speedIntra |= 3;
2996
0
    }
2997
0
    if ((m_ispTestedModes[0].bestBefore[1] == 2) || (m_ispTestedModes[0].bestBefore[2] == 2))
2998
0
    {
2999
0
      speedIntra |= 1;
3000
0
    }
3001
0
  }
3002
25.4k
}
3003
3004
} // namespace vvenc
3005
3006
//! \}
3007