Coverage Report

Created: 2026-09-01 06:57

next uncovered line (L), next uncovered region (R), next uncovered branch (B)
/work/vvenc/source/Lib/EncoderLib/EncCu.cpp
Line
Count
Source
1
/* -----------------------------------------------------------------------------
2
The copyright in this software is being made available under the Clear BSD
3
License, included below. No patent rights, trademark rights and/or
4
other Intellectual Property Rights other than the copyrights concerning
5
the Software are granted under this license.
6
7
The Clear BSD License
8
9
Copyright (c) 2019-2026, Fraunhofer-Gesellschaft zur Förderung der angewandten Forschung e.V. & The VVenC Authors.
10
All rights reserved.
11
12
Redistribution and use in source and binary forms, with or without modification,
13
are permitted (subject to the limitations in the disclaimer below) provided that
14
the following conditions are met:
15
16
     * Redistributions of source code must retain the above copyright notice,
17
     this list of conditions and the following disclaimer.
18
19
     * Redistributions in binary form must reproduce the above copyright
20
     notice, this list of conditions and the following disclaimer in the
21
     documentation and/or other materials provided with the distribution.
22
23
     * Neither the name of the copyright holder nor the names of its
24
     contributors may be used to endorse or promote products derived from this
25
     software without specific prior written permission.
26
27
NO EXPRESS OR IMPLIED LICENSES TO ANY PARTY'S PATENT RIGHTS ARE GRANTED BY
28
THIS LICENSE. THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND
29
CONTRIBUTORS "AS IS" AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT
30
LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A
31
PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR
32
CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL,
33
EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO,
34
PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR
35
BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER
36
IN CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE)
37
ARISING IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE
38
POSSIBILITY OF SUCH DAMAGE.
39
40
41
------------------------------------------------------------------------------------------- */
42
43
44
/** \file     EncCu.cpp
45
    \brief    Coding Unit (CU) encoder class
46
*/
47
48
#include "EncCu.h"
49
#include "EncLib.h"
50
#include "Analyze.h"
51
#include "EncPicture.h"
52
#include "EncModeCtrl.h"
53
#include "BitAllocation.h"
54
#include "EncStage.h"
55
56
#include "CommonLib/dtrace_codingstruct.h"
57
#include "CommonLib/Picture.h"
58
#include "CommonLib/UnitTools.h"
59
#include "CommonLib/dtrace_buffer.h"
60
#include "CommonLib/TimeProfiler.h"
61
#include "CommonLib/SearchSpaceCounter.h"
62
63
#include <mutex>
64
#include <cmath>
65
#include <algorithm>
66
67
//! \ingroup EncoderLib
68
//! \{
69
70
namespace vvenc {
71
72
const MergeIdxPair EncCu::m_GeoModeTest[GEO_MAX_NUM_CANDS] = { MergeIdxPair{0, 1}, MergeIdxPair{1, 0}, MergeIdxPair{0, 2}, MergeIdxPair{1, 2}, MergeIdxPair{2, 0},
73
                                                               MergeIdxPair{2, 1}, MergeIdxPair{0, 3}, MergeIdxPair{1, 3}, MergeIdxPair{2, 3}, MergeIdxPair{3, 0},
74
                                                               MergeIdxPair{3, 1}, MergeIdxPair{3, 2}, MergeIdxPair{0, 4}, MergeIdxPair{1, 4}, MergeIdxPair{2, 4},
75
                                                               MergeIdxPair{3, 4}, MergeIdxPair{4, 0}, MergeIdxPair{4, 1}, MergeIdxPair{4, 2}, MergeIdxPair{4, 3},
76
                                                               MergeIdxPair{0, 5}, MergeIdxPair{1, 5}, MergeIdxPair{2, 5}, MergeIdxPair{3, 5}, MergeIdxPair{4, 5},
77
                                                               MergeIdxPair{5, 0}, MergeIdxPair{5, 1}, MergeIdxPair{5, 2}, MergeIdxPair{5, 3}, MergeIdxPair{5, 4} };
78
79
80
// Shape coefSquareCUsFasterFastMedium (2 x 5 x 2 x 2 x 2): preset (faster and fast + medium) x cusize x nspred x sptype x numcoef
81
82
const double EncCu::coefSquareCUsFasterFastMedium[2][5][2][2][2] = {
83
{{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, },  {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, },  },
84
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, },  {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, },  },
85
{{{-1.00000000, -1.00000000, }, {0.07848505, 0.00225808, }, },  {{-1.00000000, -1.00000000, }, {0.07509575, 0.00204789, }, },  },
86
{{{-1.00000000, -1.00000000, }, {0.10833051, 0.00053144, }, },  {{-1.00000000, -1.00000000, }, {0.08304352, 0.00142876, }, },  },
87
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, },  {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, },  },
88
},
89
{{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, },  {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, },  },
90
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, },  {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, },  },
91
{{{0.06852235, 0.00388054, }, {0.09236045, 0.00084528, }, },  {{0.06955832, 0.00289679, }, {0.09598522, 0.00096187, }, },  },
92
{{{0.07268085, 0.00302796, }, {0.09323753, 0.00050996, }, },  {{0.06123618, 0.00471601, }, {0.09253389, 0.00046826, }, },  },
93
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, },  {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, },  },
94
},
95
};
96
97
// Shape coefSquareCUsSlowSlower (2 x 5 x 2 x 5 x 2 x 2 x 2): preset (Slow + Slower) x cusize x nspred x qtdepth x mtdepth x sptype x numcoef
98
99
const double EncCu::coefSquareCUsSlowSlower[2][5][2][5][2][2][2] = {
100
{{{{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
101
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
102
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
103
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
104
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
105
},
106
{{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
107
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
108
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
109
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
110
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
111
},
112
},
113
{{{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
114
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
115
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
116
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
117
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
118
},
119
{{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
120
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
121
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
122
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
123
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
124
},
125
},
126
{{{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
127
{{{0.06057349, 0.00447803, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
128
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{0.06948736, 0.00327774, }, {0.06396861, 0.00789923, }, }, },
129
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
130
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
131
},
132
{{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
133
{{{0.05526461, 0.00436703, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
134
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{0.05661641, 0.00546358, }, {0.06680048, 0.00670604, }, }, },
135
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
136
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
137
},
138
},
139
{{{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
140
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{0.06556926, 0.00455697, }, {0.06803347, 0.00365396, }, }, },
141
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
142
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
143
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
144
},
145
{{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
146
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{0.06394328, 0.00403533, }, {0.06688063, 0.00272684, }, }, },
147
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
148
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
149
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
150
},
151
},
152
{{{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{0.05533525, 0.00235007, }, {0.07184475, 0.00086154, }, }, },
153
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
154
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
155
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
156
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
157
},
158
{{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
159
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
160
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
161
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
162
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
163
},
164
},
165
},
166
{{{{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
167
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
168
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
169
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
170
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
171
},
172
{{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
173
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
174
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
175
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
176
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
177
},
178
},
179
{{{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
180
{{{0.04770815, 0.00774597, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
181
{{{0.05509813, 0.00812349, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
182
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{0.04719064, 0.01023887, }, {0.04641434, 0.01279769, }, }, },
183
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
184
},
185
{{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
186
{{{0.02777251, 0.00811989, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
187
{{{0.04848973, 0.00628111, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
188
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{0.04924294, 0.00769725, }, {0.04491680, 0.01120645, }, }, },
189
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
190
},
191
},
192
{{{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
193
{{{0.06856533, 0.00418949, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
194
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{0.07059085, 0.00348835, }, {0.06051544, 0.00821365, }, }, },
195
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
196
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
197
},
198
{{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
199
{{{0.07461087, 0.00229937, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
200
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{0.05893714, 0.00513168, }, {0.05809189, 0.00772994, }, }, },
201
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
202
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
203
},
204
},
205
{{{{{0.06922080, 0.00274816, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
206
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{0.06176300, 0.00374356, }, {0.06085891, 0.00361274, }, }, },
207
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
208
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
209
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
210
},
211
{{{{0.06611224, 0.00256748, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
212
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{0.06364765, 0.00282819, }, {0.05888407, 0.00320947, }, }, },
213
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
214
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
215
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
216
},
217
},
218
{{{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{0.05637817, 0.00184637, }, {0.06937475, 0.00091860, }, }, },
219
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
220
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
221
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
222
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
223
},
224
{{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
225
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
226
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
227
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
228
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
229
},
230
},
231
},
232
};
233
234
// ====================================================================================================================
235
EncCu::EncCu()
236
18.6k
  : m_CtxCache          ( nullptr )
237
18.6k
  , m_globalCtuQpVector ( nullptr )
238
18.6k
  , m_wppMutex          ( nullptr )
239
18.6k
  , m_CABACEstimator    ( nullptr )
240
18.6k
{
241
18.6k
}
242
243
void EncCu::initPic( Picture* pic )
244
4.66k
{
245
4.66k
  m_cInterSearch.setSearchRange( pic->cs->slice, *m_pcEncCfg );
246
247
4.66k
  m_wppMutex = (m_pcEncCfg->m_numThreads > 0 ) ? &pic->wppMutex : nullptr;
248
4.66k
}
249
250
void EncCu::initSlice( const Slice* slice )
251
4.66k
{
252
4.66k
  m_cTrQuant.setLambdas( slice->getLambdas() );
253
4.66k
  m_cRdCost.setLambda( slice->getLambdas()[0], slice->sps->bitDepths );
254
4.66k
}
255
256
void EncCu::setCtuEncRsrc( CABACWriter* cabacEstimator, CtxCache* ctxCache, ReuseUniMv* pReuseUniMv, BlkUniMvInfoBuffer* pBlkUniMvInfoBuffer, AffineProfList* pAffineProfList, IbcBvCand* pCachedBvs )
257
3.57k
{
258
3.57k
  m_CABACEstimator = cabacEstimator;
259
3.57k
  m_CtxCache       = ctxCache;
260
3.57k
  m_cIntraSearch.setCtuEncRsrc( cabacEstimator, ctxCache );
261
3.57k
  m_cInterSearch.setCtuEncRsrc( cabacEstimator, ctxCache, pReuseUniMv, pBlkUniMvInfoBuffer, pAffineProfList, pCachedBvs );
262
3.57k
}
263
264
void EncCu::setUpLambda (Slice& slice, const double dLambda, const int iQP, const bool setSliceLambda, const bool saveUnadjusted)
265
8.23k
{
266
  // store lambda
267
8.23k
  m_cRdCost.setLambda( dLambda, slice.sps->bitDepths );
268
269
  // for RDO
270
  // in RdCost there is only one lambda because the luma and chroma bits are not separated, instead we weight the distortion of chroma.
271
8.23k
  double dLambdas[MAX_NUM_COMP] = { dLambda };
272
24.7k
  for( uint32_t compIdx = 1; compIdx < MAX_NUM_COMP; compIdx++ )
273
16.4k
  {
274
16.4k
    const ComponentID compID = ComponentID( compIdx );
275
16.4k
    int chromaQPOffset       = slice.pps->chromaQpOffset[compID] + slice.sliceChromaQpDelta[ compID ];
276
16.4k
    int qpc = slice.sps->chromaQpMappingTable.getMappedChromaQpValue(compID, iQP) + chromaQPOffset;
277
16.4k
    double tmpWeight         = pow( 2.0, ( iQP - qpc ) / 3.0 );  // takes into account of the chroma qp mapping and chroma qp Offset
278
16.4k
    if( m_pcEncCfg->m_DepQuantEnabled/* && !( m_pcEncCfg->getLFNST() ) */)
279
16.4k
    {
280
16.4k
      tmpWeight *= ( m_pcEncCfg->m_GOPSize >= 8 ? pow( 2.0, 0.1/3.0 ) : pow( 2.0, 0.2/3.0 ) );  // increase chroma weight for dependent quantization (in order to reduce bit rate shift from chroma to luma)
281
16.4k
    }
282
16.4k
    m_cRdCost.setDistortionWeight( compID, tmpWeight );
283
16.4k
    dLambdas[compIdx] = dLambda / tmpWeight;
284
16.4k
  }
285
286
  // for RDOQ
287
8.23k
  m_cTrQuant.setLambdas( dLambdas );
288
289
  // for SAO, ALF
290
8.23k
  if (setSliceLambda)
291
4.66k
  {
292
4.66k
    slice.setLambdas( dLambdas );
293
4.66k
  }
294
8.23k
  if( saveUnadjusted )
295
8.23k
  {
296
8.23k
    m_cRdCost.saveUnadjustedLambda();
297
8.23k
  }
298
8.23k
}
299
300
void EncCu::updateLambda(const Slice& slice, const double ctuLambda, const int ctuQP, const int newQP, const bool saveUnadjusted)
301
3.08k
{
302
3.08k
  const double  corrFactor = pow (2.0, double (newQP - ctuQP) / 3.0);
303
3.08k
  const double  newLambda  = ctuLambda * corrFactor;
304
  
305
3.08k
  double newLambdas[MAX_NUM_COMP] = { newLambda };
306
9.24k
  for( uint32_t compIdx = 1; compIdx < MAX_NUM_COMP; compIdx++ )
307
6.16k
  {
308
6.16k
    const ComponentID compID = ComponentID( compIdx );
309
6.16k
    int chromaQPOffset       = slice.pps->chromaQpOffset[compID] + slice.sliceChromaQpDelta[ compID ];
310
6.16k
    int qpc = slice.sps->chromaQpMappingTable.getMappedChromaQpValue(compID, newQP) + chromaQPOffset;
311
6.16k
    double tmpWeight         = pow( 2.0, ( newQP - qpc ) / 3.0 );  // takes into account of the chroma qp mapping and chroma qp Offset
312
6.16k
    if( m_pcEncCfg->m_DepQuantEnabled/* && !( m_pcEncCfg->getLFNST() ) */)
313
6.16k
    {
314
6.16k
      tmpWeight *= ( m_pcEncCfg->m_GOPSize >= 8 ? pow( 2.0, 0.1/3.0 ) : pow( 2.0, 0.2/3.0 ) );  // increase chroma weight for dependent quantization (in order to reduce bit rate shift from chroma to luma)
315
6.16k
    }
316
6.16k
    m_cRdCost.setDistortionWeight( compID, tmpWeight );
317
6.16k
    newLambdas[compIdx] = newLambda / tmpWeight;
318
6.16k
  }
319
320
3.08k
  m_cTrQuant.setLambdas ( newLambdas);
321
3.08k
  m_cRdCost.setLambda   ( newLambda, slice.sps->bitDepths);
322
323
3.08k
  if (saveUnadjusted)
324
3.08k
  {
325
3.08k
    m_cRdCost.saveUnadjustedLambda(); // TODO hlm: check if this actually improves the overall quality
326
3.08k
  }
327
3.08k
}
328
329
void EncCu::init( const VVEncCfg& encCfg, const SPS& sps, std::vector<int>* const globalCtuQpVector, Ctx* syncPicCtx, RateCtrl* pRateCtrl )
330
18.6k
{
331
18.6k
  DecCu::init( &m_cTrQuant, &m_cIntraSearch, &m_cInterSearch, encCfg.m_internChromaFormat );
332
18.6k
  m_cRdCost.create     ();
333
18.6k
  m_cRdCost.setCostMode( encCfg.m_costMode );
334
18.6k
  if ( encCfg.m_lumaLevelToDeltaQPEnabled == 1 )
335
0
  {
336
0
    m_cRdCost.setChromaFormat( encCfg.m_internChromaFormat );
337
0
    m_cRdCost.initLumaLevelToWeightTable( encCfg.m_internalBitDepth[ CH_L ] );
338
0
    xInitLumaDeltaQpLUT();
339
0
  }
340
341
18.6k
  m_modeCtrl.init     ( encCfg, &m_cRdCost );
342
18.6k
  m_cIntraSearch.init ( encCfg, &m_cTrQuant, &m_cRdCost, &m_SortedPelUnitBufs, m_unitCache );
343
18.6k
  m_cInterSearch.init ( encCfg, &m_cTrQuant, &m_cRdCost, &m_modeCtrl, m_cIntraSearch.getSaveCSBuf() );
344
18.6k
  m_cTrQuant.init     ( nullptr, encCfg.m_RDOQ, encCfg.m_useRDOQTS, false, true, encCfg.m_quantThresholdVal );
345
346
18.6k
  m_syncPicCtx = syncPicCtx;                         ///< context storage for state of contexts at the wavefront/WPP/entropy-coding-sync second CTU of tile-row used for estimation
347
18.6k
  m_pcRateCtrl = pRateCtrl;
348
349
  // Initialise scaling lists: The encoder will only use the SPS scaling lists. The PPS will never be marked present.
350
18.6k
  const int maxLog2TrDynamicRange[ MAX_NUM_CH ] = { sps.getMaxLog2TrDynamicRange(), sps.getMaxLog2TrDynamicRange() };
351
18.6k
  m_cTrQuant.getQuant()->setFlatScalingList( maxLog2TrDynamicRange, sps.bitDepths );
352
353
18.6k
  m_pcEncCfg       = &encCfg;
354
355
18.6k
  m_GeoCostList.init( encCfg.m_maxNumGeoCand );
356
357
18.6k
  unsigned      uiMaxSize    = encCfg.m_CTUSize;
358
18.6k
  ChromaFormat  chromaFormat = encCfg.m_internChromaFormat;
359
360
18.6k
  Area ctuArea = Area( 0, 0, uiMaxSize, uiMaxSize );
361
362
18.6k
  m_mergeItemList.init( encCfg.m_maxMergeRdCandNumTotal, m_pcEncCfg->m_Geo > 1 ? 3 : 1, chromaFormat, uiMaxSize, uiMaxSize );
363
364
242k
  for( int i = 0; i < maxCuDepth; i++ )
365
223k
  {
366
223k
    Area area = Area( 0, 0, uiMaxSize >> ( i >> 1 ), uiMaxSize >> ( ( i + 1 ) >> 1 ) );
367
368
223k
    if( area.width < (1 << MIN_CU_LOG2) || area.height < (1 << MIN_CU_LOG2) )
369
18.6k
    {
370
18.6k
      m_pTempCS[i] = m_pBestCS[i] = nullptr;
371
18.6k
      continue;
372
18.6k
    }
373
374
205k
    m_pTempCS[i] = new CodingStructure( m_unitCache, nullptr );
375
205k
    m_pBestCS[i] = new CodingStructure( m_unitCache, nullptr );
376
377
205k
    m_pTempCS[i]->createForSearch( chromaFormat, area );
378
205k
    m_pBestCS[i]->createForSearch( chromaFormat, area );
379
380
205k
    m_pOrgBuffer[i].create( chromaFormat, area );
381
205k
    m_pRspBuffer[i].create( CHROMA_400, area );
382
205k
  }
383
384
18.6k
  m_pTempCS2 = new CodingStructure( m_unitCache, nullptr );
385
18.6k
  m_pBestCS2 = new CodingStructure( m_unitCache, nullptr );
386
387
18.6k
  m_pTempCS2->createForSearch( chromaFormat, ctuArea );
388
18.6k
  m_pBestCS2->createForSearch( chromaFormat, ctuArea );
389
390
18.6k
  m_cuChromaQpOffsetIdxPlus1 = 0;
391
18.6k
  m_tempQpDiff = 0;
392
18.6k
  m_globalCtuQpVector = globalCtuQpVector;
393
394
18.6k
  m_SortedPelUnitBufs.create( chromaFormat, uiMaxSize, uiMaxSize );
395
396
279k
  for( uint8_t i = 0; i < MAX_TMP_BUFS; i++)
397
261k
  {
398
261k
    m_aTmpStorageLCU[i].create(chromaFormat, Area(0, 0, uiMaxSize, uiMaxSize));
399
261k
  }
400
130k
  for (unsigned ui = 0; ui < MRG_MAX_NUM_CANDS; ui++)
401
111k
  {
402
111k
    m_acMergeTmpBuffer[ui].create(chromaFormat, Area(0, 0, uiMaxSize, uiMaxSize));
403
111k
  }
404
405
18.6k
  const unsigned maxDepth = 2 * MAX_CU_SIZE_IDX;
406
18.6k
  m_CtxBuffer.resize( maxDepth );
407
18.6k
  m_CurrCtx = 0;
408
18.6k
  if( encCfg.m_EDO )
409
18.6k
    m_dbBuffer.create( chromaFormat, Area( 0, 0, uiMaxSize, uiMaxSize ), 0, 8 );
410
411
18.6k
  m_MergeSimpleFlag = 0;
412
18.6k
  m_tileIdx = 0;
413
18.6k
}
414
415
416
void EncCu::destroy()
417
18.6k
{
418
242k
  for( int i = 0; i < maxCuDepth; i++ )
419
223k
  {
420
223k
    if( m_pTempCS[i] )
421
205k
    {
422
205k
      m_pTempCS[i]->destroy();
423
205k
      delete m_pTempCS[i]; m_pTempCS[i] = nullptr;
424
205k
    }
425
426
223k
    if( m_pBestCS[i] )
427
205k
    {
428
205k
      m_pBestCS[i]->destroy();
429
205k
      delete m_pBestCS[i]; m_pBestCS[i] = nullptr;
430
205k
    }
431
432
223k
    m_pOrgBuffer[i].destroy();
433
223k
    m_pRspBuffer[i].destroy();
434
223k
  }
435
436
18.6k
  m_pTempCS2->destroy();
437
18.6k
  m_pBestCS2->destroy();
438
439
18.6k
  delete m_pTempCS2; m_pTempCS2 = nullptr;
440
18.6k
  delete m_pBestCS2; m_pBestCS2 = nullptr;
441
442
18.6k
  m_SortedPelUnitBufs.destroy();
443
444
279k
  for( uint8_t i = 0; i < MAX_TMP_BUFS; i++)
445
261k
  {
446
261k
    m_aTmpStorageLCU[i].destroy();
447
261k
  }
448
130k
  for (unsigned ui = 0; ui < MRG_MAX_NUM_CANDS; ui++)
449
111k
  {
450
111k
    m_acMergeTmpBuffer[ui].destroy();
451
111k
  }
452
453
454
18.6k
  m_dbBuffer.destroy();
455
18.6k
}
456
457
458
EncCu::~EncCu()
459
18.6k
{
460
18.6k
  destroy();
461
18.6k
}
462
463
// ====================================================================================================================
464
// Public member functions
465
// ====================================================================================================================
466
467
void EncCu::encodeCtu( Picture* pic, int (&prevQP)[MAX_NUM_CH], uint32_t ctuXPosInCtus, uint32_t ctuYPosInCtus )
468
3.57k
{
469
3.57k
  CodingStructure&     cs          = *pic->cs;
470
3.57k
  Slice*               slice       = cs.slice;
471
3.57k
  const PreCalcValues& pcv         = *cs.pcv;
472
473
#if ENABLE_MEASURE_SEARCH_SPACE
474
  if( ctuXPosInCtus == 0 && ctuYPosInCtus == 0 )
475
  {
476
    g_searchSpaceAcc.picW = pic->lwidth();
477
    g_searchSpaceAcc.picH = pic->lheight();
478
    g_searchSpaceAcc.addSlice( slice->isIntra(), slice->depth );
479
  }
480
481
#endif
482
3.57k
  const int ctuRsAddr                 = ctuYPosInCtus * pcv.widthInCtus + ctuXPosInCtus;
483
484
3.57k
  const Position pos (ctuXPosInCtus * pcv.maxCUSize, ctuYPosInCtus * pcv.maxCUSize);
485
3.57k
  const UnitArea ctuArea( cs.area.chromaFormat, Area( pos.x, pos.y, pcv.maxCUSize, pcv.maxCUSize ) );
486
3.57k
  DTRACE_UPDATE( g_trace_ctx, std::make_pair( "ctu", ctuRsAddr ) );
487
488
3.57k
  const int tileXPosInCtus = cs.pps->tileColBd[cs.pps->ctuToTileCol[ctuXPosInCtus]];
489
3.57k
  const int tileYPosInCtus = cs.pps->tileRowBd[cs.pps->ctuToTileRow[ctuYPosInCtus]];
490
491
3.57k
  if( ( cs.slice->sliceType != VVENC_I_SLICE || cs.sps->IBC ) && ctuXPosInCtus == tileXPosInCtus )
492
2.03k
  {
493
2.03k
    const int tileRowId = cs.pps->getTileLineId( ctuXPosInCtus, ctuYPosInCtus );
494
2.03k
    cs.motionLutBuf[tileRowId].lut.resize( 0 );
495
2.03k
    cs.motionLutBuf[tileRowId].lutIbc.resize( 0 );
496
2.03k
  }
497
498
3.57k
  if( ( m_pcEncCfg->m_ensureWppBitEqual || m_pcEncCfg->m_entropyCodingSyncEnabled ) && ctuXPosInCtus == tileXPosInCtus )
499
2.03k
  {
500
2.03k
    m_CABACEstimator->initCtxModels( *slice );
501
502
2.03k
    if( m_pcEncCfg->m_entropyCodingSyncEnabled && ( ctuYPosInCtus > tileYPosInCtus ) )
503
0
    {
504
0
      m_CABACEstimator->getCtx() = m_syncPicCtx[slice->pps->getTileLineId( ctuXPosInCtus, ctuYPosInCtus - 1 )];
505
0
    }
506
507
2.03k
    prevQP[CH_L] = prevQP[CH_C] = slice->sliceQp; // hlm: call CU::predictQP() here!
508
2.03k
  }
509
1.54k
  else if( ctuXPosInCtus == tileXPosInCtus && ctuYPosInCtus == tileYPosInCtus )
510
0
  {
511
0
    m_CABACEstimator->initCtxModels( *slice );
512
0
    prevQP[CH_L] = prevQP[CH_C] = slice->sliceQp; // hlm: call CU::predictQP() here!
513
0
  }
514
515
3.57k
  xCompressCtu( cs, ctuArea, ctuRsAddr, prevQP );
516
517
3.57k
  m_CABACEstimator->resetBits();
518
3.57k
  m_CABACEstimator->coding_tree_unit( cs, ctuArea, prevQP, ctuRsAddr, true, true );
519
520
  // Store probabilities of second CTU in line into buffer - used only if wavefront-parallel-processing is enabled.
521
3.57k
  if( ctuXPosInCtus == tileXPosInCtus && m_pcEncCfg->m_entropyCodingSyncEnabled )
522
0
  {
523
0
    m_syncPicCtx[slice->pps->getTileLineId( ctuXPosInCtus, ctuYPosInCtus )] = m_CABACEstimator->getCtx();
524
0
  }
525
526
3.57k
  DTRACE_AREA_CRC( g_trace_ctx, D_CRC, cs, ctuArea );
527
3.57k
}
528
529
// ====================================================================================================================
530
// Protected member functions
531
// ====================================================================================================================
532
533
void EncCu::xCompressCtu( CodingStructure& cs, const UnitArea& area, const unsigned ctuRsAddr, const int prevQP[] )
534
3.57k
{
535
3.57k
  m_tileIdx = cs.pps->getTileIdx( area.lumaPos() );
536
537
3.57k
  m_modeCtrl.initCTUEncoding( *cs.slice, m_tileIdx );
538
539
  // init the partitioning manager
540
3.57k
  Partitioner *partitioner = &m_partitioner;
541
3.57k
  partitioner->initCtu( area, CH_L, *cs.slice );
542
  
543
3.57k
  const Position& lumaPos = area.lumaPos();
544
3.57k
  const bool leftSameTile  = lumaPos.x == 0 || m_tileIdx == cs.pps->getTileIdx( lumaPos.offset(-1, 0) );
545
3.57k
  const bool aboveSameTile = lumaPos.y == 0 || m_tileIdx == cs.pps->getTileIdx( lumaPos.offset( 0,-1) );
546
3.57k
  m_EDO = (!m_pcEncCfg->m_tileParallelCtuEnc || (leftSameTile && aboveSameTile)) ? m_pcEncCfg->m_EDO : 0;
547
  
548
3.57k
  if( m_pcEncCfg->m_IBCMode )
549
3.57k
  {
550
3.57k
    m_cInterSearch.resetCtuRecordIBC();
551
3.57k
  }
552
553
  // init current context pointer
554
3.57k
  m_CurrCtx = m_CtxBuffer.data();
555
556
3.57k
  PelStorage* orgBuffer = &m_pOrgBuffer[0];
557
3.57k
  PelStorage* rspBuffer = &m_pRspBuffer[0];
558
3.57k
  CodingStructure *tempCS =  m_pTempCS [0];
559
3.57k
  CodingStructure *bestCS =  m_pBestCS [0];
560
3.57k
  cs.initSubStructure( *tempCS, partitioner->chType, partitioner->currArea(), false, orgBuffer, rspBuffer );
561
3.57k
  cs.initSubStructure( *bestCS, partitioner->chType, partitioner->currArea(), false, orgBuffer, rspBuffer );
562
3.57k
  PROFILER_SCOPE_AND_STAGE_EXT( 1, _TPROF, P_COMPRESS_CU, tempCS, CH_L );
563
564
  // copy the relevant area
565
3.57k
  UnitArea clippedArea = clipArea( partitioner->currArea(), cs.area );
566
3.57k
  CPelUnitBuf org = cs.picture->getFilteredOrigBuffer().valid() ? cs.picture->getFiltOrigBuf( clippedArea ) : cs.picture->getOrigBuf( clippedArea );
567
3.57k
  tempCS->getOrgBuf( clippedArea ).copyFrom( org );
568
3.57k
  tempCS->currQP[CH_L] = bestCS->currQP[CH_L] =
569
3.57k
  tempCS->baseQP       = bestCS->baseQP       = cs.slice->sliceQp;
570
3.57k
  tempCS->prevQP[CH_L] = bestCS->prevQP[CH_L] = prevQP[CH_L];
571
572
3.57k
  xCompressCU( tempCS, bestCS, *partitioner );
573
  // all signals were already copied during compression if the CTU was split - at this point only the structures are copied to the top level CS
574
  
575
  // Ensure that a coding was found
576
  // Selected mode's RD-cost must be not MAX_DOUBLE.
577
3.57k
  CHECK( bestCS->cus.empty()                                   , "No possible encoding found" );
578
3.57k
  CHECK( bestCS->cus[0]->predMode == NUMBER_OF_PREDICTION_MODES, "No possible encoding found" );
579
3.57k
  CHECK( bestCS->cost             == MAX_DOUBLE                , "No possible encoding found" );
580
581
3.57k
  if ( m_wppMutex ) m_wppMutex->lock();
582
583
3.57k
  cs.useSubStructure( *bestCS, partitioner->chType, TREE_D, CS::getArea( *bestCS, area, partitioner->chType, partitioner->treeType ) );
584
585
3.57k
  if ( m_wppMutex ) m_wppMutex->unlock();
586
587
3.57k
  if( CS::isDualITree( cs ) && isChromaEnabled( cs.pcv->chrFormat ) )
588
3.57k
  {
589
3.57k
    m_CABACEstimator->getCtx() = m_CurrCtx->start;
590
591
3.57k
    partitioner->initCtu( area, CH_C, *cs.slice );
592
593
3.57k
    cs.initSubStructure( *tempCS, partitioner->chType, partitioner->currArea(), false, orgBuffer, rspBuffer );
594
3.57k
    cs.initSubStructure( *bestCS, partitioner->chType, partitioner->currArea(), false, orgBuffer, rspBuffer );
595
3.57k
    tempCS->currQP[CH_C] = bestCS->currQP[CH_C] =
596
3.57k
    tempCS->baseQP       = bestCS->baseQP       = cs.slice->sliceQp;
597
3.57k
    tempCS->prevQP[CH_C] = bestCS->prevQP[CH_C] = prevQP[CH_C];
598
599
3.57k
    xCompressCU( tempCS, bestCS, *partitioner );
600
    
601
    // Ensure that a coding was found
602
    // Selected mode's RD-cost must be not MAX_DOUBLE.
603
3.57k
    CHECK( bestCS->cus.empty()                                   , "No possible encoding found" );
604
3.57k
    CHECK( bestCS->cus[0]->predMode == NUMBER_OF_PREDICTION_MODES, "No possible encoding found" );
605
3.57k
    CHECK( bestCS->cost             == MAX_DOUBLE                , "No possible encoding found" );
606
607
3.57k
    if ( m_wppMutex ) m_wppMutex->lock();
608
609
3.57k
    cs.useSubStructure( *bestCS, partitioner->chType, TREE_D, CS::getArea( *bestCS, area, partitioner->chType, partitioner->treeType ) );
610
611
3.57k
    if ( m_wppMutex ) m_wppMutex->unlock();
612
3.57k
  }
613
614
  // reset context states and uninit context pointer
615
3.57k
  m_CABACEstimator->getCtx() = m_CurrCtx->start;
616
3.57k
  m_CurrCtx                  = 0;
617
3.57k
}
618
619
620
621
bool EncCu::xCheckBestMode( CodingStructure *&tempCS, CodingStructure *&bestCS, Partitioner &partitioner, const EncTestMode& encTestMode, const bool useEDO )
622
270k
{
623
270k
  bool bestCSUpdated = false;
624
625
270k
  if( !tempCS->cus.empty() )
626
164k
  {
627
164k
    if( tempCS->cus.size() == 1 )
628
123k
    {
629
123k
      const CodingUnit& cu = *tempCS->cus.front();
630
123k
      CHECK( cu.skip && !cu.mergeFlag, "Skip flag without a merge flag is not allowed!" );
631
123k
    }
632
633
164k
    DTRACE_BEST_MODE( tempCS, bestCS, m_cRdCost.getLambda(true), useEDO );
634
635
164k
    if( m_modeCtrl.useModeResult( encTestMode, tempCS, partitioner, useEDO ) )
636
148k
    {
637
148k
      std::swap( tempCS, bestCS );
638
      // store temp best CI for next CU coding
639
148k
      m_CurrCtx->best = m_CABACEstimator->getCtx();
640
148k
      bestCSUpdated = true;
641
148k
    }
642
164k
  }
643
644
  // reset context states
645
270k
  m_CABACEstimator->getCtx() = m_CurrCtx->start;
646
270k
  return bestCSUpdated;
647
648
270k
}
649
650
void xCheckFastCuChromaSplitting( CodingStructure*& tempCS, CodingStructure*& bestCS, Partitioner&  partitioner, ComprCUCtx& cuECtx )
651
0
{
652
0
  const uint32_t uiLPelX = tempCS->area.Cb().lumaPos().x;
653
0
  const uint32_t uiTPelY = tempCS->area.Cb().lumaPos().y;
654
655
0
  int lumaw = 0, lumah = 0;
656
0
  bool splitver      = true;
657
0
  bool splithor      = true;
658
0
  bool qtSplitChroma = true;
659
660
0
  if( partitioner.isSepTree( *tempCS ) && isChroma( partitioner.chType ) )
661
0
  {
662
0
    Position lumaRefPos( uiLPelX, uiTPelY );
663
0
    CodingUnit* colLumaCu = bestCS->lumaCS->getCU( lumaRefPos, CH_L, TREE_D );
664
665
0
    if( colLumaCu )
666
0
    {
667
0
      lumah = colLumaCu->Y().height;
668
0
      lumaw = colLumaCu->Y().width;
669
0
    }
670
0
  }
671
0
  else
672
0
  {
673
0
    return;
674
0
  }
675
676
0
  if( partitioner.getImplicitSplit( *tempCS ) != CU_DONT_SPLIT ) return;
677
678
0
  const CPelBuf orgCb = tempCS->getOrgBuf( COMP_Cb );
679
0
  const CPelBuf orgCr = tempCS->getOrgBuf( COMP_Cr );
680
681
0
  int th1 = FCBP_TH1;
682
683
0
  if( ( lumaw >> getChannelTypeScaleX( CH_C, tempCS->area.chromaFormat ) ) == orgCb.width )
684
0
  {
685
0
    if( ( bestCS->cost < ( th1*orgCb.width*orgCb.height ) ) )
686
0
    {
687
0
      splitver      = false;
688
0
      qtSplitChroma = false;
689
0
    }
690
0
  }
691
692
0
  if( ( lumah >> getChannelTypeScaleY( CH_C, tempCS->area.chromaFormat ) ) == orgCb.height )
693
0
  {
694
0
    if( ( bestCS->cost < ( th1*orgCb.width*orgCb.height ) ) )
695
0
    {
696
0
      splithor      = false;
697
0
      qtSplitChroma = false;
698
0
    }
699
0
  }
700
701
0
  cuECtx.doHorChromaSplit = splithor;
702
0
  cuECtx.doVerChromaSplit = splitver;
703
0
  cuECtx.doQtChromaSplit  = qtSplitChroma;
704
705
0
  if( orgCb.width == orgCb.height )
706
0
  {
707
0
    int varh_cb, varv_cb;
708
0
    int varh_cr, varv_cr;
709
710
0
    orgCb.calcVarianceSplit( orgCb, orgCb.width, varh_cb, varv_cb );
711
0
    orgCr.calcVarianceSplit( orgCr, orgCr.width, varh_cr, varv_cr );
712
713
0
    if( ( varh_cr*FCBP_TH2 < varv_cr * 100 ) && ( varh_cb*FCBP_TH2 < varv_cb * 100 ) )
714
0
    {
715
0
      cuECtx.doVerChromaSplit = false;
716
0
    }
717
0
    else if( ( varv_cr*FCBP_TH2 < varh_cr * 100 ) && ( varv_cb*FCBP_TH2 < varh_cb * 100 ) )
718
0
    {
719
0
      cuECtx.doHorChromaSplit = false;
720
0
    }
721
0
  }
722
0
}
723
724
void EncCu::xCompressCU( CodingStructure*& tempCS, CodingStructure*& bestCS, Partitioner& partitioner )
725
124k
{
726
124k
  const Area& lumaArea = tempCS->area.Y();
727
728
124k
  Slice&   slice      = *tempCS->slice;
729
124k
  const PPS &pps      = *tempCS->pps;
730
124k
  const SPS &sps      = *tempCS->sps;
731
124k
  const uint32_t uiLPelX  = tempCS->area.Y().lumaPos().x;
732
124k
  const uint32_t uiTPelY  = tempCS->area.Y().lumaPos().y;
733
124k
  const bool isBimEnabled = (m_pcEncCfg->m_blockImportanceMapping && !bestCS->picture->m_picShared->m_ctuBimQpOffset.empty());
734
735
124k
  m_modeCtrl.initBlk( tempCS->area, slice.pic->poc );
736
124k
  m_CABACEstimator->determineNeighborCus( *tempCS, partitioner.currArea(), partitioner.chType, partitioner.treeType );
737
738
124k
  if ((m_pcEncCfg->m_usePerceptQPA || isBimEnabled || m_pcEncCfg->m_lumaLevelToDeltaQPEnabled == 1) && ( pps.useDQP || m_pcEncCfg->m_maxDeltaQP == 0 ) && isLuma (partitioner.chType) && partitioner.currQgEnable())
739
6.65k
  {
740
6.65k
    const PreCalcValues &pcv = *pps.pcv;
741
6.65k
    Picture* const pic = bestCS->picture;
742
6.65k
    const uint32_t ctuRsAddr = getCtuAddr (partitioner.currQgPos, pcv);
743
744
6.65k
    if (partitioner.currSubdiv == 0) // CTU-level QP adaptation
745
3.57k
    {
746
3.57k
      if (m_pcEncCfg->m_usePerceptQPA)
747
3.57k
      {
748
3.57k
        if (m_pcEncCfg->m_internalUsePerceptQPATempFiltISlice == 2)
749
0
        {
750
0
          m_tempQpDiff = pic->ctuAdaptedQP[ctuRsAddr] - BitAllocation::applyQPAdaptationSubCtu (&slice, m_pcEncCfg, lumaArea, m_pcRateCtrl->getMinNoiseLevels());
751
0
        }
752
753
3.57k
        int clippedMaxDQP = 0;
754
755
3.57k
        if ((!slice.isIntra()) && (pcv.maxCUSize > 64) && // sub-CTU QPA behavior - Museum fix
756
0
            (uiLPelX + (pcv.maxCUSize >> 1) < (m_pcEncCfg->m_PadSourceWidth)) &&
757
0
            (uiTPelY + (pcv.maxCUSize >> 1) < (m_pcEncCfg->m_PadSourceHeight)))
758
0
        {
759
0
          const uint32_t h = lumaArea.height >> 1;
760
0
          const uint32_t w = lumaArea.width  >> 1;
761
0
          const int adQPTL = BitAllocation::applyQPAdaptationSubCtu (&slice, m_pcEncCfg, Area (uiLPelX + 0, uiTPelY + 0, w, h), m_pcRateCtrl->getMinNoiseLevels());
762
0
          const int adQPTR = BitAllocation::applyQPAdaptationSubCtu (&slice, m_pcEncCfg, Area (uiLPelX + w, uiTPelY + 0, w, h), m_pcRateCtrl->getMinNoiseLevels());
763
0
          const int adQPBL = BitAllocation::applyQPAdaptationSubCtu (&slice, m_pcEncCfg, Area (uiLPelX + 0, uiTPelY + h, w, h), m_pcRateCtrl->getMinNoiseLevels());
764
0
          const int adQPBR = BitAllocation::applyQPAdaptationSubCtu (&slice, m_pcEncCfg, Area (uiLPelX + w, uiTPelY + h, w, h), m_pcRateCtrl->getMinNoiseLevels());
765
766
0
          tempCS->currQP[partitioner.chType] = tempCS->baseQP =
767
0
          bestCS->currQP[partitioner.chType] = bestCS->baseQP = std::min (std::min (adQPTL, adQPTR), std::min (adQPBL, adQPBR));
768
769
0
          if (m_pcEncCfg->m_internalUsePerceptQPATempFiltISlice == 2)
770
0
          {
771
0
            if ((m_globalCtuQpVector->size() > ctuRsAddr) && (slice.TLayer == 0) && // last CTU row of non-Intra key-frame
772
0
                (m_pcEncCfg->m_IntraPeriod == 2 * m_pcEncCfg->m_GOPSize) && (ctuRsAddr >= pcv.widthInCtus) && (uiTPelY + pcv.maxCUSize > m_pcEncCfg->m_PadSourceHeight))
773
0
            {
774
0
              m_globalCtuQpVector->at (ctuRsAddr) = m_globalCtuQpVector->at (ctuRsAddr - pcv.widthInCtus); // copy the pumping reducing QP offset from the top CTU neighbor
775
0
              tempCS->currQP[partitioner.chType] = tempCS->baseQP =
776
0
              bestCS->currQP[partitioner.chType] = bestCS->baseQP = tempCS->baseQP - m_globalCtuQpVector->at (ctuRsAddr);
777
0
            }
778
0
            tempCS->currQP[partitioner.chType] = tempCS->baseQP =
779
0
            bestCS->currQP[partitioner.chType] = bestCS->baseQP = Clip3 (0, MAX_QP, tempCS->baseQP + m_tempQpDiff);
780
0
          }
781
0
        }
782
3.57k
        else
783
3.57k
        {
784
3.57k
          tempCS->currQP[partitioner.chType] = tempCS->baseQP =
785
3.57k
          bestCS->currQP[partitioner.chType] = bestCS->baseQP = pic->ctuAdaptedQP[ctuRsAddr];
786
3.57k
        }
787
788
3.57k
        int deltaQP       = slice.sliceQp - tempCS->baseQP;
789
3.57k
        if( abs( deltaQP ) > m_pcEncCfg->m_maxDeltaQP )
790
0
        {
791
0
          clippedMaxDQP = deltaQP > 0 ? slice.sliceQp - m_pcEncCfg->m_maxDeltaQP : slice.sliceQp + m_pcEncCfg->m_maxDeltaQP;
792
0
          tempCS->currQP[partitioner.chType] = tempCS->baseQP =
793
0
          bestCS->currQP[partitioner.chType] = bestCS->baseQP = clippedMaxDQP;
794
0
        }
795
796
        //the ctuAdaptedQP correspones with the ctuQpaLambda
797
3.57k
        setUpLambda (slice, pic->ctuQpaLambda[ctuRsAddr], pic->ctuAdaptedQP[ctuRsAddr], false, true);
798
3.57k
      }
799
0
      else // isBimEnabled without QPA
800
0
      {
801
0
        const int baseQp  = tempCS->baseQP;
802
0
        int adaptedLumaQP = 0;
803
0
        int lumaQPoffset  = 0;
804
        
805
0
        if( m_pcEncCfg->m_lumaLevelToDeltaQPEnabled == 1 )
806
0
        {
807
0
          uint32_t meanLuma = 0;
808
0
          if( ( uiLPelX + lumaArea.width > pps.picWidthInLumaSamples ) || ( uiTPelY + lumaArea.height > pps.picHeightInLumaSamples ) )
809
0
          {
810
0
            int wdth = ( uiLPelX + lumaArea.width > pps.picWidthInLumaSamples )   ? pps.picWidthInLumaSamples - uiLPelX  : lumaArea.width;
811
0
            int hght = ( uiTPelY + lumaArea.height > pps.picHeightInLumaSamples ) ? pps.picHeightInLumaSamples - uiTPelY : lumaArea.height;
812
            
813
0
            int sum = 0;
814
0
            for( int y = 0; y < hght; y++ )
815
0
            {
816
0
              for( int x = 0; x < wdth; x++ )
817
0
              {
818
0
                sum += tempCS->getOrgBuf().Y().at( x, y );
819
0
              }
820
0
            }
821
0
            double avg = (double) sum / ( wdth*hght );
822
0
            meanLuma = (uint32_t)( avg + 0.5 );
823
0
          }
824
0
          else
825
0
          {
826
0
            meanLuma = tempCS->getOrgBuf().Y().getAvg();
827
0
          }
828
829
0
          int lumaBD     = m_pcEncCfg->m_internalBitDepth[ CH_L ];
830
0
          int lumaIdxOrg = Clip3<int>( 0, int( 1 << lumaBD ) - 1, meanLuma );
831
0
          int lumaIdx    = lumaBD < 10 ? lumaIdxOrg << (10 - lumaBD) : lumaBD > 10 ? lumaIdxOrg >> (lumaBD - 10) : lumaIdxOrg;
832
0
          lumaQPoffset   = m_lumaLevelToDeltaQPLUT[lumaIdx];
833
0
          adaptedLumaQP  = Clip3( -sps.qpBDOffset[CH_L], MAX_QP, baseQp - lumaQPoffset );
834
0
        }
835
        
836
0
        if( isBimEnabled )
837
0
        {
838
0
          const unsigned bimQpSize = (unsigned) bestCS->picture->m_picShared->m_ctuBimQpOffset.size();
839
0
          uint32_t ctuAddr         = ctuRsAddr;
840
841
0
          double avgBimOffset = 0.0;
842
843
0
          if (bimQpSize != pcv.sizeInCtus) // re-calculate correct address of BIM CTU QP offset
844
0
          {
845
0
            const unsigned bimCtuSize  = m_pcEncCfg->m_bimCtuSize;
846
0
            const unsigned bimCtuWidth = (pcv.lumaWidth + bimCtuSize - 1) / bimCtuSize;
847
848
0
            ctuAddr = getCtuAddrFromCtuSize (partitioner.currQgPos, Log2 (bimCtuSize), bimCtuWidth);
849
0
            CHECK (ctuAddr >= bimQpSize, "ctuAddr exceeds size of m_ctuBimQpOffset");
850
851
0
            int maxY = m_pcEncCfg->m_CTUSize / bimCtuSize;
852
0
            int maxX = m_pcEncCfg->m_CTUSize / bimCtuSize;
853
0
            if( ( partitioner.currQgPos.y + maxY * bimCtuSize ) > pcv.lumaHeight )
854
0
            {
855
0
              int reduceHeight = ( partitioner.currQgPos.y + ( maxY - 1 ) * bimCtuSize ) - pcv.lumaHeight;
856
0
              int reduceY = ( reduceHeight + bimCtuSize ) / bimCtuSize;
857
0
              maxY = maxY - reduceY;
858
0
            }
859
0
            if( ( partitioner.currQgPos.x + maxX * bimCtuSize ) > pcv.lumaWidth )
860
0
            {
861
0
              int reduceWidth = ( partitioner.currQgPos.x + ( maxX - 1 ) * bimCtuSize ) - pcv.lumaWidth;
862
0
              int reduceX = ( reduceWidth + bimCtuSize ) / bimCtuSize;
863
0
              maxX = maxX - reduceX;
864
0
            }
865
0
            for (int y = 0; y < maxY * bimCtuSize; y = y + bimCtuSize)
866
0
            {
867
0
              for (int x = 0; x < maxX * bimCtuSize; x = x + bimCtuSize)
868
0
              {
869
0
                int theBlockId = ctuAddr + (y / bimCtuSize) * ((pcv.lumaWidth + bimCtuSize - 1) / bimCtuSize) + (x / bimCtuSize);
870
0
                avgBimOffset += pic->m_picShared->m_ctuBimQpOffset[theBlockId];
871
0
              }
872
0
            }
873
0
            if (maxY * maxX > 0)
874
0
            {
875
0
              avgBimOffset = avgBimOffset / (double)(maxY * maxX);
876
0
            }
877
0
            else
878
0
            {
879
0
              avgBimOffset = 0.0;
880
0
            }
881
0
          }
882
0
          else
883
0
          {
884
0
            avgBimOffset = pic->m_picShared->m_ctuBimQpOffset[ctuAddr];
885
0
          }
886
0
          int intAvgBimOffset = avgBimOffset >= 0 ? (int)(0.5 + avgBimOffset) : (int)(-0.5 + avgBimOffset);
887
0
          adaptedLumaQP       = Clip3 (-sps.qpBDOffset[CH_L], MAX_QP, baseQp - lumaQPoffset + intAvgBimOffset );
888
0
        }
889
        
890
0
        tempCS->currQP[partitioner.chType] = tempCS->baseQP =
891
0
        bestCS->currQP[partitioner.chType] = bestCS->baseQP = adaptedLumaQP;
892
        
893
0
        int deltaQP         = slice.sliceQp - adaptedLumaQP;
894
0
        if( abs( deltaQP ) > m_pcEncCfg->m_maxDeltaQP )
895
0
        {
896
0
          int clippedMaxDQP = deltaQP > 0 ? slice.sliceQp - m_pcEncCfg->m_maxDeltaQP : slice.sliceQp + m_pcEncCfg->m_maxDeltaQP;
897
0
          tempCS->currQP[partitioner.chType] = tempCS->baseQP =
898
0
          bestCS->currQP[partitioner.chType] = bestCS->baseQP = clippedMaxDQP;
899
0
        }
900
901
0
        updateLambda (slice, slice.getLambdas()[0], baseQp, adaptedLumaQP, true);
902
0
      }
903
3.57k
    }
904
3.08k
    else if (m_pcEncCfg->m_usePerceptQPA && slice.isIntra()) // currSubdiv 2 - use sub-CTU QPA
905
3.08k
    {
906
3.08k
      CHECK ((partitioner.currArea().lwidth() >= pcv.maxCUSize) || (partitioner.currArea().lheight() >= pcv.maxCUSize), "sub-CTU delta-QP error");
907
3.08k
      tempCS->currQP[partitioner.chType] = tempCS->baseQP = BitAllocation::applyQPAdaptationSubCtu (&slice, m_pcEncCfg, lumaArea, m_pcRateCtrl->getMinNoiseLevels());
908
909
3.08k
      if (m_pcEncCfg->m_internalUsePerceptQPATempFiltISlice == 2)
910
0
      {
911
0
        tempCS->currQP[partitioner.chType] = tempCS->baseQP = Clip3 (0, MAX_QP, tempCS->baseQP + m_tempQpDiff);
912
0
      }
913
914
3.08k
      int backupBaseQP  = tempCS->baseQP;
915
3.08k
      int deltaQP       = slice.sliceQp - tempCS->baseQP;
916
3.08k
      int clippedMaxDQP = 0;
917
3.08k
      if( abs( deltaQP ) > m_pcEncCfg->m_maxDeltaQP )
918
0
      {
919
0
        clippedMaxDQP = deltaQP > 0 ? slice.sliceQp - m_pcEncCfg->m_maxDeltaQP : slice.sliceQp + m_pcEncCfg->m_maxDeltaQP;
920
0
        tempCS->currQP[partitioner.chType] = tempCS->baseQP = clippedMaxDQP;
921
0
      }
922
3.08k
      updateLambda( slice, pic->ctuQpaLambda[ctuRsAddr], pic->ctuAdaptedQP[ctuRsAddr], backupBaseQP, true );
923
3.08k
    }
924
6.65k
  }
925
926
124k
  if (partitioner.currQtDepth == 0)
927
7.15k
  {
928
7.15k
    m_MergeSimpleFlag = 0;
929
7.15k
  }
930
124k
  m_modeCtrl.initCULevel( partitioner, *tempCS, m_MergeSimpleFlag );
931
124k
  m_sbtCostSave[0] = m_sbtCostSave[1] = MAX_DOUBLE;
932
933
124k
  m_CurrCtx->start = m_CABACEstimator->getCtx();
934
935
124k
  m_cuChromaQpOffsetIdxPlus1 = 0;
936
937
124k
  if( slice.chromaQpAdjEnabled && partitioner.currQgChromaEnable() )
938
0
  {
939
    // TODO M0133 : double check encoder decisions with respect to chroma QG detection and actual encode
940
0
    int cuChromaQpOffsetSubdiv = slice.isIntra() ? slice.picHeader->cuChromaQpOffsetSubdivIntra : slice.picHeader->cuChromaQpOffsetSubdivInter;
941
0
    int lgMinCuSize = sps.log2MinCodingBlockSize +
942
0
      std::max<int>(0, floorLog2(sps.CTUSize) - sps.log2MinCodingBlockSize - int((cuChromaQpOffsetSubdiv + 1) / 2));
943
0
    m_cuChromaQpOffsetIdxPlus1 = ( ( uiLPelX >> lgMinCuSize ) + ( uiTPelY >> lgMinCuSize ) ) % ( pps.chromaQpOffsetListLen + 1 );
944
0
  }
945
946
124k
  DTRACE_UPDATE( g_trace_ctx, std::make_pair( "cux", uiLPelX ) );
947
124k
  DTRACE_UPDATE( g_trace_ctx, std::make_pair( "cuy", uiTPelY ) );
948
124k
  DTRACE_UPDATE( g_trace_ctx, std::make_pair( "cuw", tempCS->area.lwidth() ) );
949
124k
  DTRACE_UPDATE( g_trace_ctx, std::make_pair( "cuh", tempCS->area.lheight() ) );
950
124k
  DTRACE( g_trace_ctx, D_COMMON, "@(%4d,%4d) [%2dx%2d]\n", tempCS->area.lx(), tempCS->area.ly(), tempCS->area.lwidth(), tempCS->area.lheight() );
951
952
124k
  if( tempCS->slice->checkLDC )
953
124k
  {
954
124k
    m_bestBcwCost[0] = m_bestBcwCost[1] = std::numeric_limits<double>::max();
955
124k
    m_bestBcwIdx[0] = m_bestBcwIdx[1] = -1;
956
124k
  }
957
958
124k
  m_cInterSearch.resetSavedAffineMotion();
959
124k
  {
960
124k
    const ComprCUCtx &cuECtx      = *m_modeCtrl.comprCUCtx;
961
124k
    const CodingStructure& cs     = *tempCS;
962
124k
    const PartSplit implicitSplit = partitioner.getImplicitSplit( cs );
963
124k
    const bool isBoundary         = implicitSplit != CU_DONT_SPLIT;
964
124k
    const bool lossless           = false;
965
124k
    int qp                        = cs.baseQP;
966
967
#if ENABLE_MEASURE_SEARCH_SPACE
968
    if( !isBoundary )
969
    {
970
      g_searchSpaceAcc.addPartition( partitioner.currArea(), partitioner.isSepTree( *tempCS ) ? partitioner.chType : MAX_NUM_CH );
971
    }
972
973
#endif
974
124k
    if( ! isBoundary )
975
79.0k
    {
976
79.0k
      if (pps.useDQP && partitioner.isSepTree (*tempCS) && isChroma (partitioner.chType))
977
54.9k
      {
978
54.9k
        const ChromaFormat chromaFm = tempCS->area.chromaFormat;
979
54.9k
        const Position chromaCentral (tempCS->area.Cb().chromaPos().offset (tempCS->area.Cb().chromaSize().width >> 1, tempCS->area.Cb().chromaSize().height >> 1));
980
54.9k
        const Position lumaRefPos (chromaCentral.x << getChannelTypeScaleX (CH_C, chromaFm), chromaCentral.y << getChannelTypeScaleY (CH_C, chromaFm));
981
54.9k
        const CodingUnit* colLumaCu = bestCS->lumaCS->getCU (lumaRefPos, CH_L, TREE_D);
982
        // update qp
983
54.9k
        qp = colLumaCu->qp;
984
54.9k
      }
985
986
79.0k
      m_cIntraSearch.reset();
987
988
79.0k
      bool isReuseCU = m_modeCtrl.isReusingCuValid( cs, partitioner, qp );
989
990
79.0k
      bool checkIbc = m_pcEncCfg->m_IBCMode && bestCS->picture->useIBC && (partitioner.chType == CH_L);
991
79.0k
      if ((m_pcEncCfg->m_IBCFastMethod>3) && (cs.area.lwidth() * cs.area.lheight()) > (16 * 16))
992
0
      {
993
0
        checkIbc = false;
994
0
      }
995
79.0k
      if( isReuseCU )
996
0
      {
997
0
        xReuseCachedResult( tempCS, bestCS, partitioner );
998
0
      }
999
79.0k
      else
1000
79.0k
      {
1001
        // add first pass modes
1002
79.0k
        if ( !slice.isIntra() && !slice.isIRAP() && !( cs.area.lwidth() == 4 && cs.area.lheight() == 4 ) && !partitioner.isConsIntra() )
1003
0
        {
1004
          // add inter modes
1005
0
          EncTestMode encTestModeSkip = { ETM_MERGE_SKIP, ETO_STANDARD, qp, lossless };
1006
0
          if (m_modeCtrl.tryMode(encTestModeSkip, cs, partitioner))
1007
0
          {
1008
0
            xCheckRDCostUnifiedMerge(tempCS, bestCS, partitioner, encTestModeSkip);
1009
1010
0
            CodingUnit* cu = bestCS->getCU(partitioner.chType, partitioner.treeType);
1011
0
            if (cu)
1012
0
              cu->mmvdSkip = cu->skip == false ? false : cu->mmvdSkip;
1013
0
          }
1014
0
          EncTestMode encTestMode = { ETM_INTER_ME, ETO_STANDARD, qp, lossless };
1015
0
          if (m_modeCtrl.tryMode(encTestMode, cs, partitioner))
1016
0
          {
1017
0
            xCheckRDCostInter(tempCS, bestCS, partitioner, encTestMode);
1018
0
          }
1019
1020
0
          if (m_pcEncCfg->m_AMVRspeed)
1021
0
          {
1022
0
            double bestIntPelCost = MAX_DOUBLE;
1023
1024
0
            EncTestMode encTestMode = {ETM_INTER_IMV, ETO_STANDARD, qp, lossless};
1025
0
            if( m_modeCtrl.tryMode( encTestMode, cs, partitioner ) )
1026
0
            {
1027
0
              const bool skipAltHpelIF = ( int( ( encTestMode.opts & ETO_IMV ) >> ETO_IMV_SHIFT ) == 4 ) && ( bestIntPelCost > 1.25 * bestCS->cost );
1028
0
              if (!skipAltHpelIF)
1029
0
              {
1030
0
                xCheckRDCostInterIMV(tempCS, bestCS, partitioner, encTestMode );
1031
0
              }
1032
0
            }
1033
0
          }
1034
0
        }
1035
1036
79.0k
        if (checkIbc && !partitioner.isConsInter())
1037
24.0k
        {
1038
24.0k
          EncTestMode encTestModeIBCMerge = { ETM_IBC_MERGE, ETO_STANDARD, qp, lossless };
1039
24.0k
          if ((m_pcEncCfg->m_IBCFastMethod < 4) && (partitioner.chType == CH_L) && m_modeCtrl.tryMode(encTestModeIBCMerge, cs, partitioner))
1040
24.0k
          {
1041
24.0k
            xCheckRDCostIBCModeMerge2Nx2N(tempCS, bestCS, partitioner, encTestModeIBCMerge);
1042
24.0k
          }
1043
1044
24.0k
          EncTestMode encTestModeIBC = { ETM_IBC, ETO_STANDARD, qp, lossless };
1045
24.0k
          if (m_modeCtrl.tryMode(encTestModeIBC, cs, partitioner))
1046
24.0k
          {
1047
24.0k
            xCheckRDCostIBCMode(tempCS, bestCS, partitioner, encTestModeIBC);
1048
24.0k
          }
1049
24.0k
        }
1050
79.0k
        if( m_EDO && bestCS->cost != MAX_DOUBLE )
1051
21.4k
        {
1052
21.4k
          xCalDebCost(*bestCS, partitioner);
1053
21.4k
        }
1054
1055
        // add intra modes
1056
79.0k
        EncTestMode encTestMode( {ETM_INTRA, ETO_STANDARD, qp, lossless} );
1057
79.0k
        if( !partitioner.isConsInter() && m_modeCtrl.tryMode( encTestMode, cs, partitioner ) )
1058
79.0k
        {
1059
79.0k
          xCheckRDCostIntra( tempCS, bestCS, partitioner, encTestMode );
1060
79.0k
        }
1061
79.0k
      } // reusing cu
1062
1063
79.0k
      m_modeCtrl.beforeSplit( partitioner );
1064
1065
79.0k
      if (cuECtx.bestCS && ((cuECtx.bestCostNoImv == (MAX_DOUBLE * .5) || cuECtx.isReusingCu) && !slice.isIntra()) )
1066
0
      {
1067
0
        m_cInterSearch.loadGlobalUniMvs( lumaArea, *pps.pcv );
1068
0
      }
1069
1070
79.0k
      if (!cs.slice->isIntra() && (partitioner.chType == CH_L) && ( m_pcEncCfg->m_qtbttSpeedUpMode & 2) && (partitioner.currQtDepth < 3) && bestCS->cus.size())
1071
0
      {
1072
0
        int flagDbefore = (bestCS->cus[0]->mergeFlag && !bestCS->cus[0]->mmvdMergeFlag && !bestCS->cus[0]->ispMode && !bestCS->cus[0]->geo) ? 1 : 0;
1073
0
        if (partitioner.currQtDepth == 0)
1074
0
        {
1075
0
          m_MergeSimpleFlag = flagDbefore;
1076
0
        }
1077
0
        else
1078
0
        {
1079
0
          int markFlag = (partitioner.currQtDepth == 1) ? 1 : 3;
1080
0
          m_MergeSimpleFlag = (flagDbefore << partitioner.currQtDepth) | (m_MergeSimpleFlag & markFlag);
1081
0
        }
1082
0
      }
1083
79.0k
    } //boundary
1084
1085
124k
    if( ( m_pcEncCfg->m_IntraPeriod == 1 ) && ( partitioner.chType == CH_C ) )
1086
0
    {
1087
0
      xCheckFastCuChromaSplitting( tempCS, bestCS, partitioner, *m_modeCtrl.comprCUCtx );
1088
0
    }
1089
    //////////////////////////////////////////////////////////////////////////
1090
    // split modes
1091
124k
    EncTestMode lastTestMode;
1092
1093
124k
    if( cuECtx.qtBeforeBt )
1094
19.9k
    {
1095
19.9k
      EncTestMode encTestMode( { ETM_SPLIT_QT, ETO_STANDARD, qp, false } );
1096
19.9k
      if( m_modeCtrl.trySplit( encTestMode, cs, partitioner, lastTestMode ) )
1097
12.8k
      {
1098
12.8k
        lastTestMode = encTestMode;
1099
12.8k
        xCheckModeSplit( tempCS, bestCS, partitioner, encTestMode );
1100
12.8k
      }
1101
19.9k
    }
1102
1103
124k
    if( partitioner.canSplit( CU_HORZ_SPLIT, cs ) )
1104
66.2k
    {
1105
      // add split modes
1106
66.2k
      EncTestMode encTestMode( { ETM_SPLIT_BT_H, ETO_STANDARD, qp, false } );
1107
66.2k
      if( m_modeCtrl.trySplit( encTestMode, cs, partitioner, lastTestMode ) )
1108
66.2k
      {
1109
66.2k
        lastTestMode = encTestMode;
1110
66.2k
        xCheckModeSplit( tempCS, bestCS, partitioner, encTestMode );
1111
66.2k
      }
1112
66.2k
    }
1113
1114
124k
    if( partitioner.canSplit( CU_VERT_SPLIT, cs ) )
1115
62.8k
    {
1116
      // add split modes
1117
62.8k
      EncTestMode encTestMode( { ETM_SPLIT_BT_V, ETO_STANDARD, qp, false } );
1118
62.8k
      if( m_modeCtrl.trySplit( encTestMode, cs, partitioner, lastTestMode ) )
1119
62.8k
      {
1120
62.8k
        lastTestMode = encTestMode;
1121
62.8k
        xCheckModeSplit( tempCS, bestCS, partitioner, encTestMode );
1122
62.8k
      }
1123
62.8k
    }
1124
1125
124k
    if( partitioner.canSplit( CU_TRIH_SPLIT, cs ) )
1126
28.0k
    {
1127
      // add split modes
1128
28.0k
      EncTestMode encTestMode( { ETM_SPLIT_TT_H, ETO_STANDARD, qp, false } );
1129
28.0k
      if( m_modeCtrl.trySplit( encTestMode, cs, partitioner, lastTestMode ) )
1130
1.50k
      {
1131
1.50k
        lastTestMode = encTestMode;
1132
1.50k
        xCheckModeSplit( tempCS, bestCS, partitioner, encTestMode );
1133
1.50k
      }
1134
28.0k
    }
1135
1136
124k
    if( partitioner.canSplit( CU_TRIV_SPLIT, cs ) )
1137
22.1k
    {
1138
      // add split modes
1139
22.1k
      EncTestMode encTestMode( { ETM_SPLIT_TT_V, ETO_STANDARD, qp, false } );
1140
22.1k
      if( m_modeCtrl.trySplit( encTestMode, cs, partitioner, lastTestMode ) )
1141
1.15k
      {
1142
1.15k
        lastTestMode = encTestMode;
1143
1.15k
        xCheckModeSplit( tempCS, bestCS, partitioner, encTestMode );
1144
1.15k
      }
1145
22.1k
    }
1146
1147
124k
    if( !cuECtx.qtBeforeBt )
1148
104k
    {
1149
104k
      EncTestMode encTestMode( { ETM_SPLIT_QT, ETO_STANDARD, qp, false } );
1150
104k
      if( m_modeCtrl.trySplit( encTestMode, cs, partitioner, lastTestMode ) )
1151
25.7k
      {
1152
25.7k
        lastTestMode = encTestMode;
1153
25.7k
        xCheckModeSplit( tempCS, bestCS, partitioner, encTestMode );
1154
25.7k
      }
1155
104k
    }
1156
124k
  }
1157
1158
124k
  if( bestCS->cus.empty() )
1159
0
  {
1160
0
    m_modeCtrl.finishCULevel( partitioner );
1161
0
    return;
1162
0
  }
1163
1164
  //////////////////////////////////////////////////////////////////////////
1165
  // Finishing CU
1166
  // set context states
1167
124k
  m_CABACEstimator->getCtx() = m_CurrCtx->best;
1168
1169
  // QP from last processed CU for further processing
1170
  //copy the qp of the last non-chroma CU
1171
124k
  int numCUInThisNode = (int)bestCS->cus.size();
1172
124k
  if( numCUInThisNode > 1 && bestCS->cus.back()->chType == CH_C && !CS::isDualITree( *bestCS ) )
1173
0
  {
1174
0
    CHECK( bestCS->cus[numCUInThisNode-2]->chType != CH_L, "wrong chType" );
1175
0
    bestCS->prevQP[partitioner.chType] = bestCS->cus[numCUInThisNode-2]->qp;
1176
0
  }
1177
124k
  else
1178
124k
  {
1179
124k
    bestCS->prevQP[partitioner.chType] = bestCS->cus.back()->qp;
1180
124k
  }
1181
124k
  if( ( !slice.isIntra() || slice.sps->IBC )
1182
124k
    && partitioner.chType == CH_L
1183
49.1k
    && bestCS->cus.size() == 1 && ( bestCS->cus.back()->predMode == MODE_INTER || bestCS->cus.back()->predMode == MODE_IBC )
1184
0
    && bestCS->area.Y() == (*bestCS->cus.back()).Y() )
1185
0
  {
1186
0
    const CodingUnit& cu = *bestCS->cus.front();
1187
0
    bool isIbcSmallBlk = CU::isIBC(cu) && (cu.lwidth() * cu.lheight() <= 16);
1188
0
    if (!cu.affine && !cu.geo && !isIbcSmallBlk)
1189
0
    {
1190
0
      const MotionInfo &mi = cu.getMotionInfo();
1191
0
      HPMVInfo hMi( mi, ( mi.interDir() == 3 ) ? cu.BcwIdx : BCW_DEFAULT, cu.imv == IMV_HPEL, CU::isIBC( cu ) );
1192
0
      cu.cs->addMiToLut( CU::isIBC( cu ) ? cu.cs->motionLut.lutIbc : cu.cs->motionLut.lut, hMi );
1193
0
    }
1194
0
  }
1195
1196
124k
  m_modeCtrl.finishCULevel( partitioner );
1197
124k
  if( m_cIntraSearch.getSaveCuCostInSCIPU() && bestCS->cus.size() == 1 )
1198
0
  {
1199
0
    m_cIntraSearch.saveCuAreaCostInSCIPU( Area( partitioner.currArea().lumaPos(), partitioner.currArea().lumaSize() ), bestCS->cost );
1200
0
  }
1201
1202
  // Assert if Best prediction mode is NONE
1203
  // Selected mode's RD-cost must be not MAX_DOUBLE.
1204
124k
  CHECK( bestCS->cus.empty()                                   , "No possible encoding found" );
1205
124k
  CHECK( bestCS->cus[0]->predMode == NUMBER_OF_PREDICTION_MODES, "No possible encoding found" );
1206
124k
  CHECK( bestCS->cost             == MAX_DOUBLE                , "No possible encoding found" );
1207
124k
}
1208
1209
1210
void EncCu::xCheckModeSplit(CodingStructure *&tempCS, CodingStructure *&bestCS, Partitioner &partitioner, const EncTestMode& encTestMode )
1211
170k
{
1212
170k
  const ModeType modeTypeParent  = partitioner.modeType;
1213
170k
  const TreeType treeTypeParent  = partitioner.treeType;
1214
170k
  const ChannelType chTypeParent = partitioner.chType;
1215
1216
170k
  int signalModeConsVal = CS::signalModeCons( *tempCS, partitioner.currArea(), getPartSplit(encTestMode), modeTypeParent);
1217
170k
  int numRoundRdo = signalModeConsVal == LDT_MODE_TYPE_SIGNAL ? 2 : 1;
1218
170k
  bool skipInterPass = false;
1219
340k
  for( int i = 0; i < numRoundRdo; i++ )
1220
170k
  {
1221
    //change cons modes
1222
170k
    if( signalModeConsVal == LDT_MODE_TYPE_SIGNAL )
1223
0
    {
1224
0
      CHECK( numRoundRdo != 2, "numRoundRdo shall be 2 - [LDT_MODE_TYPE_SIGNAL]" );
1225
0
      partitioner.modeType = (i == 0) ? MODE_TYPE_INTER : MODE_TYPE_INTRA;
1226
0
    }
1227
170k
    else if( signalModeConsVal == LDT_MODE_TYPE_INFER )
1228
0
    {
1229
0
      CHECK( numRoundRdo != 1, "numRoundRdo shall be 1 - [LDT_MODE_TYPE_INFER]" );
1230
0
      partitioner.modeType = MODE_TYPE_INTRA;
1231
0
    }
1232
170k
    else if( signalModeConsVal == LDT_MODE_TYPE_INHERIT )
1233
170k
    {
1234
170k
      CHECK( numRoundRdo != 1, "numRoundRdo shall be 1 - [LDT_MODE_TYPE_INHERIT]" );
1235
170k
      partitioner.modeType = modeTypeParent;
1236
170k
    }
1237
1238
    //for lite intra encoding fast algorithm, set the status to save inter coding info
1239
170k
    if( modeTypeParent == MODE_TYPE_ALL && partitioner.modeType == MODE_TYPE_INTER )
1240
0
    {
1241
0
      m_cIntraSearch.setSaveCuCostInSCIPU( true );
1242
0
      m_cIntraSearch.setNumCuInSCIPU( 0 );
1243
0
    }
1244
170k
    else if( modeTypeParent == MODE_TYPE_ALL && partitioner.modeType != MODE_TYPE_INTER )
1245
170k
    {
1246
170k
      m_cIntraSearch.setSaveCuCostInSCIPU( false );
1247
170k
      if( partitioner.modeType == MODE_TYPE_ALL )
1248
170k
      {
1249
170k
        m_cIntraSearch.setNumCuInSCIPU( 0 );
1250
170k
      }
1251
170k
    }
1252
1253
170k
    xCheckModeSplitInternal( tempCS, bestCS, partitioner, encTestMode, modeTypeParent, skipInterPass );
1254
    //recover cons modes
1255
170k
    partitioner.modeType = modeTypeParent;
1256
170k
    partitioner.treeType = treeTypeParent;
1257
170k
    partitioner.chType = chTypeParent;
1258
170k
    if( modeTypeParent == MODE_TYPE_ALL )
1259
170k
    {
1260
170k
      m_cIntraSearch.setSaveCuCostInSCIPU( false );
1261
170k
      if( numRoundRdo == 2 && partitioner.modeType == MODE_TYPE_INTRA )
1262
0
      {
1263
0
        m_cIntraSearch.initCuAreaCostInSCIPU();
1264
0
      }
1265
170k
    }
1266
170k
    if( skipInterPass )
1267
0
    {
1268
0
      break;
1269
0
    }
1270
170k
  }
1271
170k
}
1272
1273
void EncCu::xCheckModeSplitInternal(CodingStructure *&tempCS, CodingStructure *&bestCS, Partitioner &partitioner, const EncTestMode& encTestMode, const ModeType modeTypeParent, bool& skipInterPass )
1274
170k
{
1275
170k
  const int qp                     = encTestMode.qp;
1276
170k
  const int oldPrevQp              = tempCS->prevQP[partitioner.chType];
1277
170k
  const auto oldMotionLut          = tempCS->motionLut;
1278
                                   
1279
170k
  const PartSplit split            = getPartSplit( encTestMode );
1280
170k
  const ModeType  modeTypeChild    = partitioner.modeType;
1281
1282
170k
  CHECK( !( split == CU_QUAD_SPLIT || split == CU_HORZ_SPLIT || split == CU_VERT_SPLIT
1283
170k
         || split == CU_TRIH_SPLIT || split == CU_TRIV_SPLIT ), "invalid split type" );
1284
1285
170k
  tempCS->initStructData( qp );
1286
1287
170k
  m_CABACEstimator->getCtx()       = m_CurrCtx->start;
1288
1289
170k
  const uint16_t split_ctx_size    = Ctx::SplitFlag.size() + Ctx::SplitQtFlag.size() + Ctx::SplitHvFlag.size() + Ctx::Split12Flag.size() + Ctx::ModeConsFlag.size();
1290
170k
  const TempCtx  ctxSplitFlags     ( m_CtxCache, SubCtx( CtxSet( Ctx::SplitFlag(), split_ctx_size ), m_CABACEstimator->getCtx() ) );
1291
1292
170k
  m_CABACEstimator->determineNeighborCus( *tempCS, partitioner.currArea(), partitioner.chType, partitioner.treeType );
1293
170k
  m_CABACEstimator->resetBits           ();
1294
170k
  m_CABACEstimator->split_cu_mode       ( split, *tempCS, partitioner );
1295
170k
  partitioner     . modeType            = modeTypeParent;
1296
170k
  m_CABACEstimator->mode_constraint     ( split, *tempCS, partitioner, modeTypeChild );
1297
170k
  partitioner     . modeType            = modeTypeChild;
1298
1299
170k
  const int64_t splitBits   = m_CABACEstimator->getEstFracBits();
1300
1301
170k
  const bool chromaNotSplit = modeTypeParent == MODE_TYPE_ALL && modeTypeChild == MODE_TYPE_INTRA;
1302
170k
  const bool isChromaTooBig = isChromaEnabled( tempCS->pps->pcv->chrFormat ) && tempCS->area.Y().maxDim() > tempCS->sps->getMaxTbSize();
1303
170k
  bool       skipSplitTest  = chromaNotSplit && isChromaTooBig;
1304
1305
170k
  if( !skipSplitTest )
1306
170k
  {
1307
170k
    double         a = -1, b = -1;
1308
170k
    const unsigned w       = partitioner.currArea().lwidth();
1309
170k
    const unsigned h       = partitioner.currArea().lheight();
1310
170k
    const bool contextCond = w == h && tempCS->slice->sliceType == VVENC_B_SLICE && isLuma( partitioner.chType ) && m_pcEncCfg->m_splitCostThrParamId >= 0 && m_pcEncCfg->m_splitCostThrParamId <= 3;
1311
1312
170k
    if( contextCond )
1313
0
    {
1314
0
      uint8_t nsPredInd = m_modeCtrl.comprCUCtx->bestNsPredMode.type == ETM_INTRA;
1315
0
      uint8_t szInd     = getLog2( w ) - 3;
1316
0
      uint8_t splitInd  = split == CU_QUAD_SPLIT ? 1 : 0;
1317
0
      if ( m_pcEncCfg->m_splitCostThrParamId <= 1 )
1318
0
      {
1319
0
        a = coefSquareCUsFasterFastMedium[m_pcEncCfg->m_splitCostThrParamId][szInd][nsPredInd][splitInd][0];
1320
0
        b = coefSquareCUsFasterFastMedium[m_pcEncCfg->m_splitCostThrParamId][szInd][nsPredInd][splitInd][1];
1321
0
      }
1322
0
      else
1323
0
      {
1324
0
        uint8_t mtInd = (partitioner.currMtDepth == 0);
1325
0
        a = coefSquareCUsSlowSlower[m_pcEncCfg->m_splitCostThrParamId - 2][szInd][nsPredInd][partitioner.currQtDepth][mtInd][splitInd][0];
1326
0
        b = coefSquareCUsSlowSlower[m_pcEncCfg->m_splitCostThrParamId - 2][szInd][nsPredInd][partitioner.currQtDepth][mtInd][splitInd][1];
1327
0
      }
1328
0
    }
1329
1330
170k
    if( a > -1 && b > -1 )
1331
0
    {
1332
0
      const double bestNsCost    = m_modeCtrl.comprCUCtx->bestCostBeforeSplit == MAX_DOUBLE ? -1 : m_modeCtrl.comprCUCtx->bestCostBeforeSplit;
1333
0
      const double factor        = 1.0 + b * exp( a * qp );
1334
0
      const double predSplitCost = bestNsCost / factor + splitBits;
1335
0
      skipSplitTest              = bestNsCost >= 0 && predSplitCost >= bestNsCost;
1336
0
    }
1337
170k
    else
1338
170k
    {
1339
170k
      int numChild = 3;
1340
170k
      if( split == CU_VERT_SPLIT || split == CU_HORZ_SPLIT ) numChild--;
1341
41.2k
      else if( split == CU_QUAD_SPLIT ) numChild++;
1342
1343
170k
      int64_t approxBits = m_pcEncCfg->m_qtbttSpeedUp > 0 ? numChild << SCALE_BITS : 0;
1344
1345
170k
      const double factor     = ( tempCS->currQP[partitioner.chType] > 30                              ? 1.1  : 1.075 ) +
1346
170k
                                (   m_pcEncCfg->m_qtbttSpeedUp > 0                                     ? 0.01 : 0.0   ) +
1347
170k
                                ( ( m_pcEncCfg->m_qtbttSpeedUp > 0 && isChroma( partitioner.chType ) ) ? 0.2  : 0.0   );
1348
       
1349
170k
      const double baseCost   = bestCS->cost + bestCS->costDbOffset;
1350
170k
      const double predCost   = baseCost / factor + splitBits + approxBits;
1351
170k
      skipSplitTest           = predCost >= baseCost;
1352
170k
    }
1353
170k
  }
1354
1355
170k
  if( skipSplitTest )
1356
106k
  {
1357
106k
    m_CABACEstimator->getCtx() = SubCtx( CtxSet( Ctx::SplitFlag(), split_ctx_size ), ctxSplitFlags );
1358
106k
    xCheckBestMode( tempCS, bestCS, partitioner, encTestMode );
1359
106k
    return;
1360
106k
  }
1361
1362
64.0k
  if( partitioner.treeType == TREE_D )
1363
64.0k
  {
1364
64.0k
    if( chromaNotSplit )
1365
0
    {
1366
0
      CHECK( partitioner.chType != CH_L, "chType must be luma" );
1367
0
      partitioner.treeType = TREE_L;
1368
0
    }
1369
64.0k
    else
1370
64.0k
    {
1371
64.0k
      partitioner.treeType = TREE_D;
1372
64.0k
    }
1373
64.0k
  }
1374
1375
64.0k
  partitioner.splitCurrArea( split, *tempCS );
1376
64.0k
  bool qgEnableChildren = partitioner.currQgEnable(); // QG possible at children level
1377
1378
64.0k
  m_CurrCtx++;
1379
1380
64.0k
  AffineMVInfo tmpMVInfo;
1381
64.0k
  bool isAffMVInfoSaved = m_cInterSearch.m_AffineProfList->savePrevAffMVInfo( 0, tmpMVInfo );
1382
1383
64.0k
  BlkUniMvInfo tmpUniMvInfo;
1384
64.0k
  bool         isUniMvInfoSaved = false;
1385
64.0k
  if( !tempCS->slice->isIntra() )
1386
0
  {
1387
0
    m_cInterSearch.m_BlkUniMvInfoBuffer->savePrevUniMvInfo( tempCS->area.Y(), tmpUniMvInfo, isUniMvInfoSaved );
1388
0
  }
1389
1390
64.0k
  DeriveCtx deriveCtx = m_CABACEstimator->getDeriveCtx();
1391
1392
64.0k
  do
1393
164k
  {
1394
164k
    const auto &subCUArea  = partitioner.currArea();
1395
1396
164k
    if( tempCS->picture->Y().contains( subCUArea.lumaPos() ) )
1397
117k
    {
1398
117k
      PelStorage* orgBuffer =  &m_pOrgBuffer[partitioner.currDepth];
1399
117k
      PelStorage* rspBuffer =  &m_pRspBuffer[partitioner.currDepth];
1400
117k
      CodingStructure *tempSubCS = m_pTempCS[partitioner.currDepth];
1401
117k
      CodingStructure *bestSubCS = m_pBestCS[partitioner.currDepth];
1402
1403
117k
      tempCS->initSubStructure( *tempSubCS, partitioner.chType, subCUArea, false, orgBuffer, rspBuffer );
1404
117k
      tempCS->initSubStructure( *bestSubCS, partitioner.chType, subCUArea, false, orgBuffer, rspBuffer );
1405
1406
      // copy org buffer, need to be done after initSubStructure because of reshaping!
1407
117k
      orgBuffer->copyFrom( tempCS->getOrgBuf( subCUArea ) );
1408
1409
117k
      tempSubCS->bestParent = bestSubCS->bestParent = bestCS;
1410
1411
117k
      xCompressCU(tempSubCS, bestSubCS, partitioner );
1412
1413
117k
      tempSubCS->bestParent = bestSubCS->bestParent = nullptr;
1414
1415
117k
      if( bestSubCS->cost == MAX_DOUBLE )
1416
0
      {
1417
0
        CHECK( split == CU_QUAD_SPLIT, "Split decision reusing cannot skip quad split" );
1418
0
        tempCS->cost = MAX_DOUBLE;
1419
0
        tempCS->costDbOffset = 0;
1420
0
        m_CurrCtx--;
1421
0
        partitioner.exitCurrSplit();
1422
0
        xCheckBestMode( tempCS, bestCS, partitioner, encTestMode );
1423
0
        if( partitioner.chType == CH_L )
1424
0
        {
1425
0
          tempCS->motionLut = oldMotionLut;
1426
0
        }
1427
1428
0
        m_CABACEstimator->getDeriveCtx() = deriveCtx;
1429
0
        return;
1430
0
      }
1431
1432
117k
      tempCS->useSubStructure( *bestSubCS, partitioner.chType, TREE_D, CS::getArea( *tempCS, subCUArea, partitioner.chType, partitioner.treeType ), partitioner.hasNextPart() || chromaNotSplit );
1433
1434
117k
      if( partitioner.currQgEnable() )
1435
6.16k
      {
1436
6.16k
        tempCS->prevQP[partitioner.chType] = bestSubCS->prevQP[partitioner.chType];
1437
6.16k
      }
1438
117k
      if( partitioner.isConsInter() )
1439
0
      {
1440
0
        for( int i = 0; i < bestSubCS->cus.size(); i++ )
1441
0
        {
1442
0
          CHECK( bestSubCS->cus[i]->predMode != MODE_INTER, "all CUs must be inter mode in an Inter coding region (SCIPU)" );
1443
0
        }
1444
0
      }
1445
117k
      else if( partitioner.isConsIntra() )
1446
0
      {
1447
0
        for( int i = 0; i < bestSubCS->cus.size(); i++ )
1448
0
        {
1449
0
          CHECK( bestSubCS->cus[i]->predMode == MODE_INTER, "all CUs must not be inter mode in an Intra coding region (SCIPU)" );
1450
0
        }
1451
0
      }
1452
1453
117k
      tempSubCS->releaseIntermediateData();
1454
117k
      bestSubCS->releaseIntermediateData();
1455
117k
      if( !tempCS->slice->isIntra() && partitioner.isConsIntra() )
1456
0
      {
1457
0
        tempCS->cost = m_cRdCost.calcRdCost( tempCS->fracBits, tempCS->dist );
1458
0
        if( tempCS->cost > bestCS->cost )
1459
0
        {
1460
0
          tempCS->cost = MAX_DOUBLE;
1461
0
          tempCS->costDbOffset = 0;
1462
0
          m_CurrCtx--;
1463
0
          partitioner.exitCurrSplit();
1464
0
          if( partitioner.chType == CH_L )
1465
0
          {
1466
0
            tempCS->motionLut = oldMotionLut;
1467
0
          }
1468
1469
0
          m_CABACEstimator->getDeriveCtx() = deriveCtx;
1470
0
          return;
1471
0
        }
1472
0
      }
1473
117k
    }
1474
164k
  } while( partitioner.nextPart( *tempCS ) );
1475
1476
64.0k
  partitioner.exitCurrSplit();
1477
1478
64.0k
  m_CurrCtx--;
1479
1480
64.0k
  m_CABACEstimator->getDeriveCtx() = deriveCtx;
1481
1482
64.0k
  if( chromaNotSplit )
1483
0
  {
1484
    //Note: In local dual tree region, the chroma CU refers to the central luma CU's QP.
1485
    //If the luma CU QP shall be predQP (no residual in it and before it in the QG), it must be revised to predQP before encoding the chroma CU
1486
    //Otherwise, the chroma CU uses predQP+deltaQP in encoding but is decoded as using predQP, thus causing encoder-decoded mismatch on chroma qp.
1487
0
    if( tempCS->pps->useDQP )
1488
0
    {
1489
      //find parent CS that including all coded CUs in the QG before this node
1490
0
      CodingStructure* qgCS = tempCS;
1491
0
      bool deltaQpCodedBeforeThisNode = false;
1492
0
      if( partitioner.currArea().lumaPos() != partitioner.currQgPos )
1493
0
      {
1494
0
        int numParentNodeToQgCS = 0;
1495
0
        while( qgCS->area.lumaPos() != partitioner.currQgPos )
1496
0
        {
1497
0
          CHECK( qgCS->parent == nullptr, "parent of qgCS shall exsit" );
1498
0
          qgCS = qgCS->parent;
1499
0
          numParentNodeToQgCS++;
1500
0
        }
1501
1502
        //check whether deltaQP has been coded (in luma CU or luma&chroma CU) before this node
1503
0
        CodingStructure* parentCS = tempCS->parent;
1504
0
        for( int i = 0; i < numParentNodeToQgCS; i++ )
1505
0
        {
1506
          //checking each parent
1507
0
          CHECK( parentCS == nullptr, "parentCS shall exsit" );
1508
0
          for( const auto &cu : parentCS->cus )
1509
0
          {
1510
0
            if( cu->rootCbf && !isChroma( cu->chType ) )
1511
0
            {
1512
0
              deltaQpCodedBeforeThisNode = true;
1513
0
              break;
1514
0
            }
1515
0
          }
1516
0
          parentCS = parentCS->parent;
1517
0
        }
1518
0
      }
1519
1520
      //revise luma CU qp before the first luma CU with residual in the SCIPU to predQP
1521
0
      if( !deltaQpCodedBeforeThisNode )
1522
0
      {
1523
        //get pred QP of the QG
1524
0
        const CodingUnit* cuFirst = qgCS->getCU( CH_L, TREE_D );
1525
0
        CHECK( cuFirst->lumaPos() != partitioner.currQgPos, "First cu of the Qg is wrong" );
1526
0
        int predQp = CU::predictQP( *cuFirst, qgCS->prevQP[CH_L] );
1527
1528
        //revise to predQP
1529
0
        int firstCuHasResidual = (int)tempCS->cus.size();
1530
0
        for( int i = 0; i < tempCS->cus.size(); i++ )
1531
0
        {
1532
0
          if( tempCS->cus[i]->rootCbf )
1533
0
          {
1534
0
            firstCuHasResidual = i;
1535
0
            break;
1536
0
          }
1537
0
        }
1538
1539
0
        for( int i = 0; i < firstCuHasResidual; i++ )
1540
0
        {
1541
0
          tempCS->cus[i]->qp = predQp;
1542
0
        }
1543
0
      }
1544
0
    }
1545
0
    partitioner.chType   = CH_C;
1546
0
    partitioner.treeType = TREE_C;
1547
1548
0
    m_CurrCtx++;
1549
1550
0
    CodingStructure *tempCSChroma = m_pTempCS2;
1551
0
    CodingStructure *bestCSChroma = m_pBestCS2;
1552
1553
0
    tempCS->initSubStructure( *tempCSChroma, partitioner.chType, partitioner.currArea(), false );
1554
0
    tempCS->initSubStructure( *bestCSChroma, partitioner.chType, partitioner.currArea(), false );
1555
0
    tempCSChroma->lumaCS = tempCS;
1556
0
    bestCSChroma->lumaCS = tempCS;
1557
0
    xCompressCU( tempCSChroma, bestCSChroma, partitioner );
1558
1559
    //attach chromaCS to luma CS and update cost
1560
0
    tempCS->useSubStructure( *bestCSChroma, partitioner.chType, TREE_D, CS::getArea( *bestCSChroma, partitioner.currArea(), partitioner.chType, partitioner.treeType ), false );
1561
1562
    //release tmp resource
1563
0
    tempCSChroma->releaseIntermediateData();
1564
0
    bestCSChroma->releaseIntermediateData();
1565
1566
0
    m_CurrCtx--;
1567
    //recover luma tree status
1568
0
    partitioner.chType = CH_L;
1569
0
    partitioner.treeType = TREE_D;
1570
0
    partitioner.modeType = MODE_TYPE_ALL;
1571
0
  }
1572
1573
  // Finally, add split-signaling bits for RD-cost check
1574
64.0k
  tempCS->fracBits += splitBits; // split bits
1575
64.0k
  tempCS->cost      = m_cRdCost.calcRdCost( tempCS->fracBits, tempCS->dist );
1576
64.0k
  partitioner.modeType = modeTypeParent;
1577
1578
  // Check Delta QP bits for splitted structure
1579
64.0k
  if( !qgEnableChildren ) // check at deepest QG level only
1580
61.6k
    xCheckDQP( *tempCS, partitioner, true );
1581
1582
  // If the configuration being tested exceeds the maximum number of bytes for a slice / slice-segment, then
1583
  // a proper RD evaluation cannot be performed. Therefore, termination of the
1584
  // slice/slice-segment must be made prior to this CTU.
1585
  // This can be achieved by forcing the decision to be that of the rpcTempCU.
1586
  // The exception is each slice / slice-segment must have at least one CTU.
1587
64.0k
  if( bestCS->cost == MAX_DOUBLE )
1588
45.3k
  {
1589
45.3k
    bestCS->costDbOffset = 0;
1590
45.3k
  }
1591
1592
64.0k
  if( tempCS->cus.size() > 0 && modeTypeParent == MODE_TYPE_ALL && modeTypeChild == MODE_TYPE_INTER )
1593
0
  {
1594
0
    int areaSizeNoResiCu = 0;
1595
0
    for( int k = 0; k < tempCS->cus.size(); k++ )
1596
0
    {
1597
0
      areaSizeNoResiCu += (tempCS->cus[k]->rootCbf == false) ? tempCS->cus[k]->lumaSize().area() : 0;
1598
0
    }
1599
0
    if( areaSizeNoResiCu >= (tempCS->area.lumaSize().area() >> 1) )
1600
0
    {
1601
0
      skipInterPass = true;
1602
0
    }
1603
0
  }
1604
1605
  // RD check for sub partitioned coding structure.
1606
64.0k
  xCheckBestMode( tempCS, bestCS, partitioner, encTestMode, m_EDO );
1607
1608
64.0k
  if( isAffMVInfoSaved )
1609
0
  {
1610
0
    m_cInterSearch.m_AffineProfList->addAffMVInfo(tmpMVInfo);
1611
0
  }
1612
1613
64.0k
  if( !tempCS->slice->isIntra() && isUniMvInfoSaved )
1614
0
  {
1615
0
    m_cInterSearch.m_BlkUniMvInfoBuffer->addUniMvInfo(tmpUniMvInfo);
1616
0
  }
1617
1618
64.0k
  tempCS->motionLut = oldMotionLut;
1619
64.0k
  tempCS->releaseIntermediateData();
1620
64.0k
  tempCS->prevQP[partitioner.chType] = oldPrevQp;
1621
64.0k
}
1622
1623
1624
void EncCu::xCheckRDCostIntra( CodingStructure *&tempCS, CodingStructure *&bestCS, Partitioner &partitioner, const EncTestMode& encTestMode )
1625
79.0k
{
1626
79.0k
  PROFILER_SCOPE_AND_STAGE_EXT( 1, _TPROF, P_INTRA, tempCS, partitioner.chType );
1627
1628
79.0k
  tempCS->initStructData( encTestMode.qp, false ); // clear motion buffer
1629
1630
79.0k
  CodingUnit &cu      = tempCS->addCU( CS::getArea( *tempCS, tempCS->area, partitioner.chType, partitioner.treeType ), partitioner.chType );
1631
1632
79.0k
  partitioner.setCUData( cu );
1633
79.0k
  cu.slice            = tempCS->slice;
1634
79.0k
  cu.tileIdx          = m_tileIdx;
1635
79.0k
  cu.skip             = false;
1636
79.0k
  cu.mmvdSkip         = false;
1637
79.0k
  cu.predMode         = MODE_INTRA;
1638
79.0k
  cu.chromaQpAdj      = m_cuChromaQpOffsetIdxPlus1;
1639
79.0k
  cu.qp               = encTestMode.qp;
1640
79.0k
  cu.ispMode          = NOT_INTRA_SUBPARTITIONS;
1641
79.0k
  cu.initPuData();
1642
1643
79.0k
  m_cIntraSearch.m_ispTestedModes[0].init(0, 0, 1);
1644
79.0k
  if (m_pcEncCfg->m_FastIntraTools)
1645
79.0k
  {
1646
79.0k
    m_modeCtrl.comprCUCtx->intraWasTested = false;
1647
79.0k
    m_cIntraSearch.m_ispTestedModes[0].relatedCuIsValid = m_modeCtrl.comprCUCtx->relatedCuIsValid;
1648
79.0k
    if (!bestCS->cus.empty())
1649
21.4k
    {
1650
21.4k
      if ((bestCS->cus[0]->mergeFlag || bestCS->cus[0]->imv || bestCS->cus[0]->affine) && (!bestCS->cus[0]->ciip))
1651
21.4k
      {
1652
21.4k
        m_cIntraSearch.m_ispTestedModes[0].bestBefore[0] = -1;
1653
21.4k
      }
1654
21.4k
    }
1655
79.0k
    if (!bestCS->slice->isIntra())
1656
0
    {
1657
0
      const Position posBL = cu.Y().bottomLeft();
1658
0
      const Position posTR = cu.Y().topRight();
1659
0
      for (int i = 0; i < 2; i++)
1660
0
      {
1661
0
        const CodingUnit* neigh = i ? cu.cs->getCURestricted(posTR.offset(0, -1), cu, CH_L) :cu.cs->getCURestricted(posBL.offset(-1, 0), cu, CH_L);
1662
0
        m_cIntraSearch.m_ispTestedModes[0].bestBefore[i+1] = -1;
1663
0
        if (neigh != nullptr)
1664
0
        {
1665
0
          int bestMode = neigh->firstTU->mtsIdx[0] ? 4 : 0;
1666
0
          bestMode |= neigh->lfnstIdx ? 2 : 0;
1667
0
          bestMode |= neigh->ispMode ? 1 : 0;
1668
0
          m_cIntraSearch.m_ispTestedModes[0].bestBefore[i+1] = bestMode;
1669
0
        }
1670
0
      }
1671
0
    }
1672
79.0k
  }
1673
1674
79.0k
  tempCS->interHad    = m_modeCtrl.comprCUCtx->interHad;
1675
79.0k
  double maxCostAllowedForChroma = MAX_DOUBLE;
1676
79.0k
  if( isLuma( partitioner.chType ) )
1677
24.0k
  {
1678
24.0k
    if (!tempCS->slice->isIntra() && bestCS)
1679
0
    {
1680
0
      m_cIntraSearch.estIntraPredLumaQT(cu, partitioner, bestCS->cost);
1681
0
    }
1682
24.0k
    else
1683
24.0k
    {
1684
24.0k
      m_cIntraSearch.estIntraPredLumaQT(cu, partitioner);
1685
24.0k
    }
1686
24.0k
    if (m_pcEncCfg->m_FastIntraTools)
1687
24.0k
    {
1688
24.0k
      if (m_cIntraSearch.m_ispTestedModes[0].intraWasTested)
1689
24.0k
      {
1690
24.0k
        m_modeCtrl.comprCUCtx->intraWasTested = m_cIntraSearch.m_ispTestedModes[0].intraWasTested;
1691
24.0k
      }
1692
24.0k
    }
1693
1694
24.0k
    if( !partitioner.isSepTree( *tempCS ) )
1695
0
    {
1696
0
      tempCS->lumaCost = m_cRdCost.calcRdCost( tempCS->fracBits, tempCS->dist );
1697
0
    }
1698
24.0k
    if (m_pcEncCfg->m_usePbIntraFast && tempCS->dist == MAX_DISTORTION && tempCS->interHad == 0)
1699
0
    {
1700
      // JEM assumes only perfect reconstructions can from now on beat the inter mode
1701
0
      m_modeCtrl.comprCUCtx->interHad = 0;
1702
0
      return;
1703
0
    }
1704
24.0k
  }
1705
1706
79.0k
  if( tempCS->area.chromaFormat != CHROMA_400 && ( partitioner.chType == CH_C || !CU::isSepTree(cu) ) )
1707
54.9k
  {
1708
54.9k
    bool useIntraSubPartitions = cu.ispMode != NOT_INTRA_SUBPARTITIONS;
1709
54.9k
    Partitioner subTuPartitioner = partitioner;
1710
54.9k
    if ((m_pcEncCfg->m_ISP >= 3) && (!partitioner.isSepTree(*tempCS) && useIntraSubPartitions))
1711
0
    {
1712
0
      maxCostAllowedForChroma = bestCS->cost < MAX_DOUBLE ? bestCS->cost - tempCS->lumaCost : MAX_DOUBLE;
1713
0
    }
1714
54.9k
    m_cIntraSearch.estIntraPredChromaQT(
1715
54.9k
      cu, (!useIntraSubPartitions || (CU::isSepTree(cu) && !isLuma(CH_C))) ? partitioner : subTuPartitioner,
1716
54.9k
      maxCostAllowedForChroma);
1717
54.9k
    if ((m_pcEncCfg->m_ISP >= 3) && useIntraSubPartitions && !cu.ispMode)
1718
0
    {
1719
0
      return;
1720
0
    }
1721
54.9k
  }
1722
1723
79.0k
  cu.rootCbf = false;
1724
1725
316k
  for (uint32_t t = 0; t < getNumberValidTBlocks(*cu.cs->pcv); t++)
1726
237k
  {
1727
237k
    cu.rootCbf |= cu.firstTU->cbf[t] != 0;
1728
237k
  }
1729
1730
  // Get total bits for current mode: encode CU
1731
79.0k
  m_CABACEstimator->resetBits();
1732
1733
79.0k
  if ((!cu.cs->slice->isIntra() || cu.cs->slice->sps->IBC) && cu.Y().valid())
1734
24.0k
  {
1735
24.0k
    m_CABACEstimator->cu_skip_flag(cu);
1736
24.0k
  }
1737
79.0k
  m_CABACEstimator->pred_mode(cu);
1738
79.0k
  m_CABACEstimator->cu_pred_data(cu);
1739
1740
  // Encode Coefficients
1741
79.0k
  CUCtx cuCtx;
1742
79.0k
  cuCtx.isDQPCoded = true;
1743
79.0k
  cuCtx.isChromaQpAdjCoded = true;
1744
79.0k
  m_CABACEstimator->cu_residual(cu, partitioner, cuCtx);
1745
1746
79.0k
  tempCS->fracBits = m_CABACEstimator->getEstFracBits();
1747
79.0k
  tempCS->cost = m_cRdCost.calcRdCost(tempCS->fracBits, tempCS->dist);
1748
1749
79.0k
  xEncodeDontSplit(*tempCS, partitioner);
1750
1751
79.0k
  xCheckDQP(*tempCS, partitioner);
1752
1753
79.0k
  if( m_EDO )
1754
79.0k
  {
1755
79.0k
    xCalDebCost(*tempCS, partitioner);
1756
79.0k
  }
1757
1758
79.0k
  DTRACE_MODE_COST(*tempCS, m_cRdCost.getLambda(true));
1759
79.0k
  xCheckBestMode(tempCS, bestCS, partitioner, encTestMode, m_EDO);
1760
1761
79.0k
  STAT_COUNT_CU_MODES( partitioner.chType == CH_L, g_cuCounters1D[CU_MODES_TESTED][0][!tempCS->slice->isIntra() + tempCS->slice->depth] );
1762
79.0k
  STAT_COUNT_CU_MODES( partitioner.chType == CH_L && !tempCS->slice->isIntra(), g_cuCounters2D[CU_MODES_TESTED][Log2( tempCS->area.lheight() )][Log2( tempCS->area.lwidth() )] );
1763
79.0k
}
1764
1765
void EncCu::xCheckDQP( CodingStructure& cs, Partitioner& partitioner, bool bKeepCtx )
1766
162k
{
1767
162k
  if( !cs.pps->useDQP )
1768
0
  {
1769
0
    return;
1770
0
  }
1771
1772
162k
  if (partitioner.isSepTree(cs) && isChroma(partitioner.chType))
1773
92.1k
  {
1774
92.1k
    return;
1775
92.1k
  }
1776
1777
70.0k
  if( !partitioner.currQgEnable() ) // do not consider split or leaf/not leaf QG condition (checked by caller)
1778
63.3k
  {
1779
63.3k
    return;
1780
63.3k
  }
1781
1782
6.63k
  CodingUnit* cuFirst = cs.getCU( partitioner.chType, partitioner.treeType );
1783
1784
6.63k
  CHECK( bKeepCtx && cs.cus.size() <= 1 && partitioner.getImplicitSplit( cs ) == CU_DONT_SPLIT, "bKeepCtx should only be set in split case" );
1785
6.63k
  CHECK( !bKeepCtx && cs.cus.size() > 1, "bKeepCtx should never be set for non-split case" );
1786
6.63k
  CHECK( !cuFirst, "No CU available" );
1787
1788
6.63k
  bool hasResidual = false;
1789
6.63k
  for( const auto &cu : cs.cus )
1790
19.0k
  {
1791
    //not include the chroma CU because chroma CU is decided based on corresponding luma QP and deltaQP is not signaled at chroma CU
1792
19.0k
    if( cu->rootCbf && !isChroma( cu->chType ))
1793
1.17k
    {
1794
1.17k
      hasResidual = true;
1795
1.17k
      break;
1796
1.17k
    }
1797
19.0k
  }
1798
1799
6.63k
  int predQP = CU::predictQP( *cuFirst, cs.prevQP[partitioner.chType] );
1800
1801
6.63k
  if( hasResidual )
1802
1.17k
  {
1803
1.17k
    TempCtx ctxTemp( m_CtxCache );
1804
1.17k
    if( !bKeepCtx ) ctxTemp = SubCtx( Ctx::DeltaQP, m_CABACEstimator->getCtx() );
1805
1806
1.17k
    m_CABACEstimator->resetBits();
1807
1.17k
    m_CABACEstimator->cu_qp_delta( *cuFirst, predQP, cuFirst->qp );
1808
1809
1.17k
    cs.fracBits += m_CABACEstimator->getEstFracBits(); // dQP bits
1810
1.17k
    cs.cost      = m_cRdCost.calcRdCost(cs.fracBits, cs.dist);
1811
1812
1813
1.17k
    if( !bKeepCtx ) m_CABACEstimator->getCtx() = SubCtx( Ctx::DeltaQP, ctxTemp );
1814
1815
    // NOTE: reset QPs for CUs without residuals up to first coded CU
1816
1.17k
    for( const auto &cu : cs.cus )
1817
1.18k
    {
1818
      //not include the chroma CU because chroma CU is decided based on corresponding luma QP and deltaQP is not signaled at chroma CU
1819
1.18k
      if( cu->rootCbf && !isChroma( cu->chType ))
1820
1.17k
      {
1821
1.17k
        break;
1822
1.17k
      }
1823
8
      cu->qp = predQP;
1824
8
    }
1825
1.17k
  }
1826
5.45k
  else
1827
5.45k
  {
1828
    // No residuals: reset CU QP to predicted value
1829
5.45k
    for( const auto &cu : cs.cus )
1830
17.8k
    {
1831
17.8k
      cu->qp = predQP;
1832
17.8k
    }
1833
5.45k
  }
1834
6.63k
}
1835
1836
CodingUnit *EncCu::getCuForInterPrediction( CodingStructure *cs, const EncTestMode& encTestMode )
1837
0
{
1838
0
  CodingUnit *cu = cs->getCU( CH_L, TREE_D );
1839
1840
0
  if( cu == nullptr )
1841
0
  {
1842
0
    CHECK( cs->getCU( CH_L, TREE_D ) != nullptr, "Wrong CU/PU setting in CS" );
1843
0
    cu = &cs->addCU( cs->area, CH_L );
1844
0
  }
1845
1846
0
  cu->slice       = cs->slice;
1847
0
  cu->tileIdx     = m_tileIdx;
1848
0
  cu->skip        = false;
1849
0
  cu->mmvdSkip    = false;
1850
0
  cu->mmvdMergeFlag
1851
0
                  = false;
1852
0
  cu->geo         = false;
1853
0
  cu->predMode    = MODE_INTER;
1854
0
  cu->chromaQpAdj = m_cuChromaQpOffsetIdxPlus1;
1855
0
  cu->qp          = encTestMode.qp;
1856
0
  cu->affine      = false;
1857
0
  cu->multiRefIdx = 0;
1858
0
  cu->mipFlag     = false;
1859
0
  cu->ciip        = false;
1860
1861
0
  return cu;
1862
0
}
1863
1864
int getDmvrMvdNum( const CodingUnit &cu )
1865
0
{
1866
0
  const int dx = std::max<int>( cu.lwidth()  >> DMVR_SUBCU_SIZE_LOG2, 1 );
1867
0
  const int dy = std::max<int>( cu.lheight() >> DMVR_SUBCU_SIZE_LOG2, 1 );
1868
0
  return dx * dy;
1869
0
}
1870
1871
void EncCu::xCheckRDCostUnifiedMerge( CodingStructure *&tempCS, CodingStructure *&bestCS, Partitioner &partitioner, EncTestMode &encTestMode )
1872
0
{
1873
0
  const Slice &slice = *tempCS->slice;
1874
1875
0
  CHECK( slice.sliceType == VVENC_I_SLICE, "Merge modes not available for I-slices" );
1876
1877
0
  tempCS->initStructData( encTestMode.qp );
1878
1879
0
  MergeCtx          mergeCtx, gpmMergeCtx;
1880
0
  AffineMergeCtx    affineMergeCtx;
1881
0
  GeoComboCostList &comboList = m_comboList;
1882
0
  const SPS        &sps       = *tempCS->sps;
1883
1884
0
  if( sps.SbtMvp )
1885
0
  {
1886
0
    const Size bufSize           = g_miScaling.scale( tempCS->area.lumaSize() );
1887
0
    affineMergeCtx.subPuMvpMiBuf = MotionBuf        ( m_subPuMiBuf, bufSize );
1888
0
  }
1889
1890
0
  m_mergeBestSATDCost = MAX_DOUBLE;
1891
1892
0
  CodingUnit *cu = getCuForInterPrediction( tempCS, encTestMode );
1893
0
  partitioner.setCUData            ( *cu );
1894
0
  CU::getInterMergeCandidates      ( *cu, mergeCtx, 0 );
1895
0
  if( sps.MMVD )
1896
0
    CU::getInterMMVDMergeCandidates( *cu, mergeCtx );
1897
1898
0
  bool sameMV[MRG_MAX_NUM_CANDS] = { false, };
1899
0
  if( m_pcEncCfg->m_useFastMrg >= 2 )
1900
0
  {
1901
0
    for( int m = 0; m < mergeCtx.numValidMergeCand - 1; m++ )
1902
0
    {
1903
0
      if( !sameMV[m] )
1904
0
      {
1905
0
        for( int n = m + 1; n < mergeCtx.numValidMergeCand; n++ )
1906
0
        {
1907
0
          sameMV[n] |= mergeCtx.mvFieldNeighbours[m][0] == mergeCtx.mvFieldNeighbours[n][0]
1908
0
                    && mergeCtx.mvFieldNeighbours[m][1] == mergeCtx.mvFieldNeighbours[n][1];
1909
0
        }
1910
0
      }
1911
0
    }
1912
0
  }
1913
1914
0
  MergeBufVector mrgPredBufNoCiip;
1915
0
  MergeBufVector geoBuffer;
1916
0
  const double  sqrtLambdaForFirstPass = m_cRdCost.getMotionLambda() * FRAC_BITS_SCALE;
1917
1918
0
  const UnitArea localUnitArea( cu->chromaFormat, Area( 0, 0, cu->Y().width, cu->Y().height ) );
1919
0
  for( int i = 0; i < mergeCtx.numValidMergeCand; i++ )
1920
0
  {
1921
0
    mrgPredBufNoCiip.push_back( m_acMergeTmpBuffer[i].getCompactBuf( localUnitArea ) );
1922
0
  }
1923
1924
0
  int numMergeSatdCand = std::min( bestCS->area.lumaSize().area() >= 64 ? m_pcEncCfg->m_mergeRdCandQuotaRegular : m_pcEncCfg->m_mergeRdCandQuotaRegularSmallBlk, mergeCtx.numValidMergeCand );
1925
1926
0
  bool isCiipEnabled  = sps.CIIP && bestCS->area.lumaSize().area() >= 64 && bestCS->area.lumaSize().maxDim() < MAX_CU_SIZE;
1927
0
       isCiipEnabled &= m_pcEncCfg->m_CIIP <= 1 || !m_modeCtrl.getBlkInfo( tempCS->area ).isSkip; //5
1928
1929
0
  if( isCiipEnabled )
1930
0
  {
1931
0
    numMergeSatdCand += std::min( m_pcEncCfg->m_mergeRdCandQuotaCiip, mergeCtx.numValidMergeCand );
1932
0
  }
1933
1934
0
  const bool affineMrgAvail = ( m_pcEncCfg->m_Affine <= 2 || slice.TLayer <= 3 || m_pcEncCfg->m_SbTMVP )
1935
0
                           && ( m_pcEncCfg->m_Affine || sps.SbtMvp ) && m_pcEncCfg->m_maxNumAffineMergeCand && bestCS->area.Y().minDim() >= 8;
1936
1937
0
  if( affineMrgAvail )
1938
0
  {
1939
0
    CU::getAffineMergeCand( *cu, affineMergeCtx );
1940
0
    numMergeSatdCand += std::min( m_pcEncCfg->m_mergeRdCandQuotaSubBlk, affineMergeCtx.numValidMergeCand );
1941
0
  }
1942
1943
0
  int numSatdCandPreGeo = std::min( numMergeSatdCand, m_pcEncCfg->m_maxMergeRdCandNumTotal );
1944
0
  bool toAddGpmCand     = false;
1945
0
  if( sps.GEO && slice.isInterB() // base checks
1946
0
      && cu->lumaSize().minDim() >= GEO_MIN_CU_SIZE  && cu->lumaSize().maxDim() <= GEO_MAX_CU_SIZE && cu->lumaSize().maxDim() < 8 * cu->lumaSize().minDim() // size checks
1947
0
      && !( m_pcEncCfg->m_Geo > 2 && slice.TLayer <= 1 ) ) // speedups
1948
0
  {
1949
0
    cu->mergeFlag            = true;
1950
0
    cu->geo                  = true;
1951
0
    CU::getGeoMergeCandidates( *cu, gpmMergeCtx );
1952
0
    toAddGpmCand             = prepareGpmComboList( gpmMergeCtx, localUnitArea, sqrtLambdaForFirstPass, comboList, geoBuffer, *cu );
1953
0
    numMergeSatdCand        += toAddGpmCand ? std::min( m_pcEncCfg->m_mergeRdCandQuotaGpm, ( int ) comboList.list.size() ) : 0;
1954
0
  }
1955
1956
0
  numMergeSatdCand  = std::min( numMergeSatdCand, m_pcEncCfg->m_maxMergeRdCandNumTotal );
1957
1958
  // 1. Pass: get SATD-cost for selected candidates and reduce their count
1959
0
  m_mergeItemList.resetList( numMergeSatdCand );
1960
0
  const TempCtx ctxStart   ( m_CtxCache, m_CABACEstimator->getCtx() );
1961
0
  const DFunc   dfunc      = encTestMode.lossless ? DF_SAD : ( m_pcEncCfg->m_fastHad ? DF_HAD_fast : DF_HAD );
1962
0
  DistParam     distParam  = m_cRdCost.setDistParam( tempCS->getOrgBuf().Y(), tempCS->getOrgBuf().Y(), sps.bitDepths[CH_L], dfunc );
1963
0
  m_uiSadBestForQPA        = MAX_DISTORTION;
1964
1965
0
  addRegularCandsToPruningList( mergeCtx, localUnitArea, sqrtLambdaForFirstPass, ctxStart, distParam, *cu, sameMV, mrgPredBufNoCiip );
1966
1967
  // add CIIP candidates directly after adding regular cands
1968
0
  if( isCiipEnabled )
1969
0
  {
1970
0
    addCiipCandsToPruningList( mergeCtx, localUnitArea, sqrtLambdaForFirstPass, ctxStart, distParam, *cu, sameMV );
1971
0
  }
1972
1973
0
  if( sps.MMVD && !!m_mergeItemList.size() && !( m_pcEncCfg->m_useFastMrg >= 2 && m_mergeItemList.size() <= 1 ) )
1974
0
  {
1975
0
    addMmvdCandsToPruningList( mergeCtx, localUnitArea, sqrtLambdaForFirstPass, ctxStart, distParam, *cu );
1976
0
  }
1977
1978
0
  if( affineMergeCtx.numValidMergeCand > 0 )
1979
0
  {
1980
0
    addAffineCandsToPruningList( affineMergeCtx, localUnitArea, sqrtLambdaForFirstPass, ctxStart, distParam, *cu );
1981
0
  }
1982
1983
0
  if( m_pcEncCfg->m_useFastMrg > 0 && m_mergeItemList.size() > 0 )
1984
0
  {
1985
0
    m_mergeBestSATDCost    = m_mergeItemList.getMergeItemInList( 0 )->cost;
1986
0
    const double threshold = m_mergeBestSATDCost * MRG_FAST_RATIO[tempCS->picture->useFastMrg];
1987
0
    const   int shrinkSize = std::min( numSatdCandPreGeo, ( int ) updateRdCheckingNum( m_mergeItemList, threshold, numMergeSatdCand ) );
1988
0
    m_mergeItemList        . shrinkList( shrinkSize );
1989
0
  }
1990
0
  else
1991
0
  {
1992
0
    m_mergeItemList        . shrinkList( numSatdCandPreGeo );
1993
0
  }
1994
1995
0
  if( toAddGpmCand )
1996
0
  {
1997
0
    addGpmCandsToPruningList( gpmMergeCtx, localUnitArea, sqrtLambdaForFirstPass, ctxStart, comboList, geoBuffer, distParam, *cu );
1998
0
  }
1999
2000
0
  if(    m_pcEncCfg->m_internalUsePerceptQPATempFiltISlice == 2 && m_uiSadBestForQPA < MAX_DISTORTION && slice.TLayer == 0 // non-Intra key-frame
2001
0
      && m_pcEncCfg->m_salienceBasedOpt
2002
0
      && m_pcEncCfg->m_usePerceptQPA && partitioner.currQgEnable() && partitioner.currSubdiv == 0 ) // CTU-level luma quantization group
2003
0
  {
2004
0
    CHECK( bestCS->cost < MAX_DOUBLE, "This has to be the first test performed!" );
2005
2006
0
    const Picture *pic         = slice.pic;
2007
0
    const bool     isBIM       = m_pcEncCfg->m_RCNumPasses != 2 && m_pcEncCfg->m_blockImportanceMapping && !pic->m_picShared->m_ctuBimQpOffset.empty();
2008
0
    const uint32_t rsAddr      = getCtuAddr( partitioner.currQgPos, *pic->cs->pcv );
2009
0
    const int      pumpReducQP = BitAllocation::getCtuPumpingReducingQP( &slice, tempCS->getOrgBuf( COMP_Y ), m_uiSadBestForQPA, *m_globalCtuQpVector, rsAddr,
2010
0
                                                                         m_pcEncCfg->m_QP, isBIM );
2011
2012
0
    if( pumpReducQP != 0 ) // subtract QP offset, reduces Intra-period pumping or overcoding
2013
0
    {
2014
0
      encTestMode.qp = Clip3( 0, MAX_QP, encTestMode.qp - pumpReducQP );
2015
0
      tempCS->currQP[partitioner.chType] = tempCS->baseQP =
2016
0
      bestCS->currQP[partitioner.chType] = bestCS->baseQP = Clip3( 0, MAX_QP, tempCS->baseQP - pumpReducQP );
2017
2018
0
      updateLambda( slice, pic->ctuQpaLambda[rsAddr], pic->ctuAdaptedQP[rsAddr], tempCS->baseQP, true );
2019
0
    }
2020
0
  }
2021
2022
  // Try to limit number of candidates using SATD-costs
2023
0
  if( m_pcEncCfg->m_useFastMrg > 0 && m_mergeItemList.size() > 0 )
2024
0
  {
2025
    // shrink GEO list as well
2026
0
    const double threshold = m_mergeItemList.getMergeItemInList( 0 )->cost * MRG_FAST_RATIO[0];
2027
0
    numMergeSatdCand       = updateRdCheckingNum( m_mergeItemList, threshold, numMergeSatdCand );
2028
0
    m_mergeBestSATDCost    = m_mergeItemList.size() != 0 ? m_mergeItemList.getMergeItemInList( 0 )->cost : MAX_DOUBLE;
2029
0
  }
2030
0
  else
2031
0
  {
2032
0
    numMergeSatdCand       = std::min<int>( numMergeSatdCand, ( int ) m_mergeItemList.size() );
2033
0
  }
2034
2035
  // 2. Pass: RD checking 
2036
0
  tempCS->initStructData( encTestMode.qp );
2037
0
  m_CABACEstimator->getCtx() = ctxStart;
2038
2039
0
  double bestEndCost                            =   MAX_DOUBLE;
2040
0
  bool bestIsSkip                               =   false;
2041
0
  PelUnitBuf ciipBuf                            =   m_aTmpStorageLCU[1].getCompactBuf( *cu );
2042
0
  bool ciipChromaDone                           =   false;
2043
0
  bool isRegularTestedAsSkip[MRG_MAX_NUM_CANDS] = { false, };
2044
0
  bool geoWasTested                             =   false;
2045
0
  int  stopCand                                 =   numMergeSatdCand;
2046
2047
0
  CHECK( numMergeSatdCand > 0 && m_mergeItemList.size() == 0, "Empty merge item list is not expected" );
2048
2049
0
  for( uint32_t noResidualPass = 0; noResidualPass < 2; noResidualPass++ )
2050
0
  {
2051
0
    const bool forceNoResidual = noResidualPass == 1;
2052
0
    for( uint32_t mrgHadIdx = 0; mrgHadIdx < stopCand; mrgHadIdx++ )
2053
0
    {
2054
0
      auto mergeItem = m_mergeItemList.getMergeItemInList( mrgHadIdx );
2055
0
      CHECK( mergeItem == nullptr, "Wrong merge item" );
2056
2057
0
      const bool isCiip = mergeItem->mergeItemType == MergeItem::MergeItemType::CIIP;
2058
0
      const bool isGeo  = mergeItem->mergeItemType == MergeItem::MergeItemType::GPM;
2059
0
      const bool isRglr = mergeItem->mergeItemType == MergeItem::MergeItemType::REGULAR;
2060
0
      const bool isMmvd = mergeItem->mergeItemType == MergeItem::MergeItemType::MMVD;
2061
2062
0
      if( noResidualPass != 0 && isCiip && isRegularTestedAsSkip[mergeItem->mergeIdx] )
2063
0
      {
2064
0
        continue;
2065
0
      }
2066
2067
0
      if( noResidualPass ? mergeItem->noResidual : bestIsSkip )
2068
0
      {
2069
0
        continue;
2070
0
      }
2071
2072
0
      if( isGeo )
2073
0
      {
2074
0
        if( m_pcEncCfg->m_Geo > 2 && geoWasTested && !bestCS->cus.empty() && !bestCS->getCU( partitioner.chType, partitioner.treeType )->geo )
2075
0
        {
2076
0
          continue;
2077
0
        }
2078
2079
0
        geoWasTested = true;
2080
0
      }
2081
2082
0
      cu = getCuForInterPrediction( tempCS, encTestMode );
2083
0
      partitioner.setCUData( *cu );
2084
0
      const bool resetCiip2Regular = mergeItem->exportMergeInfo( *cu, forceNoResidual );
2085
2086
0
      if( isRglr || resetCiip2Regular )
2087
0
      {
2088
0
        if( CU::checkDMVRCondition( *cu ) ) std::copy_n( m_subPuMvOffset[mergeItem->mergeIdx].data(), getDmvrMvdNum( *cu ), cu->mvdL0SubPu );
2089
0
      }
2090
2091
0
      if( isMmvd && mergeItem->noBdofRefine )
2092
0
      {
2093
        // no BDOF refinement was made for the luma prediction, need to have luma prediction again
2094
0
        mergeItem->lumaPredReady = false;
2095
0
      }
2096
2097
0
      PelUnitBuf *predBuf1   = nullptr, *predBuf2 = isCiip ? &ciipBuf : nullptr;
2098
0
      PelUnitBuf  dstPredBuf = tempCS->getPredBuf( *cu );
2099
2100
0
      if( isGeo )
2101
0
      {
2102
0
        predBuf1 = &geoBuffer[cu->geoMergeIdx[0]];
2103
0
        predBuf2 = &geoBuffer[cu->geoMergeIdx[1]];
2104
0
      }
2105
2106
0
      if( resetCiip2Regular )
2107
0
      {
2108
0
        dstPredBuf.copyFrom( mrgPredBufNoCiip[mergeItem->mergeIdx] );
2109
0
      }
2110
0
      else
2111
0
      {
2112
0
        if( isCiip && !resetCiip2Regular && isChromaEnabled( cu->chromaFormat ) && cu->chromaSize().width > 2 )
2113
0
        {
2114
0
          if( !ciipChromaDone )
2115
0
          {
2116
0
            cu->intraDir[0] = PLANAR_IDX;
2117
0
            cu->intraDir[1] = DM_CHROMA_IDX;
2118
2119
0
            m_cIntraSearch  . initIntraPatternChType( *cu, cu->Cb() );
2120
0
            m_cIntraSearch  . predIntraAng          ( COMP_Cb, ciipBuf.Cb(), *cu );
2121
0
            m_cIntraSearch  . initIntraPatternChType( *cu, cu->Cr() );
2122
0
            m_cIntraSearch  . predIntraAng          ( COMP_Cr, ciipBuf.Cr(), *cu );
2123
2124
0
            ciipChromaDone  = true;
2125
0
          }
2126
0
        }
2127
2128
0
        if(  mergeItem->lumaPredReady ||  mergeItem->chromaPredReady )
2129
0
          dstPredBuf.copyFrom( mergeItem->getPredBuf( localUnitArea ), mergeItem->lumaPredReady, mergeItem->chromaPredReady );
2130
0
        if( !mergeItem->lumaPredReady || !mergeItem->chromaPredReady )
2131
0
          generateMergePrediction( localUnitArea, mergeItem, *cu, !mergeItem->lumaPredReady, !mergeItem->chromaPredReady, dstPredBuf, true, forceNoResidual, predBuf1, predBuf2 );
2132
0
      }
2133
2134
0
      if( !cu->mmvdSkip && !cu->ciip && !cu->affine && !cu->geo && noResidualPass != 0 )
2135
0
      {
2136
0
        CHECK( mergeItem->mergeIdx >= mergeCtx.numValidMergeCand, "out of normal merge" );
2137
0
        isRegularTestedAsSkip[mergeItem->mergeIdx] = true;
2138
0
      }
2139
2140
0
      xEncodeInterResidual( tempCS, bestCS, partitioner, encTestMode, noResidualPass, noResidualPass == 0 ? &mergeItem->noResidual : nullptr );
2141
2142
0
      if( m_pcEncCfg->m_useFastMrg >= 2 )
2143
0
      {
2144
0
        if( cu->ciip && bestCS->cost == MAX_DOUBLE && mrgHadIdx + 1 == numMergeSatdCand )
2145
0
        {
2146
0
          numMergeSatdCand = ( unsigned ) m_mergeItemList.size();
2147
0
        }
2148
      
2149
0
        if( mrgHadIdx > 0 && tempCS->cost >= bestEndCost && !cu->ciip && !isGeo )
2150
0
        {
2151
0
          stopCand = mrgHadIdx + 1;
2152
0
        }
2153
      
2154
0
        if( noResidualPass == 0 )
2155
0
        {
2156
0
          bestEndCost = std::min( bestEndCost, tempCS->cost );
2157
0
        }
2158
0
      }
2159
2160
0
      if( m_pcEncCfg->m_useFastDecisionForMerge && !bestIsSkip && !cu->ciip )
2161
0
      {
2162
0
        bestIsSkip = !bestCS->cus.empty() && bestCS->getCU( partitioner.chType, partitioner.treeType )->rootCbf == 0;
2163
0
      }
2164
2165
0
      tempCS->initStructData( encTestMode.qp );
2166
0
    }   // end loop mrgHadIdx
2167
0
  }
2168
0
}
2169
2170
unsigned int EncCu::updateRdCheckingNum( MergeItemList &mergeItemList, double threshold, unsigned int numMergeSatdCand )
2171
0
{
2172
0
  for( uint32_t i = 0; i < mergeItemList.size(); i++ )
2173
0
  {
2174
0
    const auto mergeItem = mergeItemList.getMergeItemInList( i );
2175
0
    if( mergeItem == nullptr || mergeItem->cost > threshold )
2176
0
    {
2177
0
      numMergeSatdCand = i;
2178
0
      break;
2179
0
    }
2180
0
  }
2181
0
  return std::min( numMergeSatdCand, ( unsigned ) mergeItemList.size() );
2182
0
}
2183
2184
void EncCu::generateMergePrediction( const UnitArea &unitArea, MergeItem *mergeItem, CodingUnit &pu, bool luma, bool chroma,
2185
                                     PelUnitBuf &dstBuf, bool finalRd, bool forceNoResidual, PelUnitBuf *predBuf1, PelUnitBuf *predBuf2 )
2186
0
{
2187
0
  CHECK( ( luma && mergeItem->lumaPredReady ) || ( chroma && mergeItem->chromaPredReady ), "Prediction has been avaiable" );
2188
2189
0
  pu.mcControl = ( !luma ? 4 : 0 ) | ( !chroma ? 2 : 0 );
2190
2191
0
  switch( mergeItem->mergeItemType )
2192
0
  {
2193
0
  case MergeItem::MergeItemType::REGULAR:
2194
    // here predBuf1 is predBufNoCiip
2195
0
    pu.mvRefine = true;
2196
0
    m_cInterSearch.motionCompensation( pu, dstBuf, REF_PIC_LIST_X );
2197
0
    pu.mvRefine = false;
2198
0
    if( predBuf1 != nullptr )
2199
0
    {
2200
0
      predBuf1->copyFrom( dstBuf, luma, chroma );
2201
0
    }
2202
0
    break;
2203
2204
0
  case MergeItem::MergeItemType::CIIP:
2205
0
    m_cInterSearch.motionCompensation( pu, dstBuf, REF_PIC_LIST_X );
2206
2207
0
    if( luma )
2208
0
    {
2209
      // generate intrainter Y prediction
2210
0
      dstBuf.Y().weightCiip( predBuf2->Y(), mergeItem->numCiipIntra );
2211
0
    }
2212
2213
0
    if( chroma )
2214
0
    {
2215
0
      if( pu.chromaSize().width > 2 )
2216
0
      {
2217
0
        dstBuf.Cb().weightCiip( predBuf2->Cb(), mergeItem->numCiipIntra );
2218
0
        dstBuf.Cr().weightCiip( predBuf2->Cr(), mergeItem->numCiipIntra );
2219
0
      }
2220
0
    }
2221
2222
0
    break;
2223
2224
0
  case MergeItem::MergeItemType::MMVD:
2225
0
    pu.mcControl           |= finalRd ? 0 : ( pu.mmvdMergeIdx.pos.step > 2 || m_pcEncCfg->m_MMVD > 1 ) ? 1 : 0;
2226
0
    mergeItem->noBdofRefine = pu.mccNoBdof() && pu.cs->sps->BDOF && !pu.cs->picHeader->disBdofFlag;
2227
0
    m_cInterSearch.motionCompensation( pu, dstBuf, REF_PIC_LIST_X );
2228
0
    break;
2229
2230
0
  case MergeItem::MergeItemType::SBTMVP:
2231
0
    m_cInterSearch.motionCompensation( pu, dstBuf, REF_PIC_LIST_X );
2232
0
    break;
2233
2234
0
  case MergeItem::MergeItemType::AFFINE:
2235
0
    m_cInterSearch.motionCompensation( pu, dstBuf, REF_PIC_LIST_X );
2236
0
    break;
2237
2238
0
  case MergeItem::MergeItemType::GPM:
2239
    // here predBuf1 and predBuf2 point to geoBuffer[mergeCand0] and geoBuffer[mergeCand1], respectively
2240
0
    CHECK( predBuf1 == nullptr || predBuf2 == nullptr, "Invalid input buffer to GPM" );
2241
0
    m_cInterSearch.weightedGeoBlk( pu.slice->clpRngs, pu, pu.geoSplitDir, luma && chroma ? MAX_NUM_CH : luma ? CH_L : CH_C, dstBuf, *predBuf1, *predBuf2 );
2242
0
    break;
2243
2244
0
  default:
2245
0
    THROW("Wrong merge item type");
2246
0
  }
2247
2248
0
  auto mergeItemPredBuf = mergeItem->getPredBuf( unitArea );
2249
2250
0
  if( dstBuf.Y().buf == mergeItemPredBuf.Y().buf )
2251
0
  {
2252
    // dst is the internal buffer
2253
0
    mergeItem->lumaPredReady   |= luma;
2254
0
    mergeItem->chromaPredReady |= chroma;
2255
0
  }
2256
0
  else if( finalRd && !forceNoResidual )
2257
0
  {
2258
    // at final RD stage, with and without residuals are both checked
2259
    // it makes sense to buffer the prediction
2260
0
    mergeItemPredBuf.copyFrom( dstBuf, luma, chroma );
2261
0
    mergeItem->lumaPredReady   |= luma;
2262
0
    mergeItem->chromaPredReady |= chroma;
2263
0
  }
2264
0
}
2265
2266
void EncCu::addRegularCandsToPruningList( const MergeCtx &mergeCtx, const UnitArea &localUnitArea, double sqrtLambdaForFirstPassIntra, const TempCtx &ctxStart,
2267
                                          DistParam& distParam, CodingUnit& pu, bool* sameMv, MergeBufVector& regularPred )
2268
0
{
2269
0
  pu.geo = pu.affine
2270
0
         = pu.mmvdMergeFlag = pu.mmvdSkip
2271
0
         = pu.ciip
2272
0
         = false;
2273
2274
0
  for( uint32_t uiMergeCand = 0; uiMergeCand < mergeCtx.numValidMergeCand; uiMergeCand++ )
2275
0
  {
2276
0
    if( sameMv[uiMergeCand] ) continue;
2277
2278
0
    mergeCtx.setMergeInfo   ( pu, uiMergeCand );
2279
2280
0
    if( m_pcEncCfg->m_ifpLines && // what about DMVR?
2281
0
        ( ( pu.refIdx[L0] >= 0 && !CU::isMvInRangeFPP( pu.ly(), pu.lheight(), pu.mv[L0][0].ver, m_pcEncCfg->m_ifpLines, *pu.cs->pcv ) ) ||
2282
0
          ( pu.refIdx[L1] >= 0 && !CU::isMvInRangeFPP( pu.ly(), pu.lheight(), pu.mv[L1][0].ver, m_pcEncCfg->m_ifpLines, *pu.cs->pcv ) ) ) )
2283
0
    {
2284
0
      continue;
2285
0
    }
2286
2287
0
    pu.interDir             = mergeCtx.interDirNeighbours[uiMergeCand];
2288
0
    pu.BcwIdx               = pu.interDir == 3 ? mergeCtx.BcwIdx[uiMergeCand] : BCW_DEFAULT;
2289
0
    pu.imv                  = mergeCtx.useAltHpelIf[uiMergeCand] ? IMV_HPEL : IMV_OFF;
2290
0
    CU::spanMotionInfo      ( pu );
2291
2292
0
    MergeItem *regularMerge = m_mergeItemList.allocateNewMergeItem();
2293
0
    regularMerge->importMergeInfo( mergeCtx, uiMergeCand, MergeItem::MergeItemType::REGULAR, pu );
2294
0
    auto dstBuf             = regularMerge->getPredBuf( localUnitArea );
2295
0
    generateMergePrediction ( localUnitArea, regularMerge, pu, true, true, dstBuf, false, false, &regularPred[uiMergeCand], nullptr );
2296
0
    regularMerge->cost      = calcLumaCost4MergePrediction( ctxStart, dstBuf, sqrtLambdaForFirstPassIntra, pu, distParam );
2297
0
    if( CU::checkDMVRCondition( pu ) ) std::copy_n( pu.mvdL0SubPu, getDmvrMvdNum( pu ), m_subPuMvOffset[uiMergeCand].data() );
2298
0
    m_mergeItemList         . insertMergeItemToList( regularMerge );
2299
0
  }
2300
0
}
2301
2302
void EncCu::addCiipCandsToPruningList( const MergeCtx &mergeCtx, const UnitArea &localUnitArea, double sqrtLambdaForFirstPassIntra, const TempCtx &ctxStart, DistParam &distParam, CodingUnit &pu, bool* sameMv )
2303
0
{
2304
0
  int                numCiipIntra = -1;
2305
0
  PelUnitBuf         ciipBuf      = m_aTmpStorageLCU[1].getCompactBuf( pu );
2306
2307
0
  pu.ciip        = true;
2308
0
  pu.intraDir[0] = PLANAR_IDX;
2309
0
  pu.geo         = pu.affine
2310
0
                 = pu.mmvdMergeFlag = pu.mmvdSkip
2311
0
                 = false;
2312
0
  m_cIntraSearch . initIntraPatternChType        ( pu, pu.Y() );
2313
0
  m_cIntraSearch . predIntraAng                  ( COMP_Y, ciipBuf.Y(), pu );
2314
0
  numCiipIntra   = m_cIntraSearch.getNumIntraCiip( pu );
2315
2316
0
  int nonCiipMrgCnds[MRG_MAX_NUM_CANDS] = { 0, };
2317
0
  int numNonCiipCnds                    =   0;
2318
0
  for( ; numNonCiipCnds < m_mergeItemList.size(); numNonCiipCnds++ ) nonCiipMrgCnds[numNonCiipCnds] = m_mergeItemList.getMergeItemInList( numNonCiipCnds )->mergeIdx;
2319
2320
0
  for( int i = 0; i < numNonCiipCnds; i++ )
2321
0
  {
2322
0
    const unsigned int uiMergeCand = nonCiipMrgCnds[i];
2323
2324
0
    if( sameMv[uiMergeCand] ) continue;
2325
2326
0
    mergeCtx.setMergeInfo     ( pu, uiMergeCand );
2327
2328
0
    if( m_pcEncCfg->m_ifpLines && 
2329
0
        ( ( pu.refIdx[L0] >= 0 && !CU::isMvInRangeFPP( pu.ly(), pu.lheight(), pu.mv[L0][0].ver, m_pcEncCfg->m_ifpLines, *pu.cs->pcv ) ) ||
2330
0
          ( pu.refIdx[L1] >= 0 && !CU::isMvInRangeFPP( pu.ly(), pu.lheight(), pu.mv[L1][0].ver, m_pcEncCfg->m_ifpLines, *pu.cs->pcv ) ) ) )
2331
0
    {
2332
0
      continue;
2333
0
    }
2334
2335
0
    pu.interDir               = mergeCtx.interDirNeighbours[uiMergeCand];
2336
0
    pu.BcwIdx                 = pu.interDir == 3 ? mergeCtx.BcwIdx[uiMergeCand] : BCW_DEFAULT;
2337
0
    pu.imv                    = mergeCtx.useAltHpelIf[uiMergeCand] ? IMV_HPEL : IMV_OFF;
2338
0
    CU::spanMotionInfo        ( pu );
2339
2340
0
    MergeItem* ciipMerge      = m_mergeItemList.allocateNewMergeItem();
2341
0
    ciipMerge->importMergeInfo( mergeCtx, uiMergeCand, MergeItem::MergeItemType::CIIP, pu );
2342
0
    ciipMerge->numCiipIntra   = numCiipIntra;
2343
0
    auto dstBuf               = ciipMerge->getPredBuf( localUnitArea );
2344
0
    generateMergePrediction   ( localUnitArea, ciipMerge, pu, true, false, dstBuf, false, false, nullptr, &ciipBuf );
2345
0
      ciipMerge->cost         = calcLumaCost4MergePrediction( ctxStart, dstBuf, sqrtLambdaForFirstPassIntra, pu, distParam );
2346
0
    if( !m_mergeItemList      . insertMergeItemToList( ciipMerge ) && m_pcEncCfg->m_CIIP > 1 )
2347
0
    {
2348
0
      break;
2349
0
    }
2350
0
  }
2351
0
}
2352
2353
void EncCu::addMmvdCandsToPruningList( const MergeCtx &mergeCtx, const UnitArea &localUnitArea, double sqrtLambdaForFirstPassIntra, const TempCtx& ctxStart,
2354
                                       DistParam& distParam, CodingUnit& pu )
2355
0
{
2356
0
  pu.mmvdSkip              = true;
2357
0
  pu.affine                = pu.geo
2358
0
                           = pu.ciip
2359
0
                           = false;
2360
2361
0
  int       mmvdTestNum    = mergeCtx.numValidMergeCand > 1 ? MmvdIdx::ADD_NUM : MmvdIdx::ADD_NUM >> 1;
2362
0
  int       bestDir        = 0;
2363
0
  size_t    curListSize    = m_mergeItemList.size();
2364
0
  double    bestCostMerge  = m_mergeItemList.getMergeItemInList( curListSize - 1 )->cost;
2365
0
  double    bestCostOffset = MAX_DOUBLE;
2366
0
  int       shiftCandStart = 0;
2367
2368
0
  if( m_pcEncCfg->m_MMVD == 4 )
2369
0
  {
2370
0
    const int cnd1idx = m_mergeItemList.size() == 1 ? 0 : 1;
2371
0
    const int mrgCnd0 = m_mergeItemList.getMergeItemInList(       0 )->mergeIdx;
2372
0
    const int mrgCnd1 = m_mergeItemList.getMergeItemInList( cnd1idx )->mergeIdx;
2373
2374
0
    if( mrgCnd0 > 1 && mrgCnd1 > 1 )
2375
0
    {
2376
0
      mmvdTestNum = 0;
2377
0
    }
2378
0
    else if( mrgCnd0 > 1 || mrgCnd1 > 1 )
2379
0
    {
2380
0
      int shiftCand = mrgCnd0 < 2 ? mrgCnd0 : mrgCnd1;
2381
2382
0
      if( shiftCand )
2383
0
      {
2384
0
        shiftCandStart = MMVD_MAX_REFINE_NUM;
2385
0
      }
2386
0
      else
2387
0
      {
2388
0
        mmvdTestNum    = MMVD_MAX_REFINE_NUM;
2389
0
      }
2390
0
    }
2391
0
  }
2392
2393
0
  for( int mmvdMergeCand = shiftCandStart; mmvdMergeCand < mmvdTestNum; mmvdMergeCand++ )
2394
0
  {
2395
0
    MmvdIdx mmvdIdx;
2396
0
    mmvdIdx.val = mmvdMergeCand;
2397
2398
0
    if( mmvdIdx.pos.step >= m_pcEncCfg->m_MmvdDisNum )
2399
0
    {
2400
0
      continue;
2401
0
    }
2402
2403
0
    if( m_pcEncCfg->m_MMVD > 1 )
2404
0
    {
2405
0
      int checkMMVD = xCheckMMVDCand( mmvdIdx, bestDir, mmvdTestNum, bestCostOffset, bestCostMerge, m_mergeItemList.getMergeItemInList( curListSize - 1 )->cost );
2406
0
      mmvdMergeCand = mmvdIdx.val;
2407
2408
0
      if( checkMMVD )
2409
0
      {
2410
0
        if( checkMMVD == 2 )
2411
0
        {
2412
0
          break;
2413
0
        }
2414
0
        continue;
2415
0
      }
2416
0
    }
2417
2418
0
    mergeCtx.setMmvdMergeCandiInfo( pu, mmvdIdx );
2419
2420
0
    if( m_pcEncCfg->m_ifpLines &&
2421
0
        ( ( pu.refIdx[L0] >= 0 && !CU::isMvInRangeFPP( pu.ly(), pu.lheight(), pu.mv[L0][0].ver, m_pcEncCfg->m_ifpLines, *pu.cs->pcv ) ) ||
2422
0
          ( pu.refIdx[L1] >= 0 && !CU::isMvInRangeFPP( pu.ly(), pu.lheight(), pu.mv[L1][0].ver, m_pcEncCfg->m_ifpLines, *pu.cs->pcv ) ) ) )
2423
0
    {
2424
      // skip candidate
2425
0
      continue;
2426
0
    }
2427
2428
0
    pu.interDir               = mergeCtx.interDirNeighbours[mmvdIdx.pos.baseIdx];
2429
0
    pu.BcwIdx                 = pu.interDir == 3 ? mergeCtx.BcwIdx[mmvdIdx.pos.baseIdx] : BCW_DEFAULT;
2430
0
    pu.imv                    = mergeCtx.useAltHpelIf[mmvdIdx.pos.baseIdx] ? IMV_HPEL : IMV_OFF;
2431
0
    CU::spanMotionInfo        ( pu );
2432
2433
0
    MergeItem *mmvdMerge      = m_mergeItemList.allocateNewMergeItem();
2434
0
    mmvdMerge->importMergeInfo( mergeCtx, mmvdIdx.val, MergeItem::MergeItemType::MMVD, pu );
2435
0
    auto dstBuf               = mmvdMerge->getPredBuf( localUnitArea );
2436
0
    generateMergePrediction   ( localUnitArea, mmvdMerge, pu, true, false, dstBuf, false, false, nullptr, nullptr );
2437
0
    mmvdMerge->cost           = calcLumaCost4MergePrediction( ctxStart, dstBuf, sqrtLambdaForFirstPassIntra, pu, distParam );
2438
0
    m_mergeItemList           . insertMergeItemToList( mmvdMerge );
2439
2440
0
    if( m_pcEncCfg->m_MMVD > 1 && mmvdMerge->cost < bestCostOffset )
2441
0
    {
2442
0
      bestCostOffset          = mmvdMerge->cost;
2443
0
      int CandCur             = mmvdIdx.val - MMVD_MAX_REFINE_NUM * mmvdIdx.pos.baseIdx;
2444
0
      if( CandCur < 4 )
2445
0
        bestDir               = CandCur;
2446
0
    }
2447
0
  }
2448
2449
0
  if( m_pcEncCfg->m_useFastMrg >= 2 )
2450
0
  {
2451
0
    m_mergeItemList           . shrinkList( curListSize );
2452
0
  }
2453
0
}
2454
2455
void EncCu::addAffineCandsToPruningList( AffineMergeCtx &affineMergeCtx, const UnitArea &localUnitArea, double sqrtLambdaForFirstPass,
2456
                                         const TempCtx& ctxStart, DistParam& distParam, CodingUnit& pu)
2457
0
{
2458
0
  bool sameMV[AFFINE_MRG_MAX_NUM_CANDS + 1]
2459
0
                      = { false, };
2460
0
  size_t curListSize  = m_mergeItemList.size();
2461
2462
0
  pu.mergeFlag = true;
2463
0
  pu.affine    = true;
2464
0
  pu.imv       = IMV_OFF;
2465
0
  pu.geo       = pu.mmvdMergeFlag = pu.mmvdSkip
2466
0
               = pu.ciip
2467
0
               = false;
2468
2469
0
  if( m_pcEncCfg->m_Affine > 1 )
2470
0
  {
2471
0
    for( int m = 0; m < affineMergeCtx.numValidMergeCand; m++ )
2472
0
    {
2473
0
      if( pu.cs->slice->TLayer > 3 && affineMergeCtx.mergeType[m] != MRG_TYPE_SUBPU_ATMVP )
2474
0
      {
2475
0
        sameMV[m] = m != 0;
2476
0
      }
2477
0
      else if( !sameMV[m + 1] )
2478
0
      {
2479
0
        for( int n = m + 1; n < affineMergeCtx.numValidMergeCand; n++ )
2480
0
        {
2481
0
          sameMV[n] |= affineMergeCtx.mvFieldNeighbours[m][0][0] == affineMergeCtx.mvFieldNeighbours[n][0][0]
2482
0
                    && affineMergeCtx.mvFieldNeighbours[m][1][0] == affineMergeCtx.mvFieldNeighbours[n][1][0];
2483
0
        }
2484
0
      }
2485
0
    }
2486
0
  }
2487
2488
0
  for( uint32_t mergeIdx = 0; mergeIdx < affineMergeCtx.numValidMergeCand; mergeIdx++ )
2489
0
  {
2490
0
    if( ( affineMergeCtx.mergeType[mergeIdx] != MRG_TYPE_SUBPU_ATMVP && m_pcEncCfg->m_Affine == 0 ) || sameMV[mergeIdx] )
2491
0
    {
2492
0
      continue;
2493
0
    }
2494
2495
0
    pu.mergeType              = affineMergeCtx.mergeType[mergeIdx];
2496
0
    pu.affineType             = affineMergeCtx.affineType[mergeIdx];
2497
0
    pu.interDir               = affineMergeCtx.interDirNeighbours[mergeIdx];
2498
0
    pu.BcwIdx                 = pu.interDir == 3 ? affineMergeCtx.BcwIdx[mergeIdx] : BCW_DEFAULT;
2499
2500
    // generate motion buf for IFP
2501
0
    if( affineMergeCtx.mergeType[mergeIdx] == MRG_TYPE_SUBPU_ATMVP )
2502
0
    {
2503
0
      pu.refIdx[L0]           = affineMergeCtx.mvFieldNeighbours[mergeIdx][L0][0].refIdx;
2504
0
      pu.refIdx[L1]           = affineMergeCtx.mvFieldNeighbours[mergeIdx][L1][0].refIdx;
2505
0
      pu.mv    [L0][0]        = affineMergeCtx.mvFieldNeighbours[mergeIdx][L0][0].mv;
2506
0
      pu.mv    [L1][0]        = affineMergeCtx.mvFieldNeighbours[mergeIdx][L1][0].mv;
2507
0
      CU::spanMotionInfo      ( pu, &affineMergeCtx );
2508
0
    }
2509
0
    else
2510
0
    {
2511
0
      CU::setAllAffineMvField ( pu, affineMergeCtx.mvFieldNeighbours[mergeIdx][L0], L0 );
2512
0
      CU::setAllAffineMvField ( pu, affineMergeCtx.mvFieldNeighbours[mergeIdx][L1], L1 );
2513
0
      CU::spanMotionInfo      ( pu );
2514
0
    }
2515
2516
0
    if( m_pcEncCfg->m_ifpLines && !CU::isMotionBufInRangeFPP( pu, m_pcEncCfg->m_ifpLines ) )
2517
0
    {
2518
0
      continue;
2519
0
    }
2520
2521
0
    MergeItem *mergeItem   = m_mergeItemList.allocateNewMergeItem();
2522
0
    mergeItem->importMergeInfo( affineMergeCtx, mergeIdx, affineMergeCtx.mergeType[mergeIdx] == MRG_TYPE_SUBPU_ATMVP ? MergeItem::MergeItemType::SBTMVP : MergeItem::MergeItemType::AFFINE, pu );
2523
0
    auto dstBuf            = mergeItem->getPredBuf( localUnitArea );
2524
0
    generateMergePrediction( localUnitArea, mergeItem, pu, true, false, dstBuf, false, false, nullptr, nullptr );
2525
0
    mergeItem->cost        = calcLumaCost4MergePrediction( ctxStart, dstBuf, sqrtLambdaForFirstPass, pu, distParam );
2526
0
    m_mergeItemList        . insertMergeItemToList( mergeItem );
2527
0
  }
2528
0
  if( m_pcEncCfg->m_useFastMrg >= 2 )
2529
0
  {
2530
0
    m_mergeItemList        . shrinkList( curListSize );
2531
0
  }
2532
0
}
2533
2534
void EncCu::addGpmCandsToPruningList( const MergeCtx &mergeCtx, const UnitArea &localUnitArea, double sqrtLambdaForFirstPass,
2535
                                      const TempCtx& ctxStart, const GeoComboCostList& comboList, MergeBufVector& geoBuffer, DistParam& distParam, CodingUnit& pu)
2536
0
{
2537
0
  int geoNumMrgSadCand    = std::min( GEO_MAX_TRY_WEIGHTED_SAD, ( int ) comboList.list.size() );
2538
0
  geoNumMrgSadCand        = std::min( geoNumMrgSadCand, m_pcEncCfg->m_Geo > 2 ? 10 : GEO_MAX_TRY_WEIGHTED_SAD );
2539
0
  double bestGeoCost      = MAX_DOUBLE / 2.0;
2540
0
  MergeItem* best2geo[2]  = { nullptr, nullptr };
2541
2542
0
  pu.mergeFlag = true;
2543
0
  pu.geo       = true;
2544
0
  pu.mergeType = MRG_TYPE_DEFAULT_N;
2545
0
  pu.BcwIdx    = BCW_DEFAULT;
2546
0
  pu.interDir  = 3;
2547
0
  pu.imv       = IMV_OFF;
2548
0
  pu.affine    = pu.mmvdMergeFlag = pu.mmvdSkip
2549
0
               = pu.ciip
2550
0
               = false;
2551
2552
0
  for( int candidateIdx = 0; candidateIdx < geoNumMrgSadCand; candidateIdx++ )
2553
0
  {
2554
0
    const int          splitDir     = comboList.list[candidateIdx].splitDir;
2555
0
    const MergeIdxPair mergeIdxPair { comboList.list[candidateIdx].mergeIdx0, comboList.list[candidateIdx].mergeIdx1 };
2556
0
    const int          gpmIndex     = MergeItem::getGpmUnfiedIndex( splitDir, mergeIdxPair );
2557
2558
0
    pu.mergeIdx            = gpmIndex;
2559
0
    pu.geoMergeIdx         = mergeIdxPair;
2560
0
    pu.geoSplitDir         = splitDir;
2561
0
    CU::spanGeoMotionInfo  ( pu, mergeCtx, pu.geoSplitDir, pu.geoMergeIdx[0], pu.geoMergeIdx[1] );
2562
2563
0
    MergeItem *mergeItem   = m_mergeItemList.allocateNewMergeItem();
2564
0
    mergeItem->importMergeInfo( mergeCtx, gpmIndex, MergeItem::MergeItemType::GPM, pu );
2565
0
    auto dstBuf            = mergeItem->getPredBuf( localUnitArea );
2566
0
    generateMergePrediction( localUnitArea, mergeItem, pu, true, false, dstBuf, false, false, &geoBuffer[mergeIdxPair[0]], &geoBuffer[mergeIdxPair[1]] );
2567
0
    mergeItem->cost        = calcLumaCost4MergePrediction( ctxStart, dstBuf, sqrtLambdaForFirstPass, pu, distParam );
2568
0
    bestGeoCost            = std::min( mergeItem->cost, bestGeoCost );
2569
2570
0
    if( mergeItem->cost > MRG_FAST_RATIO[0] * bestGeoCost || mergeItem->cost > m_mergeBestSATDCost )
2571
0
    {
2572
0
      m_mergeItemList      . giveBackMergeItem( mergeItem );
2573
2574
0
      if( m_pcEncCfg->m_Geo > 2 ) break;
2575
0
    }
2576
0
    else if( m_pcEncCfg->m_Geo < 2 )
2577
0
    {
2578
0
      m_mergeItemList      . insertMergeItemToList( mergeItem );
2579
0
    }
2580
0
    else
2581
0
    {
2582
0
      if( ( m_mergeItemList.size() > 0 && m_mergeItemList.getMergeItemInList( m_mergeItemList.size() - 1 )->cost <= mergeItem->cost ) ||
2583
0
        ( best2geo[1] && best2geo[1]->cost <= mergeItem->cost ) )
2584
0
      {
2585
0
        m_mergeItemList    . giveBackMergeItem( mergeItem );
2586
0
      }
2587
0
      else
2588
0
      {
2589
0
        if( !best2geo[0] || mergeItem->cost < best2geo[0]->cost )
2590
0
        {
2591
0
          if( best2geo[1] )
2592
0
            m_mergeItemList. giveBackMergeItem( best2geo[1] );
2593
2594
0
          best2geo[1] = best2geo[0]; best2geo[0] = mergeItem;
2595
0
        }
2596
0
        else
2597
0
        {
2598
0
          if( best2geo[1] ) 
2599
0
            m_mergeItemList. giveBackMergeItem( best2geo[1] );
2600
2601
0
          best2geo[1] = mergeItem;
2602
0
        }
2603
0
      }
2604
0
    }
2605
0
  }
2606
2607
0
  if( best2geo[0] )
2608
0
    m_mergeItemList        . insertMergeItemToList( best2geo[0] );
2609
0
  if( best2geo[1] )
2610
0
    m_mergeItemList        . insertMergeItemToList( best2geo[1] );
2611
0
}
2612
2613
bool EncCu::prepareGpmComboList( const MergeCtx &mergeCtx, const UnitArea &localUnitArea, double sqrtLambdaForFirstPass,
2614
                                 GeoComboCostList& comboList, MergeBufVector& geoBuffer, CodingUnit& pu )
2615
0
{
2616
0
          sqrtLambdaForFirstPass /= FRAC_BITS_SCALE;
2617
0
  const int bitsForPartitionIdx   = floorLog2(GEO_NUM_PARTITION_MODE);
2618
0
  const int maxNumMergeCandidates = std::min( ( int ) pu.cs->sps->maxNumGeoCand, MRG_MAX_NUM_CANDS );
2619
0
  DistParam distParam;
2620
  // the second arguments to setDistParam is dummy and will be updated before being used
2621
0
  DistParam  distParamWholeBlk     = m_cRdCost.setDistParam( pu.cs->getOrgBuf().Y(), pu.cs->getOrgBuf().Y(), pu.cs->sps->bitDepths[ CH_L ], DF_SAD );
2622
0
  Distortion bestWholeBlkSad       = MAX_UINT64;
2623
0
  double     bestWholeBlkCost      = MAX_DOUBLE;
2624
0
  const ClpRng&  lclpRng           = pu.slice->clpRngs[COMP_Y];
2625
0
  const unsigned rshift            = std::max<int>( 2, ( IF_INTERNAL_PREC - lclpRng.bd ) );
2626
0
  const int      offset            = ( 1 << ( rshift - 1 ) ) + IF_INTERNAL_OFFS;
2627
0
  const int      numSamples        = pu.Y().area();
2628
0
  Distortion sadWholeBlk            [GEO_MAX_NUM_UNI_CANDS];
2629
0
  int        pocMrg                 [GEO_MAX_NUM_UNI_CANDS];
2630
0
  Mv         mergeMv                [GEO_MAX_NUM_UNI_CANDS];
2631
0
  bool       isSkipThisCand         [GEO_MAX_NUM_UNI_CANDS]
2632
0
                                   = { false, };
2633
0
  bool       sameMV                 [MRG_MAX_NUM_CANDS]
2634
0
                                   = { false, };
2635
0
  MergeBufVector geoTempBuf;
2636
2637
0
  if( m_pcEncCfg->m_Geo > 2 )
2638
0
  {
2639
0
    for( int m = 0; m < maxNumMergeCandidates; m++ )
2640
0
    {
2641
0
      if( !sameMV[m] )
2642
0
      {
2643
0
        for( int n = m + 1; n < maxNumMergeCandidates; n++ )
2644
0
        {
2645
0
          sameMV[n] |= mergeCtx.mvFieldNeighbours[m][0] == mergeCtx.mvFieldNeighbours[n][0]
2646
0
                    && mergeCtx.mvFieldNeighbours[m][1] == mergeCtx.mvFieldNeighbours[n][1];
2647
0
        }
2648
0
      }
2649
0
    }
2650
0
  }
2651
2652
0
  for( uint8_t mergeCand = 0; mergeCand < maxNumMergeCandidates; mergeCand++ )
2653
0
  {
2654
0
    geoBuffer .push_back ( m_aTmpStorageLCU[2                         + mergeCand].getCompactBuf( localUnitArea ) );
2655
0
    geoTempBuf.push_back ( m_aTmpStorageLCU[2 + GEO_MAX_NUM_UNI_CANDS + mergeCand].getCompactBuf( localUnitArea ) );
2656
2657
0
    const int  listIdx    = mergeCtx.mvFieldNeighbours[mergeCand][0]      .refIdx == -1 ? 1 : 0;
2658
0
    const auto refPicList = RefPicList(listIdx);
2659
0
    const int  refIdx     = mergeCtx.mvFieldNeighbours[mergeCand][listIdx].refIdx;
2660
2661
0
    pocMrg [mergeCand]    = pu.cs->slice->getRefPic( refPicList, refIdx )->poc;
2662
0
    mergeMv[mergeCand]    = mergeCtx.mvFieldNeighbours[mergeCand][listIdx].mv;
2663
2664
0
    for( int i = 0; i < mergeCand; i++ )
2665
0
    {
2666
0
      if( pocMrg[mergeCand] == pocMrg[i] && mergeMv[mergeCand] == mergeMv[i] )
2667
0
      {
2668
0
        isSkipThisCand[mergeCand] = true;
2669
0
        break;
2670
0
      }
2671
0
    }
2672
2673
0
    if( sameMV[mergeCand] )
2674
0
    {
2675
0
      continue;
2676
0
    }
2677
2678
0
    if( m_pcEncCfg->m_ifpLines ) 
2679
0
    {
2680
0
      bool isOutOfRange  = !CU::isMvInRangeFPP( pu.ly(), pu.lheight(), mergeCtx.mvFieldNeighbours[mergeCand][0].mv.ver, m_pcEncCfg->m_ifpLines, *pu.cs->pcv );
2681
0
           isOutOfRange |= !CU::isMvInRangeFPP( pu.ly(), pu.lheight(), mergeCtx.mvFieldNeighbours[mergeCand][1].mv.ver, m_pcEncCfg->m_ifpLines, *pu.cs->pcv );
2682
2683
      // use sameMV to surpress processing of this cand later on...
2684
0
      sameMV[mergeCand] |= isOutOfRange;
2685
2686
0
      if( isOutOfRange )
2687
0
        continue;
2688
0
    }
2689
2690
0
    mergeCtx.setMergeInfo            ( pu, mergeCand );
2691
0
    CU::spanMotionInfo               ( pu );
2692
0
    m_cInterSearch.motionCompensation( pu, geoBuffer[mergeCand], REF_PIC_LIST_X );
2693
2694
0
    g_pelBufOP.roundGeo( geoBuffer[mergeCand].Y().buf, geoTempBuf[mergeCand].Y().buf, numSamples, rshift, offset, lclpRng );
2695
2696
0
    distParamWholeBlk.cur  = geoTempBuf[mergeCand].Y();
2697
0
    sadWholeBlk[mergeCand] = distParamWholeBlk.distFunc( distParamWholeBlk );
2698
2699
0
    if( sadWholeBlk[mergeCand] < bestWholeBlkSad )
2700
0
    {
2701
0
      bestWholeBlkSad  = sadWholeBlk[mergeCand];
2702
0
      int bitsCand     = mergeCand + 1;
2703
0
      bestWholeBlkCost = ( double ) bestWholeBlkSad + ( double ) bitsCand * sqrtLambdaForFirstPass;
2704
0
    }
2705
0
  }
2706
2707
0
  bool allCandsAreSame = true;
2708
0
  for( uint8_t mergeCand = 1; mergeCand < maxNumMergeCandidates; mergeCand++ )
2709
0
  {
2710
0
    allCandsAreSame &= isSkipThisCand[mergeCand];
2711
0
  }
2712
0
  if( allCandsAreSame )
2713
0
  {
2714
0
    return false;
2715
0
  }
2716
2717
0
  const int wIdx = floorLog2( pu.lwidth() )  - GEO_MIN_CU_LOG2;
2718
0
  const int hIdx = floorLog2( pu.lheight() ) - GEO_MIN_CU_LOG2;
2719
2720
0
  for( int splitDir = 0; splitDir < GEO_NUM_PARTITION_MODE; )
2721
0
  {
2722
0
    int maskStride = 0, maskStride2 = 0;
2723
0
    int stepX = 1;
2724
0
    Pel *sadMask;
2725
0
    int16_t angle = g_GeoParams[splitDir][0];
2726
    
2727
0
    if( g_angle2mirror[angle] == 2 )
2728
0
    {
2729
0
      maskStride  = -GEO_WEIGHT_MASK_SIZE;
2730
0
      maskStride2 = -( int ) pu.lwidth();
2731
0
      sadMask     = &g_globalGeoEncSADmask[g_angle2mask[g_GeoParams[splitDir][0]]]
2732
0
                      [( GEO_WEIGHT_MASK_SIZE - 1 - g_weightOffset[hIdx][wIdx][splitDir][1] ) * GEO_WEIGHT_MASK_SIZE
2733
0
                                                  + g_weightOffset[hIdx][wIdx][splitDir][0]
2734
0
                      ];
2735
0
    }
2736
0
    else if( g_angle2mirror[angle] == 1 )
2737
0
    {
2738
0
      stepX       = -1;
2739
0
      maskStride2 = pu.lwidth();
2740
0
      maskStride  = GEO_WEIGHT_MASK_SIZE;
2741
0
      sadMask     = &g_globalGeoEncSADmask[g_angle2mask[g_GeoParams[splitDir][0]]]
2742
0
                      [     GEO_WEIGHT_MASK_SIZE *     g_weightOffset[hIdx][wIdx][splitDir][1]
2743
0
                        + ( GEO_WEIGHT_MASK_SIZE - 1 - g_weightOffset[hIdx][wIdx][splitDir][0] )
2744
0
                      ];
2745
0
    }
2746
0
    else
2747
0
    {
2748
0
      maskStride  = GEO_WEIGHT_MASK_SIZE;
2749
0
      maskStride2 = -( int ) pu.lwidth();
2750
0
      sadMask     = &g_globalGeoEncSADmask[g_angle2mask[g_GeoParams[splitDir][0]]]
2751
0
                      [   g_weightOffset[hIdx][wIdx][splitDir][1] * GEO_WEIGHT_MASK_SIZE
2752
0
                        + g_weightOffset[hIdx][wIdx][splitDir][0]
2753
0
                      ];
2754
0
    }
2755
2756
0
    m_cRdCost.setDistParamGeo ( distParam, pu.cs->getOrgBuf().Y(),
2757
0
                                nullptr, 0,
2758
0
                                sadMask, maskStride, stepX, maskStride2,
2759
0
                                pu.cs->sps->bitDepths[CH_L], COMP_Y );
2760
2761
0
    for( uint8_t mergeCand = 0; mergeCand < maxNumMergeCandidates; mergeCand++ )
2762
0
    {
2763
0
      if( sameMV[mergeCand] )
2764
0
      {
2765
0
        continue;
2766
0
      }
2767
2768
0
      distParam.cur.buf         = geoTempBuf[mergeCand].Y().buf;
2769
0
      distParam.cur.stride      = geoTempBuf[mergeCand].Y().stride;
2770
0
      const Distortion sadLarge = distParam.distFunc( distParam );
2771
0
      const Distortion sadSmall = sadWholeBlk[mergeCand] - sadLarge;
2772
2773
0
      const int bitsCand        = mergeCand + 1;
2774
2775
0
      const double cost0        = ( double ) sadLarge + ( double ) bitsCand * sqrtLambdaForFirstPass;
2776
0
      const double cost1        = ( double ) sadSmall + ( double ) bitsCand * sqrtLambdaForFirstPass;
2777
2778
0
      m_GeoCostList.insert( splitDir, 0, mergeCand, cost0 );
2779
0
      m_GeoCostList.insert( splitDir, 1, mergeCand, cost1 );
2780
0
    }
2781
2782
0
    if( m_pcEncCfg->m_Geo == 4 )
2783
0
    {
2784
0
      if( splitDir == 1 )
2785
0
      {
2786
0
        splitDir += 7;
2787
0
      }
2788
0
      else if( splitDir == 35 || ( splitDir + 1 ) % 4 != 0 )
2789
0
      {
2790
0
        splitDir++;
2791
0
      }
2792
0
      else
2793
0
      {
2794
0
        splitDir += 5;
2795
0
      }
2796
0
    }
2797
0
    else
2798
0
    {
2799
0
      splitDir++;
2800
0
    }
2801
0
  }
2802
2803
0
  comboList.list.clear();
2804
2805
0
  for( int splitDir = 0; splitDir < GEO_NUM_PARTITION_MODE; )
2806
0
  {
2807
0
    for( int geoMotionIdx = 0; geoMotionIdx < maxNumMergeCandidates * ( maxNumMergeCandidates - 1 ); geoMotionIdx++ )
2808
0
    {
2809
0
      const MergeIdxPair mergeIdxPair = m_GeoModeTest[geoMotionIdx];
2810
2811
0
      if( sameMV[mergeIdxPair[0]] || sameMV[mergeIdxPair[1]] )
2812
0
      {
2813
0
        continue;
2814
0
      }
2815
2816
0
      double tempCost = m_GeoCostList.getCost( splitDir, mergeIdxPair[0], mergeIdxPair[1] );
2817
2818
0
      if( tempCost > bestWholeBlkCost )
2819
0
      {
2820
0
        continue;
2821
0
      }
2822
2823
0
      tempCost = tempCost + ( double ) bitsForPartitionIdx * sqrtLambdaForFirstPass;
2824
0
      comboList.list.push_back( GeoMergeCombo{ splitDir, mergeIdxPair[0], mergeIdxPair[1], tempCost } );
2825
0
    }
2826
2827
0
    if( m_pcEncCfg->m_Geo == 4 )
2828
0
    {
2829
0
      if( splitDir == 1 )
2830
0
      {
2831
0
        splitDir += 7;
2832
0
      }
2833
0
      else if( splitDir == 35 || ( splitDir + 1 ) % 4 != 0 )
2834
0
      {
2835
0
        splitDir++;
2836
0
      }
2837
0
      else
2838
0
      {
2839
0
        splitDir += 5;
2840
0
      }
2841
0
    }
2842
0
    else
2843
0
    {
2844
0
      splitDir++;
2845
0
    }
2846
0
  }
2847
2848
0
  if( comboList.list.empty() )
2849
0
  {
2850
0
    return false;
2851
0
  }
2852
2853
0
  comboList.sortByCost();
2854
0
  return true;
2855
0
}
2856
2857
double EncCu::calcLumaCost4MergePrediction( const TempCtx &ctxStart, const PelUnitBuf &predBuf, double lambda, CodingUnit &cu, DistParam &distParam )
2858
0
{
2859
0
  distParam.cur = predBuf.Y();
2860
0
  auto dist     = distParam.distFunc(distParam);
2861
2862
0
  m_CABACEstimator->getCtx() = ctxStart;
2863
0
  auto fracBits = xCalcPuMeBits( cu );
2864
2865
0
  double cost   = ( double ) dist + ( double ) fracBits * lambda;
2866
2867
0
  m_uiSadBestForQPA = std::min( dist, m_uiSadBestForQPA );
2868
2869
0
  return cost;
2870
0
}
2871
2872
//////////////////////////////////////////////////////////////////////////////////////////////
2873
// ibc merge/skip mode check
2874
void EncCu::xCheckRDCostIBCModeMerge2Nx2N(CodingStructure*& tempCS, CodingStructure*& bestCS, Partitioner& partitioner, const EncTestMode& encTestMode)
2875
24.0k
{
2876
24.0k
  assert(partitioner.chType != CH_C); // chroma IBC is derived
2877
24.0k
  if (tempCS->area.lwidth() == 128 || tempCS->area.lheight() == 128) // disable IBC mode larger than 64x64
2878
0
  {
2879
0
    return;
2880
0
  }
2881
2882
24.0k
  if ((m_pcEncCfg->m_IBCFastMethod > 1) && !bestCS->slice->isIntra() && (bestCS->cus.size() != 0))
2883
0
  {
2884
0
    if (bestCS->getCU(partitioner.chType, partitioner.treeType)->skip)
2885
0
    {
2886
0
      return;
2887
0
    }
2888
0
  }
2889
2890
24.0k
  const SPS& sps = *tempCS->sps;
2891
2892
24.0k
  tempCS->initStructData(encTestMode.qp);
2893
24.0k
  MergeCtx mergeCtx;
2894
2895
24.0k
  {
2896
    // first get merge candidates
2897
24.0k
    CodingUnit cu(tempCS->area);
2898
24.0k
    cu.cs = tempCS;
2899
24.0k
    cu.predMode = MODE_IBC;
2900
24.0k
    cu.slice = tempCS->slice;
2901
24.0k
    cu.tileIdx = m_tileIdx;
2902
24.0k
    cu.initPuData();
2903
24.0k
    cu.cs = tempCS;
2904
24.0k
    cu.mmvdSkip = false;
2905
24.0k
    cu.mmvdMergeFlag = false;
2906
24.0k
    cu.geo = false;
2907
24.0k
    CU::getIBCMergeCandidates(cu, mergeCtx);
2908
24.0k
  }
2909
24.0k
  int candHasNoResidual[MRG_MAX_NUM_CANDS];
2910
168k
  for (unsigned int ui = 0; ui < mergeCtx.numValidMergeCand; ui++)
2911
144k
  {
2912
144k
    candHasNoResidual[ui] = 0;
2913
144k
  }
2914
2915
24.0k
  bool                                        bestIsSkip = false;
2916
24.0k
  unsigned                                    numMrgSATDCand = mergeCtx.numValidMergeCand;
2917
24.0k
  static_vector<unsigned, MRG_MAX_NUM_CANDS>  RdModeList(MRG_MAX_NUM_CANDS);
2918
168k
  for (unsigned i = 0; i < MRG_MAX_NUM_CANDS; i++)
2919
144k
  {
2920
144k
    RdModeList[i] = i;
2921
144k
  }
2922
2923
  //{
2924
24.0k
  static_vector<double, MRG_MAX_NUM_CANDS>  candCostList(MRG_MAX_NUM_CANDS, MAX_DOUBLE);
2925
  // 1. Pass: get SATD-cost for selected candidates and reduce their count
2926
24.0k
  {
2927
24.0k
    const double sqrtLambdaForFirstPass = m_cRdCost.getMotionLambda();
2928
2929
24.0k
    CodingUnit& cu = tempCS->addCU(CS::getArea(*tempCS, tempCS->area, partitioner.chType,partitioner.treeType), partitioner.chType);
2930
2931
24.0k
    partitioner.setCUData(cu);
2932
24.0k
    cu.slice = tempCS->slice;
2933
24.0k
    cu.tileIdx = m_tileIdx;
2934
24.0k
    cu.skip = false;
2935
24.0k
    cu.predMode = MODE_IBC;
2936
24.0k
    cu.chromaQpAdj = m_cuChromaQpOffsetIdxPlus1;
2937
24.0k
    cu.qp = encTestMode.qp;
2938
24.0k
    cu.mmvdSkip = false;
2939
24.0k
    cu.geo = false;
2940
24.0k
    DistParam distParam;
2941
24.0k
    cu.initPuData();
2942
24.0k
    cu.mmvdMergeFlag = false;
2943
24.0k
    Picture* refPic = cu.slice->pic;
2944
24.0k
    const UnitArea localUnitArea(tempCS->area.chromaFormat, Area(cu.blocks[COMP_Y].x, cu.blocks[COMP_Y].y, tempCS->area.Y().width, tempCS->area.Y().height));
2945
24.0k
    const CompArea& compArea = localUnitArea.block(COMP_Y);
2946
24.0k
    const CPelBuf refBuf = refPic->getRecoBuf(compArea);
2947
24.0k
    const Pel* piRefSrch = refBuf.buf;
2948
2949
24.0k
      distParam = m_cRdCost.setDistParam(tempCS->getOrgBuf(COMP_Y), refBuf, sps.bitDepths[CH_L], DF_HAD);
2950
24.0k
    int refStride = refBuf.stride;
2951
2952
24.0k
    int numValidBv = mergeCtx.numValidMergeCand;
2953
168k
    for (unsigned int mergeCand = 0; mergeCand < mergeCtx.numValidMergeCand; mergeCand++)
2954
144k
    {
2955
144k
      mergeCtx.setMergeInfo(cu, mergeCand); // set bv info in merge mode
2956
144k
      const int cuPelX = cu.Y().x;
2957
144k
      const int cuPelY = cu.Y().y;
2958
144k
      int roiWidth     = cu.lwidth();
2959
144k
      int roiHeight    = cu.lheight();
2960
144k
      const int picWidth  = cu.cs->slice->pps->picWidthInLumaSamples;
2961
144k
      const int picHeight = cu.cs->slice->pps->picHeightInLumaSamples;
2962
144k
      const unsigned int lcuWidth = cu.cs->slice->sps->CTUSize;
2963
2964
144k
      Mv bv = cu.mv[0][0];
2965
144k
      bv.changePrecision( MV_PRECISION_INTERNAL, MV_PRECISION_INT);
2966
144k
      int xPred = bv.hor;
2967
144k
      int yPred = bv.ver;
2968
      
2969
144k
      if( !m_cInterSearch.searchBvIBC( cu, cuPelX, cuPelY, roiWidth, roiHeight, picWidth, picHeight, xPred, yPred, lcuWidth ) ) // not valid bv derived
2970
144k
      {
2971
144k
        numValidBv--;
2972
144k
        continue;
2973
144k
      }
2974
0
      CU::spanMotionInfo(cu);
2975
0
      distParam.cur.buf = piRefSrch + refStride * yPred + xPred;
2976
2977
0
      Distortion sad = distParam.distFunc(distParam);
2978
0
      unsigned int bitsCand = mergeCand + 1;
2979
0
      if (mergeCand == tempCS->sps->maxNumIBCMergeCand - 1)
2980
0
      {
2981
0
        bitsCand--;
2982
0
      }
2983
0
      double cost = (double)sad + (double)bitsCand * sqrtLambdaForFirstPass;
2984
2985
0
      updateCandList( mergeCand, cost, RdModeList, candCostList, numMrgSATDCand );
2986
0
    }
2987
2988
    // Try to limit number of candidates using SATD-costs
2989
24.0k
    if (numValidBv)
2990
0
    {
2991
0
      numMrgSATDCand = numValidBv;
2992
0
      for (unsigned int i = 1; i < numValidBv; i++)
2993
0
      {
2994
0
        if (candCostList[i] > MRG_FAST_RATIO[0] * candCostList[0])
2995
0
        {
2996
0
          numMrgSATDCand = i;
2997
0
          break;
2998
0
        }
2999
0
      }
3000
0
    }
3001
24.0k
    else
3002
24.0k
    {
3003
24.0k
      tempCS->dist = 0;
3004
24.0k
      tempCS->fracBits = 0;
3005
24.0k
      tempCS->cost = MAX_DOUBLE;
3006
24.0k
      tempCS->costDbOffset = 0;
3007
24.0k
      tempCS->initStructData(encTestMode.qp);
3008
24.0k
      return;
3009
24.0k
    }
3010
3011
0
    tempCS->initStructData(encTestMode.qp);
3012
0
  }
3013
  //}
3014
3015
3016
0
  const unsigned int iteration = 2;
3017
 // m_bestModeUpdated = tempCS->cost = bestCS->cost = false;
3018
  // 2. Pass: check candidates using full RD test
3019
0
  for (unsigned int numResidualPass = 0; numResidualPass < iteration; numResidualPass++)
3020
0
  {
3021
0
    for (unsigned int mrgHADIdx = 0; mrgHADIdx < numMrgSATDCand; mrgHADIdx++)
3022
0
    {
3023
0
      unsigned int mergeCand = RdModeList[mrgHADIdx];
3024
0
      if (!(numResidualPass == 1 && candHasNoResidual[mergeCand] == 1))
3025
0
      {
3026
0
        if (!(bestIsSkip && (numResidualPass == 0)))
3027
0
        {
3028
0
          {
3029
3030
            // first get merge candidates
3031
0
            CodingUnit& cu = tempCS->addCU(CS::getArea(*tempCS, tempCS->area, (const ChannelType)partitioner.chType,partitioner.treeType), (const ChannelType)partitioner.chType);
3032
3033
0
            partitioner.setCUData(cu);
3034
0
            cu.slice = tempCS->slice;
3035
0
            cu.tileIdx = m_tileIdx;
3036
0
            cu.skip = false;
3037
0
            cu.predMode = MODE_IBC;
3038
0
            cu.chromaQpAdj = m_cuChromaQpOffsetIdxPlus1;
3039
0
            cu.qp = encTestMode.qp;
3040
0
            cu.sbtInfo = 0;
3041
0
            cu.initPuData();
3042
0
            cu.intraDir[0] = DC_IDX; // set intra pred for ibc block
3043
0
            cu.intraDir[1] = PLANAR_IDX; // set intra pred for ibc block
3044
0
            cu.mmvdSkip = false;
3045
0
            cu.mmvdMergeFlag = false;
3046
0
            cu.geo = false;
3047
0
            mergeCtx.setMergeInfo(cu, mergeCand);
3048
0
            CU::spanMotionInfo(cu);
3049
3050
0
            assert(mergeCtx.mrgTypeNeighbours[mergeCand] == MRG_TYPE_IBC);
3051
0
            const bool chroma = !CU::isSepTree(cu);
3052
3053
            //  MC
3054
0
            cu.mcControl = chroma ? 0: 2;
3055
0
            m_cInterSearch.motionCompensationIBC(cu, tempCS->getPredBuf());
3056
0
            m_CABACEstimator->getCtx() = m_CurrCtx->start;
3057
3058
0
            m_cInterSearch.encodeResAndCalcRdInterCU(*tempCS, partitioner, (numResidualPass != 0));
3059
0
            cu.mcControl = 0;
3060
0
            xEncodeDontSplit(*tempCS, partitioner);
3061
0
            xCheckDQP(*tempCS, partitioner);
3062
0
            xCheckBestMode(tempCS, bestCS, partitioner, encTestMode);
3063
3064
0
            tempCS->initStructData(encTestMode.qp);
3065
0
          }
3066
3067
0
          if (m_pcEncCfg->m_useFastDecisionForMerge && !bestIsSkip)
3068
0
          {
3069
0
            if (bestCS->getCU(partitioner.chType, partitioner.treeType) == NULL)
3070
0
              bestIsSkip = 0;
3071
0
            else
3072
0
              bestIsSkip = bestCS->getCU(partitioner.chType, partitioner.treeType)->rootCbf == 0;
3073
0
          }
3074
0
        }
3075
0
      }
3076
0
    }
3077
0
  }
3078
0
}
3079
3080
void EncCu::xCheckRDCostIBCMode(CodingStructure*& tempCS, CodingStructure*& bestCS, Partitioner& partitioner,
3081
  const EncTestMode& encTestMode)
3082
24.0k
{
3083
24.0k
  if (tempCS->area.lwidth() == 128 || tempCS->area.lheight() == 128)   // disable IBC mode larger than 64x64
3084
0
  {
3085
0
    return;
3086
0
  }
3087
24.0k
  if ((m_pcEncCfg->m_IBCFastMethod > 1) && !bestCS->slice->isIntra() && (bestCS->cus.size() != 0))
3088
0
  {
3089
0
    if (bestCS->getCU(partitioner.chType, partitioner.treeType)->skip)
3090
0
    {
3091
0
      return;
3092
0
    }
3093
0
  }
3094
3095
24.0k
  tempCS->initStructData(encTestMode.qp);
3096
3097
24.0k
  CodingUnit& cu = tempCS->addCU(CS::getArea(*tempCS, tempCS->area, partitioner.chType, partitioner.treeType), partitioner.chType);
3098
3099
24.0k
  partitioner.setCUData(cu);
3100
24.0k
  cu.slice = tempCS->slice;
3101
24.0k
  cu.tileIdx = m_tileIdx;
3102
24.0k
  cu.skip = false;
3103
24.0k
  cu.predMode = MODE_IBC;
3104
24.0k
  cu.chromaQpAdj = m_cuChromaQpOffsetIdxPlus1;
3105
24.0k
  cu.qp = encTestMode.qp;
3106
24.0k
  cu.initPuData();
3107
24.0k
  cu.imv = IMV_OFF;
3108
24.0k
  cu.sbtInfo = 0;
3109
24.0k
  cu.mmvdSkip = false;
3110
24.0k
  cu.mmvdMergeFlag = false;
3111
3112
24.0k
  cu.intraDir[0] = DC_IDX; // set intra pred for ibc block
3113
24.0k
  cu.intraDir[1] = PLANAR_IDX; // set intra pred for ibc block
3114
3115
24.0k
  cu.interDir = 1; // use list 0 for IBC mode
3116
24.0k
  cu.refIdx[REF_PIC_LIST_0] = MAX_NUM_REF; // last idx in the list
3117
24.0k
  bool bValid = m_cInterSearch.predIBCSearch(cu, partitioner);
3118
3119
24.0k
  if (bValid)
3120
21.4k
  {
3121
21.4k
    CU::spanMotionInfo(cu);
3122
21.4k
    const bool chroma = !CU::isSepTree(cu);
3123
    //  MC
3124
21.4k
    cu.mcControl = chroma ? 0 : 2;
3125
21.4k
    m_cInterSearch.motionCompensationIBC(cu, tempCS->getPredBuf());
3126
3127
21.4k
    m_cInterSearch.encodeResAndCalcRdInterCU(*tempCS, partitioner, false);
3128
21.4k
    cu.mcControl = 0;
3129
3130
21.4k
    xEncodeDontSplit(*tempCS, partitioner);
3131
21.4k
    xCheckDQP(*tempCS, partitioner);
3132
21.4k
    xCheckBestMode(tempCS, bestCS, partitioner, encTestMode);
3133
21.4k
  } // bValid
3134
2.57k
  else
3135
2.57k
  {
3136
2.57k
    tempCS->dist = 0;
3137
2.57k
    tempCS->fracBits = 0;
3138
2.57k
    tempCS->cost = MAX_DOUBLE;
3139
2.57k
    tempCS->costDbOffset = 0;
3140
2.57k
  }
3141
24.0k
}
3142
3143
void EncCu::xCheckRDCostInter( CodingStructure *&tempCS, CodingStructure *&bestCS, Partitioner &partitioner, const EncTestMode& encTestMode )
3144
0
{
3145
0
  PROFILER_SCOPE_AND_STAGE_EXT( 1, _TPROF, P_INTER_MVD, tempCS, partitioner.chType );
3146
0
  tempCS->initStructData( encTestMode.qp );
3147
3148
0
  m_cInterSearch.setAffineModeSelected( false );
3149
3150
0
  m_cInterSearch.resetBufferedUniMotions();
3151
3152
0
  int bcwLoopNum = BCW_NUM;
3153
3154
0
  if( tempCS->area.Y().area() < BCW_SIZE_CONSTRAINT || !tempCS->slice->isInterB() || !tempCS->sps->BCW )
3155
0
  {
3156
0
    bcwLoopNum = 1;
3157
0
  }
3158
  
3159
0
  double curBestCost = bestCS->cost;
3160
0
  double equBcwCost = MAX_DOUBLE;
3161
3162
0
  for( int bcwLoopIdx = 0; bcwLoopIdx < bcwLoopNum; bcwLoopIdx++ )
3163
0
  {
3164
0
    if( m_pcEncCfg->m_BCW == 2 )
3165
0
    {
3166
0
      bool isBestInter   = m_modeCtrl.getBlkInfo( bestCS->area ).isInter;
3167
0
      uint8_t bestBcwIdx = m_modeCtrl.getBlkInfo( bestCS->area).BcwIdx;
3168
3169
0
      if( isBestInter && g_BcwSearchOrder[bcwLoopIdx] != BCW_DEFAULT && g_BcwSearchOrder[bcwLoopIdx] != bestBcwIdx )
3170
0
      {
3171
0
        continue;
3172
0
      }
3173
0
    }
3174
    
3175
0
    if( !tempCS->slice->checkLDC )
3176
0
    {
3177
0
      if( bcwLoopIdx != 0 && bcwLoopIdx != 3 && bcwLoopIdx != 4 )
3178
0
      {
3179
0
        continue;
3180
0
      }
3181
0
    }
3182
  
3183
0
    CodingUnit &cu      = tempCS->addCU( tempCS->area, partitioner.chType );
3184
3185
0
    partitioner.setCUData( cu );
3186
0
    cu.slice            = tempCS->slice;
3187
0
    cu.tileIdx          = m_tileIdx;
3188
0
    cu.skip             = false;
3189
0
    cu.mmvdSkip         = false;
3190
0
    cu.predMode         = MODE_INTER;
3191
0
    cu.chromaQpAdj      = m_cuChromaQpOffsetIdxPlus1;
3192
0
    cu.qp               = encTestMode.qp;
3193
0
    cu.initPuData();
3194
3195
0
    cu.BcwIdx = g_BcwSearchOrder[bcwLoopIdx];
3196
0
    uint8_t bcwIdx = cu.BcwIdx;
3197
0
    bool testBcw = (bcwIdx != BCW_DEFAULT);
3198
3199
0
    bool StopInterRes = (m_pcEncCfg->m_FastInferMerge >> 3) & 1;
3200
0
    StopInterRes &= bestCS->slice->TLayer > (m_pcEncCfg->m_maxTLayer - (m_pcEncCfg->m_FastInferMerge & 7));
3201
0
    double bestCostInter = StopInterRes ? m_mergeBestSATDCost : MAX_DOUBLE;
3202
3203
0
    bool stopTest = m_cInterSearch.predInterSearch(cu, partitioner, bestCostInter);
3204
3205
0
    if (StopInterRes && (bestCostInter != m_mergeBestSATDCost))
3206
0
    {
3207
0
      int L = (cu.slice->TLayer <= 2) ? 0 : (cu.slice->TLayer - 2);
3208
0
      if ((bestCostInter > MRG_FAST_RATIOMYV[L] * m_mergeBestSATDCost))
3209
0
      {
3210
0
        stopTest = true;
3211
0
      }
3212
0
    }
3213
3214
0
    if( !stopTest )
3215
0
    {
3216
0
      bcwIdx   = CU::getValidBcwIdx(cu);
3217
0
      stopTest = testBcw && bcwIdx == BCW_DEFAULT;
3218
0
    }
3219
    
3220
0
    if( stopTest )
3221
0
    {
3222
0
      tempCS->initStructData(encTestMode.qp);
3223
0
      continue;
3224
0
    }
3225
3226
0
    CHECK(!(testBcw || (!testBcw && bcwIdx == BCW_DEFAULT)), " !( bTestBcw || (!bTestBcw && bcwIdx == BCW_DEFAULT ) )");
3227
        
3228
0
    xEncodeInterResidual(tempCS, bestCS, partitioner, encTestMode, 0, 0, &equBcwCost);
3229
    
3230
0
    if( bcwIdx == BCW_DEFAULT )
3231
0
    {
3232
0
      m_cInterSearch.setAffineModeSelected( bestCS->cus.front()->affine && !bestCS->cus.front()->mergeFlag );
3233
0
    }
3234
3235
0
    tempCS->initStructData(encTestMode.qp);
3236
  
3237
0
    double skipTH = MAX_DOUBLE;
3238
0
    skipTH = (m_pcEncCfg->m_BCW == 2 ? 1.05 : MAX_DOUBLE);
3239
0
    if( equBcwCost > curBestCost * skipTH )
3240
0
    {
3241
0
      break;
3242
0
    }
3243
3244
0
    if( m_pcEncCfg->m_BCW == 2 )
3245
0
    {
3246
0
      if( ( cu.interDir != 3 && testBcw == 0 && ! m_pcEncCfg->m_picReordering )
3247
0
         || ( g_BcwSearchOrder[bcwLoopIdx] == BCW_DEFAULT && xIsBcwSkip( cu ) ) )
3248
0
      {
3249
0
        break;
3250
0
      }
3251
0
    }
3252
0
  }
3253
0
  STAT_COUNT_CU_MODES( partitioner.chType == CH_L, g_cuCounters1D[CU_MODES_TESTED][0][!tempCS->slice->isIntra() + tempCS->slice->depth] );
3254
0
  STAT_COUNT_CU_MODES( partitioner.chType == CH_L && !tempCS->slice->isIntra(), g_cuCounters2D[CU_MODES_TESTED][Log2( tempCS->area.lheight() )][Log2( tempCS->area.lwidth() )] );
3255
0
}
3256
3257
void EncCu::xCheckRDCostInterIMV(CodingStructure *&tempCS, CodingStructure *&bestCS, Partitioner &partitioner, const EncTestMode& encTestMode)
3258
0
{
3259
0
  PROFILER_SCOPE_AND_STAGE_EXT( 1, _TPROF, P_INTER_MVD_IMV, tempCS, partitioner.chType );
3260
0
  bool Test_AMVR = m_pcEncCfg->m_AMVRspeed ? true: false;
3261
0
  if (m_pcEncCfg->m_AMVRspeed > 2 && m_pcEncCfg->m_AMVRspeed < 5 && !bestCS->cus.empty() && bestCS->getCU(partitioner.chType, partitioner.treeType)->skip)
3262
0
  {
3263
0
    Test_AMVR = false;
3264
0
  }
3265
0
  else if (m_pcEncCfg->m_AMVRspeed > 4 && !bestCS->cus.empty() && bestCS->getCU(partitioner.chType, partitioner.treeType)->mergeFlag && !bestCS->getCU(partitioner.chType, partitioner.treeType)->ciip)
3266
0
  {
3267
0
    Test_AMVR = false;
3268
0
  }
3269
0
  bool Do_Limit = !bestCS->cus.empty() && (m_pcEncCfg->m_AMVRspeed == 4 || m_pcEncCfg->m_AMVRspeed == 6) ? true : false;
3270
0
  bool Do_OnceRes = !bestCS->cus.empty() && (m_pcEncCfg->m_AMVRspeed == 7) ? true : false;
3271
3272
0
  if( Test_AMVR )
3273
0
  {
3274
0
    double Fpel_cost    = m_pcEncCfg->m_AMVRspeed == 1 ? MAX_DOUBLE*0.5 : MAX_DOUBLE;
3275
0
    double costCurStart = m_pcEncCfg->m_AMVRspeed == 1 ? m_modeCtrl.comprCUCtx->bestCostNoImv : bestCS->cost;
3276
0
    double costCur      = MAX_DOUBLE;
3277
0
    double bestCostIMV  = MAX_DOUBLE;
3278
3279
0
    if (Do_OnceRes)
3280
0
    {
3281
0
      costCurStart = xCalcDistortion(bestCS, partitioner.chType, bestCS->sps->bitDepths[CH_L], 0);
3282
0
      Fpel_cost = costCurStart;
3283
0
      tempCS->initSubStructure(*m_pTempCS2, partitioner.chType, partitioner.currArea(), false);
3284
0
    }
3285
3286
0
    CodingStructure *tempCSbest = m_pTempCS2;
3287
3288
0
    m_cInterSearch.setAffineModeSelected( false );
3289
3290
0
    m_cInterSearch.resetBufferedUniMotions();
3291
3292
0
    int bcwLoopNum = (tempCS->slice->isInterB() ? BCW_NUM : 1);
3293
0
    bcwLoopNum = (tempCS->sps->BCW ? bcwLoopNum : 1);
3294
3295
0
    if( tempCS->area.lwidth() * tempCS->area.lheight() < BCW_SIZE_CONSTRAINT )
3296
0
    {
3297
0
      bcwLoopNum = 1;
3298
0
    }
3299
3300
0
    for (int i = 1; i <= IMV_HPEL; i++)
3301
0
    {
3302
0
      double curBestCost = bestCS->cost;
3303
0
      double equBcwCost  = MAX_DOUBLE;
3304
3305
0
      for( int bcwLoopIdx = 0; bcwLoopIdx < bcwLoopNum; bcwLoopIdx++ )
3306
0
      {
3307
0
        if( m_pcEncCfg->m_BCW == 2 )
3308
0
        {
3309
0
          bool isBestInter   = m_modeCtrl.getBlkInfo( bestCS->area ).isInter;
3310
0
          uint8_t bestBcwIdx = m_modeCtrl.getBlkInfo( bestCS->area).BcwIdx;
3311
3312
0
          if( isBestInter && g_BcwSearchOrder[bcwLoopIdx] != BCW_DEFAULT && g_BcwSearchOrder[bcwLoopIdx] != bestBcwIdx )
3313
0
          {
3314
0
            continue;
3315
0
          }
3316
          
3317
0
          if( tempCS->slice->checkLDC && g_BcwSearchOrder[bcwLoopIdx] != BCW_DEFAULT
3318
0
            && (m_bestBcwIdx[0] >= 0 && g_BcwSearchOrder[bcwLoopIdx] != m_bestBcwIdx[0])
3319
0
            && (m_bestBcwIdx[1] >= 0 && g_BcwSearchOrder[bcwLoopIdx] != m_bestBcwIdx[1]))
3320
0
          {
3321
0
            continue;
3322
0
          }
3323
0
        }
3324
3325
0
        if( !tempCS->slice->checkLDC )
3326
0
        {
3327
0
          if( bcwLoopIdx != 0 && bcwLoopIdx != 3 && bcwLoopIdx != 4 )
3328
0
          {
3329
0
            continue;
3330
0
          }
3331
0
        }
3332
3333
0
        bool testBcw;
3334
0
        uint8_t bcwIdx;
3335
0
        bool isEqualUni = false;
3336
3337
0
        if (i > IMV_FPEL)
3338
0
        {
3339
0
          bool nextimv = false;
3340
0
          double stopCost = i == IMV_HPEL ? 1.25 : 1.06;
3341
0
          if (Fpel_cost > stopCost * costCurStart)
3342
0
          {
3343
0
            nextimv = true;
3344
0
          }
3345
0
          if ( m_pcEncCfg->m_AMVRspeed == 1 )
3346
0
          {
3347
0
            costCurStart = bestCS->cost;
3348
0
          }
3349
0
          if (nextimv)
3350
0
          {
3351
0
            continue;
3352
0
          }
3353
0
        }
3354
3355
0
        bool Do_Search = Do_OnceRes ? false : true;
3356
3357
0
        if (Do_Limit)
3358
0
        {
3359
0
          Do_Search = i == IMV_FPEL ? true : false;
3360
3361
0
          if (i == IMV_HPEL)
3362
0
          {
3363
0
            if (bestCS->slice->TLayer > 3)
3364
0
            {
3365
0
              continue;
3366
0
            }
3367
0
            if (bestCS->getCU(partitioner.chType, partitioner.treeType)->imv != 0)
3368
0
            {
3369
0
              Do_Search = true; //do_est
3370
0
            }
3371
0
          }
3372
0
          if (bestCS->getCU(partitioner.chType, partitioner.treeType)->mmvdMergeFlag || bestCS->getCU(partitioner.chType, partitioner.treeType)->geo)
3373
0
          {
3374
0
            Do_Search = true;
3375
0
          }
3376
0
        }
3377
0
        tempCS->initStructData(encTestMode.qp);
3378
3379
0
        if (!Do_Search)
3380
0
        {
3381
0
          tempCS->copyStructure(*bestCS, partitioner.chType, TREE_D);
3382
0
        }
3383
0
        tempCS->dist = 0;
3384
0
        tempCS->fracBits = 0;
3385
0
        tempCS->cost = MAX_DOUBLE;
3386
0
        CodingUnit &cu = (Do_Search) ? tempCS->addCU(tempCS->area, partitioner.chType) : *tempCS->getCU(partitioner.chType, partitioner.treeType);
3387
0
        if (Do_Search)
3388
0
        {
3389
0
          partitioner.setCUData(cu);
3390
0
          cu.slice = tempCS->slice;
3391
0
          cu.tileIdx = m_tileIdx;
3392
0
          cu.skip = false;
3393
0
          cu.mmvdSkip = false;
3394
0
          cu.predMode = MODE_INTER;
3395
0
          cu.chromaQpAdj = m_cuChromaQpOffsetIdxPlus1;
3396
0
          cu.qp = encTestMode.qp;
3397
3398
0
          cu.initPuData();
3399
3400
0
          cu.imv = i;
3401
3402
0
          cu.BcwIdx = g_BcwSearchOrder[bcwLoopIdx];
3403
0
          bcwIdx    = cu.BcwIdx;
3404
0
          testBcw   = (bcwIdx != BCW_DEFAULT);
3405
3406
0
          cu.interDir = 10;
3407
          
3408
0
          double bestCostInter = MAX_DOUBLE;
3409
0
          m_cInterSearch.predInterSearch(cu, partitioner, bestCostInter);
3410
          
3411
0
          if ( cu.interDir <= 3 )
3412
0
          {
3413
0
            bcwIdx = CU::getValidBcwIdx(cu);
3414
0
          }
3415
0
          else
3416
0
          {
3417
0
            continue;
3418
0
          }
3419
          
3420
0
          if( testBcw && bcwIdx == BCW_DEFAULT ) // Enabled Bcw but the search results is uni.
3421
0
          {
3422
0
            continue;
3423
0
          }
3424
0
          CHECK(!(testBcw || (!testBcw && bcwIdx == BCW_DEFAULT)), " !( bTestBcw || (!bTestBcw && bcwIdx == BCW_DEFAULT ) )");
3425
3426
0
          if( m_pcEncCfg->m_BCW == 2 )
3427
0
          {
3428
0
            if( cu.interDir != 3 && testBcw == 0 )
3429
0
            {
3430
0
              isEqualUni = true;
3431
0
            }
3432
0
          }
3433
3434
0
          if (!CU::hasSubCUNonZeroMVd(cu))
3435
0
          {
3436
0
            continue;
3437
0
          }
3438
0
        }
3439
0
        else
3440
0
        {
3441
0
          cu.smvdMode = 0;
3442
0
          cu.affine = false;
3443
0
          cu.imv = i ;
3444
0
          CU::resetMVDandMV2Int(cu);
3445
0
          if (!CU::hasSubCUNonZeroMVd(cu))
3446
0
          {
3447
0
            continue;
3448
0
          }
3449
3450
0
          cu.BcwIdx = g_BcwSearchOrder[bcwLoopIdx];
3451
3452
0
          cu.mvRefine = true;
3453
0
          m_cInterSearch.motionCompensation(cu, tempCS->getPredBuf() );
3454
0
          cu.mvRefine = false;
3455
0
        }
3456
3457
0
        if( Do_OnceRes )
3458
0
        {
3459
0
          costCur = xCalcDistortion(tempCS, partitioner.chType, tempCS->sps->bitDepths[CH_L], cu.imv );
3460
0
          if (costCur < bestCostIMV)
3461
0
          {
3462
0
            bestCostIMV = costCur;
3463
0
            tempCSbest->getPredBuf().copyFrom(tempCS->getPredBuf());
3464
0
            tempCSbest->clearCUs();
3465
0
            tempCSbest->clearTUs();
3466
0
            tempCSbest->copyStructure(*tempCS, partitioner.chType, TREE_D);
3467
0
          }
3468
0
          if (i > IMV_FPEL)
3469
0
          {
3470
0
            costCurStart = costCurStart > costCur ? costCur : costCurStart;
3471
0
          }
3472
0
        }
3473
0
        else
3474
0
        {
3475
0
          xEncodeInterResidual(tempCS, bestCS, partitioner, encTestMode, 0, 0, &equBcwCost);
3476
0
          costCur = tempCS->cost;
3477
3478
0
          if (i > IMV_FPEL)
3479
0
          {
3480
0
            costCurStart = bestCS->cost;
3481
0
          }
3482
0
        }
3483
3484
0
        if (i == IMV_FPEL)
3485
0
        {
3486
0
           Fpel_cost = costCur;
3487
0
        }
3488
3489
0
        double skipTH = MAX_DOUBLE;
3490
0
        skipTH = (m_pcEncCfg->m_BCW == 2 ? 1.05 : MAX_DOUBLE);
3491
0
        if( equBcwCost > curBestCost * skipTH )
3492
0
        {
3493
0
          break;
3494
0
        }
3495
3496
0
        if( m_pcEncCfg->m_BCW == 2 )
3497
0
        {
3498
0
          if( isEqualUni == true && ! m_pcEncCfg->m_picReordering )
3499
0
          {
3500
0
            break;
3501
0
          }
3502
0
          if( g_BcwSearchOrder[bcwLoopIdx] == BCW_DEFAULT && xIsBcwSkip( cu ) )
3503
0
          {
3504
0
            break;
3505
0
          }
3506
0
        }
3507
0
      }
3508
0
    }
3509
3510
0
    if (Do_OnceRes && (bestCostIMV != MAX_DOUBLE))
3511
0
    {
3512
0
      CodingStructure* CSCandBest = tempCSbest;
3513
0
      tempCS->initStructData(bestCS->currQP[partitioner.chType]);
3514
0
      tempCS->copyStructure(*CSCandBest, partitioner.chType, TREE_D);
3515
0
      tempCS->getPredBuf().copyFrom(tempCSbest->getPredBuf());
3516
0
      tempCS->dist = 0;
3517
0
      tempCS->fracBits = 0;
3518
0
      tempCS->cost = MAX_DOUBLE;
3519
3520
0
      xEncodeInterResidual(tempCS, bestCS, partitioner, encTestMode, 0, 0, NULL);
3521
0
    }
3522
3523
0
    tempCS->initStructData(encTestMode.qp);
3524
0
  }
3525
0
  STAT_COUNT_CU_MODES( partitioner.chType == CH_L, g_cuCounters1D[CU_MODES_TESTED][0][!tempCS->slice->isIntra() + tempCS->slice->depth] );
3526
0
  STAT_COUNT_CU_MODES( partitioner.chType == CH_L && !tempCS->slice->isIntra(), g_cuCounters2D[CU_MODES_TESTED][Log2( tempCS->area.lheight() )][Log2( tempCS->area.lwidth() )] );
3527
0
}
3528
3529
void EncCu::xCalDebCost( CodingStructure &cs, Partitioner &partitioner )
3530
100k
{
3531
100k
  PROFILER_SCOPE_AND_STAGE_EXT( 1, _TPROF, P_DEBLOCK_FILTER, &cs, partitioner.chType );
3532
100k
  if ( cs.slice->deblockingFilterDisable )
3533
0
  {
3534
0
    return;
3535
0
  }
3536
3537
100k
  const ChromaFormat format = cs.area.chromaFormat;
3538
100k
  CodingUnit*            cu = cs.getCU(partitioner.chType, partitioner.treeType);
3539
100k
  const Position    lumaPos = cu->Y().valid() ? cu->Y().pos() : recalcPosition( format, cu->chType, CH_L, cu->blocks[cu->chType].pos() );
3540
100k
  bool    topEdgeAvai = lumaPos.y > 0 && ((lumaPos.y % 4) == 0);
3541
100k
  bool   leftEdgeAvai = lumaPos.x > 0 && ((lumaPos.x % 4) == 0);
3542
3543
100k
  if( ! ( topEdgeAvai || leftEdgeAvai ))
3544
21.0k
  {
3545
21.0k
    return;
3546
21.0k
  }
3547
3548
79.4k
  ComponentID compStr = ( CU::isSepTree(*cu) && !isLuma( partitioner.chType ) ) ? COMP_Cb : COMP_Y;
3549
79.4k
  ComponentID compEnd = (( CU::isSepTree(*cu) && isLuma( partitioner.chType )) || cu->chromaFormat == VVENC_CHROMA_400 ) ? COMP_Y : COMP_Cr;
3550
79.4k
  const UnitArea currCsArea = clipArea( CS::getArea( cs, cs.area, partitioner.chType, partitioner.treeType ), *cs.picture );
3551
3552
79.4k
  PelStorage&  picDbBuf = m_dbBuffer; //th we could reduce the buffer size and do some relocate
3553
3554
  //deblock neighbour pixels
3555
79.4k
  const Size     lumaSize = cu->Y().valid() ? cu->Y().size() : recalcSize( format, cu->chType, CH_L, cu->blocks[cu->chType].size() );
3556
3557
79.4k
  int verOffset = lumaPos.y > 7 ? 8 : 4;
3558
79.4k
  int horOffset = lumaPos.x > 7 ? 8 : 4;
3559
3560
79.4k
  LoopFilter::calcFilterStrengths( *cu, true );
3561
3562
79.4k
  if( m_EDO == 2 && CS::isDualITree( cs ) && isLuma( partitioner.chType ) )
3563
43.8k
  {
3564
43.8k
    m_cLoopFilter.getMaxFilterLength( *cu, verOffset, horOffset );
3565
3566
43.8k
    if( 0== (verOffset + horOffset) )
3567
0
    {
3568
0
      return;
3569
0
    }
3570
3571
43.8k
    topEdgeAvai  &= verOffset != 0;
3572
43.8k
    leftEdgeAvai &= horOffset != 0;
3573
43.8k
  }
3574
3575
79.4k
  const UnitArea  areaTop  = UnitArea( format, Area( lumaPos.x,             lumaPos.y - verOffset, lumaSize.width, verOffset       ) );
3576
79.4k
  const UnitArea  areaLeft = UnitArea( format, Area( lumaPos.x - horOffset, lumaPos.y,             horOffset,      lumaSize.height ) );
3577
3578
194k
  for ( int compIdx = compStr; compIdx <= compEnd; compIdx++ )
3579
115k
  {
3580
115k
    ComponentID compId = (ComponentID)compIdx;
3581
3582
    //Copy current CU's reco to Deblock Pic Buffer
3583
115k
    const CompArea&  compArea = currCsArea.block( compId );
3584
115k
    CompArea         locArea  = compArea;
3585
115k
    locArea.x -= cu->blocks[compIdx].x;
3586
115k
    locArea.y -= cu->blocks[compIdx].y;
3587
115k
    PelBuf dbReco = picDbBuf.getBuf( locArea );
3588
115k
      PelBuf reco = cs.getRecoBuf( compId );
3589
115k
      dbReco.copyFrom( reco );
3590
    //left neighbour
3591
115k
    if ( leftEdgeAvai )
3592
80.1k
    {
3593
80.1k
      const CompArea&  compArea = areaLeft.block(compId);
3594
80.1k
      CompArea         locArea = compArea;
3595
80.1k
      locArea.x -= cu->blocks[compIdx].x;
3596
80.1k
      locArea.y -= cu->blocks[compIdx].y;
3597
80.1k
      PelBuf dbReco = picDbBuf.getBuf( locArea );
3598
80.1k
        dbReco.copyFrom( cs.picture->getRecoBuf( compArea ) );
3599
80.1k
      }
3600
    //top neighbour
3601
115k
    if ( topEdgeAvai )
3602
83.2k
    {
3603
83.2k
      const CompArea&  compArea = areaTop.block( compId );
3604
83.2k
      CompArea         locArea = compArea;
3605
83.2k
      locArea.x -= cu->blocks[compIdx].x;
3606
83.2k
      locArea.y -= cu->blocks[compIdx].y;
3607
83.2k
      PelBuf dbReco = picDbBuf.getBuf( locArea );
3608
83.2k
        dbReco.copyFrom( cs.picture->getRecoBuf( compArea ) );
3609
83.2k
      }
3610
115k
    }
3611
3612
18.4E
  ChannelType dbChType = CU::isSepTree(*cu) ? partitioner.chType : MAX_NUM_CH;
3613
3614
79.4k
  CHECK( CU::isSepTree(*cu) && !cu->Y().valid() && partitioner.chType == CH_L, "xxx" );
3615
3616
79.4k
  if( cu->Y() .valid() ) m_cLoopFilter.setOrigin( CH_L, cu->lumaPos() );
3617
79.4k
  if( cu->chromaFormat != VVENC_CHROMA_400 && cu->Cb().valid() ) m_cLoopFilter.setOrigin( CH_C, cu->chromaPos() );
3618
3619
  //deblock
3620
79.4k
  if( leftEdgeAvai )
3621
58.3k
  {
3622
58.3k
    m_cLoopFilter.loopFilterCu( *cu, dbChType, EDGE_VER, m_dbBuffer );
3623
58.3k
  }
3624
3625
79.4k
  if( topEdgeAvai )
3626
59.7k
  {
3627
59.7k
    m_cLoopFilter.loopFilterCu( *cu, dbChType, EDGE_HOR, m_dbBuffer );
3628
59.7k
  }
3629
3630
  //calculate difference between DB_before_SSE and DB_after_SSE for neighbouring CUs
3631
79.4k
  Distortion distBeforeDb = 0, distAfterDb = 0, distCur = 0;
3632
194k
  for (int compIdx = compStr; compIdx <= compEnd; compIdx++)
3633
115k
  {
3634
115k
    ComponentID compId = (ComponentID)compIdx;
3635
115k
    {
3636
115k
      CompArea compArea = currCsArea.block( compId );
3637
115k
      CompArea         locArea  = compArea;
3638
115k
      locArea.x -= cu->blocks[compIdx].x;
3639
115k
      locArea.y -= cu->blocks[compIdx].y;
3640
115k
      CPelBuf reco      = picDbBuf.getBuf( locArea );
3641
115k
      CPelBuf org       = cs.getOrgBuf( compId );
3642
115k
      distCur += xGetDistortionDb( cs, org, reco, compArea, false );
3643
115k
    }
3644
3645
115k
    if ( leftEdgeAvai )
3646
80.1k
    {
3647
80.1k
      const CompArea&  compArea = areaLeft.block( compId );
3648
80.1k
      CompArea         locArea  = compArea;
3649
80.1k
      locArea.x -= cu->blocks[compIdx].x;
3650
80.1k
      locArea.y -= cu->blocks[compIdx].y;
3651
80.1k
      CPelBuf org    = cs.picture->getOrigBuf( compArea );
3652
80.1k
      if ( cs.picture->getFilteredOrigBuffer().valid() )
3653
0
      {
3654
0
        org = cs.picture->getFiltOrigBuf( compArea );
3655
0
      }
3656
80.1k
      CPelBuf reco   = cs.picture->getRecoBuf( compArea );
3657
80.1k
      CPelBuf recoDb = picDbBuf.getBuf( locArea );
3658
80.1k
      distBeforeDb  += xGetDistortionDb( cs, org, reco,   compArea, true );
3659
80.1k
      distAfterDb   += xGetDistortionDb( cs, org, recoDb, compArea, false  );
3660
80.1k
    }
3661
3662
115k
    if ( topEdgeAvai )
3663
83.2k
    {
3664
83.2k
      const CompArea&  compArea = areaTop.block( compId );
3665
83.2k
      CompArea         locArea  = compArea;
3666
83.2k
      locArea.x -= cu->blocks[compIdx].x;
3667
83.2k
      locArea.y -= cu->blocks[compIdx].y;
3668
83.2k
      CPelBuf org    = cs.picture->getOrigBuf( compArea );
3669
83.2k
      if ( cs.picture->getFilteredOrigBuffer().valid() )
3670
0
      {
3671
0
        org = cs.picture->getFiltOrigBuf( compArea );
3672
0
      }
3673
83.2k
      CPelBuf reco   = cs.picture->getRecoBuf( compArea );
3674
83.2k
      CPelBuf recoDb = picDbBuf.getBuf( locArea );
3675
83.2k
      distBeforeDb  += xGetDistortionDb( cs, org, reco,   compArea, true );
3676
83.2k
      distAfterDb   += xGetDistortionDb( cs, org, recoDb, compArea, false  );
3677
83.2k
    }
3678
115k
  }
3679
3680
  //updated cost
3681
79.4k
  int64_t distTmp = distCur - cs.dist + distAfterDb - distBeforeDb;
3682
79.4k
  cs.costDbOffset = distTmp < 0 ? -m_cRdCost.calcRdCost( 0, -distTmp ) : m_cRdCost.calcRdCost( 0, distTmp );
3683
79.4k
}
3684
3685
Distortion EncCu::xGetDistortionDb(CodingStructure &cs, CPelBuf& org, CPelBuf& reco, const CompArea& compArea, bool beforeDb)
3686
441k
{
3687
441k
  Distortion dist;
3688
441k
  const ComponentID compID = compArea.compID;
3689
3690
441k
  dist = m_cRdCost.getDistPart(org, reco, cs.sps->bitDepths[toChannelType(compID)], compID, DF_SSE);
3691
441k
  return dist;
3692
441k
}
3693
3694
bool checkValidMvs( const CodingUnit& cu)
3695
0
{
3696
  // clang-format off
3697
0
  const int affineShiftTab[3] =
3698
0
  {
3699
0
    MV_PRECISION_INTERNAL - MV_PRECISION_QUARTER,
3700
0
    MV_PRECISION_INTERNAL - MV_PRECISION_SIXTEENTH,
3701
0
    MV_PRECISION_INTERNAL - MV_PRECISION_INT
3702
0
  };
3703
3704
0
  const int normalShiftTab[NUM_IMV_MODES] =
3705
0
  {
3706
0
    MV_PRECISION_INTERNAL - MV_PRECISION_QUARTER,
3707
0
    MV_PRECISION_INTERNAL - MV_PRECISION_INT,
3708
0
    MV_PRECISION_INTERNAL - MV_PRECISION_4PEL,
3709
0
    MV_PRECISION_INTERNAL - MV_PRECISION_HALF,
3710
0
  };
3711
  // clang-format on
3712
3713
0
  int mvShift;
3714
3715
0
  for (int refList = 0; refList < NUM_REF_PIC_LIST_01; refList++)
3716
0
  {
3717
0
    if (cu.refIdx[refList] >= 0)
3718
0
    {
3719
0
      if (!cu.affine)
3720
0
      {
3721
0
        mvShift = normalShiftTab[cu.imv];
3722
0
        Mv signaledmvd(cu.mvd[refList][0].hor >> mvShift, cu.mvd[refList][0].ver >> mvShift);
3723
0
        if (!((signaledmvd.hor >= MVD_MIN) && (signaledmvd.hor <= MVD_MAX)) || !((signaledmvd.ver >= MVD_MIN) && (signaledmvd.ver <= MVD_MAX)))
3724
0
          return false;
3725
0
      }
3726
0
      else
3727
0
      {
3728
0
        for (int ctrlP = 1 + (cu.affineType == AFFINEMODEL_6PARAM); ctrlP >= 0; ctrlP--)
3729
0
        {
3730
0
          mvShift = affineShiftTab[cu.imv];
3731
0
          Mv signaledmvd(cu.mvd[refList][ctrlP].hor >> mvShift, cu.mvd[refList][ctrlP].ver >> mvShift);
3732
0
          if (!((signaledmvd.hor >= MVD_MIN) && (signaledmvd.hor <= MVD_MAX)) || !((signaledmvd.ver >= MVD_MIN) && (signaledmvd.ver <= MVD_MAX)))
3733
0
            return false;;
3734
0
        }
3735
0
      }
3736
0
    }
3737
0
  }
3738
  // avoid MV exceeding 18-bit dynamic range
3739
0
  const int maxMv = 1 << 17;
3740
0
  if (!cu.affine && !cu.mergeFlag)
3741
0
  {
3742
0
    if(    ( cu.refIdx[ 0 ] >= 0 && ( cu.mv[ 0 ][ 0 ].getAbsHor() >= maxMv || cu.mv[ 0 ][ 0 ].getAbsVer() >= maxMv ) )
3743
0
        || ( cu.refIdx[ 1 ] >= 0 && ( cu.mv[ 1 ][ 0 ].getAbsHor() >= maxMv || cu.mv[ 1 ][ 0 ].getAbsVer() >= maxMv ) ) )
3744
0
    {
3745
0
      return false;
3746
0
    }
3747
0
  }
3748
0
  if( cu.affine && !cu.mergeFlag )
3749
0
  {
3750
0
    for( int refList = 0; refList < NUM_REF_PIC_LIST_01; refList++ )
3751
0
    {
3752
0
      if( cu.refIdx[ refList ] >= 0 )
3753
0
      {
3754
0
        for( int ctrlP = 1 + ( cu.affineType == AFFINEMODEL_6PARAM ); ctrlP >= 0; ctrlP-- )
3755
0
        {
3756
0
          if( cu.mv[ refList ][ ctrlP ].getAbsHor() >= maxMv || cu.mv[ refList ][ ctrlP ].getAbsVer() >= maxMv )
3757
0
          {
3758
0
            return false;
3759
0
          }
3760
0
        }
3761
0
      }
3762
0
    }
3763
0
  }
3764
0
  return true;
3765
0
}
3766
3767
3768
void EncCu::xEncodeInterResidual( CodingStructure *&tempCS, CodingStructure *&bestCS, Partitioner &partitioner, const EncTestMode& encTestMode, int residualPass, bool* bestHasNonResi, double* equBcwCost )
3769
0
{
3770
0
  if( residualPass == 1 && encTestMode.lossless )
3771
0
  {
3772
0
    return;
3773
0
  }
3774
3775
0
  CodingUnit*            cu        = tempCS->getCU( partitioner.chType, partitioner.treeType );
3776
0
  double   bestCostInternal        = MAX_DOUBLE;
3777
3778
0
  if( !checkValidMvs( *cu ) )
3779
0
    return;
3780
3781
0
  double  currBestCost = MAX_DOUBLE;
3782
3783
  // For SBT
3784
0
  double     bestCost          = bestCS->cost;
3785
0
  double     bestCostBegin     = bestCS->cost;
3786
0
  const CodingUnit* prevBestCU = bestCS->getCU( partitioner.chType, partitioner.treeType );
3787
0
  uint8_t    prevBestSbt       = ( prevBestCU == nullptr ) ? 0 : prevBestCU->sbtInfo;
3788
0
  Distortion sbtOffDist        = 0;
3789
0
  bool       sbtOffRootCbf     = 0;
3790
0
  double     sbtOffCost        = MAX_DOUBLE;
3791
0
  uint8_t    currBestSbt       = 0;
3792
0
  uint8_t    histBestSbt       = MAX_UCHAR;
3793
0
  Distortion curPuSse          = MAX_DISTORTION;
3794
0
  uint8_t    numRDOTried       = 0;
3795
0
  bool       doPreAnalyzeResi  = false;
3796
0
  const bool mtsAllowed        =   tempCS->sps->MTSInter && cu->Y().maxDim() <= MTS_INTER_MAX_CU_SIZE;
3797
0
  const uint8_t sbtAllowed     = ( tempCS->pps->picWidthInLumaSamples < SBT_FAST64_WIDTH_THRESHOLD || m_pcEncCfg->m_SBT > 1 ) && cu->Y().maxDim() > 32 ? 0 : CU::checkAllowedSbt(*cu);
3798
3799
0
  if( sbtAllowed )
3800
0
  {
3801
    //SBT resolution-dependent fast algorithm: not try size-64 SBT in RDO for low-resolution sequences (now resolution below HD)
3802
0
    doPreAnalyzeResi = ( sbtAllowed || mtsAllowed ) && residualPass == 0;
3803
0
    m_cInterSearch.getBestSbt( tempCS, cu, histBestSbt, curPuSse, sbtAllowed, doPreAnalyzeResi, mtsAllowed );
3804
0
  }
3805
3806
0
  cu->skip    = false;
3807
0
  cu->sbtInfo = 0;
3808
3809
0
  const bool skipResidual = residualPass == 1;
3810
0
  if( skipResidual || histBestSbt == MAX_UCHAR || !CU::isSbtMode( histBestSbt ) )
3811
0
  {
3812
0
    m_cInterSearch.encodeResAndCalcRdInterCU( *tempCS, partitioner, skipResidual );
3813
0
    xEncodeDontSplit( *tempCS, partitioner );
3814
0
    xCheckDQP       ( *tempCS, partitioner );
3815
3816
0
    if( NULL != bestHasNonResi && (bestCostInternal > tempCS->cost) )
3817
0
    {
3818
0
      bestCostInternal = tempCS->cost;
3819
0
      if( !cu->ciip )
3820
0
        *bestHasNonResi = !cu->rootCbf;
3821
0
    }
3822
3823
0
    if( cu->rootCbf == false )
3824
0
    {
3825
0
      if( cu->ciip )
3826
0
      {
3827
0
        tempCS->cost = MAX_DOUBLE;
3828
0
        tempCS->costDbOffset = 0;
3829
0
        return;
3830
0
      }
3831
0
    }
3832
0
    currBestCost = tempCS->cost;
3833
0
    if( sbtAllowed )
3834
0
    {
3835
0
      sbtOffCost    = tempCS->cost;
3836
0
      sbtOffDist    = tempCS->dist;
3837
0
      sbtOffRootCbf = cu->rootCbf;
3838
0
      currBestSbt   = cu->firstTU->mtsIdx[COMP_Y] > MTS_SKIP ? SBT_OFF_MTS : SBT_OFF_DCT;
3839
0
      numRDOTried  += mtsAllowed ? 2 : 1;
3840
0
    }
3841
3842
0
    DTRACE_MODE_COST( *tempCS, m_cRdCost.getLambda( true ) );
3843
0
    xCheckBestMode( tempCS, bestCS, partitioner, encTestMode );
3844
3845
0
    STAT_COUNT_CU_MODES( partitioner.chType == CH_L, g_cuCounters1D[CU_RD_TESTS][0][!tempCS->slice->isIntra() + tempCS->slice->depth] );
3846
0
    STAT_COUNT_CU_MODES( partitioner.chType == CH_L && !tempCS->slice->isIntra(), g_cuCounters2D[CU_RD_TESTS][Log2( tempCS->area.lheight() )][Log2( tempCS->area.lwidth() )] );
3847
0
  }
3848
3849
0
  if( sbtAllowed && ( m_pcEncCfg->m_SBT == 1 || sbtOffRootCbf ) )
3850
0
  {
3851
0
    bool swapped = false; // avoid unwanted data copy
3852
0
    uint8_t numSbtRdo = CU::numSbtModeRdo( sbtAllowed );
3853
    //early termination if all SBT modes are not allowed
3854
    //normative
3855
0
    if( !sbtAllowed || skipResidual )
3856
0
    {
3857
0
      numSbtRdo = 0;
3858
0
    }
3859
    //fast algorithm
3860
0
    if( ( histBestSbt != MAX_UCHAR && !CU::isSbtMode( histBestSbt ) ) || m_cInterSearch.getSkipSbtAll() )
3861
0
    {
3862
0
      numSbtRdo = 0;
3863
0
    }
3864
0
    if( bestCost != MAX_DOUBLE && sbtOffCost != MAX_DOUBLE )
3865
0
    {
3866
0
      double th = 1.07;
3867
0
      if( !( prevBestSbt == 0 || m_sbtCostSave[0] == MAX_DOUBLE ) )
3868
0
      {
3869
0
        assert( m_sbtCostSave[1] <= m_sbtCostSave[0] );
3870
0
        th *= ( m_sbtCostSave[0] / m_sbtCostSave[1] );
3871
0
      }
3872
0
      if( sbtOffCost > bestCost * th )
3873
0
      {
3874
0
        numSbtRdo = 0;
3875
0
      }
3876
0
    }
3877
0
    if( !sbtOffRootCbf && sbtOffCost != MAX_DOUBLE )
3878
0
    {
3879
0
      double th = Clip3( 0.05, 0.55, ( 27 - cu->qp ) * 0.02 + 0.35 );
3880
0
      if( sbtOffCost < m_cRdCost.calcRdCost( ( cu->lwidth() * cu->lheight() ) << SCALE_BITS, 0 ) * th )
3881
0
      {
3882
0
        numSbtRdo = 0;
3883
0
      }
3884
0
    }
3885
3886
0
    if( histBestSbt != MAX_UCHAR && numSbtRdo != 0 )
3887
0
    {
3888
0
      numSbtRdo = 1;
3889
0
      m_cInterSearch.initSbtRdoOrder( CU::getSbtMode( CU::getSbtIdx( histBestSbt ), CU::getSbtPos( histBestSbt ) ) );
3890
0
    }
3891
3892
0
    for( int sbtModeIdx = 0; sbtModeIdx < numSbtRdo; sbtModeIdx++ )
3893
0
    {
3894
0
      uint8_t sbtMode = m_cInterSearch.getSbtRdoOrder( sbtModeIdx );
3895
0
      uint8_t sbtIdx = CU::getSbtIdxFromSbtMode( sbtMode );
3896
0
      uint8_t sbtPos = CU::getSbtPosFromSbtMode( sbtMode );
3897
3898
      //fast algorithm (early skip, save & load)
3899
0
      if( histBestSbt == MAX_UCHAR )
3900
0
      {
3901
0
        uint8_t skipCode = m_cInterSearch.skipSbtByRDCost( cu->lwidth(), cu->lheight(), cu->mtDepth, sbtIdx, sbtPos, bestCS->cost, sbtOffDist, sbtOffCost, sbtOffRootCbf );
3902
0
        if( skipCode != MAX_UCHAR )
3903
0
        {
3904
0
          continue;
3905
0
        }
3906
3907
0
        if( sbtModeIdx > 0 )
3908
0
        {
3909
0
          uint8_t prevSbtMode = m_cInterSearch.getSbtRdoOrder( sbtModeIdx - 1 );
3910
          //make sure the prevSbtMode is the same size as the current SBT mode (otherwise the estimated dist may not be comparable)
3911
0
          if( CU::isSameSbtSize( prevSbtMode, sbtMode ) )
3912
0
          {
3913
0
            Distortion currEstDist = m_cInterSearch.getEstDistSbt( sbtMode );
3914
0
            Distortion prevEstDist = m_cInterSearch.getEstDistSbt( prevSbtMode );
3915
0
            if( currEstDist > prevEstDist * 1.15 )
3916
0
            {
3917
0
              continue;
3918
0
            }
3919
0
          }
3920
0
        }
3921
0
      }
3922
3923
      //init tempCS and TU
3924
0
      if( bestCost == bestCS->cost ) //The first EMT pass didn't become the bestCS, so we clear the TUs generated
3925
0
      {
3926
0
        tempCS->clearTUs();
3927
0
      }
3928
0
      else if( !swapped )
3929
0
      {
3930
0
        tempCS->initStructData( encTestMode.qp );
3931
0
        tempCS->copyStructure( *bestCS, partitioner.chType, partitioner.treeType );
3932
0
        tempCS->getPredBuf().copyFrom( bestCS->getPredBuf() );
3933
0
        bestCost = bestCS->cost;
3934
0
        cu = tempCS->getCU( partitioner.chType, partitioner.treeType );
3935
0
        swapped = true;
3936
0
      }
3937
0
      else
3938
0
      {
3939
0
        tempCS->clearTUs();
3940
0
        bestCost = bestCS->cost;
3941
0
        cu = tempCS->getCU( partitioner.chType, partitioner.treeType );
3942
0
      }
3943
3944
      //we need to restart the distortion for the new tempCS, the bit count and the cost
3945
0
      tempCS->dist     = 0;
3946
0
      tempCS->fracBits = 0;
3947
0
      tempCS->cost     = MAX_DOUBLE;
3948
0
      cu->skip         = false;
3949
3950
3951
      //set SBT info
3952
0
      cu->sbtInfo = (sbtPos << 4) + sbtIdx;
3953
3954
      //try residual coding
3955
0
      m_cInterSearch.encodeResAndCalcRdInterCU( *tempCS, partitioner, skipResidual );
3956
0
      numRDOTried++;
3957
3958
0
      xEncodeDontSplit( *tempCS, partitioner );
3959
0
      xCheckDQP( *tempCS, partitioner );
3960
3961
0
      if( NULL != bestHasNonResi && ( bestCostInternal > tempCS->cost ) )
3962
0
      {
3963
0
        bestCostInternal = tempCS->cost;
3964
0
        if( !( cu->ciip ) )
3965
0
          *bestHasNonResi = !cu->rootCbf;
3966
0
      }
3967
3968
0
      if( tempCS->cost < currBestCost )
3969
0
      {
3970
0
        currBestSbt = cu->sbtInfo;
3971
0
        currBestCost = tempCS->cost;
3972
0
      }
3973
0
      else if( m_pcEncCfg->m_SBT > 2 )
3974
0
      {
3975
0
        sbtModeIdx = numSbtRdo;
3976
0
      }
3977
3978
0
      DTRACE_MODE_COST( *tempCS, m_cRdCost.getLambda( true ) );
3979
0
      xCheckBestMode( tempCS, bestCS, partitioner, encTestMode );
3980
0
      STAT_COUNT_CU_MODES( partitioner.chType == CH_L, g_cuCounters1D[CU_RD_TESTS][0][!tempCS->slice->isIntra() + tempCS->slice->depth] );
3981
0
      STAT_COUNT_CU_MODES( partitioner.chType == CH_L && !tempCS->slice->isIntra(), g_cuCounters2D[CU_RD_TESTS][Log2( tempCS->area.lheight() )][Log2( tempCS->area.lwidth() )] );
3982
0
    }
3983
3984
0
    if( bestCostBegin != bestCS->cost )
3985
0
    {
3986
0
      m_sbtCostSave[0] = sbtOffCost;
3987
0
      m_sbtCostSave[1] = currBestCost;
3988
0
    }
3989
3990
0
    if( histBestSbt == MAX_UCHAR && doPreAnalyzeResi && numRDOTried > 1 )
3991
0
    {
3992
0
      auto slsSbt = static_cast<CacheBlkInfoCtrl&>( m_modeCtrl );
3993
0
      int slShift = 4 + std::min( Log2( cu->lwidth() ) + Log2( cu->lheight() ), 9 );
3994
0
      slsSbt.saveBestSbt( cu->cs->area, (uint32_t)( curPuSse >> slShift ), currBestSbt );
3995
0
    }
3996
    
3997
0
    if( ETM_INTER_ME == encTestMode.type )
3998
0
    {
3999
0
      if( equBcwCost != NULL )
4000
0
      {
4001
0
        if( tempCS->cost < ( *equBcwCost ) && cu->BcwIdx == BCW_DEFAULT )
4002
0
        {
4003
0
          ( *equBcwCost ) = tempCS->cost;
4004
0
        }
4005
0
      }
4006
0
      else
4007
0
      {
4008
0
        CHECK( equBcwCost == NULL, "equBcwCost == NULL" );
4009
0
      }
4010
0
      if( tempCS->slice->checkLDC && !cu->imv && cu->BcwIdx != BCW_DEFAULT && tempCS->cost < m_bestBcwCost[1] )
4011
0
      {
4012
0
        if( tempCS->cost < m_bestBcwCost[0] )
4013
0
        {
4014
0
          m_bestBcwCost[1] = m_bestBcwCost[0];
4015
0
          m_bestBcwCost[0] = tempCS->cost;
4016
0
          m_bestBcwIdx[1] = m_bestBcwIdx[0];
4017
0
          m_bestBcwIdx[0] = cu->BcwIdx;
4018
0
        }
4019
0
        else
4020
0
        {
4021
0
          m_bestBcwCost[1] = tempCS->cost;
4022
0
          m_bestBcwIdx[1] = cu->BcwIdx;
4023
0
        }
4024
0
      }
4025
0
    }
4026
0
  }
4027
4028
0
  tempCS->cost = currBestCost;
4029
0
}
4030
4031
void EncCu::xEncodeDontSplit( CodingStructure &cs, Partitioner &partitioner )
4032
100k
{
4033
100k
  m_CABACEstimator->resetBits();
4034
4035
100k
  m_CABACEstimator->split_cu_mode( CU_DONT_SPLIT, cs, partitioner );
4036
100k
  if( partitioner.treeType == TREE_C )
4037
100k
    CHECK( m_CABACEstimator->getEstFracBits() != 0, "must be 0 bit" );
4038
4039
100k
  cs.fracBits += m_CABACEstimator->getEstFracBits(); // split bits
4040
100k
  cs.cost      = m_cRdCost.calcRdCost( cs.fracBits, cs.dist );
4041
100k
}
4042
4043
void EncCu::xReuseCachedResult( CodingStructure *&tempCS, CodingStructure *&bestCS, Partitioner &partitioner )
4044
0
{
4045
0
  EncTestMode cachedMode;
4046
4047
0
  if( ! m_modeCtrl.setCsFrom( *tempCS, cachedMode, partitioner ) )
4048
0
  {
4049
0
    THROW( "Should never happen!" );
4050
0
  }
4051
4052
0
  CodingUnit& cu = *tempCS->cus.front();
4053
0
  partitioner.setCUData( cu );
4054
4055
0
  if( CU::isIntra( cu ) )
4056
0
  {
4057
0
    if( isLuma( cu.chType ) )
4058
0
    {
4059
0
      cu.getMotionBuf().memset( -1 ); // clear motion buf
4060
0
    }
4061
0
    xReconIntraQT( cu );
4062
0
  }
4063
0
  else
4064
0
  {
4065
0
    xDeriveCUMV( cu );
4066
0
    xReconInter( cu );
4067
0
  }
4068
4069
0
  m_CABACEstimator->getCtx() = m_CurrCtx->start;
4070
0
  m_CABACEstimator->resetBits();
4071
4072
0
  CUCtx cuCtx;
4073
0
  cuCtx.isDQPCoded = true;
4074
0
  cuCtx.isChromaQpAdjCoded = true;
4075
0
  m_CABACEstimator->coding_unit( cu, partitioner, cuCtx );
4076
4077
0
  tempCS->fracBits = m_CABACEstimator->getEstFracBits();
4078
0
  tempCS->cost     = m_cRdCost.calcRdCost( tempCS->fracBits, tempCS->dist );
4079
4080
0
  xEncodeDontSplit( *tempCS,         partitioner );
4081
0
  xCheckDQP       ( *tempCS,         partitioner );
4082
0
  xCheckBestMode  (  tempCS, bestCS, partitioner, cachedMode, m_EDO );
4083
0
}
4084
4085
uint64_t EncCu::xCalcPuMeBits( const CodingUnit &cu )
4086
0
{
4087
0
  CHECK( !cu.mergeFlag, "Should only be used for merge!" );
4088
0
  CHECK( CU::isIBC( cu ), "Shound not be used for IBC" );
4089
4090
0
  m_CABACEstimator->resetBits();
4091
0
  m_CABACEstimator->merge_flag(cu);
4092
0
  if( cu.mergeFlag )
4093
0
  {
4094
0
    m_CABACEstimator->merge_data( cu );
4095
0
  }
4096
0
  return m_CABACEstimator->getEstFracBits();
4097
0
}
4098
4099
double EncCu::xCalcDistortion(CodingStructure *&cur_CS, ChannelType chType, int BitDepth, int imv)
4100
0
{
4101
0
  const auto currDist1 = m_cRdCost.getDistPart(cur_CS->getOrgBuf( COMP_Y ), cur_CS->getPredBuf( COMP_Y ), BitDepth, COMP_Y, m_pcEncCfg->m_fastHad ? DF_HAD_fast : DF_HAD );
4102
0
  unsigned int uiMvBits = 0;
4103
0
  unsigned imvShift = imv == IMV_HPEL ? 1 : (imv << 1);
4104
0
  const CodingUnit& cu = *cur_CS->getCU( chType, TREE_D);
4105
0
  if (cu.interDir != 2)
4106
0
  {
4107
0
    uiMvBits += m_cRdCost.getBitsOfVectorWithPredictor(cu.mvd[0][0].hor, cu.mvd[0][0].ver, imvShift + MV_FRACTIONAL_BITS_DIFF);
4108
0
  }
4109
0
  if (cu.interDir != 1)
4110
0
  {
4111
0
    uiMvBits += m_cRdCost.getBitsOfVectorWithPredictor(cu.mvd[1][0].hor, cu.mvd[1][0].ver, imvShift + MV_FRACTIONAL_BITS_DIFF);
4112
0
  }
4113
0
  return (double(currDist1) + (double)m_cRdCost.getCost(uiMvBits));
4114
0
}
4115
4116
int EncCu::xCheckMMVDCand(MmvdIdx& mmvdMergeCand, int& bestDir, int tempNum, double& bestCostOffset, double& bestCostMerge, double bestCostList )
4117
0
{
4118
0
  int baseIdx = mmvdMergeCand.val / MMVD_MAX_REFINE_NUM;
4119
0
  int CandCur = mmvdMergeCand.val - MMVD_MAX_REFINE_NUM * baseIdx;
4120
4121
0
  if( m_pcEncCfg->m_MMVD > 2 )
4122
0
  {
4123
0
    if( CandCur % 4 == 0 )
4124
0
    {
4125
0
      if( ( bestCostOffset >= bestCostMerge ) && ( CandCur >= 4 ) )
4126
0
      {
4127
0
        if( mmvdMergeCand.val > MMVD_MAX_REFINE_NUM )
4128
0
        {
4129
0
          return 2;
4130
0
        }
4131
0
        else
4132
0
        {
4133
0
          mmvdMergeCand.val = MMVD_MAX_REFINE_NUM;
4134
0
          if( tempNum == mmvdMergeCand.val )
4135
0
          {
4136
0
            return 2;
4137
0
          }
4138
0
        }
4139
0
      }
4140
      //reset
4141
0
      bestCostOffset = MAX_DOUBLE;
4142
0
      bestCostMerge  = bestCostList;
4143
0
    }
4144
0
  }
4145
4146
0
  if( mmvdMergeCand.val == MMVD_MAX_REFINE_NUM )
4147
0
  {
4148
0
    bestDir = 0;
4149
0
  }
4150
0
  if( CandCur >= 4 )
4151
0
  {
4152
0
    if( CandCur % 4 != bestDir )
4153
0
    {
4154
0
      return 1;
4155
0
    }
4156
0
  }
4157
0
  return 0;
4158
0
}
4159
4160
void EncCu::xInitLumaDeltaQpLUT()
4161
0
{
4162
  //TODO: to be discussed: add config parameters for these?
4163
0
  std::vector<int> defaultLumaLevelTodQp_QpChangePoints   =  { -3,  -2,  -1,   0,   1,   2,   3,   4,   5,   6 };
4164
0
  std::vector<int> defaultLumaLevelTodQp_LumaChangePoints =  {  0, 301, 367, 434, 501, 567, 634, 701, 767, 834 };
4165
  
4166
0
  int         lastDeltaQPValue = 0;
4167
0
  std::size_t nextSparseIndex = 0;
4168
0
  for( int index = 0; index < LUMA_LEVEL_TO_DQP_LUT_MAXSIZE; index++ )
4169
0
  {
4170
0
    while( nextSparseIndex < defaultLumaLevelTodQp_QpChangePoints.size() && index >= defaultLumaLevelTodQp_LumaChangePoints[nextSparseIndex] )
4171
0
    {
4172
0
      lastDeltaQPValue = defaultLumaLevelTodQp_QpChangePoints[nextSparseIndex];
4173
0
      nextSparseIndex++;
4174
0
    }
4175
0
    m_lumaLevelToDeltaQPLUT[index] = lastDeltaQPValue;
4176
0
  }
4177
0
}
4178
4179
MergeItem::MergeItem()
4180
149k
{
4181
4182
149k
}
4183
MergeItem::~MergeItem()
4184
149k
{
4185
4186
149k
}
4187
4188
void MergeItem::create( ChromaFormat chromaFormat, const Area &area )
4189
149k
{
4190
149k
  if( m_pelStorage.bufs.empty() )
4191
149k
  {
4192
149k
    m_pelStorage.create( chromaFormat, area );
4193
149k
    m_mvStorage .resize( area.area() >> ( MIN_CU_LOG2 << 1 ) );
4194
149k
  }
4195
4196
149k
  init();
4197
149k
}
4198
4199
void MergeItem::init()
4200
149k
{
4201
  // reset data
4202
149k
  cost        = MAX_DOUBLE;
4203
149k
  mergeIdx    = 0;
4204
149k
  bcwIdx      = 0;
4205
149k
  interDir    = 0;
4206
149k
  useAltHpelIf  = false;
4207
149k
  affineType    = AFFINEMODEL_4PARAM;
4208
149k
  mergeItemType = MergeItemType::NUM;
4209
4210
149k
  noBdofRefine  = false;
4211
149k
  noResidual    = false;
4212
4213
149k
  lumaPredReady   = false;
4214
149k
  chromaPredReady = false;
4215
149k
}
4216
4217
void MergeItem::importMergeInfo(const MergeCtx& mergeCtx, int _mergeIdx, MergeItemType _mergeItemType, CodingUnit& pu)
4218
0
{
4219
0
  mergeIdx      = _mergeIdx;
4220
0
  mergeItemType = _mergeItemType;
4221
4222
0
  if( mergeItemType != MergeItemType::GPM && mergeItemType != MergeItemType::MMVD )
4223
0
  {
4224
0
    mvField[REF_PIC_LIST_0][0] = mergeCtx.mvFieldNeighbours [mergeIdx][REF_PIC_LIST_0];
4225
0
    mvField[REF_PIC_LIST_1][0] = mergeCtx.mvFieldNeighbours [mergeIdx][REF_PIC_LIST_1];
4226
0
    interDir                   = mergeCtx.interDirNeighbours[mergeIdx];
4227
0
    bcwIdx                     = mergeCtx.BcwIdx            [mergeIdx];
4228
0
    useAltHpelIf               = mergeCtx.useAltHpelIf      [mergeIdx];
4229
0
  }
4230
4231
0
  switch( _mergeItemType )
4232
0
  {
4233
0
  case MergeItemType::REGULAR:
4234
0
  case MergeItemType::CIIP:
4235
0
    break;
4236
4237
0
  case MergeItemType::MMVD:
4238
0
  {
4239
0
    MmvdIdx candIdx;
4240
4241
0
    candIdx.val                = mergeIdx;
4242
0
    mvField[L0][0]             . setMvField( pu.mv[L0][0], pu.refIdx[0] );
4243
0
    mvField[L1][0]             . setMvField( pu.mv[L1][0], pu.refIdx[1] );
4244
0
    interDir                   = pu.interDir;
4245
0
    bcwIdx                     = pu.BcwIdx;
4246
0
    useAltHpelIf               = mergeCtx.useAltHpelIf[candIdx.pos.baseIdx];
4247
4248
0
    break;
4249
0
  }
4250
4251
0
  case MergeItemType::GPM:
4252
0
    mvField[L0][0]             . setMvField( Mv( 0, 0 ), -1 );
4253
0
    mvField[L1][0]             . setMvField( Mv( 0, 0 ), -1 );
4254
0
    bcwIdx                     = BCW_DEFAULT;
4255
0
    useAltHpelIf               = false;
4256
4257
0
    break;
4258
4259
0
  case MergeItemType::IBC:
4260
0
  default:
4261
0
    THROW( "Wrong merge item type" );
4262
0
  }
4263
4264
0
  getMvBuf( pu ).copyFrom( pu.getMotionBuf() );
4265
0
}
4266
4267
void MergeItem::importMergeInfo( const AffineMergeCtx &mergeCtx, int _mergeIdx, MergeItemType _mergeItemType, CodingUnit& pu )
4268
0
{
4269
0
  mergeIdx      = _mergeIdx;
4270
0
  mergeItemType = _mergeItemType;
4271
4272
0
  affineType    = mergeCtx.affineType         [mergeIdx];
4273
0
  interDir      = mergeCtx.interDirNeighbours [mergeIdx];
4274
0
  bcwIdx        = mergeCtx.BcwIdx             [mergeIdx];
4275
0
  useAltHpelIf  = false;
4276
4277
0
  switch( _mergeItemType )
4278
0
  {
4279
0
  case MergeItemType::SBTMVP:
4280
    // the pu motion was already generated preparing for IFP check (unconditional)
4281
0
    mvField[L0][0] . setMvField( pu.mv[L0][0], pu.refIdx[L0] );
4282
0
    mvField[L1][0] . setMvField( pu.mv[L1][0], pu.refIdx[L1] );
4283
4284
0
    break;
4285
4286
0
  case MergeItemType::AFFINE:
4287
    // the pu motion was already generated preparing for IFP check (unconditional)
4288
0
    mvField[L0][0] . setMvField( pu.mv[L0][0], pu.refIdx[L0] );
4289
0
    mvField[L0][1] . setMvField( pu.mv[L0][1], pu.refIdx[L0] );
4290
0
    mvField[L0][2] . setMvField( pu.mv[L0][2], pu.refIdx[L0] );
4291
0
    mvField[L1][0] . setMvField( pu.mv[L1][0], pu.refIdx[L1] );
4292
0
    mvField[L1][1] . setMvField( pu.mv[L1][1], pu.refIdx[L1] );
4293
0
    mvField[L1][2] . setMvField( pu.mv[L1][2], pu.refIdx[L1] );
4294
4295
0
    break;
4296
4297
0
  default:
4298
0
    THROW( "Wrong merge item type" );
4299
0
  }
4300
4301
  // the MI buf was already generated preparing for IFP check (unconditional)
4302
0
  getMvBuf( pu ).copyFrom( pu.getMotionBuf() );
4303
0
}
4304
4305
bool MergeItem::exportMergeInfo( CodingUnit &pu, bool forceNoResidual ) const
4306
0
{
4307
0
  pu.mergeFlag        = true;
4308
0
  pu.mmvdMergeFlag    = false;
4309
0
  pu.interDir         = interDir;
4310
0
  pu.mergeIdx         = mergeIdx;
4311
0
  pu.mergeType        = MRG_TYPE_DEFAULT_N;
4312
0
  pu.mv[REF_PIC_LIST_0][0]  = mvField[REF_PIC_LIST_0][0].mv;
4313
0
  pu.mv[REF_PIC_LIST_1][0]  = mvField[REF_PIC_LIST_1][0].mv;
4314
0
  pu.refIdx[REF_PIC_LIST_0] = mvField[REF_PIC_LIST_0][0].refIdx;
4315
0
  pu.refIdx[REF_PIC_LIST_1] = mvField[REF_PIC_LIST_1][0].refIdx;
4316
0
  pu.mvd[REF_PIC_LIST_0][0] = Mv();
4317
0
  pu.mvd[REF_PIC_LIST_1][0] = Mv();
4318
0
  pu.mvpIdx[REF_PIC_LIST_0] = NOT_VALID;
4319
0
  pu.mvpIdx[REF_PIC_LIST_1] = NOT_VALID;
4320
0
  pu.mvpNum[REF_PIC_LIST_0] = NOT_VALID;
4321
0
  pu.mvpNum[REF_PIC_LIST_1] = NOT_VALID;
4322
0
  pu.BcwIdx         = ( interDir == 3 ) ? bcwIdx : BCW_DEFAULT;
4323
0
  pu.mcControl      = 0;
4324
0
  pu.mmvdSkip       = false;
4325
0
  pu.affine         = false;
4326
0
  pu.affineType     = AFFINEMODEL_4PARAM;
4327
0
  pu.geo            = false;
4328
0
  pu.mtsFlag        = false;
4329
0
  pu.ciip           = false;
4330
0
  pu.imv            = ( !pu.geo && useAltHpelIf ) ? IMV_HPEL : IMV_OFF;
4331
0
  pu.mvRefine       = false;
4332
4333
0
  const bool resetCiip2Regular = mergeItemType == MergeItemType::CIIP && forceNoResidual;
4334
0
  MergeItemType updatedType    = resetCiip2Regular ? MergeItemType::REGULAR : mergeItemType;
4335
4336
0
  switch( updatedType )
4337
0
  {
4338
0
  case MergeItemType::REGULAR:
4339
0
    CU::restrictBiPredMergeCandsOne( pu );
4340
0
    break;
4341
4342
0
  case MergeItemType::CIIP:
4343
0
    CHECK( forceNoResidual, "Cannot force no residuals for CIIP" );
4344
0
    pu.ciip           = true;
4345
0
    pu.intraDir[CH_L] = PLANAR_IDX;
4346
0
    pu.intraDir[CH_C] = DM_CHROMA_IDX;
4347
0
    break;
4348
4349
0
  case MergeItemType::MMVD:
4350
0
    pu.mmvdMergeFlag    = true;
4351
0
    pu.mmvdMergeIdx.val = mergeIdx;
4352
0
    if( forceNoResidual )
4353
0
    {
4354
0
      pu.mmvdSkip       = true;
4355
0
    }
4356
0
    CU::restrictBiPredMergeCandsOne( pu );
4357
0
    break;
4358
4359
0
  case MergeItemType::SBTMVP:
4360
0
    pu.affine    = true;
4361
0
    pu.mergeType = MRG_TYPE_SUBPU_ATMVP;
4362
0
    break;
4363
4364
0
  case MergeItemType::AFFINE:
4365
0
    pu.affine     = true;
4366
0
    pu.affineType = affineType;
4367
0
    pu.mv[L0][0]  = mvField[L0][0].mv;
4368
0
    pu.mv[L1][0]  = mvField[L1][0].mv;
4369
0
    pu.mv[L0][1]  = mvField[L0][1].mv;
4370
0
    pu.mv[L1][1]  = mvField[L1][1].mv;
4371
0
    pu.mv[L0][2]  = mvField[L0][2].mv;
4372
0
    pu.mv[L1][2]  = mvField[L1][2].mv;
4373
0
    pu.refIdx[L0] = mvField[L0][0].refIdx;
4374
0
    pu.refIdx[L1] = mvField[L1][0].refIdx;
4375
0
    break;
4376
4377
0
  case MergeItemType::GPM:
4378
0
    pu.mergeIdx = -1;
4379
0
    pu.geo      = true;
4380
0
    pu.BcwIdx   = BCW_DEFAULT;
4381
0
    updateGpmIdx( mergeIdx, pu.geoSplitDir, pu.geoMergeIdx );
4382
0
    pu.imv      = IMV_OFF;
4383
0
    break;
4384
4385
0
  case MergeItemType::IBC:
4386
0
  default:
4387
0
    THROW( "Wrong merge item type" );
4388
0
  }
4389
4390
0
  pu.getMotionBuf().copyFrom( getMvBuf( pu ) );
4391
4392
0
  return resetCiip2Regular;
4393
0
}
4394
4395
MergeItemList::MergeItemList()
4396
18.6k
{
4397
4398
18.6k
}
4399
4400
MergeItemList::~MergeItemList()
4401
18.6k
{
4402
18.6k
  for( MergeItem* p : m_list )
4403
0
  {
4404
0
    delete p;
4405
0
  }
4406
18.6k
  m_list.clear();
4407
4408
18.6k
  for( MergeItem *p : m_mergeItems )
4409
149k
  {
4410
149k
    delete p;
4411
149k
  }
4412
18.6k
  m_mergeItems.clear();
4413
18.6k
}
4414
4415
void MergeItemList::init( size_t maxSize, size_t maxExtSize, ChromaFormat chromaFormat, SizeType ctuWidth, SizeType ctuHeight )
4416
18.6k
{
4417
18.6k
  CHECK( !m_mergeItems.empty() || !m_list.empty(), "MergeItemList already initialized" );
4418
4419
18.6k
  m_list      . reserve( maxSize + 1 ); // to avoid reallocation when inserting a new item
4420
18.6k
  m_mergeItems. reserve( maxSize + 1 );
4421
18.6k
  m_maxSize   = maxSize;
4422
18.6k
  m_maxExtSize= maxExtSize;
4423
18.6k
  m_numExt    = 0;
4424
4425
167k
  for( int i = 0; i < maxSize + m_maxExtSize; i++ )
4426
149k
  {
4427
149k
    MergeItem *p = new MergeItem;
4428
149k
    p->create( chromaFormat, Area{ 0, 0, ctuWidth, ctuHeight } );
4429
149k
    m_mergeItems.push_back( p );
4430
149k
  }
4431
18.6k
}
4432
4433
MergeItem *MergeItemList::allocateNewMergeItem()
4434
0
{
4435
0
  m_numExt++;
4436
0
  CHECK( m_mergeItems.empty(), "Missing merge items!" );
4437
0
  CHECK( m_numExt > m_maxExtSize, "Taking out more external items than specified during list allocation!" );
4438
0
  MergeItem *p = m_mergeItems.back();
4439
0
  m_mergeItems.pop_back();
4440
0
  p->init();
4441
0
  return p;
4442
0
}
4443
4444
bool MergeItemList::insertMergeItemToList( MergeItem *p )
4445
0
{
4446
0
  CHECK( m_list.size() + m_mergeItems.size() + m_numExt != m_maxSize + m_maxExtSize, "Wrong number of items held" );
4447
4448
0
  m_numExt--;
4449
4450
0
  if( m_list.empty() )
4451
0
  {
4452
0
    m_list.push_back( p );
4453
0
  }
4454
0
  else if( m_list.size() == m_maxTrackingNum && p->cost >= m_list.back()->cost )
4455
0
  {
4456
0
    m_mergeItems.push_back( p );
4457
0
    return false;
4458
0
  }
4459
0
  else
4460
0
  {
4461
0
    if( m_list.size() == m_maxTrackingNum )
4462
0
    {
4463
0
      m_mergeItems.push_back( m_list.back() );
4464
0
      m_list      .pop_back();
4465
0
    }
4466
0
    auto it = std::find_if( m_list.begin(), m_list.end(), [&p]( const MergeItem *mi ) { return p->cost < mi->cost; } );
4467
0
    m_list.insert( it, p );
4468
0
  }
4469
4470
0
  return true;
4471
0
}
4472
4473
void MergeItemList::giveBackMergeItem( MergeItem *p )
4474
0
{
4475
0
  CHECK( m_list.size() + m_mergeItems.size() + m_numExt != m_maxSize + m_maxExtSize, "Wrong number of items held" );
4476
4477
0
  m_numExt--;
4478
4479
0
  m_mergeItems.push_back( p );
4480
0
}
4481
4482
MergeItem *MergeItemList::getMergeItemInList( size_t index )
4483
0
{
4484
0
  return index < m_maxTrackingNum ? m_list[index] : nullptr;
4485
0
}
4486
4487
void MergeItemList::resetList( size_t maxTrackingNum )
4488
0
{
4489
0
  CHECK( maxTrackingNum > m_maxSize, "Not enough items allocated to track " << maxTrackingNum << " items" );
4490
4491
0
  for( auto p : m_list )
4492
0
  {
4493
0
    m_mergeItems.push_back( p );
4494
0
  }
4495
0
  m_list.clear  ();
4496
4497
0
  m_maxTrackingNum = maxTrackingNum;
4498
0
}
4499
4500
void MergeItemList::shrinkList( size_t reduceTo )
4501
0
{
4502
0
  CHECK( reduceTo > m_maxSize, "Not enough items allocated to track " << reduceTo << " items" );
4503
4504
0
  while( m_list.size() > reduceTo )
4505
0
  {
4506
0
    m_mergeItems.push_back( m_list.back() );
4507
0
    m_list      .pop_back();
4508
0
  }
4509
0
}
4510
4511
} // namespace vvenc
4512
4513
//! \}