Coverage Report

Created: 2026-09-14 06:44

next uncovered line (L), next uncovered region (R), next uncovered branch (B)
/work/vvenc/source/Lib/EncoderLib/EncCu.cpp
Line
Count
Source
1
/* -----------------------------------------------------------------------------
2
The copyright in this software is being made available under the Clear BSD
3
License, included below. No patent rights, trademark rights and/or
4
other Intellectual Property Rights other than the copyrights concerning
5
the Software are granted under this license.
6
7
The Clear BSD License
8
9
Copyright (c) 2019-2026, Fraunhofer-Gesellschaft zur Förderung der angewandten Forschung e.V. & The VVenC Authors.
10
All rights reserved.
11
12
Redistribution and use in source and binary forms, with or without modification,
13
are permitted (subject to the limitations in the disclaimer below) provided that
14
the following conditions are met:
15
16
     * Redistributions of source code must retain the above copyright notice,
17
     this list of conditions and the following disclaimer.
18
19
     * Redistributions in binary form must reproduce the above copyright
20
     notice, this list of conditions and the following disclaimer in the
21
     documentation and/or other materials provided with the distribution.
22
23
     * Neither the name of the copyright holder nor the names of its
24
     contributors may be used to endorse or promote products derived from this
25
     software without specific prior written permission.
26
27
NO EXPRESS OR IMPLIED LICENSES TO ANY PARTY'S PATENT RIGHTS ARE GRANTED BY
28
THIS LICENSE. THIS SOFTWARE IS PROVIDED BY THE COPYRIGHT HOLDERS AND
29
CONTRIBUTORS "AS IS" AND ANY EXPRESS OR IMPLIED WARRANTIES, INCLUDING, BUT NOT
30
LIMITED TO, THE IMPLIED WARRANTIES OF MERCHANTABILITY AND FITNESS FOR A
31
PARTICULAR PURPOSE ARE DISCLAIMED. IN NO EVENT SHALL THE COPYRIGHT HOLDER OR
32
CONTRIBUTORS BE LIABLE FOR ANY DIRECT, INDIRECT, INCIDENTAL, SPECIAL,
33
EXEMPLARY, OR CONSEQUENTIAL DAMAGES (INCLUDING, BUT NOT LIMITED TO,
34
PROCUREMENT OF SUBSTITUTE GOODS OR SERVICES; LOSS OF USE, DATA, OR PROFITS; OR
35
BUSINESS INTERRUPTION) HOWEVER CAUSED AND ON ANY THEORY OF LIABILITY, WHETHER
36
IN CONTRACT, STRICT LIABILITY, OR TORT (INCLUDING NEGLIGENCE OR OTHERWISE)
37
ARISING IN ANY WAY OUT OF THE USE OF THIS SOFTWARE, EVEN IF ADVISED OF THE
38
POSSIBILITY OF SUCH DAMAGE.
39
40
41
------------------------------------------------------------------------------------------- */
42
43
44
/** \file     EncCu.cpp
45
    \brief    Coding Unit (CU) encoder class
46
*/
47
48
#include "EncCu.h"
49
#include "EncLib.h"
50
#include "Analyze.h"
51
#include "EncPicture.h"
52
#include "EncModeCtrl.h"
53
#include "BitAllocation.h"
54
#include "EncStage.h"
55
56
#include "CommonLib/dtrace_codingstruct.h"
57
#include "CommonLib/Picture.h"
58
#include "CommonLib/UnitTools.h"
59
#include "CommonLib/dtrace_buffer.h"
60
#include "CommonLib/TimeProfiler.h"
61
#include "CommonLib/SearchSpaceCounter.h"
62
63
#include <mutex>
64
#include <cmath>
65
#include <algorithm>
66
67
//! \ingroup EncoderLib
68
//! \{
69
70
namespace vvenc {
71
72
const MergeIdxPair EncCu::m_GeoModeTest[GEO_MAX_NUM_CANDS] = { MergeIdxPair{0, 1}, MergeIdxPair{1, 0}, MergeIdxPair{0, 2}, MergeIdxPair{1, 2}, MergeIdxPair{2, 0},
73
                                                               MergeIdxPair{2, 1}, MergeIdxPair{0, 3}, MergeIdxPair{1, 3}, MergeIdxPair{2, 3}, MergeIdxPair{3, 0},
74
                                                               MergeIdxPair{3, 1}, MergeIdxPair{3, 2}, MergeIdxPair{0, 4}, MergeIdxPair{1, 4}, MergeIdxPair{2, 4},
75
                                                               MergeIdxPair{3, 4}, MergeIdxPair{4, 0}, MergeIdxPair{4, 1}, MergeIdxPair{4, 2}, MergeIdxPair{4, 3},
76
                                                               MergeIdxPair{0, 5}, MergeIdxPair{1, 5}, MergeIdxPair{2, 5}, MergeIdxPair{3, 5}, MergeIdxPair{4, 5},
77
                                                               MergeIdxPair{5, 0}, MergeIdxPair{5, 1}, MergeIdxPair{5, 2}, MergeIdxPair{5, 3}, MergeIdxPair{5, 4} };
78
79
80
// Shape coefSquareCUsFasterFastMedium (2 x 5 x 2 x 2 x 2): preset (faster and fast + medium) x cusize x nspred x sptype x numcoef
81
82
const double EncCu::coefSquareCUsFasterFastMedium[2][5][2][2][2] = {
83
{{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, },  {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, },  },
84
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, },  {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, },  },
85
{{{-1.00000000, -1.00000000, }, {0.07848505, 0.00225808, }, },  {{-1.00000000, -1.00000000, }, {0.07509575, 0.00204789, }, },  },
86
{{{-1.00000000, -1.00000000, }, {0.10833051, 0.00053144, }, },  {{-1.00000000, -1.00000000, }, {0.08304352, 0.00142876, }, },  },
87
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, },  {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, },  },
88
},
89
{{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, },  {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, },  },
90
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, },  {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, },  },
91
{{{0.06852235, 0.00388054, }, {0.09236045, 0.00084528, }, },  {{0.06955832, 0.00289679, }, {0.09598522, 0.00096187, }, },  },
92
{{{0.07268085, 0.00302796, }, {0.09323753, 0.00050996, }, },  {{0.06123618, 0.00471601, }, {0.09253389, 0.00046826, }, },  },
93
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, },  {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, },  },
94
},
95
};
96
97
// Shape coefSquareCUsSlowSlower (2 x 5 x 2 x 5 x 2 x 2 x 2): preset (Slow + Slower) x cusize x nspred x qtdepth x mtdepth x sptype x numcoef
98
99
const double EncCu::coefSquareCUsSlowSlower[2][5][2][5][2][2][2] = {
100
{{{{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
101
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
102
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
103
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
104
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
105
},
106
{{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
107
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
108
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
109
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
110
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
111
},
112
},
113
{{{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
114
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
115
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
116
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
117
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
118
},
119
{{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
120
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
121
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
122
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
123
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
124
},
125
},
126
{{{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
127
{{{0.06057349, 0.00447803, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
128
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{0.06948736, 0.00327774, }, {0.06396861, 0.00789923, }, }, },
129
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
130
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
131
},
132
{{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
133
{{{0.05526461, 0.00436703, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
134
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{0.05661641, 0.00546358, }, {0.06680048, 0.00670604, }, }, },
135
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
136
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
137
},
138
},
139
{{{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
140
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{0.06556926, 0.00455697, }, {0.06803347, 0.00365396, }, }, },
141
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
142
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
143
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
144
},
145
{{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
146
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{0.06394328, 0.00403533, }, {0.06688063, 0.00272684, }, }, },
147
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
148
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
149
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
150
},
151
},
152
{{{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{0.05533525, 0.00235007, }, {0.07184475, 0.00086154, }, }, },
153
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
154
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
155
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
156
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
157
},
158
{{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
159
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
160
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
161
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
162
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
163
},
164
},
165
},
166
{{{{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
167
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
168
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
169
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
170
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
171
},
172
{{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
173
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
174
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
175
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
176
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
177
},
178
},
179
{{{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
180
{{{0.04770815, 0.00774597, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
181
{{{0.05509813, 0.00812349, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
182
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{0.04719064, 0.01023887, }, {0.04641434, 0.01279769, }, }, },
183
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
184
},
185
{{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
186
{{{0.02777251, 0.00811989, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
187
{{{0.04848973, 0.00628111, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
188
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{0.04924294, 0.00769725, }, {0.04491680, 0.01120645, }, }, },
189
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
190
},
191
},
192
{{{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
193
{{{0.06856533, 0.00418949, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
194
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{0.07059085, 0.00348835, }, {0.06051544, 0.00821365, }, }, },
195
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
196
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
197
},
198
{{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
199
{{{0.07461087, 0.00229937, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
200
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{0.05893714, 0.00513168, }, {0.05809189, 0.00772994, }, }, },
201
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
202
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
203
},
204
},
205
{{{{{0.06922080, 0.00274816, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
206
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{0.06176300, 0.00374356, }, {0.06085891, 0.00361274, }, }, },
207
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
208
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
209
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
210
},
211
{{{{0.06611224, 0.00256748, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
212
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{0.06364765, 0.00282819, }, {0.05888407, 0.00320947, }, }, },
213
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
214
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
215
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
216
},
217
},
218
{{{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{0.05637817, 0.00184637, }, {0.06937475, 0.00091860, }, }, },
219
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
220
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
221
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
222
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
223
},
224
{{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
225
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
226
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
227
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
228
{{{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, {{-1.00000000, -1.00000000, }, {-1.00000000, -1.00000000, }, }, },
229
},
230
},
231
},
232
};
233
234
// ====================================================================================================================
235
EncCu::EncCu()
236
19.1k
  : m_CtxCache          ( nullptr )
237
19.1k
  , m_globalCtuQpVector ( nullptr )
238
19.1k
  , m_wppMutex          ( nullptr )
239
19.1k
  , m_CABACEstimator    ( nullptr )
240
19.1k
{
241
19.1k
}
242
243
void EncCu::initPic( Picture* pic )
244
4.79k
{
245
4.79k
  m_cInterSearch.setSearchRange( pic->cs->slice, *m_pcEncCfg );
246
247
4.79k
  m_wppMutex = (m_pcEncCfg->m_numThreads > 0 ) ? &pic->wppMutex : nullptr;
248
4.79k
}
249
250
void EncCu::initSlice( const Slice* slice )
251
4.79k
{
252
4.79k
  m_cTrQuant.setLambdas( slice->getLambdas() );
253
4.79k
  m_cRdCost.setLambda( slice->getLambdas()[0], slice->sps->bitDepths );
254
4.79k
}
255
256
void EncCu::setCtuEncRsrc( CABACWriter* cabacEstimator, CtxCache* ctxCache, ReuseUniMv* pReuseUniMv, BlkUniMvInfoBuffer* pBlkUniMvInfoBuffer, AffineProfList* pAffineProfList, IbcBvCand* pCachedBvs )
257
3.84k
{
258
3.84k
  m_CABACEstimator = cabacEstimator;
259
3.84k
  m_CtxCache       = ctxCache;
260
3.84k
  m_cIntraSearch.setCtuEncRsrc( cabacEstimator, ctxCache );
261
3.84k
  m_cInterSearch.setCtuEncRsrc( cabacEstimator, ctxCache, pReuseUniMv, pBlkUniMvInfoBuffer, pAffineProfList, pCachedBvs );
262
3.84k
}
263
264
void EncCu::setUpLambda (Slice& slice, const double dLambda, const int iQP, const bool setSliceLambda, const bool saveUnadjusted)
265
8.64k
{
266
  // store lambda
267
8.64k
  m_cRdCost.setLambda( dLambda, slice.sps->bitDepths );
268
269
  // for RDO
270
  // in RdCost there is only one lambda because the luma and chroma bits are not separated, instead we weight the distortion of chroma.
271
8.64k
  double dLambdas[MAX_NUM_COMP] = { dLambda };
272
25.9k
  for( uint32_t compIdx = 1; compIdx < MAX_NUM_COMP; compIdx++ )
273
17.2k
  {
274
17.2k
    const ComponentID compID = ComponentID( compIdx );
275
17.2k
    int chromaQPOffset       = slice.pps->chromaQpOffset[compID] + slice.sliceChromaQpDelta[ compID ];
276
17.2k
    int qpc = slice.sps->chromaQpMappingTable.getMappedChromaQpValue(compID, iQP) + chromaQPOffset;
277
17.2k
    double tmpWeight         = pow( 2.0, ( iQP - qpc ) / 3.0 );  // takes into account of the chroma qp mapping and chroma qp Offset
278
17.2k
    if( m_pcEncCfg->m_DepQuantEnabled/* && !( m_pcEncCfg->getLFNST() ) */)
279
17.2k
    {
280
17.2k
      tmpWeight *= ( m_pcEncCfg->m_GOPSize >= 8 ? pow( 2.0, 0.1/3.0 ) : pow( 2.0, 0.2/3.0 ) );  // increase chroma weight for dependent quantization (in order to reduce bit rate shift from chroma to luma)
281
17.2k
    }
282
17.2k
    m_cRdCost.setDistortionWeight( compID, tmpWeight );
283
17.2k
    dLambdas[compIdx] = dLambda / tmpWeight;
284
17.2k
  }
285
286
  // for RDOQ
287
8.64k
  m_cTrQuant.setLambdas( dLambdas );
288
289
  // for SAO, ALF
290
8.64k
  if (setSliceLambda)
291
4.79k
  {
292
4.79k
    slice.setLambdas( dLambdas );
293
4.79k
  }
294
8.64k
  if( saveUnadjusted )
295
8.64k
  {
296
8.64k
    m_cRdCost.saveUnadjustedLambda();
297
8.64k
  }
298
8.64k
}
299
300
void EncCu::updateLambda(const Slice& slice, const double ctuLambda, const int ctuQP, const int newQP, const bool saveUnadjusted)
301
3.44k
{
302
3.44k
  const double  corrFactor = pow (2.0, double (newQP - ctuQP) / 3.0);
303
3.44k
  const double  newLambda  = ctuLambda * corrFactor;
304
  
305
3.44k
  double newLambdas[MAX_NUM_COMP] = { newLambda };
306
10.3k
  for( uint32_t compIdx = 1; compIdx < MAX_NUM_COMP; compIdx++ )
307
6.88k
  {
308
6.88k
    const ComponentID compID = ComponentID( compIdx );
309
6.88k
    int chromaQPOffset       = slice.pps->chromaQpOffset[compID] + slice.sliceChromaQpDelta[ compID ];
310
6.88k
    int qpc = slice.sps->chromaQpMappingTable.getMappedChromaQpValue(compID, newQP) + chromaQPOffset;
311
6.88k
    double tmpWeight         = pow( 2.0, ( newQP - qpc ) / 3.0 );  // takes into account of the chroma qp mapping and chroma qp Offset
312
6.88k
    if( m_pcEncCfg->m_DepQuantEnabled/* && !( m_pcEncCfg->getLFNST() ) */)
313
6.88k
    {
314
6.88k
      tmpWeight *= ( m_pcEncCfg->m_GOPSize >= 8 ? pow( 2.0, 0.1/3.0 ) : pow( 2.0, 0.2/3.0 ) );  // increase chroma weight for dependent quantization (in order to reduce bit rate shift from chroma to luma)
315
6.88k
    }
316
6.88k
    m_cRdCost.setDistortionWeight( compID, tmpWeight );
317
6.88k
    newLambdas[compIdx] = newLambda / tmpWeight;
318
6.88k
  }
319
320
3.44k
  m_cTrQuant.setLambdas ( newLambdas);
321
3.44k
  m_cRdCost.setLambda   ( newLambda, slice.sps->bitDepths);
322
323
3.44k
  if (saveUnadjusted)
324
3.44k
  {
325
3.44k
    m_cRdCost.saveUnadjustedLambda(); // TODO hlm: check if this actually improves the overall quality
326
3.44k
  }
327
3.44k
}
328
329
void EncCu::init( const VVEncCfg& encCfg, const SPS& sps, std::vector<int>* const globalCtuQpVector, Ctx* syncPicCtx, RateCtrl* pRateCtrl )
330
19.1k
{
331
19.1k
  DecCu::init( &m_cTrQuant, &m_cIntraSearch, &m_cInterSearch, encCfg.m_internChromaFormat );
332
19.1k
  m_cRdCost.create     ();
333
19.1k
  m_cRdCost.setCostMode( encCfg.m_costMode );
334
19.1k
  if ( encCfg.m_lumaLevelToDeltaQPEnabled == 1 )
335
0
  {
336
0
    m_cRdCost.setChromaFormat( encCfg.m_internChromaFormat );
337
0
    m_cRdCost.initLumaLevelToWeightTable( encCfg.m_internalBitDepth[ CH_L ] );
338
0
    xInitLumaDeltaQpLUT();
339
0
  }
340
341
19.1k
  m_modeCtrl.init     ( encCfg, &m_cRdCost );
342
19.1k
  m_cIntraSearch.init ( encCfg, &m_cTrQuant, &m_cRdCost, &m_SortedPelUnitBufs, m_unitCache );
343
19.1k
  m_cInterSearch.init ( encCfg, &m_cTrQuant, &m_cRdCost, &m_modeCtrl, m_cIntraSearch.getSaveCSBuf() );
344
19.1k
  m_cTrQuant.init     ( nullptr, encCfg.m_RDOQ, encCfg.m_useRDOQTS, false, true, encCfg.m_quantThresholdVal );
345
346
19.1k
  m_syncPicCtx = syncPicCtx;                         ///< context storage for state of contexts at the wavefront/WPP/entropy-coding-sync second CTU of tile-row used for estimation
347
19.1k
  m_pcRateCtrl = pRateCtrl;
348
349
  // Initialise scaling lists: The encoder will only use the SPS scaling lists. The PPS will never be marked present.
350
19.1k
  const int maxLog2TrDynamicRange[ MAX_NUM_CH ] = { sps.getMaxLog2TrDynamicRange(), sps.getMaxLog2TrDynamicRange() };
351
19.1k
  m_cTrQuant.getQuant()->setFlatScalingList( maxLog2TrDynamicRange, sps.bitDepths );
352
353
19.1k
  m_pcEncCfg       = &encCfg;
354
355
19.1k
  m_GeoCostList.init( encCfg.m_maxNumGeoCand );
356
357
19.1k
  unsigned      uiMaxSize    = encCfg.m_CTUSize;
358
19.1k
  ChromaFormat  chromaFormat = encCfg.m_internChromaFormat;
359
360
19.1k
  Area ctuArea = Area( 0, 0, uiMaxSize, uiMaxSize );
361
362
19.1k
  m_mergeItemList.init( encCfg.m_maxMergeRdCandNumTotal, m_pcEncCfg->m_Geo > 1 ? 3 : 1, chromaFormat, uiMaxSize, uiMaxSize );
363
364
249k
  for( int i = 0; i < maxCuDepth; i++ )
365
230k
  {
366
230k
    Area area = Area( 0, 0, uiMaxSize >> ( i >> 1 ), uiMaxSize >> ( ( i + 1 ) >> 1 ) );
367
368
230k
    if( area.width < (1 << MIN_CU_LOG2) || area.height < (1 << MIN_CU_LOG2) )
369
19.1k
    {
370
19.1k
      m_pTempCS[i] = m_pBestCS[i] = nullptr;
371
19.1k
      continue;
372
19.1k
    }
373
374
211k
    m_pTempCS[i] = new CodingStructure( m_unitCache, nullptr );
375
211k
    m_pBestCS[i] = new CodingStructure( m_unitCache, nullptr );
376
377
211k
    m_pTempCS[i]->createForSearch( chromaFormat, area );
378
211k
    m_pBestCS[i]->createForSearch( chromaFormat, area );
379
380
211k
    m_pOrgBuffer[i].create( chromaFormat, area );
381
211k
    m_pRspBuffer[i].create( CHROMA_400, area );
382
211k
  }
383
384
19.1k
  m_pTempCS2 = new CodingStructure( m_unitCache, nullptr );
385
19.1k
  m_pBestCS2 = new CodingStructure( m_unitCache, nullptr );
386
387
19.1k
  m_pTempCS2->createForSearch( chromaFormat, ctuArea );
388
19.1k
  m_pBestCS2->createForSearch( chromaFormat, ctuArea );
389
390
19.1k
  m_cuChromaQpOffsetIdxPlus1 = 0;
391
19.1k
  m_tempQpDiff = 0;
392
19.1k
  m_globalCtuQpVector = globalCtuQpVector;
393
394
19.1k
  m_SortedPelUnitBufs.create( chromaFormat, uiMaxSize, uiMaxSize );
395
396
287k
  for( uint8_t i = 0; i < MAX_TMP_BUFS; i++)
397
268k
  {
398
268k
    m_aTmpStorageLCU[i].create(chromaFormat, Area(0, 0, uiMaxSize, uiMaxSize));
399
268k
  }
400
134k
  for (unsigned ui = 0; ui < MRG_MAX_NUM_CANDS; ui++)
401
115k
  {
402
115k
    m_acMergeTmpBuffer[ui].create(chromaFormat, Area(0, 0, uiMaxSize, uiMaxSize));
403
115k
  }
404
405
19.1k
  const unsigned maxDepth = 2 * MAX_CU_SIZE_IDX;
406
19.1k
  m_CtxBuffer.resize( maxDepth );
407
19.1k
  m_CurrCtx = 0;
408
19.1k
  if( encCfg.m_EDO )
409
19.1k
    m_dbBuffer.create( chromaFormat, Area( 0, 0, uiMaxSize, uiMaxSize ), 0, 8 );
410
411
19.1k
  m_MergeSimpleFlag = 0;
412
19.1k
  m_tileIdx = 0;
413
19.1k
}
414
415
416
void EncCu::destroy()
417
19.1k
{
418
249k
  for( int i = 0; i < maxCuDepth; i++ )
419
230k
  {
420
230k
    if( m_pTempCS[i] )
421
211k
    {
422
211k
      m_pTempCS[i]->destroy();
423
211k
      delete m_pTempCS[i]; m_pTempCS[i] = nullptr;
424
211k
    }
425
426
230k
    if( m_pBestCS[i] )
427
211k
    {
428
211k
      m_pBestCS[i]->destroy();
429
211k
      delete m_pBestCS[i]; m_pBestCS[i] = nullptr;
430
211k
    }
431
432
230k
    m_pOrgBuffer[i].destroy();
433
230k
    m_pRspBuffer[i].destroy();
434
230k
  }
435
436
19.1k
  m_pTempCS2->destroy();
437
19.1k
  m_pBestCS2->destroy();
438
439
19.1k
  delete m_pTempCS2; m_pTempCS2 = nullptr;
440
19.1k
  delete m_pBestCS2; m_pBestCS2 = nullptr;
441
442
19.1k
  m_SortedPelUnitBufs.destroy();
443
444
287k
  for( uint8_t i = 0; i < MAX_TMP_BUFS; i++)
445
268k
  {
446
268k
    m_aTmpStorageLCU[i].destroy();
447
268k
  }
448
134k
  for (unsigned ui = 0; ui < MRG_MAX_NUM_CANDS; ui++)
449
115k
  {
450
115k
    m_acMergeTmpBuffer[ui].destroy();
451
115k
  }
452
453
454
19.1k
  m_dbBuffer.destroy();
455
19.1k
}
456
457
458
EncCu::~EncCu()
459
19.1k
{
460
19.1k
  destroy();
461
19.1k
}
462
463
// ====================================================================================================================
464
// Public member functions
465
// ====================================================================================================================
466
467
void EncCu::encodeCtu( Picture* pic, int (&prevQP)[MAX_NUM_CH], uint32_t ctuXPosInCtus, uint32_t ctuYPosInCtus )
468
3.84k
{
469
3.84k
  CodingStructure&     cs          = *pic->cs;
470
3.84k
  Slice*               slice       = cs.slice;
471
3.84k
  const PreCalcValues& pcv         = *cs.pcv;
472
473
#if ENABLE_MEASURE_SEARCH_SPACE
474
  if( ctuXPosInCtus == 0 && ctuYPosInCtus == 0 )
475
  {
476
    g_searchSpaceAcc.picW = pic->lwidth();
477
    g_searchSpaceAcc.picH = pic->lheight();
478
    g_searchSpaceAcc.addSlice( slice->isIntra(), slice->depth );
479
  }
480
481
#endif
482
3.84k
  const int ctuRsAddr                 = ctuYPosInCtus * pcv.widthInCtus + ctuXPosInCtus;
483
484
3.84k
  const Position pos (ctuXPosInCtus * pcv.maxCUSize, ctuYPosInCtus * pcv.maxCUSize);
485
3.84k
  const UnitArea ctuArea( cs.area.chromaFormat, Area( pos.x, pos.y, pcv.maxCUSize, pcv.maxCUSize ) );
486
3.84k
  DTRACE_UPDATE( g_trace_ctx, std::make_pair( "ctu", ctuRsAddr ) );
487
488
3.84k
  const int tileXPosInCtus = cs.pps->tileColBd[cs.pps->ctuToTileCol[ctuXPosInCtus]];
489
3.84k
  const int tileYPosInCtus = cs.pps->tileRowBd[cs.pps->ctuToTileRow[ctuYPosInCtus]];
490
491
3.84k
  if( ( cs.slice->sliceType != VVENC_I_SLICE || cs.sps->IBC ) && ctuXPosInCtus == tileXPosInCtus )
492
2.14k
  {
493
2.14k
    const int tileRowId = cs.pps->getTileLineId( ctuXPosInCtus, ctuYPosInCtus );
494
2.14k
    cs.motionLutBuf[tileRowId].lut.resize( 0 );
495
2.14k
    cs.motionLutBuf[tileRowId].lutIbc.resize( 0 );
496
2.14k
  }
497
498
3.84k
  if( ( m_pcEncCfg->m_ensureWppBitEqual || m_pcEncCfg->m_entropyCodingSyncEnabled ) && ctuXPosInCtus == tileXPosInCtus )
499
2.14k
  {
500
2.14k
    m_CABACEstimator->initCtxModels( *slice );
501
502
2.14k
    if( m_pcEncCfg->m_entropyCodingSyncEnabled && ( ctuYPosInCtus > tileYPosInCtus ) )
503
0
    {
504
0
      m_CABACEstimator->getCtx() = m_syncPicCtx[slice->pps->getTileLineId( ctuXPosInCtus, ctuYPosInCtus - 1 )];
505
0
    }
506
507
2.14k
    prevQP[CH_L] = prevQP[CH_C] = slice->sliceQp; // hlm: call CU::predictQP() here!
508
2.14k
  }
509
1.70k
  else if( ctuXPosInCtus == tileXPosInCtus && ctuYPosInCtus == tileYPosInCtus )
510
0
  {
511
0
    m_CABACEstimator->initCtxModels( *slice );
512
0
    prevQP[CH_L] = prevQP[CH_C] = slice->sliceQp; // hlm: call CU::predictQP() here!
513
0
  }
514
515
3.84k
  xCompressCtu( cs, ctuArea, ctuRsAddr, prevQP );
516
517
3.84k
  m_CABACEstimator->resetBits();
518
3.84k
  m_CABACEstimator->coding_tree_unit( cs, ctuArea, prevQP, ctuRsAddr, true, true );
519
520
  // Store probabilities of second CTU in line into buffer - used only if wavefront-parallel-processing is enabled.
521
3.84k
  if( ctuXPosInCtus == tileXPosInCtus && m_pcEncCfg->m_entropyCodingSyncEnabled )
522
0
  {
523
0
    m_syncPicCtx[slice->pps->getTileLineId( ctuXPosInCtus, ctuYPosInCtus )] = m_CABACEstimator->getCtx();
524
0
  }
525
526
3.84k
  DTRACE_AREA_CRC( g_trace_ctx, D_CRC, cs, ctuArea );
527
3.84k
}
528
529
// ====================================================================================================================
530
// Protected member functions
531
// ====================================================================================================================
532
533
void EncCu::xCompressCtu( CodingStructure& cs, const UnitArea& area, const unsigned ctuRsAddr, const int prevQP[] )
534
3.84k
{
535
3.84k
  m_tileIdx = cs.pps->getTileIdx( area.lumaPos() );
536
537
3.84k
  m_modeCtrl.initCTUEncoding( *cs.slice, m_tileIdx );
538
539
  // init the partitioning manager
540
3.84k
  Partitioner *partitioner = &m_partitioner;
541
3.84k
  partitioner->initCtu( area, CH_L, *cs.slice );
542
  
543
3.84k
  const Position& lumaPos = area.lumaPos();
544
3.84k
  const bool leftSameTile  = lumaPos.x == 0 || m_tileIdx == cs.pps->getTileIdx( lumaPos.offset(-1, 0) );
545
3.84k
  const bool aboveSameTile = lumaPos.y == 0 || m_tileIdx == cs.pps->getTileIdx( lumaPos.offset( 0,-1) );
546
3.84k
  m_EDO = (!m_pcEncCfg->m_tileParallelCtuEnc || (leftSameTile && aboveSameTile)) ? m_pcEncCfg->m_EDO : 0;
547
  
548
3.84k
  if( m_pcEncCfg->m_IBCMode )
549
3.84k
  {
550
3.84k
    m_cInterSearch.resetCtuRecordIBC();
551
3.84k
  }
552
553
  // init current context pointer
554
3.84k
  m_CurrCtx = m_CtxBuffer.data();
555
556
3.84k
  PelStorage* orgBuffer = &m_pOrgBuffer[0];
557
3.84k
  PelStorage* rspBuffer = &m_pRspBuffer[0];
558
3.84k
  CodingStructure *tempCS =  m_pTempCS [0];
559
3.84k
  CodingStructure *bestCS =  m_pBestCS [0];
560
3.84k
  cs.initSubStructure( *tempCS, partitioner->chType, partitioner->currArea(), false, orgBuffer, rspBuffer );
561
3.84k
  cs.initSubStructure( *bestCS, partitioner->chType, partitioner->currArea(), false, orgBuffer, rspBuffer );
562
3.84k
  PROFILER_SCOPE_AND_STAGE_EXT( 1, _TPROF, P_COMPRESS_CU, tempCS, CH_L );
563
564
  // copy the relevant area
565
3.84k
  UnitArea clippedArea = clipArea( partitioner->currArea(), cs.area );
566
3.84k
  CPelUnitBuf org = cs.picture->getFilteredOrigBuffer().valid() ? cs.picture->getFiltOrigBuf( clippedArea ) : cs.picture->getOrigBuf( clippedArea );
567
3.84k
  tempCS->getOrgBuf( clippedArea ).copyFrom( org );
568
3.84k
  tempCS->currQP[CH_L] = bestCS->currQP[CH_L] =
569
3.84k
  tempCS->baseQP       = bestCS->baseQP       = cs.slice->sliceQp;
570
3.84k
  tempCS->prevQP[CH_L] = bestCS->prevQP[CH_L] = prevQP[CH_L];
571
572
3.84k
  xCompressCU( tempCS, bestCS, *partitioner );
573
  // all signals were already copied during compression if the CTU was split - at this point only the structures are copied to the top level CS
574
  
575
  // Ensure that a coding was found
576
  // Selected mode's RD-cost must be not MAX_DOUBLE.
577
3.84k
  CHECK( bestCS->cus.empty()                                   , "No possible encoding found" );
578
3.84k
  CHECK( bestCS->cus[0]->predMode == NUMBER_OF_PREDICTION_MODES, "No possible encoding found" );
579
3.84k
  CHECK( bestCS->cost             == MAX_DOUBLE                , "No possible encoding found" );
580
581
3.84k
  if ( m_wppMutex ) m_wppMutex->lock();
582
583
3.84k
  cs.useSubStructure( *bestCS, partitioner->chType, TREE_D, CS::getArea( *bestCS, area, partitioner->chType, partitioner->treeType ) );
584
585
3.84k
  if ( m_wppMutex ) m_wppMutex->unlock();
586
587
3.84k
  if( CS::isDualITree( cs ) && isChromaEnabled( cs.pcv->chrFormat ) )
588
3.84k
  {
589
3.84k
    m_CABACEstimator->getCtx() = m_CurrCtx->start;
590
591
3.84k
    partitioner->initCtu( area, CH_C, *cs.slice );
592
593
3.84k
    cs.initSubStructure( *tempCS, partitioner->chType, partitioner->currArea(), false, orgBuffer, rspBuffer );
594
3.84k
    cs.initSubStructure( *bestCS, partitioner->chType, partitioner->currArea(), false, orgBuffer, rspBuffer );
595
3.84k
    tempCS->currQP[CH_C] = bestCS->currQP[CH_C] =
596
3.84k
    tempCS->baseQP       = bestCS->baseQP       = cs.slice->sliceQp;
597
3.84k
    tempCS->prevQP[CH_C] = bestCS->prevQP[CH_C] = prevQP[CH_C];
598
599
3.84k
    xCompressCU( tempCS, bestCS, *partitioner );
600
    
601
    // Ensure that a coding was found
602
    // Selected mode's RD-cost must be not MAX_DOUBLE.
603
3.84k
    CHECK( bestCS->cus.empty()                                   , "No possible encoding found" );
604
3.84k
    CHECK( bestCS->cus[0]->predMode == NUMBER_OF_PREDICTION_MODES, "No possible encoding found" );
605
3.84k
    CHECK( bestCS->cost             == MAX_DOUBLE                , "No possible encoding found" );
606
607
3.84k
    if ( m_wppMutex ) m_wppMutex->lock();
608
609
3.84k
    cs.useSubStructure( *bestCS, partitioner->chType, TREE_D, CS::getArea( *bestCS, area, partitioner->chType, partitioner->treeType ) );
610
611
3.84k
    if ( m_wppMutex ) m_wppMutex->unlock();
612
3.84k
  }
613
614
  // reset context states and uninit context pointer
615
3.84k
  m_CABACEstimator->getCtx() = m_CurrCtx->start;
616
3.84k
  m_CurrCtx                  = 0;
617
3.84k
}
618
619
620
621
bool EncCu::xCheckBestMode( CodingStructure *&tempCS, CodingStructure *&bestCS, Partitioner &partitioner, const EncTestMode& encTestMode, const bool useEDO )
622
281k
{
623
281k
  bool bestCSUpdated = false;
624
625
281k
  if( !tempCS->cus.empty() )
626
171k
  {
627
171k
    if( tempCS->cus.size() == 1 )
628
129k
    {
629
129k
      const CodingUnit& cu = *tempCS->cus.front();
630
129k
      CHECK( cu.skip && !cu.mergeFlag, "Skip flag without a merge flag is not allowed!" );
631
129k
    }
632
633
171k
    DTRACE_BEST_MODE( tempCS, bestCS, m_cRdCost.getLambda(true), useEDO );
634
635
171k
    if( m_modeCtrl.useModeResult( encTestMode, tempCS, partitioner, useEDO ) )
636
154k
    {
637
154k
      std::swap( tempCS, bestCS );
638
      // store temp best CI for next CU coding
639
154k
      m_CurrCtx->best = m_CABACEstimator->getCtx();
640
154k
      bestCSUpdated = true;
641
154k
    }
642
171k
  }
643
644
  // reset context states
645
281k
  m_CABACEstimator->getCtx() = m_CurrCtx->start;
646
281k
  return bestCSUpdated;
647
648
281k
}
649
650
void xCheckFastCuChromaSplitting( CodingStructure*& tempCS, CodingStructure*& bestCS, Partitioner&  partitioner, ComprCUCtx& cuECtx )
651
0
{
652
0
  const uint32_t uiLPelX = tempCS->area.Cb().lumaPos().x;
653
0
  const uint32_t uiTPelY = tempCS->area.Cb().lumaPos().y;
654
655
0
  int lumaw = 0, lumah = 0;
656
0
  bool splitver      = true;
657
0
  bool splithor      = true;
658
0
  bool qtSplitChroma = true;
659
660
0
  if( partitioner.isSepTree( *tempCS ) && isChroma( partitioner.chType ) )
661
0
  {
662
0
    Position lumaRefPos( uiLPelX, uiTPelY );
663
0
    CodingUnit* colLumaCu = bestCS->lumaCS->getCU( lumaRefPos, CH_L, TREE_D );
664
665
0
    if( colLumaCu )
666
0
    {
667
0
      lumah = colLumaCu->Y().height;
668
0
      lumaw = colLumaCu->Y().width;
669
0
    }
670
0
  }
671
0
  else
672
0
  {
673
0
    return;
674
0
  }
675
676
0
  if( partitioner.getImplicitSplit( *tempCS ) != CU_DONT_SPLIT ) return;
677
678
0
  const CPelBuf orgCb = tempCS->getOrgBuf( COMP_Cb );
679
0
  const CPelBuf orgCr = tempCS->getOrgBuf( COMP_Cr );
680
681
0
  int th1 = FCBP_TH1;
682
683
0
  if( ( lumaw >> getChannelTypeScaleX( CH_C, tempCS->area.chromaFormat ) ) == orgCb.width )
684
0
  {
685
0
    if( ( bestCS->cost < ( th1*orgCb.width*orgCb.height ) ) )
686
0
    {
687
0
      splitver      = false;
688
0
      qtSplitChroma = false;
689
0
    }
690
0
  }
691
692
0
  if( ( lumah >> getChannelTypeScaleY( CH_C, tempCS->area.chromaFormat ) ) == orgCb.height )
693
0
  {
694
0
    if( ( bestCS->cost < ( th1*orgCb.width*orgCb.height ) ) )
695
0
    {
696
0
      splithor      = false;
697
0
      qtSplitChroma = false;
698
0
    }
699
0
  }
700
701
0
  cuECtx.doHorChromaSplit = splithor;
702
0
  cuECtx.doVerChromaSplit = splitver;
703
0
  cuECtx.doQtChromaSplit  = qtSplitChroma;
704
705
0
  if( orgCb.width == orgCb.height )
706
0
  {
707
0
    int varh_cb, varv_cb;
708
0
    int varh_cr, varv_cr;
709
710
0
    orgCb.calcVarianceSplit( orgCb, orgCb.width, varh_cb, varv_cb );
711
0
    orgCr.calcVarianceSplit( orgCr, orgCr.width, varh_cr, varv_cr );
712
713
0
    if( ( varh_cr*FCBP_TH2 < varv_cr * 100 ) && ( varh_cb*FCBP_TH2 < varv_cb * 100 ) )
714
0
    {
715
0
      cuECtx.doVerChromaSplit = false;
716
0
    }
717
0
    else if( ( varv_cr*FCBP_TH2 < varh_cr * 100 ) && ( varv_cb*FCBP_TH2 < varh_cb * 100 ) )
718
0
    {
719
0
      cuECtx.doHorChromaSplit = false;
720
0
    }
721
0
  }
722
0
}
723
724
void EncCu::xCompressCU( CodingStructure*& tempCS, CodingStructure*& bestCS, Partitioner& partitioner )
725
129k
{
726
129k
  const Area& lumaArea = tempCS->area.Y();
727
728
129k
  Slice&   slice      = *tempCS->slice;
729
129k
  const PPS &pps      = *tempCS->pps;
730
129k
  const SPS &sps      = *tempCS->sps;
731
129k
  const uint32_t uiLPelX  = tempCS->area.Y().lumaPos().x;
732
129k
  const uint32_t uiTPelY  = tempCS->area.Y().lumaPos().y;
733
129k
  const bool isBimEnabled = (m_pcEncCfg->m_blockImportanceMapping && !bestCS->picture->m_picShared->m_ctuBimQpOffset.empty());
734
735
129k
  m_modeCtrl.initBlk( tempCS->area, slice.pic->poc );
736
129k
  m_CABACEstimator->determineNeighborCus( *tempCS, partitioner.currArea(), partitioner.chType, partitioner.treeType );
737
738
129k
  if ((m_pcEncCfg->m_usePerceptQPA || isBimEnabled || m_pcEncCfg->m_lumaLevelToDeltaQPEnabled == 1) && ( pps.useDQP || m_pcEncCfg->m_maxDeltaQP == 0 ) && isLuma (partitioner.chType) && partitioner.currQgEnable())
739
7.29k
  {
740
7.29k
    const PreCalcValues &pcv = *pps.pcv;
741
7.29k
    Picture* const pic = bestCS->picture;
742
7.29k
    const uint32_t ctuRsAddr = getCtuAddr (partitioner.currQgPos, pcv);
743
744
7.29k
    if (partitioner.currSubdiv == 0) // CTU-level QP adaptation
745
3.84k
    {
746
3.84k
      if (m_pcEncCfg->m_usePerceptQPA)
747
3.84k
      {
748
3.84k
        if (m_pcEncCfg->m_internalUsePerceptQPATempFiltISlice == 2)
749
0
        {
750
0
          m_tempQpDiff = pic->ctuAdaptedQP[ctuRsAddr] - BitAllocation::applyQPAdaptationSubCtu (&slice, m_pcEncCfg, lumaArea, m_pcRateCtrl->getMinNoiseLevels());
751
0
        }
752
753
3.84k
        int clippedMaxDQP = 0;
754
755
3.84k
        if ((!slice.isIntra()) && (pcv.maxCUSize > 64) && // sub-CTU QPA behavior - Museum fix
756
0
            (uiLPelX + (pcv.maxCUSize >> 1) < (m_pcEncCfg->m_PadSourceWidth)) &&
757
0
            (uiTPelY + (pcv.maxCUSize >> 1) < (m_pcEncCfg->m_PadSourceHeight)))
758
0
        {
759
0
          const uint32_t h = lumaArea.height >> 1;
760
0
          const uint32_t w = lumaArea.width  >> 1;
761
0
          const int adQPTL = BitAllocation::applyQPAdaptationSubCtu (&slice, m_pcEncCfg, Area (uiLPelX + 0, uiTPelY + 0, w, h), m_pcRateCtrl->getMinNoiseLevels());
762
0
          const int adQPTR = BitAllocation::applyQPAdaptationSubCtu (&slice, m_pcEncCfg, Area (uiLPelX + w, uiTPelY + 0, w, h), m_pcRateCtrl->getMinNoiseLevels());
763
0
          const int adQPBL = BitAllocation::applyQPAdaptationSubCtu (&slice, m_pcEncCfg, Area (uiLPelX + 0, uiTPelY + h, w, h), m_pcRateCtrl->getMinNoiseLevels());
764
0
          const int adQPBR = BitAllocation::applyQPAdaptationSubCtu (&slice, m_pcEncCfg, Area (uiLPelX + w, uiTPelY + h, w, h), m_pcRateCtrl->getMinNoiseLevels());
765
766
0
          tempCS->currQP[partitioner.chType] = tempCS->baseQP =
767
0
          bestCS->currQP[partitioner.chType] = bestCS->baseQP = std::min (std::min (adQPTL, adQPTR), std::min (adQPBL, adQPBR));
768
769
0
          if (m_pcEncCfg->m_internalUsePerceptQPATempFiltISlice == 2)
770
0
          {
771
0
            if ((m_globalCtuQpVector->size() > ctuRsAddr) && (slice.TLayer == 0) && // last CTU row of non-Intra key-frame
772
0
                (m_pcEncCfg->m_IntraPeriod == 2 * m_pcEncCfg->m_GOPSize) && (ctuRsAddr >= pcv.widthInCtus) && (uiTPelY + pcv.maxCUSize > m_pcEncCfg->m_PadSourceHeight))
773
0
            {
774
0
              m_globalCtuQpVector->at (ctuRsAddr) = m_globalCtuQpVector->at (ctuRsAddr - pcv.widthInCtus); // copy the pumping reducing QP offset from the top CTU neighbor
775
0
              tempCS->currQP[partitioner.chType] = tempCS->baseQP =
776
0
              bestCS->currQP[partitioner.chType] = bestCS->baseQP = tempCS->baseQP - m_globalCtuQpVector->at (ctuRsAddr);
777
0
            }
778
0
            tempCS->currQP[partitioner.chType] = tempCS->baseQP =
779
0
            bestCS->currQP[partitioner.chType] = bestCS->baseQP = Clip3 (0, MAX_QP, tempCS->baseQP + m_tempQpDiff);
780
0
          }
781
0
        }
782
3.84k
        else
783
3.84k
        {
784
3.84k
          tempCS->currQP[partitioner.chType] = tempCS->baseQP =
785
3.84k
          bestCS->currQP[partitioner.chType] = bestCS->baseQP = pic->ctuAdaptedQP[ctuRsAddr];
786
3.84k
        }
787
788
3.84k
        int deltaQP       = slice.sliceQp - tempCS->baseQP;
789
3.84k
        if( abs( deltaQP ) > m_pcEncCfg->m_maxDeltaQP )
790
0
        {
791
0
          clippedMaxDQP = deltaQP > 0 ? slice.sliceQp - m_pcEncCfg->m_maxDeltaQP : slice.sliceQp + m_pcEncCfg->m_maxDeltaQP;
792
0
          tempCS->currQP[partitioner.chType] = tempCS->baseQP =
793
0
          bestCS->currQP[partitioner.chType] = bestCS->baseQP = clippedMaxDQP;
794
0
        }
795
796
        //the ctuAdaptedQP correspones with the ctuQpaLambda
797
3.84k
        setUpLambda (slice, pic->ctuQpaLambda[ctuRsAddr], pic->ctuAdaptedQP[ctuRsAddr], false, true);
798
3.84k
      }
799
0
      else // isBimEnabled without QPA
800
0
      {
801
0
        const int baseQp  = tempCS->baseQP;
802
0
        int adaptedLumaQP = 0;
803
0
        int lumaQPoffset  = 0;
804
        
805
0
        if( m_pcEncCfg->m_lumaLevelToDeltaQPEnabled == 1 )
806
0
        {
807
0
          uint32_t meanLuma = 0;
808
0
          if( ( uiLPelX + lumaArea.width > pps.picWidthInLumaSamples ) || ( uiTPelY + lumaArea.height > pps.picHeightInLumaSamples ) )
809
0
          {
810
0
            int wdth = ( uiLPelX + lumaArea.width > pps.picWidthInLumaSamples )   ? pps.picWidthInLumaSamples - uiLPelX  : lumaArea.width;
811
0
            int hght = ( uiTPelY + lumaArea.height > pps.picHeightInLumaSamples ) ? pps.picHeightInLumaSamples - uiTPelY : lumaArea.height;
812
            
813
0
            int sum = 0;
814
0
            for( int y = 0; y < hght; y++ )
815
0
            {
816
0
              for( int x = 0; x < wdth; x++ )
817
0
              {
818
0
                sum += tempCS->getOrgBuf().Y().at( x, y );
819
0
              }
820
0
            }
821
0
            double avg = (double) sum / ( wdth*hght );
822
0
            meanLuma = (uint32_t)( avg + 0.5 );
823
0
          }
824
0
          else
825
0
          {
826
0
            meanLuma = tempCS->getOrgBuf().Y().getAvg();
827
0
          }
828
829
0
          int lumaBD     = m_pcEncCfg->m_internalBitDepth[ CH_L ];
830
0
          int lumaIdxOrg = Clip3<int>( 0, int( 1 << lumaBD ) - 1, meanLuma );
831
0
          int lumaIdx    = lumaBD < 10 ? lumaIdxOrg << (10 - lumaBD) : lumaBD > 10 ? lumaIdxOrg >> (lumaBD - 10) : lumaIdxOrg;
832
0
          lumaQPoffset   = m_lumaLevelToDeltaQPLUT[lumaIdx];
833
0
          adaptedLumaQP  = Clip3( -sps.qpBDOffset[CH_L], MAX_QP, baseQp - lumaQPoffset );
834
0
        }
835
        
836
0
        if( isBimEnabled )
837
0
        {
838
0
          const unsigned bimQpSize = (unsigned) bestCS->picture->m_picShared->m_ctuBimQpOffset.size();
839
0
          uint32_t ctuAddr         = ctuRsAddr;
840
841
0
          double avgBimOffset = 0.0;
842
843
0
          if (bimQpSize != pcv.sizeInCtus) // re-calculate correct address of BIM CTU QP offset
844
0
          {
845
0
            const unsigned bimCtuSize  = m_pcEncCfg->m_bimCtuSize;
846
0
            const unsigned bimCtuWidth = (pcv.lumaWidth + bimCtuSize - 1) / bimCtuSize;
847
848
0
            ctuAddr = getCtuAddrFromCtuSize (partitioner.currQgPos, Log2 (bimCtuSize), bimCtuWidth);
849
0
            CHECK (ctuAddr >= bimQpSize, "ctuAddr exceeds size of m_ctuBimQpOffset");
850
851
0
            int maxY = m_pcEncCfg->m_CTUSize / bimCtuSize;
852
0
            int maxX = m_pcEncCfg->m_CTUSize / bimCtuSize;
853
0
            if( ( partitioner.currQgPos.y + maxY * bimCtuSize ) > pcv.lumaHeight )
854
0
            {
855
0
              int reduceHeight = ( partitioner.currQgPos.y + ( maxY - 1 ) * bimCtuSize ) - pcv.lumaHeight;
856
0
              int reduceY = ( reduceHeight + bimCtuSize ) / bimCtuSize;
857
0
              maxY = maxY - reduceY;
858
0
            }
859
0
            if( ( partitioner.currQgPos.x + maxX * bimCtuSize ) > pcv.lumaWidth )
860
0
            {
861
0
              int reduceWidth = ( partitioner.currQgPos.x + ( maxX - 1 ) * bimCtuSize ) - pcv.lumaWidth;
862
0
              int reduceX = ( reduceWidth + bimCtuSize ) / bimCtuSize;
863
0
              maxX = maxX - reduceX;
864
0
            }
865
0
            for (int y = 0; y < maxY * bimCtuSize; y = y + bimCtuSize)
866
0
            {
867
0
              for (int x = 0; x < maxX * bimCtuSize; x = x + bimCtuSize)
868
0
              {
869
0
                int theBlockId = ctuAddr + (y / bimCtuSize) * ((pcv.lumaWidth + bimCtuSize - 1) / bimCtuSize) + (x / bimCtuSize);
870
0
                avgBimOffset += pic->m_picShared->m_ctuBimQpOffset[theBlockId];
871
0
              }
872
0
            }
873
0
            if (maxY * maxX > 0)
874
0
            {
875
0
              avgBimOffset = avgBimOffset / (double)(maxY * maxX);
876
0
            }
877
0
            else
878
0
            {
879
0
              avgBimOffset = 0.0;
880
0
            }
881
0
          }
882
0
          else
883
0
          {
884
0
            avgBimOffset = pic->m_picShared->m_ctuBimQpOffset[ctuAddr];
885
0
          }
886
0
          int intAvgBimOffset = avgBimOffset >= 0 ? (int)(0.5 + avgBimOffset) : (int)(-0.5 + avgBimOffset);
887
0
          adaptedLumaQP       = Clip3 (-sps.qpBDOffset[CH_L], MAX_QP, baseQp - lumaQPoffset + intAvgBimOffset );
888
0
        }
889
        
890
0
        tempCS->currQP[partitioner.chType] = tempCS->baseQP =
891
0
        bestCS->currQP[partitioner.chType] = bestCS->baseQP = adaptedLumaQP;
892
        
893
0
        int deltaQP         = slice.sliceQp - adaptedLumaQP;
894
0
        if( abs( deltaQP ) > m_pcEncCfg->m_maxDeltaQP )
895
0
        {
896
0
          int clippedMaxDQP = deltaQP > 0 ? slice.sliceQp - m_pcEncCfg->m_maxDeltaQP : slice.sliceQp + m_pcEncCfg->m_maxDeltaQP;
897
0
          tempCS->currQP[partitioner.chType] = tempCS->baseQP =
898
0
          bestCS->currQP[partitioner.chType] = bestCS->baseQP = clippedMaxDQP;
899
0
        }
900
901
0
        updateLambda (slice, slice.getLambdas()[0], baseQp, adaptedLumaQP, true);
902
0
      }
903
3.84k
    }
904
3.44k
    else if (m_pcEncCfg->m_usePerceptQPA && slice.isIntra()) // currSubdiv 2 - use sub-CTU QPA
905
3.44k
    {
906
3.44k
      CHECK ((partitioner.currArea().lwidth() >= pcv.maxCUSize) || (partitioner.currArea().lheight() >= pcv.maxCUSize), "sub-CTU delta-QP error");
907
3.44k
      tempCS->currQP[partitioner.chType] = tempCS->baseQP = BitAllocation::applyQPAdaptationSubCtu (&slice, m_pcEncCfg, lumaArea, m_pcRateCtrl->getMinNoiseLevels());
908
909
3.44k
      if (m_pcEncCfg->m_internalUsePerceptQPATempFiltISlice == 2)
910
0
      {
911
0
        tempCS->currQP[partitioner.chType] = tempCS->baseQP = Clip3 (0, MAX_QP, tempCS->baseQP + m_tempQpDiff);
912
0
      }
913
914
3.44k
      int backupBaseQP  = tempCS->baseQP;
915
3.44k
      int deltaQP       = slice.sliceQp - tempCS->baseQP;
916
3.44k
      int clippedMaxDQP = 0;
917
3.44k
      if( abs( deltaQP ) > m_pcEncCfg->m_maxDeltaQP )
918
0
      {
919
0
        clippedMaxDQP = deltaQP > 0 ? slice.sliceQp - m_pcEncCfg->m_maxDeltaQP : slice.sliceQp + m_pcEncCfg->m_maxDeltaQP;
920
0
        tempCS->currQP[partitioner.chType] = tempCS->baseQP = clippedMaxDQP;
921
0
      }
922
3.44k
      updateLambda( slice, pic->ctuQpaLambda[ctuRsAddr], pic->ctuAdaptedQP[ctuRsAddr], backupBaseQP, true );
923
3.44k
    }
924
7.29k
  }
925
926
129k
  if (partitioner.currQtDepth == 0)
927
7.69k
  {
928
7.69k
    m_MergeSimpleFlag = 0;
929
7.69k
  }
930
129k
  m_modeCtrl.initCULevel( partitioner, *tempCS, m_MergeSimpleFlag );
931
129k
  m_sbtCostSave[0] = m_sbtCostSave[1] = MAX_DOUBLE;
932
933
129k
  m_CurrCtx->start = m_CABACEstimator->getCtx();
934
935
129k
  m_cuChromaQpOffsetIdxPlus1 = 0;
936
937
129k
  if( slice.chromaQpAdjEnabled && partitioner.currQgChromaEnable() )
938
0
  {
939
    // TODO M0133 : double check encoder decisions with respect to chroma QG detection and actual encode
940
0
    int cuChromaQpOffsetSubdiv = slice.isIntra() ? slice.picHeader->cuChromaQpOffsetSubdivIntra : slice.picHeader->cuChromaQpOffsetSubdivInter;
941
0
    int lgMinCuSize = sps.log2MinCodingBlockSize +
942
0
      std::max<int>(0, floorLog2(sps.CTUSize) - sps.log2MinCodingBlockSize - int((cuChromaQpOffsetSubdiv + 1) / 2));
943
0
    m_cuChromaQpOffsetIdxPlus1 = ( ( uiLPelX >> lgMinCuSize ) + ( uiTPelY >> lgMinCuSize ) ) % ( pps.chromaQpOffsetListLen + 1 );
944
0
  }
945
946
129k
  DTRACE_UPDATE( g_trace_ctx, std::make_pair( "cux", uiLPelX ) );
947
129k
  DTRACE_UPDATE( g_trace_ctx, std::make_pair( "cuy", uiTPelY ) );
948
129k
  DTRACE_UPDATE( g_trace_ctx, std::make_pair( "cuw", tempCS->area.lwidth() ) );
949
129k
  DTRACE_UPDATE( g_trace_ctx, std::make_pair( "cuh", tempCS->area.lheight() ) );
950
129k
  DTRACE( g_trace_ctx, D_COMMON, "@(%4d,%4d) [%2dx%2d]\n", tempCS->area.lx(), tempCS->area.ly(), tempCS->area.lwidth(), tempCS->area.lheight() );
951
952
129k
  if( tempCS->slice->checkLDC )
953
129k
  {
954
129k
    m_bestBcwCost[0] = m_bestBcwCost[1] = std::numeric_limits<double>::max();
955
129k
    m_bestBcwIdx[0] = m_bestBcwIdx[1] = -1;
956
129k
  }
957
958
129k
  m_cInterSearch.resetSavedAffineMotion();
959
129k
  {
960
129k
    const ComprCUCtx &cuECtx      = *m_modeCtrl.comprCUCtx;
961
129k
    const CodingStructure& cs     = *tempCS;
962
129k
    const PartSplit implicitSplit = partitioner.getImplicitSplit( cs );
963
129k
    const bool isBoundary         = implicitSplit != CU_DONT_SPLIT;
964
129k
    const bool lossless           = false;
965
129k
    int qp                        = cs.baseQP;
966
967
#if ENABLE_MEASURE_SEARCH_SPACE
968
    if( !isBoundary )
969
    {
970
      g_searchSpaceAcc.addPartition( partitioner.currArea(), partitioner.isSepTree( *tempCS ) ? partitioner.chType : MAX_NUM_CH );
971
    }
972
973
#endif
974
129k
    if( ! isBoundary )
975
81.7k
    {
976
81.7k
      if (pps.useDQP && partitioner.isSepTree (*tempCS) && isChroma (partitioner.chType))
977
56.6k
      {
978
56.6k
        const ChromaFormat chromaFm = tempCS->area.chromaFormat;
979
56.6k
        const Position chromaCentral (tempCS->area.Cb().chromaPos().offset (tempCS->area.Cb().chromaSize().width >> 1, tempCS->area.Cb().chromaSize().height >> 1));
980
56.6k
        const Position lumaRefPos (chromaCentral.x << getChannelTypeScaleX (CH_C, chromaFm), chromaCentral.y << getChannelTypeScaleY (CH_C, chromaFm));
981
56.6k
        const CodingUnit* colLumaCu = bestCS->lumaCS->getCU (lumaRefPos, CH_L, TREE_D);
982
        // update qp
983
56.6k
        qp = colLumaCu->qp;
984
56.6k
      }
985
986
81.7k
      m_cIntraSearch.reset();
987
988
81.7k
      bool isReuseCU = m_modeCtrl.isReusingCuValid( cs, partitioner, qp );
989
990
81.7k
      bool checkIbc = m_pcEncCfg->m_IBCMode && bestCS->picture->useIBC && (partitioner.chType == CH_L);
991
81.7k
      if ((m_pcEncCfg->m_IBCFastMethod>3) && (cs.area.lwidth() * cs.area.lheight()) > (16 * 16))
992
0
      {
993
0
        checkIbc = false;
994
0
      }
995
81.7k
      if( isReuseCU )
996
0
      {
997
0
        xReuseCachedResult( tempCS, bestCS, partitioner );
998
0
      }
999
81.7k
      else
1000
81.7k
      {
1001
        // add first pass modes
1002
81.7k
        if ( !slice.isIntra() && !slice.isIRAP() && !( cs.area.lwidth() == 4 && cs.area.lheight() == 4 ) && !partitioner.isConsIntra() )
1003
0
        {
1004
          // add inter modes
1005
0
          EncTestMode encTestModeSkip = { ETM_MERGE_SKIP, ETO_STANDARD, qp, lossless };
1006
0
          if (m_modeCtrl.tryMode(encTestModeSkip, cs, partitioner))
1007
0
          {
1008
0
            xCheckRDCostUnifiedMerge(tempCS, bestCS, partitioner, encTestModeSkip);
1009
1010
0
            CodingUnit* cu = bestCS->getCU(partitioner.chType, partitioner.treeType);
1011
0
            if (cu)
1012
0
              cu->mmvdSkip = cu->skip == false ? false : cu->mmvdSkip;
1013
0
          }
1014
0
          EncTestMode encTestMode = { ETM_INTER_ME, ETO_STANDARD, qp, lossless };
1015
0
          if (m_modeCtrl.tryMode(encTestMode, cs, partitioner))
1016
0
          {
1017
0
            xCheckRDCostInter(tempCS, bestCS, partitioner, encTestMode);
1018
0
          }
1019
1020
0
          if (m_pcEncCfg->m_AMVRspeed)
1021
0
          {
1022
0
            double bestIntPelCost = MAX_DOUBLE;
1023
1024
0
            EncTestMode encTestMode = {ETM_INTER_IMV, ETO_STANDARD, qp, lossless};
1025
0
            if( m_modeCtrl.tryMode( encTestMode, cs, partitioner ) )
1026
0
            {
1027
0
              const bool skipAltHpelIF = ( int( ( encTestMode.opts & ETO_IMV ) >> ETO_IMV_SHIFT ) == 4 ) && ( bestIntPelCost > 1.25 * bestCS->cost );
1028
0
              if (!skipAltHpelIF)
1029
0
              {
1030
0
                xCheckRDCostInterIMV(tempCS, bestCS, partitioner, encTestMode );
1031
0
              }
1032
0
            }
1033
0
          }
1034
0
        }
1035
1036
81.7k
        if (checkIbc && !partitioner.isConsInter())
1037
25.1k
        {
1038
25.1k
          EncTestMode encTestModeIBCMerge = { ETM_IBC_MERGE, ETO_STANDARD, qp, lossless };
1039
25.1k
          if ((m_pcEncCfg->m_IBCFastMethod < 4) && (partitioner.chType == CH_L) && m_modeCtrl.tryMode(encTestModeIBCMerge, cs, partitioner))
1040
25.1k
          {
1041
25.1k
            xCheckRDCostIBCModeMerge2Nx2N(tempCS, bestCS, partitioner, encTestModeIBCMerge);
1042
25.1k
          }
1043
1044
25.1k
          EncTestMode encTestModeIBC = { ETM_IBC, ETO_STANDARD, qp, lossless };
1045
25.1k
          if (m_modeCtrl.tryMode(encTestModeIBC, cs, partitioner))
1046
25.1k
          {
1047
25.1k
            xCheckRDCostIBCMode(tempCS, bestCS, partitioner, encTestModeIBC);
1048
25.1k
          }
1049
25.1k
        }
1050
81.7k
        if( m_EDO && bestCS->cost != MAX_DOUBLE )
1051
22.4k
        {
1052
22.4k
          xCalDebCost(*bestCS, partitioner);
1053
22.4k
        }
1054
1055
        // add intra modes
1056
81.7k
        EncTestMode encTestMode( {ETM_INTRA, ETO_STANDARD, qp, lossless} );
1057
81.7k
        if( !partitioner.isConsInter() && m_modeCtrl.tryMode( encTestMode, cs, partitioner ) )
1058
81.7k
        {
1059
81.7k
          xCheckRDCostIntra( tempCS, bestCS, partitioner, encTestMode );
1060
81.7k
        }
1061
81.7k
      } // reusing cu
1062
1063
81.7k
      m_modeCtrl.beforeSplit( partitioner );
1064
1065
81.7k
      if (cuECtx.bestCS && ((cuECtx.bestCostNoImv == (MAX_DOUBLE * .5) || cuECtx.isReusingCu) && !slice.isIntra()) )
1066
0
      {
1067
0
        m_cInterSearch.loadGlobalUniMvs( lumaArea, *pps.pcv );
1068
0
      }
1069
1070
81.7k
      if (!cs.slice->isIntra() && (partitioner.chType == CH_L) && ( m_pcEncCfg->m_qtbttSpeedUpMode & 2) && (partitioner.currQtDepth < 3) && bestCS->cus.size())
1071
0
      {
1072
0
        int flagDbefore = (bestCS->cus[0]->mergeFlag && !bestCS->cus[0]->mmvdMergeFlag && !bestCS->cus[0]->ispMode && !bestCS->cus[0]->geo) ? 1 : 0;
1073
0
        if (partitioner.currQtDepth == 0)
1074
0
        {
1075
0
          m_MergeSimpleFlag = flagDbefore;
1076
0
        }
1077
0
        else
1078
0
        {
1079
0
          int markFlag = (partitioner.currQtDepth == 1) ? 1 : 3;
1080
0
          m_MergeSimpleFlag = (flagDbefore << partitioner.currQtDepth) | (m_MergeSimpleFlag & markFlag);
1081
0
        }
1082
0
      }
1083
81.7k
    } //boundary
1084
1085
129k
    if( ( m_pcEncCfg->m_IntraPeriod == 1 ) && ( partitioner.chType == CH_C ) )
1086
0
    {
1087
0
      xCheckFastCuChromaSplitting( tempCS, bestCS, partitioner, *m_modeCtrl.comprCUCtx );
1088
0
    }
1089
    //////////////////////////////////////////////////////////////////////////
1090
    // split modes
1091
129k
    EncTestMode lastTestMode;
1092
1093
129k
    if( cuECtx.qtBeforeBt )
1094
20.7k
    {
1095
20.7k
      EncTestMode encTestMode( { ETM_SPLIT_QT, ETO_STANDARD, qp, false } );
1096
20.7k
      if( m_modeCtrl.trySplit( encTestMode, cs, partitioner, lastTestMode ) )
1097
13.5k
      {
1098
13.5k
        lastTestMode = encTestMode;
1099
13.5k
        xCheckModeSplit( tempCS, bestCS, partitioner, encTestMode );
1100
13.5k
      }
1101
20.7k
    }
1102
1103
129k
    if( partitioner.canSplit( CU_HORZ_SPLIT, cs ) )
1104
68.7k
    {
1105
      // add split modes
1106
68.7k
      EncTestMode encTestMode( { ETM_SPLIT_BT_H, ETO_STANDARD, qp, false } );
1107
68.7k
      if( m_modeCtrl.trySplit( encTestMode, cs, partitioner, lastTestMode ) )
1108
68.7k
      {
1109
68.7k
        lastTestMode = encTestMode;
1110
68.7k
        xCheckModeSplit( tempCS, bestCS, partitioner, encTestMode );
1111
68.7k
      }
1112
68.7k
    }
1113
1114
129k
    if( partitioner.canSplit( CU_VERT_SPLIT, cs ) )
1115
65.2k
    {
1116
      // add split modes
1117
65.2k
      EncTestMode encTestMode( { ETM_SPLIT_BT_V, ETO_STANDARD, qp, false } );
1118
65.2k
      if( m_modeCtrl.trySplit( encTestMode, cs, partitioner, lastTestMode ) )
1119
65.2k
      {
1120
65.2k
        lastTestMode = encTestMode;
1121
65.2k
        xCheckModeSplit( tempCS, bestCS, partitioner, encTestMode );
1122
65.2k
      }
1123
65.2k
    }
1124
1125
129k
    if( partitioner.canSplit( CU_TRIH_SPLIT, cs ) )
1126
28.5k
    {
1127
      // add split modes
1128
28.5k
      EncTestMode encTestMode( { ETM_SPLIT_TT_H, ETO_STANDARD, qp, false } );
1129
28.5k
      if( m_modeCtrl.trySplit( encTestMode, cs, partitioner, lastTestMode ) )
1130
1.51k
      {
1131
1.51k
        lastTestMode = encTestMode;
1132
1.51k
        xCheckModeSplit( tempCS, bestCS, partitioner, encTestMode );
1133
1.51k
      }
1134
28.5k
    }
1135
1136
129k
    if( partitioner.canSplit( CU_TRIV_SPLIT, cs ) )
1137
22.6k
    {
1138
      // add split modes
1139
22.6k
      EncTestMode encTestMode( { ETM_SPLIT_TT_V, ETO_STANDARD, qp, false } );
1140
22.6k
      if( m_modeCtrl.trySplit( encTestMode, cs, partitioner, lastTestMode ) )
1141
1.13k
      {
1142
1.13k
        lastTestMode = encTestMode;
1143
1.13k
        xCheckModeSplit( tempCS, bestCS, partitioner, encTestMode );
1144
1.13k
      }
1145
22.6k
    }
1146
1147
129k
    if( !cuECtx.qtBeforeBt )
1148
108k
    {
1149
108k
      EncTestMode encTestMode( { ETM_SPLIT_QT, ETO_STANDARD, qp, false } );
1150
108k
      if( m_modeCtrl.trySplit( encTestMode, cs, partitioner, lastTestMode ) )
1151
27.4k
      {
1152
27.4k
        lastTestMode = encTestMode;
1153
27.4k
        xCheckModeSplit( tempCS, bestCS, partitioner, encTestMode );
1154
27.4k
      }
1155
108k
    }
1156
129k
  }
1157
1158
129k
  if( bestCS->cus.empty() )
1159
0
  {
1160
0
    m_modeCtrl.finishCULevel( partitioner );
1161
0
    return;
1162
0
  }
1163
1164
  //////////////////////////////////////////////////////////////////////////
1165
  // Finishing CU
1166
  // set context states
1167
129k
  m_CABACEstimator->getCtx() = m_CurrCtx->best;
1168
1169
  // QP from last processed CU for further processing
1170
  //copy the qp of the last non-chroma CU
1171
129k
  int numCUInThisNode = (int)bestCS->cus.size();
1172
129k
  if( numCUInThisNode > 1 && bestCS->cus.back()->chType == CH_C && !CS::isDualITree( *bestCS ) )
1173
0
  {
1174
0
    CHECK( bestCS->cus[numCUInThisNode-2]->chType != CH_L, "wrong chType" );
1175
0
    bestCS->prevQP[partitioner.chType] = bestCS->cus[numCUInThisNode-2]->qp;
1176
0
  }
1177
129k
  else
1178
129k
  {
1179
129k
    bestCS->prevQP[partitioner.chType] = bestCS->cus.back()->qp;
1180
129k
  }
1181
129k
  if( ( !slice.isIntra() || slice.sps->IBC )
1182
129k
    && partitioner.chType == CH_L
1183
51.8k
    && bestCS->cus.size() == 1 && ( bestCS->cus.back()->predMode == MODE_INTER || bestCS->cus.back()->predMode == MODE_IBC )
1184
0
    && bestCS->area.Y() == (*bestCS->cus.back()).Y() )
1185
0
  {
1186
0
    const CodingUnit& cu = *bestCS->cus.front();
1187
0
    bool isIbcSmallBlk = CU::isIBC(cu) && (cu.lwidth() * cu.lheight() <= 16);
1188
0
    if (!cu.affine && !cu.geo && !isIbcSmallBlk)
1189
0
    {
1190
0
      const MotionInfo &mi = cu.getMotionInfo();
1191
0
      HPMVInfo hMi( mi, ( mi.interDir() == 3 ) ? cu.BcwIdx : BCW_DEFAULT, cu.imv == IMV_HPEL, CU::isIBC( cu ) );
1192
0
      cu.cs->addMiToLut( CU::isIBC( cu ) ? cu.cs->motionLut.lutIbc : cu.cs->motionLut.lut, hMi );
1193
0
    }
1194
0
  }
1195
1196
129k
  m_modeCtrl.finishCULevel( partitioner );
1197
129k
  if( m_cIntraSearch.getSaveCuCostInSCIPU() && bestCS->cus.size() == 1 )
1198
0
  {
1199
0
    m_cIntraSearch.saveCuAreaCostInSCIPU( Area( partitioner.currArea().lumaPos(), partitioner.currArea().lumaSize() ), bestCS->cost );
1200
0
  }
1201
1202
  // Assert if Best prediction mode is NONE
1203
  // Selected mode's RD-cost must be not MAX_DOUBLE.
1204
129k
  CHECK( bestCS->cus.empty()                                   , "No possible encoding found" );
1205
129k
  CHECK( bestCS->cus[0]->predMode == NUMBER_OF_PREDICTION_MODES, "No possible encoding found" );
1206
129k
  CHECK( bestCS->cost             == MAX_DOUBLE                , "No possible encoding found" );
1207
129k
}
1208
1209
1210
void EncCu::xCheckModeSplit(CodingStructure *&tempCS, CodingStructure *&bestCS, Partitioner &partitioner, const EncTestMode& encTestMode )
1211
177k
{
1212
177k
  const ModeType modeTypeParent  = partitioner.modeType;
1213
177k
  const TreeType treeTypeParent  = partitioner.treeType;
1214
177k
  const ChannelType chTypeParent = partitioner.chType;
1215
1216
177k
  int signalModeConsVal = CS::signalModeCons( *tempCS, partitioner.currArea(), getPartSplit(encTestMode), modeTypeParent);
1217
177k
  int numRoundRdo = signalModeConsVal == LDT_MODE_TYPE_SIGNAL ? 2 : 1;
1218
177k
  bool skipInterPass = false;
1219
355k
  for( int i = 0; i < numRoundRdo; i++ )
1220
177k
  {
1221
    //change cons modes
1222
177k
    if( signalModeConsVal == LDT_MODE_TYPE_SIGNAL )
1223
0
    {
1224
0
      CHECK( numRoundRdo != 2, "numRoundRdo shall be 2 - [LDT_MODE_TYPE_SIGNAL]" );
1225
0
      partitioner.modeType = (i == 0) ? MODE_TYPE_INTER : MODE_TYPE_INTRA;
1226
0
    }
1227
177k
    else if( signalModeConsVal == LDT_MODE_TYPE_INFER )
1228
0
    {
1229
0
      CHECK( numRoundRdo != 1, "numRoundRdo shall be 1 - [LDT_MODE_TYPE_INFER]" );
1230
0
      partitioner.modeType = MODE_TYPE_INTRA;
1231
0
    }
1232
177k
    else if( signalModeConsVal == LDT_MODE_TYPE_INHERIT )
1233
177k
    {
1234
177k
      CHECK( numRoundRdo != 1, "numRoundRdo shall be 1 - [LDT_MODE_TYPE_INHERIT]" );
1235
177k
      partitioner.modeType = modeTypeParent;
1236
177k
    }
1237
1238
    //for lite intra encoding fast algorithm, set the status to save inter coding info
1239
177k
    if( modeTypeParent == MODE_TYPE_ALL && partitioner.modeType == MODE_TYPE_INTER )
1240
0
    {
1241
0
      m_cIntraSearch.setSaveCuCostInSCIPU( true );
1242
0
      m_cIntraSearch.setNumCuInSCIPU( 0 );
1243
0
    }
1244
177k
    else if( modeTypeParent == MODE_TYPE_ALL && partitioner.modeType != MODE_TYPE_INTER )
1245
177k
    {
1246
177k
      m_cIntraSearch.setSaveCuCostInSCIPU( false );
1247
177k
      if( partitioner.modeType == MODE_TYPE_ALL )
1248
177k
      {
1249
177k
        m_cIntraSearch.setNumCuInSCIPU( 0 );
1250
177k
      }
1251
177k
    }
1252
1253
177k
    xCheckModeSplitInternal( tempCS, bestCS, partitioner, encTestMode, modeTypeParent, skipInterPass );
1254
    //recover cons modes
1255
177k
    partitioner.modeType = modeTypeParent;
1256
177k
    partitioner.treeType = treeTypeParent;
1257
177k
    partitioner.chType = chTypeParent;
1258
177k
    if( modeTypeParent == MODE_TYPE_ALL )
1259
177k
    {
1260
177k
      m_cIntraSearch.setSaveCuCostInSCIPU( false );
1261
177k
      if( numRoundRdo == 2 && partitioner.modeType == MODE_TYPE_INTRA )
1262
0
      {
1263
0
        m_cIntraSearch.initCuAreaCostInSCIPU();
1264
0
      }
1265
177k
    }
1266
177k
    if( skipInterPass )
1267
0
    {
1268
0
      break;
1269
0
    }
1270
177k
  }
1271
177k
}
1272
1273
void EncCu::xCheckModeSplitInternal(CodingStructure *&tempCS, CodingStructure *&bestCS, Partitioner &partitioner, const EncTestMode& encTestMode, const ModeType modeTypeParent, bool& skipInterPass )
1274
177k
{
1275
177k
  const int qp                     = encTestMode.qp;
1276
177k
  const int oldPrevQp              = tempCS->prevQP[partitioner.chType];
1277
177k
  const auto oldMotionLut          = tempCS->motionLut;
1278
                                   
1279
177k
  const PartSplit split            = getPartSplit( encTestMode );
1280
177k
  const ModeType  modeTypeChild    = partitioner.modeType;
1281
1282
177k
  CHECK( !( split == CU_QUAD_SPLIT || split == CU_HORZ_SPLIT || split == CU_VERT_SPLIT
1283
177k
         || split == CU_TRIH_SPLIT || split == CU_TRIV_SPLIT ), "invalid split type" );
1284
1285
177k
  tempCS->initStructData( qp );
1286
1287
177k
  m_CABACEstimator->getCtx()       = m_CurrCtx->start;
1288
1289
177k
  const uint16_t split_ctx_size    = Ctx::SplitFlag.size() + Ctx::SplitQtFlag.size() + Ctx::SplitHvFlag.size() + Ctx::Split12Flag.size() + Ctx::ModeConsFlag.size();
1290
177k
  const TempCtx  ctxSplitFlags     ( m_CtxCache, SubCtx( CtxSet( Ctx::SplitFlag(), split_ctx_size ), m_CABACEstimator->getCtx() ) );
1291
1292
177k
  m_CABACEstimator->determineNeighborCus( *tempCS, partitioner.currArea(), partitioner.chType, partitioner.treeType );
1293
177k
  m_CABACEstimator->resetBits           ();
1294
177k
  m_CABACEstimator->split_cu_mode       ( split, *tempCS, partitioner );
1295
177k
  partitioner     . modeType            = modeTypeParent;
1296
177k
  m_CABACEstimator->mode_constraint     ( split, *tempCS, partitioner, modeTypeChild );
1297
177k
  partitioner     . modeType            = modeTypeChild;
1298
1299
177k
  const int64_t splitBits   = m_CABACEstimator->getEstFracBits();
1300
1301
177k
  const bool chromaNotSplit = modeTypeParent == MODE_TYPE_ALL && modeTypeChild == MODE_TYPE_INTRA;
1302
177k
  const bool isChromaTooBig = isChromaEnabled( tempCS->pps->pcv->chrFormat ) && tempCS->area.Y().maxDim() > tempCS->sps->getMaxTbSize();
1303
177k
  bool       skipSplitTest  = chromaNotSplit && isChromaTooBig;
1304
1305
177k
  if( !skipSplitTest )
1306
177k
  {
1307
177k
    double         a = -1, b = -1;
1308
177k
    const unsigned w       = partitioner.currArea().lwidth();
1309
177k
    const unsigned h       = partitioner.currArea().lheight();
1310
177k
    const bool contextCond = w == h && tempCS->slice->sliceType == VVENC_B_SLICE && isLuma( partitioner.chType ) && m_pcEncCfg->m_splitCostThrParamId >= 0 && m_pcEncCfg->m_splitCostThrParamId <= 3;
1311
1312
177k
    if( contextCond )
1313
0
    {
1314
0
      uint8_t nsPredInd = m_modeCtrl.comprCUCtx->bestNsPredMode.type == ETM_INTRA;
1315
0
      uint8_t szInd     = getLog2( w ) - 3;
1316
0
      uint8_t splitInd  = split == CU_QUAD_SPLIT ? 1 : 0;
1317
0
      if ( m_pcEncCfg->m_splitCostThrParamId <= 1 )
1318
0
      {
1319
0
        a = coefSquareCUsFasterFastMedium[m_pcEncCfg->m_splitCostThrParamId][szInd][nsPredInd][splitInd][0];
1320
0
        b = coefSquareCUsFasterFastMedium[m_pcEncCfg->m_splitCostThrParamId][szInd][nsPredInd][splitInd][1];
1321
0
      }
1322
0
      else
1323
0
      {
1324
0
        uint8_t mtInd = (partitioner.currMtDepth == 0);
1325
0
        a = coefSquareCUsSlowSlower[m_pcEncCfg->m_splitCostThrParamId - 2][szInd][nsPredInd][partitioner.currQtDepth][mtInd][splitInd][0];
1326
0
        b = coefSquareCUsSlowSlower[m_pcEncCfg->m_splitCostThrParamId - 2][szInd][nsPredInd][partitioner.currQtDepth][mtInd][splitInd][1];
1327
0
      }
1328
0
    }
1329
1330
177k
    if( a > -1 && b > -1 )
1331
0
    {
1332
0
      const double bestNsCost    = m_modeCtrl.comprCUCtx->bestCostBeforeSplit == MAX_DOUBLE ? -1 : m_modeCtrl.comprCUCtx->bestCostBeforeSplit;
1333
0
      const double factor        = 1.0 + b * exp( a * qp );
1334
0
      const double predSplitCost = bestNsCost / factor + splitBits;
1335
0
      skipSplitTest              = bestNsCost >= 0 && predSplitCost >= bestNsCost;
1336
0
    }
1337
177k
    else
1338
177k
    {
1339
177k
      int numChild = 3;
1340
177k
      if( split == CU_VERT_SPLIT || split == CU_HORZ_SPLIT ) numChild--;
1341
43.6k
      else if( split == CU_QUAD_SPLIT ) numChild++;
1342
1343
177k
      int64_t approxBits = m_pcEncCfg->m_qtbttSpeedUp > 0 ? numChild << SCALE_BITS : 0;
1344
1345
177k
      const double factor     = ( tempCS->currQP[partitioner.chType] > 30                              ? 1.1  : 1.075 ) +
1346
177k
                                (   m_pcEncCfg->m_qtbttSpeedUp > 0                                     ? 0.01 : 0.0   ) +
1347
177k
                                ( ( m_pcEncCfg->m_qtbttSpeedUp > 0 && isChroma( partitioner.chType ) ) ? 0.2  : 0.0   );
1348
       
1349
177k
      const double baseCost   = bestCS->cost + bestCS->costDbOffset;
1350
177k
      const double predCost   = baseCost / factor + splitBits + approxBits;
1351
177k
      skipSplitTest           = predCost >= baseCost;
1352
177k
    }
1353
177k
  }
1354
1355
177k
  if( skipSplitTest )
1356
110k
  {
1357
110k
    m_CABACEstimator->getCtx() = SubCtx( CtxSet( Ctx::SplitFlag(), split_ctx_size ), ctxSplitFlags );
1358
110k
    xCheckBestMode( tempCS, bestCS, partitioner, encTestMode );
1359
110k
    return;
1360
110k
  }
1361
1362
67.0k
  if( partitioner.treeType == TREE_D )
1363
67.0k
  {
1364
67.0k
    if( chromaNotSplit )
1365
0
    {
1366
0
      CHECK( partitioner.chType != CH_L, "chType must be luma" );
1367
0
      partitioner.treeType = TREE_L;
1368
0
    }
1369
67.0k
    else
1370
67.0k
    {
1371
67.0k
      partitioner.treeType = TREE_D;
1372
67.0k
    }
1373
67.0k
  }
1374
1375
67.0k
  partitioner.splitCurrArea( split, *tempCS );
1376
67.0k
  bool qgEnableChildren = partitioner.currQgEnable(); // QG possible at children level
1377
1378
67.0k
  m_CurrCtx++;
1379
1380
67.0k
  AffineMVInfo tmpMVInfo;
1381
67.0k
  bool isAffMVInfoSaved = m_cInterSearch.m_AffineProfList->savePrevAffMVInfo( 0, tmpMVInfo );
1382
1383
67.0k
  BlkUniMvInfo tmpUniMvInfo;
1384
67.0k
  bool         isUniMvInfoSaved = false;
1385
67.0k
  if( !tempCS->slice->isIntra() )
1386
0
  {
1387
0
    m_cInterSearch.m_BlkUniMvInfoBuffer->savePrevUniMvInfo( tempCS->area.Y(), tmpUniMvInfo, isUniMvInfoSaved );
1388
0
  }
1389
1390
67.0k
  DeriveCtx deriveCtx = m_CABACEstimator->getDeriveCtx();
1391
1392
67.0k
  do
1393
172k
  {
1394
172k
    const auto &subCUArea  = partitioner.currArea();
1395
1396
172k
    if( tempCS->picture->Y().contains( subCUArea.lumaPos() ) )
1397
121k
    {
1398
121k
      PelStorage* orgBuffer =  &m_pOrgBuffer[partitioner.currDepth];
1399
121k
      PelStorage* rspBuffer =  &m_pRspBuffer[partitioner.currDepth];
1400
121k
      CodingStructure *tempSubCS = m_pTempCS[partitioner.currDepth];
1401
121k
      CodingStructure *bestSubCS = m_pBestCS[partitioner.currDepth];
1402
1403
121k
      tempCS->initSubStructure( *tempSubCS, partitioner.chType, subCUArea, false, orgBuffer, rspBuffer );
1404
121k
      tempCS->initSubStructure( *bestSubCS, partitioner.chType, subCUArea, false, orgBuffer, rspBuffer );
1405
1406
      // copy org buffer, need to be done after initSubStructure because of reshaping!
1407
121k
      orgBuffer->copyFrom( tempCS->getOrgBuf( subCUArea ) );
1408
1409
121k
      tempSubCS->bestParent = bestSubCS->bestParent = bestCS;
1410
1411
121k
      xCompressCU(tempSubCS, bestSubCS, partitioner );
1412
1413
121k
      tempSubCS->bestParent = bestSubCS->bestParent = nullptr;
1414
1415
121k
      if( bestSubCS->cost == MAX_DOUBLE )
1416
0
      {
1417
0
        CHECK( split == CU_QUAD_SPLIT, "Split decision reusing cannot skip quad split" );
1418
0
        tempCS->cost = MAX_DOUBLE;
1419
0
        tempCS->costDbOffset = 0;
1420
0
        m_CurrCtx--;
1421
0
        partitioner.exitCurrSplit();
1422
0
        xCheckBestMode( tempCS, bestCS, partitioner, encTestMode );
1423
0
        if( partitioner.chType == CH_L )
1424
0
        {
1425
0
          tempCS->motionLut = oldMotionLut;
1426
0
        }
1427
1428
0
        m_CABACEstimator->getDeriveCtx() = deriveCtx;
1429
0
        return;
1430
0
      }
1431
1432
121k
      tempCS->useSubStructure( *bestSubCS, partitioner.chType, TREE_D, CS::getArea( *tempCS, subCUArea, partitioner.chType, partitioner.treeType ), partitioner.hasNextPart() || chromaNotSplit );
1433
1434
121k
      if( partitioner.currQgEnable() )
1435
6.88k
      {
1436
6.88k
        tempCS->prevQP[partitioner.chType] = bestSubCS->prevQP[partitioner.chType];
1437
6.88k
      }
1438
121k
      if( partitioner.isConsInter() )
1439
0
      {
1440
0
        for( int i = 0; i < bestSubCS->cus.size(); i++ )
1441
0
        {
1442
0
          CHECK( bestSubCS->cus[i]->predMode != MODE_INTER, "all CUs must be inter mode in an Inter coding region (SCIPU)" );
1443
0
        }
1444
0
      }
1445
121k
      else if( partitioner.isConsIntra() )
1446
0
      {
1447
0
        for( int i = 0; i < bestSubCS->cus.size(); i++ )
1448
0
        {
1449
0
          CHECK( bestSubCS->cus[i]->predMode == MODE_INTER, "all CUs must not be inter mode in an Intra coding region (SCIPU)" );
1450
0
        }
1451
0
      }
1452
1453
121k
      tempSubCS->releaseIntermediateData();
1454
121k
      bestSubCS->releaseIntermediateData();
1455
121k
      if( !tempCS->slice->isIntra() && partitioner.isConsIntra() )
1456
0
      {
1457
0
        tempCS->cost = m_cRdCost.calcRdCost( tempCS->fracBits, tempCS->dist );
1458
0
        if( tempCS->cost > bestCS->cost )
1459
0
        {
1460
0
          tempCS->cost = MAX_DOUBLE;
1461
0
          tempCS->costDbOffset = 0;
1462
0
          m_CurrCtx--;
1463
0
          partitioner.exitCurrSplit();
1464
0
          if( partitioner.chType == CH_L )
1465
0
          {
1466
0
            tempCS->motionLut = oldMotionLut;
1467
0
          }
1468
1469
0
          m_CABACEstimator->getDeriveCtx() = deriveCtx;
1470
0
          return;
1471
0
        }
1472
0
      }
1473
121k
    }
1474
172k
  } while( partitioner.nextPart( *tempCS ) );
1475
1476
67.0k
  partitioner.exitCurrSplit();
1477
1478
67.0k
  m_CurrCtx--;
1479
1480
67.0k
  m_CABACEstimator->getDeriveCtx() = deriveCtx;
1481
1482
67.0k
  if( chromaNotSplit )
1483
0
  {
1484
    //Note: In local dual tree region, the chroma CU refers to the central luma CU's QP.
1485
    //If the luma CU QP shall be predQP (no residual in it and before it in the QG), it must be revised to predQP before encoding the chroma CU
1486
    //Otherwise, the chroma CU uses predQP+deltaQP in encoding but is decoded as using predQP, thus causing encoder-decoded mismatch on chroma qp.
1487
0
    if( tempCS->pps->useDQP )
1488
0
    {
1489
      //find parent CS that including all coded CUs in the QG before this node
1490
0
      CodingStructure* qgCS = tempCS;
1491
0
      bool deltaQpCodedBeforeThisNode = false;
1492
0
      if( partitioner.currArea().lumaPos() != partitioner.currQgPos )
1493
0
      {
1494
0
        int numParentNodeToQgCS = 0;
1495
0
        while( qgCS->area.lumaPos() != partitioner.currQgPos )
1496
0
        {
1497
0
          CHECK( qgCS->parent == nullptr, "parent of qgCS shall exsit" );
1498
0
          qgCS = qgCS->parent;
1499
0
          numParentNodeToQgCS++;
1500
0
        }
1501
1502
        //check whether deltaQP has been coded (in luma CU or luma&chroma CU) before this node
1503
0
        CodingStructure* parentCS = tempCS->parent;
1504
0
        for( int i = 0; i < numParentNodeToQgCS; i++ )
1505
0
        {
1506
          //checking each parent
1507
0
          CHECK( parentCS == nullptr, "parentCS shall exsit" );
1508
0
          for( const auto &cu : parentCS->cus )
1509
0
          {
1510
0
            if( cu->rootCbf && !isChroma( cu->chType ) )
1511
0
            {
1512
0
              deltaQpCodedBeforeThisNode = true;
1513
0
              break;
1514
0
            }
1515
0
          }
1516
0
          parentCS = parentCS->parent;
1517
0
        }
1518
0
      }
1519
1520
      //revise luma CU qp before the first luma CU with residual in the SCIPU to predQP
1521
0
      if( !deltaQpCodedBeforeThisNode )
1522
0
      {
1523
        //get pred QP of the QG
1524
0
        const CodingUnit* cuFirst = qgCS->getCU( CH_L, TREE_D );
1525
0
        CHECK( cuFirst->lumaPos() != partitioner.currQgPos, "First cu of the Qg is wrong" );
1526
0
        int predQp = CU::predictQP( *cuFirst, qgCS->prevQP[CH_L] );
1527
1528
        //revise to predQP
1529
0
        int firstCuHasResidual = (int)tempCS->cus.size();
1530
0
        for( int i = 0; i < tempCS->cus.size(); i++ )
1531
0
        {
1532
0
          if( tempCS->cus[i]->rootCbf )
1533
0
          {
1534
0
            firstCuHasResidual = i;
1535
0
            break;
1536
0
          }
1537
0
        }
1538
1539
0
        for( int i = 0; i < firstCuHasResidual; i++ )
1540
0
        {
1541
0
          tempCS->cus[i]->qp = predQp;
1542
0
        }
1543
0
      }
1544
0
    }
1545
0
    partitioner.chType   = CH_C;
1546
0
    partitioner.treeType = TREE_C;
1547
1548
0
    m_CurrCtx++;
1549
1550
0
    CodingStructure *tempCSChroma = m_pTempCS2;
1551
0
    CodingStructure *bestCSChroma = m_pBestCS2;
1552
1553
0
    tempCS->initSubStructure( *tempCSChroma, partitioner.chType, partitioner.currArea(), false );
1554
0
    tempCS->initSubStructure( *bestCSChroma, partitioner.chType, partitioner.currArea(), false );
1555
0
    tempCSChroma->lumaCS = tempCS;
1556
0
    bestCSChroma->lumaCS = tempCS;
1557
0
    xCompressCU( tempCSChroma, bestCSChroma, partitioner );
1558
1559
    //attach chromaCS to luma CS and update cost
1560
0
    tempCS->useSubStructure( *bestCSChroma, partitioner.chType, TREE_D, CS::getArea( *bestCSChroma, partitioner.currArea(), partitioner.chType, partitioner.treeType ), false );
1561
1562
    //release tmp resource
1563
0
    tempCSChroma->releaseIntermediateData();
1564
0
    bestCSChroma->releaseIntermediateData();
1565
1566
0
    m_CurrCtx--;
1567
    //recover luma tree status
1568
0
    partitioner.chType = CH_L;
1569
0
    partitioner.treeType = TREE_D;
1570
0
    partitioner.modeType = MODE_TYPE_ALL;
1571
0
  }
1572
1573
  // Finally, add split-signaling bits for RD-cost check
1574
67.0k
  tempCS->fracBits += splitBits; // split bits
1575
67.0k
  tempCS->cost      = m_cRdCost.calcRdCost( tempCS->fracBits, tempCS->dist );
1576
67.0k
  partitioner.modeType = modeTypeParent;
1577
1578
  // Check Delta QP bits for splitted structure
1579
67.0k
  if( !qgEnableChildren ) // check at deepest QG level only
1580
64.4k
    xCheckDQP( *tempCS, partitioner, true );
1581
1582
  // If the configuration being tested exceeds the maximum number of bytes for a slice / slice-segment, then
1583
  // a proper RD evaluation cannot be performed. Therefore, termination of the
1584
  // slice/slice-segment must be made prior to this CTU.
1585
  // This can be achieved by forcing the decision to be that of the rpcTempCU.
1586
  // The exception is each slice / slice-segment must have at least one CTU.
1587
67.0k
  if( bestCS->cost == MAX_DOUBLE )
1588
47.8k
  {
1589
47.8k
    bestCS->costDbOffset = 0;
1590
47.8k
  }
1591
1592
67.0k
  if( tempCS->cus.size() > 0 && modeTypeParent == MODE_TYPE_ALL && modeTypeChild == MODE_TYPE_INTER )
1593
0
  {
1594
0
    int areaSizeNoResiCu = 0;
1595
0
    for( int k = 0; k < tempCS->cus.size(); k++ )
1596
0
    {
1597
0
      areaSizeNoResiCu += (tempCS->cus[k]->rootCbf == false) ? tempCS->cus[k]->lumaSize().area() : 0;
1598
0
    }
1599
0
    if( areaSizeNoResiCu >= (tempCS->area.lumaSize().area() >> 1) )
1600
0
    {
1601
0
      skipInterPass = true;
1602
0
    }
1603
0
  }
1604
1605
  // RD check for sub partitioned coding structure.
1606
67.0k
  xCheckBestMode( tempCS, bestCS, partitioner, encTestMode, m_EDO );
1607
1608
67.0k
  if( isAffMVInfoSaved )
1609
0
  {
1610
0
    m_cInterSearch.m_AffineProfList->addAffMVInfo(tmpMVInfo);
1611
0
  }
1612
1613
67.0k
  if( !tempCS->slice->isIntra() && isUniMvInfoSaved )
1614
0
  {
1615
0
    m_cInterSearch.m_BlkUniMvInfoBuffer->addUniMvInfo(tmpUniMvInfo);
1616
0
  }
1617
1618
67.0k
  tempCS->motionLut = oldMotionLut;
1619
67.0k
  tempCS->releaseIntermediateData();
1620
67.0k
  tempCS->prevQP[partitioner.chType] = oldPrevQp;
1621
67.0k
}
1622
1623
1624
void EncCu::xCheckRDCostIntra( CodingStructure *&tempCS, CodingStructure *&bestCS, Partitioner &partitioner, const EncTestMode& encTestMode )
1625
81.7k
{
1626
81.7k
  PROFILER_SCOPE_AND_STAGE_EXT( 1, _TPROF, P_INTRA, tempCS, partitioner.chType );
1627
1628
81.7k
  tempCS->initStructData( encTestMode.qp, false ); // clear motion buffer
1629
1630
81.7k
  CodingUnit &cu      = tempCS->addCU( CS::getArea( *tempCS, tempCS->area, partitioner.chType, partitioner.treeType ), partitioner.chType );
1631
1632
81.7k
  partitioner.setCUData( cu );
1633
81.7k
  cu.slice            = tempCS->slice;
1634
81.7k
  cu.tileIdx          = m_tileIdx;
1635
81.7k
  cu.skip             = false;
1636
81.7k
  cu.mmvdSkip         = false;
1637
81.7k
  cu.predMode         = MODE_INTRA;
1638
81.7k
  cu.chromaQpAdj      = m_cuChromaQpOffsetIdxPlus1;
1639
81.7k
  cu.qp               = encTestMode.qp;
1640
81.7k
  cu.ispMode          = NOT_INTRA_SUBPARTITIONS;
1641
81.7k
  cu.initPuData();
1642
1643
81.7k
  m_cIntraSearch.m_ispTestedModes[0].init(0, 0, 1);
1644
81.7k
  if (m_pcEncCfg->m_FastIntraTools)
1645
81.7k
  {
1646
81.7k
    m_modeCtrl.comprCUCtx->intraWasTested = false;
1647
81.7k
    m_cIntraSearch.m_ispTestedModes[0].relatedCuIsValid = m_modeCtrl.comprCUCtx->relatedCuIsValid;
1648
81.7k
    if (!bestCS->cus.empty())
1649
22.4k
    {
1650
22.4k
      if ((bestCS->cus[0]->mergeFlag || bestCS->cus[0]->imv || bestCS->cus[0]->affine) && (!bestCS->cus[0]->ciip))
1651
22.4k
      {
1652
22.4k
        m_cIntraSearch.m_ispTestedModes[0].bestBefore[0] = -1;
1653
22.4k
      }
1654
22.4k
    }
1655
81.7k
    if (!bestCS->slice->isIntra())
1656
0
    {
1657
0
      const Position posBL = cu.Y().bottomLeft();
1658
0
      const Position posTR = cu.Y().topRight();
1659
0
      for (int i = 0; i < 2; i++)
1660
0
      {
1661
0
        const CodingUnit* neigh = i ? cu.cs->getCURestricted(posTR.offset(0, -1), cu, CH_L) :cu.cs->getCURestricted(posBL.offset(-1, 0), cu, CH_L);
1662
0
        m_cIntraSearch.m_ispTestedModes[0].bestBefore[i+1] = -1;
1663
0
        if (neigh != nullptr)
1664
0
        {
1665
0
          int bestMode = neigh->firstTU->mtsIdx[0] ? 4 : 0;
1666
0
          bestMode |= neigh->lfnstIdx ? 2 : 0;
1667
0
          bestMode |= neigh->ispMode ? 1 : 0;
1668
0
          m_cIntraSearch.m_ispTestedModes[0].bestBefore[i+1] = bestMode;
1669
0
        }
1670
0
      }
1671
0
    }
1672
81.7k
  }
1673
1674
81.7k
  tempCS->interHad    = m_modeCtrl.comprCUCtx->interHad;
1675
81.7k
  double maxCostAllowedForChroma = MAX_DOUBLE;
1676
81.7k
  if( isLuma( partitioner.chType ) )
1677
25.1k
  {
1678
25.1k
    if (!tempCS->slice->isIntra() && bestCS)
1679
0
    {
1680
0
      m_cIntraSearch.estIntraPredLumaQT(cu, partitioner, bestCS->cost);
1681
0
    }
1682
25.1k
    else
1683
25.1k
    {
1684
25.1k
      m_cIntraSearch.estIntraPredLumaQT(cu, partitioner);
1685
25.1k
    }
1686
25.1k
    if (m_pcEncCfg->m_FastIntraTools)
1687
25.1k
    {
1688
25.1k
      if (m_cIntraSearch.m_ispTestedModes[0].intraWasTested)
1689
25.1k
      {
1690
25.1k
        m_modeCtrl.comprCUCtx->intraWasTested = m_cIntraSearch.m_ispTestedModes[0].intraWasTested;
1691
25.1k
      }
1692
25.1k
    }
1693
1694
25.1k
    if( !partitioner.isSepTree( *tempCS ) )
1695
0
    {
1696
0
      tempCS->lumaCost = m_cRdCost.calcRdCost( tempCS->fracBits, tempCS->dist );
1697
0
    }
1698
25.1k
    if (m_pcEncCfg->m_usePbIntraFast && tempCS->dist == MAX_DISTORTION && tempCS->interHad == 0)
1699
0
    {
1700
      // JEM assumes only perfect reconstructions can from now on beat the inter mode
1701
0
      m_modeCtrl.comprCUCtx->interHad = 0;
1702
0
      return;
1703
0
    }
1704
25.1k
  }
1705
1706
81.7k
  if( tempCS->area.chromaFormat != CHROMA_400 && ( partitioner.chType == CH_C || !CU::isSepTree(cu) ) )
1707
56.6k
  {
1708
56.6k
    bool useIntraSubPartitions = cu.ispMode != NOT_INTRA_SUBPARTITIONS;
1709
56.6k
    Partitioner subTuPartitioner = partitioner;
1710
56.6k
    if ((m_pcEncCfg->m_ISP >= 3) && (!partitioner.isSepTree(*tempCS) && useIntraSubPartitions))
1711
0
    {
1712
0
      maxCostAllowedForChroma = bestCS->cost < MAX_DOUBLE ? bestCS->cost - tempCS->lumaCost : MAX_DOUBLE;
1713
0
    }
1714
56.6k
    m_cIntraSearch.estIntraPredChromaQT(
1715
56.6k
      cu, (!useIntraSubPartitions || (CU::isSepTree(cu) && !isLuma(CH_C))) ? partitioner : subTuPartitioner,
1716
56.6k
      maxCostAllowedForChroma);
1717
56.6k
    if ((m_pcEncCfg->m_ISP >= 3) && useIntraSubPartitions && !cu.ispMode)
1718
0
    {
1719
0
      return;
1720
0
    }
1721
56.6k
  }
1722
1723
81.7k
  cu.rootCbf = false;
1724
1725
327k
  for (uint32_t t = 0; t < getNumberValidTBlocks(*cu.cs->pcv); t++)
1726
245k
  {
1727
245k
    cu.rootCbf |= cu.firstTU->cbf[t] != 0;
1728
245k
  }
1729
1730
  // Get total bits for current mode: encode CU
1731
81.7k
  m_CABACEstimator->resetBits();
1732
1733
81.7k
  if ((!cu.cs->slice->isIntra() || cu.cs->slice->sps->IBC) && cu.Y().valid())
1734
25.1k
  {
1735
25.1k
    m_CABACEstimator->cu_skip_flag(cu);
1736
25.1k
  }
1737
81.7k
  m_CABACEstimator->pred_mode(cu);
1738
81.7k
  m_CABACEstimator->cu_pred_data(cu);
1739
1740
  // Encode Coefficients
1741
81.7k
  CUCtx cuCtx;
1742
81.7k
  cuCtx.isDQPCoded = true;
1743
81.7k
  cuCtx.isChromaQpAdjCoded = true;
1744
81.7k
  m_CABACEstimator->cu_residual(cu, partitioner, cuCtx);
1745
1746
81.7k
  tempCS->fracBits = m_CABACEstimator->getEstFracBits();
1747
81.7k
  tempCS->cost = m_cRdCost.calcRdCost(tempCS->fracBits, tempCS->dist);
1748
1749
81.7k
  xEncodeDontSplit(*tempCS, partitioner);
1750
1751
81.7k
  xCheckDQP(*tempCS, partitioner);
1752
1753
81.7k
  if( m_EDO )
1754
81.7k
  {
1755
81.7k
    xCalDebCost(*tempCS, partitioner);
1756
81.7k
  }
1757
1758
81.7k
  DTRACE_MODE_COST(*tempCS, m_cRdCost.getLambda(true));
1759
81.7k
  xCheckBestMode(tempCS, bestCS, partitioner, encTestMode, m_EDO);
1760
1761
81.7k
  STAT_COUNT_CU_MODES( partitioner.chType == CH_L, g_cuCounters1D[CU_MODES_TESTED][0][!tempCS->slice->isIntra() + tempCS->slice->depth] );
1762
81.7k
  STAT_COUNT_CU_MODES( partitioner.chType == CH_L && !tempCS->slice->isIntra(), g_cuCounters2D[CU_MODES_TESTED][Log2( tempCS->area.lheight() )][Log2( tempCS->area.lwidth() )] );
1763
81.7k
}
1764
1765
void EncCu::xCheckDQP( CodingStructure& cs, Partitioner& partitioner, bool bKeepCtx )
1766
168k
{
1767
168k
  if( !cs.pps->useDQP )
1768
0
  {
1769
0
    return;
1770
0
  }
1771
1772
168k
  if (partitioner.isSepTree(cs) && isChroma(partitioner.chType))
1773
95.1k
  {
1774
95.1k
    return;
1775
95.1k
  }
1776
1777
73.4k
  if( !partitioner.currQgEnable() ) // do not consider split or leaf/not leaf QG condition (checked by caller)
1778
66.1k
  {
1779
66.1k
    return;
1780
66.1k
  }
1781
1782
7.30k
  CodingUnit* cuFirst = cs.getCU( partitioner.chType, partitioner.treeType );
1783
1784
7.30k
  CHECK( bKeepCtx && cs.cus.size() <= 1 && partitioner.getImplicitSplit( cs ) == CU_DONT_SPLIT, "bKeepCtx should only be set in split case" );
1785
7.30k
  CHECK( !bKeepCtx && cs.cus.size() > 1, "bKeepCtx should never be set for non-split case" );
1786
7.30k
  CHECK( !cuFirst, "No CU available" );
1787
1788
7.30k
  bool hasResidual = false;
1789
7.30k
  for( const auto &cu : cs.cus )
1790
20.3k
  {
1791
    //not include the chroma CU because chroma CU is decided based on corresponding luma QP and deltaQP is not signaled at chroma CU
1792
20.3k
    if( cu->rootCbf && !isChroma( cu->chType ))
1793
1.20k
    {
1794
1.20k
      hasResidual = true;
1795
1.20k
      break;
1796
1.20k
    }
1797
20.3k
  }
1798
1799
7.30k
  int predQP = CU::predictQP( *cuFirst, cs.prevQP[partitioner.chType] );
1800
1801
7.30k
  if( hasResidual )
1802
1.20k
  {
1803
1.20k
    TempCtx ctxTemp( m_CtxCache );
1804
1.20k
    if( !bKeepCtx ) ctxTemp = SubCtx( Ctx::DeltaQP, m_CABACEstimator->getCtx() );
1805
1806
1.20k
    m_CABACEstimator->resetBits();
1807
1.20k
    m_CABACEstimator->cu_qp_delta( *cuFirst, predQP, cuFirst->qp );
1808
1809
1.20k
    cs.fracBits += m_CABACEstimator->getEstFracBits(); // dQP bits
1810
1.20k
    cs.cost      = m_cRdCost.calcRdCost(cs.fracBits, cs.dist);
1811
1812
1813
1.20k
    if( !bKeepCtx ) m_CABACEstimator->getCtx() = SubCtx( Ctx::DeltaQP, ctxTemp );
1814
1815
    // NOTE: reset QPs for CUs without residuals up to first coded CU
1816
1.20k
    for( const auto &cu : cs.cus )
1817
1.21k
    {
1818
      //not include the chroma CU because chroma CU is decided based on corresponding luma QP and deltaQP is not signaled at chroma CU
1819
1.21k
      if( cu->rootCbf && !isChroma( cu->chType ))
1820
1.20k
      {
1821
1.20k
        break;
1822
1.20k
      }
1823
6
      cu->qp = predQP;
1824
6
    }
1825
1.20k
  }
1826
6.09k
  else
1827
6.09k
  {
1828
    // No residuals: reset CU QP to predicted value
1829
6.09k
    for( const auto &cu : cs.cus )
1830
19.1k
    {
1831
19.1k
      cu->qp = predQP;
1832
19.1k
    }
1833
6.09k
  }
1834
7.30k
}
1835
1836
CodingUnit *EncCu::getCuForInterPrediction( CodingStructure *cs, const EncTestMode& encTestMode )
1837
0
{
1838
0
  CodingUnit *cu = cs->getCU( CH_L, TREE_D );
1839
1840
0
  if( cu == nullptr )
1841
0
  {
1842
0
    CHECK( cs->getCU( CH_L, TREE_D ) != nullptr, "Wrong CU/PU setting in CS" );
1843
0
    cu = &cs->addCU( cs->area, CH_L );
1844
0
  }
1845
1846
0
  cu->slice       = cs->slice;
1847
0
  cu->tileIdx     = m_tileIdx;
1848
0
  cu->skip        = false;
1849
0
  cu->mmvdSkip    = false;
1850
0
  cu->mmvdMergeFlag
1851
0
                  = false;
1852
0
  cu->geo         = false;
1853
0
  cu->predMode    = MODE_INTER;
1854
0
  cu->chromaQpAdj = m_cuChromaQpOffsetIdxPlus1;
1855
0
  cu->qp          = encTestMode.qp;
1856
0
  cu->affine      = false;
1857
0
  cu->multiRefIdx = 0;
1858
0
  cu->mipFlag     = false;
1859
0
  cu->ciip        = false;
1860
1861
0
  return cu;
1862
0
}
1863
1864
int getDmvrMvdNum( const CodingUnit &cu )
1865
0
{
1866
0
  const int dx = std::max<int>( cu.lwidth()  >> DMVR_SUBCU_SIZE_LOG2, 1 );
1867
0
  const int dy = std::max<int>( cu.lheight() >> DMVR_SUBCU_SIZE_LOG2, 1 );
1868
0
  return dx * dy;
1869
0
}
1870
1871
void EncCu::xCheckRDCostUnifiedMerge( CodingStructure *&tempCS, CodingStructure *&bestCS, Partitioner &partitioner, EncTestMode &encTestMode )
1872
0
{
1873
0
  const Slice &slice = *tempCS->slice;
1874
1875
0
  CHECK( slice.sliceType == VVENC_I_SLICE, "Merge modes not available for I-slices" );
1876
1877
0
  tempCS->initStructData( encTestMode.qp );
1878
1879
0
  MergeCtx          mergeCtx, gpmMergeCtx;
1880
0
  AffineMergeCtx    affineMergeCtx;
1881
0
  GeoComboCostList &comboList = m_comboList;
1882
0
  const SPS        &sps       = *tempCS->sps;
1883
1884
0
  if( sps.SbtMvp )
1885
0
  {
1886
0
    const Size bufSize           = g_miScaling.scale( tempCS->area.lumaSize() );
1887
0
    affineMergeCtx.subPuMvpMiBuf = MotionBuf        ( m_subPuMiBuf, bufSize );
1888
0
  }
1889
1890
0
  m_mergeBestSATDCost = MAX_DOUBLE;
1891
1892
0
  CodingUnit *cu = getCuForInterPrediction( tempCS, encTestMode );
1893
0
  partitioner.setCUData            ( *cu );
1894
0
  CU::getInterMergeCandidates      ( *cu, mergeCtx, 0 );
1895
0
  if( sps.MMVD )
1896
0
    CU::getInterMMVDMergeCandidates( *cu, mergeCtx );
1897
1898
0
  bool sameMV[MRG_MAX_NUM_CANDS] = { false, };
1899
0
  if( m_pcEncCfg->m_useFastMrg >= 2 )
1900
0
  {
1901
0
    for( int m = 0; m < mergeCtx.numValidMergeCand - 1; m++ )
1902
0
    {
1903
0
      if( !sameMV[m] )
1904
0
      {
1905
0
        for( int n = m + 1; n < mergeCtx.numValidMergeCand; n++ )
1906
0
        {
1907
0
          sameMV[n] |= mergeCtx.mvFieldNeighbours[m][0] == mergeCtx.mvFieldNeighbours[n][0]
1908
0
                    && mergeCtx.mvFieldNeighbours[m][1] == mergeCtx.mvFieldNeighbours[n][1];
1909
0
        }
1910
0
      }
1911
0
    }
1912
0
  }
1913
1914
0
  MergeBufVector mrgPredBufNoCiip;
1915
0
  MergeBufVector geoBuffer;
1916
0
  const double  sqrtLambdaForFirstPass = m_cRdCost.getMotionLambda() * FRAC_BITS_SCALE;
1917
1918
0
  const UnitArea localUnitArea( cu->chromaFormat, Area( 0, 0, cu->Y().width, cu->Y().height ) );
1919
0
  for( int i = 0; i < mergeCtx.numValidMergeCand; i++ )
1920
0
  {
1921
0
    mrgPredBufNoCiip.push_back( m_acMergeTmpBuffer[i].getCompactBuf( localUnitArea ) );
1922
0
  }
1923
1924
0
  int numMergeSatdCand = std::min( bestCS->area.lumaSize().area() >= 64 ? m_pcEncCfg->m_mergeRdCandQuotaRegular : m_pcEncCfg->m_mergeRdCandQuotaRegularSmallBlk, mergeCtx.numValidMergeCand );
1925
1926
0
  bool isCiipEnabled  = sps.CIIP && bestCS->area.lumaSize().area() >= 64 && bestCS->area.lumaSize().maxDim() < MAX_CU_SIZE;
1927
0
       isCiipEnabled &= m_pcEncCfg->m_CIIP <= 1 || !m_modeCtrl.getBlkInfo( tempCS->area ).isSkip; //5
1928
1929
0
  if( isCiipEnabled )
1930
0
  {
1931
0
    numMergeSatdCand += std::min( m_pcEncCfg->m_mergeRdCandQuotaCiip, mergeCtx.numValidMergeCand );
1932
0
  }
1933
1934
0
  const bool affineMrgAvail = ( m_pcEncCfg->m_Affine <= 2 || slice.TLayer <= 3 || m_pcEncCfg->m_SbTMVP )
1935
0
                           && ( m_pcEncCfg->m_Affine || sps.SbtMvp ) && m_pcEncCfg->m_maxNumAffineMergeCand && bestCS->area.Y().minDim() >= 8;
1936
1937
0
  if( affineMrgAvail )
1938
0
  {
1939
0
    CU::getAffineMergeCand( *cu, affineMergeCtx );
1940
0
    numMergeSatdCand += std::min( m_pcEncCfg->m_mergeRdCandQuotaSubBlk, affineMergeCtx.numValidMergeCand );
1941
0
  }
1942
1943
0
  int numSatdCandPreGeo = std::min( numMergeSatdCand, m_pcEncCfg->m_maxMergeRdCandNumTotal );
1944
0
  bool toAddGpmCand     = false;
1945
0
  if( sps.GEO && slice.isInterB() // base checks
1946
0
      && cu->lumaSize().minDim() >= GEO_MIN_CU_SIZE  && cu->lumaSize().maxDim() <= GEO_MAX_CU_SIZE && cu->lumaSize().maxDim() < 8 * cu->lumaSize().minDim() // size checks
1947
0
      && !( m_pcEncCfg->m_Geo > 2 && slice.TLayer <= 1 ) ) // speedups
1948
0
  {
1949
0
    cu->mergeFlag            = true;
1950
0
    cu->geo                  = true;
1951
0
    CU::getGeoMergeCandidates( *cu, gpmMergeCtx );
1952
0
    toAddGpmCand             = prepareGpmComboList( gpmMergeCtx, localUnitArea, sqrtLambdaForFirstPass, comboList, geoBuffer, *cu );
1953
0
    numMergeSatdCand        += toAddGpmCand ? std::min( m_pcEncCfg->m_mergeRdCandQuotaGpm, ( int ) comboList.list.size() ) : 0;
1954
0
  }
1955
1956
0
  numMergeSatdCand  = std::min( numMergeSatdCand, m_pcEncCfg->m_maxMergeRdCandNumTotal );
1957
1958
  // 1. Pass: get SATD-cost for selected candidates and reduce their count
1959
0
  m_mergeItemList.resetList( numMergeSatdCand );
1960
0
  const TempCtx ctxStart   ( m_CtxCache, m_CABACEstimator->getCtx() );
1961
0
  const DFunc   dfunc      = encTestMode.lossless ? DF_SAD : ( m_pcEncCfg->m_fastHad ? DF_HAD_fast : DF_HAD );
1962
0
  DistParam     distParam  = m_cRdCost.setDistParam( tempCS->getOrgBuf().Y(), tempCS->getOrgBuf().Y(), sps.bitDepths[CH_L], dfunc );
1963
0
  m_uiSadBestForQPA        = MAX_DISTORTION;
1964
1965
0
  addRegularCandsToPruningList( mergeCtx, localUnitArea, sqrtLambdaForFirstPass, ctxStart, distParam, *cu, sameMV, mrgPredBufNoCiip );
1966
1967
  // add CIIP candidates directly after adding regular cands
1968
0
  if( isCiipEnabled )
1969
0
  {
1970
0
    addCiipCandsToPruningList( mergeCtx, localUnitArea, sqrtLambdaForFirstPass, ctxStart, distParam, *cu, sameMV );
1971
0
  }
1972
1973
0
  if( sps.MMVD && !!m_mergeItemList.size() && !( m_pcEncCfg->m_useFastMrg >= 2 && m_mergeItemList.size() <= 1 ) )
1974
0
  {
1975
0
    addMmvdCandsToPruningList( mergeCtx, localUnitArea, sqrtLambdaForFirstPass, ctxStart, distParam, *cu );
1976
0
  }
1977
1978
0
  if( affineMergeCtx.numValidMergeCand > 0 )
1979
0
  {
1980
0
    addAffineCandsToPruningList( affineMergeCtx, localUnitArea, sqrtLambdaForFirstPass, ctxStart, distParam, *cu );
1981
0
  }
1982
1983
0
  if( m_pcEncCfg->m_useFastMrg > 0 && m_mergeItemList.size() > 0 )
1984
0
  {
1985
0
    m_mergeBestSATDCost    = m_mergeItemList.getMergeItemInList( 0 )->cost;
1986
0
    const double threshold = m_mergeBestSATDCost * MRG_FAST_RATIO[tempCS->picture->useFastMrg];
1987
0
    const   int shrinkSize = std::min( numSatdCandPreGeo, ( int ) updateRdCheckingNum( m_mergeItemList, threshold, numMergeSatdCand ) );
1988
0
    m_mergeItemList        . shrinkList( shrinkSize );
1989
0
  }
1990
0
  else
1991
0
  {
1992
0
    m_mergeItemList        . shrinkList( numSatdCandPreGeo );
1993
0
  }
1994
1995
0
  if( toAddGpmCand )
1996
0
  {
1997
0
    addGpmCandsToPruningList( gpmMergeCtx, localUnitArea, sqrtLambdaForFirstPass, ctxStart, comboList, geoBuffer, distParam, *cu );
1998
0
  }
1999
2000
0
  if(    m_pcEncCfg->m_internalUsePerceptQPATempFiltISlice == 2 && m_uiSadBestForQPA < MAX_DISTORTION && slice.TLayer == 0 // non-Intra key-frame
2001
0
      && m_pcEncCfg->m_salienceBasedOpt
2002
0
      && m_pcEncCfg->m_usePerceptQPA && partitioner.currQgEnable() && partitioner.currSubdiv == 0 ) // CTU-level luma quantization group
2003
0
  {
2004
0
    CHECK( bestCS->cost < MAX_DOUBLE, "This has to be the first test performed!" );
2005
2006
0
    const Picture *pic         = slice.pic;
2007
0
    const bool     isBIM       = m_pcEncCfg->m_RCNumPasses != 2 && m_pcEncCfg->m_blockImportanceMapping && !pic->m_picShared->m_ctuBimQpOffset.empty();
2008
0
    const uint32_t rsAddr      = getCtuAddr( partitioner.currQgPos, *pic->cs->pcv );
2009
0
    const int      pumpReducQP = BitAllocation::getCtuPumpingReducingQP( &slice, tempCS->getOrgBuf( COMP_Y ), m_uiSadBestForQPA, *m_globalCtuQpVector, rsAddr,
2010
0
                                                                         m_pcEncCfg->m_QP, isBIM );
2011
2012
0
    if( pumpReducQP != 0 ) // subtract QP offset, reduces Intra-period pumping or overcoding
2013
0
    {
2014
0
      encTestMode.qp = Clip3( 0, MAX_QP, encTestMode.qp - pumpReducQP );
2015
0
      tempCS->currQP[partitioner.chType] = tempCS->baseQP =
2016
0
      bestCS->currQP[partitioner.chType] = bestCS->baseQP = Clip3( 0, MAX_QP, tempCS->baseQP - pumpReducQP );
2017
2018
0
      updateLambda( slice, pic->ctuQpaLambda[rsAddr], pic->ctuAdaptedQP[rsAddr], tempCS->baseQP, true );
2019
0
    }
2020
0
  }
2021
2022
  // Try to limit number of candidates using SATD-costs
2023
0
  if( m_pcEncCfg->m_useFastMrg > 0 && m_mergeItemList.size() > 0 )
2024
0
  {
2025
    // shrink GEO list as well
2026
0
    const double threshold = m_mergeItemList.getMergeItemInList( 0 )->cost * MRG_FAST_RATIO[0];
2027
0
    numMergeSatdCand       = updateRdCheckingNum( m_mergeItemList, threshold, numMergeSatdCand );
2028
0
    m_mergeBestSATDCost    = m_mergeItemList.size() != 0 ? m_mergeItemList.getMergeItemInList( 0 )->cost : MAX_DOUBLE;
2029
0
  }
2030
0
  else
2031
0
  {
2032
0
    numMergeSatdCand       = std::min<int>( numMergeSatdCand, ( int ) m_mergeItemList.size() );
2033
0
  }
2034
2035
  // 2. Pass: RD checking 
2036
0
  tempCS->initStructData( encTestMode.qp );
2037
0
  m_CABACEstimator->getCtx() = ctxStart;
2038
2039
0
  double bestEndCost                            =   MAX_DOUBLE;
2040
0
  bool bestIsSkip                               =   false;
2041
0
  PelUnitBuf ciipBuf                            =   m_aTmpStorageLCU[1].getCompactBuf( *cu );
2042
0
  bool ciipChromaDone                           =   false;
2043
0
  bool isRegularTestedAsSkip[MRG_MAX_NUM_CANDS] = { false, };
2044
0
  bool geoWasTested                             =   false;
2045
0
  int  stopCand                                 =   numMergeSatdCand;
2046
2047
0
  CHECK( numMergeSatdCand > 0 && m_mergeItemList.size() == 0, "Empty merge item list is not expected" );
2048
2049
0
  for( uint32_t noResidualPass = 0; noResidualPass < 2; noResidualPass++ )
2050
0
  {
2051
0
    const bool forceNoResidual = noResidualPass == 1;
2052
0
    for( uint32_t mrgHadIdx = 0; mrgHadIdx < stopCand; mrgHadIdx++ )
2053
0
    {
2054
0
      auto mergeItem = m_mergeItemList.getMergeItemInList( mrgHadIdx );
2055
0
      CHECK( mergeItem == nullptr, "Wrong merge item" );
2056
2057
0
      const bool isCiip = mergeItem->mergeItemType == MergeItem::MergeItemType::CIIP;
2058
0
      const bool isGeo  = mergeItem->mergeItemType == MergeItem::MergeItemType::GPM;
2059
0
      const bool isRglr = mergeItem->mergeItemType == MergeItem::MergeItemType::REGULAR;
2060
0
      const bool isMmvd = mergeItem->mergeItemType == MergeItem::MergeItemType::MMVD;
2061
2062
0
      if( noResidualPass != 0 && isCiip && isRegularTestedAsSkip[mergeItem->mergeIdx] )
2063
0
      {
2064
0
        continue;
2065
0
      }
2066
2067
0
      if( noResidualPass ? mergeItem->noResidual : bestIsSkip )
2068
0
      {
2069
0
        continue;
2070
0
      }
2071
2072
0
      if( isGeo )
2073
0
      {
2074
0
        if( m_pcEncCfg->m_Geo > 2 && geoWasTested && !bestCS->cus.empty() && !bestCS->getCU( partitioner.chType, partitioner.treeType )->geo )
2075
0
        {
2076
0
          continue;
2077
0
        }
2078
2079
0
        geoWasTested = true;
2080
0
      }
2081
2082
0
      cu = getCuForInterPrediction( tempCS, encTestMode );
2083
0
      partitioner.setCUData( *cu );
2084
0
      const bool resetCiip2Regular = mergeItem->exportMergeInfo( *cu, forceNoResidual );
2085
2086
0
      if( isRglr || resetCiip2Regular )
2087
0
      {
2088
0
        if( CU::checkDMVRCondition( *cu ) ) std::copy_n( m_subPuMvOffset[mergeItem->mergeIdx].data(), getDmvrMvdNum( *cu ), cu->mvdL0SubPu );
2089
0
      }
2090
2091
0
      if( isMmvd && mergeItem->noBdofRefine )
2092
0
      {
2093
        // no BDOF refinement was made for the luma prediction, need to have luma prediction again
2094
0
        mergeItem->lumaPredReady = false;
2095
0
      }
2096
2097
0
      PelUnitBuf *predBuf1   = nullptr, *predBuf2 = isCiip ? &ciipBuf : nullptr;
2098
0
      PelUnitBuf  dstPredBuf = tempCS->getPredBuf( *cu );
2099
2100
0
      if( isGeo )
2101
0
      {
2102
0
        predBuf1 = &geoBuffer[cu->geoMergeIdx[0]];
2103
0
        predBuf2 = &geoBuffer[cu->geoMergeIdx[1]];
2104
0
      }
2105
2106
0
      if( resetCiip2Regular )
2107
0
      {
2108
0
        dstPredBuf.copyFrom( mrgPredBufNoCiip[mergeItem->mergeIdx] );
2109
0
      }
2110
0
      else
2111
0
      {
2112
0
        if( isCiip && !resetCiip2Regular && isChromaEnabled( cu->chromaFormat ) && cu->chromaSize().width > 2 )
2113
0
        {
2114
0
          if( !ciipChromaDone )
2115
0
          {
2116
0
            cu->intraDir[0] = PLANAR_IDX;
2117
0
            cu->intraDir[1] = DM_CHROMA_IDX;
2118
2119
0
            m_cIntraSearch  . initIntraPatternChType( *cu, cu->Cb() );
2120
0
            m_cIntraSearch  . predIntraAng          ( COMP_Cb, ciipBuf.Cb(), *cu );
2121
0
            m_cIntraSearch  . initIntraPatternChType( *cu, cu->Cr() );
2122
0
            m_cIntraSearch  . predIntraAng          ( COMP_Cr, ciipBuf.Cr(), *cu );
2123
2124
0
            ciipChromaDone  = true;
2125
0
          }
2126
0
        }
2127
2128
0
        if(  mergeItem->lumaPredReady ||  mergeItem->chromaPredReady )
2129
0
          dstPredBuf.copyFrom( mergeItem->getPredBuf( localUnitArea ), mergeItem->lumaPredReady, mergeItem->chromaPredReady );
2130
0
        if( !mergeItem->lumaPredReady || !mergeItem->chromaPredReady )
2131
0
          generateMergePrediction( localUnitArea, mergeItem, *cu, !mergeItem->lumaPredReady, !mergeItem->chromaPredReady, dstPredBuf, true, forceNoResidual, predBuf1, predBuf2 );
2132
0
      }
2133
2134
0
      if( !cu->mmvdSkip && !cu->ciip && !cu->affine && !cu->geo && noResidualPass != 0 )
2135
0
      {
2136
0
        CHECK( mergeItem->mergeIdx >= mergeCtx.numValidMergeCand, "out of normal merge" );
2137
0
        isRegularTestedAsSkip[mergeItem->mergeIdx] = true;
2138
0
      }
2139
2140
0
      xEncodeInterResidual( tempCS, bestCS, partitioner, encTestMode, noResidualPass, noResidualPass == 0 ? &mergeItem->noResidual : nullptr );
2141
2142
0
      if( m_pcEncCfg->m_useFastMrg >= 2 )
2143
0
      {
2144
0
        if( cu->ciip && bestCS->cost == MAX_DOUBLE && mrgHadIdx + 1 == numMergeSatdCand )
2145
0
        {
2146
0
          numMergeSatdCand = ( unsigned ) m_mergeItemList.size();
2147
0
        }
2148
      
2149
0
        if( mrgHadIdx > 0 && tempCS->cost >= bestEndCost && !cu->ciip && !isGeo )
2150
0
        {
2151
0
          stopCand = mrgHadIdx + 1;
2152
0
        }
2153
      
2154
0
        if( noResidualPass == 0 )
2155
0
        {
2156
0
          bestEndCost = std::min( bestEndCost, tempCS->cost );
2157
0
        }
2158
0
      }
2159
2160
0
      if( m_pcEncCfg->m_useFastDecisionForMerge && !bestIsSkip && !cu->ciip )
2161
0
      {
2162
0
        bestIsSkip = !bestCS->cus.empty() && bestCS->getCU( partitioner.chType, partitioner.treeType )->rootCbf == 0;
2163
0
      }
2164
2165
0
      tempCS->initStructData( encTestMode.qp );
2166
0
    }   // end loop mrgHadIdx
2167
0
  }
2168
0
}
2169
2170
unsigned int EncCu::updateRdCheckingNum( MergeItemList &mergeItemList, double threshold, unsigned int numMergeSatdCand )
2171
0
{
2172
0
  for( uint32_t i = 0; i < mergeItemList.size(); i++ )
2173
0
  {
2174
0
    const auto mergeItem = mergeItemList.getMergeItemInList( i );
2175
0
    if( mergeItem == nullptr || mergeItem->cost > threshold )
2176
0
    {
2177
0
      numMergeSatdCand = i;
2178
0
      break;
2179
0
    }
2180
0
  }
2181
0
  return std::min( numMergeSatdCand, ( unsigned ) mergeItemList.size() );
2182
0
}
2183
2184
void EncCu::generateMergePrediction( const UnitArea &unitArea, MergeItem *mergeItem, CodingUnit &pu, bool luma, bool chroma,
2185
                                     PelUnitBuf &dstBuf, bool finalRd, bool forceNoResidual, PelUnitBuf *predBuf1, PelUnitBuf *predBuf2 )
2186
0
{
2187
0
  CHECK( ( luma && mergeItem->lumaPredReady ) || ( chroma && mergeItem->chromaPredReady ), "Prediction has been avaiable" );
2188
2189
0
  pu.mcControl = ( !luma ? 4 : 0 ) | ( !chroma ? 2 : 0 );
2190
2191
0
  switch( mergeItem->mergeItemType )
2192
0
  {
2193
0
  case MergeItem::MergeItemType::REGULAR:
2194
    // here predBuf1 is predBufNoCiip
2195
0
    pu.mvRefine = true;
2196
0
    m_cInterSearch.motionCompensation( pu, dstBuf, REF_PIC_LIST_X );
2197
0
    pu.mvRefine = false;
2198
0
    if( predBuf1 != nullptr )
2199
0
    {
2200
0
      predBuf1->copyFrom( dstBuf, luma, chroma );
2201
0
    }
2202
0
    break;
2203
2204
0
  case MergeItem::MergeItemType::CIIP:
2205
0
    m_cInterSearch.motionCompensation( pu, dstBuf, REF_PIC_LIST_X );
2206
2207
0
    if( luma )
2208
0
    {
2209
      // generate intrainter Y prediction
2210
0
      dstBuf.Y().weightCiip( predBuf2->Y(), mergeItem->numCiipIntra );
2211
0
    }
2212
2213
0
    if( chroma )
2214
0
    {
2215
0
      if( pu.chromaSize().width > 2 )
2216
0
      {
2217
0
        dstBuf.Cb().weightCiip( predBuf2->Cb(), mergeItem->numCiipIntra );
2218
0
        dstBuf.Cr().weightCiip( predBuf2->Cr(), mergeItem->numCiipIntra );
2219
0
      }
2220
0
    }
2221
2222
0
    break;
2223
2224
0
  case MergeItem::MergeItemType::MMVD:
2225
0
    pu.mcControl           |= finalRd ? 0 : ( pu.mmvdMergeIdx.pos.step > 2 || m_pcEncCfg->m_MMVD > 1 ) ? 1 : 0;
2226
0
    mergeItem->noBdofRefine = pu.mccNoBdof() && pu.cs->sps->BDOF && !pu.cs->picHeader->disBdofFlag;
2227
0
    m_cInterSearch.motionCompensation( pu, dstBuf, REF_PIC_LIST_X );
2228
0
    break;
2229
2230
0
  case MergeItem::MergeItemType::SBTMVP:
2231
0
    m_cInterSearch.motionCompensation( pu, dstBuf, REF_PIC_LIST_X );
2232
0
    break;
2233
2234
0
  case MergeItem::MergeItemType::AFFINE:
2235
0
    m_cInterSearch.motionCompensation( pu, dstBuf, REF_PIC_LIST_X );
2236
0
    break;
2237
2238
0
  case MergeItem::MergeItemType::GPM:
2239
    // here predBuf1 and predBuf2 point to geoBuffer[mergeCand0] and geoBuffer[mergeCand1], respectively
2240
0
    CHECK( predBuf1 == nullptr || predBuf2 == nullptr, "Invalid input buffer to GPM" );
2241
0
    m_cInterSearch.weightedGeoBlk( pu.slice->clpRngs, pu, pu.geoSplitDir, luma && chroma ? MAX_NUM_CH : luma ? CH_L : CH_C, dstBuf, *predBuf1, *predBuf2 );
2242
0
    break;
2243
2244
0
  default:
2245
0
    THROW("Wrong merge item type");
2246
0
  }
2247
2248
0
  auto mergeItemPredBuf = mergeItem->getPredBuf( unitArea );
2249
2250
0
  if( dstBuf.Y().buf == mergeItemPredBuf.Y().buf )
2251
0
  {
2252
    // dst is the internal buffer
2253
0
    mergeItem->lumaPredReady   |= luma;
2254
0
    mergeItem->chromaPredReady |= chroma;
2255
0
  }
2256
0
  else if( finalRd && !forceNoResidual )
2257
0
  {
2258
    // at final RD stage, with and without residuals are both checked
2259
    // it makes sense to buffer the prediction
2260
0
    mergeItemPredBuf.copyFrom( dstBuf, luma, chroma );
2261
0
    mergeItem->lumaPredReady   |= luma;
2262
0
    mergeItem->chromaPredReady |= chroma;
2263
0
  }
2264
0
}
2265
2266
void EncCu::addRegularCandsToPruningList( const MergeCtx &mergeCtx, const UnitArea &localUnitArea, double sqrtLambdaForFirstPassIntra, const TempCtx &ctxStart,
2267
                                          DistParam& distParam, CodingUnit& pu, bool* sameMv, MergeBufVector& regularPred )
2268
0
{
2269
0
  pu.geo = pu.affine
2270
0
         = pu.mmvdMergeFlag = pu.mmvdSkip
2271
0
         = pu.ciip
2272
0
         = false;
2273
2274
0
  for( uint32_t uiMergeCand = 0; uiMergeCand < mergeCtx.numValidMergeCand; uiMergeCand++ )
2275
0
  {
2276
0
    if( sameMv[uiMergeCand] ) continue;
2277
2278
0
    mergeCtx.setMergeInfo   ( pu, uiMergeCand );
2279
2280
0
    if( m_pcEncCfg->m_ifpLines && // what about DMVR?
2281
0
        ( ( pu.refIdx[L0] >= 0 && !CU::isMvInRangeFPP( pu.ly(), pu.lheight(), pu.mv[L0][0].ver, m_pcEncCfg->m_ifpLines, *pu.cs->pcv ) ) ||
2282
0
          ( pu.refIdx[L1] >= 0 && !CU::isMvInRangeFPP( pu.ly(), pu.lheight(), pu.mv[L1][0].ver, m_pcEncCfg->m_ifpLines, *pu.cs->pcv ) ) ) )
2283
0
    {
2284
0
      continue;
2285
0
    }
2286
2287
0
    pu.interDir             = mergeCtx.interDirNeighbours[uiMergeCand];
2288
0
    pu.BcwIdx               = pu.interDir == 3 ? mergeCtx.BcwIdx[uiMergeCand] : BCW_DEFAULT;
2289
0
    pu.imv                  = mergeCtx.useAltHpelIf[uiMergeCand] ? IMV_HPEL : IMV_OFF;
2290
0
    CU::spanMotionInfo      ( pu );
2291
2292
0
    MergeItem *regularMerge = m_mergeItemList.allocateNewMergeItem();
2293
0
    regularMerge->importMergeInfo( mergeCtx, uiMergeCand, MergeItem::MergeItemType::REGULAR, pu );
2294
0
    auto dstBuf             = regularMerge->getPredBuf( localUnitArea );
2295
0
    generateMergePrediction ( localUnitArea, regularMerge, pu, true, true, dstBuf, false, false, &regularPred[uiMergeCand], nullptr );
2296
0
    regularMerge->cost      = calcLumaCost4MergePrediction( ctxStart, dstBuf, sqrtLambdaForFirstPassIntra, pu, distParam );
2297
0
    if( CU::checkDMVRCondition( pu ) ) std::copy_n( pu.mvdL0SubPu, getDmvrMvdNum( pu ), m_subPuMvOffset[uiMergeCand].data() );
2298
0
    m_mergeItemList         . insertMergeItemToList( regularMerge );
2299
0
  }
2300
0
}
2301
2302
void EncCu::addCiipCandsToPruningList( const MergeCtx &mergeCtx, const UnitArea &localUnitArea, double sqrtLambdaForFirstPassIntra, const TempCtx &ctxStart, DistParam &distParam, CodingUnit &pu, bool* sameMv )
2303
0
{
2304
0
  int                numCiipIntra = -1;
2305
0
  PelUnitBuf         ciipBuf      = m_aTmpStorageLCU[1].getCompactBuf( pu );
2306
2307
0
  pu.ciip        = true;
2308
0
  pu.intraDir[0] = PLANAR_IDX;
2309
0
  pu.geo         = pu.affine
2310
0
                 = pu.mmvdMergeFlag = pu.mmvdSkip
2311
0
                 = false;
2312
0
  m_cIntraSearch . initIntraPatternChType        ( pu, pu.Y() );
2313
0
  m_cIntraSearch . predIntraAng                  ( COMP_Y, ciipBuf.Y(), pu );
2314
0
  numCiipIntra   = m_cIntraSearch.getNumIntraCiip( pu );
2315
2316
0
  int nonCiipMrgCnds[MRG_MAX_NUM_CANDS] = { 0, };
2317
0
  int numNonCiipCnds                    =   0;
2318
0
  for( ; numNonCiipCnds < m_mergeItemList.size(); numNonCiipCnds++ ) nonCiipMrgCnds[numNonCiipCnds] = m_mergeItemList.getMergeItemInList( numNonCiipCnds )->mergeIdx;
2319
2320
0
  for( int i = 0; i < numNonCiipCnds; i++ )
2321
0
  {
2322
0
    const unsigned int uiMergeCand = nonCiipMrgCnds[i];
2323
2324
0
    if( sameMv[uiMergeCand] ) continue;
2325
2326
0
    mergeCtx.setMergeInfo     ( pu, uiMergeCand );
2327
2328
0
    if( m_pcEncCfg->m_ifpLines && 
2329
0
        ( ( pu.refIdx[L0] >= 0 && !CU::isMvInRangeFPP( pu.ly(), pu.lheight(), pu.mv[L0][0].ver, m_pcEncCfg->m_ifpLines, *pu.cs->pcv ) ) ||
2330
0
          ( pu.refIdx[L1] >= 0 && !CU::isMvInRangeFPP( pu.ly(), pu.lheight(), pu.mv[L1][0].ver, m_pcEncCfg->m_ifpLines, *pu.cs->pcv ) ) ) )
2331
0
    {
2332
0
      continue;
2333
0
    }
2334
2335
0
    pu.interDir               = mergeCtx.interDirNeighbours[uiMergeCand];
2336
0
    pu.BcwIdx                 = pu.interDir == 3 ? mergeCtx.BcwIdx[uiMergeCand] : BCW_DEFAULT;
2337
0
    pu.imv                    = mergeCtx.useAltHpelIf[uiMergeCand] ? IMV_HPEL : IMV_OFF;
2338
0
    CU::spanMotionInfo        ( pu );
2339
2340
0
    MergeItem* ciipMerge      = m_mergeItemList.allocateNewMergeItem();
2341
0
    ciipMerge->importMergeInfo( mergeCtx, uiMergeCand, MergeItem::MergeItemType::CIIP, pu );
2342
0
    ciipMerge->numCiipIntra   = numCiipIntra;
2343
0
    auto dstBuf               = ciipMerge->getPredBuf( localUnitArea );
2344
0
    generateMergePrediction   ( localUnitArea, ciipMerge, pu, true, false, dstBuf, false, false, nullptr, &ciipBuf );
2345
0
      ciipMerge->cost         = calcLumaCost4MergePrediction( ctxStart, dstBuf, sqrtLambdaForFirstPassIntra, pu, distParam );
2346
0
    if( !m_mergeItemList      . insertMergeItemToList( ciipMerge ) && m_pcEncCfg->m_CIIP > 1 )
2347
0
    {
2348
0
      break;
2349
0
    }
2350
0
  }
2351
0
}
2352
2353
void EncCu::addMmvdCandsToPruningList( const MergeCtx &mergeCtx, const UnitArea &localUnitArea, double sqrtLambdaForFirstPassIntra, const TempCtx& ctxStart,
2354
                                       DistParam& distParam, CodingUnit& pu )
2355
0
{
2356
0
  pu.mmvdSkip              = true;
2357
0
  pu.affine                = pu.geo
2358
0
                           = pu.ciip
2359
0
                           = false;
2360
2361
0
  int       mmvdTestNum    = mergeCtx.numValidMergeCand > 1 ? MmvdIdx::ADD_NUM : MmvdIdx::ADD_NUM >> 1;
2362
0
  int       bestDir        = 0;
2363
0
  size_t    curListSize    = m_mergeItemList.size();
2364
0
  double    bestCostMerge  = m_mergeItemList.getMergeItemInList( curListSize - 1 )->cost;
2365
0
  double    bestCostOffset = MAX_DOUBLE;
2366
0
  int       shiftCandStart = 0;
2367
2368
0
  if( m_pcEncCfg->m_MMVD == 4 )
2369
0
  {
2370
0
    const int cnd1idx = m_mergeItemList.size() == 1 ? 0 : 1;
2371
0
    const int mrgCnd0 = m_mergeItemList.getMergeItemInList(       0 )->mergeIdx;
2372
0
    const int mrgCnd1 = m_mergeItemList.getMergeItemInList( cnd1idx )->mergeIdx;
2373
2374
0
    if( mrgCnd0 > 1 && mrgCnd1 > 1 )
2375
0
    {
2376
0
      mmvdTestNum = 0;
2377
0
    }
2378
0
    else if( mrgCnd0 > 1 || mrgCnd1 > 1 )
2379
0
    {
2380
0
      int shiftCand = mrgCnd0 < 2 ? mrgCnd0 : mrgCnd1;
2381
2382
0
      if( shiftCand )
2383
0
      {
2384
0
        shiftCandStart = MMVD_MAX_REFINE_NUM;
2385
0
      }
2386
0
      else
2387
0
      {
2388
0
        mmvdTestNum    = MMVD_MAX_REFINE_NUM;
2389
0
      }
2390
0
    }
2391
0
  }
2392
2393
0
  for( int mmvdMergeCand = shiftCandStart; mmvdMergeCand < mmvdTestNum; mmvdMergeCand++ )
2394
0
  {
2395
0
    MmvdIdx mmvdIdx;
2396
0
    mmvdIdx.val = mmvdMergeCand;
2397
2398
0
    if( mmvdIdx.pos.step >= m_pcEncCfg->m_MmvdDisNum )
2399
0
    {
2400
0
      continue;
2401
0
    }
2402
2403
0
    if( m_pcEncCfg->m_MMVD > 1 )
2404
0
    {
2405
0
      int checkMMVD = xCheckMMVDCand( mmvdIdx, bestDir, mmvdTestNum, bestCostOffset, bestCostMerge, m_mergeItemList.getMergeItemInList( curListSize - 1 )->cost );
2406
0
      mmvdMergeCand = mmvdIdx.val;
2407
2408
0
      if( checkMMVD )
2409
0
      {
2410
0
        if( checkMMVD == 2 )
2411
0
        {
2412
0
          break;
2413
0
        }
2414
0
        continue;
2415
0
      }
2416
0
    }
2417
2418
0
    mergeCtx.setMmvdMergeCandiInfo( pu, mmvdIdx );
2419
2420
0
    if( m_pcEncCfg->m_ifpLines &&
2421
0
        ( ( pu.refIdx[L0] >= 0 && !CU::isMvInRangeFPP( pu.ly(), pu.lheight(), pu.mv[L0][0].ver, m_pcEncCfg->m_ifpLines, *pu.cs->pcv ) ) ||
2422
0
          ( pu.refIdx[L1] >= 0 && !CU::isMvInRangeFPP( pu.ly(), pu.lheight(), pu.mv[L1][0].ver, m_pcEncCfg->m_ifpLines, *pu.cs->pcv ) ) ) )
2423
0
    {
2424
      // skip candidate
2425
0
      continue;
2426
0
    }
2427
2428
0
    pu.interDir               = mergeCtx.interDirNeighbours[mmvdIdx.pos.baseIdx];
2429
0
    pu.BcwIdx                 = pu.interDir == 3 ? mergeCtx.BcwIdx[mmvdIdx.pos.baseIdx] : BCW_DEFAULT;
2430
0
    pu.imv                    = mergeCtx.useAltHpelIf[mmvdIdx.pos.baseIdx] ? IMV_HPEL : IMV_OFF;
2431
0
    CU::spanMotionInfo        ( pu );
2432
2433
0
    MergeItem *mmvdMerge      = m_mergeItemList.allocateNewMergeItem();
2434
0
    mmvdMerge->importMergeInfo( mergeCtx, mmvdIdx.val, MergeItem::MergeItemType::MMVD, pu );
2435
0
    auto dstBuf               = mmvdMerge->getPredBuf( localUnitArea );
2436
0
    generateMergePrediction   ( localUnitArea, mmvdMerge, pu, true, false, dstBuf, false, false, nullptr, nullptr );
2437
0
    mmvdMerge->cost           = calcLumaCost4MergePrediction( ctxStart, dstBuf, sqrtLambdaForFirstPassIntra, pu, distParam );
2438
0
    m_mergeItemList           . insertMergeItemToList( mmvdMerge );
2439
2440
0
    if( m_pcEncCfg->m_MMVD > 1 && mmvdMerge->cost < bestCostOffset )
2441
0
    {
2442
0
      bestCostOffset          = mmvdMerge->cost;
2443
0
      int CandCur             = mmvdIdx.val - MMVD_MAX_REFINE_NUM * mmvdIdx.pos.baseIdx;
2444
0
      if( CandCur < 4 )
2445
0
        bestDir               = CandCur;
2446
0
    }
2447
0
  }
2448
2449
0
  if( m_pcEncCfg->m_useFastMrg >= 2 )
2450
0
  {
2451
0
    m_mergeItemList           . shrinkList( curListSize );
2452
0
  }
2453
0
}
2454
2455
void EncCu::addAffineCandsToPruningList( AffineMergeCtx &affineMergeCtx, const UnitArea &localUnitArea, double sqrtLambdaForFirstPass,
2456
                                         const TempCtx& ctxStart, DistParam& distParam, CodingUnit& pu)
2457
0
{
2458
0
  bool sameMV[AFFINE_MRG_MAX_NUM_CANDS + 1]
2459
0
                      = { false, };
2460
0
  size_t curListSize  = m_mergeItemList.size();
2461
2462
0
  pu.mergeFlag = true;
2463
0
  pu.affine    = true;
2464
0
  pu.imv       = IMV_OFF;
2465
0
  pu.geo       = pu.mmvdMergeFlag = pu.mmvdSkip
2466
0
               = pu.ciip
2467
0
               = false;
2468
2469
0
  if( m_pcEncCfg->m_Affine > 1 )
2470
0
  {
2471
0
    for( int m = 0; m < affineMergeCtx.numValidMergeCand; m++ )
2472
0
    {
2473
0
      if( pu.cs->slice->TLayer > 3 && affineMergeCtx.mergeType[m] != MRG_TYPE_SUBPU_ATMVP )
2474
0
      {
2475
0
        sameMV[m] = m != 0;
2476
0
      }
2477
0
      else if( !sameMV[m + 1] )
2478
0
      {
2479
0
        for( int n = m + 1; n < affineMergeCtx.numValidMergeCand; n++ )
2480
0
        {
2481
0
          sameMV[n] |= affineMergeCtx.mvFieldNeighbours[m][0][0] == affineMergeCtx.mvFieldNeighbours[n][0][0]
2482
0
                    && affineMergeCtx.mvFieldNeighbours[m][1][0] == affineMergeCtx.mvFieldNeighbours[n][1][0];
2483
0
        }
2484
0
      }
2485
0
    }
2486
0
  }
2487
2488
0
  for( uint32_t mergeIdx = 0; mergeIdx < affineMergeCtx.numValidMergeCand; mergeIdx++ )
2489
0
  {
2490
0
    if( ( affineMergeCtx.mergeType[mergeIdx] != MRG_TYPE_SUBPU_ATMVP && m_pcEncCfg->m_Affine == 0 ) || sameMV[mergeIdx] )
2491
0
    {
2492
0
      continue;
2493
0
    }
2494
2495
0
    pu.mergeType              = affineMergeCtx.mergeType[mergeIdx];
2496
0
    pu.affineType             = affineMergeCtx.affineType[mergeIdx];
2497
0
    pu.interDir               = affineMergeCtx.interDirNeighbours[mergeIdx];
2498
0
    pu.BcwIdx                 = pu.interDir == 3 ? affineMergeCtx.BcwIdx[mergeIdx] : BCW_DEFAULT;
2499
2500
    // generate motion buf for IFP
2501
0
    if( affineMergeCtx.mergeType[mergeIdx] == MRG_TYPE_SUBPU_ATMVP )
2502
0
    {
2503
0
      pu.refIdx[L0]           = affineMergeCtx.mvFieldNeighbours[mergeIdx][L0][0].refIdx;
2504
0
      pu.refIdx[L1]           = affineMergeCtx.mvFieldNeighbours[mergeIdx][L1][0].refIdx;
2505
0
      pu.mv    [L0][0]        = affineMergeCtx.mvFieldNeighbours[mergeIdx][L0][0].mv;
2506
0
      pu.mv    [L1][0]        = affineMergeCtx.mvFieldNeighbours[mergeIdx][L1][0].mv;
2507
0
      CU::spanMotionInfo      ( pu, &affineMergeCtx );
2508
0
    }
2509
0
    else
2510
0
    {
2511
0
      CU::setAllAffineMvField ( pu, affineMergeCtx.mvFieldNeighbours[mergeIdx][L0], L0 );
2512
0
      CU::setAllAffineMvField ( pu, affineMergeCtx.mvFieldNeighbours[mergeIdx][L1], L1 );
2513
0
      CU::spanMotionInfo      ( pu );
2514
0
    }
2515
2516
0
    if( m_pcEncCfg->m_ifpLines && !CU::isMotionBufInRangeFPP( pu, m_pcEncCfg->m_ifpLines ) )
2517
0
    {
2518
0
      continue;
2519
0
    }
2520
2521
0
    MergeItem *mergeItem   = m_mergeItemList.allocateNewMergeItem();
2522
0
    mergeItem->importMergeInfo( affineMergeCtx, mergeIdx, affineMergeCtx.mergeType[mergeIdx] == MRG_TYPE_SUBPU_ATMVP ? MergeItem::MergeItemType::SBTMVP : MergeItem::MergeItemType::AFFINE, pu );
2523
0
    auto dstBuf            = mergeItem->getPredBuf( localUnitArea );
2524
0
    generateMergePrediction( localUnitArea, mergeItem, pu, true, false, dstBuf, false, false, nullptr, nullptr );
2525
0
    mergeItem->cost        = calcLumaCost4MergePrediction( ctxStart, dstBuf, sqrtLambdaForFirstPass, pu, distParam );
2526
0
    m_mergeItemList        . insertMergeItemToList( mergeItem );
2527
0
  }
2528
0
  if( m_pcEncCfg->m_useFastMrg >= 2 )
2529
0
  {
2530
0
    m_mergeItemList        . shrinkList( curListSize );
2531
0
  }
2532
0
}
2533
2534
void EncCu::addGpmCandsToPruningList( const MergeCtx &mergeCtx, const UnitArea &localUnitArea, double sqrtLambdaForFirstPass,
2535
                                      const TempCtx& ctxStart, const GeoComboCostList& comboList, MergeBufVector& geoBuffer, DistParam& distParam, CodingUnit& pu)
2536
0
{
2537
0
  int geoNumMrgSadCand    = std::min( GEO_MAX_TRY_WEIGHTED_SAD, ( int ) comboList.list.size() );
2538
0
  geoNumMrgSadCand        = std::min( geoNumMrgSadCand, m_pcEncCfg->m_Geo > 2 ? 10 : GEO_MAX_TRY_WEIGHTED_SAD );
2539
0
  double bestGeoCost      = MAX_DOUBLE / 2.0;
2540
0
  MergeItem* best2geo[2]  = { nullptr, nullptr };
2541
2542
0
  pu.mergeFlag = true;
2543
0
  pu.geo       = true;
2544
0
  pu.mergeType = MRG_TYPE_DEFAULT_N;
2545
0
  pu.BcwIdx    = BCW_DEFAULT;
2546
0
  pu.interDir  = 3;
2547
0
  pu.imv       = IMV_OFF;
2548
0
  pu.affine    = pu.mmvdMergeFlag = pu.mmvdSkip
2549
0
               = pu.ciip
2550
0
               = false;
2551
2552
0
  for( int candidateIdx = 0; candidateIdx < geoNumMrgSadCand; candidateIdx++ )
2553
0
  {
2554
0
    const int          splitDir     = comboList.list[candidateIdx].splitDir;
2555
0
    const MergeIdxPair mergeIdxPair { comboList.list[candidateIdx].mergeIdx0, comboList.list[candidateIdx].mergeIdx1 };
2556
0
    const int          gpmIndex     = MergeItem::getGpmUnfiedIndex( splitDir, mergeIdxPair );
2557
2558
0
    pu.mergeIdx            = gpmIndex;
2559
0
    pu.geoMergeIdx         = mergeIdxPair;
2560
0
    pu.geoSplitDir         = splitDir;
2561
0
    CU::spanGeoMotionInfo  ( pu, mergeCtx, pu.geoSplitDir, pu.geoMergeIdx[0], pu.geoMergeIdx[1] );
2562
2563
0
    MergeItem *mergeItem   = m_mergeItemList.allocateNewMergeItem();
2564
0
    mergeItem->importMergeInfo( mergeCtx, gpmIndex, MergeItem::MergeItemType::GPM, pu );
2565
0
    auto dstBuf            = mergeItem->getPredBuf( localUnitArea );
2566
0
    generateMergePrediction( localUnitArea, mergeItem, pu, true, false, dstBuf, false, false, &geoBuffer[mergeIdxPair[0]], &geoBuffer[mergeIdxPair[1]] );
2567
0
    mergeItem->cost        = calcLumaCost4MergePrediction( ctxStart, dstBuf, sqrtLambdaForFirstPass, pu, distParam );
2568
0
    bestGeoCost            = std::min( mergeItem->cost, bestGeoCost );
2569
2570
0
    if( mergeItem->cost > MRG_FAST_RATIO[0] * bestGeoCost || mergeItem->cost > m_mergeBestSATDCost )
2571
0
    {
2572
0
      m_mergeItemList      . giveBackMergeItem( mergeItem );
2573
2574
0
      if( m_pcEncCfg->m_Geo > 2 ) break;
2575
0
    }
2576
0
    else if( m_pcEncCfg->m_Geo < 2 )
2577
0
    {
2578
0
      m_mergeItemList      . insertMergeItemToList( mergeItem );
2579
0
    }
2580
0
    else
2581
0
    {
2582
0
      if( ( m_mergeItemList.size() > 0 && m_mergeItemList.getMergeItemInList( m_mergeItemList.size() - 1 )->cost <= mergeItem->cost ) ||
2583
0
        ( best2geo[1] && best2geo[1]->cost <= mergeItem->cost ) )
2584
0
      {
2585
0
        m_mergeItemList    . giveBackMergeItem( mergeItem );
2586
0
      }
2587
0
      else
2588
0
      {
2589
0
        if( !best2geo[0] || mergeItem->cost < best2geo[0]->cost )
2590
0
        {
2591
0
          if( best2geo[1] )
2592
0
            m_mergeItemList. giveBackMergeItem( best2geo[1] );
2593
2594
0
          best2geo[1] = best2geo[0]; best2geo[0] = mergeItem;
2595
0
        }
2596
0
        else
2597
0
        {
2598
0
          if( best2geo[1] ) 
2599
0
            m_mergeItemList. giveBackMergeItem( best2geo[1] );
2600
2601
0
          best2geo[1] = mergeItem;
2602
0
        }
2603
0
      }
2604
0
    }
2605
0
  }
2606
2607
0
  if( best2geo[0] )
2608
0
    m_mergeItemList        . insertMergeItemToList( best2geo[0] );
2609
0
  if( best2geo[1] )
2610
0
    m_mergeItemList        . insertMergeItemToList( best2geo[1] );
2611
0
}
2612
2613
bool EncCu::prepareGpmComboList( const MergeCtx &mergeCtx, const UnitArea &localUnitArea, double sqrtLambdaForFirstPass,
2614
                                 GeoComboCostList& comboList, MergeBufVector& geoBuffer, CodingUnit& pu )
2615
0
{
2616
0
          sqrtLambdaForFirstPass /= FRAC_BITS_SCALE;
2617
0
  const int bitsForPartitionIdx   = floorLog2(GEO_NUM_PARTITION_MODE);
2618
0
  const int maxNumMergeCandidates = std::min( ( int ) pu.cs->sps->maxNumGeoCand, MRG_MAX_NUM_CANDS );
2619
0
  DistParam distParam;
2620
  // the second arguments to setDistParam is dummy and will be updated before being used
2621
0
  DistParam  distParamWholeBlk     = m_cRdCost.setDistParam( pu.cs->getOrgBuf().Y(), pu.cs->getOrgBuf().Y(), pu.cs->sps->bitDepths[ CH_L ], DF_SAD );
2622
0
  Distortion bestWholeBlkSad       = MAX_UINT64;
2623
0
  double     bestWholeBlkCost      = MAX_DOUBLE;
2624
0
  const ClpRng&  lclpRng           = pu.slice->clpRngs[COMP_Y];
2625
0
  const unsigned rshift            = std::max<int>( 2, ( IF_INTERNAL_PREC - lclpRng.bd ) );
2626
0
  const int      offset            = ( 1 << ( rshift - 1 ) ) + IF_INTERNAL_OFFS;
2627
0
  const int      numSamples        = pu.Y().area();
2628
0
  Distortion sadWholeBlk            [GEO_MAX_NUM_UNI_CANDS];
2629
0
  int        pocMrg                 [GEO_MAX_NUM_UNI_CANDS];
2630
0
  Mv         mergeMv                [GEO_MAX_NUM_UNI_CANDS];
2631
0
  bool       isSkipThisCand         [GEO_MAX_NUM_UNI_CANDS]
2632
0
                                   = { false, };
2633
0
  bool       sameMV                 [MRG_MAX_NUM_CANDS]
2634
0
                                   = { false, };
2635
0
  MergeBufVector geoTempBuf;
2636
2637
0
  if( m_pcEncCfg->m_Geo > 2 )
2638
0
  {
2639
0
    for( int m = 0; m < maxNumMergeCandidates; m++ )
2640
0
    {
2641
0
      if( !sameMV[m] )
2642
0
      {
2643
0
        for( int n = m + 1; n < maxNumMergeCandidates; n++ )
2644
0
        {
2645
0
          sameMV[n] |= mergeCtx.mvFieldNeighbours[m][0] == mergeCtx.mvFieldNeighbours[n][0]
2646
0
                    && mergeCtx.mvFieldNeighbours[m][1] == mergeCtx.mvFieldNeighbours[n][1];
2647
0
        }
2648
0
      }
2649
0
    }
2650
0
  }
2651
2652
0
  for( uint8_t mergeCand = 0; mergeCand < maxNumMergeCandidates; mergeCand++ )
2653
0
  {
2654
0
    geoBuffer .push_back ( m_aTmpStorageLCU[2                         + mergeCand].getCompactBuf( localUnitArea ) );
2655
0
    geoTempBuf.push_back ( m_aTmpStorageLCU[2 + GEO_MAX_NUM_UNI_CANDS + mergeCand].getCompactBuf( localUnitArea ) );
2656
2657
0
    const int  listIdx    = mergeCtx.mvFieldNeighbours[mergeCand][0]      .refIdx == -1 ? 1 : 0;
2658
0
    const auto refPicList = RefPicList(listIdx);
2659
0
    const int  refIdx     = mergeCtx.mvFieldNeighbours[mergeCand][listIdx].refIdx;
2660
2661
0
    pocMrg [mergeCand]    = pu.cs->slice->getRefPic( refPicList, refIdx )->poc;
2662
0
    mergeMv[mergeCand]    = mergeCtx.mvFieldNeighbours[mergeCand][listIdx].mv;
2663
2664
0
    for( int i = 0; i < mergeCand; i++ )
2665
0
    {
2666
0
      if( pocMrg[mergeCand] == pocMrg[i] && mergeMv[mergeCand] == mergeMv[i] )
2667
0
      {
2668
0
        isSkipThisCand[mergeCand] = true;
2669
0
        break;
2670
0
      }
2671
0
    }
2672
2673
0
    if( sameMV[mergeCand] )
2674
0
    {
2675
0
      continue;
2676
0
    }
2677
2678
0
    if( m_pcEncCfg->m_ifpLines ) 
2679
0
    {
2680
0
      bool isOutOfRange  = !CU::isMvInRangeFPP( pu.ly(), pu.lheight(), mergeCtx.mvFieldNeighbours[mergeCand][0].mv.ver, m_pcEncCfg->m_ifpLines, *pu.cs->pcv );
2681
0
           isOutOfRange |= !CU::isMvInRangeFPP( pu.ly(), pu.lheight(), mergeCtx.mvFieldNeighbours[mergeCand][1].mv.ver, m_pcEncCfg->m_ifpLines, *pu.cs->pcv );
2682
2683
      // use sameMV to surpress processing of this cand later on...
2684
0
      sameMV[mergeCand] |= isOutOfRange;
2685
2686
0
      if( isOutOfRange )
2687
0
        continue;
2688
0
    }
2689
2690
0
    mergeCtx.setMergeInfo            ( pu, mergeCand );
2691
0
    CU::spanMotionInfo               ( pu );
2692
0
    m_cInterSearch.motionCompensation( pu, geoBuffer[mergeCand], REF_PIC_LIST_X );
2693
2694
0
    g_pelBufOP.roundGeo( geoBuffer[mergeCand].Y().buf, geoTempBuf[mergeCand].Y().buf, numSamples, rshift, offset, lclpRng );
2695
2696
0
    distParamWholeBlk.cur  = geoTempBuf[mergeCand].Y();
2697
0
    sadWholeBlk[mergeCand] = distParamWholeBlk.distFunc( distParamWholeBlk );
2698
2699
0
    if( sadWholeBlk[mergeCand] < bestWholeBlkSad )
2700
0
    {
2701
0
      bestWholeBlkSad  = sadWholeBlk[mergeCand];
2702
0
      int bitsCand     = mergeCand + 1;
2703
0
      bestWholeBlkCost = ( double ) bestWholeBlkSad + ( double ) bitsCand * sqrtLambdaForFirstPass;
2704
0
    }
2705
0
  }
2706
2707
0
  bool allCandsAreSame = true;
2708
0
  for( uint8_t mergeCand = 1; mergeCand < maxNumMergeCandidates; mergeCand++ )
2709
0
  {
2710
0
    allCandsAreSame &= isSkipThisCand[mergeCand];
2711
0
  }
2712
0
  if( allCandsAreSame )
2713
0
  {
2714
0
    return false;
2715
0
  }
2716
2717
0
  const int wIdx = floorLog2( pu.lwidth() )  - GEO_MIN_CU_LOG2;
2718
0
  const int hIdx = floorLog2( pu.lheight() ) - GEO_MIN_CU_LOG2;
2719
2720
0
  for( int splitDir = 0; splitDir < GEO_NUM_PARTITION_MODE; )
2721
0
  {
2722
0
    int maskStride = 0, maskStride2 = 0;
2723
0
    int stepX = 1;
2724
0
    Pel *sadMask;
2725
0
    int16_t angle = g_GeoParams[splitDir][0];
2726
    
2727
0
    if( g_angle2mirror[angle] == 2 )
2728
0
    {
2729
0
      maskStride  = -GEO_WEIGHT_MASK_SIZE;
2730
0
      maskStride2 = -( int ) pu.lwidth();
2731
0
      sadMask     = &g_globalGeoEncSADmask[g_angle2mask[g_GeoParams[splitDir][0]]]
2732
0
                      [( GEO_WEIGHT_MASK_SIZE - 1 - g_weightOffset[hIdx][wIdx][splitDir][1] ) * GEO_WEIGHT_MASK_SIZE
2733
0
                                                  + g_weightOffset[hIdx][wIdx][splitDir][0]
2734
0
                      ];
2735
0
    }
2736
0
    else if( g_angle2mirror[angle] == 1 )
2737
0
    {
2738
0
      stepX       = -1;
2739
0
      maskStride2 = pu.lwidth();
2740
0
      maskStride  = GEO_WEIGHT_MASK_SIZE;
2741
0
      sadMask     = &g_globalGeoEncSADmask[g_angle2mask[g_GeoParams[splitDir][0]]]
2742
0
                      [     GEO_WEIGHT_MASK_SIZE *     g_weightOffset[hIdx][wIdx][splitDir][1]
2743
0
                        + ( GEO_WEIGHT_MASK_SIZE - 1 - g_weightOffset[hIdx][wIdx][splitDir][0] )
2744
0
                      ];
2745
0
    }
2746
0
    else
2747
0
    {
2748
0
      maskStride  = GEO_WEIGHT_MASK_SIZE;
2749
0
      maskStride2 = -( int ) pu.lwidth();
2750
0
      sadMask     = &g_globalGeoEncSADmask[g_angle2mask[g_GeoParams[splitDir][0]]]
2751
0
                      [   g_weightOffset[hIdx][wIdx][splitDir][1] * GEO_WEIGHT_MASK_SIZE
2752
0
                        + g_weightOffset[hIdx][wIdx][splitDir][0]
2753
0
                      ];
2754
0
    }
2755
2756
0
    m_cRdCost.setDistParamGeo ( distParam, pu.cs->getOrgBuf().Y(),
2757
0
                                nullptr, 0,
2758
0
                                sadMask, maskStride, stepX, maskStride2,
2759
0
                                pu.cs->sps->bitDepths[CH_L], COMP_Y );
2760
2761
0
    for( uint8_t mergeCand = 0; mergeCand < maxNumMergeCandidates; mergeCand++ )
2762
0
    {
2763
0
      if( sameMV[mergeCand] )
2764
0
      {
2765
0
        continue;
2766
0
      }
2767
2768
0
      distParam.cur.buf         = geoTempBuf[mergeCand].Y().buf;
2769
0
      distParam.cur.stride      = geoTempBuf[mergeCand].Y().stride;
2770
0
      const Distortion sadLarge = distParam.distFunc( distParam );
2771
0
      const Distortion sadSmall = sadWholeBlk[mergeCand] - sadLarge;
2772
2773
0
      const int bitsCand        = mergeCand + 1;
2774
2775
0
      const double cost0        = ( double ) sadLarge + ( double ) bitsCand * sqrtLambdaForFirstPass;
2776
0
      const double cost1        = ( double ) sadSmall + ( double ) bitsCand * sqrtLambdaForFirstPass;
2777
2778
0
      m_GeoCostList.insert( splitDir, 0, mergeCand, cost0 );
2779
0
      m_GeoCostList.insert( splitDir, 1, mergeCand, cost1 );
2780
0
    }
2781
2782
0
    if( m_pcEncCfg->m_Geo == 4 )
2783
0
    {
2784
0
      if( splitDir == 1 )
2785
0
      {
2786
0
        splitDir += 7;
2787
0
      }
2788
0
      else if( splitDir == 35 || ( splitDir + 1 ) % 4 != 0 )
2789
0
      {
2790
0
        splitDir++;
2791
0
      }
2792
0
      else
2793
0
      {
2794
0
        splitDir += 5;
2795
0
      }
2796
0
    }
2797
0
    else
2798
0
    {
2799
0
      splitDir++;
2800
0
    }
2801
0
  }
2802
2803
0
  comboList.list.clear();
2804
2805
0
  for( int splitDir = 0; splitDir < GEO_NUM_PARTITION_MODE; )
2806
0
  {
2807
0
    for( int geoMotionIdx = 0; geoMotionIdx < maxNumMergeCandidates * ( maxNumMergeCandidates - 1 ); geoMotionIdx++ )
2808
0
    {
2809
0
      const MergeIdxPair mergeIdxPair = m_GeoModeTest[geoMotionIdx];
2810
2811
0
      if( sameMV[mergeIdxPair[0]] || sameMV[mergeIdxPair[1]] )
2812
0
      {
2813
0
        continue;
2814
0
      }
2815
2816
0
      double tempCost = m_GeoCostList.getCost( splitDir, mergeIdxPair[0], mergeIdxPair[1] );
2817
2818
0
      if( tempCost > bestWholeBlkCost )
2819
0
      {
2820
0
        continue;
2821
0
      }
2822
2823
0
      tempCost = tempCost + ( double ) bitsForPartitionIdx * sqrtLambdaForFirstPass;
2824
0
      comboList.list.push_back( GeoMergeCombo{ splitDir, mergeIdxPair[0], mergeIdxPair[1], tempCost } );
2825
0
    }
2826
2827
0
    if( m_pcEncCfg->m_Geo == 4 )
2828
0
    {
2829
0
      if( splitDir == 1 )
2830
0
      {
2831
0
        splitDir += 7;
2832
0
      }
2833
0
      else if( splitDir == 35 || ( splitDir + 1 ) % 4 != 0 )
2834
0
      {
2835
0
        splitDir++;
2836
0
      }
2837
0
      else
2838
0
      {
2839
0
        splitDir += 5;
2840
0
      }
2841
0
    }
2842
0
    else
2843
0
    {
2844
0
      splitDir++;
2845
0
    }
2846
0
  }
2847
2848
0
  if( comboList.list.empty() )
2849
0
  {
2850
0
    return false;
2851
0
  }
2852
2853
0
  comboList.sortByCost();
2854
0
  return true;
2855
0
}
2856
2857
double EncCu::calcLumaCost4MergePrediction( const TempCtx &ctxStart, const PelUnitBuf &predBuf, double lambda, CodingUnit &cu, DistParam &distParam )
2858
0
{
2859
0
  distParam.cur = predBuf.Y();
2860
0
  auto dist     = distParam.distFunc(distParam);
2861
2862
0
  m_CABACEstimator->getCtx() = ctxStart;
2863
0
  auto fracBits = xCalcPuMeBits( cu );
2864
2865
0
  double cost   = ( double ) dist + ( double ) fracBits * lambda;
2866
2867
0
  m_uiSadBestForQPA = std::min( dist, m_uiSadBestForQPA );
2868
2869
0
  return cost;
2870
0
}
2871
2872
//////////////////////////////////////////////////////////////////////////////////////////////
2873
// ibc merge/skip mode check
2874
void EncCu::xCheckRDCostIBCModeMerge2Nx2N(CodingStructure*& tempCS, CodingStructure*& bestCS, Partitioner& partitioner, const EncTestMode& encTestMode)
2875
25.1k
{
2876
25.1k
  assert(partitioner.chType != CH_C); // chroma IBC is derived
2877
25.1k
  if (tempCS->area.lwidth() == 128 || tempCS->area.lheight() == 128) // disable IBC mode larger than 64x64
2878
0
  {
2879
0
    return;
2880
0
  }
2881
2882
25.1k
  if ((m_pcEncCfg->m_IBCFastMethod > 1) && !bestCS->slice->isIntra() && (bestCS->cus.size() != 0))
2883
0
  {
2884
0
    if (bestCS->getCU(partitioner.chType, partitioner.treeType)->skip)
2885
0
    {
2886
0
      return;
2887
0
    }
2888
0
  }
2889
2890
25.1k
  const SPS& sps = *tempCS->sps;
2891
2892
25.1k
  tempCS->initStructData(encTestMode.qp);
2893
25.1k
  MergeCtx mergeCtx;
2894
2895
25.1k
  {
2896
    // first get merge candidates
2897
25.1k
    CodingUnit cu(tempCS->area);
2898
25.1k
    cu.cs = tempCS;
2899
25.1k
    cu.predMode = MODE_IBC;
2900
25.1k
    cu.slice = tempCS->slice;
2901
25.1k
    cu.tileIdx = m_tileIdx;
2902
25.1k
    cu.initPuData();
2903
25.1k
    cu.cs = tempCS;
2904
25.1k
    cu.mmvdSkip = false;
2905
25.1k
    cu.mmvdMergeFlag = false;
2906
25.1k
    cu.geo = false;
2907
25.1k
    CU::getIBCMergeCandidates(cu, mergeCtx);
2908
25.1k
  }
2909
25.1k
  int candHasNoResidual[MRG_MAX_NUM_CANDS];
2910
176k
  for (unsigned int ui = 0; ui < mergeCtx.numValidMergeCand; ui++)
2911
150k
  {
2912
150k
    candHasNoResidual[ui] = 0;
2913
150k
  }
2914
2915
25.1k
  bool                                        bestIsSkip = false;
2916
25.1k
  unsigned                                    numMrgSATDCand = mergeCtx.numValidMergeCand;
2917
25.1k
  static_vector<unsigned, MRG_MAX_NUM_CANDS>  RdModeList(MRG_MAX_NUM_CANDS);
2918
176k
  for (unsigned i = 0; i < MRG_MAX_NUM_CANDS; i++)
2919
150k
  {
2920
150k
    RdModeList[i] = i;
2921
150k
  }
2922
2923
  //{
2924
25.1k
  static_vector<double, MRG_MAX_NUM_CANDS>  candCostList(MRG_MAX_NUM_CANDS, MAX_DOUBLE);
2925
  // 1. Pass: get SATD-cost for selected candidates and reduce their count
2926
25.1k
  {
2927
25.1k
    const double sqrtLambdaForFirstPass = m_cRdCost.getMotionLambda();
2928
2929
25.1k
    CodingUnit& cu = tempCS->addCU(CS::getArea(*tempCS, tempCS->area, partitioner.chType,partitioner.treeType), partitioner.chType);
2930
2931
25.1k
    partitioner.setCUData(cu);
2932
25.1k
    cu.slice = tempCS->slice;
2933
25.1k
    cu.tileIdx = m_tileIdx;
2934
25.1k
    cu.skip = false;
2935
25.1k
    cu.predMode = MODE_IBC;
2936
25.1k
    cu.chromaQpAdj = m_cuChromaQpOffsetIdxPlus1;
2937
25.1k
    cu.qp = encTestMode.qp;
2938
25.1k
    cu.mmvdSkip = false;
2939
25.1k
    cu.geo = false;
2940
25.1k
    DistParam distParam;
2941
25.1k
    cu.initPuData();
2942
25.1k
    cu.mmvdMergeFlag = false;
2943
25.1k
    Picture* refPic = cu.slice->pic;
2944
25.1k
    const UnitArea localUnitArea(tempCS->area.chromaFormat, Area(cu.blocks[COMP_Y].x, cu.blocks[COMP_Y].y, tempCS->area.Y().width, tempCS->area.Y().height));
2945
25.1k
    const CompArea& compArea = localUnitArea.block(COMP_Y);
2946
25.1k
    const CPelBuf refBuf = refPic->getRecoBuf(compArea);
2947
25.1k
    const Pel* piRefSrch = refBuf.buf;
2948
2949
25.1k
      distParam = m_cRdCost.setDistParam(tempCS->getOrgBuf(COMP_Y), refBuf, sps.bitDepths[CH_L], DF_HAD);
2950
25.1k
    int refStride = refBuf.stride;
2951
2952
25.1k
    int numValidBv = mergeCtx.numValidMergeCand;
2953
176k
    for (unsigned int mergeCand = 0; mergeCand < mergeCtx.numValidMergeCand; mergeCand++)
2954
150k
    {
2955
150k
      mergeCtx.setMergeInfo(cu, mergeCand); // set bv info in merge mode
2956
150k
      const int cuPelX = cu.Y().x;
2957
150k
      const int cuPelY = cu.Y().y;
2958
150k
      int roiWidth     = cu.lwidth();
2959
150k
      int roiHeight    = cu.lheight();
2960
150k
      const int picWidth  = cu.cs->slice->pps->picWidthInLumaSamples;
2961
150k
      const int picHeight = cu.cs->slice->pps->picHeightInLumaSamples;
2962
150k
      const unsigned int lcuWidth = cu.cs->slice->sps->CTUSize;
2963
2964
150k
      Mv bv = cu.mv[0][0];
2965
150k
      bv.changePrecision( MV_PRECISION_INTERNAL, MV_PRECISION_INT);
2966
150k
      int xPred = bv.hor;
2967
150k
      int yPred = bv.ver;
2968
      
2969
150k
      if( !m_cInterSearch.searchBvIBC( cu, cuPelX, cuPelY, roiWidth, roiHeight, picWidth, picHeight, xPred, yPred, lcuWidth ) ) // not valid bv derived
2970
150k
      {
2971
150k
        numValidBv--;
2972
150k
        continue;
2973
150k
      }
2974
18.4E
      CU::spanMotionInfo(cu);
2975
18.4E
      distParam.cur.buf = piRefSrch + refStride * yPred + xPred;
2976
2977
18.4E
      Distortion sad = distParam.distFunc(distParam);
2978
18.4E
      unsigned int bitsCand = mergeCand + 1;
2979
18.4E
      if (mergeCand == tempCS->sps->maxNumIBCMergeCand - 1)
2980
0
      {
2981
0
        bitsCand--;
2982
0
      }
2983
18.4E
      double cost = (double)sad + (double)bitsCand * sqrtLambdaForFirstPass;
2984
2985
18.4E
      updateCandList( mergeCand, cost, RdModeList, candCostList, numMrgSATDCand );
2986
18.4E
    }
2987
2988
    // Try to limit number of candidates using SATD-costs
2989
25.1k
    if (numValidBv)
2990
0
    {
2991
0
      numMrgSATDCand = numValidBv;
2992
0
      for (unsigned int i = 1; i < numValidBv; i++)
2993
0
      {
2994
0
        if (candCostList[i] > MRG_FAST_RATIO[0] * candCostList[0])
2995
0
        {
2996
0
          numMrgSATDCand = i;
2997
0
          break;
2998
0
        }
2999
0
      }
3000
0
    }
3001
25.1k
    else
3002
25.1k
    {
3003
25.1k
      tempCS->dist = 0;
3004
25.1k
      tempCS->fracBits = 0;
3005
25.1k
      tempCS->cost = MAX_DOUBLE;
3006
25.1k
      tempCS->costDbOffset = 0;
3007
25.1k
      tempCS->initStructData(encTestMode.qp);
3008
25.1k
      return;
3009
25.1k
    }
3010
3011
0
    tempCS->initStructData(encTestMode.qp);
3012
0
  }
3013
  //}
3014
3015
3016
0
  const unsigned int iteration = 2;
3017
 // m_bestModeUpdated = tempCS->cost = bestCS->cost = false;
3018
  // 2. Pass: check candidates using full RD test
3019
0
  for (unsigned int numResidualPass = 0; numResidualPass < iteration; numResidualPass++)
3020
0
  {
3021
0
    for (unsigned int mrgHADIdx = 0; mrgHADIdx < numMrgSATDCand; mrgHADIdx++)
3022
0
    {
3023
0
      unsigned int mergeCand = RdModeList[mrgHADIdx];
3024
0
      if (!(numResidualPass == 1 && candHasNoResidual[mergeCand] == 1))
3025
0
      {
3026
0
        if (!(bestIsSkip && (numResidualPass == 0)))
3027
0
        {
3028
0
          {
3029
3030
            // first get merge candidates
3031
0
            CodingUnit& cu = tempCS->addCU(CS::getArea(*tempCS, tempCS->area, (const ChannelType)partitioner.chType,partitioner.treeType), (const ChannelType)partitioner.chType);
3032
3033
0
            partitioner.setCUData(cu);
3034
0
            cu.slice = tempCS->slice;
3035
0
            cu.tileIdx = m_tileIdx;
3036
0
            cu.skip = false;
3037
0
            cu.predMode = MODE_IBC;
3038
0
            cu.chromaQpAdj = m_cuChromaQpOffsetIdxPlus1;
3039
0
            cu.qp = encTestMode.qp;
3040
0
            cu.sbtInfo = 0;
3041
0
            cu.initPuData();
3042
0
            cu.intraDir[0] = DC_IDX; // set intra pred for ibc block
3043
0
            cu.intraDir[1] = PLANAR_IDX; // set intra pred for ibc block
3044
0
            cu.mmvdSkip = false;
3045
0
            cu.mmvdMergeFlag = false;
3046
0
            cu.geo = false;
3047
0
            mergeCtx.setMergeInfo(cu, mergeCand);
3048
0
            CU::spanMotionInfo(cu);
3049
3050
0
            assert(mergeCtx.mrgTypeNeighbours[mergeCand] == MRG_TYPE_IBC);
3051
0
            const bool chroma = !CU::isSepTree(cu);
3052
3053
            //  MC
3054
0
            cu.mcControl = chroma ? 0: 2;
3055
0
            m_cInterSearch.motionCompensationIBC(cu, tempCS->getPredBuf());
3056
0
            m_CABACEstimator->getCtx() = m_CurrCtx->start;
3057
3058
0
            m_cInterSearch.encodeResAndCalcRdInterCU(*tempCS, partitioner, (numResidualPass != 0));
3059
0
            cu.mcControl = 0;
3060
0
            xEncodeDontSplit(*tempCS, partitioner);
3061
0
            xCheckDQP(*tempCS, partitioner);
3062
0
            xCheckBestMode(tempCS, bestCS, partitioner, encTestMode);
3063
3064
0
            tempCS->initStructData(encTestMode.qp);
3065
0
          }
3066
3067
0
          if (m_pcEncCfg->m_useFastDecisionForMerge && !bestIsSkip)
3068
0
          {
3069
0
            if (bestCS->getCU(partitioner.chType, partitioner.treeType) == NULL)
3070
0
              bestIsSkip = 0;
3071
0
            else
3072
0
              bestIsSkip = bestCS->getCU(partitioner.chType, partitioner.treeType)->rootCbf == 0;
3073
0
          }
3074
0
        }
3075
0
      }
3076
0
    }
3077
0
  }
3078
0
}
3079
3080
void EncCu::xCheckRDCostIBCMode(CodingStructure*& tempCS, CodingStructure*& bestCS, Partitioner& partitioner,
3081
  const EncTestMode& encTestMode)
3082
25.1k
{
3083
25.1k
  if (tempCS->area.lwidth() == 128 || tempCS->area.lheight() == 128)   // disable IBC mode larger than 64x64
3084
0
  {
3085
0
    return;
3086
0
  }
3087
25.1k
  if ((m_pcEncCfg->m_IBCFastMethod > 1) && !bestCS->slice->isIntra() && (bestCS->cus.size() != 0))
3088
0
  {
3089
0
    if (bestCS->getCU(partitioner.chType, partitioner.treeType)->skip)
3090
0
    {
3091
0
      return;
3092
0
    }
3093
0
  }
3094
3095
25.1k
  tempCS->initStructData(encTestMode.qp);
3096
3097
25.1k
  CodingUnit& cu = tempCS->addCU(CS::getArea(*tempCS, tempCS->area, partitioner.chType, partitioner.treeType), partitioner.chType);
3098
3099
25.1k
  partitioner.setCUData(cu);
3100
25.1k
  cu.slice = tempCS->slice;
3101
25.1k
  cu.tileIdx = m_tileIdx;
3102
25.1k
  cu.skip = false;
3103
25.1k
  cu.predMode = MODE_IBC;
3104
25.1k
  cu.chromaQpAdj = m_cuChromaQpOffsetIdxPlus1;
3105
25.1k
  cu.qp = encTestMode.qp;
3106
25.1k
  cu.initPuData();
3107
25.1k
  cu.imv = IMV_OFF;
3108
25.1k
  cu.sbtInfo = 0;
3109
25.1k
  cu.mmvdSkip = false;
3110
25.1k
  cu.mmvdMergeFlag = false;
3111
3112
25.1k
  cu.intraDir[0] = DC_IDX; // set intra pred for ibc block
3113
25.1k
  cu.intraDir[1] = PLANAR_IDX; // set intra pred for ibc block
3114
3115
25.1k
  cu.interDir = 1; // use list 0 for IBC mode
3116
25.1k
  cu.refIdx[REF_PIC_LIST_0] = MAX_NUM_REF; // last idx in the list
3117
25.1k
  bool bValid = m_cInterSearch.predIBCSearch(cu, partitioner);
3118
3119
25.1k
  if (bValid)
3120
22.4k
  {
3121
22.4k
    CU::spanMotionInfo(cu);
3122
22.4k
    const bool chroma = !CU::isSepTree(cu);
3123
    //  MC
3124
22.4k
    cu.mcControl = chroma ? 0 : 2;
3125
22.4k
    m_cInterSearch.motionCompensationIBC(cu, tempCS->getPredBuf());
3126
3127
22.4k
    m_cInterSearch.encodeResAndCalcRdInterCU(*tempCS, partitioner, false);
3128
22.4k
    cu.mcControl = 0;
3129
3130
22.4k
    xEncodeDontSplit(*tempCS, partitioner);
3131
22.4k
    xCheckDQP(*tempCS, partitioner);
3132
22.4k
    xCheckBestMode(tempCS, bestCS, partitioner, encTestMode);
3133
22.4k
  } // bValid
3134
2.68k
  else
3135
2.68k
  {
3136
2.68k
    tempCS->dist = 0;
3137
2.68k
    tempCS->fracBits = 0;
3138
2.68k
    tempCS->cost = MAX_DOUBLE;
3139
2.68k
    tempCS->costDbOffset = 0;
3140
2.68k
  }
3141
25.1k
}
3142
3143
void EncCu::xCheckRDCostInter( CodingStructure *&tempCS, CodingStructure *&bestCS, Partitioner &partitioner, const EncTestMode& encTestMode )
3144
0
{
3145
0
  PROFILER_SCOPE_AND_STAGE_EXT( 1, _TPROF, P_INTER_MVD, tempCS, partitioner.chType );
3146
0
  tempCS->initStructData( encTestMode.qp );
3147
3148
0
  m_cInterSearch.setAffineModeSelected( false );
3149
3150
0
  m_cInterSearch.resetBufferedUniMotions();
3151
3152
0
  int bcwLoopNum = BCW_NUM;
3153
3154
0
  if( tempCS->area.Y().area() < BCW_SIZE_CONSTRAINT || !tempCS->slice->isInterB() || !tempCS->sps->BCW )
3155
0
  {
3156
0
    bcwLoopNum = 1;
3157
0
  }
3158
  
3159
0
  double curBestCost = bestCS->cost;
3160
0
  double equBcwCost = MAX_DOUBLE;
3161
3162
0
  for( int bcwLoopIdx = 0; bcwLoopIdx < bcwLoopNum; bcwLoopIdx++ )
3163
0
  {
3164
0
    if( m_pcEncCfg->m_BCW == 2 )
3165
0
    {
3166
0
      bool isBestInter   = m_modeCtrl.getBlkInfo( bestCS->area ).isInter;
3167
0
      uint8_t bestBcwIdx = m_modeCtrl.getBlkInfo( bestCS->area).BcwIdx;
3168
3169
0
      if( isBestInter && g_BcwSearchOrder[bcwLoopIdx] != BCW_DEFAULT && g_BcwSearchOrder[bcwLoopIdx] != bestBcwIdx )
3170
0
      {
3171
0
        continue;
3172
0
      }
3173
0
    }
3174
    
3175
0
    if( !tempCS->slice->checkLDC )
3176
0
    {
3177
0
      if( bcwLoopIdx != 0 && bcwLoopIdx != 3 && bcwLoopIdx != 4 )
3178
0
      {
3179
0
        continue;
3180
0
      }
3181
0
    }
3182
  
3183
0
    CodingUnit &cu      = tempCS->addCU( tempCS->area, partitioner.chType );
3184
3185
0
    partitioner.setCUData( cu );
3186
0
    cu.slice            = tempCS->slice;
3187
0
    cu.tileIdx          = m_tileIdx;
3188
0
    cu.skip             = false;
3189
0
    cu.mmvdSkip         = false;
3190
0
    cu.predMode         = MODE_INTER;
3191
0
    cu.chromaQpAdj      = m_cuChromaQpOffsetIdxPlus1;
3192
0
    cu.qp               = encTestMode.qp;
3193
0
    cu.initPuData();
3194
3195
0
    cu.BcwIdx = g_BcwSearchOrder[bcwLoopIdx];
3196
0
    uint8_t bcwIdx = cu.BcwIdx;
3197
0
    bool testBcw = (bcwIdx != BCW_DEFAULT);
3198
3199
0
    bool StopInterRes = (m_pcEncCfg->m_FastInferMerge >> 3) & 1;
3200
0
    StopInterRes &= bestCS->slice->TLayer > (m_pcEncCfg->m_maxTLayer - (m_pcEncCfg->m_FastInferMerge & 7));
3201
0
    double bestCostInter = StopInterRes ? m_mergeBestSATDCost : MAX_DOUBLE;
3202
3203
0
    bool stopTest = m_cInterSearch.predInterSearch(cu, partitioner, bestCostInter);
3204
3205
0
    if (StopInterRes && (bestCostInter != m_mergeBestSATDCost))
3206
0
    {
3207
0
      int L = (cu.slice->TLayer <= 2) ? 0 : (cu.slice->TLayer - 2);
3208
0
      if ((bestCostInter > MRG_FAST_RATIOMYV[L] * m_mergeBestSATDCost))
3209
0
      {
3210
0
        stopTest = true;
3211
0
      }
3212
0
    }
3213
3214
0
    if( !stopTest )
3215
0
    {
3216
0
      bcwIdx   = CU::getValidBcwIdx(cu);
3217
0
      stopTest = testBcw && bcwIdx == BCW_DEFAULT;
3218
0
    }
3219
    
3220
0
    if( stopTest )
3221
0
    {
3222
0
      tempCS->initStructData(encTestMode.qp);
3223
0
      continue;
3224
0
    }
3225
3226
0
    CHECK(!(testBcw || (!testBcw && bcwIdx == BCW_DEFAULT)), " !( bTestBcw || (!bTestBcw && bcwIdx == BCW_DEFAULT ) )");
3227
        
3228
0
    xEncodeInterResidual(tempCS, bestCS, partitioner, encTestMode, 0, 0, &equBcwCost);
3229
    
3230
0
    if( bcwIdx == BCW_DEFAULT )
3231
0
    {
3232
0
      m_cInterSearch.setAffineModeSelected( bestCS->cus.front()->affine && !bestCS->cus.front()->mergeFlag );
3233
0
    }
3234
3235
0
    tempCS->initStructData(encTestMode.qp);
3236
  
3237
0
    double skipTH = MAX_DOUBLE;
3238
0
    skipTH = (m_pcEncCfg->m_BCW == 2 ? 1.05 : MAX_DOUBLE);
3239
0
    if( equBcwCost > curBestCost * skipTH )
3240
0
    {
3241
0
      break;
3242
0
    }
3243
3244
0
    if( m_pcEncCfg->m_BCW == 2 )
3245
0
    {
3246
0
      if( ( cu.interDir != 3 && testBcw == 0 && ! m_pcEncCfg->m_picReordering )
3247
0
         || ( g_BcwSearchOrder[bcwLoopIdx] == BCW_DEFAULT && xIsBcwSkip( cu ) ) )
3248
0
      {
3249
0
        break;
3250
0
      }
3251
0
    }
3252
0
  }
3253
0
  STAT_COUNT_CU_MODES( partitioner.chType == CH_L, g_cuCounters1D[CU_MODES_TESTED][0][!tempCS->slice->isIntra() + tempCS->slice->depth] );
3254
0
  STAT_COUNT_CU_MODES( partitioner.chType == CH_L && !tempCS->slice->isIntra(), g_cuCounters2D[CU_MODES_TESTED][Log2( tempCS->area.lheight() )][Log2( tempCS->area.lwidth() )] );
3255
0
}
3256
3257
void EncCu::xCheckRDCostInterIMV(CodingStructure *&tempCS, CodingStructure *&bestCS, Partitioner &partitioner, const EncTestMode& encTestMode)
3258
0
{
3259
0
  PROFILER_SCOPE_AND_STAGE_EXT( 1, _TPROF, P_INTER_MVD_IMV, tempCS, partitioner.chType );
3260
0
  bool Test_AMVR = m_pcEncCfg->m_AMVRspeed ? true: false;
3261
0
  if (m_pcEncCfg->m_AMVRspeed > 2 && m_pcEncCfg->m_AMVRspeed < 5 && !bestCS->cus.empty() && bestCS->getCU(partitioner.chType, partitioner.treeType)->skip)
3262
0
  {
3263
0
    Test_AMVR = false;
3264
0
  }
3265
0
  else if (m_pcEncCfg->m_AMVRspeed > 4 && !bestCS->cus.empty() && bestCS->getCU(partitioner.chType, partitioner.treeType)->mergeFlag && !bestCS->getCU(partitioner.chType, partitioner.treeType)->ciip)
3266
0
  {
3267
0
    Test_AMVR = false;
3268
0
  }
3269
0
  bool Do_Limit = !bestCS->cus.empty() && (m_pcEncCfg->m_AMVRspeed == 4 || m_pcEncCfg->m_AMVRspeed == 6) ? true : false;
3270
0
  bool Do_OnceRes = !bestCS->cus.empty() && (m_pcEncCfg->m_AMVRspeed == 7) ? true : false;
3271
3272
0
  if( Test_AMVR )
3273
0
  {
3274
0
    double Fpel_cost    = m_pcEncCfg->m_AMVRspeed == 1 ? MAX_DOUBLE*0.5 : MAX_DOUBLE;
3275
0
    double costCurStart = m_pcEncCfg->m_AMVRspeed == 1 ? m_modeCtrl.comprCUCtx->bestCostNoImv : bestCS->cost;
3276
0
    double costCur      = MAX_DOUBLE;
3277
0
    double bestCostIMV  = MAX_DOUBLE;
3278
3279
0
    if (Do_OnceRes)
3280
0
    {
3281
0
      costCurStart = xCalcDistortion(bestCS, partitioner.chType, bestCS->sps->bitDepths[CH_L], 0);
3282
0
      Fpel_cost = costCurStart;
3283
0
      tempCS->initSubStructure(*m_pTempCS2, partitioner.chType, partitioner.currArea(), false);
3284
0
    }
3285
3286
0
    CodingStructure *tempCSbest = m_pTempCS2;
3287
3288
0
    m_cInterSearch.setAffineModeSelected( false );
3289
3290
0
    m_cInterSearch.resetBufferedUniMotions();
3291
3292
0
    int bcwLoopNum = (tempCS->slice->isInterB() ? BCW_NUM : 1);
3293
0
    bcwLoopNum = (tempCS->sps->BCW ? bcwLoopNum : 1);
3294
3295
0
    if( tempCS->area.lwidth() * tempCS->area.lheight() < BCW_SIZE_CONSTRAINT )
3296
0
    {
3297
0
      bcwLoopNum = 1;
3298
0
    }
3299
3300
0
    for (int i = 1; i <= IMV_HPEL; i++)
3301
0
    {
3302
0
      double curBestCost = bestCS->cost;
3303
0
      double equBcwCost  = MAX_DOUBLE;
3304
3305
0
      for( int bcwLoopIdx = 0; bcwLoopIdx < bcwLoopNum; bcwLoopIdx++ )
3306
0
      {
3307
0
        if( m_pcEncCfg->m_BCW == 2 )
3308
0
        {
3309
0
          bool isBestInter   = m_modeCtrl.getBlkInfo( bestCS->area ).isInter;
3310
0
          uint8_t bestBcwIdx = m_modeCtrl.getBlkInfo( bestCS->area).BcwIdx;
3311
3312
0
          if( isBestInter && g_BcwSearchOrder[bcwLoopIdx] != BCW_DEFAULT && g_BcwSearchOrder[bcwLoopIdx] != bestBcwIdx )
3313
0
          {
3314
0
            continue;
3315
0
          }
3316
          
3317
0
          if( tempCS->slice->checkLDC && g_BcwSearchOrder[bcwLoopIdx] != BCW_DEFAULT
3318
0
            && (m_bestBcwIdx[0] >= 0 && g_BcwSearchOrder[bcwLoopIdx] != m_bestBcwIdx[0])
3319
0
            && (m_bestBcwIdx[1] >= 0 && g_BcwSearchOrder[bcwLoopIdx] != m_bestBcwIdx[1]))
3320
0
          {
3321
0
            continue;
3322
0
          }
3323
0
        }
3324
3325
0
        if( !tempCS->slice->checkLDC )
3326
0
        {
3327
0
          if( bcwLoopIdx != 0 && bcwLoopIdx != 3 && bcwLoopIdx != 4 )
3328
0
          {
3329
0
            continue;
3330
0
          }
3331
0
        }
3332
3333
0
        bool testBcw;
3334
0
        uint8_t bcwIdx;
3335
0
        bool isEqualUni = false;
3336
3337
0
        if (i > IMV_FPEL)
3338
0
        {
3339
0
          bool nextimv = false;
3340
0
          double stopCost = i == IMV_HPEL ? 1.25 : 1.06;
3341
0
          if (Fpel_cost > stopCost * costCurStart)
3342
0
          {
3343
0
            nextimv = true;
3344
0
          }
3345
0
          if ( m_pcEncCfg->m_AMVRspeed == 1 )
3346
0
          {
3347
0
            costCurStart = bestCS->cost;
3348
0
          }
3349
0
          if (nextimv)
3350
0
          {
3351
0
            continue;
3352
0
          }
3353
0
        }
3354
3355
0
        bool Do_Search = Do_OnceRes ? false : true;
3356
3357
0
        if (Do_Limit)
3358
0
        {
3359
0
          Do_Search = i == IMV_FPEL ? true : false;
3360
3361
0
          if (i == IMV_HPEL)
3362
0
          {
3363
0
            if (bestCS->slice->TLayer > 3)
3364
0
            {
3365
0
              continue;
3366
0
            }
3367
0
            if (bestCS->getCU(partitioner.chType, partitioner.treeType)->imv != 0)
3368
0
            {
3369
0
              Do_Search = true; //do_est
3370
0
            }
3371
0
          }
3372
0
          if (bestCS->getCU(partitioner.chType, partitioner.treeType)->mmvdMergeFlag || bestCS->getCU(partitioner.chType, partitioner.treeType)->geo)
3373
0
          {
3374
0
            Do_Search = true;
3375
0
          }
3376
0
        }
3377
0
        tempCS->initStructData(encTestMode.qp);
3378
3379
0
        if (!Do_Search)
3380
0
        {
3381
0
          tempCS->copyStructure(*bestCS, partitioner.chType, TREE_D);
3382
0
        }
3383
0
        tempCS->dist = 0;
3384
0
        tempCS->fracBits = 0;
3385
0
        tempCS->cost = MAX_DOUBLE;
3386
0
        CodingUnit &cu = (Do_Search) ? tempCS->addCU(tempCS->area, partitioner.chType) : *tempCS->getCU(partitioner.chType, partitioner.treeType);
3387
0
        if (Do_Search)
3388
0
        {
3389
0
          partitioner.setCUData(cu);
3390
0
          cu.slice = tempCS->slice;
3391
0
          cu.tileIdx = m_tileIdx;
3392
0
          cu.skip = false;
3393
0
          cu.mmvdSkip = false;
3394
0
          cu.predMode = MODE_INTER;
3395
0
          cu.chromaQpAdj = m_cuChromaQpOffsetIdxPlus1;
3396
0
          cu.qp = encTestMode.qp;
3397
3398
0
          cu.initPuData();
3399
3400
0
          cu.imv = i;
3401
3402
0
          cu.BcwIdx = g_BcwSearchOrder[bcwLoopIdx];
3403
0
          bcwIdx    = cu.BcwIdx;
3404
0
          testBcw   = (bcwIdx != BCW_DEFAULT);
3405
3406
0
          cu.interDir = 10;
3407
          
3408
0
          double bestCostInter = MAX_DOUBLE;
3409
0
          m_cInterSearch.predInterSearch(cu, partitioner, bestCostInter);
3410
          
3411
0
          if ( cu.interDir <= 3 )
3412
0
          {
3413
0
            bcwIdx = CU::getValidBcwIdx(cu);
3414
0
          }
3415
0
          else
3416
0
          {
3417
0
            continue;
3418
0
          }
3419
          
3420
0
          if( testBcw && bcwIdx == BCW_DEFAULT ) // Enabled Bcw but the search results is uni.
3421
0
          {
3422
0
            continue;
3423
0
          }
3424
0
          CHECK(!(testBcw || (!testBcw && bcwIdx == BCW_DEFAULT)), " !( bTestBcw || (!bTestBcw && bcwIdx == BCW_DEFAULT ) )");
3425
3426
0
          if( m_pcEncCfg->m_BCW == 2 )
3427
0
          {
3428
0
            if( cu.interDir != 3 && testBcw == 0 )
3429
0
            {
3430
0
              isEqualUni = true;
3431
0
            }
3432
0
          }
3433
3434
0
          if (!CU::hasSubCUNonZeroMVd(cu))
3435
0
          {
3436
0
            continue;
3437
0
          }
3438
0
        }
3439
0
        else
3440
0
        {
3441
0
          cu.smvdMode = 0;
3442
0
          cu.affine = false;
3443
0
          cu.imv = i ;
3444
0
          CU::resetMVDandMV2Int(cu);
3445
0
          if (!CU::hasSubCUNonZeroMVd(cu))
3446
0
          {
3447
0
            continue;
3448
0
          }
3449
3450
0
          cu.BcwIdx = g_BcwSearchOrder[bcwLoopIdx];
3451
3452
0
          cu.mvRefine = true;
3453
0
          m_cInterSearch.motionCompensation(cu, tempCS->getPredBuf() );
3454
0
          cu.mvRefine = false;
3455
0
        }
3456
3457
0
        if( Do_OnceRes )
3458
0
        {
3459
0
          costCur = xCalcDistortion(tempCS, partitioner.chType, tempCS->sps->bitDepths[CH_L], cu.imv );
3460
0
          if (costCur < bestCostIMV)
3461
0
          {
3462
0
            bestCostIMV = costCur;
3463
0
            tempCSbest->getPredBuf().copyFrom(tempCS->getPredBuf());
3464
0
            tempCSbest->clearCUs();
3465
0
            tempCSbest->clearTUs();
3466
0
            tempCSbest->copyStructure(*tempCS, partitioner.chType, TREE_D);
3467
0
          }
3468
0
          if (i > IMV_FPEL)
3469
0
          {
3470
0
            costCurStart = costCurStart > costCur ? costCur : costCurStart;
3471
0
          }
3472
0
        }
3473
0
        else
3474
0
        {
3475
0
          xEncodeInterResidual(tempCS, bestCS, partitioner, encTestMode, 0, 0, &equBcwCost);
3476
0
          costCur = tempCS->cost;
3477
3478
0
          if (i > IMV_FPEL)
3479
0
          {
3480
0
            costCurStart = bestCS->cost;
3481
0
          }
3482
0
        }
3483
3484
0
        if (i == IMV_FPEL)
3485
0
        {
3486
0
           Fpel_cost = costCur;
3487
0
        }
3488
3489
0
        double skipTH = MAX_DOUBLE;
3490
0
        skipTH = (m_pcEncCfg->m_BCW == 2 ? 1.05 : MAX_DOUBLE);
3491
0
        if( equBcwCost > curBestCost * skipTH )
3492
0
        {
3493
0
          break;
3494
0
        }
3495
3496
0
        if( m_pcEncCfg->m_BCW == 2 )
3497
0
        {
3498
0
          if( isEqualUni == true && ! m_pcEncCfg->m_picReordering )
3499
0
          {
3500
0
            break;
3501
0
          }
3502
0
          if( g_BcwSearchOrder[bcwLoopIdx] == BCW_DEFAULT && xIsBcwSkip( cu ) )
3503
0
          {
3504
0
            break;
3505
0
          }
3506
0
        }
3507
0
      }
3508
0
    }
3509
3510
0
    if (Do_OnceRes && (bestCostIMV != MAX_DOUBLE))
3511
0
    {
3512
0
      CodingStructure* CSCandBest = tempCSbest;
3513
0
      tempCS->initStructData(bestCS->currQP[partitioner.chType]);
3514
0
      tempCS->copyStructure(*CSCandBest, partitioner.chType, TREE_D);
3515
0
      tempCS->getPredBuf().copyFrom(tempCSbest->getPredBuf());
3516
0
      tempCS->dist = 0;
3517
0
      tempCS->fracBits = 0;
3518
0
      tempCS->cost = MAX_DOUBLE;
3519
3520
0
      xEncodeInterResidual(tempCS, bestCS, partitioner, encTestMode, 0, 0, NULL);
3521
0
    }
3522
3523
0
    tempCS->initStructData(encTestMode.qp);
3524
0
  }
3525
0
  STAT_COUNT_CU_MODES( partitioner.chType == CH_L, g_cuCounters1D[CU_MODES_TESTED][0][!tempCS->slice->isIntra() + tempCS->slice->depth] );
3526
0
  STAT_COUNT_CU_MODES( partitioner.chType == CH_L && !tempCS->slice->isIntra(), g_cuCounters2D[CU_MODES_TESTED][Log2( tempCS->area.lheight() )][Log2( tempCS->area.lwidth() )] );
3527
0
}
3528
3529
void EncCu::xCalDebCost( CodingStructure &cs, Partitioner &partitioner )
3530
104k
{
3531
104k
  PROFILER_SCOPE_AND_STAGE_EXT( 1, _TPROF, P_DEBLOCK_FILTER, &cs, partitioner.chType );
3532
104k
  if ( cs.slice->deblockingFilterDisable )
3533
0
  {
3534
0
    return;
3535
0
  }
3536
3537
104k
  const ChromaFormat format = cs.area.chromaFormat;
3538
104k
  CodingUnit*            cu = cs.getCU(partitioner.chType, partitioner.treeType);
3539
104k
  const Position    lumaPos = cu->Y().valid() ? cu->Y().pos() : recalcPosition( format, cu->chType, CH_L, cu->blocks[cu->chType].pos() );
3540
104k
  bool    topEdgeAvai = lumaPos.y > 0 && ((lumaPos.y % 4) == 0);
3541
104k
  bool   leftEdgeAvai = lumaPos.x > 0 && ((lumaPos.x % 4) == 0);
3542
3543
104k
  if( ! ( topEdgeAvai || leftEdgeAvai ))
3544
21.5k
  {
3545
21.5k
    return;
3546
21.5k
  }
3547
3548
82.6k
  ComponentID compStr = ( CU::isSepTree(*cu) && !isLuma( partitioner.chType ) ) ? COMP_Cb : COMP_Y;
3549
82.6k
  ComponentID compEnd = (( CU::isSepTree(*cu) && isLuma( partitioner.chType )) || cu->chromaFormat == VVENC_CHROMA_400 ) ? COMP_Y : COMP_Cr;
3550
82.6k
  const UnitArea currCsArea = clipArea( CS::getArea( cs, cs.area, partitioner.chType, partitioner.treeType ), *cs.picture );
3551
3552
82.6k
  PelStorage&  picDbBuf = m_dbBuffer; //th we could reduce the buffer size and do some relocate
3553
3554
  //deblock neighbour pixels
3555
82.6k
  const Size     lumaSize = cu->Y().valid() ? cu->Y().size() : recalcSize( format, cu->chType, CH_L, cu->blocks[cu->chType].size() );
3556
3557
82.6k
  int verOffset = lumaPos.y > 7 ? 8 : 4;
3558
82.6k
  int horOffset = lumaPos.x > 7 ? 8 : 4;
3559
3560
82.6k
  LoopFilter::calcFilterStrengths( *cu, true );
3561
3562
82.6k
  if( m_EDO == 2 && CS::isDualITree( cs ) && isLuma( partitioner.chType ) )
3563
45.8k
  {
3564
45.8k
    m_cLoopFilter.getMaxFilterLength( *cu, verOffset, horOffset );
3565
3566
45.8k
    if( 0== (verOffset + horOffset) )
3567
0
    {
3568
0
      return;
3569
0
    }
3570
3571
45.8k
    topEdgeAvai  &= verOffset != 0;
3572
45.8k
    leftEdgeAvai &= horOffset != 0;
3573
45.8k
  }
3574
3575
82.6k
  const UnitArea  areaTop  = UnitArea( format, Area( lumaPos.x,             lumaPos.y - verOffset, lumaSize.width, verOffset       ) );
3576
82.6k
  const UnitArea  areaLeft = UnitArea( format, Area( lumaPos.x - horOffset, lumaPos.y,             horOffset,      lumaSize.height ) );
3577
3578
202k
  for ( int compIdx = compStr; compIdx <= compEnd; compIdx++ )
3579
119k
  {
3580
119k
    ComponentID compId = (ComponentID)compIdx;
3581
3582
    //Copy current CU's reco to Deblock Pic Buffer
3583
119k
    const CompArea&  compArea = currCsArea.block( compId );
3584
119k
    CompArea         locArea  = compArea;
3585
119k
    locArea.x -= cu->blocks[compIdx].x;
3586
119k
    locArea.y -= cu->blocks[compIdx].y;
3587
119k
    PelBuf dbReco = picDbBuf.getBuf( locArea );
3588
119k
      PelBuf reco = cs.getRecoBuf( compId );
3589
119k
      dbReco.copyFrom( reco );
3590
    //left neighbour
3591
119k
    if ( leftEdgeAvai )
3592
83.4k
    {
3593
83.4k
      const CompArea&  compArea = areaLeft.block(compId);
3594
83.4k
      CompArea         locArea = compArea;
3595
83.4k
      locArea.x -= cu->blocks[compIdx].x;
3596
83.4k
      locArea.y -= cu->blocks[compIdx].y;
3597
83.4k
      PelBuf dbReco = picDbBuf.getBuf( locArea );
3598
83.4k
        dbReco.copyFrom( cs.picture->getRecoBuf( compArea ) );
3599
83.4k
      }
3600
    //top neighbour
3601
119k
    if ( topEdgeAvai )
3602
86.9k
    {
3603
86.9k
      const CompArea&  compArea = areaTop.block( compId );
3604
86.9k
      CompArea         locArea = compArea;
3605
86.9k
      locArea.x -= cu->blocks[compIdx].x;
3606
86.9k
      locArea.y -= cu->blocks[compIdx].y;
3607
86.9k
      PelBuf dbReco = picDbBuf.getBuf( locArea );
3608
86.9k
        dbReco.copyFrom( cs.picture->getRecoBuf( compArea ) );
3609
86.9k
      }
3610
119k
    }
3611
3612
18.4E
  ChannelType dbChType = CU::isSepTree(*cu) ? partitioner.chType : MAX_NUM_CH;
3613
3614
82.6k
  CHECK( CU::isSepTree(*cu) && !cu->Y().valid() && partitioner.chType == CH_L, "xxx" );
3615
3616
82.6k
  if( cu->Y() .valid() ) m_cLoopFilter.setOrigin( CH_L, cu->lumaPos() );
3617
82.6k
  if( cu->chromaFormat != VVENC_CHROMA_400 && cu->Cb().valid() ) m_cLoopFilter.setOrigin( CH_C, cu->chromaPos() );
3618
3619
  //deblock
3620
82.6k
  if( leftEdgeAvai )
3621
60.8k
  {
3622
60.8k
    m_cLoopFilter.loopFilterCu( *cu, dbChType, EDGE_VER, m_dbBuffer );
3623
60.8k
  }
3624
3625
82.6k
  if( topEdgeAvai )
3626
62.6k
  {
3627
62.6k
    m_cLoopFilter.loopFilterCu( *cu, dbChType, EDGE_HOR, m_dbBuffer );
3628
62.6k
  }
3629
3630
  //calculate difference between DB_before_SSE and DB_after_SSE for neighbouring CUs
3631
82.6k
  Distortion distBeforeDb = 0, distAfterDb = 0, distCur = 0;
3632
202k
  for (int compIdx = compStr; compIdx <= compEnd; compIdx++)
3633
119k
  {
3634
119k
    ComponentID compId = (ComponentID)compIdx;
3635
119k
    {
3636
119k
      CompArea compArea = currCsArea.block( compId );
3637
119k
      CompArea         locArea  = compArea;
3638
119k
      locArea.x -= cu->blocks[compIdx].x;
3639
119k
      locArea.y -= cu->blocks[compIdx].y;
3640
119k
      CPelBuf reco      = picDbBuf.getBuf( locArea );
3641
119k
      CPelBuf org       = cs.getOrgBuf( compId );
3642
119k
      distCur += xGetDistortionDb( cs, org, reco, compArea, false );
3643
119k
    }
3644
3645
119k
    if ( leftEdgeAvai )
3646
83.4k
    {
3647
83.4k
      const CompArea&  compArea = areaLeft.block( compId );
3648
83.4k
      CompArea         locArea  = compArea;
3649
83.4k
      locArea.x -= cu->blocks[compIdx].x;
3650
83.4k
      locArea.y -= cu->blocks[compIdx].y;
3651
83.4k
      CPelBuf org    = cs.picture->getOrigBuf( compArea );
3652
83.4k
      if ( cs.picture->getFilteredOrigBuffer().valid() )
3653
0
      {
3654
0
        org = cs.picture->getFiltOrigBuf( compArea );
3655
0
      }
3656
83.4k
      CPelBuf reco   = cs.picture->getRecoBuf( compArea );
3657
83.4k
      CPelBuf recoDb = picDbBuf.getBuf( locArea );
3658
83.4k
      distBeforeDb  += xGetDistortionDb( cs, org, reco,   compArea, true );
3659
83.4k
      distAfterDb   += xGetDistortionDb( cs, org, recoDb, compArea, false  );
3660
83.4k
    }
3661
3662
119k
    if ( topEdgeAvai )
3663
86.9k
    {
3664
86.9k
      const CompArea&  compArea = areaTop.block( compId );
3665
86.9k
      CompArea         locArea  = compArea;
3666
86.9k
      locArea.x -= cu->blocks[compIdx].x;
3667
86.9k
      locArea.y -= cu->blocks[compIdx].y;
3668
86.9k
      CPelBuf org    = cs.picture->getOrigBuf( compArea );
3669
86.9k
      if ( cs.picture->getFilteredOrigBuffer().valid() )
3670
0
      {
3671
0
        org = cs.picture->getFiltOrigBuf( compArea );
3672
0
      }
3673
86.9k
      CPelBuf reco   = cs.picture->getRecoBuf( compArea );
3674
86.9k
      CPelBuf recoDb = picDbBuf.getBuf( locArea );
3675
86.9k
      distBeforeDb  += xGetDistortionDb( cs, org, reco,   compArea, true );
3676
86.9k
      distAfterDb   += xGetDistortionDb( cs, org, recoDb, compArea, false  );
3677
86.9k
    }
3678
119k
  }
3679
3680
  //updated cost
3681
82.6k
  int64_t distTmp = distCur - cs.dist + distAfterDb - distBeforeDb;
3682
82.6k
  cs.costDbOffset = distTmp < 0 ? -m_cRdCost.calcRdCost( 0, -distTmp ) : m_cRdCost.calcRdCost( 0, distTmp );
3683
82.6k
}
3684
3685
Distortion EncCu::xGetDistortionDb(CodingStructure &cs, CPelBuf& org, CPelBuf& reco, const CompArea& compArea, bool beforeDb)
3686
460k
{
3687
460k
  Distortion dist;
3688
460k
  const ComponentID compID = compArea.compID;
3689
3690
460k
  dist = m_cRdCost.getDistPart(org, reco, cs.sps->bitDepths[toChannelType(compID)], compID, DF_SSE);
3691
460k
  return dist;
3692
460k
}
3693
3694
bool checkValidMvs( const CodingUnit& cu)
3695
0
{
3696
  // clang-format off
3697
0
  const int affineShiftTab[3] =
3698
0
  {
3699
0
    MV_PRECISION_INTERNAL - MV_PRECISION_QUARTER,
3700
0
    MV_PRECISION_INTERNAL - MV_PRECISION_SIXTEENTH,
3701
0
    MV_PRECISION_INTERNAL - MV_PRECISION_INT
3702
0
  };
3703
3704
0
  const int normalShiftTab[NUM_IMV_MODES] =
3705
0
  {
3706
0
    MV_PRECISION_INTERNAL - MV_PRECISION_QUARTER,
3707
0
    MV_PRECISION_INTERNAL - MV_PRECISION_INT,
3708
0
    MV_PRECISION_INTERNAL - MV_PRECISION_4PEL,
3709
0
    MV_PRECISION_INTERNAL - MV_PRECISION_HALF,
3710
0
  };
3711
  // clang-format on
3712
3713
0
  int mvShift;
3714
3715
0
  for (int refList = 0; refList < NUM_REF_PIC_LIST_01; refList++)
3716
0
  {
3717
0
    if (cu.refIdx[refList] >= 0)
3718
0
    {
3719
0
      if (!cu.affine)
3720
0
      {
3721
0
        mvShift = normalShiftTab[cu.imv];
3722
0
        Mv signaledmvd(cu.mvd[refList][0].hor >> mvShift, cu.mvd[refList][0].ver >> mvShift);
3723
0
        if (!((signaledmvd.hor >= MVD_MIN) && (signaledmvd.hor <= MVD_MAX)) || !((signaledmvd.ver >= MVD_MIN) && (signaledmvd.ver <= MVD_MAX)))
3724
0
          return false;
3725
0
      }
3726
0
      else
3727
0
      {
3728
0
        for (int ctrlP = 1 + (cu.affineType == AFFINEMODEL_6PARAM); ctrlP >= 0; ctrlP--)
3729
0
        {
3730
0
          mvShift = affineShiftTab[cu.imv];
3731
0
          Mv signaledmvd(cu.mvd[refList][ctrlP].hor >> mvShift, cu.mvd[refList][ctrlP].ver >> mvShift);
3732
0
          if (!((signaledmvd.hor >= MVD_MIN) && (signaledmvd.hor <= MVD_MAX)) || !((signaledmvd.ver >= MVD_MIN) && (signaledmvd.ver <= MVD_MAX)))
3733
0
            return false;;
3734
0
        }
3735
0
      }
3736
0
    }
3737
0
  }
3738
  // avoid MV exceeding 18-bit dynamic range
3739
0
  const int maxMv = 1 << 17;
3740
0
  if (!cu.affine && !cu.mergeFlag)
3741
0
  {
3742
0
    if(    ( cu.refIdx[ 0 ] >= 0 && ( cu.mv[ 0 ][ 0 ].getAbsHor() >= maxMv || cu.mv[ 0 ][ 0 ].getAbsVer() >= maxMv ) )
3743
0
        || ( cu.refIdx[ 1 ] >= 0 && ( cu.mv[ 1 ][ 0 ].getAbsHor() >= maxMv || cu.mv[ 1 ][ 0 ].getAbsVer() >= maxMv ) ) )
3744
0
    {
3745
0
      return false;
3746
0
    }
3747
0
  }
3748
0
  if( cu.affine && !cu.mergeFlag )
3749
0
  {
3750
0
    for( int refList = 0; refList < NUM_REF_PIC_LIST_01; refList++ )
3751
0
    {
3752
0
      if( cu.refIdx[ refList ] >= 0 )
3753
0
      {
3754
0
        for( int ctrlP = 1 + ( cu.affineType == AFFINEMODEL_6PARAM ); ctrlP >= 0; ctrlP-- )
3755
0
        {
3756
0
          if( cu.mv[ refList ][ ctrlP ].getAbsHor() >= maxMv || cu.mv[ refList ][ ctrlP ].getAbsVer() >= maxMv )
3757
0
          {
3758
0
            return false;
3759
0
          }
3760
0
        }
3761
0
      }
3762
0
    }
3763
0
  }
3764
0
  return true;
3765
0
}
3766
3767
3768
void EncCu::xEncodeInterResidual( CodingStructure *&tempCS, CodingStructure *&bestCS, Partitioner &partitioner, const EncTestMode& encTestMode, int residualPass, bool* bestHasNonResi, double* equBcwCost )
3769
0
{
3770
0
  if( residualPass == 1 && encTestMode.lossless )
3771
0
  {
3772
0
    return;
3773
0
  }
3774
3775
0
  CodingUnit*            cu        = tempCS->getCU( partitioner.chType, partitioner.treeType );
3776
0
  double   bestCostInternal        = MAX_DOUBLE;
3777
3778
0
  if( !checkValidMvs( *cu ) )
3779
0
    return;
3780
3781
0
  double  currBestCost = MAX_DOUBLE;
3782
3783
  // For SBT
3784
0
  double     bestCost          = bestCS->cost;
3785
0
  double     bestCostBegin     = bestCS->cost;
3786
0
  const CodingUnit* prevBestCU = bestCS->getCU( partitioner.chType, partitioner.treeType );
3787
0
  uint8_t    prevBestSbt       = ( prevBestCU == nullptr ) ? 0 : prevBestCU->sbtInfo;
3788
0
  Distortion sbtOffDist        = 0;
3789
0
  bool       sbtOffRootCbf     = 0;
3790
0
  double     sbtOffCost        = MAX_DOUBLE;
3791
0
  uint8_t    currBestSbt       = 0;
3792
0
  uint8_t    histBestSbt       = MAX_UCHAR;
3793
0
  Distortion curPuSse          = MAX_DISTORTION;
3794
0
  uint8_t    numRDOTried       = 0;
3795
0
  bool       doPreAnalyzeResi  = false;
3796
0
  const bool mtsAllowed        =   tempCS->sps->MTSInter && cu->Y().maxDim() <= MTS_INTER_MAX_CU_SIZE;
3797
0
  const uint8_t sbtAllowed     = ( tempCS->pps->picWidthInLumaSamples < SBT_FAST64_WIDTH_THRESHOLD || m_pcEncCfg->m_SBT > 1 ) && cu->Y().maxDim() > 32 ? 0 : CU::checkAllowedSbt(*cu);
3798
3799
0
  if( sbtAllowed )
3800
0
  {
3801
    //SBT resolution-dependent fast algorithm: not try size-64 SBT in RDO for low-resolution sequences (now resolution below HD)
3802
0
    doPreAnalyzeResi = ( sbtAllowed || mtsAllowed ) && residualPass == 0;
3803
0
    m_cInterSearch.getBestSbt( tempCS, cu, histBestSbt, curPuSse, sbtAllowed, doPreAnalyzeResi, mtsAllowed );
3804
0
  }
3805
3806
0
  cu->skip    = false;
3807
0
  cu->sbtInfo = 0;
3808
3809
0
  const bool skipResidual = residualPass == 1;
3810
0
  if( skipResidual || histBestSbt == MAX_UCHAR || !CU::isSbtMode( histBestSbt ) )
3811
0
  {
3812
0
    m_cInterSearch.encodeResAndCalcRdInterCU( *tempCS, partitioner, skipResidual );
3813
0
    xEncodeDontSplit( *tempCS, partitioner );
3814
0
    xCheckDQP       ( *tempCS, partitioner );
3815
3816
0
    if( NULL != bestHasNonResi && (bestCostInternal > tempCS->cost) )
3817
0
    {
3818
0
      bestCostInternal = tempCS->cost;
3819
0
      if( !cu->ciip )
3820
0
        *bestHasNonResi = !cu->rootCbf;
3821
0
    }
3822
3823
0
    if( cu->rootCbf == false )
3824
0
    {
3825
0
      if( cu->ciip )
3826
0
      {
3827
0
        tempCS->cost = MAX_DOUBLE;
3828
0
        tempCS->costDbOffset = 0;
3829
0
        return;
3830
0
      }
3831
0
    }
3832
0
    currBestCost = tempCS->cost;
3833
0
    if( sbtAllowed )
3834
0
    {
3835
0
      sbtOffCost    = tempCS->cost;
3836
0
      sbtOffDist    = tempCS->dist;
3837
0
      sbtOffRootCbf = cu->rootCbf;
3838
0
      currBestSbt   = cu->firstTU->mtsIdx[COMP_Y] > MTS_SKIP ? SBT_OFF_MTS : SBT_OFF_DCT;
3839
0
      numRDOTried  += mtsAllowed ? 2 : 1;
3840
0
    }
3841
3842
0
    DTRACE_MODE_COST( *tempCS, m_cRdCost.getLambda( true ) );
3843
0
    xCheckBestMode( tempCS, bestCS, partitioner, encTestMode );
3844
3845
0
    STAT_COUNT_CU_MODES( partitioner.chType == CH_L, g_cuCounters1D[CU_RD_TESTS][0][!tempCS->slice->isIntra() + tempCS->slice->depth] );
3846
0
    STAT_COUNT_CU_MODES( partitioner.chType == CH_L && !tempCS->slice->isIntra(), g_cuCounters2D[CU_RD_TESTS][Log2( tempCS->area.lheight() )][Log2( tempCS->area.lwidth() )] );
3847
0
  }
3848
3849
0
  if( sbtAllowed && ( m_pcEncCfg->m_SBT == 1 || sbtOffRootCbf ) )
3850
0
  {
3851
0
    bool swapped = false; // avoid unwanted data copy
3852
0
    uint8_t numSbtRdo = CU::numSbtModeRdo( sbtAllowed );
3853
    //early termination if all SBT modes are not allowed
3854
    //normative
3855
0
    if( !sbtAllowed || skipResidual )
3856
0
    {
3857
0
      numSbtRdo = 0;
3858
0
    }
3859
    //fast algorithm
3860
0
    if( ( histBestSbt != MAX_UCHAR && !CU::isSbtMode( histBestSbt ) ) || m_cInterSearch.getSkipSbtAll() )
3861
0
    {
3862
0
      numSbtRdo = 0;
3863
0
    }
3864
0
    if( bestCost != MAX_DOUBLE && sbtOffCost != MAX_DOUBLE )
3865
0
    {
3866
0
      double th = 1.07;
3867
0
      if( !( prevBestSbt == 0 || m_sbtCostSave[0] == MAX_DOUBLE ) )
3868
0
      {
3869
0
        assert( m_sbtCostSave[1] <= m_sbtCostSave[0] );
3870
0
        th *= ( m_sbtCostSave[0] / m_sbtCostSave[1] );
3871
0
      }
3872
0
      if( sbtOffCost > bestCost * th )
3873
0
      {
3874
0
        numSbtRdo = 0;
3875
0
      }
3876
0
    }
3877
0
    if( !sbtOffRootCbf && sbtOffCost != MAX_DOUBLE )
3878
0
    {
3879
0
      double th = Clip3( 0.05, 0.55, ( 27 - cu->qp ) * 0.02 + 0.35 );
3880
0
      if( sbtOffCost < m_cRdCost.calcRdCost( ( cu->lwidth() * cu->lheight() ) << SCALE_BITS, 0 ) * th )
3881
0
      {
3882
0
        numSbtRdo = 0;
3883
0
      }
3884
0
    }
3885
3886
0
    if( histBestSbt != MAX_UCHAR && numSbtRdo != 0 )
3887
0
    {
3888
0
      numSbtRdo = 1;
3889
0
      m_cInterSearch.initSbtRdoOrder( CU::getSbtMode( CU::getSbtIdx( histBestSbt ), CU::getSbtPos( histBestSbt ) ) );
3890
0
    }
3891
3892
0
    for( int sbtModeIdx = 0; sbtModeIdx < numSbtRdo; sbtModeIdx++ )
3893
0
    {
3894
0
      uint8_t sbtMode = m_cInterSearch.getSbtRdoOrder( sbtModeIdx );
3895
0
      uint8_t sbtIdx = CU::getSbtIdxFromSbtMode( sbtMode );
3896
0
      uint8_t sbtPos = CU::getSbtPosFromSbtMode( sbtMode );
3897
3898
      //fast algorithm (early skip, save & load)
3899
0
      if( histBestSbt == MAX_UCHAR )
3900
0
      {
3901
0
        uint8_t skipCode = m_cInterSearch.skipSbtByRDCost( cu->lwidth(), cu->lheight(), cu->mtDepth, sbtIdx, sbtPos, bestCS->cost, sbtOffDist, sbtOffCost, sbtOffRootCbf );
3902
0
        if( skipCode != MAX_UCHAR )
3903
0
        {
3904
0
          continue;
3905
0
        }
3906
3907
0
        if( sbtModeIdx > 0 )
3908
0
        {
3909
0
          uint8_t prevSbtMode = m_cInterSearch.getSbtRdoOrder( sbtModeIdx - 1 );
3910
          //make sure the prevSbtMode is the same size as the current SBT mode (otherwise the estimated dist may not be comparable)
3911
0
          if( CU::isSameSbtSize( prevSbtMode, sbtMode ) )
3912
0
          {
3913
0
            Distortion currEstDist = m_cInterSearch.getEstDistSbt( sbtMode );
3914
0
            Distortion prevEstDist = m_cInterSearch.getEstDistSbt( prevSbtMode );
3915
0
            if( currEstDist > prevEstDist * 1.15 )
3916
0
            {
3917
0
              continue;
3918
0
            }
3919
0
          }
3920
0
        }
3921
0
      }
3922
3923
      //init tempCS and TU
3924
0
      if( bestCost == bestCS->cost ) //The first EMT pass didn't become the bestCS, so we clear the TUs generated
3925
0
      {
3926
0
        tempCS->clearTUs();
3927
0
      }
3928
0
      else if( !swapped )
3929
0
      {
3930
0
        tempCS->initStructData( encTestMode.qp );
3931
0
        tempCS->copyStructure( *bestCS, partitioner.chType, partitioner.treeType );
3932
0
        tempCS->getPredBuf().copyFrom( bestCS->getPredBuf() );
3933
0
        bestCost = bestCS->cost;
3934
0
        cu = tempCS->getCU( partitioner.chType, partitioner.treeType );
3935
0
        swapped = true;
3936
0
      }
3937
0
      else
3938
0
      {
3939
0
        tempCS->clearTUs();
3940
0
        bestCost = bestCS->cost;
3941
0
        cu = tempCS->getCU( partitioner.chType, partitioner.treeType );
3942
0
      }
3943
3944
      //we need to restart the distortion for the new tempCS, the bit count and the cost
3945
0
      tempCS->dist     = 0;
3946
0
      tempCS->fracBits = 0;
3947
0
      tempCS->cost     = MAX_DOUBLE;
3948
0
      cu->skip         = false;
3949
3950
3951
      //set SBT info
3952
0
      cu->sbtInfo = (sbtPos << 4) + sbtIdx;
3953
3954
      //try residual coding
3955
0
      m_cInterSearch.encodeResAndCalcRdInterCU( *tempCS, partitioner, skipResidual );
3956
0
      numRDOTried++;
3957
3958
0
      xEncodeDontSplit( *tempCS, partitioner );
3959
0
      xCheckDQP( *tempCS, partitioner );
3960
3961
0
      if( NULL != bestHasNonResi && ( bestCostInternal > tempCS->cost ) )
3962
0
      {
3963
0
        bestCostInternal = tempCS->cost;
3964
0
        if( !( cu->ciip ) )
3965
0
          *bestHasNonResi = !cu->rootCbf;
3966
0
      }
3967
3968
0
      if( tempCS->cost < currBestCost )
3969
0
      {
3970
0
        currBestSbt = cu->sbtInfo;
3971
0
        currBestCost = tempCS->cost;
3972
0
      }
3973
0
      else if( m_pcEncCfg->m_SBT > 2 )
3974
0
      {
3975
0
        sbtModeIdx = numSbtRdo;
3976
0
      }
3977
3978
0
      DTRACE_MODE_COST( *tempCS, m_cRdCost.getLambda( true ) );
3979
0
      xCheckBestMode( tempCS, bestCS, partitioner, encTestMode );
3980
0
      STAT_COUNT_CU_MODES( partitioner.chType == CH_L, g_cuCounters1D[CU_RD_TESTS][0][!tempCS->slice->isIntra() + tempCS->slice->depth] );
3981
0
      STAT_COUNT_CU_MODES( partitioner.chType == CH_L && !tempCS->slice->isIntra(), g_cuCounters2D[CU_RD_TESTS][Log2( tempCS->area.lheight() )][Log2( tempCS->area.lwidth() )] );
3982
0
    }
3983
3984
0
    if( bestCostBegin != bestCS->cost )
3985
0
    {
3986
0
      m_sbtCostSave[0] = sbtOffCost;
3987
0
      m_sbtCostSave[1] = currBestCost;
3988
0
    }
3989
3990
0
    if( histBestSbt == MAX_UCHAR && doPreAnalyzeResi && numRDOTried > 1 )
3991
0
    {
3992
0
      auto slsSbt = static_cast<CacheBlkInfoCtrl&>( m_modeCtrl );
3993
0
      int slShift = 4 + std::min( Log2( cu->lwidth() ) + Log2( cu->lheight() ), 9 );
3994
0
      slsSbt.saveBestSbt( cu->cs->area, (uint32_t)( curPuSse >> slShift ), currBestSbt );
3995
0
    }
3996
    
3997
0
    if( ETM_INTER_ME == encTestMode.type )
3998
0
    {
3999
0
      if( equBcwCost != NULL )
4000
0
      {
4001
0
        if( tempCS->cost < ( *equBcwCost ) && cu->BcwIdx == BCW_DEFAULT )
4002
0
        {
4003
0
          ( *equBcwCost ) = tempCS->cost;
4004
0
        }
4005
0
      }
4006
0
      else
4007
0
      {
4008
0
        CHECK( equBcwCost == NULL, "equBcwCost == NULL" );
4009
0
      }
4010
0
      if( tempCS->slice->checkLDC && !cu->imv && cu->BcwIdx != BCW_DEFAULT && tempCS->cost < m_bestBcwCost[1] )
4011
0
      {
4012
0
        if( tempCS->cost < m_bestBcwCost[0] )
4013
0
        {
4014
0
          m_bestBcwCost[1] = m_bestBcwCost[0];
4015
0
          m_bestBcwCost[0] = tempCS->cost;
4016
0
          m_bestBcwIdx[1] = m_bestBcwIdx[0];
4017
0
          m_bestBcwIdx[0] = cu->BcwIdx;
4018
0
        }
4019
0
        else
4020
0
        {
4021
0
          m_bestBcwCost[1] = tempCS->cost;
4022
0
          m_bestBcwIdx[1] = cu->BcwIdx;
4023
0
        }
4024
0
      }
4025
0
    }
4026
0
  }
4027
4028
0
  tempCS->cost = currBestCost;
4029
0
}
4030
4031
void EncCu::xEncodeDontSplit( CodingStructure &cs, Partitioner &partitioner )
4032
104k
{
4033
104k
  m_CABACEstimator->resetBits();
4034
4035
104k
  m_CABACEstimator->split_cu_mode( CU_DONT_SPLIT, cs, partitioner );
4036
104k
  if( partitioner.treeType == TREE_C )
4037
104k
    CHECK( m_CABACEstimator->getEstFracBits() != 0, "must be 0 bit" );
4038
4039
104k
  cs.fracBits += m_CABACEstimator->getEstFracBits(); // split bits
4040
104k
  cs.cost      = m_cRdCost.calcRdCost( cs.fracBits, cs.dist );
4041
104k
}
4042
4043
void EncCu::xReuseCachedResult( CodingStructure *&tempCS, CodingStructure *&bestCS, Partitioner &partitioner )
4044
0
{
4045
0
  EncTestMode cachedMode;
4046
4047
0
  if( ! m_modeCtrl.setCsFrom( *tempCS, cachedMode, partitioner ) )
4048
0
  {
4049
0
    THROW( "Should never happen!" );
4050
0
  }
4051
4052
0
  CodingUnit& cu = *tempCS->cus.front();
4053
0
  partitioner.setCUData( cu );
4054
4055
0
  if( CU::isIntra( cu ) )
4056
0
  {
4057
0
    if( isLuma( cu.chType ) )
4058
0
    {
4059
0
      cu.getMotionBuf().memset( -1 ); // clear motion buf
4060
0
    }
4061
0
    xReconIntraQT( cu );
4062
0
  }
4063
0
  else
4064
0
  {
4065
0
    xDeriveCUMV( cu );
4066
0
    xReconInter( cu );
4067
0
  }
4068
4069
0
  m_CABACEstimator->getCtx() = m_CurrCtx->start;
4070
0
  m_CABACEstimator->resetBits();
4071
4072
0
  CUCtx cuCtx;
4073
0
  cuCtx.isDQPCoded = true;
4074
0
  cuCtx.isChromaQpAdjCoded = true;
4075
0
  m_CABACEstimator->coding_unit( cu, partitioner, cuCtx );
4076
4077
0
  tempCS->fracBits = m_CABACEstimator->getEstFracBits();
4078
0
  tempCS->cost     = m_cRdCost.calcRdCost( tempCS->fracBits, tempCS->dist );
4079
4080
0
  xEncodeDontSplit( *tempCS,         partitioner );
4081
0
  xCheckDQP       ( *tempCS,         partitioner );
4082
0
  xCheckBestMode  (  tempCS, bestCS, partitioner, cachedMode, m_EDO );
4083
0
}
4084
4085
uint64_t EncCu::xCalcPuMeBits( const CodingUnit &cu )
4086
0
{
4087
0
  CHECK( !cu.mergeFlag, "Should only be used for merge!" );
4088
0
  CHECK( CU::isIBC( cu ), "Shound not be used for IBC" );
4089
4090
0
  m_CABACEstimator->resetBits();
4091
0
  m_CABACEstimator->merge_flag(cu);
4092
0
  if( cu.mergeFlag )
4093
0
  {
4094
0
    m_CABACEstimator->merge_data( cu );
4095
0
  }
4096
0
  return m_CABACEstimator->getEstFracBits();
4097
0
}
4098
4099
double EncCu::xCalcDistortion(CodingStructure *&cur_CS, ChannelType chType, int BitDepth, int imv)
4100
0
{
4101
0
  const auto currDist1 = m_cRdCost.getDistPart(cur_CS->getOrgBuf( COMP_Y ), cur_CS->getPredBuf( COMP_Y ), BitDepth, COMP_Y, m_pcEncCfg->m_fastHad ? DF_HAD_fast : DF_HAD );
4102
0
  unsigned int uiMvBits = 0;
4103
0
  unsigned imvShift = imv == IMV_HPEL ? 1 : (imv << 1);
4104
0
  const CodingUnit& cu = *cur_CS->getCU( chType, TREE_D);
4105
0
  if (cu.interDir != 2)
4106
0
  {
4107
0
    uiMvBits += m_cRdCost.getBitsOfVectorWithPredictor(cu.mvd[0][0].hor, cu.mvd[0][0].ver, imvShift + MV_FRACTIONAL_BITS_DIFF);
4108
0
  }
4109
0
  if (cu.interDir != 1)
4110
0
  {
4111
0
    uiMvBits += m_cRdCost.getBitsOfVectorWithPredictor(cu.mvd[1][0].hor, cu.mvd[1][0].ver, imvShift + MV_FRACTIONAL_BITS_DIFF);
4112
0
  }
4113
0
  return (double(currDist1) + (double)m_cRdCost.getCost(uiMvBits));
4114
0
}
4115
4116
int EncCu::xCheckMMVDCand(MmvdIdx& mmvdMergeCand, int& bestDir, int tempNum, double& bestCostOffset, double& bestCostMerge, double bestCostList )
4117
0
{
4118
0
  int baseIdx = mmvdMergeCand.val / MMVD_MAX_REFINE_NUM;
4119
0
  int CandCur = mmvdMergeCand.val - MMVD_MAX_REFINE_NUM * baseIdx;
4120
4121
0
  if( m_pcEncCfg->m_MMVD > 2 )
4122
0
  {
4123
0
    if( CandCur % 4 == 0 )
4124
0
    {
4125
0
      if( ( bestCostOffset >= bestCostMerge ) && ( CandCur >= 4 ) )
4126
0
      {
4127
0
        if( mmvdMergeCand.val > MMVD_MAX_REFINE_NUM )
4128
0
        {
4129
0
          return 2;
4130
0
        }
4131
0
        else
4132
0
        {
4133
0
          mmvdMergeCand.val = MMVD_MAX_REFINE_NUM;
4134
0
          if( tempNum == mmvdMergeCand.val )
4135
0
          {
4136
0
            return 2;
4137
0
          }
4138
0
        }
4139
0
      }
4140
      //reset
4141
0
      bestCostOffset = MAX_DOUBLE;
4142
0
      bestCostMerge  = bestCostList;
4143
0
    }
4144
0
  }
4145
4146
0
  if( mmvdMergeCand.val == MMVD_MAX_REFINE_NUM )
4147
0
  {
4148
0
    bestDir = 0;
4149
0
  }
4150
0
  if( CandCur >= 4 )
4151
0
  {
4152
0
    if( CandCur % 4 != bestDir )
4153
0
    {
4154
0
      return 1;
4155
0
    }
4156
0
  }
4157
0
  return 0;
4158
0
}
4159
4160
void EncCu::xInitLumaDeltaQpLUT()
4161
0
{
4162
  //TODO: to be discussed: add config parameters for these?
4163
0
  std::vector<int> defaultLumaLevelTodQp_QpChangePoints   =  { -3,  -2,  -1,   0,   1,   2,   3,   4,   5,   6 };
4164
0
  std::vector<int> defaultLumaLevelTodQp_LumaChangePoints =  {  0, 301, 367, 434, 501, 567, 634, 701, 767, 834 };
4165
  
4166
0
  int         lastDeltaQPValue = 0;
4167
0
  std::size_t nextSparseIndex = 0;
4168
0
  for( int index = 0; index < LUMA_LEVEL_TO_DQP_LUT_MAXSIZE; index++ )
4169
0
  {
4170
0
    while( nextSparseIndex < defaultLumaLevelTodQp_QpChangePoints.size() && index >= defaultLumaLevelTodQp_LumaChangePoints[nextSparseIndex] )
4171
0
    {
4172
0
      lastDeltaQPValue = defaultLumaLevelTodQp_QpChangePoints[nextSparseIndex];
4173
0
      nextSparseIndex++;
4174
0
    }
4175
0
    m_lumaLevelToDeltaQPLUT[index] = lastDeltaQPValue;
4176
0
  }
4177
0
}
4178
4179
MergeItem::MergeItem()
4180
153k
{
4181
4182
153k
}
4183
MergeItem::~MergeItem()
4184
153k
{
4185
4186
153k
}
4187
4188
void MergeItem::create( ChromaFormat chromaFormat, const Area &area )
4189
153k
{
4190
153k
  if( m_pelStorage.bufs.empty() )
4191
153k
  {
4192
153k
    m_pelStorage.create( chromaFormat, area );
4193
153k
    m_mvStorage .resize( area.area() >> ( MIN_CU_LOG2 << 1 ) );
4194
153k
  }
4195
4196
153k
  init();
4197
153k
}
4198
4199
void MergeItem::init()
4200
153k
{
4201
  // reset data
4202
153k
  cost        = MAX_DOUBLE;
4203
153k
  mergeIdx    = 0;
4204
153k
  bcwIdx      = 0;
4205
153k
  interDir    = 0;
4206
153k
  useAltHpelIf  = false;
4207
153k
  affineType    = AFFINEMODEL_4PARAM;
4208
153k
  mergeItemType = MergeItemType::NUM;
4209
4210
153k
  noBdofRefine  = false;
4211
153k
  noResidual    = false;
4212
4213
153k
  lumaPredReady   = false;
4214
153k
  chromaPredReady = false;
4215
153k
}
4216
4217
void MergeItem::importMergeInfo(const MergeCtx& mergeCtx, int _mergeIdx, MergeItemType _mergeItemType, CodingUnit& pu)
4218
0
{
4219
0
  mergeIdx      = _mergeIdx;
4220
0
  mergeItemType = _mergeItemType;
4221
4222
0
  if( mergeItemType != MergeItemType::GPM && mergeItemType != MergeItemType::MMVD )
4223
0
  {
4224
0
    mvField[REF_PIC_LIST_0][0] = mergeCtx.mvFieldNeighbours [mergeIdx][REF_PIC_LIST_0];
4225
0
    mvField[REF_PIC_LIST_1][0] = mergeCtx.mvFieldNeighbours [mergeIdx][REF_PIC_LIST_1];
4226
0
    interDir                   = mergeCtx.interDirNeighbours[mergeIdx];
4227
0
    bcwIdx                     = mergeCtx.BcwIdx            [mergeIdx];
4228
0
    useAltHpelIf               = mergeCtx.useAltHpelIf      [mergeIdx];
4229
0
  }
4230
4231
0
  switch( _mergeItemType )
4232
0
  {
4233
0
  case MergeItemType::REGULAR:
4234
0
  case MergeItemType::CIIP:
4235
0
    break;
4236
4237
0
  case MergeItemType::MMVD:
4238
0
  {
4239
0
    MmvdIdx candIdx;
4240
4241
0
    candIdx.val                = mergeIdx;
4242
0
    mvField[L0][0]             . setMvField( pu.mv[L0][0], pu.refIdx[0] );
4243
0
    mvField[L1][0]             . setMvField( pu.mv[L1][0], pu.refIdx[1] );
4244
0
    interDir                   = pu.interDir;
4245
0
    bcwIdx                     = pu.BcwIdx;
4246
0
    useAltHpelIf               = mergeCtx.useAltHpelIf[candIdx.pos.baseIdx];
4247
4248
0
    break;
4249
0
  }
4250
4251
0
  case MergeItemType::GPM:
4252
0
    mvField[L0][0]             . setMvField( Mv( 0, 0 ), -1 );
4253
0
    mvField[L1][0]             . setMvField( Mv( 0, 0 ), -1 );
4254
0
    bcwIdx                     = BCW_DEFAULT;
4255
0
    useAltHpelIf               = false;
4256
4257
0
    break;
4258
4259
0
  case MergeItemType::IBC:
4260
0
  default:
4261
0
    THROW( "Wrong merge item type" );
4262
0
  }
4263
4264
0
  getMvBuf( pu ).copyFrom( pu.getMotionBuf() );
4265
0
}
4266
4267
void MergeItem::importMergeInfo( const AffineMergeCtx &mergeCtx, int _mergeIdx, MergeItemType _mergeItemType, CodingUnit& pu )
4268
0
{
4269
0
  mergeIdx      = _mergeIdx;
4270
0
  mergeItemType = _mergeItemType;
4271
4272
0
  affineType    = mergeCtx.affineType         [mergeIdx];
4273
0
  interDir      = mergeCtx.interDirNeighbours [mergeIdx];
4274
0
  bcwIdx        = mergeCtx.BcwIdx             [mergeIdx];
4275
0
  useAltHpelIf  = false;
4276
4277
0
  switch( _mergeItemType )
4278
0
  {
4279
0
  case MergeItemType::SBTMVP:
4280
    // the pu motion was already generated preparing for IFP check (unconditional)
4281
0
    mvField[L0][0] . setMvField( pu.mv[L0][0], pu.refIdx[L0] );
4282
0
    mvField[L1][0] . setMvField( pu.mv[L1][0], pu.refIdx[L1] );
4283
4284
0
    break;
4285
4286
0
  case MergeItemType::AFFINE:
4287
    // the pu motion was already generated preparing for IFP check (unconditional)
4288
0
    mvField[L0][0] . setMvField( pu.mv[L0][0], pu.refIdx[L0] );
4289
0
    mvField[L0][1] . setMvField( pu.mv[L0][1], pu.refIdx[L0] );
4290
0
    mvField[L0][2] . setMvField( pu.mv[L0][2], pu.refIdx[L0] );
4291
0
    mvField[L1][0] . setMvField( pu.mv[L1][0], pu.refIdx[L1] );
4292
0
    mvField[L1][1] . setMvField( pu.mv[L1][1], pu.refIdx[L1] );
4293
0
    mvField[L1][2] . setMvField( pu.mv[L1][2], pu.refIdx[L1] );
4294
4295
0
    break;
4296
4297
0
  default:
4298
0
    THROW( "Wrong merge item type" );
4299
0
  }
4300
4301
  // the MI buf was already generated preparing for IFP check (unconditional)
4302
0
  getMvBuf( pu ).copyFrom( pu.getMotionBuf() );
4303
0
}
4304
4305
bool MergeItem::exportMergeInfo( CodingUnit &pu, bool forceNoResidual ) const
4306
0
{
4307
0
  pu.mergeFlag        = true;
4308
0
  pu.mmvdMergeFlag    = false;
4309
0
  pu.interDir         = interDir;
4310
0
  pu.mergeIdx         = mergeIdx;
4311
0
  pu.mergeType        = MRG_TYPE_DEFAULT_N;
4312
0
  pu.mv[REF_PIC_LIST_0][0]  = mvField[REF_PIC_LIST_0][0].mv;
4313
0
  pu.mv[REF_PIC_LIST_1][0]  = mvField[REF_PIC_LIST_1][0].mv;
4314
0
  pu.refIdx[REF_PIC_LIST_0] = mvField[REF_PIC_LIST_0][0].refIdx;
4315
0
  pu.refIdx[REF_PIC_LIST_1] = mvField[REF_PIC_LIST_1][0].refIdx;
4316
0
  pu.mvd[REF_PIC_LIST_0][0] = Mv();
4317
0
  pu.mvd[REF_PIC_LIST_1][0] = Mv();
4318
0
  pu.mvpIdx[REF_PIC_LIST_0] = NOT_VALID;
4319
0
  pu.mvpIdx[REF_PIC_LIST_1] = NOT_VALID;
4320
0
  pu.mvpNum[REF_PIC_LIST_0] = NOT_VALID;
4321
0
  pu.mvpNum[REF_PIC_LIST_1] = NOT_VALID;
4322
0
  pu.BcwIdx         = ( interDir == 3 ) ? bcwIdx : BCW_DEFAULT;
4323
0
  pu.mcControl      = 0;
4324
0
  pu.mmvdSkip       = false;
4325
0
  pu.affine         = false;
4326
0
  pu.affineType     = AFFINEMODEL_4PARAM;
4327
0
  pu.geo            = false;
4328
0
  pu.mtsFlag        = false;
4329
0
  pu.ciip           = false;
4330
0
  pu.imv            = ( !pu.geo && useAltHpelIf ) ? IMV_HPEL : IMV_OFF;
4331
0
  pu.mvRefine       = false;
4332
4333
0
  const bool resetCiip2Regular = mergeItemType == MergeItemType::CIIP && forceNoResidual;
4334
0
  MergeItemType updatedType    = resetCiip2Regular ? MergeItemType::REGULAR : mergeItemType;
4335
4336
0
  switch( updatedType )
4337
0
  {
4338
0
  case MergeItemType::REGULAR:
4339
0
    CU::restrictBiPredMergeCandsOne( pu );
4340
0
    break;
4341
4342
0
  case MergeItemType::CIIP:
4343
0
    CHECK( forceNoResidual, "Cannot force no residuals for CIIP" );
4344
0
    pu.ciip           = true;
4345
0
    pu.intraDir[CH_L] = PLANAR_IDX;
4346
0
    pu.intraDir[CH_C] = DM_CHROMA_IDX;
4347
0
    break;
4348
4349
0
  case MergeItemType::MMVD:
4350
0
    pu.mmvdMergeFlag    = true;
4351
0
    pu.mmvdMergeIdx.val = mergeIdx;
4352
0
    if( forceNoResidual )
4353
0
    {
4354
0
      pu.mmvdSkip       = true;
4355
0
    }
4356
0
    CU::restrictBiPredMergeCandsOne( pu );
4357
0
    break;
4358
4359
0
  case MergeItemType::SBTMVP:
4360
0
    pu.affine    = true;
4361
0
    pu.mergeType = MRG_TYPE_SUBPU_ATMVP;
4362
0
    break;
4363
4364
0
  case MergeItemType::AFFINE:
4365
0
    pu.affine     = true;
4366
0
    pu.affineType = affineType;
4367
0
    pu.mv[L0][0]  = mvField[L0][0].mv;
4368
0
    pu.mv[L1][0]  = mvField[L1][0].mv;
4369
0
    pu.mv[L0][1]  = mvField[L0][1].mv;
4370
0
    pu.mv[L1][1]  = mvField[L1][1].mv;
4371
0
    pu.mv[L0][2]  = mvField[L0][2].mv;
4372
0
    pu.mv[L1][2]  = mvField[L1][2].mv;
4373
0
    pu.refIdx[L0] = mvField[L0][0].refIdx;
4374
0
    pu.refIdx[L1] = mvField[L1][0].refIdx;
4375
0
    break;
4376
4377
0
  case MergeItemType::GPM:
4378
0
    pu.mergeIdx = -1;
4379
0
    pu.geo      = true;
4380
0
    pu.BcwIdx   = BCW_DEFAULT;
4381
0
    updateGpmIdx( mergeIdx, pu.geoSplitDir, pu.geoMergeIdx );
4382
0
    pu.imv      = IMV_OFF;
4383
0
    break;
4384
4385
0
  case MergeItemType::IBC:
4386
0
  default:
4387
0
    THROW( "Wrong merge item type" );
4388
0
  }
4389
4390
0
  pu.getMotionBuf().copyFrom( getMvBuf( pu ) );
4391
4392
0
  return resetCiip2Regular;
4393
0
}
4394
4395
MergeItemList::MergeItemList()
4396
19.1k
{
4397
4398
19.1k
}
4399
4400
MergeItemList::~MergeItemList()
4401
19.1k
{
4402
19.1k
  for( MergeItem* p : m_list )
4403
0
  {
4404
0
    delete p;
4405
0
  }
4406
19.1k
  m_list.clear();
4407
4408
19.1k
  for( MergeItem *p : m_mergeItems )
4409
153k
  {
4410
153k
    delete p;
4411
153k
  }
4412
19.1k
  m_mergeItems.clear();
4413
19.1k
}
4414
4415
void MergeItemList::init( size_t maxSize, size_t maxExtSize, ChromaFormat chromaFormat, SizeType ctuWidth, SizeType ctuHeight )
4416
19.1k
{
4417
19.1k
  CHECK( !m_mergeItems.empty() || !m_list.empty(), "MergeItemList already initialized" );
4418
4419
19.1k
  m_list      . reserve( maxSize + 1 ); // to avoid reallocation when inserting a new item
4420
19.1k
  m_mergeItems. reserve( maxSize + 1 );
4421
19.1k
  m_maxSize   = maxSize;
4422
19.1k
  m_maxExtSize= maxExtSize;
4423
19.1k
  m_numExt    = 0;
4424
4425
172k
  for( int i = 0; i < maxSize + m_maxExtSize; i++ )
4426
153k
  {
4427
153k
    MergeItem *p = new MergeItem;
4428
153k
    p->create( chromaFormat, Area{ 0, 0, ctuWidth, ctuHeight } );
4429
153k
    m_mergeItems.push_back( p );
4430
153k
  }
4431
19.1k
}
4432
4433
MergeItem *MergeItemList::allocateNewMergeItem()
4434
0
{
4435
0
  m_numExt++;
4436
0
  CHECK( m_mergeItems.empty(), "Missing merge items!" );
4437
0
  CHECK( m_numExt > m_maxExtSize, "Taking out more external items than specified during list allocation!" );
4438
0
  MergeItem *p = m_mergeItems.back();
4439
0
  m_mergeItems.pop_back();
4440
0
  p->init();
4441
0
  return p;
4442
0
}
4443
4444
bool MergeItemList::insertMergeItemToList( MergeItem *p )
4445
0
{
4446
0
  CHECK( m_list.size() + m_mergeItems.size() + m_numExt != m_maxSize + m_maxExtSize, "Wrong number of items held" );
4447
4448
0
  m_numExt--;
4449
4450
0
  if( m_list.empty() )
4451
0
  {
4452
0
    m_list.push_back( p );
4453
0
  }
4454
0
  else if( m_list.size() == m_maxTrackingNum && p->cost >= m_list.back()->cost )
4455
0
  {
4456
0
    m_mergeItems.push_back( p );
4457
0
    return false;
4458
0
  }
4459
0
  else
4460
0
  {
4461
0
    if( m_list.size() == m_maxTrackingNum )
4462
0
    {
4463
0
      m_mergeItems.push_back( m_list.back() );
4464
0
      m_list      .pop_back();
4465
0
    }
4466
0
    auto it = std::find_if( m_list.begin(), m_list.end(), [&p]( const MergeItem *mi ) { return p->cost < mi->cost; } );
4467
0
    m_list.insert( it, p );
4468
0
  }
4469
4470
0
  return true;
4471
0
}
4472
4473
void MergeItemList::giveBackMergeItem( MergeItem *p )
4474
0
{
4475
0
  CHECK( m_list.size() + m_mergeItems.size() + m_numExt != m_maxSize + m_maxExtSize, "Wrong number of items held" );
4476
4477
0
  m_numExt--;
4478
4479
0
  m_mergeItems.push_back( p );
4480
0
}
4481
4482
MergeItem *MergeItemList::getMergeItemInList( size_t index )
4483
0
{
4484
0
  return index < m_maxTrackingNum ? m_list[index] : nullptr;
4485
0
}
4486
4487
void MergeItemList::resetList( size_t maxTrackingNum )
4488
0
{
4489
0
  CHECK( maxTrackingNum > m_maxSize, "Not enough items allocated to track " << maxTrackingNum << " items" );
4490
4491
0
  for( auto p : m_list )
4492
0
  {
4493
0
    m_mergeItems.push_back( p );
4494
0
  }
4495
0
  m_list.clear  ();
4496
4497
0
  m_maxTrackingNum = maxTrackingNum;
4498
0
}
4499
4500
void MergeItemList::shrinkList( size_t reduceTo )
4501
0
{
4502
0
  CHECK( reduceTo > m_maxSize, "Not enough items allocated to track " << reduceTo << " items" );
4503
4504
0
  while( m_list.size() > reduceTo )
4505
0
  {
4506
0
    m_mergeItems.push_back( m_list.back() );
4507
0
    m_list      .pop_back();
4508
0
  }
4509
0
}
4510
4511
} // namespace vvenc
4512
4513
//! \}